mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-09-29 19:25:35 +00:00
* filer.remote.sync: stamp entries with IF_CHUNKS_EQUAL so a stale write-back cannot delete live chunks updateLocalEntry records the RemoteEntry stamp after an upload by writing the event's entry back with UpdateEntry. The filer deletes every stored chunk absent from an updated entry, so when the file was rewritten while its upload was in flight (or the event is a replay), the stale snapshot deletes the rewrite's chunks: the entry then points at the new fid with no needle behind it, and the rewrite's own upload fails and is skipped as superseded. The stamp write now carries WriteCondition IF_CHUNKS_EQUAL over the event's chunk fids, evaluated by the filer under the path lock. A refused stamp means the filer moved past this event; the superseding event follows in the log and stamps the current entry, so the refusal is logged and skipped like a superseded upload. Reproduction: weed server -filer plus a weed server -s3 remote, remote.mount, filer.remote.sync; hold the remote (docker pause) so one upload stays in flight, rewrite the file through the filer, unpause. Before: the entry's chunk is 404 on every volume server. After: the stale stamp is refused, the rewrite's chunk stays live and reads back after a vacuum. * filer.remote.sync: stamp entries with IF_ENTRY_EQUAL so stale inline content or metadata cannot be restored The IF_CHUNKS_EQUAL guard compared only the chunk fid multiset, so a rewrite that touched inline content or metadata alone still compared equal and the stale snapshot overwrote the live entry. The new clause compares the whole stored entry against the event's entry under the same path lock. * filer: route conditional UpdateEntry to the entry's owner filer Two filers locking the same path locally could still pass a stale condition on the non-owner while the owner's entry had moved on. When a condition or expected_extended precondition is set, forward the request to the entry's owner the same way conditional CreateEntry does, with is_moved bounding the hop. * filer: compare IF_ENTRY_EQUAL against the normalized expected entry FindEntry grows FileSize to the chunk extent, so a raw event entry with FileSize still zero failed the condition on an unchanged file and the stamp was skipped, letting a replay upload the object again. * filer.remote.sync: classify refused stamps by gRPC status only A FailedPrecondition substring in an unrelated error would have been swallowed as a skipped stamp; status.FromError already unwraps. * remote sync: keep the event entry intact for IF_ENTRY_EQUAL --------- Co-authored-by: Chris Lu <chrislusf@users.noreply.github.com>
172 lines
5.8 KiB
Go
172 lines
5.8 KiB
Go
package weed_server
|
|
|
|
import (
|
|
"strconv"
|
|
"strings"
|
|
"time"
|
|
|
|
"github.com/seaweedfs/seaweedfs/weed/filer"
|
|
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
|
"github.com/seaweedfs/seaweedfs/weed/s3api/s3_constants"
|
|
"google.golang.org/protobuf/proto"
|
|
)
|
|
|
|
// conditionIsSet reports whether a condition asks for any check at all.
|
|
func conditionIsSet(cond *filer_pb.WriteCondition) bool {
|
|
return cond != nil && len(cond.Clauses) > 0
|
|
}
|
|
|
|
// writeConditionSatisfied reports whether the precondition holds against the
|
|
// current entry (nil if absent), evaluated under the path lock. Every clause
|
|
// must hold (logical AND).
|
|
func writeConditionSatisfied(cond *filer_pb.WriteCondition, current *filer.Entry) bool {
|
|
for _, c := range cond.Clauses {
|
|
if !clauseSatisfied(c, current) {
|
|
return false
|
|
}
|
|
}
|
|
return true
|
|
}
|
|
|
|
// clauseSatisfied evaluates one primitive against the current entry. For the
|
|
// ETag kinds, etags is a set: IF_ETAG_MATCH holds when the current ETag equals
|
|
// any member, IF_ETAG_NOT_MATCH when it equals none. The IF_EXTENDED_* kinds are
|
|
// generic guards on an extended attribute used to enforce object-lock (legal
|
|
// hold and retention) without S3 knowledge in the filer.
|
|
func clauseSatisfied(c *filer_pb.WriteCondition_Clause, current *filer.Entry) bool {
|
|
exists := current != nil
|
|
switch c.Kind {
|
|
case filer_pb.WriteCondition_NONE:
|
|
return true
|
|
case filer_pb.WriteCondition_IF_NOT_EXISTS:
|
|
return !exists
|
|
case filer_pb.WriteCondition_IF_EXISTS:
|
|
return exists
|
|
case filer_pb.WriteCondition_IF_ETAG_MATCH:
|
|
return exists && etagInSet(storedEntryETag(current), c.Etags, c.AllowWeak)
|
|
case filer_pb.WriteCondition_IF_ETAG_NOT_MATCH:
|
|
return !exists || !etagInSet(storedEntryETag(current), c.Etags, c.AllowWeak)
|
|
case filer_pb.WriteCondition_IF_UNMODIFIED_SINCE:
|
|
return !exists || current.Attr.Mtime.Unix() <= c.UnixTime
|
|
case filer_pb.WriteCondition_IF_MODIFIED_SINCE:
|
|
return !exists || current.Attr.Mtime.Unix() > c.UnixTime
|
|
case filer_pb.WriteCondition_IF_EXTENDED_NOT_EQUAL:
|
|
if !exists {
|
|
return true
|
|
}
|
|
v, ok := current.Extended[c.ExtKey]
|
|
return !ok || string(v) != c.ExtValue
|
|
case filer_pb.WriteCondition_IF_EXTENDED_TIME_ELAPSED:
|
|
if !exists {
|
|
return true
|
|
}
|
|
// An optional gate scopes the guard: when gate_key is set, the time check
|
|
// only applies if extended[gate_key] == gate_value. This lets the gateway
|
|
// express governance bypass (enforce retention only for COMPLIANCE mode)
|
|
// without reading the entry — the filer decides under the lock.
|
|
if c.GateKey != "" {
|
|
gv, gok := current.Extended[c.GateKey]
|
|
if !gok || string(gv) != c.GateValue {
|
|
return true
|
|
}
|
|
}
|
|
v, ok := current.Extended[c.ExtKey]
|
|
if !ok {
|
|
return true
|
|
}
|
|
deadline, err := strconv.ParseInt(strings.TrimSpace(string(v)), 10, 64)
|
|
if err != nil {
|
|
// An unparseable retention deadline is treated as still in force, so
|
|
// a malformed attribute fails safe (write blocked) rather than open.
|
|
return false
|
|
}
|
|
return deadline <= time.Now().Unix()
|
|
case filer_pb.WriteCondition_IF_CHUNKS_EQUAL:
|
|
return chunkFidsEqual(current, c.Fids)
|
|
case filer_pb.WriteCondition_IF_ENTRY_EQUAL:
|
|
if !exists || c.ExpectedEntry == nil {
|
|
return !exists && c.ExpectedEntry == nil
|
|
}
|
|
// Normalize the expected entry the way FindEntry normalizes the stored
|
|
// one (e.g. FileSize grows to the chunk extent), or an unchanged entry
|
|
// can compare unequal.
|
|
return proto.Equal(current.ToProtoEntry(), filer.FromPbEntry("", c.ExpectedEntry).ToProtoEntry())
|
|
default:
|
|
// An unrecognized clause kind (e.g. from a newer client) must not be
|
|
// treated as satisfied, which would silently bypass the guard. Fail
|
|
// closed so the write is blocked rather than slipping through.
|
|
return false
|
|
}
|
|
}
|
|
|
|
// chunkFidsEqual compares the stored chunk fids (absent entry = none) against
|
|
// expected as multisets; chunk order carries no meaning for needle liveness.
|
|
func chunkFidsEqual(current *filer.Entry, expected []string) bool {
|
|
var chunks []*filer_pb.FileChunk
|
|
if current != nil {
|
|
chunks = current.GetChunks()
|
|
}
|
|
if len(chunks) != len(expected) {
|
|
return false
|
|
}
|
|
counts := make(map[string]int, len(expected))
|
|
for _, fid := range expected {
|
|
counts[fid]++
|
|
}
|
|
for _, chunk := range chunks {
|
|
fid := chunk.GetFileIdString()
|
|
if counts[fid] == 0 {
|
|
return false
|
|
}
|
|
counts[fid]--
|
|
}
|
|
return true
|
|
}
|
|
|
|
// etagInSet reports whether stored matches any candidate. A strong comparison
|
|
// (allowWeak false) treats a weak ETag as never equal; a weak comparison
|
|
// ignores the W/ marker on both sides.
|
|
func etagInSet(stored string, candidates []string, allowWeak bool) bool {
|
|
for _, c := range candidates {
|
|
if etagEqual(stored, c, allowWeak) {
|
|
return true
|
|
}
|
|
}
|
|
return false
|
|
}
|
|
|
|
func etagEqual(stored, expected string, allowWeak bool) bool {
|
|
sv, sWeak := canonicalETag(stored)
|
|
ev, eWeak := canonicalETag(expected)
|
|
// RFC 7232 strong comparison: a weak ETag on either side never matches.
|
|
if !allowWeak && (sWeak || eWeak) {
|
|
return false
|
|
}
|
|
return sv == ev
|
|
}
|
|
|
|
// canonicalETag splits off the weak (W/) marker before stripping quotes, so a
|
|
// weak ETag like W/"abc" yields ("abc", true).
|
|
func canonicalETag(etag string) (value string, weak bool) {
|
|
etag = strings.TrimSpace(etag)
|
|
if strings.HasPrefix(etag, "W/") {
|
|
return strings.Trim(etag[len("W/"):], `"`), true
|
|
}
|
|
return strings.Trim(etag, `"`), false
|
|
}
|
|
|
|
// storedEntryETag mirrors the S3 gateway's ETag precedence (the stored
|
|
// Seaweed ETag extended attribute, then the chunk/Md5 fallback) so conditional
|
|
// comparisons match what the gateway computes, without coupling the filer to
|
|
// S3 request handling.
|
|
func storedEntryETag(entry *filer.Entry) string {
|
|
if v, ok := entry.Extended[s3_constants.ExtETagKey]; ok && len(v) > 0 {
|
|
return normalizeETag(string(v))
|
|
}
|
|
return normalizeETag(filer.ETagEntry(entry))
|
|
}
|
|
|
|
func normalizeETag(etag string) string {
|
|
return strings.Trim(etag, `"`)
|
|
}
|