Compare commits

..
Author SHA1 Message Date
Chris Lu 6b5942946f fix(s3api): cancel ListEntries stream in hasChildren
hasChildren opened a server-streaming ListEntries with an uncancelled
context and returned after one Recv, leaving gRPC's per-stream client
goroutine parked. Route through filer_pb.List, which cancels the stream
on return, so the goroutine is cleaned up.
2026-05-21 14:56:36 -07:00
139 changed files with 2073 additions and 11160 deletions
+2 -2
View File
@@ -128,14 +128,14 @@ jobs:
- name: Login to Docker Hub
if: github.event_name != 'pull_request'
uses: docker/login-action@v4.2.0
uses: docker/login-action@v4.1.0
with:
username: ${{ secrets.DOCKER_USERNAME }}
password: ${{ secrets.DOCKER_PASSWORD }}
- name: Login to GHCR
if: github.event_name != 'pull_request'
uses: docker/login-action@v4.2.0
uses: docker/login-action@v4.1.0
with:
registry: ghcr.io
username: ${{ secrets.GHCR_USERNAME }}
@@ -133,7 +133,7 @@ jobs:
- name: Login to Docker Hub
if: github.event_name != 'pull_request'
uses: docker/login-action@v4.2.0
uses: docker/login-action@v4.1.0
with:
username: ${{ secrets.DOCKER_USERNAME }}
password: ${{ secrets.DOCKER_PASSWORD }}
+5 -5
View File
@@ -221,13 +221,13 @@ jobs:
buildkitd-config: /tmp/buildkitd.toml
- name: Login to Docker Hub
if: needs.setup.outputs.publish == 'true'
uses: docker/login-action@v4.2.0
uses: docker/login-action@v4.1.0
with:
username: ${{ secrets.DOCKER_USERNAME }}
password: ${{ secrets.DOCKER_PASSWORD }}
- name: Login to GHCR
if: needs.setup.outputs.publish == 'true'
uses: docker/login-action@v4.2.0
uses: docker/login-action@v4.1.0
with:
registry: ghcr.io
username: ${{ secrets.GHCR_USERNAME }}
@@ -275,7 +275,7 @@ jobs:
fi
- name: Login to GHCR
if: needs.setup.outputs.publish == 'true'
uses: docker/login-action@v4.2.0
uses: docker/login-action@v4.1.0
with:
registry: ghcr.io
username: ${{ secrets.GHCR_USERNAME }}
@@ -430,12 +430,12 @@ jobs:
ghcr.io/chrislusf/seaweedfs
tags: type=raw,value=${{ github.event_name == 'workflow_dispatch' && github.event.inputs.image_tag || 'latest' }},suffix=${{ steps.config.outputs.tag_suffix }}
- name: Login to Docker Hub
uses: docker/login-action@v4.2.0
uses: docker/login-action@v4.1.0
with:
username: ${{ secrets.DOCKER_USERNAME }}
password: ${{ secrets.DOCKER_PASSWORD }}
- name: Login to GHCR
uses: docker/login-action@v4.2.0
uses: docker/login-action@v4.1.0
with:
registry: ghcr.io
username: ${{ secrets.GHCR_USERNAME }}
@@ -50,7 +50,7 @@ jobs:
-
name: Login to Docker Hub
if: github.event_name != 'pull_request'
uses: docker/login-action@v4.2.0
uses: docker/login-action@v4.1.0
with:
username: ${{ secrets.DOCKER_USERNAME }}
password: ${{ secrets.DOCKER_PASSWORD }}
@@ -237,14 +237,14 @@ jobs:
- name: Login to Docker Hub
if: (github.event_name != 'workflow_dispatch' || github.event.inputs.variant == 'all' || github.event.inputs.variant == matrix.variant) && github.event_name != 'pull_request'
uses: docker/login-action@v4.2.0
uses: docker/login-action@v4.1.0
with:
username: ${{ secrets.DOCKER_USERNAME }}
password: ${{ secrets.DOCKER_PASSWORD }}
- name: Login to GHCR
if: (github.event_name != 'workflow_dispatch' || github.event.inputs.variant == 'all' || github.event.inputs.variant == matrix.variant) && github.event_name != 'pull_request'
uses: docker/login-action@v4.2.0
uses: docker/login-action@v4.1.0
with:
registry: ghcr.io
username: ${{ secrets.GHCR_USERNAME }}
@@ -300,14 +300,14 @@ jobs:
steps:
- name: Login to Docker Hub
if: github.event_name != 'workflow_dispatch' || github.event.inputs.variant == 'all' || github.event.inputs.variant == matrix.variant
uses: docker/login-action@v4.2.0
uses: docker/login-action@v4.1.0
with:
username: ${{ secrets.DOCKER_USERNAME }}
password: ${{ secrets.DOCKER_PASSWORD }}
- name: Login to GHCR
if: github.event_name != 'workflow_dispatch' || github.event.inputs.variant == 'all' || github.event.inputs.variant == matrix.variant
uses: docker/login-action@v4.2.0
uses: docker/login-action@v4.1.0
with:
registry: ghcr.io
username: ${{ secrets.GHCR_USERNAME }}
@@ -380,7 +380,7 @@ jobs:
variant: large_disk
steps:
- name: Login to GHCR
uses: docker/login-action@v4.2.0
uses: docker/login-action@v4.1.0
with:
registry: ghcr.io
username: ${{ secrets.GHCR_USERNAME }}
@@ -429,13 +429,13 @@ jobs:
latest_tag: latest_large_disk
steps:
- name: Login to Docker Hub
uses: docker/login-action@v4.2.0
uses: docker/login-action@v4.1.0
with:
username: ${{ secrets.DOCKER_USERNAME }}
password: ${{ secrets.DOCKER_PASSWORD }}
- name: Login to GHCR
uses: docker/login-action@v4.2.0
uses: docker/login-action@v4.1.0
with:
registry: ghcr.io
username: ${{ secrets.GHCR_USERNAME }}
@@ -88,7 +88,7 @@ jobs:
uses: docker/setup-buildx-action@4d04d5d9486b7bd6fa91e7baf45bbb4f8b9deedd # v1
- name: Login to Docker Hub
uses: docker/login-action@650006c6eb7dba73a995cc03b0b2d7f5ca915bee # v1
uses: docker/login-action@4907a6ddec9925e35a0a9e82d7399ccc52663121 # v1
with:
username: ${{ secrets.DOCKER_USERNAME }}
password: ${{ secrets.DOCKER_PASSWORD }}
-167
View File
@@ -1,167 +0,0 @@
# Design: Serializing Bucket Configuration Mutations
Issue #9651 — concurrent `PutBucketVersioning` + `PutBucketEncryption` (as Terraform
issues them in parallel) intermittently lose the encryption write.
## Root cause
The bucket's entire config lives in one filer entry, `/buckets/<name>`. Every
config API does a read-modify-write of that single entry, and the writes are not
serialized:
- `updateBucketConfig(bucket, fn)` (`s3api_bucket_config.go:468`) — sources from a
possibly-stale cached `BucketConfig`, mutates `Entry.Extended`, writes the
**whole** entry. Used by: versioning, object-lock config, lifecycle, ACL/owner.
- `UpdateBucketMetadata` → `setBucketMetadata` (`:1042`) — reads a fresh entry,
mutates `Entry.Content`, writes the **whole** entry. Used by: encryption, CORS,
tagging, ownership, policy, notification.
Two ingredients produce the lost update:
1. **No serialization** of the read→modify→write (the cache mutexes only guard the
in-memory map, not the RMW).
2. **Whole-entry rewrite from an independent snapshot** — `updateBucketConfig`
rebuilds from a stale cached `BucketConfig` whose `Content` predates the
concurrent encryption write, so writing the whole entry reverts `Content`.
Sequential calls always pass (each sees the previous write), so it only surfaces
under concurrency — and CI's slower IO widens the window (the "2 of ~12 runs").
## Goals
- No lost updates across concurrent bucket-config changes — for **all** config
fields, not just versioning/encryption.
- Correct for a single S3 gateway (the reported case) and for multiple gateways.
- Reuse the filer primitives just merged (per-path lock, `WriteCondition`,
`ObjectTransaction`); do not reintroduce a distributed lock.
- Minimal blast radius: the fix lands at the two chokepoint helpers.
## Non-goals
- Changing the one-entry-per-bucket storage model.
- Multi-filer-concurrent bucket writes (addressed only as an optional phase 3).
## The two ingredients map to two complementary fixes
### Fix A — serialize + read fresh (closes the window for whole-entry writers)
Both `updateBucketConfig` and `UpdateBucketMetadata` must run their RMW under one
per-bucket critical section, and **re-read the entry fresh from the filer inside
it** — not rebuild from the cached `BucketConfig`. The lock alone is insufficient:
without the fresh read, two serialized writers still each apply a stale snapshot.
### Fix B — field-level updates (removes the collision entirely)
The two writers touch disjoint fields (`Extended[versioning]` vs `Content`). If
each path updated only its own field instead of rewriting the whole entry, neither
could clobber the other regardless of ordering. This is the structural fix and
makes serialization a defense-in-depth concern rather than a correctness
requirement for cross-field cases.
## Where to serialize (layering)
The bucket entry is a single filer entry, so unlike object writes there is no
sharding — the question is purely the scope of the lock:
| Layer | Serializes across | Cost | Notes |
|---|---|---|---|
| 1. Gateway-local per-bucket lock | one gateway process | tiny | fixes the reported (single-gateway/CI) case |
| 2. Filer per-path lock via conditional write | all gateways on one filer | small | reuses #9640 `CreateEntry`+`WriteCondition` |
| 3. Route-by-key to bucket-key owner filer | all gateways and filers | medium | same mechanism as the object DLM-removal |
## Recommended plan (phased)
### Phase 1 — minimal fix for #9651 (gateway-local lock + fresh read)
Add a bounded per-bucket lock table to `S3ApiServer`, reusing the same
`util.LockTable` the filer uses for its per-path lock:
```go
// in S3ApiServer
bucketConfigLocks *util.LockTable[string] // serialize bucket-entry RMW
func (s3a *S3ApiServer) withBucketConfigLock(bucket string, fn func() s3err.ErrorCode) s3err.ErrorCode {
lk := s3a.bucketConfigLocks.AcquireLock("bucketConfig", bucket, util.ExclusiveLock)
defer s3a.bucketConfigLocks.ReleaseLock(bucket, lk)
return fn()
}
```
Wrap the RMW in **both** chokepoints, and inside the lock read the entry fresh:
- `updateBucketConfig`: acquire the lock; re-read `/buckets/<name>` from the filer
(not the cache); rebuild `BucketConfig` from that fresh entry; apply `fn`; write;
invalidate cache; release.
- `UpdateBucketMetadata`/`setBucketMetadata`: same lock key; it already reads fresh,
so it just needs to share the critical section.
Both must use the **same** lock keyed on `bucket`, so versioning and encryption
contend on one mutex. This closes the reported window. Limitation: only one
gateway; two gateways behind a load balancer still race.
Test: parallel `PutBucketVersioning` + `PutBucketEncryption`, assert both persist
(the exact Terraform scenario), plus an N-way parallel variant over distinct
fields.
### Phase 2 — robust across gateways (field-level + CAS via merged primitives)
Move the writers off whole-entry rewrites:
- **Extended-based config** (versioning, object-lock, ownership, tagging-in-Extended)
→ `ObjectTransaction` `PATCH_EXTENDED` on `/buckets/<name>`. The owner filer reads
the entry fresh under its per-path lock and merges only the named keys, so the
gateway never sends a whole-entry snapshot — this dissolves *both* ingredients for
these fields.
- **`Content`-based config** (encryption, CORS, tags blob) — **chosen and
implemented (b3): extend `PATCH_EXTENDED` with `set_content`.** Under the same
per-path lock the filer reads the entry fresh, merges extended attributes, and
replaces `Content`, preserving the rest. So a content write becomes a field-level
patch too — `setBucketMetadata` patches `Content`, `updateBucketConfig` patches
extended keys, and the two serialize on the lock instead of racing whole-entry
rewrites. This is cleaner than the alternatives below: no client-side retry, no
storage migration, and it reuses `ObjectTransaction`'s existing atomic lock.
- (b1, rejected) Conditional `CreateEntry` overwrite with `IF_ETAG_MATCH` + retry
(#9640): correct but needs client-side retry, and the bucket directory entry has
no reliable ETag to compare on.
- (b2, future) Migrate each per-feature config out of the single `Content` blob
into its own `Extended` key. Then even *intra-blob* writes (tags vs encryption)
stop racing. Larger migration; tracked separately.
Once all paths are field-level patches, the phase-1 gateway lock is unnecessary —
the filer enforces atomicity. (This is the path taken: phase 1 was skipped.)
### Phase 3 — multi-filer (only if needed)
If multiple filers can write `/buckets/<name>` concurrently, a filer-local per-path
lock no longer suffices. Route bucket-config writes to
`PrimaryForKey("/buckets/<name>")` (the lock-ring view) and serialize on that one
owner filer — the same route-by-key design used to take object writes off the DLM.
Overkill for rare config writes; include only if multi-filer bucket writes are real.
## Correctness summary
- Phase 1: all RMW for a bucket serialize within a gateway; the fresh read means the
second writer observes the first's change. Closes #9651 for single-gateway.
- Phase 2: `PATCH_EXTENDED` is atomic field-level merge at the filer (no snapshot);
CAS turns a concurrent `Content` write into a retry, enforced under the filer's
per-path lock — correct for any number of gateways sharing a filer.
- Phase 3: one owner filer serializes all writers — correct across filers too.
## Scope checklist (every path that RMWs the bucket entry)
All of these funnel through the two chokepoints, so fixing the chokepoints covers
them — but the fix must not leave any of them on an unserialized path:
- via `updateBucketConfig`: versioning, object-lock config, lifecycle, ACL/owner.
- via `UpdateBucketMetadata`/`setBucketMetadata`: encryption, CORS, tagging,
ownership controls, bucket policy, notification.
- bucket create/delete (`CreateEntry`/`DeleteEntry` of `/buckets/<name>`) already
go through the filer's per-path lock on `CreateEntry`; ensure they take the same
bucket lock if they also patch config.
## Cache rule (must document in code)
Under the lock, **read the entry from the filer, never rebuild from the cached
`BucketConfig`**. The cache is for reads; it must be invalidated on every write and
never be the source for an RMW. This is the single most important detail — the lock
without the fresh read does not fix the bug.
+6 -6
View File
@@ -26,7 +26,7 @@ require (
github.com/facebookgo/subset v0.0.0-20200203212716-c811ad88dec4 // indirect
github.com/fsnotify/fsnotify v1.9.0 // indirect
github.com/go-redsync/redsync/v4 v4.16.0
github.com/go-sql-driver/mysql v1.10.0
github.com/go-sql-driver/mysql v1.9.3
github.com/go-zookeeper/zk v1.0.4 // indirect
github.com/golang/protobuf v1.5.4
github.com/golang/snappy v1.0.0
@@ -48,7 +48,7 @@ require (
github.com/klauspost/compress v1.18.6
github.com/klauspost/reedsolomon v1.14.0
github.com/kurin/blazer v0.5.3
github.com/linxGnu/grocksdb v1.10.8
github.com/linxGnu/grocksdb v1.10.7
github.com/mailru/easyjson v0.9.1 // indirect
github.com/mattn/go-isatty v0.0.20 // indirect
github.com/modern-go/concurrent v0.0.0-20180306012644-bacd9c7ef1dd // indirect
@@ -91,12 +91,12 @@ require (
gocloud.dev v0.45.0
gocloud.dev/pubsub/natspubsub v0.45.0
gocloud.dev/pubsub/rabbitpubsub v0.45.0
golang.org/x/crypto v0.52.0
golang.org/x/crypto v0.51.0
golang.org/x/exp v0.0.0-20260410095643-746e56fc9e2f
golang.org/x/image v0.39.0
golang.org/x/net v0.54.0
golang.org/x/oauth2 v0.36.0
golang.org/x/sys v0.45.0
golang.org/x/sys v0.44.0
golang.org/x/text v0.37.0 // indirect
golang.org/x/tools v0.44.0 // indirect
golang.org/x/xerrors v0.0.0-20240903120638-7835f813f4da // indirect
@@ -157,7 +157,7 @@ require (
github.com/xeipuuv/gojsonschema v1.2.0
github.com/ydb-platform/ydb-go-sdk-auth-environ v0.5.1
github.com/ydb-platform/ydb-go-sdk/v3 v3.134.2
go.etcd.io/etcd/client/pkg/v3 v3.6.11
go.etcd.io/etcd/client/pkg/v3 v3.6.10
go.uber.org/atomic v1.11.0
golang.org/x/sync v0.20.0
golang.org/x/tools/godoc v0.1.0-deprecated
@@ -296,7 +296,7 @@ require (
cloud.google.com/go/compute/metadata v0.9.0 // indirect
cloud.google.com/go/iam v1.7.0 // indirect
cloud.google.com/go/monitoring v1.24.3 // indirect
filippo.io/edwards25519 v1.2.0 // indirect
filippo.io/edwards25519 v1.1.1 // indirect
github.com/Azure/azure-sdk-for-go/sdk/azcore v1.21.1
github.com/Azure/azure-sdk-for-go/sdk/azidentity v1.13.1
github.com/Azure/azure-sdk-for-go/sdk/internal v1.12.0 // indirect
+12 -12
View File
@@ -547,8 +547,8 @@ cloud.google.com/go/workflows v1.10.0/go.mod h1:fZ8LmRmZQWacon9UCX1r/g/DfAXx5VcP
dario.cat/mergo v1.0.2 h1:85+piFYR1tMbRrLcDwR18y4UKJ3aH1Tbzi24VRW1TK8=
dario.cat/mergo v1.0.2/go.mod h1:E/hbnu0NxMFBjpMIE34DRGLWqDy0g5FuKDhCb31ngxA=
dmitri.shuralyov.com/gpu/mtl v0.0.0-20190408044501-666a987793e9/go.mod h1:H6x//7gZCb22OMCxBHrMx7a5I7Hp++hsVxbQ4BYO7hU=
filippo.io/edwards25519 v1.2.0 h1:crnVqOiS4jqYleHd9vaKZ+HKtHfllngJIiOpNpoJsjo=
filippo.io/edwards25519 v1.2.0/go.mod h1:xzAOLCNug/yB62zG1bQ8uziwrIqIuxhctzJT18Q77mc=
filippo.io/edwards25519 v1.1.1 h1:YpjwWWlNmGIDyXOn8zLzqiD+9TyIlPhGFG96P39uBpw=
filippo.io/edwards25519 v1.1.1/go.mod h1:BxyFTGdWcka3PhytdK4V28tE5sGfRvvvRV7EaN4VDT4=
gioui.org v0.0.0-20210308172011-57750fc8a0a6/go.mod h1:RSH6KIUZ0p2xy5zHDxgAM4zumjgTw83q2ge/PI+yyw8=
git.sr.ht/~sbinet/gg v0.3.1/go.mod h1:KGYtlADtqsqANL9ueOFkWymvzUvLMQllU5Ixo+8v3pc=
github.com/AdaLogics/go-fuzz-headers v0.0.0-20240806141605-e8a1dd7889d6 h1:He8afgbRMd7mFxO99hRNu+6tazq8nFF9lIwo9JFroBk=
@@ -1128,8 +1128,8 @@ github.com/go-redsync/redsync/v4 v4.16.0 h1:bNcOzeHH9d3s6pghU9NJFMPrQa41f5Nx3L4Y
github.com/go-redsync/redsync/v4 v4.16.0/go.mod h1:V4gagqgyASWBZuwx4xGzu72aZNb/6Mo05byUa3mVmKQ=
github.com/go-resty/resty/v2 v2.17.2 h1:FQW5oHYcIlkCNrMD2lloGScxcHJ0gkjshV3qcQAyHQk=
github.com/go-resty/resty/v2 v2.17.2/go.mod h1:kCKZ3wWmwJaNc7S29BRtUhJwy7iqmn+2mLtQrOyQlVA=
github.com/go-sql-driver/mysql v1.10.0 h1:Q+1LV8DkHJvSYAdR83XzuhDaTykuDx0l6fkXxoWCWfw=
github.com/go-sql-driver/mysql v1.10.0/go.mod h1:M+cqaI7+xxXGG9swrdeUIoPG3Y3KCkF0pZej+SK+nWk=
github.com/go-sql-driver/mysql v1.9.3 h1:U/N249h2WzJ3Ukj8SowVFjdtZKfu9vlLZxjPXV1aweo=
github.com/go-sql-driver/mysql v1.9.3/go.mod h1:qn46aNg1333BRMNU69Lq93t8du/dwxI64Gl8i5p1WMU=
github.com/go-stack/stack v1.8.0/go.mod h1:v0f6uXyyMGvRgIKkXu+yp6POWl0qKG85gN/melR3HDY=
github.com/go-task/slim-sprig v0.0.0-20230315185526-52ccab3ef572 h1:tfuBGBXKqDEevZMzYi5KSi8KkcZtzBcTgAUUtapy0OI=
github.com/go-task/slim-sprig/v3 v3.0.0 h1:sUs3vkvUymDpBKi3qH1YSqBQk9+9D/8M2mN1vB6EwHI=
@@ -1515,8 +1515,8 @@ github.com/lib/pq v1.11.1 h1:wuChtj2hfsGmmx3nf1m7xC2XpK6OtelS2shMY+bGMtI=
github.com/lib/pq v1.11.1/go.mod h1:/p+8NSbOcwzAEI7wiMXFlgydTwcgTr3OSKMsD2BitpA=
github.com/linkedin/goavro/v2 v2.15.0 h1:pDj1UrjUOO62iXhgBiE7jQkpNIc5/tA5eZsgolMjgVI=
github.com/linkedin/goavro/v2 v2.15.0/go.mod h1:KXx+erlq+RPlGSPmLF7xGo6SAbh8sCQ53x064+ioxhk=
github.com/linxGnu/grocksdb v1.10.8 h1:Nau01Hhm/0kaVTR6d4viwD6npYbnDvZAfzwJCLzKRYo=
github.com/linxGnu/grocksdb v1.10.8/go.mod h1:OLQKZwiKwaJiAVCsOzWKvwiLwfZ5Vz8Md5TYR7t7pM8=
github.com/linxGnu/grocksdb v1.10.7 h1:fCi4qvZWo04VgFwGWmO8HQJgUVounJBy+C2TMVPU/ho=
github.com/linxGnu/grocksdb v1.10.7/go.mod h1:OLQKZwiKwaJiAVCsOzWKvwiLwfZ5Vz8Md5TYR7t7pM8=
github.com/lithammer/fuzzysearch v1.1.8 h1:/HIuJnjHuXS8bKaiTMeeDlW2/AyIWk2brx1V8LFgLN4=
github.com/lithammer/fuzzysearch v1.1.8/go.mod h1:IdqeyBClc3FFqSzYq/MXESsS4S0FsZ5ajtkr5xPLts4=
github.com/lithammer/shortuuid/v3 v3.0.7 h1:trX0KTHy4Pbwo/6ia8fscyHoGA+mf1jWbPJVuvyJQQ8=
@@ -2111,8 +2111,8 @@ go.etcd.io/bbolt v1.4.3 h1:dEadXpI6G79deX5prL3QRNP6JB8UxVkqo4UPnHaNXJo=
go.etcd.io/bbolt v1.4.3/go.mod h1:tKQlpPaYCVFctUIgFKFnAlvbmB3tpy1vkTnDWohtc0E=
go.etcd.io/etcd/api/v3 v3.6.10 h1:jlwjtELjA8yi2VWpOFH+0w0lGr3K6mVDyn0RDB9aaAY=
go.etcd.io/etcd/api/v3 v3.6.10/go.mod h1:pdV4VeFmvhdNjB4LWRkC8ReLyRBAxUOze3GarMhE2sk=
go.etcd.io/etcd/client/pkg/v3 v3.6.11 h1:e41mp315Yn3QMGPmEzCyLsMINgJXTY/dX8kM++1csxU=
go.etcd.io/etcd/client/pkg/v3 v3.6.11/go.mod h1:DysuMe/inqRyC/1tjRR6hReH/VV9Lufs27YKSKBWWJg=
go.etcd.io/etcd/client/pkg/v3 v3.6.10 h1:tBT7podcPhuVbCVkAEzx8bC5I+aqxfLwBN8/As1arrA=
go.etcd.io/etcd/client/pkg/v3 v3.6.10/go.mod h1:WEy3PpwbbEBVRdh1NVJYsuUe/8eyI21PNJRazeD8z/Y=
go.etcd.io/etcd/client/v3 v3.6.10 h1:J598zJ+C/ZPvImypmq5waj84+bovePrlZERHklf34y0=
go.etcd.io/etcd/client/v3 v3.6.10/go.mod h1:iHhUDUcEwaKs1YFq3MgmI9U4zhTVasp/vgdVbFf1RS8=
go.mongodb.org/mongo-driver v1.17.9 h1:IexDdCuuNJ3BHrELgBlyaH9p60JXAvdzWR128q+U5tU=
@@ -2216,8 +2216,8 @@ golang.org/x/crypto v0.14.0/go.mod h1:MVFd36DqK4CsrnJYDkBA3VC4m2GkXAM0PvzMCn4JQf
golang.org/x/crypto v0.19.0/go.mod h1:Iy9bg/ha4yyC70EfRS8jz+B6ybOBKMaSxLj6P6oBDfU=
golang.org/x/crypto v0.23.0/go.mod h1:CKFgDieR+mRhux2Lsu27y0fO304Db0wZe70UKqHu0v8=
golang.org/x/crypto v0.31.0/go.mod h1:kDsLvtWBEx7MV9tJOj9bnXsPbxwJQ6csT/x4KIN4Ssk=
golang.org/x/crypto v0.52.0 h1:RMs7fP2rXdep0CftQlK8Uf+kibLm7qkCcradZWYz988=
golang.org/x/crypto v0.52.0/go.mod h1:1QgfPxDqh0T2M/elOJtp9RvuR95kVjir0e6/BvEmGbc=
golang.org/x/crypto v0.51.0 h1:IBPXwPfKxY7cWQZ38ZCIRPI50YLeevDLlLnyC5wRGTI=
golang.org/x/crypto v0.51.0/go.mod h1:8AdwkbraGNABw2kOX6YFPs3WM22XqI4EXEd8g+x7Oc8=
golang.org/x/exp v0.0.0-20180321215751-8460e604b9de/go.mod h1:CJ0aWSM057203Lf6IL+f9T1iT9GByDxfZKAQTCR3kQA=
golang.org/x/exp v0.0.0-20180807140117-3d87b88a115f/go.mod h1:CJ0aWSM057203Lf6IL+f9T1iT9GByDxfZKAQTCR3kQA=
golang.org/x/exp v0.0.0-20190121172915-509febef88a4/go.mod h1:CJ0aWSM057203Lf6IL+f9T1iT9GByDxfZKAQTCR3kQA=
@@ -2511,8 +2511,8 @@ golang.org/x/sys v0.13.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
golang.org/x/sys v0.17.0/go.mod h1:/VUhepiaJMQUp4+oa/7Zr1D23ma6VTLIYjOOTFZPUcA=
golang.org/x/sys v0.20.0/go.mod h1:/VUhepiaJMQUp4+oa/7Zr1D23ma6VTLIYjOOTFZPUcA=
golang.org/x/sys v0.28.0/go.mod h1:/VUhepiaJMQUp4+oa/7Zr1D23ma6VTLIYjOOTFZPUcA=
golang.org/x/sys v0.45.0 h1:dO4czNzziLiiXplLQgBCEpCvXQ3dnkn0SdaZSYdQ+FY=
golang.org/x/sys v0.45.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw=
golang.org/x/sys v0.44.0 h1:ildZl3J4uzeKP07r2F++Op7E9B29JRUy+a27EibtBTQ=
golang.org/x/sys v0.44.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw=
golang.org/x/telemetry v0.0.0-20240228155512-f48c80bd79b2/go.mod h1:TeRTkGYfJXctD9OcfyVLyj2J3IxLnKwHJR8f4D8a3YE=
golang.org/x/telemetry v0.0.0-20260409153401-be6f6cb8b1fa h1:efT73AJZfAAUV7SOip6pWGkwJDzIGiKBZGVzHYa+ve4=
golang.org/x/telemetry v0.0.0-20260409153401-be6f6cb8b1fa/go.mod h1:kHjTxDEnAu6/Nl9lDkzjWpR+bmKfxeiRuSDlsMb70gE=
+2 -2
View File
@@ -1,6 +1,6 @@
apiVersion: v1
description: SeaweedFS
name: seaweedfs
appVersion: "4.29"
appVersion: "4.27"
# Dev note: Trigger a helm chart release by `git tag -a helm-<version>`
version: 4.29.0
version: 4.27.0
@@ -31,15 +31,6 @@ service SeaweedFiler {
rpc DeleteEntry (DeleteEntryRequest) returns (DeleteEntryResponse) {
}
rpc ObjectTransaction (ObjectTransactionRequest) returns (ObjectTransactionResponse) {
}
rpc ObjectTransactionBatch (ObjectTransactionBatchRequest) returns (ObjectTransactionBatchResponse) {
}
rpc PosixLock (PosixLockRequest) returns (PosixLockResponse) {
}
rpc AtomicRenameEntry (AtomicRenameEntryRequest) returns (AtomicRenameEntryResponse) {
}
rpc StreamRenameEntry (StreamRenameEntryRequest) returns (stream StreamRenameEntryResponse) {
@@ -231,56 +222,6 @@ message CreateEntryRequest {
bool is_from_other_cluster = 4;
repeated int32 signatures = 5;
bool skip_check_parent_directory = 6;
// Optional precondition evaluated against the current entry atomically with
// the write, under the filer's per-path lock. The caller must route the
// key's writes to this entry's owner filer for the check to be authoritative.
WriteCondition condition = 7;
}
// WriteCondition is the precondition the filer evaluates against the existing
// entry before writing, under the per-path lock. A failed condition returns
// FilerError PRECONDITION_FAILED. The client maps request semantics (e.g. RFC
// 7232) to clauses; the filer just compares.
//
// A condition is a list of clauses that ALL must hold (logical AND). One clause
// is the common case; several express what a single comparison cannot: an ETag
// set (If-Match / If-None-Match with multiple values), weak-ETag comparison, and
// compound conditions (e.g. If-Match + If-Unmodified-Since together).
message WriteCondition {
enum Kind {
NONE = 0; // unconditional
IF_NOT_EXISTS = 1; // fail if the entry exists (If-None-Match: *)
IF_EXISTS = 2; // fail if the entry is absent (If-Match: *)
IF_ETAG_MATCH = 3; // fail if absent or etag matches none of the set (If-Match)
IF_ETAG_NOT_MATCH = 4; // fail if present and etag matches any of the set (If-None-Match)
IF_UNMODIFIED_SINCE = 5; // fail if present and mtime > unix_time
IF_MODIFIED_SINCE = 6; // fail if present and mtime <= unix_time
IF_EXTENDED_NOT_EQUAL = 7; // fail if present and extended[ext_key] == ext_value
IF_EXTENDED_TIME_ELAPSED = 8; // fail if present and extended[ext_key] (unix seconds) is in the future
}
// Clause is one primitive comparison. IF_ETAG_MATCH holds when the current
// entry's ETag equals any value in etags; IF_ETAG_NOT_MATCH holds when it
// equals none. allow_weak permits weak-comparison (ignoring the W/ prefix).
//
// The IF_EXTENDED_* kinds are generic guards on an extended attribute, used
// to enforce object-lock without teaching the filer S3 semantics:
// IF_EXTENDED_NOT_EQUAL expresses a legal hold (block while a key equals a
// value), and IF_EXTENDED_TIME_ELAPSED expresses retention (block while a
// stored unix-second deadline is in the future, compared to the filer's
// clock). The caller composes these and, for governance-bypass, simply omits
// the retention clause when the bypass is authorized — the filer makes no
// authorization decision.
message Clause {
Kind kind = 1;
repeated string etags = 2; // ETag set for IF_ETAG_* kinds
int64 unix_time = 3; // bound (unix seconds) for IF_*_SINCE kinds
bool allow_weak = 4; // compare ETags ignoring the weak (W/) marker
string ext_key = 5; // extended attribute name for IF_EXTENDED_* kinds
string ext_value = 6; // blocking value for IF_EXTENDED_NOT_EQUAL
string gate_key = 7; // IF_EXTENDED_TIME_ELAPSED: only enforce when extended[gate_key] == gate_value
string gate_value = 8; // gate value (e.g. retention mode COMPLIANCE for governance bypass)
}
repeated Clause clauses = 1; // all must hold (logical AND)
}
// Structured error codes for filer entry operations.
@@ -292,132 +233,6 @@ enum FilerError {
EXISTING_IS_DIRECTORY = 3; // cannot overwrite directory with file
EXISTING_IS_FILE = 4; // cannot overwrite file with directory
ENTRY_ALREADY_EXISTS = 5; // O_EXCL and entry already exists
PRECONDITION_FAILED = 6; // WriteCondition not satisfied
}
// ObjectMutation is one entry-level change applied by ObjectTransaction. All
// mutations of a transaction run under a single per-path lock (the request's
// lock_key) and in order, so the gateway can describe a multi-entry object
// operation as one request instead of holding a distributed lock across
// several RPCs. Data-bearing writes (entries with chunks) should be written
// before the transaction; mutations here are metadata-scoped.
message ObjectMutation {
enum Type {
PUT = 0; // create or replace the entry (entry field)
DELETE = 1; // delete the entry at directory/name (no error if absent)
PATCH_EXTENDED = 2; // merge set_extended / remove delete_extended on the entry
RECOMPUTE_LATEST = 3; // scan a directory and re-point a parent entry (recompute)
}
Type type = 1;
string directory = 2;
string name = 3; // entry name for DELETE / PATCH_EXTENDED / RECOMPUTE_LATEST (the pointer entry)
Entry entry = 4; // full entry for PUT
map<string, bytes> set_extended = 5; // PATCH_EXTENDED: keys to set
repeated string delete_extended = 6; // PATCH_EXTENDED: keys to remove
bool is_delete_data = 7; // DELETE: also delete chunk data
bool is_recursive = 8; // DELETE: recurse into a directory
Recompute recompute = 9; // RECOMPUTE_LATEST parameters
bool set_content = 10; // PATCH_EXTENDED: replace Entry.content with content
bytes content = 11; // PATCH_EXTENDED: new Entry.content when set_content
bool touch_mtime = 12; // PATCH_EXTENDED: set the entry's Mtime to now (e.g. a metadata-replace copy)
}
// Recompute re-derives a pointer entry (directory/name on the mutation) from the
// current contents of a scanned directory, atomically under the transaction's
// lock. It is mechanical: the filer picks the child that sorts first or last by
// name and copies the requested fields into the pointer; it has no knowledge of
// what the entries mean. The caller (which does know the versioning scheme)
// supplies the sort direction and the key mappings. This covers re-pointing the
// latest version after a specific version is deleted, where the scan must run
// under the lock.
message Recompute {
string scan_dir = 1; // directory whose direct children are scanned
bool descending = 2; // pick the child that sorts last by name (else first)
map<string, string> copy_extended = 3; // pointer extended key -> source extended key on the chosen child
string name_to_key = 4; // if set, store the chosen child's name under this pointer key
string size_to_key = 5; // if set, store the chosen child's FileSize (decimal) under this pointer key
string mtime_to_key = 6; // if set, store the chosen child's Mtime (decimal) under this pointer key
string demote_key = 7; // if set, stamp demote_value on the prior name_to_key target when it changes
bytes demote_value = 8; // value for demote_key
string exclude_name = 9; // if set, skip this child when scanning (e.g. a version about to be deleted)
}
// ObjectTransactionRequest applies an ordered list of mutations atomically with
// respect to other writers of the same object, by holding the filer's per-path
// lock on lock_key for the whole transaction. The optional condition is checked
// first, against condition_key when set, else lock_key. Callers set route_key to
// the object's stable owner ring key; a filer that is not the owner forwards the
// transaction one hop to the owner, so a stale ring view is tolerated.
message ObjectTransactionRequest {
string lock_key = 1; // object path to lock and to evaluate the condition against
WriteCondition condition = 2; // optional precondition, checked under the lock
repeated ObjectMutation mutations = 3;
bool is_from_other_cluster = 4;
repeated int32 signatures = 5;
string condition_key = 6; // if set, evaluate the condition against this entry instead of lock_key (still locking lock_key)
string route_key = 7; // ring key identifying the owner filer; a non-owner forwards the whole transaction to it
bool is_moved = 8; // set on a forwarded transaction so the receiver applies it locally instead of forwarding again
}
message ObjectTransactionResponse {
string error = 1;
FilerError error_code = 2;
}
// PosixLockRange is one advisory byte-range lock. Owner identity is (sid, owner):
// sid is the mount session, owner the FUSE lock owner within it, so owners from
// different mounts never alias. end is inclusive (max uint64 = to EOF); is_flock
// separates the flock and fcntl namespaces, which never conflict.
message PosixLockRange {
uint64 start = 1;
uint64 end = 2;
uint32 type = 3; // 1=read, 2=write, 3=unlock
uint64 sid = 4;
uint64 owner = 5;
uint32 pid = 6; // holder pid, for get_lk reporting only
bool is_flock = 7;
}
// PosixLock routes an advisory lock operation to the inode's owner filer, which
// holds the authoritative in-memory lock table. key is the inode identity ring
// key (the file path, or hl:<HardLinkId> for a hardlink) used both to resolve the
// owner and to index the table. A non-owner filer forwards the request one hop;
// is_moved bounds it so a stale ring view cannot loop.
message PosixLockRequest {
string key = 1;
bool is_moved = 2;
PosixLockOp op = 3;
PosixLockRange lock = 4;
repeated PosixLockRange locks = 5;
bool cooling_probe = 6;
}
enum PosixLockOp {
TRY_LOCK = 0; // grant lock or report conflict (non-blocking)
UNLOCK = 1; // release lock's owner's locks over its range
GET_LK = 2; // report a conflicting lock, if any
RELEASE_POSIX_OWNER = 3; // drop the owner's fcntl locks (flush-time)
RELEASE_FLOCK_OWNER = 4; // drop the owner's flock locks (release-time)
KEEP_ALIVE = 5; // renew the session's lease on this owner (lock.sid)
}
message PosixLockResponse {
bool granted = 1; // for TRY_LOCK: whether the lock was granted
bool has_conflict = 2; // whether conflict is populated
PosixLockRange conflict = 3; // the blocking lock (TRY_LOCK conflict / GET_LK result)
}
// ObjectTransactionBatch applies several object transactions in one round trip,
// each under its own per-path lock and independent of the others (no cross-key
// atomicity). A caller groups keys that route to the same owner filer and sends
// one batch per owner, e.g. for a multi-object delete. Each response is parallel
// to its request.
message ObjectTransactionBatchRequest {
repeated ObjectTransactionRequest transactions = 1;
}
message ObjectTransactionBatchResponse {
repeated ObjectTransactionResponse responses = 1;
}
message CreateEntryResponse {
@@ -606,7 +421,6 @@ message SubscribeMetadataRequest {
repeated string directories = 10; // exact directory to watch
bool client_supports_batching = 11; // client can unpack SubscribeMetadataResponse.events
bool client_supports_metadata_chunks = 12; // client can read log file chunks from volume servers
bool client_supports_idle_heartbeat = 13; // server may send empty responses carrying the current time while the client is caught up
}
message SubscribeMetadataResponse {
string directory = 1;
+74 -27
View File
@@ -838,6 +838,8 @@ pub struct ReadQueryParams {
pub response_content_disposition: Option<String>,
/// Pretty print JSON response
pub pretty: Option<String>,
/// JSONP callback function name
pub callback: Option<String>,
}
// ============================================================================
@@ -3400,6 +3402,10 @@ fn json_response_with_params<T: Serialize>(
let is_pretty = params
.and_then(|params| params.pretty.as_ref())
.is_some_and(|value| !value.is_empty());
let callback = params
.and_then(|params| params.callback.as_ref())
.filter(|value| !value.is_empty())
.cloned();
let json_body = if is_pretty {
to_pretty_json(body)
@@ -3407,15 +3413,24 @@ fn json_response_with_params<T: Serialize>(
serde_json::to_string(body).unwrap()
};
Response::builder()
.status(status)
.header(header::CONTENT_TYPE, "application/json")
.header("X-Content-Type-Options", "nosniff")
.body(Body::from(json_body))
.unwrap()
if let Some(callback) = callback {
Response::builder()
.status(status)
.header(header::CONTENT_TYPE, "application/javascript")
.body(Body::from(format!("{}({})", callback, json_body)))
.unwrap()
} else {
Response::builder()
.status(status)
.header(header::CONTENT_TYPE, "application/json")
.body(Body::from(json_body))
.unwrap()
}
}
/// Return a JSON error response, honoring `?pretty=<any non-empty value>` for pretty-printed JSON.
/// Return a JSON error response with optional query string for pretty/JSONP support.
/// Supports `?pretty=<any non-empty value>` for pretty-printed JSON and `?callback=fn` for JSONP,
/// matching Go's writeJsonError behavior.
pub(super) fn json_error_with_query(
status: StatusCode,
msg: impl Into<String>,
@@ -3423,10 +3438,18 @@ pub(super) fn json_error_with_query(
) -> Response {
let body = serde_json::json!({"error": msg.into()});
let is_pretty = query.is_some_and(|q| {
q.split('&')
.any(|p| p.starts_with("pretty=") && p.len() > "pretty=".len())
});
let (is_pretty, callback) = if let Some(q) = query {
let pretty = q
.split('&')
.any(|p| p.starts_with("pretty=") && p.len() > "pretty=".len());
let cb = q
.split('&')
.find_map(|p| p.strip_prefix("callback="))
.map(|s| s.to_string());
(pretty, cb)
} else {
(false, None)
};
let json_body = if is_pretty {
to_pretty_json(&body)
@@ -3434,19 +3457,35 @@ pub(super) fn json_error_with_query(
serde_json::to_string(&body).unwrap()
};
Response::builder()
.status(status)
.header(header::CONTENT_TYPE, "application/json")
.header("X-Content-Type-Options", "nosniff")
.body(Body::from(json_body))
.unwrap()
if let Some(cb) = callback {
let jsonp = format!("{}({})", cb, json_body);
Response::builder()
.status(status)
.header(header::CONTENT_TYPE, "application/javascript")
.body(Body::from(jsonp))
.unwrap()
} else {
Response::builder()
.status(status)
.header(header::CONTENT_TYPE, "application/json")
.body(Body::from(json_body))
.unwrap()
}
}
/// Return a JSON response honoring `?pretty=<any non-empty value>` from a raw query string.
/// Return a JSON response with optional pretty/JSONP support from raw query string.
/// Matches Go's writeJsonQuiet behavior for write success responses.
fn json_result_with_query<T: Serialize>(status: StatusCode, body: &T, query: &str) -> Response {
let is_pretty = query
.split('&')
.any(|p| p.starts_with("pretty=") && p.len() > "pretty=".len());
let (is_pretty, callback) = {
let pretty = query
.split('&')
.any(|p| p.starts_with("pretty=") && p.len() > "pretty=".len());
let cb = query
.split('&')
.find_map(|p| p.strip_prefix("callback="))
.map(|s| s.to_string());
(pretty, cb)
};
let json_body = if is_pretty {
to_pretty_json(body)
@@ -3454,12 +3493,20 @@ fn json_result_with_query<T: Serialize>(status: StatusCode, body: &T, query: &st
serde_json::to_string(body).unwrap()
};
Response::builder()
.status(status)
.header(header::CONTENT_TYPE, "application/json")
.header("X-Content-Type-Options", "nosniff")
.body(Body::from(json_body))
.unwrap()
if let Some(cb) = callback {
let jsonp = format!("{}({})", cb, json_body);
Response::builder()
.status(status)
.header(header::CONTENT_TYPE, "application/javascript")
.body(Body::from(jsonp))
.unwrap()
} else {
Response::builder()
.status(status)
.header(header::CONTENT_TYPE, "application/json")
.body(Body::from(json_body))
.unwrap()
}
}
/// Extract JWT token from query param, Authorization header, or Cookie.
-232
View File
@@ -1,232 +0,0 @@
package fuse_dlm
import (
"errors"
"fmt"
"os"
"path/filepath"
"runtime"
"syscall"
"testing"
"time"
"github.com/seaweedfs/seaweedfs/weed/cluster/lock_manager"
"github.com/seaweedfs/seaweedfs/weed/pb"
"github.com/stretchr/testify/require"
)
// posixLockKey returns the routed-lock key a mount uses for a file at the mount
// root. Mounts run with -filer.path=/, so the file's filer path is "/"+name; the
// prefix must match mount.posixLockKeyForInode ("s3.fuse.lock:").
func posixLockKey(name string) string { return "s3.fuse.lock:/" + name }
// These tests exercise cross-mount POSIX advisory locks (flock), which the
// mounts route to the inode's owner filer because they run with -dlm
// (crossMountLocks() == lockClient != nil). They reuse the dlmTestCluster:
// master + volume + 2 filers (forming the lock ring) + 2 mounts on filer0.
//
// All flock opens are O_RDONLY on purpose: a write open (O_RDWR/O_WRONLY) would
// also take the -dlm whole-file write lock (held until close), which is a
// different mechanism — opening read-only isolates the POSIX advisory flock.
// requireForwardedLocks skips on platforms where the kernel does not forward
// advisory locks to the FUSE server. Only Linux forwards flock/fcntl (SETLK) to
// the filesystem; macFUSE handles flock in-kernel per mount, so cross-mount
// coordination can't be observed there even though the routed path is correct.
func requireForwardedLocks(t *testing.T) {
if runtime.GOOS != "linux" {
t.Skipf("advisory locks are only forwarded to the FUSE server on Linux (GOOS=%s)", runtime.GOOS)
}
}
// openFlock opens path read-only and takes a flock of type how (e.g. LOCK_EX).
// The file must already exist. The returned file holds the lock until closed.
func openFlock(path string, how int) (*os.File, error) {
f, err := os.OpenFile(path, os.O_RDONLY, 0)
if err != nil {
return nil, err
}
if err := syscall.Flock(int(f.Fd()), how); err != nil {
f.Close()
return nil, err
}
return f, nil
}
// createFile creates an empty file; the brief -dlm write lock it takes is
// released on close, before any flock test runs.
func createFile(t *testing.T, path string) {
t.Helper()
f, err := os.OpenFile(path, os.O_RDWR|os.O_CREATE|os.O_TRUNC, 0644)
require.NoError(t, err, "create %s", path)
require.NoError(t, f.Close())
}
// waitVisible waits for path to appear (cross-mount metadata propagation).
func waitVisible(t *testing.T, path string) {
t.Helper()
require.Eventually(t, func() bool {
_, err := os.Stat(path)
return err == nil
}, 15*time.Second, 200*time.Millisecond, "%s never became visible", path)
}
// tryExclusiveFlock attempts a non-blocking exclusive flock on path and
// classifies the outcome: acquired (and released again), blocked by another
// owner (EWOULDBLOCK/EAGAIN), or an unexpected error. It distinguishes a real
// "held by another" from incidental errors so the latter can't masquerade as a
// satisfied "must be blocked" assertion.
func tryExclusiveFlock(path string) (acquired, blocked bool, err error) {
f, err := os.OpenFile(path, os.O_RDONLY, 0)
if err != nil {
return false, false, err
}
defer f.Close()
if err := syscall.Flock(int(f.Fd()), syscall.LOCK_EX|syscall.LOCK_NB); err != nil {
if errors.Is(err, syscall.EWOULDBLOCK) || errors.Is(err, syscall.EAGAIN) {
return false, true, nil
}
return false, false, err
}
syscall.Flock(int(f.Fd()), syscall.LOCK_UN)
return true, false, nil
}
// requireEventuallyBlocked asserts path becomes held by another owner. It polls
// because a routed lock RPC can hit a transient EIO (a cold or forwarded gRPC
// call under load) even while the lock is genuinely held; it still fails if the
// lock is never blocked — whether it can be acquired (a double-grant) or errors
// persistently — so an incidental error can't masquerade as "held".
func requireEventuallyBlocked(t *testing.T, path string, timeout time.Duration, msg string) {
t.Helper()
require.Eventuallyf(t, func() bool { return isBlocked(path) },
timeout, 500*time.Millisecond, "%s", msg)
}
// isBlocked / isAcquirable are lenient predicates for polling, where transient
// errors during a migration are expected and simply mean "not yet".
func isBlocked(path string) bool { _, b, _ := tryExclusiveFlock(path); return b }
func isAcquirable(path string) bool {
a, _, _ := tryExclusiveFlock(path)
return a
}
// stopFiler stops filer idx and clears its command so cluster teardown does not
// try to stop it again.
func (c *dlmTestCluster) stopFiler(idx int) {
stopCmd(c.filerCmds[idx])
c.filerCmds[idx] = nil
}
// TestPosixLockCrossMount verifies a flock taken on one mount is seen by the
// other mount of the same cluster (the routed-to-owner-filer path end to end).
func TestPosixLockCrossMount(t *testing.T) {
requireForwardedLocks(t)
c := startDLMTestCluster(t)
name := "posix-xmount.lock"
path0 := filepath.Join(c.mountPoints[0], name)
path1 := filepath.Join(c.mountPoints[1], name)
createFile(t, path0)
waitVisible(t, path1)
held, err := openFlock(path0, syscall.LOCK_EX)
require.NoError(t, err, "mount0 should acquire the exclusive flock")
defer held.Close()
requireEventuallyBlocked(t, path1, 15*time.Second, "mount1 must be blocked while mount0 holds the flock")
require.NoError(t, syscall.Flock(int(held.Fd()), syscall.LOCK_UN))
require.Eventually(t, func() bool { return isAcquirable(path1) },
15*time.Second, 500*time.Millisecond,
"mount1 should acquire the flock after mount0 releases it")
}
// TestPosixLockSurvivesFilerLoss verifies advisory locks held across mounts
// survive a filer leaving the ring: ownership of the affected keys migrates to
// the surviving filer and the holding mount re-asserts them there, so the locks
// stay honored. It locks many files so that — with 2 filers — several are owned
// by filer1 and thus actually migrate when filer1 stops.
func TestPosixLockSurvivesFilerLoss(t *testing.T) {
requireForwardedLocks(t)
c := startDLMTestCluster(t)
const n = 12
// Select files via the same ring the filers run so the locked set provably
// spans both filers: filer1-owned keys must migrate when filer1 stops, while
// filer0-owned keys must keep working. Choosing by ownership (instead of
// hoping a sequential set happens to spread) keeps the migration path
// exercised on every run, independent of the cluster's dynamic ports.
ring := lock_manager.NewHashRing(lock_manager.DefaultVnodeCount)
ring.SetServers([]pb.ServerAddress{
pb.ServerAddress(c.filerAddress(0)),
pb.ServerAddress(c.filerAddress(1)),
})
filer1 := pb.ServerAddress(c.filerAddress(1))
var onFiler1, onFiler0 []string
for i := 0; (len(onFiler1) < n/2 || len(onFiler0) < n/2) && i < 1000; i++ {
name := fmt.Sprintf("posix-migrate-%d.lock", i)
if ring.GetPrimary(posixLockKey(name)) == filer1 {
onFiler1 = append(onFiler1, name)
} else {
onFiler0 = append(onFiler0, name)
}
}
require.GreaterOrEqualf(t, len(onFiler1), n/2, "need %d filer1-owned keys to exercise migration", n/2)
require.GreaterOrEqualf(t, len(onFiler0), n/2, "need %d filer0-owned keys", n/2)
names := append(onFiler1[:n/2:n/2], onFiler0[:n/2]...)
held := make([]*os.File, n)
for i, name := range names {
createFile(t, filepath.Join(c.mountPoints[0], name))
waitVisible(t, filepath.Join(c.mountPoints[1], name))
f, err := openFlock(filepath.Join(c.mountPoints[0], name), syscall.LOCK_EX)
require.NoError(t, err, "mount0 should acquire flock %s", name)
held[i] = f
defer held[i].Close()
}
require.Eventually(t, func() bool {
for _, name := range names {
if !isBlocked(filepath.Join(c.mountPoints[1], name)) {
return false
}
}
return true
}, 20*time.Second, time.Second,
"every lock must be held on mount1 before the ring change")
// Drop filer1 from the ring; keys it owned migrate to filer0.
c.stopFiler(1)
require.NoError(t, c.waitForFilerCount(1, 30*time.Second), "ring should drop to one filer")
// Poll until every lock is honored again on the surviving filer: the ring
// must propagate and the holding mount must re-assert its locks (keepalive is
// 5s). We assert only the settled state — the transient migration window is
// covered by unit tests.
require.Eventually(t, func() bool {
for _, name := range names {
if !isBlocked(filepath.Join(c.mountPoints[1], name)) {
return false
}
}
return true
}, 45*time.Second, time.Second,
"all locks must survive filer1 leaving the ring (migrate to filer0)")
// Releasing on mount0 frees them all: mount1 can then acquire each.
for _, f := range held {
require.NoError(t, syscall.Flock(int(f.Fd()), syscall.LOCK_UN))
}
require.Eventually(t, func() bool {
for _, name := range names {
if !isAcquirable(filepath.Join(c.mountPoints[1], name)) {
return false
}
}
return true
}, 20*time.Second, 500*time.Millisecond,
"mount1 should acquire every lock after mount0 releases them post-migration")
}
@@ -177,10 +177,8 @@ func testConcurrentReadWrite(t *testing.T, framework *FuseTestFramework) {
defer wg.Done()
for j := 0; j < 10; j++ {
if err := retryTransientFUSE(func() error {
_, e := os.ReadFile(mountPath)
return e
}); err != nil {
_, err := os.ReadFile(mountPath)
if err != nil {
addError(fmt.Errorf("reader %d: %v", readerID, err))
return
}
@@ -198,9 +196,8 @@ func testConcurrentReadWrite(t *testing.T, framework *FuseTestFramework) {
for j := 0; j < 5; j++ {
newData := bytes.Repeat([]byte(fmt.Sprintf("WRITER%d", writerID)), 1000)
if err := retryTransientFUSE(func() error {
return os.WriteFile(mountPath, newData, 0644)
}); err != nil {
err := os.WriteFile(mountPath, newData, 0644)
if err != nil {
addError(fmt.Errorf("writer %d: %v", writerID, err))
return
}
@@ -216,21 +213,6 @@ func testConcurrentReadWrite(t *testing.T, framework *FuseTestFramework) {
framework.AssertFileExists(filename)
}
// retryTransientFUSE retries op a few times before giving up. A concurrent
// truncating overwrite can leave a short-lived dentry/cache window where the
// entry is momentarily invisible (ENOENT) to another opener; the last error is
// returned so a genuine, persistent failure still surfaces.
func retryTransientFUSE(op func() error) error {
var err error
for attempt := 0; attempt < 5; attempt++ {
if err = op(); err == nil {
return nil
}
time.Sleep(100 * time.Millisecond)
}
return err
}
// testConcurrentDirectoryOperations tests concurrent directory operations
func testConcurrentDirectoryOperations(t *testing.T, framework *FuseTestFramework) {
numWorkers := 8
-48
View File
@@ -298,54 +298,6 @@ User Request → Load Balancer → Any S3 Gateway Instance
Allow/Deny Request
```
## Trust Policy Conditions
Step 5 above evaluates the role's trust policy against context keys derived from the
OIDC token's claims. The available keys are:
| Condition key | Source |
|---------------|--------|
| `oidc:iss` | `iss` claim (issuer URL) |
| `oidc:sub` | `sub` claim |
| `oidc:aud` | `aud` claim |
| `oidc:<claim>` | any other token claim, e.g. `oidc:roles`, `oidc:groups`, `oidc:email` |
| `aws:FederatedProvider` | the provider `name` (e.g. `keycloak-oidc`) when its configured issuer matches the token, otherwise the raw issuer URL |
| `aws:userid` | `sub` claim (same value as `oidc:sub` during trust-policy evaluation) |
| `sts:DurationSeconds` | requested session duration, when supplied |
During trust-policy evaluation `aws:userid` is the raw `sub` claim. Once the
role has been assumed, the keys seen by request authorization differ: there
`aws:userid` is a stable per-identity hash of `sub` and `iss` (see
`ComputeParentUser`), so do not assume the two contexts carry the same value.
Custom claims are always exposed under the `oidc:` prefix, so a trust policy must use
`oidc:roles` (not a bare `roles`) to match a `roles` claim:
```json
"Condition": {
"StringEquals": {
"oidc:roles": "s3-admin"
}
}
```
A multi-valued claim (such as a `roles` array) matches when any of its values equals
the condition value. The same `oidc:` keys can be interpolated into policy resources,
e.g. `arn:aws:s3:::bucket/${oidc:sub}/*`.
### roleMapping vs. trust policy
A provider's `roleMapping` and a role's trust policy apply to two different entry
points and are not interchangeable:
- **Direct OIDC** — an S3 request carrying `Authorization: Bearer <OIDC-JWT>`. The
gateway applies `roleMapping` to choose the caller's role from the token claims; the
first matching rule (or `defaultRole`) wins.
- **STS `AssumeRoleWithWebIdentity`** — the caller names the role explicitly via
`RoleArn`, and that role's trust policy decides whether the assumption is allowed.
`roleMapping` does not select the role on this path; instead the token claims are
surfaced as the `oidc:` condition keys above for the trust policy to evaluate.
## Configuration Management
### Development Environment
+3 -3
View File
@@ -37,7 +37,7 @@
"Action": ["sts:AssumeRoleWithWebIdentity"],
"Condition": {
"StringEquals": {
"oidc:roles": "s3-admin"
"roles": "s3-admin"
}
}
}
@@ -60,7 +60,7 @@
"Action": ["sts:AssumeRoleWithWebIdentity"],
"Condition": {
"StringEquals": {
"oidc:roles": "s3-read-only"
"roles": "s3-read-only"
}
}
}
@@ -83,7 +83,7 @@
"Action": ["sts:AssumeRoleWithWebIdentity"],
"Condition": {
"StringEquals": {
"oidc:roles": "s3-read-write"
"roles": "s3-read-write"
}
}
}
+1 -106
View File
@@ -22,7 +22,6 @@ import (
"os"
"os/exec"
"strings"
"sync"
"testing"
"time"
@@ -70,111 +69,7 @@ func s3Client(t *testing.T) *s3.Client {
})),
)
require.NoError(t, err)
client := s3.NewFromConfig(cfg, func(o *s3.Options) { o.UsePathStyle = true })
ensureClusterWritable(t, client)
return client
}
var clusterWritableOnce sync.Once
// ensureClusterWritable blocks until the cluster can actually serve a write,
// absorbing the volume-growth warmup window after a fresh start. The Makefile
// only waits for the server process to be up ("server up after N s"); it does
// not wait for a writable volume, so the first PutObject can race volume growth
// and fail with a transient 500 (assign volume: DeadlineExceeded) — the source
// of the lifecycle-test flakes. Probing one throwaway write here, once per
// process, warms growth so every test's first real write is past that window.
// Best-effort: if it never becomes writable, the test's own PutObject surfaces
// the failure normally.
func ensureClusterWritable(t *testing.T, c *s3.Client) {
t.Helper()
clusterWritableOnce.Do(func() {
bucket := uniqueBucket("warmup")
deadline := time.Now().Add(60 * time.Second)
// try runs fn under a bounded context so a single hung call can't block.
try := func(timeout time.Duration, fn func(ctx context.Context) error) error {
ctx, cancel := context.WithTimeout(context.Background(), timeout)
defer cancel()
return fn(ctx)
}
// probe is a try whose timeout is clamped to the time left before the
// deadline, so the whole warmup stays within the budget; ok=false means
// the budget is exhausted and the caller should stop.
probe := func(fn func(ctx context.Context) error) (err error, ok bool) {
remaining := time.Until(deadline)
if remaining <= 0 {
return nil, false
}
if remaining > 10*time.Second {
remaining = 10 * time.Second
}
return try(remaining, fn), true
}
// backoff sleeps attempt*250ms, never past the deadline.
backoff := func(attempt int) {
d := time.Duration(attempt) * 250 * time.Millisecond
if left := time.Until(deadline); d > left {
d = left
}
if d > 0 {
time.Sleep(d)
}
}
// CreateBucket is a metadata op, but on a cold cluster the filer itself may
// not be ready yet, so retry it within the deadline rather than abandoning
// the whole warmup (and the PutObject probe) on the first error.
created := false
for attempt := 1; ; attempt++ {
err, ok := probe(func(ctx context.Context) error {
_, e := c.CreateBucket(ctx, &s3.CreateBucketInput{Bucket: aws.String(bucket)})
return e
})
if !ok {
break
}
if err == nil {
created = true
break
}
backoff(attempt)
}
if !created {
t.Logf("warmup: could not create probe bucket within 60s; proceeding")
return
}
// Cleanup gets a fresh timeout (not the warmup budget) so teardown runs
// even when the probe loop used the full window.
defer try(10*time.Second, func(ctx context.Context) error {
_, e := c.DeleteBucket(ctx, &s3.DeleteBucketInput{Bucket: aws.String(bucket)})
return e
})
for attempt := 1; ; attempt++ {
err, ok := probe(func(ctx context.Context) error {
_, e := c.PutObject(ctx, &s3.PutObjectInput{
Bucket: aws.String(bucket), Key: aws.String("warmup"), Body: strings.NewReader("ok"),
})
return e
})
if !ok {
break
}
if err == nil {
try(10*time.Second, func(ctx context.Context) error {
_, e := c.DeleteObject(ctx, &s3.DeleteObjectInput{Bucket: aws.String(bucket), Key: aws.String("warmup")})
return e
})
if attempt > 1 {
t.Logf("cluster became writable after %d probe(s)", attempt)
}
return
}
backoff(attempt)
}
t.Logf("warmup: cluster not confirmed writable within 60s; proceeding")
})
return s3.NewFromConfig(cfg, func(o *s3.Options) { o.UsePathStyle = true })
}
func filerClient(t *testing.T) (filer_pb.SeaweedFilerClient, func()) {
+2 -25
View File
@@ -11,29 +11,6 @@ import (
// GrpcPortOffset is the offset weed mini uses to derive gRPC ports from HTTP ports.
const GrpcPortOffset = 10000
// miniDefaultPorts are the weed mini flag defaults (see weed/command/mini.go).
// A test only overrides services it uses; unspecified services still bind
// these defaults, so allocation must avoid handing them out (or any value
// whose gRPC offset would collide with them).
var miniDefaultPorts = []int{
9333, // master.port
8888, // filer.port
9340, // volume.port
8333, // s3.port
8181, // s3.port.iceberg
7333, // webdav.port
23646, // admin.port
}
func reservedMiniPorts() map[int]bool {
r := make(map[int]bool, len(miniDefaultPorts)*2)
for _, p := range miniDefaultPorts {
r[p] = true
r[p+GrpcPortOffset] = true
}
return r
}
// AllocatePorts allocates count unique free ports atomically.
// All listeners are held open until every port is obtained, preventing
// the OS from recycling a port between successive allocations.
@@ -86,7 +63,7 @@ func AllocateMiniPorts(count int) ([]int, error) {
minPort = 10000
maxPort = 55000
)
reserved := reservedMiniPorts()
reserved := make(map[int]bool)
ports := make([]int, 0, count)
var listeners []net.Listener
defer func() {
@@ -158,7 +135,7 @@ func AllocatePortSet(miniCount, regularCount int) (mini []int, regular []int, er
minPort = 10000
maxPort = 55000
)
reserved := reservedMiniPorts()
reserved := make(map[int]bool)
mini = make([]int, 0, miniCount)
var listeners []net.Listener
defer func() {
-23
View File
@@ -2,29 +2,6 @@ package testutil
import "testing"
// AllocateMiniPorts must never hand out a port that weed mini will reserve
// for one of its default services (or that default's gRPC offset). A real
// failure: Filer was given 33646 (Admin default 23646 + GrpcPortOffset),
// which mini then refused as "reserved for gRPC calculation".
func TestAllocateMiniPortsAvoidsMiniDefaults(t *testing.T) {
reserved := reservedMiniPorts()
for iter := 0; iter < 200; iter++ {
ports, err := AllocateMiniPorts(4)
if err != nil {
t.Fatalf("iter %d: AllocateMiniPorts: %v", iter, err)
}
for _, p := range ports {
if reserved[p] {
t.Fatalf("iter %d: allocated port %d is a mini default (or gRPC offset)", iter, p)
}
if reserved[p+GrpcPortOffset] {
t.Fatalf("iter %d: allocated port %d has gRPC offset %d colliding with a mini default",
iter, p, p+GrpcPortOffset)
}
}
}
}
func TestAllocatePortSetNoGrpcCollision(t *testing.T) {
// Run a few iterations to catch the OS-recycles-just-closed-port race
// that previously hit regular ports when the mini gRPC offset was freed
@@ -75,7 +75,7 @@ func TestStatsEndpoints(t *testing.T) {
}
}
func TestStatusPrettyJsonAndCallbackIgnored(t *testing.T) {
func TestStatusPrettyJsonAndJsonp(t *testing.T) {
if testing.Short() {
t.Skip("skipping integration test in short mode")
}
@@ -93,29 +93,29 @@ func TestStatusPrettyJsonAndCallbackIgnored(t *testing.T) {
if len(lines) < 3 {
t.Fatalf("/status?pretty=y expected multi-line indented JSON, got %d lines: %s", len(lines), string(prettyBody))
}
// Verify the body is valid JSON
var prettyPayload map[string]interface{}
if err := json.Unmarshal(prettyBody, &prettyPayload); err != nil {
t.Fatalf("/status?pretty=y is not valid JSON: %v", err)
}
// ?callback=myFunc — must be ignored; response is plain JSON with nosniff.
cbResp := framework.DoRequest(t, client, mustNewRequest(t, http.MethodGet, cluster.VolumeAdminURL()+"/status?callback=myFunc"))
cbBody := framework.ReadAllAndClose(t, cbResp)
if cbResp.StatusCode != http.StatusOK {
t.Fatalf("/status?callback=myFunc expected 200, got %d", cbResp.StatusCode)
// ?callback=myFunc — expect JSONP wrapping
jsonpResp := framework.DoRequest(t, client, mustNewRequest(t, http.MethodGet, cluster.VolumeAdminURL()+"/status?callback=myFunc"))
jsonpBody := framework.ReadAllAndClose(t, jsonpResp)
if jsonpResp.StatusCode != http.StatusOK {
t.Fatalf("/status?callback=myFunc expected 200, got %d", jsonpResp.StatusCode)
}
if ct := cbResp.Header.Get("Content-Type"); !strings.Contains(ct, "application/json") {
t.Fatalf("/status?callback=myFunc expected Content-Type application/json, got %q", ct)
bodyStr := string(jsonpBody)
if !strings.HasPrefix(bodyStr, "myFunc(") {
t.Fatalf("/status?callback=myFunc expected body to start with 'myFunc(', got prefix: %q", bodyStr[:min(len(bodyStr), 30)])
}
if nosniff := cbResp.Header.Get("X-Content-Type-Options"); nosniff != "nosniff" {
t.Fatalf("/status?callback=myFunc expected X-Content-Type-Options nosniff, got %q", nosniff)
trimmed := strings.TrimRight(bodyStr, "\n; ")
if !strings.HasSuffix(trimmed, ")") {
t.Fatalf("/status?callback=myFunc expected body to end with ')', got suffix: %q", trimmed[max(0, len(trimmed)-10):])
}
if strings.Contains(string(cbBody), "myFunc(") {
t.Fatalf("/status?callback=myFunc must not wrap response in callback; body: %q", string(cbBody))
}
var cbPayload map[string]interface{}
if err := json.Unmarshal(cbBody, &cbPayload); err != nil {
t.Fatalf("/status?callback=myFunc is not valid JSON: %v", err)
// Content-Type should be application/javascript for JSONP
if ct := jsonpResp.Header.Get("Content-Type"); !strings.Contains(ct, "javascript") {
t.Fatalf("/status?callback=myFunc expected Content-Type containing 'javascript', got %q", ct)
}
}
-52
View File
@@ -23,7 +23,6 @@ import (
"github.com/seaweedfs/seaweedfs/weed/pb/plugin_pb"
"github.com/seaweedfs/seaweedfs/weed/pb/schema_pb"
"github.com/seaweedfs/seaweedfs/weed/security"
stats_collect "github.com/seaweedfs/seaweedfs/weed/stats"
"github.com/seaweedfs/seaweedfs/weed/storage/erasure_coding"
"github.com/seaweedfs/seaweedfs/weed/storage/super_block"
"github.com/seaweedfs/seaweedfs/weed/util"
@@ -281,8 +280,6 @@ func NewAdminServer(masters string, templateFS http.FileSystem, dataDir string,
go server.monitorVacuumWorker(bgCtx)
}
go server.publishMaintenanceMetrics(bgCtx)
return server
}
@@ -367,55 +364,6 @@ func (s *AdminServer) monitorVacuumWorker(ctx context.Context) {
}
}
// publishMaintenanceMetrics periodically snapshots the maintenance queue and
// worker fleet into Prometheus gauges. Counters and durations are recorded at
// their event sites; these gauges reflect current state at scrape resolution.
func (s *AdminServer) publishMaintenanceMetrics(ctx context.Context) {
const interval = 15 * time.Second
ticker := time.NewTicker(interval)
defer ticker.Stop()
for {
select {
case <-ctx.Done():
return
case <-ticker.C:
s.collectMaintenanceMetrics()
}
}
}
func (s *AdminServer) collectMaintenanceMetrics() {
if s.maintenanceManager == nil {
return
}
stats := s.maintenanceManager.GetStats()
stats_collect.AdminMaintenanceTasksByStatus.Reset()
for status, count := range stats.TasksByStatus {
stats_collect.AdminMaintenanceTasksByStatus.WithLabelValues(string(status)).Set(float64(count))
}
stats_collect.AdminMaintenanceTasksByType.Reset()
for taskType, count := range stats.TasksByType {
stats_collect.AdminMaintenanceTasksByType.WithLabelValues(string(taskType)).Set(float64(count))
}
// NextScanTime is only meaningful while the scanner runs; GetStats computes
// it unconditionally, so clear the gauge when idle to avoid a stale value.
if s.maintenanceManager.IsRunning() && !stats.NextScanTime.IsZero() {
stats_collect.AdminMaintenanceNextScanTimestampSeconds.Set(float64(stats.NextScanTime.Unix()))
} else {
stats_collect.AdminMaintenanceNextScanTimestampSeconds.Set(0)
}
workers, usedSlots, maxSlots := s.maintenanceManager.GetWorkerSlotTotals()
stats_collect.AdminWorkersConnected.Set(float64(workers))
stats_collect.AdminWorkerSlots.WithLabelValues("used").Set(float64(usedSlots))
stats_collect.AdminWorkerSlots.WithLabelValues("max").Set(float64(maxSlots))
}
// loadTaskConfigurationsFromPersistence loads saved task configurations from protobuf files
func (s *AdminServer) loadTaskConfigurationsFromPersistence() {
if s.configPersistence == nil || !s.configPersistence.IsConfigured() {
+6 -9
View File
@@ -16,7 +16,6 @@ import (
"github.com/seaweedfs/seaweedfs/weed/pb/plugin_pb"
"github.com/seaweedfs/seaweedfs/weed/pb/worker_pb"
"github.com/seaweedfs/seaweedfs/weed/security"
stats_collect "github.com/seaweedfs/seaweedfs/weed/stats"
"github.com/seaweedfs/seaweedfs/weed/util"
"google.golang.org/grpc"
"google.golang.org/grpc/codes"
@@ -234,7 +233,6 @@ func (s *WorkerGrpcServer) WorkerStream(stream worker_pb.WorkerService_WorkerStr
}
s.connections[workerID] = conn
s.connMutex.Unlock()
stats_collect.AdminWorkerEventsTotal.WithLabelValues("registered").Inc()
// Register worker with maintenance manager
s.registerWorkerWithManager(conn)
@@ -267,11 +265,11 @@ func (s *WorkerGrpcServer) WorkerStream(stream worker_pb.WorkerService_WorkerStr
select {
case <-ctx.Done():
glog.Infof("Worker %s connection closed: %v", workerID, ctx.Err())
s.unregisterWorker(conn, "unregistered")
s.unregisterWorker(conn)
return nil
case <-connCtx.Done():
glog.Infof("Worker %s connection cancelled", workerID)
s.unregisterWorker(conn, "unregistered")
s.unregisterWorker(conn)
return nil
default:
}
@@ -287,7 +285,7 @@ func (s *WorkerGrpcServer) WorkerStream(stream worker_pb.WorkerService_WorkerStr
default:
glog.Errorf("Error receiving from worker %s: %v", workerID, err)
}
s.unregisterWorker(conn, "unregistered")
s.unregisterWorker(conn)
return err
}
@@ -340,7 +338,7 @@ func (s *WorkerGrpcServer) handleWorkerMessage(conn *WorkerConnection, msg *work
case *worker_pb.WorkerMessage_Shutdown:
glog.Infof("Worker %s shutting down: %s", workerID, m.Shutdown.Reason)
s.unregisterWorker(conn, "unregistered")
s.unregisterWorker(conn)
default:
glog.Warningf("Unknown message type from worker %s", workerID)
@@ -607,7 +605,7 @@ func (s *WorkerGrpcServer) safeCloseOutgoingChannel(conn *WorkerConnection, sour
}
// unregisterWorker removes a worker connection
func (s *WorkerGrpcServer) unregisterWorker(conn *WorkerConnection, event string) {
func (s *WorkerGrpcServer) unregisterWorker(conn *WorkerConnection) {
s.connMutex.Lock()
existingConn, exists := s.connections[conn.workerID]
if !exists {
@@ -626,7 +624,6 @@ func (s *WorkerGrpcServer) unregisterWorker(conn *WorkerConnection, event string
// Remove from map first to prevent duplicate cleanup attempts
delete(s.connections, conn.workerID)
s.connMutex.Unlock()
stats_collect.AdminWorkerEventsTotal.WithLabelValues(event).Inc()
// Cancel context to signal goroutines to stop
conn.cancel()
@@ -668,7 +665,7 @@ func (s *WorkerGrpcServer) cleanupStaleConnections() {
for _, conn := range toRemove {
glog.Warningf("Cleaning up stale worker connection: %s", conn.workerID)
s.unregisterWorker(conn, "stale_removed")
s.unregisterWorker(conn)
}
}
@@ -8,7 +8,6 @@ import (
"github.com/seaweedfs/seaweedfs/weed/glog"
"github.com/seaweedfs/seaweedfs/weed/pb/worker_pb"
stats_collect "github.com/seaweedfs/seaweedfs/weed/stats"
"github.com/seaweedfs/seaweedfs/weed/worker/tasks/balance"
"github.com/seaweedfs/seaweedfs/weed/worker/tasks/erasure_coding"
"github.com/seaweedfs/seaweedfs/weed/worker/tasks/vacuum"
@@ -316,7 +315,6 @@ func (mm *MaintenanceManager) performScan() {
glog.Infof("Starting maintenance scan...")
results, err := mm.scanner.ScanForMaintenanceTasks()
stats_collect.AdminMaintenanceLastScanTimestampSeconds.SetToCurrentTime()
if err != nil {
// Handle scan error
mm.mutex.Lock()
@@ -520,11 +518,6 @@ func (mm *MaintenanceManager) GetWorkers() []*MaintenanceWorker {
return mm.queue.GetWorkers()
}
// GetWorkerSlotTotals returns worker count and aggregate used/max task slots.
func (mm *MaintenanceManager) GetWorkerSlotTotals() (workers, used, max int) {
return mm.queue.GetWorkerSlotTotals()
}
// TriggerScan manually triggers a maintenance scan
func (mm *MaintenanceManager) TriggerScan() error {
return mm.triggerScanInternal(true)
+1 -30
View File
@@ -8,7 +8,6 @@ import (
"time"
"github.com/seaweedfs/seaweedfs/weed/glog"
stats_collect "github.com/seaweedfs/seaweedfs/weed/stats"
)
// NewMaintenanceQueue creates a new maintenance queue
@@ -483,14 +482,12 @@ func (mq *MaintenanceQueue) CompleteTask(taskID string, error string) {
// Calculate task duration
var duration time.Duration
hadStart := task.StartedAt != nil
if hadStart {
if task.StartedAt != nil {
duration = completedTime.Sub(*task.StartedAt)
}
// Capture workerID before it may be cleared during retry
originalWorkerID := task.WorkerID
taskType := string(task.Type)
var taskToSave *MaintenanceTask
var logFn func()
@@ -580,18 +577,6 @@ func (mq *MaintenanceQueue) CompleteTask(taskID string, error string) {
}
mq.mutex.Unlock()
// Record terminal-state metrics. A retry leaves the task pending, so it
// is not counted as completed or failed here.
switch taskStatus {
case TaskStatusCompleted:
stats_collect.AdminMaintenanceTasksCompletedTotal.WithLabelValues(taskType, "completed").Inc()
case TaskStatusFailed:
stats_collect.AdminMaintenanceTasksCompletedTotal.WithLabelValues(taskType, "failed").Inc()
}
if hadStart && (taskStatus == TaskStatusCompleted || taskStatus == TaskStatusFailed) {
stats_collect.AdminMaintenanceTaskDurationSeconds.WithLabelValues(taskType).Observe(duration.Seconds())
}
// Only persist non-terminal tasks (retries). Completed/failed tasks stay
// in memory for the UI but are not written to disk — they would just
// accumulate and slow down future startups.
@@ -834,20 +819,6 @@ func (mq *MaintenanceQueue) GetWorkers() []*MaintenanceWorker {
return workers
}
// GetWorkerSlotTotals aggregates worker count and used/max task slots under the
// lock, so callers don't read live worker fields that task updates mutate.
func (mq *MaintenanceQueue) GetWorkerSlotTotals() (workers, used, max int) {
mq.mutex.RLock()
defer mq.mutex.RUnlock()
for _, worker := range mq.workers {
workers++
used += worker.CurrentLoad
max += worker.MaxConcurrent
}
return
}
// generateTaskID generates a unique ID for tasks
func generateTaskID() string {
const charset = "abcdefghijklmnopqrstuvwxyz0123456789"
-64
View File
@@ -54,70 +54,6 @@ func (at *ActiveTopology) GetEffectiveAvailableCapacityDetailed(nodeID string, d
return at.getEffectiveAvailableCapacityUnsafe(disk)
}
// GetEffectiveAvailableEcShardSlots returns a disk's free EC shard slots,
// accounting for in-flight task reservations at shard granularity. Unlike the
// volume-slot views (GetDisksWithEffectiveCapacity / GetEffectiveAvailableCapacity),
// this does not truncate sub-volume shard reservations: it subtracts the full
// reservation impact (volume slots converted to shard slots, plus the raw shard
// slots) so a reservation that is not a whole multiple of ShardsPerVolumeSlot is
// not lost. It does NOT subtract the EC shards already persisted on the disk;
// callers that track those (from EcShardInfos) subtract them separately.
//
// shardsPerVolume is the number of EC shards of the target collection that fit in
// one volume slot (i.e. its data-shard count): a 4+2 volume's shards are ~1/4 of a
// volume each, so one volume slot holds 4 of them, not the default
// ShardsPerVolumeSlot. Pass <= 0 to use the default. Using the target ratio keeps
// Place from over-filling a disk for low-data-shard layouts.
func (at *ActiveTopology) GetEffectiveAvailableEcShardSlots(nodeID string, diskID uint32, shardsPerVolume int) int {
if shardsPerVolume <= 0 {
shardsPerVolume = ShardsPerVolumeSlot
}
at.mutex.RLock()
defer at.mutex.RUnlock()
diskKey := fmt.Sprintf("%s:%d", nodeID, diskID)
disk, exists := at.disks[diskKey]
if !exists || disk.DiskInfo == nil || disk.DiskInfo.DiskInfo == nil {
return 0
}
info := disk.DiskInfo.DiskInfo
base := info.MaxVolumeCount - info.VolumeCount
if base <= 0 && info.MaxVolumeCount == 0 && info.VolumeCount == 0 &&
len(info.VolumeInfos) == 0 && len(info.EcShardInfos) == 0 {
// Freshly started empty servers can report max=0 before publishing concrete
// limits; keep one provisional slot so EC placement still sees the disk,
// mirroring getEffectiveAvailableCapacityUnsafe.
base = 1
}
if base < 0 {
base = 0
}
// calculateTaskStorageImpact reports consumption as positive, so subtract it.
// Volume-slot reservations scale by the target ratio; the sub-volume shard-slot
// remainder is in default units and subtracted as-is (a small approximation).
impact := at.getEffectiveCapacityUnsafe(disk)
// impact.ShardSlots is recorded in default ShardsPerVolumeSlot units; convert it
// to the target ratio's shard slots before subtracting (identity when
// shardsPerVolume == ShardsPerVolumeSlot). Round a positive reservation up so a
// sub-slot reservation (e.g. 1 default slot against a 4-shard target) is not
// truncated to zero and wrongly counted as free.
scaledShardImpact := int64(impact.ShardSlots) * int64(shardsPerVolume)
if scaledShardImpact > 0 {
scaledShardImpact = (scaledShardImpact + int64(ShardsPerVolumeSlot) - 1) / int64(ShardsPerVolumeSlot)
} else {
scaledShardImpact /= int64(ShardsPerVolumeSlot)
}
free := base*int64(shardsPerVolume) -
int64(impact.VolumeSlots)*int64(shardsPerVolume) -
scaledShardImpact
if free < 0 {
free = 0
}
return int(free)
}
// GetEffectiveCapacityImpact returns the StorageSlotChange impact for a disk
// This shows the net impact from all pending and assigned tasks
func (at *ActiveTopology) GetEffectiveCapacityImpact(nodeID string, diskID uint32) StorageSlotChange {
+3 -55
View File
@@ -4,7 +4,6 @@ import (
"context"
"fmt"
"strings"
"sync"
"sync/atomic"
"time"
@@ -20,14 +19,6 @@ type LockClient struct {
maxLockDuration time.Duration
sleepDuration time.Duration
seedFiler pb.ServerAddress
// ring is an optional client-side view of the filer lock hash ring. When
// populated, a new lock starts at the key's primary filer instead of the
// seed filer, avoiding the seed->primary forward hop. A stale view stays
// correct: the filer forwards to the real primary as a fallback.
ringMu sync.RWMutex
ring *lock_manager.HashRing
ringVersion int64
}
func NewLockClient(grpcDialOption grpc.DialOption, seedFiler pb.ServerAddress) *LockClient {
@@ -39,49 +30,6 @@ func NewLockClient(grpcDialOption grpc.DialOption, seedFiler pb.ServerAddress) *
}
}
// SetRing mirrors the master's LockRingUpdate so the client computes the same
// primary the filers do. A non-zero version at or below the current one is
// ignored once a ring exists, dropping reordered and redundant broadcasts;
// version 0 always applies (bootstrap).
func (lc *LockClient) SetRing(servers []pb.ServerAddress, version int64) {
lc.ringMu.Lock()
defer lc.ringMu.Unlock()
if version != 0 && version <= lc.ringVersion && lc.ring != nil {
return
}
lc.ringVersion = version
if lc.ring == nil {
lc.ring = lock_manager.NewHashRing(lock_manager.DefaultVnodeCount)
}
lc.ring.SetServers(servers)
}
// hostForKey returns the filer that should own key per the current ring view,
// falling back to the seed filer when no view has been received yet.
func (lc *LockClient) hostForKey(key string) pb.ServerAddress {
lc.ringMu.RLock()
defer lc.ringMu.RUnlock()
if lc.ring == nil {
return lc.seedFiler
}
if primary := lc.ring.GetPrimary(key); primary != "" {
return primary
}
return lc.seedFiler
}
// PrimaryForKey returns the ring owner for key, or "" before any ring arrives.
// Unlike hostForKey it does not fall back to the seed, so a route-by-key caller
// stays on the distributed lock until the ring is known.
func (lc *LockClient) PrimaryForKey(key string) pb.ServerAddress {
lc.ringMu.RLock()
defer lc.ringMu.RUnlock()
if lc.ring == nil {
return ""
}
return lc.ring.GetPrimary(key)
}
type LiveLock struct {
key string
renewToken string
@@ -103,7 +51,7 @@ type LiveLock struct {
func (lc *LockClient) NewShortLivedLock(key string, owner string) (lock *LiveLock) {
lock = &LiveLock{
key: key,
hostFiler: lc.hostForKey(key),
hostFiler: lc.seedFiler,
cancelCh: make(chan struct{}),
expireAtNs: time.Now().Add(5 * time.Second).UnixNano(),
grpcDialOption: lc.grpcDialOption,
@@ -124,7 +72,7 @@ func (lc *LockClient) NewBlockingLongLivedLock(key, owner string, lockTTL time.D
}
lock := &LiveLock{
key: key,
hostFiler: lc.hostForKey(key),
hostFiler: lc.seedFiler,
cancelCh: make(chan struct{}),
expireAtNs: time.Now().Add(lockTTL).UnixNano(),
grpcDialOption: lc.grpcDialOption,
@@ -162,7 +110,7 @@ func (lc *LockClient) NewBlockingLongLivedLock(key, owner string, lockTTL time.D
func (lc *LockClient) StartLongLivedLock(key string, owner string, onLockOwnerChange func(newLockOwner string), lockTTL time.Duration) (lock *LiveLock) {
lock = &LiveLock{
key: key,
hostFiler: lc.hostForKey(key),
hostFiler: lc.seedFiler,
cancelCh: make(chan struct{}),
expireAtNs: time.Now().Add(lockTTL).UnixNano(),
grpcDialOption: lc.grpcDialOption,
-92
View File
@@ -1,92 +0,0 @@
package cluster
import (
"testing"
"github.com/seaweedfs/seaweedfs/weed/cluster/lock_manager"
"github.com/seaweedfs/seaweedfs/weed/pb"
)
// The gateway must resolve a lock key to the same primary the filers do,
// otherwise it dials the wrong filer and the lock still gets forwarded. Both
// sides use the same HashRing over the same server set, so for every key the
// client's hostForKey must equal the filer ring's GetPrimary.
func TestLockClientHostMatchesFilerRing(t *testing.T) {
servers := []pb.ServerAddress{
"filer-a:8888", "filer-b:8888", "filer-c:8888", "filer-d:8888",
}
filerRing := lock_manager.NewHashRing(lock_manager.DefaultVnodeCount)
filerRing.SetServers(servers)
lc := NewLockClient(nil, "seed:8888")
lc.SetRing(servers, 1)
for _, key := range []string{
"s3.object.write:/buckets/b/obj-0",
"s3.object.write:/buckets/b/obj-1",
"s3.object.write:/buckets/b/obj-2",
"s3.object.write:/buckets/gosbench-0/w0obj-kilo-0877",
"some/other/key",
} {
if got, want := lc.hostForKey(key), filerRing.GetPrimary(key); got != want {
t.Errorf("key %q: client host %q != filer primary %q", key, got, want)
}
}
}
// Without a ring view, the client falls back to the seed filer (which the filer
// forwards from), preserving the pre-optimization behavior.
func TestLockClientHostFallsBackToSeed(t *testing.T) {
lc := NewLockClient(nil, "seed:8888")
if got := lc.hostForKey("any-key"); got != "seed:8888" {
t.Errorf("expected seed fallback, got %q", got)
}
// An empty ring (no members yet) also falls back to the seed.
lc.SetRing(nil, 1)
if got := lc.hostForKey("any-key"); got != "seed:8888" {
t.Errorf("expected seed fallback on empty ring, got %q", got)
}
}
// A stale (older-version) update must not regress a newer ring view, while
// version 0 always applies as a bootstrap.
func TestLockClientSetRingVersionGuard(t *testing.T) {
lc := NewLockClient(nil, "seed:8888")
newer := []pb.ServerAddress{"filer-a:8888", "filer-b:8888"}
lc.SetRing(newer, 10)
primaryAt10 := lc.hostForKey("k")
// Older version is ignored.
lc.SetRing([]pb.ServerAddress{"filer-z:8888"}, 5)
if got := lc.hostForKey("k"); got != primaryAt10 {
t.Errorf("stale update applied: host changed to %q", got)
}
// version 0 is always accepted.
lc.SetRing([]pb.ServerAddress{"filer-z:8888"}, 0)
if got := lc.hostForKey("k"); got != "filer-z:8888" {
t.Errorf("bootstrap update not applied, got %q", got)
}
}
// PrimaryForKey returns "" before any ring is received (so a route-by-key
// caller falls back to the distributed lock) and the ring owner afterwards,
// unlike hostForKey which falls back to the seed.
func TestLockClientPrimaryForKey(t *testing.T) {
lc := NewLockClient(nil, "seed:8888")
if got := lc.PrimaryForKey("k"); got != "" {
t.Errorf("expected empty before ring, got %q", got)
}
lc.SetRing([]pb.ServerAddress{"filer-a:8888", "filer-b:8888"}, 1)
got := lc.PrimaryForKey("k")
if got == "" {
t.Fatal("expected an owner after ring set")
}
if got != lc.hostForKey("k") {
t.Errorf("PrimaryForKey %q disagrees with hostForKey %q", got, lc.hostForKey("k"))
}
}
-8
View File
@@ -145,15 +145,7 @@ func (hr *HashRing) rebuildRing() {
hr.vnodeToServer = make(map[uint32]pb.ServerAddress, len(hr.servers)*hr.vnodeCount)
hr.sortedHashes = make([]uint32, 0, len(hr.servers)*hr.vnodeCount)
// Sort so a vnode-hash collision resolves to the same server on every node;
// map iteration order alone is randomized per process.
servers := make([]pb.ServerAddress, 0, len(hr.servers))
for server := range hr.servers {
servers = append(servers, server)
}
sort.Slice(servers, func(i, j int) bool { return servers[i] < servers[j] })
for _, server := range servers {
for i := 0; i < hr.vnodeCount; i++ {
vnodeKey := vnodeKeyFor(server, i)
hash := hashKey(vnodeKey)
@@ -171,38 +171,3 @@ func TestHashRing_GetPrimary(t *testing.T) {
primary, _ := hr.GetPrimaryAndBackup("mykey")
assert.Equal(t, primary, hr.GetPrimary("mykey"))
}
// The ring must be identical on every node holding the same server set,
// regardless of the order servers were added or supplied. Build rings several
// ways and assert they agree on the primary for a wide range of keys.
func TestHashRing_OrderIndependent(t *testing.T) {
servers := []pb.ServerAddress{
"filer-a:8888", "filer-b:8888", "filer-c:8888", "filer-d:8888", "filer-e:8888",
}
bySet := NewHashRing(50)
bySet.SetServers(servers)
byReverse := NewHashRing(50)
rev := append([]pb.ServerAddress(nil), servers...)
for i, j := 0, len(rev)-1; i < j; i, j = i+1, j-1 {
rev[i], rev[j] = rev[j], rev[i]
}
byReverse.SetServers(rev)
byAdd := NewHashRing(50)
for _, s := range []pb.ServerAddress{"filer-c:8888", "filer-e:8888", "filer-a:8888", "filer-d:8888", "filer-b:8888"} {
byAdd.AddServer(s)
}
for i := 0; i < 5000; i++ {
key := fmt.Sprintf("s3.object.write:/buckets/b/obj-%d", i)
p := bySet.GetPrimary(key)
if got := byReverse.GetPrimary(key); got != p {
t.Fatalf("reverse-order ring disagrees on %q: %s vs %s", key, got, p)
}
if got := byAdd.GetPrimary(key); got != p {
t.Fatalf("add-order ring disagrees on %q: %s vs %s", key, got, p)
}
}
}
+6 -40
View File
@@ -12,7 +12,6 @@ import (
type LockRingSnapshot struct {
servers []pb.ServerAddress
ts time.Time
ring *HashRing // prebuilt ring for servers, so PriorOwner need not rebuild per call
}
type LockRing struct {
@@ -59,12 +58,10 @@ func (r *LockRing) SetSnapshot(servers []pb.ServerAddress, version int64) bool {
// are always consistent — prevents a concurrent SetSnapshot from
// seeing the new version but applying its servers to the old ring.
r.Ring.SetServers(servers)
// Append the snapshot under the same lock as the ring update so a concurrent
// PriorOwner always sees snapshots[0] matching r.Ring (and snapshots[1] as the
// true prior); otherwise it could pair a new ring with a stale prior snapshot.
r.addOneSnapshotLocked(servers)
r.Unlock()
r.addOneSnapshot(servers)
r.cleanupWg.Add(1)
go func() {
defer r.cleanupWg.Done()
@@ -81,16 +78,14 @@ func (r *LockRing) Version() int64 {
return r.version
}
// addOneSnapshotLocked appends a new snapshot (newest at index 0). The caller
// must hold r.Lock(), so the ring update and snapshot append are one atomic step.
func (r *LockRing) addOneSnapshotLocked(servers []pb.ServerAddress) {
func (r *LockRing) addOneSnapshot(servers []pb.ServerAddress) {
r.Lock()
defer r.Unlock()
ts := time.Now()
ring := NewHashRing(DefaultVnodeCount)
ring.SetServers(servers)
t := &LockRingSnapshot{
servers: servers,
ts: ts,
ring: ring,
}
r.snapshots = append(r.snapshots, t)
for i := len(r.snapshots) - 2; i >= 0; i-- {
@@ -153,35 +148,6 @@ func (r *LockRing) GetPrimary(key string) pb.ServerAddress {
return r.Ring.GetPrimary(key)
}
// PriorOwner returns the key's owner from the previous ring snapshot, but only
// while the ring changed within the last snapshotInterval and that owner differs
// from the current primary. This is the cooling-off window in which the previous
// owner may still hold locks the new owner has not yet rebuilt — a caller can
// consult it before granting so a fresh owner does not double-grant during a
// rebalance. Returns "" outside the window or when ownership did not move. It
// uses the snapshot's prebuilt ring, so it does not rebuild a hash ring per call.
func (r *LockRing) PriorOwner(key string) pb.ServerAddress {
r.RLock()
defer r.RUnlock()
if len(r.snapshots) < 2 {
return ""
}
if time.Since(r.snapshots[0].ts) > r.snapshotInterval {
return ""
}
current := r.Ring.GetPrimary(key)
var prior pb.ServerAddress
if pr := r.snapshots[1].ring; pr != nil {
prior = pr.GetPrimary(key)
} else {
prior = hashKeyToServer(key, r.snapshots[1].servers)
}
if prior != "" && prior != current {
return prior
}
return ""
}
// hashKeyToServer uses a temporary consistent hash ring for the given server list.
func hashKeyToServer(key string, servers []pb.ServerAddress) pb.ServerAddress {
if len(servers) == 0 {
@@ -1,63 +0,0 @@
package lock_manager
import (
"fmt"
"testing"
"time"
"github.com/seaweedfs/seaweedfs/weed/pb"
)
func TestLockRing_PriorOwner(t *testing.T) {
r := NewLockRing(5 * time.Second)
t.Cleanup(r.WaitForCleanup)
setA := []pb.ServerAddress{"s1:1", "s2:1", "s3:1"}
r.SetSnapshot(setA, 1)
// Only one snapshot: nothing to fall back to.
if got := r.PriorOwner("any"); got != "" {
t.Fatalf("single snapshot should have no prior owner, got %q", got)
}
// Add a server so some keys' ownership moves.
setB := []pb.ServerAddress{"s1:1", "s2:1", "s3:1", "s4:1"}
r.SetSnapshot(setB, 2)
var moved, stable string
for i := 0; i < 2000 && (moved == "" || stable == ""); i++ {
key := fmt.Sprintf("key-%d", i)
if r.GetPrimary(key) != hashKeyToServer(key, setA) {
if moved == "" {
moved = key
}
} else if stable == "" {
stable = key
}
}
if moved == "" || stable == "" {
t.Skip("could not find both a moved and a stable key")
}
if got, want := r.PriorOwner(moved), hashKeyToServer(moved, setA); got != want {
t.Fatalf("PriorOwner(moved)=%q, want %q", got, want)
}
if got := r.PriorOwner(stable); got != "" {
t.Fatalf("unmoved key should have no prior owner, got %q", got)
}
}
func TestLockRing_PriorOwnerExpires(t *testing.T) {
r := NewLockRing(20 * time.Millisecond)
t.Cleanup(r.WaitForCleanup)
r.SetSnapshot([]pb.ServerAddress{"s1:1", "s2:1", "s3:1"}, 1)
r.SetSnapshot([]pb.ServerAddress{"s1:1", "s2:1", "s3:1", "s4:1"}, 2)
// Past the cooling interval, the prior owner is no longer offered.
time.Sleep(40 * time.Millisecond)
for i := 0; i < 2000; i++ {
if got := r.PriorOwner(fmt.Sprintf("key-%d", i)); got != "" {
t.Fatalf("prior owner should expire after the cooling interval, got %q", got)
}
}
}
-17
View File
@@ -30,7 +30,6 @@ import (
"github.com/seaweedfs/seaweedfs/weed/glog"
"github.com/seaweedfs/seaweedfs/weed/pb"
"github.com/seaweedfs/seaweedfs/weed/security"
stats_collect "github.com/seaweedfs/seaweedfs/weed/stats"
"github.com/seaweedfs/seaweedfs/weed/util"
"github.com/seaweedfs/seaweedfs/weed/util/grace"
)
@@ -51,8 +50,6 @@ type AdminOptions struct {
dataDir *string
icebergPort *int
urlPrefix *string
metricsHttpPort *int
metricsHttpIp *string
debug *bool
debugPort *int
cpuProfile *string
@@ -73,8 +70,6 @@ func init() {
a.readOnlyPassword = cmdAdmin.Flag.String("readOnlyPassword", "", "read-only user password (optional, for view-only access; requires adminPassword to be set)")
a.icebergPort = cmdAdmin.Flag.Int("iceberg.port", 8181, "Iceberg REST Catalog port (0 to hide in UI)")
a.urlPrefix = cmdAdmin.Flag.String("urlPrefix", "", "URL path prefix when running behind a reverse proxy under a subdirectory (e.g. /seaweedfs)")
a.metricsHttpPort = cmdAdmin.Flag.Int("metricsPort", 0, "Prometheus metrics listen port")
a.metricsHttpIp = cmdAdmin.Flag.String("metricsIp", "", "metrics listen ip. If empty, listens on all interfaces.")
a.debug = cmdAdmin.Flag.Bool("debug", false, "serves runtime profiling data via pprof on the port specified by -debug.port")
a.debugPort = cmdAdmin.Flag.Int("debug.port", 6060, "http port for debugging")
a.cpuProfile = cmdAdmin.Flag.String("cpuprofile", "", "cpu profile output file")
@@ -165,12 +160,6 @@ var cmdAdmin = &Command{
weed admin -debug -debug.port=6060 -master="localhost:9333"
weed admin -cpuprofile=cpu.prof -memprofile=mem.prof -master="localhost:9333"
Metrics:
- Use -metricsPort to expose Prometheus metrics at http://<host>:<metricsPort>/metrics
- Use -metricsIp to bind the metrics endpoint to a specific ip (default: all interfaces)
- Metrics are disabled when -metricsPort is 0 (the default)
- Example: weed admin -metricsPort=9327 -master="localhost:9333"
Configuration File:
- The security.toml file is read from ".", "$HOME/.seaweedfs/",
"/usr/local/etc/seaweedfs/", or "/etc/seaweedfs/", in that order
@@ -268,12 +257,6 @@ func runAdmin(cmd *Command, args []string) bool {
}
fmt.Printf("Plugin: Enabled\n")
// Start Prometheus metrics endpoint if a port is configured
if *a.metricsHttpPort > 0 {
fmt.Printf("Metrics: http://%s/metrics\n", stats_collect.JoinHostPort(*a.metricsHttpIp, *a.metricsHttpPort))
}
go stats_collect.StartMetricsServer(*a.metricsHttpIp, *a.metricsHttpPort)
// Set up graceful shutdown
ctx, cancel := context.WithCancel(context.Background())
defer cancel()
+2 -3
View File
@@ -61,8 +61,7 @@ type MountOptions struct {
dirIdleEvictSec *int
// Distributed locking for cross-mount write coordination and POSIX
// advisory locks (flock/fcntl)
// Distributed lock for cross-mount write coordination
distributedLock *bool
// POSIX compliance options
@@ -153,7 +152,7 @@ func init() {
mountReadRetryTime = cmdMount.Flag.Duration("readRetryTime", 6*time.Second, "maximum read retry wait time")
// Distributed lock for cross-mount write coordination
mountOptions.distributedLock = cmdMount.Flag.Bool("dlm", false, "coordinate writes across mounts (only one mount writes a file at a time) and honor POSIX advisory locks (flock/fcntl) across mounts by routing them to the owner filer")
mountOptions.distributedLock = cmdMount.Flag.Bool("dlm", false, "enable distributed lock for cross-mount write coordination (only one mount can write a file at a time)")
// POSIX compliance options
mountOptions.posixDirNlink = cmdMount.Flag.Bool("posix.dirNLink", false, "report POSIX-compliant directory nlink (2 + subdirectory count); costs one directory listing per stat")
+2 -9
View File
@@ -209,11 +209,7 @@ func (f *Filer) RollbackTransaction(ctx context.Context) error {
return f.Store.RollbackTransaction(ctx)
}
// CreateEntry creates or replaces an entry. When existing is non-nil the caller
// has already fetched the current entry at this path under a path lock, and it
// is reused instead of looking the store up again; pass nil to have CreateEntry
// look it up itself.
func (f *Filer) CreateEntry(ctx context.Context, entry *Entry, existing *Entry, o_excl bool, isFromOtherCluster bool, signatures []int32, skipCreateParentDir bool, maxFilenameLength uint32) error {
func (f *Filer) CreateEntry(ctx context.Context, entry *Entry, o_excl bool, isFromOtherCluster bool, signatures []int32, skipCreateParentDir bool, maxFilenameLength uint32) error {
if string(entry.FullPath) == "/" {
return nil
@@ -231,10 +227,7 @@ func (f *Filer) CreateEntry(ctx context.Context, entry *Entry, existing *Entry,
entry.Attr.Atime = entryInitialAtime(entry.Attr)
}
oldEntry := existing
if oldEntry == nil {
oldEntry, _ = f.FindEntry(ctx, entry.FullPath)
}
oldEntry, _ := f.FindEntry(ctx, entry.FullPath)
/*
if !hasWritePermission(lastDirectoryEntry, entry) {
+2 -2
View File
@@ -28,7 +28,7 @@ func TestCreateEntryAssignsInodeWhenMissing(t *testing.T) {
},
}
err := f.CreateEntry(context.Background(), entry, nil, false, false, nil, false, f.MaxFilenameLength)
err := f.CreateEntry(context.Background(), entry, false, false, nil, false, f.MaxFilenameLength)
require.NoError(t, err)
stored, findErr := store.FindEntry(context.Background(), entry.FullPath)
@@ -48,7 +48,7 @@ func TestCreateEntryAssignsInodesToAutoCreatedParents(t *testing.T) {
},
}
err := f.CreateEntry(context.Background(), entry, nil, false, false, nil, false, f.MaxFilenameLength)
err := f.CreateEntry(context.Background(), entry, false, false, nil, false, f.MaxFilenameLength)
require.NoError(t, err)
for _, path := range []string{"/a", "/a/b", "/a/b/c.txt"} {
+1 -1
View File
@@ -96,7 +96,7 @@ func (f *Filer) maybeLazyFetchFromRemote(ctx context.Context, p util.FullPath) (
persistBaseCtx, cancelPersist := context.WithTimeout(context.Background(), 30*time.Second)
defer cancelPersist()
persistCtx := context.WithValue(persistBaseCtx, lazyFetchContextKey{}, true)
saveErr := f.CreateEntry(persistCtx, entry, nil, false, false, nil, true, f.MaxFilenameLength)
saveErr := f.CreateEntry(persistCtx, entry, false, false, nil, true, f.MaxFilenameLength)
if saveErr != nil {
glog.Warningf("maybeLazyFetchFromRemote: failed to persist filer entry for %s: %v", p, saveErr)
f.lazyFetchGroup.Forget(key)
+2 -2
View File
@@ -152,7 +152,7 @@ func (f *Filer) maybeLazyListFromRemote(ctx context.Context, p util.FullPath) {
entry.Attr.FileSize = uint64(remoteEntry.RemoteSize)
}
}
if saveErr := f.CreateEntry(persistCtx, entry, nil, false, false, nil, true, f.MaxFilenameLength); saveErr != nil {
if saveErr := f.CreateEntry(persistCtx, entry, false, false, nil, true, f.MaxFilenameLength); saveErr != nil {
glog.Warningf("maybeLazyListFromRemote: persist %s: %v", childPath, saveErr)
}
}
@@ -193,7 +193,7 @@ func (f *Filer) updateDirectoryListingSyncedAt(ctx context.Context, p util.FullP
dirEntry.Extended = make(map[string][]byte)
}
dirEntry.Extended[xattrRemoteListingSyncedAt] = []byte(fmt.Sprintf("%d", syncTime.Unix()))
if saveErr := f.CreateEntry(ctx, dirEntry, nil, false, false, nil, true, f.MaxFilenameLength); saveErr != nil {
if saveErr := f.CreateEntry(ctx, dirEntry, false, false, nil, true, f.MaxFilenameLength); saveErr != nil {
glog.Warningf("maybeLazyListFromRemote: create dir synced_at for %s: %v", p, saveErr)
}
return
+1 -1
View File
@@ -43,7 +43,7 @@ func (f *Filer) appendToFile(targetFile string, data []byte) error {
entry.Chunks = append(entry.GetChunks(), uploadResult.ToPbFileChunk(assignResult.Fid, offset, time.Now().UnixNano()))
// update the entry
err = f.CreateEntry(context.Background(), entry, nil, false, false, nil, false, f.MaxFilenameLength)
err = f.CreateEntry(context.Background(), entry, false, false, nil, false, f.MaxFilenameLength)
return err
}
+1 -1
View File
@@ -32,7 +32,7 @@ func TestCreateAndFind(t *testing.T) {
},
}
if err := testFiler.CreateEntry(ctx, entry1, nil, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
if err := testFiler.CreateEntry(ctx, entry1, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
t.Errorf("create entry %v: %v", entry1.FullPath, err)
return
}
+1 -1
View File
@@ -29,7 +29,7 @@ func TestCreateAndFind(t *testing.T) {
},
}
if err := testFiler.CreateEntry(ctx, entry1, nil, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
if err := testFiler.CreateEntry(ctx, entry1, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
t.Errorf("create entry %v: %v", entry1.FullPath, err)
return
}
+1 -1
View File
@@ -29,7 +29,7 @@ func TestCreateAndFind(t *testing.T) {
},
}
if err := testFiler.CreateEntry(ctx, entry1, nil, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
if err := testFiler.CreateEntry(ctx, entry1, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
t.Errorf("create entry %v: %v", entry1.FullPath, err)
return
}
-259
View File
@@ -1,259 +0,0 @@
package posixlock
import (
"sync"
"time"
)
// Manager is the owner filer's in-memory authority for POSIX advisory locks
// across inodes. Lock state lives here, not in replicated metadata: it is
// transient coordination, so keeping it out of the meta-log avoids churn and
// does not pollute what subscribers see (the distributed lock manager holds its
// locks the same way). A `Set` per inode key, plus a session index so a dead
// mount's locks are reaped in O(locks held) rather than by scanning every inode.
//
// key is an opaque inode identity supplied by the caller — the file's path, or
// "hl:"+hex(HardLinkId) for a hardlinked inode — so all names of one inode share
// a Set. The Manager is safe for concurrent use.
type Manager struct {
mu sync.Mutex
byKey map[string]*Set // inode key -> held locks
bySid map[uint64]map[string]bool // session -> keys it currently holds locks on
lastSeen map[uint64]time.Time // session -> last keepalive; only renewing sessions are leased
}
func NewManager() *Manager {
return &Manager{
byKey: make(map[string]*Set),
bySid: make(map[uint64]map[string]bool),
lastSeen: make(map[uint64]time.Time),
}
}
// Renew records a keepalive from a session, placing it under lease management.
// Only sessions that have renewed are subject to ReapExpired, so a session that
// never sends keepalives (e.g. before the mount keepalive exists) is never reaped.
func (m *Manager) Renew(sid uint64) {
m.mu.Lock()
defer m.mu.Unlock()
m.lastSeen[sid] = time.Now()
}
// ReapExpired releases the locks of every leased session whose last keepalive is
// older than ttl — a dead or partitioned mount. Sessions that never renewed are
// left untouched. Returns the reaped session ids.
func (m *Manager) ReapExpired(ttl time.Duration) []uint64 {
m.mu.Lock()
defer m.mu.Unlock()
cutoff := time.Now().Add(-ttl)
var reaped []uint64
for sid, seen := range m.lastSeen {
if seen.After(cutoff) {
continue
}
for key := range m.bySid[sid] {
s := m.byKey[key]
if s == nil {
continue
}
s.ReleaseSession(sid)
if s.Empty() {
delete(m.byKey, key)
}
}
delete(m.bySid, sid)
delete(m.lastSeen, sid)
reaped = append(reaped, sid)
}
return reaped
}
// TryLock grants lk on key, or returns the conflicting lock and false. The set
// is created on first use and dropped again when it empties.
func (m *Manager) TryLock(key string, lk Range) (Range, bool) {
m.mu.Lock()
defer m.mu.Unlock()
s, ok := m.byKey[key]
if !ok {
s = &Set{}
}
if c, granted := s.Acquire(lk); !granted {
return c, false
}
if !ok {
m.byKey[key] = s
}
m.index(lk.Sid, key)
return Range{}, true
}
// Track records a lock the server already granted, without arbitration. A mount
// uses it to mirror its own held locks so it can re-assert them to the inode's
// current owner filer after an ownership change or owner restart.
func (m *Manager) Track(key string, lk Range) {
m.mu.Lock()
defer m.mu.Unlock()
s, ok := m.byKey[key]
if !ok {
s = &Set{}
m.byKey[key] = s
}
s.Grant(lk)
m.index(lk.Sid, key)
}
// Snapshot returns a copy of the held locks per key. A mount calls it to drive
// re-assertion keepalives; the filer never does.
func (m *Manager) Snapshot() map[string][]Range {
m.mu.Lock()
defer m.mu.Unlock()
out := make(map[string][]Range, len(m.byKey))
for key, s := range m.byKey {
out[key] = append([]Range(nil), s.locks...)
}
return out
}
// Reassert rebuilds session sid's locks on key from the client's authoritative
// list, renewing the lease. It replaces sid's existing locks on the key, then
// re-acquires each asserted lock — arbitrating against other sessions so it never
// double-grants. Locks that lost to another session in a migration window are
// returned as conflicts. The owner filer calls this on a re-assertion keepalive.
func (m *Manager) Reassert(key string, sid uint64, locks []Range) (conflicts []Range) {
m.mu.Lock()
defer m.mu.Unlock()
m.lastSeen[sid] = time.Now()
s := m.byKey[key]
if s == nil {
if len(locks) == 0 {
return nil
}
s = &Set{}
m.byKey[key] = s
}
s.ReleaseSession(sid)
for _, lk := range locks {
lk.Sid = sid
if c, granted := s.Acquire(lk); !granted {
conflicts = append(conflicts, c)
}
}
if setHasSession(s, sid) {
m.index(sid, key)
} else {
m.deindex(sid, key)
}
if s.Empty() {
delete(m.byKey, key)
}
return conflicts
}
// Unlock releases lk's owner's locks within its namespace over lk's range.
func (m *Manager) Unlock(key string, lk Range) {
m.mu.Lock()
defer m.mu.Unlock()
s := m.byKey[key]
if s == nil {
return
}
s.Release(lk)
m.afterRelease(key, s, lk.Sid)
}
// GetLk reports the lock that would block proposed on key, if any.
func (m *Manager) GetLk(key string, proposed Range) (Range, bool) {
m.mu.Lock()
defer m.mu.Unlock()
s := m.byKey[key]
if s == nil {
return Range{}, false
}
return s.Conflict(proposed)
}
// ReleasePosixOwner drops (sid, owner)'s fcntl locks on key — the flush-time path.
func (m *Manager) ReleasePosixOwner(key string, sid, owner uint64) {
m.mu.Lock()
defer m.mu.Unlock()
s := m.byKey[key]
if s == nil {
return
}
s.ReleasePosixOwner(sid, owner)
m.afterRelease(key, s, sid)
}
// ReleaseFlockOwner drops (sid, owner)'s flock locks on key — the release-time path.
func (m *Manager) ReleaseFlockOwner(key string, sid, owner uint64) {
m.mu.Lock()
defer m.mu.Unlock()
s := m.byKey[key]
if s == nil {
return
}
s.ReleaseFlockOwner(sid, owner)
m.afterRelease(key, s, sid)
}
// ReleaseSession drops every lock held by a session across all inodes it touched,
// reaping a mount whose lease expired. O(locks held by the session).
func (m *Manager) ReleaseSession(sid uint64) {
m.mu.Lock()
defer m.mu.Unlock()
for key := range m.bySid[sid] {
s := m.byKey[key]
if s == nil {
continue
}
s.ReleaseSession(sid)
if s.Empty() {
delete(m.byKey, key)
}
}
delete(m.bySid, sid)
}
// afterRelease prunes the session index when sid no longer holds any lock on key,
// and drops the set when it empties. Only sid's presence can have changed, since
// a release only removes sid's locks.
func (m *Manager) afterRelease(key string, s *Set, sid uint64) {
if s.Empty() {
delete(m.byKey, key)
m.deindex(sid, key)
return
}
if !setHasSession(s, sid) {
m.deindex(sid, key)
}
}
func (m *Manager) index(sid uint64, key string) {
keys := m.bySid[sid]
if keys == nil {
keys = make(map[string]bool)
m.bySid[sid] = keys
}
keys[key] = true
}
func (m *Manager) deindex(sid uint64, key string) {
keys := m.bySid[sid]
if keys == nil {
return
}
delete(keys, key)
if len(keys) == 0 {
delete(m.bySid, sid)
}
}
func setHasSession(s *Set, sid uint64) bool {
for _, l := range s.Locks() {
if l.Sid == sid {
return true
}
}
return false
}
-194
View File
@@ -1,194 +0,0 @@
package posixlock
import (
"math"
"runtime"
"sync"
"sync/atomic"
"testing"
"time"
)
func TestManagerGrantAndConflict(t *testing.T) {
m := NewManager()
if _, granted := m.TryLock("a", Range{Start: 0, End: 99, Type: Write, Sid: 1, Owner: 1}); !granted {
t.Fatal("first lock should be granted")
}
if c, granted := m.TryLock("a", Range{Start: 50, End: 149, Type: Write, Sid: 2, Owner: 1}); granted {
t.Fatalf("overlapping lock from another session should conflict, got grant; conflict=%+v", c)
}
// A different key is independent.
if _, granted := m.TryLock("b", Range{Start: 0, End: 99, Type: Write, Sid: 2, Owner: 1}); !granted {
t.Fatal("lock on a different key should be granted")
}
}
func TestManagerUnlockCleansEmptyKeyAndIndex(t *testing.T) {
m := NewManager()
lk := Range{Start: 0, End: 99, Type: Write, Sid: 1, Owner: 1}
m.TryLock("a", lk)
if !m.bySid[1]["a"] {
t.Fatal("session index should record the held key")
}
m.Unlock("a", Range{Start: 0, End: 99, Type: Unlock, Sid: 1, Owner: 1})
if _, ok := m.byKey["a"]; ok {
t.Fatal("empty set should be dropped from byKey")
}
if _, ok := m.bySid[1]; ok {
t.Fatal("session index should be pruned when it holds nothing")
}
}
func TestManagerPartialUnlockKeepsIndex(t *testing.T) {
m := NewManager()
m.TryLock("a", Range{Start: 0, End: 49, Type: Write, Sid: 1, Owner: 1})
m.TryLock("a", Range{Start: 100, End: 149, Type: Write, Sid: 1, Owner: 1})
// Release one of the two ranges; the session still holds the other.
m.Unlock("a", Range{Start: 0, End: 49, Type: Unlock, Sid: 1, Owner: 1})
if !m.bySid[1]["a"] {
t.Fatal("session still holds a lock on the key; index must remain")
}
if _, ok := m.byKey["a"]; !ok {
t.Fatal("key should remain while a lock is held")
}
}
func TestManagerGetLk(t *testing.T) {
m := NewManager()
m.TryLock("a", Range{Start: 10, End: 50, Type: Write, Sid: 1, Owner: 1, Pid: 7})
c, found := m.GetLk("a", Range{Start: 30, End: 70, Type: Read, Sid: 2, Owner: 1})
if !found || c.Pid != 7 {
t.Fatalf("expected conflict from pid 7, got %+v found=%v", c, found)
}
if _, found := m.GetLk("missing", Range{Start: 0, End: 1, Type: Write, Sid: 9, Owner: 9}); found {
t.Fatal("missing key should report no conflict")
}
}
func TestManagerReleasePosixOwnerKeepsFlockAndIndex(t *testing.T) {
m := NewManager()
m.TryLock("a", Range{Start: 0, End: 99, Type: Write, Sid: 1, Owner: 1})
m.TryLock("a", Range{Start: 0, End: math.MaxUint64, Type: Write, Sid: 1, Owner: 1, IsFlock: true})
m.ReleasePosixOwner("a", 1, 1)
// flock lock for the same session remains, so the index must remain too.
if !m.bySid[1]["a"] {
t.Fatal("session still holds the flock lock; index must remain")
}
if _, found := m.GetLk("a", Range{Start: 0, End: 10, Type: Write, Sid: 2, Owner: 2, IsFlock: true}); !found {
t.Fatal("flock lock should survive ReleasePosixOwner")
}
if _, found := m.GetLk("a", Range{Start: 0, End: 10, Type: Write, Sid: 2, Owner: 2}); found {
t.Fatal("fcntl lock should be gone after ReleasePosixOwner")
}
}
func TestManagerReleaseSessionReapsAcrossKeys(t *testing.T) {
m := NewManager()
m.TryLock("a", Range{Start: 0, End: 99, Type: Write, Sid: 1, Owner: 1})
m.TryLock("b", Range{Start: 0, End: 99, Type: Write, Sid: 1, Owner: 2})
m.TryLock("b", Range{Start: 200, End: 299, Type: Write, Sid: 2, Owner: 1})
m.ReleaseSession(1)
if _, ok := m.bySid[1]; ok {
t.Fatal("reaped session should be gone from the index")
}
if _, ok := m.byKey["a"]; ok {
t.Fatal("key a held only session 1's lock and should be dropped")
}
// Session 2's lock on b survives.
if _, found := m.GetLk("b", Range{Start: 200, End: 299, Type: Write, Sid: 9, Owner: 9}); !found {
t.Fatal("session 2's lock on b should remain after reaping session 1")
}
if !m.bySid[2]["b"] {
t.Fatal("session 2 index entry should remain")
}
}
func TestManagerReapsOnlyStaleLeasedSessions(t *testing.T) {
m := NewManager()
// Session 1: holds a lock, leased but stale (renewed long ago).
m.TryLock("a", Range{Start: 0, End: 99, Type: Write, Sid: 1, Owner: 1})
m.Renew(1)
m.lastSeen[1] = time.Now().Add(-time.Hour)
// Session 2: holds a lock, leased and fresh.
m.TryLock("b", Range{Start: 0, End: 99, Type: Write, Sid: 2, Owner: 1})
m.Renew(2)
// Session 3: holds a lock but never renewed (no lease) — must not be reaped.
m.TryLock("c", Range{Start: 0, End: 99, Type: Write, Sid: 3, Owner: 1})
reaped := m.ReapExpired(30 * time.Second)
if len(reaped) != 1 || reaped[0] != 1 {
t.Fatalf("only the stale leased session should be reaped, got %v", reaped)
}
if _, ok := m.byKey["a"]; ok {
t.Fatal("stale session's lock should be gone")
}
if _, ok := m.byKey["b"]; !ok {
t.Fatal("fresh session's lock must remain")
}
if _, ok := m.byKey["c"]; !ok {
t.Fatal("never-renewed session must not be reaped")
}
if _, ok := m.lastSeen[1]; ok {
t.Fatal("reaped session's lease entry should be cleared")
}
}
// Mutual exclusion under concurrent whole-file flock churn through the Manager:
// at most one owner may believe it holds the exclusive lock at any instant.
func TestManagerConcurrentFlockMutualExclusion(t *testing.T) {
m := NewManager()
const (
key = "inode"
workers = 16
iters = 400
)
var (
wg sync.WaitGroup
holder atomic.Int64
overlap atomic.Int32
)
for w := 0; w < workers; w++ {
wg.Add(1)
go func(id int) {
defer wg.Done()
lk := Range{Start: 0, End: math.MaxUint64, Type: Write, Sid: uint64(id + 1), Owner: 1, IsFlock: true}
unlock := lk
unlock.Type = Unlock
token := int64(id + 1)
for i := 0; i < iters; i++ {
for {
if _, granted := m.TryLock(key, lk); granted {
break
}
runtime.Gosched()
}
if prev := holder.Swap(token); prev != 0 {
overlap.Add(1)
}
runtime.Gosched()
if !holder.CompareAndSwap(token, 0) {
overlap.Add(1)
}
m.Unlock(key, unlock)
}
}(w)
}
wg.Wait()
if n := overlap.Load(); n != 0 {
t.Fatalf("mutual exclusion violated %d times", n)
}
if len(m.byKey) != 0 {
t.Fatalf("all locks released; byKey should be empty, got %d", len(m.byKey))
}
if len(m.bySid) != 0 {
t.Fatalf("all locks released; bySid should be empty, got %d", len(m.bySid))
}
}
-222
View File
@@ -1,222 +0,0 @@
// Package posixlock implements the conflict, coalescing, and range-split logic
// for POSIX advisory file locks — fcntl byte-range and flock whole-file — as a
// pure per-inode lock set with no concurrency control of its own.
//
// It is the server-side authority for distributed FUSE locking: the owner filer
// for an inode holds one Set and serializes access to it under that inode's
// per-path lock, so each operation runs to completion without concurrent
// mutation. The same algorithm can back the per-mount table; blocking (SetLkw)
// and any wait queue belong to the caller, not here.
package posixlock
import (
"math"
"sort"
)
// Lock types, kept independent of the platform syscall package (whose F_RDLCK /
// F_WRLCK / F_UNLCK values differ per OS and are absent on some) so the filer
// builds everywhere. Callers map the syscall constants onto these at the edge.
// Zero is intentionally unused so a zero-value Range reads as "unset".
const (
Read uint32 = 1
Write uint32 = 2
Unlock uint32 = 3
)
// Range is one held advisory byte-range lock. Owner identity is the (Sid, Owner)
// pair: Sid is the mount session, Owner the FUSE lock owner within it, so owners
// from different mounts never alias. End is inclusive; math.MaxUint64 means EOF.
// IsFlock separates the flock and fcntl namespaces, which never conflict.
type Range struct {
Start uint64
End uint64
Type uint32
Sid uint64
Owner uint64
Pid uint32
IsFlock bool
}
func (r Range) sameOwner(o Range) bool {
return r.Sid == o.Sid && r.Owner == o.Owner
}
// Set is the authoritative set of advisory locks held on one inode. The zero
// value is an empty set. Set has no internal locking; the caller serializes it.
type Set struct {
locks []Range // sorted by Start
}
func overlap(aStart, aEnd, bStart, bEnd uint64) bool {
return aStart <= bEnd && bStart <= aEnd
}
// Conflict returns the first held lock that blocks proposed, if any. Two locks
// conflict when they share a namespace, have different owners, overlap, and at
// least one is a write lock.
func (s *Set) Conflict(proposed Range) (Range, bool) {
for _, h := range s.locks {
if h.IsFlock != proposed.IsFlock || h.sameOwner(proposed) {
continue
}
if !overlap(h.Start, h.End, proposed.Start, proposed.End) {
continue
}
if h.Type == Read && proposed.Type == Read {
continue
}
return h, true
}
return Range{}, false
}
// Acquire grants lk when it does not conflict, inserting it and coalescing the
// owner's adjacent/overlapping ranges. On conflict it returns the blocking lock
// and false, leaving the set unchanged.
func (s *Set) Acquire(lk Range) (Range, bool) {
if c, found := s.Conflict(lk); found {
return c, false
}
s.insert(lk)
return Range{}, true
}
// Grant inserts lk without a conflict check. It is for a client mirroring locks
// the server already granted (so they are conflict-free), not for arbitration.
func (s *Set) Grant(lk Range) {
s.insert(lk)
}
// insert adds lk, absorbing same-owner same-type overlaps and merging adjacent
// same-type ranges, and truncating/splitting a same-owner range of a different
// type that overlaps (an in-place type change).
func (s *Set) insert(lk Range) {
var kept []Range
for _, h := range s.locks {
if !h.sameOwner(lk) || h.IsFlock != lk.IsFlock {
kept = append(kept, h)
continue
}
if !overlap(h.Start, h.End, lk.Start, lk.End) {
// Merge only ranges that are adjacent and the same type. The
// End < MaxUint64 guards stop +1 from wrapping at EOF.
if h.Type == lk.Type && ((h.End < math.MaxUint64 && h.End+1 == lk.Start) || (lk.End < math.MaxUint64 && lk.End+1 == h.Start)) {
if h.Start < lk.Start {
lk.Start = h.Start
}
if h.End > lk.End {
lk.End = h.End
}
continue
}
kept = append(kept, h)
continue
}
if h.Type == lk.Type {
// Same type: absorb into lk by widening its range.
if h.Start < lk.Start {
lk.Start = h.Start
}
if h.End > lk.End {
lk.End = h.End
}
continue
}
// Different type: the surviving portions of h outside lk's range stay.
if h.Start < lk.Start {
left := h
left.End = lk.Start - 1
kept = append(kept, left)
}
if h.End > lk.End {
right := h
right.Start = lk.End + 1
kept = append(kept, right)
}
}
kept = append(kept, lk)
sort.Slice(kept, func(i, j int) bool { return kept[i].Start < kept[j].Start })
s.locks = kept
}
// remove drops or splits matching locks within [start,end]. A match that
// straddles the range keeps its non-overlapping head and/or tail.
func (s *Set) remove(matches func(Range) bool, start, end uint64) {
var kept []Range
for _, h := range s.locks {
if !matches(h) || !overlap(h.Start, h.End, start, end) {
kept = append(kept, h)
continue
}
if h.Start < start {
left := h
left.End = start - 1
kept = append(kept, left)
}
if h.End > end {
right := h
right.Start = end + 1
kept = append(kept, right)
}
// Fully covered: dropped.
}
s.locks = kept
}
// Release clears lk's owner's locks within lk's namespace over [lk.Start,lk.End]
// — the F_UNLCK path, which may split a straddling range.
func (s *Set) Release(lk Range) {
s.remove(func(h Range) bool {
return h.sameOwner(lk) && h.IsFlock == lk.IsFlock
}, lk.Start, lk.End)
}
// ReleaseOwner removes every lock held by (sid, owner) in both namespaces.
func (s *Set) ReleaseOwner(sid, owner uint64) {
s.remove(func(h Range) bool {
return h.Sid == sid && h.Owner == owner
}, 0, math.MaxUint64)
}
// ReleaseFlockOwner removes only the flock locks of (sid, owner) — the close-time
// path for a released file description (FUSE_RELEASE_FLOCK_UNLOCK).
func (s *Set) ReleaseFlockOwner(sid, owner uint64) {
s.remove(func(h Range) bool {
return h.IsFlock && h.Sid == sid && h.Owner == owner
}, 0, math.MaxUint64)
}
// ReleasePosixOwner removes only the fcntl locks of (sid, owner) — the close-time
// path for a flushing POSIX lock owner.
func (s *Set) ReleasePosixOwner(sid, owner uint64) {
s.remove(func(h Range) bool {
return !h.IsFlock && h.Sid == sid && h.Owner == owner
}, 0, math.MaxUint64)
}
// ReleaseSession removes every lock held by a session, reaping a mount that has
// died or disconnected (its lease expired).
func (s *Set) ReleaseSession(sid uint64) {
s.remove(func(h Range) bool { return h.Sid == sid }, 0, math.MaxUint64)
}
// HasPosix reports whether (sid, owner) holds any fcntl lock, mirroring the
// mount's flush-time check that avoids treating a lock-free flush as
// lock-sensitive.
func (s *Set) HasPosix(sid, owner uint64) bool {
for _, h := range s.locks {
if !h.IsFlock && h.Sid == sid && h.Owner == owner {
return true
}
}
return false
}
// Locks returns the held locks, sorted by Start. The slice aliases internal
// state; the caller must not mutate it.
func (s *Set) Locks() []Range { return s.locks }
// Empty reports whether no locks are held, so the caller can drop the inode's
// entry from its table.
func (s *Set) Empty() bool { return len(s.locks) == 0 }
-292
View File
@@ -1,292 +0,0 @@
package posixlock
import (
"math"
"testing"
)
// acquire is a test helper asserting the lock is granted.
func mustAcquire(t *testing.T, s *Set, lk Range) {
t.Helper()
if _, granted := s.Acquire(lk); !granted {
t.Fatalf("expected lock granted: %+v", lk)
}
}
func TestNonOverlappingLocksFromDifferentOwners(t *testing.T) {
s := &Set{}
mustAcquire(t, s, Range{Start: 0, End: 49, Type: Write, Owner: 1, Pid: 10})
mustAcquire(t, s, Range{Start: 50, End: 99, Type: Write, Owner: 2, Pid: 20})
}
func TestOverlappingReadLocksFromDifferentOwners(t *testing.T) {
s := &Set{}
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Read, Owner: 1, Pid: 10})
mustAcquire(t, s, Range{Start: 50, End: 149, Type: Read, Owner: 2, Pid: 20})
}
func TestOverlappingWriteReadConflict(t *testing.T) {
s := &Set{}
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
if _, granted := s.Acquire(Range{Start: 50, End: 149, Type: Read, Owner: 2, Pid: 20}); granted {
t.Fatal("expected conflict")
}
}
func TestOverlappingWriteWriteConflict(t *testing.T) {
s := &Set{}
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
if _, granted := s.Acquire(Range{Start: 50, End: 149, Type: Write, Owner: 2, Pid: 20}); granted {
t.Fatal("expected conflict")
}
}
func TestSameOwnerUpgradeReadToWrite(t *testing.T) {
s := &Set{}
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Read, Owner: 1, Pid: 10})
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
c, found := s.Conflict(Range{Start: 0, End: 99, Type: Write, Owner: 2, Pid: 20})
if !found || c.Type != Write {
t.Fatalf("expected conflicting write lock after upgrade, got %+v found=%v", c, found)
}
}
func TestSameOwnerDowngradeWriteToRead(t *testing.T) {
s := &Set{}
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Read, Owner: 1, Pid: 10})
// Another owner can now take a shared read lock.
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Read, Owner: 2, Pid: 20})
}
func TestLockCoalescing(t *testing.T) {
s := &Set{}
mustAcquire(t, s, Range{Start: 0, End: 9, Type: Write, Owner: 1, Pid: 10})
mustAcquire(t, s, Range{Start: 10, End: 19, Type: Write, Owner: 1, Pid: 10})
if len(s.locks) != 1 {
t.Fatalf("expected 1 coalesced lock, got %d: %+v", len(s.locks), s.locks)
}
if s.locks[0].Start != 0 || s.locks[0].End != 19 {
t.Errorf("expected coalesced [0,19], got [%d,%d]", s.locks[0].Start, s.locks[0].End)
}
}
func TestLockSplitting(t *testing.T) {
s := &Set{}
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
s.Release(Range{Start: 40, End: 59, Type: Unlock, Owner: 1, Pid: 10})
if len(s.locks) != 2 {
t.Fatalf("expected 2 locks after split, got %d: %+v", len(s.locks), s.locks)
}
if s.locks[0].Start != 0 || s.locks[0].End != 39 {
t.Errorf("expected left [0,39], got [%d,%d]", s.locks[0].Start, s.locks[0].End)
}
if s.locks[1].Start != 60 || s.locks[1].End != 99 {
t.Errorf("expected right [60,99], got [%d,%d]", s.locks[1].Start, s.locks[1].End)
}
}
func TestConflictReportsHolder(t *testing.T) {
s := &Set{}
mustAcquire(t, s, Range{Start: 10, End: 50, Type: Write, Owner: 1, Pid: 10})
c, found := s.Conflict(Range{Start: 30, End: 70, Type: Read, Owner: 2, Pid: 20})
if !found {
t.Fatal("expected a conflict")
}
if c.Type != Write || c.Pid != 10 || c.Start != 10 || c.End != 50 {
t.Fatalf("unexpected conflict report: %+v", c)
}
}
func TestConflictNoneForSharedReads(t *testing.T) {
s := &Set{}
mustAcquire(t, s, Range{Start: 10, End: 50, Type: Read, Owner: 1, Pid: 10})
if _, found := s.Conflict(Range{Start: 30, End: 70, Type: Read, Owner: 2, Pid: 20}); found {
t.Fatal("two read locks should not conflict")
}
}
func TestConflictSameOwnerNone(t *testing.T) {
s := &Set{}
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
if _, found := s.Conflict(Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10}); found {
t.Fatal("an owner should not conflict with itself")
}
}
func TestReleaseOwner(t *testing.T) {
s := &Set{}
mustAcquire(t, s, Range{Start: 0, End: 49, Type: Write, Owner: 1, Pid: 10})
mustAcquire(t, s, Range{Start: 50, End: 99, Type: Write, Owner: 1, Pid: 10})
mustAcquire(t, s, Range{Start: 200, End: 299, Type: Read, Owner: 2, Pid: 20})
s.ReleaseOwner(0, 1)
if _, found := s.Conflict(Range{Start: 0, End: 99, Type: Write, Owner: 3, Pid: 30}); found {
t.Fatal("owner 1's locks should be gone")
}
c, found := s.Conflict(Range{Start: 200, End: 299, Type: Write, Owner: 3, Pid: 30})
if !found || c.Type != Read {
t.Fatalf("owner 2's read lock should remain, got %+v found=%v", c, found)
}
}
func TestFlockAndFcntlDoNotConflict(t *testing.T) {
s := &Set{}
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
mustAcquire(t, s, Range{Start: 0, End: math.MaxUint64, Type: Write, Owner: 2, Pid: 20, IsFlock: true})
}
func TestReleasePosixOwnerKeepsFlock(t *testing.T) {
s := &Set{}
mustAcquire(t, s, Range{Start: 0, End: math.MaxUint64, Type: Write, Owner: 1, Pid: 10, IsFlock: true})
s.ReleasePosixOwner(0, 1)
if _, found := s.Conflict(Range{Start: 0, End: math.MaxUint64, Type: Write, Owner: 2, Pid: 20, IsFlock: true}); !found {
t.Fatal("flock lock should remain after ReleasePosixOwner")
}
}
func TestReleaseFlockOwnerKeepsPosix(t *testing.T) {
s := &Set{}
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
mustAcquire(t, s, Range{Start: 0, End: math.MaxUint64, Type: Write, Owner: 2, Pid: 10, IsFlock: true})
s.ReleaseFlockOwner(0, 2)
if _, found := s.Conflict(Range{Start: 0, End: 99, Type: Write, Owner: 3, Pid: 30}); !found {
t.Fatal("fcntl lock should remain after ReleaseFlockOwner")
}
if _, found := s.Conflict(Range{Start: 0, End: math.MaxUint64, Type: Write, Owner: 4, Pid: 40, IsFlock: true}); found {
t.Fatal("flock lock should be gone after ReleaseFlockOwner")
}
}
func TestHasPosixIgnoresMissingOwnerAndFlock(t *testing.T) {
s := &Set{}
if s.HasPosix(0, 1) {
t.Fatal("empty set should report no posix owner")
}
mustAcquire(t, s, Range{Start: 0, End: math.MaxUint64, Type: Write, Owner: 1, Pid: 10, IsFlock: true})
if s.HasPosix(0, 1) {
t.Fatal("a flock owner is not a posix owner")
}
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 2, Pid: 20})
if !s.HasPosix(0, 2) {
t.Fatal("posix owner should be reported")
}
}
func TestWholeFileLock(t *testing.T) {
s := &Set{}
mustAcquire(t, s, Range{Start: 0, End: math.MaxUint64, Type: Write, Owner: 1, Pid: 10})
if _, granted := s.Acquire(Range{Start: 0, End: math.MaxUint64, Type: Write, Owner: 2, Pid: 20}); granted {
t.Fatal("whole-file lock should block another owner")
}
if _, granted := s.Acquire(Range{Start: 100, End: 200, Type: Read, Owner: 2, Pid: 20}); granted {
t.Fatal("partial overlap with whole-file lock should conflict")
}
}
func TestReleaseNoExistingLocks(t *testing.T) {
s := &Set{}
s.Release(Range{Start: 0, End: 99, Type: Unlock, Owner: 1, Pid: 10})
if !s.Empty() {
t.Fatal("releasing on an empty set should be a no-op")
}
}
func TestSameOwnerReplaceDifferentType(t *testing.T) {
s := &Set{}
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
mustAcquire(t, s, Range{Start: 30, End: 60, Type: Read, Owner: 1, Pid: 10})
if len(s.locks) != 3 {
t.Fatalf("expected 3 locks after partial type change, got %d: %+v", len(s.locks), s.locks)
}
if s.locks[0].Type != Write || s.locks[0].Start != 0 || s.locks[0].End != 29 {
t.Errorf("expected write [0,29], got %+v", s.locks[0])
}
if s.locks[1].Type != Read || s.locks[1].Start != 30 || s.locks[1].End != 60 {
t.Errorf("expected read [30,60], got %+v", s.locks[1])
}
if s.locks[2].Type != Write || s.locks[2].Start != 61 || s.locks[2].End != 99 {
t.Errorf("expected write [61,99], got %+v", s.locks[2])
}
}
func TestNonAdjacentRangesNotCoalesced(t *testing.T) {
s := &Set{}
mustAcquire(t, s, Range{Start: 5, End: math.MaxUint64, Type: Write, Owner: 1, Pid: 10})
mustAcquire(t, s, Range{Start: 0, End: 2, Type: Write, Owner: 1, Pid: 10})
if len(s.locks) != 2 {
t.Fatalf("gap [3,4] should prevent coalescing, got %d: %+v", len(s.locks), s.locks)
}
if s.locks[0].Start != 0 || s.locks[0].End != 2 {
t.Errorf("expected [0,2], got [%d,%d]", s.locks[0].Start, s.locks[0].End)
}
if s.locks[1].Start != 5 || s.locks[1].End != math.MaxUint64 {
t.Errorf("expected [5,MaxUint64], got [%d,%d]", s.locks[1].Start, s.locks[1].End)
}
}
func TestAdjacencyNoOverflowAtMaxUint64(t *testing.T) {
s := &Set{}
mustAcquire(t, s, Range{Start: 100, End: math.MaxUint64, Type: Write, Owner: 1, Pid: 10})
mustAcquire(t, s, Range{Start: 0, End: 0, Type: Write, Owner: 1, Pid: 10})
if len(s.locks) != 2 {
t.Fatalf("MaxUint64+1 must not wrap and falsely merge, got %d: %+v", len(s.locks), s.locks)
}
}
// Sessions are part of owner identity: the same FUSE Owner number on two
// different mounts (Sid) is two distinct owners and must contend.
func TestTwoSessionsSameOwnerDoNotAlias(t *testing.T) {
s := &Set{}
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Sid: 1, Owner: 5, Pid: 10})
if _, granted := s.Acquire(Range{Start: 50, End: 149, Type: Write, Sid: 2, Owner: 5, Pid: 20}); granted {
t.Fatal("same Owner number on a different session must conflict, not alias")
}
// A read on session 2 against a session-1 read is fine (shared).
s2 := &Set{}
mustAcquire(t, s2, Range{Start: 0, End: 99, Type: Read, Sid: 1, Owner: 5})
mustAcquire(t, s2, Range{Start: 0, End: 99, Type: Read, Sid: 2, Owner: 5})
}
// Reaping a dead mount drops only that session's locks.
func TestReleaseSessionReapsOnlyThatSession(t *testing.T) {
s := &Set{}
mustAcquire(t, s, Range{Start: 0, End: 49, Type: Write, Sid: 1, Owner: 1, Pid: 10})
mustAcquire(t, s, Range{Start: 0, End: math.MaxUint64, Type: Write, Sid: 1, Owner: 2, Pid: 11, IsFlock: true})
mustAcquire(t, s, Range{Start: 50, End: 99, Type: Write, Sid: 2, Owner: 1, Pid: 20})
s.ReleaseSession(1)
for _, h := range s.locks {
if h.Sid == 1 {
t.Fatalf("session 1 lock survived reaping: %+v", h)
}
}
c, found := s.Conflict(Range{Start: 50, End: 99, Type: Write, Sid: 3, Owner: 9})
if !found || c.Sid != 2 {
t.Fatalf("session 2's lock should remain, got %+v found=%v", c, found)
}
}
func TestEmptyAfterReleasingAll(t *testing.T) {
s := &Set{}
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
if s.Empty() {
t.Fatal("set should not be empty with a held lock")
}
s.Release(Range{Start: 0, End: 99, Type: Unlock, Owner: 1, Pid: 10})
if !s.Empty() {
t.Fatal("set should be empty after releasing the only lock")
}
}
-82
View File
@@ -1,82 +0,0 @@
package posixlock
import (
"reflect"
"testing"
"time"
)
// A mount's tracked locks round-trip through Snapshot back to a fresh owner via
// Reassert — the owner-restart / ring-change recovery path.
func TestReassertRebuildsOnFreshOwner(t *testing.T) {
const sid = uint64(7)
// Client mirror: two granted locks on one key, one on another.
client := NewManager()
client.Track("a", Range{Start: 0, End: 99, Type: Write, Sid: sid, Owner: 1})
client.Track("a", Range{Start: 200, End: 299, Type: Read, Sid: sid, Owner: 2})
client.Track("b", Range{Start: 0, End: maxEnd, Type: Write, Sid: sid, Owner: 1, IsFlock: true})
// Fresh owner (post-restart / new ring owner) knows nothing.
owner := NewManager()
for key, locks := range client.Snapshot() {
if c := owner.Reassert(key, sid, locks); c != nil {
t.Fatalf("unexpected conflict reasserting %s: %+v", key, c)
}
}
// The owner now reports the same conflicts a foreign session would hit.
if _, granted := owner.TryLock("a", Range{Start: 50, End: 60, Type: Write, Sid: 99, Owner: 1}); granted {
t.Fatal("owner should block a foreign write after rebuild")
}
if _, granted := owner.TryLock("b", Range{Start: 0, End: 0, Type: Read, Sid: 99, Owner: 1, IsFlock: true}); granted {
t.Fatal("owner should block a foreign flock read after rebuild")
}
}
// Re-asserting every tick is idempotent: the owner's view is unchanged.
func TestReassertIdempotent(t *testing.T) {
const sid = uint64(1)
m := NewManager()
m.TryLock("k", Range{Start: 0, End: 99, Type: Write, Sid: sid, Owner: 1})
before := append([]Range(nil), m.byKey["k"].locks...)
m.Reassert("k", sid, before)
m.Reassert("k", sid, before)
if !reflect.DeepEqual(m.byKey["k"].locks, before) {
t.Fatalf("reassert not idempotent:\n got %+v\nwant %+v", m.byKey["k"].locks, before)
}
}
// A lock another session grabbed in the migration window is reported as a
// conflict and not double-granted.
func TestReassertReportsConflict(t *testing.T) {
const mine, other = uint64(1), uint64(2)
m := NewManager()
// Another mount took the lock on this (new) owner during the gap.
m.TryLock("k", Range{Start: 0, End: 99, Type: Write, Sid: other, Owner: 1})
conflicts := m.Reassert("k", mine, []Range{{Start: 0, End: 99, Type: Write, Sid: mine, Owner: 1}})
if len(conflicts) != 1 {
t.Fatalf("expected 1 conflict, got %d: %+v", len(conflicts), conflicts)
}
// The other session keeps the lock; mine was not installed.
if got := len(m.byKey["k"].locks); got != 1 {
t.Fatalf("expected only the incumbent lock, got %d", got)
}
}
// Reassert renews the lease, so a re-asserting mount is not reaped.
func TestReassertRenewsLease(t *testing.T) {
const sid = uint64(1)
m := NewManager()
m.Renew(sid)
m.Reassert("k", sid, []Range{{Start: 0, End: 9, Type: Write, Sid: sid, Owner: 1}})
if reaped := m.ReapExpired(time.Hour); len(reaped) != 0 {
t.Fatalf("freshly re-asserted session should not be reaped: %v", reaped)
}
}
const maxEnd = ^uint64(0)
@@ -10,7 +10,7 @@ import (
func (store *UniversalRedis2Store) KvPut(ctx context.Context, key []byte, value []byte) (err error) {
_, err = store.Client.Set(ctx, store.getKey(string(key)), value, 0).Result()
_, err = store.Client.Set(ctx, string(key), value, 0).Result()
if err != nil {
return fmt.Errorf("kv put: %w", err)
@@ -21,7 +21,7 @@ func (store *UniversalRedis2Store) KvPut(ctx context.Context, key []byte, value
func (store *UniversalRedis2Store) KvGet(ctx context.Context, key []byte) (value []byte, err error) {
data, err := store.Client.Get(ctx, store.getKey(string(key))).Result()
data, err := store.Client.Get(ctx, string(key)).Result()
if err == redis.Nil {
return nil, filer.ErrKvNotFound
@@ -32,7 +32,7 @@ func (store *UniversalRedis2Store) KvGet(ctx context.Context, key []byte) (value
func (store *UniversalRedis2Store) KvDelete(ctx context.Context, key []byte) (err error) {
_, err = store.Client.Del(ctx, store.getKey(string(key))).Result()
_, err = store.Client.Del(ctx, string(key)).Result()
if err != nil {
return fmt.Errorf("kv delete: %w", err)
+2 -2
View File
@@ -35,7 +35,7 @@ func TestCreateAndFind(t *testing.T) {
},
}
if err := testFiler.CreateEntry(ctx, entry1, nil, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
if err := testFiler.CreateEntry(ctx, entry1, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
t.Errorf("create entry %v: %v", entry1.FullPath, err)
return
}
@@ -149,7 +149,7 @@ func TestListDirectoryWithPrefix(t *testing.T) {
Gid: 1,
},
}
if err := testFiler.CreateEntry(ctx, entry, nil, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
if err := testFiler.CreateEntry(ctx, entry, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
t.Fatalf("Failed to create entry %s: %v", fullpath, err)
}
}
+10 -26
View File
@@ -35,12 +35,7 @@ func (fh *FileHandle) readFromChunksWithContext(ctx context.Context, buff []byte
entry := fh.GetEntry()
// IsInRemoteOnly inspects entry.Chunks, so take the LockedEntry lock the
// async uploader appends under.
entry.RLock()
remoteOnly := entry.Entry.IsInRemoteOnly()
entry.RUnlock()
if remoteOnly {
if entry.IsInRemoteOnly() {
glog.V(4).Infof("download remote entry %s", fileFullPath)
err := fh.downloadRemoteEntry(entry)
if err != nil {
@@ -49,21 +44,10 @@ func (fh *FileHandle) readFromChunksWithContext(ctx context.Context, buff []byte
}
}
// Snapshot size, inline content, and the chunk list under the LockedEntry
// lock. Async upload workers append chunks under this lock (AddChunks), so
// reading entry.Chunks / FileSize without it races with the slice
// reallocation and can crash in filer.TotalSize. The captured slice headers
// stay valid afterwards: append never mutates the old backing array, and
// truncate is excluded by the fh.entryLock held for this whole read.
entry.RLock()
pbEntry := entry.Entry
fileSize := int64(pbEntry.Attributes.FileSize)
fileSize := int64(entry.Attributes.FileSize)
if fileSize == 0 {
fileSize = int64(filer.FileSize(pbEntry))
fileSize = int64(filer.FileSize(entry.GetEntry()))
}
content := pbEntry.Content
chunks := pbEntry.Chunks
entry.RUnlock()
if fileSize == 0 {
glog.V(1).Infof("empty fh %v", fileFullPath)
@@ -75,15 +59,15 @@ func (fh *FileHandle) readFromChunksWithContext(ctx context.Context, buff []byte
return 0, 0, io.EOF
}
if offset < int64(len(content)) {
totalRead := copy(buff, content[offset:])
if offset < int64(len(entry.Content)) {
totalRead := copy(buff, entry.Content[offset:])
glog.V(4).Infof("file handle read cached %s [%d,%d] %d", fileFullPath, offset, offset+int64(totalRead), totalRead)
return int64(totalRead), 0, nil
}
// Try RDMA acceleration first if available
if fh.wfs.rdmaClient != nil && fh.wfs.option.RdmaEnabled {
totalRead, ts, err := fh.tryRDMARead(ctx, fileSize, buff, offset, chunks)
totalRead, ts, err := fh.tryRDMARead(ctx, fileSize, buff, offset, entry)
if err == nil {
glog.V(4).Infof("RDMA read successful for %s [%d,%d] %d", fileFullPath, offset, offset+int64(totalRead), totalRead)
return int64(totalRead), ts, nil
@@ -95,7 +79,7 @@ func (fh *FileHandle) readFromChunksWithContext(ctx context.Context, buff []byte
// Any failure falls through transparently. See design-weed-mount-
// peer-chunk-sharing.md §4.3.
if fh.wfs.option.PeerEnabled && fh.wfs.peerGrpcServer != nil {
totalRead, ts, err := fh.tryPeerRead(ctx, fileSize, buff, offset, chunks)
totalRead, ts, err := fh.tryPeerRead(ctx, fileSize, buff, offset, entry)
if err == nil {
glog.V(4).Infof("peer read successful for %s [%d,%d] %d", fileFullPath, offset, offset+int64(totalRead), totalRead)
return int64(totalRead), ts, nil
@@ -120,13 +104,13 @@ func (fh *FileHandle) readFromChunksWithContext(ctx context.Context, buff []byte
return int64(totalRead), ts, err
}
// tryRDMARead attempts to read file data using RDMA acceleration. chunks is a
// snapshot captured under the LockedEntry lock by the caller.
func (fh *FileHandle) tryRDMARead(ctx context.Context, fileSize int64, buff []byte, offset int64, chunks []*filer_pb.FileChunk) (int64, int64, error) {
// tryRDMARead attempts to read file data using RDMA acceleration
func (fh *FileHandle) tryRDMARead(ctx context.Context, fileSize int64, buff []byte, offset int64, entry *LockedEntry) (int64, int64, error) {
// For now, we'll try to read the chunks directly using RDMA
// This is a simplified approach - in a full implementation, we'd need to
// handle chunk boundaries, multiple chunks, etc.
chunks := entry.GetEntry().Chunks
if len(chunks) == 0 {
return 0, 0, fmt.Errorf("no chunks available for RDMA read")
}
+2 -3
View File
@@ -53,8 +53,7 @@ const maxPeerFetchChunkBytes = 64 * 1024 * 1024
// end-to-end against FileChunk.ETag.
// 4. On success, populate chunk_cache and enqueue an announce so
// other mounts can discover us as a new holder.
// chunks is a snapshot captured under the LockedEntry lock by the caller.
func (fh *FileHandle) tryPeerRead(ctx context.Context, fileSize int64, buff []byte, offset int64, chunks []*filer_pb.FileChunk) (int64, int64, error) {
func (fh *FileHandle) tryPeerRead(ctx context.Context, fileSize int64, buff []byte, offset int64, entry *LockedEntry) (int64, int64, error) {
if fh.wfs.peerRegistrar == nil || fh.wfs.peerConnPool == nil {
return 0, 0, fmt.Errorf("peer sharing not configured")
}
@@ -64,7 +63,7 @@ func (fh *FileHandle) tryPeerRead(ctx context.Context, fileSize int64, buff []by
if readStop > fileSize {
readStop = fileSize
}
dataChunks, _, err := filer.ResolveChunkManifest(ctx, fh.wfs.LookupFn(), chunks, offset, readStop)
dataChunks, _, err := filer.ResolveChunkManifest(ctx, fh.wfs.LookupFn(), entry.GetEntry().Chunks, offset, readStop)
if err != nil {
return 0, 0, fmt.Errorf("resolve manifest: %w", err)
}
+2 -14
View File
@@ -15,7 +15,6 @@ import (
"github.com/seaweedfs/seaweedfs/weed/cluster"
"github.com/seaweedfs/seaweedfs/weed/filer"
"github.com/seaweedfs/seaweedfs/weed/filer/posixlock"
"github.com/seaweedfs/seaweedfs/weed/glog"
"github.com/seaweedfs/seaweedfs/weed/mount/meta_cache"
"github.com/seaweedfs/seaweedfs/weed/mount/page_writer"
@@ -98,10 +97,8 @@ type Option struct {
// EnableDistributedLock enables DLM-based write coordination across mounts.
// When true, opening a file for write acquires a distributed lock that is
// held (with auto-renewal) until the file is closed, so only one mount can
// have a file open for writing at a time; POSIX advisory locks (flock/fcntl)
// are also routed to the inode's owner filer so they are honored across
// mounts. Disabled under writeback cache, which implies single-writer.
// held (with auto-renewal) until the file is closed. Only one mount can
// have a file open for writing at a time.
EnableDistributedLock bool
// WritebackCache enables async flush on close for improved small file write performance.
@@ -141,9 +138,6 @@ type WFS struct {
fhLockTable *util.LockTable[FileHandleId]
hardLinkLockTable *util.LockTable[string]
posixLocks *PosixLockTable
posixSid uint64 // this mount's session id, for routed-lock owner identity
posixHint *posixLockHint // local fcntl-lock hint for routed mode
posixOwn *posixlock.Manager // mirror of locks this mount holds, re-asserted via keepalive
rdmaClient *RDMAMountClient
peerRegistrar *PeerRegistrar
peerDirectory *PeerDirectory
@@ -248,9 +242,6 @@ func NewSeaweedFileSystem(option *Option) *WFS {
fhLockTable: util.NewLockTable[FileHandleId](),
hardLinkLockTable: util.NewLockTable[string](),
posixLocks: NewPosixLockTable(),
posixSid: randomPosixSid(),
posixHint: newPosixLockHint(),
posixOwn: posixlock.NewManager(),
refreshingDirs: make(map[util.FullPath]struct{}),
atimeMap: make(map[uint64]time.Time, 8192),
openMtimeCache: make(map[uint64][2]int64, 8192),
@@ -514,9 +505,6 @@ func (wfs *WFS) StartBackgroundTasks() error {
go wfs.loopFlushDirtyMetadata()
go wfs.loopEvictIdleDirCache()
go wfs.loopProactiveFlush()
if wfs.crossMountLocks() {
go wfs.loopRenewPosixLeases()
}
return nil
}
+2 -24
View File
@@ -23,21 +23,10 @@ func (wfs *WFS) GetAttr(cancel <-chan struct{}, input *fuse.GetAttrIn, out *fuse
}
inode := input.NodeId
path, fh, entry, status := wfs.maybeReadEntry(inode)
path, _, entry, status := wfs.maybeReadEntry(inode)
if status == fuse.OK {
out.AttrValid = wfs.attrValidSec
// When an open handle owns the entry, async upload workers append
// chunks under the LockedEntry lock; take it for reading so FileSize
// does not iterate the chunk slice mid-reallocation. Re-read under the
// lock in case SetEntry swapped the pointer since maybeReadEntry.
if fh != nil {
fh.entry.RLock()
entry = fh.entry.Entry
}
wfs.setAttrByPbEntry(&out.Attr, inode, entry, true)
if fh != nil {
fh.entry.RUnlock()
}
wfs.applyInMemoryAtime(&out.Attr, inode)
if entry.IsDirectory {
wfs.applyInMemoryDirMtime(&out.Attr, inode)
@@ -51,9 +40,7 @@ func (wfs *WFS) GetAttr(cancel <-chan struct{}, input *fuse.GetAttrIn, out *fuse
out.AttrValid = wfs.attrValidSec
// Use shared lock to prevent race with Write operations
fhActiveLock := wfs.fhLockTable.AcquireLock("GetAttr", fh.fh, util.SharedLock)
fh.entry.RLock()
wfs.setAttrByPbEntry(&out.Attr, inode, fh.entry.Entry, true)
fh.entry.RUnlock()
wfs.setAttrByPbEntry(&out.Attr, inode, fh.entry.GetEntry(), true)
wfs.fhLockTable.ReleaseLock(fh.fh, fhActiveLock)
wfs.applyInMemoryAtime(&out.Attr, inode)
out.Nlink = 0
@@ -78,15 +65,6 @@ func (wfs *WFS) SetAttr(cancel <-chan struct{}, input *fuse.SetAttrIn, out *fuse
if fh != nil {
fh.entryLock.Lock()
defer fh.entryLock.Unlock()
// entry is the handle's shared LockedEntry.Entry. Async upload workers
// mutate its Chunks slice under the LockedEntry lock (AddChunks); hold
// that same lock so the truncate and FileSize reads below don't tear
// against a concurrent append. Re-read under the lock in case SetEntry
// swapped the pointer since maybeReadEntry, so we don't mutate an
// orphaned entry and lose the update.
fh.entry.Lock()
defer fh.entry.Unlock()
entry = fh.entry.Entry
}
wormEnforced, wormEnabled := wfs.wormEnforcedForEntry(path, entry)
-152
View File
@@ -1,152 +0,0 @@
package mount
import (
"sync"
"testing"
"github.com/seaweedfs/go-fuse/v2/fuse"
"github.com/seaweedfs/seaweedfs/weed/filer"
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
"github.com/seaweedfs/seaweedfs/weed/util"
)
// TestAttrChunkRace guards the locking around an open handle's chunk slice.
//
// With writebackCache, async upload workers append chunks to an open file
// handle's shared entry under the LockedEntry lock (FileHandle.AddChunks),
// while metadata ops compute the file size by iterating entry.Chunks. SetAttr
// and GetAttr used to read that slice without the LockedEntry lock, so a
// concurrent append that reallocated the backing array produced a torn slice
// read and a nil pointer dereference in filer.TotalSize. Run under -race.
func TestAttrChunkRace(t *testing.T) {
wfs := &WFS{
option: &Option{},
inodeToPath: NewInodeToPath(util.FullPath("/"), 0),
fhMap: NewFileHandleToInode(),
openMtimeCache: make(map[uint64][2]int64, 8),
}
const inode = uint64(42)
fullPath := util.FullPath("/dir/sample.txt")
wfs.inodeToPath.Lookup(fullPath, 1, false, false, inode, true)
entry := &filer_pb.Entry{
Name: "sample.txt",
Attributes: &filer_pb.FuseAttributes{FileMode: 0644},
}
chunkGroup, err := filer.NewChunkGroup(nil, nil, nil, 1)
if err != nil {
t.Fatalf("NewChunkGroup: %v", err)
}
fh := &FileHandle{
fh: FileHandleId(1),
inode: inode,
wfs: wfs,
entry: &LockedEntry{Entry: entry},
entryChunkGroup: chunkGroup,
}
wfs.fhMap.inode2fh[inode] = fh
wfs.fhMap.fh2inode[fh.fh] = inode
const iterations = 2000
var wg sync.WaitGroup
wg.Add(3)
// Async uploader: append chunks, reallocating the backing array.
go func() {
defer wg.Done()
for i := 0; i < iterations; i++ {
fh.AddChunks([]*filer_pb.FileChunk{{FileId: "x", Offset: int64(i), Size: 1}})
}
}()
// SetAttr: mtime-only recomputes FileSize by iterating chunks; a shrinking
// size takes the truncate path that rewrites entry.Chunks under the lock.
go func() {
defer wg.Done()
for i := 0; i < iterations; i++ {
in := &fuse.SetAttrIn{}
in.NodeId = inode
if i%2 == 0 {
in.Valid = fuse.FATTR_MTIME
in.Mtime = uint64(i)
} else {
in.Valid = fuse.FATTR_SIZE
in.Size = uint64(i % 8)
}
var out fuse.AttrOut
wfs.SetAttr(nil, in, &out)
}
}()
// GetAttr also computes FileSize by iterating chunks.
go func() {
defer wg.Done()
for i := 0; i < iterations; i++ {
in := &fuse.GetAttrIn{}
in.NodeId = inode
var out fuse.AttrOut
wfs.GetAttr(nil, in, &out)
}
}()
wg.Wait()
}
// TestReadFromChunksRace guards the read path's chunk-slice access. The read
// path holds fh.entryLock (which excludes SetAttr) but not the LockedEntry lock
// the async uploader appends under, so readFromChunks used to compute FileSize
// and walk entry.Chunks while AddChunks reallocated the slice. Run under -race.
func TestReadFromChunksRace(t *testing.T) {
wfs := &WFS{
option: &Option{},
inodeToPath: NewInodeToPath(util.FullPath("/"), 0),
fhMap: NewFileHandleToInode(),
}
const inode = uint64(42)
fullPath := util.FullPath("/dir/sample.txt")
wfs.inodeToPath.Lookup(fullPath, 1, false, false, inode, true)
// FileSize 0 forces readFromChunks down the filer.FileSize(chunks) branch.
entry := &filer_pb.Entry{
Name: "sample.txt",
Attributes: &filer_pb.FuseAttributes{FileMode: 0644},
}
chunkGroup, err := filer.NewChunkGroup(nil, nil, nil, 1)
if err != nil {
t.Fatalf("NewChunkGroup: %v", err)
}
fh := &FileHandle{
fh: FileHandleId(1),
inode: inode,
wfs: wfs,
entry: &LockedEntry{Entry: entry},
entryChunkGroup: chunkGroup,
}
wfs.fhMap.inode2fh[inode] = fh
wfs.fhMap.fh2inode[fh.fh] = inode
const iterations = 2000
var wg sync.WaitGroup
wg.Add(2)
go func() {
defer wg.Done()
for i := 0; i < iterations; i++ {
fh.AddChunks([]*filer_pb.FileChunk{{FileId: "x", Offset: int64(i), Size: 1}})
}
}()
// A read past EOF returns before touching the volume tier, but only after
// the racy size/chunk snapshot has run.
go func() {
defer wg.Done()
buff := make([]byte, 16)
for i := 0; i < iterations; i++ {
fh.readFromChunks(buff, 1<<62)
}
}()
wg.Wait()
}
+1 -1
View File
@@ -148,7 +148,7 @@ func (wfs *WFS) Release(cancel <-chan struct{}, in *fuse.ReleaseIn) {
}
}
if in.ReleaseFlags&fuse.FUSE_RELEASE_FLOCK_UNLOCK != 0 {
wfs.releaseFlockOwner(in.NodeId, in.LockOwner)
wfs.posixLocks.ReleaseFlockOwner(in.NodeId, in.LockOwner)
}
wfs.ReleaseHandle(FileHandleId(in.Fh))
}
-9
View File
@@ -10,9 +10,6 @@ import (
// If a conflict exists, the conflicting lock is returned in out.
// If no conflict, out.Lk.Typ is set to F_UNLCK.
func (wfs *WFS) GetLk(cancel <-chan struct{}, in *fuse.LkIn, out *fuse.LkOut) fuse.Status {
if wfs.crossMountLocks() {
return wfs.routedGetLk(cancel, in, out)
}
proposed := lockRange{
Start: in.Lk.Start,
End: in.Lk.End,
@@ -28,9 +25,6 @@ func (wfs *WFS) GetLk(cancel <-chan struct{}, in *fuse.LkIn, out *fuse.LkOut) fu
// SetLk sets or clears a POSIX lock (non-blocking).
// Returns EAGAIN if the lock conflicts with an existing lock from another owner.
func (wfs *WFS) SetLk(cancel <-chan struct{}, in *fuse.LkIn) fuse.Status {
if wfs.crossMountLocks() {
return wfs.routedSetLk(cancel, in)
}
lk := lockRange{
Start: in.Lk.Start,
End: in.Lk.End,
@@ -45,9 +39,6 @@ func (wfs *WFS) SetLk(cancel <-chan struct{}, in *fuse.LkIn) fuse.Status {
// SetLkw sets a POSIX lock (blocking).
// Waits until the lock can be acquired or the request is cancelled.
func (wfs *WFS) SetLkw(cancel <-chan struct{}, in *fuse.LkIn) fuse.Status {
if wfs.crossMountLocks() {
return wfs.routedSetLkw(cancel, in)
}
lk := lockRange{
Start: in.Lk.Start,
End: in.Lk.End,
+3 -3
View File
@@ -60,7 +60,7 @@ func (wfs *WFS) Flush(cancel <-chan struct{}, in *fuse.FlushIn) fuse.Status {
// If handle is not found, it might have been already released
// This is not an error condition for FLUSH
if in.LockOwner != 0 {
wfs.releasePosixOwner(in.NodeId, in.LockOwner)
wfs.posixLocks.ReleasePosixOwner(in.NodeId, in.LockOwner)
}
return fuse.OK
}
@@ -69,11 +69,11 @@ func (wfs *WFS) Flush(cancel <-chan struct{}, in *fuse.FlushIn) fuse.Status {
// did not hold byte-range locks. Only force the synchronous close path when
// this owner actually has POSIX locks to release; otherwise writebackCache
// would silently degrade to a blocking flush for ordinary close().
hasPosixLocks := wfs.hasPosixOwner(in.NodeId, in.LockOwner)
hasPosixLocks := wfs.posixLocks.HasPosixOwner(in.NodeId, in.LockOwner)
allowAsync := !hasPosixLocks
status := wfs.doFlush(fh, in.Uid, in.Gid, allowAsync)
if in.LockOwner != 0 {
wfs.releasePosixOwner(in.NodeId, in.LockOwner)
wfs.posixLocks.ReleasePosixOwner(in.NodeId, in.LockOwner)
}
return status
}
-426
View File
@@ -1,426 +0,0 @@
package mount
import (
"context"
"encoding/binary"
"encoding/hex"
"math/rand/v2"
"sync"
"syscall"
"time"
"github.com/seaweedfs/go-fuse/v2/fuse"
"github.com/seaweedfs/seaweedfs/weed/filer/posixlock"
"github.com/seaweedfs/seaweedfs/weed/glog"
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
"github.com/seaweedfs/seaweedfs/weed/util"
)
// randomPosixSid returns a process-stable, cluster-unique-enough session id for
// this mount, used to namespace lock owners so the same FUSE owner value on a
// different mount never aliases, and to scope lease-based reaping.
func randomPosixSid() uint64 {
return binary.BigEndian.Uint64(util.RandomBytes(8))
}
// Routed POSIX locking: when -dlm is set, advisory locks are serialized on the
// inode's owner filer instead of in this mount's local table, so locks are
// honored across mounts. The mount calls its filer and relies on filer-side
// forwarding to reach the owner. Blocking (SetLkw) is client-side polling — there
// is no server-side wait queue.
const (
posixLockMinBackoff = 5 * time.Millisecond
posixLockMaxBackoff = 200 * time.Millisecond
posixLockKeyPrefix = "s3.fuse.lock:"
// posixLockReleaseTimeout bounds the background unlock/release RPCs. They run
// off the syscall path (close/flush) and must not be cancelled by an
// interrupt, but a deadline keeps a slow or unreachable filer from blocking
// close indefinitely. It also bounds each keepalive RPC.
posixLockReleaseTimeout = 5 * time.Second
// posixKeepaliveInterval renews the session lease well within the filer's
// posixLockSessionTTL (15s) so a live mount is never reaped.
posixKeepaliveInterval = 5 * time.Second
)
// posixLockHint records, per inode, which owners this mount has taken fcntl locks
// for. Flush consults it to force a synchronous close and route a release without
// an RPC on every close(). It is a superset hint: an extra sync flush or a no-op
// release RPC is harmless, a missed release is not — so it is added on grant and
// cleared on close (which drops the owner's POSIX locks per POSIX semantics).
type posixLockHint struct {
mu sync.Mutex
m map[uint64]map[uint64]struct{}
}
func newPosixLockHint() *posixLockHint {
return &posixLockHint{m: make(map[uint64]map[uint64]struct{})}
}
func (h *posixLockHint) add(inode, owner uint64) {
h.mu.Lock()
defer h.mu.Unlock()
owners := h.m[inode]
if owners == nil {
owners = make(map[uint64]struct{})
h.m[inode] = owners
}
owners[owner] = struct{}{}
}
func (h *posixLockHint) has(inode, owner uint64) bool {
h.mu.Lock()
defer h.mu.Unlock()
_, ok := h.m[inode][owner]
return ok
}
func (h *posixLockHint) drop(inode, owner uint64) {
h.mu.Lock()
defer h.mu.Unlock()
if owners := h.m[inode]; owners != nil {
delete(owners, owner)
if len(owners) == 0 {
delete(h.m, inode)
}
}
}
// posixLockKeyForInode resolves a FUSE inode to its cluster-stable lock identity:
// the HardLinkId for a hardlinked file (so all names share one owner and table),
// else the path. POSIX locks are inode-scoped, so this must never be the FUSE
// NodeId (mount-local) for a multi-named inode.
func (wfs *WFS) posixLockKeyForInode(inode uint64) (string, bool) {
path, status := wfs.inodeToPath.GetPath(inode)
if status != fuse.OK {
return "", false
}
if entry, st := wfs.maybeLoadEntry(path); st == fuse.OK && entry != nil && len(entry.HardLinkId) > 0 {
return posixLockKeyPrefix + "hl:" + hex.EncodeToString(entry.HardLinkId), true
}
return posixLockKeyPrefix + string(path), true
}
func posixLockTypeToWire(typ uint32) uint32 {
switch typ {
case syscall.F_RDLCK:
return posixlock.Read
case syscall.F_WRLCK:
return posixlock.Write
default:
return posixlock.Unlock
}
}
func posixLockTypeFromWire(typ uint32) uint32 {
switch typ {
case posixlock.Read:
return syscall.F_RDLCK
case posixlock.Write:
return syscall.F_WRLCK
default:
return syscall.F_UNLCK
}
}
func (wfs *WFS) posixRangeFromLkIn(in *fuse.LkIn) posixlock.Range {
return posixlock.Range{
Start: in.Lk.Start,
End: in.Lk.End,
Type: posixLockTypeToWire(in.Lk.Typ),
Sid: wfs.posixSid,
Owner: in.Owner,
Pid: in.Lk.Pid,
IsFlock: in.LkFlags&fuse.FUSE_LK_FLOCK != 0,
}
}
// posixLockContext derives a context from the FUSE cancel channel so an
// interrupted lock syscall aborts its in-flight RPC. The caller must call the
// returned func to release the watcher goroutine.
func posixLockContext(cancel <-chan struct{}) (context.Context, context.CancelFunc) {
ctx, cancelFn := context.WithCancel(context.Background())
if cancel != nil {
go func() {
select {
case <-cancel:
cancelFn()
case <-ctx.Done():
}
}()
}
return ctx, cancelFn
}
func (wfs *WFS) callPosixLock(ctx context.Context, key string, op filer_pb.PosixLockOp, lk posixlock.Range) (*filer_pb.PosixLockResponse, error) {
var resp *filer_pb.PosixLockResponse
err := wfs.WithFilerClient(false, func(client filer_pb.SeaweedFilerClient) error {
var e error
resp, e = client.PosixLock(ctx, &filer_pb.PosixLockRequest{
Key: key,
Op: op,
Lock: &filer_pb.PosixLockRange{
Start: lk.Start, End: lk.End, Type: lk.Type,
Sid: lk.Sid, Owner: lk.Owner, Pid: lk.Pid, IsFlock: lk.IsFlock,
},
})
return e
})
return resp, err
}
func (wfs *WFS) routedGetLk(cancel <-chan struct{}, in *fuse.LkIn, out *fuse.LkOut) fuse.Status {
key, ok := wfs.posixLockKeyForInode(in.NodeId)
if !ok {
return fuse.EINVAL
}
ctx, done := posixLockContext(cancel)
defer done()
resp, err := wfs.callPosixLock(ctx, key, filer_pb.PosixLockOp_GET_LK, wfs.posixRangeFromLkIn(in))
if err != nil {
glog.Warningf("routed GetLk %s: %v", key, err)
return fuse.EIO
}
if resp.GetHasConflict() {
c := resp.GetConflict()
out.Lk.Start, out.Lk.End, out.Lk.Pid = c.GetStart(), c.GetEnd(), c.GetPid()
out.Lk.Typ = posixLockTypeFromWire(c.GetType())
} else {
out.Lk.Typ = syscall.F_UNLCK
}
return fuse.OK
}
func (wfs *WFS) routedSetLk(cancel <-chan struct{}, in *fuse.LkIn) fuse.Status {
key, ok := wfs.posixLockKeyForInode(in.NodeId)
if !ok {
return fuse.EINVAL
}
lk := wfs.posixRangeFromLkIn(in)
if lk.Type == posixlock.Unlock {
return wfs.routedUnlock(key, lk)
}
ctx, done := posixLockContext(cancel)
defer done()
resp, err := wfs.callPosixLock(ctx, key, filer_pb.PosixLockOp_TRY_LOCK, lk)
if err != nil {
glog.Warningf("routed SetLk %s: %v", key, err)
return fuse.EIO
}
if !resp.GetGranted() {
return fuse.EAGAIN
}
wfs.recordPosixGrant(key, in.NodeId, lk)
return fuse.OK
}
func (wfs *WFS) routedSetLkw(cancel <-chan struct{}, in *fuse.LkIn) fuse.Status {
key, ok := wfs.posixLockKeyForInode(in.NodeId)
if !ok {
return fuse.EINVAL
}
lk := wfs.posixRangeFromLkIn(in)
if lk.Type == posixlock.Unlock {
return wfs.routedUnlock(key, lk)
}
ctx, done := posixLockContext(cancel)
defer done()
status := posixPollAcquire(cancel, func() (bool, error) {
resp, err := wfs.callPosixLock(ctx, key, filer_pb.PosixLockOp_TRY_LOCK, lk)
if err != nil {
return false, err
}
return resp.GetGranted(), nil
})
if status == fuse.OK {
wfs.recordPosixGrant(key, in.NodeId, lk)
}
return status
}
// recordPosixGrant notes a newly granted lock: the flush hint (fcntl only) and
// the own-lock mirror, which the keepalive re-asserts to the key's owner so the
// lease is renewed and the owner can rebuild after a takeover or restart.
func (wfs *WFS) recordPosixGrant(key string, inode uint64, lk posixlock.Range) {
if !lk.IsFlock {
wfs.posixHint.add(inode, lk.Owner)
}
wfs.posixOwn.Track(key, lk)
}
func (wfs *WFS) routedUnlock(key string, lk posixlock.Range) fuse.Status {
// Drop the range from the mirror first so a keepalive re-assertion can't
// resurrect it; if the RPC below fails, the next re-assertion reconciles the
// owner to this (released) state anyway.
wfs.posixOwn.Unlock(key, lk)
// A release must complete even if the syscall was interrupted; cancelling it
// would leak the lock on the owner filer. Bound it so a stuck filer can't
// hang close() forever.
ctx, cancel := context.WithTimeout(context.Background(), posixLockReleaseTimeout)
defer cancel()
if _, err := wfs.callPosixLock(ctx, key, filer_pb.PosixLockOp_UNLOCK, lk); err != nil {
glog.Warningf("routed unlock %s: %v", key, err)
return fuse.EIO
}
return fuse.OK
}
// posixPollAcquire retries try with capped, jittered backoff until it is granted,
// the cancel channel fires (EINTR), or try errors (EIO). This is the client-side
// stand-in for a server-side wait queue.
func posixPollAcquire(cancel <-chan struct{}, try func() (bool, error)) fuse.Status {
backoff := posixLockMinBackoff
for {
select {
case <-cancel:
return fuse.EINTR
default:
}
granted, err := try()
if err != nil {
return fuse.EIO
}
if granted {
return fuse.OK
}
timer := time.NewTimer(backoff + time.Duration(rand.Int64N(int64(backoff))))
select {
case <-cancel:
timer.Stop()
return fuse.EINTR
case <-timer.C:
}
if backoff < posixLockMaxBackoff {
if backoff *= 2; backoff > posixLockMaxBackoff {
backoff = posixLockMaxBackoff
}
}
}
}
func (wfs *WFS) routedReleasePosixOwner(inode, owner uint64) {
if !wfs.posixHint.has(inode, owner) {
return
}
key, ok := wfs.posixLockKeyForInode(inode)
if !ok {
return
}
wfs.posixOwn.ReleasePosixOwner(key, wfs.posixSid, owner)
ctx, cancel := context.WithTimeout(context.Background(), posixLockReleaseTimeout)
defer cancel()
if _, err := wfs.callPosixLock(ctx, key, filer_pb.PosixLockOp_RELEASE_POSIX_OWNER, posixlock.Range{Sid: wfs.posixSid, Owner: owner}); err != nil {
// Keep the hint so a later flush retries the release; dropping it on a
// transient failure would strand the lock until the owner filer's
// session-lease reaping expires it.
glog.Warningf("routed release posix owner %s: %v", key, err)
return
}
wfs.posixHint.drop(inode, owner)
}
func (wfs *WFS) routedReleaseFlockOwner(inode, owner uint64) {
key, ok := wfs.posixLockKeyForInode(inode)
if !ok {
return
}
wfs.posixOwn.ReleaseFlockOwner(key, wfs.posixSid, owner)
ctx, cancel := context.WithTimeout(context.Background(), posixLockReleaseTimeout)
defer cancel()
if _, err := wfs.callPosixLock(ctx, key, filer_pb.PosixLockOp_RELEASE_FLOCK_OWNER, posixlock.Range{Sid: wfs.posixSid, Owner: owner}); err != nil {
glog.Warningf("routed release flock owner %s: %v", key, err)
}
}
// crossMountLocks reports whether advisory locks are routed to the inode's owner
// filer rather than served from the per-mount local table. It tracks -dlm: lock
// coordination rides the same switch as whole-file write coordination, and is
// off under writeback cache (which implies single-writer, so lockClient is nil).
func (wfs *WFS) crossMountLocks() bool {
return wfs.lockClient != nil
}
// releasePosixOwner / releaseFlockOwner / hasPosixOwner dispatch to the routed
// authority or the local table based on crossMountLocks, keeping the Flush and
// Release call sites flag-agnostic.
func (wfs *WFS) releasePosixOwner(inode, owner uint64) {
if wfs.crossMountLocks() {
wfs.routedReleasePosixOwner(inode, owner)
return
}
wfs.posixLocks.ReleasePosixOwner(inode, owner)
}
func (wfs *WFS) releaseFlockOwner(inode, owner uint64) {
if wfs.crossMountLocks() {
wfs.routedReleaseFlockOwner(inode, owner)
return
}
wfs.posixLocks.ReleaseFlockOwner(inode, owner)
}
func (wfs *WFS) hasPosixOwner(inode, owner uint64) bool {
if wfs.crossMountLocks() {
return owner != 0 && wfs.posixHint.has(inode, owner)
}
return wfs.posixLocks.HasPosixOwner(inode, owner)
}
// callPosixReassert sends the mount's held locks on key to the key's current
// owner filer as a KEEP_ALIVE, which renews the lease and lets the owner rebuild
// its in-memory state after a takeover or restart.
func (wfs *WFS) callPosixReassert(ctx context.Context, key string, locks []posixlock.Range) error {
pbLocks := make([]*filer_pb.PosixLockRange, 0, len(locks))
for _, l := range locks {
pbLocks = append(pbLocks, &filer_pb.PosixLockRange{
Start: l.Start, End: l.End, Type: l.Type,
Sid: l.Sid, Owner: l.Owner, Pid: l.Pid, IsFlock: l.IsFlock,
})
}
return wfs.WithFilerClient(false, func(client filer_pb.SeaweedFilerClient) error {
_, e := client.PosixLock(ctx, &filer_pb.PosixLockRequest{
Key: key,
Op: filer_pb.PosixLockOp_KEEP_ALIVE,
Lock: &filer_pb.PosixLockRange{Sid: wfs.posixSid},
Locks: pbLocks,
})
return e
})
}
// posixKeepaliveConcurrency bounds the parallel keepalive RPCs per tick. A mount
// holding locks on many inodes must renew every lease well within the filer TTL;
// dispatching them concurrently keeps the round trip from scaling with lock count.
const posixKeepaliveConcurrency = 32
// loopRenewPosixLeases re-asserts this mount's held locks to the owner filer of
// every key, which renews the session lease and rebuilds the owner's state after
// a ring change or owner restart. A dead mount stops re-asserting and the owners'
// sweepers reclaim its locks after the TTL.
func (wfs *WFS) loopRenewPosixLeases() {
ticker := time.NewTicker(posixKeepaliveInterval)
defer ticker.Stop()
for range ticker.C {
held := wfs.posixOwn.Snapshot()
var wg sync.WaitGroup
sem := make(chan struct{}, posixKeepaliveConcurrency)
for key, locks := range held {
wg.Add(1)
sem <- struct{}{}
go func() {
defer wg.Done()
defer func() { <-sem }()
// Bound each re-assertion so a stuck filer can't block wg.Wait and
// stall the whole tick, which would let other keys' leases expire
// and get reaped.
ctx, cancel := context.WithTimeout(context.Background(), posixLockReleaseTimeout)
defer cancel()
if err := wfs.callPosixReassert(ctx, key, locks); err != nil {
glog.V(2).Infof("posix reassert %s: %v", key, err)
}
}()
}
wg.Wait()
}
}
@@ -1,94 +0,0 @@
package mount
import (
"syscall"
"testing"
"time"
"github.com/seaweedfs/go-fuse/v2/fuse"
"github.com/seaweedfs/seaweedfs/weed/filer/posixlock"
)
func TestPosixLockTypeMapping(t *testing.T) {
cases := []struct {
sys uint32
wire uint32
}{
{syscall.F_RDLCK, posixlock.Read},
{syscall.F_WRLCK, posixlock.Write},
{syscall.F_UNLCK, posixlock.Unlock},
}
for _, c := range cases {
if got := posixLockTypeToWire(c.sys); got != c.wire {
t.Errorf("toWire(%d) = %d, want %d", c.sys, got, c.wire)
}
if got := posixLockTypeFromWire(c.wire); got != c.sys {
t.Errorf("fromWire(%d) = %d, want %d", c.wire, got, c.sys)
}
}
}
func TestPosixPollAcquireGrantedImmediately(t *testing.T) {
calls := 0
st := posixPollAcquire(nil, func() (bool, error) { calls++; return true, nil })
if st != fuse.OK || calls != 1 {
t.Fatalf("immediate grant: status=%v calls=%d", st, calls)
}
}
func TestPosixPollAcquireRetriesThenGrants(t *testing.T) {
calls := 0
st := posixPollAcquire(nil, func() (bool, error) {
calls++
return calls >= 3, nil
})
if st != fuse.OK || calls != 3 {
t.Fatalf("retry then grant: status=%v calls=%d", st, calls)
}
}
func TestPosixPollAcquireError(t *testing.T) {
if st := posixPollAcquire(nil, func() (bool, error) { return false, syscall.EIO }); st != fuse.EIO {
t.Fatalf("error should map to EIO, got %v", st)
}
}
func TestPosixPollAcquireCancel(t *testing.T) {
cancel := make(chan struct{})
close(cancel)
done := make(chan fuse.Status, 1)
go func() {
done <- posixPollAcquire(cancel, func() (bool, error) { return false, nil })
}()
select {
case st := <-done:
if st != fuse.EINTR {
t.Fatalf("cancel should map to EINTR, got %v", st)
}
case <-time.After(2 * time.Second):
t.Fatal("poll did not return on cancel")
}
}
func TestPosixLockHint(t *testing.T) {
h := newPosixLockHint()
if h.has(1, 2) {
t.Fatal("empty hint should not report a lock")
}
h.add(1, 2)
h.add(1, 3)
if !h.has(1, 2) || !h.has(1, 3) {
t.Fatal("added owners should be reported")
}
h.drop(1, 2)
if h.has(1, 2) {
t.Fatal("dropped owner should be gone")
}
if !h.has(1, 3) {
t.Fatal("sibling owner should remain")
}
h.drop(1, 3)
if _, ok := h.m[1]; ok {
t.Fatal("inode entry should be removed when its last owner drops")
}
}
-191
View File
@@ -31,15 +31,6 @@ service SeaweedFiler {
rpc DeleteEntry (DeleteEntryRequest) returns (DeleteEntryResponse) {
}
rpc ObjectTransaction (ObjectTransactionRequest) returns (ObjectTransactionResponse) {
}
rpc ObjectTransactionBatch (ObjectTransactionBatchRequest) returns (ObjectTransactionBatchResponse) {
}
rpc PosixLock (PosixLockRequest) returns (PosixLockResponse) {
}
rpc AtomicRenameEntry (AtomicRenameEntryRequest) returns (AtomicRenameEntryResponse) {
}
rpc StreamRenameEntry (StreamRenameEntryRequest) returns (stream StreamRenameEntryResponse) {
@@ -231,56 +222,6 @@ message CreateEntryRequest {
bool is_from_other_cluster = 4;
repeated int32 signatures = 5;
bool skip_check_parent_directory = 6;
// Optional precondition evaluated against the current entry atomically with
// the write, under the filer's per-path lock. The caller must route the
// key's writes to this entry's owner filer for the check to be authoritative.
WriteCondition condition = 7;
}
// WriteCondition is the precondition the filer evaluates against the existing
// entry before writing, under the per-path lock. A failed condition returns
// FilerError PRECONDITION_FAILED. The client maps request semantics (e.g. RFC
// 7232) to clauses; the filer just compares.
//
// A condition is a list of clauses that ALL must hold (logical AND). One clause
// is the common case; several express what a single comparison cannot: an ETag
// set (If-Match / If-None-Match with multiple values), weak-ETag comparison, and
// compound conditions (e.g. If-Match + If-Unmodified-Since together).
message WriteCondition {
enum Kind {
NONE = 0; // unconditional
IF_NOT_EXISTS = 1; // fail if the entry exists (If-None-Match: *)
IF_EXISTS = 2; // fail if the entry is absent (If-Match: *)
IF_ETAG_MATCH = 3; // fail if absent or etag matches none of the set (If-Match)
IF_ETAG_NOT_MATCH = 4; // fail if present and etag matches any of the set (If-None-Match)
IF_UNMODIFIED_SINCE = 5; // fail if present and mtime > unix_time
IF_MODIFIED_SINCE = 6; // fail if present and mtime <= unix_time
IF_EXTENDED_NOT_EQUAL = 7; // fail if present and extended[ext_key] == ext_value
IF_EXTENDED_TIME_ELAPSED = 8; // fail if present and extended[ext_key] (unix seconds) is in the future
}
// Clause is one primitive comparison. IF_ETAG_MATCH holds when the current
// entry's ETag equals any value in etags; IF_ETAG_NOT_MATCH holds when it
// equals none. allow_weak permits weak-comparison (ignoring the W/ prefix).
//
// The IF_EXTENDED_* kinds are generic guards on an extended attribute, used
// to enforce object-lock without teaching the filer S3 semantics:
// IF_EXTENDED_NOT_EQUAL expresses a legal hold (block while a key equals a
// value), and IF_EXTENDED_TIME_ELAPSED expresses retention (block while a
// stored unix-second deadline is in the future, compared to the filer's
// clock). The caller composes these and, for governance-bypass, simply omits
// the retention clause when the bypass is authorized — the filer makes no
// authorization decision.
message Clause {
Kind kind = 1;
repeated string etags = 2; // ETag set for IF_ETAG_* kinds
int64 unix_time = 3; // bound (unix seconds) for IF_*_SINCE kinds
bool allow_weak = 4; // compare ETags ignoring the weak (W/) marker
string ext_key = 5; // extended attribute name for IF_EXTENDED_* kinds
string ext_value = 6; // blocking value for IF_EXTENDED_NOT_EQUAL
string gate_key = 7; // IF_EXTENDED_TIME_ELAPSED: only enforce when extended[gate_key] == gate_value
string gate_value = 8; // gate value (e.g. retention mode COMPLIANCE for governance bypass)
}
repeated Clause clauses = 1; // all must hold (logical AND)
}
// Structured error codes for filer entry operations.
@@ -292,138 +233,6 @@ enum FilerError {
EXISTING_IS_DIRECTORY = 3; // cannot overwrite directory with file
EXISTING_IS_FILE = 4; // cannot overwrite file with directory
ENTRY_ALREADY_EXISTS = 5; // O_EXCL and entry already exists
PRECONDITION_FAILED = 6; // WriteCondition not satisfied
}
// ObjectMutation is one entry-level change applied by ObjectTransaction. All
// mutations of a transaction run under a single per-path lock (the request's
// lock_key) and in order, so the gateway can describe a multi-entry object
// operation as one request instead of holding a distributed lock across
// several RPCs. Data-bearing writes (entries with chunks) should be written
// before the transaction; mutations here are metadata-scoped.
message ObjectMutation {
enum Type {
PUT = 0; // create or replace the entry (entry field)
DELETE = 1; // delete the entry at directory/name (no error if absent)
PATCH_EXTENDED = 2; // merge set_extended / remove delete_extended on the entry
RECOMPUTE_LATEST = 3; // scan a directory and re-point a parent entry (recompute)
}
Type type = 1;
string directory = 2;
string name = 3; // entry name for DELETE / PATCH_EXTENDED / RECOMPUTE_LATEST (the pointer entry)
Entry entry = 4; // full entry for PUT
map<string, bytes> set_extended = 5; // PATCH_EXTENDED: keys to set
repeated string delete_extended = 6; // PATCH_EXTENDED: keys to remove
bool is_delete_data = 7; // DELETE: also delete chunk data
bool is_recursive = 8; // DELETE: recurse into a directory
Recompute recompute = 9; // RECOMPUTE_LATEST parameters
bool set_content = 10; // PATCH_EXTENDED: replace Entry.content with content
bytes content = 11; // PATCH_EXTENDED: new Entry.content when set_content
bool touch_mtime = 12; // PATCH_EXTENDED: set the entry's Mtime to now (e.g. a metadata-replace copy)
}
// Recompute re-derives a pointer entry (directory/name on the mutation) from the
// current contents of a scanned directory, atomically under the transaction's
// lock. It is mechanical: the filer picks the child that sorts first or last by
// name and copies the requested fields into the pointer; it has no knowledge of
// what the entries mean. The caller (which does know the versioning scheme)
// supplies the sort direction and the key mappings. This covers re-pointing the
// latest version after a specific version is deleted, where the scan must run
// under the lock.
message Recompute {
string scan_dir = 1; // directory whose direct children are scanned
bool descending = 2; // pick the child that sorts last by name (else first)
map<string, string> copy_extended = 3; // pointer extended key -> source extended key on the chosen child
string name_to_key = 4; // if set, store the chosen child's name under this pointer key
string size_to_key = 5; // if set, store the chosen child's FileSize (decimal) under this pointer key
string mtime_to_key = 6; // if set, store the chosen child's Mtime (decimal) under this pointer key
string demote_key = 7; // if set, stamp demote_value on the prior name_to_key target when it changes
bytes demote_value = 8; // value for demote_key
string exclude_name = 9; // if set, skip this child when scanning (e.g. a version about to be deleted)
}
// ObjectTransactionRequest applies an ordered list of mutations atomically with
// respect to other writers of the same object, by holding the filer's per-path
// lock on lock_key for the whole transaction. The optional condition is checked
// first, against condition_key when set, else lock_key. Callers set route_key to
// the object's stable owner ring key; a filer that is not the owner forwards the
// transaction one hop to the owner, so a stale ring view is tolerated.
message ObjectTransactionRequest {
string lock_key = 1; // object path to lock and to evaluate the condition against
WriteCondition condition = 2; // optional precondition, checked under the lock
repeated ObjectMutation mutations = 3;
bool is_from_other_cluster = 4;
repeated int32 signatures = 5;
string condition_key = 6; // if set, evaluate the condition against this entry instead of lock_key (still locking lock_key)
string route_key = 7; // ring key identifying the owner filer; a non-owner forwards the whole transaction to it
bool is_moved = 8; // set on a forwarded transaction so the receiver applies it locally instead of forwarding again
}
message ObjectTransactionResponse {
string error = 1;
FilerError error_code = 2;
}
// PosixLockRange is one advisory byte-range lock. Owner identity is (sid, owner):
// sid is the mount session, owner the FUSE lock owner within it, so owners from
// different mounts never alias. end is inclusive (max uint64 = to EOF); is_flock
// separates the flock and fcntl namespaces, which never conflict.
message PosixLockRange {
uint64 start = 1;
uint64 end = 2;
uint32 type = 3; // 1=read, 2=write, 3=unlock
uint64 sid = 4;
uint64 owner = 5;
uint32 pid = 6; // holder pid, for get_lk reporting only
bool is_flock = 7;
}
// PosixLock routes an advisory lock operation to the inode's owner filer, which
// holds the authoritative in-memory lock table. key is the inode identity ring
// key (the file path, or hl:<HardLinkId> for a hardlink) used both to resolve the
// owner and to index the table. A non-owner filer forwards the request one hop;
// is_moved bounds it so a stale ring view cannot loop.
message PosixLockRequest {
string key = 1;
bool is_moved = 2;
PosixLockOp op = 3;
PosixLockRange lock = 4;
// locks carries the full set a mount holds on key for a KEEP_ALIVE
// re-assertion, so the current owner filer can rebuild its in-memory state
// after an ownership change or restart. lock.sid identifies the session.
repeated PosixLockRange locks = 5;
// cooling_probe marks a dual-read a new owner sends to the previous owner
// during a ring change, so the previous owner answers from local state
// without itself cooling-off (no recursion).
bool cooling_probe = 6;
}
enum PosixLockOp {
TRY_LOCK = 0; // grant lock or report conflict (non-blocking)
UNLOCK = 1; // release lock's owner's locks over its range
GET_LK = 2; // report a conflicting lock, if any
RELEASE_POSIX_OWNER = 3; // drop the owner's fcntl locks (flush-time)
RELEASE_FLOCK_OWNER = 4; // drop the owner's flock locks (release-time)
KEEP_ALIVE = 5; // renew the session's lease on this owner (lock.sid)
}
message PosixLockResponse {
bool granted = 1; // for TRY_LOCK: whether the lock was granted
bool has_conflict = 2; // whether conflict is populated
PosixLockRange conflict = 3; // the blocking lock (TRY_LOCK conflict / GET_LK result)
}
// ObjectTransactionBatch applies several object transactions in one round trip,
// each under its own per-path lock and independent of the others (no cross-key
// atomicity). A caller groups keys that route to the same owner filer and sends
// one batch per owner, e.g. for a multi-object delete. Each response is parallel
// to its request.
message ObjectTransactionBatchRequest {
repeated ObjectTransactionRequest transactions = 1;
}
message ObjectTransactionBatchResponse {
repeated ObjectTransactionResponse responses = 1;
}
message CreateEntryResponse {
+407 -1676
View File
File diff suppressed because it is too large Load Diff
+1 -115
View File
@@ -1,7 +1,7 @@
// Code generated by protoc-gen-go-grpc. DO NOT EDIT.
// versions:
// - protoc-gen-go-grpc v1.6.2
// - protoc v6.33.4
// - protoc v7.34.1
// source: filer.proto
package filer_pb
@@ -26,9 +26,6 @@ const (
SeaweedFiler_TouchAccessTime_FullMethodName = "/filer_pb.SeaweedFiler/TouchAccessTime"
SeaweedFiler_AppendToEntry_FullMethodName = "/filer_pb.SeaweedFiler/AppendToEntry"
SeaweedFiler_DeleteEntry_FullMethodName = "/filer_pb.SeaweedFiler/DeleteEntry"
SeaweedFiler_ObjectTransaction_FullMethodName = "/filer_pb.SeaweedFiler/ObjectTransaction"
SeaweedFiler_ObjectTransactionBatch_FullMethodName = "/filer_pb.SeaweedFiler/ObjectTransactionBatch"
SeaweedFiler_PosixLock_FullMethodName = "/filer_pb.SeaweedFiler/PosixLock"
SeaweedFiler_AtomicRenameEntry_FullMethodName = "/filer_pb.SeaweedFiler/AtomicRenameEntry"
SeaweedFiler_StreamRenameEntry_FullMethodName = "/filer_pb.SeaweedFiler/StreamRenameEntry"
SeaweedFiler_StreamMutateEntry_FullMethodName = "/filer_pb.SeaweedFiler/StreamMutateEntry"
@@ -65,9 +62,6 @@ type SeaweedFilerClient interface {
TouchAccessTime(ctx context.Context, in *TouchAccessTimeRequest, opts ...grpc.CallOption) (*TouchAccessTimeResponse, error)
AppendToEntry(ctx context.Context, in *AppendToEntryRequest, opts ...grpc.CallOption) (*AppendToEntryResponse, error)
DeleteEntry(ctx context.Context, in *DeleteEntryRequest, opts ...grpc.CallOption) (*DeleteEntryResponse, error)
ObjectTransaction(ctx context.Context, in *ObjectTransactionRequest, opts ...grpc.CallOption) (*ObjectTransactionResponse, error)
ObjectTransactionBatch(ctx context.Context, in *ObjectTransactionBatchRequest, opts ...grpc.CallOption) (*ObjectTransactionBatchResponse, error)
PosixLock(ctx context.Context, in *PosixLockRequest, opts ...grpc.CallOption) (*PosixLockResponse, error)
AtomicRenameEntry(ctx context.Context, in *AtomicRenameEntryRequest, opts ...grpc.CallOption) (*AtomicRenameEntryResponse, error)
StreamRenameEntry(ctx context.Context, in *StreamRenameEntryRequest, opts ...grpc.CallOption) (grpc.ServerStreamingClient[StreamRenameEntryResponse], error)
StreamMutateEntry(ctx context.Context, opts ...grpc.CallOption) (grpc.BidiStreamingClient[StreamMutateEntryRequest, StreamMutateEntryResponse], error)
@@ -183,36 +177,6 @@ func (c *seaweedFilerClient) DeleteEntry(ctx context.Context, in *DeleteEntryReq
return out, nil
}
func (c *seaweedFilerClient) ObjectTransaction(ctx context.Context, in *ObjectTransactionRequest, opts ...grpc.CallOption) (*ObjectTransactionResponse, error) {
cOpts := append([]grpc.CallOption{grpc.StaticMethod()}, opts...)
out := new(ObjectTransactionResponse)
err := c.cc.Invoke(ctx, SeaweedFiler_ObjectTransaction_FullMethodName, in, out, cOpts...)
if err != nil {
return nil, err
}
return out, nil
}
func (c *seaweedFilerClient) ObjectTransactionBatch(ctx context.Context, in *ObjectTransactionBatchRequest, opts ...grpc.CallOption) (*ObjectTransactionBatchResponse, error) {
cOpts := append([]grpc.CallOption{grpc.StaticMethod()}, opts...)
out := new(ObjectTransactionBatchResponse)
err := c.cc.Invoke(ctx, SeaweedFiler_ObjectTransactionBatch_FullMethodName, in, out, cOpts...)
if err != nil {
return nil, err
}
return out, nil
}
func (c *seaweedFilerClient) PosixLock(ctx context.Context, in *PosixLockRequest, opts ...grpc.CallOption) (*PosixLockResponse, error) {
cOpts := append([]grpc.CallOption{grpc.StaticMethod()}, opts...)
out := new(PosixLockResponse)
err := c.cc.Invoke(ctx, SeaweedFiler_PosixLock_FullMethodName, in, out, cOpts...)
if err != nil {
return nil, err
}
return out, nil
}
func (c *seaweedFilerClient) AtomicRenameEntry(ctx context.Context, in *AtomicRenameEntryRequest, opts ...grpc.CallOption) (*AtomicRenameEntryResponse, error) {
cOpts := append([]grpc.CallOption{grpc.StaticMethod()}, opts...)
out := new(AtomicRenameEntryResponse)
@@ -493,9 +457,6 @@ type SeaweedFilerServer interface {
TouchAccessTime(context.Context, *TouchAccessTimeRequest) (*TouchAccessTimeResponse, error)
AppendToEntry(context.Context, *AppendToEntryRequest) (*AppendToEntryResponse, error)
DeleteEntry(context.Context, *DeleteEntryRequest) (*DeleteEntryResponse, error)
ObjectTransaction(context.Context, *ObjectTransactionRequest) (*ObjectTransactionResponse, error)
ObjectTransactionBatch(context.Context, *ObjectTransactionBatchRequest) (*ObjectTransactionBatchResponse, error)
PosixLock(context.Context, *PosixLockRequest) (*PosixLockResponse, error)
AtomicRenameEntry(context.Context, *AtomicRenameEntryRequest) (*AtomicRenameEntryResponse, error)
StreamRenameEntry(*StreamRenameEntryRequest, grpc.ServerStreamingServer[StreamRenameEntryResponse]) error
StreamMutateEntry(grpc.BidiStreamingServer[StreamMutateEntryRequest, StreamMutateEntryResponse]) error
@@ -553,15 +514,6 @@ func (UnimplementedSeaweedFilerServer) AppendToEntry(context.Context, *AppendToE
func (UnimplementedSeaweedFilerServer) DeleteEntry(context.Context, *DeleteEntryRequest) (*DeleteEntryResponse, error) {
return nil, status.Error(codes.Unimplemented, "method DeleteEntry not implemented")
}
func (UnimplementedSeaweedFilerServer) ObjectTransaction(context.Context, *ObjectTransactionRequest) (*ObjectTransactionResponse, error) {
return nil, status.Error(codes.Unimplemented, "method ObjectTransaction not implemented")
}
func (UnimplementedSeaweedFilerServer) ObjectTransactionBatch(context.Context, *ObjectTransactionBatchRequest) (*ObjectTransactionBatchResponse, error) {
return nil, status.Error(codes.Unimplemented, "method ObjectTransactionBatch not implemented")
}
func (UnimplementedSeaweedFilerServer) PosixLock(context.Context, *PosixLockRequest) (*PosixLockResponse, error) {
return nil, status.Error(codes.Unimplemented, "method PosixLock not implemented")
}
func (UnimplementedSeaweedFilerServer) AtomicRenameEntry(context.Context, *AtomicRenameEntryRequest) (*AtomicRenameEntryResponse, error) {
return nil, status.Error(codes.Unimplemented, "method AtomicRenameEntry not implemented")
}
@@ -771,60 +723,6 @@ func _SeaweedFiler_DeleteEntry_Handler(srv interface{}, ctx context.Context, dec
return interceptor(ctx, in, info, handler)
}
func _SeaweedFiler_ObjectTransaction_Handler(srv interface{}, ctx context.Context, dec func(interface{}) error, interceptor grpc.UnaryServerInterceptor) (interface{}, error) {
in := new(ObjectTransactionRequest)
if err := dec(in); err != nil {
return nil, err
}
if interceptor == nil {
return srv.(SeaweedFilerServer).ObjectTransaction(ctx, in)
}
info := &grpc.UnaryServerInfo{
Server: srv,
FullMethod: SeaweedFiler_ObjectTransaction_FullMethodName,
}
handler := func(ctx context.Context, req interface{}) (interface{}, error) {
return srv.(SeaweedFilerServer).ObjectTransaction(ctx, req.(*ObjectTransactionRequest))
}
return interceptor(ctx, in, info, handler)
}
func _SeaweedFiler_ObjectTransactionBatch_Handler(srv interface{}, ctx context.Context, dec func(interface{}) error, interceptor grpc.UnaryServerInterceptor) (interface{}, error) {
in := new(ObjectTransactionBatchRequest)
if err := dec(in); err != nil {
return nil, err
}
if interceptor == nil {
return srv.(SeaweedFilerServer).ObjectTransactionBatch(ctx, in)
}
info := &grpc.UnaryServerInfo{
Server: srv,
FullMethod: SeaweedFiler_ObjectTransactionBatch_FullMethodName,
}
handler := func(ctx context.Context, req interface{}) (interface{}, error) {
return srv.(SeaweedFilerServer).ObjectTransactionBatch(ctx, req.(*ObjectTransactionBatchRequest))
}
return interceptor(ctx, in, info, handler)
}
func _SeaweedFiler_PosixLock_Handler(srv interface{}, ctx context.Context, dec func(interface{}) error, interceptor grpc.UnaryServerInterceptor) (interface{}, error) {
in := new(PosixLockRequest)
if err := dec(in); err != nil {
return nil, err
}
if interceptor == nil {
return srv.(SeaweedFilerServer).PosixLock(ctx, in)
}
info := &grpc.UnaryServerInfo{
Server: srv,
FullMethod: SeaweedFiler_PosixLock_FullMethodName,
}
handler := func(ctx context.Context, req interface{}) (interface{}, error) {
return srv.(SeaweedFilerServer).PosixLock(ctx, req.(*PosixLockRequest))
}
return interceptor(ctx, in, info, handler)
}
func _SeaweedFiler_AtomicRenameEntry_Handler(srv interface{}, ctx context.Context, dec func(interface{}) error, interceptor grpc.UnaryServerInterceptor) (interface{}, error) {
in := new(AtomicRenameEntryRequest)
if err := dec(in); err != nil {
@@ -1231,18 +1129,6 @@ var SeaweedFiler_ServiceDesc = grpc.ServiceDesc{
MethodName: "DeleteEntry",
Handler: _SeaweedFiler_DeleteEntry_Handler,
},
{
MethodName: "ObjectTransaction",
Handler: _SeaweedFiler_ObjectTransaction_Handler,
},
{
MethodName: "ObjectTransactionBatch",
Handler: _SeaweedFiler_ObjectTransactionBatch_Handler,
},
{
MethodName: "PosixLock",
Handler: _SeaweedFiler_PosixLock_Handler,
},
{
MethodName: "AtomicRenameEntry",
Handler: _SeaweedFiler_AtomicRenameEntry_Handler,
-1
View File
@@ -374,7 +374,6 @@ message ErasureCodingTaskConfig {
int32 min_volume_size_mb = 3; // Minimum volume size for EC
string collection_filter = 4; // Only process volumes from specific collections
repeated string preferred_tags = 5; // Disk tags to prioritize for EC shard placement
string replica_placement = 6; // EC shard replica placement (e.g. "020"); empty falls back to master default replication
}
// BalanceTaskConfig contains balance-specific configuration
+2 -11
View File
@@ -2960,7 +2960,6 @@ type ErasureCodingTaskConfig struct {
MinVolumeSizeMb int32 `protobuf:"varint,3,opt,name=min_volume_size_mb,json=minVolumeSizeMb,proto3" json:"min_volume_size_mb,omitempty"` // Minimum volume size for EC
CollectionFilter string `protobuf:"bytes,4,opt,name=collection_filter,json=collectionFilter,proto3" json:"collection_filter,omitempty"` // Only process volumes from specific collections
PreferredTags []string `protobuf:"bytes,5,rep,name=preferred_tags,json=preferredTags,proto3" json:"preferred_tags,omitempty"` // Disk tags to prioritize for EC shard placement
ReplicaPlacement string `protobuf:"bytes,6,opt,name=replica_placement,json=replicaPlacement,proto3" json:"replica_placement,omitempty"` // EC shard replica placement (e.g. "020"); empty falls back to master default replication
unknownFields protoimpl.UnknownFields
sizeCache protoimpl.SizeCache
}
@@ -3030,13 +3029,6 @@ func (x *ErasureCodingTaskConfig) GetPreferredTags() []string {
return nil
}
func (x *ErasureCodingTaskConfig) GetReplicaPlacement() string {
if x != nil {
return x.ReplicaPlacement
}
return ""
}
// BalanceTaskConfig contains balance-specific configuration
type BalanceTaskConfig struct {
state protoimpl.MessageState `protogen:"open.v1"`
@@ -4226,14 +4218,13 @@ const file_worker_proto_rawDesc = "" +
"\x10VacuumTaskConfig\x12+\n" +
"\x11garbage_threshold\x18\x01 \x01(\x01R\x10garbageThreshold\x12/\n" +
"\x14min_volume_age_hours\x18\x02 \x01(\x05R\x11minVolumeAgeHours\x120\n" +
"\x14min_interval_seconds\x18\x03 \x01(\x05R\x12minIntervalSeconds\"\x9a\x02\n" +
"\x14min_interval_seconds\x18\x03 \x01(\x05R\x12minIntervalSeconds\"\xed\x01\n" +
"\x17ErasureCodingTaskConfig\x12%\n" +
"\x0efullness_ratio\x18\x01 \x01(\x01R\rfullnessRatio\x12*\n" +
"\x11quiet_for_seconds\x18\x02 \x01(\x05R\x0fquietForSeconds\x12+\n" +
"\x12min_volume_size_mb\x18\x03 \x01(\x05R\x0fminVolumeSizeMb\x12+\n" +
"\x11collection_filter\x18\x04 \x01(\tR\x10collectionFilter\x12%\n" +
"\x0epreferred_tags\x18\x05 \x03(\tR\rpreferredTags\x12+\n" +
"\x11replica_placement\x18\x06 \x01(\tR\x10replicaPlacement\"n\n" +
"\x0epreferred_tags\x18\x05 \x03(\tR\rpreferredTags\"n\n" +
"\x11BalanceTaskConfig\x12/\n" +
"\x13imbalance_threshold\x18\x01 \x01(\x01R\x12imbalanceThreshold\x12(\n" +
"\x10min_server_count\x18\x02 \x01(\x05R\x0eminServerCount\"I\n" +
@@ -299,9 +299,6 @@ func (s *s3RemoteStorageClient) WriteFile(loc *remote_pb.RemoteStorageLocation,
Body: reader,
Tagging: awsTags,
}
if entry.Attributes != nil && entry.Attributes.Mime != "" {
uploadInput.ContentType = aws.String(entry.Attributes.Mime)
}
if s.conf.S3StorageClass != "" {
uploadInput.StorageClass = aws.String(s.conf.S3StorageClass)
}
@@ -1,15 +1,10 @@
package s3
import (
"bytes"
"io"
"net/http"
"strings"
"testing"
"github.com/aws/aws-sdk-go/aws/credentials"
awss3 "github.com/aws/aws-sdk-go/service/s3"
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
"github.com/seaweedfs/seaweedfs/weed/pb/remote_pb"
"github.com/seaweedfs/seaweedfs/weed/remote_storage"
"github.com/stretchr/testify/require"
@@ -70,89 +65,3 @@ func TestS3ErrRemoteObjectNotFoundIsAccessible(t *testing.T) {
require.Error(t, remote_storage.ErrRemoteObjectNotFound)
require.Equal(t, "remote object not found", remote_storage.ErrRemoteObjectNotFound.Error())
}
// captureRoundTripper records the PUT request that the s3manager uploader
// sends, and short-circuits all calls with a 200 so the SDK is satisfied.
type captureRoundTripper struct {
uploadReq *http.Request
}
func (c *captureRoundTripper) RoundTrip(req *http.Request) (*http.Response, error) {
if req.Method == http.MethodPut {
c.uploadReq = req.Clone(req.Context())
}
if req.Body != nil {
_, _ = io.Copy(io.Discard, req.Body)
_ = req.Body.Close()
}
return &http.Response{
StatusCode: http.StatusOK,
Body: io.NopCloser(strings.NewReader("")),
Header: http.Header{
"ETag": []string{"\"etag\""},
},
Request: req,
}, nil
}
func (c *captureRoundTripper) uploadContentType() string {
if c.uploadReq == nil {
return ""
}
return c.uploadReq.Header.Get("Content-Type")
}
func newCapturingS3Client(t *testing.T) (*s3RemoteStorageClient, *captureRoundTripper) {
t.Helper()
rt := &captureRoundTripper{}
conf := &remote_pb.RemoteConf{
Name: "test",
S3Region: "us-east-1",
S3Endpoint: "https://example.invalid",
S3ForcePathStyle: true,
S3AccessKey: "test-key",
S3SecretKey: "test-secret",
}
httpClient := &http.Client{Transport: rt}
rs, err := MakeWithHTTPClient(conf, httpClient)
require.NoError(t, err)
return rs.(*s3RemoteStorageClient), rt
}
func TestS3WriteFilePassesMimeAsContentType(t *testing.T) {
client, rt := newCapturingS3Client(t)
loc := &remote_pb.RemoteStorageLocation{
Name: "test",
Bucket: "bucket",
Path: "/dir/test.html",
}
entry := &filer_pb.Entry{
Attributes: &filer_pb.FuseAttributes{Mime: "text/html"},
}
_, err := client.WriteFile(loc, entry, bytes.NewReader([]byte("<html></html>")))
require.NoError(t, err)
require.NotNil(t, rt.uploadReq, "uploader should have issued a PUT")
require.Equal(t, "text/html", rt.uploadContentType(), "Content-Type should match entry.Attributes.Mime")
}
func TestS3WriteFileOmitsContentTypeWhenMimeMissing(t *testing.T) {
client, rt := newCapturingS3Client(t)
loc := &remote_pb.RemoteStorageLocation{
Name: "test",
Bucket: "bucket",
Path: "/dir/test.bin",
}
entry := &filer_pb.Entry{
Attributes: &filer_pb.FuseAttributes{},
}
_, err := client.WriteFile(loc, entry, bytes.NewReader([]byte("data")))
require.NoError(t, err)
require.NotNil(t, rt.uploadReq, "uploader should have issued a PUT")
// When entry.Attributes.Mime is empty we don't force a Content-Type so the
// remote can apply its own default rather than getting a misleading one.
require.Equal(t, "", rt.uploadContentType())
}
+1 -25
View File
@@ -120,7 +120,7 @@ func (fs *FilerSink) replicateOneChunk(sourceChunk *filer_pb.FileChunk, path str
fileId, err := fs.fetchAndWrite(sourceChunk, path, sourceMtime)
if err != nil {
return nil, fmt.Errorf("copy %s: %w", sourceChunk.GetFileIdString(), err)
return nil, fmt.Errorf("copy %s: %v", sourceChunk.GetFileIdString(), err)
}
return &filer_pb.FileChunk{
@@ -292,10 +292,6 @@ func (fs *FilerSink) fetchAndWrite(sourceChunk *filer_pb.FileChunk, path string,
fullData = data
}
if err := validateReplicatedReadSize(sourceChunk, len(fullData)); err != nil {
return err
}
transferStatus.mu.Lock()
transferStatus.BytesReceived = int64(len(fullData))
transferStatus.Status = "uploading"
@@ -339,14 +335,6 @@ func (fs *FilerSink) fetchAndWrite(sourceChunk *filer_pb.FileChunk, path string,
fileId = currentFileId
return nil
}, func(retryErr error) (shouldContinue bool) {
if errors.Is(retryErr, errChunkSizeMismatch) {
glog.V(0).Infof("permanent size mismatch replicating %s for %s: %v",
sourceChunk.GetFileIdString(), path, retryErr)
transferStatus.mu.Lock()
transferStatus.LastErr = retryErr.Error()
transferStatus.mu.Unlock()
return false
}
if fs.hasSourceNewerVersion(path, sourceMtime) {
glog.V(1).Infof("skip retrying stale source %s for %s: %v", sourceChunk.GetFileIdString(), path, retryErr)
return false
@@ -400,18 +388,6 @@ func isEofError(err error) bool {
return errors.Is(err, io.ErrUnexpectedEOF) || errors.Is(err, io.EOF)
}
// errChunkSizeMismatch is a permanent (non-retriable) replication failure.
var errChunkSizeMismatch = errors.New("chunk size mismatch")
func validateReplicatedReadSize(sourceChunk *filer_pb.FileChunk, readSize int) error {
if uint64(readSize) != sourceChunk.Size {
return fmt.Errorf("%w: read %s got %d bytes, source metadata says %d",
errChunkSizeMismatch, sourceChunk.GetFileIdString(),
readSize, sourceChunk.Size)
}
return nil
}
func (fs *FilerSink) buildUploadUrl(host, fileId string) string {
if fs.writeChunkByFiler {
return fmt.Sprintf("http://%s/?proxyChunkId=%s", fs.address, fileId)
@@ -1,27 +1,12 @@
package filersink
import (
"errors"
"net/http"
"net/http/httptest"
"os"
"strings"
"sync/atomic"
"testing"
"time"
"github.com/seaweedfs/seaweedfs/weed/operation"
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
"github.com/seaweedfs/seaweedfs/weed/replication/source"
"github.com/seaweedfs/seaweedfs/weed/util"
util_http "github.com/seaweedfs/seaweedfs/weed/util/http"
)
func TestMain(m *testing.M) {
util_http.InitGlobalHttpClient()
os.Exit(m.Run())
}
func TestTargetPathToSourcePath(t *testing.T) {
tests := []struct {
name string
@@ -92,191 +77,3 @@ func TestTargetPathToSourcePath(t *testing.T) {
})
}
}
// FilerSink must reject chunks whose received byte count disagrees with the
// source filer metadata, instead of silently writing 0-byte needles with the
// source size in the destination metadata.
func TestValidateReplicatedChunkSize(t *testing.T) {
const fid = "74,047d16a94aa581"
tests := []struct {
name string
expectedSize uint64
readSize int
wantErr bool
}{
{
name: "healthy",
expectedSize: 5171,
readSize: 5171,
wantErr: false,
},
{
name: "legitimately empty file",
expectedSize: 0,
readSize: 0,
wantErr: false,
},
{
name: "zero-byte read for non-empty source",
expectedSize: 5171,
readSize: 0,
wantErr: true,
},
{
name: "short read",
expectedSize: 5171,
readSize: 100,
wantErr: true,
},
{
name: "over-read (server returned more than metadata)",
expectedSize: 5171,
readSize: 8192,
wantErr: true,
},
}
for _, tc := range tests {
t.Run(tc.name, func(t *testing.T) {
chunk := &filer_pb.FileChunk{FileId: fid, Size: tc.expectedSize}
gotErr := validateReplicatedReadSize(chunk, tc.readSize)
if tc.wantErr {
if gotErr == nil {
t.Fatalf("expected error, got nil (read=%d expected=%d)",
tc.readSize, tc.expectedSize)
}
if !errors.Is(gotErr, errChunkSizeMismatch) {
t.Fatalf("expected errChunkSizeMismatch, got %v", gotErr)
}
if !strings.Contains(gotErr.Error(), fid) {
t.Fatalf("error %q does not mention chunk id %q", gotErr, fid)
}
return
}
if gotErr != nil {
t.Fatalf("unexpected read-size error: %v", gotErr)
}
})
}
}
// End-to-end regression :
// a source volume that responds 200 OK with Content-Length: 0
// for a chunk that filer metadata claims is 5171 bytes must be rejected
// by fetchAndWrite with a (non-retriable) size mismatch error,
// instead of being silently propagated to the destination as a 0-byte needle.
func TestFetchAndWriteRejectsZeroByteSource(t *testing.T) {
const fid = "74,047d16a94aa581"
const expectedSize uint64 = 5171
// Shorten retry backoff so a fail-fast test that briefly enters the retry
// loop doesn't pay the production 1s+ wait. Scoped to this test so any
// future test in the package keeps the production constant.
prevRetryWaitTime := util.RetryWaitTime
util.RetryWaitTime = 100 * time.Millisecond
t.Cleanup(func() { util.RetryWaitTime = prevRetryWaitTime })
var hits atomic.Int32
sourceServer := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
hits.Add(1)
w.Header().Set("Content-Type", "application/octet-stream")
w.WriteHeader(http.StatusOK)
// Intentionally write no body — mimic the buggy volume response.
}))
defer sourceServer.Close()
serverAddr := strings.TrimPrefix(sourceServer.URL, "http://")
filerSrc := &source.FilerSource{}
if err := filerSrc.DoInitialize(serverAddr, serverAddr, "/", true); err != nil {
t.Fatalf("filerSource.DoInitialize: %v", err)
}
fs := &FilerSink{
filerSource: filerSrc,
address: serverAddr,
dir: "/dst",
executor: util.NewLimitedConcurrentExecutor(1),
}
fs.SetUploader(operation.NewUploaderWithHttpClient(http.DefaultClient))
sourceChunk := &filer_pb.FileChunk{
FileId: fid,
Size: expectedSize,
}
done := make(chan struct {
fileId string
err error
}, 1)
go func() {
gotFileId, gotErr := fs.fetchAndWrite(sourceChunk, "/dst/index.bin", 0)
done <- struct {
fileId string
err error
}{gotFileId, gotErr}
}()
select {
case result := <-done:
if result.err == nil {
t.Fatalf("expected size mismatch error, got nil (fileId=%q)", result.fileId)
}
if !errors.Is(result.err, errChunkSizeMismatch) {
t.Fatalf("expected errChunkSizeMismatch, got %v", result.err)
}
if !strings.Contains(result.err.Error(), "5171") {
t.Fatalf("error %q does not mention expected size 5171", result.err)
}
if !strings.Contains(result.err.Error(), fid) {
t.Fatalf("error %q does not mention chunk id %q", result.err, fid)
}
if h := hits.Load(); h != 1 {
t.Fatalf("expected exactly 1 source hit (fail-fast), got %d", h)
}
case <-time.After(5 * time.Second):
t.Fatalf("fetchAndWrite did not return within 5s (retry loop not aborted on size mismatch); hits=%d", hits.Load())
}
}
// Lock in that the errChunkSizeMismatch sentinel survives the wrap in
// replicateOneChunk + pass-through in util.Retry, so filer_sink.go's
// errors.Is check actually fires.
func TestReplicateChunksPreservesSizeMismatchSentinel(t *testing.T) {
const fid = "74,047d16a94aa581"
const expectedSize uint64 = 5171
sourceServer := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
w.Header().Set("Content-Type", "application/octet-stream")
w.WriteHeader(http.StatusOK)
}))
defer sourceServer.Close()
serverAddr := strings.TrimPrefix(sourceServer.URL, "http://")
filerSrc := &source.FilerSource{}
if err := filerSrc.DoInitialize(serverAddr, serverAddr, "/", true); err != nil {
t.Fatalf("filerSource.DoInitialize: %v", err)
}
fs := &FilerSink{
filerSource: filerSrc,
address: serverAddr,
dir: "/dst",
executor: util.NewLimitedConcurrentExecutor(1),
}
fs.SetUploader(operation.NewUploaderWithHttpClient(http.DefaultClient))
sourceChunks := []*filer_pb.FileChunk{{FileId: fid, Size: expectedSize}}
_, err := fs.replicateChunks(nil, sourceChunks, "/dst/index.bin", 0)
if err == nil {
t.Fatal("expected error from replicateChunks, got nil")
}
if !errors.Is(err, errChunkSizeMismatch) {
t.Fatalf("error chain broken: errors.Is(err, errChunkSizeMismatch) = false; got %v", err)
}
}
+3 -13
View File
@@ -2,7 +2,6 @@ package filersink
import (
"context"
"errors"
"fmt"
"math"
"sync"
@@ -192,12 +191,7 @@ func (fs *FilerSink) CreateEntry(key string, entry *filer_pb.Entry, signatures [
replicatedChunks, err := fs.replicateChunks(context.Background(), entry.GetChunks(), key, getEntryMtime(entry))
if err != nil {
// Don't swallow size-mismatch: source bytes disagree with source
// metadata, so committing would propagate corruption silently.
if errors.Is(err, errChunkSizeMismatch) {
glog.Errorf("refuse to replicate entry with corrupt chunk %s: %v", key, err)
return err
}
// only warning here since the source chunk may have been deleted already
glog.Warningf("replicate entry chunks %s: %v", key, err)
return nil
}
@@ -265,8 +259,8 @@ func (fs *FilerSink) UpdateEntry(key string, oldEntry *filer_pb.Entry, newParent
// this usually happens when the messages are not ordered
glog.V(2).Infof("late updates %s", key)
} else {
// source-side chunks resolve via source filer; sink volume IDs may collide.
deletedChunks, newChunks, err := compareChunks(context.Background(), filer.LookupFn(fs.filerSource), oldEntry, newEntry)
// find out what changed
deletedChunks, newChunks, err := compareChunks(context.Background(), filer.LookupFn(fs), oldEntry, newEntry)
if err != nil {
return true, fmt.Errorf("replicate %s compare chunks error: %v", key, err)
}
@@ -280,10 +274,6 @@ func (fs *FilerSink) UpdateEntry(key string, oldEntry *filer_pb.Entry, newParent
// replicate the chunks that are new in the source
replicatedChunks, err := fs.replicateChunks(context.Background(), newChunks, key, getEntryMtime(newEntry))
if err != nil {
if errors.Is(err, errChunkSizeMismatch) {
glog.Errorf("refuse to replicate entry with corrupt chunk %s: %v", key, err)
return true, err
}
glog.Warningf("replicate entry chunks %s: %v", key, err)
return true, nil
}
+7 -33
View File
@@ -553,14 +553,9 @@ func (s3a *S3ApiServer) completeMultipartUpload(r *http.Request, input *s3.Compl
uploadDirectory := s3a.genUploadsFolder(*input.Bucket) + "/" + *input.UploadId
entryName, dirName := s3a.getEntryNameAndDir(input)
var completionState *multipartCompletionState
// Route the completion's writes to the object's owner filer when known, off
// the distributed lock. Idempotent replay is handled gateway-side in
// prepareMultipartCompletionState (it returns the existing result when the
// object already carries this UploadId), so the lock is not needed to dedupe
// retries. With no owner yet (no ring), keep the lock as the bootstrap path.
owner := s3a.objectWriteOwner(*input.Bucket, *input.Key)
routeKey := s3a.objectRouteKey(*input.Bucket, *input.Key)
completionBody := func() s3err.ErrorCode {
finalizeCode := s3a.withObjectWriteLock(*input.Bucket, *input.Key, func() s3err.ErrorCode {
return s3a.checkConditionalHeaders(r, *input.Bucket, *input.Key)
}, func() s3err.ErrorCode {
var prepCode s3err.ErrorCode
completionState, output, prepCode = s3a.prepareMultipartCompletionState(r, input, uploadDirectory, entryName, dirName, completedPartNumbers, completedPartMap, maxPartNo)
if prepCode != s3err.ErrNone || output != nil {
@@ -656,16 +651,7 @@ func (s3a *S3ApiServer) completeMultipartUpload(r *http.Request, input *s3.Compl
// Update the .versions directory metadata to indicate this is the latest version
// Pass entry to cache its metadata for single-scan list efficiency
// Route the pointer flip to the owner (off the lock) via
// RECOMPUTE_LATEST; the just-written version file is the newest.
if owner != "" {
if code := s3a.routedVersionedFinalize(owner, *input.Bucket, *input.Key, useInvertedFormat); code != s3err.ErrNone {
if rollbackErr := s3a.rollbackMultipartVersion(versionDir, versionFileName); rollbackErr != nil {
glog.Errorf("completeMultipartUpload: failed to rollback version %s for %s/%s after routed finalize error: %v", versionId, *input.Bucket, *input.Key, rollbackErr)
}
return code
}
} else if err := s3a.updateLatestVersionInDirectory(*input.Bucket, *input.Key, versionId, versionFileName, versionEntryForCache); err != nil {
if err := s3a.updateLatestVersionInDirectory(*input.Bucket, *input.Key, versionId, versionFileName, versionEntryForCache); err != nil {
if rollbackErr := s3a.rollbackMultipartVersion(versionDir, versionFileName); rollbackErr != nil {
glog.Errorf("completeMultipartUpload: failed to rollback version %s for %s/%s after latest pointer update error: %v", versionId, *input.Bucket, *input.Key, rollbackErr)
}
@@ -689,7 +675,7 @@ func (s3a *S3ApiServer) completeMultipartUpload(r *http.Request, input *s3.Compl
if versioningState == s3_constants.VersioningSuspended {
// For suspended versioning, add "null" version ID metadata and return "null" version ID
if err := s3a.writeMultipartObject(owner, routeKey, dirName, entryName, completionState.finalParts, func(entry *filer_pb.Entry) {
if err := s3a.mkFile(dirName, entryName, completionState.finalParts, func(entry *filer_pb.Entry) {
if entry.Extended == nil {
entry.Extended = make(map[string][]byte)
}
@@ -753,7 +739,7 @@ func (s3a *S3ApiServer) completeMultipartUpload(r *http.Request, input *s3.Compl
}
// For non-versioned buckets, create main object file
if err := s3a.writeMultipartObject(owner, routeKey, dirName, entryName, completionState.finalParts, func(entry *filer_pb.Entry) {
if err := s3a.mkFile(dirName, entryName, completionState.finalParts, func(entry *filer_pb.Entry) {
if entry.Extended == nil {
entry.Extended = make(map[string][]byte)
}
@@ -816,19 +802,7 @@ func (s3a *S3ApiServer) completeMultipartUpload(r *http.Request, input *s3.Compl
ChecksumValue: completionState.checksumValue,
}
return s3err.ErrNone
}
var finalizeCode s3err.ErrorCode
if owner != "" {
if code := s3a.checkConditionalHeaders(r, *input.Bucket, *input.Key); code != s3err.ErrNone {
finalizeCode = code
} else {
finalizeCode = completionBody()
}
} else {
finalizeCode = s3a.withObjectWriteLock(*input.Bucket, *input.Key, func() s3err.ErrorCode {
return s3a.checkConditionalHeaders(r, *input.Bucket, *input.Key)
}, completionBody)
}
})
if finalizeCode != s3err.ErrNone {
return nil, finalizeCode
}
-70
View File
@@ -1,70 +0,0 @@
package iceberg
import (
"net/http"
"strings"
"github.com/gorilla/mux"
"github.com/seaweedfs/seaweedfs/weed/s3api/s3_constants"
)
// validateRequestPath rejects Iceberg REST requests whose captured
// {prefix}/{namespace}/{table} mux vars would produce a parent-directory
// traversal when joined into a filer path. The iceberg router runs with
// SkipClean(true), so `..` survives routing; downstream path.Join calls
// (stageCreateMarkerDir, location builders, etc.) then collapse it and
// escape the table-bucket directory.
//
// {prefix} maps to a table-bucket name; {table} is a single path segment;
// {namespace} is unit-separator (0x1F) joined parts that get flattened into
// a single dotted name for the on-disk layout — each part is validated
// individually.
func validateRequestPath(next http.Handler) http.Handler {
return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
vars := mux.Vars(r)
// Use the comma-ok form so vars only checked when the matched route
// actually captures them; when captured, an empty value is itself a
// rejection because downstream path.Join would collapse it.
if prefix, ok := vars["prefix"]; ok {
if prefix == "" || !s3_constants.IsValidBucketName(prefix) {
writeError(w, http.StatusBadRequest, "BadRequest", "invalid prefix")
return
}
}
if table, ok := vars["table"]; ok {
if table == "" || !isValidNameSegment(table) {
writeError(w, http.StatusBadRequest, "BadRequest", "invalid table name")
return
}
}
if ns, ok := vars["namespace"]; ok {
if ns == "" {
writeError(w, http.StatusBadRequest, "BadRequest", "invalid namespace")
return
}
// Reject leading/trailing/consecutive unit separators so distinct
// inputs cannot collapse to the same parsed namespace via
// parseNamespace's empty-part filter.
for _, part := range strings.Split(ns, "\x1F") {
if part == "" || !isValidNameSegment(part) {
writeError(w, http.StatusBadRequest, "BadRequest", "invalid namespace")
return
}
}
}
next.ServeHTTP(w, r)
})
}
// isValidNameSegment rejects a single path-segment value (bucket prefix slot,
// table name, or one namespace part) that would be unsafe to embed in a filer
// path: `.`, `..`, embedded slash/backslash, or NUL.
func isValidNameSegment(s string) bool {
if s == "" {
return true
}
if s == "." || s == ".." {
return false
}
return !strings.ContainsAny(s, "/\\\x00")
}
-116
View File
@@ -1,116 +0,0 @@
package iceberg
import (
"net/http"
"net/http/httptest"
"testing"
"github.com/gorilla/mux"
)
func TestValidateRequestPath_RejectsTraversal(t *testing.T) {
tests := []struct {
name string
rawPath string
wantCode int
}{
{"clean namespace+table passes", "/v1/namespaces/sales/tables/orders", http.StatusOK},
{"clean prefixed passes", "/v1/wh/namespaces/sales/tables/orders", http.StatusOK},
{"clean namespace only passes", "/v1/namespaces/sales", http.StatusOK},
// SkipClean(true) means raw `..` survives routing — these are the
// realistic traversal shapes the middleware must catch.
{"dotdot as prefix var rejected", "/v1/../namespaces/sales", http.StatusBadRequest},
{"dotdot as namespace var rejected", "/v1/namespaces/..", http.StatusBadRequest},
{"dotdot as namespace var prefixed rejected", "/v1/wh/namespaces/..", http.StatusBadRequest},
{"dotdot as table var rejected", "/v1/namespaces/sales/tables/..", http.StatusBadRequest},
{"dot as table var rejected", "/v1/namespaces/sales/tables/.", http.StatusBadRequest},
// Iceberg clients send the 0x1F unit separator percent-encoded; mux
// decodes it before the middleware sees the namespace var.
{"unit-sep namespace with dotdot part rejected", "/v1/namespaces/sales%1F..%1Fevil", http.StatusBadRequest},
{"leading unit-sep namespace rejected", "/v1/namespaces/%1Fsales", http.StatusBadRequest},
{"trailing unit-sep namespace rejected", "/v1/namespaces/sales%1F", http.StatusBadRequest},
{"consecutive unit-sep namespace rejected", "/v1/namespaces/sales%1F%1Fevil", http.StatusBadRequest},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
router := mux.NewRouter().SkipClean(true)
router.Use(validateRequestPath)
handlerCalled := false
pass := func(w http.ResponseWriter, r *http.Request) {
handlerCalled = true
w.WriteHeader(http.StatusOK)
}
router.HandleFunc("/v1/namespaces/{namespace}", pass)
router.HandleFunc("/v1/namespaces/{namespace}/tables/{table}", pass)
router.HandleFunc("/v1/{prefix}/namespaces/{namespace}", pass)
router.HandleFunc("/v1/{prefix}/namespaces/{namespace}/tables/{table}", pass)
req := httptest.NewRequest(http.MethodGet, tt.rawPath, nil)
rr := httptest.NewRecorder()
router.ServeHTTP(rr, req)
if rr.Code != tt.wantCode {
t.Fatalf("path %q: got status %d, want %d (body=%q)", tt.rawPath, rr.Code, tt.wantCode, rr.Body.String())
}
if tt.wantCode == http.StatusBadRequest && handlerCalled {
t.Fatalf("path %q: inner handler reached despite rejection", tt.rawPath)
}
})
}
}
// Defense-in-depth: if a future route or middleware ever leaves one of the
// captured vars empty, the middleware must still reject the request. The
// default mux regex won't normally allow this.
func TestValidateRequestPath_RejectsEmptyCapturedVars(t *testing.T) {
tests := []struct {
name string
vars map[string]string
}{
{"empty prefix", map[string]string{"prefix": "", "namespace": "ns"}},
{"empty table", map[string]string{"namespace": "ns", "table": ""}},
{"empty namespace", map[string]string{"namespace": ""}},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
handlerCalled := false
h := validateRequestPath(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
handlerCalled = true
}))
req := mux.SetURLVars(httptest.NewRequest(http.MethodGet, "/", nil), tt.vars)
rr := httptest.NewRecorder()
h.ServeHTTP(rr, req)
if handlerCalled {
t.Fatalf("vars %v: inner handler reached despite empty capture", tt.vars)
}
})
}
}
func TestIsValidNameSegment(t *testing.T) {
tests := []struct {
name string
input string
want bool
}{
{"empty ok", "", true},
{"plain", "orders", true},
{"with dot inside", "my.table", true},
{"hidden", ".hidden", true},
{"bare dot", ".", false},
{"bare dotdot", "..", false},
{"contains slash", "foo/bar", false},
{"contains backslash", "foo\\bar", false},
{"contains nul", "foo\x00bar", false},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
if got := isValidNameSegment(tt.input); got != tt.want {
t.Errorf("isValidNameSegment(%q) = %v, want %v", tt.input, got, tt.want)
}
})
}
}
-7
View File
@@ -71,13 +71,6 @@ func (s *Server) RegisterRoutes(router *mux.Router) {
// Add middleware to log all requests/responses
router.Use(loggingMiddleware)
// Reject `..`/`.`/NUL in {prefix}/{namespace}/{table} vars before any
// handler runs. The router uses SkipClean(true), so traversal segments
// would otherwise reach path.Join in stage-marker / location builders.
// Registered after loggingMiddleware so rejected requests still get
// audit-logged.
router.Use(validateRequestPath)
// Configuration endpoint - no auth needed for config
router.HandleFunc("/v1/config", s.handleConfig).Methods(http.MethodGet)
-35
View File
@@ -177,41 +177,6 @@ func GetBucketAndObject(r *http.Request) (bucket, object string) {
return
}
// IsValidObjectKey rejects S3 object keys that — after normalization
// (backslash→slash, slash collapse) — contain a `.` or `..` path segment, or
// embed a NUL byte. Such keys are collapsed by filepath.Join inside the filer
// and would escape the bucket directory, so they must not reach the gRPC layer.
// Gorilla mux URL-decodes captured vars before this runs, so `%2e%2e` is
// already `..` here.
func IsValidObjectKey(object string) bool {
if object == "" {
return true
}
if strings.ContainsRune(object, '\x00') {
return false
}
object = strings.ReplaceAll(object, "\\", "/")
for _, seg := range strings.Split(object, "/") {
if seg == "." || seg == ".." {
return false
}
}
return true
}
// IsValidBucketName rejects bucket names captured from the URL path that are
// unsafe to use in filer path construction (`.`, `..`, contain `/` or `\`, or
// embed NUL). This is a path-safety check, not a full S3 naming-rule check.
func IsValidBucketName(bucket string) bool {
if bucket == "" {
return true
}
if bucket == "." || bucket == ".." {
return false
}
return !strings.ContainsAny(bucket, "/\\\x00")
}
// NormalizeObjectKey normalizes object keys by removing duplicate slashes and converting backslashes.
// This normalizes keys from various sources (URL path, form values, etc.) to a consistent format.
// It also converts Windows-style backslashes to forward slashes for cross-platform compatibility.
-59
View File
@@ -89,65 +89,6 @@ func TestNormalizeObjectKey(t *testing.T) {
}
}
func TestIsValidObjectKey(t *testing.T) {
tests := []struct {
name string
input string
want bool
}{
{"empty", "", true},
{"plain", "folder/file.txt", true},
{"leading slash", "/folder/file.txt", true},
{"trailing slash", "folder/", true},
{"hidden file ok", ".hidden", true},
{"dotdot in name ok", "..hidden", true},
{"double dots inside name", "foo..bar/baz", true},
{"bare dotdot", "..", false},
{"bare dot", ".", false},
{"leading dotdot segment", "../evil-bucket/test.txt", false},
{"leading dot-slash", "./evil/test.txt", false},
{"nested dotdot segment", "good/../evil/test.txt", false},
{"trailing dotdot segment", "good/..", false},
{"backslash dotdot", "..\\evil\\test.txt", false},
{"mixed-slash dotdot", "good\\..\\evil/test.txt", false},
{"dotdot after duplicate slash", "good//../evil", false},
{"nul byte", "foo\x00bar", false},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
if got := IsValidObjectKey(tt.input); got != tt.want {
t.Errorf("IsValidObjectKey(%q) = %v, want %v", tt.input, got, tt.want)
}
})
}
}
func TestIsValidBucketName(t *testing.T) {
tests := []struct {
name string
input string
want bool
}{
{"empty ok", "", true},
{"plain", "my-bucket", true},
{"name containing dots", "my.bucket.name", true},
{"bare dot", ".", false},
{"bare dotdot", "..", false},
{"with slash", "evil/bucket", false},
{"with backslash", "evil\\bucket", false},
{"with nul", "evil\x00bucket", false},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
if got := IsValidBucketName(tt.input); got != tt.want {
t.Errorf("IsValidBucketName(%q) = %v, want %v", tt.input, got, tt.want)
}
})
}
}
func TestRemoveDuplicateSlashes(t *testing.T) {
tests := []struct {
name string
+36 -70
View File
@@ -1,7 +1,6 @@
package s3api
import (
"bytes"
"context"
"encoding/json"
"errors"
@@ -16,7 +15,6 @@ import (
"github.com/seaweedfs/seaweedfs/weed/glog"
"github.com/seaweedfs/seaweedfs/weed/kms"
"github.com/seaweedfs/seaweedfs/weed/pb"
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
"github.com/seaweedfs/seaweedfs/weed/pb/s3_pb"
"github.com/seaweedfs/seaweedfs/weed/s3api/cors"
@@ -46,8 +44,8 @@ type BucketConfig struct {
// rather than this cache.
LifecycleTTL *LifecycleTTLResolver
KMSKeyCache *BucketKMSCache // Per-bucket KMS key cache for SSE-KMS operations
LastModified time.Time
Entry *filer_pb.Entry
LastModified time.Time
Entry *filer_pb.Entry
}
// BucketKMSCache represents per-bucket KMS key caching for SSE-KMS operations
@@ -526,74 +524,26 @@ func (s3a *S3ApiServer) updateBucketConfig(bucket string, updateFn func(*BucketC
bucket, s3_constants.ExtObjectLockEnabledKey, string(nextConfig.Entry.Extended[s3_constants.ExtObjectLockEnabledKey]))
}
// Patch only the changed/removed extended keys, leaving Entry.content
// untouched so a concurrent content write (e.g. encryption) is preserved.
oldExt := config.Entry.GetExtended()
newExt := nextConfig.Entry.Extended
set := make(map[string][]byte)
for k, v := range newExt {
if ov, ok := oldExt[k]; !ok || !bytes.Equal(ov, v) {
set[k] = v
}
}
var del []string
for k := range oldExt {
if _, ok := newExt[k]; !ok {
del = append(del, k)
}
}
glog.V(3).Infof("updateBucketConfig: patching %d/%d extended keys for bucket %s", len(set), len(del), bucket)
if err := s3a.patchBucketEntry(bucket, &filer_pb.ObjectMutation{SetExtended: set, DeleteExtended: del}); err != nil {
glog.Errorf("updateBucketConfig: failed to patch bucket entry for %s: %v", bucket, err)
// Save to filer
glog.V(3).Infof("updateBucketConfig: saving entry to filer for bucket %s", bucket)
err := s3a.updateEntry(s3a.bucketRoot(bucket), nextConfig.Entry)
if err != nil {
glog.Errorf("updateBucketConfig: failed to update bucket entry for %s: %v", bucket, err)
return s3err.ErrInternalError
}
glog.V(3).Infof("updateBucketConfig: saved entry to filer for bucket %s", bucket)
// Invalidate rather than cache nextConfig: its content may be stale relative
// to a concurrent content write. The next read re-fetches the merged entry.
if s3a.bucketConfigCache != nil {
s3a.bucketConfigCache.Remove(bucket)
s3a.bucketConfigCache.RemoveNegativeCache(bucket)
}
// Update cache. Re-derive every Extended-backed field from the
// just-saved Entry — the user's update fn may have flipped, added,
// or cleared bytes (e.g. PutBucketLifecycle / DeleteBucketLifecycle
// rewrites the lifecycle XML key) and the resolver / parsed configs
// must follow.
s3a.populateBucketConfigDerivedFields(nextConfig)
s3a.bucketConfigCache.Set(bucket, nextConfig)
return s3err.ErrNone
}
// patchBucketEntry applies a field-level PATCH_EXTENDED mutation to the bucket's
// entry via ObjectTransaction, routed to the bucket's owner filer so its per-path
// lock serializes concurrent config writes cluster-wide rather than racing
// whole-entry rewrites. A nil/empty mutation is a no-op.
func (s3a *S3ApiServer) patchBucketEntry(bucket string, m *filer_pb.ObjectMutation) error {
if m == nil || (len(m.SetExtended) == 0 && len(m.DeleteExtended) == 0 && !m.SetContent) {
return nil
}
dir := s3a.option.BucketsPath
bucketPath := dir + "/" + bucket
m.Type = filer_pb.ObjectMutation_PATCH_EXTENDED
m.Directory = dir
m.Name = bucket
req := &filer_pb.ObjectTransactionRequest{
LockKey: bucketPath,
RouteKey: objectWriteRouteKeyPrefix + bucketPath,
Mutations: []*filer_pb.ObjectMutation{m},
}
txn := func(client filer_pb.SeaweedFilerClient) error {
resp, err := client.ObjectTransaction(context.Background(), req)
if err != nil {
return err
}
if resp.Error != "" {
return fmt.Errorf("patch bucket %s: %s", bucket, resp.Error)
}
return nil
}
if s3a.objectWriteLockClient != nil {
if owner := s3a.objectWriteLockClient.PrimaryForKey(objectWriteRouteKeyPrefix + bucketPath); owner != "" {
return pb.WithFilerClient(false, 0, owner, s3a.option.GrpcDialOption, txn)
}
}
return s3a.WithFilerClient(false, txn)
}
func cloneBucketConfig(config *BucketConfig) *BucketConfig {
if config == nil {
return nil
@@ -1120,11 +1070,27 @@ func (s3a *S3ApiServer) setBucketMetadata(bucket string, metadata *BucketMetadat
return fmt.Errorf("failed to marshal bucket metadata to protobuf: %w", err)
}
// Patch only Entry.content so a concurrent extended-attribute write
// (e.g. versioning) is preserved.
err = s3a.patchBucketEntry(bucket, &filer_pb.ObjectMutation{
SetContent: true,
Content: metadataBytes,
// Update the bucket entry with new content
err = s3a.WithFilerClient(false, func(client filer_pb.SeaweedFilerClient) error {
// Get current bucket entry
entry, err := s3a.getBucketEntry(bucket)
if err != nil {
return fmt.Errorf("error retrieving bucket directory %s: %w", bucket, err)
}
if entry == nil {
return fmt.Errorf("bucket directory not found %s", bucket)
}
// Update content with metadata
entry.Content = metadataBytes
request := &filer_pb.UpdateEntryRequest{
Directory: s3a.bucketRoot(bucket),
Entry: entry,
}
_, err = client.UpdateEntry(context.Background(), request)
return err
})
// Invalidate cache after successful update
+3 -5
View File
@@ -324,7 +324,7 @@ func removeDuplicateSlashes(object string) string {
// hasChildren("bucket", "empty-dir") where no children exist → false
//
// Performance: ~1-5ms per call (one gRPC LIST request with Limit=1)
func (s3a *S3ApiServer) hasChildren(ctx context.Context, bucket, prefix string) bool {
func (s3a *S3ApiServer) hasChildren(bucket, prefix string) bool {
// Clean up prefix: remove leading slashes
cleanPrefix := strings.TrimPrefix(prefix, "/")
@@ -334,10 +334,8 @@ func (s3a *S3ApiServer) hasChildren(ctx context.Context, bucket, prefix string)
// List one child object. filer_pb.List cancels the underlying ListEntries
// stream when it returns, so gRPC's per-stream client goroutine is not leaked.
// The caller's request context is propagated so the probe is cancelled if the
// client disconnects.
found := false
err := filer_pb.List(ctx, s3a, fullPath, "", func(*filer_pb.Entry, bool) error {
err := filer_pb.List(context.Background(), s3a, fullPath, "", func(*filer_pb.Entry, bool) error {
found = true
return nil
}, "", true, 1)
@@ -2322,7 +2320,7 @@ func (s3a *S3ApiServer) HeadObjectHandler(w http.ResponseWriter, r *http.Request
}
if isZeroByteFile {
// Check if it has children (making it an implicit directory)
if s3a.hasChildren(r.Context(), bucket, object) {
if s3a.hasChildren(bucket, object) {
// This is an implicit directory with children
// Return 404 to force clients (like s3fs) to use LIST-based discovery
s3err.WriteErrorResponse(w, r, s3err.ErrNoSuchKey)
+15 -42
View File
@@ -158,12 +158,9 @@ func (s3a *S3ApiServer) CopyObjectHandler(w http.ResponseWriter, r *http.Request
if sameDestination && (replaceMeta || replaceTagging) && s3a.canUseMetadataOnlySelfCopy(entry, r, dstBucket, dstObject) {
var dstVersionId string
var etag string
// A non-versioned in-place metadata replace routes to the owner as a
// serialized PATCH (off the distributed lock); versioned/suspended (which
// create a new version) and the no-owner bootstrap keep the lock.
owner := s3a.objectWriteOwner(dstBucket, dstObject)
routeInPlace := owner != "" && dstVersioningState == ""
selfCopyBody := func() s3err.ErrorCode {
updateCode := s3a.withObjectWriteLock(dstBucket, dstObject, func() s3err.ErrorCode {
return s3a.checkConditionalHeaders(r, dstBucket, dstObject)
}, func() s3err.ErrorCode {
currentEntry, currentErr := s3a.resolveCopySourceEntry(srcBucket, srcObject, srcVersionId, srcVersioningState)
if currentErr != nil || currentEntry.IsDirectory {
return s3err.ErrInvalidCopySource
@@ -171,41 +168,26 @@ func (s3a *S3ApiServer) CopyObjectHandler(w http.ResponseWriter, r *http.Request
if errCode := s3a.validateConditionalCopyHeaders(r, currentEntry); errCode != s3err.ErrNone {
return errCode
}
updatedMetadata, metadataErr := processMetadataBytes(r.Header, currentEntry.Extended, replaceMeta, replaceTagging)
if metadataErr != nil {
glog.Errorf("CopyObjectHandler ValidateTags error %s: %v", r.URL, metadataErr)
updatedEntry := cloneProtoEntry(currentEntry)
updatedMetadata, metadataErr := processMetadataBytes(r.Header, updatedEntry.Extended, replaceMeta, replaceTagging)
currentErr = metadataErr
if currentErr != nil {
glog.Errorf("CopyObjectHandler ValidateTags error %s: %v", r.URL, currentErr)
return s3err.ErrInvalidTag
}
if routeInPlace {
if err := s3a.routedMetadataReplace(owner, dstBucket, dstObject, currentEntry, updatedMetadata); err != nil {
return filerErrorToS3Error(err)
}
etag = getEtagFromEntry(currentEntry)
return s3err.ErrNone
}
updatedEntry := cloneProtoEntry(currentEntry)
updatedEntry.Extended = mergeCopyMetadata(updatedEntry.Extended, updatedMetadata)
if updatedEntry.Attributes == nil {
updatedEntry.Attributes = &filer_pb.FuseAttributes{}
}
updatedEntry.Attributes.Mtime = t.Unix()
var finErr error
dstVersionId, etag, finErr = s3a.finalizeCopyDestination(dstBucket, dstObject, dstVersioningState, updatedEntry)
if finErr != nil {
return filerErrorToS3Error(finErr)
dstVersionId, etag, currentErr = s3a.finalizeCopyDestination(dstBucket, dstObject, dstVersioningState, updatedEntry)
if currentErr != nil {
return filerErrorToS3Error(currentErr)
}
return s3err.ErrNone
}
var updateCode s3err.ErrorCode
if routeInPlace {
if updateCode = s3a.checkConditionalHeaders(r, dstBucket, dstObject); updateCode == s3err.ErrNone {
updateCode = selfCopyBody()
}
} else {
updateCode = s3a.withObjectWriteLock(dstBucket, dstObject, func() s3err.ErrorCode {
return s3a.checkConditionalHeaders(r, dstBucket, dstObject)
}, selfCopyBody)
}
})
if updateCode != s3err.ErrNone {
s3err.WriteErrorResponse(w, r, updateCode)
return
@@ -462,16 +444,7 @@ func (s3a *S3ApiServer) finalizeCopyDestination(dstBucket, dstObject, dstVersion
return "", "", err
}
// Route the pointer flip to the owner filer when known (off the
// distributed lock); RECOMPUTE_LATEST picks the just-written version.
if owner := s3a.objectWriteOwner(dstBucket, normalizedObject); owner != "" {
if code := s3a.routedVersionedFinalize(owner, dstBucket, normalizedObject, isNewFormatVersionId(versionId)); code != s3err.ErrNone {
if rollbackErr := s3a.rollbackCopyVersion(bucketDir, versionObjectPath); rollbackErr != nil {
glog.Errorf("CopyObjectHandler: failed to rollback version %s for %s/%s after routed finalize error: %v", versionId, dstBucket, normalizedObject, rollbackErr)
}
return "", "", fmt.Errorf("routed finalize for %s/%s: code %d", dstBucket, normalizedObject, code)
}
} else if err = s3a.updateLatestVersionInDirectory(dstBucket, normalizedObject, versionId, versionFileName, dstEntry); err != nil {
if err = s3a.updateLatestVersionInDirectory(dstBucket, normalizedObject, versionId, versionFileName, dstEntry); err != nil {
if rollbackErr := s3a.rollbackCopyVersion(bucketDir, versionObjectPath); rollbackErr != nil {
glog.Errorf("CopyObjectHandler: failed to rollback version %s for %s/%s after latest pointer update error: %v", versionId, dstBucket, normalizedObject, rollbackErr)
}
@@ -397,7 +397,7 @@ func (s3a *S3ApiServer) copyObjectPartViaReencryption(
filePath := s3a.genPartUploadPath(dstBucket, uploadID, partID)
// Copy-part is an MPU part write under .uploads/<id>/<n>; lifecycle
// TTL only applies to the eventual completed object. Pass 0.
tag, code, putSSE := s3a.putToFiler(cloned, filePath, srcReader, dstBucket, "", partID, 0, nil, false)
tag, code, putSSE := s3a.putToFiler(cloned, filePath, srcReader, dstBucket, "", partID, 0, nil)
if code != s3err.ErrNone {
return "", SSEResponseMetadata{}, code
}
+25 -88
View File
@@ -214,96 +214,33 @@ func (s3a *S3ApiServer) DeleteObjectHandler(w http.ResponseWriter, r *http.Reque
}
var deleteResult deleteMutationResult
var deleteCode s3err.ErrorCode
// Fast path: route the delete to the owner filer under its per-path lock;
// routedObjectOwner excludes versioned/object-lock buckets.
deleteHandled := false
if !versioningConfigured {
if cond, condOk := buildDeleteCondition(r); condOk {
if owner, ownerOk := s3a.routedObjectOwner(bucket, object); ownerOk {
resp, err := s3a.routedDelete(owner, bucket, object, cond)
switch {
case err != nil:
glog.Warningf("DeleteObjectHandler: routed delete to %s failed for %s/%s, falling back to lock: %v", owner, bucket, object, err)
case resp.ErrorCode == filer_pb.FilerError_PRECONDITION_FAILED:
deleteCode, deleteHandled = s3err.ErrPreconditionFailed, true
case resp.Error != "":
// Non-precondition error: fall back (the lock path handles cases
// the raw delete cannot, e.g. a non-empty directory marker).
glog.Warningf("DeleteObjectHandler: routed delete to %s returned %q for %s/%s, falling back to lock", owner, resp.Error, bucket, object)
default:
deleteCode, deleteHandled = s3err.ErrNone, true
}
deleteCode := s3a.withObjectWriteLock(bucket, object, func() s3err.ErrorCode {
return s3a.checkDeleteIfMatch(bucket, object, versionId, versioningState, r.Header.Get(s3_constants.IfMatch), s3err.ErrPreconditionFailed)
}, func() s3err.ErrorCode {
if versioningConfigured {
result, errCode := s3a.deleteVersionedObject(r, bucket, object, versionId, versioningState)
if errCode != s3err.ErrNone {
return errCode
}
}
}
// Versioned/suspended delete with no specific version: route off the lock when
// the bucket has an owner. createDeleteMarker routes its own pointer flip; a
// delete marker never removes a locked version, so object-lock buckets route
// here too. The If-Match precondition was already checked above.
if !deleteHandled && versioningConfigured && versionId == "" {
if owner := s3a.routableWriteOwner(bucket, object); owner != "" {
deleteResult, deleteCode = s3a.deleteVersionedObject(r, bucket, object, versionId, versioningState)
deleteHandled = true
}
}
// Specific-version delete: route off the lock. A real version recomputes the
// .versions pointer excluding it and deletes the version file; the null
// version is the regular object entry, deleted directly. Object-lock buckets
// gate the delete on the version's WORM guards, evaluated on the owner — for
// governance bypass the retention guard is scoped to COMPLIANCE so the filer
// allows a governance-mode delete while still denying compliance and legal
// hold, without the gateway reading the version.
if !deleteHandled && versionId != "" {
worm, lockErr := s3a.isObjectLockEnabled(bucket)
bypass := worm && s3a.evaluateGovernanceBypassRequest(r, bucket, object)
if lockErr == nil {
if owner := s3a.objectWriteOwner(bucket, object); owner != "" {
deleteResult.versionId = versionId
if ve, vErr := s3a.getSpecificObjectVersion(bucket, object, versionId); vErr == nil && ve != nil && ve.Extended != nil {
if dm, ok := ve.Extended[s3_constants.ExtDeleteMarkerKey]; ok && string(dm) == "true" {
deleteResult.deleteMarker = true
}
}
if versionId == "null" {
deleteCode = s3a.routedDeleteNullVersion(owner, bucket, object, worm, bypass)
} else {
deleteCode = s3a.routedDeleteSpecificVersion(owner, bucket, object, versionId, worm, bypass)
}
deleteHandled = true
}
}
}
if !deleteHandled {
deleteCode = s3a.withObjectWriteLock(bucket, object, func() s3err.ErrorCode {
return s3a.checkDeleteIfMatch(bucket, object, versionId, versioningState, r.Header.Get(s3_constants.IfMatch), s3err.ErrPreconditionFailed)
}, func() s3err.ErrorCode {
if versioningConfigured {
result, errCode := s3a.deleteVersionedObject(r, bucket, object, versionId, versioningState)
if errCode != s3err.ErrNone {
return errCode
}
deleteResult = result
return s3err.ErrNone
}
governanceBypassAllowed := s3a.evaluateGovernanceBypassRequest(r, bucket, object)
if err := s3a.enforceObjectLockProtections(r, bucket, object, "", governanceBypassAllowed); err != nil {
glog.V(2).Infof("DeleteObjectHandler: object lock check failed for %s/%s: %v", bucket, object, err)
return s3err.ErrAccessDenied
}
if err := s3a.WithFilerClient(false, func(client filer_pb.SeaweedFilerClient) error {
return s3a.deleteUnversionedObjectWithClient(client, bucket, object, false)
}); err != nil {
glog.Errorf("DeleteObjectHandler: failed to delete %s/%s: %v", bucket, object, err)
return s3err.ErrInternalError
}
deleteResult = result
return s3err.ErrNone
})
}
}
governanceBypassAllowed := s3a.evaluateGovernanceBypassRequest(r, bucket, object)
if err := s3a.enforceObjectLockProtections(r, bucket, object, "", governanceBypassAllowed); err != nil {
glog.V(2).Infof("DeleteObjectHandler: object lock check failed for %s/%s: %v", bucket, object, err)
return s3err.ErrAccessDenied
}
if err := s3a.WithFilerClient(false, func(client filer_pb.SeaweedFilerClient) error {
return s3a.deleteUnversionedObjectWithClient(client, bucket, object, false)
}); err != nil {
glog.Errorf("DeleteObjectHandler: failed to delete %s/%s: %v", bucket, object, err)
return s3err.ErrInternalError
}
return s3err.ErrNone
})
if deleteCode != s3err.ErrNone {
s3err.WriteErrorResponse(w, r, deleteCode)
return
+4 -4
View File
@@ -109,7 +109,7 @@ func (s3a *S3ApiServer) ListObjectsV2Handler(w http.ResponseWriter, r *http.Requ
// Adjust marker if it ends with delimiter to skip all entries with that prefix
marker = adjustMarkerForDelimiter(marker, delimiter)
response, err := s3a.listFilerEntries(r.Context(), bucket, originalPrefix, maxKeys, marker, delimiter, encodingTypeUrl, fetchOwner)
response, err := s3a.listFilerEntries(bucket, originalPrefix, maxKeys, marker, delimiter, encodingTypeUrl, fetchOwner)
if err != nil {
s3err.WriteErrorResponse(w, r, s3err.ErrInternalError)
@@ -173,7 +173,7 @@ func (s3a *S3ApiServer) ListObjectsV1Handler(w http.ResponseWriter, r *http.Requ
// Adjust marker if it ends with delimiter to skip all entries with that prefix
marker = adjustMarkerForDelimiter(marker, delimiter)
response, err := s3a.listFilerEntries(r.Context(), bucket, originalPrefix, uint16(maxKeys), marker, delimiter, encodingTypeUrl, true)
response, err := s3a.listFilerEntries(bucket, originalPrefix, uint16(maxKeys), marker, delimiter, encodingTypeUrl, true)
if err != nil {
s3err.WriteErrorResponse(w, r, s3err.ErrInternalError)
@@ -232,7 +232,7 @@ func sanitizeV1MarkerEcho(response *ListBucketResult, marker string, encodingTyp
}
}
func (s3a *S3ApiServer) listFilerEntries(ctx context.Context, bucket string, originalPrefix string, maxKeys uint16, originalMarker string, delimiter string, encodingTypeUrl bool, fetchOwner bool) (response ListBucketResult, err error) {
func (s3a *S3ApiServer) listFilerEntries(bucket string, originalPrefix string, maxKeys uint16, originalMarker string, delimiter string, encodingTypeUrl bool, fetchOwner bool) (response ListBucketResult, err error) {
// convert full path prefix into directory name and prefix for entry name
requestDir, prefix, marker := normalizePrefixMarker(originalPrefix, originalMarker)
bucketPrefix := s3a.bucketPrefix(bucket)
@@ -307,7 +307,7 @@ func (s3a *S3ApiServer) listFilerEntries(ctx context.Context, bucket string, ori
if normalizedPrefix != "" {
relativePath := strings.TrimPrefix(fmt.Sprintf("%s/%s", dir, entry.Name), bucketPrefix)
relativePath = strings.TrimPrefix(relativePath, "/")
if normalizedPrefix == relativePath && !s3a.hasChildren(ctx, bucket, relativePath) && !entry.IsDirectoryKeyObject() {
if normalizedPrefix == relativePath && !s3a.hasChildren(bucket, relativePath) && !entry.IsDirectoryKeyObject() {
return
}
}
@@ -457,7 +457,7 @@ func (s3a *S3ApiServer) PutObjectPartHandler(w http.ResponseWriter, r *http.Requ
// transient .uploads/<id>/<n> path, and a part write would otherwise
// start the TTL clock before CompleteMultipartUpload ever assembled
// the object.
etag, errCode, sseMetadata := s3a.putToFiler(r, filePath, dataReader, bucket, "", partID, 0, nil, false)
etag, errCode, sseMetadata := s3a.putToFiler(r, filePath, dataReader, bucket, "", partID, 0, nil)
if errCode != s3err.ErrNone {
glog.Errorf("PutObjectPart: putToFiler failed with error code %v for bucket=%s, object=%s, partNumber=%d",
errCode, bucket, object, partID)
@@ -135,7 +135,7 @@ func (s3a *S3ApiServer) PostPolicyBucketHandler(w http.ResponseWriter, r *http.R
// fields and boundaries inflates ContentLength relative to the
// object body, which would mis-evaluate any size-filtered rule.
ttlSec := s3a.lifecycleTTLForObjectWrite(bucket, object, fileSize)
etag, errCode, sseMetadata := s3a.putToFiler(r, filePath, fileBody, bucket, object, 1, ttlSec, nil, false)
etag, errCode, sseMetadata := s3a.putToFiler(r, filePath, fileBody, bucket, object, 1, ttlSec, nil)
if errCode != s3err.ErrNone {
s3err.WriteErrorResponse(w, r, errCode)
+12 -36
View File
@@ -297,7 +297,7 @@ func (s3a *S3ApiServer) PutObjectHandler(w http.ResponseWriter, r *http.Request)
}
ttlSec := s3a.lifecycleTTLForObjectWrite(bucket, object, r.ContentLength)
etag, errCode, sseMetadata := s3a.putToFiler(r, filePath, dataReader, bucket, object, 1, ttlSec, nil, false)
etag, errCode, sseMetadata := s3a.putToFiler(r, filePath, dataReader, bucket, object, 1, ttlSec, nil)
if errCode != s3err.ErrNone {
s3err.WriteErrorResponse(w, r, errCode)
@@ -359,7 +359,7 @@ func (s3a *S3ApiServer) withObjectWriteLock(bucket, object string, preconditionF
// pass 0 because their own keys aren't the user-visible object the rule
// targets and a part write would otherwise bind a TTL clock starting
// before CompleteMultipartUpload.
func (s3a *S3ApiServer) putToFiler(r *http.Request, filePath string, dataReader io.Reader, bucket string, object string, partNumber int, lifecycleTTLSec int32, afterCreate func(entry *filer_pb.Entry) s3err.ErrorCode, uniqueWritePath bool) (etag string, code s3err.ErrorCode, sseMetadata SSEResponseMetadata) {
func (s3a *S3ApiServer) putToFiler(r *http.Request, filePath string, dataReader io.Reader, bucket string, object string, partNumber int, lifecycleTTLSec int32, afterCreate func(entry *filer_pb.Entry) s3err.ErrorCode) (etag string, code s3err.ErrorCode, sseMetadata SSEResponseMetadata) {
// NEW OPTIMIZATION: Write directly to volume servers, bypassing filer proxy
// This eliminates the filer proxy overhead for PUT operations
// Note: filePath is now passed directly instead of URL (no parsing needed)
@@ -802,7 +802,7 @@ func (s3a *S3ApiServer) putToFiler(r *http.Request, filePath string, dataReader
}
return s3a.checkConditionalHeaders(r, bucket, object)
}
createUnderLock := func() s3err.ErrorCode {
createCode := s3a.withObjectWriteLock(bucket, object, preconditionFn, func() s3err.ErrorCode {
createErr = s3a.WithFilerClient(false, func(client filer_pb.SeaweedFilerClient) error {
req := &filer_pb.CreateEntryRequest{
Directory: path.Dir(filePath),
@@ -831,36 +831,7 @@ func (s3a *S3ApiServer) putToFiler(r *http.Request, filePath string, dataReader
}
}
return s3err.ErrNone
}
// Route the create to the object's owner filer, whose per-path lock
// serializes it, then run afterCreate (e.g. a versioned finalize that routes
// itself). Conditional/object-lock/non-reducible cases fall back to the
// distributed lock.
var createCode s3err.ErrorCode
routed := false
if owner := s3a.routableWriteOwner(bucket, object); owner != "" {
if cond, ok := routeWriteCondition(r, uniqueWritePath); ok {
resp, err := s3a.routedPut(owner, s3a.objectRouteKey(bucket, object), filePath, entry, cond)
switch {
case err != nil:
glog.Warningf("putToFiler: routed PUT to %s failed for %s, falling back to lock: %v", owner, filePath, err)
case resp.ErrorCode == filer_pb.FilerError_PRECONDITION_FAILED:
createCode, routed = s3err.ErrPreconditionFailed, true
case resp.Error != "":
// Non-precondition mutation error: fall back so the lock path maps it.
glog.Warningf("putToFiler: routed PUT to %s returned %q for %s, falling back to lock", owner, resp.Error, filePath)
default:
entryCreated, routed, createCode = true, true, s3err.ErrNone
if afterCreate != nil {
createCode = afterCreate(entry)
}
}
}
}
if !routed {
createCode = s3a.withObjectWriteLock(bucket, object, preconditionFn, createUnderLock)
}
})
if createCode != s3err.ErrNone {
if createErr != nil {
glog.Errorf("putToFiler: failed to create entry for %s: %v", filePath, createErr)
@@ -1316,7 +1287,7 @@ func (s3a *S3ApiServer) putSuspendedVersioningObject(r *http.Request, bucket, ob
glog.Warningf("putSuspendedVersioningObject: failed to update IsLatest flags: %v", err)
}
return s3err.ErrNone
}, false)
})
if errCode != s3err.ErrNone {
glog.Errorf("putSuspendedVersioningObject: failed to upload object: %v", errCode)
return "", errCode, SSEResponseMetadata{}
@@ -1482,8 +1453,13 @@ func (s3a *S3ApiServer) putVersionedObject(r *http.Request, bucket, object strin
// Versioned bucket: resolver returns 0 by construction. Pass 0
// directly — versioned objects sit on regular volumes and the
// lifecycle worker handles their expiration.
etag, errCode, sseMetadata = s3a.putToFiler(r, versionFilePath, body, bucket, normalizedObject, 1, 0,
s3a.versionedAfterCreate(bucket, normalizedObject, versionId, versionFileName, useInvertedFormat), true)
etag, errCode, sseMetadata = s3a.putToFiler(r, versionFilePath, body, bucket, normalizedObject, 1, 0, func(versionEntry *filer_pb.Entry) s3err.ErrorCode {
if err := s3a.updateLatestVersionInDirectory(bucket, normalizedObject, versionId, versionFileName, versionEntry); err != nil {
glog.Errorf("putVersionedObject: failed to update latest version in directory: %v", err)
return s3err.ErrInternalError
}
return s3err.ErrNone
})
if errCode != s3err.ErrNone {
glog.Errorf("putVersionedObject: failed to upload version: %v", errCode)
return "", "", errCode, SSEResponseMetadata{}
-272
View File
@@ -1,272 +0,0 @@
package s3api
import (
"context"
"fmt"
"net/http"
"path"
"strings"
"time"
"github.com/seaweedfs/seaweedfs/weed/pb"
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
"github.com/seaweedfs/seaweedfs/weed/s3api/s3_constants"
"github.com/seaweedfs/seaweedfs/weed/s3api/s3err"
"github.com/seaweedfs/seaweedfs/weed/util"
)
// objectWriteRouteKeyPrefix namespaces an object's full path into the ring key
// used to resolve and forward its writes. Shared by every routed builder so the
// gateway and filer hash the same key.
const objectWriteRouteKeyPrefix = "s3.object.write:"
// objectRouteKey is the ring key the gateway hashes to resolve an object's owner
// filer. It is also sent as route_key on each routed transaction, so a non-owner
// filer (reached because the gateway's ring view was stale) forwards the
// transaction to the owner. All of an object's writes share this key.
func (s3a *S3ApiServer) objectRouteKey(bucket, object string) string {
return objectWriteRouteKeyPrefix + s3a.toFilerPath(bucket, object)
}
// routableWriteOwner returns the owner filer for an object's writes, or "" to
// keep them on the distributed lock. All writes to one object (versioned,
// suspended, non-versioned) share the owner. Any lookup error falls back.
func (s3a *S3ApiServer) routableWriteOwner(bucket, object string) pb.ServerAddress {
if object == "" || s3a.objectWriteLockClient == nil {
return ""
}
// Object-lock PUTs route: a versioned PUT creates a new version (never an
// overwrite of a locked one), and a non-versioned overwrite is WORM-checked
// gateway-side before dispatch. WORM-checked deletes use routedObjectOwner.
return s3a.objectWriteLockClient.PrimaryForKey(s3a.objectRouteKey(bucket, object))
}
// routedObjectOwner is routableWriteOwner restricted to non-versioned,
// non-object-lock buckets, for the unversioned DELETE fast path.
func (s3a *S3ApiServer) routedObjectOwner(bucket, object string) (pb.ServerAddress, bool) {
if configured, err := s3a.isVersioningConfigured(bucket); err != nil || configured {
return "", false
}
// An unversioned object-lock delete enforces WORM in the lock path; keep it
// on the lock rather than routing past the check.
if locked, err := s3a.isObjectLockEnabled(bucket); err != nil || locked {
return "", false
}
owner := s3a.routableWriteOwner(bucket, object)
return owner, owner != ""
}
// routeWriteCondition reduces the request's conditional headers for a routed
// create. A unique version path carries no precondition and only routes when the
// request is unconditional (a conditional versioned write must check the latest,
// which the lock path does); an overwrite carries the reduced condition.
func routeWriteCondition(r *http.Request, uniqueWritePath bool) (*filer_pb.WriteCondition, bool) {
cond, ok := buildWriteCondition(r)
if !ok {
return nil, false
}
if uniqueWritePath && cond != nil {
return nil, false
}
return cond, true
}
// buildWriteCondition reduces the request's conditional headers to a
// WriteCondition. ok=false (combined headers, time conditions, ETag lists, weak
// ETags) keeps gateway-side evaluation under the lock; a nil condition with
// ok=true means unconditional.
func buildWriteCondition(r *http.Request) (*filer_pb.WriteCondition, bool) {
headers, errCode := parseConditionalHeaders(r)
if errCode != s3err.ErrNone {
return nil, false
}
if !headers.isSet {
return nil, true
}
if !headers.ifModifiedSince.IsZero() || !headers.ifUnmodifiedSince.IsZero() {
return nil, false
}
hasMatch := headers.ifMatch != ""
hasNoneMatch := headers.ifNoneMatch != ""
switch {
case hasMatch && !hasNoneMatch:
if headers.ifMatch == "*" {
return clause(filer_pb.WriteCondition_IF_EXISTS), true
}
if etag, single := singleStrongETag(headers.ifMatch); single {
return etagClause(filer_pb.WriteCondition_IF_ETAG_MATCH, etag), true
}
return nil, false
case hasNoneMatch && !hasMatch:
if headers.ifNoneMatch == "*" {
return clause(filer_pb.WriteCondition_IF_NOT_EXISTS), true
}
if etag, single := singleStrongETag(headers.ifNoneMatch); single {
return etagClause(filer_pb.WriteCondition_IF_ETAG_NOT_MATCH, etag), true
}
return nil, false
default:
return nil, false
}
}
// buildDeleteCondition reduces a DeleteObject's If-Match header to a condition;
// DeleteObject honors only If-Match, matching checkDeleteIfMatch.
func buildDeleteCondition(r *http.Request) (*filer_pb.WriteCondition, bool) {
ifMatch := strings.TrimSpace(r.Header.Get(s3_constants.IfMatch))
switch {
case ifMatch == "":
return nil, true
case ifMatch == "*":
return clause(filer_pb.WriteCondition_IF_EXISTS), true
default:
if etag, single := singleStrongETag(ifMatch); single {
return etagClause(filer_pb.WriteCondition_IF_ETAG_MATCH, etag), true
}
return nil, false
}
}
func clause(kind filer_pb.WriteCondition_Kind) *filer_pb.WriteCondition {
return &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{{Kind: kind}}}
}
func etagClause(kind filer_pb.WriteCondition_Kind, etag string) *filer_pb.WriteCondition {
return &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{{Kind: kind, Etags: []string{etag}}}}
}
// singleStrongETag returns the normalized ETag when v carries exactly one strong
// ETag, and false for ETag lists or weak ("W/") ETags.
func singleStrongETag(v string) (string, bool) {
v = strings.TrimSpace(v)
if strings.Contains(v, ",") {
return "", false
}
if strings.HasPrefix(v, "W/") || strings.HasPrefix(v, "w/") {
return "", false
}
return strings.Trim(v, `"`), true
}
func (s3a *S3ApiServer) objectTxnOnFiler(owner pb.ServerAddress, req *filer_pb.ObjectTransactionRequest) (*filer_pb.ObjectTransactionResponse, error) {
var resp *filer_pb.ObjectTransactionResponse
err := pb.WithFilerClient(false, 0, owner, s3a.option.GrpcDialOption, func(client filer_pb.SeaweedFilerClient) error {
var e error
resp, e = client.ObjectTransaction(context.Background(), req)
return e
})
return resp, err
}
// routedPut writes an object entry as a one-mutation ObjectTransaction on the
// owner filer. lock_key is the object's full path so the transaction shares the
// per-path lock with a concurrent create or delete of the same key.
func (s3a *S3ApiServer) routedPut(owner pb.ServerAddress, routeKey, filePath string, entry *filer_pb.Entry, cond *filer_pb.WriteCondition) (*filer_pb.ObjectTransactionResponse, error) {
return s3a.objectTxnOnFiler(owner, &filer_pb.ObjectTransactionRequest{
LockKey: filePath,
RouteKey: routeKey,
Condition: cond,
Mutations: []*filer_pb.ObjectMutation{{
Type: filer_pb.ObjectMutation_PUT,
Directory: path.Dir(filePath),
Entry: entry,
}},
})
}
// routedMkFile builds an entry like filer_pb.MkFile and writes it through a
// routed PUT on the owner filer, for callers that would otherwise mkFile to the
// default filer (e.g. multipart completion of a non-versioned object).
func (s3a *S3ApiServer) routedMkFile(owner pb.ServerAddress, routeKey, parentDir, name string, chunks []*filer_pb.FileChunk, fn func(*filer_pb.Entry)) error {
now := time.Now().Unix()
entry := &filer_pb.Entry{
Name: name,
Attributes: &filer_pb.FuseAttributes{
Mtime: now,
Crtime: now,
FileMode: uint32(0770),
Uid: filer_pb.OS_UID,
Gid: filer_pb.OS_GID,
},
Chunks: chunks,
}
if fn != nil {
fn(entry)
}
resp, err := s3a.routedPut(owner, routeKey, parentDir+"/"+name, entry, nil)
if err != nil {
return err
}
if resp.Error != "" {
return fmt.Errorf("routed mkfile %s/%s: %s", parentDir, name, resp.Error)
}
return nil
}
// writeMultipartObject writes a completed multipart object entry, routed to the
// owner when known (so it serializes with concurrent writes to the same key)
// and falling back to a plain mkFile otherwise. routeKey must be the same key the
// caller used to resolve owner, so owner selection and forwarding stay consistent.
func (s3a *S3ApiServer) writeMultipartObject(owner pb.ServerAddress, routeKey, dir, name string, chunks []*filer_pb.FileChunk, fn func(*filer_pb.Entry)) error {
if owner != "" {
return s3a.routedMkFile(owner, routeKey, dir, name, chunks, fn)
}
return s3a.mkFile(dir, name, chunks, fn)
}
func (s3a *S3ApiServer) routedDelete(owner pb.ServerAddress, bucket, object string, cond *filer_pb.WriteCondition) (*filer_pb.ObjectTransactionResponse, error) {
// NewFullPath normalizes a trailing-slash directory-marker key (e.g. "dir/")
// to the entry name "dir", matching deleteUnversionedObjectWithClient.
fullpath := util.NewFullPath(s3a.bucketDir(bucket), object)
dir, name := fullpath.DirAndName()
return s3a.objectTxnOnFiler(owner, &filer_pb.ObjectTransactionRequest{
LockKey: string(fullpath),
RouteKey: s3a.objectRouteKey(bucket, object),
Condition: cond,
Mutations: []*filer_pb.ObjectMutation{{
Type: filer_pb.ObjectMutation_DELETE,
Directory: dir,
Name: name,
IsDeleteData: true,
}},
})
}
// routedMetadataReplace applies a metadata-only self-copy (REPLACE directive) to
// an existing object in place via a routed PATCH_EXTENDED. The owner merges the
// new managed metadata onto a fresh read of the entry under its per-path lock —
// so a concurrent change to non-managed keys (legal hold, retention, version id)
// is preserved rather than clobbered by a whole-entry rewrite — and bumps mtime.
// updatedMetadata is the full managed-metadata set (processMetadataBytes); the
// delete list is the managed keys the replace dropped.
func (s3a *S3ApiServer) routedMetadataReplace(owner pb.ServerAddress, bucket, object string, current *filer_pb.Entry, updatedMetadata map[string][]byte) error {
fullpath := util.NewFullPath(s3a.bucketDir(bucket), object)
dir, name := fullpath.DirAndName()
var del []string
for k := range current.Extended {
if isManagedCopyMetadataKey(k) {
if _, keep := updatedMetadata[k]; !keep {
del = append(del, k)
}
}
}
resp, err := s3a.objectTxnOnFiler(owner, &filer_pb.ObjectTransactionRequest{
LockKey: string(fullpath),
RouteKey: s3a.objectRouteKey(bucket, object),
Mutations: []*filer_pb.ObjectMutation{{
Type: filer_pb.ObjectMutation_PATCH_EXTENDED,
Directory: dir,
Name: name,
SetExtended: updatedMetadata,
DeleteExtended: del,
TouchMtime: true,
}},
})
if err != nil {
return err
}
if resp.Error != "" {
return fmt.Errorf("routed metadata replace %s/%s: %s", bucket, object, resp.Error)
}
return nil
}
@@ -1,177 +0,0 @@
package s3api
import (
"net/http"
"testing"
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
"github.com/seaweedfs/seaweedfs/weed/s3api/s3_constants"
)
func reqWith(headers map[string]string) *http.Request {
r, _ := http.NewRequest(http.MethodPut, "/b/o", nil)
for k, v := range headers {
r.Header.Set(k, v)
}
return r
}
// oneClause returns the single clause of cond, failing if it does not hold
// exactly one.
func oneClause(t *testing.T, cond *filer_pb.WriteCondition) *filer_pb.WriteCondition_Clause {
t.Helper()
if cond == nil {
t.Fatal("expected a condition, got nil")
}
if len(cond.Clauses) != 1 {
t.Fatalf("expected 1 clause, got %d", len(cond.Clauses))
}
return cond.Clauses[0]
}
func TestBuildWriteCondition(t *testing.T) {
t.Run("no headers is unconditional", func(t *testing.T) {
cond, ok := buildWriteCondition(reqWith(nil))
if !ok || cond != nil {
t.Fatalf("want (nil, true), got (%v, %v)", cond, ok)
}
})
t.Run("If-None-Match * to IF_NOT_EXISTS", func(t *testing.T) {
cond, ok := buildWriteCondition(reqWith(map[string]string{s3_constants.IfNoneMatch: "*"}))
if !ok {
t.Fatal("want ok")
}
if c := oneClause(t, cond); c.Kind != filer_pb.WriteCondition_IF_NOT_EXISTS {
t.Fatalf("kind = %v", c.Kind)
}
})
t.Run("If-Match * to IF_EXISTS", func(t *testing.T) {
cond, ok := buildWriteCondition(reqWith(map[string]string{s3_constants.IfMatch: "*"}))
if !ok {
t.Fatal("want ok")
}
if c := oneClause(t, cond); c.Kind != filer_pb.WriteCondition_IF_EXISTS {
t.Fatalf("kind = %v", c.Kind)
}
})
t.Run("If-Match strong etag to IF_ETAG_MATCH", func(t *testing.T) {
cond, ok := buildWriteCondition(reqWith(map[string]string{s3_constants.IfMatch: `"abc123"`}))
if !ok {
t.Fatal("want ok")
}
c := oneClause(t, cond)
if c.Kind != filer_pb.WriteCondition_IF_ETAG_MATCH || len(c.Etags) != 1 || c.Etags[0] != "abc123" {
t.Fatalf("clause = %+v", c)
}
})
t.Run("If-None-Match strong etag to IF_ETAG_NOT_MATCH", func(t *testing.T) {
cond, ok := buildWriteCondition(reqWith(map[string]string{s3_constants.IfNoneMatch: `"abc123"`}))
if !ok {
t.Fatal("want ok")
}
c := oneClause(t, cond)
if c.Kind != filer_pb.WriteCondition_IF_ETAG_NOT_MATCH || len(c.Etags) != 1 || c.Etags[0] != "abc123" {
t.Fatalf("clause = %+v", c)
}
})
t.Run("weak etag falls back", func(t *testing.T) {
if _, ok := buildWriteCondition(reqWith(map[string]string{s3_constants.IfMatch: `W/"abc"`})); ok {
t.Fatal("weak etag must not take the fast path")
}
})
t.Run("etag list falls back", func(t *testing.T) {
if _, ok := buildWriteCondition(reqWith(map[string]string{s3_constants.IfMatch: `"a","b"`})); ok {
t.Fatal("etag list must not take the fast path")
}
})
t.Run("both match and none-match falls back", func(t *testing.T) {
if _, ok := buildWriteCondition(reqWith(map[string]string{
s3_constants.IfMatch: "*",
s3_constants.IfNoneMatch: "*",
})); ok {
t.Fatal("ambiguous combination must not take the fast path")
}
})
t.Run("time-based falls back", func(t *testing.T) {
if _, ok := buildWriteCondition(reqWith(map[string]string{
"If-Unmodified-Since": "Wed, 21 Oct 2015 07:28:00 GMT",
})); ok {
t.Fatal("time condition must not take the fast path")
}
})
}
func TestBuildDeleteCondition(t *testing.T) {
t.Run("no If-Match is unconditional", func(t *testing.T) {
cond, ok := buildDeleteCondition(reqWith(nil))
if !ok || cond != nil {
t.Fatalf("want (nil, true), got (%v, %v)", cond, ok)
}
})
t.Run("If-Match * to IF_EXISTS", func(t *testing.T) {
cond, ok := buildDeleteCondition(reqWith(map[string]string{s3_constants.IfMatch: "*"}))
if !ok {
t.Fatal("want ok")
}
if c := oneClause(t, cond); c.Kind != filer_pb.WriteCondition_IF_EXISTS {
t.Fatalf("kind = %v", c.Kind)
}
})
t.Run("If-Match etag to IF_ETAG_MATCH", func(t *testing.T) {
cond, ok := buildDeleteCondition(reqWith(map[string]string{s3_constants.IfMatch: `"e"`}))
if !ok {
t.Fatal("want ok")
}
if c := oneClause(t, cond); c.Kind != filer_pb.WriteCondition_IF_ETAG_MATCH || c.Etags[0] != "e" {
t.Fatalf("clause = %+v", c)
}
})
t.Run("weak etag falls back", func(t *testing.T) {
if _, ok := buildDeleteCondition(reqWith(map[string]string{s3_constants.IfMatch: `W/"e"`})); ok {
t.Fatal("weak etag must not take the fast path")
}
})
}
func TestSingleStrongETag(t *testing.T) {
cases := []struct {
in string
want string
single bool
}{
{`"abc"`, "abc", true},
{` "abc" `, "abc", true},
{`abc`, "abc", true},
{`W/"abc"`, "", false},
{`w/"abc"`, "", false},
{`"a","b"`, "", false},
}
for _, c := range cases {
got, single := singleStrongETag(c.in)
if single != c.single || (single && got != c.want) {
t.Errorf("singleStrongETag(%q) = (%q, %v), want (%q, %v)", c.in, got, single, c.want, c.single)
}
}
}
func TestRouteWriteCondition(t *testing.T) {
// Unconditional routes either way.
if c, ok := routeWriteCondition(reqWith(nil), false); !ok || c != nil {
t.Fatalf("overwrite unconditional: got (%v,%v)", c, ok)
}
if c, ok := routeWriteCondition(reqWith(nil), true); !ok || c != nil {
t.Fatalf("unique unconditional: got (%v,%v)", c, ok)
}
// An overwrite carries a reducible condition.
if c, ok := routeWriteCondition(reqWith(map[string]string{s3_constants.IfMatch: `"e"`}), false); !ok || c == nil {
t.Fatalf("overwrite conditional should route: got (%v,%v)", c, ok)
}
// A conditional unique (versioned) write bails to the lock path.
if _, ok := routeWriteCondition(reqWith(map[string]string{s3_constants.IfMatch: `"e"`}), true); ok {
t.Fatal("conditional unique write must not route")
}
// A non-reducible condition bails regardless.
if _, ok := routeWriteCondition(reqWith(map[string]string{s3_constants.IfMatch: `W/"e"`}), false); ok {
t.Fatal("weak etag must not route")
}
}
@@ -1,188 +0,0 @@
package s3api
import (
"strconv"
"time"
"github.com/seaweedfs/seaweedfs/weed/glog"
"github.com/seaweedfs/seaweedfs/weed/pb"
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
"github.com/seaweedfs/seaweedfs/weed/s3api/s3_constants"
"github.com/seaweedfs/seaweedfs/weed/s3api/s3err"
"github.com/seaweedfs/seaweedfs/weed/util"
)
// objectWriteOwner resolves the filer that owns all of an object's writes,
// regardless of versioning state, or "" when no ring view is available. Normal,
// suspended, and versioned writes to the same object hash to one owner and
// serialize on its per-path lock.
func (s3a *S3ApiServer) objectWriteOwner(bucket, object string) pb.ServerAddress {
if s3a.objectWriteLockClient == nil {
return ""
}
return s3a.objectWriteLockClient.PrimaryForKey(s3a.objectRouteKey(bucket, object))
}
// latestPointerRecompute builds the RECOMPUTE_LATEST mutation that re-derives an
// object's .versions pointer. excludeName, when set, omits a version about to be
// deleted (so the pointer is repointed before the blob is removed); demote, when
// set, stamps the displaced prior latest with NoncurrentSinceNs.
func (s3a *S3ApiServer) latestPointerRecompute(bucket, object string, useInvertedFormat bool, excludeName string, demote bool) *filer_pb.ObjectMutation {
versionsPath := s3a.toFilerPath(bucket, object+s3_constants.VersionsFolder)
vdir, vname := util.FullPath(versionsPath).DirAndName()
rc := &filer_pb.Recompute{
ScanDir: versionsPath,
// Inverted ids sort newest-first, so the newest is the first ascending
// entry; legacy ids sort oldest-first (scan to the last).
Descending: !useInvertedFormat,
NameToKey: s3_constants.ExtLatestVersionFileNameKey,
SizeToKey: s3_constants.ExtLatestVersionSizeKey,
MtimeToKey: s3_constants.ExtLatestVersionMtimeKey,
CopyExtended: map[string]string{
s3_constants.ExtLatestVersionIdKey: s3_constants.ExtVersionIdKey,
s3_constants.ExtLatestVersionETagKey: s3_constants.ExtETagKey,
s3_constants.ExtLatestVersionOwnerKey: s3_constants.ExtAmzOwnerKey,
s3_constants.ExtLatestVersionIsDeleteMarker: s3_constants.ExtDeleteMarkerKey,
},
ExcludeName: excludeName,
}
if demote {
rc.DemoteKey = s3_constants.ExtNoncurrentSinceNsKey
rc.DemoteValue = []byte(strconv.FormatInt(time.Now().UnixNano(), 10))
}
return &filer_pb.ObjectMutation{
Type: filer_pb.ObjectMutation_RECOMPUTE_LATEST,
Directory: vdir,
Name: vname,
Recompute: rc,
}
}
// routedVersionedFinalize flips the .versions pointer to the newest version and
// demotes the prior latest, atomically under the object's per-path lock on the
// owner filer, via a single RECOMPUTE_LATEST. The version file is already
// written; the owner re-derives the pointer by scanning the directory.
func (s3a *S3ApiServer) routedVersionedFinalize(owner pb.ServerAddress, bucket, object string, useInvertedFormat bool) s3err.ErrorCode {
req := &filer_pb.ObjectTransactionRequest{
LockKey: s3a.toFilerPath(bucket, object),
RouteKey: s3a.objectRouteKey(bucket, object),
Mutations: []*filer_pb.ObjectMutation{s3a.latestPointerRecompute(bucket, object, useInvertedFormat, "", true)},
}
resp, err := s3a.objectTxnOnFiler(owner, req)
switch {
case err != nil:
glog.Errorf("routedVersionedFinalize: %s/%s on %s: %v", bucket, object, owner, err)
return s3err.ErrInternalError
case resp.Error != "":
glog.Errorf("routedVersionedFinalize: %s/%s: %s", bucket, object, resp.Error)
return s3err.ErrInternalError
default:
return s3err.ErrNone
}
}
// wormDeleteCondition returns the object-lock guards for a delete, or nil when
// the bucket has no object lock. Legal hold always blocks. Retention blocks
// while not elapsed; with governance bypass the retention guard is gated to
// COMPLIANCE mode, so a governance-mode version becomes deletable while a
// compliance-mode one stays protected — the filer decides from the version's
// mode under the lock, so the gateway never has to read it.
func wormDeleteCondition(worm, bypass bool) *filer_pb.WriteCondition {
if !worm {
return nil
}
retention := &filer_pb.WriteCondition_Clause{
Kind: filer_pb.WriteCondition_IF_EXTENDED_TIME_ELAPSED,
ExtKey: s3_constants.ExtRetentionUntilDateKey,
}
if bypass {
retention.GateKey = s3_constants.ExtObjectLockModeKey
retention.GateValue = s3_constants.RetentionModeCompliance
}
return &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{
{Kind: filer_pb.WriteCondition_IF_EXTENDED_NOT_EQUAL, ExtKey: s3_constants.ExtLegalHoldKey, ExtValue: s3_constants.LegalHoldOn},
retention,
}}
}
// routedDeleteSpecificVersion deletes one version off the distributed lock: in a
// single transaction on the owner it recomputes the .versions pointer excluding
// the version (repoint-before-delete, so a crash leaves a recoverable orphan
// rather than a dangling pointer) and deletes the version file. lock_key is the
// object (serializing the pointer recompute); for object-lock buckets the
// condition gates the delete on the version's WORM guards evaluated on the owner.
func (s3a *S3ApiServer) routedDeleteSpecificVersion(owner pb.ServerAddress, bucket, object, versionId string, worm, bypass bool) s3err.ErrorCode {
versionFileName := s3a.getVersionFileName(versionId)
versionsPath := s3a.toFilerPath(bucket, object+s3_constants.VersionsFolder)
cond := wormDeleteCondition(worm, bypass)
req := &filer_pb.ObjectTransactionRequest{
LockKey: s3a.toFilerPath(bucket, object),
RouteKey: s3a.objectRouteKey(bucket, object),
ConditionKey: versionsPath + "/" + versionFileName,
Condition: cond,
Mutations: []*filer_pb.ObjectMutation{
s3a.latestPointerRecompute(bucket, object, isNewFormatVersionId(versionId), versionFileName, false),
{Type: filer_pb.ObjectMutation_DELETE, Directory: versionsPath, Name: versionFileName, IsDeleteData: true},
},
}
resp, err := s3a.objectTxnOnFiler(owner, req)
switch {
case err != nil:
glog.Errorf("routedDeleteSpecificVersion: %s/%s %s on %s: %v", bucket, object, versionId, owner, err)
return s3err.ErrInternalError
case resp.ErrorCode == filer_pb.FilerError_PRECONDITION_FAILED:
// Legal hold or retention in force on the version.
return s3err.ErrAccessDenied
case resp.Error != "":
glog.Errorf("routedDeleteSpecificVersion: %s/%s %s: %s", bucket, object, versionId, resp.Error)
return s3err.ErrInternalError
default:
return s3err.ErrNone
}
}
// routedDeleteNullVersion deletes the null version (the regular object entry, not
// a .versions file) off the distributed lock. There is no pointer to recompute;
// the WORM guards, when present, gate the delete on the object entry itself
// (condition defaults to lock_key).
func (s3a *S3ApiServer) routedDeleteNullVersion(owner pb.ServerAddress, bucket, object string, worm, bypass bool) s3err.ErrorCode {
fullpath := util.NewFullPath(s3a.bucketDir(bucket), object)
dir, name := fullpath.DirAndName()
resp, err := s3a.objectTxnOnFiler(owner, &filer_pb.ObjectTransactionRequest{
LockKey: string(fullpath),
RouteKey: s3a.objectRouteKey(bucket, object),
Condition: wormDeleteCondition(worm, bypass),
Mutations: []*filer_pb.ObjectMutation{
{Type: filer_pb.ObjectMutation_DELETE, Directory: dir, Name: name, IsDeleteData: true},
},
})
switch {
case err != nil:
glog.Errorf("routedDeleteNullVersion: %s/%s on %s: %v", bucket, object, owner, err)
return s3err.ErrInternalError
case resp.ErrorCode == filer_pb.FilerError_PRECONDITION_FAILED:
return s3err.ErrAccessDenied
case resp.Error != "":
glog.Errorf("routedDeleteNullVersion: %s/%s: %s", bucket, object, resp.Error)
return s3err.ErrInternalError
default:
return s3err.ErrNone
}
}
// versionedAfterCreate returns the putToFiler hook that finalizes a versioned
// write: the routed RECOMPUTE_LATEST when the owner is known, else the existing
// lock-free updateLatestVersionInDirectory.
func (s3a *S3ApiServer) versionedAfterCreate(bucket, object, versionId, versionFileName string, useInvertedFormat bool) func(*filer_pb.Entry) s3err.ErrorCode {
owner := s3a.objectWriteOwner(bucket, object)
return func(versionEntry *filer_pb.Entry) s3err.ErrorCode {
if owner != "" {
return s3a.routedVersionedFinalize(owner, bucket, object, useInvertedFormat)
}
if err := s3a.updateLatestVersionInDirectory(bucket, object, versionId, versionFileName, versionEntry); err != nil {
glog.Errorf("putVersionedObject: failed to update latest version in directory: %v", err)
return s3err.ErrInternalError
}
return s3err.ErrNone
}
}
+2 -7
View File
@@ -242,13 +242,8 @@ func (s3a *S3ApiServer) createDeleteMarker(bucket, object string) (string, error
},
Extended: deleteMarkerExtended,
}
// Route the pointer flip to the owner filer when known (off the distributed
// lock); RECOMPUTE_LATEST picks the just-written marker as the new latest.
if owner := s3a.objectWriteOwner(bucket, cleanObject); owner != "" {
if code := s3a.routedVersionedFinalize(owner, bucket, cleanObject, useInvertedFormat); code != s3err.ErrNone {
return "", fmt.Errorf("createDeleteMarker: routed finalize failed for %s/%s: code %d", bucket, object, code)
}
} else if err = s3a.updateLatestVersionInDirectory(bucket, cleanObject, versionId, versionFileName, deleteMarkerEntry); err != nil {
err = s3a.updateLatestVersionInDirectory(bucket, cleanObject, versionId, versionFileName, deleteMarkerEntry)
if err != nil {
glog.Errorf("createDeleteMarker: failed to update latest version in directory: %v", err)
return "", fmt.Errorf("failed to update latest version in directory: %w", err)
}
-38
View File
@@ -1,38 +0,0 @@
package s3api
import (
"net/http"
"github.com/gorilla/mux"
"github.com/seaweedfs/seaweedfs/weed/s3api/s3_constants"
"github.com/seaweedfs/seaweedfs/weed/s3api/s3err"
)
// validateRequestPath rejects requests whose captured {bucket}/{object} mux
// vars would normalize to a parent-directory traversal once joined into a
// filer path. The router runs with mux.NewRouter().SkipClean(true), so
// segments like `..` survive routing; the filer's util.JoinPath later collapses
// them via filepath.Join. Without this guard, `GET /bucket-A/../evil-bucket/k`
// matches as bucket=bucket-A, object=../evil-bucket/k, the filer resolves the
// read against evil-bucket, while IAM authorizes against bucket-A.
func validateRequestPath(next http.Handler) http.Handler {
return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
vars := mux.Vars(r)
// When a var is in the matched route it must be non-empty: an empty
// bucket would let downstream path.Join collapse it and let the object
// key pick the bucket.
if bucket, ok := vars["bucket"]; ok {
if bucket == "" || !s3_constants.IsValidBucketName(bucket) {
s3err.WriteErrorResponse(w, r, s3err.ErrInvalidRequest)
return
}
}
if object, ok := vars["object"]; ok {
if object == "" || !s3_constants.IsValidObjectKey(object) {
s3err.WriteErrorResponse(w, r, s3err.ErrInvalidRequest)
return
}
}
next.ServeHTTP(w, r)
})
}
-87
View File
@@ -1,87 +0,0 @@
package s3api
import (
"net/http"
"net/http/httptest"
"testing"
"github.com/gorilla/mux"
)
func TestValidateRequestPath_RejectsTraversal(t *testing.T) {
tests := []struct {
name string
// rawPath is sent as the Request-URI; net/http.NewRequest does not
// rewrite the path, so `..` segments survive into mux when the router
// is built with SkipClean(true) — matching the production setup in
// weed/command/s3.go.
rawPath string
wantCode int
}{
{"clean path passes", "/bucket-a/folder/file.txt", http.StatusOK},
{"bucket only passes", "/bucket-a", http.StatusOK},
{"trailing slash passes", "/bucket-a/folder/", http.StatusOK},
{"leading dotdot rejected", "/bucket-a/../evil-bucket/test.txt", http.StatusBadRequest},
{"nested dotdot rejected", "/bucket-a/good/../evil/test.txt", http.StatusBadRequest},
{"backslash dotdot rejected", "/bucket-a/..\\evil\\test.txt", http.StatusBadRequest},
{"percent-encoded dotdot rejected", "/bucket-a/%2e%2e/evil/test.txt", http.StatusBadRequest},
{"bare dot object rejected", "/bucket-a/./evil/test.txt", http.StatusBadRequest},
{"dotdot bucket rejected", "/../buckets/evil", http.StatusBadRequest},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
router := mux.NewRouter().SkipClean(true)
sub := router.PathPrefix("/{bucket}").Subrouter()
sub.Use(validateRequestPath)
handlerCalled := false
pass := func(w http.ResponseWriter, r *http.Request) {
handlerCalled = true
w.WriteHeader(http.StatusOK)
}
// Mirror the production routes: /{bucket}/{object:(?s).+} for
// object-scoped requests, bare /{bucket} for bucket-scoped ones.
sub.Path("/{object:(?s).+}").HandlerFunc(pass)
sub.Path("").HandlerFunc(pass)
req := httptest.NewRequest(http.MethodGet, tt.rawPath, nil)
rr := httptest.NewRecorder()
router.ServeHTTP(rr, req)
if rr.Code != tt.wantCode {
t.Fatalf("path %q: got status %d, want %d (body=%q)", tt.rawPath, rr.Code, tt.wantCode, rr.Body.String())
}
if tt.wantCode == http.StatusBadRequest && handlerCalled {
t.Fatalf("path %q: inner handler reached despite rejection", tt.rawPath)
}
})
}
}
// Defense-in-depth: a future router or middleware that captures the {bucket}
// or {object} mux var as an empty string must still be rejected, even though
// mux's default `[^/]+` regex won't match an empty segment from a real URL.
func TestValidateRequestPath_RejectsEmptyCapturedVars(t *testing.T) {
tests := []struct {
name string
vars map[string]string
}{
{"empty bucket", map[string]string{"bucket": "", "object": "key"}},
{"empty object", map[string]string{"bucket": "bucket-a", "object": ""}},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
handlerCalled := false
h := validateRequestPath(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
handlerCalled = true
}))
req := mux.SetURLVars(httptest.NewRequest(http.MethodGet, "/", nil), tt.vars)
rr := httptest.NewRecorder()
h.ServeHTTP(rr, req)
if handlerCalled {
t.Fatalf("vars %v: inner handler reached despite empty capture", tt.vars)
}
})
}
}
+3 -33
View File
@@ -25,7 +25,6 @@ import (
"github.com/seaweedfs/seaweedfs/weed/iam/policy"
"github.com/seaweedfs/seaweedfs/weed/iam/sts"
"github.com/seaweedfs/seaweedfs/weed/pb"
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
"github.com/seaweedfs/seaweedfs/weed/pb/s3_lifecycle_pb"
"github.com/seaweedfs/seaweedfs/weed/pb/s3_pb"
"github.com/seaweedfs/seaweedfs/weed/s3api/policy_engine"
@@ -96,8 +95,6 @@ type S3ApiServer struct {
stsHandlers *STSHandlers // STS HTTP handlers for AssumeRoleWithWebIdentity
cipher bool // encrypt data on volume servers
newObjectWriteLock func(bucket, object string) objectWriteLock
// objectWriteLockClient resolves a key's owner filer for route-by-key.
objectWriteLockClient *cluster.LockClient
// Shared ReaderCache used by the S3 GET streaming path. It lives for the
// lifetime of the server so that concurrent and repeat reads share a
// single in-flight download per chunk, and so that no per-request
@@ -155,8 +152,6 @@ func NewS3ApiServerWithStore(router *mux.Router, option *S3ApiServerOption, expl
// Uses the battle-tested vidMap with filer-based lookups
// Supports multiple filer addresses with automatic failover for high availability
var filerClient *wdclient.FilerClient
var masterClient *wdclient.MasterClient
var objectWriteLockClient *cluster.LockClient
if len(option.Masters) > 0 {
// Enable filer discovery via master
masterMap := make(map[string]pb.ServerAddress)
@@ -167,22 +162,7 @@ func NewS3ApiServerWithStore(router *mux.Router, option *S3ApiServerOption, expl
if clientHost == "0.0.0.0" || clientHost == "" {
clientHost = util.DetectedHostAddress()
}
masterClient = wdclient.NewMasterClient(option.GrpcDialOption, option.FilerGroup, cluster.S3Type, pb.ServerAddress(util.JoinHostPort(clientHost, option.GrpcPort)), "", "", *pb.NewServiceDiscoveryFromMap(masterMap))
// Build the object-write lock client and subscribe to the master's
// lock-ring updates BEFORE starting the master loop, so the initial
// LockRingUpdate sent on connect isn't dropped (the master only delivers
// it once per connect). The masterClient already filters updates to this
// server's filer group.
if len(option.Filers) > 0 {
objectWriteLockClient = cluster.NewLockClient(option.GrpcDialOption, option.Filers[0])
masterClient.SetOnLockRingUpdateFn(func(update *master_pb.LockRingUpdate) {
servers := make([]pb.ServerAddress, 0, len(update.Servers))
for _, s := range update.Servers {
servers = append(servers, pb.ServerAddress(s))
}
objectWriteLockClient.SetRing(servers, update.Version)
})
}
masterClient := wdclient.NewMasterClient(option.GrpcDialOption, option.FilerGroup, cluster.S3Type, pb.ServerAddress(util.JoinHostPort(clientHost, option.GrpcPort)), "", "", *pb.NewServiceDiscoveryFromMap(masterMap))
// Start the master client connection loop - required for GetMaster() to work
go masterClient.KeepConnectedToMaster(context.Background())
@@ -283,14 +263,9 @@ func NewS3ApiServerWithStore(router *mux.Router, option *S3ApiServerOption, expl
}
if len(option.Filers) > 0 {
// Reuse the lock client built in the masters block (already subscribed to
// ring updates); create a plain one when no masters are configured.
if objectWriteLockClient == nil {
objectWriteLockClient = cluster.NewLockClient(option.GrpcDialOption, option.Filers[0])
}
s3ApiServer.objectWriteLockClient = objectWriteLockClient
objectWriteLockClient := cluster.NewLockClient(option.GrpcDialOption, option.Filers[0])
s3ApiServer.newObjectWriteLock = func(bucket, object string) objectWriteLock {
lockKey := objectWriteRouteKeyPrefix + s3ApiServer.toFilerPath(bucket, object)
lockKey := fmt.Sprintf("s3.object.write:%s", s3ApiServer.toFilerPath(bucket, object))
owner := fmt.Sprintf("s3api-%d", s3ApiServer.randomClientId)
lock := objectWriteLockClient.NewShortLivedLock(lockKey, owner)
if err := lock.AttemptToLock(objectWriteLockTTL); err != nil {
@@ -735,11 +710,6 @@ func (s3a *S3ApiServer) registerRouter(router *mux.Router) {
corsMiddleware := s3a.getCORSMiddleware()
for _, bucket := range routers {
// Reject `..`/`.`/NUL in {bucket} or {object} vars before any handler
// runs. SkipClean(true) keeps `..` in the matched path; the filer would
// otherwise collapse it via filepath.Join and cross bucket boundaries.
bucket.Use(validateRequestPath)
// Apply CORS middleware to bucket routers for automatic CORS header handling
bucket.Use(corsMiddleware.Handler)
+26 -4
View File
@@ -110,10 +110,32 @@ func writeJson(w http.ResponseWriter, r *http.Request, httpStatus int, obj inter
r.Method, r.URL.String(), httpStatus, string(bytes))
}
w.Header().Set("Content-Type", "application/json")
w.Header().Set("X-Content-Type-Options", "nosniff")
w.WriteHeader(httpStatus)
_, err = w.Write(bytes)
callback := r.FormValue("callback")
if callback == "" {
w.Header().Set("Content-Type", "application/json")
w.WriteHeader(httpStatus)
if httpStatus == http.StatusNotModified {
return
}
_, err = w.Write(bytes)
} else {
w.Header().Set("Content-Type", "application/javascript")
w.WriteHeader(httpStatus)
if httpStatus == http.StatusNotModified {
return
}
if _, err = w.Write([]uint8(callback)); err != nil {
return
}
if _, err = w.Write([]uint8("(")); err != nil {
return
}
fmt.Fprint(w, string(bytes))
if _, err = w.Write([]uint8(")")); err != nil {
return
}
}
return
}
-32
View File
@@ -1,8 +1,6 @@
package weed_server
import (
"net/http"
"net/http/httptest"
"strings"
"testing"
)
@@ -31,33 +29,3 @@ func TestParseURL(t *testing.T) {
}
}
}
func TestWriteJsonNoJSONP(t *testing.T) {
// callback= must be ignored; response is always application/json with nosniff.
cases := []string{"", "myCb", "<script>alert(1)</script>"}
for _, cb := range cases {
t.Run("callback="+cb, func(t *testing.T) {
url := "/x"
if cb != "" {
url += "?callback=" + cb
}
r := httptest.NewRequest(http.MethodGet, url, nil)
w := httptest.NewRecorder()
if err := writeJson(w, r, http.StatusOK, map[string]string{"k": "v"}); err != nil {
t.Fatalf("writeJson: %v", err)
}
if w.Code != http.StatusOK {
t.Errorf("status: got %d want 200", w.Code)
}
if got := w.Header().Get("Content-Type"); got != "application/json" {
t.Errorf("Content-Type: got %q want application/json", got)
}
if got := w.Header().Get("X-Content-Type-Options"); got != "nosniff" {
t.Errorf("X-Content-Type-Options: got %q want nosniff", got)
}
if got := w.Body.String(); got != `{"k":"v"}` {
t.Errorf("body: got %q want %q", got, `{"k":"v"}`)
}
})
}
}
+3 -330
View File
@@ -5,17 +5,15 @@ import (
"context"
"errors"
"fmt"
"math"
"os"
"path/filepath"
"strconv"
"time"
"github.com/seaweedfs/seaweedfs/weed/cluster"
"github.com/seaweedfs/seaweedfs/weed/filer"
"github.com/seaweedfs/seaweedfs/weed/glog"
"github.com/seaweedfs/seaweedfs/weed/operation"
"github.com/seaweedfs/seaweedfs/weed/pb"
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
"github.com/seaweedfs/seaweedfs/weed/storage/needle"
@@ -188,38 +186,8 @@ func (fs *FilerServer) CreateEntry(ctx context.Context, req *filer_pb.CreateEntr
newEntry.TtlSec = 0
}
// Serialize concurrent mutations to the same path on this filer so the
// read (existence/condition) and the write are atomic. Callers route a
// key's writes to this owner filer, making this local lock sufficient.
fullpath := newEntry.FullPath
pathLock := fs.entryLockTable.AcquireLock("CreateEntry", fullpath, util.ExclusiveLock)
defer fs.entryLockTable.ReleaseLock(fullpath, pathLock)
// Evaluate the optional precondition against the current entry while the
// path lock is held, so the check and the write are atomic on this filer.
// The fetched entry is then handed to CreateEntry below so it does not look
// the same path up again under the lock.
var existing *filer.Entry
if conditionIsSet(req.Condition) {
current, findErr := fs.filer.FindEntry(ctx, fullpath)
if findErr != nil && findErr != filer_pb.ErrNotFound {
return &filer_pb.CreateEntryResponse{}, fmt.Errorf("CreateEntry condition check %s: %w", fullpath, findErr)
}
if findErr == filer_pb.ErrNotFound {
current = nil
}
if !writeConditionSatisfied(req.Condition, current) {
glog.V(3).InfofCtx(ctx, "CreateEntry %s: precondition failed: %v", fullpath, req.Condition)
return &filer_pb.CreateEntryResponse{
Error: "precondition failed",
ErrorCode: filer_pb.FilerError_PRECONDITION_FAILED,
}, nil
}
existing = current
}
ctx, eventSink := filer.WithMetadataEventSink(ctx)
createErr := fs.filer.CreateEntry(ctx, newEntry, existing, req.OExcl, req.IsFromOtherCluster, req.Signatures, req.SkipCheckParentDirectory, so.MaxFileNameLength)
createErr := fs.filer.CreateEntry(ctx, newEntry, req.OExcl, req.IsFromOtherCluster, req.Signatures, req.SkipCheckParentDirectory, so.MaxFileNameLength)
if createErr == nil {
fs.filer.DeleteChunksNotRecursive(garbage)
@@ -244,301 +212,6 @@ func (fs *FilerServer) CreateEntry(ctx context.Context, req *filer_pb.CreateEntr
return
}
// ObjectTransaction applies an ordered list of entry mutations atomically with
// respect to other writers of the same object, by holding the per-path lock on
// lock_key for the whole call. The optional condition is checked first, against
// the entry at lock_key. This lets a caller describe a multi-entry object
// operation (e.g. delete the null version + write a delete marker + flip the
// latest pointer) as one request, replacing a distributed lock held across
// several RPCs. Callers must route the object's writes to its owner filer for
// the lock to be authoritative.
func (fs *FilerServer) ObjectTransaction(ctx context.Context, req *filer_pb.ObjectTransactionRequest) (*filer_pb.ObjectTransactionResponse, error) {
if req.LockKey == "" {
return &filer_pb.ObjectTransactionResponse{Error: "lock_key is required"}, nil
}
// Route-by-key: if this filer is not the ring owner of route_key, forward the
// whole transaction to the owner so its per-path lock is the single
// serialization point — even when the caller's ring view was stale. is_moved
// bounds this to one hop: a forwarded transaction is applied locally, so two
// filers that disagree on the owner during a ring change cannot loop.
if req.RouteKey != "" && !req.IsMoved && fs.filer.Dlm != nil {
if owner := fs.filer.Dlm.LockRing.GetPrimary(req.RouteKey); owner != "" && owner != fs.option.Host {
// Rebuild rather than copy the request struct (it carries a mutex);
// the pointer/slice fields are shared since the original is not mutated.
forwarded := &filer_pb.ObjectTransactionRequest{
LockKey: req.LockKey,
Condition: req.Condition,
Mutations: req.Mutations,
IsFromOtherCluster: req.IsFromOtherCluster,
Signatures: req.Signatures,
ConditionKey: req.ConditionKey,
RouteKey: req.RouteKey,
IsMoved: true,
}
glog.V(2).InfofCtx(ctx, "ObjectTransaction %s: forwarding to owner %s", req.LockKey, owner)
var resp *filer_pb.ObjectTransactionResponse
err := pb.WithFilerClient(false, 0, owner, fs.grpcDialOption, func(client filer_pb.SeaweedFilerClient) error {
var e error
resp, e = client.ObjectTransaction(ctx, forwarded)
return e
})
if err != nil {
return &filer_pb.ObjectTransactionResponse{}, err
}
return resp, nil
}
}
lockPath := util.FullPath(req.LockKey)
pathLock := fs.entryLockTable.AcquireLock("ObjectTransaction", lockPath, util.ExclusiveLock)
defer fs.entryLockTable.ReleaseLock(lockPath, pathLock)
if conditionIsSet(req.Condition) {
// The condition is evaluated against condition_key when set (e.g. a
// version entry whose WORM guards gate the delete), while the lock stays
// on lock_key (the object, serializing the pointer recompute).
conditionPath := lockPath
if req.ConditionKey != "" {
conditionPath = util.FullPath(req.ConditionKey)
}
current, findErr := fs.filer.FindEntry(ctx, conditionPath)
if findErr != nil && findErr != filer_pb.ErrNotFound {
return &filer_pb.ObjectTransactionResponse{}, fmt.Errorf("ObjectTransaction condition %s: %w", conditionPath, findErr)
}
if findErr == filer_pb.ErrNotFound {
current = nil
}
if !writeConditionSatisfied(req.Condition, current) {
glog.V(3).InfofCtx(ctx, "ObjectTransaction %s: precondition failed", conditionPath)
return &filer_pb.ObjectTransactionResponse{
Error: "precondition failed",
ErrorCode: filer_pb.FilerError_PRECONDITION_FAILED,
}, nil
}
}
for i, m := range req.Mutations {
if err := fs.applyObjectMutation(ctx, m, req.IsFromOtherCluster, req.Signatures); err != nil {
glog.V(2).InfofCtx(ctx, "ObjectTransaction %s mutation %d (%v): %v", lockPath, i, m.Type, err)
return &filer_pb.ObjectTransactionResponse{Error: fmt.Sprintf("mutation %d: %v", i, err)}, nil
}
}
return &filer_pb.ObjectTransactionResponse{}, nil
}
// ObjectTransactionBatch applies several object transactions in one round trip,
// each under its own per-path lock and independent of the others. A failed
// transaction (precondition or mutation error) is reported in its own response
// without aborting the rest, matching S3 multi-object semantics where each key
// succeeds or fails on its own.
func (fs *FilerServer) ObjectTransactionBatch(ctx context.Context, req *filer_pb.ObjectTransactionBatchRequest) (*filer_pb.ObjectTransactionBatchResponse, error) {
if req == nil {
return nil, status.Error(codes.InvalidArgument, "request is required")
}
resp := &filer_pb.ObjectTransactionBatchResponse{
Responses: make([]*filer_pb.ObjectTransactionResponse, 0, len(req.Transactions)),
}
for _, txn := range req.Transactions {
// Stop early if the caller went away; the request still holds the
// unprocessed transactions, so it is retried rather than lost.
if err := ctx.Err(); err != nil {
return nil, err
}
if txn == nil {
resp.Responses = append(resp.Responses, &filer_pb.ObjectTransactionResponse{Error: "nil transaction"})
continue
}
one, err := fs.ObjectTransaction(ctx, txn)
if err != nil {
// A transport-level error on one transaction is surfaced as that
// transaction's error; the batch RPC itself still succeeds.
one = &filer_pb.ObjectTransactionResponse{Error: err.Error()}
}
resp.Responses = append(resp.Responses, one)
}
return resp, nil
}
// applyObjectMutation applies a single mutation while the transaction's path
// lock is held. PUT entries are expected to be fully prepared by the caller
// (chunks resolved); mutations here are metadata-scoped. A DELETE of an absent
// entry and a PATCH of an absent entry are no-ops, so transactions are
// idempotent on replay.
func (fs *FilerServer) applyObjectMutation(ctx context.Context, m *filer_pb.ObjectMutation, fromOtherCluster bool, signatures []int32) error {
switch m.Type {
case filer_pb.ObjectMutation_PUT:
if m.Entry == nil {
return fmt.Errorf("PUT requires an entry")
}
newEntry := filer.FromPbEntry(m.Directory, m.Entry)
return fs.filer.CreateEntry(ctx, newEntry, nil, false, fromOtherCluster, signatures, false, fs.filer.MaxFilenameLength)
case filer_pb.ObjectMutation_DELETE:
fullpath := util.NewFullPath(m.Directory, m.Name)
err := fs.filer.DeleteEntryMetaAndData(ctx, fullpath, m.IsRecursive, false, m.IsDeleteData, fromOtherCluster, signatures, 0)
if err == filer_pb.ErrNotFound {
return nil
}
return err
case filer_pb.ObjectMutation_PATCH_EXTENDED:
fullpath := util.NewFullPath(m.Directory, m.Name)
oldEntry, err := fs.filer.FindEntry(ctx, fullpath)
if err == filer_pb.ErrNotFound {
return nil
}
if err != nil {
return err
}
// Patch a copy so oldEntry still reflects the pre-update state for the
// metadata notification's diff.
newEntry := oldEntry.ShallowClone()
newEntry.Extended = make(map[string][]byte, len(oldEntry.Extended))
for k, v := range oldEntry.Extended {
newEntry.Extended[k] = v
}
for k, v := range m.SetExtended {
newEntry.Extended[k] = v
}
for _, k := range m.DeleteExtended {
delete(newEntry.Extended, k)
}
if m.SetContent {
newEntry.Content = m.Content
// Keep FileSize consistent with content for files; some stores and
// tools read the attribute directly. Directories carry no file size.
if !newEntry.IsDirectory() {
newEntry.FileSize = uint64(len(m.Content))
}
}
if m.TouchMtime {
newEntry.Attr.Mtime = time.Now()
}
if err := fs.filer.UpdateEntry(ctx, oldEntry, newEntry); err != nil {
return err
}
// Emit the metadata event so the update replicates and subscribers see it,
// matching the UpdateEntry handler.
fs.filer.NotifyUpdateEvent(ctx, oldEntry, newEntry, true, fromOtherCluster, signatures)
return nil
case filer_pb.ObjectMutation_RECOMPUTE_LATEST:
return fs.applyRecomputeLatest(ctx, m)
default:
return fmt.Errorf("unknown mutation type %v", m.Type)
}
}
// applyRecomputeLatest re-derives the pointer entry (m.Directory/m.Name) from the
// current contents of recompute.scan_dir, under the transaction's lock. It is
// mechanical: pick the child that sorts last (descending) or first by name, copy
// the mapped extended keys from it into the pointer, and store its name under
// name_to_key. When the scanned directory is empty the pointer keys are cleared.
// The caller, which knows the versioning scheme, supplies the direction and the
// key mappings. A missing pointer entry is a no-op (idempotent on replay).
func (fs *FilerServer) applyRecomputeLatest(ctx context.Context, m *filer_pb.ObjectMutation) error {
rc := m.Recompute
if rc == nil {
return fmt.Errorf("RECOMPUTE_LATEST requires recompute parameters")
}
pointer, err := fs.filer.FindEntry(ctx, util.NewFullPath(m.Directory, m.Name))
if err == filer_pb.ErrNotFound {
return nil
}
if err != nil {
return err
}
if pointer.Extended == nil {
pointer.Extended = make(map[string][]byte)
}
// Remember the prior chosen child so it can be demoted once the pointer moves.
var priorName string
if rc.NameToKey != "" {
priorName = string(pointer.Extended[rc.NameToKey])
}
// The store streams entries ascending by name. For the lowest-name pick we
// only need the first entry, so cap the listing at one; for the highest-name
// pick we must scan all and keep the last (the store has no reverse order).
// With exclude_name set the first child may be the excluded one, so the cap
// is lifted to find the first non-excluded entry.
limit := int64(math.MaxInt32)
if !rc.Descending && rc.ExcludeName == "" {
limit = 1
}
var chosen *filer.Entry
_, listErr := fs.filer.StreamListDirectoryEntries(ctx, util.FullPath(rc.ScanDir), "", false, limit, "", "", "", func(entry *filer.Entry) (bool, error) {
if rc.ExcludeName != "" && entry.Name() == rc.ExcludeName {
return true, nil
}
chosen = entry
return rc.Descending, nil
})
if listErr != nil {
return listErr
}
cleared := []string{rc.NameToKey, rc.SizeToKey, rc.MtimeToKey}
if chosen == nil {
for pointerKey := range rc.CopyExtended {
delete(pointer.Extended, pointerKey)
}
for _, k := range cleared {
if k != "" {
delete(pointer.Extended, k)
}
}
} else {
for pointerKey, sourceKey := range rc.CopyExtended {
if v, ok := chosen.Extended[sourceKey]; ok {
pointer.Extended[pointerKey] = v
} else {
delete(pointer.Extended, pointerKey)
}
}
if rc.NameToKey != "" {
pointer.Extended[rc.NameToKey] = []byte(chosen.Name())
}
if rc.SizeToKey != "" {
pointer.Extended[rc.SizeToKey] = []byte(strconv.FormatUint(chosen.FileSize, 10))
}
if rc.MtimeToKey != "" {
pointer.Extended[rc.MtimeToKey] = []byte(strconv.FormatInt(chosen.Mtime.Unix(), 10))
}
}
if err := fs.filer.UpdateEntry(ctx, pointer, pointer); err != nil {
return err
}
// Stamp the displaced prior child (e.g. NoncurrentSinceNs for lifecycle).
newName := ""
if chosen != nil {
newName = chosen.Name()
}
if rc.DemoteKey != "" && priorName != "" && priorName != newName {
priorEntry, perr := fs.filer.FindEntry(ctx, util.NewFullPath(rc.ScanDir, priorName))
if perr == filer_pb.ErrNotFound {
return nil
}
if perr != nil {
return perr
}
if priorEntry.Extended == nil {
priorEntry.Extended = make(map[string][]byte)
}
priorEntry.Extended[rc.DemoteKey] = rc.DemoteValue
return fs.filer.UpdateEntry(ctx, priorEntry, priorEntry)
}
return nil
}
func (fs *FilerServer) UpdateEntry(ctx context.Context, req *filer_pb.UpdateEntryRequest) (*filer_pb.UpdateEntryResponse, error) {
glog.V(4).InfofCtx(ctx, "UpdateEntry %v", req)
@@ -693,7 +366,7 @@ func (fs *FilerServer) AppendToEntry(ctx context.Context, req *filer_pb.AppendTo
glog.V(0).InfofCtx(ctx, "MaybeManifestize: %v", err)
}
err = fs.filer.CreateEntry(context.Background(), entry, nil, false, false, nil, false, fs.filer.MaxFilenameLength)
err = fs.filer.CreateEntry(context.Background(), entry, false, false, nil, false, fs.filer.MaxFilenameLength)
return &filer_pb.AppendToEntryResponse{}, err
}
-136
View File
@@ -1,136 +0,0 @@
package weed_server
import (
"strconv"
"strings"
"time"
"github.com/seaweedfs/seaweedfs/weed/filer"
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
"github.com/seaweedfs/seaweedfs/weed/s3api/s3_constants"
)
// conditionIsSet reports whether a condition asks for any check at all.
func conditionIsSet(cond *filer_pb.WriteCondition) bool {
return cond != nil && len(cond.Clauses) > 0
}
// writeConditionSatisfied reports whether the precondition holds against the
// current entry (nil if absent), evaluated under the path lock. Every clause
// must hold (logical AND).
func writeConditionSatisfied(cond *filer_pb.WriteCondition, current *filer.Entry) bool {
for _, c := range cond.Clauses {
if !clauseSatisfied(c, current) {
return false
}
}
return true
}
// clauseSatisfied evaluates one primitive against the current entry. For the
// ETag kinds, etags is a set: IF_ETAG_MATCH holds when the current ETag equals
// any member, IF_ETAG_NOT_MATCH when it equals none. The IF_EXTENDED_* kinds are
// generic guards on an extended attribute used to enforce object-lock (legal
// hold and retention) without S3 knowledge in the filer.
func clauseSatisfied(c *filer_pb.WriteCondition_Clause, current *filer.Entry) bool {
exists := current != nil
switch c.Kind {
case filer_pb.WriteCondition_NONE:
return true
case filer_pb.WriteCondition_IF_NOT_EXISTS:
return !exists
case filer_pb.WriteCondition_IF_EXISTS:
return exists
case filer_pb.WriteCondition_IF_ETAG_MATCH:
return exists && etagInSet(storedEntryETag(current), c.Etags, c.AllowWeak)
case filer_pb.WriteCondition_IF_ETAG_NOT_MATCH:
return !exists || !etagInSet(storedEntryETag(current), c.Etags, c.AllowWeak)
case filer_pb.WriteCondition_IF_UNMODIFIED_SINCE:
return !exists || current.Attr.Mtime.Unix() <= c.UnixTime
case filer_pb.WriteCondition_IF_MODIFIED_SINCE:
return !exists || current.Attr.Mtime.Unix() > c.UnixTime
case filer_pb.WriteCondition_IF_EXTENDED_NOT_EQUAL:
if !exists {
return true
}
v, ok := current.Extended[c.ExtKey]
return !ok || string(v) != c.ExtValue
case filer_pb.WriteCondition_IF_EXTENDED_TIME_ELAPSED:
if !exists {
return true
}
// An optional gate scopes the guard: when gate_key is set, the time check
// only applies if extended[gate_key] == gate_value. This lets the gateway
// express governance bypass (enforce retention only for COMPLIANCE mode)
// without reading the entry — the filer decides under the lock.
if c.GateKey != "" {
gv, gok := current.Extended[c.GateKey]
if !gok || string(gv) != c.GateValue {
return true
}
}
v, ok := current.Extended[c.ExtKey]
if !ok {
return true
}
deadline, err := strconv.ParseInt(strings.TrimSpace(string(v)), 10, 64)
if err != nil {
// An unparseable retention deadline is treated as still in force, so
// a malformed attribute fails safe (write blocked) rather than open.
return false
}
return deadline <= time.Now().Unix()
default:
// An unrecognized clause kind (e.g. from a newer client) must not be
// treated as satisfied, which would silently bypass the guard. Fail
// closed so the write is blocked rather than slipping through.
return false
}
}
// etagInSet reports whether stored matches any candidate. A strong comparison
// (allowWeak false) treats a weak ETag as never equal; a weak comparison
// ignores the W/ marker on both sides.
func etagInSet(stored string, candidates []string, allowWeak bool) bool {
for _, c := range candidates {
if etagEqual(stored, c, allowWeak) {
return true
}
}
return false
}
func etagEqual(stored, expected string, allowWeak bool) bool {
sv, sWeak := canonicalETag(stored)
ev, eWeak := canonicalETag(expected)
// RFC 7232 strong comparison: a weak ETag on either side never matches.
if !allowWeak && (sWeak || eWeak) {
return false
}
return sv == ev
}
// canonicalETag splits off the weak (W/) marker before stripping quotes, so a
// weak ETag like W/"abc" yields ("abc", true).
func canonicalETag(etag string) (value string, weak bool) {
etag = strings.TrimSpace(etag)
if strings.HasPrefix(etag, "W/") {
return strings.Trim(etag[len("W/"):], `"`), true
}
return strings.Trim(etag, `"`), false
}
// storedEntryETag mirrors the S3 gateway's ETag precedence (the stored
// Seaweed ETag extended attribute, then the chunk/Md5 fallback) so conditional
// comparisons match what the gateway computes, without coupling the filer to
// S3 request handling.
func storedEntryETag(entry *filer.Entry) string {
if v, ok := entry.Extended[s3_constants.ExtETagKey]; ok && len(v) > 0 {
return normalizeETag(string(v))
}
return normalizeETag(filer.ETagEntry(entry))
}
func normalizeETag(etag string) string {
return strings.Trim(etag, `"`)
}
@@ -1,286 +0,0 @@
package weed_server
import (
"context"
"strconv"
"testing"
"time"
"github.com/seaweedfs/seaweedfs/weed/filer"
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
"github.com/seaweedfs/seaweedfs/weed/s3api/s3_constants"
"github.com/seaweedfs/seaweedfs/weed/util"
)
func entryWithETag(etag string, mtime time.Time) *filer.Entry {
return &filer.Entry{
FullPath: "/test/obj",
Attr: filer.Attr{Mtime: mtime},
Extended: map[string][]byte{s3_constants.ExtETagKey: []byte(etag)},
}
}
// one wraps a single clause into a condition.
func one(c *filer_pb.WriteCondition_Clause) *filer_pb.WriteCondition {
return &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{c}}
}
func TestWriteConditionSatisfied(t *testing.T) {
base := time.Unix(1700000000, 0)
present := entryWithETag("abc", base)
cases := []struct {
name string
cond *filer_pb.WriteCondition
cur *filer.Entry
want bool
}{
{"empty-absent", &filer_pb.WriteCondition{}, nil, true},
{"ifnotexists-absent", one(&filer_pb.WriteCondition_Clause{Kind: filer_pb.WriteCondition_IF_NOT_EXISTS}), nil, true},
{"ifnotexists-present", one(&filer_pb.WriteCondition_Clause{Kind: filer_pb.WriteCondition_IF_NOT_EXISTS}), present, false},
{"ifexists-absent", one(&filer_pb.WriteCondition_Clause{Kind: filer_pb.WriteCondition_IF_EXISTS}), nil, false},
{"ifexists-present", one(&filer_pb.WriteCondition_Clause{Kind: filer_pb.WriteCondition_IF_EXISTS}), present, true},
{"etagmatch-hit", one(&filer_pb.WriteCondition_Clause{Kind: filer_pb.WriteCondition_IF_ETAG_MATCH, Etags: []string{`"abc"`}}), present, true},
{"etagmatch-miss", one(&filer_pb.WriteCondition_Clause{Kind: filer_pb.WriteCondition_IF_ETAG_MATCH, Etags: []string{`"zzz"`}}), present, false},
{"etagmatch-absent", one(&filer_pb.WriteCondition_Clause{Kind: filer_pb.WriteCondition_IF_ETAG_MATCH, Etags: []string{`"abc"`}}), nil, false},
{"etagnotmatch-hit", one(&filer_pb.WriteCondition_Clause{Kind: filer_pb.WriteCondition_IF_ETAG_NOT_MATCH, Etags: []string{`"abc"`}}), present, false},
{"etagnotmatch-miss", one(&filer_pb.WriteCondition_Clause{Kind: filer_pb.WriteCondition_IF_ETAG_NOT_MATCH, Etags: []string{`"zzz"`}}), present, true},
{"etagnotmatch-absent", one(&filer_pb.WriteCondition_Clause{Kind: filer_pb.WriteCondition_IF_ETAG_NOT_MATCH, Etags: []string{`"abc"`}}), nil, true},
{"unmodsince-ok", one(&filer_pb.WriteCondition_Clause{Kind: filer_pb.WriteCondition_IF_UNMODIFIED_SINCE, UnixTime: base.Unix()}), present, true},
{"unmodsince-fail", one(&filer_pb.WriteCondition_Clause{Kind: filer_pb.WriteCondition_IF_UNMODIFIED_SINCE, UnixTime: base.Unix() - 1}), present, false},
{"modsince-ok", one(&filer_pb.WriteCondition_Clause{Kind: filer_pb.WriteCondition_IF_MODIFIED_SINCE, UnixTime: base.Unix() - 1}), present, true},
{"modsince-fail", one(&filer_pb.WriteCondition_Clause{Kind: filer_pb.WriteCondition_IF_MODIFIED_SINCE, UnixTime: base.Unix()}), present, false},
}
for _, tc := range cases {
if got := writeConditionSatisfied(tc.cond, tc.cur); got != tc.want {
t.Errorf("%s: got %v want %v", tc.name, got, tc.want)
}
}
}
func TestWriteConditionClauses(t *testing.T) {
base := time.Unix(1700000000, 0)
present := entryWithETag("abc", base)
matchAny := func(etags ...string) *filer_pb.WriteCondition {
return &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{
{Kind: filer_pb.WriteCondition_IF_ETAG_MATCH, Etags: etags},
}}
}
noneOf := func(etags ...string) *filer_pb.WriteCondition {
return &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{
{Kind: filer_pb.WriteCondition_IF_ETAG_NOT_MATCH, Etags: etags},
}}
}
cases := []struct {
name string
cond *filer_pb.WriteCondition
cur *filer.Entry
want bool
}{
{"set-match-hit", matchAny(`"x"`, `"abc"`, `"y"`), present, true},
{"set-match-miss", matchAny(`"x"`, `"y"`), present, false},
{"set-none-clean", noneOf(`"x"`, `"y"`), present, true},
{"set-none-hit", noneOf(`"abc"`, `"y"`), present, false},
// A weak request ETag never matches under strong comparison.
{"weak-strong-fails", matchAny(`W/"abc"`), present, false},
// allow_weak compares ignoring the W/ marker.
{"weak-allowed", &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{
{Kind: filer_pb.WriteCondition_IF_ETAG_MATCH, Etags: []string{`W/"abc"`}, AllowWeak: true},
}}, present, true},
// Compound clauses are ANDed.
{"compound-ok", &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{
{Kind: filer_pb.WriteCondition_IF_ETAG_MATCH, Etags: []string{`"abc"`}},
{Kind: filer_pb.WriteCondition_IF_UNMODIFIED_SINCE, UnixTime: base.Unix()},
}}, present, true},
{"compound-second-fails", &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{
{Kind: filer_pb.WriteCondition_IF_ETAG_MATCH, Etags: []string{`"abc"`}},
{Kind: filer_pb.WriteCondition_IF_UNMODIFIED_SINCE, UnixTime: base.Unix() - 1},
}}, present, false},
}
for _, tc := range cases {
if got := writeConditionSatisfied(tc.cond, tc.cur); got != tc.want {
t.Errorf("%s: got %v want %v", tc.name, got, tc.want)
}
}
}
// The generic IF_EXTENDED_* guards express object-lock without S3 knowledge in
// the filer: a legal hold (IF_EXTENDED_NOT_EQUAL) and a retention deadline
// (IF_EXTENDED_TIME_ELAPSED).
func TestWriteConditionObjectLockGuards(t *testing.T) {
now := time.Now()
withExt := func(ext map[string]string) *filer.Entry {
e := &filer.Entry{FullPath: "/test/obj", Attr: filer.Attr{Mtime: now}, Extended: map[string][]byte{}}
for k, v := range ext {
e.Extended[k] = []byte(v)
}
return e
}
legalHold := &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{
{Kind: filer_pb.WriteCondition_IF_EXTENDED_NOT_EQUAL, ExtKey: "lock-hold", ExtValue: "ON"},
}}
retention := &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{
{Kind: filer_pb.WriteCondition_IF_EXTENDED_TIME_ELAPSED, ExtKey: "retain-until"},
}}
// Governance bypass: the retention guard is gated to COMPLIANCE mode, so a
// governance-mode (or unmoded) entry is deletable while compliance stays
// protected.
gatedRetention := &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{
{Kind: filer_pb.WriteCondition_IF_EXTENDED_TIME_ELAPSED, ExtKey: "retain-until", GateKey: "lock-mode", GateValue: "COMPLIANCE"},
}}
future := strconv.FormatInt(now.Add(time.Hour).Unix(), 10)
past := strconv.FormatInt(now.Add(-time.Hour).Unix(), 10)
cases := []struct {
name string
cond *filer_pb.WriteCondition
cur *filer.Entry
want bool
}{
{"hold-on-blocks", legalHold, withExt(map[string]string{"lock-hold": "ON"}), false},
{"hold-off-allows", legalHold, withExt(map[string]string{"lock-hold": "OFF"}), true},
{"hold-absent-allows", legalHold, withExt(nil), true},
{"hold-on-new-object", legalHold, nil, true}, // nothing to protect yet
{"retain-future-blocks", retention, withExt(map[string]string{"retain-until": future}), false},
{"retain-past-allows", retention, withExt(map[string]string{"retain-until": past}), true},
{"retain-absent-allows", retention, withExt(nil), true},
{"retain-malformed-blocks", retention, withExt(map[string]string{"retain-until": "soon"}), false},
// Governance bypass (gated to COMPLIANCE): compliance still blocks, but
// governance and unmoded entries become deletable despite future retention.
{"bypass-compliance-blocks", gatedRetention, withExt(map[string]string{"retain-until": future, "lock-mode": "COMPLIANCE"}), false},
{"bypass-governance-allows", gatedRetention, withExt(map[string]string{"retain-until": future, "lock-mode": "GOVERNANCE"}), true},
{"bypass-no-mode-allows", gatedRetention, withExt(map[string]string{"retain-until": future}), true},
// Composed WORM guard: legal hold AND retention, both clear -> allowed.
{"worm-both-clear", &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{
{Kind: filer_pb.WriteCondition_IF_EXTENDED_NOT_EQUAL, ExtKey: "lock-hold", ExtValue: "ON"},
{Kind: filer_pb.WriteCondition_IF_EXTENDED_TIME_ELAPSED, ExtKey: "retain-until"},
}}, withExt(map[string]string{"lock-hold": "OFF", "retain-until": past}), true},
// Either guard tripping blocks the whole WORM condition.
{"worm-hold-trips", &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{
{Kind: filer_pb.WriteCondition_IF_EXTENDED_NOT_EQUAL, ExtKey: "lock-hold", ExtValue: "ON"},
{Kind: filer_pb.WriteCondition_IF_EXTENDED_TIME_ELAPSED, ExtKey: "retain-until"},
}}, withExt(map[string]string{"lock-hold": "ON", "retain-until": past}), false},
}
for _, tc := range cases {
if got := writeConditionSatisfied(tc.cond, tc.cur); got != tc.want {
t.Errorf("%s: got %v want %v", tc.name, got, tc.want)
}
}
}
// An unrecognized clause kind (e.g. from a newer client) fails closed, so a
// guard can't be silently bypassed by an older filer. NONE stays a no-op.
func TestWriteConditionUnknownKindFailsClosed(t *testing.T) {
present := entryWithETag("abc", time.Now())
unknown := &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{
{Kind: filer_pb.WriteCondition_Kind(9999)},
}}
if writeConditionSatisfied(unknown, present) {
t.Error("unknown clause kind must not be satisfied (fail closed) for an existing entry")
}
if writeConditionSatisfied(unknown, nil) {
t.Error("unknown clause kind must not be satisfied (fail closed) for an absent entry")
}
none := &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{
{Kind: filer_pb.WriteCondition_NONE},
}}
if !writeConditionSatisfied(none, present) {
t.Error("a NONE clause must be satisfied (no-op)")
}
}
// storedEntryETag prefers the stored Seaweed ETag attribute and falls back to
// the Md5-derived ETag, matching the S3 gateway.
func TestStoredEntryETag(t *testing.T) {
withExt := entryWithETag("explicit", time.Unix(0, 0))
if got := storedEntryETag(withExt); got != "explicit" {
t.Errorf("extended etag: got %q", got)
}
md5Only := &filer.Entry{Attr: filer.Attr{Md5: []byte{0xab, 0xcd}}}
if got := storedEntryETag(md5Only); got != "abcd" {
t.Errorf("md5 fallback: got %q", got)
}
}
// The CreateEntry handler enforces the precondition atomically: a matching
// If-Match overwrites, a non-matching one returns PRECONDITION_FAILED.
func TestCreateEntryConditionEnforced(t *testing.T) {
store := newRenameTestStore()
store.entries["/test/obj"] = &filer.Entry{
FullPath: "/test/obj",
Attr: filer.Attr{Inode: 1, Mtime: time.Unix(1700000000, 0)},
Extended: map[string][]byte{s3_constants.ExtETagKey: []byte("abc")},
}
f := newRenameTestFiler(store)
f.DirBucketsPath = "/buckets"
fs := &FilerServer{filer: f, option: &FilerOption{}, entryLockTable: util.NewLockTable[util.FullPath]()}
req := func(etag string) *filer_pb.CreateEntryRequest {
return &filer_pb.CreateEntryRequest{
Directory: "/test",
SkipCheckParentDirectory: true,
Entry: &filer_pb.Entry{
Name: "obj",
Attributes: &filer_pb.FuseAttributes{Mtime: 1700000001, FileMode: 0644, Inode: 2},
},
Condition: one(&filer_pb.WriteCondition_Clause{Kind: filer_pb.WriteCondition_IF_ETAG_MATCH, Etags: []string{etag}}),
}
}
resp, err := fs.CreateEntry(context.Background(), req(`"zzz"`))
if err != nil {
t.Fatalf("unexpected err: %v", err)
}
if resp.ErrorCode != filer_pb.FilerError_PRECONDITION_FAILED {
t.Fatalf("mismatched etag: want PRECONDITION_FAILED, got %v (%q)", resp.ErrorCode, resp.Error)
}
resp, err = fs.CreateEntry(context.Background(), req(`"abc"`))
if err != nil {
t.Fatalf("unexpected err: %v", err)
}
if resp.Error != "" {
t.Fatalf("matching etag should overwrite, got error %q", resp.Error)
}
}
// CreateEntry reuses a provided existing entry instead of reading the store
// again; passing nil makes it look the path up itself.
func TestCreateEntryReusesProvidedExisting(t *testing.T) {
existing := &filer.Entry{
FullPath: "/test/obj",
Attr: filer.Attr{Inode: 1, Mtime: time.Unix(1700000000, 0), Crtime: time.Unix(1700000000, 0)},
Extended: map[string][]byte{s3_constants.ExtETagKey: []byte("abc")},
}
newEntry := func() *filer.Entry {
return &filer.Entry{
FullPath: "/test/obj",
Attr: filer.Attr{Inode: 1, Mtime: time.Unix(1700000001, 0)},
}
}
// Provided existing vs nil: the only difference is CreateEntry's own lookup,
// so providing it must save exactly one path read (other store-layer reads
// during the overwrite are the same in both runs).
store := newRenameTestStore()
store.entries["/test/obj"] = existing.ShallowClone()
f := newRenameTestFiler(store)
if err := f.CreateEntry(context.Background(), newEntry(), existing, false, false, nil, true, f.MaxFilenameLength); err != nil {
t.Fatalf("create with existing: %v", err)
}
withExisting := store.findEntryCallCount("/test/obj")
store2 := newRenameTestStore()
store2.entries["/test/obj"] = existing.ShallowClone()
f2 := newRenameTestFiler(store2)
if err := f2.CreateEntry(context.Background(), newEntry(), nil, false, false, nil, true, f2.MaxFilenameLength); err != nil {
t.Fatalf("create with nil: %v", err)
}
withNil := store2.findEntryCallCount("/test/obj")
if withNil != withExisting+1 {
t.Fatalf("providing existing should save one path lookup: existing=%d nil=%d", withExisting, withNil)
}
}
@@ -1,68 +0,0 @@
package weed_server
import (
"context"
"sync"
"sync/atomic"
"testing"
"time"
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
"github.com/seaweedfs/seaweedfs/weed/util"
)
// Concurrent OExcl creates for the same path must yield exactly one winner. The
// filer's CreateEntry is a FindEntry-then-Insert; without the per-path lock both
// racers observe "not found" and both insert. The exclusive entry lock makes the
// check-then-act atomic so the losers see ErrEntryAlreadyExists.
func TestCreateEntryOExclSerialized(t *testing.T) {
store := newRenameTestStore()
store.findDelay = 5 * time.Millisecond
f := newRenameTestFiler(store)
f.DirBucketsPath = "/buckets"
fs := &FilerServer{
filer: f,
option: &FilerOption{},
entryLockTable: util.NewLockTable[util.FullPath](),
}
const racers = 8
var success, alreadyExists, unexpected int32
var wg sync.WaitGroup
start := make(chan struct{})
for i := 0; i < racers; i++ {
wg.Add(1)
go func() {
defer wg.Done()
<-start
resp, err := fs.CreateEntry(context.Background(), &filer_pb.CreateEntryRequest{
Directory: "/test",
OExcl: true,
SkipCheckParentDirectory: true,
Entry: &filer_pb.Entry{
Name: "obj",
Attributes: &filer_pb.FuseAttributes{Mtime: 1700000000, FileMode: 0644, Inode: 1},
},
})
switch {
case err != nil:
atomic.AddInt32(&unexpected, 1)
case resp.Error == "":
atomic.AddInt32(&success, 1)
case resp.ErrorCode == filer_pb.FilerError_ENTRY_ALREADY_EXISTS:
atomic.AddInt32(&alreadyExists, 1)
default:
atomic.AddInt32(&unexpected, 1)
}
}()
}
close(start)
wg.Wait()
// Exactly one winner; every loser fails with ENTRY_ALREADY_EXISTS and nothing
// else, so an unrelated failure can't masquerade as a passing test.
if success != 1 || alreadyExists != racers-1 || unexpected != 0 {
t.Fatalf("winners=%d already_exists=%d unexpected=%d (racers=%d)", success, alreadyExists, unexpected, racers)
}
}
@@ -1,720 +0,0 @@
package weed_server
import (
"context"
"net"
"strconv"
"testing"
"time"
"github.com/seaweedfs/seaweedfs/weed/cluster/lock_manager"
"github.com/seaweedfs/seaweedfs/weed/filer"
"github.com/seaweedfs/seaweedfs/weed/pb"
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
"github.com/seaweedfs/seaweedfs/weed/s3api/s3_constants"
"github.com/seaweedfs/seaweedfs/weed/util"
"google.golang.org/grpc"
"google.golang.org/grpc/credentials/insecure"
)
func newTxnTestServer(seed map[string]*filer.Entry) (*FilerServer, *renameTestStore) {
store := newRenameTestStore()
for path, entry := range seed {
entry.FullPath = util.FullPath(path)
store.entries[path] = entry
}
f := newRenameTestFiler(store)
f.DirBucketsPath = "/buckets"
fs := &FilerServer{filer: f, option: &FilerOption{}, entryLockTable: util.NewLockTable[util.FullPath]()}
return fs, store
}
// A versioned delete is a multi-entry object operation: drop the null version,
// write a delete marker, and flip the latest pointer. ObjectTransaction applies
// all three atomically under one lock keyed on the object path.
func TestObjectTransactionMultiEntry(t *testing.T) {
now := time.Unix(1700000000, 0)
fs, store := newTxnTestServer(map[string]*filer.Entry{
"/buckets/b/obj": {
Attr: filer.Attr{Inode: 1, Mtime: now, Crtime: now, Mode: 0644},
Extended: map[string][]byte{s3_constants.ExtETagKey: []byte("abc")},
},
"/buckets/b/obj/.versions": {
Attr: filer.Attr{Inode: 2, Mtime: now, Crtime: now, Mode: 0755 | (1 << 31)},
Extended: map[string][]byte{"latest": []byte("v1")},
},
})
req := &filer_pb.ObjectTransactionRequest{
LockKey: "/buckets/b/obj",
Mutations: []*filer_pb.ObjectMutation{
{Type: filer_pb.ObjectMutation_DELETE, Directory: "/buckets/b", Name: "obj"},
{Type: filer_pb.ObjectMutation_PUT, Directory: "/buckets/b/obj/.versions", Entry: &filer_pb.Entry{
Name: "marker",
Attributes: &filer_pb.FuseAttributes{Mtime: now.Unix(), FileMode: 0644, Inode: 3},
Extended: map[string][]byte{"isDeleteMarker": []byte("true")},
}},
{Type: filer_pb.ObjectMutation_PATCH_EXTENDED, Directory: "/buckets/b/obj", Name: ".versions",
SetExtended: map[string][]byte{"latest": []byte("marker")},
DeleteExtended: nil},
},
}
resp, err := fs.ObjectTransaction(context.Background(), req)
if err != nil {
t.Fatalf("unexpected err: %v", err)
}
if resp.Error != "" {
t.Fatalf("unexpected response error: %q", resp.Error)
}
if _, ok := store.entries["/buckets/b/obj"]; ok {
t.Errorf("null version should be deleted")
}
if _, ok := store.entries["/buckets/b/obj/.versions/marker"]; !ok {
t.Errorf("delete marker should be created")
}
if got := string(store.entries["/buckets/b/obj/.versions"].Extended["latest"]); got != "marker" {
t.Errorf("latest pointer = %q, want marker", got)
}
}
// A PATCH_EXTENDED mutation emits a metadata event (so the change replicates and
// subscribers see it), carrying both the prior and updated state in the diff.
func TestObjectTransactionPatchNotifies(t *testing.T) {
queue := &captureQueue{}
swapNotificationQueue(t, queue)
now := time.Unix(1700000000, 0)
fs, _ := newTxnTestServer(map[string]*filer.Entry{
"/buckets/b/obj/.versions": {
Attr: filer.Attr{Inode: 2, Mtime: now, Crtime: now, Mode: 0755 | (1 << 31)},
Extended: map[string][]byte{"latest": []byte("v1")},
},
})
resp, err := fs.ObjectTransaction(context.Background(), &filer_pb.ObjectTransactionRequest{
LockKey: "/buckets/b/obj",
Mutations: []*filer_pb.ObjectMutation{
{Type: filer_pb.ObjectMutation_PATCH_EXTENDED, Directory: "/buckets/b/obj", Name: ".versions",
SetExtended: map[string][]byte{"latest": []byte("v2")}},
},
})
if err != nil || resp.Error != "" {
t.Fatalf("txn failed: err=%v resp=%q", err, resp.Error)
}
events := queue.snapshot()
if len(events) != 1 {
t.Fatalf("expected 1 metadata event from PATCH_EXTENDED, got %d", len(events))
}
ev := events[0].notification
if ev.NewEntry == nil || string(ev.NewEntry.Extended["latest"]) != "v2" {
t.Fatalf("event new entry latest = %q, want v2", ev.GetNewEntry().GetExtended()["latest"])
}
if ev.OldEntry == nil || string(ev.OldEntry.Extended["latest"]) != "v1" {
t.Fatalf("event old entry latest = %q, want v1 (clone must preserve prior state)", ev.GetOldEntry().GetExtended()["latest"])
}
}
// PATCH_EXTENDED with set_content replaces Entry.Content while merging Extended
// and preserving the rest; without set_content, Content is left untouched.
func TestObjectTransactionPatchContent(t *testing.T) {
now := time.Unix(1700000000, 0)
fs, store := newTxnTestServer(map[string]*filer.Entry{
"/buckets/b": {
Attr: filer.Attr{Inode: 1, Mtime: now, Crtime: now, Mode: 0755 | (1 << 31)},
Extended: map[string][]byte{"versioning": []byte("Enabled")},
Content: []byte("old-content"),
},
})
// set_content replaces Content and merges an Extended key, preserving the
// existing versioning key.
resp, err := fs.ObjectTransaction(context.Background(), &filer_pb.ObjectTransactionRequest{
LockKey: "/buckets/b",
Mutations: []*filer_pb.ObjectMutation{
{Type: filer_pb.ObjectMutation_PATCH_EXTENDED, Directory: "/buckets", Name: "b",
SetContent: true, Content: []byte("encryption-blob"),
SetExtended: map[string][]byte{"cors": []byte("yes")}},
},
})
if err != nil || resp.Error != "" {
t.Fatalf("patch set_content failed: err=%v resp=%q", err, resp.Error)
}
e := store.entries["/buckets/b"]
if string(e.Content) != "encryption-blob" {
t.Fatalf("content = %q, want encryption-blob", e.Content)
}
if string(e.Extended["versioning"]) != "Enabled" || string(e.Extended["cors"]) != "yes" {
t.Fatalf("extended not merged: %v", e.Extended)
}
// A PATCH without set_content must not disturb Content.
resp, err = fs.ObjectTransaction(context.Background(), &filer_pb.ObjectTransactionRequest{
LockKey: "/buckets/b",
Mutations: []*filer_pb.ObjectMutation{
{Type: filer_pb.ObjectMutation_PATCH_EXTENDED, Directory: "/buckets", Name: "b",
SetExtended: map[string][]byte{"versioning": []byte("Suspended")}},
},
})
if err != nil || resp.Error != "" {
t.Fatalf("patch extended-only failed: err=%v resp=%q", err, resp.Error)
}
e = store.entries["/buckets/b"]
if string(e.Content) != "encryption-blob" {
t.Fatalf("content clobbered by extended-only patch: %q", e.Content)
}
if string(e.Extended["versioning"]) != "Suspended" {
t.Fatalf("versioning = %q, want Suspended", e.Extended["versioning"])
}
if e.FileSize != 0 {
t.Fatalf("directory FileSize must stay 0, got %d", e.FileSize)
}
// For a file, set_content syncs FileSize to the new content length, even when
// the content shrinks.
store.entries["/file"] = &filer.Entry{
FullPath: "/file",
Attr: filer.Attr{Inode: 9, Mtime: now, Crtime: now, Mode: 0644, FileSize: 100},
Content: []byte("xxxxxxxxxxxxxxx"),
}
resp, err = fs.ObjectTransaction(context.Background(), &filer_pb.ObjectTransactionRequest{
LockKey: "/file",
Mutations: []*filer_pb.ObjectMutation{
{Type: filer_pb.ObjectMutation_PATCH_EXTENDED, Directory: "/", Name: "file",
SetContent: true, Content: []byte("short")},
},
})
if err != nil || resp.Error != "" {
t.Fatalf("file patch failed: err=%v resp=%q", err, resp.Error)
}
if f := store.entries["/file"]; string(f.Content) != "short" || f.FileSize != uint64(len("short")) {
t.Fatalf("file content=%q FileSize=%d, want short/5", f.Content, f.FileSize)
}
}
// A failing precondition aborts before any mutation is applied.
func TestObjectTransactionPreconditionAborts(t *testing.T) {
now := time.Unix(1700000000, 0)
fs, store := newTxnTestServer(map[string]*filer.Entry{
"/buckets/b/obj": {
Attr: filer.Attr{Inode: 1, Mtime: now, Crtime: now, Mode: 0644},
Extended: map[string][]byte{s3_constants.ExtETagKey: []byte("abc")},
},
})
req := &filer_pb.ObjectTransactionRequest{
LockKey: "/buckets/b/obj",
Condition: &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{
{Kind: filer_pb.WriteCondition_IF_ETAG_MATCH, Etags: []string{`"zzz"`}},
}},
Mutations: []*filer_pb.ObjectMutation{
{Type: filer_pb.ObjectMutation_DELETE, Directory: "/buckets/b", Name: "obj"},
},
}
resp, err := fs.ObjectTransaction(context.Background(), req)
if err != nil {
t.Fatalf("unexpected err: %v", err)
}
if resp.ErrorCode != filer_pb.FilerError_PRECONDITION_FAILED {
t.Fatalf("want PRECONDITION_FAILED, got %v (%q)", resp.ErrorCode, resp.Error)
}
if _, ok := store.entries["/buckets/b/obj"]; !ok {
t.Errorf("object must survive a failed precondition")
}
}
// Deleting the latest version and recomputing re-points the pointer at the new
// highest-named remaining version; the scan runs under the transaction lock.
func TestObjectTransactionRecomputeLatest(t *testing.T) {
now := time.Unix(1700000000, 0)
ver := func(id string) *filer.Entry {
return &filer.Entry{
Attr: filer.Attr{Inode: 10, Mtime: now, Crtime: now, Mode: 0644},
Extended: map[string][]byte{"vid": []byte(id), "etag": []byte("etag-" + id)},
}
}
fs, store := newTxnTestServer(map[string]*filer.Entry{
"/buckets/b/obj/.versions": {
Attr: filer.Attr{Inode: 2, Mtime: now, Crtime: now, Mode: 0755 | (1 << 31)},
Extended: map[string][]byte{
"latestVid": []byte("v3"), "latestEtag": []byte("etag-v3"), "latestName": []byte("v3.ver"),
},
},
"/buckets/b/obj/.versions/v1.ver": ver("v1"),
"/buckets/b/obj/.versions/v2.ver": ver("v2"),
"/buckets/b/obj/.versions/v3.ver": ver("v3"),
})
recompute := func() *filer_pb.ObjectMutation {
return &filer_pb.ObjectMutation{
Type: filer_pb.ObjectMutation_RECOMPUTE_LATEST, Directory: "/buckets/b/obj", Name: ".versions",
Recompute: &filer_pb.Recompute{
ScanDir: "/buckets/b/obj/.versions",
Descending: true,
CopyExtended: map[string]string{"latestVid": "vid", "latestEtag": "etag"},
NameToKey: "latestName",
},
}
}
// Delete the latest (v3); recompute should pick v2.
resp, err := fs.ObjectTransaction(context.Background(), &filer_pb.ObjectTransactionRequest{
LockKey: "/buckets/b/obj",
Mutations: []*filer_pb.ObjectMutation{
{Type: filer_pb.ObjectMutation_DELETE, Directory: "/buckets/b/obj/.versions", Name: "v3.ver"},
recompute(),
},
})
if err != nil || resp.Error != "" {
t.Fatalf("txn failed: err=%v resp=%q", err, resp.Error)
}
ptr := store.entries["/buckets/b/obj/.versions"].Extended
if string(ptr["latestVid"]) != "v2" || string(ptr["latestEtag"]) != "etag-v2" || string(ptr["latestName"]) != "v2.ver" {
t.Fatalf("after deleting v3, pointer = vid:%s etag:%s name:%s; want v2",
ptr["latestVid"], ptr["latestEtag"], ptr["latestName"])
}
// Delete the remaining versions; recompute on an empty dir clears the pointer.
resp, err = fs.ObjectTransaction(context.Background(), &filer_pb.ObjectTransactionRequest{
LockKey: "/buckets/b/obj",
Mutations: []*filer_pb.ObjectMutation{
{Type: filer_pb.ObjectMutation_DELETE, Directory: "/buckets/b/obj/.versions", Name: "v2.ver"},
{Type: filer_pb.ObjectMutation_DELETE, Directory: "/buckets/b/obj/.versions", Name: "v1.ver"},
recompute(),
},
})
if err != nil || resp.Error != "" {
t.Fatalf("txn failed: err=%v resp=%q", err, resp.Error)
}
ptr = store.entries["/buckets/b/obj/.versions"].Extended
for _, k := range []string{"latestVid", "latestEtag", "latestName"} {
if _, ok := ptr[k]; ok {
t.Errorf("pointer key %q should be cleared when no versions remain", k)
}
}
}
// With descending=false the lowest-named child is chosen (the listing is capped
// at one entry).
func TestObjectTransactionRecomputeAscending(t *testing.T) {
now := time.Unix(1700000000, 0)
ver := func(id string) *filer.Entry {
return &filer.Entry{Attr: filer.Attr{Inode: 10, Mtime: now, Crtime: now, Mode: 0644}, Extended: map[string][]byte{"vid": []byte(id)}}
}
fs, store := newTxnTestServer(map[string]*filer.Entry{
"/buckets/b/obj/.versions": {Attr: filer.Attr{Inode: 2, Mtime: now, Crtime: now, Mode: 0755 | (1 << 31)}, Extended: map[string][]byte{}},
"/buckets/b/obj/.versions/v1.ver": ver("v1"),
"/buckets/b/obj/.versions/v2.ver": ver("v2"),
})
resp, err := fs.ObjectTransaction(context.Background(), &filer_pb.ObjectTransactionRequest{
LockKey: "/buckets/b/obj",
Mutations: []*filer_pb.ObjectMutation{
{Type: filer_pb.ObjectMutation_RECOMPUTE_LATEST, Directory: "/buckets/b/obj", Name: ".versions",
Recompute: &filer_pb.Recompute{
ScanDir: "/buckets/b/obj/.versions",
Descending: false,
CopyExtended: map[string]string{"latestVid": "vid"},
}},
},
})
if err != nil || resp.Error != "" {
t.Fatalf("txn failed: err=%v resp=%q", err, resp.Error)
}
if got := string(store.entries["/buckets/b/obj/.versions"].Extended["latestVid"]); got != "v1" {
t.Fatalf("ascending recompute latestVid = %q, want v1 (lowest)", got)
}
}
// A batch applies each transaction independently: one failed precondition does
// not abort the others, matching S3 multi-object delete semantics.
func TestObjectTransactionBatchIndependent(t *testing.T) {
now := time.Unix(1700000000, 0)
obj := func(inode uint64) *filer.Entry {
return &filer.Entry{
Attr: filer.Attr{Inode: inode, Mtime: now, Crtime: now, Mode: 0644},
Extended: map[string][]byte{s3_constants.ExtETagKey: []byte("abc")},
}
}
fs, store := newTxnTestServer(map[string]*filer.Entry{
"/buckets/b/a": obj(1),
"/buckets/b/c": obj(3),
})
del := func(name string, cond *filer_pb.WriteCondition) *filer_pb.ObjectTransactionRequest {
return &filer_pb.ObjectTransactionRequest{
LockKey: "/buckets/b/" + name,
Condition: cond,
Mutations: []*filer_pb.ObjectMutation{
{Type: filer_pb.ObjectMutation_DELETE, Directory: "/buckets/b", Name: name},
},
}
}
resp, err := fs.ObjectTransactionBatch(context.Background(), &filer_pb.ObjectTransactionBatchRequest{
Transactions: []*filer_pb.ObjectTransactionRequest{
del("a", nil),
del("c", &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{
{Kind: filer_pb.WriteCondition_IF_ETAG_MATCH, Etags: []string{`"zzz"`}},
}}),
},
})
if err != nil {
t.Fatalf("unexpected err: %v", err)
}
if len(resp.Responses) != 2 {
t.Fatalf("want 2 responses, got %d", len(resp.Responses))
}
if resp.Responses[0].Error != "" {
t.Errorf("delete a should succeed: %q", resp.Responses[0].Error)
}
if resp.Responses[1].ErrorCode != filer_pb.FilerError_PRECONDITION_FAILED {
t.Errorf("delete c should fail precondition, got %v", resp.Responses[1].ErrorCode)
}
if _, ok := store.entries["/buckets/b/a"]; ok {
t.Errorf("a should be deleted")
}
if _, ok := store.entries["/buckets/b/c"]; !ok {
t.Errorf("c should survive its failed precondition")
}
}
// A nil transaction in a batch yields an error response in its slot rather than
// panicking, keeping responses parallel to the requests.
func TestObjectTransactionBatchNilTransaction(t *testing.T) {
now := time.Unix(1700000000, 0)
fs, store := newTxnTestServer(map[string]*filer.Entry{
"/buckets/b/a": {Attr: filer.Attr{Inode: 1, Mtime: now, Crtime: now, Mode: 0644}},
})
resp, err := fs.ObjectTransactionBatch(context.Background(), &filer_pb.ObjectTransactionBatchRequest{
Transactions: []*filer_pb.ObjectTransactionRequest{
nil,
{LockKey: "/buckets/b/a", Mutations: []*filer_pb.ObjectMutation{
{Type: filer_pb.ObjectMutation_DELETE, Directory: "/buckets/b", Name: "a"},
}},
},
})
if err != nil {
t.Fatalf("unexpected err: %v", err)
}
if len(resp.Responses) != 2 {
t.Fatalf("want 2 responses (parallel to requests), got %d", len(resp.Responses))
}
if resp.Responses[0].Error == "" {
t.Errorf("nil transaction should produce an error response")
}
if resp.Responses[1].Error != "" {
t.Errorf("valid transaction should succeed: %q", resp.Responses[1].Error)
}
if _, ok := store.entries["/buckets/b/a"]; ok {
t.Errorf("a should be deleted by the valid transaction")
}
}
// DELETE and PATCH of an absent entry are no-ops, so a replayed transaction
// does not error.
func TestObjectTransactionIdempotentNoops(t *testing.T) {
fs, _ := newTxnTestServer(nil)
req := &filer_pb.ObjectTransactionRequest{
LockKey: "/buckets/b/obj",
Mutations: []*filer_pb.ObjectMutation{
{Type: filer_pb.ObjectMutation_DELETE, Directory: "/buckets/b", Name: "obj"},
{Type: filer_pb.ObjectMutation_PATCH_EXTENDED, Directory: "/buckets/b/obj", Name: ".versions",
SetExtended: map[string][]byte{"latest": []byte("x")}},
},
}
resp, err := fs.ObjectTransaction(context.Background(), req)
if err != nil {
t.Fatalf("unexpected err: %v", err)
}
if resp.Error != "" {
t.Fatalf("no-op mutations should not error: %q", resp.Error)
}
}
// RECOMPUTE_LATEST copies the chosen child's size/mtime to the pointer and
// stamps the demote key on the prior latest when the pointer moves.
func TestObjectTransactionRecomputeDemoteAndAttrs(t *testing.T) {
t0 := time.Unix(1700000000, 0)
t1 := time.Unix(1700000100, 0)
mk := func(inode uint64, mt time.Time, size uint64, id string) *filer.Entry {
return &filer.Entry{
Attr: filer.Attr{Inode: inode, Mtime: mt, Crtime: mt, Mode: 0644, FileSize: size},
Extended: map[string][]byte{"vid": []byte(id)},
}
}
fs, store := newTxnTestServer(map[string]*filer.Entry{
"/buckets/b/obj/.versions": {
Attr: filer.Attr{Inode: 2, Mtime: t0, Crtime: t0, Mode: 0755 | (1 << 31)},
Extended: map[string][]byte{"latestName": []byte("v1.ver"), "latestVid": []byte("v1")},
},
"/buckets/b/obj/.versions/v1.ver": mk(10, t0, 100, "v1"),
"/buckets/b/obj/.versions/v2.ver": mk(11, t1, 250, "v2"),
})
resp, err := fs.ObjectTransaction(context.Background(), &filer_pb.ObjectTransactionRequest{
LockKey: "/buckets/b/obj",
Mutations: []*filer_pb.ObjectMutation{{
Type: filer_pb.ObjectMutation_RECOMPUTE_LATEST, Directory: "/buckets/b/obj", Name: ".versions",
Recompute: &filer_pb.Recompute{
ScanDir: "/buckets/b/obj/.versions",
Descending: true,
CopyExtended: map[string]string{"latestVid": "vid"},
NameToKey: "latestName",
SizeToKey: "latestSize",
MtimeToKey: "latestMtime",
DemoteKey: "noncurrentSince",
DemoteValue: []byte("999"),
},
}},
})
if err != nil || resp.Error != "" {
t.Fatalf("txn failed: err=%v resp=%q", err, resp.Error)
}
ptr := store.entries["/buckets/b/obj/.versions"].Extended
if string(ptr["latestName"]) != "v2.ver" || string(ptr["latestVid"]) != "v2" {
t.Fatalf("pointer not moved to v2: name=%s vid=%s", ptr["latestName"], ptr["latestVid"])
}
if string(ptr["latestSize"]) != "250" {
t.Errorf("latestSize = %s, want 250", ptr["latestSize"])
}
if want := strconv.FormatInt(t1.Unix(), 10); string(ptr["latestMtime"]) != want {
t.Errorf("latestMtime = %s, want %s", ptr["latestMtime"], want)
}
if got := store.entries["/buckets/b/obj/.versions/v1.ver"].Extended["noncurrentSince"]; string(got) != "999" {
t.Errorf("prior latest v1.ver noncurrentSince = %q, want 999", got)
}
if _, ok := store.entries["/buckets/b/obj/.versions/v2.ver"].Extended["noncurrentSince"]; ok {
t.Errorf("new latest v2.ver should not be demoted")
}
}
// A version-specific delete locks the object (condition_key checks WORM on the
// version), recomputes the pointer excluding the version (repoint-before-delete),
// then deletes it. A legal-hold guard blocks the delete and preserves the entry.
func TestObjectTransactionVersionDeleteWithWorm(t *testing.T) {
now := time.Unix(1700000000, 0)
ver := func(inode uint64, ext map[string][]byte) *filer.Entry {
return &filer.Entry{Attr: filer.Attr{Inode: inode, Mtime: now, Crtime: now, Mode: 0644}, Extended: ext}
}
seed := func(latestLocked bool) map[string]*filer.Entry {
vcExt := map[string][]byte{"vid": []byte("v3")}
if latestLocked {
vcExt["legalhold"] = []byte("ON")
}
return map[string]*filer.Entry{
"/buckets/b/obj/.versions": {
Attr: filer.Attr{Inode: 2, Mtime: now, Crtime: now, Mode: 0755 | (1 << 31)},
Extended: map[string][]byte{"latestName": []byte("v_c"), "latestVid": []byte("v3")},
},
"/buckets/b/obj/.versions/v_a": ver(10, map[string][]byte{"vid": []byte("v1")}),
"/buckets/b/obj/.versions/v_b": ver(11, map[string][]byte{"vid": []byte("v2")}),
"/buckets/b/obj/.versions/v_c": ver(12, vcExt),
}
}
mkReq := func() *filer_pb.ObjectTransactionRequest {
return &filer_pb.ObjectTransactionRequest{
LockKey: "/buckets/b/obj",
ConditionKey: "/buckets/b/obj/.versions/v_c",
Condition: &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{
{Kind: filer_pb.WriteCondition_IF_EXTENDED_NOT_EQUAL, ExtKey: "legalhold", ExtValue: "ON"},
}},
Mutations: []*filer_pb.ObjectMutation{
{Type: filer_pb.ObjectMutation_RECOMPUTE_LATEST, Directory: "/buckets/b/obj", Name: ".versions",
Recompute: &filer_pb.Recompute{ScanDir: "/buckets/b/obj/.versions", Descending: true, ExcludeName: "v_c",
NameToKey: "latestName", CopyExtended: map[string]string{"latestVid": "vid"}}},
{Type: filer_pb.ObjectMutation_DELETE, Directory: "/buckets/b/obj/.versions", Name: "v_c"},
},
}
}
// Legal hold ON: the WORM guard blocks; version and pointer untouched.
fs, store := newTxnTestServer(seed(true))
resp, err := fs.ObjectTransaction(context.Background(), mkReq())
if err != nil {
t.Fatalf("err: %v", err)
}
if resp.ErrorCode != filer_pb.FilerError_PRECONDITION_FAILED {
t.Fatalf("locked version delete should fail precondition, got code=%v err=%q", resp.ErrorCode, resp.Error)
}
if _, ok := store.entries["/buckets/b/obj/.versions/v_c"]; !ok {
t.Errorf("locked version must not be deleted")
}
if got := string(store.entries["/buckets/b/obj/.versions"].Extended["latestName"]); got != "v_c" {
t.Errorf("pointer must be unchanged when delete is blocked, got %s", got)
}
// No legal hold: pointer recomputes to v_b (excluding v_c), then v_c is deleted.
fs, store = newTxnTestServer(seed(false))
resp, err = fs.ObjectTransaction(context.Background(), mkReq())
if err != nil || resp.Error != "" {
t.Fatalf("unlocked delete failed: err=%v resp=%q", err, resp.Error)
}
if _, ok := store.entries["/buckets/b/obj/.versions/v_c"]; ok {
t.Errorf("unlocked version should be deleted")
}
ptr := store.entries["/buckets/b/obj/.versions"].Extended
if string(ptr["latestName"]) != "v_b" || string(ptr["latestVid"]) != "v2" {
t.Errorf("pointer should recompute to v_b/v2, got name=%s vid=%s", ptr["latestName"], ptr["latestVid"])
}
}
// PATCH_EXTENDED with touch_mtime bumps the entry's Mtime (a metadata-replace
// copy) while merging Extended.
func TestObjectTransactionPatchTouchMtime(t *testing.T) {
old := time.Unix(1600000000, 0)
fs, store := newTxnTestServer(map[string]*filer.Entry{
"/buckets/b/obj": {
FullPath: "/buckets/b/obj",
Attr: filer.Attr{Inode: 1, Mtime: old, Crtime: old, Mode: 0644},
Extended: map[string][]byte{"X-Amz-Meta-old": []byte("1")},
},
})
resp, err := fs.ObjectTransaction(context.Background(), &filer_pb.ObjectTransactionRequest{
LockKey: "/buckets/b/obj",
Mutations: []*filer_pb.ObjectMutation{{
Type: filer_pb.ObjectMutation_PATCH_EXTENDED, Directory: "/buckets/b", Name: "obj",
SetExtended: map[string][]byte{"X-Amz-Meta-new": []byte("2")},
DeleteExtended: []string{"X-Amz-Meta-old"},
TouchMtime: true,
}},
})
if err != nil || resp.Error != "" {
t.Fatalf("patch failed: err=%v resp=%q", err, resp.Error)
}
e := store.entries["/buckets/b/obj"]
if !e.Attr.Mtime.After(old) {
t.Errorf("touch_mtime should bump Mtime past %v, got %v", old, e.Attr.Mtime)
}
if _, ok := e.Extended["X-Amz-Meta-old"]; ok {
t.Errorf("old meta should be deleted")
}
if string(e.Extended["X-Amz-Meta-new"]) != "2" {
t.Errorf("new meta not set: %v", e.Extended)
}
}
// withRing attaches a Dlm whose ring contains exactly the given servers and sets
// the filer's own host, so route_key resolution in ObjectTransaction is decided
// by who owns the single-server ring.
func withRing(fs *FilerServer, self pb.ServerAddress, servers ...pb.ServerAddress) {
dlm := lock_manager.NewDistributedLockManager(self)
dlm.LockRing.SetSnapshot(servers, 1)
fs.filer.Dlm = dlm
fs.option.Host = self
}
// When this filer owns route_key, the transaction applies locally rather than
// forwarding to itself.
func TestObjectTransactionRouteKeyOwnerAppliesLocally(t *testing.T) {
self := pb.ServerAddress("localhost:1")
fs, store := newTxnTestServer(map[string]*filer.Entry{
"/buckets/b/obj": {FullPath: "/buckets/b/obj", Attr: filer.Attr{Inode: 1, Mode: 0644}},
})
withRing(fs, self, self)
resp, err := fs.ObjectTransaction(context.Background(), &filer_pb.ObjectTransactionRequest{
LockKey: "/buckets/b/obj",
RouteKey: "s3.object.write:/buckets/b/obj",
Mutations: []*filer_pb.ObjectMutation{{
Type: filer_pb.ObjectMutation_PATCH_EXTENDED, Directory: "/buckets/b", Name: "obj",
SetExtended: map[string][]byte{"X-Amz-Meta-k": []byte("v")},
}},
})
if err != nil || resp.Error != "" {
t.Fatalf("txn failed: err=%v resp=%q", err, resp.Error)
}
if string(store.entries["/buckets/b/obj"].Extended["X-Amz-Meta-k"]) != "v" {
t.Errorf("mutation should have applied locally: %v", store.entries["/buckets/b/obj"].Extended)
}
}
// A forwarded transaction (is_moved) applies locally even when the ring names a
// different owner: is_moved bounds forwarding to a single hop, so two filers that
// disagree on the owner during a ring change cannot loop. If is_moved were
// ignored, this would attempt to dial the bogus owner instead of applying.
func TestObjectTransactionIsMovedSkipsForward(t *testing.T) {
self := pb.ServerAddress("localhost:1")
other := pb.ServerAddress("localhost:2")
fs, store := newTxnTestServer(map[string]*filer.Entry{
"/buckets/b/obj": {FullPath: "/buckets/b/obj", Attr: filer.Attr{Inode: 1, Mode: 0644}},
})
withRing(fs, self, other) // ring owner is "other", not self
resp, err := fs.ObjectTransaction(context.Background(), &filer_pb.ObjectTransactionRequest{
LockKey: "/buckets/b/obj",
RouteKey: "s3.object.write:/buckets/b/obj",
IsMoved: true,
Mutations: []*filer_pb.ObjectMutation{{
Type: filer_pb.ObjectMutation_PATCH_EXTENDED, Directory: "/buckets/b", Name: "obj",
SetExtended: map[string][]byte{"X-Amz-Meta-k": []byte("v")},
}},
})
if err != nil || resp.Error != "" {
t.Fatalf("txn failed: err=%v resp=%q", err, resp.Error)
}
if string(store.entries["/buckets/b/obj"].Extended["X-Amz-Meta-k"]) != "v" {
t.Errorf("forwarded txn should apply locally: %v", store.entries["/buckets/b/obj"].Extended)
}
}
// End-to-end forward hop: a non-owner filer dials the ring owner and the owner
// applies the transaction. The owner's own ring points back at the (bogus)
// sender, so it would re-forward and fail to dial unless is_moved is set on the
// forwarded request — making this also assert that one-hop bound over the wire.
func TestObjectTransactionForwardsToOwner(t *testing.T) {
owner, ownerStore := newTxnTestServer(map[string]*filer.Entry{
"/buckets/b/obj": {FullPath: "/buckets/b/obj", Attr: filer.Attr{Inode: 1, Mode: 0644}},
})
lis, err := net.Listen("tcp", "127.0.0.1:0")
if err != nil {
t.Fatalf("listen: %v", err)
}
// Pin the grpc port to the real listener (ToGrpcAddress otherwise adds the
// +10000 convention, which dials nothing).
port := lis.Addr().(*net.TCPAddr).Port
ownerAddr := pb.NewServerAddressWithGrpcPort(lis.Addr().String(), port)
sender := pb.ServerAddress("127.0.0.1:1") // bogus: nothing listens here
// owner's ring points back at the sender; only is_moved keeps it from
// re-forwarding to (and failing to dial) that bogus address.
withRing(owner, ownerAddr, sender)
owner.grpcDialOption = grpc.WithTransportCredentials(insecure.NewCredentials())
srv := grpc.NewServer()
filer_pb.RegisterSeaweedFilerServer(srv, owner)
go srv.Serve(lis)
t.Cleanup(srv.Stop)
self, selfStore := newTxnTestServer(nil)
withRing(self, sender, ownerAddr) // ring owner is the real owner; self forwards
self.grpcDialOption = grpc.WithTransportCredentials(insecure.NewCredentials())
ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second)
defer cancel()
resp, err := self.ObjectTransaction(ctx, &filer_pb.ObjectTransactionRequest{
LockKey: "/buckets/b/obj",
RouteKey: "s3.object.write:/buckets/b/obj",
Mutations: []*filer_pb.ObjectMutation{{
Type: filer_pb.ObjectMutation_PATCH_EXTENDED, Directory: "/buckets/b", Name: "obj",
SetExtended: map[string][]byte{"X-Amz-Meta-k": []byte("v")},
}},
})
if err != nil || resp.Error != "" {
t.Fatalf("forwarded txn failed: err=%v resp=%q", err, resp.Error)
}
if string(ownerStore.entries["/buckets/b/obj"].Extended["X-Amz-Meta-k"]) != "v" {
t.Errorf("owner should have applied the forwarded mutation: %v", ownerStore.entries["/buckets/b/obj"].Extended)
}
if _, ok := selfStore.entries["/buckets/b/obj"]; ok {
t.Errorf("non-owner must forward, not apply locally")
}
}
-248
View File
@@ -1,248 +0,0 @@
package weed_server
import (
"context"
"fmt"
"time"
"github.com/seaweedfs/seaweedfs/weed/filer/posixlock"
"github.com/seaweedfs/seaweedfs/weed/glog"
"github.com/seaweedfs/seaweedfs/weed/pb"
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
)
const (
// posixLockSessionTTL is how long a mount's lease survives without a
// keepalive before its locks are reaped; posixLockSweepInterval is how often
// each filer checks. The mount renews well within the TTL.
posixLockSessionTTL = 15 * time.Second
posixLockSweepInterval = 5 * time.Second
// posixCoolingProbeTimeout bounds the dual-read probe to the prior owner so a
// slow peer can't stall a non-blocking lock call during the cooling window.
posixCoolingProbeTimeout = 2 * time.Second
// posixLockWarmup is how long after a (re)start the owner defers would-be
// grants while mounts re-assert their held locks, so it does not double-grant
// a lock its fresh, still-empty state has not yet rebuilt. Must exceed the
// mount keepalive interval and stay below the TTL.
posixLockWarmup = 10 * time.Second
)
// startPosixLockSweeper periodically reaps the locks of leased sessions (mounts)
// that stopped sending keepalives. Sessions that never renew are never reaped, so
// this is inert until mounts run with -posixLock.
func (fs *FilerServer) startPosixLockSweeper() {
fs.posixLockReadyAt.Store(time.Now().UnixNano())
fs.posixLockSweeperStop = make(chan struct{})
go func() {
ticker := time.NewTicker(posixLockSweepInterval)
defer ticker.Stop()
for {
select {
case <-fs.posixLockSweeperStop:
return
case <-ticker.C:
if reaped := fs.posixLocks.ReapExpired(posixLockSessionTTL); len(reaped) > 0 {
glog.V(2).Infof("posix lock: reaped %d expired session(s): %v", len(reaped), reaped)
}
}
}
}()
}
// PosixLock applies one advisory lock operation against the in-memory lock table
// of the inode's owner filer. The owner is resolved from req.Key on the same
// ring the gateway and DLM use; a non-owner filer forwards the request one hop
// (is_moved bounds it), so the owner's table stays the single authority even when
// the caller's ring view is stale. Blocking (SetLkw) is the caller's job — this
// is strictly non-blocking try/release/query.
func (fs *FilerServer) PosixLock(ctx context.Context, req *filer_pb.PosixLockRequest) (*filer_pb.PosixLockResponse, error) {
if req.Key == "" {
return &filer_pb.PosixLockResponse{}, fmt.Errorf("key is required")
}
if req.Lock == nil {
return &filer_pb.PosixLockResponse{}, fmt.Errorf("lock is required")
}
if !req.IsMoved && fs.filer.Dlm != nil {
if owner := fs.filer.Dlm.LockRing.GetPrimary(req.Key); owner != "" && owner != fs.option.Host {
forwarded := &filer_pb.PosixLockRequest{
Key: req.Key,
IsMoved: true,
Op: req.Op,
Lock: req.Lock,
Locks: req.Locks,
}
glog.V(4).InfofCtx(ctx, "PosixLock %s op=%v: forwarding to owner %s", req.Key, req.Op, owner)
var resp *filer_pb.PosixLockResponse
err := pb.WithFilerClient(false, 0, owner, fs.grpcDialOption, func(client filer_pb.SeaweedFilerClient) error {
var e error
resp, e = client.PosixLock(ctx, forwarded)
return e
})
if err != nil {
return &filer_pb.PosixLockResponse{}, err
}
return resp, nil
}
}
lk := posixlock.Range{
Start: req.Lock.GetStart(),
End: req.Lock.GetEnd(),
Type: req.Lock.GetType(),
Sid: req.Lock.GetSid(),
Owner: req.Lock.GetOwner(),
Pid: req.Lock.GetPid(),
IsFlock: req.Lock.GetIsFlock(),
}
resp := &filer_pb.PosixLockResponse{}
switch req.Op {
case filer_pb.PosixLockOp_TRY_LOCK:
// A fresh owner's state may be incomplete (post-restart warm-up, or a ring
// change whose previous owner can't be reached): report a known conflict,
// else defer the grant so the client retries rather than risk a
// double-grant. A deferred grant becomes EAGAIN for non-blocking SetLk and
// a retry for the blocking SetLkw poll — never a spurious grant.
if !req.CoolingProbe {
if c, deferGrant := fs.posixCoolingOrWarmup(ctx, req.Key, lk); c != nil {
resp.HasConflict = true
resp.Conflict = c
break
} else if deferGrant {
break
}
}
if c, granted := fs.posixLocks.TryLock(req.Key, lk); granted {
resp.Granted = true
} else {
resp.HasConflict = true
resp.Conflict = posixRangeToPb(c)
}
case filer_pb.PosixLockOp_UNLOCK:
fs.posixLocks.Unlock(req.Key, lk)
case filer_pb.PosixLockOp_GET_LK:
// A query is best-effort: report a known conflict but never defer.
if !req.CoolingProbe {
if c, _ := fs.posixCoolingOrWarmup(ctx, req.Key, lk); c != nil {
resp.HasConflict = true
resp.Conflict = c
break
}
}
if c, found := fs.posixLocks.GetLk(req.Key, lk); found {
resp.HasConflict = true
resp.Conflict = posixRangeToPb(c)
}
case filer_pb.PosixLockOp_RELEASE_POSIX_OWNER:
fs.posixLocks.ReleasePosixOwner(req.Key, lk.Sid, lk.Owner)
case filer_pb.PosixLockOp_RELEASE_FLOCK_OWNER:
fs.posixLocks.ReleaseFlockOwner(req.Key, lk.Sid, lk.Owner)
case filer_pb.PosixLockOp_KEEP_ALIVE:
// A re-assertion carries the mount's held locks on this key so the owner
// can rebuild its in-memory state after an ownership change or restart; a
// bare keepalive just renews the lease.
if len(req.Locks) > 0 {
held := make([]posixlock.Range, 0, len(req.Locks))
for _, l := range req.Locks {
held = append(held, posixRangeFromPb(l))
}
if conflicts := fs.posixLocks.Reassert(req.Key, lk.Sid, held); len(conflicts) > 0 {
glog.Warningf("posix reassert %s sid %d: %d lock(s) lost to another session: %+v", req.Key, lk.Sid, len(conflicts), conflicts)
}
} else {
fs.posixLocks.Renew(lk.Sid)
}
default:
return &filer_pb.PosixLockResponse{}, fmt.Errorf("unknown posix lock op %v", req.Op)
}
return resp, nil
}
// posixWarmingUp reports whether this filer is still within posixLockWarmup of
// when it began serving POSIX locks. The zero readyAt (e.g. in tests) is never
// warming up.
func (fs *FilerServer) posixWarmingUp() bool {
readyAt := fs.posixLockReadyAt.Load()
return readyAt != 0 && time.Since(time.Unix(0, readyAt)) < posixLockWarmup
}
// posixCoolingOrWarmup decides whether a fresh owner can trust its local state
// for key before granting. It returns a known blocking lock (conflict != nil),
// or asks the caller to defer the grant (deferGrant) when the state may be
// incomplete and no conflict can be confirmed. It returns (nil, false) when the
// owner is authoritative and should consult its local table.
//
// - Warm-up: just after a (re)start the owner is still rebuilding from
// re-assertions, so only a locally-visible conflict is trustworthy; a
// would-be grant is deferred.
// - Ring change (cooling window): the previous owner may still hold a lock this
// owner hasn't rebuilt. Ask it (marked cooling_probe so it answers locally
// without recursing), under a short deadline so a slow peer can't stall the
// non-blocking lock path. If it is unreachable — typically because it
// crashed, which caused the change — we cannot confirm, so we defer rather
// than risk a double-grant; re-assertion rebuilds this owner before the
// window ends.
func (fs *FilerServer) posixCoolingOrWarmup(ctx context.Context, key string, lk posixlock.Range) (conflict *filer_pb.PosixLockRange, deferGrant bool) {
if fs.posixWarmingUp() {
if c, found := fs.posixLocks.GetLk(key, lk); found {
return posixRangeToPb(c), false
}
return nil, true
}
if fs.filer.Dlm == nil {
return nil, false
}
prior := fs.filer.Dlm.LockRing.PriorOwner(key)
if prior == "" || prior == fs.option.Host {
return nil, false
}
probeCtx, cancel := context.WithTimeout(ctx, posixCoolingProbeTimeout)
defer cancel()
var resp *filer_pb.PosixLockResponse
err := pb.WithFilerClient(false, 0, prior, fs.grpcDialOption, func(client filer_pb.SeaweedFilerClient) error {
var e error
resp, e = client.PosixLock(probeCtx, &filer_pb.PosixLockRequest{
Key: key,
IsMoved: true,
CoolingProbe: true,
Op: filer_pb.PosixLockOp_GET_LK,
Lock: posixRangeToPb(lk),
})
return e
})
if err != nil {
// Cannot confirm — defer rather than risk a double-grant (the prior owner
// likely crashed; re-assertion rebuilds this owner before the window ends).
glog.V(2).InfofCtx(ctx, "posix cooling probe %s -> %s: %v (deferring)", key, prior, err)
return nil, true
}
if resp.GetHasConflict() {
return resp.GetConflict(), false
}
return nil, false
}
func posixRangeFromPb(l *filer_pb.PosixLockRange) posixlock.Range {
return posixlock.Range{
Start: l.GetStart(),
End: l.GetEnd(),
Type: l.GetType(),
Sid: l.GetSid(),
Owner: l.GetOwner(),
Pid: l.GetPid(),
IsFlock: l.GetIsFlock(),
}
}
func posixRangeToPb(r posixlock.Range) *filer_pb.PosixLockRange {
return &filer_pb.PosixLockRange{
Start: r.Start,
End: r.End,
Type: r.Type,
Sid: r.Sid,
Owner: r.Owner,
Pid: r.Pid,
IsFlock: r.IsFlock,
}
}
@@ -1,178 +0,0 @@
package weed_server
import (
"context"
"net"
"testing"
"time"
"github.com/seaweedfs/seaweedfs/weed/filer/posixlock"
"github.com/seaweedfs/seaweedfs/weed/pb"
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
"google.golang.org/grpc"
"google.golang.org/grpc/credentials/insecure"
)
func newPosixTestServer() *FilerServer {
fs, _ := newTxnTestServer(nil)
fs.posixLocks = posixlock.NewManager()
return fs
}
func pbLock(start, end uint64, typ uint32, sid, owner uint64, pid uint32, flock bool) *filer_pb.PosixLockRange {
return &filer_pb.PosixLockRange{Start: start, End: end, Type: typ, Sid: sid, Owner: owner, Pid: pid, IsFlock: flock}
}
func posixOp(t *testing.T, fs *FilerServer, op filer_pb.PosixLockOp, lk *filer_pb.PosixLockRange) *filer_pb.PosixLockResponse {
t.Helper()
resp, err := fs.PosixLock(context.Background(), &filer_pb.PosixLockRequest{Key: "s3.fuse.lock:/x", Op: op, Lock: lk})
if err != nil {
t.Fatalf("PosixLock op=%v: %v", op, err)
}
return resp
}
func TestPosixLockGrantAndConflict(t *testing.T) {
fs := newPosixTestServer()
if r := posixOp(t, fs, filer_pb.PosixLockOp_TRY_LOCK, pbLock(0, 99, posixlock.Write, 1, 1, 7, false)); !r.Granted {
t.Fatal("first lock should be granted")
}
// Conflicting writer from another session: rejected, conflict reported.
r := posixOp(t, fs, filer_pb.PosixLockOp_TRY_LOCK, pbLock(50, 149, posixlock.Write, 2, 1, 8, false))
if r.Granted {
t.Fatal("overlapping lock from another session should conflict")
}
if !r.HasConflict || r.Conflict.GetPid() != 7 || r.Conflict.GetStart() != 0 || r.Conflict.GetEnd() != 99 {
t.Fatalf("conflict not reported correctly: %+v", r.Conflict)
}
}
func TestPosixLockUnlockThenReacquire(t *testing.T) {
fs := newPosixTestServer()
posixOp(t, fs, filer_pb.PosixLockOp_TRY_LOCK, pbLock(0, 99, posixlock.Write, 1, 1, 7, false))
posixOp(t, fs, filer_pb.PosixLockOp_UNLOCK, pbLock(0, 99, posixlock.Unlock, 1, 1, 7, false))
if r := posixOp(t, fs, filer_pb.PosixLockOp_TRY_LOCK, pbLock(0, 99, posixlock.Write, 2, 1, 8, false)); !r.Granted {
t.Fatal("lock should be grantable after the holder unlocked")
}
}
func TestPosixLockGetLk(t *testing.T) {
fs := newPosixTestServer()
posixOp(t, fs, filer_pb.PosixLockOp_TRY_LOCK, pbLock(10, 50, posixlock.Write, 1, 1, 7, false))
r := posixOp(t, fs, filer_pb.PosixLockOp_GET_LK, pbLock(30, 70, posixlock.Read, 2, 1, 8, false))
if !r.HasConflict || r.Conflict.GetPid() != 7 {
t.Fatalf("GET_LK should report the holder, got %+v", r)
}
// No conflict for the same owner.
r = posixOp(t, fs, filer_pb.PosixLockOp_GET_LK, pbLock(30, 70, posixlock.Read, 1, 1, 7, false))
if r.HasConflict {
t.Fatal("an owner should not conflict with itself")
}
}
func TestPosixLockReleasePosixOwnerKeepsFlock(t *testing.T) {
fs := newPosixTestServer()
posixOp(t, fs, filer_pb.PosixLockOp_TRY_LOCK, pbLock(0, 99, posixlock.Write, 1, 1, 7, false))
posixOp(t, fs, filer_pb.PosixLockOp_TRY_LOCK, pbLock(0, 1<<63, posixlock.Write, 1, 1, 7, true))
posixOp(t, fs, filer_pb.PosixLockOp_RELEASE_POSIX_OWNER, pbLock(0, 0, posixlock.Unlock, 1, 1, 7, false))
// fcntl gone, flock remains.
if r := posixOp(t, fs, filer_pb.PosixLockOp_TRY_LOCK, pbLock(0, 99, posixlock.Write, 2, 1, 8, false)); !r.Granted {
t.Fatal("fcntl lock should be gone after RELEASE_POSIX_OWNER")
}
if r := posixOp(t, fs, filer_pb.PosixLockOp_GET_LK, pbLock(0, 10, posixlock.Write, 3, 3, 9, true)); !r.HasConflict {
t.Fatal("flock lock should survive RELEASE_POSIX_OWNER")
}
}
func TestPosixLockKeepAlive(t *testing.T) {
fs := newPosixTestServer()
resp, err := fs.PosixLock(context.Background(), &filer_pb.PosixLockRequest{
Key: "s3.fuse.lock:/x", Op: filer_pb.PosixLockOp_KEEP_ALIVE,
Lock: pbLock(0, 0, posixlock.Unlock, 7, 0, 0, false),
})
if err != nil || resp == nil {
t.Fatalf("keep_alive should succeed: err=%v", err)
}
// A renewed session that goes stale is reaped; a never-renewed one is not.
fs.posixLocks.TryLock("s3.fuse.lock:/x", posixlock.Range{Start: 0, End: 9, Type: posixlock.Write, Sid: 7, Owner: 1})
if reaped := fs.posixLocks.ReapExpired(0); len(reaped) != 1 || reaped[0] != 7 {
t.Fatalf("renewed session 7 should be reapable at ttl=0, got %v", reaped)
}
}
// A request whose key is owned by another filer is forwarded to it; the owner
// applies it and the sender does not. The owner's ring points back at the bogus
// sender, so without is_moved on the forwarded hop it would re-forward and fail.
func TestPosixLockForwardsToOwner(t *testing.T) {
const key = "s3.fuse.lock:/x"
owner := newPosixTestServer()
lis, err := net.Listen("tcp", "127.0.0.1:0")
if err != nil {
t.Fatalf("listen: %v", err)
}
port := lis.Addr().(*net.TCPAddr).Port
ownerAddr := pb.NewServerAddressWithGrpcPort(lis.Addr().String(), port)
sender := pb.ServerAddress("127.0.0.1:1")
withRing(owner, ownerAddr, sender)
owner.grpcDialOption = grpc.WithTransportCredentials(insecure.NewCredentials())
srv := grpc.NewServer()
filer_pb.RegisterSeaweedFilerServer(srv, owner)
go srv.Serve(lis)
t.Cleanup(srv.Stop)
self := newPosixTestServer()
withRing(self, sender, ownerAddr)
self.grpcDialOption = grpc.WithTransportCredentials(insecure.NewCredentials())
ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second)
defer cancel()
resp, err := self.PosixLock(ctx, &filer_pb.PosixLockRequest{
Key: key, Op: filer_pb.PosixLockOp_TRY_LOCK,
Lock: pbLock(0, 99, posixlock.Write, 1, 1, 7, false),
})
if err != nil || !resp.GetGranted() {
t.Fatalf("forwarded TRY_LOCK: err=%v granted=%v", err, resp.GetGranted())
}
// The lock landed on the owner: a conflicting acquire there is rejected.
if _, granted := owner.posixLocks.TryLock(key, posixlock.Range{Start: 50, End: 149, Type: posixlock.Write, Sid: 2, Owner: 1}); granted {
t.Fatal("owner should hold the forwarded lock")
}
// The sender did not apply locally.
if _, granted := self.posixLocks.TryLock(key, posixlock.Range{Start: 0, End: 99, Type: posixlock.Write, Sid: 9, Owner: 9}); !granted {
t.Fatal("sender must forward, not apply locally")
}
}
func TestPosixLockWarmupDefersGrants(t *testing.T) {
fs := newPosixTestServer()
fs.posixLockReadyAt.Store(time.Now().UnixNano()) // warming up
// A would-be grant is deferred (not granted, no conflict) so the client retries.
r := posixOp(t, fs, filer_pb.PosixLockOp_TRY_LOCK, pbLock(0, 99, posixlock.Write, 1, 1, 7, false))
if r.Granted {
t.Fatal("grant should be deferred during warm-up")
}
if r.HasConflict {
t.Fatalf("deferred grant should report no conflict, got %+v", r.Conflict)
}
// A lock the owner already knows about is still reported as a conflict.
fs.posixLocks.TryLock("s3.fuse.lock:/x", posixlock.Range{Start: 0, End: 99, Type: posixlock.Write, Sid: 9, Owner: 1, Pid: 5})
if r := posixOp(t, fs, filer_pb.PosixLockOp_TRY_LOCK, pbLock(50, 60, posixlock.Write, 1, 1, 7, false)); r.Granted || !r.HasConflict {
t.Fatalf("known conflict should be reported during warm-up: %+v", r)
}
// After warm-up, grants resume.
posixOp(t, fs, filer_pb.PosixLockOp_UNLOCK, pbLock(0, 99, posixlock.Unlock, 9, 1, 5, false))
fs.posixLockReadyAt.Store(time.Now().Add(-2 * posixLockWarmup).UnixNano())
if r := posixOp(t, fs, filer_pb.PosixLockOp_TRY_LOCK, pbLock(0, 99, posixlock.Write, 1, 1, 7, false)); !r.Granted {
t.Fatal("grant should succeed after warm-up")
}
}
+1 -1
View File
@@ -245,7 +245,7 @@ func (fs *FilerServer) moveSelfEntry(ctx context.Context, stream filer_pb.Seawee
return fmt.Errorf("insert entry %s: %v", newEntry.FullPath, createErr)
}
} else {
if createErr := fs.filer.CreateEntry(filer.WithSuppressedMetadataEvents(ctx), newEntry, nil, false, false, signatures, false, fs.filer.MaxFilenameLength); createErr != nil {
if createErr := fs.filer.CreateEntry(filer.WithSuppressedMetadataEvents(ctx), newEntry, false, false, signatures, false, fs.filer.MaxFilenameLength); createErr != nil {
return createErr
}
}

Some files were not shown because too many files have changed in this diff Show More