mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-10-07 15:15:52 +00:00
Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
6b5942946f |
@@ -128,14 +128,14 @@ jobs:
|
||||
|
||||
- name: Login to Docker Hub
|
||||
if: github.event_name != 'pull_request'
|
||||
uses: docker/login-action@v4.2.0
|
||||
uses: docker/login-action@v4.1.0
|
||||
with:
|
||||
username: ${{ secrets.DOCKER_USERNAME }}
|
||||
password: ${{ secrets.DOCKER_PASSWORD }}
|
||||
|
||||
- name: Login to GHCR
|
||||
if: github.event_name != 'pull_request'
|
||||
uses: docker/login-action@v4.2.0
|
||||
uses: docker/login-action@v4.1.0
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ secrets.GHCR_USERNAME }}
|
||||
|
||||
@@ -133,7 +133,7 @@ jobs:
|
||||
|
||||
- name: Login to Docker Hub
|
||||
if: github.event_name != 'pull_request'
|
||||
uses: docker/login-action@v4.2.0
|
||||
uses: docker/login-action@v4.1.0
|
||||
with:
|
||||
username: ${{ secrets.DOCKER_USERNAME }}
|
||||
password: ${{ secrets.DOCKER_PASSWORD }}
|
||||
|
||||
@@ -221,13 +221,13 @@ jobs:
|
||||
buildkitd-config: /tmp/buildkitd.toml
|
||||
- name: Login to Docker Hub
|
||||
if: needs.setup.outputs.publish == 'true'
|
||||
uses: docker/login-action@v4.2.0
|
||||
uses: docker/login-action@v4.1.0
|
||||
with:
|
||||
username: ${{ secrets.DOCKER_USERNAME }}
|
||||
password: ${{ secrets.DOCKER_PASSWORD }}
|
||||
- name: Login to GHCR
|
||||
if: needs.setup.outputs.publish == 'true'
|
||||
uses: docker/login-action@v4.2.0
|
||||
uses: docker/login-action@v4.1.0
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ secrets.GHCR_USERNAME }}
|
||||
@@ -275,7 +275,7 @@ jobs:
|
||||
fi
|
||||
- name: Login to GHCR
|
||||
if: needs.setup.outputs.publish == 'true'
|
||||
uses: docker/login-action@v4.2.0
|
||||
uses: docker/login-action@v4.1.0
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ secrets.GHCR_USERNAME }}
|
||||
@@ -430,12 +430,12 @@ jobs:
|
||||
ghcr.io/chrislusf/seaweedfs
|
||||
tags: type=raw,value=${{ github.event_name == 'workflow_dispatch' && github.event.inputs.image_tag || 'latest' }},suffix=${{ steps.config.outputs.tag_suffix }}
|
||||
- name: Login to Docker Hub
|
||||
uses: docker/login-action@v4.2.0
|
||||
uses: docker/login-action@v4.1.0
|
||||
with:
|
||||
username: ${{ secrets.DOCKER_USERNAME }}
|
||||
password: ${{ secrets.DOCKER_PASSWORD }}
|
||||
- name: Login to GHCR
|
||||
uses: docker/login-action@v4.2.0
|
||||
uses: docker/login-action@v4.1.0
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ secrets.GHCR_USERNAME }}
|
||||
|
||||
@@ -50,7 +50,7 @@ jobs:
|
||||
-
|
||||
name: Login to Docker Hub
|
||||
if: github.event_name != 'pull_request'
|
||||
uses: docker/login-action@v4.2.0
|
||||
uses: docker/login-action@v4.1.0
|
||||
with:
|
||||
username: ${{ secrets.DOCKER_USERNAME }}
|
||||
password: ${{ secrets.DOCKER_PASSWORD }}
|
||||
|
||||
@@ -237,14 +237,14 @@ jobs:
|
||||
|
||||
- name: Login to Docker Hub
|
||||
if: (github.event_name != 'workflow_dispatch' || github.event.inputs.variant == 'all' || github.event.inputs.variant == matrix.variant) && github.event_name != 'pull_request'
|
||||
uses: docker/login-action@v4.2.0
|
||||
uses: docker/login-action@v4.1.0
|
||||
with:
|
||||
username: ${{ secrets.DOCKER_USERNAME }}
|
||||
password: ${{ secrets.DOCKER_PASSWORD }}
|
||||
|
||||
- name: Login to GHCR
|
||||
if: (github.event_name != 'workflow_dispatch' || github.event.inputs.variant == 'all' || github.event.inputs.variant == matrix.variant) && github.event_name != 'pull_request'
|
||||
uses: docker/login-action@v4.2.0
|
||||
uses: docker/login-action@v4.1.0
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ secrets.GHCR_USERNAME }}
|
||||
@@ -300,14 +300,14 @@ jobs:
|
||||
steps:
|
||||
- name: Login to Docker Hub
|
||||
if: github.event_name != 'workflow_dispatch' || github.event.inputs.variant == 'all' || github.event.inputs.variant == matrix.variant
|
||||
uses: docker/login-action@v4.2.0
|
||||
uses: docker/login-action@v4.1.0
|
||||
with:
|
||||
username: ${{ secrets.DOCKER_USERNAME }}
|
||||
password: ${{ secrets.DOCKER_PASSWORD }}
|
||||
|
||||
- name: Login to GHCR
|
||||
if: github.event_name != 'workflow_dispatch' || github.event.inputs.variant == 'all' || github.event.inputs.variant == matrix.variant
|
||||
uses: docker/login-action@v4.2.0
|
||||
uses: docker/login-action@v4.1.0
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ secrets.GHCR_USERNAME }}
|
||||
@@ -380,7 +380,7 @@ jobs:
|
||||
variant: large_disk
|
||||
steps:
|
||||
- name: Login to GHCR
|
||||
uses: docker/login-action@v4.2.0
|
||||
uses: docker/login-action@v4.1.0
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ secrets.GHCR_USERNAME }}
|
||||
@@ -429,13 +429,13 @@ jobs:
|
||||
latest_tag: latest_large_disk
|
||||
steps:
|
||||
- name: Login to Docker Hub
|
||||
uses: docker/login-action@v4.2.0
|
||||
uses: docker/login-action@v4.1.0
|
||||
with:
|
||||
username: ${{ secrets.DOCKER_USERNAME }}
|
||||
password: ${{ secrets.DOCKER_PASSWORD }}
|
||||
|
||||
- name: Login to GHCR
|
||||
uses: docker/login-action@v4.2.0
|
||||
uses: docker/login-action@v4.1.0
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ secrets.GHCR_USERNAME }}
|
||||
|
||||
@@ -88,7 +88,7 @@ jobs:
|
||||
uses: docker/setup-buildx-action@4d04d5d9486b7bd6fa91e7baf45bbb4f8b9deedd # v1
|
||||
|
||||
- name: Login to Docker Hub
|
||||
uses: docker/login-action@650006c6eb7dba73a995cc03b0b2d7f5ca915bee # v1
|
||||
uses: docker/login-action@4907a6ddec9925e35a0a9e82d7399ccc52663121 # v1
|
||||
with:
|
||||
username: ${{ secrets.DOCKER_USERNAME }}
|
||||
password: ${{ secrets.DOCKER_PASSWORD }}
|
||||
|
||||
@@ -1,167 +0,0 @@
|
||||
# Design: Serializing Bucket Configuration Mutations
|
||||
|
||||
Issue #9651 — concurrent `PutBucketVersioning` + `PutBucketEncryption` (as Terraform
|
||||
issues them in parallel) intermittently lose the encryption write.
|
||||
|
||||
## Root cause
|
||||
|
||||
The bucket's entire config lives in one filer entry, `/buckets/<name>`. Every
|
||||
config API does a read-modify-write of that single entry, and the writes are not
|
||||
serialized:
|
||||
|
||||
- `updateBucketConfig(bucket, fn)` (`s3api_bucket_config.go:468`) — sources from a
|
||||
possibly-stale cached `BucketConfig`, mutates `Entry.Extended`, writes the
|
||||
**whole** entry. Used by: versioning, object-lock config, lifecycle, ACL/owner.
|
||||
- `UpdateBucketMetadata` → `setBucketMetadata` (`:1042`) — reads a fresh entry,
|
||||
mutates `Entry.Content`, writes the **whole** entry. Used by: encryption, CORS,
|
||||
tagging, ownership, policy, notification.
|
||||
|
||||
Two ingredients produce the lost update:
|
||||
|
||||
1. **No serialization** of the read→modify→write (the cache mutexes only guard the
|
||||
in-memory map, not the RMW).
|
||||
2. **Whole-entry rewrite from an independent snapshot** — `updateBucketConfig`
|
||||
rebuilds from a stale cached `BucketConfig` whose `Content` predates the
|
||||
concurrent encryption write, so writing the whole entry reverts `Content`.
|
||||
|
||||
Sequential calls always pass (each sees the previous write), so it only surfaces
|
||||
under concurrency — and CI's slower IO widens the window (the "2 of ~12 runs").
|
||||
|
||||
## Goals
|
||||
|
||||
- No lost updates across concurrent bucket-config changes — for **all** config
|
||||
fields, not just versioning/encryption.
|
||||
- Correct for a single S3 gateway (the reported case) and for multiple gateways.
|
||||
- Reuse the filer primitives just merged (per-path lock, `WriteCondition`,
|
||||
`ObjectTransaction`); do not reintroduce a distributed lock.
|
||||
- Minimal blast radius: the fix lands at the two chokepoint helpers.
|
||||
|
||||
## Non-goals
|
||||
|
||||
- Changing the one-entry-per-bucket storage model.
|
||||
- Multi-filer-concurrent bucket writes (addressed only as an optional phase 3).
|
||||
|
||||
## The two ingredients map to two complementary fixes
|
||||
|
||||
### Fix A — serialize + read fresh (closes the window for whole-entry writers)
|
||||
|
||||
Both `updateBucketConfig` and `UpdateBucketMetadata` must run their RMW under one
|
||||
per-bucket critical section, and **re-read the entry fresh from the filer inside
|
||||
it** — not rebuild from the cached `BucketConfig`. The lock alone is insufficient:
|
||||
without the fresh read, two serialized writers still each apply a stale snapshot.
|
||||
|
||||
### Fix B — field-level updates (removes the collision entirely)
|
||||
|
||||
The two writers touch disjoint fields (`Extended[versioning]` vs `Content`). If
|
||||
each path updated only its own field instead of rewriting the whole entry, neither
|
||||
could clobber the other regardless of ordering. This is the structural fix and
|
||||
makes serialization a defense-in-depth concern rather than a correctness
|
||||
requirement for cross-field cases.
|
||||
|
||||
## Where to serialize (layering)
|
||||
|
||||
The bucket entry is a single filer entry, so unlike object writes there is no
|
||||
sharding — the question is purely the scope of the lock:
|
||||
|
||||
| Layer | Serializes across | Cost | Notes |
|
||||
|---|---|---|---|
|
||||
| 1. Gateway-local per-bucket lock | one gateway process | tiny | fixes the reported (single-gateway/CI) case |
|
||||
| 2. Filer per-path lock via conditional write | all gateways on one filer | small | reuses #9640 `CreateEntry`+`WriteCondition` |
|
||||
| 3. Route-by-key to bucket-key owner filer | all gateways and filers | medium | same mechanism as the object DLM-removal |
|
||||
|
||||
## Recommended plan (phased)
|
||||
|
||||
### Phase 1 — minimal fix for #9651 (gateway-local lock + fresh read)
|
||||
|
||||
Add a bounded per-bucket lock table to `S3ApiServer`, reusing the same
|
||||
`util.LockTable` the filer uses for its per-path lock:
|
||||
|
||||
```go
|
||||
// in S3ApiServer
|
||||
bucketConfigLocks *util.LockTable[string] // serialize bucket-entry RMW
|
||||
|
||||
func (s3a *S3ApiServer) withBucketConfigLock(bucket string, fn func() s3err.ErrorCode) s3err.ErrorCode {
|
||||
lk := s3a.bucketConfigLocks.AcquireLock("bucketConfig", bucket, util.ExclusiveLock)
|
||||
defer s3a.bucketConfigLocks.ReleaseLock(bucket, lk)
|
||||
return fn()
|
||||
}
|
||||
```
|
||||
|
||||
Wrap the RMW in **both** chokepoints, and inside the lock read the entry fresh:
|
||||
|
||||
- `updateBucketConfig`: acquire the lock; re-read `/buckets/<name>` from the filer
|
||||
(not the cache); rebuild `BucketConfig` from that fresh entry; apply `fn`; write;
|
||||
invalidate cache; release.
|
||||
- `UpdateBucketMetadata`/`setBucketMetadata`: same lock key; it already reads fresh,
|
||||
so it just needs to share the critical section.
|
||||
|
||||
Both must use the **same** lock keyed on `bucket`, so versioning and encryption
|
||||
contend on one mutex. This closes the reported window. Limitation: only one
|
||||
gateway; two gateways behind a load balancer still race.
|
||||
|
||||
Test: parallel `PutBucketVersioning` + `PutBucketEncryption`, assert both persist
|
||||
(the exact Terraform scenario), plus an N-way parallel variant over distinct
|
||||
fields.
|
||||
|
||||
### Phase 2 — robust across gateways (field-level + CAS via merged primitives)
|
||||
|
||||
Move the writers off whole-entry rewrites:
|
||||
|
||||
- **Extended-based config** (versioning, object-lock, ownership, tagging-in-Extended)
|
||||
→ `ObjectTransaction` `PATCH_EXTENDED` on `/buckets/<name>`. The owner filer reads
|
||||
the entry fresh under its per-path lock and merges only the named keys, so the
|
||||
gateway never sends a whole-entry snapshot — this dissolves *both* ingredients for
|
||||
these fields.
|
||||
- **`Content`-based config** (encryption, CORS, tags blob) — **chosen and
|
||||
implemented (b3): extend `PATCH_EXTENDED` with `set_content`.** Under the same
|
||||
per-path lock the filer reads the entry fresh, merges extended attributes, and
|
||||
replaces `Content`, preserving the rest. So a content write becomes a field-level
|
||||
patch too — `setBucketMetadata` patches `Content`, `updateBucketConfig` patches
|
||||
extended keys, and the two serialize on the lock instead of racing whole-entry
|
||||
rewrites. This is cleaner than the alternatives below: no client-side retry, no
|
||||
storage migration, and it reuses `ObjectTransaction`'s existing atomic lock.
|
||||
- (b1, rejected) Conditional `CreateEntry` overwrite with `IF_ETAG_MATCH` + retry
|
||||
(#9640): correct but needs client-side retry, and the bucket directory entry has
|
||||
no reliable ETag to compare on.
|
||||
- (b2, future) Migrate each per-feature config out of the single `Content` blob
|
||||
into its own `Extended` key. Then even *intra-blob* writes (tags vs encryption)
|
||||
stop racing. Larger migration; tracked separately.
|
||||
|
||||
Once all paths are field-level patches, the phase-1 gateway lock is unnecessary —
|
||||
the filer enforces atomicity. (This is the path taken: phase 1 was skipped.)
|
||||
|
||||
### Phase 3 — multi-filer (only if needed)
|
||||
|
||||
If multiple filers can write `/buckets/<name>` concurrently, a filer-local per-path
|
||||
lock no longer suffices. Route bucket-config writes to
|
||||
`PrimaryForKey("/buckets/<name>")` (the lock-ring view) and serialize on that one
|
||||
owner filer — the same route-by-key design used to take object writes off the DLM.
|
||||
Overkill for rare config writes; include only if multi-filer bucket writes are real.
|
||||
|
||||
## Correctness summary
|
||||
|
||||
- Phase 1: all RMW for a bucket serialize within a gateway; the fresh read means the
|
||||
second writer observes the first's change. Closes #9651 for single-gateway.
|
||||
- Phase 2: `PATCH_EXTENDED` is atomic field-level merge at the filer (no snapshot);
|
||||
CAS turns a concurrent `Content` write into a retry, enforced under the filer's
|
||||
per-path lock — correct for any number of gateways sharing a filer.
|
||||
- Phase 3: one owner filer serializes all writers — correct across filers too.
|
||||
|
||||
## Scope checklist (every path that RMWs the bucket entry)
|
||||
|
||||
All of these funnel through the two chokepoints, so fixing the chokepoints covers
|
||||
them — but the fix must not leave any of them on an unserialized path:
|
||||
|
||||
- via `updateBucketConfig`: versioning, object-lock config, lifecycle, ACL/owner.
|
||||
- via `UpdateBucketMetadata`/`setBucketMetadata`: encryption, CORS, tagging,
|
||||
ownership controls, bucket policy, notification.
|
||||
- bucket create/delete (`CreateEntry`/`DeleteEntry` of `/buckets/<name>`) already
|
||||
go through the filer's per-path lock on `CreateEntry`; ensure they take the same
|
||||
bucket lock if they also patch config.
|
||||
|
||||
## Cache rule (must document in code)
|
||||
|
||||
Under the lock, **read the entry from the filer, never rebuild from the cached
|
||||
`BucketConfig`**. The cache is for reads; it must be invalidated on every write and
|
||||
never be the source for an RMW. This is the single most important detail — the lock
|
||||
without the fresh read does not fix the bug.
|
||||
@@ -26,7 +26,7 @@ require (
|
||||
github.com/facebookgo/subset v0.0.0-20200203212716-c811ad88dec4 // indirect
|
||||
github.com/fsnotify/fsnotify v1.9.0 // indirect
|
||||
github.com/go-redsync/redsync/v4 v4.16.0
|
||||
github.com/go-sql-driver/mysql v1.10.0
|
||||
github.com/go-sql-driver/mysql v1.9.3
|
||||
github.com/go-zookeeper/zk v1.0.4 // indirect
|
||||
github.com/golang/protobuf v1.5.4
|
||||
github.com/golang/snappy v1.0.0
|
||||
@@ -48,7 +48,7 @@ require (
|
||||
github.com/klauspost/compress v1.18.6
|
||||
github.com/klauspost/reedsolomon v1.14.0
|
||||
github.com/kurin/blazer v0.5.3
|
||||
github.com/linxGnu/grocksdb v1.10.8
|
||||
github.com/linxGnu/grocksdb v1.10.7
|
||||
github.com/mailru/easyjson v0.9.1 // indirect
|
||||
github.com/mattn/go-isatty v0.0.20 // indirect
|
||||
github.com/modern-go/concurrent v0.0.0-20180306012644-bacd9c7ef1dd // indirect
|
||||
@@ -91,12 +91,12 @@ require (
|
||||
gocloud.dev v0.45.0
|
||||
gocloud.dev/pubsub/natspubsub v0.45.0
|
||||
gocloud.dev/pubsub/rabbitpubsub v0.45.0
|
||||
golang.org/x/crypto v0.52.0
|
||||
golang.org/x/crypto v0.51.0
|
||||
golang.org/x/exp v0.0.0-20260410095643-746e56fc9e2f
|
||||
golang.org/x/image v0.39.0
|
||||
golang.org/x/net v0.54.0
|
||||
golang.org/x/oauth2 v0.36.0
|
||||
golang.org/x/sys v0.45.0
|
||||
golang.org/x/sys v0.44.0
|
||||
golang.org/x/text v0.37.0 // indirect
|
||||
golang.org/x/tools v0.44.0 // indirect
|
||||
golang.org/x/xerrors v0.0.0-20240903120638-7835f813f4da // indirect
|
||||
@@ -157,7 +157,7 @@ require (
|
||||
github.com/xeipuuv/gojsonschema v1.2.0
|
||||
github.com/ydb-platform/ydb-go-sdk-auth-environ v0.5.1
|
||||
github.com/ydb-platform/ydb-go-sdk/v3 v3.134.2
|
||||
go.etcd.io/etcd/client/pkg/v3 v3.6.11
|
||||
go.etcd.io/etcd/client/pkg/v3 v3.6.10
|
||||
go.uber.org/atomic v1.11.0
|
||||
golang.org/x/sync v0.20.0
|
||||
golang.org/x/tools/godoc v0.1.0-deprecated
|
||||
@@ -296,7 +296,7 @@ require (
|
||||
cloud.google.com/go/compute/metadata v0.9.0 // indirect
|
||||
cloud.google.com/go/iam v1.7.0 // indirect
|
||||
cloud.google.com/go/monitoring v1.24.3 // indirect
|
||||
filippo.io/edwards25519 v1.2.0 // indirect
|
||||
filippo.io/edwards25519 v1.1.1 // indirect
|
||||
github.com/Azure/azure-sdk-for-go/sdk/azcore v1.21.1
|
||||
github.com/Azure/azure-sdk-for-go/sdk/azidentity v1.13.1
|
||||
github.com/Azure/azure-sdk-for-go/sdk/internal v1.12.0 // indirect
|
||||
|
||||
@@ -547,8 +547,8 @@ cloud.google.com/go/workflows v1.10.0/go.mod h1:fZ8LmRmZQWacon9UCX1r/g/DfAXx5VcP
|
||||
dario.cat/mergo v1.0.2 h1:85+piFYR1tMbRrLcDwR18y4UKJ3aH1Tbzi24VRW1TK8=
|
||||
dario.cat/mergo v1.0.2/go.mod h1:E/hbnu0NxMFBjpMIE34DRGLWqDy0g5FuKDhCb31ngxA=
|
||||
dmitri.shuralyov.com/gpu/mtl v0.0.0-20190408044501-666a987793e9/go.mod h1:H6x//7gZCb22OMCxBHrMx7a5I7Hp++hsVxbQ4BYO7hU=
|
||||
filippo.io/edwards25519 v1.2.0 h1:crnVqOiS4jqYleHd9vaKZ+HKtHfllngJIiOpNpoJsjo=
|
||||
filippo.io/edwards25519 v1.2.0/go.mod h1:xzAOLCNug/yB62zG1bQ8uziwrIqIuxhctzJT18Q77mc=
|
||||
filippo.io/edwards25519 v1.1.1 h1:YpjwWWlNmGIDyXOn8zLzqiD+9TyIlPhGFG96P39uBpw=
|
||||
filippo.io/edwards25519 v1.1.1/go.mod h1:BxyFTGdWcka3PhytdK4V28tE5sGfRvvvRV7EaN4VDT4=
|
||||
gioui.org v0.0.0-20210308172011-57750fc8a0a6/go.mod h1:RSH6KIUZ0p2xy5zHDxgAM4zumjgTw83q2ge/PI+yyw8=
|
||||
git.sr.ht/~sbinet/gg v0.3.1/go.mod h1:KGYtlADtqsqANL9ueOFkWymvzUvLMQllU5Ixo+8v3pc=
|
||||
github.com/AdaLogics/go-fuzz-headers v0.0.0-20240806141605-e8a1dd7889d6 h1:He8afgbRMd7mFxO99hRNu+6tazq8nFF9lIwo9JFroBk=
|
||||
@@ -1128,8 +1128,8 @@ github.com/go-redsync/redsync/v4 v4.16.0 h1:bNcOzeHH9d3s6pghU9NJFMPrQa41f5Nx3L4Y
|
||||
github.com/go-redsync/redsync/v4 v4.16.0/go.mod h1:V4gagqgyASWBZuwx4xGzu72aZNb/6Mo05byUa3mVmKQ=
|
||||
github.com/go-resty/resty/v2 v2.17.2 h1:FQW5oHYcIlkCNrMD2lloGScxcHJ0gkjshV3qcQAyHQk=
|
||||
github.com/go-resty/resty/v2 v2.17.2/go.mod h1:kCKZ3wWmwJaNc7S29BRtUhJwy7iqmn+2mLtQrOyQlVA=
|
||||
github.com/go-sql-driver/mysql v1.10.0 h1:Q+1LV8DkHJvSYAdR83XzuhDaTykuDx0l6fkXxoWCWfw=
|
||||
github.com/go-sql-driver/mysql v1.10.0/go.mod h1:M+cqaI7+xxXGG9swrdeUIoPG3Y3KCkF0pZej+SK+nWk=
|
||||
github.com/go-sql-driver/mysql v1.9.3 h1:U/N249h2WzJ3Ukj8SowVFjdtZKfu9vlLZxjPXV1aweo=
|
||||
github.com/go-sql-driver/mysql v1.9.3/go.mod h1:qn46aNg1333BRMNU69Lq93t8du/dwxI64Gl8i5p1WMU=
|
||||
github.com/go-stack/stack v1.8.0/go.mod h1:v0f6uXyyMGvRgIKkXu+yp6POWl0qKG85gN/melR3HDY=
|
||||
github.com/go-task/slim-sprig v0.0.0-20230315185526-52ccab3ef572 h1:tfuBGBXKqDEevZMzYi5KSi8KkcZtzBcTgAUUtapy0OI=
|
||||
github.com/go-task/slim-sprig/v3 v3.0.0 h1:sUs3vkvUymDpBKi3qH1YSqBQk9+9D/8M2mN1vB6EwHI=
|
||||
@@ -1515,8 +1515,8 @@ github.com/lib/pq v1.11.1 h1:wuChtj2hfsGmmx3nf1m7xC2XpK6OtelS2shMY+bGMtI=
|
||||
github.com/lib/pq v1.11.1/go.mod h1:/p+8NSbOcwzAEI7wiMXFlgydTwcgTr3OSKMsD2BitpA=
|
||||
github.com/linkedin/goavro/v2 v2.15.0 h1:pDj1UrjUOO62iXhgBiE7jQkpNIc5/tA5eZsgolMjgVI=
|
||||
github.com/linkedin/goavro/v2 v2.15.0/go.mod h1:KXx+erlq+RPlGSPmLF7xGo6SAbh8sCQ53x064+ioxhk=
|
||||
github.com/linxGnu/grocksdb v1.10.8 h1:Nau01Hhm/0kaVTR6d4viwD6npYbnDvZAfzwJCLzKRYo=
|
||||
github.com/linxGnu/grocksdb v1.10.8/go.mod h1:OLQKZwiKwaJiAVCsOzWKvwiLwfZ5Vz8Md5TYR7t7pM8=
|
||||
github.com/linxGnu/grocksdb v1.10.7 h1:fCi4qvZWo04VgFwGWmO8HQJgUVounJBy+C2TMVPU/ho=
|
||||
github.com/linxGnu/grocksdb v1.10.7/go.mod h1:OLQKZwiKwaJiAVCsOzWKvwiLwfZ5Vz8Md5TYR7t7pM8=
|
||||
github.com/lithammer/fuzzysearch v1.1.8 h1:/HIuJnjHuXS8bKaiTMeeDlW2/AyIWk2brx1V8LFgLN4=
|
||||
github.com/lithammer/fuzzysearch v1.1.8/go.mod h1:IdqeyBClc3FFqSzYq/MXESsS4S0FsZ5ajtkr5xPLts4=
|
||||
github.com/lithammer/shortuuid/v3 v3.0.7 h1:trX0KTHy4Pbwo/6ia8fscyHoGA+mf1jWbPJVuvyJQQ8=
|
||||
@@ -2111,8 +2111,8 @@ go.etcd.io/bbolt v1.4.3 h1:dEadXpI6G79deX5prL3QRNP6JB8UxVkqo4UPnHaNXJo=
|
||||
go.etcd.io/bbolt v1.4.3/go.mod h1:tKQlpPaYCVFctUIgFKFnAlvbmB3tpy1vkTnDWohtc0E=
|
||||
go.etcd.io/etcd/api/v3 v3.6.10 h1:jlwjtELjA8yi2VWpOFH+0w0lGr3K6mVDyn0RDB9aaAY=
|
||||
go.etcd.io/etcd/api/v3 v3.6.10/go.mod h1:pdV4VeFmvhdNjB4LWRkC8ReLyRBAxUOze3GarMhE2sk=
|
||||
go.etcd.io/etcd/client/pkg/v3 v3.6.11 h1:e41mp315Yn3QMGPmEzCyLsMINgJXTY/dX8kM++1csxU=
|
||||
go.etcd.io/etcd/client/pkg/v3 v3.6.11/go.mod h1:DysuMe/inqRyC/1tjRR6hReH/VV9Lufs27YKSKBWWJg=
|
||||
go.etcd.io/etcd/client/pkg/v3 v3.6.10 h1:tBT7podcPhuVbCVkAEzx8bC5I+aqxfLwBN8/As1arrA=
|
||||
go.etcd.io/etcd/client/pkg/v3 v3.6.10/go.mod h1:WEy3PpwbbEBVRdh1NVJYsuUe/8eyI21PNJRazeD8z/Y=
|
||||
go.etcd.io/etcd/client/v3 v3.6.10 h1:J598zJ+C/ZPvImypmq5waj84+bovePrlZERHklf34y0=
|
||||
go.etcd.io/etcd/client/v3 v3.6.10/go.mod h1:iHhUDUcEwaKs1YFq3MgmI9U4zhTVasp/vgdVbFf1RS8=
|
||||
go.mongodb.org/mongo-driver v1.17.9 h1:IexDdCuuNJ3BHrELgBlyaH9p60JXAvdzWR128q+U5tU=
|
||||
@@ -2216,8 +2216,8 @@ golang.org/x/crypto v0.14.0/go.mod h1:MVFd36DqK4CsrnJYDkBA3VC4m2GkXAM0PvzMCn4JQf
|
||||
golang.org/x/crypto v0.19.0/go.mod h1:Iy9bg/ha4yyC70EfRS8jz+B6ybOBKMaSxLj6P6oBDfU=
|
||||
golang.org/x/crypto v0.23.0/go.mod h1:CKFgDieR+mRhux2Lsu27y0fO304Db0wZe70UKqHu0v8=
|
||||
golang.org/x/crypto v0.31.0/go.mod h1:kDsLvtWBEx7MV9tJOj9bnXsPbxwJQ6csT/x4KIN4Ssk=
|
||||
golang.org/x/crypto v0.52.0 h1:RMs7fP2rXdep0CftQlK8Uf+kibLm7qkCcradZWYz988=
|
||||
golang.org/x/crypto v0.52.0/go.mod h1:1QgfPxDqh0T2M/elOJtp9RvuR95kVjir0e6/BvEmGbc=
|
||||
golang.org/x/crypto v0.51.0 h1:IBPXwPfKxY7cWQZ38ZCIRPI50YLeevDLlLnyC5wRGTI=
|
||||
golang.org/x/crypto v0.51.0/go.mod h1:8AdwkbraGNABw2kOX6YFPs3WM22XqI4EXEd8g+x7Oc8=
|
||||
golang.org/x/exp v0.0.0-20180321215751-8460e604b9de/go.mod h1:CJ0aWSM057203Lf6IL+f9T1iT9GByDxfZKAQTCR3kQA=
|
||||
golang.org/x/exp v0.0.0-20180807140117-3d87b88a115f/go.mod h1:CJ0aWSM057203Lf6IL+f9T1iT9GByDxfZKAQTCR3kQA=
|
||||
golang.org/x/exp v0.0.0-20190121172915-509febef88a4/go.mod h1:CJ0aWSM057203Lf6IL+f9T1iT9GByDxfZKAQTCR3kQA=
|
||||
@@ -2511,8 +2511,8 @@ golang.org/x/sys v0.13.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.17.0/go.mod h1:/VUhepiaJMQUp4+oa/7Zr1D23ma6VTLIYjOOTFZPUcA=
|
||||
golang.org/x/sys v0.20.0/go.mod h1:/VUhepiaJMQUp4+oa/7Zr1D23ma6VTLIYjOOTFZPUcA=
|
||||
golang.org/x/sys v0.28.0/go.mod h1:/VUhepiaJMQUp4+oa/7Zr1D23ma6VTLIYjOOTFZPUcA=
|
||||
golang.org/x/sys v0.45.0 h1:dO4czNzziLiiXplLQgBCEpCvXQ3dnkn0SdaZSYdQ+FY=
|
||||
golang.org/x/sys v0.45.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw=
|
||||
golang.org/x/sys v0.44.0 h1:ildZl3J4uzeKP07r2F++Op7E9B29JRUy+a27EibtBTQ=
|
||||
golang.org/x/sys v0.44.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw=
|
||||
golang.org/x/telemetry v0.0.0-20240228155512-f48c80bd79b2/go.mod h1:TeRTkGYfJXctD9OcfyVLyj2J3IxLnKwHJR8f4D8a3YE=
|
||||
golang.org/x/telemetry v0.0.0-20260409153401-be6f6cb8b1fa h1:efT73AJZfAAUV7SOip6pWGkwJDzIGiKBZGVzHYa+ve4=
|
||||
golang.org/x/telemetry v0.0.0-20260409153401-be6f6cb8b1fa/go.mod h1:kHjTxDEnAu6/Nl9lDkzjWpR+bmKfxeiRuSDlsMb70gE=
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
apiVersion: v1
|
||||
description: SeaweedFS
|
||||
name: seaweedfs
|
||||
appVersion: "4.29"
|
||||
appVersion: "4.27"
|
||||
# Dev note: Trigger a helm chart release by `git tag -a helm-<version>`
|
||||
version: 4.29.0
|
||||
version: 4.27.0
|
||||
|
||||
@@ -31,15 +31,6 @@ service SeaweedFiler {
|
||||
rpc DeleteEntry (DeleteEntryRequest) returns (DeleteEntryResponse) {
|
||||
}
|
||||
|
||||
rpc ObjectTransaction (ObjectTransactionRequest) returns (ObjectTransactionResponse) {
|
||||
}
|
||||
|
||||
rpc ObjectTransactionBatch (ObjectTransactionBatchRequest) returns (ObjectTransactionBatchResponse) {
|
||||
}
|
||||
|
||||
rpc PosixLock (PosixLockRequest) returns (PosixLockResponse) {
|
||||
}
|
||||
|
||||
rpc AtomicRenameEntry (AtomicRenameEntryRequest) returns (AtomicRenameEntryResponse) {
|
||||
}
|
||||
rpc StreamRenameEntry (StreamRenameEntryRequest) returns (stream StreamRenameEntryResponse) {
|
||||
@@ -231,56 +222,6 @@ message CreateEntryRequest {
|
||||
bool is_from_other_cluster = 4;
|
||||
repeated int32 signatures = 5;
|
||||
bool skip_check_parent_directory = 6;
|
||||
// Optional precondition evaluated against the current entry atomically with
|
||||
// the write, under the filer's per-path lock. The caller must route the
|
||||
// key's writes to this entry's owner filer for the check to be authoritative.
|
||||
WriteCondition condition = 7;
|
||||
}
|
||||
|
||||
// WriteCondition is the precondition the filer evaluates against the existing
|
||||
// entry before writing, under the per-path lock. A failed condition returns
|
||||
// FilerError PRECONDITION_FAILED. The client maps request semantics (e.g. RFC
|
||||
// 7232) to clauses; the filer just compares.
|
||||
//
|
||||
// A condition is a list of clauses that ALL must hold (logical AND). One clause
|
||||
// is the common case; several express what a single comparison cannot: an ETag
|
||||
// set (If-Match / If-None-Match with multiple values), weak-ETag comparison, and
|
||||
// compound conditions (e.g. If-Match + If-Unmodified-Since together).
|
||||
message WriteCondition {
|
||||
enum Kind {
|
||||
NONE = 0; // unconditional
|
||||
IF_NOT_EXISTS = 1; // fail if the entry exists (If-None-Match: *)
|
||||
IF_EXISTS = 2; // fail if the entry is absent (If-Match: *)
|
||||
IF_ETAG_MATCH = 3; // fail if absent or etag matches none of the set (If-Match)
|
||||
IF_ETAG_NOT_MATCH = 4; // fail if present and etag matches any of the set (If-None-Match)
|
||||
IF_UNMODIFIED_SINCE = 5; // fail if present and mtime > unix_time
|
||||
IF_MODIFIED_SINCE = 6; // fail if present and mtime <= unix_time
|
||||
IF_EXTENDED_NOT_EQUAL = 7; // fail if present and extended[ext_key] == ext_value
|
||||
IF_EXTENDED_TIME_ELAPSED = 8; // fail if present and extended[ext_key] (unix seconds) is in the future
|
||||
}
|
||||
// Clause is one primitive comparison. IF_ETAG_MATCH holds when the current
|
||||
// entry's ETag equals any value in etags; IF_ETAG_NOT_MATCH holds when it
|
||||
// equals none. allow_weak permits weak-comparison (ignoring the W/ prefix).
|
||||
//
|
||||
// The IF_EXTENDED_* kinds are generic guards on an extended attribute, used
|
||||
// to enforce object-lock without teaching the filer S3 semantics:
|
||||
// IF_EXTENDED_NOT_EQUAL expresses a legal hold (block while a key equals a
|
||||
// value), and IF_EXTENDED_TIME_ELAPSED expresses retention (block while a
|
||||
// stored unix-second deadline is in the future, compared to the filer's
|
||||
// clock). The caller composes these and, for governance-bypass, simply omits
|
||||
// the retention clause when the bypass is authorized — the filer makes no
|
||||
// authorization decision.
|
||||
message Clause {
|
||||
Kind kind = 1;
|
||||
repeated string etags = 2; // ETag set for IF_ETAG_* kinds
|
||||
int64 unix_time = 3; // bound (unix seconds) for IF_*_SINCE kinds
|
||||
bool allow_weak = 4; // compare ETags ignoring the weak (W/) marker
|
||||
string ext_key = 5; // extended attribute name for IF_EXTENDED_* kinds
|
||||
string ext_value = 6; // blocking value for IF_EXTENDED_NOT_EQUAL
|
||||
string gate_key = 7; // IF_EXTENDED_TIME_ELAPSED: only enforce when extended[gate_key] == gate_value
|
||||
string gate_value = 8; // gate value (e.g. retention mode COMPLIANCE for governance bypass)
|
||||
}
|
||||
repeated Clause clauses = 1; // all must hold (logical AND)
|
||||
}
|
||||
|
||||
// Structured error codes for filer entry operations.
|
||||
@@ -292,132 +233,6 @@ enum FilerError {
|
||||
EXISTING_IS_DIRECTORY = 3; // cannot overwrite directory with file
|
||||
EXISTING_IS_FILE = 4; // cannot overwrite file with directory
|
||||
ENTRY_ALREADY_EXISTS = 5; // O_EXCL and entry already exists
|
||||
PRECONDITION_FAILED = 6; // WriteCondition not satisfied
|
||||
}
|
||||
|
||||
// ObjectMutation is one entry-level change applied by ObjectTransaction. All
|
||||
// mutations of a transaction run under a single per-path lock (the request's
|
||||
// lock_key) and in order, so the gateway can describe a multi-entry object
|
||||
// operation as one request instead of holding a distributed lock across
|
||||
// several RPCs. Data-bearing writes (entries with chunks) should be written
|
||||
// before the transaction; mutations here are metadata-scoped.
|
||||
message ObjectMutation {
|
||||
enum Type {
|
||||
PUT = 0; // create or replace the entry (entry field)
|
||||
DELETE = 1; // delete the entry at directory/name (no error if absent)
|
||||
PATCH_EXTENDED = 2; // merge set_extended / remove delete_extended on the entry
|
||||
RECOMPUTE_LATEST = 3; // scan a directory and re-point a parent entry (recompute)
|
||||
}
|
||||
Type type = 1;
|
||||
string directory = 2;
|
||||
string name = 3; // entry name for DELETE / PATCH_EXTENDED / RECOMPUTE_LATEST (the pointer entry)
|
||||
Entry entry = 4; // full entry for PUT
|
||||
map<string, bytes> set_extended = 5; // PATCH_EXTENDED: keys to set
|
||||
repeated string delete_extended = 6; // PATCH_EXTENDED: keys to remove
|
||||
bool is_delete_data = 7; // DELETE: also delete chunk data
|
||||
bool is_recursive = 8; // DELETE: recurse into a directory
|
||||
Recompute recompute = 9; // RECOMPUTE_LATEST parameters
|
||||
bool set_content = 10; // PATCH_EXTENDED: replace Entry.content with content
|
||||
bytes content = 11; // PATCH_EXTENDED: new Entry.content when set_content
|
||||
bool touch_mtime = 12; // PATCH_EXTENDED: set the entry's Mtime to now (e.g. a metadata-replace copy)
|
||||
}
|
||||
|
||||
// Recompute re-derives a pointer entry (directory/name on the mutation) from the
|
||||
// current contents of a scanned directory, atomically under the transaction's
|
||||
// lock. It is mechanical: the filer picks the child that sorts first or last by
|
||||
// name and copies the requested fields into the pointer; it has no knowledge of
|
||||
// what the entries mean. The caller (which does know the versioning scheme)
|
||||
// supplies the sort direction and the key mappings. This covers re-pointing the
|
||||
// latest version after a specific version is deleted, where the scan must run
|
||||
// under the lock.
|
||||
message Recompute {
|
||||
string scan_dir = 1; // directory whose direct children are scanned
|
||||
bool descending = 2; // pick the child that sorts last by name (else first)
|
||||
map<string, string> copy_extended = 3; // pointer extended key -> source extended key on the chosen child
|
||||
string name_to_key = 4; // if set, store the chosen child's name under this pointer key
|
||||
string size_to_key = 5; // if set, store the chosen child's FileSize (decimal) under this pointer key
|
||||
string mtime_to_key = 6; // if set, store the chosen child's Mtime (decimal) under this pointer key
|
||||
string demote_key = 7; // if set, stamp demote_value on the prior name_to_key target when it changes
|
||||
bytes demote_value = 8; // value for demote_key
|
||||
string exclude_name = 9; // if set, skip this child when scanning (e.g. a version about to be deleted)
|
||||
}
|
||||
|
||||
// ObjectTransactionRequest applies an ordered list of mutations atomically with
|
||||
// respect to other writers of the same object, by holding the filer's per-path
|
||||
// lock on lock_key for the whole transaction. The optional condition is checked
|
||||
// first, against condition_key when set, else lock_key. Callers set route_key to
|
||||
// the object's stable owner ring key; a filer that is not the owner forwards the
|
||||
// transaction one hop to the owner, so a stale ring view is tolerated.
|
||||
message ObjectTransactionRequest {
|
||||
string lock_key = 1; // object path to lock and to evaluate the condition against
|
||||
WriteCondition condition = 2; // optional precondition, checked under the lock
|
||||
repeated ObjectMutation mutations = 3;
|
||||
bool is_from_other_cluster = 4;
|
||||
repeated int32 signatures = 5;
|
||||
string condition_key = 6; // if set, evaluate the condition against this entry instead of lock_key (still locking lock_key)
|
||||
string route_key = 7; // ring key identifying the owner filer; a non-owner forwards the whole transaction to it
|
||||
bool is_moved = 8; // set on a forwarded transaction so the receiver applies it locally instead of forwarding again
|
||||
}
|
||||
|
||||
message ObjectTransactionResponse {
|
||||
string error = 1;
|
||||
FilerError error_code = 2;
|
||||
}
|
||||
|
||||
// PosixLockRange is one advisory byte-range lock. Owner identity is (sid, owner):
|
||||
// sid is the mount session, owner the FUSE lock owner within it, so owners from
|
||||
// different mounts never alias. end is inclusive (max uint64 = to EOF); is_flock
|
||||
// separates the flock and fcntl namespaces, which never conflict.
|
||||
message PosixLockRange {
|
||||
uint64 start = 1;
|
||||
uint64 end = 2;
|
||||
uint32 type = 3; // 1=read, 2=write, 3=unlock
|
||||
uint64 sid = 4;
|
||||
uint64 owner = 5;
|
||||
uint32 pid = 6; // holder pid, for get_lk reporting only
|
||||
bool is_flock = 7;
|
||||
}
|
||||
|
||||
// PosixLock routes an advisory lock operation to the inode's owner filer, which
|
||||
// holds the authoritative in-memory lock table. key is the inode identity ring
|
||||
// key (the file path, or hl:<HardLinkId> for a hardlink) used both to resolve the
|
||||
// owner and to index the table. A non-owner filer forwards the request one hop;
|
||||
// is_moved bounds it so a stale ring view cannot loop.
|
||||
message PosixLockRequest {
|
||||
string key = 1;
|
||||
bool is_moved = 2;
|
||||
PosixLockOp op = 3;
|
||||
PosixLockRange lock = 4;
|
||||
repeated PosixLockRange locks = 5;
|
||||
bool cooling_probe = 6;
|
||||
}
|
||||
|
||||
enum PosixLockOp {
|
||||
TRY_LOCK = 0; // grant lock or report conflict (non-blocking)
|
||||
UNLOCK = 1; // release lock's owner's locks over its range
|
||||
GET_LK = 2; // report a conflicting lock, if any
|
||||
RELEASE_POSIX_OWNER = 3; // drop the owner's fcntl locks (flush-time)
|
||||
RELEASE_FLOCK_OWNER = 4; // drop the owner's flock locks (release-time)
|
||||
KEEP_ALIVE = 5; // renew the session's lease on this owner (lock.sid)
|
||||
}
|
||||
|
||||
message PosixLockResponse {
|
||||
bool granted = 1; // for TRY_LOCK: whether the lock was granted
|
||||
bool has_conflict = 2; // whether conflict is populated
|
||||
PosixLockRange conflict = 3; // the blocking lock (TRY_LOCK conflict / GET_LK result)
|
||||
}
|
||||
|
||||
// ObjectTransactionBatch applies several object transactions in one round trip,
|
||||
// each under its own per-path lock and independent of the others (no cross-key
|
||||
// atomicity). A caller groups keys that route to the same owner filer and sends
|
||||
// one batch per owner, e.g. for a multi-object delete. Each response is parallel
|
||||
// to its request.
|
||||
message ObjectTransactionBatchRequest {
|
||||
repeated ObjectTransactionRequest transactions = 1;
|
||||
}
|
||||
|
||||
message ObjectTransactionBatchResponse {
|
||||
repeated ObjectTransactionResponse responses = 1;
|
||||
}
|
||||
|
||||
message CreateEntryResponse {
|
||||
@@ -606,7 +421,6 @@ message SubscribeMetadataRequest {
|
||||
repeated string directories = 10; // exact directory to watch
|
||||
bool client_supports_batching = 11; // client can unpack SubscribeMetadataResponse.events
|
||||
bool client_supports_metadata_chunks = 12; // client can read log file chunks from volume servers
|
||||
bool client_supports_idle_heartbeat = 13; // server may send empty responses carrying the current time while the client is caught up
|
||||
}
|
||||
message SubscribeMetadataResponse {
|
||||
string directory = 1;
|
||||
|
||||
@@ -838,6 +838,8 @@ pub struct ReadQueryParams {
|
||||
pub response_content_disposition: Option<String>,
|
||||
/// Pretty print JSON response
|
||||
pub pretty: Option<String>,
|
||||
/// JSONP callback function name
|
||||
pub callback: Option<String>,
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
@@ -3400,6 +3402,10 @@ fn json_response_with_params<T: Serialize>(
|
||||
let is_pretty = params
|
||||
.and_then(|params| params.pretty.as_ref())
|
||||
.is_some_and(|value| !value.is_empty());
|
||||
let callback = params
|
||||
.and_then(|params| params.callback.as_ref())
|
||||
.filter(|value| !value.is_empty())
|
||||
.cloned();
|
||||
|
||||
let json_body = if is_pretty {
|
||||
to_pretty_json(body)
|
||||
@@ -3407,15 +3413,24 @@ fn json_response_with_params<T: Serialize>(
|
||||
serde_json::to_string(body).unwrap()
|
||||
};
|
||||
|
||||
Response::builder()
|
||||
.status(status)
|
||||
.header(header::CONTENT_TYPE, "application/json")
|
||||
.header("X-Content-Type-Options", "nosniff")
|
||||
.body(Body::from(json_body))
|
||||
.unwrap()
|
||||
if let Some(callback) = callback {
|
||||
Response::builder()
|
||||
.status(status)
|
||||
.header(header::CONTENT_TYPE, "application/javascript")
|
||||
.body(Body::from(format!("{}({})", callback, json_body)))
|
||||
.unwrap()
|
||||
} else {
|
||||
Response::builder()
|
||||
.status(status)
|
||||
.header(header::CONTENT_TYPE, "application/json")
|
||||
.body(Body::from(json_body))
|
||||
.unwrap()
|
||||
}
|
||||
}
|
||||
|
||||
/// Return a JSON error response, honoring `?pretty=<any non-empty value>` for pretty-printed JSON.
|
||||
/// Return a JSON error response with optional query string for pretty/JSONP support.
|
||||
/// Supports `?pretty=<any non-empty value>` for pretty-printed JSON and `?callback=fn` for JSONP,
|
||||
/// matching Go's writeJsonError behavior.
|
||||
pub(super) fn json_error_with_query(
|
||||
status: StatusCode,
|
||||
msg: impl Into<String>,
|
||||
@@ -3423,10 +3438,18 @@ pub(super) fn json_error_with_query(
|
||||
) -> Response {
|
||||
let body = serde_json::json!({"error": msg.into()});
|
||||
|
||||
let is_pretty = query.is_some_and(|q| {
|
||||
q.split('&')
|
||||
.any(|p| p.starts_with("pretty=") && p.len() > "pretty=".len())
|
||||
});
|
||||
let (is_pretty, callback) = if let Some(q) = query {
|
||||
let pretty = q
|
||||
.split('&')
|
||||
.any(|p| p.starts_with("pretty=") && p.len() > "pretty=".len());
|
||||
let cb = q
|
||||
.split('&')
|
||||
.find_map(|p| p.strip_prefix("callback="))
|
||||
.map(|s| s.to_string());
|
||||
(pretty, cb)
|
||||
} else {
|
||||
(false, None)
|
||||
};
|
||||
|
||||
let json_body = if is_pretty {
|
||||
to_pretty_json(&body)
|
||||
@@ -3434,19 +3457,35 @@ pub(super) fn json_error_with_query(
|
||||
serde_json::to_string(&body).unwrap()
|
||||
};
|
||||
|
||||
Response::builder()
|
||||
.status(status)
|
||||
.header(header::CONTENT_TYPE, "application/json")
|
||||
.header("X-Content-Type-Options", "nosniff")
|
||||
.body(Body::from(json_body))
|
||||
.unwrap()
|
||||
if let Some(cb) = callback {
|
||||
let jsonp = format!("{}({})", cb, json_body);
|
||||
Response::builder()
|
||||
.status(status)
|
||||
.header(header::CONTENT_TYPE, "application/javascript")
|
||||
.body(Body::from(jsonp))
|
||||
.unwrap()
|
||||
} else {
|
||||
Response::builder()
|
||||
.status(status)
|
||||
.header(header::CONTENT_TYPE, "application/json")
|
||||
.body(Body::from(json_body))
|
||||
.unwrap()
|
||||
}
|
||||
}
|
||||
|
||||
/// Return a JSON response honoring `?pretty=<any non-empty value>` from a raw query string.
|
||||
/// Return a JSON response with optional pretty/JSONP support from raw query string.
|
||||
/// Matches Go's writeJsonQuiet behavior for write success responses.
|
||||
fn json_result_with_query<T: Serialize>(status: StatusCode, body: &T, query: &str) -> Response {
|
||||
let is_pretty = query
|
||||
.split('&')
|
||||
.any(|p| p.starts_with("pretty=") && p.len() > "pretty=".len());
|
||||
let (is_pretty, callback) = {
|
||||
let pretty = query
|
||||
.split('&')
|
||||
.any(|p| p.starts_with("pretty=") && p.len() > "pretty=".len());
|
||||
let cb = query
|
||||
.split('&')
|
||||
.find_map(|p| p.strip_prefix("callback="))
|
||||
.map(|s| s.to_string());
|
||||
(pretty, cb)
|
||||
};
|
||||
|
||||
let json_body = if is_pretty {
|
||||
to_pretty_json(body)
|
||||
@@ -3454,12 +3493,20 @@ fn json_result_with_query<T: Serialize>(status: StatusCode, body: &T, query: &st
|
||||
serde_json::to_string(body).unwrap()
|
||||
};
|
||||
|
||||
Response::builder()
|
||||
.status(status)
|
||||
.header(header::CONTENT_TYPE, "application/json")
|
||||
.header("X-Content-Type-Options", "nosniff")
|
||||
.body(Body::from(json_body))
|
||||
.unwrap()
|
||||
if let Some(cb) = callback {
|
||||
let jsonp = format!("{}({})", cb, json_body);
|
||||
Response::builder()
|
||||
.status(status)
|
||||
.header(header::CONTENT_TYPE, "application/javascript")
|
||||
.body(Body::from(jsonp))
|
||||
.unwrap()
|
||||
} else {
|
||||
Response::builder()
|
||||
.status(status)
|
||||
.header(header::CONTENT_TYPE, "application/json")
|
||||
.body(Body::from(json_body))
|
||||
.unwrap()
|
||||
}
|
||||
}
|
||||
|
||||
/// Extract JWT token from query param, Authorization header, or Cookie.
|
||||
|
||||
@@ -1,232 +0,0 @@
|
||||
package fuse_dlm
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"runtime"
|
||||
"syscall"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/cluster/lock_manager"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// posixLockKey returns the routed-lock key a mount uses for a file at the mount
|
||||
// root. Mounts run with -filer.path=/, so the file's filer path is "/"+name; the
|
||||
// prefix must match mount.posixLockKeyForInode ("s3.fuse.lock:").
|
||||
func posixLockKey(name string) string { return "s3.fuse.lock:/" + name }
|
||||
|
||||
// These tests exercise cross-mount POSIX advisory locks (flock), which the
|
||||
// mounts route to the inode's owner filer because they run with -dlm
|
||||
// (crossMountLocks() == lockClient != nil). They reuse the dlmTestCluster:
|
||||
// master + volume + 2 filers (forming the lock ring) + 2 mounts on filer0.
|
||||
//
|
||||
// All flock opens are O_RDONLY on purpose: a write open (O_RDWR/O_WRONLY) would
|
||||
// also take the -dlm whole-file write lock (held until close), which is a
|
||||
// different mechanism — opening read-only isolates the POSIX advisory flock.
|
||||
|
||||
// requireForwardedLocks skips on platforms where the kernel does not forward
|
||||
// advisory locks to the FUSE server. Only Linux forwards flock/fcntl (SETLK) to
|
||||
// the filesystem; macFUSE handles flock in-kernel per mount, so cross-mount
|
||||
// coordination can't be observed there even though the routed path is correct.
|
||||
func requireForwardedLocks(t *testing.T) {
|
||||
if runtime.GOOS != "linux" {
|
||||
t.Skipf("advisory locks are only forwarded to the FUSE server on Linux (GOOS=%s)", runtime.GOOS)
|
||||
}
|
||||
}
|
||||
|
||||
// openFlock opens path read-only and takes a flock of type how (e.g. LOCK_EX).
|
||||
// The file must already exist. The returned file holds the lock until closed.
|
||||
func openFlock(path string, how int) (*os.File, error) {
|
||||
f, err := os.OpenFile(path, os.O_RDONLY, 0)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := syscall.Flock(int(f.Fd()), how); err != nil {
|
||||
f.Close()
|
||||
return nil, err
|
||||
}
|
||||
return f, nil
|
||||
}
|
||||
|
||||
// createFile creates an empty file; the brief -dlm write lock it takes is
|
||||
// released on close, before any flock test runs.
|
||||
func createFile(t *testing.T, path string) {
|
||||
t.Helper()
|
||||
f, err := os.OpenFile(path, os.O_RDWR|os.O_CREATE|os.O_TRUNC, 0644)
|
||||
require.NoError(t, err, "create %s", path)
|
||||
require.NoError(t, f.Close())
|
||||
}
|
||||
|
||||
// waitVisible waits for path to appear (cross-mount metadata propagation).
|
||||
func waitVisible(t *testing.T, path string) {
|
||||
t.Helper()
|
||||
require.Eventually(t, func() bool {
|
||||
_, err := os.Stat(path)
|
||||
return err == nil
|
||||
}, 15*time.Second, 200*time.Millisecond, "%s never became visible", path)
|
||||
}
|
||||
|
||||
// tryExclusiveFlock attempts a non-blocking exclusive flock on path and
|
||||
// classifies the outcome: acquired (and released again), blocked by another
|
||||
// owner (EWOULDBLOCK/EAGAIN), or an unexpected error. It distinguishes a real
|
||||
// "held by another" from incidental errors so the latter can't masquerade as a
|
||||
// satisfied "must be blocked" assertion.
|
||||
func tryExclusiveFlock(path string) (acquired, blocked bool, err error) {
|
||||
f, err := os.OpenFile(path, os.O_RDONLY, 0)
|
||||
if err != nil {
|
||||
return false, false, err
|
||||
}
|
||||
defer f.Close()
|
||||
if err := syscall.Flock(int(f.Fd()), syscall.LOCK_EX|syscall.LOCK_NB); err != nil {
|
||||
if errors.Is(err, syscall.EWOULDBLOCK) || errors.Is(err, syscall.EAGAIN) {
|
||||
return false, true, nil
|
||||
}
|
||||
return false, false, err
|
||||
}
|
||||
syscall.Flock(int(f.Fd()), syscall.LOCK_UN)
|
||||
return true, false, nil
|
||||
}
|
||||
|
||||
// requireEventuallyBlocked asserts path becomes held by another owner. It polls
|
||||
// because a routed lock RPC can hit a transient EIO (a cold or forwarded gRPC
|
||||
// call under load) even while the lock is genuinely held; it still fails if the
|
||||
// lock is never blocked — whether it can be acquired (a double-grant) or errors
|
||||
// persistently — so an incidental error can't masquerade as "held".
|
||||
func requireEventuallyBlocked(t *testing.T, path string, timeout time.Duration, msg string) {
|
||||
t.Helper()
|
||||
require.Eventuallyf(t, func() bool { return isBlocked(path) },
|
||||
timeout, 500*time.Millisecond, "%s", msg)
|
||||
}
|
||||
|
||||
// isBlocked / isAcquirable are lenient predicates for polling, where transient
|
||||
// errors during a migration are expected and simply mean "not yet".
|
||||
func isBlocked(path string) bool { _, b, _ := tryExclusiveFlock(path); return b }
|
||||
func isAcquirable(path string) bool {
|
||||
a, _, _ := tryExclusiveFlock(path)
|
||||
return a
|
||||
}
|
||||
|
||||
// stopFiler stops filer idx and clears its command so cluster teardown does not
|
||||
// try to stop it again.
|
||||
func (c *dlmTestCluster) stopFiler(idx int) {
|
||||
stopCmd(c.filerCmds[idx])
|
||||
c.filerCmds[idx] = nil
|
||||
}
|
||||
|
||||
// TestPosixLockCrossMount verifies a flock taken on one mount is seen by the
|
||||
// other mount of the same cluster (the routed-to-owner-filer path end to end).
|
||||
func TestPosixLockCrossMount(t *testing.T) {
|
||||
requireForwardedLocks(t)
|
||||
c := startDLMTestCluster(t)
|
||||
|
||||
name := "posix-xmount.lock"
|
||||
path0 := filepath.Join(c.mountPoints[0], name)
|
||||
path1 := filepath.Join(c.mountPoints[1], name)
|
||||
|
||||
createFile(t, path0)
|
||||
waitVisible(t, path1)
|
||||
|
||||
held, err := openFlock(path0, syscall.LOCK_EX)
|
||||
require.NoError(t, err, "mount0 should acquire the exclusive flock")
|
||||
defer held.Close()
|
||||
|
||||
requireEventuallyBlocked(t, path1, 15*time.Second, "mount1 must be blocked while mount0 holds the flock")
|
||||
|
||||
require.NoError(t, syscall.Flock(int(held.Fd()), syscall.LOCK_UN))
|
||||
require.Eventually(t, func() bool { return isAcquirable(path1) },
|
||||
15*time.Second, 500*time.Millisecond,
|
||||
"mount1 should acquire the flock after mount0 releases it")
|
||||
}
|
||||
|
||||
// TestPosixLockSurvivesFilerLoss verifies advisory locks held across mounts
|
||||
// survive a filer leaving the ring: ownership of the affected keys migrates to
|
||||
// the surviving filer and the holding mount re-asserts them there, so the locks
|
||||
// stay honored. It locks many files so that — with 2 filers — several are owned
|
||||
// by filer1 and thus actually migrate when filer1 stops.
|
||||
func TestPosixLockSurvivesFilerLoss(t *testing.T) {
|
||||
requireForwardedLocks(t)
|
||||
c := startDLMTestCluster(t)
|
||||
|
||||
const n = 12
|
||||
|
||||
// Select files via the same ring the filers run so the locked set provably
|
||||
// spans both filers: filer1-owned keys must migrate when filer1 stops, while
|
||||
// filer0-owned keys must keep working. Choosing by ownership (instead of
|
||||
// hoping a sequential set happens to spread) keeps the migration path
|
||||
// exercised on every run, independent of the cluster's dynamic ports.
|
||||
ring := lock_manager.NewHashRing(lock_manager.DefaultVnodeCount)
|
||||
ring.SetServers([]pb.ServerAddress{
|
||||
pb.ServerAddress(c.filerAddress(0)),
|
||||
pb.ServerAddress(c.filerAddress(1)),
|
||||
})
|
||||
filer1 := pb.ServerAddress(c.filerAddress(1))
|
||||
var onFiler1, onFiler0 []string
|
||||
for i := 0; (len(onFiler1) < n/2 || len(onFiler0) < n/2) && i < 1000; i++ {
|
||||
name := fmt.Sprintf("posix-migrate-%d.lock", i)
|
||||
if ring.GetPrimary(posixLockKey(name)) == filer1 {
|
||||
onFiler1 = append(onFiler1, name)
|
||||
} else {
|
||||
onFiler0 = append(onFiler0, name)
|
||||
}
|
||||
}
|
||||
require.GreaterOrEqualf(t, len(onFiler1), n/2, "need %d filer1-owned keys to exercise migration", n/2)
|
||||
require.GreaterOrEqualf(t, len(onFiler0), n/2, "need %d filer0-owned keys", n/2)
|
||||
names := append(onFiler1[:n/2:n/2], onFiler0[:n/2]...)
|
||||
|
||||
held := make([]*os.File, n)
|
||||
for i, name := range names {
|
||||
createFile(t, filepath.Join(c.mountPoints[0], name))
|
||||
waitVisible(t, filepath.Join(c.mountPoints[1], name))
|
||||
f, err := openFlock(filepath.Join(c.mountPoints[0], name), syscall.LOCK_EX)
|
||||
require.NoError(t, err, "mount0 should acquire flock %s", name)
|
||||
held[i] = f
|
||||
defer held[i].Close()
|
||||
}
|
||||
|
||||
require.Eventually(t, func() bool {
|
||||
for _, name := range names {
|
||||
if !isBlocked(filepath.Join(c.mountPoints[1], name)) {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}, 20*time.Second, time.Second,
|
||||
"every lock must be held on mount1 before the ring change")
|
||||
|
||||
// Drop filer1 from the ring; keys it owned migrate to filer0.
|
||||
c.stopFiler(1)
|
||||
require.NoError(t, c.waitForFilerCount(1, 30*time.Second), "ring should drop to one filer")
|
||||
|
||||
// Poll until every lock is honored again on the surviving filer: the ring
|
||||
// must propagate and the holding mount must re-assert its locks (keepalive is
|
||||
// 5s). We assert only the settled state — the transient migration window is
|
||||
// covered by unit tests.
|
||||
require.Eventually(t, func() bool {
|
||||
for _, name := range names {
|
||||
if !isBlocked(filepath.Join(c.mountPoints[1], name)) {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}, 45*time.Second, time.Second,
|
||||
"all locks must survive filer1 leaving the ring (migrate to filer0)")
|
||||
|
||||
// Releasing on mount0 frees them all: mount1 can then acquire each.
|
||||
for _, f := range held {
|
||||
require.NoError(t, syscall.Flock(int(f.Fd()), syscall.LOCK_UN))
|
||||
}
|
||||
require.Eventually(t, func() bool {
|
||||
for _, name := range names {
|
||||
if !isAcquirable(filepath.Join(c.mountPoints[1], name)) {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}, 20*time.Second, 500*time.Millisecond,
|
||||
"mount1 should acquire every lock after mount0 releases them post-migration")
|
||||
}
|
||||
@@ -177,10 +177,8 @@ func testConcurrentReadWrite(t *testing.T, framework *FuseTestFramework) {
|
||||
defer wg.Done()
|
||||
|
||||
for j := 0; j < 10; j++ {
|
||||
if err := retryTransientFUSE(func() error {
|
||||
_, e := os.ReadFile(mountPath)
|
||||
return e
|
||||
}); err != nil {
|
||||
_, err := os.ReadFile(mountPath)
|
||||
if err != nil {
|
||||
addError(fmt.Errorf("reader %d: %v", readerID, err))
|
||||
return
|
||||
}
|
||||
@@ -198,9 +196,8 @@ func testConcurrentReadWrite(t *testing.T, framework *FuseTestFramework) {
|
||||
|
||||
for j := 0; j < 5; j++ {
|
||||
newData := bytes.Repeat([]byte(fmt.Sprintf("WRITER%d", writerID)), 1000)
|
||||
if err := retryTransientFUSE(func() error {
|
||||
return os.WriteFile(mountPath, newData, 0644)
|
||||
}); err != nil {
|
||||
err := os.WriteFile(mountPath, newData, 0644)
|
||||
if err != nil {
|
||||
addError(fmt.Errorf("writer %d: %v", writerID, err))
|
||||
return
|
||||
}
|
||||
@@ -216,21 +213,6 @@ func testConcurrentReadWrite(t *testing.T, framework *FuseTestFramework) {
|
||||
framework.AssertFileExists(filename)
|
||||
}
|
||||
|
||||
// retryTransientFUSE retries op a few times before giving up. A concurrent
|
||||
// truncating overwrite can leave a short-lived dentry/cache window where the
|
||||
// entry is momentarily invisible (ENOENT) to another opener; the last error is
|
||||
// returned so a genuine, persistent failure still surfaces.
|
||||
func retryTransientFUSE(op func() error) error {
|
||||
var err error
|
||||
for attempt := 0; attempt < 5; attempt++ {
|
||||
if err = op(); err == nil {
|
||||
return nil
|
||||
}
|
||||
time.Sleep(100 * time.Millisecond)
|
||||
}
|
||||
return err
|
||||
}
|
||||
|
||||
// testConcurrentDirectoryOperations tests concurrent directory operations
|
||||
func testConcurrentDirectoryOperations(t *testing.T, framework *FuseTestFramework) {
|
||||
numWorkers := 8
|
||||
|
||||
@@ -298,54 +298,6 @@ User Request → Load Balancer → Any S3 Gateway Instance
|
||||
Allow/Deny Request
|
||||
```
|
||||
|
||||
## Trust Policy Conditions
|
||||
|
||||
Step 5 above evaluates the role's trust policy against context keys derived from the
|
||||
OIDC token's claims. The available keys are:
|
||||
|
||||
| Condition key | Source |
|
||||
|---------------|--------|
|
||||
| `oidc:iss` | `iss` claim (issuer URL) |
|
||||
| `oidc:sub` | `sub` claim |
|
||||
| `oidc:aud` | `aud` claim |
|
||||
| `oidc:<claim>` | any other token claim, e.g. `oidc:roles`, `oidc:groups`, `oidc:email` |
|
||||
| `aws:FederatedProvider` | the provider `name` (e.g. `keycloak-oidc`) when its configured issuer matches the token, otherwise the raw issuer URL |
|
||||
| `aws:userid` | `sub` claim (same value as `oidc:sub` during trust-policy evaluation) |
|
||||
| `sts:DurationSeconds` | requested session duration, when supplied |
|
||||
|
||||
During trust-policy evaluation `aws:userid` is the raw `sub` claim. Once the
|
||||
role has been assumed, the keys seen by request authorization differ: there
|
||||
`aws:userid` is a stable per-identity hash of `sub` and `iss` (see
|
||||
`ComputeParentUser`), so do not assume the two contexts carry the same value.
|
||||
|
||||
Custom claims are always exposed under the `oidc:` prefix, so a trust policy must use
|
||||
`oidc:roles` (not a bare `roles`) to match a `roles` claim:
|
||||
|
||||
```json
|
||||
"Condition": {
|
||||
"StringEquals": {
|
||||
"oidc:roles": "s3-admin"
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
A multi-valued claim (such as a `roles` array) matches when any of its values equals
|
||||
the condition value. The same `oidc:` keys can be interpolated into policy resources,
|
||||
e.g. `arn:aws:s3:::bucket/${oidc:sub}/*`.
|
||||
|
||||
### roleMapping vs. trust policy
|
||||
|
||||
A provider's `roleMapping` and a role's trust policy apply to two different entry
|
||||
points and are not interchangeable:
|
||||
|
||||
- **Direct OIDC** — an S3 request carrying `Authorization: Bearer <OIDC-JWT>`. The
|
||||
gateway applies `roleMapping` to choose the caller's role from the token claims; the
|
||||
first matching rule (or `defaultRole`) wins.
|
||||
- **STS `AssumeRoleWithWebIdentity`** — the caller names the role explicitly via
|
||||
`RoleArn`, and that role's trust policy decides whether the assumption is allowed.
|
||||
`roleMapping` does not select the role on this path; instead the token claims are
|
||||
surfaced as the `oidc:` condition keys above for the trust policy to evaluate.
|
||||
|
||||
## Configuration Management
|
||||
|
||||
### Development Environment
|
||||
|
||||
@@ -37,7 +37,7 @@
|
||||
"Action": ["sts:AssumeRoleWithWebIdentity"],
|
||||
"Condition": {
|
||||
"StringEquals": {
|
||||
"oidc:roles": "s3-admin"
|
||||
"roles": "s3-admin"
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -60,7 +60,7 @@
|
||||
"Action": ["sts:AssumeRoleWithWebIdentity"],
|
||||
"Condition": {
|
||||
"StringEquals": {
|
||||
"oidc:roles": "s3-read-only"
|
||||
"roles": "s3-read-only"
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -83,7 +83,7 @@
|
||||
"Action": ["sts:AssumeRoleWithWebIdentity"],
|
||||
"Condition": {
|
||||
"StringEquals": {
|
||||
"oidc:roles": "s3-read-write"
|
||||
"roles": "s3-read-write"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -22,7 +22,6 @@ import (
|
||||
"os"
|
||||
"os/exec"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
@@ -70,111 +69,7 @@ func s3Client(t *testing.T) *s3.Client {
|
||||
})),
|
||||
)
|
||||
require.NoError(t, err)
|
||||
client := s3.NewFromConfig(cfg, func(o *s3.Options) { o.UsePathStyle = true })
|
||||
ensureClusterWritable(t, client)
|
||||
return client
|
||||
}
|
||||
|
||||
var clusterWritableOnce sync.Once
|
||||
|
||||
// ensureClusterWritable blocks until the cluster can actually serve a write,
|
||||
// absorbing the volume-growth warmup window after a fresh start. The Makefile
|
||||
// only waits for the server process to be up ("server up after N s"); it does
|
||||
// not wait for a writable volume, so the first PutObject can race volume growth
|
||||
// and fail with a transient 500 (assign volume: DeadlineExceeded) — the source
|
||||
// of the lifecycle-test flakes. Probing one throwaway write here, once per
|
||||
// process, warms growth so every test's first real write is past that window.
|
||||
// Best-effort: if it never becomes writable, the test's own PutObject surfaces
|
||||
// the failure normally.
|
||||
func ensureClusterWritable(t *testing.T, c *s3.Client) {
|
||||
t.Helper()
|
||||
clusterWritableOnce.Do(func() {
|
||||
bucket := uniqueBucket("warmup")
|
||||
deadline := time.Now().Add(60 * time.Second)
|
||||
|
||||
// try runs fn under a bounded context so a single hung call can't block.
|
||||
try := func(timeout time.Duration, fn func(ctx context.Context) error) error {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), timeout)
|
||||
defer cancel()
|
||||
return fn(ctx)
|
||||
}
|
||||
// probe is a try whose timeout is clamped to the time left before the
|
||||
// deadline, so the whole warmup stays within the budget; ok=false means
|
||||
// the budget is exhausted and the caller should stop.
|
||||
probe := func(fn func(ctx context.Context) error) (err error, ok bool) {
|
||||
remaining := time.Until(deadline)
|
||||
if remaining <= 0 {
|
||||
return nil, false
|
||||
}
|
||||
if remaining > 10*time.Second {
|
||||
remaining = 10 * time.Second
|
||||
}
|
||||
return try(remaining, fn), true
|
||||
}
|
||||
// backoff sleeps attempt*250ms, never past the deadline.
|
||||
backoff := func(attempt int) {
|
||||
d := time.Duration(attempt) * 250 * time.Millisecond
|
||||
if left := time.Until(deadline); d > left {
|
||||
d = left
|
||||
}
|
||||
if d > 0 {
|
||||
time.Sleep(d)
|
||||
}
|
||||
}
|
||||
|
||||
// CreateBucket is a metadata op, but on a cold cluster the filer itself may
|
||||
// not be ready yet, so retry it within the deadline rather than abandoning
|
||||
// the whole warmup (and the PutObject probe) on the first error.
|
||||
created := false
|
||||
for attempt := 1; ; attempt++ {
|
||||
err, ok := probe(func(ctx context.Context) error {
|
||||
_, e := c.CreateBucket(ctx, &s3.CreateBucketInput{Bucket: aws.String(bucket)})
|
||||
return e
|
||||
})
|
||||
if !ok {
|
||||
break
|
||||
}
|
||||
if err == nil {
|
||||
created = true
|
||||
break
|
||||
}
|
||||
backoff(attempt)
|
||||
}
|
||||
if !created {
|
||||
t.Logf("warmup: could not create probe bucket within 60s; proceeding")
|
||||
return
|
||||
}
|
||||
// Cleanup gets a fresh timeout (not the warmup budget) so teardown runs
|
||||
// even when the probe loop used the full window.
|
||||
defer try(10*time.Second, func(ctx context.Context) error {
|
||||
_, e := c.DeleteBucket(ctx, &s3.DeleteBucketInput{Bucket: aws.String(bucket)})
|
||||
return e
|
||||
})
|
||||
|
||||
for attempt := 1; ; attempt++ {
|
||||
err, ok := probe(func(ctx context.Context) error {
|
||||
_, e := c.PutObject(ctx, &s3.PutObjectInput{
|
||||
Bucket: aws.String(bucket), Key: aws.String("warmup"), Body: strings.NewReader("ok"),
|
||||
})
|
||||
return e
|
||||
})
|
||||
if !ok {
|
||||
break
|
||||
}
|
||||
if err == nil {
|
||||
try(10*time.Second, func(ctx context.Context) error {
|
||||
_, e := c.DeleteObject(ctx, &s3.DeleteObjectInput{Bucket: aws.String(bucket), Key: aws.String("warmup")})
|
||||
return e
|
||||
})
|
||||
if attempt > 1 {
|
||||
t.Logf("cluster became writable after %d probe(s)", attempt)
|
||||
}
|
||||
return
|
||||
}
|
||||
backoff(attempt)
|
||||
}
|
||||
t.Logf("warmup: cluster not confirmed writable within 60s; proceeding")
|
||||
})
|
||||
return s3.NewFromConfig(cfg, func(o *s3.Options) { o.UsePathStyle = true })
|
||||
}
|
||||
|
||||
func filerClient(t *testing.T) (filer_pb.SeaweedFilerClient, func()) {
|
||||
|
||||
+2
-25
@@ -11,29 +11,6 @@ import (
|
||||
// GrpcPortOffset is the offset weed mini uses to derive gRPC ports from HTTP ports.
|
||||
const GrpcPortOffset = 10000
|
||||
|
||||
// miniDefaultPorts are the weed mini flag defaults (see weed/command/mini.go).
|
||||
// A test only overrides services it uses; unspecified services still bind
|
||||
// these defaults, so allocation must avoid handing them out (or any value
|
||||
// whose gRPC offset would collide with them).
|
||||
var miniDefaultPorts = []int{
|
||||
9333, // master.port
|
||||
8888, // filer.port
|
||||
9340, // volume.port
|
||||
8333, // s3.port
|
||||
8181, // s3.port.iceberg
|
||||
7333, // webdav.port
|
||||
23646, // admin.port
|
||||
}
|
||||
|
||||
func reservedMiniPorts() map[int]bool {
|
||||
r := make(map[int]bool, len(miniDefaultPorts)*2)
|
||||
for _, p := range miniDefaultPorts {
|
||||
r[p] = true
|
||||
r[p+GrpcPortOffset] = true
|
||||
}
|
||||
return r
|
||||
}
|
||||
|
||||
// AllocatePorts allocates count unique free ports atomically.
|
||||
// All listeners are held open until every port is obtained, preventing
|
||||
// the OS from recycling a port between successive allocations.
|
||||
@@ -86,7 +63,7 @@ func AllocateMiniPorts(count int) ([]int, error) {
|
||||
minPort = 10000
|
||||
maxPort = 55000
|
||||
)
|
||||
reserved := reservedMiniPorts()
|
||||
reserved := make(map[int]bool)
|
||||
ports := make([]int, 0, count)
|
||||
var listeners []net.Listener
|
||||
defer func() {
|
||||
@@ -158,7 +135,7 @@ func AllocatePortSet(miniCount, regularCount int) (mini []int, regular []int, er
|
||||
minPort = 10000
|
||||
maxPort = 55000
|
||||
)
|
||||
reserved := reservedMiniPorts()
|
||||
reserved := make(map[int]bool)
|
||||
mini = make([]int, 0, miniCount)
|
||||
var listeners []net.Listener
|
||||
defer func() {
|
||||
|
||||
@@ -2,29 +2,6 @@ package testutil
|
||||
|
||||
import "testing"
|
||||
|
||||
// AllocateMiniPorts must never hand out a port that weed mini will reserve
|
||||
// for one of its default services (or that default's gRPC offset). A real
|
||||
// failure: Filer was given 33646 (Admin default 23646 + GrpcPortOffset),
|
||||
// which mini then refused as "reserved for gRPC calculation".
|
||||
func TestAllocateMiniPortsAvoidsMiniDefaults(t *testing.T) {
|
||||
reserved := reservedMiniPorts()
|
||||
for iter := 0; iter < 200; iter++ {
|
||||
ports, err := AllocateMiniPorts(4)
|
||||
if err != nil {
|
||||
t.Fatalf("iter %d: AllocateMiniPorts: %v", iter, err)
|
||||
}
|
||||
for _, p := range ports {
|
||||
if reserved[p] {
|
||||
t.Fatalf("iter %d: allocated port %d is a mini default (or gRPC offset)", iter, p)
|
||||
}
|
||||
if reserved[p+GrpcPortOffset] {
|
||||
t.Fatalf("iter %d: allocated port %d has gRPC offset %d colliding with a mini default",
|
||||
iter, p, p+GrpcPortOffset)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestAllocatePortSetNoGrpcCollision(t *testing.T) {
|
||||
// Run a few iterations to catch the OS-recycles-just-closed-port race
|
||||
// that previously hit regular ports when the mini gRPC offset was freed
|
||||
|
||||
@@ -75,7 +75,7 @@ func TestStatsEndpoints(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestStatusPrettyJsonAndCallbackIgnored(t *testing.T) {
|
||||
func TestStatusPrettyJsonAndJsonp(t *testing.T) {
|
||||
if testing.Short() {
|
||||
t.Skip("skipping integration test in short mode")
|
||||
}
|
||||
@@ -93,29 +93,29 @@ func TestStatusPrettyJsonAndCallbackIgnored(t *testing.T) {
|
||||
if len(lines) < 3 {
|
||||
t.Fatalf("/status?pretty=y expected multi-line indented JSON, got %d lines: %s", len(lines), string(prettyBody))
|
||||
}
|
||||
// Verify the body is valid JSON
|
||||
var prettyPayload map[string]interface{}
|
||||
if err := json.Unmarshal(prettyBody, &prettyPayload); err != nil {
|
||||
t.Fatalf("/status?pretty=y is not valid JSON: %v", err)
|
||||
}
|
||||
|
||||
// ?callback=myFunc — must be ignored; response is plain JSON with nosniff.
|
||||
cbResp := framework.DoRequest(t, client, mustNewRequest(t, http.MethodGet, cluster.VolumeAdminURL()+"/status?callback=myFunc"))
|
||||
cbBody := framework.ReadAllAndClose(t, cbResp)
|
||||
if cbResp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("/status?callback=myFunc expected 200, got %d", cbResp.StatusCode)
|
||||
// ?callback=myFunc — expect JSONP wrapping
|
||||
jsonpResp := framework.DoRequest(t, client, mustNewRequest(t, http.MethodGet, cluster.VolumeAdminURL()+"/status?callback=myFunc"))
|
||||
jsonpBody := framework.ReadAllAndClose(t, jsonpResp)
|
||||
if jsonpResp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("/status?callback=myFunc expected 200, got %d", jsonpResp.StatusCode)
|
||||
}
|
||||
if ct := cbResp.Header.Get("Content-Type"); !strings.Contains(ct, "application/json") {
|
||||
t.Fatalf("/status?callback=myFunc expected Content-Type application/json, got %q", ct)
|
||||
bodyStr := string(jsonpBody)
|
||||
if !strings.HasPrefix(bodyStr, "myFunc(") {
|
||||
t.Fatalf("/status?callback=myFunc expected body to start with 'myFunc(', got prefix: %q", bodyStr[:min(len(bodyStr), 30)])
|
||||
}
|
||||
if nosniff := cbResp.Header.Get("X-Content-Type-Options"); nosniff != "nosniff" {
|
||||
t.Fatalf("/status?callback=myFunc expected X-Content-Type-Options nosniff, got %q", nosniff)
|
||||
trimmed := strings.TrimRight(bodyStr, "\n; ")
|
||||
if !strings.HasSuffix(trimmed, ")") {
|
||||
t.Fatalf("/status?callback=myFunc expected body to end with ')', got suffix: %q", trimmed[max(0, len(trimmed)-10):])
|
||||
}
|
||||
if strings.Contains(string(cbBody), "myFunc(") {
|
||||
t.Fatalf("/status?callback=myFunc must not wrap response in callback; body: %q", string(cbBody))
|
||||
}
|
||||
var cbPayload map[string]interface{}
|
||||
if err := json.Unmarshal(cbBody, &cbPayload); err != nil {
|
||||
t.Fatalf("/status?callback=myFunc is not valid JSON: %v", err)
|
||||
// Content-Type should be application/javascript for JSONP
|
||||
if ct := jsonpResp.Header.Get("Content-Type"); !strings.Contains(ct, "javascript") {
|
||||
t.Fatalf("/status?callback=myFunc expected Content-Type containing 'javascript', got %q", ct)
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -23,7 +23,6 @@ import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/plugin_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/schema_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/security"
|
||||
stats_collect "github.com/seaweedfs/seaweedfs/weed/stats"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/erasure_coding"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/super_block"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
@@ -281,8 +280,6 @@ func NewAdminServer(masters string, templateFS http.FileSystem, dataDir string,
|
||||
go server.monitorVacuumWorker(bgCtx)
|
||||
}
|
||||
|
||||
go server.publishMaintenanceMetrics(bgCtx)
|
||||
|
||||
return server
|
||||
}
|
||||
|
||||
@@ -367,55 +364,6 @@ func (s *AdminServer) monitorVacuumWorker(ctx context.Context) {
|
||||
}
|
||||
}
|
||||
|
||||
// publishMaintenanceMetrics periodically snapshots the maintenance queue and
|
||||
// worker fleet into Prometheus gauges. Counters and durations are recorded at
|
||||
// their event sites; these gauges reflect current state at scrape resolution.
|
||||
func (s *AdminServer) publishMaintenanceMetrics(ctx context.Context) {
|
||||
const interval = 15 * time.Second
|
||||
ticker := time.NewTicker(interval)
|
||||
defer ticker.Stop()
|
||||
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case <-ticker.C:
|
||||
s.collectMaintenanceMetrics()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func (s *AdminServer) collectMaintenanceMetrics() {
|
||||
if s.maintenanceManager == nil {
|
||||
return
|
||||
}
|
||||
|
||||
stats := s.maintenanceManager.GetStats()
|
||||
|
||||
stats_collect.AdminMaintenanceTasksByStatus.Reset()
|
||||
for status, count := range stats.TasksByStatus {
|
||||
stats_collect.AdminMaintenanceTasksByStatus.WithLabelValues(string(status)).Set(float64(count))
|
||||
}
|
||||
|
||||
stats_collect.AdminMaintenanceTasksByType.Reset()
|
||||
for taskType, count := range stats.TasksByType {
|
||||
stats_collect.AdminMaintenanceTasksByType.WithLabelValues(string(taskType)).Set(float64(count))
|
||||
}
|
||||
|
||||
// NextScanTime is only meaningful while the scanner runs; GetStats computes
|
||||
// it unconditionally, so clear the gauge when idle to avoid a stale value.
|
||||
if s.maintenanceManager.IsRunning() && !stats.NextScanTime.IsZero() {
|
||||
stats_collect.AdminMaintenanceNextScanTimestampSeconds.Set(float64(stats.NextScanTime.Unix()))
|
||||
} else {
|
||||
stats_collect.AdminMaintenanceNextScanTimestampSeconds.Set(0)
|
||||
}
|
||||
|
||||
workers, usedSlots, maxSlots := s.maintenanceManager.GetWorkerSlotTotals()
|
||||
stats_collect.AdminWorkersConnected.Set(float64(workers))
|
||||
stats_collect.AdminWorkerSlots.WithLabelValues("used").Set(float64(usedSlots))
|
||||
stats_collect.AdminWorkerSlots.WithLabelValues("max").Set(float64(maxSlots))
|
||||
}
|
||||
|
||||
// loadTaskConfigurationsFromPersistence loads saved task configurations from protobuf files
|
||||
func (s *AdminServer) loadTaskConfigurationsFromPersistence() {
|
||||
if s.configPersistence == nil || !s.configPersistence.IsConfigured() {
|
||||
|
||||
@@ -16,7 +16,6 @@ import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/plugin_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/worker_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/security"
|
||||
stats_collect "github.com/seaweedfs/seaweedfs/weed/stats"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
"google.golang.org/grpc"
|
||||
"google.golang.org/grpc/codes"
|
||||
@@ -234,7 +233,6 @@ func (s *WorkerGrpcServer) WorkerStream(stream worker_pb.WorkerService_WorkerStr
|
||||
}
|
||||
s.connections[workerID] = conn
|
||||
s.connMutex.Unlock()
|
||||
stats_collect.AdminWorkerEventsTotal.WithLabelValues("registered").Inc()
|
||||
|
||||
// Register worker with maintenance manager
|
||||
s.registerWorkerWithManager(conn)
|
||||
@@ -267,11 +265,11 @@ func (s *WorkerGrpcServer) WorkerStream(stream worker_pb.WorkerService_WorkerStr
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
glog.Infof("Worker %s connection closed: %v", workerID, ctx.Err())
|
||||
s.unregisterWorker(conn, "unregistered")
|
||||
s.unregisterWorker(conn)
|
||||
return nil
|
||||
case <-connCtx.Done():
|
||||
glog.Infof("Worker %s connection cancelled", workerID)
|
||||
s.unregisterWorker(conn, "unregistered")
|
||||
s.unregisterWorker(conn)
|
||||
return nil
|
||||
default:
|
||||
}
|
||||
@@ -287,7 +285,7 @@ func (s *WorkerGrpcServer) WorkerStream(stream worker_pb.WorkerService_WorkerStr
|
||||
default:
|
||||
glog.Errorf("Error receiving from worker %s: %v", workerID, err)
|
||||
}
|
||||
s.unregisterWorker(conn, "unregistered")
|
||||
s.unregisterWorker(conn)
|
||||
return err
|
||||
}
|
||||
|
||||
@@ -340,7 +338,7 @@ func (s *WorkerGrpcServer) handleWorkerMessage(conn *WorkerConnection, msg *work
|
||||
|
||||
case *worker_pb.WorkerMessage_Shutdown:
|
||||
glog.Infof("Worker %s shutting down: %s", workerID, m.Shutdown.Reason)
|
||||
s.unregisterWorker(conn, "unregistered")
|
||||
s.unregisterWorker(conn)
|
||||
|
||||
default:
|
||||
glog.Warningf("Unknown message type from worker %s", workerID)
|
||||
@@ -607,7 +605,7 @@ func (s *WorkerGrpcServer) safeCloseOutgoingChannel(conn *WorkerConnection, sour
|
||||
}
|
||||
|
||||
// unregisterWorker removes a worker connection
|
||||
func (s *WorkerGrpcServer) unregisterWorker(conn *WorkerConnection, event string) {
|
||||
func (s *WorkerGrpcServer) unregisterWorker(conn *WorkerConnection) {
|
||||
s.connMutex.Lock()
|
||||
existingConn, exists := s.connections[conn.workerID]
|
||||
if !exists {
|
||||
@@ -626,7 +624,6 @@ func (s *WorkerGrpcServer) unregisterWorker(conn *WorkerConnection, event string
|
||||
// Remove from map first to prevent duplicate cleanup attempts
|
||||
delete(s.connections, conn.workerID)
|
||||
s.connMutex.Unlock()
|
||||
stats_collect.AdminWorkerEventsTotal.WithLabelValues(event).Inc()
|
||||
|
||||
// Cancel context to signal goroutines to stop
|
||||
conn.cancel()
|
||||
@@ -668,7 +665,7 @@ func (s *WorkerGrpcServer) cleanupStaleConnections() {
|
||||
|
||||
for _, conn := range toRemove {
|
||||
glog.Warningf("Cleaning up stale worker connection: %s", conn.workerID)
|
||||
s.unregisterWorker(conn, "stale_removed")
|
||||
s.unregisterWorker(conn)
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -8,7 +8,6 @@ import (
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/worker_pb"
|
||||
stats_collect "github.com/seaweedfs/seaweedfs/weed/stats"
|
||||
"github.com/seaweedfs/seaweedfs/weed/worker/tasks/balance"
|
||||
"github.com/seaweedfs/seaweedfs/weed/worker/tasks/erasure_coding"
|
||||
"github.com/seaweedfs/seaweedfs/weed/worker/tasks/vacuum"
|
||||
@@ -316,7 +315,6 @@ func (mm *MaintenanceManager) performScan() {
|
||||
glog.Infof("Starting maintenance scan...")
|
||||
|
||||
results, err := mm.scanner.ScanForMaintenanceTasks()
|
||||
stats_collect.AdminMaintenanceLastScanTimestampSeconds.SetToCurrentTime()
|
||||
if err != nil {
|
||||
// Handle scan error
|
||||
mm.mutex.Lock()
|
||||
@@ -520,11 +518,6 @@ func (mm *MaintenanceManager) GetWorkers() []*MaintenanceWorker {
|
||||
return mm.queue.GetWorkers()
|
||||
}
|
||||
|
||||
// GetWorkerSlotTotals returns worker count and aggregate used/max task slots.
|
||||
func (mm *MaintenanceManager) GetWorkerSlotTotals() (workers, used, max int) {
|
||||
return mm.queue.GetWorkerSlotTotals()
|
||||
}
|
||||
|
||||
// TriggerScan manually triggers a maintenance scan
|
||||
func (mm *MaintenanceManager) TriggerScan() error {
|
||||
return mm.triggerScanInternal(true)
|
||||
|
||||
@@ -8,7 +8,6 @@ import (
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
stats_collect "github.com/seaweedfs/seaweedfs/weed/stats"
|
||||
)
|
||||
|
||||
// NewMaintenanceQueue creates a new maintenance queue
|
||||
@@ -483,14 +482,12 @@ func (mq *MaintenanceQueue) CompleteTask(taskID string, error string) {
|
||||
|
||||
// Calculate task duration
|
||||
var duration time.Duration
|
||||
hadStart := task.StartedAt != nil
|
||||
if hadStart {
|
||||
if task.StartedAt != nil {
|
||||
duration = completedTime.Sub(*task.StartedAt)
|
||||
}
|
||||
|
||||
// Capture workerID before it may be cleared during retry
|
||||
originalWorkerID := task.WorkerID
|
||||
taskType := string(task.Type)
|
||||
|
||||
var taskToSave *MaintenanceTask
|
||||
var logFn func()
|
||||
@@ -580,18 +577,6 @@ func (mq *MaintenanceQueue) CompleteTask(taskID string, error string) {
|
||||
}
|
||||
mq.mutex.Unlock()
|
||||
|
||||
// Record terminal-state metrics. A retry leaves the task pending, so it
|
||||
// is not counted as completed or failed here.
|
||||
switch taskStatus {
|
||||
case TaskStatusCompleted:
|
||||
stats_collect.AdminMaintenanceTasksCompletedTotal.WithLabelValues(taskType, "completed").Inc()
|
||||
case TaskStatusFailed:
|
||||
stats_collect.AdminMaintenanceTasksCompletedTotal.WithLabelValues(taskType, "failed").Inc()
|
||||
}
|
||||
if hadStart && (taskStatus == TaskStatusCompleted || taskStatus == TaskStatusFailed) {
|
||||
stats_collect.AdminMaintenanceTaskDurationSeconds.WithLabelValues(taskType).Observe(duration.Seconds())
|
||||
}
|
||||
|
||||
// Only persist non-terminal tasks (retries). Completed/failed tasks stay
|
||||
// in memory for the UI but are not written to disk — they would just
|
||||
// accumulate and slow down future startups.
|
||||
@@ -834,20 +819,6 @@ func (mq *MaintenanceQueue) GetWorkers() []*MaintenanceWorker {
|
||||
return workers
|
||||
}
|
||||
|
||||
// GetWorkerSlotTotals aggregates worker count and used/max task slots under the
|
||||
// lock, so callers don't read live worker fields that task updates mutate.
|
||||
func (mq *MaintenanceQueue) GetWorkerSlotTotals() (workers, used, max int) {
|
||||
mq.mutex.RLock()
|
||||
defer mq.mutex.RUnlock()
|
||||
|
||||
for _, worker := range mq.workers {
|
||||
workers++
|
||||
used += worker.CurrentLoad
|
||||
max += worker.MaxConcurrent
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
// generateTaskID generates a unique ID for tasks
|
||||
func generateTaskID() string {
|
||||
const charset = "abcdefghijklmnopqrstuvwxyz0123456789"
|
||||
|
||||
@@ -54,70 +54,6 @@ func (at *ActiveTopology) GetEffectiveAvailableCapacityDetailed(nodeID string, d
|
||||
return at.getEffectiveAvailableCapacityUnsafe(disk)
|
||||
}
|
||||
|
||||
// GetEffectiveAvailableEcShardSlots returns a disk's free EC shard slots,
|
||||
// accounting for in-flight task reservations at shard granularity. Unlike the
|
||||
// volume-slot views (GetDisksWithEffectiveCapacity / GetEffectiveAvailableCapacity),
|
||||
// this does not truncate sub-volume shard reservations: it subtracts the full
|
||||
// reservation impact (volume slots converted to shard slots, plus the raw shard
|
||||
// slots) so a reservation that is not a whole multiple of ShardsPerVolumeSlot is
|
||||
// not lost. It does NOT subtract the EC shards already persisted on the disk;
|
||||
// callers that track those (from EcShardInfos) subtract them separately.
|
||||
//
|
||||
// shardsPerVolume is the number of EC shards of the target collection that fit in
|
||||
// one volume slot (i.e. its data-shard count): a 4+2 volume's shards are ~1/4 of a
|
||||
// volume each, so one volume slot holds 4 of them, not the default
|
||||
// ShardsPerVolumeSlot. Pass <= 0 to use the default. Using the target ratio keeps
|
||||
// Place from over-filling a disk for low-data-shard layouts.
|
||||
func (at *ActiveTopology) GetEffectiveAvailableEcShardSlots(nodeID string, diskID uint32, shardsPerVolume int) int {
|
||||
if shardsPerVolume <= 0 {
|
||||
shardsPerVolume = ShardsPerVolumeSlot
|
||||
}
|
||||
|
||||
at.mutex.RLock()
|
||||
defer at.mutex.RUnlock()
|
||||
|
||||
diskKey := fmt.Sprintf("%s:%d", nodeID, diskID)
|
||||
disk, exists := at.disks[diskKey]
|
||||
if !exists || disk.DiskInfo == nil || disk.DiskInfo.DiskInfo == nil {
|
||||
return 0
|
||||
}
|
||||
|
||||
info := disk.DiskInfo.DiskInfo
|
||||
base := info.MaxVolumeCount - info.VolumeCount
|
||||
if base <= 0 && info.MaxVolumeCount == 0 && info.VolumeCount == 0 &&
|
||||
len(info.VolumeInfos) == 0 && len(info.EcShardInfos) == 0 {
|
||||
// Freshly started empty servers can report max=0 before publishing concrete
|
||||
// limits; keep one provisional slot so EC placement still sees the disk,
|
||||
// mirroring getEffectiveAvailableCapacityUnsafe.
|
||||
base = 1
|
||||
}
|
||||
if base < 0 {
|
||||
base = 0
|
||||
}
|
||||
// calculateTaskStorageImpact reports consumption as positive, so subtract it.
|
||||
// Volume-slot reservations scale by the target ratio; the sub-volume shard-slot
|
||||
// remainder is in default units and subtracted as-is (a small approximation).
|
||||
impact := at.getEffectiveCapacityUnsafe(disk)
|
||||
// impact.ShardSlots is recorded in default ShardsPerVolumeSlot units; convert it
|
||||
// to the target ratio's shard slots before subtracting (identity when
|
||||
// shardsPerVolume == ShardsPerVolumeSlot). Round a positive reservation up so a
|
||||
// sub-slot reservation (e.g. 1 default slot against a 4-shard target) is not
|
||||
// truncated to zero and wrongly counted as free.
|
||||
scaledShardImpact := int64(impact.ShardSlots) * int64(shardsPerVolume)
|
||||
if scaledShardImpact > 0 {
|
||||
scaledShardImpact = (scaledShardImpact + int64(ShardsPerVolumeSlot) - 1) / int64(ShardsPerVolumeSlot)
|
||||
} else {
|
||||
scaledShardImpact /= int64(ShardsPerVolumeSlot)
|
||||
}
|
||||
free := base*int64(shardsPerVolume) -
|
||||
int64(impact.VolumeSlots)*int64(shardsPerVolume) -
|
||||
scaledShardImpact
|
||||
if free < 0 {
|
||||
free = 0
|
||||
}
|
||||
return int(free)
|
||||
}
|
||||
|
||||
// GetEffectiveCapacityImpact returns the StorageSlotChange impact for a disk
|
||||
// This shows the net impact from all pending and assigned tasks
|
||||
func (at *ActiveTopology) GetEffectiveCapacityImpact(nodeID string, diskID uint32) StorageSlotChange {
|
||||
|
||||
@@ -4,7 +4,6 @@ import (
|
||||
"context"
|
||||
"fmt"
|
||||
"strings"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
|
||||
@@ -20,14 +19,6 @@ type LockClient struct {
|
||||
maxLockDuration time.Duration
|
||||
sleepDuration time.Duration
|
||||
seedFiler pb.ServerAddress
|
||||
|
||||
// ring is an optional client-side view of the filer lock hash ring. When
|
||||
// populated, a new lock starts at the key's primary filer instead of the
|
||||
// seed filer, avoiding the seed->primary forward hop. A stale view stays
|
||||
// correct: the filer forwards to the real primary as a fallback.
|
||||
ringMu sync.RWMutex
|
||||
ring *lock_manager.HashRing
|
||||
ringVersion int64
|
||||
}
|
||||
|
||||
func NewLockClient(grpcDialOption grpc.DialOption, seedFiler pb.ServerAddress) *LockClient {
|
||||
@@ -39,49 +30,6 @@ func NewLockClient(grpcDialOption grpc.DialOption, seedFiler pb.ServerAddress) *
|
||||
}
|
||||
}
|
||||
|
||||
// SetRing mirrors the master's LockRingUpdate so the client computes the same
|
||||
// primary the filers do. A non-zero version at or below the current one is
|
||||
// ignored once a ring exists, dropping reordered and redundant broadcasts;
|
||||
// version 0 always applies (bootstrap).
|
||||
func (lc *LockClient) SetRing(servers []pb.ServerAddress, version int64) {
|
||||
lc.ringMu.Lock()
|
||||
defer lc.ringMu.Unlock()
|
||||
if version != 0 && version <= lc.ringVersion && lc.ring != nil {
|
||||
return
|
||||
}
|
||||
lc.ringVersion = version
|
||||
if lc.ring == nil {
|
||||
lc.ring = lock_manager.NewHashRing(lock_manager.DefaultVnodeCount)
|
||||
}
|
||||
lc.ring.SetServers(servers)
|
||||
}
|
||||
|
||||
// hostForKey returns the filer that should own key per the current ring view,
|
||||
// falling back to the seed filer when no view has been received yet.
|
||||
func (lc *LockClient) hostForKey(key string) pb.ServerAddress {
|
||||
lc.ringMu.RLock()
|
||||
defer lc.ringMu.RUnlock()
|
||||
if lc.ring == nil {
|
||||
return lc.seedFiler
|
||||
}
|
||||
if primary := lc.ring.GetPrimary(key); primary != "" {
|
||||
return primary
|
||||
}
|
||||
return lc.seedFiler
|
||||
}
|
||||
|
||||
// PrimaryForKey returns the ring owner for key, or "" before any ring arrives.
|
||||
// Unlike hostForKey it does not fall back to the seed, so a route-by-key caller
|
||||
// stays on the distributed lock until the ring is known.
|
||||
func (lc *LockClient) PrimaryForKey(key string) pb.ServerAddress {
|
||||
lc.ringMu.RLock()
|
||||
defer lc.ringMu.RUnlock()
|
||||
if lc.ring == nil {
|
||||
return ""
|
||||
}
|
||||
return lc.ring.GetPrimary(key)
|
||||
}
|
||||
|
||||
type LiveLock struct {
|
||||
key string
|
||||
renewToken string
|
||||
@@ -103,7 +51,7 @@ type LiveLock struct {
|
||||
func (lc *LockClient) NewShortLivedLock(key string, owner string) (lock *LiveLock) {
|
||||
lock = &LiveLock{
|
||||
key: key,
|
||||
hostFiler: lc.hostForKey(key),
|
||||
hostFiler: lc.seedFiler,
|
||||
cancelCh: make(chan struct{}),
|
||||
expireAtNs: time.Now().Add(5 * time.Second).UnixNano(),
|
||||
grpcDialOption: lc.grpcDialOption,
|
||||
@@ -124,7 +72,7 @@ func (lc *LockClient) NewBlockingLongLivedLock(key, owner string, lockTTL time.D
|
||||
}
|
||||
lock := &LiveLock{
|
||||
key: key,
|
||||
hostFiler: lc.hostForKey(key),
|
||||
hostFiler: lc.seedFiler,
|
||||
cancelCh: make(chan struct{}),
|
||||
expireAtNs: time.Now().Add(lockTTL).UnixNano(),
|
||||
grpcDialOption: lc.grpcDialOption,
|
||||
@@ -162,7 +110,7 @@ func (lc *LockClient) NewBlockingLongLivedLock(key, owner string, lockTTL time.D
|
||||
func (lc *LockClient) StartLongLivedLock(key string, owner string, onLockOwnerChange func(newLockOwner string), lockTTL time.Duration) (lock *LiveLock) {
|
||||
lock = &LiveLock{
|
||||
key: key,
|
||||
hostFiler: lc.hostForKey(key),
|
||||
hostFiler: lc.seedFiler,
|
||||
cancelCh: make(chan struct{}),
|
||||
expireAtNs: time.Now().Add(lockTTL).UnixNano(),
|
||||
grpcDialOption: lc.grpcDialOption,
|
||||
|
||||
@@ -1,92 +0,0 @@
|
||||
package cluster
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/cluster/lock_manager"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
)
|
||||
|
||||
// The gateway must resolve a lock key to the same primary the filers do,
|
||||
// otherwise it dials the wrong filer and the lock still gets forwarded. Both
|
||||
// sides use the same HashRing over the same server set, so for every key the
|
||||
// client's hostForKey must equal the filer ring's GetPrimary.
|
||||
func TestLockClientHostMatchesFilerRing(t *testing.T) {
|
||||
servers := []pb.ServerAddress{
|
||||
"filer-a:8888", "filer-b:8888", "filer-c:8888", "filer-d:8888",
|
||||
}
|
||||
|
||||
filerRing := lock_manager.NewHashRing(lock_manager.DefaultVnodeCount)
|
||||
filerRing.SetServers(servers)
|
||||
|
||||
lc := NewLockClient(nil, "seed:8888")
|
||||
lc.SetRing(servers, 1)
|
||||
|
||||
for _, key := range []string{
|
||||
"s3.object.write:/buckets/b/obj-0",
|
||||
"s3.object.write:/buckets/b/obj-1",
|
||||
"s3.object.write:/buckets/b/obj-2",
|
||||
"s3.object.write:/buckets/gosbench-0/w0obj-kilo-0877",
|
||||
"some/other/key",
|
||||
} {
|
||||
if got, want := lc.hostForKey(key), filerRing.GetPrimary(key); got != want {
|
||||
t.Errorf("key %q: client host %q != filer primary %q", key, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Without a ring view, the client falls back to the seed filer (which the filer
|
||||
// forwards from), preserving the pre-optimization behavior.
|
||||
func TestLockClientHostFallsBackToSeed(t *testing.T) {
|
||||
lc := NewLockClient(nil, "seed:8888")
|
||||
if got := lc.hostForKey("any-key"); got != "seed:8888" {
|
||||
t.Errorf("expected seed fallback, got %q", got)
|
||||
}
|
||||
|
||||
// An empty ring (no members yet) also falls back to the seed.
|
||||
lc.SetRing(nil, 1)
|
||||
if got := lc.hostForKey("any-key"); got != "seed:8888" {
|
||||
t.Errorf("expected seed fallback on empty ring, got %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
// A stale (older-version) update must not regress a newer ring view, while
|
||||
// version 0 always applies as a bootstrap.
|
||||
func TestLockClientSetRingVersionGuard(t *testing.T) {
|
||||
lc := NewLockClient(nil, "seed:8888")
|
||||
|
||||
newer := []pb.ServerAddress{"filer-a:8888", "filer-b:8888"}
|
||||
lc.SetRing(newer, 10)
|
||||
primaryAt10 := lc.hostForKey("k")
|
||||
|
||||
// Older version is ignored.
|
||||
lc.SetRing([]pb.ServerAddress{"filer-z:8888"}, 5)
|
||||
if got := lc.hostForKey("k"); got != primaryAt10 {
|
||||
t.Errorf("stale update applied: host changed to %q", got)
|
||||
}
|
||||
|
||||
// version 0 is always accepted.
|
||||
lc.SetRing([]pb.ServerAddress{"filer-z:8888"}, 0)
|
||||
if got := lc.hostForKey("k"); got != "filer-z:8888" {
|
||||
t.Errorf("bootstrap update not applied, got %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
// PrimaryForKey returns "" before any ring is received (so a route-by-key
|
||||
// caller falls back to the distributed lock) and the ring owner afterwards,
|
||||
// unlike hostForKey which falls back to the seed.
|
||||
func TestLockClientPrimaryForKey(t *testing.T) {
|
||||
lc := NewLockClient(nil, "seed:8888")
|
||||
if got := lc.PrimaryForKey("k"); got != "" {
|
||||
t.Errorf("expected empty before ring, got %q", got)
|
||||
}
|
||||
|
||||
lc.SetRing([]pb.ServerAddress{"filer-a:8888", "filer-b:8888"}, 1)
|
||||
got := lc.PrimaryForKey("k")
|
||||
if got == "" {
|
||||
t.Fatal("expected an owner after ring set")
|
||||
}
|
||||
if got != lc.hostForKey("k") {
|
||||
t.Errorf("PrimaryForKey %q disagrees with hostForKey %q", got, lc.hostForKey("k"))
|
||||
}
|
||||
}
|
||||
@@ -145,15 +145,7 @@ func (hr *HashRing) rebuildRing() {
|
||||
hr.vnodeToServer = make(map[uint32]pb.ServerAddress, len(hr.servers)*hr.vnodeCount)
|
||||
hr.sortedHashes = make([]uint32, 0, len(hr.servers)*hr.vnodeCount)
|
||||
|
||||
// Sort so a vnode-hash collision resolves to the same server on every node;
|
||||
// map iteration order alone is randomized per process.
|
||||
servers := make([]pb.ServerAddress, 0, len(hr.servers))
|
||||
for server := range hr.servers {
|
||||
servers = append(servers, server)
|
||||
}
|
||||
sort.Slice(servers, func(i, j int) bool { return servers[i] < servers[j] })
|
||||
|
||||
for _, server := range servers {
|
||||
for i := 0; i < hr.vnodeCount; i++ {
|
||||
vnodeKey := vnodeKeyFor(server, i)
|
||||
hash := hashKey(vnodeKey)
|
||||
|
||||
@@ -171,38 +171,3 @@ func TestHashRing_GetPrimary(t *testing.T) {
|
||||
primary, _ := hr.GetPrimaryAndBackup("mykey")
|
||||
assert.Equal(t, primary, hr.GetPrimary("mykey"))
|
||||
}
|
||||
|
||||
// The ring must be identical on every node holding the same server set,
|
||||
// regardless of the order servers were added or supplied. Build rings several
|
||||
// ways and assert they agree on the primary for a wide range of keys.
|
||||
func TestHashRing_OrderIndependent(t *testing.T) {
|
||||
servers := []pb.ServerAddress{
|
||||
"filer-a:8888", "filer-b:8888", "filer-c:8888", "filer-d:8888", "filer-e:8888",
|
||||
}
|
||||
|
||||
bySet := NewHashRing(50)
|
||||
bySet.SetServers(servers)
|
||||
|
||||
byReverse := NewHashRing(50)
|
||||
rev := append([]pb.ServerAddress(nil), servers...)
|
||||
for i, j := 0, len(rev)-1; i < j; i, j = i+1, j-1 {
|
||||
rev[i], rev[j] = rev[j], rev[i]
|
||||
}
|
||||
byReverse.SetServers(rev)
|
||||
|
||||
byAdd := NewHashRing(50)
|
||||
for _, s := range []pb.ServerAddress{"filer-c:8888", "filer-e:8888", "filer-a:8888", "filer-d:8888", "filer-b:8888"} {
|
||||
byAdd.AddServer(s)
|
||||
}
|
||||
|
||||
for i := 0; i < 5000; i++ {
|
||||
key := fmt.Sprintf("s3.object.write:/buckets/b/obj-%d", i)
|
||||
p := bySet.GetPrimary(key)
|
||||
if got := byReverse.GetPrimary(key); got != p {
|
||||
t.Fatalf("reverse-order ring disagrees on %q: %s vs %s", key, got, p)
|
||||
}
|
||||
if got := byAdd.GetPrimary(key); got != p {
|
||||
t.Fatalf("add-order ring disagrees on %q: %s vs %s", key, got, p)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -12,7 +12,6 @@ import (
|
||||
type LockRingSnapshot struct {
|
||||
servers []pb.ServerAddress
|
||||
ts time.Time
|
||||
ring *HashRing // prebuilt ring for servers, so PriorOwner need not rebuild per call
|
||||
}
|
||||
|
||||
type LockRing struct {
|
||||
@@ -59,12 +58,10 @@ func (r *LockRing) SetSnapshot(servers []pb.ServerAddress, version int64) bool {
|
||||
// are always consistent — prevents a concurrent SetSnapshot from
|
||||
// seeing the new version but applying its servers to the old ring.
|
||||
r.Ring.SetServers(servers)
|
||||
// Append the snapshot under the same lock as the ring update so a concurrent
|
||||
// PriorOwner always sees snapshots[0] matching r.Ring (and snapshots[1] as the
|
||||
// true prior); otherwise it could pair a new ring with a stale prior snapshot.
|
||||
r.addOneSnapshotLocked(servers)
|
||||
r.Unlock()
|
||||
|
||||
r.addOneSnapshot(servers)
|
||||
|
||||
r.cleanupWg.Add(1)
|
||||
go func() {
|
||||
defer r.cleanupWg.Done()
|
||||
@@ -81,16 +78,14 @@ func (r *LockRing) Version() int64 {
|
||||
return r.version
|
||||
}
|
||||
|
||||
// addOneSnapshotLocked appends a new snapshot (newest at index 0). The caller
|
||||
// must hold r.Lock(), so the ring update and snapshot append are one atomic step.
|
||||
func (r *LockRing) addOneSnapshotLocked(servers []pb.ServerAddress) {
|
||||
func (r *LockRing) addOneSnapshot(servers []pb.ServerAddress) {
|
||||
r.Lock()
|
||||
defer r.Unlock()
|
||||
|
||||
ts := time.Now()
|
||||
ring := NewHashRing(DefaultVnodeCount)
|
||||
ring.SetServers(servers)
|
||||
t := &LockRingSnapshot{
|
||||
servers: servers,
|
||||
ts: ts,
|
||||
ring: ring,
|
||||
}
|
||||
r.snapshots = append(r.snapshots, t)
|
||||
for i := len(r.snapshots) - 2; i >= 0; i-- {
|
||||
@@ -153,35 +148,6 @@ func (r *LockRing) GetPrimary(key string) pb.ServerAddress {
|
||||
return r.Ring.GetPrimary(key)
|
||||
}
|
||||
|
||||
// PriorOwner returns the key's owner from the previous ring snapshot, but only
|
||||
// while the ring changed within the last snapshotInterval and that owner differs
|
||||
// from the current primary. This is the cooling-off window in which the previous
|
||||
// owner may still hold locks the new owner has not yet rebuilt — a caller can
|
||||
// consult it before granting so a fresh owner does not double-grant during a
|
||||
// rebalance. Returns "" outside the window or when ownership did not move. It
|
||||
// uses the snapshot's prebuilt ring, so it does not rebuild a hash ring per call.
|
||||
func (r *LockRing) PriorOwner(key string) pb.ServerAddress {
|
||||
r.RLock()
|
||||
defer r.RUnlock()
|
||||
if len(r.snapshots) < 2 {
|
||||
return ""
|
||||
}
|
||||
if time.Since(r.snapshots[0].ts) > r.snapshotInterval {
|
||||
return ""
|
||||
}
|
||||
current := r.Ring.GetPrimary(key)
|
||||
var prior pb.ServerAddress
|
||||
if pr := r.snapshots[1].ring; pr != nil {
|
||||
prior = pr.GetPrimary(key)
|
||||
} else {
|
||||
prior = hashKeyToServer(key, r.snapshots[1].servers)
|
||||
}
|
||||
if prior != "" && prior != current {
|
||||
return prior
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// hashKeyToServer uses a temporary consistent hash ring for the given server list.
|
||||
func hashKeyToServer(key string, servers []pb.ServerAddress) pb.ServerAddress {
|
||||
if len(servers) == 0 {
|
||||
|
||||
@@ -1,63 +0,0 @@
|
||||
package lock_manager
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
)
|
||||
|
||||
func TestLockRing_PriorOwner(t *testing.T) {
|
||||
r := NewLockRing(5 * time.Second)
|
||||
t.Cleanup(r.WaitForCleanup)
|
||||
|
||||
setA := []pb.ServerAddress{"s1:1", "s2:1", "s3:1"}
|
||||
r.SetSnapshot(setA, 1)
|
||||
|
||||
// Only one snapshot: nothing to fall back to.
|
||||
if got := r.PriorOwner("any"); got != "" {
|
||||
t.Fatalf("single snapshot should have no prior owner, got %q", got)
|
||||
}
|
||||
|
||||
// Add a server so some keys' ownership moves.
|
||||
setB := []pb.ServerAddress{"s1:1", "s2:1", "s3:1", "s4:1"}
|
||||
r.SetSnapshot(setB, 2)
|
||||
|
||||
var moved, stable string
|
||||
for i := 0; i < 2000 && (moved == "" || stable == ""); i++ {
|
||||
key := fmt.Sprintf("key-%d", i)
|
||||
if r.GetPrimary(key) != hashKeyToServer(key, setA) {
|
||||
if moved == "" {
|
||||
moved = key
|
||||
}
|
||||
} else if stable == "" {
|
||||
stable = key
|
||||
}
|
||||
}
|
||||
if moved == "" || stable == "" {
|
||||
t.Skip("could not find both a moved and a stable key")
|
||||
}
|
||||
|
||||
if got, want := r.PriorOwner(moved), hashKeyToServer(moved, setA); got != want {
|
||||
t.Fatalf("PriorOwner(moved)=%q, want %q", got, want)
|
||||
}
|
||||
if got := r.PriorOwner(stable); got != "" {
|
||||
t.Fatalf("unmoved key should have no prior owner, got %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestLockRing_PriorOwnerExpires(t *testing.T) {
|
||||
r := NewLockRing(20 * time.Millisecond)
|
||||
t.Cleanup(r.WaitForCleanup)
|
||||
r.SetSnapshot([]pb.ServerAddress{"s1:1", "s2:1", "s3:1"}, 1)
|
||||
r.SetSnapshot([]pb.ServerAddress{"s1:1", "s2:1", "s3:1", "s4:1"}, 2)
|
||||
|
||||
// Past the cooling interval, the prior owner is no longer offered.
|
||||
time.Sleep(40 * time.Millisecond)
|
||||
for i := 0; i < 2000; i++ {
|
||||
if got := r.PriorOwner(fmt.Sprintf("key-%d", i)); got != "" {
|
||||
t.Fatalf("prior owner should expire after the cooling interval, got %q", got)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -30,7 +30,6 @@ import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/security"
|
||||
stats_collect "github.com/seaweedfs/seaweedfs/weed/stats"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util/grace"
|
||||
)
|
||||
@@ -51,8 +50,6 @@ type AdminOptions struct {
|
||||
dataDir *string
|
||||
icebergPort *int
|
||||
urlPrefix *string
|
||||
metricsHttpPort *int
|
||||
metricsHttpIp *string
|
||||
debug *bool
|
||||
debugPort *int
|
||||
cpuProfile *string
|
||||
@@ -73,8 +70,6 @@ func init() {
|
||||
a.readOnlyPassword = cmdAdmin.Flag.String("readOnlyPassword", "", "read-only user password (optional, for view-only access; requires adminPassword to be set)")
|
||||
a.icebergPort = cmdAdmin.Flag.Int("iceberg.port", 8181, "Iceberg REST Catalog port (0 to hide in UI)")
|
||||
a.urlPrefix = cmdAdmin.Flag.String("urlPrefix", "", "URL path prefix when running behind a reverse proxy under a subdirectory (e.g. /seaweedfs)")
|
||||
a.metricsHttpPort = cmdAdmin.Flag.Int("metricsPort", 0, "Prometheus metrics listen port")
|
||||
a.metricsHttpIp = cmdAdmin.Flag.String("metricsIp", "", "metrics listen ip. If empty, listens on all interfaces.")
|
||||
a.debug = cmdAdmin.Flag.Bool("debug", false, "serves runtime profiling data via pprof on the port specified by -debug.port")
|
||||
a.debugPort = cmdAdmin.Flag.Int("debug.port", 6060, "http port for debugging")
|
||||
a.cpuProfile = cmdAdmin.Flag.String("cpuprofile", "", "cpu profile output file")
|
||||
@@ -165,12 +160,6 @@ var cmdAdmin = &Command{
|
||||
weed admin -debug -debug.port=6060 -master="localhost:9333"
|
||||
weed admin -cpuprofile=cpu.prof -memprofile=mem.prof -master="localhost:9333"
|
||||
|
||||
Metrics:
|
||||
- Use -metricsPort to expose Prometheus metrics at http://<host>:<metricsPort>/metrics
|
||||
- Use -metricsIp to bind the metrics endpoint to a specific ip (default: all interfaces)
|
||||
- Metrics are disabled when -metricsPort is 0 (the default)
|
||||
- Example: weed admin -metricsPort=9327 -master="localhost:9333"
|
||||
|
||||
Configuration File:
|
||||
- The security.toml file is read from ".", "$HOME/.seaweedfs/",
|
||||
"/usr/local/etc/seaweedfs/", or "/etc/seaweedfs/", in that order
|
||||
@@ -268,12 +257,6 @@ func runAdmin(cmd *Command, args []string) bool {
|
||||
}
|
||||
fmt.Printf("Plugin: Enabled\n")
|
||||
|
||||
// Start Prometheus metrics endpoint if a port is configured
|
||||
if *a.metricsHttpPort > 0 {
|
||||
fmt.Printf("Metrics: http://%s/metrics\n", stats_collect.JoinHostPort(*a.metricsHttpIp, *a.metricsHttpPort))
|
||||
}
|
||||
go stats_collect.StartMetricsServer(*a.metricsHttpIp, *a.metricsHttpPort)
|
||||
|
||||
// Set up graceful shutdown
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
|
||||
@@ -61,8 +61,7 @@ type MountOptions struct {
|
||||
|
||||
dirIdleEvictSec *int
|
||||
|
||||
// Distributed locking for cross-mount write coordination and POSIX
|
||||
// advisory locks (flock/fcntl)
|
||||
// Distributed lock for cross-mount write coordination
|
||||
distributedLock *bool
|
||||
|
||||
// POSIX compliance options
|
||||
@@ -153,7 +152,7 @@ func init() {
|
||||
mountReadRetryTime = cmdMount.Flag.Duration("readRetryTime", 6*time.Second, "maximum read retry wait time")
|
||||
|
||||
// Distributed lock for cross-mount write coordination
|
||||
mountOptions.distributedLock = cmdMount.Flag.Bool("dlm", false, "coordinate writes across mounts (only one mount writes a file at a time) and honor POSIX advisory locks (flock/fcntl) across mounts by routing them to the owner filer")
|
||||
mountOptions.distributedLock = cmdMount.Flag.Bool("dlm", false, "enable distributed lock for cross-mount write coordination (only one mount can write a file at a time)")
|
||||
|
||||
// POSIX compliance options
|
||||
mountOptions.posixDirNlink = cmdMount.Flag.Bool("posix.dirNLink", false, "report POSIX-compliant directory nlink (2 + subdirectory count); costs one directory listing per stat")
|
||||
|
||||
+2
-9
@@ -209,11 +209,7 @@ func (f *Filer) RollbackTransaction(ctx context.Context) error {
|
||||
return f.Store.RollbackTransaction(ctx)
|
||||
}
|
||||
|
||||
// CreateEntry creates or replaces an entry. When existing is non-nil the caller
|
||||
// has already fetched the current entry at this path under a path lock, and it
|
||||
// is reused instead of looking the store up again; pass nil to have CreateEntry
|
||||
// look it up itself.
|
||||
func (f *Filer) CreateEntry(ctx context.Context, entry *Entry, existing *Entry, o_excl bool, isFromOtherCluster bool, signatures []int32, skipCreateParentDir bool, maxFilenameLength uint32) error {
|
||||
func (f *Filer) CreateEntry(ctx context.Context, entry *Entry, o_excl bool, isFromOtherCluster bool, signatures []int32, skipCreateParentDir bool, maxFilenameLength uint32) error {
|
||||
|
||||
if string(entry.FullPath) == "/" {
|
||||
return nil
|
||||
@@ -231,10 +227,7 @@ func (f *Filer) CreateEntry(ctx context.Context, entry *Entry, existing *Entry,
|
||||
entry.Attr.Atime = entryInitialAtime(entry.Attr)
|
||||
}
|
||||
|
||||
oldEntry := existing
|
||||
if oldEntry == nil {
|
||||
oldEntry, _ = f.FindEntry(ctx, entry.FullPath)
|
||||
}
|
||||
oldEntry, _ := f.FindEntry(ctx, entry.FullPath)
|
||||
|
||||
/*
|
||||
if !hasWritePermission(lastDirectoryEntry, entry) {
|
||||
|
||||
@@ -28,7 +28,7 @@ func TestCreateEntryAssignsInodeWhenMissing(t *testing.T) {
|
||||
},
|
||||
}
|
||||
|
||||
err := f.CreateEntry(context.Background(), entry, nil, false, false, nil, false, f.MaxFilenameLength)
|
||||
err := f.CreateEntry(context.Background(), entry, false, false, nil, false, f.MaxFilenameLength)
|
||||
require.NoError(t, err)
|
||||
|
||||
stored, findErr := store.FindEntry(context.Background(), entry.FullPath)
|
||||
@@ -48,7 +48,7 @@ func TestCreateEntryAssignsInodesToAutoCreatedParents(t *testing.T) {
|
||||
},
|
||||
}
|
||||
|
||||
err := f.CreateEntry(context.Background(), entry, nil, false, false, nil, false, f.MaxFilenameLength)
|
||||
err := f.CreateEntry(context.Background(), entry, false, false, nil, false, f.MaxFilenameLength)
|
||||
require.NoError(t, err)
|
||||
|
||||
for _, path := range []string{"/a", "/a/b", "/a/b/c.txt"} {
|
||||
|
||||
@@ -96,7 +96,7 @@ func (f *Filer) maybeLazyFetchFromRemote(ctx context.Context, p util.FullPath) (
|
||||
persistBaseCtx, cancelPersist := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
defer cancelPersist()
|
||||
persistCtx := context.WithValue(persistBaseCtx, lazyFetchContextKey{}, true)
|
||||
saveErr := f.CreateEntry(persistCtx, entry, nil, false, false, nil, true, f.MaxFilenameLength)
|
||||
saveErr := f.CreateEntry(persistCtx, entry, false, false, nil, true, f.MaxFilenameLength)
|
||||
if saveErr != nil {
|
||||
glog.Warningf("maybeLazyFetchFromRemote: failed to persist filer entry for %s: %v", p, saveErr)
|
||||
f.lazyFetchGroup.Forget(key)
|
||||
|
||||
@@ -152,7 +152,7 @@ func (f *Filer) maybeLazyListFromRemote(ctx context.Context, p util.FullPath) {
|
||||
entry.Attr.FileSize = uint64(remoteEntry.RemoteSize)
|
||||
}
|
||||
}
|
||||
if saveErr := f.CreateEntry(persistCtx, entry, nil, false, false, nil, true, f.MaxFilenameLength); saveErr != nil {
|
||||
if saveErr := f.CreateEntry(persistCtx, entry, false, false, nil, true, f.MaxFilenameLength); saveErr != nil {
|
||||
glog.Warningf("maybeLazyListFromRemote: persist %s: %v", childPath, saveErr)
|
||||
}
|
||||
}
|
||||
@@ -193,7 +193,7 @@ func (f *Filer) updateDirectoryListingSyncedAt(ctx context.Context, p util.FullP
|
||||
dirEntry.Extended = make(map[string][]byte)
|
||||
}
|
||||
dirEntry.Extended[xattrRemoteListingSyncedAt] = []byte(fmt.Sprintf("%d", syncTime.Unix()))
|
||||
if saveErr := f.CreateEntry(ctx, dirEntry, nil, false, false, nil, true, f.MaxFilenameLength); saveErr != nil {
|
||||
if saveErr := f.CreateEntry(ctx, dirEntry, false, false, nil, true, f.MaxFilenameLength); saveErr != nil {
|
||||
glog.Warningf("maybeLazyListFromRemote: create dir synced_at for %s: %v", p, saveErr)
|
||||
}
|
||||
return
|
||||
|
||||
@@ -43,7 +43,7 @@ func (f *Filer) appendToFile(targetFile string, data []byte) error {
|
||||
entry.Chunks = append(entry.GetChunks(), uploadResult.ToPbFileChunk(assignResult.Fid, offset, time.Now().UnixNano()))
|
||||
|
||||
// update the entry
|
||||
err = f.CreateEntry(context.Background(), entry, nil, false, false, nil, false, f.MaxFilenameLength)
|
||||
err = f.CreateEntry(context.Background(), entry, false, false, nil, false, f.MaxFilenameLength)
|
||||
|
||||
return err
|
||||
}
|
||||
|
||||
@@ -32,7 +32,7 @@ func TestCreateAndFind(t *testing.T) {
|
||||
},
|
||||
}
|
||||
|
||||
if err := testFiler.CreateEntry(ctx, entry1, nil, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
|
||||
if err := testFiler.CreateEntry(ctx, entry1, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
|
||||
t.Errorf("create entry %v: %v", entry1.FullPath, err)
|
||||
return
|
||||
}
|
||||
|
||||
@@ -29,7 +29,7 @@ func TestCreateAndFind(t *testing.T) {
|
||||
},
|
||||
}
|
||||
|
||||
if err := testFiler.CreateEntry(ctx, entry1, nil, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
|
||||
if err := testFiler.CreateEntry(ctx, entry1, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
|
||||
t.Errorf("create entry %v: %v", entry1.FullPath, err)
|
||||
return
|
||||
}
|
||||
|
||||
@@ -29,7 +29,7 @@ func TestCreateAndFind(t *testing.T) {
|
||||
},
|
||||
}
|
||||
|
||||
if err := testFiler.CreateEntry(ctx, entry1, nil, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
|
||||
if err := testFiler.CreateEntry(ctx, entry1, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
|
||||
t.Errorf("create entry %v: %v", entry1.FullPath, err)
|
||||
return
|
||||
}
|
||||
|
||||
@@ -1,259 +0,0 @@
|
||||
package posixlock
|
||||
|
||||
import (
|
||||
"sync"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Manager is the owner filer's in-memory authority for POSIX advisory locks
|
||||
// across inodes. Lock state lives here, not in replicated metadata: it is
|
||||
// transient coordination, so keeping it out of the meta-log avoids churn and
|
||||
// does not pollute what subscribers see (the distributed lock manager holds its
|
||||
// locks the same way). A `Set` per inode key, plus a session index so a dead
|
||||
// mount's locks are reaped in O(locks held) rather than by scanning every inode.
|
||||
//
|
||||
// key is an opaque inode identity supplied by the caller — the file's path, or
|
||||
// "hl:"+hex(HardLinkId) for a hardlinked inode — so all names of one inode share
|
||||
// a Set. The Manager is safe for concurrent use.
|
||||
type Manager struct {
|
||||
mu sync.Mutex
|
||||
byKey map[string]*Set // inode key -> held locks
|
||||
bySid map[uint64]map[string]bool // session -> keys it currently holds locks on
|
||||
lastSeen map[uint64]time.Time // session -> last keepalive; only renewing sessions are leased
|
||||
}
|
||||
|
||||
func NewManager() *Manager {
|
||||
return &Manager{
|
||||
byKey: make(map[string]*Set),
|
||||
bySid: make(map[uint64]map[string]bool),
|
||||
lastSeen: make(map[uint64]time.Time),
|
||||
}
|
||||
}
|
||||
|
||||
// Renew records a keepalive from a session, placing it under lease management.
|
||||
// Only sessions that have renewed are subject to ReapExpired, so a session that
|
||||
// never sends keepalives (e.g. before the mount keepalive exists) is never reaped.
|
||||
func (m *Manager) Renew(sid uint64) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
m.lastSeen[sid] = time.Now()
|
||||
}
|
||||
|
||||
// ReapExpired releases the locks of every leased session whose last keepalive is
|
||||
// older than ttl — a dead or partitioned mount. Sessions that never renewed are
|
||||
// left untouched. Returns the reaped session ids.
|
||||
func (m *Manager) ReapExpired(ttl time.Duration) []uint64 {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
cutoff := time.Now().Add(-ttl)
|
||||
var reaped []uint64
|
||||
for sid, seen := range m.lastSeen {
|
||||
if seen.After(cutoff) {
|
||||
continue
|
||||
}
|
||||
for key := range m.bySid[sid] {
|
||||
s := m.byKey[key]
|
||||
if s == nil {
|
||||
continue
|
||||
}
|
||||
s.ReleaseSession(sid)
|
||||
if s.Empty() {
|
||||
delete(m.byKey, key)
|
||||
}
|
||||
}
|
||||
delete(m.bySid, sid)
|
||||
delete(m.lastSeen, sid)
|
||||
reaped = append(reaped, sid)
|
||||
}
|
||||
return reaped
|
||||
}
|
||||
|
||||
// TryLock grants lk on key, or returns the conflicting lock and false. The set
|
||||
// is created on first use and dropped again when it empties.
|
||||
func (m *Manager) TryLock(key string, lk Range) (Range, bool) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
s, ok := m.byKey[key]
|
||||
if !ok {
|
||||
s = &Set{}
|
||||
}
|
||||
if c, granted := s.Acquire(lk); !granted {
|
||||
return c, false
|
||||
}
|
||||
if !ok {
|
||||
m.byKey[key] = s
|
||||
}
|
||||
m.index(lk.Sid, key)
|
||||
return Range{}, true
|
||||
}
|
||||
|
||||
// Track records a lock the server already granted, without arbitration. A mount
|
||||
// uses it to mirror its own held locks so it can re-assert them to the inode's
|
||||
// current owner filer after an ownership change or owner restart.
|
||||
func (m *Manager) Track(key string, lk Range) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
s, ok := m.byKey[key]
|
||||
if !ok {
|
||||
s = &Set{}
|
||||
m.byKey[key] = s
|
||||
}
|
||||
s.Grant(lk)
|
||||
m.index(lk.Sid, key)
|
||||
}
|
||||
|
||||
// Snapshot returns a copy of the held locks per key. A mount calls it to drive
|
||||
// re-assertion keepalives; the filer never does.
|
||||
func (m *Manager) Snapshot() map[string][]Range {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
out := make(map[string][]Range, len(m.byKey))
|
||||
for key, s := range m.byKey {
|
||||
out[key] = append([]Range(nil), s.locks...)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// Reassert rebuilds session sid's locks on key from the client's authoritative
|
||||
// list, renewing the lease. It replaces sid's existing locks on the key, then
|
||||
// re-acquires each asserted lock — arbitrating against other sessions so it never
|
||||
// double-grants. Locks that lost to another session in a migration window are
|
||||
// returned as conflicts. The owner filer calls this on a re-assertion keepalive.
|
||||
func (m *Manager) Reassert(key string, sid uint64, locks []Range) (conflicts []Range) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
m.lastSeen[sid] = time.Now()
|
||||
|
||||
s := m.byKey[key]
|
||||
if s == nil {
|
||||
if len(locks) == 0 {
|
||||
return nil
|
||||
}
|
||||
s = &Set{}
|
||||
m.byKey[key] = s
|
||||
}
|
||||
s.ReleaseSession(sid)
|
||||
for _, lk := range locks {
|
||||
lk.Sid = sid
|
||||
if c, granted := s.Acquire(lk); !granted {
|
||||
conflicts = append(conflicts, c)
|
||||
}
|
||||
}
|
||||
if setHasSession(s, sid) {
|
||||
m.index(sid, key)
|
||||
} else {
|
||||
m.deindex(sid, key)
|
||||
}
|
||||
if s.Empty() {
|
||||
delete(m.byKey, key)
|
||||
}
|
||||
return conflicts
|
||||
}
|
||||
|
||||
// Unlock releases lk's owner's locks within its namespace over lk's range.
|
||||
func (m *Manager) Unlock(key string, lk Range) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
s := m.byKey[key]
|
||||
if s == nil {
|
||||
return
|
||||
}
|
||||
s.Release(lk)
|
||||
m.afterRelease(key, s, lk.Sid)
|
||||
}
|
||||
|
||||
// GetLk reports the lock that would block proposed on key, if any.
|
||||
func (m *Manager) GetLk(key string, proposed Range) (Range, bool) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
s := m.byKey[key]
|
||||
if s == nil {
|
||||
return Range{}, false
|
||||
}
|
||||
return s.Conflict(proposed)
|
||||
}
|
||||
|
||||
// ReleasePosixOwner drops (sid, owner)'s fcntl locks on key — the flush-time path.
|
||||
func (m *Manager) ReleasePosixOwner(key string, sid, owner uint64) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
s := m.byKey[key]
|
||||
if s == nil {
|
||||
return
|
||||
}
|
||||
s.ReleasePosixOwner(sid, owner)
|
||||
m.afterRelease(key, s, sid)
|
||||
}
|
||||
|
||||
// ReleaseFlockOwner drops (sid, owner)'s flock locks on key — the release-time path.
|
||||
func (m *Manager) ReleaseFlockOwner(key string, sid, owner uint64) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
s := m.byKey[key]
|
||||
if s == nil {
|
||||
return
|
||||
}
|
||||
s.ReleaseFlockOwner(sid, owner)
|
||||
m.afterRelease(key, s, sid)
|
||||
}
|
||||
|
||||
// ReleaseSession drops every lock held by a session across all inodes it touched,
|
||||
// reaping a mount whose lease expired. O(locks held by the session).
|
||||
func (m *Manager) ReleaseSession(sid uint64) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
for key := range m.bySid[sid] {
|
||||
s := m.byKey[key]
|
||||
if s == nil {
|
||||
continue
|
||||
}
|
||||
s.ReleaseSession(sid)
|
||||
if s.Empty() {
|
||||
delete(m.byKey, key)
|
||||
}
|
||||
}
|
||||
delete(m.bySid, sid)
|
||||
}
|
||||
|
||||
// afterRelease prunes the session index when sid no longer holds any lock on key,
|
||||
// and drops the set when it empties. Only sid's presence can have changed, since
|
||||
// a release only removes sid's locks.
|
||||
func (m *Manager) afterRelease(key string, s *Set, sid uint64) {
|
||||
if s.Empty() {
|
||||
delete(m.byKey, key)
|
||||
m.deindex(sid, key)
|
||||
return
|
||||
}
|
||||
if !setHasSession(s, sid) {
|
||||
m.deindex(sid, key)
|
||||
}
|
||||
}
|
||||
|
||||
func (m *Manager) index(sid uint64, key string) {
|
||||
keys := m.bySid[sid]
|
||||
if keys == nil {
|
||||
keys = make(map[string]bool)
|
||||
m.bySid[sid] = keys
|
||||
}
|
||||
keys[key] = true
|
||||
}
|
||||
|
||||
func (m *Manager) deindex(sid uint64, key string) {
|
||||
keys := m.bySid[sid]
|
||||
if keys == nil {
|
||||
return
|
||||
}
|
||||
delete(keys, key)
|
||||
if len(keys) == 0 {
|
||||
delete(m.bySid, sid)
|
||||
}
|
||||
}
|
||||
|
||||
func setHasSession(s *Set, sid uint64) bool {
|
||||
for _, l := range s.Locks() {
|
||||
if l.Sid == sid {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
@@ -1,194 +0,0 @@
|
||||
package posixlock
|
||||
|
||||
import (
|
||||
"math"
|
||||
"runtime"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
func TestManagerGrantAndConflict(t *testing.T) {
|
||||
m := NewManager()
|
||||
if _, granted := m.TryLock("a", Range{Start: 0, End: 99, Type: Write, Sid: 1, Owner: 1}); !granted {
|
||||
t.Fatal("first lock should be granted")
|
||||
}
|
||||
if c, granted := m.TryLock("a", Range{Start: 50, End: 149, Type: Write, Sid: 2, Owner: 1}); granted {
|
||||
t.Fatalf("overlapping lock from another session should conflict, got grant; conflict=%+v", c)
|
||||
}
|
||||
// A different key is independent.
|
||||
if _, granted := m.TryLock("b", Range{Start: 0, End: 99, Type: Write, Sid: 2, Owner: 1}); !granted {
|
||||
t.Fatal("lock on a different key should be granted")
|
||||
}
|
||||
}
|
||||
|
||||
func TestManagerUnlockCleansEmptyKeyAndIndex(t *testing.T) {
|
||||
m := NewManager()
|
||||
lk := Range{Start: 0, End: 99, Type: Write, Sid: 1, Owner: 1}
|
||||
m.TryLock("a", lk)
|
||||
|
||||
if !m.bySid[1]["a"] {
|
||||
t.Fatal("session index should record the held key")
|
||||
}
|
||||
m.Unlock("a", Range{Start: 0, End: 99, Type: Unlock, Sid: 1, Owner: 1})
|
||||
|
||||
if _, ok := m.byKey["a"]; ok {
|
||||
t.Fatal("empty set should be dropped from byKey")
|
||||
}
|
||||
if _, ok := m.bySid[1]; ok {
|
||||
t.Fatal("session index should be pruned when it holds nothing")
|
||||
}
|
||||
}
|
||||
|
||||
func TestManagerPartialUnlockKeepsIndex(t *testing.T) {
|
||||
m := NewManager()
|
||||
m.TryLock("a", Range{Start: 0, End: 49, Type: Write, Sid: 1, Owner: 1})
|
||||
m.TryLock("a", Range{Start: 100, End: 149, Type: Write, Sid: 1, Owner: 1})
|
||||
// Release one of the two ranges; the session still holds the other.
|
||||
m.Unlock("a", Range{Start: 0, End: 49, Type: Unlock, Sid: 1, Owner: 1})
|
||||
|
||||
if !m.bySid[1]["a"] {
|
||||
t.Fatal("session still holds a lock on the key; index must remain")
|
||||
}
|
||||
if _, ok := m.byKey["a"]; !ok {
|
||||
t.Fatal("key should remain while a lock is held")
|
||||
}
|
||||
}
|
||||
|
||||
func TestManagerGetLk(t *testing.T) {
|
||||
m := NewManager()
|
||||
m.TryLock("a", Range{Start: 10, End: 50, Type: Write, Sid: 1, Owner: 1, Pid: 7})
|
||||
c, found := m.GetLk("a", Range{Start: 30, End: 70, Type: Read, Sid: 2, Owner: 1})
|
||||
if !found || c.Pid != 7 {
|
||||
t.Fatalf("expected conflict from pid 7, got %+v found=%v", c, found)
|
||||
}
|
||||
if _, found := m.GetLk("missing", Range{Start: 0, End: 1, Type: Write, Sid: 9, Owner: 9}); found {
|
||||
t.Fatal("missing key should report no conflict")
|
||||
}
|
||||
}
|
||||
|
||||
func TestManagerReleasePosixOwnerKeepsFlockAndIndex(t *testing.T) {
|
||||
m := NewManager()
|
||||
m.TryLock("a", Range{Start: 0, End: 99, Type: Write, Sid: 1, Owner: 1})
|
||||
m.TryLock("a", Range{Start: 0, End: math.MaxUint64, Type: Write, Sid: 1, Owner: 1, IsFlock: true})
|
||||
|
||||
m.ReleasePosixOwner("a", 1, 1)
|
||||
|
||||
// flock lock for the same session remains, so the index must remain too.
|
||||
if !m.bySid[1]["a"] {
|
||||
t.Fatal("session still holds the flock lock; index must remain")
|
||||
}
|
||||
if _, found := m.GetLk("a", Range{Start: 0, End: 10, Type: Write, Sid: 2, Owner: 2, IsFlock: true}); !found {
|
||||
t.Fatal("flock lock should survive ReleasePosixOwner")
|
||||
}
|
||||
if _, found := m.GetLk("a", Range{Start: 0, End: 10, Type: Write, Sid: 2, Owner: 2}); found {
|
||||
t.Fatal("fcntl lock should be gone after ReleasePosixOwner")
|
||||
}
|
||||
}
|
||||
|
||||
func TestManagerReleaseSessionReapsAcrossKeys(t *testing.T) {
|
||||
m := NewManager()
|
||||
m.TryLock("a", Range{Start: 0, End: 99, Type: Write, Sid: 1, Owner: 1})
|
||||
m.TryLock("b", Range{Start: 0, End: 99, Type: Write, Sid: 1, Owner: 2})
|
||||
m.TryLock("b", Range{Start: 200, End: 299, Type: Write, Sid: 2, Owner: 1})
|
||||
|
||||
m.ReleaseSession(1)
|
||||
|
||||
if _, ok := m.bySid[1]; ok {
|
||||
t.Fatal("reaped session should be gone from the index")
|
||||
}
|
||||
if _, ok := m.byKey["a"]; ok {
|
||||
t.Fatal("key a held only session 1's lock and should be dropped")
|
||||
}
|
||||
// Session 2's lock on b survives.
|
||||
if _, found := m.GetLk("b", Range{Start: 200, End: 299, Type: Write, Sid: 9, Owner: 9}); !found {
|
||||
t.Fatal("session 2's lock on b should remain after reaping session 1")
|
||||
}
|
||||
if !m.bySid[2]["b"] {
|
||||
t.Fatal("session 2 index entry should remain")
|
||||
}
|
||||
}
|
||||
|
||||
func TestManagerReapsOnlyStaleLeasedSessions(t *testing.T) {
|
||||
m := NewManager()
|
||||
// Session 1: holds a lock, leased but stale (renewed long ago).
|
||||
m.TryLock("a", Range{Start: 0, End: 99, Type: Write, Sid: 1, Owner: 1})
|
||||
m.Renew(1)
|
||||
m.lastSeen[1] = time.Now().Add(-time.Hour)
|
||||
// Session 2: holds a lock, leased and fresh.
|
||||
m.TryLock("b", Range{Start: 0, End: 99, Type: Write, Sid: 2, Owner: 1})
|
||||
m.Renew(2)
|
||||
// Session 3: holds a lock but never renewed (no lease) — must not be reaped.
|
||||
m.TryLock("c", Range{Start: 0, End: 99, Type: Write, Sid: 3, Owner: 1})
|
||||
|
||||
reaped := m.ReapExpired(30 * time.Second)
|
||||
|
||||
if len(reaped) != 1 || reaped[0] != 1 {
|
||||
t.Fatalf("only the stale leased session should be reaped, got %v", reaped)
|
||||
}
|
||||
if _, ok := m.byKey["a"]; ok {
|
||||
t.Fatal("stale session's lock should be gone")
|
||||
}
|
||||
if _, ok := m.byKey["b"]; !ok {
|
||||
t.Fatal("fresh session's lock must remain")
|
||||
}
|
||||
if _, ok := m.byKey["c"]; !ok {
|
||||
t.Fatal("never-renewed session must not be reaped")
|
||||
}
|
||||
if _, ok := m.lastSeen[1]; ok {
|
||||
t.Fatal("reaped session's lease entry should be cleared")
|
||||
}
|
||||
}
|
||||
|
||||
// Mutual exclusion under concurrent whole-file flock churn through the Manager:
|
||||
// at most one owner may believe it holds the exclusive lock at any instant.
|
||||
func TestManagerConcurrentFlockMutualExclusion(t *testing.T) {
|
||||
m := NewManager()
|
||||
const (
|
||||
key = "inode"
|
||||
workers = 16
|
||||
iters = 400
|
||||
)
|
||||
var (
|
||||
wg sync.WaitGroup
|
||||
holder atomic.Int64
|
||||
overlap atomic.Int32
|
||||
)
|
||||
for w := 0; w < workers; w++ {
|
||||
wg.Add(1)
|
||||
go func(id int) {
|
||||
defer wg.Done()
|
||||
lk := Range{Start: 0, End: math.MaxUint64, Type: Write, Sid: uint64(id + 1), Owner: 1, IsFlock: true}
|
||||
unlock := lk
|
||||
unlock.Type = Unlock
|
||||
token := int64(id + 1)
|
||||
for i := 0; i < iters; i++ {
|
||||
for {
|
||||
if _, granted := m.TryLock(key, lk); granted {
|
||||
break
|
||||
}
|
||||
runtime.Gosched()
|
||||
}
|
||||
if prev := holder.Swap(token); prev != 0 {
|
||||
overlap.Add(1)
|
||||
}
|
||||
runtime.Gosched()
|
||||
if !holder.CompareAndSwap(token, 0) {
|
||||
overlap.Add(1)
|
||||
}
|
||||
m.Unlock(key, unlock)
|
||||
}
|
||||
}(w)
|
||||
}
|
||||
wg.Wait()
|
||||
if n := overlap.Load(); n != 0 {
|
||||
t.Fatalf("mutual exclusion violated %d times", n)
|
||||
}
|
||||
if len(m.byKey) != 0 {
|
||||
t.Fatalf("all locks released; byKey should be empty, got %d", len(m.byKey))
|
||||
}
|
||||
if len(m.bySid) != 0 {
|
||||
t.Fatalf("all locks released; bySid should be empty, got %d", len(m.bySid))
|
||||
}
|
||||
}
|
||||
@@ -1,222 +0,0 @@
|
||||
// Package posixlock implements the conflict, coalescing, and range-split logic
|
||||
// for POSIX advisory file locks — fcntl byte-range and flock whole-file — as a
|
||||
// pure per-inode lock set with no concurrency control of its own.
|
||||
//
|
||||
// It is the server-side authority for distributed FUSE locking: the owner filer
|
||||
// for an inode holds one Set and serializes access to it under that inode's
|
||||
// per-path lock, so each operation runs to completion without concurrent
|
||||
// mutation. The same algorithm can back the per-mount table; blocking (SetLkw)
|
||||
// and any wait queue belong to the caller, not here.
|
||||
package posixlock
|
||||
|
||||
import (
|
||||
"math"
|
||||
"sort"
|
||||
)
|
||||
|
||||
// Lock types, kept independent of the platform syscall package (whose F_RDLCK /
|
||||
// F_WRLCK / F_UNLCK values differ per OS and are absent on some) so the filer
|
||||
// builds everywhere. Callers map the syscall constants onto these at the edge.
|
||||
// Zero is intentionally unused so a zero-value Range reads as "unset".
|
||||
const (
|
||||
Read uint32 = 1
|
||||
Write uint32 = 2
|
||||
Unlock uint32 = 3
|
||||
)
|
||||
|
||||
// Range is one held advisory byte-range lock. Owner identity is the (Sid, Owner)
|
||||
// pair: Sid is the mount session, Owner the FUSE lock owner within it, so owners
|
||||
// from different mounts never alias. End is inclusive; math.MaxUint64 means EOF.
|
||||
// IsFlock separates the flock and fcntl namespaces, which never conflict.
|
||||
type Range struct {
|
||||
Start uint64
|
||||
End uint64
|
||||
Type uint32
|
||||
Sid uint64
|
||||
Owner uint64
|
||||
Pid uint32
|
||||
IsFlock bool
|
||||
}
|
||||
|
||||
func (r Range) sameOwner(o Range) bool {
|
||||
return r.Sid == o.Sid && r.Owner == o.Owner
|
||||
}
|
||||
|
||||
// Set is the authoritative set of advisory locks held on one inode. The zero
|
||||
// value is an empty set. Set has no internal locking; the caller serializes it.
|
||||
type Set struct {
|
||||
locks []Range // sorted by Start
|
||||
}
|
||||
|
||||
func overlap(aStart, aEnd, bStart, bEnd uint64) bool {
|
||||
return aStart <= bEnd && bStart <= aEnd
|
||||
}
|
||||
|
||||
// Conflict returns the first held lock that blocks proposed, if any. Two locks
|
||||
// conflict when they share a namespace, have different owners, overlap, and at
|
||||
// least one is a write lock.
|
||||
func (s *Set) Conflict(proposed Range) (Range, bool) {
|
||||
for _, h := range s.locks {
|
||||
if h.IsFlock != proposed.IsFlock || h.sameOwner(proposed) {
|
||||
continue
|
||||
}
|
||||
if !overlap(h.Start, h.End, proposed.Start, proposed.End) {
|
||||
continue
|
||||
}
|
||||
if h.Type == Read && proposed.Type == Read {
|
||||
continue
|
||||
}
|
||||
return h, true
|
||||
}
|
||||
return Range{}, false
|
||||
}
|
||||
|
||||
// Acquire grants lk when it does not conflict, inserting it and coalescing the
|
||||
// owner's adjacent/overlapping ranges. On conflict it returns the blocking lock
|
||||
// and false, leaving the set unchanged.
|
||||
func (s *Set) Acquire(lk Range) (Range, bool) {
|
||||
if c, found := s.Conflict(lk); found {
|
||||
return c, false
|
||||
}
|
||||
s.insert(lk)
|
||||
return Range{}, true
|
||||
}
|
||||
|
||||
// Grant inserts lk without a conflict check. It is for a client mirroring locks
|
||||
// the server already granted (so they are conflict-free), not for arbitration.
|
||||
func (s *Set) Grant(lk Range) {
|
||||
s.insert(lk)
|
||||
}
|
||||
|
||||
// insert adds lk, absorbing same-owner same-type overlaps and merging adjacent
|
||||
// same-type ranges, and truncating/splitting a same-owner range of a different
|
||||
// type that overlaps (an in-place type change).
|
||||
func (s *Set) insert(lk Range) {
|
||||
var kept []Range
|
||||
for _, h := range s.locks {
|
||||
if !h.sameOwner(lk) || h.IsFlock != lk.IsFlock {
|
||||
kept = append(kept, h)
|
||||
continue
|
||||
}
|
||||
if !overlap(h.Start, h.End, lk.Start, lk.End) {
|
||||
// Merge only ranges that are adjacent and the same type. The
|
||||
// End < MaxUint64 guards stop +1 from wrapping at EOF.
|
||||
if h.Type == lk.Type && ((h.End < math.MaxUint64 && h.End+1 == lk.Start) || (lk.End < math.MaxUint64 && lk.End+1 == h.Start)) {
|
||||
if h.Start < lk.Start {
|
||||
lk.Start = h.Start
|
||||
}
|
||||
if h.End > lk.End {
|
||||
lk.End = h.End
|
||||
}
|
||||
continue
|
||||
}
|
||||
kept = append(kept, h)
|
||||
continue
|
||||
}
|
||||
if h.Type == lk.Type {
|
||||
// Same type: absorb into lk by widening its range.
|
||||
if h.Start < lk.Start {
|
||||
lk.Start = h.Start
|
||||
}
|
||||
if h.End > lk.End {
|
||||
lk.End = h.End
|
||||
}
|
||||
continue
|
||||
}
|
||||
// Different type: the surviving portions of h outside lk's range stay.
|
||||
if h.Start < lk.Start {
|
||||
left := h
|
||||
left.End = lk.Start - 1
|
||||
kept = append(kept, left)
|
||||
}
|
||||
if h.End > lk.End {
|
||||
right := h
|
||||
right.Start = lk.End + 1
|
||||
kept = append(kept, right)
|
||||
}
|
||||
}
|
||||
kept = append(kept, lk)
|
||||
sort.Slice(kept, func(i, j int) bool { return kept[i].Start < kept[j].Start })
|
||||
s.locks = kept
|
||||
}
|
||||
|
||||
// remove drops or splits matching locks within [start,end]. A match that
|
||||
// straddles the range keeps its non-overlapping head and/or tail.
|
||||
func (s *Set) remove(matches func(Range) bool, start, end uint64) {
|
||||
var kept []Range
|
||||
for _, h := range s.locks {
|
||||
if !matches(h) || !overlap(h.Start, h.End, start, end) {
|
||||
kept = append(kept, h)
|
||||
continue
|
||||
}
|
||||
if h.Start < start {
|
||||
left := h
|
||||
left.End = start - 1
|
||||
kept = append(kept, left)
|
||||
}
|
||||
if h.End > end {
|
||||
right := h
|
||||
right.Start = end + 1
|
||||
kept = append(kept, right)
|
||||
}
|
||||
// Fully covered: dropped.
|
||||
}
|
||||
s.locks = kept
|
||||
}
|
||||
|
||||
// Release clears lk's owner's locks within lk's namespace over [lk.Start,lk.End]
|
||||
// — the F_UNLCK path, which may split a straddling range.
|
||||
func (s *Set) Release(lk Range) {
|
||||
s.remove(func(h Range) bool {
|
||||
return h.sameOwner(lk) && h.IsFlock == lk.IsFlock
|
||||
}, lk.Start, lk.End)
|
||||
}
|
||||
|
||||
// ReleaseOwner removes every lock held by (sid, owner) in both namespaces.
|
||||
func (s *Set) ReleaseOwner(sid, owner uint64) {
|
||||
s.remove(func(h Range) bool {
|
||||
return h.Sid == sid && h.Owner == owner
|
||||
}, 0, math.MaxUint64)
|
||||
}
|
||||
|
||||
// ReleaseFlockOwner removes only the flock locks of (sid, owner) — the close-time
|
||||
// path for a released file description (FUSE_RELEASE_FLOCK_UNLOCK).
|
||||
func (s *Set) ReleaseFlockOwner(sid, owner uint64) {
|
||||
s.remove(func(h Range) bool {
|
||||
return h.IsFlock && h.Sid == sid && h.Owner == owner
|
||||
}, 0, math.MaxUint64)
|
||||
}
|
||||
|
||||
// ReleasePosixOwner removes only the fcntl locks of (sid, owner) — the close-time
|
||||
// path for a flushing POSIX lock owner.
|
||||
func (s *Set) ReleasePosixOwner(sid, owner uint64) {
|
||||
s.remove(func(h Range) bool {
|
||||
return !h.IsFlock && h.Sid == sid && h.Owner == owner
|
||||
}, 0, math.MaxUint64)
|
||||
}
|
||||
|
||||
// ReleaseSession removes every lock held by a session, reaping a mount that has
|
||||
// died or disconnected (its lease expired).
|
||||
func (s *Set) ReleaseSession(sid uint64) {
|
||||
s.remove(func(h Range) bool { return h.Sid == sid }, 0, math.MaxUint64)
|
||||
}
|
||||
|
||||
// HasPosix reports whether (sid, owner) holds any fcntl lock, mirroring the
|
||||
// mount's flush-time check that avoids treating a lock-free flush as
|
||||
// lock-sensitive.
|
||||
func (s *Set) HasPosix(sid, owner uint64) bool {
|
||||
for _, h := range s.locks {
|
||||
if !h.IsFlock && h.Sid == sid && h.Owner == owner {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// Locks returns the held locks, sorted by Start. The slice aliases internal
|
||||
// state; the caller must not mutate it.
|
||||
func (s *Set) Locks() []Range { return s.locks }
|
||||
|
||||
// Empty reports whether no locks are held, so the caller can drop the inode's
|
||||
// entry from its table.
|
||||
func (s *Set) Empty() bool { return len(s.locks) == 0 }
|
||||
@@ -1,292 +0,0 @@
|
||||
package posixlock
|
||||
|
||||
import (
|
||||
"math"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// acquire is a test helper asserting the lock is granted.
|
||||
func mustAcquire(t *testing.T, s *Set, lk Range) {
|
||||
t.Helper()
|
||||
if _, granted := s.Acquire(lk); !granted {
|
||||
t.Fatalf("expected lock granted: %+v", lk)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNonOverlappingLocksFromDifferentOwners(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 49, Type: Write, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 50, End: 99, Type: Write, Owner: 2, Pid: 20})
|
||||
}
|
||||
|
||||
func TestOverlappingReadLocksFromDifferentOwners(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Read, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 50, End: 149, Type: Read, Owner: 2, Pid: 20})
|
||||
}
|
||||
|
||||
func TestOverlappingWriteReadConflict(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
|
||||
if _, granted := s.Acquire(Range{Start: 50, End: 149, Type: Read, Owner: 2, Pid: 20}); granted {
|
||||
t.Fatal("expected conflict")
|
||||
}
|
||||
}
|
||||
|
||||
func TestOverlappingWriteWriteConflict(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
|
||||
if _, granted := s.Acquire(Range{Start: 50, End: 149, Type: Write, Owner: 2, Pid: 20}); granted {
|
||||
t.Fatal("expected conflict")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSameOwnerUpgradeReadToWrite(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Read, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
|
||||
|
||||
c, found := s.Conflict(Range{Start: 0, End: 99, Type: Write, Owner: 2, Pid: 20})
|
||||
if !found || c.Type != Write {
|
||||
t.Fatalf("expected conflicting write lock after upgrade, got %+v found=%v", c, found)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSameOwnerDowngradeWriteToRead(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Read, Owner: 1, Pid: 10})
|
||||
// Another owner can now take a shared read lock.
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Read, Owner: 2, Pid: 20})
|
||||
}
|
||||
|
||||
func TestLockCoalescing(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 9, Type: Write, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 10, End: 19, Type: Write, Owner: 1, Pid: 10})
|
||||
|
||||
if len(s.locks) != 1 {
|
||||
t.Fatalf("expected 1 coalesced lock, got %d: %+v", len(s.locks), s.locks)
|
||||
}
|
||||
if s.locks[0].Start != 0 || s.locks[0].End != 19 {
|
||||
t.Errorf("expected coalesced [0,19], got [%d,%d]", s.locks[0].Start, s.locks[0].End)
|
||||
}
|
||||
}
|
||||
|
||||
func TestLockSplitting(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
|
||||
s.Release(Range{Start: 40, End: 59, Type: Unlock, Owner: 1, Pid: 10})
|
||||
|
||||
if len(s.locks) != 2 {
|
||||
t.Fatalf("expected 2 locks after split, got %d: %+v", len(s.locks), s.locks)
|
||||
}
|
||||
if s.locks[0].Start != 0 || s.locks[0].End != 39 {
|
||||
t.Errorf("expected left [0,39], got [%d,%d]", s.locks[0].Start, s.locks[0].End)
|
||||
}
|
||||
if s.locks[1].Start != 60 || s.locks[1].End != 99 {
|
||||
t.Errorf("expected right [60,99], got [%d,%d]", s.locks[1].Start, s.locks[1].End)
|
||||
}
|
||||
}
|
||||
|
||||
func TestConflictReportsHolder(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 10, End: 50, Type: Write, Owner: 1, Pid: 10})
|
||||
|
||||
c, found := s.Conflict(Range{Start: 30, End: 70, Type: Read, Owner: 2, Pid: 20})
|
||||
if !found {
|
||||
t.Fatal("expected a conflict")
|
||||
}
|
||||
if c.Type != Write || c.Pid != 10 || c.Start != 10 || c.End != 50 {
|
||||
t.Fatalf("unexpected conflict report: %+v", c)
|
||||
}
|
||||
}
|
||||
|
||||
func TestConflictNoneForSharedReads(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 10, End: 50, Type: Read, Owner: 1, Pid: 10})
|
||||
if _, found := s.Conflict(Range{Start: 30, End: 70, Type: Read, Owner: 2, Pid: 20}); found {
|
||||
t.Fatal("two read locks should not conflict")
|
||||
}
|
||||
}
|
||||
|
||||
func TestConflictSameOwnerNone(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
|
||||
if _, found := s.Conflict(Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10}); found {
|
||||
t.Fatal("an owner should not conflict with itself")
|
||||
}
|
||||
}
|
||||
|
||||
func TestReleaseOwner(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 49, Type: Write, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 50, End: 99, Type: Write, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 200, End: 299, Type: Read, Owner: 2, Pid: 20})
|
||||
|
||||
s.ReleaseOwner(0, 1)
|
||||
|
||||
if _, found := s.Conflict(Range{Start: 0, End: 99, Type: Write, Owner: 3, Pid: 30}); found {
|
||||
t.Fatal("owner 1's locks should be gone")
|
||||
}
|
||||
c, found := s.Conflict(Range{Start: 200, End: 299, Type: Write, Owner: 3, Pid: 30})
|
||||
if !found || c.Type != Read {
|
||||
t.Fatalf("owner 2's read lock should remain, got %+v found=%v", c, found)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFlockAndFcntlDoNotConflict(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 0, End: math.MaxUint64, Type: Write, Owner: 2, Pid: 20, IsFlock: true})
|
||||
}
|
||||
|
||||
func TestReleasePosixOwnerKeepsFlock(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: math.MaxUint64, Type: Write, Owner: 1, Pid: 10, IsFlock: true})
|
||||
s.ReleasePosixOwner(0, 1)
|
||||
if _, found := s.Conflict(Range{Start: 0, End: math.MaxUint64, Type: Write, Owner: 2, Pid: 20, IsFlock: true}); !found {
|
||||
t.Fatal("flock lock should remain after ReleasePosixOwner")
|
||||
}
|
||||
}
|
||||
|
||||
func TestReleaseFlockOwnerKeepsPosix(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 0, End: math.MaxUint64, Type: Write, Owner: 2, Pid: 10, IsFlock: true})
|
||||
|
||||
s.ReleaseFlockOwner(0, 2)
|
||||
|
||||
if _, found := s.Conflict(Range{Start: 0, End: 99, Type: Write, Owner: 3, Pid: 30}); !found {
|
||||
t.Fatal("fcntl lock should remain after ReleaseFlockOwner")
|
||||
}
|
||||
if _, found := s.Conflict(Range{Start: 0, End: math.MaxUint64, Type: Write, Owner: 4, Pid: 40, IsFlock: true}); found {
|
||||
t.Fatal("flock lock should be gone after ReleaseFlockOwner")
|
||||
}
|
||||
}
|
||||
|
||||
func TestHasPosixIgnoresMissingOwnerAndFlock(t *testing.T) {
|
||||
s := &Set{}
|
||||
if s.HasPosix(0, 1) {
|
||||
t.Fatal("empty set should report no posix owner")
|
||||
}
|
||||
mustAcquire(t, s, Range{Start: 0, End: math.MaxUint64, Type: Write, Owner: 1, Pid: 10, IsFlock: true})
|
||||
if s.HasPosix(0, 1) {
|
||||
t.Fatal("a flock owner is not a posix owner")
|
||||
}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 2, Pid: 20})
|
||||
if !s.HasPosix(0, 2) {
|
||||
t.Fatal("posix owner should be reported")
|
||||
}
|
||||
}
|
||||
|
||||
func TestWholeFileLock(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: math.MaxUint64, Type: Write, Owner: 1, Pid: 10})
|
||||
if _, granted := s.Acquire(Range{Start: 0, End: math.MaxUint64, Type: Write, Owner: 2, Pid: 20}); granted {
|
||||
t.Fatal("whole-file lock should block another owner")
|
||||
}
|
||||
if _, granted := s.Acquire(Range{Start: 100, End: 200, Type: Read, Owner: 2, Pid: 20}); granted {
|
||||
t.Fatal("partial overlap with whole-file lock should conflict")
|
||||
}
|
||||
}
|
||||
|
||||
func TestReleaseNoExistingLocks(t *testing.T) {
|
||||
s := &Set{}
|
||||
s.Release(Range{Start: 0, End: 99, Type: Unlock, Owner: 1, Pid: 10})
|
||||
if !s.Empty() {
|
||||
t.Fatal("releasing on an empty set should be a no-op")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSameOwnerReplaceDifferentType(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 30, End: 60, Type: Read, Owner: 1, Pid: 10})
|
||||
|
||||
if len(s.locks) != 3 {
|
||||
t.Fatalf("expected 3 locks after partial type change, got %d: %+v", len(s.locks), s.locks)
|
||||
}
|
||||
if s.locks[0].Type != Write || s.locks[0].Start != 0 || s.locks[0].End != 29 {
|
||||
t.Errorf("expected write [0,29], got %+v", s.locks[0])
|
||||
}
|
||||
if s.locks[1].Type != Read || s.locks[1].Start != 30 || s.locks[1].End != 60 {
|
||||
t.Errorf("expected read [30,60], got %+v", s.locks[1])
|
||||
}
|
||||
if s.locks[2].Type != Write || s.locks[2].Start != 61 || s.locks[2].End != 99 {
|
||||
t.Errorf("expected write [61,99], got %+v", s.locks[2])
|
||||
}
|
||||
}
|
||||
|
||||
func TestNonAdjacentRangesNotCoalesced(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 5, End: math.MaxUint64, Type: Write, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 0, End: 2, Type: Write, Owner: 1, Pid: 10})
|
||||
|
||||
if len(s.locks) != 2 {
|
||||
t.Fatalf("gap [3,4] should prevent coalescing, got %d: %+v", len(s.locks), s.locks)
|
||||
}
|
||||
if s.locks[0].Start != 0 || s.locks[0].End != 2 {
|
||||
t.Errorf("expected [0,2], got [%d,%d]", s.locks[0].Start, s.locks[0].End)
|
||||
}
|
||||
if s.locks[1].Start != 5 || s.locks[1].End != math.MaxUint64 {
|
||||
t.Errorf("expected [5,MaxUint64], got [%d,%d]", s.locks[1].Start, s.locks[1].End)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAdjacencyNoOverflowAtMaxUint64(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 100, End: math.MaxUint64, Type: Write, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 0, End: 0, Type: Write, Owner: 1, Pid: 10})
|
||||
|
||||
if len(s.locks) != 2 {
|
||||
t.Fatalf("MaxUint64+1 must not wrap and falsely merge, got %d: %+v", len(s.locks), s.locks)
|
||||
}
|
||||
}
|
||||
|
||||
// Sessions are part of owner identity: the same FUSE Owner number on two
|
||||
// different mounts (Sid) is two distinct owners and must contend.
|
||||
func TestTwoSessionsSameOwnerDoNotAlias(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Sid: 1, Owner: 5, Pid: 10})
|
||||
|
||||
if _, granted := s.Acquire(Range{Start: 50, End: 149, Type: Write, Sid: 2, Owner: 5, Pid: 20}); granted {
|
||||
t.Fatal("same Owner number on a different session must conflict, not alias")
|
||||
}
|
||||
// A read on session 2 against a session-1 read is fine (shared).
|
||||
s2 := &Set{}
|
||||
mustAcquire(t, s2, Range{Start: 0, End: 99, Type: Read, Sid: 1, Owner: 5})
|
||||
mustAcquire(t, s2, Range{Start: 0, End: 99, Type: Read, Sid: 2, Owner: 5})
|
||||
}
|
||||
|
||||
// Reaping a dead mount drops only that session's locks.
|
||||
func TestReleaseSessionReapsOnlyThatSession(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 49, Type: Write, Sid: 1, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 0, End: math.MaxUint64, Type: Write, Sid: 1, Owner: 2, Pid: 11, IsFlock: true})
|
||||
mustAcquire(t, s, Range{Start: 50, End: 99, Type: Write, Sid: 2, Owner: 1, Pid: 20})
|
||||
|
||||
s.ReleaseSession(1)
|
||||
|
||||
for _, h := range s.locks {
|
||||
if h.Sid == 1 {
|
||||
t.Fatalf("session 1 lock survived reaping: %+v", h)
|
||||
}
|
||||
}
|
||||
c, found := s.Conflict(Range{Start: 50, End: 99, Type: Write, Sid: 3, Owner: 9})
|
||||
if !found || c.Sid != 2 {
|
||||
t.Fatalf("session 2's lock should remain, got %+v found=%v", c, found)
|
||||
}
|
||||
}
|
||||
|
||||
func TestEmptyAfterReleasingAll(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
|
||||
if s.Empty() {
|
||||
t.Fatal("set should not be empty with a held lock")
|
||||
}
|
||||
s.Release(Range{Start: 0, End: 99, Type: Unlock, Owner: 1, Pid: 10})
|
||||
if !s.Empty() {
|
||||
t.Fatal("set should be empty after releasing the only lock")
|
||||
}
|
||||
}
|
||||
@@ -1,82 +0,0 @@
|
||||
package posixlock
|
||||
|
||||
import (
|
||||
"reflect"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// A mount's tracked locks round-trip through Snapshot back to a fresh owner via
|
||||
// Reassert — the owner-restart / ring-change recovery path.
|
||||
func TestReassertRebuildsOnFreshOwner(t *testing.T) {
|
||||
const sid = uint64(7)
|
||||
|
||||
// Client mirror: two granted locks on one key, one on another.
|
||||
client := NewManager()
|
||||
client.Track("a", Range{Start: 0, End: 99, Type: Write, Sid: sid, Owner: 1})
|
||||
client.Track("a", Range{Start: 200, End: 299, Type: Read, Sid: sid, Owner: 2})
|
||||
client.Track("b", Range{Start: 0, End: maxEnd, Type: Write, Sid: sid, Owner: 1, IsFlock: true})
|
||||
|
||||
// Fresh owner (post-restart / new ring owner) knows nothing.
|
||||
owner := NewManager()
|
||||
for key, locks := range client.Snapshot() {
|
||||
if c := owner.Reassert(key, sid, locks); c != nil {
|
||||
t.Fatalf("unexpected conflict reasserting %s: %+v", key, c)
|
||||
}
|
||||
}
|
||||
|
||||
// The owner now reports the same conflicts a foreign session would hit.
|
||||
if _, granted := owner.TryLock("a", Range{Start: 50, End: 60, Type: Write, Sid: 99, Owner: 1}); granted {
|
||||
t.Fatal("owner should block a foreign write after rebuild")
|
||||
}
|
||||
if _, granted := owner.TryLock("b", Range{Start: 0, End: 0, Type: Read, Sid: 99, Owner: 1, IsFlock: true}); granted {
|
||||
t.Fatal("owner should block a foreign flock read after rebuild")
|
||||
}
|
||||
}
|
||||
|
||||
// Re-asserting every tick is idempotent: the owner's view is unchanged.
|
||||
func TestReassertIdempotent(t *testing.T) {
|
||||
const sid = uint64(1)
|
||||
m := NewManager()
|
||||
m.TryLock("k", Range{Start: 0, End: 99, Type: Write, Sid: sid, Owner: 1})
|
||||
before := append([]Range(nil), m.byKey["k"].locks...)
|
||||
|
||||
m.Reassert("k", sid, before)
|
||||
m.Reassert("k", sid, before)
|
||||
|
||||
if !reflect.DeepEqual(m.byKey["k"].locks, before) {
|
||||
t.Fatalf("reassert not idempotent:\n got %+v\nwant %+v", m.byKey["k"].locks, before)
|
||||
}
|
||||
}
|
||||
|
||||
// A lock another session grabbed in the migration window is reported as a
|
||||
// conflict and not double-granted.
|
||||
func TestReassertReportsConflict(t *testing.T) {
|
||||
const mine, other = uint64(1), uint64(2)
|
||||
m := NewManager()
|
||||
// Another mount took the lock on this (new) owner during the gap.
|
||||
m.TryLock("k", Range{Start: 0, End: 99, Type: Write, Sid: other, Owner: 1})
|
||||
|
||||
conflicts := m.Reassert("k", mine, []Range{{Start: 0, End: 99, Type: Write, Sid: mine, Owner: 1}})
|
||||
if len(conflicts) != 1 {
|
||||
t.Fatalf("expected 1 conflict, got %d: %+v", len(conflicts), conflicts)
|
||||
}
|
||||
// The other session keeps the lock; mine was not installed.
|
||||
if got := len(m.byKey["k"].locks); got != 1 {
|
||||
t.Fatalf("expected only the incumbent lock, got %d", got)
|
||||
}
|
||||
}
|
||||
|
||||
// Reassert renews the lease, so a re-asserting mount is not reaped.
|
||||
func TestReassertRenewsLease(t *testing.T) {
|
||||
const sid = uint64(1)
|
||||
m := NewManager()
|
||||
m.Renew(sid)
|
||||
m.Reassert("k", sid, []Range{{Start: 0, End: 9, Type: Write, Sid: sid, Owner: 1}})
|
||||
|
||||
if reaped := m.ReapExpired(time.Hour); len(reaped) != 0 {
|
||||
t.Fatalf("freshly re-asserted session should not be reaped: %v", reaped)
|
||||
}
|
||||
}
|
||||
|
||||
const maxEnd = ^uint64(0)
|
||||
@@ -10,7 +10,7 @@ import (
|
||||
|
||||
func (store *UniversalRedis2Store) KvPut(ctx context.Context, key []byte, value []byte) (err error) {
|
||||
|
||||
_, err = store.Client.Set(ctx, store.getKey(string(key)), value, 0).Result()
|
||||
_, err = store.Client.Set(ctx, string(key), value, 0).Result()
|
||||
|
||||
if err != nil {
|
||||
return fmt.Errorf("kv put: %w", err)
|
||||
@@ -21,7 +21,7 @@ func (store *UniversalRedis2Store) KvPut(ctx context.Context, key []byte, value
|
||||
|
||||
func (store *UniversalRedis2Store) KvGet(ctx context.Context, key []byte) (value []byte, err error) {
|
||||
|
||||
data, err := store.Client.Get(ctx, store.getKey(string(key))).Result()
|
||||
data, err := store.Client.Get(ctx, string(key)).Result()
|
||||
|
||||
if err == redis.Nil {
|
||||
return nil, filer.ErrKvNotFound
|
||||
@@ -32,7 +32,7 @@ func (store *UniversalRedis2Store) KvGet(ctx context.Context, key []byte) (value
|
||||
|
||||
func (store *UniversalRedis2Store) KvDelete(ctx context.Context, key []byte) (err error) {
|
||||
|
||||
_, err = store.Client.Del(ctx, store.getKey(string(key))).Result()
|
||||
_, err = store.Client.Del(ctx, string(key)).Result()
|
||||
|
||||
if err != nil {
|
||||
return fmt.Errorf("kv delete: %w", err)
|
||||
|
||||
@@ -35,7 +35,7 @@ func TestCreateAndFind(t *testing.T) {
|
||||
},
|
||||
}
|
||||
|
||||
if err := testFiler.CreateEntry(ctx, entry1, nil, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
|
||||
if err := testFiler.CreateEntry(ctx, entry1, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
|
||||
t.Errorf("create entry %v: %v", entry1.FullPath, err)
|
||||
return
|
||||
}
|
||||
@@ -149,7 +149,7 @@ func TestListDirectoryWithPrefix(t *testing.T) {
|
||||
Gid: 1,
|
||||
},
|
||||
}
|
||||
if err := testFiler.CreateEntry(ctx, entry, nil, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
|
||||
if err := testFiler.CreateEntry(ctx, entry, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
|
||||
t.Fatalf("Failed to create entry %s: %v", fullpath, err)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -35,12 +35,7 @@ func (fh *FileHandle) readFromChunksWithContext(ctx context.Context, buff []byte
|
||||
|
||||
entry := fh.GetEntry()
|
||||
|
||||
// IsInRemoteOnly inspects entry.Chunks, so take the LockedEntry lock the
|
||||
// async uploader appends under.
|
||||
entry.RLock()
|
||||
remoteOnly := entry.Entry.IsInRemoteOnly()
|
||||
entry.RUnlock()
|
||||
if remoteOnly {
|
||||
if entry.IsInRemoteOnly() {
|
||||
glog.V(4).Infof("download remote entry %s", fileFullPath)
|
||||
err := fh.downloadRemoteEntry(entry)
|
||||
if err != nil {
|
||||
@@ -49,21 +44,10 @@ func (fh *FileHandle) readFromChunksWithContext(ctx context.Context, buff []byte
|
||||
}
|
||||
}
|
||||
|
||||
// Snapshot size, inline content, and the chunk list under the LockedEntry
|
||||
// lock. Async upload workers append chunks under this lock (AddChunks), so
|
||||
// reading entry.Chunks / FileSize without it races with the slice
|
||||
// reallocation and can crash in filer.TotalSize. The captured slice headers
|
||||
// stay valid afterwards: append never mutates the old backing array, and
|
||||
// truncate is excluded by the fh.entryLock held for this whole read.
|
||||
entry.RLock()
|
||||
pbEntry := entry.Entry
|
||||
fileSize := int64(pbEntry.Attributes.FileSize)
|
||||
fileSize := int64(entry.Attributes.FileSize)
|
||||
if fileSize == 0 {
|
||||
fileSize = int64(filer.FileSize(pbEntry))
|
||||
fileSize = int64(filer.FileSize(entry.GetEntry()))
|
||||
}
|
||||
content := pbEntry.Content
|
||||
chunks := pbEntry.Chunks
|
||||
entry.RUnlock()
|
||||
|
||||
if fileSize == 0 {
|
||||
glog.V(1).Infof("empty fh %v", fileFullPath)
|
||||
@@ -75,15 +59,15 @@ func (fh *FileHandle) readFromChunksWithContext(ctx context.Context, buff []byte
|
||||
return 0, 0, io.EOF
|
||||
}
|
||||
|
||||
if offset < int64(len(content)) {
|
||||
totalRead := copy(buff, content[offset:])
|
||||
if offset < int64(len(entry.Content)) {
|
||||
totalRead := copy(buff, entry.Content[offset:])
|
||||
glog.V(4).Infof("file handle read cached %s [%d,%d] %d", fileFullPath, offset, offset+int64(totalRead), totalRead)
|
||||
return int64(totalRead), 0, nil
|
||||
}
|
||||
|
||||
// Try RDMA acceleration first if available
|
||||
if fh.wfs.rdmaClient != nil && fh.wfs.option.RdmaEnabled {
|
||||
totalRead, ts, err := fh.tryRDMARead(ctx, fileSize, buff, offset, chunks)
|
||||
totalRead, ts, err := fh.tryRDMARead(ctx, fileSize, buff, offset, entry)
|
||||
if err == nil {
|
||||
glog.V(4).Infof("RDMA read successful for %s [%d,%d] %d", fileFullPath, offset, offset+int64(totalRead), totalRead)
|
||||
return int64(totalRead), ts, nil
|
||||
@@ -95,7 +79,7 @@ func (fh *FileHandle) readFromChunksWithContext(ctx context.Context, buff []byte
|
||||
// Any failure falls through transparently. See design-weed-mount-
|
||||
// peer-chunk-sharing.md §4.3.
|
||||
if fh.wfs.option.PeerEnabled && fh.wfs.peerGrpcServer != nil {
|
||||
totalRead, ts, err := fh.tryPeerRead(ctx, fileSize, buff, offset, chunks)
|
||||
totalRead, ts, err := fh.tryPeerRead(ctx, fileSize, buff, offset, entry)
|
||||
if err == nil {
|
||||
glog.V(4).Infof("peer read successful for %s [%d,%d] %d", fileFullPath, offset, offset+int64(totalRead), totalRead)
|
||||
return int64(totalRead), ts, nil
|
||||
@@ -120,13 +104,13 @@ func (fh *FileHandle) readFromChunksWithContext(ctx context.Context, buff []byte
|
||||
return int64(totalRead), ts, err
|
||||
}
|
||||
|
||||
// tryRDMARead attempts to read file data using RDMA acceleration. chunks is a
|
||||
// snapshot captured under the LockedEntry lock by the caller.
|
||||
func (fh *FileHandle) tryRDMARead(ctx context.Context, fileSize int64, buff []byte, offset int64, chunks []*filer_pb.FileChunk) (int64, int64, error) {
|
||||
// tryRDMARead attempts to read file data using RDMA acceleration
|
||||
func (fh *FileHandle) tryRDMARead(ctx context.Context, fileSize int64, buff []byte, offset int64, entry *LockedEntry) (int64, int64, error) {
|
||||
// For now, we'll try to read the chunks directly using RDMA
|
||||
// This is a simplified approach - in a full implementation, we'd need to
|
||||
// handle chunk boundaries, multiple chunks, etc.
|
||||
|
||||
chunks := entry.GetEntry().Chunks
|
||||
if len(chunks) == 0 {
|
||||
return 0, 0, fmt.Errorf("no chunks available for RDMA read")
|
||||
}
|
||||
|
||||
@@ -53,8 +53,7 @@ const maxPeerFetchChunkBytes = 64 * 1024 * 1024
|
||||
// end-to-end against FileChunk.ETag.
|
||||
// 4. On success, populate chunk_cache and enqueue an announce so
|
||||
// other mounts can discover us as a new holder.
|
||||
// chunks is a snapshot captured under the LockedEntry lock by the caller.
|
||||
func (fh *FileHandle) tryPeerRead(ctx context.Context, fileSize int64, buff []byte, offset int64, chunks []*filer_pb.FileChunk) (int64, int64, error) {
|
||||
func (fh *FileHandle) tryPeerRead(ctx context.Context, fileSize int64, buff []byte, offset int64, entry *LockedEntry) (int64, int64, error) {
|
||||
if fh.wfs.peerRegistrar == nil || fh.wfs.peerConnPool == nil {
|
||||
return 0, 0, fmt.Errorf("peer sharing not configured")
|
||||
}
|
||||
@@ -64,7 +63,7 @@ func (fh *FileHandle) tryPeerRead(ctx context.Context, fileSize int64, buff []by
|
||||
if readStop > fileSize {
|
||||
readStop = fileSize
|
||||
}
|
||||
dataChunks, _, err := filer.ResolveChunkManifest(ctx, fh.wfs.LookupFn(), chunks, offset, readStop)
|
||||
dataChunks, _, err := filer.ResolveChunkManifest(ctx, fh.wfs.LookupFn(), entry.GetEntry().Chunks, offset, readStop)
|
||||
if err != nil {
|
||||
return 0, 0, fmt.Errorf("resolve manifest: %w", err)
|
||||
}
|
||||
|
||||
+2
-14
@@ -15,7 +15,6 @@ import (
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/cluster"
|
||||
"github.com/seaweedfs/seaweedfs/weed/filer"
|
||||
"github.com/seaweedfs/seaweedfs/weed/filer/posixlock"
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/mount/meta_cache"
|
||||
"github.com/seaweedfs/seaweedfs/weed/mount/page_writer"
|
||||
@@ -98,10 +97,8 @@ type Option struct {
|
||||
|
||||
// EnableDistributedLock enables DLM-based write coordination across mounts.
|
||||
// When true, opening a file for write acquires a distributed lock that is
|
||||
// held (with auto-renewal) until the file is closed, so only one mount can
|
||||
// have a file open for writing at a time; POSIX advisory locks (flock/fcntl)
|
||||
// are also routed to the inode's owner filer so they are honored across
|
||||
// mounts. Disabled under writeback cache, which implies single-writer.
|
||||
// held (with auto-renewal) until the file is closed. Only one mount can
|
||||
// have a file open for writing at a time.
|
||||
EnableDistributedLock bool
|
||||
|
||||
// WritebackCache enables async flush on close for improved small file write performance.
|
||||
@@ -141,9 +138,6 @@ type WFS struct {
|
||||
fhLockTable *util.LockTable[FileHandleId]
|
||||
hardLinkLockTable *util.LockTable[string]
|
||||
posixLocks *PosixLockTable
|
||||
posixSid uint64 // this mount's session id, for routed-lock owner identity
|
||||
posixHint *posixLockHint // local fcntl-lock hint for routed mode
|
||||
posixOwn *posixlock.Manager // mirror of locks this mount holds, re-asserted via keepalive
|
||||
rdmaClient *RDMAMountClient
|
||||
peerRegistrar *PeerRegistrar
|
||||
peerDirectory *PeerDirectory
|
||||
@@ -248,9 +242,6 @@ func NewSeaweedFileSystem(option *Option) *WFS {
|
||||
fhLockTable: util.NewLockTable[FileHandleId](),
|
||||
hardLinkLockTable: util.NewLockTable[string](),
|
||||
posixLocks: NewPosixLockTable(),
|
||||
posixSid: randomPosixSid(),
|
||||
posixHint: newPosixLockHint(),
|
||||
posixOwn: posixlock.NewManager(),
|
||||
refreshingDirs: make(map[util.FullPath]struct{}),
|
||||
atimeMap: make(map[uint64]time.Time, 8192),
|
||||
openMtimeCache: make(map[uint64][2]int64, 8192),
|
||||
@@ -514,9 +505,6 @@ func (wfs *WFS) StartBackgroundTasks() error {
|
||||
go wfs.loopFlushDirtyMetadata()
|
||||
go wfs.loopEvictIdleDirCache()
|
||||
go wfs.loopProactiveFlush()
|
||||
if wfs.crossMountLocks() {
|
||||
go wfs.loopRenewPosixLeases()
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -23,21 +23,10 @@ func (wfs *WFS) GetAttr(cancel <-chan struct{}, input *fuse.GetAttrIn, out *fuse
|
||||
}
|
||||
|
||||
inode := input.NodeId
|
||||
path, fh, entry, status := wfs.maybeReadEntry(inode)
|
||||
path, _, entry, status := wfs.maybeReadEntry(inode)
|
||||
if status == fuse.OK {
|
||||
out.AttrValid = wfs.attrValidSec
|
||||
// When an open handle owns the entry, async upload workers append
|
||||
// chunks under the LockedEntry lock; take it for reading so FileSize
|
||||
// does not iterate the chunk slice mid-reallocation. Re-read under the
|
||||
// lock in case SetEntry swapped the pointer since maybeReadEntry.
|
||||
if fh != nil {
|
||||
fh.entry.RLock()
|
||||
entry = fh.entry.Entry
|
||||
}
|
||||
wfs.setAttrByPbEntry(&out.Attr, inode, entry, true)
|
||||
if fh != nil {
|
||||
fh.entry.RUnlock()
|
||||
}
|
||||
wfs.applyInMemoryAtime(&out.Attr, inode)
|
||||
if entry.IsDirectory {
|
||||
wfs.applyInMemoryDirMtime(&out.Attr, inode)
|
||||
@@ -51,9 +40,7 @@ func (wfs *WFS) GetAttr(cancel <-chan struct{}, input *fuse.GetAttrIn, out *fuse
|
||||
out.AttrValid = wfs.attrValidSec
|
||||
// Use shared lock to prevent race with Write operations
|
||||
fhActiveLock := wfs.fhLockTable.AcquireLock("GetAttr", fh.fh, util.SharedLock)
|
||||
fh.entry.RLock()
|
||||
wfs.setAttrByPbEntry(&out.Attr, inode, fh.entry.Entry, true)
|
||||
fh.entry.RUnlock()
|
||||
wfs.setAttrByPbEntry(&out.Attr, inode, fh.entry.GetEntry(), true)
|
||||
wfs.fhLockTable.ReleaseLock(fh.fh, fhActiveLock)
|
||||
wfs.applyInMemoryAtime(&out.Attr, inode)
|
||||
out.Nlink = 0
|
||||
@@ -78,15 +65,6 @@ func (wfs *WFS) SetAttr(cancel <-chan struct{}, input *fuse.SetAttrIn, out *fuse
|
||||
if fh != nil {
|
||||
fh.entryLock.Lock()
|
||||
defer fh.entryLock.Unlock()
|
||||
// entry is the handle's shared LockedEntry.Entry. Async upload workers
|
||||
// mutate its Chunks slice under the LockedEntry lock (AddChunks); hold
|
||||
// that same lock so the truncate and FileSize reads below don't tear
|
||||
// against a concurrent append. Re-read under the lock in case SetEntry
|
||||
// swapped the pointer since maybeReadEntry, so we don't mutate an
|
||||
// orphaned entry and lose the update.
|
||||
fh.entry.Lock()
|
||||
defer fh.entry.Unlock()
|
||||
entry = fh.entry.Entry
|
||||
}
|
||||
|
||||
wormEnforced, wormEnabled := wfs.wormEnforcedForEntry(path, entry)
|
||||
|
||||
@@ -1,152 +0,0 @@
|
||||
package mount
|
||||
|
||||
import (
|
||||
"sync"
|
||||
"testing"
|
||||
|
||||
"github.com/seaweedfs/go-fuse/v2/fuse"
|
||||
"github.com/seaweedfs/seaweedfs/weed/filer"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
)
|
||||
|
||||
// TestAttrChunkRace guards the locking around an open handle's chunk slice.
|
||||
//
|
||||
// With writebackCache, async upload workers append chunks to an open file
|
||||
// handle's shared entry under the LockedEntry lock (FileHandle.AddChunks),
|
||||
// while metadata ops compute the file size by iterating entry.Chunks. SetAttr
|
||||
// and GetAttr used to read that slice without the LockedEntry lock, so a
|
||||
// concurrent append that reallocated the backing array produced a torn slice
|
||||
// read and a nil pointer dereference in filer.TotalSize. Run under -race.
|
||||
func TestAttrChunkRace(t *testing.T) {
|
||||
wfs := &WFS{
|
||||
option: &Option{},
|
||||
inodeToPath: NewInodeToPath(util.FullPath("/"), 0),
|
||||
fhMap: NewFileHandleToInode(),
|
||||
openMtimeCache: make(map[uint64][2]int64, 8),
|
||||
}
|
||||
|
||||
const inode = uint64(42)
|
||||
fullPath := util.FullPath("/dir/sample.txt")
|
||||
wfs.inodeToPath.Lookup(fullPath, 1, false, false, inode, true)
|
||||
|
||||
entry := &filer_pb.Entry{
|
||||
Name: "sample.txt",
|
||||
Attributes: &filer_pb.FuseAttributes{FileMode: 0644},
|
||||
}
|
||||
chunkGroup, err := filer.NewChunkGroup(nil, nil, nil, 1)
|
||||
if err != nil {
|
||||
t.Fatalf("NewChunkGroup: %v", err)
|
||||
}
|
||||
fh := &FileHandle{
|
||||
fh: FileHandleId(1),
|
||||
inode: inode,
|
||||
wfs: wfs,
|
||||
entry: &LockedEntry{Entry: entry},
|
||||
entryChunkGroup: chunkGroup,
|
||||
}
|
||||
wfs.fhMap.inode2fh[inode] = fh
|
||||
wfs.fhMap.fh2inode[fh.fh] = inode
|
||||
|
||||
const iterations = 2000
|
||||
var wg sync.WaitGroup
|
||||
wg.Add(3)
|
||||
|
||||
// Async uploader: append chunks, reallocating the backing array.
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
for i := 0; i < iterations; i++ {
|
||||
fh.AddChunks([]*filer_pb.FileChunk{{FileId: "x", Offset: int64(i), Size: 1}})
|
||||
}
|
||||
}()
|
||||
|
||||
// SetAttr: mtime-only recomputes FileSize by iterating chunks; a shrinking
|
||||
// size takes the truncate path that rewrites entry.Chunks under the lock.
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
for i := 0; i < iterations; i++ {
|
||||
in := &fuse.SetAttrIn{}
|
||||
in.NodeId = inode
|
||||
if i%2 == 0 {
|
||||
in.Valid = fuse.FATTR_MTIME
|
||||
in.Mtime = uint64(i)
|
||||
} else {
|
||||
in.Valid = fuse.FATTR_SIZE
|
||||
in.Size = uint64(i % 8)
|
||||
}
|
||||
var out fuse.AttrOut
|
||||
wfs.SetAttr(nil, in, &out)
|
||||
}
|
||||
}()
|
||||
|
||||
// GetAttr also computes FileSize by iterating chunks.
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
for i := 0; i < iterations; i++ {
|
||||
in := &fuse.GetAttrIn{}
|
||||
in.NodeId = inode
|
||||
var out fuse.AttrOut
|
||||
wfs.GetAttr(nil, in, &out)
|
||||
}
|
||||
}()
|
||||
|
||||
wg.Wait()
|
||||
}
|
||||
|
||||
// TestReadFromChunksRace guards the read path's chunk-slice access. The read
|
||||
// path holds fh.entryLock (which excludes SetAttr) but not the LockedEntry lock
|
||||
// the async uploader appends under, so readFromChunks used to compute FileSize
|
||||
// and walk entry.Chunks while AddChunks reallocated the slice. Run under -race.
|
||||
func TestReadFromChunksRace(t *testing.T) {
|
||||
wfs := &WFS{
|
||||
option: &Option{},
|
||||
inodeToPath: NewInodeToPath(util.FullPath("/"), 0),
|
||||
fhMap: NewFileHandleToInode(),
|
||||
}
|
||||
|
||||
const inode = uint64(42)
|
||||
fullPath := util.FullPath("/dir/sample.txt")
|
||||
wfs.inodeToPath.Lookup(fullPath, 1, false, false, inode, true)
|
||||
|
||||
// FileSize 0 forces readFromChunks down the filer.FileSize(chunks) branch.
|
||||
entry := &filer_pb.Entry{
|
||||
Name: "sample.txt",
|
||||
Attributes: &filer_pb.FuseAttributes{FileMode: 0644},
|
||||
}
|
||||
chunkGroup, err := filer.NewChunkGroup(nil, nil, nil, 1)
|
||||
if err != nil {
|
||||
t.Fatalf("NewChunkGroup: %v", err)
|
||||
}
|
||||
fh := &FileHandle{
|
||||
fh: FileHandleId(1),
|
||||
inode: inode,
|
||||
wfs: wfs,
|
||||
entry: &LockedEntry{Entry: entry},
|
||||
entryChunkGroup: chunkGroup,
|
||||
}
|
||||
wfs.fhMap.inode2fh[inode] = fh
|
||||
wfs.fhMap.fh2inode[fh.fh] = inode
|
||||
|
||||
const iterations = 2000
|
||||
var wg sync.WaitGroup
|
||||
wg.Add(2)
|
||||
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
for i := 0; i < iterations; i++ {
|
||||
fh.AddChunks([]*filer_pb.FileChunk{{FileId: "x", Offset: int64(i), Size: 1}})
|
||||
}
|
||||
}()
|
||||
|
||||
// A read past EOF returns before touching the volume tier, but only after
|
||||
// the racy size/chunk snapshot has run.
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
buff := make([]byte, 16)
|
||||
for i := 0; i < iterations; i++ {
|
||||
fh.readFromChunks(buff, 1<<62)
|
||||
}
|
||||
}()
|
||||
|
||||
wg.Wait()
|
||||
}
|
||||
@@ -148,7 +148,7 @@ func (wfs *WFS) Release(cancel <-chan struct{}, in *fuse.ReleaseIn) {
|
||||
}
|
||||
}
|
||||
if in.ReleaseFlags&fuse.FUSE_RELEASE_FLOCK_UNLOCK != 0 {
|
||||
wfs.releaseFlockOwner(in.NodeId, in.LockOwner)
|
||||
wfs.posixLocks.ReleaseFlockOwner(in.NodeId, in.LockOwner)
|
||||
}
|
||||
wfs.ReleaseHandle(FileHandleId(in.Fh))
|
||||
}
|
||||
|
||||
@@ -10,9 +10,6 @@ import (
|
||||
// If a conflict exists, the conflicting lock is returned in out.
|
||||
// If no conflict, out.Lk.Typ is set to F_UNLCK.
|
||||
func (wfs *WFS) GetLk(cancel <-chan struct{}, in *fuse.LkIn, out *fuse.LkOut) fuse.Status {
|
||||
if wfs.crossMountLocks() {
|
||||
return wfs.routedGetLk(cancel, in, out)
|
||||
}
|
||||
proposed := lockRange{
|
||||
Start: in.Lk.Start,
|
||||
End: in.Lk.End,
|
||||
@@ -28,9 +25,6 @@ func (wfs *WFS) GetLk(cancel <-chan struct{}, in *fuse.LkIn, out *fuse.LkOut) fu
|
||||
// SetLk sets or clears a POSIX lock (non-blocking).
|
||||
// Returns EAGAIN if the lock conflicts with an existing lock from another owner.
|
||||
func (wfs *WFS) SetLk(cancel <-chan struct{}, in *fuse.LkIn) fuse.Status {
|
||||
if wfs.crossMountLocks() {
|
||||
return wfs.routedSetLk(cancel, in)
|
||||
}
|
||||
lk := lockRange{
|
||||
Start: in.Lk.Start,
|
||||
End: in.Lk.End,
|
||||
@@ -45,9 +39,6 @@ func (wfs *WFS) SetLk(cancel <-chan struct{}, in *fuse.LkIn) fuse.Status {
|
||||
// SetLkw sets a POSIX lock (blocking).
|
||||
// Waits until the lock can be acquired or the request is cancelled.
|
||||
func (wfs *WFS) SetLkw(cancel <-chan struct{}, in *fuse.LkIn) fuse.Status {
|
||||
if wfs.crossMountLocks() {
|
||||
return wfs.routedSetLkw(cancel, in)
|
||||
}
|
||||
lk := lockRange{
|
||||
Start: in.Lk.Start,
|
||||
End: in.Lk.End,
|
||||
|
||||
@@ -60,7 +60,7 @@ func (wfs *WFS) Flush(cancel <-chan struct{}, in *fuse.FlushIn) fuse.Status {
|
||||
// If handle is not found, it might have been already released
|
||||
// This is not an error condition for FLUSH
|
||||
if in.LockOwner != 0 {
|
||||
wfs.releasePosixOwner(in.NodeId, in.LockOwner)
|
||||
wfs.posixLocks.ReleasePosixOwner(in.NodeId, in.LockOwner)
|
||||
}
|
||||
return fuse.OK
|
||||
}
|
||||
@@ -69,11 +69,11 @@ func (wfs *WFS) Flush(cancel <-chan struct{}, in *fuse.FlushIn) fuse.Status {
|
||||
// did not hold byte-range locks. Only force the synchronous close path when
|
||||
// this owner actually has POSIX locks to release; otherwise writebackCache
|
||||
// would silently degrade to a blocking flush for ordinary close().
|
||||
hasPosixLocks := wfs.hasPosixOwner(in.NodeId, in.LockOwner)
|
||||
hasPosixLocks := wfs.posixLocks.HasPosixOwner(in.NodeId, in.LockOwner)
|
||||
allowAsync := !hasPosixLocks
|
||||
status := wfs.doFlush(fh, in.Uid, in.Gid, allowAsync)
|
||||
if in.LockOwner != 0 {
|
||||
wfs.releasePosixOwner(in.NodeId, in.LockOwner)
|
||||
wfs.posixLocks.ReleasePosixOwner(in.NodeId, in.LockOwner)
|
||||
}
|
||||
return status
|
||||
}
|
||||
|
||||
@@ -1,426 +0,0 @@
|
||||
package mount
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/binary"
|
||||
"encoding/hex"
|
||||
"math/rand/v2"
|
||||
"sync"
|
||||
"syscall"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/go-fuse/v2/fuse"
|
||||
"github.com/seaweedfs/seaweedfs/weed/filer/posixlock"
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
)
|
||||
|
||||
// randomPosixSid returns a process-stable, cluster-unique-enough session id for
|
||||
// this mount, used to namespace lock owners so the same FUSE owner value on a
|
||||
// different mount never aliases, and to scope lease-based reaping.
|
||||
func randomPosixSid() uint64 {
|
||||
return binary.BigEndian.Uint64(util.RandomBytes(8))
|
||||
}
|
||||
|
||||
// Routed POSIX locking: when -dlm is set, advisory locks are serialized on the
|
||||
// inode's owner filer instead of in this mount's local table, so locks are
|
||||
// honored across mounts. The mount calls its filer and relies on filer-side
|
||||
// forwarding to reach the owner. Blocking (SetLkw) is client-side polling — there
|
||||
// is no server-side wait queue.
|
||||
|
||||
const (
|
||||
posixLockMinBackoff = 5 * time.Millisecond
|
||||
posixLockMaxBackoff = 200 * time.Millisecond
|
||||
posixLockKeyPrefix = "s3.fuse.lock:"
|
||||
// posixLockReleaseTimeout bounds the background unlock/release RPCs. They run
|
||||
// off the syscall path (close/flush) and must not be cancelled by an
|
||||
// interrupt, but a deadline keeps a slow or unreachable filer from blocking
|
||||
// close indefinitely. It also bounds each keepalive RPC.
|
||||
posixLockReleaseTimeout = 5 * time.Second
|
||||
// posixKeepaliveInterval renews the session lease well within the filer's
|
||||
// posixLockSessionTTL (15s) so a live mount is never reaped.
|
||||
posixKeepaliveInterval = 5 * time.Second
|
||||
)
|
||||
|
||||
// posixLockHint records, per inode, which owners this mount has taken fcntl locks
|
||||
// for. Flush consults it to force a synchronous close and route a release without
|
||||
// an RPC on every close(). It is a superset hint: an extra sync flush or a no-op
|
||||
// release RPC is harmless, a missed release is not — so it is added on grant and
|
||||
// cleared on close (which drops the owner's POSIX locks per POSIX semantics).
|
||||
type posixLockHint struct {
|
||||
mu sync.Mutex
|
||||
m map[uint64]map[uint64]struct{}
|
||||
}
|
||||
|
||||
func newPosixLockHint() *posixLockHint {
|
||||
return &posixLockHint{m: make(map[uint64]map[uint64]struct{})}
|
||||
}
|
||||
|
||||
func (h *posixLockHint) add(inode, owner uint64) {
|
||||
h.mu.Lock()
|
||||
defer h.mu.Unlock()
|
||||
owners := h.m[inode]
|
||||
if owners == nil {
|
||||
owners = make(map[uint64]struct{})
|
||||
h.m[inode] = owners
|
||||
}
|
||||
owners[owner] = struct{}{}
|
||||
}
|
||||
|
||||
func (h *posixLockHint) has(inode, owner uint64) bool {
|
||||
h.mu.Lock()
|
||||
defer h.mu.Unlock()
|
||||
_, ok := h.m[inode][owner]
|
||||
return ok
|
||||
}
|
||||
|
||||
func (h *posixLockHint) drop(inode, owner uint64) {
|
||||
h.mu.Lock()
|
||||
defer h.mu.Unlock()
|
||||
if owners := h.m[inode]; owners != nil {
|
||||
delete(owners, owner)
|
||||
if len(owners) == 0 {
|
||||
delete(h.m, inode)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// posixLockKeyForInode resolves a FUSE inode to its cluster-stable lock identity:
|
||||
// the HardLinkId for a hardlinked file (so all names share one owner and table),
|
||||
// else the path. POSIX locks are inode-scoped, so this must never be the FUSE
|
||||
// NodeId (mount-local) for a multi-named inode.
|
||||
func (wfs *WFS) posixLockKeyForInode(inode uint64) (string, bool) {
|
||||
path, status := wfs.inodeToPath.GetPath(inode)
|
||||
if status != fuse.OK {
|
||||
return "", false
|
||||
}
|
||||
if entry, st := wfs.maybeLoadEntry(path); st == fuse.OK && entry != nil && len(entry.HardLinkId) > 0 {
|
||||
return posixLockKeyPrefix + "hl:" + hex.EncodeToString(entry.HardLinkId), true
|
||||
}
|
||||
return posixLockKeyPrefix + string(path), true
|
||||
}
|
||||
|
||||
func posixLockTypeToWire(typ uint32) uint32 {
|
||||
switch typ {
|
||||
case syscall.F_RDLCK:
|
||||
return posixlock.Read
|
||||
case syscall.F_WRLCK:
|
||||
return posixlock.Write
|
||||
default:
|
||||
return posixlock.Unlock
|
||||
}
|
||||
}
|
||||
|
||||
func posixLockTypeFromWire(typ uint32) uint32 {
|
||||
switch typ {
|
||||
case posixlock.Read:
|
||||
return syscall.F_RDLCK
|
||||
case posixlock.Write:
|
||||
return syscall.F_WRLCK
|
||||
default:
|
||||
return syscall.F_UNLCK
|
||||
}
|
||||
}
|
||||
|
||||
func (wfs *WFS) posixRangeFromLkIn(in *fuse.LkIn) posixlock.Range {
|
||||
return posixlock.Range{
|
||||
Start: in.Lk.Start,
|
||||
End: in.Lk.End,
|
||||
Type: posixLockTypeToWire(in.Lk.Typ),
|
||||
Sid: wfs.posixSid,
|
||||
Owner: in.Owner,
|
||||
Pid: in.Lk.Pid,
|
||||
IsFlock: in.LkFlags&fuse.FUSE_LK_FLOCK != 0,
|
||||
}
|
||||
}
|
||||
|
||||
// posixLockContext derives a context from the FUSE cancel channel so an
|
||||
// interrupted lock syscall aborts its in-flight RPC. The caller must call the
|
||||
// returned func to release the watcher goroutine.
|
||||
func posixLockContext(cancel <-chan struct{}) (context.Context, context.CancelFunc) {
|
||||
ctx, cancelFn := context.WithCancel(context.Background())
|
||||
if cancel != nil {
|
||||
go func() {
|
||||
select {
|
||||
case <-cancel:
|
||||
cancelFn()
|
||||
case <-ctx.Done():
|
||||
}
|
||||
}()
|
||||
}
|
||||
return ctx, cancelFn
|
||||
}
|
||||
|
||||
func (wfs *WFS) callPosixLock(ctx context.Context, key string, op filer_pb.PosixLockOp, lk posixlock.Range) (*filer_pb.PosixLockResponse, error) {
|
||||
var resp *filer_pb.PosixLockResponse
|
||||
err := wfs.WithFilerClient(false, func(client filer_pb.SeaweedFilerClient) error {
|
||||
var e error
|
||||
resp, e = client.PosixLock(ctx, &filer_pb.PosixLockRequest{
|
||||
Key: key,
|
||||
Op: op,
|
||||
Lock: &filer_pb.PosixLockRange{
|
||||
Start: lk.Start, End: lk.End, Type: lk.Type,
|
||||
Sid: lk.Sid, Owner: lk.Owner, Pid: lk.Pid, IsFlock: lk.IsFlock,
|
||||
},
|
||||
})
|
||||
return e
|
||||
})
|
||||
return resp, err
|
||||
}
|
||||
|
||||
func (wfs *WFS) routedGetLk(cancel <-chan struct{}, in *fuse.LkIn, out *fuse.LkOut) fuse.Status {
|
||||
key, ok := wfs.posixLockKeyForInode(in.NodeId)
|
||||
if !ok {
|
||||
return fuse.EINVAL
|
||||
}
|
||||
ctx, done := posixLockContext(cancel)
|
||||
defer done()
|
||||
resp, err := wfs.callPosixLock(ctx, key, filer_pb.PosixLockOp_GET_LK, wfs.posixRangeFromLkIn(in))
|
||||
if err != nil {
|
||||
glog.Warningf("routed GetLk %s: %v", key, err)
|
||||
return fuse.EIO
|
||||
}
|
||||
if resp.GetHasConflict() {
|
||||
c := resp.GetConflict()
|
||||
out.Lk.Start, out.Lk.End, out.Lk.Pid = c.GetStart(), c.GetEnd(), c.GetPid()
|
||||
out.Lk.Typ = posixLockTypeFromWire(c.GetType())
|
||||
} else {
|
||||
out.Lk.Typ = syscall.F_UNLCK
|
||||
}
|
||||
return fuse.OK
|
||||
}
|
||||
|
||||
func (wfs *WFS) routedSetLk(cancel <-chan struct{}, in *fuse.LkIn) fuse.Status {
|
||||
key, ok := wfs.posixLockKeyForInode(in.NodeId)
|
||||
if !ok {
|
||||
return fuse.EINVAL
|
||||
}
|
||||
lk := wfs.posixRangeFromLkIn(in)
|
||||
if lk.Type == posixlock.Unlock {
|
||||
return wfs.routedUnlock(key, lk)
|
||||
}
|
||||
ctx, done := posixLockContext(cancel)
|
||||
defer done()
|
||||
resp, err := wfs.callPosixLock(ctx, key, filer_pb.PosixLockOp_TRY_LOCK, lk)
|
||||
if err != nil {
|
||||
glog.Warningf("routed SetLk %s: %v", key, err)
|
||||
return fuse.EIO
|
||||
}
|
||||
if !resp.GetGranted() {
|
||||
return fuse.EAGAIN
|
||||
}
|
||||
wfs.recordPosixGrant(key, in.NodeId, lk)
|
||||
return fuse.OK
|
||||
}
|
||||
|
||||
func (wfs *WFS) routedSetLkw(cancel <-chan struct{}, in *fuse.LkIn) fuse.Status {
|
||||
key, ok := wfs.posixLockKeyForInode(in.NodeId)
|
||||
if !ok {
|
||||
return fuse.EINVAL
|
||||
}
|
||||
lk := wfs.posixRangeFromLkIn(in)
|
||||
if lk.Type == posixlock.Unlock {
|
||||
return wfs.routedUnlock(key, lk)
|
||||
}
|
||||
ctx, done := posixLockContext(cancel)
|
||||
defer done()
|
||||
status := posixPollAcquire(cancel, func() (bool, error) {
|
||||
resp, err := wfs.callPosixLock(ctx, key, filer_pb.PosixLockOp_TRY_LOCK, lk)
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
return resp.GetGranted(), nil
|
||||
})
|
||||
if status == fuse.OK {
|
||||
wfs.recordPosixGrant(key, in.NodeId, lk)
|
||||
}
|
||||
return status
|
||||
}
|
||||
|
||||
// recordPosixGrant notes a newly granted lock: the flush hint (fcntl only) and
|
||||
// the own-lock mirror, which the keepalive re-asserts to the key's owner so the
|
||||
// lease is renewed and the owner can rebuild after a takeover or restart.
|
||||
func (wfs *WFS) recordPosixGrant(key string, inode uint64, lk posixlock.Range) {
|
||||
if !lk.IsFlock {
|
||||
wfs.posixHint.add(inode, lk.Owner)
|
||||
}
|
||||
wfs.posixOwn.Track(key, lk)
|
||||
}
|
||||
|
||||
func (wfs *WFS) routedUnlock(key string, lk posixlock.Range) fuse.Status {
|
||||
// Drop the range from the mirror first so a keepalive re-assertion can't
|
||||
// resurrect it; if the RPC below fails, the next re-assertion reconciles the
|
||||
// owner to this (released) state anyway.
|
||||
wfs.posixOwn.Unlock(key, lk)
|
||||
// A release must complete even if the syscall was interrupted; cancelling it
|
||||
// would leak the lock on the owner filer. Bound it so a stuck filer can't
|
||||
// hang close() forever.
|
||||
ctx, cancel := context.WithTimeout(context.Background(), posixLockReleaseTimeout)
|
||||
defer cancel()
|
||||
if _, err := wfs.callPosixLock(ctx, key, filer_pb.PosixLockOp_UNLOCK, lk); err != nil {
|
||||
glog.Warningf("routed unlock %s: %v", key, err)
|
||||
return fuse.EIO
|
||||
}
|
||||
return fuse.OK
|
||||
}
|
||||
|
||||
// posixPollAcquire retries try with capped, jittered backoff until it is granted,
|
||||
// the cancel channel fires (EINTR), or try errors (EIO). This is the client-side
|
||||
// stand-in for a server-side wait queue.
|
||||
func posixPollAcquire(cancel <-chan struct{}, try func() (bool, error)) fuse.Status {
|
||||
backoff := posixLockMinBackoff
|
||||
for {
|
||||
select {
|
||||
case <-cancel:
|
||||
return fuse.EINTR
|
||||
default:
|
||||
}
|
||||
granted, err := try()
|
||||
if err != nil {
|
||||
return fuse.EIO
|
||||
}
|
||||
if granted {
|
||||
return fuse.OK
|
||||
}
|
||||
timer := time.NewTimer(backoff + time.Duration(rand.Int64N(int64(backoff))))
|
||||
select {
|
||||
case <-cancel:
|
||||
timer.Stop()
|
||||
return fuse.EINTR
|
||||
case <-timer.C:
|
||||
}
|
||||
if backoff < posixLockMaxBackoff {
|
||||
if backoff *= 2; backoff > posixLockMaxBackoff {
|
||||
backoff = posixLockMaxBackoff
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func (wfs *WFS) routedReleasePosixOwner(inode, owner uint64) {
|
||||
if !wfs.posixHint.has(inode, owner) {
|
||||
return
|
||||
}
|
||||
key, ok := wfs.posixLockKeyForInode(inode)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
wfs.posixOwn.ReleasePosixOwner(key, wfs.posixSid, owner)
|
||||
ctx, cancel := context.WithTimeout(context.Background(), posixLockReleaseTimeout)
|
||||
defer cancel()
|
||||
if _, err := wfs.callPosixLock(ctx, key, filer_pb.PosixLockOp_RELEASE_POSIX_OWNER, posixlock.Range{Sid: wfs.posixSid, Owner: owner}); err != nil {
|
||||
// Keep the hint so a later flush retries the release; dropping it on a
|
||||
// transient failure would strand the lock until the owner filer's
|
||||
// session-lease reaping expires it.
|
||||
glog.Warningf("routed release posix owner %s: %v", key, err)
|
||||
return
|
||||
}
|
||||
wfs.posixHint.drop(inode, owner)
|
||||
}
|
||||
|
||||
func (wfs *WFS) routedReleaseFlockOwner(inode, owner uint64) {
|
||||
key, ok := wfs.posixLockKeyForInode(inode)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
wfs.posixOwn.ReleaseFlockOwner(key, wfs.posixSid, owner)
|
||||
ctx, cancel := context.WithTimeout(context.Background(), posixLockReleaseTimeout)
|
||||
defer cancel()
|
||||
if _, err := wfs.callPosixLock(ctx, key, filer_pb.PosixLockOp_RELEASE_FLOCK_OWNER, posixlock.Range{Sid: wfs.posixSid, Owner: owner}); err != nil {
|
||||
glog.Warningf("routed release flock owner %s: %v", key, err)
|
||||
}
|
||||
}
|
||||
|
||||
// crossMountLocks reports whether advisory locks are routed to the inode's owner
|
||||
// filer rather than served from the per-mount local table. It tracks -dlm: lock
|
||||
// coordination rides the same switch as whole-file write coordination, and is
|
||||
// off under writeback cache (which implies single-writer, so lockClient is nil).
|
||||
func (wfs *WFS) crossMountLocks() bool {
|
||||
return wfs.lockClient != nil
|
||||
}
|
||||
|
||||
// releasePosixOwner / releaseFlockOwner / hasPosixOwner dispatch to the routed
|
||||
// authority or the local table based on crossMountLocks, keeping the Flush and
|
||||
// Release call sites flag-agnostic.
|
||||
|
||||
func (wfs *WFS) releasePosixOwner(inode, owner uint64) {
|
||||
if wfs.crossMountLocks() {
|
||||
wfs.routedReleasePosixOwner(inode, owner)
|
||||
return
|
||||
}
|
||||
wfs.posixLocks.ReleasePosixOwner(inode, owner)
|
||||
}
|
||||
|
||||
func (wfs *WFS) releaseFlockOwner(inode, owner uint64) {
|
||||
if wfs.crossMountLocks() {
|
||||
wfs.routedReleaseFlockOwner(inode, owner)
|
||||
return
|
||||
}
|
||||
wfs.posixLocks.ReleaseFlockOwner(inode, owner)
|
||||
}
|
||||
|
||||
func (wfs *WFS) hasPosixOwner(inode, owner uint64) bool {
|
||||
if wfs.crossMountLocks() {
|
||||
return owner != 0 && wfs.posixHint.has(inode, owner)
|
||||
}
|
||||
return wfs.posixLocks.HasPosixOwner(inode, owner)
|
||||
}
|
||||
|
||||
// callPosixReassert sends the mount's held locks on key to the key's current
|
||||
// owner filer as a KEEP_ALIVE, which renews the lease and lets the owner rebuild
|
||||
// its in-memory state after a takeover or restart.
|
||||
func (wfs *WFS) callPosixReassert(ctx context.Context, key string, locks []posixlock.Range) error {
|
||||
pbLocks := make([]*filer_pb.PosixLockRange, 0, len(locks))
|
||||
for _, l := range locks {
|
||||
pbLocks = append(pbLocks, &filer_pb.PosixLockRange{
|
||||
Start: l.Start, End: l.End, Type: l.Type,
|
||||
Sid: l.Sid, Owner: l.Owner, Pid: l.Pid, IsFlock: l.IsFlock,
|
||||
})
|
||||
}
|
||||
return wfs.WithFilerClient(false, func(client filer_pb.SeaweedFilerClient) error {
|
||||
_, e := client.PosixLock(ctx, &filer_pb.PosixLockRequest{
|
||||
Key: key,
|
||||
Op: filer_pb.PosixLockOp_KEEP_ALIVE,
|
||||
Lock: &filer_pb.PosixLockRange{Sid: wfs.posixSid},
|
||||
Locks: pbLocks,
|
||||
})
|
||||
return e
|
||||
})
|
||||
}
|
||||
|
||||
// posixKeepaliveConcurrency bounds the parallel keepalive RPCs per tick. A mount
|
||||
// holding locks on many inodes must renew every lease well within the filer TTL;
|
||||
// dispatching them concurrently keeps the round trip from scaling with lock count.
|
||||
const posixKeepaliveConcurrency = 32
|
||||
|
||||
// loopRenewPosixLeases re-asserts this mount's held locks to the owner filer of
|
||||
// every key, which renews the session lease and rebuilds the owner's state after
|
||||
// a ring change or owner restart. A dead mount stops re-asserting and the owners'
|
||||
// sweepers reclaim its locks after the TTL.
|
||||
func (wfs *WFS) loopRenewPosixLeases() {
|
||||
ticker := time.NewTicker(posixKeepaliveInterval)
|
||||
defer ticker.Stop()
|
||||
for range ticker.C {
|
||||
held := wfs.posixOwn.Snapshot()
|
||||
var wg sync.WaitGroup
|
||||
sem := make(chan struct{}, posixKeepaliveConcurrency)
|
||||
for key, locks := range held {
|
||||
wg.Add(1)
|
||||
sem <- struct{}{}
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
defer func() { <-sem }()
|
||||
// Bound each re-assertion so a stuck filer can't block wg.Wait and
|
||||
// stall the whole tick, which would let other keys' leases expire
|
||||
// and get reaped.
|
||||
ctx, cancel := context.WithTimeout(context.Background(), posixLockReleaseTimeout)
|
||||
defer cancel()
|
||||
if err := wfs.callPosixReassert(ctx, key, locks); err != nil {
|
||||
glog.V(2).Infof("posix reassert %s: %v", key, err)
|
||||
}
|
||||
}()
|
||||
}
|
||||
wg.Wait()
|
||||
}
|
||||
}
|
||||
@@ -1,94 +0,0 @@
|
||||
package mount
|
||||
|
||||
import (
|
||||
"syscall"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/go-fuse/v2/fuse"
|
||||
"github.com/seaweedfs/seaweedfs/weed/filer/posixlock"
|
||||
)
|
||||
|
||||
func TestPosixLockTypeMapping(t *testing.T) {
|
||||
cases := []struct {
|
||||
sys uint32
|
||||
wire uint32
|
||||
}{
|
||||
{syscall.F_RDLCK, posixlock.Read},
|
||||
{syscall.F_WRLCK, posixlock.Write},
|
||||
{syscall.F_UNLCK, posixlock.Unlock},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if got := posixLockTypeToWire(c.sys); got != c.wire {
|
||||
t.Errorf("toWire(%d) = %d, want %d", c.sys, got, c.wire)
|
||||
}
|
||||
if got := posixLockTypeFromWire(c.wire); got != c.sys {
|
||||
t.Errorf("fromWire(%d) = %d, want %d", c.wire, got, c.sys)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestPosixPollAcquireGrantedImmediately(t *testing.T) {
|
||||
calls := 0
|
||||
st := posixPollAcquire(nil, func() (bool, error) { calls++; return true, nil })
|
||||
if st != fuse.OK || calls != 1 {
|
||||
t.Fatalf("immediate grant: status=%v calls=%d", st, calls)
|
||||
}
|
||||
}
|
||||
|
||||
func TestPosixPollAcquireRetriesThenGrants(t *testing.T) {
|
||||
calls := 0
|
||||
st := posixPollAcquire(nil, func() (bool, error) {
|
||||
calls++
|
||||
return calls >= 3, nil
|
||||
})
|
||||
if st != fuse.OK || calls != 3 {
|
||||
t.Fatalf("retry then grant: status=%v calls=%d", st, calls)
|
||||
}
|
||||
}
|
||||
|
||||
func TestPosixPollAcquireError(t *testing.T) {
|
||||
if st := posixPollAcquire(nil, func() (bool, error) { return false, syscall.EIO }); st != fuse.EIO {
|
||||
t.Fatalf("error should map to EIO, got %v", st)
|
||||
}
|
||||
}
|
||||
|
||||
func TestPosixPollAcquireCancel(t *testing.T) {
|
||||
cancel := make(chan struct{})
|
||||
close(cancel)
|
||||
done := make(chan fuse.Status, 1)
|
||||
go func() {
|
||||
done <- posixPollAcquire(cancel, func() (bool, error) { return false, nil })
|
||||
}()
|
||||
select {
|
||||
case st := <-done:
|
||||
if st != fuse.EINTR {
|
||||
t.Fatalf("cancel should map to EINTR, got %v", st)
|
||||
}
|
||||
case <-time.After(2 * time.Second):
|
||||
t.Fatal("poll did not return on cancel")
|
||||
}
|
||||
}
|
||||
|
||||
func TestPosixLockHint(t *testing.T) {
|
||||
h := newPosixLockHint()
|
||||
if h.has(1, 2) {
|
||||
t.Fatal("empty hint should not report a lock")
|
||||
}
|
||||
h.add(1, 2)
|
||||
h.add(1, 3)
|
||||
if !h.has(1, 2) || !h.has(1, 3) {
|
||||
t.Fatal("added owners should be reported")
|
||||
}
|
||||
h.drop(1, 2)
|
||||
if h.has(1, 2) {
|
||||
t.Fatal("dropped owner should be gone")
|
||||
}
|
||||
if !h.has(1, 3) {
|
||||
t.Fatal("sibling owner should remain")
|
||||
}
|
||||
h.drop(1, 3)
|
||||
if _, ok := h.m[1]; ok {
|
||||
t.Fatal("inode entry should be removed when its last owner drops")
|
||||
}
|
||||
}
|
||||
@@ -31,15 +31,6 @@ service SeaweedFiler {
|
||||
rpc DeleteEntry (DeleteEntryRequest) returns (DeleteEntryResponse) {
|
||||
}
|
||||
|
||||
rpc ObjectTransaction (ObjectTransactionRequest) returns (ObjectTransactionResponse) {
|
||||
}
|
||||
|
||||
rpc ObjectTransactionBatch (ObjectTransactionBatchRequest) returns (ObjectTransactionBatchResponse) {
|
||||
}
|
||||
|
||||
rpc PosixLock (PosixLockRequest) returns (PosixLockResponse) {
|
||||
}
|
||||
|
||||
rpc AtomicRenameEntry (AtomicRenameEntryRequest) returns (AtomicRenameEntryResponse) {
|
||||
}
|
||||
rpc StreamRenameEntry (StreamRenameEntryRequest) returns (stream StreamRenameEntryResponse) {
|
||||
@@ -231,56 +222,6 @@ message CreateEntryRequest {
|
||||
bool is_from_other_cluster = 4;
|
||||
repeated int32 signatures = 5;
|
||||
bool skip_check_parent_directory = 6;
|
||||
// Optional precondition evaluated against the current entry atomically with
|
||||
// the write, under the filer's per-path lock. The caller must route the
|
||||
// key's writes to this entry's owner filer for the check to be authoritative.
|
||||
WriteCondition condition = 7;
|
||||
}
|
||||
|
||||
// WriteCondition is the precondition the filer evaluates against the existing
|
||||
// entry before writing, under the per-path lock. A failed condition returns
|
||||
// FilerError PRECONDITION_FAILED. The client maps request semantics (e.g. RFC
|
||||
// 7232) to clauses; the filer just compares.
|
||||
//
|
||||
// A condition is a list of clauses that ALL must hold (logical AND). One clause
|
||||
// is the common case; several express what a single comparison cannot: an ETag
|
||||
// set (If-Match / If-None-Match with multiple values), weak-ETag comparison, and
|
||||
// compound conditions (e.g. If-Match + If-Unmodified-Since together).
|
||||
message WriteCondition {
|
||||
enum Kind {
|
||||
NONE = 0; // unconditional
|
||||
IF_NOT_EXISTS = 1; // fail if the entry exists (If-None-Match: *)
|
||||
IF_EXISTS = 2; // fail if the entry is absent (If-Match: *)
|
||||
IF_ETAG_MATCH = 3; // fail if absent or etag matches none of the set (If-Match)
|
||||
IF_ETAG_NOT_MATCH = 4; // fail if present and etag matches any of the set (If-None-Match)
|
||||
IF_UNMODIFIED_SINCE = 5; // fail if present and mtime > unix_time
|
||||
IF_MODIFIED_SINCE = 6; // fail if present and mtime <= unix_time
|
||||
IF_EXTENDED_NOT_EQUAL = 7; // fail if present and extended[ext_key] == ext_value
|
||||
IF_EXTENDED_TIME_ELAPSED = 8; // fail if present and extended[ext_key] (unix seconds) is in the future
|
||||
}
|
||||
// Clause is one primitive comparison. IF_ETAG_MATCH holds when the current
|
||||
// entry's ETag equals any value in etags; IF_ETAG_NOT_MATCH holds when it
|
||||
// equals none. allow_weak permits weak-comparison (ignoring the W/ prefix).
|
||||
//
|
||||
// The IF_EXTENDED_* kinds are generic guards on an extended attribute, used
|
||||
// to enforce object-lock without teaching the filer S3 semantics:
|
||||
// IF_EXTENDED_NOT_EQUAL expresses a legal hold (block while a key equals a
|
||||
// value), and IF_EXTENDED_TIME_ELAPSED expresses retention (block while a
|
||||
// stored unix-second deadline is in the future, compared to the filer's
|
||||
// clock). The caller composes these and, for governance-bypass, simply omits
|
||||
// the retention clause when the bypass is authorized — the filer makes no
|
||||
// authorization decision.
|
||||
message Clause {
|
||||
Kind kind = 1;
|
||||
repeated string etags = 2; // ETag set for IF_ETAG_* kinds
|
||||
int64 unix_time = 3; // bound (unix seconds) for IF_*_SINCE kinds
|
||||
bool allow_weak = 4; // compare ETags ignoring the weak (W/) marker
|
||||
string ext_key = 5; // extended attribute name for IF_EXTENDED_* kinds
|
||||
string ext_value = 6; // blocking value for IF_EXTENDED_NOT_EQUAL
|
||||
string gate_key = 7; // IF_EXTENDED_TIME_ELAPSED: only enforce when extended[gate_key] == gate_value
|
||||
string gate_value = 8; // gate value (e.g. retention mode COMPLIANCE for governance bypass)
|
||||
}
|
||||
repeated Clause clauses = 1; // all must hold (logical AND)
|
||||
}
|
||||
|
||||
// Structured error codes for filer entry operations.
|
||||
@@ -292,138 +233,6 @@ enum FilerError {
|
||||
EXISTING_IS_DIRECTORY = 3; // cannot overwrite directory with file
|
||||
EXISTING_IS_FILE = 4; // cannot overwrite file with directory
|
||||
ENTRY_ALREADY_EXISTS = 5; // O_EXCL and entry already exists
|
||||
PRECONDITION_FAILED = 6; // WriteCondition not satisfied
|
||||
}
|
||||
|
||||
// ObjectMutation is one entry-level change applied by ObjectTransaction. All
|
||||
// mutations of a transaction run under a single per-path lock (the request's
|
||||
// lock_key) and in order, so the gateway can describe a multi-entry object
|
||||
// operation as one request instead of holding a distributed lock across
|
||||
// several RPCs. Data-bearing writes (entries with chunks) should be written
|
||||
// before the transaction; mutations here are metadata-scoped.
|
||||
message ObjectMutation {
|
||||
enum Type {
|
||||
PUT = 0; // create or replace the entry (entry field)
|
||||
DELETE = 1; // delete the entry at directory/name (no error if absent)
|
||||
PATCH_EXTENDED = 2; // merge set_extended / remove delete_extended on the entry
|
||||
RECOMPUTE_LATEST = 3; // scan a directory and re-point a parent entry (recompute)
|
||||
}
|
||||
Type type = 1;
|
||||
string directory = 2;
|
||||
string name = 3; // entry name for DELETE / PATCH_EXTENDED / RECOMPUTE_LATEST (the pointer entry)
|
||||
Entry entry = 4; // full entry for PUT
|
||||
map<string, bytes> set_extended = 5; // PATCH_EXTENDED: keys to set
|
||||
repeated string delete_extended = 6; // PATCH_EXTENDED: keys to remove
|
||||
bool is_delete_data = 7; // DELETE: also delete chunk data
|
||||
bool is_recursive = 8; // DELETE: recurse into a directory
|
||||
Recompute recompute = 9; // RECOMPUTE_LATEST parameters
|
||||
bool set_content = 10; // PATCH_EXTENDED: replace Entry.content with content
|
||||
bytes content = 11; // PATCH_EXTENDED: new Entry.content when set_content
|
||||
bool touch_mtime = 12; // PATCH_EXTENDED: set the entry's Mtime to now (e.g. a metadata-replace copy)
|
||||
}
|
||||
|
||||
// Recompute re-derives a pointer entry (directory/name on the mutation) from the
|
||||
// current contents of a scanned directory, atomically under the transaction's
|
||||
// lock. It is mechanical: the filer picks the child that sorts first or last by
|
||||
// name and copies the requested fields into the pointer; it has no knowledge of
|
||||
// what the entries mean. The caller (which does know the versioning scheme)
|
||||
// supplies the sort direction and the key mappings. This covers re-pointing the
|
||||
// latest version after a specific version is deleted, where the scan must run
|
||||
// under the lock.
|
||||
message Recompute {
|
||||
string scan_dir = 1; // directory whose direct children are scanned
|
||||
bool descending = 2; // pick the child that sorts last by name (else first)
|
||||
map<string, string> copy_extended = 3; // pointer extended key -> source extended key on the chosen child
|
||||
string name_to_key = 4; // if set, store the chosen child's name under this pointer key
|
||||
string size_to_key = 5; // if set, store the chosen child's FileSize (decimal) under this pointer key
|
||||
string mtime_to_key = 6; // if set, store the chosen child's Mtime (decimal) under this pointer key
|
||||
string demote_key = 7; // if set, stamp demote_value on the prior name_to_key target when it changes
|
||||
bytes demote_value = 8; // value for demote_key
|
||||
string exclude_name = 9; // if set, skip this child when scanning (e.g. a version about to be deleted)
|
||||
}
|
||||
|
||||
// ObjectTransactionRequest applies an ordered list of mutations atomically with
|
||||
// respect to other writers of the same object, by holding the filer's per-path
|
||||
// lock on lock_key for the whole transaction. The optional condition is checked
|
||||
// first, against condition_key when set, else lock_key. Callers set route_key to
|
||||
// the object's stable owner ring key; a filer that is not the owner forwards the
|
||||
// transaction one hop to the owner, so a stale ring view is tolerated.
|
||||
message ObjectTransactionRequest {
|
||||
string lock_key = 1; // object path to lock and to evaluate the condition against
|
||||
WriteCondition condition = 2; // optional precondition, checked under the lock
|
||||
repeated ObjectMutation mutations = 3;
|
||||
bool is_from_other_cluster = 4;
|
||||
repeated int32 signatures = 5;
|
||||
string condition_key = 6; // if set, evaluate the condition against this entry instead of lock_key (still locking lock_key)
|
||||
string route_key = 7; // ring key identifying the owner filer; a non-owner forwards the whole transaction to it
|
||||
bool is_moved = 8; // set on a forwarded transaction so the receiver applies it locally instead of forwarding again
|
||||
}
|
||||
|
||||
message ObjectTransactionResponse {
|
||||
string error = 1;
|
||||
FilerError error_code = 2;
|
||||
}
|
||||
|
||||
// PosixLockRange is one advisory byte-range lock. Owner identity is (sid, owner):
|
||||
// sid is the mount session, owner the FUSE lock owner within it, so owners from
|
||||
// different mounts never alias. end is inclusive (max uint64 = to EOF); is_flock
|
||||
// separates the flock and fcntl namespaces, which never conflict.
|
||||
message PosixLockRange {
|
||||
uint64 start = 1;
|
||||
uint64 end = 2;
|
||||
uint32 type = 3; // 1=read, 2=write, 3=unlock
|
||||
uint64 sid = 4;
|
||||
uint64 owner = 5;
|
||||
uint32 pid = 6; // holder pid, for get_lk reporting only
|
||||
bool is_flock = 7;
|
||||
}
|
||||
|
||||
// PosixLock routes an advisory lock operation to the inode's owner filer, which
|
||||
// holds the authoritative in-memory lock table. key is the inode identity ring
|
||||
// key (the file path, or hl:<HardLinkId> for a hardlink) used both to resolve the
|
||||
// owner and to index the table. A non-owner filer forwards the request one hop;
|
||||
// is_moved bounds it so a stale ring view cannot loop.
|
||||
message PosixLockRequest {
|
||||
string key = 1;
|
||||
bool is_moved = 2;
|
||||
PosixLockOp op = 3;
|
||||
PosixLockRange lock = 4;
|
||||
// locks carries the full set a mount holds on key for a KEEP_ALIVE
|
||||
// re-assertion, so the current owner filer can rebuild its in-memory state
|
||||
// after an ownership change or restart. lock.sid identifies the session.
|
||||
repeated PosixLockRange locks = 5;
|
||||
// cooling_probe marks a dual-read a new owner sends to the previous owner
|
||||
// during a ring change, so the previous owner answers from local state
|
||||
// without itself cooling-off (no recursion).
|
||||
bool cooling_probe = 6;
|
||||
}
|
||||
|
||||
enum PosixLockOp {
|
||||
TRY_LOCK = 0; // grant lock or report conflict (non-blocking)
|
||||
UNLOCK = 1; // release lock's owner's locks over its range
|
||||
GET_LK = 2; // report a conflicting lock, if any
|
||||
RELEASE_POSIX_OWNER = 3; // drop the owner's fcntl locks (flush-time)
|
||||
RELEASE_FLOCK_OWNER = 4; // drop the owner's flock locks (release-time)
|
||||
KEEP_ALIVE = 5; // renew the session's lease on this owner (lock.sid)
|
||||
}
|
||||
|
||||
message PosixLockResponse {
|
||||
bool granted = 1; // for TRY_LOCK: whether the lock was granted
|
||||
bool has_conflict = 2; // whether conflict is populated
|
||||
PosixLockRange conflict = 3; // the blocking lock (TRY_LOCK conflict / GET_LK result)
|
||||
}
|
||||
|
||||
// ObjectTransactionBatch applies several object transactions in one round trip,
|
||||
// each under its own per-path lock and independent of the others (no cross-key
|
||||
// atomicity). A caller groups keys that route to the same owner filer and sends
|
||||
// one batch per owner, e.g. for a multi-object delete. Each response is parallel
|
||||
// to its request.
|
||||
message ObjectTransactionBatchRequest {
|
||||
repeated ObjectTransactionRequest transactions = 1;
|
||||
}
|
||||
|
||||
message ObjectTransactionBatchResponse {
|
||||
repeated ObjectTransactionResponse responses = 1;
|
||||
}
|
||||
|
||||
message CreateEntryResponse {
|
||||
|
||||
+407
-1676
File diff suppressed because it is too large
Load Diff
@@ -1,7 +1,7 @@
|
||||
// Code generated by protoc-gen-go-grpc. DO NOT EDIT.
|
||||
// versions:
|
||||
// - protoc-gen-go-grpc v1.6.2
|
||||
// - protoc v6.33.4
|
||||
// - protoc v7.34.1
|
||||
// source: filer.proto
|
||||
|
||||
package filer_pb
|
||||
@@ -26,9 +26,6 @@ const (
|
||||
SeaweedFiler_TouchAccessTime_FullMethodName = "/filer_pb.SeaweedFiler/TouchAccessTime"
|
||||
SeaweedFiler_AppendToEntry_FullMethodName = "/filer_pb.SeaweedFiler/AppendToEntry"
|
||||
SeaweedFiler_DeleteEntry_FullMethodName = "/filer_pb.SeaweedFiler/DeleteEntry"
|
||||
SeaweedFiler_ObjectTransaction_FullMethodName = "/filer_pb.SeaweedFiler/ObjectTransaction"
|
||||
SeaweedFiler_ObjectTransactionBatch_FullMethodName = "/filer_pb.SeaweedFiler/ObjectTransactionBatch"
|
||||
SeaweedFiler_PosixLock_FullMethodName = "/filer_pb.SeaweedFiler/PosixLock"
|
||||
SeaweedFiler_AtomicRenameEntry_FullMethodName = "/filer_pb.SeaweedFiler/AtomicRenameEntry"
|
||||
SeaweedFiler_StreamRenameEntry_FullMethodName = "/filer_pb.SeaweedFiler/StreamRenameEntry"
|
||||
SeaweedFiler_StreamMutateEntry_FullMethodName = "/filer_pb.SeaweedFiler/StreamMutateEntry"
|
||||
@@ -65,9 +62,6 @@ type SeaweedFilerClient interface {
|
||||
TouchAccessTime(ctx context.Context, in *TouchAccessTimeRequest, opts ...grpc.CallOption) (*TouchAccessTimeResponse, error)
|
||||
AppendToEntry(ctx context.Context, in *AppendToEntryRequest, opts ...grpc.CallOption) (*AppendToEntryResponse, error)
|
||||
DeleteEntry(ctx context.Context, in *DeleteEntryRequest, opts ...grpc.CallOption) (*DeleteEntryResponse, error)
|
||||
ObjectTransaction(ctx context.Context, in *ObjectTransactionRequest, opts ...grpc.CallOption) (*ObjectTransactionResponse, error)
|
||||
ObjectTransactionBatch(ctx context.Context, in *ObjectTransactionBatchRequest, opts ...grpc.CallOption) (*ObjectTransactionBatchResponse, error)
|
||||
PosixLock(ctx context.Context, in *PosixLockRequest, opts ...grpc.CallOption) (*PosixLockResponse, error)
|
||||
AtomicRenameEntry(ctx context.Context, in *AtomicRenameEntryRequest, opts ...grpc.CallOption) (*AtomicRenameEntryResponse, error)
|
||||
StreamRenameEntry(ctx context.Context, in *StreamRenameEntryRequest, opts ...grpc.CallOption) (grpc.ServerStreamingClient[StreamRenameEntryResponse], error)
|
||||
StreamMutateEntry(ctx context.Context, opts ...grpc.CallOption) (grpc.BidiStreamingClient[StreamMutateEntryRequest, StreamMutateEntryResponse], error)
|
||||
@@ -183,36 +177,6 @@ func (c *seaweedFilerClient) DeleteEntry(ctx context.Context, in *DeleteEntryReq
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func (c *seaweedFilerClient) ObjectTransaction(ctx context.Context, in *ObjectTransactionRequest, opts ...grpc.CallOption) (*ObjectTransactionResponse, error) {
|
||||
cOpts := append([]grpc.CallOption{grpc.StaticMethod()}, opts...)
|
||||
out := new(ObjectTransactionResponse)
|
||||
err := c.cc.Invoke(ctx, SeaweedFiler_ObjectTransaction_FullMethodName, in, out, cOpts...)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func (c *seaweedFilerClient) ObjectTransactionBatch(ctx context.Context, in *ObjectTransactionBatchRequest, opts ...grpc.CallOption) (*ObjectTransactionBatchResponse, error) {
|
||||
cOpts := append([]grpc.CallOption{grpc.StaticMethod()}, opts...)
|
||||
out := new(ObjectTransactionBatchResponse)
|
||||
err := c.cc.Invoke(ctx, SeaweedFiler_ObjectTransactionBatch_FullMethodName, in, out, cOpts...)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func (c *seaweedFilerClient) PosixLock(ctx context.Context, in *PosixLockRequest, opts ...grpc.CallOption) (*PosixLockResponse, error) {
|
||||
cOpts := append([]grpc.CallOption{grpc.StaticMethod()}, opts...)
|
||||
out := new(PosixLockResponse)
|
||||
err := c.cc.Invoke(ctx, SeaweedFiler_PosixLock_FullMethodName, in, out, cOpts...)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func (c *seaweedFilerClient) AtomicRenameEntry(ctx context.Context, in *AtomicRenameEntryRequest, opts ...grpc.CallOption) (*AtomicRenameEntryResponse, error) {
|
||||
cOpts := append([]grpc.CallOption{grpc.StaticMethod()}, opts...)
|
||||
out := new(AtomicRenameEntryResponse)
|
||||
@@ -493,9 +457,6 @@ type SeaweedFilerServer interface {
|
||||
TouchAccessTime(context.Context, *TouchAccessTimeRequest) (*TouchAccessTimeResponse, error)
|
||||
AppendToEntry(context.Context, *AppendToEntryRequest) (*AppendToEntryResponse, error)
|
||||
DeleteEntry(context.Context, *DeleteEntryRequest) (*DeleteEntryResponse, error)
|
||||
ObjectTransaction(context.Context, *ObjectTransactionRequest) (*ObjectTransactionResponse, error)
|
||||
ObjectTransactionBatch(context.Context, *ObjectTransactionBatchRequest) (*ObjectTransactionBatchResponse, error)
|
||||
PosixLock(context.Context, *PosixLockRequest) (*PosixLockResponse, error)
|
||||
AtomicRenameEntry(context.Context, *AtomicRenameEntryRequest) (*AtomicRenameEntryResponse, error)
|
||||
StreamRenameEntry(*StreamRenameEntryRequest, grpc.ServerStreamingServer[StreamRenameEntryResponse]) error
|
||||
StreamMutateEntry(grpc.BidiStreamingServer[StreamMutateEntryRequest, StreamMutateEntryResponse]) error
|
||||
@@ -553,15 +514,6 @@ func (UnimplementedSeaweedFilerServer) AppendToEntry(context.Context, *AppendToE
|
||||
func (UnimplementedSeaweedFilerServer) DeleteEntry(context.Context, *DeleteEntryRequest) (*DeleteEntryResponse, error) {
|
||||
return nil, status.Error(codes.Unimplemented, "method DeleteEntry not implemented")
|
||||
}
|
||||
func (UnimplementedSeaweedFilerServer) ObjectTransaction(context.Context, *ObjectTransactionRequest) (*ObjectTransactionResponse, error) {
|
||||
return nil, status.Error(codes.Unimplemented, "method ObjectTransaction not implemented")
|
||||
}
|
||||
func (UnimplementedSeaweedFilerServer) ObjectTransactionBatch(context.Context, *ObjectTransactionBatchRequest) (*ObjectTransactionBatchResponse, error) {
|
||||
return nil, status.Error(codes.Unimplemented, "method ObjectTransactionBatch not implemented")
|
||||
}
|
||||
func (UnimplementedSeaweedFilerServer) PosixLock(context.Context, *PosixLockRequest) (*PosixLockResponse, error) {
|
||||
return nil, status.Error(codes.Unimplemented, "method PosixLock not implemented")
|
||||
}
|
||||
func (UnimplementedSeaweedFilerServer) AtomicRenameEntry(context.Context, *AtomicRenameEntryRequest) (*AtomicRenameEntryResponse, error) {
|
||||
return nil, status.Error(codes.Unimplemented, "method AtomicRenameEntry not implemented")
|
||||
}
|
||||
@@ -771,60 +723,6 @@ func _SeaweedFiler_DeleteEntry_Handler(srv interface{}, ctx context.Context, dec
|
||||
return interceptor(ctx, in, info, handler)
|
||||
}
|
||||
|
||||
func _SeaweedFiler_ObjectTransaction_Handler(srv interface{}, ctx context.Context, dec func(interface{}) error, interceptor grpc.UnaryServerInterceptor) (interface{}, error) {
|
||||
in := new(ObjectTransactionRequest)
|
||||
if err := dec(in); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if interceptor == nil {
|
||||
return srv.(SeaweedFilerServer).ObjectTransaction(ctx, in)
|
||||
}
|
||||
info := &grpc.UnaryServerInfo{
|
||||
Server: srv,
|
||||
FullMethod: SeaweedFiler_ObjectTransaction_FullMethodName,
|
||||
}
|
||||
handler := func(ctx context.Context, req interface{}) (interface{}, error) {
|
||||
return srv.(SeaweedFilerServer).ObjectTransaction(ctx, req.(*ObjectTransactionRequest))
|
||||
}
|
||||
return interceptor(ctx, in, info, handler)
|
||||
}
|
||||
|
||||
func _SeaweedFiler_ObjectTransactionBatch_Handler(srv interface{}, ctx context.Context, dec func(interface{}) error, interceptor grpc.UnaryServerInterceptor) (interface{}, error) {
|
||||
in := new(ObjectTransactionBatchRequest)
|
||||
if err := dec(in); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if interceptor == nil {
|
||||
return srv.(SeaweedFilerServer).ObjectTransactionBatch(ctx, in)
|
||||
}
|
||||
info := &grpc.UnaryServerInfo{
|
||||
Server: srv,
|
||||
FullMethod: SeaweedFiler_ObjectTransactionBatch_FullMethodName,
|
||||
}
|
||||
handler := func(ctx context.Context, req interface{}) (interface{}, error) {
|
||||
return srv.(SeaweedFilerServer).ObjectTransactionBatch(ctx, req.(*ObjectTransactionBatchRequest))
|
||||
}
|
||||
return interceptor(ctx, in, info, handler)
|
||||
}
|
||||
|
||||
func _SeaweedFiler_PosixLock_Handler(srv interface{}, ctx context.Context, dec func(interface{}) error, interceptor grpc.UnaryServerInterceptor) (interface{}, error) {
|
||||
in := new(PosixLockRequest)
|
||||
if err := dec(in); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if interceptor == nil {
|
||||
return srv.(SeaweedFilerServer).PosixLock(ctx, in)
|
||||
}
|
||||
info := &grpc.UnaryServerInfo{
|
||||
Server: srv,
|
||||
FullMethod: SeaweedFiler_PosixLock_FullMethodName,
|
||||
}
|
||||
handler := func(ctx context.Context, req interface{}) (interface{}, error) {
|
||||
return srv.(SeaweedFilerServer).PosixLock(ctx, req.(*PosixLockRequest))
|
||||
}
|
||||
return interceptor(ctx, in, info, handler)
|
||||
}
|
||||
|
||||
func _SeaweedFiler_AtomicRenameEntry_Handler(srv interface{}, ctx context.Context, dec func(interface{}) error, interceptor grpc.UnaryServerInterceptor) (interface{}, error) {
|
||||
in := new(AtomicRenameEntryRequest)
|
||||
if err := dec(in); err != nil {
|
||||
@@ -1231,18 +1129,6 @@ var SeaweedFiler_ServiceDesc = grpc.ServiceDesc{
|
||||
MethodName: "DeleteEntry",
|
||||
Handler: _SeaweedFiler_DeleteEntry_Handler,
|
||||
},
|
||||
{
|
||||
MethodName: "ObjectTransaction",
|
||||
Handler: _SeaweedFiler_ObjectTransaction_Handler,
|
||||
},
|
||||
{
|
||||
MethodName: "ObjectTransactionBatch",
|
||||
Handler: _SeaweedFiler_ObjectTransactionBatch_Handler,
|
||||
},
|
||||
{
|
||||
MethodName: "PosixLock",
|
||||
Handler: _SeaweedFiler_PosixLock_Handler,
|
||||
},
|
||||
{
|
||||
MethodName: "AtomicRenameEntry",
|
||||
Handler: _SeaweedFiler_AtomicRenameEntry_Handler,
|
||||
|
||||
@@ -374,7 +374,6 @@ message ErasureCodingTaskConfig {
|
||||
int32 min_volume_size_mb = 3; // Minimum volume size for EC
|
||||
string collection_filter = 4; // Only process volumes from specific collections
|
||||
repeated string preferred_tags = 5; // Disk tags to prioritize for EC shard placement
|
||||
string replica_placement = 6; // EC shard replica placement (e.g. "020"); empty falls back to master default replication
|
||||
}
|
||||
|
||||
// BalanceTaskConfig contains balance-specific configuration
|
||||
|
||||
@@ -2960,7 +2960,6 @@ type ErasureCodingTaskConfig struct {
|
||||
MinVolumeSizeMb int32 `protobuf:"varint,3,opt,name=min_volume_size_mb,json=minVolumeSizeMb,proto3" json:"min_volume_size_mb,omitempty"` // Minimum volume size for EC
|
||||
CollectionFilter string `protobuf:"bytes,4,opt,name=collection_filter,json=collectionFilter,proto3" json:"collection_filter,omitempty"` // Only process volumes from specific collections
|
||||
PreferredTags []string `protobuf:"bytes,5,rep,name=preferred_tags,json=preferredTags,proto3" json:"preferred_tags,omitempty"` // Disk tags to prioritize for EC shard placement
|
||||
ReplicaPlacement string `protobuf:"bytes,6,opt,name=replica_placement,json=replicaPlacement,proto3" json:"replica_placement,omitempty"` // EC shard replica placement (e.g. "020"); empty falls back to master default replication
|
||||
unknownFields protoimpl.UnknownFields
|
||||
sizeCache protoimpl.SizeCache
|
||||
}
|
||||
@@ -3030,13 +3029,6 @@ func (x *ErasureCodingTaskConfig) GetPreferredTags() []string {
|
||||
return nil
|
||||
}
|
||||
|
||||
func (x *ErasureCodingTaskConfig) GetReplicaPlacement() string {
|
||||
if x != nil {
|
||||
return x.ReplicaPlacement
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// BalanceTaskConfig contains balance-specific configuration
|
||||
type BalanceTaskConfig struct {
|
||||
state protoimpl.MessageState `protogen:"open.v1"`
|
||||
@@ -4226,14 +4218,13 @@ const file_worker_proto_rawDesc = "" +
|
||||
"\x10VacuumTaskConfig\x12+\n" +
|
||||
"\x11garbage_threshold\x18\x01 \x01(\x01R\x10garbageThreshold\x12/\n" +
|
||||
"\x14min_volume_age_hours\x18\x02 \x01(\x05R\x11minVolumeAgeHours\x120\n" +
|
||||
"\x14min_interval_seconds\x18\x03 \x01(\x05R\x12minIntervalSeconds\"\x9a\x02\n" +
|
||||
"\x14min_interval_seconds\x18\x03 \x01(\x05R\x12minIntervalSeconds\"\xed\x01\n" +
|
||||
"\x17ErasureCodingTaskConfig\x12%\n" +
|
||||
"\x0efullness_ratio\x18\x01 \x01(\x01R\rfullnessRatio\x12*\n" +
|
||||
"\x11quiet_for_seconds\x18\x02 \x01(\x05R\x0fquietForSeconds\x12+\n" +
|
||||
"\x12min_volume_size_mb\x18\x03 \x01(\x05R\x0fminVolumeSizeMb\x12+\n" +
|
||||
"\x11collection_filter\x18\x04 \x01(\tR\x10collectionFilter\x12%\n" +
|
||||
"\x0epreferred_tags\x18\x05 \x03(\tR\rpreferredTags\x12+\n" +
|
||||
"\x11replica_placement\x18\x06 \x01(\tR\x10replicaPlacement\"n\n" +
|
||||
"\x0epreferred_tags\x18\x05 \x03(\tR\rpreferredTags\"n\n" +
|
||||
"\x11BalanceTaskConfig\x12/\n" +
|
||||
"\x13imbalance_threshold\x18\x01 \x01(\x01R\x12imbalanceThreshold\x12(\n" +
|
||||
"\x10min_server_count\x18\x02 \x01(\x05R\x0eminServerCount\"I\n" +
|
||||
|
||||
@@ -299,9 +299,6 @@ func (s *s3RemoteStorageClient) WriteFile(loc *remote_pb.RemoteStorageLocation,
|
||||
Body: reader,
|
||||
Tagging: awsTags,
|
||||
}
|
||||
if entry.Attributes != nil && entry.Attributes.Mime != "" {
|
||||
uploadInput.ContentType = aws.String(entry.Attributes.Mime)
|
||||
}
|
||||
if s.conf.S3StorageClass != "" {
|
||||
uploadInput.StorageClass = aws.String(s.conf.S3StorageClass)
|
||||
}
|
||||
|
||||
@@ -1,15 +1,10 @@
|
||||
package s3
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"io"
|
||||
"net/http"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/aws/aws-sdk-go/aws/credentials"
|
||||
awss3 "github.com/aws/aws-sdk-go/service/s3"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/remote_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/remote_storage"
|
||||
"github.com/stretchr/testify/require"
|
||||
@@ -70,89 +65,3 @@ func TestS3ErrRemoteObjectNotFoundIsAccessible(t *testing.T) {
|
||||
require.Error(t, remote_storage.ErrRemoteObjectNotFound)
|
||||
require.Equal(t, "remote object not found", remote_storage.ErrRemoteObjectNotFound.Error())
|
||||
}
|
||||
|
||||
// captureRoundTripper records the PUT request that the s3manager uploader
|
||||
// sends, and short-circuits all calls with a 200 so the SDK is satisfied.
|
||||
type captureRoundTripper struct {
|
||||
uploadReq *http.Request
|
||||
}
|
||||
|
||||
func (c *captureRoundTripper) RoundTrip(req *http.Request) (*http.Response, error) {
|
||||
if req.Method == http.MethodPut {
|
||||
c.uploadReq = req.Clone(req.Context())
|
||||
}
|
||||
if req.Body != nil {
|
||||
_, _ = io.Copy(io.Discard, req.Body)
|
||||
_ = req.Body.Close()
|
||||
}
|
||||
return &http.Response{
|
||||
StatusCode: http.StatusOK,
|
||||
Body: io.NopCloser(strings.NewReader("")),
|
||||
Header: http.Header{
|
||||
"ETag": []string{"\"etag\""},
|
||||
},
|
||||
Request: req,
|
||||
}, nil
|
||||
}
|
||||
|
||||
func (c *captureRoundTripper) uploadContentType() string {
|
||||
if c.uploadReq == nil {
|
||||
return ""
|
||||
}
|
||||
return c.uploadReq.Header.Get("Content-Type")
|
||||
}
|
||||
|
||||
func newCapturingS3Client(t *testing.T) (*s3RemoteStorageClient, *captureRoundTripper) {
|
||||
t.Helper()
|
||||
rt := &captureRoundTripper{}
|
||||
conf := &remote_pb.RemoteConf{
|
||||
Name: "test",
|
||||
S3Region: "us-east-1",
|
||||
S3Endpoint: "https://example.invalid",
|
||||
S3ForcePathStyle: true,
|
||||
S3AccessKey: "test-key",
|
||||
S3SecretKey: "test-secret",
|
||||
}
|
||||
httpClient := &http.Client{Transport: rt}
|
||||
rs, err := MakeWithHTTPClient(conf, httpClient)
|
||||
require.NoError(t, err)
|
||||
return rs.(*s3RemoteStorageClient), rt
|
||||
}
|
||||
|
||||
func TestS3WriteFilePassesMimeAsContentType(t *testing.T) {
|
||||
client, rt := newCapturingS3Client(t)
|
||||
loc := &remote_pb.RemoteStorageLocation{
|
||||
Name: "test",
|
||||
Bucket: "bucket",
|
||||
Path: "/dir/test.html",
|
||||
}
|
||||
entry := &filer_pb.Entry{
|
||||
Attributes: &filer_pb.FuseAttributes{Mime: "text/html"},
|
||||
}
|
||||
|
||||
_, err := client.WriteFile(loc, entry, bytes.NewReader([]byte("<html></html>")))
|
||||
require.NoError(t, err)
|
||||
|
||||
require.NotNil(t, rt.uploadReq, "uploader should have issued a PUT")
|
||||
require.Equal(t, "text/html", rt.uploadContentType(), "Content-Type should match entry.Attributes.Mime")
|
||||
}
|
||||
|
||||
func TestS3WriteFileOmitsContentTypeWhenMimeMissing(t *testing.T) {
|
||||
client, rt := newCapturingS3Client(t)
|
||||
loc := &remote_pb.RemoteStorageLocation{
|
||||
Name: "test",
|
||||
Bucket: "bucket",
|
||||
Path: "/dir/test.bin",
|
||||
}
|
||||
entry := &filer_pb.Entry{
|
||||
Attributes: &filer_pb.FuseAttributes{},
|
||||
}
|
||||
|
||||
_, err := client.WriteFile(loc, entry, bytes.NewReader([]byte("data")))
|
||||
require.NoError(t, err)
|
||||
|
||||
require.NotNil(t, rt.uploadReq, "uploader should have issued a PUT")
|
||||
// When entry.Attributes.Mime is empty we don't force a Content-Type so the
|
||||
// remote can apply its own default rather than getting a misleading one.
|
||||
require.Equal(t, "", rt.uploadContentType())
|
||||
}
|
||||
|
||||
@@ -120,7 +120,7 @@ func (fs *FilerSink) replicateOneChunk(sourceChunk *filer_pb.FileChunk, path str
|
||||
|
||||
fileId, err := fs.fetchAndWrite(sourceChunk, path, sourceMtime)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("copy %s: %w", sourceChunk.GetFileIdString(), err)
|
||||
return nil, fmt.Errorf("copy %s: %v", sourceChunk.GetFileIdString(), err)
|
||||
}
|
||||
|
||||
return &filer_pb.FileChunk{
|
||||
@@ -292,10 +292,6 @@ func (fs *FilerSink) fetchAndWrite(sourceChunk *filer_pb.FileChunk, path string,
|
||||
fullData = data
|
||||
}
|
||||
|
||||
if err := validateReplicatedReadSize(sourceChunk, len(fullData)); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
transferStatus.mu.Lock()
|
||||
transferStatus.BytesReceived = int64(len(fullData))
|
||||
transferStatus.Status = "uploading"
|
||||
@@ -339,14 +335,6 @@ func (fs *FilerSink) fetchAndWrite(sourceChunk *filer_pb.FileChunk, path string,
|
||||
fileId = currentFileId
|
||||
return nil
|
||||
}, func(retryErr error) (shouldContinue bool) {
|
||||
if errors.Is(retryErr, errChunkSizeMismatch) {
|
||||
glog.V(0).Infof("permanent size mismatch replicating %s for %s: %v",
|
||||
sourceChunk.GetFileIdString(), path, retryErr)
|
||||
transferStatus.mu.Lock()
|
||||
transferStatus.LastErr = retryErr.Error()
|
||||
transferStatus.mu.Unlock()
|
||||
return false
|
||||
}
|
||||
if fs.hasSourceNewerVersion(path, sourceMtime) {
|
||||
glog.V(1).Infof("skip retrying stale source %s for %s: %v", sourceChunk.GetFileIdString(), path, retryErr)
|
||||
return false
|
||||
@@ -400,18 +388,6 @@ func isEofError(err error) bool {
|
||||
return errors.Is(err, io.ErrUnexpectedEOF) || errors.Is(err, io.EOF)
|
||||
}
|
||||
|
||||
// errChunkSizeMismatch is a permanent (non-retriable) replication failure.
|
||||
var errChunkSizeMismatch = errors.New("chunk size mismatch")
|
||||
|
||||
func validateReplicatedReadSize(sourceChunk *filer_pb.FileChunk, readSize int) error {
|
||||
if uint64(readSize) != sourceChunk.Size {
|
||||
return fmt.Errorf("%w: read %s got %d bytes, source metadata says %d",
|
||||
errChunkSizeMismatch, sourceChunk.GetFileIdString(),
|
||||
readSize, sourceChunk.Size)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (fs *FilerSink) buildUploadUrl(host, fileId string) string {
|
||||
if fs.writeChunkByFiler {
|
||||
return fmt.Sprintf("http://%s/?proxyChunkId=%s", fs.address, fileId)
|
||||
|
||||
@@ -1,27 +1,12 @@
|
||||
package filersink
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"os"
|
||||
"strings"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/operation"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/replication/source"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
util_http "github.com/seaweedfs/seaweedfs/weed/util/http"
|
||||
)
|
||||
|
||||
func TestMain(m *testing.M) {
|
||||
util_http.InitGlobalHttpClient()
|
||||
os.Exit(m.Run())
|
||||
}
|
||||
|
||||
func TestTargetPathToSourcePath(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
@@ -92,191 +77,3 @@ func TestTargetPathToSourcePath(t *testing.T) {
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// FilerSink must reject chunks whose received byte count disagrees with the
|
||||
// source filer metadata, instead of silently writing 0-byte needles with the
|
||||
// source size in the destination metadata.
|
||||
func TestValidateReplicatedChunkSize(t *testing.T) {
|
||||
const fid = "74,047d16a94aa581"
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
expectedSize uint64
|
||||
readSize int
|
||||
wantErr bool
|
||||
}{
|
||||
{
|
||||
name: "healthy",
|
||||
expectedSize: 5171,
|
||||
readSize: 5171,
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "legitimately empty file",
|
||||
expectedSize: 0,
|
||||
readSize: 0,
|
||||
wantErr: false,
|
||||
},
|
||||
{
|
||||
name: "zero-byte read for non-empty source",
|
||||
expectedSize: 5171,
|
||||
readSize: 0,
|
||||
wantErr: true,
|
||||
},
|
||||
{
|
||||
name: "short read",
|
||||
expectedSize: 5171,
|
||||
readSize: 100,
|
||||
wantErr: true,
|
||||
},
|
||||
{
|
||||
name: "over-read (server returned more than metadata)",
|
||||
expectedSize: 5171,
|
||||
readSize: 8192,
|
||||
wantErr: true,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
chunk := &filer_pb.FileChunk{FileId: fid, Size: tc.expectedSize}
|
||||
|
||||
gotErr := validateReplicatedReadSize(chunk, tc.readSize)
|
||||
|
||||
if tc.wantErr {
|
||||
if gotErr == nil {
|
||||
t.Fatalf("expected error, got nil (read=%d expected=%d)",
|
||||
tc.readSize, tc.expectedSize)
|
||||
}
|
||||
if !errors.Is(gotErr, errChunkSizeMismatch) {
|
||||
t.Fatalf("expected errChunkSizeMismatch, got %v", gotErr)
|
||||
}
|
||||
if !strings.Contains(gotErr.Error(), fid) {
|
||||
t.Fatalf("error %q does not mention chunk id %q", gotErr, fid)
|
||||
}
|
||||
return
|
||||
}
|
||||
if gotErr != nil {
|
||||
t.Fatalf("unexpected read-size error: %v", gotErr)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// End-to-end regression :
|
||||
// a source volume that responds 200 OK with Content-Length: 0
|
||||
// for a chunk that filer metadata claims is 5171 bytes must be rejected
|
||||
// by fetchAndWrite with a (non-retriable) size mismatch error,
|
||||
// instead of being silently propagated to the destination as a 0-byte needle.
|
||||
func TestFetchAndWriteRejectsZeroByteSource(t *testing.T) {
|
||||
const fid = "74,047d16a94aa581"
|
||||
const expectedSize uint64 = 5171
|
||||
|
||||
// Shorten retry backoff so a fail-fast test that briefly enters the retry
|
||||
// loop doesn't pay the production 1s+ wait. Scoped to this test so any
|
||||
// future test in the package keeps the production constant.
|
||||
prevRetryWaitTime := util.RetryWaitTime
|
||||
util.RetryWaitTime = 100 * time.Millisecond
|
||||
t.Cleanup(func() { util.RetryWaitTime = prevRetryWaitTime })
|
||||
|
||||
var hits atomic.Int32
|
||||
sourceServer := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
hits.Add(1)
|
||||
w.Header().Set("Content-Type", "application/octet-stream")
|
||||
w.WriteHeader(http.StatusOK)
|
||||
// Intentionally write no body — mimic the buggy volume response.
|
||||
}))
|
||||
defer sourceServer.Close()
|
||||
|
||||
serverAddr := strings.TrimPrefix(sourceServer.URL, "http://")
|
||||
|
||||
filerSrc := &source.FilerSource{}
|
||||
if err := filerSrc.DoInitialize(serverAddr, serverAddr, "/", true); err != nil {
|
||||
t.Fatalf("filerSource.DoInitialize: %v", err)
|
||||
}
|
||||
|
||||
fs := &FilerSink{
|
||||
filerSource: filerSrc,
|
||||
address: serverAddr,
|
||||
dir: "/dst",
|
||||
executor: util.NewLimitedConcurrentExecutor(1),
|
||||
}
|
||||
fs.SetUploader(operation.NewUploaderWithHttpClient(http.DefaultClient))
|
||||
|
||||
sourceChunk := &filer_pb.FileChunk{
|
||||
FileId: fid,
|
||||
Size: expectedSize,
|
||||
}
|
||||
|
||||
done := make(chan struct {
|
||||
fileId string
|
||||
err error
|
||||
}, 1)
|
||||
go func() {
|
||||
gotFileId, gotErr := fs.fetchAndWrite(sourceChunk, "/dst/index.bin", 0)
|
||||
done <- struct {
|
||||
fileId string
|
||||
err error
|
||||
}{gotFileId, gotErr}
|
||||
}()
|
||||
|
||||
select {
|
||||
case result := <-done:
|
||||
if result.err == nil {
|
||||
t.Fatalf("expected size mismatch error, got nil (fileId=%q)", result.fileId)
|
||||
}
|
||||
if !errors.Is(result.err, errChunkSizeMismatch) {
|
||||
t.Fatalf("expected errChunkSizeMismatch, got %v", result.err)
|
||||
}
|
||||
if !strings.Contains(result.err.Error(), "5171") {
|
||||
t.Fatalf("error %q does not mention expected size 5171", result.err)
|
||||
}
|
||||
if !strings.Contains(result.err.Error(), fid) {
|
||||
t.Fatalf("error %q does not mention chunk id %q", result.err, fid)
|
||||
}
|
||||
if h := hits.Load(); h != 1 {
|
||||
t.Fatalf("expected exactly 1 source hit (fail-fast), got %d", h)
|
||||
}
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatalf("fetchAndWrite did not return within 5s (retry loop not aborted on size mismatch); hits=%d", hits.Load())
|
||||
}
|
||||
}
|
||||
|
||||
// Lock in that the errChunkSizeMismatch sentinel survives the wrap in
|
||||
// replicateOneChunk + pass-through in util.Retry, so filer_sink.go's
|
||||
// errors.Is check actually fires.
|
||||
func TestReplicateChunksPreservesSizeMismatchSentinel(t *testing.T) {
|
||||
const fid = "74,047d16a94aa581"
|
||||
const expectedSize uint64 = 5171
|
||||
|
||||
sourceServer := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.Header().Set("Content-Type", "application/octet-stream")
|
||||
w.WriteHeader(http.StatusOK)
|
||||
}))
|
||||
defer sourceServer.Close()
|
||||
|
||||
serverAddr := strings.TrimPrefix(sourceServer.URL, "http://")
|
||||
|
||||
filerSrc := &source.FilerSource{}
|
||||
if err := filerSrc.DoInitialize(serverAddr, serverAddr, "/", true); err != nil {
|
||||
t.Fatalf("filerSource.DoInitialize: %v", err)
|
||||
}
|
||||
|
||||
fs := &FilerSink{
|
||||
filerSource: filerSrc,
|
||||
address: serverAddr,
|
||||
dir: "/dst",
|
||||
executor: util.NewLimitedConcurrentExecutor(1),
|
||||
}
|
||||
fs.SetUploader(operation.NewUploaderWithHttpClient(http.DefaultClient))
|
||||
|
||||
sourceChunks := []*filer_pb.FileChunk{{FileId: fid, Size: expectedSize}}
|
||||
|
||||
_, err := fs.replicateChunks(nil, sourceChunks, "/dst/index.bin", 0)
|
||||
if err == nil {
|
||||
t.Fatal("expected error from replicateChunks, got nil")
|
||||
}
|
||||
if !errors.Is(err, errChunkSizeMismatch) {
|
||||
t.Fatalf("error chain broken: errors.Is(err, errChunkSizeMismatch) = false; got %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2,7 +2,6 @@ package filersink
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"math"
|
||||
"sync"
|
||||
@@ -192,12 +191,7 @@ func (fs *FilerSink) CreateEntry(key string, entry *filer_pb.Entry, signatures [
|
||||
replicatedChunks, err := fs.replicateChunks(context.Background(), entry.GetChunks(), key, getEntryMtime(entry))
|
||||
|
||||
if err != nil {
|
||||
// Don't swallow size-mismatch: source bytes disagree with source
|
||||
// metadata, so committing would propagate corruption silently.
|
||||
if errors.Is(err, errChunkSizeMismatch) {
|
||||
glog.Errorf("refuse to replicate entry with corrupt chunk %s: %v", key, err)
|
||||
return err
|
||||
}
|
||||
// only warning here since the source chunk may have been deleted already
|
||||
glog.Warningf("replicate entry chunks %s: %v", key, err)
|
||||
return nil
|
||||
}
|
||||
@@ -265,8 +259,8 @@ func (fs *FilerSink) UpdateEntry(key string, oldEntry *filer_pb.Entry, newParent
|
||||
// this usually happens when the messages are not ordered
|
||||
glog.V(2).Infof("late updates %s", key)
|
||||
} else {
|
||||
// source-side chunks resolve via source filer; sink volume IDs may collide.
|
||||
deletedChunks, newChunks, err := compareChunks(context.Background(), filer.LookupFn(fs.filerSource), oldEntry, newEntry)
|
||||
// find out what changed
|
||||
deletedChunks, newChunks, err := compareChunks(context.Background(), filer.LookupFn(fs), oldEntry, newEntry)
|
||||
if err != nil {
|
||||
return true, fmt.Errorf("replicate %s compare chunks error: %v", key, err)
|
||||
}
|
||||
@@ -280,10 +274,6 @@ func (fs *FilerSink) UpdateEntry(key string, oldEntry *filer_pb.Entry, newParent
|
||||
// replicate the chunks that are new in the source
|
||||
replicatedChunks, err := fs.replicateChunks(context.Background(), newChunks, key, getEntryMtime(newEntry))
|
||||
if err != nil {
|
||||
if errors.Is(err, errChunkSizeMismatch) {
|
||||
glog.Errorf("refuse to replicate entry with corrupt chunk %s: %v", key, err)
|
||||
return true, err
|
||||
}
|
||||
glog.Warningf("replicate entry chunks %s: %v", key, err)
|
||||
return true, nil
|
||||
}
|
||||
|
||||
@@ -553,14 +553,9 @@ func (s3a *S3ApiServer) completeMultipartUpload(r *http.Request, input *s3.Compl
|
||||
uploadDirectory := s3a.genUploadsFolder(*input.Bucket) + "/" + *input.UploadId
|
||||
entryName, dirName := s3a.getEntryNameAndDir(input)
|
||||
var completionState *multipartCompletionState
|
||||
// Route the completion's writes to the object's owner filer when known, off
|
||||
// the distributed lock. Idempotent replay is handled gateway-side in
|
||||
// prepareMultipartCompletionState (it returns the existing result when the
|
||||
// object already carries this UploadId), so the lock is not needed to dedupe
|
||||
// retries. With no owner yet (no ring), keep the lock as the bootstrap path.
|
||||
owner := s3a.objectWriteOwner(*input.Bucket, *input.Key)
|
||||
routeKey := s3a.objectRouteKey(*input.Bucket, *input.Key)
|
||||
completionBody := func() s3err.ErrorCode {
|
||||
finalizeCode := s3a.withObjectWriteLock(*input.Bucket, *input.Key, func() s3err.ErrorCode {
|
||||
return s3a.checkConditionalHeaders(r, *input.Bucket, *input.Key)
|
||||
}, func() s3err.ErrorCode {
|
||||
var prepCode s3err.ErrorCode
|
||||
completionState, output, prepCode = s3a.prepareMultipartCompletionState(r, input, uploadDirectory, entryName, dirName, completedPartNumbers, completedPartMap, maxPartNo)
|
||||
if prepCode != s3err.ErrNone || output != nil {
|
||||
@@ -656,16 +651,7 @@ func (s3a *S3ApiServer) completeMultipartUpload(r *http.Request, input *s3.Compl
|
||||
|
||||
// Update the .versions directory metadata to indicate this is the latest version
|
||||
// Pass entry to cache its metadata for single-scan list efficiency
|
||||
// Route the pointer flip to the owner (off the lock) via
|
||||
// RECOMPUTE_LATEST; the just-written version file is the newest.
|
||||
if owner != "" {
|
||||
if code := s3a.routedVersionedFinalize(owner, *input.Bucket, *input.Key, useInvertedFormat); code != s3err.ErrNone {
|
||||
if rollbackErr := s3a.rollbackMultipartVersion(versionDir, versionFileName); rollbackErr != nil {
|
||||
glog.Errorf("completeMultipartUpload: failed to rollback version %s for %s/%s after routed finalize error: %v", versionId, *input.Bucket, *input.Key, rollbackErr)
|
||||
}
|
||||
return code
|
||||
}
|
||||
} else if err := s3a.updateLatestVersionInDirectory(*input.Bucket, *input.Key, versionId, versionFileName, versionEntryForCache); err != nil {
|
||||
if err := s3a.updateLatestVersionInDirectory(*input.Bucket, *input.Key, versionId, versionFileName, versionEntryForCache); err != nil {
|
||||
if rollbackErr := s3a.rollbackMultipartVersion(versionDir, versionFileName); rollbackErr != nil {
|
||||
glog.Errorf("completeMultipartUpload: failed to rollback version %s for %s/%s after latest pointer update error: %v", versionId, *input.Bucket, *input.Key, rollbackErr)
|
||||
}
|
||||
@@ -689,7 +675,7 @@ func (s3a *S3ApiServer) completeMultipartUpload(r *http.Request, input *s3.Compl
|
||||
|
||||
if versioningState == s3_constants.VersioningSuspended {
|
||||
// For suspended versioning, add "null" version ID metadata and return "null" version ID
|
||||
if err := s3a.writeMultipartObject(owner, routeKey, dirName, entryName, completionState.finalParts, func(entry *filer_pb.Entry) {
|
||||
if err := s3a.mkFile(dirName, entryName, completionState.finalParts, func(entry *filer_pb.Entry) {
|
||||
if entry.Extended == nil {
|
||||
entry.Extended = make(map[string][]byte)
|
||||
}
|
||||
@@ -753,7 +739,7 @@ func (s3a *S3ApiServer) completeMultipartUpload(r *http.Request, input *s3.Compl
|
||||
}
|
||||
|
||||
// For non-versioned buckets, create main object file
|
||||
if err := s3a.writeMultipartObject(owner, routeKey, dirName, entryName, completionState.finalParts, func(entry *filer_pb.Entry) {
|
||||
if err := s3a.mkFile(dirName, entryName, completionState.finalParts, func(entry *filer_pb.Entry) {
|
||||
if entry.Extended == nil {
|
||||
entry.Extended = make(map[string][]byte)
|
||||
}
|
||||
@@ -816,19 +802,7 @@ func (s3a *S3ApiServer) completeMultipartUpload(r *http.Request, input *s3.Compl
|
||||
ChecksumValue: completionState.checksumValue,
|
||||
}
|
||||
return s3err.ErrNone
|
||||
}
|
||||
var finalizeCode s3err.ErrorCode
|
||||
if owner != "" {
|
||||
if code := s3a.checkConditionalHeaders(r, *input.Bucket, *input.Key); code != s3err.ErrNone {
|
||||
finalizeCode = code
|
||||
} else {
|
||||
finalizeCode = completionBody()
|
||||
}
|
||||
} else {
|
||||
finalizeCode = s3a.withObjectWriteLock(*input.Bucket, *input.Key, func() s3err.ErrorCode {
|
||||
return s3a.checkConditionalHeaders(r, *input.Bucket, *input.Key)
|
||||
}, completionBody)
|
||||
}
|
||||
})
|
||||
if finalizeCode != s3err.ErrNone {
|
||||
return nil, finalizeCode
|
||||
}
|
||||
|
||||
@@ -1,70 +0,0 @@
|
||||
package iceberg
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
"strings"
|
||||
|
||||
"github.com/gorilla/mux"
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api/s3_constants"
|
||||
)
|
||||
|
||||
// validateRequestPath rejects Iceberg REST requests whose captured
|
||||
// {prefix}/{namespace}/{table} mux vars would produce a parent-directory
|
||||
// traversal when joined into a filer path. The iceberg router runs with
|
||||
// SkipClean(true), so `..` survives routing; downstream path.Join calls
|
||||
// (stageCreateMarkerDir, location builders, etc.) then collapse it and
|
||||
// escape the table-bucket directory.
|
||||
//
|
||||
// {prefix} maps to a table-bucket name; {table} is a single path segment;
|
||||
// {namespace} is unit-separator (0x1F) joined parts that get flattened into
|
||||
// a single dotted name for the on-disk layout — each part is validated
|
||||
// individually.
|
||||
func validateRequestPath(next http.Handler) http.Handler {
|
||||
return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
vars := mux.Vars(r)
|
||||
// Use the comma-ok form so vars only checked when the matched route
|
||||
// actually captures them; when captured, an empty value is itself a
|
||||
// rejection because downstream path.Join would collapse it.
|
||||
if prefix, ok := vars["prefix"]; ok {
|
||||
if prefix == "" || !s3_constants.IsValidBucketName(prefix) {
|
||||
writeError(w, http.StatusBadRequest, "BadRequest", "invalid prefix")
|
||||
return
|
||||
}
|
||||
}
|
||||
if table, ok := vars["table"]; ok {
|
||||
if table == "" || !isValidNameSegment(table) {
|
||||
writeError(w, http.StatusBadRequest, "BadRequest", "invalid table name")
|
||||
return
|
||||
}
|
||||
}
|
||||
if ns, ok := vars["namespace"]; ok {
|
||||
if ns == "" {
|
||||
writeError(w, http.StatusBadRequest, "BadRequest", "invalid namespace")
|
||||
return
|
||||
}
|
||||
// Reject leading/trailing/consecutive unit separators so distinct
|
||||
// inputs cannot collapse to the same parsed namespace via
|
||||
// parseNamespace's empty-part filter.
|
||||
for _, part := range strings.Split(ns, "\x1F") {
|
||||
if part == "" || !isValidNameSegment(part) {
|
||||
writeError(w, http.StatusBadRequest, "BadRequest", "invalid namespace")
|
||||
return
|
||||
}
|
||||
}
|
||||
}
|
||||
next.ServeHTTP(w, r)
|
||||
})
|
||||
}
|
||||
|
||||
// isValidNameSegment rejects a single path-segment value (bucket prefix slot,
|
||||
// table name, or one namespace part) that would be unsafe to embed in a filer
|
||||
// path: `.`, `..`, embedded slash/backslash, or NUL.
|
||||
func isValidNameSegment(s string) bool {
|
||||
if s == "" {
|
||||
return true
|
||||
}
|
||||
if s == "." || s == ".." {
|
||||
return false
|
||||
}
|
||||
return !strings.ContainsAny(s, "/\\\x00")
|
||||
}
|
||||
@@ -1,116 +0,0 @@
|
||||
package iceberg
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"testing"
|
||||
|
||||
"github.com/gorilla/mux"
|
||||
)
|
||||
|
||||
func TestValidateRequestPath_RejectsTraversal(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
rawPath string
|
||||
wantCode int
|
||||
}{
|
||||
{"clean namespace+table passes", "/v1/namespaces/sales/tables/orders", http.StatusOK},
|
||||
{"clean prefixed passes", "/v1/wh/namespaces/sales/tables/orders", http.StatusOK},
|
||||
{"clean namespace only passes", "/v1/namespaces/sales", http.StatusOK},
|
||||
|
||||
// SkipClean(true) means raw `..` survives routing — these are the
|
||||
// realistic traversal shapes the middleware must catch.
|
||||
{"dotdot as prefix var rejected", "/v1/../namespaces/sales", http.StatusBadRequest},
|
||||
{"dotdot as namespace var rejected", "/v1/namespaces/..", http.StatusBadRequest},
|
||||
{"dotdot as namespace var prefixed rejected", "/v1/wh/namespaces/..", http.StatusBadRequest},
|
||||
{"dotdot as table var rejected", "/v1/namespaces/sales/tables/..", http.StatusBadRequest},
|
||||
{"dot as table var rejected", "/v1/namespaces/sales/tables/.", http.StatusBadRequest},
|
||||
// Iceberg clients send the 0x1F unit separator percent-encoded; mux
|
||||
// decodes it before the middleware sees the namespace var.
|
||||
{"unit-sep namespace with dotdot part rejected", "/v1/namespaces/sales%1F..%1Fevil", http.StatusBadRequest},
|
||||
{"leading unit-sep namespace rejected", "/v1/namespaces/%1Fsales", http.StatusBadRequest},
|
||||
{"trailing unit-sep namespace rejected", "/v1/namespaces/sales%1F", http.StatusBadRequest},
|
||||
{"consecutive unit-sep namespace rejected", "/v1/namespaces/sales%1F%1Fevil", http.StatusBadRequest},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
router := mux.NewRouter().SkipClean(true)
|
||||
router.Use(validateRequestPath)
|
||||
handlerCalled := false
|
||||
pass := func(w http.ResponseWriter, r *http.Request) {
|
||||
handlerCalled = true
|
||||
w.WriteHeader(http.StatusOK)
|
||||
}
|
||||
router.HandleFunc("/v1/namespaces/{namespace}", pass)
|
||||
router.HandleFunc("/v1/namespaces/{namespace}/tables/{table}", pass)
|
||||
router.HandleFunc("/v1/{prefix}/namespaces/{namespace}", pass)
|
||||
router.HandleFunc("/v1/{prefix}/namespaces/{namespace}/tables/{table}", pass)
|
||||
|
||||
req := httptest.NewRequest(http.MethodGet, tt.rawPath, nil)
|
||||
rr := httptest.NewRecorder()
|
||||
router.ServeHTTP(rr, req)
|
||||
|
||||
if rr.Code != tt.wantCode {
|
||||
t.Fatalf("path %q: got status %d, want %d (body=%q)", tt.rawPath, rr.Code, tt.wantCode, rr.Body.String())
|
||||
}
|
||||
if tt.wantCode == http.StatusBadRequest && handlerCalled {
|
||||
t.Fatalf("path %q: inner handler reached despite rejection", tt.rawPath)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// Defense-in-depth: if a future route or middleware ever leaves one of the
|
||||
// captured vars empty, the middleware must still reject the request. The
|
||||
// default mux regex won't normally allow this.
|
||||
func TestValidateRequestPath_RejectsEmptyCapturedVars(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
vars map[string]string
|
||||
}{
|
||||
{"empty prefix", map[string]string{"prefix": "", "namespace": "ns"}},
|
||||
{"empty table", map[string]string{"namespace": "ns", "table": ""}},
|
||||
{"empty namespace", map[string]string{"namespace": ""}},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
handlerCalled := false
|
||||
h := validateRequestPath(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
handlerCalled = true
|
||||
}))
|
||||
req := mux.SetURLVars(httptest.NewRequest(http.MethodGet, "/", nil), tt.vars)
|
||||
rr := httptest.NewRecorder()
|
||||
h.ServeHTTP(rr, req)
|
||||
if handlerCalled {
|
||||
t.Fatalf("vars %v: inner handler reached despite empty capture", tt.vars)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestIsValidNameSegment(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
input string
|
||||
want bool
|
||||
}{
|
||||
{"empty ok", "", true},
|
||||
{"plain", "orders", true},
|
||||
{"with dot inside", "my.table", true},
|
||||
{"hidden", ".hidden", true},
|
||||
|
||||
{"bare dot", ".", false},
|
||||
{"bare dotdot", "..", false},
|
||||
{"contains slash", "foo/bar", false},
|
||||
{"contains backslash", "foo\\bar", false},
|
||||
{"contains nul", "foo\x00bar", false},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
if got := isValidNameSegment(tt.input); got != tt.want {
|
||||
t.Errorf("isValidNameSegment(%q) = %v, want %v", tt.input, got, tt.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -71,13 +71,6 @@ func (s *Server) RegisterRoutes(router *mux.Router) {
|
||||
// Add middleware to log all requests/responses
|
||||
router.Use(loggingMiddleware)
|
||||
|
||||
// Reject `..`/`.`/NUL in {prefix}/{namespace}/{table} vars before any
|
||||
// handler runs. The router uses SkipClean(true), so traversal segments
|
||||
// would otherwise reach path.Join in stage-marker / location builders.
|
||||
// Registered after loggingMiddleware so rejected requests still get
|
||||
// audit-logged.
|
||||
router.Use(validateRequestPath)
|
||||
|
||||
// Configuration endpoint - no auth needed for config
|
||||
router.HandleFunc("/v1/config", s.handleConfig).Methods(http.MethodGet)
|
||||
|
||||
|
||||
@@ -177,41 +177,6 @@ func GetBucketAndObject(r *http.Request) (bucket, object string) {
|
||||
return
|
||||
}
|
||||
|
||||
// IsValidObjectKey rejects S3 object keys that — after normalization
|
||||
// (backslash→slash, slash collapse) — contain a `.` or `..` path segment, or
|
||||
// embed a NUL byte. Such keys are collapsed by filepath.Join inside the filer
|
||||
// and would escape the bucket directory, so they must not reach the gRPC layer.
|
||||
// Gorilla mux URL-decodes captured vars before this runs, so `%2e%2e` is
|
||||
// already `..` here.
|
||||
func IsValidObjectKey(object string) bool {
|
||||
if object == "" {
|
||||
return true
|
||||
}
|
||||
if strings.ContainsRune(object, '\x00') {
|
||||
return false
|
||||
}
|
||||
object = strings.ReplaceAll(object, "\\", "/")
|
||||
for _, seg := range strings.Split(object, "/") {
|
||||
if seg == "." || seg == ".." {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// IsValidBucketName rejects bucket names captured from the URL path that are
|
||||
// unsafe to use in filer path construction (`.`, `..`, contain `/` or `\`, or
|
||||
// embed NUL). This is a path-safety check, not a full S3 naming-rule check.
|
||||
func IsValidBucketName(bucket string) bool {
|
||||
if bucket == "" {
|
||||
return true
|
||||
}
|
||||
if bucket == "." || bucket == ".." {
|
||||
return false
|
||||
}
|
||||
return !strings.ContainsAny(bucket, "/\\\x00")
|
||||
}
|
||||
|
||||
// NormalizeObjectKey normalizes object keys by removing duplicate slashes and converting backslashes.
|
||||
// This normalizes keys from various sources (URL path, form values, etc.) to a consistent format.
|
||||
// It also converts Windows-style backslashes to forward slashes for cross-platform compatibility.
|
||||
|
||||
@@ -89,65 +89,6 @@ func TestNormalizeObjectKey(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestIsValidObjectKey(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
input string
|
||||
want bool
|
||||
}{
|
||||
{"empty", "", true},
|
||||
{"plain", "folder/file.txt", true},
|
||||
{"leading slash", "/folder/file.txt", true},
|
||||
{"trailing slash", "folder/", true},
|
||||
{"hidden file ok", ".hidden", true},
|
||||
{"dotdot in name ok", "..hidden", true},
|
||||
{"double dots inside name", "foo..bar/baz", true},
|
||||
|
||||
{"bare dotdot", "..", false},
|
||||
{"bare dot", ".", false},
|
||||
{"leading dotdot segment", "../evil-bucket/test.txt", false},
|
||||
{"leading dot-slash", "./evil/test.txt", false},
|
||||
{"nested dotdot segment", "good/../evil/test.txt", false},
|
||||
{"trailing dotdot segment", "good/..", false},
|
||||
{"backslash dotdot", "..\\evil\\test.txt", false},
|
||||
{"mixed-slash dotdot", "good\\..\\evil/test.txt", false},
|
||||
{"dotdot after duplicate slash", "good//../evil", false},
|
||||
{"nul byte", "foo\x00bar", false},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
if got := IsValidObjectKey(tt.input); got != tt.want {
|
||||
t.Errorf("IsValidObjectKey(%q) = %v, want %v", tt.input, got, tt.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestIsValidBucketName(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
input string
|
||||
want bool
|
||||
}{
|
||||
{"empty ok", "", true},
|
||||
{"plain", "my-bucket", true},
|
||||
{"name containing dots", "my.bucket.name", true},
|
||||
|
||||
{"bare dot", ".", false},
|
||||
{"bare dotdot", "..", false},
|
||||
{"with slash", "evil/bucket", false},
|
||||
{"with backslash", "evil\\bucket", false},
|
||||
{"with nul", "evil\x00bucket", false},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
if got := IsValidBucketName(tt.input); got != tt.want {
|
||||
t.Errorf("IsValidBucketName(%q) = %v, want %v", tt.input, got, tt.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestRemoveDuplicateSlashes(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
package s3api
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
@@ -16,7 +15,6 @@ import (
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/kms"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/s3_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api/cors"
|
||||
@@ -46,8 +44,8 @@ type BucketConfig struct {
|
||||
// rather than this cache.
|
||||
LifecycleTTL *LifecycleTTLResolver
|
||||
KMSKeyCache *BucketKMSCache // Per-bucket KMS key cache for SSE-KMS operations
|
||||
LastModified time.Time
|
||||
Entry *filer_pb.Entry
|
||||
LastModified time.Time
|
||||
Entry *filer_pb.Entry
|
||||
}
|
||||
|
||||
// BucketKMSCache represents per-bucket KMS key caching for SSE-KMS operations
|
||||
@@ -526,74 +524,26 @@ func (s3a *S3ApiServer) updateBucketConfig(bucket string, updateFn func(*BucketC
|
||||
bucket, s3_constants.ExtObjectLockEnabledKey, string(nextConfig.Entry.Extended[s3_constants.ExtObjectLockEnabledKey]))
|
||||
}
|
||||
|
||||
// Patch only the changed/removed extended keys, leaving Entry.content
|
||||
// untouched so a concurrent content write (e.g. encryption) is preserved.
|
||||
oldExt := config.Entry.GetExtended()
|
||||
newExt := nextConfig.Entry.Extended
|
||||
set := make(map[string][]byte)
|
||||
for k, v := range newExt {
|
||||
if ov, ok := oldExt[k]; !ok || !bytes.Equal(ov, v) {
|
||||
set[k] = v
|
||||
}
|
||||
}
|
||||
var del []string
|
||||
for k := range oldExt {
|
||||
if _, ok := newExt[k]; !ok {
|
||||
del = append(del, k)
|
||||
}
|
||||
}
|
||||
glog.V(3).Infof("updateBucketConfig: patching %d/%d extended keys for bucket %s", len(set), len(del), bucket)
|
||||
if err := s3a.patchBucketEntry(bucket, &filer_pb.ObjectMutation{SetExtended: set, DeleteExtended: del}); err != nil {
|
||||
glog.Errorf("updateBucketConfig: failed to patch bucket entry for %s: %v", bucket, err)
|
||||
// Save to filer
|
||||
glog.V(3).Infof("updateBucketConfig: saving entry to filer for bucket %s", bucket)
|
||||
err := s3a.updateEntry(s3a.bucketRoot(bucket), nextConfig.Entry)
|
||||
if err != nil {
|
||||
glog.Errorf("updateBucketConfig: failed to update bucket entry for %s: %v", bucket, err)
|
||||
return s3err.ErrInternalError
|
||||
}
|
||||
glog.V(3).Infof("updateBucketConfig: saved entry to filer for bucket %s", bucket)
|
||||
|
||||
// Invalidate rather than cache nextConfig: its content may be stale relative
|
||||
// to a concurrent content write. The next read re-fetches the merged entry.
|
||||
if s3a.bucketConfigCache != nil {
|
||||
s3a.bucketConfigCache.Remove(bucket)
|
||||
s3a.bucketConfigCache.RemoveNegativeCache(bucket)
|
||||
}
|
||||
// Update cache. Re-derive every Extended-backed field from the
|
||||
// just-saved Entry — the user's update fn may have flipped, added,
|
||||
// or cleared bytes (e.g. PutBucketLifecycle / DeleteBucketLifecycle
|
||||
// rewrites the lifecycle XML key) and the resolver / parsed configs
|
||||
// must follow.
|
||||
s3a.populateBucketConfigDerivedFields(nextConfig)
|
||||
s3a.bucketConfigCache.Set(bucket, nextConfig)
|
||||
|
||||
return s3err.ErrNone
|
||||
}
|
||||
|
||||
// patchBucketEntry applies a field-level PATCH_EXTENDED mutation to the bucket's
|
||||
// entry via ObjectTransaction, routed to the bucket's owner filer so its per-path
|
||||
// lock serializes concurrent config writes cluster-wide rather than racing
|
||||
// whole-entry rewrites. A nil/empty mutation is a no-op.
|
||||
func (s3a *S3ApiServer) patchBucketEntry(bucket string, m *filer_pb.ObjectMutation) error {
|
||||
if m == nil || (len(m.SetExtended) == 0 && len(m.DeleteExtended) == 0 && !m.SetContent) {
|
||||
return nil
|
||||
}
|
||||
dir := s3a.option.BucketsPath
|
||||
bucketPath := dir + "/" + bucket
|
||||
m.Type = filer_pb.ObjectMutation_PATCH_EXTENDED
|
||||
m.Directory = dir
|
||||
m.Name = bucket
|
||||
req := &filer_pb.ObjectTransactionRequest{
|
||||
LockKey: bucketPath,
|
||||
RouteKey: objectWriteRouteKeyPrefix + bucketPath,
|
||||
Mutations: []*filer_pb.ObjectMutation{m},
|
||||
}
|
||||
txn := func(client filer_pb.SeaweedFilerClient) error {
|
||||
resp, err := client.ObjectTransaction(context.Background(), req)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if resp.Error != "" {
|
||||
return fmt.Errorf("patch bucket %s: %s", bucket, resp.Error)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
if s3a.objectWriteLockClient != nil {
|
||||
if owner := s3a.objectWriteLockClient.PrimaryForKey(objectWriteRouteKeyPrefix + bucketPath); owner != "" {
|
||||
return pb.WithFilerClient(false, 0, owner, s3a.option.GrpcDialOption, txn)
|
||||
}
|
||||
}
|
||||
return s3a.WithFilerClient(false, txn)
|
||||
}
|
||||
|
||||
func cloneBucketConfig(config *BucketConfig) *BucketConfig {
|
||||
if config == nil {
|
||||
return nil
|
||||
@@ -1120,11 +1070,27 @@ func (s3a *S3ApiServer) setBucketMetadata(bucket string, metadata *BucketMetadat
|
||||
return fmt.Errorf("failed to marshal bucket metadata to protobuf: %w", err)
|
||||
}
|
||||
|
||||
// Patch only Entry.content so a concurrent extended-attribute write
|
||||
// (e.g. versioning) is preserved.
|
||||
err = s3a.patchBucketEntry(bucket, &filer_pb.ObjectMutation{
|
||||
SetContent: true,
|
||||
Content: metadataBytes,
|
||||
// Update the bucket entry with new content
|
||||
err = s3a.WithFilerClient(false, func(client filer_pb.SeaweedFilerClient) error {
|
||||
// Get current bucket entry
|
||||
entry, err := s3a.getBucketEntry(bucket)
|
||||
if err != nil {
|
||||
return fmt.Errorf("error retrieving bucket directory %s: %w", bucket, err)
|
||||
}
|
||||
if entry == nil {
|
||||
return fmt.Errorf("bucket directory not found %s", bucket)
|
||||
}
|
||||
|
||||
// Update content with metadata
|
||||
entry.Content = metadataBytes
|
||||
|
||||
request := &filer_pb.UpdateEntryRequest{
|
||||
Directory: s3a.bucketRoot(bucket),
|
||||
Entry: entry,
|
||||
}
|
||||
|
||||
_, err = client.UpdateEntry(context.Background(), request)
|
||||
return err
|
||||
})
|
||||
|
||||
// Invalidate cache after successful update
|
||||
|
||||
@@ -324,7 +324,7 @@ func removeDuplicateSlashes(object string) string {
|
||||
// hasChildren("bucket", "empty-dir") where no children exist → false
|
||||
//
|
||||
// Performance: ~1-5ms per call (one gRPC LIST request with Limit=1)
|
||||
func (s3a *S3ApiServer) hasChildren(ctx context.Context, bucket, prefix string) bool {
|
||||
func (s3a *S3ApiServer) hasChildren(bucket, prefix string) bool {
|
||||
// Clean up prefix: remove leading slashes
|
||||
cleanPrefix := strings.TrimPrefix(prefix, "/")
|
||||
|
||||
@@ -334,10 +334,8 @@ func (s3a *S3ApiServer) hasChildren(ctx context.Context, bucket, prefix string)
|
||||
|
||||
// List one child object. filer_pb.List cancels the underlying ListEntries
|
||||
// stream when it returns, so gRPC's per-stream client goroutine is not leaked.
|
||||
// The caller's request context is propagated so the probe is cancelled if the
|
||||
// client disconnects.
|
||||
found := false
|
||||
err := filer_pb.List(ctx, s3a, fullPath, "", func(*filer_pb.Entry, bool) error {
|
||||
err := filer_pb.List(context.Background(), s3a, fullPath, "", func(*filer_pb.Entry, bool) error {
|
||||
found = true
|
||||
return nil
|
||||
}, "", true, 1)
|
||||
@@ -2322,7 +2320,7 @@ func (s3a *S3ApiServer) HeadObjectHandler(w http.ResponseWriter, r *http.Request
|
||||
}
|
||||
if isZeroByteFile {
|
||||
// Check if it has children (making it an implicit directory)
|
||||
if s3a.hasChildren(r.Context(), bucket, object) {
|
||||
if s3a.hasChildren(bucket, object) {
|
||||
// This is an implicit directory with children
|
||||
// Return 404 to force clients (like s3fs) to use LIST-based discovery
|
||||
s3err.WriteErrorResponse(w, r, s3err.ErrNoSuchKey)
|
||||
|
||||
@@ -158,12 +158,9 @@ func (s3a *S3ApiServer) CopyObjectHandler(w http.ResponseWriter, r *http.Request
|
||||
if sameDestination && (replaceMeta || replaceTagging) && s3a.canUseMetadataOnlySelfCopy(entry, r, dstBucket, dstObject) {
|
||||
var dstVersionId string
|
||||
var etag string
|
||||
// A non-versioned in-place metadata replace routes to the owner as a
|
||||
// serialized PATCH (off the distributed lock); versioned/suspended (which
|
||||
// create a new version) and the no-owner bootstrap keep the lock.
|
||||
owner := s3a.objectWriteOwner(dstBucket, dstObject)
|
||||
routeInPlace := owner != "" && dstVersioningState == ""
|
||||
selfCopyBody := func() s3err.ErrorCode {
|
||||
updateCode := s3a.withObjectWriteLock(dstBucket, dstObject, func() s3err.ErrorCode {
|
||||
return s3a.checkConditionalHeaders(r, dstBucket, dstObject)
|
||||
}, func() s3err.ErrorCode {
|
||||
currentEntry, currentErr := s3a.resolveCopySourceEntry(srcBucket, srcObject, srcVersionId, srcVersioningState)
|
||||
if currentErr != nil || currentEntry.IsDirectory {
|
||||
return s3err.ErrInvalidCopySource
|
||||
@@ -171,41 +168,26 @@ func (s3a *S3ApiServer) CopyObjectHandler(w http.ResponseWriter, r *http.Request
|
||||
if errCode := s3a.validateConditionalCopyHeaders(r, currentEntry); errCode != s3err.ErrNone {
|
||||
return errCode
|
||||
}
|
||||
updatedMetadata, metadataErr := processMetadataBytes(r.Header, currentEntry.Extended, replaceMeta, replaceTagging)
|
||||
if metadataErr != nil {
|
||||
glog.Errorf("CopyObjectHandler ValidateTags error %s: %v", r.URL, metadataErr)
|
||||
|
||||
updatedEntry := cloneProtoEntry(currentEntry)
|
||||
updatedMetadata, metadataErr := processMetadataBytes(r.Header, updatedEntry.Extended, replaceMeta, replaceTagging)
|
||||
currentErr = metadataErr
|
||||
if currentErr != nil {
|
||||
glog.Errorf("CopyObjectHandler ValidateTags error %s: %v", r.URL, currentErr)
|
||||
return s3err.ErrInvalidTag
|
||||
}
|
||||
if routeInPlace {
|
||||
if err := s3a.routedMetadataReplace(owner, dstBucket, dstObject, currentEntry, updatedMetadata); err != nil {
|
||||
return filerErrorToS3Error(err)
|
||||
}
|
||||
etag = getEtagFromEntry(currentEntry)
|
||||
return s3err.ErrNone
|
||||
}
|
||||
updatedEntry := cloneProtoEntry(currentEntry)
|
||||
updatedEntry.Extended = mergeCopyMetadata(updatedEntry.Extended, updatedMetadata)
|
||||
if updatedEntry.Attributes == nil {
|
||||
updatedEntry.Attributes = &filer_pb.FuseAttributes{}
|
||||
}
|
||||
updatedEntry.Attributes.Mtime = t.Unix()
|
||||
var finErr error
|
||||
dstVersionId, etag, finErr = s3a.finalizeCopyDestination(dstBucket, dstObject, dstVersioningState, updatedEntry)
|
||||
if finErr != nil {
|
||||
return filerErrorToS3Error(finErr)
|
||||
|
||||
dstVersionId, etag, currentErr = s3a.finalizeCopyDestination(dstBucket, dstObject, dstVersioningState, updatedEntry)
|
||||
if currentErr != nil {
|
||||
return filerErrorToS3Error(currentErr)
|
||||
}
|
||||
return s3err.ErrNone
|
||||
}
|
||||
var updateCode s3err.ErrorCode
|
||||
if routeInPlace {
|
||||
if updateCode = s3a.checkConditionalHeaders(r, dstBucket, dstObject); updateCode == s3err.ErrNone {
|
||||
updateCode = selfCopyBody()
|
||||
}
|
||||
} else {
|
||||
updateCode = s3a.withObjectWriteLock(dstBucket, dstObject, func() s3err.ErrorCode {
|
||||
return s3a.checkConditionalHeaders(r, dstBucket, dstObject)
|
||||
}, selfCopyBody)
|
||||
}
|
||||
})
|
||||
if updateCode != s3err.ErrNone {
|
||||
s3err.WriteErrorResponse(w, r, updateCode)
|
||||
return
|
||||
@@ -462,16 +444,7 @@ func (s3a *S3ApiServer) finalizeCopyDestination(dstBucket, dstObject, dstVersion
|
||||
return "", "", err
|
||||
}
|
||||
|
||||
// Route the pointer flip to the owner filer when known (off the
|
||||
// distributed lock); RECOMPUTE_LATEST picks the just-written version.
|
||||
if owner := s3a.objectWriteOwner(dstBucket, normalizedObject); owner != "" {
|
||||
if code := s3a.routedVersionedFinalize(owner, dstBucket, normalizedObject, isNewFormatVersionId(versionId)); code != s3err.ErrNone {
|
||||
if rollbackErr := s3a.rollbackCopyVersion(bucketDir, versionObjectPath); rollbackErr != nil {
|
||||
glog.Errorf("CopyObjectHandler: failed to rollback version %s for %s/%s after routed finalize error: %v", versionId, dstBucket, normalizedObject, rollbackErr)
|
||||
}
|
||||
return "", "", fmt.Errorf("routed finalize for %s/%s: code %d", dstBucket, normalizedObject, code)
|
||||
}
|
||||
} else if err = s3a.updateLatestVersionInDirectory(dstBucket, normalizedObject, versionId, versionFileName, dstEntry); err != nil {
|
||||
if err = s3a.updateLatestVersionInDirectory(dstBucket, normalizedObject, versionId, versionFileName, dstEntry); err != nil {
|
||||
if rollbackErr := s3a.rollbackCopyVersion(bucketDir, versionObjectPath); rollbackErr != nil {
|
||||
glog.Errorf("CopyObjectHandler: failed to rollback version %s for %s/%s after latest pointer update error: %v", versionId, dstBucket, normalizedObject, rollbackErr)
|
||||
}
|
||||
|
||||
@@ -397,7 +397,7 @@ func (s3a *S3ApiServer) copyObjectPartViaReencryption(
|
||||
filePath := s3a.genPartUploadPath(dstBucket, uploadID, partID)
|
||||
// Copy-part is an MPU part write under .uploads/<id>/<n>; lifecycle
|
||||
// TTL only applies to the eventual completed object. Pass 0.
|
||||
tag, code, putSSE := s3a.putToFiler(cloned, filePath, srcReader, dstBucket, "", partID, 0, nil, false)
|
||||
tag, code, putSSE := s3a.putToFiler(cloned, filePath, srcReader, dstBucket, "", partID, 0, nil)
|
||||
if code != s3err.ErrNone {
|
||||
return "", SSEResponseMetadata{}, code
|
||||
}
|
||||
|
||||
@@ -214,96 +214,33 @@ func (s3a *S3ApiServer) DeleteObjectHandler(w http.ResponseWriter, r *http.Reque
|
||||
}
|
||||
|
||||
var deleteResult deleteMutationResult
|
||||
var deleteCode s3err.ErrorCode
|
||||
|
||||
// Fast path: route the delete to the owner filer under its per-path lock;
|
||||
// routedObjectOwner excludes versioned/object-lock buckets.
|
||||
deleteHandled := false
|
||||
if !versioningConfigured {
|
||||
if cond, condOk := buildDeleteCondition(r); condOk {
|
||||
if owner, ownerOk := s3a.routedObjectOwner(bucket, object); ownerOk {
|
||||
resp, err := s3a.routedDelete(owner, bucket, object, cond)
|
||||
switch {
|
||||
case err != nil:
|
||||
glog.Warningf("DeleteObjectHandler: routed delete to %s failed for %s/%s, falling back to lock: %v", owner, bucket, object, err)
|
||||
case resp.ErrorCode == filer_pb.FilerError_PRECONDITION_FAILED:
|
||||
deleteCode, deleteHandled = s3err.ErrPreconditionFailed, true
|
||||
case resp.Error != "":
|
||||
// Non-precondition error: fall back (the lock path handles cases
|
||||
// the raw delete cannot, e.g. a non-empty directory marker).
|
||||
glog.Warningf("DeleteObjectHandler: routed delete to %s returned %q for %s/%s, falling back to lock", owner, resp.Error, bucket, object)
|
||||
default:
|
||||
deleteCode, deleteHandled = s3err.ErrNone, true
|
||||
}
|
||||
deleteCode := s3a.withObjectWriteLock(bucket, object, func() s3err.ErrorCode {
|
||||
return s3a.checkDeleteIfMatch(bucket, object, versionId, versioningState, r.Header.Get(s3_constants.IfMatch), s3err.ErrPreconditionFailed)
|
||||
}, func() s3err.ErrorCode {
|
||||
if versioningConfigured {
|
||||
result, errCode := s3a.deleteVersionedObject(r, bucket, object, versionId, versioningState)
|
||||
if errCode != s3err.ErrNone {
|
||||
return errCode
|
||||
}
|
||||
}
|
||||
}
|
||||
// Versioned/suspended delete with no specific version: route off the lock when
|
||||
// the bucket has an owner. createDeleteMarker routes its own pointer flip; a
|
||||
// delete marker never removes a locked version, so object-lock buckets route
|
||||
// here too. The If-Match precondition was already checked above.
|
||||
if !deleteHandled && versioningConfigured && versionId == "" {
|
||||
if owner := s3a.routableWriteOwner(bucket, object); owner != "" {
|
||||
deleteResult, deleteCode = s3a.deleteVersionedObject(r, bucket, object, versionId, versioningState)
|
||||
deleteHandled = true
|
||||
}
|
||||
}
|
||||
// Specific-version delete: route off the lock. A real version recomputes the
|
||||
// .versions pointer excluding it and deletes the version file; the null
|
||||
// version is the regular object entry, deleted directly. Object-lock buckets
|
||||
// gate the delete on the version's WORM guards, evaluated on the owner — for
|
||||
// governance bypass the retention guard is scoped to COMPLIANCE so the filer
|
||||
// allows a governance-mode delete while still denying compliance and legal
|
||||
// hold, without the gateway reading the version.
|
||||
if !deleteHandled && versionId != "" {
|
||||
worm, lockErr := s3a.isObjectLockEnabled(bucket)
|
||||
bypass := worm && s3a.evaluateGovernanceBypassRequest(r, bucket, object)
|
||||
if lockErr == nil {
|
||||
if owner := s3a.objectWriteOwner(bucket, object); owner != "" {
|
||||
deleteResult.versionId = versionId
|
||||
if ve, vErr := s3a.getSpecificObjectVersion(bucket, object, versionId); vErr == nil && ve != nil && ve.Extended != nil {
|
||||
if dm, ok := ve.Extended[s3_constants.ExtDeleteMarkerKey]; ok && string(dm) == "true" {
|
||||
deleteResult.deleteMarker = true
|
||||
}
|
||||
}
|
||||
if versionId == "null" {
|
||||
deleteCode = s3a.routedDeleteNullVersion(owner, bucket, object, worm, bypass)
|
||||
} else {
|
||||
deleteCode = s3a.routedDeleteSpecificVersion(owner, bucket, object, versionId, worm, bypass)
|
||||
}
|
||||
deleteHandled = true
|
||||
}
|
||||
}
|
||||
}
|
||||
if !deleteHandled {
|
||||
deleteCode = s3a.withObjectWriteLock(bucket, object, func() s3err.ErrorCode {
|
||||
return s3a.checkDeleteIfMatch(bucket, object, versionId, versioningState, r.Header.Get(s3_constants.IfMatch), s3err.ErrPreconditionFailed)
|
||||
}, func() s3err.ErrorCode {
|
||||
if versioningConfigured {
|
||||
result, errCode := s3a.deleteVersionedObject(r, bucket, object, versionId, versioningState)
|
||||
if errCode != s3err.ErrNone {
|
||||
return errCode
|
||||
}
|
||||
deleteResult = result
|
||||
return s3err.ErrNone
|
||||
}
|
||||
|
||||
governanceBypassAllowed := s3a.evaluateGovernanceBypassRequest(r, bucket, object)
|
||||
if err := s3a.enforceObjectLockProtections(r, bucket, object, "", governanceBypassAllowed); err != nil {
|
||||
glog.V(2).Infof("DeleteObjectHandler: object lock check failed for %s/%s: %v", bucket, object, err)
|
||||
return s3err.ErrAccessDenied
|
||||
}
|
||||
|
||||
if err := s3a.WithFilerClient(false, func(client filer_pb.SeaweedFilerClient) error {
|
||||
return s3a.deleteUnversionedObjectWithClient(client, bucket, object, false)
|
||||
}); err != nil {
|
||||
glog.Errorf("DeleteObjectHandler: failed to delete %s/%s: %v", bucket, object, err)
|
||||
return s3err.ErrInternalError
|
||||
}
|
||||
|
||||
deleteResult = result
|
||||
return s3err.ErrNone
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
governanceBypassAllowed := s3a.evaluateGovernanceBypassRequest(r, bucket, object)
|
||||
if err := s3a.enforceObjectLockProtections(r, bucket, object, "", governanceBypassAllowed); err != nil {
|
||||
glog.V(2).Infof("DeleteObjectHandler: object lock check failed for %s/%s: %v", bucket, object, err)
|
||||
return s3err.ErrAccessDenied
|
||||
}
|
||||
|
||||
if err := s3a.WithFilerClient(false, func(client filer_pb.SeaweedFilerClient) error {
|
||||
return s3a.deleteUnversionedObjectWithClient(client, bucket, object, false)
|
||||
}); err != nil {
|
||||
glog.Errorf("DeleteObjectHandler: failed to delete %s/%s: %v", bucket, object, err)
|
||||
return s3err.ErrInternalError
|
||||
}
|
||||
|
||||
return s3err.ErrNone
|
||||
})
|
||||
if deleteCode != s3err.ErrNone {
|
||||
s3err.WriteErrorResponse(w, r, deleteCode)
|
||||
return
|
||||
|
||||
@@ -109,7 +109,7 @@ func (s3a *S3ApiServer) ListObjectsV2Handler(w http.ResponseWriter, r *http.Requ
|
||||
// Adjust marker if it ends with delimiter to skip all entries with that prefix
|
||||
marker = adjustMarkerForDelimiter(marker, delimiter)
|
||||
|
||||
response, err := s3a.listFilerEntries(r.Context(), bucket, originalPrefix, maxKeys, marker, delimiter, encodingTypeUrl, fetchOwner)
|
||||
response, err := s3a.listFilerEntries(bucket, originalPrefix, maxKeys, marker, delimiter, encodingTypeUrl, fetchOwner)
|
||||
|
||||
if err != nil {
|
||||
s3err.WriteErrorResponse(w, r, s3err.ErrInternalError)
|
||||
@@ -173,7 +173,7 @@ func (s3a *S3ApiServer) ListObjectsV1Handler(w http.ResponseWriter, r *http.Requ
|
||||
// Adjust marker if it ends with delimiter to skip all entries with that prefix
|
||||
marker = adjustMarkerForDelimiter(marker, delimiter)
|
||||
|
||||
response, err := s3a.listFilerEntries(r.Context(), bucket, originalPrefix, uint16(maxKeys), marker, delimiter, encodingTypeUrl, true)
|
||||
response, err := s3a.listFilerEntries(bucket, originalPrefix, uint16(maxKeys), marker, delimiter, encodingTypeUrl, true)
|
||||
|
||||
if err != nil {
|
||||
s3err.WriteErrorResponse(w, r, s3err.ErrInternalError)
|
||||
@@ -232,7 +232,7 @@ func sanitizeV1MarkerEcho(response *ListBucketResult, marker string, encodingTyp
|
||||
}
|
||||
}
|
||||
|
||||
func (s3a *S3ApiServer) listFilerEntries(ctx context.Context, bucket string, originalPrefix string, maxKeys uint16, originalMarker string, delimiter string, encodingTypeUrl bool, fetchOwner bool) (response ListBucketResult, err error) {
|
||||
func (s3a *S3ApiServer) listFilerEntries(bucket string, originalPrefix string, maxKeys uint16, originalMarker string, delimiter string, encodingTypeUrl bool, fetchOwner bool) (response ListBucketResult, err error) {
|
||||
// convert full path prefix into directory name and prefix for entry name
|
||||
requestDir, prefix, marker := normalizePrefixMarker(originalPrefix, originalMarker)
|
||||
bucketPrefix := s3a.bucketPrefix(bucket)
|
||||
@@ -307,7 +307,7 @@ func (s3a *S3ApiServer) listFilerEntries(ctx context.Context, bucket string, ori
|
||||
if normalizedPrefix != "" {
|
||||
relativePath := strings.TrimPrefix(fmt.Sprintf("%s/%s", dir, entry.Name), bucketPrefix)
|
||||
relativePath = strings.TrimPrefix(relativePath, "/")
|
||||
if normalizedPrefix == relativePath && !s3a.hasChildren(ctx, bucket, relativePath) && !entry.IsDirectoryKeyObject() {
|
||||
if normalizedPrefix == relativePath && !s3a.hasChildren(bucket, relativePath) && !entry.IsDirectoryKeyObject() {
|
||||
return
|
||||
}
|
||||
}
|
||||
|
||||
@@ -457,7 +457,7 @@ func (s3a *S3ApiServer) PutObjectPartHandler(w http.ResponseWriter, r *http.Requ
|
||||
// transient .uploads/<id>/<n> path, and a part write would otherwise
|
||||
// start the TTL clock before CompleteMultipartUpload ever assembled
|
||||
// the object.
|
||||
etag, errCode, sseMetadata := s3a.putToFiler(r, filePath, dataReader, bucket, "", partID, 0, nil, false)
|
||||
etag, errCode, sseMetadata := s3a.putToFiler(r, filePath, dataReader, bucket, "", partID, 0, nil)
|
||||
if errCode != s3err.ErrNone {
|
||||
glog.Errorf("PutObjectPart: putToFiler failed with error code %v for bucket=%s, object=%s, partNumber=%d",
|
||||
errCode, bucket, object, partID)
|
||||
|
||||
@@ -135,7 +135,7 @@ func (s3a *S3ApiServer) PostPolicyBucketHandler(w http.ResponseWriter, r *http.R
|
||||
// fields and boundaries inflates ContentLength relative to the
|
||||
// object body, which would mis-evaluate any size-filtered rule.
|
||||
ttlSec := s3a.lifecycleTTLForObjectWrite(bucket, object, fileSize)
|
||||
etag, errCode, sseMetadata := s3a.putToFiler(r, filePath, fileBody, bucket, object, 1, ttlSec, nil, false)
|
||||
etag, errCode, sseMetadata := s3a.putToFiler(r, filePath, fileBody, bucket, object, 1, ttlSec, nil)
|
||||
|
||||
if errCode != s3err.ErrNone {
|
||||
s3err.WriteErrorResponse(w, r, errCode)
|
||||
|
||||
@@ -297,7 +297,7 @@ func (s3a *S3ApiServer) PutObjectHandler(w http.ResponseWriter, r *http.Request)
|
||||
}
|
||||
|
||||
ttlSec := s3a.lifecycleTTLForObjectWrite(bucket, object, r.ContentLength)
|
||||
etag, errCode, sseMetadata := s3a.putToFiler(r, filePath, dataReader, bucket, object, 1, ttlSec, nil, false)
|
||||
etag, errCode, sseMetadata := s3a.putToFiler(r, filePath, dataReader, bucket, object, 1, ttlSec, nil)
|
||||
|
||||
if errCode != s3err.ErrNone {
|
||||
s3err.WriteErrorResponse(w, r, errCode)
|
||||
@@ -359,7 +359,7 @@ func (s3a *S3ApiServer) withObjectWriteLock(bucket, object string, preconditionF
|
||||
// pass 0 because their own keys aren't the user-visible object the rule
|
||||
// targets and a part write would otherwise bind a TTL clock starting
|
||||
// before CompleteMultipartUpload.
|
||||
func (s3a *S3ApiServer) putToFiler(r *http.Request, filePath string, dataReader io.Reader, bucket string, object string, partNumber int, lifecycleTTLSec int32, afterCreate func(entry *filer_pb.Entry) s3err.ErrorCode, uniqueWritePath bool) (etag string, code s3err.ErrorCode, sseMetadata SSEResponseMetadata) {
|
||||
func (s3a *S3ApiServer) putToFiler(r *http.Request, filePath string, dataReader io.Reader, bucket string, object string, partNumber int, lifecycleTTLSec int32, afterCreate func(entry *filer_pb.Entry) s3err.ErrorCode) (etag string, code s3err.ErrorCode, sseMetadata SSEResponseMetadata) {
|
||||
// NEW OPTIMIZATION: Write directly to volume servers, bypassing filer proxy
|
||||
// This eliminates the filer proxy overhead for PUT operations
|
||||
// Note: filePath is now passed directly instead of URL (no parsing needed)
|
||||
@@ -802,7 +802,7 @@ func (s3a *S3ApiServer) putToFiler(r *http.Request, filePath string, dataReader
|
||||
}
|
||||
return s3a.checkConditionalHeaders(r, bucket, object)
|
||||
}
|
||||
createUnderLock := func() s3err.ErrorCode {
|
||||
createCode := s3a.withObjectWriteLock(bucket, object, preconditionFn, func() s3err.ErrorCode {
|
||||
createErr = s3a.WithFilerClient(false, func(client filer_pb.SeaweedFilerClient) error {
|
||||
req := &filer_pb.CreateEntryRequest{
|
||||
Directory: path.Dir(filePath),
|
||||
@@ -831,36 +831,7 @@ func (s3a *S3ApiServer) putToFiler(r *http.Request, filePath string, dataReader
|
||||
}
|
||||
}
|
||||
return s3err.ErrNone
|
||||
}
|
||||
|
||||
// Route the create to the object's owner filer, whose per-path lock
|
||||
// serializes it, then run afterCreate (e.g. a versioned finalize that routes
|
||||
// itself). Conditional/object-lock/non-reducible cases fall back to the
|
||||
// distributed lock.
|
||||
var createCode s3err.ErrorCode
|
||||
routed := false
|
||||
if owner := s3a.routableWriteOwner(bucket, object); owner != "" {
|
||||
if cond, ok := routeWriteCondition(r, uniqueWritePath); ok {
|
||||
resp, err := s3a.routedPut(owner, s3a.objectRouteKey(bucket, object), filePath, entry, cond)
|
||||
switch {
|
||||
case err != nil:
|
||||
glog.Warningf("putToFiler: routed PUT to %s failed for %s, falling back to lock: %v", owner, filePath, err)
|
||||
case resp.ErrorCode == filer_pb.FilerError_PRECONDITION_FAILED:
|
||||
createCode, routed = s3err.ErrPreconditionFailed, true
|
||||
case resp.Error != "":
|
||||
// Non-precondition mutation error: fall back so the lock path maps it.
|
||||
glog.Warningf("putToFiler: routed PUT to %s returned %q for %s, falling back to lock", owner, resp.Error, filePath)
|
||||
default:
|
||||
entryCreated, routed, createCode = true, true, s3err.ErrNone
|
||||
if afterCreate != nil {
|
||||
createCode = afterCreate(entry)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if !routed {
|
||||
createCode = s3a.withObjectWriteLock(bucket, object, preconditionFn, createUnderLock)
|
||||
}
|
||||
})
|
||||
if createCode != s3err.ErrNone {
|
||||
if createErr != nil {
|
||||
glog.Errorf("putToFiler: failed to create entry for %s: %v", filePath, createErr)
|
||||
@@ -1316,7 +1287,7 @@ func (s3a *S3ApiServer) putSuspendedVersioningObject(r *http.Request, bucket, ob
|
||||
glog.Warningf("putSuspendedVersioningObject: failed to update IsLatest flags: %v", err)
|
||||
}
|
||||
return s3err.ErrNone
|
||||
}, false)
|
||||
})
|
||||
if errCode != s3err.ErrNone {
|
||||
glog.Errorf("putSuspendedVersioningObject: failed to upload object: %v", errCode)
|
||||
return "", errCode, SSEResponseMetadata{}
|
||||
@@ -1482,8 +1453,13 @@ func (s3a *S3ApiServer) putVersionedObject(r *http.Request, bucket, object strin
|
||||
// Versioned bucket: resolver returns 0 by construction. Pass 0
|
||||
// directly — versioned objects sit on regular volumes and the
|
||||
// lifecycle worker handles their expiration.
|
||||
etag, errCode, sseMetadata = s3a.putToFiler(r, versionFilePath, body, bucket, normalizedObject, 1, 0,
|
||||
s3a.versionedAfterCreate(bucket, normalizedObject, versionId, versionFileName, useInvertedFormat), true)
|
||||
etag, errCode, sseMetadata = s3a.putToFiler(r, versionFilePath, body, bucket, normalizedObject, 1, 0, func(versionEntry *filer_pb.Entry) s3err.ErrorCode {
|
||||
if err := s3a.updateLatestVersionInDirectory(bucket, normalizedObject, versionId, versionFileName, versionEntry); err != nil {
|
||||
glog.Errorf("putVersionedObject: failed to update latest version in directory: %v", err)
|
||||
return s3err.ErrInternalError
|
||||
}
|
||||
return s3err.ErrNone
|
||||
})
|
||||
if errCode != s3err.ErrNone {
|
||||
glog.Errorf("putVersionedObject: failed to upload version: %v", errCode)
|
||||
return "", "", errCode, SSEResponseMetadata{}
|
||||
|
||||
@@ -1,272 +0,0 @@
|
||||
package s3api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"path"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api/s3_constants"
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api/s3err"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
)
|
||||
|
||||
// objectWriteRouteKeyPrefix namespaces an object's full path into the ring key
|
||||
// used to resolve and forward its writes. Shared by every routed builder so the
|
||||
// gateway and filer hash the same key.
|
||||
const objectWriteRouteKeyPrefix = "s3.object.write:"
|
||||
|
||||
// objectRouteKey is the ring key the gateway hashes to resolve an object's owner
|
||||
// filer. It is also sent as route_key on each routed transaction, so a non-owner
|
||||
// filer (reached because the gateway's ring view was stale) forwards the
|
||||
// transaction to the owner. All of an object's writes share this key.
|
||||
func (s3a *S3ApiServer) objectRouteKey(bucket, object string) string {
|
||||
return objectWriteRouteKeyPrefix + s3a.toFilerPath(bucket, object)
|
||||
}
|
||||
|
||||
// routableWriteOwner returns the owner filer for an object's writes, or "" to
|
||||
// keep them on the distributed lock. All writes to one object (versioned,
|
||||
// suspended, non-versioned) share the owner. Any lookup error falls back.
|
||||
func (s3a *S3ApiServer) routableWriteOwner(bucket, object string) pb.ServerAddress {
|
||||
if object == "" || s3a.objectWriteLockClient == nil {
|
||||
return ""
|
||||
}
|
||||
// Object-lock PUTs route: a versioned PUT creates a new version (never an
|
||||
// overwrite of a locked one), and a non-versioned overwrite is WORM-checked
|
||||
// gateway-side before dispatch. WORM-checked deletes use routedObjectOwner.
|
||||
return s3a.objectWriteLockClient.PrimaryForKey(s3a.objectRouteKey(bucket, object))
|
||||
}
|
||||
|
||||
// routedObjectOwner is routableWriteOwner restricted to non-versioned,
|
||||
// non-object-lock buckets, for the unversioned DELETE fast path.
|
||||
func (s3a *S3ApiServer) routedObjectOwner(bucket, object string) (pb.ServerAddress, bool) {
|
||||
if configured, err := s3a.isVersioningConfigured(bucket); err != nil || configured {
|
||||
return "", false
|
||||
}
|
||||
// An unversioned object-lock delete enforces WORM in the lock path; keep it
|
||||
// on the lock rather than routing past the check.
|
||||
if locked, err := s3a.isObjectLockEnabled(bucket); err != nil || locked {
|
||||
return "", false
|
||||
}
|
||||
owner := s3a.routableWriteOwner(bucket, object)
|
||||
return owner, owner != ""
|
||||
}
|
||||
|
||||
// routeWriteCondition reduces the request's conditional headers for a routed
|
||||
// create. A unique version path carries no precondition and only routes when the
|
||||
// request is unconditional (a conditional versioned write must check the latest,
|
||||
// which the lock path does); an overwrite carries the reduced condition.
|
||||
func routeWriteCondition(r *http.Request, uniqueWritePath bool) (*filer_pb.WriteCondition, bool) {
|
||||
cond, ok := buildWriteCondition(r)
|
||||
if !ok {
|
||||
return nil, false
|
||||
}
|
||||
if uniqueWritePath && cond != nil {
|
||||
return nil, false
|
||||
}
|
||||
return cond, true
|
||||
}
|
||||
|
||||
// buildWriteCondition reduces the request's conditional headers to a
|
||||
// WriteCondition. ok=false (combined headers, time conditions, ETag lists, weak
|
||||
// ETags) keeps gateway-side evaluation under the lock; a nil condition with
|
||||
// ok=true means unconditional.
|
||||
func buildWriteCondition(r *http.Request) (*filer_pb.WriteCondition, bool) {
|
||||
headers, errCode := parseConditionalHeaders(r)
|
||||
if errCode != s3err.ErrNone {
|
||||
return nil, false
|
||||
}
|
||||
if !headers.isSet {
|
||||
return nil, true
|
||||
}
|
||||
if !headers.ifModifiedSince.IsZero() || !headers.ifUnmodifiedSince.IsZero() {
|
||||
return nil, false
|
||||
}
|
||||
hasMatch := headers.ifMatch != ""
|
||||
hasNoneMatch := headers.ifNoneMatch != ""
|
||||
switch {
|
||||
case hasMatch && !hasNoneMatch:
|
||||
if headers.ifMatch == "*" {
|
||||
return clause(filer_pb.WriteCondition_IF_EXISTS), true
|
||||
}
|
||||
if etag, single := singleStrongETag(headers.ifMatch); single {
|
||||
return etagClause(filer_pb.WriteCondition_IF_ETAG_MATCH, etag), true
|
||||
}
|
||||
return nil, false
|
||||
case hasNoneMatch && !hasMatch:
|
||||
if headers.ifNoneMatch == "*" {
|
||||
return clause(filer_pb.WriteCondition_IF_NOT_EXISTS), true
|
||||
}
|
||||
if etag, single := singleStrongETag(headers.ifNoneMatch); single {
|
||||
return etagClause(filer_pb.WriteCondition_IF_ETAG_NOT_MATCH, etag), true
|
||||
}
|
||||
return nil, false
|
||||
default:
|
||||
return nil, false
|
||||
}
|
||||
}
|
||||
|
||||
// buildDeleteCondition reduces a DeleteObject's If-Match header to a condition;
|
||||
// DeleteObject honors only If-Match, matching checkDeleteIfMatch.
|
||||
func buildDeleteCondition(r *http.Request) (*filer_pb.WriteCondition, bool) {
|
||||
ifMatch := strings.TrimSpace(r.Header.Get(s3_constants.IfMatch))
|
||||
switch {
|
||||
case ifMatch == "":
|
||||
return nil, true
|
||||
case ifMatch == "*":
|
||||
return clause(filer_pb.WriteCondition_IF_EXISTS), true
|
||||
default:
|
||||
if etag, single := singleStrongETag(ifMatch); single {
|
||||
return etagClause(filer_pb.WriteCondition_IF_ETAG_MATCH, etag), true
|
||||
}
|
||||
return nil, false
|
||||
}
|
||||
}
|
||||
|
||||
func clause(kind filer_pb.WriteCondition_Kind) *filer_pb.WriteCondition {
|
||||
return &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{{Kind: kind}}}
|
||||
}
|
||||
|
||||
func etagClause(kind filer_pb.WriteCondition_Kind, etag string) *filer_pb.WriteCondition {
|
||||
return &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{{Kind: kind, Etags: []string{etag}}}}
|
||||
}
|
||||
|
||||
// singleStrongETag returns the normalized ETag when v carries exactly one strong
|
||||
// ETag, and false for ETag lists or weak ("W/") ETags.
|
||||
func singleStrongETag(v string) (string, bool) {
|
||||
v = strings.TrimSpace(v)
|
||||
if strings.Contains(v, ",") {
|
||||
return "", false
|
||||
}
|
||||
if strings.HasPrefix(v, "W/") || strings.HasPrefix(v, "w/") {
|
||||
return "", false
|
||||
}
|
||||
return strings.Trim(v, `"`), true
|
||||
}
|
||||
|
||||
func (s3a *S3ApiServer) objectTxnOnFiler(owner pb.ServerAddress, req *filer_pb.ObjectTransactionRequest) (*filer_pb.ObjectTransactionResponse, error) {
|
||||
var resp *filer_pb.ObjectTransactionResponse
|
||||
err := pb.WithFilerClient(false, 0, owner, s3a.option.GrpcDialOption, func(client filer_pb.SeaweedFilerClient) error {
|
||||
var e error
|
||||
resp, e = client.ObjectTransaction(context.Background(), req)
|
||||
return e
|
||||
})
|
||||
return resp, err
|
||||
}
|
||||
|
||||
// routedPut writes an object entry as a one-mutation ObjectTransaction on the
|
||||
// owner filer. lock_key is the object's full path so the transaction shares the
|
||||
// per-path lock with a concurrent create or delete of the same key.
|
||||
func (s3a *S3ApiServer) routedPut(owner pb.ServerAddress, routeKey, filePath string, entry *filer_pb.Entry, cond *filer_pb.WriteCondition) (*filer_pb.ObjectTransactionResponse, error) {
|
||||
return s3a.objectTxnOnFiler(owner, &filer_pb.ObjectTransactionRequest{
|
||||
LockKey: filePath,
|
||||
RouteKey: routeKey,
|
||||
Condition: cond,
|
||||
Mutations: []*filer_pb.ObjectMutation{{
|
||||
Type: filer_pb.ObjectMutation_PUT,
|
||||
Directory: path.Dir(filePath),
|
||||
Entry: entry,
|
||||
}},
|
||||
})
|
||||
}
|
||||
|
||||
// routedMkFile builds an entry like filer_pb.MkFile and writes it through a
|
||||
// routed PUT on the owner filer, for callers that would otherwise mkFile to the
|
||||
// default filer (e.g. multipart completion of a non-versioned object).
|
||||
func (s3a *S3ApiServer) routedMkFile(owner pb.ServerAddress, routeKey, parentDir, name string, chunks []*filer_pb.FileChunk, fn func(*filer_pb.Entry)) error {
|
||||
now := time.Now().Unix()
|
||||
entry := &filer_pb.Entry{
|
||||
Name: name,
|
||||
Attributes: &filer_pb.FuseAttributes{
|
||||
Mtime: now,
|
||||
Crtime: now,
|
||||
FileMode: uint32(0770),
|
||||
Uid: filer_pb.OS_UID,
|
||||
Gid: filer_pb.OS_GID,
|
||||
},
|
||||
Chunks: chunks,
|
||||
}
|
||||
if fn != nil {
|
||||
fn(entry)
|
||||
}
|
||||
resp, err := s3a.routedPut(owner, routeKey, parentDir+"/"+name, entry, nil)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if resp.Error != "" {
|
||||
return fmt.Errorf("routed mkfile %s/%s: %s", parentDir, name, resp.Error)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// writeMultipartObject writes a completed multipart object entry, routed to the
|
||||
// owner when known (so it serializes with concurrent writes to the same key)
|
||||
// and falling back to a plain mkFile otherwise. routeKey must be the same key the
|
||||
// caller used to resolve owner, so owner selection and forwarding stay consistent.
|
||||
func (s3a *S3ApiServer) writeMultipartObject(owner pb.ServerAddress, routeKey, dir, name string, chunks []*filer_pb.FileChunk, fn func(*filer_pb.Entry)) error {
|
||||
if owner != "" {
|
||||
return s3a.routedMkFile(owner, routeKey, dir, name, chunks, fn)
|
||||
}
|
||||
return s3a.mkFile(dir, name, chunks, fn)
|
||||
}
|
||||
|
||||
func (s3a *S3ApiServer) routedDelete(owner pb.ServerAddress, bucket, object string, cond *filer_pb.WriteCondition) (*filer_pb.ObjectTransactionResponse, error) {
|
||||
// NewFullPath normalizes a trailing-slash directory-marker key (e.g. "dir/")
|
||||
// to the entry name "dir", matching deleteUnversionedObjectWithClient.
|
||||
fullpath := util.NewFullPath(s3a.bucketDir(bucket), object)
|
||||
dir, name := fullpath.DirAndName()
|
||||
return s3a.objectTxnOnFiler(owner, &filer_pb.ObjectTransactionRequest{
|
||||
LockKey: string(fullpath),
|
||||
RouteKey: s3a.objectRouteKey(bucket, object),
|
||||
Condition: cond,
|
||||
Mutations: []*filer_pb.ObjectMutation{{
|
||||
Type: filer_pb.ObjectMutation_DELETE,
|
||||
Directory: dir,
|
||||
Name: name,
|
||||
IsDeleteData: true,
|
||||
}},
|
||||
})
|
||||
}
|
||||
|
||||
// routedMetadataReplace applies a metadata-only self-copy (REPLACE directive) to
|
||||
// an existing object in place via a routed PATCH_EXTENDED. The owner merges the
|
||||
// new managed metadata onto a fresh read of the entry under its per-path lock —
|
||||
// so a concurrent change to non-managed keys (legal hold, retention, version id)
|
||||
// is preserved rather than clobbered by a whole-entry rewrite — and bumps mtime.
|
||||
// updatedMetadata is the full managed-metadata set (processMetadataBytes); the
|
||||
// delete list is the managed keys the replace dropped.
|
||||
func (s3a *S3ApiServer) routedMetadataReplace(owner pb.ServerAddress, bucket, object string, current *filer_pb.Entry, updatedMetadata map[string][]byte) error {
|
||||
fullpath := util.NewFullPath(s3a.bucketDir(bucket), object)
|
||||
dir, name := fullpath.DirAndName()
|
||||
var del []string
|
||||
for k := range current.Extended {
|
||||
if isManagedCopyMetadataKey(k) {
|
||||
if _, keep := updatedMetadata[k]; !keep {
|
||||
del = append(del, k)
|
||||
}
|
||||
}
|
||||
}
|
||||
resp, err := s3a.objectTxnOnFiler(owner, &filer_pb.ObjectTransactionRequest{
|
||||
LockKey: string(fullpath),
|
||||
RouteKey: s3a.objectRouteKey(bucket, object),
|
||||
Mutations: []*filer_pb.ObjectMutation{{
|
||||
Type: filer_pb.ObjectMutation_PATCH_EXTENDED,
|
||||
Directory: dir,
|
||||
Name: name,
|
||||
SetExtended: updatedMetadata,
|
||||
DeleteExtended: del,
|
||||
TouchMtime: true,
|
||||
}},
|
||||
})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if resp.Error != "" {
|
||||
return fmt.Errorf("routed metadata replace %s/%s: %s", bucket, object, resp.Error)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
@@ -1,177 +0,0 @@
|
||||
package s3api
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
"testing"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api/s3_constants"
|
||||
)
|
||||
|
||||
func reqWith(headers map[string]string) *http.Request {
|
||||
r, _ := http.NewRequest(http.MethodPut, "/b/o", nil)
|
||||
for k, v := range headers {
|
||||
r.Header.Set(k, v)
|
||||
}
|
||||
return r
|
||||
}
|
||||
|
||||
// oneClause returns the single clause of cond, failing if it does not hold
|
||||
// exactly one.
|
||||
func oneClause(t *testing.T, cond *filer_pb.WriteCondition) *filer_pb.WriteCondition_Clause {
|
||||
t.Helper()
|
||||
if cond == nil {
|
||||
t.Fatal("expected a condition, got nil")
|
||||
}
|
||||
if len(cond.Clauses) != 1 {
|
||||
t.Fatalf("expected 1 clause, got %d", len(cond.Clauses))
|
||||
}
|
||||
return cond.Clauses[0]
|
||||
}
|
||||
|
||||
func TestBuildWriteCondition(t *testing.T) {
|
||||
t.Run("no headers is unconditional", func(t *testing.T) {
|
||||
cond, ok := buildWriteCondition(reqWith(nil))
|
||||
if !ok || cond != nil {
|
||||
t.Fatalf("want (nil, true), got (%v, %v)", cond, ok)
|
||||
}
|
||||
})
|
||||
t.Run("If-None-Match * to IF_NOT_EXISTS", func(t *testing.T) {
|
||||
cond, ok := buildWriteCondition(reqWith(map[string]string{s3_constants.IfNoneMatch: "*"}))
|
||||
if !ok {
|
||||
t.Fatal("want ok")
|
||||
}
|
||||
if c := oneClause(t, cond); c.Kind != filer_pb.WriteCondition_IF_NOT_EXISTS {
|
||||
t.Fatalf("kind = %v", c.Kind)
|
||||
}
|
||||
})
|
||||
t.Run("If-Match * to IF_EXISTS", func(t *testing.T) {
|
||||
cond, ok := buildWriteCondition(reqWith(map[string]string{s3_constants.IfMatch: "*"}))
|
||||
if !ok {
|
||||
t.Fatal("want ok")
|
||||
}
|
||||
if c := oneClause(t, cond); c.Kind != filer_pb.WriteCondition_IF_EXISTS {
|
||||
t.Fatalf("kind = %v", c.Kind)
|
||||
}
|
||||
})
|
||||
t.Run("If-Match strong etag to IF_ETAG_MATCH", func(t *testing.T) {
|
||||
cond, ok := buildWriteCondition(reqWith(map[string]string{s3_constants.IfMatch: `"abc123"`}))
|
||||
if !ok {
|
||||
t.Fatal("want ok")
|
||||
}
|
||||
c := oneClause(t, cond)
|
||||
if c.Kind != filer_pb.WriteCondition_IF_ETAG_MATCH || len(c.Etags) != 1 || c.Etags[0] != "abc123" {
|
||||
t.Fatalf("clause = %+v", c)
|
||||
}
|
||||
})
|
||||
t.Run("If-None-Match strong etag to IF_ETAG_NOT_MATCH", func(t *testing.T) {
|
||||
cond, ok := buildWriteCondition(reqWith(map[string]string{s3_constants.IfNoneMatch: `"abc123"`}))
|
||||
if !ok {
|
||||
t.Fatal("want ok")
|
||||
}
|
||||
c := oneClause(t, cond)
|
||||
if c.Kind != filer_pb.WriteCondition_IF_ETAG_NOT_MATCH || len(c.Etags) != 1 || c.Etags[0] != "abc123" {
|
||||
t.Fatalf("clause = %+v", c)
|
||||
}
|
||||
})
|
||||
t.Run("weak etag falls back", func(t *testing.T) {
|
||||
if _, ok := buildWriteCondition(reqWith(map[string]string{s3_constants.IfMatch: `W/"abc"`})); ok {
|
||||
t.Fatal("weak etag must not take the fast path")
|
||||
}
|
||||
})
|
||||
t.Run("etag list falls back", func(t *testing.T) {
|
||||
if _, ok := buildWriteCondition(reqWith(map[string]string{s3_constants.IfMatch: `"a","b"`})); ok {
|
||||
t.Fatal("etag list must not take the fast path")
|
||||
}
|
||||
})
|
||||
t.Run("both match and none-match falls back", func(t *testing.T) {
|
||||
if _, ok := buildWriteCondition(reqWith(map[string]string{
|
||||
s3_constants.IfMatch: "*",
|
||||
s3_constants.IfNoneMatch: "*",
|
||||
})); ok {
|
||||
t.Fatal("ambiguous combination must not take the fast path")
|
||||
}
|
||||
})
|
||||
t.Run("time-based falls back", func(t *testing.T) {
|
||||
if _, ok := buildWriteCondition(reqWith(map[string]string{
|
||||
"If-Unmodified-Since": "Wed, 21 Oct 2015 07:28:00 GMT",
|
||||
})); ok {
|
||||
t.Fatal("time condition must not take the fast path")
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestBuildDeleteCondition(t *testing.T) {
|
||||
t.Run("no If-Match is unconditional", func(t *testing.T) {
|
||||
cond, ok := buildDeleteCondition(reqWith(nil))
|
||||
if !ok || cond != nil {
|
||||
t.Fatalf("want (nil, true), got (%v, %v)", cond, ok)
|
||||
}
|
||||
})
|
||||
t.Run("If-Match * to IF_EXISTS", func(t *testing.T) {
|
||||
cond, ok := buildDeleteCondition(reqWith(map[string]string{s3_constants.IfMatch: "*"}))
|
||||
if !ok {
|
||||
t.Fatal("want ok")
|
||||
}
|
||||
if c := oneClause(t, cond); c.Kind != filer_pb.WriteCondition_IF_EXISTS {
|
||||
t.Fatalf("kind = %v", c.Kind)
|
||||
}
|
||||
})
|
||||
t.Run("If-Match etag to IF_ETAG_MATCH", func(t *testing.T) {
|
||||
cond, ok := buildDeleteCondition(reqWith(map[string]string{s3_constants.IfMatch: `"e"`}))
|
||||
if !ok {
|
||||
t.Fatal("want ok")
|
||||
}
|
||||
if c := oneClause(t, cond); c.Kind != filer_pb.WriteCondition_IF_ETAG_MATCH || c.Etags[0] != "e" {
|
||||
t.Fatalf("clause = %+v", c)
|
||||
}
|
||||
})
|
||||
t.Run("weak etag falls back", func(t *testing.T) {
|
||||
if _, ok := buildDeleteCondition(reqWith(map[string]string{s3_constants.IfMatch: `W/"e"`})); ok {
|
||||
t.Fatal("weak etag must not take the fast path")
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestSingleStrongETag(t *testing.T) {
|
||||
cases := []struct {
|
||||
in string
|
||||
want string
|
||||
single bool
|
||||
}{
|
||||
{`"abc"`, "abc", true},
|
||||
{` "abc" `, "abc", true},
|
||||
{`abc`, "abc", true},
|
||||
{`W/"abc"`, "", false},
|
||||
{`w/"abc"`, "", false},
|
||||
{`"a","b"`, "", false},
|
||||
}
|
||||
for _, c := range cases {
|
||||
got, single := singleStrongETag(c.in)
|
||||
if single != c.single || (single && got != c.want) {
|
||||
t.Errorf("singleStrongETag(%q) = (%q, %v), want (%q, %v)", c.in, got, single, c.want, c.single)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestRouteWriteCondition(t *testing.T) {
|
||||
// Unconditional routes either way.
|
||||
if c, ok := routeWriteCondition(reqWith(nil), false); !ok || c != nil {
|
||||
t.Fatalf("overwrite unconditional: got (%v,%v)", c, ok)
|
||||
}
|
||||
if c, ok := routeWriteCondition(reqWith(nil), true); !ok || c != nil {
|
||||
t.Fatalf("unique unconditional: got (%v,%v)", c, ok)
|
||||
}
|
||||
// An overwrite carries a reducible condition.
|
||||
if c, ok := routeWriteCondition(reqWith(map[string]string{s3_constants.IfMatch: `"e"`}), false); !ok || c == nil {
|
||||
t.Fatalf("overwrite conditional should route: got (%v,%v)", c, ok)
|
||||
}
|
||||
// A conditional unique (versioned) write bails to the lock path.
|
||||
if _, ok := routeWriteCondition(reqWith(map[string]string{s3_constants.IfMatch: `"e"`}), true); ok {
|
||||
t.Fatal("conditional unique write must not route")
|
||||
}
|
||||
// A non-reducible condition bails regardless.
|
||||
if _, ok := routeWriteCondition(reqWith(map[string]string{s3_constants.IfMatch: `W/"e"`}), false); ok {
|
||||
t.Fatal("weak etag must not route")
|
||||
}
|
||||
}
|
||||
@@ -1,188 +0,0 @@
|
||||
package s3api
|
||||
|
||||
import (
|
||||
"strconv"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api/s3_constants"
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api/s3err"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
)
|
||||
|
||||
// objectWriteOwner resolves the filer that owns all of an object's writes,
|
||||
// regardless of versioning state, or "" when no ring view is available. Normal,
|
||||
// suspended, and versioned writes to the same object hash to one owner and
|
||||
// serialize on its per-path lock.
|
||||
func (s3a *S3ApiServer) objectWriteOwner(bucket, object string) pb.ServerAddress {
|
||||
if s3a.objectWriteLockClient == nil {
|
||||
return ""
|
||||
}
|
||||
return s3a.objectWriteLockClient.PrimaryForKey(s3a.objectRouteKey(bucket, object))
|
||||
}
|
||||
|
||||
// latestPointerRecompute builds the RECOMPUTE_LATEST mutation that re-derives an
|
||||
// object's .versions pointer. excludeName, when set, omits a version about to be
|
||||
// deleted (so the pointer is repointed before the blob is removed); demote, when
|
||||
// set, stamps the displaced prior latest with NoncurrentSinceNs.
|
||||
func (s3a *S3ApiServer) latestPointerRecompute(bucket, object string, useInvertedFormat bool, excludeName string, demote bool) *filer_pb.ObjectMutation {
|
||||
versionsPath := s3a.toFilerPath(bucket, object+s3_constants.VersionsFolder)
|
||||
vdir, vname := util.FullPath(versionsPath).DirAndName()
|
||||
rc := &filer_pb.Recompute{
|
||||
ScanDir: versionsPath,
|
||||
// Inverted ids sort newest-first, so the newest is the first ascending
|
||||
// entry; legacy ids sort oldest-first (scan to the last).
|
||||
Descending: !useInvertedFormat,
|
||||
NameToKey: s3_constants.ExtLatestVersionFileNameKey,
|
||||
SizeToKey: s3_constants.ExtLatestVersionSizeKey,
|
||||
MtimeToKey: s3_constants.ExtLatestVersionMtimeKey,
|
||||
CopyExtended: map[string]string{
|
||||
s3_constants.ExtLatestVersionIdKey: s3_constants.ExtVersionIdKey,
|
||||
s3_constants.ExtLatestVersionETagKey: s3_constants.ExtETagKey,
|
||||
s3_constants.ExtLatestVersionOwnerKey: s3_constants.ExtAmzOwnerKey,
|
||||
s3_constants.ExtLatestVersionIsDeleteMarker: s3_constants.ExtDeleteMarkerKey,
|
||||
},
|
||||
ExcludeName: excludeName,
|
||||
}
|
||||
if demote {
|
||||
rc.DemoteKey = s3_constants.ExtNoncurrentSinceNsKey
|
||||
rc.DemoteValue = []byte(strconv.FormatInt(time.Now().UnixNano(), 10))
|
||||
}
|
||||
return &filer_pb.ObjectMutation{
|
||||
Type: filer_pb.ObjectMutation_RECOMPUTE_LATEST,
|
||||
Directory: vdir,
|
||||
Name: vname,
|
||||
Recompute: rc,
|
||||
}
|
||||
}
|
||||
|
||||
// routedVersionedFinalize flips the .versions pointer to the newest version and
|
||||
// demotes the prior latest, atomically under the object's per-path lock on the
|
||||
// owner filer, via a single RECOMPUTE_LATEST. The version file is already
|
||||
// written; the owner re-derives the pointer by scanning the directory.
|
||||
func (s3a *S3ApiServer) routedVersionedFinalize(owner pb.ServerAddress, bucket, object string, useInvertedFormat bool) s3err.ErrorCode {
|
||||
req := &filer_pb.ObjectTransactionRequest{
|
||||
LockKey: s3a.toFilerPath(bucket, object),
|
||||
RouteKey: s3a.objectRouteKey(bucket, object),
|
||||
Mutations: []*filer_pb.ObjectMutation{s3a.latestPointerRecompute(bucket, object, useInvertedFormat, "", true)},
|
||||
}
|
||||
resp, err := s3a.objectTxnOnFiler(owner, req)
|
||||
switch {
|
||||
case err != nil:
|
||||
glog.Errorf("routedVersionedFinalize: %s/%s on %s: %v", bucket, object, owner, err)
|
||||
return s3err.ErrInternalError
|
||||
case resp.Error != "":
|
||||
glog.Errorf("routedVersionedFinalize: %s/%s: %s", bucket, object, resp.Error)
|
||||
return s3err.ErrInternalError
|
||||
default:
|
||||
return s3err.ErrNone
|
||||
}
|
||||
}
|
||||
|
||||
// wormDeleteCondition returns the object-lock guards for a delete, or nil when
|
||||
// the bucket has no object lock. Legal hold always blocks. Retention blocks
|
||||
// while not elapsed; with governance bypass the retention guard is gated to
|
||||
// COMPLIANCE mode, so a governance-mode version becomes deletable while a
|
||||
// compliance-mode one stays protected — the filer decides from the version's
|
||||
// mode under the lock, so the gateway never has to read it.
|
||||
func wormDeleteCondition(worm, bypass bool) *filer_pb.WriteCondition {
|
||||
if !worm {
|
||||
return nil
|
||||
}
|
||||
retention := &filer_pb.WriteCondition_Clause{
|
||||
Kind: filer_pb.WriteCondition_IF_EXTENDED_TIME_ELAPSED,
|
||||
ExtKey: s3_constants.ExtRetentionUntilDateKey,
|
||||
}
|
||||
if bypass {
|
||||
retention.GateKey = s3_constants.ExtObjectLockModeKey
|
||||
retention.GateValue = s3_constants.RetentionModeCompliance
|
||||
}
|
||||
return &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{
|
||||
{Kind: filer_pb.WriteCondition_IF_EXTENDED_NOT_EQUAL, ExtKey: s3_constants.ExtLegalHoldKey, ExtValue: s3_constants.LegalHoldOn},
|
||||
retention,
|
||||
}}
|
||||
}
|
||||
|
||||
// routedDeleteSpecificVersion deletes one version off the distributed lock: in a
|
||||
// single transaction on the owner it recomputes the .versions pointer excluding
|
||||
// the version (repoint-before-delete, so a crash leaves a recoverable orphan
|
||||
// rather than a dangling pointer) and deletes the version file. lock_key is the
|
||||
// object (serializing the pointer recompute); for object-lock buckets the
|
||||
// condition gates the delete on the version's WORM guards evaluated on the owner.
|
||||
func (s3a *S3ApiServer) routedDeleteSpecificVersion(owner pb.ServerAddress, bucket, object, versionId string, worm, bypass bool) s3err.ErrorCode {
|
||||
versionFileName := s3a.getVersionFileName(versionId)
|
||||
versionsPath := s3a.toFilerPath(bucket, object+s3_constants.VersionsFolder)
|
||||
cond := wormDeleteCondition(worm, bypass)
|
||||
req := &filer_pb.ObjectTransactionRequest{
|
||||
LockKey: s3a.toFilerPath(bucket, object),
|
||||
RouteKey: s3a.objectRouteKey(bucket, object),
|
||||
ConditionKey: versionsPath + "/" + versionFileName,
|
||||
Condition: cond,
|
||||
Mutations: []*filer_pb.ObjectMutation{
|
||||
s3a.latestPointerRecompute(bucket, object, isNewFormatVersionId(versionId), versionFileName, false),
|
||||
{Type: filer_pb.ObjectMutation_DELETE, Directory: versionsPath, Name: versionFileName, IsDeleteData: true},
|
||||
},
|
||||
}
|
||||
resp, err := s3a.objectTxnOnFiler(owner, req)
|
||||
switch {
|
||||
case err != nil:
|
||||
glog.Errorf("routedDeleteSpecificVersion: %s/%s %s on %s: %v", bucket, object, versionId, owner, err)
|
||||
return s3err.ErrInternalError
|
||||
case resp.ErrorCode == filer_pb.FilerError_PRECONDITION_FAILED:
|
||||
// Legal hold or retention in force on the version.
|
||||
return s3err.ErrAccessDenied
|
||||
case resp.Error != "":
|
||||
glog.Errorf("routedDeleteSpecificVersion: %s/%s %s: %s", bucket, object, versionId, resp.Error)
|
||||
return s3err.ErrInternalError
|
||||
default:
|
||||
return s3err.ErrNone
|
||||
}
|
||||
}
|
||||
|
||||
// routedDeleteNullVersion deletes the null version (the regular object entry, not
|
||||
// a .versions file) off the distributed lock. There is no pointer to recompute;
|
||||
// the WORM guards, when present, gate the delete on the object entry itself
|
||||
// (condition defaults to lock_key).
|
||||
func (s3a *S3ApiServer) routedDeleteNullVersion(owner pb.ServerAddress, bucket, object string, worm, bypass bool) s3err.ErrorCode {
|
||||
fullpath := util.NewFullPath(s3a.bucketDir(bucket), object)
|
||||
dir, name := fullpath.DirAndName()
|
||||
resp, err := s3a.objectTxnOnFiler(owner, &filer_pb.ObjectTransactionRequest{
|
||||
LockKey: string(fullpath),
|
||||
RouteKey: s3a.objectRouteKey(bucket, object),
|
||||
Condition: wormDeleteCondition(worm, bypass),
|
||||
Mutations: []*filer_pb.ObjectMutation{
|
||||
{Type: filer_pb.ObjectMutation_DELETE, Directory: dir, Name: name, IsDeleteData: true},
|
||||
},
|
||||
})
|
||||
switch {
|
||||
case err != nil:
|
||||
glog.Errorf("routedDeleteNullVersion: %s/%s on %s: %v", bucket, object, owner, err)
|
||||
return s3err.ErrInternalError
|
||||
case resp.ErrorCode == filer_pb.FilerError_PRECONDITION_FAILED:
|
||||
return s3err.ErrAccessDenied
|
||||
case resp.Error != "":
|
||||
glog.Errorf("routedDeleteNullVersion: %s/%s: %s", bucket, object, resp.Error)
|
||||
return s3err.ErrInternalError
|
||||
default:
|
||||
return s3err.ErrNone
|
||||
}
|
||||
}
|
||||
|
||||
// versionedAfterCreate returns the putToFiler hook that finalizes a versioned
|
||||
// write: the routed RECOMPUTE_LATEST when the owner is known, else the existing
|
||||
// lock-free updateLatestVersionInDirectory.
|
||||
func (s3a *S3ApiServer) versionedAfterCreate(bucket, object, versionId, versionFileName string, useInvertedFormat bool) func(*filer_pb.Entry) s3err.ErrorCode {
|
||||
owner := s3a.objectWriteOwner(bucket, object)
|
||||
return func(versionEntry *filer_pb.Entry) s3err.ErrorCode {
|
||||
if owner != "" {
|
||||
return s3a.routedVersionedFinalize(owner, bucket, object, useInvertedFormat)
|
||||
}
|
||||
if err := s3a.updateLatestVersionInDirectory(bucket, object, versionId, versionFileName, versionEntry); err != nil {
|
||||
glog.Errorf("putVersionedObject: failed to update latest version in directory: %v", err)
|
||||
return s3err.ErrInternalError
|
||||
}
|
||||
return s3err.ErrNone
|
||||
}
|
||||
}
|
||||
@@ -242,13 +242,8 @@ func (s3a *S3ApiServer) createDeleteMarker(bucket, object string) (string, error
|
||||
},
|
||||
Extended: deleteMarkerExtended,
|
||||
}
|
||||
// Route the pointer flip to the owner filer when known (off the distributed
|
||||
// lock); RECOMPUTE_LATEST picks the just-written marker as the new latest.
|
||||
if owner := s3a.objectWriteOwner(bucket, cleanObject); owner != "" {
|
||||
if code := s3a.routedVersionedFinalize(owner, bucket, cleanObject, useInvertedFormat); code != s3err.ErrNone {
|
||||
return "", fmt.Errorf("createDeleteMarker: routed finalize failed for %s/%s: code %d", bucket, object, code)
|
||||
}
|
||||
} else if err = s3a.updateLatestVersionInDirectory(bucket, cleanObject, versionId, versionFileName, deleteMarkerEntry); err != nil {
|
||||
err = s3a.updateLatestVersionInDirectory(bucket, cleanObject, versionId, versionFileName, deleteMarkerEntry)
|
||||
if err != nil {
|
||||
glog.Errorf("createDeleteMarker: failed to update latest version in directory: %v", err)
|
||||
return "", fmt.Errorf("failed to update latest version in directory: %w", err)
|
||||
}
|
||||
|
||||
@@ -1,38 +0,0 @@
|
||||
package s3api
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
|
||||
"github.com/gorilla/mux"
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api/s3_constants"
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api/s3err"
|
||||
)
|
||||
|
||||
// validateRequestPath rejects requests whose captured {bucket}/{object} mux
|
||||
// vars would normalize to a parent-directory traversal once joined into a
|
||||
// filer path. The router runs with mux.NewRouter().SkipClean(true), so
|
||||
// segments like `..` survive routing; the filer's util.JoinPath later collapses
|
||||
// them via filepath.Join. Without this guard, `GET /bucket-A/../evil-bucket/k`
|
||||
// matches as bucket=bucket-A, object=../evil-bucket/k, the filer resolves the
|
||||
// read against evil-bucket, while IAM authorizes against bucket-A.
|
||||
func validateRequestPath(next http.Handler) http.Handler {
|
||||
return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
vars := mux.Vars(r)
|
||||
// When a var is in the matched route it must be non-empty: an empty
|
||||
// bucket would let downstream path.Join collapse it and let the object
|
||||
// key pick the bucket.
|
||||
if bucket, ok := vars["bucket"]; ok {
|
||||
if bucket == "" || !s3_constants.IsValidBucketName(bucket) {
|
||||
s3err.WriteErrorResponse(w, r, s3err.ErrInvalidRequest)
|
||||
return
|
||||
}
|
||||
}
|
||||
if object, ok := vars["object"]; ok {
|
||||
if object == "" || !s3_constants.IsValidObjectKey(object) {
|
||||
s3err.WriteErrorResponse(w, r, s3err.ErrInvalidRequest)
|
||||
return
|
||||
}
|
||||
}
|
||||
next.ServeHTTP(w, r)
|
||||
})
|
||||
}
|
||||
@@ -1,87 +0,0 @@
|
||||
package s3api
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"testing"
|
||||
|
||||
"github.com/gorilla/mux"
|
||||
)
|
||||
|
||||
func TestValidateRequestPath_RejectsTraversal(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
// rawPath is sent as the Request-URI; net/http.NewRequest does not
|
||||
// rewrite the path, so `..` segments survive into mux when the router
|
||||
// is built with SkipClean(true) — matching the production setup in
|
||||
// weed/command/s3.go.
|
||||
rawPath string
|
||||
wantCode int
|
||||
}{
|
||||
{"clean path passes", "/bucket-a/folder/file.txt", http.StatusOK},
|
||||
{"bucket only passes", "/bucket-a", http.StatusOK},
|
||||
{"trailing slash passes", "/bucket-a/folder/", http.StatusOK},
|
||||
|
||||
{"leading dotdot rejected", "/bucket-a/../evil-bucket/test.txt", http.StatusBadRequest},
|
||||
{"nested dotdot rejected", "/bucket-a/good/../evil/test.txt", http.StatusBadRequest},
|
||||
{"backslash dotdot rejected", "/bucket-a/..\\evil\\test.txt", http.StatusBadRequest},
|
||||
{"percent-encoded dotdot rejected", "/bucket-a/%2e%2e/evil/test.txt", http.StatusBadRequest},
|
||||
{"bare dot object rejected", "/bucket-a/./evil/test.txt", http.StatusBadRequest},
|
||||
{"dotdot bucket rejected", "/../buckets/evil", http.StatusBadRequest},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
router := mux.NewRouter().SkipClean(true)
|
||||
sub := router.PathPrefix("/{bucket}").Subrouter()
|
||||
sub.Use(validateRequestPath)
|
||||
handlerCalled := false
|
||||
pass := func(w http.ResponseWriter, r *http.Request) {
|
||||
handlerCalled = true
|
||||
w.WriteHeader(http.StatusOK)
|
||||
}
|
||||
// Mirror the production routes: /{bucket}/{object:(?s).+} for
|
||||
// object-scoped requests, bare /{bucket} for bucket-scoped ones.
|
||||
sub.Path("/{object:(?s).+}").HandlerFunc(pass)
|
||||
sub.Path("").HandlerFunc(pass)
|
||||
|
||||
req := httptest.NewRequest(http.MethodGet, tt.rawPath, nil)
|
||||
rr := httptest.NewRecorder()
|
||||
router.ServeHTTP(rr, req)
|
||||
|
||||
if rr.Code != tt.wantCode {
|
||||
t.Fatalf("path %q: got status %d, want %d (body=%q)", tt.rawPath, rr.Code, tt.wantCode, rr.Body.String())
|
||||
}
|
||||
if tt.wantCode == http.StatusBadRequest && handlerCalled {
|
||||
t.Fatalf("path %q: inner handler reached despite rejection", tt.rawPath)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// Defense-in-depth: a future router or middleware that captures the {bucket}
|
||||
// or {object} mux var as an empty string must still be rejected, even though
|
||||
// mux's default `[^/]+` regex won't match an empty segment from a real URL.
|
||||
func TestValidateRequestPath_RejectsEmptyCapturedVars(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
vars map[string]string
|
||||
}{
|
||||
{"empty bucket", map[string]string{"bucket": "", "object": "key"}},
|
||||
{"empty object", map[string]string{"bucket": "bucket-a", "object": ""}},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
handlerCalled := false
|
||||
h := validateRequestPath(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
handlerCalled = true
|
||||
}))
|
||||
req := mux.SetURLVars(httptest.NewRequest(http.MethodGet, "/", nil), tt.vars)
|
||||
rr := httptest.NewRecorder()
|
||||
h.ServeHTTP(rr, req)
|
||||
if handlerCalled {
|
||||
t.Fatalf("vars %v: inner handler reached despite empty capture", tt.vars)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -25,7 +25,6 @@ import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/iam/policy"
|
||||
"github.com/seaweedfs/seaweedfs/weed/iam/sts"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/s3_lifecycle_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/s3_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api/policy_engine"
|
||||
@@ -96,8 +95,6 @@ type S3ApiServer struct {
|
||||
stsHandlers *STSHandlers // STS HTTP handlers for AssumeRoleWithWebIdentity
|
||||
cipher bool // encrypt data on volume servers
|
||||
newObjectWriteLock func(bucket, object string) objectWriteLock
|
||||
// objectWriteLockClient resolves a key's owner filer for route-by-key.
|
||||
objectWriteLockClient *cluster.LockClient
|
||||
// Shared ReaderCache used by the S3 GET streaming path. It lives for the
|
||||
// lifetime of the server so that concurrent and repeat reads share a
|
||||
// single in-flight download per chunk, and so that no per-request
|
||||
@@ -155,8 +152,6 @@ func NewS3ApiServerWithStore(router *mux.Router, option *S3ApiServerOption, expl
|
||||
// Uses the battle-tested vidMap with filer-based lookups
|
||||
// Supports multiple filer addresses with automatic failover for high availability
|
||||
var filerClient *wdclient.FilerClient
|
||||
var masterClient *wdclient.MasterClient
|
||||
var objectWriteLockClient *cluster.LockClient
|
||||
if len(option.Masters) > 0 {
|
||||
// Enable filer discovery via master
|
||||
masterMap := make(map[string]pb.ServerAddress)
|
||||
@@ -167,22 +162,7 @@ func NewS3ApiServerWithStore(router *mux.Router, option *S3ApiServerOption, expl
|
||||
if clientHost == "0.0.0.0" || clientHost == "" {
|
||||
clientHost = util.DetectedHostAddress()
|
||||
}
|
||||
masterClient = wdclient.NewMasterClient(option.GrpcDialOption, option.FilerGroup, cluster.S3Type, pb.ServerAddress(util.JoinHostPort(clientHost, option.GrpcPort)), "", "", *pb.NewServiceDiscoveryFromMap(masterMap))
|
||||
// Build the object-write lock client and subscribe to the master's
|
||||
// lock-ring updates BEFORE starting the master loop, so the initial
|
||||
// LockRingUpdate sent on connect isn't dropped (the master only delivers
|
||||
// it once per connect). The masterClient already filters updates to this
|
||||
// server's filer group.
|
||||
if len(option.Filers) > 0 {
|
||||
objectWriteLockClient = cluster.NewLockClient(option.GrpcDialOption, option.Filers[0])
|
||||
masterClient.SetOnLockRingUpdateFn(func(update *master_pb.LockRingUpdate) {
|
||||
servers := make([]pb.ServerAddress, 0, len(update.Servers))
|
||||
for _, s := range update.Servers {
|
||||
servers = append(servers, pb.ServerAddress(s))
|
||||
}
|
||||
objectWriteLockClient.SetRing(servers, update.Version)
|
||||
})
|
||||
}
|
||||
masterClient := wdclient.NewMasterClient(option.GrpcDialOption, option.FilerGroup, cluster.S3Type, pb.ServerAddress(util.JoinHostPort(clientHost, option.GrpcPort)), "", "", *pb.NewServiceDiscoveryFromMap(masterMap))
|
||||
// Start the master client connection loop - required for GetMaster() to work
|
||||
go masterClient.KeepConnectedToMaster(context.Background())
|
||||
|
||||
@@ -283,14 +263,9 @@ func NewS3ApiServerWithStore(router *mux.Router, option *S3ApiServerOption, expl
|
||||
}
|
||||
|
||||
if len(option.Filers) > 0 {
|
||||
// Reuse the lock client built in the masters block (already subscribed to
|
||||
// ring updates); create a plain one when no masters are configured.
|
||||
if objectWriteLockClient == nil {
|
||||
objectWriteLockClient = cluster.NewLockClient(option.GrpcDialOption, option.Filers[0])
|
||||
}
|
||||
s3ApiServer.objectWriteLockClient = objectWriteLockClient
|
||||
objectWriteLockClient := cluster.NewLockClient(option.GrpcDialOption, option.Filers[0])
|
||||
s3ApiServer.newObjectWriteLock = func(bucket, object string) objectWriteLock {
|
||||
lockKey := objectWriteRouteKeyPrefix + s3ApiServer.toFilerPath(bucket, object)
|
||||
lockKey := fmt.Sprintf("s3.object.write:%s", s3ApiServer.toFilerPath(bucket, object))
|
||||
owner := fmt.Sprintf("s3api-%d", s3ApiServer.randomClientId)
|
||||
lock := objectWriteLockClient.NewShortLivedLock(lockKey, owner)
|
||||
if err := lock.AttemptToLock(objectWriteLockTTL); err != nil {
|
||||
@@ -735,11 +710,6 @@ func (s3a *S3ApiServer) registerRouter(router *mux.Router) {
|
||||
corsMiddleware := s3a.getCORSMiddleware()
|
||||
|
||||
for _, bucket := range routers {
|
||||
// Reject `..`/`.`/NUL in {bucket} or {object} vars before any handler
|
||||
// runs. SkipClean(true) keeps `..` in the matched path; the filer would
|
||||
// otherwise collapse it via filepath.Join and cross bucket boundaries.
|
||||
bucket.Use(validateRequestPath)
|
||||
|
||||
// Apply CORS middleware to bucket routers for automatic CORS header handling
|
||||
bucket.Use(corsMiddleware.Handler)
|
||||
|
||||
|
||||
+26
-4
@@ -110,10 +110,32 @@ func writeJson(w http.ResponseWriter, r *http.Request, httpStatus int, obj inter
|
||||
r.Method, r.URL.String(), httpStatus, string(bytes))
|
||||
}
|
||||
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
w.Header().Set("X-Content-Type-Options", "nosniff")
|
||||
w.WriteHeader(httpStatus)
|
||||
_, err = w.Write(bytes)
|
||||
callback := r.FormValue("callback")
|
||||
if callback == "" {
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
w.WriteHeader(httpStatus)
|
||||
if httpStatus == http.StatusNotModified {
|
||||
return
|
||||
}
|
||||
_, err = w.Write(bytes)
|
||||
} else {
|
||||
w.Header().Set("Content-Type", "application/javascript")
|
||||
w.WriteHeader(httpStatus)
|
||||
if httpStatus == http.StatusNotModified {
|
||||
return
|
||||
}
|
||||
if _, err = w.Write([]uint8(callback)); err != nil {
|
||||
return
|
||||
}
|
||||
if _, err = w.Write([]uint8("(")); err != nil {
|
||||
return
|
||||
}
|
||||
fmt.Fprint(w, string(bytes))
|
||||
if _, err = w.Write([]uint8(")")); err != nil {
|
||||
return
|
||||
}
|
||||
}
|
||||
|
||||
return
|
||||
}
|
||||
|
||||
|
||||
@@ -1,8 +1,6 @@
|
||||
package weed_server
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
@@ -31,33 +29,3 @@ func TestParseURL(t *testing.T) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestWriteJsonNoJSONP(t *testing.T) {
|
||||
// callback= must be ignored; response is always application/json with nosniff.
|
||||
cases := []string{"", "myCb", "<script>alert(1)</script>"}
|
||||
for _, cb := range cases {
|
||||
t.Run("callback="+cb, func(t *testing.T) {
|
||||
url := "/x"
|
||||
if cb != "" {
|
||||
url += "?callback=" + cb
|
||||
}
|
||||
r := httptest.NewRequest(http.MethodGet, url, nil)
|
||||
w := httptest.NewRecorder()
|
||||
if err := writeJson(w, r, http.StatusOK, map[string]string{"k": "v"}); err != nil {
|
||||
t.Fatalf("writeJson: %v", err)
|
||||
}
|
||||
if w.Code != http.StatusOK {
|
||||
t.Errorf("status: got %d want 200", w.Code)
|
||||
}
|
||||
if got := w.Header().Get("Content-Type"); got != "application/json" {
|
||||
t.Errorf("Content-Type: got %q want application/json", got)
|
||||
}
|
||||
if got := w.Header().Get("X-Content-Type-Options"); got != "nosniff" {
|
||||
t.Errorf("X-Content-Type-Options: got %q want nosniff", got)
|
||||
}
|
||||
if got := w.Body.String(); got != `{"k":"v"}` {
|
||||
t.Errorf("body: got %q want %q", got, `{"k":"v"}`)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
@@ -5,17 +5,15 @@ import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"math"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/cluster"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/filer"
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/operation"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/needle"
|
||||
@@ -188,38 +186,8 @@ func (fs *FilerServer) CreateEntry(ctx context.Context, req *filer_pb.CreateEntr
|
||||
newEntry.TtlSec = 0
|
||||
}
|
||||
|
||||
// Serialize concurrent mutations to the same path on this filer so the
|
||||
// read (existence/condition) and the write are atomic. Callers route a
|
||||
// key's writes to this owner filer, making this local lock sufficient.
|
||||
fullpath := newEntry.FullPath
|
||||
pathLock := fs.entryLockTable.AcquireLock("CreateEntry", fullpath, util.ExclusiveLock)
|
||||
defer fs.entryLockTable.ReleaseLock(fullpath, pathLock)
|
||||
|
||||
// Evaluate the optional precondition against the current entry while the
|
||||
// path lock is held, so the check and the write are atomic on this filer.
|
||||
// The fetched entry is then handed to CreateEntry below so it does not look
|
||||
// the same path up again under the lock.
|
||||
var existing *filer.Entry
|
||||
if conditionIsSet(req.Condition) {
|
||||
current, findErr := fs.filer.FindEntry(ctx, fullpath)
|
||||
if findErr != nil && findErr != filer_pb.ErrNotFound {
|
||||
return &filer_pb.CreateEntryResponse{}, fmt.Errorf("CreateEntry condition check %s: %w", fullpath, findErr)
|
||||
}
|
||||
if findErr == filer_pb.ErrNotFound {
|
||||
current = nil
|
||||
}
|
||||
if !writeConditionSatisfied(req.Condition, current) {
|
||||
glog.V(3).InfofCtx(ctx, "CreateEntry %s: precondition failed: %v", fullpath, req.Condition)
|
||||
return &filer_pb.CreateEntryResponse{
|
||||
Error: "precondition failed",
|
||||
ErrorCode: filer_pb.FilerError_PRECONDITION_FAILED,
|
||||
}, nil
|
||||
}
|
||||
existing = current
|
||||
}
|
||||
|
||||
ctx, eventSink := filer.WithMetadataEventSink(ctx)
|
||||
createErr := fs.filer.CreateEntry(ctx, newEntry, existing, req.OExcl, req.IsFromOtherCluster, req.Signatures, req.SkipCheckParentDirectory, so.MaxFileNameLength)
|
||||
createErr := fs.filer.CreateEntry(ctx, newEntry, req.OExcl, req.IsFromOtherCluster, req.Signatures, req.SkipCheckParentDirectory, so.MaxFileNameLength)
|
||||
|
||||
if createErr == nil {
|
||||
fs.filer.DeleteChunksNotRecursive(garbage)
|
||||
@@ -244,301 +212,6 @@ func (fs *FilerServer) CreateEntry(ctx context.Context, req *filer_pb.CreateEntr
|
||||
return
|
||||
}
|
||||
|
||||
// ObjectTransaction applies an ordered list of entry mutations atomically with
|
||||
// respect to other writers of the same object, by holding the per-path lock on
|
||||
// lock_key for the whole call. The optional condition is checked first, against
|
||||
// the entry at lock_key. This lets a caller describe a multi-entry object
|
||||
// operation (e.g. delete the null version + write a delete marker + flip the
|
||||
// latest pointer) as one request, replacing a distributed lock held across
|
||||
// several RPCs. Callers must route the object's writes to its owner filer for
|
||||
// the lock to be authoritative.
|
||||
func (fs *FilerServer) ObjectTransaction(ctx context.Context, req *filer_pb.ObjectTransactionRequest) (*filer_pb.ObjectTransactionResponse, error) {
|
||||
if req.LockKey == "" {
|
||||
return &filer_pb.ObjectTransactionResponse{Error: "lock_key is required"}, nil
|
||||
}
|
||||
|
||||
// Route-by-key: if this filer is not the ring owner of route_key, forward the
|
||||
// whole transaction to the owner so its per-path lock is the single
|
||||
// serialization point — even when the caller's ring view was stale. is_moved
|
||||
// bounds this to one hop: a forwarded transaction is applied locally, so two
|
||||
// filers that disagree on the owner during a ring change cannot loop.
|
||||
if req.RouteKey != "" && !req.IsMoved && fs.filer.Dlm != nil {
|
||||
if owner := fs.filer.Dlm.LockRing.GetPrimary(req.RouteKey); owner != "" && owner != fs.option.Host {
|
||||
// Rebuild rather than copy the request struct (it carries a mutex);
|
||||
// the pointer/slice fields are shared since the original is not mutated.
|
||||
forwarded := &filer_pb.ObjectTransactionRequest{
|
||||
LockKey: req.LockKey,
|
||||
Condition: req.Condition,
|
||||
Mutations: req.Mutations,
|
||||
IsFromOtherCluster: req.IsFromOtherCluster,
|
||||
Signatures: req.Signatures,
|
||||
ConditionKey: req.ConditionKey,
|
||||
RouteKey: req.RouteKey,
|
||||
IsMoved: true,
|
||||
}
|
||||
glog.V(2).InfofCtx(ctx, "ObjectTransaction %s: forwarding to owner %s", req.LockKey, owner)
|
||||
var resp *filer_pb.ObjectTransactionResponse
|
||||
err := pb.WithFilerClient(false, 0, owner, fs.grpcDialOption, func(client filer_pb.SeaweedFilerClient) error {
|
||||
var e error
|
||||
resp, e = client.ObjectTransaction(ctx, forwarded)
|
||||
return e
|
||||
})
|
||||
if err != nil {
|
||||
return &filer_pb.ObjectTransactionResponse{}, err
|
||||
}
|
||||
return resp, nil
|
||||
}
|
||||
}
|
||||
|
||||
lockPath := util.FullPath(req.LockKey)
|
||||
pathLock := fs.entryLockTable.AcquireLock("ObjectTransaction", lockPath, util.ExclusiveLock)
|
||||
defer fs.entryLockTable.ReleaseLock(lockPath, pathLock)
|
||||
|
||||
if conditionIsSet(req.Condition) {
|
||||
// The condition is evaluated against condition_key when set (e.g. a
|
||||
// version entry whose WORM guards gate the delete), while the lock stays
|
||||
// on lock_key (the object, serializing the pointer recompute).
|
||||
conditionPath := lockPath
|
||||
if req.ConditionKey != "" {
|
||||
conditionPath = util.FullPath(req.ConditionKey)
|
||||
}
|
||||
current, findErr := fs.filer.FindEntry(ctx, conditionPath)
|
||||
if findErr != nil && findErr != filer_pb.ErrNotFound {
|
||||
return &filer_pb.ObjectTransactionResponse{}, fmt.Errorf("ObjectTransaction condition %s: %w", conditionPath, findErr)
|
||||
}
|
||||
if findErr == filer_pb.ErrNotFound {
|
||||
current = nil
|
||||
}
|
||||
if !writeConditionSatisfied(req.Condition, current) {
|
||||
glog.V(3).InfofCtx(ctx, "ObjectTransaction %s: precondition failed", conditionPath)
|
||||
return &filer_pb.ObjectTransactionResponse{
|
||||
Error: "precondition failed",
|
||||
ErrorCode: filer_pb.FilerError_PRECONDITION_FAILED,
|
||||
}, nil
|
||||
}
|
||||
}
|
||||
|
||||
for i, m := range req.Mutations {
|
||||
if err := fs.applyObjectMutation(ctx, m, req.IsFromOtherCluster, req.Signatures); err != nil {
|
||||
glog.V(2).InfofCtx(ctx, "ObjectTransaction %s mutation %d (%v): %v", lockPath, i, m.Type, err)
|
||||
return &filer_pb.ObjectTransactionResponse{Error: fmt.Sprintf("mutation %d: %v", i, err)}, nil
|
||||
}
|
||||
}
|
||||
|
||||
return &filer_pb.ObjectTransactionResponse{}, nil
|
||||
}
|
||||
|
||||
// ObjectTransactionBatch applies several object transactions in one round trip,
|
||||
// each under its own per-path lock and independent of the others. A failed
|
||||
// transaction (precondition or mutation error) is reported in its own response
|
||||
// without aborting the rest, matching S3 multi-object semantics where each key
|
||||
// succeeds or fails on its own.
|
||||
func (fs *FilerServer) ObjectTransactionBatch(ctx context.Context, req *filer_pb.ObjectTransactionBatchRequest) (*filer_pb.ObjectTransactionBatchResponse, error) {
|
||||
if req == nil {
|
||||
return nil, status.Error(codes.InvalidArgument, "request is required")
|
||||
}
|
||||
resp := &filer_pb.ObjectTransactionBatchResponse{
|
||||
Responses: make([]*filer_pb.ObjectTransactionResponse, 0, len(req.Transactions)),
|
||||
}
|
||||
for _, txn := range req.Transactions {
|
||||
// Stop early if the caller went away; the request still holds the
|
||||
// unprocessed transactions, so it is retried rather than lost.
|
||||
if err := ctx.Err(); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if txn == nil {
|
||||
resp.Responses = append(resp.Responses, &filer_pb.ObjectTransactionResponse{Error: "nil transaction"})
|
||||
continue
|
||||
}
|
||||
one, err := fs.ObjectTransaction(ctx, txn)
|
||||
if err != nil {
|
||||
// A transport-level error on one transaction is surfaced as that
|
||||
// transaction's error; the batch RPC itself still succeeds.
|
||||
one = &filer_pb.ObjectTransactionResponse{Error: err.Error()}
|
||||
}
|
||||
resp.Responses = append(resp.Responses, one)
|
||||
}
|
||||
return resp, nil
|
||||
}
|
||||
|
||||
// applyObjectMutation applies a single mutation while the transaction's path
|
||||
// lock is held. PUT entries are expected to be fully prepared by the caller
|
||||
// (chunks resolved); mutations here are metadata-scoped. A DELETE of an absent
|
||||
// entry and a PATCH of an absent entry are no-ops, so transactions are
|
||||
// idempotent on replay.
|
||||
func (fs *FilerServer) applyObjectMutation(ctx context.Context, m *filer_pb.ObjectMutation, fromOtherCluster bool, signatures []int32) error {
|
||||
switch m.Type {
|
||||
case filer_pb.ObjectMutation_PUT:
|
||||
if m.Entry == nil {
|
||||
return fmt.Errorf("PUT requires an entry")
|
||||
}
|
||||
newEntry := filer.FromPbEntry(m.Directory, m.Entry)
|
||||
return fs.filer.CreateEntry(ctx, newEntry, nil, false, fromOtherCluster, signatures, false, fs.filer.MaxFilenameLength)
|
||||
|
||||
case filer_pb.ObjectMutation_DELETE:
|
||||
fullpath := util.NewFullPath(m.Directory, m.Name)
|
||||
err := fs.filer.DeleteEntryMetaAndData(ctx, fullpath, m.IsRecursive, false, m.IsDeleteData, fromOtherCluster, signatures, 0)
|
||||
if err == filer_pb.ErrNotFound {
|
||||
return nil
|
||||
}
|
||||
return err
|
||||
|
||||
case filer_pb.ObjectMutation_PATCH_EXTENDED:
|
||||
fullpath := util.NewFullPath(m.Directory, m.Name)
|
||||
oldEntry, err := fs.filer.FindEntry(ctx, fullpath)
|
||||
if err == filer_pb.ErrNotFound {
|
||||
return nil
|
||||
}
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
// Patch a copy so oldEntry still reflects the pre-update state for the
|
||||
// metadata notification's diff.
|
||||
newEntry := oldEntry.ShallowClone()
|
||||
newEntry.Extended = make(map[string][]byte, len(oldEntry.Extended))
|
||||
for k, v := range oldEntry.Extended {
|
||||
newEntry.Extended[k] = v
|
||||
}
|
||||
for k, v := range m.SetExtended {
|
||||
newEntry.Extended[k] = v
|
||||
}
|
||||
for _, k := range m.DeleteExtended {
|
||||
delete(newEntry.Extended, k)
|
||||
}
|
||||
if m.SetContent {
|
||||
newEntry.Content = m.Content
|
||||
// Keep FileSize consistent with content for files; some stores and
|
||||
// tools read the attribute directly. Directories carry no file size.
|
||||
if !newEntry.IsDirectory() {
|
||||
newEntry.FileSize = uint64(len(m.Content))
|
||||
}
|
||||
}
|
||||
if m.TouchMtime {
|
||||
newEntry.Attr.Mtime = time.Now()
|
||||
}
|
||||
if err := fs.filer.UpdateEntry(ctx, oldEntry, newEntry); err != nil {
|
||||
return err
|
||||
}
|
||||
// Emit the metadata event so the update replicates and subscribers see it,
|
||||
// matching the UpdateEntry handler.
|
||||
fs.filer.NotifyUpdateEvent(ctx, oldEntry, newEntry, true, fromOtherCluster, signatures)
|
||||
return nil
|
||||
|
||||
case filer_pb.ObjectMutation_RECOMPUTE_LATEST:
|
||||
return fs.applyRecomputeLatest(ctx, m)
|
||||
|
||||
default:
|
||||
return fmt.Errorf("unknown mutation type %v", m.Type)
|
||||
}
|
||||
}
|
||||
|
||||
// applyRecomputeLatest re-derives the pointer entry (m.Directory/m.Name) from the
|
||||
// current contents of recompute.scan_dir, under the transaction's lock. It is
|
||||
// mechanical: pick the child that sorts last (descending) or first by name, copy
|
||||
// the mapped extended keys from it into the pointer, and store its name under
|
||||
// name_to_key. When the scanned directory is empty the pointer keys are cleared.
|
||||
// The caller, which knows the versioning scheme, supplies the direction and the
|
||||
// key mappings. A missing pointer entry is a no-op (idempotent on replay).
|
||||
func (fs *FilerServer) applyRecomputeLatest(ctx context.Context, m *filer_pb.ObjectMutation) error {
|
||||
rc := m.Recompute
|
||||
if rc == nil {
|
||||
return fmt.Errorf("RECOMPUTE_LATEST requires recompute parameters")
|
||||
}
|
||||
|
||||
pointer, err := fs.filer.FindEntry(ctx, util.NewFullPath(m.Directory, m.Name))
|
||||
if err == filer_pb.ErrNotFound {
|
||||
return nil
|
||||
}
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if pointer.Extended == nil {
|
||||
pointer.Extended = make(map[string][]byte)
|
||||
}
|
||||
|
||||
// Remember the prior chosen child so it can be demoted once the pointer moves.
|
||||
var priorName string
|
||||
if rc.NameToKey != "" {
|
||||
priorName = string(pointer.Extended[rc.NameToKey])
|
||||
}
|
||||
|
||||
// The store streams entries ascending by name. For the lowest-name pick we
|
||||
// only need the first entry, so cap the listing at one; for the highest-name
|
||||
// pick we must scan all and keep the last (the store has no reverse order).
|
||||
// With exclude_name set the first child may be the excluded one, so the cap
|
||||
// is lifted to find the first non-excluded entry.
|
||||
limit := int64(math.MaxInt32)
|
||||
if !rc.Descending && rc.ExcludeName == "" {
|
||||
limit = 1
|
||||
}
|
||||
var chosen *filer.Entry
|
||||
_, listErr := fs.filer.StreamListDirectoryEntries(ctx, util.FullPath(rc.ScanDir), "", false, limit, "", "", "", func(entry *filer.Entry) (bool, error) {
|
||||
if rc.ExcludeName != "" && entry.Name() == rc.ExcludeName {
|
||||
return true, nil
|
||||
}
|
||||
chosen = entry
|
||||
return rc.Descending, nil
|
||||
})
|
||||
if listErr != nil {
|
||||
return listErr
|
||||
}
|
||||
|
||||
cleared := []string{rc.NameToKey, rc.SizeToKey, rc.MtimeToKey}
|
||||
if chosen == nil {
|
||||
for pointerKey := range rc.CopyExtended {
|
||||
delete(pointer.Extended, pointerKey)
|
||||
}
|
||||
for _, k := range cleared {
|
||||
if k != "" {
|
||||
delete(pointer.Extended, k)
|
||||
}
|
||||
}
|
||||
} else {
|
||||
for pointerKey, sourceKey := range rc.CopyExtended {
|
||||
if v, ok := chosen.Extended[sourceKey]; ok {
|
||||
pointer.Extended[pointerKey] = v
|
||||
} else {
|
||||
delete(pointer.Extended, pointerKey)
|
||||
}
|
||||
}
|
||||
if rc.NameToKey != "" {
|
||||
pointer.Extended[rc.NameToKey] = []byte(chosen.Name())
|
||||
}
|
||||
if rc.SizeToKey != "" {
|
||||
pointer.Extended[rc.SizeToKey] = []byte(strconv.FormatUint(chosen.FileSize, 10))
|
||||
}
|
||||
if rc.MtimeToKey != "" {
|
||||
pointer.Extended[rc.MtimeToKey] = []byte(strconv.FormatInt(chosen.Mtime.Unix(), 10))
|
||||
}
|
||||
}
|
||||
|
||||
if err := fs.filer.UpdateEntry(ctx, pointer, pointer); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
// Stamp the displaced prior child (e.g. NoncurrentSinceNs for lifecycle).
|
||||
newName := ""
|
||||
if chosen != nil {
|
||||
newName = chosen.Name()
|
||||
}
|
||||
if rc.DemoteKey != "" && priorName != "" && priorName != newName {
|
||||
priorEntry, perr := fs.filer.FindEntry(ctx, util.NewFullPath(rc.ScanDir, priorName))
|
||||
if perr == filer_pb.ErrNotFound {
|
||||
return nil
|
||||
}
|
||||
if perr != nil {
|
||||
return perr
|
||||
}
|
||||
if priorEntry.Extended == nil {
|
||||
priorEntry.Extended = make(map[string][]byte)
|
||||
}
|
||||
priorEntry.Extended[rc.DemoteKey] = rc.DemoteValue
|
||||
return fs.filer.UpdateEntry(ctx, priorEntry, priorEntry)
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
func (fs *FilerServer) UpdateEntry(ctx context.Context, req *filer_pb.UpdateEntryRequest) (*filer_pb.UpdateEntryResponse, error) {
|
||||
|
||||
glog.V(4).InfofCtx(ctx, "UpdateEntry %v", req)
|
||||
@@ -693,7 +366,7 @@ func (fs *FilerServer) AppendToEntry(ctx context.Context, req *filer_pb.AppendTo
|
||||
glog.V(0).InfofCtx(ctx, "MaybeManifestize: %v", err)
|
||||
}
|
||||
|
||||
err = fs.filer.CreateEntry(context.Background(), entry, nil, false, false, nil, false, fs.filer.MaxFilenameLength)
|
||||
err = fs.filer.CreateEntry(context.Background(), entry, false, false, nil, false, fs.filer.MaxFilenameLength)
|
||||
|
||||
return &filer_pb.AppendToEntryResponse{}, err
|
||||
}
|
||||
|
||||
@@ -1,136 +0,0 @@
|
||||
package weed_server
|
||||
|
||||
import (
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/filer"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api/s3_constants"
|
||||
)
|
||||
|
||||
// conditionIsSet reports whether a condition asks for any check at all.
|
||||
func conditionIsSet(cond *filer_pb.WriteCondition) bool {
|
||||
return cond != nil && len(cond.Clauses) > 0
|
||||
}
|
||||
|
||||
// writeConditionSatisfied reports whether the precondition holds against the
|
||||
// current entry (nil if absent), evaluated under the path lock. Every clause
|
||||
// must hold (logical AND).
|
||||
func writeConditionSatisfied(cond *filer_pb.WriteCondition, current *filer.Entry) bool {
|
||||
for _, c := range cond.Clauses {
|
||||
if !clauseSatisfied(c, current) {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// clauseSatisfied evaluates one primitive against the current entry. For the
|
||||
// ETag kinds, etags is a set: IF_ETAG_MATCH holds when the current ETag equals
|
||||
// any member, IF_ETAG_NOT_MATCH when it equals none. The IF_EXTENDED_* kinds are
|
||||
// generic guards on an extended attribute used to enforce object-lock (legal
|
||||
// hold and retention) without S3 knowledge in the filer.
|
||||
func clauseSatisfied(c *filer_pb.WriteCondition_Clause, current *filer.Entry) bool {
|
||||
exists := current != nil
|
||||
switch c.Kind {
|
||||
case filer_pb.WriteCondition_NONE:
|
||||
return true
|
||||
case filer_pb.WriteCondition_IF_NOT_EXISTS:
|
||||
return !exists
|
||||
case filer_pb.WriteCondition_IF_EXISTS:
|
||||
return exists
|
||||
case filer_pb.WriteCondition_IF_ETAG_MATCH:
|
||||
return exists && etagInSet(storedEntryETag(current), c.Etags, c.AllowWeak)
|
||||
case filer_pb.WriteCondition_IF_ETAG_NOT_MATCH:
|
||||
return !exists || !etagInSet(storedEntryETag(current), c.Etags, c.AllowWeak)
|
||||
case filer_pb.WriteCondition_IF_UNMODIFIED_SINCE:
|
||||
return !exists || current.Attr.Mtime.Unix() <= c.UnixTime
|
||||
case filer_pb.WriteCondition_IF_MODIFIED_SINCE:
|
||||
return !exists || current.Attr.Mtime.Unix() > c.UnixTime
|
||||
case filer_pb.WriteCondition_IF_EXTENDED_NOT_EQUAL:
|
||||
if !exists {
|
||||
return true
|
||||
}
|
||||
v, ok := current.Extended[c.ExtKey]
|
||||
return !ok || string(v) != c.ExtValue
|
||||
case filer_pb.WriteCondition_IF_EXTENDED_TIME_ELAPSED:
|
||||
if !exists {
|
||||
return true
|
||||
}
|
||||
// An optional gate scopes the guard: when gate_key is set, the time check
|
||||
// only applies if extended[gate_key] == gate_value. This lets the gateway
|
||||
// express governance bypass (enforce retention only for COMPLIANCE mode)
|
||||
// without reading the entry — the filer decides under the lock.
|
||||
if c.GateKey != "" {
|
||||
gv, gok := current.Extended[c.GateKey]
|
||||
if !gok || string(gv) != c.GateValue {
|
||||
return true
|
||||
}
|
||||
}
|
||||
v, ok := current.Extended[c.ExtKey]
|
||||
if !ok {
|
||||
return true
|
||||
}
|
||||
deadline, err := strconv.ParseInt(strings.TrimSpace(string(v)), 10, 64)
|
||||
if err != nil {
|
||||
// An unparseable retention deadline is treated as still in force, so
|
||||
// a malformed attribute fails safe (write blocked) rather than open.
|
||||
return false
|
||||
}
|
||||
return deadline <= time.Now().Unix()
|
||||
default:
|
||||
// An unrecognized clause kind (e.g. from a newer client) must not be
|
||||
// treated as satisfied, which would silently bypass the guard. Fail
|
||||
// closed so the write is blocked rather than slipping through.
|
||||
return false
|
||||
}
|
||||
}
|
||||
|
||||
// etagInSet reports whether stored matches any candidate. A strong comparison
|
||||
// (allowWeak false) treats a weak ETag as never equal; a weak comparison
|
||||
// ignores the W/ marker on both sides.
|
||||
func etagInSet(stored string, candidates []string, allowWeak bool) bool {
|
||||
for _, c := range candidates {
|
||||
if etagEqual(stored, c, allowWeak) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func etagEqual(stored, expected string, allowWeak bool) bool {
|
||||
sv, sWeak := canonicalETag(stored)
|
||||
ev, eWeak := canonicalETag(expected)
|
||||
// RFC 7232 strong comparison: a weak ETag on either side never matches.
|
||||
if !allowWeak && (sWeak || eWeak) {
|
||||
return false
|
||||
}
|
||||
return sv == ev
|
||||
}
|
||||
|
||||
// canonicalETag splits off the weak (W/) marker before stripping quotes, so a
|
||||
// weak ETag like W/"abc" yields ("abc", true).
|
||||
func canonicalETag(etag string) (value string, weak bool) {
|
||||
etag = strings.TrimSpace(etag)
|
||||
if strings.HasPrefix(etag, "W/") {
|
||||
return strings.Trim(etag[len("W/"):], `"`), true
|
||||
}
|
||||
return strings.Trim(etag, `"`), false
|
||||
}
|
||||
|
||||
// storedEntryETag mirrors the S3 gateway's ETag precedence (the stored
|
||||
// Seaweed ETag extended attribute, then the chunk/Md5 fallback) so conditional
|
||||
// comparisons match what the gateway computes, without coupling the filer to
|
||||
// S3 request handling.
|
||||
func storedEntryETag(entry *filer.Entry) string {
|
||||
if v, ok := entry.Extended[s3_constants.ExtETagKey]; ok && len(v) > 0 {
|
||||
return normalizeETag(string(v))
|
||||
}
|
||||
return normalizeETag(filer.ETagEntry(entry))
|
||||
}
|
||||
|
||||
func normalizeETag(etag string) string {
|
||||
return strings.Trim(etag, `"`)
|
||||
}
|
||||
@@ -1,286 +0,0 @@
|
||||
package weed_server
|
||||
|
||||
import (
|
||||
"context"
|
||||
"strconv"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/filer"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api/s3_constants"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
)
|
||||
|
||||
func entryWithETag(etag string, mtime time.Time) *filer.Entry {
|
||||
return &filer.Entry{
|
||||
FullPath: "/test/obj",
|
||||
Attr: filer.Attr{Mtime: mtime},
|
||||
Extended: map[string][]byte{s3_constants.ExtETagKey: []byte(etag)},
|
||||
}
|
||||
}
|
||||
|
||||
// one wraps a single clause into a condition.
|
||||
func one(c *filer_pb.WriteCondition_Clause) *filer_pb.WriteCondition {
|
||||
return &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{c}}
|
||||
}
|
||||
|
||||
func TestWriteConditionSatisfied(t *testing.T) {
|
||||
base := time.Unix(1700000000, 0)
|
||||
present := entryWithETag("abc", base)
|
||||
|
||||
cases := []struct {
|
||||
name string
|
||||
cond *filer_pb.WriteCondition
|
||||
cur *filer.Entry
|
||||
want bool
|
||||
}{
|
||||
{"empty-absent", &filer_pb.WriteCondition{}, nil, true},
|
||||
{"ifnotexists-absent", one(&filer_pb.WriteCondition_Clause{Kind: filer_pb.WriteCondition_IF_NOT_EXISTS}), nil, true},
|
||||
{"ifnotexists-present", one(&filer_pb.WriteCondition_Clause{Kind: filer_pb.WriteCondition_IF_NOT_EXISTS}), present, false},
|
||||
{"ifexists-absent", one(&filer_pb.WriteCondition_Clause{Kind: filer_pb.WriteCondition_IF_EXISTS}), nil, false},
|
||||
{"ifexists-present", one(&filer_pb.WriteCondition_Clause{Kind: filer_pb.WriteCondition_IF_EXISTS}), present, true},
|
||||
{"etagmatch-hit", one(&filer_pb.WriteCondition_Clause{Kind: filer_pb.WriteCondition_IF_ETAG_MATCH, Etags: []string{`"abc"`}}), present, true},
|
||||
{"etagmatch-miss", one(&filer_pb.WriteCondition_Clause{Kind: filer_pb.WriteCondition_IF_ETAG_MATCH, Etags: []string{`"zzz"`}}), present, false},
|
||||
{"etagmatch-absent", one(&filer_pb.WriteCondition_Clause{Kind: filer_pb.WriteCondition_IF_ETAG_MATCH, Etags: []string{`"abc"`}}), nil, false},
|
||||
{"etagnotmatch-hit", one(&filer_pb.WriteCondition_Clause{Kind: filer_pb.WriteCondition_IF_ETAG_NOT_MATCH, Etags: []string{`"abc"`}}), present, false},
|
||||
{"etagnotmatch-miss", one(&filer_pb.WriteCondition_Clause{Kind: filer_pb.WriteCondition_IF_ETAG_NOT_MATCH, Etags: []string{`"zzz"`}}), present, true},
|
||||
{"etagnotmatch-absent", one(&filer_pb.WriteCondition_Clause{Kind: filer_pb.WriteCondition_IF_ETAG_NOT_MATCH, Etags: []string{`"abc"`}}), nil, true},
|
||||
{"unmodsince-ok", one(&filer_pb.WriteCondition_Clause{Kind: filer_pb.WriteCondition_IF_UNMODIFIED_SINCE, UnixTime: base.Unix()}), present, true},
|
||||
{"unmodsince-fail", one(&filer_pb.WriteCondition_Clause{Kind: filer_pb.WriteCondition_IF_UNMODIFIED_SINCE, UnixTime: base.Unix() - 1}), present, false},
|
||||
{"modsince-ok", one(&filer_pb.WriteCondition_Clause{Kind: filer_pb.WriteCondition_IF_MODIFIED_SINCE, UnixTime: base.Unix() - 1}), present, true},
|
||||
{"modsince-fail", one(&filer_pb.WriteCondition_Clause{Kind: filer_pb.WriteCondition_IF_MODIFIED_SINCE, UnixTime: base.Unix()}), present, false},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
if got := writeConditionSatisfied(tc.cond, tc.cur); got != tc.want {
|
||||
t.Errorf("%s: got %v want %v", tc.name, got, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestWriteConditionClauses(t *testing.T) {
|
||||
base := time.Unix(1700000000, 0)
|
||||
present := entryWithETag("abc", base)
|
||||
|
||||
matchAny := func(etags ...string) *filer_pb.WriteCondition {
|
||||
return &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{
|
||||
{Kind: filer_pb.WriteCondition_IF_ETAG_MATCH, Etags: etags},
|
||||
}}
|
||||
}
|
||||
noneOf := func(etags ...string) *filer_pb.WriteCondition {
|
||||
return &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{
|
||||
{Kind: filer_pb.WriteCondition_IF_ETAG_NOT_MATCH, Etags: etags},
|
||||
}}
|
||||
}
|
||||
|
||||
cases := []struct {
|
||||
name string
|
||||
cond *filer_pb.WriteCondition
|
||||
cur *filer.Entry
|
||||
want bool
|
||||
}{
|
||||
{"set-match-hit", matchAny(`"x"`, `"abc"`, `"y"`), present, true},
|
||||
{"set-match-miss", matchAny(`"x"`, `"y"`), present, false},
|
||||
{"set-none-clean", noneOf(`"x"`, `"y"`), present, true},
|
||||
{"set-none-hit", noneOf(`"abc"`, `"y"`), present, false},
|
||||
// A weak request ETag never matches under strong comparison.
|
||||
{"weak-strong-fails", matchAny(`W/"abc"`), present, false},
|
||||
// allow_weak compares ignoring the W/ marker.
|
||||
{"weak-allowed", &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{
|
||||
{Kind: filer_pb.WriteCondition_IF_ETAG_MATCH, Etags: []string{`W/"abc"`}, AllowWeak: true},
|
||||
}}, present, true},
|
||||
// Compound clauses are ANDed.
|
||||
{"compound-ok", &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{
|
||||
{Kind: filer_pb.WriteCondition_IF_ETAG_MATCH, Etags: []string{`"abc"`}},
|
||||
{Kind: filer_pb.WriteCondition_IF_UNMODIFIED_SINCE, UnixTime: base.Unix()},
|
||||
}}, present, true},
|
||||
{"compound-second-fails", &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{
|
||||
{Kind: filer_pb.WriteCondition_IF_ETAG_MATCH, Etags: []string{`"abc"`}},
|
||||
{Kind: filer_pb.WriteCondition_IF_UNMODIFIED_SINCE, UnixTime: base.Unix() - 1},
|
||||
}}, present, false},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
if got := writeConditionSatisfied(tc.cond, tc.cur); got != tc.want {
|
||||
t.Errorf("%s: got %v want %v", tc.name, got, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The generic IF_EXTENDED_* guards express object-lock without S3 knowledge in
|
||||
// the filer: a legal hold (IF_EXTENDED_NOT_EQUAL) and a retention deadline
|
||||
// (IF_EXTENDED_TIME_ELAPSED).
|
||||
func TestWriteConditionObjectLockGuards(t *testing.T) {
|
||||
now := time.Now()
|
||||
withExt := func(ext map[string]string) *filer.Entry {
|
||||
e := &filer.Entry{FullPath: "/test/obj", Attr: filer.Attr{Mtime: now}, Extended: map[string][]byte{}}
|
||||
for k, v := range ext {
|
||||
e.Extended[k] = []byte(v)
|
||||
}
|
||||
return e
|
||||
}
|
||||
legalHold := &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{
|
||||
{Kind: filer_pb.WriteCondition_IF_EXTENDED_NOT_EQUAL, ExtKey: "lock-hold", ExtValue: "ON"},
|
||||
}}
|
||||
retention := &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{
|
||||
{Kind: filer_pb.WriteCondition_IF_EXTENDED_TIME_ELAPSED, ExtKey: "retain-until"},
|
||||
}}
|
||||
// Governance bypass: the retention guard is gated to COMPLIANCE mode, so a
|
||||
// governance-mode (or unmoded) entry is deletable while compliance stays
|
||||
// protected.
|
||||
gatedRetention := &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{
|
||||
{Kind: filer_pb.WriteCondition_IF_EXTENDED_TIME_ELAPSED, ExtKey: "retain-until", GateKey: "lock-mode", GateValue: "COMPLIANCE"},
|
||||
}}
|
||||
future := strconv.FormatInt(now.Add(time.Hour).Unix(), 10)
|
||||
past := strconv.FormatInt(now.Add(-time.Hour).Unix(), 10)
|
||||
|
||||
cases := []struct {
|
||||
name string
|
||||
cond *filer_pb.WriteCondition
|
||||
cur *filer.Entry
|
||||
want bool
|
||||
}{
|
||||
{"hold-on-blocks", legalHold, withExt(map[string]string{"lock-hold": "ON"}), false},
|
||||
{"hold-off-allows", legalHold, withExt(map[string]string{"lock-hold": "OFF"}), true},
|
||||
{"hold-absent-allows", legalHold, withExt(nil), true},
|
||||
{"hold-on-new-object", legalHold, nil, true}, // nothing to protect yet
|
||||
{"retain-future-blocks", retention, withExt(map[string]string{"retain-until": future}), false},
|
||||
{"retain-past-allows", retention, withExt(map[string]string{"retain-until": past}), true},
|
||||
{"retain-absent-allows", retention, withExt(nil), true},
|
||||
{"retain-malformed-blocks", retention, withExt(map[string]string{"retain-until": "soon"}), false},
|
||||
// Governance bypass (gated to COMPLIANCE): compliance still blocks, but
|
||||
// governance and unmoded entries become deletable despite future retention.
|
||||
{"bypass-compliance-blocks", gatedRetention, withExt(map[string]string{"retain-until": future, "lock-mode": "COMPLIANCE"}), false},
|
||||
{"bypass-governance-allows", gatedRetention, withExt(map[string]string{"retain-until": future, "lock-mode": "GOVERNANCE"}), true},
|
||||
{"bypass-no-mode-allows", gatedRetention, withExt(map[string]string{"retain-until": future}), true},
|
||||
// Composed WORM guard: legal hold AND retention, both clear -> allowed.
|
||||
{"worm-both-clear", &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{
|
||||
{Kind: filer_pb.WriteCondition_IF_EXTENDED_NOT_EQUAL, ExtKey: "lock-hold", ExtValue: "ON"},
|
||||
{Kind: filer_pb.WriteCondition_IF_EXTENDED_TIME_ELAPSED, ExtKey: "retain-until"},
|
||||
}}, withExt(map[string]string{"lock-hold": "OFF", "retain-until": past}), true},
|
||||
// Either guard tripping blocks the whole WORM condition.
|
||||
{"worm-hold-trips", &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{
|
||||
{Kind: filer_pb.WriteCondition_IF_EXTENDED_NOT_EQUAL, ExtKey: "lock-hold", ExtValue: "ON"},
|
||||
{Kind: filer_pb.WriteCondition_IF_EXTENDED_TIME_ELAPSED, ExtKey: "retain-until"},
|
||||
}}, withExt(map[string]string{"lock-hold": "ON", "retain-until": past}), false},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
if got := writeConditionSatisfied(tc.cond, tc.cur); got != tc.want {
|
||||
t.Errorf("%s: got %v want %v", tc.name, got, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// An unrecognized clause kind (e.g. from a newer client) fails closed, so a
|
||||
// guard can't be silently bypassed by an older filer. NONE stays a no-op.
|
||||
func TestWriteConditionUnknownKindFailsClosed(t *testing.T) {
|
||||
present := entryWithETag("abc", time.Now())
|
||||
unknown := &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{
|
||||
{Kind: filer_pb.WriteCondition_Kind(9999)},
|
||||
}}
|
||||
if writeConditionSatisfied(unknown, present) {
|
||||
t.Error("unknown clause kind must not be satisfied (fail closed) for an existing entry")
|
||||
}
|
||||
if writeConditionSatisfied(unknown, nil) {
|
||||
t.Error("unknown clause kind must not be satisfied (fail closed) for an absent entry")
|
||||
}
|
||||
none := &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{
|
||||
{Kind: filer_pb.WriteCondition_NONE},
|
||||
}}
|
||||
if !writeConditionSatisfied(none, present) {
|
||||
t.Error("a NONE clause must be satisfied (no-op)")
|
||||
}
|
||||
}
|
||||
|
||||
// storedEntryETag prefers the stored Seaweed ETag attribute and falls back to
|
||||
// the Md5-derived ETag, matching the S3 gateway.
|
||||
func TestStoredEntryETag(t *testing.T) {
|
||||
withExt := entryWithETag("explicit", time.Unix(0, 0))
|
||||
if got := storedEntryETag(withExt); got != "explicit" {
|
||||
t.Errorf("extended etag: got %q", got)
|
||||
}
|
||||
md5Only := &filer.Entry{Attr: filer.Attr{Md5: []byte{0xab, 0xcd}}}
|
||||
if got := storedEntryETag(md5Only); got != "abcd" {
|
||||
t.Errorf("md5 fallback: got %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
// The CreateEntry handler enforces the precondition atomically: a matching
|
||||
// If-Match overwrites, a non-matching one returns PRECONDITION_FAILED.
|
||||
func TestCreateEntryConditionEnforced(t *testing.T) {
|
||||
store := newRenameTestStore()
|
||||
store.entries["/test/obj"] = &filer.Entry{
|
||||
FullPath: "/test/obj",
|
||||
Attr: filer.Attr{Inode: 1, Mtime: time.Unix(1700000000, 0)},
|
||||
Extended: map[string][]byte{s3_constants.ExtETagKey: []byte("abc")},
|
||||
}
|
||||
f := newRenameTestFiler(store)
|
||||
f.DirBucketsPath = "/buckets"
|
||||
fs := &FilerServer{filer: f, option: &FilerOption{}, entryLockTable: util.NewLockTable[util.FullPath]()}
|
||||
|
||||
req := func(etag string) *filer_pb.CreateEntryRequest {
|
||||
return &filer_pb.CreateEntryRequest{
|
||||
Directory: "/test",
|
||||
SkipCheckParentDirectory: true,
|
||||
Entry: &filer_pb.Entry{
|
||||
Name: "obj",
|
||||
Attributes: &filer_pb.FuseAttributes{Mtime: 1700000001, FileMode: 0644, Inode: 2},
|
||||
},
|
||||
Condition: one(&filer_pb.WriteCondition_Clause{Kind: filer_pb.WriteCondition_IF_ETAG_MATCH, Etags: []string{etag}}),
|
||||
}
|
||||
}
|
||||
|
||||
resp, err := fs.CreateEntry(context.Background(), req(`"zzz"`))
|
||||
if err != nil {
|
||||
t.Fatalf("unexpected err: %v", err)
|
||||
}
|
||||
if resp.ErrorCode != filer_pb.FilerError_PRECONDITION_FAILED {
|
||||
t.Fatalf("mismatched etag: want PRECONDITION_FAILED, got %v (%q)", resp.ErrorCode, resp.Error)
|
||||
}
|
||||
|
||||
resp, err = fs.CreateEntry(context.Background(), req(`"abc"`))
|
||||
if err != nil {
|
||||
t.Fatalf("unexpected err: %v", err)
|
||||
}
|
||||
if resp.Error != "" {
|
||||
t.Fatalf("matching etag should overwrite, got error %q", resp.Error)
|
||||
}
|
||||
}
|
||||
|
||||
// CreateEntry reuses a provided existing entry instead of reading the store
|
||||
// again; passing nil makes it look the path up itself.
|
||||
func TestCreateEntryReusesProvidedExisting(t *testing.T) {
|
||||
existing := &filer.Entry{
|
||||
FullPath: "/test/obj",
|
||||
Attr: filer.Attr{Inode: 1, Mtime: time.Unix(1700000000, 0), Crtime: time.Unix(1700000000, 0)},
|
||||
Extended: map[string][]byte{s3_constants.ExtETagKey: []byte("abc")},
|
||||
}
|
||||
newEntry := func() *filer.Entry {
|
||||
return &filer.Entry{
|
||||
FullPath: "/test/obj",
|
||||
Attr: filer.Attr{Inode: 1, Mtime: time.Unix(1700000001, 0)},
|
||||
}
|
||||
}
|
||||
|
||||
// Provided existing vs nil: the only difference is CreateEntry's own lookup,
|
||||
// so providing it must save exactly one path read (other store-layer reads
|
||||
// during the overwrite are the same in both runs).
|
||||
store := newRenameTestStore()
|
||||
store.entries["/test/obj"] = existing.ShallowClone()
|
||||
f := newRenameTestFiler(store)
|
||||
if err := f.CreateEntry(context.Background(), newEntry(), existing, false, false, nil, true, f.MaxFilenameLength); err != nil {
|
||||
t.Fatalf("create with existing: %v", err)
|
||||
}
|
||||
withExisting := store.findEntryCallCount("/test/obj")
|
||||
|
||||
store2 := newRenameTestStore()
|
||||
store2.entries["/test/obj"] = existing.ShallowClone()
|
||||
f2 := newRenameTestFiler(store2)
|
||||
if err := f2.CreateEntry(context.Background(), newEntry(), nil, false, false, nil, true, f2.MaxFilenameLength); err != nil {
|
||||
t.Fatalf("create with nil: %v", err)
|
||||
}
|
||||
withNil := store2.findEntryCallCount("/test/obj")
|
||||
|
||||
if withNil != withExisting+1 {
|
||||
t.Fatalf("providing existing should save one path lookup: existing=%d nil=%d", withExisting, withNil)
|
||||
}
|
||||
}
|
||||
@@ -1,68 +0,0 @@
|
||||
package weed_server
|
||||
|
||||
import (
|
||||
"context"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
)
|
||||
|
||||
// Concurrent OExcl creates for the same path must yield exactly one winner. The
|
||||
// filer's CreateEntry is a FindEntry-then-Insert; without the per-path lock both
|
||||
// racers observe "not found" and both insert. The exclusive entry lock makes the
|
||||
// check-then-act atomic so the losers see ErrEntryAlreadyExists.
|
||||
func TestCreateEntryOExclSerialized(t *testing.T) {
|
||||
store := newRenameTestStore()
|
||||
store.findDelay = 5 * time.Millisecond
|
||||
f := newRenameTestFiler(store)
|
||||
f.DirBucketsPath = "/buckets"
|
||||
|
||||
fs := &FilerServer{
|
||||
filer: f,
|
||||
option: &FilerOption{},
|
||||
entryLockTable: util.NewLockTable[util.FullPath](),
|
||||
}
|
||||
|
||||
const racers = 8
|
||||
var success, alreadyExists, unexpected int32
|
||||
var wg sync.WaitGroup
|
||||
start := make(chan struct{})
|
||||
for i := 0; i < racers; i++ {
|
||||
wg.Add(1)
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
<-start
|
||||
resp, err := fs.CreateEntry(context.Background(), &filer_pb.CreateEntryRequest{
|
||||
Directory: "/test",
|
||||
OExcl: true,
|
||||
SkipCheckParentDirectory: true,
|
||||
Entry: &filer_pb.Entry{
|
||||
Name: "obj",
|
||||
Attributes: &filer_pb.FuseAttributes{Mtime: 1700000000, FileMode: 0644, Inode: 1},
|
||||
},
|
||||
})
|
||||
switch {
|
||||
case err != nil:
|
||||
atomic.AddInt32(&unexpected, 1)
|
||||
case resp.Error == "":
|
||||
atomic.AddInt32(&success, 1)
|
||||
case resp.ErrorCode == filer_pb.FilerError_ENTRY_ALREADY_EXISTS:
|
||||
atomic.AddInt32(&alreadyExists, 1)
|
||||
default:
|
||||
atomic.AddInt32(&unexpected, 1)
|
||||
}
|
||||
}()
|
||||
}
|
||||
close(start)
|
||||
wg.Wait()
|
||||
|
||||
// Exactly one winner; every loser fails with ENTRY_ALREADY_EXISTS and nothing
|
||||
// else, so an unrelated failure can't masquerade as a passing test.
|
||||
if success != 1 || alreadyExists != racers-1 || unexpected != 0 {
|
||||
t.Fatalf("winners=%d already_exists=%d unexpected=%d (racers=%d)", success, alreadyExists, unexpected, racers)
|
||||
}
|
||||
}
|
||||
@@ -1,720 +0,0 @@
|
||||
package weed_server
|
||||
|
||||
import (
|
||||
"context"
|
||||
"net"
|
||||
"strconv"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/cluster/lock_manager"
|
||||
"github.com/seaweedfs/seaweedfs/weed/filer"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api/s3_constants"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
"google.golang.org/grpc"
|
||||
"google.golang.org/grpc/credentials/insecure"
|
||||
)
|
||||
|
||||
func newTxnTestServer(seed map[string]*filer.Entry) (*FilerServer, *renameTestStore) {
|
||||
store := newRenameTestStore()
|
||||
for path, entry := range seed {
|
||||
entry.FullPath = util.FullPath(path)
|
||||
store.entries[path] = entry
|
||||
}
|
||||
f := newRenameTestFiler(store)
|
||||
f.DirBucketsPath = "/buckets"
|
||||
fs := &FilerServer{filer: f, option: &FilerOption{}, entryLockTable: util.NewLockTable[util.FullPath]()}
|
||||
return fs, store
|
||||
}
|
||||
|
||||
// A versioned delete is a multi-entry object operation: drop the null version,
|
||||
// write a delete marker, and flip the latest pointer. ObjectTransaction applies
|
||||
// all three atomically under one lock keyed on the object path.
|
||||
func TestObjectTransactionMultiEntry(t *testing.T) {
|
||||
now := time.Unix(1700000000, 0)
|
||||
fs, store := newTxnTestServer(map[string]*filer.Entry{
|
||||
"/buckets/b/obj": {
|
||||
Attr: filer.Attr{Inode: 1, Mtime: now, Crtime: now, Mode: 0644},
|
||||
Extended: map[string][]byte{s3_constants.ExtETagKey: []byte("abc")},
|
||||
},
|
||||
"/buckets/b/obj/.versions": {
|
||||
Attr: filer.Attr{Inode: 2, Mtime: now, Crtime: now, Mode: 0755 | (1 << 31)},
|
||||
Extended: map[string][]byte{"latest": []byte("v1")},
|
||||
},
|
||||
})
|
||||
|
||||
req := &filer_pb.ObjectTransactionRequest{
|
||||
LockKey: "/buckets/b/obj",
|
||||
Mutations: []*filer_pb.ObjectMutation{
|
||||
{Type: filer_pb.ObjectMutation_DELETE, Directory: "/buckets/b", Name: "obj"},
|
||||
{Type: filer_pb.ObjectMutation_PUT, Directory: "/buckets/b/obj/.versions", Entry: &filer_pb.Entry{
|
||||
Name: "marker",
|
||||
Attributes: &filer_pb.FuseAttributes{Mtime: now.Unix(), FileMode: 0644, Inode: 3},
|
||||
Extended: map[string][]byte{"isDeleteMarker": []byte("true")},
|
||||
}},
|
||||
{Type: filer_pb.ObjectMutation_PATCH_EXTENDED, Directory: "/buckets/b/obj", Name: ".versions",
|
||||
SetExtended: map[string][]byte{"latest": []byte("marker")},
|
||||
DeleteExtended: nil},
|
||||
},
|
||||
}
|
||||
|
||||
resp, err := fs.ObjectTransaction(context.Background(), req)
|
||||
if err != nil {
|
||||
t.Fatalf("unexpected err: %v", err)
|
||||
}
|
||||
if resp.Error != "" {
|
||||
t.Fatalf("unexpected response error: %q", resp.Error)
|
||||
}
|
||||
|
||||
if _, ok := store.entries["/buckets/b/obj"]; ok {
|
||||
t.Errorf("null version should be deleted")
|
||||
}
|
||||
if _, ok := store.entries["/buckets/b/obj/.versions/marker"]; !ok {
|
||||
t.Errorf("delete marker should be created")
|
||||
}
|
||||
if got := string(store.entries["/buckets/b/obj/.versions"].Extended["latest"]); got != "marker" {
|
||||
t.Errorf("latest pointer = %q, want marker", got)
|
||||
}
|
||||
}
|
||||
|
||||
// A PATCH_EXTENDED mutation emits a metadata event (so the change replicates and
|
||||
// subscribers see it), carrying both the prior and updated state in the diff.
|
||||
func TestObjectTransactionPatchNotifies(t *testing.T) {
|
||||
queue := &captureQueue{}
|
||||
swapNotificationQueue(t, queue)
|
||||
|
||||
now := time.Unix(1700000000, 0)
|
||||
fs, _ := newTxnTestServer(map[string]*filer.Entry{
|
||||
"/buckets/b/obj/.versions": {
|
||||
Attr: filer.Attr{Inode: 2, Mtime: now, Crtime: now, Mode: 0755 | (1 << 31)},
|
||||
Extended: map[string][]byte{"latest": []byte("v1")},
|
||||
},
|
||||
})
|
||||
|
||||
resp, err := fs.ObjectTransaction(context.Background(), &filer_pb.ObjectTransactionRequest{
|
||||
LockKey: "/buckets/b/obj",
|
||||
Mutations: []*filer_pb.ObjectMutation{
|
||||
{Type: filer_pb.ObjectMutation_PATCH_EXTENDED, Directory: "/buckets/b/obj", Name: ".versions",
|
||||
SetExtended: map[string][]byte{"latest": []byte("v2")}},
|
||||
},
|
||||
})
|
||||
if err != nil || resp.Error != "" {
|
||||
t.Fatalf("txn failed: err=%v resp=%q", err, resp.Error)
|
||||
}
|
||||
|
||||
events := queue.snapshot()
|
||||
if len(events) != 1 {
|
||||
t.Fatalf("expected 1 metadata event from PATCH_EXTENDED, got %d", len(events))
|
||||
}
|
||||
ev := events[0].notification
|
||||
if ev.NewEntry == nil || string(ev.NewEntry.Extended["latest"]) != "v2" {
|
||||
t.Fatalf("event new entry latest = %q, want v2", ev.GetNewEntry().GetExtended()["latest"])
|
||||
}
|
||||
if ev.OldEntry == nil || string(ev.OldEntry.Extended["latest"]) != "v1" {
|
||||
t.Fatalf("event old entry latest = %q, want v1 (clone must preserve prior state)", ev.GetOldEntry().GetExtended()["latest"])
|
||||
}
|
||||
}
|
||||
|
||||
// PATCH_EXTENDED with set_content replaces Entry.Content while merging Extended
|
||||
// and preserving the rest; without set_content, Content is left untouched.
|
||||
func TestObjectTransactionPatchContent(t *testing.T) {
|
||||
now := time.Unix(1700000000, 0)
|
||||
fs, store := newTxnTestServer(map[string]*filer.Entry{
|
||||
"/buckets/b": {
|
||||
Attr: filer.Attr{Inode: 1, Mtime: now, Crtime: now, Mode: 0755 | (1 << 31)},
|
||||
Extended: map[string][]byte{"versioning": []byte("Enabled")},
|
||||
Content: []byte("old-content"),
|
||||
},
|
||||
})
|
||||
|
||||
// set_content replaces Content and merges an Extended key, preserving the
|
||||
// existing versioning key.
|
||||
resp, err := fs.ObjectTransaction(context.Background(), &filer_pb.ObjectTransactionRequest{
|
||||
LockKey: "/buckets/b",
|
||||
Mutations: []*filer_pb.ObjectMutation{
|
||||
{Type: filer_pb.ObjectMutation_PATCH_EXTENDED, Directory: "/buckets", Name: "b",
|
||||
SetContent: true, Content: []byte("encryption-blob"),
|
||||
SetExtended: map[string][]byte{"cors": []byte("yes")}},
|
||||
},
|
||||
})
|
||||
if err != nil || resp.Error != "" {
|
||||
t.Fatalf("patch set_content failed: err=%v resp=%q", err, resp.Error)
|
||||
}
|
||||
e := store.entries["/buckets/b"]
|
||||
if string(e.Content) != "encryption-blob" {
|
||||
t.Fatalf("content = %q, want encryption-blob", e.Content)
|
||||
}
|
||||
if string(e.Extended["versioning"]) != "Enabled" || string(e.Extended["cors"]) != "yes" {
|
||||
t.Fatalf("extended not merged: %v", e.Extended)
|
||||
}
|
||||
|
||||
// A PATCH without set_content must not disturb Content.
|
||||
resp, err = fs.ObjectTransaction(context.Background(), &filer_pb.ObjectTransactionRequest{
|
||||
LockKey: "/buckets/b",
|
||||
Mutations: []*filer_pb.ObjectMutation{
|
||||
{Type: filer_pb.ObjectMutation_PATCH_EXTENDED, Directory: "/buckets", Name: "b",
|
||||
SetExtended: map[string][]byte{"versioning": []byte("Suspended")}},
|
||||
},
|
||||
})
|
||||
if err != nil || resp.Error != "" {
|
||||
t.Fatalf("patch extended-only failed: err=%v resp=%q", err, resp.Error)
|
||||
}
|
||||
e = store.entries["/buckets/b"]
|
||||
if string(e.Content) != "encryption-blob" {
|
||||
t.Fatalf("content clobbered by extended-only patch: %q", e.Content)
|
||||
}
|
||||
if string(e.Extended["versioning"]) != "Suspended" {
|
||||
t.Fatalf("versioning = %q, want Suspended", e.Extended["versioning"])
|
||||
}
|
||||
if e.FileSize != 0 {
|
||||
t.Fatalf("directory FileSize must stay 0, got %d", e.FileSize)
|
||||
}
|
||||
|
||||
// For a file, set_content syncs FileSize to the new content length, even when
|
||||
// the content shrinks.
|
||||
store.entries["/file"] = &filer.Entry{
|
||||
FullPath: "/file",
|
||||
Attr: filer.Attr{Inode: 9, Mtime: now, Crtime: now, Mode: 0644, FileSize: 100},
|
||||
Content: []byte("xxxxxxxxxxxxxxx"),
|
||||
}
|
||||
resp, err = fs.ObjectTransaction(context.Background(), &filer_pb.ObjectTransactionRequest{
|
||||
LockKey: "/file",
|
||||
Mutations: []*filer_pb.ObjectMutation{
|
||||
{Type: filer_pb.ObjectMutation_PATCH_EXTENDED, Directory: "/", Name: "file",
|
||||
SetContent: true, Content: []byte("short")},
|
||||
},
|
||||
})
|
||||
if err != nil || resp.Error != "" {
|
||||
t.Fatalf("file patch failed: err=%v resp=%q", err, resp.Error)
|
||||
}
|
||||
if f := store.entries["/file"]; string(f.Content) != "short" || f.FileSize != uint64(len("short")) {
|
||||
t.Fatalf("file content=%q FileSize=%d, want short/5", f.Content, f.FileSize)
|
||||
}
|
||||
}
|
||||
|
||||
// A failing precondition aborts before any mutation is applied.
|
||||
func TestObjectTransactionPreconditionAborts(t *testing.T) {
|
||||
now := time.Unix(1700000000, 0)
|
||||
fs, store := newTxnTestServer(map[string]*filer.Entry{
|
||||
"/buckets/b/obj": {
|
||||
Attr: filer.Attr{Inode: 1, Mtime: now, Crtime: now, Mode: 0644},
|
||||
Extended: map[string][]byte{s3_constants.ExtETagKey: []byte("abc")},
|
||||
},
|
||||
})
|
||||
|
||||
req := &filer_pb.ObjectTransactionRequest{
|
||||
LockKey: "/buckets/b/obj",
|
||||
Condition: &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{
|
||||
{Kind: filer_pb.WriteCondition_IF_ETAG_MATCH, Etags: []string{`"zzz"`}},
|
||||
}},
|
||||
Mutations: []*filer_pb.ObjectMutation{
|
||||
{Type: filer_pb.ObjectMutation_DELETE, Directory: "/buckets/b", Name: "obj"},
|
||||
},
|
||||
}
|
||||
|
||||
resp, err := fs.ObjectTransaction(context.Background(), req)
|
||||
if err != nil {
|
||||
t.Fatalf("unexpected err: %v", err)
|
||||
}
|
||||
if resp.ErrorCode != filer_pb.FilerError_PRECONDITION_FAILED {
|
||||
t.Fatalf("want PRECONDITION_FAILED, got %v (%q)", resp.ErrorCode, resp.Error)
|
||||
}
|
||||
if _, ok := store.entries["/buckets/b/obj"]; !ok {
|
||||
t.Errorf("object must survive a failed precondition")
|
||||
}
|
||||
}
|
||||
|
||||
// Deleting the latest version and recomputing re-points the pointer at the new
|
||||
// highest-named remaining version; the scan runs under the transaction lock.
|
||||
func TestObjectTransactionRecomputeLatest(t *testing.T) {
|
||||
now := time.Unix(1700000000, 0)
|
||||
ver := func(id string) *filer.Entry {
|
||||
return &filer.Entry{
|
||||
Attr: filer.Attr{Inode: 10, Mtime: now, Crtime: now, Mode: 0644},
|
||||
Extended: map[string][]byte{"vid": []byte(id), "etag": []byte("etag-" + id)},
|
||||
}
|
||||
}
|
||||
fs, store := newTxnTestServer(map[string]*filer.Entry{
|
||||
"/buckets/b/obj/.versions": {
|
||||
Attr: filer.Attr{Inode: 2, Mtime: now, Crtime: now, Mode: 0755 | (1 << 31)},
|
||||
Extended: map[string][]byte{
|
||||
"latestVid": []byte("v3"), "latestEtag": []byte("etag-v3"), "latestName": []byte("v3.ver"),
|
||||
},
|
||||
},
|
||||
"/buckets/b/obj/.versions/v1.ver": ver("v1"),
|
||||
"/buckets/b/obj/.versions/v2.ver": ver("v2"),
|
||||
"/buckets/b/obj/.versions/v3.ver": ver("v3"),
|
||||
})
|
||||
|
||||
recompute := func() *filer_pb.ObjectMutation {
|
||||
return &filer_pb.ObjectMutation{
|
||||
Type: filer_pb.ObjectMutation_RECOMPUTE_LATEST, Directory: "/buckets/b/obj", Name: ".versions",
|
||||
Recompute: &filer_pb.Recompute{
|
||||
ScanDir: "/buckets/b/obj/.versions",
|
||||
Descending: true,
|
||||
CopyExtended: map[string]string{"latestVid": "vid", "latestEtag": "etag"},
|
||||
NameToKey: "latestName",
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// Delete the latest (v3); recompute should pick v2.
|
||||
resp, err := fs.ObjectTransaction(context.Background(), &filer_pb.ObjectTransactionRequest{
|
||||
LockKey: "/buckets/b/obj",
|
||||
Mutations: []*filer_pb.ObjectMutation{
|
||||
{Type: filer_pb.ObjectMutation_DELETE, Directory: "/buckets/b/obj/.versions", Name: "v3.ver"},
|
||||
recompute(),
|
||||
},
|
||||
})
|
||||
if err != nil || resp.Error != "" {
|
||||
t.Fatalf("txn failed: err=%v resp=%q", err, resp.Error)
|
||||
}
|
||||
ptr := store.entries["/buckets/b/obj/.versions"].Extended
|
||||
if string(ptr["latestVid"]) != "v2" || string(ptr["latestEtag"]) != "etag-v2" || string(ptr["latestName"]) != "v2.ver" {
|
||||
t.Fatalf("after deleting v3, pointer = vid:%s etag:%s name:%s; want v2",
|
||||
ptr["latestVid"], ptr["latestEtag"], ptr["latestName"])
|
||||
}
|
||||
|
||||
// Delete the remaining versions; recompute on an empty dir clears the pointer.
|
||||
resp, err = fs.ObjectTransaction(context.Background(), &filer_pb.ObjectTransactionRequest{
|
||||
LockKey: "/buckets/b/obj",
|
||||
Mutations: []*filer_pb.ObjectMutation{
|
||||
{Type: filer_pb.ObjectMutation_DELETE, Directory: "/buckets/b/obj/.versions", Name: "v2.ver"},
|
||||
{Type: filer_pb.ObjectMutation_DELETE, Directory: "/buckets/b/obj/.versions", Name: "v1.ver"},
|
||||
recompute(),
|
||||
},
|
||||
})
|
||||
if err != nil || resp.Error != "" {
|
||||
t.Fatalf("txn failed: err=%v resp=%q", err, resp.Error)
|
||||
}
|
||||
ptr = store.entries["/buckets/b/obj/.versions"].Extended
|
||||
for _, k := range []string{"latestVid", "latestEtag", "latestName"} {
|
||||
if _, ok := ptr[k]; ok {
|
||||
t.Errorf("pointer key %q should be cleared when no versions remain", k)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// With descending=false the lowest-named child is chosen (the listing is capped
|
||||
// at one entry).
|
||||
func TestObjectTransactionRecomputeAscending(t *testing.T) {
|
||||
now := time.Unix(1700000000, 0)
|
||||
ver := func(id string) *filer.Entry {
|
||||
return &filer.Entry{Attr: filer.Attr{Inode: 10, Mtime: now, Crtime: now, Mode: 0644}, Extended: map[string][]byte{"vid": []byte(id)}}
|
||||
}
|
||||
fs, store := newTxnTestServer(map[string]*filer.Entry{
|
||||
"/buckets/b/obj/.versions": {Attr: filer.Attr{Inode: 2, Mtime: now, Crtime: now, Mode: 0755 | (1 << 31)}, Extended: map[string][]byte{}},
|
||||
"/buckets/b/obj/.versions/v1.ver": ver("v1"),
|
||||
"/buckets/b/obj/.versions/v2.ver": ver("v2"),
|
||||
})
|
||||
|
||||
resp, err := fs.ObjectTransaction(context.Background(), &filer_pb.ObjectTransactionRequest{
|
||||
LockKey: "/buckets/b/obj",
|
||||
Mutations: []*filer_pb.ObjectMutation{
|
||||
{Type: filer_pb.ObjectMutation_RECOMPUTE_LATEST, Directory: "/buckets/b/obj", Name: ".versions",
|
||||
Recompute: &filer_pb.Recompute{
|
||||
ScanDir: "/buckets/b/obj/.versions",
|
||||
Descending: false,
|
||||
CopyExtended: map[string]string{"latestVid": "vid"},
|
||||
}},
|
||||
},
|
||||
})
|
||||
if err != nil || resp.Error != "" {
|
||||
t.Fatalf("txn failed: err=%v resp=%q", err, resp.Error)
|
||||
}
|
||||
if got := string(store.entries["/buckets/b/obj/.versions"].Extended["latestVid"]); got != "v1" {
|
||||
t.Fatalf("ascending recompute latestVid = %q, want v1 (lowest)", got)
|
||||
}
|
||||
}
|
||||
|
||||
// A batch applies each transaction independently: one failed precondition does
|
||||
// not abort the others, matching S3 multi-object delete semantics.
|
||||
func TestObjectTransactionBatchIndependent(t *testing.T) {
|
||||
now := time.Unix(1700000000, 0)
|
||||
obj := func(inode uint64) *filer.Entry {
|
||||
return &filer.Entry{
|
||||
Attr: filer.Attr{Inode: inode, Mtime: now, Crtime: now, Mode: 0644},
|
||||
Extended: map[string][]byte{s3_constants.ExtETagKey: []byte("abc")},
|
||||
}
|
||||
}
|
||||
fs, store := newTxnTestServer(map[string]*filer.Entry{
|
||||
"/buckets/b/a": obj(1),
|
||||
"/buckets/b/c": obj(3),
|
||||
})
|
||||
|
||||
del := func(name string, cond *filer_pb.WriteCondition) *filer_pb.ObjectTransactionRequest {
|
||||
return &filer_pb.ObjectTransactionRequest{
|
||||
LockKey: "/buckets/b/" + name,
|
||||
Condition: cond,
|
||||
Mutations: []*filer_pb.ObjectMutation{
|
||||
{Type: filer_pb.ObjectMutation_DELETE, Directory: "/buckets/b", Name: name},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
resp, err := fs.ObjectTransactionBatch(context.Background(), &filer_pb.ObjectTransactionBatchRequest{
|
||||
Transactions: []*filer_pb.ObjectTransactionRequest{
|
||||
del("a", nil),
|
||||
del("c", &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{
|
||||
{Kind: filer_pb.WriteCondition_IF_ETAG_MATCH, Etags: []string{`"zzz"`}},
|
||||
}}),
|
||||
},
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("unexpected err: %v", err)
|
||||
}
|
||||
if len(resp.Responses) != 2 {
|
||||
t.Fatalf("want 2 responses, got %d", len(resp.Responses))
|
||||
}
|
||||
if resp.Responses[0].Error != "" {
|
||||
t.Errorf("delete a should succeed: %q", resp.Responses[0].Error)
|
||||
}
|
||||
if resp.Responses[1].ErrorCode != filer_pb.FilerError_PRECONDITION_FAILED {
|
||||
t.Errorf("delete c should fail precondition, got %v", resp.Responses[1].ErrorCode)
|
||||
}
|
||||
if _, ok := store.entries["/buckets/b/a"]; ok {
|
||||
t.Errorf("a should be deleted")
|
||||
}
|
||||
if _, ok := store.entries["/buckets/b/c"]; !ok {
|
||||
t.Errorf("c should survive its failed precondition")
|
||||
}
|
||||
}
|
||||
|
||||
// A nil transaction in a batch yields an error response in its slot rather than
|
||||
// panicking, keeping responses parallel to the requests.
|
||||
func TestObjectTransactionBatchNilTransaction(t *testing.T) {
|
||||
now := time.Unix(1700000000, 0)
|
||||
fs, store := newTxnTestServer(map[string]*filer.Entry{
|
||||
"/buckets/b/a": {Attr: filer.Attr{Inode: 1, Mtime: now, Crtime: now, Mode: 0644}},
|
||||
})
|
||||
|
||||
resp, err := fs.ObjectTransactionBatch(context.Background(), &filer_pb.ObjectTransactionBatchRequest{
|
||||
Transactions: []*filer_pb.ObjectTransactionRequest{
|
||||
nil,
|
||||
{LockKey: "/buckets/b/a", Mutations: []*filer_pb.ObjectMutation{
|
||||
{Type: filer_pb.ObjectMutation_DELETE, Directory: "/buckets/b", Name: "a"},
|
||||
}},
|
||||
},
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("unexpected err: %v", err)
|
||||
}
|
||||
if len(resp.Responses) != 2 {
|
||||
t.Fatalf("want 2 responses (parallel to requests), got %d", len(resp.Responses))
|
||||
}
|
||||
if resp.Responses[0].Error == "" {
|
||||
t.Errorf("nil transaction should produce an error response")
|
||||
}
|
||||
if resp.Responses[1].Error != "" {
|
||||
t.Errorf("valid transaction should succeed: %q", resp.Responses[1].Error)
|
||||
}
|
||||
if _, ok := store.entries["/buckets/b/a"]; ok {
|
||||
t.Errorf("a should be deleted by the valid transaction")
|
||||
}
|
||||
}
|
||||
|
||||
// DELETE and PATCH of an absent entry are no-ops, so a replayed transaction
|
||||
// does not error.
|
||||
func TestObjectTransactionIdempotentNoops(t *testing.T) {
|
||||
fs, _ := newTxnTestServer(nil)
|
||||
|
||||
req := &filer_pb.ObjectTransactionRequest{
|
||||
LockKey: "/buckets/b/obj",
|
||||
Mutations: []*filer_pb.ObjectMutation{
|
||||
{Type: filer_pb.ObjectMutation_DELETE, Directory: "/buckets/b", Name: "obj"},
|
||||
{Type: filer_pb.ObjectMutation_PATCH_EXTENDED, Directory: "/buckets/b/obj", Name: ".versions",
|
||||
SetExtended: map[string][]byte{"latest": []byte("x")}},
|
||||
},
|
||||
}
|
||||
|
||||
resp, err := fs.ObjectTransaction(context.Background(), req)
|
||||
if err != nil {
|
||||
t.Fatalf("unexpected err: %v", err)
|
||||
}
|
||||
if resp.Error != "" {
|
||||
t.Fatalf("no-op mutations should not error: %q", resp.Error)
|
||||
}
|
||||
}
|
||||
|
||||
// RECOMPUTE_LATEST copies the chosen child's size/mtime to the pointer and
|
||||
// stamps the demote key on the prior latest when the pointer moves.
|
||||
func TestObjectTransactionRecomputeDemoteAndAttrs(t *testing.T) {
|
||||
t0 := time.Unix(1700000000, 0)
|
||||
t1 := time.Unix(1700000100, 0)
|
||||
mk := func(inode uint64, mt time.Time, size uint64, id string) *filer.Entry {
|
||||
return &filer.Entry{
|
||||
Attr: filer.Attr{Inode: inode, Mtime: mt, Crtime: mt, Mode: 0644, FileSize: size},
|
||||
Extended: map[string][]byte{"vid": []byte(id)},
|
||||
}
|
||||
}
|
||||
fs, store := newTxnTestServer(map[string]*filer.Entry{
|
||||
"/buckets/b/obj/.versions": {
|
||||
Attr: filer.Attr{Inode: 2, Mtime: t0, Crtime: t0, Mode: 0755 | (1 << 31)},
|
||||
Extended: map[string][]byte{"latestName": []byte("v1.ver"), "latestVid": []byte("v1")},
|
||||
},
|
||||
"/buckets/b/obj/.versions/v1.ver": mk(10, t0, 100, "v1"),
|
||||
"/buckets/b/obj/.versions/v2.ver": mk(11, t1, 250, "v2"),
|
||||
})
|
||||
|
||||
resp, err := fs.ObjectTransaction(context.Background(), &filer_pb.ObjectTransactionRequest{
|
||||
LockKey: "/buckets/b/obj",
|
||||
Mutations: []*filer_pb.ObjectMutation{{
|
||||
Type: filer_pb.ObjectMutation_RECOMPUTE_LATEST, Directory: "/buckets/b/obj", Name: ".versions",
|
||||
Recompute: &filer_pb.Recompute{
|
||||
ScanDir: "/buckets/b/obj/.versions",
|
||||
Descending: true,
|
||||
CopyExtended: map[string]string{"latestVid": "vid"},
|
||||
NameToKey: "latestName",
|
||||
SizeToKey: "latestSize",
|
||||
MtimeToKey: "latestMtime",
|
||||
DemoteKey: "noncurrentSince",
|
||||
DemoteValue: []byte("999"),
|
||||
},
|
||||
}},
|
||||
})
|
||||
if err != nil || resp.Error != "" {
|
||||
t.Fatalf("txn failed: err=%v resp=%q", err, resp.Error)
|
||||
}
|
||||
|
||||
ptr := store.entries["/buckets/b/obj/.versions"].Extended
|
||||
if string(ptr["latestName"]) != "v2.ver" || string(ptr["latestVid"]) != "v2" {
|
||||
t.Fatalf("pointer not moved to v2: name=%s vid=%s", ptr["latestName"], ptr["latestVid"])
|
||||
}
|
||||
if string(ptr["latestSize"]) != "250" {
|
||||
t.Errorf("latestSize = %s, want 250", ptr["latestSize"])
|
||||
}
|
||||
if want := strconv.FormatInt(t1.Unix(), 10); string(ptr["latestMtime"]) != want {
|
||||
t.Errorf("latestMtime = %s, want %s", ptr["latestMtime"], want)
|
||||
}
|
||||
if got := store.entries["/buckets/b/obj/.versions/v1.ver"].Extended["noncurrentSince"]; string(got) != "999" {
|
||||
t.Errorf("prior latest v1.ver noncurrentSince = %q, want 999", got)
|
||||
}
|
||||
if _, ok := store.entries["/buckets/b/obj/.versions/v2.ver"].Extended["noncurrentSince"]; ok {
|
||||
t.Errorf("new latest v2.ver should not be demoted")
|
||||
}
|
||||
}
|
||||
|
||||
// A version-specific delete locks the object (condition_key checks WORM on the
|
||||
// version), recomputes the pointer excluding the version (repoint-before-delete),
|
||||
// then deletes it. A legal-hold guard blocks the delete and preserves the entry.
|
||||
func TestObjectTransactionVersionDeleteWithWorm(t *testing.T) {
|
||||
now := time.Unix(1700000000, 0)
|
||||
ver := func(inode uint64, ext map[string][]byte) *filer.Entry {
|
||||
return &filer.Entry{Attr: filer.Attr{Inode: inode, Mtime: now, Crtime: now, Mode: 0644}, Extended: ext}
|
||||
}
|
||||
seed := func(latestLocked bool) map[string]*filer.Entry {
|
||||
vcExt := map[string][]byte{"vid": []byte("v3")}
|
||||
if latestLocked {
|
||||
vcExt["legalhold"] = []byte("ON")
|
||||
}
|
||||
return map[string]*filer.Entry{
|
||||
"/buckets/b/obj/.versions": {
|
||||
Attr: filer.Attr{Inode: 2, Mtime: now, Crtime: now, Mode: 0755 | (1 << 31)},
|
||||
Extended: map[string][]byte{"latestName": []byte("v_c"), "latestVid": []byte("v3")},
|
||||
},
|
||||
"/buckets/b/obj/.versions/v_a": ver(10, map[string][]byte{"vid": []byte("v1")}),
|
||||
"/buckets/b/obj/.versions/v_b": ver(11, map[string][]byte{"vid": []byte("v2")}),
|
||||
"/buckets/b/obj/.versions/v_c": ver(12, vcExt),
|
||||
}
|
||||
}
|
||||
mkReq := func() *filer_pb.ObjectTransactionRequest {
|
||||
return &filer_pb.ObjectTransactionRequest{
|
||||
LockKey: "/buckets/b/obj",
|
||||
ConditionKey: "/buckets/b/obj/.versions/v_c",
|
||||
Condition: &filer_pb.WriteCondition{Clauses: []*filer_pb.WriteCondition_Clause{
|
||||
{Kind: filer_pb.WriteCondition_IF_EXTENDED_NOT_EQUAL, ExtKey: "legalhold", ExtValue: "ON"},
|
||||
}},
|
||||
Mutations: []*filer_pb.ObjectMutation{
|
||||
{Type: filer_pb.ObjectMutation_RECOMPUTE_LATEST, Directory: "/buckets/b/obj", Name: ".versions",
|
||||
Recompute: &filer_pb.Recompute{ScanDir: "/buckets/b/obj/.versions", Descending: true, ExcludeName: "v_c",
|
||||
NameToKey: "latestName", CopyExtended: map[string]string{"latestVid": "vid"}}},
|
||||
{Type: filer_pb.ObjectMutation_DELETE, Directory: "/buckets/b/obj/.versions", Name: "v_c"},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// Legal hold ON: the WORM guard blocks; version and pointer untouched.
|
||||
fs, store := newTxnTestServer(seed(true))
|
||||
resp, err := fs.ObjectTransaction(context.Background(), mkReq())
|
||||
if err != nil {
|
||||
t.Fatalf("err: %v", err)
|
||||
}
|
||||
if resp.ErrorCode != filer_pb.FilerError_PRECONDITION_FAILED {
|
||||
t.Fatalf("locked version delete should fail precondition, got code=%v err=%q", resp.ErrorCode, resp.Error)
|
||||
}
|
||||
if _, ok := store.entries["/buckets/b/obj/.versions/v_c"]; !ok {
|
||||
t.Errorf("locked version must not be deleted")
|
||||
}
|
||||
if got := string(store.entries["/buckets/b/obj/.versions"].Extended["latestName"]); got != "v_c" {
|
||||
t.Errorf("pointer must be unchanged when delete is blocked, got %s", got)
|
||||
}
|
||||
|
||||
// No legal hold: pointer recomputes to v_b (excluding v_c), then v_c is deleted.
|
||||
fs, store = newTxnTestServer(seed(false))
|
||||
resp, err = fs.ObjectTransaction(context.Background(), mkReq())
|
||||
if err != nil || resp.Error != "" {
|
||||
t.Fatalf("unlocked delete failed: err=%v resp=%q", err, resp.Error)
|
||||
}
|
||||
if _, ok := store.entries["/buckets/b/obj/.versions/v_c"]; ok {
|
||||
t.Errorf("unlocked version should be deleted")
|
||||
}
|
||||
ptr := store.entries["/buckets/b/obj/.versions"].Extended
|
||||
if string(ptr["latestName"]) != "v_b" || string(ptr["latestVid"]) != "v2" {
|
||||
t.Errorf("pointer should recompute to v_b/v2, got name=%s vid=%s", ptr["latestName"], ptr["latestVid"])
|
||||
}
|
||||
}
|
||||
|
||||
// PATCH_EXTENDED with touch_mtime bumps the entry's Mtime (a metadata-replace
|
||||
// copy) while merging Extended.
|
||||
func TestObjectTransactionPatchTouchMtime(t *testing.T) {
|
||||
old := time.Unix(1600000000, 0)
|
||||
fs, store := newTxnTestServer(map[string]*filer.Entry{
|
||||
"/buckets/b/obj": {
|
||||
FullPath: "/buckets/b/obj",
|
||||
Attr: filer.Attr{Inode: 1, Mtime: old, Crtime: old, Mode: 0644},
|
||||
Extended: map[string][]byte{"X-Amz-Meta-old": []byte("1")},
|
||||
},
|
||||
})
|
||||
resp, err := fs.ObjectTransaction(context.Background(), &filer_pb.ObjectTransactionRequest{
|
||||
LockKey: "/buckets/b/obj",
|
||||
Mutations: []*filer_pb.ObjectMutation{{
|
||||
Type: filer_pb.ObjectMutation_PATCH_EXTENDED, Directory: "/buckets/b", Name: "obj",
|
||||
SetExtended: map[string][]byte{"X-Amz-Meta-new": []byte("2")},
|
||||
DeleteExtended: []string{"X-Amz-Meta-old"},
|
||||
TouchMtime: true,
|
||||
}},
|
||||
})
|
||||
if err != nil || resp.Error != "" {
|
||||
t.Fatalf("patch failed: err=%v resp=%q", err, resp.Error)
|
||||
}
|
||||
e := store.entries["/buckets/b/obj"]
|
||||
if !e.Attr.Mtime.After(old) {
|
||||
t.Errorf("touch_mtime should bump Mtime past %v, got %v", old, e.Attr.Mtime)
|
||||
}
|
||||
if _, ok := e.Extended["X-Amz-Meta-old"]; ok {
|
||||
t.Errorf("old meta should be deleted")
|
||||
}
|
||||
if string(e.Extended["X-Amz-Meta-new"]) != "2" {
|
||||
t.Errorf("new meta not set: %v", e.Extended)
|
||||
}
|
||||
}
|
||||
|
||||
// withRing attaches a Dlm whose ring contains exactly the given servers and sets
|
||||
// the filer's own host, so route_key resolution in ObjectTransaction is decided
|
||||
// by who owns the single-server ring.
|
||||
func withRing(fs *FilerServer, self pb.ServerAddress, servers ...pb.ServerAddress) {
|
||||
dlm := lock_manager.NewDistributedLockManager(self)
|
||||
dlm.LockRing.SetSnapshot(servers, 1)
|
||||
fs.filer.Dlm = dlm
|
||||
fs.option.Host = self
|
||||
}
|
||||
|
||||
// When this filer owns route_key, the transaction applies locally rather than
|
||||
// forwarding to itself.
|
||||
func TestObjectTransactionRouteKeyOwnerAppliesLocally(t *testing.T) {
|
||||
self := pb.ServerAddress("localhost:1")
|
||||
fs, store := newTxnTestServer(map[string]*filer.Entry{
|
||||
"/buckets/b/obj": {FullPath: "/buckets/b/obj", Attr: filer.Attr{Inode: 1, Mode: 0644}},
|
||||
})
|
||||
withRing(fs, self, self)
|
||||
|
||||
resp, err := fs.ObjectTransaction(context.Background(), &filer_pb.ObjectTransactionRequest{
|
||||
LockKey: "/buckets/b/obj",
|
||||
RouteKey: "s3.object.write:/buckets/b/obj",
|
||||
Mutations: []*filer_pb.ObjectMutation{{
|
||||
Type: filer_pb.ObjectMutation_PATCH_EXTENDED, Directory: "/buckets/b", Name: "obj",
|
||||
SetExtended: map[string][]byte{"X-Amz-Meta-k": []byte("v")},
|
||||
}},
|
||||
})
|
||||
if err != nil || resp.Error != "" {
|
||||
t.Fatalf("txn failed: err=%v resp=%q", err, resp.Error)
|
||||
}
|
||||
if string(store.entries["/buckets/b/obj"].Extended["X-Amz-Meta-k"]) != "v" {
|
||||
t.Errorf("mutation should have applied locally: %v", store.entries["/buckets/b/obj"].Extended)
|
||||
}
|
||||
}
|
||||
|
||||
// A forwarded transaction (is_moved) applies locally even when the ring names a
|
||||
// different owner: is_moved bounds forwarding to a single hop, so two filers that
|
||||
// disagree on the owner during a ring change cannot loop. If is_moved were
|
||||
// ignored, this would attempt to dial the bogus owner instead of applying.
|
||||
func TestObjectTransactionIsMovedSkipsForward(t *testing.T) {
|
||||
self := pb.ServerAddress("localhost:1")
|
||||
other := pb.ServerAddress("localhost:2")
|
||||
fs, store := newTxnTestServer(map[string]*filer.Entry{
|
||||
"/buckets/b/obj": {FullPath: "/buckets/b/obj", Attr: filer.Attr{Inode: 1, Mode: 0644}},
|
||||
})
|
||||
withRing(fs, self, other) // ring owner is "other", not self
|
||||
|
||||
resp, err := fs.ObjectTransaction(context.Background(), &filer_pb.ObjectTransactionRequest{
|
||||
LockKey: "/buckets/b/obj",
|
||||
RouteKey: "s3.object.write:/buckets/b/obj",
|
||||
IsMoved: true,
|
||||
Mutations: []*filer_pb.ObjectMutation{{
|
||||
Type: filer_pb.ObjectMutation_PATCH_EXTENDED, Directory: "/buckets/b", Name: "obj",
|
||||
SetExtended: map[string][]byte{"X-Amz-Meta-k": []byte("v")},
|
||||
}},
|
||||
})
|
||||
if err != nil || resp.Error != "" {
|
||||
t.Fatalf("txn failed: err=%v resp=%q", err, resp.Error)
|
||||
}
|
||||
if string(store.entries["/buckets/b/obj"].Extended["X-Amz-Meta-k"]) != "v" {
|
||||
t.Errorf("forwarded txn should apply locally: %v", store.entries["/buckets/b/obj"].Extended)
|
||||
}
|
||||
}
|
||||
|
||||
// End-to-end forward hop: a non-owner filer dials the ring owner and the owner
|
||||
// applies the transaction. The owner's own ring points back at the (bogus)
|
||||
// sender, so it would re-forward and fail to dial unless is_moved is set on the
|
||||
// forwarded request — making this also assert that one-hop bound over the wire.
|
||||
func TestObjectTransactionForwardsToOwner(t *testing.T) {
|
||||
owner, ownerStore := newTxnTestServer(map[string]*filer.Entry{
|
||||
"/buckets/b/obj": {FullPath: "/buckets/b/obj", Attr: filer.Attr{Inode: 1, Mode: 0644}},
|
||||
})
|
||||
|
||||
lis, err := net.Listen("tcp", "127.0.0.1:0")
|
||||
if err != nil {
|
||||
t.Fatalf("listen: %v", err)
|
||||
}
|
||||
// Pin the grpc port to the real listener (ToGrpcAddress otherwise adds the
|
||||
// +10000 convention, which dials nothing).
|
||||
port := lis.Addr().(*net.TCPAddr).Port
|
||||
ownerAddr := pb.NewServerAddressWithGrpcPort(lis.Addr().String(), port)
|
||||
sender := pb.ServerAddress("127.0.0.1:1") // bogus: nothing listens here
|
||||
|
||||
// owner's ring points back at the sender; only is_moved keeps it from
|
||||
// re-forwarding to (and failing to dial) that bogus address.
|
||||
withRing(owner, ownerAddr, sender)
|
||||
owner.grpcDialOption = grpc.WithTransportCredentials(insecure.NewCredentials())
|
||||
|
||||
srv := grpc.NewServer()
|
||||
filer_pb.RegisterSeaweedFilerServer(srv, owner)
|
||||
go srv.Serve(lis)
|
||||
t.Cleanup(srv.Stop)
|
||||
|
||||
self, selfStore := newTxnTestServer(nil)
|
||||
withRing(self, sender, ownerAddr) // ring owner is the real owner; self forwards
|
||||
self.grpcDialOption = grpc.WithTransportCredentials(insecure.NewCredentials())
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second)
|
||||
defer cancel()
|
||||
resp, err := self.ObjectTransaction(ctx, &filer_pb.ObjectTransactionRequest{
|
||||
LockKey: "/buckets/b/obj",
|
||||
RouteKey: "s3.object.write:/buckets/b/obj",
|
||||
Mutations: []*filer_pb.ObjectMutation{{
|
||||
Type: filer_pb.ObjectMutation_PATCH_EXTENDED, Directory: "/buckets/b", Name: "obj",
|
||||
SetExtended: map[string][]byte{"X-Amz-Meta-k": []byte("v")},
|
||||
}},
|
||||
})
|
||||
if err != nil || resp.Error != "" {
|
||||
t.Fatalf("forwarded txn failed: err=%v resp=%q", err, resp.Error)
|
||||
}
|
||||
if string(ownerStore.entries["/buckets/b/obj"].Extended["X-Amz-Meta-k"]) != "v" {
|
||||
t.Errorf("owner should have applied the forwarded mutation: %v", ownerStore.entries["/buckets/b/obj"].Extended)
|
||||
}
|
||||
if _, ok := selfStore.entries["/buckets/b/obj"]; ok {
|
||||
t.Errorf("non-owner must forward, not apply locally")
|
||||
}
|
||||
}
|
||||
@@ -1,248 +0,0 @@
|
||||
package weed_server
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/filer/posixlock"
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
)
|
||||
|
||||
const (
|
||||
// posixLockSessionTTL is how long a mount's lease survives without a
|
||||
// keepalive before its locks are reaped; posixLockSweepInterval is how often
|
||||
// each filer checks. The mount renews well within the TTL.
|
||||
posixLockSessionTTL = 15 * time.Second
|
||||
posixLockSweepInterval = 5 * time.Second
|
||||
// posixCoolingProbeTimeout bounds the dual-read probe to the prior owner so a
|
||||
// slow peer can't stall a non-blocking lock call during the cooling window.
|
||||
posixCoolingProbeTimeout = 2 * time.Second
|
||||
// posixLockWarmup is how long after a (re)start the owner defers would-be
|
||||
// grants while mounts re-assert their held locks, so it does not double-grant
|
||||
// a lock its fresh, still-empty state has not yet rebuilt. Must exceed the
|
||||
// mount keepalive interval and stay below the TTL.
|
||||
posixLockWarmup = 10 * time.Second
|
||||
)
|
||||
|
||||
// startPosixLockSweeper periodically reaps the locks of leased sessions (mounts)
|
||||
// that stopped sending keepalives. Sessions that never renew are never reaped, so
|
||||
// this is inert until mounts run with -posixLock.
|
||||
func (fs *FilerServer) startPosixLockSweeper() {
|
||||
fs.posixLockReadyAt.Store(time.Now().UnixNano())
|
||||
fs.posixLockSweeperStop = make(chan struct{})
|
||||
go func() {
|
||||
ticker := time.NewTicker(posixLockSweepInterval)
|
||||
defer ticker.Stop()
|
||||
for {
|
||||
select {
|
||||
case <-fs.posixLockSweeperStop:
|
||||
return
|
||||
case <-ticker.C:
|
||||
if reaped := fs.posixLocks.ReapExpired(posixLockSessionTTL); len(reaped) > 0 {
|
||||
glog.V(2).Infof("posix lock: reaped %d expired session(s): %v", len(reaped), reaped)
|
||||
}
|
||||
}
|
||||
}
|
||||
}()
|
||||
}
|
||||
|
||||
// PosixLock applies one advisory lock operation against the in-memory lock table
|
||||
// of the inode's owner filer. The owner is resolved from req.Key on the same
|
||||
// ring the gateway and DLM use; a non-owner filer forwards the request one hop
|
||||
// (is_moved bounds it), so the owner's table stays the single authority even when
|
||||
// the caller's ring view is stale. Blocking (SetLkw) is the caller's job — this
|
||||
// is strictly non-blocking try/release/query.
|
||||
func (fs *FilerServer) PosixLock(ctx context.Context, req *filer_pb.PosixLockRequest) (*filer_pb.PosixLockResponse, error) {
|
||||
if req.Key == "" {
|
||||
return &filer_pb.PosixLockResponse{}, fmt.Errorf("key is required")
|
||||
}
|
||||
if req.Lock == nil {
|
||||
return &filer_pb.PosixLockResponse{}, fmt.Errorf("lock is required")
|
||||
}
|
||||
|
||||
if !req.IsMoved && fs.filer.Dlm != nil {
|
||||
if owner := fs.filer.Dlm.LockRing.GetPrimary(req.Key); owner != "" && owner != fs.option.Host {
|
||||
forwarded := &filer_pb.PosixLockRequest{
|
||||
Key: req.Key,
|
||||
IsMoved: true,
|
||||
Op: req.Op,
|
||||
Lock: req.Lock,
|
||||
Locks: req.Locks,
|
||||
}
|
||||
glog.V(4).InfofCtx(ctx, "PosixLock %s op=%v: forwarding to owner %s", req.Key, req.Op, owner)
|
||||
var resp *filer_pb.PosixLockResponse
|
||||
err := pb.WithFilerClient(false, 0, owner, fs.grpcDialOption, func(client filer_pb.SeaweedFilerClient) error {
|
||||
var e error
|
||||
resp, e = client.PosixLock(ctx, forwarded)
|
||||
return e
|
||||
})
|
||||
if err != nil {
|
||||
return &filer_pb.PosixLockResponse{}, err
|
||||
}
|
||||
return resp, nil
|
||||
}
|
||||
}
|
||||
|
||||
lk := posixlock.Range{
|
||||
Start: req.Lock.GetStart(),
|
||||
End: req.Lock.GetEnd(),
|
||||
Type: req.Lock.GetType(),
|
||||
Sid: req.Lock.GetSid(),
|
||||
Owner: req.Lock.GetOwner(),
|
||||
Pid: req.Lock.GetPid(),
|
||||
IsFlock: req.Lock.GetIsFlock(),
|
||||
}
|
||||
|
||||
resp := &filer_pb.PosixLockResponse{}
|
||||
switch req.Op {
|
||||
case filer_pb.PosixLockOp_TRY_LOCK:
|
||||
// A fresh owner's state may be incomplete (post-restart warm-up, or a ring
|
||||
// change whose previous owner can't be reached): report a known conflict,
|
||||
// else defer the grant so the client retries rather than risk a
|
||||
// double-grant. A deferred grant becomes EAGAIN for non-blocking SetLk and
|
||||
// a retry for the blocking SetLkw poll — never a spurious grant.
|
||||
if !req.CoolingProbe {
|
||||
if c, deferGrant := fs.posixCoolingOrWarmup(ctx, req.Key, lk); c != nil {
|
||||
resp.HasConflict = true
|
||||
resp.Conflict = c
|
||||
break
|
||||
} else if deferGrant {
|
||||
break
|
||||
}
|
||||
}
|
||||
if c, granted := fs.posixLocks.TryLock(req.Key, lk); granted {
|
||||
resp.Granted = true
|
||||
} else {
|
||||
resp.HasConflict = true
|
||||
resp.Conflict = posixRangeToPb(c)
|
||||
}
|
||||
case filer_pb.PosixLockOp_UNLOCK:
|
||||
fs.posixLocks.Unlock(req.Key, lk)
|
||||
case filer_pb.PosixLockOp_GET_LK:
|
||||
// A query is best-effort: report a known conflict but never defer.
|
||||
if !req.CoolingProbe {
|
||||
if c, _ := fs.posixCoolingOrWarmup(ctx, req.Key, lk); c != nil {
|
||||
resp.HasConflict = true
|
||||
resp.Conflict = c
|
||||
break
|
||||
}
|
||||
}
|
||||
if c, found := fs.posixLocks.GetLk(req.Key, lk); found {
|
||||
resp.HasConflict = true
|
||||
resp.Conflict = posixRangeToPb(c)
|
||||
}
|
||||
case filer_pb.PosixLockOp_RELEASE_POSIX_OWNER:
|
||||
fs.posixLocks.ReleasePosixOwner(req.Key, lk.Sid, lk.Owner)
|
||||
case filer_pb.PosixLockOp_RELEASE_FLOCK_OWNER:
|
||||
fs.posixLocks.ReleaseFlockOwner(req.Key, lk.Sid, lk.Owner)
|
||||
case filer_pb.PosixLockOp_KEEP_ALIVE:
|
||||
// A re-assertion carries the mount's held locks on this key so the owner
|
||||
// can rebuild its in-memory state after an ownership change or restart; a
|
||||
// bare keepalive just renews the lease.
|
||||
if len(req.Locks) > 0 {
|
||||
held := make([]posixlock.Range, 0, len(req.Locks))
|
||||
for _, l := range req.Locks {
|
||||
held = append(held, posixRangeFromPb(l))
|
||||
}
|
||||
if conflicts := fs.posixLocks.Reassert(req.Key, lk.Sid, held); len(conflicts) > 0 {
|
||||
glog.Warningf("posix reassert %s sid %d: %d lock(s) lost to another session: %+v", req.Key, lk.Sid, len(conflicts), conflicts)
|
||||
}
|
||||
} else {
|
||||
fs.posixLocks.Renew(lk.Sid)
|
||||
}
|
||||
default:
|
||||
return &filer_pb.PosixLockResponse{}, fmt.Errorf("unknown posix lock op %v", req.Op)
|
||||
}
|
||||
return resp, nil
|
||||
}
|
||||
|
||||
// posixWarmingUp reports whether this filer is still within posixLockWarmup of
|
||||
// when it began serving POSIX locks. The zero readyAt (e.g. in tests) is never
|
||||
// warming up.
|
||||
func (fs *FilerServer) posixWarmingUp() bool {
|
||||
readyAt := fs.posixLockReadyAt.Load()
|
||||
return readyAt != 0 && time.Since(time.Unix(0, readyAt)) < posixLockWarmup
|
||||
}
|
||||
|
||||
// posixCoolingOrWarmup decides whether a fresh owner can trust its local state
|
||||
// for key before granting. It returns a known blocking lock (conflict != nil),
|
||||
// or asks the caller to defer the grant (deferGrant) when the state may be
|
||||
// incomplete and no conflict can be confirmed. It returns (nil, false) when the
|
||||
// owner is authoritative and should consult its local table.
|
||||
//
|
||||
// - Warm-up: just after a (re)start the owner is still rebuilding from
|
||||
// re-assertions, so only a locally-visible conflict is trustworthy; a
|
||||
// would-be grant is deferred.
|
||||
// - Ring change (cooling window): the previous owner may still hold a lock this
|
||||
// owner hasn't rebuilt. Ask it (marked cooling_probe so it answers locally
|
||||
// without recursing), under a short deadline so a slow peer can't stall the
|
||||
// non-blocking lock path. If it is unreachable — typically because it
|
||||
// crashed, which caused the change — we cannot confirm, so we defer rather
|
||||
// than risk a double-grant; re-assertion rebuilds this owner before the
|
||||
// window ends.
|
||||
func (fs *FilerServer) posixCoolingOrWarmup(ctx context.Context, key string, lk posixlock.Range) (conflict *filer_pb.PosixLockRange, deferGrant bool) {
|
||||
if fs.posixWarmingUp() {
|
||||
if c, found := fs.posixLocks.GetLk(key, lk); found {
|
||||
return posixRangeToPb(c), false
|
||||
}
|
||||
return nil, true
|
||||
}
|
||||
if fs.filer.Dlm == nil {
|
||||
return nil, false
|
||||
}
|
||||
prior := fs.filer.Dlm.LockRing.PriorOwner(key)
|
||||
if prior == "" || prior == fs.option.Host {
|
||||
return nil, false
|
||||
}
|
||||
probeCtx, cancel := context.WithTimeout(ctx, posixCoolingProbeTimeout)
|
||||
defer cancel()
|
||||
var resp *filer_pb.PosixLockResponse
|
||||
err := pb.WithFilerClient(false, 0, prior, fs.grpcDialOption, func(client filer_pb.SeaweedFilerClient) error {
|
||||
var e error
|
||||
resp, e = client.PosixLock(probeCtx, &filer_pb.PosixLockRequest{
|
||||
Key: key,
|
||||
IsMoved: true,
|
||||
CoolingProbe: true,
|
||||
Op: filer_pb.PosixLockOp_GET_LK,
|
||||
Lock: posixRangeToPb(lk),
|
||||
})
|
||||
return e
|
||||
})
|
||||
if err != nil {
|
||||
// Cannot confirm — defer rather than risk a double-grant (the prior owner
|
||||
// likely crashed; re-assertion rebuilds this owner before the window ends).
|
||||
glog.V(2).InfofCtx(ctx, "posix cooling probe %s -> %s: %v (deferring)", key, prior, err)
|
||||
return nil, true
|
||||
}
|
||||
if resp.GetHasConflict() {
|
||||
return resp.GetConflict(), false
|
||||
}
|
||||
return nil, false
|
||||
}
|
||||
|
||||
func posixRangeFromPb(l *filer_pb.PosixLockRange) posixlock.Range {
|
||||
return posixlock.Range{
|
||||
Start: l.GetStart(),
|
||||
End: l.GetEnd(),
|
||||
Type: l.GetType(),
|
||||
Sid: l.GetSid(),
|
||||
Owner: l.GetOwner(),
|
||||
Pid: l.GetPid(),
|
||||
IsFlock: l.GetIsFlock(),
|
||||
}
|
||||
}
|
||||
|
||||
func posixRangeToPb(r posixlock.Range) *filer_pb.PosixLockRange {
|
||||
return &filer_pb.PosixLockRange{
|
||||
Start: r.Start,
|
||||
End: r.End,
|
||||
Type: r.Type,
|
||||
Sid: r.Sid,
|
||||
Owner: r.Owner,
|
||||
Pid: r.Pid,
|
||||
IsFlock: r.IsFlock,
|
||||
}
|
||||
}
|
||||
@@ -1,178 +0,0 @@
|
||||
package weed_server
|
||||
|
||||
import (
|
||||
"context"
|
||||
"net"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/filer/posixlock"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"google.golang.org/grpc"
|
||||
"google.golang.org/grpc/credentials/insecure"
|
||||
)
|
||||
|
||||
func newPosixTestServer() *FilerServer {
|
||||
fs, _ := newTxnTestServer(nil)
|
||||
fs.posixLocks = posixlock.NewManager()
|
||||
return fs
|
||||
}
|
||||
|
||||
func pbLock(start, end uint64, typ uint32, sid, owner uint64, pid uint32, flock bool) *filer_pb.PosixLockRange {
|
||||
return &filer_pb.PosixLockRange{Start: start, End: end, Type: typ, Sid: sid, Owner: owner, Pid: pid, IsFlock: flock}
|
||||
}
|
||||
|
||||
func posixOp(t *testing.T, fs *FilerServer, op filer_pb.PosixLockOp, lk *filer_pb.PosixLockRange) *filer_pb.PosixLockResponse {
|
||||
t.Helper()
|
||||
resp, err := fs.PosixLock(context.Background(), &filer_pb.PosixLockRequest{Key: "s3.fuse.lock:/x", Op: op, Lock: lk})
|
||||
if err != nil {
|
||||
t.Fatalf("PosixLock op=%v: %v", op, err)
|
||||
}
|
||||
return resp
|
||||
}
|
||||
|
||||
func TestPosixLockGrantAndConflict(t *testing.T) {
|
||||
fs := newPosixTestServer()
|
||||
|
||||
if r := posixOp(t, fs, filer_pb.PosixLockOp_TRY_LOCK, pbLock(0, 99, posixlock.Write, 1, 1, 7, false)); !r.Granted {
|
||||
t.Fatal("first lock should be granted")
|
||||
}
|
||||
// Conflicting writer from another session: rejected, conflict reported.
|
||||
r := posixOp(t, fs, filer_pb.PosixLockOp_TRY_LOCK, pbLock(50, 149, posixlock.Write, 2, 1, 8, false))
|
||||
if r.Granted {
|
||||
t.Fatal("overlapping lock from another session should conflict")
|
||||
}
|
||||
if !r.HasConflict || r.Conflict.GetPid() != 7 || r.Conflict.GetStart() != 0 || r.Conflict.GetEnd() != 99 {
|
||||
t.Fatalf("conflict not reported correctly: %+v", r.Conflict)
|
||||
}
|
||||
}
|
||||
|
||||
func TestPosixLockUnlockThenReacquire(t *testing.T) {
|
||||
fs := newPosixTestServer()
|
||||
posixOp(t, fs, filer_pb.PosixLockOp_TRY_LOCK, pbLock(0, 99, posixlock.Write, 1, 1, 7, false))
|
||||
posixOp(t, fs, filer_pb.PosixLockOp_UNLOCK, pbLock(0, 99, posixlock.Unlock, 1, 1, 7, false))
|
||||
if r := posixOp(t, fs, filer_pb.PosixLockOp_TRY_LOCK, pbLock(0, 99, posixlock.Write, 2, 1, 8, false)); !r.Granted {
|
||||
t.Fatal("lock should be grantable after the holder unlocked")
|
||||
}
|
||||
}
|
||||
|
||||
func TestPosixLockGetLk(t *testing.T) {
|
||||
fs := newPosixTestServer()
|
||||
posixOp(t, fs, filer_pb.PosixLockOp_TRY_LOCK, pbLock(10, 50, posixlock.Write, 1, 1, 7, false))
|
||||
|
||||
r := posixOp(t, fs, filer_pb.PosixLockOp_GET_LK, pbLock(30, 70, posixlock.Read, 2, 1, 8, false))
|
||||
if !r.HasConflict || r.Conflict.GetPid() != 7 {
|
||||
t.Fatalf("GET_LK should report the holder, got %+v", r)
|
||||
}
|
||||
// No conflict for the same owner.
|
||||
r = posixOp(t, fs, filer_pb.PosixLockOp_GET_LK, pbLock(30, 70, posixlock.Read, 1, 1, 7, false))
|
||||
if r.HasConflict {
|
||||
t.Fatal("an owner should not conflict with itself")
|
||||
}
|
||||
}
|
||||
|
||||
func TestPosixLockReleasePosixOwnerKeepsFlock(t *testing.T) {
|
||||
fs := newPosixTestServer()
|
||||
posixOp(t, fs, filer_pb.PosixLockOp_TRY_LOCK, pbLock(0, 99, posixlock.Write, 1, 1, 7, false))
|
||||
posixOp(t, fs, filer_pb.PosixLockOp_TRY_LOCK, pbLock(0, 1<<63, posixlock.Write, 1, 1, 7, true))
|
||||
|
||||
posixOp(t, fs, filer_pb.PosixLockOp_RELEASE_POSIX_OWNER, pbLock(0, 0, posixlock.Unlock, 1, 1, 7, false))
|
||||
|
||||
// fcntl gone, flock remains.
|
||||
if r := posixOp(t, fs, filer_pb.PosixLockOp_TRY_LOCK, pbLock(0, 99, posixlock.Write, 2, 1, 8, false)); !r.Granted {
|
||||
t.Fatal("fcntl lock should be gone after RELEASE_POSIX_OWNER")
|
||||
}
|
||||
if r := posixOp(t, fs, filer_pb.PosixLockOp_GET_LK, pbLock(0, 10, posixlock.Write, 3, 3, 9, true)); !r.HasConflict {
|
||||
t.Fatal("flock lock should survive RELEASE_POSIX_OWNER")
|
||||
}
|
||||
}
|
||||
|
||||
func TestPosixLockKeepAlive(t *testing.T) {
|
||||
fs := newPosixTestServer()
|
||||
resp, err := fs.PosixLock(context.Background(), &filer_pb.PosixLockRequest{
|
||||
Key: "s3.fuse.lock:/x", Op: filer_pb.PosixLockOp_KEEP_ALIVE,
|
||||
Lock: pbLock(0, 0, posixlock.Unlock, 7, 0, 0, false),
|
||||
})
|
||||
if err != nil || resp == nil {
|
||||
t.Fatalf("keep_alive should succeed: err=%v", err)
|
||||
}
|
||||
// A renewed session that goes stale is reaped; a never-renewed one is not.
|
||||
fs.posixLocks.TryLock("s3.fuse.lock:/x", posixlock.Range{Start: 0, End: 9, Type: posixlock.Write, Sid: 7, Owner: 1})
|
||||
if reaped := fs.posixLocks.ReapExpired(0); len(reaped) != 1 || reaped[0] != 7 {
|
||||
t.Fatalf("renewed session 7 should be reapable at ttl=0, got %v", reaped)
|
||||
}
|
||||
}
|
||||
|
||||
// A request whose key is owned by another filer is forwarded to it; the owner
|
||||
// applies it and the sender does not. The owner's ring points back at the bogus
|
||||
// sender, so without is_moved on the forwarded hop it would re-forward and fail.
|
||||
func TestPosixLockForwardsToOwner(t *testing.T) {
|
||||
const key = "s3.fuse.lock:/x"
|
||||
owner := newPosixTestServer()
|
||||
|
||||
lis, err := net.Listen("tcp", "127.0.0.1:0")
|
||||
if err != nil {
|
||||
t.Fatalf("listen: %v", err)
|
||||
}
|
||||
port := lis.Addr().(*net.TCPAddr).Port
|
||||
ownerAddr := pb.NewServerAddressWithGrpcPort(lis.Addr().String(), port)
|
||||
sender := pb.ServerAddress("127.0.0.1:1")
|
||||
|
||||
withRing(owner, ownerAddr, sender)
|
||||
owner.grpcDialOption = grpc.WithTransportCredentials(insecure.NewCredentials())
|
||||
srv := grpc.NewServer()
|
||||
filer_pb.RegisterSeaweedFilerServer(srv, owner)
|
||||
go srv.Serve(lis)
|
||||
t.Cleanup(srv.Stop)
|
||||
|
||||
self := newPosixTestServer()
|
||||
withRing(self, sender, ownerAddr)
|
||||
self.grpcDialOption = grpc.WithTransportCredentials(insecure.NewCredentials())
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second)
|
||||
defer cancel()
|
||||
resp, err := self.PosixLock(ctx, &filer_pb.PosixLockRequest{
|
||||
Key: key, Op: filer_pb.PosixLockOp_TRY_LOCK,
|
||||
Lock: pbLock(0, 99, posixlock.Write, 1, 1, 7, false),
|
||||
})
|
||||
if err != nil || !resp.GetGranted() {
|
||||
t.Fatalf("forwarded TRY_LOCK: err=%v granted=%v", err, resp.GetGranted())
|
||||
}
|
||||
|
||||
// The lock landed on the owner: a conflicting acquire there is rejected.
|
||||
if _, granted := owner.posixLocks.TryLock(key, posixlock.Range{Start: 50, End: 149, Type: posixlock.Write, Sid: 2, Owner: 1}); granted {
|
||||
t.Fatal("owner should hold the forwarded lock")
|
||||
}
|
||||
// The sender did not apply locally.
|
||||
if _, granted := self.posixLocks.TryLock(key, posixlock.Range{Start: 0, End: 99, Type: posixlock.Write, Sid: 9, Owner: 9}); !granted {
|
||||
t.Fatal("sender must forward, not apply locally")
|
||||
}
|
||||
}
|
||||
|
||||
func TestPosixLockWarmupDefersGrants(t *testing.T) {
|
||||
fs := newPosixTestServer()
|
||||
fs.posixLockReadyAt.Store(time.Now().UnixNano()) // warming up
|
||||
|
||||
// A would-be grant is deferred (not granted, no conflict) so the client retries.
|
||||
r := posixOp(t, fs, filer_pb.PosixLockOp_TRY_LOCK, pbLock(0, 99, posixlock.Write, 1, 1, 7, false))
|
||||
if r.Granted {
|
||||
t.Fatal("grant should be deferred during warm-up")
|
||||
}
|
||||
if r.HasConflict {
|
||||
t.Fatalf("deferred grant should report no conflict, got %+v", r.Conflict)
|
||||
}
|
||||
|
||||
// A lock the owner already knows about is still reported as a conflict.
|
||||
fs.posixLocks.TryLock("s3.fuse.lock:/x", posixlock.Range{Start: 0, End: 99, Type: posixlock.Write, Sid: 9, Owner: 1, Pid: 5})
|
||||
if r := posixOp(t, fs, filer_pb.PosixLockOp_TRY_LOCK, pbLock(50, 60, posixlock.Write, 1, 1, 7, false)); r.Granted || !r.HasConflict {
|
||||
t.Fatalf("known conflict should be reported during warm-up: %+v", r)
|
||||
}
|
||||
|
||||
// After warm-up, grants resume.
|
||||
posixOp(t, fs, filer_pb.PosixLockOp_UNLOCK, pbLock(0, 99, posixlock.Unlock, 9, 1, 5, false))
|
||||
fs.posixLockReadyAt.Store(time.Now().Add(-2 * posixLockWarmup).UnixNano())
|
||||
if r := posixOp(t, fs, filer_pb.PosixLockOp_TRY_LOCK, pbLock(0, 99, posixlock.Write, 1, 1, 7, false)); !r.Granted {
|
||||
t.Fatal("grant should succeed after warm-up")
|
||||
}
|
||||
}
|
||||
@@ -245,7 +245,7 @@ func (fs *FilerServer) moveSelfEntry(ctx context.Context, stream filer_pb.Seawee
|
||||
return fmt.Errorf("insert entry %s: %v", newEntry.FullPath, createErr)
|
||||
}
|
||||
} else {
|
||||
if createErr := fs.filer.CreateEntry(filer.WithSuppressedMetadataEvents(ctx), newEntry, nil, false, false, signatures, false, fs.filer.MaxFilenameLength); createErr != nil {
|
||||
if createErr := fs.filer.CreateEntry(filer.WithSuppressedMetadataEvents(ctx), newEntry, false, false, signatures, false, fs.filer.MaxFilenameLength); createErr != nil {
|
||||
return createErr
|
||||
}
|
||||
}
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user