mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-10-10 00:25:52 +00:00
Compare commits
87
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
a4bf9fe47f | ||
|
|
b3ea57b5d7 | ||
|
|
cd68313929 | ||
|
|
675020b342 | ||
|
|
7919cc7ca0 | ||
|
|
1e91a99f79 | ||
|
|
4f17c6661a | ||
|
|
29eec2f111 | ||
|
|
8fd7c524c7 | ||
|
|
77dcb20a74 | ||
|
|
dd1b428789 | ||
|
|
1355c7a102 | ||
|
|
f72c5ec5d3 | ||
|
|
96f521addc | ||
|
|
584da4cd10 | ||
|
|
56b9df937c | ||
|
|
e8ed043d2b | ||
|
|
502fef6b50 | ||
|
|
b21c263328 | ||
|
|
c9868dcf2f | ||
|
|
85ca3cb757 | ||
|
|
a3c0baa9b0 | ||
|
|
881226a81b | ||
|
|
f8caaa4464 | ||
|
|
c97b69f8a4 | ||
|
|
3976264391 | ||
|
|
3481f13f54 | ||
|
|
68cae26c0b | ||
|
|
fef49c2d75 | ||
|
|
564b94796a | ||
|
|
475ae2b443 | ||
|
|
e8e7cd6fac | ||
|
|
0f1e50f9ec | ||
|
|
2a4923e7e8 | ||
|
|
25beb7ec48 | ||
|
|
6fc212cedb | ||
|
|
1f0c366583 | ||
|
|
fa7056dc6f | ||
|
|
eeda7181aa | ||
|
|
4b9d46b5ad | ||
|
|
5bac8b9281 | ||
|
|
db954b5503 | ||
|
|
32aa70ab59 | ||
|
|
f9bc6adf98 | ||
|
|
f037fc4dce | ||
|
|
b4d2224e97 | ||
|
|
83195fc111 | ||
|
|
091aad59dc | ||
|
|
dc5621d2ae | ||
|
|
e2203b2a0b | ||
|
|
e71bac55e9 | ||
|
|
bf022ca018 | ||
|
|
b18d3dc96c | ||
|
|
bce76e6e21 | ||
|
|
21f2699624 | ||
|
|
d1665750e1 | ||
|
|
0566fbd552 | ||
|
|
d4e39b499b | ||
|
|
adfd731bb8 | ||
|
|
917a87928c | ||
|
|
8fa769f29a | ||
|
|
7c635c4508 | ||
|
|
fbdcec1cba | ||
|
|
0accff0e4a | ||
|
|
9021225591 | ||
|
|
5b42287c22 | ||
|
|
3392493f0a | ||
|
|
d82b3a8d6a | ||
|
|
39e9294907 | ||
|
|
3825035f07 | ||
|
|
83b7ea5e7b | ||
|
|
eae8f33db5 | ||
|
|
2c2b2d4d3e | ||
|
|
cd15ae1395 | ||
|
|
3f6410fdc3 | ||
|
|
87fdea5330 | ||
|
|
303c2be38d | ||
|
|
9b9fdb5b76 | ||
|
|
7e4691f2dc | ||
|
|
391f543ff2 | ||
|
|
afcc491517 | ||
|
|
a5d0e4a735 | ||
|
|
a17dca7009 | ||
|
|
024b59fb31 | ||
|
|
5af7d12f04 | ||
|
|
4385b86bf1 | ||
|
|
c00aa90990 |
@@ -128,14 +128,14 @@ jobs:
|
||||
|
||||
- name: Login to Docker Hub
|
||||
if: github.event_name != 'pull_request'
|
||||
uses: docker/login-action@v4.1.0
|
||||
uses: docker/login-action@v4.2.0
|
||||
with:
|
||||
username: ${{ secrets.DOCKER_USERNAME }}
|
||||
password: ${{ secrets.DOCKER_PASSWORD }}
|
||||
|
||||
- name: Login to GHCR
|
||||
if: github.event_name != 'pull_request'
|
||||
uses: docker/login-action@v4.1.0
|
||||
uses: docker/login-action@v4.2.0
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ secrets.GHCR_USERNAME }}
|
||||
|
||||
@@ -133,7 +133,7 @@ jobs:
|
||||
|
||||
- name: Login to Docker Hub
|
||||
if: github.event_name != 'pull_request'
|
||||
uses: docker/login-action@v4.1.0
|
||||
uses: docker/login-action@v4.2.0
|
||||
with:
|
||||
username: ${{ secrets.DOCKER_USERNAME }}
|
||||
password: ${{ secrets.DOCKER_PASSWORD }}
|
||||
|
||||
@@ -221,13 +221,13 @@ jobs:
|
||||
buildkitd-config: /tmp/buildkitd.toml
|
||||
- name: Login to Docker Hub
|
||||
if: needs.setup.outputs.publish == 'true'
|
||||
uses: docker/login-action@v4.1.0
|
||||
uses: docker/login-action@v4.2.0
|
||||
with:
|
||||
username: ${{ secrets.DOCKER_USERNAME }}
|
||||
password: ${{ secrets.DOCKER_PASSWORD }}
|
||||
- name: Login to GHCR
|
||||
if: needs.setup.outputs.publish == 'true'
|
||||
uses: docker/login-action@v4.1.0
|
||||
uses: docker/login-action@v4.2.0
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ secrets.GHCR_USERNAME }}
|
||||
@@ -275,7 +275,7 @@ jobs:
|
||||
fi
|
||||
- name: Login to GHCR
|
||||
if: needs.setup.outputs.publish == 'true'
|
||||
uses: docker/login-action@v4.1.0
|
||||
uses: docker/login-action@v4.2.0
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ secrets.GHCR_USERNAME }}
|
||||
@@ -430,12 +430,12 @@ jobs:
|
||||
ghcr.io/chrislusf/seaweedfs
|
||||
tags: type=raw,value=${{ github.event_name == 'workflow_dispatch' && github.event.inputs.image_tag || 'latest' }},suffix=${{ steps.config.outputs.tag_suffix }}
|
||||
- name: Login to Docker Hub
|
||||
uses: docker/login-action@v4.1.0
|
||||
uses: docker/login-action@v4.2.0
|
||||
with:
|
||||
username: ${{ secrets.DOCKER_USERNAME }}
|
||||
password: ${{ secrets.DOCKER_PASSWORD }}
|
||||
- name: Login to GHCR
|
||||
uses: docker/login-action@v4.1.0
|
||||
uses: docker/login-action@v4.2.0
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ secrets.GHCR_USERNAME }}
|
||||
|
||||
@@ -50,7 +50,7 @@ jobs:
|
||||
-
|
||||
name: Login to Docker Hub
|
||||
if: github.event_name != 'pull_request'
|
||||
uses: docker/login-action@v4.1.0
|
||||
uses: docker/login-action@v4.2.0
|
||||
with:
|
||||
username: ${{ secrets.DOCKER_USERNAME }}
|
||||
password: ${{ secrets.DOCKER_PASSWORD }}
|
||||
|
||||
@@ -237,14 +237,14 @@ jobs:
|
||||
|
||||
- name: Login to Docker Hub
|
||||
if: (github.event_name != 'workflow_dispatch' || github.event.inputs.variant == 'all' || github.event.inputs.variant == matrix.variant) && github.event_name != 'pull_request'
|
||||
uses: docker/login-action@v4.1.0
|
||||
uses: docker/login-action@v4.2.0
|
||||
with:
|
||||
username: ${{ secrets.DOCKER_USERNAME }}
|
||||
password: ${{ secrets.DOCKER_PASSWORD }}
|
||||
|
||||
- name: Login to GHCR
|
||||
if: (github.event_name != 'workflow_dispatch' || github.event.inputs.variant == 'all' || github.event.inputs.variant == matrix.variant) && github.event_name != 'pull_request'
|
||||
uses: docker/login-action@v4.1.0
|
||||
uses: docker/login-action@v4.2.0
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ secrets.GHCR_USERNAME }}
|
||||
@@ -300,14 +300,14 @@ jobs:
|
||||
steps:
|
||||
- name: Login to Docker Hub
|
||||
if: github.event_name != 'workflow_dispatch' || github.event.inputs.variant == 'all' || github.event.inputs.variant == matrix.variant
|
||||
uses: docker/login-action@v4.1.0
|
||||
uses: docker/login-action@v4.2.0
|
||||
with:
|
||||
username: ${{ secrets.DOCKER_USERNAME }}
|
||||
password: ${{ secrets.DOCKER_PASSWORD }}
|
||||
|
||||
- name: Login to GHCR
|
||||
if: github.event_name != 'workflow_dispatch' || github.event.inputs.variant == 'all' || github.event.inputs.variant == matrix.variant
|
||||
uses: docker/login-action@v4.1.0
|
||||
uses: docker/login-action@v4.2.0
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ secrets.GHCR_USERNAME }}
|
||||
@@ -380,7 +380,7 @@ jobs:
|
||||
variant: large_disk
|
||||
steps:
|
||||
- name: Login to GHCR
|
||||
uses: docker/login-action@v4.1.0
|
||||
uses: docker/login-action@v4.2.0
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ secrets.GHCR_USERNAME }}
|
||||
@@ -429,13 +429,13 @@ jobs:
|
||||
latest_tag: latest_large_disk
|
||||
steps:
|
||||
- name: Login to Docker Hub
|
||||
uses: docker/login-action@v4.1.0
|
||||
uses: docker/login-action@v4.2.0
|
||||
with:
|
||||
username: ${{ secrets.DOCKER_USERNAME }}
|
||||
password: ${{ secrets.DOCKER_PASSWORD }}
|
||||
|
||||
- name: Login to GHCR
|
||||
uses: docker/login-action@v4.1.0
|
||||
uses: docker/login-action@v4.2.0
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ secrets.GHCR_USERNAME }}
|
||||
|
||||
@@ -88,7 +88,7 @@ jobs:
|
||||
uses: docker/setup-buildx-action@4d04d5d9486b7bd6fa91e7baf45bbb4f8b9deedd # v1
|
||||
|
||||
- name: Login to Docker Hub
|
||||
uses: docker/login-action@4907a6ddec9925e35a0a9e82d7399ccc52663121 # v1
|
||||
uses: docker/login-action@650006c6eb7dba73a995cc03b0b2d7f5ca915bee # v1
|
||||
with:
|
||||
username: ${{ secrets.DOCKER_USERNAME }}
|
||||
password: ${{ secrets.DOCKER_PASSWORD }}
|
||||
|
||||
@@ -1,49 +0,0 @@
|
||||
name: EC Integration Tests
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [ master ]
|
||||
paths:
|
||||
- 'weed/admin/**'
|
||||
- 'weed/worker/**'
|
||||
- 'test/erasure_coding/admin_dockertest/**'
|
||||
- '.github/workflows/ec-integration.yml'
|
||||
pull_request:
|
||||
branches: [ master ]
|
||||
paths:
|
||||
- 'weed/admin/**'
|
||||
- 'weed/worker/**'
|
||||
- 'test/erasure_coding/admin_dockertest/**'
|
||||
- '.github/workflows/ec-integration.yml'
|
||||
|
||||
jobs:
|
||||
ec-integration-test:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 15
|
||||
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Set up Go
|
||||
uses: actions/setup-go@v6
|
||||
with:
|
||||
go-version-file: 'go.mod'
|
||||
|
||||
- name: Build weed binary
|
||||
run: |
|
||||
cd weed
|
||||
go build -o ../weed_bin
|
||||
|
||||
- name: Run EC integration tests
|
||||
run: |
|
||||
cd test/erasure_coding/admin_dockertest
|
||||
go test -v -timeout 15m ec_integration_test.go
|
||||
|
||||
- name: Upload test logs on failure
|
||||
if: failure()
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: ec-test-logs
|
||||
path: test/erasure_coding/admin_dockertest/tmp/logs/
|
||||
retention-days: 7
|
||||
@@ -0,0 +1,120 @@
|
||||
name: "Samba on FUSE Integration"
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [ master, main ]
|
||||
paths:
|
||||
- 'weed/mount/**'
|
||||
- 'weed/filer/**'
|
||||
- 'weed/cluster/**'
|
||||
- 'test/samba/**'
|
||||
- '.github/workflows/samba-integration.yml'
|
||||
pull_request:
|
||||
branches: [ master, main ]
|
||||
paths:
|
||||
- 'weed/mount/**'
|
||||
- 'weed/filer/**'
|
||||
- 'weed/cluster/**'
|
||||
- 'test/samba/**'
|
||||
- '.github/workflows/samba-integration.yml'
|
||||
workflow_dispatch:
|
||||
|
||||
concurrency:
|
||||
group: samba-integration/${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
samba-integration:
|
||||
name: samba-integration
|
||||
runs-on: ubuntu-22.04
|
||||
timeout-minutes: 45
|
||||
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Set up Go
|
||||
uses: actions/setup-go@v6
|
||||
with:
|
||||
go-version-file: 'go.mod'
|
||||
|
||||
- name: Start local Docker registry
|
||||
run: docker run -d --restart=always -p 5000:5000 --name registry registry:2
|
||||
|
||||
- name: Set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@v4
|
||||
with:
|
||||
driver-opts: network=host
|
||||
|
||||
- name: Build weed race binary
|
||||
run: |
|
||||
cd docker
|
||||
make binary_race
|
||||
|
||||
- name: Build SeaweedFS e2e image
|
||||
uses: docker/build-push-action@v7
|
||||
with:
|
||||
context: docker
|
||||
file: docker/Dockerfile.e2e
|
||||
tags: localhost:5000/chrislusf/seaweedfs:e2e
|
||||
push: true
|
||||
cache-from: type=gha,scope=samba-e2e
|
||||
cache-to: type=gha,mode=max,scope=samba-e2e
|
||||
|
||||
- name: Tag e2e image for docker compose
|
||||
run: |
|
||||
docker pull localhost:5000/chrislusf/seaweedfs:e2e
|
||||
docker tag localhost:5000/chrislusf/seaweedfs:e2e chrislusf/seaweedfs:e2e
|
||||
|
||||
- name: Build samba image
|
||||
uses: docker/build-push-action@v7
|
||||
with:
|
||||
context: test/samba
|
||||
build-contexts: |
|
||||
chrislusf/seaweedfs:e2e=docker-image://localhost:5000/chrislusf/seaweedfs:e2e
|
||||
tags: localhost:5000/chrislusf/seaweedfs:samba
|
||||
push: true
|
||||
cache-from: type=gha,scope=samba-harness
|
||||
cache-to: type=gha,mode=max,scope=samba-harness
|
||||
|
||||
- name: Tag samba image for docker compose
|
||||
run: |
|
||||
docker pull localhost:5000/chrislusf/seaweedfs:samba
|
||||
docker tag localhost:5000/chrislusf/seaweedfs:samba chrislusf/seaweedfs:samba
|
||||
|
||||
- name: Start SeaweedFS cluster and Samba
|
||||
run: |
|
||||
docker compose -f test/samba/docker-compose.yml up --wait
|
||||
|
||||
- name: Run Samba test battery
|
||||
run: |
|
||||
set -o pipefail
|
||||
docker compose -f test/samba/docker-compose.yml exec -T samba \
|
||||
/run_inside_container.sh 2>&1 | tee /tmp/samba-output.log
|
||||
|
||||
- name: Collect logs
|
||||
if: always()
|
||||
run: |
|
||||
mkdir -p /tmp/samba-docker-logs
|
||||
for svc in master volume filer samba; do
|
||||
docker compose -f test/samba/docker-compose.yml logs "$svc" \
|
||||
> "/tmp/samba-docker-logs/${svc}.log" 2>&1 || true
|
||||
done
|
||||
|
||||
- name: Tear down
|
||||
if: always()
|
||||
run: |
|
||||
docker compose -f test/samba/docker-compose.yml down -v
|
||||
|
||||
- name: Upload logs
|
||||
if: always()
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: samba-integration-results
|
||||
path: |
|
||||
/tmp/samba-output.log
|
||||
/tmp/samba-docker-logs/
|
||||
retention-days: 7
|
||||
@@ -0,0 +1,167 @@
|
||||
# Design: Serializing Bucket Configuration Mutations
|
||||
|
||||
Issue #9651 — concurrent `PutBucketVersioning` + `PutBucketEncryption` (as Terraform
|
||||
issues them in parallel) intermittently lose the encryption write.
|
||||
|
||||
## Root cause
|
||||
|
||||
The bucket's entire config lives in one filer entry, `/buckets/<name>`. Every
|
||||
config API does a read-modify-write of that single entry, and the writes are not
|
||||
serialized:
|
||||
|
||||
- `updateBucketConfig(bucket, fn)` (`s3api_bucket_config.go:468`) — sources from a
|
||||
possibly-stale cached `BucketConfig`, mutates `Entry.Extended`, writes the
|
||||
**whole** entry. Used by: versioning, object-lock config, lifecycle, ACL/owner.
|
||||
- `UpdateBucketMetadata` → `setBucketMetadata` (`:1042`) — reads a fresh entry,
|
||||
mutates `Entry.Content`, writes the **whole** entry. Used by: encryption, CORS,
|
||||
tagging, ownership, policy, notification.
|
||||
|
||||
Two ingredients produce the lost update:
|
||||
|
||||
1. **No serialization** of the read→modify→write (the cache mutexes only guard the
|
||||
in-memory map, not the RMW).
|
||||
2. **Whole-entry rewrite from an independent snapshot** — `updateBucketConfig`
|
||||
rebuilds from a stale cached `BucketConfig` whose `Content` predates the
|
||||
concurrent encryption write, so writing the whole entry reverts `Content`.
|
||||
|
||||
Sequential calls always pass (each sees the previous write), so it only surfaces
|
||||
under concurrency — and CI's slower IO widens the window (the "2 of ~12 runs").
|
||||
|
||||
## Goals
|
||||
|
||||
- No lost updates across concurrent bucket-config changes — for **all** config
|
||||
fields, not just versioning/encryption.
|
||||
- Correct for a single S3 gateway (the reported case) and for multiple gateways.
|
||||
- Reuse the filer primitives just merged (per-path lock, `WriteCondition`,
|
||||
`ObjectTransaction`); do not reintroduce a distributed lock.
|
||||
- Minimal blast radius: the fix lands at the two chokepoint helpers.
|
||||
|
||||
## Non-goals
|
||||
|
||||
- Changing the one-entry-per-bucket storage model.
|
||||
- Multi-filer-concurrent bucket writes (addressed only as an optional phase 3).
|
||||
|
||||
## The two ingredients map to two complementary fixes
|
||||
|
||||
### Fix A — serialize + read fresh (closes the window for whole-entry writers)
|
||||
|
||||
Both `updateBucketConfig` and `UpdateBucketMetadata` must run their RMW under one
|
||||
per-bucket critical section, and **re-read the entry fresh from the filer inside
|
||||
it** — not rebuild from the cached `BucketConfig`. The lock alone is insufficient:
|
||||
without the fresh read, two serialized writers still each apply a stale snapshot.
|
||||
|
||||
### Fix B — field-level updates (removes the collision entirely)
|
||||
|
||||
The two writers touch disjoint fields (`Extended[versioning]` vs `Content`). If
|
||||
each path updated only its own field instead of rewriting the whole entry, neither
|
||||
could clobber the other regardless of ordering. This is the structural fix and
|
||||
makes serialization a defense-in-depth concern rather than a correctness
|
||||
requirement for cross-field cases.
|
||||
|
||||
## Where to serialize (layering)
|
||||
|
||||
The bucket entry is a single filer entry, so unlike object writes there is no
|
||||
sharding — the question is purely the scope of the lock:
|
||||
|
||||
| Layer | Serializes across | Cost | Notes |
|
||||
|---|---|---|---|
|
||||
| 1. Gateway-local per-bucket lock | one gateway process | tiny | fixes the reported (single-gateway/CI) case |
|
||||
| 2. Filer per-path lock via conditional write | all gateways on one filer | small | reuses #9640 `CreateEntry`+`WriteCondition` |
|
||||
| 3. Route-by-key to bucket-key owner filer | all gateways and filers | medium | same mechanism as the object DLM-removal |
|
||||
|
||||
## Recommended plan (phased)
|
||||
|
||||
### Phase 1 — minimal fix for #9651 (gateway-local lock + fresh read)
|
||||
|
||||
Add a bounded per-bucket lock table to `S3ApiServer`, reusing the same
|
||||
`util.LockTable` the filer uses for its per-path lock:
|
||||
|
||||
```go
|
||||
// in S3ApiServer
|
||||
bucketConfigLocks *util.LockTable[string] // serialize bucket-entry RMW
|
||||
|
||||
func (s3a *S3ApiServer) withBucketConfigLock(bucket string, fn func() s3err.ErrorCode) s3err.ErrorCode {
|
||||
lk := s3a.bucketConfigLocks.AcquireLock("bucketConfig", bucket, util.ExclusiveLock)
|
||||
defer s3a.bucketConfigLocks.ReleaseLock(bucket, lk)
|
||||
return fn()
|
||||
}
|
||||
```
|
||||
|
||||
Wrap the RMW in **both** chokepoints, and inside the lock read the entry fresh:
|
||||
|
||||
- `updateBucketConfig`: acquire the lock; re-read `/buckets/<name>` from the filer
|
||||
(not the cache); rebuild `BucketConfig` from that fresh entry; apply `fn`; write;
|
||||
invalidate cache; release.
|
||||
- `UpdateBucketMetadata`/`setBucketMetadata`: same lock key; it already reads fresh,
|
||||
so it just needs to share the critical section.
|
||||
|
||||
Both must use the **same** lock keyed on `bucket`, so versioning and encryption
|
||||
contend on one mutex. This closes the reported window. Limitation: only one
|
||||
gateway; two gateways behind a load balancer still race.
|
||||
|
||||
Test: parallel `PutBucketVersioning` + `PutBucketEncryption`, assert both persist
|
||||
(the exact Terraform scenario), plus an N-way parallel variant over distinct
|
||||
fields.
|
||||
|
||||
### Phase 2 — robust across gateways (field-level + CAS via merged primitives)
|
||||
|
||||
Move the writers off whole-entry rewrites:
|
||||
|
||||
- **Extended-based config** (versioning, object-lock, ownership, tagging-in-Extended)
|
||||
→ `ObjectTransaction` `PATCH_EXTENDED` on `/buckets/<name>`. The owner filer reads
|
||||
the entry fresh under its per-path lock and merges only the named keys, so the
|
||||
gateway never sends a whole-entry snapshot — this dissolves *both* ingredients for
|
||||
these fields.
|
||||
- **`Content`-based config** (encryption, CORS, tags blob) — **chosen and
|
||||
implemented (b3): extend `PATCH_EXTENDED` with `set_content`.** Under the same
|
||||
per-path lock the filer reads the entry fresh, merges extended attributes, and
|
||||
replaces `Content`, preserving the rest. So a content write becomes a field-level
|
||||
patch too — `setBucketMetadata` patches `Content`, `updateBucketConfig` patches
|
||||
extended keys, and the two serialize on the lock instead of racing whole-entry
|
||||
rewrites. This is cleaner than the alternatives below: no client-side retry, no
|
||||
storage migration, and it reuses `ObjectTransaction`'s existing atomic lock.
|
||||
- (b1, rejected) Conditional `CreateEntry` overwrite with `IF_ETAG_MATCH` + retry
|
||||
(#9640): correct but needs client-side retry, and the bucket directory entry has
|
||||
no reliable ETag to compare on.
|
||||
- (b2, future) Migrate each per-feature config out of the single `Content` blob
|
||||
into its own `Extended` key. Then even *intra-blob* writes (tags vs encryption)
|
||||
stop racing. Larger migration; tracked separately.
|
||||
|
||||
Once all paths are field-level patches, the phase-1 gateway lock is unnecessary —
|
||||
the filer enforces atomicity. (This is the path taken: phase 1 was skipped.)
|
||||
|
||||
### Phase 3 — multi-filer (only if needed)
|
||||
|
||||
If multiple filers can write `/buckets/<name>` concurrently, a filer-local per-path
|
||||
lock no longer suffices. Route bucket-config writes to
|
||||
`PrimaryForKey("/buckets/<name>")` (the lock-ring view) and serialize on that one
|
||||
owner filer — the same route-by-key design used to take object writes off the DLM.
|
||||
Overkill for rare config writes; include only if multi-filer bucket writes are real.
|
||||
|
||||
## Correctness summary
|
||||
|
||||
- Phase 1: all RMW for a bucket serialize within a gateway; the fresh read means the
|
||||
second writer observes the first's change. Closes #9651 for single-gateway.
|
||||
- Phase 2: `PATCH_EXTENDED` is atomic field-level merge at the filer (no snapshot);
|
||||
CAS turns a concurrent `Content` write into a retry, enforced under the filer's
|
||||
per-path lock — correct for any number of gateways sharing a filer.
|
||||
- Phase 3: one owner filer serializes all writers — correct across filers too.
|
||||
|
||||
## Scope checklist (every path that RMWs the bucket entry)
|
||||
|
||||
All of these funnel through the two chokepoints, so fixing the chokepoints covers
|
||||
them — but the fix must not leave any of them on an unserialized path:
|
||||
|
||||
- via `updateBucketConfig`: versioning, object-lock config, lifecycle, ACL/owner.
|
||||
- via `UpdateBucketMetadata`/`setBucketMetadata`: encryption, CORS, tagging,
|
||||
ownership controls, bucket policy, notification.
|
||||
- bucket create/delete (`CreateEntry`/`DeleteEntry` of `/buckets/<name>`) already
|
||||
go through the filer's per-path lock on `CreateEntry`; ensure they take the same
|
||||
bucket lock if they also patch config.
|
||||
|
||||
## Cache rule (must document in code)
|
||||
|
||||
Under the lock, **read the entry from the filer, never rebuild from the cached
|
||||
`BucketConfig`**. The cache is for reads; it must be invalidated on every write and
|
||||
never be the source for an RMW. This is the single most important detail — the lock
|
||||
without the fresh read does not fix the bug.
|
||||
@@ -42,6 +42,10 @@ RUN if [ -f "/prebuilt/weed-volume-${TARGETARCH}" ]; then \
|
||||
echo "Skipping Rust build for $TARGETARCH (unsupported)" && \
|
||||
touch /weed-volume; \
|
||||
fi
|
||||
# Pre-built binaries arrive via GitHub Actions artifacts, which drop the
|
||||
# executable bit, so the copied file is 0644 and exec fails with "Permission
|
||||
# denied". Restore it (no-op for the empty placeholder, which stays size 0).
|
||||
RUN chmod 0755 /weed-volume
|
||||
|
||||
FROM alpine AS final
|
||||
LABEL author="Chris Lu"
|
||||
|
||||
@@ -26,7 +26,7 @@ require (
|
||||
github.com/facebookgo/subset v0.0.0-20200203212716-c811ad88dec4 // indirect
|
||||
github.com/fsnotify/fsnotify v1.9.0 // indirect
|
||||
github.com/go-redsync/redsync/v4 v4.16.0
|
||||
github.com/go-sql-driver/mysql v1.9.3
|
||||
github.com/go-sql-driver/mysql v1.10.0
|
||||
github.com/go-zookeeper/zk v1.0.4 // indirect
|
||||
github.com/golang/protobuf v1.5.4
|
||||
github.com/golang/snappy v1.0.0
|
||||
@@ -48,7 +48,7 @@ require (
|
||||
github.com/klauspost/compress v1.18.6
|
||||
github.com/klauspost/reedsolomon v1.14.0
|
||||
github.com/kurin/blazer v0.5.3
|
||||
github.com/linxGnu/grocksdb v1.10.7
|
||||
github.com/linxGnu/grocksdb v1.10.8
|
||||
github.com/mailru/easyjson v0.9.1 // indirect
|
||||
github.com/mattn/go-isatty v0.0.20 // indirect
|
||||
github.com/modern-go/concurrent v0.0.0-20180306012644-bacd9c7ef1dd // indirect
|
||||
@@ -91,12 +91,12 @@ require (
|
||||
gocloud.dev v0.45.0
|
||||
gocloud.dev/pubsub/natspubsub v0.45.0
|
||||
gocloud.dev/pubsub/rabbitpubsub v0.45.0
|
||||
golang.org/x/crypto v0.51.0
|
||||
golang.org/x/crypto v0.52.0
|
||||
golang.org/x/exp v0.0.0-20260410095643-746e56fc9e2f
|
||||
golang.org/x/image v0.39.0
|
||||
golang.org/x/net v0.54.0
|
||||
golang.org/x/oauth2 v0.36.0
|
||||
golang.org/x/sys v0.44.0
|
||||
golang.org/x/sys v0.45.0
|
||||
golang.org/x/text v0.37.0 // indirect
|
||||
golang.org/x/tools v0.44.0 // indirect
|
||||
golang.org/x/xerrors v0.0.0-20240903120638-7835f813f4da // indirect
|
||||
@@ -157,7 +157,7 @@ require (
|
||||
github.com/xeipuuv/gojsonschema v1.2.0
|
||||
github.com/ydb-platform/ydb-go-sdk-auth-environ v0.5.1
|
||||
github.com/ydb-platform/ydb-go-sdk/v3 v3.134.2
|
||||
go.etcd.io/etcd/client/pkg/v3 v3.6.10
|
||||
go.etcd.io/etcd/client/pkg/v3 v3.6.11
|
||||
go.uber.org/atomic v1.11.0
|
||||
golang.org/x/sync v0.20.0
|
||||
golang.org/x/tools/godoc v0.1.0-deprecated
|
||||
@@ -296,7 +296,7 @@ require (
|
||||
cloud.google.com/go/compute/metadata v0.9.0 // indirect
|
||||
cloud.google.com/go/iam v1.7.0 // indirect
|
||||
cloud.google.com/go/monitoring v1.24.3 // indirect
|
||||
filippo.io/edwards25519 v1.1.1 // indirect
|
||||
filippo.io/edwards25519 v1.2.0 // indirect
|
||||
github.com/Azure/azure-sdk-for-go/sdk/azcore v1.21.1
|
||||
github.com/Azure/azure-sdk-for-go/sdk/azidentity v1.13.1
|
||||
github.com/Azure/azure-sdk-for-go/sdk/internal v1.12.0 // indirect
|
||||
|
||||
@@ -547,8 +547,8 @@ cloud.google.com/go/workflows v1.10.0/go.mod h1:fZ8LmRmZQWacon9UCX1r/g/DfAXx5VcP
|
||||
dario.cat/mergo v1.0.2 h1:85+piFYR1tMbRrLcDwR18y4UKJ3aH1Tbzi24VRW1TK8=
|
||||
dario.cat/mergo v1.0.2/go.mod h1:E/hbnu0NxMFBjpMIE34DRGLWqDy0g5FuKDhCb31ngxA=
|
||||
dmitri.shuralyov.com/gpu/mtl v0.0.0-20190408044501-666a987793e9/go.mod h1:H6x//7gZCb22OMCxBHrMx7a5I7Hp++hsVxbQ4BYO7hU=
|
||||
filippo.io/edwards25519 v1.1.1 h1:YpjwWWlNmGIDyXOn8zLzqiD+9TyIlPhGFG96P39uBpw=
|
||||
filippo.io/edwards25519 v1.1.1/go.mod h1:BxyFTGdWcka3PhytdK4V28tE5sGfRvvvRV7EaN4VDT4=
|
||||
filippo.io/edwards25519 v1.2.0 h1:crnVqOiS4jqYleHd9vaKZ+HKtHfllngJIiOpNpoJsjo=
|
||||
filippo.io/edwards25519 v1.2.0/go.mod h1:xzAOLCNug/yB62zG1bQ8uziwrIqIuxhctzJT18Q77mc=
|
||||
gioui.org v0.0.0-20210308172011-57750fc8a0a6/go.mod h1:RSH6KIUZ0p2xy5zHDxgAM4zumjgTw83q2ge/PI+yyw8=
|
||||
git.sr.ht/~sbinet/gg v0.3.1/go.mod h1:KGYtlADtqsqANL9ueOFkWymvzUvLMQllU5Ixo+8v3pc=
|
||||
github.com/AdaLogics/go-fuzz-headers v0.0.0-20240806141605-e8a1dd7889d6 h1:He8afgbRMd7mFxO99hRNu+6tazq8nFF9lIwo9JFroBk=
|
||||
@@ -1128,8 +1128,8 @@ github.com/go-redsync/redsync/v4 v4.16.0 h1:bNcOzeHH9d3s6pghU9NJFMPrQa41f5Nx3L4Y
|
||||
github.com/go-redsync/redsync/v4 v4.16.0/go.mod h1:V4gagqgyASWBZuwx4xGzu72aZNb/6Mo05byUa3mVmKQ=
|
||||
github.com/go-resty/resty/v2 v2.17.2 h1:FQW5oHYcIlkCNrMD2lloGScxcHJ0gkjshV3qcQAyHQk=
|
||||
github.com/go-resty/resty/v2 v2.17.2/go.mod h1:kCKZ3wWmwJaNc7S29BRtUhJwy7iqmn+2mLtQrOyQlVA=
|
||||
github.com/go-sql-driver/mysql v1.9.3 h1:U/N249h2WzJ3Ukj8SowVFjdtZKfu9vlLZxjPXV1aweo=
|
||||
github.com/go-sql-driver/mysql v1.9.3/go.mod h1:qn46aNg1333BRMNU69Lq93t8du/dwxI64Gl8i5p1WMU=
|
||||
github.com/go-sql-driver/mysql v1.10.0 h1:Q+1LV8DkHJvSYAdR83XzuhDaTykuDx0l6fkXxoWCWfw=
|
||||
github.com/go-sql-driver/mysql v1.10.0/go.mod h1:M+cqaI7+xxXGG9swrdeUIoPG3Y3KCkF0pZej+SK+nWk=
|
||||
github.com/go-stack/stack v1.8.0/go.mod h1:v0f6uXyyMGvRgIKkXu+yp6POWl0qKG85gN/melR3HDY=
|
||||
github.com/go-task/slim-sprig v0.0.0-20230315185526-52ccab3ef572 h1:tfuBGBXKqDEevZMzYi5KSi8KkcZtzBcTgAUUtapy0OI=
|
||||
github.com/go-task/slim-sprig/v3 v3.0.0 h1:sUs3vkvUymDpBKi3qH1YSqBQk9+9D/8M2mN1vB6EwHI=
|
||||
@@ -1515,8 +1515,8 @@ github.com/lib/pq v1.11.1 h1:wuChtj2hfsGmmx3nf1m7xC2XpK6OtelS2shMY+bGMtI=
|
||||
github.com/lib/pq v1.11.1/go.mod h1:/p+8NSbOcwzAEI7wiMXFlgydTwcgTr3OSKMsD2BitpA=
|
||||
github.com/linkedin/goavro/v2 v2.15.0 h1:pDj1UrjUOO62iXhgBiE7jQkpNIc5/tA5eZsgolMjgVI=
|
||||
github.com/linkedin/goavro/v2 v2.15.0/go.mod h1:KXx+erlq+RPlGSPmLF7xGo6SAbh8sCQ53x064+ioxhk=
|
||||
github.com/linxGnu/grocksdb v1.10.7 h1:fCi4qvZWo04VgFwGWmO8HQJgUVounJBy+C2TMVPU/ho=
|
||||
github.com/linxGnu/grocksdb v1.10.7/go.mod h1:OLQKZwiKwaJiAVCsOzWKvwiLwfZ5Vz8Md5TYR7t7pM8=
|
||||
github.com/linxGnu/grocksdb v1.10.8 h1:Nau01Hhm/0kaVTR6d4viwD6npYbnDvZAfzwJCLzKRYo=
|
||||
github.com/linxGnu/grocksdb v1.10.8/go.mod h1:OLQKZwiKwaJiAVCsOzWKvwiLwfZ5Vz8Md5TYR7t7pM8=
|
||||
github.com/lithammer/fuzzysearch v1.1.8 h1:/HIuJnjHuXS8bKaiTMeeDlW2/AyIWk2brx1V8LFgLN4=
|
||||
github.com/lithammer/fuzzysearch v1.1.8/go.mod h1:IdqeyBClc3FFqSzYq/MXESsS4S0FsZ5ajtkr5xPLts4=
|
||||
github.com/lithammer/shortuuid/v3 v3.0.7 h1:trX0KTHy4Pbwo/6ia8fscyHoGA+mf1jWbPJVuvyJQQ8=
|
||||
@@ -2111,8 +2111,8 @@ go.etcd.io/bbolt v1.4.3 h1:dEadXpI6G79deX5prL3QRNP6JB8UxVkqo4UPnHaNXJo=
|
||||
go.etcd.io/bbolt v1.4.3/go.mod h1:tKQlpPaYCVFctUIgFKFnAlvbmB3tpy1vkTnDWohtc0E=
|
||||
go.etcd.io/etcd/api/v3 v3.6.10 h1:jlwjtELjA8yi2VWpOFH+0w0lGr3K6mVDyn0RDB9aaAY=
|
||||
go.etcd.io/etcd/api/v3 v3.6.10/go.mod h1:pdV4VeFmvhdNjB4LWRkC8ReLyRBAxUOze3GarMhE2sk=
|
||||
go.etcd.io/etcd/client/pkg/v3 v3.6.10 h1:tBT7podcPhuVbCVkAEzx8bC5I+aqxfLwBN8/As1arrA=
|
||||
go.etcd.io/etcd/client/pkg/v3 v3.6.10/go.mod h1:WEy3PpwbbEBVRdh1NVJYsuUe/8eyI21PNJRazeD8z/Y=
|
||||
go.etcd.io/etcd/client/pkg/v3 v3.6.11 h1:e41mp315Yn3QMGPmEzCyLsMINgJXTY/dX8kM++1csxU=
|
||||
go.etcd.io/etcd/client/pkg/v3 v3.6.11/go.mod h1:DysuMe/inqRyC/1tjRR6hReH/VV9Lufs27YKSKBWWJg=
|
||||
go.etcd.io/etcd/client/v3 v3.6.10 h1:J598zJ+C/ZPvImypmq5waj84+bovePrlZERHklf34y0=
|
||||
go.etcd.io/etcd/client/v3 v3.6.10/go.mod h1:iHhUDUcEwaKs1YFq3MgmI9U4zhTVasp/vgdVbFf1RS8=
|
||||
go.mongodb.org/mongo-driver v1.17.9 h1:IexDdCuuNJ3BHrELgBlyaH9p60JXAvdzWR128q+U5tU=
|
||||
@@ -2216,8 +2216,8 @@ golang.org/x/crypto v0.14.0/go.mod h1:MVFd36DqK4CsrnJYDkBA3VC4m2GkXAM0PvzMCn4JQf
|
||||
golang.org/x/crypto v0.19.0/go.mod h1:Iy9bg/ha4yyC70EfRS8jz+B6ybOBKMaSxLj6P6oBDfU=
|
||||
golang.org/x/crypto v0.23.0/go.mod h1:CKFgDieR+mRhux2Lsu27y0fO304Db0wZe70UKqHu0v8=
|
||||
golang.org/x/crypto v0.31.0/go.mod h1:kDsLvtWBEx7MV9tJOj9bnXsPbxwJQ6csT/x4KIN4Ssk=
|
||||
golang.org/x/crypto v0.51.0 h1:IBPXwPfKxY7cWQZ38ZCIRPI50YLeevDLlLnyC5wRGTI=
|
||||
golang.org/x/crypto v0.51.0/go.mod h1:8AdwkbraGNABw2kOX6YFPs3WM22XqI4EXEd8g+x7Oc8=
|
||||
golang.org/x/crypto v0.52.0 h1:RMs7fP2rXdep0CftQlK8Uf+kibLm7qkCcradZWYz988=
|
||||
golang.org/x/crypto v0.52.0/go.mod h1:1QgfPxDqh0T2M/elOJtp9RvuR95kVjir0e6/BvEmGbc=
|
||||
golang.org/x/exp v0.0.0-20180321215751-8460e604b9de/go.mod h1:CJ0aWSM057203Lf6IL+f9T1iT9GByDxfZKAQTCR3kQA=
|
||||
golang.org/x/exp v0.0.0-20180807140117-3d87b88a115f/go.mod h1:CJ0aWSM057203Lf6IL+f9T1iT9GByDxfZKAQTCR3kQA=
|
||||
golang.org/x/exp v0.0.0-20190121172915-509febef88a4/go.mod h1:CJ0aWSM057203Lf6IL+f9T1iT9GByDxfZKAQTCR3kQA=
|
||||
@@ -2511,8 +2511,8 @@ golang.org/x/sys v0.13.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.17.0/go.mod h1:/VUhepiaJMQUp4+oa/7Zr1D23ma6VTLIYjOOTFZPUcA=
|
||||
golang.org/x/sys v0.20.0/go.mod h1:/VUhepiaJMQUp4+oa/7Zr1D23ma6VTLIYjOOTFZPUcA=
|
||||
golang.org/x/sys v0.28.0/go.mod h1:/VUhepiaJMQUp4+oa/7Zr1D23ma6VTLIYjOOTFZPUcA=
|
||||
golang.org/x/sys v0.44.0 h1:ildZl3J4uzeKP07r2F++Op7E9B29JRUy+a27EibtBTQ=
|
||||
golang.org/x/sys v0.44.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw=
|
||||
golang.org/x/sys v0.45.0 h1:dO4czNzziLiiXplLQgBCEpCvXQ3dnkn0SdaZSYdQ+FY=
|
||||
golang.org/x/sys v0.45.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw=
|
||||
golang.org/x/telemetry v0.0.0-20240228155512-f48c80bd79b2/go.mod h1:TeRTkGYfJXctD9OcfyVLyj2J3IxLnKwHJR8f4D8a3YE=
|
||||
golang.org/x/telemetry v0.0.0-20260409153401-be6f6cb8b1fa h1:efT73AJZfAAUV7SOip6pWGkwJDzIGiKBZGVzHYa+ve4=
|
||||
golang.org/x/telemetry v0.0.0-20260409153401-be6f6cb8b1fa/go.mod h1:kHjTxDEnAu6/Nl9lDkzjWpR+bmKfxeiRuSDlsMb70gE=
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
apiVersion: v1
|
||||
description: SeaweedFS
|
||||
name: seaweedfs
|
||||
appVersion: "4.27"
|
||||
appVersion: "4.29"
|
||||
# Dev note: Trigger a helm chart release by `git tag -a helm-<version>`
|
||||
version: 4.27.0
|
||||
version: 4.29.0
|
||||
|
||||
@@ -137,6 +137,9 @@ spec:
|
||||
- "/bin/sh"
|
||||
- "-ec"
|
||||
- |
|
||||
{{- if $volume.rust }}
|
||||
exec /usr/bin/weed-volume \
|
||||
{{- else }}
|
||||
exec /usr/bin/weed \
|
||||
{{- if $volume.logs }}
|
||||
-logdir=/logs \
|
||||
@@ -149,6 +152,7 @@ spec:
|
||||
-v={{ $.Values.global.seaweedfs.loggingLevel }} \
|
||||
{{- end }}
|
||||
volume \
|
||||
{{- end }}
|
||||
-port={{ $volume.port }} \
|
||||
{{- if $volume.metricsPort }}
|
||||
-metricsPort={{ $volume.metricsPort }} \
|
||||
@@ -180,7 +184,7 @@ spec:
|
||||
{{- if $volume.imagesFixOrientation }}
|
||||
-images.fix.orientation \
|
||||
{{- end }}
|
||||
{{- if $volume.pulseSeconds }}
|
||||
{{- if and $volume.pulseSeconds (not $volume.rust) }}
|
||||
-pulseSeconds={{ $volume.pulseSeconds }} \
|
||||
{{- end }}
|
||||
{{- if $volume.index }}
|
||||
|
||||
@@ -305,6 +305,11 @@ volume:
|
||||
enabled: true
|
||||
imageOverride: null
|
||||
restartPolicy: null
|
||||
# Run the Rust volume server (/usr/bin/weed-volume) instead of the Go one.
|
||||
# Requires an image that ships the Rust binary (amd64/arm64). The Go-only
|
||||
# log flags (-logtostderr/-logdir/-v) and -pulseSeconds are dropped; set log
|
||||
# level via the RUST_LOG env var in extraEnvironmentVars if needed.
|
||||
rust: false
|
||||
port: 8080
|
||||
grpcPort: 18080
|
||||
metricsPort: 9327
|
||||
|
||||
@@ -31,6 +31,15 @@ service SeaweedFiler {
|
||||
rpc DeleteEntry (DeleteEntryRequest) returns (DeleteEntryResponse) {
|
||||
}
|
||||
|
||||
rpc ObjectTransaction (ObjectTransactionRequest) returns (ObjectTransactionResponse) {
|
||||
}
|
||||
|
||||
rpc ObjectTransactionBatch (ObjectTransactionBatchRequest) returns (ObjectTransactionBatchResponse) {
|
||||
}
|
||||
|
||||
rpc PosixLock (PosixLockRequest) returns (PosixLockResponse) {
|
||||
}
|
||||
|
||||
rpc AtomicRenameEntry (AtomicRenameEntryRequest) returns (AtomicRenameEntryResponse) {
|
||||
}
|
||||
rpc StreamRenameEntry (StreamRenameEntryRequest) returns (stream StreamRenameEntryResponse) {
|
||||
@@ -222,6 +231,56 @@ message CreateEntryRequest {
|
||||
bool is_from_other_cluster = 4;
|
||||
repeated int32 signatures = 5;
|
||||
bool skip_check_parent_directory = 6;
|
||||
// Optional precondition evaluated against the current entry atomically with
|
||||
// the write, under the filer's per-path lock. The caller must route the
|
||||
// key's writes to this entry's owner filer for the check to be authoritative.
|
||||
WriteCondition condition = 7;
|
||||
}
|
||||
|
||||
// WriteCondition is the precondition the filer evaluates against the existing
|
||||
// entry before writing, under the per-path lock. A failed condition returns
|
||||
// FilerError PRECONDITION_FAILED. The client maps request semantics (e.g. RFC
|
||||
// 7232) to clauses; the filer just compares.
|
||||
//
|
||||
// A condition is a list of clauses that ALL must hold (logical AND). One clause
|
||||
// is the common case; several express what a single comparison cannot: an ETag
|
||||
// set (If-Match / If-None-Match with multiple values), weak-ETag comparison, and
|
||||
// compound conditions (e.g. If-Match + If-Unmodified-Since together).
|
||||
message WriteCondition {
|
||||
enum Kind {
|
||||
NONE = 0; // unconditional
|
||||
IF_NOT_EXISTS = 1; // fail if the entry exists (If-None-Match: *)
|
||||
IF_EXISTS = 2; // fail if the entry is absent (If-Match: *)
|
||||
IF_ETAG_MATCH = 3; // fail if absent or etag matches none of the set (If-Match)
|
||||
IF_ETAG_NOT_MATCH = 4; // fail if present and etag matches any of the set (If-None-Match)
|
||||
IF_UNMODIFIED_SINCE = 5; // fail if present and mtime > unix_time
|
||||
IF_MODIFIED_SINCE = 6; // fail if present and mtime <= unix_time
|
||||
IF_EXTENDED_NOT_EQUAL = 7; // fail if present and extended[ext_key] == ext_value
|
||||
IF_EXTENDED_TIME_ELAPSED = 8; // fail if present and extended[ext_key] (unix seconds) is in the future
|
||||
}
|
||||
// Clause is one primitive comparison. IF_ETAG_MATCH holds when the current
|
||||
// entry's ETag equals any value in etags; IF_ETAG_NOT_MATCH holds when it
|
||||
// equals none. allow_weak permits weak-comparison (ignoring the W/ prefix).
|
||||
//
|
||||
// The IF_EXTENDED_* kinds are generic guards on an extended attribute, used
|
||||
// to enforce object-lock without teaching the filer S3 semantics:
|
||||
// IF_EXTENDED_NOT_EQUAL expresses a legal hold (block while a key equals a
|
||||
// value), and IF_EXTENDED_TIME_ELAPSED expresses retention (block while a
|
||||
// stored unix-second deadline is in the future, compared to the filer's
|
||||
// clock). The caller composes these and, for governance-bypass, simply omits
|
||||
// the retention clause when the bypass is authorized — the filer makes no
|
||||
// authorization decision.
|
||||
message Clause {
|
||||
Kind kind = 1;
|
||||
repeated string etags = 2; // ETag set for IF_ETAG_* kinds
|
||||
int64 unix_time = 3; // bound (unix seconds) for IF_*_SINCE kinds
|
||||
bool allow_weak = 4; // compare ETags ignoring the weak (W/) marker
|
||||
string ext_key = 5; // extended attribute name for IF_EXTENDED_* kinds
|
||||
string ext_value = 6; // blocking value for IF_EXTENDED_NOT_EQUAL
|
||||
string gate_key = 7; // IF_EXTENDED_TIME_ELAPSED: only enforce when extended[gate_key] == gate_value
|
||||
string gate_value = 8; // gate value (e.g. retention mode COMPLIANCE for governance bypass)
|
||||
}
|
||||
repeated Clause clauses = 1; // all must hold (logical AND)
|
||||
}
|
||||
|
||||
// Structured error codes for filer entry operations.
|
||||
@@ -233,6 +292,132 @@ enum FilerError {
|
||||
EXISTING_IS_DIRECTORY = 3; // cannot overwrite directory with file
|
||||
EXISTING_IS_FILE = 4; // cannot overwrite file with directory
|
||||
ENTRY_ALREADY_EXISTS = 5; // O_EXCL and entry already exists
|
||||
PRECONDITION_FAILED = 6; // WriteCondition not satisfied
|
||||
}
|
||||
|
||||
// ObjectMutation is one entry-level change applied by ObjectTransaction. All
|
||||
// mutations of a transaction run under a single per-path lock (the request's
|
||||
// lock_key) and in order, so the gateway can describe a multi-entry object
|
||||
// operation as one request instead of holding a distributed lock across
|
||||
// several RPCs. Data-bearing writes (entries with chunks) should be written
|
||||
// before the transaction; mutations here are metadata-scoped.
|
||||
message ObjectMutation {
|
||||
enum Type {
|
||||
PUT = 0; // create or replace the entry (entry field)
|
||||
DELETE = 1; // delete the entry at directory/name (no error if absent)
|
||||
PATCH_EXTENDED = 2; // merge set_extended / remove delete_extended on the entry
|
||||
RECOMPUTE_LATEST = 3; // scan a directory and re-point a parent entry (recompute)
|
||||
}
|
||||
Type type = 1;
|
||||
string directory = 2;
|
||||
string name = 3; // entry name for DELETE / PATCH_EXTENDED / RECOMPUTE_LATEST (the pointer entry)
|
||||
Entry entry = 4; // full entry for PUT
|
||||
map<string, bytes> set_extended = 5; // PATCH_EXTENDED: keys to set
|
||||
repeated string delete_extended = 6; // PATCH_EXTENDED: keys to remove
|
||||
bool is_delete_data = 7; // DELETE: also delete chunk data
|
||||
bool is_recursive = 8; // DELETE: recurse into a directory
|
||||
Recompute recompute = 9; // RECOMPUTE_LATEST parameters
|
||||
bool set_content = 10; // PATCH_EXTENDED: replace Entry.content with content
|
||||
bytes content = 11; // PATCH_EXTENDED: new Entry.content when set_content
|
||||
bool touch_mtime = 12; // PATCH_EXTENDED: set the entry's Mtime to now (e.g. a metadata-replace copy)
|
||||
}
|
||||
|
||||
// Recompute re-derives a pointer entry (directory/name on the mutation) from the
|
||||
// current contents of a scanned directory, atomically under the transaction's
|
||||
// lock. It is mechanical: the filer picks the child that sorts first or last by
|
||||
// name and copies the requested fields into the pointer; it has no knowledge of
|
||||
// what the entries mean. The caller (which does know the versioning scheme)
|
||||
// supplies the sort direction and the key mappings. This covers re-pointing the
|
||||
// latest version after a specific version is deleted, where the scan must run
|
||||
// under the lock.
|
||||
message Recompute {
|
||||
string scan_dir = 1; // directory whose direct children are scanned
|
||||
bool descending = 2; // pick the child that sorts last by name (else first)
|
||||
map<string, string> copy_extended = 3; // pointer extended key -> source extended key on the chosen child
|
||||
string name_to_key = 4; // if set, store the chosen child's name under this pointer key
|
||||
string size_to_key = 5; // if set, store the chosen child's FileSize (decimal) under this pointer key
|
||||
string mtime_to_key = 6; // if set, store the chosen child's Mtime (decimal) under this pointer key
|
||||
string demote_key = 7; // if set, stamp demote_value on the prior name_to_key target when it changes
|
||||
bytes demote_value = 8; // value for demote_key
|
||||
string exclude_name = 9; // if set, skip this child when scanning (e.g. a version about to be deleted)
|
||||
}
|
||||
|
||||
// ObjectTransactionRequest applies an ordered list of mutations atomically with
|
||||
// respect to other writers of the same object, by holding the filer's per-path
|
||||
// lock on lock_key for the whole transaction. The optional condition is checked
|
||||
// first, against condition_key when set, else lock_key. Callers set route_key to
|
||||
// the object's stable owner ring key; a filer that is not the owner forwards the
|
||||
// transaction one hop to the owner, so a stale ring view is tolerated.
|
||||
message ObjectTransactionRequest {
|
||||
string lock_key = 1; // object path to lock and to evaluate the condition against
|
||||
WriteCondition condition = 2; // optional precondition, checked under the lock
|
||||
repeated ObjectMutation mutations = 3;
|
||||
bool is_from_other_cluster = 4;
|
||||
repeated int32 signatures = 5;
|
||||
string condition_key = 6; // if set, evaluate the condition against this entry instead of lock_key (still locking lock_key)
|
||||
string route_key = 7; // ring key identifying the owner filer; a non-owner forwards the whole transaction to it
|
||||
bool is_moved = 8; // set on a forwarded transaction so the receiver applies it locally instead of forwarding again
|
||||
}
|
||||
|
||||
message ObjectTransactionResponse {
|
||||
string error = 1;
|
||||
FilerError error_code = 2;
|
||||
}
|
||||
|
||||
// PosixLockRange is one advisory byte-range lock. Owner identity is (sid, owner):
|
||||
// sid is the mount session, owner the FUSE lock owner within it, so owners from
|
||||
// different mounts never alias. end is inclusive (max uint64 = to EOF); is_flock
|
||||
// separates the flock and fcntl namespaces, which never conflict.
|
||||
message PosixLockRange {
|
||||
uint64 start = 1;
|
||||
uint64 end = 2;
|
||||
uint32 type = 3; // 1=read, 2=write, 3=unlock
|
||||
uint64 sid = 4;
|
||||
uint64 owner = 5;
|
||||
uint32 pid = 6; // holder pid, for get_lk reporting only
|
||||
bool is_flock = 7;
|
||||
}
|
||||
|
||||
// PosixLock routes an advisory lock operation to the inode's owner filer, which
|
||||
// holds the authoritative in-memory lock table. key is the inode identity ring
|
||||
// key (the file path, or hl:<HardLinkId> for a hardlink) used both to resolve the
|
||||
// owner and to index the table. A non-owner filer forwards the request one hop;
|
||||
// is_moved bounds it so a stale ring view cannot loop.
|
||||
message PosixLockRequest {
|
||||
string key = 1;
|
||||
bool is_moved = 2;
|
||||
PosixLockOp op = 3;
|
||||
PosixLockRange lock = 4;
|
||||
repeated PosixLockRange locks = 5;
|
||||
bool cooling_probe = 6;
|
||||
}
|
||||
|
||||
enum PosixLockOp {
|
||||
TRY_LOCK = 0; // grant lock or report conflict (non-blocking)
|
||||
UNLOCK = 1; // release lock's owner's locks over its range
|
||||
GET_LK = 2; // report a conflicting lock, if any
|
||||
RELEASE_POSIX_OWNER = 3; // drop the owner's fcntl locks (flush-time)
|
||||
RELEASE_FLOCK_OWNER = 4; // drop the owner's flock locks (release-time)
|
||||
KEEP_ALIVE = 5; // renew the session's lease on this owner (lock.sid)
|
||||
}
|
||||
|
||||
message PosixLockResponse {
|
||||
bool granted = 1; // for TRY_LOCK: whether the lock was granted
|
||||
bool has_conflict = 2; // whether conflict is populated
|
||||
PosixLockRange conflict = 3; // the blocking lock (TRY_LOCK conflict / GET_LK result)
|
||||
}
|
||||
|
||||
// ObjectTransactionBatch applies several object transactions in one round trip,
|
||||
// each under its own per-path lock and independent of the others (no cross-key
|
||||
// atomicity). A caller groups keys that route to the same owner filer and sends
|
||||
// one batch per owner, e.g. for a multi-object delete. Each response is parallel
|
||||
// to its request.
|
||||
message ObjectTransactionBatchRequest {
|
||||
repeated ObjectTransactionRequest transactions = 1;
|
||||
}
|
||||
|
||||
message ObjectTransactionBatchResponse {
|
||||
repeated ObjectTransactionResponse responses = 1;
|
||||
}
|
||||
|
||||
message CreateEntryResponse {
|
||||
@@ -421,6 +606,7 @@ message SubscribeMetadataRequest {
|
||||
repeated string directories = 10; // exact directory to watch
|
||||
bool client_supports_batching = 11; // client can unpack SubscribeMetadataResponse.events
|
||||
bool client_supports_metadata_chunks = 12; // client can read log file chunks from volume servers
|
||||
bool client_supports_idle_heartbeat = 13; // server may send empty responses carrying the current time while the client is caught up
|
||||
}
|
||||
message SubscribeMetadataResponse {
|
||||
string directory = 1;
|
||||
|
||||
@@ -838,8 +838,6 @@ pub struct ReadQueryParams {
|
||||
pub response_content_disposition: Option<String>,
|
||||
/// Pretty print JSON response
|
||||
pub pretty: Option<String>,
|
||||
/// JSONP callback function name
|
||||
pub callback: Option<String>,
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
@@ -3402,10 +3400,6 @@ fn json_response_with_params<T: Serialize>(
|
||||
let is_pretty = params
|
||||
.and_then(|params| params.pretty.as_ref())
|
||||
.is_some_and(|value| !value.is_empty());
|
||||
let callback = params
|
||||
.and_then(|params| params.callback.as_ref())
|
||||
.filter(|value| !value.is_empty())
|
||||
.cloned();
|
||||
|
||||
let json_body = if is_pretty {
|
||||
to_pretty_json(body)
|
||||
@@ -3413,24 +3407,15 @@ fn json_response_with_params<T: Serialize>(
|
||||
serde_json::to_string(body).unwrap()
|
||||
};
|
||||
|
||||
if let Some(callback) = callback {
|
||||
Response::builder()
|
||||
.status(status)
|
||||
.header(header::CONTENT_TYPE, "application/javascript")
|
||||
.body(Body::from(format!("{}({})", callback, json_body)))
|
||||
.unwrap()
|
||||
} else {
|
||||
Response::builder()
|
||||
.status(status)
|
||||
.header(header::CONTENT_TYPE, "application/json")
|
||||
.body(Body::from(json_body))
|
||||
.unwrap()
|
||||
}
|
||||
Response::builder()
|
||||
.status(status)
|
||||
.header(header::CONTENT_TYPE, "application/json")
|
||||
.header("X-Content-Type-Options", "nosniff")
|
||||
.body(Body::from(json_body))
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
/// Return a JSON error response with optional query string for pretty/JSONP support.
|
||||
/// Supports `?pretty=<any non-empty value>` for pretty-printed JSON and `?callback=fn` for JSONP,
|
||||
/// matching Go's writeJsonError behavior.
|
||||
/// Return a JSON error response, honoring `?pretty=<any non-empty value>` for pretty-printed JSON.
|
||||
pub(super) fn json_error_with_query(
|
||||
status: StatusCode,
|
||||
msg: impl Into<String>,
|
||||
@@ -3438,18 +3423,10 @@ pub(super) fn json_error_with_query(
|
||||
) -> Response {
|
||||
let body = serde_json::json!({"error": msg.into()});
|
||||
|
||||
let (is_pretty, callback) = if let Some(q) = query {
|
||||
let pretty = q
|
||||
.split('&')
|
||||
.any(|p| p.starts_with("pretty=") && p.len() > "pretty=".len());
|
||||
let cb = q
|
||||
.split('&')
|
||||
.find_map(|p| p.strip_prefix("callback="))
|
||||
.map(|s| s.to_string());
|
||||
(pretty, cb)
|
||||
} else {
|
||||
(false, None)
|
||||
};
|
||||
let is_pretty = query.is_some_and(|q| {
|
||||
q.split('&')
|
||||
.any(|p| p.starts_with("pretty=") && p.len() > "pretty=".len())
|
||||
});
|
||||
|
||||
let json_body = if is_pretty {
|
||||
to_pretty_json(&body)
|
||||
@@ -3457,35 +3434,19 @@ pub(super) fn json_error_with_query(
|
||||
serde_json::to_string(&body).unwrap()
|
||||
};
|
||||
|
||||
if let Some(cb) = callback {
|
||||
let jsonp = format!("{}({})", cb, json_body);
|
||||
Response::builder()
|
||||
.status(status)
|
||||
.header(header::CONTENT_TYPE, "application/javascript")
|
||||
.body(Body::from(jsonp))
|
||||
.unwrap()
|
||||
} else {
|
||||
Response::builder()
|
||||
.status(status)
|
||||
.header(header::CONTENT_TYPE, "application/json")
|
||||
.body(Body::from(json_body))
|
||||
.unwrap()
|
||||
}
|
||||
Response::builder()
|
||||
.status(status)
|
||||
.header(header::CONTENT_TYPE, "application/json")
|
||||
.header("X-Content-Type-Options", "nosniff")
|
||||
.body(Body::from(json_body))
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
/// Return a JSON response with optional pretty/JSONP support from raw query string.
|
||||
/// Matches Go's writeJsonQuiet behavior for write success responses.
|
||||
/// Return a JSON response honoring `?pretty=<any non-empty value>` from a raw query string.
|
||||
fn json_result_with_query<T: Serialize>(status: StatusCode, body: &T, query: &str) -> Response {
|
||||
let (is_pretty, callback) = {
|
||||
let pretty = query
|
||||
.split('&')
|
||||
.any(|p| p.starts_with("pretty=") && p.len() > "pretty=".len());
|
||||
let cb = query
|
||||
.split('&')
|
||||
.find_map(|p| p.strip_prefix("callback="))
|
||||
.map(|s| s.to_string());
|
||||
(pretty, cb)
|
||||
};
|
||||
let is_pretty = query
|
||||
.split('&')
|
||||
.any(|p| p.starts_with("pretty=") && p.len() > "pretty=".len());
|
||||
|
||||
let json_body = if is_pretty {
|
||||
to_pretty_json(body)
|
||||
@@ -3493,20 +3454,12 @@ fn json_result_with_query<T: Serialize>(status: StatusCode, body: &T, query: &st
|
||||
serde_json::to_string(body).unwrap()
|
||||
};
|
||||
|
||||
if let Some(cb) = callback {
|
||||
let jsonp = format!("{}({})", cb, json_body);
|
||||
Response::builder()
|
||||
.status(status)
|
||||
.header(header::CONTENT_TYPE, "application/javascript")
|
||||
.body(Body::from(jsonp))
|
||||
.unwrap()
|
||||
} else {
|
||||
Response::builder()
|
||||
.status(status)
|
||||
.header(header::CONTENT_TYPE, "application/json")
|
||||
.body(Body::from(json_body))
|
||||
.unwrap()
|
||||
}
|
||||
Response::builder()
|
||||
.status(status)
|
||||
.header(header::CONTENT_TYPE, "application/json")
|
||||
.header("X-Content-Type-Options", "nosniff")
|
||||
.body(Body::from(json_body))
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
/// Extract JWT token from query param, Authorization header, or Cookie.
|
||||
|
||||
@@ -3643,6 +3643,67 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_scrub_empty_volume() {
|
||||
// Mirror of Go's TestScrubVolumeData "zero-size volume without index"
|
||||
// case (weed/storage/volume_checking_test.go): a freshly created /
|
||||
// pre-allocated volume has a superblock-only .dat and a zero-size .idx,
|
||||
// and must scrub clean instead of being flagged as corrupt.
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
let v = make_test_volume(dir);
|
||||
|
||||
// .dat holds only the superblock; .idx is empty.
|
||||
assert_eq!(v.dat_file_size().unwrap(), SUPER_BLOCK_SIZE as u64);
|
||||
|
||||
let (files_checked, broken) = v.scrub().unwrap();
|
||||
assert_eq!(files_checked, 0);
|
||||
assert!(
|
||||
broken.is_empty(),
|
||||
"empty volume should scrub clean, got {:?}",
|
||||
broken
|
||||
);
|
||||
|
||||
// The index-only mode must agree.
|
||||
let (idx_checked, idx_broken) = v.scrub_index().unwrap();
|
||||
assert_eq!(idx_checked, 0);
|
||||
assert!(
|
||||
idx_broken.is_empty(),
|
||||
"empty volume should scrub_index clean, got {:?}",
|
||||
idx_broken
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_scrub_healthy_volume() {
|
||||
// Mirror of Go's TestScrubVolumeData "healthy volume" case: a volume
|
||||
// with live needles scrubs clean and the .dat size accounting matches.
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
let mut v = make_test_volume(dir);
|
||||
|
||||
for i in 1..=5 {
|
||||
let data = format!("needle data {}", i);
|
||||
let mut n = Needle {
|
||||
id: NeedleId(i),
|
||||
cookie: Cookie(i as u32),
|
||||
data: data.as_bytes().to_vec(),
|
||||
data_size: data.len() as u32,
|
||||
..Needle::default()
|
||||
};
|
||||
v.write_needle(&mut n, true).unwrap();
|
||||
}
|
||||
v.sync_to_disk().unwrap();
|
||||
|
||||
let (files_checked, broken) = v.scrub().unwrap();
|
||||
assert_eq!(files_checked, 5);
|
||||
assert!(
|
||||
broken.is_empty(),
|
||||
"healthy volume should scrub clean, got {:?}",
|
||||
broken
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_volume_multiple_needles() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
|
||||
@@ -0,0 +1,265 @@
|
||||
package erasure_coding
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"path/filepath"
|
||||
"regexp"
|
||||
"strconv"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/shell"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/needle"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
"google.golang.org/grpc"
|
||||
)
|
||||
|
||||
// TestMultiDiskECBalanceNoShardLoss is the end-to-end regression for issue 9593.
|
||||
// It runs a real cluster of multi-disk volume servers (3 servers x 4 disks),
|
||||
// EC-encodes a volume, then runs ec.balance, asserting hard invariants the older
|
||||
// integration tests only logged:
|
||||
//
|
||||
// - after encode the full set of 14 EC shards exists,
|
||||
// - ec.balance never loses a shard (still 14 distinct shards afterwards),
|
||||
// - shards end up spread across more than one disk per node, and
|
||||
// - cluster.status counts physical disks (not one per node) and matches the
|
||||
// real on-disk distribution.
|
||||
func TestMultiDiskECBalanceNoShardLoss(t *testing.T) {
|
||||
if testing.Short() {
|
||||
t.Skip("Skipping multi-disk EC integration test in short mode")
|
||||
}
|
||||
|
||||
testDir := t.TempDir()
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 240*time.Second)
|
||||
defer cancel()
|
||||
|
||||
cluster, err := startMultiDiskCluster(ctx, testDir)
|
||||
require.NoError(t, err)
|
||||
defer cluster.Stop()
|
||||
|
||||
require.NoError(t, waitForServer("127.0.0.1:9334", 30*time.Second))
|
||||
for i := 0; i < 3; i++ {
|
||||
require.NoError(t, waitForServer(fmt.Sprintf("127.0.0.1:809%d", i), 30*time.Second))
|
||||
}
|
||||
t.Log("waiting for multi-disk volume servers to register...")
|
||||
time.Sleep(10 * time.Second)
|
||||
|
||||
commandEnv := shell.NewCommandEnv(&shell.ShellOptions{
|
||||
Masters: stringPtr("127.0.0.1:9334"),
|
||||
GrpcDialOption: grpc.WithInsecure(),
|
||||
FilerGroup: stringPtr("default"),
|
||||
})
|
||||
connectToMasterAndSync(ctx, t, commandEnv)
|
||||
|
||||
// Upload enough small files that the volume holds real data to encode.
|
||||
var volumeId needle.VolumeId
|
||||
for retry := 0; retry < 5; retry++ {
|
||||
volumeId, err = uploadTestDataToMaster([]byte(strings.Repeat("multidisk-ec-9593 ", 64)), "127.0.0.1:9334")
|
||||
if err == nil {
|
||||
break
|
||||
}
|
||||
time.Sleep(3 * time.Second)
|
||||
}
|
||||
require.NoError(t, err, "failed to upload test data")
|
||||
for i := 0; i < 40; i++ {
|
||||
if _, e := uploadTestDataToMaster([]byte(strings.Repeat("filler ", 128)), "127.0.0.1:9334"); e != nil {
|
||||
break
|
||||
}
|
||||
}
|
||||
t.Logf("using volume %d", volumeId)
|
||||
time.Sleep(3 * time.Second)
|
||||
|
||||
// Populate every server's disks with volumes so the encode can see and target
|
||||
// each physical disk. The master only enumerates disks that already hold a
|
||||
// volume or EC shard — an empty disk leaves no trace in the topology (heartbeats
|
||||
// aggregate capacity per disk type, not per physical disk). ec.encode therefore
|
||||
// spreads a volume's shards only across the disks the master already knows hold
|
||||
// data on each node; if a node's data sits on a single disk, all its shards land
|
||||
// there and ec.balance cannot redistribute them (it has no within-node
|
||||
// cross-disk move). So spreading must be set up before encoding.
|
||||
//
|
||||
// volume.grow only tops up toward a writable target and stops on the first
|
||||
// allocation error, so a single -count grow can create far fewer volumes than
|
||||
// asked and leave a node on one disk. Grow repeatedly on the nodes that have not
|
||||
// spread yet (the volume server places each new volume on its least-loaded disk)
|
||||
// until the master's topology shows every node holding volumes on at least two
|
||||
// physical disks. This makes the multi-disk layout — and thus the post-encode
|
||||
// disk spread — deterministic instead of racing volume-growth and heartbeat.
|
||||
require.Eventually(t, func() bool {
|
||||
spread := nodeVolumeDiskCounts(t, commandEnv)
|
||||
if len(spread) == 3 && allAtLeast(spread, 2) {
|
||||
return true
|
||||
}
|
||||
for i := 0; i < 3; i++ {
|
||||
server := fmt.Sprintf("127.0.0.1:809%d", i)
|
||||
if spread[server] < 2 {
|
||||
captureCommandOutput(t, shell.Commands[findCommandIndex("volume.grow")],
|
||||
[]string{"-collection", "test", "-dataNode", server, "-count", "4"}, commandEnv)
|
||||
}
|
||||
}
|
||||
return false
|
||||
}, 60*time.Second, 2*time.Second,
|
||||
"volumes never spread across >=2 disks on all 3 nodes")
|
||||
|
||||
locked, unlock := tryLockWithTimeout(t, commandEnv, 15*time.Second)
|
||||
require.True(t, locked, "could not acquire shell lock")
|
||||
defer unlock()
|
||||
|
||||
// EC-encode the volume.
|
||||
out, err := captureCommandOutput(t, shell.Commands[findCommandIndex("ec.encode")],
|
||||
[]string{"-volumeId", fmt.Sprintf("%d", volumeId), "-collection", "test", "-force"}, commandEnv)
|
||||
t.Logf("ec.encode output:\n%s", out)
|
||||
require.NoError(t, err, "ec.encode failed")
|
||||
|
||||
// All 14 shards must exist after encoding.
|
||||
require.Eventually(t, func() bool {
|
||||
return len(collectDistinctShardIDs(testDir, uint32(volumeId))) == erasureShardCount
|
||||
}, 30*time.Second, time.Second, "expected all %d EC shards after encode, got %v",
|
||||
erasureShardCount, collectDistinctShardIDs(testDir, uint32(volumeId)))
|
||||
|
||||
beforeBalance := collectDistinctShardIDs(testDir, uint32(volumeId))
|
||||
t.Logf("after encode: %d distinct shards on %d disks", len(beforeBalance), disksWithShards(testDir, uint32(volumeId)))
|
||||
|
||||
// Run ec.balance.
|
||||
out, err = captureCommandOutput(t, shell.Commands[findCommandIndex("ec.balance")],
|
||||
[]string{"-collection", "test", "-force"}, commandEnv)
|
||||
t.Logf("ec.balance output:\n%s", out)
|
||||
require.NoError(t, err, "ec.balance failed")
|
||||
time.Sleep(3 * time.Second)
|
||||
|
||||
// The core regression: ec.balance must not lose any shard.
|
||||
afterBalance := collectDistinctShardIDs(testDir, uint32(volumeId))
|
||||
require.Equal(t, erasureShardCount, len(afterBalance),
|
||||
"ec.balance lost shards on multi-disk nodes: had %v, now %v", sortedKeysOf(beforeBalance), sortedKeysOf(afterBalance))
|
||||
|
||||
// Shards must be spread across more than one physical disk per node overall.
|
||||
usedDisks := disksWithShards(testDir, uint32(volumeId))
|
||||
assert.Greater(t, usedDisks, 3, "EC shards should span more than one disk per node (got %d disks across 3 nodes)", usedDisks)
|
||||
|
||||
// cluster.status must count physical disks, not collapse to one per node: it
|
||||
// must report at least the disks actually holding this volume's shards (which
|
||||
// is already >3 across the 3 nodes). Before the fix it reported 3 (node count).
|
||||
require.Eventually(t, func() bool {
|
||||
n, ok := clusterStatusDiskCount(t, commandEnv)
|
||||
return ok && n >= usedDisks
|
||||
}, 30*time.Second, 2*time.Second, "cluster.status never reported the >=%d physical disks holding shards (multi-disk count)", usedDisks)
|
||||
|
||||
n, _ := clusterStatusDiskCount(t, commandEnv)
|
||||
t.Logf("cluster.status reports %d physical disks (>= %d holding this volume's shards)", n, usedDisks)
|
||||
}
|
||||
|
||||
const erasureShardCount = 14 // 10 data + 4 parity
|
||||
|
||||
// collectDistinctShardIDs returns the set of EC shard ids present for a volume
|
||||
// across every disk of every server in the multi-disk test layout.
|
||||
func collectDistinctShardIDs(testDir string, volumeId uint32) map[int]bool {
|
||||
ids := map[int]bool{}
|
||||
for server := 0; server < 3; server++ {
|
||||
for disk := 0; disk < 4; disk++ {
|
||||
diskDir := filepath.Join(testDir, fmt.Sprintf("server%d_disk%d", server, disk))
|
||||
files, err := listECShardFiles(diskDir, volumeId)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
for _, f := range files {
|
||||
i := strings.LastIndex(f, ".ec")
|
||||
if i < 0 {
|
||||
continue
|
||||
}
|
||||
if n, err := strconv.Atoi(f[i+3:]); err == nil && n >= 0 && n < erasureShardCount {
|
||||
ids[n] = true
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return ids
|
||||
}
|
||||
|
||||
// disksWithShards counts how many physical disks hold at least one shard.
|
||||
func disksWithShards(testDir string, volumeId uint32) int {
|
||||
n := 0
|
||||
for _, disks := range countShardsPerDisk(testDir, volumeId) {
|
||||
for _, c := range disks {
|
||||
if c > 0 {
|
||||
n++
|
||||
}
|
||||
}
|
||||
}
|
||||
return n
|
||||
}
|
||||
|
||||
// nodeVolumeDiskCounts returns, per volume server id, how many distinct physical
|
||||
// disks hold at least one volume according to the master's topology. The master
|
||||
// only enumerates disks that already hold a volume or EC shard (heartbeats
|
||||
// aggregate capacity per disk type, not per physical disk), so this reports the
|
||||
// disks ec.encode can actually spread a volume's shards across on each node.
|
||||
func nodeVolumeDiskCounts(t *testing.T, commandEnv *shell.CommandEnv) map[string]int {
|
||||
t.Helper()
|
||||
var resp *master_pb.VolumeListResponse
|
||||
err := commandEnv.MasterClient.WithClient(false, func(client master_pb.SeaweedClient) error {
|
||||
var e error
|
||||
resp, e = client.VolumeList(context.Background(), &master_pb.VolumeListRequest{})
|
||||
return e
|
||||
})
|
||||
counts := map[string]int{}
|
||||
if err != nil || resp.GetTopologyInfo() == nil {
|
||||
return counts
|
||||
}
|
||||
for _, dc := range resp.GetTopologyInfo().GetDataCenterInfos() {
|
||||
for _, r := range dc.GetRackInfos() {
|
||||
for _, dn := range r.GetDataNodeInfos() {
|
||||
disks := map[uint32]bool{}
|
||||
for _, di := range dn.GetDiskInfos() {
|
||||
for _, vi := range di.GetVolumeInfos() {
|
||||
disks[vi.GetDiskId()] = true
|
||||
}
|
||||
}
|
||||
counts[dn.Id] = len(disks)
|
||||
}
|
||||
}
|
||||
}
|
||||
return counts
|
||||
}
|
||||
|
||||
func allAtLeast(counts map[string]int, min int) bool {
|
||||
for _, c := range counts {
|
||||
if c < min {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
var diskCountRe = regexp.MustCompile(`(\d+)\s+disks?`)
|
||||
|
||||
// clusterStatusDiskCount runs cluster.status and parses the reported disk count.
|
||||
func clusterStatusDiskCount(t *testing.T, commandEnv *shell.CommandEnv) (int, bool) {
|
||||
t.Helper()
|
||||
out, err := captureCommandOutput(t, shell.Commands[findCommandIndex("cluster.status")], []string{}, commandEnv)
|
||||
if err != nil {
|
||||
return 0, false
|
||||
}
|
||||
m := diskCountRe.FindStringSubmatch(out)
|
||||
if m == nil {
|
||||
return 0, false
|
||||
}
|
||||
n, err := strconv.Atoi(m[1])
|
||||
return n, err == nil
|
||||
}
|
||||
|
||||
func sortedKeysOf(m map[int]bool) []int {
|
||||
out := make([]int, 0, len(m))
|
||||
for k := range m {
|
||||
out = append(out, k)
|
||||
}
|
||||
for i := 1; i < len(out); i++ {
|
||||
for j := i; j > 0 && out[j-1] > out[j]; j-- {
|
||||
out[j-1], out[j] = out[j], out[j-1]
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
@@ -0,0 +1,232 @@
|
||||
package fuse_dlm
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"runtime"
|
||||
"syscall"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/cluster/lock_manager"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// posixLockKey returns the routed-lock key a mount uses for a file at the mount
|
||||
// root. Mounts run with -filer.path=/, so the file's filer path is "/"+name; the
|
||||
// prefix must match mount.posixLockKeyForInode ("s3.fuse.lock:").
|
||||
func posixLockKey(name string) string { return "s3.fuse.lock:/" + name }
|
||||
|
||||
// These tests exercise cross-mount POSIX advisory locks (flock), which the
|
||||
// mounts route to the inode's owner filer because they run with -dlm
|
||||
// (crossMountLocks() == lockClient != nil). They reuse the dlmTestCluster:
|
||||
// master + volume + 2 filers (forming the lock ring) + 2 mounts on filer0.
|
||||
//
|
||||
// All flock opens are O_RDONLY on purpose: a write open (O_RDWR/O_WRONLY) would
|
||||
// also take the -dlm whole-file write lock (held until close), which is a
|
||||
// different mechanism — opening read-only isolates the POSIX advisory flock.
|
||||
|
||||
// requireForwardedLocks skips on platforms where the kernel does not forward
|
||||
// advisory locks to the FUSE server. Only Linux forwards flock/fcntl (SETLK) to
|
||||
// the filesystem; macFUSE handles flock in-kernel per mount, so cross-mount
|
||||
// coordination can't be observed there even though the routed path is correct.
|
||||
func requireForwardedLocks(t *testing.T) {
|
||||
if runtime.GOOS != "linux" {
|
||||
t.Skipf("advisory locks are only forwarded to the FUSE server on Linux (GOOS=%s)", runtime.GOOS)
|
||||
}
|
||||
}
|
||||
|
||||
// openFlock opens path read-only and takes a flock of type how (e.g. LOCK_EX).
|
||||
// The file must already exist. The returned file holds the lock until closed.
|
||||
func openFlock(path string, how int) (*os.File, error) {
|
||||
f, err := os.OpenFile(path, os.O_RDONLY, 0)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := syscall.Flock(int(f.Fd()), how); err != nil {
|
||||
f.Close()
|
||||
return nil, err
|
||||
}
|
||||
return f, nil
|
||||
}
|
||||
|
||||
// createFile creates an empty file; the brief -dlm write lock it takes is
|
||||
// released on close, before any flock test runs.
|
||||
func createFile(t *testing.T, path string) {
|
||||
t.Helper()
|
||||
f, err := os.OpenFile(path, os.O_RDWR|os.O_CREATE|os.O_TRUNC, 0644)
|
||||
require.NoError(t, err, "create %s", path)
|
||||
require.NoError(t, f.Close())
|
||||
}
|
||||
|
||||
// waitVisible waits for path to appear (cross-mount metadata propagation).
|
||||
func waitVisible(t *testing.T, path string) {
|
||||
t.Helper()
|
||||
require.Eventually(t, func() bool {
|
||||
_, err := os.Stat(path)
|
||||
return err == nil
|
||||
}, 15*time.Second, 200*time.Millisecond, "%s never became visible", path)
|
||||
}
|
||||
|
||||
// tryExclusiveFlock attempts a non-blocking exclusive flock on path and
|
||||
// classifies the outcome: acquired (and released again), blocked by another
|
||||
// owner (EWOULDBLOCK/EAGAIN), or an unexpected error. It distinguishes a real
|
||||
// "held by another" from incidental errors so the latter can't masquerade as a
|
||||
// satisfied "must be blocked" assertion.
|
||||
func tryExclusiveFlock(path string) (acquired, blocked bool, err error) {
|
||||
f, err := os.OpenFile(path, os.O_RDONLY, 0)
|
||||
if err != nil {
|
||||
return false, false, err
|
||||
}
|
||||
defer f.Close()
|
||||
if err := syscall.Flock(int(f.Fd()), syscall.LOCK_EX|syscall.LOCK_NB); err != nil {
|
||||
if errors.Is(err, syscall.EWOULDBLOCK) || errors.Is(err, syscall.EAGAIN) {
|
||||
return false, true, nil
|
||||
}
|
||||
return false, false, err
|
||||
}
|
||||
syscall.Flock(int(f.Fd()), syscall.LOCK_UN)
|
||||
return true, false, nil
|
||||
}
|
||||
|
||||
// requireEventuallyBlocked asserts path becomes held by another owner. It polls
|
||||
// because a routed lock RPC can hit a transient EIO (a cold or forwarded gRPC
|
||||
// call under load) even while the lock is genuinely held; it still fails if the
|
||||
// lock is never blocked — whether it can be acquired (a double-grant) or errors
|
||||
// persistently — so an incidental error can't masquerade as "held".
|
||||
func requireEventuallyBlocked(t *testing.T, path string, timeout time.Duration, msg string) {
|
||||
t.Helper()
|
||||
require.Eventuallyf(t, func() bool { return isBlocked(path) },
|
||||
timeout, 500*time.Millisecond, "%s", msg)
|
||||
}
|
||||
|
||||
// isBlocked / isAcquirable are lenient predicates for polling, where transient
|
||||
// errors during a migration are expected and simply mean "not yet".
|
||||
func isBlocked(path string) bool { _, b, _ := tryExclusiveFlock(path); return b }
|
||||
func isAcquirable(path string) bool {
|
||||
a, _, _ := tryExclusiveFlock(path)
|
||||
return a
|
||||
}
|
||||
|
||||
// stopFiler stops filer idx and clears its command so cluster teardown does not
|
||||
// try to stop it again.
|
||||
func (c *dlmTestCluster) stopFiler(idx int) {
|
||||
stopCmd(c.filerCmds[idx])
|
||||
c.filerCmds[idx] = nil
|
||||
}
|
||||
|
||||
// TestPosixLockCrossMount verifies a flock taken on one mount is seen by the
|
||||
// other mount of the same cluster (the routed-to-owner-filer path end to end).
|
||||
func TestPosixLockCrossMount(t *testing.T) {
|
||||
requireForwardedLocks(t)
|
||||
c := startDLMTestCluster(t)
|
||||
|
||||
name := "posix-xmount.lock"
|
||||
path0 := filepath.Join(c.mountPoints[0], name)
|
||||
path1 := filepath.Join(c.mountPoints[1], name)
|
||||
|
||||
createFile(t, path0)
|
||||
waitVisible(t, path1)
|
||||
|
||||
held, err := openFlock(path0, syscall.LOCK_EX)
|
||||
require.NoError(t, err, "mount0 should acquire the exclusive flock")
|
||||
defer held.Close()
|
||||
|
||||
requireEventuallyBlocked(t, path1, 15*time.Second, "mount1 must be blocked while mount0 holds the flock")
|
||||
|
||||
require.NoError(t, syscall.Flock(int(held.Fd()), syscall.LOCK_UN))
|
||||
require.Eventually(t, func() bool { return isAcquirable(path1) },
|
||||
15*time.Second, 500*time.Millisecond,
|
||||
"mount1 should acquire the flock after mount0 releases it")
|
||||
}
|
||||
|
||||
// TestPosixLockSurvivesFilerLoss verifies advisory locks held across mounts
|
||||
// survive a filer leaving the ring: ownership of the affected keys migrates to
|
||||
// the surviving filer and the holding mount re-asserts them there, so the locks
|
||||
// stay honored. It locks many files so that — with 2 filers — several are owned
|
||||
// by filer1 and thus actually migrate when filer1 stops.
|
||||
func TestPosixLockSurvivesFilerLoss(t *testing.T) {
|
||||
requireForwardedLocks(t)
|
||||
c := startDLMTestCluster(t)
|
||||
|
||||
const n = 12
|
||||
|
||||
// Select files via the same ring the filers run so the locked set provably
|
||||
// spans both filers: filer1-owned keys must migrate when filer1 stops, while
|
||||
// filer0-owned keys must keep working. Choosing by ownership (instead of
|
||||
// hoping a sequential set happens to spread) keeps the migration path
|
||||
// exercised on every run, independent of the cluster's dynamic ports.
|
||||
ring := lock_manager.NewHashRing(lock_manager.DefaultVnodeCount)
|
||||
ring.SetServers([]pb.ServerAddress{
|
||||
pb.ServerAddress(c.filerAddress(0)),
|
||||
pb.ServerAddress(c.filerAddress(1)),
|
||||
})
|
||||
filer1 := pb.ServerAddress(c.filerAddress(1))
|
||||
var onFiler1, onFiler0 []string
|
||||
for i := 0; (len(onFiler1) < n/2 || len(onFiler0) < n/2) && i < 1000; i++ {
|
||||
name := fmt.Sprintf("posix-migrate-%d.lock", i)
|
||||
if ring.GetPrimary(posixLockKey(name)) == filer1 {
|
||||
onFiler1 = append(onFiler1, name)
|
||||
} else {
|
||||
onFiler0 = append(onFiler0, name)
|
||||
}
|
||||
}
|
||||
require.GreaterOrEqualf(t, len(onFiler1), n/2, "need %d filer1-owned keys to exercise migration", n/2)
|
||||
require.GreaterOrEqualf(t, len(onFiler0), n/2, "need %d filer0-owned keys", n/2)
|
||||
names := append(onFiler1[:n/2:n/2], onFiler0[:n/2]...)
|
||||
|
||||
held := make([]*os.File, n)
|
||||
for i, name := range names {
|
||||
createFile(t, filepath.Join(c.mountPoints[0], name))
|
||||
waitVisible(t, filepath.Join(c.mountPoints[1], name))
|
||||
f, err := openFlock(filepath.Join(c.mountPoints[0], name), syscall.LOCK_EX)
|
||||
require.NoError(t, err, "mount0 should acquire flock %s", name)
|
||||
held[i] = f
|
||||
defer held[i].Close()
|
||||
}
|
||||
|
||||
require.Eventually(t, func() bool {
|
||||
for _, name := range names {
|
||||
if !isBlocked(filepath.Join(c.mountPoints[1], name)) {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}, 20*time.Second, time.Second,
|
||||
"every lock must be held on mount1 before the ring change")
|
||||
|
||||
// Drop filer1 from the ring; keys it owned migrate to filer0.
|
||||
c.stopFiler(1)
|
||||
require.NoError(t, c.waitForFilerCount(1, 30*time.Second), "ring should drop to one filer")
|
||||
|
||||
// Poll until every lock is honored again on the surviving filer: the ring
|
||||
// must propagate and the holding mount must re-assert its locks (keepalive is
|
||||
// 5s). We assert only the settled state — the transient migration window is
|
||||
// covered by unit tests.
|
||||
require.Eventually(t, func() bool {
|
||||
for _, name := range names {
|
||||
if !isBlocked(filepath.Join(c.mountPoints[1], name)) {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}, 45*time.Second, time.Second,
|
||||
"all locks must survive filer1 leaving the ring (migrate to filer0)")
|
||||
|
||||
// Releasing on mount0 frees them all: mount1 can then acquire each.
|
||||
for _, f := range held {
|
||||
require.NoError(t, syscall.Flock(int(f.Fd()), syscall.LOCK_UN))
|
||||
}
|
||||
require.Eventually(t, func() bool {
|
||||
for _, name := range names {
|
||||
if !isAcquirable(filepath.Join(c.mountPoints[1], name)) {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}, 20*time.Second, 500*time.Millisecond,
|
||||
"mount1 should acquire every lock after mount0 releases them post-migration")
|
||||
}
|
||||
@@ -177,8 +177,10 @@ func testConcurrentReadWrite(t *testing.T, framework *FuseTestFramework) {
|
||||
defer wg.Done()
|
||||
|
||||
for j := 0; j < 10; j++ {
|
||||
_, err := os.ReadFile(mountPath)
|
||||
if err != nil {
|
||||
if err := retryTransientFUSE(func() error {
|
||||
_, e := os.ReadFile(mountPath)
|
||||
return e
|
||||
}); err != nil {
|
||||
addError(fmt.Errorf("reader %d: %v", readerID, err))
|
||||
return
|
||||
}
|
||||
@@ -196,8 +198,9 @@ func testConcurrentReadWrite(t *testing.T, framework *FuseTestFramework) {
|
||||
|
||||
for j := 0; j < 5; j++ {
|
||||
newData := bytes.Repeat([]byte(fmt.Sprintf("WRITER%d", writerID)), 1000)
|
||||
err := os.WriteFile(mountPath, newData, 0644)
|
||||
if err != nil {
|
||||
if err := retryTransientFUSE(func() error {
|
||||
return os.WriteFile(mountPath, newData, 0644)
|
||||
}); err != nil {
|
||||
addError(fmt.Errorf("writer %d: %v", writerID, err))
|
||||
return
|
||||
}
|
||||
@@ -213,6 +216,21 @@ func testConcurrentReadWrite(t *testing.T, framework *FuseTestFramework) {
|
||||
framework.AssertFileExists(filename)
|
||||
}
|
||||
|
||||
// retryTransientFUSE retries op a few times before giving up. A concurrent
|
||||
// truncating overwrite can leave a short-lived dentry/cache window where the
|
||||
// entry is momentarily invisible (ENOENT) to another opener; the last error is
|
||||
// returned so a genuine, persistent failure still surfaces.
|
||||
func retryTransientFUSE(op func() error) error {
|
||||
var err error
|
||||
for attempt := 0; attempt < 5; attempt++ {
|
||||
if err = op(); err == nil {
|
||||
return nil
|
||||
}
|
||||
time.Sleep(100 * time.Millisecond)
|
||||
}
|
||||
return err
|
||||
}
|
||||
|
||||
// testConcurrentDirectoryOperations tests concurrent directory operations
|
||||
func testConcurrentDirectoryOperations(t *testing.T, framework *FuseTestFramework) {
|
||||
numWorkers := 8
|
||||
|
||||
@@ -207,8 +207,10 @@ func buildVolumeListResponse(t *testing.T, spec topologySpec, volumeID uint32) *
|
||||
t.Helper()
|
||||
|
||||
volumeSizeLimitMB := uint64(100)
|
||||
volumeSize := uint64(90) * 1024 * 1024
|
||||
volumeModifiedAt := time.Now().Add(-10 * time.Minute).Unix()
|
||||
// Exceed the default fullness (0.95) and quiet (1h) thresholds so volumes are
|
||||
// EC-eligible.
|
||||
volumeSize := uint64(96) * 1024 * 1024
|
||||
volumeModifiedAt := time.Now().Add(-2 * time.Hour).Unix()
|
||||
|
||||
diskTypes := spec.diskTypes
|
||||
if len(diskTypes) == 0 {
|
||||
|
||||
@@ -29,9 +29,11 @@ func TestErasureCodingDetectionLargeTopology(t *testing.T) {
|
||||
}
|
||||
|
||||
nodesPerRack := serverCount / rackCount
|
||||
eligibleSize := uint64(90) * 1024 * 1024
|
||||
// Eligible volumes must exceed the default fullness (0.95) and quiet (1h)
|
||||
// thresholds; ineligible ones fall below the fullness threshold.
|
||||
eligibleSize := uint64(96) * 1024 * 1024
|
||||
ineligibleSize := uint64(10) * 1024 * 1024
|
||||
modifiedAt := time.Now().Add(-10 * time.Minute).Unix()
|
||||
modifiedAt := time.Now().Add(-2 * time.Hour).Unix()
|
||||
|
||||
volumeID := uint32(1)
|
||||
dataCenters := make([]*master_pb.DataCenterInfo, 0, 1)
|
||||
|
||||
@@ -298,6 +298,54 @@ User Request → Load Balancer → Any S3 Gateway Instance
|
||||
Allow/Deny Request
|
||||
```
|
||||
|
||||
## Trust Policy Conditions
|
||||
|
||||
Step 5 above evaluates the role's trust policy against context keys derived from the
|
||||
OIDC token's claims. The available keys are:
|
||||
|
||||
| Condition key | Source |
|
||||
|---------------|--------|
|
||||
| `oidc:iss` | `iss` claim (issuer URL) |
|
||||
| `oidc:sub` | `sub` claim |
|
||||
| `oidc:aud` | `aud` claim |
|
||||
| `oidc:<claim>` | any other token claim, e.g. `oidc:roles`, `oidc:groups`, `oidc:email` |
|
||||
| `aws:FederatedProvider` | the provider `name` (e.g. `keycloak-oidc`) when its configured issuer matches the token, otherwise the raw issuer URL |
|
||||
| `aws:userid` | `sub` claim (same value as `oidc:sub` during trust-policy evaluation) |
|
||||
| `sts:DurationSeconds` | requested session duration, when supplied |
|
||||
|
||||
During trust-policy evaluation `aws:userid` is the raw `sub` claim. Once the
|
||||
role has been assumed, the keys seen by request authorization differ: there
|
||||
`aws:userid` is a stable per-identity hash of `sub` and `iss` (see
|
||||
`ComputeParentUser`), so do not assume the two contexts carry the same value.
|
||||
|
||||
Custom claims are always exposed under the `oidc:` prefix, so a trust policy must use
|
||||
`oidc:roles` (not a bare `roles`) to match a `roles` claim:
|
||||
|
||||
```json
|
||||
"Condition": {
|
||||
"StringEquals": {
|
||||
"oidc:roles": "s3-admin"
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
A multi-valued claim (such as a `roles` array) matches when any of its values equals
|
||||
the condition value. The same `oidc:` keys can be interpolated into policy resources,
|
||||
e.g. `arn:aws:s3:::bucket/${oidc:sub}/*`.
|
||||
|
||||
### roleMapping vs. trust policy
|
||||
|
||||
A provider's `roleMapping` and a role's trust policy apply to two different entry
|
||||
points and are not interchangeable:
|
||||
|
||||
- **Direct OIDC** — an S3 request carrying `Authorization: Bearer <OIDC-JWT>`. The
|
||||
gateway applies `roleMapping` to choose the caller's role from the token claims; the
|
||||
first matching rule (or `defaultRole`) wins.
|
||||
- **STS `AssumeRoleWithWebIdentity`** — the caller names the role explicitly via
|
||||
`RoleArn`, and that role's trust policy decides whether the assumption is allowed.
|
||||
`roleMapping` does not select the role on this path; instead the token claims are
|
||||
surfaced as the `oidc:` condition keys above for the trust policy to evaluate.
|
||||
|
||||
## Configuration Management
|
||||
|
||||
### Development Environment
|
||||
|
||||
@@ -37,7 +37,7 @@
|
||||
"Action": ["sts:AssumeRoleWithWebIdentity"],
|
||||
"Condition": {
|
||||
"StringEquals": {
|
||||
"roles": "s3-admin"
|
||||
"oidc:roles": "s3-admin"
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -60,7 +60,7 @@
|
||||
"Action": ["sts:AssumeRoleWithWebIdentity"],
|
||||
"Condition": {
|
||||
"StringEquals": {
|
||||
"roles": "s3-read-only"
|
||||
"oidc:roles": "s3-read-only"
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -83,7 +83,7 @@
|
||||
"Action": ["sts:AssumeRoleWithWebIdentity"],
|
||||
"Condition": {
|
||||
"StringEquals": {
|
||||
"roles": "s3-read-write"
|
||||
"oidc:roles": "s3-read-write"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -22,6 +22,7 @@ import (
|
||||
"os"
|
||||
"os/exec"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
@@ -69,7 +70,111 @@ func s3Client(t *testing.T) *s3.Client {
|
||||
})),
|
||||
)
|
||||
require.NoError(t, err)
|
||||
return s3.NewFromConfig(cfg, func(o *s3.Options) { o.UsePathStyle = true })
|
||||
client := s3.NewFromConfig(cfg, func(o *s3.Options) { o.UsePathStyle = true })
|
||||
ensureClusterWritable(t, client)
|
||||
return client
|
||||
}
|
||||
|
||||
var clusterWritableOnce sync.Once
|
||||
|
||||
// ensureClusterWritable blocks until the cluster can actually serve a write,
|
||||
// absorbing the volume-growth warmup window after a fresh start. The Makefile
|
||||
// only waits for the server process to be up ("server up after N s"); it does
|
||||
// not wait for a writable volume, so the first PutObject can race volume growth
|
||||
// and fail with a transient 500 (assign volume: DeadlineExceeded) — the source
|
||||
// of the lifecycle-test flakes. Probing one throwaway write here, once per
|
||||
// process, warms growth so every test's first real write is past that window.
|
||||
// Best-effort: if it never becomes writable, the test's own PutObject surfaces
|
||||
// the failure normally.
|
||||
func ensureClusterWritable(t *testing.T, c *s3.Client) {
|
||||
t.Helper()
|
||||
clusterWritableOnce.Do(func() {
|
||||
bucket := uniqueBucket("warmup")
|
||||
deadline := time.Now().Add(60 * time.Second)
|
||||
|
||||
// try runs fn under a bounded context so a single hung call can't block.
|
||||
try := func(timeout time.Duration, fn func(ctx context.Context) error) error {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), timeout)
|
||||
defer cancel()
|
||||
return fn(ctx)
|
||||
}
|
||||
// probe is a try whose timeout is clamped to the time left before the
|
||||
// deadline, so the whole warmup stays within the budget; ok=false means
|
||||
// the budget is exhausted and the caller should stop.
|
||||
probe := func(fn func(ctx context.Context) error) (err error, ok bool) {
|
||||
remaining := time.Until(deadline)
|
||||
if remaining <= 0 {
|
||||
return nil, false
|
||||
}
|
||||
if remaining > 10*time.Second {
|
||||
remaining = 10 * time.Second
|
||||
}
|
||||
return try(remaining, fn), true
|
||||
}
|
||||
// backoff sleeps attempt*250ms, never past the deadline.
|
||||
backoff := func(attempt int) {
|
||||
d := time.Duration(attempt) * 250 * time.Millisecond
|
||||
if left := time.Until(deadline); d > left {
|
||||
d = left
|
||||
}
|
||||
if d > 0 {
|
||||
time.Sleep(d)
|
||||
}
|
||||
}
|
||||
|
||||
// CreateBucket is a metadata op, but on a cold cluster the filer itself may
|
||||
// not be ready yet, so retry it within the deadline rather than abandoning
|
||||
// the whole warmup (and the PutObject probe) on the first error.
|
||||
created := false
|
||||
for attempt := 1; ; attempt++ {
|
||||
err, ok := probe(func(ctx context.Context) error {
|
||||
_, e := c.CreateBucket(ctx, &s3.CreateBucketInput{Bucket: aws.String(bucket)})
|
||||
return e
|
||||
})
|
||||
if !ok {
|
||||
break
|
||||
}
|
||||
if err == nil {
|
||||
created = true
|
||||
break
|
||||
}
|
||||
backoff(attempt)
|
||||
}
|
||||
if !created {
|
||||
t.Logf("warmup: could not create probe bucket within 60s; proceeding")
|
||||
return
|
||||
}
|
||||
// Cleanup gets a fresh timeout (not the warmup budget) so teardown runs
|
||||
// even when the probe loop used the full window.
|
||||
defer try(10*time.Second, func(ctx context.Context) error {
|
||||
_, e := c.DeleteBucket(ctx, &s3.DeleteBucketInput{Bucket: aws.String(bucket)})
|
||||
return e
|
||||
})
|
||||
|
||||
for attempt := 1; ; attempt++ {
|
||||
err, ok := probe(func(ctx context.Context) error {
|
||||
_, e := c.PutObject(ctx, &s3.PutObjectInput{
|
||||
Bucket: aws.String(bucket), Key: aws.String("warmup"), Body: strings.NewReader("ok"),
|
||||
})
|
||||
return e
|
||||
})
|
||||
if !ok {
|
||||
break
|
||||
}
|
||||
if err == nil {
|
||||
try(10*time.Second, func(ctx context.Context) error {
|
||||
_, e := c.DeleteObject(ctx, &s3.DeleteObjectInput{Bucket: aws.String(bucket), Key: aws.String("warmup")})
|
||||
return e
|
||||
})
|
||||
if attempt > 1 {
|
||||
t.Logf("cluster became writable after %d probe(s)", attempt)
|
||||
}
|
||||
return
|
||||
}
|
||||
backoff(attempt)
|
||||
}
|
||||
t.Logf("warmup: cluster not confirmed writable within 60s; proceeding")
|
||||
})
|
||||
}
|
||||
|
||||
func filerClient(t *testing.T) (filer_pb.SeaweedFilerClient, func()) {
|
||||
|
||||
@@ -0,0 +1,20 @@
|
||||
FROM chrislusf/seaweedfs:e2e
|
||||
|
||||
RUN apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 update && \
|
||||
DEBIAN_FRONTEND=noninteractive apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 install -y \
|
||||
--no-install-recommends \
|
||||
--no-install-suggests \
|
||||
samba \
|
||||
smbclient \
|
||||
python3-minimal \
|
||||
&& apt-get clean \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
COPY smb.conf.template /smb.conf.template
|
||||
COPY smb_tests.sh /smb_tests.sh
|
||||
COPY lock_tests.sh /lock_tests.sh
|
||||
COPY entrypoint.sh /entrypoint.sh
|
||||
COPY run_inside_container.sh /run_inside_container.sh
|
||||
RUN chmod +x /smb_tests.sh /lock_tests.sh /entrypoint.sh /run_inside_container.sh
|
||||
|
||||
ENTRYPOINT ["/entrypoint.sh"]
|
||||
@@ -0,0 +1,96 @@
|
||||
# Samba on FUSE integration test
|
||||
|
||||
Exports a SeaweedFS FUSE mount over SMB with Samba's `smbd` and drives it with
|
||||
`smbclient`, verifying that SMB file operations work correctly on top of the
|
||||
mount and that data stays consistent across both protocols.
|
||||
|
||||
## What it checks
|
||||
|
||||
The functional battery in `smb_tests.sh` covers:
|
||||
|
||||
- connecting to the share and listing the root
|
||||
- 1 MiB upload/download round-trip with content verification
|
||||
- subdirectory creation and writes into it
|
||||
- file rename
|
||||
- 64 MiB upload/download (exercises SeaweedFS chunk splitting)
|
||||
- recursive upload of a directory tree
|
||||
- cross-protocol consistency: files written over SMB appear on the FUSE mount
|
||||
with identical content, and files written directly on the FUSE mount are
|
||||
readable over SMB
|
||||
- deleting files and directory trees
|
||||
|
||||
The locking / concurrency battery in `lock_tests.sh` covers the harder cases a
|
||||
network-filesystem backend has to get right:
|
||||
|
||||
- **POSIX `fcntl` byte-range locking** on the FUSE mount: a held exclusive lock
|
||||
denies a conflicting lock, allows a non-overlapping range, and is reacquirable
|
||||
after release (exercises the mount's `SetLk`/`GetLk`)
|
||||
- **Distributed locking** (`-dlm`): a file held open for writing on one mount
|
||||
blocks a writer on a second mount until it is released
|
||||
- **Distributed-lock integrity**: concurrent writers to the same file from two
|
||||
mounts leave exactly one intact payload, never a torn mix
|
||||
- **Concurrency**: parallel writers to distinct files all succeed
|
||||
|
||||
Both FUSE mounts are started with `-dlm` (distributed lock manager). The second
|
||||
mount (`/mnt/seaweedfs2`) exists only to contend with the smbd-backed mount in
|
||||
the distributed-locking tests; both see the same filer path, so `.../share` is
|
||||
the same data on each.
|
||||
|
||||
> Note on DLM semantics: `-dlm` coordinates *write access* (one mount writes a
|
||||
> file at a time) and guarantees writes are not torn. It does not guarantee
|
||||
> which concurrent writer wins or instant cross-mount read convergence — the
|
||||
> holder's buffered data is flushed on close, asynchronously to lock release.
|
||||
> When a holder closes the file, a writer on another mount acquires the freed
|
||||
> lock within ~1s and completes.
|
||||
|
||||
## Layout
|
||||
|
||||
| File | Purpose |
|
||||
| --- | --- |
|
||||
| `smb_tests.sh` | SMB functional battery. Shared by both runners. |
|
||||
| `lock_tests.sh` | SMB locking / concurrency battery. Shared by both runners. |
|
||||
| `smb.conf.template` | Samba config; placeholders are filled in at run time. |
|
||||
| `run.sh` | Local runner: `weed mini` + two `-dlm` mounts + `smbd` + both batteries, all as the current user on unprivileged ports. |
|
||||
| `entrypoint.sh` | Container entrypoint: starts two `-dlm` FUSE mounts and runs `smbd`. |
|
||||
| `run_inside_container.sh` | Runs both batteries inside the container against the local `smbd`. |
|
||||
| `Dockerfile` | Adds Samba to the `chrislusf/seaweedfs:e2e` image. |
|
||||
| `docker-compose.yml` | master + volume + filer + samba services. |
|
||||
|
||||
## Running locally
|
||||
|
||||
Requirements: `weed` on `$PATH`, `fusermount3`, and Samba's `smbd` /
|
||||
`smbclient` / `smbpasswd` (Debian/Ubuntu: `apt-get install samba smbclient`).
|
||||
|
||||
```sh
|
||||
test/samba/run.sh
|
||||
```
|
||||
|
||||
No `sudo` is needed: `smbd` runs as the current user on port 4450 and all state
|
||||
lives under a temp work dir that is cleaned up on exit.
|
||||
|
||||
## Running with Docker
|
||||
|
||||
Mirrors the CI job. Requires `/dev/fuse` and `SYS_ADMIN` (provided in the
|
||||
compose file).
|
||||
|
||||
```sh
|
||||
# build the base e2e image first (from the repo's docker/ dir)
|
||||
docker compose -f test/samba/docker-compose.yml up --wait
|
||||
docker compose -f test/samba/docker-compose.yml exec -T samba /run_inside_container.sh
|
||||
docker compose -f test/samba/docker-compose.yml down -v
|
||||
```
|
||||
|
||||
## CI
|
||||
|
||||
`.github/workflows/samba-integration.yml` runs on changes to `weed/mount/**`,
|
||||
`weed/filer/**`, or `test/samba/**`. It builds the e2e image, builds the Samba
|
||||
harness image on top, brings up the cluster, runs the battery, and uploads
|
||||
server logs as artifacts.
|
||||
|
||||
## Notes
|
||||
|
||||
- The share disables Samba's DOS-attribute / xattr mapping and oplocks. The
|
||||
SeaweedFS FUSE mount does not implement that surface, and leaving it on
|
||||
produces `NT_STATUS_NOT_SUPPORTED` errors unrelated to data integrity.
|
||||
- The share path is a subdirectory of the mount (`.../share`) so the runner can
|
||||
verify SMB-side operations directly on the FUSE side.
|
||||
@@ -0,0 +1,60 @@
|
||||
services:
|
||||
master:
|
||||
image: chrislusf/seaweedfs:e2e
|
||||
command: "-v=4 master -ip=master -ip.bind=0.0.0.0 -raftBootstrap"
|
||||
healthcheck:
|
||||
test: ["CMD", "curl", "--fail", "-I", "http://localhost:9333/cluster/healthz"]
|
||||
interval: 2s
|
||||
timeout: 10s
|
||||
retries: 30
|
||||
start_period: 10s
|
||||
|
||||
volume:
|
||||
image: chrislusf/seaweedfs:e2e
|
||||
command: "-v=4 volume -master=master:9333 -ip=volume -ip.bind=0.0.0.0 -preStopSeconds=1"
|
||||
healthcheck:
|
||||
test: ["CMD", "curl", "--fail", "-I", "http://localhost:8080/healthz"]
|
||||
interval: 2s
|
||||
timeout: 10s
|
||||
retries: 15
|
||||
start_period: 5s
|
||||
depends_on:
|
||||
master:
|
||||
condition: service_healthy
|
||||
|
||||
filer:
|
||||
image: chrislusf/seaweedfs:e2e
|
||||
command: "-v=4 filer -master=master:9333 -ip=filer -ip.bind=0.0.0.0"
|
||||
healthcheck:
|
||||
test: ["CMD", "curl", "--fail", "-I", "http://localhost:8888/healthz"]
|
||||
interval: 2s
|
||||
timeout: 10s
|
||||
retries: 15
|
||||
start_period: 5s
|
||||
depends_on:
|
||||
volume:
|
||||
condition: service_healthy
|
||||
|
||||
samba:
|
||||
image: chrislusf/seaweedfs:samba
|
||||
build:
|
||||
context: .
|
||||
environment:
|
||||
FILER: filer:8888
|
||||
cap_add:
|
||||
- SYS_ADMIN
|
||||
devices:
|
||||
- /dev/fuse
|
||||
security_opt:
|
||||
- apparmor:unconfined
|
||||
healthcheck:
|
||||
test:
|
||||
- "CMD-SHELL"
|
||||
- "mountpoint -q /mnt/seaweedfs && mountpoint -q /mnt/seaweedfs2 && smbclient -L 127.0.0.1 -p 445 -U smbtest%smbtest -m SMB3 >/dev/null 2>&1"
|
||||
interval: 3s
|
||||
timeout: 10s
|
||||
retries: 20
|
||||
start_period: 15s
|
||||
depends_on:
|
||||
filer:
|
||||
condition: service_healthy
|
||||
Executable
+76
@@ -0,0 +1,76 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# Entrypoint for the samba test container.
|
||||
#
|
||||
# Mounts SeaweedFS over FUSE twice, both with distributed locking (-dlm) so the
|
||||
# locking tests can exercise cross-mount write coordination:
|
||||
# - MOUNT_DIR (/mnt/seaweedfs) is exported over SMB by smbd
|
||||
# - MOUNT2_DIR (/mnt/seaweedfs2) is a second, independent mount of the same
|
||||
# filer used to contend with the SMB writer
|
||||
#
|
||||
# Both mounts see the same filer path, so .../share is the same data on each.
|
||||
# smbd runs in the foreground (as root, which owns the mounts), so the share
|
||||
# uses "force user = root".
|
||||
set -euo pipefail
|
||||
|
||||
FILER="${FILER:-filer:8888}"
|
||||
MOUNT_DIR="${MOUNT_DIR:-/mnt/seaweedfs}"
|
||||
MOUNT2_DIR="${MOUNT2_DIR:-/mnt/seaweedfs2}"
|
||||
SHARE_DIR="${MOUNT_DIR}/share"
|
||||
STATE_DIR="${STATE_DIR:-/var/lib/samba-test}"
|
||||
SMB_PORT="${SMB_PORT:-445}"
|
||||
SMB_USER="${SMB_USER:-smbtest}"
|
||||
SMB_PASS="${SMB_PASS:-smbtest}"
|
||||
|
||||
mkdir -p "${MOUNT_DIR}" "${MOUNT2_DIR}" \
|
||||
"${STATE_DIR}/private" "${STATE_DIR}/state" "${STATE_DIR}/cache" \
|
||||
"${STATE_DIR}/lock" "${STATE_DIR}/pid" "${STATE_DIR}/ncalrpc"
|
||||
|
||||
# mount_seaweedfs <mountpoint> <logfile> — mount with -dlm and wait for it.
|
||||
mount_seaweedfs() {
|
||||
local dir="$1" log="$2"
|
||||
echo "==> Mounting SeaweedFS (${FILER}) at ${dir} with -dlm"
|
||||
weed -v=1 mount \
|
||||
-filer="${FILER}" \
|
||||
-dir="${dir}" \
|
||||
-filer.path=/ \
|
||||
-dirAutoCreate \
|
||||
-allowOthers \
|
||||
-dlm \
|
||||
>"${log}" 2>&1 &
|
||||
local pid=$!
|
||||
for _ in $(seq 1 120); do
|
||||
if mountpoint -q "${dir}"; then
|
||||
return 0
|
||||
fi
|
||||
if ! kill -0 "${pid}" 2>/dev/null; then
|
||||
echo "weed mount (${dir}) exited early; log tail:" >&2
|
||||
tail -n 100 "${log}" >&2 || true
|
||||
exit 1
|
||||
fi
|
||||
sleep 0.5
|
||||
done
|
||||
echo "FUSE mount ${dir} did not come up" >&2
|
||||
tail -n 100 "${log}" >&2 || true
|
||||
exit 1
|
||||
}
|
||||
|
||||
mount_seaweedfs "${MOUNT_DIR}" /var/log/weed-mount.log
|
||||
mount_seaweedfs "${MOUNT2_DIR}" /var/log/weed-mount2.log
|
||||
|
||||
mkdir -p "${SHARE_DIR}"
|
||||
chmod 0777 "${SHARE_DIR}"
|
||||
|
||||
# --- configure and start smbd ----------------------------------------------
|
||||
echo "==> Configuring Samba share on port ${SMB_PORT}"
|
||||
sed -e "s#@SHARE_PATH@#${SHARE_DIR}#g" \
|
||||
-e "s#@STATE_DIR@#${STATE_DIR}#g" \
|
||||
-e "s#@SMB_PORT@#${SMB_PORT}#g" \
|
||||
-e "s#@FORCE_USER@#root#g" \
|
||||
/smb.conf.template >/etc/samba/smb.conf
|
||||
|
||||
id -u "${SMB_USER}" >/dev/null 2>&1 || useradd -M -s /usr/sbin/nologin "${SMB_USER}"
|
||||
printf '%s\n%s\n' "${SMB_PASS}" "${SMB_PASS}" | smbpasswd -a -s "${SMB_USER}"
|
||||
|
||||
echo "==> Starting smbd"
|
||||
exec smbd -F --no-process-group -s /etc/samba/smb.conf
|
||||
Executable
+218
@@ -0,0 +1,218 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# Locking / concurrency test battery for Samba on a SeaweedFS FUSE mount.
|
||||
#
|
||||
# Covers the challenges a network-filesystem backend has to get right:
|
||||
# 1. POSIX fcntl byte-range locking on the FUSE mount (SetLk/GetLk)
|
||||
# 2. Distributed locking (-dlm): a write held open on one mount blocks a
|
||||
# writer on another mount until it is released
|
||||
# 3. Distributed locking integrity: concurrent writers to the same file from
|
||||
# two mounts produce intact (non-torn) data
|
||||
# 4. Concurrent writers to distinct files all succeed
|
||||
#
|
||||
# Required env:
|
||||
# SMB_USER, SMB_PASS samba credentials
|
||||
# MOUNT_SHARE dir on the smbd-backed FUSE mount (mount 1)
|
||||
# MOUNT2_SHARE dir on the second FUSE mount (mount 2)
|
||||
# Optional env:
|
||||
# SMB_HOST (127.0.0.1), SMB_SHARE (seaweedfs), SMB_PORT (445)
|
||||
set -uo pipefail
|
||||
|
||||
SMB_HOST="${SMB_HOST:-127.0.0.1}"
|
||||
SMB_SHARE="${SMB_SHARE:-seaweedfs}"
|
||||
SMB_PORT="${SMB_PORT:-445}"
|
||||
SMB_USER="${SMB_USER:?SMB_USER is required}"
|
||||
SMB_PASS="${SMB_PASS:?SMB_PASS is required}"
|
||||
MOUNT_SHARE="${MOUNT_SHARE:?MOUNT_SHARE is required}"
|
||||
MOUNT2_SHARE="${MOUNT2_SHARE:?MOUNT2_SHARE is required}"
|
||||
|
||||
WORK="$(mktemp -d /tmp/samba-locktest.XXXXXX)"
|
||||
trap 'rm -rf "${WORK}"' EXIT
|
||||
|
||||
PASS=0
|
||||
FAIL=0
|
||||
pass() { printf ' [PASS] %s\n' "$1"; PASS=$((PASS + 1)); }
|
||||
fail() { printf ' [FAIL] %s\n' "$1"; FAIL=$((FAIL + 1)); }
|
||||
|
||||
smb() {
|
||||
smbclient "//${SMB_HOST}/${SMB_SHARE}" -p "${SMB_PORT}" \
|
||||
-U "${SMB_USER}%${SMB_PASS}" -m SMB3 -c "$1"
|
||||
}
|
||||
md5() { md5sum "$1" | awk '{print $1}'; }
|
||||
|
||||
# 1. POSIX fcntl byte-range locking on the FUSE mount ------------------------
|
||||
# Exercises the mount's SetLk/GetLk via two processes contending over fcntl
|
||||
# (F_SETLK) byte-range locks. python3's fcntl.lockf issues real POSIX locks.
|
||||
echo "==> 1. POSIX fcntl byte-range locking (FUSE mount SetLk/GetLk)"
|
||||
lockfile="${MOUNT_SHARE}/fcntl_lock.dat"
|
||||
: >"${lockfile}"
|
||||
fcntl_out="$(python3 - "${lockfile}" <<'PY'
|
||||
import fcntl, os, sys
|
||||
|
||||
path = sys.argv[1]
|
||||
parent_to_child_r, parent_to_child_w = os.pipe() # release signal
|
||||
child_to_parent_r, child_to_parent_w = os.pipe() # locked signal
|
||||
|
||||
pid = os.fork()
|
||||
if pid == 0: # child: hold an exclusive lock on [0,100)
|
||||
fd = os.open(path, os.O_RDWR | os.O_CREAT, 0o644)
|
||||
fcntl.lockf(fd, fcntl.LOCK_EX, 100, 0, 0)
|
||||
os.write(child_to_parent_w, b"L")
|
||||
os.read(parent_to_child_r, 1) # wait until parent says release
|
||||
fcntl.lockf(fd, fcntl.LOCK_UN, 100, 0, 0)
|
||||
os.close(fd)
|
||||
os._exit(0)
|
||||
|
||||
# parent
|
||||
os.read(child_to_parent_r, 1) # wait until child holds the lock
|
||||
fd = os.open(path, os.O_RDWR | os.O_CREAT, 0o644)
|
||||
results = []
|
||||
|
||||
# a. a conflicting exclusive lock must be denied while the child holds it
|
||||
try:
|
||||
fcntl.lockf(fd, fcntl.LOCK_EX | fcntl.LOCK_NB, 100, 0, 0)
|
||||
fcntl.lockf(fd, fcntl.LOCK_UN, 100, 0, 0)
|
||||
results.append(("conflicting exclusive lock denied while held", False))
|
||||
except OSError:
|
||||
results.append(("conflicting exclusive lock denied while held", True))
|
||||
|
||||
# b. a non-overlapping range must be grantable
|
||||
try:
|
||||
fcntl.lockf(fd, fcntl.LOCK_EX | fcntl.LOCK_NB, 100, 200, 0)
|
||||
fcntl.lockf(fd, fcntl.LOCK_UN, 100, 200, 0)
|
||||
results.append(("non-overlapping range lock granted", True))
|
||||
except OSError:
|
||||
results.append(("non-overlapping range lock granted", False))
|
||||
|
||||
# c. after the holder releases, the lock must be acquirable
|
||||
os.write(parent_to_child_w, b"R")
|
||||
os.waitpid(pid, 0)
|
||||
try:
|
||||
fcntl.lockf(fd, fcntl.LOCK_EX | fcntl.LOCK_NB, 100, 0, 0)
|
||||
fcntl.lockf(fd, fcntl.LOCK_UN, 100, 0, 0)
|
||||
results.append(("lock acquirable after holder releases", True))
|
||||
except OSError:
|
||||
results.append(("lock acquirable after holder releases", False))
|
||||
|
||||
for name, ok in results:
|
||||
print((" [PASS] " if ok else " [FAIL] ") + name)
|
||||
sys.exit(0 if all(ok for _, ok in results) else 1)
|
||||
PY
|
||||
)"
|
||||
echo "${fcntl_out}"
|
||||
PASS=$((PASS + $(grep -c '\[PASS\]' <<<"${fcntl_out}")))
|
||||
FAIL=$((FAIL + $(grep -c '\[FAIL\]' <<<"${fcntl_out}")))
|
||||
|
||||
# 2. Distributed lock blocks a cross-mount writer, then hands it off ----------
|
||||
# mount 2 holds a file open for writing (holding the DLM lock on its path).
|
||||
# An SMB put of the same file goes through mount 1 and must (a) block while
|
||||
# mount 2 holds it and (b) succeed once mount 2 releases, leaving the SMB
|
||||
# writer's payload on disk. smbclient gets a long client timeout (-t) so we are
|
||||
# testing the lock handoff itself, not smbclient's own ~20s default timeout.
|
||||
echo "==> 2. distributed lock: cross-mount write coordination"
|
||||
dlmfile="dlm_coord.bin"
|
||||
newdata="${WORK}/dlm_new.bin"
|
||||
head -c 4096 /dev/urandom >"${newdata}"
|
||||
|
||||
# Hold the file open for writing on mount 2 via fd 9 -> holds the DLM lock.
|
||||
exec 9>"${MOUNT2_SHARE}/${dlmfile}"
|
||||
printf 'held-by-mount2' >&9
|
||||
|
||||
# Start the SMB write; record its real exit code when it returns. The subshell
|
||||
# must NOT inherit fd 9 (9>&-): otherwise the SMB writer keeps the file open and
|
||||
# waits on a DLM lock held by its own inherited descriptor, deadlocking the
|
||||
# handoff this test is meant to exercise.
|
||||
rm -f "${WORK}/dlm_put.rc"
|
||||
(
|
||||
smbclient "//${SMB_HOST}/${SMB_SHARE}" -p "${SMB_PORT}" \
|
||||
-U "${SMB_USER}%${SMB_PASS}" -m SMB3 -t 120 \
|
||||
-c "put ${newdata} ${dlmfile}" >/dev/null 2>&1
|
||||
echo "$?" >"${WORK}/dlm_put.rc"
|
||||
) 9>&- &
|
||||
smb_bg=$!
|
||||
|
||||
sleep 4
|
||||
if [[ ! -f "${WORK}/dlm_put.rc" ]]; then
|
||||
pass "SMB write blocks while another mount holds the file open"
|
||||
else
|
||||
fail "SMB write returned early instead of blocking (rc=$(cat "${WORK}/dlm_put.rc"))"
|
||||
fi
|
||||
|
||||
# Release mount 2's DLM lock; the blocked SMB write must now complete.
|
||||
exec 9>&-
|
||||
|
||||
# Wait (bounded) for the SMB put to finish so a stuck handoff fails the test
|
||||
# instead of hanging the suite.
|
||||
put_rc="timeout"
|
||||
for _ in $(seq 1 20); do
|
||||
if [[ -f "${WORK}/dlm_put.rc" ]]; then
|
||||
put_rc="$(cat "${WORK}/dlm_put.rc")"
|
||||
break
|
||||
fi
|
||||
sleep 1
|
||||
done
|
||||
kill "${smb_bg}" 2>/dev/null
|
||||
wait "${smb_bg}" 2>/dev/null
|
||||
|
||||
if [[ "${put_rc}" == "0" ]]; then
|
||||
pass "blocked SMB write succeeds after the other mount releases"
|
||||
else
|
||||
fail "blocked SMB write succeeds after the other mount releases (rc=${put_rc})"
|
||||
fi
|
||||
|
||||
# A correct handoff leaves the SMB writer's payload on disk: mount 1 acquired
|
||||
# the lock and wrote after mount 2 released.
|
||||
got="${WORK}/dlm_got.bin"
|
||||
if smb "get ${dlmfile} ${got}" >/dev/null 2>&1 && [[ "$(md5 "${got}")" == "$(md5 "${newdata}")" ]]; then
|
||||
pass "post-release content is the SMB writer's payload (correct handoff)"
|
||||
else
|
||||
fail "post-release content is the SMB writer's payload (correct handoff)"
|
||||
fi
|
||||
|
||||
# 3. Distributed lock integrity: concurrent writers, same file ---------------
|
||||
# An SMB writer (mount 1) and a direct writer (mount 2) race on one file. DLM
|
||||
# serializes them, so the result must be exactly one of the two payloads.
|
||||
echo "==> 3. distributed lock: concurrent writers produce intact data"
|
||||
racefile="dlm_race.bin"
|
||||
payloadA="${WORK}/dlm_raceA.bin"
|
||||
head -c 1048576 /dev/urandom >"${payloadA}"
|
||||
payloadB="direct-write-from-mount2-payload"
|
||||
(smb "put ${payloadA} ${racefile}" >/dev/null 2>&1) &
|
||||
(printf '%s' "${payloadB}" >"${MOUNT2_SHARE}/${racefile}") &
|
||||
wait
|
||||
racegot="${WORK}/dlm_race_got.bin"
|
||||
if smb "get ${racefile} ${racegot}" >/dev/null 2>&1 &&
|
||||
{ [[ "$(md5 "${racegot}")" == "$(md5 "${payloadA}")" ]] || [[ "$(cat "${racegot}")" == "${payloadB}" ]]; }; then
|
||||
pass "concurrent same-file writers leave one intact payload"
|
||||
else
|
||||
fail "concurrent same-file writers leave one intact payload"
|
||||
fi
|
||||
|
||||
# 4. Concurrent writers to distinct files ------------------------------------
|
||||
echo "==> 4. concurrent writers to distinct files"
|
||||
n=6
|
||||
declare -a srcs=()
|
||||
for i in $(seq 1 "${n}"); do
|
||||
s="${WORK}/cc_${i}.bin"
|
||||
head -c 1048576 /dev/urandom >"${s}"
|
||||
srcs+=("${s}")
|
||||
(smb "put ${s} concurrent_${i}.bin" >/dev/null 2>&1) &
|
||||
done
|
||||
wait
|
||||
all_ok=true
|
||||
for i in $(seq 1 "${n}"); do
|
||||
g="${WORK}/cc_got_${i}.bin"
|
||||
if ! smb "get concurrent_${i}.bin ${g}" >/dev/null 2>&1 ||
|
||||
[[ "$(md5 "${srcs[$((i - 1))]}")" != "$(md5 "${g}")" ]]; then
|
||||
all_ok=false
|
||||
fi
|
||||
done
|
||||
if ${all_ok}; then
|
||||
pass "${n} concurrent distinct-file writes all intact"
|
||||
else
|
||||
fail "${n} concurrent distinct-file writes all intact"
|
||||
fi
|
||||
|
||||
echo
|
||||
echo "==> Summary: ${PASS} passed, ${FAIL} failed"
|
||||
[[ "${FAIL}" -eq 0 ]]
|
||||
Executable
+203
@@ -0,0 +1,203 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# Run the SMB (Samba) integration test against a SeaweedFS FUSE mount.
|
||||
#
|
||||
# Pipeline:
|
||||
# 1. start a self-contained "weed mini" (master + volume + filer in one)
|
||||
# 2. mount the filesystem with "weed mount"
|
||||
# 3. export a subdirectory of the mount over SMB with smbd
|
||||
# 4. drive the share with smbclient (test/samba/smb_tests.sh)
|
||||
#
|
||||
# Everything runs as the current user on unprivileged ports, so no sudo is
|
||||
# required. State lives under a temp work dir and is removed on exit.
|
||||
#
|
||||
# Requirements: weed in $PATH, fusermount3, and Samba's smbd / smbclient /
|
||||
# smbpasswd (Debian/Ubuntu: apt-get install samba smbclient).
|
||||
#
|
||||
# Usage:
|
||||
# test/samba/run.sh
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
WEED_BIN="${WEED_BIN:-weed}"
|
||||
WORK_DIR="${WORK_DIR:-$(mktemp -d /tmp/seaweedfs-samba.XXXXXX)}"
|
||||
MOUNT_DIR="${MOUNT_DIR:-${WORK_DIR}/mnt}"
|
||||
MOUNT2_DIR="${MOUNT2_DIR:-${WORK_DIR}/mnt2}"
|
||||
DATA_DIR="${DATA_DIR:-${WORK_DIR}/data}"
|
||||
LOG_DIR="${LOG_DIR:-${WORK_DIR}/logs}"
|
||||
STATE_DIR="${WORK_DIR}/samba"
|
||||
SHARE_DIR="${MOUNT_DIR}/share"
|
||||
SHARE_DIR2="${MOUNT2_DIR}/share"
|
||||
|
||||
FILER_PORT="${FILER_PORT:-28888}"
|
||||
FILER_ADDR="127.0.0.1:${FILER_PORT}"
|
||||
SMB_PORT="${SMB_PORT:-4450}"
|
||||
SMB_SHARE="seaweedfs"
|
||||
SMB_USER="${SMB_USER:-$(id -un)}"
|
||||
SMB_PASS="${SMB_PASS:-seaweedfs}"
|
||||
|
||||
SMBD_BIN="$(command -v smbd || echo /usr/sbin/smbd)"
|
||||
SMBPASSWD_BIN="$(command -v smbpasswd || echo /usr/bin/smbpasswd)"
|
||||
|
||||
CI_LOG_DIR="/tmp/seaweedfs-samba-logs"
|
||||
|
||||
mini_pid=""
|
||||
mount_pid=""
|
||||
mount2_pid=""
|
||||
smbd_pid=""
|
||||
|
||||
unmount_dir() {
|
||||
local dir="$1"
|
||||
if mountpoint -q "${dir}" 2>/dev/null; then
|
||||
fusermount3 -u "${dir}" 2>/dev/null ||
|
||||
fusermount -u "${dir}" 2>/dev/null || true
|
||||
fi
|
||||
}
|
||||
|
||||
cleanup() {
|
||||
set +e
|
||||
if [[ -n "${smbd_pid}" ]] && kill -0 "${smbd_pid}" 2>/dev/null; then
|
||||
kill -TERM "${smbd_pid}" 2>/dev/null || true
|
||||
wait "${smbd_pid}" 2>/dev/null || true
|
||||
fi
|
||||
for p in "${mount_pid}" "${mount2_pid}"; do
|
||||
if [[ -n "${p}" ]] && kill -0 "${p}" 2>/dev/null; then
|
||||
kill -TERM "${p}" 2>/dev/null || true
|
||||
wait "${p}" 2>/dev/null || true
|
||||
fi
|
||||
done
|
||||
unmount_dir "${MOUNT_DIR}"
|
||||
unmount_dir "${MOUNT2_DIR}"
|
||||
if [[ -n "${mini_pid}" ]] && kill -0 "${mini_pid}" 2>/dev/null; then
|
||||
kill -TERM "${mini_pid}" 2>/dev/null || true
|
||||
wait "${mini_pid}" 2>/dev/null || true
|
||||
fi
|
||||
# Copy logs to a fixed path for CI artifact upload.
|
||||
mkdir -p "${CI_LOG_DIR}"
|
||||
cp "${LOG_DIR}"/*.log "${LOG_DIR}"/*.out "${STATE_DIR}/smbd.log" "${CI_LOG_DIR}/" 2>/dev/null || true
|
||||
}
|
||||
trap cleanup EXIT INT TERM
|
||||
|
||||
mkdir -p "${MOUNT_DIR}" "${MOUNT2_DIR}" "${DATA_DIR}" "${LOG_DIR}" \
|
||||
"${STATE_DIR}/private" "${STATE_DIR}/state" "${STATE_DIR}/cache" \
|
||||
"${STATE_DIR}/lock" "${STATE_DIR}/pid" "${STATE_DIR}/ncalrpc"
|
||||
|
||||
# --- 1. weed mini -----------------------------------------------------------
|
||||
echo "==> Starting weed mini on ${FILER_ADDR}"
|
||||
"${WEED_BIN}" mini \
|
||||
-dir="${DATA_DIR}" \
|
||||
-ip=127.0.0.1 \
|
||||
-filer.port="${FILER_PORT}" \
|
||||
-s3=false \
|
||||
-webdav=false \
|
||||
-admin.ui=false \
|
||||
>"${LOG_DIR}/mini.log" 2>&1 &
|
||||
mini_pid=$!
|
||||
|
||||
for i in $(seq 1 60); do
|
||||
if (echo >"/dev/tcp/127.0.0.1/${FILER_PORT}") 2>/dev/null; then
|
||||
break
|
||||
fi
|
||||
if ! kill -0 "${mini_pid}" 2>/dev/null; then
|
||||
echo "weed mini exited early; log tail:" >&2
|
||||
tail -n 100 "${LOG_DIR}/mini.log" >&2 || true
|
||||
exit 1
|
||||
fi
|
||||
sleep 0.5
|
||||
done
|
||||
if ! (echo >"/dev/tcp/127.0.0.1/${FILER_PORT}") 2>/dev/null; then
|
||||
echo "weed mini filer did not become reachable within 30s; log tail:" >&2
|
||||
tail -n 100 "${LOG_DIR}/mini.log" >&2 || true
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# --- 2. weed mount (two mounts, both with -dlm) -----------------------------
|
||||
# mount_with_dlm <mountpoint> <logfile> <pid-var-name>
|
||||
mount_with_dlm() {
|
||||
local dir="$1" log="$2" pidvar="$3" pid
|
||||
echo "==> Mounting SeaweedFS at ${dir} with -dlm"
|
||||
"${WEED_BIN}" mount \
|
||||
-filer="${FILER_ADDR}" \
|
||||
-dir="${dir}" \
|
||||
-filer.path=/ \
|
||||
-dirAutoCreate \
|
||||
-dlm \
|
||||
>"${log}" 2>&1 &
|
||||
pid=$!
|
||||
printf -v "${pidvar}" '%s' "${pid}"
|
||||
for _ in $(seq 1 60); do
|
||||
if mountpoint -q "${dir}"; then
|
||||
return 0
|
||||
fi
|
||||
if ! kill -0 "${pid}" 2>/dev/null; then
|
||||
echo "weed mount (${dir}) exited early; log tail:" >&2
|
||||
tail -n 100 "${log}" >&2 || true
|
||||
exit 1
|
||||
fi
|
||||
sleep 0.5
|
||||
done
|
||||
echo "FUSE mount ${dir} did not come up within 30s" >&2
|
||||
tail -n 100 "${log}" >&2 || true
|
||||
exit 1
|
||||
}
|
||||
|
||||
mount_with_dlm "${MOUNT_DIR}" "${LOG_DIR}/mount.log" mount_pid
|
||||
mount_with_dlm "${MOUNT2_DIR}" "${LOG_DIR}/mount2.log" mount2_pid
|
||||
|
||||
mkdir -p "${SHARE_DIR}"
|
||||
|
||||
# --- 3. smbd ----------------------------------------------------------------
|
||||
echo "==> Generating smb.conf and starting smbd on port ${SMB_PORT}"
|
||||
SMB_CONF="${STATE_DIR}/smb.conf"
|
||||
sed -e "s#@SHARE_PATH@#${SHARE_DIR}#g" \
|
||||
-e "s#@STATE_DIR@#${STATE_DIR}#g" \
|
||||
-e "s#@SMB_PORT@#${SMB_PORT}#g" \
|
||||
-e "s#@FORCE_USER@#${SMB_USER}#g" \
|
||||
"${SCRIPT_DIR}/smb.conf.template" >"${SMB_CONF}"
|
||||
|
||||
printf '%s\n%s\n' "${SMB_PASS}" "${SMB_PASS}" |
|
||||
"${SMBPASSWD_BIN}" -c "${SMB_CONF}" -a -s "${SMB_USER}"
|
||||
|
||||
"${SMBD_BIN}" -F --no-process-group -s "${SMB_CONF}" >"${LOG_DIR}/smbd.out" 2>&1 &
|
||||
smbd_pid=$!
|
||||
|
||||
for i in $(seq 1 60); do
|
||||
if (echo >"/dev/tcp/127.0.0.1/${SMB_PORT}") 2>/dev/null; then
|
||||
break
|
||||
fi
|
||||
if ! kill -0 "${smbd_pid}" 2>/dev/null; then
|
||||
echo "smbd exited early; log tail:" >&2
|
||||
tail -n 100 "${LOG_DIR}/smbd.out" "${STATE_DIR}/smbd.log" 2>/dev/null >&2 || true
|
||||
exit 1
|
||||
fi
|
||||
sleep 0.5
|
||||
done
|
||||
if ! (echo >"/dev/tcp/127.0.0.1/${SMB_PORT}") 2>/dev/null; then
|
||||
echo "smbd did not become reachable within 30s; log tail:" >&2
|
||||
tail -n 100 "${LOG_DIR}/smbd.out" "${STATE_DIR}/smbd.log" 2>/dev/null >&2 || true
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# --- 4. run the test batteries ---------------------------------------------
|
||||
rc=0
|
||||
|
||||
echo "==> Running SMB functional test battery"
|
||||
SMB_HOST=127.0.0.1 \
|
||||
SMB_SHARE="${SMB_SHARE}" \
|
||||
SMB_PORT="${SMB_PORT}" \
|
||||
SMB_USER="${SMB_USER}" \
|
||||
SMB_PASS="${SMB_PASS}" \
|
||||
SHARE_FS_PATH="${SHARE_DIR}" \
|
||||
"${SCRIPT_DIR}/smb_tests.sh" || rc=1
|
||||
|
||||
echo "==> Running SMB locking / concurrency test battery"
|
||||
SMB_HOST=127.0.0.1 \
|
||||
SMB_SHARE="${SMB_SHARE}" \
|
||||
SMB_PORT="${SMB_PORT}" \
|
||||
SMB_USER="${SMB_USER}" \
|
||||
SMB_PASS="${SMB_PASS}" \
|
||||
MOUNT_SHARE="${SHARE_DIR}" \
|
||||
MOUNT2_SHARE="${SHARE_DIR2}" \
|
||||
"${SCRIPT_DIR}/lock_tests.sh" || rc=1
|
||||
|
||||
exit "${rc}"
|
||||
Executable
+24
@@ -0,0 +1,24 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# Runs the SMB test batteries inside the samba container against the local smbd,
|
||||
# which serves /mnt/seaweedfs/share over a SeaweedFS FUSE mount. A second FUSE
|
||||
# mount (/mnt/seaweedfs2) backs the distributed-locking tests.
|
||||
# Invoked via: docker compose exec samba /run_inside_container.sh
|
||||
set -euo pipefail
|
||||
|
||||
export SMB_HOST=127.0.0.1
|
||||
export SMB_SHARE=seaweedfs
|
||||
export SMB_PORT="${SMB_PORT:-445}"
|
||||
export SMB_USER="${SMB_USER:-smbtest}"
|
||||
export SMB_PASS="${SMB_PASS:-smbtest}"
|
||||
export SHARE_FS_PATH="${SHARE_FS_PATH:-/mnt/seaweedfs/share}"
|
||||
export MOUNT_SHARE="${MOUNT_SHARE:-/mnt/seaweedfs/share}"
|
||||
export MOUNT2_SHARE="${MOUNT2_SHARE:-/mnt/seaweedfs2/share}"
|
||||
|
||||
rc=0
|
||||
echo "############ SMB functional tests ############"
|
||||
/smb_tests.sh || rc=1
|
||||
echo
|
||||
echo "############ SMB locking / concurrency tests ############"
|
||||
/lock_tests.sh || rc=1
|
||||
exit "${rc}"
|
||||
@@ -0,0 +1,52 @@
|
||||
[global]
|
||||
server role = standalone server
|
||||
workgroup = WORKGROUP
|
||||
server string = SeaweedFS FUSE Samba test
|
||||
security = user
|
||||
server min protocol = SMB2
|
||||
smb ports = @SMB_PORT@
|
||||
bind interfaces only = yes
|
||||
interfaces = lo 127.0.0.1
|
||||
|
||||
# Self-contained state so smbd can run rootless and leaves nothing behind
|
||||
# outside the test work directory.
|
||||
private dir = @STATE_DIR@/private
|
||||
state directory = @STATE_DIR@/state
|
||||
cache directory = @STATE_DIR@/cache
|
||||
lock directory = @STATE_DIR@/lock
|
||||
pid directory = @STATE_DIR@/pid
|
||||
ncalrpc dir = @STATE_DIR@/ncalrpc
|
||||
log file = @STATE_DIR@/smbd.log
|
||||
log level = 1
|
||||
usershare max shares = 0
|
||||
|
||||
# No printing subsystem in a file-server test.
|
||||
load printers = no
|
||||
printing = bsd
|
||||
printcap name = /dev/null
|
||||
disable spoolss = yes
|
||||
|
||||
# The SeaweedFS FUSE mount does not implement the full xattr / DOS-attribute
|
||||
# surface Samba uses by default. Disabling these avoids spurious
|
||||
# NT_STATUS_NOT_SUPPORTED / EOPNOTSUPP errors unrelated to data integrity.
|
||||
ea support = no
|
||||
store dos attributes = no
|
||||
map archive = no
|
||||
map hidden = no
|
||||
map system = no
|
||||
map readonly = no
|
||||
|
||||
# A network-filesystem backend should not advertise local oplocks/leases.
|
||||
oplocks = no
|
||||
level2 oplocks = no
|
||||
kernel oplocks = no
|
||||
posix locking = no
|
||||
|
||||
[seaweedfs]
|
||||
path = @SHARE_PATH@
|
||||
comment = SeaweedFS share backed by a FUSE mount
|
||||
browseable = yes
|
||||
read only = no
|
||||
create mask = 0644
|
||||
directory mask = 0755
|
||||
force user = @FORCE_USER@
|
||||
Executable
+172
@@ -0,0 +1,172 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# SMB protocol test battery against a Samba share backed by a SeaweedFS FUSE
|
||||
# mount. Driven both by the local runner (test/samba/run.sh) and by the Docker
|
||||
# harness (run_inside_container.sh).
|
||||
#
|
||||
# Required env:
|
||||
# SMB_USER samba username
|
||||
# SMB_PASS samba password
|
||||
# Optional env:
|
||||
# SMB_HOST samba host (default 127.0.0.1)
|
||||
# SMB_SHARE share name (default seaweedfs)
|
||||
# SMB_PORT smbd port (default 445)
|
||||
# SHARE_FS_PATH directory on the FUSE mount that backs the share. When set,
|
||||
# the suite also checks cross-protocol consistency: data written
|
||||
# over SMB is visible on the FUSE mount, and vice versa.
|
||||
set -uo pipefail
|
||||
|
||||
SMB_HOST="${SMB_HOST:-127.0.0.1}"
|
||||
SMB_SHARE="${SMB_SHARE:-seaweedfs}"
|
||||
SMB_PORT="${SMB_PORT:-445}"
|
||||
SMB_USER="${SMB_USER:?SMB_USER is required}"
|
||||
SMB_PASS="${SMB_PASS:?SMB_PASS is required}"
|
||||
SHARE_FS_PATH="${SHARE_FS_PATH:-}"
|
||||
|
||||
WORK="$(mktemp -d /tmp/samba-smbtest.XXXXXX)"
|
||||
trap 'rm -rf "${WORK}"' EXIT
|
||||
|
||||
PASS=0
|
||||
FAIL=0
|
||||
pass() { printf ' [PASS] %s\n' "$1"; PASS=$((PASS + 1)); }
|
||||
fail() { printf ' [FAIL] %s\n' "$1"; FAIL=$((FAIL + 1)); }
|
||||
|
||||
# Run one or more smbclient commands (separated by ';') against the share.
|
||||
smb() {
|
||||
smbclient "//${SMB_HOST}/${SMB_SHARE}" -p "${SMB_PORT}" \
|
||||
-U "${SMB_USER}%${SMB_PASS}" -m SMB3 -c "$1"
|
||||
}
|
||||
|
||||
md5() { md5sum "$1" | awk '{print $1}'; }
|
||||
|
||||
echo "==> Target //${SMB_HOST}/${SMB_SHARE} (port ${SMB_PORT}) as ${SMB_USER}"
|
||||
[[ -n "${SHARE_FS_PATH}" ]] && echo "==> Cross-protocol checks against ${SHARE_FS_PATH}"
|
||||
|
||||
# 1. Connectivity ------------------------------------------------------------
|
||||
echo "==> 1. connectivity"
|
||||
if smb "ls" >/dev/null 2>&1; then
|
||||
pass "connect and list share root"
|
||||
else
|
||||
fail "connect and list share root"
|
||||
fi
|
||||
|
||||
# 2. Upload / download round-trip -------------------------------------------
|
||||
echo "==> 2. upload / download round-trip"
|
||||
src="${WORK}/src.bin"
|
||||
head -c 1048576 /dev/urandom >"${src}" # 1 MiB
|
||||
if smb "put ${src} roundtrip.bin" >/dev/null 2>&1; then
|
||||
pass "put 1 MiB file"
|
||||
else
|
||||
fail "put 1 MiB file"
|
||||
fi
|
||||
got="${WORK}/got.bin"
|
||||
if smb "get roundtrip.bin ${got}" >/dev/null 2>&1 && [[ "$(md5 "${src}")" == "$(md5 "${got}")" ]]; then
|
||||
pass "get returns identical content"
|
||||
else
|
||||
fail "get returns identical content"
|
||||
fi
|
||||
if [[ -n "${SHARE_FS_PATH}" ]]; then
|
||||
if [[ -f "${SHARE_FS_PATH}/roundtrip.bin" ]] && [[ "$(md5 "${SHARE_FS_PATH}/roundtrip.bin")" == "$(md5 "${src}")" ]]; then
|
||||
pass "SMB-written file visible on FUSE mount with identical content"
|
||||
else
|
||||
fail "SMB-written file visible on FUSE mount with identical content"
|
||||
fi
|
||||
fi
|
||||
|
||||
# 3. Directory operations ----------------------------------------------------
|
||||
echo "==> 3. directory operations"
|
||||
if smb "mkdir docs; cd docs; put ${src} nested.bin; ls" >/dev/null 2>&1; then
|
||||
pass "mkdir + put into subdirectory"
|
||||
else
|
||||
fail "mkdir + put into subdirectory"
|
||||
fi
|
||||
if [[ -z "${SHARE_FS_PATH}" || -f "${SHARE_FS_PATH}/docs/nested.bin" ]]; then
|
||||
pass "nested file present"
|
||||
else
|
||||
fail "nested file present"
|
||||
fi
|
||||
|
||||
# 4. Rename ------------------------------------------------------------------
|
||||
echo "==> 4. rename"
|
||||
if smb "rename roundtrip.bin renamed.bin" >/dev/null 2>&1; then
|
||||
pass "rename file"
|
||||
else
|
||||
fail "rename file"
|
||||
fi
|
||||
renback="${WORK}/renamed.bin"
|
||||
if smb "get renamed.bin ${renback}" >/dev/null 2>&1 && [[ "$(md5 "${renback}")" == "$(md5 "${src}")" ]]; then
|
||||
pass "renamed file readable with original content"
|
||||
else
|
||||
fail "renamed file readable with original content"
|
||||
fi
|
||||
if [[ -n "${SHARE_FS_PATH}" ]]; then
|
||||
if [[ -f "${SHARE_FS_PATH}/renamed.bin" && ! -e "${SHARE_FS_PATH}/roundtrip.bin" ]]; then
|
||||
pass "rename reflected on FUSE mount"
|
||||
else
|
||||
fail "rename reflected on FUSE mount"
|
||||
fi
|
||||
fi
|
||||
|
||||
# 5. Large file (exercises SeaweedFS chunking) -------------------------------
|
||||
echo "==> 5. large file (SeaweedFS chunking)"
|
||||
big="${WORK}/big.bin"
|
||||
head -c 67108864 /dev/urandom >"${big}" # 64 MiB
|
||||
bigback="${WORK}/big.back"
|
||||
if smb "put ${big} big.bin" >/dev/null 2>&1 &&
|
||||
smb "get big.bin ${bigback}" >/dev/null 2>&1 &&
|
||||
[[ "$(md5 "${big}")" == "$(md5 "${bigback}")" ]]; then
|
||||
pass "64 MiB put/get round-trip"
|
||||
else
|
||||
fail "64 MiB put/get round-trip"
|
||||
fi
|
||||
|
||||
# 6. Recursive upload --------------------------------------------------------
|
||||
echo "==> 6. recursive upload"
|
||||
tree="${WORK}/tree"
|
||||
mkdir -p "${tree}/a/b"
|
||||
echo one >"${tree}/f1.txt"
|
||||
echo two >"${tree}/a/f2.txt"
|
||||
echo three >"${tree}/a/b/f3.txt"
|
||||
if (cd "${WORK}" && smb "recurse ON; prompt OFF; mput tree" >/dev/null 2>&1) &&
|
||||
{ [[ -z "${SHARE_FS_PATH}" ]] || [[ -f "${SHARE_FS_PATH}/tree/a/b/f3.txt" ]]; }; then
|
||||
pass "recursive mput"
|
||||
else
|
||||
fail "recursive mput"
|
||||
fi
|
||||
|
||||
# 7. Cross-protocol read (FUSE writes, SMB reads) ----------------------------
|
||||
if [[ -n "${SHARE_FS_PATH}" ]]; then
|
||||
echo "==> 7. cross-protocol read (FUSE write -> SMB read)"
|
||||
echo "written-via-fuse" >"${SHARE_FS_PATH}/from_fuse.txt"
|
||||
cpb="${WORK}/from_fuse.back"
|
||||
if smb "get from_fuse.txt ${cpb}" >/dev/null 2>&1 && grep -q written-via-fuse "${cpb}"; then
|
||||
pass "FUSE-written file readable over SMB"
|
||||
else
|
||||
fail "FUSE-written file readable over SMB"
|
||||
fi
|
||||
fi
|
||||
|
||||
# 8. Delete ------------------------------------------------------------------
|
||||
echo "==> 8. delete"
|
||||
smb "del renamed.bin" >/dev/null 2>&1
|
||||
smb "del big.bin" >/dev/null 2>&1
|
||||
smb "deltree docs" >/dev/null 2>&1
|
||||
smb "deltree tree" >/dev/null 2>&1
|
||||
if [[ -n "${SHARE_FS_PATH}" ]]; then
|
||||
if [[ ! -e "${SHARE_FS_PATH}/renamed.bin" && ! -e "${SHARE_FS_PATH}/big.bin" &&
|
||||
! -e "${SHARE_FS_PATH}/docs" && ! -e "${SHARE_FS_PATH}/tree" ]]; then
|
||||
pass "delete files and directory trees"
|
||||
else
|
||||
fail "delete files and directory trees"
|
||||
fi
|
||||
else
|
||||
if ! smb "get renamed.bin /dev/null" >/dev/null 2>&1; then
|
||||
pass "deleted file no longer retrievable"
|
||||
else
|
||||
fail "deleted file no longer retrievable"
|
||||
fi
|
||||
fi
|
||||
|
||||
echo
|
||||
echo "==> Summary: ${PASS} passed, ${FAIL} failed"
|
||||
[[ "${FAIL}" -eq 0 ]]
|
||||
+25
-2
@@ -11,6 +11,29 @@ import (
|
||||
// GrpcPortOffset is the offset weed mini uses to derive gRPC ports from HTTP ports.
|
||||
const GrpcPortOffset = 10000
|
||||
|
||||
// miniDefaultPorts are the weed mini flag defaults (see weed/command/mini.go).
|
||||
// A test only overrides services it uses; unspecified services still bind
|
||||
// these defaults, so allocation must avoid handing them out (or any value
|
||||
// whose gRPC offset would collide with them).
|
||||
var miniDefaultPorts = []int{
|
||||
9333, // master.port
|
||||
8888, // filer.port
|
||||
9340, // volume.port
|
||||
8333, // s3.port
|
||||
8181, // s3.port.iceberg
|
||||
7333, // webdav.port
|
||||
23646, // admin.port
|
||||
}
|
||||
|
||||
func reservedMiniPorts() map[int]bool {
|
||||
r := make(map[int]bool, len(miniDefaultPorts)*2)
|
||||
for _, p := range miniDefaultPorts {
|
||||
r[p] = true
|
||||
r[p+GrpcPortOffset] = true
|
||||
}
|
||||
return r
|
||||
}
|
||||
|
||||
// AllocatePorts allocates count unique free ports atomically.
|
||||
// All listeners are held open until every port is obtained, preventing
|
||||
// the OS from recycling a port between successive allocations.
|
||||
@@ -63,7 +86,7 @@ func AllocateMiniPorts(count int) ([]int, error) {
|
||||
minPort = 10000
|
||||
maxPort = 55000
|
||||
)
|
||||
reserved := make(map[int]bool)
|
||||
reserved := reservedMiniPorts()
|
||||
ports := make([]int, 0, count)
|
||||
var listeners []net.Listener
|
||||
defer func() {
|
||||
@@ -135,7 +158,7 @@ func AllocatePortSet(miniCount, regularCount int) (mini []int, regular []int, er
|
||||
minPort = 10000
|
||||
maxPort = 55000
|
||||
)
|
||||
reserved := make(map[int]bool)
|
||||
reserved := reservedMiniPorts()
|
||||
mini = make([]int, 0, miniCount)
|
||||
var listeners []net.Listener
|
||||
defer func() {
|
||||
|
||||
@@ -2,6 +2,29 @@ package testutil
|
||||
|
||||
import "testing"
|
||||
|
||||
// AllocateMiniPorts must never hand out a port that weed mini will reserve
|
||||
// for one of its default services (or that default's gRPC offset). A real
|
||||
// failure: Filer was given 33646 (Admin default 23646 + GrpcPortOffset),
|
||||
// which mini then refused as "reserved for gRPC calculation".
|
||||
func TestAllocateMiniPortsAvoidsMiniDefaults(t *testing.T) {
|
||||
reserved := reservedMiniPorts()
|
||||
for iter := 0; iter < 200; iter++ {
|
||||
ports, err := AllocateMiniPorts(4)
|
||||
if err != nil {
|
||||
t.Fatalf("iter %d: AllocateMiniPorts: %v", iter, err)
|
||||
}
|
||||
for _, p := range ports {
|
||||
if reserved[p] {
|
||||
t.Fatalf("iter %d: allocated port %d is a mini default (or gRPC offset)", iter, p)
|
||||
}
|
||||
if reserved[p+GrpcPortOffset] {
|
||||
t.Fatalf("iter %d: allocated port %d has gRPC offset %d colliding with a mini default",
|
||||
iter, p, p+GrpcPortOffset)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestAllocatePortSetNoGrpcCollision(t *testing.T) {
|
||||
// Run a few iterations to catch the OS-recycles-just-closed-port race
|
||||
// that previously hit regular ports when the mini gRPC offset was freed
|
||||
|
||||
@@ -75,7 +75,7 @@ func TestStatsEndpoints(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestStatusPrettyJsonAndJsonp(t *testing.T) {
|
||||
func TestStatusPrettyJsonAndCallbackIgnored(t *testing.T) {
|
||||
if testing.Short() {
|
||||
t.Skip("skipping integration test in short mode")
|
||||
}
|
||||
@@ -93,29 +93,29 @@ func TestStatusPrettyJsonAndJsonp(t *testing.T) {
|
||||
if len(lines) < 3 {
|
||||
t.Fatalf("/status?pretty=y expected multi-line indented JSON, got %d lines: %s", len(lines), string(prettyBody))
|
||||
}
|
||||
// Verify the body is valid JSON
|
||||
var prettyPayload map[string]interface{}
|
||||
if err := json.Unmarshal(prettyBody, &prettyPayload); err != nil {
|
||||
t.Fatalf("/status?pretty=y is not valid JSON: %v", err)
|
||||
}
|
||||
|
||||
// ?callback=myFunc — expect JSONP wrapping
|
||||
jsonpResp := framework.DoRequest(t, client, mustNewRequest(t, http.MethodGet, cluster.VolumeAdminURL()+"/status?callback=myFunc"))
|
||||
jsonpBody := framework.ReadAllAndClose(t, jsonpResp)
|
||||
if jsonpResp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("/status?callback=myFunc expected 200, got %d", jsonpResp.StatusCode)
|
||||
// ?callback=myFunc — must be ignored; response is plain JSON with nosniff.
|
||||
cbResp := framework.DoRequest(t, client, mustNewRequest(t, http.MethodGet, cluster.VolumeAdminURL()+"/status?callback=myFunc"))
|
||||
cbBody := framework.ReadAllAndClose(t, cbResp)
|
||||
if cbResp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("/status?callback=myFunc expected 200, got %d", cbResp.StatusCode)
|
||||
}
|
||||
bodyStr := string(jsonpBody)
|
||||
if !strings.HasPrefix(bodyStr, "myFunc(") {
|
||||
t.Fatalf("/status?callback=myFunc expected body to start with 'myFunc(', got prefix: %q", bodyStr[:min(len(bodyStr), 30)])
|
||||
if ct := cbResp.Header.Get("Content-Type"); !strings.Contains(ct, "application/json") {
|
||||
t.Fatalf("/status?callback=myFunc expected Content-Type application/json, got %q", ct)
|
||||
}
|
||||
trimmed := strings.TrimRight(bodyStr, "\n; ")
|
||||
if !strings.HasSuffix(trimmed, ")") {
|
||||
t.Fatalf("/status?callback=myFunc expected body to end with ')', got suffix: %q", trimmed[max(0, len(trimmed)-10):])
|
||||
if nosniff := cbResp.Header.Get("X-Content-Type-Options"); nosniff != "nosniff" {
|
||||
t.Fatalf("/status?callback=myFunc expected X-Content-Type-Options nosniff, got %q", nosniff)
|
||||
}
|
||||
// Content-Type should be application/javascript for JSONP
|
||||
if ct := jsonpResp.Header.Get("Content-Type"); !strings.Contains(ct, "javascript") {
|
||||
t.Fatalf("/status?callback=myFunc expected Content-Type containing 'javascript', got %q", ct)
|
||||
if strings.Contains(string(cbBody), "myFunc(") {
|
||||
t.Fatalf("/status?callback=myFunc must not wrap response in callback; body: %q", string(cbBody))
|
||||
}
|
||||
var cbPayload map[string]interface{}
|
||||
if err := json.Unmarshal(cbBody, &cbPayload); err != nil {
|
||||
t.Fatalf("/status?callback=myFunc is not valid JSON: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -2,6 +2,8 @@ package volume_server_http_test
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"testing"
|
||||
"time"
|
||||
@@ -44,6 +46,17 @@ func TestReplicatedUploadSucceedsImmediatelyAfterAllocate(t *testing.T) {
|
||||
fid := framework.NewFileID(volumeID, 881001, 0x0B0C0D0E)
|
||||
payload := []byte("replicated-upload-after-allocate")
|
||||
|
||||
// The master only learns about replica locations through volume-server
|
||||
// heartbeats, which lag behind the direct AllocateVolume gRPC calls above.
|
||||
// In production a client obtains its fid from the master assign flow, which
|
||||
// guarantees the master already knows every replica; this test crafts the
|
||||
// fid by hand, so the replicated write would otherwise look up the master
|
||||
// before the second replica is registered and fail with a 500. Wait until
|
||||
// the master reports both replicas before uploading.
|
||||
if !waitForMasterReplicaCount(t, client, clusterHarness.MasterURL(), volumeID, 2, 10*time.Second) {
|
||||
t.Fatalf("master did not report 2 replica locations for volume %d within deadline", volumeID)
|
||||
}
|
||||
|
||||
uploadResp := framework.UploadBytes(t, client, clusterHarness.VolumeAdminURL(0), fid, payload)
|
||||
_ = framework.ReadAllAndClose(t, uploadResp)
|
||||
if uploadResp.StatusCode != http.StatusCreated {
|
||||
@@ -61,3 +74,29 @@ func TestReplicatedUploadSucceedsImmediatelyAfterAllocate(t *testing.T) {
|
||||
t.Fatalf("replica body mismatch: got %q want %q", string(replicaBody), string(payload))
|
||||
}
|
||||
}
|
||||
|
||||
// waitForMasterReplicaCount polls the master volume lookup until it reports at
|
||||
// least want locations for volumeID, or the timeout elapses.
|
||||
func waitForMasterReplicaCount(t testing.TB, client *http.Client, masterURL string, volumeID uint32, want int, timeout time.Duration) bool {
|
||||
t.Helper()
|
||||
|
||||
lookupURL := fmt.Sprintf("%s/dir/lookup?volumeId=%d", masterURL, volumeID)
|
||||
deadline := time.Now().Add(timeout)
|
||||
for time.Now().Before(deadline) {
|
||||
resp := framework.DoRequest(t, client, mustNewRequest(t, http.MethodGet, lookupURL))
|
||||
body := framework.ReadAllAndClose(t, resp)
|
||||
if resp.StatusCode == http.StatusOK {
|
||||
var result struct {
|
||||
Locations []struct {
|
||||
Url string `json:"url"`
|
||||
} `json:"locations"`
|
||||
}
|
||||
if err := json.Unmarshal(body, &result); err == nil && len(result.Locations) >= want {
|
||||
return true
|
||||
}
|
||||
}
|
||||
time.Sleep(200 * time.Millisecond)
|
||||
}
|
||||
|
||||
return false
|
||||
}
|
||||
|
||||
@@ -23,6 +23,7 @@ import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/plugin_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/schema_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/security"
|
||||
stats_collect "github.com/seaweedfs/seaweedfs/weed/stats"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/erasure_coding"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/super_block"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
@@ -280,6 +281,8 @@ func NewAdminServer(masters string, templateFS http.FileSystem, dataDir string,
|
||||
go server.monitorVacuumWorker(bgCtx)
|
||||
}
|
||||
|
||||
go server.publishMaintenanceMetrics(bgCtx)
|
||||
|
||||
return server
|
||||
}
|
||||
|
||||
@@ -364,6 +367,55 @@ func (s *AdminServer) monitorVacuumWorker(ctx context.Context) {
|
||||
}
|
||||
}
|
||||
|
||||
// publishMaintenanceMetrics periodically snapshots the maintenance queue and
|
||||
// worker fleet into Prometheus gauges. Counters and durations are recorded at
|
||||
// their event sites; these gauges reflect current state at scrape resolution.
|
||||
func (s *AdminServer) publishMaintenanceMetrics(ctx context.Context) {
|
||||
const interval = 15 * time.Second
|
||||
ticker := time.NewTicker(interval)
|
||||
defer ticker.Stop()
|
||||
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case <-ticker.C:
|
||||
s.collectMaintenanceMetrics()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func (s *AdminServer) collectMaintenanceMetrics() {
|
||||
if s.maintenanceManager == nil {
|
||||
return
|
||||
}
|
||||
|
||||
stats := s.maintenanceManager.GetStats()
|
||||
|
||||
stats_collect.AdminMaintenanceTasksByStatus.Reset()
|
||||
for status, count := range stats.TasksByStatus {
|
||||
stats_collect.AdminMaintenanceTasksByStatus.WithLabelValues(string(status)).Set(float64(count))
|
||||
}
|
||||
|
||||
stats_collect.AdminMaintenanceTasksByType.Reset()
|
||||
for taskType, count := range stats.TasksByType {
|
||||
stats_collect.AdminMaintenanceTasksByType.WithLabelValues(string(taskType)).Set(float64(count))
|
||||
}
|
||||
|
||||
// NextScanTime is only meaningful while the scanner runs; GetStats computes
|
||||
// it unconditionally, so clear the gauge when idle to avoid a stale value.
|
||||
if s.maintenanceManager.IsRunning() && !stats.NextScanTime.IsZero() {
|
||||
stats_collect.AdminMaintenanceNextScanTimestampSeconds.Set(float64(stats.NextScanTime.Unix()))
|
||||
} else {
|
||||
stats_collect.AdminMaintenanceNextScanTimestampSeconds.Set(0)
|
||||
}
|
||||
|
||||
workers, usedSlots, maxSlots := s.maintenanceManager.GetWorkerSlotTotals()
|
||||
stats_collect.AdminWorkersConnected.Set(float64(workers))
|
||||
stats_collect.AdminWorkerSlots.WithLabelValues("used").Set(float64(usedSlots))
|
||||
stats_collect.AdminWorkerSlots.WithLabelValues("max").Set(float64(maxSlots))
|
||||
}
|
||||
|
||||
// loadTaskConfigurationsFromPersistence loads saved task configurations from protobuf files
|
||||
func (s *AdminServer) loadTaskConfigurationsFromPersistence() {
|
||||
if s.configPersistence == nil || !s.configPersistence.IsConfigured() {
|
||||
|
||||
@@ -16,7 +16,6 @@ import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/admin/plugin"
|
||||
"github.com/seaweedfs/seaweedfs/weed/cluster"
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/plugin_pb"
|
||||
"google.golang.org/protobuf/encoding/protojson"
|
||||
@@ -735,8 +734,8 @@ func (s *AdminServer) parseOrBuildClusterContext(raw json.RawMessage) (*plugin_p
|
||||
if len(contextMessage.MasterGrpcAddresses) == 0 {
|
||||
contextMessage.MasterGrpcAddresses = append(contextMessage.MasterGrpcAddresses, fallback.MasterGrpcAddresses...)
|
||||
}
|
||||
if len(contextMessage.FilerGrpcAddresses) == 0 {
|
||||
contextMessage.FilerGrpcAddresses = append(contextMessage.FilerGrpcAddresses, fallback.FilerGrpcAddresses...)
|
||||
if len(contextMessage.FilerAddresses) == 0 {
|
||||
contextMessage.FilerAddresses = append(contextMessage.FilerAddresses, fallback.FilerAddresses...)
|
||||
}
|
||||
if len(contextMessage.VolumeGrpcAddresses) == 0 {
|
||||
contextMessage.VolumeGrpcAddresses = append(contextMessage.VolumeGrpcAddresses, fallback.VolumeGrpcAddresses...)
|
||||
@@ -755,7 +754,7 @@ func (s *AdminServer) parseOrBuildClusterContext(raw json.RawMessage) (*plugin_p
|
||||
func (s *AdminServer) buildDefaultPluginClusterContext() *plugin_pb.ClusterContext {
|
||||
clusterContext := &plugin_pb.ClusterContext{
|
||||
MasterGrpcAddresses: make([]string, 0),
|
||||
FilerGrpcAddresses: make([]string, 0),
|
||||
FilerAddresses: make([]string, 0),
|
||||
VolumeGrpcAddresses: make([]string, 0),
|
||||
S3GrpcAddresses: make([]string, 0),
|
||||
Metadata: map[string]string{
|
||||
@@ -768,20 +767,20 @@ func (s *AdminServer) buildDefaultPluginClusterContext() *plugin_pb.ClusterConte
|
||||
clusterContext.MasterGrpcAddresses = append(clusterContext.MasterGrpcAddresses, masterAddress)
|
||||
}
|
||||
|
||||
// Master returns filers in dual-port form (host:httpPort.grpcPort);
|
||||
// workers dial these directly, so collapse to host:grpcPort first.
|
||||
// Master returns filers in pb.ServerAddress form (host:httpPort.grpcPort).
|
||||
// Forward that verbatim; each worker converts to a gRPC or HTTP address as
|
||||
// it needs (dialing wants gRPC, the admin shell wants the ServerAddress).
|
||||
filerSeen := map[string]struct{}{}
|
||||
for _, filer := range s.GetAllFilers() {
|
||||
filer = strings.TrimSpace(filer)
|
||||
if filer == "" {
|
||||
continue
|
||||
}
|
||||
grpcAddr := pb.ServerAddress(filer).ToGrpcAddress()
|
||||
if _, exists := filerSeen[grpcAddr]; exists {
|
||||
if _, exists := filerSeen[filer]; exists {
|
||||
continue
|
||||
}
|
||||
filerSeen[grpcAddr] = struct{}{}
|
||||
clusterContext.FilerGrpcAddresses = append(clusterContext.FilerGrpcAddresses, grpcAddr)
|
||||
filerSeen[filer] = struct{}{}
|
||||
clusterContext.FilerAddresses = append(clusterContext.FilerAddresses, filer)
|
||||
}
|
||||
|
||||
volumeSeen := map[string]struct{}{}
|
||||
@@ -829,7 +828,7 @@ func (s *AdminServer) buildDefaultPluginClusterContext() *plugin_pb.ClusterConte
|
||||
}
|
||||
|
||||
sort.Strings(clusterContext.MasterGrpcAddresses)
|
||||
sort.Strings(clusterContext.FilerGrpcAddresses)
|
||||
sort.Strings(clusterContext.FilerAddresses)
|
||||
sort.Strings(clusterContext.VolumeGrpcAddresses)
|
||||
sort.Strings(clusterContext.S3GrpcAddresses)
|
||||
|
||||
|
||||
@@ -16,6 +16,7 @@ import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/plugin_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/worker_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/security"
|
||||
stats_collect "github.com/seaweedfs/seaweedfs/weed/stats"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
"google.golang.org/grpc"
|
||||
"google.golang.org/grpc/codes"
|
||||
@@ -233,6 +234,7 @@ func (s *WorkerGrpcServer) WorkerStream(stream worker_pb.WorkerService_WorkerStr
|
||||
}
|
||||
s.connections[workerID] = conn
|
||||
s.connMutex.Unlock()
|
||||
stats_collect.AdminWorkerEventsTotal.WithLabelValues("registered").Inc()
|
||||
|
||||
// Register worker with maintenance manager
|
||||
s.registerWorkerWithManager(conn)
|
||||
@@ -265,11 +267,11 @@ func (s *WorkerGrpcServer) WorkerStream(stream worker_pb.WorkerService_WorkerStr
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
glog.Infof("Worker %s connection closed: %v", workerID, ctx.Err())
|
||||
s.unregisterWorker(conn)
|
||||
s.unregisterWorker(conn, "unregistered")
|
||||
return nil
|
||||
case <-connCtx.Done():
|
||||
glog.Infof("Worker %s connection cancelled", workerID)
|
||||
s.unregisterWorker(conn)
|
||||
s.unregisterWorker(conn, "unregistered")
|
||||
return nil
|
||||
default:
|
||||
}
|
||||
@@ -285,7 +287,7 @@ func (s *WorkerGrpcServer) WorkerStream(stream worker_pb.WorkerService_WorkerStr
|
||||
default:
|
||||
glog.Errorf("Error receiving from worker %s: %v", workerID, err)
|
||||
}
|
||||
s.unregisterWorker(conn)
|
||||
s.unregisterWorker(conn, "unregistered")
|
||||
return err
|
||||
}
|
||||
|
||||
@@ -338,7 +340,7 @@ func (s *WorkerGrpcServer) handleWorkerMessage(conn *WorkerConnection, msg *work
|
||||
|
||||
case *worker_pb.WorkerMessage_Shutdown:
|
||||
glog.Infof("Worker %s shutting down: %s", workerID, m.Shutdown.Reason)
|
||||
s.unregisterWorker(conn)
|
||||
s.unregisterWorker(conn, "unregistered")
|
||||
|
||||
default:
|
||||
glog.Warningf("Unknown message type from worker %s", workerID)
|
||||
@@ -605,7 +607,7 @@ func (s *WorkerGrpcServer) safeCloseOutgoingChannel(conn *WorkerConnection, sour
|
||||
}
|
||||
|
||||
// unregisterWorker removes a worker connection
|
||||
func (s *WorkerGrpcServer) unregisterWorker(conn *WorkerConnection) {
|
||||
func (s *WorkerGrpcServer) unregisterWorker(conn *WorkerConnection, event string) {
|
||||
s.connMutex.Lock()
|
||||
existingConn, exists := s.connections[conn.workerID]
|
||||
if !exists {
|
||||
@@ -624,6 +626,7 @@ func (s *WorkerGrpcServer) unregisterWorker(conn *WorkerConnection) {
|
||||
// Remove from map first to prevent duplicate cleanup attempts
|
||||
delete(s.connections, conn.workerID)
|
||||
s.connMutex.Unlock()
|
||||
stats_collect.AdminWorkerEventsTotal.WithLabelValues(event).Inc()
|
||||
|
||||
// Cancel context to signal goroutines to stop
|
||||
conn.cancel()
|
||||
@@ -665,7 +668,7 @@ func (s *WorkerGrpcServer) cleanupStaleConnections() {
|
||||
|
||||
for _, conn := range toRemove {
|
||||
glog.Warningf("Cleaning up stale worker connection: %s", conn.workerID)
|
||||
s.unregisterWorker(conn)
|
||||
s.unregisterWorker(conn, "stale_removed")
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -26,6 +26,10 @@ type MaintenanceIntegration struct {
|
||||
// Active topology for task detection and target selection
|
||||
activeTopology *topology.ActiveTopology
|
||||
|
||||
// Master's default replication, refreshed by the scanner each cycle and
|
||||
// passed to detectors as the replica-placement fallback (matches the shell).
|
||||
defaultReplicaPlacement string
|
||||
|
||||
// Type conversion maps
|
||||
taskTypeMap map[types.TaskType]MaintenanceTaskType
|
||||
revTaskTypeMap map[MaintenanceTaskType]types.TaskType
|
||||
@@ -219,9 +223,10 @@ func (s *MaintenanceIntegration) ScanWithTaskDetectors(volumeMetrics []*types.Vo
|
||||
|
||||
// Create cluster info
|
||||
clusterInfo := &types.ClusterInfo{
|
||||
TotalVolumes: len(filteredMetrics),
|
||||
LastUpdated: time.Now(),
|
||||
ActiveTopology: s.activeTopology, // Provide ActiveTopology for destination planning
|
||||
TotalVolumes: len(filteredMetrics),
|
||||
LastUpdated: time.Now(),
|
||||
ActiveTopology: s.activeTopology, // Provide ActiveTopology for destination planning
|
||||
DefaultReplicaPlacement: s.defaultReplicaPlacement,
|
||||
}
|
||||
|
||||
// Run detection for each registered task type
|
||||
@@ -271,6 +276,12 @@ func (s *MaintenanceIntegration) ScanWithTaskDetectors(volumeMetrics []*types.Vo
|
||||
return allResults, nil
|
||||
}
|
||||
|
||||
// SetDefaultReplicaPlacement records the master's default replication so detectors
|
||||
// can use it as the replica-placement fallback (matching the shell).
|
||||
func (s *MaintenanceIntegration) SetDefaultReplicaPlacement(replicaPlacement string) {
|
||||
s.defaultReplicaPlacement = replicaPlacement
|
||||
}
|
||||
|
||||
// UpdateTopologyInfo updates the volume shard tracker with topology information for empty servers
|
||||
func (s *MaintenanceIntegration) UpdateTopologyInfo(topologyInfo *master_pb.TopologyInfo) error {
|
||||
// Log topology details before update for diagnostics
|
||||
|
||||
@@ -8,6 +8,7 @@ import (
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/worker_pb"
|
||||
stats_collect "github.com/seaweedfs/seaweedfs/weed/stats"
|
||||
"github.com/seaweedfs/seaweedfs/weed/worker/tasks/balance"
|
||||
"github.com/seaweedfs/seaweedfs/weed/worker/tasks/erasure_coding"
|
||||
"github.com/seaweedfs/seaweedfs/weed/worker/tasks/vacuum"
|
||||
@@ -315,6 +316,7 @@ func (mm *MaintenanceManager) performScan() {
|
||||
glog.Infof("Starting maintenance scan...")
|
||||
|
||||
results, err := mm.scanner.ScanForMaintenanceTasks()
|
||||
stats_collect.AdminMaintenanceLastScanTimestampSeconds.SetToCurrentTime()
|
||||
if err != nil {
|
||||
// Handle scan error
|
||||
mm.mutex.Lock()
|
||||
@@ -518,6 +520,11 @@ func (mm *MaintenanceManager) GetWorkers() []*MaintenanceWorker {
|
||||
return mm.queue.GetWorkers()
|
||||
}
|
||||
|
||||
// GetWorkerSlotTotals returns worker count and aggregate used/max task slots.
|
||||
func (mm *MaintenanceManager) GetWorkerSlotTotals() (workers, used, max int) {
|
||||
return mm.queue.GetWorkerSlotTotals()
|
||||
}
|
||||
|
||||
// TriggerScan manually triggers a maintenance scan
|
||||
func (mm *MaintenanceManager) TriggerScan() error {
|
||||
return mm.triggerScanInternal(true)
|
||||
|
||||
@@ -8,6 +8,7 @@ import (
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
stats_collect "github.com/seaweedfs/seaweedfs/weed/stats"
|
||||
)
|
||||
|
||||
// NewMaintenanceQueue creates a new maintenance queue
|
||||
@@ -482,12 +483,14 @@ func (mq *MaintenanceQueue) CompleteTask(taskID string, error string) {
|
||||
|
||||
// Calculate task duration
|
||||
var duration time.Duration
|
||||
if task.StartedAt != nil {
|
||||
hadStart := task.StartedAt != nil
|
||||
if hadStart {
|
||||
duration = completedTime.Sub(*task.StartedAt)
|
||||
}
|
||||
|
||||
// Capture workerID before it may be cleared during retry
|
||||
originalWorkerID := task.WorkerID
|
||||
taskType := string(task.Type)
|
||||
|
||||
var taskToSave *MaintenanceTask
|
||||
var logFn func()
|
||||
@@ -577,6 +580,18 @@ func (mq *MaintenanceQueue) CompleteTask(taskID string, error string) {
|
||||
}
|
||||
mq.mutex.Unlock()
|
||||
|
||||
// Record terminal-state metrics. A retry leaves the task pending, so it
|
||||
// is not counted as completed or failed here.
|
||||
switch taskStatus {
|
||||
case TaskStatusCompleted:
|
||||
stats_collect.AdminMaintenanceTasksCompletedTotal.WithLabelValues(taskType, "completed").Inc()
|
||||
case TaskStatusFailed:
|
||||
stats_collect.AdminMaintenanceTasksCompletedTotal.WithLabelValues(taskType, "failed").Inc()
|
||||
}
|
||||
if hadStart && (taskStatus == TaskStatusCompleted || taskStatus == TaskStatusFailed) {
|
||||
stats_collect.AdminMaintenanceTaskDurationSeconds.WithLabelValues(taskType).Observe(duration.Seconds())
|
||||
}
|
||||
|
||||
// Only persist non-terminal tasks (retries). Completed/failed tasks stay
|
||||
// in memory for the UI but are not written to disk — they would just
|
||||
// accumulate and slow down future startups.
|
||||
@@ -819,6 +834,20 @@ func (mq *MaintenanceQueue) GetWorkers() []*MaintenanceWorker {
|
||||
return workers
|
||||
}
|
||||
|
||||
// GetWorkerSlotTotals aggregates worker count and used/max task slots under the
|
||||
// lock, so callers don't read live worker fields that task updates mutate.
|
||||
func (mq *MaintenanceQueue) GetWorkerSlotTotals() (workers, used, max int) {
|
||||
mq.mutex.RLock()
|
||||
defer mq.mutex.RUnlock()
|
||||
|
||||
for _, worker := range mq.workers {
|
||||
workers++
|
||||
used += worker.CurrentLoad
|
||||
max += worker.MaxConcurrent
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
// generateTaskID generates a unique ID for tasks
|
||||
func generateTaskID() string {
|
||||
const charset = "abcdefghijklmnopqrstuvwxyz0123456789"
|
||||
|
||||
@@ -51,6 +51,10 @@ func (ms *MaintenanceScanner) ScanForMaintenanceTasks() ([]*TaskDetectionResult,
|
||||
}
|
||||
}
|
||||
|
||||
// Refresh the master's default replication so detectors can use it as the
|
||||
// replica-placement fallback (matches the shell ec.balance default).
|
||||
ms.integration.SetDefaultReplicaPlacement(ms.getDefaultReplicaPlacement())
|
||||
|
||||
// Use task detection system with complete cluster information
|
||||
results, err := ms.integration.ScanWithTaskDetectors(taskMetrics)
|
||||
if err != nil {
|
||||
@@ -67,6 +71,26 @@ func (ms *MaintenanceScanner) ScanForMaintenanceTasks() ([]*TaskDetectionResult,
|
||||
return []*TaskDetectionResult{}, nil
|
||||
}
|
||||
|
||||
// getDefaultReplicaPlacement reads the master's configured default replication,
|
||||
// used by detectors as the replica-placement fallback. Returns "" on error so
|
||||
// detectors fall back to even spread rather than failing the scan.
|
||||
func (ms *MaintenanceScanner) getDefaultReplicaPlacement() string {
|
||||
var replicaPlacement string
|
||||
err := ms.adminClient.WithMasterClient(func(client master_pb.SeaweedClient) error {
|
||||
resp, err := client.GetMasterConfiguration(context.Background(), &master_pb.GetMasterConfigurationRequest{})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
replicaPlacement = resp.DefaultReplication
|
||||
return nil
|
||||
})
|
||||
if err != nil {
|
||||
glog.V(1).Infof("could not fetch master default replication: %v", err)
|
||||
return ""
|
||||
}
|
||||
return replicaPlacement
|
||||
}
|
||||
|
||||
// getVolumeHealthMetrics collects health information for all volumes.
|
||||
// Returns metrics in task-system format directly (no intermediate copy) and
|
||||
// the topology info for updating the active topology.
|
||||
|
||||
@@ -113,7 +113,7 @@ func cloneClusterContext(in *plugin_pb.ClusterContext) *plugin_pb.ClusterContext
|
||||
}
|
||||
out := &plugin_pb.ClusterContext{
|
||||
MasterGrpcAddresses: in.MasterGrpcAddresses,
|
||||
FilerGrpcAddresses: in.FilerGrpcAddresses,
|
||||
FilerAddresses: in.FilerAddresses,
|
||||
VolumeGrpcAddresses: in.VolumeGrpcAddresses,
|
||||
S3GrpcAddresses: in.S3GrpcAddresses,
|
||||
}
|
||||
|
||||
@@ -54,6 +54,70 @@ func (at *ActiveTopology) GetEffectiveAvailableCapacityDetailed(nodeID string, d
|
||||
return at.getEffectiveAvailableCapacityUnsafe(disk)
|
||||
}
|
||||
|
||||
// GetEffectiveAvailableEcShardSlots returns a disk's free EC shard slots,
|
||||
// accounting for in-flight task reservations at shard granularity. Unlike the
|
||||
// volume-slot views (GetDisksWithEffectiveCapacity / GetEffectiveAvailableCapacity),
|
||||
// this does not truncate sub-volume shard reservations: it subtracts the full
|
||||
// reservation impact (volume slots converted to shard slots, plus the raw shard
|
||||
// slots) so a reservation that is not a whole multiple of ShardsPerVolumeSlot is
|
||||
// not lost. It does NOT subtract the EC shards already persisted on the disk;
|
||||
// callers that track those (from EcShardInfos) subtract them separately.
|
||||
//
|
||||
// shardsPerVolume is the number of EC shards of the target collection that fit in
|
||||
// one volume slot (i.e. its data-shard count): a 4+2 volume's shards are ~1/4 of a
|
||||
// volume each, so one volume slot holds 4 of them, not the default
|
||||
// ShardsPerVolumeSlot. Pass <= 0 to use the default. Using the target ratio keeps
|
||||
// Place from over-filling a disk for low-data-shard layouts.
|
||||
func (at *ActiveTopology) GetEffectiveAvailableEcShardSlots(nodeID string, diskID uint32, shardsPerVolume int) int {
|
||||
if shardsPerVolume <= 0 {
|
||||
shardsPerVolume = ShardsPerVolumeSlot
|
||||
}
|
||||
|
||||
at.mutex.RLock()
|
||||
defer at.mutex.RUnlock()
|
||||
|
||||
diskKey := fmt.Sprintf("%s:%d", nodeID, diskID)
|
||||
disk, exists := at.disks[diskKey]
|
||||
if !exists || disk.DiskInfo == nil || disk.DiskInfo.DiskInfo == nil {
|
||||
return 0
|
||||
}
|
||||
|
||||
info := disk.DiskInfo.DiskInfo
|
||||
base := info.MaxVolumeCount - info.VolumeCount
|
||||
if base <= 0 && info.MaxVolumeCount == 0 && info.VolumeCount == 0 &&
|
||||
len(info.VolumeInfos) == 0 && len(info.EcShardInfos) == 0 {
|
||||
// Freshly started empty servers can report max=0 before publishing concrete
|
||||
// limits; keep one provisional slot so EC placement still sees the disk,
|
||||
// mirroring getEffectiveAvailableCapacityUnsafe.
|
||||
base = 1
|
||||
}
|
||||
if base < 0 {
|
||||
base = 0
|
||||
}
|
||||
// calculateTaskStorageImpact reports consumption as positive, so subtract it.
|
||||
// Volume-slot reservations scale by the target ratio; the sub-volume shard-slot
|
||||
// remainder is in default units and subtracted as-is (a small approximation).
|
||||
impact := at.getEffectiveCapacityUnsafe(disk)
|
||||
// impact.ShardSlots is recorded in default ShardsPerVolumeSlot units; convert it
|
||||
// to the target ratio's shard slots before subtracting (identity when
|
||||
// shardsPerVolume == ShardsPerVolumeSlot). Round a positive reservation up so a
|
||||
// sub-slot reservation (e.g. 1 default slot against a 4-shard target) is not
|
||||
// truncated to zero and wrongly counted as free.
|
||||
scaledShardImpact := int64(impact.ShardSlots) * int64(shardsPerVolume)
|
||||
if scaledShardImpact > 0 {
|
||||
scaledShardImpact = (scaledShardImpact + int64(ShardsPerVolumeSlot) - 1) / int64(ShardsPerVolumeSlot)
|
||||
} else {
|
||||
scaledShardImpact /= int64(ShardsPerVolumeSlot)
|
||||
}
|
||||
free := base*int64(shardsPerVolume) -
|
||||
int64(impact.VolumeSlots)*int64(shardsPerVolume) -
|
||||
scaledShardImpact
|
||||
if free < 0 {
|
||||
free = 0
|
||||
}
|
||||
return int(free)
|
||||
}
|
||||
|
||||
// GetEffectiveCapacityImpact returns the StorageSlotChange impact for a disk
|
||||
// This shows the net impact from all pending and assigned tasks
|
||||
func (at *ActiveTopology) GetEffectiveCapacityImpact(nodeID string, diskID uint32) StorageSlotChange {
|
||||
|
||||
@@ -0,0 +1,54 @@
|
||||
package topology
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
|
||||
)
|
||||
|
||||
// TestCountTopologyResources_multiDiskPerNode covers the case where the master
|
||||
// keys DiskInfos by disk type, so several same-type physical disks on a node
|
||||
// collapse into a single DiskInfo entry. Counting len(DiskInfos) under-reports
|
||||
// the physical disk count and disagrees with the per-disk activeDisk map that
|
||||
// the rest of the admin topology builds via SplitByPhysicalDisk.
|
||||
func TestCountTopologyResources_multiDiskPerNode(t *testing.T) {
|
||||
makeNode := func(id string) *master_pb.DataNodeInfo {
|
||||
var ecShardInfos []*master_pb.VolumeEcShardInformationMessage
|
||||
for diskId := uint32(0); diskId < 6; diskId++ {
|
||||
ecShardInfos = append(ecShardInfos, &master_pb.VolumeEcShardInformationMessage{
|
||||
Id: diskId + 1,
|
||||
DiskId: diskId,
|
||||
EcIndexBits: 1,
|
||||
})
|
||||
}
|
||||
return &master_pb.DataNodeInfo{
|
||||
Id: id,
|
||||
DiskInfos: map[string]*master_pb.DiskInfo{
|
||||
"": {Type: "", MaxVolumeCount: 60, EcShardInfos: ecShardInfos},
|
||||
},
|
||||
}
|
||||
}
|
||||
topo := &master_pb.TopologyInfo{
|
||||
Id: "multi_disk_topo",
|
||||
DataCenterInfos: []*master_pb.DataCenterInfo{{
|
||||
Id: "dc1",
|
||||
RackInfos: []*master_pb.RackInfo{{
|
||||
Id: "rack1",
|
||||
DataNodeInfos: []*master_pb.DataNodeInfo{
|
||||
makeNode("node1"), makeNode("node2"), makeNode("node3"),
|
||||
},
|
||||
}},
|
||||
}},
|
||||
}
|
||||
|
||||
dcCount, nodeCount, diskCount := CountTopologyResources(topo)
|
||||
if dcCount != 1 {
|
||||
t.Errorf("dcCount = %d, want 1", dcCount)
|
||||
}
|
||||
if nodeCount != 3 {
|
||||
t.Errorf("nodeCount = %d, want 3", nodeCount)
|
||||
}
|
||||
if diskCount != 18 {
|
||||
t.Errorf("diskCount = %d, want 18 (6 physical disks x 3 nodes)", diskCount)
|
||||
}
|
||||
}
|
||||
@@ -19,7 +19,12 @@ func CountTopologyResources(topologyInfo *master_pb.TopologyInfo) (dcCount, node
|
||||
for _, rack := range dc.RackInfos {
|
||||
nodeCount += len(rack.DataNodeInfos)
|
||||
for _, node := range rack.DataNodeInfos {
|
||||
diskCount += len(node.DiskInfos)
|
||||
// DiskInfos is keyed by disk type, so same-type physical disks
|
||||
// collapse into one entry. Count physical disks so the number
|
||||
// matches the per-disk activeDisk map.
|
||||
for _, diskInfo := range node.DiskInfos {
|
||||
diskCount += len(diskInfo.SplitByPhysicalDisk())
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+103
-21
@@ -4,6 +4,7 @@ import (
|
||||
"context"
|
||||
"fmt"
|
||||
"strings"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
|
||||
@@ -11,7 +12,6 @@ import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
"google.golang.org/grpc"
|
||||
)
|
||||
|
||||
@@ -20,6 +20,14 @@ type LockClient struct {
|
||||
maxLockDuration time.Duration
|
||||
sleepDuration time.Duration
|
||||
seedFiler pb.ServerAddress
|
||||
|
||||
// ring is an optional client-side view of the filer lock hash ring. When
|
||||
// populated, a new lock starts at the key's primary filer instead of the
|
||||
// seed filer, avoiding the seed->primary forward hop. A stale view stays
|
||||
// correct: the filer forwards to the real primary as a fallback.
|
||||
ringMu sync.RWMutex
|
||||
ring *lock_manager.HashRing
|
||||
ringVersion int64
|
||||
}
|
||||
|
||||
func NewLockClient(grpcDialOption grpc.DialOption, seedFiler pb.ServerAddress) *LockClient {
|
||||
@@ -31,12 +39,56 @@ func NewLockClient(grpcDialOption grpc.DialOption, seedFiler pb.ServerAddress) *
|
||||
}
|
||||
}
|
||||
|
||||
// SetRing mirrors the master's LockRingUpdate so the client computes the same
|
||||
// primary the filers do. A non-zero version at or below the current one is
|
||||
// ignored once a ring exists, dropping reordered and redundant broadcasts;
|
||||
// version 0 always applies (bootstrap).
|
||||
func (lc *LockClient) SetRing(servers []pb.ServerAddress, version int64) {
|
||||
lc.ringMu.Lock()
|
||||
defer lc.ringMu.Unlock()
|
||||
if version != 0 && version <= lc.ringVersion && lc.ring != nil {
|
||||
return
|
||||
}
|
||||
lc.ringVersion = version
|
||||
if lc.ring == nil {
|
||||
lc.ring = lock_manager.NewHashRing(lock_manager.DefaultVnodeCount)
|
||||
}
|
||||
lc.ring.SetServers(servers)
|
||||
}
|
||||
|
||||
// hostForKey returns the filer that should own key per the current ring view,
|
||||
// falling back to the seed filer when no view has been received yet.
|
||||
func (lc *LockClient) hostForKey(key string) pb.ServerAddress {
|
||||
lc.ringMu.RLock()
|
||||
defer lc.ringMu.RUnlock()
|
||||
if lc.ring == nil {
|
||||
return lc.seedFiler
|
||||
}
|
||||
if primary := lc.ring.GetPrimary(key); primary != "" {
|
||||
return primary
|
||||
}
|
||||
return lc.seedFiler
|
||||
}
|
||||
|
||||
// PrimaryForKey returns the ring owner for key, or "" before any ring arrives.
|
||||
// Unlike hostForKey it does not fall back to the seed, so a route-by-key caller
|
||||
// stays on the distributed lock until the ring is known.
|
||||
func (lc *LockClient) PrimaryForKey(key string) pb.ServerAddress {
|
||||
lc.ringMu.RLock()
|
||||
defer lc.ringMu.RUnlock()
|
||||
if lc.ring == nil {
|
||||
return ""
|
||||
}
|
||||
return lc.ring.GetPrimary(key)
|
||||
}
|
||||
|
||||
type LiveLock struct {
|
||||
key string
|
||||
renewToken string
|
||||
expireAtNs int64
|
||||
hostFiler pb.ServerAddress
|
||||
cancelCh chan struct{}
|
||||
renewalDone chan struct{} // closed when the renewal goroutine exits; nil if there is none
|
||||
grpcDialOption grpc.DialOption
|
||||
isLocked int32 // 0 = unlocked, 1 = locked; use atomic operations
|
||||
self string
|
||||
@@ -51,7 +103,7 @@ type LiveLock struct {
|
||||
func (lc *LockClient) NewShortLivedLock(key string, owner string) (lock *LiveLock) {
|
||||
lock = &LiveLock{
|
||||
key: key,
|
||||
hostFiler: lc.seedFiler,
|
||||
hostFiler: lc.hostForKey(key),
|
||||
cancelCh: make(chan struct{}),
|
||||
expireAtNs: time.Now().Add(5 * time.Second).UnixNano(),
|
||||
grpcDialOption: lc.grpcDialOption,
|
||||
@@ -72,7 +124,7 @@ func (lc *LockClient) NewBlockingLongLivedLock(key, owner string, lockTTL time.D
|
||||
}
|
||||
lock := &LiveLock{
|
||||
key: key,
|
||||
hostFiler: lc.seedFiler,
|
||||
hostFiler: lc.hostForKey(key),
|
||||
cancelCh: make(chan struct{}),
|
||||
expireAtNs: time.Now().Add(lockTTL).UnixNano(),
|
||||
grpcDialOption: lc.grpcDialOption,
|
||||
@@ -83,7 +135,9 @@ func (lc *LockClient) NewBlockingLongLivedLock(key, owner string, lockTTL time.D
|
||||
// Block until acquired
|
||||
lock.retryUntilLocked(lockTTL)
|
||||
// Start renewal goroutine using a ticker for interruptible sleep
|
||||
lock.renewalDone = make(chan struct{})
|
||||
go func() {
|
||||
defer close(lock.renewalDone)
|
||||
renewInterval := lockTTL / 2
|
||||
ticker := time.NewTicker(renewInterval)
|
||||
defer ticker.Stop()
|
||||
@@ -108,7 +162,7 @@ func (lc *LockClient) NewBlockingLongLivedLock(key, owner string, lockTTL time.D
|
||||
func (lc *LockClient) StartLongLivedLock(key string, owner string, onLockOwnerChange func(newLockOwner string), lockTTL time.Duration) (lock *LiveLock) {
|
||||
lock = &LiveLock{
|
||||
key: key,
|
||||
hostFiler: lc.seedFiler,
|
||||
hostFiler: lc.hostForKey(key),
|
||||
cancelCh: make(chan struct{}),
|
||||
expireAtNs: time.Now().Add(lockTTL).UnixNano(),
|
||||
grpcDialOption: lc.grpcDialOption,
|
||||
@@ -119,7 +173,9 @@ func (lc *LockClient) StartLongLivedLock(key string, owner string, onLockOwnerCh
|
||||
if lock.lockTTL == 0 {
|
||||
lock.lockTTL = lock_manager.LiveLockTTL
|
||||
}
|
||||
lock.renewalDone = make(chan struct{})
|
||||
go func() {
|
||||
defer close(lock.renewalDone)
|
||||
renewInterval := lock.lockTTL / 2
|
||||
isLocked := false
|
||||
lockOwner := ""
|
||||
@@ -149,30 +205,39 @@ func (lc *LockClient) StartLongLivedLock(key string, owner string, onLockOwnerCh
|
||||
onLockOwnerChange(lock.LockOwner())
|
||||
lockOwner = lock.LockOwner()
|
||||
}
|
||||
// Sleep until the next attempt, but wake immediately on Stop() so
|
||||
// the goroutine exits and closes renewalDone before Stop()'s bounded
|
||||
// wait elapses. An uninterruptible sleep here (up to 5*renewInterval
|
||||
// when unlocked) can outlast that wait and break the shutdown
|
||||
// synchronization.
|
||||
sleepFor := renewInterval
|
||||
if !isLocked {
|
||||
sleepFor = 5 * renewInterval
|
||||
}
|
||||
timer := time.NewTimer(sleepFor)
|
||||
select {
|
||||
case <-lock.cancelCh:
|
||||
timer.Stop()
|
||||
return
|
||||
default:
|
||||
if isLocked {
|
||||
time.Sleep(renewInterval)
|
||||
} else {
|
||||
time.Sleep(5 * renewInterval)
|
||||
}
|
||||
case <-timer.C:
|
||||
}
|
||||
}
|
||||
}()
|
||||
return
|
||||
}
|
||||
|
||||
// retryUntilLocked blocks until the lock is acquired, polling at the steady
|
||||
// short cadence that AttemptToLock already enforces on contention (~1s). It
|
||||
// deliberately avoids util.RetryUntil's exponential backoff (which grows to
|
||||
// several seconds): when a holder on another mount releases the lock, the
|
||||
// waiter must pick it up promptly, otherwise cross-mount write handoff stalls
|
||||
// long enough to time out clients.
|
||||
func (lock *LiveLock) retryUntilLocked(lockDuration time.Duration) {
|
||||
util.RetryUntil("create lock:"+lock.key, func() error {
|
||||
return lock.AttemptToLock(lockDuration)
|
||||
}, func(err error) (shouldContinue bool) {
|
||||
if err != nil {
|
||||
glog.Warningf("create lock %s: %s", lock.key, err)
|
||||
for lock.renewToken == "" {
|
||||
if err := lock.AttemptToLock(lockDuration); err != nil {
|
||||
glog.V(1).Infof("create lock %s: %v", lock.key, err)
|
||||
}
|
||||
return lock.renewToken == ""
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func (lock *LiveLock) AttemptToLock(lockDuration time.Duration) error {
|
||||
@@ -226,10 +291,27 @@ func (lock *LiveLock) Stop() error {
|
||||
close(lock.cancelCh)
|
||||
}
|
||||
|
||||
// Wait a brief moment for the goroutine to see the closed channel
|
||||
// This reduces the race condition window where the goroutine might
|
||||
// attempt one more lock operation after we've released the lock
|
||||
time.Sleep(10 * time.Millisecond)
|
||||
// Wait for the renewal goroutine to fully exit before unlocking. A renewal
|
||||
// in flight when we close cancelCh rotates renewToken on the server; if we
|
||||
// then unlock with the token we read here, the unlock fails with a token
|
||||
// mismatch and the lock lingers until its TTL expires — blocking other
|
||||
// mounts waiting on the same file. Waiting for the goroutine to return also
|
||||
// makes the renewToken read below race-free (channel close = happens-before).
|
||||
if lock.renewalDone != nil {
|
||||
select {
|
||||
case <-lock.renewalDone:
|
||||
case <-time.After(lock.lockTTL + 2*time.Second):
|
||||
// The renewal goroutine is wedged, almost certainly in a stuck
|
||||
// renewal RPC. Do not unlock here: the renewToken may be rotated
|
||||
// when that RPC finally returns, so an unlock sent now could race
|
||||
// it, be rejected on a stale token, and leave the lock lingering
|
||||
// anyway. cancelCh is closed, so the goroutine stops renewing once
|
||||
// its in-flight call returns and the lock then expires within its
|
||||
// TTL on its own.
|
||||
glog.Warningf("lock %s: renewal goroutine still running at shutdown; letting lock expire via TTL", lock.key)
|
||||
return nil
|
||||
}
|
||||
}
|
||||
|
||||
// Also release the lock if held
|
||||
// Note: We intentionally don't clear renewToken here because
|
||||
|
||||
@@ -0,0 +1,92 @@
|
||||
package cluster
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/cluster/lock_manager"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
)
|
||||
|
||||
// The gateway must resolve a lock key to the same primary the filers do,
|
||||
// otherwise it dials the wrong filer and the lock still gets forwarded. Both
|
||||
// sides use the same HashRing over the same server set, so for every key the
|
||||
// client's hostForKey must equal the filer ring's GetPrimary.
|
||||
func TestLockClientHostMatchesFilerRing(t *testing.T) {
|
||||
servers := []pb.ServerAddress{
|
||||
"filer-a:8888", "filer-b:8888", "filer-c:8888", "filer-d:8888",
|
||||
}
|
||||
|
||||
filerRing := lock_manager.NewHashRing(lock_manager.DefaultVnodeCount)
|
||||
filerRing.SetServers(servers)
|
||||
|
||||
lc := NewLockClient(nil, "seed:8888")
|
||||
lc.SetRing(servers, 1)
|
||||
|
||||
for _, key := range []string{
|
||||
"s3.object.write:/buckets/b/obj-0",
|
||||
"s3.object.write:/buckets/b/obj-1",
|
||||
"s3.object.write:/buckets/b/obj-2",
|
||||
"s3.object.write:/buckets/gosbench-0/w0obj-kilo-0877",
|
||||
"some/other/key",
|
||||
} {
|
||||
if got, want := lc.hostForKey(key), filerRing.GetPrimary(key); got != want {
|
||||
t.Errorf("key %q: client host %q != filer primary %q", key, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Without a ring view, the client falls back to the seed filer (which the filer
|
||||
// forwards from), preserving the pre-optimization behavior.
|
||||
func TestLockClientHostFallsBackToSeed(t *testing.T) {
|
||||
lc := NewLockClient(nil, "seed:8888")
|
||||
if got := lc.hostForKey("any-key"); got != "seed:8888" {
|
||||
t.Errorf("expected seed fallback, got %q", got)
|
||||
}
|
||||
|
||||
// An empty ring (no members yet) also falls back to the seed.
|
||||
lc.SetRing(nil, 1)
|
||||
if got := lc.hostForKey("any-key"); got != "seed:8888" {
|
||||
t.Errorf("expected seed fallback on empty ring, got %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
// A stale (older-version) update must not regress a newer ring view, while
|
||||
// version 0 always applies as a bootstrap.
|
||||
func TestLockClientSetRingVersionGuard(t *testing.T) {
|
||||
lc := NewLockClient(nil, "seed:8888")
|
||||
|
||||
newer := []pb.ServerAddress{"filer-a:8888", "filer-b:8888"}
|
||||
lc.SetRing(newer, 10)
|
||||
primaryAt10 := lc.hostForKey("k")
|
||||
|
||||
// Older version is ignored.
|
||||
lc.SetRing([]pb.ServerAddress{"filer-z:8888"}, 5)
|
||||
if got := lc.hostForKey("k"); got != primaryAt10 {
|
||||
t.Errorf("stale update applied: host changed to %q", got)
|
||||
}
|
||||
|
||||
// version 0 is always accepted.
|
||||
lc.SetRing([]pb.ServerAddress{"filer-z:8888"}, 0)
|
||||
if got := lc.hostForKey("k"); got != "filer-z:8888" {
|
||||
t.Errorf("bootstrap update not applied, got %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
// PrimaryForKey returns "" before any ring is received (so a route-by-key
|
||||
// caller falls back to the distributed lock) and the ring owner afterwards,
|
||||
// unlike hostForKey which falls back to the seed.
|
||||
func TestLockClientPrimaryForKey(t *testing.T) {
|
||||
lc := NewLockClient(nil, "seed:8888")
|
||||
if got := lc.PrimaryForKey("k"); got != "" {
|
||||
t.Errorf("expected empty before ring, got %q", got)
|
||||
}
|
||||
|
||||
lc.SetRing([]pb.ServerAddress{"filer-a:8888", "filer-b:8888"}, 1)
|
||||
got := lc.PrimaryForKey("k")
|
||||
if got == "" {
|
||||
t.Fatal("expected an owner after ring set")
|
||||
}
|
||||
if got != lc.hostForKey("k") {
|
||||
t.Errorf("PrimaryForKey %q disagrees with hostForKey %q", got, lc.hostForKey("k"))
|
||||
}
|
||||
}
|
||||
@@ -145,7 +145,15 @@ func (hr *HashRing) rebuildRing() {
|
||||
hr.vnodeToServer = make(map[uint32]pb.ServerAddress, len(hr.servers)*hr.vnodeCount)
|
||||
hr.sortedHashes = make([]uint32, 0, len(hr.servers)*hr.vnodeCount)
|
||||
|
||||
// Sort so a vnode-hash collision resolves to the same server on every node;
|
||||
// map iteration order alone is randomized per process.
|
||||
servers := make([]pb.ServerAddress, 0, len(hr.servers))
|
||||
for server := range hr.servers {
|
||||
servers = append(servers, server)
|
||||
}
|
||||
sort.Slice(servers, func(i, j int) bool { return servers[i] < servers[j] })
|
||||
|
||||
for _, server := range servers {
|
||||
for i := 0; i < hr.vnodeCount; i++ {
|
||||
vnodeKey := vnodeKeyFor(server, i)
|
||||
hash := hashKey(vnodeKey)
|
||||
|
||||
@@ -171,3 +171,38 @@ func TestHashRing_GetPrimary(t *testing.T) {
|
||||
primary, _ := hr.GetPrimaryAndBackup("mykey")
|
||||
assert.Equal(t, primary, hr.GetPrimary("mykey"))
|
||||
}
|
||||
|
||||
// The ring must be identical on every node holding the same server set,
|
||||
// regardless of the order servers were added or supplied. Build rings several
|
||||
// ways and assert they agree on the primary for a wide range of keys.
|
||||
func TestHashRing_OrderIndependent(t *testing.T) {
|
||||
servers := []pb.ServerAddress{
|
||||
"filer-a:8888", "filer-b:8888", "filer-c:8888", "filer-d:8888", "filer-e:8888",
|
||||
}
|
||||
|
||||
bySet := NewHashRing(50)
|
||||
bySet.SetServers(servers)
|
||||
|
||||
byReverse := NewHashRing(50)
|
||||
rev := append([]pb.ServerAddress(nil), servers...)
|
||||
for i, j := 0, len(rev)-1; i < j; i, j = i+1, j-1 {
|
||||
rev[i], rev[j] = rev[j], rev[i]
|
||||
}
|
||||
byReverse.SetServers(rev)
|
||||
|
||||
byAdd := NewHashRing(50)
|
||||
for _, s := range []pb.ServerAddress{"filer-c:8888", "filer-e:8888", "filer-a:8888", "filer-d:8888", "filer-b:8888"} {
|
||||
byAdd.AddServer(s)
|
||||
}
|
||||
|
||||
for i := 0; i < 5000; i++ {
|
||||
key := fmt.Sprintf("s3.object.write:/buckets/b/obj-%d", i)
|
||||
p := bySet.GetPrimary(key)
|
||||
if got := byReverse.GetPrimary(key); got != p {
|
||||
t.Fatalf("reverse-order ring disagrees on %q: %s vs %s", key, got, p)
|
||||
}
|
||||
if got := byAdd.GetPrimary(key); got != p {
|
||||
t.Fatalf("add-order ring disagrees on %q: %s vs %s", key, got, p)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -12,6 +12,7 @@ import (
|
||||
type LockRingSnapshot struct {
|
||||
servers []pb.ServerAddress
|
||||
ts time.Time
|
||||
ring *HashRing // prebuilt ring for servers, so PriorOwner need not rebuild per call
|
||||
}
|
||||
|
||||
type LockRing struct {
|
||||
@@ -58,10 +59,12 @@ func (r *LockRing) SetSnapshot(servers []pb.ServerAddress, version int64) bool {
|
||||
// are always consistent — prevents a concurrent SetSnapshot from
|
||||
// seeing the new version but applying its servers to the old ring.
|
||||
r.Ring.SetServers(servers)
|
||||
// Append the snapshot under the same lock as the ring update so a concurrent
|
||||
// PriorOwner always sees snapshots[0] matching r.Ring (and snapshots[1] as the
|
||||
// true prior); otherwise it could pair a new ring with a stale prior snapshot.
|
||||
r.addOneSnapshotLocked(servers)
|
||||
r.Unlock()
|
||||
|
||||
r.addOneSnapshot(servers)
|
||||
|
||||
r.cleanupWg.Add(1)
|
||||
go func() {
|
||||
defer r.cleanupWg.Done()
|
||||
@@ -78,14 +81,16 @@ func (r *LockRing) Version() int64 {
|
||||
return r.version
|
||||
}
|
||||
|
||||
func (r *LockRing) addOneSnapshot(servers []pb.ServerAddress) {
|
||||
r.Lock()
|
||||
defer r.Unlock()
|
||||
|
||||
// addOneSnapshotLocked appends a new snapshot (newest at index 0). The caller
|
||||
// must hold r.Lock(), so the ring update and snapshot append are one atomic step.
|
||||
func (r *LockRing) addOneSnapshotLocked(servers []pb.ServerAddress) {
|
||||
ts := time.Now()
|
||||
ring := NewHashRing(DefaultVnodeCount)
|
||||
ring.SetServers(servers)
|
||||
t := &LockRingSnapshot{
|
||||
servers: servers,
|
||||
ts: ts,
|
||||
ring: ring,
|
||||
}
|
||||
r.snapshots = append(r.snapshots, t)
|
||||
for i := len(r.snapshots) - 2; i >= 0; i-- {
|
||||
@@ -148,6 +153,35 @@ func (r *LockRing) GetPrimary(key string) pb.ServerAddress {
|
||||
return r.Ring.GetPrimary(key)
|
||||
}
|
||||
|
||||
// PriorOwner returns the key's owner from the previous ring snapshot, but only
|
||||
// while the ring changed within the last snapshotInterval and that owner differs
|
||||
// from the current primary. This is the cooling-off window in which the previous
|
||||
// owner may still hold locks the new owner has not yet rebuilt — a caller can
|
||||
// consult it before granting so a fresh owner does not double-grant during a
|
||||
// rebalance. Returns "" outside the window or when ownership did not move. It
|
||||
// uses the snapshot's prebuilt ring, so it does not rebuild a hash ring per call.
|
||||
func (r *LockRing) PriorOwner(key string) pb.ServerAddress {
|
||||
r.RLock()
|
||||
defer r.RUnlock()
|
||||
if len(r.snapshots) < 2 {
|
||||
return ""
|
||||
}
|
||||
if time.Since(r.snapshots[0].ts) > r.snapshotInterval {
|
||||
return ""
|
||||
}
|
||||
current := r.Ring.GetPrimary(key)
|
||||
var prior pb.ServerAddress
|
||||
if pr := r.snapshots[1].ring; pr != nil {
|
||||
prior = pr.GetPrimary(key)
|
||||
} else {
|
||||
prior = hashKeyToServer(key, r.snapshots[1].servers)
|
||||
}
|
||||
if prior != "" && prior != current {
|
||||
return prior
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// hashKeyToServer uses a temporary consistent hash ring for the given server list.
|
||||
func hashKeyToServer(key string, servers []pb.ServerAddress) pb.ServerAddress {
|
||||
if len(servers) == 0 {
|
||||
|
||||
@@ -0,0 +1,63 @@
|
||||
package lock_manager
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
)
|
||||
|
||||
func TestLockRing_PriorOwner(t *testing.T) {
|
||||
r := NewLockRing(5 * time.Second)
|
||||
t.Cleanup(r.WaitForCleanup)
|
||||
|
||||
setA := []pb.ServerAddress{"s1:1", "s2:1", "s3:1"}
|
||||
r.SetSnapshot(setA, 1)
|
||||
|
||||
// Only one snapshot: nothing to fall back to.
|
||||
if got := r.PriorOwner("any"); got != "" {
|
||||
t.Fatalf("single snapshot should have no prior owner, got %q", got)
|
||||
}
|
||||
|
||||
// Add a server so some keys' ownership moves.
|
||||
setB := []pb.ServerAddress{"s1:1", "s2:1", "s3:1", "s4:1"}
|
||||
r.SetSnapshot(setB, 2)
|
||||
|
||||
var moved, stable string
|
||||
for i := 0; i < 2000 && (moved == "" || stable == ""); i++ {
|
||||
key := fmt.Sprintf("key-%d", i)
|
||||
if r.GetPrimary(key) != hashKeyToServer(key, setA) {
|
||||
if moved == "" {
|
||||
moved = key
|
||||
}
|
||||
} else if stable == "" {
|
||||
stable = key
|
||||
}
|
||||
}
|
||||
if moved == "" || stable == "" {
|
||||
t.Skip("could not find both a moved and a stable key")
|
||||
}
|
||||
|
||||
if got, want := r.PriorOwner(moved), hashKeyToServer(moved, setA); got != want {
|
||||
t.Fatalf("PriorOwner(moved)=%q, want %q", got, want)
|
||||
}
|
||||
if got := r.PriorOwner(stable); got != "" {
|
||||
t.Fatalf("unmoved key should have no prior owner, got %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestLockRing_PriorOwnerExpires(t *testing.T) {
|
||||
r := NewLockRing(20 * time.Millisecond)
|
||||
t.Cleanup(r.WaitForCleanup)
|
||||
r.SetSnapshot([]pb.ServerAddress{"s1:1", "s2:1", "s3:1"}, 1)
|
||||
r.SetSnapshot([]pb.ServerAddress{"s1:1", "s2:1", "s3:1", "s4:1"}, 2)
|
||||
|
||||
// Past the cooling interval, the prior owner is no longer offered.
|
||||
time.Sleep(40 * time.Millisecond)
|
||||
for i := 0; i < 2000; i++ {
|
||||
if got := r.PriorOwner(fmt.Sprintf("key-%d", i)); got != "" {
|
||||
t.Fatalf("prior owner should expire after the cooling interval, got %q", got)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -30,6 +30,7 @@ import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/security"
|
||||
stats_collect "github.com/seaweedfs/seaweedfs/weed/stats"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util/grace"
|
||||
)
|
||||
@@ -50,6 +51,8 @@ type AdminOptions struct {
|
||||
dataDir *string
|
||||
icebergPort *int
|
||||
urlPrefix *string
|
||||
metricsHttpPort *int
|
||||
metricsHttpIp *string
|
||||
debug *bool
|
||||
debugPort *int
|
||||
cpuProfile *string
|
||||
@@ -70,6 +73,8 @@ func init() {
|
||||
a.readOnlyPassword = cmdAdmin.Flag.String("readOnlyPassword", "", "read-only user password (optional, for view-only access; requires adminPassword to be set)")
|
||||
a.icebergPort = cmdAdmin.Flag.Int("iceberg.port", 8181, "Iceberg REST Catalog port (0 to hide in UI)")
|
||||
a.urlPrefix = cmdAdmin.Flag.String("urlPrefix", "", "URL path prefix when running behind a reverse proxy under a subdirectory (e.g. /seaweedfs)")
|
||||
a.metricsHttpPort = cmdAdmin.Flag.Int("metricsPort", 0, "Prometheus metrics listen port")
|
||||
a.metricsHttpIp = cmdAdmin.Flag.String("metricsIp", "", "metrics listen ip. If empty, listens on all interfaces.")
|
||||
a.debug = cmdAdmin.Flag.Bool("debug", false, "serves runtime profiling data via pprof on the port specified by -debug.port")
|
||||
a.debugPort = cmdAdmin.Flag.Int("debug.port", 6060, "http port for debugging")
|
||||
a.cpuProfile = cmdAdmin.Flag.String("cpuprofile", "", "cpu profile output file")
|
||||
@@ -160,6 +165,12 @@ var cmdAdmin = &Command{
|
||||
weed admin -debug -debug.port=6060 -master="localhost:9333"
|
||||
weed admin -cpuprofile=cpu.prof -memprofile=mem.prof -master="localhost:9333"
|
||||
|
||||
Metrics:
|
||||
- Use -metricsPort to expose Prometheus metrics at http://<host>:<metricsPort>/metrics
|
||||
- Use -metricsIp to bind the metrics endpoint to a specific ip (default: all interfaces)
|
||||
- Metrics are disabled when -metricsPort is 0 (the default)
|
||||
- Example: weed admin -metricsPort=9327 -master="localhost:9333"
|
||||
|
||||
Configuration File:
|
||||
- The security.toml file is read from ".", "$HOME/.seaweedfs/",
|
||||
"/usr/local/etc/seaweedfs/", or "/etc/seaweedfs/", in that order
|
||||
@@ -257,6 +268,12 @@ func runAdmin(cmd *Command, args []string) bool {
|
||||
}
|
||||
fmt.Printf("Plugin: Enabled\n")
|
||||
|
||||
// Start Prometheus metrics endpoint if a port is configured
|
||||
if *a.metricsHttpPort > 0 {
|
||||
fmt.Printf("Metrics: http://%s/metrics\n", stats_collect.JoinHostPort(*a.metricsHttpIp, *a.metricsHttpPort))
|
||||
}
|
||||
go stats_collect.StartMetricsServer(*a.metricsHttpIp, *a.metricsHttpPort)
|
||||
|
||||
// Set up graceful shutdown
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
|
||||
@@ -439,6 +439,13 @@ func doSubscribeFilerMetaChanges(clientId int32, clientEpoch int32, sourceGrpcDi
|
||||
StartTsNs: sourceFilerOffsetTsNs,
|
||||
StopTsNs: 0,
|
||||
EventErrorType: pb.RetryForeverOnError,
|
||||
// While the source has only read activity it emits no metadata events, so
|
||||
// the watermark above never advances and sync_offset would look stuck.
|
||||
// The idle heartbeat moves the gauge to the source's current time once we
|
||||
// are caught up, so now-sync_offset reflects real lag and stays alertable.
|
||||
OnIdleHeartbeat: func(tsNs int64) {
|
||||
statsCollect.FilerSyncOffsetGauge.WithLabelValues(sourceFiler.String(), targetFiler.String(), clientName, sourcePath).Set(float64(tsNs))
|
||||
},
|
||||
}
|
||||
|
||||
return pb.FollowMetadata(sourceFiler, sourceGrpcDialOption, metadataFollowOption, processEventFnWithOffset)
|
||||
|
||||
@@ -2,6 +2,7 @@ package command
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"io"
|
||||
"io/fs"
|
||||
"os"
|
||||
"path"
|
||||
@@ -9,12 +10,15 @@ import (
|
||||
"strings"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/volume_server_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/backend"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/erasure_coding"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/needle"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/needle_map"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/super_block"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/types"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/volume_info"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
)
|
||||
|
||||
@@ -27,6 +31,7 @@ var cmdFix = &Command{
|
||||
Short: "run weed tool fix on files or whole folders to recreate index file(s) if corrupted",
|
||||
Long: `Fix runs the SeaweedFS fix command on local dat files ( or remote files) or whole folders to re-create the index .idx file. If fixing remote files, you need to synchronize master.toml to the same directory on the current node as on the master node.
|
||||
You Need to stop the volume server when running this command.
|
||||
Use -ecx to rebuild a lost EC index (.ecx) — and the .vif when missing — from the local .ec## shards.
|
||||
`,
|
||||
}
|
||||
|
||||
@@ -36,6 +41,9 @@ var (
|
||||
fixIncludeDeleted = cmdFix.Flag.Bool("includeDeleted", true, "include deleted entries in the index file")
|
||||
fixIgnoreError = cmdFix.Flag.Bool("ignoreError", false, "an optional, if true will be processed despite errors")
|
||||
fixRemoteFile = cmdFix.Flag.Bool("remoteFile", false, "an optional, if true will not try to load the local .dat file, but only the remote file")
|
||||
fixGenerateEcx = cmdFix.Flag.Bool("ecx", false, "regenerate a lost EC index (.ecx) — and the .vif when missing — from the local .ec## shards (missing shards are reconstructed from parity when enough survive). Run with the volume server stopped.")
|
||||
fixEcDataShards = cmdFix.Flag.Int("ecDataShards", 0, "EC data shard count for -ecx (0 = read from .vif, otherwise default 10)")
|
||||
fixEcParityShards = cmdFix.Flag.Int("ecParityShards", 0, "EC parity shard count for -ecx (0 = read from .vif, infer from shard count, otherwise default 4)")
|
||||
)
|
||||
|
||||
type VolumeFileScanner4Fix struct {
|
||||
@@ -130,6 +138,45 @@ func runFix(cmd *Command, args []string) bool {
|
||||
}
|
||||
doFixOneVolume(basePath, baseFileName, collection, volumeId, *fixIncludeDeleted)
|
||||
}
|
||||
|
||||
if *fixGenerateEcx {
|
||||
if !fixEcxFromShardsInDir(basePath, files) {
|
||||
return false
|
||||
}
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// fixEcxFromShardsInDir finds EC volumes in files (identified by their .ec00
|
||||
// data shard) and regenerates the .ecx (and .vif when missing) for each,
|
||||
// honoring the -collection and -volumeId filters.
|
||||
func fixEcxFromShardsInDir(basePath string, files []fs.DirEntry) bool {
|
||||
const shard0Ext = ".ec00"
|
||||
for _, file := range files {
|
||||
if !strings.HasSuffix(file.Name(), shard0Ext) {
|
||||
continue
|
||||
}
|
||||
if *fixVolumeCollection != "" {
|
||||
if !strings.HasPrefix(file.Name(), *fixVolumeCollection+"_") {
|
||||
continue
|
||||
}
|
||||
}
|
||||
baseFileName := file.Name()[:len(file.Name())-len(shard0Ext)]
|
||||
collection, volumeIdStr := "", baseFileName
|
||||
if sepIndex := strings.LastIndex(baseFileName, "_"); sepIndex > 0 {
|
||||
collection = baseFileName[:sepIndex]
|
||||
volumeIdStr = baseFileName[sepIndex+1:]
|
||||
}
|
||||
volumeId, parseErr := strconv.ParseInt(volumeIdStr, 10, 64)
|
||||
if parseErr != nil {
|
||||
fmt.Printf("Failed to parse volume id from %s: %v\n", baseFileName, parseErr)
|
||||
return false
|
||||
}
|
||||
if *fixVolumeId != 0 && *fixVolumeId != volumeId {
|
||||
continue
|
||||
}
|
||||
doFixEcxFromShards(basePath, baseFileName, collection, volumeId)
|
||||
}
|
||||
return true
|
||||
}
|
||||
@@ -202,3 +249,252 @@ func doFixOneVolume(basepath string, baseFileName string, collection string, vol
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// doFixEcxFromShards rebuilds the sealed EC index (.ecx) for one EC volume
|
||||
// directly from its local shards when both the .ecx and the original .dat are
|
||||
// gone but the shards survive. When some data shards are missing but at least
|
||||
// dataShards shards survive in total, the missing shards are first reconstructed
|
||||
// from the survivors via Reed-Solomon. It then de-stripes the data shards into a
|
||||
// temporary .dat, scans the needles, and writes a fresh ascending-sorted .ecx
|
||||
// that matches what WriteSortedFileFromIdx emits at encode time (live entries
|
||||
// only). When the .vif is also missing it is regenerated from the inferred EC
|
||||
// ratio and the .dat size discovered during the scan.
|
||||
func doFixEcxFromShards(basePath, baseFileName, collection string, volumeId int64) {
|
||||
base := path.Join(basePath, baseFileName)
|
||||
|
||||
fail := func(err error) {
|
||||
if *fixIgnoreError {
|
||||
glog.Error(err)
|
||||
} else {
|
||||
glog.Fatal(err)
|
||||
}
|
||||
}
|
||||
|
||||
ecxName := base + ".ecx"
|
||||
if info, err := os.Stat(ecxName); err == nil && info.Size() > 0 {
|
||||
glog.Infof("volume %d: %s already exists (%d bytes), skipping; remove it first to force regeneration", volumeId, ecxName, info.Size())
|
||||
return
|
||||
}
|
||||
|
||||
// Discover which shards are present and their common size. Reed-Solomon
|
||||
// requires every shard to be the same size.
|
||||
present := make([]bool, erasure_coding.MaxShardCount)
|
||||
presentCount := 0
|
||||
maxPresentIdx := -1
|
||||
var shardSize int64
|
||||
for i := 0; i < erasure_coding.MaxShardCount; i++ {
|
||||
info, statErr := os.Stat(base + erasure_coding.ToExt(i))
|
||||
if statErr != nil || info.Size() == 0 {
|
||||
continue
|
||||
}
|
||||
if shardSize == 0 {
|
||||
shardSize = info.Size()
|
||||
} else if info.Size() != shardSize {
|
||||
fail(fmt.Errorf("volume %d: shard %s size %d does not match %d", volumeId, base+erasure_coding.ToExt(i), info.Size(), shardSize))
|
||||
return
|
||||
}
|
||||
present[i] = true
|
||||
presentCount++
|
||||
maxPresentIdx = i
|
||||
}
|
||||
if presentCount == 0 {
|
||||
fail(fmt.Errorf("volume %d: no EC shards found under %s", volumeId, base))
|
||||
return
|
||||
}
|
||||
|
||||
// Resolve the EC ratio and the original .dat size.
|
||||
// Priority: explicit flags > existing .vif > defaults (10+4).
|
||||
vifName := base + ".vif"
|
||||
vifExists := util.FileExists(vifName)
|
||||
dataShards := erasure_coding.DataShardsCount
|
||||
parityShards := erasure_coding.ParityShardsCount
|
||||
var datFileSize int64
|
||||
if vifExists {
|
||||
// MaybeLoadVolumeInfo returns a non-nil error when the .vif exists but
|
||||
// cannot be read or unmarshalled; fail loudly rather than silently
|
||||
// falling back to defaults (which would be wrong for a custom ratio).
|
||||
if vi, _, found, loadErr := volume_info.MaybeLoadVolumeInfo(vifName); loadErr != nil {
|
||||
fail(fmt.Errorf("volume %d: read %s: %w", volumeId, vifName, loadErr))
|
||||
return
|
||||
} else if found && vi != nil {
|
||||
if cfg := vi.GetEcShardConfig(); cfg != nil && cfg.GetDataShards() > 0 {
|
||||
dataShards = int(cfg.GetDataShards())
|
||||
parityShards = int(cfg.GetParityShards())
|
||||
}
|
||||
datFileSize = vi.GetDatFileSize()
|
||||
}
|
||||
}
|
||||
if *fixEcDataShards > 0 {
|
||||
dataShards = *fixEcDataShards
|
||||
}
|
||||
if *fixEcParityShards > 0 {
|
||||
parityShards = *fixEcParityShards
|
||||
}
|
||||
// Ensure the configured total covers every shard index actually present
|
||||
// (a custom-ratio volume with more than the default 14 shards and no .vif).
|
||||
// This never lowers parity below the default, so the common 10+4 case stays
|
||||
// correct for any subset of missing shards.
|
||||
if maxPresentIdx+1 > dataShards+parityShards {
|
||||
parityShards = maxPresentIdx + 1 - dataShards
|
||||
}
|
||||
if dataShards <= 0 || parityShards <= 0 || dataShards+parityShards > erasure_coding.MaxShardCount {
|
||||
fail(fmt.Errorf("volume %d: cannot determine EC ratio (data=%d parity=%d); set -ecDataShards/-ecParityShards", volumeId, dataShards, parityShards))
|
||||
return
|
||||
}
|
||||
|
||||
// Need at least dataShards shards (any data+parity mix) to recover anything.
|
||||
if presentCount < dataShards {
|
||||
fail(fmt.Errorf("volume %d: only %d shards present, need at least %d (data shards) to recover", volumeId, presentCount, dataShards))
|
||||
return
|
||||
}
|
||||
|
||||
// If any data shard is missing, reconstruct the missing shards from the
|
||||
// survivors via Reed-Solomon before de-striping. This writes the rebuilt
|
||||
// shard files back to disk, fully repairing the volume locally.
|
||||
dataComplete := true
|
||||
for i := 0; i < dataShards; i++ {
|
||||
if !present[i] {
|
||||
dataComplete = false
|
||||
break
|
||||
}
|
||||
}
|
||||
if !dataComplete {
|
||||
ctx := &erasure_coding.ECContext{DataShards: dataShards, ParityShards: parityShards}
|
||||
glog.Infof("volume %d: %d/%d shards present; reconstructing missing shards (%s) before index rebuild", volumeId, presentCount, dataShards+parityShards, ctx.String())
|
||||
if _, err := erasure_coding.RebuildEcFilesWithContext(base, ctx); err != nil {
|
||||
fail(fmt.Errorf("volume %d: reconstruct missing shards from %d survivors: %w", volumeId, presentCount, err))
|
||||
return
|
||||
}
|
||||
}
|
||||
|
||||
// Collect the data shards (now all present).
|
||||
shardFileNames := make([]string, dataShards)
|
||||
for i := 0; i < dataShards; i++ {
|
||||
shardPath := base + erasure_coding.ToExt(i)
|
||||
if !util.FileExists(shardPath) {
|
||||
fail(fmt.Errorf("volume %d: data shard %s still missing after reconstruction", volumeId, shardPath))
|
||||
return
|
||||
}
|
||||
shardFileNames[i] = shardPath
|
||||
}
|
||||
|
||||
// Without a recorded original size, reconstruct the fully padded layout; the
|
||||
// scan below detects the trailing zero padding and recovers the true size.
|
||||
reconstructSize := datFileSize
|
||||
if reconstructSize <= 0 {
|
||||
reconstructSize = int64(dataShards) * shardSize
|
||||
glog.V(0).Infof("volume %d: no .dat size in .vif; reconstructing padded .dat (%d bytes) from %d data shards", volumeId, reconstructSize, dataShards)
|
||||
}
|
||||
|
||||
// De-stripe the data shards into a temporary .dat next to the shards.
|
||||
tmpBase := base + ".ecxrecover"
|
||||
tmpDat := tmpBase + ".dat"
|
||||
if err := erasure_coding.WriteDatFile(tmpBase, reconstructSize, shardFileNames); err != nil {
|
||||
os.Remove(tmpDat)
|
||||
fail(fmt.Errorf("volume %d: reconstruct .dat from data shards: %w", volumeId, err))
|
||||
return
|
||||
}
|
||||
defer os.Remove(tmpDat)
|
||||
|
||||
realDatSize, version, err := writeEcxFromDat(tmpDat, ecxName)
|
||||
if err != nil {
|
||||
os.Remove(ecxName)
|
||||
fail(fmt.Errorf("volume %d: build .ecx from reconstructed .dat: %w", volumeId, err))
|
||||
return
|
||||
}
|
||||
glog.Infof("volume %d: wrote %s from %d data shards", volumeId, ecxName, dataShards)
|
||||
|
||||
// Regenerate the .vif when missing so the volume can mount and future
|
||||
// rebuilds know the EC ratio and original .dat size.
|
||||
if !vifExists {
|
||||
size := datFileSize
|
||||
if size <= 0 {
|
||||
size = realDatSize
|
||||
}
|
||||
volumeInfo := &volume_server_pb.VolumeInfo{
|
||||
Version: uint32(version),
|
||||
DatFileSize: size,
|
||||
EcShardConfig: &volume_server_pb.EcShardConfig{
|
||||
DataShards: uint32(dataShards),
|
||||
ParityShards: uint32(parityShards),
|
||||
},
|
||||
}
|
||||
if err := volume_info.SaveVolumeInfo(vifName, volumeInfo); err != nil {
|
||||
fail(fmt.Errorf("volume %d: write %s: %w", volumeId, vifName, err))
|
||||
return
|
||||
}
|
||||
glog.Infof("volume %d: wrote %s (version %d, datFileSize %d, ec %d+%d)", volumeId, vifName, version, size, dataShards, parityShards)
|
||||
}
|
||||
}
|
||||
|
||||
// writeEcxFromDat scans a (reconstructed) .dat and writes an ascending-sorted
|
||||
// .ecx containing only live needles — the same on-disk shape
|
||||
// WriteSortedFileFromIdx produces when an EC volume is first encoded. It returns
|
||||
// the physical .dat size (the offset where the EC zero padding begins) and the
|
||||
// volume version read from the superblock.
|
||||
func writeEcxFromDat(datPath, ecxPath string) (datFileSize int64, version needle.Version, err error) {
|
||||
f, err := os.OpenFile(datPath, os.O_RDONLY, 0644)
|
||||
if err != nil {
|
||||
return 0, 0, fmt.Errorf("open %s: %w", datPath, err)
|
||||
}
|
||||
datBackend := backend.NewDiskFile(f)
|
||||
defer datBackend.Close()
|
||||
|
||||
superBlock, err := super_block.ReadSuperBlock(datBackend)
|
||||
if err != nil {
|
||||
return 0, 0, fmt.Errorf("read superblock: %w", err)
|
||||
}
|
||||
version = superBlock.Version
|
||||
|
||||
fileSize, _, err := datBackend.GetStat()
|
||||
if err != nil {
|
||||
return 0, version, fmt.Errorf("stat %s: %w", datPath, err)
|
||||
}
|
||||
|
||||
nm := needle_map.NewMemDb()
|
||||
defer nm.Close()
|
||||
|
||||
offset := int64(superBlock.BlockSize())
|
||||
for offset < fileSize {
|
||||
n, _, rest, readErr := needle.ReadNeedleHeader(datBackend, version, offset)
|
||||
if readErr != nil {
|
||||
if readErr == io.EOF {
|
||||
break
|
||||
}
|
||||
return 0, version, fmt.Errorf("read needle header at offset %d: %w", offset, readErr)
|
||||
}
|
||||
// EC encoding zero-pads the tail of the last block row. An all-zero
|
||||
// header marks the start of that padding, i.e. the end of real needles.
|
||||
if n.Cookie == 0 && n.Id == 0 && n.Size == 0 {
|
||||
break
|
||||
}
|
||||
if n.Size.IsValid() {
|
||||
if pe := nm.Set(n.Id, types.ToOffset(offset), n.Size); pe != nil {
|
||||
return 0, version, fmt.Errorf("set needle %d: %w", n.Id, pe)
|
||||
}
|
||||
} else {
|
||||
// Deleted/invalid: drop it so the .ecx carries only live entries,
|
||||
// matching the encode-time WriteSortedFileFromIdx behavior.
|
||||
if pe := nm.Delete(n.Id); pe != nil {
|
||||
return 0, version, fmt.Errorf("delete needle %d: %w", n.Id, pe)
|
||||
}
|
||||
}
|
||||
offset += types.NeedleHeaderSize + rest
|
||||
}
|
||||
datFileSize = offset
|
||||
|
||||
ecxFile, err := os.OpenFile(ecxPath, os.O_TRUNC|os.O_CREATE|os.O_WRONLY, 0644)
|
||||
if err != nil {
|
||||
return 0, version, fmt.Errorf("open %s: %w", ecxPath, err)
|
||||
}
|
||||
defer ecxFile.Close()
|
||||
|
||||
if err := nm.AscendingVisit(func(value needle_map.NeedleValue) error {
|
||||
_, writeErr := ecxFile.Write(value.ToBytes())
|
||||
return writeErr
|
||||
}); err != nil {
|
||||
return 0, version, fmt.Errorf("write %s: %w", ecxPath, err)
|
||||
}
|
||||
|
||||
return datFileSize, version, nil
|
||||
}
|
||||
|
||||
@@ -0,0 +1,273 @@
|
||||
package command
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"math/rand"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/volume_server_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/backend"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/erasure_coding"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/needle"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/needle_map"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/super_block"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/types"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/volume_info"
|
||||
)
|
||||
|
||||
// buildAndEncodeTestEcVolume writes a small volume (.dat + .idx), EC-encodes it
|
||||
// into .ec00..ec13, and produces the canonical sorted .ecx. It returns the base
|
||||
// path, the canonical .ecx bytes, and the original .dat size. A couple of
|
||||
// needles are deleted so the .ecx must exclude them (live entries only).
|
||||
func buildAndEncodeTestEcVolume(t *testing.T, dir, baseName string) (base string, canonicalEcx []byte, origDatSize int64) {
|
||||
t.Helper()
|
||||
base = filepath.Join(dir, baseName)
|
||||
version := needle.GetCurrentVersion()
|
||||
|
||||
df, err := os.OpenFile(base+".dat", os.O_RDWR|os.O_CREATE|os.O_TRUNC, 0644)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
datBackend := backend.NewDiskFile(df)
|
||||
sb := super_block.SuperBlock{
|
||||
Version: version,
|
||||
ReplicaPlacement: &super_block.ReplicaPlacement{},
|
||||
Ttl: &needle.TTL{},
|
||||
}
|
||||
if _, err := datBackend.WriteAt(sb.Bytes(), 0); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
nm := needle_map.NewMemDb()
|
||||
for i := uint64(1); i <= 12; i++ {
|
||||
n := new(needle.Needle)
|
||||
n.Id = types.Uint64ToNeedleId(i)
|
||||
n.Data = make([]byte, 200+int(i))
|
||||
rand.Read(n.Data)
|
||||
n.Checksum = needle.NewCRC(n.Data)
|
||||
offset, _, _, err := n.Append(datBackend, version)
|
||||
if err != nil {
|
||||
t.Fatalf("append needle %d: %v", i, err)
|
||||
}
|
||||
// Store n.Size (the on-disk header size), exactly what the volume
|
||||
// server records in its .idx (volume_write.go: nm.Put(..., n.Size)).
|
||||
if err := nm.Set(n.Id, types.ToOffset(int64(offset)), n.Size); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
// Delete ids 3 and 8: append an empty needle (delete record) and drop them
|
||||
// from the index, exactly as the encode-time .ecx would reflect.
|
||||
for _, id := range []uint64{3, 8} {
|
||||
n := new(needle.Needle)
|
||||
n.Id = types.Uint64ToNeedleId(id)
|
||||
if _, _, _, err := n.Append(datBackend, version); err != nil {
|
||||
t.Fatalf("append delete record %d: %v", id, err)
|
||||
}
|
||||
if err := nm.Delete(types.Uint64ToNeedleId(id)); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
if err := datBackend.Sync(); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
datInfo, err := df.Stat()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
origDatSize = datInfo.Size()
|
||||
datBackend.Close()
|
||||
|
||||
idxFile, err := os.OpenFile(base+".idx", os.O_WRONLY|os.O_CREATE|os.O_TRUNC, 0644)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := nm.AscendingVisit(func(v needle_map.NeedleValue) error {
|
||||
_, e := idxFile.Write(v.ToBytes())
|
||||
return e
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
idxFile.Close()
|
||||
nm.Close()
|
||||
|
||||
if err := erasure_coding.WriteEcFiles(base); err != nil {
|
||||
t.Fatalf("WriteEcFiles: %v", err)
|
||||
}
|
||||
if err := erasure_coding.WriteSortedFileFromIdx(base, ".ecx"); err != nil {
|
||||
t.Fatalf("WriteSortedFileFromIdx: %v", err)
|
||||
}
|
||||
canonicalEcx, err = os.ReadFile(base + ".ecx")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(canonicalEcx) == 0 {
|
||||
t.Fatal("canonical .ecx is empty")
|
||||
}
|
||||
return base, canonicalEcx, origDatSize
|
||||
}
|
||||
|
||||
// TestFixEcxFromShards verifies the .ecx and .vif are rebuilt purely from the
|
||||
// data shards when every index/metadata file has been lost.
|
||||
func TestFixEcxFromShards(t *testing.T) {
|
||||
oldData, oldParity := *fixEcDataShards, *fixEcParityShards
|
||||
*fixEcDataShards, *fixEcParityShards = 0, 0
|
||||
*fixIgnoreError = true
|
||||
t.Cleanup(func() {
|
||||
*fixIgnoreError = false
|
||||
*fixEcDataShards, *fixEcParityShards = oldData, oldParity
|
||||
})
|
||||
|
||||
dir := t.TempDir()
|
||||
const volumeId = 7
|
||||
base, canonical, origDatSize := buildAndEncodeTestEcVolume(t, dir, "7")
|
||||
|
||||
// Disaster: keep only the shards.
|
||||
for _, ext := range []string{".ecx", ".ecj", ".idx", ".dat", ".vif"} {
|
||||
if err := os.Remove(base + ext); err != nil && !os.IsNotExist(err) {
|
||||
t.Fatalf("remove %s: %v", base+ext, err)
|
||||
}
|
||||
}
|
||||
|
||||
doFixEcxFromShards(dir, "7", "", volumeId)
|
||||
|
||||
recovered, err := os.ReadFile(base + ".ecx")
|
||||
if err != nil {
|
||||
t.Fatalf("recovered .ecx not written: %v", err)
|
||||
}
|
||||
if !bytes.Equal(canonical, recovered) {
|
||||
t.Fatalf(".ecx mismatch: canonical %d bytes, recovered %d bytes", len(canonical), len(recovered))
|
||||
}
|
||||
|
||||
// The reconstructed temporary .dat must not be left behind.
|
||||
if _, err := os.Stat(base + ".ecxrecover.dat"); !os.IsNotExist(err) {
|
||||
t.Fatalf("temporary reconstructed .dat was not cleaned up")
|
||||
}
|
||||
|
||||
// .vif must be regenerated with the default ratio and the original .dat size.
|
||||
vi, _, found, err := volume_info.MaybeLoadVolumeInfo(base + ".vif")
|
||||
if err != nil || !found {
|
||||
t.Fatalf(".vif not regenerated: found=%v err=%v", found, err)
|
||||
}
|
||||
if got := int(vi.GetEcShardConfig().GetDataShards()); got != erasure_coding.DataShardsCount {
|
||||
t.Fatalf("data shards = %d, want %d", got, erasure_coding.DataShardsCount)
|
||||
}
|
||||
if got := int(vi.GetEcShardConfig().GetParityShards()); got != erasure_coding.ParityShardsCount {
|
||||
t.Fatalf("parity shards = %d, want %d", got, erasure_coding.ParityShardsCount)
|
||||
}
|
||||
if vi.GetDatFileSize() != origDatSize {
|
||||
t.Fatalf("dat size = %d, want %d", vi.GetDatFileSize(), origDatSize)
|
||||
}
|
||||
}
|
||||
|
||||
// TestFixEcxFromShardsWithVif verifies that when the .vif survives (recording
|
||||
// the exact .dat size and EC ratio) the .ecx is rebuilt from it and the .vif is
|
||||
// left untouched.
|
||||
func TestFixEcxFromShardsWithVif(t *testing.T) {
|
||||
oldData, oldParity := *fixEcDataShards, *fixEcParityShards
|
||||
*fixEcDataShards, *fixEcParityShards = 0, 0
|
||||
*fixIgnoreError = true
|
||||
t.Cleanup(func() {
|
||||
*fixIgnoreError = false
|
||||
*fixEcDataShards, *fixEcParityShards = oldData, oldParity
|
||||
})
|
||||
|
||||
dir := t.TempDir()
|
||||
const volumeId = 9
|
||||
base, canonical, origDatSize := buildAndEncodeTestEcVolume(t, dir, "9")
|
||||
|
||||
// Write a .vif as the volume server would after encoding.
|
||||
if err := volume_info.SaveVolumeInfo(base+".vif", &volume_server_pb.VolumeInfo{
|
||||
Version: uint32(needle.GetCurrentVersion()),
|
||||
DatFileSize: origDatSize,
|
||||
EcShardConfig: &volume_server_pb.EcShardConfig{
|
||||
DataShards: uint32(erasure_coding.DataShardsCount),
|
||||
ParityShards: uint32(erasure_coding.ParityShardsCount),
|
||||
},
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
vifBefore, err := os.ReadFile(base + ".vif")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// Lose the index but keep the surviving .vif and shards.
|
||||
for _, ext := range []string{".ecx", ".ecj", ".idx", ".dat"} {
|
||||
if err := os.Remove(base + ext); err != nil && !os.IsNotExist(err) {
|
||||
t.Fatalf("remove %s: %v", base+ext, err)
|
||||
}
|
||||
}
|
||||
|
||||
doFixEcxFromShards(dir, "9", "", volumeId)
|
||||
|
||||
recovered, err := os.ReadFile(base + ".ecx")
|
||||
if err != nil {
|
||||
t.Fatalf("recovered .ecx not written: %v", err)
|
||||
}
|
||||
if !bytes.Equal(canonical, recovered) {
|
||||
t.Fatalf(".ecx mismatch: canonical %d bytes, recovered %d bytes", len(canonical), len(recovered))
|
||||
}
|
||||
|
||||
// An existing .vif must be left untouched.
|
||||
vifAfter, err := os.ReadFile(base + ".vif")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !bytes.Equal(vifBefore, vifAfter) {
|
||||
t.Fatalf("existing .vif was modified")
|
||||
}
|
||||
}
|
||||
|
||||
// TestFixEcxFromShardsMissingShards verifies that when some shards (including a
|
||||
// couple of data shards) are lost but at least dataShards survive, the missing
|
||||
// shards are reconstructed from parity and the .ecx is still rebuilt correctly.
|
||||
func TestFixEcxFromShardsMissingShards(t *testing.T) {
|
||||
oldData, oldParity := *fixEcDataShards, *fixEcParityShards
|
||||
*fixEcDataShards, *fixEcParityShards = 0, 0
|
||||
*fixIgnoreError = true
|
||||
t.Cleanup(func() {
|
||||
*fixIgnoreError = false
|
||||
*fixEcDataShards, *fixEcParityShards = oldData, oldParity
|
||||
})
|
||||
|
||||
dir := t.TempDir()
|
||||
const volumeId = 11
|
||||
base, canonical, origDatSize := buildAndEncodeTestEcVolume(t, dir, "11")
|
||||
|
||||
// Lose every index/metadata file plus three shards (two data: .ec02, .ec05;
|
||||
// one parity: .ec11), keeping 11 of 14 — enough to reconstruct. The highest
|
||||
// shard (.ec13) is kept so the default 10+4 ratio is inferred without a .vif.
|
||||
for _, ext := range []string{".ecx", ".ecj", ".idx", ".dat", ".vif",
|
||||
erasure_coding.ToExt(2), erasure_coding.ToExt(5), erasure_coding.ToExt(11)} {
|
||||
if err := os.Remove(base + ext); err != nil && !os.IsNotExist(err) {
|
||||
t.Fatalf("remove %s: %v", base+ext, err)
|
||||
}
|
||||
}
|
||||
|
||||
doFixEcxFromShards(dir, "11", "", volumeId)
|
||||
|
||||
recovered, err := os.ReadFile(base + ".ecx")
|
||||
if err != nil {
|
||||
t.Fatalf("recovered .ecx not written: %v", err)
|
||||
}
|
||||
if !bytes.Equal(canonical, recovered) {
|
||||
t.Fatalf(".ecx mismatch: canonical %d bytes, recovered %d bytes", len(canonical), len(recovered))
|
||||
}
|
||||
|
||||
// The missing shards must have been reconstructed on disk.
|
||||
for _, idx := range []int{2, 5, 11} {
|
||||
if info, err := os.Stat(base + erasure_coding.ToExt(idx)); err != nil || info.Size() == 0 {
|
||||
t.Fatalf("missing shard %d was not reconstructed: err=%v", idx, err)
|
||||
}
|
||||
}
|
||||
|
||||
vi, _, found, err := volume_info.MaybeLoadVolumeInfo(base + ".vif")
|
||||
if err != nil || !found {
|
||||
t.Fatalf(".vif not regenerated: found=%v err=%v", found, err)
|
||||
}
|
||||
if vi.GetDatFileSize() != origDatSize {
|
||||
t.Fatalf("dat size = %d, want %d", vi.GetDatFileSize(), origDatSize)
|
||||
}
|
||||
}
|
||||
@@ -61,7 +61,8 @@ type MountOptions struct {
|
||||
|
||||
dirIdleEvictSec *int
|
||||
|
||||
// Distributed lock for cross-mount write coordination
|
||||
// Distributed locking for cross-mount write coordination and POSIX
|
||||
// advisory locks (flock/fcntl)
|
||||
distributedLock *bool
|
||||
|
||||
// POSIX compliance options
|
||||
@@ -152,7 +153,7 @@ func init() {
|
||||
mountReadRetryTime = cmdMount.Flag.Duration("readRetryTime", 6*time.Second, "maximum read retry wait time")
|
||||
|
||||
// Distributed lock for cross-mount write coordination
|
||||
mountOptions.distributedLock = cmdMount.Flag.Bool("dlm", false, "enable distributed lock for cross-mount write coordination (only one mount can write a file at a time)")
|
||||
mountOptions.distributedLock = cmdMount.Flag.Bool("dlm", false, "coordinate writes across mounts (only one mount writes a file at a time) and honor POSIX advisory locks (flock/fcntl) across mounts by routing them to the owner filer")
|
||||
|
||||
// POSIX compliance options
|
||||
mountOptions.posixDirNlink = cmdMount.Flag.Bool("posix.dirNLink", false, "report POSIX-compliant directory nlink (2 + subdirectory count); costs one directory listing per stat")
|
||||
|
||||
+9
-2
@@ -209,7 +209,11 @@ func (f *Filer) RollbackTransaction(ctx context.Context) error {
|
||||
return f.Store.RollbackTransaction(ctx)
|
||||
}
|
||||
|
||||
func (f *Filer) CreateEntry(ctx context.Context, entry *Entry, o_excl bool, isFromOtherCluster bool, signatures []int32, skipCreateParentDir bool, maxFilenameLength uint32) error {
|
||||
// CreateEntry creates or replaces an entry. When existing is non-nil the caller
|
||||
// has already fetched the current entry at this path under a path lock, and it
|
||||
// is reused instead of looking the store up again; pass nil to have CreateEntry
|
||||
// look it up itself.
|
||||
func (f *Filer) CreateEntry(ctx context.Context, entry *Entry, existing *Entry, o_excl bool, isFromOtherCluster bool, signatures []int32, skipCreateParentDir bool, maxFilenameLength uint32) error {
|
||||
|
||||
if string(entry.FullPath) == "/" {
|
||||
return nil
|
||||
@@ -227,7 +231,10 @@ func (f *Filer) CreateEntry(ctx context.Context, entry *Entry, o_excl bool, isFr
|
||||
entry.Attr.Atime = entryInitialAtime(entry.Attr)
|
||||
}
|
||||
|
||||
oldEntry, _ := f.FindEntry(ctx, entry.FullPath)
|
||||
oldEntry := existing
|
||||
if oldEntry == nil {
|
||||
oldEntry, _ = f.FindEntry(ctx, entry.FullPath)
|
||||
}
|
||||
|
||||
/*
|
||||
if !hasWritePermission(lastDirectoryEntry, entry) {
|
||||
|
||||
@@ -28,7 +28,7 @@ func TestCreateEntryAssignsInodeWhenMissing(t *testing.T) {
|
||||
},
|
||||
}
|
||||
|
||||
err := f.CreateEntry(context.Background(), entry, false, false, nil, false, f.MaxFilenameLength)
|
||||
err := f.CreateEntry(context.Background(), entry, nil, false, false, nil, false, f.MaxFilenameLength)
|
||||
require.NoError(t, err)
|
||||
|
||||
stored, findErr := store.FindEntry(context.Background(), entry.FullPath)
|
||||
@@ -48,7 +48,7 @@ func TestCreateEntryAssignsInodesToAutoCreatedParents(t *testing.T) {
|
||||
},
|
||||
}
|
||||
|
||||
err := f.CreateEntry(context.Background(), entry, false, false, nil, false, f.MaxFilenameLength)
|
||||
err := f.CreateEntry(context.Background(), entry, nil, false, false, nil, false, f.MaxFilenameLength)
|
||||
require.NoError(t, err)
|
||||
|
||||
for _, path := range []string{"/a", "/a/b", "/a/b/c.txt"} {
|
||||
|
||||
@@ -96,7 +96,7 @@ func (f *Filer) maybeLazyFetchFromRemote(ctx context.Context, p util.FullPath) (
|
||||
persistBaseCtx, cancelPersist := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
defer cancelPersist()
|
||||
persistCtx := context.WithValue(persistBaseCtx, lazyFetchContextKey{}, true)
|
||||
saveErr := f.CreateEntry(persistCtx, entry, false, false, nil, true, f.MaxFilenameLength)
|
||||
saveErr := f.CreateEntry(persistCtx, entry, nil, false, false, nil, true, f.MaxFilenameLength)
|
||||
if saveErr != nil {
|
||||
glog.Warningf("maybeLazyFetchFromRemote: failed to persist filer entry for %s: %v", p, saveErr)
|
||||
f.lazyFetchGroup.Forget(key)
|
||||
|
||||
@@ -152,7 +152,7 @@ func (f *Filer) maybeLazyListFromRemote(ctx context.Context, p util.FullPath) {
|
||||
entry.Attr.FileSize = uint64(remoteEntry.RemoteSize)
|
||||
}
|
||||
}
|
||||
if saveErr := f.CreateEntry(persistCtx, entry, false, false, nil, true, f.MaxFilenameLength); saveErr != nil {
|
||||
if saveErr := f.CreateEntry(persistCtx, entry, nil, false, false, nil, true, f.MaxFilenameLength); saveErr != nil {
|
||||
glog.Warningf("maybeLazyListFromRemote: persist %s: %v", childPath, saveErr)
|
||||
}
|
||||
}
|
||||
@@ -193,7 +193,7 @@ func (f *Filer) updateDirectoryListingSyncedAt(ctx context.Context, p util.FullP
|
||||
dirEntry.Extended = make(map[string][]byte)
|
||||
}
|
||||
dirEntry.Extended[xattrRemoteListingSyncedAt] = []byte(fmt.Sprintf("%d", syncTime.Unix()))
|
||||
if saveErr := f.CreateEntry(ctx, dirEntry, false, false, nil, true, f.MaxFilenameLength); saveErr != nil {
|
||||
if saveErr := f.CreateEntry(ctx, dirEntry, nil, false, false, nil, true, f.MaxFilenameLength); saveErr != nil {
|
||||
glog.Warningf("maybeLazyListFromRemote: create dir synced_at for %s: %v", p, saveErr)
|
||||
}
|
||||
return
|
||||
|
||||
@@ -43,7 +43,7 @@ func (f *Filer) appendToFile(targetFile string, data []byte) error {
|
||||
entry.Chunks = append(entry.GetChunks(), uploadResult.ToPbFileChunk(assignResult.Fid, offset, time.Now().UnixNano()))
|
||||
|
||||
// update the entry
|
||||
err = f.CreateEntry(context.Background(), entry, false, false, nil, false, f.MaxFilenameLength)
|
||||
err = f.CreateEntry(context.Background(), entry, nil, false, false, nil, false, f.MaxFilenameLength)
|
||||
|
||||
return err
|
||||
}
|
||||
|
||||
@@ -32,7 +32,7 @@ func TestCreateAndFind(t *testing.T) {
|
||||
},
|
||||
}
|
||||
|
||||
if err := testFiler.CreateEntry(ctx, entry1, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
|
||||
if err := testFiler.CreateEntry(ctx, entry1, nil, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
|
||||
t.Errorf("create entry %v: %v", entry1.FullPath, err)
|
||||
return
|
||||
}
|
||||
|
||||
@@ -29,7 +29,7 @@ func TestCreateAndFind(t *testing.T) {
|
||||
},
|
||||
}
|
||||
|
||||
if err := testFiler.CreateEntry(ctx, entry1, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
|
||||
if err := testFiler.CreateEntry(ctx, entry1, nil, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
|
||||
t.Errorf("create entry %v: %v", entry1.FullPath, err)
|
||||
return
|
||||
}
|
||||
|
||||
@@ -29,7 +29,7 @@ func TestCreateAndFind(t *testing.T) {
|
||||
},
|
||||
}
|
||||
|
||||
if err := testFiler.CreateEntry(ctx, entry1, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
|
||||
if err := testFiler.CreateEntry(ctx, entry1, nil, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
|
||||
t.Errorf("create entry %v: %v", entry1.FullPath, err)
|
||||
return
|
||||
}
|
||||
|
||||
@@ -33,6 +33,8 @@ func (store *MysqlStore) GetName() string {
|
||||
}
|
||||
|
||||
func (store *MysqlStore) Initialize(configuration util.Configuration, prefix string) (err error) {
|
||||
// Absent key keeps a pooled default; an explicit 0 disables the idle pool.
|
||||
configuration.SetDefault(prefix+"connection_max_idle", 2)
|
||||
return store.initialize(
|
||||
configuration.GetString(prefix+"dsn"),
|
||||
configuration.GetString(prefix+"upsertQuery"),
|
||||
|
||||
@@ -33,6 +33,8 @@ func (store *MysqlStore2) GetName() string {
|
||||
}
|
||||
|
||||
func (store *MysqlStore2) Initialize(configuration util.Configuration, prefix string) (err error) {
|
||||
// Absent key keeps a pooled default; an explicit 0 disables the idle pool.
|
||||
configuration.SetDefault(prefix+"connection_max_idle", 2)
|
||||
return store.initialize(
|
||||
configuration.GetString(prefix+"createTable"),
|
||||
configuration.GetString(prefix+"upsertQuery"),
|
||||
|
||||
@@ -0,0 +1,259 @@
|
||||
package posixlock
|
||||
|
||||
import (
|
||||
"sync"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Manager is the owner filer's in-memory authority for POSIX advisory locks
|
||||
// across inodes. Lock state lives here, not in replicated metadata: it is
|
||||
// transient coordination, so keeping it out of the meta-log avoids churn and
|
||||
// does not pollute what subscribers see (the distributed lock manager holds its
|
||||
// locks the same way). A `Set` per inode key, plus a session index so a dead
|
||||
// mount's locks are reaped in O(locks held) rather than by scanning every inode.
|
||||
//
|
||||
// key is an opaque inode identity supplied by the caller — the file's path, or
|
||||
// "hl:"+hex(HardLinkId) for a hardlinked inode — so all names of one inode share
|
||||
// a Set. The Manager is safe for concurrent use.
|
||||
type Manager struct {
|
||||
mu sync.Mutex
|
||||
byKey map[string]*Set // inode key -> held locks
|
||||
bySid map[uint64]map[string]bool // session -> keys it currently holds locks on
|
||||
lastSeen map[uint64]time.Time // session -> last keepalive; only renewing sessions are leased
|
||||
}
|
||||
|
||||
func NewManager() *Manager {
|
||||
return &Manager{
|
||||
byKey: make(map[string]*Set),
|
||||
bySid: make(map[uint64]map[string]bool),
|
||||
lastSeen: make(map[uint64]time.Time),
|
||||
}
|
||||
}
|
||||
|
||||
// Renew records a keepalive from a session, placing it under lease management.
|
||||
// Only sessions that have renewed are subject to ReapExpired, so a session that
|
||||
// never sends keepalives (e.g. before the mount keepalive exists) is never reaped.
|
||||
func (m *Manager) Renew(sid uint64) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
m.lastSeen[sid] = time.Now()
|
||||
}
|
||||
|
||||
// ReapExpired releases the locks of every leased session whose last keepalive is
|
||||
// older than ttl — a dead or partitioned mount. Sessions that never renewed are
|
||||
// left untouched. Returns the reaped session ids.
|
||||
func (m *Manager) ReapExpired(ttl time.Duration) []uint64 {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
cutoff := time.Now().Add(-ttl)
|
||||
var reaped []uint64
|
||||
for sid, seen := range m.lastSeen {
|
||||
if seen.After(cutoff) {
|
||||
continue
|
||||
}
|
||||
for key := range m.bySid[sid] {
|
||||
s := m.byKey[key]
|
||||
if s == nil {
|
||||
continue
|
||||
}
|
||||
s.ReleaseSession(sid)
|
||||
if s.Empty() {
|
||||
delete(m.byKey, key)
|
||||
}
|
||||
}
|
||||
delete(m.bySid, sid)
|
||||
delete(m.lastSeen, sid)
|
||||
reaped = append(reaped, sid)
|
||||
}
|
||||
return reaped
|
||||
}
|
||||
|
||||
// TryLock grants lk on key, or returns the conflicting lock and false. The set
|
||||
// is created on first use and dropped again when it empties.
|
||||
func (m *Manager) TryLock(key string, lk Range) (Range, bool) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
s, ok := m.byKey[key]
|
||||
if !ok {
|
||||
s = &Set{}
|
||||
}
|
||||
if c, granted := s.Acquire(lk); !granted {
|
||||
return c, false
|
||||
}
|
||||
if !ok {
|
||||
m.byKey[key] = s
|
||||
}
|
||||
m.index(lk.Sid, key)
|
||||
return Range{}, true
|
||||
}
|
||||
|
||||
// Track records a lock the server already granted, without arbitration. A mount
|
||||
// uses it to mirror its own held locks so it can re-assert them to the inode's
|
||||
// current owner filer after an ownership change or owner restart.
|
||||
func (m *Manager) Track(key string, lk Range) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
s, ok := m.byKey[key]
|
||||
if !ok {
|
||||
s = &Set{}
|
||||
m.byKey[key] = s
|
||||
}
|
||||
s.Grant(lk)
|
||||
m.index(lk.Sid, key)
|
||||
}
|
||||
|
||||
// Snapshot returns a copy of the held locks per key. A mount calls it to drive
|
||||
// re-assertion keepalives; the filer never does.
|
||||
func (m *Manager) Snapshot() map[string][]Range {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
out := make(map[string][]Range, len(m.byKey))
|
||||
for key, s := range m.byKey {
|
||||
out[key] = append([]Range(nil), s.locks...)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// Reassert rebuilds session sid's locks on key from the client's authoritative
|
||||
// list, renewing the lease. It replaces sid's existing locks on the key, then
|
||||
// re-acquires each asserted lock — arbitrating against other sessions so it never
|
||||
// double-grants. Locks that lost to another session in a migration window are
|
||||
// returned as conflicts. The owner filer calls this on a re-assertion keepalive.
|
||||
func (m *Manager) Reassert(key string, sid uint64, locks []Range) (conflicts []Range) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
m.lastSeen[sid] = time.Now()
|
||||
|
||||
s := m.byKey[key]
|
||||
if s == nil {
|
||||
if len(locks) == 0 {
|
||||
return nil
|
||||
}
|
||||
s = &Set{}
|
||||
m.byKey[key] = s
|
||||
}
|
||||
s.ReleaseSession(sid)
|
||||
for _, lk := range locks {
|
||||
lk.Sid = sid
|
||||
if c, granted := s.Acquire(lk); !granted {
|
||||
conflicts = append(conflicts, c)
|
||||
}
|
||||
}
|
||||
if setHasSession(s, sid) {
|
||||
m.index(sid, key)
|
||||
} else {
|
||||
m.deindex(sid, key)
|
||||
}
|
||||
if s.Empty() {
|
||||
delete(m.byKey, key)
|
||||
}
|
||||
return conflicts
|
||||
}
|
||||
|
||||
// Unlock releases lk's owner's locks within its namespace over lk's range.
|
||||
func (m *Manager) Unlock(key string, lk Range) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
s := m.byKey[key]
|
||||
if s == nil {
|
||||
return
|
||||
}
|
||||
s.Release(lk)
|
||||
m.afterRelease(key, s, lk.Sid)
|
||||
}
|
||||
|
||||
// GetLk reports the lock that would block proposed on key, if any.
|
||||
func (m *Manager) GetLk(key string, proposed Range) (Range, bool) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
s := m.byKey[key]
|
||||
if s == nil {
|
||||
return Range{}, false
|
||||
}
|
||||
return s.Conflict(proposed)
|
||||
}
|
||||
|
||||
// ReleasePosixOwner drops (sid, owner)'s fcntl locks on key — the flush-time path.
|
||||
func (m *Manager) ReleasePosixOwner(key string, sid, owner uint64) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
s := m.byKey[key]
|
||||
if s == nil {
|
||||
return
|
||||
}
|
||||
s.ReleasePosixOwner(sid, owner)
|
||||
m.afterRelease(key, s, sid)
|
||||
}
|
||||
|
||||
// ReleaseFlockOwner drops (sid, owner)'s flock locks on key — the release-time path.
|
||||
func (m *Manager) ReleaseFlockOwner(key string, sid, owner uint64) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
s := m.byKey[key]
|
||||
if s == nil {
|
||||
return
|
||||
}
|
||||
s.ReleaseFlockOwner(sid, owner)
|
||||
m.afterRelease(key, s, sid)
|
||||
}
|
||||
|
||||
// ReleaseSession drops every lock held by a session across all inodes it touched,
|
||||
// reaping a mount whose lease expired. O(locks held by the session).
|
||||
func (m *Manager) ReleaseSession(sid uint64) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
for key := range m.bySid[sid] {
|
||||
s := m.byKey[key]
|
||||
if s == nil {
|
||||
continue
|
||||
}
|
||||
s.ReleaseSession(sid)
|
||||
if s.Empty() {
|
||||
delete(m.byKey, key)
|
||||
}
|
||||
}
|
||||
delete(m.bySid, sid)
|
||||
}
|
||||
|
||||
// afterRelease prunes the session index when sid no longer holds any lock on key,
|
||||
// and drops the set when it empties. Only sid's presence can have changed, since
|
||||
// a release only removes sid's locks.
|
||||
func (m *Manager) afterRelease(key string, s *Set, sid uint64) {
|
||||
if s.Empty() {
|
||||
delete(m.byKey, key)
|
||||
m.deindex(sid, key)
|
||||
return
|
||||
}
|
||||
if !setHasSession(s, sid) {
|
||||
m.deindex(sid, key)
|
||||
}
|
||||
}
|
||||
|
||||
func (m *Manager) index(sid uint64, key string) {
|
||||
keys := m.bySid[sid]
|
||||
if keys == nil {
|
||||
keys = make(map[string]bool)
|
||||
m.bySid[sid] = keys
|
||||
}
|
||||
keys[key] = true
|
||||
}
|
||||
|
||||
func (m *Manager) deindex(sid uint64, key string) {
|
||||
keys := m.bySid[sid]
|
||||
if keys == nil {
|
||||
return
|
||||
}
|
||||
delete(keys, key)
|
||||
if len(keys) == 0 {
|
||||
delete(m.bySid, sid)
|
||||
}
|
||||
}
|
||||
|
||||
func setHasSession(s *Set, sid uint64) bool {
|
||||
for _, l := range s.Locks() {
|
||||
if l.Sid == sid {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
@@ -0,0 +1,194 @@
|
||||
package posixlock
|
||||
|
||||
import (
|
||||
"math"
|
||||
"runtime"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
func TestManagerGrantAndConflict(t *testing.T) {
|
||||
m := NewManager()
|
||||
if _, granted := m.TryLock("a", Range{Start: 0, End: 99, Type: Write, Sid: 1, Owner: 1}); !granted {
|
||||
t.Fatal("first lock should be granted")
|
||||
}
|
||||
if c, granted := m.TryLock("a", Range{Start: 50, End: 149, Type: Write, Sid: 2, Owner: 1}); granted {
|
||||
t.Fatalf("overlapping lock from another session should conflict, got grant; conflict=%+v", c)
|
||||
}
|
||||
// A different key is independent.
|
||||
if _, granted := m.TryLock("b", Range{Start: 0, End: 99, Type: Write, Sid: 2, Owner: 1}); !granted {
|
||||
t.Fatal("lock on a different key should be granted")
|
||||
}
|
||||
}
|
||||
|
||||
func TestManagerUnlockCleansEmptyKeyAndIndex(t *testing.T) {
|
||||
m := NewManager()
|
||||
lk := Range{Start: 0, End: 99, Type: Write, Sid: 1, Owner: 1}
|
||||
m.TryLock("a", lk)
|
||||
|
||||
if !m.bySid[1]["a"] {
|
||||
t.Fatal("session index should record the held key")
|
||||
}
|
||||
m.Unlock("a", Range{Start: 0, End: 99, Type: Unlock, Sid: 1, Owner: 1})
|
||||
|
||||
if _, ok := m.byKey["a"]; ok {
|
||||
t.Fatal("empty set should be dropped from byKey")
|
||||
}
|
||||
if _, ok := m.bySid[1]; ok {
|
||||
t.Fatal("session index should be pruned when it holds nothing")
|
||||
}
|
||||
}
|
||||
|
||||
func TestManagerPartialUnlockKeepsIndex(t *testing.T) {
|
||||
m := NewManager()
|
||||
m.TryLock("a", Range{Start: 0, End: 49, Type: Write, Sid: 1, Owner: 1})
|
||||
m.TryLock("a", Range{Start: 100, End: 149, Type: Write, Sid: 1, Owner: 1})
|
||||
// Release one of the two ranges; the session still holds the other.
|
||||
m.Unlock("a", Range{Start: 0, End: 49, Type: Unlock, Sid: 1, Owner: 1})
|
||||
|
||||
if !m.bySid[1]["a"] {
|
||||
t.Fatal("session still holds a lock on the key; index must remain")
|
||||
}
|
||||
if _, ok := m.byKey["a"]; !ok {
|
||||
t.Fatal("key should remain while a lock is held")
|
||||
}
|
||||
}
|
||||
|
||||
func TestManagerGetLk(t *testing.T) {
|
||||
m := NewManager()
|
||||
m.TryLock("a", Range{Start: 10, End: 50, Type: Write, Sid: 1, Owner: 1, Pid: 7})
|
||||
c, found := m.GetLk("a", Range{Start: 30, End: 70, Type: Read, Sid: 2, Owner: 1})
|
||||
if !found || c.Pid != 7 {
|
||||
t.Fatalf("expected conflict from pid 7, got %+v found=%v", c, found)
|
||||
}
|
||||
if _, found := m.GetLk("missing", Range{Start: 0, End: 1, Type: Write, Sid: 9, Owner: 9}); found {
|
||||
t.Fatal("missing key should report no conflict")
|
||||
}
|
||||
}
|
||||
|
||||
func TestManagerReleasePosixOwnerKeepsFlockAndIndex(t *testing.T) {
|
||||
m := NewManager()
|
||||
m.TryLock("a", Range{Start: 0, End: 99, Type: Write, Sid: 1, Owner: 1})
|
||||
m.TryLock("a", Range{Start: 0, End: math.MaxUint64, Type: Write, Sid: 1, Owner: 1, IsFlock: true})
|
||||
|
||||
m.ReleasePosixOwner("a", 1, 1)
|
||||
|
||||
// flock lock for the same session remains, so the index must remain too.
|
||||
if !m.bySid[1]["a"] {
|
||||
t.Fatal("session still holds the flock lock; index must remain")
|
||||
}
|
||||
if _, found := m.GetLk("a", Range{Start: 0, End: 10, Type: Write, Sid: 2, Owner: 2, IsFlock: true}); !found {
|
||||
t.Fatal("flock lock should survive ReleasePosixOwner")
|
||||
}
|
||||
if _, found := m.GetLk("a", Range{Start: 0, End: 10, Type: Write, Sid: 2, Owner: 2}); found {
|
||||
t.Fatal("fcntl lock should be gone after ReleasePosixOwner")
|
||||
}
|
||||
}
|
||||
|
||||
func TestManagerReleaseSessionReapsAcrossKeys(t *testing.T) {
|
||||
m := NewManager()
|
||||
m.TryLock("a", Range{Start: 0, End: 99, Type: Write, Sid: 1, Owner: 1})
|
||||
m.TryLock("b", Range{Start: 0, End: 99, Type: Write, Sid: 1, Owner: 2})
|
||||
m.TryLock("b", Range{Start: 200, End: 299, Type: Write, Sid: 2, Owner: 1})
|
||||
|
||||
m.ReleaseSession(1)
|
||||
|
||||
if _, ok := m.bySid[1]; ok {
|
||||
t.Fatal("reaped session should be gone from the index")
|
||||
}
|
||||
if _, ok := m.byKey["a"]; ok {
|
||||
t.Fatal("key a held only session 1's lock and should be dropped")
|
||||
}
|
||||
// Session 2's lock on b survives.
|
||||
if _, found := m.GetLk("b", Range{Start: 200, End: 299, Type: Write, Sid: 9, Owner: 9}); !found {
|
||||
t.Fatal("session 2's lock on b should remain after reaping session 1")
|
||||
}
|
||||
if !m.bySid[2]["b"] {
|
||||
t.Fatal("session 2 index entry should remain")
|
||||
}
|
||||
}
|
||||
|
||||
func TestManagerReapsOnlyStaleLeasedSessions(t *testing.T) {
|
||||
m := NewManager()
|
||||
// Session 1: holds a lock, leased but stale (renewed long ago).
|
||||
m.TryLock("a", Range{Start: 0, End: 99, Type: Write, Sid: 1, Owner: 1})
|
||||
m.Renew(1)
|
||||
m.lastSeen[1] = time.Now().Add(-time.Hour)
|
||||
// Session 2: holds a lock, leased and fresh.
|
||||
m.TryLock("b", Range{Start: 0, End: 99, Type: Write, Sid: 2, Owner: 1})
|
||||
m.Renew(2)
|
||||
// Session 3: holds a lock but never renewed (no lease) — must not be reaped.
|
||||
m.TryLock("c", Range{Start: 0, End: 99, Type: Write, Sid: 3, Owner: 1})
|
||||
|
||||
reaped := m.ReapExpired(30 * time.Second)
|
||||
|
||||
if len(reaped) != 1 || reaped[0] != 1 {
|
||||
t.Fatalf("only the stale leased session should be reaped, got %v", reaped)
|
||||
}
|
||||
if _, ok := m.byKey["a"]; ok {
|
||||
t.Fatal("stale session's lock should be gone")
|
||||
}
|
||||
if _, ok := m.byKey["b"]; !ok {
|
||||
t.Fatal("fresh session's lock must remain")
|
||||
}
|
||||
if _, ok := m.byKey["c"]; !ok {
|
||||
t.Fatal("never-renewed session must not be reaped")
|
||||
}
|
||||
if _, ok := m.lastSeen[1]; ok {
|
||||
t.Fatal("reaped session's lease entry should be cleared")
|
||||
}
|
||||
}
|
||||
|
||||
// Mutual exclusion under concurrent whole-file flock churn through the Manager:
|
||||
// at most one owner may believe it holds the exclusive lock at any instant.
|
||||
func TestManagerConcurrentFlockMutualExclusion(t *testing.T) {
|
||||
m := NewManager()
|
||||
const (
|
||||
key = "inode"
|
||||
workers = 16
|
||||
iters = 400
|
||||
)
|
||||
var (
|
||||
wg sync.WaitGroup
|
||||
holder atomic.Int64
|
||||
overlap atomic.Int32
|
||||
)
|
||||
for w := 0; w < workers; w++ {
|
||||
wg.Add(1)
|
||||
go func(id int) {
|
||||
defer wg.Done()
|
||||
lk := Range{Start: 0, End: math.MaxUint64, Type: Write, Sid: uint64(id + 1), Owner: 1, IsFlock: true}
|
||||
unlock := lk
|
||||
unlock.Type = Unlock
|
||||
token := int64(id + 1)
|
||||
for i := 0; i < iters; i++ {
|
||||
for {
|
||||
if _, granted := m.TryLock(key, lk); granted {
|
||||
break
|
||||
}
|
||||
runtime.Gosched()
|
||||
}
|
||||
if prev := holder.Swap(token); prev != 0 {
|
||||
overlap.Add(1)
|
||||
}
|
||||
runtime.Gosched()
|
||||
if !holder.CompareAndSwap(token, 0) {
|
||||
overlap.Add(1)
|
||||
}
|
||||
m.Unlock(key, unlock)
|
||||
}
|
||||
}(w)
|
||||
}
|
||||
wg.Wait()
|
||||
if n := overlap.Load(); n != 0 {
|
||||
t.Fatalf("mutual exclusion violated %d times", n)
|
||||
}
|
||||
if len(m.byKey) != 0 {
|
||||
t.Fatalf("all locks released; byKey should be empty, got %d", len(m.byKey))
|
||||
}
|
||||
if len(m.bySid) != 0 {
|
||||
t.Fatalf("all locks released; bySid should be empty, got %d", len(m.bySid))
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,222 @@
|
||||
// Package posixlock implements the conflict, coalescing, and range-split logic
|
||||
// for POSIX advisory file locks — fcntl byte-range and flock whole-file — as a
|
||||
// pure per-inode lock set with no concurrency control of its own.
|
||||
//
|
||||
// It is the server-side authority for distributed FUSE locking: the owner filer
|
||||
// for an inode holds one Set and serializes access to it under that inode's
|
||||
// per-path lock, so each operation runs to completion without concurrent
|
||||
// mutation. The same algorithm can back the per-mount table; blocking (SetLkw)
|
||||
// and any wait queue belong to the caller, not here.
|
||||
package posixlock
|
||||
|
||||
import (
|
||||
"math"
|
||||
"sort"
|
||||
)
|
||||
|
||||
// Lock types, kept independent of the platform syscall package (whose F_RDLCK /
|
||||
// F_WRLCK / F_UNLCK values differ per OS and are absent on some) so the filer
|
||||
// builds everywhere. Callers map the syscall constants onto these at the edge.
|
||||
// Zero is intentionally unused so a zero-value Range reads as "unset".
|
||||
const (
|
||||
Read uint32 = 1
|
||||
Write uint32 = 2
|
||||
Unlock uint32 = 3
|
||||
)
|
||||
|
||||
// Range is one held advisory byte-range lock. Owner identity is the (Sid, Owner)
|
||||
// pair: Sid is the mount session, Owner the FUSE lock owner within it, so owners
|
||||
// from different mounts never alias. End is inclusive; math.MaxUint64 means EOF.
|
||||
// IsFlock separates the flock and fcntl namespaces, which never conflict.
|
||||
type Range struct {
|
||||
Start uint64
|
||||
End uint64
|
||||
Type uint32
|
||||
Sid uint64
|
||||
Owner uint64
|
||||
Pid uint32
|
||||
IsFlock bool
|
||||
}
|
||||
|
||||
func (r Range) sameOwner(o Range) bool {
|
||||
return r.Sid == o.Sid && r.Owner == o.Owner
|
||||
}
|
||||
|
||||
// Set is the authoritative set of advisory locks held on one inode. The zero
|
||||
// value is an empty set. Set has no internal locking; the caller serializes it.
|
||||
type Set struct {
|
||||
locks []Range // sorted by Start
|
||||
}
|
||||
|
||||
func overlap(aStart, aEnd, bStart, bEnd uint64) bool {
|
||||
return aStart <= bEnd && bStart <= aEnd
|
||||
}
|
||||
|
||||
// Conflict returns the first held lock that blocks proposed, if any. Two locks
|
||||
// conflict when they share a namespace, have different owners, overlap, and at
|
||||
// least one is a write lock.
|
||||
func (s *Set) Conflict(proposed Range) (Range, bool) {
|
||||
for _, h := range s.locks {
|
||||
if h.IsFlock != proposed.IsFlock || h.sameOwner(proposed) {
|
||||
continue
|
||||
}
|
||||
if !overlap(h.Start, h.End, proposed.Start, proposed.End) {
|
||||
continue
|
||||
}
|
||||
if h.Type == Read && proposed.Type == Read {
|
||||
continue
|
||||
}
|
||||
return h, true
|
||||
}
|
||||
return Range{}, false
|
||||
}
|
||||
|
||||
// Acquire grants lk when it does not conflict, inserting it and coalescing the
|
||||
// owner's adjacent/overlapping ranges. On conflict it returns the blocking lock
|
||||
// and false, leaving the set unchanged.
|
||||
func (s *Set) Acquire(lk Range) (Range, bool) {
|
||||
if c, found := s.Conflict(lk); found {
|
||||
return c, false
|
||||
}
|
||||
s.insert(lk)
|
||||
return Range{}, true
|
||||
}
|
||||
|
||||
// Grant inserts lk without a conflict check. It is for a client mirroring locks
|
||||
// the server already granted (so they are conflict-free), not for arbitration.
|
||||
func (s *Set) Grant(lk Range) {
|
||||
s.insert(lk)
|
||||
}
|
||||
|
||||
// insert adds lk, absorbing same-owner same-type overlaps and merging adjacent
|
||||
// same-type ranges, and truncating/splitting a same-owner range of a different
|
||||
// type that overlaps (an in-place type change).
|
||||
func (s *Set) insert(lk Range) {
|
||||
var kept []Range
|
||||
for _, h := range s.locks {
|
||||
if !h.sameOwner(lk) || h.IsFlock != lk.IsFlock {
|
||||
kept = append(kept, h)
|
||||
continue
|
||||
}
|
||||
if !overlap(h.Start, h.End, lk.Start, lk.End) {
|
||||
// Merge only ranges that are adjacent and the same type. The
|
||||
// End < MaxUint64 guards stop +1 from wrapping at EOF.
|
||||
if h.Type == lk.Type && ((h.End < math.MaxUint64 && h.End+1 == lk.Start) || (lk.End < math.MaxUint64 && lk.End+1 == h.Start)) {
|
||||
if h.Start < lk.Start {
|
||||
lk.Start = h.Start
|
||||
}
|
||||
if h.End > lk.End {
|
||||
lk.End = h.End
|
||||
}
|
||||
continue
|
||||
}
|
||||
kept = append(kept, h)
|
||||
continue
|
||||
}
|
||||
if h.Type == lk.Type {
|
||||
// Same type: absorb into lk by widening its range.
|
||||
if h.Start < lk.Start {
|
||||
lk.Start = h.Start
|
||||
}
|
||||
if h.End > lk.End {
|
||||
lk.End = h.End
|
||||
}
|
||||
continue
|
||||
}
|
||||
// Different type: the surviving portions of h outside lk's range stay.
|
||||
if h.Start < lk.Start {
|
||||
left := h
|
||||
left.End = lk.Start - 1
|
||||
kept = append(kept, left)
|
||||
}
|
||||
if h.End > lk.End {
|
||||
right := h
|
||||
right.Start = lk.End + 1
|
||||
kept = append(kept, right)
|
||||
}
|
||||
}
|
||||
kept = append(kept, lk)
|
||||
sort.Slice(kept, func(i, j int) bool { return kept[i].Start < kept[j].Start })
|
||||
s.locks = kept
|
||||
}
|
||||
|
||||
// remove drops or splits matching locks within [start,end]. A match that
|
||||
// straddles the range keeps its non-overlapping head and/or tail.
|
||||
func (s *Set) remove(matches func(Range) bool, start, end uint64) {
|
||||
var kept []Range
|
||||
for _, h := range s.locks {
|
||||
if !matches(h) || !overlap(h.Start, h.End, start, end) {
|
||||
kept = append(kept, h)
|
||||
continue
|
||||
}
|
||||
if h.Start < start {
|
||||
left := h
|
||||
left.End = start - 1
|
||||
kept = append(kept, left)
|
||||
}
|
||||
if h.End > end {
|
||||
right := h
|
||||
right.Start = end + 1
|
||||
kept = append(kept, right)
|
||||
}
|
||||
// Fully covered: dropped.
|
||||
}
|
||||
s.locks = kept
|
||||
}
|
||||
|
||||
// Release clears lk's owner's locks within lk's namespace over [lk.Start,lk.End]
|
||||
// — the F_UNLCK path, which may split a straddling range.
|
||||
func (s *Set) Release(lk Range) {
|
||||
s.remove(func(h Range) bool {
|
||||
return h.sameOwner(lk) && h.IsFlock == lk.IsFlock
|
||||
}, lk.Start, lk.End)
|
||||
}
|
||||
|
||||
// ReleaseOwner removes every lock held by (sid, owner) in both namespaces.
|
||||
func (s *Set) ReleaseOwner(sid, owner uint64) {
|
||||
s.remove(func(h Range) bool {
|
||||
return h.Sid == sid && h.Owner == owner
|
||||
}, 0, math.MaxUint64)
|
||||
}
|
||||
|
||||
// ReleaseFlockOwner removes only the flock locks of (sid, owner) — the close-time
|
||||
// path for a released file description (FUSE_RELEASE_FLOCK_UNLOCK).
|
||||
func (s *Set) ReleaseFlockOwner(sid, owner uint64) {
|
||||
s.remove(func(h Range) bool {
|
||||
return h.IsFlock && h.Sid == sid && h.Owner == owner
|
||||
}, 0, math.MaxUint64)
|
||||
}
|
||||
|
||||
// ReleasePosixOwner removes only the fcntl locks of (sid, owner) — the close-time
|
||||
// path for a flushing POSIX lock owner.
|
||||
func (s *Set) ReleasePosixOwner(sid, owner uint64) {
|
||||
s.remove(func(h Range) bool {
|
||||
return !h.IsFlock && h.Sid == sid && h.Owner == owner
|
||||
}, 0, math.MaxUint64)
|
||||
}
|
||||
|
||||
// ReleaseSession removes every lock held by a session, reaping a mount that has
|
||||
// died or disconnected (its lease expired).
|
||||
func (s *Set) ReleaseSession(sid uint64) {
|
||||
s.remove(func(h Range) bool { return h.Sid == sid }, 0, math.MaxUint64)
|
||||
}
|
||||
|
||||
// HasPosix reports whether (sid, owner) holds any fcntl lock, mirroring the
|
||||
// mount's flush-time check that avoids treating a lock-free flush as
|
||||
// lock-sensitive.
|
||||
func (s *Set) HasPosix(sid, owner uint64) bool {
|
||||
for _, h := range s.locks {
|
||||
if !h.IsFlock && h.Sid == sid && h.Owner == owner {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// Locks returns the held locks, sorted by Start. The slice aliases internal
|
||||
// state; the caller must not mutate it.
|
||||
func (s *Set) Locks() []Range { return s.locks }
|
||||
|
||||
// Empty reports whether no locks are held, so the caller can drop the inode's
|
||||
// entry from its table.
|
||||
func (s *Set) Empty() bool { return len(s.locks) == 0 }
|
||||
@@ -0,0 +1,292 @@
|
||||
package posixlock
|
||||
|
||||
import (
|
||||
"math"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// acquire is a test helper asserting the lock is granted.
|
||||
func mustAcquire(t *testing.T, s *Set, lk Range) {
|
||||
t.Helper()
|
||||
if _, granted := s.Acquire(lk); !granted {
|
||||
t.Fatalf("expected lock granted: %+v", lk)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNonOverlappingLocksFromDifferentOwners(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 49, Type: Write, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 50, End: 99, Type: Write, Owner: 2, Pid: 20})
|
||||
}
|
||||
|
||||
func TestOverlappingReadLocksFromDifferentOwners(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Read, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 50, End: 149, Type: Read, Owner: 2, Pid: 20})
|
||||
}
|
||||
|
||||
func TestOverlappingWriteReadConflict(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
|
||||
if _, granted := s.Acquire(Range{Start: 50, End: 149, Type: Read, Owner: 2, Pid: 20}); granted {
|
||||
t.Fatal("expected conflict")
|
||||
}
|
||||
}
|
||||
|
||||
func TestOverlappingWriteWriteConflict(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
|
||||
if _, granted := s.Acquire(Range{Start: 50, End: 149, Type: Write, Owner: 2, Pid: 20}); granted {
|
||||
t.Fatal("expected conflict")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSameOwnerUpgradeReadToWrite(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Read, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
|
||||
|
||||
c, found := s.Conflict(Range{Start: 0, End: 99, Type: Write, Owner: 2, Pid: 20})
|
||||
if !found || c.Type != Write {
|
||||
t.Fatalf("expected conflicting write lock after upgrade, got %+v found=%v", c, found)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSameOwnerDowngradeWriteToRead(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Read, Owner: 1, Pid: 10})
|
||||
// Another owner can now take a shared read lock.
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Read, Owner: 2, Pid: 20})
|
||||
}
|
||||
|
||||
func TestLockCoalescing(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 9, Type: Write, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 10, End: 19, Type: Write, Owner: 1, Pid: 10})
|
||||
|
||||
if len(s.locks) != 1 {
|
||||
t.Fatalf("expected 1 coalesced lock, got %d: %+v", len(s.locks), s.locks)
|
||||
}
|
||||
if s.locks[0].Start != 0 || s.locks[0].End != 19 {
|
||||
t.Errorf("expected coalesced [0,19], got [%d,%d]", s.locks[0].Start, s.locks[0].End)
|
||||
}
|
||||
}
|
||||
|
||||
func TestLockSplitting(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
|
||||
s.Release(Range{Start: 40, End: 59, Type: Unlock, Owner: 1, Pid: 10})
|
||||
|
||||
if len(s.locks) != 2 {
|
||||
t.Fatalf("expected 2 locks after split, got %d: %+v", len(s.locks), s.locks)
|
||||
}
|
||||
if s.locks[0].Start != 0 || s.locks[0].End != 39 {
|
||||
t.Errorf("expected left [0,39], got [%d,%d]", s.locks[0].Start, s.locks[0].End)
|
||||
}
|
||||
if s.locks[1].Start != 60 || s.locks[1].End != 99 {
|
||||
t.Errorf("expected right [60,99], got [%d,%d]", s.locks[1].Start, s.locks[1].End)
|
||||
}
|
||||
}
|
||||
|
||||
func TestConflictReportsHolder(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 10, End: 50, Type: Write, Owner: 1, Pid: 10})
|
||||
|
||||
c, found := s.Conflict(Range{Start: 30, End: 70, Type: Read, Owner: 2, Pid: 20})
|
||||
if !found {
|
||||
t.Fatal("expected a conflict")
|
||||
}
|
||||
if c.Type != Write || c.Pid != 10 || c.Start != 10 || c.End != 50 {
|
||||
t.Fatalf("unexpected conflict report: %+v", c)
|
||||
}
|
||||
}
|
||||
|
||||
func TestConflictNoneForSharedReads(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 10, End: 50, Type: Read, Owner: 1, Pid: 10})
|
||||
if _, found := s.Conflict(Range{Start: 30, End: 70, Type: Read, Owner: 2, Pid: 20}); found {
|
||||
t.Fatal("two read locks should not conflict")
|
||||
}
|
||||
}
|
||||
|
||||
func TestConflictSameOwnerNone(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
|
||||
if _, found := s.Conflict(Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10}); found {
|
||||
t.Fatal("an owner should not conflict with itself")
|
||||
}
|
||||
}
|
||||
|
||||
func TestReleaseOwner(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 49, Type: Write, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 50, End: 99, Type: Write, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 200, End: 299, Type: Read, Owner: 2, Pid: 20})
|
||||
|
||||
s.ReleaseOwner(0, 1)
|
||||
|
||||
if _, found := s.Conflict(Range{Start: 0, End: 99, Type: Write, Owner: 3, Pid: 30}); found {
|
||||
t.Fatal("owner 1's locks should be gone")
|
||||
}
|
||||
c, found := s.Conflict(Range{Start: 200, End: 299, Type: Write, Owner: 3, Pid: 30})
|
||||
if !found || c.Type != Read {
|
||||
t.Fatalf("owner 2's read lock should remain, got %+v found=%v", c, found)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFlockAndFcntlDoNotConflict(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 0, End: math.MaxUint64, Type: Write, Owner: 2, Pid: 20, IsFlock: true})
|
||||
}
|
||||
|
||||
func TestReleasePosixOwnerKeepsFlock(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: math.MaxUint64, Type: Write, Owner: 1, Pid: 10, IsFlock: true})
|
||||
s.ReleasePosixOwner(0, 1)
|
||||
if _, found := s.Conflict(Range{Start: 0, End: math.MaxUint64, Type: Write, Owner: 2, Pid: 20, IsFlock: true}); !found {
|
||||
t.Fatal("flock lock should remain after ReleasePosixOwner")
|
||||
}
|
||||
}
|
||||
|
||||
func TestReleaseFlockOwnerKeepsPosix(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 0, End: math.MaxUint64, Type: Write, Owner: 2, Pid: 10, IsFlock: true})
|
||||
|
||||
s.ReleaseFlockOwner(0, 2)
|
||||
|
||||
if _, found := s.Conflict(Range{Start: 0, End: 99, Type: Write, Owner: 3, Pid: 30}); !found {
|
||||
t.Fatal("fcntl lock should remain after ReleaseFlockOwner")
|
||||
}
|
||||
if _, found := s.Conflict(Range{Start: 0, End: math.MaxUint64, Type: Write, Owner: 4, Pid: 40, IsFlock: true}); found {
|
||||
t.Fatal("flock lock should be gone after ReleaseFlockOwner")
|
||||
}
|
||||
}
|
||||
|
||||
func TestHasPosixIgnoresMissingOwnerAndFlock(t *testing.T) {
|
||||
s := &Set{}
|
||||
if s.HasPosix(0, 1) {
|
||||
t.Fatal("empty set should report no posix owner")
|
||||
}
|
||||
mustAcquire(t, s, Range{Start: 0, End: math.MaxUint64, Type: Write, Owner: 1, Pid: 10, IsFlock: true})
|
||||
if s.HasPosix(0, 1) {
|
||||
t.Fatal("a flock owner is not a posix owner")
|
||||
}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 2, Pid: 20})
|
||||
if !s.HasPosix(0, 2) {
|
||||
t.Fatal("posix owner should be reported")
|
||||
}
|
||||
}
|
||||
|
||||
func TestWholeFileLock(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: math.MaxUint64, Type: Write, Owner: 1, Pid: 10})
|
||||
if _, granted := s.Acquire(Range{Start: 0, End: math.MaxUint64, Type: Write, Owner: 2, Pid: 20}); granted {
|
||||
t.Fatal("whole-file lock should block another owner")
|
||||
}
|
||||
if _, granted := s.Acquire(Range{Start: 100, End: 200, Type: Read, Owner: 2, Pid: 20}); granted {
|
||||
t.Fatal("partial overlap with whole-file lock should conflict")
|
||||
}
|
||||
}
|
||||
|
||||
func TestReleaseNoExistingLocks(t *testing.T) {
|
||||
s := &Set{}
|
||||
s.Release(Range{Start: 0, End: 99, Type: Unlock, Owner: 1, Pid: 10})
|
||||
if !s.Empty() {
|
||||
t.Fatal("releasing on an empty set should be a no-op")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSameOwnerReplaceDifferentType(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 30, End: 60, Type: Read, Owner: 1, Pid: 10})
|
||||
|
||||
if len(s.locks) != 3 {
|
||||
t.Fatalf("expected 3 locks after partial type change, got %d: %+v", len(s.locks), s.locks)
|
||||
}
|
||||
if s.locks[0].Type != Write || s.locks[0].Start != 0 || s.locks[0].End != 29 {
|
||||
t.Errorf("expected write [0,29], got %+v", s.locks[0])
|
||||
}
|
||||
if s.locks[1].Type != Read || s.locks[1].Start != 30 || s.locks[1].End != 60 {
|
||||
t.Errorf("expected read [30,60], got %+v", s.locks[1])
|
||||
}
|
||||
if s.locks[2].Type != Write || s.locks[2].Start != 61 || s.locks[2].End != 99 {
|
||||
t.Errorf("expected write [61,99], got %+v", s.locks[2])
|
||||
}
|
||||
}
|
||||
|
||||
func TestNonAdjacentRangesNotCoalesced(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 5, End: math.MaxUint64, Type: Write, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 0, End: 2, Type: Write, Owner: 1, Pid: 10})
|
||||
|
||||
if len(s.locks) != 2 {
|
||||
t.Fatalf("gap [3,4] should prevent coalescing, got %d: %+v", len(s.locks), s.locks)
|
||||
}
|
||||
if s.locks[0].Start != 0 || s.locks[0].End != 2 {
|
||||
t.Errorf("expected [0,2], got [%d,%d]", s.locks[0].Start, s.locks[0].End)
|
||||
}
|
||||
if s.locks[1].Start != 5 || s.locks[1].End != math.MaxUint64 {
|
||||
t.Errorf("expected [5,MaxUint64], got [%d,%d]", s.locks[1].Start, s.locks[1].End)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAdjacencyNoOverflowAtMaxUint64(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 100, End: math.MaxUint64, Type: Write, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 0, End: 0, Type: Write, Owner: 1, Pid: 10})
|
||||
|
||||
if len(s.locks) != 2 {
|
||||
t.Fatalf("MaxUint64+1 must not wrap and falsely merge, got %d: %+v", len(s.locks), s.locks)
|
||||
}
|
||||
}
|
||||
|
||||
// Sessions are part of owner identity: the same FUSE Owner number on two
|
||||
// different mounts (Sid) is two distinct owners and must contend.
|
||||
func TestTwoSessionsSameOwnerDoNotAlias(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Sid: 1, Owner: 5, Pid: 10})
|
||||
|
||||
if _, granted := s.Acquire(Range{Start: 50, End: 149, Type: Write, Sid: 2, Owner: 5, Pid: 20}); granted {
|
||||
t.Fatal("same Owner number on a different session must conflict, not alias")
|
||||
}
|
||||
// A read on session 2 against a session-1 read is fine (shared).
|
||||
s2 := &Set{}
|
||||
mustAcquire(t, s2, Range{Start: 0, End: 99, Type: Read, Sid: 1, Owner: 5})
|
||||
mustAcquire(t, s2, Range{Start: 0, End: 99, Type: Read, Sid: 2, Owner: 5})
|
||||
}
|
||||
|
||||
// Reaping a dead mount drops only that session's locks.
|
||||
func TestReleaseSessionReapsOnlyThatSession(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 49, Type: Write, Sid: 1, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 0, End: math.MaxUint64, Type: Write, Sid: 1, Owner: 2, Pid: 11, IsFlock: true})
|
||||
mustAcquire(t, s, Range{Start: 50, End: 99, Type: Write, Sid: 2, Owner: 1, Pid: 20})
|
||||
|
||||
s.ReleaseSession(1)
|
||||
|
||||
for _, h := range s.locks {
|
||||
if h.Sid == 1 {
|
||||
t.Fatalf("session 1 lock survived reaping: %+v", h)
|
||||
}
|
||||
}
|
||||
c, found := s.Conflict(Range{Start: 50, End: 99, Type: Write, Sid: 3, Owner: 9})
|
||||
if !found || c.Sid != 2 {
|
||||
t.Fatalf("session 2's lock should remain, got %+v found=%v", c, found)
|
||||
}
|
||||
}
|
||||
|
||||
func TestEmptyAfterReleasingAll(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
|
||||
if s.Empty() {
|
||||
t.Fatal("set should not be empty with a held lock")
|
||||
}
|
||||
s.Release(Range{Start: 0, End: 99, Type: Unlock, Owner: 1, Pid: 10})
|
||||
if !s.Empty() {
|
||||
t.Fatal("set should be empty after releasing the only lock")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,82 @@
|
||||
package posixlock
|
||||
|
||||
import (
|
||||
"reflect"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// A mount's tracked locks round-trip through Snapshot back to a fresh owner via
|
||||
// Reassert — the owner-restart / ring-change recovery path.
|
||||
func TestReassertRebuildsOnFreshOwner(t *testing.T) {
|
||||
const sid = uint64(7)
|
||||
|
||||
// Client mirror: two granted locks on one key, one on another.
|
||||
client := NewManager()
|
||||
client.Track("a", Range{Start: 0, End: 99, Type: Write, Sid: sid, Owner: 1})
|
||||
client.Track("a", Range{Start: 200, End: 299, Type: Read, Sid: sid, Owner: 2})
|
||||
client.Track("b", Range{Start: 0, End: maxEnd, Type: Write, Sid: sid, Owner: 1, IsFlock: true})
|
||||
|
||||
// Fresh owner (post-restart / new ring owner) knows nothing.
|
||||
owner := NewManager()
|
||||
for key, locks := range client.Snapshot() {
|
||||
if c := owner.Reassert(key, sid, locks); c != nil {
|
||||
t.Fatalf("unexpected conflict reasserting %s: %+v", key, c)
|
||||
}
|
||||
}
|
||||
|
||||
// The owner now reports the same conflicts a foreign session would hit.
|
||||
if _, granted := owner.TryLock("a", Range{Start: 50, End: 60, Type: Write, Sid: 99, Owner: 1}); granted {
|
||||
t.Fatal("owner should block a foreign write after rebuild")
|
||||
}
|
||||
if _, granted := owner.TryLock("b", Range{Start: 0, End: 0, Type: Read, Sid: 99, Owner: 1, IsFlock: true}); granted {
|
||||
t.Fatal("owner should block a foreign flock read after rebuild")
|
||||
}
|
||||
}
|
||||
|
||||
// Re-asserting every tick is idempotent: the owner's view is unchanged.
|
||||
func TestReassertIdempotent(t *testing.T) {
|
||||
const sid = uint64(1)
|
||||
m := NewManager()
|
||||
m.TryLock("k", Range{Start: 0, End: 99, Type: Write, Sid: sid, Owner: 1})
|
||||
before := append([]Range(nil), m.byKey["k"].locks...)
|
||||
|
||||
m.Reassert("k", sid, before)
|
||||
m.Reassert("k", sid, before)
|
||||
|
||||
if !reflect.DeepEqual(m.byKey["k"].locks, before) {
|
||||
t.Fatalf("reassert not idempotent:\n got %+v\nwant %+v", m.byKey["k"].locks, before)
|
||||
}
|
||||
}
|
||||
|
||||
// A lock another session grabbed in the migration window is reported as a
|
||||
// conflict and not double-granted.
|
||||
func TestReassertReportsConflict(t *testing.T) {
|
||||
const mine, other = uint64(1), uint64(2)
|
||||
m := NewManager()
|
||||
// Another mount took the lock on this (new) owner during the gap.
|
||||
m.TryLock("k", Range{Start: 0, End: 99, Type: Write, Sid: other, Owner: 1})
|
||||
|
||||
conflicts := m.Reassert("k", mine, []Range{{Start: 0, End: 99, Type: Write, Sid: mine, Owner: 1}})
|
||||
if len(conflicts) != 1 {
|
||||
t.Fatalf("expected 1 conflict, got %d: %+v", len(conflicts), conflicts)
|
||||
}
|
||||
// The other session keeps the lock; mine was not installed.
|
||||
if got := len(m.byKey["k"].locks); got != 1 {
|
||||
t.Fatalf("expected only the incumbent lock, got %d", got)
|
||||
}
|
||||
}
|
||||
|
||||
// Reassert renews the lease, so a re-asserting mount is not reaped.
|
||||
func TestReassertRenewsLease(t *testing.T) {
|
||||
const sid = uint64(1)
|
||||
m := NewManager()
|
||||
m.Renew(sid)
|
||||
m.Reassert("k", sid, []Range{{Start: 0, End: 9, Type: Write, Sid: sid, Owner: 1}})
|
||||
|
||||
if reaped := m.ReapExpired(time.Hour); len(reaped) != 0 {
|
||||
t.Fatalf("freshly re-asserted session should not be reaped: %v", reaped)
|
||||
}
|
||||
}
|
||||
|
||||
const maxEnd = ^uint64(0)
|
||||
@@ -28,6 +28,8 @@ func (store *PostgresStore) GetName() string {
|
||||
}
|
||||
|
||||
func (store *PostgresStore) Initialize(configuration util.Configuration, prefix string) (err error) {
|
||||
// Absent key keeps a pooled default; an explicit 0 disables the idle pool.
|
||||
configuration.SetDefault(prefix+"connection_max_idle", 2)
|
||||
return store.initialize(
|
||||
configuration.GetString(prefix+"upsertQuery"),
|
||||
configuration.GetBool(prefix+"enableUpsert"),
|
||||
|
||||
@@ -33,6 +33,8 @@ func (store *PostgresStore2) GetName() string {
|
||||
}
|
||||
|
||||
func (store *PostgresStore2) Initialize(configuration util.Configuration, prefix string) (err error) {
|
||||
// Absent key keeps a pooled default; an explicit 0 disables the idle pool.
|
||||
configuration.SetDefault(prefix+"connection_max_idle", 2)
|
||||
return store.initialize(
|
||||
configuration.GetString(prefix+"createTable"),
|
||||
configuration.GetString(prefix+"upsertQuery"),
|
||||
|
||||
@@ -10,7 +10,7 @@ import (
|
||||
|
||||
func (store *UniversalRedis2Store) KvPut(ctx context.Context, key []byte, value []byte) (err error) {
|
||||
|
||||
_, err = store.Client.Set(ctx, string(key), value, 0).Result()
|
||||
_, err = store.Client.Set(ctx, store.getKey(string(key)), value, 0).Result()
|
||||
|
||||
if err != nil {
|
||||
return fmt.Errorf("kv put: %w", err)
|
||||
@@ -21,7 +21,7 @@ func (store *UniversalRedis2Store) KvPut(ctx context.Context, key []byte, value
|
||||
|
||||
func (store *UniversalRedis2Store) KvGet(ctx context.Context, key []byte) (value []byte, err error) {
|
||||
|
||||
data, err := store.Client.Get(ctx, string(key)).Result()
|
||||
data, err := store.Client.Get(ctx, store.getKey(string(key))).Result()
|
||||
|
||||
if err == redis.Nil {
|
||||
return nil, filer.ErrKvNotFound
|
||||
@@ -32,7 +32,7 @@ func (store *UniversalRedis2Store) KvGet(ctx context.Context, key []byte) (value
|
||||
|
||||
func (store *UniversalRedis2Store) KvDelete(ctx context.Context, key []byte) (err error) {
|
||||
|
||||
_, err = store.Client.Del(ctx, string(key)).Result()
|
||||
_, err = store.Client.Del(ctx, store.getKey(string(key))).Result()
|
||||
|
||||
if err != nil {
|
||||
return fmt.Errorf("kv delete: %w", err)
|
||||
|
||||
@@ -127,7 +127,7 @@ func (nl *ItemList) WriteName(name string) error {
|
||||
// collect names before name, add them to X
|
||||
namesToX, err := nl.NodeRangeBeforeExclusive(prevNodeReference, name)
|
||||
if err != nil {
|
||||
return nil
|
||||
return err
|
||||
}
|
||||
// delete skiplist reference to old node
|
||||
if _, err := nl.skipList.DeleteByKey(prevNodeReference.Key); err != nil {
|
||||
@@ -136,32 +136,32 @@ func (nl *ItemList) WriteName(name string) error {
|
||||
// add namesToY and name to a new X
|
||||
namesToX = append(namesToX, name)
|
||||
if err := nl.ItemAdd([]byte(namesToX[0]), 0, namesToX...); err != nil {
|
||||
return nil
|
||||
return err
|
||||
}
|
||||
// remove names less than name from current Y
|
||||
if err := nl.NodeDeleteBeforeExclusive(prevNodeReference, name); err != nil {
|
||||
return nil
|
||||
return err
|
||||
}
|
||||
|
||||
// point skip list to current Y
|
||||
if err := nl.ItemAdd(lookupKey, prevNodeReference.ElementPointer); err != nil {
|
||||
return nil
|
||||
return err
|
||||
}
|
||||
return nil
|
||||
} else {
|
||||
// collect names after name, add them to Y
|
||||
namesToY, err := nl.NodeRangeAfterExclusive(prevNodeReference, name)
|
||||
if err != nil {
|
||||
return nil
|
||||
return err
|
||||
}
|
||||
// add namesToY and name to a new Y
|
||||
namesToY = append(namesToY, name)
|
||||
if err := nl.ItemAdd(lookupKey, 0, namesToY...); err != nil {
|
||||
return nil
|
||||
return err
|
||||
}
|
||||
// remove names after name from current X
|
||||
if err := nl.NodeDeleteAfterExclusive(prevNodeReference, name); err != nil {
|
||||
return nil
|
||||
return err
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -35,7 +35,7 @@ func TestCreateAndFind(t *testing.T) {
|
||||
},
|
||||
}
|
||||
|
||||
if err := testFiler.CreateEntry(ctx, entry1, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
|
||||
if err := testFiler.CreateEntry(ctx, entry1, nil, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
|
||||
t.Errorf("create entry %v: %v", entry1.FullPath, err)
|
||||
return
|
||||
}
|
||||
@@ -149,7 +149,7 @@ func TestListDirectoryWithPrefix(t *testing.T) {
|
||||
Gid: 1,
|
||||
},
|
||||
}
|
||||
if err := testFiler.CreateEntry(ctx, entry, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
|
||||
if err := testFiler.CreateEntry(ctx, entry, nil, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
|
||||
t.Fatalf("Failed to create entry %s: %v", fullpath, err)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -35,7 +35,12 @@ func (fh *FileHandle) readFromChunksWithContext(ctx context.Context, buff []byte
|
||||
|
||||
entry := fh.GetEntry()
|
||||
|
||||
if entry.IsInRemoteOnly() {
|
||||
// IsInRemoteOnly inspects entry.Chunks, so take the LockedEntry lock the
|
||||
// async uploader appends under.
|
||||
entry.RLock()
|
||||
remoteOnly := entry.Entry.IsInRemoteOnly()
|
||||
entry.RUnlock()
|
||||
if remoteOnly {
|
||||
glog.V(4).Infof("download remote entry %s", fileFullPath)
|
||||
err := fh.downloadRemoteEntry(entry)
|
||||
if err != nil {
|
||||
@@ -44,10 +49,21 @@ func (fh *FileHandle) readFromChunksWithContext(ctx context.Context, buff []byte
|
||||
}
|
||||
}
|
||||
|
||||
fileSize := int64(entry.Attributes.FileSize)
|
||||
// Snapshot size, inline content, and the chunk list under the LockedEntry
|
||||
// lock. Async upload workers append chunks under this lock (AddChunks), so
|
||||
// reading entry.Chunks / FileSize without it races with the slice
|
||||
// reallocation and can crash in filer.TotalSize. The captured slice headers
|
||||
// stay valid afterwards: append never mutates the old backing array, and
|
||||
// truncate is excluded by the fh.entryLock held for this whole read.
|
||||
entry.RLock()
|
||||
pbEntry := entry.Entry
|
||||
fileSize := int64(pbEntry.Attributes.FileSize)
|
||||
if fileSize == 0 {
|
||||
fileSize = int64(filer.FileSize(entry.GetEntry()))
|
||||
fileSize = int64(filer.FileSize(pbEntry))
|
||||
}
|
||||
content := pbEntry.Content
|
||||
chunks := pbEntry.Chunks
|
||||
entry.RUnlock()
|
||||
|
||||
if fileSize == 0 {
|
||||
glog.V(1).Infof("empty fh %v", fileFullPath)
|
||||
@@ -59,15 +75,15 @@ func (fh *FileHandle) readFromChunksWithContext(ctx context.Context, buff []byte
|
||||
return 0, 0, io.EOF
|
||||
}
|
||||
|
||||
if offset < int64(len(entry.Content)) {
|
||||
totalRead := copy(buff, entry.Content[offset:])
|
||||
if offset < int64(len(content)) {
|
||||
totalRead := copy(buff, content[offset:])
|
||||
glog.V(4).Infof("file handle read cached %s [%d,%d] %d", fileFullPath, offset, offset+int64(totalRead), totalRead)
|
||||
return int64(totalRead), 0, nil
|
||||
}
|
||||
|
||||
// Try RDMA acceleration first if available
|
||||
if fh.wfs.rdmaClient != nil && fh.wfs.option.RdmaEnabled {
|
||||
totalRead, ts, err := fh.tryRDMARead(ctx, fileSize, buff, offset, entry)
|
||||
totalRead, ts, err := fh.tryRDMARead(ctx, fileSize, buff, offset, chunks)
|
||||
if err == nil {
|
||||
glog.V(4).Infof("RDMA read successful for %s [%d,%d] %d", fileFullPath, offset, offset+int64(totalRead), totalRead)
|
||||
return int64(totalRead), ts, nil
|
||||
@@ -79,7 +95,7 @@ func (fh *FileHandle) readFromChunksWithContext(ctx context.Context, buff []byte
|
||||
// Any failure falls through transparently. See design-weed-mount-
|
||||
// peer-chunk-sharing.md §4.3.
|
||||
if fh.wfs.option.PeerEnabled && fh.wfs.peerGrpcServer != nil {
|
||||
totalRead, ts, err := fh.tryPeerRead(ctx, fileSize, buff, offset, entry)
|
||||
totalRead, ts, err := fh.tryPeerRead(ctx, fileSize, buff, offset, chunks)
|
||||
if err == nil {
|
||||
glog.V(4).Infof("peer read successful for %s [%d,%d] %d", fileFullPath, offset, offset+int64(totalRead), totalRead)
|
||||
return int64(totalRead), ts, nil
|
||||
@@ -104,13 +120,13 @@ func (fh *FileHandle) readFromChunksWithContext(ctx context.Context, buff []byte
|
||||
return int64(totalRead), ts, err
|
||||
}
|
||||
|
||||
// tryRDMARead attempts to read file data using RDMA acceleration
|
||||
func (fh *FileHandle) tryRDMARead(ctx context.Context, fileSize int64, buff []byte, offset int64, entry *LockedEntry) (int64, int64, error) {
|
||||
// tryRDMARead attempts to read file data using RDMA acceleration. chunks is a
|
||||
// snapshot captured under the LockedEntry lock by the caller.
|
||||
func (fh *FileHandle) tryRDMARead(ctx context.Context, fileSize int64, buff []byte, offset int64, chunks []*filer_pb.FileChunk) (int64, int64, error) {
|
||||
// For now, we'll try to read the chunks directly using RDMA
|
||||
// This is a simplified approach - in a full implementation, we'd need to
|
||||
// handle chunk boundaries, multiple chunks, etc.
|
||||
|
||||
chunks := entry.GetEntry().Chunks
|
||||
if len(chunks) == 0 {
|
||||
return 0, 0, fmt.Errorf("no chunks available for RDMA read")
|
||||
}
|
||||
|
||||
@@ -53,7 +53,8 @@ const maxPeerFetchChunkBytes = 64 * 1024 * 1024
|
||||
// end-to-end against FileChunk.ETag.
|
||||
// 4. On success, populate chunk_cache and enqueue an announce so
|
||||
// other mounts can discover us as a new holder.
|
||||
func (fh *FileHandle) tryPeerRead(ctx context.Context, fileSize int64, buff []byte, offset int64, entry *LockedEntry) (int64, int64, error) {
|
||||
// chunks is a snapshot captured under the LockedEntry lock by the caller.
|
||||
func (fh *FileHandle) tryPeerRead(ctx context.Context, fileSize int64, buff []byte, offset int64, chunks []*filer_pb.FileChunk) (int64, int64, error) {
|
||||
if fh.wfs.peerRegistrar == nil || fh.wfs.peerConnPool == nil {
|
||||
return 0, 0, fmt.Errorf("peer sharing not configured")
|
||||
}
|
||||
@@ -63,7 +64,7 @@ func (fh *FileHandle) tryPeerRead(ctx context.Context, fileSize int64, buff []by
|
||||
if readStop > fileSize {
|
||||
readStop = fileSize
|
||||
}
|
||||
dataChunks, _, err := filer.ResolveChunkManifest(ctx, fh.wfs.LookupFn(), entry.GetEntry().Chunks, offset, readStop)
|
||||
dataChunks, _, err := filer.ResolveChunkManifest(ctx, fh.wfs.LookupFn(), chunks, offset, readStop)
|
||||
if err != nil {
|
||||
return 0, 0, fmt.Errorf("resolve manifest: %w", err)
|
||||
}
|
||||
|
||||
+14
-2
@@ -15,6 +15,7 @@ import (
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/cluster"
|
||||
"github.com/seaweedfs/seaweedfs/weed/filer"
|
||||
"github.com/seaweedfs/seaweedfs/weed/filer/posixlock"
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/mount/meta_cache"
|
||||
"github.com/seaweedfs/seaweedfs/weed/mount/page_writer"
|
||||
@@ -97,8 +98,10 @@ type Option struct {
|
||||
|
||||
// EnableDistributedLock enables DLM-based write coordination across mounts.
|
||||
// When true, opening a file for write acquires a distributed lock that is
|
||||
// held (with auto-renewal) until the file is closed. Only one mount can
|
||||
// have a file open for writing at a time.
|
||||
// held (with auto-renewal) until the file is closed, so only one mount can
|
||||
// have a file open for writing at a time; POSIX advisory locks (flock/fcntl)
|
||||
// are also routed to the inode's owner filer so they are honored across
|
||||
// mounts. Disabled under writeback cache, which implies single-writer.
|
||||
EnableDistributedLock bool
|
||||
|
||||
// WritebackCache enables async flush on close for improved small file write performance.
|
||||
@@ -138,6 +141,9 @@ type WFS struct {
|
||||
fhLockTable *util.LockTable[FileHandleId]
|
||||
hardLinkLockTable *util.LockTable[string]
|
||||
posixLocks *PosixLockTable
|
||||
posixSid uint64 // this mount's session id, for routed-lock owner identity
|
||||
posixHint *posixLockHint // local fcntl-lock hint for routed mode
|
||||
posixOwn *posixlock.Manager // mirror of locks this mount holds, re-asserted via keepalive
|
||||
rdmaClient *RDMAMountClient
|
||||
peerRegistrar *PeerRegistrar
|
||||
peerDirectory *PeerDirectory
|
||||
@@ -242,6 +248,9 @@ func NewSeaweedFileSystem(option *Option) *WFS {
|
||||
fhLockTable: util.NewLockTable[FileHandleId](),
|
||||
hardLinkLockTable: util.NewLockTable[string](),
|
||||
posixLocks: NewPosixLockTable(),
|
||||
posixSid: randomPosixSid(),
|
||||
posixHint: newPosixLockHint(),
|
||||
posixOwn: posixlock.NewManager(),
|
||||
refreshingDirs: make(map[util.FullPath]struct{}),
|
||||
atimeMap: make(map[uint64]time.Time, 8192),
|
||||
openMtimeCache: make(map[uint64][2]int64, 8192),
|
||||
@@ -505,6 +514,9 @@ func (wfs *WFS) StartBackgroundTasks() error {
|
||||
go wfs.loopFlushDirtyMetadata()
|
||||
go wfs.loopEvictIdleDirCache()
|
||||
go wfs.loopProactiveFlush()
|
||||
if wfs.crossMountLocks() {
|
||||
go wfs.loopRenewPosixLeases()
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -23,10 +23,21 @@ func (wfs *WFS) GetAttr(cancel <-chan struct{}, input *fuse.GetAttrIn, out *fuse
|
||||
}
|
||||
|
||||
inode := input.NodeId
|
||||
path, _, entry, status := wfs.maybeReadEntry(inode)
|
||||
path, fh, entry, status := wfs.maybeReadEntry(inode)
|
||||
if status == fuse.OK {
|
||||
out.AttrValid = wfs.attrValidSec
|
||||
// When an open handle owns the entry, async upload workers append
|
||||
// chunks under the LockedEntry lock; take it for reading so FileSize
|
||||
// does not iterate the chunk slice mid-reallocation. Re-read under the
|
||||
// lock in case SetEntry swapped the pointer since maybeReadEntry.
|
||||
if fh != nil {
|
||||
fh.entry.RLock()
|
||||
entry = fh.entry.Entry
|
||||
}
|
||||
wfs.setAttrByPbEntry(&out.Attr, inode, entry, true)
|
||||
if fh != nil {
|
||||
fh.entry.RUnlock()
|
||||
}
|
||||
wfs.applyInMemoryAtime(&out.Attr, inode)
|
||||
if entry.IsDirectory {
|
||||
wfs.applyInMemoryDirMtime(&out.Attr, inode)
|
||||
@@ -40,7 +51,9 @@ func (wfs *WFS) GetAttr(cancel <-chan struct{}, input *fuse.GetAttrIn, out *fuse
|
||||
out.AttrValid = wfs.attrValidSec
|
||||
// Use shared lock to prevent race with Write operations
|
||||
fhActiveLock := wfs.fhLockTable.AcquireLock("GetAttr", fh.fh, util.SharedLock)
|
||||
wfs.setAttrByPbEntry(&out.Attr, inode, fh.entry.GetEntry(), true)
|
||||
fh.entry.RLock()
|
||||
wfs.setAttrByPbEntry(&out.Attr, inode, fh.entry.Entry, true)
|
||||
fh.entry.RUnlock()
|
||||
wfs.fhLockTable.ReleaseLock(fh.fh, fhActiveLock)
|
||||
wfs.applyInMemoryAtime(&out.Attr, inode)
|
||||
out.Nlink = 0
|
||||
@@ -65,6 +78,15 @@ func (wfs *WFS) SetAttr(cancel <-chan struct{}, input *fuse.SetAttrIn, out *fuse
|
||||
if fh != nil {
|
||||
fh.entryLock.Lock()
|
||||
defer fh.entryLock.Unlock()
|
||||
// entry is the handle's shared LockedEntry.Entry. Async upload workers
|
||||
// mutate its Chunks slice under the LockedEntry lock (AddChunks); hold
|
||||
// that same lock so the truncate and FileSize reads below don't tear
|
||||
// against a concurrent append. Re-read under the lock in case SetEntry
|
||||
// swapped the pointer since maybeReadEntry, so we don't mutate an
|
||||
// orphaned entry and lose the update.
|
||||
fh.entry.Lock()
|
||||
defer fh.entry.Unlock()
|
||||
entry = fh.entry.Entry
|
||||
}
|
||||
|
||||
wormEnforced, wormEnabled := wfs.wormEnforcedForEntry(path, entry)
|
||||
|
||||
@@ -0,0 +1,152 @@
|
||||
package mount
|
||||
|
||||
import (
|
||||
"sync"
|
||||
"testing"
|
||||
|
||||
"github.com/seaweedfs/go-fuse/v2/fuse"
|
||||
"github.com/seaweedfs/seaweedfs/weed/filer"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
)
|
||||
|
||||
// TestAttrChunkRace guards the locking around an open handle's chunk slice.
|
||||
//
|
||||
// With writebackCache, async upload workers append chunks to an open file
|
||||
// handle's shared entry under the LockedEntry lock (FileHandle.AddChunks),
|
||||
// while metadata ops compute the file size by iterating entry.Chunks. SetAttr
|
||||
// and GetAttr used to read that slice without the LockedEntry lock, so a
|
||||
// concurrent append that reallocated the backing array produced a torn slice
|
||||
// read and a nil pointer dereference in filer.TotalSize. Run under -race.
|
||||
func TestAttrChunkRace(t *testing.T) {
|
||||
wfs := &WFS{
|
||||
option: &Option{},
|
||||
inodeToPath: NewInodeToPath(util.FullPath("/"), 0),
|
||||
fhMap: NewFileHandleToInode(),
|
||||
openMtimeCache: make(map[uint64][2]int64, 8),
|
||||
}
|
||||
|
||||
const inode = uint64(42)
|
||||
fullPath := util.FullPath("/dir/sample.txt")
|
||||
wfs.inodeToPath.Lookup(fullPath, 1, false, false, inode, true)
|
||||
|
||||
entry := &filer_pb.Entry{
|
||||
Name: "sample.txt",
|
||||
Attributes: &filer_pb.FuseAttributes{FileMode: 0644},
|
||||
}
|
||||
chunkGroup, err := filer.NewChunkGroup(nil, nil, nil, 1)
|
||||
if err != nil {
|
||||
t.Fatalf("NewChunkGroup: %v", err)
|
||||
}
|
||||
fh := &FileHandle{
|
||||
fh: FileHandleId(1),
|
||||
inode: inode,
|
||||
wfs: wfs,
|
||||
entry: &LockedEntry{Entry: entry},
|
||||
entryChunkGroup: chunkGroup,
|
||||
}
|
||||
wfs.fhMap.inode2fh[inode] = fh
|
||||
wfs.fhMap.fh2inode[fh.fh] = inode
|
||||
|
||||
const iterations = 2000
|
||||
var wg sync.WaitGroup
|
||||
wg.Add(3)
|
||||
|
||||
// Async uploader: append chunks, reallocating the backing array.
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
for i := 0; i < iterations; i++ {
|
||||
fh.AddChunks([]*filer_pb.FileChunk{{FileId: "x", Offset: int64(i), Size: 1}})
|
||||
}
|
||||
}()
|
||||
|
||||
// SetAttr: mtime-only recomputes FileSize by iterating chunks; a shrinking
|
||||
// size takes the truncate path that rewrites entry.Chunks under the lock.
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
for i := 0; i < iterations; i++ {
|
||||
in := &fuse.SetAttrIn{}
|
||||
in.NodeId = inode
|
||||
if i%2 == 0 {
|
||||
in.Valid = fuse.FATTR_MTIME
|
||||
in.Mtime = uint64(i)
|
||||
} else {
|
||||
in.Valid = fuse.FATTR_SIZE
|
||||
in.Size = uint64(i % 8)
|
||||
}
|
||||
var out fuse.AttrOut
|
||||
wfs.SetAttr(nil, in, &out)
|
||||
}
|
||||
}()
|
||||
|
||||
// GetAttr also computes FileSize by iterating chunks.
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
for i := 0; i < iterations; i++ {
|
||||
in := &fuse.GetAttrIn{}
|
||||
in.NodeId = inode
|
||||
var out fuse.AttrOut
|
||||
wfs.GetAttr(nil, in, &out)
|
||||
}
|
||||
}()
|
||||
|
||||
wg.Wait()
|
||||
}
|
||||
|
||||
// TestReadFromChunksRace guards the read path's chunk-slice access. The read
|
||||
// path holds fh.entryLock (which excludes SetAttr) but not the LockedEntry lock
|
||||
// the async uploader appends under, so readFromChunks used to compute FileSize
|
||||
// and walk entry.Chunks while AddChunks reallocated the slice. Run under -race.
|
||||
func TestReadFromChunksRace(t *testing.T) {
|
||||
wfs := &WFS{
|
||||
option: &Option{},
|
||||
inodeToPath: NewInodeToPath(util.FullPath("/"), 0),
|
||||
fhMap: NewFileHandleToInode(),
|
||||
}
|
||||
|
||||
const inode = uint64(42)
|
||||
fullPath := util.FullPath("/dir/sample.txt")
|
||||
wfs.inodeToPath.Lookup(fullPath, 1, false, false, inode, true)
|
||||
|
||||
// FileSize 0 forces readFromChunks down the filer.FileSize(chunks) branch.
|
||||
entry := &filer_pb.Entry{
|
||||
Name: "sample.txt",
|
||||
Attributes: &filer_pb.FuseAttributes{FileMode: 0644},
|
||||
}
|
||||
chunkGroup, err := filer.NewChunkGroup(nil, nil, nil, 1)
|
||||
if err != nil {
|
||||
t.Fatalf("NewChunkGroup: %v", err)
|
||||
}
|
||||
fh := &FileHandle{
|
||||
fh: FileHandleId(1),
|
||||
inode: inode,
|
||||
wfs: wfs,
|
||||
entry: &LockedEntry{Entry: entry},
|
||||
entryChunkGroup: chunkGroup,
|
||||
}
|
||||
wfs.fhMap.inode2fh[inode] = fh
|
||||
wfs.fhMap.fh2inode[fh.fh] = inode
|
||||
|
||||
const iterations = 2000
|
||||
var wg sync.WaitGroup
|
||||
wg.Add(2)
|
||||
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
for i := 0; i < iterations; i++ {
|
||||
fh.AddChunks([]*filer_pb.FileChunk{{FileId: "x", Offset: int64(i), Size: 1}})
|
||||
}
|
||||
}()
|
||||
|
||||
// A read past EOF returns before touching the volume tier, but only after
|
||||
// the racy size/chunk snapshot has run.
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
buff := make([]byte, 16)
|
||||
for i := 0; i < iterations; i++ {
|
||||
fh.readFromChunks(buff, 1<<62)
|
||||
}
|
||||
}()
|
||||
|
||||
wg.Wait()
|
||||
}
|
||||
@@ -148,7 +148,7 @@ func (wfs *WFS) Release(cancel <-chan struct{}, in *fuse.ReleaseIn) {
|
||||
}
|
||||
}
|
||||
if in.ReleaseFlags&fuse.FUSE_RELEASE_FLOCK_UNLOCK != 0 {
|
||||
wfs.posixLocks.ReleaseFlockOwner(in.NodeId, in.LockOwner)
|
||||
wfs.releaseFlockOwner(in.NodeId, in.LockOwner)
|
||||
}
|
||||
wfs.ReleaseHandle(FileHandleId(in.Fh))
|
||||
}
|
||||
|
||||
@@ -10,6 +10,9 @@ import (
|
||||
// If a conflict exists, the conflicting lock is returned in out.
|
||||
// If no conflict, out.Lk.Typ is set to F_UNLCK.
|
||||
func (wfs *WFS) GetLk(cancel <-chan struct{}, in *fuse.LkIn, out *fuse.LkOut) fuse.Status {
|
||||
if wfs.crossMountLocks() {
|
||||
return wfs.routedGetLk(cancel, in, out)
|
||||
}
|
||||
proposed := lockRange{
|
||||
Start: in.Lk.Start,
|
||||
End: in.Lk.End,
|
||||
@@ -25,6 +28,9 @@ func (wfs *WFS) GetLk(cancel <-chan struct{}, in *fuse.LkIn, out *fuse.LkOut) fu
|
||||
// SetLk sets or clears a POSIX lock (non-blocking).
|
||||
// Returns EAGAIN if the lock conflicts with an existing lock from another owner.
|
||||
func (wfs *WFS) SetLk(cancel <-chan struct{}, in *fuse.LkIn) fuse.Status {
|
||||
if wfs.crossMountLocks() {
|
||||
return wfs.routedSetLk(cancel, in)
|
||||
}
|
||||
lk := lockRange{
|
||||
Start: in.Lk.Start,
|
||||
End: in.Lk.End,
|
||||
@@ -39,6 +45,9 @@ func (wfs *WFS) SetLk(cancel <-chan struct{}, in *fuse.LkIn) fuse.Status {
|
||||
// SetLkw sets a POSIX lock (blocking).
|
||||
// Waits until the lock can be acquired or the request is cancelled.
|
||||
func (wfs *WFS) SetLkw(cancel <-chan struct{}, in *fuse.LkIn) fuse.Status {
|
||||
if wfs.crossMountLocks() {
|
||||
return wfs.routedSetLkw(cancel, in)
|
||||
}
|
||||
lk := lockRange{
|
||||
Start: in.Lk.Start,
|
||||
End: in.Lk.End,
|
||||
|
||||
@@ -60,7 +60,7 @@ func (wfs *WFS) Flush(cancel <-chan struct{}, in *fuse.FlushIn) fuse.Status {
|
||||
// If handle is not found, it might have been already released
|
||||
// This is not an error condition for FLUSH
|
||||
if in.LockOwner != 0 {
|
||||
wfs.posixLocks.ReleasePosixOwner(in.NodeId, in.LockOwner)
|
||||
wfs.releasePosixOwner(in.NodeId, in.LockOwner)
|
||||
}
|
||||
return fuse.OK
|
||||
}
|
||||
@@ -69,11 +69,11 @@ func (wfs *WFS) Flush(cancel <-chan struct{}, in *fuse.FlushIn) fuse.Status {
|
||||
// did not hold byte-range locks. Only force the synchronous close path when
|
||||
// this owner actually has POSIX locks to release; otherwise writebackCache
|
||||
// would silently degrade to a blocking flush for ordinary close().
|
||||
hasPosixLocks := wfs.posixLocks.HasPosixOwner(in.NodeId, in.LockOwner)
|
||||
hasPosixLocks := wfs.hasPosixOwner(in.NodeId, in.LockOwner)
|
||||
allowAsync := !hasPosixLocks
|
||||
status := wfs.doFlush(fh, in.Uid, in.Gid, allowAsync)
|
||||
if in.LockOwner != 0 {
|
||||
wfs.posixLocks.ReleasePosixOwner(in.NodeId, in.LockOwner)
|
||||
wfs.releasePosixOwner(in.NodeId, in.LockOwner)
|
||||
}
|
||||
return status
|
||||
}
|
||||
|
||||
@@ -0,0 +1,426 @@
|
||||
package mount
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/binary"
|
||||
"encoding/hex"
|
||||
"math/rand/v2"
|
||||
"sync"
|
||||
"syscall"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/go-fuse/v2/fuse"
|
||||
"github.com/seaweedfs/seaweedfs/weed/filer/posixlock"
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
)
|
||||
|
||||
// randomPosixSid returns a process-stable, cluster-unique-enough session id for
|
||||
// this mount, used to namespace lock owners so the same FUSE owner value on a
|
||||
// different mount never aliases, and to scope lease-based reaping.
|
||||
func randomPosixSid() uint64 {
|
||||
return binary.BigEndian.Uint64(util.RandomBytes(8))
|
||||
}
|
||||
|
||||
// Routed POSIX locking: when -dlm is set, advisory locks are serialized on the
|
||||
// inode's owner filer instead of in this mount's local table, so locks are
|
||||
// honored across mounts. The mount calls its filer and relies on filer-side
|
||||
// forwarding to reach the owner. Blocking (SetLkw) is client-side polling — there
|
||||
// is no server-side wait queue.
|
||||
|
||||
const (
|
||||
posixLockMinBackoff = 5 * time.Millisecond
|
||||
posixLockMaxBackoff = 200 * time.Millisecond
|
||||
posixLockKeyPrefix = "s3.fuse.lock:"
|
||||
// posixLockReleaseTimeout bounds the background unlock/release RPCs. They run
|
||||
// off the syscall path (close/flush) and must not be cancelled by an
|
||||
// interrupt, but a deadline keeps a slow or unreachable filer from blocking
|
||||
// close indefinitely. It also bounds each keepalive RPC.
|
||||
posixLockReleaseTimeout = 5 * time.Second
|
||||
// posixKeepaliveInterval renews the session lease well within the filer's
|
||||
// posixLockSessionTTL (15s) so a live mount is never reaped.
|
||||
posixKeepaliveInterval = 5 * time.Second
|
||||
)
|
||||
|
||||
// posixLockHint records, per inode, which owners this mount has taken fcntl locks
|
||||
// for. Flush consults it to force a synchronous close and route a release without
|
||||
// an RPC on every close(). It is a superset hint: an extra sync flush or a no-op
|
||||
// release RPC is harmless, a missed release is not — so it is added on grant and
|
||||
// cleared on close (which drops the owner's POSIX locks per POSIX semantics).
|
||||
type posixLockHint struct {
|
||||
mu sync.Mutex
|
||||
m map[uint64]map[uint64]struct{}
|
||||
}
|
||||
|
||||
func newPosixLockHint() *posixLockHint {
|
||||
return &posixLockHint{m: make(map[uint64]map[uint64]struct{})}
|
||||
}
|
||||
|
||||
func (h *posixLockHint) add(inode, owner uint64) {
|
||||
h.mu.Lock()
|
||||
defer h.mu.Unlock()
|
||||
owners := h.m[inode]
|
||||
if owners == nil {
|
||||
owners = make(map[uint64]struct{})
|
||||
h.m[inode] = owners
|
||||
}
|
||||
owners[owner] = struct{}{}
|
||||
}
|
||||
|
||||
func (h *posixLockHint) has(inode, owner uint64) bool {
|
||||
h.mu.Lock()
|
||||
defer h.mu.Unlock()
|
||||
_, ok := h.m[inode][owner]
|
||||
return ok
|
||||
}
|
||||
|
||||
func (h *posixLockHint) drop(inode, owner uint64) {
|
||||
h.mu.Lock()
|
||||
defer h.mu.Unlock()
|
||||
if owners := h.m[inode]; owners != nil {
|
||||
delete(owners, owner)
|
||||
if len(owners) == 0 {
|
||||
delete(h.m, inode)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// posixLockKeyForInode resolves a FUSE inode to its cluster-stable lock identity:
|
||||
// the HardLinkId for a hardlinked file (so all names share one owner and table),
|
||||
// else the path. POSIX locks are inode-scoped, so this must never be the FUSE
|
||||
// NodeId (mount-local) for a multi-named inode.
|
||||
func (wfs *WFS) posixLockKeyForInode(inode uint64) (string, bool) {
|
||||
path, status := wfs.inodeToPath.GetPath(inode)
|
||||
if status != fuse.OK {
|
||||
return "", false
|
||||
}
|
||||
if entry, st := wfs.maybeLoadEntry(path); st == fuse.OK && entry != nil && len(entry.HardLinkId) > 0 {
|
||||
return posixLockKeyPrefix + "hl:" + hex.EncodeToString(entry.HardLinkId), true
|
||||
}
|
||||
return posixLockKeyPrefix + string(path), true
|
||||
}
|
||||
|
||||
func posixLockTypeToWire(typ uint32) uint32 {
|
||||
switch typ {
|
||||
case syscall.F_RDLCK:
|
||||
return posixlock.Read
|
||||
case syscall.F_WRLCK:
|
||||
return posixlock.Write
|
||||
default:
|
||||
return posixlock.Unlock
|
||||
}
|
||||
}
|
||||
|
||||
func posixLockTypeFromWire(typ uint32) uint32 {
|
||||
switch typ {
|
||||
case posixlock.Read:
|
||||
return syscall.F_RDLCK
|
||||
case posixlock.Write:
|
||||
return syscall.F_WRLCK
|
||||
default:
|
||||
return syscall.F_UNLCK
|
||||
}
|
||||
}
|
||||
|
||||
func (wfs *WFS) posixRangeFromLkIn(in *fuse.LkIn) posixlock.Range {
|
||||
return posixlock.Range{
|
||||
Start: in.Lk.Start,
|
||||
End: in.Lk.End,
|
||||
Type: posixLockTypeToWire(in.Lk.Typ),
|
||||
Sid: wfs.posixSid,
|
||||
Owner: in.Owner,
|
||||
Pid: in.Lk.Pid,
|
||||
IsFlock: in.LkFlags&fuse.FUSE_LK_FLOCK != 0,
|
||||
}
|
||||
}
|
||||
|
||||
// posixLockContext derives a context from the FUSE cancel channel so an
|
||||
// interrupted lock syscall aborts its in-flight RPC. The caller must call the
|
||||
// returned func to release the watcher goroutine.
|
||||
func posixLockContext(cancel <-chan struct{}) (context.Context, context.CancelFunc) {
|
||||
ctx, cancelFn := context.WithCancel(context.Background())
|
||||
if cancel != nil {
|
||||
go func() {
|
||||
select {
|
||||
case <-cancel:
|
||||
cancelFn()
|
||||
case <-ctx.Done():
|
||||
}
|
||||
}()
|
||||
}
|
||||
return ctx, cancelFn
|
||||
}
|
||||
|
||||
func (wfs *WFS) callPosixLock(ctx context.Context, key string, op filer_pb.PosixLockOp, lk posixlock.Range) (*filer_pb.PosixLockResponse, error) {
|
||||
var resp *filer_pb.PosixLockResponse
|
||||
err := wfs.WithFilerClient(false, func(client filer_pb.SeaweedFilerClient) error {
|
||||
var e error
|
||||
resp, e = client.PosixLock(ctx, &filer_pb.PosixLockRequest{
|
||||
Key: key,
|
||||
Op: op,
|
||||
Lock: &filer_pb.PosixLockRange{
|
||||
Start: lk.Start, End: lk.End, Type: lk.Type,
|
||||
Sid: lk.Sid, Owner: lk.Owner, Pid: lk.Pid, IsFlock: lk.IsFlock,
|
||||
},
|
||||
})
|
||||
return e
|
||||
})
|
||||
return resp, err
|
||||
}
|
||||
|
||||
func (wfs *WFS) routedGetLk(cancel <-chan struct{}, in *fuse.LkIn, out *fuse.LkOut) fuse.Status {
|
||||
key, ok := wfs.posixLockKeyForInode(in.NodeId)
|
||||
if !ok {
|
||||
return fuse.EINVAL
|
||||
}
|
||||
ctx, done := posixLockContext(cancel)
|
||||
defer done()
|
||||
resp, err := wfs.callPosixLock(ctx, key, filer_pb.PosixLockOp_GET_LK, wfs.posixRangeFromLkIn(in))
|
||||
if err != nil {
|
||||
glog.Warningf("routed GetLk %s: %v", key, err)
|
||||
return fuse.EIO
|
||||
}
|
||||
if resp.GetHasConflict() {
|
||||
c := resp.GetConflict()
|
||||
out.Lk.Start, out.Lk.End, out.Lk.Pid = c.GetStart(), c.GetEnd(), c.GetPid()
|
||||
out.Lk.Typ = posixLockTypeFromWire(c.GetType())
|
||||
} else {
|
||||
out.Lk.Typ = syscall.F_UNLCK
|
||||
}
|
||||
return fuse.OK
|
||||
}
|
||||
|
||||
func (wfs *WFS) routedSetLk(cancel <-chan struct{}, in *fuse.LkIn) fuse.Status {
|
||||
key, ok := wfs.posixLockKeyForInode(in.NodeId)
|
||||
if !ok {
|
||||
return fuse.EINVAL
|
||||
}
|
||||
lk := wfs.posixRangeFromLkIn(in)
|
||||
if lk.Type == posixlock.Unlock {
|
||||
return wfs.routedUnlock(key, lk)
|
||||
}
|
||||
ctx, done := posixLockContext(cancel)
|
||||
defer done()
|
||||
resp, err := wfs.callPosixLock(ctx, key, filer_pb.PosixLockOp_TRY_LOCK, lk)
|
||||
if err != nil {
|
||||
glog.Warningf("routed SetLk %s: %v", key, err)
|
||||
return fuse.EIO
|
||||
}
|
||||
if !resp.GetGranted() {
|
||||
return fuse.EAGAIN
|
||||
}
|
||||
wfs.recordPosixGrant(key, in.NodeId, lk)
|
||||
return fuse.OK
|
||||
}
|
||||
|
||||
func (wfs *WFS) routedSetLkw(cancel <-chan struct{}, in *fuse.LkIn) fuse.Status {
|
||||
key, ok := wfs.posixLockKeyForInode(in.NodeId)
|
||||
if !ok {
|
||||
return fuse.EINVAL
|
||||
}
|
||||
lk := wfs.posixRangeFromLkIn(in)
|
||||
if lk.Type == posixlock.Unlock {
|
||||
return wfs.routedUnlock(key, lk)
|
||||
}
|
||||
ctx, done := posixLockContext(cancel)
|
||||
defer done()
|
||||
status := posixPollAcquire(cancel, func() (bool, error) {
|
||||
resp, err := wfs.callPosixLock(ctx, key, filer_pb.PosixLockOp_TRY_LOCK, lk)
|
||||
if err != nil {
|
||||
return false, err
|
||||
}
|
||||
return resp.GetGranted(), nil
|
||||
})
|
||||
if status == fuse.OK {
|
||||
wfs.recordPosixGrant(key, in.NodeId, lk)
|
||||
}
|
||||
return status
|
||||
}
|
||||
|
||||
// recordPosixGrant notes a newly granted lock: the flush hint (fcntl only) and
|
||||
// the own-lock mirror, which the keepalive re-asserts to the key's owner so the
|
||||
// lease is renewed and the owner can rebuild after a takeover or restart.
|
||||
func (wfs *WFS) recordPosixGrant(key string, inode uint64, lk posixlock.Range) {
|
||||
if !lk.IsFlock {
|
||||
wfs.posixHint.add(inode, lk.Owner)
|
||||
}
|
||||
wfs.posixOwn.Track(key, lk)
|
||||
}
|
||||
|
||||
func (wfs *WFS) routedUnlock(key string, lk posixlock.Range) fuse.Status {
|
||||
// Drop the range from the mirror first so a keepalive re-assertion can't
|
||||
// resurrect it; if the RPC below fails, the next re-assertion reconciles the
|
||||
// owner to this (released) state anyway.
|
||||
wfs.posixOwn.Unlock(key, lk)
|
||||
// A release must complete even if the syscall was interrupted; cancelling it
|
||||
// would leak the lock on the owner filer. Bound it so a stuck filer can't
|
||||
// hang close() forever.
|
||||
ctx, cancel := context.WithTimeout(context.Background(), posixLockReleaseTimeout)
|
||||
defer cancel()
|
||||
if _, err := wfs.callPosixLock(ctx, key, filer_pb.PosixLockOp_UNLOCK, lk); err != nil {
|
||||
glog.Warningf("routed unlock %s: %v", key, err)
|
||||
return fuse.EIO
|
||||
}
|
||||
return fuse.OK
|
||||
}
|
||||
|
||||
// posixPollAcquire retries try with capped, jittered backoff until it is granted,
|
||||
// the cancel channel fires (EINTR), or try errors (EIO). This is the client-side
|
||||
// stand-in for a server-side wait queue.
|
||||
func posixPollAcquire(cancel <-chan struct{}, try func() (bool, error)) fuse.Status {
|
||||
backoff := posixLockMinBackoff
|
||||
for {
|
||||
select {
|
||||
case <-cancel:
|
||||
return fuse.EINTR
|
||||
default:
|
||||
}
|
||||
granted, err := try()
|
||||
if err != nil {
|
||||
return fuse.EIO
|
||||
}
|
||||
if granted {
|
||||
return fuse.OK
|
||||
}
|
||||
timer := time.NewTimer(backoff + time.Duration(rand.Int64N(int64(backoff))))
|
||||
select {
|
||||
case <-cancel:
|
||||
timer.Stop()
|
||||
return fuse.EINTR
|
||||
case <-timer.C:
|
||||
}
|
||||
if backoff < posixLockMaxBackoff {
|
||||
if backoff *= 2; backoff > posixLockMaxBackoff {
|
||||
backoff = posixLockMaxBackoff
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func (wfs *WFS) routedReleasePosixOwner(inode, owner uint64) {
|
||||
if !wfs.posixHint.has(inode, owner) {
|
||||
return
|
||||
}
|
||||
key, ok := wfs.posixLockKeyForInode(inode)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
wfs.posixOwn.ReleasePosixOwner(key, wfs.posixSid, owner)
|
||||
ctx, cancel := context.WithTimeout(context.Background(), posixLockReleaseTimeout)
|
||||
defer cancel()
|
||||
if _, err := wfs.callPosixLock(ctx, key, filer_pb.PosixLockOp_RELEASE_POSIX_OWNER, posixlock.Range{Sid: wfs.posixSid, Owner: owner}); err != nil {
|
||||
// Keep the hint so a later flush retries the release; dropping it on a
|
||||
// transient failure would strand the lock until the owner filer's
|
||||
// session-lease reaping expires it.
|
||||
glog.Warningf("routed release posix owner %s: %v", key, err)
|
||||
return
|
||||
}
|
||||
wfs.posixHint.drop(inode, owner)
|
||||
}
|
||||
|
||||
func (wfs *WFS) routedReleaseFlockOwner(inode, owner uint64) {
|
||||
key, ok := wfs.posixLockKeyForInode(inode)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
wfs.posixOwn.ReleaseFlockOwner(key, wfs.posixSid, owner)
|
||||
ctx, cancel := context.WithTimeout(context.Background(), posixLockReleaseTimeout)
|
||||
defer cancel()
|
||||
if _, err := wfs.callPosixLock(ctx, key, filer_pb.PosixLockOp_RELEASE_FLOCK_OWNER, posixlock.Range{Sid: wfs.posixSid, Owner: owner}); err != nil {
|
||||
glog.Warningf("routed release flock owner %s: %v", key, err)
|
||||
}
|
||||
}
|
||||
|
||||
// crossMountLocks reports whether advisory locks are routed to the inode's owner
|
||||
// filer rather than served from the per-mount local table. It tracks -dlm: lock
|
||||
// coordination rides the same switch as whole-file write coordination, and is
|
||||
// off under writeback cache (which implies single-writer, so lockClient is nil).
|
||||
func (wfs *WFS) crossMountLocks() bool {
|
||||
return wfs.lockClient != nil
|
||||
}
|
||||
|
||||
// releasePosixOwner / releaseFlockOwner / hasPosixOwner dispatch to the routed
|
||||
// authority or the local table based on crossMountLocks, keeping the Flush and
|
||||
// Release call sites flag-agnostic.
|
||||
|
||||
func (wfs *WFS) releasePosixOwner(inode, owner uint64) {
|
||||
if wfs.crossMountLocks() {
|
||||
wfs.routedReleasePosixOwner(inode, owner)
|
||||
return
|
||||
}
|
||||
wfs.posixLocks.ReleasePosixOwner(inode, owner)
|
||||
}
|
||||
|
||||
func (wfs *WFS) releaseFlockOwner(inode, owner uint64) {
|
||||
if wfs.crossMountLocks() {
|
||||
wfs.routedReleaseFlockOwner(inode, owner)
|
||||
return
|
||||
}
|
||||
wfs.posixLocks.ReleaseFlockOwner(inode, owner)
|
||||
}
|
||||
|
||||
func (wfs *WFS) hasPosixOwner(inode, owner uint64) bool {
|
||||
if wfs.crossMountLocks() {
|
||||
return owner != 0 && wfs.posixHint.has(inode, owner)
|
||||
}
|
||||
return wfs.posixLocks.HasPosixOwner(inode, owner)
|
||||
}
|
||||
|
||||
// callPosixReassert sends the mount's held locks on key to the key's current
|
||||
// owner filer as a KEEP_ALIVE, which renews the lease and lets the owner rebuild
|
||||
// its in-memory state after a takeover or restart.
|
||||
func (wfs *WFS) callPosixReassert(ctx context.Context, key string, locks []posixlock.Range) error {
|
||||
pbLocks := make([]*filer_pb.PosixLockRange, 0, len(locks))
|
||||
for _, l := range locks {
|
||||
pbLocks = append(pbLocks, &filer_pb.PosixLockRange{
|
||||
Start: l.Start, End: l.End, Type: l.Type,
|
||||
Sid: l.Sid, Owner: l.Owner, Pid: l.Pid, IsFlock: l.IsFlock,
|
||||
})
|
||||
}
|
||||
return wfs.WithFilerClient(false, func(client filer_pb.SeaweedFilerClient) error {
|
||||
_, e := client.PosixLock(ctx, &filer_pb.PosixLockRequest{
|
||||
Key: key,
|
||||
Op: filer_pb.PosixLockOp_KEEP_ALIVE,
|
||||
Lock: &filer_pb.PosixLockRange{Sid: wfs.posixSid},
|
||||
Locks: pbLocks,
|
||||
})
|
||||
return e
|
||||
})
|
||||
}
|
||||
|
||||
// posixKeepaliveConcurrency bounds the parallel keepalive RPCs per tick. A mount
|
||||
// holding locks on many inodes must renew every lease well within the filer TTL;
|
||||
// dispatching them concurrently keeps the round trip from scaling with lock count.
|
||||
const posixKeepaliveConcurrency = 32
|
||||
|
||||
// loopRenewPosixLeases re-asserts this mount's held locks to the owner filer of
|
||||
// every key, which renews the session lease and rebuilds the owner's state after
|
||||
// a ring change or owner restart. A dead mount stops re-asserting and the owners'
|
||||
// sweepers reclaim its locks after the TTL.
|
||||
func (wfs *WFS) loopRenewPosixLeases() {
|
||||
ticker := time.NewTicker(posixKeepaliveInterval)
|
||||
defer ticker.Stop()
|
||||
for range ticker.C {
|
||||
held := wfs.posixOwn.Snapshot()
|
||||
var wg sync.WaitGroup
|
||||
sem := make(chan struct{}, posixKeepaliveConcurrency)
|
||||
for key, locks := range held {
|
||||
wg.Add(1)
|
||||
sem <- struct{}{}
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
defer func() { <-sem }()
|
||||
// Bound each re-assertion so a stuck filer can't block wg.Wait and
|
||||
// stall the whole tick, which would let other keys' leases expire
|
||||
// and get reaped.
|
||||
ctx, cancel := context.WithTimeout(context.Background(), posixLockReleaseTimeout)
|
||||
defer cancel()
|
||||
if err := wfs.callPosixReassert(ctx, key, locks); err != nil {
|
||||
glog.V(2).Infof("posix reassert %s: %v", key, err)
|
||||
}
|
||||
}()
|
||||
}
|
||||
wg.Wait()
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,94 @@
|
||||
package mount
|
||||
|
||||
import (
|
||||
"syscall"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/go-fuse/v2/fuse"
|
||||
"github.com/seaweedfs/seaweedfs/weed/filer/posixlock"
|
||||
)
|
||||
|
||||
func TestPosixLockTypeMapping(t *testing.T) {
|
||||
cases := []struct {
|
||||
sys uint32
|
||||
wire uint32
|
||||
}{
|
||||
{syscall.F_RDLCK, posixlock.Read},
|
||||
{syscall.F_WRLCK, posixlock.Write},
|
||||
{syscall.F_UNLCK, posixlock.Unlock},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if got := posixLockTypeToWire(c.sys); got != c.wire {
|
||||
t.Errorf("toWire(%d) = %d, want %d", c.sys, got, c.wire)
|
||||
}
|
||||
if got := posixLockTypeFromWire(c.wire); got != c.sys {
|
||||
t.Errorf("fromWire(%d) = %d, want %d", c.wire, got, c.sys)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestPosixPollAcquireGrantedImmediately(t *testing.T) {
|
||||
calls := 0
|
||||
st := posixPollAcquire(nil, func() (bool, error) { calls++; return true, nil })
|
||||
if st != fuse.OK || calls != 1 {
|
||||
t.Fatalf("immediate grant: status=%v calls=%d", st, calls)
|
||||
}
|
||||
}
|
||||
|
||||
func TestPosixPollAcquireRetriesThenGrants(t *testing.T) {
|
||||
calls := 0
|
||||
st := posixPollAcquire(nil, func() (bool, error) {
|
||||
calls++
|
||||
return calls >= 3, nil
|
||||
})
|
||||
if st != fuse.OK || calls != 3 {
|
||||
t.Fatalf("retry then grant: status=%v calls=%d", st, calls)
|
||||
}
|
||||
}
|
||||
|
||||
func TestPosixPollAcquireError(t *testing.T) {
|
||||
if st := posixPollAcquire(nil, func() (bool, error) { return false, syscall.EIO }); st != fuse.EIO {
|
||||
t.Fatalf("error should map to EIO, got %v", st)
|
||||
}
|
||||
}
|
||||
|
||||
func TestPosixPollAcquireCancel(t *testing.T) {
|
||||
cancel := make(chan struct{})
|
||||
close(cancel)
|
||||
done := make(chan fuse.Status, 1)
|
||||
go func() {
|
||||
done <- posixPollAcquire(cancel, func() (bool, error) { return false, nil })
|
||||
}()
|
||||
select {
|
||||
case st := <-done:
|
||||
if st != fuse.EINTR {
|
||||
t.Fatalf("cancel should map to EINTR, got %v", st)
|
||||
}
|
||||
case <-time.After(2 * time.Second):
|
||||
t.Fatal("poll did not return on cancel")
|
||||
}
|
||||
}
|
||||
|
||||
func TestPosixLockHint(t *testing.T) {
|
||||
h := newPosixLockHint()
|
||||
if h.has(1, 2) {
|
||||
t.Fatal("empty hint should not report a lock")
|
||||
}
|
||||
h.add(1, 2)
|
||||
h.add(1, 3)
|
||||
if !h.has(1, 2) || !h.has(1, 3) {
|
||||
t.Fatal("added owners should be reported")
|
||||
}
|
||||
h.drop(1, 2)
|
||||
if h.has(1, 2) {
|
||||
t.Fatal("dropped owner should be gone")
|
||||
}
|
||||
if !h.has(1, 3) {
|
||||
t.Fatal("sibling owner should remain")
|
||||
}
|
||||
h.drop(1, 3)
|
||||
if _, ok := h.m[1]; ok {
|
||||
t.Fatal("inode entry should be removed when its last owner drops")
|
||||
}
|
||||
}
|
||||
@@ -31,6 +31,15 @@ service SeaweedFiler {
|
||||
rpc DeleteEntry (DeleteEntryRequest) returns (DeleteEntryResponse) {
|
||||
}
|
||||
|
||||
rpc ObjectTransaction (ObjectTransactionRequest) returns (ObjectTransactionResponse) {
|
||||
}
|
||||
|
||||
rpc ObjectTransactionBatch (ObjectTransactionBatchRequest) returns (ObjectTransactionBatchResponse) {
|
||||
}
|
||||
|
||||
rpc PosixLock (PosixLockRequest) returns (PosixLockResponse) {
|
||||
}
|
||||
|
||||
rpc AtomicRenameEntry (AtomicRenameEntryRequest) returns (AtomicRenameEntryResponse) {
|
||||
}
|
||||
rpc StreamRenameEntry (StreamRenameEntryRequest) returns (stream StreamRenameEntryResponse) {
|
||||
@@ -222,6 +231,56 @@ message CreateEntryRequest {
|
||||
bool is_from_other_cluster = 4;
|
||||
repeated int32 signatures = 5;
|
||||
bool skip_check_parent_directory = 6;
|
||||
// Optional precondition evaluated against the current entry atomically with
|
||||
// the write, under the filer's per-path lock. The caller must route the
|
||||
// key's writes to this entry's owner filer for the check to be authoritative.
|
||||
WriteCondition condition = 7;
|
||||
}
|
||||
|
||||
// WriteCondition is the precondition the filer evaluates against the existing
|
||||
// entry before writing, under the per-path lock. A failed condition returns
|
||||
// FilerError PRECONDITION_FAILED. The client maps request semantics (e.g. RFC
|
||||
// 7232) to clauses; the filer just compares.
|
||||
//
|
||||
// A condition is a list of clauses that ALL must hold (logical AND). One clause
|
||||
// is the common case; several express what a single comparison cannot: an ETag
|
||||
// set (If-Match / If-None-Match with multiple values), weak-ETag comparison, and
|
||||
// compound conditions (e.g. If-Match + If-Unmodified-Since together).
|
||||
message WriteCondition {
|
||||
enum Kind {
|
||||
NONE = 0; // unconditional
|
||||
IF_NOT_EXISTS = 1; // fail if the entry exists (If-None-Match: *)
|
||||
IF_EXISTS = 2; // fail if the entry is absent (If-Match: *)
|
||||
IF_ETAG_MATCH = 3; // fail if absent or etag matches none of the set (If-Match)
|
||||
IF_ETAG_NOT_MATCH = 4; // fail if present and etag matches any of the set (If-None-Match)
|
||||
IF_UNMODIFIED_SINCE = 5; // fail if present and mtime > unix_time
|
||||
IF_MODIFIED_SINCE = 6; // fail if present and mtime <= unix_time
|
||||
IF_EXTENDED_NOT_EQUAL = 7; // fail if present and extended[ext_key] == ext_value
|
||||
IF_EXTENDED_TIME_ELAPSED = 8; // fail if present and extended[ext_key] (unix seconds) is in the future
|
||||
}
|
||||
// Clause is one primitive comparison. IF_ETAG_MATCH holds when the current
|
||||
// entry's ETag equals any value in etags; IF_ETAG_NOT_MATCH holds when it
|
||||
// equals none. allow_weak permits weak-comparison (ignoring the W/ prefix).
|
||||
//
|
||||
// The IF_EXTENDED_* kinds are generic guards on an extended attribute, used
|
||||
// to enforce object-lock without teaching the filer S3 semantics:
|
||||
// IF_EXTENDED_NOT_EQUAL expresses a legal hold (block while a key equals a
|
||||
// value), and IF_EXTENDED_TIME_ELAPSED expresses retention (block while a
|
||||
// stored unix-second deadline is in the future, compared to the filer's
|
||||
// clock). The caller composes these and, for governance-bypass, simply omits
|
||||
// the retention clause when the bypass is authorized — the filer makes no
|
||||
// authorization decision.
|
||||
message Clause {
|
||||
Kind kind = 1;
|
||||
repeated string etags = 2; // ETag set for IF_ETAG_* kinds
|
||||
int64 unix_time = 3; // bound (unix seconds) for IF_*_SINCE kinds
|
||||
bool allow_weak = 4; // compare ETags ignoring the weak (W/) marker
|
||||
string ext_key = 5; // extended attribute name for IF_EXTENDED_* kinds
|
||||
string ext_value = 6; // blocking value for IF_EXTENDED_NOT_EQUAL
|
||||
string gate_key = 7; // IF_EXTENDED_TIME_ELAPSED: only enforce when extended[gate_key] == gate_value
|
||||
string gate_value = 8; // gate value (e.g. retention mode COMPLIANCE for governance bypass)
|
||||
}
|
||||
repeated Clause clauses = 1; // all must hold (logical AND)
|
||||
}
|
||||
|
||||
// Structured error codes for filer entry operations.
|
||||
@@ -233,6 +292,138 @@ enum FilerError {
|
||||
EXISTING_IS_DIRECTORY = 3; // cannot overwrite directory with file
|
||||
EXISTING_IS_FILE = 4; // cannot overwrite file with directory
|
||||
ENTRY_ALREADY_EXISTS = 5; // O_EXCL and entry already exists
|
||||
PRECONDITION_FAILED = 6; // WriteCondition not satisfied
|
||||
}
|
||||
|
||||
// ObjectMutation is one entry-level change applied by ObjectTransaction. All
|
||||
// mutations of a transaction run under a single per-path lock (the request's
|
||||
// lock_key) and in order, so the gateway can describe a multi-entry object
|
||||
// operation as one request instead of holding a distributed lock across
|
||||
// several RPCs. Data-bearing writes (entries with chunks) should be written
|
||||
// before the transaction; mutations here are metadata-scoped.
|
||||
message ObjectMutation {
|
||||
enum Type {
|
||||
PUT = 0; // create or replace the entry (entry field)
|
||||
DELETE = 1; // delete the entry at directory/name (no error if absent)
|
||||
PATCH_EXTENDED = 2; // merge set_extended / remove delete_extended on the entry
|
||||
RECOMPUTE_LATEST = 3; // scan a directory and re-point a parent entry (recompute)
|
||||
}
|
||||
Type type = 1;
|
||||
string directory = 2;
|
||||
string name = 3; // entry name for DELETE / PATCH_EXTENDED / RECOMPUTE_LATEST (the pointer entry)
|
||||
Entry entry = 4; // full entry for PUT
|
||||
map<string, bytes> set_extended = 5; // PATCH_EXTENDED: keys to set
|
||||
repeated string delete_extended = 6; // PATCH_EXTENDED: keys to remove
|
||||
bool is_delete_data = 7; // DELETE: also delete chunk data
|
||||
bool is_recursive = 8; // DELETE: recurse into a directory
|
||||
Recompute recompute = 9; // RECOMPUTE_LATEST parameters
|
||||
bool set_content = 10; // PATCH_EXTENDED: replace Entry.content with content
|
||||
bytes content = 11; // PATCH_EXTENDED: new Entry.content when set_content
|
||||
bool touch_mtime = 12; // PATCH_EXTENDED: set the entry's Mtime to now (e.g. a metadata-replace copy)
|
||||
}
|
||||
|
||||
// Recompute re-derives a pointer entry (directory/name on the mutation) from the
|
||||
// current contents of a scanned directory, atomically under the transaction's
|
||||
// lock. It is mechanical: the filer picks the child that sorts first or last by
|
||||
// name and copies the requested fields into the pointer; it has no knowledge of
|
||||
// what the entries mean. The caller (which does know the versioning scheme)
|
||||
// supplies the sort direction and the key mappings. This covers re-pointing the
|
||||
// latest version after a specific version is deleted, where the scan must run
|
||||
// under the lock.
|
||||
message Recompute {
|
||||
string scan_dir = 1; // directory whose direct children are scanned
|
||||
bool descending = 2; // pick the child that sorts last by name (else first)
|
||||
map<string, string> copy_extended = 3; // pointer extended key -> source extended key on the chosen child
|
||||
string name_to_key = 4; // if set, store the chosen child's name under this pointer key
|
||||
string size_to_key = 5; // if set, store the chosen child's FileSize (decimal) under this pointer key
|
||||
string mtime_to_key = 6; // if set, store the chosen child's Mtime (decimal) under this pointer key
|
||||
string demote_key = 7; // if set, stamp demote_value on the prior name_to_key target when it changes
|
||||
bytes demote_value = 8; // value for demote_key
|
||||
string exclude_name = 9; // if set, skip this child when scanning (e.g. a version about to be deleted)
|
||||
}
|
||||
|
||||
// ObjectTransactionRequest applies an ordered list of mutations atomically with
|
||||
// respect to other writers of the same object, by holding the filer's per-path
|
||||
// lock on lock_key for the whole transaction. The optional condition is checked
|
||||
// first, against condition_key when set, else lock_key. Callers set route_key to
|
||||
// the object's stable owner ring key; a filer that is not the owner forwards the
|
||||
// transaction one hop to the owner, so a stale ring view is tolerated.
|
||||
message ObjectTransactionRequest {
|
||||
string lock_key = 1; // object path to lock and to evaluate the condition against
|
||||
WriteCondition condition = 2; // optional precondition, checked under the lock
|
||||
repeated ObjectMutation mutations = 3;
|
||||
bool is_from_other_cluster = 4;
|
||||
repeated int32 signatures = 5;
|
||||
string condition_key = 6; // if set, evaluate the condition against this entry instead of lock_key (still locking lock_key)
|
||||
string route_key = 7; // ring key identifying the owner filer; a non-owner forwards the whole transaction to it
|
||||
bool is_moved = 8; // set on a forwarded transaction so the receiver applies it locally instead of forwarding again
|
||||
}
|
||||
|
||||
message ObjectTransactionResponse {
|
||||
string error = 1;
|
||||
FilerError error_code = 2;
|
||||
}
|
||||
|
||||
// PosixLockRange is one advisory byte-range lock. Owner identity is (sid, owner):
|
||||
// sid is the mount session, owner the FUSE lock owner within it, so owners from
|
||||
// different mounts never alias. end is inclusive (max uint64 = to EOF); is_flock
|
||||
// separates the flock and fcntl namespaces, which never conflict.
|
||||
message PosixLockRange {
|
||||
uint64 start = 1;
|
||||
uint64 end = 2;
|
||||
uint32 type = 3; // 1=read, 2=write, 3=unlock
|
||||
uint64 sid = 4;
|
||||
uint64 owner = 5;
|
||||
uint32 pid = 6; // holder pid, for get_lk reporting only
|
||||
bool is_flock = 7;
|
||||
}
|
||||
|
||||
// PosixLock routes an advisory lock operation to the inode's owner filer, which
|
||||
// holds the authoritative in-memory lock table. key is the inode identity ring
|
||||
// key (the file path, or hl:<HardLinkId> for a hardlink) used both to resolve the
|
||||
// owner and to index the table. A non-owner filer forwards the request one hop;
|
||||
// is_moved bounds it so a stale ring view cannot loop.
|
||||
message PosixLockRequest {
|
||||
string key = 1;
|
||||
bool is_moved = 2;
|
||||
PosixLockOp op = 3;
|
||||
PosixLockRange lock = 4;
|
||||
// locks carries the full set a mount holds on key for a KEEP_ALIVE
|
||||
// re-assertion, so the current owner filer can rebuild its in-memory state
|
||||
// after an ownership change or restart. lock.sid identifies the session.
|
||||
repeated PosixLockRange locks = 5;
|
||||
// cooling_probe marks a dual-read a new owner sends to the previous owner
|
||||
// during a ring change, so the previous owner answers from local state
|
||||
// without itself cooling-off (no recursion).
|
||||
bool cooling_probe = 6;
|
||||
}
|
||||
|
||||
enum PosixLockOp {
|
||||
TRY_LOCK = 0; // grant lock or report conflict (non-blocking)
|
||||
UNLOCK = 1; // release lock's owner's locks over its range
|
||||
GET_LK = 2; // report a conflicting lock, if any
|
||||
RELEASE_POSIX_OWNER = 3; // drop the owner's fcntl locks (flush-time)
|
||||
RELEASE_FLOCK_OWNER = 4; // drop the owner's flock locks (release-time)
|
||||
KEEP_ALIVE = 5; // renew the session's lease on this owner (lock.sid)
|
||||
}
|
||||
|
||||
message PosixLockResponse {
|
||||
bool granted = 1; // for TRY_LOCK: whether the lock was granted
|
||||
bool has_conflict = 2; // whether conflict is populated
|
||||
PosixLockRange conflict = 3; // the blocking lock (TRY_LOCK conflict / GET_LK result)
|
||||
}
|
||||
|
||||
// ObjectTransactionBatch applies several object transactions in one round trip,
|
||||
// each under its own per-path lock and independent of the others (no cross-key
|
||||
// atomicity). A caller groups keys that route to the same owner filer and sends
|
||||
// one batch per owner, e.g. for a multi-object delete. Each response is parallel
|
||||
// to its request.
|
||||
message ObjectTransactionBatchRequest {
|
||||
repeated ObjectTransactionRequest transactions = 1;
|
||||
}
|
||||
|
||||
message ObjectTransactionBatchResponse {
|
||||
repeated ObjectTransactionResponse responses = 1;
|
||||
}
|
||||
|
||||
message CreateEntryResponse {
|
||||
@@ -421,6 +612,7 @@ message SubscribeMetadataRequest {
|
||||
repeated string directories = 10; // exact directory to watch
|
||||
bool client_supports_batching = 11; // client can unpack SubscribeMetadataResponse.events
|
||||
bool client_supports_metadata_chunks = 12; // client can read log file chunks from volume servers
|
||||
bool client_supports_idle_heartbeat = 13; // server may send empty responses carrying the current time while the client is caught up
|
||||
}
|
||||
message SubscribeMetadataResponse {
|
||||
string directory = 1;
|
||||
|
||||
+1688
-410
File diff suppressed because it is too large
Load Diff
@@ -1,7 +1,7 @@
|
||||
// Code generated by protoc-gen-go-grpc. DO NOT EDIT.
|
||||
// versions:
|
||||
// - protoc-gen-go-grpc v1.6.2
|
||||
// - protoc v7.34.1
|
||||
// - protoc v6.33.4
|
||||
// source: filer.proto
|
||||
|
||||
package filer_pb
|
||||
@@ -26,6 +26,9 @@ const (
|
||||
SeaweedFiler_TouchAccessTime_FullMethodName = "/filer_pb.SeaweedFiler/TouchAccessTime"
|
||||
SeaweedFiler_AppendToEntry_FullMethodName = "/filer_pb.SeaweedFiler/AppendToEntry"
|
||||
SeaweedFiler_DeleteEntry_FullMethodName = "/filer_pb.SeaweedFiler/DeleteEntry"
|
||||
SeaweedFiler_ObjectTransaction_FullMethodName = "/filer_pb.SeaweedFiler/ObjectTransaction"
|
||||
SeaweedFiler_ObjectTransactionBatch_FullMethodName = "/filer_pb.SeaweedFiler/ObjectTransactionBatch"
|
||||
SeaweedFiler_PosixLock_FullMethodName = "/filer_pb.SeaweedFiler/PosixLock"
|
||||
SeaweedFiler_AtomicRenameEntry_FullMethodName = "/filer_pb.SeaweedFiler/AtomicRenameEntry"
|
||||
SeaweedFiler_StreamRenameEntry_FullMethodName = "/filer_pb.SeaweedFiler/StreamRenameEntry"
|
||||
SeaweedFiler_StreamMutateEntry_FullMethodName = "/filer_pb.SeaweedFiler/StreamMutateEntry"
|
||||
@@ -62,6 +65,9 @@ type SeaweedFilerClient interface {
|
||||
TouchAccessTime(ctx context.Context, in *TouchAccessTimeRequest, opts ...grpc.CallOption) (*TouchAccessTimeResponse, error)
|
||||
AppendToEntry(ctx context.Context, in *AppendToEntryRequest, opts ...grpc.CallOption) (*AppendToEntryResponse, error)
|
||||
DeleteEntry(ctx context.Context, in *DeleteEntryRequest, opts ...grpc.CallOption) (*DeleteEntryResponse, error)
|
||||
ObjectTransaction(ctx context.Context, in *ObjectTransactionRequest, opts ...grpc.CallOption) (*ObjectTransactionResponse, error)
|
||||
ObjectTransactionBatch(ctx context.Context, in *ObjectTransactionBatchRequest, opts ...grpc.CallOption) (*ObjectTransactionBatchResponse, error)
|
||||
PosixLock(ctx context.Context, in *PosixLockRequest, opts ...grpc.CallOption) (*PosixLockResponse, error)
|
||||
AtomicRenameEntry(ctx context.Context, in *AtomicRenameEntryRequest, opts ...grpc.CallOption) (*AtomicRenameEntryResponse, error)
|
||||
StreamRenameEntry(ctx context.Context, in *StreamRenameEntryRequest, opts ...grpc.CallOption) (grpc.ServerStreamingClient[StreamRenameEntryResponse], error)
|
||||
StreamMutateEntry(ctx context.Context, opts ...grpc.CallOption) (grpc.BidiStreamingClient[StreamMutateEntryRequest, StreamMutateEntryResponse], error)
|
||||
@@ -177,6 +183,36 @@ func (c *seaweedFilerClient) DeleteEntry(ctx context.Context, in *DeleteEntryReq
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func (c *seaweedFilerClient) ObjectTransaction(ctx context.Context, in *ObjectTransactionRequest, opts ...grpc.CallOption) (*ObjectTransactionResponse, error) {
|
||||
cOpts := append([]grpc.CallOption{grpc.StaticMethod()}, opts...)
|
||||
out := new(ObjectTransactionResponse)
|
||||
err := c.cc.Invoke(ctx, SeaweedFiler_ObjectTransaction_FullMethodName, in, out, cOpts...)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func (c *seaweedFilerClient) ObjectTransactionBatch(ctx context.Context, in *ObjectTransactionBatchRequest, opts ...grpc.CallOption) (*ObjectTransactionBatchResponse, error) {
|
||||
cOpts := append([]grpc.CallOption{grpc.StaticMethod()}, opts...)
|
||||
out := new(ObjectTransactionBatchResponse)
|
||||
err := c.cc.Invoke(ctx, SeaweedFiler_ObjectTransactionBatch_FullMethodName, in, out, cOpts...)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func (c *seaweedFilerClient) PosixLock(ctx context.Context, in *PosixLockRequest, opts ...grpc.CallOption) (*PosixLockResponse, error) {
|
||||
cOpts := append([]grpc.CallOption{grpc.StaticMethod()}, opts...)
|
||||
out := new(PosixLockResponse)
|
||||
err := c.cc.Invoke(ctx, SeaweedFiler_PosixLock_FullMethodName, in, out, cOpts...)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func (c *seaweedFilerClient) AtomicRenameEntry(ctx context.Context, in *AtomicRenameEntryRequest, opts ...grpc.CallOption) (*AtomicRenameEntryResponse, error) {
|
||||
cOpts := append([]grpc.CallOption{grpc.StaticMethod()}, opts...)
|
||||
out := new(AtomicRenameEntryResponse)
|
||||
@@ -457,6 +493,9 @@ type SeaweedFilerServer interface {
|
||||
TouchAccessTime(context.Context, *TouchAccessTimeRequest) (*TouchAccessTimeResponse, error)
|
||||
AppendToEntry(context.Context, *AppendToEntryRequest) (*AppendToEntryResponse, error)
|
||||
DeleteEntry(context.Context, *DeleteEntryRequest) (*DeleteEntryResponse, error)
|
||||
ObjectTransaction(context.Context, *ObjectTransactionRequest) (*ObjectTransactionResponse, error)
|
||||
ObjectTransactionBatch(context.Context, *ObjectTransactionBatchRequest) (*ObjectTransactionBatchResponse, error)
|
||||
PosixLock(context.Context, *PosixLockRequest) (*PosixLockResponse, error)
|
||||
AtomicRenameEntry(context.Context, *AtomicRenameEntryRequest) (*AtomicRenameEntryResponse, error)
|
||||
StreamRenameEntry(*StreamRenameEntryRequest, grpc.ServerStreamingServer[StreamRenameEntryResponse]) error
|
||||
StreamMutateEntry(grpc.BidiStreamingServer[StreamMutateEntryRequest, StreamMutateEntryResponse]) error
|
||||
@@ -514,6 +553,15 @@ func (UnimplementedSeaweedFilerServer) AppendToEntry(context.Context, *AppendToE
|
||||
func (UnimplementedSeaweedFilerServer) DeleteEntry(context.Context, *DeleteEntryRequest) (*DeleteEntryResponse, error) {
|
||||
return nil, status.Error(codes.Unimplemented, "method DeleteEntry not implemented")
|
||||
}
|
||||
func (UnimplementedSeaweedFilerServer) ObjectTransaction(context.Context, *ObjectTransactionRequest) (*ObjectTransactionResponse, error) {
|
||||
return nil, status.Error(codes.Unimplemented, "method ObjectTransaction not implemented")
|
||||
}
|
||||
func (UnimplementedSeaweedFilerServer) ObjectTransactionBatch(context.Context, *ObjectTransactionBatchRequest) (*ObjectTransactionBatchResponse, error) {
|
||||
return nil, status.Error(codes.Unimplemented, "method ObjectTransactionBatch not implemented")
|
||||
}
|
||||
func (UnimplementedSeaweedFilerServer) PosixLock(context.Context, *PosixLockRequest) (*PosixLockResponse, error) {
|
||||
return nil, status.Error(codes.Unimplemented, "method PosixLock not implemented")
|
||||
}
|
||||
func (UnimplementedSeaweedFilerServer) AtomicRenameEntry(context.Context, *AtomicRenameEntryRequest) (*AtomicRenameEntryResponse, error) {
|
||||
return nil, status.Error(codes.Unimplemented, "method AtomicRenameEntry not implemented")
|
||||
}
|
||||
@@ -723,6 +771,60 @@ func _SeaweedFiler_DeleteEntry_Handler(srv interface{}, ctx context.Context, dec
|
||||
return interceptor(ctx, in, info, handler)
|
||||
}
|
||||
|
||||
func _SeaweedFiler_ObjectTransaction_Handler(srv interface{}, ctx context.Context, dec func(interface{}) error, interceptor grpc.UnaryServerInterceptor) (interface{}, error) {
|
||||
in := new(ObjectTransactionRequest)
|
||||
if err := dec(in); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if interceptor == nil {
|
||||
return srv.(SeaweedFilerServer).ObjectTransaction(ctx, in)
|
||||
}
|
||||
info := &grpc.UnaryServerInfo{
|
||||
Server: srv,
|
||||
FullMethod: SeaweedFiler_ObjectTransaction_FullMethodName,
|
||||
}
|
||||
handler := func(ctx context.Context, req interface{}) (interface{}, error) {
|
||||
return srv.(SeaweedFilerServer).ObjectTransaction(ctx, req.(*ObjectTransactionRequest))
|
||||
}
|
||||
return interceptor(ctx, in, info, handler)
|
||||
}
|
||||
|
||||
func _SeaweedFiler_ObjectTransactionBatch_Handler(srv interface{}, ctx context.Context, dec func(interface{}) error, interceptor grpc.UnaryServerInterceptor) (interface{}, error) {
|
||||
in := new(ObjectTransactionBatchRequest)
|
||||
if err := dec(in); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if interceptor == nil {
|
||||
return srv.(SeaweedFilerServer).ObjectTransactionBatch(ctx, in)
|
||||
}
|
||||
info := &grpc.UnaryServerInfo{
|
||||
Server: srv,
|
||||
FullMethod: SeaweedFiler_ObjectTransactionBatch_FullMethodName,
|
||||
}
|
||||
handler := func(ctx context.Context, req interface{}) (interface{}, error) {
|
||||
return srv.(SeaweedFilerServer).ObjectTransactionBatch(ctx, req.(*ObjectTransactionBatchRequest))
|
||||
}
|
||||
return interceptor(ctx, in, info, handler)
|
||||
}
|
||||
|
||||
func _SeaweedFiler_PosixLock_Handler(srv interface{}, ctx context.Context, dec func(interface{}) error, interceptor grpc.UnaryServerInterceptor) (interface{}, error) {
|
||||
in := new(PosixLockRequest)
|
||||
if err := dec(in); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if interceptor == nil {
|
||||
return srv.(SeaweedFilerServer).PosixLock(ctx, in)
|
||||
}
|
||||
info := &grpc.UnaryServerInfo{
|
||||
Server: srv,
|
||||
FullMethod: SeaweedFiler_PosixLock_FullMethodName,
|
||||
}
|
||||
handler := func(ctx context.Context, req interface{}) (interface{}, error) {
|
||||
return srv.(SeaweedFilerServer).PosixLock(ctx, req.(*PosixLockRequest))
|
||||
}
|
||||
return interceptor(ctx, in, info, handler)
|
||||
}
|
||||
|
||||
func _SeaweedFiler_AtomicRenameEntry_Handler(srv interface{}, ctx context.Context, dec func(interface{}) error, interceptor grpc.UnaryServerInterceptor) (interface{}, error) {
|
||||
in := new(AtomicRenameEntryRequest)
|
||||
if err := dec(in); err != nil {
|
||||
@@ -1129,6 +1231,18 @@ var SeaweedFiler_ServiceDesc = grpc.ServiceDesc{
|
||||
MethodName: "DeleteEntry",
|
||||
Handler: _SeaweedFiler_DeleteEntry_Handler,
|
||||
},
|
||||
{
|
||||
MethodName: "ObjectTransaction",
|
||||
Handler: _SeaweedFiler_ObjectTransaction_Handler,
|
||||
},
|
||||
{
|
||||
MethodName: "ObjectTransactionBatch",
|
||||
Handler: _SeaweedFiler_ObjectTransactionBatch_Handler,
|
||||
},
|
||||
{
|
||||
MethodName: "PosixLock",
|
||||
Handler: _SeaweedFiler_PosixLock_Handler,
|
||||
},
|
||||
{
|
||||
MethodName: "AtomicRenameEntry",
|
||||
Handler: _SeaweedFiler_AtomicRenameEntry_Handler,
|
||||
|
||||
@@ -38,6 +38,12 @@ type MetadataFollowOption struct {
|
||||
// the server sends log file chunk fids instead of streaming events,
|
||||
// and the client reads directly from volume servers.
|
||||
LogFileReaderFn LogFileReaderFn
|
||||
// OnIdleHeartbeat, when non-nil, opts in to idle heartbeats: while the
|
||||
// subscriber is caught up the server periodically sends an empty response
|
||||
// carrying the current time, and this is called with that timestamp. It is
|
||||
// a freshness signal only and does not advance StartTsNs, so the resume
|
||||
// checkpoint stays on the last real event.
|
||||
OnIdleHeartbeat func(tsNs int64)
|
||||
}
|
||||
|
||||
type ProcessMetadataFunc func(resp *filer_pb.SubscribeMetadataResponse) error
|
||||
@@ -77,6 +83,7 @@ func makeSubscribeMetadataFunc(option *MetadataFollowOption, processEventFn Proc
|
||||
UntilNs: option.StopTsNs,
|
||||
ClientSupportsBatching: true,
|
||||
ClientSupportsMetadataChunks: option.LogFileReaderFn != nil,
|
||||
ClientSupportsIdleHeartbeat: option.OnIdleHeartbeat != nil,
|
||||
})
|
||||
if err != nil {
|
||||
return fmt.Errorf("subscribe: %w", err)
|
||||
@@ -138,6 +145,17 @@ func makeSubscribeMetadataFunc(option *MetadataFollowOption, processEventFn Proc
|
||||
pendingRefs = nil
|
||||
}
|
||||
|
||||
// Idle heartbeat: the source is caught up and has no new events, so
|
||||
// it sends an empty response carrying the current time. Surface it as
|
||||
// a freshness signal but leave option.StartTsNs untouched so a restart
|
||||
// still resumes from the last real event.
|
||||
if resp.EventNotification == nil && len(resp.Events) == 0 && resp.TsNs > 0 {
|
||||
if option.OnIdleHeartbeat != nil {
|
||||
option.OnIdleHeartbeat(resp.TsNs)
|
||||
}
|
||||
continue
|
||||
}
|
||||
|
||||
// Process the first event (always present in top-level fields)
|
||||
if resp.EventNotification != nil {
|
||||
if err := processEventFn(resp); err != nil {
|
||||
|
||||
@@ -348,7 +348,9 @@ message JobResult {
|
||||
|
||||
message ClusterContext {
|
||||
repeated string master_grpc_addresses = 1;
|
||||
repeated string filer_grpc_addresses = 2;
|
||||
// Filers in pb.ServerAddress form (host:httpPort.grpcPort). Consumers
|
||||
// convert to a gRPC or HTTP address as needed; see weed/pb/server_address.go.
|
||||
repeated string filer_addresses = 2;
|
||||
repeated string volume_grpc_addresses = 3;
|
||||
map<string, string> metadata = 4;
|
||||
repeated string s3_grpc_addresses = 5;
|
||||
|
||||
@@ -3564,10 +3564,12 @@ func (x *JobResult) GetSummary() string {
|
||||
type ClusterContext struct {
|
||||
state protoimpl.MessageState `protogen:"open.v1"`
|
||||
MasterGrpcAddresses []string `protobuf:"bytes,1,rep,name=master_grpc_addresses,json=masterGrpcAddresses,proto3" json:"master_grpc_addresses,omitempty"`
|
||||
FilerGrpcAddresses []string `protobuf:"bytes,2,rep,name=filer_grpc_addresses,json=filerGrpcAddresses,proto3" json:"filer_grpc_addresses,omitempty"`
|
||||
VolumeGrpcAddresses []string `protobuf:"bytes,3,rep,name=volume_grpc_addresses,json=volumeGrpcAddresses,proto3" json:"volume_grpc_addresses,omitempty"`
|
||||
Metadata map[string]string `protobuf:"bytes,4,rep,name=metadata,proto3" json:"metadata,omitempty" protobuf_key:"bytes,1,opt,name=key" protobuf_val:"bytes,2,opt,name=value"`
|
||||
S3GrpcAddresses []string `protobuf:"bytes,5,rep,name=s3_grpc_addresses,json=s3GrpcAddresses,proto3" json:"s3_grpc_addresses,omitempty"`
|
||||
// Filers in pb.ServerAddress form (host:httpPort.grpcPort). Consumers
|
||||
// convert to a gRPC or HTTP address as needed; see weed/pb/server_address.go.
|
||||
FilerAddresses []string `protobuf:"bytes,2,rep,name=filer_addresses,json=filerAddresses,proto3" json:"filer_addresses,omitempty"`
|
||||
VolumeGrpcAddresses []string `protobuf:"bytes,3,rep,name=volume_grpc_addresses,json=volumeGrpcAddresses,proto3" json:"volume_grpc_addresses,omitempty"`
|
||||
Metadata map[string]string `protobuf:"bytes,4,rep,name=metadata,proto3" json:"metadata,omitempty" protobuf_key:"bytes,1,opt,name=key" protobuf_val:"bytes,2,opt,name=value"`
|
||||
S3GrpcAddresses []string `protobuf:"bytes,5,rep,name=s3_grpc_addresses,json=s3GrpcAddresses,proto3" json:"s3_grpc_addresses,omitempty"`
|
||||
unknownFields protoimpl.UnknownFields
|
||||
sizeCache protoimpl.SizeCache
|
||||
}
|
||||
@@ -3609,9 +3611,9 @@ func (x *ClusterContext) GetMasterGrpcAddresses() []string {
|
||||
return nil
|
||||
}
|
||||
|
||||
func (x *ClusterContext) GetFilerGrpcAddresses() []string {
|
||||
func (x *ClusterContext) GetFilerAddresses() []string {
|
||||
if x != nil {
|
||||
return x.FilerGrpcAddresses
|
||||
return x.FilerAddresses
|
||||
}
|
||||
return nil
|
||||
}
|
||||
@@ -4273,10 +4275,10 @@ const file_plugin_proto_rawDesc = "" +
|
||||
"\asummary\x18\x02 \x01(\tR\asummary\x1aT\n" +
|
||||
"\x11OutputValuesEntry\x12\x10\n" +
|
||||
"\x03key\x18\x01 \x01(\tR\x03key\x12)\n" +
|
||||
"\x05value\x18\x02 \x01(\v2\x13.plugin.ConfigValueR\x05value:\x028\x01\"\xd5\x02\n" +
|
||||
"\x05value\x18\x02 \x01(\v2\x13.plugin.ConfigValueR\x05value:\x028\x01\"\xcc\x02\n" +
|
||||
"\x0eClusterContext\x122\n" +
|
||||
"\x15master_grpc_addresses\x18\x01 \x03(\tR\x13masterGrpcAddresses\x120\n" +
|
||||
"\x14filer_grpc_addresses\x18\x02 \x03(\tR\x12filerGrpcAddresses\x122\n" +
|
||||
"\x15master_grpc_addresses\x18\x01 \x03(\tR\x13masterGrpcAddresses\x12'\n" +
|
||||
"\x0ffiler_addresses\x18\x02 \x03(\tR\x0efilerAddresses\x122\n" +
|
||||
"\x15volume_grpc_addresses\x18\x03 \x03(\tR\x13volumeGrpcAddresses\x12@\n" +
|
||||
"\bmetadata\x18\x04 \x03(\v2$.plugin.ClusterContext.MetadataEntryR\bmetadata\x12*\n" +
|
||||
"\x11s3_grpc_addresses\x18\x05 \x03(\tR\x0fs3GrpcAddresses\x1a;\n" +
|
||||
|
||||
@@ -374,6 +374,7 @@ message ErasureCodingTaskConfig {
|
||||
int32 min_volume_size_mb = 3; // Minimum volume size for EC
|
||||
string collection_filter = 4; // Only process volumes from specific collections
|
||||
repeated string preferred_tags = 5; // Disk tags to prioritize for EC shard placement
|
||||
string replica_placement = 6; // EC shard replica placement (e.g. "020"); empty falls back to master default replication
|
||||
}
|
||||
|
||||
// BalanceTaskConfig contains balance-specific configuration
|
||||
@@ -413,6 +414,7 @@ message EcBalanceTaskConfig {
|
||||
string collection_filter = 3; // Collection filter
|
||||
string disk_type = 4; // Disk type filter
|
||||
repeated string preferred_tags = 5; // Preferred disk tags for placement
|
||||
string replica_placement = 6; // EC shard replica placement (e.g. "020"); empty falls back to master default replication
|
||||
}
|
||||
|
||||
// ========== Task Persistence Messages ==========
|
||||
|
||||
@@ -2960,6 +2960,7 @@ type ErasureCodingTaskConfig struct {
|
||||
MinVolumeSizeMb int32 `protobuf:"varint,3,opt,name=min_volume_size_mb,json=minVolumeSizeMb,proto3" json:"min_volume_size_mb,omitempty"` // Minimum volume size for EC
|
||||
CollectionFilter string `protobuf:"bytes,4,opt,name=collection_filter,json=collectionFilter,proto3" json:"collection_filter,omitempty"` // Only process volumes from specific collections
|
||||
PreferredTags []string `protobuf:"bytes,5,rep,name=preferred_tags,json=preferredTags,proto3" json:"preferred_tags,omitempty"` // Disk tags to prioritize for EC shard placement
|
||||
ReplicaPlacement string `protobuf:"bytes,6,opt,name=replica_placement,json=replicaPlacement,proto3" json:"replica_placement,omitempty"` // EC shard replica placement (e.g. "020"); empty falls back to master default replication
|
||||
unknownFields protoimpl.UnknownFields
|
||||
sizeCache protoimpl.SizeCache
|
||||
}
|
||||
@@ -3029,6 +3030,13 @@ func (x *ErasureCodingTaskConfig) GetPreferredTags() []string {
|
||||
return nil
|
||||
}
|
||||
|
||||
func (x *ErasureCodingTaskConfig) GetReplicaPlacement() string {
|
||||
if x != nil {
|
||||
return x.ReplicaPlacement
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// BalanceTaskConfig contains balance-specific configuration
|
||||
type BalanceTaskConfig struct {
|
||||
state protoimpl.MessageState `protogen:"open.v1"`
|
||||
@@ -3297,6 +3305,7 @@ type EcBalanceTaskConfig struct {
|
||||
CollectionFilter string `protobuf:"bytes,3,opt,name=collection_filter,json=collectionFilter,proto3" json:"collection_filter,omitempty"` // Collection filter
|
||||
DiskType string `protobuf:"bytes,4,opt,name=disk_type,json=diskType,proto3" json:"disk_type,omitempty"` // Disk type filter
|
||||
PreferredTags []string `protobuf:"bytes,5,rep,name=preferred_tags,json=preferredTags,proto3" json:"preferred_tags,omitempty"` // Preferred disk tags for placement
|
||||
ReplicaPlacement string `protobuf:"bytes,6,opt,name=replica_placement,json=replicaPlacement,proto3" json:"replica_placement,omitempty"` // EC shard replica placement (e.g. "020"); empty falls back to master default replication
|
||||
unknownFields protoimpl.UnknownFields
|
||||
sizeCache protoimpl.SizeCache
|
||||
}
|
||||
@@ -3366,6 +3375,13 @@ func (x *EcBalanceTaskConfig) GetPreferredTags() []string {
|
||||
return nil
|
||||
}
|
||||
|
||||
func (x *EcBalanceTaskConfig) GetReplicaPlacement() string {
|
||||
if x != nil {
|
||||
return x.ReplicaPlacement
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// MaintenanceTaskData represents complete task state for persistence
|
||||
type MaintenanceTaskData struct {
|
||||
state protoimpl.MessageState `protogen:"open.v1"`
|
||||
@@ -4210,13 +4226,14 @@ const file_worker_proto_rawDesc = "" +
|
||||
"\x10VacuumTaskConfig\x12+\n" +
|
||||
"\x11garbage_threshold\x18\x01 \x01(\x01R\x10garbageThreshold\x12/\n" +
|
||||
"\x14min_volume_age_hours\x18\x02 \x01(\x05R\x11minVolumeAgeHours\x120\n" +
|
||||
"\x14min_interval_seconds\x18\x03 \x01(\x05R\x12minIntervalSeconds\"\xed\x01\n" +
|
||||
"\x14min_interval_seconds\x18\x03 \x01(\x05R\x12minIntervalSeconds\"\x9a\x02\n" +
|
||||
"\x17ErasureCodingTaskConfig\x12%\n" +
|
||||
"\x0efullness_ratio\x18\x01 \x01(\x01R\rfullnessRatio\x12*\n" +
|
||||
"\x11quiet_for_seconds\x18\x02 \x01(\x05R\x0fquietForSeconds\x12+\n" +
|
||||
"\x12min_volume_size_mb\x18\x03 \x01(\x05R\x0fminVolumeSizeMb\x12+\n" +
|
||||
"\x11collection_filter\x18\x04 \x01(\tR\x10collectionFilter\x12%\n" +
|
||||
"\x0epreferred_tags\x18\x05 \x03(\tR\rpreferredTags\"n\n" +
|
||||
"\x0epreferred_tags\x18\x05 \x03(\tR\rpreferredTags\x12+\n" +
|
||||
"\x11replica_placement\x18\x06 \x01(\tR\x10replicaPlacement\"n\n" +
|
||||
"\x11BalanceTaskConfig\x12/\n" +
|
||||
"\x13imbalance_threshold\x18\x01 \x01(\x01R\x12imbalanceThreshold\x12(\n" +
|
||||
"\x10min_server_count\x18\x02 \x01(\x05R\x0eminServerCount\"I\n" +
|
||||
@@ -4238,13 +4255,14 @@ const file_worker_proto_rawDesc = "" +
|
||||
"\x0esource_disk_id\x18\x05 \x01(\rR\fsourceDiskId\x12\x1f\n" +
|
||||
"\vtarget_node\x18\x06 \x01(\tR\n" +
|
||||
"targetNode\x12$\n" +
|
||||
"\x0etarget_disk_id\x18\a \x01(\rR\ftargetDiskId\"\xe1\x01\n" +
|
||||
"\x0etarget_disk_id\x18\a \x01(\rR\ftargetDiskId\"\x8e\x02\n" +
|
||||
"\x13EcBalanceTaskConfig\x12/\n" +
|
||||
"\x13imbalance_threshold\x18\x01 \x01(\x01R\x12imbalanceThreshold\x12(\n" +
|
||||
"\x10min_server_count\x18\x02 \x01(\x05R\x0eminServerCount\x12+\n" +
|
||||
"\x11collection_filter\x18\x03 \x01(\tR\x10collectionFilter\x12\x1b\n" +
|
||||
"\tdisk_type\x18\x04 \x01(\tR\bdiskType\x12%\n" +
|
||||
"\x0epreferred_tags\x18\x05 \x03(\tR\rpreferredTags\"\xae\a\n" +
|
||||
"\x0epreferred_tags\x18\x05 \x03(\tR\rpreferredTags\x12+\n" +
|
||||
"\x11replica_placement\x18\x06 \x01(\tR\x10replicaPlacement\"\xae\a\n" +
|
||||
"\x13MaintenanceTaskData\x12\x0e\n" +
|
||||
"\x02id\x18\x01 \x01(\tR\x02id\x12\x12\n" +
|
||||
"\x04type\x18\x02 \x01(\tR\x04type\x12\x1a\n" +
|
||||
|
||||
@@ -19,9 +19,9 @@ import (
|
||||
)
|
||||
|
||||
const (
|
||||
adminScriptJobType = "admin_script"
|
||||
maxAdminScriptOutputBytes = 16 * 1024
|
||||
defaultAdminScriptRunMins = 17
|
||||
adminScriptJobType = "admin_script"
|
||||
maxAdminScriptOutputBytes = 16 * 1024
|
||||
defaultAdminScriptRunMins = 17
|
||||
adminScriptDetectTickMinutes = 17
|
||||
)
|
||||
|
||||
@@ -546,13 +546,31 @@ func (h *AdminScriptHandler) buildAdminScriptCommandEnv(
|
||||
ctx context.Context,
|
||||
clusterContext *plugin_pb.ClusterContext,
|
||||
) (*shell.CommandEnv, context.CancelFunc, error) {
|
||||
options, err := h.buildAdminScriptShellOptions(clusterContext)
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
|
||||
commandEnv := shell.NewCommandEnv(&options)
|
||||
commandEnv.ForceNoLock()
|
||||
|
||||
ctx, cancel := context.WithCancel(ctx)
|
||||
go commandEnv.MasterClient.KeepConnectedToMaster(ctx)
|
||||
|
||||
return commandEnv, cancel, nil
|
||||
}
|
||||
|
||||
// buildAdminScriptShellOptions maps a cluster context onto shell.ShellOptions.
|
||||
// It is split out from buildAdminScriptCommandEnv so the address handling is
|
||||
// testable without standing up a master client.
|
||||
func (h *AdminScriptHandler) buildAdminScriptShellOptions(clusterContext *plugin_pb.ClusterContext) (shell.ShellOptions, error) {
|
||||
if clusterContext == nil {
|
||||
return nil, nil, fmt.Errorf("cluster context is required")
|
||||
return shell.ShellOptions{}, fmt.Errorf("cluster context is required")
|
||||
}
|
||||
|
||||
masters := normalizeAddressList(clusterContext.MasterGrpcAddresses)
|
||||
if len(masters) == 0 {
|
||||
return nil, nil, fmt.Errorf("missing master addresses for admin script")
|
||||
return shell.ShellOptions{}, fmt.Errorf("missing master addresses for admin script")
|
||||
}
|
||||
|
||||
filerGroup := ""
|
||||
@@ -564,20 +582,17 @@ func (h *AdminScriptHandler) buildAdminScriptCommandEnv(
|
||||
Directory: "/",
|
||||
}
|
||||
|
||||
filers := normalizeAddressList(clusterContext.FilerGrpcAddresses)
|
||||
// FilerAddresses are pb.ServerAddress strings (host:httpPort.grpcPort).
|
||||
// ShellOptions.FilerAddress is itself a ServerAddress; shell commands derive
|
||||
// the gRPC or HTTP port from it as needed, so store it verbatim.
|
||||
filers := normalizeAddressList(clusterContext.FilerAddresses)
|
||||
if len(filers) > 0 {
|
||||
options.FilerAddress = pb.ServerAddress(filers[0])
|
||||
} else {
|
||||
glog.V(1).Infof("admin script worker missing filer address; filer-dependent commands may fail")
|
||||
}
|
||||
|
||||
commandEnv := shell.NewCommandEnv(&options)
|
||||
commandEnv.ForceNoLock()
|
||||
|
||||
ctx, cancel := context.WithCancel(ctx)
|
||||
go commandEnv.MasterClient.KeepConnectedToMaster(ctx)
|
||||
|
||||
return commandEnv, cancel, nil
|
||||
return options, nil
|
||||
}
|
||||
|
||||
func normalizeAddressList(addresses []string) []string {
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user