mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-10-10 16:45:51 +00:00
Compare commits
119
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
a4bf9fe47f | ||
|
|
b3ea57b5d7 | ||
|
|
cd68313929 | ||
|
|
675020b342 | ||
|
|
7919cc7ca0 | ||
|
|
1e91a99f79 | ||
|
|
4f17c6661a | ||
|
|
29eec2f111 | ||
|
|
8fd7c524c7 | ||
|
|
77dcb20a74 | ||
|
|
dd1b428789 | ||
|
|
1355c7a102 | ||
|
|
f72c5ec5d3 | ||
|
|
96f521addc | ||
|
|
584da4cd10 | ||
|
|
56b9df937c | ||
|
|
e8ed043d2b | ||
|
|
502fef6b50 | ||
|
|
b21c263328 | ||
|
|
c9868dcf2f | ||
|
|
85ca3cb757 | ||
|
|
a3c0baa9b0 | ||
|
|
881226a81b | ||
|
|
f8caaa4464 | ||
|
|
c97b69f8a4 | ||
|
|
3976264391 | ||
|
|
3481f13f54 | ||
|
|
68cae26c0b | ||
|
|
fef49c2d75 | ||
|
|
564b94796a | ||
|
|
475ae2b443 | ||
|
|
e8e7cd6fac | ||
|
|
0f1e50f9ec | ||
|
|
2a4923e7e8 | ||
|
|
25beb7ec48 | ||
|
|
6fc212cedb | ||
|
|
1f0c366583 | ||
|
|
fa7056dc6f | ||
|
|
eeda7181aa | ||
|
|
4b9d46b5ad | ||
|
|
5bac8b9281 | ||
|
|
db954b5503 | ||
|
|
32aa70ab59 | ||
|
|
f9bc6adf98 | ||
|
|
f037fc4dce | ||
|
|
b4d2224e97 | ||
|
|
83195fc111 | ||
|
|
091aad59dc | ||
|
|
dc5621d2ae | ||
|
|
e2203b2a0b | ||
|
|
e71bac55e9 | ||
|
|
bf022ca018 | ||
|
|
b18d3dc96c | ||
|
|
bce76e6e21 | ||
|
|
21f2699624 | ||
|
|
d1665750e1 | ||
|
|
0566fbd552 | ||
|
|
d4e39b499b | ||
|
|
adfd731bb8 | ||
|
|
917a87928c | ||
|
|
8fa769f29a | ||
|
|
7c635c4508 | ||
|
|
fbdcec1cba | ||
|
|
0accff0e4a | ||
|
|
9021225591 | ||
|
|
5b42287c22 | ||
|
|
3392493f0a | ||
|
|
d82b3a8d6a | ||
|
|
39e9294907 | ||
|
|
3825035f07 | ||
|
|
83b7ea5e7b | ||
|
|
eae8f33db5 | ||
|
|
2c2b2d4d3e | ||
|
|
cd15ae1395 | ||
|
|
3f6410fdc3 | ||
|
|
87fdea5330 | ||
|
|
303c2be38d | ||
|
|
9b9fdb5b76 | ||
|
|
7e4691f2dc | ||
|
|
391f543ff2 | ||
|
|
afcc491517 | ||
|
|
a5d0e4a735 | ||
|
|
a17dca7009 | ||
|
|
024b59fb31 | ||
|
|
5af7d12f04 | ||
|
|
4385b86bf1 | ||
|
|
c00aa90990 | ||
|
|
e332b97d52 | ||
|
|
868849392c | ||
|
|
a4415c39aa | ||
|
|
9914e6af30 | ||
|
|
cc5ef1b741 | ||
|
|
37b6a14b0d | ||
|
|
cee2bf697c | ||
|
|
285025eb73 | ||
|
|
77ac781bbd | ||
|
|
f72983c1fd | ||
|
|
cfc08fbf6c | ||
|
|
d57de6dc20 | ||
|
|
4476cb282b | ||
|
|
b63610cf8f | ||
|
|
c61d227613 | ||
|
|
7c252e1f16 | ||
|
|
7c5296dfb1 | ||
|
|
58c3fa802c | ||
|
|
d3f80444df | ||
|
|
0dc65e7069 | ||
|
|
18c6c24e47 | ||
|
|
120901c883 | ||
|
|
a79880ed41 | ||
|
|
f5aa776742 | ||
|
|
f3d6633aac | ||
|
|
68794fb94c | ||
|
|
af8d4e00ee | ||
|
|
41b6ad002b | ||
|
|
a761441926 | ||
|
|
37e6263efe | ||
|
|
3d872a1416 | ||
|
|
4d04609bb8 |
@@ -128,14 +128,14 @@ jobs:
|
||||
|
||||
- name: Login to Docker Hub
|
||||
if: github.event_name != 'pull_request'
|
||||
uses: docker/login-action@v4.1.0
|
||||
uses: docker/login-action@v4.2.0
|
||||
with:
|
||||
username: ${{ secrets.DOCKER_USERNAME }}
|
||||
password: ${{ secrets.DOCKER_PASSWORD }}
|
||||
|
||||
- name: Login to GHCR
|
||||
if: github.event_name != 'pull_request'
|
||||
uses: docker/login-action@v4.1.0
|
||||
uses: docker/login-action@v4.2.0
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ secrets.GHCR_USERNAME }}
|
||||
|
||||
@@ -133,7 +133,7 @@ jobs:
|
||||
|
||||
- name: Login to Docker Hub
|
||||
if: github.event_name != 'pull_request'
|
||||
uses: docker/login-action@v4.1.0
|
||||
uses: docker/login-action@v4.2.0
|
||||
with:
|
||||
username: ${{ secrets.DOCKER_USERNAME }}
|
||||
password: ${{ secrets.DOCKER_PASSWORD }}
|
||||
|
||||
@@ -221,13 +221,13 @@ jobs:
|
||||
buildkitd-config: /tmp/buildkitd.toml
|
||||
- name: Login to Docker Hub
|
||||
if: needs.setup.outputs.publish == 'true'
|
||||
uses: docker/login-action@v4.1.0
|
||||
uses: docker/login-action@v4.2.0
|
||||
with:
|
||||
username: ${{ secrets.DOCKER_USERNAME }}
|
||||
password: ${{ secrets.DOCKER_PASSWORD }}
|
||||
- name: Login to GHCR
|
||||
if: needs.setup.outputs.publish == 'true'
|
||||
uses: docker/login-action@v4.1.0
|
||||
uses: docker/login-action@v4.2.0
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ secrets.GHCR_USERNAME }}
|
||||
@@ -275,7 +275,7 @@ jobs:
|
||||
fi
|
||||
- name: Login to GHCR
|
||||
if: needs.setup.outputs.publish == 'true'
|
||||
uses: docker/login-action@v4.1.0
|
||||
uses: docker/login-action@v4.2.0
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ secrets.GHCR_USERNAME }}
|
||||
@@ -430,12 +430,12 @@ jobs:
|
||||
ghcr.io/chrislusf/seaweedfs
|
||||
tags: type=raw,value=${{ github.event_name == 'workflow_dispatch' && github.event.inputs.image_tag || 'latest' }},suffix=${{ steps.config.outputs.tag_suffix }}
|
||||
- name: Login to Docker Hub
|
||||
uses: docker/login-action@v4.1.0
|
||||
uses: docker/login-action@v4.2.0
|
||||
with:
|
||||
username: ${{ secrets.DOCKER_USERNAME }}
|
||||
password: ${{ secrets.DOCKER_PASSWORD }}
|
||||
- name: Login to GHCR
|
||||
uses: docker/login-action@v4.1.0
|
||||
uses: docker/login-action@v4.2.0
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ secrets.GHCR_USERNAME }}
|
||||
|
||||
@@ -50,7 +50,7 @@ jobs:
|
||||
-
|
||||
name: Login to Docker Hub
|
||||
if: github.event_name != 'pull_request'
|
||||
uses: docker/login-action@v4.1.0
|
||||
uses: docker/login-action@v4.2.0
|
||||
with:
|
||||
username: ${{ secrets.DOCKER_USERNAME }}
|
||||
password: ${{ secrets.DOCKER_PASSWORD }}
|
||||
|
||||
@@ -237,14 +237,14 @@ jobs:
|
||||
|
||||
- name: Login to Docker Hub
|
||||
if: (github.event_name != 'workflow_dispatch' || github.event.inputs.variant == 'all' || github.event.inputs.variant == matrix.variant) && github.event_name != 'pull_request'
|
||||
uses: docker/login-action@v4.1.0
|
||||
uses: docker/login-action@v4.2.0
|
||||
with:
|
||||
username: ${{ secrets.DOCKER_USERNAME }}
|
||||
password: ${{ secrets.DOCKER_PASSWORD }}
|
||||
|
||||
- name: Login to GHCR
|
||||
if: (github.event_name != 'workflow_dispatch' || github.event.inputs.variant == 'all' || github.event.inputs.variant == matrix.variant) && github.event_name != 'pull_request'
|
||||
uses: docker/login-action@v4.1.0
|
||||
uses: docker/login-action@v4.2.0
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ secrets.GHCR_USERNAME }}
|
||||
@@ -300,14 +300,14 @@ jobs:
|
||||
steps:
|
||||
- name: Login to Docker Hub
|
||||
if: github.event_name != 'workflow_dispatch' || github.event.inputs.variant == 'all' || github.event.inputs.variant == matrix.variant
|
||||
uses: docker/login-action@v4.1.0
|
||||
uses: docker/login-action@v4.2.0
|
||||
with:
|
||||
username: ${{ secrets.DOCKER_USERNAME }}
|
||||
password: ${{ secrets.DOCKER_PASSWORD }}
|
||||
|
||||
- name: Login to GHCR
|
||||
if: github.event_name != 'workflow_dispatch' || github.event.inputs.variant == 'all' || github.event.inputs.variant == matrix.variant
|
||||
uses: docker/login-action@v4.1.0
|
||||
uses: docker/login-action@v4.2.0
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ secrets.GHCR_USERNAME }}
|
||||
@@ -380,7 +380,7 @@ jobs:
|
||||
variant: large_disk
|
||||
steps:
|
||||
- name: Login to GHCR
|
||||
uses: docker/login-action@v4.1.0
|
||||
uses: docker/login-action@v4.2.0
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ secrets.GHCR_USERNAME }}
|
||||
@@ -429,13 +429,13 @@ jobs:
|
||||
latest_tag: latest_large_disk
|
||||
steps:
|
||||
- name: Login to Docker Hub
|
||||
uses: docker/login-action@v4.1.0
|
||||
uses: docker/login-action@v4.2.0
|
||||
with:
|
||||
username: ${{ secrets.DOCKER_USERNAME }}
|
||||
password: ${{ secrets.DOCKER_PASSWORD }}
|
||||
|
||||
- name: Login to GHCR
|
||||
uses: docker/login-action@v4.1.0
|
||||
uses: docker/login-action@v4.2.0
|
||||
with:
|
||||
registry: ghcr.io
|
||||
username: ${{ secrets.GHCR_USERNAME }}
|
||||
|
||||
@@ -88,7 +88,7 @@ jobs:
|
||||
uses: docker/setup-buildx-action@4d04d5d9486b7bd6fa91e7baf45bbb4f8b9deedd # v1
|
||||
|
||||
- name: Login to Docker Hub
|
||||
uses: docker/login-action@4907a6ddec9925e35a0a9e82d7399ccc52663121 # v1
|
||||
uses: docker/login-action@650006c6eb7dba73a995cc03b0b2d7f5ca915bee # v1
|
||||
with:
|
||||
username: ${{ secrets.DOCKER_USERNAME }}
|
||||
password: ${{ secrets.DOCKER_PASSWORD }}
|
||||
|
||||
@@ -1,49 +0,0 @@
|
||||
name: EC Integration Tests
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [ master ]
|
||||
paths:
|
||||
- 'weed/admin/**'
|
||||
- 'weed/worker/**'
|
||||
- 'test/erasure_coding/admin_dockertest/**'
|
||||
- '.github/workflows/ec-integration.yml'
|
||||
pull_request:
|
||||
branches: [ master ]
|
||||
paths:
|
||||
- 'weed/admin/**'
|
||||
- 'weed/worker/**'
|
||||
- 'test/erasure_coding/admin_dockertest/**'
|
||||
- '.github/workflows/ec-integration.yml'
|
||||
|
||||
jobs:
|
||||
ec-integration-test:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 15
|
||||
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Set up Go
|
||||
uses: actions/setup-go@v6
|
||||
with:
|
||||
go-version-file: 'go.mod'
|
||||
|
||||
- name: Build weed binary
|
||||
run: |
|
||||
cd weed
|
||||
go build -o ../weed_bin
|
||||
|
||||
- name: Run EC integration tests
|
||||
run: |
|
||||
cd test/erasure_coding/admin_dockertest
|
||||
go test -v -timeout 15m ec_integration_test.go
|
||||
|
||||
- name: Upload test logs on failure
|
||||
if: failure()
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: ec-test-logs
|
||||
path: test/erasure_coding/admin_dockertest/tmp/logs/
|
||||
retention-days: 7
|
||||
@@ -0,0 +1,110 @@
|
||||
name: "S3 SDK V2 Route Disambiguation Tests"
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [ master ]
|
||||
paths:
|
||||
- 'weed/s3api/**'
|
||||
- 'test/s3/sdk_v2_routing/**'
|
||||
- '.github/workflows/s3-sdk-v2-routing-tests.yml'
|
||||
pull_request:
|
||||
branches: [ master ]
|
||||
paths:
|
||||
- 'weed/s3api/**'
|
||||
- 'test/s3/sdk_v2_routing/**'
|
||||
- '.github/workflows/s3-sdk-v2-routing-tests.yml'
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.head_ref || github.ref }}/s3-sdk-v2-routing-tests
|
||||
cancel-in-progress: true
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
s3-sdk-v2-routing-tests:
|
||||
name: S3 SDK V2 Routing Tests
|
||||
runs-on: ubuntu-22.04
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- name: Check out code
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Set up Go
|
||||
uses: actions/setup-go@v6
|
||||
with:
|
||||
go-version-file: 'go.mod'
|
||||
|
||||
- name: Install SeaweedFS
|
||||
run: |
|
||||
cd weed && go install -buildvcs=false
|
||||
|
||||
- name: Start weed mini (S3 on :8333)
|
||||
# Pins the regression for issue #9559: AWS SDK V2 / Hadoop s3a
|
||||
# listing a bucket literally named "buckets" must get an XML
|
||||
# ListObjectsV2 response, not the JSON ListTableBuckets body
|
||||
# served by the S3 Tables REST endpoint on the same path.
|
||||
run: |
|
||||
mkdir -p /tmp/seaweedfs-sdk-v2-routing
|
||||
cat > /tmp/seaweedfs-sdk-v2-routing-s3.json <<'JSON'
|
||||
{
|
||||
"identities": [
|
||||
{
|
||||
"name": "admin",
|
||||
"credentials": [
|
||||
{"accessKey": "some_access_key1", "secretKey": "some_secret_key1"}
|
||||
],
|
||||
"actions": ["Admin", "Read", "Write"]
|
||||
}
|
||||
]
|
||||
}
|
||||
JSON
|
||||
AWS_ACCESS_KEY_ID=some_access_key1 \
|
||||
AWS_SECRET_ACCESS_KEY=some_secret_key1 \
|
||||
weed mini \
|
||||
-dir=/tmp/seaweedfs-sdk-v2-routing \
|
||||
-s3.port=8333 \
|
||||
-s3.config=/tmp/seaweedfs-sdk-v2-routing-s3.json \
|
||||
-ip=127.0.0.1 \
|
||||
> /tmp/weed-mini.log 2>&1 &
|
||||
echo $! > /tmp/weed-mini.pid
|
||||
|
||||
for i in $(seq 1 30); do
|
||||
if curl -s -o /dev/null -w "%{http_code}" http://127.0.0.1:8333/ | grep -qE "^(200|403)$"; then
|
||||
echo "weed mini is ready"
|
||||
exit 0
|
||||
fi
|
||||
sleep 1
|
||||
done
|
||||
echo "weed mini failed to start within 30s"
|
||||
tail -50 /tmp/weed-mini.log
|
||||
exit 1
|
||||
|
||||
- name: Run SDK V2 routing tests
|
||||
env:
|
||||
S3_ENDPOINT: http://127.0.0.1:8333
|
||||
AWS_ACCESS_KEY_ID: some_access_key1
|
||||
AWS_SECRET_ACCESS_KEY: some_secret_key1
|
||||
AWS_REGION: us-east-1
|
||||
run: go test -v -timeout=5m ./test/s3/sdk_v2_routing/...
|
||||
|
||||
- name: Stop weed mini
|
||||
if: always()
|
||||
run: |
|
||||
if [ -f /tmp/weed-mini.pid ]; then
|
||||
kill "$(cat /tmp/weed-mini.pid)" 2>/dev/null || true
|
||||
fi
|
||||
|
||||
- name: Show server log on failure
|
||||
if: failure()
|
||||
run: |
|
||||
echo "=== weed mini log (last 200 lines) ==="
|
||||
tail -n 200 /tmp/weed-mini.log 2>/dev/null || echo "no log available"
|
||||
|
||||
- name: Archive log
|
||||
if: failure()
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: s3-sdk-v2-routing-server-log
|
||||
path: /tmp/weed-mini.log
|
||||
retention-days: 3
|
||||
@@ -0,0 +1,120 @@
|
||||
name: "Samba on FUSE Integration"
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [ master, main ]
|
||||
paths:
|
||||
- 'weed/mount/**'
|
||||
- 'weed/filer/**'
|
||||
- 'weed/cluster/**'
|
||||
- 'test/samba/**'
|
||||
- '.github/workflows/samba-integration.yml'
|
||||
pull_request:
|
||||
branches: [ master, main ]
|
||||
paths:
|
||||
- 'weed/mount/**'
|
||||
- 'weed/filer/**'
|
||||
- 'weed/cluster/**'
|
||||
- 'test/samba/**'
|
||||
- '.github/workflows/samba-integration.yml'
|
||||
workflow_dispatch:
|
||||
|
||||
concurrency:
|
||||
group: samba-integration/${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
samba-integration:
|
||||
name: samba-integration
|
||||
runs-on: ubuntu-22.04
|
||||
timeout-minutes: 45
|
||||
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v6
|
||||
|
||||
- name: Set up Go
|
||||
uses: actions/setup-go@v6
|
||||
with:
|
||||
go-version-file: 'go.mod'
|
||||
|
||||
- name: Start local Docker registry
|
||||
run: docker run -d --restart=always -p 5000:5000 --name registry registry:2
|
||||
|
||||
- name: Set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@v4
|
||||
with:
|
||||
driver-opts: network=host
|
||||
|
||||
- name: Build weed race binary
|
||||
run: |
|
||||
cd docker
|
||||
make binary_race
|
||||
|
||||
- name: Build SeaweedFS e2e image
|
||||
uses: docker/build-push-action@v7
|
||||
with:
|
||||
context: docker
|
||||
file: docker/Dockerfile.e2e
|
||||
tags: localhost:5000/chrislusf/seaweedfs:e2e
|
||||
push: true
|
||||
cache-from: type=gha,scope=samba-e2e
|
||||
cache-to: type=gha,mode=max,scope=samba-e2e
|
||||
|
||||
- name: Tag e2e image for docker compose
|
||||
run: |
|
||||
docker pull localhost:5000/chrislusf/seaweedfs:e2e
|
||||
docker tag localhost:5000/chrislusf/seaweedfs:e2e chrislusf/seaweedfs:e2e
|
||||
|
||||
- name: Build samba image
|
||||
uses: docker/build-push-action@v7
|
||||
with:
|
||||
context: test/samba
|
||||
build-contexts: |
|
||||
chrislusf/seaweedfs:e2e=docker-image://localhost:5000/chrislusf/seaweedfs:e2e
|
||||
tags: localhost:5000/chrislusf/seaweedfs:samba
|
||||
push: true
|
||||
cache-from: type=gha,scope=samba-harness
|
||||
cache-to: type=gha,mode=max,scope=samba-harness
|
||||
|
||||
- name: Tag samba image for docker compose
|
||||
run: |
|
||||
docker pull localhost:5000/chrislusf/seaweedfs:samba
|
||||
docker tag localhost:5000/chrislusf/seaweedfs:samba chrislusf/seaweedfs:samba
|
||||
|
||||
- name: Start SeaweedFS cluster and Samba
|
||||
run: |
|
||||
docker compose -f test/samba/docker-compose.yml up --wait
|
||||
|
||||
- name: Run Samba test battery
|
||||
run: |
|
||||
set -o pipefail
|
||||
docker compose -f test/samba/docker-compose.yml exec -T samba \
|
||||
/run_inside_container.sh 2>&1 | tee /tmp/samba-output.log
|
||||
|
||||
- name: Collect logs
|
||||
if: always()
|
||||
run: |
|
||||
mkdir -p /tmp/samba-docker-logs
|
||||
for svc in master volume filer samba; do
|
||||
docker compose -f test/samba/docker-compose.yml logs "$svc" \
|
||||
> "/tmp/samba-docker-logs/${svc}.log" 2>&1 || true
|
||||
done
|
||||
|
||||
- name: Tear down
|
||||
if: always()
|
||||
run: |
|
||||
docker compose -f test/samba/docker-compose.yml down -v
|
||||
|
||||
- name: Upload logs
|
||||
if: always()
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: samba-integration-results
|
||||
path: |
|
||||
/tmp/samba-output.log
|
||||
/tmp/samba-docker-logs/
|
||||
retention-days: 7
|
||||
@@ -509,20 +509,22 @@ SeaweedFS Filer uses off-the-shelf stores, such as MySql, Postgres, Sqlite, Mong
|
||||
|
||||
### Compared to MinIO ###
|
||||
|
||||
MinIO follows AWS S3 closely and is ideal for testing for S3 API. It has good UI, policies, versionings, etc. SeaweedFS is trying to catch up here. It is also possible to put MinIO as a gateway in front of SeaweedFS later.
|
||||
Please note, as Apr 25, 2026 MinIO ceased developement. It's strongly discouraged to use that unmaintained software with multiple security bugs.
|
||||
|
||||
MinIO metadata are in simple files. Each file write will incur extra writes to corresponding meta file.
|
||||
MinIO followed AWS S3 closely and was ideal for testing for S3 API. It had good UI, policies, versionings, etc. SeaweedFS is trying to catch up here.
|
||||
|
||||
MinIO does not have optimization for lots of small files. The files are simply stored as is to local disks.
|
||||
MinIO metadata were in simple files. Each file write will incur extra writes to corresponding meta file.
|
||||
|
||||
MinIO did not have optimization for lots of small files. The files were simply stored as is to local disks.
|
||||
Plus the extra meta file and shards for erasure coding, it only amplifies the LOSF problem.
|
||||
|
||||
MinIO has multiple disk IO to read one file. SeaweedFS has O(1) disk reads, even for erasure coded files.
|
||||
MinIO had multiple disk IO to read one file. SeaweedFS has O(1) disk reads, even for erasure coded files.
|
||||
|
||||
MinIO has full-time erasure coding. SeaweedFS uses replication on hot data for faster speed and optionally applies erasure coding on warm data.
|
||||
MinIO had full-time erasure coding. SeaweedFS uses replication on hot data for faster speed and optionally applies erasure coding on warm data.
|
||||
|
||||
MinIO does not have POSIX-like API support.
|
||||
MinIO did not have POSIX-like API support.
|
||||
|
||||
MinIO has specific requirements on storage layout. It is not flexible to adjust capacity. In SeaweedFS, just start one volume server pointing to the master. That's all.
|
||||
MinIO had specific requirements on storage layout. It is not flexible to adjust capacity. In SeaweedFS, just start one volume server pointing to the master. That's all.
|
||||
|
||||
## Dev Plan ##
|
||||
|
||||
|
||||
@@ -0,0 +1,167 @@
|
||||
# Design: Serializing Bucket Configuration Mutations
|
||||
|
||||
Issue #9651 — concurrent `PutBucketVersioning` + `PutBucketEncryption` (as Terraform
|
||||
issues them in parallel) intermittently lose the encryption write.
|
||||
|
||||
## Root cause
|
||||
|
||||
The bucket's entire config lives in one filer entry, `/buckets/<name>`. Every
|
||||
config API does a read-modify-write of that single entry, and the writes are not
|
||||
serialized:
|
||||
|
||||
- `updateBucketConfig(bucket, fn)` (`s3api_bucket_config.go:468`) — sources from a
|
||||
possibly-stale cached `BucketConfig`, mutates `Entry.Extended`, writes the
|
||||
**whole** entry. Used by: versioning, object-lock config, lifecycle, ACL/owner.
|
||||
- `UpdateBucketMetadata` → `setBucketMetadata` (`:1042`) — reads a fresh entry,
|
||||
mutates `Entry.Content`, writes the **whole** entry. Used by: encryption, CORS,
|
||||
tagging, ownership, policy, notification.
|
||||
|
||||
Two ingredients produce the lost update:
|
||||
|
||||
1. **No serialization** of the read→modify→write (the cache mutexes only guard the
|
||||
in-memory map, not the RMW).
|
||||
2. **Whole-entry rewrite from an independent snapshot** — `updateBucketConfig`
|
||||
rebuilds from a stale cached `BucketConfig` whose `Content` predates the
|
||||
concurrent encryption write, so writing the whole entry reverts `Content`.
|
||||
|
||||
Sequential calls always pass (each sees the previous write), so it only surfaces
|
||||
under concurrency — and CI's slower IO widens the window (the "2 of ~12 runs").
|
||||
|
||||
## Goals
|
||||
|
||||
- No lost updates across concurrent bucket-config changes — for **all** config
|
||||
fields, not just versioning/encryption.
|
||||
- Correct for a single S3 gateway (the reported case) and for multiple gateways.
|
||||
- Reuse the filer primitives just merged (per-path lock, `WriteCondition`,
|
||||
`ObjectTransaction`); do not reintroduce a distributed lock.
|
||||
- Minimal blast radius: the fix lands at the two chokepoint helpers.
|
||||
|
||||
## Non-goals
|
||||
|
||||
- Changing the one-entry-per-bucket storage model.
|
||||
- Multi-filer-concurrent bucket writes (addressed only as an optional phase 3).
|
||||
|
||||
## The two ingredients map to two complementary fixes
|
||||
|
||||
### Fix A — serialize + read fresh (closes the window for whole-entry writers)
|
||||
|
||||
Both `updateBucketConfig` and `UpdateBucketMetadata` must run their RMW under one
|
||||
per-bucket critical section, and **re-read the entry fresh from the filer inside
|
||||
it** — not rebuild from the cached `BucketConfig`. The lock alone is insufficient:
|
||||
without the fresh read, two serialized writers still each apply a stale snapshot.
|
||||
|
||||
### Fix B — field-level updates (removes the collision entirely)
|
||||
|
||||
The two writers touch disjoint fields (`Extended[versioning]` vs `Content`). If
|
||||
each path updated only its own field instead of rewriting the whole entry, neither
|
||||
could clobber the other regardless of ordering. This is the structural fix and
|
||||
makes serialization a defense-in-depth concern rather than a correctness
|
||||
requirement for cross-field cases.
|
||||
|
||||
## Where to serialize (layering)
|
||||
|
||||
The bucket entry is a single filer entry, so unlike object writes there is no
|
||||
sharding — the question is purely the scope of the lock:
|
||||
|
||||
| Layer | Serializes across | Cost | Notes |
|
||||
|---|---|---|---|
|
||||
| 1. Gateway-local per-bucket lock | one gateway process | tiny | fixes the reported (single-gateway/CI) case |
|
||||
| 2. Filer per-path lock via conditional write | all gateways on one filer | small | reuses #9640 `CreateEntry`+`WriteCondition` |
|
||||
| 3. Route-by-key to bucket-key owner filer | all gateways and filers | medium | same mechanism as the object DLM-removal |
|
||||
|
||||
## Recommended plan (phased)
|
||||
|
||||
### Phase 1 — minimal fix for #9651 (gateway-local lock + fresh read)
|
||||
|
||||
Add a bounded per-bucket lock table to `S3ApiServer`, reusing the same
|
||||
`util.LockTable` the filer uses for its per-path lock:
|
||||
|
||||
```go
|
||||
// in S3ApiServer
|
||||
bucketConfigLocks *util.LockTable[string] // serialize bucket-entry RMW
|
||||
|
||||
func (s3a *S3ApiServer) withBucketConfigLock(bucket string, fn func() s3err.ErrorCode) s3err.ErrorCode {
|
||||
lk := s3a.bucketConfigLocks.AcquireLock("bucketConfig", bucket, util.ExclusiveLock)
|
||||
defer s3a.bucketConfigLocks.ReleaseLock(bucket, lk)
|
||||
return fn()
|
||||
}
|
||||
```
|
||||
|
||||
Wrap the RMW in **both** chokepoints, and inside the lock read the entry fresh:
|
||||
|
||||
- `updateBucketConfig`: acquire the lock; re-read `/buckets/<name>` from the filer
|
||||
(not the cache); rebuild `BucketConfig` from that fresh entry; apply `fn`; write;
|
||||
invalidate cache; release.
|
||||
- `UpdateBucketMetadata`/`setBucketMetadata`: same lock key; it already reads fresh,
|
||||
so it just needs to share the critical section.
|
||||
|
||||
Both must use the **same** lock keyed on `bucket`, so versioning and encryption
|
||||
contend on one mutex. This closes the reported window. Limitation: only one
|
||||
gateway; two gateways behind a load balancer still race.
|
||||
|
||||
Test: parallel `PutBucketVersioning` + `PutBucketEncryption`, assert both persist
|
||||
(the exact Terraform scenario), plus an N-way parallel variant over distinct
|
||||
fields.
|
||||
|
||||
### Phase 2 — robust across gateways (field-level + CAS via merged primitives)
|
||||
|
||||
Move the writers off whole-entry rewrites:
|
||||
|
||||
- **Extended-based config** (versioning, object-lock, ownership, tagging-in-Extended)
|
||||
→ `ObjectTransaction` `PATCH_EXTENDED` on `/buckets/<name>`. The owner filer reads
|
||||
the entry fresh under its per-path lock and merges only the named keys, so the
|
||||
gateway never sends a whole-entry snapshot — this dissolves *both* ingredients for
|
||||
these fields.
|
||||
- **`Content`-based config** (encryption, CORS, tags blob) — **chosen and
|
||||
implemented (b3): extend `PATCH_EXTENDED` with `set_content`.** Under the same
|
||||
per-path lock the filer reads the entry fresh, merges extended attributes, and
|
||||
replaces `Content`, preserving the rest. So a content write becomes a field-level
|
||||
patch too — `setBucketMetadata` patches `Content`, `updateBucketConfig` patches
|
||||
extended keys, and the two serialize on the lock instead of racing whole-entry
|
||||
rewrites. This is cleaner than the alternatives below: no client-side retry, no
|
||||
storage migration, and it reuses `ObjectTransaction`'s existing atomic lock.
|
||||
- (b1, rejected) Conditional `CreateEntry` overwrite with `IF_ETAG_MATCH` + retry
|
||||
(#9640): correct but needs client-side retry, and the bucket directory entry has
|
||||
no reliable ETag to compare on.
|
||||
- (b2, future) Migrate each per-feature config out of the single `Content` blob
|
||||
into its own `Extended` key. Then even *intra-blob* writes (tags vs encryption)
|
||||
stop racing. Larger migration; tracked separately.
|
||||
|
||||
Once all paths are field-level patches, the phase-1 gateway lock is unnecessary —
|
||||
the filer enforces atomicity. (This is the path taken: phase 1 was skipped.)
|
||||
|
||||
### Phase 3 — multi-filer (only if needed)
|
||||
|
||||
If multiple filers can write `/buckets/<name>` concurrently, a filer-local per-path
|
||||
lock no longer suffices. Route bucket-config writes to
|
||||
`PrimaryForKey("/buckets/<name>")` (the lock-ring view) and serialize on that one
|
||||
owner filer — the same route-by-key design used to take object writes off the DLM.
|
||||
Overkill for rare config writes; include only if multi-filer bucket writes are real.
|
||||
|
||||
## Correctness summary
|
||||
|
||||
- Phase 1: all RMW for a bucket serialize within a gateway; the fresh read means the
|
||||
second writer observes the first's change. Closes #9651 for single-gateway.
|
||||
- Phase 2: `PATCH_EXTENDED` is atomic field-level merge at the filer (no snapshot);
|
||||
CAS turns a concurrent `Content` write into a retry, enforced under the filer's
|
||||
per-path lock — correct for any number of gateways sharing a filer.
|
||||
- Phase 3: one owner filer serializes all writers — correct across filers too.
|
||||
|
||||
## Scope checklist (every path that RMWs the bucket entry)
|
||||
|
||||
All of these funnel through the two chokepoints, so fixing the chokepoints covers
|
||||
them — but the fix must not leave any of them on an unserialized path:
|
||||
|
||||
- via `updateBucketConfig`: versioning, object-lock config, lifecycle, ACL/owner.
|
||||
- via `UpdateBucketMetadata`/`setBucketMetadata`: encryption, CORS, tagging,
|
||||
ownership controls, bucket policy, notification.
|
||||
- bucket create/delete (`CreateEntry`/`DeleteEntry` of `/buckets/<name>`) already
|
||||
go through the filer's per-path lock on `CreateEntry`; ensure they take the same
|
||||
bucket lock if they also patch config.
|
||||
|
||||
## Cache rule (must document in code)
|
||||
|
||||
Under the lock, **read the entry from the filer, never rebuild from the cached
|
||||
`BucketConfig`**. The cache is for reads; it must be invalidated on every write and
|
||||
never be the source for an RMW. This is the single most important detail — the lock
|
||||
without the fresh read does not fix the bug.
|
||||
@@ -42,6 +42,10 @@ RUN if [ -f "/prebuilt/weed-volume-${TARGETARCH}" ]; then \
|
||||
echo "Skipping Rust build for $TARGETARCH (unsupported)" && \
|
||||
touch /weed-volume; \
|
||||
fi
|
||||
# Pre-built binaries arrive via GitHub Actions artifacts, which drop the
|
||||
# executable bit, so the copied file is 0644 and exec fails with "Permission
|
||||
# denied". Restore it (no-op for the empty placeholder, which stays size 0).
|
||||
RUN chmod 0755 /weed-volume
|
||||
|
||||
FROM alpine AS final
|
||||
LABEL author="Chris Lu"
|
||||
|
||||
@@ -15,7 +15,6 @@ require (
|
||||
github.com/coreos/go-semver v0.3.1 // indirect
|
||||
github.com/coreos/go-systemd/v22 v22.6.0 // indirect
|
||||
github.com/davecgh/go-spew v1.1.2-0.20180830191138-d8f796af33cc // indirect
|
||||
github.com/dgryski/go-rendezvous v0.0.0-20200823014737-9f7001d12a5f // indirect
|
||||
github.com/dustin/go-humanize v1.0.1
|
||||
github.com/eapache/go-resiliency v1.6.0 // indirect
|
||||
github.com/eapache/go-xerial-snappy v0.0.0-20230731223053-c322873962e3 // indirect
|
||||
@@ -27,7 +26,7 @@ require (
|
||||
github.com/facebookgo/subset v0.0.0-20200203212716-c811ad88dec4 // indirect
|
||||
github.com/fsnotify/fsnotify v1.9.0 // indirect
|
||||
github.com/go-redsync/redsync/v4 v4.16.0
|
||||
github.com/go-sql-driver/mysql v1.9.3
|
||||
github.com/go-sql-driver/mysql v1.10.0
|
||||
github.com/go-zookeeper/zk v1.0.4 // indirect
|
||||
github.com/golang/protobuf v1.5.4
|
||||
github.com/golang/snappy v1.0.0
|
||||
@@ -49,7 +48,7 @@ require (
|
||||
github.com/klauspost/compress v1.18.6
|
||||
github.com/klauspost/reedsolomon v1.14.0
|
||||
github.com/kurin/blazer v0.5.3
|
||||
github.com/linxGnu/grocksdb v1.10.7
|
||||
github.com/linxGnu/grocksdb v1.10.8
|
||||
github.com/mailru/easyjson v0.9.1 // indirect
|
||||
github.com/mattn/go-isatty v0.0.20 // indirect
|
||||
github.com/modern-go/concurrent v0.0.0-20180306012644-bacd9c7ef1dd // indirect
|
||||
@@ -92,13 +91,13 @@ require (
|
||||
gocloud.dev v0.45.0
|
||||
gocloud.dev/pubsub/natspubsub v0.45.0
|
||||
gocloud.dev/pubsub/rabbitpubsub v0.45.0
|
||||
golang.org/x/crypto v0.50.0
|
||||
golang.org/x/crypto v0.52.0
|
||||
golang.org/x/exp v0.0.0-20260410095643-746e56fc9e2f
|
||||
golang.org/x/image v0.39.0
|
||||
golang.org/x/net v0.53.0
|
||||
golang.org/x/net v0.54.0
|
||||
golang.org/x/oauth2 v0.36.0
|
||||
golang.org/x/sys v0.43.0
|
||||
golang.org/x/text v0.36.0 // indirect
|
||||
golang.org/x/sys v0.45.0
|
||||
golang.org/x/text v0.37.0 // indirect
|
||||
golang.org/x/tools v0.44.0 // indirect
|
||||
golang.org/x/xerrors v0.0.0-20240903120638-7835f813f4da // indirect
|
||||
google.golang.org/api v0.278.0
|
||||
@@ -123,11 +122,11 @@ require (
|
||||
github.com/apple/foundationdb/bindings/go v0.0.0-20250911184653-27f7192f47c3
|
||||
github.com/arangodb/go-driver v1.6.9
|
||||
github.com/armon/go-metrics v0.4.1
|
||||
github.com/aws/aws-sdk-go-v2 v1.41.6
|
||||
github.com/aws/aws-sdk-go-v2 v1.41.7
|
||||
github.com/aws/aws-sdk-go-v2/config v1.32.14
|
||||
github.com/aws/aws-sdk-go-v2/credentials v1.19.14
|
||||
github.com/aws/aws-sdk-go-v2/service/s3 v1.99.0
|
||||
github.com/cognusion/imaging v1.0.2
|
||||
github.com/aws/aws-sdk-go-v2/service/s3 v1.101.0
|
||||
github.com/cognusion/imaging v1.0.3
|
||||
github.com/fluent/fluent-logger-golang v1.10.1
|
||||
github.com/getsentry/sentry-go v0.44.1
|
||||
github.com/go-git/go-billy/v5 v5.9.0
|
||||
@@ -141,12 +140,12 @@ require (
|
||||
github.com/linkedin/goavro/v2 v2.15.0
|
||||
github.com/minio/crc64nvme v1.1.1
|
||||
github.com/orcaman/concurrent-map/v2 v2.0.1
|
||||
github.com/parquet-go/parquet-go v0.28.0
|
||||
github.com/parquet-go/parquet-go v0.30.1
|
||||
github.com/pkg/sftp v1.13.10
|
||||
github.com/rabbitmq/amqp091-go v1.11.0
|
||||
github.com/rclone/rclone v1.74.1
|
||||
github.com/rdleal/intervalst v1.5.0
|
||||
github.com/redis/go-redis/v9 v9.18.0
|
||||
github.com/redis/go-redis/v9 v9.19.0
|
||||
github.com/schollz/progressbar/v3 v3.19.0
|
||||
github.com/seaweedfs/go-fuse/v2 v2.9.3
|
||||
github.com/shirou/gopsutil/v4 v4.26.3
|
||||
@@ -158,7 +157,7 @@ require (
|
||||
github.com/xeipuuv/gojsonschema v1.2.0
|
||||
github.com/ydb-platform/ydb-go-sdk-auth-environ v0.5.1
|
||||
github.com/ydb-platform/ydb-go-sdk/v3 v3.134.2
|
||||
go.etcd.io/etcd/client/pkg/v3 v3.6.10
|
||||
go.etcd.io/etcd/client/pkg/v3 v3.6.11
|
||||
go.uber.org/atomic v1.11.0
|
||||
golang.org/x/sync v0.20.0
|
||||
golang.org/x/tools/godoc v0.1.0-deprecated
|
||||
@@ -297,14 +296,14 @@ require (
|
||||
cloud.google.com/go/compute/metadata v0.9.0 // indirect
|
||||
cloud.google.com/go/iam v1.7.0 // indirect
|
||||
cloud.google.com/go/monitoring v1.24.3 // indirect
|
||||
filippo.io/edwards25519 v1.1.1 // indirect
|
||||
github.com/Azure/azure-sdk-for-go/sdk/azcore v1.21.0
|
||||
filippo.io/edwards25519 v1.2.0 // indirect
|
||||
github.com/Azure/azure-sdk-for-go/sdk/azcore v1.21.1
|
||||
github.com/Azure/azure-sdk-for-go/sdk/azidentity v1.13.1
|
||||
github.com/Azure/azure-sdk-for-go/sdk/internal v1.11.2 // indirect
|
||||
github.com/Azure/azure-sdk-for-go/sdk/storage/azblob v1.6.4
|
||||
github.com/Azure/azure-sdk-for-go/sdk/internal v1.12.0 // indirect
|
||||
github.com/Azure/azure-sdk-for-go/sdk/storage/azblob v1.7.0
|
||||
github.com/Azure/azure-sdk-for-go/sdk/storage/azfile v1.5.4 // indirect
|
||||
github.com/Azure/go-ntlmssp v0.1.1 // indirect
|
||||
github.com/AzureAD/microsoft-authentication-library-for-go v1.6.0 // indirect
|
||||
github.com/AzureAD/microsoft-authentication-library-for-go v1.7.2 // indirect
|
||||
github.com/Files-com/files-sdk-go/v3 v3.3.82 // indirect
|
||||
github.com/GoogleCloudPlatform/opentelemetry-operations-go/detectors/gcp v1.31.0 // indirect
|
||||
github.com/GoogleCloudPlatform/opentelemetry-operations-go/exporter/metric v0.55.0 // indirect
|
||||
@@ -324,17 +323,17 @@ require (
|
||||
github.com/andybalholm/cascadia v1.3.3 // indirect
|
||||
github.com/appscode/go-querystring v0.0.0-20170504095604-0126cfb3f1dc // indirect
|
||||
github.com/arangodb/go-velocypack v0.0.0-20200318135517-5af53c29c67e // indirect
|
||||
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.8 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.10 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.18.21 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/feature/s3/manager v1.22.13 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/internal/configsources v1.4.21 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.7.21 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/internal/configsources v1.4.23 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.7.23 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/internal/ini v1.8.6 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/internal/v4a v1.4.22 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.7 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.13 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.13.21 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.21 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/internal/v4a v1.4.24 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.9 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.15 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.13.23 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.23 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/sns v1.39.7 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/sqs v1.42.17 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/sso v1.30.15 // indirect
|
||||
@@ -499,7 +498,7 @@ require (
|
||||
go.opentelemetry.io/otel/trace v1.43.0 // indirect
|
||||
go.uber.org/multierr v1.11.0 // indirect
|
||||
go.uber.org/zap v1.27.1 // indirect
|
||||
golang.org/x/term v0.42.0
|
||||
golang.org/x/term v0.43.0
|
||||
golang.org/x/time v0.15.0
|
||||
google.golang.org/genproto/googleapis/api v0.0.0-20260401024825-9d38bb4040a9 // indirect
|
||||
google.golang.org/genproto/googleapis/rpc v0.0.0-20260427160629-7cedc36a6bc4 // indirect
|
||||
|
||||
@@ -547,28 +547,28 @@ cloud.google.com/go/workflows v1.10.0/go.mod h1:fZ8LmRmZQWacon9UCX1r/g/DfAXx5VcP
|
||||
dario.cat/mergo v1.0.2 h1:85+piFYR1tMbRrLcDwR18y4UKJ3aH1Tbzi24VRW1TK8=
|
||||
dario.cat/mergo v1.0.2/go.mod h1:E/hbnu0NxMFBjpMIE34DRGLWqDy0g5FuKDhCb31ngxA=
|
||||
dmitri.shuralyov.com/gpu/mtl v0.0.0-20190408044501-666a987793e9/go.mod h1:H6x//7gZCb22OMCxBHrMx7a5I7Hp++hsVxbQ4BYO7hU=
|
||||
filippo.io/edwards25519 v1.1.1 h1:YpjwWWlNmGIDyXOn8zLzqiD+9TyIlPhGFG96P39uBpw=
|
||||
filippo.io/edwards25519 v1.1.1/go.mod h1:BxyFTGdWcka3PhytdK4V28tE5sGfRvvvRV7EaN4VDT4=
|
||||
filippo.io/edwards25519 v1.2.0 h1:crnVqOiS4jqYleHd9vaKZ+HKtHfllngJIiOpNpoJsjo=
|
||||
filippo.io/edwards25519 v1.2.0/go.mod h1:xzAOLCNug/yB62zG1bQ8uziwrIqIuxhctzJT18Q77mc=
|
||||
gioui.org v0.0.0-20210308172011-57750fc8a0a6/go.mod h1:RSH6KIUZ0p2xy5zHDxgAM4zumjgTw83q2ge/PI+yyw8=
|
||||
git.sr.ht/~sbinet/gg v0.3.1/go.mod h1:KGYtlADtqsqANL9ueOFkWymvzUvLMQllU5Ixo+8v3pc=
|
||||
github.com/AdaLogics/go-fuzz-headers v0.0.0-20240806141605-e8a1dd7889d6 h1:He8afgbRMd7mFxO99hRNu+6tazq8nFF9lIwo9JFroBk=
|
||||
github.com/AdaLogics/go-fuzz-headers v0.0.0-20240806141605-e8a1dd7889d6/go.mod h1:8o94RPi1/7XTJvwPpRSzSUedZrtlirdB3r9Z20bi2f8=
|
||||
github.com/Azure/azure-sdk-for-go/sdk/azcore v1.21.0 h1:fou+2+WFTib47nS+nz/ozhEBnvU96bKHy6LjRsY4E28=
|
||||
github.com/Azure/azure-sdk-for-go/sdk/azcore v1.21.0/go.mod h1:t76Ruy8AHvUAC8GfMWJMa0ElSbuIcO03NLpynfbgsPA=
|
||||
github.com/Azure/azure-sdk-for-go/sdk/azcore v1.21.1 h1:jHb/wfvRikGdxMXYV3QG/SzUOPYN9KEUUuC0Yd0/vC0=
|
||||
github.com/Azure/azure-sdk-for-go/sdk/azcore v1.21.1/go.mod h1:pzBXCYn05zvYIrwLgtK8Ap8QcjRg+0i76tMQdWN6wOk=
|
||||
github.com/Azure/azure-sdk-for-go/sdk/azidentity v1.13.1 h1:Hk5QBxZQC1jb2Fwj6mpzme37xbCDdNTxU7O9eb5+LB4=
|
||||
github.com/Azure/azure-sdk-for-go/sdk/azidentity v1.13.1/go.mod h1:IYus9qsFobWIc2YVwe/WPjcnyCkPKtnHAqUYeebc8z0=
|
||||
github.com/Azure/azure-sdk-for-go/sdk/azidentity/cache v0.3.2 h1:yz1bePFlP5Vws5+8ez6T3HWXPmwOK7Yvq8QxDBD3SKY=
|
||||
github.com/Azure/azure-sdk-for-go/sdk/azidentity/cache v0.3.2/go.mod h1:Pa9ZNPuoNu/GztvBSKk9J1cDJW6vk/n0zLtV4mgd8N8=
|
||||
github.com/Azure/azure-sdk-for-go/sdk/internal v1.11.2 h1:9iefClla7iYpfYWdzPCRDozdmndjTm8DXdpCzPajMgA=
|
||||
github.com/Azure/azure-sdk-for-go/sdk/internal v1.11.2/go.mod h1:XtLgD3ZD34DAaVIIAyG3objl5DynM3CQ/vMcbBNJZGI=
|
||||
github.com/Azure/azure-sdk-for-go/sdk/internal v1.12.0 h1:fhqpLE3UEXi9lPaBRpQ6XuRW0nU7hgg4zlmZZa+a9q4=
|
||||
github.com/Azure/azure-sdk-for-go/sdk/internal v1.12.0/go.mod h1:7dCRMLwisfRH3dBupKeNCioWYUZ4SS09Z14H+7i8ZoY=
|
||||
github.com/Azure/azure-sdk-for-go/sdk/keyvault/azkeys v0.10.0 h1:m/sWOGCREuSBqg2htVQTBY8nOZpyajYztF0vUvSZTuM=
|
||||
github.com/Azure/azure-sdk-for-go/sdk/keyvault/azkeys v0.10.0/go.mod h1:Pu5Zksi2KrU7LPbZbNINx6fuVrUp/ffvpxdDj+i8LeE=
|
||||
github.com/Azure/azure-sdk-for-go/sdk/keyvault/internal v0.7.1 h1:FbH3BbSb4bvGluTesZZ+ttN/MDsnMmQP36OSnDuSXqw=
|
||||
github.com/Azure/azure-sdk-for-go/sdk/keyvault/internal v0.7.1/go.mod h1:9V2j0jn9jDEkCkv8w/bKTNppX/d0FVA1ud77xCIP4KA=
|
||||
github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/storage/armstorage v1.8.1 h1:/Zt+cDPnpC3OVDm/JKLOs7M2DKmLRIIp3XIx9pHHiig=
|
||||
github.com/Azure/azure-sdk-for-go/sdk/resourcemanager/storage/armstorage v1.8.1/go.mod h1:Ng3urmn6dYe8gnbCMoHHVl5APYz2txho3koEkV2o2HA=
|
||||
github.com/Azure/azure-sdk-for-go/sdk/storage/azblob v1.6.4 h1:jWQK1GI+LeGGUKBADtcH2rRqPxYB1Ljwms5gFA2LqrM=
|
||||
github.com/Azure/azure-sdk-for-go/sdk/storage/azblob v1.6.4/go.mod h1:8mwH4klAm9DUgR2EEHyEEAQlRDvLPyg5fQry3y+cDew=
|
||||
github.com/Azure/azure-sdk-for-go/sdk/storage/azblob v1.7.0 h1:BM85pSYlVYQHdq00nxyPoOkyLF5NArJG3bOsrmbwr4k=
|
||||
github.com/Azure/azure-sdk-for-go/sdk/storage/azblob v1.7.0/go.mod h1:QYjP2cB7ZYtS/8jAbE0VSBZde/tjExqGjp+8JY6/+ts=
|
||||
github.com/Azure/azure-sdk-for-go/sdk/storage/azfile v1.5.4 h1:tZh20RjgfMxKBxJiIS75iTVAKIUxrST5X2dVHMTptL4=
|
||||
github.com/Azure/azure-sdk-for-go/sdk/storage/azfile v1.5.4/go.mod h1:vGYAk36rhMVCfTP7v+RVruCR0zmPe6S+36KRpDCLySw=
|
||||
github.com/Azure/go-ansiterm v0.0.0-20250102033503-faa5f7b0171c h1:udKWzYgxTojEKWjV8V+WSxDXJ4NFATAsZjh8iIbsQIg=
|
||||
@@ -581,8 +581,8 @@ github.com/Azure/go-ntlmssp v0.1.1 h1:l+FM/EEMb0U9QZE7mKNEDw5Mu3mFiaa2GKOoTSsNDP
|
||||
github.com/Azure/go-ntlmssp v0.1.1/go.mod h1:NYqdhxd/8aAct/s4qSYZEerdPuH1liG2/X9DiVTbhpk=
|
||||
github.com/AzureAD/microsoft-authentication-extensions-for-go/cache v0.1.1 h1:WJTmL004Abzc5wDB5VtZG2PJk5ndYDgVacGqfirKxjM=
|
||||
github.com/AzureAD/microsoft-authentication-extensions-for-go/cache v0.1.1/go.mod h1:tCcJZ0uHAmvjsVYzEFivsRTN00oz5BEsRgQHu5JZ9WE=
|
||||
github.com/AzureAD/microsoft-authentication-library-for-go v1.6.0 h1:XRzhVemXdgvJqCH0sFfrBUTnUJSBrBf7++ypk+twtRs=
|
||||
github.com/AzureAD/microsoft-authentication-library-for-go v1.6.0/go.mod h1:HKpQxkWaGLJ+D/5H8QRpyQXA1eKjxkFlOMwck5+33Jk=
|
||||
github.com/AzureAD/microsoft-authentication-library-for-go v1.7.2 h1:RHK7bS+HQMslb1sZpAokUt+zTVmue0hKSs2C791hhzU=
|
||||
github.com/AzureAD/microsoft-authentication-library-for-go v1.7.2/go.mod h1:HKpQxkWaGLJ+D/5H8QRpyQXA1eKjxkFlOMwck5+33Jk=
|
||||
github.com/BurntSushi/toml v0.3.1/go.mod h1:xHWCNGjB5oqiDr8zfno3MHue2Ht5sIBksp03qcyfWMU=
|
||||
github.com/BurntSushi/xgb v0.0.0-20160522181843-27f122750802/go.mod h1:IVnqGOEym/WlBOVXweHU+Q+/VP0lqqI8lqeDx9IjBqo=
|
||||
github.com/Codefor/geohash v0.0.0-20140723084247-1b41c28e3a9d h1:iG9B49Q218F/XxXNRM7k/vWf7MKmLIS8AcJV9cGN4nA=
|
||||
@@ -715,10 +715,10 @@ github.com/armon/go-metrics v0.4.1/go.mod h1:E6amYzXo6aW1tqzoZGT755KkbgrJsSdpwZ+
|
||||
github.com/atomicgo/cursor v0.0.1/go.mod h1:cBON2QmmrysudxNBFthvMtN32r3jxVRIvzkUiF/RuIk=
|
||||
github.com/aws/aws-sdk-go v1.55.8 h1:JRmEUbU52aJQZ2AjX4q4Wu7t4uZjOu71uyNmaWlUkJQ=
|
||||
github.com/aws/aws-sdk-go v1.55.8/go.mod h1:ZkViS9AqA6otK+JBBNH2++sx1sgxrPKcSzPPvQkUtXk=
|
||||
github.com/aws/aws-sdk-go-v2 v1.41.6 h1:1AX0AthnBQzMx1vbmir3Y4WsnJgiydmnJjiLu+LvXOg=
|
||||
github.com/aws/aws-sdk-go-v2 v1.41.6/go.mod h1:dy0UzBIfwSeot4grGvY1AqFWN5zgziMmWGzysDnHFcQ=
|
||||
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.8 h1:eBMB84YGghSocM7PsjmmPffTa+1FBUeNvGvFou6V/4o=
|
||||
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.8/go.mod h1:lyw7GFp3qENLh7kwzf7iMzAxDn+NzjXEAGjKS2UOKqI=
|
||||
github.com/aws/aws-sdk-go-v2 v1.41.7 h1:DWpAJt66FmnnaRIOT/8ASTucrvuDPZASqhhLey6tLY8=
|
||||
github.com/aws/aws-sdk-go-v2 v1.41.7/go.mod h1:4LAfZOPHNVNQEckOACQx60Y8pSRjIkNZQz1w92xpMJc=
|
||||
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.10 h1:gx1AwW1Iyk9Z9dD9F4akX5gnN3QZwUB20GGKH/I+Rho=
|
||||
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.10/go.mod h1:qqY157uZoqm5OXq/amuaBJyC9hgBCBQnsaWnPe905GY=
|
||||
github.com/aws/aws-sdk-go-v2/config v1.32.14 h1:opVIRo/ZbbI8OIqSOKmpFaY7IwfFUOCCXBsUpJOwDdI=
|
||||
github.com/aws/aws-sdk-go-v2/config v1.32.14/go.mod h1:U4/V0uKxh0Tl5sxmCBZ3AecYny4UNlVmObYjKuuaiOo=
|
||||
github.com/aws/aws-sdk-go-v2/credentials v1.19.14 h1:n+UcGWAIZHkXzYt87uMFBv/l8THYELoX6gVcUvgl6fI=
|
||||
@@ -727,24 +727,24 @@ github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.18.21 h1:NUS3K4BTDArQqNu2ih7yeD
|
||||
github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.18.21/go.mod h1:YWNWJQNjKigKY1RHVJCuupeWDrrHjRqHm0N9rdrWzYI=
|
||||
github.com/aws/aws-sdk-go-v2/feature/s3/manager v1.22.13 h1:uMC4oL6G3MNhodo358QEqSDjrgvzV3TUQ58nyQSGq2E=
|
||||
github.com/aws/aws-sdk-go-v2/feature/s3/manager v1.22.13/go.mod h1:Cer86AE2686DvVUe57LPve3jUBmbujuaonSX8pNzGgw=
|
||||
github.com/aws/aws-sdk-go-v2/internal/configsources v1.4.21 h1:Rgg6wvjjtX8bNHcvi9OnXWwcE0a2vGpbwmtICOsvcf4=
|
||||
github.com/aws/aws-sdk-go-v2/internal/configsources v1.4.21/go.mod h1:A/kJFst/nm//cyqonihbdpQZwiUhhzpqTsdbhDdRF9c=
|
||||
github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.7.21 h1:PEgGVtPoB6NTpPrBgqSE5hE/o47Ij9qk/SEZFbUOe9A=
|
||||
github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.7.21/go.mod h1:p+hz+PRAYlY3zcpJhPwXlLC4C+kqn70WIHwnzAfs6ps=
|
||||
github.com/aws/aws-sdk-go-v2/internal/configsources v1.4.23 h1:GpT/TrnBYuE5gan2cZbTtvP+JlHsutdmlV2YfEyNde0=
|
||||
github.com/aws/aws-sdk-go-v2/internal/configsources v1.4.23/go.mod h1:xYWD6BS9ywC5bS3sz9Xh04whO/hzK2plt2Zkyrp4JuA=
|
||||
github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.7.23 h1:bpd8vxhlQi2r1hiueOw02f/duEPTMK59Q4QMAoTTtTo=
|
||||
github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.7.23/go.mod h1:15DfR2nw+CRHIk0tqNyifu3G1YdAOy68RftkhMDDwYk=
|
||||
github.com/aws/aws-sdk-go-v2/internal/ini v1.8.6 h1:qYQ4pzQ2Oz6WpQ8T3HvGHnZydA72MnLuFK9tJwmrbHw=
|
||||
github.com/aws/aws-sdk-go-v2/internal/ini v1.8.6/go.mod h1:O3h0IK87yXci+kg6flUKzJnWeziQUKciKrLjcatSNcY=
|
||||
github.com/aws/aws-sdk-go-v2/internal/v4a v1.4.22 h1:rWyie/PxDRIdhNf4DzRk0lvjVOqFJuNnO8WwaIRVxzQ=
|
||||
github.com/aws/aws-sdk-go-v2/internal/v4a v1.4.22/go.mod h1:zd/JsJ4P7oGfUhXn1VyLqaRZwPmZwg44Jf2dS84Dm3Y=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.7 h1:5EniKhLZe4xzL7a+fU3C2tfUN4nWIqlLesfrjkuPFTY=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.7/go.mod h1:x0nZssQ3qZSnIcePWLvcoFisRXJzcTVvYpAAdYX8+GI=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.13 h1:JRaIgADQS/U6uXDqlPiefP32yXTda7Kqfx+LgspooZM=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.13/go.mod h1:CEuVn5WqOMilYl+tbccq8+N2ieCy0gVn3OtRb0vBNNM=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.13.21 h1:c31//R3xgIJMSC8S6hEVq+38DcvUlgFY0FM6mSI5oto=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.13.21/go.mod h1:r6+pf23ouCB718FUxaqzZdbpYFyDtehyZcmP5KL9FkA=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.21 h1:ZlvrNcHSFFWURB8avufQq9gFsheUgjVD9536obIknfM=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.21/go.mod h1:cv3TNhVrssKR0O/xxLJVRfd2oazSnZnkUeTf6ctUwfQ=
|
||||
github.com/aws/aws-sdk-go-v2/service/s3 v1.99.0 h1:hlSuz394kV0vhv9drL5lhuEFbEOEP1VyQpy15qWh1Pk=
|
||||
github.com/aws/aws-sdk-go-v2/service/s3 v1.99.0/go.mod h1:uoA43SdFwacedBfSgfFSjjCvYe8aYBS7EnU5GZ/YKMM=
|
||||
github.com/aws/aws-sdk-go-v2/internal/v4a v1.4.24 h1:OQqn11BtaYv1WLUowvcA30MpzIu8Ti4pcLPIIyoKZrA=
|
||||
github.com/aws/aws-sdk-go-v2/internal/v4a v1.4.24/go.mod h1:X5ZJyfwVrWA96GzPmUCWFQaEARPR7gCrpq2E92PJwAE=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.9 h1:FLudkZLt5ci0ozzgkVo8BJGwvqNaZbTWb3UcucAateA=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.9/go.mod h1:w7wZ/s9qK7c8g4al+UyoF1Sp/Z45UwMGcqIzLWVQHWk=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.15 h1:ieLCO1JxUWuxTZ1cRd0GAaeX7O6cIxnwk7tc1LsQhC4=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.15/go.mod h1:e3IzZvQ3kAWNykvE0Tr0RDZCMFInMvhku3qNpcIQXhM=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.13.23 h1:pbrxO/kuIwgEsOPLkaHu0O+m4fNgLU8B3vxQ+72jTPw=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.13.23/go.mod h1:/CMNUqoj46HpS3MNRDEDIwcgEnrtZlKRaHNaHxIFpNA=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.23 h1:03xatSQO4+AM1lTAbnRg5OK528EUg744nW7F73U8DKw=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.23/go.mod h1:M8l3mwgx5ToK7wot2sBBce/ojzgnPzZXUV445gTSyE8=
|
||||
github.com/aws/aws-sdk-go-v2/service/s3 v1.101.0 h1:etqBTKY581iwLL/H/S2sVgk3C9lAsTJFeXWFDsDcWOU=
|
||||
github.com/aws/aws-sdk-go-v2/service/s3 v1.101.0/go.mod h1:L2dcoOgS2VSgbPLvpak2NyUPsO1TBN7M45Z4H7DlRc4=
|
||||
github.com/aws/aws-sdk-go-v2/service/signin v1.0.9 h1:QKZH0S178gCmFEgst8hN0mCX1KxLgHBKKY/CLqwP8lg=
|
||||
github.com/aws/aws-sdk-go-v2/service/signin v1.0.9/go.mod h1:7yuQJoT+OoH8aqIxw9vwF+8KpvLZ8AWmvmUWHsGQZvI=
|
||||
github.com/aws/aws-sdk-go-v2/service/sns v1.39.7 h1:fovS7qGMT+BBSuifkySdVaMWxXTyaYT6qaBx/1y6Ij4=
|
||||
@@ -866,8 +866,8 @@ github.com/cockroachdb/redact v1.1.5 h1:u1PMllDkdFfPWaNGMyLD1+so+aq3uUItthCFqzwP
|
||||
github.com/cockroachdb/redact v1.1.5/go.mod h1:BVNblN9mBWFyMyqK1k3AAiSxhvhfK2oOZZ2lK+dpvRg=
|
||||
github.com/cockroachdb/version v0.0.0-20250314144055-3860cd14adf2 h1:8Vfw2iNEpYIV6aLtMwT5UOGuPmp9MKlEKWKFTuB+MPU=
|
||||
github.com/cockroachdb/version v0.0.0-20250314144055-3860cd14adf2/go.mod h1:P9WiZOdQ1R/ZZDL0WzF5wlyRvrjtfhNOwMZymFpBwjE=
|
||||
github.com/cognusion/imaging v1.0.2 h1:BQwBV8V8eF3+dwffp8Udl9xF1JKh5Z0z5JkJwAi98Mc=
|
||||
github.com/cognusion/imaging v1.0.2/go.mod h1:mj7FvH7cT2dlFogQOSUQRtotBxJ4gFQ2ySMSmBm5dSk=
|
||||
github.com/cognusion/imaging v1.0.3 h1:nHyIeEVDV8JkBbuhgx8iSBW72W8rHbDEyruA0Jh7Lnk=
|
||||
github.com/cognusion/imaging v1.0.3/go.mod h1:38tFLFhGK81ORThZG8dVXPtp1uhd+3xM1WXuMn+unOA=
|
||||
github.com/colinmarc/hdfs/v2 v2.4.0 h1:v6R8oBx/Wu9fHpdPoJJjpGSUxo8NhHIwrwsfhFvU9W0=
|
||||
github.com/colinmarc/hdfs/v2 v2.4.0/go.mod h1:0NAO+/3knbMx6+5pCv+Hcbaz4xn/Zzbn9+WIib2rKVI=
|
||||
github.com/compose-spec/compose-go/v2 v2.9.0 h1:UHSv/QHlo6QJtrT4igF1rdORgIUhDo1gWuyJUoiNNIM=
|
||||
@@ -1128,8 +1128,8 @@ github.com/go-redsync/redsync/v4 v4.16.0 h1:bNcOzeHH9d3s6pghU9NJFMPrQa41f5Nx3L4Y
|
||||
github.com/go-redsync/redsync/v4 v4.16.0/go.mod h1:V4gagqgyASWBZuwx4xGzu72aZNb/6Mo05byUa3mVmKQ=
|
||||
github.com/go-resty/resty/v2 v2.17.2 h1:FQW5oHYcIlkCNrMD2lloGScxcHJ0gkjshV3qcQAyHQk=
|
||||
github.com/go-resty/resty/v2 v2.17.2/go.mod h1:kCKZ3wWmwJaNc7S29BRtUhJwy7iqmn+2mLtQrOyQlVA=
|
||||
github.com/go-sql-driver/mysql v1.9.3 h1:U/N249h2WzJ3Ukj8SowVFjdtZKfu9vlLZxjPXV1aweo=
|
||||
github.com/go-sql-driver/mysql v1.9.3/go.mod h1:qn46aNg1333BRMNU69Lq93t8du/dwxI64Gl8i5p1WMU=
|
||||
github.com/go-sql-driver/mysql v1.10.0 h1:Q+1LV8DkHJvSYAdR83XzuhDaTykuDx0l6fkXxoWCWfw=
|
||||
github.com/go-sql-driver/mysql v1.10.0/go.mod h1:M+cqaI7+xxXGG9swrdeUIoPG3Y3KCkF0pZej+SK+nWk=
|
||||
github.com/go-stack/stack v1.8.0/go.mod h1:v0f6uXyyMGvRgIKkXu+yp6POWl0qKG85gN/melR3HDY=
|
||||
github.com/go-task/slim-sprig v0.0.0-20230315185526-52ccab3ef572 h1:tfuBGBXKqDEevZMzYi5KSi8KkcZtzBcTgAUUtapy0OI=
|
||||
github.com/go-task/slim-sprig/v3 v3.0.0 h1:sUs3vkvUymDpBKi3qH1YSqBQk9+9D/8M2mN1vB6EwHI=
|
||||
@@ -1515,8 +1515,8 @@ github.com/lib/pq v1.11.1 h1:wuChtj2hfsGmmx3nf1m7xC2XpK6OtelS2shMY+bGMtI=
|
||||
github.com/lib/pq v1.11.1/go.mod h1:/p+8NSbOcwzAEI7wiMXFlgydTwcgTr3OSKMsD2BitpA=
|
||||
github.com/linkedin/goavro/v2 v2.15.0 h1:pDj1UrjUOO62iXhgBiE7jQkpNIc5/tA5eZsgolMjgVI=
|
||||
github.com/linkedin/goavro/v2 v2.15.0/go.mod h1:KXx+erlq+RPlGSPmLF7xGo6SAbh8sCQ53x064+ioxhk=
|
||||
github.com/linxGnu/grocksdb v1.10.7 h1:fCi4qvZWo04VgFwGWmO8HQJgUVounJBy+C2TMVPU/ho=
|
||||
github.com/linxGnu/grocksdb v1.10.7/go.mod h1:OLQKZwiKwaJiAVCsOzWKvwiLwfZ5Vz8Md5TYR7t7pM8=
|
||||
github.com/linxGnu/grocksdb v1.10.8 h1:Nau01Hhm/0kaVTR6d4viwD6npYbnDvZAfzwJCLzKRYo=
|
||||
github.com/linxGnu/grocksdb v1.10.8/go.mod h1:OLQKZwiKwaJiAVCsOzWKvwiLwfZ5Vz8Md5TYR7t7pM8=
|
||||
github.com/lithammer/fuzzysearch v1.1.8 h1:/HIuJnjHuXS8bKaiTMeeDlW2/AyIWk2brx1V8LFgLN4=
|
||||
github.com/lithammer/fuzzysearch v1.1.8/go.mod h1:IdqeyBClc3FFqSzYq/MXESsS4S0FsZ5ajtkr5xPLts4=
|
||||
github.com/lithammer/shortuuid/v3 v3.0.7 h1:trX0KTHy4Pbwo/6ia8fscyHoGA+mf1jWbPJVuvyJQQ8=
|
||||
@@ -1667,8 +1667,8 @@ github.com/parquet-go/bitpack v1.0.0 h1:AUqzlKzPPXf2bCdjfj4sTeacrUwsT7NlcYDMUQxP
|
||||
github.com/parquet-go/bitpack v1.0.0/go.mod h1:XnVk9TH+O40eOOmvpAVZ7K2ocQFrQwysLMnc6M/8lgs=
|
||||
github.com/parquet-go/jsonlite v1.0.0 h1:87QNdi56wOfsE5bdgas0vRzHPxfJgzrXGml1zZdd7VU=
|
||||
github.com/parquet-go/jsonlite v1.0.0/go.mod h1:nDjpkpL4EOtqs6NQugUsi0Rleq9sW/OtC1NnZEnxzF0=
|
||||
github.com/parquet-go/parquet-go v0.28.0 h1:ECyksyv8T2pOrlLsN7aWJIoQakyk/HtxQ2lchgS4els=
|
||||
github.com/parquet-go/parquet-go v0.28.0/go.mod h1:navtkAYr2LGoJVp141oXPlO/sxLvaOe3la2JEoD8+rg=
|
||||
github.com/parquet-go/parquet-go v0.30.1 h1:Oy6ganNrAdFiVwy7wNmWagfPTWA2X9Z3tVHBc7JtuX8=
|
||||
github.com/parquet-go/parquet-go v0.30.1/go.mod h1:navtkAYr2LGoJVp141oXPlO/sxLvaOe3la2JEoD8+rg=
|
||||
github.com/pascaldekloe/goe v0.1.0 h1:cBOtyMzM9HTpWjXfbbunk26uA6nG3a8n06Wieeh0MwY=
|
||||
github.com/pascaldekloe/goe v0.1.0/go.mod h1:lzWF7FIEvWOWxwDKqyGYQf6ZUaNfKdP144TG7ZOy1lc=
|
||||
github.com/patrickmn/go-cache v2.1.0+incompatible h1:HRMgzkcYKYpi3C8ajMPV8OFXaaRUnok+kx1WdO15EQc=
|
||||
@@ -1795,8 +1795,8 @@ github.com/rcrowley/go-metrics v0.0.0-20201227073835-cf1acfcdf475 h1:N/ElC8H3+5X
|
||||
github.com/rcrowley/go-metrics v0.0.0-20201227073835-cf1acfcdf475/go.mod h1:bCqnVzQkZxMG4s8nGwiZ5l3QUCyqpo9Y+/ZMZ9VjZe4=
|
||||
github.com/rdleal/intervalst v1.5.0 h1:SEB9bCFz5IqD1yhfH1Wv8IBnY/JQxDplwkxHjT6hamU=
|
||||
github.com/rdleal/intervalst v1.5.0/go.mod h1:xO89Z6BC+LQDH+IPQQw/OESt5UADgFD41tYMUINGpxQ=
|
||||
github.com/redis/go-redis/v9 v9.18.0 h1:pMkxYPkEbMPwRdenAzUNyFNrDgHx9U+DrBabWNfSRQs=
|
||||
github.com/redis/go-redis/v9 v9.18.0/go.mod h1:k3ufPphLU5YXwNTUcCRXGxUoF1fqxnhFQmscfkCoDA0=
|
||||
github.com/redis/go-redis/v9 v9.19.0 h1:XPVaaPSnG6RhYf7p+rmSa9zZfeVAnWsH5h3lxthOm/k=
|
||||
github.com/redis/go-redis/v9 v9.19.0/go.mod h1:v/M13XI1PVCDcm01VtPFOADfZtHf8YW3baQf57KlIkA=
|
||||
github.com/redis/rueidis v1.0.71 h1:pODtnAR5GAB7j4ekhldZ29HKOxe4Hph0GTDGk1ayEQY=
|
||||
github.com/redis/rueidis v1.0.71/go.mod h1:lfdcZzJ1oKGKL37vh9fO3ymwt+0TdjkkUCJxbgpmcgQ=
|
||||
github.com/redis/rueidis/rueidiscompat v1.0.71 h1:wNZ//kEjMZgBM0KCk7ncOX8KmAgROU2kDdDNpwheG4w=
|
||||
@@ -2111,8 +2111,8 @@ go.etcd.io/bbolt v1.4.3 h1:dEadXpI6G79deX5prL3QRNP6JB8UxVkqo4UPnHaNXJo=
|
||||
go.etcd.io/bbolt v1.4.3/go.mod h1:tKQlpPaYCVFctUIgFKFnAlvbmB3tpy1vkTnDWohtc0E=
|
||||
go.etcd.io/etcd/api/v3 v3.6.10 h1:jlwjtELjA8yi2VWpOFH+0w0lGr3K6mVDyn0RDB9aaAY=
|
||||
go.etcd.io/etcd/api/v3 v3.6.10/go.mod h1:pdV4VeFmvhdNjB4LWRkC8ReLyRBAxUOze3GarMhE2sk=
|
||||
go.etcd.io/etcd/client/pkg/v3 v3.6.10 h1:tBT7podcPhuVbCVkAEzx8bC5I+aqxfLwBN8/As1arrA=
|
||||
go.etcd.io/etcd/client/pkg/v3 v3.6.10/go.mod h1:WEy3PpwbbEBVRdh1NVJYsuUe/8eyI21PNJRazeD8z/Y=
|
||||
go.etcd.io/etcd/client/pkg/v3 v3.6.11 h1:e41mp315Yn3QMGPmEzCyLsMINgJXTY/dX8kM++1csxU=
|
||||
go.etcd.io/etcd/client/pkg/v3 v3.6.11/go.mod h1:DysuMe/inqRyC/1tjRR6hReH/VV9Lufs27YKSKBWWJg=
|
||||
go.etcd.io/etcd/client/v3 v3.6.10 h1:J598zJ+C/ZPvImypmq5waj84+bovePrlZERHklf34y0=
|
||||
go.etcd.io/etcd/client/v3 v3.6.10/go.mod h1:iHhUDUcEwaKs1YFq3MgmI9U4zhTVasp/vgdVbFf1RS8=
|
||||
go.mongodb.org/mongo-driver v1.17.9 h1:IexDdCuuNJ3BHrELgBlyaH9p60JXAvdzWR128q+U5tU=
|
||||
@@ -2216,8 +2216,8 @@ golang.org/x/crypto v0.14.0/go.mod h1:MVFd36DqK4CsrnJYDkBA3VC4m2GkXAM0PvzMCn4JQf
|
||||
golang.org/x/crypto v0.19.0/go.mod h1:Iy9bg/ha4yyC70EfRS8jz+B6ybOBKMaSxLj6P6oBDfU=
|
||||
golang.org/x/crypto v0.23.0/go.mod h1:CKFgDieR+mRhux2Lsu27y0fO304Db0wZe70UKqHu0v8=
|
||||
golang.org/x/crypto v0.31.0/go.mod h1:kDsLvtWBEx7MV9tJOj9bnXsPbxwJQ6csT/x4KIN4Ssk=
|
||||
golang.org/x/crypto v0.50.0 h1:zO47/JPrL6vsNkINmLoo/PH1gcxpls50DNogFvB5ZGI=
|
||||
golang.org/x/crypto v0.50.0/go.mod h1:3muZ7vA7PBCE6xgPX7nkzzjiUq87kRItoJQM1Yo8S+Q=
|
||||
golang.org/x/crypto v0.52.0 h1:RMs7fP2rXdep0CftQlK8Uf+kibLm7qkCcradZWYz988=
|
||||
golang.org/x/crypto v0.52.0/go.mod h1:1QgfPxDqh0T2M/elOJtp9RvuR95kVjir0e6/BvEmGbc=
|
||||
golang.org/x/exp v0.0.0-20180321215751-8460e604b9de/go.mod h1:CJ0aWSM057203Lf6IL+f9T1iT9GByDxfZKAQTCR3kQA=
|
||||
golang.org/x/exp v0.0.0-20180807140117-3d87b88a115f/go.mod h1:CJ0aWSM057203Lf6IL+f9T1iT9GByDxfZKAQTCR3kQA=
|
||||
golang.org/x/exp v0.0.0-20190121172915-509febef88a4/go.mod h1:CJ0aWSM057203Lf6IL+f9T1iT9GByDxfZKAQTCR3kQA=
|
||||
@@ -2352,8 +2352,8 @@ golang.org/x/net v0.16.0/go.mod h1:NxSsAGuq816PNPmqtQdLE42eU2Fs7NoRIZrHJAlaCOE=
|
||||
golang.org/x/net v0.21.0/go.mod h1:bIjVDfnllIU7BJ2DNgfnXvpSvtn8VRwhlsaeUTyUS44=
|
||||
golang.org/x/net v0.25.0/go.mod h1:JkAGAh7GEvH74S6FOH42FLoXpXbE/aqXSrIQjXgsiwM=
|
||||
golang.org/x/net v0.33.0/go.mod h1:HXLR5J+9DxmrqMwG9qjGCxZ+zKXxBru04zlTvWlWuN4=
|
||||
golang.org/x/net v0.53.0 h1:d+qAbo5L0orcWAr0a9JweQpjXF19LMXJE8Ey7hwOdUA=
|
||||
golang.org/x/net v0.53.0/go.mod h1:JvMuJH7rrdiCfbeHoo3fCQU24Lf5JJwT9W3sJFulfgs=
|
||||
golang.org/x/net v0.54.0 h1:2zJIZAxAHV/OHCDTCOHAYehQzLfSXuf/5SoL/Dv6w/w=
|
||||
golang.org/x/net v0.54.0/go.mod h1:Sj4oj8jK6XmHpBZU/zWHw3BV3abl4Kvi+Ut7cQcY+cQ=
|
||||
golang.org/x/oauth2 v0.0.0-20180821212333-d2e6202438be/go.mod h1:N/0e6XlmueqKjAGxoOufVs8QHGRruUQn6yWY3a++T0U=
|
||||
golang.org/x/oauth2 v0.0.0-20190226205417-e64efc72b421/go.mod h1:gOpvHmFTYa4IltrdGE7lF6nIHvwfUNPOp7c8zoXwtLw=
|
||||
golang.org/x/oauth2 v0.0.0-20190604053449-0f29369cfe45/go.mod h1:gOpvHmFTYa4IltrdGE7lF6nIHvwfUNPOp7c8zoXwtLw=
|
||||
@@ -2511,8 +2511,8 @@ golang.org/x/sys v0.13.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.17.0/go.mod h1:/VUhepiaJMQUp4+oa/7Zr1D23ma6VTLIYjOOTFZPUcA=
|
||||
golang.org/x/sys v0.20.0/go.mod h1:/VUhepiaJMQUp4+oa/7Zr1D23ma6VTLIYjOOTFZPUcA=
|
||||
golang.org/x/sys v0.28.0/go.mod h1:/VUhepiaJMQUp4+oa/7Zr1D23ma6VTLIYjOOTFZPUcA=
|
||||
golang.org/x/sys v0.43.0 h1:Rlag2XtaFTxp19wS8MXlJwTvoh8ArU6ezoyFsMyCTNI=
|
||||
golang.org/x/sys v0.43.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw=
|
||||
golang.org/x/sys v0.45.0 h1:dO4czNzziLiiXplLQgBCEpCvXQ3dnkn0SdaZSYdQ+FY=
|
||||
golang.org/x/sys v0.45.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw=
|
||||
golang.org/x/telemetry v0.0.0-20240228155512-f48c80bd79b2/go.mod h1:TeRTkGYfJXctD9OcfyVLyj2J3IxLnKwHJR8f4D8a3YE=
|
||||
golang.org/x/telemetry v0.0.0-20260409153401-be6f6cb8b1fa h1:efT73AJZfAAUV7SOip6pWGkwJDzIGiKBZGVzHYa+ve4=
|
||||
golang.org/x/telemetry v0.0.0-20260409153401-be6f6cb8b1fa/go.mod h1:kHjTxDEnAu6/Nl9lDkzjWpR+bmKfxeiRuSDlsMb70gE=
|
||||
@@ -2531,8 +2531,8 @@ golang.org/x/term v0.13.0/go.mod h1:LTmsnFJwVN6bCy1rVCoS+qHT1HhALEFxKncY3WNNh4U=
|
||||
golang.org/x/term v0.17.0/go.mod h1:lLRBjIVuehSbZlaOtGMbcMncT+aqLLLmKrsjNrUguwk=
|
||||
golang.org/x/term v0.20.0/go.mod h1:8UkIAJTvZgivsXaD6/pH6U9ecQzZ45awqEOzuCvwpFY=
|
||||
golang.org/x/term v0.27.0/go.mod h1:iMsnZpn0cago0GOrHO2+Y7u7JPn5AylBrcoWkElMTSM=
|
||||
golang.org/x/term v0.42.0 h1:UiKe+zDFmJobeJ5ggPwOshJIVt6/Ft0rcfrXZDLWAWY=
|
||||
golang.org/x/term v0.42.0/go.mod h1:Dq/D+snpsbazcBG5+F9Q1n2rXV8Ma+71xEjTRufARgY=
|
||||
golang.org/x/term v0.43.0 h1:S4RLU2sB31O/NCl+zFN9Aru9A/Cq2aqKpTZJ6B+DwT4=
|
||||
golang.org/x/term v0.43.0/go.mod h1:lrhlHNdQJHO+1qVYiHfFKVuVioJIheAc3fBSMFYEIsk=
|
||||
golang.org/x/text v0.0.0-20170915032832-14c0d48ead0c/go.mod h1:NqM8EUOU14njkJ3fqMW+pc6Ldnwhi/IjpwHt7yyuwOQ=
|
||||
golang.org/x/text v0.3.0/go.mod h1:NqM8EUOU14njkJ3fqMW+pc6Ldnwhi/IjpwHt7yyuwOQ=
|
||||
golang.org/x/text v0.3.1-0.20180807135948-17ff2d5776d2/go.mod h1:NqM8EUOU14njkJ3fqMW+pc6Ldnwhi/IjpwHt7yyuwOQ=
|
||||
@@ -2553,8 +2553,8 @@ golang.org/x/text v0.13.0/go.mod h1:TvPlkZtksWOMsz7fbANvkp4WM8x/WCo/om8BMLbz+aE=
|
||||
golang.org/x/text v0.14.0/go.mod h1:18ZOQIKpY8NJVqYksKHtTdi31H5itFRjB5/qKTNYzSU=
|
||||
golang.org/x/text v0.15.0/go.mod h1:18ZOQIKpY8NJVqYksKHtTdi31H5itFRjB5/qKTNYzSU=
|
||||
golang.org/x/text v0.21.0/go.mod h1:4IBbMaMmOPCJ8SecivzSH54+73PCFmPWxNTLm+vZkEQ=
|
||||
golang.org/x/text v0.36.0 h1:JfKh3XmcRPqZPKevfXVpI1wXPTqbkE5f7JA92a55Yxg=
|
||||
golang.org/x/text v0.36.0/go.mod h1:NIdBknypM8iqVmPiuco0Dh6P5Jcdk8lJL0CUebqK164=
|
||||
golang.org/x/text v0.37.0 h1:Cqjiwd9eSg8e0QAkyCaQTNHFIIzWtidPahFWR83rTrc=
|
||||
golang.org/x/text v0.37.0/go.mod h1:a5sjxXGs9hsn/AJVwuElvCAo9v8QYLzvavO5z2PiM38=
|
||||
golang.org/x/time v0.0.0-20181108054448-85acf8d2951c/go.mod h1:tRJNPiyCQ0inRvYxbN9jk5I+vvW/OXSQhTDSoE431IQ=
|
||||
golang.org/x/time v0.0.0-20190308202827-9d24e82272b4/go.mod h1:tRJNPiyCQ0inRvYxbN9jk5I+vvW/OXSQhTDSoE431IQ=
|
||||
golang.org/x/time v0.0.0-20191024005414-555d28b269f0/go.mod h1:tRJNPiyCQ0inRvYxbN9jk5I+vvW/OXSQhTDSoE431IQ=
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
apiVersion: v1
|
||||
description: SeaweedFS
|
||||
name: seaweedfs
|
||||
appVersion: "4.26"
|
||||
appVersion: "4.29"
|
||||
# Dev note: Trigger a helm chart release by `git tag -a helm-<version>`
|
||||
version: 4.26.0
|
||||
version: 4.29.0
|
||||
|
||||
@@ -137,6 +137,9 @@ spec:
|
||||
- "/bin/sh"
|
||||
- "-ec"
|
||||
- |
|
||||
{{- if $volume.rust }}
|
||||
exec /usr/bin/weed-volume \
|
||||
{{- else }}
|
||||
exec /usr/bin/weed \
|
||||
{{- if $volume.logs }}
|
||||
-logdir=/logs \
|
||||
@@ -149,6 +152,7 @@ spec:
|
||||
-v={{ $.Values.global.seaweedfs.loggingLevel }} \
|
||||
{{- end }}
|
||||
volume \
|
||||
{{- end }}
|
||||
-port={{ $volume.port }} \
|
||||
{{- if $volume.metricsPort }}
|
||||
-metricsPort={{ $volume.metricsPort }} \
|
||||
@@ -180,7 +184,7 @@ spec:
|
||||
{{- if $volume.imagesFixOrientation }}
|
||||
-images.fix.orientation \
|
||||
{{- end }}
|
||||
{{- if $volume.pulseSeconds }}
|
||||
{{- if and $volume.pulseSeconds (not $volume.rust) }}
|
||||
-pulseSeconds={{ $volume.pulseSeconds }} \
|
||||
{{- end }}
|
||||
{{- if $volume.index }}
|
||||
|
||||
@@ -305,6 +305,11 @@ volume:
|
||||
enabled: true
|
||||
imageOverride: null
|
||||
restartPolicy: null
|
||||
# Run the Rust volume server (/usr/bin/weed-volume) instead of the Go one.
|
||||
# Requires an image that ships the Rust binary (amd64/arm64). The Go-only
|
||||
# log flags (-logtostderr/-logdir/-v) and -pulseSeconds are dropped; set log
|
||||
# level via the RUST_LOG env var in extraEnvironmentVars if needed.
|
||||
rust: false
|
||||
port: 8080
|
||||
grpcPort: 18080
|
||||
metricsPort: 9327
|
||||
|
||||
@@ -22,12 +22,24 @@ service SeaweedFiler {
|
||||
rpc UpdateEntry (UpdateEntryRequest) returns (UpdateEntryResponse) {
|
||||
}
|
||||
|
||||
rpc TouchAccessTime (TouchAccessTimeRequest) returns (TouchAccessTimeResponse) {
|
||||
}
|
||||
|
||||
rpc AppendToEntry (AppendToEntryRequest) returns (AppendToEntryResponse) {
|
||||
}
|
||||
|
||||
rpc DeleteEntry (DeleteEntryRequest) returns (DeleteEntryResponse) {
|
||||
}
|
||||
|
||||
rpc ObjectTransaction (ObjectTransactionRequest) returns (ObjectTransactionResponse) {
|
||||
}
|
||||
|
||||
rpc ObjectTransactionBatch (ObjectTransactionBatchRequest) returns (ObjectTransactionBatchResponse) {
|
||||
}
|
||||
|
||||
rpc PosixLock (PosixLockRequest) returns (PosixLockResponse) {
|
||||
}
|
||||
|
||||
rpc AtomicRenameEntry (AtomicRenameEntryRequest) returns (AtomicRenameEntryResponse) {
|
||||
}
|
||||
rpc StreamRenameEntry (StreamRenameEntryRequest) returns (stream StreamRenameEntryResponse) {
|
||||
@@ -208,6 +220,8 @@ message FuseAttributes {
|
||||
int32 mtime_ns = 19; // nanosecond component of mtime (0-999999999)
|
||||
int32 ctime_ns = 20; // nanosecond component of ctime (0-999999999)
|
||||
int32 crtime_ns = 21; // nanosecond component of crtime (0-999999999)
|
||||
int64 atime = 22; // unix time in seconds, last access time
|
||||
int32 atime_ns = 23; // nanosecond component of atime (0-999999999)
|
||||
}
|
||||
|
||||
message CreateEntryRequest {
|
||||
@@ -217,6 +231,56 @@ message CreateEntryRequest {
|
||||
bool is_from_other_cluster = 4;
|
||||
repeated int32 signatures = 5;
|
||||
bool skip_check_parent_directory = 6;
|
||||
// Optional precondition evaluated against the current entry atomically with
|
||||
// the write, under the filer's per-path lock. The caller must route the
|
||||
// key's writes to this entry's owner filer for the check to be authoritative.
|
||||
WriteCondition condition = 7;
|
||||
}
|
||||
|
||||
// WriteCondition is the precondition the filer evaluates against the existing
|
||||
// entry before writing, under the per-path lock. A failed condition returns
|
||||
// FilerError PRECONDITION_FAILED. The client maps request semantics (e.g. RFC
|
||||
// 7232) to clauses; the filer just compares.
|
||||
//
|
||||
// A condition is a list of clauses that ALL must hold (logical AND). One clause
|
||||
// is the common case; several express what a single comparison cannot: an ETag
|
||||
// set (If-Match / If-None-Match with multiple values), weak-ETag comparison, and
|
||||
// compound conditions (e.g. If-Match + If-Unmodified-Since together).
|
||||
message WriteCondition {
|
||||
enum Kind {
|
||||
NONE = 0; // unconditional
|
||||
IF_NOT_EXISTS = 1; // fail if the entry exists (If-None-Match: *)
|
||||
IF_EXISTS = 2; // fail if the entry is absent (If-Match: *)
|
||||
IF_ETAG_MATCH = 3; // fail if absent or etag matches none of the set (If-Match)
|
||||
IF_ETAG_NOT_MATCH = 4; // fail if present and etag matches any of the set (If-None-Match)
|
||||
IF_UNMODIFIED_SINCE = 5; // fail if present and mtime > unix_time
|
||||
IF_MODIFIED_SINCE = 6; // fail if present and mtime <= unix_time
|
||||
IF_EXTENDED_NOT_EQUAL = 7; // fail if present and extended[ext_key] == ext_value
|
||||
IF_EXTENDED_TIME_ELAPSED = 8; // fail if present and extended[ext_key] (unix seconds) is in the future
|
||||
}
|
||||
// Clause is one primitive comparison. IF_ETAG_MATCH holds when the current
|
||||
// entry's ETag equals any value in etags; IF_ETAG_NOT_MATCH holds when it
|
||||
// equals none. allow_weak permits weak-comparison (ignoring the W/ prefix).
|
||||
//
|
||||
// The IF_EXTENDED_* kinds are generic guards on an extended attribute, used
|
||||
// to enforce object-lock without teaching the filer S3 semantics:
|
||||
// IF_EXTENDED_NOT_EQUAL expresses a legal hold (block while a key equals a
|
||||
// value), and IF_EXTENDED_TIME_ELAPSED expresses retention (block while a
|
||||
// stored unix-second deadline is in the future, compared to the filer's
|
||||
// clock). The caller composes these and, for governance-bypass, simply omits
|
||||
// the retention clause when the bypass is authorized — the filer makes no
|
||||
// authorization decision.
|
||||
message Clause {
|
||||
Kind kind = 1;
|
||||
repeated string etags = 2; // ETag set for IF_ETAG_* kinds
|
||||
int64 unix_time = 3; // bound (unix seconds) for IF_*_SINCE kinds
|
||||
bool allow_weak = 4; // compare ETags ignoring the weak (W/) marker
|
||||
string ext_key = 5; // extended attribute name for IF_EXTENDED_* kinds
|
||||
string ext_value = 6; // blocking value for IF_EXTENDED_NOT_EQUAL
|
||||
string gate_key = 7; // IF_EXTENDED_TIME_ELAPSED: only enforce when extended[gate_key] == gate_value
|
||||
string gate_value = 8; // gate value (e.g. retention mode COMPLIANCE for governance bypass)
|
||||
}
|
||||
repeated Clause clauses = 1; // all must hold (logical AND)
|
||||
}
|
||||
|
||||
// Structured error codes for filer entry operations.
|
||||
@@ -228,6 +292,132 @@ enum FilerError {
|
||||
EXISTING_IS_DIRECTORY = 3; // cannot overwrite directory with file
|
||||
EXISTING_IS_FILE = 4; // cannot overwrite file with directory
|
||||
ENTRY_ALREADY_EXISTS = 5; // O_EXCL and entry already exists
|
||||
PRECONDITION_FAILED = 6; // WriteCondition not satisfied
|
||||
}
|
||||
|
||||
// ObjectMutation is one entry-level change applied by ObjectTransaction. All
|
||||
// mutations of a transaction run under a single per-path lock (the request's
|
||||
// lock_key) and in order, so the gateway can describe a multi-entry object
|
||||
// operation as one request instead of holding a distributed lock across
|
||||
// several RPCs. Data-bearing writes (entries with chunks) should be written
|
||||
// before the transaction; mutations here are metadata-scoped.
|
||||
message ObjectMutation {
|
||||
enum Type {
|
||||
PUT = 0; // create or replace the entry (entry field)
|
||||
DELETE = 1; // delete the entry at directory/name (no error if absent)
|
||||
PATCH_EXTENDED = 2; // merge set_extended / remove delete_extended on the entry
|
||||
RECOMPUTE_LATEST = 3; // scan a directory and re-point a parent entry (recompute)
|
||||
}
|
||||
Type type = 1;
|
||||
string directory = 2;
|
||||
string name = 3; // entry name for DELETE / PATCH_EXTENDED / RECOMPUTE_LATEST (the pointer entry)
|
||||
Entry entry = 4; // full entry for PUT
|
||||
map<string, bytes> set_extended = 5; // PATCH_EXTENDED: keys to set
|
||||
repeated string delete_extended = 6; // PATCH_EXTENDED: keys to remove
|
||||
bool is_delete_data = 7; // DELETE: also delete chunk data
|
||||
bool is_recursive = 8; // DELETE: recurse into a directory
|
||||
Recompute recompute = 9; // RECOMPUTE_LATEST parameters
|
||||
bool set_content = 10; // PATCH_EXTENDED: replace Entry.content with content
|
||||
bytes content = 11; // PATCH_EXTENDED: new Entry.content when set_content
|
||||
bool touch_mtime = 12; // PATCH_EXTENDED: set the entry's Mtime to now (e.g. a metadata-replace copy)
|
||||
}
|
||||
|
||||
// Recompute re-derives a pointer entry (directory/name on the mutation) from the
|
||||
// current contents of a scanned directory, atomically under the transaction's
|
||||
// lock. It is mechanical: the filer picks the child that sorts first or last by
|
||||
// name and copies the requested fields into the pointer; it has no knowledge of
|
||||
// what the entries mean. The caller (which does know the versioning scheme)
|
||||
// supplies the sort direction and the key mappings. This covers re-pointing the
|
||||
// latest version after a specific version is deleted, where the scan must run
|
||||
// under the lock.
|
||||
message Recompute {
|
||||
string scan_dir = 1; // directory whose direct children are scanned
|
||||
bool descending = 2; // pick the child that sorts last by name (else first)
|
||||
map<string, string> copy_extended = 3; // pointer extended key -> source extended key on the chosen child
|
||||
string name_to_key = 4; // if set, store the chosen child's name under this pointer key
|
||||
string size_to_key = 5; // if set, store the chosen child's FileSize (decimal) under this pointer key
|
||||
string mtime_to_key = 6; // if set, store the chosen child's Mtime (decimal) under this pointer key
|
||||
string demote_key = 7; // if set, stamp demote_value on the prior name_to_key target when it changes
|
||||
bytes demote_value = 8; // value for demote_key
|
||||
string exclude_name = 9; // if set, skip this child when scanning (e.g. a version about to be deleted)
|
||||
}
|
||||
|
||||
// ObjectTransactionRequest applies an ordered list of mutations atomically with
|
||||
// respect to other writers of the same object, by holding the filer's per-path
|
||||
// lock on lock_key for the whole transaction. The optional condition is checked
|
||||
// first, against condition_key when set, else lock_key. Callers set route_key to
|
||||
// the object's stable owner ring key; a filer that is not the owner forwards the
|
||||
// transaction one hop to the owner, so a stale ring view is tolerated.
|
||||
message ObjectTransactionRequest {
|
||||
string lock_key = 1; // object path to lock and to evaluate the condition against
|
||||
WriteCondition condition = 2; // optional precondition, checked under the lock
|
||||
repeated ObjectMutation mutations = 3;
|
||||
bool is_from_other_cluster = 4;
|
||||
repeated int32 signatures = 5;
|
||||
string condition_key = 6; // if set, evaluate the condition against this entry instead of lock_key (still locking lock_key)
|
||||
string route_key = 7; // ring key identifying the owner filer; a non-owner forwards the whole transaction to it
|
||||
bool is_moved = 8; // set on a forwarded transaction so the receiver applies it locally instead of forwarding again
|
||||
}
|
||||
|
||||
message ObjectTransactionResponse {
|
||||
string error = 1;
|
||||
FilerError error_code = 2;
|
||||
}
|
||||
|
||||
// PosixLockRange is one advisory byte-range lock. Owner identity is (sid, owner):
|
||||
// sid is the mount session, owner the FUSE lock owner within it, so owners from
|
||||
// different mounts never alias. end is inclusive (max uint64 = to EOF); is_flock
|
||||
// separates the flock and fcntl namespaces, which never conflict.
|
||||
message PosixLockRange {
|
||||
uint64 start = 1;
|
||||
uint64 end = 2;
|
||||
uint32 type = 3; // 1=read, 2=write, 3=unlock
|
||||
uint64 sid = 4;
|
||||
uint64 owner = 5;
|
||||
uint32 pid = 6; // holder pid, for get_lk reporting only
|
||||
bool is_flock = 7;
|
||||
}
|
||||
|
||||
// PosixLock routes an advisory lock operation to the inode's owner filer, which
|
||||
// holds the authoritative in-memory lock table. key is the inode identity ring
|
||||
// key (the file path, or hl:<HardLinkId> for a hardlink) used both to resolve the
|
||||
// owner and to index the table. A non-owner filer forwards the request one hop;
|
||||
// is_moved bounds it so a stale ring view cannot loop.
|
||||
message PosixLockRequest {
|
||||
string key = 1;
|
||||
bool is_moved = 2;
|
||||
PosixLockOp op = 3;
|
||||
PosixLockRange lock = 4;
|
||||
repeated PosixLockRange locks = 5;
|
||||
bool cooling_probe = 6;
|
||||
}
|
||||
|
||||
enum PosixLockOp {
|
||||
TRY_LOCK = 0; // grant lock or report conflict (non-blocking)
|
||||
UNLOCK = 1; // release lock's owner's locks over its range
|
||||
GET_LK = 2; // report a conflicting lock, if any
|
||||
RELEASE_POSIX_OWNER = 3; // drop the owner's fcntl locks (flush-time)
|
||||
RELEASE_FLOCK_OWNER = 4; // drop the owner's flock locks (release-time)
|
||||
KEEP_ALIVE = 5; // renew the session's lease on this owner (lock.sid)
|
||||
}
|
||||
|
||||
message PosixLockResponse {
|
||||
bool granted = 1; // for TRY_LOCK: whether the lock was granted
|
||||
bool has_conflict = 2; // whether conflict is populated
|
||||
PosixLockRange conflict = 3; // the blocking lock (TRY_LOCK conflict / GET_LK result)
|
||||
}
|
||||
|
||||
// ObjectTransactionBatch applies several object transactions in one round trip,
|
||||
// each under its own per-path lock and independent of the others (no cross-key
|
||||
// atomicity). A caller groups keys that route to the same owner filer and sends
|
||||
// one batch per owner, e.g. for a multi-object delete. Each response is parallel
|
||||
// to its request.
|
||||
message ObjectTransactionBatchRequest {
|
||||
repeated ObjectTransactionRequest transactions = 1;
|
||||
}
|
||||
|
||||
message ObjectTransactionBatchResponse {
|
||||
repeated ObjectTransactionResponse responses = 1;
|
||||
}
|
||||
|
||||
message CreateEntryResponse {
|
||||
@@ -247,6 +437,16 @@ message UpdateEntryResponse {
|
||||
SubscribeMetadataResponse metadata_event = 1;
|
||||
}
|
||||
|
||||
message TouchAccessTimeRequest {
|
||||
string directory = 1;
|
||||
string name = 2;
|
||||
int64 client_atime_ns = 3; // nanoseconds since epoch; filer may override with relatime
|
||||
}
|
||||
message TouchAccessTimeResponse {
|
||||
int64 persisted_atime_ns = 1; // nanoseconds since epoch; 0 if no update was performed
|
||||
bool updated = 2;
|
||||
}
|
||||
|
||||
message AppendToEntryRequest {
|
||||
string directory = 1;
|
||||
string entry_name = 2;
|
||||
@@ -406,6 +606,7 @@ message SubscribeMetadataRequest {
|
||||
repeated string directories = 10; // exact directory to watch
|
||||
bool client_supports_batching = 11; // client can unpack SubscribeMetadataResponse.events
|
||||
bool client_supports_metadata_chunks = 12; // client can read log file chunks from volume servers
|
||||
bool client_supports_idle_heartbeat = 13; // server may send empty responses carrying the current time while the client is caught up
|
||||
}
|
||||
message SubscribeMetadataResponse {
|
||||
string directory = 1;
|
||||
|
||||
@@ -838,8 +838,6 @@ pub struct ReadQueryParams {
|
||||
pub response_content_disposition: Option<String>,
|
||||
/// Pretty print JSON response
|
||||
pub pretty: Option<String>,
|
||||
/// JSONP callback function name
|
||||
pub callback: Option<String>,
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
@@ -3402,10 +3400,6 @@ fn json_response_with_params<T: Serialize>(
|
||||
let is_pretty = params
|
||||
.and_then(|params| params.pretty.as_ref())
|
||||
.is_some_and(|value| !value.is_empty());
|
||||
let callback = params
|
||||
.and_then(|params| params.callback.as_ref())
|
||||
.filter(|value| !value.is_empty())
|
||||
.cloned();
|
||||
|
||||
let json_body = if is_pretty {
|
||||
to_pretty_json(body)
|
||||
@@ -3413,24 +3407,15 @@ fn json_response_with_params<T: Serialize>(
|
||||
serde_json::to_string(body).unwrap()
|
||||
};
|
||||
|
||||
if let Some(callback) = callback {
|
||||
Response::builder()
|
||||
.status(status)
|
||||
.header(header::CONTENT_TYPE, "application/javascript")
|
||||
.body(Body::from(format!("{}({})", callback, json_body)))
|
||||
.unwrap()
|
||||
} else {
|
||||
Response::builder()
|
||||
.status(status)
|
||||
.header(header::CONTENT_TYPE, "application/json")
|
||||
.body(Body::from(json_body))
|
||||
.unwrap()
|
||||
}
|
||||
Response::builder()
|
||||
.status(status)
|
||||
.header(header::CONTENT_TYPE, "application/json")
|
||||
.header("X-Content-Type-Options", "nosniff")
|
||||
.body(Body::from(json_body))
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
/// Return a JSON error response with optional query string for pretty/JSONP support.
|
||||
/// Supports `?pretty=<any non-empty value>` for pretty-printed JSON and `?callback=fn` for JSONP,
|
||||
/// matching Go's writeJsonError behavior.
|
||||
/// Return a JSON error response, honoring `?pretty=<any non-empty value>` for pretty-printed JSON.
|
||||
pub(super) fn json_error_with_query(
|
||||
status: StatusCode,
|
||||
msg: impl Into<String>,
|
||||
@@ -3438,18 +3423,10 @@ pub(super) fn json_error_with_query(
|
||||
) -> Response {
|
||||
let body = serde_json::json!({"error": msg.into()});
|
||||
|
||||
let (is_pretty, callback) = if let Some(q) = query {
|
||||
let pretty = q
|
||||
.split('&')
|
||||
.any(|p| p.starts_with("pretty=") && p.len() > "pretty=".len());
|
||||
let cb = q
|
||||
.split('&')
|
||||
.find_map(|p| p.strip_prefix("callback="))
|
||||
.map(|s| s.to_string());
|
||||
(pretty, cb)
|
||||
} else {
|
||||
(false, None)
|
||||
};
|
||||
let is_pretty = query.is_some_and(|q| {
|
||||
q.split('&')
|
||||
.any(|p| p.starts_with("pretty=") && p.len() > "pretty=".len())
|
||||
});
|
||||
|
||||
let json_body = if is_pretty {
|
||||
to_pretty_json(&body)
|
||||
@@ -3457,35 +3434,19 @@ pub(super) fn json_error_with_query(
|
||||
serde_json::to_string(&body).unwrap()
|
||||
};
|
||||
|
||||
if let Some(cb) = callback {
|
||||
let jsonp = format!("{}({})", cb, json_body);
|
||||
Response::builder()
|
||||
.status(status)
|
||||
.header(header::CONTENT_TYPE, "application/javascript")
|
||||
.body(Body::from(jsonp))
|
||||
.unwrap()
|
||||
} else {
|
||||
Response::builder()
|
||||
.status(status)
|
||||
.header(header::CONTENT_TYPE, "application/json")
|
||||
.body(Body::from(json_body))
|
||||
.unwrap()
|
||||
}
|
||||
Response::builder()
|
||||
.status(status)
|
||||
.header(header::CONTENT_TYPE, "application/json")
|
||||
.header("X-Content-Type-Options", "nosniff")
|
||||
.body(Body::from(json_body))
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
/// Return a JSON response with optional pretty/JSONP support from raw query string.
|
||||
/// Matches Go's writeJsonQuiet behavior for write success responses.
|
||||
/// Return a JSON response honoring `?pretty=<any non-empty value>` from a raw query string.
|
||||
fn json_result_with_query<T: Serialize>(status: StatusCode, body: &T, query: &str) -> Response {
|
||||
let (is_pretty, callback) = {
|
||||
let pretty = query
|
||||
.split('&')
|
||||
.any(|p| p.starts_with("pretty=") && p.len() > "pretty=".len());
|
||||
let cb = query
|
||||
.split('&')
|
||||
.find_map(|p| p.strip_prefix("callback="))
|
||||
.map(|s| s.to_string());
|
||||
(pretty, cb)
|
||||
};
|
||||
let is_pretty = query
|
||||
.split('&')
|
||||
.any(|p| p.starts_with("pretty=") && p.len() > "pretty=".len());
|
||||
|
||||
let json_body = if is_pretty {
|
||||
to_pretty_json(body)
|
||||
@@ -3493,20 +3454,12 @@ fn json_result_with_query<T: Serialize>(status: StatusCode, body: &T, query: &st
|
||||
serde_json::to_string(body).unwrap()
|
||||
};
|
||||
|
||||
if let Some(cb) = callback {
|
||||
let jsonp = format!("{}({})", cb, json_body);
|
||||
Response::builder()
|
||||
.status(status)
|
||||
.header(header::CONTENT_TYPE, "application/javascript")
|
||||
.body(Body::from(jsonp))
|
||||
.unwrap()
|
||||
} else {
|
||||
Response::builder()
|
||||
.status(status)
|
||||
.header(header::CONTENT_TYPE, "application/json")
|
||||
.body(Body::from(json_body))
|
||||
.unwrap()
|
||||
}
|
||||
Response::builder()
|
||||
.status(status)
|
||||
.header(header::CONTENT_TYPE, "application/json")
|
||||
.header("X-Content-Type-Options", "nosniff")
|
||||
.body(Body::from(json_body))
|
||||
.unwrap()
|
||||
}
|
||||
|
||||
/// Extract JWT token from query param, Authorization header, or Cookie.
|
||||
|
||||
@@ -212,6 +212,13 @@ impl CompactNeedleMap {
|
||||
self.idx_file_offset = offset;
|
||||
}
|
||||
|
||||
/// True when an .idx file writer is attached. A read-only load leaves
|
||||
/// this `false` — set_writable() must reattach a writer or subsequent
|
||||
/// puts silently skip the disk append.
|
||||
pub fn has_idx_writer(&self) -> bool {
|
||||
self.idx_file.is_some()
|
||||
}
|
||||
|
||||
// ---- Map operations ----
|
||||
|
||||
/// Insert or update an entry. Appends to .idx file if present.
|
||||
@@ -705,6 +712,11 @@ impl RedbNeedleMap {
|
||||
self.idx_file_offset = offset;
|
||||
}
|
||||
|
||||
/// True when an .idx file writer is attached. See CompactNeedleMap.
|
||||
pub fn has_idx_writer(&self) -> bool {
|
||||
self.idx_file.is_some()
|
||||
}
|
||||
|
||||
// ---- Map operations ----
|
||||
|
||||
/// Insert or update an entry. Writes to idx file first, then redb.
|
||||
@@ -1000,6 +1012,14 @@ impl NeedleMap {
|
||||
}
|
||||
}
|
||||
|
||||
/// True when an .idx file writer is attached.
|
||||
pub fn has_idx_writer(&self) -> bool {
|
||||
match self {
|
||||
NeedleMap::InMemory(nm) => nm.has_idx_writer(),
|
||||
NeedleMap::Redb(nm) => nm.has_idx_writer(),
|
||||
}
|
||||
}
|
||||
|
||||
/// Content byte count.
|
||||
pub fn content_size(&self) -> u64 {
|
||||
match self {
|
||||
|
||||
@@ -2122,7 +2122,34 @@ impl Volume {
|
||||
}
|
||||
|
||||
/// Mark this volume as writable (allow writes and deletes).
|
||||
///
|
||||
/// If the volume booted with .vif ReadOnly=true, `load_index` built the
|
||||
/// needle map without an .idx writer attached, so subsequent puts would
|
||||
/// silently skip the on-disk append and only mutate in-memory state —
|
||||
/// surviving until the next restart, then vanishing. Re-attach a writer
|
||||
/// here so writes persist again.
|
||||
pub fn set_writable(&mut self) -> Result<(), VolumeError> {
|
||||
// Attach the writer (if missing) before flipping the flag — otherwise
|
||||
// a transient open/metadata failure would leave the volume marked
|
||||
// writable with no .idx writer, and subsequent puts would silently
|
||||
// skip the on-disk append and vanish on the next restart.
|
||||
let needs_idx_writer = self
|
||||
.nm
|
||||
.as_ref()
|
||||
.map(|nm| !nm.has_idx_writer())
|
||||
.unwrap_or(false);
|
||||
if needs_idx_writer {
|
||||
let idx_path = self.file_name(".idx");
|
||||
let write_file = OpenOptions::new()
|
||||
.write(true)
|
||||
.append(true)
|
||||
.create(true)
|
||||
.open(&idx_path)?;
|
||||
let idx_size = write_file.metadata()?.len();
|
||||
if let Some(ref mut nm) = self.nm {
|
||||
nm.set_idx_file(Box::new(write_file), idx_size);
|
||||
}
|
||||
}
|
||||
self.no_write_or_delete = false;
|
||||
self.save_vif()
|
||||
}
|
||||
@@ -3565,6 +3592,118 @@ mod tests {
|
||||
assert!(matches!(err, VolumeError::Deleted));
|
||||
}
|
||||
|
||||
// Guard the Rust integrity-check tombstone path against the Go regression
|
||||
// where verifyDeletedNeedleIntegrity forwarded TombstoneFileSize into the
|
||||
// needle-size check, mismatched against the on-disk Size=0 header, and
|
||||
// sent every volume with a trailing deletion read-only on load. The Rust
|
||||
// check guards its size comparison with !size.is_deleted(); this test
|
||||
// keeps that guarantee from silently regressing.
|
||||
#[test]
|
||||
fn test_check_volume_data_integrity_with_deletion_tombstone() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
|
||||
{
|
||||
let mut v = make_test_volume(dir);
|
||||
for i in 1..=3 {
|
||||
let data = format!("data {}", i);
|
||||
let mut n = Needle {
|
||||
id: NeedleId(i),
|
||||
cookie: Cookie(i as u32),
|
||||
data: data.as_bytes().to_vec(),
|
||||
data_size: data.len() as u32,
|
||||
..Needle::default()
|
||||
};
|
||||
v.write_needle(&mut n, true).unwrap();
|
||||
}
|
||||
v.delete_needle(&mut Needle {
|
||||
id: NeedleId(2),
|
||||
cookie: Cookie(2),
|
||||
..Needle::default()
|
||||
})
|
||||
.unwrap();
|
||||
v.sync_to_disk().unwrap();
|
||||
}
|
||||
|
||||
let v = Volume::new(
|
||||
dir,
|
||||
dir,
|
||||
"",
|
||||
VolumeId(1),
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
)
|
||||
.unwrap();
|
||||
assert!(
|
||||
!v.is_no_write_or_delete(),
|
||||
"volume should not be read-only after reload with trailing deletion tombstone"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_scrub_empty_volume() {
|
||||
// Mirror of Go's TestScrubVolumeData "zero-size volume without index"
|
||||
// case (weed/storage/volume_checking_test.go): a freshly created /
|
||||
// pre-allocated volume has a superblock-only .dat and a zero-size .idx,
|
||||
// and must scrub clean instead of being flagged as corrupt.
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
let v = make_test_volume(dir);
|
||||
|
||||
// .dat holds only the superblock; .idx is empty.
|
||||
assert_eq!(v.dat_file_size().unwrap(), SUPER_BLOCK_SIZE as u64);
|
||||
|
||||
let (files_checked, broken) = v.scrub().unwrap();
|
||||
assert_eq!(files_checked, 0);
|
||||
assert!(
|
||||
broken.is_empty(),
|
||||
"empty volume should scrub clean, got {:?}",
|
||||
broken
|
||||
);
|
||||
|
||||
// The index-only mode must agree.
|
||||
let (idx_checked, idx_broken) = v.scrub_index().unwrap();
|
||||
assert_eq!(idx_checked, 0);
|
||||
assert!(
|
||||
idx_broken.is_empty(),
|
||||
"empty volume should scrub_index clean, got {:?}",
|
||||
idx_broken
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_scrub_healthy_volume() {
|
||||
// Mirror of Go's TestScrubVolumeData "healthy volume" case: a volume
|
||||
// with live needles scrubs clean and the .dat size accounting matches.
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
let mut v = make_test_volume(dir);
|
||||
|
||||
for i in 1..=5 {
|
||||
let data = format!("needle data {}", i);
|
||||
let mut n = Needle {
|
||||
id: NeedleId(i),
|
||||
cookie: Cookie(i as u32),
|
||||
data: data.as_bytes().to_vec(),
|
||||
data_size: data.len() as u32,
|
||||
..Needle::default()
|
||||
};
|
||||
v.write_needle(&mut n, true).unwrap();
|
||||
}
|
||||
v.sync_to_disk().unwrap();
|
||||
|
||||
let (files_checked, broken) = v.scrub().unwrap();
|
||||
assert_eq!(files_checked, 5);
|
||||
assert!(
|
||||
broken.is_empty(),
|
||||
"healthy volume should scrub clean, got {:?}",
|
||||
broken
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_volume_multiple_needles() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
@@ -4191,6 +4330,91 @@ mod tests {
|
||||
assert!(v.no_write_can_delete);
|
||||
}
|
||||
|
||||
// A volume booted with .vif ReadOnly=true used to come back stuck —
|
||||
// load_index_inmemory built the CompactNeedleMap without an .idx writer
|
||||
// attached, and set_writable only flipped the flag and rewrote .vif.
|
||||
// The next put silently skipped the .idx append, so the write landed in
|
||||
// memory only and was lost on the next restart.
|
||||
#[test]
|
||||
fn test_set_writable_reattaches_idx_writer_after_persisted_readonly() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
|
||||
{
|
||||
let mut v = make_test_volume(dir);
|
||||
let mut n = Needle {
|
||||
id: NeedleId(1),
|
||||
cookie: Cookie(1),
|
||||
data: b"initial".to_vec(),
|
||||
data_size: 7,
|
||||
..Needle::default()
|
||||
};
|
||||
v.write_needle(&mut n, true).unwrap();
|
||||
v.set_read_only_persist(true).unwrap();
|
||||
v.sync_to_disk().unwrap();
|
||||
}
|
||||
|
||||
let mut v = Volume::new(
|
||||
dir,
|
||||
dir,
|
||||
"",
|
||||
VolumeId(1),
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
)
|
||||
.unwrap();
|
||||
assert!(
|
||||
v.no_write_or_delete,
|
||||
"reloaded volume should be read-only from .vif"
|
||||
);
|
||||
assert!(
|
||||
!v.nm.as_ref().unwrap().has_idx_writer(),
|
||||
"read-only load should not attach an .idx writer"
|
||||
);
|
||||
|
||||
v.set_writable().unwrap();
|
||||
assert!(!v.is_read_only());
|
||||
assert!(
|
||||
v.nm.as_ref().unwrap().has_idx_writer(),
|
||||
"set_writable must reattach the .idx writer or post-restart writes vanish"
|
||||
);
|
||||
|
||||
let mut n = Needle {
|
||||
id: NeedleId(2),
|
||||
cookie: Cookie(2),
|
||||
data: b"after-mark-writable".to_vec(),
|
||||
data_size: 19,
|
||||
..Needle::default()
|
||||
};
|
||||
v.write_needle(&mut n, true).unwrap();
|
||||
v.sync_to_disk().unwrap();
|
||||
|
||||
// Reload one more time — the .idx must contain the post-mark-writable
|
||||
// entry, not just have it in memory.
|
||||
drop(v);
|
||||
let v = Volume::new(
|
||||
dir,
|
||||
dir,
|
||||
"",
|
||||
VolumeId(1),
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
)
|
||||
.unwrap();
|
||||
let mut probe = Needle {
|
||||
id: NeedleId(2),
|
||||
..Needle::default()
|
||||
};
|
||||
v.read_needle(&mut probe).unwrap();
|
||||
assert_eq!(std::str::from_utf8(&probe.data).unwrap(), "after-mark-writable");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_load_vif_defaults_local_version_and_bytes_offset() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
|
||||
@@ -0,0 +1,265 @@
|
||||
package erasure_coding
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"path/filepath"
|
||||
"regexp"
|
||||
"strconv"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/shell"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/needle"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
"google.golang.org/grpc"
|
||||
)
|
||||
|
||||
// TestMultiDiskECBalanceNoShardLoss is the end-to-end regression for issue 9593.
|
||||
// It runs a real cluster of multi-disk volume servers (3 servers x 4 disks),
|
||||
// EC-encodes a volume, then runs ec.balance, asserting hard invariants the older
|
||||
// integration tests only logged:
|
||||
//
|
||||
// - after encode the full set of 14 EC shards exists,
|
||||
// - ec.balance never loses a shard (still 14 distinct shards afterwards),
|
||||
// - shards end up spread across more than one disk per node, and
|
||||
// - cluster.status counts physical disks (not one per node) and matches the
|
||||
// real on-disk distribution.
|
||||
func TestMultiDiskECBalanceNoShardLoss(t *testing.T) {
|
||||
if testing.Short() {
|
||||
t.Skip("Skipping multi-disk EC integration test in short mode")
|
||||
}
|
||||
|
||||
testDir := t.TempDir()
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 240*time.Second)
|
||||
defer cancel()
|
||||
|
||||
cluster, err := startMultiDiskCluster(ctx, testDir)
|
||||
require.NoError(t, err)
|
||||
defer cluster.Stop()
|
||||
|
||||
require.NoError(t, waitForServer("127.0.0.1:9334", 30*time.Second))
|
||||
for i := 0; i < 3; i++ {
|
||||
require.NoError(t, waitForServer(fmt.Sprintf("127.0.0.1:809%d", i), 30*time.Second))
|
||||
}
|
||||
t.Log("waiting for multi-disk volume servers to register...")
|
||||
time.Sleep(10 * time.Second)
|
||||
|
||||
commandEnv := shell.NewCommandEnv(&shell.ShellOptions{
|
||||
Masters: stringPtr("127.0.0.1:9334"),
|
||||
GrpcDialOption: grpc.WithInsecure(),
|
||||
FilerGroup: stringPtr("default"),
|
||||
})
|
||||
connectToMasterAndSync(ctx, t, commandEnv)
|
||||
|
||||
// Upload enough small files that the volume holds real data to encode.
|
||||
var volumeId needle.VolumeId
|
||||
for retry := 0; retry < 5; retry++ {
|
||||
volumeId, err = uploadTestDataToMaster([]byte(strings.Repeat("multidisk-ec-9593 ", 64)), "127.0.0.1:9334")
|
||||
if err == nil {
|
||||
break
|
||||
}
|
||||
time.Sleep(3 * time.Second)
|
||||
}
|
||||
require.NoError(t, err, "failed to upload test data")
|
||||
for i := 0; i < 40; i++ {
|
||||
if _, e := uploadTestDataToMaster([]byte(strings.Repeat("filler ", 128)), "127.0.0.1:9334"); e != nil {
|
||||
break
|
||||
}
|
||||
}
|
||||
t.Logf("using volume %d", volumeId)
|
||||
time.Sleep(3 * time.Second)
|
||||
|
||||
// Populate every server's disks with volumes so the encode can see and target
|
||||
// each physical disk. The master only enumerates disks that already hold a
|
||||
// volume or EC shard — an empty disk leaves no trace in the topology (heartbeats
|
||||
// aggregate capacity per disk type, not per physical disk). ec.encode therefore
|
||||
// spreads a volume's shards only across the disks the master already knows hold
|
||||
// data on each node; if a node's data sits on a single disk, all its shards land
|
||||
// there and ec.balance cannot redistribute them (it has no within-node
|
||||
// cross-disk move). So spreading must be set up before encoding.
|
||||
//
|
||||
// volume.grow only tops up toward a writable target and stops on the first
|
||||
// allocation error, so a single -count grow can create far fewer volumes than
|
||||
// asked and leave a node on one disk. Grow repeatedly on the nodes that have not
|
||||
// spread yet (the volume server places each new volume on its least-loaded disk)
|
||||
// until the master's topology shows every node holding volumes on at least two
|
||||
// physical disks. This makes the multi-disk layout — and thus the post-encode
|
||||
// disk spread — deterministic instead of racing volume-growth and heartbeat.
|
||||
require.Eventually(t, func() bool {
|
||||
spread := nodeVolumeDiskCounts(t, commandEnv)
|
||||
if len(spread) == 3 && allAtLeast(spread, 2) {
|
||||
return true
|
||||
}
|
||||
for i := 0; i < 3; i++ {
|
||||
server := fmt.Sprintf("127.0.0.1:809%d", i)
|
||||
if spread[server] < 2 {
|
||||
captureCommandOutput(t, shell.Commands[findCommandIndex("volume.grow")],
|
||||
[]string{"-collection", "test", "-dataNode", server, "-count", "4"}, commandEnv)
|
||||
}
|
||||
}
|
||||
return false
|
||||
}, 60*time.Second, 2*time.Second,
|
||||
"volumes never spread across >=2 disks on all 3 nodes")
|
||||
|
||||
locked, unlock := tryLockWithTimeout(t, commandEnv, 15*time.Second)
|
||||
require.True(t, locked, "could not acquire shell lock")
|
||||
defer unlock()
|
||||
|
||||
// EC-encode the volume.
|
||||
out, err := captureCommandOutput(t, shell.Commands[findCommandIndex("ec.encode")],
|
||||
[]string{"-volumeId", fmt.Sprintf("%d", volumeId), "-collection", "test", "-force"}, commandEnv)
|
||||
t.Logf("ec.encode output:\n%s", out)
|
||||
require.NoError(t, err, "ec.encode failed")
|
||||
|
||||
// All 14 shards must exist after encoding.
|
||||
require.Eventually(t, func() bool {
|
||||
return len(collectDistinctShardIDs(testDir, uint32(volumeId))) == erasureShardCount
|
||||
}, 30*time.Second, time.Second, "expected all %d EC shards after encode, got %v",
|
||||
erasureShardCount, collectDistinctShardIDs(testDir, uint32(volumeId)))
|
||||
|
||||
beforeBalance := collectDistinctShardIDs(testDir, uint32(volumeId))
|
||||
t.Logf("after encode: %d distinct shards on %d disks", len(beforeBalance), disksWithShards(testDir, uint32(volumeId)))
|
||||
|
||||
// Run ec.balance.
|
||||
out, err = captureCommandOutput(t, shell.Commands[findCommandIndex("ec.balance")],
|
||||
[]string{"-collection", "test", "-force"}, commandEnv)
|
||||
t.Logf("ec.balance output:\n%s", out)
|
||||
require.NoError(t, err, "ec.balance failed")
|
||||
time.Sleep(3 * time.Second)
|
||||
|
||||
// The core regression: ec.balance must not lose any shard.
|
||||
afterBalance := collectDistinctShardIDs(testDir, uint32(volumeId))
|
||||
require.Equal(t, erasureShardCount, len(afterBalance),
|
||||
"ec.balance lost shards on multi-disk nodes: had %v, now %v", sortedKeysOf(beforeBalance), sortedKeysOf(afterBalance))
|
||||
|
||||
// Shards must be spread across more than one physical disk per node overall.
|
||||
usedDisks := disksWithShards(testDir, uint32(volumeId))
|
||||
assert.Greater(t, usedDisks, 3, "EC shards should span more than one disk per node (got %d disks across 3 nodes)", usedDisks)
|
||||
|
||||
// cluster.status must count physical disks, not collapse to one per node: it
|
||||
// must report at least the disks actually holding this volume's shards (which
|
||||
// is already >3 across the 3 nodes). Before the fix it reported 3 (node count).
|
||||
require.Eventually(t, func() bool {
|
||||
n, ok := clusterStatusDiskCount(t, commandEnv)
|
||||
return ok && n >= usedDisks
|
||||
}, 30*time.Second, 2*time.Second, "cluster.status never reported the >=%d physical disks holding shards (multi-disk count)", usedDisks)
|
||||
|
||||
n, _ := clusterStatusDiskCount(t, commandEnv)
|
||||
t.Logf("cluster.status reports %d physical disks (>= %d holding this volume's shards)", n, usedDisks)
|
||||
}
|
||||
|
||||
const erasureShardCount = 14 // 10 data + 4 parity
|
||||
|
||||
// collectDistinctShardIDs returns the set of EC shard ids present for a volume
|
||||
// across every disk of every server in the multi-disk test layout.
|
||||
func collectDistinctShardIDs(testDir string, volumeId uint32) map[int]bool {
|
||||
ids := map[int]bool{}
|
||||
for server := 0; server < 3; server++ {
|
||||
for disk := 0; disk < 4; disk++ {
|
||||
diskDir := filepath.Join(testDir, fmt.Sprintf("server%d_disk%d", server, disk))
|
||||
files, err := listECShardFiles(diskDir, volumeId)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
for _, f := range files {
|
||||
i := strings.LastIndex(f, ".ec")
|
||||
if i < 0 {
|
||||
continue
|
||||
}
|
||||
if n, err := strconv.Atoi(f[i+3:]); err == nil && n >= 0 && n < erasureShardCount {
|
||||
ids[n] = true
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return ids
|
||||
}
|
||||
|
||||
// disksWithShards counts how many physical disks hold at least one shard.
|
||||
func disksWithShards(testDir string, volumeId uint32) int {
|
||||
n := 0
|
||||
for _, disks := range countShardsPerDisk(testDir, volumeId) {
|
||||
for _, c := range disks {
|
||||
if c > 0 {
|
||||
n++
|
||||
}
|
||||
}
|
||||
}
|
||||
return n
|
||||
}
|
||||
|
||||
// nodeVolumeDiskCounts returns, per volume server id, how many distinct physical
|
||||
// disks hold at least one volume according to the master's topology. The master
|
||||
// only enumerates disks that already hold a volume or EC shard (heartbeats
|
||||
// aggregate capacity per disk type, not per physical disk), so this reports the
|
||||
// disks ec.encode can actually spread a volume's shards across on each node.
|
||||
func nodeVolumeDiskCounts(t *testing.T, commandEnv *shell.CommandEnv) map[string]int {
|
||||
t.Helper()
|
||||
var resp *master_pb.VolumeListResponse
|
||||
err := commandEnv.MasterClient.WithClient(false, func(client master_pb.SeaweedClient) error {
|
||||
var e error
|
||||
resp, e = client.VolumeList(context.Background(), &master_pb.VolumeListRequest{})
|
||||
return e
|
||||
})
|
||||
counts := map[string]int{}
|
||||
if err != nil || resp.GetTopologyInfo() == nil {
|
||||
return counts
|
||||
}
|
||||
for _, dc := range resp.GetTopologyInfo().GetDataCenterInfos() {
|
||||
for _, r := range dc.GetRackInfos() {
|
||||
for _, dn := range r.GetDataNodeInfos() {
|
||||
disks := map[uint32]bool{}
|
||||
for _, di := range dn.GetDiskInfos() {
|
||||
for _, vi := range di.GetVolumeInfos() {
|
||||
disks[vi.GetDiskId()] = true
|
||||
}
|
||||
}
|
||||
counts[dn.Id] = len(disks)
|
||||
}
|
||||
}
|
||||
}
|
||||
return counts
|
||||
}
|
||||
|
||||
func allAtLeast(counts map[string]int, min int) bool {
|
||||
for _, c := range counts {
|
||||
if c < min {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
var diskCountRe = regexp.MustCompile(`(\d+)\s+disks?`)
|
||||
|
||||
// clusterStatusDiskCount runs cluster.status and parses the reported disk count.
|
||||
func clusterStatusDiskCount(t *testing.T, commandEnv *shell.CommandEnv) (int, bool) {
|
||||
t.Helper()
|
||||
out, err := captureCommandOutput(t, shell.Commands[findCommandIndex("cluster.status")], []string{}, commandEnv)
|
||||
if err != nil {
|
||||
return 0, false
|
||||
}
|
||||
m := diskCountRe.FindStringSubmatch(out)
|
||||
if m == nil {
|
||||
return 0, false
|
||||
}
|
||||
n, err := strconv.Atoi(m[1])
|
||||
return n, err == nil
|
||||
}
|
||||
|
||||
func sortedKeysOf(m map[int]bool) []int {
|
||||
out := make([]int, 0, len(m))
|
||||
for k := range m {
|
||||
out = append(out, k)
|
||||
}
|
||||
for i := 1; i < len(out); i++ {
|
||||
for j := i; j > 0 && out[j-1] > out[j]; j-- {
|
||||
out[j-1], out[j] = out[j], out[j-1]
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
@@ -0,0 +1,232 @@
|
||||
package fuse_dlm
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"runtime"
|
||||
"syscall"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/cluster/lock_manager"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// posixLockKey returns the routed-lock key a mount uses for a file at the mount
|
||||
// root. Mounts run with -filer.path=/, so the file's filer path is "/"+name; the
|
||||
// prefix must match mount.posixLockKeyForInode ("s3.fuse.lock:").
|
||||
func posixLockKey(name string) string { return "s3.fuse.lock:/" + name }
|
||||
|
||||
// These tests exercise cross-mount POSIX advisory locks (flock), which the
|
||||
// mounts route to the inode's owner filer because they run with -dlm
|
||||
// (crossMountLocks() == lockClient != nil). They reuse the dlmTestCluster:
|
||||
// master + volume + 2 filers (forming the lock ring) + 2 mounts on filer0.
|
||||
//
|
||||
// All flock opens are O_RDONLY on purpose: a write open (O_RDWR/O_WRONLY) would
|
||||
// also take the -dlm whole-file write lock (held until close), which is a
|
||||
// different mechanism — opening read-only isolates the POSIX advisory flock.
|
||||
|
||||
// requireForwardedLocks skips on platforms where the kernel does not forward
|
||||
// advisory locks to the FUSE server. Only Linux forwards flock/fcntl (SETLK) to
|
||||
// the filesystem; macFUSE handles flock in-kernel per mount, so cross-mount
|
||||
// coordination can't be observed there even though the routed path is correct.
|
||||
func requireForwardedLocks(t *testing.T) {
|
||||
if runtime.GOOS != "linux" {
|
||||
t.Skipf("advisory locks are only forwarded to the FUSE server on Linux (GOOS=%s)", runtime.GOOS)
|
||||
}
|
||||
}
|
||||
|
||||
// openFlock opens path read-only and takes a flock of type how (e.g. LOCK_EX).
|
||||
// The file must already exist. The returned file holds the lock until closed.
|
||||
func openFlock(path string, how int) (*os.File, error) {
|
||||
f, err := os.OpenFile(path, os.O_RDONLY, 0)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if err := syscall.Flock(int(f.Fd()), how); err != nil {
|
||||
f.Close()
|
||||
return nil, err
|
||||
}
|
||||
return f, nil
|
||||
}
|
||||
|
||||
// createFile creates an empty file; the brief -dlm write lock it takes is
|
||||
// released on close, before any flock test runs.
|
||||
func createFile(t *testing.T, path string) {
|
||||
t.Helper()
|
||||
f, err := os.OpenFile(path, os.O_RDWR|os.O_CREATE|os.O_TRUNC, 0644)
|
||||
require.NoError(t, err, "create %s", path)
|
||||
require.NoError(t, f.Close())
|
||||
}
|
||||
|
||||
// waitVisible waits for path to appear (cross-mount metadata propagation).
|
||||
func waitVisible(t *testing.T, path string) {
|
||||
t.Helper()
|
||||
require.Eventually(t, func() bool {
|
||||
_, err := os.Stat(path)
|
||||
return err == nil
|
||||
}, 15*time.Second, 200*time.Millisecond, "%s never became visible", path)
|
||||
}
|
||||
|
||||
// tryExclusiveFlock attempts a non-blocking exclusive flock on path and
|
||||
// classifies the outcome: acquired (and released again), blocked by another
|
||||
// owner (EWOULDBLOCK/EAGAIN), or an unexpected error. It distinguishes a real
|
||||
// "held by another" from incidental errors so the latter can't masquerade as a
|
||||
// satisfied "must be blocked" assertion.
|
||||
func tryExclusiveFlock(path string) (acquired, blocked bool, err error) {
|
||||
f, err := os.OpenFile(path, os.O_RDONLY, 0)
|
||||
if err != nil {
|
||||
return false, false, err
|
||||
}
|
||||
defer f.Close()
|
||||
if err := syscall.Flock(int(f.Fd()), syscall.LOCK_EX|syscall.LOCK_NB); err != nil {
|
||||
if errors.Is(err, syscall.EWOULDBLOCK) || errors.Is(err, syscall.EAGAIN) {
|
||||
return false, true, nil
|
||||
}
|
||||
return false, false, err
|
||||
}
|
||||
syscall.Flock(int(f.Fd()), syscall.LOCK_UN)
|
||||
return true, false, nil
|
||||
}
|
||||
|
||||
// requireEventuallyBlocked asserts path becomes held by another owner. It polls
|
||||
// because a routed lock RPC can hit a transient EIO (a cold or forwarded gRPC
|
||||
// call under load) even while the lock is genuinely held; it still fails if the
|
||||
// lock is never blocked — whether it can be acquired (a double-grant) or errors
|
||||
// persistently — so an incidental error can't masquerade as "held".
|
||||
func requireEventuallyBlocked(t *testing.T, path string, timeout time.Duration, msg string) {
|
||||
t.Helper()
|
||||
require.Eventuallyf(t, func() bool { return isBlocked(path) },
|
||||
timeout, 500*time.Millisecond, "%s", msg)
|
||||
}
|
||||
|
||||
// isBlocked / isAcquirable are lenient predicates for polling, where transient
|
||||
// errors during a migration are expected and simply mean "not yet".
|
||||
func isBlocked(path string) bool { _, b, _ := tryExclusiveFlock(path); return b }
|
||||
func isAcquirable(path string) bool {
|
||||
a, _, _ := tryExclusiveFlock(path)
|
||||
return a
|
||||
}
|
||||
|
||||
// stopFiler stops filer idx and clears its command so cluster teardown does not
|
||||
// try to stop it again.
|
||||
func (c *dlmTestCluster) stopFiler(idx int) {
|
||||
stopCmd(c.filerCmds[idx])
|
||||
c.filerCmds[idx] = nil
|
||||
}
|
||||
|
||||
// TestPosixLockCrossMount verifies a flock taken on one mount is seen by the
|
||||
// other mount of the same cluster (the routed-to-owner-filer path end to end).
|
||||
func TestPosixLockCrossMount(t *testing.T) {
|
||||
requireForwardedLocks(t)
|
||||
c := startDLMTestCluster(t)
|
||||
|
||||
name := "posix-xmount.lock"
|
||||
path0 := filepath.Join(c.mountPoints[0], name)
|
||||
path1 := filepath.Join(c.mountPoints[1], name)
|
||||
|
||||
createFile(t, path0)
|
||||
waitVisible(t, path1)
|
||||
|
||||
held, err := openFlock(path0, syscall.LOCK_EX)
|
||||
require.NoError(t, err, "mount0 should acquire the exclusive flock")
|
||||
defer held.Close()
|
||||
|
||||
requireEventuallyBlocked(t, path1, 15*time.Second, "mount1 must be blocked while mount0 holds the flock")
|
||||
|
||||
require.NoError(t, syscall.Flock(int(held.Fd()), syscall.LOCK_UN))
|
||||
require.Eventually(t, func() bool { return isAcquirable(path1) },
|
||||
15*time.Second, 500*time.Millisecond,
|
||||
"mount1 should acquire the flock after mount0 releases it")
|
||||
}
|
||||
|
||||
// TestPosixLockSurvivesFilerLoss verifies advisory locks held across mounts
|
||||
// survive a filer leaving the ring: ownership of the affected keys migrates to
|
||||
// the surviving filer and the holding mount re-asserts them there, so the locks
|
||||
// stay honored. It locks many files so that — with 2 filers — several are owned
|
||||
// by filer1 and thus actually migrate when filer1 stops.
|
||||
func TestPosixLockSurvivesFilerLoss(t *testing.T) {
|
||||
requireForwardedLocks(t)
|
||||
c := startDLMTestCluster(t)
|
||||
|
||||
const n = 12
|
||||
|
||||
// Select files via the same ring the filers run so the locked set provably
|
||||
// spans both filers: filer1-owned keys must migrate when filer1 stops, while
|
||||
// filer0-owned keys must keep working. Choosing by ownership (instead of
|
||||
// hoping a sequential set happens to spread) keeps the migration path
|
||||
// exercised on every run, independent of the cluster's dynamic ports.
|
||||
ring := lock_manager.NewHashRing(lock_manager.DefaultVnodeCount)
|
||||
ring.SetServers([]pb.ServerAddress{
|
||||
pb.ServerAddress(c.filerAddress(0)),
|
||||
pb.ServerAddress(c.filerAddress(1)),
|
||||
})
|
||||
filer1 := pb.ServerAddress(c.filerAddress(1))
|
||||
var onFiler1, onFiler0 []string
|
||||
for i := 0; (len(onFiler1) < n/2 || len(onFiler0) < n/2) && i < 1000; i++ {
|
||||
name := fmt.Sprintf("posix-migrate-%d.lock", i)
|
||||
if ring.GetPrimary(posixLockKey(name)) == filer1 {
|
||||
onFiler1 = append(onFiler1, name)
|
||||
} else {
|
||||
onFiler0 = append(onFiler0, name)
|
||||
}
|
||||
}
|
||||
require.GreaterOrEqualf(t, len(onFiler1), n/2, "need %d filer1-owned keys to exercise migration", n/2)
|
||||
require.GreaterOrEqualf(t, len(onFiler0), n/2, "need %d filer0-owned keys", n/2)
|
||||
names := append(onFiler1[:n/2:n/2], onFiler0[:n/2]...)
|
||||
|
||||
held := make([]*os.File, n)
|
||||
for i, name := range names {
|
||||
createFile(t, filepath.Join(c.mountPoints[0], name))
|
||||
waitVisible(t, filepath.Join(c.mountPoints[1], name))
|
||||
f, err := openFlock(filepath.Join(c.mountPoints[0], name), syscall.LOCK_EX)
|
||||
require.NoError(t, err, "mount0 should acquire flock %s", name)
|
||||
held[i] = f
|
||||
defer held[i].Close()
|
||||
}
|
||||
|
||||
require.Eventually(t, func() bool {
|
||||
for _, name := range names {
|
||||
if !isBlocked(filepath.Join(c.mountPoints[1], name)) {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}, 20*time.Second, time.Second,
|
||||
"every lock must be held on mount1 before the ring change")
|
||||
|
||||
// Drop filer1 from the ring; keys it owned migrate to filer0.
|
||||
c.stopFiler(1)
|
||||
require.NoError(t, c.waitForFilerCount(1, 30*time.Second), "ring should drop to one filer")
|
||||
|
||||
// Poll until every lock is honored again on the surviving filer: the ring
|
||||
// must propagate and the holding mount must re-assert its locks (keepalive is
|
||||
// 5s). We assert only the settled state — the transient migration window is
|
||||
// covered by unit tests.
|
||||
require.Eventually(t, func() bool {
|
||||
for _, name := range names {
|
||||
if !isBlocked(filepath.Join(c.mountPoints[1], name)) {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}, 45*time.Second, time.Second,
|
||||
"all locks must survive filer1 leaving the ring (migrate to filer0)")
|
||||
|
||||
// Releasing on mount0 frees them all: mount1 can then acquire each.
|
||||
for _, f := range held {
|
||||
require.NoError(t, syscall.Flock(int(f.Fd()), syscall.LOCK_UN))
|
||||
}
|
||||
require.Eventually(t, func() bool {
|
||||
for _, name := range names {
|
||||
if !isAcquirable(filepath.Join(c.mountPoints[1], name)) {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}, 20*time.Second, 500*time.Millisecond,
|
||||
"mount1 should acquire every lock after mount0 releases them post-migration")
|
||||
}
|
||||
@@ -177,8 +177,10 @@ func testConcurrentReadWrite(t *testing.T, framework *FuseTestFramework) {
|
||||
defer wg.Done()
|
||||
|
||||
for j := 0; j < 10; j++ {
|
||||
_, err := os.ReadFile(mountPath)
|
||||
if err != nil {
|
||||
if err := retryTransientFUSE(func() error {
|
||||
_, e := os.ReadFile(mountPath)
|
||||
return e
|
||||
}); err != nil {
|
||||
addError(fmt.Errorf("reader %d: %v", readerID, err))
|
||||
return
|
||||
}
|
||||
@@ -196,8 +198,9 @@ func testConcurrentReadWrite(t *testing.T, framework *FuseTestFramework) {
|
||||
|
||||
for j := 0; j < 5; j++ {
|
||||
newData := bytes.Repeat([]byte(fmt.Sprintf("WRITER%d", writerID)), 1000)
|
||||
err := os.WriteFile(mountPath, newData, 0644)
|
||||
if err != nil {
|
||||
if err := retryTransientFUSE(func() error {
|
||||
return os.WriteFile(mountPath, newData, 0644)
|
||||
}); err != nil {
|
||||
addError(fmt.Errorf("writer %d: %v", writerID, err))
|
||||
return
|
||||
}
|
||||
@@ -213,6 +216,21 @@ func testConcurrentReadWrite(t *testing.T, framework *FuseTestFramework) {
|
||||
framework.AssertFileExists(filename)
|
||||
}
|
||||
|
||||
// retryTransientFUSE retries op a few times before giving up. A concurrent
|
||||
// truncating overwrite can leave a short-lived dentry/cache window where the
|
||||
// entry is momentarily invisible (ENOENT) to another opener; the last error is
|
||||
// returned so a genuine, persistent failure still surfaces.
|
||||
func retryTransientFUSE(op func() error) error {
|
||||
var err error
|
||||
for attempt := 0; attempt < 5; attempt++ {
|
||||
if err = op(); err == nil {
|
||||
return nil
|
||||
}
|
||||
time.Sleep(100 * time.Millisecond)
|
||||
}
|
||||
return err
|
||||
}
|
||||
|
||||
// testConcurrentDirectoryOperations tests concurrent directory operations
|
||||
func testConcurrentDirectoryOperations(t *testing.T, framework *FuseTestFramework) {
|
||||
numWorkers := 8
|
||||
|
||||
@@ -207,8 +207,10 @@ func buildVolumeListResponse(t *testing.T, spec topologySpec, volumeID uint32) *
|
||||
t.Helper()
|
||||
|
||||
volumeSizeLimitMB := uint64(100)
|
||||
volumeSize := uint64(90) * 1024 * 1024
|
||||
volumeModifiedAt := time.Now().Add(-10 * time.Minute).Unix()
|
||||
// Exceed the default fullness (0.95) and quiet (1h) thresholds so volumes are
|
||||
// EC-eligible.
|
||||
volumeSize := uint64(96) * 1024 * 1024
|
||||
volumeModifiedAt := time.Now().Add(-2 * time.Hour).Unix()
|
||||
|
||||
diskTypes := spec.diskTypes
|
||||
if len(diskTypes) == 0 {
|
||||
|
||||
@@ -29,9 +29,11 @@ func TestErasureCodingDetectionLargeTopology(t *testing.T) {
|
||||
}
|
||||
|
||||
nodesPerRack := serverCount / rackCount
|
||||
eligibleSize := uint64(90) * 1024 * 1024
|
||||
// Eligible volumes must exceed the default fullness (0.95) and quiet (1h)
|
||||
// thresholds; ineligible ones fall below the fullness threshold.
|
||||
eligibleSize := uint64(96) * 1024 * 1024
|
||||
ineligibleSize := uint64(10) * 1024 * 1024
|
||||
modifiedAt := time.Now().Add(-10 * time.Minute).Unix()
|
||||
modifiedAt := time.Now().Add(-2 * time.Hour).Unix()
|
||||
|
||||
volumeID := uint32(1)
|
||||
dataCenters := make([]*master_pb.DataCenterInfo, 0, 1)
|
||||
|
||||
@@ -298,6 +298,54 @@ User Request → Load Balancer → Any S3 Gateway Instance
|
||||
Allow/Deny Request
|
||||
```
|
||||
|
||||
## Trust Policy Conditions
|
||||
|
||||
Step 5 above evaluates the role's trust policy against context keys derived from the
|
||||
OIDC token's claims. The available keys are:
|
||||
|
||||
| Condition key | Source |
|
||||
|---------------|--------|
|
||||
| `oidc:iss` | `iss` claim (issuer URL) |
|
||||
| `oidc:sub` | `sub` claim |
|
||||
| `oidc:aud` | `aud` claim |
|
||||
| `oidc:<claim>` | any other token claim, e.g. `oidc:roles`, `oidc:groups`, `oidc:email` |
|
||||
| `aws:FederatedProvider` | the provider `name` (e.g. `keycloak-oidc`) when its configured issuer matches the token, otherwise the raw issuer URL |
|
||||
| `aws:userid` | `sub` claim (same value as `oidc:sub` during trust-policy evaluation) |
|
||||
| `sts:DurationSeconds` | requested session duration, when supplied |
|
||||
|
||||
During trust-policy evaluation `aws:userid` is the raw `sub` claim. Once the
|
||||
role has been assumed, the keys seen by request authorization differ: there
|
||||
`aws:userid` is a stable per-identity hash of `sub` and `iss` (see
|
||||
`ComputeParentUser`), so do not assume the two contexts carry the same value.
|
||||
|
||||
Custom claims are always exposed under the `oidc:` prefix, so a trust policy must use
|
||||
`oidc:roles` (not a bare `roles`) to match a `roles` claim:
|
||||
|
||||
```json
|
||||
"Condition": {
|
||||
"StringEquals": {
|
||||
"oidc:roles": "s3-admin"
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
A multi-valued claim (such as a `roles` array) matches when any of its values equals
|
||||
the condition value. The same `oidc:` keys can be interpolated into policy resources,
|
||||
e.g. `arn:aws:s3:::bucket/${oidc:sub}/*`.
|
||||
|
||||
### roleMapping vs. trust policy
|
||||
|
||||
A provider's `roleMapping` and a role's trust policy apply to two different entry
|
||||
points and are not interchangeable:
|
||||
|
||||
- **Direct OIDC** — an S3 request carrying `Authorization: Bearer <OIDC-JWT>`. The
|
||||
gateway applies `roleMapping` to choose the caller's role from the token claims; the
|
||||
first matching rule (or `defaultRole`) wins.
|
||||
- **STS `AssumeRoleWithWebIdentity`** — the caller names the role explicitly via
|
||||
`RoleArn`, and that role's trust policy decides whether the assumption is allowed.
|
||||
`roleMapping` does not select the role on this path; instead the token claims are
|
||||
surfaced as the `oidc:` condition keys above for the trust policy to evaluate.
|
||||
|
||||
## Configuration Management
|
||||
|
||||
### Development Environment
|
||||
|
||||
@@ -37,7 +37,7 @@
|
||||
"Action": ["sts:AssumeRoleWithWebIdentity"],
|
||||
"Condition": {
|
||||
"StringEquals": {
|
||||
"roles": "s3-admin"
|
||||
"oidc:roles": "s3-admin"
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -60,7 +60,7 @@
|
||||
"Action": ["sts:AssumeRoleWithWebIdentity"],
|
||||
"Condition": {
|
||||
"StringEquals": {
|
||||
"roles": "s3-read-only"
|
||||
"oidc:roles": "s3-read-only"
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -83,7 +83,7 @@
|
||||
"Action": ["sts:AssumeRoleWithWebIdentity"],
|
||||
"Condition": {
|
||||
"StringEquals": {
|
||||
"roles": "s3-read-write"
|
||||
"oidc:roles": "s3-read-write"
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,309 @@
|
||||
package iam
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
mathrand "math/rand"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/aws/aws-sdk-go/aws"
|
||||
"github.com/aws/aws-sdk-go/aws/awserr"
|
||||
"github.com/aws/aws-sdk-go/service/iam"
|
||||
"github.com/aws/aws-sdk-go/service/s3"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// uniqueResourceSuffix returns a lowercased per-test, per-invocation suffix
|
||||
// safe for use in IAM resource and S3 bucket names. Avoids EntityAlreadyExists
|
||||
// / BucketAlreadyExists collisions when integration jobs retry or run in
|
||||
// parallel against a shared stack.
|
||||
func uniqueResourceSuffix(t *testing.T) string {
|
||||
name := strings.ToLower(t.Name())
|
||||
name = strings.ReplaceAll(name, "/", "-")
|
||||
name = strings.ReplaceAll(name, "_", "-")
|
||||
return fmt.Sprintf("%s-%d", name, mathrand.Intn(10000))
|
||||
}
|
||||
|
||||
// isAccessDenied returns true when err is an AWS error with code "AccessDenied".
|
||||
// Used to gate deny-path polling so transient setup errors don't end the
|
||||
// Eventually loop prematurely.
|
||||
func isAccessDenied(err error) bool {
|
||||
if err == nil {
|
||||
return false
|
||||
}
|
||||
awsErr, ok := err.(awserr.Error)
|
||||
return ok && awsErr.Code() == "AccessDenied"
|
||||
}
|
||||
|
||||
// TestIAMUserInlinePolicySourceIpCondition verifies that an aws:SourceIp condition
|
||||
// on a user inline policy is honored. Tests run from localhost (127.0.0.1), so a
|
||||
// policy that only allows access from a non-loopback CIDR must deny the request,
|
||||
// and a policy that allows access from 127.0.0.0/8 must allow it.
|
||||
func TestIAMUserInlinePolicySourceIpCondition(t *testing.T) {
|
||||
framework := NewS3IAMTestFramework(t)
|
||||
defer framework.Cleanup()
|
||||
|
||||
iamClient, err := framework.CreateIAMClientWithJWT("admin-user", "TestAdminRole")
|
||||
require.NoError(t, err)
|
||||
|
||||
suffix := uniqueResourceSuffix(t)
|
||||
userName := "user-" + suffix
|
||||
policyName := "policy-" + suffix
|
||||
bucketName := "bucket-" + suffix
|
||||
|
||||
_, err = iamClient.CreateUser(&iam.CreateUserInput{UserName: aws.String(userName)})
|
||||
require.NoError(t, err)
|
||||
|
||||
keyResp, err := iamClient.CreateAccessKey(&iam.CreateAccessKeyInput{
|
||||
UserName: aws.String(userName),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
accessKeyId := *keyResp.AccessKey.AccessKeyId
|
||||
secretKey := *keyResp.AccessKey.SecretAccessKey
|
||||
|
||||
userS3 := createS3Client(t, accessKeyId, secretKey)
|
||||
|
||||
adminS3, err := framework.CreateS3ClientWithJWT("admin-user", "TestAdminRole")
|
||||
require.NoError(t, err)
|
||||
require.NoError(t, framework.CreateBucketWithCleanup(adminS3, bucketName))
|
||||
|
||||
t.Cleanup(func() {
|
||||
if _, err := iamClient.DeleteUserPolicy(&iam.DeleteUserPolicyInput{
|
||||
UserName: aws.String(userName),
|
||||
PolicyName: aws.String(policyName),
|
||||
}); err != nil {
|
||||
t.Logf("cleanup: failed to delete user policy: %v", err)
|
||||
}
|
||||
if _, err := iamClient.DeleteAccessKey(&iam.DeleteAccessKeyInput{
|
||||
UserName: aws.String(userName),
|
||||
AccessKeyId: keyResp.AccessKey.AccessKeyId,
|
||||
}); err != nil {
|
||||
t.Logf("cleanup: failed to delete access key: %v", err)
|
||||
}
|
||||
if _, err := iamClient.DeleteUser(&iam.DeleteUserInput{UserName: aws.String(userName)}); err != nil {
|
||||
t.Logf("cleanup: failed to delete user: %v", err)
|
||||
}
|
||||
})
|
||||
|
||||
policyDoc := func(cidrs ...string) string {
|
||||
quoted := make([]string, len(cidrs))
|
||||
for i, c := range cidrs {
|
||||
quoted[i] = `"` + c + `"`
|
||||
}
|
||||
return `{
|
||||
"Version":"2012-10-17",
|
||||
"Statement":[{
|
||||
"Effect":"Allow",
|
||||
"Action":"s3:*",
|
||||
"Resource":["arn:aws:s3:::` + bucketName + `","arn:aws:s3:::` + bucketName + `/*"],
|
||||
"Condition":{"IpAddress":{"aws:SourceIp":[` + strings.Join(quoted, ",") + `]}}
|
||||
}]
|
||||
}`
|
||||
}
|
||||
|
||||
t.Run("denies_when_source_ip_does_not_match", func(t *testing.T) {
|
||||
// SourceIp 198.51.100.0/24 is RFC5737 TEST-NET-2; the test client is on
|
||||
// loopback (127.0.0.1 or ::1 depending on resolver), so the condition
|
||||
// must fail and the action must be denied.
|
||||
_, err = iamClient.PutUserPolicy(&iam.PutUserPolicyInput{
|
||||
UserName: aws.String(userName),
|
||||
PolicyName: aws.String(policyName),
|
||||
PolicyDocument: aws.String(policyDoc("198.51.100.0/24")),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
|
||||
var lastErr error
|
||||
require.Eventually(t, func() bool {
|
||||
_, lastErr = userS3.PutObject(&s3.PutObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String("denied.txt"),
|
||||
Body: aws.ReadSeekCloser(strings.NewReader("nope")),
|
||||
})
|
||||
return isAccessDenied(lastErr)
|
||||
}, 10*time.Second, 500*time.Millisecond,
|
||||
"PutObject must be denied with AccessDenied when aws:SourceIp condition does not match (last error: %v)", lastErr)
|
||||
})
|
||||
|
||||
t.Run("allows_when_source_ip_matches", func(t *testing.T) {
|
||||
// Cover both IPv4 and IPv6 loopback: on CI runners `localhost` may
|
||||
// resolve to ::1 first, in which case a 127.0.0.0/8-only allow would
|
||||
// silently never match and the test would hang.
|
||||
_, err = iamClient.PutUserPolicy(&iam.PutUserPolicyInput{
|
||||
UserName: aws.String(userName),
|
||||
PolicyName: aws.String(policyName),
|
||||
PolicyDocument: aws.String(policyDoc("127.0.0.0/8", "::1/128")),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
|
||||
require.Eventually(t, func() bool {
|
||||
_, err := userS3.PutObject(&s3.PutObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String("allowed.txt"),
|
||||
Body: aws.ReadSeekCloser(strings.NewReader("ok")),
|
||||
})
|
||||
return err == nil
|
||||
}, 10*time.Second, 500*time.Millisecond,
|
||||
"PutObject must succeed when aws:SourceIp condition matches the loopback range")
|
||||
})
|
||||
}
|
||||
|
||||
// TestIAMGroupInlinePolicyEnforcement verifies that PutGroupPolicy is supported
|
||||
// and that the resulting inline policy is enforced for members of the group,
|
||||
// including its Condition block.
|
||||
func TestIAMGroupInlinePolicyEnforcement(t *testing.T) {
|
||||
framework := NewS3IAMTestFramework(t)
|
||||
defer framework.Cleanup()
|
||||
|
||||
iamClient, err := framework.CreateIAMClientWithJWT("admin-user", "TestAdminRole")
|
||||
require.NoError(t, err)
|
||||
|
||||
suffix := uniqueResourceSuffix(t)
|
||||
groupName := "group-" + suffix
|
||||
userName := "user-" + suffix
|
||||
policyName := "policy-" + suffix
|
||||
bucketName := "bucket-" + suffix
|
||||
|
||||
_, err = iamClient.CreateUser(&iam.CreateUserInput{UserName: aws.String(userName)})
|
||||
require.NoError(t, err)
|
||||
|
||||
keyResp, err := iamClient.CreateAccessKey(&iam.CreateAccessKeyInput{
|
||||
UserName: aws.String(userName),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
|
||||
_, err = iamClient.CreateGroup(&iam.CreateGroupInput{GroupName: aws.String(groupName)})
|
||||
require.NoError(t, err)
|
||||
|
||||
_, err = iamClient.AddUserToGroup(&iam.AddUserToGroupInput{
|
||||
GroupName: aws.String(groupName),
|
||||
UserName: aws.String(userName),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
|
||||
userS3 := createS3Client(t, *keyResp.AccessKey.AccessKeyId, *keyResp.AccessKey.SecretAccessKey)
|
||||
|
||||
adminS3, err := framework.CreateS3ClientWithJWT("admin-user", "TestAdminRole")
|
||||
require.NoError(t, err)
|
||||
require.NoError(t, framework.CreateBucketWithCleanup(adminS3, bucketName))
|
||||
|
||||
t.Cleanup(func() {
|
||||
if _, err := iamClient.DeleteGroupPolicy(&iam.DeleteGroupPolicyInput{
|
||||
GroupName: aws.String(groupName),
|
||||
PolicyName: aws.String(policyName),
|
||||
}); err != nil {
|
||||
t.Logf("cleanup: failed to delete group policy: %v", err)
|
||||
}
|
||||
if _, err := iamClient.RemoveUserFromGroup(&iam.RemoveUserFromGroupInput{
|
||||
GroupName: aws.String(groupName),
|
||||
UserName: aws.String(userName),
|
||||
}); err != nil {
|
||||
t.Logf("cleanup: failed to remove user from group: %v", err)
|
||||
}
|
||||
if _, err := iamClient.DeleteAccessKey(&iam.DeleteAccessKeyInput{
|
||||
UserName: aws.String(userName),
|
||||
AccessKeyId: keyResp.AccessKey.AccessKeyId,
|
||||
}); err != nil {
|
||||
t.Logf("cleanup: failed to delete access key: %v", err)
|
||||
}
|
||||
if _, err := iamClient.DeleteUser(&iam.DeleteUserInput{UserName: aws.String(userName)}); err != nil {
|
||||
t.Logf("cleanup: failed to delete user: %v", err)
|
||||
}
|
||||
if _, err := iamClient.DeleteGroup(&iam.DeleteGroupInput{GroupName: aws.String(groupName)}); err != nil {
|
||||
t.Logf("cleanup: failed to delete group: %v", err)
|
||||
}
|
||||
})
|
||||
|
||||
// Cover both IPv4 and IPv6 loopback in the allow CIDR list: on CI runners
|
||||
// `localhost` may resolve to ::1 first, in which case a 127.0.0.0/8-only
|
||||
// allow would silently never match and the test would hang.
|
||||
allowDoc := `{
|
||||
"Version":"2012-10-17",
|
||||
"Statement":[{
|
||||
"Effect":"Allow",
|
||||
"Action":"s3:*",
|
||||
"Resource":["arn:aws:s3:::` + bucketName + `","arn:aws:s3:::` + bucketName + `/*"],
|
||||
"Condition":{"IpAddress":{"aws:SourceIp":["127.0.0.0/8","::1/128"]}}
|
||||
}]
|
||||
}`
|
||||
denyDoc := `{
|
||||
"Version":"2012-10-17",
|
||||
"Statement":[{
|
||||
"Effect":"Allow",
|
||||
"Action":"s3:*",
|
||||
"Resource":["arn:aws:s3:::` + bucketName + `","arn:aws:s3:::` + bucketName + `/*"],
|
||||
"Condition":{"IpAddress":{"aws:SourceIp":"198.51.100.0/24"}}
|
||||
}]
|
||||
}`
|
||||
|
||||
t.Run("crud_round_trip", func(t *testing.T) {
|
||||
_, err := iamClient.PutGroupPolicy(&iam.PutGroupPolicyInput{
|
||||
GroupName: aws.String(groupName),
|
||||
PolicyName: aws.String(policyName),
|
||||
PolicyDocument: aws.String(allowDoc),
|
||||
})
|
||||
require.NoError(t, err, "PutGroupPolicy must succeed (no longer NotImplemented)")
|
||||
|
||||
listResp, err := iamClient.ListGroupPolicies(&iam.ListGroupPoliciesInput{
|
||||
GroupName: aws.String(groupName),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
found := false
|
||||
for _, name := range listResp.PolicyNames {
|
||||
if name != nil && *name == policyName {
|
||||
found = true
|
||||
break
|
||||
}
|
||||
}
|
||||
assert.True(t, found, "ListGroupPolicies must return the freshly added policy")
|
||||
|
||||
getResp, err := iamClient.GetGroupPolicy(&iam.GetGroupPolicyInput{
|
||||
GroupName: aws.String(groupName),
|
||||
PolicyName: aws.String(policyName),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
require.NotNil(t, getResp.PolicyDocument)
|
||||
assert.Contains(t, *getResp.PolicyDocument, "aws:SourceIp",
|
||||
"GetGroupPolicy must round-trip the Condition block")
|
||||
})
|
||||
|
||||
t.Run("enforces_allow_when_condition_matches", func(t *testing.T) {
|
||||
_, err := iamClient.PutGroupPolicy(&iam.PutGroupPolicyInput{
|
||||
GroupName: aws.String(groupName),
|
||||
PolicyName: aws.String(policyName),
|
||||
PolicyDocument: aws.String(allowDoc),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
|
||||
require.Eventually(t, func() bool {
|
||||
_, err := userS3.PutObject(&s3.PutObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String("group-allowed.txt"),
|
||||
Body: aws.ReadSeekCloser(strings.NewReader("ok")),
|
||||
})
|
||||
return err == nil
|
||||
}, 10*time.Second, 500*time.Millisecond,
|
||||
"group member must be allowed when the group policy condition matches")
|
||||
})
|
||||
|
||||
t.Run("enforces_deny_when_condition_does_not_match", func(t *testing.T) {
|
||||
_, err := iamClient.PutGroupPolicy(&iam.PutGroupPolicyInput{
|
||||
GroupName: aws.String(groupName),
|
||||
PolicyName: aws.String(policyName),
|
||||
PolicyDocument: aws.String(denyDoc),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
|
||||
var lastErr error
|
||||
require.Eventually(t, func() bool {
|
||||
_, lastErr = userS3.PutObject(&s3.PutObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String("group-denied.txt"),
|
||||
Body: aws.ReadSeekCloser(strings.NewReader("nope")),
|
||||
})
|
||||
return isAccessDenied(lastErr)
|
||||
}, 10*time.Second, 500*time.Millisecond,
|
||||
"group member must be denied with AccessDenied when the group policy condition does not match (last error: %v)", lastErr)
|
||||
})
|
||||
}
|
||||
@@ -22,6 +22,7 @@ import (
|
||||
"os"
|
||||
"os/exec"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
@@ -69,7 +70,111 @@ func s3Client(t *testing.T) *s3.Client {
|
||||
})),
|
||||
)
|
||||
require.NoError(t, err)
|
||||
return s3.NewFromConfig(cfg, func(o *s3.Options) { o.UsePathStyle = true })
|
||||
client := s3.NewFromConfig(cfg, func(o *s3.Options) { o.UsePathStyle = true })
|
||||
ensureClusterWritable(t, client)
|
||||
return client
|
||||
}
|
||||
|
||||
var clusterWritableOnce sync.Once
|
||||
|
||||
// ensureClusterWritable blocks until the cluster can actually serve a write,
|
||||
// absorbing the volume-growth warmup window after a fresh start. The Makefile
|
||||
// only waits for the server process to be up ("server up after N s"); it does
|
||||
// not wait for a writable volume, so the first PutObject can race volume growth
|
||||
// and fail with a transient 500 (assign volume: DeadlineExceeded) — the source
|
||||
// of the lifecycle-test flakes. Probing one throwaway write here, once per
|
||||
// process, warms growth so every test's first real write is past that window.
|
||||
// Best-effort: if it never becomes writable, the test's own PutObject surfaces
|
||||
// the failure normally.
|
||||
func ensureClusterWritable(t *testing.T, c *s3.Client) {
|
||||
t.Helper()
|
||||
clusterWritableOnce.Do(func() {
|
||||
bucket := uniqueBucket("warmup")
|
||||
deadline := time.Now().Add(60 * time.Second)
|
||||
|
||||
// try runs fn under a bounded context so a single hung call can't block.
|
||||
try := func(timeout time.Duration, fn func(ctx context.Context) error) error {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), timeout)
|
||||
defer cancel()
|
||||
return fn(ctx)
|
||||
}
|
||||
// probe is a try whose timeout is clamped to the time left before the
|
||||
// deadline, so the whole warmup stays within the budget; ok=false means
|
||||
// the budget is exhausted and the caller should stop.
|
||||
probe := func(fn func(ctx context.Context) error) (err error, ok bool) {
|
||||
remaining := time.Until(deadline)
|
||||
if remaining <= 0 {
|
||||
return nil, false
|
||||
}
|
||||
if remaining > 10*time.Second {
|
||||
remaining = 10 * time.Second
|
||||
}
|
||||
return try(remaining, fn), true
|
||||
}
|
||||
// backoff sleeps attempt*250ms, never past the deadline.
|
||||
backoff := func(attempt int) {
|
||||
d := time.Duration(attempt) * 250 * time.Millisecond
|
||||
if left := time.Until(deadline); d > left {
|
||||
d = left
|
||||
}
|
||||
if d > 0 {
|
||||
time.Sleep(d)
|
||||
}
|
||||
}
|
||||
|
||||
// CreateBucket is a metadata op, but on a cold cluster the filer itself may
|
||||
// not be ready yet, so retry it within the deadline rather than abandoning
|
||||
// the whole warmup (and the PutObject probe) on the first error.
|
||||
created := false
|
||||
for attempt := 1; ; attempt++ {
|
||||
err, ok := probe(func(ctx context.Context) error {
|
||||
_, e := c.CreateBucket(ctx, &s3.CreateBucketInput{Bucket: aws.String(bucket)})
|
||||
return e
|
||||
})
|
||||
if !ok {
|
||||
break
|
||||
}
|
||||
if err == nil {
|
||||
created = true
|
||||
break
|
||||
}
|
||||
backoff(attempt)
|
||||
}
|
||||
if !created {
|
||||
t.Logf("warmup: could not create probe bucket within 60s; proceeding")
|
||||
return
|
||||
}
|
||||
// Cleanup gets a fresh timeout (not the warmup budget) so teardown runs
|
||||
// even when the probe loop used the full window.
|
||||
defer try(10*time.Second, func(ctx context.Context) error {
|
||||
_, e := c.DeleteBucket(ctx, &s3.DeleteBucketInput{Bucket: aws.String(bucket)})
|
||||
return e
|
||||
})
|
||||
|
||||
for attempt := 1; ; attempt++ {
|
||||
err, ok := probe(func(ctx context.Context) error {
|
||||
_, e := c.PutObject(ctx, &s3.PutObjectInput{
|
||||
Bucket: aws.String(bucket), Key: aws.String("warmup"), Body: strings.NewReader("ok"),
|
||||
})
|
||||
return e
|
||||
})
|
||||
if !ok {
|
||||
break
|
||||
}
|
||||
if err == nil {
|
||||
try(10*time.Second, func(ctx context.Context) error {
|
||||
_, e := c.DeleteObject(ctx, &s3.DeleteObjectInput{Bucket: aws.String(bucket), Key: aws.String("warmup")})
|
||||
return e
|
||||
})
|
||||
if attempt > 1 {
|
||||
t.Logf("cluster became writable after %d probe(s)", attempt)
|
||||
}
|
||||
return
|
||||
}
|
||||
backoff(attempt)
|
||||
}
|
||||
t.Logf("warmup: cluster not confirmed writable within 60s; proceeding")
|
||||
})
|
||||
}
|
||||
|
||||
func filerClient(t *testing.T) (filer_pb.SeaweedFilerClient, func()) {
|
||||
|
||||
@@ -0,0 +1,41 @@
|
||||
# AWS SDK V2 Route Disambiguation Integration Tests
|
||||
#
|
||||
# Pins the regression for the route collision between the regular S3 API
|
||||
# and the S3 Tables REST API on shared top-level paths (/buckets,
|
||||
# /get-table). The tests use the real AWS SDK V2 for Go so the SDK's own
|
||||
# XML deserializer is the assertion — a JSON body produces an SDK error
|
||||
# before any test code runs.
|
||||
#
|
||||
# Prerequisites:
|
||||
# - SeaweedFS running with S3 API enabled on port 8333
|
||||
# - Go 1.21+
|
||||
#
|
||||
# Usage:
|
||||
# make test - Run the SDK V2 routing tests
|
||||
# make test-verbose - Run with verbose output
|
||||
# make clean - Clean test cache
|
||||
|
||||
.PHONY: all test test-verbose clean help
|
||||
|
||||
S3_ENDPOINT ?= http://127.0.0.1:8333
|
||||
|
||||
all: test
|
||||
|
||||
test:
|
||||
@echo "Running SDK V2 routing tests against $(S3_ENDPOINT)..."
|
||||
S3_ENDPOINT=$(S3_ENDPOINT) go test -v -timeout 5m ./...
|
||||
|
||||
test-verbose:
|
||||
S3_ENDPOINT=$(S3_ENDPOINT) go test -v -timeout 5m -count=1 ./...
|
||||
|
||||
clean:
|
||||
go clean -testcache
|
||||
|
||||
help:
|
||||
@echo "AWS SDK V2 Route Disambiguation Tests"
|
||||
@echo "Targets:"
|
||||
@echo " test Run the routing tests"
|
||||
@echo " test-verbose Run with verbose output"
|
||||
@echo " clean Clean test cache"
|
||||
@echo "Environment Variables:"
|
||||
@echo " S3_ENDPOINT S3 endpoint URL (default: http://127.0.0.1:8333)"
|
||||
@@ -0,0 +1,210 @@
|
||||
// Package sdkv2routing_test exercises route disambiguation between the
|
||||
// regular S3 API and the S3 Tables REST API on top-level paths the two
|
||||
// share (/buckets, /get-table). The bug it pins:
|
||||
//
|
||||
// When a user has an S3 bucket named "buckets" (or "get-table"), a
|
||||
// path-style ListObjectsV2 request sent by AWS SDK V2 / Hadoop s3a /
|
||||
// Spark would be routed to the S3 Tables ListTableBuckets handler and
|
||||
// receive a JSON body. AWS SDK V2 then fails XML parsing with
|
||||
// "Unexpected character '{' (code 123) in prolog".
|
||||
//
|
||||
// These tests use the real AWS SDK V2 for Go, so the SDK's own
|
||||
// deserializer is the assertion: if the server returns the wrong
|
||||
// content type, the SDK errors out before any test assertion runs.
|
||||
package sdkv2routing_test
|
||||
|
||||
import (
|
||||
"context"
|
||||
"io"
|
||||
"os"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/aws/aws-sdk-go-v2/aws"
|
||||
"github.com/aws/aws-sdk-go-v2/config"
|
||||
"github.com/aws/aws-sdk-go-v2/credentials"
|
||||
"github.com/aws/aws-sdk-go-v2/service/s3"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
const (
|
||||
defaultEndpoint = "http://127.0.0.1:8333"
|
||||
defaultAccessKey = "some_access_key1"
|
||||
defaultSecretKey = "some_secret_key1"
|
||||
defaultRegion = "us-east-1"
|
||||
)
|
||||
|
||||
func getS3Client(t *testing.T) *s3.Client {
|
||||
t.Helper()
|
||||
|
||||
endpoint := os.Getenv("S3_ENDPOINT")
|
||||
if endpoint == "" {
|
||||
endpoint = defaultEndpoint
|
||||
}
|
||||
accessKey := os.Getenv("AWS_ACCESS_KEY_ID")
|
||||
if accessKey == "" {
|
||||
accessKey = defaultAccessKey
|
||||
}
|
||||
secretKey := os.Getenv("AWS_SECRET_ACCESS_KEY")
|
||||
if secretKey == "" {
|
||||
secretKey = defaultSecretKey
|
||||
}
|
||||
region := os.Getenv("AWS_REGION")
|
||||
if region == "" {
|
||||
region = defaultRegion
|
||||
}
|
||||
|
||||
cfg, err := config.LoadDefaultConfig(context.TODO(),
|
||||
config.WithRegion(region),
|
||||
config.WithCredentialsProvider(credentials.NewStaticCredentialsProvider(accessKey, secretKey, "")),
|
||||
config.WithEndpointResolverWithOptions(aws.EndpointResolverWithOptionsFunc(
|
||||
func(service, region string, options ...interface{}) (aws.Endpoint, error) {
|
||||
return aws.Endpoint{
|
||||
URL: endpoint,
|
||||
SigningRegion: defaultRegion,
|
||||
HostnameImmutable: true,
|
||||
}, nil
|
||||
})),
|
||||
)
|
||||
require.NoError(t, err)
|
||||
|
||||
return s3.NewFromConfig(cfg, func(o *s3.Options) {
|
||||
o.UsePathStyle = true
|
||||
})
|
||||
}
|
||||
|
||||
// ensureBucket creates bucket if it doesn't already exist. It tolerates
|
||||
// BucketAlreadyOwnedByYou / BucketAlreadyExists so the tests are
|
||||
// idempotent across local re-runs.
|
||||
func ensureBucket(t *testing.T, ctx context.Context, client *s3.Client, bucket string) {
|
||||
t.Helper()
|
||||
_, err := client.CreateBucket(ctx, &s3.CreateBucketInput{Bucket: aws.String(bucket)})
|
||||
if err == nil {
|
||||
return
|
||||
}
|
||||
msg := err.Error()
|
||||
if strings.Contains(msg, "BucketAlreadyOwnedByYou") || strings.Contains(msg, "BucketAlreadyExists") {
|
||||
return
|
||||
}
|
||||
t.Fatalf("CreateBucket(%q) failed: %v", bucket, err)
|
||||
}
|
||||
|
||||
// deleteBucket best-effort cleans up a bucket and any objects in it.
|
||||
// Test does not fail if cleanup fails — the next run is idempotent.
|
||||
func deleteBucket(ctx context.Context, client *s3.Client, bucket string) {
|
||||
paginator := s3.NewListObjectsV2Paginator(client, &s3.ListObjectsV2Input{Bucket: aws.String(bucket)})
|
||||
for paginator.HasMorePages() {
|
||||
page, err := paginator.NextPage(ctx)
|
||||
if err != nil {
|
||||
break
|
||||
}
|
||||
for _, obj := range page.Contents {
|
||||
client.DeleteObject(ctx, &s3.DeleteObjectInput{Bucket: aws.String(bucket), Key: obj.Key})
|
||||
}
|
||||
}
|
||||
client.DeleteBucket(ctx, &s3.DeleteBucketInput{Bucket: aws.String(bucket)})
|
||||
}
|
||||
|
||||
// TestListObjectsV2_OnBucketNamedBuckets is the direct reproducer for
|
||||
// issue #9559: Spark / Hadoop s3a does a ListObjectsV2 against bucket
|
||||
// "buckets" via AWS SDK V2, which fails with
|
||||
// "Could not parse XML response. ... Unexpected character '{' (code 123)
|
||||
// in prolog" when SeaweedFS routes the request to the JSON-returning
|
||||
// ListTableBuckets handler. The SDK's response deserializer is the
|
||||
// real assertion here — a JSON body produces an SDK error before we
|
||||
// reach require.NoError.
|
||||
func TestListObjectsV2_OnBucketNamedBuckets(t *testing.T) {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
defer cancel()
|
||||
client := getS3Client(t)
|
||||
|
||||
const bucket = "buckets"
|
||||
ensureBucket(t, ctx, client, bucket)
|
||||
t.Cleanup(func() { deleteBucket(ctx, client, bucket) })
|
||||
|
||||
out, err := client.ListObjectsV2(ctx, &s3.ListObjectsV2Input{
|
||||
Bucket: aws.String(bucket),
|
||||
Prefix: aws.String("logs/"),
|
||||
})
|
||||
require.NoError(t, err, "AWS SDK V2 must parse the response as XML")
|
||||
require.NotNil(t, out)
|
||||
}
|
||||
|
||||
// TestPutGetObject_OnBucketNamedBuckets exercises the full read/write
|
||||
// round-trip on the colliding bucket name. PutObject and GetObject go
|
||||
// through different routes than ListObjectsV2, and verifying them
|
||||
// guards against future regressions that re-route only some verbs.
|
||||
func TestPutGetObject_OnBucketNamedBuckets(t *testing.T) {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
defer cancel()
|
||||
client := getS3Client(t)
|
||||
|
||||
const bucket = "buckets"
|
||||
const key = "logs/hello.txt"
|
||||
const body = "hello from issue 9559"
|
||||
|
||||
ensureBucket(t, ctx, client, bucket)
|
||||
t.Cleanup(func() { deleteBucket(ctx, client, bucket) })
|
||||
|
||||
_, err := client.PutObject(ctx, &s3.PutObjectInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String(key),
|
||||
Body: strings.NewReader(body),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
|
||||
out, err := client.GetObject(ctx, &s3.GetObjectInput{Bucket: aws.String(bucket), Key: aws.String(key)})
|
||||
require.NoError(t, err)
|
||||
defer out.Body.Close()
|
||||
|
||||
got, err := io.ReadAll(out.Body)
|
||||
require.NoError(t, err)
|
||||
require.Equal(t, body, string(got))
|
||||
}
|
||||
|
||||
// TestListObjectsV2_OnBucketNamedGetTable covers the second colliding
|
||||
// path. The S3 Tables GET /get-table endpoint shares its path with a
|
||||
// bucket literally named "get-table", which is a legal S3 bucket name.
|
||||
func TestListObjectsV2_OnBucketNamedGetTable(t *testing.T) {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
defer cancel()
|
||||
client := getS3Client(t)
|
||||
|
||||
const bucket = "get-table"
|
||||
ensureBucket(t, ctx, client, bucket)
|
||||
t.Cleanup(func() { deleteBucket(ctx, client, bucket) })
|
||||
|
||||
out, err := client.ListObjectsV2(ctx, &s3.ListObjectsV2Input{Bucket: aws.String(bucket)})
|
||||
require.NoError(t, err)
|
||||
require.NotNil(t, out)
|
||||
}
|
||||
|
||||
// TestCreateAndListBuckets_ServiceLevel verifies the SDK's service-level
|
||||
// ListBuckets still parses as XML when a bucket named "buckets" exists.
|
||||
// ListBuckets goes through GET / (root) which is unaffected by the
|
||||
// /buckets route collision — this is a guard against the matcher
|
||||
// accidentally widening to top-level paths.
|
||||
func TestCreateAndListBuckets_ServiceLevel(t *testing.T) {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
defer cancel()
|
||||
client := getS3Client(t)
|
||||
|
||||
const bucket = "buckets"
|
||||
ensureBucket(t, ctx, client, bucket)
|
||||
t.Cleanup(func() { deleteBucket(ctx, client, bucket) })
|
||||
|
||||
out, err := client.ListBuckets(ctx, &s3.ListBucketsInput{})
|
||||
require.NoError(t, err)
|
||||
require.NotNil(t, out)
|
||||
|
||||
found := false
|
||||
for _, b := range out.Buckets {
|
||||
if aws.ToString(b.Name) == bucket {
|
||||
found = true
|
||||
break
|
||||
}
|
||||
}
|
||||
require.True(t, found, "bucket %q must appear in service-level ListBuckets", bucket)
|
||||
}
|
||||
|
||||
@@ -17,6 +17,9 @@ import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/aws/aws-sdk-go/aws/credentials"
|
||||
v4 "github.com/aws/aws-sdk-go/aws/signer/v4"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/test/testutil"
|
||||
)
|
||||
|
||||
@@ -451,19 +454,22 @@ func icebergPath(prefix, path string) string {
|
||||
return withPrefix
|
||||
}
|
||||
|
||||
// createTableBucket creates a table bucket via the S3Tables REST API
|
||||
// createTableBucket creates a table bucket via the S3Tables REST API.
|
||||
// The request is AWS V4 signed for SERVICE=s3tables so the S3 Tables
|
||||
// route matcher accepts it; signing with regular SERVICE=s3 would let
|
||||
// the request fall through to the S3 CreateBucket handler.
|
||||
func createTableBucket(t *testing.T, env *TestEnvironment, bucketName string) {
|
||||
t.Helper()
|
||||
|
||||
// Use S3Tables REST API to create the bucket
|
||||
endpoint := fmt.Sprintf("http://localhost:%d/buckets", env.s3Port)
|
||||
|
||||
reqBody := fmt.Sprintf(`{"name":"%s"}`, bucketName)
|
||||
|
||||
req, err := http.NewRequest(http.MethodPut, endpoint, strings.NewReader(reqBody))
|
||||
if err != nil {
|
||||
t.Fatalf("Failed to create request: %v", err)
|
||||
}
|
||||
req.Header.Set("Content-Type", "application/x-amz-json-1.1")
|
||||
signS3TablesRequest(t, req, reqBody)
|
||||
|
||||
resp, err := http.DefaultClient.Do(req)
|
||||
if err != nil {
|
||||
@@ -480,6 +486,20 @@ func createTableBucket(t *testing.T, env *TestEnvironment, bucketName string) {
|
||||
t.Logf("Created table bucket %s", bucketName)
|
||||
}
|
||||
|
||||
// signS3TablesRequest signs req with AWS V4 for SERVICE=s3tables. The
|
||||
// underlying weed mini instance runs in default-allow mode so the
|
||||
// signature itself is not verified; only the credential scope matters,
|
||||
// because the S3 Tables route matcher requires SERVICE=s3tables to
|
||||
// distinguish S3 Tables traffic from regular S3 calls on the same paths.
|
||||
func signS3TablesRequest(t *testing.T, req *http.Request, body string) {
|
||||
t.Helper()
|
||||
creds := credentials.NewStaticCredentials("test-ak", "test-sk", "")
|
||||
signer := v4.NewSigner(creds)
|
||||
if _, err := signer.Sign(req, strings.NewReader(body), "s3tables", "us-east-1", time.Now()); err != nil {
|
||||
t.Fatalf("Failed to sign S3 Tables request: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// randomSuffix returns a short random hex suffix for unique resource naming.
|
||||
func randomSuffix() string {
|
||||
return fmt.Sprintf("%x", time.Now().UnixNano()&0xffffffff)
|
||||
|
||||
@@ -411,7 +411,12 @@ func createIcebergTable(t *testing.T, env *TestEnvironment, bucketName, namespac
|
||||
func listFilerContents(t *testing.T, env *TestEnvironment, path string) {
|
||||
t.Helper()
|
||||
|
||||
cmd := exec.Command("weed", "shell",
|
||||
// Bound diagnostic listing so a hung weed shell during cleanup can't
|
||||
// burn the whole 20-minute test timeout.
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
defer cancel()
|
||||
|
||||
cmd := exec.CommandContext(ctx, "weed", "shell",
|
||||
fmt.Sprintf("-master=%s", env.hostMasterAddress()),
|
||||
)
|
||||
cmd.Stdin = strings.NewReader(fmt.Sprintf("fs.ls -R %s\nexit\n", path))
|
||||
|
||||
@@ -0,0 +1,20 @@
|
||||
FROM chrislusf/seaweedfs:e2e
|
||||
|
||||
RUN apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 update && \
|
||||
DEBIAN_FRONTEND=noninteractive apt-get -o Acquire::Retries=5 -o Acquire::http::Timeout=30 install -y \
|
||||
--no-install-recommends \
|
||||
--no-install-suggests \
|
||||
samba \
|
||||
smbclient \
|
||||
python3-minimal \
|
||||
&& apt-get clean \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
COPY smb.conf.template /smb.conf.template
|
||||
COPY smb_tests.sh /smb_tests.sh
|
||||
COPY lock_tests.sh /lock_tests.sh
|
||||
COPY entrypoint.sh /entrypoint.sh
|
||||
COPY run_inside_container.sh /run_inside_container.sh
|
||||
RUN chmod +x /smb_tests.sh /lock_tests.sh /entrypoint.sh /run_inside_container.sh
|
||||
|
||||
ENTRYPOINT ["/entrypoint.sh"]
|
||||
@@ -0,0 +1,96 @@
|
||||
# Samba on FUSE integration test
|
||||
|
||||
Exports a SeaweedFS FUSE mount over SMB with Samba's `smbd` and drives it with
|
||||
`smbclient`, verifying that SMB file operations work correctly on top of the
|
||||
mount and that data stays consistent across both protocols.
|
||||
|
||||
## What it checks
|
||||
|
||||
The functional battery in `smb_tests.sh` covers:
|
||||
|
||||
- connecting to the share and listing the root
|
||||
- 1 MiB upload/download round-trip with content verification
|
||||
- subdirectory creation and writes into it
|
||||
- file rename
|
||||
- 64 MiB upload/download (exercises SeaweedFS chunk splitting)
|
||||
- recursive upload of a directory tree
|
||||
- cross-protocol consistency: files written over SMB appear on the FUSE mount
|
||||
with identical content, and files written directly on the FUSE mount are
|
||||
readable over SMB
|
||||
- deleting files and directory trees
|
||||
|
||||
The locking / concurrency battery in `lock_tests.sh` covers the harder cases a
|
||||
network-filesystem backend has to get right:
|
||||
|
||||
- **POSIX `fcntl` byte-range locking** on the FUSE mount: a held exclusive lock
|
||||
denies a conflicting lock, allows a non-overlapping range, and is reacquirable
|
||||
after release (exercises the mount's `SetLk`/`GetLk`)
|
||||
- **Distributed locking** (`-dlm`): a file held open for writing on one mount
|
||||
blocks a writer on a second mount until it is released
|
||||
- **Distributed-lock integrity**: concurrent writers to the same file from two
|
||||
mounts leave exactly one intact payload, never a torn mix
|
||||
- **Concurrency**: parallel writers to distinct files all succeed
|
||||
|
||||
Both FUSE mounts are started with `-dlm` (distributed lock manager). The second
|
||||
mount (`/mnt/seaweedfs2`) exists only to contend with the smbd-backed mount in
|
||||
the distributed-locking tests; both see the same filer path, so `.../share` is
|
||||
the same data on each.
|
||||
|
||||
> Note on DLM semantics: `-dlm` coordinates *write access* (one mount writes a
|
||||
> file at a time) and guarantees writes are not torn. It does not guarantee
|
||||
> which concurrent writer wins or instant cross-mount read convergence — the
|
||||
> holder's buffered data is flushed on close, asynchronously to lock release.
|
||||
> When a holder closes the file, a writer on another mount acquires the freed
|
||||
> lock within ~1s and completes.
|
||||
|
||||
## Layout
|
||||
|
||||
| File | Purpose |
|
||||
| --- | --- |
|
||||
| `smb_tests.sh` | SMB functional battery. Shared by both runners. |
|
||||
| `lock_tests.sh` | SMB locking / concurrency battery. Shared by both runners. |
|
||||
| `smb.conf.template` | Samba config; placeholders are filled in at run time. |
|
||||
| `run.sh` | Local runner: `weed mini` + two `-dlm` mounts + `smbd` + both batteries, all as the current user on unprivileged ports. |
|
||||
| `entrypoint.sh` | Container entrypoint: starts two `-dlm` FUSE mounts and runs `smbd`. |
|
||||
| `run_inside_container.sh` | Runs both batteries inside the container against the local `smbd`. |
|
||||
| `Dockerfile` | Adds Samba to the `chrislusf/seaweedfs:e2e` image. |
|
||||
| `docker-compose.yml` | master + volume + filer + samba services. |
|
||||
|
||||
## Running locally
|
||||
|
||||
Requirements: `weed` on `$PATH`, `fusermount3`, and Samba's `smbd` /
|
||||
`smbclient` / `smbpasswd` (Debian/Ubuntu: `apt-get install samba smbclient`).
|
||||
|
||||
```sh
|
||||
test/samba/run.sh
|
||||
```
|
||||
|
||||
No `sudo` is needed: `smbd` runs as the current user on port 4450 and all state
|
||||
lives under a temp work dir that is cleaned up on exit.
|
||||
|
||||
## Running with Docker
|
||||
|
||||
Mirrors the CI job. Requires `/dev/fuse` and `SYS_ADMIN` (provided in the
|
||||
compose file).
|
||||
|
||||
```sh
|
||||
# build the base e2e image first (from the repo's docker/ dir)
|
||||
docker compose -f test/samba/docker-compose.yml up --wait
|
||||
docker compose -f test/samba/docker-compose.yml exec -T samba /run_inside_container.sh
|
||||
docker compose -f test/samba/docker-compose.yml down -v
|
||||
```
|
||||
|
||||
## CI
|
||||
|
||||
`.github/workflows/samba-integration.yml` runs on changes to `weed/mount/**`,
|
||||
`weed/filer/**`, or `test/samba/**`. It builds the e2e image, builds the Samba
|
||||
harness image on top, brings up the cluster, runs the battery, and uploads
|
||||
server logs as artifacts.
|
||||
|
||||
## Notes
|
||||
|
||||
- The share disables Samba's DOS-attribute / xattr mapping and oplocks. The
|
||||
SeaweedFS FUSE mount does not implement that surface, and leaving it on
|
||||
produces `NT_STATUS_NOT_SUPPORTED` errors unrelated to data integrity.
|
||||
- The share path is a subdirectory of the mount (`.../share`) so the runner can
|
||||
verify SMB-side operations directly on the FUSE side.
|
||||
@@ -0,0 +1,60 @@
|
||||
services:
|
||||
master:
|
||||
image: chrislusf/seaweedfs:e2e
|
||||
command: "-v=4 master -ip=master -ip.bind=0.0.0.0 -raftBootstrap"
|
||||
healthcheck:
|
||||
test: ["CMD", "curl", "--fail", "-I", "http://localhost:9333/cluster/healthz"]
|
||||
interval: 2s
|
||||
timeout: 10s
|
||||
retries: 30
|
||||
start_period: 10s
|
||||
|
||||
volume:
|
||||
image: chrislusf/seaweedfs:e2e
|
||||
command: "-v=4 volume -master=master:9333 -ip=volume -ip.bind=0.0.0.0 -preStopSeconds=1"
|
||||
healthcheck:
|
||||
test: ["CMD", "curl", "--fail", "-I", "http://localhost:8080/healthz"]
|
||||
interval: 2s
|
||||
timeout: 10s
|
||||
retries: 15
|
||||
start_period: 5s
|
||||
depends_on:
|
||||
master:
|
||||
condition: service_healthy
|
||||
|
||||
filer:
|
||||
image: chrislusf/seaweedfs:e2e
|
||||
command: "-v=4 filer -master=master:9333 -ip=filer -ip.bind=0.0.0.0"
|
||||
healthcheck:
|
||||
test: ["CMD", "curl", "--fail", "-I", "http://localhost:8888/healthz"]
|
||||
interval: 2s
|
||||
timeout: 10s
|
||||
retries: 15
|
||||
start_period: 5s
|
||||
depends_on:
|
||||
volume:
|
||||
condition: service_healthy
|
||||
|
||||
samba:
|
||||
image: chrislusf/seaweedfs:samba
|
||||
build:
|
||||
context: .
|
||||
environment:
|
||||
FILER: filer:8888
|
||||
cap_add:
|
||||
- SYS_ADMIN
|
||||
devices:
|
||||
- /dev/fuse
|
||||
security_opt:
|
||||
- apparmor:unconfined
|
||||
healthcheck:
|
||||
test:
|
||||
- "CMD-SHELL"
|
||||
- "mountpoint -q /mnt/seaweedfs && mountpoint -q /mnt/seaweedfs2 && smbclient -L 127.0.0.1 -p 445 -U smbtest%smbtest -m SMB3 >/dev/null 2>&1"
|
||||
interval: 3s
|
||||
timeout: 10s
|
||||
retries: 20
|
||||
start_period: 15s
|
||||
depends_on:
|
||||
filer:
|
||||
condition: service_healthy
|
||||
Executable
+76
@@ -0,0 +1,76 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# Entrypoint for the samba test container.
|
||||
#
|
||||
# Mounts SeaweedFS over FUSE twice, both with distributed locking (-dlm) so the
|
||||
# locking tests can exercise cross-mount write coordination:
|
||||
# - MOUNT_DIR (/mnt/seaweedfs) is exported over SMB by smbd
|
||||
# - MOUNT2_DIR (/mnt/seaweedfs2) is a second, independent mount of the same
|
||||
# filer used to contend with the SMB writer
|
||||
#
|
||||
# Both mounts see the same filer path, so .../share is the same data on each.
|
||||
# smbd runs in the foreground (as root, which owns the mounts), so the share
|
||||
# uses "force user = root".
|
||||
set -euo pipefail
|
||||
|
||||
FILER="${FILER:-filer:8888}"
|
||||
MOUNT_DIR="${MOUNT_DIR:-/mnt/seaweedfs}"
|
||||
MOUNT2_DIR="${MOUNT2_DIR:-/mnt/seaweedfs2}"
|
||||
SHARE_DIR="${MOUNT_DIR}/share"
|
||||
STATE_DIR="${STATE_DIR:-/var/lib/samba-test}"
|
||||
SMB_PORT="${SMB_PORT:-445}"
|
||||
SMB_USER="${SMB_USER:-smbtest}"
|
||||
SMB_PASS="${SMB_PASS:-smbtest}"
|
||||
|
||||
mkdir -p "${MOUNT_DIR}" "${MOUNT2_DIR}" \
|
||||
"${STATE_DIR}/private" "${STATE_DIR}/state" "${STATE_DIR}/cache" \
|
||||
"${STATE_DIR}/lock" "${STATE_DIR}/pid" "${STATE_DIR}/ncalrpc"
|
||||
|
||||
# mount_seaweedfs <mountpoint> <logfile> — mount with -dlm and wait for it.
|
||||
mount_seaweedfs() {
|
||||
local dir="$1" log="$2"
|
||||
echo "==> Mounting SeaweedFS (${FILER}) at ${dir} with -dlm"
|
||||
weed -v=1 mount \
|
||||
-filer="${FILER}" \
|
||||
-dir="${dir}" \
|
||||
-filer.path=/ \
|
||||
-dirAutoCreate \
|
||||
-allowOthers \
|
||||
-dlm \
|
||||
>"${log}" 2>&1 &
|
||||
local pid=$!
|
||||
for _ in $(seq 1 120); do
|
||||
if mountpoint -q "${dir}"; then
|
||||
return 0
|
||||
fi
|
||||
if ! kill -0 "${pid}" 2>/dev/null; then
|
||||
echo "weed mount (${dir}) exited early; log tail:" >&2
|
||||
tail -n 100 "${log}" >&2 || true
|
||||
exit 1
|
||||
fi
|
||||
sleep 0.5
|
||||
done
|
||||
echo "FUSE mount ${dir} did not come up" >&2
|
||||
tail -n 100 "${log}" >&2 || true
|
||||
exit 1
|
||||
}
|
||||
|
||||
mount_seaweedfs "${MOUNT_DIR}" /var/log/weed-mount.log
|
||||
mount_seaweedfs "${MOUNT2_DIR}" /var/log/weed-mount2.log
|
||||
|
||||
mkdir -p "${SHARE_DIR}"
|
||||
chmod 0777 "${SHARE_DIR}"
|
||||
|
||||
# --- configure and start smbd ----------------------------------------------
|
||||
echo "==> Configuring Samba share on port ${SMB_PORT}"
|
||||
sed -e "s#@SHARE_PATH@#${SHARE_DIR}#g" \
|
||||
-e "s#@STATE_DIR@#${STATE_DIR}#g" \
|
||||
-e "s#@SMB_PORT@#${SMB_PORT}#g" \
|
||||
-e "s#@FORCE_USER@#root#g" \
|
||||
/smb.conf.template >/etc/samba/smb.conf
|
||||
|
||||
id -u "${SMB_USER}" >/dev/null 2>&1 || useradd -M -s /usr/sbin/nologin "${SMB_USER}"
|
||||
printf '%s\n%s\n' "${SMB_PASS}" "${SMB_PASS}" | smbpasswd -a -s "${SMB_USER}"
|
||||
|
||||
echo "==> Starting smbd"
|
||||
exec smbd -F --no-process-group -s /etc/samba/smb.conf
|
||||
Executable
+218
@@ -0,0 +1,218 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# Locking / concurrency test battery for Samba on a SeaweedFS FUSE mount.
|
||||
#
|
||||
# Covers the challenges a network-filesystem backend has to get right:
|
||||
# 1. POSIX fcntl byte-range locking on the FUSE mount (SetLk/GetLk)
|
||||
# 2. Distributed locking (-dlm): a write held open on one mount blocks a
|
||||
# writer on another mount until it is released
|
||||
# 3. Distributed locking integrity: concurrent writers to the same file from
|
||||
# two mounts produce intact (non-torn) data
|
||||
# 4. Concurrent writers to distinct files all succeed
|
||||
#
|
||||
# Required env:
|
||||
# SMB_USER, SMB_PASS samba credentials
|
||||
# MOUNT_SHARE dir on the smbd-backed FUSE mount (mount 1)
|
||||
# MOUNT2_SHARE dir on the second FUSE mount (mount 2)
|
||||
# Optional env:
|
||||
# SMB_HOST (127.0.0.1), SMB_SHARE (seaweedfs), SMB_PORT (445)
|
||||
set -uo pipefail
|
||||
|
||||
SMB_HOST="${SMB_HOST:-127.0.0.1}"
|
||||
SMB_SHARE="${SMB_SHARE:-seaweedfs}"
|
||||
SMB_PORT="${SMB_PORT:-445}"
|
||||
SMB_USER="${SMB_USER:?SMB_USER is required}"
|
||||
SMB_PASS="${SMB_PASS:?SMB_PASS is required}"
|
||||
MOUNT_SHARE="${MOUNT_SHARE:?MOUNT_SHARE is required}"
|
||||
MOUNT2_SHARE="${MOUNT2_SHARE:?MOUNT2_SHARE is required}"
|
||||
|
||||
WORK="$(mktemp -d /tmp/samba-locktest.XXXXXX)"
|
||||
trap 'rm -rf "${WORK}"' EXIT
|
||||
|
||||
PASS=0
|
||||
FAIL=0
|
||||
pass() { printf ' [PASS] %s\n' "$1"; PASS=$((PASS + 1)); }
|
||||
fail() { printf ' [FAIL] %s\n' "$1"; FAIL=$((FAIL + 1)); }
|
||||
|
||||
smb() {
|
||||
smbclient "//${SMB_HOST}/${SMB_SHARE}" -p "${SMB_PORT}" \
|
||||
-U "${SMB_USER}%${SMB_PASS}" -m SMB3 -c "$1"
|
||||
}
|
||||
md5() { md5sum "$1" | awk '{print $1}'; }
|
||||
|
||||
# 1. POSIX fcntl byte-range locking on the FUSE mount ------------------------
|
||||
# Exercises the mount's SetLk/GetLk via two processes contending over fcntl
|
||||
# (F_SETLK) byte-range locks. python3's fcntl.lockf issues real POSIX locks.
|
||||
echo "==> 1. POSIX fcntl byte-range locking (FUSE mount SetLk/GetLk)"
|
||||
lockfile="${MOUNT_SHARE}/fcntl_lock.dat"
|
||||
: >"${lockfile}"
|
||||
fcntl_out="$(python3 - "${lockfile}" <<'PY'
|
||||
import fcntl, os, sys
|
||||
|
||||
path = sys.argv[1]
|
||||
parent_to_child_r, parent_to_child_w = os.pipe() # release signal
|
||||
child_to_parent_r, child_to_parent_w = os.pipe() # locked signal
|
||||
|
||||
pid = os.fork()
|
||||
if pid == 0: # child: hold an exclusive lock on [0,100)
|
||||
fd = os.open(path, os.O_RDWR | os.O_CREAT, 0o644)
|
||||
fcntl.lockf(fd, fcntl.LOCK_EX, 100, 0, 0)
|
||||
os.write(child_to_parent_w, b"L")
|
||||
os.read(parent_to_child_r, 1) # wait until parent says release
|
||||
fcntl.lockf(fd, fcntl.LOCK_UN, 100, 0, 0)
|
||||
os.close(fd)
|
||||
os._exit(0)
|
||||
|
||||
# parent
|
||||
os.read(child_to_parent_r, 1) # wait until child holds the lock
|
||||
fd = os.open(path, os.O_RDWR | os.O_CREAT, 0o644)
|
||||
results = []
|
||||
|
||||
# a. a conflicting exclusive lock must be denied while the child holds it
|
||||
try:
|
||||
fcntl.lockf(fd, fcntl.LOCK_EX | fcntl.LOCK_NB, 100, 0, 0)
|
||||
fcntl.lockf(fd, fcntl.LOCK_UN, 100, 0, 0)
|
||||
results.append(("conflicting exclusive lock denied while held", False))
|
||||
except OSError:
|
||||
results.append(("conflicting exclusive lock denied while held", True))
|
||||
|
||||
# b. a non-overlapping range must be grantable
|
||||
try:
|
||||
fcntl.lockf(fd, fcntl.LOCK_EX | fcntl.LOCK_NB, 100, 200, 0)
|
||||
fcntl.lockf(fd, fcntl.LOCK_UN, 100, 200, 0)
|
||||
results.append(("non-overlapping range lock granted", True))
|
||||
except OSError:
|
||||
results.append(("non-overlapping range lock granted", False))
|
||||
|
||||
# c. after the holder releases, the lock must be acquirable
|
||||
os.write(parent_to_child_w, b"R")
|
||||
os.waitpid(pid, 0)
|
||||
try:
|
||||
fcntl.lockf(fd, fcntl.LOCK_EX | fcntl.LOCK_NB, 100, 0, 0)
|
||||
fcntl.lockf(fd, fcntl.LOCK_UN, 100, 0, 0)
|
||||
results.append(("lock acquirable after holder releases", True))
|
||||
except OSError:
|
||||
results.append(("lock acquirable after holder releases", False))
|
||||
|
||||
for name, ok in results:
|
||||
print((" [PASS] " if ok else " [FAIL] ") + name)
|
||||
sys.exit(0 if all(ok for _, ok in results) else 1)
|
||||
PY
|
||||
)"
|
||||
echo "${fcntl_out}"
|
||||
PASS=$((PASS + $(grep -c '\[PASS\]' <<<"${fcntl_out}")))
|
||||
FAIL=$((FAIL + $(grep -c '\[FAIL\]' <<<"${fcntl_out}")))
|
||||
|
||||
# 2. Distributed lock blocks a cross-mount writer, then hands it off ----------
|
||||
# mount 2 holds a file open for writing (holding the DLM lock on its path).
|
||||
# An SMB put of the same file goes through mount 1 and must (a) block while
|
||||
# mount 2 holds it and (b) succeed once mount 2 releases, leaving the SMB
|
||||
# writer's payload on disk. smbclient gets a long client timeout (-t) so we are
|
||||
# testing the lock handoff itself, not smbclient's own ~20s default timeout.
|
||||
echo "==> 2. distributed lock: cross-mount write coordination"
|
||||
dlmfile="dlm_coord.bin"
|
||||
newdata="${WORK}/dlm_new.bin"
|
||||
head -c 4096 /dev/urandom >"${newdata}"
|
||||
|
||||
# Hold the file open for writing on mount 2 via fd 9 -> holds the DLM lock.
|
||||
exec 9>"${MOUNT2_SHARE}/${dlmfile}"
|
||||
printf 'held-by-mount2' >&9
|
||||
|
||||
# Start the SMB write; record its real exit code when it returns. The subshell
|
||||
# must NOT inherit fd 9 (9>&-): otherwise the SMB writer keeps the file open and
|
||||
# waits on a DLM lock held by its own inherited descriptor, deadlocking the
|
||||
# handoff this test is meant to exercise.
|
||||
rm -f "${WORK}/dlm_put.rc"
|
||||
(
|
||||
smbclient "//${SMB_HOST}/${SMB_SHARE}" -p "${SMB_PORT}" \
|
||||
-U "${SMB_USER}%${SMB_PASS}" -m SMB3 -t 120 \
|
||||
-c "put ${newdata} ${dlmfile}" >/dev/null 2>&1
|
||||
echo "$?" >"${WORK}/dlm_put.rc"
|
||||
) 9>&- &
|
||||
smb_bg=$!
|
||||
|
||||
sleep 4
|
||||
if [[ ! -f "${WORK}/dlm_put.rc" ]]; then
|
||||
pass "SMB write blocks while another mount holds the file open"
|
||||
else
|
||||
fail "SMB write returned early instead of blocking (rc=$(cat "${WORK}/dlm_put.rc"))"
|
||||
fi
|
||||
|
||||
# Release mount 2's DLM lock; the blocked SMB write must now complete.
|
||||
exec 9>&-
|
||||
|
||||
# Wait (bounded) for the SMB put to finish so a stuck handoff fails the test
|
||||
# instead of hanging the suite.
|
||||
put_rc="timeout"
|
||||
for _ in $(seq 1 20); do
|
||||
if [[ -f "${WORK}/dlm_put.rc" ]]; then
|
||||
put_rc="$(cat "${WORK}/dlm_put.rc")"
|
||||
break
|
||||
fi
|
||||
sleep 1
|
||||
done
|
||||
kill "${smb_bg}" 2>/dev/null
|
||||
wait "${smb_bg}" 2>/dev/null
|
||||
|
||||
if [[ "${put_rc}" == "0" ]]; then
|
||||
pass "blocked SMB write succeeds after the other mount releases"
|
||||
else
|
||||
fail "blocked SMB write succeeds after the other mount releases (rc=${put_rc})"
|
||||
fi
|
||||
|
||||
# A correct handoff leaves the SMB writer's payload on disk: mount 1 acquired
|
||||
# the lock and wrote after mount 2 released.
|
||||
got="${WORK}/dlm_got.bin"
|
||||
if smb "get ${dlmfile} ${got}" >/dev/null 2>&1 && [[ "$(md5 "${got}")" == "$(md5 "${newdata}")" ]]; then
|
||||
pass "post-release content is the SMB writer's payload (correct handoff)"
|
||||
else
|
||||
fail "post-release content is the SMB writer's payload (correct handoff)"
|
||||
fi
|
||||
|
||||
# 3. Distributed lock integrity: concurrent writers, same file ---------------
|
||||
# An SMB writer (mount 1) and a direct writer (mount 2) race on one file. DLM
|
||||
# serializes them, so the result must be exactly one of the two payloads.
|
||||
echo "==> 3. distributed lock: concurrent writers produce intact data"
|
||||
racefile="dlm_race.bin"
|
||||
payloadA="${WORK}/dlm_raceA.bin"
|
||||
head -c 1048576 /dev/urandom >"${payloadA}"
|
||||
payloadB="direct-write-from-mount2-payload"
|
||||
(smb "put ${payloadA} ${racefile}" >/dev/null 2>&1) &
|
||||
(printf '%s' "${payloadB}" >"${MOUNT2_SHARE}/${racefile}") &
|
||||
wait
|
||||
racegot="${WORK}/dlm_race_got.bin"
|
||||
if smb "get ${racefile} ${racegot}" >/dev/null 2>&1 &&
|
||||
{ [[ "$(md5 "${racegot}")" == "$(md5 "${payloadA}")" ]] || [[ "$(cat "${racegot}")" == "${payloadB}" ]]; }; then
|
||||
pass "concurrent same-file writers leave one intact payload"
|
||||
else
|
||||
fail "concurrent same-file writers leave one intact payload"
|
||||
fi
|
||||
|
||||
# 4. Concurrent writers to distinct files ------------------------------------
|
||||
echo "==> 4. concurrent writers to distinct files"
|
||||
n=6
|
||||
declare -a srcs=()
|
||||
for i in $(seq 1 "${n}"); do
|
||||
s="${WORK}/cc_${i}.bin"
|
||||
head -c 1048576 /dev/urandom >"${s}"
|
||||
srcs+=("${s}")
|
||||
(smb "put ${s} concurrent_${i}.bin" >/dev/null 2>&1) &
|
||||
done
|
||||
wait
|
||||
all_ok=true
|
||||
for i in $(seq 1 "${n}"); do
|
||||
g="${WORK}/cc_got_${i}.bin"
|
||||
if ! smb "get concurrent_${i}.bin ${g}" >/dev/null 2>&1 ||
|
||||
[[ "$(md5 "${srcs[$((i - 1))]}")" != "$(md5 "${g}")" ]]; then
|
||||
all_ok=false
|
||||
fi
|
||||
done
|
||||
if ${all_ok}; then
|
||||
pass "${n} concurrent distinct-file writes all intact"
|
||||
else
|
||||
fail "${n} concurrent distinct-file writes all intact"
|
||||
fi
|
||||
|
||||
echo
|
||||
echo "==> Summary: ${PASS} passed, ${FAIL} failed"
|
||||
[[ "${FAIL}" -eq 0 ]]
|
||||
Executable
+203
@@ -0,0 +1,203 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# Run the SMB (Samba) integration test against a SeaweedFS FUSE mount.
|
||||
#
|
||||
# Pipeline:
|
||||
# 1. start a self-contained "weed mini" (master + volume + filer in one)
|
||||
# 2. mount the filesystem with "weed mount"
|
||||
# 3. export a subdirectory of the mount over SMB with smbd
|
||||
# 4. drive the share with smbclient (test/samba/smb_tests.sh)
|
||||
#
|
||||
# Everything runs as the current user on unprivileged ports, so no sudo is
|
||||
# required. State lives under a temp work dir and is removed on exit.
|
||||
#
|
||||
# Requirements: weed in $PATH, fusermount3, and Samba's smbd / smbclient /
|
||||
# smbpasswd (Debian/Ubuntu: apt-get install samba smbclient).
|
||||
#
|
||||
# Usage:
|
||||
# test/samba/run.sh
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
WEED_BIN="${WEED_BIN:-weed}"
|
||||
WORK_DIR="${WORK_DIR:-$(mktemp -d /tmp/seaweedfs-samba.XXXXXX)}"
|
||||
MOUNT_DIR="${MOUNT_DIR:-${WORK_DIR}/mnt}"
|
||||
MOUNT2_DIR="${MOUNT2_DIR:-${WORK_DIR}/mnt2}"
|
||||
DATA_DIR="${DATA_DIR:-${WORK_DIR}/data}"
|
||||
LOG_DIR="${LOG_DIR:-${WORK_DIR}/logs}"
|
||||
STATE_DIR="${WORK_DIR}/samba"
|
||||
SHARE_DIR="${MOUNT_DIR}/share"
|
||||
SHARE_DIR2="${MOUNT2_DIR}/share"
|
||||
|
||||
FILER_PORT="${FILER_PORT:-28888}"
|
||||
FILER_ADDR="127.0.0.1:${FILER_PORT}"
|
||||
SMB_PORT="${SMB_PORT:-4450}"
|
||||
SMB_SHARE="seaweedfs"
|
||||
SMB_USER="${SMB_USER:-$(id -un)}"
|
||||
SMB_PASS="${SMB_PASS:-seaweedfs}"
|
||||
|
||||
SMBD_BIN="$(command -v smbd || echo /usr/sbin/smbd)"
|
||||
SMBPASSWD_BIN="$(command -v smbpasswd || echo /usr/bin/smbpasswd)"
|
||||
|
||||
CI_LOG_DIR="/tmp/seaweedfs-samba-logs"
|
||||
|
||||
mini_pid=""
|
||||
mount_pid=""
|
||||
mount2_pid=""
|
||||
smbd_pid=""
|
||||
|
||||
unmount_dir() {
|
||||
local dir="$1"
|
||||
if mountpoint -q "${dir}" 2>/dev/null; then
|
||||
fusermount3 -u "${dir}" 2>/dev/null ||
|
||||
fusermount -u "${dir}" 2>/dev/null || true
|
||||
fi
|
||||
}
|
||||
|
||||
cleanup() {
|
||||
set +e
|
||||
if [[ -n "${smbd_pid}" ]] && kill -0 "${smbd_pid}" 2>/dev/null; then
|
||||
kill -TERM "${smbd_pid}" 2>/dev/null || true
|
||||
wait "${smbd_pid}" 2>/dev/null || true
|
||||
fi
|
||||
for p in "${mount_pid}" "${mount2_pid}"; do
|
||||
if [[ -n "${p}" ]] && kill -0 "${p}" 2>/dev/null; then
|
||||
kill -TERM "${p}" 2>/dev/null || true
|
||||
wait "${p}" 2>/dev/null || true
|
||||
fi
|
||||
done
|
||||
unmount_dir "${MOUNT_DIR}"
|
||||
unmount_dir "${MOUNT2_DIR}"
|
||||
if [[ -n "${mini_pid}" ]] && kill -0 "${mini_pid}" 2>/dev/null; then
|
||||
kill -TERM "${mini_pid}" 2>/dev/null || true
|
||||
wait "${mini_pid}" 2>/dev/null || true
|
||||
fi
|
||||
# Copy logs to a fixed path for CI artifact upload.
|
||||
mkdir -p "${CI_LOG_DIR}"
|
||||
cp "${LOG_DIR}"/*.log "${LOG_DIR}"/*.out "${STATE_DIR}/smbd.log" "${CI_LOG_DIR}/" 2>/dev/null || true
|
||||
}
|
||||
trap cleanup EXIT INT TERM
|
||||
|
||||
mkdir -p "${MOUNT_DIR}" "${MOUNT2_DIR}" "${DATA_DIR}" "${LOG_DIR}" \
|
||||
"${STATE_DIR}/private" "${STATE_DIR}/state" "${STATE_DIR}/cache" \
|
||||
"${STATE_DIR}/lock" "${STATE_DIR}/pid" "${STATE_DIR}/ncalrpc"
|
||||
|
||||
# --- 1. weed mini -----------------------------------------------------------
|
||||
echo "==> Starting weed mini on ${FILER_ADDR}"
|
||||
"${WEED_BIN}" mini \
|
||||
-dir="${DATA_DIR}" \
|
||||
-ip=127.0.0.1 \
|
||||
-filer.port="${FILER_PORT}" \
|
||||
-s3=false \
|
||||
-webdav=false \
|
||||
-admin.ui=false \
|
||||
>"${LOG_DIR}/mini.log" 2>&1 &
|
||||
mini_pid=$!
|
||||
|
||||
for i in $(seq 1 60); do
|
||||
if (echo >"/dev/tcp/127.0.0.1/${FILER_PORT}") 2>/dev/null; then
|
||||
break
|
||||
fi
|
||||
if ! kill -0 "${mini_pid}" 2>/dev/null; then
|
||||
echo "weed mini exited early; log tail:" >&2
|
||||
tail -n 100 "${LOG_DIR}/mini.log" >&2 || true
|
||||
exit 1
|
||||
fi
|
||||
sleep 0.5
|
||||
done
|
||||
if ! (echo >"/dev/tcp/127.0.0.1/${FILER_PORT}") 2>/dev/null; then
|
||||
echo "weed mini filer did not become reachable within 30s; log tail:" >&2
|
||||
tail -n 100 "${LOG_DIR}/mini.log" >&2 || true
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# --- 2. weed mount (two mounts, both with -dlm) -----------------------------
|
||||
# mount_with_dlm <mountpoint> <logfile> <pid-var-name>
|
||||
mount_with_dlm() {
|
||||
local dir="$1" log="$2" pidvar="$3" pid
|
||||
echo "==> Mounting SeaweedFS at ${dir} with -dlm"
|
||||
"${WEED_BIN}" mount \
|
||||
-filer="${FILER_ADDR}" \
|
||||
-dir="${dir}" \
|
||||
-filer.path=/ \
|
||||
-dirAutoCreate \
|
||||
-dlm \
|
||||
>"${log}" 2>&1 &
|
||||
pid=$!
|
||||
printf -v "${pidvar}" '%s' "${pid}"
|
||||
for _ in $(seq 1 60); do
|
||||
if mountpoint -q "${dir}"; then
|
||||
return 0
|
||||
fi
|
||||
if ! kill -0 "${pid}" 2>/dev/null; then
|
||||
echo "weed mount (${dir}) exited early; log tail:" >&2
|
||||
tail -n 100 "${log}" >&2 || true
|
||||
exit 1
|
||||
fi
|
||||
sleep 0.5
|
||||
done
|
||||
echo "FUSE mount ${dir} did not come up within 30s" >&2
|
||||
tail -n 100 "${log}" >&2 || true
|
||||
exit 1
|
||||
}
|
||||
|
||||
mount_with_dlm "${MOUNT_DIR}" "${LOG_DIR}/mount.log" mount_pid
|
||||
mount_with_dlm "${MOUNT2_DIR}" "${LOG_DIR}/mount2.log" mount2_pid
|
||||
|
||||
mkdir -p "${SHARE_DIR}"
|
||||
|
||||
# --- 3. smbd ----------------------------------------------------------------
|
||||
echo "==> Generating smb.conf and starting smbd on port ${SMB_PORT}"
|
||||
SMB_CONF="${STATE_DIR}/smb.conf"
|
||||
sed -e "s#@SHARE_PATH@#${SHARE_DIR}#g" \
|
||||
-e "s#@STATE_DIR@#${STATE_DIR}#g" \
|
||||
-e "s#@SMB_PORT@#${SMB_PORT}#g" \
|
||||
-e "s#@FORCE_USER@#${SMB_USER}#g" \
|
||||
"${SCRIPT_DIR}/smb.conf.template" >"${SMB_CONF}"
|
||||
|
||||
printf '%s\n%s\n' "${SMB_PASS}" "${SMB_PASS}" |
|
||||
"${SMBPASSWD_BIN}" -c "${SMB_CONF}" -a -s "${SMB_USER}"
|
||||
|
||||
"${SMBD_BIN}" -F --no-process-group -s "${SMB_CONF}" >"${LOG_DIR}/smbd.out" 2>&1 &
|
||||
smbd_pid=$!
|
||||
|
||||
for i in $(seq 1 60); do
|
||||
if (echo >"/dev/tcp/127.0.0.1/${SMB_PORT}") 2>/dev/null; then
|
||||
break
|
||||
fi
|
||||
if ! kill -0 "${smbd_pid}" 2>/dev/null; then
|
||||
echo "smbd exited early; log tail:" >&2
|
||||
tail -n 100 "${LOG_DIR}/smbd.out" "${STATE_DIR}/smbd.log" 2>/dev/null >&2 || true
|
||||
exit 1
|
||||
fi
|
||||
sleep 0.5
|
||||
done
|
||||
if ! (echo >"/dev/tcp/127.0.0.1/${SMB_PORT}") 2>/dev/null; then
|
||||
echo "smbd did not become reachable within 30s; log tail:" >&2
|
||||
tail -n 100 "${LOG_DIR}/smbd.out" "${STATE_DIR}/smbd.log" 2>/dev/null >&2 || true
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# --- 4. run the test batteries ---------------------------------------------
|
||||
rc=0
|
||||
|
||||
echo "==> Running SMB functional test battery"
|
||||
SMB_HOST=127.0.0.1 \
|
||||
SMB_SHARE="${SMB_SHARE}" \
|
||||
SMB_PORT="${SMB_PORT}" \
|
||||
SMB_USER="${SMB_USER}" \
|
||||
SMB_PASS="${SMB_PASS}" \
|
||||
SHARE_FS_PATH="${SHARE_DIR}" \
|
||||
"${SCRIPT_DIR}/smb_tests.sh" || rc=1
|
||||
|
||||
echo "==> Running SMB locking / concurrency test battery"
|
||||
SMB_HOST=127.0.0.1 \
|
||||
SMB_SHARE="${SMB_SHARE}" \
|
||||
SMB_PORT="${SMB_PORT}" \
|
||||
SMB_USER="${SMB_USER}" \
|
||||
SMB_PASS="${SMB_PASS}" \
|
||||
MOUNT_SHARE="${SHARE_DIR}" \
|
||||
MOUNT2_SHARE="${SHARE_DIR2}" \
|
||||
"${SCRIPT_DIR}/lock_tests.sh" || rc=1
|
||||
|
||||
exit "${rc}"
|
||||
Executable
+24
@@ -0,0 +1,24 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# Runs the SMB test batteries inside the samba container against the local smbd,
|
||||
# which serves /mnt/seaweedfs/share over a SeaweedFS FUSE mount. A second FUSE
|
||||
# mount (/mnt/seaweedfs2) backs the distributed-locking tests.
|
||||
# Invoked via: docker compose exec samba /run_inside_container.sh
|
||||
set -euo pipefail
|
||||
|
||||
export SMB_HOST=127.0.0.1
|
||||
export SMB_SHARE=seaweedfs
|
||||
export SMB_PORT="${SMB_PORT:-445}"
|
||||
export SMB_USER="${SMB_USER:-smbtest}"
|
||||
export SMB_PASS="${SMB_PASS:-smbtest}"
|
||||
export SHARE_FS_PATH="${SHARE_FS_PATH:-/mnt/seaweedfs/share}"
|
||||
export MOUNT_SHARE="${MOUNT_SHARE:-/mnt/seaweedfs/share}"
|
||||
export MOUNT2_SHARE="${MOUNT2_SHARE:-/mnt/seaweedfs2/share}"
|
||||
|
||||
rc=0
|
||||
echo "############ SMB functional tests ############"
|
||||
/smb_tests.sh || rc=1
|
||||
echo
|
||||
echo "############ SMB locking / concurrency tests ############"
|
||||
/lock_tests.sh || rc=1
|
||||
exit "${rc}"
|
||||
@@ -0,0 +1,52 @@
|
||||
[global]
|
||||
server role = standalone server
|
||||
workgroup = WORKGROUP
|
||||
server string = SeaweedFS FUSE Samba test
|
||||
security = user
|
||||
server min protocol = SMB2
|
||||
smb ports = @SMB_PORT@
|
||||
bind interfaces only = yes
|
||||
interfaces = lo 127.0.0.1
|
||||
|
||||
# Self-contained state so smbd can run rootless and leaves nothing behind
|
||||
# outside the test work directory.
|
||||
private dir = @STATE_DIR@/private
|
||||
state directory = @STATE_DIR@/state
|
||||
cache directory = @STATE_DIR@/cache
|
||||
lock directory = @STATE_DIR@/lock
|
||||
pid directory = @STATE_DIR@/pid
|
||||
ncalrpc dir = @STATE_DIR@/ncalrpc
|
||||
log file = @STATE_DIR@/smbd.log
|
||||
log level = 1
|
||||
usershare max shares = 0
|
||||
|
||||
# No printing subsystem in a file-server test.
|
||||
load printers = no
|
||||
printing = bsd
|
||||
printcap name = /dev/null
|
||||
disable spoolss = yes
|
||||
|
||||
# The SeaweedFS FUSE mount does not implement the full xattr / DOS-attribute
|
||||
# surface Samba uses by default. Disabling these avoids spurious
|
||||
# NT_STATUS_NOT_SUPPORTED / EOPNOTSUPP errors unrelated to data integrity.
|
||||
ea support = no
|
||||
store dos attributes = no
|
||||
map archive = no
|
||||
map hidden = no
|
||||
map system = no
|
||||
map readonly = no
|
||||
|
||||
# A network-filesystem backend should not advertise local oplocks/leases.
|
||||
oplocks = no
|
||||
level2 oplocks = no
|
||||
kernel oplocks = no
|
||||
posix locking = no
|
||||
|
||||
[seaweedfs]
|
||||
path = @SHARE_PATH@
|
||||
comment = SeaweedFS share backed by a FUSE mount
|
||||
browseable = yes
|
||||
read only = no
|
||||
create mask = 0644
|
||||
directory mask = 0755
|
||||
force user = @FORCE_USER@
|
||||
Executable
+172
@@ -0,0 +1,172 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# SMB protocol test battery against a Samba share backed by a SeaweedFS FUSE
|
||||
# mount. Driven both by the local runner (test/samba/run.sh) and by the Docker
|
||||
# harness (run_inside_container.sh).
|
||||
#
|
||||
# Required env:
|
||||
# SMB_USER samba username
|
||||
# SMB_PASS samba password
|
||||
# Optional env:
|
||||
# SMB_HOST samba host (default 127.0.0.1)
|
||||
# SMB_SHARE share name (default seaweedfs)
|
||||
# SMB_PORT smbd port (default 445)
|
||||
# SHARE_FS_PATH directory on the FUSE mount that backs the share. When set,
|
||||
# the suite also checks cross-protocol consistency: data written
|
||||
# over SMB is visible on the FUSE mount, and vice versa.
|
||||
set -uo pipefail
|
||||
|
||||
SMB_HOST="${SMB_HOST:-127.0.0.1}"
|
||||
SMB_SHARE="${SMB_SHARE:-seaweedfs}"
|
||||
SMB_PORT="${SMB_PORT:-445}"
|
||||
SMB_USER="${SMB_USER:?SMB_USER is required}"
|
||||
SMB_PASS="${SMB_PASS:?SMB_PASS is required}"
|
||||
SHARE_FS_PATH="${SHARE_FS_PATH:-}"
|
||||
|
||||
WORK="$(mktemp -d /tmp/samba-smbtest.XXXXXX)"
|
||||
trap 'rm -rf "${WORK}"' EXIT
|
||||
|
||||
PASS=0
|
||||
FAIL=0
|
||||
pass() { printf ' [PASS] %s\n' "$1"; PASS=$((PASS + 1)); }
|
||||
fail() { printf ' [FAIL] %s\n' "$1"; FAIL=$((FAIL + 1)); }
|
||||
|
||||
# Run one or more smbclient commands (separated by ';') against the share.
|
||||
smb() {
|
||||
smbclient "//${SMB_HOST}/${SMB_SHARE}" -p "${SMB_PORT}" \
|
||||
-U "${SMB_USER}%${SMB_PASS}" -m SMB3 -c "$1"
|
||||
}
|
||||
|
||||
md5() { md5sum "$1" | awk '{print $1}'; }
|
||||
|
||||
echo "==> Target //${SMB_HOST}/${SMB_SHARE} (port ${SMB_PORT}) as ${SMB_USER}"
|
||||
[[ -n "${SHARE_FS_PATH}" ]] && echo "==> Cross-protocol checks against ${SHARE_FS_PATH}"
|
||||
|
||||
# 1. Connectivity ------------------------------------------------------------
|
||||
echo "==> 1. connectivity"
|
||||
if smb "ls" >/dev/null 2>&1; then
|
||||
pass "connect and list share root"
|
||||
else
|
||||
fail "connect and list share root"
|
||||
fi
|
||||
|
||||
# 2. Upload / download round-trip -------------------------------------------
|
||||
echo "==> 2. upload / download round-trip"
|
||||
src="${WORK}/src.bin"
|
||||
head -c 1048576 /dev/urandom >"${src}" # 1 MiB
|
||||
if smb "put ${src} roundtrip.bin" >/dev/null 2>&1; then
|
||||
pass "put 1 MiB file"
|
||||
else
|
||||
fail "put 1 MiB file"
|
||||
fi
|
||||
got="${WORK}/got.bin"
|
||||
if smb "get roundtrip.bin ${got}" >/dev/null 2>&1 && [[ "$(md5 "${src}")" == "$(md5 "${got}")" ]]; then
|
||||
pass "get returns identical content"
|
||||
else
|
||||
fail "get returns identical content"
|
||||
fi
|
||||
if [[ -n "${SHARE_FS_PATH}" ]]; then
|
||||
if [[ -f "${SHARE_FS_PATH}/roundtrip.bin" ]] && [[ "$(md5 "${SHARE_FS_PATH}/roundtrip.bin")" == "$(md5 "${src}")" ]]; then
|
||||
pass "SMB-written file visible on FUSE mount with identical content"
|
||||
else
|
||||
fail "SMB-written file visible on FUSE mount with identical content"
|
||||
fi
|
||||
fi
|
||||
|
||||
# 3. Directory operations ----------------------------------------------------
|
||||
echo "==> 3. directory operations"
|
||||
if smb "mkdir docs; cd docs; put ${src} nested.bin; ls" >/dev/null 2>&1; then
|
||||
pass "mkdir + put into subdirectory"
|
||||
else
|
||||
fail "mkdir + put into subdirectory"
|
||||
fi
|
||||
if [[ -z "${SHARE_FS_PATH}" || -f "${SHARE_FS_PATH}/docs/nested.bin" ]]; then
|
||||
pass "nested file present"
|
||||
else
|
||||
fail "nested file present"
|
||||
fi
|
||||
|
||||
# 4. Rename ------------------------------------------------------------------
|
||||
echo "==> 4. rename"
|
||||
if smb "rename roundtrip.bin renamed.bin" >/dev/null 2>&1; then
|
||||
pass "rename file"
|
||||
else
|
||||
fail "rename file"
|
||||
fi
|
||||
renback="${WORK}/renamed.bin"
|
||||
if smb "get renamed.bin ${renback}" >/dev/null 2>&1 && [[ "$(md5 "${renback}")" == "$(md5 "${src}")" ]]; then
|
||||
pass "renamed file readable with original content"
|
||||
else
|
||||
fail "renamed file readable with original content"
|
||||
fi
|
||||
if [[ -n "${SHARE_FS_PATH}" ]]; then
|
||||
if [[ -f "${SHARE_FS_PATH}/renamed.bin" && ! -e "${SHARE_FS_PATH}/roundtrip.bin" ]]; then
|
||||
pass "rename reflected on FUSE mount"
|
||||
else
|
||||
fail "rename reflected on FUSE mount"
|
||||
fi
|
||||
fi
|
||||
|
||||
# 5. Large file (exercises SeaweedFS chunking) -------------------------------
|
||||
echo "==> 5. large file (SeaweedFS chunking)"
|
||||
big="${WORK}/big.bin"
|
||||
head -c 67108864 /dev/urandom >"${big}" # 64 MiB
|
||||
bigback="${WORK}/big.back"
|
||||
if smb "put ${big} big.bin" >/dev/null 2>&1 &&
|
||||
smb "get big.bin ${bigback}" >/dev/null 2>&1 &&
|
||||
[[ "$(md5 "${big}")" == "$(md5 "${bigback}")" ]]; then
|
||||
pass "64 MiB put/get round-trip"
|
||||
else
|
||||
fail "64 MiB put/get round-trip"
|
||||
fi
|
||||
|
||||
# 6. Recursive upload --------------------------------------------------------
|
||||
echo "==> 6. recursive upload"
|
||||
tree="${WORK}/tree"
|
||||
mkdir -p "${tree}/a/b"
|
||||
echo one >"${tree}/f1.txt"
|
||||
echo two >"${tree}/a/f2.txt"
|
||||
echo three >"${tree}/a/b/f3.txt"
|
||||
if (cd "${WORK}" && smb "recurse ON; prompt OFF; mput tree" >/dev/null 2>&1) &&
|
||||
{ [[ -z "${SHARE_FS_PATH}" ]] || [[ -f "${SHARE_FS_PATH}/tree/a/b/f3.txt" ]]; }; then
|
||||
pass "recursive mput"
|
||||
else
|
||||
fail "recursive mput"
|
||||
fi
|
||||
|
||||
# 7. Cross-protocol read (FUSE writes, SMB reads) ----------------------------
|
||||
if [[ -n "${SHARE_FS_PATH}" ]]; then
|
||||
echo "==> 7. cross-protocol read (FUSE write -> SMB read)"
|
||||
echo "written-via-fuse" >"${SHARE_FS_PATH}/from_fuse.txt"
|
||||
cpb="${WORK}/from_fuse.back"
|
||||
if smb "get from_fuse.txt ${cpb}" >/dev/null 2>&1 && grep -q written-via-fuse "${cpb}"; then
|
||||
pass "FUSE-written file readable over SMB"
|
||||
else
|
||||
fail "FUSE-written file readable over SMB"
|
||||
fi
|
||||
fi
|
||||
|
||||
# 8. Delete ------------------------------------------------------------------
|
||||
echo "==> 8. delete"
|
||||
smb "del renamed.bin" >/dev/null 2>&1
|
||||
smb "del big.bin" >/dev/null 2>&1
|
||||
smb "deltree docs" >/dev/null 2>&1
|
||||
smb "deltree tree" >/dev/null 2>&1
|
||||
if [[ -n "${SHARE_FS_PATH}" ]]; then
|
||||
if [[ ! -e "${SHARE_FS_PATH}/renamed.bin" && ! -e "${SHARE_FS_PATH}/big.bin" &&
|
||||
! -e "${SHARE_FS_PATH}/docs" && ! -e "${SHARE_FS_PATH}/tree" ]]; then
|
||||
pass "delete files and directory trees"
|
||||
else
|
||||
fail "delete files and directory trees"
|
||||
fi
|
||||
else
|
||||
if ! smb "get renamed.bin /dev/null" >/dev/null 2>&1; then
|
||||
pass "deleted file no longer retrievable"
|
||||
else
|
||||
fail "deleted file no longer retrievable"
|
||||
fi
|
||||
fi
|
||||
|
||||
echo
|
||||
echo "==> Summary: ${PASS} passed, ${FAIL} failed"
|
||||
[[ "${FAIL}" -eq 0 ]]
|
||||
+37
-7
@@ -11,6 +11,29 @@ import (
|
||||
// GrpcPortOffset is the offset weed mini uses to derive gRPC ports from HTTP ports.
|
||||
const GrpcPortOffset = 10000
|
||||
|
||||
// miniDefaultPorts are the weed mini flag defaults (see weed/command/mini.go).
|
||||
// A test only overrides services it uses; unspecified services still bind
|
||||
// these defaults, so allocation must avoid handing them out (or any value
|
||||
// whose gRPC offset would collide with them).
|
||||
var miniDefaultPorts = []int{
|
||||
9333, // master.port
|
||||
8888, // filer.port
|
||||
9340, // volume.port
|
||||
8333, // s3.port
|
||||
8181, // s3.port.iceberg
|
||||
7333, // webdav.port
|
||||
23646, // admin.port
|
||||
}
|
||||
|
||||
func reservedMiniPorts() map[int]bool {
|
||||
r := make(map[int]bool, len(miniDefaultPorts)*2)
|
||||
for _, p := range miniDefaultPorts {
|
||||
r[p] = true
|
||||
r[p+GrpcPortOffset] = true
|
||||
}
|
||||
return r
|
||||
}
|
||||
|
||||
// AllocatePorts allocates count unique free ports atomically.
|
||||
// All listeners are held open until every port is obtained, preventing
|
||||
// the OS from recycling a port between successive allocations.
|
||||
@@ -51,12 +74,19 @@ func MustAllocatePorts(t *testing.T, count int) []int {
|
||||
// from recycling ports between allocations. Use this when ports will be
|
||||
// passed to weed mini without explicit gRPC port flags, so mini will
|
||||
// derive gRPC ports as HTTP + 10000.
|
||||
//
|
||||
// Listeners are bound on all interfaces (":port") rather than 127.0.0.1
|
||||
// to match weed mini's availability check (isPortAvailable). A port can
|
||||
// be free on loopback but held by another process on a different
|
||||
// interface; reserving only on loopback lets mini's check fail and
|
||||
// trigger gRPC port shifting, which then causes weed shell to dial the
|
||||
// wrong port and hang.
|
||||
func AllocateMiniPorts(count int) ([]int, error) {
|
||||
const (
|
||||
minPort = 10000
|
||||
maxPort = 55000
|
||||
)
|
||||
reserved := make(map[int]bool)
|
||||
reserved := reservedMiniPorts()
|
||||
ports := make([]int, 0, count)
|
||||
var listeners []net.Listener
|
||||
defer func() {
|
||||
@@ -75,12 +105,12 @@ func AllocateMiniPorts(count int) ([]int, error) {
|
||||
continue
|
||||
}
|
||||
|
||||
l1, err := net.Listen("tcp", fmt.Sprintf("127.0.0.1:%d", port))
|
||||
l1, err := net.Listen("tcp", fmt.Sprintf(":%d", port))
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
|
||||
l2, err := net.Listen("tcp", fmt.Sprintf("127.0.0.1:%d", grpcPort))
|
||||
l2, err := net.Listen("tcp", fmt.Sprintf(":%d", grpcPort))
|
||||
if err != nil {
|
||||
l1.Close()
|
||||
continue
|
||||
@@ -128,7 +158,7 @@ func AllocatePortSet(miniCount, regularCount int) (mini []int, regular []int, er
|
||||
minPort = 10000
|
||||
maxPort = 55000
|
||||
)
|
||||
reserved := make(map[int]bool)
|
||||
reserved := reservedMiniPorts()
|
||||
mini = make([]int, 0, miniCount)
|
||||
var listeners []net.Listener
|
||||
defer func() {
|
||||
@@ -145,11 +175,11 @@ func AllocatePortSet(miniCount, regularCount int) (mini []int, regular []int, er
|
||||
if reserved[port] || reserved[grpcPort] {
|
||||
continue
|
||||
}
|
||||
l1, lErr := net.Listen("tcp", fmt.Sprintf("127.0.0.1:%d", port))
|
||||
l1, lErr := net.Listen("tcp", fmt.Sprintf(":%d", port))
|
||||
if lErr != nil {
|
||||
continue
|
||||
}
|
||||
l2, lErr := net.Listen("tcp", fmt.Sprintf("127.0.0.1:%d", grpcPort))
|
||||
l2, lErr := net.Listen("tcp", fmt.Sprintf(":%d", grpcPort))
|
||||
if lErr != nil {
|
||||
l1.Close()
|
||||
continue
|
||||
@@ -168,7 +198,7 @@ func AllocatePortSet(miniCount, regularCount int) (mini []int, regular []int, er
|
||||
|
||||
regular = make([]int, 0, regularCount)
|
||||
for i := 0; i < regularCount; i++ {
|
||||
l, lErr := net.Listen("tcp", "127.0.0.1:0")
|
||||
l, lErr := net.Listen("tcp", ":0")
|
||||
if lErr != nil {
|
||||
return nil, nil, lErr
|
||||
}
|
||||
|
||||
@@ -2,6 +2,29 @@ package testutil
|
||||
|
||||
import "testing"
|
||||
|
||||
// AllocateMiniPorts must never hand out a port that weed mini will reserve
|
||||
// for one of its default services (or that default's gRPC offset). A real
|
||||
// failure: Filer was given 33646 (Admin default 23646 + GrpcPortOffset),
|
||||
// which mini then refused as "reserved for gRPC calculation".
|
||||
func TestAllocateMiniPortsAvoidsMiniDefaults(t *testing.T) {
|
||||
reserved := reservedMiniPorts()
|
||||
for iter := 0; iter < 200; iter++ {
|
||||
ports, err := AllocateMiniPorts(4)
|
||||
if err != nil {
|
||||
t.Fatalf("iter %d: AllocateMiniPorts: %v", iter, err)
|
||||
}
|
||||
for _, p := range ports {
|
||||
if reserved[p] {
|
||||
t.Fatalf("iter %d: allocated port %d is a mini default (or gRPC offset)", iter, p)
|
||||
}
|
||||
if reserved[p+GrpcPortOffset] {
|
||||
t.Fatalf("iter %d: allocated port %d has gRPC offset %d colliding with a mini default",
|
||||
iter, p, p+GrpcPortOffset)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestAllocatePortSetNoGrpcCollision(t *testing.T) {
|
||||
// Run a few iterations to catch the OS-recycles-just-closed-port race
|
||||
// that previously hit regular ports when the mini gRPC offset was freed
|
||||
|
||||
@@ -136,8 +136,6 @@ func TestEcLifecycleAcrossMultipleDisks(t *testing.T) {
|
||||
conn2, grpcClient2 := framework.DialVolumeServer(t, clusterHarness.VolumeGRPCAddress())
|
||||
defer conn2.Close()
|
||||
|
||||
// VolumeEcShardsInfo only sees one disk's EcVolume; filesystem layout is
|
||||
// the ground truth for the whole-store shard count.
|
||||
postReconcileLayout := scanShardLayout(t, dataDirs, collection, volumeID)
|
||||
if got, want := totalShardsInLayout(postReconcileLayout), erasure_coding.TotalShardsCount; got != want {
|
||||
t.Fatalf("post-reconcile: total shards on disk mismatch: got %d, want %d (layout=%v)", got, want, postReconcileLayout)
|
||||
@@ -148,11 +146,28 @@ func TestEcLifecycleAcrossMultipleDisks(t *testing.T) {
|
||||
if got, want := len(postReconcileLayout[1]), splitAt; got != want {
|
||||
t.Fatalf("post-reconcile: disk 1 shard count drift: got %d, want %d (layout=%v)", got, want, postReconcileLayout)
|
||||
}
|
||||
if _, err := grpcClient2.VolumeEcShardsInfo(ctx, &volume_server_pb.VolumeEcShardsInfoRequest{
|
||||
// VolumeEcShardsInfo must walk every DiskLocation and report the full
|
||||
// shard set — the verification step in ec_task.go gates source-volume
|
||||
// deletion on this RPC returning a complete shard inventory.
|
||||
infoResp, err := grpcClient2.VolumeEcShardsInfo(ctx, &volume_server_pb.VolumeEcShardsInfoRequest{
|
||||
VolumeId: volumeID,
|
||||
}); err != nil {
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("VolumeEcShardsInfo after redistribute restart: %v", err)
|
||||
}
|
||||
if got, want := len(infoResp.GetEcShardInfos()), erasure_coding.TotalShardsCount; got != want {
|
||||
t.Fatalf("VolumeEcShardsInfo after redistribute restart: got %d shards, want %d (per-disk layout=%v)",
|
||||
got, want, postReconcileLayout)
|
||||
}
|
||||
gotShardIds := make(map[uint32]struct{}, len(infoResp.GetEcShardInfos()))
|
||||
for _, info := range infoResp.GetEcShardInfos() {
|
||||
gotShardIds[info.GetShardId()] = struct{}{}
|
||||
}
|
||||
for shardId := uint32(0); shardId < uint32(erasure_coding.TotalShardsCount); shardId++ {
|
||||
if _, ok := gotShardIds[shardId]; !ok {
|
||||
t.Fatalf("VolumeEcShardsInfo missing shard %d (per-disk layout=%v)", shardId, postReconcileLayout)
|
||||
}
|
||||
}
|
||||
for _, n := range needles {
|
||||
verifyHTTPRead(t, httpClient, clusterHarness.VolumeAdminURL(), n.fid, n.payload, "after-cross-disk-reconcile")
|
||||
}
|
||||
|
||||
@@ -0,0 +1,175 @@
|
||||
package volume_server_grpc_test
|
||||
|
||||
import (
|
||||
"context"
|
||||
"net/http"
|
||||
"sort"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/test/volume_server/framework"
|
||||
"github.com/seaweedfs/seaweedfs/test/volume_server/matrix"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/volume_server_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/erasure_coding"
|
||||
"google.golang.org/grpc"
|
||||
"google.golang.org/grpc/credentials/insecure"
|
||||
)
|
||||
|
||||
// TestVolumeEcShardsInfoReturnsAllShardsAcrossDisks drives the full path
|
||||
// behind the ec.encode source-deletion gate. A multi-disk volume server
|
||||
// ends up with EC shards split across disks (each registers its own
|
||||
// EcVolume entry in DiskLocation.ecVolumes), and the volume server's
|
||||
// VolumeEcShardsInfo RPC must walk every DiskLocation rather than
|
||||
// reporting whichever disk Store.FindEcVolume picks first.
|
||||
//
|
||||
// Pre-fix, verifyEcShardsBeforeDelete refused to delete source volumes —
|
||||
// the shard-bitmap union across destinations fell short of dataShards +
|
||||
// parityShards because each destination only reported shards on one of
|
||||
// its disks. With the handler fix, the same VerifyShardsAcrossServers
|
||||
// call returns a complete bitmap and the gate opens.
|
||||
func TestVolumeEcShardsInfoReturnsAllShardsAcrossDisks(t *testing.T) {
|
||||
if testing.Short() {
|
||||
t.Skip("skipping integration test in short mode")
|
||||
}
|
||||
|
||||
const (
|
||||
dataDirCount = 2
|
||||
volumeID = uint32(9558)
|
||||
collection = "ec-multi-disk-verify"
|
||||
)
|
||||
|
||||
clusterHarness := framework.StartSingleVolumeClusterWithDataDirs(t, matrix.P1(), dataDirCount)
|
||||
dataDirs := clusterHarness.VolumeDataDirs()
|
||||
if len(dataDirs) != dataDirCount {
|
||||
t.Fatalf("expected %d data dirs, got %d: %v", dataDirCount, len(dataDirs), dataDirs)
|
||||
}
|
||||
|
||||
conn, grpcClient := framework.DialVolumeServer(t, clusterHarness.VolumeGRPCAddress())
|
||||
defer conn.Close()
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 60*time.Second)
|
||||
defer cancel()
|
||||
|
||||
framework.AllocateVolume(t, grpcClient, volumeID, collection)
|
||||
|
||||
httpClient := framework.NewHTTPClient()
|
||||
needles := []struct {
|
||||
fid string
|
||||
payload []byte
|
||||
}{
|
||||
{framework.NewFileID(volumeID, 9559, 0xC0FFEE01), bytesOfLen(64, 0xB1)},
|
||||
{framework.NewFileID(volumeID, 9560, 0xC0FFEE02), bytesOfLen(8192, 0xB2)},
|
||||
{framework.NewFileID(volumeID, 9561, 0xC0FFEE03), bytesOfLen(131072, 0xB3)},
|
||||
}
|
||||
for _, n := range needles {
|
||||
resp := framework.UploadBytes(t, httpClient, clusterHarness.VolumeAdminURL(), n.fid, n.payload)
|
||||
_ = framework.ReadAllAndClose(t, resp)
|
||||
if resp.StatusCode != http.StatusCreated {
|
||||
t.Fatalf("upload %s expected 201, got %d", n.fid, resp.StatusCode)
|
||||
}
|
||||
}
|
||||
|
||||
if _, err := grpcClient.VolumeEcShardsGenerate(ctx, &volume_server_pb.VolumeEcShardsGenerateRequest{
|
||||
VolumeId: volumeID,
|
||||
Collection: collection,
|
||||
}); err != nil {
|
||||
t.Fatalf("VolumeEcShardsGenerate: %v", err)
|
||||
}
|
||||
|
||||
// Generate places every shard plus the .ecx/.ecj/.vif on the .dat's
|
||||
// disk (disk 0). Mount all 14 there first so the next step's restart
|
||||
// has a steady starting state.
|
||||
allShards := make([]uint32, erasure_coding.TotalShardsCount)
|
||||
for i := range allShards {
|
||||
allShards[i] = uint32(i)
|
||||
}
|
||||
if _, err := grpcClient.VolumeEcShardsMount(ctx, &volume_server_pb.VolumeEcShardsMountRequest{
|
||||
VolumeId: volumeID,
|
||||
Collection: collection,
|
||||
ShardIds: allShards,
|
||||
}); err != nil {
|
||||
t.Fatalf("VolumeEcShardsMount all shards: %v", err)
|
||||
}
|
||||
|
||||
// Drop the .dat so the EC shards are the only data path — mirrors the
|
||||
// real ec.encode flow before verifyEcShardsBeforeDelete fires.
|
||||
if _, err := grpcClient.VolumeDelete(ctx, &volume_server_pb.VolumeDeleteRequest{
|
||||
VolumeId: volumeID,
|
||||
}); err != nil {
|
||||
t.Fatalf("VolumeDelete (drop .dat): %v", err)
|
||||
}
|
||||
|
||||
// Move half the shards onto disk 1, leaving .ecx on disk 0. After
|
||||
// restart, the cross-disk reconcile path attaches each disk's shards
|
||||
// against its own EcVolume entry — the exact in-memory shape the bug
|
||||
// reporter saw on a multi-disk destination.
|
||||
clusterHarness.StopVolumeServer()
|
||||
const splitAt = 7
|
||||
for shard := 0; shard < splitAt; shard++ {
|
||||
movedFile(t, dataDirs[0], dataDirs[1], collection, volumeID, erasure_coding.ToExt(shard))
|
||||
}
|
||||
if fileExistsIn(dataDirs[1], collection, volumeID, ".ecx") {
|
||||
t.Fatalf("setup: .ecx must stay on disk 0 to exercise the multi-disk path")
|
||||
}
|
||||
|
||||
clusterHarness.RestartVolumeServer()
|
||||
conn2, grpcClient2 := framework.DialVolumeServer(t, clusterHarness.VolumeGRPCAddress())
|
||||
defer conn2.Close()
|
||||
|
||||
postReconcileLayout := scanShardLayout(t, dataDirs, collection, volumeID)
|
||||
if got, want := totalShardsInLayout(postReconcileLayout), erasure_coding.TotalShardsCount; got != want {
|
||||
t.Fatalf("post-reconcile: total shards on disk mismatch: got %d, want %d (layout=%v)", got, want, postReconcileLayout)
|
||||
}
|
||||
if len(postReconcileLayout[0]) == 0 || len(postReconcileLayout[1]) == 0 {
|
||||
t.Fatalf("post-reconcile: expected shards on BOTH disks, got per-disk layout %v", postReconcileLayout)
|
||||
}
|
||||
|
||||
// Direct RPC assertion: VolumeEcShardsInfo must report every shard
|
||||
// the server holds, not just the ones registered against the first
|
||||
// matching DiskLocation.
|
||||
infoResp, err := grpcClient2.VolumeEcShardsInfo(ctx, &volume_server_pb.VolumeEcShardsInfoRequest{
|
||||
VolumeId: volumeID,
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("VolumeEcShardsInfo: %v", err)
|
||||
}
|
||||
gotShardIds := make([]int, 0, len(infoResp.GetEcShardInfos()))
|
||||
for _, info := range infoResp.GetEcShardInfos() {
|
||||
if info.GetVolumeId() != volumeID {
|
||||
t.Errorf("EcShardInfo VolumeId=%d, want %d", info.GetVolumeId(), volumeID)
|
||||
}
|
||||
gotShardIds = append(gotShardIds, int(info.GetShardId()))
|
||||
}
|
||||
sort.Ints(gotShardIds)
|
||||
wantShardIds := make([]int, erasure_coding.TotalShardsCount)
|
||||
for i := range wantShardIds {
|
||||
wantShardIds[i] = i
|
||||
}
|
||||
if len(gotShardIds) != len(wantShardIds) {
|
||||
t.Fatalf("VolumeEcShardsInfo returned %d shards (ids=%v), want %d (ids=%v) — per-disk layout=%v",
|
||||
len(gotShardIds), gotShardIds, len(wantShardIds), wantShardIds, postReconcileLayout)
|
||||
}
|
||||
for i, sid := range wantShardIds {
|
||||
if gotShardIds[i] != sid {
|
||||
t.Fatalf("VolumeEcShardsInfo shard ids=%v, want %v (per-disk layout=%v)",
|
||||
gotShardIds, wantShardIds, postReconcileLayout)
|
||||
}
|
||||
}
|
||||
|
||||
// End-to-end assertion via the same helper the worker uses to gate
|
||||
// source-volume deletion (weed/worker/tasks/erasure_coding/ec_task.go
|
||||
// verifyEcShardsBeforeDelete). The union across destinations is what
|
||||
// RequireFullShardSet measures; with one destination that holds every
|
||||
// shard, the union must cover dataShards + parityShards.
|
||||
dialOption := grpc.WithTransportCredentials(insecure.NewCredentials())
|
||||
servers := []string{clusterHarness.VolumeServerAddress()}
|
||||
union, perServer := erasure_coding.VerifyShardsAcrossServers(ctx, volumeID, servers, dialOption)
|
||||
if err := erasure_coding.RequireFullShardSet(volumeID, union, erasure_coding.TotalShardsCount); err != nil {
|
||||
t.Fatalf("verifyEcShardsBeforeDelete-equivalent gate failed: %v\nper-server inventory: %s\nper-disk layout: %v",
|
||||
err, erasure_coding.SummarizeShardInventory(perServer), postReconcileLayout)
|
||||
}
|
||||
if got, want := union.Count(), erasure_coding.TotalShardsCount; got != want {
|
||||
t.Fatalf("VerifyShardsAcrossServers union covered %d/%d shards (per-server=%s, layout=%v)",
|
||||
got, want, erasure_coding.SummarizeShardInventory(perServer), postReconcileLayout)
|
||||
}
|
||||
}
|
||||
@@ -75,7 +75,7 @@ func TestStatsEndpoints(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestStatusPrettyJsonAndJsonp(t *testing.T) {
|
||||
func TestStatusPrettyJsonAndCallbackIgnored(t *testing.T) {
|
||||
if testing.Short() {
|
||||
t.Skip("skipping integration test in short mode")
|
||||
}
|
||||
@@ -93,29 +93,29 @@ func TestStatusPrettyJsonAndJsonp(t *testing.T) {
|
||||
if len(lines) < 3 {
|
||||
t.Fatalf("/status?pretty=y expected multi-line indented JSON, got %d lines: %s", len(lines), string(prettyBody))
|
||||
}
|
||||
// Verify the body is valid JSON
|
||||
var prettyPayload map[string]interface{}
|
||||
if err := json.Unmarshal(prettyBody, &prettyPayload); err != nil {
|
||||
t.Fatalf("/status?pretty=y is not valid JSON: %v", err)
|
||||
}
|
||||
|
||||
// ?callback=myFunc — expect JSONP wrapping
|
||||
jsonpResp := framework.DoRequest(t, client, mustNewRequest(t, http.MethodGet, cluster.VolumeAdminURL()+"/status?callback=myFunc"))
|
||||
jsonpBody := framework.ReadAllAndClose(t, jsonpResp)
|
||||
if jsonpResp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("/status?callback=myFunc expected 200, got %d", jsonpResp.StatusCode)
|
||||
// ?callback=myFunc — must be ignored; response is plain JSON with nosniff.
|
||||
cbResp := framework.DoRequest(t, client, mustNewRequest(t, http.MethodGet, cluster.VolumeAdminURL()+"/status?callback=myFunc"))
|
||||
cbBody := framework.ReadAllAndClose(t, cbResp)
|
||||
if cbResp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("/status?callback=myFunc expected 200, got %d", cbResp.StatusCode)
|
||||
}
|
||||
bodyStr := string(jsonpBody)
|
||||
if !strings.HasPrefix(bodyStr, "myFunc(") {
|
||||
t.Fatalf("/status?callback=myFunc expected body to start with 'myFunc(', got prefix: %q", bodyStr[:min(len(bodyStr), 30)])
|
||||
if ct := cbResp.Header.Get("Content-Type"); !strings.Contains(ct, "application/json") {
|
||||
t.Fatalf("/status?callback=myFunc expected Content-Type application/json, got %q", ct)
|
||||
}
|
||||
trimmed := strings.TrimRight(bodyStr, "\n; ")
|
||||
if !strings.HasSuffix(trimmed, ")") {
|
||||
t.Fatalf("/status?callback=myFunc expected body to end with ')', got suffix: %q", trimmed[max(0, len(trimmed)-10):])
|
||||
if nosniff := cbResp.Header.Get("X-Content-Type-Options"); nosniff != "nosniff" {
|
||||
t.Fatalf("/status?callback=myFunc expected X-Content-Type-Options nosniff, got %q", nosniff)
|
||||
}
|
||||
// Content-Type should be application/javascript for JSONP
|
||||
if ct := jsonpResp.Header.Get("Content-Type"); !strings.Contains(ct, "javascript") {
|
||||
t.Fatalf("/status?callback=myFunc expected Content-Type containing 'javascript', got %q", ct)
|
||||
if strings.Contains(string(cbBody), "myFunc(") {
|
||||
t.Fatalf("/status?callback=myFunc must not wrap response in callback; body: %q", string(cbBody))
|
||||
}
|
||||
var cbPayload map[string]interface{}
|
||||
if err := json.Unmarshal(cbBody, &cbPayload); err != nil {
|
||||
t.Fatalf("/status?callback=myFunc is not valid JSON: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -2,6 +2,8 @@ package volume_server_http_test
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"testing"
|
||||
"time"
|
||||
@@ -44,6 +46,17 @@ func TestReplicatedUploadSucceedsImmediatelyAfterAllocate(t *testing.T) {
|
||||
fid := framework.NewFileID(volumeID, 881001, 0x0B0C0D0E)
|
||||
payload := []byte("replicated-upload-after-allocate")
|
||||
|
||||
// The master only learns about replica locations through volume-server
|
||||
// heartbeats, which lag behind the direct AllocateVolume gRPC calls above.
|
||||
// In production a client obtains its fid from the master assign flow, which
|
||||
// guarantees the master already knows every replica; this test crafts the
|
||||
// fid by hand, so the replicated write would otherwise look up the master
|
||||
// before the second replica is registered and fail with a 500. Wait until
|
||||
// the master reports both replicas before uploading.
|
||||
if !waitForMasterReplicaCount(t, client, clusterHarness.MasterURL(), volumeID, 2, 10*time.Second) {
|
||||
t.Fatalf("master did not report 2 replica locations for volume %d within deadline", volumeID)
|
||||
}
|
||||
|
||||
uploadResp := framework.UploadBytes(t, client, clusterHarness.VolumeAdminURL(0), fid, payload)
|
||||
_ = framework.ReadAllAndClose(t, uploadResp)
|
||||
if uploadResp.StatusCode != http.StatusCreated {
|
||||
@@ -61,3 +74,29 @@ func TestReplicatedUploadSucceedsImmediatelyAfterAllocate(t *testing.T) {
|
||||
t.Fatalf("replica body mismatch: got %q want %q", string(replicaBody), string(payload))
|
||||
}
|
||||
}
|
||||
|
||||
// waitForMasterReplicaCount polls the master volume lookup until it reports at
|
||||
// least want locations for volumeID, or the timeout elapses.
|
||||
func waitForMasterReplicaCount(t testing.TB, client *http.Client, masterURL string, volumeID uint32, want int, timeout time.Duration) bool {
|
||||
t.Helper()
|
||||
|
||||
lookupURL := fmt.Sprintf("%s/dir/lookup?volumeId=%d", masterURL, volumeID)
|
||||
deadline := time.Now().Add(timeout)
|
||||
for time.Now().Before(deadline) {
|
||||
resp := framework.DoRequest(t, client, mustNewRequest(t, http.MethodGet, lookupURL))
|
||||
body := framework.ReadAllAndClose(t, resp)
|
||||
if resp.StatusCode == http.StatusOK {
|
||||
var result struct {
|
||||
Locations []struct {
|
||||
Url string `json:"url"`
|
||||
} `json:"locations"`
|
||||
}
|
||||
if err := json.Unmarshal(body, &result); err == nil && len(result.Locations) >= want {
|
||||
return true
|
||||
}
|
||||
}
|
||||
time.Sleep(200 * time.Millisecond)
|
||||
}
|
||||
|
||||
return false
|
||||
}
|
||||
|
||||
@@ -23,6 +23,7 @@ import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/plugin_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/schema_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/security"
|
||||
stats_collect "github.com/seaweedfs/seaweedfs/weed/stats"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/erasure_coding"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/super_block"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
@@ -280,6 +281,8 @@ func NewAdminServer(masters string, templateFS http.FileSystem, dataDir string,
|
||||
go server.monitorVacuumWorker(bgCtx)
|
||||
}
|
||||
|
||||
go server.publishMaintenanceMetrics(bgCtx)
|
||||
|
||||
return server
|
||||
}
|
||||
|
||||
@@ -364,6 +367,55 @@ func (s *AdminServer) monitorVacuumWorker(ctx context.Context) {
|
||||
}
|
||||
}
|
||||
|
||||
// publishMaintenanceMetrics periodically snapshots the maintenance queue and
|
||||
// worker fleet into Prometheus gauges. Counters and durations are recorded at
|
||||
// their event sites; these gauges reflect current state at scrape resolution.
|
||||
func (s *AdminServer) publishMaintenanceMetrics(ctx context.Context) {
|
||||
const interval = 15 * time.Second
|
||||
ticker := time.NewTicker(interval)
|
||||
defer ticker.Stop()
|
||||
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case <-ticker.C:
|
||||
s.collectMaintenanceMetrics()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func (s *AdminServer) collectMaintenanceMetrics() {
|
||||
if s.maintenanceManager == nil {
|
||||
return
|
||||
}
|
||||
|
||||
stats := s.maintenanceManager.GetStats()
|
||||
|
||||
stats_collect.AdminMaintenanceTasksByStatus.Reset()
|
||||
for status, count := range stats.TasksByStatus {
|
||||
stats_collect.AdminMaintenanceTasksByStatus.WithLabelValues(string(status)).Set(float64(count))
|
||||
}
|
||||
|
||||
stats_collect.AdminMaintenanceTasksByType.Reset()
|
||||
for taskType, count := range stats.TasksByType {
|
||||
stats_collect.AdminMaintenanceTasksByType.WithLabelValues(string(taskType)).Set(float64(count))
|
||||
}
|
||||
|
||||
// NextScanTime is only meaningful while the scanner runs; GetStats computes
|
||||
// it unconditionally, so clear the gauge when idle to avoid a stale value.
|
||||
if s.maintenanceManager.IsRunning() && !stats.NextScanTime.IsZero() {
|
||||
stats_collect.AdminMaintenanceNextScanTimestampSeconds.Set(float64(stats.NextScanTime.Unix()))
|
||||
} else {
|
||||
stats_collect.AdminMaintenanceNextScanTimestampSeconds.Set(0)
|
||||
}
|
||||
|
||||
workers, usedSlots, maxSlots := s.maintenanceManager.GetWorkerSlotTotals()
|
||||
stats_collect.AdminWorkersConnected.Set(float64(workers))
|
||||
stats_collect.AdminWorkerSlots.WithLabelValues("used").Set(float64(usedSlots))
|
||||
stats_collect.AdminWorkerSlots.WithLabelValues("max").Set(float64(maxSlots))
|
||||
}
|
||||
|
||||
// loadTaskConfigurationsFromPersistence loads saved task configurations from protobuf files
|
||||
func (s *AdminServer) loadTaskConfigurationsFromPersistence() {
|
||||
if s.configPersistence == nil || !s.configPersistence.IsConfigured() {
|
||||
|
||||
@@ -12,6 +12,8 @@ import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/volume_server_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/wdclient"
|
||||
"google.golang.org/grpc"
|
||||
)
|
||||
|
||||
// WithMasterClient executes a function with a master client connection
|
||||
@@ -38,6 +40,21 @@ func (s *AdminServer) WithVolumeServerClient(address pb.ServerAddress, f func(cl
|
||||
})
|
||||
}
|
||||
|
||||
// GetMasterClient returns the admin server's wdclient.MasterClient. It is used
|
||||
// by file browser download paths that stream chunks straight from the volume
|
||||
// servers via filer.PrepareStreamContent so they keep working when the filer
|
||||
// has -disableHttp=true.
|
||||
func (s *AdminServer) GetMasterClient() *wdclient.MasterClient {
|
||||
return s.masterClient
|
||||
}
|
||||
|
||||
// GetGrpcDialOption returns the dial option used for all admin-originated
|
||||
// gRPC connections (TLS or insecure). File browser uploads need this when
|
||||
// they perform the assign + volume HTTP POST + create-entry flow.
|
||||
func (s *AdminServer) GetGrpcDialOption() grpc.DialOption {
|
||||
return s.grpcDialOption
|
||||
}
|
||||
|
||||
// GetFilerAddress returns a filer address, discovering from masters if needed
|
||||
func (s *AdminServer) GetFilerAddress() string {
|
||||
// Discover filers from masters
|
||||
|
||||
@@ -16,7 +16,6 @@ import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/admin/plugin"
|
||||
"github.com/seaweedfs/seaweedfs/weed/cluster"
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/plugin_pb"
|
||||
"google.golang.org/protobuf/encoding/protojson"
|
||||
@@ -735,8 +734,8 @@ func (s *AdminServer) parseOrBuildClusterContext(raw json.RawMessage) (*plugin_p
|
||||
if len(contextMessage.MasterGrpcAddresses) == 0 {
|
||||
contextMessage.MasterGrpcAddresses = append(contextMessage.MasterGrpcAddresses, fallback.MasterGrpcAddresses...)
|
||||
}
|
||||
if len(contextMessage.FilerGrpcAddresses) == 0 {
|
||||
contextMessage.FilerGrpcAddresses = append(contextMessage.FilerGrpcAddresses, fallback.FilerGrpcAddresses...)
|
||||
if len(contextMessage.FilerAddresses) == 0 {
|
||||
contextMessage.FilerAddresses = append(contextMessage.FilerAddresses, fallback.FilerAddresses...)
|
||||
}
|
||||
if len(contextMessage.VolumeGrpcAddresses) == 0 {
|
||||
contextMessage.VolumeGrpcAddresses = append(contextMessage.VolumeGrpcAddresses, fallback.VolumeGrpcAddresses...)
|
||||
@@ -755,7 +754,7 @@ func (s *AdminServer) parseOrBuildClusterContext(raw json.RawMessage) (*plugin_p
|
||||
func (s *AdminServer) buildDefaultPluginClusterContext() *plugin_pb.ClusterContext {
|
||||
clusterContext := &plugin_pb.ClusterContext{
|
||||
MasterGrpcAddresses: make([]string, 0),
|
||||
FilerGrpcAddresses: make([]string, 0),
|
||||
FilerAddresses: make([]string, 0),
|
||||
VolumeGrpcAddresses: make([]string, 0),
|
||||
S3GrpcAddresses: make([]string, 0),
|
||||
Metadata: map[string]string{
|
||||
@@ -768,20 +767,20 @@ func (s *AdminServer) buildDefaultPluginClusterContext() *plugin_pb.ClusterConte
|
||||
clusterContext.MasterGrpcAddresses = append(clusterContext.MasterGrpcAddresses, masterAddress)
|
||||
}
|
||||
|
||||
// Master returns filers in dual-port form (host:httpPort.grpcPort);
|
||||
// workers dial these directly, so collapse to host:grpcPort first.
|
||||
// Master returns filers in pb.ServerAddress form (host:httpPort.grpcPort).
|
||||
// Forward that verbatim; each worker converts to a gRPC or HTTP address as
|
||||
// it needs (dialing wants gRPC, the admin shell wants the ServerAddress).
|
||||
filerSeen := map[string]struct{}{}
|
||||
for _, filer := range s.GetAllFilers() {
|
||||
filer = strings.TrimSpace(filer)
|
||||
if filer == "" {
|
||||
continue
|
||||
}
|
||||
grpcAddr := pb.ServerAddress(filer).ToGrpcAddress()
|
||||
if _, exists := filerSeen[grpcAddr]; exists {
|
||||
if _, exists := filerSeen[filer]; exists {
|
||||
continue
|
||||
}
|
||||
filerSeen[grpcAddr] = struct{}{}
|
||||
clusterContext.FilerGrpcAddresses = append(clusterContext.FilerGrpcAddresses, grpcAddr)
|
||||
filerSeen[filer] = struct{}{}
|
||||
clusterContext.FilerAddresses = append(clusterContext.FilerAddresses, filer)
|
||||
}
|
||||
|
||||
volumeSeen := map[string]struct{}{}
|
||||
@@ -829,7 +828,7 @@ func (s *AdminServer) buildDefaultPluginClusterContext() *plugin_pb.ClusterConte
|
||||
}
|
||||
|
||||
sort.Strings(clusterContext.MasterGrpcAddresses)
|
||||
sort.Strings(clusterContext.FilerGrpcAddresses)
|
||||
sort.Strings(clusterContext.FilerAddresses)
|
||||
sort.Strings(clusterContext.VolumeGrpcAddresses)
|
||||
sort.Strings(clusterContext.S3GrpcAddresses)
|
||||
|
||||
|
||||
@@ -16,6 +16,7 @@ import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/plugin_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/worker_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/security"
|
||||
stats_collect "github.com/seaweedfs/seaweedfs/weed/stats"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
"google.golang.org/grpc"
|
||||
"google.golang.org/grpc/codes"
|
||||
@@ -233,6 +234,7 @@ func (s *WorkerGrpcServer) WorkerStream(stream worker_pb.WorkerService_WorkerStr
|
||||
}
|
||||
s.connections[workerID] = conn
|
||||
s.connMutex.Unlock()
|
||||
stats_collect.AdminWorkerEventsTotal.WithLabelValues("registered").Inc()
|
||||
|
||||
// Register worker with maintenance manager
|
||||
s.registerWorkerWithManager(conn)
|
||||
@@ -265,11 +267,11 @@ func (s *WorkerGrpcServer) WorkerStream(stream worker_pb.WorkerService_WorkerStr
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
glog.Infof("Worker %s connection closed: %v", workerID, ctx.Err())
|
||||
s.unregisterWorker(conn)
|
||||
s.unregisterWorker(conn, "unregistered")
|
||||
return nil
|
||||
case <-connCtx.Done():
|
||||
glog.Infof("Worker %s connection cancelled", workerID)
|
||||
s.unregisterWorker(conn)
|
||||
s.unregisterWorker(conn, "unregistered")
|
||||
return nil
|
||||
default:
|
||||
}
|
||||
@@ -285,7 +287,7 @@ func (s *WorkerGrpcServer) WorkerStream(stream worker_pb.WorkerService_WorkerStr
|
||||
default:
|
||||
glog.Errorf("Error receiving from worker %s: %v", workerID, err)
|
||||
}
|
||||
s.unregisterWorker(conn)
|
||||
s.unregisterWorker(conn, "unregistered")
|
||||
return err
|
||||
}
|
||||
|
||||
@@ -338,7 +340,7 @@ func (s *WorkerGrpcServer) handleWorkerMessage(conn *WorkerConnection, msg *work
|
||||
|
||||
case *worker_pb.WorkerMessage_Shutdown:
|
||||
glog.Infof("Worker %s shutting down: %s", workerID, m.Shutdown.Reason)
|
||||
s.unregisterWorker(conn)
|
||||
s.unregisterWorker(conn, "unregistered")
|
||||
|
||||
default:
|
||||
glog.Warningf("Unknown message type from worker %s", workerID)
|
||||
@@ -605,7 +607,7 @@ func (s *WorkerGrpcServer) safeCloseOutgoingChannel(conn *WorkerConnection, sour
|
||||
}
|
||||
|
||||
// unregisterWorker removes a worker connection
|
||||
func (s *WorkerGrpcServer) unregisterWorker(conn *WorkerConnection) {
|
||||
func (s *WorkerGrpcServer) unregisterWorker(conn *WorkerConnection, event string) {
|
||||
s.connMutex.Lock()
|
||||
existingConn, exists := s.connections[conn.workerID]
|
||||
if !exists {
|
||||
@@ -624,6 +626,7 @@ func (s *WorkerGrpcServer) unregisterWorker(conn *WorkerConnection) {
|
||||
// Remove from map first to prevent duplicate cleanup attempts
|
||||
delete(s.connections, conn.workerID)
|
||||
s.connMutex.Unlock()
|
||||
stats_collect.AdminWorkerEventsTotal.WithLabelValues(event).Inc()
|
||||
|
||||
// Cancel context to signal goroutines to stop
|
||||
conn.cancel()
|
||||
@@ -665,7 +668,7 @@ func (s *WorkerGrpcServer) cleanupStaleConnections() {
|
||||
|
||||
for _, conn := range toRemove {
|
||||
glog.Warningf("Cleaning up stale worker connection: %s", conn.workerID)
|
||||
s.unregisterWorker(conn)
|
||||
s.unregisterWorker(conn, "stale_removed")
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,291 @@
|
||||
package handlers
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"fmt"
|
||||
"io"
|
||||
"mime"
|
||||
"net/http"
|
||||
"path"
|
||||
"strconv"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/filer"
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/operation"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/security"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
)
|
||||
|
||||
// Admin file browser upload chunk sizing — kept in sync with the values
|
||||
// s3api uses so files end up split into the same fid-sized pieces the rest of
|
||||
// the cluster expects.
|
||||
const (
|
||||
adminUploadChunkSize = 8 * 1024 * 1024
|
||||
adminUploadSmallFileLimit = 256 * 1024
|
||||
)
|
||||
|
||||
// File browser handlers backed by the filer gRPC service. They bypass the
|
||||
// filer's HTTP listener so the UI keeps working when the filer is started
|
||||
// with -disableHttp=true; chunk bytes still flow through the volume server
|
||||
// HTTP endpoints (which run on their own ports).
|
||||
|
||||
// fetchFileContentGrpc reads file content via the filer gRPC service, looking
|
||||
// the entry up and then streaming the chunks straight from the volume servers.
|
||||
// When maxBytes > 0 the stream is truncated to that many bytes — used by the
|
||||
// "is this text?" sniff so unknown-MIME files don't get fully downloaded.
|
||||
func (h *FileBrowserHandlers) fetchFileContentGrpc(ctx context.Context, filePath string, maxBytes int) (string, error) {
|
||||
cleanFilePath, err := h.validateAndCleanFilePath(filePath)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
|
||||
entry, err := h.lookupEntry(ctx, cleanFilePath)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
if entry.IsDirectory {
|
||||
return "", fmt.Errorf("path is a directory")
|
||||
}
|
||||
|
||||
size := int64(filer.FileSize(entry))
|
||||
streamSize := size
|
||||
if maxBytes > 0 && streamSize > int64(maxBytes) {
|
||||
streamSize = int64(maxBytes)
|
||||
}
|
||||
|
||||
var buf bytes.Buffer
|
||||
if err := h.streamEntryContent(ctx, entry, streamSize, &buf); err != nil {
|
||||
return "", err
|
||||
}
|
||||
return buf.String(), nil
|
||||
}
|
||||
|
||||
// downloadFileGrpc streams a file via gRPC + volume server HTTP. The
|
||||
// response writer receives the canonical attachment headers and the raw
|
||||
// bytes; this replaces the HTTP-to-filer proxy that used to run in
|
||||
// DownloadFile.
|
||||
func (h *FileBrowserHandlers) downloadFileGrpc(ctx context.Context, filePath string, w http.ResponseWriter) error {
|
||||
cleanFilePath, err := h.validateAndCleanFilePath(filePath)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
entry, err := h.lookupEntry(ctx, cleanFilePath)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if entry.IsDirectory {
|
||||
return fmt.Errorf("path is a directory")
|
||||
}
|
||||
|
||||
size := int64(filer.FileSize(entry))
|
||||
|
||||
fileName := path.Base(cleanFilePath)
|
||||
w.Header().Set("Content-Disposition", mime.FormatMediaType("attachment", map[string]string{"filename": fileName}))
|
||||
|
||||
contentType := ""
|
||||
if entry.Attributes != nil {
|
||||
contentType = entry.Attributes.Mime
|
||||
}
|
||||
if contentType == "" {
|
||||
contentType = "application/octet-stream"
|
||||
}
|
||||
w.Header().Set("Content-Type", contentType)
|
||||
w.Header().Set("Content-Length", strconv.FormatInt(size, 10))
|
||||
w.WriteHeader(http.StatusOK)
|
||||
|
||||
return h.streamEntryContent(ctx, entry, size, w)
|
||||
}
|
||||
|
||||
// uploadFileGrpc streams the upload to volume servers in 8 MiB chunks via the
|
||||
// shared chunked-upload helper, then registers the assembled entry through the
|
||||
// filer gRPC service. Bytes never enter the admin process's heap as a whole —
|
||||
// each chunk is sized to adminUploadChunkSize. Small files (< 256 KiB) are
|
||||
// stored inline on the entry, matching the S3 server's behaviour.
|
||||
func (h *FileBrowserHandlers) uploadFileGrpc(ctx context.Context, filePath string, fileName string, mimeType string, reader io.Reader) error {
|
||||
cleanFilePath, err := h.validateAndCleanFilePath(filePath)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
dir := path.Dir(cleanFilePath)
|
||||
if dir == "." {
|
||||
dir = "/"
|
||||
}
|
||||
entryName := path.Base(cleanFilePath)
|
||||
if mimeType == "" {
|
||||
mimeType = "application/octet-stream"
|
||||
}
|
||||
|
||||
assignFunc := func(ctx context.Context, count int, expectedDataSize uint64) (*operation.VolumeAssignRequest, *operation.AssignResult, error) {
|
||||
var assignResp *filer_pb.AssignVolumeResponse
|
||||
err := h.adminServer.WithFilerClient(func(client filer_pb.SeaweedFilerClient) error {
|
||||
resp, assignErr := client.AssignVolume(ctx, &filer_pb.AssignVolumeRequest{
|
||||
Count: int32(count),
|
||||
Path: cleanFilePath,
|
||||
ExpectedDataSize: expectedDataSize,
|
||||
})
|
||||
if assignErr != nil {
|
||||
return assignErr
|
||||
}
|
||||
if resp.Error != "" {
|
||||
return fmt.Errorf("%s", resp.Error)
|
||||
}
|
||||
assignResp = resp
|
||||
return nil
|
||||
})
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
if assignResp.Location == nil || assignResp.FileId == "" {
|
||||
return nil, nil, fmt.Errorf("assign volume returned empty location")
|
||||
}
|
||||
return nil, &operation.AssignResult{
|
||||
Fid: assignResp.FileId,
|
||||
Url: assignResp.Location.Url,
|
||||
PublicUrl: assignResp.Location.PublicUrl,
|
||||
Count: uint64(count),
|
||||
Auth: security.EncodedJwt(assignResp.Auth),
|
||||
}, nil
|
||||
}
|
||||
|
||||
chunkResult, err := operation.UploadReaderInChunks(ctx, reader, &operation.ChunkedUploadOption{
|
||||
ChunkSize: adminUploadChunkSize,
|
||||
SmallFileLimit: adminUploadSmallFileLimit,
|
||||
SaveSmallInline: true,
|
||||
MimeType: mimeType,
|
||||
AssignFunc: assignFunc,
|
||||
})
|
||||
if err != nil {
|
||||
// Partial chunks come back even on error so we can clean them up rather
|
||||
// than leaving orphaned data on volume servers.
|
||||
if chunkResult != nil && len(chunkResult.FileChunks) > 0 {
|
||||
h.deleteOrphanedChunks(chunkResult.FileChunks)
|
||||
}
|
||||
return fmt.Errorf("upload: %w", err)
|
||||
}
|
||||
|
||||
now := time.Now()
|
||||
entry := &filer_pb.Entry{
|
||||
Name: entryName,
|
||||
Attributes: &filer_pb.FuseAttributes{
|
||||
FileSize: uint64(chunkResult.TotalSize),
|
||||
Mtime: now.Unix(),
|
||||
Crtime: now.Unix(),
|
||||
FileMode: 0644,
|
||||
Mime: mimeType,
|
||||
},
|
||||
}
|
||||
if len(chunkResult.SmallContent) > 0 {
|
||||
entry.Content = chunkResult.SmallContent
|
||||
} else {
|
||||
entry.Chunks = chunkResult.FileChunks
|
||||
}
|
||||
|
||||
err = h.adminServer.WithFilerClient(func(client filer_pb.SeaweedFilerClient) error {
|
||||
_, createErr := client.CreateEntry(ctx, &filer_pb.CreateEntryRequest{
|
||||
Directory: dir,
|
||||
Entry: entry,
|
||||
})
|
||||
return createErr
|
||||
})
|
||||
if err != nil {
|
||||
if len(chunkResult.FileChunks) > 0 {
|
||||
h.deleteOrphanedChunks(chunkResult.FileChunks)
|
||||
}
|
||||
return fmt.Errorf("create entry: %w", err)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// deleteOrphanedChunks best-effort removes the chunk fids when an upload
|
||||
// fails partway through. Errors are logged; we can't surface them past the
|
||||
// caller's primary failure.
|
||||
func (h *FileBrowserHandlers) deleteOrphanedChunks(chunks []*filer_pb.FileChunk) {
|
||||
fileIds := make([]string, 0, len(chunks))
|
||||
for _, c := range chunks {
|
||||
if fid := c.GetFileIdString(); fid != "" {
|
||||
fileIds = append(fileIds, fid)
|
||||
}
|
||||
}
|
||||
if len(fileIds) == 0 {
|
||||
return
|
||||
}
|
||||
master := h.adminServer.GetMasterClient()
|
||||
results := operation.DeleteFileIds(master.GetMaster, false, h.adminServer.GetGrpcDialOption(), fileIds)
|
||||
for _, r := range results {
|
||||
if r.Error != "" {
|
||||
glog.Warningf("admin file browser: orphan chunk %s cleanup: %s", r.FileId, r.Error)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func (h *FileBrowserHandlers) lookupEntry(ctx context.Context, cleanFilePath string) (*filer_pb.Entry, error) {
|
||||
dir := path.Dir(cleanFilePath)
|
||||
if dir == "." {
|
||||
dir = "/"
|
||||
}
|
||||
name := path.Base(cleanFilePath)
|
||||
var entry *filer_pb.Entry
|
||||
err := h.adminServer.WithFilerClient(func(client filer_pb.SeaweedFilerClient) error {
|
||||
resp, lookupErr := client.LookupDirectoryEntry(ctx, &filer_pb.LookupDirectoryEntryRequest{
|
||||
Directory: dir,
|
||||
Name: name,
|
||||
})
|
||||
if lookupErr != nil {
|
||||
return lookupErr
|
||||
}
|
||||
if resp.Entry == nil {
|
||||
return fmt.Errorf("not found")
|
||||
}
|
||||
entry = resp.Entry
|
||||
return nil
|
||||
})
|
||||
return entry, err
|
||||
}
|
||||
|
||||
func (h *FileBrowserHandlers) streamEntryContent(ctx context.Context, entry *filer_pb.Entry, size int64, w io.Writer) error {
|
||||
if size == 0 {
|
||||
// Inline content (small files stored directly on the entry) skip the
|
||||
// chunk pipeline entirely.
|
||||
if len(entry.Content) > 0 {
|
||||
_, err := w.Write(entry.Content)
|
||||
return err
|
||||
}
|
||||
return nil
|
||||
}
|
||||
if len(entry.Content) > 0 && len(entry.GetChunks()) == 0 {
|
||||
_, err := w.Write(entry.Content)
|
||||
return err
|
||||
}
|
||||
|
||||
streamFn, err := filer.PrepareStreamContentWithThrottler(
|
||||
ctx,
|
||||
h.adminServer.GetMasterClient(),
|
||||
volumeServerReadJwt,
|
||||
entry.GetChunks(),
|
||||
0,
|
||||
size,
|
||||
0,
|
||||
)
|
||||
if err != nil {
|
||||
return fmt.Errorf("prepare stream: %w", err)
|
||||
}
|
||||
return streamFn(w)
|
||||
}
|
||||
|
||||
// volumeServerReadJwt mints a per-fileId Bearer token for reads against a
|
||||
// volume server when jwt.signing.read.key is configured. The volume servers
|
||||
// are unaware of jwt.filer_signing.read.key — that one only gates the filer
|
||||
// HTTP surface, which this code path doesn't touch.
|
||||
func volumeServerReadJwt(fileId string) string {
|
||||
v := util.GetViper()
|
||||
signingKey := security.SigningKey(v.GetString("jwt.signing.read.key"))
|
||||
if len(signingKey) == 0 {
|
||||
return ""
|
||||
}
|
||||
expiresAfterSec := v.GetInt("jwt.signing.read.expires_after_seconds")
|
||||
return string(security.GenJwtForVolumeServer(signingKey, expiresAfterSec, fileId))
|
||||
}
|
||||
@@ -1,15 +1,10 @@
|
||||
package handlers
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"fmt"
|
||||
"io"
|
||||
"mime"
|
||||
"mime/multipart"
|
||||
"net"
|
||||
"net/http"
|
||||
"net/url"
|
||||
"os"
|
||||
"path"
|
||||
"path/filepath"
|
||||
@@ -21,9 +16,7 @@ import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/admin/view/app"
|
||||
"github.com/seaweedfs/seaweedfs/weed/admin/view/layout"
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/security"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util/http/client"
|
||||
)
|
||||
@@ -49,15 +42,6 @@ func NewFileBrowserHandlers(adminServer *dash.AdminServer) *FileBrowserHandlers
|
||||
}
|
||||
}
|
||||
|
||||
// newClientWithTimeout creates a temporary http.Client with the specified timeout,
|
||||
// reusing the TLS transport from the shared httpClient.
|
||||
func (h *FileBrowserHandlers) newClientWithTimeout(timeout time.Duration) http.Client {
|
||||
return http.Client{
|
||||
Transport: h.httpClient.Client.Transport,
|
||||
Timeout: timeout,
|
||||
}
|
||||
}
|
||||
|
||||
// ShowFileBrowser renders the file browser page
|
||||
func (h *FileBrowserHandlers) ShowFileBrowser(w http.ResponseWriter, r *http.Request) {
|
||||
// Get path from query parameter, default to root
|
||||
@@ -309,7 +293,7 @@ func (h *FileBrowserHandlers) UploadFile(w http.ResponseWriter, r *http.Request)
|
||||
}
|
||||
|
||||
// Upload file to filer
|
||||
err = h.uploadFileToFiler(fullPath, fileHeader)
|
||||
err = h.uploadFileToFiler(r.Context(), fullPath, fileHeader)
|
||||
|
||||
if err != nil {
|
||||
failedUploads = append(failedUploads, fmt.Sprintf("%s: %v", fileName, err))
|
||||
@@ -346,139 +330,21 @@ func (h *FileBrowserHandlers) UploadFile(w http.ResponseWriter, r *http.Request)
|
||||
}
|
||||
}
|
||||
|
||||
// uploadFileToFiler uploads a file directly to the filer using multipart form data
|
||||
func (h *FileBrowserHandlers) uploadFileToFiler(filePath string, fileHeader *multipart.FileHeader) error {
|
||||
// Get filer address from admin server
|
||||
filerAddress := h.adminServer.GetFilerAddress()
|
||||
if filerAddress == "" {
|
||||
return fmt.Errorf("filer address not configured")
|
||||
}
|
||||
|
||||
// Validate and sanitize the filer address
|
||||
if err := h.validateFilerAddress(filerAddress); err != nil {
|
||||
return fmt.Errorf("invalid filer address: %w", err)
|
||||
}
|
||||
filerHttpAddress := pb.ServerAddress(filerAddress).ToHttpAddress()
|
||||
|
||||
// Validate and sanitize the file path
|
||||
cleanFilePath, err := h.validateAndCleanFilePath(filePath)
|
||||
if err != nil {
|
||||
return fmt.Errorf("invalid file path: %w", err)
|
||||
}
|
||||
|
||||
// Open the file
|
||||
// uploadFileToFiler uploads a file to the cluster via filer gRPC + volume
|
||||
// HTTP. This works whether or not the filer is running with -disableHttp=true,
|
||||
// since the bytes never traverse the filer's HTTP listener. The multipart
|
||||
// file is streamed through the chunked uploader, so the admin process never
|
||||
// buffers the entire payload in memory. The caller passes the request
|
||||
// context so a client disconnect cancels the in-flight chunk uploads instead
|
||||
// of letting them run to completion against the volume servers.
|
||||
func (h *FileBrowserHandlers) uploadFileToFiler(ctx context.Context, filePath string, fileHeader *multipart.FileHeader) error {
|
||||
file, err := fileHeader.Open()
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to open file: %w", err)
|
||||
}
|
||||
defer file.Close()
|
||||
|
||||
// Create multipart form data
|
||||
var body bytes.Buffer
|
||||
writer := multipart.NewWriter(&body)
|
||||
|
||||
// Create form file field with normalized base filename
|
||||
// Use path.Base (not filepath.Base) since cleanFilePath uses URL path semantics
|
||||
baseFileName := path.Base(cleanFilePath)
|
||||
part, err := writer.CreateFormFile("file", baseFileName)
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to create form file: %w", err)
|
||||
}
|
||||
|
||||
// Copy file content to form
|
||||
_, err = io.Copy(part, file)
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to copy file content: %w", err)
|
||||
}
|
||||
|
||||
// Close the writer to finalize the form
|
||||
err = writer.Close()
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to close multipart writer: %w", err)
|
||||
}
|
||||
|
||||
// Create the upload URL - the httpClient will normalize to the correct scheme (http/https)
|
||||
// based on the https.client configuration in security.toml
|
||||
uploadURL := filerFileURL(filerHttpAddress, cleanFilePath)
|
||||
|
||||
// Normalize the URL scheme based on TLS configuration
|
||||
uploadURL, err = h.httpClient.NormalizeHttpScheme(uploadURL)
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to normalize URL scheme: %w", err)
|
||||
}
|
||||
|
||||
// Create HTTP request
|
||||
req, err := http.NewRequest("POST", uploadURL, &body)
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to create request: %w", err)
|
||||
}
|
||||
|
||||
// Set content type with boundary
|
||||
req.Header.Set("Content-Type", writer.FormDataContentType())
|
||||
|
||||
// Add JWT Token to Authorization Header
|
||||
h.setupFilerJwtAuth(req, "jwt.filer_signing.key", "jwt.filer_signing.expires_after_seconds", "filer upload")
|
||||
|
||||
// Send request using TLS-aware HTTP client with 60s timeout for large file uploads
|
||||
// lgtm[go/ssrf]
|
||||
// Safe: filerAddress validated by validateFilerAddress() to match configured filer
|
||||
// Safe: cleanFilePath validated and cleaned by validateAndCleanFilePath() to prevent path traversal
|
||||
client := h.newClientWithTimeout(60 * time.Second)
|
||||
resp, err := client.Do(req)
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to upload file: %w", err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
|
||||
// Check response
|
||||
if resp.StatusCode != http.StatusOK && resp.StatusCode != http.StatusCreated {
|
||||
responseBody, _ := io.ReadAll(resp.Body)
|
||||
return fmt.Errorf("upload failed with status %d: %s", resp.StatusCode, string(responseBody))
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// validateFilerAddress validates that the filer address is safe to use
|
||||
func (h *FileBrowserHandlers) validateFilerAddress(address string) error {
|
||||
if address == "" {
|
||||
return fmt.Errorf("filer address cannot be empty")
|
||||
}
|
||||
|
||||
// CRITICAL: Only allow the configured filer address to prevent SSRF
|
||||
configuredFiler := h.adminServer.GetFilerAddress()
|
||||
normalizedAddress := pb.ServerAddress(address).ToHttpAddress()
|
||||
normalizedConfigured := pb.ServerAddress(configuredFiler).ToHttpAddress()
|
||||
if normalizedAddress != normalizedConfigured {
|
||||
return fmt.Errorf("address does not match configured filer: got %s, expected %s", address, configuredFiler)
|
||||
}
|
||||
|
||||
// Parse the normalized HTTP address to validate it's a proper host:port format.
|
||||
host, port, err := net.SplitHostPort(normalizedAddress)
|
||||
if err != nil {
|
||||
return fmt.Errorf("invalid address format: %w", err)
|
||||
}
|
||||
|
||||
// Validate host is not empty
|
||||
if host == "" {
|
||||
return fmt.Errorf("host cannot be empty")
|
||||
}
|
||||
|
||||
// Validate port is numeric and in valid range
|
||||
if port == "" {
|
||||
return fmt.Errorf("port cannot be empty")
|
||||
}
|
||||
|
||||
portNum, err := strconv.Atoi(port)
|
||||
if err != nil {
|
||||
return fmt.Errorf("invalid port number: %w", err)
|
||||
}
|
||||
|
||||
if portNum < 1 || portNum > 65535 {
|
||||
return fmt.Errorf("port number must be between 1 and 65535")
|
||||
}
|
||||
|
||||
return nil
|
||||
return h.uploadFileGrpc(ctx, filePath, fileHeader.Filename, fileHeader.Header.Get("Content-Type"), file)
|
||||
}
|
||||
|
||||
// validateAndCleanFilePath validates and cleans the file path to prevent path traversal
|
||||
@@ -507,161 +373,56 @@ func (h *FileBrowserHandlers) validateAndCleanFilePath(filePath string) (string,
|
||||
return cleanPath, nil
|
||||
}
|
||||
|
||||
// filerFileURL joins the filer HTTP address with a validated file path, URL-escaping
|
||||
// the path so that control characters and other bytes that are legal in S3 object keys
|
||||
// cannot inject into the HTTP request target.
|
||||
func filerFileURL(filerHttpAddress, cleanFilePath string) string {
|
||||
return filerHttpAddress + (&url.URL{Path: cleanFilePath}).EscapedPath()
|
||||
}
|
||||
|
||||
// fetchFileContent fetches file content from the filer and returns the content or an error.
|
||||
// fetchFileContent fetches file content via the filer gRPC service. It is
|
||||
// used for the "view as text" path, so the maxBytes cap matches the 1 MB
|
||||
// limit the caller already applies before invoking us.
|
||||
func (h *FileBrowserHandlers) fetchFileContent(filePath string, timeout time.Duration) (string, error) {
|
||||
filerAddress := h.adminServer.GetFilerAddress()
|
||||
if filerAddress == "" {
|
||||
return "", fmt.Errorf("filer address not configured")
|
||||
}
|
||||
|
||||
if err := h.validateFilerAddress(filerAddress); err != nil {
|
||||
return "", fmt.Errorf("invalid filer address configuration: %w", err)
|
||||
}
|
||||
filerHttpAddress := pb.ServerAddress(filerAddress).ToHttpAddress()
|
||||
|
||||
cleanFilePath, err := h.validateAndCleanFilePath(filePath)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
|
||||
// Create the file URL with proper scheme based on TLS configuration
|
||||
fileURL := filerFileURL(filerHttpAddress, cleanFilePath)
|
||||
fileURL, err = h.httpClient.NormalizeHttpScheme(fileURL)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("failed to construct file URL: %w", err)
|
||||
}
|
||||
|
||||
// lgtm[go/ssrf]
|
||||
// Safe: filerAddress validated by validateFilerAddress() to match configured filer
|
||||
// Safe: cleanFilePath validated and cleaned by validateAndCleanFilePath() to prevent path traversal
|
||||
client := h.newClientWithTimeout(timeout)
|
||||
req, err := http.NewRequest("GET", fileURL, nil)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("failed to create request: %w", err)
|
||||
}
|
||||
h.addFilerJwtAuthHeader(req)
|
||||
resp, err := client.Do(req)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("failed to fetch file from filer: %w", err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
body, err := io.ReadAll(resp.Body)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("filer returned status %d but failed to read response body: %w", resp.StatusCode, err)
|
||||
}
|
||||
return "", fmt.Errorf("filer returned status %d: %s", resp.StatusCode, string(body))
|
||||
}
|
||||
|
||||
contentBytes, err := io.ReadAll(resp.Body)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("failed to read file content: %w", err)
|
||||
}
|
||||
|
||||
return string(contentBytes), nil
|
||||
ctx, cancel := context.WithTimeout(context.Background(), timeout)
|
||||
defer cancel()
|
||||
return h.fetchFileContentGrpc(ctx, filePath, 0)
|
||||
}
|
||||
|
||||
// DownloadFile handles file download requests by proxying through the Admin UI server
|
||||
// This ensures mTLS works correctly since the Admin UI server has the client certificates
|
||||
// DownloadFile streams a file straight from the volume servers via the filer
|
||||
// gRPC service, so the admin file browser keeps working even when the filer
|
||||
// is started with -disableHttp=true.
|
||||
func (h *FileBrowserHandlers) DownloadFile(w http.ResponseWriter, r *http.Request) {
|
||||
filePath := r.URL.Query().Get("path")
|
||||
if filePath == "" {
|
||||
writeJSONError(w, http.StatusBadRequest, "File path is required")
|
||||
return
|
||||
}
|
||||
|
||||
// Get filer address
|
||||
filerAddress := h.adminServer.GetFilerAddress()
|
||||
if filerAddress == "" {
|
||||
writeJSONError(w, http.StatusInternalServerError, "Filer address not configured")
|
||||
return
|
||||
}
|
||||
|
||||
// Validate filer address to prevent SSRF
|
||||
if err := h.validateFilerAddress(filerAddress); err != nil {
|
||||
writeJSONError(w, http.StatusInternalServerError, "Invalid filer address configuration")
|
||||
return
|
||||
}
|
||||
filerHttpAddress := pb.ServerAddress(filerAddress).ToHttpAddress()
|
||||
|
||||
// Validate and sanitize the file path
|
||||
cleanFilePath, err := h.validateAndCleanFilePath(filePath)
|
||||
if err != nil {
|
||||
writeJSONError(w, http.StatusBadRequest, "Invalid file path: "+err.Error())
|
||||
return
|
||||
}
|
||||
|
||||
// Create the download URL with proper scheme based on TLS configuration
|
||||
downloadURL := filerFileURL(filerHttpAddress, cleanFilePath)
|
||||
downloadURL, err = h.httpClient.NormalizeHttpScheme(downloadURL)
|
||||
if err != nil {
|
||||
writeJSONError(w, http.StatusInternalServerError, "Failed to construct download URL: "+err.Error())
|
||||
return
|
||||
}
|
||||
|
||||
// Proxy the download through the Admin UI server to support mTLS
|
||||
// lgtm[go/ssrf]
|
||||
// Safe: filerAddress validated by validateFilerAddress() to match configured filer
|
||||
// Safe: cleanFilePath validated and cleaned by validateAndCleanFilePath() to prevent path traversal
|
||||
// Use request context so download is cancelled when client disconnects
|
||||
req, err := http.NewRequestWithContext(r.Context(), "GET", downloadURL, nil)
|
||||
if err != nil {
|
||||
writeJSONError(w, http.StatusInternalServerError, "Failed to create request: "+err.Error())
|
||||
return
|
||||
}
|
||||
client := h.newClientWithTimeout(5 * time.Minute) // Longer timeout for large file downloads
|
||||
|
||||
h.addFilerJwtAuthHeader(req)
|
||||
|
||||
resp, err := client.Do(req)
|
||||
if err != nil {
|
||||
writeJSONError(w, http.StatusBadGateway, "Failed to fetch file from filer: "+err.Error())
|
||||
return
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
body, err := io.ReadAll(resp.Body)
|
||||
if err != nil {
|
||||
writeJSONError(w, resp.StatusCode, fmt.Sprintf("Filer returned status %d but failed to read response body: %v", resp.StatusCode, err))
|
||||
tracker := &responseWriteTracker{ResponseWriter: w}
|
||||
if err := h.downloadFileGrpc(r.Context(), filePath, tracker); err != nil {
|
||||
// Once bytes have been written we can't switch to a JSON error body
|
||||
// without corrupting the partial response — log and stop. Before any
|
||||
// write the response is still uncommitted, so a 502 with details is
|
||||
// safe.
|
||||
if tracker.committed {
|
||||
glog.Errorf("Error streaming file download: %v", err)
|
||||
return
|
||||
}
|
||||
writeJSONError(w, resp.StatusCode, fmt.Sprintf("Filer returned status %d: %s", resp.StatusCode, string(body)))
|
||||
return
|
||||
writeJSONError(w, http.StatusBadGateway, "Failed to fetch file: "+err.Error())
|
||||
}
|
||||
}
|
||||
|
||||
// Set headers for file download
|
||||
fileName := filepath.Base(cleanFilePath)
|
||||
// Use mime.FormatMediaType for RFC 6266 compliant Content-Disposition,
|
||||
// properly handling non-ASCII characters and special characters
|
||||
w.Header().Set("Content-Disposition", mime.FormatMediaType("attachment", map[string]string{"filename": fileName}))
|
||||
// responseWriteTracker wraps http.ResponseWriter to record whether the
|
||||
// response has been committed (status line + headers sent). DownloadFile
|
||||
// uses this instead of probing Header() so future header-setting code
|
||||
// reorganization can't silently break the "did we already send bytes?"
|
||||
// detection.
|
||||
type responseWriteTracker struct {
|
||||
http.ResponseWriter
|
||||
committed bool
|
||||
}
|
||||
|
||||
// Use content type from filer response, or default to octet-stream
|
||||
contentType := resp.Header.Get("Content-Type")
|
||||
if contentType == "" {
|
||||
contentType = "application/octet-stream"
|
||||
}
|
||||
w.Header().Set("Content-Type", contentType)
|
||||
func (t *responseWriteTracker) WriteHeader(code int) {
|
||||
t.committed = true
|
||||
t.ResponseWriter.WriteHeader(code)
|
||||
}
|
||||
|
||||
// Set content length if available
|
||||
if resp.ContentLength > 0 {
|
||||
w.Header().Set("Content-Length", fmt.Sprintf("%d", resp.ContentLength))
|
||||
}
|
||||
|
||||
// Stream the response body to the client
|
||||
w.WriteHeader(http.StatusOK)
|
||||
_, err = io.Copy(w, resp.Body)
|
||||
if err != nil {
|
||||
glog.Errorf("Error streaming file download: %v", err)
|
||||
}
|
||||
func (t *responseWriteTracker) Write(p []byte) (int, error) {
|
||||
t.committed = true
|
||||
return t.ResponseWriter.Write(p)
|
||||
}
|
||||
|
||||
// ViewFile handles file viewing requests (for text files, images, etc.)
|
||||
@@ -883,64 +644,16 @@ func (h *FileBrowserHandlers) formatBytes(bytes int64) string {
|
||||
|
||||
// Helper function to check if a file is likely a text file by checking content
|
||||
func (h *FileBrowserHandlers) isLikelyTextFile(filePath string, maxCheckSize int64) bool {
|
||||
filerAddress := h.adminServer.GetFilerAddress()
|
||||
if filerAddress == "" {
|
||||
return false
|
||||
}
|
||||
|
||||
// Validate filer address to prevent SSRF
|
||||
if err := h.validateFilerAddress(filerAddress); err != nil {
|
||||
glog.Errorf("Invalid filer address: %v", err)
|
||||
return false
|
||||
}
|
||||
filerHttpAddress := pb.ServerAddress(filerAddress).ToHttpAddress()
|
||||
|
||||
cleanFilePath, err := h.validateAndCleanFilePath(filePath)
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 10*time.Second)
|
||||
defer cancel()
|
||||
content, err := h.fetchFileContentGrpc(ctx, filePath, int(maxCheckSize))
|
||||
if err != nil {
|
||||
return false
|
||||
}
|
||||
|
||||
// Create the file URL with proper scheme based on TLS configuration
|
||||
fileURL := filerFileURL(filerHttpAddress, cleanFilePath)
|
||||
fileURL, err = h.httpClient.NormalizeHttpScheme(fileURL)
|
||||
if err != nil {
|
||||
glog.Errorf("Failed to normalize URL scheme: %v", err)
|
||||
return false
|
||||
if len(content) == 0 {
|
||||
return true
|
||||
}
|
||||
|
||||
// lgtm[go/ssrf]
|
||||
// Safe: filerAddress validated by validateFilerAddress() to match configured filer
|
||||
// Safe: cleanFilePath validated and cleaned by validateAndCleanFilePath() to prevent path traversal
|
||||
client := h.newClientWithTimeout(10 * time.Second)
|
||||
req, err := http.NewRequest("GET", fileURL, nil)
|
||||
if err != nil {
|
||||
glog.Errorf("Failed to create request: %v", err)
|
||||
return false
|
||||
}
|
||||
h.addFilerJwtAuthHeader(req)
|
||||
resp, err := client.Do(req)
|
||||
if err != nil {
|
||||
return false
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
return false
|
||||
}
|
||||
|
||||
// Read first few bytes to check if it's text
|
||||
buffer := make([]byte, min(maxCheckSize, 512))
|
||||
n, err := resp.Body.Read(buffer)
|
||||
if err != nil && err != io.EOF {
|
||||
return false
|
||||
}
|
||||
|
||||
if n == 0 {
|
||||
return true // Empty file can be considered text
|
||||
}
|
||||
|
||||
// Check if content is printable text
|
||||
return h.isPrintableText(buffer[:n])
|
||||
return h.isPrintableText([]byte(content))
|
||||
}
|
||||
|
||||
// Helper function to check if content is printable text
|
||||
@@ -973,35 +686,3 @@ func min(a, b int64) int64 {
|
||||
return b
|
||||
}
|
||||
|
||||
// setupFilerJwtAuth generates a JWT token and adds it to the request Authorization header if configured.
|
||||
func (h *FileBrowserHandlers) setupFilerJwtAuth(req *http.Request, keyPath, expiresPath, operation string) {
|
||||
// Load security configuration
|
||||
v := util.GetViper()
|
||||
|
||||
// Read Filer JWT token from security.toml
|
||||
signingKey := security.SigningKey(v.GetString(keyPath))
|
||||
expiresAfterSec := v.GetInt(expiresPath)
|
||||
|
||||
// Generate JWT token to authenticate with Filer
|
||||
var jwtToken security.EncodedJwt
|
||||
if len(signingKey) > 0 {
|
||||
jwtToken = security.GenJwtForFilerServer(signingKey, expiresAfterSec)
|
||||
glog.V(4).Infof("Generated JWT token for %s (expires in %d sec)", operation, expiresAfterSec)
|
||||
} else {
|
||||
if v.GetString("jwt.signing.key") != "" {
|
||||
glog.Warningf("JWT %s key not configured, but general JWT security is enabled. %s without authentication.", keyPath, operation)
|
||||
} else {
|
||||
glog.V(1).Infof("No JWT signing key configured, %s without authentication", operation)
|
||||
}
|
||||
}
|
||||
|
||||
// Add JWT Token to Authorization Header
|
||||
if jwtToken != "" {
|
||||
req.Header.Set("Authorization", fmt.Sprintf("Bearer %s", string(jwtToken)))
|
||||
glog.V(4).Infof("Added JWT authorization header for %s", operation)
|
||||
}
|
||||
}
|
||||
|
||||
func (h *FileBrowserHandlers) addFilerJwtAuthHeader(req *http.Request) {
|
||||
h.setupFilerJwtAuth(req, "jwt.filer_signing.read.key", "jwt.filer_signing.read.expires_after_seconds", "filer request")
|
||||
}
|
||||
|
||||
@@ -42,21 +42,3 @@ func TestValidateAndCleanFilePath_RejectsEmpty(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestFilerFileURL_EscapesControlChars(t *testing.T) {
|
||||
cases := []struct {
|
||||
addr string
|
||||
path string
|
||||
want string
|
||||
}{
|
||||
{"http://127.0.0.1:8888", "/buckets/profilebuilder/3testGB.zip\n ", "http://127.0.0.1:8888/buckets/profilebuilder/3testGB.zip%0A%20"},
|
||||
{"http://127.0.0.1:8888", "/buckets/profilebuilder/file\rname", "http://127.0.0.1:8888/buckets/profilebuilder/file%0Dname"},
|
||||
{"http://127.0.0.1:8888", "/buckets/profilebuilder/file\x00name", "http://127.0.0.1:8888/buckets/profilebuilder/file%00name"},
|
||||
// Plain path round-trips unchanged.
|
||||
{"http://h:1", "/a/b.txt", "http://h:1/a/b.txt"},
|
||||
}
|
||||
for _, tc := range cases {
|
||||
if got := filerFileURL(tc.addr, tc.path); got != tc.want {
|
||||
t.Errorf("filerFileURL(%q, %q) = %q, want %q", tc.addr, tc.path, got, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -26,6 +26,10 @@ type MaintenanceIntegration struct {
|
||||
// Active topology for task detection and target selection
|
||||
activeTopology *topology.ActiveTopology
|
||||
|
||||
// Master's default replication, refreshed by the scanner each cycle and
|
||||
// passed to detectors as the replica-placement fallback (matches the shell).
|
||||
defaultReplicaPlacement string
|
||||
|
||||
// Type conversion maps
|
||||
taskTypeMap map[types.TaskType]MaintenanceTaskType
|
||||
revTaskTypeMap map[MaintenanceTaskType]types.TaskType
|
||||
@@ -219,9 +223,10 @@ func (s *MaintenanceIntegration) ScanWithTaskDetectors(volumeMetrics []*types.Vo
|
||||
|
||||
// Create cluster info
|
||||
clusterInfo := &types.ClusterInfo{
|
||||
TotalVolumes: len(filteredMetrics),
|
||||
LastUpdated: time.Now(),
|
||||
ActiveTopology: s.activeTopology, // Provide ActiveTopology for destination planning
|
||||
TotalVolumes: len(filteredMetrics),
|
||||
LastUpdated: time.Now(),
|
||||
ActiveTopology: s.activeTopology, // Provide ActiveTopology for destination planning
|
||||
DefaultReplicaPlacement: s.defaultReplicaPlacement,
|
||||
}
|
||||
|
||||
// Run detection for each registered task type
|
||||
@@ -271,6 +276,12 @@ func (s *MaintenanceIntegration) ScanWithTaskDetectors(volumeMetrics []*types.Vo
|
||||
return allResults, nil
|
||||
}
|
||||
|
||||
// SetDefaultReplicaPlacement records the master's default replication so detectors
|
||||
// can use it as the replica-placement fallback (matching the shell).
|
||||
func (s *MaintenanceIntegration) SetDefaultReplicaPlacement(replicaPlacement string) {
|
||||
s.defaultReplicaPlacement = replicaPlacement
|
||||
}
|
||||
|
||||
// UpdateTopologyInfo updates the volume shard tracker with topology information for empty servers
|
||||
func (s *MaintenanceIntegration) UpdateTopologyInfo(topologyInfo *master_pb.TopologyInfo) error {
|
||||
// Log topology details before update for diagnostics
|
||||
|
||||
@@ -8,6 +8,7 @@ import (
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/worker_pb"
|
||||
stats_collect "github.com/seaweedfs/seaweedfs/weed/stats"
|
||||
"github.com/seaweedfs/seaweedfs/weed/worker/tasks/balance"
|
||||
"github.com/seaweedfs/seaweedfs/weed/worker/tasks/erasure_coding"
|
||||
"github.com/seaweedfs/seaweedfs/weed/worker/tasks/vacuum"
|
||||
@@ -315,6 +316,7 @@ func (mm *MaintenanceManager) performScan() {
|
||||
glog.Infof("Starting maintenance scan...")
|
||||
|
||||
results, err := mm.scanner.ScanForMaintenanceTasks()
|
||||
stats_collect.AdminMaintenanceLastScanTimestampSeconds.SetToCurrentTime()
|
||||
if err != nil {
|
||||
// Handle scan error
|
||||
mm.mutex.Lock()
|
||||
@@ -518,6 +520,11 @@ func (mm *MaintenanceManager) GetWorkers() []*MaintenanceWorker {
|
||||
return mm.queue.GetWorkers()
|
||||
}
|
||||
|
||||
// GetWorkerSlotTotals returns worker count and aggregate used/max task slots.
|
||||
func (mm *MaintenanceManager) GetWorkerSlotTotals() (workers, used, max int) {
|
||||
return mm.queue.GetWorkerSlotTotals()
|
||||
}
|
||||
|
||||
// TriggerScan manually triggers a maintenance scan
|
||||
func (mm *MaintenanceManager) TriggerScan() error {
|
||||
return mm.triggerScanInternal(true)
|
||||
|
||||
@@ -8,6 +8,7 @@ import (
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
stats_collect "github.com/seaweedfs/seaweedfs/weed/stats"
|
||||
)
|
||||
|
||||
// NewMaintenanceQueue creates a new maintenance queue
|
||||
@@ -482,12 +483,14 @@ func (mq *MaintenanceQueue) CompleteTask(taskID string, error string) {
|
||||
|
||||
// Calculate task duration
|
||||
var duration time.Duration
|
||||
if task.StartedAt != nil {
|
||||
hadStart := task.StartedAt != nil
|
||||
if hadStart {
|
||||
duration = completedTime.Sub(*task.StartedAt)
|
||||
}
|
||||
|
||||
// Capture workerID before it may be cleared during retry
|
||||
originalWorkerID := task.WorkerID
|
||||
taskType := string(task.Type)
|
||||
|
||||
var taskToSave *MaintenanceTask
|
||||
var logFn func()
|
||||
@@ -577,6 +580,18 @@ func (mq *MaintenanceQueue) CompleteTask(taskID string, error string) {
|
||||
}
|
||||
mq.mutex.Unlock()
|
||||
|
||||
// Record terminal-state metrics. A retry leaves the task pending, so it
|
||||
// is not counted as completed or failed here.
|
||||
switch taskStatus {
|
||||
case TaskStatusCompleted:
|
||||
stats_collect.AdminMaintenanceTasksCompletedTotal.WithLabelValues(taskType, "completed").Inc()
|
||||
case TaskStatusFailed:
|
||||
stats_collect.AdminMaintenanceTasksCompletedTotal.WithLabelValues(taskType, "failed").Inc()
|
||||
}
|
||||
if hadStart && (taskStatus == TaskStatusCompleted || taskStatus == TaskStatusFailed) {
|
||||
stats_collect.AdminMaintenanceTaskDurationSeconds.WithLabelValues(taskType).Observe(duration.Seconds())
|
||||
}
|
||||
|
||||
// Only persist non-terminal tasks (retries). Completed/failed tasks stay
|
||||
// in memory for the UI but are not written to disk — they would just
|
||||
// accumulate and slow down future startups.
|
||||
@@ -819,6 +834,20 @@ func (mq *MaintenanceQueue) GetWorkers() []*MaintenanceWorker {
|
||||
return workers
|
||||
}
|
||||
|
||||
// GetWorkerSlotTotals aggregates worker count and used/max task slots under the
|
||||
// lock, so callers don't read live worker fields that task updates mutate.
|
||||
func (mq *MaintenanceQueue) GetWorkerSlotTotals() (workers, used, max int) {
|
||||
mq.mutex.RLock()
|
||||
defer mq.mutex.RUnlock()
|
||||
|
||||
for _, worker := range mq.workers {
|
||||
workers++
|
||||
used += worker.CurrentLoad
|
||||
max += worker.MaxConcurrent
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
// generateTaskID generates a unique ID for tasks
|
||||
func generateTaskID() string {
|
||||
const charset = "abcdefghijklmnopqrstuvwxyz0123456789"
|
||||
|
||||
@@ -51,6 +51,10 @@ func (ms *MaintenanceScanner) ScanForMaintenanceTasks() ([]*TaskDetectionResult,
|
||||
}
|
||||
}
|
||||
|
||||
// Refresh the master's default replication so detectors can use it as the
|
||||
// replica-placement fallback (matches the shell ec.balance default).
|
||||
ms.integration.SetDefaultReplicaPlacement(ms.getDefaultReplicaPlacement())
|
||||
|
||||
// Use task detection system with complete cluster information
|
||||
results, err := ms.integration.ScanWithTaskDetectors(taskMetrics)
|
||||
if err != nil {
|
||||
@@ -67,6 +71,26 @@ func (ms *MaintenanceScanner) ScanForMaintenanceTasks() ([]*TaskDetectionResult,
|
||||
return []*TaskDetectionResult{}, nil
|
||||
}
|
||||
|
||||
// getDefaultReplicaPlacement reads the master's configured default replication,
|
||||
// used by detectors as the replica-placement fallback. Returns "" on error so
|
||||
// detectors fall back to even spread rather than failing the scan.
|
||||
func (ms *MaintenanceScanner) getDefaultReplicaPlacement() string {
|
||||
var replicaPlacement string
|
||||
err := ms.adminClient.WithMasterClient(func(client master_pb.SeaweedClient) error {
|
||||
resp, err := client.GetMasterConfiguration(context.Background(), &master_pb.GetMasterConfigurationRequest{})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
replicaPlacement = resp.DefaultReplication
|
||||
return nil
|
||||
})
|
||||
if err != nil {
|
||||
glog.V(1).Infof("could not fetch master default replication: %v", err)
|
||||
return ""
|
||||
}
|
||||
return replicaPlacement
|
||||
}
|
||||
|
||||
// getVolumeHealthMetrics collects health information for all volumes.
|
||||
// Returns metrics in task-system format directly (no intermediate copy) and
|
||||
// the topology info for updating the active topology.
|
||||
|
||||
@@ -113,7 +113,7 @@ func cloneClusterContext(in *plugin_pb.ClusterContext) *plugin_pb.ClusterContext
|
||||
}
|
||||
out := &plugin_pb.ClusterContext{
|
||||
MasterGrpcAddresses: in.MasterGrpcAddresses,
|
||||
FilerGrpcAddresses: in.FilerGrpcAddresses,
|
||||
FilerAddresses: in.FilerAddresses,
|
||||
VolumeGrpcAddresses: in.VolumeGrpcAddresses,
|
||||
S3GrpcAddresses: in.S3GrpcAddresses,
|
||||
}
|
||||
|
||||
@@ -115,6 +115,7 @@ func buildErasureCodingExecutionPlan(params *worker_pb.TaskParams) map[string]in
|
||||
source.DataCenter,
|
||||
source.Rack,
|
||||
source.VolumeId,
|
||||
source.DiskId,
|
||||
source.ShardIds,
|
||||
dataShards,
|
||||
))
|
||||
@@ -132,6 +133,7 @@ func buildErasureCodingExecutionPlan(params *worker_pb.TaskParams) map[string]in
|
||||
target.DataCenter,
|
||||
target.Rack,
|
||||
target.VolumeId,
|
||||
target.DiskId,
|
||||
target.ShardIds,
|
||||
dataShards,
|
||||
))
|
||||
@@ -147,6 +149,7 @@ func buildErasureCodingExecutionPlan(params *worker_pb.TaskParams) map[string]in
|
||||
"target_data_center": strings.TrimSpace(target.DataCenter),
|
||||
"target_rack": strings.TrimSpace(target.Rack),
|
||||
"target_volume_id": int(target.VolumeId),
|
||||
"target_disk_id": int(target.DiskId),
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -182,6 +185,7 @@ func buildExecutionEndpoint(
|
||||
dataCenter string,
|
||||
rack string,
|
||||
volumeID uint32,
|
||||
diskID uint32,
|
||||
shardIDs []uint32,
|
||||
dataShardCount int,
|
||||
) map[string]interface{} {
|
||||
@@ -201,6 +205,7 @@ func buildExecutionEndpoint(
|
||||
"data_center": strings.TrimSpace(dataCenter),
|
||||
"rack": strings.TrimSpace(rack),
|
||||
"volume_id": int(volumeID),
|
||||
"disk_id": int(diskID),
|
||||
"shard_ids": allShards,
|
||||
"data_shard_ids": dataShards,
|
||||
"parity_shard_ids": parityShards,
|
||||
|
||||
@@ -423,6 +423,7 @@ func TestTrackExecutionStartStoresErasureCodingExecutionPlan(t *testing.T) {
|
||||
DataCenter: "dc1",
|
||||
Rack: "rack1",
|
||||
VolumeId: 29,
|
||||
DiskId: 5,
|
||||
},
|
||||
},
|
||||
Targets: []*worker_pb.TaskTarget{
|
||||
@@ -431,6 +432,7 @@ func TestTrackExecutionStartStoresErasureCodingExecutionPlan(t *testing.T) {
|
||||
DataCenter: "dc1",
|
||||
Rack: "rack2",
|
||||
VolumeId: 29,
|
||||
DiskId: 2,
|
||||
ShardIds: []uint32{0, 10},
|
||||
},
|
||||
{
|
||||
@@ -438,6 +440,7 @@ func TestTrackExecutionStartStoresErasureCodingExecutionPlan(t *testing.T) {
|
||||
DataCenter: "dc2",
|
||||
Rack: "rack3",
|
||||
VolumeId: 29,
|
||||
DiskId: 3,
|
||||
ShardIds: []uint32{1, 11},
|
||||
},
|
||||
},
|
||||
@@ -486,10 +489,28 @@ func TestTrackExecutionStartStoresErasureCodingExecutionPlan(t *testing.T) {
|
||||
if plan["volume_id"] != float64(29) {
|
||||
t.Fatalf("unexpected execution plan volume id: %+v", plan["volume_id"])
|
||||
}
|
||||
sourcesRaw, ok := plan["sources"].([]interface{})
|
||||
if !ok || len(sourcesRaw) != 1 {
|
||||
t.Fatalf("unexpected sources in execution plan: %+v", plan["sources"])
|
||||
}
|
||||
firstSource, ok := sourcesRaw[0].(map[string]interface{})
|
||||
if !ok {
|
||||
t.Fatalf("unexpected source payload: %+v", sourcesRaw[0])
|
||||
}
|
||||
if firstSource["disk_id"] != float64(5) {
|
||||
t.Fatalf("unexpected source disk_id: %+v", firstSource["disk_id"])
|
||||
}
|
||||
targets, ok := plan["targets"].([]interface{})
|
||||
if !ok || len(targets) != 2 {
|
||||
t.Fatalf("unexpected targets in execution plan: %+v", plan["targets"])
|
||||
}
|
||||
firstTarget, ok := targets[0].(map[string]interface{})
|
||||
if !ok {
|
||||
t.Fatalf("unexpected target payload: %+v", targets[0])
|
||||
}
|
||||
if firstTarget["disk_id"] != float64(2) {
|
||||
t.Fatalf("unexpected target disk_id: %+v", firstTarget["disk_id"])
|
||||
}
|
||||
assignments, ok := plan["shard_assignments"].([]interface{})
|
||||
if !ok || len(assignments) != 4 {
|
||||
t.Fatalf("unexpected shard assignments in execution plan: %+v", plan["shard_assignments"])
|
||||
@@ -501,6 +522,16 @@ func TestTrackExecutionStartStoresErasureCodingExecutionPlan(t *testing.T) {
|
||||
if firstAssignment["shard_id"] != float64(0) || firstAssignment["kind"] != "data" {
|
||||
t.Fatalf("unexpected first assignment: %+v", firstAssignment)
|
||||
}
|
||||
if firstAssignment["target_disk_id"] != float64(2) {
|
||||
t.Fatalf("unexpected first assignment target_disk_id: %+v", firstAssignment["target_disk_id"])
|
||||
}
|
||||
secondAssignment, ok := assignments[1].(map[string]interface{})
|
||||
if !ok {
|
||||
t.Fatalf("unexpected second assignment payload: %+v", assignments[1])
|
||||
}
|
||||
if secondAssignment["shard_id"] != float64(1) || secondAssignment["target_disk_id"] != float64(3) {
|
||||
t.Fatalf("unexpected second assignment: %+v", secondAssignment)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBuildJobDetailIncludesActivitiesAndRunRecord(t *testing.T) {
|
||||
|
||||
@@ -54,6 +54,70 @@ func (at *ActiveTopology) GetEffectiveAvailableCapacityDetailed(nodeID string, d
|
||||
return at.getEffectiveAvailableCapacityUnsafe(disk)
|
||||
}
|
||||
|
||||
// GetEffectiveAvailableEcShardSlots returns a disk's free EC shard slots,
|
||||
// accounting for in-flight task reservations at shard granularity. Unlike the
|
||||
// volume-slot views (GetDisksWithEffectiveCapacity / GetEffectiveAvailableCapacity),
|
||||
// this does not truncate sub-volume shard reservations: it subtracts the full
|
||||
// reservation impact (volume slots converted to shard slots, plus the raw shard
|
||||
// slots) so a reservation that is not a whole multiple of ShardsPerVolumeSlot is
|
||||
// not lost. It does NOT subtract the EC shards already persisted on the disk;
|
||||
// callers that track those (from EcShardInfos) subtract them separately.
|
||||
//
|
||||
// shardsPerVolume is the number of EC shards of the target collection that fit in
|
||||
// one volume slot (i.e. its data-shard count): a 4+2 volume's shards are ~1/4 of a
|
||||
// volume each, so one volume slot holds 4 of them, not the default
|
||||
// ShardsPerVolumeSlot. Pass <= 0 to use the default. Using the target ratio keeps
|
||||
// Place from over-filling a disk for low-data-shard layouts.
|
||||
func (at *ActiveTopology) GetEffectiveAvailableEcShardSlots(nodeID string, diskID uint32, shardsPerVolume int) int {
|
||||
if shardsPerVolume <= 0 {
|
||||
shardsPerVolume = ShardsPerVolumeSlot
|
||||
}
|
||||
|
||||
at.mutex.RLock()
|
||||
defer at.mutex.RUnlock()
|
||||
|
||||
diskKey := fmt.Sprintf("%s:%d", nodeID, diskID)
|
||||
disk, exists := at.disks[diskKey]
|
||||
if !exists || disk.DiskInfo == nil || disk.DiskInfo.DiskInfo == nil {
|
||||
return 0
|
||||
}
|
||||
|
||||
info := disk.DiskInfo.DiskInfo
|
||||
base := info.MaxVolumeCount - info.VolumeCount
|
||||
if base <= 0 && info.MaxVolumeCount == 0 && info.VolumeCount == 0 &&
|
||||
len(info.VolumeInfos) == 0 && len(info.EcShardInfos) == 0 {
|
||||
// Freshly started empty servers can report max=0 before publishing concrete
|
||||
// limits; keep one provisional slot so EC placement still sees the disk,
|
||||
// mirroring getEffectiveAvailableCapacityUnsafe.
|
||||
base = 1
|
||||
}
|
||||
if base < 0 {
|
||||
base = 0
|
||||
}
|
||||
// calculateTaskStorageImpact reports consumption as positive, so subtract it.
|
||||
// Volume-slot reservations scale by the target ratio; the sub-volume shard-slot
|
||||
// remainder is in default units and subtracted as-is (a small approximation).
|
||||
impact := at.getEffectiveCapacityUnsafe(disk)
|
||||
// impact.ShardSlots is recorded in default ShardsPerVolumeSlot units; convert it
|
||||
// to the target ratio's shard slots before subtracting (identity when
|
||||
// shardsPerVolume == ShardsPerVolumeSlot). Round a positive reservation up so a
|
||||
// sub-slot reservation (e.g. 1 default slot against a 4-shard target) is not
|
||||
// truncated to zero and wrongly counted as free.
|
||||
scaledShardImpact := int64(impact.ShardSlots) * int64(shardsPerVolume)
|
||||
if scaledShardImpact > 0 {
|
||||
scaledShardImpact = (scaledShardImpact + int64(ShardsPerVolumeSlot) - 1) / int64(ShardsPerVolumeSlot)
|
||||
} else {
|
||||
scaledShardImpact /= int64(ShardsPerVolumeSlot)
|
||||
}
|
||||
free := base*int64(shardsPerVolume) -
|
||||
int64(impact.VolumeSlots)*int64(shardsPerVolume) -
|
||||
scaledShardImpact
|
||||
if free < 0 {
|
||||
free = 0
|
||||
}
|
||||
return int(free)
|
||||
}
|
||||
|
||||
// GetEffectiveCapacityImpact returns the StorageSlotChange impact for a disk
|
||||
// This shows the net impact from all pending and assigned tasks
|
||||
func (at *ActiveTopology) GetEffectiveCapacityImpact(nodeID string, diskID uint32) StorageSlotChange {
|
||||
|
||||
@@ -0,0 +1,54 @@
|
||||
package topology
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
|
||||
)
|
||||
|
||||
// TestCountTopologyResources_multiDiskPerNode covers the case where the master
|
||||
// keys DiskInfos by disk type, so several same-type physical disks on a node
|
||||
// collapse into a single DiskInfo entry. Counting len(DiskInfos) under-reports
|
||||
// the physical disk count and disagrees with the per-disk activeDisk map that
|
||||
// the rest of the admin topology builds via SplitByPhysicalDisk.
|
||||
func TestCountTopologyResources_multiDiskPerNode(t *testing.T) {
|
||||
makeNode := func(id string) *master_pb.DataNodeInfo {
|
||||
var ecShardInfos []*master_pb.VolumeEcShardInformationMessage
|
||||
for diskId := uint32(0); diskId < 6; diskId++ {
|
||||
ecShardInfos = append(ecShardInfos, &master_pb.VolumeEcShardInformationMessage{
|
||||
Id: diskId + 1,
|
||||
DiskId: diskId,
|
||||
EcIndexBits: 1,
|
||||
})
|
||||
}
|
||||
return &master_pb.DataNodeInfo{
|
||||
Id: id,
|
||||
DiskInfos: map[string]*master_pb.DiskInfo{
|
||||
"": {Type: "", MaxVolumeCount: 60, EcShardInfos: ecShardInfos},
|
||||
},
|
||||
}
|
||||
}
|
||||
topo := &master_pb.TopologyInfo{
|
||||
Id: "multi_disk_topo",
|
||||
DataCenterInfos: []*master_pb.DataCenterInfo{{
|
||||
Id: "dc1",
|
||||
RackInfos: []*master_pb.RackInfo{{
|
||||
Id: "rack1",
|
||||
DataNodeInfos: []*master_pb.DataNodeInfo{
|
||||
makeNode("node1"), makeNode("node2"), makeNode("node3"),
|
||||
},
|
||||
}},
|
||||
}},
|
||||
}
|
||||
|
||||
dcCount, nodeCount, diskCount := CountTopologyResources(topo)
|
||||
if dcCount != 1 {
|
||||
t.Errorf("dcCount = %d, want 1", dcCount)
|
||||
}
|
||||
if nodeCount != 3 {
|
||||
t.Errorf("nodeCount = %d, want 3", nodeCount)
|
||||
}
|
||||
if diskCount != 18 {
|
||||
t.Errorf("diskCount = %d, want 18 (6 physical disks x 3 nodes)", diskCount)
|
||||
}
|
||||
}
|
||||
@@ -9,74 +9,6 @@ import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/erasure_coding"
|
||||
)
|
||||
|
||||
// splitDiskInfoByPhysicalDisk returns one master_pb.DiskInfo per physical
|
||||
// disk_id observed in VolumeInfos / EcShardInfos. Multiple same-type physical
|
||||
// disks collapse to one DiskInfo at the master; per-volume/per-shard records
|
||||
// keep the original disk_id and are the authoritative signal here. Capacity
|
||||
// is split evenly — the wire format doesn't carry per-disk capacity yet.
|
||||
func splitDiskInfoByPhysicalDisk(diskInfo *master_pb.DiskInfo) []*master_pb.DiskInfo {
|
||||
if diskInfo == nil {
|
||||
return nil
|
||||
}
|
||||
|
||||
// Records with DiskId=0 and a non-zero outer DiskId belong to the outer
|
||||
// disk — handles older payloads / fixtures that omit the per-record id.
|
||||
normalize := func(id uint32) uint32 {
|
||||
if id == 0 && diskInfo.DiskId != 0 {
|
||||
return diskInfo.DiskId
|
||||
}
|
||||
return id
|
||||
}
|
||||
|
||||
diskIDs := make(map[uint32]struct{})
|
||||
for _, vi := range diskInfo.VolumeInfos {
|
||||
diskIDs[normalize(vi.DiskId)] = struct{}{}
|
||||
}
|
||||
for _, eci := range diskInfo.EcShardInfos {
|
||||
diskIDs[normalize(eci.DiskId)] = struct{}{}
|
||||
}
|
||||
if len(diskIDs) == 0 {
|
||||
diskIDs[diskInfo.DiskId] = struct{}{}
|
||||
}
|
||||
|
||||
if len(diskIDs) == 1 {
|
||||
for diskID := range diskIDs {
|
||||
if diskID == diskInfo.DiskId {
|
||||
return []*master_pb.DiskInfo{diskInfo}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
perDiskVolumes := make(map[uint32][]*master_pb.VolumeInformationMessage)
|
||||
for _, vi := range diskInfo.VolumeInfos {
|
||||
perDiskVolumes[normalize(vi.DiskId)] = append(perDiskVolumes[normalize(vi.DiskId)], vi)
|
||||
}
|
||||
perDiskShards := make(map[uint32][]*master_pb.VolumeEcShardInformationMessage)
|
||||
for _, eci := range diskInfo.EcShardInfos {
|
||||
perDiskShards[normalize(eci.DiskId)] = append(perDiskShards[normalize(eci.DiskId)], eci)
|
||||
}
|
||||
|
||||
count := int64(len(diskIDs))
|
||||
share := func(total int64) int64 { return total / count }
|
||||
|
||||
result := make([]*master_pb.DiskInfo, 0, len(diskIDs))
|
||||
for diskID := range diskIDs {
|
||||
result = append(result, &master_pb.DiskInfo{
|
||||
Type: diskInfo.Type,
|
||||
MaxVolumeCount: share(diskInfo.MaxVolumeCount),
|
||||
VolumeCount: int64(len(perDiskVolumes[diskID])),
|
||||
FreeVolumeCount: share(diskInfo.FreeVolumeCount),
|
||||
ActiveVolumeCount: share(diskInfo.ActiveVolumeCount),
|
||||
RemoteVolumeCount: share(diskInfo.RemoteVolumeCount),
|
||||
VolumeInfos: perDiskVolumes[diskID],
|
||||
EcShardInfos: perDiskShards[diskID],
|
||||
DiskId: diskID,
|
||||
Tags: append([]string(nil), diskInfo.Tags...),
|
||||
})
|
||||
}
|
||||
return result
|
||||
}
|
||||
|
||||
// CountTopologyResources counts datacenters, nodes, and disks in topology info
|
||||
func CountTopologyResources(topologyInfo *master_pb.TopologyInfo) (dcCount, nodeCount, diskCount int) {
|
||||
if topologyInfo == nil {
|
||||
@@ -87,7 +19,12 @@ func CountTopologyResources(topologyInfo *master_pb.TopologyInfo) (dcCount, node
|
||||
for _, rack := range dc.RackInfos {
|
||||
nodeCount += len(rack.DataNodeInfos)
|
||||
for _, node := range rack.DataNodeInfos {
|
||||
diskCount += len(node.DiskInfos)
|
||||
// DiskInfos is keyed by disk type, so same-type physical disks
|
||||
// collapse into one entry. Count physical disks so the number
|
||||
// matches the per-disk activeDisk map.
|
||||
for _, diskInfo := range node.DiskInfos {
|
||||
diskCount += len(diskInfo.SplitByPhysicalDisk())
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -142,10 +79,10 @@ func (at *ActiveTopology) UpdateTopology(topologyInfo *master_pb.TopologyInfo) e
|
||||
disks: make(map[uint32]*activeDisk),
|
||||
}
|
||||
|
||||
// One activeDisk per physical disk_id (#9369): the master keys
|
||||
// One activeDisk per physical disk_id: the master keys
|
||||
// DiskInfos by disk type, so same-type disks must be split out.
|
||||
for diskType, diskInfo := range nodeInfo.DiskInfos {
|
||||
perDiskInfos := splitDiskInfoByPhysicalDisk(diskInfo)
|
||||
perDiskInfos := diskInfo.SplitByPhysicalDisk()
|
||||
for _, perDisk := range perDiskInfos {
|
||||
disk := &activeDisk{
|
||||
DiskInfo: &DiskInfo{
|
||||
|
||||
+103
-21
@@ -4,6 +4,7 @@ import (
|
||||
"context"
|
||||
"fmt"
|
||||
"strings"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
|
||||
@@ -11,7 +12,6 @@ import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
"google.golang.org/grpc"
|
||||
)
|
||||
|
||||
@@ -20,6 +20,14 @@ type LockClient struct {
|
||||
maxLockDuration time.Duration
|
||||
sleepDuration time.Duration
|
||||
seedFiler pb.ServerAddress
|
||||
|
||||
// ring is an optional client-side view of the filer lock hash ring. When
|
||||
// populated, a new lock starts at the key's primary filer instead of the
|
||||
// seed filer, avoiding the seed->primary forward hop. A stale view stays
|
||||
// correct: the filer forwards to the real primary as a fallback.
|
||||
ringMu sync.RWMutex
|
||||
ring *lock_manager.HashRing
|
||||
ringVersion int64
|
||||
}
|
||||
|
||||
func NewLockClient(grpcDialOption grpc.DialOption, seedFiler pb.ServerAddress) *LockClient {
|
||||
@@ -31,12 +39,56 @@ func NewLockClient(grpcDialOption grpc.DialOption, seedFiler pb.ServerAddress) *
|
||||
}
|
||||
}
|
||||
|
||||
// SetRing mirrors the master's LockRingUpdate so the client computes the same
|
||||
// primary the filers do. A non-zero version at or below the current one is
|
||||
// ignored once a ring exists, dropping reordered and redundant broadcasts;
|
||||
// version 0 always applies (bootstrap).
|
||||
func (lc *LockClient) SetRing(servers []pb.ServerAddress, version int64) {
|
||||
lc.ringMu.Lock()
|
||||
defer lc.ringMu.Unlock()
|
||||
if version != 0 && version <= lc.ringVersion && lc.ring != nil {
|
||||
return
|
||||
}
|
||||
lc.ringVersion = version
|
||||
if lc.ring == nil {
|
||||
lc.ring = lock_manager.NewHashRing(lock_manager.DefaultVnodeCount)
|
||||
}
|
||||
lc.ring.SetServers(servers)
|
||||
}
|
||||
|
||||
// hostForKey returns the filer that should own key per the current ring view,
|
||||
// falling back to the seed filer when no view has been received yet.
|
||||
func (lc *LockClient) hostForKey(key string) pb.ServerAddress {
|
||||
lc.ringMu.RLock()
|
||||
defer lc.ringMu.RUnlock()
|
||||
if lc.ring == nil {
|
||||
return lc.seedFiler
|
||||
}
|
||||
if primary := lc.ring.GetPrimary(key); primary != "" {
|
||||
return primary
|
||||
}
|
||||
return lc.seedFiler
|
||||
}
|
||||
|
||||
// PrimaryForKey returns the ring owner for key, or "" before any ring arrives.
|
||||
// Unlike hostForKey it does not fall back to the seed, so a route-by-key caller
|
||||
// stays on the distributed lock until the ring is known.
|
||||
func (lc *LockClient) PrimaryForKey(key string) pb.ServerAddress {
|
||||
lc.ringMu.RLock()
|
||||
defer lc.ringMu.RUnlock()
|
||||
if lc.ring == nil {
|
||||
return ""
|
||||
}
|
||||
return lc.ring.GetPrimary(key)
|
||||
}
|
||||
|
||||
type LiveLock struct {
|
||||
key string
|
||||
renewToken string
|
||||
expireAtNs int64
|
||||
hostFiler pb.ServerAddress
|
||||
cancelCh chan struct{}
|
||||
renewalDone chan struct{} // closed when the renewal goroutine exits; nil if there is none
|
||||
grpcDialOption grpc.DialOption
|
||||
isLocked int32 // 0 = unlocked, 1 = locked; use atomic operations
|
||||
self string
|
||||
@@ -51,7 +103,7 @@ type LiveLock struct {
|
||||
func (lc *LockClient) NewShortLivedLock(key string, owner string) (lock *LiveLock) {
|
||||
lock = &LiveLock{
|
||||
key: key,
|
||||
hostFiler: lc.seedFiler,
|
||||
hostFiler: lc.hostForKey(key),
|
||||
cancelCh: make(chan struct{}),
|
||||
expireAtNs: time.Now().Add(5 * time.Second).UnixNano(),
|
||||
grpcDialOption: lc.grpcDialOption,
|
||||
@@ -72,7 +124,7 @@ func (lc *LockClient) NewBlockingLongLivedLock(key, owner string, lockTTL time.D
|
||||
}
|
||||
lock := &LiveLock{
|
||||
key: key,
|
||||
hostFiler: lc.seedFiler,
|
||||
hostFiler: lc.hostForKey(key),
|
||||
cancelCh: make(chan struct{}),
|
||||
expireAtNs: time.Now().Add(lockTTL).UnixNano(),
|
||||
grpcDialOption: lc.grpcDialOption,
|
||||
@@ -83,7 +135,9 @@ func (lc *LockClient) NewBlockingLongLivedLock(key, owner string, lockTTL time.D
|
||||
// Block until acquired
|
||||
lock.retryUntilLocked(lockTTL)
|
||||
// Start renewal goroutine using a ticker for interruptible sleep
|
||||
lock.renewalDone = make(chan struct{})
|
||||
go func() {
|
||||
defer close(lock.renewalDone)
|
||||
renewInterval := lockTTL / 2
|
||||
ticker := time.NewTicker(renewInterval)
|
||||
defer ticker.Stop()
|
||||
@@ -108,7 +162,7 @@ func (lc *LockClient) NewBlockingLongLivedLock(key, owner string, lockTTL time.D
|
||||
func (lc *LockClient) StartLongLivedLock(key string, owner string, onLockOwnerChange func(newLockOwner string), lockTTL time.Duration) (lock *LiveLock) {
|
||||
lock = &LiveLock{
|
||||
key: key,
|
||||
hostFiler: lc.seedFiler,
|
||||
hostFiler: lc.hostForKey(key),
|
||||
cancelCh: make(chan struct{}),
|
||||
expireAtNs: time.Now().Add(lockTTL).UnixNano(),
|
||||
grpcDialOption: lc.grpcDialOption,
|
||||
@@ -119,7 +173,9 @@ func (lc *LockClient) StartLongLivedLock(key string, owner string, onLockOwnerCh
|
||||
if lock.lockTTL == 0 {
|
||||
lock.lockTTL = lock_manager.LiveLockTTL
|
||||
}
|
||||
lock.renewalDone = make(chan struct{})
|
||||
go func() {
|
||||
defer close(lock.renewalDone)
|
||||
renewInterval := lock.lockTTL / 2
|
||||
isLocked := false
|
||||
lockOwner := ""
|
||||
@@ -149,30 +205,39 @@ func (lc *LockClient) StartLongLivedLock(key string, owner string, onLockOwnerCh
|
||||
onLockOwnerChange(lock.LockOwner())
|
||||
lockOwner = lock.LockOwner()
|
||||
}
|
||||
// Sleep until the next attempt, but wake immediately on Stop() so
|
||||
// the goroutine exits and closes renewalDone before Stop()'s bounded
|
||||
// wait elapses. An uninterruptible sleep here (up to 5*renewInterval
|
||||
// when unlocked) can outlast that wait and break the shutdown
|
||||
// synchronization.
|
||||
sleepFor := renewInterval
|
||||
if !isLocked {
|
||||
sleepFor = 5 * renewInterval
|
||||
}
|
||||
timer := time.NewTimer(sleepFor)
|
||||
select {
|
||||
case <-lock.cancelCh:
|
||||
timer.Stop()
|
||||
return
|
||||
default:
|
||||
if isLocked {
|
||||
time.Sleep(renewInterval)
|
||||
} else {
|
||||
time.Sleep(5 * renewInterval)
|
||||
}
|
||||
case <-timer.C:
|
||||
}
|
||||
}
|
||||
}()
|
||||
return
|
||||
}
|
||||
|
||||
// retryUntilLocked blocks until the lock is acquired, polling at the steady
|
||||
// short cadence that AttemptToLock already enforces on contention (~1s). It
|
||||
// deliberately avoids util.RetryUntil's exponential backoff (which grows to
|
||||
// several seconds): when a holder on another mount releases the lock, the
|
||||
// waiter must pick it up promptly, otherwise cross-mount write handoff stalls
|
||||
// long enough to time out clients.
|
||||
func (lock *LiveLock) retryUntilLocked(lockDuration time.Duration) {
|
||||
util.RetryUntil("create lock:"+lock.key, func() error {
|
||||
return lock.AttemptToLock(lockDuration)
|
||||
}, func(err error) (shouldContinue bool) {
|
||||
if err != nil {
|
||||
glog.Warningf("create lock %s: %s", lock.key, err)
|
||||
for lock.renewToken == "" {
|
||||
if err := lock.AttemptToLock(lockDuration); err != nil {
|
||||
glog.V(1).Infof("create lock %s: %v", lock.key, err)
|
||||
}
|
||||
return lock.renewToken == ""
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func (lock *LiveLock) AttemptToLock(lockDuration time.Duration) error {
|
||||
@@ -226,10 +291,27 @@ func (lock *LiveLock) Stop() error {
|
||||
close(lock.cancelCh)
|
||||
}
|
||||
|
||||
// Wait a brief moment for the goroutine to see the closed channel
|
||||
// This reduces the race condition window where the goroutine might
|
||||
// attempt one more lock operation after we've released the lock
|
||||
time.Sleep(10 * time.Millisecond)
|
||||
// Wait for the renewal goroutine to fully exit before unlocking. A renewal
|
||||
// in flight when we close cancelCh rotates renewToken on the server; if we
|
||||
// then unlock with the token we read here, the unlock fails with a token
|
||||
// mismatch and the lock lingers until its TTL expires — blocking other
|
||||
// mounts waiting on the same file. Waiting for the goroutine to return also
|
||||
// makes the renewToken read below race-free (channel close = happens-before).
|
||||
if lock.renewalDone != nil {
|
||||
select {
|
||||
case <-lock.renewalDone:
|
||||
case <-time.After(lock.lockTTL + 2*time.Second):
|
||||
// The renewal goroutine is wedged, almost certainly in a stuck
|
||||
// renewal RPC. Do not unlock here: the renewToken may be rotated
|
||||
// when that RPC finally returns, so an unlock sent now could race
|
||||
// it, be rejected on a stale token, and leave the lock lingering
|
||||
// anyway. cancelCh is closed, so the goroutine stops renewing once
|
||||
// its in-flight call returns and the lock then expires within its
|
||||
// TTL on its own.
|
||||
glog.Warningf("lock %s: renewal goroutine still running at shutdown; letting lock expire via TTL", lock.key)
|
||||
return nil
|
||||
}
|
||||
}
|
||||
|
||||
// Also release the lock if held
|
||||
// Note: We intentionally don't clear renewToken here because
|
||||
|
||||
@@ -0,0 +1,92 @@
|
||||
package cluster
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/cluster/lock_manager"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
)
|
||||
|
||||
// The gateway must resolve a lock key to the same primary the filers do,
|
||||
// otherwise it dials the wrong filer and the lock still gets forwarded. Both
|
||||
// sides use the same HashRing over the same server set, so for every key the
|
||||
// client's hostForKey must equal the filer ring's GetPrimary.
|
||||
func TestLockClientHostMatchesFilerRing(t *testing.T) {
|
||||
servers := []pb.ServerAddress{
|
||||
"filer-a:8888", "filer-b:8888", "filer-c:8888", "filer-d:8888",
|
||||
}
|
||||
|
||||
filerRing := lock_manager.NewHashRing(lock_manager.DefaultVnodeCount)
|
||||
filerRing.SetServers(servers)
|
||||
|
||||
lc := NewLockClient(nil, "seed:8888")
|
||||
lc.SetRing(servers, 1)
|
||||
|
||||
for _, key := range []string{
|
||||
"s3.object.write:/buckets/b/obj-0",
|
||||
"s3.object.write:/buckets/b/obj-1",
|
||||
"s3.object.write:/buckets/b/obj-2",
|
||||
"s3.object.write:/buckets/gosbench-0/w0obj-kilo-0877",
|
||||
"some/other/key",
|
||||
} {
|
||||
if got, want := lc.hostForKey(key), filerRing.GetPrimary(key); got != want {
|
||||
t.Errorf("key %q: client host %q != filer primary %q", key, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Without a ring view, the client falls back to the seed filer (which the filer
|
||||
// forwards from), preserving the pre-optimization behavior.
|
||||
func TestLockClientHostFallsBackToSeed(t *testing.T) {
|
||||
lc := NewLockClient(nil, "seed:8888")
|
||||
if got := lc.hostForKey("any-key"); got != "seed:8888" {
|
||||
t.Errorf("expected seed fallback, got %q", got)
|
||||
}
|
||||
|
||||
// An empty ring (no members yet) also falls back to the seed.
|
||||
lc.SetRing(nil, 1)
|
||||
if got := lc.hostForKey("any-key"); got != "seed:8888" {
|
||||
t.Errorf("expected seed fallback on empty ring, got %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
// A stale (older-version) update must not regress a newer ring view, while
|
||||
// version 0 always applies as a bootstrap.
|
||||
func TestLockClientSetRingVersionGuard(t *testing.T) {
|
||||
lc := NewLockClient(nil, "seed:8888")
|
||||
|
||||
newer := []pb.ServerAddress{"filer-a:8888", "filer-b:8888"}
|
||||
lc.SetRing(newer, 10)
|
||||
primaryAt10 := lc.hostForKey("k")
|
||||
|
||||
// Older version is ignored.
|
||||
lc.SetRing([]pb.ServerAddress{"filer-z:8888"}, 5)
|
||||
if got := lc.hostForKey("k"); got != primaryAt10 {
|
||||
t.Errorf("stale update applied: host changed to %q", got)
|
||||
}
|
||||
|
||||
// version 0 is always accepted.
|
||||
lc.SetRing([]pb.ServerAddress{"filer-z:8888"}, 0)
|
||||
if got := lc.hostForKey("k"); got != "filer-z:8888" {
|
||||
t.Errorf("bootstrap update not applied, got %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
// PrimaryForKey returns "" before any ring is received (so a route-by-key
|
||||
// caller falls back to the distributed lock) and the ring owner afterwards,
|
||||
// unlike hostForKey which falls back to the seed.
|
||||
func TestLockClientPrimaryForKey(t *testing.T) {
|
||||
lc := NewLockClient(nil, "seed:8888")
|
||||
if got := lc.PrimaryForKey("k"); got != "" {
|
||||
t.Errorf("expected empty before ring, got %q", got)
|
||||
}
|
||||
|
||||
lc.SetRing([]pb.ServerAddress{"filer-a:8888", "filer-b:8888"}, 1)
|
||||
got := lc.PrimaryForKey("k")
|
||||
if got == "" {
|
||||
t.Fatal("expected an owner after ring set")
|
||||
}
|
||||
if got != lc.hostForKey("k") {
|
||||
t.Errorf("PrimaryForKey %q disagrees with hostForKey %q", got, lc.hostForKey("k"))
|
||||
}
|
||||
}
|
||||
@@ -145,7 +145,15 @@ func (hr *HashRing) rebuildRing() {
|
||||
hr.vnodeToServer = make(map[uint32]pb.ServerAddress, len(hr.servers)*hr.vnodeCount)
|
||||
hr.sortedHashes = make([]uint32, 0, len(hr.servers)*hr.vnodeCount)
|
||||
|
||||
// Sort so a vnode-hash collision resolves to the same server on every node;
|
||||
// map iteration order alone is randomized per process.
|
||||
servers := make([]pb.ServerAddress, 0, len(hr.servers))
|
||||
for server := range hr.servers {
|
||||
servers = append(servers, server)
|
||||
}
|
||||
sort.Slice(servers, func(i, j int) bool { return servers[i] < servers[j] })
|
||||
|
||||
for _, server := range servers {
|
||||
for i := 0; i < hr.vnodeCount; i++ {
|
||||
vnodeKey := vnodeKeyFor(server, i)
|
||||
hash := hashKey(vnodeKey)
|
||||
|
||||
@@ -171,3 +171,38 @@ func TestHashRing_GetPrimary(t *testing.T) {
|
||||
primary, _ := hr.GetPrimaryAndBackup("mykey")
|
||||
assert.Equal(t, primary, hr.GetPrimary("mykey"))
|
||||
}
|
||||
|
||||
// The ring must be identical on every node holding the same server set,
|
||||
// regardless of the order servers were added or supplied. Build rings several
|
||||
// ways and assert they agree on the primary for a wide range of keys.
|
||||
func TestHashRing_OrderIndependent(t *testing.T) {
|
||||
servers := []pb.ServerAddress{
|
||||
"filer-a:8888", "filer-b:8888", "filer-c:8888", "filer-d:8888", "filer-e:8888",
|
||||
}
|
||||
|
||||
bySet := NewHashRing(50)
|
||||
bySet.SetServers(servers)
|
||||
|
||||
byReverse := NewHashRing(50)
|
||||
rev := append([]pb.ServerAddress(nil), servers...)
|
||||
for i, j := 0, len(rev)-1; i < j; i, j = i+1, j-1 {
|
||||
rev[i], rev[j] = rev[j], rev[i]
|
||||
}
|
||||
byReverse.SetServers(rev)
|
||||
|
||||
byAdd := NewHashRing(50)
|
||||
for _, s := range []pb.ServerAddress{"filer-c:8888", "filer-e:8888", "filer-a:8888", "filer-d:8888", "filer-b:8888"} {
|
||||
byAdd.AddServer(s)
|
||||
}
|
||||
|
||||
for i := 0; i < 5000; i++ {
|
||||
key := fmt.Sprintf("s3.object.write:/buckets/b/obj-%d", i)
|
||||
p := bySet.GetPrimary(key)
|
||||
if got := byReverse.GetPrimary(key); got != p {
|
||||
t.Fatalf("reverse-order ring disagrees on %q: %s vs %s", key, got, p)
|
||||
}
|
||||
if got := byAdd.GetPrimary(key); got != p {
|
||||
t.Fatalf("add-order ring disagrees on %q: %s vs %s", key, got, p)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -12,6 +12,7 @@ import (
|
||||
type LockRingSnapshot struct {
|
||||
servers []pb.ServerAddress
|
||||
ts time.Time
|
||||
ring *HashRing // prebuilt ring for servers, so PriorOwner need not rebuild per call
|
||||
}
|
||||
|
||||
type LockRing struct {
|
||||
@@ -58,10 +59,12 @@ func (r *LockRing) SetSnapshot(servers []pb.ServerAddress, version int64) bool {
|
||||
// are always consistent — prevents a concurrent SetSnapshot from
|
||||
// seeing the new version but applying its servers to the old ring.
|
||||
r.Ring.SetServers(servers)
|
||||
// Append the snapshot under the same lock as the ring update so a concurrent
|
||||
// PriorOwner always sees snapshots[0] matching r.Ring (and snapshots[1] as the
|
||||
// true prior); otherwise it could pair a new ring with a stale prior snapshot.
|
||||
r.addOneSnapshotLocked(servers)
|
||||
r.Unlock()
|
||||
|
||||
r.addOneSnapshot(servers)
|
||||
|
||||
r.cleanupWg.Add(1)
|
||||
go func() {
|
||||
defer r.cleanupWg.Done()
|
||||
@@ -78,14 +81,16 @@ func (r *LockRing) Version() int64 {
|
||||
return r.version
|
||||
}
|
||||
|
||||
func (r *LockRing) addOneSnapshot(servers []pb.ServerAddress) {
|
||||
r.Lock()
|
||||
defer r.Unlock()
|
||||
|
||||
// addOneSnapshotLocked appends a new snapshot (newest at index 0). The caller
|
||||
// must hold r.Lock(), so the ring update and snapshot append are one atomic step.
|
||||
func (r *LockRing) addOneSnapshotLocked(servers []pb.ServerAddress) {
|
||||
ts := time.Now()
|
||||
ring := NewHashRing(DefaultVnodeCount)
|
||||
ring.SetServers(servers)
|
||||
t := &LockRingSnapshot{
|
||||
servers: servers,
|
||||
ts: ts,
|
||||
ring: ring,
|
||||
}
|
||||
r.snapshots = append(r.snapshots, t)
|
||||
for i := len(r.snapshots) - 2; i >= 0; i-- {
|
||||
@@ -148,6 +153,35 @@ func (r *LockRing) GetPrimary(key string) pb.ServerAddress {
|
||||
return r.Ring.GetPrimary(key)
|
||||
}
|
||||
|
||||
// PriorOwner returns the key's owner from the previous ring snapshot, but only
|
||||
// while the ring changed within the last snapshotInterval and that owner differs
|
||||
// from the current primary. This is the cooling-off window in which the previous
|
||||
// owner may still hold locks the new owner has not yet rebuilt — a caller can
|
||||
// consult it before granting so a fresh owner does not double-grant during a
|
||||
// rebalance. Returns "" outside the window or when ownership did not move. It
|
||||
// uses the snapshot's prebuilt ring, so it does not rebuild a hash ring per call.
|
||||
func (r *LockRing) PriorOwner(key string) pb.ServerAddress {
|
||||
r.RLock()
|
||||
defer r.RUnlock()
|
||||
if len(r.snapshots) < 2 {
|
||||
return ""
|
||||
}
|
||||
if time.Since(r.snapshots[0].ts) > r.snapshotInterval {
|
||||
return ""
|
||||
}
|
||||
current := r.Ring.GetPrimary(key)
|
||||
var prior pb.ServerAddress
|
||||
if pr := r.snapshots[1].ring; pr != nil {
|
||||
prior = pr.GetPrimary(key)
|
||||
} else {
|
||||
prior = hashKeyToServer(key, r.snapshots[1].servers)
|
||||
}
|
||||
if prior != "" && prior != current {
|
||||
return prior
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// hashKeyToServer uses a temporary consistent hash ring for the given server list.
|
||||
func hashKeyToServer(key string, servers []pb.ServerAddress) pb.ServerAddress {
|
||||
if len(servers) == 0 {
|
||||
|
||||
@@ -0,0 +1,63 @@
|
||||
package lock_manager
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
)
|
||||
|
||||
func TestLockRing_PriorOwner(t *testing.T) {
|
||||
r := NewLockRing(5 * time.Second)
|
||||
t.Cleanup(r.WaitForCleanup)
|
||||
|
||||
setA := []pb.ServerAddress{"s1:1", "s2:1", "s3:1"}
|
||||
r.SetSnapshot(setA, 1)
|
||||
|
||||
// Only one snapshot: nothing to fall back to.
|
||||
if got := r.PriorOwner("any"); got != "" {
|
||||
t.Fatalf("single snapshot should have no prior owner, got %q", got)
|
||||
}
|
||||
|
||||
// Add a server so some keys' ownership moves.
|
||||
setB := []pb.ServerAddress{"s1:1", "s2:1", "s3:1", "s4:1"}
|
||||
r.SetSnapshot(setB, 2)
|
||||
|
||||
var moved, stable string
|
||||
for i := 0; i < 2000 && (moved == "" || stable == ""); i++ {
|
||||
key := fmt.Sprintf("key-%d", i)
|
||||
if r.GetPrimary(key) != hashKeyToServer(key, setA) {
|
||||
if moved == "" {
|
||||
moved = key
|
||||
}
|
||||
} else if stable == "" {
|
||||
stable = key
|
||||
}
|
||||
}
|
||||
if moved == "" || stable == "" {
|
||||
t.Skip("could not find both a moved and a stable key")
|
||||
}
|
||||
|
||||
if got, want := r.PriorOwner(moved), hashKeyToServer(moved, setA); got != want {
|
||||
t.Fatalf("PriorOwner(moved)=%q, want %q", got, want)
|
||||
}
|
||||
if got := r.PriorOwner(stable); got != "" {
|
||||
t.Fatalf("unmoved key should have no prior owner, got %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestLockRing_PriorOwnerExpires(t *testing.T) {
|
||||
r := NewLockRing(20 * time.Millisecond)
|
||||
t.Cleanup(r.WaitForCleanup)
|
||||
r.SetSnapshot([]pb.ServerAddress{"s1:1", "s2:1", "s3:1"}, 1)
|
||||
r.SetSnapshot([]pb.ServerAddress{"s1:1", "s2:1", "s3:1", "s4:1"}, 2)
|
||||
|
||||
// Past the cooling interval, the prior owner is no longer offered.
|
||||
time.Sleep(40 * time.Millisecond)
|
||||
for i := 0; i < 2000; i++ {
|
||||
if got := r.PriorOwner(fmt.Sprintf("key-%d", i)); got != "" {
|
||||
t.Fatalf("prior owner should expire after the cooling interval, got %q", got)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -30,6 +30,7 @@ import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/security"
|
||||
stats_collect "github.com/seaweedfs/seaweedfs/weed/stats"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util/grace"
|
||||
)
|
||||
@@ -50,6 +51,8 @@ type AdminOptions struct {
|
||||
dataDir *string
|
||||
icebergPort *int
|
||||
urlPrefix *string
|
||||
metricsHttpPort *int
|
||||
metricsHttpIp *string
|
||||
debug *bool
|
||||
debugPort *int
|
||||
cpuProfile *string
|
||||
@@ -70,6 +73,8 @@ func init() {
|
||||
a.readOnlyPassword = cmdAdmin.Flag.String("readOnlyPassword", "", "read-only user password (optional, for view-only access; requires adminPassword to be set)")
|
||||
a.icebergPort = cmdAdmin.Flag.Int("iceberg.port", 8181, "Iceberg REST Catalog port (0 to hide in UI)")
|
||||
a.urlPrefix = cmdAdmin.Flag.String("urlPrefix", "", "URL path prefix when running behind a reverse proxy under a subdirectory (e.g. /seaweedfs)")
|
||||
a.metricsHttpPort = cmdAdmin.Flag.Int("metricsPort", 0, "Prometheus metrics listen port")
|
||||
a.metricsHttpIp = cmdAdmin.Flag.String("metricsIp", "", "metrics listen ip. If empty, listens on all interfaces.")
|
||||
a.debug = cmdAdmin.Flag.Bool("debug", false, "serves runtime profiling data via pprof on the port specified by -debug.port")
|
||||
a.debugPort = cmdAdmin.Flag.Int("debug.port", 6060, "http port for debugging")
|
||||
a.cpuProfile = cmdAdmin.Flag.String("cpuprofile", "", "cpu profile output file")
|
||||
@@ -160,6 +165,12 @@ var cmdAdmin = &Command{
|
||||
weed admin -debug -debug.port=6060 -master="localhost:9333"
|
||||
weed admin -cpuprofile=cpu.prof -memprofile=mem.prof -master="localhost:9333"
|
||||
|
||||
Metrics:
|
||||
- Use -metricsPort to expose Prometheus metrics at http://<host>:<metricsPort>/metrics
|
||||
- Use -metricsIp to bind the metrics endpoint to a specific ip (default: all interfaces)
|
||||
- Metrics are disabled when -metricsPort is 0 (the default)
|
||||
- Example: weed admin -metricsPort=9327 -master="localhost:9333"
|
||||
|
||||
Configuration File:
|
||||
- The security.toml file is read from ".", "$HOME/.seaweedfs/",
|
||||
"/usr/local/etc/seaweedfs/", or "/etc/seaweedfs/", in that order
|
||||
@@ -257,6 +268,12 @@ func runAdmin(cmd *Command, args []string) bool {
|
||||
}
|
||||
fmt.Printf("Plugin: Enabled\n")
|
||||
|
||||
// Start Prometheus metrics endpoint if a port is configured
|
||||
if *a.metricsHttpPort > 0 {
|
||||
fmt.Printf("Metrics: http://%s/metrics\n", stats_collect.JoinHostPort(*a.metricsHttpIp, *a.metricsHttpPort))
|
||||
}
|
||||
go stats_collect.StartMetricsServer(*a.metricsHttpIp, *a.metricsHttpPort)
|
||||
|
||||
// Set up graceful shutdown
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
|
||||
@@ -236,6 +236,12 @@ func runFiler(cmd *Command, args []string) bool {
|
||||
filerSftpOptions.resolvePaths()
|
||||
util.LoadSecurityConfiguration()
|
||||
|
||||
// Share the S3 static identity config file with the filer regardless of
|
||||
// whether the embedded S3 gateway runs on this node: the IAM gRPC service
|
||||
// the admin UI and weed shell talk to is wired up unconditionally, and it
|
||||
// needs the same identities the S3 server would load from -s3.config.
|
||||
f.s3ConfigFile = filerS3Options.config
|
||||
|
||||
switch {
|
||||
case *f.metricsHttpIp != "":
|
||||
// noting to do, use f.metricsHttpIp
|
||||
|
||||
@@ -439,6 +439,13 @@ func doSubscribeFilerMetaChanges(clientId int32, clientEpoch int32, sourceGrpcDi
|
||||
StartTsNs: sourceFilerOffsetTsNs,
|
||||
StopTsNs: 0,
|
||||
EventErrorType: pb.RetryForeverOnError,
|
||||
// While the source has only read activity it emits no metadata events, so
|
||||
// the watermark above never advances and sync_offset would look stuck.
|
||||
// The idle heartbeat moves the gauge to the source's current time once we
|
||||
// are caught up, so now-sync_offset reflects real lag and stays alertable.
|
||||
OnIdleHeartbeat: func(tsNs int64) {
|
||||
statsCollect.FilerSyncOffsetGauge.WithLabelValues(sourceFiler.String(), targetFiler.String(), clientName, sourcePath).Set(float64(tsNs))
|
||||
},
|
||||
}
|
||||
|
||||
return pb.FollowMetadata(sourceFiler, sourceGrpcDialOption, metadataFollowOption, processEventFnWithOffset)
|
||||
|
||||
@@ -2,6 +2,7 @@ package command
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"io"
|
||||
"io/fs"
|
||||
"os"
|
||||
"path"
|
||||
@@ -9,12 +10,15 @@ import (
|
||||
"strings"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/volume_server_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/backend"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/erasure_coding"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/needle"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/needle_map"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/super_block"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/types"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/volume_info"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
)
|
||||
|
||||
@@ -27,6 +31,7 @@ var cmdFix = &Command{
|
||||
Short: "run weed tool fix on files or whole folders to recreate index file(s) if corrupted",
|
||||
Long: `Fix runs the SeaweedFS fix command on local dat files ( or remote files) or whole folders to re-create the index .idx file. If fixing remote files, you need to synchronize master.toml to the same directory on the current node as on the master node.
|
||||
You Need to stop the volume server when running this command.
|
||||
Use -ecx to rebuild a lost EC index (.ecx) — and the .vif when missing — from the local .ec## shards.
|
||||
`,
|
||||
}
|
||||
|
||||
@@ -36,6 +41,9 @@ var (
|
||||
fixIncludeDeleted = cmdFix.Flag.Bool("includeDeleted", true, "include deleted entries in the index file")
|
||||
fixIgnoreError = cmdFix.Flag.Bool("ignoreError", false, "an optional, if true will be processed despite errors")
|
||||
fixRemoteFile = cmdFix.Flag.Bool("remoteFile", false, "an optional, if true will not try to load the local .dat file, but only the remote file")
|
||||
fixGenerateEcx = cmdFix.Flag.Bool("ecx", false, "regenerate a lost EC index (.ecx) — and the .vif when missing — from the local .ec## shards (missing shards are reconstructed from parity when enough survive). Run with the volume server stopped.")
|
||||
fixEcDataShards = cmdFix.Flag.Int("ecDataShards", 0, "EC data shard count for -ecx (0 = read from .vif, otherwise default 10)")
|
||||
fixEcParityShards = cmdFix.Flag.Int("ecParityShards", 0, "EC parity shard count for -ecx (0 = read from .vif, infer from shard count, otherwise default 4)")
|
||||
)
|
||||
|
||||
type VolumeFileScanner4Fix struct {
|
||||
@@ -130,6 +138,45 @@ func runFix(cmd *Command, args []string) bool {
|
||||
}
|
||||
doFixOneVolume(basePath, baseFileName, collection, volumeId, *fixIncludeDeleted)
|
||||
}
|
||||
|
||||
if *fixGenerateEcx {
|
||||
if !fixEcxFromShardsInDir(basePath, files) {
|
||||
return false
|
||||
}
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// fixEcxFromShardsInDir finds EC volumes in files (identified by their .ec00
|
||||
// data shard) and regenerates the .ecx (and .vif when missing) for each,
|
||||
// honoring the -collection and -volumeId filters.
|
||||
func fixEcxFromShardsInDir(basePath string, files []fs.DirEntry) bool {
|
||||
const shard0Ext = ".ec00"
|
||||
for _, file := range files {
|
||||
if !strings.HasSuffix(file.Name(), shard0Ext) {
|
||||
continue
|
||||
}
|
||||
if *fixVolumeCollection != "" {
|
||||
if !strings.HasPrefix(file.Name(), *fixVolumeCollection+"_") {
|
||||
continue
|
||||
}
|
||||
}
|
||||
baseFileName := file.Name()[:len(file.Name())-len(shard0Ext)]
|
||||
collection, volumeIdStr := "", baseFileName
|
||||
if sepIndex := strings.LastIndex(baseFileName, "_"); sepIndex > 0 {
|
||||
collection = baseFileName[:sepIndex]
|
||||
volumeIdStr = baseFileName[sepIndex+1:]
|
||||
}
|
||||
volumeId, parseErr := strconv.ParseInt(volumeIdStr, 10, 64)
|
||||
if parseErr != nil {
|
||||
fmt.Printf("Failed to parse volume id from %s: %v\n", baseFileName, parseErr)
|
||||
return false
|
||||
}
|
||||
if *fixVolumeId != 0 && *fixVolumeId != volumeId {
|
||||
continue
|
||||
}
|
||||
doFixEcxFromShards(basePath, baseFileName, collection, volumeId)
|
||||
}
|
||||
return true
|
||||
}
|
||||
@@ -202,3 +249,252 @@ func doFixOneVolume(basepath string, baseFileName string, collection string, vol
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// doFixEcxFromShards rebuilds the sealed EC index (.ecx) for one EC volume
|
||||
// directly from its local shards when both the .ecx and the original .dat are
|
||||
// gone but the shards survive. When some data shards are missing but at least
|
||||
// dataShards shards survive in total, the missing shards are first reconstructed
|
||||
// from the survivors via Reed-Solomon. It then de-stripes the data shards into a
|
||||
// temporary .dat, scans the needles, and writes a fresh ascending-sorted .ecx
|
||||
// that matches what WriteSortedFileFromIdx emits at encode time (live entries
|
||||
// only). When the .vif is also missing it is regenerated from the inferred EC
|
||||
// ratio and the .dat size discovered during the scan.
|
||||
func doFixEcxFromShards(basePath, baseFileName, collection string, volumeId int64) {
|
||||
base := path.Join(basePath, baseFileName)
|
||||
|
||||
fail := func(err error) {
|
||||
if *fixIgnoreError {
|
||||
glog.Error(err)
|
||||
} else {
|
||||
glog.Fatal(err)
|
||||
}
|
||||
}
|
||||
|
||||
ecxName := base + ".ecx"
|
||||
if info, err := os.Stat(ecxName); err == nil && info.Size() > 0 {
|
||||
glog.Infof("volume %d: %s already exists (%d bytes), skipping; remove it first to force regeneration", volumeId, ecxName, info.Size())
|
||||
return
|
||||
}
|
||||
|
||||
// Discover which shards are present and their common size. Reed-Solomon
|
||||
// requires every shard to be the same size.
|
||||
present := make([]bool, erasure_coding.MaxShardCount)
|
||||
presentCount := 0
|
||||
maxPresentIdx := -1
|
||||
var shardSize int64
|
||||
for i := 0; i < erasure_coding.MaxShardCount; i++ {
|
||||
info, statErr := os.Stat(base + erasure_coding.ToExt(i))
|
||||
if statErr != nil || info.Size() == 0 {
|
||||
continue
|
||||
}
|
||||
if shardSize == 0 {
|
||||
shardSize = info.Size()
|
||||
} else if info.Size() != shardSize {
|
||||
fail(fmt.Errorf("volume %d: shard %s size %d does not match %d", volumeId, base+erasure_coding.ToExt(i), info.Size(), shardSize))
|
||||
return
|
||||
}
|
||||
present[i] = true
|
||||
presentCount++
|
||||
maxPresentIdx = i
|
||||
}
|
||||
if presentCount == 0 {
|
||||
fail(fmt.Errorf("volume %d: no EC shards found under %s", volumeId, base))
|
||||
return
|
||||
}
|
||||
|
||||
// Resolve the EC ratio and the original .dat size.
|
||||
// Priority: explicit flags > existing .vif > defaults (10+4).
|
||||
vifName := base + ".vif"
|
||||
vifExists := util.FileExists(vifName)
|
||||
dataShards := erasure_coding.DataShardsCount
|
||||
parityShards := erasure_coding.ParityShardsCount
|
||||
var datFileSize int64
|
||||
if vifExists {
|
||||
// MaybeLoadVolumeInfo returns a non-nil error when the .vif exists but
|
||||
// cannot be read or unmarshalled; fail loudly rather than silently
|
||||
// falling back to defaults (which would be wrong for a custom ratio).
|
||||
if vi, _, found, loadErr := volume_info.MaybeLoadVolumeInfo(vifName); loadErr != nil {
|
||||
fail(fmt.Errorf("volume %d: read %s: %w", volumeId, vifName, loadErr))
|
||||
return
|
||||
} else if found && vi != nil {
|
||||
if cfg := vi.GetEcShardConfig(); cfg != nil && cfg.GetDataShards() > 0 {
|
||||
dataShards = int(cfg.GetDataShards())
|
||||
parityShards = int(cfg.GetParityShards())
|
||||
}
|
||||
datFileSize = vi.GetDatFileSize()
|
||||
}
|
||||
}
|
||||
if *fixEcDataShards > 0 {
|
||||
dataShards = *fixEcDataShards
|
||||
}
|
||||
if *fixEcParityShards > 0 {
|
||||
parityShards = *fixEcParityShards
|
||||
}
|
||||
// Ensure the configured total covers every shard index actually present
|
||||
// (a custom-ratio volume with more than the default 14 shards and no .vif).
|
||||
// This never lowers parity below the default, so the common 10+4 case stays
|
||||
// correct for any subset of missing shards.
|
||||
if maxPresentIdx+1 > dataShards+parityShards {
|
||||
parityShards = maxPresentIdx + 1 - dataShards
|
||||
}
|
||||
if dataShards <= 0 || parityShards <= 0 || dataShards+parityShards > erasure_coding.MaxShardCount {
|
||||
fail(fmt.Errorf("volume %d: cannot determine EC ratio (data=%d parity=%d); set -ecDataShards/-ecParityShards", volumeId, dataShards, parityShards))
|
||||
return
|
||||
}
|
||||
|
||||
// Need at least dataShards shards (any data+parity mix) to recover anything.
|
||||
if presentCount < dataShards {
|
||||
fail(fmt.Errorf("volume %d: only %d shards present, need at least %d (data shards) to recover", volumeId, presentCount, dataShards))
|
||||
return
|
||||
}
|
||||
|
||||
// If any data shard is missing, reconstruct the missing shards from the
|
||||
// survivors via Reed-Solomon before de-striping. This writes the rebuilt
|
||||
// shard files back to disk, fully repairing the volume locally.
|
||||
dataComplete := true
|
||||
for i := 0; i < dataShards; i++ {
|
||||
if !present[i] {
|
||||
dataComplete = false
|
||||
break
|
||||
}
|
||||
}
|
||||
if !dataComplete {
|
||||
ctx := &erasure_coding.ECContext{DataShards: dataShards, ParityShards: parityShards}
|
||||
glog.Infof("volume %d: %d/%d shards present; reconstructing missing shards (%s) before index rebuild", volumeId, presentCount, dataShards+parityShards, ctx.String())
|
||||
if _, err := erasure_coding.RebuildEcFilesWithContext(base, ctx); err != nil {
|
||||
fail(fmt.Errorf("volume %d: reconstruct missing shards from %d survivors: %w", volumeId, presentCount, err))
|
||||
return
|
||||
}
|
||||
}
|
||||
|
||||
// Collect the data shards (now all present).
|
||||
shardFileNames := make([]string, dataShards)
|
||||
for i := 0; i < dataShards; i++ {
|
||||
shardPath := base + erasure_coding.ToExt(i)
|
||||
if !util.FileExists(shardPath) {
|
||||
fail(fmt.Errorf("volume %d: data shard %s still missing after reconstruction", volumeId, shardPath))
|
||||
return
|
||||
}
|
||||
shardFileNames[i] = shardPath
|
||||
}
|
||||
|
||||
// Without a recorded original size, reconstruct the fully padded layout; the
|
||||
// scan below detects the trailing zero padding and recovers the true size.
|
||||
reconstructSize := datFileSize
|
||||
if reconstructSize <= 0 {
|
||||
reconstructSize = int64(dataShards) * shardSize
|
||||
glog.V(0).Infof("volume %d: no .dat size in .vif; reconstructing padded .dat (%d bytes) from %d data shards", volumeId, reconstructSize, dataShards)
|
||||
}
|
||||
|
||||
// De-stripe the data shards into a temporary .dat next to the shards.
|
||||
tmpBase := base + ".ecxrecover"
|
||||
tmpDat := tmpBase + ".dat"
|
||||
if err := erasure_coding.WriteDatFile(tmpBase, reconstructSize, shardFileNames); err != nil {
|
||||
os.Remove(tmpDat)
|
||||
fail(fmt.Errorf("volume %d: reconstruct .dat from data shards: %w", volumeId, err))
|
||||
return
|
||||
}
|
||||
defer os.Remove(tmpDat)
|
||||
|
||||
realDatSize, version, err := writeEcxFromDat(tmpDat, ecxName)
|
||||
if err != nil {
|
||||
os.Remove(ecxName)
|
||||
fail(fmt.Errorf("volume %d: build .ecx from reconstructed .dat: %w", volumeId, err))
|
||||
return
|
||||
}
|
||||
glog.Infof("volume %d: wrote %s from %d data shards", volumeId, ecxName, dataShards)
|
||||
|
||||
// Regenerate the .vif when missing so the volume can mount and future
|
||||
// rebuilds know the EC ratio and original .dat size.
|
||||
if !vifExists {
|
||||
size := datFileSize
|
||||
if size <= 0 {
|
||||
size = realDatSize
|
||||
}
|
||||
volumeInfo := &volume_server_pb.VolumeInfo{
|
||||
Version: uint32(version),
|
||||
DatFileSize: size,
|
||||
EcShardConfig: &volume_server_pb.EcShardConfig{
|
||||
DataShards: uint32(dataShards),
|
||||
ParityShards: uint32(parityShards),
|
||||
},
|
||||
}
|
||||
if err := volume_info.SaveVolumeInfo(vifName, volumeInfo); err != nil {
|
||||
fail(fmt.Errorf("volume %d: write %s: %w", volumeId, vifName, err))
|
||||
return
|
||||
}
|
||||
glog.Infof("volume %d: wrote %s (version %d, datFileSize %d, ec %d+%d)", volumeId, vifName, version, size, dataShards, parityShards)
|
||||
}
|
||||
}
|
||||
|
||||
// writeEcxFromDat scans a (reconstructed) .dat and writes an ascending-sorted
|
||||
// .ecx containing only live needles — the same on-disk shape
|
||||
// WriteSortedFileFromIdx produces when an EC volume is first encoded. It returns
|
||||
// the physical .dat size (the offset where the EC zero padding begins) and the
|
||||
// volume version read from the superblock.
|
||||
func writeEcxFromDat(datPath, ecxPath string) (datFileSize int64, version needle.Version, err error) {
|
||||
f, err := os.OpenFile(datPath, os.O_RDONLY, 0644)
|
||||
if err != nil {
|
||||
return 0, 0, fmt.Errorf("open %s: %w", datPath, err)
|
||||
}
|
||||
datBackend := backend.NewDiskFile(f)
|
||||
defer datBackend.Close()
|
||||
|
||||
superBlock, err := super_block.ReadSuperBlock(datBackend)
|
||||
if err != nil {
|
||||
return 0, 0, fmt.Errorf("read superblock: %w", err)
|
||||
}
|
||||
version = superBlock.Version
|
||||
|
||||
fileSize, _, err := datBackend.GetStat()
|
||||
if err != nil {
|
||||
return 0, version, fmt.Errorf("stat %s: %w", datPath, err)
|
||||
}
|
||||
|
||||
nm := needle_map.NewMemDb()
|
||||
defer nm.Close()
|
||||
|
||||
offset := int64(superBlock.BlockSize())
|
||||
for offset < fileSize {
|
||||
n, _, rest, readErr := needle.ReadNeedleHeader(datBackend, version, offset)
|
||||
if readErr != nil {
|
||||
if readErr == io.EOF {
|
||||
break
|
||||
}
|
||||
return 0, version, fmt.Errorf("read needle header at offset %d: %w", offset, readErr)
|
||||
}
|
||||
// EC encoding zero-pads the tail of the last block row. An all-zero
|
||||
// header marks the start of that padding, i.e. the end of real needles.
|
||||
if n.Cookie == 0 && n.Id == 0 && n.Size == 0 {
|
||||
break
|
||||
}
|
||||
if n.Size.IsValid() {
|
||||
if pe := nm.Set(n.Id, types.ToOffset(offset), n.Size); pe != nil {
|
||||
return 0, version, fmt.Errorf("set needle %d: %w", n.Id, pe)
|
||||
}
|
||||
} else {
|
||||
// Deleted/invalid: drop it so the .ecx carries only live entries,
|
||||
// matching the encode-time WriteSortedFileFromIdx behavior.
|
||||
if pe := nm.Delete(n.Id); pe != nil {
|
||||
return 0, version, fmt.Errorf("delete needle %d: %w", n.Id, pe)
|
||||
}
|
||||
}
|
||||
offset += types.NeedleHeaderSize + rest
|
||||
}
|
||||
datFileSize = offset
|
||||
|
||||
ecxFile, err := os.OpenFile(ecxPath, os.O_TRUNC|os.O_CREATE|os.O_WRONLY, 0644)
|
||||
if err != nil {
|
||||
return 0, version, fmt.Errorf("open %s: %w", ecxPath, err)
|
||||
}
|
||||
defer ecxFile.Close()
|
||||
|
||||
if err := nm.AscendingVisit(func(value needle_map.NeedleValue) error {
|
||||
_, writeErr := ecxFile.Write(value.ToBytes())
|
||||
return writeErr
|
||||
}); err != nil {
|
||||
return 0, version, fmt.Errorf("write %s: %w", ecxPath, err)
|
||||
}
|
||||
|
||||
return datFileSize, version, nil
|
||||
}
|
||||
|
||||
@@ -0,0 +1,273 @@
|
||||
package command
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"math/rand"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/volume_server_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/backend"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/erasure_coding"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/needle"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/needle_map"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/super_block"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/types"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/volume_info"
|
||||
)
|
||||
|
||||
// buildAndEncodeTestEcVolume writes a small volume (.dat + .idx), EC-encodes it
|
||||
// into .ec00..ec13, and produces the canonical sorted .ecx. It returns the base
|
||||
// path, the canonical .ecx bytes, and the original .dat size. A couple of
|
||||
// needles are deleted so the .ecx must exclude them (live entries only).
|
||||
func buildAndEncodeTestEcVolume(t *testing.T, dir, baseName string) (base string, canonicalEcx []byte, origDatSize int64) {
|
||||
t.Helper()
|
||||
base = filepath.Join(dir, baseName)
|
||||
version := needle.GetCurrentVersion()
|
||||
|
||||
df, err := os.OpenFile(base+".dat", os.O_RDWR|os.O_CREATE|os.O_TRUNC, 0644)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
datBackend := backend.NewDiskFile(df)
|
||||
sb := super_block.SuperBlock{
|
||||
Version: version,
|
||||
ReplicaPlacement: &super_block.ReplicaPlacement{},
|
||||
Ttl: &needle.TTL{},
|
||||
}
|
||||
if _, err := datBackend.WriteAt(sb.Bytes(), 0); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
nm := needle_map.NewMemDb()
|
||||
for i := uint64(1); i <= 12; i++ {
|
||||
n := new(needle.Needle)
|
||||
n.Id = types.Uint64ToNeedleId(i)
|
||||
n.Data = make([]byte, 200+int(i))
|
||||
rand.Read(n.Data)
|
||||
n.Checksum = needle.NewCRC(n.Data)
|
||||
offset, _, _, err := n.Append(datBackend, version)
|
||||
if err != nil {
|
||||
t.Fatalf("append needle %d: %v", i, err)
|
||||
}
|
||||
// Store n.Size (the on-disk header size), exactly what the volume
|
||||
// server records in its .idx (volume_write.go: nm.Put(..., n.Size)).
|
||||
if err := nm.Set(n.Id, types.ToOffset(int64(offset)), n.Size); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
// Delete ids 3 and 8: append an empty needle (delete record) and drop them
|
||||
// from the index, exactly as the encode-time .ecx would reflect.
|
||||
for _, id := range []uint64{3, 8} {
|
||||
n := new(needle.Needle)
|
||||
n.Id = types.Uint64ToNeedleId(id)
|
||||
if _, _, _, err := n.Append(datBackend, version); err != nil {
|
||||
t.Fatalf("append delete record %d: %v", id, err)
|
||||
}
|
||||
if err := nm.Delete(types.Uint64ToNeedleId(id)); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
}
|
||||
if err := datBackend.Sync(); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
datInfo, err := df.Stat()
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
origDatSize = datInfo.Size()
|
||||
datBackend.Close()
|
||||
|
||||
idxFile, err := os.OpenFile(base+".idx", os.O_WRONLY|os.O_CREATE|os.O_TRUNC, 0644)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := nm.AscendingVisit(func(v needle_map.NeedleValue) error {
|
||||
_, e := idxFile.Write(v.ToBytes())
|
||||
return e
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
idxFile.Close()
|
||||
nm.Close()
|
||||
|
||||
if err := erasure_coding.WriteEcFiles(base); err != nil {
|
||||
t.Fatalf("WriteEcFiles: %v", err)
|
||||
}
|
||||
if err := erasure_coding.WriteSortedFileFromIdx(base, ".ecx"); err != nil {
|
||||
t.Fatalf("WriteSortedFileFromIdx: %v", err)
|
||||
}
|
||||
canonicalEcx, err = os.ReadFile(base + ".ecx")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(canonicalEcx) == 0 {
|
||||
t.Fatal("canonical .ecx is empty")
|
||||
}
|
||||
return base, canonicalEcx, origDatSize
|
||||
}
|
||||
|
||||
// TestFixEcxFromShards verifies the .ecx and .vif are rebuilt purely from the
|
||||
// data shards when every index/metadata file has been lost.
|
||||
func TestFixEcxFromShards(t *testing.T) {
|
||||
oldData, oldParity := *fixEcDataShards, *fixEcParityShards
|
||||
*fixEcDataShards, *fixEcParityShards = 0, 0
|
||||
*fixIgnoreError = true
|
||||
t.Cleanup(func() {
|
||||
*fixIgnoreError = false
|
||||
*fixEcDataShards, *fixEcParityShards = oldData, oldParity
|
||||
})
|
||||
|
||||
dir := t.TempDir()
|
||||
const volumeId = 7
|
||||
base, canonical, origDatSize := buildAndEncodeTestEcVolume(t, dir, "7")
|
||||
|
||||
// Disaster: keep only the shards.
|
||||
for _, ext := range []string{".ecx", ".ecj", ".idx", ".dat", ".vif"} {
|
||||
if err := os.Remove(base + ext); err != nil && !os.IsNotExist(err) {
|
||||
t.Fatalf("remove %s: %v", base+ext, err)
|
||||
}
|
||||
}
|
||||
|
||||
doFixEcxFromShards(dir, "7", "", volumeId)
|
||||
|
||||
recovered, err := os.ReadFile(base + ".ecx")
|
||||
if err != nil {
|
||||
t.Fatalf("recovered .ecx not written: %v", err)
|
||||
}
|
||||
if !bytes.Equal(canonical, recovered) {
|
||||
t.Fatalf(".ecx mismatch: canonical %d bytes, recovered %d bytes", len(canonical), len(recovered))
|
||||
}
|
||||
|
||||
// The reconstructed temporary .dat must not be left behind.
|
||||
if _, err := os.Stat(base + ".ecxrecover.dat"); !os.IsNotExist(err) {
|
||||
t.Fatalf("temporary reconstructed .dat was not cleaned up")
|
||||
}
|
||||
|
||||
// .vif must be regenerated with the default ratio and the original .dat size.
|
||||
vi, _, found, err := volume_info.MaybeLoadVolumeInfo(base + ".vif")
|
||||
if err != nil || !found {
|
||||
t.Fatalf(".vif not regenerated: found=%v err=%v", found, err)
|
||||
}
|
||||
if got := int(vi.GetEcShardConfig().GetDataShards()); got != erasure_coding.DataShardsCount {
|
||||
t.Fatalf("data shards = %d, want %d", got, erasure_coding.DataShardsCount)
|
||||
}
|
||||
if got := int(vi.GetEcShardConfig().GetParityShards()); got != erasure_coding.ParityShardsCount {
|
||||
t.Fatalf("parity shards = %d, want %d", got, erasure_coding.ParityShardsCount)
|
||||
}
|
||||
if vi.GetDatFileSize() != origDatSize {
|
||||
t.Fatalf("dat size = %d, want %d", vi.GetDatFileSize(), origDatSize)
|
||||
}
|
||||
}
|
||||
|
||||
// TestFixEcxFromShardsWithVif verifies that when the .vif survives (recording
|
||||
// the exact .dat size and EC ratio) the .ecx is rebuilt from it and the .vif is
|
||||
// left untouched.
|
||||
func TestFixEcxFromShardsWithVif(t *testing.T) {
|
||||
oldData, oldParity := *fixEcDataShards, *fixEcParityShards
|
||||
*fixEcDataShards, *fixEcParityShards = 0, 0
|
||||
*fixIgnoreError = true
|
||||
t.Cleanup(func() {
|
||||
*fixIgnoreError = false
|
||||
*fixEcDataShards, *fixEcParityShards = oldData, oldParity
|
||||
})
|
||||
|
||||
dir := t.TempDir()
|
||||
const volumeId = 9
|
||||
base, canonical, origDatSize := buildAndEncodeTestEcVolume(t, dir, "9")
|
||||
|
||||
// Write a .vif as the volume server would after encoding.
|
||||
if err := volume_info.SaveVolumeInfo(base+".vif", &volume_server_pb.VolumeInfo{
|
||||
Version: uint32(needle.GetCurrentVersion()),
|
||||
DatFileSize: origDatSize,
|
||||
EcShardConfig: &volume_server_pb.EcShardConfig{
|
||||
DataShards: uint32(erasure_coding.DataShardsCount),
|
||||
ParityShards: uint32(erasure_coding.ParityShardsCount),
|
||||
},
|
||||
}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
vifBefore, err := os.ReadFile(base + ".vif")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
// Lose the index but keep the surviving .vif and shards.
|
||||
for _, ext := range []string{".ecx", ".ecj", ".idx", ".dat"} {
|
||||
if err := os.Remove(base + ext); err != nil && !os.IsNotExist(err) {
|
||||
t.Fatalf("remove %s: %v", base+ext, err)
|
||||
}
|
||||
}
|
||||
|
||||
doFixEcxFromShards(dir, "9", "", volumeId)
|
||||
|
||||
recovered, err := os.ReadFile(base + ".ecx")
|
||||
if err != nil {
|
||||
t.Fatalf("recovered .ecx not written: %v", err)
|
||||
}
|
||||
if !bytes.Equal(canonical, recovered) {
|
||||
t.Fatalf(".ecx mismatch: canonical %d bytes, recovered %d bytes", len(canonical), len(recovered))
|
||||
}
|
||||
|
||||
// An existing .vif must be left untouched.
|
||||
vifAfter, err := os.ReadFile(base + ".vif")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !bytes.Equal(vifBefore, vifAfter) {
|
||||
t.Fatalf("existing .vif was modified")
|
||||
}
|
||||
}
|
||||
|
||||
// TestFixEcxFromShardsMissingShards verifies that when some shards (including a
|
||||
// couple of data shards) are lost but at least dataShards survive, the missing
|
||||
// shards are reconstructed from parity and the .ecx is still rebuilt correctly.
|
||||
func TestFixEcxFromShardsMissingShards(t *testing.T) {
|
||||
oldData, oldParity := *fixEcDataShards, *fixEcParityShards
|
||||
*fixEcDataShards, *fixEcParityShards = 0, 0
|
||||
*fixIgnoreError = true
|
||||
t.Cleanup(func() {
|
||||
*fixIgnoreError = false
|
||||
*fixEcDataShards, *fixEcParityShards = oldData, oldParity
|
||||
})
|
||||
|
||||
dir := t.TempDir()
|
||||
const volumeId = 11
|
||||
base, canonical, origDatSize := buildAndEncodeTestEcVolume(t, dir, "11")
|
||||
|
||||
// Lose every index/metadata file plus three shards (two data: .ec02, .ec05;
|
||||
// one parity: .ec11), keeping 11 of 14 — enough to reconstruct. The highest
|
||||
// shard (.ec13) is kept so the default 10+4 ratio is inferred without a .vif.
|
||||
for _, ext := range []string{".ecx", ".ecj", ".idx", ".dat", ".vif",
|
||||
erasure_coding.ToExt(2), erasure_coding.ToExt(5), erasure_coding.ToExt(11)} {
|
||||
if err := os.Remove(base + ext); err != nil && !os.IsNotExist(err) {
|
||||
t.Fatalf("remove %s: %v", base+ext, err)
|
||||
}
|
||||
}
|
||||
|
||||
doFixEcxFromShards(dir, "11", "", volumeId)
|
||||
|
||||
recovered, err := os.ReadFile(base + ".ecx")
|
||||
if err != nil {
|
||||
t.Fatalf("recovered .ecx not written: %v", err)
|
||||
}
|
||||
if !bytes.Equal(canonical, recovered) {
|
||||
t.Fatalf(".ecx mismatch: canonical %d bytes, recovered %d bytes", len(canonical), len(recovered))
|
||||
}
|
||||
|
||||
// The missing shards must have been reconstructed on disk.
|
||||
for _, idx := range []int{2, 5, 11} {
|
||||
if info, err := os.Stat(base + erasure_coding.ToExt(idx)); err != nil || info.Size() == 0 {
|
||||
t.Fatalf("missing shard %d was not reconstructed: err=%v", idx, err)
|
||||
}
|
||||
}
|
||||
|
||||
vi, _, found, err := volume_info.MaybeLoadVolumeInfo(base + ".vif")
|
||||
if err != nil || !found {
|
||||
t.Fatalf(".vif not regenerated: found=%v err=%v", found, err)
|
||||
}
|
||||
if vi.GetDatFileSize() != origDatSize {
|
||||
t.Fatalf("dat size = %d, want %d", vi.GetDatFileSize(), origDatSize)
|
||||
}
|
||||
}
|
||||
@@ -90,7 +90,6 @@ type miniProgress struct {
|
||||
elapsed map[string]time.Duration
|
||||
isTTY bool
|
||||
rendered bool
|
||||
closed bool
|
||||
}
|
||||
|
||||
const miniProgressNameWidth = 12
|
||||
@@ -141,7 +140,6 @@ func (p *miniProgress) update(name, state string) {
|
||||
func (p *miniProgress) starting(name string) { p.update(name, "starting") }
|
||||
func (p *miniProgress) ready(name string) { p.update(name, "ready") }
|
||||
func (p *miniProgress) failed(name string) { p.update(name, "failed") }
|
||||
func (p *miniProgress) stopping(name string) { p.update(name, "stopping") }
|
||||
func (p *miniProgress) stopped(name string) { p.update(name, "stopped") }
|
||||
|
||||
// renderLocked redraws every row in the board. Caller must hold p.mu and
|
||||
@@ -149,9 +147,6 @@ func (p *miniProgress) stopped(name string) { p.update(name, "stopped") }
|
||||
// ESC[2K clears the line — leaving the cursor parked one line below the last
|
||||
// row so the next print (welcome banner, shutdown messages) flows naturally.
|
||||
func (p *miniProgress) renderLocked() {
|
||||
if p.closed {
|
||||
return
|
||||
}
|
||||
if p.rendered {
|
||||
fmt.Printf("\033[%dA", len(p.order))
|
||||
}
|
||||
@@ -194,14 +189,6 @@ func (p *miniProgress) reset(services []string, initialState string) {
|
||||
}
|
||||
}
|
||||
|
||||
// close prevents further redraws (e.g. before a Fatalf so the panic doesn't
|
||||
// repaint the board). Safe to call multiple times.
|
||||
func (p *miniProgress) close() {
|
||||
p.mu.Lock()
|
||||
defer p.mu.Unlock()
|
||||
p.closed = true
|
||||
}
|
||||
|
||||
// reportMiniStopped marks a service as fully stopped on the progress board.
|
||||
// Safe to call when the board is nil (mini not running) — used from defers
|
||||
// in service goroutines so a normal exit (Ctrl+C, ctx cancel) flips the row.
|
||||
@@ -266,21 +253,6 @@ func resetMiniClients() {
|
||||
miniClients = s
|
||||
}
|
||||
|
||||
// onMiniClientsShutdown runs fn when mini shutdown is triggered, and tracks
|
||||
// it so the interrupt hook can wait for it to drain. No-op outside mini.
|
||||
func onMiniClientsShutdown(fn func()) {
|
||||
s := miniClients
|
||||
if s == nil {
|
||||
return
|
||||
}
|
||||
s.wg.Add(1)
|
||||
go func() {
|
||||
defer s.wg.Done()
|
||||
<-s.ctx.Done()
|
||||
fn()
|
||||
}()
|
||||
}
|
||||
|
||||
// trackMiniClient registers an externally-managed goroutine (one that
|
||||
// observes miniClientsCtx() itself) so the interrupt hook waits for it.
|
||||
// The caller invokes the returned done func when the goroutine exits.
|
||||
|
||||
@@ -61,7 +61,8 @@ type MountOptions struct {
|
||||
|
||||
dirIdleEvictSec *int
|
||||
|
||||
// Distributed lock for cross-mount write coordination
|
||||
// Distributed locking for cross-mount write coordination and POSIX
|
||||
// advisory locks (flock/fcntl)
|
||||
distributedLock *bool
|
||||
|
||||
// POSIX compliance options
|
||||
@@ -152,7 +153,7 @@ func init() {
|
||||
mountReadRetryTime = cmdMount.Flag.Duration("readRetryTime", 6*time.Second, "maximum read retry wait time")
|
||||
|
||||
// Distributed lock for cross-mount write coordination
|
||||
mountOptions.distributedLock = cmdMount.Flag.Bool("dlm", false, "enable distributed lock for cross-mount write coordination (only one mount can write a file at a time)")
|
||||
mountOptions.distributedLock = cmdMount.Flag.Bool("dlm", false, "coordinate writes across mounts (only one mount writes a file at a time) and honor POSIX advisory locks (flock/fcntl) across mounts by routing them to the owner filer")
|
||||
|
||||
// POSIX compliance options
|
||||
mountOptions.posixDirNlink = cmdMount.Flag.Bool("posix.dirNLink", false, "report POSIX-compliant directory nlink (2 + subdirectory count); costs one directory listing per stat")
|
||||
|
||||
@@ -34,6 +34,16 @@ func (store *MemoryStore) LoadConfiguration(ctx context.Context) (*iam_pb.S3ApiC
|
||||
})
|
||||
}
|
||||
|
||||
// Groups are written via CreateGroup / AddUserToGroup / AttachGroupPolicy
|
||||
// (never via SaveConfiguration), so they live in store.groups regardless
|
||||
// of the bulk-config path. Surface them here so consumers that reload
|
||||
// through cm.LoadConfiguration see a faithful snapshot — without this,
|
||||
// iam.groups would never repopulate on memory-store reloads and group
|
||||
// policy evaluation would silently no-op.
|
||||
for _, g := range store.groups {
|
||||
config.Groups = append(config.Groups, cloneGroup(g))
|
||||
}
|
||||
|
||||
return config, nil
|
||||
}
|
||||
|
||||
|
||||
@@ -13,6 +13,7 @@ type Attr struct {
|
||||
Mtime time.Time // time of last modification
|
||||
Crtime time.Time // time of creation (OS X only)
|
||||
Ctime time.Time // time of last inode change
|
||||
Atime time.Time // time of last access
|
||||
Mode os.FileMode // file mode
|
||||
Uid uint32 // owner uid
|
||||
Gid uint32 // group gid
|
||||
|
||||
@@ -85,6 +85,8 @@ func EntryAttributeToPb(entry *Entry) *filer_pb.FuseAttributes {
|
||||
MtimeNs: int32(entry.Attr.Mtime.Nanosecond()),
|
||||
Ctime: entry.Attr.Ctime.Unix(),
|
||||
CtimeNs: int32(entry.Attr.Ctime.Nanosecond()),
|
||||
Atime: atimeSecondsForPb(entry.Attr),
|
||||
AtimeNs: atimeNanosForPb(entry.Attr),
|
||||
FileMode: uint32(entry.Attr.Mode),
|
||||
Uid: entry.Uid,
|
||||
Gid: entry.Gid,
|
||||
@@ -100,6 +102,20 @@ func EntryAttributeToPb(entry *Entry) *filer_pb.FuseAttributes {
|
||||
}
|
||||
}
|
||||
|
||||
func atimeSecondsForPb(attr Attr) int64 {
|
||||
if attr.Atime.IsZero() {
|
||||
return 0
|
||||
}
|
||||
return attr.Atime.Unix()
|
||||
}
|
||||
|
||||
func atimeNanosForPb(attr Attr) int32 {
|
||||
if attr.Atime.IsZero() {
|
||||
return 0
|
||||
}
|
||||
return int32(attr.Atime.Nanosecond())
|
||||
}
|
||||
|
||||
// EntryAttributeToExistingPb fills an existing FuseAttributes to avoid allocation.
|
||||
// Safe to call with nil attr (will return early without populating).
|
||||
func EntryAttributeToExistingPb(entry *Entry, attr *filer_pb.FuseAttributes) {
|
||||
@@ -111,6 +127,8 @@ func EntryAttributeToExistingPb(entry *Entry, attr *filer_pb.FuseAttributes) {
|
||||
attr.MtimeNs = int32(entry.Attr.Mtime.Nanosecond())
|
||||
attr.Ctime = entry.Attr.Ctime.Unix()
|
||||
attr.CtimeNs = int32(entry.Attr.Ctime.Nanosecond())
|
||||
attr.Atime = atimeSecondsForPb(entry.Attr)
|
||||
attr.AtimeNs = atimeNanosForPb(entry.Attr)
|
||||
attr.FileMode = uint32(entry.Attr.Mode)
|
||||
attr.Uid = entry.Uid
|
||||
attr.Gid = entry.Gid
|
||||
@@ -140,6 +158,11 @@ func PbToEntryAttribute(attr *filer_pb.FuseAttributes) Attr {
|
||||
} else {
|
||||
t.Ctime = t.Mtime
|
||||
}
|
||||
if attr.Atime != 0 || attr.AtimeNs != 0 {
|
||||
t.Atime = time.Unix(attr.Atime, int64(attr.AtimeNs))
|
||||
} else {
|
||||
t.Atime = t.Mtime
|
||||
}
|
||||
t.Mode = os.FileMode(attr.FileMode)
|
||||
t.Uid = attr.Uid
|
||||
t.Gid = attr.Gid
|
||||
|
||||
@@ -0,0 +1,65 @@
|
||||
package filer
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
)
|
||||
|
||||
func TestEntryCodec_AtimeRoundTrip(t *testing.T) {
|
||||
mtime := time.Unix(1_700_000_000, 123_456_789)
|
||||
atime := time.Unix(1_700_001_000, 987_654_321)
|
||||
|
||||
source := &Entry{
|
||||
FullPath: util.FullPath("/bucket/object"),
|
||||
Attr: Attr{
|
||||
Mtime: mtime,
|
||||
Ctime: mtime,
|
||||
Atime: atime,
|
||||
},
|
||||
}
|
||||
pb := EntryAttributeToPb(source)
|
||||
if pb.Atime != atime.Unix() {
|
||||
t.Fatalf("expected proto atime %d, got %d", atime.Unix(), pb.Atime)
|
||||
}
|
||||
if pb.AtimeNs != int32(atime.Nanosecond()) {
|
||||
t.Fatalf("expected proto atime_ns %d, got %d", atime.Nanosecond(), pb.AtimeNs)
|
||||
}
|
||||
|
||||
decoded := PbToEntryAttribute(pb)
|
||||
if !decoded.Atime.Equal(atime) {
|
||||
t.Fatalf("expected decoded atime %v, got %v", atime, decoded.Atime)
|
||||
}
|
||||
}
|
||||
|
||||
func TestEntryCodec_AtimeZeroFallsBackToMtime(t *testing.T) {
|
||||
mtime := time.Unix(1_700_000_000, 0)
|
||||
pb := &filer_pb.FuseAttributes{
|
||||
Mtime: mtime.Unix(),
|
||||
MtimeNs: int32(mtime.Nanosecond()),
|
||||
}
|
||||
decoded := PbToEntryAttribute(pb)
|
||||
if !decoded.Atime.Equal(mtime) {
|
||||
t.Fatalf("expected atime to fall back to mtime %v, got %v", mtime, decoded.Atime)
|
||||
}
|
||||
}
|
||||
|
||||
// Atime in the first second of the unix epoch encodes as Atime=0 with
|
||||
// AtimeNs>0; the decode path must treat that as a valid timestamp rather than
|
||||
// falling back to Mtime.
|
||||
func TestEntryCodec_AtimeSubSecondEpochPreserved(t *testing.T) {
|
||||
mtime := time.Unix(1_700_000_000, 0)
|
||||
pb := &filer_pb.FuseAttributes{
|
||||
Mtime: mtime.Unix(),
|
||||
MtimeNs: int32(mtime.Nanosecond()),
|
||||
Atime: 0,
|
||||
AtimeNs: 500_000,
|
||||
}
|
||||
decoded := PbToEntryAttribute(pb)
|
||||
want := time.Unix(0, 500_000)
|
||||
if !decoded.Atime.Equal(want) {
|
||||
t.Fatalf("expected sub-second-epoch atime %v, got %v", want, decoded.Atime)
|
||||
}
|
||||
}
|
||||
+23
-2
@@ -209,7 +209,11 @@ func (f *Filer) RollbackTransaction(ctx context.Context) error {
|
||||
return f.Store.RollbackTransaction(ctx)
|
||||
}
|
||||
|
||||
func (f *Filer) CreateEntry(ctx context.Context, entry *Entry, o_excl bool, isFromOtherCluster bool, signatures []int32, skipCreateParentDir bool, maxFilenameLength uint32) error {
|
||||
// CreateEntry creates or replaces an entry. When existing is non-nil the caller
|
||||
// has already fetched the current entry at this path under a path lock, and it
|
||||
// is reused instead of looking the store up again; pass nil to have CreateEntry
|
||||
// look it up itself.
|
||||
func (f *Filer) CreateEntry(ctx context.Context, entry *Entry, existing *Entry, o_excl bool, isFromOtherCluster bool, signatures []int32, skipCreateParentDir bool, maxFilenameLength uint32) error {
|
||||
|
||||
if string(entry.FullPath) == "/" {
|
||||
return nil
|
||||
@@ -223,7 +227,14 @@ func (f *Filer) CreateEntry(ctx context.Context, entry *Entry, o_excl bool, isFr
|
||||
entry.Attr.TtlSec = 0
|
||||
}
|
||||
|
||||
oldEntry, _ := f.FindEntry(ctx, entry.FullPath)
|
||||
if entry.Attr.Atime.IsZero() {
|
||||
entry.Attr.Atime = entryInitialAtime(entry.Attr)
|
||||
}
|
||||
|
||||
oldEntry := existing
|
||||
if oldEntry == nil {
|
||||
oldEntry, _ = f.FindEntry(ctx, entry.FullPath)
|
||||
}
|
||||
|
||||
/*
|
||||
if !hasWritePermission(lastDirectoryEntry, entry) {
|
||||
@@ -371,9 +382,19 @@ func (f *Filer) UpdateEntry(ctx context.Context, oldEntry, entry *Entry) (err er
|
||||
return fmt.Errorf("%s: %w", oldEntry.FullPath, filer_pb.ErrExistingIsFile)
|
||||
}
|
||||
}
|
||||
if entry.Attr.Atime.IsZero() {
|
||||
entry.Attr.Atime = entryInitialAtime(entry.Attr)
|
||||
}
|
||||
return f.Store.UpdateEntry(ctx, entry)
|
||||
}
|
||||
|
||||
func entryInitialAtime(attr Attr) time.Time {
|
||||
if !attr.Mtime.IsZero() {
|
||||
return attr.Mtime
|
||||
}
|
||||
return attr.Crtime
|
||||
}
|
||||
|
||||
var (
|
||||
Root = &Entry{
|
||||
FullPath: "/",
|
||||
|
||||
@@ -28,7 +28,7 @@ func TestCreateEntryAssignsInodeWhenMissing(t *testing.T) {
|
||||
},
|
||||
}
|
||||
|
||||
err := f.CreateEntry(context.Background(), entry, false, false, nil, false, f.MaxFilenameLength)
|
||||
err := f.CreateEntry(context.Background(), entry, nil, false, false, nil, false, f.MaxFilenameLength)
|
||||
require.NoError(t, err)
|
||||
|
||||
stored, findErr := store.FindEntry(context.Background(), entry.FullPath)
|
||||
@@ -48,7 +48,7 @@ func TestCreateEntryAssignsInodesToAutoCreatedParents(t *testing.T) {
|
||||
},
|
||||
}
|
||||
|
||||
err := f.CreateEntry(context.Background(), entry, false, false, nil, false, f.MaxFilenameLength)
|
||||
err := f.CreateEntry(context.Background(), entry, nil, false, false, nil, false, f.MaxFilenameLength)
|
||||
require.NoError(t, err)
|
||||
|
||||
for _, path := range []string{"/a", "/a/b", "/a/b/c.txt"} {
|
||||
|
||||
@@ -96,7 +96,7 @@ func (f *Filer) maybeLazyFetchFromRemote(ctx context.Context, p util.FullPath) (
|
||||
persistBaseCtx, cancelPersist := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
defer cancelPersist()
|
||||
persistCtx := context.WithValue(persistBaseCtx, lazyFetchContextKey{}, true)
|
||||
saveErr := f.CreateEntry(persistCtx, entry, false, false, nil, true, f.MaxFilenameLength)
|
||||
saveErr := f.CreateEntry(persistCtx, entry, nil, false, false, nil, true, f.MaxFilenameLength)
|
||||
if saveErr != nil {
|
||||
glog.Warningf("maybeLazyFetchFromRemote: failed to persist filer entry for %s: %v", p, saveErr)
|
||||
f.lazyFetchGroup.Forget(key)
|
||||
|
||||
@@ -152,7 +152,7 @@ func (f *Filer) maybeLazyListFromRemote(ctx context.Context, p util.FullPath) {
|
||||
entry.Attr.FileSize = uint64(remoteEntry.RemoteSize)
|
||||
}
|
||||
}
|
||||
if saveErr := f.CreateEntry(persistCtx, entry, false, false, nil, true, f.MaxFilenameLength); saveErr != nil {
|
||||
if saveErr := f.CreateEntry(persistCtx, entry, nil, false, false, nil, true, f.MaxFilenameLength); saveErr != nil {
|
||||
glog.Warningf("maybeLazyListFromRemote: persist %s: %v", childPath, saveErr)
|
||||
}
|
||||
}
|
||||
@@ -193,7 +193,7 @@ func (f *Filer) updateDirectoryListingSyncedAt(ctx context.Context, p util.FullP
|
||||
dirEntry.Extended = make(map[string][]byte)
|
||||
}
|
||||
dirEntry.Extended[xattrRemoteListingSyncedAt] = []byte(fmt.Sprintf("%d", syncTime.Unix()))
|
||||
if saveErr := f.CreateEntry(ctx, dirEntry, false, false, nil, true, f.MaxFilenameLength); saveErr != nil {
|
||||
if saveErr := f.CreateEntry(ctx, dirEntry, nil, false, false, nil, true, f.MaxFilenameLength); saveErr != nil {
|
||||
glog.Warningf("maybeLazyListFromRemote: create dir synced_at for %s: %v", p, saveErr)
|
||||
}
|
||||
return
|
||||
|
||||
@@ -43,7 +43,7 @@ func (f *Filer) appendToFile(targetFile string, data []byte) error {
|
||||
entry.Chunks = append(entry.GetChunks(), uploadResult.ToPbFileChunk(assignResult.Fid, offset, time.Now().UnixNano()))
|
||||
|
||||
// update the entry
|
||||
err = f.CreateEntry(context.Background(), entry, false, false, nil, false, f.MaxFilenameLength)
|
||||
err = f.CreateEntry(context.Background(), entry, nil, false, false, nil, false, f.MaxFilenameLength)
|
||||
|
||||
return err
|
||||
}
|
||||
|
||||
@@ -32,7 +32,7 @@ func TestCreateAndFind(t *testing.T) {
|
||||
},
|
||||
}
|
||||
|
||||
if err := testFiler.CreateEntry(ctx, entry1, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
|
||||
if err := testFiler.CreateEntry(ctx, entry1, nil, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
|
||||
t.Errorf("create entry %v: %v", entry1.FullPath, err)
|
||||
return
|
||||
}
|
||||
|
||||
@@ -29,7 +29,7 @@ func TestCreateAndFind(t *testing.T) {
|
||||
},
|
||||
}
|
||||
|
||||
if err := testFiler.CreateEntry(ctx, entry1, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
|
||||
if err := testFiler.CreateEntry(ctx, entry1, nil, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
|
||||
t.Errorf("create entry %v: %v", entry1.FullPath, err)
|
||||
return
|
||||
}
|
||||
|
||||
@@ -29,7 +29,7 @@ func TestCreateAndFind(t *testing.T) {
|
||||
},
|
||||
}
|
||||
|
||||
if err := testFiler.CreateEntry(ctx, entry1, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
|
||||
if err := testFiler.CreateEntry(ctx, entry1, nil, false, false, nil, false, testFiler.MaxFilenameLength); err != nil {
|
||||
t.Errorf("create entry %v: %v", entry1.FullPath, err)
|
||||
return
|
||||
}
|
||||
|
||||
@@ -33,6 +33,8 @@ func (store *MysqlStore) GetName() string {
|
||||
}
|
||||
|
||||
func (store *MysqlStore) Initialize(configuration util.Configuration, prefix string) (err error) {
|
||||
// Absent key keeps a pooled default; an explicit 0 disables the idle pool.
|
||||
configuration.SetDefault(prefix+"connection_max_idle", 2)
|
||||
return store.initialize(
|
||||
configuration.GetString(prefix+"dsn"),
|
||||
configuration.GetString(prefix+"upsertQuery"),
|
||||
|
||||
@@ -33,6 +33,8 @@ func (store *MysqlStore2) GetName() string {
|
||||
}
|
||||
|
||||
func (store *MysqlStore2) Initialize(configuration util.Configuration, prefix string) (err error) {
|
||||
// Absent key keeps a pooled default; an explicit 0 disables the idle pool.
|
||||
configuration.SetDefault(prefix+"connection_max_idle", 2)
|
||||
return store.initialize(
|
||||
configuration.GetString(prefix+"createTable"),
|
||||
configuration.GetString(prefix+"upsertQuery"),
|
||||
|
||||
@@ -0,0 +1,259 @@
|
||||
package posixlock
|
||||
|
||||
import (
|
||||
"sync"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Manager is the owner filer's in-memory authority for POSIX advisory locks
|
||||
// across inodes. Lock state lives here, not in replicated metadata: it is
|
||||
// transient coordination, so keeping it out of the meta-log avoids churn and
|
||||
// does not pollute what subscribers see (the distributed lock manager holds its
|
||||
// locks the same way). A `Set` per inode key, plus a session index so a dead
|
||||
// mount's locks are reaped in O(locks held) rather than by scanning every inode.
|
||||
//
|
||||
// key is an opaque inode identity supplied by the caller — the file's path, or
|
||||
// "hl:"+hex(HardLinkId) for a hardlinked inode — so all names of one inode share
|
||||
// a Set. The Manager is safe for concurrent use.
|
||||
type Manager struct {
|
||||
mu sync.Mutex
|
||||
byKey map[string]*Set // inode key -> held locks
|
||||
bySid map[uint64]map[string]bool // session -> keys it currently holds locks on
|
||||
lastSeen map[uint64]time.Time // session -> last keepalive; only renewing sessions are leased
|
||||
}
|
||||
|
||||
func NewManager() *Manager {
|
||||
return &Manager{
|
||||
byKey: make(map[string]*Set),
|
||||
bySid: make(map[uint64]map[string]bool),
|
||||
lastSeen: make(map[uint64]time.Time),
|
||||
}
|
||||
}
|
||||
|
||||
// Renew records a keepalive from a session, placing it under lease management.
|
||||
// Only sessions that have renewed are subject to ReapExpired, so a session that
|
||||
// never sends keepalives (e.g. before the mount keepalive exists) is never reaped.
|
||||
func (m *Manager) Renew(sid uint64) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
m.lastSeen[sid] = time.Now()
|
||||
}
|
||||
|
||||
// ReapExpired releases the locks of every leased session whose last keepalive is
|
||||
// older than ttl — a dead or partitioned mount. Sessions that never renewed are
|
||||
// left untouched. Returns the reaped session ids.
|
||||
func (m *Manager) ReapExpired(ttl time.Duration) []uint64 {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
cutoff := time.Now().Add(-ttl)
|
||||
var reaped []uint64
|
||||
for sid, seen := range m.lastSeen {
|
||||
if seen.After(cutoff) {
|
||||
continue
|
||||
}
|
||||
for key := range m.bySid[sid] {
|
||||
s := m.byKey[key]
|
||||
if s == nil {
|
||||
continue
|
||||
}
|
||||
s.ReleaseSession(sid)
|
||||
if s.Empty() {
|
||||
delete(m.byKey, key)
|
||||
}
|
||||
}
|
||||
delete(m.bySid, sid)
|
||||
delete(m.lastSeen, sid)
|
||||
reaped = append(reaped, sid)
|
||||
}
|
||||
return reaped
|
||||
}
|
||||
|
||||
// TryLock grants lk on key, or returns the conflicting lock and false. The set
|
||||
// is created on first use and dropped again when it empties.
|
||||
func (m *Manager) TryLock(key string, lk Range) (Range, bool) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
s, ok := m.byKey[key]
|
||||
if !ok {
|
||||
s = &Set{}
|
||||
}
|
||||
if c, granted := s.Acquire(lk); !granted {
|
||||
return c, false
|
||||
}
|
||||
if !ok {
|
||||
m.byKey[key] = s
|
||||
}
|
||||
m.index(lk.Sid, key)
|
||||
return Range{}, true
|
||||
}
|
||||
|
||||
// Track records a lock the server already granted, without arbitration. A mount
|
||||
// uses it to mirror its own held locks so it can re-assert them to the inode's
|
||||
// current owner filer after an ownership change or owner restart.
|
||||
func (m *Manager) Track(key string, lk Range) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
s, ok := m.byKey[key]
|
||||
if !ok {
|
||||
s = &Set{}
|
||||
m.byKey[key] = s
|
||||
}
|
||||
s.Grant(lk)
|
||||
m.index(lk.Sid, key)
|
||||
}
|
||||
|
||||
// Snapshot returns a copy of the held locks per key. A mount calls it to drive
|
||||
// re-assertion keepalives; the filer never does.
|
||||
func (m *Manager) Snapshot() map[string][]Range {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
out := make(map[string][]Range, len(m.byKey))
|
||||
for key, s := range m.byKey {
|
||||
out[key] = append([]Range(nil), s.locks...)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// Reassert rebuilds session sid's locks on key from the client's authoritative
|
||||
// list, renewing the lease. It replaces sid's existing locks on the key, then
|
||||
// re-acquires each asserted lock — arbitrating against other sessions so it never
|
||||
// double-grants. Locks that lost to another session in a migration window are
|
||||
// returned as conflicts. The owner filer calls this on a re-assertion keepalive.
|
||||
func (m *Manager) Reassert(key string, sid uint64, locks []Range) (conflicts []Range) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
m.lastSeen[sid] = time.Now()
|
||||
|
||||
s := m.byKey[key]
|
||||
if s == nil {
|
||||
if len(locks) == 0 {
|
||||
return nil
|
||||
}
|
||||
s = &Set{}
|
||||
m.byKey[key] = s
|
||||
}
|
||||
s.ReleaseSession(sid)
|
||||
for _, lk := range locks {
|
||||
lk.Sid = sid
|
||||
if c, granted := s.Acquire(lk); !granted {
|
||||
conflicts = append(conflicts, c)
|
||||
}
|
||||
}
|
||||
if setHasSession(s, sid) {
|
||||
m.index(sid, key)
|
||||
} else {
|
||||
m.deindex(sid, key)
|
||||
}
|
||||
if s.Empty() {
|
||||
delete(m.byKey, key)
|
||||
}
|
||||
return conflicts
|
||||
}
|
||||
|
||||
// Unlock releases lk's owner's locks within its namespace over lk's range.
|
||||
func (m *Manager) Unlock(key string, lk Range) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
s := m.byKey[key]
|
||||
if s == nil {
|
||||
return
|
||||
}
|
||||
s.Release(lk)
|
||||
m.afterRelease(key, s, lk.Sid)
|
||||
}
|
||||
|
||||
// GetLk reports the lock that would block proposed on key, if any.
|
||||
func (m *Manager) GetLk(key string, proposed Range) (Range, bool) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
s := m.byKey[key]
|
||||
if s == nil {
|
||||
return Range{}, false
|
||||
}
|
||||
return s.Conflict(proposed)
|
||||
}
|
||||
|
||||
// ReleasePosixOwner drops (sid, owner)'s fcntl locks on key — the flush-time path.
|
||||
func (m *Manager) ReleasePosixOwner(key string, sid, owner uint64) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
s := m.byKey[key]
|
||||
if s == nil {
|
||||
return
|
||||
}
|
||||
s.ReleasePosixOwner(sid, owner)
|
||||
m.afterRelease(key, s, sid)
|
||||
}
|
||||
|
||||
// ReleaseFlockOwner drops (sid, owner)'s flock locks on key — the release-time path.
|
||||
func (m *Manager) ReleaseFlockOwner(key string, sid, owner uint64) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
s := m.byKey[key]
|
||||
if s == nil {
|
||||
return
|
||||
}
|
||||
s.ReleaseFlockOwner(sid, owner)
|
||||
m.afterRelease(key, s, sid)
|
||||
}
|
||||
|
||||
// ReleaseSession drops every lock held by a session across all inodes it touched,
|
||||
// reaping a mount whose lease expired. O(locks held by the session).
|
||||
func (m *Manager) ReleaseSession(sid uint64) {
|
||||
m.mu.Lock()
|
||||
defer m.mu.Unlock()
|
||||
for key := range m.bySid[sid] {
|
||||
s := m.byKey[key]
|
||||
if s == nil {
|
||||
continue
|
||||
}
|
||||
s.ReleaseSession(sid)
|
||||
if s.Empty() {
|
||||
delete(m.byKey, key)
|
||||
}
|
||||
}
|
||||
delete(m.bySid, sid)
|
||||
}
|
||||
|
||||
// afterRelease prunes the session index when sid no longer holds any lock on key,
|
||||
// and drops the set when it empties. Only sid's presence can have changed, since
|
||||
// a release only removes sid's locks.
|
||||
func (m *Manager) afterRelease(key string, s *Set, sid uint64) {
|
||||
if s.Empty() {
|
||||
delete(m.byKey, key)
|
||||
m.deindex(sid, key)
|
||||
return
|
||||
}
|
||||
if !setHasSession(s, sid) {
|
||||
m.deindex(sid, key)
|
||||
}
|
||||
}
|
||||
|
||||
func (m *Manager) index(sid uint64, key string) {
|
||||
keys := m.bySid[sid]
|
||||
if keys == nil {
|
||||
keys = make(map[string]bool)
|
||||
m.bySid[sid] = keys
|
||||
}
|
||||
keys[key] = true
|
||||
}
|
||||
|
||||
func (m *Manager) deindex(sid uint64, key string) {
|
||||
keys := m.bySid[sid]
|
||||
if keys == nil {
|
||||
return
|
||||
}
|
||||
delete(keys, key)
|
||||
if len(keys) == 0 {
|
||||
delete(m.bySid, sid)
|
||||
}
|
||||
}
|
||||
|
||||
func setHasSession(s *Set, sid uint64) bool {
|
||||
for _, l := range s.Locks() {
|
||||
if l.Sid == sid {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
@@ -0,0 +1,194 @@
|
||||
package posixlock
|
||||
|
||||
import (
|
||||
"math"
|
||||
"runtime"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
func TestManagerGrantAndConflict(t *testing.T) {
|
||||
m := NewManager()
|
||||
if _, granted := m.TryLock("a", Range{Start: 0, End: 99, Type: Write, Sid: 1, Owner: 1}); !granted {
|
||||
t.Fatal("first lock should be granted")
|
||||
}
|
||||
if c, granted := m.TryLock("a", Range{Start: 50, End: 149, Type: Write, Sid: 2, Owner: 1}); granted {
|
||||
t.Fatalf("overlapping lock from another session should conflict, got grant; conflict=%+v", c)
|
||||
}
|
||||
// A different key is independent.
|
||||
if _, granted := m.TryLock("b", Range{Start: 0, End: 99, Type: Write, Sid: 2, Owner: 1}); !granted {
|
||||
t.Fatal("lock on a different key should be granted")
|
||||
}
|
||||
}
|
||||
|
||||
func TestManagerUnlockCleansEmptyKeyAndIndex(t *testing.T) {
|
||||
m := NewManager()
|
||||
lk := Range{Start: 0, End: 99, Type: Write, Sid: 1, Owner: 1}
|
||||
m.TryLock("a", lk)
|
||||
|
||||
if !m.bySid[1]["a"] {
|
||||
t.Fatal("session index should record the held key")
|
||||
}
|
||||
m.Unlock("a", Range{Start: 0, End: 99, Type: Unlock, Sid: 1, Owner: 1})
|
||||
|
||||
if _, ok := m.byKey["a"]; ok {
|
||||
t.Fatal("empty set should be dropped from byKey")
|
||||
}
|
||||
if _, ok := m.bySid[1]; ok {
|
||||
t.Fatal("session index should be pruned when it holds nothing")
|
||||
}
|
||||
}
|
||||
|
||||
func TestManagerPartialUnlockKeepsIndex(t *testing.T) {
|
||||
m := NewManager()
|
||||
m.TryLock("a", Range{Start: 0, End: 49, Type: Write, Sid: 1, Owner: 1})
|
||||
m.TryLock("a", Range{Start: 100, End: 149, Type: Write, Sid: 1, Owner: 1})
|
||||
// Release one of the two ranges; the session still holds the other.
|
||||
m.Unlock("a", Range{Start: 0, End: 49, Type: Unlock, Sid: 1, Owner: 1})
|
||||
|
||||
if !m.bySid[1]["a"] {
|
||||
t.Fatal("session still holds a lock on the key; index must remain")
|
||||
}
|
||||
if _, ok := m.byKey["a"]; !ok {
|
||||
t.Fatal("key should remain while a lock is held")
|
||||
}
|
||||
}
|
||||
|
||||
func TestManagerGetLk(t *testing.T) {
|
||||
m := NewManager()
|
||||
m.TryLock("a", Range{Start: 10, End: 50, Type: Write, Sid: 1, Owner: 1, Pid: 7})
|
||||
c, found := m.GetLk("a", Range{Start: 30, End: 70, Type: Read, Sid: 2, Owner: 1})
|
||||
if !found || c.Pid != 7 {
|
||||
t.Fatalf("expected conflict from pid 7, got %+v found=%v", c, found)
|
||||
}
|
||||
if _, found := m.GetLk("missing", Range{Start: 0, End: 1, Type: Write, Sid: 9, Owner: 9}); found {
|
||||
t.Fatal("missing key should report no conflict")
|
||||
}
|
||||
}
|
||||
|
||||
func TestManagerReleasePosixOwnerKeepsFlockAndIndex(t *testing.T) {
|
||||
m := NewManager()
|
||||
m.TryLock("a", Range{Start: 0, End: 99, Type: Write, Sid: 1, Owner: 1})
|
||||
m.TryLock("a", Range{Start: 0, End: math.MaxUint64, Type: Write, Sid: 1, Owner: 1, IsFlock: true})
|
||||
|
||||
m.ReleasePosixOwner("a", 1, 1)
|
||||
|
||||
// flock lock for the same session remains, so the index must remain too.
|
||||
if !m.bySid[1]["a"] {
|
||||
t.Fatal("session still holds the flock lock; index must remain")
|
||||
}
|
||||
if _, found := m.GetLk("a", Range{Start: 0, End: 10, Type: Write, Sid: 2, Owner: 2, IsFlock: true}); !found {
|
||||
t.Fatal("flock lock should survive ReleasePosixOwner")
|
||||
}
|
||||
if _, found := m.GetLk("a", Range{Start: 0, End: 10, Type: Write, Sid: 2, Owner: 2}); found {
|
||||
t.Fatal("fcntl lock should be gone after ReleasePosixOwner")
|
||||
}
|
||||
}
|
||||
|
||||
func TestManagerReleaseSessionReapsAcrossKeys(t *testing.T) {
|
||||
m := NewManager()
|
||||
m.TryLock("a", Range{Start: 0, End: 99, Type: Write, Sid: 1, Owner: 1})
|
||||
m.TryLock("b", Range{Start: 0, End: 99, Type: Write, Sid: 1, Owner: 2})
|
||||
m.TryLock("b", Range{Start: 200, End: 299, Type: Write, Sid: 2, Owner: 1})
|
||||
|
||||
m.ReleaseSession(1)
|
||||
|
||||
if _, ok := m.bySid[1]; ok {
|
||||
t.Fatal("reaped session should be gone from the index")
|
||||
}
|
||||
if _, ok := m.byKey["a"]; ok {
|
||||
t.Fatal("key a held only session 1's lock and should be dropped")
|
||||
}
|
||||
// Session 2's lock on b survives.
|
||||
if _, found := m.GetLk("b", Range{Start: 200, End: 299, Type: Write, Sid: 9, Owner: 9}); !found {
|
||||
t.Fatal("session 2's lock on b should remain after reaping session 1")
|
||||
}
|
||||
if !m.bySid[2]["b"] {
|
||||
t.Fatal("session 2 index entry should remain")
|
||||
}
|
||||
}
|
||||
|
||||
func TestManagerReapsOnlyStaleLeasedSessions(t *testing.T) {
|
||||
m := NewManager()
|
||||
// Session 1: holds a lock, leased but stale (renewed long ago).
|
||||
m.TryLock("a", Range{Start: 0, End: 99, Type: Write, Sid: 1, Owner: 1})
|
||||
m.Renew(1)
|
||||
m.lastSeen[1] = time.Now().Add(-time.Hour)
|
||||
// Session 2: holds a lock, leased and fresh.
|
||||
m.TryLock("b", Range{Start: 0, End: 99, Type: Write, Sid: 2, Owner: 1})
|
||||
m.Renew(2)
|
||||
// Session 3: holds a lock but never renewed (no lease) — must not be reaped.
|
||||
m.TryLock("c", Range{Start: 0, End: 99, Type: Write, Sid: 3, Owner: 1})
|
||||
|
||||
reaped := m.ReapExpired(30 * time.Second)
|
||||
|
||||
if len(reaped) != 1 || reaped[0] != 1 {
|
||||
t.Fatalf("only the stale leased session should be reaped, got %v", reaped)
|
||||
}
|
||||
if _, ok := m.byKey["a"]; ok {
|
||||
t.Fatal("stale session's lock should be gone")
|
||||
}
|
||||
if _, ok := m.byKey["b"]; !ok {
|
||||
t.Fatal("fresh session's lock must remain")
|
||||
}
|
||||
if _, ok := m.byKey["c"]; !ok {
|
||||
t.Fatal("never-renewed session must not be reaped")
|
||||
}
|
||||
if _, ok := m.lastSeen[1]; ok {
|
||||
t.Fatal("reaped session's lease entry should be cleared")
|
||||
}
|
||||
}
|
||||
|
||||
// Mutual exclusion under concurrent whole-file flock churn through the Manager:
|
||||
// at most one owner may believe it holds the exclusive lock at any instant.
|
||||
func TestManagerConcurrentFlockMutualExclusion(t *testing.T) {
|
||||
m := NewManager()
|
||||
const (
|
||||
key = "inode"
|
||||
workers = 16
|
||||
iters = 400
|
||||
)
|
||||
var (
|
||||
wg sync.WaitGroup
|
||||
holder atomic.Int64
|
||||
overlap atomic.Int32
|
||||
)
|
||||
for w := 0; w < workers; w++ {
|
||||
wg.Add(1)
|
||||
go func(id int) {
|
||||
defer wg.Done()
|
||||
lk := Range{Start: 0, End: math.MaxUint64, Type: Write, Sid: uint64(id + 1), Owner: 1, IsFlock: true}
|
||||
unlock := lk
|
||||
unlock.Type = Unlock
|
||||
token := int64(id + 1)
|
||||
for i := 0; i < iters; i++ {
|
||||
for {
|
||||
if _, granted := m.TryLock(key, lk); granted {
|
||||
break
|
||||
}
|
||||
runtime.Gosched()
|
||||
}
|
||||
if prev := holder.Swap(token); prev != 0 {
|
||||
overlap.Add(1)
|
||||
}
|
||||
runtime.Gosched()
|
||||
if !holder.CompareAndSwap(token, 0) {
|
||||
overlap.Add(1)
|
||||
}
|
||||
m.Unlock(key, unlock)
|
||||
}
|
||||
}(w)
|
||||
}
|
||||
wg.Wait()
|
||||
if n := overlap.Load(); n != 0 {
|
||||
t.Fatalf("mutual exclusion violated %d times", n)
|
||||
}
|
||||
if len(m.byKey) != 0 {
|
||||
t.Fatalf("all locks released; byKey should be empty, got %d", len(m.byKey))
|
||||
}
|
||||
if len(m.bySid) != 0 {
|
||||
t.Fatalf("all locks released; bySid should be empty, got %d", len(m.bySid))
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,222 @@
|
||||
// Package posixlock implements the conflict, coalescing, and range-split logic
|
||||
// for POSIX advisory file locks — fcntl byte-range and flock whole-file — as a
|
||||
// pure per-inode lock set with no concurrency control of its own.
|
||||
//
|
||||
// It is the server-side authority for distributed FUSE locking: the owner filer
|
||||
// for an inode holds one Set and serializes access to it under that inode's
|
||||
// per-path lock, so each operation runs to completion without concurrent
|
||||
// mutation. The same algorithm can back the per-mount table; blocking (SetLkw)
|
||||
// and any wait queue belong to the caller, not here.
|
||||
package posixlock
|
||||
|
||||
import (
|
||||
"math"
|
||||
"sort"
|
||||
)
|
||||
|
||||
// Lock types, kept independent of the platform syscall package (whose F_RDLCK /
|
||||
// F_WRLCK / F_UNLCK values differ per OS and are absent on some) so the filer
|
||||
// builds everywhere. Callers map the syscall constants onto these at the edge.
|
||||
// Zero is intentionally unused so a zero-value Range reads as "unset".
|
||||
const (
|
||||
Read uint32 = 1
|
||||
Write uint32 = 2
|
||||
Unlock uint32 = 3
|
||||
)
|
||||
|
||||
// Range is one held advisory byte-range lock. Owner identity is the (Sid, Owner)
|
||||
// pair: Sid is the mount session, Owner the FUSE lock owner within it, so owners
|
||||
// from different mounts never alias. End is inclusive; math.MaxUint64 means EOF.
|
||||
// IsFlock separates the flock and fcntl namespaces, which never conflict.
|
||||
type Range struct {
|
||||
Start uint64
|
||||
End uint64
|
||||
Type uint32
|
||||
Sid uint64
|
||||
Owner uint64
|
||||
Pid uint32
|
||||
IsFlock bool
|
||||
}
|
||||
|
||||
func (r Range) sameOwner(o Range) bool {
|
||||
return r.Sid == o.Sid && r.Owner == o.Owner
|
||||
}
|
||||
|
||||
// Set is the authoritative set of advisory locks held on one inode. The zero
|
||||
// value is an empty set. Set has no internal locking; the caller serializes it.
|
||||
type Set struct {
|
||||
locks []Range // sorted by Start
|
||||
}
|
||||
|
||||
func overlap(aStart, aEnd, bStart, bEnd uint64) bool {
|
||||
return aStart <= bEnd && bStart <= aEnd
|
||||
}
|
||||
|
||||
// Conflict returns the first held lock that blocks proposed, if any. Two locks
|
||||
// conflict when they share a namespace, have different owners, overlap, and at
|
||||
// least one is a write lock.
|
||||
func (s *Set) Conflict(proposed Range) (Range, bool) {
|
||||
for _, h := range s.locks {
|
||||
if h.IsFlock != proposed.IsFlock || h.sameOwner(proposed) {
|
||||
continue
|
||||
}
|
||||
if !overlap(h.Start, h.End, proposed.Start, proposed.End) {
|
||||
continue
|
||||
}
|
||||
if h.Type == Read && proposed.Type == Read {
|
||||
continue
|
||||
}
|
||||
return h, true
|
||||
}
|
||||
return Range{}, false
|
||||
}
|
||||
|
||||
// Acquire grants lk when it does not conflict, inserting it and coalescing the
|
||||
// owner's adjacent/overlapping ranges. On conflict it returns the blocking lock
|
||||
// and false, leaving the set unchanged.
|
||||
func (s *Set) Acquire(lk Range) (Range, bool) {
|
||||
if c, found := s.Conflict(lk); found {
|
||||
return c, false
|
||||
}
|
||||
s.insert(lk)
|
||||
return Range{}, true
|
||||
}
|
||||
|
||||
// Grant inserts lk without a conflict check. It is for a client mirroring locks
|
||||
// the server already granted (so they are conflict-free), not for arbitration.
|
||||
func (s *Set) Grant(lk Range) {
|
||||
s.insert(lk)
|
||||
}
|
||||
|
||||
// insert adds lk, absorbing same-owner same-type overlaps and merging adjacent
|
||||
// same-type ranges, and truncating/splitting a same-owner range of a different
|
||||
// type that overlaps (an in-place type change).
|
||||
func (s *Set) insert(lk Range) {
|
||||
var kept []Range
|
||||
for _, h := range s.locks {
|
||||
if !h.sameOwner(lk) || h.IsFlock != lk.IsFlock {
|
||||
kept = append(kept, h)
|
||||
continue
|
||||
}
|
||||
if !overlap(h.Start, h.End, lk.Start, lk.End) {
|
||||
// Merge only ranges that are adjacent and the same type. The
|
||||
// End < MaxUint64 guards stop +1 from wrapping at EOF.
|
||||
if h.Type == lk.Type && ((h.End < math.MaxUint64 && h.End+1 == lk.Start) || (lk.End < math.MaxUint64 && lk.End+1 == h.Start)) {
|
||||
if h.Start < lk.Start {
|
||||
lk.Start = h.Start
|
||||
}
|
||||
if h.End > lk.End {
|
||||
lk.End = h.End
|
||||
}
|
||||
continue
|
||||
}
|
||||
kept = append(kept, h)
|
||||
continue
|
||||
}
|
||||
if h.Type == lk.Type {
|
||||
// Same type: absorb into lk by widening its range.
|
||||
if h.Start < lk.Start {
|
||||
lk.Start = h.Start
|
||||
}
|
||||
if h.End > lk.End {
|
||||
lk.End = h.End
|
||||
}
|
||||
continue
|
||||
}
|
||||
// Different type: the surviving portions of h outside lk's range stay.
|
||||
if h.Start < lk.Start {
|
||||
left := h
|
||||
left.End = lk.Start - 1
|
||||
kept = append(kept, left)
|
||||
}
|
||||
if h.End > lk.End {
|
||||
right := h
|
||||
right.Start = lk.End + 1
|
||||
kept = append(kept, right)
|
||||
}
|
||||
}
|
||||
kept = append(kept, lk)
|
||||
sort.Slice(kept, func(i, j int) bool { return kept[i].Start < kept[j].Start })
|
||||
s.locks = kept
|
||||
}
|
||||
|
||||
// remove drops or splits matching locks within [start,end]. A match that
|
||||
// straddles the range keeps its non-overlapping head and/or tail.
|
||||
func (s *Set) remove(matches func(Range) bool, start, end uint64) {
|
||||
var kept []Range
|
||||
for _, h := range s.locks {
|
||||
if !matches(h) || !overlap(h.Start, h.End, start, end) {
|
||||
kept = append(kept, h)
|
||||
continue
|
||||
}
|
||||
if h.Start < start {
|
||||
left := h
|
||||
left.End = start - 1
|
||||
kept = append(kept, left)
|
||||
}
|
||||
if h.End > end {
|
||||
right := h
|
||||
right.Start = end + 1
|
||||
kept = append(kept, right)
|
||||
}
|
||||
// Fully covered: dropped.
|
||||
}
|
||||
s.locks = kept
|
||||
}
|
||||
|
||||
// Release clears lk's owner's locks within lk's namespace over [lk.Start,lk.End]
|
||||
// — the F_UNLCK path, which may split a straddling range.
|
||||
func (s *Set) Release(lk Range) {
|
||||
s.remove(func(h Range) bool {
|
||||
return h.sameOwner(lk) && h.IsFlock == lk.IsFlock
|
||||
}, lk.Start, lk.End)
|
||||
}
|
||||
|
||||
// ReleaseOwner removes every lock held by (sid, owner) in both namespaces.
|
||||
func (s *Set) ReleaseOwner(sid, owner uint64) {
|
||||
s.remove(func(h Range) bool {
|
||||
return h.Sid == sid && h.Owner == owner
|
||||
}, 0, math.MaxUint64)
|
||||
}
|
||||
|
||||
// ReleaseFlockOwner removes only the flock locks of (sid, owner) — the close-time
|
||||
// path for a released file description (FUSE_RELEASE_FLOCK_UNLOCK).
|
||||
func (s *Set) ReleaseFlockOwner(sid, owner uint64) {
|
||||
s.remove(func(h Range) bool {
|
||||
return h.IsFlock && h.Sid == sid && h.Owner == owner
|
||||
}, 0, math.MaxUint64)
|
||||
}
|
||||
|
||||
// ReleasePosixOwner removes only the fcntl locks of (sid, owner) — the close-time
|
||||
// path for a flushing POSIX lock owner.
|
||||
func (s *Set) ReleasePosixOwner(sid, owner uint64) {
|
||||
s.remove(func(h Range) bool {
|
||||
return !h.IsFlock && h.Sid == sid && h.Owner == owner
|
||||
}, 0, math.MaxUint64)
|
||||
}
|
||||
|
||||
// ReleaseSession removes every lock held by a session, reaping a mount that has
|
||||
// died or disconnected (its lease expired).
|
||||
func (s *Set) ReleaseSession(sid uint64) {
|
||||
s.remove(func(h Range) bool { return h.Sid == sid }, 0, math.MaxUint64)
|
||||
}
|
||||
|
||||
// HasPosix reports whether (sid, owner) holds any fcntl lock, mirroring the
|
||||
// mount's flush-time check that avoids treating a lock-free flush as
|
||||
// lock-sensitive.
|
||||
func (s *Set) HasPosix(sid, owner uint64) bool {
|
||||
for _, h := range s.locks {
|
||||
if !h.IsFlock && h.Sid == sid && h.Owner == owner {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// Locks returns the held locks, sorted by Start. The slice aliases internal
|
||||
// state; the caller must not mutate it.
|
||||
func (s *Set) Locks() []Range { return s.locks }
|
||||
|
||||
// Empty reports whether no locks are held, so the caller can drop the inode's
|
||||
// entry from its table.
|
||||
func (s *Set) Empty() bool { return len(s.locks) == 0 }
|
||||
@@ -0,0 +1,292 @@
|
||||
package posixlock
|
||||
|
||||
import (
|
||||
"math"
|
||||
"testing"
|
||||
)
|
||||
|
||||
// acquire is a test helper asserting the lock is granted.
|
||||
func mustAcquire(t *testing.T, s *Set, lk Range) {
|
||||
t.Helper()
|
||||
if _, granted := s.Acquire(lk); !granted {
|
||||
t.Fatalf("expected lock granted: %+v", lk)
|
||||
}
|
||||
}
|
||||
|
||||
func TestNonOverlappingLocksFromDifferentOwners(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 49, Type: Write, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 50, End: 99, Type: Write, Owner: 2, Pid: 20})
|
||||
}
|
||||
|
||||
func TestOverlappingReadLocksFromDifferentOwners(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Read, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 50, End: 149, Type: Read, Owner: 2, Pid: 20})
|
||||
}
|
||||
|
||||
func TestOverlappingWriteReadConflict(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
|
||||
if _, granted := s.Acquire(Range{Start: 50, End: 149, Type: Read, Owner: 2, Pid: 20}); granted {
|
||||
t.Fatal("expected conflict")
|
||||
}
|
||||
}
|
||||
|
||||
func TestOverlappingWriteWriteConflict(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
|
||||
if _, granted := s.Acquire(Range{Start: 50, End: 149, Type: Write, Owner: 2, Pid: 20}); granted {
|
||||
t.Fatal("expected conflict")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSameOwnerUpgradeReadToWrite(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Read, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
|
||||
|
||||
c, found := s.Conflict(Range{Start: 0, End: 99, Type: Write, Owner: 2, Pid: 20})
|
||||
if !found || c.Type != Write {
|
||||
t.Fatalf("expected conflicting write lock after upgrade, got %+v found=%v", c, found)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSameOwnerDowngradeWriteToRead(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Read, Owner: 1, Pid: 10})
|
||||
// Another owner can now take a shared read lock.
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Read, Owner: 2, Pid: 20})
|
||||
}
|
||||
|
||||
func TestLockCoalescing(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 9, Type: Write, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 10, End: 19, Type: Write, Owner: 1, Pid: 10})
|
||||
|
||||
if len(s.locks) != 1 {
|
||||
t.Fatalf("expected 1 coalesced lock, got %d: %+v", len(s.locks), s.locks)
|
||||
}
|
||||
if s.locks[0].Start != 0 || s.locks[0].End != 19 {
|
||||
t.Errorf("expected coalesced [0,19], got [%d,%d]", s.locks[0].Start, s.locks[0].End)
|
||||
}
|
||||
}
|
||||
|
||||
func TestLockSplitting(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
|
||||
s.Release(Range{Start: 40, End: 59, Type: Unlock, Owner: 1, Pid: 10})
|
||||
|
||||
if len(s.locks) != 2 {
|
||||
t.Fatalf("expected 2 locks after split, got %d: %+v", len(s.locks), s.locks)
|
||||
}
|
||||
if s.locks[0].Start != 0 || s.locks[0].End != 39 {
|
||||
t.Errorf("expected left [0,39], got [%d,%d]", s.locks[0].Start, s.locks[0].End)
|
||||
}
|
||||
if s.locks[1].Start != 60 || s.locks[1].End != 99 {
|
||||
t.Errorf("expected right [60,99], got [%d,%d]", s.locks[1].Start, s.locks[1].End)
|
||||
}
|
||||
}
|
||||
|
||||
func TestConflictReportsHolder(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 10, End: 50, Type: Write, Owner: 1, Pid: 10})
|
||||
|
||||
c, found := s.Conflict(Range{Start: 30, End: 70, Type: Read, Owner: 2, Pid: 20})
|
||||
if !found {
|
||||
t.Fatal("expected a conflict")
|
||||
}
|
||||
if c.Type != Write || c.Pid != 10 || c.Start != 10 || c.End != 50 {
|
||||
t.Fatalf("unexpected conflict report: %+v", c)
|
||||
}
|
||||
}
|
||||
|
||||
func TestConflictNoneForSharedReads(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 10, End: 50, Type: Read, Owner: 1, Pid: 10})
|
||||
if _, found := s.Conflict(Range{Start: 30, End: 70, Type: Read, Owner: 2, Pid: 20}); found {
|
||||
t.Fatal("two read locks should not conflict")
|
||||
}
|
||||
}
|
||||
|
||||
func TestConflictSameOwnerNone(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
|
||||
if _, found := s.Conflict(Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10}); found {
|
||||
t.Fatal("an owner should not conflict with itself")
|
||||
}
|
||||
}
|
||||
|
||||
func TestReleaseOwner(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 49, Type: Write, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 50, End: 99, Type: Write, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 200, End: 299, Type: Read, Owner: 2, Pid: 20})
|
||||
|
||||
s.ReleaseOwner(0, 1)
|
||||
|
||||
if _, found := s.Conflict(Range{Start: 0, End: 99, Type: Write, Owner: 3, Pid: 30}); found {
|
||||
t.Fatal("owner 1's locks should be gone")
|
||||
}
|
||||
c, found := s.Conflict(Range{Start: 200, End: 299, Type: Write, Owner: 3, Pid: 30})
|
||||
if !found || c.Type != Read {
|
||||
t.Fatalf("owner 2's read lock should remain, got %+v found=%v", c, found)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFlockAndFcntlDoNotConflict(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 0, End: math.MaxUint64, Type: Write, Owner: 2, Pid: 20, IsFlock: true})
|
||||
}
|
||||
|
||||
func TestReleasePosixOwnerKeepsFlock(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: math.MaxUint64, Type: Write, Owner: 1, Pid: 10, IsFlock: true})
|
||||
s.ReleasePosixOwner(0, 1)
|
||||
if _, found := s.Conflict(Range{Start: 0, End: math.MaxUint64, Type: Write, Owner: 2, Pid: 20, IsFlock: true}); !found {
|
||||
t.Fatal("flock lock should remain after ReleasePosixOwner")
|
||||
}
|
||||
}
|
||||
|
||||
func TestReleaseFlockOwnerKeepsPosix(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 0, End: math.MaxUint64, Type: Write, Owner: 2, Pid: 10, IsFlock: true})
|
||||
|
||||
s.ReleaseFlockOwner(0, 2)
|
||||
|
||||
if _, found := s.Conflict(Range{Start: 0, End: 99, Type: Write, Owner: 3, Pid: 30}); !found {
|
||||
t.Fatal("fcntl lock should remain after ReleaseFlockOwner")
|
||||
}
|
||||
if _, found := s.Conflict(Range{Start: 0, End: math.MaxUint64, Type: Write, Owner: 4, Pid: 40, IsFlock: true}); found {
|
||||
t.Fatal("flock lock should be gone after ReleaseFlockOwner")
|
||||
}
|
||||
}
|
||||
|
||||
func TestHasPosixIgnoresMissingOwnerAndFlock(t *testing.T) {
|
||||
s := &Set{}
|
||||
if s.HasPosix(0, 1) {
|
||||
t.Fatal("empty set should report no posix owner")
|
||||
}
|
||||
mustAcquire(t, s, Range{Start: 0, End: math.MaxUint64, Type: Write, Owner: 1, Pid: 10, IsFlock: true})
|
||||
if s.HasPosix(0, 1) {
|
||||
t.Fatal("a flock owner is not a posix owner")
|
||||
}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 2, Pid: 20})
|
||||
if !s.HasPosix(0, 2) {
|
||||
t.Fatal("posix owner should be reported")
|
||||
}
|
||||
}
|
||||
|
||||
func TestWholeFileLock(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: math.MaxUint64, Type: Write, Owner: 1, Pid: 10})
|
||||
if _, granted := s.Acquire(Range{Start: 0, End: math.MaxUint64, Type: Write, Owner: 2, Pid: 20}); granted {
|
||||
t.Fatal("whole-file lock should block another owner")
|
||||
}
|
||||
if _, granted := s.Acquire(Range{Start: 100, End: 200, Type: Read, Owner: 2, Pid: 20}); granted {
|
||||
t.Fatal("partial overlap with whole-file lock should conflict")
|
||||
}
|
||||
}
|
||||
|
||||
func TestReleaseNoExistingLocks(t *testing.T) {
|
||||
s := &Set{}
|
||||
s.Release(Range{Start: 0, End: 99, Type: Unlock, Owner: 1, Pid: 10})
|
||||
if !s.Empty() {
|
||||
t.Fatal("releasing on an empty set should be a no-op")
|
||||
}
|
||||
}
|
||||
|
||||
func TestSameOwnerReplaceDifferentType(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 30, End: 60, Type: Read, Owner: 1, Pid: 10})
|
||||
|
||||
if len(s.locks) != 3 {
|
||||
t.Fatalf("expected 3 locks after partial type change, got %d: %+v", len(s.locks), s.locks)
|
||||
}
|
||||
if s.locks[0].Type != Write || s.locks[0].Start != 0 || s.locks[0].End != 29 {
|
||||
t.Errorf("expected write [0,29], got %+v", s.locks[0])
|
||||
}
|
||||
if s.locks[1].Type != Read || s.locks[1].Start != 30 || s.locks[1].End != 60 {
|
||||
t.Errorf("expected read [30,60], got %+v", s.locks[1])
|
||||
}
|
||||
if s.locks[2].Type != Write || s.locks[2].Start != 61 || s.locks[2].End != 99 {
|
||||
t.Errorf("expected write [61,99], got %+v", s.locks[2])
|
||||
}
|
||||
}
|
||||
|
||||
func TestNonAdjacentRangesNotCoalesced(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 5, End: math.MaxUint64, Type: Write, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 0, End: 2, Type: Write, Owner: 1, Pid: 10})
|
||||
|
||||
if len(s.locks) != 2 {
|
||||
t.Fatalf("gap [3,4] should prevent coalescing, got %d: %+v", len(s.locks), s.locks)
|
||||
}
|
||||
if s.locks[0].Start != 0 || s.locks[0].End != 2 {
|
||||
t.Errorf("expected [0,2], got [%d,%d]", s.locks[0].Start, s.locks[0].End)
|
||||
}
|
||||
if s.locks[1].Start != 5 || s.locks[1].End != math.MaxUint64 {
|
||||
t.Errorf("expected [5,MaxUint64], got [%d,%d]", s.locks[1].Start, s.locks[1].End)
|
||||
}
|
||||
}
|
||||
|
||||
func TestAdjacencyNoOverflowAtMaxUint64(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 100, End: math.MaxUint64, Type: Write, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 0, End: 0, Type: Write, Owner: 1, Pid: 10})
|
||||
|
||||
if len(s.locks) != 2 {
|
||||
t.Fatalf("MaxUint64+1 must not wrap and falsely merge, got %d: %+v", len(s.locks), s.locks)
|
||||
}
|
||||
}
|
||||
|
||||
// Sessions are part of owner identity: the same FUSE Owner number on two
|
||||
// different mounts (Sid) is two distinct owners and must contend.
|
||||
func TestTwoSessionsSameOwnerDoNotAlias(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Sid: 1, Owner: 5, Pid: 10})
|
||||
|
||||
if _, granted := s.Acquire(Range{Start: 50, End: 149, Type: Write, Sid: 2, Owner: 5, Pid: 20}); granted {
|
||||
t.Fatal("same Owner number on a different session must conflict, not alias")
|
||||
}
|
||||
// A read on session 2 against a session-1 read is fine (shared).
|
||||
s2 := &Set{}
|
||||
mustAcquire(t, s2, Range{Start: 0, End: 99, Type: Read, Sid: 1, Owner: 5})
|
||||
mustAcquire(t, s2, Range{Start: 0, End: 99, Type: Read, Sid: 2, Owner: 5})
|
||||
}
|
||||
|
||||
// Reaping a dead mount drops only that session's locks.
|
||||
func TestReleaseSessionReapsOnlyThatSession(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 49, Type: Write, Sid: 1, Owner: 1, Pid: 10})
|
||||
mustAcquire(t, s, Range{Start: 0, End: math.MaxUint64, Type: Write, Sid: 1, Owner: 2, Pid: 11, IsFlock: true})
|
||||
mustAcquire(t, s, Range{Start: 50, End: 99, Type: Write, Sid: 2, Owner: 1, Pid: 20})
|
||||
|
||||
s.ReleaseSession(1)
|
||||
|
||||
for _, h := range s.locks {
|
||||
if h.Sid == 1 {
|
||||
t.Fatalf("session 1 lock survived reaping: %+v", h)
|
||||
}
|
||||
}
|
||||
c, found := s.Conflict(Range{Start: 50, End: 99, Type: Write, Sid: 3, Owner: 9})
|
||||
if !found || c.Sid != 2 {
|
||||
t.Fatalf("session 2's lock should remain, got %+v found=%v", c, found)
|
||||
}
|
||||
}
|
||||
|
||||
func TestEmptyAfterReleasingAll(t *testing.T) {
|
||||
s := &Set{}
|
||||
mustAcquire(t, s, Range{Start: 0, End: 99, Type: Write, Owner: 1, Pid: 10})
|
||||
if s.Empty() {
|
||||
t.Fatal("set should not be empty with a held lock")
|
||||
}
|
||||
s.Release(Range{Start: 0, End: 99, Type: Unlock, Owner: 1, Pid: 10})
|
||||
if !s.Empty() {
|
||||
t.Fatal("set should be empty after releasing the only lock")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,82 @@
|
||||
package posixlock
|
||||
|
||||
import (
|
||||
"reflect"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// A mount's tracked locks round-trip through Snapshot back to a fresh owner via
|
||||
// Reassert — the owner-restart / ring-change recovery path.
|
||||
func TestReassertRebuildsOnFreshOwner(t *testing.T) {
|
||||
const sid = uint64(7)
|
||||
|
||||
// Client mirror: two granted locks on one key, one on another.
|
||||
client := NewManager()
|
||||
client.Track("a", Range{Start: 0, End: 99, Type: Write, Sid: sid, Owner: 1})
|
||||
client.Track("a", Range{Start: 200, End: 299, Type: Read, Sid: sid, Owner: 2})
|
||||
client.Track("b", Range{Start: 0, End: maxEnd, Type: Write, Sid: sid, Owner: 1, IsFlock: true})
|
||||
|
||||
// Fresh owner (post-restart / new ring owner) knows nothing.
|
||||
owner := NewManager()
|
||||
for key, locks := range client.Snapshot() {
|
||||
if c := owner.Reassert(key, sid, locks); c != nil {
|
||||
t.Fatalf("unexpected conflict reasserting %s: %+v", key, c)
|
||||
}
|
||||
}
|
||||
|
||||
// The owner now reports the same conflicts a foreign session would hit.
|
||||
if _, granted := owner.TryLock("a", Range{Start: 50, End: 60, Type: Write, Sid: 99, Owner: 1}); granted {
|
||||
t.Fatal("owner should block a foreign write after rebuild")
|
||||
}
|
||||
if _, granted := owner.TryLock("b", Range{Start: 0, End: 0, Type: Read, Sid: 99, Owner: 1, IsFlock: true}); granted {
|
||||
t.Fatal("owner should block a foreign flock read after rebuild")
|
||||
}
|
||||
}
|
||||
|
||||
// Re-asserting every tick is idempotent: the owner's view is unchanged.
|
||||
func TestReassertIdempotent(t *testing.T) {
|
||||
const sid = uint64(1)
|
||||
m := NewManager()
|
||||
m.TryLock("k", Range{Start: 0, End: 99, Type: Write, Sid: sid, Owner: 1})
|
||||
before := append([]Range(nil), m.byKey["k"].locks...)
|
||||
|
||||
m.Reassert("k", sid, before)
|
||||
m.Reassert("k", sid, before)
|
||||
|
||||
if !reflect.DeepEqual(m.byKey["k"].locks, before) {
|
||||
t.Fatalf("reassert not idempotent:\n got %+v\nwant %+v", m.byKey["k"].locks, before)
|
||||
}
|
||||
}
|
||||
|
||||
// A lock another session grabbed in the migration window is reported as a
|
||||
// conflict and not double-granted.
|
||||
func TestReassertReportsConflict(t *testing.T) {
|
||||
const mine, other = uint64(1), uint64(2)
|
||||
m := NewManager()
|
||||
// Another mount took the lock on this (new) owner during the gap.
|
||||
m.TryLock("k", Range{Start: 0, End: 99, Type: Write, Sid: other, Owner: 1})
|
||||
|
||||
conflicts := m.Reassert("k", mine, []Range{{Start: 0, End: 99, Type: Write, Sid: mine, Owner: 1}})
|
||||
if len(conflicts) != 1 {
|
||||
t.Fatalf("expected 1 conflict, got %d: %+v", len(conflicts), conflicts)
|
||||
}
|
||||
// The other session keeps the lock; mine was not installed.
|
||||
if got := len(m.byKey["k"].locks); got != 1 {
|
||||
t.Fatalf("expected only the incumbent lock, got %d", got)
|
||||
}
|
||||
}
|
||||
|
||||
// Reassert renews the lease, so a re-asserting mount is not reaped.
|
||||
func TestReassertRenewsLease(t *testing.T) {
|
||||
const sid = uint64(1)
|
||||
m := NewManager()
|
||||
m.Renew(sid)
|
||||
m.Reassert("k", sid, []Range{{Start: 0, End: 9, Type: Write, Sid: sid, Owner: 1}})
|
||||
|
||||
if reaped := m.ReapExpired(time.Hour); len(reaped) != 0 {
|
||||
t.Fatalf("freshly re-asserted session should not be reaped: %v", reaped)
|
||||
}
|
||||
}
|
||||
|
||||
const maxEnd = ^uint64(0)
|
||||
@@ -28,6 +28,8 @@ func (store *PostgresStore) GetName() string {
|
||||
}
|
||||
|
||||
func (store *PostgresStore) Initialize(configuration util.Configuration, prefix string) (err error) {
|
||||
// Absent key keeps a pooled default; an explicit 0 disables the idle pool.
|
||||
configuration.SetDefault(prefix+"connection_max_idle", 2)
|
||||
return store.initialize(
|
||||
configuration.GetString(prefix+"upsertQuery"),
|
||||
configuration.GetBool(prefix+"enableUpsert"),
|
||||
|
||||
@@ -33,6 +33,8 @@ func (store *PostgresStore2) GetName() string {
|
||||
}
|
||||
|
||||
func (store *PostgresStore2) Initialize(configuration util.Configuration, prefix string) (err error) {
|
||||
// Absent key keeps a pooled default; an explicit 0 disables the idle pool.
|
||||
configuration.SetDefault(prefix+"connection_max_idle", 2)
|
||||
return store.initialize(
|
||||
configuration.GetString(prefix+"createTable"),
|
||||
configuration.GetString(prefix+"upsertQuery"),
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user