mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-10-08 07:35:50 +00:00
Compare commits
99
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
530be3e373 | ||
|
|
9b3b12c607 | ||
|
|
3abdef3202 | ||
|
|
43fd5b8d82 | ||
|
|
150a69fe11 | ||
|
|
4fec65d949 | ||
|
|
f564918685 | ||
|
|
f9289f0570 | ||
|
|
4303b3aa4c | ||
|
|
a0ee7ba314 | ||
|
|
a976b21010 | ||
|
|
e57f8c4d87 | ||
|
|
67691a1eea | ||
|
|
02353444ac | ||
|
|
68944e83a3 | ||
|
|
00310f6588 | ||
|
|
5c9c424a84 | ||
|
|
5218e68554 | ||
|
|
00cffa028c | ||
|
|
a261f90e18 | ||
|
|
be29f44d87 | ||
|
|
2864bc0fe8 | ||
|
|
ab95d58b7c | ||
|
|
2f641a63d6 | ||
|
|
80fd3635d2 | ||
|
|
7129e1178e | ||
|
|
3c1e8ca7a8 | ||
|
|
0f3ba98e11 | ||
|
|
80a26020d7 | ||
|
|
5389f61cef | ||
|
|
8ad2f29e3e | ||
|
|
c58bd0dfd3 | ||
|
|
f7680cf812 | ||
|
|
4299fdf578 | ||
|
|
3e9fc9e75b | ||
|
|
afce0a3dd3 | ||
|
|
83d44be0f3 | ||
|
|
975cec9228 | ||
|
|
f31a026b2a | ||
|
|
4914c14982 | ||
|
|
2f6c237238 | ||
|
|
df4995b894 | ||
|
|
b750853c42 | ||
|
|
317e756b9a | ||
|
|
5b79f51e3c | ||
|
|
635f69a821 | ||
|
|
11791fad6a | ||
|
|
56d2f05ccd | ||
|
|
8c1be63c92 | ||
|
|
bb9942c646 | ||
|
|
f0afcf904d | ||
|
|
c1ccbcda13 | ||
|
|
94a68fa9b9 | ||
|
|
b3a8701989 | ||
|
|
b9ad62fc16 | ||
|
|
d848b8ed00 | ||
|
|
196c71b613 | ||
|
|
559ec33498 | ||
|
|
d6397fdc50 | ||
|
|
264030c08c | ||
|
|
8099e71934 | ||
|
|
63eb67c9da | ||
|
|
93efc64af6 | ||
|
|
f2b08e47e5 | ||
|
|
5079d926e4 | ||
|
|
a9e1b57dcd | ||
|
|
e2608edda4 | ||
|
|
0f2ecb766f | ||
|
|
6848cdf9e1 | ||
|
|
ca62d4297b | ||
|
|
4bb40732bb | ||
|
|
8ff2e0777e | ||
|
|
ac876eef21 | ||
|
|
ddf009ffe7 | ||
|
|
818f3bb71b | ||
|
|
7643f4f541 | ||
|
|
f1ed270942 | ||
|
|
7dbbdac030 | ||
|
|
6d676eda67 | ||
|
|
d002481037 | ||
|
|
1df8c05bc3 | ||
|
|
44ba070d83 | ||
|
|
062238bb5c | ||
|
|
b77c42ff32 | ||
|
|
26fc90187e | ||
|
|
ef463fe1af | ||
|
|
17ad5a1419 | ||
|
|
b03419ee92 | ||
|
|
f849b7c823 | ||
|
|
f15b980976 | ||
|
|
110b485bae | ||
|
|
06dda12e4b | ||
|
|
a93a1ab2eb | ||
|
|
cd1e738422 | ||
|
|
3dec359d6b | ||
|
|
f6a3286b32 | ||
|
|
01bb3b3053 | ||
|
|
5769057af3 | ||
|
|
4160b92864 |
@@ -27,7 +27,7 @@ jobs:
|
||||
|
||||
# Initializes the CodeQL tools for scanning.
|
||||
- name: Initialize CodeQL
|
||||
uses: github/codeql-action/init@v4.38.0
|
||||
uses: github/codeql-action/init@v4.38.1
|
||||
# Override language selection by uncommenting this and choosing your languages
|
||||
with:
|
||||
languages: go
|
||||
@@ -35,7 +35,7 @@ jobs:
|
||||
# Autobuild attempts to build any compiled languages (C/C++, C#, or Java).
|
||||
# If this step fails, then you should remove it and run the build manually (see below).
|
||||
- name: Autobuild
|
||||
uses: github/codeql-action/autobuild@v4.38.0
|
||||
uses: github/codeql-action/autobuild@v4.38.1
|
||||
|
||||
# ℹ️ Command-line programs to run using the OS shell.
|
||||
# 📚 See https://docs.github.com/en/actions/using-workflows/workflow-syntax-for-github-actions#jobsjob_idstepsrun
|
||||
@@ -49,4 +49,4 @@ jobs:
|
||||
# make release
|
||||
|
||||
- name: Perform CodeQL Analysis
|
||||
uses: github/codeql-action/analyze@v4.38.0
|
||||
uses: github/codeql-action/analyze@v4.38.1
|
||||
|
||||
@@ -6,6 +6,7 @@ on:
|
||||
paths:
|
||||
- 'weed/**'
|
||||
- 'seaweed-volume/**'
|
||||
- 'seaweed-common/**'
|
||||
- 'seaweed-worker/**'
|
||||
- 'docker/**'
|
||||
- 'go.mod'
|
||||
@@ -152,7 +153,7 @@ jobs:
|
||||
org.opencontainers.image.vendor=Chris Lu
|
||||
|
||||
- name: Set up QEMU
|
||||
uses: docker/setup-qemu-action@v4.3.0
|
||||
uses: docker/setup-qemu-action@v4.4.0
|
||||
|
||||
- name: Create BuildKit config
|
||||
run: |
|
||||
|
||||
@@ -129,7 +129,7 @@ jobs:
|
||||
echo "seaweedfs_ref=$seaweed" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Set up QEMU
|
||||
uses: docker/setup-qemu-action@v4.3.0
|
||||
uses: docker/setup-qemu-action@v4.4.0
|
||||
|
||||
- name: Set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@v4
|
||||
|
||||
@@ -236,7 +236,7 @@ jobs:
|
||||
org.opencontainers.image.vendor=Chris Lu
|
||||
- name: Set up QEMU
|
||||
if: matrix.platform != 'amd64'
|
||||
uses: docker/setup-qemu-action@v4.3.0
|
||||
uses: docker/setup-qemu-action@v4.4.0
|
||||
- name: Create BuildKit config
|
||||
run: |
|
||||
cat > /tmp/buildkitd.toml <<EOF
|
||||
@@ -405,7 +405,7 @@ jobs:
|
||||
output: trivy-results.sarif
|
||||
exit-code: '0'
|
||||
- name: Upload Trivy scan results to GitHub Security
|
||||
uses: github/codeql-action/upload-sarif@v4.38.0
|
||||
uses: github/codeql-action/upload-sarif@v4.38.1
|
||||
if: always()
|
||||
with:
|
||||
sarif_file: trivy-results.sarif
|
||||
|
||||
@@ -46,7 +46,7 @@ jobs:
|
||||
org.opencontainers.image.vendor=Chris Lu
|
||||
-
|
||||
name: Set up QEMU
|
||||
uses: docker/setup-qemu-action@v4.3.0
|
||||
uses: docker/setup-qemu-action@v4.4.0
|
||||
-
|
||||
name: Set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@v4
|
||||
|
||||
@@ -231,7 +231,7 @@ jobs:
|
||||
|
||||
- name: Set up QEMU
|
||||
if: (github.event_name != 'workflow_dispatch' || github.event.inputs.variant == 'all' || github.event.inputs.variant == matrix.variant) && matrix.qemu
|
||||
uses: docker/setup-qemu-action@v4.3.0
|
||||
uses: docker/setup-qemu-action@v4.4.0
|
||||
|
||||
- name: Create BuildKit config
|
||||
if: github.event_name != 'workflow_dispatch' || github.event.inputs.variant == 'all' || github.event.inputs.variant == matrix.variant
|
||||
@@ -456,7 +456,7 @@ jobs:
|
||||
|
||||
- name: Upload Trivy scan results to GitHub Security
|
||||
if: always()
|
||||
uses: github/codeql-action/upload-sarif@v4.38.0
|
||||
uses: github/codeql-action/upload-sarif@v4.38.1
|
||||
with:
|
||||
sarif_file: trivy-results.sarif
|
||||
category: trivy-${{ matrix.variant }}
|
||||
|
||||
@@ -85,7 +85,7 @@ jobs:
|
||||
echo "seaweedfs_ref=$seaweed" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Set up QEMU
|
||||
uses: docker/setup-qemu-action@1f40c72289eff860ee54a304f1438e3cff362e0a # v1
|
||||
uses: docker/setup-qemu-action@99012661954931238ded8c8b007157a8430204e1 # v1
|
||||
|
||||
- name: Set up Docker Buildx
|
||||
uses: docker/setup-buildx-action@4d04d5d9486b7bd6fa91e7baf45bbb4f8b9deedd # v1
|
||||
|
||||
@@ -116,6 +116,20 @@ jobs:
|
||||
grep -q "security-config" /tmp/security.yaml
|
||||
echo "Security configuration renders correctly"
|
||||
|
||||
echo ""
|
||||
echo "=== Testing admin.allowInsecureBind satisfies the admin auth render guard ==="
|
||||
helm template test $CHART_DIR --set admin.enabled=true --set admin.allowInsecureBind=true \
|
||||
> /tmp/admin-allow-insecure-bind.yaml
|
||||
grep -q -- "-allowInsecureBind" /tmp/admin-allow-insecure-bind.yaml
|
||||
echo "admin.allowInsecureBind renders -allowInsecureBind and passes the render guard"
|
||||
|
||||
if helm template test $CHART_DIR --set admin.enabled=true > /tmp/admin-no-auth.yaml 2>/tmp/admin-no-auth.err; then
|
||||
echo "FAIL: admin.enabled=true with no auth configured should fail to render"
|
||||
exit 1
|
||||
fi
|
||||
grep -q "admin.allowInsecureBind" /tmp/admin-no-auth.err
|
||||
echo "admin with no auth configured still fails the render guard, and the guard mentions admin.allowInsecureBind"
|
||||
|
||||
echo ""
|
||||
echo "=== Testing JWT expiration overrides ==="
|
||||
helm template test $CHART_DIR \
|
||||
|
||||
@@ -8,6 +8,7 @@ on:
|
||||
- 'go.mod'
|
||||
- 'go.sum'
|
||||
- 'seaweed-volume/**'
|
||||
- 'seaweed-common/**'
|
||||
- 'test/perf/**'
|
||||
- '.github/workflows/performance.yml'
|
||||
workflow_dispatch:
|
||||
|
||||
@@ -5,6 +5,7 @@ on:
|
||||
branches: [ master ]
|
||||
paths:
|
||||
- 'seaweed-volume/**'
|
||||
- 'seaweed-common/**'
|
||||
- 'test/volume_server/**'
|
||||
- 'weed/pb/volume_server.proto'
|
||||
- 'weed/pb/volume_server_pb/**'
|
||||
@@ -13,6 +14,7 @@ on:
|
||||
branches: [ master, main ]
|
||||
paths:
|
||||
- 'seaweed-volume/**'
|
||||
- 'seaweed-common/**'
|
||||
- 'test/volume_server/**'
|
||||
- 'weed/pb/volume_server.proto'
|
||||
- 'weed/pb/volume_server_pb/**'
|
||||
@@ -43,7 +45,7 @@ jobs:
|
||||
|
||||
- name: Filter changed paths
|
||||
id: filter
|
||||
uses: dorny/paths-filter@v3
|
||||
uses: dorny/paths-filter@v4
|
||||
with:
|
||||
filters: |
|
||||
rust:
|
||||
@@ -75,7 +77,7 @@ jobs:
|
||||
~/.cargo/registry
|
||||
~/.cargo/git
|
||||
seaweed-volume/target
|
||||
key: rust-${{ steps.toolchain.outputs.fingerprint }}-${{ hashFiles('seaweed-volume/Cargo.lock') }}
|
||||
key: rust-${{ steps.toolchain.outputs.fingerprint }}-${{ hashFiles('seaweed-volume/Cargo.lock', 'seaweed-common/Cargo.lock') }}
|
||||
restore-keys: |
|
||||
rust-${{ steps.toolchain.outputs.fingerprint }}-
|
||||
|
||||
@@ -93,6 +95,16 @@ jobs:
|
||||
# - name: Check formatting
|
||||
# run: cd seaweed-volume && cargo fmt --check
|
||||
|
||||
# seaweed-common is a path dependency of this crate, not a member of its
|
||||
# workspace, so the run below does not reach its own tests. It builds into
|
||||
# this job's cached target directory, and the cache key above covers the
|
||||
# shared crate's lock, so the aws-lc-sys that rustls pulls in is restored
|
||||
# with the cache instead of compiled from scratch on every run.
|
||||
- name: Run shared-crate unit tests
|
||||
env:
|
||||
CARGO_TARGET_DIR: ${{ github.workspace }}/seaweed-volume/target
|
||||
run: cd seaweed-common && cargo test
|
||||
|
||||
- name: Run Rust unit tests
|
||||
run: cd seaweed-volume && cargo test
|
||||
|
||||
@@ -168,7 +180,7 @@ jobs:
|
||||
~/.cargo/registry
|
||||
~/.cargo/git
|
||||
seaweed-volume/target
|
||||
key: rust-${{ steps.toolchain.outputs.fingerprint }}-${{ hashFiles('seaweed-volume/Cargo.lock') }}
|
||||
key: rust-${{ steps.toolchain.outputs.fingerprint }}-${{ hashFiles('seaweed-volume/Cargo.lock', 'seaweed-common/Cargo.lock') }}
|
||||
restore-keys: |
|
||||
rust-${{ steps.toolchain.outputs.fingerprint }}-
|
||||
|
||||
@@ -250,7 +262,7 @@ jobs:
|
||||
~/.cargo/registry
|
||||
~/.cargo/git
|
||||
seaweed-volume/target
|
||||
key: rust-${{ steps.toolchain.outputs.fingerprint }}-${{ hashFiles('seaweed-volume/Cargo.lock') }}
|
||||
key: rust-${{ steps.toolchain.outputs.fingerprint }}-${{ hashFiles('seaweed-volume/Cargo.lock', 'seaweed-common/Cargo.lock') }}
|
||||
restore-keys: |
|
||||
rust-${{ steps.toolchain.outputs.fingerprint }}-
|
||||
|
||||
|
||||
@@ -5,12 +5,14 @@ on:
|
||||
branches: [ master ]
|
||||
paths:
|
||||
- 'seaweed-worker/**'
|
||||
- 'seaweed-common/**'
|
||||
- 'weed/pb/plugin.proto'
|
||||
- '.github/workflows/rust-worker-tests.yml'
|
||||
push:
|
||||
branches: [ master, main ]
|
||||
paths:
|
||||
- 'seaweed-worker/**'
|
||||
- 'seaweed-common/**'
|
||||
- 'weed/pb/plugin.proto'
|
||||
- '.github/workflows/rust-worker-tests.yml'
|
||||
|
||||
@@ -49,7 +51,7 @@ jobs:
|
||||
~/.cargo/registry
|
||||
~/.cargo/git
|
||||
seaweed-worker/target/release
|
||||
key: rust-worker-${{ steps.toolchain.outputs.fingerprint }}-${{ hashFiles('seaweed-worker/Cargo.lock') }}
|
||||
key: rust-worker-${{ steps.toolchain.outputs.fingerprint }}-${{ hashFiles('seaweed-worker/Cargo.lock', 'seaweed-common/Cargo.lock') }}
|
||||
restore-keys: |
|
||||
rust-worker-${{ steps.toolchain.outputs.fingerprint }}-
|
||||
|
||||
@@ -84,6 +86,17 @@ jobs:
|
||||
# - name: Check formatting
|
||||
# run: cd seaweed-worker && cargo fmt --all --check
|
||||
|
||||
# seaweed-common is a path dependency of core and lance, not a member of
|
||||
# this workspace, so `--workspace` below does not reach its own tests.
|
||||
# Release and this job's cached target directory, and the cache key above
|
||||
# covers the shared crate's lock. That lock pins the same rustls and
|
||||
# aws-lc-sys this workspace resolves, so the release build above has
|
||||
# already paid for them.
|
||||
- name: Run shared-crate unit tests
|
||||
env:
|
||||
CARGO_TARGET_DIR: ${{ github.workspace }}/seaweed-worker/target
|
||||
run: cd seaweed-common && cargo test --release
|
||||
|
||||
# The tests that need a live gateway skip themselves without one, the way
|
||||
# the Go integration tests skip without Docker; the lifecycle suite in
|
||||
# test/s3tables/lifecycle is what runs them against a real cluster.
|
||||
|
||||
@@ -5,6 +5,7 @@ on:
|
||||
branches: [ master ]
|
||||
paths:
|
||||
- 'seaweed-volume/**'
|
||||
- 'seaweed-common/**'
|
||||
- '.github/workflows/rust_binaries_dev.yml'
|
||||
|
||||
permissions:
|
||||
|
||||
@@ -0,0 +1,106 @@
|
||||
name: "Snowflake S3Compat API tests"
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [ master ]
|
||||
paths:
|
||||
- 'weed/s3api/**'
|
||||
- 'weed/filer/**'
|
||||
- 'weed/server/**'
|
||||
- 'weed/iam/**'
|
||||
- 'weed/command/**'
|
||||
- 'weed/storage/**'
|
||||
- 'weed/operation/**'
|
||||
- 'weed/wdclient/**'
|
||||
- 'weed/cluster/**'
|
||||
- 'weed/pb/**'
|
||||
- 'test/s3/snowflake/**'
|
||||
- 'go.mod'
|
||||
- 'go.sum'
|
||||
- '.github/workflows/s3-snowflake-tests.yml'
|
||||
pull_request:
|
||||
branches: [ master ]
|
||||
paths:
|
||||
- 'weed/s3api/**'
|
||||
- 'weed/filer/**'
|
||||
- 'weed/server/**'
|
||||
- 'weed/iam/**'
|
||||
- 'weed/command/**'
|
||||
- 'weed/storage/**'
|
||||
- 'weed/operation/**'
|
||||
- 'weed/wdclient/**'
|
||||
- 'weed/cluster/**'
|
||||
- 'weed/pb/**'
|
||||
- 'test/s3/snowflake/**'
|
||||
- 'go.mod'
|
||||
- 'go.sum'
|
||||
- '.github/workflows/s3-snowflake-tests.yml'
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.event.pull_request.number || github.ref }}/s3-snowflake-tests
|
||||
cancel-in-progress: true
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
snowflake-s3compat-tests:
|
||||
name: Snowflake S3Compat API tests
|
||||
runs-on: ubuntu-22.04
|
||||
timeout-minutes: 30
|
||||
env:
|
||||
WORK_DIR: /tmp/seaweedfs-snowflake-tests
|
||||
steps:
|
||||
- name: Check out code
|
||||
uses: actions/checkout@v7
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- name: Set up Go
|
||||
uses: actions/setup-go@v7
|
||||
with:
|
||||
go-version-file: 'go.mod'
|
||||
id: go
|
||||
|
||||
- name: Set up Java
|
||||
uses: actions/setup-java@v6
|
||||
with:
|
||||
java-version: '17'
|
||||
distribution: 'temurin'
|
||||
cache: 'maven'
|
||||
|
||||
- name: Install SeaweedFS
|
||||
run: |
|
||||
cd weed
|
||||
go install -buildvcs=false
|
||||
weed version
|
||||
|
||||
- name: Run Snowflake S3Compat API tests
|
||||
timeout-minutes: 20
|
||||
run: |
|
||||
# Starts weed server, creates the buckets/objects the suite needs,
|
||||
# clones the upstream suite, and runs mvn -Dtest=S3CompatApiTest.
|
||||
bash test/s3/snowflake/run.sh
|
||||
|
||||
- name: Show logs on failure
|
||||
if: failure()
|
||||
run: |
|
||||
echo "=== SeaweedFS Server Log ==="
|
||||
tail -200 "$WORK_DIR/weed.log" || echo "No server log"
|
||||
echo ""
|
||||
echo "=== Surefire results ==="
|
||||
cat "$WORK_DIR"/snowflake-s3compat-api-test-suite/s3compatapi/target/surefire-reports/*.txt 2>/dev/null || echo "No surefire reports"
|
||||
|
||||
- name: Upload test results
|
||||
if: always()
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: snowflake-s3compat-surefire-reports
|
||||
path: /tmp/seaweedfs-snowflake-tests/snowflake-s3compat-api-test-suite/s3compatapi/target/surefire-reports/
|
||||
retention-days: 14
|
||||
|
||||
- name: Cleanup
|
||||
if: always()
|
||||
run: |
|
||||
pkill -9 -f "weed server" || true
|
||||
rm -rf "$WORK_DIR" || true
|
||||
@@ -439,6 +439,117 @@ jobs:
|
||||
path: test/s3tables/catalog_clickhouse/test-output.log
|
||||
retention-days: 3
|
||||
|
||||
olake-iceberg-catalog-tests:
|
||||
name: OLake Iceberg Catalog Integration Tests (${{ matrix.tag }})
|
||||
runs-on: ubuntu-22.04
|
||||
timeout-minutes: 30
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
include:
|
||||
# Pinned baseline, and latest so new OLake releases are exercised
|
||||
# without a code change. OLake's Iceberg writer is a Java sidecar
|
||||
# whose Iceberg version moves independently of the Go release, so
|
||||
# the latest leg is the one that catches library drift.
|
||||
- olake-image: olakego/source-postgres:v0.10.1
|
||||
tag: "v0.10.1"
|
||||
- olake-image: olakego/source-postgres:latest
|
||||
tag: latest
|
||||
|
||||
steps:
|
||||
- name: Check out code
|
||||
uses: actions/checkout@v7
|
||||
|
||||
- name: Set up Go
|
||||
uses: actions/setup-go@v7
|
||||
with:
|
||||
go-version-file: 'go.mod'
|
||||
id: go
|
||||
|
||||
- name: Configure Docker Hub mirror
|
||||
run: |
|
||||
echo '{"registry-mirrors": ["https://mirror.gcr.io"]}' | sudo tee /etc/docker/daemon.json
|
||||
sudo systemctl restart docker
|
||||
|
||||
- name: Pre-pull images
|
||||
run: |
|
||||
pull() { for i in 1 2 3; do docker pull "$1" && return 0; sleep 15; done; return 1; }
|
||||
pull ${{ matrix.olake-image }}
|
||||
pull postgres:16
|
||||
pull python:3.11-slim
|
||||
|
||||
- name: Run go mod tidy
|
||||
run: go mod tidy
|
||||
|
||||
- name: Install SeaweedFS
|
||||
run: |
|
||||
go install -buildvcs=false ./weed
|
||||
|
||||
- name: Run OLake Iceberg Catalog Integration Tests
|
||||
timeout-minutes: 25
|
||||
working-directory: test/s3tables/catalog_olake
|
||||
env:
|
||||
OLAKE_IMAGE: ${{ matrix.olake-image }}
|
||||
run: |
|
||||
set -x
|
||||
set -o pipefail
|
||||
echo "=== System Information ==="
|
||||
uname -a
|
||||
free -h
|
||||
df -h
|
||||
docker info
|
||||
echo "=== Starting OLake Iceberg Catalog Tests ==="
|
||||
|
||||
go test -v -timeout 20m . 2>&1 | tee test-output.log || {
|
||||
echo "OLake Iceberg catalog integration tests failed"
|
||||
exit 1
|
||||
}
|
||||
|
||||
# The suite skips itself when Docker is unavailable, so a green job is not
|
||||
# by itself evidence that anything ran. Assert execution explicitly.
|
||||
- name: Assert the suite actually ran
|
||||
working-directory: test/s3tables/catalog_olake
|
||||
run: |
|
||||
log=test-output.log
|
||||
if [ ! -f "$log" ]; then
|
||||
echo "::error::no test-output.log; the suite did not run"
|
||||
exit 1
|
||||
fi
|
||||
passes=$(grep -c '^--- PASS' "$log" || true)
|
||||
skips=$(grep -c '^--- SKIP' "$log" || true)
|
||||
echo "top-level PASS=$passes SKIP=$skips"
|
||||
if [ "$skips" -gt 0 ]; then
|
||||
echo "::error::the OLake suite skipped $skips top-level test(s); the environment it needs was not provisioned, so this job proves nothing"
|
||||
grep '^--- SKIP' "$log" | head -20
|
||||
exit 1
|
||||
fi
|
||||
if [ "$passes" -lt 1 ]; then
|
||||
echo "::error::the OLake suite recorded no passing top-level test"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
- name: Show test output on failure
|
||||
if: failure()
|
||||
working-directory: test/s3tables/catalog_olake
|
||||
run: |
|
||||
echo "=== Test Output ==="
|
||||
if [ -f test-output.log ]; then
|
||||
tail -200 test-output.log
|
||||
fi
|
||||
|
||||
echo "=== Process information ==="
|
||||
ps aux | grep -E "(weed|test|docker|olake|postgres)" || true
|
||||
echo "=== Containers ==="
|
||||
docker ps -a | head -30 || true
|
||||
|
||||
- name: Upload test logs on failure
|
||||
if: failure()
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: olake-iceberg-catalog-test-logs-${{ matrix.tag }}
|
||||
path: test/s3tables/catalog_olake/test-output.log
|
||||
retention-days: 3
|
||||
|
||||
polaris-integration-tests:
|
||||
name: Polaris Integration Tests
|
||||
runs-on: ubuntu-22.04
|
||||
|
||||
@@ -21,6 +21,7 @@ One `weed` binary serves an S3 object store, a POSIX file system, and a lakehous
|
||||
|
||||
- [Download Binaries for different platforms](https://github.com/seaweedfs/seaweedfs/releases/latest)
|
||||
- [Wiki Documentation](https://github.com/seaweedfs/seaweedfs/wiki)
|
||||
- [HTTP REST API](REST_API.md) for the filer, master, and volume servers
|
||||
- Community: [Slack](https://join.slack.com/t/seaweedfs/shared_invite/enQtMzI4MTMwMjU2MzA3LTEyYzZmZWYzOGQ3MDJlZWMzYmI0OTE4OTJiZjJjODBmMzUxNmYwODg0YjY3MTNlMjBmZDQ1NzQ5NDJhZWI2ZmY), [Twitter](https://twitter.com/SeaweedFS), [Telegram](https://t.me/Seaweedfs), [Reddit](https://www.reddit.com/r/SeaweedFS/), [Mailing List](https://groups.google.com/d/forum/seaweedfs)
|
||||
- [SeaweedFS White Paper](https://github.com/seaweedfs/seaweedfs/wiki/SeaweedFS_Architecture.pdf) and introduction slides: [2025.5](https://docs.google.com/presentation/d/1tdkp45J01oRV68dIm4yoTXKJDof-EhainlA0LMXexQE/edit?usp=sharing), [2021.5](https://docs.google.com/presentation/d/1DcxKWlINc-HNCjhYeERkpGXXm6nTCES8mi2W5G0Z4Ts/edit?usp=sharing), [2019.3](https://www.slideshare.net/chrislusf/seaweedfs-introduction)
|
||||
|
||||
|
||||
+344
@@ -0,0 +1,344 @@
|
||||
# SeaweedFS HTTP REST API
|
||||
|
||||
SeaweedFS exposes three HTTP surfaces:
|
||||
|
||||
| Service | Default port | Addressing |
|
||||
|---------|--------------|------------|
|
||||
| Filer | 8888 | File system paths (`/dir/name`) |
|
||||
| Master | 9333 | File id assignment and cluster topology |
|
||||
| Volume server | 8080 | File content by file id (`vid,fid`) |
|
||||
|
||||
Most clients only need the filer API (paths) or the S3 API. The master and
|
||||
volume APIs are the lower-level blob store interface.
|
||||
|
||||
Conventions applying to all three:
|
||||
|
||||
- Responses are JSON unless noted otherwise. Append `&pretty=y` to pretty-print.
|
||||
- A file id (`fid`) has the form `volumeId,fileKeyCookie`, e.g. `3,01637037d6`.
|
||||
An optional suffix selects a reserved id from a `count` assignment
|
||||
(`3,01637037d6_1`, `_2`, ...), and an optional extension
|
||||
(`3,01637037d6.jpg`) sets the content type on reads.
|
||||
- `replication` is a 3-digit replica placement `xyz`: `x` copies in other
|
||||
data centers, `y` on other racks in the same data center, `z` on other
|
||||
volume servers on the same rack. `000` = no replication, `001` = one copy
|
||||
on the same rack, `010` = one copy on a different rack, `100` = one copy in
|
||||
another data center, `200` = two copies in two other data centers, `110` =
|
||||
one copy in another data center plus one on another rack.
|
||||
- `ttl` units: `m` minute, `h` hour, `d` day, `w` week, `M` month, `y` year.
|
||||
|
||||
## Filer API (port 8888)
|
||||
|
||||
The filer presents a POSIX-like namespace over the volume servers.
|
||||
|
||||
### Upload a file
|
||||
|
||||
```bash
|
||||
# PUT the raw body to the target path
|
||||
curl -T /home/chris/myphoto.jpg "http://localhost:8888/dir/myphoto.jpg"
|
||||
|
||||
# or POST as multipart form (the part filename becomes the entry name)
|
||||
curl -F file=@/home/chris/myphoto.jpg "http://localhost:8888/dir/"
|
||||
```
|
||||
|
||||
Response `201 Created`:
|
||||
|
||||
```json
|
||||
{"name":"myphoto.jpg","size":43234,"eTag":"0x6c656...","mtime":"...","chunks":[...]}
|
||||
```
|
||||
|
||||
Query parameters:
|
||||
|
||||
| Parameter | Description | Default |
|
||||
|-----------|-------------|---------|
|
||||
| `collection` | collection name | empty |
|
||||
| `replication` | replica placement code | filer default |
|
||||
| `ttl` | file expiration, e.g. `3d` | never |
|
||||
| `disk` | disk type to store on | filer default |
|
||||
| `fsync` | `true` fsyncs on the volume server | false |
|
||||
| `dataCenter` | preferred data center | empty |
|
||||
| `rack` | preferred rack | empty |
|
||||
| `dataNode` | preferred volume server | empty |
|
||||
| `saveInside` | store small content inside the metadata instead of a volume | false |
|
||||
| `maxMB` | split the upload into chunks of this many MB | filer `-maxMB` |
|
||||
| `mode` | unix permission bits, e.g. `0644` | `0660` |
|
||||
| `op` | `append` appends to an existing file | overwrite |
|
||||
| `skipCheckParentDir` | `true` skips the parent-directory existence check | false |
|
||||
|
||||
### Create a directory
|
||||
|
||||
```bash
|
||||
curl -X POST "http://localhost:8888/dir/newdir/"
|
||||
```
|
||||
|
||||
A POST to a path ending in `/` with no content creates the directory,
|
||||
including missing parents.
|
||||
|
||||
### Read a file
|
||||
|
||||
```bash
|
||||
curl "http://localhost:8888/dir/myphoto.jpg"
|
||||
```
|
||||
|
||||
Supports `Range` requests (`Accept-Ranges: bytes`), `ETag`, and the
|
||||
`If-None-Match` / `If-Modified-Since` conditional headers. `HEAD` returns
|
||||
headers only. Entry headers stored as extended attributes are echoed back,
|
||||
minus internal `Seaweed-` and `xattr-` keys.
|
||||
|
||||
Entry metadata instead of content:
|
||||
|
||||
```bash
|
||||
curl "http://localhost:8888/dir/myphoto.jpg?metadata=true"
|
||||
```
|
||||
|
||||
`metadata=true&resolveManifest=true` additionally resolves chunked-manifest
|
||||
entries into their real chunk list.
|
||||
|
||||
### List a directory
|
||||
|
||||
```bash
|
||||
curl -H "Accept: application/json" "http://localhost:8888/dir/?limit=10&lastFileName=a.jpg"
|
||||
```
|
||||
|
||||
| Parameter | Description | Default |
|
||||
|-----------|-------------|---------|
|
||||
| `limit` | max entries per page | filer `-dirListLimit` |
|
||||
| `lastFileName` | resume listing after this entry name | empty |
|
||||
| `namePattern` | include only names matching the wildcard | empty |
|
||||
| `namePatternExclude` | exclude names matching the wildcard | empty |
|
||||
|
||||
The JSON response carries `Path`, `Entries`, `Limit`, `LastFileName`,
|
||||
`ShouldDisplayLoadMore`, and `EmptyFolder`. Without the `Accept` header the
|
||||
filer renders its HTML browser.
|
||||
|
||||
### Move and copy
|
||||
|
||||
```bash
|
||||
curl -X POST "http://localhost:8888/dir/newname.jpg?mv.from=/dir/myphoto.jpg"
|
||||
curl -X POST "http://localhost:8888/dir/copy.jpg?cp.from=/dir/myphoto.jpg"
|
||||
```
|
||||
|
||||
`mv.from` renames or moves the source to the request path (`204 No Content`).
|
||||
`cp.from` copies it.
|
||||
|
||||
### Append
|
||||
|
||||
```bash
|
||||
curl -T chunk2.bin "http://localhost:8888/dir/file.bin?op=append"
|
||||
```
|
||||
|
||||
### Delete
|
||||
|
||||
```bash
|
||||
curl -X DELETE "http://localhost:8888/dir/myphoto.jpg"
|
||||
curl -X DELETE "http://localhost:8888/dir/?recursive=true"
|
||||
```
|
||||
|
||||
| Parameter | Description | Default |
|
||||
|-----------|-------------|---------|
|
||||
| `recursive` | delete a non-empty directory tree | false; when the filer runs with `filer.options.recursive_delete=true`, deletes are recursive unless `recursive=false` |
|
||||
| `ignoreRecursiveError` | keep deleting remaining entries after an error | false |
|
||||
| `skipChunkDeletion` | remove only the metadata, keep volume data | false |
|
||||
|
||||
### Tagging
|
||||
|
||||
Tags are carried as `Seaweed-`-prefixed request headers, not query
|
||||
parameters; `?tagging` selects the tagging handler and `?tagging=K1,K2`
|
||||
lists the keys to remove. Header names are canonicalized on write
|
||||
(`Seaweed-k1` is stored as `Seaweed-K1`), and the delete list is matched
|
||||
case-sensitively against the stored names.
|
||||
|
||||
```bash
|
||||
curl -X PUT -H "Seaweed-k1: v1" -H "Seaweed-k2: v2" "http://localhost:8888/dir/file.jpg?tagging"
|
||||
curl -X DELETE "http://localhost:8888/dir/file.jpg?tagging=K1,K2"
|
||||
```
|
||||
|
||||
### Read by file id
|
||||
|
||||
```bash
|
||||
curl "http://localhost:8888/?proxyChunkId=3,01637037d6"
|
||||
```
|
||||
|
||||
The filer proxies the chunk read to the right volume server, so only the
|
||||
filer port needs to be exposed.
|
||||
|
||||
### Resumable uploads
|
||||
|
||||
The filer serves the [TUS protocol](https://tus.io/) for resumable uploads
|
||||
(`POST`, `PATCH`, `HEAD` on upload URLs). It is enabled by default at
|
||||
`/.tus`; `-tusBasePath` changes the endpoint base path.
|
||||
|
||||
### Health
|
||||
|
||||
`GET /healthz` and `GET /readyz` return `200 OK`.
|
||||
|
||||
## Master API (port 9333)
|
||||
|
||||
Write-affecting endpoints are automatically proxied to the current leader, so
|
||||
any master in the quorum can serve them.
|
||||
|
||||
### Assign a file id
|
||||
|
||||
```bash
|
||||
curl "http://localhost:9333/dir/assign?count=1&replication=001&collection=turbo&dataCenter=dc1&ttl=3d&disk=ssd"
|
||||
{"count":1,"fid":"3,01637037d6","url":"127.0.0.1:8080","publicUrl":"localhost:8080"}
|
||||
```
|
||||
|
||||
Upload the file content to `http://<url>/<fid>` afterwards. With `count>1`,
|
||||
use `<fid>_1`, `<fid>_2`, ... for the additional ids.
|
||||
|
||||
| Parameter | Description | Default |
|
||||
|-----------|-------------|---------|
|
||||
| `count` | file ids to reserve | 1 |
|
||||
| `collection` | collection name | empty |
|
||||
| `dataCenter` | preferred data center | empty |
|
||||
| `rack` | preferred rack | empty |
|
||||
| `dataNode` | preferred volume server | empty |
|
||||
| `replication` | replica placement | master `-defaultReplication` |
|
||||
| `ttl` | file expiration, e.g. `3d` | never |
|
||||
| `disk` | disk type | empty |
|
||||
| `dataSize` | expected file size in bytes | 0 |
|
||||
| `preallocate` | bytes to preallocate for new volumes | master `-volumePreallocate` |
|
||||
| `writableVolumeCount` | grow this many volumes when none are writable | master default |
|
||||
| `memoryMapMaxSizeMb` | memory-mapped file size (Windows) | 0 |
|
||||
|
||||
### Look up a volume or file id
|
||||
|
||||
```bash
|
||||
curl "http://localhost:9333/dir/lookup?volumeId=3"
|
||||
{"locations":[{"url":"localhost:8080","publicUrl":"localhost:8080"}]}
|
||||
```
|
||||
|
||||
| Parameter | Description | Default |
|
||||
|-----------|-------------|---------|
|
||||
| `volumeId` | volume id; a full `vid,fid` is accepted too | required |
|
||||
| `fileId` | like `volumeId`, but also returns a write JWT when security is on | empty |
|
||||
| `collection` | speeds up the lookup | empty |
|
||||
| `read` | `yes` generates a read JWT instead of a write JWT | empty |
|
||||
|
||||
### Store a file in one call
|
||||
|
||||
```bash
|
||||
curl -F file=@/home/chris/report.pdf "http://localhost:9333/submit?collection=turbo&replication=001"
|
||||
{"fileName":"report.pdf","fid":"3,01637037d6","fileUrl":"localhost:8080/3,01637037d6","size":43234,"eTag":"0x6c656..."}
|
||||
```
|
||||
|
||||
`POST /submit` accepts multipart file data plus the `dir/assign` placement
|
||||
parameters (`count`, `collection`, `dataCenter`, `rack`, `replication`,
|
||||
`ttl`, `disk`), assigns a file id, uploads to the volume server, and returns
|
||||
the result.
|
||||
|
||||
### Redirect to a file
|
||||
|
||||
```bash
|
||||
curl -v "http://localhost:9333/3,01637037d6"
|
||||
```
|
||||
|
||||
`GET /{fileId}` answers `308 Permanent Redirect` to a volume server holding
|
||||
the file, preserving the query string (e.g. image-resize parameters).
|
||||
|
||||
### Cluster status
|
||||
|
||||
```bash
|
||||
curl "http://localhost:9333/dir/status?pretty=y" # full topology tree
|
||||
curl "http://localhost:9333/vol/status?pretty=y" # every volume on every node
|
||||
curl "http://localhost:9333/collection/info?collection=turbo"
|
||||
curl "http://localhost:9333/collection/info?collection=turbo&detail=true"
|
||||
```
|
||||
|
||||
`collection/info` returns aggregated `TotalSize`, `FileCount`, `UsedSize`,
|
||||
`VolumeCount`; `detail=true` splits them per volume layout.
|
||||
|
||||
### Grow volumes
|
||||
|
||||
```bash
|
||||
curl "http://localhost:9333/vol/grow?count=4&replication=001&collection=turbo&ttl=5d&disk=ssd&dataCenter=dc1&rack=rack1"
|
||||
{"count":4}
|
||||
```
|
||||
|
||||
`count` is required; the placement parameters match `dir/assign`. One volume
|
||||
serves one write at a time, so pre-allocated volumes raise write concurrency.
|
||||
|
||||
### Vacuum deleted space
|
||||
|
||||
```bash
|
||||
curl "http://localhost:9333/vol/vacuum?garbageThreshold=0.4"
|
||||
```
|
||||
|
||||
| Parameter | Description | Default |
|
||||
|-----------|-------------|---------|
|
||||
| `garbageThreshold` | minimum deleted-bytes ratio before a volume is compacted | master `-garbageThreshold` (0.3) |
|
||||
|
||||
Vacuuming makes a volume read-only, copies live needles to a new volume, and
|
||||
swaps it in.
|
||||
|
||||
### Delete a collection
|
||||
|
||||
```bash
|
||||
curl "http://localhost:9333/col/delete?collection=benchmark"
|
||||
```
|
||||
|
||||
Deletes all volumes of the collection, including erasure-coded shards.
|
||||
`204 No Content` on success.
|
||||
|
||||
### Health
|
||||
|
||||
```bash
|
||||
curl -I "http://localhost:9333/healthz" # liveness
|
||||
curl -I "http://localhost:9333/readyz" # readiness
|
||||
curl "http://localhost:9333/" # web UI
|
||||
```
|
||||
|
||||
## Volume server API (port 8080)
|
||||
|
||||
The volume server stores file content by file id. Clients normally get the
|
||||
volume URL from `dir/assign` or `dir/lookup`.
|
||||
|
||||
### Upload
|
||||
|
||||
```bash
|
||||
curl -F file=@/home/chris/myphoto.jpg "http://127.0.0.1:8080/3,01637037d6"
|
||||
{"name":"myphoto.jpg","size":43234,"eTag":"0x6c656...","mime":"image/jpeg","contentMd5":"..."}
|
||||
```
|
||||
|
||||
PUT or POST the body (or a multipart `file` part) to `/{vid},{fid}`.
|
||||
`204 No Content` is returned when the content is unchanged. `?ts=<unix>`
|
||||
sets the stored modification time.
|
||||
|
||||
### Read
|
||||
|
||||
```bash
|
||||
curl "http://127.0.0.1:8080/3,01637037d6"
|
||||
curl "http://127.0.0.1:8080/3,01637037d6.jpg" # sets Content-Type from the extension
|
||||
```
|
||||
|
||||
Supports `Range` and `HEAD`. Image files can be resized server-side:
|
||||
|
||||
| Parameter | Description |
|
||||
|-----------|-------------|
|
||||
| `width`, `height` | resize bounds in pixels |
|
||||
| `mode` | `fit` (contain) or `fill` (cover); omitted resizes to `width`/`height` |
|
||||
| `crop_x1`, `crop_y1`, `crop_x2`, `crop_y2` | explicit crop rectangle |
|
||||
| `cm` | `false` returns the chunk-manifest blob instead of resolving it |
|
||||
| `readDeleted` | `true` reads soft-deleted needles |
|
||||
| `collection` | passed through redirects for the right volume |
|
||||
|
||||
### Delete
|
||||
|
||||
```bash
|
||||
curl -X DELETE "http://127.0.0.1:8080/3,01637037d6"
|
||||
{"size":43234}
|
||||
```
|
||||
|
||||
`?ts=<unix>` sets the deletion timestamp. Replicated volumes propagate the
|
||||
delete to every replica.
|
||||
|
||||
### Status
|
||||
|
||||
```bash
|
||||
curl "http://localhost:8080/status?pretty=y" # disk and volume inventory
|
||||
curl -I "http://localhost:8080/healthz" # liveness/readiness
|
||||
```
|
||||
|
||||
`OPTIONS` preflights answer CORS headers. When `-port.public` differs from
|
||||
`-port`, the volume server opens a separate read-only public listener on
|
||||
that port; `-publicUrl` sets the address it advertises to clients.
|
||||
@@ -14,6 +14,9 @@ RUN cd /go/src/github.com/seaweedfs/seaweedfs && \
|
||||
git checkout $BRANCH) || \
|
||||
(echo "ERROR: Branch/commit $BRANCH not found in repository" && \
|
||||
echo "Available branches:" && git branch -a && exit 1))
|
||||
# seaweed-common only exists on revisions that have it; a BRANCH predating it
|
||||
# still needs the directory so the COPY into rust_builder below never fails.
|
||||
RUN mkdir -p /go/src/github.com/seaweedfs/seaweedfs/seaweed-common
|
||||
ARG TARGETOS TARGETARCH TARGETVARIANT
|
||||
RUN cd /go/src/github.com/seaweedfs/seaweedfs/weed \
|
||||
&& export LDFLAGS="-X github.com/seaweedfs/seaweedfs/weed/util/version.COMMIT=$(git rev-parse --short HEAD)" \
|
||||
@@ -31,6 +34,9 @@ ARG TAGS
|
||||
COPY weed-volume-prebuilt/ /prebuilt/
|
||||
COPY weed-worker-prebuilt/ /prebuilt-worker/
|
||||
COPY --from=builder /go/src/github.com/seaweedfs/seaweedfs/seaweed-volume /build/seaweed-volume
|
||||
# seaweed-common is a path dependency of seaweed-volume that lives beside it,
|
||||
# so the source build below needs it in the same relative position.
|
||||
COPY --from=builder /go/src/github.com/seaweedfs/seaweedfs/seaweed-common /build/seaweed-common
|
||||
COPY --from=builder /go/src/github.com/seaweedfs/seaweedfs/weed /build/weed
|
||||
WORKDIR /build/seaweed-volume
|
||||
RUN if [ -f "/prebuilt/weed-volume-${TARGETARCH}" ]; then \
|
||||
|
||||
@@ -70,7 +70,7 @@ require (
|
||||
github.com/spf13/afero v1.15.0 // indirect
|
||||
github.com/spf13/cast v1.10.0 // indirect
|
||||
github.com/spf13/viper v1.21.0
|
||||
github.com/stretchr/testify v1.11.1
|
||||
github.com/stretchr/testify v1.12.1
|
||||
github.com/stvp/tempredis v0.0.0-20181119212430-b82af8480203
|
||||
github.com/syndtr/goleveldb v1.0.1-0.20190318030020-c3a204f8e965
|
||||
github.com/tidwall/gjson v1.18.0
|
||||
@@ -90,7 +90,7 @@ require (
|
||||
gocloud.dev v0.46.0
|
||||
gocloud.dev/pubsub/natspubsub v0.46.0
|
||||
gocloud.dev/pubsub/rabbitpubsub v0.46.0
|
||||
golang.org/x/crypto v0.56.0
|
||||
golang.org/x/crypto v0.57.0
|
||||
golang.org/x/exp v0.0.0-20260709172345-9ea1abe57597
|
||||
golang.org/x/image v0.46.0
|
||||
golang.org/x/net v0.58.0
|
||||
@@ -111,7 +111,7 @@ require (
|
||||
)
|
||||
|
||||
require (
|
||||
cloud.google.com/go/kms v1.33.0
|
||||
cloud.google.com/go/kms v1.34.0
|
||||
github.com/Azure/azure-sdk-for-go/sdk/keyvault/azkeys v0.10.0
|
||||
github.com/DATA-DOG/go-sqlmock v1.5.2
|
||||
github.com/Jille/raft-grpc-transport v1.6.1
|
||||
@@ -151,7 +151,7 @@ require (
|
||||
github.com/seaweedfs/go-fuse/v2 v2.9.4
|
||||
github.com/shirou/gopsutil/v4 v4.26.7
|
||||
github.com/tarantool/go-option v1.1.0
|
||||
github.com/tarantool/go-tarantool/v3 v3.0.1
|
||||
github.com/tarantool/go-tarantool/v3 v3.0.2
|
||||
github.com/testcontainers/testcontainers-go v0.44.0
|
||||
github.com/tikv/client-go/v2 v2.0.7
|
||||
github.com/twmb/avro v1.8.0
|
||||
@@ -159,7 +159,7 @@ require (
|
||||
github.com/ydb-platform/ydb-go-sdk-auth-environ v0.5.2
|
||||
github.com/ydb-platform/ydb-go-sdk/v3 v3.151.1
|
||||
go.etcd.io/etcd/client/pkg/v3 v3.7.1
|
||||
go.uber.org/atomic v1.11.0
|
||||
go.uber.org/atomic v1.12.0
|
||||
golang.org/x/sync v0.23.0
|
||||
golang.org/x/tools/godoc v0.1.0-deprecated
|
||||
google.golang.org/grpc/security/advancedtls v1.0.0
|
||||
@@ -289,7 +289,7 @@ require (
|
||||
go.opentelemetry.io/proto/otlp v1.11.0 // indirect
|
||||
go.uber.org/mock v0.5.2 // indirect
|
||||
go.yaml.in/yaml/v2 v2.4.4 // indirect
|
||||
go.yaml.in/yaml/v3 v3.0.4 // indirect
|
||||
go.yaml.in/yaml/v3 v3.0.5 // indirect
|
||||
golang.org/x/mod v0.41.0 // indirect
|
||||
gonum.org/v1/gonum v0.17.0 // indirect
|
||||
)
|
||||
@@ -340,7 +340,7 @@ require (
|
||||
github.com/aws/aws-sdk-go-v2/service/sqs v1.42.24 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/sso v1.38.0 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/ssooidc v1.43.0 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/sts v1.50.0
|
||||
github.com/aws/aws-sdk-go-v2/service/sts v1.51.0
|
||||
github.com/aws/smithy-go v1.28.1
|
||||
github.com/boltdb/bolt v1.3.1 // indirect
|
||||
github.com/bradenaw/juniper v0.15.3 // indirect
|
||||
@@ -499,7 +499,7 @@ require (
|
||||
go.opentelemetry.io/otel/trace v1.45.0 // indirect
|
||||
go.uber.org/multierr v1.11.0 // indirect
|
||||
go.uber.org/zap v1.27.1 // indirect
|
||||
golang.org/x/term v0.45.0
|
||||
golang.org/x/term v0.46.0
|
||||
golang.org/x/time v0.15.0
|
||||
google.golang.org/genproto/googleapis/api v0.0.0-20260817212433-ac3dfec99bb1 // indirect
|
||||
google.golang.org/genproto/googleapis/rpc v0.0.0-20260819154853-08b0e4226688 // indirect
|
||||
|
||||
@@ -298,8 +298,8 @@ cloud.google.com/go/kms v1.4.0/go.mod h1:fajBHndQ+6ubNw6Ss2sSd+SWvjL26RNo/dr7uxs
|
||||
cloud.google.com/go/kms v1.5.0/go.mod h1:QJS2YY0eJGBg3mnDfuaCyLauWwBJiHRboYxJ++1xJNg=
|
||||
cloud.google.com/go/kms v1.6.0/go.mod h1:Jjy850yySiasBUDi6KFUwUv2n1+o7QZFyuUJg6OgjA0=
|
||||
cloud.google.com/go/kms v1.9.0/go.mod h1:qb1tPTgfF9RQP8e1wq4cLFErVuTJv7UsSC915J8dh3w=
|
||||
cloud.google.com/go/kms v1.33.0 h1:pG0X78m212b2pv9N4fdMoUO69LuZGQ9kSvn8sHBOFAo=
|
||||
cloud.google.com/go/kms v1.33.0/go.mod h1:CSGvW6GnMQbY+1nOHcIzhMtHSbExXlOmCKjWtYVjcpA=
|
||||
cloud.google.com/go/kms v1.34.0 h1:mxWcXEiyjxwFH5gclulLx+B8Y2OEpKJRZ5FOF78c2XE=
|
||||
cloud.google.com/go/kms v1.34.0/go.mod h1:FbxZWUiihmyjxlaBha84OK5+fmJHPrS6F5/mBFdJk6A=
|
||||
cloud.google.com/go/language v1.4.0/go.mod h1:F9dRpNFQmJbkaop6g0JhSBXCNlO90e1KWx5iDdxbWic=
|
||||
cloud.google.com/go/language v1.6.0/go.mod h1:6dJ8t3B+lUYfStgls25GusK04NLh3eDLQnWM3mdEbhI=
|
||||
cloud.google.com/go/language v1.7.0/go.mod h1:DJ6dYN/W+SQOjF8e1hLQXMF21AkH2w9wiPzPCJa2MIE=
|
||||
@@ -750,8 +750,8 @@ github.com/aws/aws-sdk-go-v2/service/sso v1.38.0 h1:JGeeBcMlhg1xtOXYpeCaTQBZObtX
|
||||
github.com/aws/aws-sdk-go-v2/service/sso v1.38.0/go.mod h1:XwteswG9EOMRFm73UT0t+MbTwyLxMrEXkU6e+v92Lzo=
|
||||
github.com/aws/aws-sdk-go-v2/service/ssooidc v1.43.0 h1:obhahQXDEdVEv8y5bTKXR30LVaxYe1kyYM0L7l2Iq+k=
|
||||
github.com/aws/aws-sdk-go-v2/service/ssooidc v1.43.0/go.mod h1:6twZZ/aXHNy1vXUO8koUbp++MYzMASkOgEBdkbJYmO0=
|
||||
github.com/aws/aws-sdk-go-v2/service/sts v1.50.0 h1:khXV3+K5D3f4e8xtplaRdSFn1bEg3gj5EBHQvbCOZbQ=
|
||||
github.com/aws/aws-sdk-go-v2/service/sts v1.50.0/go.mod h1:/8JRcdTt//hG0Q4BTmGbuOplT7ABe+5rdtqUHqXvYIM=
|
||||
github.com/aws/aws-sdk-go-v2/service/sts v1.51.0 h1:Zpnqa6XtrNzXZnwbdCqHOXpXhMsa01ql/pcRQ1sb4hk=
|
||||
github.com/aws/aws-sdk-go-v2/service/sts v1.51.0/go.mod h1:/8JRcdTt//hG0Q4BTmGbuOplT7ABe+5rdtqUHqXvYIM=
|
||||
github.com/aws/smithy-go v1.28.1 h1:R/nXH00c8qcfCzQVELtRw+eLQWtzv+VAIEFJ1/xxXlQ=
|
||||
github.com/aws/smithy-go v1.28.1/go.mod h1:YE2RhdIuDbA5E5bTdciG9KrW3+TiEONeUWCqxX9i1Fc=
|
||||
github.com/bahlo/generic-list-go v0.2.0 h1:5sz/EEAK+ls5wF+NeqDpk5+iNdMDXrh3z3nPnH1Wvgk=
|
||||
@@ -1897,8 +1897,8 @@ github.com/stretchr/testify v1.8.1/go.mod h1:w2LPCIKwWwSfY2zedu0+kehJoqGctiVI29o
|
||||
github.com/stretchr/testify v1.8.2/go.mod h1:w2LPCIKwWwSfY2zedu0+kehJoqGctiVI29o6fzry7u4=
|
||||
github.com/stretchr/testify v1.8.3/go.mod h1:sz/lmYIOXD/1dqDmKjjqLyZ2RngseejIcXlSw2iwfAo=
|
||||
github.com/stretchr/testify v1.8.4/go.mod h1:sz/lmYIOXD/1dqDmKjjqLyZ2RngseejIcXlSw2iwfAo=
|
||||
github.com/stretchr/testify v1.11.1 h1:7s2iGBzp5EwR7/aIZr8ao5+dra3wiQyKjjFuvgVKu7U=
|
||||
github.com/stretchr/testify v1.11.1/go.mod h1:wZwfW3scLgRK+23gO65QZefKpKQRnfz6sD981Nm4B6U=
|
||||
github.com/stretchr/testify v1.12.1 h1:EuwCh5fleGS7H32xRwO3wRGT7DxrDhLAT6FF8MpWDWE=
|
||||
github.com/stretchr/testify v1.12.1/go.mod h1:MDEgiDPPsNp5cuIrHPPCyornHKgEVbtFUmoNlxoYthg=
|
||||
github.com/stvp/tempredis v0.0.0-20181119212430-b82af8480203 h1:QVqDTf3h2WHt08YuiTGPZLls0Wq99X9bWd0Q5ZSBesM=
|
||||
github.com/stvp/tempredis v0.0.0-20181119212430-b82af8480203/go.mod h1:oqN97ltKNihBbwlX8dLpwxCl3+HnXKV/R0e+sRLd9C8=
|
||||
github.com/subosito/gotenv v1.6.0 h1:9NlTDc1FTs4qu0DDq7AEtTPNw6SVm7uBMsUCUjABIf8=
|
||||
@@ -1918,8 +1918,8 @@ github.com/tarantool/go-iproto v1.1.0 h1:HULVOIHsiehI+FnHfM7wMDntuzUddO09DKqu2Wn
|
||||
github.com/tarantool/go-iproto v1.1.0/go.mod h1:LNCtdyZxojUed8SbOiYHoc3v9NvaZTB7p96hUySMlIo=
|
||||
github.com/tarantool/go-option v1.1.0 h1:ShoOhNsdL41sRpm4hXCRDjV8H0WzPkd4UnKhLKbW//w=
|
||||
github.com/tarantool/go-option v1.1.0/go.mod h1:hMr9z2JXOWlgdCBpCPSL2nwp8718GKYvNBJ+ZuzJbCo=
|
||||
github.com/tarantool/go-tarantool/v3 v3.0.1 h1:vaUX4xmVmXh2dIJ/LqlX1MXK3iYqAqV6YiE54Wwl/qg=
|
||||
github.com/tarantool/go-tarantool/v3 v3.0.1/go.mod h1:TXxLWhUCgdxXFfelnTSkq+goKRTRj660zxq4/WXPe8k=
|
||||
github.com/tarantool/go-tarantool/v3 v3.0.2 h1:9ZtHllun80QX7KS9tqd3dXa+QAu2BfqFtX1ZWcNnaW8=
|
||||
github.com/tarantool/go-tarantool/v3 v3.0.2/go.mod h1:TXxLWhUCgdxXFfelnTSkq+goKRTRj660zxq4/WXPe8k=
|
||||
github.com/testcontainers/testcontainers-go v0.44.0 h1:/Fwh6HY1mIikhnm9e7HwoxGycx0lzRAE0f5VQpjFxzI=
|
||||
github.com/testcontainers/testcontainers-go v0.44.0/go.mod h1:IcnwQrYTO86xHXu5bvMaBH7ATlbS3Qn1M1QWW3c66rE=
|
||||
github.com/testcontainers/testcontainers-go/modules/compose v0.44.0 h1:8YcW51jhgpkkiRVe10Wj9TCBthJmoNpU2fK5WSf7TQ8=
|
||||
@@ -2145,8 +2145,8 @@ go.opentelemetry.io/proto/otlp v1.11.0/go.mod h1:SmVizdCOAm3XBtG1g1NnOdhW6jtddT7
|
||||
go.uber.org/atomic v1.6.0/go.mod h1:sABNBOSYdrvTF6hTgEIbc7YasKWGhgEQZyfxyTvoXHQ=
|
||||
go.uber.org/atomic v1.7.0/go.mod h1:fEN4uk6kAWBTFdckzkM89CLk9XfWZrxpCo0nPH17wJc=
|
||||
go.uber.org/atomic v1.9.0/go.mod h1:fEN4uk6kAWBTFdckzkM89CLk9XfWZrxpCo0nPH17wJc=
|
||||
go.uber.org/atomic v1.11.0 h1:ZvwS0R+56ePWxUNi+Atn9dWONBPp/AUETXlHW0DxSjE=
|
||||
go.uber.org/atomic v1.11.0/go.mod h1:LUxbIzbOniOlMKjJjyPfpl4v+PKK2cNJn91OQbhoJI0=
|
||||
go.uber.org/atomic v1.12.0 h1:BvcXdFKuviU4fTL/f+SxdQ5qJX/Jix8pAkgdUcb3XOE=
|
||||
go.uber.org/atomic v1.12.0/go.mod h1:I6c4cg+6HCxRjfjSsYtApoFILnpc0CGUdGkXVqbYVNk=
|
||||
go.uber.org/goleak v1.1.10/go.mod h1:8a7PlsEVH3e/a/GLqe5IIrQx6GzcnRmZEufDUTk4A7A=
|
||||
go.uber.org/goleak v1.1.12/go.mod h1:cwTWslyiVhfpKIDGSZEM2HlOvcqm+tG4zioyIeLoqMQ=
|
||||
go.uber.org/goleak v1.3.0 h1:2K3zAYmnTNqV73imy9J1T3WC+gmCePx2hEGkimedGto=
|
||||
@@ -2163,8 +2163,8 @@ go.uber.org/zap v1.27.1 h1:08RqriUEv8+ArZRYSTXy1LeBScaMpVSTBhCeaZYfMYc=
|
||||
go.uber.org/zap v1.27.1/go.mod h1:GB2qFLM7cTU87MWRP2mPIjqfIDnGu+VIO4V/SdhGo2E=
|
||||
go.yaml.in/yaml/v2 v2.4.4 h1:tuyd0P+2Ont/d6e2rl3be67goVK4R6deVxCUX5vyPaQ=
|
||||
go.yaml.in/yaml/v2 v2.4.4/go.mod h1:gMZqIpDtDqOfM0uNfy0SkpRhvUryYH0Z6wdMYcacYXQ=
|
||||
go.yaml.in/yaml/v3 v3.0.4 h1:tfq32ie2Jv2UxXFdLJdh3jXuOzWiL1fo0bu/FbuKpbc=
|
||||
go.yaml.in/yaml/v3 v3.0.4/go.mod h1:DhzuOOF2ATzADvBadXxruRBLzYTpT36CKvDb3+aBEFg=
|
||||
go.yaml.in/yaml/v3 v3.0.5 h1:N6y/pJk8buWs9NY5ERU2HSMfm+IuD/OtfdAnq6kESPw=
|
||||
go.yaml.in/yaml/v3 v3.0.5/go.mod h1:HVTZu1O7/Vkt2N+BFy8Zza+lnLsABggaTM2ZpNIGuKg=
|
||||
go.yaml.in/yaml/v4 v4.0.0-rc.6 h1:1h7H1ohdUh93/FyE4YaDa1Zh64K6VVbjF4K6WUxMtH4=
|
||||
go.yaml.in/yaml/v4 v4.0.0-rc.6/go.mod h1:aZqd9kCMsGL7AuUv/m/PvWLdg5sjJsZ4oHDEnfPPfY0=
|
||||
gocloud.dev v0.46.0 h1:niIuZwSjMtBx8K+ITB2s5kZullB13PGOS2ZoQPZxQ4Q=
|
||||
@@ -2193,8 +2193,8 @@ golang.org/x/crypto v0.6.0/go.mod h1:OFC/31mSvZgRz0V1QTNCzfAI1aIRzbiufJtkMIlEp58
|
||||
golang.org/x/crypto v0.7.0/go.mod h1:pYwdfH91IfpZVANVyUOhSIPZaFoJGxTFbZhFTx+dXZU=
|
||||
golang.org/x/crypto v0.13.0/go.mod h1:y6Z2r+Rw4iayiXXAIxJIDAJ1zMW4yaTpebo8fPOliYc=
|
||||
golang.org/x/crypto v0.14.0/go.mod h1:MVFd36DqK4CsrnJYDkBA3VC4m2GkXAM0PvzMCn4JQf4=
|
||||
golang.org/x/crypto v0.56.0 h1:GUh5Ii4J5jtcseSMiRqr1jXCNHoxjeV9Fmekc2oLy6Y=
|
||||
golang.org/x/crypto v0.56.0/go.mod h1:OMW5y6CY9l38uPLmxU6l6pwcXp1obtLo3e6gT7gQR2I=
|
||||
golang.org/x/crypto v0.57.0 h1:3ZVCjf8Ggz7zneR/EHRVx68Ctf+2pmIMP2UFhh9cC6M=
|
||||
golang.org/x/crypto v0.57.0/go.mod h1:Fdz0i5U6CoizGwLda9DttjSk6qlZo25zYNtR+ycvuZA=
|
||||
golang.org/x/exp v0.0.0-20180321215751-8460e604b9de/go.mod h1:CJ0aWSM057203Lf6IL+f9T1iT9GByDxfZKAQTCR3kQA=
|
||||
golang.org/x/exp v0.0.0-20180807140117-3d87b88a115f/go.mod h1:CJ0aWSM057203Lf6IL+f9T1iT9GByDxfZKAQTCR3kQA=
|
||||
golang.org/x/exp v0.0.0-20190121172915-509febef88a4/go.mod h1:CJ0aWSM057203Lf6IL+f9T1iT9GByDxfZKAQTCR3kQA=
|
||||
@@ -2491,8 +2491,8 @@ golang.org/x/term v0.6.0/go.mod h1:m6U89DPEgQRMq3DNkDClhWw02AUbt2daBVO4cn4Hv9U=
|
||||
golang.org/x/term v0.8.0/go.mod h1:xPskH00ivmX89bAKVGSKKtLOWNx2+17Eiy94tnKShWo=
|
||||
golang.org/x/term v0.12.0/go.mod h1:owVbMEjm3cBLCHdkQu9b1opXd4ETQWc3BhuQGKgXgvU=
|
||||
golang.org/x/term v0.13.0/go.mod h1:LTmsnFJwVN6bCy1rVCoS+qHT1HhALEFxKncY3WNNh4U=
|
||||
golang.org/x/term v0.45.0 h1:NwWyBmoJCbfTHpxrWoZ9C6/VxOf7ic219I8xZZFdrf0=
|
||||
golang.org/x/term v0.45.0/go.mod h1:9aqxs0blBcrm/n0L9QW0aRVD+ktan8ssZromtqJC43w=
|
||||
golang.org/x/term v0.46.0 h1:3+OXuTbaKDgwk8jTi3aSLHRlmWqHEUDUtxnbFigO4YE=
|
||||
golang.org/x/term v0.46.0/go.mod h1:+K02xbkittuwc0Am4abfA3Fc+XRGXkvBXNO88NCXPoc=
|
||||
golang.org/x/text v0.0.0-20170915032832-14c0d48ead0c/go.mod h1:NqM8EUOU14njkJ3fqMW+pc6Ldnwhi/IjpwHt7yyuwOQ=
|
||||
golang.org/x/text v0.3.0/go.mod h1:NqM8EUOU14njkJ3fqMW+pc6Ldnwhi/IjpwHt7yyuwOQ=
|
||||
golang.org/x/text v0.3.1-0.20180807135948-17ff2d5776d2/go.mod h1:NqM8EUOU14njkJ3fqMW+pc6Ldnwhi/IjpwHt7yyuwOQ=
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
apiVersion: v1
|
||||
description: SeaweedFS
|
||||
name: seaweedfs
|
||||
appVersion: "4.47"
|
||||
appVersion: "4.48"
|
||||
# Dev note: Trigger a helm chart release by `git tag -a helm-<version>`
|
||||
version: 4.47.1
|
||||
version: 4.48.0
|
||||
|
||||
@@ -376,7 +376,10 @@ start on a non-loopback address without `-adminPassword`, so the chart fails at
|
||||
render time if `admin.ip` is non-loopback and authentication is not configured via
|
||||
`admin.secret.adminPassword`, `admin.secret.existingSecret`, or
|
||||
`WEED_ADMIN_PASSWORD` supplied through `admin.extraEnvironmentVars` /
|
||||
`admin.secretExtraEnvironmentVars`. The whole `127.0.0.0/8` range and `::1` are
|
||||
`admin.secretExtraEnvironmentVars`. Setting `admin.allowInsecureBind` renders
|
||||
`-allowInsecureBind` and bypasses this guard; it leaves the admin API
|
||||
unauthenticated on the network, so use it only when access is otherwise
|
||||
restricted (e.g. network policies). The whole `127.0.0.0/8` range and `::1` are
|
||||
treated as loopback (matching `weed admin`); `localhost` is treated as
|
||||
non-loopback. Set `admin.ip` to a loopback address only if you also replace the
|
||||
httpGet probes (e.g. with an `exec` probe that checks `127.0.0.1`).
|
||||
|
||||
@@ -9,7 +9,7 @@
|
||||
{{- $adminAuthEnabled := include "seaweedfs.admin.authEnabled" . }}
|
||||
{{- $adminIp := .Values.admin.ip | default "0.0.0.0" }}
|
||||
{{- if and (not (include "seaweedfs.admin.isLoopbackIp" $adminIp)) (ne $adminAuthEnabled "true") }}
|
||||
{{- fail (printf "admin.ip is set to %q (non-loopback) but admin authentication is not configured. Since `weed admin` 4.46 refuses to bind a non-loopback address without authentication, the admin container would exit on startup. Set admin.secret.adminPassword or admin.secret.existingSecret, or supply WEED_ADMIN_PASSWORD via admin.extraEnvironmentVars / admin.secretExtraEnvironmentVars, or set admin.ip to a loopback address such as 127.0.0.1 (note: a loopback bind makes the chart's httpGet readiness/liveness probes fail)." $adminIp) -}}
|
||||
{{- fail (printf "admin.ip is set to %q (non-loopback) but admin authentication is not configured. Since `weed admin` 4.46 refuses to bind a non-loopback address without authentication, the admin container would exit on startup. Set admin.secret.adminPassword or admin.secret.existingSecret, or supply WEED_ADMIN_PASSWORD via admin.extraEnvironmentVars / admin.secretExtraEnvironmentVars, or set admin.ip to a loopback address such as 127.0.0.1 (note: a loopback bind makes the chart's httpGet readiness/liveness probes fail), or set admin.allowInsecureBind to true to opt out via -allowInsecureBind (INSECURE: exposes the admin API unauthenticated on the network)." $adminIp) -}}
|
||||
{{- end }}
|
||||
apiVersion: apps/v1
|
||||
kind: StatefulSet
|
||||
@@ -176,14 +176,17 @@ spec:
|
||||
-dataDir={{ .Values.admin.dataDir }} \
|
||||
{{- end }}
|
||||
{{- if .Values.admin.masters }}
|
||||
-masters={{ .Values.admin.masters }}{{- if or $urlPrefix .Values.admin.extraArgs }} \{{ end }}
|
||||
-masters={{ .Values.admin.masters }} \
|
||||
{{- else if .Values.global.seaweedfs.masterServer }}
|
||||
-masters={{ .Values.global.seaweedfs.masterServer }}{{- if or $urlPrefix .Values.admin.extraArgs }} \{{ end }}
|
||||
-masters={{ .Values.global.seaweedfs.masterServer }} \
|
||||
{{- else }}
|
||||
-masters={{ range $index := until (.Values.master.replicas | int) }}${SEAWEEDFS_FULLNAME}-master-{{ $index }}.${SEAWEEDFS_FULLNAME}-master.{{ $.Release.Namespace }}:{{ $.Values.master.port }}{{ if lt $index (sub ($.Values.master.replicas | int) 1) }},{{ end }}{{ end }}{{- if or $urlPrefix .Values.admin.extraArgs }} \{{ end }}
|
||||
-masters={{ range $index := until (.Values.master.replicas | int) }}${SEAWEEDFS_FULLNAME}-master-{{ $index }}.${SEAWEEDFS_FULLNAME}-master.{{ $.Release.Namespace }}:{{ $.Values.master.port }}{{ if lt $index (sub ($.Values.master.replicas | int) 1) }},{{ end }}{{ end }} \
|
||||
{{- end }}
|
||||
{{- if $urlPrefix }}
|
||||
-urlPrefix={{ $urlPrefix }}{{- if .Values.admin.extraArgs }} \{{ end }}
|
||||
-urlPrefix={{ $urlPrefix }} \
|
||||
{{- end }}
|
||||
{{- if .Values.admin.allowInsecureBind }}
|
||||
-allowInsecureBind \
|
||||
{{- end }}
|
||||
{{- range $index, $arg := .Values.admin.extraArgs }}
|
||||
{{ $arg }}{{- if lt $index (sub (len $.Values.admin.extraArgs) 1) }} \{{ end }}
|
||||
|
||||
@@ -105,13 +105,13 @@ true
|
||||
{{- end -}}
|
||||
{{- end -}}
|
||||
|
||||
{{/* Whether admin authentication is enabled from any supported source:
|
||||
admin.secret (adminPassword or existingSecret), or WEED_ADMIN_PASSWORD
|
||||
supplied via extraEnvironmentVars / secretExtraEnvironmentVars (which
|
||||
weed admin picks up through viper's AutomaticEnv). A secret-backed
|
||||
entry counts as enabled even though the chart cannot read its value. */}}
|
||||
{{/* Whether the admin non-loopback bind guard is satisfied: admin.secret
|
||||
(adminPassword or existingSecret), WEED_ADMIN_PASSWORD via
|
||||
extraEnvironmentVars / secretExtraEnvironmentVars, or
|
||||
admin.allowInsecureBind. A secret-backed entry counts as enabled even
|
||||
though the chart cannot read its value. */}}
|
||||
{{- define "seaweedfs.admin.authEnabled" -}}
|
||||
{{- if or .Values.admin.secret.existingSecret .Values.admin.secret.adminPassword -}}
|
||||
{{- if or .Values.admin.secret.existingSecret .Values.admin.secret.adminPassword .Values.admin.allowInsecureBind -}}
|
||||
true
|
||||
{{- else -}}
|
||||
{{- $merged := dict -}}
|
||||
|
||||
@@ -1339,10 +1339,11 @@ admin:
|
||||
# kubelet's httpGet readiness/liveness probes (which dial the pod IP) to ever
|
||||
# succeed. "0.0.0.0" restores the pre-4.46 behaviour of listening on all
|
||||
# interfaces. A non-loopback address requires authentication: set
|
||||
# admin.secret.adminPassword or admin.secret.existingSecret, or supply
|
||||
# admin.secret.adminPassword or admin.secret.existingSecret, supply
|
||||
# WEED_ADMIN_PASSWORD via admin.extraEnvironmentVars /
|
||||
# admin.secretExtraEnvironmentVars; otherwise the admin container will exit
|
||||
# with a clear error rather than silently staying unready. The whole
|
||||
# admin.secretExtraEnvironmentVars, or opt out with admin.allowInsecureBind;
|
||||
# otherwise the admin container will exit with a clear error rather than
|
||||
# silently staying unready. The whole
|
||||
# 127.0.0.0/8 range and ::1 are treated as loopback (matching weed admin).
|
||||
# Set to a loopback address only if you also replace the httpGet probes.
|
||||
# Note: the -ip flag requires SeaweedFS 4.46 or newer; pinning
|
||||
@@ -1350,6 +1351,9 @@ admin:
|
||||
ip: "0.0.0.0"
|
||||
loggingOverrideLevel: null
|
||||
|
||||
# INSECURE: allow binding a non-loopback ip without authentication.
|
||||
allowInsecureBind: false
|
||||
|
||||
# Admin authentication
|
||||
secret:
|
||||
# Name of an existing secret containing admin credentials. If set, adminUser and adminPassword below are ignored.
|
||||
|
||||
+1021
-1026
File diff suppressed because it is too large
Load Diff
|
Before Width: | Height: | Size: 53 KiB After Width: | Height: | Size: 53 KiB |
Generated
+293
@@ -0,0 +1,293 @@
|
||||
# This file is automatically @generated by Cargo.
|
||||
# It is not intended for manual editing.
|
||||
version = 4
|
||||
|
||||
[[package]]
|
||||
name = "aws-lc-rs"
|
||||
version = "1.18.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ce2b2dcc879c3bae0d371e77c99f2238400ef24ec001394befa67b6e543add9e"
|
||||
dependencies = [
|
||||
"aws-lc-sys",
|
||||
"zeroize",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "aws-lc-sys"
|
||||
version = "0.44.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f09fae7be8bb3174e05c6afdb34199e6dc0c7c04ba9fa237b1967adfbde27483"
|
||||
dependencies = [
|
||||
"cc",
|
||||
"cmake",
|
||||
"dunce",
|
||||
"fs_extra",
|
||||
"pkg-config",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "cc"
|
||||
version = "1.4.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "509591b7bcd67f4ef775afad7662703b4935daaa6ec0e5605cfb1090b32a2b6d"
|
||||
dependencies = [
|
||||
"find-msvc-tools",
|
||||
"jobserver",
|
||||
"libc",
|
||||
"shlex",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "cfg-if"
|
||||
version = "1.0.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9330f8b2ff13f34540b44e946ef35111825727b38d33286ef986142615121801"
|
||||
|
||||
[[package]]
|
||||
name = "cmake"
|
||||
version = "0.1.58"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "c0f78a02292a74a88ac736019ab962ece0bc380e3f977bf72e376c5d78ff0678"
|
||||
dependencies = [
|
||||
"cc",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "dunce"
|
||||
version = "1.0.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "92773504d58c093f6de2459af4af33faa518c13451eb8f2b5698ed3d36e7c813"
|
||||
|
||||
[[package]]
|
||||
name = "find-msvc-tools"
|
||||
version = "0.1.11"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d45db016d36b838f563236e9193d0ee6ce38f3f68b6c94e914b4929c96bbb890"
|
||||
|
||||
[[package]]
|
||||
name = "fs_extra"
|
||||
version = "1.3.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "42703706b716c37f96a77aea830392ad231f44c9e9a67872fa5548707e11b11c"
|
||||
|
||||
[[package]]
|
||||
name = "getrandom"
|
||||
version = "0.2.17"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ff2abc00be7fca6ebc474524697ae276ad847ad0a6b3faa4bcb027e9a4614ad0"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"libc",
|
||||
"wasi",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "getrandom"
|
||||
version = "0.4.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "300e883d756b2e4ec94e02791f39b04b522276138852cfc41d9fb7e904106099"
|
||||
dependencies = [
|
||||
"cfg-if",
|
||||
"libc",
|
||||
"r-efi",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "jobserver"
|
||||
version = "0.1.35"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "1c00acbd29eabad4a2392fa0e921c874934dbbf4194312ad20f04a0ed67a3cb3"
|
||||
dependencies = [
|
||||
"getrandom 0.4.3",
|
||||
"libc",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "libc"
|
||||
version = "0.2.189"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2"
|
||||
|
||||
[[package]]
|
||||
name = "log"
|
||||
version = "0.4.33"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0ceec5bc11778974d1bcb055b18002eba7f4b3518b6a0081b3af5f21666da9ad"
|
||||
|
||||
[[package]]
|
||||
name = "once_cell"
|
||||
version = "1.21.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50"
|
||||
|
||||
[[package]]
|
||||
name = "pkg-config"
|
||||
version = "0.3.34"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f6b464fbc74e149a392436b17d523f769e057cb6877f6a5c4618bc6f11800548"
|
||||
|
||||
[[package]]
|
||||
name = "r-efi"
|
||||
version = "6.0.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f8dcc9c7d52a811697d2151c701e0d08956f92b0e24136cf4cf27b57a6a0d9bf"
|
||||
|
||||
[[package]]
|
||||
name = "ring"
|
||||
version = "0.17.14"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "a4689e6c2294d81e88dc6261c768b63bc4fcdb852be6d1352498b114f61383b7"
|
||||
dependencies = [
|
||||
"cc",
|
||||
"cfg-if",
|
||||
"getrandom 0.2.17",
|
||||
"libc",
|
||||
"untrusted",
|
||||
"windows-sys",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rustls"
|
||||
version = "0.23.43"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0283386ce02abc0151e1761d08802dfe86c173b0b494af5cbc086574e453da06"
|
||||
dependencies = [
|
||||
"aws-lc-rs",
|
||||
"log",
|
||||
"once_cell",
|
||||
"rustls-pki-types",
|
||||
"rustls-webpki",
|
||||
"subtle",
|
||||
"zeroize",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rustls-pki-types"
|
||||
version = "1.15.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "2f4925028c7eb5d1fcdaf196971378ed9d2c1c4efc7dc5d011256f76c99c0a96"
|
||||
dependencies = [
|
||||
"zeroize",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rustls-webpki"
|
||||
version = "0.103.14"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0527518605e68109d875e248ea259b6758801cf165e4b2c2733ae3b51f12535a"
|
||||
dependencies = [
|
||||
"aws-lc-rs",
|
||||
"ring",
|
||||
"rustls-pki-types",
|
||||
"untrusted",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "seaweed-common"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"rustls",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "shlex"
|
||||
version = "2.0.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba"
|
||||
|
||||
[[package]]
|
||||
name = "subtle"
|
||||
version = "2.6.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "13c2bddecc57b384dee18652358fb23172facb8a2c51ccc10d74c157bdea3292"
|
||||
|
||||
[[package]]
|
||||
name = "untrusted"
|
||||
version = "0.9.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "8ecb6da28b8a351d773b68d5825ac39017e680750f980f3a1a85cd8dd28a47c1"
|
||||
|
||||
[[package]]
|
||||
name = "wasi"
|
||||
version = "0.11.1+wasi-snapshot-preview1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ccf3ec651a847eb01de73ccad15eb7d99f80485de043efb2f370cd654f4ea44b"
|
||||
|
||||
[[package]]
|
||||
name = "windows-sys"
|
||||
version = "0.52.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "282be5f36a8ce781fad8c8ae18fa3f9beff57ec1b52cb3de0789201425d9a33d"
|
||||
dependencies = [
|
||||
"windows-targets",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "windows-targets"
|
||||
version = "0.52.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9b724f72796e036ab90c1021d4780d4d3d648aca59e491e6b98e725b84e99973"
|
||||
dependencies = [
|
||||
"windows_aarch64_gnullvm",
|
||||
"windows_aarch64_msvc",
|
||||
"windows_i686_gnu",
|
||||
"windows_i686_gnullvm",
|
||||
"windows_i686_msvc",
|
||||
"windows_x86_64_gnu",
|
||||
"windows_x86_64_gnullvm",
|
||||
"windows_x86_64_msvc",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "windows_aarch64_gnullvm"
|
||||
version = "0.52.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "32a4622180e7a0ec044bb555404c800bc9fd9ec262ec147edd5989ccd0c02cd3"
|
||||
|
||||
[[package]]
|
||||
name = "windows_aarch64_msvc"
|
||||
version = "0.52.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "09ec2a7bb152e2252b53fa7803150007879548bc709c039df7627cabbd05d469"
|
||||
|
||||
[[package]]
|
||||
name = "windows_i686_gnu"
|
||||
version = "0.52.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "8e9b5ad5ab802e97eb8e295ac6720e509ee4c243f69d781394014ebfe8bbfa0b"
|
||||
|
||||
[[package]]
|
||||
name = "windows_i686_gnullvm"
|
||||
version = "0.52.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0eee52d38c090b3caa76c563b86c3a4bd71ef1a819287c19d586d7334ae8ed66"
|
||||
|
||||
[[package]]
|
||||
name = "windows_i686_msvc"
|
||||
version = "0.52.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "240948bc05c5e7c6dabba28bf89d89ffce3e303022809e73deaefe4f6ec56c66"
|
||||
|
||||
[[package]]
|
||||
name = "windows_x86_64_gnu"
|
||||
version = "0.52.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "147a5c80aabfbf0c7d901cb5895d1de30ef2907eb21fbbab29ca94c5b08b1a78"
|
||||
|
||||
[[package]]
|
||||
name = "windows_x86_64_gnullvm"
|
||||
version = "0.52.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "24d5b23dc417412679681396f2b49f3de8c1473deb516bd34410872eff51ed0d"
|
||||
|
||||
[[package]]
|
||||
name = "windows_x86_64_msvc"
|
||||
version = "0.52.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "589f6da84c646204747d1270a2a5661ea66ed1cced2631d546fdfb155959f9ec"
|
||||
|
||||
[[package]]
|
||||
name = "zeroize"
|
||||
version = "1.9.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e13c156562582aa81c60cb29407084cdb54c4164760106ab78e6c5b0858cf64e"
|
||||
@@ -0,0 +1,27 @@
|
||||
[package]
|
||||
name = "seaweed-common"
|
||||
version = "0.1.0"
|
||||
edition = "2024"
|
||||
# The lower of the two consumers' floors (seaweed-volume 1.91.1,
|
||||
# seaweed-worker 1.94.1), so depending on this crate cannot raise either
|
||||
# tree's MSRV. Verified with `cargo +1.91.1 check --all-targets`.
|
||||
rust-version = "1.91.1"
|
||||
description = "Helpers shared by the SeaweedFS Rust volume server and the Rust plugin workers"
|
||||
|
||||
# There is no root manifest: seaweed-volume and seaweed-worker are separate
|
||||
# cargo trees with their own lockfiles, and this crate is a path dependency of
|
||||
# both rather than a member of either. Keeping the lint policy identical in all
|
||||
# three manifests is what stops them drifting.
|
||||
[lints.clippy]
|
||||
# Protobuf message literals keep `..Default::default()` on purpose: it is
|
||||
# what lets a proto gain a field without touching every constructor.
|
||||
needless_update = "allow"
|
||||
|
||||
[dependencies]
|
||||
# The same requirement both consumers already write. Cargo unifies all
|
||||
# semver-compatible `rustls = "0.23"` requirements into one crate per binary,
|
||||
# which is what makes `install_default_crypto_provider` write the same
|
||||
# process-wide static the consuming crate reads. rustls is already in both
|
||||
# trees (the volume server directly, seaweed-worker-core through tonic's
|
||||
# `tls-aws-lc`), so this adds no crate to either graph.
|
||||
rustls = "0.23"
|
||||
@@ -0,0 +1,307 @@
|
||||
//! SeaweedFS server addresses, the way the Go tree does them.
|
||||
//!
|
||||
//! An operator gives a SeaweedFS process an HTTP address and the gRPC port is
|
||||
//! derived from it rather than asked for separately: `host:port` means gRPC on
|
||||
//! `port + 10000`, and the explicit `host:port.grpcPort` form names it outright.
|
||||
//! Dialling the HTTP port by mistake fails as "frame with invalid size", which
|
||||
//! reads like a protocol bug rather than a wrong port, so the rule is worth its
|
||||
//! own module. Mirrors `pb.ServerToGrpcAddress` in
|
||||
//! `weed/pb/grpc_client_server.go`.
|
||||
//!
|
||||
//! The volume server and the workers each had their own copy of this and the
|
||||
//! copies had drifted: the worker's bracketed IPv6 literals and the volume
|
||||
//! server's did not, so `::1:19333` produced `::1:29333`, which the HTTP
|
||||
//! authority parser rejects. One implementation, two thin wrappers.
|
||||
|
||||
use std::fmt;
|
||||
use std::num::ParseIntError;
|
||||
|
||||
/// SeaweedFS's HTTP↔gRPC port-offset convention.
|
||||
pub const GRPC_PORT_OFFSET: u16 = 10000;
|
||||
|
||||
/// Why an address could not be turned into a gRPC address.
|
||||
///
|
||||
/// The `Display` text is the volume server's original wording, because its
|
||||
/// `parse_grpc_address` wrapper hands it straight to callers that put it in a
|
||||
/// `Status` or an `io::Error`.
|
||||
#[derive(Debug, Clone, PartialEq, Eq)]
|
||||
#[non_exhaustive]
|
||||
pub enum AddressError {
|
||||
/// No `:` at all, so there is no port to translate.
|
||||
MissingPort(String),
|
||||
/// The HTTP port of the `host:port.grpcPort` form is not a `u16`. It is
|
||||
/// validated even though it is then discarded, so that a malformed address
|
||||
/// is rejected here instead of failing later as an opaque connect error.
|
||||
InvalidHttpPort { port: String, source: ParseIntError },
|
||||
/// The gRPC port of the `host:port.grpcPort` form is not a `u16`.
|
||||
InvalidGrpcPort { port: String, source: ParseIntError },
|
||||
/// The port of the `host:port` form is not a `u16`.
|
||||
InvalidPort { port: String, source: ParseIntError },
|
||||
/// `port + GRPC_PORT_OFFSET` leaves the TCP port range, e.g. `host:60000`.
|
||||
/// Without the check the cast would wrap silently.
|
||||
ImplicitGrpcPortOutOfRange(u16),
|
||||
}
|
||||
|
||||
impl fmt::Display for AddressError {
|
||||
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
|
||||
match self {
|
||||
Self::MissingPort(address) => write!(f, "cannot parse address: {address}"),
|
||||
Self::InvalidHttpPort { port, source } => {
|
||||
write!(f, "invalid http port {port:?}: {source}")
|
||||
}
|
||||
Self::InvalidGrpcPort { port, source } => {
|
||||
write!(f, "invalid grpc port {port:?}: {source}")
|
||||
}
|
||||
Self::InvalidPort { port, source } => write!(f, "invalid port {port:?}: {source}"),
|
||||
Self::ImplicitGrpcPortOutOfRange(port) => write!(
|
||||
f,
|
||||
"implicit grpc port out of range: {port} + {GRPC_PORT_OFFSET} = {}",
|
||||
u32::from(*port) + u32::from(GRPC_PORT_OFFSET)
|
||||
),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl std::error::Error for AddressError {
|
||||
fn source(&self) -> Option<&(dyn std::error::Error + 'static)> {
|
||||
match self {
|
||||
Self::InvalidHttpPort { source, .. }
|
||||
| Self::InvalidGrpcPort { source, .. }
|
||||
| Self::InvalidPort { source, .. } => Some(source),
|
||||
Self::MissingPort(_) | Self::ImplicitGrpcPortOutOfRange(_) => None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Turn a SeaweedFS server address (`"host:port.grpcPort"` or `"host:port"`)
|
||||
/// into the `host:grpcPort` form the endpoint builders expect.
|
||||
///
|
||||
/// With the trailing `.grpcPort` segment that segment *is* the gRPC port;
|
||||
/// without it the gRPC port is `port + GRPC_PORT_OFFSET`. An unbracketed IPv6
|
||||
/// literal comes back bracketed, because otherwise the port reads as part of
|
||||
/// the address.
|
||||
pub fn to_grpc_address(server: &str) -> Result<String, AddressError> {
|
||||
// rfind, not find: an IPv6 literal is full of colons and the port is after
|
||||
// the last one.
|
||||
let colon_idx = server
|
||||
.rfind(':')
|
||||
.ok_or_else(|| AddressError::MissingPort(server.to_string()))?;
|
||||
let host = &server[..colon_idx];
|
||||
let port_part = &server[colon_idx + 1..];
|
||||
|
||||
// rfind again rather than split_once: the host may be an IPv4 address, and
|
||||
// only the part after the last colon is being split here anyway.
|
||||
if let Some(dot_idx) = port_part.rfind('.') {
|
||||
let http_port = &port_part[..dot_idx];
|
||||
let grpc_port = &port_part[dot_idx + 1..];
|
||||
http_port
|
||||
.parse::<u16>()
|
||||
.map_err(|source| AddressError::InvalidHttpPort {
|
||||
port: http_port.to_string(),
|
||||
source,
|
||||
})?;
|
||||
let grpc_port =
|
||||
grpc_port
|
||||
.parse::<u16>()
|
||||
.map_err(|source| AddressError::InvalidGrpcPort {
|
||||
port: grpc_port.to_string(),
|
||||
source,
|
||||
})?;
|
||||
return Ok(join_host_port(host, grpc_port));
|
||||
}
|
||||
|
||||
let port: u16 = port_part
|
||||
.parse()
|
||||
.map_err(|source| AddressError::InvalidPort {
|
||||
port: port_part.to_string(),
|
||||
source,
|
||||
})?;
|
||||
let grpc_port = port
|
||||
.checked_add(GRPC_PORT_OFFSET)
|
||||
.ok_or(AddressError::ImplicitGrpcPortOutOfRange(port))?;
|
||||
Ok(join_host_port(host, grpc_port))
|
||||
}
|
||||
|
||||
/// Join a host and a port, bracketing an IPv6 literal that is not bracketed
|
||||
/// already. Public because the address rule is not the only place that has to
|
||||
/// put a host and a port back together.
|
||||
pub fn join_host_port(host: &str, port: u16) -> String {
|
||||
// An IPv6 literal has to keep its brackets or the port reads as part of it.
|
||||
if host.contains(':') && !host.starts_with('[') {
|
||||
format!("[{host}]:{port}")
|
||||
} else {
|
||||
format!("{host}:{port}")
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::{AddressError, GRPC_PORT_OFFSET, join_host_port, to_grpc_address};
|
||||
|
||||
// ---- the volume server's cases -------------------------------------
|
||||
|
||||
#[test]
|
||||
fn dotted_form_states_the_grpc_port() {
|
||||
assert_eq!(
|
||||
to_grpc_address("127.0.0.1:8080.18080").unwrap(),
|
||||
"127.0.0.1:18080"
|
||||
);
|
||||
assert_eq!(
|
||||
to_grpc_address("192.168.1.66:8080.18080").unwrap(),
|
||||
"192.168.1.66:18080"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn implicit_form_adds_the_offset() {
|
||||
assert_eq!(
|
||||
to_grpc_address("127.0.0.1:8080").unwrap(),
|
||||
"127.0.0.1:18080"
|
||||
);
|
||||
assert_eq!(
|
||||
to_grpc_address("192.168.1.66:8080").unwrap(),
|
||||
"192.168.1.66:18080"
|
||||
);
|
||||
assert_eq!(
|
||||
to_grpc_address("localhost:9333").unwrap(),
|
||||
"localhost:19333"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_dotted_grpc_port_comes_back_normalised() {
|
||||
// The volume server's copy validated this segment as a u16 and then
|
||||
// emitted the original text, so a padded or signed port produced an
|
||||
// authority the URI parser rejects. The parsed value is emitted now.
|
||||
assert_eq!(to_grpc_address("host:8080.018080").unwrap(), "host:18080");
|
||||
assert_eq!(to_grpc_address("host:8080.+18080").unwrap(), "host:18080");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn an_ipv4_host_is_not_confused_with_the_dotted_port() {
|
||||
// Regression: a naive split on '.' breaks on IP addresses.
|
||||
assert_eq!(
|
||||
to_grpc_address("10.0.0.1:8080.18080").unwrap(),
|
||||
"10.0.0.1:18080"
|
||||
);
|
||||
assert_eq!(to_grpc_address("10.0.0.1:8080").unwrap(), "10.0.0.1:18080");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rejects_a_non_numeric_http_port_in_the_dotted_form() {
|
||||
let err = to_grpc_address("host:abc.18080").unwrap_err();
|
||||
assert!(
|
||||
matches!(err, AddressError::InvalidHttpPort { .. }),
|
||||
"{err:?}"
|
||||
);
|
||||
assert!(err.to_string().contains("invalid http port"), "{err}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rejects_a_non_numeric_grpc_port_in_the_dotted_form() {
|
||||
let err = to_grpc_address("host:8080.xyz").unwrap_err();
|
||||
assert!(
|
||||
matches!(err, AddressError::InvalidGrpcPort { .. }),
|
||||
"{err:?}"
|
||||
);
|
||||
assert!(err.to_string().contains("invalid grpc port"), "{err}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rejects_an_implicit_port_that_leaves_the_tcp_range() {
|
||||
let err = to_grpc_address("127.0.0.1:60000").unwrap_err();
|
||||
assert!(
|
||||
matches!(err, AddressError::ImplicitGrpcPortOutOfRange(60000)),
|
||||
"{err:?}"
|
||||
);
|
||||
assert!(err.to_string().contains("out of range"), "{err}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn the_messages_are_the_volume_servers_wording_verbatim() {
|
||||
// parse_grpc_address hands these straight to callers that put them in a
|
||||
// Status or an io::Error, so the whole string is the contract, not just
|
||||
// the substring the older tests match on. Only the two variants whose
|
||||
// text is entirely ours are pinned exactly; the other three end in a
|
||||
// std ParseIntError message, which is std's to reword.
|
||||
assert_eq!(
|
||||
to_grpc_address("127.0.0.1:60000").unwrap_err().to_string(),
|
||||
"implicit grpc port out of range: 60000 + 10000 = 70000"
|
||||
);
|
||||
assert_eq!(
|
||||
to_grpc_address("hostname").unwrap_err().to_string(),
|
||||
"cannot parse address: hostname"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rejects_an_address_without_a_port() {
|
||||
for source in ["hostname", "no-colon", "localhost"] {
|
||||
let err = to_grpc_address(source).unwrap_err();
|
||||
assert!(matches!(err, AddressError::MissingPort(_)), "{err:?}");
|
||||
assert!(err.to_string().contains("cannot parse"), "{err}");
|
||||
}
|
||||
}
|
||||
|
||||
// ---- the worker's cases --------------------------------------------
|
||||
|
||||
#[test]
|
||||
fn derives_the_grpc_port() {
|
||||
assert_eq!(
|
||||
to_grpc_address("localhost:23646").unwrap(),
|
||||
"localhost:33646"
|
||||
);
|
||||
assert_eq!(
|
||||
to_grpc_address("127.0.0.1:9333").unwrap(),
|
||||
"127.0.0.1:19333"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn honours_an_explicit_grpc_port() {
|
||||
assert_eq!(
|
||||
to_grpc_address("localhost:23646.33999").unwrap(),
|
||||
"localhost:33999"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rejects_what_it_cannot_parse() {
|
||||
let err = to_grpc_address("localhost:notaport").unwrap_err();
|
||||
assert!(matches!(err, AddressError::InvalidPort { .. }), "{err:?}");
|
||||
assert!(err.to_string().contains("invalid port"), "{err}");
|
||||
}
|
||||
|
||||
// ---- IPv6, which only the worker's copy handled --------------------
|
||||
|
||||
#[test]
|
||||
fn brackets_ipv6_literals() {
|
||||
assert_eq!(to_grpc_address("::1:23646").unwrap(), "[::1]:33646");
|
||||
assert_eq!(to_grpc_address("::1:9333").unwrap(), "[::1]:19333");
|
||||
assert_eq!(
|
||||
to_grpc_address("fe80::1:9333.19333").unwrap(),
|
||||
"[fe80::1]:19333"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn leaves_an_already_bracketed_literal_alone() {
|
||||
assert_eq!(to_grpc_address("[::1]:9333").unwrap(), "[::1]:19333");
|
||||
assert_eq!(to_grpc_address("[::1]:9333.19333").unwrap(), "[::1]:19333");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn join_host_port_brackets_only_unbracketed_literals() {
|
||||
assert_eq!(join_host_port("127.0.0.1", 19333), "127.0.0.1:19333");
|
||||
assert_eq!(join_host_port("localhost", 19333), "localhost:19333");
|
||||
assert_eq!(join_host_port("::1", 19333), "[::1]:19333");
|
||||
assert_eq!(join_host_port("[::1]", 19333), "[::1]:19333");
|
||||
}
|
||||
|
||||
// ---- the offset itself ---------------------------------------------
|
||||
|
||||
#[test]
|
||||
fn the_offset_is_the_seaweedfs_convention() {
|
||||
assert_eq!(GRPC_PORT_OFFSET, 10000);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,11 @@
|
||||
//! Helpers the SeaweedFS Rust volume server and the Rust plugin workers both need.
|
||||
//!
|
||||
//! `seaweed-volume` and `seaweed-worker` are separate cargo trees with separate
|
||||
//! lockfiles and no root manifest, so anything both of them need was, until this
|
||||
//! crate existed, written twice. The two things in here are the ones where a
|
||||
//! second copy is a correctness risk rather than a typing cost: the HTTP↔gRPC
|
||||
//! address rule, which two copies had already drifted on, and the process-wide
|
||||
//! rustls provider, which only works if every binary installs the same one.
|
||||
|
||||
pub mod address;
|
||||
pub mod tls;
|
||||
@@ -0,0 +1,28 @@
|
||||
//! The process-wide rustls crypto provider.
|
||||
//!
|
||||
//! Both binaries link aws-lc-rs and ring transitively — in the volume server
|
||||
//! through the AWS SDK and reqwest, in the lance worker through lance's `aws`
|
||||
//! backend and reqwest — so rustls cannot auto-select a provider and tonic's
|
||||
//! client TLS panics on first use. Each binary has to pin one, and it has to be
|
||||
//! the same one, which is why the choice lives here rather than in either tree.
|
||||
|
||||
use rustls::crypto::aws_lc_rs;
|
||||
|
||||
/// Pin rustls's process-wide default provider to aws-lc-rs, matching the
|
||||
/// volume server's TLS config. Idempotent: the first call wins and every
|
||||
/// later one is a no-op, so callers do not have to coordinate.
|
||||
pub fn install_default_crypto_provider() {
|
||||
let _ = aws_lc_rs::default_provider().install_default();
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::install_default_crypto_provider;
|
||||
|
||||
#[test]
|
||||
fn installing_is_idempotent_and_leaves_a_default_behind() {
|
||||
install_default_crypto_provider();
|
||||
install_default_crypto_provider();
|
||||
assert!(rustls::crypto::CryptoProvider::get_default().is_some());
|
||||
}
|
||||
}
|
||||
Generated
+9
@@ -3487,6 +3487,13 @@ version = "1.2.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "94143f37725109f92c262ed2cf5e59bce7498c01bcc1502d7b9afe439a4e9f49"
|
||||
|
||||
[[package]]
|
||||
name = "seaweed-common"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"rustls",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "sec1"
|
||||
version = "0.3.0"
|
||||
@@ -4548,6 +4555,7 @@ dependencies = [
|
||||
"aws-config",
|
||||
"aws-credential-types",
|
||||
"aws-sdk-s3",
|
||||
"aws-smithy-runtime-api",
|
||||
"aws-types",
|
||||
"axum",
|
||||
"base64",
|
||||
@@ -4584,6 +4592,7 @@ dependencies = [
|
||||
"rustls",
|
||||
"rustls-pemfile",
|
||||
"rusty-leveldb",
|
||||
"seaweed-common",
|
||||
"serde",
|
||||
"serde_json",
|
||||
"serde_urlencoded",
|
||||
|
||||
@@ -27,8 +27,15 @@ redb-experimental-cursor = ["redb/experimental_cursor"]
|
||||
# Protobuf message literals keep `..Default::default()` on purpose: it is
|
||||
# what lets a proto gain a field without touching every constructor.
|
||||
needless_update = "allow"
|
||||
# Every `unsafe` block states its precondition, right above the block.
|
||||
undocumented_unsafe_blocks = "warn"
|
||||
|
||||
[dependencies]
|
||||
# Helpers the Rust plugin workers (seaweed-worker) need as well. A path
|
||||
# dependency because the two trees are separate cargo workspaces with no
|
||||
# common root manifest.
|
||||
seaweed-common = { path = "../seaweed-common" }
|
||||
|
||||
# Async runtime
|
||||
tokio = { version = "1", features = ["full"] }
|
||||
tokio-stream = { version = "0.1", features = ["net"] }
|
||||
@@ -150,6 +157,9 @@ windows-sys = { version = "0.61", features = ["Win32_Storage_FileSystem"] }
|
||||
|
||||
[dev-dependencies]
|
||||
tempfile = "3"
|
||||
# Already a transitive dependency of aws-sdk-s3 at a single locked version;
|
||||
# needed directly only for the canned HttpClient in remote_storage::s3 tests.
|
||||
aws-smithy-runtime-api = "1"
|
||||
|
||||
[build-dependencies]
|
||||
tonic-prost-build = "0.14"
|
||||
|
||||
@@ -259,6 +259,10 @@ message VolumeDeleteRequest {
|
||||
// when true, do not remove the cloud-tier object backing the volume.
|
||||
// used for moves where another server is taking over the same .vif.
|
||||
bool keep_remote_data = 3;
|
||||
// when true, delete only if every needle is deleted: the volume held
|
||||
// data once but nothing is live anymore. Passing either check,
|
||||
// only_empty or this one, is enough to delete.
|
||||
bool only_garbage = 4;
|
||||
}
|
||||
message VolumeDeleteResponse {
|
||||
}
|
||||
@@ -471,6 +475,7 @@ message VolumeEcShardsDeleteRequest {
|
||||
repeated uint32 shard_ids = 3;
|
||||
bool full_teardown = 4; // pre-encode cleanup: wipe every EC artifact + generation for this volume, not just shard_ids
|
||||
int64 encode_ts_ns = 5; // full_teardown generation fence: delete only a disk whose .vif generation is strictly OLDER than this; preserve same-or-newer, generation 0, and an unreadable .vif. 0 => wipe-all (shell pre-encode / pre-upgrade)
|
||||
uint32 delete_generations_older_than = 6; // post-commit cleanup: delete only staged <base>.*.v<N> artifacts with N strictly below this; 0 disables
|
||||
}
|
||||
message VolumeEcShardsDeleteResponse {
|
||||
bool full_teardown_done = 1; // set by a new server that performed full_teardown; absent from an old server lets the caller detect the silent no-op
|
||||
|
||||
+647
-304
File diff suppressed because it is too large
Load Diff
@@ -12,7 +12,10 @@ use seaweed_volume::security::tls::{
|
||||
use seaweed_volume::security::{Guard, SigningKey};
|
||||
#[cfg(unix)]
|
||||
use seaweed_volume::server::debug::build_debug_router;
|
||||
use seaweed_volume::server::grpc_client::load_outgoing_grpc_tls;
|
||||
use seaweed_volume::server::grpc_client::{
|
||||
GRPC_INITIAL_WINDOW_SIZE, GRPC_KEEPALIVE_INTERVAL, GRPC_KEEPALIVE_TIMEOUT,
|
||||
GRPC_MAX_MESSAGE_SIZE, load_outgoing_grpc_tls,
|
||||
};
|
||||
use seaweed_volume::server::grpc_server::VolumeGrpcService;
|
||||
#[cfg(unix)]
|
||||
use seaweed_volume::server::profiling::CpuProfileSession;
|
||||
@@ -31,10 +34,10 @@ type CpuProfileParam = Option<CpuProfileSession>;
|
||||
#[cfg(not(unix))]
|
||||
type CpuProfileParam = Option<()>;
|
||||
|
||||
const GRPC_MAX_MESSAGE_SIZE: usize = 1 << 30;
|
||||
const GRPC_KEEPALIVE_INTERVAL: std::time::Duration = std::time::Duration::from_secs(60);
|
||||
const GRPC_KEEPALIVE_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(20);
|
||||
const GRPC_INITIAL_WINDOW_SIZE: u32 = 16 * 1024 * 1024;
|
||||
// The two settings that only make sense for the inbound server. The rest of
|
||||
// this server's HTTP/2 tuning — keepalive, window sizes, message size — is
|
||||
// imported from `server::grpc_client` above, which is also what the outgoing
|
||||
// clients dial with, so the two directions cannot drift apart.
|
||||
const GRPC_MAX_HEADER_LIST_SIZE: u32 = 8 * 1024 * 1024;
|
||||
const GRPC_MAX_CONCURRENT_STREAMS: u32 = 1000;
|
||||
|
||||
@@ -343,9 +346,6 @@ async fn run(
|
||||
pre_stop_seconds: config.pre_stop_seconds,
|
||||
volume_state_notify: tokio::sync::Notify::new(),
|
||||
write_queue: std::sync::OnceLock::new(),
|
||||
s3_tier_registry: std::sync::RwLock::new(
|
||||
seaweed_volume::remote_storage::s3_tier::S3TierRegistry::new(),
|
||||
),
|
||||
read_mode: config.read_mode,
|
||||
allow_untrusted_remote_endpoints: config.allow_untrusted_remote_endpoints,
|
||||
master_url,
|
||||
|
||||
@@ -7,9 +7,7 @@ pub mod endpoint_guard;
|
||||
pub mod s3;
|
||||
pub mod s3_tier;
|
||||
|
||||
pub use endpoint_guard::{
|
||||
guarded_tcp_connect, validate_remote_endpoint, validate_replica_target,
|
||||
};
|
||||
pub use endpoint_guard::{guarded_tcp_connect, validate_remote_endpoint, validate_replica_target};
|
||||
|
||||
use crate::pb::remote_pb::{RemoteConf, RemoteStorageLocation};
|
||||
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
|
||||
use aws_sdk_s3::Client;
|
||||
use aws_sdk_s3::config::{BehaviorVersion, Credentials, Region};
|
||||
use aws_sdk_s3::error::{DisplayErrorContext, SdkError};
|
||||
use aws_sdk_s3::primitives::ByteStream;
|
||||
|
||||
use super::{RemoteEntry, RemoteStorageClient, RemoteStorageError};
|
||||
@@ -25,6 +26,23 @@ impl S3RemoteStorageClient {
|
||||
endpoint: &str,
|
||||
force_path_style: bool,
|
||||
) -> Self {
|
||||
let client = Client::from_conf(
|
||||
Self::config_builder(access_key, secret_key, region, endpoint, force_path_style)
|
||||
.build(),
|
||||
);
|
||||
|
||||
S3RemoteStorageClient { client, conf }
|
||||
}
|
||||
|
||||
/// Build the SDK config for the given credentials and endpoint. Split out so
|
||||
/// tests can attach a canned HTTP client before building the [`Client`].
|
||||
fn config_builder(
|
||||
access_key: &str,
|
||||
secret_key: &str,
|
||||
region: &str,
|
||||
endpoint: &str,
|
||||
force_path_style: bool,
|
||||
) -> aws_sdk_s3::config::Builder {
|
||||
let region = if region.is_empty() {
|
||||
"us-east-1"
|
||||
} else {
|
||||
@@ -49,9 +67,7 @@ impl S3RemoteStorageClient {
|
||||
s3_config = s3_config.endpoint_url(endpoint);
|
||||
}
|
||||
|
||||
let client = Client::from_conf(s3_config.build());
|
||||
|
||||
S3RemoteStorageClient { client, conf }
|
||||
s3_config
|
||||
}
|
||||
}
|
||||
|
||||
@@ -75,13 +91,14 @@ impl RemoteStorageClient for S3RemoteStorageClient {
|
||||
req = req.range(format!("bytes={}-", offset));
|
||||
}
|
||||
|
||||
let resp = req.send().await.map_err(|e| {
|
||||
let msg = format!("{}", e);
|
||||
if msg.contains("NoSuchKey") || msg.contains("404") {
|
||||
let resp = req.send().await.map_err(|e| match e {
|
||||
// Go compares `aerr.Code()` to NoSuchKey on GET
|
||||
// (s3_storage_client.go:436): a bare 404 maps to "NotFound"
|
||||
// and stays a generic error, as it does here.
|
||||
SdkError::ServiceError(ref se) if se.err().is_no_such_key() => {
|
||||
RemoteStorageError::ObjectNotFound(format!("{}/{}", loc.bucket, key))
|
||||
} else {
|
||||
RemoteStorageError::Other(format!("s3 get object: {}", e))
|
||||
}
|
||||
e => RemoteStorageError::Other(format!("s3 get object: {}", DisplayErrorContext(&e))),
|
||||
})?;
|
||||
|
||||
let data = resp
|
||||
@@ -108,7 +125,9 @@ impl RemoteStorageClient for S3RemoteStorageClient {
|
||||
.body(ByteStream::from(data.to_vec()))
|
||||
.send()
|
||||
.await
|
||||
.map_err(|e| RemoteStorageError::Other(format!("s3 put object: {}", e)))?;
|
||||
.map_err(|e| {
|
||||
RemoteStorageError::Other(format!("s3 put object: {}", DisplayErrorContext(&e)))
|
||||
})?;
|
||||
|
||||
Ok(RemoteEntry {
|
||||
size: data.len() as i64,
|
||||
@@ -134,13 +153,18 @@ impl RemoteStorageClient for S3RemoteStorageClient {
|
||||
.key(key)
|
||||
.send()
|
||||
.await
|
||||
.map_err(|e| {
|
||||
let msg = format!("{}", e);
|
||||
if msg.contains("404") || msg.contains("NotFound") {
|
||||
.map_err(|e| match e {
|
||||
// Go checks only the raw HTTP status on HEAD
|
||||
// (s3_storage_client.go:373): a HEAD response carries no
|
||||
// error body, so a 404 is not-found whatever code the SDK
|
||||
// assigns, and a non-404 is not.
|
||||
SdkError::ServiceError(ref se) if se.raw().status().as_u16() == 404 => {
|
||||
RemoteStorageError::ObjectNotFound(format!("{}/{}", loc.bucket, key))
|
||||
} else {
|
||||
RemoteStorageError::Other(format!("s3 head object: {}", e))
|
||||
}
|
||||
e => RemoteStorageError::Other(format!(
|
||||
"s3 head object: {}",
|
||||
DisplayErrorContext(&e)
|
||||
)),
|
||||
})?;
|
||||
|
||||
Ok(RemoteEntry {
|
||||
@@ -160,18 +184,17 @@ impl RemoteStorageClient for S3RemoteStorageClient {
|
||||
.key(key)
|
||||
.send()
|
||||
.await
|
||||
.map_err(|e| RemoteStorageError::Other(format!("s3 delete object: {}", e)))?;
|
||||
.map_err(|e| {
|
||||
RemoteStorageError::Other(format!("s3 delete object: {}", DisplayErrorContext(&e)))
|
||||
})?;
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
async fn list_buckets(&self) -> Result<Vec<String>, RemoteStorageError> {
|
||||
let resp = self
|
||||
.client
|
||||
.list_buckets()
|
||||
.send()
|
||||
.await
|
||||
.map_err(|e| RemoteStorageError::Other(format!("s3 list buckets: {}", e)))?;
|
||||
let resp = self.client.list_buckets().send().await.map_err(|e| {
|
||||
RemoteStorageError::Other(format!("s3 list buckets: {}", DisplayErrorContext(&e)))
|
||||
})?;
|
||||
|
||||
Ok(resp
|
||||
.buckets()
|
||||
@@ -184,3 +207,178 @@ impl RemoteStorageClient for S3RemoteStorageClient {
|
||||
&self.conf
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
pub(crate) mod tests {
|
||||
use super::*;
|
||||
use aws_sdk_s3::config::http::{HttpRequest, HttpResponse};
|
||||
use aws_sdk_s3::config::retry::RetryConfig;
|
||||
use aws_sdk_s3::config::{HttpClient, RuntimeComponents};
|
||||
use aws_sdk_s3::primitives::SdkBody;
|
||||
use aws_smithy_runtime_api::client::http::{
|
||||
HttpConnector, HttpConnectorFuture, HttpConnectorSettings, SharedHttpConnector,
|
||||
};
|
||||
use aws_smithy_runtime_api::http::StatusCode;
|
||||
|
||||
/// An SDK HTTP client that answers every request with one canned response,
|
||||
/// so the error-mapping paths can be exercised without a network or a
|
||||
/// running S3 server.
|
||||
#[derive(Debug, Clone)]
|
||||
pub(crate) struct CannedResponse {
|
||||
pub(crate) status: u16,
|
||||
pub(crate) body: &'static str,
|
||||
}
|
||||
|
||||
impl HttpConnector for CannedResponse {
|
||||
fn call(&self, _request: HttpRequest) -> HttpConnectorFuture {
|
||||
let status = StatusCode::try_from(self.status).expect("valid HTTP status");
|
||||
HttpConnectorFuture::ready(Ok(HttpResponse::new(status, SdkBody::from(self.body))))
|
||||
}
|
||||
}
|
||||
|
||||
impl HttpClient for CannedResponse {
|
||||
fn http_connector(
|
||||
&self,
|
||||
_settings: &HttpConnectorSettings,
|
||||
_components: &RuntimeComponents,
|
||||
) -> SharedHttpConnector {
|
||||
SharedHttpConnector::new(self.clone())
|
||||
}
|
||||
}
|
||||
|
||||
fn client_with(status: u16, body: &'static str) -> S3RemoteStorageClient {
|
||||
let config = S3RemoteStorageClient::config_builder(
|
||||
"AKIATEST",
|
||||
"secret",
|
||||
"us-east-1",
|
||||
"http://127.0.0.1:1",
|
||||
true,
|
||||
)
|
||||
.http_client(CannedResponse { status, body })
|
||||
.retry_config(RetryConfig::disabled())
|
||||
.build();
|
||||
S3RemoteStorageClient {
|
||||
client: Client::from_conf(config),
|
||||
conf: RemoteConf::default(),
|
||||
}
|
||||
}
|
||||
|
||||
fn location() -> RemoteStorageLocation {
|
||||
RemoteStorageLocation {
|
||||
name: "remote".to_string(),
|
||||
bucket: "bucket".to_string(),
|
||||
path: "/dir/missing".to_string(),
|
||||
..Default::default()
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) const NO_SUCH_KEY: &str = r#"<?xml version="1.0" encoding="UTF-8"?>
|
||||
<Error><Code>NoSuchKey</Code><Message>The specified key does not exist.</Message><Key>dir/missing</Key></Error>"#;
|
||||
|
||||
const NOT_FOUND_BODY: &str = r#"<?xml version="1.0" encoding="UTF-8"?>
|
||||
<Error><Code>NotFound</Code><Message>Not Found</Message></Error>"#;
|
||||
|
||||
const ACCESS_DENIED: &str = r#"<?xml version="1.0" encoding="UTF-8"?>
|
||||
<Error><Code>AccessDenied</Code><Message>Access Denied</Message></Error>"#;
|
||||
|
||||
#[tokio::test]
|
||||
async fn get_no_such_key_is_object_not_found() {
|
||||
let err = client_with(404, NO_SUCH_KEY)
|
||||
.read_file(&location(), 0, 0)
|
||||
.await
|
||||
.unwrap_err();
|
||||
assert!(
|
||||
matches!(&err, RemoteStorageError::ObjectNotFound(path) if path == "bucket/dir/missing"),
|
||||
"expected ObjectNotFound, got {err:?}"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn get_bare_404_is_not_object_not_found() {
|
||||
// Go compares codes, not statuses, on GET: a body-less 404 stays generic.
|
||||
let err = client_with(404, "")
|
||||
.read_file(&location(), 0, 0)
|
||||
.await
|
||||
.unwrap_err();
|
||||
assert!(
|
||||
matches!(err, RemoteStorageError::Other(_)),
|
||||
"expected Other, got {err:?}"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn head_404_is_object_not_found() {
|
||||
let err = client_with(404, "")
|
||||
.stat_file(&location())
|
||||
.await
|
||||
.unwrap_err();
|
||||
assert!(
|
||||
matches!(&err, RemoteStorageError::ObjectNotFound(path) if path == "bucket/dir/missing"),
|
||||
"expected ObjectNotFound, got {err:?}"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn head_404_with_foreign_error_body_is_object_not_found() {
|
||||
// The raw status check makes a 404 not-found whatever body it carries.
|
||||
let err = client_with(404, NO_SUCH_KEY)
|
||||
.stat_file(&location())
|
||||
.await
|
||||
.unwrap_err();
|
||||
assert!(
|
||||
matches!(err, RemoteStorageError::ObjectNotFound(_)),
|
||||
"expected ObjectNotFound, got {err:?}"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn head_not_found_code_on_a_non_404_status_is_not_object_not_found() {
|
||||
// A NotFound body on a non-404 status stays an error, as in Go.
|
||||
let err = client_with(400, NOT_FOUND_BODY)
|
||||
.stat_file(&location())
|
||||
.await
|
||||
.unwrap_err();
|
||||
assert!(
|
||||
matches!(&err, RemoteStorageError::Other(msg) if msg.contains("NotFound")),
|
||||
"expected Other naming the code, got {err:?}"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn get_access_denied_keeps_service_error_code() {
|
||||
let err = client_with(403, ACCESS_DENIED)
|
||||
.read_file(&location(), 0, 0)
|
||||
.await
|
||||
.unwrap_err();
|
||||
let msg = err.to_string();
|
||||
assert!(
|
||||
matches!(err, RemoteStorageError::Other(_)),
|
||||
"expected Other, got {err:?}"
|
||||
);
|
||||
assert!(
|
||||
msg.contains("AccessDenied"),
|
||||
"message should carry the S3 error code, got: {msg}"
|
||||
);
|
||||
assert!(
|
||||
!msg.ends_with("service error"),
|
||||
"message should not be the bare SdkError Display, got: {msg}"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn head_access_denied_keeps_service_error_code() {
|
||||
let err = client_with(403, ACCESS_DENIED)
|
||||
.stat_file(&location())
|
||||
.await
|
||||
.unwrap_err();
|
||||
let msg = err.to_string();
|
||||
assert!(
|
||||
matches!(err, RemoteStorageError::Other(_)),
|
||||
"expected Other, got {err:?}"
|
||||
);
|
||||
assert!(
|
||||
msg.contains("AccessDenied"),
|
||||
"message should carry the S3 error code, got: {msg}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -8,7 +8,11 @@ use std::future::Future;
|
||||
use std::sync::{Arc, OnceLock, RwLock};
|
||||
|
||||
use aws_sdk_s3::Client;
|
||||
use aws_sdk_s3::config::http::HttpResponse;
|
||||
use aws_sdk_s3::config::{BehaviorVersion, Credentials, Region};
|
||||
use aws_sdk_s3::error::{DisplayErrorContext, SdkError};
|
||||
use aws_sdk_s3::operation::get_object::GetObjectError;
|
||||
use aws_sdk_s3::operation::head_object::HeadObjectError;
|
||||
use aws_sdk_s3::types::{CompletedMultipartUpload, CompletedPart};
|
||||
use tokio::io::{AsyncReadExt, AsyncSeekExt, AsyncWriteExt};
|
||||
use tokio::sync::Semaphore;
|
||||
@@ -16,6 +20,53 @@ use tokio::sync::Semaphore;
|
||||
/// Concurrency limit for multipart upload/download (matches Go's s3manager).
|
||||
const CONCURRENCY: usize = 5;
|
||||
|
||||
/// A tier transfer failure. The variant is what callers match on; the
|
||||
/// message is the operator-facing text.
|
||||
#[derive(Debug, thiserror::Error)]
|
||||
pub enum TierError {
|
||||
/// The remote object does not exist.
|
||||
#[error("{0}")]
|
||||
NotFound(String),
|
||||
/// An S3 request or a local file operation failed.
|
||||
#[error("{0}")]
|
||||
Io(String),
|
||||
/// The tier I/O runtime could not be built or dropped the task.
|
||||
#[error("{0}")]
|
||||
RuntimeUnavailable(String),
|
||||
/// The progress callback asked to stop.
|
||||
#[error("{0}")]
|
||||
Aborted(String),
|
||||
}
|
||||
|
||||
// Not-found rules as in remote_storage/s3.rs: HEAD by the raw 404 status,
|
||||
// GET by the NoSuchKey code only.
|
||||
fn head_object_error(key: &str, e: SdkError<HeadObjectError, HttpResponse>) -> TierError {
|
||||
let message = format!("failed to head object {}: {}", key, DisplayErrorContext(&e));
|
||||
match e {
|
||||
SdkError::ServiceError(ref se) if se.raw().status().as_u16() == 404 => {
|
||||
TierError::NotFound(message)
|
||||
}
|
||||
_ => TierError::Io(message),
|
||||
}
|
||||
}
|
||||
|
||||
fn get_object_error(
|
||||
key: &str,
|
||||
range: &str,
|
||||
e: SdkError<GetObjectError, HttpResponse>,
|
||||
) -> TierError {
|
||||
let message = format!(
|
||||
"failed to get object {} range {}: {}",
|
||||
key,
|
||||
range,
|
||||
DisplayErrorContext(&e)
|
||||
);
|
||||
match e {
|
||||
SdkError::ServiceError(ref se) if se.err().is_no_such_key() => TierError::NotFound(message),
|
||||
_ => TierError::Io(message),
|
||||
}
|
||||
}
|
||||
|
||||
/// Configuration for an S3 tier backend.
|
||||
#[derive(Debug, Clone)]
|
||||
pub struct S3TierConfig {
|
||||
@@ -89,7 +140,7 @@ impl S3TierBackend {
|
||||
&self,
|
||||
file_path: &str,
|
||||
progress_fn: F,
|
||||
) -> Result<(String, u64), String>
|
||||
) -> Result<(String, u64), TierError>
|
||||
where
|
||||
F: FnMut(i64, f32) -> Result<(), String> + Send + Sync + 'static,
|
||||
{
|
||||
@@ -97,7 +148,7 @@ impl S3TierBackend {
|
||||
|
||||
let metadata = tokio::fs::metadata(file_path)
|
||||
.await
|
||||
.map_err(|e| format!("failed to stat file {}: {}", file_path, e))?;
|
||||
.map_err(|e| TierError::Io(format!("failed to stat file {}: {}", file_path, e)))?;
|
||||
let file_size = metadata.len();
|
||||
|
||||
// Calculate part size: start at 64MB, scale up for very large files (matches Go)
|
||||
@@ -119,11 +170,16 @@ impl S3TierBackend {
|
||||
)
|
||||
.send()
|
||||
.await
|
||||
.map_err(|e| format!("failed to create multipart upload: {}", e))?;
|
||||
.map_err(|e| {
|
||||
TierError::Io(format!(
|
||||
"failed to create multipart upload: {}",
|
||||
DisplayErrorContext(&e)
|
||||
))
|
||||
})?;
|
||||
|
||||
let upload_id = create_resp
|
||||
.upload_id()
|
||||
.ok_or_else(|| "no upload_id in multipart upload response".to_string())?
|
||||
.ok_or_else(|| TierError::Io("no upload_id in multipart upload response".to_string()))?
|
||||
.to_string();
|
||||
|
||||
// Build list of (part_number, offset, size) for all parts
|
||||
@@ -159,19 +215,21 @@ impl S3TierBackend {
|
||||
let _permit = sem
|
||||
.acquire()
|
||||
.await
|
||||
.map_err(|e| format!("semaphore error: {}", e))?;
|
||||
.map_err(|e| TierError::Io(format!("semaphore error: {}", e)))?;
|
||||
|
||||
// Read this part's data from the file at the correct offset
|
||||
let mut file = tokio::fs::File::open(&fp)
|
||||
.await
|
||||
.map_err(|e| format!("failed to open file {}: {}", fp, e))?;
|
||||
.map_err(|e| TierError::Io(format!("failed to open file {}: {}", fp, e)))?;
|
||||
file.seek(std::io::SeekFrom::Start(off))
|
||||
.await
|
||||
.map_err(|e| format!("failed to seek to offset {}: {}", off, e))?;
|
||||
.map_err(|e| {
|
||||
TierError::Io(format!("failed to seek to offset {}: {}", off, e))
|
||||
})?;
|
||||
let mut buf = vec![0u8; size];
|
||||
file.read_exact(&mut buf)
|
||||
.await
|
||||
.map_err(|e| format!("failed to read file at offset {}: {}", off, e))?;
|
||||
file.read_exact(&mut buf).await.map_err(|e| {
|
||||
TierError::Io(format!("failed to read file at offset {}: {}", off, e))
|
||||
})?;
|
||||
|
||||
let upload_part_resp = client
|
||||
.upload_part()
|
||||
@@ -183,7 +241,12 @@ impl S3TierBackend {
|
||||
.send()
|
||||
.await
|
||||
.map_err(|e| {
|
||||
format!("failed to upload part {} at offset {}: {}", pn, off, e)
|
||||
TierError::Io(format!(
|
||||
"failed to upload part {} at offset {}: {}",
|
||||
pn,
|
||||
off,
|
||||
DisplayErrorContext(&e)
|
||||
))
|
||||
})?;
|
||||
|
||||
let e_tag = upload_part_resp.e_tag().unwrap_or_default().to_string();
|
||||
@@ -202,9 +265,9 @@ impl S3TierBackend {
|
||||
};
|
||||
(guard.1)(uploaded as i64, pct)
|
||||
};
|
||||
progress_result?;
|
||||
progress_result.map_err(TierError::Aborted)?;
|
||||
|
||||
Ok::<_, String>(
|
||||
Ok::<_, TierError>(
|
||||
CompletedPart::builder()
|
||||
.e_tag(e_tag)
|
||||
.part_number(pn)
|
||||
@@ -219,7 +282,7 @@ impl S3TierBackend {
|
||||
for handle in handles {
|
||||
let part = handle
|
||||
.await
|
||||
.map_err(|e| format!("upload task panicked: {}", e))??;
|
||||
.map_err(|e| TierError::Io(format!("upload task panicked: {}", e)))??;
|
||||
completed_parts.push(part);
|
||||
}
|
||||
|
||||
@@ -236,9 +299,14 @@ impl S3TierBackend {
|
||||
.multipart_upload(completed_upload)
|
||||
.send()
|
||||
.await
|
||||
.map_err(|e| format!("failed to complete multipart upload: {}", e))?;
|
||||
.map_err(|e| {
|
||||
TierError::Io(format!(
|
||||
"failed to complete multipart upload: {}",
|
||||
DisplayErrorContext(&e)
|
||||
))
|
||||
})?;
|
||||
|
||||
Ok::<(), String>(())
|
||||
Ok::<(), TierError>(())
|
||||
}
|
||||
.await;
|
||||
|
||||
@@ -281,7 +349,7 @@ impl S3TierBackend {
|
||||
dest_path: &str,
|
||||
key: &str,
|
||||
progress_fn: F,
|
||||
) -> Result<u64, String>
|
||||
) -> Result<u64, TierError>
|
||||
where
|
||||
F: FnMut(i64, f32) -> Result<(), String> + Send + Sync + 'static,
|
||||
{
|
||||
@@ -293,7 +361,7 @@ impl S3TierBackend {
|
||||
.key(key)
|
||||
.send()
|
||||
.await
|
||||
.map_err(|e| format!("failed to head object {}: {}", key, e))?;
|
||||
.map_err(|e| head_object_error(key, e))?;
|
||||
|
||||
let file_size = head_resp.content_length().unwrap_or(0) as u64;
|
||||
|
||||
@@ -305,10 +373,12 @@ impl S3TierBackend {
|
||||
.truncate(true)
|
||||
.open(dest_path)
|
||||
.await
|
||||
.map_err(|e| format!("failed to open dest file {}: {}", dest_path, e))?;
|
||||
.map_err(|e| {
|
||||
TierError::Io(format!("failed to open dest file {}: {}", dest_path, e))
|
||||
})?;
|
||||
file.set_len(file_size)
|
||||
.await
|
||||
.map_err(|e| format!("failed to set file length: {}", e))?;
|
||||
.map_err(|e| TierError::Io(format!("failed to set file length: {}", e)))?;
|
||||
}
|
||||
|
||||
let part_size: u64 = 64 * 1024 * 1024;
|
||||
@@ -344,7 +414,7 @@ impl S3TierBackend {
|
||||
let _permit = sem
|
||||
.acquire()
|
||||
.await
|
||||
.map_err(|e| format!("semaphore error: {}", e))?;
|
||||
.map_err(|e| TierError::Io(format!("semaphore error: {}", e)))?;
|
||||
|
||||
let end = off + size - 1;
|
||||
let range = format!("bytes={}-{}", off, end);
|
||||
@@ -356,13 +426,13 @@ impl S3TierBackend {
|
||||
.range(&range)
|
||||
.send()
|
||||
.await
|
||||
.map_err(|e| format!("failed to get object {} range {}: {}", key, range, e))?;
|
||||
.map_err(|e| get_object_error(&key, &range, e))?;
|
||||
|
||||
let body = get_resp
|
||||
.body
|
||||
.collect()
|
||||
.await
|
||||
.map_err(|e| format!("failed to read body: {}", e))?;
|
||||
.map_err(|e| TierError::Io(format!("failed to read body: {}", e)))?;
|
||||
let bytes = body.into_bytes();
|
||||
|
||||
// Write at the correct offset (like Go's WriteAt)
|
||||
@@ -370,13 +440,17 @@ impl S3TierBackend {
|
||||
.write(true)
|
||||
.open(&dp)
|
||||
.await
|
||||
.map_err(|e| format!("failed to open dest file {}: {}", dp, e))?;
|
||||
.map_err(|e| {
|
||||
TierError::Io(format!("failed to open dest file {}: {}", dp, e))
|
||||
})?;
|
||||
file.seek(std::io::SeekFrom::Start(off))
|
||||
.await
|
||||
.map_err(|e| format!("failed to seek to offset {}: {}", off, e))?;
|
||||
.map_err(|e| {
|
||||
TierError::Io(format!("failed to seek to offset {}: {}", off, e))
|
||||
})?;
|
||||
file.write_all(&bytes)
|
||||
.await
|
||||
.map_err(|e| format!("failed to write to {}: {}", dp, e))?;
|
||||
.map_err(|e| TierError::Io(format!("failed to write to {}: {}", dp, e)))?;
|
||||
|
||||
// Report progress. The lock is released before the result is
|
||||
// propagated so an aborting callback cannot poison the mutex
|
||||
@@ -392,9 +466,9 @@ impl S3TierBackend {
|
||||
};
|
||||
(guard.1)(downloaded as i64, pct)
|
||||
};
|
||||
progress_result?;
|
||||
progress_result.map_err(TierError::Aborted)?;
|
||||
|
||||
Ok::<_, String>(())
|
||||
Ok::<_, TierError>(())
|
||||
}));
|
||||
}
|
||||
|
||||
@@ -402,7 +476,7 @@ impl S3TierBackend {
|
||||
for handle in handles {
|
||||
handle
|
||||
.await
|
||||
.map_err(|e| format!("download task panicked: {}", e))??;
|
||||
.map_err(|e| TierError::Io(format!("download task panicked: {}", e)))??;
|
||||
}
|
||||
|
||||
// fsync the file so its content is durable before the caller trims the .vif
|
||||
@@ -411,16 +485,21 @@ impl S3TierBackend {
|
||||
.write(true)
|
||||
.open(dest_path)
|
||||
.await
|
||||
.map_err(|e| format!("failed to open {} for fsync: {}", dest_path, e))?;
|
||||
.map_err(|e| TierError::Io(format!("failed to open {} for fsync: {}", dest_path, e)))?;
|
||||
synced
|
||||
.sync_all()
|
||||
.await
|
||||
.map_err(|e| format!("failed to fsync {}: {}", dest_path, e))?;
|
||||
.map_err(|e| TierError::Io(format!("failed to fsync {}: {}", dest_path, e)))?;
|
||||
|
||||
Ok(file_size)
|
||||
}
|
||||
|
||||
pub async fn read_range(&self, key: &str, offset: u64, size: usize) -> Result<Vec<u8>, String> {
|
||||
pub async fn read_range(
|
||||
&self,
|
||||
key: &str,
|
||||
offset: u64,
|
||||
size: usize,
|
||||
) -> Result<Vec<u8>, TierError> {
|
||||
let end = offset + (size as u64).saturating_sub(1);
|
||||
let range = format!("bytes={}-{}", offset, end);
|
||||
let resp = self
|
||||
@@ -431,29 +510,35 @@ impl S3TierBackend {
|
||||
.range(&range)
|
||||
.send()
|
||||
.await
|
||||
.map_err(|e| format!("failed to get object {} range {}: {}", key, range, e))?;
|
||||
.map_err(|e| get_object_error(key, &range, e))?;
|
||||
|
||||
let body = resp
|
||||
.body
|
||||
.collect()
|
||||
.await
|
||||
.map_err(|e| format!("failed to read object {} body: {}", key, e))?;
|
||||
.map_err(|e| TierError::Io(format!("failed to read object {} body: {}", key, e)))?;
|
||||
Ok(body.into_bytes().to_vec())
|
||||
}
|
||||
|
||||
/// Delete a file from S3.
|
||||
pub async fn delete_file(&self, key: &str) -> Result<(), String> {
|
||||
pub async fn delete_file(&self, key: &str) -> Result<(), TierError> {
|
||||
self.client
|
||||
.delete_object()
|
||||
.bucket(&self.bucket)
|
||||
.key(key)
|
||||
.send()
|
||||
.await
|
||||
.map_err(|e| format!("failed to delete object {}: {}", key, e))?;
|
||||
.map_err(|e| {
|
||||
TierError::Io(format!(
|
||||
"failed to delete object {}: {}",
|
||||
key,
|
||||
DisplayErrorContext(&e)
|
||||
))
|
||||
})?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub fn delete_file_blocking(&self, key: &str) -> Result<(), String> {
|
||||
pub fn delete_file_blocking(&self, key: &str) -> Result<(), TierError> {
|
||||
let client = self.client.clone();
|
||||
let bucket = self.bucket.clone();
|
||||
let key = key.to_string();
|
||||
@@ -464,7 +549,13 @@ impl S3TierBackend {
|
||||
.key(&key)
|
||||
.send()
|
||||
.await
|
||||
.map_err(|e| format!("failed to delete object {}: {}", key, e))?;
|
||||
.map_err(|e| {
|
||||
TierError::Io(format!(
|
||||
"failed to delete object {}: {}",
|
||||
key,
|
||||
DisplayErrorContext(&e)
|
||||
))
|
||||
})?;
|
||||
Ok(())
|
||||
})
|
||||
}
|
||||
@@ -474,7 +565,7 @@ impl S3TierBackend {
|
||||
key: &str,
|
||||
offset: u64,
|
||||
size: usize,
|
||||
) -> Result<Vec<u8>, String> {
|
||||
) -> Result<Vec<u8>, TierError> {
|
||||
let client = self.client.clone();
|
||||
let bucket = self.bucket.clone();
|
||||
let key = key.to_string();
|
||||
@@ -488,13 +579,12 @@ impl S3TierBackend {
|
||||
.range(&range)
|
||||
.send()
|
||||
.await
|
||||
.map_err(|e| format!("failed to get object {} range {}: {}", key, range, e))?;
|
||||
.map_err(|e| get_object_error(&key, &range, e))?;
|
||||
|
||||
let body = resp
|
||||
.body
|
||||
.collect()
|
||||
.await
|
||||
.map_err(|e| format!("failed to read object {} body: {}", key, e))?;
|
||||
let body =
|
||||
resp.body.collect().await.map_err(|e| {
|
||||
TierError::Io(format!("failed to read object {} body: {}", key, e))
|
||||
})?;
|
||||
Ok(body.into_bytes().to_vec())
|
||||
})
|
||||
}
|
||||
@@ -555,18 +645,279 @@ pub fn global_s3_tier_registry() -> &'static RwLock<S3TierRegistry> {
|
||||
GLOBAL_S3_TIER_REGISTRY.get_or_init(|| RwLock::new(S3TierRegistry::new()))
|
||||
}
|
||||
|
||||
fn block_on_tier_future<F, T>(future: F) -> Result<T, String>
|
||||
where
|
||||
F: Future<Output = Result<T, String>> + Send + 'static,
|
||||
T: Send + 'static,
|
||||
{
|
||||
std::thread::spawn(move || {
|
||||
let runtime = tokio::runtime::Builder::new_current_thread()
|
||||
/// The one process-wide runtime for tiered-S3 I/O issued from synchronous
|
||||
/// storage code. A per-call runtime tore down the SDK's pooled connections
|
||||
/// after every 64 KiB chunk, re-dialing TLS per read; a long-lived runtime
|
||||
/// keeps the pool warm.
|
||||
///
|
||||
/// Built on first use. A build failure is returned, not cached or panicked:
|
||||
/// callers sit inside `Volume::destroy` and needle reads, whose own error
|
||||
/// paths must run, and a later call may succeed.
|
||||
static TIER_RUNTIME: std::sync::Mutex<Option<tokio::runtime::Runtime>> =
|
||||
std::sync::Mutex::new(None);
|
||||
|
||||
fn tier_handle() -> Result<tokio::runtime::Handle, TierError> {
|
||||
let mut slot = TIER_RUNTIME
|
||||
.lock()
|
||||
.unwrap_or_else(|poisoned| poisoned.into_inner());
|
||||
if slot.is_none() {
|
||||
let runtime = tokio::runtime::Builder::new_multi_thread()
|
||||
.worker_threads(2)
|
||||
.thread_name("tier-io")
|
||||
.enable_all()
|
||||
.build()
|
||||
.map_err(|e| format!("failed to build tokio runtime: {}", e))?;
|
||||
runtime.block_on(future)
|
||||
})
|
||||
.join()
|
||||
.map_err(|_| "tier runtime thread panicked".to_string())?
|
||||
.map_err(|e| {
|
||||
TierError::RuntimeUnavailable(format!(
|
||||
"failed to build the tier I/O tokio runtime: {}",
|
||||
e
|
||||
))
|
||||
})?;
|
||||
*slot = Some(runtime);
|
||||
}
|
||||
Ok(slot.as_ref().expect("just initialised").handle().clone())
|
||||
}
|
||||
|
||||
/// Run `future` on the tier runtime and block the calling thread until it
|
||||
/// finishes. The caller may be a worker of *another* tokio runtime, so this
|
||||
/// waits on a channel rather than `Handle::block_on`, which panics when
|
||||
/// called from inside any runtime context.
|
||||
fn block_on_tier_future<F, T>(future: F) -> Result<T, TierError>
|
||||
where
|
||||
F: Future<Output = Result<T, TierError>> + Send + 'static,
|
||||
T: Send + 'static,
|
||||
{
|
||||
let handle = tier_handle()?;
|
||||
let task = handle.spawn(future);
|
||||
let (tx, rx) = std::sync::mpsc::sync_channel(1);
|
||||
handle.spawn(async move {
|
||||
// The receiver only goes away if the caller was unwound; nothing to
|
||||
// report then.
|
||||
let _ = tx.send(task.await);
|
||||
});
|
||||
match rx.recv() {
|
||||
Ok(Ok(result)) => result,
|
||||
Ok(Err(join_error)) => Err(describe_join_error(join_error)),
|
||||
Err(_) => Err(TierError::RuntimeUnavailable(
|
||||
"tier I/O runtime dropped the task before it finished".to_string(),
|
||||
)),
|
||||
}
|
||||
}
|
||||
|
||||
/// Turn a `JoinError` into a message that keeps the panic payload, so an
|
||||
/// SDK panic surfaces as "boom" rather than a fixed "thread panicked".
|
||||
fn describe_join_error(join_error: tokio::task::JoinError) -> TierError {
|
||||
if join_error.is_panic() {
|
||||
let payload = join_error.into_panic();
|
||||
let message = if let Some(s) = payload.downcast_ref::<&str>() {
|
||||
(*s).to_string()
|
||||
} else if let Some(s) = payload.downcast_ref::<String>() {
|
||||
s.clone()
|
||||
} else {
|
||||
"non-string panic payload".to_string()
|
||||
};
|
||||
TierError::Io(format!("tier I/O task panicked: {}", message))
|
||||
} else {
|
||||
TierError::RuntimeUnavailable(format!("tier I/O task failed: {}", join_error))
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::remote_storage::s3::tests::{CannedResponse, NO_SUCH_KEY};
|
||||
use std::collections::HashSet;
|
||||
use tokio::runtime::Handle;
|
||||
|
||||
fn probe() -> Result<(tokio::runtime::Id, Option<String>), TierError> {
|
||||
block_on_tier_future(async {
|
||||
Ok((
|
||||
Handle::current().id(),
|
||||
std::thread::current().name().map(str::to_string),
|
||||
))
|
||||
})
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn block_on_tier_future_reuses_one_runtime() {
|
||||
let (first_runtime, first_thread) = probe().expect("first call");
|
||||
let (second_runtime, second_thread) = probe().expect("second call");
|
||||
assert_eq!(
|
||||
first_runtime, second_runtime,
|
||||
"each call must run on the same long-lived tier runtime"
|
||||
);
|
||||
assert_eq!(first_thread.as_deref(), Some("tier-io"));
|
||||
assert_eq!(second_thread.as_deref(), Some("tier-io"));
|
||||
|
||||
let mut runtimes = HashSet::new();
|
||||
for _ in 0..20 {
|
||||
let (id, _) = probe().expect("probe");
|
||||
runtimes.insert(id);
|
||||
}
|
||||
assert_eq!(runtimes.len(), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn block_on_tier_future_returns_the_value_and_the_error() {
|
||||
assert_eq!(block_on_tier_future(async { Ok(7u32) }).unwrap(), 7);
|
||||
let err = block_on_tier_future::<_, u32>(async { Err(TierError::NotFound("nope".into())) })
|
||||
.unwrap_err();
|
||||
assert!(
|
||||
matches!(&err, TierError::NotFound(m) if m == "nope"),
|
||||
"{err:?}"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn block_on_tier_future_works_from_a_std_thread() {
|
||||
let (id, _) = std::thread::spawn(probe)
|
||||
.join()
|
||||
.expect("probe thread")
|
||||
.expect("probe");
|
||||
assert_eq!(id, tier_handle().expect("tier runtime").id());
|
||||
}
|
||||
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
|
||||
async fn block_on_tier_future_works_from_spawn_blocking() {
|
||||
let (id, _) = tokio::task::spawn_blocking(probe)
|
||||
.await
|
||||
.expect("spawn_blocking")
|
||||
.expect("probe");
|
||||
assert_eq!(id, tier_handle().expect("tier runtime").id());
|
||||
assert_ne!(id, Handle::current().id());
|
||||
}
|
||||
|
||||
// Called straight from another runtime's async context: the case that
|
||||
// would panic with `Handle::block_on` ("Cannot start a runtime from
|
||||
// within a runtime").
|
||||
#[tokio::test]
|
||||
async fn block_on_tier_future_works_from_a_current_thread_runtime() {
|
||||
let (id, _) = probe().expect("probe");
|
||||
assert_eq!(id, tier_handle().expect("tier runtime").id());
|
||||
assert_ne!(id, Handle::current().id());
|
||||
}
|
||||
|
||||
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
|
||||
async fn block_on_tier_future_works_from_a_multi_thread_runtime_worker() {
|
||||
let (id, _) = probe().expect("probe");
|
||||
assert_eq!(id, tier_handle().expect("tier runtime").id());
|
||||
assert_ne!(id, Handle::current().id());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn block_on_tier_future_reports_the_panic_payload() {
|
||||
let err = block_on_tier_future::<_, ()>(async {
|
||||
if std::hint::black_box(true) {
|
||||
panic!("boom {}", 42);
|
||||
}
|
||||
Ok(())
|
||||
})
|
||||
.expect_err("a panicking future must be an error");
|
||||
assert!(matches!(err, TierError::Io(_)), "got: {err:?}");
|
||||
assert!(err.to_string().contains("boom 42"), "got: {err}");
|
||||
assert!(err.to_string().contains("panicked"), "got: {err}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn block_on_tier_future_reports_a_str_panic_payload() {
|
||||
let err = block_on_tier_future::<_, ()>(async {
|
||||
if std::hint::black_box(true) {
|
||||
panic!("static boom");
|
||||
}
|
||||
Ok(())
|
||||
})
|
||||
.expect_err("a panicking future must be an error");
|
||||
assert!(err.to_string().contains("static boom"), "got: {err}");
|
||||
}
|
||||
|
||||
fn backend_answering(status: u16, body: &'static str) -> S3TierBackend {
|
||||
let config = aws_sdk_s3::Config::builder()
|
||||
.behavior_version(BehaviorVersion::latest())
|
||||
.region(Region::new("us-east-1"))
|
||||
.credentials_provider(Credentials::new("AKIATEST", "secret", None, None, "test"))
|
||||
.endpoint_url("http://127.0.0.1:1")
|
||||
.force_path_style(true)
|
||||
.http_client(CannedResponse { status, body })
|
||||
.retry_config(aws_sdk_s3::config::retry::RetryConfig::disabled())
|
||||
.build();
|
||||
S3TierBackend {
|
||||
client: Client::from_conf(config),
|
||||
bucket: "bucket".to_string(),
|
||||
storage_class: "STANDARD".to_string(),
|
||||
}
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn download_head_404_is_not_found() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let dest = tmp.path().join("1.dat");
|
||||
let err = backend_answering(404, "")
|
||||
.download_file(dest.to_str().unwrap(), "missing", |_, _| Ok(()))
|
||||
.await
|
||||
.unwrap_err();
|
||||
assert!(matches!(err, TierError::NotFound(_)), "{err:?}");
|
||||
assert!(
|
||||
err.to_string()
|
||||
.starts_with("failed to head object missing: "),
|
||||
"{err}"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn download_head_403_is_io() {
|
||||
let tmp = tempfile::tempdir().unwrap();
|
||||
let dest = tmp.path().join("1.dat");
|
||||
let err = backend_answering(403, "")
|
||||
.download_file(dest.to_str().unwrap(), "denied", |_, _| Ok(()))
|
||||
.await
|
||||
.unwrap_err();
|
||||
assert!(matches!(err, TierError::Io(_)), "{err:?}");
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn read_range_no_such_key_is_not_found() {
|
||||
let err = backend_answering(404, NO_SUCH_KEY)
|
||||
.read_range("missing", 0, 8)
|
||||
.await
|
||||
.unwrap_err();
|
||||
assert!(matches!(err, TierError::NotFound(_)), "{err:?}");
|
||||
assert!(
|
||||
err.to_string()
|
||||
.starts_with("failed to get object missing range bytes=0-7: "),
|
||||
"{err}"
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn read_range_bare_404_is_io() {
|
||||
// As in Go, GET is not-found by the NoSuchKey code, not the status.
|
||||
let err = backend_answering(404, "")
|
||||
.read_range("missing", 0, 8)
|
||||
.await
|
||||
.unwrap_err();
|
||||
assert!(matches!(err, TierError::Io(_)), "{err:?}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn read_range_blocking_no_such_key_is_not_found() {
|
||||
let err = backend_answering(404, NO_SUCH_KEY)
|
||||
.read_range_blocking("missing", 0, 8)
|
||||
.unwrap_err();
|
||||
assert!(matches!(err, TierError::NotFound(_)), "{err:?}");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn backend_name_to_type_id_splits_on_dot() {
|
||||
assert_eq!(
|
||||
backend_name_to_type_id("s3"),
|
||||
("s3".to_string(), "default".to_string())
|
||||
);
|
||||
assert_eq!(
|
||||
backend_name_to_type_id("s3.eu"),
|
||||
("s3".to_string(), "eu".to_string())
|
||||
);
|
||||
assert_eq!(
|
||||
backend_name_to_type_id("s3.a.b"),
|
||||
(String::new(), String::new())
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -120,10 +120,11 @@ impl ClientCertVerifier for CommonNameVerifier {
|
||||
|
||||
// aws-lc-rs and ring both get linked transitively, so rustls can't auto-select
|
||||
// a provider and tonic's client TLS panics on first use. Pin the default to
|
||||
// aws-lc-rs, matching the server config. Idempotent.
|
||||
pub fn install_default_crypto_provider() {
|
||||
let _ = aws_lc_rs::default_provider().install_default();
|
||||
}
|
||||
// aws-lc-rs, matching the server config. Idempotent. The body lives in
|
||||
// seaweed-common so this binary and the Rust plugin workers cannot end up
|
||||
// installing different providers; re-exported here so callers keep their
|
||||
// import path.
|
||||
pub use seaweed_common::tls::install_default_crypto_provider;
|
||||
|
||||
pub fn build_rustls_server_config(
|
||||
cert_path: &str,
|
||||
|
||||
@@ -1,16 +1,42 @@
|
||||
//! Construction of the volume server's *outgoing* gRPC clients: TLS material,
|
||||
//! endpoint tuning, dial bounds, and the three client constructors every call
|
||||
//! site goes through.
|
||||
//!
|
||||
//! The keepalive, window-size and message-size constants below are shared with
|
||||
//! the *inbound* server built in `main.rs`, which imports them from here rather
|
||||
//! than declaring its own. Changing one therefore changes both directions at
|
||||
//! once, which is deliberate: a volume server talks to its peers with the same
|
||||
//! HTTP/2 settings it offers them.
|
||||
|
||||
use std::error::Error;
|
||||
use std::fmt;
|
||||
use std::time::Duration;
|
||||
|
||||
use hyper::http::Uri;
|
||||
use tonic::service::interceptor::InterceptedService;
|
||||
use tonic::transport::{Certificate, Channel, ClientTlsConfig, Endpoint, Identity};
|
||||
use tonic::{Request, Status};
|
||||
|
||||
use crate::config::VolumeServerConfig;
|
||||
use crate::pb::filer_pb::seaweed_filer_client::SeaweedFilerClient;
|
||||
use crate::pb::master_pb::seaweed_client::SeaweedClient;
|
||||
use crate::pb::volume_server_pb::volume_server_client::VolumeServerClient;
|
||||
use crate::server::request_id::outgoing_request_id_interceptor;
|
||||
|
||||
pub const GRPC_MAX_MESSAGE_SIZE: usize = 1 << 30;
|
||||
const GRPC_KEEPALIVE_INTERVAL: Duration = Duration::from_secs(60);
|
||||
const GRPC_KEEPALIVE_TIMEOUT: Duration = Duration::from_secs(20);
|
||||
const GRPC_INITIAL_WINDOW_SIZE: u32 = 16 * 1024 * 1024;
|
||||
pub const GRPC_KEEPALIVE_INTERVAL: Duration = Duration::from_secs(60);
|
||||
pub const GRPC_KEEPALIVE_TIMEOUT: Duration = Duration::from_secs(20);
|
||||
pub const GRPC_INITIAL_WINDOW_SIZE: u32 = 16 * 1024 * 1024;
|
||||
|
||||
/// Bound on the TCP connect of every outgoing dial. `build_grpc_endpoint` is
|
||||
/// private and `connect_channel` is the only way out of this module, so every
|
||||
/// call site picks this up whether it thinks about timeouts or not.
|
||||
///
|
||||
/// It bounds the TCP handshake only — tonic hands it to
|
||||
/// `HttpConnector::set_connect_timeout`. A peer that completes the handshake
|
||||
/// and then stalls in the TLS or HTTP/2 exchange is not covered; callers that
|
||||
/// need that bound wrap the whole dial (see `connect_ping_target`).
|
||||
const GRPC_CONNECT_TIMEOUT: Duration = Duration::from_secs(5);
|
||||
|
||||
#[derive(Clone, Debug)]
|
||||
pub struct OutgoingGrpcTlsConfig {
|
||||
@@ -81,7 +107,7 @@ pub fn grpc_endpoint_uri(grpc_host_port: &str, tls: Option<&OutgoingGrpcTlsConfi
|
||||
format!("{}://{}", scheme, grpc_host_port)
|
||||
}
|
||||
|
||||
pub fn build_grpc_endpoint(
|
||||
fn build_grpc_endpoint(
|
||||
grpc_host_port: &str,
|
||||
tls: Option<&OutgoingGrpcTlsConfig>,
|
||||
) -> Result<Endpoint, GrpcClientError> {
|
||||
@@ -149,6 +175,152 @@ pub async fn connect_guarded(
|
||||
.map_err(|e| GrpcClientError(format!("connect {} failed: {}", target, e)))
|
||||
}
|
||||
|
||||
/// How a dial is bounded.
|
||||
///
|
||||
/// `connect_timeout` is handed to the TCP connector. `request_timeout` becomes
|
||||
/// [`Endpoint::timeout`], which tonic installs as a `GrpcTimeout` layer in
|
||||
/// front of *every* request the resulting channel carries — it is not a
|
||||
/// property of one call.
|
||||
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
|
||||
pub struct GrpcDialOptions {
|
||||
/// Bound on establishing the connection to the peer.
|
||||
pub connect_timeout: Duration,
|
||||
/// Deadline applied to each RPC on the channel, or `None` to leave them
|
||||
/// unbounded.
|
||||
pub request_timeout: Option<Duration>,
|
||||
}
|
||||
|
||||
impl GrpcDialOptions {
|
||||
/// A short request/response call: connect within 5 s, answer within 10 s.
|
||||
pub fn unary() -> Self {
|
||||
Self {
|
||||
connect_timeout: GRPC_CONNECT_TIMEOUT,
|
||||
request_timeout: Some(Duration::from_secs(10)),
|
||||
}
|
||||
}
|
||||
|
||||
/// A call the peer may take a while to answer: connect within 5 s, answer
|
||||
/// within 30 s.
|
||||
pub fn long() -> Self {
|
||||
Self {
|
||||
connect_timeout: GRPC_CONNECT_TIMEOUT,
|
||||
request_timeout: Some(Duration::from_secs(30)),
|
||||
}
|
||||
}
|
||||
|
||||
/// A bounded connect with no deadline on the RPCs themselves.
|
||||
///
|
||||
/// `request_timeout` must stay `None` here. [`Endpoint::timeout`] is not a
|
||||
/// transfer budget: tonic layers it as a `GrpcTimeout` around the
|
||||
/// response future, which resolves when the server's *first response
|
||||
/// headers* arrive, so it bounds how long the peer may take to start
|
||||
/// answering — per request, for every request the channel carries. A 10 s
|
||||
/// value picked to suit one short call would therefore also be the header
|
||||
/// deadline for the `VolumeCopy` that shares the dial, and a busy source
|
||||
/// that takes longer than that to open its file would lose the whole copy.
|
||||
/// `VolumeCopy`, `VolumeTailSender` and `VolumeEcShardsCopy` have never
|
||||
/// carried one.
|
||||
pub fn stream() -> Self {
|
||||
Self {
|
||||
connect_timeout: GRPC_CONNECT_TIMEOUT,
|
||||
request_timeout: None,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Dial a peer and return a connected channel.
|
||||
///
|
||||
/// The error carries only the transport failure: every caller already wraps it
|
||||
/// with the address and the operation it was attempting.
|
||||
pub async fn connect_channel(
|
||||
grpc_host_port: &str,
|
||||
tls: Option<&OutgoingGrpcTlsConfig>,
|
||||
opts: GrpcDialOptions,
|
||||
) -> Result<Channel, GrpcClientError> {
|
||||
let mut endpoint =
|
||||
build_grpc_endpoint(grpc_host_port, tls)?.connect_timeout(opts.connect_timeout);
|
||||
if let Some(request_timeout) = opts.request_timeout {
|
||||
endpoint = endpoint.timeout(request_timeout);
|
||||
}
|
||||
endpoint
|
||||
.connect()
|
||||
.await
|
||||
.map_err(|e| GrpcClientError(e.to_string()))
|
||||
}
|
||||
|
||||
/// Dial a copy/tail source and return a connected channel, re-validating every
|
||||
/// resolved address at connect time.
|
||||
///
|
||||
/// The guarded equivalent of [`connect_channel`]: same `opts` bounds, but the
|
||||
/// dial goes through [`connect_guarded`] so a source address that passed
|
||||
/// validation cannot be re-pointed by DNS between the check and the connect.
|
||||
/// The bounds are applied to the endpoint *before* delegating, so the
|
||||
/// `allow_untrusted` opt-out is timed too.
|
||||
///
|
||||
/// `target` is the caller-facing source address (the unparsed
|
||||
/// `"ip:port.grpcPort"` form), which is what the guard pins against; the error
|
||||
/// carries only the transport failure, as every caller already wraps it with
|
||||
/// the address and the operation it was attempting.
|
||||
pub async fn connect_channel_guarded(
|
||||
grpc_host_port: &str,
|
||||
target: &str,
|
||||
tls: Option<&OutgoingGrpcTlsConfig>,
|
||||
opts: GrpcDialOptions,
|
||||
allow_untrusted: bool,
|
||||
) -> Result<Channel, GrpcClientError> {
|
||||
let mut endpoint =
|
||||
build_grpc_endpoint(grpc_host_port, tls)?.connect_timeout(opts.connect_timeout);
|
||||
if let Some(request_timeout) = opts.request_timeout {
|
||||
endpoint = endpoint.timeout(request_timeout);
|
||||
}
|
||||
connect_guarded(endpoint, target, allow_untrusted).await
|
||||
}
|
||||
|
||||
/// The outgoing request-id interceptor as a concrete type, so the client
|
||||
/// aliases below can name it.
|
||||
pub type RequestIdInterceptor = fn(Request<()>) -> Result<Request<()>, Status>;
|
||||
|
||||
/// A volume-server client with the request-id interceptor attached.
|
||||
pub type VolumeServerGrpcClient =
|
||||
VolumeServerClient<InterceptedService<Channel, RequestIdInterceptor>>;
|
||||
/// A master client with the request-id interceptor attached.
|
||||
pub type MasterGrpcClient = SeaweedClient<InterceptedService<Channel, RequestIdInterceptor>>;
|
||||
/// A filer client with the request-id interceptor attached.
|
||||
pub type FilerGrpcClient = SeaweedFilerClient<InterceptedService<Channel, RequestIdInterceptor>>;
|
||||
|
||||
/// Wrap a connected channel in a volume-server client that forwards the
|
||||
/// current request id and lifts both message-size limits.
|
||||
pub fn volume_server_client(channel: Channel) -> VolumeServerGrpcClient {
|
||||
VolumeServerClient::with_interceptor(
|
||||
channel,
|
||||
outgoing_request_id_interceptor as RequestIdInterceptor,
|
||||
)
|
||||
.max_decoding_message_size(GRPC_MAX_MESSAGE_SIZE)
|
||||
.max_encoding_message_size(GRPC_MAX_MESSAGE_SIZE)
|
||||
}
|
||||
|
||||
/// Wrap a connected channel in a master client that forwards the current
|
||||
/// request id and lifts both message-size limits.
|
||||
pub fn master_client(channel: Channel) -> MasterGrpcClient {
|
||||
SeaweedClient::with_interceptor(
|
||||
channel,
|
||||
outgoing_request_id_interceptor as RequestIdInterceptor,
|
||||
)
|
||||
.max_decoding_message_size(GRPC_MAX_MESSAGE_SIZE)
|
||||
.max_encoding_message_size(GRPC_MAX_MESSAGE_SIZE)
|
||||
}
|
||||
|
||||
/// Wrap a connected channel in a filer client that forwards the current
|
||||
/// request id and lifts both message-size limits.
|
||||
pub fn filer_client(channel: Channel) -> FilerGrpcClient {
|
||||
SeaweedFilerClient::with_interceptor(
|
||||
channel,
|
||||
outgoing_request_id_interceptor as RequestIdInterceptor,
|
||||
)
|
||||
.max_decoding_message_size(GRPC_MAX_MESSAGE_SIZE)
|
||||
.max_encoding_message_size(GRPC_MAX_MESSAGE_SIZE)
|
||||
}
|
||||
|
||||
/// Parse a SeaweedFS server address (`"ip:port.grpcPort"` or
|
||||
/// `"ip:port"`) into the `host:grpcPort` form `build_grpc_endpoint`
|
||||
/// expects. With the trailing `.grpcPort` segment, that segment IS
|
||||
@@ -158,53 +330,27 @@ pub async fn connect_guarded(
|
||||
/// Shared between `grpc_server.rs` and the distributed-EC-read path
|
||||
/// in `store_ec.rs` — keep this as the single source of truth so the
|
||||
/// HTTP↔gRPC port translation can't drift between callers.
|
||||
///
|
||||
/// The rule itself lives in `seaweed_common::address`, which the Rust
|
||||
/// plugin workers share; this wrapper only flattens the typed error
|
||||
/// back to the `String` its callers already handle. Unbracketed IPv6
|
||||
/// literals come back bracketed, which this copy used to get wrong.
|
||||
pub fn parse_grpc_address(source: &str) -> Result<String, String> {
|
||||
let colon_idx = source
|
||||
.rfind(':')
|
||||
.ok_or_else(|| format!("cannot parse address: {}", source))?;
|
||||
let host = &source[..colon_idx];
|
||||
let port_part = &source[colon_idx + 1..];
|
||||
|
||||
if let Some(dot_idx) = port_part.rfind('.') {
|
||||
// Format: "ip:port.grpcPort". Validate BOTH ports as u16
|
||||
// so a malformed HTTP port (e.g. `host:abc.18080`) is
|
||||
// rejected here rather than tripping a downstream
|
||||
// `build_grpc_endpoint` URI parse failure with a less
|
||||
// useful error.
|
||||
let http_port = &port_part[..dot_idx];
|
||||
let grpc_port = &port_part[dot_idx + 1..];
|
||||
http_port
|
||||
.parse::<u16>()
|
||||
.map_err(|e| format!("invalid http port {:?}: {}", http_port, e))?;
|
||||
grpc_port
|
||||
.parse::<u16>()
|
||||
.map_err(|e| format!("invalid grpc port {:?}: {}", grpc_port, e))?;
|
||||
return Ok(format!("{}:{}", host, grpc_port));
|
||||
}
|
||||
|
||||
// Format: "ip:port" → grpc = port + 10000. Reject inputs whose
|
||||
// implicit grpc port would overflow the TCP port range (e.g.
|
||||
// `host:60000` produces 70000 — invalid). Without this check
|
||||
// the cast silently wraps and the endpoint call later fails
|
||||
// with an opaque connection error.
|
||||
let port: u16 = port_part
|
||||
.parse()
|
||||
.map_err(|e| format!("invalid port {:?}: {}", port_part, e))?;
|
||||
let grpc_port = port as u32 + 10000;
|
||||
if grpc_port > u16::MAX as u32 {
|
||||
return Err(format!(
|
||||
"implicit grpc port out of range: {} + 10000 = {}",
|
||||
port, grpc_port
|
||||
));
|
||||
}
|
||||
Ok(format!("{}:{}", host, grpc_port))
|
||||
seaweed_common::address::to_grpc_address(source).map_err(|e| e.to_string())
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::{build_grpc_endpoint, grpc_endpoint_uri, load_outgoing_grpc_tls};
|
||||
use super::{
|
||||
GrpcDialOptions, build_grpc_endpoint, connect_channel, grpc_endpoint_uri,
|
||||
load_outgoing_grpc_tls, volume_server_client,
|
||||
};
|
||||
use crate::config::{NeedleMapKind, ReadMode, VolumeServerConfig};
|
||||
use crate::pb::volume_server_pb;
|
||||
use crate::security::tls::TlsPolicy;
|
||||
use crate::server::request_id::scope_request_id;
|
||||
use std::sync::{Arc, Mutex};
|
||||
use std::time::Duration;
|
||||
|
||||
const TEST_CERT_PEM: &str = "-----BEGIN CERTIFICATE-----\nMIIBPDCB76ADAgECAhRuRPQgeAu43BT/M7EfAWSdapVdYDAFBgMrZXAwFDESMBAG\nA1UEAwwJbG9jYWxob3N0MB4XDTI2MDcwNTE2MTUyOVoXDTM2MDcwMjE2MTUyOVow\nFDESMBAGA1UEAwwJbG9jYWxob3N0MCowBQYDK2VwAyEAr/3bNIFI+8V32oCiY6y+\nXRFmZpdNQ2g//VtRkT+nQg+jUzBRMB0GA1UdDgQWBBTsy9tLf1zPiXCQfgci6zNi\ndEzRSjAfBgNVHSMEGDAWgBTsy9tLf1zPiXCQfgci6zNidEzRSjAPBgNVHRMBAf8E\nBTADAQH/MAUGAytlcANBAIvsdw0IbvOBBkb9cd7BfMJfIP9pQQrAL03pCRWJFnFh\nSysaLVgFXI4T078IiaM874oO+iB+5vNbWEpc7CkGow4=\n-----END CERTIFICATE-----\n";
|
||||
const TEST_KEY_PEM: &str = "-----BEGIN PRIVATE KEY-----\nMC4CAQAwBQYDK2VwBCIEIHbyn71Kk+Y7KT3sBctit7uZpErpoH6qDbFj6P8qGaZH\n-----END PRIVATE KEY-----\n";
|
||||
@@ -399,4 +545,131 @@ mod tests {
|
||||
let err = parse_grpc_address("hostname").unwrap_err();
|
||||
assert!(err.contains("cannot parse"), "{}", err);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_parse_grpc_address_brackets_ipv6_literals() {
|
||||
use super::parse_grpc_address;
|
||||
// This used to come back as `::1:29333`, which is not a valid
|
||||
// authority: `build_grpc_endpoint` reads the last colon as the port
|
||||
// separator and rejects the rest.
|
||||
assert_eq!(parse_grpc_address("::1:19333").unwrap(), "[::1]:29333");
|
||||
assert_eq!(parse_grpc_address("::1:9333.19333").unwrap(), "[::1]:19333");
|
||||
// Already bracketed, so it is left alone.
|
||||
assert_eq!(parse_grpc_address("[::1]:9333").unwrap(), "[::1]:19333");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_build_grpc_endpoint_accepts_an_ipv6_master_address() {
|
||||
use super::parse_grpc_address;
|
||||
let endpoint = build_grpc_endpoint(&parse_grpc_address("::1:9333").unwrap(), None).unwrap();
|
||||
assert_eq!(endpoint.uri().port_u16(), Some(19333));
|
||||
}
|
||||
/// A minimal HTTP/2 server that records the gRPC request headers it is
|
||||
/// sent and answers every call with a trailers-only `unimplemented`. It is
|
||||
/// enough to prove what a helper-built client puts on the wire, without
|
||||
/// standing up the whole `VolumeServer` service behind a tonic server.
|
||||
async fn serve_header_capture() -> (u16, Arc<Mutex<Option<String>>>) {
|
||||
use hyper::service::service_fn;
|
||||
use hyper_util::rt::{TokioExecutor, TokioIo};
|
||||
|
||||
let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
|
||||
let port = listener.local_addr().unwrap().port();
|
||||
let seen: Arc<Mutex<Option<String>>> = Arc::new(Mutex::new(None));
|
||||
let captured = Arc::clone(&seen);
|
||||
|
||||
tokio::spawn(async move {
|
||||
while let Ok((stream, _)) = listener.accept().await {
|
||||
let captured = Arc::clone(&captured);
|
||||
tokio::spawn(async move {
|
||||
let _ = hyper::server::conn::http2::Builder::new(TokioExecutor::new())
|
||||
.serve_connection(
|
||||
TokioIo::new(stream),
|
||||
service_fn(move |req: hyper::Request<hyper::body::Incoming>| {
|
||||
let captured = Arc::clone(&captured);
|
||||
async move {
|
||||
let value = req
|
||||
.headers()
|
||||
.get("x-amz-request-id")
|
||||
.and_then(|v| v.to_str().ok())
|
||||
.map(str::to_string);
|
||||
*captured.lock().unwrap() = value;
|
||||
Ok::<_, std::convert::Infallible>(
|
||||
hyper::http::Response::builder()
|
||||
.status(200)
|
||||
.header("content-type", "application/grpc")
|
||||
.header("grpc-status", "12")
|
||||
.body(tonic::body::Body::empty())
|
||||
.unwrap(),
|
||||
)
|
||||
}
|
||||
}),
|
||||
)
|
||||
.await;
|
||||
});
|
||||
}
|
||||
});
|
||||
|
||||
(port, seen)
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_helper_built_client_sends_the_scoped_request_id() {
|
||||
let (port, seen) = serve_header_capture().await;
|
||||
|
||||
let channel = connect_channel(
|
||||
&format!("127.0.0.1:{}", port),
|
||||
None,
|
||||
GrpcDialOptions::unary(),
|
||||
)
|
||||
.await
|
||||
.expect("dial the header-capturing server");
|
||||
|
||||
let mut client = volume_server_client(channel);
|
||||
// The interceptor has a request id to forward only inside a scope, so
|
||||
// the call has to run inside one for this to test anything.
|
||||
let _ = scope_request_id("REQUEST-ID-ON-THE-WIRE".to_string(), async move {
|
||||
client
|
||||
.ping(volume_server_pb::PingRequest {
|
||||
target: String::new(),
|
||||
target_type: String::new(),
|
||||
})
|
||||
.await
|
||||
})
|
||||
.await;
|
||||
|
||||
assert_eq!(
|
||||
seen.lock().unwrap().as_deref(),
|
||||
Some("REQUEST-ID-ON-THE-WIRE"),
|
||||
"a client built by volume_server_client must carry the outgoing request id"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_dial_presets_match_the_call_sites_they_replace() {
|
||||
assert_eq!(
|
||||
GrpcDialOptions::unary().connect_timeout,
|
||||
Duration::from_secs(5)
|
||||
);
|
||||
assert_eq!(
|
||||
GrpcDialOptions::unary().request_timeout,
|
||||
Some(Duration::from_secs(10))
|
||||
);
|
||||
assert_eq!(
|
||||
GrpcDialOptions::long().connect_timeout,
|
||||
Duration::from_secs(5)
|
||||
);
|
||||
assert_eq!(
|
||||
GrpcDialOptions::long().request_timeout,
|
||||
Some(Duration::from_secs(30))
|
||||
);
|
||||
assert_eq!(
|
||||
GrpcDialOptions::stream().connect_timeout,
|
||||
Duration::from_secs(5)
|
||||
);
|
||||
assert_eq!(
|
||||
GrpcDialOptions::stream().request_timeout,
|
||||
None,
|
||||
"a streaming dial must not put a per-request deadline on the channel"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -12,10 +12,9 @@ use std::time::{Duration, SystemTime, UNIX_EPOCH};
|
||||
use tokio::sync::broadcast;
|
||||
use tracing::{error, info, warn};
|
||||
|
||||
use super::grpc_client::{GRPC_MAX_MESSAGE_SIZE, build_grpc_endpoint};
|
||||
use super::grpc_client::{GrpcDialOptions, connect_channel, master_client};
|
||||
use super::volume_server::VolumeServerState;
|
||||
use crate::pb::master_pb;
|
||||
use crate::pb::master_pb::seaweed_client::SeaweedClient;
|
||||
use crate::pb::volume_server_pb;
|
||||
use crate::remote_storage::s3_tier::{S3TierBackend, S3TierConfig};
|
||||
use crate::storage::store::Store;
|
||||
@@ -24,10 +23,11 @@ use crate::storage::volume_report::VolumeReportKey;
|
||||
use crate::storage::volume_report_hash::report_hash;
|
||||
|
||||
const DUPLICATE_UUID_RETRY_MESSAGE: &str = "duplicate UUIDs detected, retrying connection";
|
||||
const VOLUME_IO_ERROR_TOLERANCE: i32 = 3;
|
||||
const MAX_DUPLICATE_UUID_RETRIES: u32 = 3;
|
||||
const MAX_TTL_VOLUME_REMOVAL_DELAY: u32 = 10;
|
||||
|
||||
/// Configuration for the heartbeat client.
|
||||
#[derive(Clone)]
|
||||
pub struct HeartbeatConfig {
|
||||
pub ip: String,
|
||||
pub port: u16,
|
||||
@@ -222,7 +222,7 @@ async fn check_with_master(config: &HeartbeatConfig, state: &Arc<VolumeServerSta
|
||||
if changed {
|
||||
state.metrics_notify.notify_waiters();
|
||||
}
|
||||
apply_storage_backends(state, &resp.storage_backends);
|
||||
apply_storage_backends(&resp.storage_backends);
|
||||
info!(
|
||||
"Got master configuration from {}: metrics_address={}, metrics_interval={}s",
|
||||
master_addr, resp.metrics_address, resp.metrics_interval_seconds
|
||||
@@ -246,17 +246,8 @@ pub async fn try_get_master_configuration(
|
||||
grpc_addr: &str,
|
||||
tls: Option<&super::grpc_client::OutgoingGrpcTlsConfig>,
|
||||
) -> Result<master_pb::GetMasterConfigurationResponse, Box<dyn std::error::Error>> {
|
||||
let channel = build_grpc_endpoint(grpc_addr, tls)?
|
||||
.connect_timeout(Duration::from_secs(5))
|
||||
.timeout(Duration::from_secs(10))
|
||||
.connect()
|
||||
.await?;
|
||||
let mut client = SeaweedClient::with_interceptor(
|
||||
channel,
|
||||
super::request_id::outgoing_request_id_interceptor,
|
||||
)
|
||||
.max_decoding_message_size(GRPC_MAX_MESSAGE_SIZE)
|
||||
.max_encoding_message_size(GRPC_MAX_MESSAGE_SIZE);
|
||||
let channel = connect_channel(grpc_addr, tls, GrpcDialOptions::unary()).await?;
|
||||
let mut client = master_client(channel);
|
||||
let resp = client
|
||||
.get_master_configuration(master_pb::GetMasterConfigurationRequest {})
|
||||
.await?;
|
||||
@@ -391,18 +382,14 @@ async fn do_heartbeat(
|
||||
pulse: Duration,
|
||||
shutdown_rx: &mut broadcast::Receiver<()>,
|
||||
) -> Result<Option<String>, Box<dyn std::error::Error>> {
|
||||
let channel = build_grpc_endpoint(grpc_addr, state.outgoing_grpc_tls.as_ref())?
|
||||
.connect_timeout(Duration::from_secs(5))
|
||||
.timeout(Duration::from_secs(30))
|
||||
.connect()
|
||||
.await?;
|
||||
|
||||
let mut client = SeaweedClient::with_interceptor(
|
||||
channel,
|
||||
super::request_id::outgoing_request_id_interceptor,
|
||||
let channel = connect_channel(
|
||||
grpc_addr,
|
||||
state.outgoing_grpc_tls.as_ref(),
|
||||
GrpcDialOptions::long(),
|
||||
)
|
||||
.max_decoding_message_size(GRPC_MAX_MESSAGE_SIZE)
|
||||
.max_encoding_message_size(GRPC_MAX_MESSAGE_SIZE);
|
||||
.await?;
|
||||
|
||||
let mut client = master_client(channel);
|
||||
|
||||
let (tx, rx) = tokio::sync::mpsc::channel::<master_pb::Heartbeat>(32);
|
||||
|
||||
@@ -411,7 +398,8 @@ async fn do_heartbeat(
|
||||
state.store.read().unwrap().volume_report.reset();
|
||||
|
||||
// Keep track of what we sent, to generate delta updates
|
||||
let (initial_hb, initial_volumes) = collect_heartbeat_with_snapshot(config, state);
|
||||
let (initial_hb, initial_volumes) =
|
||||
off_runtime(config, state, collect_heartbeat_with_snapshot).await?;
|
||||
let mut last_volumes: HashMap<u32, VolumeIdentity> = volume_identities(&initial_volumes);
|
||||
let mut last_ec_shards = {
|
||||
let store = state.store.read().unwrap();
|
||||
@@ -477,7 +465,8 @@ async fn do_heartbeat(
|
||||
};
|
||||
if changed {
|
||||
let (adjusted_hb, adjusted_volumes) =
|
||||
collect_heartbeat_with_snapshot(config, state);
|
||||
off_runtime(config, state, collect_heartbeat_with_snapshot)
|
||||
.await?;
|
||||
last_volumes = volume_identities(&adjusted_volumes);
|
||||
last_ec_shards = {
|
||||
let store = state.store.read().unwrap();
|
||||
@@ -511,7 +500,8 @@ async fn do_heartbeat(
|
||||
let s = state.store.read().unwrap();
|
||||
s.maybe_adjust_volume_max();
|
||||
}
|
||||
let (current_hb, current_volumes) = collect_heartbeat_with_snapshot(config, state);
|
||||
let (current_hb, current_volumes) =
|
||||
off_runtime(config, state, collect_heartbeat_with_snapshot).await?;
|
||||
last_volumes = volume_identities(¤t_volumes);
|
||||
last_ec_shards = {
|
||||
let store = state.store.read().unwrap();
|
||||
@@ -540,7 +530,7 @@ async fn do_heartbeat(
|
||||
info!("Heartbeat stopping");
|
||||
return Ok(None);
|
||||
}
|
||||
let held_volumes = collect_volume_snapshot(config, state);
|
||||
let held_volumes = off_runtime(config, state, collect_volume_snapshot).await?;
|
||||
let current_volumes = volume_identities(&held_volumes);
|
||||
let current_ec_shards = {
|
||||
let store = state.store.read().unwrap();
|
||||
@@ -674,16 +664,15 @@ fn apply_metrics_push_settings(
|
||||
true
|
||||
}
|
||||
|
||||
fn apply_storage_backends(
|
||||
state: &VolumeServerState,
|
||||
storage_backends: &[master_pb::StorageBackend],
|
||||
) {
|
||||
/// Registers the master's S3 storage backends in the process-wide tier
|
||||
/// registry, the single place both the tier-move handlers and `Volume` itself
|
||||
/// resolve a backend from.
|
||||
fn apply_storage_backends(storage_backends: &[master_pb::StorageBackend]) {
|
||||
if storage_backends.is_empty() {
|
||||
return;
|
||||
}
|
||||
|
||||
let mut registry = state.s3_tier_registry.write().unwrap();
|
||||
let mut global_registry = crate::remote_storage::s3_tier::global_s3_tier_registry()
|
||||
let mut registry = crate::remote_storage::s3_tier::global_s3_tier_registry()
|
||||
.write()
|
||||
.unwrap();
|
||||
for backend in storage_backends {
|
||||
@@ -714,7 +703,6 @@ fn apply_storage_backends(
|
||||
backend.id.as_str()
|
||||
};
|
||||
register_s3_backend(&mut registry, backend, backend_id, &config);
|
||||
register_s3_backend(&mut global_registry, backend, backend_id, &config);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -798,6 +786,16 @@ fn volume_identities(
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Runs a store pass on the blocking pool: it stats every volume's files.
|
||||
async fn off_runtime<T: Send + 'static>(
|
||||
config: &HeartbeatConfig,
|
||||
state: &Arc<VolumeServerState>,
|
||||
pass: fn(&HeartbeatConfig, &Arc<VolumeServerState>) -> T,
|
||||
) -> Result<T, tokio::task::JoinError> {
|
||||
let (config, state) = (config.clone(), state.clone());
|
||||
tokio::task::spawn_blocking(move || pass(&config, &state)).await
|
||||
}
|
||||
|
||||
/// Collect volume information into a Heartbeat message.
|
||||
fn collect_heartbeat_with_snapshot(
|
||||
config: &HeartbeatConfig,
|
||||
@@ -806,15 +804,32 @@ fn collect_heartbeat_with_snapshot(
|
||||
master_pb::Heartbeat,
|
||||
Vec<master_pb::VolumeInformationMessage>,
|
||||
) {
|
||||
let mut store = state.store.write().unwrap();
|
||||
let (ec_shards, deleted_ec_shards) = store.delete_expired_ec_volumes();
|
||||
build_heartbeat_with_ec_status(
|
||||
config,
|
||||
&mut store,
|
||||
deleted_ec_shards,
|
||||
ec_shards.is_empty(),
|
||||
true,
|
||||
)
|
||||
let (_, expired_ec) = state.store.read().unwrap().find_expired_ec_volumes();
|
||||
let mut deleted_ec_shards = Vec::new();
|
||||
if !expired_ec.is_empty() {
|
||||
deleted_ec_shards = state
|
||||
.store
|
||||
.write()
|
||||
.unwrap()
|
||||
.remove_expired_ec_volumes(expired_ec)
|
||||
.0;
|
||||
}
|
||||
#[cfg(test)]
|
||||
{
|
||||
let store_id = state.store.read().unwrap().id.clone();
|
||||
read_phase_hook::park(read_phase_hook::Point::BeforeVolumePass, &store_id);
|
||||
}
|
||||
let (heartbeat, volumes, actions) = {
|
||||
let store = state.store.read().unwrap();
|
||||
#[cfg(test)]
|
||||
read_phase_hook::park(read_phase_hook::Point::VolumePass, &store.id);
|
||||
// Taken with the volume list: a shard mounted since the EC phase must
|
||||
// not go out as "no EC shards", which clears it on the master.
|
||||
let has_no_ec_shards = !has_reportable_ec_shards(&store);
|
||||
build_heartbeat_with_ec_status(config, &store, deleted_ec_shards, has_no_ec_shards, true)
|
||||
};
|
||||
apply_volume_actions(state, actions);
|
||||
(heartbeat, volumes)
|
||||
}
|
||||
|
||||
/// Lists the volumes the server holds without touching reporting state or
|
||||
@@ -824,8 +839,100 @@ fn collect_volume_snapshot(
|
||||
config: &HeartbeatConfig,
|
||||
state: &Arc<VolumeServerState>,
|
||||
) -> Vec<master_pb::VolumeInformationMessage> {
|
||||
let mut store = state.store.write().unwrap();
|
||||
build_heartbeat_with_ec_status(config, &mut store, Vec::new(), true, false).1
|
||||
let (_, volumes, actions) = build_heartbeat_with_ec_status(
|
||||
config,
|
||||
&state.store.read().unwrap(),
|
||||
Vec::new(),
|
||||
true,
|
||||
false,
|
||||
);
|
||||
apply_volume_actions(state, actions);
|
||||
volumes
|
||||
}
|
||||
|
||||
/// Store changes a heartbeat pass decides on under the read lock, applied
|
||||
/// under a short write lock afterwards. Entries are (disk index, volume id).
|
||||
#[derive(Default)]
|
||||
struct VolumeActions {
|
||||
delete_expired: Vec<(usize, VolumeId)>,
|
||||
quarantine: Vec<(usize, VolumeId)>,
|
||||
}
|
||||
|
||||
fn apply_volume_actions(state: &VolumeServerState, actions: VolumeActions) {
|
||||
if actions.delete_expired.is_empty() && actions.quarantine.is_empty() {
|
||||
return;
|
||||
}
|
||||
apply_volume_actions_to(&mut state.store.write().unwrap(), actions);
|
||||
}
|
||||
|
||||
/// Each target is re-checked: it may have been written to, replaced or removed
|
||||
/// since the read pass chose it.
|
||||
fn apply_volume_actions_to(store: &mut Store, actions: VolumeActions) {
|
||||
let volume_size_limit = store.volume_size_limit.load(Ordering::Relaxed);
|
||||
for (disk_id, vid) in actions.delete_expired {
|
||||
let Some(loc) = store.locations.get_mut(disk_id) else {
|
||||
continue;
|
||||
};
|
||||
let still_expired = loc.find_volume(vid).is_some_and(|vol| {
|
||||
!vol.should_quarantine()
|
||||
&& vol.is_expired(vol.dat_file_size().unwrap_or(0), volume_size_limit)
|
||||
&& vol.is_expired_long_enough(MAX_TTL_VOLUME_REMOVAL_DELAY)
|
||||
});
|
||||
if still_expired {
|
||||
let _ = loc.delete_volume(vid, false, false, false);
|
||||
}
|
||||
}
|
||||
for (disk_id, vid) in actions.quarantine {
|
||||
if let Some(vol) = store
|
||||
.locations
|
||||
.get_mut(disk_id)
|
||||
.and_then(|loc| loc.find_volume_mut(vid))
|
||||
&& vol.should_quarantine()
|
||||
{
|
||||
vol.set_no_write_or_delete(true);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod read_phase_hook {
|
||||
use std::sync::Mutex;
|
||||
use std::sync::mpsc::Receiver;
|
||||
use tokio::sync::oneshot::Sender;
|
||||
|
||||
#[derive(Clone, Copy, PartialEq)]
|
||||
pub(super) enum Point {
|
||||
/// Between the EC phase and the volume pass, holding no lock.
|
||||
BeforeVolumePass,
|
||||
/// Inside the volume pass, holding the store read lock.
|
||||
VolumePass,
|
||||
}
|
||||
|
||||
type Park = (Point, String, Sender<()>, Receiver<()>);
|
||||
static ARMED: Mutex<Vec<Park>> = Mutex::new(Vec::new());
|
||||
|
||||
/// Parks the next pass over the store with this id at `point`, announcing
|
||||
/// itself on `entered` and waiting until `release` is dropped.
|
||||
pub(super) fn arm(point: Point, store_id: &str, entered: Sender<()>, release: Receiver<()>) {
|
||||
ARMED
|
||||
.lock()
|
||||
.unwrap()
|
||||
.push((point, store_id.to_string(), entered, release));
|
||||
}
|
||||
|
||||
pub(super) fn park(point: Point, store_id: &str) {
|
||||
let armed = {
|
||||
let mut armed = ARMED.lock().unwrap();
|
||||
armed
|
||||
.iter()
|
||||
.position(|(p, id, _, _)| *p == point && !id.is_empty() && id == store_id)
|
||||
.map(|i| armed.swap_remove(i))
|
||||
};
|
||||
if let Some((_, _, entered, release)) = armed {
|
||||
let _ = entered.send(());
|
||||
let _ = release.recv();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// The heartbeat alone, without the volume snapshot the send loop pairs it
|
||||
@@ -868,27 +975,42 @@ fn collect_location_metadata(
|
||||
(location_uuids, disk_tags)
|
||||
}
|
||||
|
||||
/// Whether a heartbeat would report any EC shard: Go's non-empty
|
||||
/// `ecVolumeMessages` from `deleteExpiredEcVolumes`.
|
||||
fn has_reportable_ec_shards(store: &Store) -> bool {
|
||||
store.locations.iter().any(|loc| {
|
||||
loc.ec_volumes().any(|(_, ec_vol)| {
|
||||
!ec_vol.is_time_to_destroy()
|
||||
&& !ec_vol.should_quarantine()
|
||||
&& ec_vol.shards.iter().any(Option::is_some)
|
||||
})
|
||||
})
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
fn build_heartbeat(config: &HeartbeatConfig, store: &mut Store) -> master_pb::Heartbeat {
|
||||
let has_no_ec_shards = collect_live_ec_shards(store, false).is_empty();
|
||||
build_heartbeat_with_ec_status(config, store, Vec::new(), has_no_ec_shards, true).0
|
||||
let (heartbeat, _, actions) =
|
||||
build_heartbeat_with_ec_status(config, store, Vec::new(), has_no_ec_shards, true);
|
||||
apply_volume_actions_to(store, actions);
|
||||
heartbeat
|
||||
}
|
||||
|
||||
/// Returns the heartbeat to send and, separately, every volume held. The
|
||||
/// caller derives mount and unmount deltas by diffing successive snapshots, so
|
||||
/// it must not be handed the partial list a heartbeat may carry.
|
||||
/// Returns the heartbeat to send, every volume held, and the store changes
|
||||
/// the pass decided on. The caller derives mount and unmount deltas by diffing
|
||||
/// successive snapshots, so it must not be handed the partial list a heartbeat
|
||||
/// may carry.
|
||||
fn build_heartbeat_with_ec_status(
|
||||
config: &HeartbeatConfig,
|
||||
store: &mut Store,
|
||||
store: &Store,
|
||||
deleted_ec_shards: Vec<master_pb::VolumeEcShardInformationMessage>,
|
||||
has_no_ec_shards: bool,
|
||||
commit_report: bool,
|
||||
) -> (
|
||||
master_pb::Heartbeat,
|
||||
Vec<master_pb::VolumeInformationMessage>,
|
||||
VolumeActions,
|
||||
) {
|
||||
const MAX_TTL_VOLUME_REMOVAL_DELAY: u32 = 10;
|
||||
|
||||
#[derive(Default)]
|
||||
struct ReadOnlyCounts {
|
||||
is_read_only: u32,
|
||||
@@ -918,8 +1040,9 @@ fn build_heartbeat_with_ec_status(
|
||||
|
||||
// Per-disk effective max for DiskTag, captured alongside the per-type sum.
|
||||
let mut disk_max_by_id = vec![0i32; store.locations.len()];
|
||||
let mut actions = VolumeActions::default();
|
||||
|
||||
for (disk_id, loc) in store.locations.iter_mut().enumerate() {
|
||||
for (disk_id, loc) in store.locations.iter().enumerate() {
|
||||
let disk_type_str = loc.disk_type.to_string();
|
||||
let mut effective_max_count = loc.max_volume_count.load(Ordering::Relaxed);
|
||||
if loc.is_disk_space_low.load(Ordering::Relaxed) {
|
||||
@@ -943,8 +1066,6 @@ fn build_heartbeat_with_ec_status(
|
||||
*disk_free_bytes.entry(disk_type_str).or_insert(0) +=
|
||||
loc.disk_free_bytes.load(Ordering::Relaxed);
|
||||
|
||||
let mut delete_vids = Vec::new();
|
||||
let mut quarantine_vids: Vec<VolumeId> = Vec::new();
|
||||
for (_, vol) in loc.iter_volumes() {
|
||||
let cur_max = vol.max_file_key();
|
||||
if cur_max > max_file_key {
|
||||
@@ -954,8 +1075,8 @@ fn build_heartbeat_with_ec_status(
|
||||
let volume_size = vol.dat_file_size().unwrap_or(0);
|
||||
let mut should_delete_volume = false;
|
||||
|
||||
let (_, io_count, io_quarantined) = vol.get_io_error_state();
|
||||
if io_quarantined || io_count >= VOLUME_IO_ERROR_TOLERANCE {
|
||||
if vol.should_quarantine() {
|
||||
let (_, io_count, io_quarantined) = vol.get_io_error_state();
|
||||
if !io_quarantined {
|
||||
vol.mark_io_quarantined();
|
||||
warn!(
|
||||
@@ -964,7 +1085,7 @@ fn build_heartbeat_with_ec_status(
|
||||
);
|
||||
}
|
||||
quarantined_volumes += 1;
|
||||
quarantine_vids.push(vol.id);
|
||||
actions.quarantine.push((disk_id, vol.id));
|
||||
continue;
|
||||
} else if !vol.is_expired(volume_size, volume_size_limit) {
|
||||
// Detect phantom volumes: the .dat was unlinked from disk but is still
|
||||
@@ -973,7 +1094,7 @@ fn build_heartbeat_with_ec_status(
|
||||
// whose .dat legitimately lives in cloud storage. Only a present .dat is
|
||||
// cached for 30s; a missing one is re-checked every heartbeat so the volume
|
||||
// stays suppressed until the file returns. See issues/10004
|
||||
if vol.file_count() > 0 && !vol.has_remote_file {
|
||||
if vol.file_count() > 0 && !vol.has_remote_file() {
|
||||
const DISK_CHECK_INTERVAL_NS: i64 = 30 * 1_000_000_000;
|
||||
let now_ns = SystemTime::now()
|
||||
.duration_since(UNIX_EPOCH)
|
||||
@@ -1032,7 +1153,7 @@ fn build_heartbeat_with_ec_status(
|
||||
}
|
||||
volumes.push(volume_message);
|
||||
} else if vol.is_expired_long_enough(MAX_TTL_VOLUME_REMOVAL_DELAY) {
|
||||
delete_vids.push(vol.id);
|
||||
actions.delete_expired.push((disk_id, vol.id));
|
||||
should_delete_volume = true;
|
||||
}
|
||||
|
||||
@@ -1063,16 +1184,6 @@ fn build_heartbeat_with_ec_status(
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for vid in delete_vids {
|
||||
let _ = loc.delete_volume(vid, false, false);
|
||||
}
|
||||
|
||||
for vid in quarantine_vids {
|
||||
if let Some(vol) = loc.find_volume_mut(vid) {
|
||||
vol.set_no_write_or_delete(true);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Update disk size and read-only gauges
|
||||
@@ -1169,7 +1280,7 @@ fn build_heartbeat_with_ec_status(
|
||||
disk_tags,
|
||||
..Default::default()
|
||||
};
|
||||
(heartbeat, volumes)
|
||||
(heartbeat, volumes, actions)
|
||||
}
|
||||
|
||||
fn collect_live_ec_shards(
|
||||
@@ -1250,7 +1361,6 @@ mod tests {
|
||||
READ_ONLY_LABEL_NO_WRITE_CAN_DELETE, READ_ONLY_LABEL_NO_WRITE_OR_DELETE,
|
||||
READ_ONLY_VOLUME_GAUGE,
|
||||
};
|
||||
use crate::remote_storage::s3_tier::S3TierRegistry;
|
||||
use crate::security::{Guard, SigningKey};
|
||||
use crate::storage::needle_map::NeedleMapKind;
|
||||
use crate::storage::types::{DiskType, VolumeId};
|
||||
@@ -1302,7 +1412,6 @@ mod tests {
|
||||
pre_stop_seconds: 0,
|
||||
volume_state_notify: tokio::sync::Notify::new(),
|
||||
write_queue: std::sync::OnceLock::new(),
|
||||
s3_tier_registry: std::sync::RwLock::new(S3TierRegistry::new()),
|
||||
read_mode: ReadMode::Local,
|
||||
allow_untrusted_remote_endpoints: false,
|
||||
master_url: String::new(),
|
||||
@@ -1604,7 +1713,7 @@ mod tests {
|
||||
|
||||
// What the notify path does: collect a snapshot, send a message of its own.
|
||||
let snapshot =
|
||||
build_heartbeat_with_ec_status(&test_config(), &mut store, Vec::new(), true, false).1;
|
||||
build_heartbeat_with_ec_status(&test_config(), &store, Vec::new(), true, false).1;
|
||||
assert_eq!(snapshot.len(), 3);
|
||||
|
||||
let heartbeat = build_heartbeat(&test_config(), &mut store);
|
||||
@@ -1688,8 +1797,9 @@ mod tests {
|
||||
{
|
||||
let (_, volume) = store.find_volume_mut(VolumeId(17)).unwrap();
|
||||
volume.set_read_only().unwrap();
|
||||
volume.volume_info.files.push(Default::default());
|
||||
volume.refresh_remote_write_mode().unwrap();
|
||||
volume
|
||||
.update_remote_files(|files| files.push(Default::default()))
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
let heartbeat = build_heartbeat(&test_config(), &mut store);
|
||||
@@ -1796,7 +1906,7 @@ mod tests {
|
||||
1.0
|
||||
);
|
||||
|
||||
assert!(store.unmount_volume(VolumeId(21)));
|
||||
assert!(store.unmount_volume(VolumeId(21)).unwrap());
|
||||
build_heartbeat(&test_config(), &mut store);
|
||||
|
||||
assert_eq!(
|
||||
@@ -1867,6 +1977,8 @@ mod tests {
|
||||
|
||||
let shard_path = format!("{}/ec_metrics_case_27.ec00", dir);
|
||||
std::fs::write(&shard_path, b"ec-shard").unwrap();
|
||||
// An EC volume needs its .ecx to mount.
|
||||
std::fs::write(format!("{}/ec_metrics_case_27.ecx", dir), [0u8; 16]).unwrap();
|
||||
store.locations[0]
|
||||
.mount_ec_shards(VolumeId(27), "ec_metrics_case", &[0], "")
|
||||
.unwrap();
|
||||
@@ -1911,6 +2023,8 @@ mod tests {
|
||||
.unwrap();
|
||||
|
||||
std::fs::write(format!("{}/expired_heartbeat_ec_31.ec00", dir), b"expired").unwrap();
|
||||
// An EC volume needs its .ecx to mount.
|
||||
std::fs::write(format!("{}/expired_heartbeat_ec_31.ecx", dir), [0u8; 16]).unwrap();
|
||||
store.locations[0]
|
||||
.mount_ec_shards(VolumeId(31), "expired_heartbeat_ec", &[0], "")
|
||||
.unwrap();
|
||||
@@ -2026,6 +2140,235 @@ mod tests {
|
||||
assert!(volume.is_no_write_or_delete());
|
||||
}
|
||||
|
||||
// The pass only reads the store, so it must not shut out the readers that
|
||||
// serve traffic while it stats every volume.
|
||||
#[tokio::test]
|
||||
async fn test_heartbeat_collection_leaves_the_store_readable() {
|
||||
let temp_dir = tempfile::tempdir().unwrap();
|
||||
let mut store = reporting_store(temp_dir.path().to_str().unwrap(), 2);
|
||||
store.id = "heartbeat-read-phase-park".to_string();
|
||||
let state = test_state_with_store(store);
|
||||
|
||||
let (entered_tx, entered_rx) = tokio::sync::oneshot::channel();
|
||||
let (release_tx, release_rx) = std::sync::mpsc::channel::<()>();
|
||||
read_phase_hook::arm(
|
||||
read_phase_hook::Point::VolumePass,
|
||||
"heartbeat-read-phase-park",
|
||||
entered_tx,
|
||||
release_rx,
|
||||
);
|
||||
let collection = {
|
||||
let state = state.clone();
|
||||
tokio::spawn(async move {
|
||||
off_runtime(&test_config(), &state, collect_heartbeat_with_snapshot).await
|
||||
})
|
||||
};
|
||||
tokio::time::timeout(Duration::from_secs(10), entered_rx)
|
||||
.await
|
||||
.expect("the pass never reached its read phase")
|
||||
.unwrap();
|
||||
|
||||
let readable = state.store.try_read().is_ok();
|
||||
drop(release_tx);
|
||||
let (heartbeat, volumes) = collection.await.unwrap().unwrap();
|
||||
|
||||
assert!(readable, "a parked heartbeat pass shut out store readers");
|
||||
assert_eq!(heartbeat.volumes.len(), 2);
|
||||
assert_eq!(volumes.len(), 2);
|
||||
}
|
||||
|
||||
// What the read pass decided on can go stale before the write lock is
|
||||
// taken: each action must re-check its target, not act on whatever now
|
||||
// holds the id.
|
||||
#[test]
|
||||
fn test_volume_actions_skip_volumes_changed_since_the_read_pass() {
|
||||
let temp_dir = tempfile::tempdir().unwrap();
|
||||
let dir = temp_dir.path().to_str().unwrap();
|
||||
|
||||
let mut store = Store::new(NeedleMapKind::InMemory);
|
||||
store
|
||||
.add_location(
|
||||
dir,
|
||||
dir,
|
||||
8,
|
||||
DiskType::HardDrive,
|
||||
MinFreeSpace::Percent(1.0),
|
||||
Vec::new(),
|
||||
)
|
||||
.unwrap();
|
||||
store.volume_size_limit.store(1, Ordering::Relaxed);
|
||||
let now = SystemTime::now()
|
||||
.duration_since(UNIX_EPOCH)
|
||||
.unwrap_or_default()
|
||||
.as_secs();
|
||||
for id in [61, 62, 63] {
|
||||
store
|
||||
.add_volume(
|
||||
VolumeId(id),
|
||||
DiskType::HardDrive,
|
||||
&VolumeSpec {
|
||||
collection: "stale_action_case",
|
||||
ttl: Some(crate::storage::needle::ttl::TTL::read("20m").unwrap()),
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
let (_, volume) = store.find_volume_mut(VolumeId(id)).unwrap();
|
||||
volume.set_last_io_error_for_test(None);
|
||||
volume.set_last_modified_ts_for_test(now.saturating_sub(60 * 60));
|
||||
std::fs::OpenOptions::new()
|
||||
.write(true)
|
||||
.open(volume.dat_path())
|
||||
.unwrap()
|
||||
.set_len((crate::storage::super_block::SUPER_BLOCK_SIZE + 1) as u64)
|
||||
.unwrap();
|
||||
}
|
||||
store
|
||||
.add_volume(
|
||||
VolumeId(64),
|
||||
DiskType::HardDrive,
|
||||
&VolumeSpec {
|
||||
collection: "stale_action_case",
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
let (_, volume) = store.find_volume_mut(VolumeId(64)).unwrap();
|
||||
volume.set_last_io_error_for_test(Some("input/output error"));
|
||||
|
||||
let (heartbeat, _, actions) =
|
||||
build_heartbeat_with_ec_status(&test_config(), &store, Vec::new(), true, true);
|
||||
assert!(heartbeat.volumes.is_empty());
|
||||
assert_eq!(actions.delete_expired.len(), 3);
|
||||
assert_eq!(actions.quarantine, vec![(0, VolumeId(64))]);
|
||||
|
||||
// Between the phases: 61 is deleted by someone else, 62 and 64 are
|
||||
// replaced by fresh copies under the same ids; 63 is left alone.
|
||||
store
|
||||
.delete_volume(VolumeId(61), false, false, false)
|
||||
.unwrap();
|
||||
for id in [62, 64] {
|
||||
store
|
||||
.delete_volume(VolumeId(id), false, false, false)
|
||||
.unwrap();
|
||||
store
|
||||
.add_volume(
|
||||
VolumeId(id),
|
||||
DiskType::HardDrive,
|
||||
&VolumeSpec {
|
||||
collection: "stale_action_case",
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
apply_volume_actions_to(&mut store, actions);
|
||||
|
||||
assert!(!store.has_volume(VolumeId(61)));
|
||||
assert!(store.has_volume(VolumeId(62)), "a fresh copy was deleted");
|
||||
assert!(
|
||||
!store.has_volume(VolumeId(63)),
|
||||
"the expired volume survived"
|
||||
);
|
||||
let (_, fresh) = store.find_volume(VolumeId(64)).unwrap();
|
||||
assert!(
|
||||
!fresh.is_no_write_or_delete(),
|
||||
"a fresh copy was quarantined"
|
||||
);
|
||||
}
|
||||
|
||||
// A shard mounted after the EC phase is held when the volume list is
|
||||
// taken; reporting "no EC shards" alongside it would clear it on the master.
|
||||
#[tokio::test]
|
||||
async fn test_ec_shard_mounted_after_the_ec_phase_is_not_reported_absent() {
|
||||
let temp_dir = tempfile::tempdir().unwrap();
|
||||
let dir = temp_dir.path().to_str().unwrap();
|
||||
let mut store = reporting_store(dir, 1);
|
||||
store.id = "heartbeat-ec-mount-between-passes".to_string();
|
||||
let state = test_state_with_store(store);
|
||||
std::fs::write(format!("{}/ec_mount_race_73.ec00", dir), b"shard").unwrap();
|
||||
std::fs::write(format!("{}/ec_mount_race_73.ecx", dir), [0u8; 16]).unwrap();
|
||||
|
||||
let (entered_tx, entered_rx) = tokio::sync::oneshot::channel();
|
||||
let (release_tx, release_rx) = std::sync::mpsc::channel::<()>();
|
||||
read_phase_hook::arm(
|
||||
read_phase_hook::Point::BeforeVolumePass,
|
||||
"heartbeat-ec-mount-between-passes",
|
||||
entered_tx,
|
||||
release_rx,
|
||||
);
|
||||
let collection = {
|
||||
let state = state.clone();
|
||||
tokio::spawn(async move {
|
||||
off_runtime(&test_config(), &state, collect_heartbeat_with_snapshot).await
|
||||
})
|
||||
};
|
||||
tokio::time::timeout(Duration::from_secs(10), entered_rx)
|
||||
.await
|
||||
.expect("the pass never finished its EC phase")
|
||||
.unwrap();
|
||||
|
||||
state.store.write().unwrap().locations[0]
|
||||
.mount_ec_shards(VolumeId(73), "ec_mount_race", &[0], "")
|
||||
.unwrap();
|
||||
drop(release_tx);
|
||||
let (heartbeat, _) = collection.await.unwrap().unwrap();
|
||||
|
||||
assert!(
|
||||
!heartbeat.has_no_ec_shards,
|
||||
"a mounted EC shard was reported absent"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_expired_ec_volume_gone_before_removal_is_not_reported_deleted() {
|
||||
let temp_dir = tempfile::tempdir().unwrap();
|
||||
let dir = temp_dir.path().to_str().unwrap();
|
||||
|
||||
let mut store = Store::new(NeedleMapKind::InMemory);
|
||||
store
|
||||
.add_location(
|
||||
dir,
|
||||
dir,
|
||||
8,
|
||||
DiskType::HardDrive,
|
||||
MinFreeSpace::Percent(1.0),
|
||||
Vec::new(),
|
||||
)
|
||||
.unwrap();
|
||||
for id in [71, 72] {
|
||||
std::fs::write(format!("{}/stale_ec_case_{}.ec00", dir, id), b"expired").unwrap();
|
||||
std::fs::write(format!("{}/stale_ec_case_{}.ecx", dir, id), [0u8; 16]).unwrap();
|
||||
store.locations[0]
|
||||
.mount_ec_shards(VolumeId(id), "stale_ec_case", &[0], "")
|
||||
.unwrap();
|
||||
store
|
||||
.find_ec_volume_mut(VolumeId(id))
|
||||
.unwrap()
|
||||
.expire_at_sec = 1;
|
||||
}
|
||||
|
||||
let (ec_shards, mut expired) = store.find_expired_ec_volumes();
|
||||
expired.sort();
|
||||
assert!(ec_shards.is_empty());
|
||||
assert_eq!(expired, vec![(0, VolumeId(71)), (0, VolumeId(72))]);
|
||||
|
||||
// Between the phases: 71 is destroyed by someone else, 72 is remounted
|
||||
// without an expiry.
|
||||
store.remove_ec_volume(VolumeId(71)).unwrap().destroy();
|
||||
store.remove_ec_volume(VolumeId(72)).unwrap();
|
||||
store.locations[0]
|
||||
.mount_ec_shards(VolumeId(72), "stale_ec_case", &[0], "")
|
||||
.unwrap();
|
||||
let (deleted, still_held) = store.remove_expired_ec_volumes(expired);
|
||||
|
||||
assert!(deleted.is_empty());
|
||||
assert_eq!(still_held.len(), 1);
|
||||
assert_eq!(still_held[0].id, 72);
|
||||
assert!(store.has_ec_volume(VolumeId(72)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_build_heartbeat_includes_remote_storage_name_and_key() {
|
||||
let temp_dir = tempfile::tempdir().unwrap();
|
||||
@@ -2054,15 +2397,15 @@ mod tests {
|
||||
.unwrap();
|
||||
let (_, volume) = store.find_volume_mut(VolumeId(71)).unwrap();
|
||||
volume
|
||||
.volume_info
|
||||
.files
|
||||
.push(crate::storage::volume::PbRemoteFile {
|
||||
backend_type: "s3".to_string(),
|
||||
backend_id: "archive".to_string(),
|
||||
key: "volumes/71.dat".to_string(),
|
||||
..Default::default()
|
||||
});
|
||||
volume.refresh_remote_write_mode().unwrap();
|
||||
.update_remote_files(|files| {
|
||||
files.push(crate::storage::volume::PbRemoteFile {
|
||||
backend_type: "s3".to_string(),
|
||||
backend_id: "archive".to_string(),
|
||||
key: "volumes/71.dat".to_string(),
|
||||
..Default::default()
|
||||
})
|
||||
})
|
||||
.unwrap();
|
||||
|
||||
let heartbeat = build_heartbeat(&test_config(), &mut store);
|
||||
|
||||
@@ -2071,65 +2414,56 @@ mod tests {
|
||||
assert_eq!(heartbeat.volumes[0].remote_storage_key, "volumes/71.dat");
|
||||
}
|
||||
|
||||
// Not hermetic, and cannot be made so cheaply: `register_s3_backend` skips
|
||||
// a name that is already registered, so had another test put `s3` or
|
||||
// `s3.default` in the process-wide registry first, this would pass without
|
||||
// proving that this call registered anything. Removing them afterwards is
|
||||
// no better — unlike the tier tests' unique ids, the bare `s3` alias is the
|
||||
// production one. Nothing else in the tree registers those two names.
|
||||
#[test]
|
||||
fn test_apply_storage_backends_registers_s3_default_aliases() {
|
||||
let state = test_state_with_store(Store::new(NeedleMapKind::InMemory));
|
||||
// Do not call clear() on the global registry — other tests may be
|
||||
// running concurrently. Just register our entries and verify them.
|
||||
|
||||
apply_storage_backends(
|
||||
&state,
|
||||
&[master_pb::StorageBackend {
|
||||
r#type: "s3".to_string(),
|
||||
id: "default".to_string(),
|
||||
properties: std::collections::HashMap::from([
|
||||
("aws_access_key_id".to_string(), "access".to_string()),
|
||||
("aws_secret_access_key".to_string(), "secret".to_string()),
|
||||
("bucket".to_string(), "bucket-a".to_string()),
|
||||
("region".to_string(), "us-west-2".to_string()),
|
||||
("endpoint".to_string(), "http://127.0.0.1:8333".to_string()),
|
||||
("storage_class".to_string(), "STANDARD".to_string()),
|
||||
("force_path_style".to_string(), "false".to_string()),
|
||||
]),
|
||||
}],
|
||||
);
|
||||
apply_storage_backends(&[master_pb::StorageBackend {
|
||||
r#type: "s3".to_string(),
|
||||
id: "default".to_string(),
|
||||
properties: std::collections::HashMap::from([
|
||||
("aws_access_key_id".to_string(), "access".to_string()),
|
||||
("aws_secret_access_key".to_string(), "secret".to_string()),
|
||||
("bucket".to_string(), "bucket-a".to_string()),
|
||||
("region".to_string(), "us-west-2".to_string()),
|
||||
("endpoint".to_string(), "http://127.0.0.1:8333".to_string()),
|
||||
("storage_class".to_string(), "STANDARD".to_string()),
|
||||
("force_path_style".to_string(), "false".to_string()),
|
||||
]),
|
||||
}]);
|
||||
|
||||
let registry = state.s3_tier_registry.read().unwrap();
|
||||
assert!(registry.get("s3.default").is_some());
|
||||
assert!(registry.get("s3").is_some());
|
||||
let global_registry = crate::remote_storage::s3_tier::global_s3_tier_registry()
|
||||
let registry = crate::remote_storage::s3_tier::global_s3_tier_registry()
|
||||
.read()
|
||||
.unwrap();
|
||||
assert!(global_registry.get("s3.default").is_some());
|
||||
assert!(global_registry.get("s3").is_some());
|
||||
assert!(registry.get("s3.default").is_some());
|
||||
assert!(registry.get("s3").is_some());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_apply_storage_backends_ignores_unsupported_types() {
|
||||
let state = test_state_with_store(Store::new(NeedleMapKind::InMemory));
|
||||
// Do not call clear() on the global registry — other tests may be
|
||||
// running concurrently.
|
||||
|
||||
apply_storage_backends(
|
||||
&state,
|
||||
&[master_pb::StorageBackend {
|
||||
r#type: "rclone".to_string(),
|
||||
id: "default".to_string(),
|
||||
properties: std::collections::HashMap::new(),
|
||||
}],
|
||||
);
|
||||
apply_storage_backends(&[master_pb::StorageBackend {
|
||||
r#type: "rclone".to_string(),
|
||||
id: "default".to_string(),
|
||||
properties: std::collections::HashMap::new(),
|
||||
}]);
|
||||
|
||||
// The per-state registry is freshly created and should have no entries
|
||||
// since "rclone" is unsupported.
|
||||
let registry = state.s3_tier_registry.read().unwrap();
|
||||
assert!(registry.names().is_empty());
|
||||
// Only check that the unsupported type was not added to the global
|
||||
// registry. Other tests may have their own entries present.
|
||||
let global_registry = crate::remote_storage::s3_tier::global_s3_tier_registry()
|
||||
let registry = crate::remote_storage::s3_tier::global_s3_tier_registry()
|
||||
.read()
|
||||
.unwrap();
|
||||
assert!(global_registry.get("rclone.default").is_none());
|
||||
assert!(global_registry.get("rclone").is_none());
|
||||
assert!(registry.get("rclone.default").is_none());
|
||||
assert!(registry.get("rclone").is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -2221,6 +2555,8 @@ mod tests {
|
||||
let previous = collect_ec_shard_delta_messages(&store);
|
||||
|
||||
std::fs::write(format!("{}/ec_delta_case_81.ec00", dir), b"delta").unwrap();
|
||||
// An EC volume needs its .ecx to mount.
|
||||
std::fs::write(format!("{}/ec_delta_case_81.ecx", dir), [0u8; 16]).unwrap();
|
||||
store.locations[0]
|
||||
.mount_ec_shards(VolumeId(81), "ec_delta_case", &[0], "")
|
||||
.unwrap();
|
||||
|
||||
@@ -36,16 +36,20 @@ pub fn collect_mem_status() -> volume_server_pb::MemStatus {
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
fn get_system_memory_linux() -> Option<(u64, u64)> {
|
||||
unsafe {
|
||||
let mut info: libc::sysinfo = std::mem::zeroed();
|
||||
if libc::sysinfo(&mut info) == 0 {
|
||||
let unit = info.mem_unit as u64;
|
||||
let total = info.totalram as u64 * unit;
|
||||
let free = info.freeram as u64 * unit;
|
||||
return Some((total, free));
|
||||
}
|
||||
// SAFETY: `libc::sysinfo` is plain data — integers and trailing padding,
|
||||
// no pointers and no restricted niches — so the all-zero value is a valid
|
||||
// one for the kernel to overwrite.
|
||||
let mut info: libc::sysinfo = unsafe { std::mem::zeroed() };
|
||||
// SAFETY: `&mut info` is a live, aligned, exclusive pointer to a
|
||||
// `sysinfo` that the kernel only writes through, and its fields are read
|
||||
// below only after the call reports success.
|
||||
if unsafe { libc::sysinfo(&mut info) } != 0 {
|
||||
return None;
|
||||
}
|
||||
None
|
||||
let unit = info.mem_unit as u64;
|
||||
let total = info.totalram as u64 * unit;
|
||||
let free = info.freeram as u64 * unit;
|
||||
Some((total, free))
|
||||
}
|
||||
|
||||
#[cfg(target_os = "linux")]
|
||||
|
||||
@@ -1,3 +1,8 @@
|
||||
use tonic::Status;
|
||||
|
||||
use crate::remote_storage::s3_tier::TierError;
|
||||
use crate::storage::volume::VolumeError;
|
||||
|
||||
#[cfg(unix)]
|
||||
pub mod debug;
|
||||
pub mod grpc_client;
|
||||
@@ -13,3 +18,102 @@ pub mod store_ec;
|
||||
pub mod ui;
|
||||
pub mod volume_server;
|
||||
pub mod write_queue;
|
||||
|
||||
/// Map a storage error onto the gRPC code that describes it.
|
||||
impl From<VolumeError> for Status {
|
||||
fn from(err: VolumeError) -> Self {
|
||||
let message = err.to_string();
|
||||
match err {
|
||||
VolumeError::NotFound
|
||||
| VolumeError::VolumeNotFound(_)
|
||||
| VolumeError::Tier(TierError::NotFound(_)) => Status::not_found(message),
|
||||
VolumeError::ReadOnly | VolumeError::NotEmpty => Status::failed_precondition(message),
|
||||
VolumeError::InsufficientSpace { .. } => Status::resource_exhausted(message),
|
||||
VolumeError::AlreadyExists => Status::already_exists(message),
|
||||
_ => Status::internal(message),
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Same mapping, with the RPC's own context prefixed (`compact volume 7: ...`).
|
||||
pub fn status_with_context(context: &str, err: VolumeError) -> Status {
|
||||
let status = Status::from(err);
|
||||
Status::new(status.code(), format!("{context}: {}", status.message()))
|
||||
}
|
||||
|
||||
/// Render a configured disk directory as an absolute path for display, so the
|
||||
/// status JSON and the UI show the same thing for a relative `-dir`. Falls
|
||||
/// back to the configured spelling when the current directory cannot be read.
|
||||
pub(crate) fn absolute_display_path(path: &str) -> String {
|
||||
let p = std::path::Path::new(path);
|
||||
if p.is_absolute() {
|
||||
return path.to_string();
|
||||
}
|
||||
std::env::current_dir()
|
||||
.map(|cwd| cwd.join(p).to_string_lossy().to_string())
|
||||
.unwrap_or_else(|_| path.to_string())
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::storage::types::VolumeId;
|
||||
|
||||
#[test]
|
||||
fn test_volume_error_maps_to_grpc_code() {
|
||||
use tonic::Code;
|
||||
|
||||
let code = |e: VolumeError| Status::from(e).code();
|
||||
assert_eq!(
|
||||
code(VolumeError::VolumeNotFound(VolumeId(7))),
|
||||
Code::NotFound
|
||||
);
|
||||
assert_eq!(code(VolumeError::NotFound), Code::NotFound);
|
||||
assert_eq!(code(VolumeError::ReadOnly), Code::FailedPrecondition);
|
||||
assert_eq!(
|
||||
code(VolumeError::InsufficientSpace {
|
||||
vid: VolumeId(7),
|
||||
required: 2,
|
||||
free: 1,
|
||||
}),
|
||||
Code::ResourceExhausted
|
||||
);
|
||||
assert_eq!(code(VolumeError::AlreadyExists), Code::AlreadyExists);
|
||||
assert_eq!(code(VolumeError::NotInitialized), Code::Internal);
|
||||
assert_eq!(
|
||||
code(TierError::NotFound("gone".into()).into()),
|
||||
Code::NotFound
|
||||
);
|
||||
for tier in [
|
||||
TierError::Io("io".into()),
|
||||
TierError::RuntimeUnavailable("rt".into()),
|
||||
TierError::Aborted("bye".into()),
|
||||
] {
|
||||
assert_eq!(code(tier.into()), Code::Internal);
|
||||
}
|
||||
|
||||
let status = status_with_context(
|
||||
"backend s3.default copy file /data/1.dat",
|
||||
TierError::NotFound("failed to head object k: NotFound".into()).into(),
|
||||
);
|
||||
assert_eq!(status.code(), Code::NotFound);
|
||||
assert_eq!(
|
||||
status.message(),
|
||||
"backend s3.default copy file /data/1.dat: failed to head object k: NotFound"
|
||||
);
|
||||
|
||||
let status = status_with_context(
|
||||
"compact volume 7",
|
||||
VolumeError::InsufficientSpace {
|
||||
vid: VolumeId(7),
|
||||
required: 2,
|
||||
free: 1,
|
||||
},
|
||||
);
|
||||
assert_eq!(status.code(), Code::ResourceExhausted);
|
||||
assert_eq!(
|
||||
status.message(),
|
||||
"compact volume 7: not enough free space: required 2, free 1"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -26,7 +26,7 @@
|
||||
//! cache write-back briefly reacquires the EcVolume's internal
|
||||
//! `RwLock` so we do not contend with the Store-level lock at all.
|
||||
|
||||
use std::collections::HashMap;
|
||||
use std::collections::{HashMap, HashSet};
|
||||
use std::fs;
|
||||
use std::io;
|
||||
use std::sync::Arc;
|
||||
@@ -38,12 +38,13 @@ use reed_solomon_erasure::galois_8::ReedSolomon;
|
||||
use tokio::sync::Semaphore;
|
||||
use tonic::Request;
|
||||
|
||||
use crate::pb::master_pb::{self, LookupEcVolumeRequest, seaweed_client::SeaweedClient};
|
||||
use crate::pb::master_pb::{self, LookupEcVolumeRequest};
|
||||
use crate::pb::volume_server_pb::{
|
||||
CopyFileRequest, VolumeEcShardReadRequest, volume_server_client::VolumeServerClient,
|
||||
CopyFileRequest, VolumeEcBlobDeleteRequest, VolumeEcShardReadRequest,
|
||||
};
|
||||
use crate::server::grpc_client::{
|
||||
GrpcDialOptions, connect_channel, master_client, parse_grpc_address, volume_server_client,
|
||||
};
|
||||
use crate::server::grpc_client::{GRPC_MAX_MESSAGE_SIZE, build_grpc_endpoint, parse_grpc_address};
|
||||
use crate::server::request_id::outgoing_request_id_interceptor;
|
||||
use crate::server::volume_server::{VolumeServerState, to_http_address};
|
||||
use crate::storage::erasure_coding::ec_shard::{ShardId, shard_id_try_from};
|
||||
use crate::storage::needle::needle::{Needle, NeedleError, get_actual_size};
|
||||
@@ -254,6 +255,237 @@ pub async fn read_ec_shard_needle_distributed(
|
||||
Ok(Some(n))
|
||||
}
|
||||
|
||||
/// What one EC delete RPC carries — `VolumeEcBlobDeleteRequest` minus tonic.
|
||||
struct EcDeleteTarget<'a> {
|
||||
vid: VolumeId,
|
||||
collection: &'a str,
|
||||
version: Version,
|
||||
needle_id: NeedleId,
|
||||
}
|
||||
|
||||
/// `Store.doDeleteNeedleFromAtLeastOneRemoteEcShards` in Go: journal the
|
||||
/// tombstone on one holder of the needle's primary data shard, falling back
|
||||
/// to any other shard holder when the primary has none. Exactly one node
|
||||
/// journals — replicas of a shard hold identical .ecx copies, so journaling
|
||||
/// on more than one would double the reported delete count.
|
||||
///
|
||||
/// `NotFound` means the volume or needle is gone; other errors mean every
|
||||
/// reachable holder failed or no shard has a holder at all.
|
||||
pub async fn delete_ec_shard_needle_distributed(
|
||||
state: &Arc<VolumeServerState>,
|
||||
vid: VolumeId,
|
||||
needle_id: NeedleId,
|
||||
) -> io::Result<()> {
|
||||
let (
|
||||
primary_shard_id,
|
||||
collection,
|
||||
version,
|
||||
total_shards,
|
||||
local_shards,
|
||||
data_shards,
|
||||
encode_ts_ns,
|
||||
refreshed_at,
|
||||
cached_locations,
|
||||
) = {
|
||||
let store = state.store.read().unwrap();
|
||||
let ecv = store.find_ec_volume(vid).ok_or_else(|| {
|
||||
io::Error::new(
|
||||
io::ErrorKind::NotFound,
|
||||
format!("ec volume {} not mounted", vid.0),
|
||||
)
|
||||
})?;
|
||||
let (_, _, intervals) = ecv.locate_needle(needle_id)?.ok_or_else(|| {
|
||||
io::Error::new(
|
||||
io::ErrorKind::NotFound,
|
||||
format!("needle {} not in ec volume {}", needle_id, vid.0),
|
||||
)
|
||||
})?;
|
||||
let (shard_id, _) = intervals
|
||||
.first()
|
||||
.map(|i| ecv.interval_to_shard_id_and_offset(i))
|
||||
.ok_or_else(|| io::Error::new(io::ErrorKind::NotFound, "no intervals for needle"))?;
|
||||
let (cached_locations, refreshed_at) = ecv.shard_locations_snapshot();
|
||||
(
|
||||
shard_id,
|
||||
ecv.collection.clone(),
|
||||
ecv.version,
|
||||
ecv.data_shards + ecv.parity_shards,
|
||||
local_shard_ids(ecv),
|
||||
ecv.data_shards as usize,
|
||||
ecv.encode_ts_ns,
|
||||
refreshed_at,
|
||||
cached_locations,
|
||||
)
|
||||
};
|
||||
let target = EcDeleteTarget {
|
||||
vid,
|
||||
collection: &collection,
|
||||
version,
|
||||
needle_id,
|
||||
};
|
||||
|
||||
// Holder addresses come from the same staleness-gated cache as the read
|
||||
// path: a master LookupEcVolume only when due, merged on a complete reply.
|
||||
let mut locations = cached_locations;
|
||||
if claim_shard_locations_refresh(
|
||||
state,
|
||||
vid,
|
||||
&locations,
|
||||
refreshed_at,
|
||||
data_shards,
|
||||
total_shards as usize,
|
||||
) {
|
||||
match cached_lookup_ec_shard_locations(state, vid).await {
|
||||
Ok(fresh) => {
|
||||
match write_back_shard_locations(state, vid, fresh, data_shards, encode_ts_ns) {
|
||||
Some(merged) => locations = merged,
|
||||
None => mark_shard_locations_stale(state, vid),
|
||||
}
|
||||
}
|
||||
Err(_) => mark_shard_locations_stale(state, vid),
|
||||
}
|
||||
}
|
||||
|
||||
match delete_on_ec_shard_holders(state, &locations, &local_shards, primary_shard_id, &target)
|
||||
.await
|
||||
{
|
||||
Ok(true) => return Ok(()),
|
||||
Err(e) => return Err(e),
|
||||
Ok(false) => {}
|
||||
}
|
||||
|
||||
for shard_id in 0..total_shards {
|
||||
let Ok(shard_id) = shard_id_try_from(shard_id) else {
|
||||
continue;
|
||||
};
|
||||
if shard_id == primary_shard_id {
|
||||
continue;
|
||||
}
|
||||
if let Ok(true) =
|
||||
delete_on_ec_shard_holders(state, &locations, &local_shards, shard_id, &target).await
|
||||
{
|
||||
return Ok(());
|
||||
}
|
||||
}
|
||||
|
||||
Err(io::Error::other(format!(
|
||||
"ec volume {}: no shard holder could journal the delete",
|
||||
vid.0
|
||||
)))
|
||||
}
|
||||
|
||||
/// `doDeleteNeedleFromRemoteEcShardServers` in Go. `Ok(false)` is the
|
||||
/// shard-missing signal — no live holder anywhere — that triggers the
|
||||
/// caller's fallback walk over the remaining shards.
|
||||
async fn delete_on_ec_shard_holders(
|
||||
state: &Arc<VolumeServerState>,
|
||||
locations: &HashMap<ShardId, Vec<String>>,
|
||||
local_shards: &HashSet<ShardId>,
|
||||
shard_id: ShardId,
|
||||
target: &EcDeleteTarget<'_>,
|
||||
) -> io::Result<bool> {
|
||||
let addrs = locations.get(&shard_id);
|
||||
if !local_shards.contains(&shard_id) && addrs.is_none_or(|a| a.is_empty()) {
|
||||
return Ok(false);
|
||||
}
|
||||
|
||||
let mut last_err = None;
|
||||
if local_shards.contains(&shard_id) {
|
||||
match journal_delete_local(state, target.vid, target.needle_id) {
|
||||
Ok(()) => return Ok(true),
|
||||
// Nothing was committed — the volume unmounted or remounted
|
||||
// without the needle — so it is safe to fall back to other
|
||||
// shard holders, unlike an RPC failure which may have landed.
|
||||
Err(e) if e.kind() == io::ErrorKind::NotFound => return Ok(false),
|
||||
Err(e) => last_err = Some(e),
|
||||
}
|
||||
}
|
||||
if let Some(addrs) = addrs {
|
||||
let self_http = to_http_address(&state.self_url);
|
||||
for addr in addrs {
|
||||
// A stale self entry: the loopback RPC would journal on this same
|
||||
// volume, which the local attempt above already covered.
|
||||
if to_http_address(addr).as_ref() == self_http.as_ref() {
|
||||
continue;
|
||||
}
|
||||
match delete_on_remote_ec_shard(state, addr, target).await {
|
||||
Ok(()) => return Ok(true),
|
||||
Err(e) => last_err = Some(e),
|
||||
}
|
||||
}
|
||||
}
|
||||
match last_err {
|
||||
Some(e) => Err(e),
|
||||
None => Ok(false),
|
||||
}
|
||||
}
|
||||
|
||||
/// `doDeleteNeedleFromRemoteEcShard` in Go — one `VolumeEcBlobDelete` RPC.
|
||||
async fn delete_on_remote_ec_shard(
|
||||
state: &Arc<VolumeServerState>,
|
||||
addr: &str,
|
||||
target: &EcDeleteTarget<'_>,
|
||||
) -> io::Result<()> {
|
||||
let grpc_addr =
|
||||
parse_grpc_address(addr).map_err(|e| io::Error::new(io::ErrorKind::InvalidInput, e))?;
|
||||
let channel = connect_channel(
|
||||
&grpc_addr,
|
||||
state.outgoing_grpc_tls.as_ref(),
|
||||
GrpcDialOptions::unary(),
|
||||
)
|
||||
.await
|
||||
.map_err(|e| io::Error::other(format!("connect to {}: {}", addr, e)))?;
|
||||
let mut client = volume_server_client(channel);
|
||||
client
|
||||
.volume_ec_blob_delete(Request::new(VolumeEcBlobDeleteRequest {
|
||||
volume_id: target.vid.0,
|
||||
collection: target.collection.to_string(),
|
||||
file_key: target.needle_id.0,
|
||||
version: target.version.0 as u32,
|
||||
}))
|
||||
.await
|
||||
.map_err(|e| io::Error::other(format!("volume_ec_blob_delete on {}: {}", addr, e)))?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Journals on the local volume — what the `VolumeEcBlobDelete` handler runs
|
||||
/// when this server is the shard holder. An absent needle is an error, not a
|
||||
/// no-op: `journal_delete` would accept it silently, but here it means the
|
||||
/// volume remounted as a different generation mid-delete and the tombstone
|
||||
/// should go to a replica that still has the needle.
|
||||
fn journal_delete_local(
|
||||
state: &Arc<VolumeServerState>,
|
||||
vid: VolumeId,
|
||||
needle_id: NeedleId,
|
||||
) -> io::Result<()> {
|
||||
let mut store = state.store.write().unwrap();
|
||||
let ecv = store.find_ec_volume_mut(vid).ok_or_else(|| {
|
||||
io::Error::new(
|
||||
io::ErrorKind::NotFound,
|
||||
format!("ec volume {} unmounted", vid.0),
|
||||
)
|
||||
})?;
|
||||
match ecv.find_needle_from_ecx(needle_id)? {
|
||||
None => {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::NotFound,
|
||||
format!("needle {} not in local ecx", needle_id),
|
||||
));
|
||||
}
|
||||
Some((_, size)) if size.is_deleted() => return Ok(()),
|
||||
Some(_) => {}
|
||||
}
|
||||
ecv.journal_delete(needle_id)
|
||||
}
|
||||
|
||||
fn local_shard_ids(ecv: &crate::storage::erasure_coding::EcVolume) -> HashSet<ShardId> {
|
||||
ecv.shards
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter_map(|(i, s)| s.as_ref().map(|_| i as ShardId))
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// FULL EC scrub: verify every needle's bytes across local AND remote shards,
|
||||
/// without decoding (so genuine shard faults are reported rather than healed).
|
||||
/// Mirrors Go's `Store.ScrubEcVolume`. Returns (rows walked, broken shards,
|
||||
@@ -324,9 +556,9 @@ pub async fn scrub_ec_volume_distributed(
|
||||
// mounted volume's encode_ts_ns no longer matches, abort like a
|
||||
// mid-scan unmount rather than mixing generations.
|
||||
let encode_ts_ns = ecv.encode_ts_ns;
|
||||
// Bind to locals so the inner RwLock/Mutex guards drop before the block ends.
|
||||
let cached_locations = ecv.shard_locations.read().unwrap().clone();
|
||||
let cache_refreshed_at = *ecv.shard_locations_refresh_time.lock().unwrap();
|
||||
// One read section, so the map and the refresh time it is aged against
|
||||
// describe the same lookup.
|
||||
let (cached_locations, cache_refreshed_at) = ecv.shard_locations_snapshot();
|
||||
let data_shards = ecv.data_shards as usize;
|
||||
let total_shards = (ecv.data_shards + ecv.parity_shards) as usize;
|
||||
(
|
||||
@@ -428,7 +660,7 @@ pub async fn scrub_ec_volume_distributed(
|
||||
);
|
||||
}
|
||||
};
|
||||
ecv.shard_locations.read().unwrap().clone()
|
||||
ecv.shard_locations_snapshot().0
|
||||
};
|
||||
|
||||
// Walk the .ecx (private fd captured under the lock, no lock held) for the
|
||||
@@ -773,8 +1005,7 @@ fn build_snapshot(
|
||||
}
|
||||
let actual = get_actual_size(size, ecv.version);
|
||||
let interval_results = read_local_intervals(ecv, intervals);
|
||||
let cached_locations = ecv.shard_locations.read().unwrap().clone();
|
||||
let cache_refreshed_at = *ecv.shard_locations_refresh_time.lock().unwrap();
|
||||
let (cached_locations, cache_refreshed_at) = ecv.shard_locations_snapshot();
|
||||
|
||||
Ok(Snapshot {
|
||||
data_shards: ecv.data_shards,
|
||||
@@ -826,14 +1057,14 @@ fn needs_refresh(
|
||||
fn mark_shard_locations_stale(state: &Arc<VolumeServerState>, vid: VolumeId) {
|
||||
let store = state.store.read().unwrap();
|
||||
if let Some(ecv) = store.find_ec_volume(vid) {
|
||||
*ecv.shard_locations_stale.lock().unwrap() = true;
|
||||
ecv.mark_shard_locations_stale();
|
||||
}
|
||||
}
|
||||
|
||||
/// Decide whether the cached map is due a master lookup and, when it is, consume
|
||||
/// its stale mark in the same critical section. A mark raised from here on
|
||||
/// belongs to the next refresh: the read that raised it has disproved the map
|
||||
/// this lookup is about to install.
|
||||
/// Decide whether the caller's snapshot is due a master lookup and, when it is,
|
||||
/// consume the cache's stale mark in the same critical section. A mark raised
|
||||
/// from here on belongs to the next refresh: the read that raised it has
|
||||
/// disproved the map this lookup is about to install.
|
||||
fn claim_shard_locations_refresh(
|
||||
state: &Arc<VolumeServerState>,
|
||||
vid: VolumeId,
|
||||
@@ -846,12 +1077,9 @@ fn claim_shard_locations_refresh(
|
||||
let Some(ecv) = store.find_ec_volume(vid) else {
|
||||
return needs_refresh(locations, refreshed_at, false, data_shards, total_shards);
|
||||
};
|
||||
let mut stale = ecv.shard_locations_stale.lock().unwrap();
|
||||
let refresh = needs_refresh(locations, refreshed_at, *stale, data_shards, total_shards);
|
||||
if refresh {
|
||||
*stale = false;
|
||||
}
|
||||
refresh
|
||||
ecv.claim_shard_locations_refresh(|stale| {
|
||||
needs_refresh(locations, refreshed_at, stale, data_shards, total_shards)
|
||||
})
|
||||
}
|
||||
|
||||
async fn cached_lookup_ec_shard_locations(
|
||||
@@ -872,18 +1100,15 @@ async fn cached_lookup_ec_shard_locations(
|
||||
|
||||
let grpc_addr =
|
||||
parse_grpc_address(&master).map_err(|e| io::Error::new(io::ErrorKind::InvalidInput, e))?;
|
||||
let endpoint = build_grpc_endpoint(&grpc_addr, state.outgoing_grpc_tls.as_ref())
|
||||
.map_err(|e| io::Error::other(e.to_string()))?;
|
||||
let channel = endpoint
|
||||
.connect_timeout(Duration::from_secs(5))
|
||||
.timeout(Duration::from_secs(10))
|
||||
.connect()
|
||||
.await
|
||||
.map_err(|e| io::Error::other(format!("master connect: {}", e)))?;
|
||||
let channel = connect_channel(
|
||||
&grpc_addr,
|
||||
state.outgoing_grpc_tls.as_ref(),
|
||||
GrpcDialOptions::unary(),
|
||||
)
|
||||
.await
|
||||
.map_err(|e| io::Error::other(format!("master connect: {}", e)))?;
|
||||
|
||||
let mut client = SeaweedClient::with_interceptor(channel, outgoing_request_id_interceptor)
|
||||
.max_decoding_message_size(GRPC_MAX_MESSAGE_SIZE)
|
||||
.max_encoding_message_size(GRPC_MAX_MESSAGE_SIZE);
|
||||
let mut client = master_client(channel);
|
||||
|
||||
let resp = client
|
||||
.lookup_ec_volume(Request::new(LookupEcVolumeRequest { volume_id: vid.0 }))
|
||||
@@ -1064,14 +1289,13 @@ async fn do_read_remote_ec_shard_interval(
|
||||
} = iv;
|
||||
let grpc_addr =
|
||||
parse_grpc_address(source).map_err(|e| io::Error::new(io::ErrorKind::InvalidInput, e))?;
|
||||
let endpoint = build_grpc_endpoint(&grpc_addr, state.outgoing_grpc_tls.as_ref())
|
||||
.map_err(|e| io::Error::other(e.to_string()))?;
|
||||
let channel = endpoint
|
||||
.connect_timeout(Duration::from_secs(5))
|
||||
.timeout(Duration::from_secs(30))
|
||||
.connect()
|
||||
.await
|
||||
.map_err(|e| io::Error::other(format!("connect to {}: {}", source, e)))?;
|
||||
let channel = connect_channel(
|
||||
&grpc_addr,
|
||||
state.outgoing_grpc_tls.as_ref(),
|
||||
GrpcDialOptions::long(),
|
||||
)
|
||||
.await
|
||||
.map_err(|e| io::Error::other(format!("connect to {}: {}", source, e)))?;
|
||||
|
||||
// TODO(grpc-jwt): clusters with `jwt.signing.key` configured will
|
||||
// reject peer-to-peer VolumeEcShardRead calls until the Rust
|
||||
@@ -1081,9 +1305,7 @@ async fn do_read_remote_ec_shard_interval(
|
||||
// here in isolation would split the credential plumbing across
|
||||
// call sites. Re-visit when outgoing JWT signing lands as a
|
||||
// server-wide helper.
|
||||
let mut client = VolumeServerClient::with_interceptor(channel, outgoing_request_id_interceptor)
|
||||
.max_decoding_message_size(GRPC_MAX_MESSAGE_SIZE)
|
||||
.max_encoding_message_size(GRPC_MAX_MESSAGE_SIZE);
|
||||
let mut client = volume_server_client(channel);
|
||||
|
||||
let req = VolumeEcShardReadRequest {
|
||||
volume_id: vid.0,
|
||||
@@ -1471,16 +1693,14 @@ async fn fetch_ec_index_from_one_peer(
|
||||
) -> io::Result<()> {
|
||||
let grpc_addr =
|
||||
parse_grpc_address(peer).map_err(|e| io::Error::new(io::ErrorKind::InvalidInput, e))?;
|
||||
let channel = build_grpc_endpoint(&grpc_addr, state.outgoing_grpc_tls.as_ref())
|
||||
.map_err(|e| io::Error::other(e.to_string()))?
|
||||
.connect_timeout(Duration::from_secs(5))
|
||||
.timeout(Duration::from_secs(30))
|
||||
.connect()
|
||||
.await
|
||||
.map_err(|e| io::Error::other(format!("connect {}: {}", peer, e)))?;
|
||||
let mut client = VolumeServerClient::with_interceptor(channel, outgoing_request_id_interceptor)
|
||||
.max_decoding_message_size(GRPC_MAX_MESSAGE_SIZE)
|
||||
.max_encoding_message_size(GRPC_MAX_MESSAGE_SIZE);
|
||||
let channel = connect_channel(
|
||||
&grpc_addr,
|
||||
state.outgoing_grpc_tls.as_ref(),
|
||||
GrpcDialOptions::long(),
|
||||
)
|
||||
.await
|
||||
.map_err(|e| io::Error::other(format!("connect {}: {}", peer, e)))?;
|
||||
let mut client = volume_server_client(channel);
|
||||
|
||||
let copy_req = |ext: &str, ignore_not_found: bool| CopyFileRequest {
|
||||
volume_id: m.vid.0,
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
use std::fmt::Write as _;
|
||||
|
||||
use crate::server::absolute_display_path;
|
||||
use crate::server::server_stats;
|
||||
use crate::server::volume_server::VolumeServerState;
|
||||
use crate::storage::store::Store;
|
||||
@@ -450,16 +451,6 @@ fn collect_ui_data(
|
||||
(disk_rows, volumes, remote_volumes, ec_volumes)
|
||||
}
|
||||
|
||||
fn absolute_display_path(path: &str) -> String {
|
||||
let p = std::path::Path::new(path);
|
||||
if p.is_absolute() {
|
||||
return path.to_string();
|
||||
}
|
||||
std::env::current_dir()
|
||||
.map(|cwd| cwd.join(p).to_string_lossy().to_string())
|
||||
.unwrap_or_else(|_| path.to_string())
|
||||
}
|
||||
|
||||
fn join_i64(values: &[i64]) -> String {
|
||||
values
|
||||
.iter()
|
||||
|
||||
@@ -73,8 +73,6 @@ pub struct VolumeServerState {
|
||||
pub volume_state_notify: tokio::sync::Notify,
|
||||
/// Optional batched write queue for improved throughput under load.
|
||||
pub write_queue: std::sync::OnceLock<WriteQueue>,
|
||||
/// Registry of S3 tier backends for tiered storage operations.
|
||||
pub s3_tier_registry: std::sync::RwLock<crate::remote_storage::s3_tier::S3TierRegistry>,
|
||||
/// Read mode: local, proxy, or redirect for non-local volumes.
|
||||
pub read_mode: ReadMode,
|
||||
/// If true, FetchAndWriteNeedle skips remote S3 endpoint validation,
|
||||
|
||||
@@ -207,9 +207,6 @@ mod tests {
|
||||
pre_stop_seconds: 0,
|
||||
volume_state_notify: tokio::sync::Notify::new(),
|
||||
write_queue: std::sync::OnceLock::new(),
|
||||
s3_tier_registry: std::sync::RwLock::new(
|
||||
crate::remote_storage::s3_tier::S3TierRegistry::new(),
|
||||
),
|
||||
read_mode: crate::config::ReadMode::Local,
|
||||
allow_untrusted_remote_endpoints: false,
|
||||
master_url: String::new(),
|
||||
|
||||
@@ -18,7 +18,7 @@ use crate::storage::erasure_coding::ec_shard::{
|
||||
DATA_SHARDS_COUNT, ERASURE_CODING_LARGE_BLOCK_SIZE, ERASURE_CODING_SMALL_BLOCK_SIZE,
|
||||
EcVolumeShard, ShardId,
|
||||
};
|
||||
use crate::storage::erasure_coding::ec_volume::EcVolume;
|
||||
use crate::storage::erasure_coding::ec_volume::{EcVolume, is_usable_ecx_file};
|
||||
use crate::storage::needle_map::NeedleMapKind;
|
||||
use crate::storage::super_block::SUPER_BLOCK_SIZE;
|
||||
use crate::storage::types::*;
|
||||
@@ -479,6 +479,13 @@ impl DiskLocation {
|
||||
if self.idx_directory != self.directory {
|
||||
remove_bitrot_sidecars(&idx_base)?;
|
||||
}
|
||||
|
||||
// Staged 2PC generations (<base>.ecNN.v<N>, versioned .ecx/.ecj/.vif)
|
||||
// belong to this volume's EC state too; leaving them orphans the files.
|
||||
crate::storage::erasure_coding::ec_shard::remove_ec_generation_files(&base, 0)?;
|
||||
if self.idx_directory != self.directory {
|
||||
crate::storage::erasure_coding::ec_shard::remove_ec_generation_files(&idx_base, 0)?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -598,13 +605,20 @@ impl DiskLocation {
|
||||
&mut self,
|
||||
vid: VolumeId,
|
||||
only_empty: bool,
|
||||
only_garbage: bool,
|
||||
keep_remote_data: bool,
|
||||
) -> Result<(), VolumeError> {
|
||||
// Refuse before removing: a refused destroy must leave it mounted.
|
||||
if let Some(v) = self.volumes.get(&vid)
|
||||
&& v.is_compacting()
|
||||
{
|
||||
return Err(v.compacting_error());
|
||||
}
|
||||
if let Some(mut v) = self.volumes.remove(&vid) {
|
||||
crate::metrics::VOLUME_GAUGE
|
||||
.with_label_values(&[&v.collection, "volume"])
|
||||
.dec();
|
||||
v.destroy(only_empty, keep_remote_data)?;
|
||||
v.destroy(only_empty, only_garbage, keep_remote_data)?;
|
||||
Ok(())
|
||||
} else {
|
||||
Err(VolumeError::NotFound)
|
||||
@@ -625,7 +639,7 @@ impl DiskLocation {
|
||||
crate::metrics::VOLUME_GAUGE
|
||||
.with_label_values(&[&v.collection, "volume"])
|
||||
.dec();
|
||||
if let Err(e) = v.destroy(false, false) {
|
||||
if let Err(e) = v.destroy(false, false, false) {
|
||||
warn!(volume_id = vid.0, error = %e, "delete collection: failed to destroy volume");
|
||||
}
|
||||
}
|
||||
@@ -779,21 +793,17 @@ impl DiskLocation {
|
||||
/// Mirrors `DiskLocation.HasEcxFileOnDisk` in
|
||||
/// `weed/storage/disk_location_ec.go`. Skips entries that are
|
||||
/// directories so a stray dir named `<collection>_<vid>.ecx` doesn't
|
||||
/// register as a present index file.
|
||||
/// register as a present index file. A 0-byte `.ecx` is a corrupt stub
|
||||
/// left by a failed EC distribute copy; it must not steer placement
|
||||
/// toward this disk, so it counts as absent (Go requires `Size() > 0`).
|
||||
pub fn has_ecx_file_on_disk(&self, collection: &str, vid: VolumeId) -> bool {
|
||||
let idx_base = volume_file_name(&self.idx_directory, collection, vid);
|
||||
let idx_path = format!("{}.ecx", idx_base);
|
||||
if let Ok(meta) = fs::metadata(&idx_path)
|
||||
&& !meta.is_dir()
|
||||
{
|
||||
if is_usable_ecx_file(&format!("{}.ecx", idx_base)) {
|
||||
return true;
|
||||
}
|
||||
if self.idx_directory != self.directory {
|
||||
let data_base = volume_file_name(&self.directory, collection, vid);
|
||||
let data_path = format!("{}.ecx", data_base);
|
||||
if let Ok(meta) = fs::metadata(&data_path)
|
||||
&& !meta.is_dir()
|
||||
{
|
||||
if is_usable_ecx_file(&format!("{}.ecx", data_base)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
@@ -805,6 +815,20 @@ impl DiskLocation {
|
||||
self.ec_volumes.remove(&vid)
|
||||
}
|
||||
|
||||
/// Drop the in-memory EC volume for vid and close its descriptors without
|
||||
/// deleting files, so a following unlink frees the inodes instead of
|
||||
/// leaving open fds serving the old bytes. Mirrors Go's unloadEcVolume.
|
||||
pub fn unload_ec_volume(&mut self, vid: VolumeId) {
|
||||
if let Some(mut ec_vol) = self.ec_volumes.remove(&vid) {
|
||||
for _ in 0..ec_vol.shard_count() {
|
||||
crate::metrics::VOLUME_GAUGE
|
||||
.with_label_values(&[&ec_vol.collection, "ec_shards"])
|
||||
.dec();
|
||||
}
|
||||
ec_vol.close();
|
||||
}
|
||||
}
|
||||
|
||||
/// Mount EC shards for a volume on this location.
|
||||
///
|
||||
/// `source_disk_type` is the source volume's disk type carried on the
|
||||
@@ -1133,15 +1157,20 @@ pub fn get_disk_stats(path: &str) -> (u64, u64) {
|
||||
Ok(p) => p,
|
||||
Err(_) => return (0, 0),
|
||||
};
|
||||
unsafe {
|
||||
let mut stat: libc::statvfs = std::mem::zeroed();
|
||||
if libc::statvfs(c_path.as_ptr(), &mut stat) == 0 {
|
||||
let all = stat.f_blocks as u64 * stat.f_frsize as u64;
|
||||
let free = stat.f_bavail as u64 * stat.f_frsize as u64;
|
||||
return (all, free);
|
||||
}
|
||||
// SAFETY: `libc::statvfs` is plain data — integers and reserved
|
||||
// padding, no pointers and no restricted niches — so the all-zero
|
||||
// value is a valid one for the call to overwrite.
|
||||
let mut stat: libc::statvfs = unsafe { std::mem::zeroed() };
|
||||
// SAFETY: `c_path` is a live NUL-terminated `CString` that outlives
|
||||
// the call, and `&mut stat` is a live, aligned, exclusive pointer the
|
||||
// kernel only writes through; the fields are read below only after
|
||||
// the call reports success.
|
||||
if unsafe { libc::statvfs(c_path.as_ptr(), &mut stat) } != 0 {
|
||||
return (0, 0);
|
||||
}
|
||||
(0, 0)
|
||||
let all = stat.f_blocks as u64 * stat.f_frsize as u64;
|
||||
let free = stat.f_bavail as u64 * stat.f_frsize as u64;
|
||||
(all, free)
|
||||
}
|
||||
#[cfg(windows)]
|
||||
{
|
||||
@@ -1732,7 +1761,7 @@ mod tests {
|
||||
.unwrap();
|
||||
assert_eq!(loc.volumes_len(), 2);
|
||||
|
||||
loc.delete_volume(VolumeId(1), false, false).unwrap();
|
||||
loc.delete_volume(VolumeId(1), false, false, false).unwrap();
|
||||
assert_eq!(loc.volumes_len(), 1);
|
||||
assert!(loc.find_volume(VolumeId(1)).is_none());
|
||||
}
|
||||
@@ -1785,6 +1814,34 @@ mod tests {
|
||||
assert!(loc.find_volume(VolumeId(3)).is_some());
|
||||
}
|
||||
|
||||
/// A 0-byte `.ecx` is the stub a failed EC distribute copy leaves behind.
|
||||
/// Go's HasEcxFileOnDisk requires Size() > 0 so the stub cannot pin
|
||||
/// placement to a disk that has no usable index.
|
||||
#[test]
|
||||
fn test_has_ecx_file_on_disk_ignores_zero_byte_stub() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let data = tmp.path().join("data");
|
||||
let idx = tmp.path().join("idx");
|
||||
fs::create_dir_all(&data).unwrap();
|
||||
fs::create_dir_all(&idx).unwrap();
|
||||
let loc = DiskLocation::new(
|
||||
data.to_str().unwrap(),
|
||||
idx.to_str().unwrap(),
|
||||
10,
|
||||
DiskType::HardDrive,
|
||||
MinFreeSpace::Percent(1.0),
|
||||
Vec::new(),
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
fs::write(idx.join("pics_7.ecx"), b"").unwrap();
|
||||
assert!(!loc.has_ecx_file_on_disk("pics", VolumeId(7)));
|
||||
|
||||
// A real index in the data dir still counts, stub or no stub.
|
||||
fs::write(data.join("pics_7.ecx"), [0u8; 16]).unwrap();
|
||||
assert!(loc.has_ecx_file_on_disk("pics", VolumeId(7)));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_disk_location_delete_collection_removes_ec_volumes() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
@@ -1801,6 +1858,8 @@ mod tests {
|
||||
|
||||
let shard_path = format!("{}/pics_7.ec00", dir);
|
||||
std::fs::write(&shard_path, b"ec-shard").unwrap();
|
||||
// An EC volume needs its .ecx to mount.
|
||||
std::fs::write(format!("{}/pics_7.ecx", dir), [0u8; 16]).unwrap();
|
||||
|
||||
loc.mount_ec_shards(VolumeId(7), "pics", &[0], "").unwrap();
|
||||
assert!(loc.has_ec_volume(VolumeId(7)));
|
||||
@@ -1838,6 +1897,7 @@ mod tests {
|
||||
// mount_ec_shards with source_disk_type="ssd" — simulating the
|
||||
// VolumeEcShardsMount RPC path.
|
||||
std::fs::write(format!("{}/pics_7.ec00", dir), b"ec-shard").unwrap();
|
||||
std::fs::write(format!("{}/pics_7.ecx", dir), [0u8; 16]).unwrap();
|
||||
loc.mount_ec_shards(VolumeId(7), "pics", &[0], "ssd")
|
||||
.unwrap();
|
||||
{
|
||||
@@ -1937,6 +1997,7 @@ mod tests {
|
||||
// A collection name unique to this test: the gauge is process-global
|
||||
// and sibling tests running in parallel touch other labels.
|
||||
std::fs::write(format!("{}/dupmount_11.ec00", dir), b"shard bytes").unwrap();
|
||||
std::fs::write(format!("{}/dupmount_11.ecx", dir), [0u8; 16]).unwrap();
|
||||
let gauge = crate::metrics::VOLUME_GAUGE.with_label_values(&["dupmount", "ec_shards"]);
|
||||
let before = gauge.get();
|
||||
|
||||
|
||||
@@ -195,6 +195,104 @@ impl ShardBits {
|
||||
}
|
||||
}
|
||||
|
||||
/// Parses the generation of a 2PC-staged `<base>.v<N>` file: `None` means the
|
||||
/// name is not a generation file of `base`.
|
||||
pub fn ec_file_generation(name: &str, base: &str) -> Option<u32> {
|
||||
let suffix = name.strip_prefix(&format!("{}.v", base))?;
|
||||
match suffix.parse::<u32>() {
|
||||
Ok(g) if g > 0 => Some(g),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
/// Removes 2PC generation files staged under `base`:
|
||||
/// `<base>.ecNN.v<N>`, `<base>.ecx.v<N>`, `<base>.ecj.v<N>`, `<base>.ecsum.v<N>`
|
||||
/// and `<base>.vif.v<N>`. `generations_older_than == 0` removes every
|
||||
/// generation; otherwise only generations strictly below it. Returns the
|
||||
/// first real removal failure. Mirrors Go's `RemoveEcGenerationFiles`.
|
||||
pub fn remove_ec_generation_files(base: &str, generations_older_than: u32) -> io::Result<()> {
|
||||
let path = std::path::Path::new(base);
|
||||
let (Some(parent), Some(fname)) = (path.parent(), path.file_name()) else {
|
||||
return Ok(());
|
||||
};
|
||||
let ec_prefix = format!("{}.ec", fname.to_string_lossy());
|
||||
let vif_name = format!("{}.vif", fname.to_string_lossy());
|
||||
let mut first_err: Option<io::Error> = None;
|
||||
let mut record = |res: io::Result<()>| {
|
||||
if let Err(e) = res
|
||||
&& first_err.is_none()
|
||||
{
|
||||
first_err = Some(e);
|
||||
}
|
||||
};
|
||||
match fs::read_dir(parent) {
|
||||
Ok(entries) => {
|
||||
for entry in entries {
|
||||
let entry = match entry {
|
||||
Ok(entry) => entry,
|
||||
Err(e) => {
|
||||
// A skipped entry means an incomplete sweep; report it
|
||||
// instead of pretending the cleanup finished.
|
||||
record(Err(e));
|
||||
continue;
|
||||
}
|
||||
};
|
||||
let name = entry.file_name().to_string_lossy().into_owned();
|
||||
let Some((artifact, _)) = name.rsplit_once(".v") else {
|
||||
continue;
|
||||
};
|
||||
if artifact != vif_name && !artifact.starts_with(&ec_prefix) {
|
||||
continue;
|
||||
}
|
||||
let Some(generation) = ec_file_generation(&name, artifact) else {
|
||||
continue;
|
||||
};
|
||||
if generations_older_than > 0 && generation >= generations_older_than {
|
||||
continue;
|
||||
}
|
||||
record(match fs::remove_file(entry.path()) {
|
||||
Err(e) if e.kind() != io::ErrorKind::NotFound => Err(e),
|
||||
_ => Ok(()),
|
||||
});
|
||||
}
|
||||
}
|
||||
Err(e) if e.kind() != io::ErrorKind::NotFound => record(Err(e)),
|
||||
Err(_) => {}
|
||||
}
|
||||
match first_err {
|
||||
Some(e) => Err(e),
|
||||
None => Ok(()),
|
||||
}
|
||||
}
|
||||
|
||||
/// Removes every staged generation `<shard_file>.v<N>` of one shard file.
|
||||
/// Returns true when at least one generation file was removed.
|
||||
pub fn remove_ec_shard_generations(shard_file: &str) -> io::Result<bool> {
|
||||
let path = std::path::Path::new(shard_file);
|
||||
let (Some(parent), Some(fname)) = (path.parent(), path.file_name()) else {
|
||||
return Ok(false);
|
||||
};
|
||||
let fname = fname.to_string_lossy().into_owned();
|
||||
let mut removed = false;
|
||||
match fs::read_dir(parent) {
|
||||
Ok(entries) => {
|
||||
for entry in entries {
|
||||
let entry = entry?;
|
||||
let name = entry.file_name().to_string_lossy().into_owned();
|
||||
if ec_file_generation(&name, &fname).is_some() {
|
||||
match fs::remove_file(entry.path()) {
|
||||
Err(e) if e.kind() != io::ErrorKind::NotFound => return Err(e),
|
||||
_ => removed = true,
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
Err(e) if e.kind() != io::ErrorKind::NotFound => return Err(e),
|
||||
Err(_) => {}
|
||||
}
|
||||
Ok(removed)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
@@ -7,19 +7,48 @@ use std::collections::{HashMap, HashSet};
|
||||
use std::fs::{self, File, OpenOptions};
|
||||
use std::io::{self, Write};
|
||||
use std::sync::RwLock;
|
||||
use std::time::{SystemTime, UNIX_EPOCH};
|
||||
use std::time::{Instant, SystemTime, UNIX_EPOCH};
|
||||
|
||||
use crate::pb::master_pb;
|
||||
use crate::storage::erasure_coding::ec_locate;
|
||||
use crate::storage::erasure_coding::ec_shard::*;
|
||||
use crate::storage::io::read_exact_at;
|
||||
use crate::storage::io_error::IoErrorTracker;
|
||||
use crate::storage::needle::needle::{Needle, NeedleError, get_actual_size};
|
||||
use crate::storage::types::*;
|
||||
use crate::storage::volume_open::open_volume_file;
|
||||
|
||||
/// An erasure-coded volume managing its local shards and index.
|
||||
pub const IO_ERROR_TOLERANCE: i32 = 3;
|
||||
/// The shard-location cache: where each shard lives, when that was last learned
|
||||
/// from the master, and whether a read has since disproved it. One struct behind
|
||||
/// one lock, because the freshness heuristic judges all three together (the map
|
||||
/// and its time from the reader's own snapshot, see
|
||||
/// `EcVolume::claim_shard_locations_refresh`) — a map published ahead of its
|
||||
/// refresh time reads as a fresh map stamped with the previous lookup, and a
|
||||
/// half-populated map must never look fresh at all.
|
||||
#[derive(Default)]
|
||||
pub(crate) struct ShardLocationCache {
|
||||
/// Maps shard ID -> list of server addresses where that shard exists.
|
||||
/// Used for distributed EC reads across the cluster.
|
||||
locations: HashMap<ShardId, Vec<String>>,
|
||||
/// Wall-clock timestamp of the most recent successful `LookupEcVolume`
|
||||
/// refresh of `locations`. `None` until the first refresh. Drives the
|
||||
/// staleness heuristic in `cached_lookup_ec_shard_locations` (mirrors Go's
|
||||
/// `ShardLocationsRefreshTime`).
|
||||
refreshed_at: Option<Instant>,
|
||||
/// Marks the map for a prompt re-check: a read that failed against a cached
|
||||
/// location has disproved what the map claims, and the normal freshness
|
||||
/// window is far too long to serve from a map known to be wrong. Mirrors the
|
||||
/// invalidation Go's `forgetShardId` performs. A field under the map's lock
|
||||
/// rather than an atomic so the refresh can consume the mark in one critical
|
||||
/// section, and a mark raised meanwhile survives for the next refresh.
|
||||
stale: bool,
|
||||
}
|
||||
|
||||
/// Bytes read per positional read when seeding `deleted_needles` from `.ecj`.
|
||||
/// A multiple of `NEEDLE_ID_SIZE`; 1 MiB is 131072 entries per syscall.
|
||||
const ECJ_LOAD_CHUNK_BYTES: usize = 1 << 20;
|
||||
|
||||
/// An erasure-coded volume managing its local shards and index.
|
||||
pub struct EcVolume {
|
||||
pub volume_id: VolumeId,
|
||||
pub collection: String,
|
||||
@@ -50,25 +79,11 @@ pub struct EcVolume {
|
||||
pub disk_type: DiskType,
|
||||
/// Directory where .ecx/.ecj were actually found (may differ from dir_idx after fallback).
|
||||
ecx_actual_dir: String,
|
||||
/// Maps shard ID -> list of server addresses where that shard exists.
|
||||
/// Used for distributed EC reads across the cluster. Wrapped in
|
||||
/// `RwLock` so the read path can refresh the map (under master
|
||||
/// lookup) without holding the Store write lock — mirrors Go's
|
||||
/// `ShardLocationsLock sync.RWMutex` in `weed/storage/erasure_coding/ec_volume.go`.
|
||||
pub shard_locations: std::sync::RwLock<HashMap<ShardId, Vec<String>>>,
|
||||
/// Wall-clock timestamp of the most recent successful
|
||||
/// `LookupEcVolume` refresh of `shard_locations`. `None` until the
|
||||
/// first refresh. Drives the staleness heuristic in
|
||||
/// `cached_lookup_ec_shard_locations` (mirrors Go's
|
||||
/// `ShardLocationsRefreshTime`).
|
||||
pub shard_locations_refresh_time: std::sync::Mutex<Option<std::time::Instant>>,
|
||||
/// Marks the map for a prompt re-check: a read that failed against a cached
|
||||
/// location has disproved what the map claims, and the normal freshness
|
||||
/// window is far too long to serve from a map known to be wrong. Mirrors the
|
||||
/// invalidation Go's `forgetShardId` performs. A mutex rather than an atomic
|
||||
/// so the refresh can judge the map and consume the mark in one critical
|
||||
/// section, and a mark raised meanwhile survives for the next refresh.
|
||||
pub shard_locations_stale: std::sync::Mutex<bool>,
|
||||
/// Where each shard lives and how far that is to be trusted, all under one
|
||||
/// `RwLock` so the read path can refresh it (under master lookup) without
|
||||
/// holding the Store write lock — mirrors Go's `ShardLocationsLock
|
||||
/// sync.RWMutex` in `weed/storage/erasure_coding/ec_volume.go`.
|
||||
shard_locations: RwLock<ShardLocationCache>,
|
||||
/// EC volume expiration time (unix epoch seconds), set during EC encode from TTL.
|
||||
pub expire_at_sec: u64,
|
||||
/// Encode-run identity (unix nanos) loaded from the .vif EcShardConfig. A read
|
||||
@@ -92,9 +107,9 @@ pub struct EcVolume {
|
||||
/// sidecar other than the one those two fields actually hold.
|
||||
pub(crate) bitrot_source_dir: String,
|
||||
|
||||
io_error_count: std::sync::atomic::AtomicI32,
|
||||
io_error_quarantined: std::sync::atomic::AtomicBool,
|
||||
last_io_error: std::sync::Mutex<Option<String>>,
|
||||
/// Consecutive storage-media errors and the quarantine they lead to,
|
||||
/// for EC volume health monitoring.
|
||||
io_errors: IoErrorTracker,
|
||||
}
|
||||
|
||||
/// Locate the `.vif` for a (collection, vid) by preferring the data dir
|
||||
@@ -368,8 +383,26 @@ pub fn validate_block_size(block_size: i64) -> io::Result<()> {
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Size of the `.ecx` at `path`, or `None` when it is absent or a directory.
|
||||
/// Mirrors Go's `statEcxSize`.
|
||||
pub(crate) fn ecx_file_size(path: &str) -> Option<u64> {
|
||||
match std::fs::metadata(path) {
|
||||
Ok(meta) if !meta.is_dir() => Some(meta.len()),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether `path` is an `.ecx` that can steer placement, ownership or a copy:
|
||||
/// a regular file with content. A 0-byte `.ecx` is what a failed EC distribute
|
||||
/// copy leaves behind, and Go treats it as absent at every such decision
|
||||
/// (`HasEcxFileOnDisk`, `findEcxIdxDirForVolume`, `indexEcxOwners`) so the scan
|
||||
/// moves on to a sibling disk that may hold a valid index.
|
||||
pub(crate) fn is_usable_ecx_file(path: &str) -> bool {
|
||||
ecx_file_size(path).is_some_and(|size| size > 0)
|
||||
}
|
||||
|
||||
impl EcVolume {
|
||||
/// Create a new EcVolume. Loads .ecx index and .ecj journal if present.
|
||||
/// Create a new EcVolume. Opens the .ecx index (required) and the .ecj journal.
|
||||
pub fn new(
|
||||
dir: &str,
|
||||
dir_idx: &str,
|
||||
@@ -441,42 +474,73 @@ impl EcVolume {
|
||||
deleted_needles: RwLock::new(HashSet::new()),
|
||||
disk_type: DiskType::default(),
|
||||
ecx_actual_dir: dir_idx.to_string(),
|
||||
shard_locations: std::sync::RwLock::new(HashMap::new()),
|
||||
shard_locations_refresh_time: std::sync::Mutex::new(None),
|
||||
shard_locations_stale: std::sync::Mutex::new(false),
|
||||
shard_locations: RwLock::new(ShardLocationCache::default()),
|
||||
expire_at_sec,
|
||||
encode_ts_ns,
|
||||
bitrot: None,
|
||||
bitrot_status: crate::storage::erasure_coding::ec_bitrot::BitrotStatus::Off,
|
||||
bitrot_source_dir: String::new(),
|
||||
io_error_count: std::sync::atomic::AtomicI32::new(0),
|
||||
io_error_quarantined: std::sync::atomic::AtomicBool::new(false),
|
||||
last_io_error: std::sync::Mutex::new(None),
|
||||
io_errors: IoErrorTracker::default(),
|
||||
};
|
||||
|
||||
// Open .ecx file (sorted index) in read/write mode for in-place deletion marking.
|
||||
// Matches Go which opens ecx for writing via MarkNeedleDeleted.
|
||||
let ecx_path = vol.ecx_file_name();
|
||||
if std::path::Path::new(&ecx_path).exists() {
|
||||
let file = open_volume_file(OpenOptions::new().read(true).write(true), &ecx_path)?;
|
||||
vol.ecx_file_size = file.metadata()?.len() as i64;
|
||||
vol.ecx_file = Some(file);
|
||||
} else if dir_idx != dir {
|
||||
// Fall back to data directory if .ecx was created before -dir.idx was configured
|
||||
let data_base = crate::storage::volume::volume_file_name(dir, collection, volume_id);
|
||||
let fallback_ecx = format!("{}.ecx", data_base);
|
||||
if std::path::Path::new(&fallback_ecx).exists() {
|
||||
tracing::info!(
|
||||
//
|
||||
// Resolve it the way Go's NewEcVolume does: prefer a non-empty copy,
|
||||
// the one co-located with the shard data first, then the caller's
|
||||
// index directory — either the shared -dir.idx dir or a sibling disk
|
||||
// that owns the .ecx when this disk holds only a 0-byte stub left by an
|
||||
// interrupted copy. A 0-byte .ecx is also a legitimate empty index, so
|
||||
// it yields only to a non-empty copy elsewhere, never to a mere
|
||||
// absence. No .ecx at all fails the mount with NotFound (Go wraps
|
||||
// os.ErrNotExist): an EcVolume without an index would advertise shards
|
||||
// that no read can ever serve.
|
||||
let local_ecx = format!(
|
||||
"{}.ecx",
|
||||
crate::storage::volume::volume_file_name(dir, collection, volume_id)
|
||||
);
|
||||
let shared_ecx = vol.ecx_file_name();
|
||||
let local_size = ecx_file_size(&local_ecx);
|
||||
let shared_size = if dir_idx != dir {
|
||||
ecx_file_size(&shared_ecx)
|
||||
} else {
|
||||
None
|
||||
};
|
||||
let use_local = match (local_size, shared_size) {
|
||||
(Some(n), _) if n > 0 => true,
|
||||
(_, Some(n)) if n > 0 => {
|
||||
tracing::debug!(
|
||||
volume_id = volume_id.0,
|
||||
"ecx file not found in idx dir, falling back to data dir"
|
||||
"ecx not local at {}, using {}",
|
||||
local_ecx,
|
||||
shared_ecx
|
||||
);
|
||||
let file =
|
||||
open_volume_file(OpenOptions::new().read(true).write(true), &fallback_ecx)?;
|
||||
vol.ecx_file_size = file.metadata()?.len() as i64;
|
||||
vol.ecx_file = Some(file);
|
||||
vol.ecx_actual_dir = dir.to_string();
|
||||
false
|
||||
}
|
||||
}
|
||||
// Only 0-byte copies exist: an empty index, local first.
|
||||
(Some(_), _) => true,
|
||||
(None, Some(_)) => false,
|
||||
(None, None) => {
|
||||
let tried = if dir_idx != dir {
|
||||
format!("{} (or {})", local_ecx, shared_ecx)
|
||||
} else {
|
||||
local_ecx
|
||||
};
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::NotFound,
|
||||
format!("cannot open ec volume index {}: not found", tried),
|
||||
));
|
||||
}
|
||||
};
|
||||
let ecx_path = if use_local {
|
||||
vol.ecx_actual_dir = dir.to_string();
|
||||
local_ecx
|
||||
} else {
|
||||
shared_ecx
|
||||
};
|
||||
let file = open_volume_file(OpenOptions::new().read(true).write(true), &ecx_path)?;
|
||||
vol.ecx_file_size = file.metadata()?.len() as i64;
|
||||
vol.ecx_file = Some(file);
|
||||
|
||||
// Open .ecj file (deletion journal) — use ecx_actual_dir for consistency.
|
||||
// Note: Go does NOT replay .ecj into .ecx at volume load (RebuildEcxFile
|
||||
@@ -487,6 +551,45 @@ impl EcVolume {
|
||||
let ecj_base =
|
||||
crate::storage::volume::volume_file_name(&vol.ecx_actual_dir, collection, volume_id);
|
||||
let ecj_path = format!("{}.ecj", ecj_base);
|
||||
|
||||
// Repair a torn tail BEFORE the append handle exists.
|
||||
//
|
||||
// The file is a flat array of fixed-size records and the journal handle
|
||||
// is in append mode, so every write lands at the physical end. A
|
||||
// trailing partial record therefore knocks every later append out of
|
||||
// alignment: the loader below skips the partial bytes, but the next
|
||||
// mount decodes those bytes together with the leading bytes of a real
|
||||
// entry, yielding one garbage id and silently dropping the delete that
|
||||
// followed the tear. Truncating to a whole number of records costs at
|
||||
// most one incomplete id that was never readable anyway.
|
||||
//
|
||||
// This deliberately uses its own read+write handle rather than the
|
||||
// append handle opened below: on Windows, `append(true)` requests
|
||||
// FILE_APPEND_DATA *without* FILE_WRITE_DATA (and `.write(true)` is
|
||||
// subsumed by `.append(true)`), so SetEndOfFile through that handle
|
||||
// fails with ERROR_ACCESS_DENIED. The handle is scoped so it is closed
|
||||
// again before the append handle opens.
|
||||
{
|
||||
let repair = open_volume_file(
|
||||
OpenOptions::new().read(true).write(true).create(true),
|
||||
&ecj_path,
|
||||
)?;
|
||||
let on_disk = repair.metadata()?.len() as i64;
|
||||
let ragged = on_disk % NEEDLE_ID_SIZE as i64;
|
||||
if ragged != 0 {
|
||||
let whole = on_disk - ragged;
|
||||
tracing::warn!(
|
||||
volume_id = volume_id.0,
|
||||
collection = %collection,
|
||||
on_disk_bytes = on_disk,
|
||||
truncated_to = whole,
|
||||
"truncating torn .ecj tail so later appends stay aligned",
|
||||
);
|
||||
repair.set_len(whole as u64)?;
|
||||
repair.sync_all()?;
|
||||
}
|
||||
}
|
||||
|
||||
let ecj_file = open_volume_file(
|
||||
OpenOptions::new()
|
||||
.read(true)
|
||||
@@ -686,6 +789,14 @@ impl EcVolume {
|
||||
/// from `new()` under exclusive ownership of the just-constructed
|
||||
/// EcVolume, so locking is not strictly required — but we take the
|
||||
/// write lock anyway for symmetry with later mutations.
|
||||
///
|
||||
/// Read in large chunks. This previously issued one `NEEDLE_ID_SIZE`-byte
|
||||
/// positional read per entry, which is fine for a healthy journal (a few
|
||||
/// KB) and catastrophic for a bloated one: at the 1.51 TB seen in
|
||||
/// production that is ~188e9 syscalls, so the server spun at 100% of one
|
||||
/// core with a 31 MB RSS — the set stays small because the ids repeat —
|
||||
/// and never opened its HTTP port, which made the master unregister every
|
||||
/// volume it held. Chunked reads cut that by ~`ECJ_LOAD_CHUNK_BYTES / 8`.
|
||||
fn load_deleted_needles_from_ecj(&mut self) -> io::Result<()> {
|
||||
let ecj_file = match self.ecj_file.as_ref() {
|
||||
Some(f) => f,
|
||||
@@ -694,35 +805,51 @@ impl EcVolume {
|
||||
if self.ecj_file_size < NEEDLE_ID_SIZE as i64 {
|
||||
return Ok(());
|
||||
}
|
||||
let mut buf = [0u8; NEEDLE_ID_SIZE];
|
||||
let mut set = self
|
||||
.deleted_needles
|
||||
.write()
|
||||
.map_err(|_| io::Error::other("deleted_needles lock poisoned"))?;
|
||||
let mut off: i64 = 0;
|
||||
while off + NEEDLE_ID_SIZE as i64 <= self.ecj_file_size {
|
||||
|
||||
// Build into a local set and merge at the end. The previous version
|
||||
// held the `deleted_needles` write lock for the whole scan, which on a
|
||||
// bloated journal is the entire (unbounded) startup.
|
||||
let mut loaded: HashSet<NeedleId> = HashSet::new();
|
||||
let mut buf = vec![0u8; ECJ_LOAD_CHUNK_BYTES];
|
||||
let end = self.ecj_file_size as u64;
|
||||
let mut off: u64 = 0;
|
||||
while off + NEEDLE_ID_SIZE as u64 <= end {
|
||||
// Whole entries only; a trailing partial record is ignored, as the
|
||||
// per-entry loop did by construction.
|
||||
let mut want = std::cmp::min(ECJ_LOAD_CHUNK_BYTES as u64, end - off) as usize;
|
||||
want -= want % NEEDLE_ID_SIZE;
|
||||
if want == 0 {
|
||||
break;
|
||||
}
|
||||
#[cfg(unix)]
|
||||
{
|
||||
use std::os::unix::fs::FileExt;
|
||||
ecj_file.read_exact_at(&mut buf, off as u64)?;
|
||||
ecj_file.read_exact_at(&mut buf[..want], off)?;
|
||||
}
|
||||
#[cfg(windows)]
|
||||
{
|
||||
// Positional read so concurrent readers of the shared .ecj
|
||||
// handle can't interleave seek/read. Mirrors the
|
||||
// read_exact_at helper at the bottom of this file.
|
||||
read_exact_at(ecj_file, &mut buf, off as u64)?;
|
||||
read_exact_at(ecj_file, &mut buf[..want], off)?;
|
||||
}
|
||||
#[cfg(not(any(unix, windows)))]
|
||||
{
|
||||
compile_error!("Platform not supported: only unix and windows are supported");
|
||||
}
|
||||
set.insert(NeedleId::from_bytes(&buf));
|
||||
off += NEEDLE_ID_SIZE as i64;
|
||||
for entry in buf[..want].chunks_exact(NEEDLE_ID_SIZE) {
|
||||
loaded.insert(NeedleId::from_bytes(entry));
|
||||
}
|
||||
off += want as u64;
|
||||
}
|
||||
|
||||
let mut set = self
|
||||
.deleted_needles
|
||||
.write()
|
||||
.map_err(|_| io::Error::other("deleted_needles lock poisoned"))?;
|
||||
set.extend(loaded);
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Returns (file_count, delete_count) for this EC volume. Mirrors Go's
|
||||
/// `EcVolume.FileAndDeleteCount`:
|
||||
///
|
||||
@@ -938,37 +1065,6 @@ impl EcVolume {
|
||||
|
||||
// ---- Shard locations (distributed tracking) ----
|
||||
|
||||
/// Set the list of server addresses for a single shard ID. Does
|
||||
/// NOT touch `shard_locations_refresh_time` — a per-shard write
|
||||
/// from inside a multi-shard population (e.g. iterating the
|
||||
/// `LookupEcVolume` response shard-by-shard) would otherwise
|
||||
/// flip the staleness flag while the map is still incomplete,
|
||||
/// letting a concurrent reader observe `needs_refresh == false`
|
||||
/// against a half-populated cache and return NotFound for the
|
||||
/// not-yet-inserted shards.
|
||||
///
|
||||
/// Callers writing back a whole `LookupEcVolume` reply should use
|
||||
/// [`Self::merge_shard_locations`] instead — it upserts the reply's
|
||||
/// shards under the write lock and advances the refresh timestamp in
|
||||
/// one step, retaining cached shards the reply omits.
|
||||
pub fn set_shard_locations(&self, shard_id: ShardId, locations: Vec<String>) {
|
||||
self.shard_locations
|
||||
.write()
|
||||
.unwrap()
|
||||
.insert(shard_id, locations);
|
||||
}
|
||||
|
||||
/// Atomically replace the entire shard-locations map and stamp
|
||||
/// the refresh time. Used by the distributed-read path's
|
||||
/// post-`LookupEcVolume` write-back so the cache transitions
|
||||
/// from old → fresh in a single observable step — concurrent
|
||||
/// readers either see the full prior map or the full new map,
|
||||
/// never an intermediate state with the freshness flag flipped.
|
||||
pub fn replace_shard_locations(&self, locations: HashMap<ShardId, Vec<String>>) {
|
||||
*self.shard_locations.write().unwrap() = locations;
|
||||
*self.shard_locations_refresh_time.lock().unwrap() = Some(std::time::Instant::now());
|
||||
}
|
||||
|
||||
/// Merge a fresh `LookupEcVolume` reply into the shard-locations cache and
|
||||
/// stamp the refresh time, returning a clone of the resulting map.
|
||||
///
|
||||
@@ -977,73 +1073,76 @@ impl EcVolume {
|
||||
/// from the reply keep their previously-cached locations — so a reply that
|
||||
/// passes the data-shard completeness guard but omits a shard already in cache
|
||||
/// does not drop that shard's known location (unlike a full replace).
|
||||
pub fn merge_shard_locations(
|
||||
///
|
||||
/// Map and refresh time advance in one write section, so no reader can pair
|
||||
/// the merged map with the previous lookup's timestamp.
|
||||
pub(crate) fn merge_shard_locations(
|
||||
&self,
|
||||
locations: HashMap<ShardId, Vec<String>>,
|
||||
) -> HashMap<ShardId, Vec<String>> {
|
||||
let merged = {
|
||||
let mut guard = self.shard_locations.write().unwrap();
|
||||
for (shard_id, addrs) in locations {
|
||||
guard.insert(shard_id, addrs);
|
||||
}
|
||||
guard.clone()
|
||||
};
|
||||
*self.shard_locations_refresh_time.lock().unwrap() = Some(std::time::Instant::now());
|
||||
merged
|
||||
let mut cache = self.shard_locations.write().unwrap();
|
||||
for (shard_id, addrs) in locations {
|
||||
cache.locations.insert(shard_id, addrs);
|
||||
}
|
||||
cache.refreshed_at = Some(Instant::now());
|
||||
cache.locations.clone()
|
||||
}
|
||||
|
||||
/// Get a cloned list of server addresses for a given shard ID.
|
||||
pub fn get_shard_locations(&self, shard_id: ShardId) -> Vec<String> {
|
||||
self.shard_locations
|
||||
.read()
|
||||
.unwrap()
|
||||
.get(&shard_id)
|
||||
.cloned()
|
||||
.unwrap_or_default()
|
||||
/// The cached map and the time it was learned, read in one section. Both
|
||||
/// halves describe the same refresh, which is what the freshness heuristic
|
||||
/// assumes when it ages the map against the timestamp.
|
||||
pub(crate) fn shard_locations_snapshot(
|
||||
&self,
|
||||
) -> (HashMap<ShardId, Vec<String>>, Option<Instant>) {
|
||||
let cache = self.shard_locations.read().unwrap();
|
||||
(cache.locations.clone(), cache.refreshed_at)
|
||||
}
|
||||
|
||||
// ---- I/O error tracking (mirrors Go's EcVolume IoErrorTracker) ----
|
||||
/// Mark the cached map for a prompt re-check after a read failed against one
|
||||
/// of its locations.
|
||||
pub(crate) fn mark_shard_locations_stale(&self) {
|
||||
self.shard_locations.write().unwrap().stale = true;
|
||||
}
|
||||
|
||||
/// Put the stale mark to `decide`, and clear it only if `decide` says the
|
||||
/// master lookup is going ahead — both in one critical section, so the mark
|
||||
/// is consumed exactly once by the refresh that answers for it and a mark
|
||||
/// raised meanwhile survives for the next one. `decide` owns the rest of the
|
||||
/// freshness rule; it judges the map its caller will actually read from,
|
||||
/// which is not necessarily the one cached here by the time it runs.
|
||||
///
|
||||
/// `decide` runs under the cache write lock and must not acquire other locks:
|
||||
/// every caller reaches this while already holding the store read lock, so a
|
||||
/// cache-write → store-read inside `decide` would invert that order.
|
||||
pub(crate) fn claim_shard_locations_refresh(&self, decide: impl FnOnce(bool) -> bool) -> bool {
|
||||
let mut cache = self.shard_locations.write().unwrap();
|
||||
let refresh = decide(cache.stale);
|
||||
if refresh {
|
||||
cache.stale = false;
|
||||
}
|
||||
refresh
|
||||
}
|
||||
|
||||
// ---- I/O error tracking ----
|
||||
|
||||
pub fn check_read_write_error(&self, err: Option<&io::Error>) {
|
||||
use std::sync::atomic::Ordering;
|
||||
if let Some(e) = err
|
||||
&& crate::storage::volume::is_storage_io_error(e)
|
||||
{
|
||||
self.io_error_count.fetch_add(1, Ordering::Relaxed);
|
||||
if let Ok(mut guard) = self.last_io_error.lock() {
|
||||
*guard = Some(e.to_string());
|
||||
}
|
||||
crate::metrics::STORAGE_IO_ERROR_COUNTER.inc();
|
||||
return;
|
||||
}
|
||||
self.io_error_count.store(0, Ordering::Relaxed);
|
||||
if let Ok(mut guard) = self.last_io_error.lock()
|
||||
&& guard.is_some()
|
||||
{
|
||||
*guard = None;
|
||||
}
|
||||
self.io_errors.check_read_write_error(err);
|
||||
}
|
||||
|
||||
pub fn get_io_error_state(&self) -> (Option<String>, i32, bool) {
|
||||
use std::sync::atomic::Ordering;
|
||||
let err = self.last_io_error.lock().ok().and_then(|g| g.clone());
|
||||
let count = self.io_error_count.load(Ordering::Relaxed);
|
||||
let quarantined = self.io_error_quarantined.load(Ordering::Relaxed);
|
||||
(err, count, quarantined)
|
||||
self.io_errors.get_io_error_state()
|
||||
}
|
||||
|
||||
pub fn should_quarantine(&self) -> bool {
|
||||
self.io_errors.should_quarantine()
|
||||
}
|
||||
|
||||
pub fn mark_io_quarantined(&self) {
|
||||
self.io_error_quarantined
|
||||
.store(true, std::sync::atomic::Ordering::Relaxed);
|
||||
self.io_errors.mark_io_quarantined();
|
||||
}
|
||||
|
||||
pub fn reset_io_error_state(&self) {
|
||||
use std::sync::atomic::Ordering;
|
||||
self.io_error_count.store(0, Ordering::Relaxed);
|
||||
self.io_error_quarantined.store(false, Ordering::Relaxed);
|
||||
if let Ok(mut guard) = self.last_io_error.lock() {
|
||||
*guard = None;
|
||||
}
|
||||
self.io_errors.reset_io_error_state();
|
||||
}
|
||||
|
||||
// ---- Index operations ----
|
||||
@@ -1574,9 +1673,21 @@ impl EcVolume {
|
||||
// write_all may have extended the file on disk before
|
||||
// sync_all failed; truncate back to the known-good size so
|
||||
// the on-disk journal never drifts past `deleted_needles`.
|
||||
if let Some(ecj) = self.ecj_file.as_mut()
|
||||
&& let Err(trunc_err) = ecj.set_len(prev_ecj_size as u64)
|
||||
{
|
||||
// Uses its own write handle: on Windows the append handle
|
||||
// lacks FILE_WRITE_DATA, so set_len through it fails with
|
||||
// ERROR_ACCESS_DENIED and the rollback would silently not
|
||||
// happen.
|
||||
let ecj_path = format!(
|
||||
"{}.ecj",
|
||||
crate::storage::volume::volume_file_name(
|
||||
&self.ecx_actual_dir,
|
||||
&self.collection,
|
||||
self.volume_id,
|
||||
)
|
||||
);
|
||||
let rollback = open_volume_file(OpenOptions::new().write(true), &ecj_path)
|
||||
.and_then(|f| f.set_len(prev_ecj_size as u64).and_then(|_| f.sync_all()));
|
||||
if let Err(trunc_err) = rollback {
|
||||
tracing::error!(
|
||||
volume_id = self.volume_id.0,
|
||||
needle_id = needle_id.0,
|
||||
@@ -1806,6 +1917,63 @@ mod tests {
|
||||
use super::*;
|
||||
use tempfile::TempDir;
|
||||
|
||||
/// Go's NewEcVolume fails with os.ErrNotExist when neither directory has an
|
||||
/// `.ecx`. Mounting anyway advertises shards every read then fails on with
|
||||
/// "ecx file not open", and zeroes the size `add_shard`'s 0-byte guard needs.
|
||||
#[test]
|
||||
fn test_new_without_ecx_is_not_found() {
|
||||
let data = TempDir::new().unwrap();
|
||||
let idx = TempDir::new().unwrap();
|
||||
let (dir, dir_idx) = (data.path().to_str().unwrap(), idx.path().to_str().unwrap());
|
||||
std::fs::write(format!("{}/7.ec00", dir), b"shard").unwrap();
|
||||
|
||||
for idx_dir in [dir, dir_idx] {
|
||||
let err = EcVolume::new(dir, idx_dir, "", VolumeId(7))
|
||||
.err()
|
||||
.expect("an EC volume without an .ecx must not mount");
|
||||
assert_eq!(err.kind(), io::ErrorKind::NotFound, "{}", err);
|
||||
}
|
||||
}
|
||||
|
||||
/// A 0-byte `.ecx` stub left by an interrupted copy yields to a non-empty
|
||||
/// copy in the other directory, whichever side the stub is on.
|
||||
#[test]
|
||||
fn test_new_prefers_non_empty_ecx_over_zero_byte_stub() {
|
||||
for stub_in_idx_dir in [true, false] {
|
||||
let data = TempDir::new().unwrap();
|
||||
let idx = TempDir::new().unwrap();
|
||||
let (dir, dir_idx) = (data.path().to_str().unwrap(), idx.path().to_str().unwrap());
|
||||
let (stub_dir, valid_dir) = if stub_in_idx_dir {
|
||||
(dir_idx, dir)
|
||||
} else {
|
||||
(dir, dir_idx)
|
||||
};
|
||||
std::fs::write(format!("{}/7.ecx", stub_dir), b"").unwrap();
|
||||
std::fs::write(
|
||||
format!("{}/7.ecx", valid_dir),
|
||||
vec![0u8; NEEDLE_MAP_ENTRY_SIZE],
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
let vol = EcVolume::new(dir, dir_idx, "", VolumeId(7)).unwrap();
|
||||
assert_eq!(vol.ecx_actual_dir(), valid_dir);
|
||||
assert_eq!(vol.ecx_file_size, NEEDLE_MAP_ENTRY_SIZE as i64);
|
||||
}
|
||||
}
|
||||
|
||||
/// With no other copy a 0-byte `.ecx` is a legitimate empty index (a volume
|
||||
/// whose needles were all deleted before encoding) and still mounts, as in Go.
|
||||
#[test]
|
||||
fn test_new_accepts_lone_zero_byte_ecx_as_empty_index() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
std::fs::write(format!("{}/7.ecx", dir), b"").unwrap();
|
||||
|
||||
let vol = EcVolume::new(dir, dir, "", VolumeId(7)).unwrap();
|
||||
assert_eq!(vol.ecx_file_size, 0);
|
||||
assert!(!is_usable_ecx_file(&vol.ecx_file_name()));
|
||||
}
|
||||
|
||||
/// `destroy()` must remove co-located `.ecsum` sidecars (Go Destroy parity).
|
||||
/// Without this, `collection.delete` leaves orphaned bitrot files that
|
||||
/// inflate EC-health scanners after the shards are gone.
|
||||
@@ -2354,6 +2522,122 @@ mod tests {
|
||||
assert_eq!((fc, dc), (2, 2));
|
||||
}
|
||||
|
||||
/// Write a raw `.ecj` containing `ids` repeated `repeats` times, i.e. the
|
||||
/// shape the append paths produce when a peer's whole journal is
|
||||
/// concatenated onto this one over and over.
|
||||
fn write_bloated_ecj(
|
||||
dir: &str,
|
||||
collection: &str,
|
||||
vid: VolumeId,
|
||||
ids: &[NeedleId],
|
||||
repeats: usize,
|
||||
) {
|
||||
let base = crate::storage::volume::volume_file_name(dir, collection, vid);
|
||||
let mut one = vec![0u8; ids.len() * NEEDLE_ID_SIZE];
|
||||
for (i, id) in ids.iter().enumerate() {
|
||||
id.to_bytes(&mut one[i * NEEDLE_ID_SIZE..(i + 1) * NEEDLE_ID_SIZE]);
|
||||
}
|
||||
let mut f = File::create(format!("{}.ecj", base)).unwrap();
|
||||
for _ in 0..repeats {
|
||||
f.write_all(&one).unwrap();
|
||||
}
|
||||
f.sync_all().unwrap();
|
||||
}
|
||||
/// A journal whose length is not a whole number of entries must not lose
|
||||
/// the entries that ARE complete, and must not panic.
|
||||
#[test]
|
||||
fn test_ecj_with_trailing_partial_entry() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
write_ecx_file(dir, "", VolumeId(40), &[]);
|
||||
|
||||
let base = crate::storage::volume::volume_file_name(dir, "", VolumeId(40));
|
||||
let ids: Vec<NeedleId> = (1..=3).map(NeedleId).collect();
|
||||
let mut bytes = vec![0u8; ids.len() * NEEDLE_ID_SIZE];
|
||||
for (i, id) in ids.iter().enumerate() {
|
||||
id.to_bytes(&mut bytes[i * NEEDLE_ID_SIZE..(i + 1) * NEEDLE_ID_SIZE]);
|
||||
}
|
||||
bytes.extend_from_slice(&[0xAB, 0xCD, 0xEF]); // torn tail
|
||||
std::fs::write(format!("{}.ecj", base), &bytes).unwrap();
|
||||
|
||||
let vol = EcVolume::new(dir, dir, "", VolumeId(40)).unwrap();
|
||||
assert_eq!(vol.read_deleted_needles().unwrap(), ids);
|
||||
}
|
||||
|
||||
/// A torn tail must be truncated at mount, not merely skipped. The handle
|
||||
/// is in append mode, so leaving the partial bytes in place would push
|
||||
/// every later append out of alignment: the delete taken after the tear
|
||||
/// would decode as garbage on the next mount and be silently lost.
|
||||
#[test]
|
||||
fn test_torn_ecj_tail_is_repaired_so_later_deletes_survive() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
// journal_delete only appends on a live->tombstone transition.
|
||||
write_ecx_file(
|
||||
dir,
|
||||
"",
|
||||
VolumeId(42),
|
||||
&[(NeedleId(7), Offset::from_actual_offset(8), Size(10))],
|
||||
);
|
||||
|
||||
let base = crate::storage::volume::volume_file_name(dir, "", VolumeId(42));
|
||||
let ecj_path = format!("{}.ecj", base);
|
||||
let ids: Vec<NeedleId> = (1..=3).map(NeedleId).collect();
|
||||
let mut bytes = vec![0u8; ids.len() * NEEDLE_ID_SIZE];
|
||||
for (i, id) in ids.iter().enumerate() {
|
||||
id.to_bytes(&mut bytes[i * NEEDLE_ID_SIZE..(i + 1) * NEEDLE_ID_SIZE]);
|
||||
}
|
||||
bytes.extend_from_slice(&[0xAB, 0xCD, 0xEF]); // torn tail
|
||||
std::fs::write(&ecj_path, &bytes).unwrap();
|
||||
|
||||
let mut vol = EcVolume::new(dir, dir, "", VolumeId(42)).unwrap();
|
||||
|
||||
// The tear is gone from disk, not just ignored in memory.
|
||||
assert_eq!(
|
||||
std::fs::metadata(&ecj_path).unwrap().len(),
|
||||
(ids.len() * NEEDLE_ID_SIZE) as u64,
|
||||
"torn tail should have been truncated at mount",
|
||||
);
|
||||
|
||||
vol.journal_delete(NeedleId(7)).unwrap();
|
||||
drop(vol);
|
||||
|
||||
// The delete taken after the repair must survive a remount.
|
||||
let vol2 = EcVolume::new(dir, dir, "", VolumeId(42)).unwrap();
|
||||
let deleted = vol2.read_deleted_needles().unwrap();
|
||||
assert!(
|
||||
deleted.contains(&NeedleId(7)),
|
||||
"delete after a torn tail was lost: {:?}",
|
||||
deleted,
|
||||
);
|
||||
assert_eq!(
|
||||
deleted.len(),
|
||||
ids.len() + 1,
|
||||
"misaligned decode: {:?}",
|
||||
deleted
|
||||
);
|
||||
}
|
||||
|
||||
/// A journal spanning several read chunks must load every entry — guards
|
||||
/// the chunk-boundary arithmetic in the buffered loader.
|
||||
#[test]
|
||||
fn test_ecj_spanning_multiple_read_chunks() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
write_ecx_file(dir, "", VolumeId(41), &[]);
|
||||
|
||||
// 2.5 chunks' worth of DISTINCT ids, so nothing is masked by dedup and
|
||||
// the total is not a chunk multiple.
|
||||
let n = (ECJ_LOAD_CHUNK_BYTES / NEEDLE_ID_SIZE) * 5 / 2;
|
||||
let ids: Vec<NeedleId> = (1..=n as u64).map(NeedleId).collect();
|
||||
write_bloated_ecj(dir, "", VolumeId(41), &ids, 1);
|
||||
|
||||
let vol = EcVolume::new(dir, dir, "", VolumeId(41)).unwrap();
|
||||
let deleted = vol.read_deleted_needles().unwrap();
|
||||
assert_eq!(deleted.len(), n, "entries lost across a chunk boundary");
|
||||
assert_eq!(deleted, ids);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_journal_delete_wrong_cookie() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
@@ -3508,6 +3792,93 @@ mod tests {
|
||||
errs
|
||||
);
|
||||
}
|
||||
|
||||
/// Mount a bare EC volume: an empty .ecx is enough to exercise the
|
||||
/// shard-location cache, which is pure in-memory state.
|
||||
fn mount_bare_ec_volume(dir: &str) -> EcVolume {
|
||||
let base = crate::storage::volume::volume_file_name(dir, "", VolumeId(1));
|
||||
std::fs::write(format!("{}.ecx", base), b"").unwrap();
|
||||
EcVolume::new(dir, dir, "", VolumeId(1)).unwrap()
|
||||
}
|
||||
|
||||
/// The map and the refresh time it was learned at must become visible in the
|
||||
/// same step. A merge that published the map first and stamped the time
|
||||
/// afterwards let a reader in between pair a fresh map with the previous
|
||||
/// refresh time -- and the freshness heuristic judges exactly that pair, so
|
||||
/// the reader re-looked-up a map that had just been refreshed.
|
||||
#[test]
|
||||
fn merge_shard_locations_publishes_map_and_time_together() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let vol = mount_bare_ec_volume(tmp.path().to_str().unwrap());
|
||||
|
||||
let (locations, refreshed_at) = vol.shard_locations_snapshot();
|
||||
assert!(locations.is_empty(), "a fresh mount knows no locations");
|
||||
assert!(
|
||||
refreshed_at.is_none(),
|
||||
"a volume that never refreshed has no refresh time"
|
||||
);
|
||||
|
||||
let reply: HashMap<ShardId, Vec<String>> =
|
||||
HashMap::from([(0, vec!["a:1".to_string()]), (1, vec!["b:2".to_string()])]);
|
||||
let merged = vol.merge_shard_locations(reply.clone());
|
||||
assert_eq!(merged, reply, "merge returns the resulting map");
|
||||
|
||||
let (locations, refreshed_at) = vol.shard_locations_snapshot();
|
||||
assert_eq!(locations, reply, "the snapshot reports the merged map");
|
||||
assert!(
|
||||
refreshed_at.is_some(),
|
||||
"the merge that produced that map also stamped its refresh time"
|
||||
);
|
||||
|
||||
// Upsert, not replace: a reply that omits a cached shard keeps it.
|
||||
// `before` is taken outside the merge so the assertion fails if the
|
||||
// second merge did not re-stamp: `later >= refreshed_at` would hold on
|
||||
// the first merge's stamp alone.
|
||||
let before = Instant::now();
|
||||
let merged = vol.merge_shard_locations(HashMap::from([(1, vec!["c:3".to_string()])]));
|
||||
assert_eq!(merged.get(&0), Some(&vec!["a:1".to_string()]));
|
||||
assert_eq!(merged.get(&1), Some(&vec!["c:3".to_string()]));
|
||||
let (locations, later) = vol.shard_locations_snapshot();
|
||||
assert_eq!(locations, merged, "the snapshot reports the merged map");
|
||||
assert!(
|
||||
later.unwrap() >= before,
|
||||
"the second merge re-stamped the refresh time"
|
||||
);
|
||||
}
|
||||
|
||||
/// The stale mark is consumed by the refresh that acts on it, exactly once:
|
||||
/// otherwise every later read keeps re-looking-up a map no one has disproved
|
||||
/// since. A claim that declines the refresh must leave the mark standing.
|
||||
#[test]
|
||||
fn claim_shard_locations_refresh_consumes_the_stale_mark_once() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let vol = mount_bare_ec_volume(tmp.path().to_str().unwrap());
|
||||
|
||||
assert!(
|
||||
!vol.claim_shard_locations_refresh(|stale| stale),
|
||||
"an untouched cache carries no mark"
|
||||
);
|
||||
|
||||
vol.mark_shard_locations_stale();
|
||||
assert!(
|
||||
vol.claim_shard_locations_refresh(|stale| stale),
|
||||
"the mark is visible to the next claim"
|
||||
);
|
||||
assert!(
|
||||
!vol.claim_shard_locations_refresh(|stale| stale),
|
||||
"the claim that acted on the mark consumed it"
|
||||
);
|
||||
|
||||
vol.mark_shard_locations_stale();
|
||||
assert!(
|
||||
!vol.claim_shard_locations_refresh(|_| false),
|
||||
"the refresh decision stays the caller's"
|
||||
);
|
||||
assert!(
|
||||
vol.claim_shard_locations_refresh(|stale| stale),
|
||||
"a claim that did not refresh leaves the mark for the next one"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
|
||||
@@ -21,12 +21,26 @@ where
|
||||
let mut buf = vec![0u8; NEEDLE_MAP_ENTRY_SIZE * ROWS_TO_READ];
|
||||
|
||||
loop {
|
||||
let count = match reader.read(&mut buf) {
|
||||
Ok(0) => return Ok(()),
|
||||
Ok(n) => n,
|
||||
Err(ref e) if e.kind() == io::ErrorKind::UnexpectedEof => return Ok(()),
|
||||
Err(e) => return Err(e),
|
||||
};
|
||||
// Fill the batch before decoding: `read` may return a count that is
|
||||
// not a multiple of the entry size, and a split entry would misalign
|
||||
// every later row. Go is immune: `ReadAt` fills or errors.
|
||||
let mut count = 0;
|
||||
let mut eof = false;
|
||||
while count < buf.len() {
|
||||
match reader.read(&mut buf[count..]) {
|
||||
Ok(0) => {
|
||||
eof = true;
|
||||
break;
|
||||
}
|
||||
Ok(n) => count += n,
|
||||
Err(ref e) if e.kind() == io::ErrorKind::Interrupted => continue,
|
||||
Err(ref e) if e.kind() == io::ErrorKind::UnexpectedEof => {
|
||||
eof = true;
|
||||
break;
|
||||
}
|
||||
Err(e) => return Err(e),
|
||||
}
|
||||
}
|
||||
|
||||
let mut i = 0;
|
||||
while i + NEEDLE_MAP_ENTRY_SIZE <= count {
|
||||
@@ -34,6 +48,11 @@ where
|
||||
f(key, offset, size)?;
|
||||
i += NEEDLE_MAP_ENTRY_SIZE;
|
||||
}
|
||||
|
||||
// A trailing partial entry at EOF is ignored, as Go does on `io.EOF`.
|
||||
if eof {
|
||||
return Ok(());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -177,6 +196,111 @@ mod tests {
|
||||
data
|
||||
}
|
||||
|
||||
/// Reader that hands back at most `chunk` bytes per `read`. 7 is coprime
|
||||
/// with the 17-byte entry size, so nearly every read ends mid-entry. With
|
||||
/// `interrupts`, every other call fails with `ErrorKind::Interrupted`.
|
||||
struct ShortReader {
|
||||
inner: Cursor<Vec<u8>>,
|
||||
chunk: usize,
|
||||
interrupts: bool,
|
||||
interrupt_next: bool,
|
||||
}
|
||||
|
||||
impl ShortReader {
|
||||
fn new(data: Vec<u8>, interrupts: bool) -> Self {
|
||||
ShortReader {
|
||||
inner: Cursor::new(data),
|
||||
chunk: 7,
|
||||
interrupts,
|
||||
interrupt_next: false,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
impl Read for ShortReader {
|
||||
fn read(&mut self, buf: &mut [u8]) -> io::Result<usize> {
|
||||
if self.interrupt_next {
|
||||
self.interrupt_next = false;
|
||||
return Err(io::Error::from(io::ErrorKind::Interrupted));
|
||||
}
|
||||
self.interrupt_next = self.interrupts;
|
||||
let n = buf.len().min(self.chunk);
|
||||
self.inner.read(&mut buf[..n])
|
||||
}
|
||||
}
|
||||
|
||||
impl Seek for ShortReader {
|
||||
fn seek(&mut self, pos: SeekFrom) -> io::Result<u64> {
|
||||
self.inner.seek(pos)
|
||||
}
|
||||
}
|
||||
|
||||
fn walk_all<R: Read + Seek>(reader: &mut R, start_from: u64) -> Vec<(NeedleId, i64, Size)> {
|
||||
let mut collected = Vec::new();
|
||||
walk_index_file(reader, start_from, |key, offset, size| {
|
||||
collected.push((key, offset.to_actual_offset(), size));
|
||||
Ok(())
|
||||
})
|
||||
.unwrap();
|
||||
collected
|
||||
}
|
||||
|
||||
/// More than one ROWS_TO_READ batch, so the walk crosses a buffer refill.
|
||||
fn many_entries() -> Vec<(NeedleId, Offset, Size)> {
|
||||
(0..(ROWS_TO_READ as u64 * 2 + 37))
|
||||
.map(|i| {
|
||||
(
|
||||
NeedleId(i * 7 + 1),
|
||||
Offset::from_actual_offset(i as i64 * 128),
|
||||
Size(i as i32 + 1),
|
||||
)
|
||||
})
|
||||
.collect()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_walk_index_file_short_reads_keep_alignment() {
|
||||
let data = idx_bytes(&many_entries());
|
||||
let expected = walk_all(&mut Cursor::new(data.clone()), 0);
|
||||
assert_eq!(expected.len(), ROWS_TO_READ * 2 + 37);
|
||||
|
||||
let mut short = ShortReader::new(data, false);
|
||||
assert_eq!(walk_all(&mut short, 0), expected);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_walk_index_file_retries_interrupted_reads() {
|
||||
let data = idx_bytes(&many_entries());
|
||||
let expected = walk_all(&mut Cursor::new(data.clone()), 0);
|
||||
|
||||
let mut short = ShortReader::new(data, true);
|
||||
assert_eq!(walk_all(&mut short, 0), expected);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_walk_index_file_short_reads_start_from() {
|
||||
let data = idx_bytes(&many_entries());
|
||||
let expected = walk_all(&mut Cursor::new(data.clone()), 0);
|
||||
|
||||
let start = ROWS_TO_READ as u64 + 5;
|
||||
let mut short = ShortReader::new(data, false);
|
||||
assert_eq!(walk_all(&mut short, start), expected[start as usize..]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_walk_index_file_ignores_trailing_partial_entry() {
|
||||
// A torn final entry is dropped without an error, as Go does on io.EOF.
|
||||
let entries = many_entries();
|
||||
let mut data = idx_bytes(&entries);
|
||||
data.extend_from_slice(&[0xAB; NEEDLE_MAP_ENTRY_SIZE - 1]);
|
||||
|
||||
let expected = walk_all(&mut Cursor::new(idx_bytes(&entries)), 0);
|
||||
assert_eq!(walk_all(&mut Cursor::new(data.clone()), 0), expected);
|
||||
|
||||
let mut short = ShortReader::new(data, false);
|
||||
assert_eq!(walk_all(&mut short, 0), expected);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_check_index_file_clean() {
|
||||
let data = idx_bytes(&[
|
||||
|
||||
@@ -0,0 +1,202 @@
|
||||
//! Consecutive storage-media error tracking shared by `Volume` and
|
||||
//! `EcVolume`. Mirrors Go's `weed/storage/io_error.go`.
|
||||
|
||||
use std::io;
|
||||
use std::sync::Mutex;
|
||||
use std::sync::atomic::{AtomicBool, AtomicI32, Ordering};
|
||||
|
||||
/// Consecutive storage-media errors allowed before the volume is quarantined.
|
||||
pub(crate) const IO_ERROR_TOLERANCE: i32 = 3;
|
||||
|
||||
/// Returns true for I/O errors that indicate faulty storage media, not
|
||||
/// transient/network failures. On Unix this is EIO; on Windows it covers
|
||||
/// ERROR_CRC and ERROR_IO_DEVICE, which the kernel returns for failing disks.
|
||||
pub(crate) fn is_storage_io_error(e: &io::Error) -> bool {
|
||||
#[cfg(unix)]
|
||||
{
|
||||
e.raw_os_error() == Some(libc::EIO)
|
||||
}
|
||||
#[cfg(windows)]
|
||||
{
|
||||
const ERROR_CRC: i32 = 23;
|
||||
const ERROR_IO_DEVICE: i32 = 1117;
|
||||
return e.raw_os_error() == Some(ERROR_CRC) || e.raw_os_error() == Some(ERROR_IO_DEVICE);
|
||||
}
|
||||
#[cfg(not(any(unix, windows)))]
|
||||
{
|
||||
false
|
||||
}
|
||||
}
|
||||
|
||||
/// Consecutive storage-media error state for one volume. `quarantined` is
|
||||
/// sticky: once set it survives later successful I/O and is lifted only by
|
||||
/// `reset_io_error_state`.
|
||||
#[derive(Default)]
|
||||
pub(crate) struct IoErrorTracker {
|
||||
last: Mutex<Option<String>>,
|
||||
count: AtomicI32,
|
||||
quarantined: AtomicBool,
|
||||
}
|
||||
|
||||
impl IoErrorTracker {
|
||||
/// `Some(e)` records a failure, `None` a success. Only storage-media
|
||||
/// failures count; every other outcome clears the count and last error.
|
||||
pub(crate) fn check_read_write_error(&self, err: Option<&io::Error>) {
|
||||
if let Some(e) = err
|
||||
&& is_storage_io_error(e)
|
||||
{
|
||||
self.count.fetch_add(1, Ordering::Relaxed);
|
||||
if let Ok(mut guard) = self.last.lock() {
|
||||
*guard = Some(e.to_string());
|
||||
}
|
||||
crate::metrics::STORAGE_IO_ERROR_COUNTER.inc();
|
||||
return;
|
||||
}
|
||||
self.count.store(0, Ordering::Relaxed);
|
||||
if let Ok(mut guard) = self.last.lock()
|
||||
&& guard.is_some()
|
||||
{
|
||||
*guard = None;
|
||||
}
|
||||
}
|
||||
|
||||
/// The last recorded error, the consecutive count, and the quarantine flag.
|
||||
pub(crate) fn get_io_error_state(&self) -> (Option<String>, i32, bool) {
|
||||
let err = self.last.lock().ok().and_then(|g| g.clone());
|
||||
let count = self.count.load(Ordering::Relaxed);
|
||||
let quarantined = self.quarantined.load(Ordering::Relaxed);
|
||||
(err, count, quarantined)
|
||||
}
|
||||
|
||||
pub(crate) fn should_quarantine(&self) -> bool {
|
||||
self.quarantined.load(Ordering::Relaxed)
|
||||
|| self.count.load(Ordering::Relaxed) >= IO_ERROR_TOLERANCE
|
||||
}
|
||||
|
||||
pub(crate) fn mark_io_quarantined(&self) {
|
||||
self.quarantined.store(true, Ordering::Relaxed);
|
||||
}
|
||||
|
||||
pub(crate) fn reset_io_error_state(&self) {
|
||||
self.count.store(0, Ordering::Relaxed);
|
||||
self.quarantined.store(false, Ordering::Relaxed);
|
||||
if let Ok(mut guard) = self.last.lock() {
|
||||
*guard = None;
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
pub(crate) fn set_last_io_error_for_test(&self, err: Option<&str>) {
|
||||
if let Ok(mut guard) = self.last.lock() {
|
||||
*guard = err.map(|value| value.to_string());
|
||||
}
|
||||
if err.is_some() {
|
||||
self.count.store(IO_ERROR_TOLERANCE, Ordering::Relaxed);
|
||||
} else {
|
||||
self.count.store(0, Ordering::Relaxed);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The tracker only reacts to errors `is_storage_io_error` recognises, which is
|
||||
// nothing at all on a platform that is neither Unix nor Windows.
|
||||
#[cfg(all(test, any(unix, windows)))]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// An OS error the platform reports for failing storage media.
|
||||
#[cfg(unix)]
|
||||
fn media_error() -> io::Error {
|
||||
io::Error::from_raw_os_error(libc::EIO)
|
||||
}
|
||||
|
||||
/// An OS error the platform reports for failing storage media.
|
||||
#[cfg(windows)]
|
||||
fn media_error() -> io::Error {
|
||||
const ERROR_IO_DEVICE: i32 = 1117;
|
||||
io::Error::from_raw_os_error(ERROR_IO_DEVICE)
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn check_read_write_error_counts_consecutive_media_errors() {
|
||||
let tracker = IoErrorTracker::default();
|
||||
tracker.check_read_write_error(Some(&media_error()));
|
||||
tracker.check_read_write_error(Some(&media_error()));
|
||||
|
||||
let (last, count, quarantined) = tracker.get_io_error_state();
|
||||
assert_eq!(last, Some(media_error().to_string()));
|
||||
assert_eq!(count, 2);
|
||||
assert!(!quarantined);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn success_clears_the_count_and_the_last_error() {
|
||||
let tracker = IoErrorTracker::default();
|
||||
tracker.check_read_write_error(Some(&media_error()));
|
||||
tracker.check_read_write_error(None);
|
||||
|
||||
assert_eq!(tracker.get_io_error_state(), (None, 0, false));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn non_media_error_clears_the_count() {
|
||||
let tracker = IoErrorTracker::default();
|
||||
tracker.check_read_write_error(Some(&media_error()));
|
||||
tracker.check_read_write_error(Some(&io::Error::new(
|
||||
io::ErrorKind::NotFound,
|
||||
"no such file",
|
||||
)));
|
||||
|
||||
assert_eq!(tracker.get_io_error_state(), (None, 0, false));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn should_quarantine_only_once_the_tolerance_is_reached() {
|
||||
let tracker = IoErrorTracker::default();
|
||||
for _ in 1..IO_ERROR_TOLERANCE {
|
||||
tracker.check_read_write_error(Some(&media_error()));
|
||||
assert!(!tracker.should_quarantine());
|
||||
}
|
||||
tracker.check_read_write_error(Some(&media_error()));
|
||||
assert!(tracker.should_quarantine());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn quarantine_survives_later_successful_io() {
|
||||
let tracker = IoErrorTracker::default();
|
||||
tracker.mark_io_quarantined();
|
||||
tracker.check_read_write_error(None);
|
||||
|
||||
assert_eq!(tracker.get_io_error_state(), (None, 0, true));
|
||||
assert!(tracker.should_quarantine());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn reset_io_error_state_lifts_the_quarantine() {
|
||||
let tracker = IoErrorTracker::default();
|
||||
tracker.check_read_write_error(Some(&media_error()));
|
||||
tracker.mark_io_quarantined();
|
||||
tracker.reset_io_error_state();
|
||||
|
||||
assert_eq!(tracker.get_io_error_state(), (None, 0, false));
|
||||
assert!(!tracker.should_quarantine());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_helper_arms_a_sustained_error() {
|
||||
let tracker = IoErrorTracker::default();
|
||||
tracker.set_last_io_error_for_test(Some("input/output error"));
|
||||
assert!(tracker.should_quarantine());
|
||||
assert_eq!(
|
||||
tracker.get_io_error_state(),
|
||||
(
|
||||
Some("input/output error".to_string()),
|
||||
IO_ERROR_TOLERANCE,
|
||||
false
|
||||
)
|
||||
);
|
||||
|
||||
tracker.set_last_io_error_for_test(None);
|
||||
assert_eq!(tracker.get_io_error_state(), (None, 0, false));
|
||||
}
|
||||
}
|
||||
@@ -2,6 +2,7 @@ pub mod disk_location;
|
||||
pub mod erasure_coding;
|
||||
pub mod idx;
|
||||
pub(crate) mod io;
|
||||
pub(crate) mod io_error;
|
||||
pub mod needle;
|
||||
pub mod needle_map;
|
||||
pub mod store;
|
||||
|
||||
@@ -749,6 +749,14 @@ pub fn parse_needle_id_cookie(s: &str) -> Result<(NeedleId, Cookie), String> {
|
||||
(s, None)
|
||||
};
|
||||
|
||||
// Every length check and the split below are in BYTES, so a multi-byte
|
||||
// character would let `split` land inside one and panic the slice. Hex is
|
||||
// ASCII by definition; reject anything else up front, as Go's ParseUint
|
||||
// does a step later.
|
||||
if !hex_part.is_ascii() {
|
||||
return Err("KeyHash must be ASCII hex.".to_string());
|
||||
}
|
||||
|
||||
// Go: len(key_hash_string) <= CookieSize*2 => error (must be > 8 hex chars)
|
||||
if hex_part.len() <= COOKIE_SIZE * 2 {
|
||||
return Err("KeyHash is too short.".to_string());
|
||||
@@ -828,6 +836,30 @@ pub enum NeedleError {
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// A fid whose hex part carries multi-byte UTF-8 must be rejected, not
|
||||
/// panic. `split` is a byte offset into `hex_part`; before the ASCII guard
|
||||
/// `&hex_part[..split]` could land inside a character. `GET /3,ééééa` is
|
||||
/// nine bytes, so it passes the length checks and splits at byte 1 —
|
||||
/// halfway through the first `é`. Go's `ParseUint` just errors.
|
||||
#[test]
|
||||
fn parse_needle_id_cookie_rejects_non_ascii_instead_of_panicking() {
|
||||
for s in ["ééééa", "ééééaaaaa", "0123456é9abc", "ééééa_1"] {
|
||||
assert!(
|
||||
parse_needle_id_cookie(s).is_err(),
|
||||
"non-ASCII fid {:?} must be an error",
|
||||
s
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// The ASCII guard must not change any accepted input.
|
||||
#[test]
|
||||
fn parse_needle_id_cookie_still_accepts_ascii_hex() {
|
||||
let (id, cookie) = parse_needle_id_cookie("01637037d6").unwrap();
|
||||
assert_eq!(id, NeedleId(0x01));
|
||||
assert_eq!(cookie, Cookie(0x637037d6));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_parse_header() {
|
||||
let mut buf = [0u8; NEEDLE_HEADER_SIZE];
|
||||
|
||||
@@ -80,6 +80,12 @@ impl TTL {
|
||||
if s.is_empty() {
|
||||
return Ok(TTL::EMPTY);
|
||||
}
|
||||
// The unit is read as the last BYTE and the count as everything before
|
||||
// it, so a trailing multi-byte character would split inside itself and
|
||||
// panic. A TTL is digits plus a one-letter unit; reject the rest.
|
||||
if !s.is_ascii() {
|
||||
return Err(format!("invalid TTL {:?}: must be ASCII", s));
|
||||
}
|
||||
let last_byte = s.as_bytes()[s.len() - 1];
|
||||
let (num_str, unit_byte) = if last_byte.is_ascii_digit() {
|
||||
// All digits — default to minutes (matching Go)
|
||||
@@ -240,6 +246,16 @@ impl fmt::Display for TTL {
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
/// `?ttl=5%C3%A9` must be an error, not a panic. The unit is taken as the
|
||||
/// last *byte*, so a trailing multi-byte character made `&s[..s.len()-1]`
|
||||
/// split inside it.
|
||||
#[test]
|
||||
fn ttl_read_rejects_non_ascii_instead_of_panicking() {
|
||||
for s in ["5é", "é", "3🦀", "12é"] {
|
||||
assert!(TTL::read(s).is_err(), "non-ASCII TTL {:?} must error", s);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_ttl_parse() {
|
||||
let ttl = TTL::read("3m").unwrap();
|
||||
|
||||
@@ -247,6 +247,8 @@ impl CompactNeedleMap {
|
||||
pub fn load_from_idx<R: Read + Seek>(reader: &mut R, version: Version) -> io::Result<Self> {
|
||||
let mut nm = CompactNeedleMap::new();
|
||||
idx::walk_index_file(reader, 0, |key, offset, size| {
|
||||
// A read-only load attaches no writer, so this is its only size.
|
||||
nm.idx_file_offset += NEEDLE_MAP_ENTRY_SIZE as u64;
|
||||
nm.metric.maybe_set_max_needle_end(offset, size, version);
|
||||
if offset.is_zero() || size.is_deleted() {
|
||||
nm.delete_from_map(key);
|
||||
@@ -417,9 +419,9 @@ impl CompactNeedleMap {
|
||||
}
|
||||
|
||||
/// Visit all entries in ascending order by needle ID.
|
||||
pub fn ascending_visit<F>(&self, f: F) -> Result<(), String>
|
||||
pub fn ascending_visit<F, E>(&self, f: F) -> Result<(), E>
|
||||
where
|
||||
F: FnMut(NeedleId, &NeedleValue) -> Result<(), String>,
|
||||
F: FnMut(NeedleId, &NeedleValue) -> Result<(), E>,
|
||||
{
|
||||
self.map.ascending_visit(f)
|
||||
}
|
||||
@@ -1203,9 +1205,10 @@ impl RedbNeedleMap {
|
||||
}
|
||||
|
||||
/// Visit all entries in ascending order by needle ID.
|
||||
pub fn ascending_visit<F>(&self, mut f: F) -> Result<(), String>
|
||||
pub fn ascending_visit<F, E>(&self, mut f: F) -> Result<(), E>
|
||||
where
|
||||
F: FnMut(NeedleId, &NeedleValue) -> Result<(), String>,
|
||||
F: FnMut(NeedleId, &NeedleValue) -> Result<(), E>,
|
||||
E: From<String>,
|
||||
{
|
||||
let txn = self
|
||||
.db_or_err()
|
||||
@@ -1385,6 +1388,18 @@ impl NeedleMap {
|
||||
}
|
||||
}
|
||||
|
||||
/// Skew the live file count away from what the `.idx` holds, so tests
|
||||
/// can build a volume whose reported count disagrees with a reload.
|
||||
#[cfg(test)]
|
||||
pub(crate) fn add_file_count_for_test(&self, delta: i64) {
|
||||
let metric = match self {
|
||||
NeedleMap::InMemory(nm) => &nm.metric,
|
||||
NeedleMap::Redb(nm) => &nm.metric,
|
||||
NeedleMap::SortedFile(_) => panic!("sorted-file needle maps are read-only"),
|
||||
};
|
||||
metric.file_count.fetch_add(delta, Ordering::Relaxed);
|
||||
}
|
||||
|
||||
/// Largest (offset + actual size) seen during the load walk; 0 if the
|
||||
/// map is empty. Used at volume load to detect .idx entries that
|
||||
/// reference past the end of .dat (issue #8928) without a second scan.
|
||||
@@ -1443,9 +1458,10 @@ impl NeedleMap {
|
||||
}
|
||||
|
||||
/// Visit all entries in ascending order by needle ID.
|
||||
pub fn ascending_visit<F>(&self, f: F) -> Result<(), String>
|
||||
pub fn ascending_visit<F, E>(&self, f: F) -> Result<(), E>
|
||||
where
|
||||
F: FnMut(NeedleId, &NeedleValue) -> Result<(), String>,
|
||||
F: FnMut(NeedleId, &NeedleValue) -> Result<(), E>,
|
||||
E: From<String>,
|
||||
{
|
||||
match self {
|
||||
NeedleMap::InMemory(nm) => nm.ascending_visit(f),
|
||||
@@ -1466,7 +1482,7 @@ impl NeedleMap {
|
||||
// The visitor never fails, so neither can this.
|
||||
let _ = nm.ascending_visit(|id, nv| {
|
||||
entries.push((id, *nv));
|
||||
Ok(())
|
||||
Ok::<(), std::convert::Infallible>(())
|
||||
});
|
||||
Ok(entries)
|
||||
}
|
||||
@@ -1838,7 +1854,7 @@ mod tests {
|
||||
let mut live = 0u64;
|
||||
nm.ascending_visit(|_, _| {
|
||||
live += 1;
|
||||
Ok(())
|
||||
Ok::<(), String>(())
|
||||
})
|
||||
.unwrap();
|
||||
assert_eq!(live, N - 1);
|
||||
@@ -2051,7 +2067,7 @@ mod tests {
|
||||
let mut visited = Vec::new();
|
||||
nm.ascending_visit(|id, nv| {
|
||||
visited.push((id, nv.size));
|
||||
Ok(())
|
||||
Ok::<(), String>(())
|
||||
})
|
||||
.unwrap();
|
||||
|
||||
|
||||
@@ -317,9 +317,11 @@ impl SortedFileNeedleMap {
|
||||
Ok(())
|
||||
}
|
||||
|
||||
pub fn ascending_visit<F>(&self, mut f: F) -> Result<(), String>
|
||||
/// Visit all live entries in ascending order by needle ID.
|
||||
pub fn ascending_visit<F, E>(&self, mut f: F) -> Result<(), E>
|
||||
where
|
||||
F: FnMut(NeedleId, &NeedleValue) -> Result<(), String>,
|
||||
F: FnMut(NeedleId, &NeedleValue) -> Result<(), E>,
|
||||
E: From<String>,
|
||||
{
|
||||
let mut visit_error = None;
|
||||
self.visit_live_entries(|id, nv| {
|
||||
@@ -329,7 +331,7 @@ impl SortedFileNeedleMap {
|
||||
}
|
||||
Ok(())
|
||||
})
|
||||
.map_err(|e| visit_error.take().unwrap_or_else(|| e.to_string()))
|
||||
.map_err(|e| visit_error.take().unwrap_or_else(|| E::from(e.to_string())))
|
||||
}
|
||||
|
||||
pub fn iter_entries(&self) -> io::Result<Vec<(NeedleId, NeedleValue)>> {
|
||||
@@ -1035,7 +1037,7 @@ mod tests {
|
||||
let mut visited = Vec::new();
|
||||
m.ascending_visit(|id, _| {
|
||||
visited.push(id);
|
||||
Ok(())
|
||||
Ok::<(), String>(())
|
||||
})
|
||||
.unwrap();
|
||||
assert_eq!(visited, vec![NeedleId(2)]);
|
||||
|
||||
+692
-142
File diff suppressed because it is too large
Load Diff
@@ -131,6 +131,12 @@ impl Store {
|
||||
let Some(base) = name.strip_suffix(".ecx") else {
|
||||
continue;
|
||||
};
|
||||
// A 0-byte .ecx is a corrupt stub from a failed copy, not a
|
||||
// credible owner — skip it so the scan keeps looking for a
|
||||
// real index on a sibling disk (Go's indexEcxOwners).
|
||||
if !ent.metadata().is_ok_and(|m| m.len() > 0) {
|
||||
continue;
|
||||
}
|
||||
let Some((collection, vid)) = parse_collection_volume_id_pub(base) else {
|
||||
continue;
|
||||
};
|
||||
@@ -417,4 +423,33 @@ mod tests {
|
||||
let post = fs::read(dir0.join(format!("{}_{}.ecx", collection, vid))).unwrap();
|
||||
assert_eq!(post, ecx_local, "mirror overwrote dir0's existing .ecx");
|
||||
}
|
||||
|
||||
/// The mirror shares Go's indexEcxOwners, which skips a 0-byte `.ecx`:
|
||||
/// a stub must not be chosen as the source to mirror from.
|
||||
#[test]
|
||||
fn mirror_owner_index_skips_zero_byte_ecx() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir0 = tmp.path().join("data0");
|
||||
let dir1 = tmp.path().join("data1");
|
||||
fs::create_dir_all(&dir0).unwrap();
|
||||
fs::create_dir_all(&dir1).unwrap();
|
||||
|
||||
let collection = "video-recordings";
|
||||
let vid = 4123u32;
|
||||
plant_ecx(&dir0, collection, vid, b"");
|
||||
plant_ecx(&dir1, collection, vid, &[0xA1u8; 20]);
|
||||
|
||||
let mut store = Store::new(NeedleMapKind::InMemory);
|
||||
add_loc(&mut store, &dir0);
|
||||
add_loc(&mut store, &dir1);
|
||||
|
||||
let owners = store.index_ecx_owners_for_mirror();
|
||||
let owner = owners
|
||||
.get(&EcKey {
|
||||
collection: collection.to_string(),
|
||||
vid: VolumeId(vid),
|
||||
})
|
||||
.expect("the valid .ecx on disk 1 must be indexed");
|
||||
assert_eq!(owner.location, 1);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -37,6 +37,7 @@ pub(crate) struct EcVolumeMissingIndex {
|
||||
pub data_dir: String,
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
pub(crate) fn ec_local_ecx_path(dir: &str, collection: &str, vid: VolumeId) -> String {
|
||||
if collection.is_empty() {
|
||||
format!("{}/{}.ecx", dir, vid.0)
|
||||
@@ -117,10 +118,9 @@ impl Store {
|
||||
);
|
||||
continue;
|
||||
};
|
||||
let local_ecx = ec_local_ecx_path(&loc.idx_directory, &key.collection, key.vid);
|
||||
let local_ecx_in_data = ec_local_ecx_path(&loc.directory, &key.collection, key.vid);
|
||||
let use_local_idx = std::path::Path::new(&local_ecx).exists()
|
||||
|| std::path::Path::new(&local_ecx_in_data).exists();
|
||||
// A 0-byte local stub is not a mirrored index (Go gates this fast
|
||||
// path on HasEcxFileOnDisk); mount against the owner instead.
|
||||
let use_local_idx = loc.has_ecx_file_on_disk(&key.collection, key.vid);
|
||||
|
||||
if !use_local_idx && owner.location == loc_idx && owner.idx_dir == loc.idx_directory
|
||||
{
|
||||
@@ -422,6 +422,12 @@ impl Store {
|
||||
let Some(base) = name.strip_suffix(".ecx") else {
|
||||
continue;
|
||||
};
|
||||
// A 0-byte .ecx is a corrupt stub from a failed copy, not a
|
||||
// credible owner — skip it so the scan keeps looking for a
|
||||
// real index on a sibling disk (Go's indexEcxOwners).
|
||||
if !ent.metadata().is_ok_and(|m| m.len() > 0) {
|
||||
continue;
|
||||
}
|
||||
let Some((collection, vid)) = parse_collection_volume_id_pub(base) else {
|
||||
continue;
|
||||
};
|
||||
@@ -639,6 +645,34 @@ mod tests {
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
/// A 0-byte `.ecx` is not a credible owner (Go's indexEcxOwners skips
|
||||
/// it): picking the stub would hide the valid index on the sibling disk.
|
||||
#[test]
|
||||
fn test_index_ecx_owners_skips_zero_byte_stub() {
|
||||
let (store, _tmp) = make_test_store(2, None);
|
||||
let d0 = store.locations[0].directory.clone();
|
||||
let d1 = store.locations[1].directory.clone();
|
||||
std::fs::write(ec_local_ecx_path(&d0, "pics", VolumeId(7)), b"").unwrap();
|
||||
write_index_files(&d1, "pics", 7, 10, 4);
|
||||
|
||||
let owners = store.index_ecx_owners();
|
||||
let owner = owners
|
||||
.get(&EcKey {
|
||||
collection: "pics".to_string(),
|
||||
vid: VolumeId(7),
|
||||
})
|
||||
.expect("the valid .ecx on disk 1 must be indexed");
|
||||
assert_eq!(owner.location, 1);
|
||||
assert_eq!(owner.idx_dir, d1);
|
||||
|
||||
// A stub with no real index anywhere owns nothing.
|
||||
std::fs::write(ec_local_ecx_path(&d0, "pics", VolumeId(8)), b"").unwrap();
|
||||
assert!(!store.index_ecx_owners().contains_key(&EcKey {
|
||||
collection: "pics".to_string(),
|
||||
vid: VolumeId(8),
|
||||
}));
|
||||
}
|
||||
|
||||
/// An empty `.dat` (<= a superblock, i.e. zero needles) for an EC volume
|
||||
/// is a leftover stub from the pre-fix loader. It must be swept on startup,
|
||||
/// not loaded as a phantom empty volume. With the same vid's stub on two
|
||||
@@ -971,6 +1005,56 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
/// dir0 holds orphan shards next to a 0-byte `.ecx` stub from a failed
|
||||
/// copy; the real index is on dir1. The stub must not count as a
|
||||
/// locally-mirrored index (Go gates that fast path on HasEcxFileOnDisk),
|
||||
/// or the shards get registered against an empty index.
|
||||
#[test]
|
||||
fn test_reconcile_ignores_zero_byte_local_ecx_stub() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir0 = tmp.path().join("data0");
|
||||
let dir1 = tmp.path().join("data1");
|
||||
std::fs::create_dir_all(&dir0).unwrap();
|
||||
std::fs::create_dir_all(&dir1).unwrap();
|
||||
|
||||
let collection = "grafana-loki";
|
||||
let vid = 1094u32;
|
||||
|
||||
write_shard(dir0.to_str().unwrap(), collection, vid, 0);
|
||||
write_shard(dir1.to_str().unwrap(), collection, vid, 1);
|
||||
write_index_files(dir1.to_str().unwrap(), collection, vid, 10, 4);
|
||||
|
||||
let mut store = Store::new(NeedleMapKind::InMemory);
|
||||
for dir in [&dir0, &dir1] {
|
||||
store
|
||||
.add_location(
|
||||
dir.to_str().unwrap(),
|
||||
dir.to_str().unwrap(),
|
||||
100,
|
||||
DiskType::HardDrive,
|
||||
MinFreeSpace::Percent(0.0),
|
||||
Vec::new(),
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
// Plant the stub after the startup scan so only the reconcile decision
|
||||
// is under test, then drop dir0's mount and reconcile again.
|
||||
store.locations[0].remove_ec_volume(VolumeId(vid));
|
||||
std::fs::write(
|
||||
ec_local_ecx_path(dir0.to_str().unwrap(), collection, VolumeId(vid)),
|
||||
b"",
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
store.reconcile_ec_shards_across_disks();
|
||||
|
||||
let ev0 = store.locations[0]
|
||||
.find_ec_volume(VolumeId(vid))
|
||||
.expect("dir0's shard must be mounted against the owner's index");
|
||||
assert!(ev0.has_shard(0));
|
||||
assert_eq!(ev0.ecx_actual_dir(), dir1.to_str().unwrap());
|
||||
}
|
||||
|
||||
/// PR 9244 review case: idx_directory is configured but the
|
||||
/// owner's .ecx / .ecj / .vif live in the owner's data dir
|
||||
/// (the legacy "written before -dir.idx was set" layout). The
|
||||
@@ -1392,7 +1476,7 @@ mod tests {
|
||||
let vid = VolumeId(7004);
|
||||
let collection = "grafana-loki";
|
||||
|
||||
store.delete_ec_shards(vid, collection, &[1]);
|
||||
store.delete_ec_shards(vid, collection, &[1]).unwrap();
|
||||
|
||||
// Shard 1 file is gone on disk 1.
|
||||
let p1 = format!(
|
||||
|
||||
@@ -221,6 +221,35 @@ mod tests {
|
||||
use super::*;
|
||||
use crate::storage::types::*;
|
||||
|
||||
/// Multi-byte input must be an error, not a panic: `to_digit` on the
|
||||
/// leading characters rejects it before `chars[2]` is ever indexed.
|
||||
#[test]
|
||||
fn replica_placement_rejects_non_ascii_instead_of_panicking() {
|
||||
for s in ["é", "0é", "é0", "🦀", "ééé"] {
|
||||
assert!(
|
||||
ReplicaPlacement::from_string(s).is_err(),
|
||||
"non-ASCII replication {:?} must error",
|
||||
s
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// The ASCII guard must not change any accepted input, including the
|
||||
/// zero-padding shorthands.
|
||||
#[test]
|
||||
fn replica_placement_still_accepts_ascii_shorthands() {
|
||||
assert_eq!(
|
||||
ReplicaPlacement::from_string("1").unwrap(),
|
||||
ReplicaPlacement::from_string("001").unwrap()
|
||||
);
|
||||
assert_eq!(
|
||||
ReplicaPlacement::from_string("01").unwrap(),
|
||||
ReplicaPlacement::from_string("001").unwrap()
|
||||
);
|
||||
let rp = ReplicaPlacement::from_string("010").unwrap();
|
||||
assert_eq!(rp.diff_rack_count, 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_super_block_round_trip() {
|
||||
let sb = SuperBlock {
|
||||
|
||||
+1547
-659
File diff suppressed because it is too large
Load Diff
@@ -122,7 +122,19 @@ impl Volume {
|
||||
} else {
|
||||
found.remove(&id);
|
||||
}
|
||||
offset += NEEDLE_HEADER_SIZE as i64 + needle_body_length(size, version);
|
||||
let record_size = NEEDLE_HEADER_SIZE as i64 + needle_body_length(size, version);
|
||||
// A corrupt header can make the record length zero or negative;
|
||||
// the scan cannot advance past it.
|
||||
if record_size <= 0 {
|
||||
return Err(VolumeError::Io(io::Error::new(
|
||||
io::ErrorKind::InvalidData,
|
||||
format!(
|
||||
"corrupt needle header at offset {offset}: size {}, record length {record_size}",
|
||||
size.0
|
||||
),
|
||||
)));
|
||||
}
|
||||
offset += record_size;
|
||||
}
|
||||
|
||||
Ok((found, order))
|
||||
@@ -426,4 +438,29 @@ mod tests {
|
||||
"needle 1 should have been recovered"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_scan_dat_head_fails_at_a_header_it_cannot_advance_past() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
write_test_volume(dir, 2);
|
||||
|
||||
let mut dat = OpenOptions::new()
|
||||
.append(true)
|
||||
.open(format!("{}/1.dat", dir))
|
||||
.unwrap();
|
||||
let mut corrupt = [0u8; NEEDLE_HEADER_SIZE];
|
||||
NeedleId(99).to_bytes(&mut corrupt[4..12]);
|
||||
Size(-100).to_bytes(&mut corrupt[12..16]);
|
||||
dat.write_all(&corrupt).unwrap();
|
||||
drop(dat);
|
||||
|
||||
let v = open_volume(dir);
|
||||
let first = v.super_block.block_size() as i64;
|
||||
let err = v.scan_dat_head(v.version(), first, 10).unwrap_err();
|
||||
assert!(
|
||||
matches!(&err, VolumeError::Io(e) if e.kind() == io::ErrorKind::InvalidData),
|
||||
"expected a corrupt-data error, got {err:?}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -112,9 +112,6 @@ fn build_test_state(
|
||||
pre_stop_seconds: 0,
|
||||
volume_state_notify: tokio::sync::Notify::new(),
|
||||
write_queue: std::sync::OnceLock::new(),
|
||||
s3_tier_registry: std::sync::RwLock::new(
|
||||
seaweed_volume::remote_storage::s3_tier::S3TierRegistry::new(),
|
||||
),
|
||||
read_mode: seaweed_volume::config::ReadMode::Local,
|
||||
allow_untrusted_remote_endpoints: false,
|
||||
master_url,
|
||||
@@ -1043,3 +1040,317 @@ async fn chunk_manifest_expands_chunk_stored_on_ec_volume() {
|
||||
assert_eq!(response.status(), StatusCode::OK);
|
||||
assert_eq!(body_bytes(response).await, chunk_data);
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// HTTP DELETE on an EC volume whose shards are not all mounted locally
|
||||
//
|
||||
// The delete handler used to validate the cookie with the local-only
|
||||
// `EcVolume::read_ec_shard_needle`, which errors "ec shard N not available
|
||||
// locally" for any interval held by a peer. Every such error was mapped to 500
|
||||
// and no `.ecj` tombstone was written, so on a standard 10+4 spread across 14
|
||||
// servers no HTTP delete of an EC needle could ever succeed.
|
||||
//
|
||||
// A single node can exercise that path without peers: mount 13 of the 14
|
||||
// shards, leaving out the one holding the needle's interval. The distributed
|
||||
// reader seeds its Reed-Solomon buffers from locally mounted siblings (Phase 0
|
||||
// in `store_ec.rs`), so with >= 10 survivors it reconstructs without any peer
|
||||
// fan-out — while the local-only read still fails outright.
|
||||
// ============================================================================
|
||||
|
||||
#[tokio::test]
|
||||
async fn delete_on_ec_volume_succeeds_when_the_needles_shard_is_not_mounted() {
|
||||
use seaweed_volume::storage::erasure_coding::ec_encoder::write_ec_files;
|
||||
use seaweed_volume::storage::erasure_coding::ec_shard::ShardId;
|
||||
use seaweed_volume::storage::needle::needle::{FileId, Needle};
|
||||
use seaweed_volume::storage::types::{Cookie, NeedleId};
|
||||
use seaweed_volume::storage::volume::{Volume, VolumeSpec};
|
||||
|
||||
let (state, tmp) = test_state();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
|
||||
let data: Vec<u8> = (0..4096u32).map(|i| (i % 251) as u8).collect();
|
||||
let nid = NeedleId(0x5c);
|
||||
let cookie = Cookie(0x0badc0de);
|
||||
|
||||
// Build regular volume 4, then EC-encode it. The volume is standalone and
|
||||
// never registered in the store, so afterwards it exists only as shards.
|
||||
{
|
||||
let mut v = Volume::new(
|
||||
dir,
|
||||
dir,
|
||||
VolumeId(4),
|
||||
NeedleMapKind::InMemory,
|
||||
&VolumeSpec::default(),
|
||||
)
|
||||
.unwrap();
|
||||
let mut n = Needle {
|
||||
id: nid,
|
||||
cookie,
|
||||
data: data.clone(),
|
||||
data_size: data.len() as u32,
|
||||
..Needle::default()
|
||||
};
|
||||
v.write_needle(&mut n, true, false).unwrap();
|
||||
v.sync_to_disk().unwrap();
|
||||
v.close();
|
||||
}
|
||||
write_ec_files(dir, dir, "", VolumeId(4), 10, 4).unwrap();
|
||||
|
||||
// Mount shards 1..=13 only. A needle at .dat offset 0 lives in shard 0's
|
||||
// first small block, so the interval this delete needs is deliberately the
|
||||
// one shard that is absent.
|
||||
{
|
||||
let mut store = state.store.write().unwrap();
|
||||
let shard_ids: Vec<ShardId> = (1..14).collect();
|
||||
store.mount_ec_shards(VolumeId(4), "", &shard_ids).unwrap();
|
||||
}
|
||||
|
||||
let fid = FileId::new(VolumeId(4), nid, cookie).to_string();
|
||||
let app = build_admin_router(state.clone());
|
||||
let response = app
|
||||
.oneshot(
|
||||
Request::builder()
|
||||
.method("DELETE")
|
||||
.uri(format!("/{}", fid))
|
||||
.body(Body::empty())
|
||||
.unwrap(),
|
||||
)
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(
|
||||
response.status(),
|
||||
StatusCode::ACCEPTED,
|
||||
"DELETE of an EC needle must reconstruct through the distributed \
|
||||
reader instead of failing 500 on a non-local shard"
|
||||
);
|
||||
|
||||
// The tombstone must actually have landed: a later GET is a 404.
|
||||
let app = build_admin_router(state.clone());
|
||||
let response = app
|
||||
.oneshot(
|
||||
Request::builder()
|
||||
.uri(format!("/{}", fid))
|
||||
.body(Body::empty())
|
||||
.unwrap(),
|
||||
)
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
response.status(),
|
||||
StatusCode::NOT_FOUND,
|
||||
"the delete must have been journalled, not just answered 202"
|
||||
);
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// Hostile response-header override params must not panic the handler
|
||||
//
|
||||
// The `response-*` query params are attacker-controlled and were inserted with
|
||||
// `parse().unwrap()`. `%0A` decodes to a newline, `HeaderValue::from_str`
|
||||
// rejects it, and the unwrap panicked the connection task — unauthenticated.
|
||||
// The override must simply be skipped.
|
||||
// ============================================================================
|
||||
|
||||
#[tokio::test]
|
||||
async fn hostile_response_header_overrides_are_skipped_not_panicked() {
|
||||
let (state, _tmp) = test_state();
|
||||
let uri = "/1,01637037d6";
|
||||
|
||||
let app = build_admin_router(state.clone());
|
||||
let response = app
|
||||
.oneshot(
|
||||
Request::builder()
|
||||
.method("POST")
|
||||
.uri(uri)
|
||||
.body(Body::from(b"payload".to_vec()))
|
||||
.unwrap(),
|
||||
)
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(response.status(), StatusCode::CREATED);
|
||||
|
||||
// One request per override param, each carrying a raw newline.
|
||||
for param in [
|
||||
"response-cache-control",
|
||||
"response-content-encoding",
|
||||
"response-expires",
|
||||
"response-content-language",
|
||||
"response-content-disposition",
|
||||
"response-content-type",
|
||||
] {
|
||||
let app = build_admin_router(state.clone());
|
||||
let response = app
|
||||
.oneshot(
|
||||
Request::builder()
|
||||
.uri(format!("{}?{}=%0Aevil", uri, param))
|
||||
.body(Body::empty())
|
||||
.unwrap(),
|
||||
)
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
response.status(),
|
||||
StatusCode::OK,
|
||||
"{} with a newline must be ignored, not panic",
|
||||
param
|
||||
);
|
||||
assert_eq!(body_bytes(response).await, b"payload".to_vec());
|
||||
}
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// Non-ASCII in the fid and in ?ttl= must be rejected, not panic
|
||||
//
|
||||
// `parse_needle_id_cookie` split the hex by BYTE offset and `TTL::read` took
|
||||
// the unit as the last BYTE, so a multi-byte character split inside itself.
|
||||
// Both are reachable unauthenticated from the request line / query string.
|
||||
// ============================================================================
|
||||
|
||||
#[tokio::test]
|
||||
async fn non_ascii_fid_and_ttl_are_rejected_not_panicked() {
|
||||
let (state, _tmp) = test_state();
|
||||
|
||||
// A fid whose hex part is multi-byte UTF-8.
|
||||
let app = build_admin_router(state.clone());
|
||||
let response = app
|
||||
.oneshot(
|
||||
Request::builder()
|
||||
.uri("/1,%C3%A9%C3%A9%C3%A9%C3%A9a")
|
||||
.body(Body::empty())
|
||||
.unwrap(),
|
||||
)
|
||||
.await
|
||||
.unwrap();
|
||||
assert!(
|
||||
response.status().is_client_error() || response.status().is_server_error(),
|
||||
"non-ASCII fid must produce an error status, got {}",
|
||||
response.status()
|
||||
);
|
||||
|
||||
// A TTL whose unit character is multi-byte. The upload path does
|
||||
// `TTL::read(..).ok()`, so *any* unparseable TTL is simply dropped and the
|
||||
// write succeeds — the point here is that a non-ASCII one now takes that
|
||||
// same road instead of panicking. Assert it matches an ASCII-invalid TTL
|
||||
// rather than inventing a stricter contract than the handler has.
|
||||
let mut statuses = Vec::new();
|
||||
// Distinct needle ids: reusing one id with a different cookie is a
|
||||
// cookie-mismatch overwrite, which would mask what this test measures.
|
||||
for (fid, ttl) in [("/1,03637037d7", "5%C3%A9"), ("/1,04637037d8", "5z")] {
|
||||
let app = build_admin_router(state.clone());
|
||||
let response = app
|
||||
.oneshot(
|
||||
Request::builder()
|
||||
.method("POST")
|
||||
.uri(format!("{}?ttl={}", fid, ttl))
|
||||
.body(Body::from(b"x".to_vec()))
|
||||
.unwrap(),
|
||||
)
|
||||
.await
|
||||
.unwrap();
|
||||
statuses.push(response.status());
|
||||
}
|
||||
assert_eq!(
|
||||
statuses[0], statuses[1],
|
||||
"a non-ASCII ttl must behave like any other invalid ttl, not panic"
|
||||
);
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// The write queue answers an upload with the needle's real ETag
|
||||
//
|
||||
// The queue worker computes the CRC on the needle it was handed, so the handler
|
||||
// has to know the checksum before it submits. Without that every queued upload
|
||||
// came back as "00000000". The direct path is the reference: same payload, same
|
||||
// ETag, for a plain body and for one the handler gzips before storing.
|
||||
// ============================================================================
|
||||
|
||||
#[tokio::test]
|
||||
async fn write_queue_upload_returns_same_etag_as_direct_write() {
|
||||
use seaweed_volume::server::write_queue::WriteQueue;
|
||||
|
||||
let (direct_state, _direct_tmp) = test_state();
|
||||
let (queued_state, _queued_tmp) = test_state();
|
||||
let wq = WriteQueue::new(queued_state.clone(), 128);
|
||||
let _ = queued_state.write_queue.set(wq);
|
||||
|
||||
let compressible = "seaweedfs ".repeat(200).into_bytes();
|
||||
let uploads: [(&str, &[u8]); 2] = [
|
||||
("/1,01637037d6", b"hello, seaweedfs!"),
|
||||
("/1/02637037d6/notes.txt", &compressible),
|
||||
];
|
||||
|
||||
for (uri, payload) in uploads {
|
||||
let mut etags = Vec::new();
|
||||
for state in [&direct_state, &queued_state] {
|
||||
let app = build_admin_router(state.clone());
|
||||
let response = app
|
||||
.oneshot(
|
||||
Request::builder()
|
||||
.method("POST")
|
||||
.uri(uri)
|
||||
.body(Body::from(payload.to_vec()))
|
||||
.unwrap(),
|
||||
)
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(response.status(), StatusCode::CREATED);
|
||||
|
||||
let header = response
|
||||
.headers()
|
||||
.get("ETag")
|
||||
.expect("upload response has no ETag")
|
||||
.to_str()
|
||||
.unwrap()
|
||||
.to_string();
|
||||
let body = body_bytes(response).await;
|
||||
let json: serde_json::Value =
|
||||
serde_json::from_slice(&body).expect("POST response is not valid JSON");
|
||||
let etag = json["eTag"].as_str().unwrap().to_string();
|
||||
assert_eq!(header, format!("\"{}\"", etag));
|
||||
etags.push(etag);
|
||||
}
|
||||
assert_ne!(etags[0], "00000000", "{}: direct ETag is the zero CRC", uri);
|
||||
assert_eq!(
|
||||
etags[1], etags[0],
|
||||
"{}: queued upload must return the direct path's ETag",
|
||||
uri
|
||||
);
|
||||
}
|
||||
|
||||
// The second upload really was stored gzipped, so its ETag is the CRC of
|
||||
// the compressed bytes on both paths.
|
||||
let app = build_admin_router(queued_state.clone());
|
||||
let response = app
|
||||
.oneshot(
|
||||
Request::builder()
|
||||
.uri(uploads[1].0)
|
||||
.header("Accept-Encoding", "gzip")
|
||||
.body(Body::empty())
|
||||
.unwrap(),
|
||||
)
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(response.status(), StatusCode::OK);
|
||||
assert_eq!(response.headers()["Content-Encoding"], "gzip");
|
||||
|
||||
// Re-uploading the same bytes is the unchanged path: 204 with the same ETag.
|
||||
let (uri, payload) = uploads[0];
|
||||
let mut etags = Vec::new();
|
||||
for state in [&direct_state, &queued_state] {
|
||||
let app = build_admin_router(state.clone());
|
||||
let response = app
|
||||
.oneshot(
|
||||
Request::builder()
|
||||
.method("POST")
|
||||
.uri(uri)
|
||||
.body(Body::from(payload.to_vec()))
|
||||
.unwrap(),
|
||||
)
|
||||
.await
|
||||
.unwrap();
|
||||
assert_eq!(response.status(), StatusCode::NO_CONTENT);
|
||||
etags.push(response.headers()["ETag"].to_str().unwrap().to_string());
|
||||
}
|
||||
assert_eq!(etags[1], etags[0], "unchanged upload ETag differs");
|
||||
}
|
||||
|
||||
Generated
+9
-1
@@ -5329,6 +5329,13 @@ version = "1.2.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "94143f37725109f92c262ed2cf5e59bce7498c01bcc1502d7b9afe439a4e9f49"
|
||||
|
||||
[[package]]
|
||||
name = "seaweed-common"
|
||||
version = "0.1.0"
|
||||
dependencies = [
|
||||
"rustls",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "seaweed-worker-core"
|
||||
version = "0.1.0"
|
||||
@@ -5340,6 +5347,7 @@ dependencies = [
|
||||
"prost",
|
||||
"prost-types",
|
||||
"protoc-bin-vendored",
|
||||
"seaweed-common",
|
||||
"tokio",
|
||||
"tokio-stream",
|
||||
"tonic",
|
||||
@@ -6449,7 +6457,7 @@ dependencies = [
|
||||
"lance-table",
|
||||
"prometheus",
|
||||
"reqwest 0.12.28",
|
||||
"rustls",
|
||||
"seaweed-common",
|
||||
"seaweed-worker-core",
|
||||
"seaweed-worker-sort",
|
||||
"serde",
|
||||
|
||||
@@ -20,6 +20,8 @@ rust-version = "1.94.1"
|
||||
# Protobuf message literals keep `..Default::default()` on purpose: it is
|
||||
# what lets a proto gain a field without touching every constructor.
|
||||
needless_update = "allow"
|
||||
# Every `unsafe` block states its precondition, right above the block.
|
||||
undocumented_unsafe_blocks = "warn"
|
||||
|
||||
[workspace.dependencies]
|
||||
anyhow = "1"
|
||||
|
||||
@@ -9,6 +9,10 @@ one.
|
||||
crates/core the contract: stream, handshake, heartbeat, registry, config forms
|
||||
crates/lance maintenance jobs for Lance tables, and a binary
|
||||
|
||||
It also depends on `../seaweed-common`, a small crate outside this workspace
|
||||
holding the few helpers the Rust volume server needs identically — the
|
||||
HTTP<->gRPC address rule and the rustls provider choice.
|
||||
|
||||
`core` knows nothing about any job. A second worker is a new crate beside
|
||||
`lance` that depends on it, not a fork of the protocol.
|
||||
|
||||
|
||||
@@ -9,6 +9,9 @@ description = "SeaweedFS plugin.proto worker contract"
|
||||
name = "seaweed_worker_core"
|
||||
|
||||
[dependencies]
|
||||
# Helpers the Rust volume server needs as well. A path dependency because
|
||||
# the two trees are separate cargo workspaces with no common root manifest.
|
||||
seaweed-common = { path = "../../../seaweed-common" }
|
||||
anyhow.workspace = true
|
||||
async-trait.workspace = true
|
||||
prost.workspace = true
|
||||
|
||||
@@ -5,32 +5,15 @@
|
||||
//! instead fails as "frame with invalid size", which reads like a protocol bug
|
||||
//! rather than a wrong port, so getting this right is worth its own module.
|
||||
//! Mirrors pb.ServerToGrpcAddress in weed/pb/grpc_client_server.go.
|
||||
|
||||
const GRPC_PORT_OFFSET: u16 = 10000;
|
||||
//!
|
||||
//! The rule itself now lives in `seaweed_common::address`, shared with the Rust
|
||||
//! volume server, which had its own copy of it. What stays here is the `Option`
|
||||
//! shape this crate's callers expect, and the tests that pin it.
|
||||
|
||||
/// Converts `host:port` to the gRPC address, and accepts the explicit
|
||||
/// `host:port.grpcPort` form the Go side also understands.
|
||||
pub fn server_to_grpc_address(server: &str) -> Option<String> {
|
||||
let (host, port_part) = server.rsplit_once(':')?;
|
||||
|
||||
// "port.grpcPort" states the gRPC port outright.
|
||||
if let Some((_, grpc_port)) = port_part.split_once('.')
|
||||
&& let Ok(port) = grpc_port.parse::<u16>()
|
||||
{
|
||||
return Some(join_host_port(host, port));
|
||||
}
|
||||
|
||||
let port: u16 = port_part.parse().ok()?;
|
||||
Some(join_host_port(host, port.checked_add(GRPC_PORT_OFFSET)?))
|
||||
}
|
||||
|
||||
fn join_host_port(host: &str, port: u16) -> String {
|
||||
// An IPv6 literal has to keep its brackets or the port reads as part of it.
|
||||
if host.contains(':') && !host.starts_with('[') {
|
||||
format!("[{host}]:{port}")
|
||||
} else {
|
||||
format!("{host}:{port}")
|
||||
}
|
||||
seaweed_common::address::to_grpc_address(server).ok()
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
@@ -70,4 +53,11 @@ mod tests {
|
||||
assert!(server_to_grpc_address("localhost").is_none());
|
||||
assert!(server_to_grpc_address("localhost:notaport").is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rejects_a_dotted_form_whose_http_port_is_not_a_port() {
|
||||
// Tightened by the move to seaweed-common: this copy used to ignore the
|
||||
// HTTP port of the dotted form and answer Some("host:18080").
|
||||
assert!(server_to_grpc_address("host:abc.18080").is_none());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -16,6 +16,10 @@ name = "weed-worker"
|
||||
path = "src/main.rs"
|
||||
|
||||
[dependencies]
|
||||
# Helpers the Rust volume server needs as well. A path dependency because
|
||||
# the two trees are separate cargo workspaces with no common root manifest.
|
||||
# It also carries the rustls requirement this crate used to state itself.
|
||||
seaweed-common = { path = "../../../seaweed-common" }
|
||||
seaweed-worker-core = { path = "../core" }
|
||||
seaweed-worker-sort = { path = "../sort" }
|
||||
prometheus.workspace = true
|
||||
@@ -36,7 +40,6 @@ anyhow.workspace = true
|
||||
async-trait.workspace = true
|
||||
clap = { version = "4", features = ["derive", "env"] }
|
||||
reqwest = { version = "0.12", default-features = false, features = ["json", "rustls-tls"] }
|
||||
rustls = "0.23"
|
||||
serde = { version = "1", features = ["derive"] }
|
||||
serde_json = "1"
|
||||
tokio.workspace = true
|
||||
|
||||
@@ -1,12 +1,10 @@
|
||||
use rustls::crypto::aws_lc_rs;
|
||||
|
||||
// aws-lc-rs and ring both get linked transitively (lance's aws backend pulls
|
||||
// aws-lc-rs, reqwest's rustls-tls pulls ring), so rustls can't auto-select a
|
||||
// provider and tonic's client TLS panics on first use. Pin the default to
|
||||
// aws-lc-rs, matching the Rust volume server. Idempotent.
|
||||
pub fn install_default_crypto_provider() {
|
||||
let _ = aws_lc_rs::default_provider().install_default();
|
||||
}
|
||||
// aws-lc-rs. The body lives in seaweed-common so this binary and the Rust
|
||||
// volume server cannot end up installing different providers; re-exported here
|
||||
// so callers keep their import path.
|
||||
pub use seaweed_common::tls::install_default_crypto_provider;
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
|
||||
@@ -0,0 +1,291 @@
|
||||
package checksum_test
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"crypto/sha256"
|
||||
"encoding/base64"
|
||||
"errors"
|
||||
"testing"
|
||||
|
||||
"github.com/aws/aws-sdk-go-v2/aws"
|
||||
"github.com/aws/aws-sdk-go-v2/config"
|
||||
"github.com/aws/aws-sdk-go-v2/credentials"
|
||||
"github.com/aws/aws-sdk-go-v2/service/s3"
|
||||
"github.com/aws/aws-sdk-go-v2/service/s3/types"
|
||||
"github.com/aws/smithy-go"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// newWhenRequiredChecksumClient returns a client that only sends checksum
|
||||
// headers when the request asks for them, so tests can upload parts with no
|
||||
// checksum headers at all.
|
||||
func newWhenRequiredChecksumClient(t *testing.T) *s3.Client {
|
||||
t.Helper()
|
||||
|
||||
cfg, err := config.LoadDefaultConfig(context.TODO(),
|
||||
config.WithRegion(defaultConfig.Region),
|
||||
config.WithCredentialsProvider(credentials.NewStaticCredentialsProvider(
|
||||
defaultConfig.AccessKey, defaultConfig.SecretKey, "")),
|
||||
)
|
||||
require.NoError(t, err)
|
||||
return s3.NewFromConfig(cfg, func(o *s3.Options) {
|
||||
o.UsePathStyle = true
|
||||
o.BaseEndpoint = aws.String(defaultConfig.Endpoint)
|
||||
o.RequestChecksumCalculation = aws.RequestChecksumCalculationWhenRequired
|
||||
})
|
||||
}
|
||||
|
||||
// UploadPart without checksum headers inherits the algorithm declared at
|
||||
// CreateMultipartUpload, so the part still gets a checksum and the upload can
|
||||
// be completed with it (AWS behavior).
|
||||
func TestMultipartPartInheritsChecksumAlgorithm(t *testing.T) {
|
||||
client := newWhenRequiredChecksumClient(t)
|
||||
|
||||
bucket := uniqueBucket()
|
||||
createBucket(t, client, bucket)
|
||||
defer cleanupBucket(t, client, bucket)
|
||||
|
||||
body := bytes.Repeat([]byte("y"), 1024)
|
||||
key := "inherit-algo"
|
||||
|
||||
create, err := client.CreateMultipartUpload(context.Background(), &s3.CreateMultipartUploadInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String(key),
|
||||
ChecksumAlgorithm: types.ChecksumAlgorithmSha256,
|
||||
})
|
||||
require.NoError(t, err)
|
||||
|
||||
part, err := client.UploadPart(context.Background(), &s3.UploadPartInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String(key),
|
||||
UploadId: create.UploadId,
|
||||
PartNumber: aws.Int32(1),
|
||||
Body: bytes.NewReader(body),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
require.NotEmpty(t, aws.ToString(part.ChecksumSHA256))
|
||||
|
||||
done, err := client.CompleteMultipartUpload(context.Background(), &s3.CompleteMultipartUploadInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String(key),
|
||||
UploadId: create.UploadId,
|
||||
MultipartUpload: &types.CompletedMultipartUpload{Parts: []types.CompletedPart{{
|
||||
ETag: part.ETag,
|
||||
PartNumber: aws.Int32(1),
|
||||
ChecksumSHA256: part.ChecksumSHA256,
|
||||
}}},
|
||||
})
|
||||
require.NoError(t, err)
|
||||
require.Equal(t, types.ChecksumTypeComposite, done.ChecksumType)
|
||||
require.NotEmpty(t, aws.ToString(done.ChecksumSHA256))
|
||||
}
|
||||
|
||||
// Mimics the .NET repro in https://github.com/seaweedfs/seaweedfs/issues/11401:
|
||||
// UploadPart carries an explicit client-computed x-amz-checksum-sha256 value,
|
||||
// CompleteMultipartUpload echoes it back per part, and HeadObject returns the
|
||||
// object checksum when x-amz-checksum-mode: ENABLED is sent (same as AWS).
|
||||
func TestIssue11401(t *testing.T) {
|
||||
client := getS3Client(t)
|
||||
|
||||
bucket := uniqueBucket()
|
||||
createBucket(t, client, bucket)
|
||||
defer cleanupBucket(t, client, bucket)
|
||||
|
||||
body := bytes.Repeat([]byte("x"), 1024)
|
||||
checksum := base64.StdEncoding.EncodeToString(func() []byte { s := sha256.Sum256(body); return s[:] }())
|
||||
key := "issue-11401"
|
||||
|
||||
create, err := client.CreateMultipartUpload(context.Background(), &s3.CreateMultipartUploadInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String(key),
|
||||
ContentType: aws.String("application/octet-stream"),
|
||||
ChecksumAlgorithm: types.ChecksumAlgorithmSha256,
|
||||
})
|
||||
require.NoError(t, err)
|
||||
|
||||
part, err := client.UploadPart(context.Background(), &s3.UploadPartInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String(key),
|
||||
UploadId: create.UploadId,
|
||||
PartNumber: aws.Int32(1),
|
||||
Body: bytes.NewReader(body),
|
||||
ChecksumSHA256: aws.String(checksum),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
require.Equal(t, checksum, aws.ToString(part.ChecksumSHA256))
|
||||
|
||||
done, err := client.CompleteMultipartUpload(context.Background(), &s3.CompleteMultipartUploadInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String(key),
|
||||
UploadId: create.UploadId,
|
||||
MultipartUpload: &types.CompletedMultipartUpload{Parts: []types.CompletedPart{{
|
||||
ETag: part.ETag,
|
||||
PartNumber: aws.Int32(1),
|
||||
ChecksumSHA256: part.ChecksumSHA256,
|
||||
}}},
|
||||
})
|
||||
require.NoError(t, err)
|
||||
require.Equal(t, types.ChecksumTypeComposite, done.ChecksumType)
|
||||
require.NotEmpty(t, aws.ToString(done.ChecksumSHA256))
|
||||
|
||||
head, err := client.HeadObject(context.Background(), &s3.HeadObjectInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String(key),
|
||||
ChecksumMode: types.ChecksumModeEnabled,
|
||||
})
|
||||
require.NoError(t, err)
|
||||
require.Equal(t, aws.ToString(done.ChecksumSHA256), aws.ToString(head.ChecksumSHA256))
|
||||
|
||||
headNoMode, err := client.HeadObject(context.Background(), &s3.HeadObjectInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String(key),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
require.Empty(t, aws.ToString(headNoMode.ChecksumSHA256))
|
||||
}
|
||||
|
||||
// CompleteMultipartUpload validates the per-part checksums it is given, like
|
||||
// AWS: missing checksums fail with InvalidRequest and wrong ones with
|
||||
// BadDigest.
|
||||
func TestCompleteMultipartUploadValidatesPartChecksums(t *testing.T) {
|
||||
client := newWhenRequiredChecksumClient(t)
|
||||
|
||||
bucket := uniqueBucket()
|
||||
createBucket(t, client, bucket)
|
||||
defer cleanupBucket(t, client, bucket)
|
||||
|
||||
body := bytes.Repeat([]byte("z"), 1024)
|
||||
key := "validate-parts"
|
||||
|
||||
create, err := client.CreateMultipartUpload(context.Background(), &s3.CreateMultipartUploadInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String(key),
|
||||
ChecksumAlgorithm: types.ChecksumAlgorithmSha256,
|
||||
})
|
||||
require.NoError(t, err)
|
||||
|
||||
part, err := client.UploadPart(context.Background(), &s3.UploadPartInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String(key),
|
||||
UploadId: create.UploadId,
|
||||
PartNumber: aws.Int32(1),
|
||||
Body: bytes.NewReader(body),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
require.NotEmpty(t, aws.ToString(part.ChecksumSHA256))
|
||||
|
||||
complete := func(checksum *string) error {
|
||||
_, err := client.CompleteMultipartUpload(context.Background(), &s3.CompleteMultipartUploadInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String(key),
|
||||
UploadId: create.UploadId,
|
||||
MultipartUpload: &types.CompletedMultipartUpload{Parts: []types.CompletedPart{{
|
||||
ETag: part.ETag,
|
||||
PartNumber: aws.Int32(1),
|
||||
ChecksumSHA256: checksum,
|
||||
}}},
|
||||
})
|
||||
return err
|
||||
}
|
||||
|
||||
var apiErr smithy.APIError
|
||||
|
||||
err = complete(nil)
|
||||
require.Error(t, err)
|
||||
require.True(t, errors.As(err, &apiErr))
|
||||
require.Equal(t, "InvalidRequest", apiErr.ErrorCode())
|
||||
|
||||
wrong := base64.StdEncoding.EncodeToString(func() []byte { s := sha256.Sum256([]byte("other")); return s[:] }())
|
||||
err = complete(aws.String(wrong))
|
||||
require.Error(t, err)
|
||||
require.True(t, errors.As(err, &apiErr))
|
||||
require.Equal(t, "BadDigest", apiErr.ErrorCode())
|
||||
|
||||
require.NoError(t, complete(part.ChecksumSHA256))
|
||||
}
|
||||
|
||||
// FULL_OBJECT uploads need no per-part checksums in the complete request; the
|
||||
// whole-object checksum travels in a request header and is validated against
|
||||
// the computed value (BadDigest on mismatch), as on AWS.
|
||||
func TestCompleteMultipartUploadFullObjectChecksum(t *testing.T) {
|
||||
client := newWhenRequiredChecksumClient(t)
|
||||
|
||||
bucket := uniqueBucket()
|
||||
createBucket(t, client, bucket)
|
||||
defer cleanupBucket(t, client, bucket)
|
||||
|
||||
body := bytes.Repeat([]byte("w"), 1024)
|
||||
key := "full-object"
|
||||
|
||||
create, err := client.CreateMultipartUpload(context.Background(), &s3.CreateMultipartUploadInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String(key),
|
||||
ChecksumAlgorithm: types.ChecksumAlgorithmCrc64nvme,
|
||||
})
|
||||
require.NoError(t, err)
|
||||
require.Equal(t, types.ChecksumTypeFullObject, create.ChecksumType)
|
||||
|
||||
part, err := client.UploadPart(context.Background(), &s3.UploadPartInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String(key),
|
||||
UploadId: create.UploadId,
|
||||
PartNumber: aws.Int32(1),
|
||||
Body: bytes.NewReader(body),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
require.NotEmpty(t, aws.ToString(part.ChecksumCRC64NVME))
|
||||
|
||||
complete := func(objectChecksum *string) error {
|
||||
_, err := client.CompleteMultipartUpload(context.Background(), &s3.CompleteMultipartUploadInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String(key),
|
||||
UploadId: create.UploadId,
|
||||
ChecksumCRC64NVME: objectChecksum,
|
||||
MultipartUpload: &types.CompletedMultipartUpload{Parts: []types.CompletedPart{{
|
||||
ETag: part.ETag,
|
||||
PartNumber: aws.Int32(1),
|
||||
}}},
|
||||
})
|
||||
return err
|
||||
}
|
||||
|
||||
var apiErr smithy.APIError
|
||||
wrong := base64.StdEncoding.EncodeToString(func() []byte { s := sha256.Sum256(body); return s[:] }())
|
||||
err = complete(aws.String(wrong))
|
||||
require.Error(t, err)
|
||||
require.True(t, errors.As(err, &apiErr))
|
||||
require.Equal(t, "BadDigest", apiErr.ErrorCode())
|
||||
|
||||
require.NoError(t, complete(part.ChecksumCRC64NVME))
|
||||
}
|
||||
|
||||
// UploadPart with a checksum algorithm that conflicts with the one declared at
|
||||
// CreateMultipartUpload is rejected, as on AWS.
|
||||
func TestMultipartPartConflictingAlgorithm(t *testing.T) {
|
||||
client := newWhenRequiredChecksumClient(t)
|
||||
|
||||
bucket := uniqueBucket()
|
||||
createBucket(t, client, bucket)
|
||||
defer cleanupBucket(t, client, bucket)
|
||||
|
||||
create, err := client.CreateMultipartUpload(context.Background(), &s3.CreateMultipartUploadInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String("conflict"),
|
||||
ChecksumAlgorithm: types.ChecksumAlgorithmSha256,
|
||||
})
|
||||
require.NoError(t, err)
|
||||
|
||||
_, err = client.UploadPart(context.Background(), &s3.UploadPartInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String("conflict"),
|
||||
UploadId: create.UploadId,
|
||||
PartNumber: aws.Int32(1),
|
||||
Body: bytes.NewReader([]byte("data")),
|
||||
ChecksumAlgorithm: types.ChecksumAlgorithmCrc32,
|
||||
})
|
||||
var apiErr smithy.APIError
|
||||
require.Error(t, err)
|
||||
require.True(t, errors.As(err, &apiErr))
|
||||
require.Equal(t, "InvalidRequest", apiErr.ErrorCode())
|
||||
}
|
||||
@@ -0,0 +1,33 @@
|
||||
# Snowflake S3Compat API test suite
|
||||
|
||||
Integration tests that run the upstream
|
||||
[Snowflake s3compat API test suite](https://github.com/snowflakedb/snowflake-s3compat-api-test-suite)
|
||||
against SeaweedFS. The suite covers `getBucketLocation`, `getObject` (including
|
||||
range reads), `getObjectMetadata`, `putObject` (including a 5 GB upload),
|
||||
`listObjectsV2` (including paged listing of >1000 objects), `deleteObject`,
|
||||
`deleteObjects`, `copyObject`, and `generatePresignedUrl`.
|
||||
|
||||
## Running locally
|
||||
|
||||
Requires `weed` (or `WEED_BIN`), the `aws` CLI, `mvn`, and JDK 11+ on `PATH`.
|
||||
|
||||
```sh
|
||||
(cd weed && go install -buildvcs=false) # build weed first
|
||||
bash test/s3/snowflake/run.sh
|
||||
```
|
||||
|
||||
`run.sh` starts a `weed server` with S3 enabled, calls `prepare.sh` to create
|
||||
the fixtures, clones the suite into a scratch dir, and runs
|
||||
`mvn -Dtest=S3CompatApiTest`. Set `WORK_DIR` to keep the server log and suite
|
||||
clone around, `SKIP_SERVER_START=1` with `ENDPOINT_URL` to run against an
|
||||
already-running server, and see the top of `run.sh` for the other overrides.
|
||||
|
||||
## Notes
|
||||
|
||||
- The server runs with `-s3.autoCreateBucket=false` so PUTs to a missing bucket
|
||||
return `NoSuchBucket` like AWS; the suite asserts this.
|
||||
- The suite forces virtual-hosted-style bucket addressing, which requires
|
||||
wildcard DNS that does not exist for a local endpoint. `run.sh` switches it
|
||||
to path-style access with a `sed` patch.
|
||||
- `NOT_ACCESSIBLE_BUCKET` is a real bucket carrying a deny-all bucket policy,
|
||||
which is how the suite's `AccessDenied` negative tests are satisfied.
|
||||
Executable
+81
@@ -0,0 +1,81 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# Prepares the fixtures required by the Snowflake s3compat API test suite
|
||||
# (https://github.com/snowflakedb/snowflake-s3compat-api-test-suite) against a
|
||||
# running SeaweedFS S3 endpoint:
|
||||
#
|
||||
# BUCKET_NAME_1 versioning-enabled bucket the suite writes to
|
||||
# NOT_ACCESSIBLE_BUCKET bucket with a deny-all bucket policy; the suite
|
||||
# expects 403 AccessDenied for every operation on it
|
||||
# PREFIX_FOR_PAGE_LISTING prefix under BUCKET_NAME_1 holding more than 1000
|
||||
# objects (the suite asserts paged listing works)
|
||||
#
|
||||
# Requires: aws CLI on PATH, AWS_ACCESS_KEY_ID / AWS_SECRET_ACCESS_KEY set.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
ENDPOINT_URL="${ENDPOINT_URL:-http://127.0.0.1:8333}"
|
||||
BUCKET_NAME_1="${BUCKET_NAME_1:-sf-snowflake-test}"
|
||||
NOT_ACCESSIBLE_BUCKET="${NOT_ACCESSIBLE_BUCKET:-sf-denied-bucket}"
|
||||
PREFIX_FOR_PAGE_LISTING="${PREFIX_FOR_PAGE_LISTING:-test-suite/page-listing/}"
|
||||
PAGE_LISTING_TOTAL_SIZE="${PAGE_LISTING_TOTAL_SIZE:-1100}"
|
||||
|
||||
aws="aws --endpoint-url $ENDPOINT_URL"
|
||||
|
||||
echo "Creating test bucket $BUCKET_NAME_1 with versioning enabled"
|
||||
$aws s3api create-bucket --bucket "$BUCKET_NAME_1"
|
||||
$aws s3api put-bucket-versioning --bucket "$BUCKET_NAME_1" --versioning-configuration Status=Enabled
|
||||
$aws s3api get-bucket-versioning --bucket "$BUCKET_NAME_1"
|
||||
|
||||
echo "Creating not-accessible bucket $NOT_ACCESSIBLE_BUCKET with a deny-all bucket policy"
|
||||
$aws s3api create-bucket --bucket "$NOT_ACCESSIBLE_BUCKET"
|
||||
POLICY_FILE="$(mktemp)"
|
||||
cat > "$POLICY_FILE" <<EOF
|
||||
{
|
||||
"Version": "2012-10-17",
|
||||
"Statement": [
|
||||
{
|
||||
"Sid": "DenyAll",
|
||||
"Effect": "Deny",
|
||||
"Principal": "*",
|
||||
"Action": "s3:*",
|
||||
"Resource": [
|
||||
"arn:aws:s3:::${NOT_ACCESSIBLE_BUCKET}",
|
||||
"arn:aws:s3:::${NOT_ACCESSIBLE_BUCKET}/*"
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
EOF
|
||||
$aws s3api put-bucket-policy --bucket "$NOT_ACCESSIBLE_BUCKET" --policy "file://$POLICY_FILE"
|
||||
rm -f "$POLICY_FILE"
|
||||
|
||||
if OUT="$($aws s3api get-bucket-location --bucket "$NOT_ACCESSIBLE_BUCKET" 2>&1)"; then
|
||||
echo "ERROR: expected AccessDenied on $NOT_ACCESSIBLE_BUCKET, got success" >&2
|
||||
exit 1
|
||||
elif ! echo "$OUT" | grep -q "AccessDenied"; then
|
||||
echo "ERROR: expected AccessDenied on $NOT_ACCESSIBLE_BUCKET, got: $OUT" >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "Verified $NOT_ACCESSIBLE_BUCKET denies access"
|
||||
|
||||
if [ "$PAGE_LISTING_TOTAL_SIZE" -le 1000 ]; then
|
||||
echo "ERROR: PAGE_LISTING_TOTAL_SIZE must be > 1000, got $PAGE_LISTING_TOTAL_SIZE" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo "Uploading $PAGE_LISTING_TOTAL_SIZE objects to s3://$BUCKET_NAME_1/$PREFIX_FOR_PAGE_LISTING"
|
||||
WORKDIR="$(mktemp -d)"
|
||||
for i in $(seq 1 "$PAGE_LISTING_TOTAL_SIZE"); do
|
||||
echo "object-$i" > "$WORKDIR/file_$(printf %05d "$i").txt"
|
||||
done
|
||||
$aws s3 sync "$WORKDIR" "s3://$BUCKET_NAME_1/$PREFIX_FOR_PAGE_LISTING" --quiet
|
||||
rm -rf "$WORKDIR"
|
||||
|
||||
COUNT=$($aws s3 ls "s3://$BUCKET_NAME_1/$PREFIX_FOR_PAGE_LISTING" | wc -l | tr -d ' ')
|
||||
if [ "$COUNT" != "$PAGE_LISTING_TOTAL_SIZE" ]; then
|
||||
echo "ERROR: expected $PAGE_LISTING_TOTAL_SIZE objects under $PREFIX_FOR_PAGE_LISTING, found $COUNT" >&2
|
||||
exit 1
|
||||
fi
|
||||
echo "Verified $COUNT objects under s3://$BUCKET_NAME_1/$PREFIX_FOR_PAGE_LISTING"
|
||||
echo "Fixtures ready."
|
||||
Executable
+140
@@ -0,0 +1,140 @@
|
||||
#!/usr/bin/env bash
|
||||
#
|
||||
# Runs the Snowflake s3compat API test suite
|
||||
# (https://github.com/snowflakedb/snowflake-s3compat-api-test-suite) against a
|
||||
# locally-built SeaweedFS server.
|
||||
#
|
||||
# Required on PATH: weed (or WEED_BIN), aws, mvn, java, git.
|
||||
#
|
||||
# Env overrides:
|
||||
# WEED_BIN path to the weed binary (default: weed)
|
||||
# WORK_DIR scratch dir for data + suite clone (default: mktemp -d,
|
||||
# removed on success, kept on failure)
|
||||
# MASTER_PORT master http port (default: 9333)
|
||||
# VOLUME_PORT volume http port (default: 8080)
|
||||
# FILER_PORT filer http port (default: 8888)
|
||||
# S3_PORT s3 endpoint port (default: 8333)
|
||||
# METRICS_PORT metrics http port (default: 9324)
|
||||
# SKIP_SERVER_START if set, do not start weed; prepare and run the suite
|
||||
# against ENDPOINT_URL
|
||||
# SUITE_REPO git url of the test suite (default: upstream)
|
||||
# SUITE_REV suite commit to check out (default: pinned SHA)
|
||||
#
|
||||
# The suite env vars (BUCKET_NAME_1, PREFIX_FOR_PAGE_LISTING,
|
||||
# PAGE_LISTING_TOTAL_SIZE, NOT_ACCESSIBLE_BUCKET) default to the same values
|
||||
# prepare.sh uses.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
|
||||
WEED_BIN="${WEED_BIN:-weed}"
|
||||
MASTER_PORT="${MASTER_PORT:-9333}"
|
||||
VOLUME_PORT="${VOLUME_PORT:-8080}"
|
||||
FILER_PORT="${FILER_PORT:-8888}"
|
||||
S3_PORT="${S3_PORT:-8333}"
|
||||
METRICS_PORT="${METRICS_PORT:-9324}"
|
||||
ENDPOINT_URL="${ENDPOINT_URL:-http://127.0.0.1:$S3_PORT}"
|
||||
SUITE_REPO="${SUITE_REPO:-https://github.com/snowflakedb/snowflake-s3compat-api-test-suite.git}"
|
||||
# Pinned upstream revision verified against SeaweedFS; bump deliberately.
|
||||
SUITE_REV="${SUITE_REV:-8ae535b35fff0d8a72e21bba4e51281ac991cab9}"
|
||||
WORK_DIR_CREATED=""
|
||||
if [ -z "${WORK_DIR:-}" ]; then
|
||||
WORK_DIR="$(mktemp -d)"
|
||||
WORK_DIR_CREATED=1
|
||||
fi
|
||||
|
||||
export BUCKET_NAME_1="${BUCKET_NAME_1:-sf-snowflake-test}"
|
||||
export NOT_ACCESSIBLE_BUCKET="${NOT_ACCESSIBLE_BUCKET:-sf-denied-bucket}"
|
||||
export PREFIX_FOR_PAGE_LISTING="${PREFIX_FOR_PAGE_LISTING:-test-suite/page-listing/}"
|
||||
export PAGE_LISTING_TOTAL_SIZE="${PAGE_LISTING_TOTAL_SIZE:-1100}"
|
||||
export ENDPOINT_URL
|
||||
|
||||
# The suite reads credentials from these variables.
|
||||
export S3COMPAT_ACCESS_KEY="${S3COMPAT_ACCESS_KEY:-snowflake_compat_access}"
|
||||
export S3COMPAT_SECRET_KEY="${S3COMPAT_SECRET_KEY:-snowflake_compat_secret}"
|
||||
export AWS_ACCESS_KEY_ID="$S3COMPAT_ACCESS_KEY"
|
||||
export AWS_SECRET_ACCESS_KEY="$S3COMPAT_SECRET_KEY"
|
||||
|
||||
WEED_PID=""
|
||||
cleanup() {
|
||||
status=$?
|
||||
if [ -n "$WEED_PID" ]; then
|
||||
kill "$WEED_PID" 2>/dev/null || true
|
||||
sleep 2
|
||||
kill -9 "$WEED_PID" 2>/dev/null || true
|
||||
fi
|
||||
if [ -n "$WORK_DIR_CREATED" ]; then
|
||||
if [ "$status" -eq 0 ]; then
|
||||
rm -rf "$WORK_DIR"
|
||||
else
|
||||
echo "Work dir kept for debugging: $WORK_DIR" >&2
|
||||
fi
|
||||
fi
|
||||
}
|
||||
trap cleanup EXIT
|
||||
|
||||
wait_for_url() {
|
||||
local url="$1" name="$2"
|
||||
for i in $(seq 1 30); do
|
||||
if curl -s "$url" > /dev/null 2>&1; then
|
||||
echo "$name is ready"
|
||||
return 0
|
||||
fi
|
||||
echo "Waiting for $name... ($i/30)"
|
||||
sleep 2
|
||||
done
|
||||
echo "ERROR: $name did not become ready" >&2
|
||||
return 1
|
||||
}
|
||||
|
||||
if [ -z "${SKIP_SERVER_START:-}" ]; then
|
||||
WEED_DATA_DIR="$WORK_DIR/data"
|
||||
mkdir -p "$WEED_DATA_DIR"
|
||||
|
||||
echo "Starting SeaweedFS (data dir: $WEED_DATA_DIR)"
|
||||
"$WEED_BIN" server -filer -filer.maxMB=64 -s3 -ip 127.0.0.1 -ip.bind 127.0.0.1 \
|
||||
-dir="$WEED_DATA_DIR" \
|
||||
-master.raftHashicorp -master.electionTimeout 1s -master.volumeSizeLimitMB=5000 \
|
||||
-volume.max=4 -volume.preStopSeconds=1 \
|
||||
-master.peers=none \
|
||||
-master.port="$MASTER_PORT" -volume.port="$VOLUME_PORT" -filer.port="$FILER_PORT" -s3.port="$S3_PORT" \
|
||||
-metricsPort="$METRICS_PORT" \
|
||||
-s3.allowDeleteBucketNotEmpty=true \
|
||||
-s3.autoCreateBucket=false \
|
||||
-s3.port.iceberg=0 -s3.port.lance=0 \
|
||||
-s3.config="$SCRIPT_DIR/s3.json" \
|
||||
> "$WORK_DIR/weed.log" 2>&1 &
|
||||
WEED_PID=$!
|
||||
|
||||
wait_for_url "http://127.0.0.1:$MASTER_PORT/cluster/status" "Master server"
|
||||
wait_for_url "http://127.0.0.1:$VOLUME_PORT/status" "Volume server"
|
||||
wait_for_url "http://127.0.0.1:$FILER_PORT/" "Filer"
|
||||
wait_for_url "$ENDPOINT_URL/" "S3 API"
|
||||
echo "All SeaweedFS components are ready"
|
||||
fi
|
||||
|
||||
"$SCRIPT_DIR/prepare.sh"
|
||||
|
||||
SUITE_DIR="$WORK_DIR/snowflake-s3compat-api-test-suite"
|
||||
if [ ! -d "$SUITE_DIR" ]; then
|
||||
git init -q "$SUITE_DIR"
|
||||
git -C "$SUITE_DIR" remote add origin "$SUITE_REPO"
|
||||
git -C "$SUITE_DIR" fetch -q --depth 1 origin "$SUITE_REV"
|
||||
git -C "$SUITE_DIR" checkout -q FETCH_HEAD
|
||||
fi
|
||||
|
||||
# The suite forces virtual-hosted style bucket addressing, which needs wildcard
|
||||
# DNS (<bucket>.<endpoint>) that does not exist for a local endpoint. Switch it
|
||||
# to path-style access, which SeaweedFS supports.
|
||||
STORAGE_CLIENT="$SUITE_DIR/s3compatapi/src/main/java/com/snowflake/s3compatapitestsuite/compatapi/S3CompatStorageClient.java"
|
||||
sed -i.bak 's/setPathStyleAccess(false)/setPathStyleAccess(true)/' "$STORAGE_CLIENT"
|
||||
rm -f "$STORAGE_CLIENT.bak"
|
||||
grep -q 'setPathStyleAccess(true)' "$STORAGE_CLIENT"
|
||||
|
||||
cd "$SUITE_DIR/s3compatapi"
|
||||
|
||||
END_POINT="$ENDPOINT_URL" \
|
||||
REGION_1=us-east-1 \
|
||||
REGION_2=us-west-2 \
|
||||
mvn -B test -Dtest=S3CompatApiTest
|
||||
@@ -0,0 +1,20 @@
|
||||
{
|
||||
"identities": [
|
||||
{
|
||||
"name": "snowflake_compat_admin",
|
||||
"credentials": [
|
||||
{
|
||||
"accessKey": "snowflake_compat_access",
|
||||
"secretKey": "snowflake_compat_secret"
|
||||
}
|
||||
],
|
||||
"actions": [
|
||||
"Admin",
|
||||
"Read",
|
||||
"List",
|
||||
"Tagging",
|
||||
"Write"
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,14 @@
|
||||
# Strict Iceberg reader used by the OLake integration test. PyIceberg is
|
||||
# deliberate here: it rejects manifests that omit the spec's field ids and
|
||||
# parquet without either field ids or a name mapping, so a passing read proves
|
||||
# the catalog served something every engine can consume -- not just something
|
||||
# OLake itself can read back.
|
||||
FROM python:3.11-slim
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
RUN pip install --no-cache-dir "pyiceberg[s3fs]==0.11.1" "pyarrow==25.0.0"
|
||||
|
||||
COPY inspect_table.py /app/
|
||||
|
||||
ENTRYPOINT ["python3", "/app/inspect_table.py"]
|
||||
@@ -0,0 +1,104 @@
|
||||
# OLake Iceberg Catalog Integration Test
|
||||
|
||||
An integration test for [OLake](https://github.com/datazip-inc/olake) against
|
||||
SeaweedFS's Iceberg REST Catalog, in the same shape as `catalog_clickhouse`.
|
||||
|
||||
## Why OLake, given we already test five engines
|
||||
|
||||
Two things here are covered by nothing else in this directory.
|
||||
|
||||
**It is a strict Java Iceberg client.** OLake does not write Iceberg from Go —
|
||||
its Go process spawns a Java sidecar over gRPC and writes through the official
|
||||
Apache Iceberg library, because the Go library has no equality deletes and CDC
|
||||
needs them. So this test exercises the client class that
|
||||
`weed/s3api/iceberg/metadata_compliance.go` exists to serve: the one
|
||||
that fails with *"Cannot parse missing long current-snapshot-id"* when the
|
||||
catalog omits spec-required keys that `iceberg-go` strips via `omitempty`.
|
||||
|
||||
**It produces equality deletes.** OLake is a CDC tool. Its upsert path commits
|
||||
`operation=overwrite` with an equality-delete file, and a delete manifest
|
||||
alongside the data manifests. ClickHouse, StarRocks, Doris and DuckDB all only
|
||||
append.
|
||||
|
||||
## What it asserts
|
||||
|
||||
`TestOLakeIcebergCatalog` runs six subtests against a `weed mini` cluster with
|
||||
a pre-created table bucket and a Postgres source:
|
||||
|
||||
| Subtest | What a failure means |
|
||||
|---|---|
|
||||
| `CheckDestination` | `olake check` did not reach `SUCCEEDED`, or it passed without ever loading `org.apache.iceberg.rest.RESTSessionCatalog` — the second case means the destination was never actually contacted. |
|
||||
| `Discover` | OLake could not enumerate the source table, or wrote no `streams.json`. |
|
||||
| `FullSyncAppendsRows` | The sync read fewer than the three seeded rows, or committed no Iceberg snapshot. |
|
||||
| `StrictReaderSeesRows` | PyIceberg could not read back what the Java writer committed, or the values differ. This is the data path, not just metadata. |
|
||||
| `UpsertProducesEqualityDelete` | After an `UPDATE` and a re-sync, no snapshot recorded an overwrite carrying equality deletes, or the current snapshot has no delete manifest. |
|
||||
| `CompliantWriterNeedsNoRepair` | The catalog rewrote a manifest the official Iceberg Java writer produced. That is a regression in the repair gate, not a problem with OLake. |
|
||||
|
||||
## The one thing this test deliberately does not check
|
||||
|
||||
It does **not** assert that a reader sees the updated row and no duplicate.
|
||||
|
||||
PyIceberg refuses to scan a table carrying equality deletes
|
||||
([apache/iceberg#6568](https://github.com/apache/iceberg/issues/6568)) while
|
||||
reading its metadata perfectly well — so a rows-mode read after the upsert would
|
||||
raise, not pass. The alternative is an engine that applies equality deletes,
|
||||
which for StarRocks means a 3 GB image and 12 GB of RAM in CI.
|
||||
|
||||
The line drawn instead: **recording the commit correctly is the catalog's
|
||||
contract; applying deletes on read is the query engine's.** The metadata
|
||||
assertions cover our half.
|
||||
|
||||
This was verified once by hand outside CI, on 2026-09-23, with StarRocks 4.1.4
|
||||
attached to the same catalog: after the upsert it read 3 rows / 3 distinct ids
|
||||
with `id=1` showing the updated value and `_op_type=u`. If someone later wants
|
||||
that inside the gate, add a reader that supports equality deletes — **do not**
|
||||
"upgrade" this test to a PyIceberg rows-mode read after the upsert. It would not
|
||||
pass; and if PyIceberg ever starts silently skipping deletes instead of raising,
|
||||
it would pass by not looking.
|
||||
|
||||
## What the config proves
|
||||
|
||||
Nothing in the destination config is SeaweedFS-specific:
|
||||
|
||||
```json
|
||||
{
|
||||
"catalog_type": "rest",
|
||||
"rest_catalog_url": "http://HOST:ICEBERG_PORT",
|
||||
"iceberg_s3_path": "s3://olake-tables",
|
||||
"s3_endpoint": "http://HOST:S3_PORT",
|
||||
"rest_auth_type": "oauth2",
|
||||
"oauth2_uri": "http://HOST:ICEBERG_PORT/v1/oauth/tokens",
|
||||
"credential": "ACCESS_KEY:SECRET_KEY"
|
||||
}
|
||||
```
|
||||
|
||||
`catalog_type` is the generic `rest`, auth is the standard OAuth2
|
||||
client-credentials flow, and `s3_path_style` does not even need setting —
|
||||
OLake turns it on by itself whenever `s3_endpoint` is non-empty. As of
|
||||
OLake v0.10.1 this works with no change on either side.
|
||||
|
||||
## Running it
|
||||
|
||||
```sh
|
||||
go test ./test/s3tables/catalog_olake/ -run TestOLakeIcebergCatalog -v -timeout 25m
|
||||
```
|
||||
|
||||
Needs Docker and a `weed` binary (at `weed/weed` under the repo root, or on
|
||||
`PATH`). Takes about 35 seconds. It skips rather than fails when Docker is
|
||||
absent, and `SEAWEEDFS_SKIP_OLAKE_TESTS=1` skips it outright.
|
||||
|
||||
Overrides: `OLAKE_IMAGE` (default `olakego/source-postgres:latest`),
|
||||
`POSTGRES_IMAGE` (default `postgres:16`).
|
||||
|
||||
## In CI
|
||||
|
||||
Runs as `olake-iceberg-catalog-tests` in `.github/workflows/s3-tables-tests.yml`,
|
||||
on a matrix of a pinned image plus `latest` — the same shape the ClickHouse job
|
||||
uses, and for the same reason: OLake's Iceberg writer is a Java sidecar whose
|
||||
library version moves independently of the Go release, so the `latest` leg is
|
||||
what catches drift in the client rather than in OLake itself.
|
||||
|
||||
The job asserts the suite actually ran — at least one top-level `--- PASS` and
|
||||
zero `--- SKIP` — rather than trusting a green exit. This suite skips itself
|
||||
when Docker is unavailable, and a skipped suite reporting success is how a gate
|
||||
quietly stops being one.
|
||||
@@ -0,0 +1,91 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Inspect an Iceberg table written by OLake, through the SeaweedFS REST catalog.
|
||||
|
||||
Two modes, because a strict reader cannot do both:
|
||||
|
||||
rows -- scan the table and print "id,region,amount" per row, ordered by
|
||||
id. Only valid while the table has no equality deletes.
|
||||
snapshots -- print one line per snapshot with its operation and delete-file
|
||||
counters, plus the manifest content kinds of the current
|
||||
snapshot.
|
||||
|
||||
The split exists because PyIceberg refuses to scan a table carrying equality
|
||||
deletes (apache/iceberg#6568) while reading its metadata perfectly well. OLake
|
||||
is a CDC tool, so its upsert path produces exactly those deletes -- asserting
|
||||
the commit landed is the catalog's concern, and applying deletes on read is the
|
||||
query engine's.
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import sys
|
||||
|
||||
from pyiceberg.catalog import load_catalog
|
||||
|
||||
|
||||
def main() -> int:
|
||||
p = argparse.ArgumentParser()
|
||||
p.add_argument("mode", choices=["rows", "snapshots"])
|
||||
p.add_argument("--catalog-url", required=True)
|
||||
p.add_argument("--warehouse", required=True)
|
||||
p.add_argument("--prefix", required=True)
|
||||
p.add_argument("--s3-endpoint", required=True)
|
||||
p.add_argument("--access-key", required=True)
|
||||
p.add_argument("--secret-key", required=True)
|
||||
p.add_argument("--region", default="us-east-1")
|
||||
p.add_argument("--namespace", action="append", required=True)
|
||||
p.add_argument("--table", required=True)
|
||||
args = p.parse_args()
|
||||
|
||||
catalog = load_catalog(
|
||||
"rest",
|
||||
**{
|
||||
"type": "rest",
|
||||
"uri": args.catalog_url,
|
||||
"warehouse": args.warehouse,
|
||||
"prefix": args.prefix,
|
||||
"credential": f"{args.access_key}:{args.secret_key}",
|
||||
"s3.access-key-id": args.access_key,
|
||||
"s3.secret-access-key": args.secret_key,
|
||||
"s3.endpoint": args.s3_endpoint,
|
||||
"s3.region": args.region,
|
||||
"s3.path-style-access": "true",
|
||||
},
|
||||
)
|
||||
|
||||
table = catalog.load_table(tuple(args.namespace) + (args.table,))
|
||||
|
||||
if args.mode == "rows":
|
||||
data = table.scan().to_arrow().to_pydict()
|
||||
# amount is decimal(10,2) in Postgres but arrives here as a float, whose
|
||||
# repr drops trailing zeros (120.5, not 120.50). Format it to the source
|
||||
# scale so the expected values in the Go test stay readable.
|
||||
for row_id, region, amount in sorted(
|
||||
zip(data["id"], data["region"], data["amount"])
|
||||
):
|
||||
print("%s,%s,%.2f" % (row_id, region, float(amount)))
|
||||
return 0
|
||||
|
||||
print("format-version=%d" % table.metadata.format_version)
|
||||
ids = table.metadata.schemas[-1].identifier_field_ids
|
||||
print("identifier-field-ids=%s" % ",".join(str(i) for i in ids))
|
||||
for snap in table.metadata.snapshots:
|
||||
s = snap.summary
|
||||
print(
|
||||
"snapshot operation=%s total-delete-files=%s added-delete-files=%s "
|
||||
"added-equality-deletes=%s total-records=%s"
|
||||
% (
|
||||
s.operation,
|
||||
s.get("total-delete-files", "0"),
|
||||
s.get("added-delete-files", "0"),
|
||||
s.get("added-equality-deletes", "0"),
|
||||
s.get("total-records", "0"),
|
||||
)
|
||||
)
|
||||
current = table.current_snapshot()
|
||||
kinds = [str(m.content).rsplit(".", 1)[-1] for m in current.manifests(table.io)]
|
||||
print("current-manifest-kinds=%s" % ",".join(kinds))
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,623 @@
|
||||
// Package catalog_olake provides an integration test for OLake
|
||||
// (github.com/datazip-inc/olake) against the SeaweedFS Iceberg REST Catalog.
|
||||
//
|
||||
// OLake matters here for two reasons that no other engine in this directory
|
||||
// covers. First, it does not write Iceberg from Go: it spawns a Java sidecar
|
||||
// over gRPC and writes through the official Apache Iceberg library, so this is
|
||||
// a strict Java client -- the class that iceberg/metadata_compliance.go exists
|
||||
// to serve. Second, it is a CDC tool, so its upsert path emits equality
|
||||
// deletes, which neither the ClickHouse nor the Doris test exercises.
|
||||
package catalog_olake
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/test/testutil"
|
||||
)
|
||||
|
||||
const (
|
||||
olakeDefaultImage = "olakego/source-postgres:latest"
|
||||
postgresDefaultImage = "postgres:16"
|
||||
readerImage = "seaweedfs-olake-reader:test"
|
||||
|
||||
tableBucketName = "olake-tables"
|
||||
sourceDatabase = "olakedb"
|
||||
sourceUser = "olake"
|
||||
sourcePassword = "olakepw"
|
||||
|
||||
// OLake derives the destination namespace from the source: it joins the
|
||||
// connector name, database and schema. Asserting the derived name rather
|
||||
// than configuring one keeps the test honest about what OLake actually
|
||||
// does with our catalog.
|
||||
expectedNamespace = "postgres_olakedb_public"
|
||||
expectedTable = "orders"
|
||||
|
||||
postgresStartTimeout = 90 * time.Second
|
||||
olakeRunTimeout = 6 * time.Minute
|
||||
)
|
||||
|
||||
type TestEnvironment struct {
|
||||
seaweedDir string
|
||||
weedBinary string
|
||||
dataDir string
|
||||
configDir string
|
||||
bindIP string
|
||||
|
||||
masterPort int
|
||||
masterGrpcPort int
|
||||
volumePort int
|
||||
volumeGrpcPort int
|
||||
filerPort int
|
||||
filerGrpcPort int
|
||||
s3Port int
|
||||
s3GrpcPort int
|
||||
icebergPort int
|
||||
postgresPort int
|
||||
|
||||
accessKey string
|
||||
secretKey string
|
||||
|
||||
weedProcess *exec.Cmd
|
||||
weedCancel func()
|
||||
postgresContainer string
|
||||
}
|
||||
|
||||
func TestOLakeIcebergCatalog(t *testing.T) {
|
||||
requireOLakeRuntime(t)
|
||||
|
||||
env := NewTestEnvironment(t)
|
||||
defer env.Cleanup(t)
|
||||
|
||||
env.StartSeaweedFS(t)
|
||||
env.startPostgres(t)
|
||||
env.seedSource(t)
|
||||
|
||||
buildReaderImage(t)
|
||||
env.writeOLakeConfigs(t)
|
||||
|
||||
t.Run("CheckDestination", func(t *testing.T) {
|
||||
out := env.runOLake(t, "check",
|
||||
"--config", "/mnt/config/source.json",
|
||||
"--destination", "/mnt/config/destination.json")
|
||||
if !strings.Contains(out, `"status":"SUCCEEDED"`) {
|
||||
t.Fatalf("olake check did not report SUCCEEDED.\n%s", tailLines(out, 40))
|
||||
}
|
||||
// The Java sidecar, not the Go process, is what talks to our catalog.
|
||||
// If this line is missing the check passed without exercising the
|
||||
// Iceberg REST path at all.
|
||||
if !strings.Contains(out, "org.apache.iceberg.rest.RESTSessionCatalog") {
|
||||
t.Errorf("check passed but never loaded the Iceberg REST catalog; "+
|
||||
"the destination may not have been contacted.\n%s", tailLines(out, 40))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("Discover", func(t *testing.T) {
|
||||
out := env.runOLake(t, "discover",
|
||||
"--config", "/mnt/config/source.json",
|
||||
"--destination", "/mnt/config/destination.json")
|
||||
if !strings.Contains(out, `"stream_name":"`+expectedTable+`"`) {
|
||||
t.Fatalf("discover did not report the %s stream.\n%s", expectedTable, tailLines(out, 20))
|
||||
}
|
||||
if _, err := os.Stat(filepath.Join(env.configDir, "streams.json")); err != nil {
|
||||
t.Fatalf("discover did not write streams.json: %v", err)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("FullSyncAppendsRows", func(t *testing.T) {
|
||||
out := env.runOLake(t, "sync",
|
||||
"--config", "/mnt/config/source.json",
|
||||
"--destination", "/mnt/config/destination.json",
|
||||
"--streams", "/mnt/config/streams.json",
|
||||
"--state", "/mnt/config/state.json")
|
||||
if !strings.Contains(out, "Total records read: 3") {
|
||||
t.Fatalf("sync did not read the three seeded rows.\n%s", tailLines(out, 30))
|
||||
}
|
||||
if !strings.Contains(out, "Committed snapshot") {
|
||||
t.Fatalf("sync never committed an Iceberg snapshot.\n%s", tailLines(out, 30))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("StrictReaderSeesRows", func(t *testing.T) {
|
||||
rows := env.readTable(t, "rows")
|
||||
want := []string{"1,us-east,120.50", "2,us-west,87.20", "3,eu-west,210.00"}
|
||||
got := nonEmptyLines(rows)
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("PyIceberg read %d rows, want %d.\n%s", len(got), len(want), rows)
|
||||
}
|
||||
for i := range want {
|
||||
if got[i] != want[i] {
|
||||
t.Errorf("row %d = %q, want %q", i, got[i], want[i])
|
||||
}
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("UpsertProducesEqualityDelete", func(t *testing.T) {
|
||||
// Push the row past the incremental cursor so the next sync re-emits it
|
||||
// as an update rather than skipping it.
|
||||
env.execSQL(t, "UPDATE orders SET amount=999.99, region='ap-south' WHERE id=1")
|
||||
|
||||
out := env.runOLake(t, "sync",
|
||||
"--config", "/mnt/config/source.json",
|
||||
"--destination", "/mnt/config/destination.json",
|
||||
"--streams", "/mnt/config/streams.json",
|
||||
"--state", "/mnt/config/state.json")
|
||||
if !strings.Contains(out, "delete files") {
|
||||
t.Fatalf("re-sync committed no delete files.\n%s", tailLines(out, 30))
|
||||
}
|
||||
|
||||
meta := env.readTable(t, "snapshots")
|
||||
if !strings.Contains(meta, "format-version=2") {
|
||||
t.Errorf("expected Iceberg format-version 2, got:\n%s", meta)
|
||||
}
|
||||
if !strings.Contains(meta, "identifier-field-ids=") ||
|
||||
strings.Contains(meta, "identifier-field-ids=\n") {
|
||||
t.Errorf("table carries no identifier fields, so OLake could not "+
|
||||
"have upserted:\n%s", meta)
|
||||
}
|
||||
|
||||
var sawOverwrite bool
|
||||
for _, line := range nonEmptyLines(meta) {
|
||||
if !strings.HasPrefix(line, "snapshot operation=") {
|
||||
continue
|
||||
}
|
||||
if strings.Contains(line, "operation=Operation.OVERWRITE") &&
|
||||
!strings.Contains(line, "added-equality-deletes=0") {
|
||||
sawOverwrite = true
|
||||
}
|
||||
}
|
||||
if !sawOverwrite {
|
||||
t.Errorf("no snapshot recorded an overwrite carrying equality "+
|
||||
"deletes:\n%s", meta)
|
||||
}
|
||||
if !strings.Contains(meta, "current-manifest-kinds=") ||
|
||||
!strings.Contains(meta, "DELETES") {
|
||||
t.Errorf("current snapshot has no delete manifest:\n%s", meta)
|
||||
}
|
||||
// Deliberately NOT asserted here: that a reader sees the updated value
|
||||
// and no duplicate row. PyIceberg refuses to scan a table carrying
|
||||
// equality deletes (apache/iceberg#6568), so a rows-mode read would
|
||||
// fail rather than pass, and swapping in an engine that can apply them
|
||||
// costs this test a multi-gigabyte image. Applying deletes on read is
|
||||
// the engine's contract; recording the commit correctly is ours, and
|
||||
// that is what the assertions above cover. See the README.
|
||||
})
|
||||
|
||||
t.Run("CompliantWriterNeedsNoRepair", func(t *testing.T) {
|
||||
names := env.listTableMetadata(t)
|
||||
if len(names) == 0 {
|
||||
t.Fatalf("no metadata files found for %s.%s", expectedNamespace, expectedTable)
|
||||
}
|
||||
for _, n := range names {
|
||||
if strings.HasPrefix(n, "repaired-") {
|
||||
t.Errorf("catalog repaired a manifest written by OLake (%s); "+
|
||||
"the official Iceberg Java writer is expected to be "+
|
||||
"spec-compliant, so this is a regression in the repair "+
|
||||
"gate rather than in OLake", n)
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func NewTestEnvironment(t *testing.T) *TestEnvironment {
|
||||
t.Helper()
|
||||
|
||||
wd, err := os.Getwd()
|
||||
if err != nil {
|
||||
t.Fatalf("Failed to get working directory: %v", err)
|
||||
}
|
||||
|
||||
seaweedDir := wd
|
||||
for i := 0; i < 6; i++ {
|
||||
if _, err := os.Stat(filepath.Join(seaweedDir, "go.mod")); err == nil {
|
||||
break
|
||||
}
|
||||
seaweedDir = filepath.Dir(seaweedDir)
|
||||
}
|
||||
|
||||
weedBinary := filepath.Join(seaweedDir, "weed", "weed")
|
||||
if info, err := os.Stat(weedBinary); err != nil || info.IsDir() {
|
||||
weedBinary = filepath.Join(seaweedDir, "weed", "weed", "weed")
|
||||
if info, err := os.Stat(weedBinary); err != nil || info.IsDir() {
|
||||
weedBinary = "weed"
|
||||
if _, err := exec.LookPath(weedBinary); err != nil {
|
||||
t.Skip("weed binary not found, skipping integration test")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
dataDir, err := os.MkdirTemp("", "seaweed-olake-test-*")
|
||||
if err != nil {
|
||||
t.Fatalf("Failed to create temp dir: %v", err)
|
||||
}
|
||||
configDir := filepath.Join(dataDir, "olake")
|
||||
if err := os.MkdirAll(configDir, 0755); err != nil {
|
||||
t.Fatalf("Failed to create config dir: %v", err)
|
||||
}
|
||||
|
||||
// 9 for the mini cluster, 1 for the Postgres source mapped on the host.
|
||||
ports := testutil.MustAllocatePorts(t, 10)
|
||||
|
||||
return &TestEnvironment{
|
||||
seaweedDir: seaweedDir,
|
||||
weedBinary: weedBinary,
|
||||
dataDir: dataDir,
|
||||
configDir: configDir,
|
||||
bindIP: testutil.FindBindIP(),
|
||||
masterPort: ports[0],
|
||||
masterGrpcPort: ports[1],
|
||||
volumePort: ports[2],
|
||||
volumeGrpcPort: ports[3],
|
||||
filerPort: ports[4],
|
||||
filerGrpcPort: ports[5],
|
||||
s3Port: ports[6],
|
||||
s3GrpcPort: ports[7],
|
||||
icebergPort: ports[8],
|
||||
postgresPort: ports[9],
|
||||
accessKey: "AKIAIOSFODNN7EXAMPLE",
|
||||
secretKey: "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY",
|
||||
}
|
||||
}
|
||||
|
||||
func (env *TestEnvironment) StartSeaweedFS(t *testing.T) {
|
||||
t.Helper()
|
||||
|
||||
iamConfigPath, err := testutil.WriteIAMConfig(env.dataDir, env.accessKey, env.secretKey)
|
||||
if err != nil {
|
||||
t.Fatalf("Failed to create IAM config: %v", err)
|
||||
}
|
||||
|
||||
cmd := exec.Command(env.weedBinary, "mini",
|
||||
"-master.port", fmt.Sprintf("%d", env.masterPort),
|
||||
"-master.port.grpc", fmt.Sprintf("%d", env.masterGrpcPort),
|
||||
"-volume.port", fmt.Sprintf("%d", env.volumePort),
|
||||
"-volume.port.grpc", fmt.Sprintf("%d", env.volumeGrpcPort),
|
||||
"-filer.port", fmt.Sprintf("%d", env.filerPort),
|
||||
"-filer.port.grpc", fmt.Sprintf("%d", env.filerGrpcPort),
|
||||
"-s3.port", fmt.Sprintf("%d", env.s3Port),
|
||||
"-s3.port.grpc", fmt.Sprintf("%d", env.s3GrpcPort),
|
||||
"-s3.port.iceberg", fmt.Sprintf("%d", env.icebergPort),
|
||||
"-s3.config", iamConfigPath,
|
||||
// Pre-create the table bucket the way an operator would, rather than
|
||||
// reaching for the S3 Tables control plane from the test.
|
||||
"-tableBucket", tableBucketName,
|
||||
"-ip", env.bindIP,
|
||||
"-ip.bind", "0.0.0.0",
|
||||
"-dir", env.dataDir,
|
||||
)
|
||||
cmd.Dir = env.dataDir
|
||||
cmd.Stdout = os.Stdout
|
||||
cmd.Stderr = os.Stderr
|
||||
cmd.Env = append(os.Environ(),
|
||||
"AWS_ACCESS_KEY_ID="+env.accessKey,
|
||||
"AWS_SECRET_ACCESS_KEY="+env.secretKey,
|
||||
)
|
||||
|
||||
if err := cmd.Start(); err != nil {
|
||||
t.Fatalf("Failed to start SeaweedFS: %v", err)
|
||||
}
|
||||
env.weedProcess = cmd
|
||||
env.weedCancel = func() {
|
||||
if cmd.Process != nil {
|
||||
_ = cmd.Process.Kill()
|
||||
}
|
||||
}
|
||||
|
||||
url := fmt.Sprintf("http://%s:%d/v1/config", env.bindIP, env.icebergPort)
|
||||
if !waitForService(url, 45*time.Second) {
|
||||
t.Fatalf("Iceberg REST API did not become ready at %s", url)
|
||||
}
|
||||
}
|
||||
|
||||
func (env *TestEnvironment) startPostgres(t *testing.T) {
|
||||
t.Helper()
|
||||
|
||||
name := fmt.Sprintf("olake-pg-%d", env.postgresPort)
|
||||
_ = exec.Command("docker", "rm", "-f", name).Run()
|
||||
|
||||
cmd := exec.Command("docker", "run", "-d", "--name", name,
|
||||
"-e", "POSTGRES_USER="+sourceUser,
|
||||
"-e", "POSTGRES_PASSWORD="+sourcePassword,
|
||||
"-e", "POSTGRES_DB="+sourceDatabase,
|
||||
"-p", fmt.Sprintf("%d:5432", env.postgresPort),
|
||||
postgresImage(),
|
||||
)
|
||||
if out, err := cmd.CombinedOutput(); err != nil {
|
||||
t.Fatalf("Failed to start Postgres: %v\n%s", err, out)
|
||||
}
|
||||
env.postgresContainer = name
|
||||
|
||||
deadline := time.Now().Add(postgresStartTimeout)
|
||||
for time.Now().Before(deadline) {
|
||||
probe := exec.Command("docker", "exec", name,
|
||||
"pg_isready", "-U", sourceUser, "-d", sourceDatabase)
|
||||
if err := probe.Run(); err == nil {
|
||||
return
|
||||
}
|
||||
time.Sleep(2 * time.Second)
|
||||
}
|
||||
t.Fatalf("Postgres did not become ready within %s", postgresStartTimeout)
|
||||
}
|
||||
|
||||
func (env *TestEnvironment) seedSource(t *testing.T) {
|
||||
t.Helper()
|
||||
env.execSQL(t, `
|
||||
CREATE TABLE orders (
|
||||
id int PRIMARY KEY,
|
||||
region text,
|
||||
amount numeric(10,2),
|
||||
order_ts timestamp
|
||||
);
|
||||
INSERT INTO orders VALUES
|
||||
(1,'us-east',120.50,'2026-07-27 09:15'),
|
||||
(2,'us-west', 87.20,'2026-07-27 09:20'),
|
||||
(3,'eu-west',210.00,'2026-07-27 09:31');
|
||||
ALTER TABLE orders REPLICA IDENTITY FULL;`)
|
||||
}
|
||||
|
||||
func (env *TestEnvironment) execSQL(t *testing.T, sqlText string) {
|
||||
t.Helper()
|
||||
cmd := exec.Command("docker", "exec", env.postgresContainer,
|
||||
"psql", "-v", "ON_ERROR_STOP=1", "-U", sourceUser, "-d", sourceDatabase,
|
||||
"-c", sqlText)
|
||||
if out, err := cmd.CombinedOutput(); err != nil {
|
||||
t.Fatalf("psql failed: %v\n%s", err, out)
|
||||
}
|
||||
}
|
||||
|
||||
// writeOLakeConfigs writes the source and destination configs. The destination
|
||||
// is the point of this test: nothing in it is SeaweedFS-specific. catalog_type
|
||||
// is the generic "rest", auth is the standard OAuth2 client-credentials flow,
|
||||
// and path-style access is not even set here because OLake turns it on by
|
||||
// itself whenever s3_endpoint is non-empty.
|
||||
func (env *TestEnvironment) writeOLakeConfigs(t *testing.T) {
|
||||
t.Helper()
|
||||
|
||||
source := map[string]any{
|
||||
"host": env.bindIP,
|
||||
"port": env.postgresPort,
|
||||
"database": sourceDatabase,
|
||||
"username": sourceUser,
|
||||
"password": sourcePassword,
|
||||
"jdbc_url_params": map[string]any{},
|
||||
"ssl": map[string]any{"mode": "disable"},
|
||||
"update_method": map[string]any{"type": "Standalone"},
|
||||
"max_threads": 2,
|
||||
"retry_count": 0,
|
||||
}
|
||||
|
||||
catalogURL := fmt.Sprintf("http://%s:%d", env.bindIP, env.icebergPort)
|
||||
destination := map[string]any{
|
||||
"type": "ICEBERG",
|
||||
"writer": map[string]any{
|
||||
"catalog_type": "rest",
|
||||
"rest_catalog_url": catalogURL,
|
||||
"catalog_name": "olake",
|
||||
"iceberg_s3_path": "s3://" + tableBucketName,
|
||||
"s3_endpoint": fmt.Sprintf("http://%s:%d", env.bindIP, env.s3Port),
|
||||
"s3_use_ssl": false,
|
||||
"s3_path_style": true,
|
||||
"aws_region": "us-east-1",
|
||||
"aws_access_key": env.accessKey,
|
||||
"aws_secret_key": env.secretKey,
|
||||
"rest_auth_type": "oauth2",
|
||||
"oauth2_uri": catalogURL + "/v1/oauth/tokens",
|
||||
"credential": env.accessKey + ":" + env.secretKey,
|
||||
},
|
||||
}
|
||||
|
||||
writeJSON(t, filepath.Join(env.configDir, "source.json"), source)
|
||||
writeJSON(t, filepath.Join(env.configDir, "destination.json"), destination)
|
||||
if err := os.WriteFile(filepath.Join(env.configDir, "state.json"), []byte("{}\n"), 0644); err != nil {
|
||||
t.Fatalf("Failed to write state.json: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func (env *TestEnvironment) runOLake(t *testing.T, args ...string) string {
|
||||
t.Helper()
|
||||
|
||||
full := append([]string{
|
||||
"run", "--rm",
|
||||
"-v", dockerMount(env.configDir) + ":/mnt/config",
|
||||
olakeImage(),
|
||||
}, args...)
|
||||
|
||||
cmd := exec.Command("docker", full...)
|
||||
done := make(chan struct{})
|
||||
var out []byte
|
||||
var err error
|
||||
go func() {
|
||||
out, err = cmd.CombinedOutput()
|
||||
close(done)
|
||||
}()
|
||||
select {
|
||||
case <-done:
|
||||
case <-time.After(olakeRunTimeout):
|
||||
if cmd.Process != nil {
|
||||
_ = cmd.Process.Kill()
|
||||
}
|
||||
t.Fatalf("olake %s did not finish within %s", args[0], olakeRunTimeout)
|
||||
}
|
||||
if err != nil {
|
||||
t.Fatalf("olake %s failed: %v\n%s", args[0], err, tailLines(string(out), 40))
|
||||
}
|
||||
return string(out)
|
||||
}
|
||||
|
||||
func (env *TestEnvironment) readTable(t *testing.T, mode string) string {
|
||||
t.Helper()
|
||||
|
||||
cmd := exec.Command("docker", "run", "--rm", readerImage, mode,
|
||||
"--catalog-url", fmt.Sprintf("http://%s:%d", env.bindIP, env.icebergPort),
|
||||
"--warehouse", "s3://"+tableBucketName,
|
||||
"--prefix", tableBucketName,
|
||||
"--s3-endpoint", fmt.Sprintf("http://%s:%d", env.bindIP, env.s3Port),
|
||||
"--access-key", env.accessKey,
|
||||
"--secret-key", env.secretKey,
|
||||
"--namespace", expectedNamespace,
|
||||
"--table", expectedTable,
|
||||
)
|
||||
out, err := cmd.CombinedOutput()
|
||||
if err != nil {
|
||||
t.Fatalf("reader (%s) failed: %v\n%s", mode, err, tailLines(string(out), 30))
|
||||
}
|
||||
return string(out)
|
||||
}
|
||||
|
||||
// listTableMetadata lists the table's metadata directory through the filer so
|
||||
// the repair assertion looks at real objects rather than at what the catalog
|
||||
// reports about itself.
|
||||
func (env *TestEnvironment) listTableMetadata(t *testing.T) []string {
|
||||
t.Helper()
|
||||
|
||||
url := fmt.Sprintf("http://%s:%d/buckets/%s/%s/%s/metadata/?limit=200",
|
||||
env.bindIP, env.filerPort, tableBucketName, expectedNamespace, expectedTable)
|
||||
req, err := http.NewRequest(http.MethodGet, url, nil)
|
||||
if err != nil {
|
||||
t.Fatalf("Failed to build filer request: %v", err)
|
||||
}
|
||||
req.Header.Set("Accept", "application/json")
|
||||
|
||||
resp, err := http.DefaultClient.Do(req)
|
||||
if err != nil {
|
||||
t.Fatalf("filer listing failed: %v", err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
body, _ := io.ReadAll(resp.Body)
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("filer listing returned %d: %s", resp.StatusCode, body)
|
||||
}
|
||||
|
||||
var parsed struct {
|
||||
Entries []struct {
|
||||
FullPath string `json:"FullPath"`
|
||||
} `json:"Entries"`
|
||||
}
|
||||
if err := json.Unmarshal(body, &parsed); err != nil {
|
||||
t.Fatalf("Failed to parse filer listing: %v\n%s", err, body)
|
||||
}
|
||||
|
||||
names := make([]string, 0, len(parsed.Entries))
|
||||
for _, e := range parsed.Entries {
|
||||
names = append(names, e.FullPath[strings.LastIndex(e.FullPath, "/")+1:])
|
||||
}
|
||||
return names
|
||||
}
|
||||
|
||||
func (env *TestEnvironment) Cleanup(t *testing.T) {
|
||||
t.Helper()
|
||||
|
||||
if env.postgresContainer != "" {
|
||||
_ = exec.Command("docker", "rm", "-f", env.postgresContainer).Run()
|
||||
}
|
||||
if env.weedCancel != nil {
|
||||
env.weedCancel()
|
||||
}
|
||||
if env.weedProcess != nil {
|
||||
_ = env.weedProcess.Wait()
|
||||
}
|
||||
if env.dataDir != "" {
|
||||
_ = os.RemoveAll(env.dataDir)
|
||||
}
|
||||
}
|
||||
|
||||
func buildReaderImage(t *testing.T) {
|
||||
t.Helper()
|
||||
|
||||
wd, err := os.Getwd()
|
||||
if err != nil {
|
||||
t.Fatalf("Failed to get working directory: %v", err)
|
||||
}
|
||||
cmd := exec.Command("docker", "build",
|
||||
"-f", filepath.Join(wd, "Dockerfile.reader"),
|
||||
"-t", readerImage, wd)
|
||||
if out, err := cmd.CombinedOutput(); err != nil {
|
||||
t.Fatalf("Failed to build reader image: %v\n%s", err, tailLines(string(out), 30))
|
||||
}
|
||||
}
|
||||
|
||||
func olakeImage() string {
|
||||
if v := os.Getenv("OLAKE_IMAGE"); v != "" {
|
||||
return v
|
||||
}
|
||||
return olakeDefaultImage
|
||||
}
|
||||
|
||||
func postgresImage() string {
|
||||
if v := os.Getenv("POSTGRES_IMAGE"); v != "" {
|
||||
return v
|
||||
}
|
||||
return postgresDefaultImage
|
||||
}
|
||||
|
||||
func requireOLakeRuntime(t *testing.T) {
|
||||
t.Helper()
|
||||
if os.Getenv("SEAWEEDFS_SKIP_OLAKE_TESTS") != "" {
|
||||
t.Skip("SEAWEEDFS_SKIP_OLAKE_TESTS set")
|
||||
}
|
||||
if _, err := exec.LookPath("docker"); err != nil {
|
||||
t.Skip("docker not available, skipping OLake integration test")
|
||||
}
|
||||
if err := exec.Command("docker", "info").Run(); err != nil {
|
||||
t.Skip("docker daemon not reachable, skipping OLake integration test")
|
||||
}
|
||||
}
|
||||
|
||||
func waitForService(url string, timeout time.Duration) bool {
|
||||
deadline := time.Now().Add(timeout)
|
||||
client := &http.Client{Timeout: 3 * time.Second}
|
||||
for time.Now().Before(deadline) {
|
||||
resp, err := client.Get(url)
|
||||
if err == nil {
|
||||
resp.Body.Close()
|
||||
if resp.StatusCode < 500 {
|
||||
return true
|
||||
}
|
||||
}
|
||||
time.Sleep(time.Second)
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
func writeJSON(t *testing.T, path string, v any) {
|
||||
t.Helper()
|
||||
body, err := json.MarshalIndent(v, "", " ")
|
||||
if err != nil {
|
||||
t.Fatalf("Failed to marshal %s: %v", path, err)
|
||||
}
|
||||
if err := os.WriteFile(path, append(body, '\n'), 0644); err != nil {
|
||||
t.Fatalf("Failed to write %s: %v", path, err)
|
||||
}
|
||||
}
|
||||
|
||||
// dockerMount normalises a host path for a -v bind mount. Docker Desktop
|
||||
// accepts forward slashes on Windows; the native separator it does not.
|
||||
func dockerMount(path string) string {
|
||||
return strings.ReplaceAll(path, `\`, "/")
|
||||
}
|
||||
|
||||
func nonEmptyLines(s string) []string {
|
||||
var out []string
|
||||
for _, line := range strings.Split(s, "\n") {
|
||||
if trimmed := strings.TrimSpace(line); trimmed != "" {
|
||||
out = append(out, trimmed)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func tailLines(s string, n int) string {
|
||||
lines := strings.Split(strings.TrimRight(s, "\n"), "\n")
|
||||
if len(lines) > n {
|
||||
lines = lines[len(lines)-n:]
|
||||
}
|
||||
return strings.Join(lines, "\n")
|
||||
}
|
||||
+68
-12
@@ -3,6 +3,7 @@ package cluster
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"slices"
|
||||
"strings"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
@@ -27,6 +28,7 @@ type LockClient struct {
|
||||
// correct: the filer forwards to the real primary as a fallback.
|
||||
ringMu sync.RWMutex
|
||||
ring *lock_manager.HashRing
|
||||
ringServers []pb.ServerAddress
|
||||
ringVersion int64
|
||||
|
||||
// priorRing is the ring before the most recent change, kept for priorWindow so a
|
||||
@@ -35,15 +37,24 @@ type LockClient struct {
|
||||
priorRing *lock_manager.HashRing
|
||||
ringChangedAt time.Time
|
||||
priorWindow time.Duration
|
||||
|
||||
// noLockServerRetryPeriod bounds retries when every filer reports "no
|
||||
// lock server found": an empty lock ring is a systemic fault that waiting
|
||||
// on a lock holder cannot resolve, unlike ordinary contention. The bound
|
||||
// is long enough to ride out a master leader change (ring reset + fresh
|
||||
// snapshot) but short enough that a write fails instead of hanging
|
||||
// forever.
|
||||
noLockServerRetryPeriod time.Duration
|
||||
}
|
||||
|
||||
func NewLockClient(grpcDialOption grpc.DialOption, seedFiler pb.ServerAddress) *LockClient {
|
||||
return &LockClient{
|
||||
grpcDialOption: grpcDialOption,
|
||||
maxLockDuration: 5 * time.Second,
|
||||
sleepDuration: 2473 * time.Millisecond,
|
||||
seedFiler: seedFiler,
|
||||
priorWindow: 5 * time.Second,
|
||||
grpcDialOption: grpcDialOption,
|
||||
maxLockDuration: 5 * time.Second,
|
||||
sleepDuration: 2473 * time.Millisecond,
|
||||
seedFiler: seedFiler,
|
||||
priorWindow: 5 * time.Second,
|
||||
noLockServerRetryPeriod: 15 * time.Second,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -58,10 +69,16 @@ func (lc *LockClient) SetRing(servers []pb.ServerAddress, version int64) {
|
||||
return
|
||||
}
|
||||
lc.ringVersion = version
|
||||
sorted := slices.Clone(servers)
|
||||
slices.Sort(sorted)
|
||||
if slices.Equal(sorted, lc.ringServers) {
|
||||
return
|
||||
}
|
||||
lc.ringServers = sorted
|
||||
// Build a fresh ring (not an in-place mutation) so the outgoing ring survives as
|
||||
// priorRing with its own servers for the cooling-off window.
|
||||
newRing := lock_manager.NewHashRing(lock_manager.DefaultVnodeCount)
|
||||
newRing.SetServers(servers)
|
||||
newRing.SetServers(sorted)
|
||||
if lc.ring != nil {
|
||||
lc.priorRing = lc.ring
|
||||
lc.ringChangedAt = time.Now()
|
||||
@@ -69,6 +86,18 @@ func (lc *LockClient) SetRing(servers []pb.ServerAddress, version int64) {
|
||||
lc.ring = newRing
|
||||
}
|
||||
|
||||
// ResetRing clears only the version gate so the first update from a
|
||||
// different master applies unconditionally: ring versions are per-master
|
||||
// monotonic and a high version accepted from a former leader must not
|
||||
// reject the new leader's snapshot. The last ring keeps routing during the
|
||||
// gap rather than falling back to the seed filer, and the arriving ring
|
||||
// becomes the prior ring for the cooling-off window.
|
||||
func (lc *LockClient) ResetRing() {
|
||||
lc.ringMu.Lock()
|
||||
defer lc.ringMu.Unlock()
|
||||
lc.ringVersion = 0
|
||||
}
|
||||
|
||||
// hostForKey returns the filer that should own key per the current ring view,
|
||||
// falling back to the seed filer when no view has been received yet.
|
||||
func (lc *LockClient) hostForKey(key string) pb.ServerAddress {
|
||||
@@ -133,7 +162,9 @@ type LiveLock struct {
|
||||
consecutiveFailures int // Track connection failures to trigger fallback
|
||||
}
|
||||
|
||||
// NewShortLivedLock creates a lock with a 5-second duration
|
||||
// NewShortLivedLock creates a lock with a 5-second duration.
|
||||
// It returns nil when the lock cannot be acquired because no lock server
|
||||
// exists; ordinary contention is still waited out.
|
||||
func (lc *LockClient) NewShortLivedLock(key string, owner string) (lock *LiveLock) {
|
||||
lock = &LiveLock{
|
||||
key: key,
|
||||
@@ -144,7 +175,10 @@ func (lc *LockClient) NewShortLivedLock(key string, owner string) (lock *LiveLoc
|
||||
self: owner,
|
||||
lc: lc,
|
||||
}
|
||||
lock.retryUntilLocked(5 * time.Second)
|
||||
if err := lock.retryUntilLocked(5 * time.Second); err != nil {
|
||||
glog.Warningf("create lock %s: %v", key, err)
|
||||
return nil
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
@@ -167,7 +201,10 @@ func (lc *LockClient) NewBlockingLongLivedLock(key, owner string, lockTTL time.D
|
||||
lockTTL: lockTTL,
|
||||
}
|
||||
// Block until acquired
|
||||
lock.retryUntilLocked(lockTTL)
|
||||
if err := lock.retryUntilLocked(lockTTL); err != nil {
|
||||
glog.Warningf("create lock %s: %v", key, err)
|
||||
return nil
|
||||
}
|
||||
// Start renewal goroutine using a ticker for interruptible sleep
|
||||
lock.renewalDone = make(chan struct{})
|
||||
go func() {
|
||||
@@ -266,12 +303,31 @@ func (lc *LockClient) StartLongLivedLock(key string, owner string, onLockOwnerCh
|
||||
// several seconds): when a holder on another mount releases the lock, the
|
||||
// waiter must pick it up promptly, otherwise cross-mount write handoff stalls
|
||||
// long enough to time out clients.
|
||||
func (lock *LiveLock) retryUntilLocked(lockDuration time.Duration) {
|
||||
func (lock *LiveLock) retryUntilLocked(lockDuration time.Duration) error {
|
||||
var unavailableSince time.Time
|
||||
for lock.renewToken == "" {
|
||||
if err := lock.AttemptToLock(lockDuration); err != nil {
|
||||
glog.V(1).Infof("create lock %s: %v", lock.key, err)
|
||||
err := lock.AttemptToLock(lockDuration)
|
||||
if err == nil {
|
||||
unavailableSince = time.Time{}
|
||||
continue
|
||||
}
|
||||
glog.V(1).Infof("create lock %s: %v", lock.key, err)
|
||||
if strings.Contains(err.Error(), "lock already owned") {
|
||||
// Ordinary contention: a reachable server holds the lock, so
|
||||
// waiting is the point and has no bound.
|
||||
unavailableSince = time.Time{}
|
||||
continue
|
||||
}
|
||||
// Anything else — "no lock server found", a dead ring member refusing
|
||||
// connections — is a systemic fault waiting cannot fix; give up once
|
||||
// it persists past the retry period.
|
||||
if unavailableSince.IsZero() {
|
||||
unavailableSince = time.Now()
|
||||
} else if time.Since(unavailableSince) > lock.lc.noLockServerRetryPeriod {
|
||||
return err
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func (lock *LiveLock) AttemptToLock(lockDuration time.Duration) error {
|
||||
|
||||
@@ -1,13 +1,18 @@
|
||||
package cluster
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"net"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/cluster/lock_manager"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"google.golang.org/grpc"
|
||||
"google.golang.org/grpc/credentials/insecure"
|
||||
)
|
||||
|
||||
// The gateway must resolve a lock key to the same primary the filers do,
|
||||
@@ -152,6 +157,135 @@ func TestLockClientPriorOwnerForKeyExpires(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// A master change clears the version gate so the new leader's (lower-versioned)
|
||||
// snapshot applies — versions are only comparable within one master's stream.
|
||||
func TestLockClientResetRing(t *testing.T) {
|
||||
lc := NewLockClient(nil, "seed:8888")
|
||||
|
||||
lc.SetRing([]pb.ServerAddress{"filer-a:8888", "filer-b:8888"}, 100)
|
||||
lc.ResetRing()
|
||||
|
||||
// The last ring keeps routing during the gap; only version acceptance
|
||||
// is reset so the new leader's (lower-versioned) snapshot applies.
|
||||
if got := lc.hostForKey("k"); got == "seed:8888" {
|
||||
t.Fatal("expected the previous ring to keep routing after reset")
|
||||
}
|
||||
lc.SetRing([]pb.ServerAddress{"filer-z:8888"}, 50)
|
||||
if got := lc.hostForKey("k"); got != "filer-z:8888" {
|
||||
t.Fatalf("lower version from new master not applied, got %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
type noLockServerFiler struct {
|
||||
filer_pb.UnimplementedSeaweedFilerServer
|
||||
}
|
||||
|
||||
func (s *noLockServerFiler) DistributedLock(ctx context.Context, req *filer_pb.LockRequest) (*filer_pb.LockResponse, error) {
|
||||
return &filer_pb.LockResponse{Error: lock_manager.NoLockServerError.Error()}, nil
|
||||
}
|
||||
|
||||
// When every filer reports an empty lock ring, lock acquisition must fail
|
||||
// after a bounded period instead of hanging the write forever.
|
||||
func TestNewShortLivedLockFailsFastOnNoLockServer(t *testing.T) {
|
||||
listener, err := net.Listen("tcp", "127.0.0.1:0")
|
||||
if err != nil {
|
||||
t.Fatalf("listen: %v", err)
|
||||
}
|
||||
grpcServer := grpc.NewServer()
|
||||
filer_pb.RegisterSeaweedFilerServer(grpcServer, &noLockServerFiler{})
|
||||
go grpcServer.Serve(listener)
|
||||
defer grpcServer.Stop()
|
||||
|
||||
dialOption := grpc.WithTransportCredentials(insecure.NewCredentials())
|
||||
host, port, err := net.SplitHostPort(listener.Addr().String())
|
||||
if err != nil {
|
||||
t.Fatalf("split host port: %v", err)
|
||||
}
|
||||
// "host:httpPort.grpcPort" dials the fake filer's port directly.
|
||||
lc := NewLockClient(dialOption, pb.ServerAddress(fmt.Sprintf("%s:0.%s", host, port)))
|
||||
lc.noLockServerRetryPeriod = 200 * time.Millisecond
|
||||
|
||||
start := time.Now()
|
||||
lock := lc.NewShortLivedLock("test-key", "test-owner")
|
||||
elapsed := time.Since(start)
|
||||
|
||||
if lock != nil {
|
||||
t.Fatal("expected nil lock when no lock server exists")
|
||||
}
|
||||
if elapsed > 10*time.Second {
|
||||
t.Fatalf("lock acquisition took %v, expected fail-fast", elapsed)
|
||||
}
|
||||
}
|
||||
|
||||
type contendedLockFiler struct {
|
||||
filer_pb.UnimplementedSeaweedFilerServer
|
||||
}
|
||||
|
||||
func (s *contendedLockFiler) DistributedLock(ctx context.Context, req *filer_pb.LockRequest) (*filer_pb.LockResponse, error) {
|
||||
return &filer_pb.LockResponse{Error: "lock already owned by someone"}, nil
|
||||
}
|
||||
|
||||
// A lock held by another owner is ordinary contention: acquisition waits it
|
||||
// out rather than failing on the unavailability bound.
|
||||
func TestNewShortLivedLockWaitsOutContention(t *testing.T) {
|
||||
listener, err := net.Listen("tcp", "127.0.0.1:0")
|
||||
if err != nil {
|
||||
t.Fatalf("listen: %v", err)
|
||||
}
|
||||
grpcServer := grpc.NewServer()
|
||||
filer_pb.RegisterSeaweedFilerServer(grpcServer, &contendedLockFiler{})
|
||||
go grpcServer.Serve(listener)
|
||||
defer grpcServer.Stop()
|
||||
|
||||
dialOption := grpc.WithTransportCredentials(insecure.NewCredentials())
|
||||
host, port, err := net.SplitHostPort(listener.Addr().String())
|
||||
if err != nil {
|
||||
t.Fatalf("split host port: %v", err)
|
||||
}
|
||||
lc := NewLockClient(dialOption, pb.ServerAddress(fmt.Sprintf("%s:0.%s", host, port)))
|
||||
lc.noLockServerRetryPeriod = 200 * time.Millisecond
|
||||
|
||||
done := make(chan *LiveLock, 1)
|
||||
go func() {
|
||||
done <- lc.NewShortLivedLock("test-key", "test-owner")
|
||||
}()
|
||||
select {
|
||||
case <-done:
|
||||
t.Fatal("ordinary lock contention must not hit the unavailability bound")
|
||||
case <-time.After(3 * lc.noLockServerRetryPeriod):
|
||||
}
|
||||
}
|
||||
|
||||
// A ring member that refuses connections is unavailability, not contention:
|
||||
// acquisition fails on the same bound as "no lock server found".
|
||||
func TestNewShortLivedLockFailsFastOnUnreachableFiler(t *testing.T) {
|
||||
listener, err := net.Listen("tcp", "127.0.0.1:0")
|
||||
if err != nil {
|
||||
t.Fatalf("listen: %v", err)
|
||||
}
|
||||
addr := listener.Addr().String()
|
||||
listener.Close()
|
||||
|
||||
dialOption := grpc.WithTransportCredentials(insecure.NewCredentials())
|
||||
host, port, err := net.SplitHostPort(addr)
|
||||
if err != nil {
|
||||
t.Fatalf("split host port: %v", err)
|
||||
}
|
||||
lc := NewLockClient(dialOption, pb.ServerAddress(fmt.Sprintf("%s:0.%s", host, port)))
|
||||
lc.noLockServerRetryPeriod = 200 * time.Millisecond
|
||||
|
||||
start := time.Now()
|
||||
lock := lc.NewShortLivedLock("test-key", "test-owner")
|
||||
elapsed := time.Since(start)
|
||||
|
||||
if lock != nil {
|
||||
t.Fatal("expected nil lock when the ring member is unreachable")
|
||||
}
|
||||
if elapsed > 10*time.Second {
|
||||
t.Fatalf("lock acquisition took %v, expected fail-fast", elapsed)
|
||||
}
|
||||
}
|
||||
|
||||
// LiveLock.generation is a fencing token written and read with 64-bit atomic
|
||||
// operations. On 32-bit platforms (GOARCH=386 and GOARCH=arm) a 64-bit atomic
|
||||
// op requires an 8-byte-aligned address, which Go only guarantees for the
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
package lock_manager
|
||||
|
||||
import (
|
||||
"slices"
|
||||
"sort"
|
||||
"sync"
|
||||
"time"
|
||||
@@ -54,6 +55,14 @@ func (r *LockRing) SetSnapshot(servers []pb.ServerAddress, version int64) bool {
|
||||
r.Unlock()
|
||||
return false
|
||||
}
|
||||
// An unchanged member list is only a version refresh: installing it as a
|
||||
// new snapshot would run the topology-change callback and restart the
|
||||
// prior-owner window on every periodic rebroadcast.
|
||||
if len(r.snapshots) > 0 && slices.Equal(servers, r.snapshots[0].servers) {
|
||||
r.version = version
|
||||
r.Unlock()
|
||||
return true
|
||||
}
|
||||
r.version = version
|
||||
// Update the ring while holding the lock so version and ring state
|
||||
// are always consistent — prevents a concurrent SetSnapshot from
|
||||
@@ -74,6 +83,19 @@ func (r *LockRing) SetSnapshot(servers []pb.ServerAddress, version int64) bool {
|
||||
return true
|
||||
}
|
||||
|
||||
// Reset clears only the version gate so the first update from a different
|
||||
// master always applies: ring versions are per-master monotonic, and a high
|
||||
// version accepted from a former leader must not reject the new leader's
|
||||
// view. The ring itself stays installed — writes keep routing to the last
|
||||
// known owner during the gap instead of every filer treating itself as the
|
||||
// owner, and the arriving snapshot transitions off it with the usual
|
||||
// prior-owner window.
|
||||
func (r *LockRing) Reset() {
|
||||
r.Lock()
|
||||
defer r.Unlock()
|
||||
r.version = 0
|
||||
}
|
||||
|
||||
// Version returns the current ring version.
|
||||
func (r *LockRing) Version() int64 {
|
||||
r.RLock()
|
||||
|
||||
@@ -96,3 +96,42 @@ func TestLockRing_VersionRejectsStale(t *testing.T) {
|
||||
assert.True(t, ok)
|
||||
assert.Equal(t, 1, len(r.GetSnapshot()))
|
||||
}
|
||||
|
||||
func TestLockRing_SetSnapshotUnchangedOnlyBumpsVersion(t *testing.T) {
|
||||
r := NewLockRing(100 * time.Millisecond)
|
||||
callbacks := 0
|
||||
r.SetTakeSnapshotCallback(func(snapshot []pb.ServerAddress) { callbacks++ })
|
||||
|
||||
assert.True(t, r.SetSnapshot([]pb.ServerAddress{"a:1", "b:2"}, 100))
|
||||
assert.Equal(t, 1, callbacks)
|
||||
|
||||
// A periodic rebroadcast with the same members refreshes the version
|
||||
// without a new snapshot or another topology-change callback.
|
||||
assert.True(t, r.SetSnapshot([]pb.ServerAddress{"b:2", "a:1"}, 200))
|
||||
assert.Equal(t, int64(200), r.Version())
|
||||
assert.Equal(t, 1, r.GetSnapshotCount())
|
||||
assert.Equal(t, 1, callbacks, "unchanged ring must not fire the topology callback")
|
||||
|
||||
assert.True(t, r.SetSnapshot([]pb.ServerAddress{"a:1", "b:2", "c:3"}, 300))
|
||||
assert.Equal(t, 2, callbacks)
|
||||
}
|
||||
|
||||
func TestLockRing_Reset(t *testing.T) {
|
||||
r := NewLockRing(100 * time.Millisecond)
|
||||
|
||||
// A high version accepted from a former leader must not reject the new
|
||||
// leader's view once the client has moved masters.
|
||||
ok := r.SetSnapshot([]pb.ServerAddress{"a:1", "b:2"}, 100)
|
||||
assert.True(t, ok)
|
||||
|
||||
r.Reset()
|
||||
assert.Equal(t, int64(0), r.Version())
|
||||
// The operational ring survives the reset: writes keep routing to the
|
||||
// last known owner until the new leader's snapshot arrives.
|
||||
assert.Equal(t, 2, len(r.GetSnapshot()))
|
||||
assert.NotEqual(t, "", string(r.GetPrimary("key")))
|
||||
|
||||
ok = r.SetSnapshot([]pb.ServerAddress{"c:1"}, 50)
|
||||
assert.True(t, ok, "lower version from a different master must apply after reset")
|
||||
assert.Equal(t, 1, len(r.GetSnapshot()))
|
||||
}
|
||||
|
||||
@@ -11,29 +11,38 @@ import (
|
||||
|
||||
const LockRingStabilizationInterval = 1 * time.Second
|
||||
|
||||
// LockRingRebroadcastInterval is how often the ring is re-sent even when
|
||||
// membership has not changed. Broadcasts are otherwise purely event-driven,
|
||||
// so a single lost or poisoned update would be permanent without this.
|
||||
const LockRingRebroadcastInterval = 30 * time.Second
|
||||
|
||||
// LockRingManager tracks filer membership for the distributed lock ring.
|
||||
// It batches rapid topology changes (e.g., node drop + join) with a
|
||||
// stabilization timer, then broadcasts the complete member list atomically
|
||||
// so filers receive a single consistent ring update instead of multiple
|
||||
// intermediate states.
|
||||
type LockRingManager struct {
|
||||
mu sync.Mutex
|
||||
members map[FilerGroupName]map[pb.ServerAddress]struct{}
|
||||
version map[FilerGroupName]int64
|
||||
lastBroadcast map[FilerGroupName]*master_pb.LockRingUpdate
|
||||
pendingTimer map[FilerGroupName]*time.Timer
|
||||
broadcastFn func(resp *master_pb.KeepConnectedResponse)
|
||||
stabilizeDelay time.Duration
|
||||
mu sync.Mutex
|
||||
members map[FilerGroupName]map[pb.ServerAddress]struct{}
|
||||
version map[FilerGroupName]int64
|
||||
lastBroadcast map[FilerGroupName]*master_pb.LockRingUpdate
|
||||
pendingTimer map[FilerGroupName]*time.Timer
|
||||
rebroadcastTimer map[FilerGroupName]*time.Timer
|
||||
broadcastFn func(resp *master_pb.KeepConnectedResponse)
|
||||
stabilizeDelay time.Duration
|
||||
rebroadcastInterval time.Duration
|
||||
}
|
||||
|
||||
func NewLockRingManager(broadcastFn func(resp *master_pb.KeepConnectedResponse)) *LockRingManager {
|
||||
return &LockRingManager{
|
||||
members: make(map[FilerGroupName]map[pb.ServerAddress]struct{}),
|
||||
version: make(map[FilerGroupName]int64),
|
||||
lastBroadcast: make(map[FilerGroupName]*master_pb.LockRingUpdate),
|
||||
pendingTimer: make(map[FilerGroupName]*time.Timer),
|
||||
broadcastFn: broadcastFn,
|
||||
stabilizeDelay: LockRingStabilizationInterval,
|
||||
members: make(map[FilerGroupName]map[pb.ServerAddress]struct{}),
|
||||
version: make(map[FilerGroupName]int64),
|
||||
lastBroadcast: make(map[FilerGroupName]*master_pb.LockRingUpdate),
|
||||
pendingTimer: make(map[FilerGroupName]*time.Timer),
|
||||
rebroadcastTimer: make(map[FilerGroupName]*time.Timer),
|
||||
broadcastFn: broadcastFn,
|
||||
stabilizeDelay: LockRingStabilizationInterval,
|
||||
rebroadcastInterval: LockRingRebroadcastInterval,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -116,16 +125,64 @@ func (lrm *LockRingManager) scheduleBroadcast(filerGroup FilerGroupName) {
|
||||
|
||||
func (lrm *LockRingManager) doBroadcast(filerGroup FilerGroupName) {
|
||||
lrm.mu.Lock()
|
||||
delete(lrm.pendingTimer, filerGroup)
|
||||
lrm.mu.Unlock()
|
||||
lrm.emit(filerGroup)
|
||||
}
|
||||
|
||||
// rebroadcast re-sends the current ring unless a membership broadcast is
|
||||
// still stabilizing — emitting mid-window would publish an intermediate
|
||||
// topology that the pending timer immediately replaces. The check and the
|
||||
// update must sit in one critical section or a membership change can slip
|
||||
// a pending timer in between.
|
||||
func (lrm *LockRingManager) rebroadcast(filerGroup FilerGroupName) {
|
||||
lrm.mu.Lock()
|
||||
var update *master_pb.LockRingUpdate
|
||||
if _, pending := lrm.pendingTimer[filerGroup]; !pending {
|
||||
update = lrm.nextBroadcastUpdate(filerGroup)
|
||||
}
|
||||
lrm.mu.Unlock()
|
||||
lrm.sendUpdate(filerGroup, update)
|
||||
}
|
||||
|
||||
func (lrm *LockRingManager) emit(filerGroup FilerGroupName) {
|
||||
lrm.mu.Lock()
|
||||
update := lrm.nextBroadcastUpdate(filerGroup)
|
||||
lrm.mu.Unlock()
|
||||
lrm.sendUpdate(filerGroup, update)
|
||||
}
|
||||
|
||||
func (lrm *LockRingManager) sendUpdate(filerGroup FilerGroupName, update *master_pb.LockRingUpdate) {
|
||||
if update == nil {
|
||||
return
|
||||
}
|
||||
glog.V(0).Infof("LockRing: broadcasting ring update for group %q version %d: %v", filerGroup, update.Version, update.Servers)
|
||||
if lrm.broadcastFn != nil {
|
||||
lrm.broadcastFn(&master_pb.KeepConnectedResponse{
|
||||
LockRingUpdate: update,
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// nextBroadcastUpdate stamps the current members into an update and re-arms
|
||||
// the periodic rebroadcast. It returns nil for an empty member list: an empty
|
||||
// lock ring is never usable, so a late "last member removed" event from a
|
||||
// former leader must not propagate and wedge every lock client. The last
|
||||
// non-empty broadcast stays in lastBroadcast for reconnecting clients.
|
||||
// Caller must hold lrm.mu.
|
||||
func (lrm *LockRingManager) nextBroadcastUpdate(filerGroup FilerGroupName) *master_pb.LockRingUpdate {
|
||||
members := lrm.members[filerGroup]
|
||||
if len(members) == 0 {
|
||||
return nil
|
||||
}
|
||||
// Use wall-clock nanoseconds so the version survives master restarts
|
||||
// without persistence — a restarted master produces a version greater
|
||||
// than any pre-restart value (assuming clocks don't jump backward).
|
||||
version := time.Now().UnixNano()
|
||||
lrm.version[filerGroup] = version
|
||||
servers := make([]string, 0)
|
||||
if members, ok := lrm.members[filerGroup]; ok {
|
||||
for addr := range members {
|
||||
servers = append(servers, string(addr))
|
||||
}
|
||||
servers := make([]string, 0, len(members))
|
||||
for addr := range members {
|
||||
servers = append(servers, string(addr))
|
||||
}
|
||||
update := &master_pb.LockRingUpdate{
|
||||
FilerGroup: string(filerGroup),
|
||||
@@ -133,16 +190,13 @@ func (lrm *LockRingManager) doBroadcast(filerGroup FilerGroupName) {
|
||||
Version: version,
|
||||
}
|
||||
lrm.lastBroadcast[filerGroup] = update
|
||||
delete(lrm.pendingTimer, filerGroup)
|
||||
lrm.mu.Unlock()
|
||||
|
||||
glog.V(0).Infof("LockRing: broadcasting ring update for group %q version %d: %v", filerGroup, version, servers)
|
||||
|
||||
if lrm.broadcastFn != nil {
|
||||
lrm.broadcastFn(&master_pb.KeepConnectedResponse{
|
||||
LockRingUpdate: update,
|
||||
})
|
||||
if timer, ok := lrm.rebroadcastTimer[filerGroup]; ok {
|
||||
timer.Stop()
|
||||
}
|
||||
lrm.rebroadcastTimer[filerGroup] = time.AfterFunc(lrm.rebroadcastInterval, func() {
|
||||
lrm.rebroadcast(filerGroup)
|
||||
})
|
||||
return update
|
||||
}
|
||||
|
||||
// FlushPending fires any pending timer immediately (for testing or shutdown).
|
||||
|
||||
@@ -213,6 +213,111 @@ func TestLockRingManager_NoBroadcastWithoutFn(t *testing.T) {
|
||||
time.Sleep(50 * time.Millisecond) // should not panic
|
||||
}
|
||||
|
||||
func TestLockRingManager_EmptyRingNotBroadcast(t *testing.T) {
|
||||
var mu sync.Mutex
|
||||
var broadcasts []*master_pb.LockRingUpdate
|
||||
|
||||
lrm := NewLockRingManager(func(resp *master_pb.KeepConnectedResponse) {
|
||||
mu.Lock()
|
||||
if resp.LockRingUpdate != nil {
|
||||
broadcasts = append(broadcasts, resp.LockRingUpdate)
|
||||
}
|
||||
mu.Unlock()
|
||||
})
|
||||
lrm.stabilizeDelay = 50 * time.Millisecond
|
||||
|
||||
group := FilerGroupName("default")
|
||||
|
||||
lrm.AddServer(group, "filer1:8888")
|
||||
lrm.FlushPending(group)
|
||||
|
||||
mu.Lock()
|
||||
broadcasts = nil
|
||||
mu.Unlock()
|
||||
|
||||
// Removing the last member must not propagate an empty ring.
|
||||
lrm.RemoveServer(group, "filer1:8888")
|
||||
time.Sleep(100 * time.Millisecond)
|
||||
|
||||
mu.Lock()
|
||||
assert.Equal(t, 0, len(broadcasts), "empty ring must not be broadcast")
|
||||
mu.Unlock()
|
||||
|
||||
// The last non-empty snapshot is still served to reconnecting clients.
|
||||
update := lrm.GetLastUpdate(group)
|
||||
require.NotNil(t, update)
|
||||
assert.Equal(t, []string{"filer1:8888"}, update.Servers)
|
||||
}
|
||||
|
||||
func TestLockRingManager_PeriodicRebroadcast(t *testing.T) {
|
||||
var mu sync.Mutex
|
||||
var broadcasts []*master_pb.LockRingUpdate
|
||||
|
||||
lrm := NewLockRingManager(func(resp *master_pb.KeepConnectedResponse) {
|
||||
mu.Lock()
|
||||
if resp.LockRingUpdate != nil {
|
||||
broadcasts = append(broadcasts, resp.LockRingUpdate)
|
||||
}
|
||||
mu.Unlock()
|
||||
})
|
||||
lrm.stabilizeDelay = 20 * time.Millisecond
|
||||
lrm.rebroadcastInterval = 60 * time.Millisecond
|
||||
|
||||
group := FilerGroupName("default")
|
||||
lrm.AddServer(group, "filer1:8888")
|
||||
|
||||
// Without any further membership change, the ring keeps being re-sent so
|
||||
// a lost or poisoned update cannot be permanent.
|
||||
time.Sleep(200 * time.Millisecond)
|
||||
|
||||
mu.Lock()
|
||||
require.GreaterOrEqual(t, len(broadcasts), 2, "ring should rebroadcast periodically")
|
||||
for i := 1; i < len(broadcasts); i++ {
|
||||
assert.Greater(t, broadcasts[i].Version, broadcasts[i-1].Version)
|
||||
}
|
||||
mu.Unlock()
|
||||
}
|
||||
|
||||
func TestLockRingManager_RebroadcastDefersToPendingStabilization(t *testing.T) {
|
||||
var mu sync.Mutex
|
||||
var broadcasts []*master_pb.LockRingUpdate
|
||||
|
||||
lrm := NewLockRingManager(func(resp *master_pb.KeepConnectedResponse) {
|
||||
mu.Lock()
|
||||
if resp.LockRingUpdate != nil {
|
||||
broadcasts = append(broadcasts, resp.LockRingUpdate)
|
||||
}
|
||||
mu.Unlock()
|
||||
})
|
||||
lrm.stabilizeDelay = 100 * time.Millisecond
|
||||
lrm.rebroadcastInterval = 30 * time.Millisecond
|
||||
|
||||
group := FilerGroupName("default")
|
||||
lrm.AddServer(group, "filer1:8888")
|
||||
lrm.FlushPending(group)
|
||||
|
||||
mu.Lock()
|
||||
require.Len(t, broadcasts, 1)
|
||||
mu.Unlock()
|
||||
|
||||
// A membership change just before the periodic tick: the rebroadcast must
|
||||
// not publish the unsettled ring ahead of the stabilization timer.
|
||||
lrm.RemoveServer(group, "filer1:8888")
|
||||
lrm.AddServer(group, "filer2:8888")
|
||||
time.Sleep(2 * lrm.rebroadcastInterval)
|
||||
|
||||
mu.Lock()
|
||||
assert.Len(t, broadcasts, 1, "rebroadcast during stabilization should be deferred")
|
||||
mu.Unlock()
|
||||
|
||||
time.Sleep(2 * lrm.stabilizeDelay)
|
||||
|
||||
mu.Lock()
|
||||
require.GreaterOrEqual(t, len(broadcasts), 2)
|
||||
assert.Equal(t, []string{"filer2:8888"}, broadcasts[1].Servers)
|
||||
mu.Unlock()
|
||||
}
|
||||
|
||||
func TestLockRingManager_GetLastUpdateReturnsBroadcastState(t *testing.T) {
|
||||
lrm := NewLockRingManager(nil)
|
||||
|
||||
|
||||
@@ -156,7 +156,7 @@ func backupFromLocation(volumeServer pb.ServerAddress, grpcDialOption grpc.DialO
|
||||
|
||||
// If local volume is larger than remote, recreate it
|
||||
if datSize > stats.TailOffset {
|
||||
if err := v.Destroy(false, false); err != nil {
|
||||
if err := v.Destroy(false, false, false); err != nil {
|
||||
v.Close()
|
||||
return fmt.Errorf("destroying volume: %w", err), false
|
||||
}
|
||||
|
||||
@@ -20,6 +20,8 @@ import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/replication/source"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
"google.golang.org/grpc"
|
||||
"google.golang.org/grpc/codes"
|
||||
"google.golang.org/grpc/status"
|
||||
"google.golang.org/protobuf/proto"
|
||||
)
|
||||
|
||||
@@ -41,14 +43,6 @@ func followUpdatesAndUploadToRemote(option *RemoteSyncOptions, filerSource *sour
|
||||
|
||||
var lastLogTsNs = time.Now().UnixNano()
|
||||
processEventFnWithOffset := pb.AddOffsetFunc(func(resp *filer_pb.SubscribeMetadataResponse) error {
|
||||
if resp.EventNotification.NewEntry != nil {
|
||||
if *option.storageClass == "" {
|
||||
delete(resp.EventNotification.NewEntry.Extended, s3_constants.AmzStorageClass)
|
||||
} else {
|
||||
resp.EventNotification.NewEntry.Extended[s3_constants.AmzStorageClass] = []byte(*option.storageClass)
|
||||
}
|
||||
}
|
||||
|
||||
processor.AddSyncJob(resp)
|
||||
return nil
|
||||
}, 3*time.Second, func(counter int64, lastTsNs int64) error {
|
||||
@@ -175,10 +169,10 @@ func (option *RemoteSyncOptions) makeEventProcessor(remoteStorage *remote_pb.Rem
|
||||
dest := toRemoteStorageLocation(util.FullPath(mountedDir), util.NewFullPath(parentPath, entryName), remoteStorageMountLocation)
|
||||
if message.NewEntry.IsDirectory {
|
||||
glog.V(0).Infof("mkdir %s", remote_storage.FormatLocation(dest))
|
||||
return client.WriteDirectory(dest, message.NewEntry)
|
||||
return client.WriteDirectory(dest, remoteWriteEntry(message.NewEntry, *option.storageClass))
|
||||
}
|
||||
glog.V(0).Infof("create %s", remote_storage.FormatLocation(dest))
|
||||
remoteEntry, writeErr := retriedWriteFile(client, filerSource, message.NewParentPath, message.NewEntry, dest)
|
||||
remoteEntry, writeErr := retriedWriteFile(client, filerSource, message.NewParentPath, remoteWriteEntry(message.NewEntry, *option.storageClass), dest)
|
||||
if errors.Is(writeErr, errSuperseded) {
|
||||
glog.Errorf("skipping %s: %v", remote_storage.FormatLocation(dest), writeErr)
|
||||
return nil
|
||||
@@ -213,7 +207,7 @@ func (option *RemoteSyncOptions) makeEventProcessor(remoteStorage *remote_pb.Rem
|
||||
return client.DeleteFile(dest)
|
||||
}
|
||||
if message.OldEntry != nil && message.NewEntry != nil {
|
||||
return processUpdateEvent(option, filerSource, client, mountedDir, remoteStorageMountLocation, resp)
|
||||
return processUpdateEvent(option, filerSource, *option.storageClass, client, mountedDir, remoteStorageMountLocation, resp)
|
||||
}
|
||||
|
||||
return nil
|
||||
@@ -224,6 +218,7 @@ func (option *RemoteSyncOptions) makeEventProcessor(remoteStorage *remote_pb.Rem
|
||||
func processUpdateEvent(
|
||||
filerClient filer_pb.FilerClient,
|
||||
filerSource filer_pb.FilerClient,
|
||||
storageClass string,
|
||||
client remote_storage.RemoteStorageClient,
|
||||
mountedDir string,
|
||||
remoteStorageMountLocation *remote_pb.RemoteStorageLocation,
|
||||
@@ -244,7 +239,7 @@ func processUpdateEvent(
|
||||
return nil
|
||||
}
|
||||
if message.NewEntry.IsDirectory {
|
||||
return client.WriteDirectory(dest, message.NewEntry)
|
||||
return client.WriteDirectory(dest, remoteWriteEntry(message.NewEntry, storageClass))
|
||||
}
|
||||
if isMetadataOnlyUpdate(resp.Directory, message) {
|
||||
remoteEntry, err := liveRemoteEntry(filerClient, message.NewParentPath, message.NewEntry)
|
||||
@@ -257,7 +252,7 @@ func processUpdateEvent(
|
||||
}
|
||||
if remoteEntry != nil {
|
||||
glog.V(2).Infof("update meta: %+v", resp)
|
||||
return client.UpdateFileMetadata(dest, message.OldEntry, message.NewEntry)
|
||||
return client.UpdateFileMetadata(dest, message.OldEntry, remoteWriteEntry(message.NewEntry, storageClass))
|
||||
}
|
||||
glog.V(0).Infof("never replicated, uploading %s", remote_storage.FormatLocation(dest))
|
||||
}
|
||||
@@ -277,7 +272,7 @@ func processUpdateEvent(
|
||||
}
|
||||
}
|
||||
}
|
||||
remoteEntry, writeErr := retriedWriteFile(client, filerSource, message.NewParentPath, message.NewEntry, dest)
|
||||
remoteEntry, writeErr := retriedWriteFile(client, filerSource, message.NewParentPath, remoteWriteEntry(message.NewEntry, storageClass), dest)
|
||||
if errors.Is(writeErr, errSuperseded) {
|
||||
glog.Errorf("skipping %s: %v", remote_storage.FormatLocation(dest), writeErr)
|
||||
return nil
|
||||
@@ -417,16 +412,66 @@ func shouldSendToRemote(entry *filer_pb.Entry) bool {
|
||||
return false
|
||||
}
|
||||
|
||||
// remoteWriteEntry returns the entry as remote storage should see it: the
|
||||
// storage class attribute is dropped, or overridden by -storageClass. The
|
||||
// event entry is left untouched so updateLocalEntry still compares the entry
|
||||
// the filer stored.
|
||||
func remoteWriteEntry(entry *filer_pb.Entry, storageClass string) *filer_pb.Entry {
|
||||
clone := proto.Clone(entry).(*filer_pb.Entry)
|
||||
if storageClass == "" {
|
||||
delete(clone.Extended, s3_constants.AmzStorageClass)
|
||||
} else {
|
||||
if clone.Extended == nil {
|
||||
clone.Extended = map[string][]byte{}
|
||||
}
|
||||
clone.Extended[s3_constants.AmzStorageClass] = []byte(storageClass)
|
||||
}
|
||||
return clone
|
||||
}
|
||||
|
||||
// updateLocalEntry stamps the entry an event described with its RemoteEntry.
|
||||
// The write carries IF_ENTRY_EQUAL over the event's entry: the filer deletes
|
||||
// every stored chunk absent from an updated entry, so a snapshot older than
|
||||
// the live entry (the file was rewritten while its upload was in flight, or
|
||||
// the event is a replay) would delete the live chunks. A failed precondition
|
||||
// means the filer moved past this event; the event that superseded it follows
|
||||
// in the log and stamps the current entry, so the stale stamp is skipped the
|
||||
// same way a superseded upload is.
|
||||
func updateLocalEntry(filerClient filer_pb.FilerClient, dir string, entry *filer_pb.Entry, remoteEntry *filer_pb.RemoteEntry) error {
|
||||
remoteEntry.LastLocalSyncTsNs = time.Now().UnixNano()
|
||||
expected := proto.Clone(entry).(*filer_pb.Entry)
|
||||
entry.RemoteEntry = remoteEntry
|
||||
return filerClient.WithFilerClient(false, func(client filer_pb.SeaweedFilerClient) error {
|
||||
err := filerClient.WithFilerClient(false, func(client filer_pb.SeaweedFilerClient) error {
|
||||
_, err := client.UpdateEntry(context.Background(), &filer_pb.UpdateEntryRequest{
|
||||
Directory: dir,
|
||||
Entry: entry,
|
||||
Condition: ifEntryEqual(expected),
|
||||
})
|
||||
return err
|
||||
})
|
||||
if isFailedPrecondition(err) {
|
||||
glog.Errorf("skipping stale stamp of %s: %v", util.NewFullPath(dir, entry.Name), err)
|
||||
return nil
|
||||
}
|
||||
return err
|
||||
}
|
||||
|
||||
// ifEntryEqual builds the precondition that the stored entry still equals the
|
||||
// one the event described: chunk fids, inline content, and metadata alike.
|
||||
func ifEntryEqual(entry *filer_pb.Entry) *filer_pb.WriteCondition {
|
||||
return &filer_pb.WriteCondition{
|
||||
Clauses: []*filer_pb.WriteCondition_Clause{{Kind: filer_pb.WriteCondition_IF_ENTRY_EQUAL, ExpectedEntry: entry}},
|
||||
}
|
||||
}
|
||||
|
||||
// isFailedPrecondition reports a write condition the filer refused, through
|
||||
// any wrapping WithFilerClient added.
|
||||
func isFailedPrecondition(err error) bool {
|
||||
if err == nil {
|
||||
return false
|
||||
}
|
||||
st, ok := status.FromError(err)
|
||||
return ok && st.Code() == codes.FailedPrecondition
|
||||
}
|
||||
|
||||
func isMultipartUploadFile(dir string, name string) bool {
|
||||
|
||||
@@ -16,6 +16,8 @@ import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api/s3_constants"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
"google.golang.org/grpc"
|
||||
"google.golang.org/grpc/codes"
|
||||
"google.golang.org/grpc/status"
|
||||
"google.golang.org/protobuf/proto"
|
||||
)
|
||||
|
||||
@@ -684,7 +686,7 @@ func TestRenameWithInheritedRemoteEntryWritesNewKey(t *testing.T) {
|
||||
|
||||
remote := &recordingRemote{}
|
||||
filerClient := &stubFilerClient{}
|
||||
if err := processUpdateEvent(filerClient, filerClient, remote, mountedDir, mountLoc, resp); err != nil {
|
||||
if err := processUpdateEvent(filerClient, filerClient, "", remote, mountedDir, mountLoc, resp); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
@@ -731,7 +733,7 @@ func TestRenameRemoteOnlyEntrySkipsEmptyUpload(t *testing.T) {
|
||||
|
||||
remote := &recordingRemote{}
|
||||
filerClient := &stubFilerClient{}
|
||||
if err := processUpdateEvent(filerClient, filerClient, remote, mountedDir, mountLoc, resp); err != nil {
|
||||
if err := processUpdateEvent(filerClient, filerClient, "", remote, mountedDir, mountLoc, resp); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(remote.writes) != 0 {
|
||||
@@ -764,7 +766,7 @@ func TestRenameDeleteOldKeyFailureReturnsError(t *testing.T) {
|
||||
deleteErr := errors.New("AccessDenied: Access Denied")
|
||||
remote := &recordingRemote{deleteErr: deleteErr}
|
||||
filerClient := &stubFilerClient{}
|
||||
err := processUpdateEvent(filerClient, filerClient, remote, mountedDir, mountLoc, resp)
|
||||
err := processUpdateEvent(filerClient, filerClient, "", remote, mountedDir, mountLoc, resp)
|
||||
if !errors.Is(err, deleteErr) {
|
||||
t.Errorf("err = %v, want the delete failure returned so the event is retried", err)
|
||||
}
|
||||
@@ -800,7 +802,7 @@ func TestRenameDeleteOldKeyNotFoundStillWrites(t *testing.T) {
|
||||
|
||||
remote := &recordingRemote{deleteErr: remote_storage.ErrRemoteObjectNotFound}
|
||||
filerClient := &stubFilerClient{}
|
||||
if err := processUpdateEvent(filerClient, filerClient, remote, mountedDir, mountLoc, resp); err != nil {
|
||||
if err := processUpdateEvent(filerClient, filerClient, "", remote, mountedDir, mountLoc, resp); err != nil {
|
||||
t.Fatalf("err = %v, want nil: an already-deleted old key must not block the write", err)
|
||||
}
|
||||
wantWrite := &remote_pb.RemoteStorageLocation{Name: "gcs", Bucket: "bucket", Path: "/b/dst/probe.bin"}
|
||||
@@ -808,3 +810,73 @@ func TestRenameDeleteOldKeyNotFoundStillWrites(t *testing.T) {
|
||||
t.Errorf("writes = %+v, want the new key %s written", remote.writes, remote_storage.FormatLocation(wantWrite))
|
||||
}
|
||||
}
|
||||
|
||||
func TestIfEntryEqualCarriesTheEventEntry(t *testing.T) {
|
||||
entry := &filer_pb.Entry{
|
||||
Name: "f",
|
||||
Content: []byte("inline"),
|
||||
Chunks: []*filer_pb.FileChunk{
|
||||
{FileId: "3,01a"},
|
||||
{Fid: &filer_pb.FileId{VolumeId: 4, FileKey: 0x2b, Cookie: 0x0c}},
|
||||
},
|
||||
}
|
||||
cond := ifEntryEqual(entry)
|
||||
if len(cond.Clauses) != 1 || cond.Clauses[0].Kind != filer_pb.WriteCondition_IF_ENTRY_EQUAL {
|
||||
t.Fatalf("condition = %v, want one IF_ENTRY_EQUAL clause", cond)
|
||||
}
|
||||
if got := cond.Clauses[0].ExpectedEntry; !proto.Equal(got, entry) {
|
||||
t.Fatalf("expected_entry = %v, want %v", got, entry)
|
||||
}
|
||||
}
|
||||
|
||||
// The remote-bound entry loses (or gains) the storage class attribute while
|
||||
// the event entry keeps it, so IF_ENTRY_EQUAL still sees the entry the filer
|
||||
// stored — otherwise every stamp on an S3-written object reads as stale.
|
||||
func TestRemoteWriteEntryLeavesEventEntryUntouched(t *testing.T) {
|
||||
entry := &filer_pb.Entry{
|
||||
Name: "f",
|
||||
Attributes: &filer_pb.FuseAttributes{Mtime: 1},
|
||||
Extended: map[string][]byte{s3_constants.AmzStorageClass: []byte("STANDARD")},
|
||||
}
|
||||
|
||||
stripped := remoteWriteEntry(entry, "")
|
||||
if _, ok := stripped.Extended[s3_constants.AmzStorageClass]; ok {
|
||||
t.Fatalf("remote entry still carries %s", s3_constants.AmzStorageClass)
|
||||
}
|
||||
if string(entry.Extended[s3_constants.AmzStorageClass]) != "STANDARD" {
|
||||
t.Fatalf("event entry Extended mutated: %v", entry.Extended)
|
||||
}
|
||||
|
||||
overridden := remoteWriteEntry(entry, "GLACIER")
|
||||
if got := string(overridden.Extended[s3_constants.AmzStorageClass]); got != "GLACIER" {
|
||||
t.Fatalf("override = %q, want GLACIER", got)
|
||||
}
|
||||
if string(entry.Extended[s3_constants.AmzStorageClass]) != "STANDARD" {
|
||||
t.Fatalf("event entry Extended mutated: %v", entry.Extended)
|
||||
}
|
||||
|
||||
bare := &filer_pb.Entry{Name: "g", Attributes: &filer_pb.FuseAttributes{Mtime: 1}}
|
||||
if got := string(remoteWriteEntry(bare, "GLACIER").Extended[s3_constants.AmzStorageClass]); got != "GLACIER" {
|
||||
t.Fatalf("override on nil Extended = %q, want GLACIER", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestIsFailedPrecondition(t *testing.T) {
|
||||
refused := status.Error(codes.FailedPrecondition, "precondition failed: /buckets/b/f")
|
||||
cases := []struct {
|
||||
name string
|
||||
err error
|
||||
want bool
|
||||
}{
|
||||
{"nil", nil, false},
|
||||
{"status", refused, true},
|
||||
{"wrapped status", fmt.Errorf("update entry: %w", refused), true},
|
||||
{"message only", errors.New("rpc error: code = FailedPrecondition desc = precondition failed: /f"), false},
|
||||
{"other", status.Error(codes.Unavailable, "filer down"), false},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if got := isFailedPrecondition(c.err); got != c.want {
|
||||
t.Errorf("%s: isFailedPrecondition = %v, want %v", c.name, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+23
-17
@@ -46,17 +46,19 @@ const (
|
||||
)
|
||||
|
||||
type MasterOptions struct {
|
||||
port *int
|
||||
portGrpc *int
|
||||
ip *string
|
||||
ipBind *string
|
||||
metaFolder *string
|
||||
peers *string
|
||||
mastersDeprecated *string // deprecated, for backward compatibility in master.follower
|
||||
volumeSizeLimitMB *uint
|
||||
fileSizeLimitMB *int
|
||||
volumePreallocate *bool
|
||||
maxParallelVacuumPerServer *int
|
||||
port *int
|
||||
portGrpc *int
|
||||
ip *string
|
||||
ipBind *string
|
||||
metaFolder *string
|
||||
peers *string
|
||||
mastersDeprecated *string // deprecated, for backward compatibility in master.follower
|
||||
volumeSizeLimitMB *uint
|
||||
fileSizeLimitMB *int
|
||||
volumePreallocate *bool
|
||||
maxParallelVacuumPerServer *int
|
||||
vacuumIntervalSeconds *int
|
||||
vacuumDeleteEmptyAfterSeconds *int
|
||||
// pulseSeconds *int
|
||||
defaultReplication *string
|
||||
garbageThreshold *float64
|
||||
@@ -93,6 +95,8 @@ func init() {
|
||||
m.fileSizeLimitMB = cmdMaster.Flag.Int("fileSizeLimitMB", 256, "limit the file size accepted by /submit, should match the volume servers' -fileSizeLimitMB (-volume.fileSizeLimitMB under weed server or weed mini, which set this for you)")
|
||||
m.volumePreallocate = cmdMaster.Flag.Bool("volumePreallocate", false, "Preallocate disk space for volumes.")
|
||||
m.maxParallelVacuumPerServer = cmdMaster.Flag.Int("maxParallelVacuumPerServer", 1, "maximum number of volumes to vacuum in parallel per volume server")
|
||||
m.vacuumIntervalSeconds = cmdMaster.Flag.Int("vacuumIntervalSeconds", 840, "seconds between automatic vacuum sweeps")
|
||||
m.vacuumDeleteEmptyAfterSeconds = cmdMaster.Flag.Int("vacuumDeleteEmptyAfterSeconds", 0, "automatic sweep deletes volume copies that stay empty this many seconds; 0 disables")
|
||||
// m.pulseSeconds = cmdMaster.Flag.Int("pulseSeconds", 5, "number of seconds between heartbeats")
|
||||
m.defaultReplication = cmdMaster.Flag.String("defaultReplication", "", "Default replication type if not specified.")
|
||||
m.garbageThreshold = cmdMaster.Flag.Float64("garbageThreshold", 0.3, "threshold to vacuum and reclaim spaces")
|
||||
@@ -467,12 +471,14 @@ func peerIndex(self pb.ServerAddress, peers []pb.ServerAddress) int {
|
||||
func (m *MasterOptions) toMasterOption(whiteList []string) *weed_server.MasterOption {
|
||||
masterAddress := pb.NewServerAddress(*m.ip, *m.port, *m.portGrpc)
|
||||
return &weed_server.MasterOption{
|
||||
Master: masterAddress,
|
||||
MetaFolder: *m.metaFolder,
|
||||
VolumeSizeLimitMB: uint32(*m.volumeSizeLimitMB),
|
||||
FileSizeLimitMB: *m.fileSizeLimitMB,
|
||||
VolumePreallocate: *m.volumePreallocate,
|
||||
MaxParallelVacuumPerServer: *m.maxParallelVacuumPerServer,
|
||||
Master: masterAddress,
|
||||
MetaFolder: *m.metaFolder,
|
||||
VolumeSizeLimitMB: uint32(*m.volumeSizeLimitMB),
|
||||
FileSizeLimitMB: *m.fileSizeLimitMB,
|
||||
VolumePreallocate: *m.volumePreallocate,
|
||||
MaxParallelVacuumPerServer: *m.maxParallelVacuumPerServer,
|
||||
VacuumIntervalSeconds: *m.vacuumIntervalSeconds,
|
||||
VacuumDeleteEmptyAfterSeconds: *m.vacuumDeleteEmptyAfterSeconds,
|
||||
// PulseSeconds: *m.pulseSeconds,
|
||||
DefaultReplicaPlacement: *m.defaultReplication,
|
||||
GarbageThreshold: *m.garbageThreshold,
|
||||
|
||||
@@ -47,6 +47,8 @@ func init() {
|
||||
mf.metricsIntervalSec = aws.Int(0)
|
||||
mf.raftResumeState = aws.Bool(false)
|
||||
mf.maxParallelVacuumPerServer = aws.Int(1)
|
||||
mf.vacuumIntervalSeconds = aws.Int(840)
|
||||
mf.vacuumDeleteEmptyAfterSeconds = aws.Int(0)
|
||||
mf.telemetryUrl = aws.String("https://telemetry.seaweedfs.com/api/collect")
|
||||
mf.telemetryEnabled = aws.Bool(false)
|
||||
}
|
||||
|
||||
@@ -427,6 +427,8 @@ func initMiniMasterFlags() {
|
||||
miniMasterOptions.volumeSizeLimitMB = cmdMini.Flag.Uint("master.volumeSizeLimitMB", defaultMiniVolumeSizeMB, "Master stops directing writes to oversized volumes (default: 128MB for mini)")
|
||||
miniMasterOptions.volumePreallocate = cmdMini.Flag.Bool("master.volumePreallocate", false, "Preallocate disk space for volumes.")
|
||||
miniMasterOptions.maxParallelVacuumPerServer = cmdMini.Flag.Int("master.maxParallelVacuumPerServer", 1, "maximum number of volumes to vacuum in parallel on one volume server")
|
||||
miniMasterOptions.vacuumIntervalSeconds = cmdMini.Flag.Int("master.vacuumIntervalSeconds", 840, "seconds between automatic vacuum sweeps")
|
||||
miniMasterOptions.vacuumDeleteEmptyAfterSeconds = cmdMini.Flag.Int("master.vacuumDeleteEmptyAfterSeconds", 0, "automatic sweep deletes volume copies that stay empty this many seconds; 0 disables")
|
||||
miniMasterOptions.defaultReplication = cmdMini.Flag.String("master.defaultReplication", "", "Default replication type if not specified.")
|
||||
miniMasterOptions.garbageThreshold = cmdMini.Flag.Float64("master.garbageThreshold", 0.3, "threshold to vacuum and reclaim spaces")
|
||||
miniMasterOptions.metricsAddress = cmdMini.Flag.String("master.metrics.address", "", "Prometheus gateway address")
|
||||
|
||||
@@ -618,6 +618,7 @@ func (s3opt *S3Options) startLanceServer(s3ApiServer *s3api.S3ApiServer) {
|
||||
lanceRouter.Use(util_http.EscapeSemicolonsInQuery)
|
||||
|
||||
lanceServer := lance.NewServer(s3ApiServer, s3ApiServer)
|
||||
lanceServer.SetCredentialValidator(s3ApiServer)
|
||||
if s3opt.icebergCredentialRole != nil && *s3opt.icebergCredentialRole != "" {
|
||||
lanceServer.SetCredentialVendor(lanceCredentialVendor{s3ApiServer})
|
||||
}
|
||||
|
||||
@@ -41,6 +41,7 @@ copy_2 = 6 # create 2 x 6 = 12 actual volumes
|
||||
copy_3 = 3 # create 3 x 3 = 9 actual volumes
|
||||
copy_other = 1 # create n x 1 = n actual volumes
|
||||
threshold = 0.9 # create threshold
|
||||
reservation_timeout = "5m"# capacity reservation timeout before unreleased reservations expire
|
||||
disable = false # disables volume growth if true
|
||||
|
||||
# configuration flags for replication
|
||||
|
||||
@@ -98,6 +98,8 @@ func init() {
|
||||
masterOptions.volumeSizeLimitMB = cmdServer.Flag.Uint("master.volumeSizeLimitMB", util.DefaultVolumeSizeLimitMB, "Master stops directing writes to oversized volumes.")
|
||||
masterOptions.volumePreallocate = cmdServer.Flag.Bool("master.volumePreallocate", false, "Preallocate disk space for volumes.")
|
||||
masterOptions.maxParallelVacuumPerServer = cmdServer.Flag.Int("master.maxParallelVacuumPerServer", 1, "maximum number of volumes to vacuum in parallel on one volume server")
|
||||
masterOptions.vacuumIntervalSeconds = cmdServer.Flag.Int("master.vacuumIntervalSeconds", 840, "seconds between automatic vacuum sweeps")
|
||||
masterOptions.vacuumDeleteEmptyAfterSeconds = cmdServer.Flag.Int("master.vacuumDeleteEmptyAfterSeconds", 0, "automatic sweep deletes volume copies that stay empty this many seconds; 0 disables")
|
||||
masterOptions.defaultReplication = cmdServer.Flag.String("master.defaultReplication", "", "Default replication type if not specified.")
|
||||
masterOptions.garbageThreshold = cmdServer.Flag.Float64("master.garbageThreshold", 0.3, "threshold to vacuum and reclaim spaces")
|
||||
masterOptions.metricsAddress = cmdServer.Flag.String("master.metrics.address", "", "Prometheus gateway address")
|
||||
|
||||
@@ -43,7 +43,7 @@ func markVolumeReplicaWritable(ctx context.Context, grpcDialOption grpc.DialOpti
|
||||
// deleteVolume removes the volume from sourceVolumeServer via the canonical
|
||||
// volume_move helper.
|
||||
func deleteVolume(ctx context.Context, grpcDialOption grpc.DialOption, volumeId needle.VolumeId, sourceVolumeServer pb.ServerAddress, onlyEmpty bool, keepRemoteData bool) (err error) {
|
||||
return volume_move.NewMover(grpcDialOption).DeleteVolume(ctx, volumeId, sourceVolumeServer, onlyEmpty, keepRemoteData)
|
||||
return volume_move.NewMover(grpcDialOption).DeleteVolume(ctx, volumeId, sourceVolumeServer, onlyEmpty, false, keepRemoteData)
|
||||
}
|
||||
|
||||
func ChunkVolumeIds(volumeIds []needle.VolumeId, batchSize int) [][]needle.VolumeId {
|
||||
|
||||
+19
-2
@@ -7,6 +7,7 @@ import (
|
||||
"os"
|
||||
"sort"
|
||||
"strings"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/remote_storage"
|
||||
@@ -73,6 +74,11 @@ type Filer struct {
|
||||
EmptyFolderCleanupDelay time.Duration
|
||||
persistedLogCache *persistedLogCache
|
||||
metaLogInflight metaLogInflight
|
||||
remoteTombstones *remoteDeletionTombstones
|
||||
// remoteTombstonesDone, when non-nil, is closed once the startup tombstone
|
||||
// rebuild finishes; lazy remote reads wait on it so a pending delete
|
||||
// cannot resurrect in the gap.
|
||||
remoteTombstonesDone atomic.Pointer[chan struct{}]
|
||||
}
|
||||
|
||||
func NewFiler(masters pb.ServerDiscovery, grpcDialOption grpc.DialOption, filerHost pb.ServerAddress, filerGroup string, collection string, replication string, dataCenter string, maxFilenameLength uint32, notifyFn func()) *Filer {
|
||||
@@ -88,6 +94,7 @@ func NewFiler(masters pb.ServerDiscovery, grpcDialOption grpc.DialOption, filerH
|
||||
deletionQuit: make(chan struct{}),
|
||||
DeletionRetryQueue: NewDeletionRetryQueue(),
|
||||
persistedLogCache: newPersistedLogCache(persistedLogCacheMaxBytes),
|
||||
remoteTombstones: newRemoteDeletionTombstones(),
|
||||
}
|
||||
if f.UniqueFilerId < 0 {
|
||||
f.UniqueFilerId = -f.UniqueFilerId
|
||||
@@ -174,6 +181,10 @@ func (f *Filer) AggregateFromPeers(self pb.ServerAddress, existingNodes []*maste
|
||||
glog.V(0).Infof("LockRing: applying master ring update v%d: %v", update.Version, servers)
|
||||
f.Dlm.LockRing.SetSnapshot(servers, update.Version)
|
||||
})
|
||||
f.MasterClient.SetOnMasterChangeFn(func(previous, current pb.ServerAddress) {
|
||||
glog.V(0).Infof("LockRing: master changed %s -> %s, resetting ring", previous, current)
|
||||
f.Dlm.LockRing.Reset()
|
||||
})
|
||||
|
||||
// Subscribe to the local filer first: its events reach the aggregated
|
||||
// buffer only through this subscription, and the peer watermarks must
|
||||
@@ -265,7 +276,13 @@ func (f *Filer) CreateEntry(ctx context.Context, entry *Entry, existing *Entry,
|
||||
|
||||
oldEntry := existing
|
||||
if oldEntry == nil {
|
||||
oldEntry, _ = f.FindEntry(ctx, entry.FullPath)
|
||||
var findErr error
|
||||
oldEntry, findErr = f.FindEntry(ctx, entry.FullPath)
|
||||
if o_excl && findErr != nil && !errors.Is(findErr, filer_pb.ErrNotFound) {
|
||||
// An exclusive create cannot decide whether the path exists when
|
||||
// the lookup itself failed; proceeding would upsert over it.
|
||||
return fmt.Errorf("find entry %s: %w", entry.FullPath, findErr)
|
||||
}
|
||||
}
|
||||
|
||||
/*
|
||||
@@ -589,7 +606,7 @@ func (f *Filer) doListDirectoryEntries(ctx context.Context, p util.FullPath, sta
|
||||
lastFileName, err = f.Store.ListDirectoryPrefixedEntries(ctx, p, startFileName, inclusive, limit, prefix, func(entry *Entry) (bool, error) {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
glog.Errorf("Context is done.")
|
||||
glog.V(1).InfofCtx(ctx, "listing %q canceled: %v", p, ctx.Err())
|
||||
return false, fmt.Errorf("context canceled: %w", ctx.Err())
|
||||
default:
|
||||
if entry.TtlSec > 0 && !entry.IsDirectory() {
|
||||
|
||||
@@ -13,6 +13,7 @@ import (
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/cluster"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util/log_buffer"
|
||||
@@ -96,7 +97,7 @@ func newFilerWithFakeMaster(t *testing.T) (*Filer, *hookedStore, *collectionDele
|
||||
|
||||
mc := wdclient.NewMasterClient(
|
||||
grpc.WithTransportCredentials(insecure.NewCredentials()),
|
||||
"test", cluster.FilerType, pb.ServerAddress("localhost:0"), "", "",
|
||||
"", cluster.FilerType, pb.ServerAddress("localhost:0"), "", "",
|
||||
*pb.NewServiceDiscoveryFromMap(map[string]pb.ServerAddress{"m": masterAddress}),
|
||||
)
|
||||
|
||||
@@ -181,3 +182,206 @@ func TestDeleteEntryMetaAndDataDeletesCollectionWhenTheRequestIsCancelledMidDele
|
||||
t.Error("the bucket entry survived the delete")
|
||||
}
|
||||
}
|
||||
|
||||
func seedBucket(t *testing.T, store *hookedStore, path util.FullPath) {
|
||||
t.Helper()
|
||||
if err := store.InsertEntry(context.Background(), &Entry{
|
||||
FullPath: path,
|
||||
Attr: Attr{Mode: os.ModeDir | 0755},
|
||||
}); err != nil {
|
||||
t.Fatalf("seed bucket %s: %v", path, err)
|
||||
}
|
||||
}
|
||||
|
||||
// Two buckets resolving to one collection: deleting either must leave the
|
||||
// collection for the other. Previously the delete dropped the collection
|
||||
// named after the bucket regardless of where its data actually lived.
|
||||
func TestDeleteBucketKeepsSharedCollection(t *testing.T) {
|
||||
f, store, master := newFilerWithFakeMaster(t)
|
||||
f.FilerConf.SetLocationConf(&filer_pb.FilerConf_PathConf{
|
||||
LocationPrefix: "/buckets",
|
||||
Collection: "shared",
|
||||
})
|
||||
seedBucket(t, store, util.FullPath("/buckets/a"))
|
||||
seedBucket(t, store, util.FullPath("/buckets/b"))
|
||||
|
||||
if err := f.DeleteEntryMetaAndData(context.Background(), "/buckets/a", true, false, true, false, nil, 0); err != nil {
|
||||
t.Fatalf("DeleteEntryMetaAndData: %v", err)
|
||||
}
|
||||
|
||||
select {
|
||||
case call := <-master.calls:
|
||||
t.Fatalf("shared collection was deleted: %q", call.name)
|
||||
default:
|
||||
}
|
||||
if store.getEntry("/buckets/b") == nil {
|
||||
t.Error("the surviving bucket's entry is gone")
|
||||
}
|
||||
}
|
||||
|
||||
// A bucket named after a collection other buckets resolve to is still just a
|
||||
// bucket: deleting it must not take the shared collection down with it.
|
||||
func TestDeleteBucketNamedAfterSharedCollection(t *testing.T) {
|
||||
f, store, master := newFilerWithFakeMaster(t)
|
||||
f.FilerConf.SetLocationConf(&filer_pb.FilerConf_PathConf{
|
||||
LocationPrefix: "/buckets",
|
||||
Collection: "shared",
|
||||
})
|
||||
seedBucket(t, store, util.FullPath("/buckets/shared"))
|
||||
seedBucket(t, store, util.FullPath("/buckets/other"))
|
||||
|
||||
if err := f.DeleteEntryMetaAndData(context.Background(), "/buckets/shared", true, false, true, false, nil, 0); err != nil {
|
||||
t.Fatalf("DeleteEntryMetaAndData: %v", err)
|
||||
}
|
||||
|
||||
select {
|
||||
case call := <-master.calls:
|
||||
t.Fatalf("collection backing other buckets was deleted: %q", call.name)
|
||||
default:
|
||||
}
|
||||
}
|
||||
|
||||
// A rule pointing a non-bucket path at the same collection keeps it: the
|
||||
// collection serves files the bucket delete must not orphan.
|
||||
func TestDeleteBucketKeepsCollectionUsedByNonBucketPath(t *testing.T) {
|
||||
f, store, master := newFilerWithFakeMaster(t)
|
||||
f.FilerConf.SetLocationConf(&filer_pb.FilerConf_PathConf{
|
||||
LocationPrefix: "/buckets/a",
|
||||
Collection: "cold",
|
||||
})
|
||||
f.FilerConf.SetLocationConf(&filer_pb.FilerConf_PathConf{
|
||||
LocationPrefix: "/archives",
|
||||
Collection: "cold",
|
||||
})
|
||||
seedBucket(t, store, util.FullPath("/buckets/a"))
|
||||
|
||||
if err := f.DeleteEntryMetaAndData(context.Background(), "/buckets/a", true, false, true, false, nil, 0); err != nil {
|
||||
t.Fatalf("DeleteEntryMetaAndData: %v", err)
|
||||
}
|
||||
|
||||
select {
|
||||
case call := <-master.calls:
|
||||
t.Fatalf("collection used by /archives was deleted: %q", call.name)
|
||||
default:
|
||||
}
|
||||
}
|
||||
|
||||
// A broad prefix rule covering the whole tree keeps the collection even for a
|
||||
// lone bucket: the same collection backs non-bucket paths too.
|
||||
func TestDeleteBucketKeepsCollectionFromBroadRule(t *testing.T) {
|
||||
f, store, master := newFilerWithFakeMaster(t)
|
||||
f.FilerConf.SetLocationConf(&filer_pb.FilerConf_PathConf{
|
||||
LocationPrefix: "/",
|
||||
Collection: "everything",
|
||||
})
|
||||
seedBucket(t, store, util.FullPath("/buckets/a"))
|
||||
|
||||
if err := f.DeleteEntryMetaAndData(context.Background(), "/buckets/a", true, false, true, false, nil, 0); err != nil {
|
||||
t.Fatalf("DeleteEntryMetaAndData: %v", err)
|
||||
}
|
||||
|
||||
select {
|
||||
case call := <-master.calls:
|
||||
t.Fatalf("collection from a / rule was deleted: %q", call.name)
|
||||
default:
|
||||
}
|
||||
}
|
||||
|
||||
// A bucket resolving to the filer's default collection keeps it: rule-less
|
||||
// writes outside buckets land there too, so it is never one bucket's alone.
|
||||
func TestDeleteBucketKeepsDefaultCollection(t *testing.T) {
|
||||
f, store, master := newFilerWithFakeMaster(t)
|
||||
f.metaLogCollection = "everything"
|
||||
f.FilerConf.SetLocationConf(&filer_pb.FilerConf_PathConf{
|
||||
LocationPrefix: "/buckets/a",
|
||||
Collection: "everything",
|
||||
})
|
||||
seedBucket(t, store, util.FullPath("/buckets/a"))
|
||||
|
||||
if err := f.DeleteEntryMetaAndData(context.Background(), "/buckets/a", true, false, true, false, nil, 0); err != nil {
|
||||
t.Fatalf("DeleteEntryMetaAndData: %v", err)
|
||||
}
|
||||
|
||||
select {
|
||||
case call := <-master.calls:
|
||||
t.Fatalf("the filer's default collection was deleted: %q", call.name)
|
||||
default:
|
||||
}
|
||||
}
|
||||
|
||||
// A rule nested under a surviving bucket keeps the collection: the other
|
||||
// bucket resolves elsewhere at its root, but objects deeper inside it still
|
||||
// land in the shared collection.
|
||||
func TestDeleteBucketKeepsCollectionFromNestedSiblingRule(t *testing.T) {
|
||||
f, store, master := newFilerWithFakeMaster(t)
|
||||
f.FilerConf.SetLocationConf(&filer_pb.FilerConf_PathConf{
|
||||
LocationPrefix: "/buckets/a",
|
||||
Collection: "shared",
|
||||
})
|
||||
f.FilerConf.SetLocationConf(&filer_pb.FilerConf_PathConf{
|
||||
LocationPrefix: "/buckets/b/deep",
|
||||
Collection: "shared",
|
||||
})
|
||||
seedBucket(t, store, util.FullPath("/buckets/a"))
|
||||
seedBucket(t, store, util.FullPath("/buckets/b"))
|
||||
|
||||
if err := f.DeleteEntryMetaAndData(context.Background(), "/buckets/a", true, false, true, false, nil, 0); err != nil {
|
||||
t.Fatalf("DeleteEntryMetaAndData: %v", err)
|
||||
}
|
||||
|
||||
select {
|
||||
case call := <-master.calls:
|
||||
t.Fatalf("collection used under /buckets/b/deep was deleted: %q", call.name)
|
||||
default:
|
||||
}
|
||||
}
|
||||
|
||||
// A grouped gateway writes to <group>_<bucket> regardless of the storage
|
||||
// rules, so that is the collection the delete must drop -- and a rule-named
|
||||
// collection the bucket never used must survive.
|
||||
func TestDeleteBucketUnderFilerGroup(t *testing.T) {
|
||||
f, store, master := newFilerWithFakeMaster(t)
|
||||
f.MasterClient.FilerGroup = "tenant1"
|
||||
f.FilerConf.SetLocationConf(&filer_pb.FilerConf_PathConf{
|
||||
LocationPrefix: "/buckets/photos",
|
||||
Collection: "archive",
|
||||
})
|
||||
seedBucket(t, store, util.FullPath("/buckets/photos"))
|
||||
|
||||
if err := f.DeleteEntryMetaAndData(context.Background(), "/buckets/photos", true, false, true, false, nil, 0); err != nil {
|
||||
t.Fatalf("DeleteEntryMetaAndData: %v", err)
|
||||
}
|
||||
|
||||
select {
|
||||
case call := <-master.calls:
|
||||
if call.name != "tenant1_photos" {
|
||||
t.Fatalf("CollectionDelete = %q, want %q", call.name, "tenant1_photos")
|
||||
}
|
||||
case <-time.After(20 * time.Second):
|
||||
t.Fatal("CollectionDelete never reached the master")
|
||||
}
|
||||
}
|
||||
|
||||
// A collection only the deleted bucket resolves to is dropped under its real
|
||||
// name, so a dedicated custom collection does not leak its volumes.
|
||||
func TestDeleteBucketDeletesResolvedCollection(t *testing.T) {
|
||||
f, store, master := newFilerWithFakeMaster(t)
|
||||
f.FilerConf.SetLocationConf(&filer_pb.FilerConf_PathConf{
|
||||
LocationPrefix: "/buckets/only",
|
||||
Collection: "custom",
|
||||
})
|
||||
seedBucket(t, store, util.FullPath("/buckets/only"))
|
||||
|
||||
if err := f.DeleteEntryMetaAndData(context.Background(), "/buckets/only", true, false, true, false, nil, 0); err != nil {
|
||||
t.Fatalf("DeleteEntryMetaAndData: %v", err)
|
||||
}
|
||||
|
||||
select {
|
||||
case call := <-master.calls:
|
||||
if call.name != "custom" {
|
||||
t.Fatalf("CollectionDelete = %q, want %q", call.name, "custom")
|
||||
}
|
||||
case <-time.After(20 * time.Second):
|
||||
t.Fatal("CollectionDelete never reached the master")
|
||||
}
|
||||
}
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user