mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-10-09 16:15:51 +00:00
Compare commits
115
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
1cb659019e | ||
|
|
9ca9dd2379 | ||
|
|
87474c2f21 | ||
|
|
b8049bc633 | ||
|
|
23d424248d | ||
|
|
49ee13635b | ||
|
|
8bd5ec37d1 | ||
|
|
1971fb8cb0 | ||
|
|
8c3695fd4a | ||
|
|
3a3d513cb8 | ||
|
|
86ee10a080 | ||
|
|
909f5cabc7 | ||
|
|
f740210235 | ||
|
|
32df246a81 | ||
|
|
d3b8030a69 | ||
|
|
9bafeb6139 | ||
|
|
74b520113e | ||
|
|
83a929720a | ||
|
|
88c873ecd4 | ||
|
|
93666c90e9 | ||
|
|
3967ca23be | ||
|
|
fcc2ea61d3 | ||
|
|
ba5b14b457 | ||
|
|
7dc3835b02 | ||
|
|
c858e01a09 | ||
|
|
cd5013f116 | ||
|
|
624deaf3a4 | ||
|
|
7bb0a1c127 | ||
|
|
af6f69740c | ||
|
|
95248f7492 | ||
|
|
23241cf0f1 | ||
|
|
60893c5ef3 | ||
|
|
9e06e1d0f9 | ||
|
|
742b2f5896 | ||
|
|
902a12fd6f | ||
|
|
d850f36513 | ||
|
|
eed3c27d15 | ||
|
|
fdd8bd9478 | ||
|
|
99cf7a66df | ||
|
|
2a97e08caa | ||
|
|
ab8b34720a | ||
|
|
0b5fff2ccd | ||
|
|
28862c866e | ||
|
|
bc06505b40 | ||
|
|
e9a464840c | ||
|
|
d8a189f07f | ||
|
|
2d25c39da4 | ||
|
|
8dcdb70594 | ||
|
|
f5f1dcbd8c | ||
|
|
3431bdcb74 | ||
|
|
12fd60f92e | ||
|
|
da087f77b3 | ||
|
|
eb3bbfeb1f | ||
|
|
a02c0024e5 | ||
|
|
b77d954f55 | ||
|
|
7658305c76 | ||
|
|
627b5e9d59 | ||
|
|
368b2035b2 | ||
|
|
b58d52ac16 | ||
|
|
e482e67971 | ||
|
|
ef4c9d9178 | ||
|
|
70c3adb983 | ||
|
|
b77431c142 | ||
|
|
50b388771a | ||
|
|
44115c1051 | ||
|
|
c69bb10407 | ||
|
|
68f0793b6f | ||
|
|
40f77503d0 | ||
|
|
d9d7d0be74 | ||
|
|
b3be2f5449 | ||
|
|
4a2879abad | ||
|
|
2a70532d0d | ||
|
|
d2c470af1b | ||
|
|
115756dd41 | ||
|
|
d9d5fab35b | ||
|
|
d8545997c2 | ||
|
|
9121177f48 | ||
|
|
9ba3d473d7 | ||
|
|
e36b01c18f | ||
|
|
dfa75ce231 | ||
|
|
61d588e455 | ||
|
|
a3afe4460b | ||
|
|
a0ddf22f17 | ||
|
|
863fec6c3f | ||
|
|
46ce2c45a2 | ||
|
|
51eb5333d3 | ||
|
|
69cc2869ad | ||
|
|
68ec8ca655 | ||
|
|
e931cccc7b | ||
|
|
c80664ec21 | ||
|
|
71a8c77a36 | ||
|
|
9c8d3b6a81 | ||
|
|
36c97344ef | ||
|
|
74038e1b14 | ||
|
|
cf0dba334c | ||
|
|
c167af541e | ||
|
|
0f85d005ad | ||
|
|
173adbc291 | ||
|
|
fa3bd5b5a7 | ||
|
|
5ebc9c9f4b | ||
|
|
3b10e43d5d | ||
|
|
9d06f2c378 | ||
|
|
9d4270f118 | ||
|
|
3b8931c2f6 | ||
|
|
8d8a25b1cf | ||
|
|
c58795354a | ||
|
|
8d2c0273bd | ||
|
|
f710b6003a | ||
|
|
f3caf6e7da | ||
|
|
004fc32503 | ||
|
|
641fc8b031 | ||
|
|
228500fe37 | ||
|
|
7ebf2ebac3 | ||
|
|
cc8364a03e | ||
|
|
c3f4d799b5 |
@@ -27,7 +27,7 @@ jobs:
|
||||
|
||||
# Initializes the CodeQL tools for scanning.
|
||||
- name: Initialize CodeQL
|
||||
uses: github/codeql-action/init@v4.37.6
|
||||
uses: github/codeql-action/init@v4.37.9
|
||||
# Override language selection by uncommenting this and choosing your languages
|
||||
with:
|
||||
languages: go
|
||||
@@ -35,7 +35,7 @@ jobs:
|
||||
# Autobuild attempts to build any compiled languages (C/C++, C#, or Java).
|
||||
# If this step fails, then you should remove it and run the build manually (see below).
|
||||
- name: Autobuild
|
||||
uses: github/codeql-action/autobuild@v4.37.6
|
||||
uses: github/codeql-action/autobuild@v4.37.9
|
||||
|
||||
# ℹ️ Command-line programs to run using the OS shell.
|
||||
# 📚 See https://docs.github.com/en/actions/using-workflows/workflow-syntax-for-github-actions#jobsjob_idstepsrun
|
||||
@@ -49,4 +49,4 @@ jobs:
|
||||
# make release
|
||||
|
||||
- name: Perform CodeQL Analysis
|
||||
uses: github/codeql-action/analyze@v4.37.6
|
||||
uses: github/codeql-action/analyze@v4.37.9
|
||||
|
||||
@@ -6,6 +6,7 @@ on:
|
||||
paths:
|
||||
- 'weed/**'
|
||||
- 'seaweed-volume/**'
|
||||
- 'seaweed-worker/**'
|
||||
- 'docker/**'
|
||||
- 'go.mod'
|
||||
- 'go.sum'
|
||||
@@ -16,7 +17,7 @@ permissions:
|
||||
|
||||
jobs:
|
||||
|
||||
# ── Pre-build Rust volume server binaries natively ──────────────────
|
||||
# ── Pre-build the Rust binaries natively ────────────────────────────
|
||||
build-rust-binaries:
|
||||
runs-on: ubuntu-22.04
|
||||
strategy:
|
||||
@@ -55,10 +56,26 @@ jobs:
|
||||
~/.cargo/registry
|
||||
~/.cargo/git
|
||||
seaweed-volume/target
|
||||
key: rust-docker-dev-${{ matrix.target }}-${{ hashFiles('seaweed-volume/Cargo.lock') }}
|
||||
seaweed-worker/target/${{ matrix.target }}/release
|
||||
key: rust-docker-dev-${{ matrix.target }}-${{ hashFiles('seaweed-volume/Cargo.lock', 'seaweed-worker/Cargo.lock') }}
|
||||
restore-keys: |
|
||||
rust-docker-dev-${{ matrix.target }}-
|
||||
|
||||
# lance's build scripts compile their own protos and look for a protoc.
|
||||
# Point them at the one protoc-bin-vendored ships, which seaweed-worker's
|
||||
# own build already uses, so no job depends on a system package and every
|
||||
# build sees the same version.
|
||||
- name: Use the vendored protoc
|
||||
run: |
|
||||
cd seaweed-worker
|
||||
cargo fetch
|
||||
# The version from the lock, not whatever else a restored cache holds.
|
||||
version=$(awk '/^name = "protoc-bin-vendored-linux-x86_64"$/{found=1; next} found && /^version = /{gsub(/"/,"",$3); print $3; exit}' Cargo.lock)
|
||||
test -n "$version" || { echo "protoc-bin-vendored-linux-x86_64 is not in Cargo.lock" >&2; exit 1; }
|
||||
protoc=$(find ~/.cargo/registry/src -path "*protoc-bin-vendored-linux-x86_64-$version/bin/protoc" | head -1)
|
||||
test -x "$protoc" || { echo "no vendored protoc $version in the registry" >&2; exit 1; }
|
||||
echo "PROTOC=$protoc" >> "$GITHUB_ENV"
|
||||
|
||||
- name: Build normal variant
|
||||
env:
|
||||
SEAWEEDFS_COMMIT: ${{ github.sha }}
|
||||
@@ -67,11 +84,19 @@ jobs:
|
||||
cargo build --release --target ${{ matrix.target }} --no-default-features
|
||||
cp target/${{ matrix.target }}/release/weed-volume ../weed-volume-normal-${{ matrix.arch }}
|
||||
|
||||
- name: Build the Rust maintenance worker
|
||||
run: |
|
||||
cd seaweed-worker
|
||||
cargo build --release -p weed-lance-worker --target ${{ matrix.target }}
|
||||
cp target/${{ matrix.target }}/release/weed-worker ../weed-worker-${{ matrix.arch }}
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: rust-volume-${{ matrix.arch }}
|
||||
path: weed-volume-normal-${{ matrix.arch }}
|
||||
name: rust-bins-${{ matrix.arch }}
|
||||
path: |
|
||||
weed-volume-normal-${{ matrix.arch }}
|
||||
weed-worker-${{ matrix.arch }}
|
||||
|
||||
build-dev-containers:
|
||||
needs: [build-rust-binaries]
|
||||
@@ -84,7 +109,7 @@ jobs:
|
||||
- name: Download pre-built Rust binaries
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
pattern: rust-volume-*
|
||||
pattern: rust-bins-*
|
||||
merge-multiple: true
|
||||
path: ./rust-bins
|
||||
|
||||
@@ -98,7 +123,16 @@ jobs:
|
||||
echo "Placed pre-built Rust binary for ${arch}"
|
||||
fi
|
||||
done
|
||||
mkdir -p docker/weed-worker-prebuilt
|
||||
for arch in amd64 arm64; do
|
||||
src="./rust-bins/weed-worker-${arch}"
|
||||
if [ -f "$src" ]; then
|
||||
cp "$src" "docker/weed-worker-prebuilt/weed-worker-${arch}"
|
||||
echo "Placed pre-built Rust worker for ${arch}"
|
||||
fi
|
||||
done
|
||||
ls -la docker/weed-volume-prebuilt/
|
||||
ls -la docker/weed-worker-prebuilt/
|
||||
|
||||
- name: Docker meta
|
||||
id: docker_meta
|
||||
|
||||
@@ -59,7 +59,7 @@ jobs:
|
||||
echo "publish=true" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
# ── Pre-build Rust volume server binaries natively ──────────────────
|
||||
# ── Pre-build the Rust binaries natively ────────────────────────────
|
||||
build-rust-binaries:
|
||||
runs-on: ubuntu-22.04
|
||||
strategy:
|
||||
@@ -100,10 +100,26 @@ jobs:
|
||||
~/.cargo/registry
|
||||
~/.cargo/git
|
||||
seaweed-volume/target
|
||||
key: rust-docker-${{ matrix.target }}-${{ hashFiles('seaweed-volume/Cargo.lock') }}
|
||||
seaweed-worker/target/${{ matrix.target }}/release
|
||||
key: rust-docker-${{ matrix.target }}-${{ hashFiles('seaweed-volume/Cargo.lock', 'seaweed-worker/Cargo.lock') }}
|
||||
restore-keys: |
|
||||
rust-docker-${{ matrix.target }}-
|
||||
|
||||
# lance's build scripts compile their own protos and look for a protoc.
|
||||
# Point them at the one protoc-bin-vendored ships, which seaweed-worker's
|
||||
# own build already uses, so no job depends on a system package and every
|
||||
# build sees the same version.
|
||||
- name: Use the vendored protoc
|
||||
run: |
|
||||
cd seaweed-worker
|
||||
cargo fetch
|
||||
# The version from the lock, not whatever else a restored cache holds.
|
||||
version=$(awk '/^name = "protoc-bin-vendored-linux-x86_64"$/{found=1; next} found && /^version = /{gsub(/"/,"",$3); print $3; exit}' Cargo.lock)
|
||||
test -n "$version" || { echo "protoc-bin-vendored-linux-x86_64 is not in Cargo.lock" >&2; exit 1; }
|
||||
protoc=$(find ~/.cargo/registry/src -path "*protoc-bin-vendored-linux-x86_64-$version/bin/protoc" | head -1)
|
||||
test -x "$protoc" || { echo "no vendored protoc $version in the registry" >&2; exit 1; }
|
||||
echo "PROTOC=$protoc" >> "$GITHUB_ENV"
|
||||
|
||||
- name: Build large-disk variant
|
||||
env:
|
||||
SEAWEEDFS_COMMIT: ${{ github.sha }}
|
||||
@@ -120,13 +136,20 @@ jobs:
|
||||
cargo build --release --target ${{ matrix.target }} --no-default-features
|
||||
cp target/${{ matrix.target }}/release/weed-volume ../weed-volume-normal-${{ matrix.arch }}
|
||||
|
||||
- name: Build the Rust maintenance worker
|
||||
run: |
|
||||
cd seaweed-worker
|
||||
cargo build --release -p weed-lance-worker --target ${{ matrix.target }}
|
||||
cp target/${{ matrix.target }}/release/weed-worker ../weed-worker-${{ matrix.arch }}
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: rust-volume-${{ matrix.arch }}
|
||||
name: rust-bins-${{ matrix.arch }}
|
||||
path: |
|
||||
weed-volume-large-disk-${{ matrix.arch }}
|
||||
weed-volume-normal-${{ matrix.arch }}
|
||||
weed-worker-${{ matrix.arch }}
|
||||
|
||||
build:
|
||||
needs: [setup, build-rust-binaries]
|
||||
@@ -174,7 +197,7 @@ jobs:
|
||||
- name: Download pre-built Rust binaries
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
pattern: rust-volume-*
|
||||
pattern: rust-bins-*
|
||||
merge-multiple: true
|
||||
path: ./rust-bins
|
||||
|
||||
@@ -188,7 +211,16 @@ jobs:
|
||||
echo "Placed pre-built Rust binary for ${arch}"
|
||||
fi
|
||||
done
|
||||
mkdir -p docker/weed-worker-prebuilt
|
||||
for arch in amd64 arm64; do
|
||||
src="./rust-bins/weed-worker-${arch}"
|
||||
if [ -f "$src" ]; then
|
||||
cp "$src" "docker/weed-worker-prebuilt/weed-worker-${arch}"
|
||||
echo "Placed pre-built Rust worker for ${arch}"
|
||||
fi
|
||||
done
|
||||
ls -la docker/weed-volume-prebuilt/
|
||||
ls -la docker/weed-worker-prebuilt/
|
||||
|
||||
- name: Docker meta
|
||||
id: docker_meta
|
||||
@@ -286,7 +318,7 @@ jobs:
|
||||
if: needs.setup.outputs.publish != 'true'
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
pattern: rust-volume-*
|
||||
pattern: rust-bins-*
|
||||
merge-multiple: true
|
||||
path: ./rust-bins
|
||||
- name: Place Rust binaries in Docker context for local scan
|
||||
@@ -304,7 +336,16 @@ jobs:
|
||||
echo "Placed pre-built Rust binary for ${arch}"
|
||||
fi
|
||||
done
|
||||
mkdir -p docker/weed-worker-prebuilt
|
||||
for arch in amd64 arm64; do
|
||||
src="./rust-bins/weed-worker-${arch}"
|
||||
if [ -f "$src" ]; then
|
||||
cp "$src" "docker/weed-worker-prebuilt/weed-worker-${arch}"
|
||||
echo "Placed pre-built Rust worker for ${arch}"
|
||||
fi
|
||||
done
|
||||
ls -la docker/weed-volume-prebuilt/
|
||||
ls -la docker/weed-worker-prebuilt/
|
||||
- name: Create BuildKit config for local scan build
|
||||
if: needs.setup.outputs.publish != 'true'
|
||||
run: |
|
||||
@@ -364,7 +405,7 @@ jobs:
|
||||
output: trivy-results.sarif
|
||||
exit-code: '0'
|
||||
- name: Upload Trivy scan results to GitHub Security
|
||||
uses: github/codeql-action/upload-sarif@v4.37.6
|
||||
uses: github/codeql-action/upload-sarif@v4.37.9
|
||||
if: always()
|
||||
with:
|
||||
sarif_file: trivy-results.sarif
|
||||
|
||||
@@ -42,9 +42,10 @@ concurrency:
|
||||
|
||||
jobs:
|
||||
|
||||
# ── Pre-build Rust volume server binaries natively ──────────────────
|
||||
# Cross-compiles for amd64 and arm64 without QEMU, turning a 5-hour
|
||||
# emulated cargo build into ~15 minutes of native compilation.
|
||||
# ── Pre-build the Rust binaries natively ────────────────────────────
|
||||
# The volume server and the Rust maintenance worker, cross-compiled for
|
||||
# amd64 and arm64 without QEMU, turning a 5-hour emulated cargo build into
|
||||
# ~15 minutes of native compilation.
|
||||
build-rust-binaries:
|
||||
runs-on: ubuntu-22.04
|
||||
strategy:
|
||||
@@ -83,10 +84,26 @@ jobs:
|
||||
~/.cargo/registry
|
||||
~/.cargo/git
|
||||
seaweed-volume/target
|
||||
key: rust-docker-${{ matrix.target }}-${{ hashFiles('seaweed-volume/Cargo.lock') }}
|
||||
seaweed-worker/target/${{ matrix.target }}/release
|
||||
key: rust-docker-${{ matrix.target }}-${{ hashFiles('seaweed-volume/Cargo.lock', 'seaweed-worker/Cargo.lock') }}
|
||||
restore-keys: |
|
||||
rust-docker-${{ matrix.target }}-
|
||||
|
||||
# lance's build scripts compile their own protos and look for a protoc.
|
||||
# Point them at the one protoc-bin-vendored ships, which seaweed-worker's
|
||||
# own build already uses, so no job depends on a system package and every
|
||||
# build sees the same version.
|
||||
- name: Use the vendored protoc
|
||||
run: |
|
||||
cd seaweed-worker
|
||||
cargo fetch
|
||||
# The version from the lock, not whatever else a restored cache holds.
|
||||
version=$(awk '/^name = "protoc-bin-vendored-linux-x86_64"$/{found=1; next} found && /^version = /{gsub(/"/,"",$3); print $3; exit}' Cargo.lock)
|
||||
test -n "$version" || { echo "protoc-bin-vendored-linux-x86_64 is not in Cargo.lock" >&2; exit 1; }
|
||||
protoc=$(find ~/.cargo/registry/src -path "*protoc-bin-vendored-linux-x86_64-$version/bin/protoc" | head -1)
|
||||
test -x "$protoc" || { echo "no vendored protoc $version in the registry" >&2; exit 1; }
|
||||
echo "PROTOC=$protoc" >> "$GITHUB_ENV"
|
||||
|
||||
- name: Build large-disk variant
|
||||
env:
|
||||
SEAWEEDFS_COMMIT: ${{ github.sha }}
|
||||
@@ -103,13 +120,20 @@ jobs:
|
||||
cargo build --release --target ${{ matrix.target }} --no-default-features
|
||||
cp target/${{ matrix.target }}/release/weed-volume ../weed-volume-normal-${{ matrix.arch }}
|
||||
|
||||
- name: Build the Rust maintenance worker
|
||||
run: |
|
||||
cd seaweed-worker
|
||||
cargo build --release -p weed-lance-worker --target ${{ matrix.target }}
|
||||
cp target/${{ matrix.target }}/release/weed-worker ../weed-worker-${{ matrix.arch }}
|
||||
|
||||
- name: Upload artifacts
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: rust-volume-${{ matrix.arch }}
|
||||
name: rust-bins-${{ matrix.arch }}
|
||||
path: |
|
||||
weed-volume-large-disk-${{ matrix.arch }}
|
||||
weed-volume-normal-${{ matrix.arch }}
|
||||
weed-worker-${{ matrix.arch }}
|
||||
|
||||
# One job per (variant, platform) on a native runner, pushed by digest;
|
||||
# the merge job stitches the digests into one multi-arch tag.
|
||||
@@ -155,7 +179,7 @@ jobs:
|
||||
if: github.event_name != 'workflow_dispatch' || github.event.inputs.variant == 'all' || github.event.inputs.variant == matrix.variant
|
||||
uses: actions/download-artifact@v8
|
||||
with:
|
||||
pattern: rust-volume-*
|
||||
pattern: rust-bins-*
|
||||
merge-multiple: true
|
||||
path: ./rust-bins
|
||||
|
||||
@@ -170,7 +194,16 @@ jobs:
|
||||
echo "Placed pre-built Rust binary for ${arch}"
|
||||
fi
|
||||
done
|
||||
mkdir -p docker/weed-worker-prebuilt
|
||||
for arch in amd64 arm64; do
|
||||
src="./rust-bins/weed-worker-${arch}"
|
||||
if [ -f "$src" ]; then
|
||||
cp "$src" "docker/weed-worker-prebuilt/weed-worker-${arch}"
|
||||
echo "Placed pre-built Rust worker for ${arch}"
|
||||
fi
|
||||
done
|
||||
ls -la docker/weed-volume-prebuilt/
|
||||
ls -la docker/weed-worker-prebuilt/
|
||||
|
||||
- name: Free Disk Space
|
||||
if: github.event_name != 'workflow_dispatch' || github.event.inputs.variant == 'all' || github.event.inputs.variant == matrix.variant
|
||||
@@ -397,7 +430,7 @@ jobs:
|
||||
|
||||
- name: Upload Trivy scan results to GitHub Security
|
||||
if: always()
|
||||
uses: github/codeql-action/upload-sarif@v4.37.6
|
||||
uses: github/codeql-action/upload-sarif@v4.37.9
|
||||
with:
|
||||
sarif_file: trivy-results.sarif
|
||||
category: trivy-${{ matrix.variant }}
|
||||
|
||||
@@ -3,10 +3,10 @@ name: "helm: lint and test charts"
|
||||
on:
|
||||
push:
|
||||
branches: [ master ]
|
||||
paths: ['k8s/**']
|
||||
paths: ['k8s/**', '.github/workflows/helm_ci.yml']
|
||||
pull_request:
|
||||
branches: [ master ]
|
||||
paths: ['k8s/**']
|
||||
paths: ['k8s/**', '.github/workflows/helm_ci.yml']
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
@@ -1549,11 +1549,37 @@ jobs:
|
||||
|
||||
echo "All template rendering tests passed!"
|
||||
|
||||
- name: Resolve an image tag that is published
|
||||
run: |
|
||||
set -e
|
||||
# A release bumps appVersion on master ~40 minutes before the container
|
||||
# build publishes that tag, and this workflow runs on the bump commit.
|
||||
# Install the last released image for the length of that window instead
|
||||
# of failing on ImagePullBackOff.
|
||||
IMAGE=$(helm template test k8s/charts/seaweedfs \
|
||||
-s templates/master/master-statefulset.yaml | awk '$1 == "image:" {print $2; exit}')
|
||||
REPO=${IMAGE%:*}
|
||||
TAG=${IMAGE##*:}
|
||||
# Anything but a published tag installs latest, which between releases
|
||||
# is the same digest as the chart's own appVersion, so a registry blip
|
||||
# costs nothing while failing the job on one would cost a red build.
|
||||
STATUS=$(curl -sSL --connect-timeout 5 --max-time 15 -o /dev/null \
|
||||
-w '%{http_code}' "https://hub.docker.com/v2/repositories/$REPO/tags/$TAG" || true)
|
||||
if [ "$STATUS" = 200 ]; then
|
||||
echo "installing $IMAGE"
|
||||
else
|
||||
echo "$IMAGE is unavailable (HTTP ${STATUS:-none}), installing $REPO:latest"
|
||||
TAG=latest
|
||||
fi
|
||||
echo "IMAGE_TAG=$TAG" >> $GITHUB_ENV
|
||||
|
||||
- name: Create kind cluster
|
||||
uses: helm/kind-action@v1.14.0
|
||||
|
||||
- name: Run chart-testing (install)
|
||||
run: ct install --target-branch ${{ github.event.repository.default_branch }} --all --chart-dirs k8s/charts
|
||||
run: |
|
||||
ct install --target-branch ${{ github.event.repository.default_branch }} --all --chart-dirs k8s/charts \
|
||||
--helm-extra-set-args "--set=image.tag=$IMAGE_TAG"
|
||||
|
||||
- name: Verify SFTP host key secret lifecycle
|
||||
run: |
|
||||
@@ -1561,7 +1587,7 @@ jobs:
|
||||
CHART_DIR="k8s/charts/seaweedfs"
|
||||
NS="sftp-hostkey"
|
||||
SECRET="hk-seaweedfs-sftp-ssh-secret"
|
||||
SFTP_ARGS="--set sftp.enabled=true --set master.enabled=false --set volume.enabled=false --set filer.enabled=false"
|
||||
SFTP_ARGS="--set image.tag=$IMAGE_TAG --set sftp.enabled=true --set master.enabled=false --set volume.enabled=false --set filer.enabled=false"
|
||||
kubectl create namespace "$NS"
|
||||
|
||||
echo "=== install generates a host key, upgrade keeps it ==="
|
||||
@@ -1638,6 +1664,7 @@ jobs:
|
||||
# release if the hook Job does not finish, so a clean install is the
|
||||
# assertion.
|
||||
helm install np $CHART_DIR -n "$NS" --wait --timeout 8m \
|
||||
--set image.tag=$IMAGE_TAG \
|
||||
--set s3.enabled=true \
|
||||
--set s3.createBuckets[0].name=testbucket \
|
||||
--set networkPolicy.enabled=true \
|
||||
|
||||
@@ -34,7 +34,7 @@ jobs:
|
||||
id: go
|
||||
|
||||
- name: Set up Java
|
||||
uses: actions/setup-java@v5
|
||||
uses: actions/setup-java@v6
|
||||
with:
|
||||
java-version: ${{ matrix.java }}
|
||||
distribution: 'temurin'
|
||||
|
||||
@@ -73,7 +73,7 @@ jobs:
|
||||
echo "version=${VERSION}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Set up JDK 17
|
||||
uses: actions/setup-java@v5
|
||||
uses: actions/setup-java@v6
|
||||
with:
|
||||
java-version: '17'
|
||||
distribution: 'temurin'
|
||||
|
||||
@@ -26,7 +26,7 @@ jobs:
|
||||
uses: actions/checkout@v7
|
||||
|
||||
- name: Set up Java
|
||||
uses: actions/setup-java@v5
|
||||
uses: actions/setup-java@v6
|
||||
with:
|
||||
java-version: ${{ matrix.java }}
|
||||
distribution: 'temurin'
|
||||
|
||||
@@ -131,7 +131,7 @@ jobs:
|
||||
|
||||
- name: Results
|
||||
if: always()
|
||||
uses: actions/upload-artifact@v4
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: results-windows
|
||||
path: C:\results-*.json
|
||||
@@ -196,7 +196,7 @@ jobs:
|
||||
|
||||
- name: Results
|
||||
if: always()
|
||||
uses: actions/upload-artifact@v4
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: results-linux
|
||||
path: /tmp/results-linux.json
|
||||
|
||||
@@ -0,0 +1,82 @@
|
||||
name: "Rust Plugin Worker Tests"
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
branches: [ master ]
|
||||
paths:
|
||||
- 'seaweed-worker/**'
|
||||
- 'weed/pb/plugin.proto'
|
||||
- '.github/workflows/rust-worker-tests.yml'
|
||||
push:
|
||||
branches: [ master, main ]
|
||||
paths:
|
||||
- 'seaweed-worker/**'
|
||||
- 'weed/pb/plugin.proto'
|
||||
- '.github/workflows/rust-worker-tests.yml'
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.head_ref || github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
rust-worker-build:
|
||||
name: Rust Plugin Worker Build and Unit Tests
|
||||
runs-on: ubuntu-22.04
|
||||
timeout-minutes: 45
|
||||
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v7
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- name: Install Rust toolchain
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
|
||||
# cargo tracks its own inputs but not the runner's C toolchain, so a cached
|
||||
# target/ can carry C objects built against a different glibc than we link against.
|
||||
- name: Fingerprint build toolchain
|
||||
id: toolchain
|
||||
run: echo "fingerprint=$(getconf GNU_LIBC_VERSION | tr ' ' '-')-rustc-$(rustc -V | awk '{print $2}')" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Cache cargo registry and target
|
||||
uses: actions/cache@v6
|
||||
with:
|
||||
path: |
|
||||
~/.cargo/registry
|
||||
~/.cargo/git
|
||||
seaweed-worker/target/release
|
||||
key: rust-worker-${{ steps.toolchain.outputs.fingerprint }}-${{ hashFiles('seaweed-worker/Cargo.lock') }}
|
||||
restore-keys: |
|
||||
rust-worker-${{ steps.toolchain.outputs.fingerprint }}-
|
||||
|
||||
# lance's build scripts compile their own protos and look for a protoc.
|
||||
# Point them at the one protoc-bin-vendored ships, which seaweed-worker's
|
||||
# own build already uses, so no job depends on a system package and every
|
||||
# build sees the same version.
|
||||
- name: Use the vendored protoc
|
||||
run: |
|
||||
cd seaweed-worker
|
||||
cargo fetch
|
||||
# The version from the lock, not whatever else a restored cache holds.
|
||||
version=$(awk '/^name = "protoc-bin-vendored-linux-x86_64"$/{found=1; next} found && /^version = /{gsub(/"/,"",$3); print $3; exit}' Cargo.lock)
|
||||
test -n "$version" || { echo "protoc-bin-vendored-linux-x86_64 is not in Cargo.lock" >&2; exit 1; }
|
||||
protoc=$(find ~/.cargo/registry/src -path "*protoc-bin-vendored-linux-x86_64-$version/bin/protoc" | head -1)
|
||||
test -x "$protoc" || { echo "no vendored protoc $version in the registry" >&2; exit 1; }
|
||||
echo "PROTOC=$protoc" >> "$GITHUB_ENV"
|
||||
|
||||
# The release profile is what ships, and it is where the release and the
|
||||
# container builds would otherwise discover a break for the first time.
|
||||
- name: Build the plugin workers
|
||||
run: cd seaweed-worker && cargo build --release
|
||||
|
||||
# The tests that need a live gateway skip themselves without one, the way
|
||||
# the Go integration tests skip without Docker; the lifecycle suite in
|
||||
# test/s3tables/lifecycle is what runs them against a real cluster.
|
||||
# Release, so this reuses the build above rather than compiling lance,
|
||||
# arrow and datafusion a second time in another profile.
|
||||
- name: Run unit tests
|
||||
run: cd seaweed-worker && cargo test --release --workspace
|
||||
@@ -1,4 +1,4 @@
|
||||
name: "rust: build versioned volume server binaries"
|
||||
name: "rust: build versioned binaries"
|
||||
|
||||
on:
|
||||
push:
|
||||
@@ -113,6 +113,102 @@ jobs:
|
||||
weed-volume_${{ matrix.asset_suffix }}.tar.gz
|
||||
weed-volume_${{ matrix.asset_suffix }}.tar.gz.md5
|
||||
|
||||
# The Rust maintenance worker: Linux only, because it runs beside the cluster
|
||||
# it maintains rather than on a laptop, and its dependency tree (lance, arrow,
|
||||
# datafusion) makes every extra target an expensive build.
|
||||
build-rust-worker-linux:
|
||||
permissions:
|
||||
contents: write
|
||||
runs-on: ubuntu-22.04
|
||||
strategy:
|
||||
matrix:
|
||||
include:
|
||||
- target: x86_64-unknown-linux-gnu
|
||||
asset_suffix: linux_amd64
|
||||
- target: aarch64-unknown-linux-gnu
|
||||
asset_suffix: linux_arm64
|
||||
cross: true
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v7
|
||||
with:
|
||||
# The upload step is handed a token explicitly; a cargo build script
|
||||
# should not find another one sitting in the checkout's git config.
|
||||
persist-credentials: false
|
||||
|
||||
- name: Install Rust toolchain
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
with:
|
||||
targets: ${{ matrix.target }}
|
||||
|
||||
- name: Install cross-compilation tools
|
||||
if: matrix.cross
|
||||
run: |
|
||||
sudo dpkg --add-architecture arm64
|
||||
sudo sed -i 's/^deb /deb [arch=amd64] /' /etc/apt/sources.list
|
||||
echo "deb [arch=arm64] http://ports.ubuntu.com/ jammy main restricted universe multiverse" | sudo tee /etc/apt/sources.list.d/arm64.list
|
||||
echo "deb [arch=arm64] http://ports.ubuntu.com/ jammy-updates main restricted universe multiverse" | sudo tee -a /etc/apt/sources.list.d/arm64.list
|
||||
sudo apt-get update
|
||||
sudo apt-get install -y gcc-aarch64-linux-gnu
|
||||
echo "CARGO_TARGET_AARCH64_UNKNOWN_LINUX_GNU_LINKER=aarch64-linux-gnu-gcc" >> "$GITHUB_ENV"
|
||||
|
||||
- name: Cache cargo registry and target
|
||||
uses: actions/cache@v6
|
||||
with:
|
||||
path: |
|
||||
~/.cargo/registry
|
||||
~/.cargo/git
|
||||
seaweed-worker/target/${{ matrix.target }}/release
|
||||
key: rust-worker-release-${{ matrix.target }}-${{ hashFiles('seaweed-worker/Cargo.lock') }}
|
||||
restore-keys: |
|
||||
rust-worker-release-${{ matrix.target }}-
|
||||
|
||||
# lance's build scripts compile their own protos and look for a protoc.
|
||||
# Point them at the one protoc-bin-vendored ships, which seaweed-worker's
|
||||
# own build already uses, so no job depends on a system package and every
|
||||
# build sees the same version.
|
||||
- name: Use the vendored protoc
|
||||
run: |
|
||||
cd seaweed-worker
|
||||
cargo fetch
|
||||
# The version from the lock, not whatever else a restored cache holds.
|
||||
version=$(awk '/^name = "protoc-bin-vendored-linux-x86_64"$/{found=1; next} found && /^version = /{gsub(/"/,"",$3); print $3; exit}' Cargo.lock)
|
||||
test -n "$version" || { echo "protoc-bin-vendored-linux-x86_64 is not in Cargo.lock" >&2; exit 1; }
|
||||
protoc=$(find ~/.cargo/registry/src -path "*protoc-bin-vendored-linux-x86_64-$version/bin/protoc" | head -1)
|
||||
test -x "$protoc" || { echo "no vendored protoc $version in the registry" >&2; exit 1; }
|
||||
echo "PROTOC=$protoc" >> "$GITHUB_ENV"
|
||||
|
||||
- name: Build the Rust maintenance worker
|
||||
run: |
|
||||
cd seaweed-worker
|
||||
cargo build --release -p weed-lance-worker --target ${{ matrix.target }}
|
||||
|
||||
- name: Package binary
|
||||
run: |
|
||||
cp seaweed-worker/target/${{ matrix.target }}/release/weed-worker weed-worker
|
||||
tar czf weed-worker_${{ matrix.asset_suffix }}.tar.gz weed-worker
|
||||
rm weed-worker
|
||||
md5sum weed-worker_${{ matrix.asset_suffix }}.tar.gz > weed-worker_${{ matrix.asset_suffix }}.tar.gz.md5
|
||||
|
||||
- name: Upload release assets
|
||||
if: startsWith(github.ref, 'refs/tags/')
|
||||
uses: softprops/action-gh-release@v3
|
||||
with:
|
||||
files: |
|
||||
weed-worker_${{ matrix.asset_suffix }}.tar.gz
|
||||
weed-worker_${{ matrix.asset_suffix }}.tar.gz.md5
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
|
||||
- name: Upload artifacts
|
||||
if: ${{ !startsWith(github.ref, 'refs/tags/') }}
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: rust-worker-${{ matrix.asset_suffix }}
|
||||
path: |
|
||||
weed-worker_${{ matrix.asset_suffix }}.tar.gz
|
||||
weed-worker_${{ matrix.asset_suffix }}.tar.gz.md5
|
||||
|
||||
build-rust-volume-darwin:
|
||||
permissions:
|
||||
contents: write
|
||||
|
||||
@@ -64,6 +64,14 @@ jobs:
|
||||
echo "=== Running S3 Empty Directory Marker Tests ==="
|
||||
go test -v -timeout=180s -run TestS3ListObjectsEmptyDirectoryMarkers ./...
|
||||
|
||||
- name: Run S3 Prefix Object Tests
|
||||
timeout-minutes: 15
|
||||
working-directory: test/s3/normal
|
||||
run: |
|
||||
set -x
|
||||
echo "=== Running S3 Prefix Object Tests ==="
|
||||
go test -v -timeout=180s -run TestS3PrefixObjectKeys ./...
|
||||
|
||||
- name: Run IAM Integration Tests
|
||||
timeout-minutes: 15
|
||||
working-directory: test/s3/normal
|
||||
|
||||
@@ -34,7 +34,7 @@ jobs:
|
||||
uses: actions/checkout@v7
|
||||
|
||||
- name: Set up JDK 11
|
||||
uses: actions/setup-java@v5
|
||||
uses: actions/setup-java@v6
|
||||
with:
|
||||
java-version: '11'
|
||||
distribution: 'temurin'
|
||||
|
||||
@@ -29,6 +29,7 @@ FROM alpine:3.23 as rust_builder
|
||||
ARG TARGETARCH
|
||||
ARG TAGS
|
||||
COPY weed-volume-prebuilt/ /prebuilt/
|
||||
COPY weed-worker-prebuilt/ /prebuilt-worker/
|
||||
COPY --from=builder /go/src/github.com/seaweedfs/seaweedfs/seaweed-volume /build/seaweed-volume
|
||||
COPY --from=builder /go/src/github.com/seaweedfs/seaweedfs/weed /build/weed
|
||||
WORKDIR /build/seaweed-volume
|
||||
@@ -47,16 +48,30 @@ RUN if [ -f "/prebuilt/weed-volume-${TARGETARCH}" ]; then \
|
||||
echo "Skipping Rust build for $TARGETARCH (unsupported)" && \
|
||||
touch /weed-volume; \
|
||||
fi
|
||||
# The Rust maintenance worker is taken pre-built or not at all: the lance jobs
|
||||
# it carries pull in arrow and datafusion, a far larger dependency tree than
|
||||
# the image build can carry, so an architecture CI did not build for gets the
|
||||
# same empty placeholder the entrypoint refuses to exec.
|
||||
RUN if [ -f "/prebuilt-worker/weed-worker-${TARGETARCH}" ]; then \
|
||||
echo "Using pre-built Rust worker for ${TARGETARCH}" && \
|
||||
cp "/prebuilt-worker/weed-worker-${TARGETARCH}" /weed-worker; \
|
||||
else \
|
||||
echo "No pre-built Rust worker for ${TARGETARCH}" && \
|
||||
touch /weed-worker; \
|
||||
fi
|
||||
|
||||
# Pre-built binaries arrive via GitHub Actions artifacts, which drop the
|
||||
# executable bit, so the copied file is 0644 and exec fails with "Permission
|
||||
# denied". Restore it (no-op for the empty placeholder, which stays size 0).
|
||||
RUN chmod 0755 /weed-volume
|
||||
# denied". Restore it (no-op for the empty placeholders, which stay size 0).
|
||||
RUN chmod 0755 /weed-volume /weed-worker
|
||||
|
||||
FROM alpine AS final
|
||||
LABEL author="Chris Lu"
|
||||
COPY --from=builder /go/bin/weed /usr/bin/
|
||||
# Copy Rust volume server binary (real binary on amd64/arm64, empty placeholder on other platforms)
|
||||
COPY --from=rust_builder /weed-volume /usr/bin/weed-volume
|
||||
# Same for the Rust maintenance worker, which serves Lance table buckets
|
||||
COPY --from=rust_builder /weed-worker /usr/bin/weed-worker
|
||||
RUN mkdir -p /etc/seaweedfs
|
||||
COPY --from=builder /go/src/github.com/seaweedfs/seaweedfs/docker/filer.toml /etc/seaweedfs/filer.toml
|
||||
COPY --from=builder /go/src/github.com/seaweedfs/seaweedfs/docker/entrypoint.sh /entrypoint.sh
|
||||
|
||||
@@ -90,6 +90,16 @@ case "$1" in
|
||||
exec /usr/bin/weed-volume $ARGS $@
|
||||
;;
|
||||
|
||||
'worker-rust')
|
||||
shift
|
||||
if [ ! -s /usr/bin/weed-worker ]; then
|
||||
echo "Error: Rust maintenance worker is not available on this platform ($(uname -m))." >&2
|
||||
echo "Use 'worker' for the Go maintenance worker instead." >&2
|
||||
exit 1
|
||||
fi
|
||||
exec /usr/bin/weed-worker "$@"
|
||||
;;
|
||||
|
||||
'server')
|
||||
ARGS="-dir=/data -volume.max=0 -master.volumeSizeLimitMB=1024"
|
||||
if isArgPassed "-volume.max" "$@"; then
|
||||
|
||||
@@ -4,7 +4,7 @@ go 1.26
|
||||
|
||||
require (
|
||||
cloud.google.com/go v0.123.0 // indirect
|
||||
cloud.google.com/go/pubsub v1.51.0
|
||||
cloud.google.com/go/pubsub v1.51.1
|
||||
cloud.google.com/go/storage v1.64.0
|
||||
github.com/Shopify/sarama v1.38.1
|
||||
github.com/aws/aws-sdk-go v1.55.8
|
||||
@@ -32,7 +32,7 @@ require (
|
||||
github.com/google/btree v1.1.3
|
||||
github.com/google/uuid v1.6.0
|
||||
github.com/google/wire v0.7.0 // indirect
|
||||
github.com/googleapis/gax-go/v2 v2.23.0 // indirect
|
||||
github.com/googleapis/gax-go/v2 v2.24.0 // indirect
|
||||
github.com/gorilla/mux v1.8.1
|
||||
github.com/hashicorp/errwrap v1.1.0 // indirect
|
||||
github.com/hashicorp/go-multierror v1.1.1 // indirect
|
||||
@@ -64,7 +64,7 @@ require (
|
||||
github.com/prometheus/procfs v0.21.1
|
||||
github.com/rcrowley/go-metrics v0.0.0-20201227073835-cf1acfcdf475 // indirect
|
||||
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec // indirect
|
||||
github.com/seaweedfs/goexif v1.0.3
|
||||
github.com/seaweedfs/goexif v2.0.0+incompatible
|
||||
github.com/seaweedfs/raft v1.2.0
|
||||
github.com/sirupsen/logrus v1.9.4 // indirect
|
||||
github.com/spf13/afero v1.15.0 // indirect
|
||||
@@ -99,15 +99,15 @@ require (
|
||||
golang.org/x/text v0.41.0 // indirect
|
||||
golang.org/x/tools v0.48.0 // indirect
|
||||
golang.org/x/xerrors v0.0.0-20240903120638-7835f813f4da // indirect
|
||||
google.golang.org/api v0.293.0
|
||||
google.golang.org/genproto v0.0.0-20260519071638-aa98bba5eb94 // indirect
|
||||
google.golang.org/grpc v1.84.0-dev.0.20260723093437-b6eac429d7b6
|
||||
google.golang.org/protobuf v1.36.11
|
||||
google.golang.org/api v0.294.0
|
||||
google.golang.org/genproto v0.0.0-20260715232425-e75dac1f907d // indirect
|
||||
google.golang.org/grpc v1.85.0-dev
|
||||
google.golang.org/protobuf v1.36.12
|
||||
gopkg.in/inf.v0 v0.9.1 // indirect
|
||||
modernc.org/b v1.0.0 // indirect
|
||||
modernc.org/mathutil v1.7.1 // indirect
|
||||
modernc.org/memory v1.11.0 // indirect
|
||||
modernc.org/sqlite v1.56.0
|
||||
modernc.org/sqlite v1.57.0
|
||||
)
|
||||
|
||||
require (
|
||||
@@ -122,14 +122,14 @@ require (
|
||||
github.com/apple/foundationdb/bindings/go v0.0.0-20250911184653-27f7192f47c3
|
||||
github.com/arangodb/go-driver v1.6.9
|
||||
github.com/armon/go-metrics v0.4.1
|
||||
github.com/aws/aws-sdk-go-v2 v1.43.5
|
||||
github.com/aws/aws-sdk-go-v2 v1.45.1
|
||||
github.com/aws/aws-sdk-go-v2/config v1.32.33
|
||||
github.com/aws/aws-sdk-go-v2/credentials v1.19.34
|
||||
github.com/aws/aws-sdk-go-v2/service/s3 v1.105.2
|
||||
github.com/cespare/xxhash/v2 v2.3.0
|
||||
github.com/cognusion/imaging v1.0.4
|
||||
github.com/fluent/fluent-logger-golang v1.10.1
|
||||
github.com/getsentry/sentry-go v0.44.1
|
||||
github.com/getsentry/sentry-go v0.48.0
|
||||
github.com/go-ldap/ldap/v3 v3.4.13
|
||||
github.com/golang-jwt/jwt/v5 v5.3.1
|
||||
github.com/google/flatbuffers/go v0.0.0-20230108230133-3b8644d32c50
|
||||
@@ -143,20 +143,20 @@ require (
|
||||
github.com/orcaman/concurrent-map/v2 v2.0.1
|
||||
github.com/parquet-go/parquet-go v0.32.0
|
||||
github.com/pkg/sftp v1.13.11
|
||||
github.com/rabbitmq/amqp091-go v1.13.0
|
||||
github.com/rabbitmq/amqp091-go v1.14.0
|
||||
github.com/rclone/rclone v1.75.0
|
||||
github.com/rdleal/intervalst v1.5.0
|
||||
github.com/redis/go-redis/v9 v9.21.0
|
||||
github.com/schollz/progressbar/v3 v3.19.1
|
||||
github.com/seaweedfs/go-fuse/v2 v2.9.4
|
||||
github.com/shirou/gopsutil/v4 v4.26.6
|
||||
github.com/shirou/gopsutil/v4 v4.26.7
|
||||
github.com/tarantool/go-option v1.1.0
|
||||
github.com/tarantool/go-tarantool/v3 v3.0.0
|
||||
github.com/tarantool/go-tarantool/v3 v3.0.1
|
||||
github.com/testcontainers/testcontainers-go v0.43.0
|
||||
github.com/tikv/client-go/v2 v2.0.7
|
||||
github.com/xeipuuv/gojsonschema v1.2.0
|
||||
github.com/ydb-platform/ydb-go-sdk-auth-environ v0.5.2
|
||||
github.com/ydb-platform/ydb-go-sdk/v3 v3.147.1
|
||||
github.com/ydb-platform/ydb-go-sdk/v3 v3.151.1
|
||||
go.etcd.io/etcd/client/pkg/v3 v3.7.1
|
||||
go.uber.org/atomic v1.11.0
|
||||
golang.org/x/sync v0.22.0
|
||||
@@ -193,7 +193,7 @@ require (
|
||||
github.com/cenkalti/backoff/v5 v5.0.3 // indirect
|
||||
github.com/clipperhouse/uax29/v2 v2.7.0 // indirect
|
||||
github.com/cockroachdb/apd/v3 v3.2.1 // indirect
|
||||
github.com/cockroachdb/errors v1.11.3 // indirect
|
||||
github.com/cockroachdb/errors v1.14.0 // indirect
|
||||
github.com/cockroachdb/logtags v0.0.0-20241215232642-bb51bb14a506 // indirect
|
||||
github.com/cockroachdb/redact v1.1.5 // indirect
|
||||
github.com/cockroachdb/version v0.0.0-20250314144055-3860cd14adf2 // indirect
|
||||
@@ -238,12 +238,12 @@ require (
|
||||
github.com/lpar/calendar v0.2.0 // indirect
|
||||
github.com/magiconair/properties v1.8.10 // indirect
|
||||
github.com/moby/docker-image-spec v1.3.1 // indirect
|
||||
github.com/moby/go-archive v0.2.0 // indirect
|
||||
github.com/moby/go-archive v0.3.0 // indirect
|
||||
github.com/moby/moby/api v1.54.2 // indirect
|
||||
github.com/moby/moby/client v0.4.0 // indirect
|
||||
github.com/moby/patternmatcher v0.6.1 // indirect
|
||||
github.com/moby/sys/sequential v0.6.0 // indirect
|
||||
github.com/moby/sys/user v0.4.0 // indirect
|
||||
github.com/moby/sys/sequential v0.7.0 // indirect
|
||||
github.com/moby/sys/user v0.4.1 // indirect
|
||||
github.com/moby/sys/userns v0.1.0 // indirect
|
||||
github.com/moby/term v0.5.2 // indirect
|
||||
github.com/oklog/ulid/v2 v2.1.1 // indirect
|
||||
@@ -260,6 +260,7 @@ require (
|
||||
github.com/rclone/Proton-API-Bridge v1.0.4 // indirect
|
||||
github.com/rclone/go-proton-api v1.0.3 // indirect
|
||||
github.com/rogpeppe/go-internal v1.15.0 // indirect
|
||||
github.com/rwcarlsen/goexif v0.0.0-20190401172101-9e8deecbddbd // indirect
|
||||
github.com/ryanuber/go-glob v1.0.0 // indirect
|
||||
github.com/sasha-s/go-deadlock v0.3.1 // indirect
|
||||
github.com/smarty/assertions v1.15.0 // indirect
|
||||
@@ -291,11 +292,11 @@ require (
|
||||
|
||||
require (
|
||||
cel.dev/expr v0.25.2 // indirect
|
||||
cloud.google.com/go/auth v0.23.0 // indirect
|
||||
cloud.google.com/go/auth v0.23.2 // indirect
|
||||
cloud.google.com/go/auth/oauth2adapt v0.2.8 // indirect
|
||||
cloud.google.com/go/compute/metadata v0.9.0 // indirect
|
||||
cloud.google.com/go/iam v1.11.0 // indirect
|
||||
cloud.google.com/go/monitoring v1.29.0 // indirect
|
||||
cloud.google.com/go/iam v1.12.0 // indirect
|
||||
cloud.google.com/go/monitoring v1.30.0 // indirect
|
||||
filippo.io/edwards25519 v1.2.0 // indirect
|
||||
github.com/Azure/azure-sdk-for-go/sdk/azcore v1.22.0
|
||||
github.com/Azure/azure-sdk-for-go/sdk/azidentity v1.14.0
|
||||
@@ -336,7 +337,7 @@ require (
|
||||
github.com/aws/aws-sdk-go-v2/service/sso v1.33.4 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/ssooidc v1.38.4 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/sts v1.45.4
|
||||
github.com/aws/smithy-go v1.27.7
|
||||
github.com/aws/smithy-go v1.28.1
|
||||
github.com/boltdb/bolt v1.3.1 // indirect
|
||||
github.com/bradenaw/juniper v0.15.3 // indirect
|
||||
github.com/buengese/sgzip v0.1.1 // indirect
|
||||
@@ -354,7 +355,7 @@ require (
|
||||
github.com/d4l3k/messagediff v1.2.1 // indirect
|
||||
github.com/dgryski/go-farm v0.0.0-20200201041132-a6ae2369ad13 // indirect
|
||||
github.com/dropbox/dropbox-sdk-go-unofficial/v6 v6.4.0 // indirect
|
||||
github.com/ebitengine/purego v0.10.1 // indirect
|
||||
github.com/ebitengine/purego v0.10.2 // indirect
|
||||
github.com/elastic/gosigar v0.14.3 // indirect
|
||||
github.com/emersion/go-message v0.18.2 // indirect
|
||||
github.com/emersion/go-vcard v0.0.0-20260618161152-d854b7e0e2d3 // indirect
|
||||
@@ -474,7 +475,7 @@ require (
|
||||
github.com/winfsp/cgofuse v1.6.1-0.20260126094232-f2c4fccdb286
|
||||
github.com/xanzy/ssh-agent v0.3.3 // indirect
|
||||
github.com/yandex-cloud/go-genproto v0.0.0-20211115083454-9ca41db5ed9e // indirect
|
||||
github.com/ydb-platform/ydb-go-genproto v0.0.0-20260428144813-1c07baab7f7b // indirect
|
||||
github.com/ydb-platform/ydb-go-genproto v0.0.0-20260810122915-65bfd5c4b705 // indirect
|
||||
github.com/ydb-platform/ydb-go-yc v0.12.1 // indirect
|
||||
github.com/ydb-platform/ydb-go-yc-metadata v0.6.1 // indirect
|
||||
github.com/yunify/qingstor-sdk-go/v3 v3.2.0 // indirect
|
||||
@@ -496,8 +497,8 @@ require (
|
||||
go.uber.org/zap v1.27.1 // indirect
|
||||
golang.org/x/term v0.45.0
|
||||
golang.org/x/time v0.15.0
|
||||
google.golang.org/genproto/googleapis/api v0.0.0-20260706201446-f0a921348800 // indirect
|
||||
google.golang.org/genproto/googleapis/rpc v0.0.0-20260807164820-c8921c73eeea // indirect
|
||||
google.golang.org/genproto/googleapis/api v0.0.0-20260715232425-e75dac1f907d // indirect
|
||||
google.golang.org/genproto/googleapis/rpc v0.0.0-20260819154853-08b0e4226688 // indirect
|
||||
gopkg.in/natefinch/lumberjack.v2 v2.2.1 // indirect
|
||||
gopkg.in/validator.v2 v2.0.1 // indirect
|
||||
gopkg.in/yaml.v2 v2.4.0 // indirect
|
||||
|
||||
@@ -94,8 +94,8 @@ cloud.google.com/go/assuredworkloads v1.7.0/go.mod h1:z/736/oNmtGAyU47reJgGN+KVo
|
||||
cloud.google.com/go/assuredworkloads v1.8.0/go.mod h1:AsX2cqyNCOvEQC8RMPnoc0yEarXQk6WEKkxYfL6kGIo=
|
||||
cloud.google.com/go/assuredworkloads v1.9.0/go.mod h1:kFuI1P78bplYtT77Tb1hi0FMxM0vVpRC7VVoJC3ZoT0=
|
||||
cloud.google.com/go/assuredworkloads v1.10.0/go.mod h1:kwdUQuXcedVdsIaKgKTp9t0UJkE5+PAVNhdQm4ZVq2E=
|
||||
cloud.google.com/go/auth v0.23.0 h1:6Gg1CMgpgubRG7DGz5Vf1pcoNo8RfiRiRAPS4crTp54=
|
||||
cloud.google.com/go/auth v0.23.0/go.mod h1:4DhBRcqvtljQN3dJ57qtqbib5ZGCYE5f2crfiiC2EM0=
|
||||
cloud.google.com/go/auth v0.23.2 h1:pxSCpfiji41hpzpPdMCftEUCezpgpqmmDdYiAjCKXxo=
|
||||
cloud.google.com/go/auth v0.23.2/go.mod h1:4DhBRcqvtljQN3dJ57qtqbib5ZGCYE5f2crfiiC2EM0=
|
||||
cloud.google.com/go/auth/oauth2adapt v0.2.8 h1:keo8NaayQZ6wimpNSmW5OPc283g65QNIiLpZnkHRbnc=
|
||||
cloud.google.com/go/auth/oauth2adapt v0.2.8/go.mod h1:XQ9y31RkqZCcwJWNSx2Xvric3RrU88hAYYbjDWYDL+c=
|
||||
cloud.google.com/go/automl v1.5.0/go.mod h1:34EjfoFGMZ5sgJ9EoLsRtdPSNZLcfflJR39VbVNS2M0=
|
||||
@@ -283,8 +283,8 @@ cloud.google.com/go/iam v0.7.0/go.mod h1:H5Br8wRaDGNc8XP3keLc4unfUUZeyH3Sfl9XpQE
|
||||
cloud.google.com/go/iam v0.8.0/go.mod h1:lga0/y3iH6CX7sYqypWJ33hf7kkfXJag67naqGESjkE=
|
||||
cloud.google.com/go/iam v0.11.0/go.mod h1:9PiLDanza5D+oWFZiH1uG+RnRCfEGKoyl6yo4cgWZGY=
|
||||
cloud.google.com/go/iam v0.12.0/go.mod h1:knyHGviacl11zrtZUoDuYpDgLjvr28sLQaG0YB2GYAY=
|
||||
cloud.google.com/go/iam v1.11.0 h1:KieQ9Pb+LLPak1O3Rv3GgCxhnmkYf7Xyh0P5HfF1jFM=
|
||||
cloud.google.com/go/iam v1.11.0/go.mod h1:KP+nKGugNJW4LcLx1uEZcq1ok5sQHFaQehQNl4QDgV4=
|
||||
cloud.google.com/go/iam v1.12.0 h1:Aki3bX9aHUDKPHfnRJfDcTdVedvy6quGBQcTqx3DRXk=
|
||||
cloud.google.com/go/iam v1.12.0/go.mod h1:FEZ4lXpADAC2AIpQY7LANNjjwyQ2jK439CI2VaD+sLY=
|
||||
cloud.google.com/go/iap v1.4.0/go.mod h1:RGFwRJdihTINIe4wZ2iCP0zF/qu18ZwyKxrhMhygBEc=
|
||||
cloud.google.com/go/iap v1.5.0/go.mod h1:UH/CGgKd4KyohZL5Pt0jSKE4m3FR51qg6FKQ/z/Ix9A=
|
||||
cloud.google.com/go/iap v1.6.0/go.mod h1:NSuvI9C/j7UdjGjIde7t7HBz+QTwBcapPE07+sSRcLk=
|
||||
@@ -310,8 +310,8 @@ cloud.google.com/go/lifesciences v0.6.0/go.mod h1:ddj6tSX/7BOnhxCSd3ZcETvtNr8NZ6
|
||||
cloud.google.com/go/lifesciences v0.8.0/go.mod h1:lFxiEOMqII6XggGbOnKiyZ7IBwoIqA84ClvoezaA/bo=
|
||||
cloud.google.com/go/logging v1.6.1/go.mod h1:5ZO0mHHbvm8gEmeEUHrmDlTDSu5imF6MUP9OfilNXBw=
|
||||
cloud.google.com/go/logging v1.7.0/go.mod h1:3xjP2CjkM3ZkO73aj4ASA5wRPGGCRrPIAeNqVNkzY8M=
|
||||
cloud.google.com/go/logging v1.18.0 h1:KhzZq+1cSkPH9YUaKLLhLtQxIHitVayBmk0sGfoM9+k=
|
||||
cloud.google.com/go/logging v1.18.0/go.mod h1:ZGKnpBaURITh+g/uom2VhbiFoFWvejcrHPDhxFtU/gI=
|
||||
cloud.google.com/go/logging v1.19.0 h1:NCqhdVUg3wQ8Cobdf16FDSuTGi3+6+hdSBHrY5TsR6Q=
|
||||
cloud.google.com/go/logging v1.19.0/go.mod h1:i40NZCHC9Gqvod4yE+yQfDWwlgwW/SrshkkGibCHxcA=
|
||||
cloud.google.com/go/longrunning v0.1.1/go.mod h1:UUFxuDWkv22EuY93jjmDMFT5GPQKeFVJBIF6QlTqdsE=
|
||||
cloud.google.com/go/longrunning v0.3.0/go.mod h1:qth9Y41RRSUE69rDcOn6DdK3HfQfsUI0YSmW3iIlLJc=
|
||||
cloud.google.com/go/longrunning v0.4.1/go.mod h1:4iWDqhBZ70CvZ6BfETbvam3T8FMvLK+eFj0E6AaRQTo=
|
||||
@@ -338,8 +338,8 @@ cloud.google.com/go/metastore v1.10.0/go.mod h1:fPEnH3g4JJAk+gMRnrAnoqyv2lpUCqJP
|
||||
cloud.google.com/go/monitoring v1.7.0/go.mod h1:HpYse6kkGo//7p6sT0wsIC6IBDET0RhIsnmlA53dvEk=
|
||||
cloud.google.com/go/monitoring v1.8.0/go.mod h1:E7PtoMJ1kQXWxPjB6mv2fhC5/15jInuulFdYYtlcvT4=
|
||||
cloud.google.com/go/monitoring v1.12.0/go.mod h1:yx8Jj2fZNEkL/GYZyTLS4ZtZEZN8WtDEiEqG4kLK50w=
|
||||
cloud.google.com/go/monitoring v1.29.0 h1:AHhDsFaSax1/4k+qlIDX/SDGe6hggnfXJ9dkgD9qBPY=
|
||||
cloud.google.com/go/monitoring v1.29.0/go.mod h1:72NOVjJXHY/HBfoLT0+qlCZBT059+9VXLeAnL2PeeVM=
|
||||
cloud.google.com/go/monitoring v1.30.0 h1:r/d+JUbyKmJ8b07iznuKfzVzrIXTWxHQ3lBRm3x2LlY=
|
||||
cloud.google.com/go/monitoring v1.30.0/go.mod h1:htlUR0QWVMrjFzZmN4LGnMAve9xB/eduwjmINxVZ8RM=
|
||||
cloud.google.com/go/networkconnectivity v1.4.0/go.mod h1:nOl7YL8odKyAOtzNX73/M5/mGZgqqMeryi6UPZTk/rA=
|
||||
cloud.google.com/go/networkconnectivity v1.5.0/go.mod h1:3GzqJx7uhtlM3kln0+x5wyFvuVH1pIBJjhCpjzSt75o=
|
||||
cloud.google.com/go/networkconnectivity v1.6.0/go.mod h1:OJOoEXW+0LAxHh89nXd64uGG+FbQoeH8DtxCHVOMlaM=
|
||||
@@ -391,8 +391,8 @@ cloud.google.com/go/pubsub v1.3.1/go.mod h1:i+ucay31+CNRpDW4Lu78I4xXG+O1r/MAHgjp
|
||||
cloud.google.com/go/pubsub v1.26.0/go.mod h1:QgBH3U/jdJy/ftjPhTkyXNj543Tin1pRYcdcPRnFIRI=
|
||||
cloud.google.com/go/pubsub v1.27.1/go.mod h1:hQN39ymbV9geqBnfQq6Xf63yNhUAhv9CZhzp5O6qsW0=
|
||||
cloud.google.com/go/pubsub v1.28.0/go.mod h1:vuXFpwaVoIPQMGXqRyUQigu/AX1S3IWugR9xznmcXX8=
|
||||
cloud.google.com/go/pubsub v1.51.0 h1:XOaCejsqX7EEtUdQz+WPag66wWsUUGliyCOfGPKfo90=
|
||||
cloud.google.com/go/pubsub v1.51.0/go.mod h1:NERXf11sd82UV3VnflcUj8POIyQUXT/QwrKlxD8di/I=
|
||||
cloud.google.com/go/pubsub v1.51.1 h1:R3G1wCOxBO7jRpL8x2pdZMv1GAJDF6ax/m2zPOtvTNE=
|
||||
cloud.google.com/go/pubsub v1.51.1/go.mod h1:y2T0IKtW1iWwVvazYaRpqOAFO4gy2+O7dTDt9TWY/5U=
|
||||
cloud.google.com/go/pubsub/v2 v2.6.0 h1:8pjR0id+GTB+krKx5G6AGJoYrHog58w2Q89PCOrfM64=
|
||||
cloud.google.com/go/pubsub/v2 v2.6.0/go.mod h1:4anqvV/w8Pcgu2tO0qr2XgsF3GXHowzryfQ5gOnVmWY=
|
||||
cloud.google.com/go/pubsublite v1.5.0/go.mod h1:xapqNQ1CuLfGi23Yda/9l4bBCKz/wC3KIJ5gKcxveZg=
|
||||
@@ -707,8 +707,8 @@ github.com/armon/go-metrics v0.4.1/go.mod h1:E6amYzXo6aW1tqzoZGT755KkbgrJsSdpwZ+
|
||||
github.com/atomicgo/cursor v0.0.1/go.mod h1:cBON2QmmrysudxNBFthvMtN32r3jxVRIvzkUiF/RuIk=
|
||||
github.com/aws/aws-sdk-go v1.55.8 h1:JRmEUbU52aJQZ2AjX4q4Wu7t4uZjOu71uyNmaWlUkJQ=
|
||||
github.com/aws/aws-sdk-go v1.55.8/go.mod h1:ZkViS9AqA6otK+JBBNH2++sx1sgxrPKcSzPPvQkUtXk=
|
||||
github.com/aws/aws-sdk-go-v2 v1.43.5 h1:yKT5GYnFWhuDo+DqKvE5ZPwVn3RjC4MAeBtZGlh6AVM=
|
||||
github.com/aws/aws-sdk-go-v2 v1.43.5/go.mod h1:wZjAJppCntyOGgVSmgVTfDyRJK5PHOasO6Wsy8U7Axk=
|
||||
github.com/aws/aws-sdk-go-v2 v1.45.1 h1:iIoG3NaLhV6UZpPXyPXlDj2I9oS8tV/nMcMnITCC6Ks=
|
||||
github.com/aws/aws-sdk-go-v2 v1.45.1/go.mod h1:bttEH6JqnUL8LepvDVfdrds/fZ5bCIxzpe3abyUrhDU=
|
||||
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.14 h1:3IZY0XAJquT3aHzbkHfPzy4ACPcEjVG0x87KOwtpqGY=
|
||||
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.14/go.mod h1:zwM6veDkhGgQFqkBy+uT28AAYpLu+uFMlPl+rCg/73E=
|
||||
github.com/aws/aws-sdk-go-v2/config v1.32.33 h1:M1m/Q6f0OKDEDGwhiNOqx1OjTdrewe3v+GDbHmKczWk=
|
||||
@@ -749,8 +749,8 @@ github.com/aws/aws-sdk-go-v2/service/ssooidc v1.38.4 h1:AsbZcJAQPRmHDJG8K1N0pof/
|
||||
github.com/aws/aws-sdk-go-v2/service/ssooidc v1.38.4/go.mod h1:6imqztH0//t0mKbl6yWl7swSEl7F/w32oAmqB3vP1ag=
|
||||
github.com/aws/aws-sdk-go-v2/service/sts v1.45.4 h1:w/AryDYMjSUANSQ2uoZxJovUsMTwWJNTv3IMex30Y+4=
|
||||
github.com/aws/aws-sdk-go-v2/service/sts v1.45.4/go.mod h1:WeBiAa67azG7Su9Vf+ChGDBLiAozJCXzdjXiPBUwtbc=
|
||||
github.com/aws/smithy-go v1.27.7 h1:Zgj5z4LfcDYoQIVk+n/yGdTkP/2y6ZT5vYxe0fp7bqE=
|
||||
github.com/aws/smithy-go v1.27.7/go.mod h1:YE2RhdIuDbA5E5bTdciG9KrW3+TiEONeUWCqxX9i1Fc=
|
||||
github.com/aws/smithy-go v1.28.1 h1:R/nXH00c8qcfCzQVELtRw+eLQWtzv+VAIEFJ1/xxXlQ=
|
||||
github.com/aws/smithy-go v1.28.1/go.mod h1:YE2RhdIuDbA5E5bTdciG9KrW3+TiEONeUWCqxX9i1Fc=
|
||||
github.com/bahlo/generic-list-go v0.2.0 h1:5sz/EEAK+ls5wF+NeqDpk5+iNdMDXrh3z3nPnH1Wvgk=
|
||||
github.com/bahlo/generic-list-go v0.2.0/go.mod h1:2KvAjgMlE5NNynlg/5iLrrCCZ2+5xWbdbCW3pNTGyYg=
|
||||
github.com/bazelbuild/rules_go v0.46.0 h1:CTefzjN/D3Cdn3rkrM6qMWuQj59OBcuOjyIp3m4hZ7s=
|
||||
@@ -848,8 +848,8 @@ github.com/cncf/xds/go v0.0.0-20260202195803-dba9d589def2 h1:aBangftG7EVZoUb69Os
|
||||
github.com/cncf/xds/go v0.0.0-20260202195803-dba9d589def2/go.mod h1:qwXFYgsP6T7XnJtbKlf1HP8AjxZZyzxMmc+Lq5GjlU4=
|
||||
github.com/cockroachdb/apd/v3 v3.2.1 h1:U+8j7t0axsIgvQUqthuNm82HIrYXodOV2iWLWtEaIwg=
|
||||
github.com/cockroachdb/apd/v3 v3.2.1/go.mod h1:klXJcjp+FffLTHlhIG69tezTDvdP065naDsHzKhYSqc=
|
||||
github.com/cockroachdb/errors v1.11.3 h1:5bA+k2Y6r+oz/6Z/RFlNeVCesGARKuC6YymtcDrbC/I=
|
||||
github.com/cockroachdb/errors v1.11.3/go.mod h1:m4UIW4CDjx+R5cybPsNrRbreomiFqt8o1h1wUVazSd8=
|
||||
github.com/cockroachdb/errors v1.14.0 h1:EfdVEJpN3z8rPMo43Yit59LxoiIa470fSXpZXuEs+ZI=
|
||||
github.com/cockroachdb/errors v1.14.0/go.mod h1:xRa70jZ9sNBQmISt5KmJmAD++E4dQHm89oCRiZGEdq0=
|
||||
github.com/cockroachdb/logtags v0.0.0-20241215232642-bb51bb14a506 h1:ASDL+UJcILMqgNeV5jiqR4j+sTuvQNHdf2chuKj1M5k=
|
||||
github.com/cockroachdb/logtags v0.0.0-20241215232642-bb51bb14a506/go.mod h1:Mw7HqKr2kdtu6aYGn3tPmAftiP3QPX63LdK/zcariIo=
|
||||
github.com/cockroachdb/redact v1.1.5 h1:u1PMllDkdFfPWaNGMyLD1+so+aq3uUItthCFqzwPJ30=
|
||||
@@ -956,8 +956,8 @@ github.com/eapache/go-xerial-snappy v0.0.0-20230731223053-c322873962e3 h1:Oy0F4A
|
||||
github.com/eapache/go-xerial-snappy v0.0.0-20230731223053-c322873962e3/go.mod h1:YvSRo5mw33fLEx1+DlK6L2VV43tJt5Eyel9n9XBcR+0=
|
||||
github.com/eapache/queue v1.1.0 h1:YOEu7KNc61ntiQlcEeUIoDTJ2o8mQznoNvUhiigpIqc=
|
||||
github.com/eapache/queue v1.1.0/go.mod h1:6eCeP0CKFpHLu8blIFXhExK/dRa7WDZfr6jVFPTqq+I=
|
||||
github.com/ebitengine/purego v0.10.1 h1:dewVBCBT2GaMu1SrNTYxQhgQBethzfhiwvZiLGP/qyY=
|
||||
github.com/ebitengine/purego v0.10.1/go.mod h1:iIjxzd6CiRiOG0UyXP+V1+jWqUXVjPKLAI0mRfJZTmQ=
|
||||
github.com/ebitengine/purego v0.10.2 h1:W809HbnvzAxgdm+aOvlSekrM16wGCdT/e76+9tS7gzE=
|
||||
github.com/ebitengine/purego v0.10.2/go.mod h1:iIjxzd6CiRiOG0UyXP+V1+jWqUXVjPKLAI0mRfJZTmQ=
|
||||
github.com/eiannone/keyboard v0.0.0-20220611211555-0d226195f203 h1:XBBHcIb256gUJtLmY22n99HaZTz+r2Z51xUPi01m3wg=
|
||||
github.com/eiannone/keyboard v0.0.0-20220611211555-0d226195f203/go.mod h1:E1jcSv8FaEny+OP/5k9UxZVw9YFWGj7eI4KR/iOBqCg=
|
||||
github.com/elastic/gosigar v0.14.3 h1:xwkKwPia+hSfg9GqrCUKYdId102m9qTJIIr7egmK/uo=
|
||||
@@ -1031,8 +1031,8 @@ github.com/gabriel-vasile/mimetype v1.4.13 h1:46nXokslUBsAJE/wMsp5gtO500a4F3Nkz9
|
||||
github.com/gabriel-vasile/mimetype v1.4.13/go.mod h1:d+9Oxyo1wTzWdyVUPMmXFvp4F9tea18J8ufA774AB3s=
|
||||
github.com/geoffgarside/ber v1.2.0 h1:/loowoRcs/MWLYmGX9QtIAbA+V/FrnVLsMMPhwiRm64=
|
||||
github.com/geoffgarside/ber v1.2.0/go.mod h1:jVPKeCbj6MvQZhwLYsGwaGI52oUorHoHKNecGT85ZCc=
|
||||
github.com/getsentry/sentry-go v0.44.1 h1:/cPtrA5qB7uMRrhgSn9TYtcEF36auGP3Y6+ThvD/yaI=
|
||||
github.com/getsentry/sentry-go v0.44.1/go.mod h1:XDotiNZbgf5U8bPDUAfvcFmOnMQQceESxyKaObSssW0=
|
||||
github.com/getsentry/sentry-go v0.48.0 h1:FRZNr7Uk1C86ev1bSJmYlUkL9oyivQA6YOcdYfaaMmY=
|
||||
github.com/getsentry/sentry-go v0.48.0/go.mod h1:E5UkA5wp1qR2+MDydNYlVeUiNN2xEdjYMidkgf0Qoss=
|
||||
github.com/ghodss/yaml v1.0.0/go.mod h1:4dBDuWmgqj2HViK6kFavaiC9ZROes6MMH2rRYeMEF04=
|
||||
github.com/gin-contrib/sse v1.1.0 h1:n0w2GMuUpWDVp7qSpvze6fAu9iRxJY4Hmj6AmBOU05w=
|
||||
github.com/gin-contrib/sse v1.1.0/go.mod h1:hxRZ5gVpWMT7Z0B0gSNYqqsSCNIJMjzvm6fqCz9vjwM=
|
||||
@@ -1267,8 +1267,8 @@ github.com/googleapis/gax-go/v2 v2.4.0/go.mod h1:XOTVJ59hdnfJLIP/dh8n5CGryZR2LxK
|
||||
github.com/googleapis/gax-go/v2 v2.5.1/go.mod h1:h6B0KMMFNtI2ddbGJn3T3ZbwkeT6yqEF02fYlzkUCyo=
|
||||
github.com/googleapis/gax-go/v2 v2.6.0/go.mod h1:1mjbznJAPHFpesgE5ucqfYEscaz5kMdcIDwU/6+DDoY=
|
||||
github.com/googleapis/gax-go/v2 v2.7.0/go.mod h1:TEop28CZZQ2y+c0VxMUmu1lV+fQx57QpBWsYpwqHJx8=
|
||||
github.com/googleapis/gax-go/v2 v2.23.0 h1:Tchl7qkvE7Ip3y+ztvNufYFvkfqTe7NfLTYGIdJRLuE=
|
||||
github.com/googleapis/gax-go/v2 v2.23.0/go.mod h1:rBQKOVJCdb8IFEzg+FCwlt1LP/xMDGuqUXhUG+XMXEg=
|
||||
github.com/googleapis/gax-go/v2 v2.24.0 h1:myMaPYyF9MecEmvQqMqomIwn9t/4KCZN9qnwsS76wlg=
|
||||
github.com/googleapis/gax-go/v2 v2.24.0/go.mod h1:IaTHBDd7NHxSCiu0vEs8pQZu4dGZrWwuSoxCnk16OFM=
|
||||
github.com/googleapis/go-type-adapters v1.0.0/go.mod h1:zHW75FOG2aur7gAO2B+MLby+cLsWGBF62rFAi7WjWO4=
|
||||
github.com/googleapis/google-cloud-go-testing v0.0.0-20200911160855-bcd43fbb19e8/go.mod h1:dvDLG8qkwmyD9a/MJJN3XJcT3xFxOKAvTZGvuZmac9g=
|
||||
github.com/gookit/assert v0.1.1 h1:lh3GcawXe/p+cU7ESTZ5Ui3Sm/x8JWpIis4/1aF0mY0=
|
||||
@@ -1543,8 +1543,8 @@ github.com/moby/buildkit v0.29.0 h1:wxLEFbCOJntEDjSNNN2YWd8zxltZxT5muDQ0LzpbtpU=
|
||||
github.com/moby/buildkit v0.29.0/go.mod h1:Dmv2FeDe34t75QuzeU87rBoZpAAkcpT5zeu4hXzmASc=
|
||||
github.com/moby/docker-image-spec v1.3.1 h1:jMKff3w6PgbfSa69GfNg+zN/XLhfXJGnEx3Nl2EsFP0=
|
||||
github.com/moby/docker-image-spec v1.3.1/go.mod h1:eKmb5VW8vQEh/BAr2yvVNvuiJuY6UIocYsFu/DxxRpo=
|
||||
github.com/moby/go-archive v0.2.0 h1:zg5QDUM2mi0JIM9fdQZWC7U8+2ZfixfTYoHL7rWUcP8=
|
||||
github.com/moby/go-archive v0.2.0/go.mod h1:mNeivT14o8xU+5q1YnNrkQVpK+dnNe/K6fHqnTg4qPU=
|
||||
github.com/moby/go-archive v0.3.0 h1:nos4BtzzUIqB406BgQnWGMI4qib9BZ8XUHU+ucv/n1c=
|
||||
github.com/moby/go-archive v0.3.0/go.mod h1:Npdv43fFqlhZW7Xo8fbm3ZMYFvAGNviUPqX21VERbcE=
|
||||
github.com/moby/locker v1.0.1 h1:fOXqR41zeveg4fFODix+1Ch4mj/gT0NE1XJbp/epuBg=
|
||||
github.com/moby/locker v1.0.1/go.mod h1:S7SDdo5zpBK84bzzVlKr2V0hz+7x9hWbYC/kq7oQppc=
|
||||
github.com/moby/moby/api v1.54.2 h1:wiat9QAhnDQjA7wk1kh/TqHz2I1uUA7M7t9SAl/JNXg=
|
||||
@@ -1559,14 +1559,14 @@ github.com/moby/sys/capability v0.4.0 h1:4D4mI6KlNtWMCM1Z/K0i7RV1FkX+DBDHKVJpCnd
|
||||
github.com/moby/sys/capability v0.4.0/go.mod h1:4g9IK291rVkms3LKCDOoYlnV8xKwoDTpIrNEE35Wq0I=
|
||||
github.com/moby/sys/mountinfo v0.7.2 h1:1shs6aH5s4o5H2zQLn796ADW1wMrIwHsyJ2v9KouLrg=
|
||||
github.com/moby/sys/mountinfo v0.7.2/go.mod h1:1YOa8w8Ih7uW0wALDUgT1dTTSBrZ+HiBLGws92L2RU4=
|
||||
github.com/moby/sys/sequential v0.6.0 h1:qrx7XFUd/5DxtqcoH1h438hF5TmOvzC/lspjy7zgvCU=
|
||||
github.com/moby/sys/sequential v0.6.0/go.mod h1:uyv8EUTrca5PnDsdMGXhZe6CCe8U/UiTWd+lL+7b/Ko=
|
||||
github.com/moby/sys/sequential v0.7.0 h1:ASQNGNROJSuOO6LL6bPHbKvuZu6NU8P4ldPWk31zj/8=
|
||||
github.com/moby/sys/sequential v0.7.0/go.mod h1:NfSTAp6V3fw4tmkD62PEcOKeZKquXT8VKCkf7aVR79o=
|
||||
github.com/moby/sys/signal v0.7.1 h1:PrQxdvxcGijdo6UXXo/lU/TvHUWyPhj7UOpSo8tuvk0=
|
||||
github.com/moby/sys/signal v0.7.1/go.mod h1:Se1VGehYokAkrSQwL4tDzHvETwUZlnY7S5XtQ50mQp8=
|
||||
github.com/moby/sys/symlink v0.3.0 h1:GZX89mEZ9u53f97npBy4Rc3vJKj7JBDj/PN2I22GrNU=
|
||||
github.com/moby/sys/symlink v0.3.0/go.mod h1:3eNdhduHmYPcgsJtZXW1W4XUJdZGBIkttZ8xKqPUJq0=
|
||||
github.com/moby/sys/user v0.4.0 h1:jhcMKit7SA80hivmFJcbB1vqmw//wU61Zdui2eQXuMs=
|
||||
github.com/moby/sys/user v0.4.0/go.mod h1:bG+tYYYJgaMtRKgEmuueC0hJEAZWwtIbZTB+85uoHjs=
|
||||
github.com/moby/sys/user v0.4.1 h1:RgjRlaDKi/Xmyrz4t8lyzXT6v2ooFeO/7xtchmhVWE0=
|
||||
github.com/moby/sys/user v0.4.1/go.mod h1:E9QsW5WRe1kUAf7kW8hXKwu1uhsZEAdPLYHYSDudF4Y=
|
||||
github.com/moby/sys/userns v0.1.0 h1:tVLXkFOxVu9A64/yh59slHVv9ahO9UIev4JZusOLG/g=
|
||||
github.com/moby/sys/userns v0.1.0/go.mod h1:IHUYgu/kao6N8YZlp9Cf444ySSvCmDlmzUcYfDHOl28=
|
||||
github.com/moby/term v0.5.2 h1:6qk3FJAFDs6i/q3W/pQ97SX192qKfZgGjCQqfCJkgzQ=
|
||||
@@ -1749,8 +1749,8 @@ github.com/quic-go/qpack v0.6.0 h1:g7W+BMYynC1LbYLSqRt8PBg5Tgwxn214ZZR34VIOjz8=
|
||||
github.com/quic-go/qpack v0.6.0/go.mod h1:lUpLKChi8njB4ty2bFLX2x4gzDqXwUpaO1DP9qMDZII=
|
||||
github.com/quic-go/quic-go v0.59.0 h1:OLJkp1Mlm/aS7dpKgTc6cnpynnD2Xg7C1pwL6vy/SAw=
|
||||
github.com/quic-go/quic-go v0.59.0/go.mod h1:upnsH4Ju1YkqpLXC305eW3yDZ4NfnNbmQRCMWS58IKU=
|
||||
github.com/rabbitmq/amqp091-go v1.13.0 h1:L8NA1WtF76C6KA3LAoufjfLgbist/If1UQYcsOjtxXA=
|
||||
github.com/rabbitmq/amqp091-go v1.13.0/go.mod h1:Hy4jKW5kQART1u+JkDTF9YYOQUHXqMuhrgxOEeS7G4o=
|
||||
github.com/rabbitmq/amqp091-go v1.14.0 h1:RSaT7aOKt/OrkVUyswPDW29lnRz9psuGmfZFBmLqLek=
|
||||
github.com/rabbitmq/amqp091-go v1.14.0/go.mod h1:Hy4jKW5kQART1u+JkDTF9YYOQUHXqMuhrgxOEeS7G4o=
|
||||
github.com/rclone/Proton-API-Bridge v1.0.4 h1:uGQJRjQC1hVLd5kqLsXc6CWO6oqrVeLoKQYoHapEZDg=
|
||||
github.com/rclone/Proton-API-Bridge v1.0.4/go.mod h1:VTPBYZotKAeDLlAzxU2O/s14NXk9FxUt9hn1jhH2iY8=
|
||||
github.com/rclone/go-proton-api v1.0.3 h1:3gBTzR+j0dYiTwtj9yKIdN/aV3W2a8KIPKp0GArojyQ=
|
||||
@@ -1790,6 +1790,8 @@ github.com/rs/zerolog v1.34.0 h1:k43nTLIwcTVQAncfCw4KZ2VY6ukYoZaBPNOE8txlOeY=
|
||||
github.com/rs/zerolog v1.34.0/go.mod h1:bJsvje4Z08ROH4Nhs5iH600c3IkWhwp44iRc54W6wYQ=
|
||||
github.com/ruudk/golang-pdf417 v0.0.0-20181029194003-1af4ab5afa58/go.mod h1:6lfFZQK844Gfx8o5WFuvpxWRwnSoipWe/p622j1v06w=
|
||||
github.com/ruudk/golang-pdf417 v0.0.0-20201230142125-a7e3863a1245/go.mod h1:pQAZKsJ8yyVxGRWYNEm9oFB8ieLgKFnamEyDmSA0BRk=
|
||||
github.com/rwcarlsen/goexif v0.0.0-20190401172101-9e8deecbddbd h1:CmH9+J6ZSsIjUK3dcGsnCnO41eRBOnY12zwkn5qVwgc=
|
||||
github.com/rwcarlsen/goexif v0.0.0-20190401172101-9e8deecbddbd/go.mod h1:hPqNNc0+uJM6H+SuU8sEs5K5IQeKccPqeSjfgcKGgPk=
|
||||
github.com/ryanuber/go-glob v1.0.0 h1:iQh3xXAumdQ+4Ufa5b25cRpC5TYKlno6hsv6Cb3pkBk=
|
||||
github.com/ryanuber/go-glob v1.0.0/go.mod h1:807d1WSdnB0XRJzKNil9Om6lcp/3a0v4qIHxIXzX/Yc=
|
||||
github.com/sabhiram/go-gitignore v0.0.0-20210923224102-525f6e181f06 h1:OkMGxebDjyw0ULyrTYWeN0UNCCkmCWfjPnIA2W6oviI=
|
||||
@@ -1808,8 +1810,8 @@ github.com/seaweedfs/cockroachdb-parser v0.0.0-20260225204133-2f342c5ea564 h1:Tg
|
||||
github.com/seaweedfs/cockroachdb-parser v0.0.0-20260225204133-2f342c5ea564/go.mod h1:JSKCh6uCHBz91lQYFYHCyTrSVIPge4SUFVn28iwMNB0=
|
||||
github.com/seaweedfs/go-fuse/v2 v2.9.4 h1:ACyloiuopdhRSjdLLeSWbsVaemMPskORaRF01TY6GyM=
|
||||
github.com/seaweedfs/go-fuse/v2 v2.9.4/go.mod h1:zABdmWEa6A0bwaBeEOBUeUkGIZlxUhcdv+V1Dcc/U/I=
|
||||
github.com/seaweedfs/goexif v1.0.3 h1:ve/OjI7dxPW8X9YQsv3JuVMaxEyF9Rvfd04ouL+Bz30=
|
||||
github.com/seaweedfs/goexif v1.0.3/go.mod h1:Oni780Z236sXpIQzk1XoJlTwqrJ02smEin9zQeff7Fk=
|
||||
github.com/seaweedfs/goexif v2.0.0+incompatible h1:x8pckiT12QQhifwhDQpeISgDfsqmQ6VR4LFPQ64JRps=
|
||||
github.com/seaweedfs/goexif v2.0.0+incompatible/go.mod h1:Oni780Z236sXpIQzk1XoJlTwqrJ02smEin9zQeff7Fk=
|
||||
github.com/seaweedfs/raft v1.2.0 h1:Ez4Hw9ifBbTT7wg54DvGHBjw1vRlTb4roH0TKl0Oj9Y=
|
||||
github.com/seaweedfs/raft v1.2.0/go.mod h1:fgs/rAVEzjQ7e04XMzG3eJhwZZRmBW+2uRtjakeCGeU=
|
||||
github.com/secure-systems-lab/go-securesystemslib v0.10.0 h1:l+H5ErcW0PAehBNrBxoGv1jjNpGYdZ9RcheFkB2WI14=
|
||||
@@ -1820,8 +1822,8 @@ github.com/sergi/go-diff v1.2.0 h1:XU+rvMAioB0UC3q1MFrIQy4Vo5/4VsRDQQXHsEya6xQ=
|
||||
github.com/sergi/go-diff v1.2.0/go.mod h1:STckp+ISIX8hZLjrqAeVduY0gWCT9IjLuqbuNXdaHfM=
|
||||
github.com/shibumi/go-pathspec v1.3.0 h1:QUyMZhFo0Md5B8zV8x2tesohbb5kfbpTi9rBnKh5dkI=
|
||||
github.com/shibumi/go-pathspec v1.3.0/go.mod h1:Xutfslp817l2I1cZvgcfeMQJG5QnU2lh5tVaaMCl3jE=
|
||||
github.com/shirou/gopsutil/v4 v4.26.6 h1:Mzr/npDtQC/xpeEuQKHZt8Zo9CmPvhTj8nkR8w5TLDs=
|
||||
github.com/shirou/gopsutil/v4 v4.26.6/go.mod h1:LZ6ewCSkBqUpvSOf+LsTGnRinC6iaNUNMGBtDkJBaLQ=
|
||||
github.com/shirou/gopsutil/v4 v4.26.7 h1:IXzpHz/dkMRYAhKkOXr1HB6SuzWU3eoyyeWe7g3bNZc=
|
||||
github.com/shirou/gopsutil/v4 v4.26.7/go.mod h1:5O9FjBiXoTDFatIWjZZosqj4pV0DRtLx598xGbBehzM=
|
||||
github.com/sigstore/sigstore v1.10.4 h1:ytOmxMgLdcUed3w1SbbZOgcxqwMG61lh1TmZLN+WeZE=
|
||||
github.com/sigstore/sigstore v1.10.4/go.mod h1:tDiyrdOref3q6qJxm2G+JHghqfmvifB7hw+EReAfnbI=
|
||||
github.com/sigstore/sigstore-go v1.1.4 h1:wTTsgCHOfqiEzVyBYA6mDczGtBkN7cM8mPpjJj5QvMg=
|
||||
@@ -1907,8 +1909,8 @@ github.com/tarantool/go-iproto v1.1.0 h1:HULVOIHsiehI+FnHfM7wMDntuzUddO09DKqu2Wn
|
||||
github.com/tarantool/go-iproto v1.1.0/go.mod h1:LNCtdyZxojUed8SbOiYHoc3v9NvaZTB7p96hUySMlIo=
|
||||
github.com/tarantool/go-option v1.1.0 h1:ShoOhNsdL41sRpm4hXCRDjV8H0WzPkd4UnKhLKbW//w=
|
||||
github.com/tarantool/go-option v1.1.0/go.mod h1:hMr9z2JXOWlgdCBpCPSL2nwp8718GKYvNBJ+ZuzJbCo=
|
||||
github.com/tarantool/go-tarantool/v3 v3.0.0 h1:zsIXS4nvSSXqZWtN/f1Bfuq973j6J85O5U/8RYQCRcE=
|
||||
github.com/tarantool/go-tarantool/v3 v3.0.0/go.mod h1:TXxLWhUCgdxXFfelnTSkq+goKRTRj660zxq4/WXPe8k=
|
||||
github.com/tarantool/go-tarantool/v3 v3.0.1 h1:vaUX4xmVmXh2dIJ/LqlX1MXK3iYqAqV6YiE54Wwl/qg=
|
||||
github.com/tarantool/go-tarantool/v3 v3.0.1/go.mod h1:TXxLWhUCgdxXFfelnTSkq+goKRTRj660zxq4/WXPe8k=
|
||||
github.com/testcontainers/testcontainers-go v0.43.0 h1:oEQx5MW2DGd9z3AeEQfB2lPM0eLs7ztyaGRu75bFo5A=
|
||||
github.com/testcontainers/testcontainers-go v0.43.0/go.mod h1:+VxkT2NQnKOZPKi6praMuMKYHYyOGXr0XSBSlSMCzFo=
|
||||
github.com/testcontainers/testcontainers-go/modules/compose v0.42.0 h1:+t1ZN31TD36cwxmeLqGwe7wIdvblBm0Z+vlj4SX8Mv0=
|
||||
@@ -2030,14 +2032,14 @@ github.com/yandex-cloud/go-genproto v0.0.0-20211115083454-9ca41db5ed9e h1:9LPdmD
|
||||
github.com/yandex-cloud/go-genproto v0.0.0-20211115083454-9ca41db5ed9e/go.mod h1:HEUYX/p8966tMUHHT+TsS0hF/Ca/NYwqprC5WXSDMfE=
|
||||
github.com/ydb-platform/ydb-go-genproto v0.0.0-20221215182650-986f9d10542f/go.mod h1:Er+FePu1dNUieD+XTMDduGpQuCPssK5Q4BjF+IIXJ3I=
|
||||
github.com/ydb-platform/ydb-go-genproto v0.0.0-20230528143953-42c825ace222/go.mod h1:Er+FePu1dNUieD+XTMDduGpQuCPssK5Q4BjF+IIXJ3I=
|
||||
github.com/ydb-platform/ydb-go-genproto v0.0.0-20260428144813-1c07baab7f7b h1:xeiobG1riqe6dTMZuTcOvwOY0BbFidSk3gRLyVeLfJA=
|
||||
github.com/ydb-platform/ydb-go-genproto v0.0.0-20260428144813-1c07baab7f7b/go.mod h1:Er+FePu1dNUieD+XTMDduGpQuCPssK5Q4BjF+IIXJ3I=
|
||||
github.com/ydb-platform/ydb-go-genproto v0.0.0-20260810122915-65bfd5c4b705 h1:7VKlOrBIQ8L8acJ9wFCt6lWzLDbOhc05VLpL31743tU=
|
||||
github.com/ydb-platform/ydb-go-genproto v0.0.0-20260810122915-65bfd5c4b705/go.mod h1:Er+FePu1dNUieD+XTMDduGpQuCPssK5Q4BjF+IIXJ3I=
|
||||
github.com/ydb-platform/ydb-go-sdk-auth-environ v0.5.2 h1:e2nGQPGC5OEPBWlMnLPFpdgyqNA8iTOPiZShCrv6944=
|
||||
github.com/ydb-platform/ydb-go-sdk-auth-environ v0.5.2/go.mod h1:9YzkhlIymWaJGX6KMU3vh5sOf3UKbCXkG/ZdjaI3zNM=
|
||||
github.com/ydb-platform/ydb-go-sdk/v3 v3.44.0/go.mod h1:oSLwnuilwIpaF5bJJMAofnGgzPJusoI3zWMNb8I+GnM=
|
||||
github.com/ydb-platform/ydb-go-sdk/v3 v3.47.3/go.mod h1:bWnOIcUHd7+Sl7DN+yhyY1H/I61z53GczvwJgXMgvj0=
|
||||
github.com/ydb-platform/ydb-go-sdk/v3 v3.147.1 h1:WRxyl1UdFD6EmFlujwrHm4sX9ktYb/adZ/SW+WWNOEg=
|
||||
github.com/ydb-platform/ydb-go-sdk/v3 v3.147.1/go.mod h1:b9NEO6mgaiqsnOMkS003uS82XsKh6GL+ZTFfPqXWz+c=
|
||||
github.com/ydb-platform/ydb-go-sdk/v3 v3.151.1 h1:T+fB2ZDHpYIGC7DWjK+rzZHLoexBF0zG/n3Q9DGNskY=
|
||||
github.com/ydb-platform/ydb-go-sdk/v3 v3.151.1/go.mod h1:dJXJ1u00IqO8Vsph8fWmYj8b1E7jyphnvJPqA69emBY=
|
||||
github.com/ydb-platform/ydb-go-yc v0.12.1 h1:qw3Fa+T81+Kpu5Io2vYHJOwcrYrVjgJlT6t/0dOXJrA=
|
||||
github.com/ydb-platform/ydb-go-yc v0.12.1/go.mod h1:t/ZA4ECdgPWjAb4jyDe8AzQZB5dhpGbi3iCahFaNwBY=
|
||||
github.com/ydb-platform/ydb-go-yc-metadata v0.6.1 h1:9E5q8Nsy2RiJMZDNVy0A3KUrIMBPakJ2VgloeWbcI84=
|
||||
@@ -2657,8 +2659,8 @@ google.golang.org/api v0.106.0/go.mod h1:2Ts0XTHNVWxypznxWOYUeI4g3WdP9Pk2Qk58+a/
|
||||
google.golang.org/api v0.107.0/go.mod h1:2Ts0XTHNVWxypznxWOYUeI4g3WdP9Pk2Qk58+a/O9MY=
|
||||
google.golang.org/api v0.108.0/go.mod h1:2Ts0XTHNVWxypznxWOYUeI4g3WdP9Pk2Qk58+a/O9MY=
|
||||
google.golang.org/api v0.110.0/go.mod h1:7FC4Vvx1Mooxh8C5HWjzZHcavuS2f6pmJpZx60ca7iI=
|
||||
google.golang.org/api v0.293.0 h1:p9XIWOf63U4OgYx120ZwVU8+vl4XTPmWfgVPnmOAS9w=
|
||||
google.golang.org/api v0.293.0/go.mod h1:6n5tjEB1gzwniZTepZ0g5u+wM7Bof5GeULCx/zh8ZE0=
|
||||
google.golang.org/api v0.294.0 h1:8gASjJxdtcIieB3OqbkLcF0FfbXVNqKtU5iozD1ssvA=
|
||||
google.golang.org/api v0.294.0/go.mod h1:02qB8+Ox1ZFzcaKFMguy1nQLJmSIyvV6Ff4txJEXtl4=
|
||||
google.golang.org/appengine v1.1.0/go.mod h1:EbEs0AVv82hx2wNQdGPgUI5lhzA/G0D9YwlJXL52JkM=
|
||||
google.golang.org/appengine v1.4.0/go.mod h1:xpcJRLb0r/rnEns0DIKYYv+WjYCduHsrkT7/EB5XEv4=
|
||||
google.golang.org/appengine v1.5.0/go.mod h1:xpcJRLb0r/rnEns0DIKYYv+WjYCduHsrkT7/EB5XEv4=
|
||||
@@ -2792,12 +2794,12 @@ google.golang.org/genproto v0.0.0-20230209215440-0dfe4f8abfcc/go.mod h1:RGgjbofJ
|
||||
google.golang.org/genproto v0.0.0-20230216225411-c8e22ba71e44/go.mod h1:8B0gmkoRebU8ukX6HP+4wrVQUY1+6PkQ44BSyIlflHA=
|
||||
google.golang.org/genproto v0.0.0-20230222225845-10f96fb3dbec/go.mod h1:3Dl5ZL0q0isWJt+FVcfpQyirqemEuLAK/iFvg1UP1Hw=
|
||||
google.golang.org/genproto v0.0.0-20230306155012-7f2fa6fef1f4/go.mod h1:NWraEVixdDnqcqQ30jipen1STv2r/n24Wb7twVTGR4s=
|
||||
google.golang.org/genproto v0.0.0-20260519071638-aa98bba5eb94 h1:YJjbgu+dkp5kUJLfpMyCLfBIWZb/FcJyuLeo1gVBOuo=
|
||||
google.golang.org/genproto v0.0.0-20260519071638-aa98bba5eb94/go.mod h1:RRHjglSYABVCWpQ7USCpdfhcd9t4PkajvVwyynZizTc=
|
||||
google.golang.org/genproto/googleapis/api v0.0.0-20260706201446-f0a921348800 h1:admdQBe8jR3VWhBsUrAOaF2Qw6K/+p5pSm1GN8+6Fw4=
|
||||
google.golang.org/genproto/googleapis/api v0.0.0-20260706201446-f0a921348800/go.mod h1:FPk7EXUKMtImne7AmknoYjT4QXqKIzzRbeQIXzLk6fQ=
|
||||
google.golang.org/genproto/googleapis/rpc v0.0.0-20260807164820-c8921c73eeea h1:kVhQEPTpKQahD5+JSBTfBB19wcgQTTjAIn45MBqnyHk=
|
||||
google.golang.org/genproto/googleapis/rpc v0.0.0-20260807164820-c8921c73eeea/go.mod h1:4Hqkh8ycfw05ld/3BWL7rJOSfebL2Q+DVDeRgYgxUU8=
|
||||
google.golang.org/genproto v0.0.0-20260715232425-e75dac1f907d h1:C9v1o0/4quuhOAfmRXA2j+we0PqZIp8traLdeogF3Ms=
|
||||
google.golang.org/genproto v0.0.0-20260715232425-e75dac1f907d/go.mod h1:Wz2wFJntZFmLGo7pLDXZ3wYk5hyc0Mb+SkHhDDXT+lU=
|
||||
google.golang.org/genproto/googleapis/api v0.0.0-20260715232425-e75dac1f907d h1:QwnJwPte4XXAkhPu26LTDIahnsMSUV0kK8HkxbC+Pc4=
|
||||
google.golang.org/genproto/googleapis/api v0.0.0-20260715232425-e75dac1f907d/go.mod h1:WRrQ7/7N19PypuT0fxLOL5Lq0waoiRri4FbtHDEKrGE=
|
||||
google.golang.org/genproto/googleapis/rpc v0.0.0-20260819154853-08b0e4226688 h1:cYNAzI2sUwhmCcoj9TxvihSrqsxt6uIkj3rDRhSDmW4=
|
||||
google.golang.org/genproto/googleapis/rpc v0.0.0-20260819154853-08b0e4226688/go.mod h1:DjtHYE8FKJLivXcBEjGwndXfIC23G0VpXiXKqG179uA=
|
||||
google.golang.org/grpc v1.19.0/go.mod h1:mqu4LbDTu4XGKhr4mRzUsmM4RtVoemTSY81AxZiDr8c=
|
||||
google.golang.org/grpc v1.20.1/go.mod h1:10oTOabMzJvdu6/UiuZezV6QK5dSlG84ov/aaiqXj38=
|
||||
google.golang.org/grpc v1.21.1/go.mod h1:oYelfM1adQP15Ek0mdvEgi9Df8B9CZIaU1084ijfRaM=
|
||||
@@ -2838,8 +2840,8 @@ google.golang.org/grpc v1.51.0/go.mod h1:wgNDFcnuBGmxLKI/qn4T+m5BtEBYXJPvibbUPsA
|
||||
google.golang.org/grpc v1.52.0/go.mod h1:pu6fVzoFb+NBYNAvQL08ic+lvB2IojljRYuun5vorUY=
|
||||
google.golang.org/grpc v1.53.0/go.mod h1:OnIrk0ipVdj4N5d9IUoFUx72/VlD7+jUsHwZgwSMQpw=
|
||||
google.golang.org/grpc v1.55.0/go.mod h1:iYEXKGkEBhg1PjZQvoYEVPTDkHo1/bjTnfwTeGONTY8=
|
||||
google.golang.org/grpc v1.84.0-dev.0.20260723093437-b6eac429d7b6 h1:HfjjkdGIa8u9sP9EW5WCygy0kQDuTI/Tax4j//t24Fo=
|
||||
google.golang.org/grpc v1.84.0-dev.0.20260723093437-b6eac429d7b6/go.mod h1:ljCht0DrxQrXBDRTZp52Qxh3Ffk8CdYm2sj4O2QN2C0=
|
||||
google.golang.org/grpc v1.85.0-dev h1:HxkDyKIIZPpFnroC56tQv5gNuKTmVvi0t7TzOf5zt7g=
|
||||
google.golang.org/grpc v1.85.0-dev/go.mod h1:ljCht0DrxQrXBDRTZp52Qxh3Ffk8CdYm2sj4O2QN2C0=
|
||||
google.golang.org/grpc/cmd/protoc-gen-go-grpc v1.1.0/go.mod h1:6Kw0yEErY5E/yWrBtf03jp27GLLJujG4z/JK95pnjjw=
|
||||
google.golang.org/grpc/examples v0.0.0-20250407062114-b368379ef8f6 h1:ExN12ndbJ608cboPYflpTny6mXSzPrDLh0iTaVrRrds=
|
||||
google.golang.org/grpc/examples v0.0.0-20250407062114-b368379ef8f6/go.mod h1:6ytKWczdvnpnO+m+JiG9NjEDzR1FJfsnmJdG7B8QVZ8=
|
||||
@@ -2861,8 +2863,8 @@ google.golang.org/protobuf v1.27.1/go.mod h1:9q0QmTI4eRPtz6boOQmLYwt+qCgq0jsYwAQ
|
||||
google.golang.org/protobuf v1.28.0/go.mod h1:HV8QOd/L58Z+nl8r43ehVNZIU/HEI6OcFqwMG9pJV4I=
|
||||
google.golang.org/protobuf v1.28.1/go.mod h1:HV8QOd/L58Z+nl8r43ehVNZIU/HEI6OcFqwMG9pJV4I=
|
||||
google.golang.org/protobuf v1.30.0/go.mod h1:HV8QOd/L58Z+nl8r43ehVNZIU/HEI6OcFqwMG9pJV4I=
|
||||
google.golang.org/protobuf v1.36.11 h1:fV6ZwhNocDyBLK0dj+fg8ektcVegBBuEolpbTQyBNVE=
|
||||
google.golang.org/protobuf v1.36.11/go.mod h1:HTf+CrKn2C3g5S8VImy6tdcUvCska2kB7j23XfzDpco=
|
||||
google.golang.org/protobuf v1.36.12 h1:pJOKDDOyeXErUroCihFAd5LQuwXBSpVnKGrj5o/fwxc=
|
||||
google.golang.org/protobuf v1.36.12/go.mod h1:HTf+CrKn2C3g5S8VImy6tdcUvCska2kB7j23XfzDpco=
|
||||
gopkg.in/alecthomas/kingpin.v2 v2.2.6/go.mod h1:FMv+mEhP44yOT+4EoQTLFTRgOQ1FBLkstjWtayDeSgw=
|
||||
gopkg.in/check.v1 v0.0.0-20161208181325-20d25e280405/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0=
|
||||
gopkg.in/check.v1 v1.0.0-20180628173108-788fd7840127/go.mod h1:Co6ibVJAznAaIkqp8huTwlJQCZ016jof/cbN4VW5Yz0=
|
||||
@@ -2962,8 +2964,8 @@ modernc.org/opt v0.2.0/go.mod h1:03fq9lsNfvkYSfxrfUhZCWPk1lm4cq4N+Bh//bEtgns=
|
||||
modernc.org/sortutil v1.2.1 h1:+xyoGf15mM3NMlPDnFqrteY07klSFxLElE2PVuWIJ7w=
|
||||
modernc.org/sortutil v1.2.1/go.mod h1:7ZI3a3REbai7gzCLcotuw9AC4VZVpYMjDzETGsSMqJE=
|
||||
modernc.org/sqlite v1.18.1/go.mod h1:6ho+Gow7oX5V+OiOQ6Tr4xeqbx13UZ6t+Fw9IRUG4d4=
|
||||
modernc.org/sqlite v1.56.0 h1:/D8e2RfFqoy/Zc6PuC76U28zFwmI/sYx1Kjm4yEn9e0=
|
||||
modernc.org/sqlite v1.56.0/go.mod h1:yCJ2cmAaIkHQ25oXWrF8H4O1lIfPYPR26yCEDj2P3pQ=
|
||||
modernc.org/sqlite v1.57.0 h1:qNQP6xnx5M0ISNtlnxoOX0+cD5bJ0/gr9aMmndFczzg=
|
||||
modernc.org/sqlite v1.57.0/go.mod h1:yCJ2cmAaIkHQ25oXWrF8H4O1lIfPYPR26yCEDj2P3pQ=
|
||||
modernc.org/strutil v1.1.0/go.mod h1:lstksw84oURvj9y3tn8lGvRxyRC1S2+g5uuIzNfIOBs=
|
||||
modernc.org/strutil v1.1.1/go.mod h1:DE+MQQ/hjKBZS2zNInV5hhcipt5rLPWkmpbGeW5mmdw=
|
||||
modernc.org/strutil v1.1.3/go.mod h1:MEHNA7PdEnEwLvspRMtWTNnp2nnyvMfkimT1NKNAGbw=
|
||||
|
||||
+84
-2
@@ -9,7 +9,7 @@
|
||||
# curl -fsSL ... | bash -s -- --version 4.34 --dir /usr/local/bin
|
||||
#
|
||||
# Options:
|
||||
# --component COMP Which binary to install: weed, volume-rust, all (default: weed)
|
||||
# --component COMP Which binary to install: weed, volume-rust, worker-rust, all (default: weed)
|
||||
# --version VER Release version tag (default: latest)
|
||||
# --large-disk Use large disk variant (5-byte offset, 8TB max volume)
|
||||
# --dir DIR Installation directory (default: /usr/local/bin)
|
||||
@@ -22,6 +22,7 @@ COMPONENT="weed"
|
||||
VERSION=""
|
||||
LARGE_DISK=false
|
||||
INSTALL_DIR="/usr/local/bin"
|
||||
WORKER_INSTALLED=false
|
||||
|
||||
# Colors (if terminal supports them)
|
||||
if [ -t 1 ]; then
|
||||
@@ -128,6 +129,33 @@ rust_asset_name() {
|
||||
fi
|
||||
}
|
||||
|
||||
# Does a release carry this asset? Answers 0 for yes and 1 for a 404, and stops
|
||||
# the installer on anything else: a rate limit or a network blip must not read
|
||||
# as "this release predates the binary" and quietly skip it.
|
||||
asset_exists() {
|
||||
local url="$1" code=""
|
||||
if command -v curl &>/dev/null; then
|
||||
code="$(curl -sL -o /dev/null -I -w '%{http_code}' "$url" || true)"
|
||||
elif command -v wget &>/dev/null; then
|
||||
# --spider's exit status folds 404 in with every other server error, so
|
||||
# read the status line itself; -S prints one per redirect hop.
|
||||
code="$(wget -S --spider -q -O /dev/null "$url" 2>&1 | awk '/^ *HTTP\// {c=$2} END {print c}')"
|
||||
fi
|
||||
|
||||
case "$code" in
|
||||
200) return 0 ;;
|
||||
404) return 1 ;;
|
||||
*) error "Could not check ${url} (HTTP ${code:-none}). Retry, or install components one at a time." ;;
|
||||
esac
|
||||
}
|
||||
|
||||
# Build Rust maintenance worker asset name. No large-disk variant: the worker
|
||||
# maintains tables through the namespace and never opens a volume file.
|
||||
worker_asset_name() {
|
||||
local os="$1" arch="$2"
|
||||
echo "weed-worker_${os}_${arch}.tar.gz"
|
||||
}
|
||||
|
||||
# Install a single component
|
||||
install_component() {
|
||||
local component="$1" os="$2" arch="$3"
|
||||
@@ -204,10 +232,48 @@ install_component() {
|
||||
ok "Installed weed-volume to ${INSTALL_DIR}/${dest_name}"
|
||||
;;
|
||||
|
||||
worker-rust)
|
||||
# Published for linux only: the worker runs beside the cluster it
|
||||
# maintains, and its dependency tree makes every extra target an
|
||||
# expensive build.
|
||||
case "$os" in
|
||||
linux) ;;
|
||||
*) error "Rust maintenance worker is not available for ${os}. Supported: linux" ;;
|
||||
esac
|
||||
case "$arch" in
|
||||
amd64|arm64) ;;
|
||||
*) error "Rust maintenance worker is not available for ${arch}. Supported: amd64, arm64" ;;
|
||||
esac
|
||||
|
||||
asset_name="$(worker_asset_name "$os" "$arch")"
|
||||
download_url="https://github.com/${REPO}/releases/download/${VERSION}/${asset_name}"
|
||||
download "$download_url" "${tmpdir}/${asset_name}"
|
||||
|
||||
info "Extracting ${asset_name}..."
|
||||
tar xzf "${tmpdir}/${asset_name}" -C "$tmpdir"
|
||||
|
||||
local worker_bin
|
||||
worker_bin="$(find "$tmpdir" -name 'weed-worker' -type f | head -1)"
|
||||
if [ -z "$worker_bin" ]; then
|
||||
error "Could not find weed-worker binary in archive"
|
||||
fi
|
||||
|
||||
chmod +x "$worker_bin"
|
||||
install_binary "$worker_bin" "weed-worker"
|
||||
WORKER_INSTALLED=true
|
||||
ok "Installed weed-worker to ${INSTALL_DIR}/weed-worker"
|
||||
;;
|
||||
|
||||
*)
|
||||
error "Unknown component: ${component}. Use: weed, volume-rust, all"
|
||||
error "Unknown component: ${component}. Use: weed, volume-rust, worker-rust, all"
|
||||
;;
|
||||
esac
|
||||
|
||||
# The trap is per-process, so a later component's would replace this one and
|
||||
# leave the earlier extraction behind. Clean up here and hand the trap back;
|
||||
# the error paths above exit, which still fires it.
|
||||
rm -rf "$tmpdir"
|
||||
trap - EXIT
|
||||
}
|
||||
|
||||
# Copy binary to install dir, using sudo if needed
|
||||
@@ -251,6 +317,16 @@ main() {
|
||||
all)
|
||||
install_component "weed" "$os" "$arch"
|
||||
install_component "volume-rust" "$os" "$arch"
|
||||
# The worker is published for linux amd64/arm64 only, and only by
|
||||
# releases new enough to carry it; skip either case rather than fail
|
||||
# an install that has already put two binaries in place.
|
||||
if [ "$os" != "linux" ] || { [ "$arch" != "amd64" ] && [ "$arch" != "arm64" ]; }; then
|
||||
warn "Skipping the Rust maintenance worker: no build for ${os}/${arch}"
|
||||
elif ! asset_exists "https://github.com/${REPO}/releases/download/${VERSION}/$(worker_asset_name "$os" "$arch")"; then
|
||||
warn "Skipping the Rust maintenance worker: ${VERSION} does not carry one"
|
||||
else
|
||||
install_component "worker-rust" "$os" "$arch"
|
||||
fi
|
||||
;;
|
||||
*)
|
||||
install_component "$COMPONENT" "$os" "$arch"
|
||||
@@ -265,11 +341,17 @@ main() {
|
||||
if [ "$COMPONENT" = "volume-rust" ] || [ "$COMPONENT" = "all" ]; then
|
||||
info " weed-volume: ${INSTALL_DIR}/weed-volume"
|
||||
fi
|
||||
if [ "$WORKER_INSTALLED" = true ]; then
|
||||
info " weed-worker: ${INSTALL_DIR}/weed-worker"
|
||||
fi
|
||||
echo ""
|
||||
info "Quick start:"
|
||||
info " weed master # Start master server"
|
||||
info " weed volume -mserver=localhost:9333 # Start Go volume server"
|
||||
info " weed-volume -mserver localhost:9333 # Start Rust volume server"
|
||||
if [ "$WORKER_INSTALLED" = true ]; then
|
||||
info " weed-worker --admin localhost:23646 # Start the Rust maintenance worker"
|
||||
fi
|
||||
}
|
||||
|
||||
main
|
||||
|
||||
@@ -21,6 +21,7 @@ spec:
|
||||
{{- $nodePorts = .Values.admin.service.nodePorts | default dict }}
|
||||
{{- end }}
|
||||
type: {{ $serviceType }}
|
||||
{{- include "seaweedfs.service.loadBalancerFields" .Values.admin.service }}
|
||||
ports:
|
||||
- name: "http"
|
||||
port: {{ .Values.admin.port }}
|
||||
|
||||
@@ -21,6 +21,7 @@ spec:
|
||||
{{- $nodePorts = .Values.allInOne.service.nodePorts | default dict }}
|
||||
{{- end }}
|
||||
type: {{ $serviceType }}
|
||||
{{- include "seaweedfs.service.loadBalancerFields" .Values.allInOne.service }}
|
||||
internalTrafficPolicy: {{ .Values.allInOne.service.internalTrafficPolicy | default "Cluster" }}
|
||||
{{- if and (semverCompare ">=1.31-0" .Capabilities.KubeVersion.GitVersion) .Values.allInOne.s3.trafficDistribution }}
|
||||
trafficDistribution: {{ include "seaweedfs.trafficDistribution" (dict "value" .Values.allInOne.s3.trafficDistribution "Capabilities" .Capabilities) }}
|
||||
|
||||
@@ -149,6 +149,8 @@ spec:
|
||||
{{- if .Values.s3.icebergPort }}
|
||||
-port.iceberg={{ .Values.s3.icebergPort }} \
|
||||
{{- end }}
|
||||
{{- /* rendered even when 0: an if would drop the flag and weed would serve its default */}}
|
||||
-port.lance={{ .Values.s3.lancePort | default 0 }} \
|
||||
{{- range .Values.s3.extraArgs }}
|
||||
{{ . }} \
|
||||
{{- end }}
|
||||
@@ -198,6 +200,10 @@ spec:
|
||||
- containerPort: {{ .Values.s3.icebergPort }}
|
||||
name: swfs-iceberg
|
||||
{{- end }}
|
||||
{{- if .Values.s3.lancePort }}
|
||||
- containerPort: {{ .Values.s3.lancePort }}
|
||||
name: swfs-lance
|
||||
{{- end }}
|
||||
{{- if .Values.s3.metricsPort }}
|
||||
- containerPort: {{ .Values.s3.metricsPort }}
|
||||
name: metrics
|
||||
|
||||
@@ -0,0 +1,61 @@
|
||||
{{- define "seaweedfs.s3.lance.ingress.paths" -}}
|
||||
paths:
|
||||
- path: {{ .Values.s3.lanceIngress.path | quote }}
|
||||
pathType: {{ .Values.s3.lanceIngress.pathType | quote }}
|
||||
backend:
|
||||
{{- if semverCompare ">=1.19-0" .Capabilities.KubeVersion.GitVersion }}
|
||||
service:
|
||||
name: {{ include "seaweedfs.componentName" (list . "s3") }}
|
||||
port:
|
||||
number: {{ .Values.s3.lancePort }}
|
||||
{{- else }}
|
||||
serviceName: {{ include "seaweedfs.componentName" (list . "s3") }}
|
||||
servicePort: {{ .Values.s3.lancePort }}
|
||||
{{- end }}
|
||||
{{- end -}}
|
||||
{{- if and .Values.s3.enabled .Values.s3.lancePort .Values.s3.lanceIngress.enabled }}
|
||||
{{- $hosts := list }}
|
||||
{{- if kindIs "slice" .Values.s3.lanceIngress.host }}
|
||||
{{- $hosts = .Values.s3.lanceIngress.host }}
|
||||
{{- else if .Values.s3.lanceIngress.host }}
|
||||
{{- $hosts = list .Values.s3.lanceIngress.host }}
|
||||
{{- end }}
|
||||
{{- if semverCompare ">=1.19-0" .Capabilities.KubeVersion.GitVersion }}
|
||||
apiVersion: networking.k8s.io/v1
|
||||
{{- else if semverCompare ">=1.14-0" .Capabilities.KubeVersion.GitVersion }}
|
||||
apiVersion: networking.k8s.io/v1beta1
|
||||
{{- else }}
|
||||
apiVersion: extensions/v1beta1
|
||||
{{- end }}
|
||||
kind: Ingress
|
||||
metadata:
|
||||
name: ingress-{{ include "seaweedfs.fullname" . }}-s3-lance
|
||||
namespace: {{ .Release.Namespace }}
|
||||
{{- with .Values.s3.lanceIngress.annotations }}
|
||||
annotations:
|
||||
{{- toYaml . | nindent 4 }}
|
||||
{{- end }}
|
||||
labels:
|
||||
app.kubernetes.io/name: {{ template "seaweedfs.name" . }}
|
||||
helm.sh/chart: {{ .Chart.Name }}-{{ .Chart.Version | replace "+" "_" }}
|
||||
app.kubernetes.io/managed-by: {{ .Release.Service }}
|
||||
app.kubernetes.io/instance: {{ .Release.Name }}
|
||||
app.kubernetes.io/component: s3-lance
|
||||
spec:
|
||||
{{- if .Values.s3.lanceIngress.className }}
|
||||
ingressClassName: {{ .Values.s3.lanceIngress.className | quote }}
|
||||
{{- end }}
|
||||
tls:
|
||||
{{ .Values.s3.lanceIngress.tls | default list | toYaml | nindent 6}}
|
||||
rules:
|
||||
{{- if $hosts }}
|
||||
{{- range $host := $hosts }}
|
||||
- host: {{ $host | quote }}
|
||||
http:
|
||||
{{- include "seaweedfs.s3.lance.ingress.paths" $ | nindent 6 }}
|
||||
{{- end }}
|
||||
{{- else }}
|
||||
- http:
|
||||
{{- include "seaweedfs.s3.lance.ingress.paths" . | nindent 4 }}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
@@ -21,6 +21,7 @@ spec:
|
||||
{{- $nodePorts = .Values.s3.service.nodePorts | default dict }}
|
||||
{{- end }}
|
||||
type: {{ $serviceType }}
|
||||
{{- include "seaweedfs.service.loadBalancerFields" .Values.s3.service }}
|
||||
internalTrafficPolicy: {{ .Values.s3.internalTrafficPolicy | default "Cluster" }}
|
||||
{{- $td := .Values.s3.trafficDistribution | default .Values.filer.s3.trafficDistribution }}
|
||||
{{- if and (semverCompare ">=1.31-0" .Capabilities.KubeVersion.GitVersion) $td }}
|
||||
@@ -43,6 +44,15 @@ spec:
|
||||
{{- end }}
|
||||
protocol: TCP
|
||||
{{- end }}
|
||||
{{- if and .Values.s3.enabled .Values.s3.lancePort }}
|
||||
- name: "swfs-lance"
|
||||
port: {{ .Values.s3.lancePort }}
|
||||
targetPort: {{ .Values.s3.lancePort }}
|
||||
{{- if $nodePorts.lance }}
|
||||
nodePort: {{ $nodePorts.lance }}
|
||||
{{- end }}
|
||||
protocol: TCP
|
||||
{{- end }}
|
||||
{{- if and .Values.s3.enabled .Values.s3.httpsPort }}
|
||||
- name: "swfs-s3-tls"
|
||||
port: {{ .Values.s3.httpsPort }}
|
||||
|
||||
@@ -21,6 +21,7 @@ spec:
|
||||
{{- $nodePorts = .Values.sftp.service.nodePorts | default dict }}
|
||||
{{- end }}
|
||||
type: {{ $serviceType }}
|
||||
{{- include "seaweedfs.service.loadBalancerFields" .Values.sftp.service }}
|
||||
internalTrafficPolicy: {{ .Values.sftp.internalTrafficPolicy | default "Cluster" }}
|
||||
ports:
|
||||
- name: "swfs-sftp"
|
||||
|
||||
@@ -148,6 +148,15 @@ true
|
||||
{{- end -}}
|
||||
{{- end -}}
|
||||
|
||||
{{/* Lance namespace URL the worker's Lance container maintains; empty when unreachable */}}
|
||||
{{- define "seaweedfs.worker.lanceNamespaceUrl" -}}
|
||||
{{- if .Values.worker.namespaceUrl -}}
|
||||
{{- .Values.worker.namespaceUrl -}}
|
||||
{{- else if and .Values.s3.enabled .Values.s3.lancePort -}}
|
||||
{{- printf "http://%s.%s:%d" (include "seaweedfs.componentName" (list . "s3")) .Release.Namespace (int .Values.s3.lancePort) -}}
|
||||
{{- end -}}
|
||||
{{- end -}}
|
||||
|
||||
{{/* Return the proper volume image */}}
|
||||
{{- define "seaweedfs.volume.image" -}}
|
||||
{{- if .Values.volume.imageOverride -}}
|
||||
@@ -548,3 +557,23 @@ true
|
||||
{{- and (eq .value "PreferClose") (semverCompare ">=1.35-0" .Capabilities.KubeVersion.GitVersion) | ternary "PreferSameZone" .value -}}
|
||||
{{- end -}}
|
||||
{{- end -}}
|
||||
|
||||
{{/*
|
||||
Render LoadBalancer-specific service fields (loadBalancerClass, loadBalancerIP,
|
||||
loadBalancerSourceRanges), only when the service type is LoadBalancer.
|
||||
Usage: {{ include "seaweedfs.service.loadBalancerFields" .Values.s3.service }}
|
||||
*/}}
|
||||
{{- define "seaweedfs.service.loadBalancerFields" -}}
|
||||
{{- if eq (.type | default "ClusterIP") "LoadBalancer" }}
|
||||
{{- with .loadBalancerClass }}
|
||||
loadBalancerClass: {{ . }}
|
||||
{{- end }}
|
||||
{{- with .loadBalancerIP }}
|
||||
loadBalancerIP: {{ . }}
|
||||
{{- end }}
|
||||
{{- with .loadBalancerSourceRanges }}
|
||||
loadBalancerSourceRanges:
|
||||
{{- toYaml . | nindent 4 }}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
{{- end -}}
|
||||
|
||||
@@ -73,6 +73,9 @@
|
||||
{{- if .Values.s3.icebergPort }}
|
||||
{{- $ports = append $ports .Values.s3.icebergPort }}
|
||||
{{- end }}
|
||||
{{- if .Values.s3.lancePort }}
|
||||
{{- $ports = append $ports .Values.s3.lancePort }}
|
||||
{{- end }}
|
||||
{{- if .Values.s3.metricsPort }}
|
||||
{{- $ports = append $ports .Values.s3.metricsPort }}
|
||||
{{- end }}
|
||||
@@ -97,6 +100,9 @@
|
||||
{{- if .Values.worker.metricsPort }}
|
||||
{{- $ports = append $ports .Values.worker.metricsPort }}
|
||||
{{- end }}
|
||||
{{- if and .Values.worker.lanceMetricsPort (include "seaweedfs.worker.lanceNamespaceUrl" .) }}
|
||||
{{- $ports = append $ports .Values.worker.lanceMetricsPort }}
|
||||
{{- end }}
|
||||
{{- $targets = append $targets (dict "component" "worker" "ports" $ports) }}
|
||||
{{- end }}
|
||||
|
||||
|
||||
@@ -223,6 +223,81 @@ spec:
|
||||
{{- if .Values.worker.containerSecurityContext.enabled }}
|
||||
securityContext: {{- omit .Values.worker.containerSecurityContext "enabled" | toYaml | nindent 12 }}
|
||||
{{- end }}
|
||||
{{- if include "seaweedfs.worker.lanceNamespaceUrl" . }}
|
||||
- name: worker-lance
|
||||
image: {{ template "seaweedfs.worker.image" . }}
|
||||
imagePullPolicy: {{ default "IfNotPresent" .Values.global.seaweedfs.imagePullPolicy }}
|
||||
env:
|
||||
- name: POD_NAME
|
||||
valueFrom:
|
||||
fieldRef:
|
||||
fieldPath: metadata.name
|
||||
{{- /* the URL crosses a shell line; metacharacters must arrive as data, not syntax */}}
|
||||
- name: LANCE_NAMESPACE_URL
|
||||
value: {{ include "seaweedfs.worker.lanceNamespaceUrl" . | quote }}
|
||||
command:
|
||||
- "/bin/sh"
|
||||
- "-ec"
|
||||
- |
|
||||
{{- /* the armv7/386 placeholder is empty; exec of it becomes the shell and exits 0 */}}
|
||||
if [ ! -s /usr/bin/weed-worker ]; then
|
||||
echo "the Rust worker is not available on this platform ($(uname -m)); it ships for amd64 and arm64" >&2
|
||||
exit 1
|
||||
fi
|
||||
exec /usr/bin/weed-worker \
|
||||
--id="$POD_NAME" \
|
||||
{{- if .Values.worker.adminServer }}
|
||||
--admin={{ .Values.worker.adminServer }} \
|
||||
{{- else }}
|
||||
--admin={{ template "seaweedfs.fullname" . }}-admin.{{ .Release.Namespace }}:{{ .Values.admin.port }}{{ if .Values.admin.grpcPort }}.{{ .Values.admin.grpcPort }}{{ end }} \
|
||||
{{- end }}
|
||||
--namespace="$LANCE_NAMESPACE_URL" \
|
||||
{{- if .Values.global.seaweedfs.enableSecurity }}
|
||||
--tls-ca=/usr/local/share/ca-certificates/ca/tls.crt \
|
||||
--tls-cert=/usr/local/share/ca-certificates/worker/tls.crt \
|
||||
--tls-key=/usr/local/share/ca-certificates/worker/tls.key \
|
||||
{{- end }}
|
||||
{{- if .Values.worker.lanceMetricsPort }}
|
||||
--metrics-port={{ .Values.worker.lanceMetricsPort }} \
|
||||
--metrics-ip=0.0.0.0 \
|
||||
{{- end }}
|
||||
--max-concurrency={{ .Values.worker.maxExecute }}
|
||||
{{- if .Values.global.seaweedfs.enableSecurity }}
|
||||
volumeMounts:
|
||||
- name: ca-cert
|
||||
readOnly: true
|
||||
mountPath: /usr/local/share/ca-certificates/ca/
|
||||
- name: worker-cert
|
||||
readOnly: true
|
||||
mountPath: /usr/local/share/ca-certificates/worker/
|
||||
{{- end }}
|
||||
{{- if .Values.worker.lanceMetricsPort }}
|
||||
ports:
|
||||
- containerPort: {{ .Values.worker.lanceMetricsPort }}
|
||||
name: lance-metrics
|
||||
livenessProbe:
|
||||
httpGet:
|
||||
path: /health
|
||||
port: lance-metrics
|
||||
initialDelaySeconds: 30
|
||||
periodSeconds: 60
|
||||
successThreshold: 1
|
||||
failureThreshold: 5
|
||||
timeoutSeconds: 10
|
||||
readinessProbe:
|
||||
httpGet:
|
||||
path: /ready
|
||||
port: lance-metrics
|
||||
initialDelaySeconds: 20
|
||||
periodSeconds: 15
|
||||
successThreshold: 1
|
||||
failureThreshold: 3
|
||||
timeoutSeconds: 10
|
||||
{{- end }}
|
||||
{{- if .Values.worker.containerSecurityContext.enabled }}
|
||||
securityContext: {{- omit .Values.worker.containerSecurityContext "enabled" | toYaml | nindent 12 }}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
{{- if .Values.worker.sidecars }}
|
||||
{{- include "seaweedfs.tplvalues.render" (dict "value" .Values.worker.sidecars "context" $) | nindent 8 }}
|
||||
{{- end }}
|
||||
|
||||
@@ -12,13 +12,22 @@ metadata:
|
||||
app.kubernetes.io/component: worker
|
||||
spec:
|
||||
clusterIP: None # Headless service
|
||||
{{- if .Values.worker.metricsPort }}
|
||||
{{- $lanceMetrics := and .Values.worker.lanceMetricsPort (include "seaweedfs.worker.lanceNamespaceUrl" .) }}
|
||||
{{- if or .Values.worker.metricsPort $lanceMetrics }}
|
||||
ports:
|
||||
{{- if .Values.worker.metricsPort }}
|
||||
- name: "metrics"
|
||||
port: {{ .Values.worker.metricsPort }}
|
||||
targetPort: {{ .Values.worker.metricsPort }}
|
||||
protocol: TCP
|
||||
{{- end }}
|
||||
{{- if $lanceMetrics }}
|
||||
- name: "lance-metrics"
|
||||
port: {{ .Values.worker.lanceMetricsPort }}
|
||||
targetPort: {{ .Values.worker.lanceMetricsPort }}
|
||||
protocol: TCP
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
selector:
|
||||
app.kubernetes.io/name: {{ template "seaweedfs.name" . }}
|
||||
app.kubernetes.io/instance: {{ .Release.Name }}
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
{{- include "seaweedfs.compat" . -}}
|
||||
{{- if .Values.worker.enabled }}
|
||||
{{- if .Values.worker.metricsPort }}
|
||||
{{- $lanceMetrics := and .Values.worker.lanceMetricsPort (include "seaweedfs.worker.lanceNamespaceUrl" .) }}
|
||||
{{- if or .Values.worker.metricsPort $lanceMetrics }}
|
||||
{{- if .Values.global.seaweedfs.monitoring.enabled }}
|
||||
apiVersion: monitoring.coreos.com/v1
|
||||
kind: ServiceMonitor
|
||||
@@ -22,9 +23,16 @@ metadata:
|
||||
{{- end }}
|
||||
spec:
|
||||
endpoints:
|
||||
{{- if .Values.worker.metricsPort }}
|
||||
- interval: 30s
|
||||
port: metrics
|
||||
scrapeTimeout: 5s
|
||||
{{- end }}
|
||||
{{- if $lanceMetrics }}
|
||||
- interval: 30s
|
||||
port: lance-metrics
|
||||
scrapeTimeout: 5s
|
||||
{{- end }}
|
||||
selector:
|
||||
matchLabels:
|
||||
app.kubernetes.io/name: {{ template "seaweedfs.name" . }}
|
||||
|
||||
@@ -1006,6 +1006,9 @@ s3:
|
||||
# Iceberg catalog REST port (Apache Iceberg REST Catalog API)
|
||||
# Set to a port number to enable, or 0/null to disable
|
||||
icebergPort: null
|
||||
# Lance Namespace port; weed serves 9101 by default, 0 disables it
|
||||
# (and, unless worker.namespaceUrl points elsewhere, the worker's Lance container)
|
||||
lancePort: 9101
|
||||
loggingOverrideLevel: null
|
||||
# enable user & permission to s3 (need to inject to all services)
|
||||
enableAuth: false
|
||||
@@ -1184,11 +1187,16 @@ s3:
|
||||
# Service settings
|
||||
service:
|
||||
type: ClusterIP
|
||||
# used only when type is LoadBalancer
|
||||
loadBalancerClass: ""
|
||||
loadBalancerIP: ""
|
||||
loadBalancerSourceRanges: []
|
||||
# fixed nodePorts, used only when type is NodePort or LoadBalancer
|
||||
nodePorts:
|
||||
http: null
|
||||
https: null
|
||||
iceberg: null
|
||||
lance: null
|
||||
metrics: null
|
||||
|
||||
icebergIngress:
|
||||
@@ -1200,6 +1208,15 @@ s3:
|
||||
annotations: {}
|
||||
tls: []
|
||||
|
||||
lanceIngress:
|
||||
enabled: false
|
||||
className: ""
|
||||
host: "seaweedfs-lance.cluster.local"
|
||||
path: "/"
|
||||
pathType: Prefix
|
||||
annotations: {}
|
||||
tls: []
|
||||
|
||||
sftp:
|
||||
enabled: false
|
||||
imageOverride: null
|
||||
@@ -1288,6 +1305,10 @@ sftp:
|
||||
# Service settings
|
||||
service:
|
||||
type: ClusterIP
|
||||
# used only when type is LoadBalancer
|
||||
loadBalancerClass: ""
|
||||
loadBalancerIP: ""
|
||||
loadBalancerSourceRanges: []
|
||||
# fixed nodePorts, used only when type is NodePort or LoadBalancer
|
||||
nodePorts:
|
||||
sftp: null
|
||||
@@ -1430,6 +1451,10 @@ admin:
|
||||
service:
|
||||
type: ClusterIP
|
||||
annotations: {}
|
||||
# used only when type is LoadBalancer
|
||||
loadBalancerClass: ""
|
||||
loadBalancerIP: ""
|
||||
loadBalancerSourceRanges: []
|
||||
# fixed nodePorts, used only when type is NodePort or LoadBalancer
|
||||
nodePorts:
|
||||
http: null
|
||||
@@ -1448,6 +1473,15 @@ worker:
|
||||
metricsPort: 9327
|
||||
metricsIp: "" # If empty, defaults to 0.0.0.0
|
||||
|
||||
# The lance_* jobs run in their own container, /usr/bin/weed-worker.
|
||||
# amd64/arm64 only; pin mixed clusters with worker.affinity/nodeSelector.
|
||||
|
||||
# Lance namespace URL override; empty derives it from s3.lancePort.
|
||||
namespaceUrl: ""
|
||||
|
||||
# Metrics port for the Lance worker container; the Go worker keeps metricsPort
|
||||
lanceMetricsPort: 9328
|
||||
|
||||
# Admin server to connect to
|
||||
adminServer: ""
|
||||
|
||||
@@ -1654,6 +1688,10 @@ allInOne:
|
||||
annotations: {} # Annotations for the service
|
||||
type: ClusterIP # Service type (ClusterIP, NodePort, LoadBalancer)
|
||||
internalTrafficPolicy: Cluster # Internal traffic policy
|
||||
# used only when type is LoadBalancer
|
||||
loadBalancerClass: ""
|
||||
loadBalancerIP: ""
|
||||
loadBalancerSourceRanges: []
|
||||
# fixed nodePorts, used only when type is NodePort or LoadBalancer
|
||||
nodePorts:
|
||||
master: null
|
||||
|
||||
@@ -27,6 +27,8 @@ service Seaweed {
|
||||
}
|
||||
rpc VolumeList (VolumeListRequest) returns (VolumeListResponse) {
|
||||
}
|
||||
rpc VolumeListStream (VolumeListRequest) returns (stream VolumeListStreamResponse) {
|
||||
}
|
||||
rpc LookupEcVolume (LookupEcVolumeRequest) returns (LookupEcVolumeResponse) {
|
||||
}
|
||||
rpc VacuumVolume (VacuumVolumeRequest) returns (VacuumVolumeResponse) {
|
||||
@@ -406,12 +408,44 @@ message TopologyInfo {
|
||||
map<string, DiskInfo> diskInfos = 3;
|
||||
}
|
||||
message VolumeListRequest {
|
||||
// Empty and zero take everything. Only the volumes and ec shards listed
|
||||
// under a disk are selected; the topology and its disk counters are always
|
||||
// reported in full.
|
||||
string collection = 1;
|
||||
repeated uint32 volume_ids = 2;
|
||||
// The one collection the empty string cannot name. A named collection wins.
|
||||
bool default_collection_only = 3;
|
||||
// Empty and zero take everything. Wildcards are supported.
|
||||
string remote_storage_name = 4;
|
||||
bool local_volume_only = 5;
|
||||
// The topology, its disks and their counters alone, without the volumes
|
||||
// and ec shards. Selecting volumes above contradicts this and is refused.
|
||||
// A master that predates this field ignores it and answers in full.
|
||||
bool topology_only = 6;
|
||||
}
|
||||
message VolumeListResponse {
|
||||
TopologyInfo topology_info = 1;
|
||||
uint64 volume_size_limit_mb = 2;
|
||||
}
|
||||
|
||||
// VolumeListStream answers the same request as VolumeList without either end
|
||||
// holding every volume in the cluster at once. At 800k volumes the reply is
|
||||
// 36MB on the wire but 305MB as messages, which the master built in full
|
||||
// before sending any of it.
|
||||
message VolumeListStreamResponse {
|
||||
// Sent once, first, listing no volumes: the topology, its disks and their
|
||||
// counters. Every message after carries volumes for one of those disks.
|
||||
VolumeListResponse header = 1;
|
||||
// Which disk this batch is from. A disk arrives over as many batches as it
|
||||
// takes, so append rather than assign.
|
||||
string data_center = 2;
|
||||
string rack = 3;
|
||||
string data_node = 4;
|
||||
string disk_type = 5;
|
||||
repeated VolumeInformationMessage volume_infos = 6;
|
||||
repeated VolumeEcShardInformationMessage ec_shard_infos = 7;
|
||||
}
|
||||
|
||||
message LookupEcVolumeRequest {
|
||||
uint32 volume_id = 1;
|
||||
}
|
||||
|
||||
@@ -539,6 +539,7 @@ message VolumeEcShardsInfoResponse {
|
||||
uint64 volume_size = 2;
|
||||
uint64 file_count = 3;
|
||||
uint64 file_deleted_count = 4;
|
||||
EcShardConfig ec_shard_config = 10; // the layout this holder serves reads through; a binary predating a field reports it as unset
|
||||
}
|
||||
|
||||
message EcShardInfo {
|
||||
@@ -612,6 +613,7 @@ message EcShardConfig {
|
||||
uint32 data_shards = 1; // Number of data shards (e.g., 10)
|
||||
uint32 parity_shards = 2; // Number of parity shards (e.g., 4)
|
||||
int64 encode_ts_ns = 3; // encode time (unix nanos); a read served from a shard of a different encode run is rejected
|
||||
int64 block_size = 4; // uniform block layout: each shard is a single contiguous block of this many bytes; 0 = legacy 1GiB/1MiB two-tier layout
|
||||
}
|
||||
// EcBitrotProtection is the entire content of a bitrot checksum sidecar
|
||||
// (<base>.ecsum for the legacy generation, <base>.ecsum.v<N> for vacuum
|
||||
@@ -714,6 +716,7 @@ enum VolumeScrubMode {
|
||||
FULL = 2;
|
||||
LOCAL = 3;
|
||||
CHECKSUM = 4; // EC only: verify each local shard's raw bytes against the bitrot checksum sidecar
|
||||
READS = 5; // like FULL, but EC intervals no shard can serve are reconstructed from parity
|
||||
}
|
||||
|
||||
message ScrubVolumeRequest {
|
||||
|
||||
@@ -38,6 +38,7 @@ fn scrub_mode_label(mode: i32) -> &'static str {
|
||||
2 => "FULL",
|
||||
3 => "LOCAL",
|
||||
4 => "CHECKSUM",
|
||||
5 => "READS",
|
||||
_ => "UNKNOWN",
|
||||
}
|
||||
}
|
||||
@@ -91,6 +92,57 @@ pub fn load_state_file(
|
||||
volume_server_pb::VolumeServerState::decode(data.as_slice()).ok()
|
||||
}
|
||||
|
||||
/// One disk location's stake in an EC volume, as seen by the rebuild handler.
|
||||
struct LocInfo {
|
||||
dir: String,
|
||||
idx_dir: String,
|
||||
shard_count: usize,
|
||||
has_ecx: bool,
|
||||
}
|
||||
|
||||
/// Picks the location a rebuild should write into — the one holding an `.ecx`
|
||||
/// and the most shards — and returns every other directory it may have to read
|
||||
/// from. Shards are only half of what the rebuild needs: a split
|
||||
/// `-dir`/`-dir.idx` layout keeps `.ecx`/`.ecj`/`.vif` with the INDEX, and on a
|
||||
/// multi-disk server the chosen disk may hold nothing but shards while this
|
||||
/// volume's `.vif` or generation-0 `.ecsum` sits on a sibling. Miss those and
|
||||
/// the layout resolution falls back to 10+4 with the legacy striping and
|
||||
/// reconstructs through the wrong matrix, so both directories of every other
|
||||
/// location are listed. The rebuild's own two are passed separately by the
|
||||
/// caller and dropped here, along with empties and duplicates.
|
||||
///
|
||||
/// Returns `None` when no location holds an `.ecx`, i.e. there is nothing to
|
||||
/// rebuild from.
|
||||
fn select_rebuild_location(loc_infos: &[LocInfo]) -> Option<(usize, Vec<String>)> {
|
||||
let mut rebuild_loc_idx: Option<usize> = None;
|
||||
let mut other_dirs: Vec<String> = Vec::new();
|
||||
|
||||
for (i, info) in loc_infos.iter().enumerate() {
|
||||
let better = info.has_ecx
|
||||
&& rebuild_loc_idx
|
||||
.is_none_or(|prev| info.shard_count > loc_infos[prev].shard_count);
|
||||
if better {
|
||||
if let Some(prev) = rebuild_loc_idx {
|
||||
other_dirs.push(loc_infos[prev].dir.clone());
|
||||
other_dirs.push(loc_infos[prev].idx_dir.clone());
|
||||
}
|
||||
rebuild_loc_idx = Some(i);
|
||||
} else {
|
||||
other_dirs.push(info.dir.clone());
|
||||
other_dirs.push(info.idx_dir.clone());
|
||||
}
|
||||
}
|
||||
|
||||
let rebuild_loc_idx = rebuild_loc_idx?;
|
||||
let rebuild_dir = &loc_infos[rebuild_loc_idx].dir;
|
||||
let rebuild_idx_dir = &loc_infos[rebuild_loc_idx].idx_dir;
|
||||
other_dirs.retain(|d| !d.is_empty() && d != rebuild_dir && d != rebuild_idx_dir);
|
||||
other_dirs.sort();
|
||||
other_dirs.dedup();
|
||||
|
||||
Some((rebuild_loc_idx, other_dirs))
|
||||
}
|
||||
|
||||
struct WriteThrottler {
|
||||
bytes_per_second: i64,
|
||||
last_size_counter: i64,
|
||||
@@ -1768,11 +1820,14 @@ impl VolumeServer for VolumeGrpcService {
|
||||
}
|
||||
Some(store.locations[info.disk_id as usize].directory.clone())
|
||||
} else {
|
||||
// The mounted-volume refusal above means no disk
|
||||
// holds an in-memory claim here.
|
||||
store
|
||||
.find_ec_shard_target_location(
|
||||
&info.collection,
|
||||
vid,
|
||||
DATA_SHARDS_COUNT as u32,
|
||||
&[],
|
||||
)
|
||||
.map(|i| store.locations[i].directory.clone())
|
||||
};
|
||||
@@ -2382,13 +2437,21 @@ impl VolumeServer for VolumeGrpcService {
|
||||
)
|
||||
};
|
||||
|
||||
// Check existing .vif for EC shard config (matching Go's MaybeLoadVolumeInfo)
|
||||
let (data_shards, parity_shards) =
|
||||
// Check existing .vif for EC shard config (matching Go's MaybeLoadVolumeInfo).
|
||||
// The block size is recomputed by the encode for the current .dat, so
|
||||
// only the ratio is carried over from a prior config.
|
||||
let (data_shards, parity_shards, _) =
|
||||
crate::storage::erasure_coding::ec_volume::read_ec_shard_config(
|
||||
&dir, &idx_dir, collection, vid,
|
||||
);
|
||||
)
|
||||
.map_err(|e| {
|
||||
tonic::Status::internal(format!(
|
||||
"read ec shard config for volume {}: {}",
|
||||
vid.0, e
|
||||
))
|
||||
})?;
|
||||
|
||||
if let Err(e) = crate::storage::erasure_coding::ec_encoder::write_ec_files(
|
||||
let block_size = match crate::storage::erasure_coding::ec_encoder::write_ec_files(
|
||||
&dir,
|
||||
&idx_dir,
|
||||
collection,
|
||||
@@ -2396,16 +2459,19 @@ impl VolumeServer for VolumeGrpcService {
|
||||
data_shards as usize,
|
||||
parity_shards as usize,
|
||||
) {
|
||||
// Cleanup partially-created .ecNN and .ecx files on failure (matching Go defer)
|
||||
let base = crate::storage::volume::volume_file_name(&dir, collection, vid);
|
||||
let total_shards = data_shards + parity_shards;
|
||||
for i in 0..total_shards {
|
||||
let shard_path = format!("{}.ec{:02}", base, i);
|
||||
let _ = std::fs::remove_file(&shard_path);
|
||||
Ok(block_size) => block_size,
|
||||
Err(e) => {
|
||||
// Cleanup partially-created .ecNN and .ecx files on failure (matching Go defer)
|
||||
let base = crate::storage::volume::volume_file_name(&dir, collection, vid);
|
||||
let total_shards = data_shards + parity_shards;
|
||||
for i in 0..total_shards {
|
||||
let shard_path = format!("{}.ec{:02}", base, i);
|
||||
let _ = std::fs::remove_file(&shard_path);
|
||||
}
|
||||
let _ = std::fs::remove_file(format!("{}.ecx", base));
|
||||
return Err(Status::internal(e.to_string()));
|
||||
}
|
||||
let _ = std::fs::remove_file(format!("{}.ecx", base));
|
||||
return Err(Status::internal(e.to_string()));
|
||||
}
|
||||
};
|
||||
|
||||
// Write .vif file with EC shard metadata
|
||||
{
|
||||
@@ -2424,6 +2490,7 @@ impl VolumeServer for VolumeGrpcService {
|
||||
.duration_since(std::time::UNIX_EPOCH)
|
||||
.unwrap_or_default()
|
||||
.as_nanos() as i64,
|
||||
block_size,
|
||||
}),
|
||||
..Default::default()
|
||||
};
|
||||
@@ -2457,13 +2524,6 @@ impl VolumeServer for VolumeGrpcService {
|
||||
format!("{}_{}", collection, vid.0)
|
||||
};
|
||||
|
||||
struct LocInfo {
|
||||
dir: String,
|
||||
idx_dir: String,
|
||||
shard_count: usize,
|
||||
has_ecx: bool,
|
||||
}
|
||||
|
||||
let store = self.state.store.read().unwrap();
|
||||
let mut loc_infos: Vec<LocInfo> = Vec::new();
|
||||
|
||||
@@ -2511,26 +2571,8 @@ impl VolumeServer for VolumeGrpcService {
|
||||
));
|
||||
}
|
||||
|
||||
// Pick rebuild location: has .ecx and most shards
|
||||
let mut rebuild_loc_idx: Option<usize> = None;
|
||||
let mut other_dirs: Vec<String> = Vec::new();
|
||||
|
||||
for (i, info) in loc_infos.iter().enumerate() {
|
||||
if info.has_ecx
|
||||
&& (rebuild_loc_idx.is_none()
|
||||
|| info.shard_count > loc_infos[rebuild_loc_idx.unwrap()].shard_count)
|
||||
{
|
||||
if let Some(prev) = rebuild_loc_idx {
|
||||
other_dirs.push(loc_infos[prev].dir.clone());
|
||||
}
|
||||
rebuild_loc_idx = Some(i);
|
||||
} else {
|
||||
other_dirs.push(info.dir.clone());
|
||||
}
|
||||
}
|
||||
|
||||
let rebuild_loc_idx = match rebuild_loc_idx {
|
||||
Some(i) => i,
|
||||
let (rebuild_loc_idx, other_dirs) = match select_rebuild_location(&loc_infos) {
|
||||
Some(picked) => picked,
|
||||
None => {
|
||||
return Ok(Response::new(
|
||||
volume_server_pb::VolumeEcShardsRebuildResponse {
|
||||
@@ -2544,13 +2586,37 @@ impl VolumeServer for VolumeGrpcService {
|
||||
let rebuild_idx_dir = loc_infos[rebuild_loc_idx].idx_dir.clone();
|
||||
|
||||
// Determine data/parity shard config from rebuild dir
|
||||
let (data_shards, parity_shards) =
|
||||
crate::storage::erasure_coding::ec_volume::read_ec_shard_config(
|
||||
// The encode-time .dat size resolves the row count the ecx rebuild
|
||||
// de-stripes with; 0 leaves it to infer from the padded shard extent.
|
||||
// Both lookups search the sibling disks too: the rebuild writes into one
|
||||
// location, but a multi-disk server may keep this volume's .vif or its
|
||||
// generation-0 .ecsum on another, and defaulting to 10+4 with the
|
||||
// legacy layout would reconstruct through the wrong matrix.
|
||||
let dat_file_size = crate::storage::erasure_coding::ec_volume::load_vif_info_across_dirs(
|
||||
&rebuild_dir,
|
||||
&rebuild_idx_dir,
|
||||
&other_dirs,
|
||||
collection,
|
||||
vid,
|
||||
)
|
||||
.ok()
|
||||
.flatten()
|
||||
.map(|(v, _)| v.dat_file_size)
|
||||
.unwrap_or(0);
|
||||
let (data_shards, parity_shards, block_size) =
|
||||
crate::storage::erasure_coding::ec_volume::read_ec_shard_config_across_dirs(
|
||||
&rebuild_dir,
|
||||
&rebuild_idx_dir,
|
||||
&other_dirs,
|
||||
collection,
|
||||
vid,
|
||||
);
|
||||
)
|
||||
.map_err(|e| {
|
||||
tonic::Status::internal(format!(
|
||||
"read ec shard config for volume {}: {}",
|
||||
vid.0, e
|
||||
))
|
||||
})?;
|
||||
let total_shards = data_shards + parity_shards;
|
||||
|
||||
// Check which shards are missing (check rebuild dir and all other dirs)
|
||||
@@ -2589,8 +2655,17 @@ impl VolumeServer for VolumeGrpcService {
|
||||
|
||||
// Rebuild missing shards, searching all locations for input shards.
|
||||
// Pass other_dirs so shards on sibling disks are found even when the
|
||||
// primary rebuild dir doesn't hold them.
|
||||
let other_dir_refs: Vec<&str> = other_dirs.iter().map(|s| s.as_str()).collect();
|
||||
// primary rebuild dir doesn't hold them. This one takes a single flat
|
||||
// list — the shape Go's RebuildEcFiles uses — so unlike the resolvers
|
||||
// above it cannot be handed the rebuild's own index directory
|
||||
// separately, and a split -dir/-dir.idx location keeps its .ecx and
|
||||
// .vif there. Go's additionalDirs carries that directory for the same
|
||||
// reason.
|
||||
let mut rebuild_search_dirs: Vec<String> = other_dirs.clone();
|
||||
if !rebuild_idx_dir.is_empty() && rebuild_idx_dir != rebuild_dir {
|
||||
rebuild_search_dirs.push(rebuild_idx_dir.clone());
|
||||
}
|
||||
let other_dir_refs: Vec<&str> = rebuild_search_dirs.iter().map(|s| s.as_str()).collect();
|
||||
crate::storage::erasure_coding::ec_encoder::rebuild_ec_files(
|
||||
&rebuild_dir,
|
||||
collection,
|
||||
@@ -2628,6 +2703,8 @@ impl VolumeServer for VolumeGrpcService {
|
||||
collection,
|
||||
vid,
|
||||
data_shards as usize,
|
||||
block_size,
|
||||
dat_file_size,
|
||||
&ecx_dir_refs,
|
||||
)
|
||||
.map_err(|e| Status::internal(format!("RebuildEcxFile: {}", e)))?;
|
||||
@@ -2652,13 +2729,15 @@ impl VolumeServer for VolumeGrpcService {
|
||||
// When disk_id > 0: use that specific location.
|
||||
// When disk_id == 0 (unset): auto-select via
|
||||
// find_ec_shard_target_location, which prefers a disk that
|
||||
// already has the EC volume mounted, then a disk that owns the
|
||||
// .ecx on disk (volume not yet mounted — relevant for
|
||||
// ec.rebuild, where only the first shard carries .ecx and
|
||||
// subsequent shards must land on the same disk; see #9212),
|
||||
// then any HDD, then any disk. Pass the build's default
|
||||
// data-shard count; the helper takes it as a parameter so
|
||||
// custom-ratio builds can swap it.
|
||||
// already owns one of the shards being copied (a retried move
|
||||
// must overwrite in place, not leave two disks of this server
|
||||
// claiming the same shard), then a disk that already has the EC
|
||||
// volume mounted, then a disk that owns the .ecx on disk (volume
|
||||
// not yet mounted — relevant for ec.rebuild, where only the
|
||||
// first shard carries .ecx and subsequent shards must land on
|
||||
// the same disk; see #9212), then any HDD, then any disk. Pass
|
||||
// the build's default data-shard count; the helper takes it as a
|
||||
// parameter so custom-ratio builds can swap it.
|
||||
let (dest_dir, dest_idx_dir) = {
|
||||
let store = self.state.store.read().unwrap();
|
||||
let count = store.locations.len();
|
||||
@@ -2674,10 +2753,27 @@ impl VolumeServer for VolumeGrpcService {
|
||||
let loc = &store.locations[req.disk_id as usize];
|
||||
(loc.directory.clone(), loc.idx_directory.clone())
|
||||
} else {
|
||||
// A batch whose requested shards are already owned by
|
||||
// different local disks has no single correct destination:
|
||||
// writing them all to one disk would duplicate the other
|
||||
// disks' claims. Refuse so the caller splits the batch per
|
||||
// shard (or chooses explicitly via disk_id).
|
||||
let owners = store.ec_shard_owner_disks(vid, &req.shard_ids);
|
||||
if owners.len() > 1 {
|
||||
let dirs: Vec<&str> = owners
|
||||
.iter()
|
||||
.map(|&i| store.locations[i].directory.as_str())
|
||||
.collect();
|
||||
return Err(Status::failed_precondition(format!(
|
||||
"volume {} shards {:?} are already owned by multiple local disks {:?}: no single destination; copy per shard or pass disk_id",
|
||||
req.volume_id, req.shard_ids, dirs
|
||||
)));
|
||||
}
|
||||
match store.find_ec_shard_target_location(
|
||||
&req.collection,
|
||||
vid,
|
||||
DATA_SHARDS_COUNT as u32,
|
||||
&req.shard_ids,
|
||||
) {
|
||||
Some(i) => {
|
||||
let loc = &store.locations[i];
|
||||
@@ -3014,6 +3110,27 @@ impl VolumeServer for VolumeGrpcService {
|
||||
Status::internal(format!("mount {}.{}: {}", req.volume_id, shard_id, e))
|
||||
})?;
|
||||
}
|
||||
// A delivery can bring the checksum manifest alongside the shards, but
|
||||
// the receive path only writes the file. When this server already had
|
||||
// the volume mounted, the EcVolume in memory keeps whatever protection
|
||||
// state it resolved at mount — off, for a volume whose sidecar arrives
|
||||
// now — until a remount. Re-resolve it here, where the shards it
|
||||
// describes have just been added.
|
||||
//
|
||||
// Every per-disk runtime, not just the first: a vid mounts as one
|
||||
// EcVolume per disk, the delivery lands the .ecsum on one of them, and
|
||||
// the first-match lookup would leave the siblings reporting no
|
||||
// protection. Each re-resolves against its own data and index
|
||||
// directories, so a shared -dir.idx reaches all of them.
|
||||
// Resolving across every EC metadata directory is what makes that
|
||||
// reload mean something: startup mirroring gives each shard-bearing
|
||||
// disk its own .ecx/.ecj/.vif but deliberately not the sidecar, so a
|
||||
// runtime restricted to its own two directories would find nothing
|
||||
// however often it reloaded. One delivered copy, reachable from all.
|
||||
let ec_metadata_dirs = store.ec_metadata_dirs();
|
||||
for ec_vol in store.find_all_ec_volumes_mut(vid) {
|
||||
ec_vol.reload_bitrot_sidecar(&ec_metadata_dirs);
|
||||
}
|
||||
drop(store);
|
||||
self.state.volume_state_notify.notify_one();
|
||||
|
||||
@@ -3348,6 +3465,8 @@ impl VolumeServer for VolumeGrpcService {
|
||||
let ecx_dir = ec_vol.ecx_actual_dir().to_string();
|
||||
let collection = ec_vol.collection.clone();
|
||||
let vif_dat_file_size = ec_vol.dat_file_size;
|
||||
let (large_block_size, small_block_size) =
|
||||
(ec_vol.large_block_size(), ec_vol.small_block_size());
|
||||
// shard_dirs[i] is guaranteed Some for i in 0..data_shards by
|
||||
// the check above; collect concrete dirs for the decoder.
|
||||
let per_shard_dirs: Vec<String> = shard_dirs[..data_shards]
|
||||
@@ -3381,6 +3500,8 @@ impl VolumeServer for VolumeGrpcService {
|
||||
vif_dat_file_size,
|
||||
data_shards,
|
||||
&per_shard_dirs,
|
||||
large_block_size as usize,
|
||||
small_block_size as usize,
|
||||
)
|
||||
.map_err(|e| Status::internal(format!("WriteDatFile: {}", e)))?;
|
||||
|
||||
@@ -3439,12 +3560,24 @@ impl VolumeServer for VolumeGrpcService {
|
||||
.walk_ecx_stats()
|
||||
.map_err(|e| Status::internal(e.to_string()))?;
|
||||
|
||||
// The layout this holder serves reads through, as Go reports it: a
|
||||
// coordinator cannot otherwise tell a holder that understands the
|
||||
// uniform block layout from one that dropped the unknown .vif field
|
||||
// and mounted the volume as legacy.
|
||||
let ec_shard_config = Some(volume_server_pb::EcShardConfig {
|
||||
data_shards: ec_vol.data_shards,
|
||||
parity_shards: ec_vol.parity_shards,
|
||||
encode_ts_ns: 0,
|
||||
block_size: ec_vol.block_size,
|
||||
});
|
||||
|
||||
Ok(Response::new(
|
||||
volume_server_pb::VolumeEcShardsInfoResponse {
|
||||
ec_shard_infos: shard_infos,
|
||||
volume_size,
|
||||
file_count,
|
||||
file_deleted_count,
|
||||
ec_shard_config,
|
||||
},
|
||||
))
|
||||
}
|
||||
@@ -3575,7 +3708,12 @@ impl VolumeServer for VolumeGrpcService {
|
||||
modified_time: dat_modified_secs,
|
||||
extension: ".dat".to_string(),
|
||||
});
|
||||
vol.refresh_remote_write_mode();
|
||||
vol.refresh_remote_write_mode().map_err(|e| {
|
||||
Status::internal(format!(
|
||||
"volume {} failed to refresh write mode: {}",
|
||||
vid, e
|
||||
))
|
||||
})?;
|
||||
|
||||
if let Err(e) = vol.save_volume_info() {
|
||||
return Err(Status::internal(format!(
|
||||
@@ -3761,10 +3899,40 @@ impl VolumeServer for VolumeGrpcService {
|
||||
)));
|
||||
}
|
||||
|
||||
if !vol.volume_info.files.is_empty() {
|
||||
vol.volume_info.files.remove(0);
|
||||
// Snapshot the remote reference before dropping it: the
|
||||
// refresh below can fail, and a half-applied transition
|
||||
// leaves the volume claiming local while the remote backend
|
||||
// is still attached and the on-disk .vif still says remote
|
||||
// — a state a retry reads as "already on local disk" and
|
||||
// refuses to finish.
|
||||
let removed_remote = if vol.volume_info.files.is_empty() {
|
||||
None
|
||||
} else {
|
||||
Some(vol.volume_info.files.remove(0))
|
||||
};
|
||||
// Swaps the read-only sorted map out before the volume is
|
||||
// published as writable; without it the first write would
|
||||
// append to the local .dat and then fail to index.
|
||||
if let Err(e) = vol.refresh_remote_write_mode() {
|
||||
if let Some(remote) = removed_remote {
|
||||
vol.volume_info.files.insert(0, remote);
|
||||
}
|
||||
// Put the derived flags and the needle map back where
|
||||
// the restored reference says they belong. Best effort:
|
||||
// if even this fails the volume stays pinned read-only,
|
||||
// which is the safe end of the transition.
|
||||
if let Err(restore_err) = vol.refresh_remote_write_mode() {
|
||||
tracing::warn!(
|
||||
volume_id = vid.0,
|
||||
error = %restore_err,
|
||||
"tier-down rollback could not restore the remote write mode",
|
||||
);
|
||||
}
|
||||
return Err(Status::internal(format!(
|
||||
"volume {} failed to refresh write mode: {}",
|
||||
vid, e
|
||||
)));
|
||||
}
|
||||
vol.refresh_remote_write_mode();
|
||||
|
||||
if let Err(e) = vol.save_volume_info() {
|
||||
return Err(Status::internal(format!(
|
||||
@@ -4062,7 +4230,7 @@ impl VolumeServer for VolumeGrpcService {
|
||||
// Validate mode
|
||||
let mode = req.mode;
|
||||
match mode {
|
||||
1 | 2 | 3 => {} // INDEX=1, FULL=2, LOCAL=3
|
||||
1 | 2 | 3 | 5 => {} // INDEX=1, FULL=2, LOCAL=3, READS=5 (FULL for regular volumes)
|
||||
_ => {
|
||||
return Err(Status::invalid_argument(format!(
|
||||
"unsupported volume scrub mode {}",
|
||||
@@ -4162,7 +4330,7 @@ impl VolumeServer for VolumeGrpcService {
|
||||
// Validate mode
|
||||
let mode = req.mode;
|
||||
match mode {
|
||||
1 | 2 | 3 | 4 => {} // INDEX=1, FULL=2, LOCAL=3, CHECKSUM=4
|
||||
1 | 2 | 3 | 4 | 5 => {} // INDEX=1, FULL=2, LOCAL=3, CHECKSUM=4, READS=5
|
||||
_ => {
|
||||
return Err(Status::invalid_argument(format!(
|
||||
"unsupported EC volume scrub mode {}",
|
||||
@@ -4171,6 +4339,14 @@ impl VolumeServer for VolumeGrpcService {
|
||||
}
|
||||
}
|
||||
|
||||
// Only the modes that walk needles can be strict about deleted ones.
|
||||
let force_deleted_needles_check = req.force_deleted_needles_check;
|
||||
if force_deleted_needles_check && mode != 2 && mode != 5 {
|
||||
return Err(Status::invalid_argument(
|
||||
"deleted needle checks are only supported for FULL and READS scrubs",
|
||||
));
|
||||
}
|
||||
|
||||
// Collect the volume ids under a brief lock, then release it: FULL (mode 2)
|
||||
// reads remote shards and must not hold the !Send store guard across .await.
|
||||
let vids: Vec<VolumeId> = {
|
||||
@@ -4212,8 +4388,8 @@ impl VolumeServer for VolumeGrpcService {
|
||||
}
|
||||
}
|
||||
}
|
||||
2 => {
|
||||
// FULL: Go-parity per-needle local+remote walk, PLUS a TEMPORARY
|
||||
2 | 5 => {
|
||||
// FULL/READS: Go-parity per-needle local+remote walk, PLUS a TEMPORARY
|
||||
// local Reed-Solomon parity check. The needle walk only reads
|
||||
// DATA-shard intervals of LIVE needles, so on its own it can't
|
||||
// catch silent bitrot in a PARITY shard or an unwalked cold
|
||||
@@ -4245,8 +4421,13 @@ impl VolumeServer for VolumeGrpcService {
|
||||
|
||||
// (1) Per-needle local+remote walk (Go ScrubEcVolume parity).
|
||||
let (files, mut shard_infos, mut errs) =
|
||||
crate::server::store_ec::scrub_ec_volume_distributed(&self.state, vid, false)
|
||||
.await;
|
||||
crate::server::store_ec::scrub_ec_volume_distributed(
|
||||
&self.state,
|
||||
vid,
|
||||
force_deleted_needles_check,
|
||||
mode == 5,
|
||||
)
|
||||
.await;
|
||||
total_files += files as u64; // count comes from the needle walk only
|
||||
|
||||
// (2) Local parity check, gated on all-shards-local. Blocking RS
|
||||
@@ -5037,6 +5218,110 @@ mod tests {
|
||||
use tempfile::TempDir;
|
||||
use tokio_stream::StreamExt;
|
||||
|
||||
fn loc(dir: &str, idx_dir: &str, shard_count: usize, has_ecx: bool) -> LocInfo {
|
||||
LocInfo {
|
||||
dir: dir.to_string(),
|
||||
idx_dir: idx_dir.to_string(),
|
||||
shard_count,
|
||||
has_ecx,
|
||||
}
|
||||
}
|
||||
|
||||
// The rebuild reads its shards from one directory but resolves the volume's
|
||||
// layout -- ratio and uniform block size -- from the .vif or the
|
||||
// generation-0 .ecsum, which on a multi-disk server may sit anywhere. Every
|
||||
// directory that could hold one has to be in the search list, or the
|
||||
// resolution silently falls back to 10+4 with the legacy striping and
|
||||
// reconstructs through the wrong matrix.
|
||||
// The rebuild's own data and index directories are handed to the resolvers
|
||||
// as their own arguments, so they are deliberately absent from this list --
|
||||
// unlike Go, whose resolver takes a single directory list and therefore
|
||||
// carries the rebuild's index directory inside it.
|
||||
#[test]
|
||||
fn select_rebuild_location_excludes_the_rebuilds_own_dirs() {
|
||||
for infos in [
|
||||
vec![loc("/data1", "/idx1", 3, true)],
|
||||
vec![loc("/data1", "/data1", 3, true)],
|
||||
] {
|
||||
let (idx, others) = select_rebuild_location(&infos).expect("a location with .ecx");
|
||||
assert_eq!(idx, 0);
|
||||
assert!(others.is_empty(), "got {:?}", others);
|
||||
}
|
||||
}
|
||||
|
||||
// The case two reviewers flagged: a sibling holding only shards while its
|
||||
// index directory holds this volume's .vif.
|
||||
#[test]
|
||||
fn select_rebuild_location_searches_a_siblings_index_dir_not_just_its_data_dir() {
|
||||
let infos = vec![
|
||||
loc("/data1", "/data1", 5, true),
|
||||
loc("/data2", "/idx2", 2, false),
|
||||
];
|
||||
let (idx, others) = select_rebuild_location(&infos).expect("a location with .ecx");
|
||||
assert_eq!(idx, 0);
|
||||
assert_eq!(others, vec!["/data2".to_string(), "/idx2".to_string()]);
|
||||
}
|
||||
|
||||
// Several disks pointed at one index directory is a normal -dir.idx
|
||||
// deployment; the shared directory is worth searching but only once.
|
||||
#[test]
|
||||
fn select_rebuild_location_lists_a_shared_index_dir_once() {
|
||||
let infos = vec![
|
||||
loc("/data1", "/data1", 5, true),
|
||||
loc("/data2", "/shared-idx", 2, false),
|
||||
loc("/data3", "/shared-idx", 1, false),
|
||||
];
|
||||
let (_, others) = select_rebuild_location(&infos).expect("a location with .ecx");
|
||||
assert_eq!(
|
||||
others,
|
||||
vec![
|
||||
"/data2".to_string(),
|
||||
"/data3".to_string(),
|
||||
"/shared-idx".to_string()
|
||||
]
|
||||
);
|
||||
}
|
||||
|
||||
// When the shared index directory is the rebuild's own it drops out, since
|
||||
// the caller passes it separately.
|
||||
#[test]
|
||||
fn select_rebuild_location_omits_a_shared_index_dir_it_rebuilds_into() {
|
||||
let infos = vec![
|
||||
loc("/data1", "/shared-idx", 5, true),
|
||||
loc("/data2", "/shared-idx", 2, false),
|
||||
];
|
||||
let (_, others) = select_rebuild_location(&infos).expect("a location with .ecx");
|
||||
assert_eq!(others, vec!["/data2".to_string()]);
|
||||
}
|
||||
|
||||
// The winner moves as a fuller location turns up; the one it displaces
|
||||
// still has to be searched, index directory included.
|
||||
#[test]
|
||||
fn select_rebuild_location_keeps_the_displaced_winners_dirs() {
|
||||
let infos = vec![
|
||||
loc("/data1", "/idx1", 2, true),
|
||||
loc("/data2", "/idx2", 9, true),
|
||||
];
|
||||
let (idx, others) = select_rebuild_location(&infos).expect("a location with .ecx");
|
||||
assert_eq!(idx, 1, "the fuller location wins");
|
||||
assert_eq!(others, vec!["/data1".to_string(), "/idx1".to_string()]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn select_rebuild_location_drops_empty_dirs() {
|
||||
let infos = vec![loc("/data1", "", 3, true), loc("/data2", "", 1, false)];
|
||||
let (_, others) = select_rebuild_location(&infos).expect("a location with .ecx");
|
||||
assert_eq!(others, vec!["/data2".to_string()]);
|
||||
}
|
||||
|
||||
// Nothing carries an .ecx: there is no index to rebuild the shards against,
|
||||
// so the caller answers with an empty rebuild rather than guessing.
|
||||
#[test]
|
||||
fn select_rebuild_location_is_none_without_an_ecx() {
|
||||
let infos = vec![loc("/data1", "/idx1", 3, false)];
|
||||
assert!(select_rebuild_location(&infos).is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_parse_grpc_address_with_explicit_grpc_port() {
|
||||
// Format: "ip:port.grpcPort" — used by SeaweedFS for source_data_node
|
||||
@@ -6053,4 +6338,89 @@ mod tests {
|
||||
assert!(vif.expire_at_sec >= before + ttl.to_seconds());
|
||||
assert!(vif.expire_at_sec <= before + ttl.to_seconds() + 5);
|
||||
}
|
||||
|
||||
async fn scrub_ec_volume_1(
|
||||
service: &VolumeGrpcService,
|
||||
mode: i32,
|
||||
) -> volume_server_pb::ScrubEcVolumeResponse {
|
||||
service
|
||||
.scrub_ec_volume(Request::new(volume_server_pb::ScrubEcVolumeRequest {
|
||||
mode,
|
||||
volume_ids: vec![1],
|
||||
force_deleted_needles_check: false,
|
||||
}))
|
||||
.await
|
||||
.unwrap()
|
||||
.into_inner()
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn test_scrub_ec_volume_reads_mode_reconstructs_missing_shard() {
|
||||
let (service, _tmp) = make_local_service_with_volume("", None);
|
||||
service
|
||||
.volume_ec_shards_generate(Request::new(
|
||||
volume_server_pb::VolumeEcShardsGenerateRequest {
|
||||
volume_id: 1,
|
||||
collection: String::new(),
|
||||
},
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
service
|
||||
.volume_ec_shards_mount(Request::new(
|
||||
volume_server_pb::VolumeEcShardsMountRequest {
|
||||
volume_id: 1,
|
||||
collection: String::new(),
|
||||
shard_ids: (0..14).collect(),
|
||||
source_disk_type: String::new(),
|
||||
recover_missing_index: false,
|
||||
},
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
// Seed the shard-location cache so the scrub skips the master lookup.
|
||||
// The explicit-gRPC-port form targets port 1, so remote reads fail fast
|
||||
// with connection-refused instead of resolving to a live local port.
|
||||
{
|
||||
let store = service.state.store.read().unwrap();
|
||||
let ecv = store.find_ec_volume(VolumeId(1)).unwrap();
|
||||
let mut locs = ecv.shard_locations.write().unwrap();
|
||||
for sid in 0u8..14 {
|
||||
locs.insert(sid, vec!["127.0.0.1:255.1".to_string()]);
|
||||
}
|
||||
*ecv.shard_locations_refresh_time.lock().unwrap() =
|
||||
Some(std::time::Instant::now());
|
||||
}
|
||||
|
||||
// All shards local: FULL is clean.
|
||||
let resp = scrub_ec_volume_1(&service, 2).await;
|
||||
assert!(resp.broken_volume_ids.is_empty(), "{:?}", resp.details);
|
||||
assert_eq!(resp.total_files, 1);
|
||||
|
||||
service
|
||||
.volume_ec_shards_unmount(Request::new(
|
||||
volume_server_pb::VolumeEcShardsUnmountRequest {
|
||||
volume_id: 1,
|
||||
shard_ids: vec![0],
|
||||
encode_ts_ns: 0,
|
||||
},
|
||||
))
|
||||
.await
|
||||
.unwrap();
|
||||
|
||||
// With a shard unreadable, FULL flags it AND fails the needle...
|
||||
let resp = scrub_ec_volume_1(&service, 2).await;
|
||||
assert_eq!(resp.broken_volume_ids, vec![1]);
|
||||
assert!(resp.broken_shard_infos.iter().any(|s| s.shard_id == 0));
|
||||
assert!(!resp.details.is_empty());
|
||||
|
||||
// ...while READS still reports the shard broken but reconstructs the
|
||||
// interval from the local survivors, so no needle errors surface.
|
||||
let resp = scrub_ec_volume_1(&service, 5).await;
|
||||
assert_eq!(resp.broken_volume_ids, vec![1]);
|
||||
assert!(resp.broken_shard_infos.iter().any(|s| s.shard_id == 0));
|
||||
assert!(resp.details.is_empty(), "{:?}", resp.details);
|
||||
assert_eq!(resp.total_files, 1);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1589,12 +1589,15 @@ async fn get_or_head_handler_inner(
|
||||
}
|
||||
|
||||
/// Handle HTTP Range requests. Returns 206 Partial Content or 416 Range Not Satisfiable.
|
||||
#[derive(Clone, Copy)]
|
||||
#[derive(Clone, Copy, Debug)]
|
||||
struct HttpRange {
|
||||
start: i64,
|
||||
length: i64,
|
||||
}
|
||||
|
||||
// Returned when the first-byte-pos of every byte-range-spec is at or past the content size.
|
||||
const RANGE_NO_OVERLAP: &str = "invalid range: failed to overlap";
|
||||
|
||||
fn parse_range_header(s: &str, size: i64) -> Result<Vec<HttpRange>, &'static str> {
|
||||
if s.is_empty() {
|
||||
return Ok(Vec::new());
|
||||
@@ -1604,6 +1607,7 @@ fn parse_range_header(s: &str, size: i64) -> Result<Vec<HttpRange>, &'static str
|
||||
return Err("invalid range");
|
||||
}
|
||||
let mut ranges = Vec::new();
|
||||
let mut no_overlap = false;
|
||||
for part in s[PREFIX.len()..].split(',') {
|
||||
let part = part.trim();
|
||||
if part.is_empty() {
|
||||
@@ -1627,9 +1631,13 @@ fn parse_range_header(s: &str, size: i64) -> Result<Vec<HttpRange>, &'static str
|
||||
r.length = size - r.start;
|
||||
} else {
|
||||
let i = start_str.parse::<i64>().map_err(|_| "invalid range")?;
|
||||
if i > size || i < 0 {
|
||||
if i < 0 {
|
||||
return Err("invalid range");
|
||||
}
|
||||
if i >= size {
|
||||
no_overlap = true;
|
||||
continue;
|
||||
}
|
||||
r.start = i;
|
||||
if end_str.is_empty() {
|
||||
r.length = size - r.start;
|
||||
@@ -1646,6 +1654,9 @@ fn parse_range_header(s: &str, size: i64) -> Result<Vec<HttpRange>, &'static str
|
||||
}
|
||||
ranges.push(r);
|
||||
}
|
||||
if no_overlap && ranges.is_empty() {
|
||||
return Err(RANGE_NO_OVERLAP);
|
||||
}
|
||||
Ok(ranges)
|
||||
}
|
||||
|
||||
@@ -1679,7 +1690,15 @@ fn handle_range_request(
|
||||
let total = data.len() as i64;
|
||||
let ranges = match parse_range_header(range_str, total) {
|
||||
Ok(r) => r,
|
||||
Err(msg) => return range_error_response(headers, msg),
|
||||
Err(msg) => {
|
||||
if msg == RANGE_NO_OVERLAP {
|
||||
headers.insert(
|
||||
"Content-Range",
|
||||
format!("bytes */{}", total).parse().unwrap(),
|
||||
);
|
||||
}
|
||||
return range_error_response(headers, msg);
|
||||
}
|
||||
};
|
||||
|
||||
// Go's ProcessRangeRequest returns nil (empty body) for empty or oversized ranges
|
||||
@@ -1762,7 +1781,15 @@ fn handle_range_request_from_source(
|
||||
let total = info.data_size as i64;
|
||||
let ranges = match parse_range_header(range_str, total) {
|
||||
Ok(r) => r,
|
||||
Err(msg) => return range_error_response(headers, msg),
|
||||
Err(msg) => {
|
||||
if msg == RANGE_NO_OVERLAP {
|
||||
headers.insert(
|
||||
"Content-Range",
|
||||
format!("bytes */{}", total).parse().unwrap(),
|
||||
);
|
||||
}
|
||||
return range_error_response(headers, msg);
|
||||
}
|
||||
};
|
||||
|
||||
if ranges.is_empty() {
|
||||
@@ -3912,6 +3939,25 @@ mod tests {
|
||||
assert!(parse_url_path("").is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_parse_range_header_no_overlap() {
|
||||
assert_eq!(
|
||||
parse_range_header("bytes=10-", 10).unwrap_err(),
|
||||
RANGE_NO_OVERLAP
|
||||
);
|
||||
assert_eq!(
|
||||
parse_range_header("bytes=100-", 10).unwrap_err(),
|
||||
RANGE_NO_OVERLAP
|
||||
);
|
||||
// 416 only when every range fails to overlap
|
||||
let ranges = parse_range_header("bytes=10-,0-1", 10).unwrap();
|
||||
assert_eq!(ranges.len(), 1);
|
||||
assert_eq!((ranges[0].start, ranges[0].length), (0, 2));
|
||||
// an end past the size is clamped, still satisfiable
|
||||
let ranges = parse_range_header("bytes=5-100", 10).unwrap();
|
||||
assert_eq!((ranges[0].start, ranges[0].length), (5, 5));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_extract_jwt_bearer() {
|
||||
let mut headers = HeaderMap::new();
|
||||
|
||||
@@ -398,8 +398,7 @@ async fn do_heartbeat(
|
||||
|
||||
// Keep track of what we sent, to generate delta updates
|
||||
let (initial_hb, initial_volumes) = collect_heartbeat_with_snapshot(config, state);
|
||||
let mut last_volumes: HashMap<u32, master_pb::VolumeInformationMessage> =
|
||||
initial_volumes.iter().map(|v| (v.id, v.clone())).collect();
|
||||
let mut last_volumes: HashMap<u32, VolumeIdentity> = volume_identities(&initial_volumes);
|
||||
let mut last_ec_shards = {
|
||||
let store = state.store.read().unwrap();
|
||||
collect_ec_shard_delta_messages(&store)
|
||||
@@ -466,8 +465,7 @@ async fn do_heartbeat(
|
||||
if changed {
|
||||
let (adjusted_hb, adjusted_volumes) =
|
||||
collect_heartbeat_with_snapshot(config, state);
|
||||
last_volumes =
|
||||
adjusted_volumes.iter().map(|v| (v.id, v.clone())).collect();
|
||||
last_volumes = volume_identities(&adjusted_volumes);
|
||||
last_ec_shards = {
|
||||
let store = state.store.read().unwrap();
|
||||
collect_ec_shard_delta_messages(&store)
|
||||
@@ -501,7 +499,7 @@ async fn do_heartbeat(
|
||||
s.maybe_adjust_volume_max();
|
||||
}
|
||||
let (current_hb, current_volumes) = collect_heartbeat_with_snapshot(config, state);
|
||||
last_volumes = current_volumes.iter().map(|v| (v.id, v.clone())).collect();
|
||||
last_volumes = volume_identities(¤t_volumes);
|
||||
last_ec_shards = {
|
||||
let store = state.store.read().unwrap();
|
||||
collect_ec_shard_delta_messages(&store)
|
||||
@@ -530,7 +528,7 @@ async fn do_heartbeat(
|
||||
return Ok(None);
|
||||
}
|
||||
let held_volumes = collect_volume_snapshot(config, state);
|
||||
let current_volumes: HashMap<u32, _> = held_volumes.iter().map(|v| (v.id, v.clone())).collect();
|
||||
let current_volumes = volume_identities(&held_volumes);
|
||||
let current_ec_shards = {
|
||||
let store = state.store.read().unwrap();
|
||||
collect_ec_shard_delta_messages(&store)
|
||||
@@ -541,29 +539,13 @@ async fn do_heartbeat(
|
||||
|
||||
for (id, vol) in ¤t_volumes {
|
||||
if !last_volumes.contains_key(id) {
|
||||
new_vols.push(master_pb::VolumeShortInformationMessage {
|
||||
id: *id,
|
||||
collection: vol.collection.clone(),
|
||||
version: vol.version,
|
||||
replica_placement: vol.replica_placement,
|
||||
ttl: vol.ttl,
|
||||
disk_type: vol.disk_type.clone(),
|
||||
disk_id: vol.disk_id,
|
||||
});
|
||||
new_vols.push(vol.to_short_message(*id));
|
||||
}
|
||||
}
|
||||
|
||||
for (id, vol) in &last_volumes {
|
||||
if !current_volumes.contains_key(id) {
|
||||
del_vols.push(master_pb::VolumeShortInformationMessage {
|
||||
id: *id,
|
||||
collection: vol.collection.clone(),
|
||||
version: vol.version,
|
||||
replica_placement: vol.replica_placement,
|
||||
ttl: vol.ttl,
|
||||
disk_type: vol.disk_type.clone(),
|
||||
disk_id: vol.disk_id,
|
||||
});
|
||||
del_vols.push(vol.to_short_message(*id));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -744,6 +726,54 @@ fn parse_bool_property(value: Option<&String>) -> bool {
|
||||
.unwrap_or(true)
|
||||
}
|
||||
|
||||
/// What a mount or unmount delta has to name, which is far less than the
|
||||
/// information message the heartbeat carries. A server holding millions of
|
||||
/// volumes cannot keep a whole message for each just to notice one leave; the
|
||||
/// Go report state keeps the same fields for the same reason.
|
||||
#[derive(Clone)]
|
||||
struct VolumeIdentity {
|
||||
collection: String,
|
||||
disk_type: String,
|
||||
version: u32,
|
||||
replica_placement: u32,
|
||||
ttl: u32,
|
||||
disk_id: u32,
|
||||
}
|
||||
|
||||
impl VolumeIdentity {
|
||||
fn of(v: &master_pb::VolumeInformationMessage) -> Self {
|
||||
Self {
|
||||
collection: v.collection.clone(),
|
||||
disk_type: v.disk_type.clone(),
|
||||
version: v.version,
|
||||
replica_placement: v.replica_placement,
|
||||
ttl: v.ttl,
|
||||
disk_id: v.disk_id,
|
||||
}
|
||||
}
|
||||
|
||||
fn to_short_message(&self, id: u32) -> master_pb::VolumeShortInformationMessage {
|
||||
master_pb::VolumeShortInformationMessage {
|
||||
id,
|
||||
collection: self.collection.clone(),
|
||||
version: self.version,
|
||||
replica_placement: self.replica_placement,
|
||||
ttl: self.ttl,
|
||||
disk_type: self.disk_type.clone(),
|
||||
disk_id: self.disk_id,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fn volume_identities(
|
||||
volumes: &[master_pb::VolumeInformationMessage],
|
||||
) -> HashMap<u32, VolumeIdentity> {
|
||||
volumes
|
||||
.iter()
|
||||
.map(|v| (v.id, VolumeIdentity::of(v)))
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Collect volume information into a Heartbeat message.
|
||||
fn collect_heartbeat_with_snapshot(
|
||||
config: &HeartbeatConfig,
|
||||
@@ -1578,7 +1608,7 @@ mod tests {
|
||||
let (_, volume) = store.find_volume_mut(VolumeId(17)).unwrap();
|
||||
volume.set_read_only().unwrap();
|
||||
volume.volume_info.files.push(Default::default());
|
||||
volume.refresh_remote_write_mode();
|
||||
volume.refresh_remote_write_mode().unwrap();
|
||||
}
|
||||
|
||||
let heartbeat = build_heartbeat(&test_config(), &mut store);
|
||||
@@ -1944,7 +1974,7 @@ mod tests {
|
||||
key: "volumes/71.dat".to_string(),
|
||||
..Default::default()
|
||||
});
|
||||
volume.refresh_remote_write_mode();
|
||||
volume.refresh_remote_write_mode().unwrap();
|
||||
|
||||
let heartbeat = build_heartbeat(&test_config(), &mut store);
|
||||
|
||||
|
||||
@@ -33,7 +33,9 @@ use std::sync::Arc;
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
use futures::future::join_all;
|
||||
use futures::stream::{self, StreamExt};
|
||||
use reed_solomon_erasure::galois_8::ReedSolomon;
|
||||
use tokio::sync::Semaphore;
|
||||
use tonic::Request;
|
||||
|
||||
use crate::pb::master_pb::{self, seaweed_client::SeaweedClient, LookupEcVolumeRequest};
|
||||
@@ -49,6 +51,19 @@ use crate::storage::store_ec_reconcile::EcVolumeMissingIndex;
|
||||
use crate::storage::types::*;
|
||||
use crate::storage::volume::volume_file_name;
|
||||
|
||||
/// Bounds the fan-out of a single needle read. Mirrors Go's
|
||||
/// `ecIntervalReadConcurrency`.
|
||||
const INTERVAL_READ_CONCURRENCY: usize = 8;
|
||||
|
||||
/// Bounds the bytes EC recovery holds in flight across every concurrent read.
|
||||
/// Recovery is the one read path that multiplies the served bytes — it keeps an
|
||||
/// interval-sized buffer per shard alive until Reed-Solomon runs — and a peer
|
||||
/// that is slow to fail holds each of them for the whole gRPC timeout, so a
|
||||
/// burst of reads during a network blip walked the server into an OOM. Mirrors
|
||||
/// Go's `ecRecoverBudget`.
|
||||
const EC_RECOVER_BUDGET: usize = 256 << 20;
|
||||
static EC_RECOVER_SEM: Semaphore = Semaphore::const_new(EC_RECOVER_BUDGET);
|
||||
|
||||
/// One interval's data after Phase A.
|
||||
enum IntervalResult {
|
||||
/// Already read from a locally-mounted shard.
|
||||
@@ -90,7 +105,7 @@ pub async fn read_ec_shard_needle_distributed(
|
||||
// intervals, and read any locally-mounted shard intervals. We must
|
||||
// not `.await` while holding this guard (std::sync::RwLockReadGuard
|
||||
// is !Send).
|
||||
let snapshot = match snapshot_under_lock(state, vid, needle_id)? {
|
||||
let mut snapshot = match snapshot_under_lock(state, vid, needle_id)? {
|
||||
Some(s) => s,
|
||||
None => return Ok(None),
|
||||
};
|
||||
@@ -106,7 +121,9 @@ pub async fn read_ec_shard_needle_distributed(
|
||||
|
||||
let mut shard_locations = snapshot.cached_locations.clone();
|
||||
if any_remote
|
||||
&& needs_refresh(
|
||||
&& claim_shard_locations_refresh(
|
||||
state,
|
||||
vid,
|
||||
&shard_locations,
|
||||
snapshot.cache_refreshed_at,
|
||||
snapshot.data_shards as usize,
|
||||
@@ -117,17 +134,21 @@ pub async fn read_ec_shard_needle_distributed(
|
||||
Ok(fresh) => {
|
||||
// A complete reply merges into the cache; an incomplete one
|
||||
// (< data_shards) is left unwritten — keep the prior cache.
|
||||
if let Some(merged) =
|
||||
write_back_shard_locations(state, vid, fresh, snapshot.data_shards as usize)
|
||||
match write_back_shard_locations(state, vid, fresh, snapshot.data_shards as usize)
|
||||
{
|
||||
shard_locations = merged;
|
||||
Some(merged) => shard_locations = merged,
|
||||
// An incomplete reply leaves the cache unwritten and its refresh
|
||||
// time unadvanced, so the mark this refresh consumed goes back.
|
||||
None => mark_shard_locations_stale(state, vid),
|
||||
}
|
||||
}
|
||||
Err(e) => {
|
||||
// Lookup failed — proceed with cached values. If cache
|
||||
// is empty, the remote fetch below will fail and we
|
||||
// surface a NotFound (matching Go's behavior when no
|
||||
// locations are known).
|
||||
// locations are known). The mark goes back: nothing
|
||||
// answered for it, and the map stays disproved.
|
||||
mark_shard_locations_stale(state, vid);
|
||||
tracing::warn!(
|
||||
"ec lookup failed for volume {}: {} — using cached locations ({} entries)",
|
||||
vid.0,
|
||||
@@ -138,39 +159,56 @@ pub async fn read_ec_shard_needle_distributed(
|
||||
}
|
||||
}
|
||||
|
||||
// Phase C — fetch missing intervals, reconstructing when the
|
||||
// direct peer read fails.
|
||||
let mut assembled: Vec<Vec<u8>> = Vec::with_capacity(snapshot.intervals.len());
|
||||
for res in snapshot.intervals {
|
||||
match res {
|
||||
IntervalResult::Local(buf) => assembled.push(buf),
|
||||
IntervalResult::NeedRemote {
|
||||
shard_id,
|
||||
shard_offset,
|
||||
size,
|
||||
} => {
|
||||
let (buf, is_deleted) = fetch_one_interval(
|
||||
state,
|
||||
vid,
|
||||
needle_id,
|
||||
// Phase C — fetch missing intervals, reconstructing when the direct peer
|
||||
// read fails. Blocks that follow each other in the .dat live on different
|
||||
// shards, so a needle spanning several of them costs one round trip per
|
||||
// block when fetched in sequence; `buffered` keeps the order while letting
|
||||
// INTERVAL_READ_CONCURRENCY of them fly at once.
|
||||
let data_shards = snapshot.data_shards as usize;
|
||||
let parity_shards = snapshot.parity_shards as usize;
|
||||
let encode_ts_ns = snapshot.encode_ts_ns;
|
||||
let intervals = std::mem::take(&mut snapshot.intervals);
|
||||
let fetched: Vec<io::Result<(Vec<u8>, bool)>> = stream::iter(intervals.into_iter().map(|res| {
|
||||
let shard_locations = &shard_locations;
|
||||
async move {
|
||||
match res {
|
||||
IntervalResult::Local(buf) => Ok((buf, false)),
|
||||
IntervalResult::NeedRemote {
|
||||
shard_id,
|
||||
shard_offset,
|
||||
size,
|
||||
&shard_locations,
|
||||
snapshot.data_shards as usize,
|
||||
snapshot.parity_shards as usize,
|
||||
snapshot.encode_ts_ns,
|
||||
)
|
||||
.await?;
|
||||
// A peer reports the needle deleted (a cross-server window where the
|
||||
// local index still shows it live): treat as not-found rather than
|
||||
// serving zeros, mirroring Go's ErrorDeleted.
|
||||
if is_deleted {
|
||||
return Ok(None);
|
||||
} => {
|
||||
fetch_one_interval(
|
||||
state,
|
||||
vid,
|
||||
needle_id,
|
||||
shard_id,
|
||||
shard_offset,
|
||||
size,
|
||||
shard_locations,
|
||||
data_shards,
|
||||
parity_shards,
|
||||
encode_ts_ns,
|
||||
)
|
||||
.await
|
||||
}
|
||||
assembled.push(buf);
|
||||
}
|
||||
}
|
||||
}))
|
||||
.buffered(INTERVAL_READ_CONCURRENCY)
|
||||
.collect()
|
||||
.await;
|
||||
|
||||
let mut assembled: Vec<Vec<u8>> = Vec::with_capacity(fetched.len());
|
||||
for res in fetched {
|
||||
let (buf, is_deleted) = res?;
|
||||
// A peer reports the needle deleted (a cross-server window where the
|
||||
// local index still shows it live): treat as not-found rather than
|
||||
// serving zeros, mirroring Go's ErrorDeleted.
|
||||
if is_deleted {
|
||||
return Ok(None);
|
||||
}
|
||||
assembled.push(buf);
|
||||
}
|
||||
|
||||
// Phase D — assemble and parse the Needle. Mirrors the tail of
|
||||
@@ -208,7 +246,9 @@ pub async fn read_ec_shard_needle_distributed(
|
||||
/// without decoding (so genuine shard faults are reported rather than healed).
|
||||
/// Mirrors Go's `Store.ScrubEcVolume`. Returns (rows walked, broken shards,
|
||||
/// errors). `force_deleted_needles_check` disables the benign delete-state
|
||||
/// size-mismatch suppression.
|
||||
/// size-mismatch suppression. `recover_unreadable` (READS mode) rebuilds an
|
||||
/// unreadable interval from the surviving shards: the same shards are reported
|
||||
/// broken, but only needles parity can no longer recover become errors.
|
||||
///
|
||||
/// Shard locations are refreshed once up front. Each needle is then processed via
|
||||
/// `scrub_snapshot_under_lock` + lock-drop + no-reconstruct `read_remote_ec_shard_interval`,
|
||||
@@ -217,10 +257,19 @@ pub async fn scrub_ec_volume_distributed(
|
||||
state: &Arc<VolumeServerState>,
|
||||
vid: VolumeId,
|
||||
force_deleted_needles_check: bool,
|
||||
recover_unreadable: bool,
|
||||
) -> (i64, Vec<crate::pb::volume_server_pb::EcShardInfo>, Vec<String>) {
|
||||
// Phase A — under the Store read lock, run the index scrub and grab the
|
||||
// paths/scalars + shard-location staleness; release the lock before any await.
|
||||
let (ecx_path, collection, seed_errs, cached_locations, cache_refreshed_at, data_shards, total_shards) = {
|
||||
let (
|
||||
ecx_path,
|
||||
collection,
|
||||
seed_errs,
|
||||
cached_locations,
|
||||
cache_refreshed_at,
|
||||
data_shards,
|
||||
total_shards,
|
||||
) = {
|
||||
let store = state.store.read().unwrap();
|
||||
let ecv = match store.find_ec_volume(vid) {
|
||||
Some(v) => v,
|
||||
@@ -255,10 +304,18 @@ pub async fn scrub_ec_volume_distributed(
|
||||
// cachedLookupEcShardLocations). A partial reply (< data_shards locations, a
|
||||
// master mid-recovery) or a failed lookup is a hard, retryable error — never
|
||||
// overwrite a good cache with a partial map or storm a down master per needle.
|
||||
if needs_refresh(&cached_locations, cache_refreshed_at, data_shards, total_shards) {
|
||||
if claim_shard_locations_refresh(
|
||||
state,
|
||||
vid,
|
||||
&cached_locations,
|
||||
cache_refreshed_at,
|
||||
data_shards,
|
||||
total_shards,
|
||||
) {
|
||||
match cached_lookup_ec_shard_locations(state, vid).await {
|
||||
Ok(fresh) => {
|
||||
if write_back_shard_locations(state, vid, fresh, data_shards).is_none() {
|
||||
mark_shard_locations_stale(state, vid);
|
||||
return (
|
||||
0,
|
||||
Vec::new(),
|
||||
@@ -270,11 +327,12 @@ pub async fn scrub_ec_volume_distributed(
|
||||
}
|
||||
}
|
||||
Err(e) => {
|
||||
mark_shard_locations_stale(state, vid);
|
||||
return (
|
||||
0,
|
||||
Vec::new(),
|
||||
vec![format!("failed to locate shard via master grpc: {}", e)],
|
||||
)
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -340,8 +398,9 @@ pub async fn scrub_ec_volume_distributed(
|
||||
}
|
||||
};
|
||||
|
||||
// Read each interval local-then-remote WITHOUT reconstructing: we verify
|
||||
// the shards are valid, we do not heal them. Locations refreshed above.
|
||||
// Read each interval local-then-remote. Neither read decodes: the point is to
|
||||
// find shards that are themselves broken, not to heal around them. READS then
|
||||
// rebuilds what it could not read. Locations refreshed above.
|
||||
let n_intervals = snapshot.intervals.len();
|
||||
let mut data: Vec<u8> = Vec::with_capacity(snapshot.actual_size);
|
||||
for (i, res) in snapshot.intervals.iter().enumerate() {
|
||||
@@ -371,11 +430,9 @@ pub async fn scrub_ec_volume_distributed(
|
||||
// -> the delete-state suppression (mirrors Go's pre-zeroed buffer).
|
||||
Ok((_, true)) => data.resize(data.len() + *ssize, 0),
|
||||
Ok((buf, false)) => data.extend_from_slice(&buf),
|
||||
Err(_) => {
|
||||
errs.push(format!(
|
||||
"failed to read EC shard {} for needle {} on volume {} (interval {}/{})",
|
||||
shard_id, id.0, vid.0, i + 1, n_intervals
|
||||
));
|
||||
Err(read_err) => {
|
||||
// The shard is broken whether or not the needle survives it,
|
||||
// so report it either way.
|
||||
broken_shards.insert(
|
||||
*shard_id,
|
||||
crate::pb::volume_server_pb::EcShardInfo {
|
||||
@@ -386,7 +443,41 @@ pub async fn scrub_ec_volume_distributed(
|
||||
..Default::default()
|
||||
},
|
||||
);
|
||||
break;
|
||||
if !recover_unreadable {
|
||||
errs.push(format!(
|
||||
"failed to read EC shard {} for needle {} on volume {} (interval {}/{}): {}",
|
||||
shard_id, id.0, vid.0, i + 1, n_intervals, read_err
|
||||
));
|
||||
break;
|
||||
}
|
||||
match recover_one_remote_ec_shard_interval(
|
||||
state,
|
||||
vid,
|
||||
id,
|
||||
*shard_id,
|
||||
*shard_offset,
|
||||
*ssize,
|
||||
&locations,
|
||||
data_shards,
|
||||
total_shards - data_shards,
|
||||
snapshot.encode_ts_ns,
|
||||
)
|
||||
.await
|
||||
{
|
||||
// Same as the direct read above: a holder reporting the
|
||||
// needle deleted is authoritative and answers with no
|
||||
// bytes, so zero-fill and let the delete-state
|
||||
// suppression have it.
|
||||
Ok((_, true)) => data.resize(data.len() + *ssize, 0),
|
||||
Ok((buf, false)) => data.extend_from_slice(&buf),
|
||||
Err(e) => {
|
||||
errs.push(format!(
|
||||
"failed to recover EC shard {} for needle {} on volume {} (interval {}/{}): {}",
|
||||
shard_id, id.0, vid.0, i + 1, n_intervals, e
|
||||
));
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -508,7 +599,7 @@ fn read_local_intervals(
|
||||
) -> Vec<IntervalResult> {
|
||||
let mut interval_results = Vec::with_capacity(intervals.len());
|
||||
for interval in intervals {
|
||||
let (shard_id, shard_offset) = interval.to_shard_id_and_offset(ecv.data_shards);
|
||||
let (shard_id, shard_offset) = ecv.interval_to_shard_id_and_offset(interval);
|
||||
let buf_size = interval.size as usize;
|
||||
let local = ecv.shards.get(shard_id as usize).and_then(|s| s.as_ref());
|
||||
match local {
|
||||
@@ -571,6 +662,7 @@ fn build_snapshot(
|
||||
fn needs_refresh(
|
||||
locations: &HashMap<ShardId, Vec<String>>,
|
||||
refreshed_at: Option<Instant>,
|
||||
stale: bool,
|
||||
data_shards: usize,
|
||||
total_shards: usize,
|
||||
) -> bool {
|
||||
@@ -580,16 +672,53 @@ fn needs_refresh(
|
||||
None => return true,
|
||||
};
|
||||
let shard_count = locations.len();
|
||||
if shard_count < data_shards && age < Duration::from_secs(11) {
|
||||
return false;
|
||||
// A complete map is trusted longest. One short of data_shards, or one a
|
||||
// failed read has just disproved, is re-checked promptly: until it is, reads
|
||||
// keep aiming at a location the shard has left.
|
||||
let ttl = if stale || shard_count < data_shards {
|
||||
Duration::from_secs(11)
|
||||
} else if shard_count == total_shards {
|
||||
Duration::from_secs(37 * 60)
|
||||
} else {
|
||||
Duration::from_secs(7 * 60)
|
||||
};
|
||||
age >= ttl
|
||||
}
|
||||
|
||||
/// Mark the cached shard map for a prompt re-check after a read failed against
|
||||
/// one of its locations. Go drops the entry outright in `forgetShardId`, which
|
||||
/// costs it the direct read until the map is re-learned; here the entry stays
|
||||
/// (a dead peer just fails fast on the next attempt) and only the freshness
|
||||
/// window is cut, so a shard that has moved is picked up in seconds either way.
|
||||
fn mark_shard_locations_stale(state: &Arc<VolumeServerState>, vid: VolumeId) {
|
||||
let store = state.store.read().unwrap();
|
||||
if let Some(ecv) = store.find_ec_volume(vid) {
|
||||
*ecv.shard_locations_stale.lock().unwrap() = true;
|
||||
}
|
||||
if shard_count == total_shards && age < Duration::from_secs(37 * 60) {
|
||||
return false;
|
||||
}
|
||||
|
||||
/// Decide whether the cached map is due a master lookup and, when it is, consume
|
||||
/// its stale mark in the same critical section. A mark raised from here on
|
||||
/// belongs to the next refresh: the read that raised it has disproved the map
|
||||
/// this lookup is about to install.
|
||||
fn claim_shard_locations_refresh(
|
||||
state: &Arc<VolumeServerState>,
|
||||
vid: VolumeId,
|
||||
locations: &HashMap<ShardId, Vec<String>>,
|
||||
refreshed_at: Option<Instant>,
|
||||
data_shards: usize,
|
||||
total_shards: usize,
|
||||
) -> bool {
|
||||
let store = state.store.read().unwrap();
|
||||
let Some(ecv) = store.find_ec_volume(vid) else {
|
||||
return needs_refresh(locations, refreshed_at, false, data_shards, total_shards);
|
||||
};
|
||||
let mut stale = ecv.shard_locations_stale.lock().unwrap();
|
||||
let refresh = needs_refresh(locations, refreshed_at, *stale, data_shards, total_shards);
|
||||
if refresh {
|
||||
*stale = false;
|
||||
}
|
||||
if shard_count >= data_shards && age < Duration::from_secs(7 * 60) {
|
||||
return false;
|
||||
}
|
||||
true
|
||||
refresh
|
||||
}
|
||||
|
||||
async fn cached_lookup_ec_shard_locations(
|
||||
@@ -725,6 +854,9 @@ async fn fetch_one_interval(
|
||||
sources,
|
||||
e
|
||||
);
|
||||
// Reconstruction below skips this very shard, so nothing else
|
||||
// invalidates the location that just failed.
|
||||
mark_shard_locations_stale(state, vid);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -732,7 +864,7 @@ async fn fetch_one_interval(
|
||||
|
||||
// Reconstruct: fan-out reads to every other shard at the same
|
||||
// (shard_offset, size). Mirrors `recoverOneRemoteEcShardInterval`.
|
||||
let buf = recover_one_remote_ec_shard_interval(
|
||||
recover_one_remote_ec_shard_interval(
|
||||
state,
|
||||
vid,
|
||||
needle_id,
|
||||
@@ -744,8 +876,7 @@ async fn fetch_one_interval(
|
||||
parity_shards,
|
||||
expected_encode_ts_ns,
|
||||
)
|
||||
.await?;
|
||||
Ok((buf, false))
|
||||
.await
|
||||
}
|
||||
|
||||
async fn read_remote_ec_shard_interval(
|
||||
@@ -905,7 +1036,7 @@ async fn recover_one_remote_ec_shard_interval(
|
||||
data_shards: usize,
|
||||
parity_shards: usize,
|
||||
expected_encode_ts_ns: i64,
|
||||
) -> io::Result<Vec<u8>> {
|
||||
) -> io::Result<(Vec<u8>, bool)> {
|
||||
let total_shards = data_shards + parity_shards;
|
||||
let rs = ReedSolomon::new(data_shards, parity_shards).map_err(|e| {
|
||||
io::Error::new(
|
||||
@@ -914,89 +1045,136 @@ async fn recover_one_remote_ec_shard_interval(
|
||||
)
|
||||
})?;
|
||||
|
||||
// Charge the buffers this recovery is about to hold against the budget, so a
|
||||
// burst of them queues here rather than on the heap. An interval whose
|
||||
// fan-out outgrows the whole budget takes all of it and so runs alone,
|
||||
// rather than waiting on permits that can never be granted.
|
||||
let _permit = EC_RECOVER_SEM
|
||||
.acquire_many((size * data_shards).min(EC_RECOVER_BUDGET) as u32)
|
||||
.await
|
||||
.map_err(|e| {
|
||||
io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!(
|
||||
"ec recover budget for shard {}.{}: {}",
|
||||
vid.0, shard_id_to_recover, e
|
||||
),
|
||||
)
|
||||
})?;
|
||||
|
||||
let mut bufs: Vec<Option<Vec<u8>>> = vec![None; total_shards];
|
||||
|
||||
// Phase 0: seed bufs from LOCALLY mounted shards. If this node
|
||||
// already holds enough sibling shards, reconstruction completes
|
||||
// without any peer fan-out — and even with a cold/incomplete
|
||||
// shard_locations cache or a failed master lookup, local
|
||||
// survivors still contribute. Mirrors Go's
|
||||
// recoverOneRemoteEcShardInterval behaviour, which is implicitly
|
||||
// local-aware because the Store fan-out targets ALL known
|
||||
// locations (including the caller's own server address); the
|
||||
// Rust port had been remote-only, so reconstructing with a cold
|
||||
// cache failed even when enough siblings were on disk.
|
||||
// survivors still contribute.
|
||||
let mut available = 0usize;
|
||||
{
|
||||
let store = state.store.read().unwrap();
|
||||
if let Some(ecv) = store.find_ec_volume(vid) {
|
||||
for sid in 0..total_shards {
|
||||
if sid as ShardId == shard_id_to_recover {
|
||||
continue;
|
||||
}
|
||||
if let Some(Some(shard)) = ecv.shards.get(sid) {
|
||||
let mut buf = vec![0u8; size];
|
||||
if shard.read_at(&mut buf, shard_offset as u64).map(|n| n == size).unwrap_or(false) {
|
||||
bufs[sid] = Some(buf);
|
||||
}
|
||||
for sid in 0..total_shards {
|
||||
if available >= data_shards {
|
||||
break;
|
||||
}
|
||||
if sid as ShardId == shard_id_to_recover {
|
||||
continue;
|
||||
}
|
||||
// Resolve the shard together with the EcVolume on the disk that owns
|
||||
// it: a reconciled volume has its shards split across data dirs. A
|
||||
// shard from a different encode run must not be fed to Reed-Solomon;
|
||||
// lenient only when the caller carries no identity (pre-upgrade).
|
||||
// Mirrors Go's `readLocalEcShardInterval`.
|
||||
let owner = match store.find_ec_volume_with_shard(vid, sid as u32) {
|
||||
Some(ecv) if expected_encode_ts_ns == 0 || ecv.encode_ts_ns == expected_encode_ts_ns => ecv,
|
||||
_ => continue,
|
||||
};
|
||||
if let Some(Some(shard)) = owner.shards.get(sid) {
|
||||
let mut buf = vec![0u8; size];
|
||||
if shard.read_at(&mut buf, shard_offset as u64).map(|n| n == size).unwrap_or(false) {
|
||||
bufs[sid] = Some(buf);
|
||||
available += 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Phase 1: remote fan-out — one task per known shard location
|
||||
// we DON'T already have locally and DON'T need to recover.
|
||||
let mut tasks = Vec::new();
|
||||
for (sid, locs) in shard_locations {
|
||||
if *sid == shard_id_to_recover || locs.is_empty() {
|
||||
continue;
|
||||
}
|
||||
if bufs[*sid as usize].is_some() {
|
||||
continue;
|
||||
}
|
||||
let sid = *sid;
|
||||
let locs = locs.clone();
|
||||
let state = state.clone();
|
||||
tasks.push(async move {
|
||||
let res = read_remote_ec_shard_interval(
|
||||
&state,
|
||||
&locs,
|
||||
vid,
|
||||
needle_id,
|
||||
sid,
|
||||
shard_offset,
|
||||
size,
|
||||
expected_encode_ts_ns,
|
||||
)
|
||||
.await;
|
||||
(sid, res)
|
||||
});
|
||||
}
|
||||
let results = join_all(tasks).await;
|
||||
// Phase 1: remote fan-out over the shard locations we DON'T already have
|
||||
// locally and DON'T need to recover. Reconstruction consumes data_shards
|
||||
// shards, so reading every remaining one holds a third more buffers than
|
||||
// that and asks a third more of peers that may already be struggling: fetch
|
||||
// what is still missing, and widen only if some of those reads fail.
|
||||
let mut candidates: Vec<(ShardId, Vec<String>)> = shard_locations
|
||||
.iter()
|
||||
.filter(|(sid, locs)| {
|
||||
**sid != shard_id_to_recover
|
||||
&& (**sid as usize) < total_shards
|
||||
&& !locs.is_empty()
|
||||
&& bufs[**sid as usize].is_none()
|
||||
})
|
||||
.map(|(sid, locs)| (*sid, locs.clone()))
|
||||
.collect();
|
||||
|
||||
for (sid, res) in results {
|
||||
match res {
|
||||
// Exclude a deleted shard from reconstruction (Go gates on a full
|
||||
// read): feeding the empty/zero buffer into Reed-Solomon would
|
||||
// corrupt the recovered shard.
|
||||
Ok((buf, is_deleted)) => {
|
||||
if !is_deleted && (sid as usize) < total_shards {
|
||||
bufs[sid as usize] = Some(buf);
|
||||
}
|
||||
}
|
||||
Err(e) => {
|
||||
tracing::debug!(
|
||||
"recover: read {}.{} for needle {} failed: {}",
|
||||
vid.0,
|
||||
sid,
|
||||
let mut any_deleted = false;
|
||||
while available < data_shards && !candidates.is_empty() {
|
||||
let rest = candidates.split_off((data_shards - available).min(candidates.len()));
|
||||
let wave = std::mem::replace(&mut candidates, rest);
|
||||
let results = join_all(wave.into_iter().map(|(sid, locs)| {
|
||||
let state = state.clone();
|
||||
async move {
|
||||
let res = read_remote_ec_shard_interval(
|
||||
&state,
|
||||
&locs,
|
||||
vid,
|
||||
needle_id,
|
||||
e
|
||||
);
|
||||
sid,
|
||||
shard_offset,
|
||||
size,
|
||||
expected_encode_ts_ns,
|
||||
)
|
||||
.await;
|
||||
(sid, res)
|
||||
}
|
||||
}))
|
||||
.await;
|
||||
|
||||
for (sid, res) in results {
|
||||
match res {
|
||||
// Exclude a deleted shard from reconstruction (Go gates on a full
|
||||
// read): feeding the empty/zero buffer into Reed-Solomon would
|
||||
// corrupt the recovered shard.
|
||||
Ok((buf, is_deleted)) => {
|
||||
if is_deleted {
|
||||
any_deleted = true;
|
||||
continue;
|
||||
}
|
||||
bufs[sid as usize] = Some(buf);
|
||||
available += 1;
|
||||
}
|
||||
Err(e) => {
|
||||
tracing::debug!(
|
||||
"recover: read {}.{} for needle {} failed: {}",
|
||||
vid.0,
|
||||
sid,
|
||||
needle_id,
|
||||
e
|
||||
);
|
||||
mark_shard_locations_stale(state, vid);
|
||||
}
|
||||
}
|
||||
}
|
||||
if any_deleted {
|
||||
// every shard of a deleted needle answers deleted, so another wave cannot help
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
let available = bufs.iter().filter(|b| b.is_some()).count();
|
||||
if available < data_shards {
|
||||
// A holder reporting the needle deleted is authoritative -- deletes are
|
||||
// never invented and never undone -- so answer that rather than the
|
||||
// failure to gather shards of a needle that is gone.
|
||||
if any_deleted {
|
||||
return Ok((Vec::new(), true));
|
||||
}
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!(
|
||||
@@ -1017,7 +1195,7 @@ async fn recover_one_remote_ec_shard_interval(
|
||||
})?;
|
||||
|
||||
match bufs.into_iter().nth(shard_id_to_recover as usize).flatten() {
|
||||
Some(buf) => Ok(buf),
|
||||
Some(buf) => Ok((buf, any_deleted)),
|
||||
None => Err(io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!(
|
||||
@@ -1255,3 +1433,34 @@ async fn drain_copy_stream(
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn locations(count: usize) -> HashMap<ShardId, Vec<String>> {
|
||||
(0..count)
|
||||
.map(|sid| (sid as ShardId, vec!["127.0.0.1:8080".to_string()]))
|
||||
.collect()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn needs_refresh_re_checks_a_map_a_failed_read_disproved() {
|
||||
let just_now = Some(Instant::now());
|
||||
let aged = Some(Instant::now() - Duration::from_secs(12));
|
||||
|
||||
// A complete map is trusted for a long time, and one shard short still
|
||||
// outlasts a 12-second gap.
|
||||
assert!(!needs_refresh(&locations(14), aged, false, 10, 14));
|
||||
assert!(!needs_refresh(&locations(13), aged, false, 10, 14));
|
||||
// Disproved by a read, the same maps are re-checked within seconds.
|
||||
assert!(needs_refresh(&locations(14), aged, true, 10, 14));
|
||||
assert!(needs_refresh(&locations(13), aged, true, 10, 14));
|
||||
// But the mark buys one prompt re-check, not a lookup per read.
|
||||
assert!(!needs_refresh(&locations(14), just_now, true, 10, 14));
|
||||
// A map short of the data shards is re-checked promptly regardless.
|
||||
assert!(needs_refresh(&locations(9), aged, false, 10, 14));
|
||||
// An unrefreshed cache always looks up.
|
||||
assert!(needs_refresh(&locations(0), None, false, 10, 14));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -843,7 +843,8 @@ impl DiskLocation {
|
||||
// .ecx open error, .ecj create error, malformed .vif) would
|
||||
// have to panic via unwrap(). Build the EcVolume up front and
|
||||
// propagate the error to the caller.
|
||||
if !self.ec_volumes.contains_key(&vid) {
|
||||
let created = !self.ec_volumes.contains_key(&vid);
|
||||
if created {
|
||||
let ec_vol = EcVolume::new(&dir, idx_dir, collection, vid)
|
||||
.map_err(VolumeError::Io)?;
|
||||
self.ec_volumes.insert(vid, ec_vol);
|
||||
@@ -866,9 +867,27 @@ impl DiskLocation {
|
||||
}
|
||||
|
||||
for &shard_id in shard_ids {
|
||||
// A mount retry re-listing a shard this volume already holds:
|
||||
// keep the existing registration (mirrors Go's AddEcVolumeShard
|
||||
// added=false) — re-adding would replace a serving fd and bump
|
||||
// the ec_shards gauge without growing the mounted count.
|
||||
if ec_vol.has_shard(shard_id as u8) {
|
||||
continue;
|
||||
}
|
||||
let mut shard = EcVolumeShard::new(&dir, collection, vid, shard_id as u8);
|
||||
shard.disk_type = ec_vol.disk_type.clone();
|
||||
ec_vol.add_shard(shard).map_err(VolumeError::Io)?;
|
||||
if let Err(e) = ec_vol.add_shard(shard) {
|
||||
// The shard was dropped (its descriptors closed) inside the
|
||||
// failed add. If this call just created the EcVolume and it
|
||||
// holds nothing, remove it too — a zero-shard registration
|
||||
// would advertise a mount that serves no data while pinning
|
||||
// its descriptors.
|
||||
let now_empty = ec_vol.shard_count() == 0;
|
||||
if created && now_empty {
|
||||
self.ec_volumes.remove(&vid);
|
||||
}
|
||||
return Err(VolumeError::Io(e));
|
||||
}
|
||||
crate::metrics::VOLUME_GAUGE
|
||||
.with_label_values(&[collection, "ec_shards"])
|
||||
.inc();
|
||||
@@ -921,27 +940,32 @@ impl DiskLocation {
|
||||
// double-count for those filenames.
|
||||
let mut seen: HashSet<String> = HashSet::new();
|
||||
let mut entries: Vec<String> = Vec::new();
|
||||
for ent in fs::read_dir(&self.directory)? {
|
||||
let ent = ent?;
|
||||
if ent.file_type().map(|ft| ft.is_dir()).unwrap_or(false) {
|
||||
continue;
|
||||
}
|
||||
let name = ent.file_name().to_string_lossy().into_owned();
|
||||
if seen.insert(name.clone()) {
|
||||
entries.push(name);
|
||||
}
|
||||
}
|
||||
if self.idx_directory != self.directory {
|
||||
for ent in fs::read_dir(&self.idx_directory)? {
|
||||
// Keep only the shard and index files this scan acts on: a disk of
|
||||
// regular volumes has millions of .dat/.idx/.vif names that would
|
||||
// otherwise each cost a String here and a slot in the sort below.
|
||||
let mut collect = |dir: &str| -> io::Result<()> {
|
||||
for ent in fs::read_dir(dir)? {
|
||||
let ent = ent?;
|
||||
if ent.file_type().map(|ft| ft.is_dir()).unwrap_or(false) {
|
||||
continue;
|
||||
}
|
||||
let name = ent.file_name().to_string_lossy().into_owned();
|
||||
let Some(dot) = name.rfind('.') else {
|
||||
continue;
|
||||
};
|
||||
let ext = &name[dot..];
|
||||
if parse_ec_shard_extension(ext).is_none() && ext != ".ecx" {
|
||||
continue;
|
||||
}
|
||||
if seen.insert(name.clone()) {
|
||||
entries.push(name);
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
};
|
||||
collect(&self.directory)?;
|
||||
if self.idx_directory != self.directory {
|
||||
collect(&self.idx_directory)?;
|
||||
}
|
||||
entries.sort();
|
||||
|
||||
@@ -1774,6 +1798,94 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
/// A refused shard (a 0-byte file beside an index with entries) on the
|
||||
/// FIRST mount of a volume must not leave the just-created zero-shard
|
||||
/// EcVolume registered — it would advertise a mount serving no data,
|
||||
/// pin the .ecx/.ecj descriptors, and make placement's mounted tier
|
||||
/// prefer this disk. A volume that already holds shards keeps them
|
||||
/// (the mount RPC's pre-existing first-error-aborts contract).
|
||||
#[test]
|
||||
fn test_mount_ec_shards_refused_shard_removes_created_empty_volume() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
let mut loc = DiskLocation::new(
|
||||
dir,
|
||||
dir,
|
||||
10,
|
||||
DiskType::HardDrive,
|
||||
MinFreeSpace::Percent(1.0),
|
||||
Vec::new(),
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
// An index with one 16-byte entry and a 0-byte shard: the mount is
|
||||
// refused, and the EcVolume created for it must be unregistered.
|
||||
std::fs::write(format!("{}/pics_9.ecx", dir), [0u8; 16]).unwrap();
|
||||
std::fs::write(format!("{}/pics_9.ec00", dir), b"").unwrap();
|
||||
let err = loc
|
||||
.mount_ec_shards(VolumeId(9), "pics", &[0], "")
|
||||
.expect_err("a 0-byte shard beside an index with entries must refuse the mount");
|
||||
assert!(
|
||||
err.to_string().contains("empty (0 bytes)"),
|
||||
"want the empty-shard refusal, got: {}",
|
||||
err
|
||||
);
|
||||
assert!(
|
||||
loc.find_ec_volume(VolumeId(9)).is_none(),
|
||||
"a refused first mount must not leave a zero-shard EcVolume registered",
|
||||
);
|
||||
|
||||
// With a valid shard mounted, a later refused shard keeps the
|
||||
// existing registration intact.
|
||||
std::fs::write(format!("{}/pics_9.ec01", dir), b"good bytes").unwrap();
|
||||
loc.mount_ec_shards(VolumeId(9), "pics", &[1], "").unwrap();
|
||||
loc.mount_ec_shards(VolumeId(9), "pics", &[0], "")
|
||||
.expect_err("the 0-byte shard stays refused");
|
||||
assert_eq!(
|
||||
loc.find_ec_volume(VolumeId(9)).map(|v| v.shard_count()),
|
||||
Some(1),
|
||||
"an existing volume keeps its valid shards when a later shard is refused",
|
||||
);
|
||||
}
|
||||
|
||||
/// A mount retry re-listing an already mounted shard must keep the
|
||||
/// existing registration and not bump the ec_shards gauge — the Rust
|
||||
/// twin of Go's AddEcVolumeShard added=false handling.
|
||||
#[test]
|
||||
fn test_mount_ec_shards_duplicate_keeps_registration_and_gauge() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
let mut loc = DiskLocation::new(
|
||||
dir,
|
||||
dir,
|
||||
10,
|
||||
DiskType::HardDrive,
|
||||
MinFreeSpace::Percent(1.0),
|
||||
Vec::new(),
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
// A collection name unique to this test: the gauge is process-global
|
||||
// and sibling tests running in parallel touch other labels.
|
||||
std::fs::write(format!("{}/dupmount_11.ec00", dir), b"shard bytes").unwrap();
|
||||
let gauge = crate::metrics::VOLUME_GAUGE.with_label_values(&["dupmount", "ec_shards"]);
|
||||
let before = gauge.get();
|
||||
|
||||
loc.mount_ec_shards(VolumeId(11), "dupmount", &[0], "").unwrap();
|
||||
loc.mount_ec_shards(VolumeId(11), "dupmount", &[0], "")
|
||||
.expect("a duplicate mount must succeed as a no-op");
|
||||
|
||||
assert_eq!(
|
||||
loc.find_ec_volume(VolumeId(11)).map(|v| v.shard_count()),
|
||||
Some(1),
|
||||
);
|
||||
assert_eq!(
|
||||
gauge.get(),
|
||||
before + 1.0,
|
||||
"the duplicate mount must not bump the ec_shards gauge",
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_disk_location_persists_directory_uuid_and_tags() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
|
||||
@@ -465,6 +465,28 @@ pub fn resolve_status(
|
||||
}
|
||||
}
|
||||
|
||||
/// Whether a generation-matching sidecar agrees with the geometry the volume is
|
||||
/// mounted with. Both files record the layout the generation was encoded with,
|
||||
/// so a disagreement means one of them is wrong and reads through the other
|
||||
/// would land at the wrong shard offsets — the caller fails the mount rather
|
||||
/// than merely dropping protection. A sidecar that records no EC config has
|
||||
/// nothing to contradict.
|
||||
pub fn geometry_matches(
|
||||
prot: &EcBitrotProtection,
|
||||
data_shards: usize,
|
||||
parity_shards: usize,
|
||||
block_size: i64,
|
||||
) -> bool {
|
||||
match &prot.ec_shard_config {
|
||||
None => true,
|
||||
Some(cfg) => {
|
||||
cfg.data_shards as usize == data_shards
|
||||
&& cfg.parity_shards as usize == parity_shards
|
||||
&& cfg.block_size == block_size
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Returns the [`EcShardChecksums`] entry for a shard id, or `None`.
|
||||
pub fn shard_checksums(prot: &EcBitrotProtection, shard_id: u32) -> Option<&EcShardChecksums> {
|
||||
prot.shards.iter().find(|s| s.shard_id == shard_id)
|
||||
@@ -539,11 +561,12 @@ fn read_full_at(f: &File, buf: &mut [u8], offset: u64) -> io::Result<()> {
|
||||
|
||||
/// Builds the `EcShardConfig` proto for the given layout. The bitrot sidecar
|
||||
/// carries its own top-level encode_uuid, so the nested config leaves it empty.
|
||||
pub fn ec_shard_config(data_shards: u32, parity_shards: u32) -> EcShardConfig {
|
||||
pub fn ec_shard_config(data_shards: u32, parity_shards: u32, block_size: i64) -> EcShardConfig {
|
||||
EcShardConfig {
|
||||
data_shards,
|
||||
parity_shards,
|
||||
encode_ts_ns: 0,
|
||||
block_size,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -567,6 +590,7 @@ mod tests {
|
||||
data_shards: 10,
|
||||
parity_shards: 4,
|
||||
encode_ts_ns: 0,
|
||||
block_size: 0,
|
||||
}),
|
||||
shards: vec![
|
||||
EcShardChecksums {
|
||||
@@ -712,7 +736,7 @@ mod tests {
|
||||
algorithm: ChecksumAlgorithm::ChecksumCrc32c as i32,
|
||||
block_size: DEFAULT_BITROT_BLOCK_SIZE as u32,
|
||||
generation: 0,
|
||||
ec_shard_config: Some(ec_shard_config(10, 4)),
|
||||
ec_shard_config: Some(ec_shard_config(10, 4, 0)),
|
||||
shards: vec![EcShardChecksums {
|
||||
shard_id: 0,
|
||||
covered_size: covered,
|
||||
@@ -750,7 +774,7 @@ mod tests {
|
||||
algorithm: ChecksumAlgorithm::ChecksumCrc32c as i32,
|
||||
block_size: DEFAULT_BITROT_BLOCK_SIZE as u32,
|
||||
generation: 0,
|
||||
ec_shard_config: Some(ec_shard_config(10, 4)),
|
||||
ec_shard_config: Some(ec_shard_config(10, 4, 0)),
|
||||
shards: vec![EcShardChecksums {
|
||||
shard_id: 0,
|
||||
covered_size: 5,
|
||||
@@ -794,7 +818,7 @@ mod tests {
|
||||
algorithm: ChecksumAlgorithm::ChecksumCrc32c as i32,
|
||||
block_size: DEFAULT_BITROT_BLOCK_SIZE as u32,
|
||||
generation: 0,
|
||||
ec_shard_config: Some(ec_shard_config(10, 4)),
|
||||
ec_shard_config: Some(ec_shard_config(10, 4, 0)),
|
||||
shards,
|
||||
encode_uuid: vec![0u8; 16],
|
||||
}
|
||||
|
||||
@@ -80,6 +80,8 @@ pub fn write_dat_file_from_shards(
|
||||
dat_file_size: i64,
|
||||
encoded_dat_file_size: i64,
|
||||
data_shards: usize,
|
||||
large_block_size: usize,
|
||||
small_block_size: usize,
|
||||
) -> io::Result<()> {
|
||||
let dirs: Vec<String> = (0..data_shards).map(|_| dir.to_string()).collect();
|
||||
write_dat_file_from_shards_with_dirs(
|
||||
@@ -90,6 +92,8 @@ pub fn write_dat_file_from_shards(
|
||||
encoded_dat_file_size,
|
||||
data_shards,
|
||||
&dirs,
|
||||
large_block_size,
|
||||
small_block_size,
|
||||
)
|
||||
}
|
||||
|
||||
@@ -113,7 +117,10 @@ pub fn write_dat_file_from_shards(
|
||||
/// boundary, and deriving the layout from the shrunk extent would read
|
||||
/// the shards in the wrong block order. Pass zero when the .vif does
|
||||
/// not record the encode-time size to infer the layout from the shard
|
||||
/// size.
|
||||
/// size. `large_block_size`/`small_block_size` are the volume's shard
|
||||
/// block layout, e.g. `EcVolume::large_block_size()` /
|
||||
/// `small_block_size()` from its .vif EC config.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
pub fn write_dat_file_from_shards_with_dirs(
|
||||
dat_dir: &str,
|
||||
collection: &str,
|
||||
@@ -122,6 +129,8 @@ pub fn write_dat_file_from_shards_with_dirs(
|
||||
encoded_dat_file_size: i64,
|
||||
data_shards: usize,
|
||||
shard_dirs: &[String],
|
||||
large_block_size: usize,
|
||||
small_block_size: usize,
|
||||
) -> io::Result<()> {
|
||||
write_dat_file(
|
||||
dat_dir,
|
||||
@@ -131,8 +140,8 @@ pub fn write_dat_file_from_shards_with_dirs(
|
||||
encoded_dat_file_size,
|
||||
data_shards,
|
||||
shard_dirs,
|
||||
ERASURE_CODING_LARGE_BLOCK_SIZE,
|
||||
ERASURE_CODING_SMALL_BLOCK_SIZE,
|
||||
large_block_size,
|
||||
small_block_size,
|
||||
)
|
||||
}
|
||||
|
||||
@@ -412,7 +421,9 @@ mod tests {
|
||||
// Encode to EC
|
||||
let data_shards = 10;
|
||||
let parity_shards = 4;
|
||||
ec_encoder::write_ec_files(dir, dir, "", VolumeId(1), data_shards, parity_shards).unwrap();
|
||||
let block_size =
|
||||
ec_encoder::write_ec_files(dir, dir, "", VolumeId(1), data_shards, parity_shards)
|
||||
.unwrap();
|
||||
|
||||
// Delete original .dat and .idx
|
||||
std::fs::remove_file(format!("{}/1.dat", dir)).unwrap();
|
||||
@@ -426,6 +437,8 @@ mod tests {
|
||||
original_dat_size as i64,
|
||||
original_dat_size as i64,
|
||||
data_shards,
|
||||
block_size as usize,
|
||||
block_size as usize,
|
||||
)
|
||||
.unwrap();
|
||||
write_idx_file_from_ec_index(dir, "", VolumeId(1)).unwrap();
|
||||
@@ -472,7 +485,16 @@ mod tests {
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
// No shard files exist, so de-striping must fail and publish nothing:
|
||||
// neither the final .dat nor a partial .dat.tmp may remain.
|
||||
let res = write_dat_file_from_shards(dir, "", VolumeId(7), 100, 100, 10);
|
||||
let res = write_dat_file_from_shards(
|
||||
dir,
|
||||
"",
|
||||
VolumeId(7),
|
||||
100,
|
||||
100,
|
||||
10,
|
||||
ERASURE_CODING_LARGE_BLOCK_SIZE,
|
||||
ERASURE_CODING_SMALL_BLOCK_SIZE,
|
||||
);
|
||||
assert!(res.is_err());
|
||||
assert!(!std::path::Path::new(&format!("{}/7.dat", dir)).exists());
|
||||
assert!(!std::path::Path::new(&format!("{}/7.dat.tmp", dir)).exists());
|
||||
@@ -525,6 +547,7 @@ mod tests {
|
||||
&mut builders,
|
||||
data_shards,
|
||||
parity_shards,
|
||||
SMALL,
|
||||
LARGE,
|
||||
SMALL,
|
||||
)
|
||||
@@ -616,6 +639,7 @@ mod tests {
|
||||
&mut builders,
|
||||
data_shards,
|
||||
parity_shards,
|
||||
SMALL,
|
||||
LARGE,
|
||||
SMALL,
|
||||
)
|
||||
|
||||
@@ -25,6 +25,10 @@ use crate::storage::volume::volume_file_name;
|
||||
///
|
||||
/// Creates .ec00-.ec13 files in the same directory.
|
||||
/// Also creates a sorted .ecx index from the .idx file.
|
||||
///
|
||||
/// Always encodes with the uniform block layout, sized for this .dat, and
|
||||
/// returns the block size so the caller can persist it to .vif. Mirrors Go's
|
||||
/// WriteEcFiles.
|
||||
pub fn write_ec_files(
|
||||
dir: &str,
|
||||
idx_dir: &str,
|
||||
@@ -32,7 +36,7 @@ pub fn write_ec_files(
|
||||
volume_id: VolumeId,
|
||||
data_shards: usize,
|
||||
parity_shards: usize,
|
||||
) -> io::Result<()> {
|
||||
) -> io::Result<i64> {
|
||||
let base = volume_file_name(dir, collection, volume_id);
|
||||
let dat_path = format!("{}.dat", base);
|
||||
let idx_base = volume_file_name(idx_dir, collection, volume_id);
|
||||
@@ -66,7 +70,7 @@ pub fn write_ec_files(
|
||||
.map(|_| ShardChecksumBuilder::new(DEFAULT_BITROT_BLOCK_SIZE as i64))
|
||||
.collect();
|
||||
|
||||
// Encode in large blocks, then small blocks
|
||||
let block_size = uniform_block_size(dat_size, data_shards);
|
||||
encode_dat_file(
|
||||
&dat_file,
|
||||
dat_size,
|
||||
@@ -75,8 +79,9 @@ pub fn write_ec_files(
|
||||
&mut builders,
|
||||
data_shards,
|
||||
parity_shards,
|
||||
ERASURE_CODING_LARGE_BLOCK_SIZE,
|
||||
ERASURE_CODING_SMALL_BLOCK_SIZE,
|
||||
ENCODE_BUFFER_SIZE,
|
||||
block_size as usize,
|
||||
block_size as usize,
|
||||
)?;
|
||||
|
||||
// Close all shards
|
||||
@@ -103,6 +108,7 @@ pub fn write_ec_files(
|
||||
ec_shard_config: Some(ec_bitrot::ec_shard_config(
|
||||
data_shards as u32,
|
||||
parity_shards as u32,
|
||||
block_size,
|
||||
)),
|
||||
shards: shard_checksums,
|
||||
encode_uuid: ec_bitrot::new_encode_uuid(),
|
||||
@@ -120,7 +126,19 @@ pub fn write_ec_files(
|
||||
);
|
||||
}
|
||||
|
||||
Ok(())
|
||||
Ok(block_size)
|
||||
}
|
||||
|
||||
/// uniform_block_size returns the per-shard block size of the uniform layout
|
||||
/// for a .dat of the given size: ceil(dat_file_size/data_shards) rounded up to
|
||||
/// a whole small block. For every input this equals the legacy layout's padded
|
||||
/// shard size, so only the byte placement differs between the two layouts,
|
||||
/// never the shard length. Mirrors Go's UniformBlockSize.
|
||||
pub fn uniform_block_size(dat_file_size: i64, data_shards: usize) -> i64 {
|
||||
let small = ERASURE_CODING_SMALL_BLOCK_SIZE as i64;
|
||||
let per_shard = (dat_file_size + data_shards as i64 - 1) / data_shards as i64;
|
||||
let blocks = ((per_shard + small - 1) / small).max(1);
|
||||
blocks * small
|
||||
}
|
||||
|
||||
/// Rebuild missing EC shard files from existing shards using Reed-Solomon reconstruct.
|
||||
@@ -370,7 +388,7 @@ pub fn verify_ec_shards(
|
||||
}
|
||||
|
||||
/// Write sorted .ecx index from .idx file.
|
||||
fn write_sorted_ecx_from_idx(idx_path: &str, ecx_path: &str) -> io::Result<()> {
|
||||
pub(crate) fn write_sorted_ecx_from_idx(idx_path: &str, ecx_path: &str) -> io::Result<()> {
|
||||
if !std::path::Path::new(idx_path).exists() {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::NotFound,
|
||||
@@ -421,6 +439,8 @@ pub fn rebuild_ecx_file(
|
||||
collection: &str,
|
||||
volume_id: VolumeId,
|
||||
data_shards: usize,
|
||||
block_size: i64,
|
||||
dat_file_size: i64,
|
||||
additional_dirs: &[&str],
|
||||
) -> io::Result<()> {
|
||||
use crate::storage::needle::needle::get_actual_size;
|
||||
@@ -463,10 +483,42 @@ pub fn rebuild_ecx_file(
|
||||
// Determine total logical data size from shard sizes
|
||||
let shard_size = shards.iter().map(|s| s.file_size()).max().unwrap_or(0);
|
||||
let total_data_size = shard_size as i64 * data_shards as i64;
|
||||
// The volume's shard block layout: the .vif-recorded uniform block size,
|
||||
// or the legacy two-tier sizes when 0. The row count comes from the shard
|
||||
// length; -1 disambiguates a legacy shard that is an exact large-block
|
||||
// multiple (mirrors the ecdFileSize-1 fallback in the read path).
|
||||
let (large_block, small_block) = if block_size > 0 {
|
||||
(block_size, block_size)
|
||||
} else {
|
||||
(
|
||||
ERASURE_CODING_LARGE_BLOCK_SIZE as i64,
|
||||
ERASURE_CODING_SMALL_BLOCK_SIZE as i64,
|
||||
)
|
||||
};
|
||||
// The row count the de-stripe walks with. The encode-time .dat size is the
|
||||
// authority — the same value the read path divides by data_shards — and the
|
||||
// padded extent is only a fallback: under the legacy layout a shard that is
|
||||
// an exact large-block multiple reads as one row too many, which
|
||||
// re-interprets its last large row as small blocks and scrambles the
|
||||
// recovered offsets. Subtracting one keeps that fallback on the safe side of
|
||||
// the boundary, exactly as the read path's own fallback does.
|
||||
let locate_shard_size = if dat_file_size > 0 {
|
||||
dat_file_size / data_shards as i64
|
||||
} else {
|
||||
(shard_size as i64 - 1).max(0)
|
||||
};
|
||||
|
||||
// Read version from superblock (first byte of logical data)
|
||||
let mut sb_buf = [0u8; SUPER_BLOCK_SIZE];
|
||||
read_from_data_shards(&shards, &mut sb_buf, 0, data_shards)?;
|
||||
read_from_data_shards(
|
||||
&shards,
|
||||
&mut sb_buf,
|
||||
0,
|
||||
data_shards,
|
||||
locate_shard_size,
|
||||
large_block,
|
||||
small_block,
|
||||
)?;
|
||||
let version = Version(sb_buf[0]);
|
||||
|
||||
// Walk needles starting after superblock
|
||||
@@ -475,10 +527,30 @@ pub fn rebuild_ecx_file(
|
||||
let mut entries: Vec<(NeedleId, Offset, Size)> = Vec::new();
|
||||
|
||||
while offset + header_size as i64 <= total_data_size {
|
||||
// Read needle header (cookie + needle_id + size = 16 bytes)
|
||||
// Read needle header (cookie + needle_id + size = 16 bytes).
|
||||
// A read failure is NOT the end of the data — every offset in
|
||||
// range maps into the shards, so an error means a truncated or
|
||||
// unreadable shard. Publishing the entries collected so far as
|
||||
// a successful .ecx would hand out a silently incomplete
|
||||
// recovery index; propagate instead. (The scan still ends
|
||||
// normally on the zero-cookie tail below.)
|
||||
let mut header_buf = [0u8; NEEDLE_HEADER_SIZE];
|
||||
if read_from_data_shards(&shards, &mut header_buf, offset as u64, data_shards).is_err() {
|
||||
break;
|
||||
if let Err(e) = read_from_data_shards(
|
||||
&shards,
|
||||
&mut header_buf,
|
||||
offset as u64,
|
||||
data_shards,
|
||||
locate_shard_size,
|
||||
large_block,
|
||||
small_block,
|
||||
) {
|
||||
for s in &mut shards {
|
||||
s.close();
|
||||
}
|
||||
return Err(io::Error::new(
|
||||
e.kind(),
|
||||
format!("scan needle header at offset {}: {}", offset, e),
|
||||
));
|
||||
}
|
||||
|
||||
let cookie = Cookie::from_bytes(&header_buf[..COOKIE_SIZE]);
|
||||
@@ -532,58 +604,83 @@ pub fn rebuild_ecx_file(
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Read bytes from EC data shards at a logical offset in the .dat file.
|
||||
/// Read bytes from EC data shards at a logical offset in the .dat file,
|
||||
/// resolving the shard/offset through the volume's block layout via
|
||||
/// locate_data — the same mapping the read path uses.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
fn read_from_data_shards(
|
||||
shards: &[EcVolumeShard],
|
||||
buf: &mut [u8],
|
||||
logical_offset: u64,
|
||||
data_shards: usize,
|
||||
locate_shard_size: i64,
|
||||
large_block_size: i64,
|
||||
small_block_size: i64,
|
||||
) -> io::Result<()> {
|
||||
let small_block = ERASURE_CODING_SMALL_BLOCK_SIZE as u64;
|
||||
let data_shards_u64 = data_shards as u64;
|
||||
|
||||
let mut bytes_read = 0u64;
|
||||
let mut remaining = buf.len() as u64;
|
||||
let mut current_offset = logical_offset;
|
||||
|
||||
while remaining > 0 {
|
||||
// Determine which shard and at what shard-offset this logical offset maps to.
|
||||
// The data is interleaved: large blocks first, then small blocks.
|
||||
// For simplicity, use the small block size for all calculations since
|
||||
// large blocks are multiples of small blocks.
|
||||
let row_size = small_block * data_shards_u64;
|
||||
let row_index = current_offset / row_size;
|
||||
let row_offset = current_offset % row_size;
|
||||
let shard_index = (row_offset / small_block) as usize;
|
||||
let shard_offset = row_index * small_block + (row_offset % small_block);
|
||||
|
||||
if shard_index >= data_shards {
|
||||
let intervals = crate::storage::erasure_coding::ec_locate::locate_data(
|
||||
logical_offset as i64,
|
||||
Size(buf.len() as i32),
|
||||
locate_shard_size,
|
||||
data_shards as u32,
|
||||
large_block_size,
|
||||
small_block_size,
|
||||
);
|
||||
let mut bytes_read = 0usize;
|
||||
for interval in &intervals {
|
||||
let (shard_id, shard_offset) =
|
||||
interval.to_shard_id_and_offset(data_shards as u32, large_block_size, small_block_size);
|
||||
if shard_id as usize >= data_shards {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::InvalidInput,
|
||||
"shard index out of range",
|
||||
));
|
||||
}
|
||||
|
||||
// How many bytes can we read from this position in this shard block
|
||||
let bytes_left_in_block = small_block - (row_offset % small_block);
|
||||
let to_read = remaining.min(bytes_left_in_block) as usize;
|
||||
|
||||
let dest = &mut buf[bytes_read as usize..bytes_read as usize + to_read];
|
||||
shards[shard_index].read_at(dest, shard_offset)?;
|
||||
|
||||
bytes_read += to_read as u64;
|
||||
remaining -= to_read as u64;
|
||||
current_offset += to_read as u64;
|
||||
let to_read = interval.size as usize;
|
||||
let dest = &mut buf[bytes_read..bytes_read + to_read];
|
||||
// Exact-read semantics: read_at may legally return fewer bytes
|
||||
// than requested, and treating a short read as complete leaves
|
||||
// the tail of `dest` as whatever the buffer held before. Loop
|
||||
// until filled; zero bytes inside the mapped range means the
|
||||
// shard is truncated — an error, not an end.
|
||||
let mut filled = 0usize;
|
||||
while filled < to_read {
|
||||
let n = shards[shard_id as usize]
|
||||
.read_at(&mut dest[filled..], shard_offset as u64 + filled as u64)?;
|
||||
if n == 0 {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::UnexpectedEof,
|
||||
format!(
|
||||
"short read from data shard {}: {} of {} bytes at offset {}",
|
||||
shard_id, filled, to_read, shard_offset
|
||||
),
|
||||
));
|
||||
}
|
||||
filled += n;
|
||||
}
|
||||
bytes_read += to_read;
|
||||
}
|
||||
if bytes_read != buf.len() {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::UnexpectedEof,
|
||||
"short read from data shards",
|
||||
));
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Buffer size for one encode sub-batch per shard, mirroring Go's 256KB
|
||||
/// bufferSize in WriteEcFiles. A block is processed in block_size/buffer_size
|
||||
/// sub-batches, so memory stays at total_shards * 256KB no matter how large
|
||||
/// the uniform block is.
|
||||
const ENCODE_BUFFER_SIZE: usize = 256 * 1024;
|
||||
|
||||
/// Encode the .dat file data into shard files.
|
||||
///
|
||||
/// Uses a two-phase approach matching Go's ec_encoder.go:
|
||||
/// 1. Process as many large blocks (1GB) as possible
|
||||
/// 2. Process remaining data with small blocks (1MB)
|
||||
/// 1. Process as many large blocks as possible
|
||||
/// 2. Process remaining data with small blocks
|
||||
///
|
||||
/// `buffer_size` must divide both block sizes.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
pub(crate) fn encode_dat_file(
|
||||
dat_file: &File,
|
||||
@@ -593,44 +690,50 @@ pub(crate) fn encode_dat_file(
|
||||
builders: &mut [ShardChecksumBuilder],
|
||||
data_shards: usize,
|
||||
parity_shards: usize,
|
||||
buffer_size: usize,
|
||||
large_block_size: usize,
|
||||
small_block_size: usize,
|
||||
) -> io::Result<()> {
|
||||
let total_shards = data_shards + parity_shards;
|
||||
let mut buffers: Vec<Vec<u8>> = (0..total_shards)
|
||||
.map(|_| vec![0u8; buffer_size])
|
||||
.collect();
|
||||
|
||||
let mut remaining = dat_size;
|
||||
let mut offset: u64 = 0;
|
||||
|
||||
// Phase 1: Process large blocks (1GB each) while enough data remains
|
||||
// Phase 1: process whole large-block rows while enough data remains
|
||||
let large_row_size = large_block_size * data_shards;
|
||||
|
||||
while remaining >= large_row_size as i64 {
|
||||
encode_one_batch(
|
||||
encode_data(
|
||||
dat_file,
|
||||
offset,
|
||||
large_block_size,
|
||||
rs,
|
||||
&mut buffers,
|
||||
shards,
|
||||
builders,
|
||||
data_shards,
|
||||
parity_shards,
|
||||
)?;
|
||||
offset += large_row_size as u64;
|
||||
remaining -= large_row_size as i64;
|
||||
}
|
||||
|
||||
// Phase 2: Process remaining data with small blocks (1MB each)
|
||||
// Phase 2: process remaining data with small blocks
|
||||
let small_row_size = small_block_size * data_shards;
|
||||
|
||||
while remaining > 0 {
|
||||
let to_process = remaining.min(small_row_size as i64);
|
||||
encode_one_batch(
|
||||
encode_data(
|
||||
dat_file,
|
||||
offset,
|
||||
small_block_size,
|
||||
rs,
|
||||
&mut buffers,
|
||||
shards,
|
||||
builders,
|
||||
data_shards,
|
||||
parity_shards,
|
||||
)?;
|
||||
offset += to_process as u64;
|
||||
remaining -= to_process;
|
||||
@@ -639,61 +742,71 @@ pub(crate) fn encode_dat_file(
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Encode one batch (row) of data.
|
||||
/// Encode one row of blocks, streaming it in ENCODE_BUFFER_SIZE sub-batches so
|
||||
/// arbitrarily large blocks never require block-sized allocations. Mirrors
|
||||
/// Go's encodeData.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
fn encode_data(
|
||||
dat_file: &File,
|
||||
row_offset: u64,
|
||||
block_size: usize,
|
||||
rs: &ReedSolomon,
|
||||
buffers: &mut [Vec<u8>],
|
||||
shards: &mut [EcVolumeShard],
|
||||
builders: &mut [ShardChecksumBuilder],
|
||||
data_shards: usize,
|
||||
) -> io::Result<()> {
|
||||
let buffer_size = buffers[0].len();
|
||||
if block_size % buffer_size != 0 {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::InvalidInput,
|
||||
format!(
|
||||
"unexpected block size {} buffer size {}",
|
||||
block_size, buffer_size
|
||||
),
|
||||
));
|
||||
}
|
||||
let batch_count = block_size / buffer_size;
|
||||
for b in 0..batch_count {
|
||||
encode_one_batch(
|
||||
dat_file,
|
||||
row_offset + (b * buffer_size) as u64,
|
||||
block_size,
|
||||
rs,
|
||||
buffers,
|
||||
shards,
|
||||
builders,
|
||||
data_shards,
|
||||
)?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Encode one sub-batch: the same buffer-sized slice of every shard's block in
|
||||
/// this row. Mirrors Go's encodeDataOneBatch.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
fn encode_one_batch(
|
||||
dat_file: &File,
|
||||
offset: u64,
|
||||
block_size: usize,
|
||||
rs: &ReedSolomon,
|
||||
buffers: &mut [Vec<u8>],
|
||||
shards: &mut [EcVolumeShard],
|
||||
builders: &mut [ShardChecksumBuilder],
|
||||
data_shards: usize,
|
||||
parity_shards: usize,
|
||||
) -> io::Result<()> {
|
||||
let total_shards = data_shards + parity_shards;
|
||||
// Each batch allocates block_size * total_shards bytes.
|
||||
// With large blocks (1 GiB) this is 14 GiB -- guard against OOM.
|
||||
let total_alloc = block_size.checked_mul(total_shards).ok_or_else(|| {
|
||||
io::Error::new(
|
||||
io::ErrorKind::InvalidInput,
|
||||
"block_size * shard count overflows usize",
|
||||
)
|
||||
})?;
|
||||
// Large-block encoding uses 1 GiB * 14 shards = 14 GiB; allow up to 16 GiB.
|
||||
const MAX_BATCH_ALLOC: usize = 16 * 1024 * 1024 * 1024; // 16 GiB safety limit
|
||||
if total_alloc > MAX_BATCH_ALLOC {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::InvalidInput,
|
||||
format!(
|
||||
"batch allocation too large ({} bytes, limit {} bytes); block_size={} shards={}",
|
||||
total_alloc, MAX_BATCH_ALLOC, block_size, total_shards,
|
||||
),
|
||||
));
|
||||
}
|
||||
|
||||
// Allocate buffers for all shards
|
||||
let mut buffers: Vec<Vec<u8>> = (0..total_shards).map(|_| vec![0u8; block_size]).collect();
|
||||
|
||||
// Read data shards from .dat file
|
||||
// Read data shards from the .dat file, zero-filling past EOF — the buffers
|
||||
// are reused across batches, so the tail must be cleared explicitly.
|
||||
for i in 0..data_shards {
|
||||
let read_offset = offset + (i * block_size) as u64;
|
||||
|
||||
#[cfg(unix)]
|
||||
{
|
||||
use std::os::unix::fs::FileExt;
|
||||
dat_file.read_at(&mut buffers[i], read_offset)?;
|
||||
}
|
||||
|
||||
#[cfg(not(unix))]
|
||||
{
|
||||
let mut f = dat_file.try_clone()?;
|
||||
f.seek(SeekFrom::Start(read_offset))?;
|
||||
f.read(&mut buffers[i])?;
|
||||
let n = read_at_most(dat_file, &mut buffers[i], read_offset)?;
|
||||
for b in buffers[i][n..].iter_mut() {
|
||||
*b = 0;
|
||||
}
|
||||
}
|
||||
|
||||
// Encode parity shards
|
||||
rs.encode(&mut buffers).map_err(|e| {
|
||||
rs.encode(&mut *buffers).map_err(|e| {
|
||||
io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!("reed-solomon encode: {:?}", e),
|
||||
@@ -710,6 +823,29 @@ fn encode_one_batch(
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Read into `buf` at `offset` until it is full or EOF; returns bytes read.
|
||||
fn read_at_most(dat_file: &File, buf: &mut [u8], offset: u64) -> io::Result<usize> {
|
||||
let mut n = 0;
|
||||
while n < buf.len() {
|
||||
#[cfg(unix)]
|
||||
let r = {
|
||||
use std::os::unix::fs::FileExt;
|
||||
dat_file.read_at(&mut buf[n..], offset + n as u64)?
|
||||
};
|
||||
#[cfg(not(unix))]
|
||||
let r = {
|
||||
let mut f = dat_file.try_clone()?;
|
||||
f.seek(SeekFrom::Start(offset + n as u64))?;
|
||||
f.read(&mut buf[n..])?
|
||||
};
|
||||
if r == 0 {
|
||||
break;
|
||||
}
|
||||
n += r;
|
||||
}
|
||||
Ok(n)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
@@ -1020,7 +1156,7 @@ mod tests {
|
||||
|
||||
// Without additional_dirs, rebuild must fail: shards 1, 3, 6 are not
|
||||
// in primary and the full logical .dat content can't be reconstructed.
|
||||
let res = rebuild_ecx_file(&primary, "", VolumeId(1), 10, &[]);
|
||||
let res = rebuild_ecx_file(&primary, "", VolumeId(1), 10, 0, 0, &[]);
|
||||
assert!(
|
||||
res.is_err(),
|
||||
"ecx rebuild without additional_dirs must fail when data shards are on another disk"
|
||||
@@ -1031,7 +1167,7 @@ mod tests {
|
||||
);
|
||||
|
||||
// With additional_dirs pointing at the secondary, the rebuild must succeed.
|
||||
rebuild_ecx_file(&primary, "", VolumeId(1), 10, &[secondary.as_str()]).unwrap();
|
||||
rebuild_ecx_file(&primary, "", VolumeId(1), 10, 0, 0, &[secondary.as_str()]).unwrap();
|
||||
|
||||
assert!(
|
||||
std::path::Path::new(&ecx_path).exists(),
|
||||
@@ -1043,6 +1179,118 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
// A uniform-layout volume (block size > 1MiB) must have its .ecx rebuilt
|
||||
// through the recorded geometry; the legacy 1MiB mapping would scan
|
||||
// garbage past the first block boundary.
|
||||
#[test]
|
||||
fn test_rebuild_ecx_file_uniform_layout() {
|
||||
use crate::storage::needle_map::NeedleMapKind;
|
||||
use crate::storage::volume::Volume;
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap().to_string();
|
||||
let mut v = Volume::new(
|
||||
&dir,
|
||||
&dir,
|
||||
"",
|
||||
VolumeId(2),
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
)
|
||||
.unwrap();
|
||||
for i in 1u64..=12 {
|
||||
let data: Vec<u8> = (0..2 << 20)
|
||||
.map(|b| ((b as u64).wrapping_mul(2654435761).wrapping_add(i) >> 8) as u8)
|
||||
.collect();
|
||||
let mut n = Needle {
|
||||
id: NeedleId(i),
|
||||
cookie: Cookie(i as u32),
|
||||
data: data.clone(),
|
||||
data_size: data.len() as u32,
|
||||
..Needle::default()
|
||||
};
|
||||
v.write_needle(&mut n, true, false).unwrap();
|
||||
}
|
||||
v.sync_to_disk().unwrap();
|
||||
v.close();
|
||||
let block_size = write_ec_files(&dir, &dir, "", VolumeId(2), 10, 4).unwrap();
|
||||
assert!(
|
||||
block_size > ERASURE_CODING_SMALL_BLOCK_SIZE as i64,
|
||||
"fixture must diverge from the legacy layout"
|
||||
);
|
||||
|
||||
let ecx_path = format!("{}/2.ecx", dir);
|
||||
let canonical = std::fs::read(&ecx_path).unwrap();
|
||||
std::fs::remove_file(&ecx_path).unwrap();
|
||||
|
||||
rebuild_ecx_file(&dir, "", VolumeId(2), 10, block_size, 0, &[]).unwrap();
|
||||
let rebuilt = std::fs::read(&ecx_path).unwrap();
|
||||
assert_eq!(canonical, rebuilt, "rebuilt .ecx must match the encode-time .ecx");
|
||||
}
|
||||
|
||||
// A truncated data shard must FAIL the .ecx rebuild, not publish the
|
||||
// entries scanned so far as a successful (silently incomplete) index.
|
||||
#[test]
|
||||
fn test_rebuild_ecx_file_fails_on_truncated_shard() {
|
||||
use crate::storage::needle_map::NeedleMapKind;
|
||||
use crate::storage::volume::Volume;
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap().to_string();
|
||||
let mut v = Volume::new(
|
||||
&dir,
|
||||
&dir,
|
||||
"",
|
||||
VolumeId(3),
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
)
|
||||
.unwrap();
|
||||
for i in 1u64..=12 {
|
||||
let data: Vec<u8> = (0..2 << 20)
|
||||
.map(|b| ((b as u64).wrapping_mul(2654435761).wrapping_add(i) >> 8) as u8)
|
||||
.collect();
|
||||
let mut n = Needle {
|
||||
id: NeedleId(i),
|
||||
cookie: Cookie(i as u32),
|
||||
data: data.clone(),
|
||||
data_size: data.len() as u32,
|
||||
..Needle::default()
|
||||
};
|
||||
v.write_needle(&mut n, true, false).unwrap();
|
||||
}
|
||||
v.sync_to_disk().unwrap();
|
||||
v.close();
|
||||
let block_size = write_ec_files(&dir, &dir, "", VolumeId(3), 10, 4).unwrap();
|
||||
|
||||
let ecx_path = format!("{}/3.ecx", dir);
|
||||
std::fs::remove_file(&ecx_path).unwrap();
|
||||
|
||||
// Truncate shard 0 to just the superblock: the scan's very first
|
||||
// needle-header read (offset SUPER_BLOCK_SIZE, shard 0 under the
|
||||
// uniform layout) lands in the missing region. The pre-fix code
|
||||
// broke the scan there and published an EMPTY .ecx as success.
|
||||
let shard_path = format!("{}/3.ec00", dir);
|
||||
let f = std::fs::OpenOptions::new()
|
||||
.write(true)
|
||||
.open(&shard_path)
|
||||
.unwrap();
|
||||
f.set_len(crate::storage::super_block::SUPER_BLOCK_SIZE as u64)
|
||||
.unwrap();
|
||||
drop(f);
|
||||
|
||||
let res = rebuild_ecx_file(&dir, "", VolumeId(3), 10, block_size, 0, &[]);
|
||||
assert!(res.is_err(), "rebuild over a truncated shard must fail");
|
||||
assert!(
|
||||
!std::path::Path::new(&ecx_path).exists(),
|
||||
"a failed rebuild must not leave a partial .ecx behind"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_reed_solomon_basic() {
|
||||
let data_shards = 10;
|
||||
|
||||
@@ -18,21 +18,26 @@ pub struct Interval {
|
||||
}
|
||||
|
||||
impl Interval {
|
||||
pub fn to_shard_id_and_offset(&self, data_shards: u32) -> (ShardId, i64) {
|
||||
pub fn to_shard_id_and_offset(
|
||||
&self,
|
||||
data_shards: u32,
|
||||
large_block_size: i64,
|
||||
small_block_size: i64,
|
||||
) -> (ShardId, i64) {
|
||||
let data_shards_usize = data_shards as usize;
|
||||
let shard_id = (self.block_index % data_shards_usize) as ShardId;
|
||||
let row_index = self.block_index / data_shards_usize;
|
||||
|
||||
let block_size = if self.is_large_block {
|
||||
ERASURE_CODING_LARGE_BLOCK_SIZE as i64
|
||||
large_block_size
|
||||
} else {
|
||||
ERASURE_CODING_SMALL_BLOCK_SIZE as i64
|
||||
small_block_size
|
||||
};
|
||||
|
||||
let mut offset = row_index as i64 * block_size + self.inner_block_offset;
|
||||
if !self.is_large_block {
|
||||
// Small blocks come after large blocks in the shard file
|
||||
offset += self.large_block_rows_count as i64 * ERASURE_CODING_LARGE_BLOCK_SIZE as i64;
|
||||
offset += self.large_block_rows_count as i64 * large_block_size;
|
||||
}
|
||||
|
||||
(shard_id, offset)
|
||||
@@ -42,7 +47,14 @@ impl Interval {
|
||||
/// Locate the EC shard intervals needed to read data at the given offset and size.
|
||||
///
|
||||
/// `shard_size` is the size of a single shard file.
|
||||
pub fn locate_data(offset: i64, size: Size, shard_size: i64, data_shards: u32) -> Vec<Interval> {
|
||||
pub fn locate_data(
|
||||
offset: i64,
|
||||
size: Size,
|
||||
shard_size: i64,
|
||||
data_shards: u32,
|
||||
large_block_size: i64,
|
||||
small_block_size: i64,
|
||||
) -> Vec<Interval> {
|
||||
let mut intervals = Vec::new();
|
||||
let data_size = size.0 as i64;
|
||||
|
||||
@@ -50,17 +62,14 @@ pub fn locate_data(offset: i64, size: Size, shard_size: i64, data_shards: u32) -
|
||||
return intervals;
|
||||
}
|
||||
|
||||
let large_block_size = ERASURE_CODING_LARGE_BLOCK_SIZE as i64;
|
||||
let small_block_size = ERASURE_CODING_SMALL_BLOCK_SIZE as i64;
|
||||
let large_row_size = large_block_size * data_shards as i64;
|
||||
let small_row_size = small_block_size * data_shards as i64;
|
||||
|
||||
// Number of large block rows
|
||||
let n_large_block_rows = if shard_size > 0 {
|
||||
((shard_size - 1) / large_block_size) as usize
|
||||
} else {
|
||||
0
|
||||
};
|
||||
// Number of large block rows. Mirrors Go's shardDatSize/largeBlockLength:
|
||||
// the caller's ecd-size fallback already subtracts 1 to disambiguate the
|
||||
// exact-multiple case, so no further -1 here — a shard size that IS an
|
||||
// exact multiple (dat_file_size path) means real full large rows.
|
||||
let n_large_block_rows = (shard_size / large_block_size) as usize;
|
||||
let large_section_size = n_large_block_rows as i64 * large_row_size;
|
||||
|
||||
let mut remaining_offset = offset;
|
||||
@@ -150,7 +159,11 @@ mod tests {
|
||||
is_large_block: true,
|
||||
large_block_rows_count: 1,
|
||||
};
|
||||
let (shard_id, offset) = interval.to_shard_id_and_offset(data_shards);
|
||||
let (shard_id, offset) = interval.to_shard_id_and_offset(
|
||||
data_shards,
|
||||
ERASURE_CODING_LARGE_BLOCK_SIZE as i64,
|
||||
ERASURE_CODING_SMALL_BLOCK_SIZE as i64,
|
||||
);
|
||||
assert_eq!(shard_id, 0);
|
||||
assert_eq!(offset, 100);
|
||||
|
||||
@@ -162,7 +175,11 @@ mod tests {
|
||||
is_large_block: true,
|
||||
large_block_rows_count: 1,
|
||||
};
|
||||
let (shard_id, _offset) = interval.to_shard_id_and_offset(data_shards);
|
||||
let (shard_id, _offset) = interval.to_shard_id_and_offset(
|
||||
data_shards,
|
||||
ERASURE_CODING_LARGE_BLOCK_SIZE as i64,
|
||||
ERASURE_CODING_SMALL_BLOCK_SIZE as i64,
|
||||
);
|
||||
assert_eq!(shard_id, 5);
|
||||
|
||||
// Block index 12 (data_shards=10) → row_index 1, shard_id 2
|
||||
@@ -173,7 +190,11 @@ mod tests {
|
||||
is_large_block: true,
|
||||
large_block_rows_count: 5,
|
||||
};
|
||||
let (shard_id, offset) = interval.to_shard_id_and_offset(data_shards);
|
||||
let (shard_id, offset) = interval.to_shard_id_and_offset(
|
||||
data_shards,
|
||||
ERASURE_CODING_LARGE_BLOCK_SIZE as i64,
|
||||
ERASURE_CODING_SMALL_BLOCK_SIZE as i64,
|
||||
);
|
||||
assert_eq!(shard_id, 2); // 12 % 10 = 2
|
||||
assert_eq!(offset, large_block_size + 200); // row 1 offset + inner_block_offset
|
||||
|
||||
@@ -185,7 +206,11 @@ mod tests {
|
||||
is_large_block: true,
|
||||
large_block_rows_count: 2,
|
||||
};
|
||||
let (shard_id, offset) = interval.to_shard_id_and_offset(data_shards);
|
||||
let (shard_id, offset) = interval.to_shard_id_and_offset(
|
||||
data_shards,
|
||||
ERASURE_CODING_LARGE_BLOCK_SIZE as i64,
|
||||
ERASURE_CODING_SMALL_BLOCK_SIZE as i64,
|
||||
);
|
||||
assert_eq!(shard_id, 0);
|
||||
assert_eq!(offset, ERASURE_CODING_LARGE_BLOCK_SIZE as i64); // row 1 offset
|
||||
}
|
||||
@@ -193,7 +218,14 @@ mod tests {
|
||||
#[test]
|
||||
fn test_locate_data_small_file() {
|
||||
// Small file: 100 bytes at offset 50, shard size = 1MB
|
||||
let intervals = locate_data(50, Size(100), 1024 * 1024, 10);
|
||||
let intervals = locate_data(
|
||||
50,
|
||||
Size(100),
|
||||
1024 * 1024,
|
||||
10,
|
||||
ERASURE_CODING_LARGE_BLOCK_SIZE as i64,
|
||||
ERASURE_CODING_SMALL_BLOCK_SIZE as i64,
|
||||
);
|
||||
assert!(!intervals.is_empty());
|
||||
|
||||
// Should be a single small block interval (no large block rows for 1MB shard)
|
||||
@@ -203,7 +235,14 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn test_locate_data_empty() {
|
||||
let intervals = locate_data(0, Size(0), 1024 * 1024, 10);
|
||||
let intervals = locate_data(
|
||||
0,
|
||||
Size(0),
|
||||
1024 * 1024,
|
||||
10,
|
||||
ERASURE_CODING_LARGE_BLOCK_SIZE as i64,
|
||||
ERASURE_CODING_SMALL_BLOCK_SIZE as i64,
|
||||
);
|
||||
assert!(intervals.is_empty());
|
||||
}
|
||||
|
||||
@@ -216,7 +255,11 @@ mod tests {
|
||||
is_large_block: false,
|
||||
large_block_rows_count: 2,
|
||||
};
|
||||
let (_shard_id, offset) = interval.to_shard_id_and_offset(10);
|
||||
let (_shard_id, offset) = interval.to_shard_id_and_offset(
|
||||
10,
|
||||
ERASURE_CODING_LARGE_BLOCK_SIZE as i64,
|
||||
ERASURE_CODING_SMALL_BLOCK_SIZE as i64,
|
||||
);
|
||||
// Should be after 2 large block rows
|
||||
assert_eq!(offset, 2 * ERASURE_CODING_LARGE_BLOCK_SIZE as i64);
|
||||
}
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -14,7 +14,10 @@ use std::path::Path;
|
||||
use std::sync::atomic::{AtomicI64, AtomicU64, Ordering};
|
||||
|
||||
mod compact_map;
|
||||
pub mod file_pool;
|
||||
pub mod sorted_file;
|
||||
use compact_map::CompactMap;
|
||||
use sorted_file::SortedFileNeedleMap;
|
||||
|
||||
use redb::{Database, Durability, ReadableDatabase, ReadableTable, TableDefinition};
|
||||
|
||||
@@ -750,10 +753,12 @@ impl RedbNeedleMap {
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Look up a needle.
|
||||
pub fn get(&self, key: NeedleId) -> Option<NeedleValue> {
|
||||
/// Look up a needle. A redb failure is an ERROR, not an absent needle:
|
||||
/// answering "not found" would turn a database problem into a read miss
|
||||
/// and let a delete report success without recording a tombstone.
|
||||
pub fn get(&self, key: NeedleId) -> io::Result<Option<NeedleValue>> {
|
||||
let key_u64: u64 = key.into();
|
||||
self.get_internal(key_u64).ok().flatten()
|
||||
self.get_internal(key_u64)
|
||||
}
|
||||
|
||||
/// Internal get that returns io::Result for error propagation.
|
||||
@@ -939,33 +944,31 @@ impl RedbNeedleMap {
|
||||
}
|
||||
|
||||
/// Collect all entries as a Vec for iteration (used by volume.rs iter patterns).
|
||||
pub fn collect_entries(&self) -> Vec<(NeedleId, NeedleValue)> {
|
||||
pub fn collect_entries(&self) -> io::Result<Vec<(NeedleId, NeedleValue)>> {
|
||||
let mut result = Vec::new();
|
||||
let txn: redb::ReadTransaction = match self.db.begin_read() {
|
||||
Ok(t) => t,
|
||||
Err(_) => return result,
|
||||
};
|
||||
let table = match txn.open_table(NEEDLE_TABLE) {
|
||||
Ok(t) => t,
|
||||
Err(_) => return result,
|
||||
};
|
||||
let iter = match table.iter() {
|
||||
Ok(i) => i,
|
||||
Err(_) => return result,
|
||||
};
|
||||
let txn: redb::ReadTransaction = self
|
||||
.db
|
||||
.begin_read()
|
||||
.map_err(|e| io::Error::other(format!("redb begin_read: {e}")))?;
|
||||
let table = txn
|
||||
.open_table(NEEDLE_TABLE)
|
||||
.map_err(|e| io::Error::other(format!("redb open_table: {e}")))?;
|
||||
let iter = table
|
||||
.iter()
|
||||
.map_err(|e| io::Error::other(format!("redb iter: {e}")))?;
|
||||
for entry in iter {
|
||||
if let Ok((key_guard, val_guard)) = entry {
|
||||
let key_u64: u64 = key_guard.value();
|
||||
let bytes: &[u8] = val_guard.value();
|
||||
if bytes.len() == PACKED_NEEDLE_VALUE_SIZE {
|
||||
let mut arr = [0u8; PACKED_NEEDLE_VALUE_SIZE];
|
||||
arr.copy_from_slice(bytes);
|
||||
let nv = unpack_needle_value(&arr);
|
||||
result.push((NeedleId(key_u64), nv));
|
||||
}
|
||||
let (key_guard, val_guard) =
|
||||
entry.map_err(|e| io::Error::other(format!("redb entry: {e}")))?;
|
||||
let key_u64: u64 = key_guard.value();
|
||||
let bytes: &[u8] = val_guard.value();
|
||||
if bytes.len() == PACKED_NEEDLE_VALUE_SIZE {
|
||||
let mut arr = [0u8; PACKED_NEEDLE_VALUE_SIZE];
|
||||
arr.copy_from_slice(bytes);
|
||||
let nv = unpack_needle_value(&arr);
|
||||
result.push((NeedleId(key_u64), nv));
|
||||
}
|
||||
}
|
||||
result
|
||||
Ok(result)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -977,6 +980,10 @@ impl RedbNeedleMap {
|
||||
pub enum NeedleMap {
|
||||
InMemory(CompactNeedleMap),
|
||||
Redb(RedbNeedleMap),
|
||||
/// Read-only volumes — including every cloud-tiered one — search the sorted
|
||||
/// `.sdx` on disk instead of holding an index in RAM. Mirrors Go's
|
||||
/// `SortedFileNeedleMap`.
|
||||
SortedFile(SortedFileNeedleMap),
|
||||
}
|
||||
|
||||
impl NeedleMap {
|
||||
@@ -985,14 +992,18 @@ impl NeedleMap {
|
||||
match self {
|
||||
NeedleMap::InMemory(nm) => nm.put(key, offset, size),
|
||||
NeedleMap::Redb(nm) => nm.put(key, offset, size),
|
||||
NeedleMap::SortedFile(nm) => nm.put(key, offset, size),
|
||||
}
|
||||
}
|
||||
|
||||
/// Look up a needle.
|
||||
pub fn get(&self, key: NeedleId) -> Option<NeedleValue> {
|
||||
/// Look up a needle. Disk- and database-backed maps report their own
|
||||
/// failures rather than folding them into "not found" — see the notes on
|
||||
/// `RedbNeedleMap::get` and `SortedFileNeedleMap::get`.
|
||||
pub fn get(&self, key: NeedleId) -> io::Result<Option<NeedleValue>> {
|
||||
match self {
|
||||
NeedleMap::InMemory(nm) => nm.get(key),
|
||||
NeedleMap::InMemory(nm) => Ok(nm.get(key)),
|
||||
NeedleMap::Redb(nm) => nm.get(key),
|
||||
NeedleMap::SortedFile(nm) => nm.get(key),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1001,6 +1012,7 @@ impl NeedleMap {
|
||||
match self {
|
||||
NeedleMap::InMemory(nm) => nm.delete(key, offset),
|
||||
NeedleMap::Redb(nm) => nm.delete(key, offset),
|
||||
NeedleMap::SortedFile(nm) => nm.delete(key, offset),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1009,6 +1021,9 @@ impl NeedleMap {
|
||||
match self {
|
||||
NeedleMap::InMemory(nm) => nm.set_idx_file(file, offset),
|
||||
NeedleMap::Redb(nm) => nm.set_idx_file(file, offset),
|
||||
// The sorted map borrows its .idx per append, so there is no
|
||||
// long-lived writer to install.
|
||||
NeedleMap::SortedFile(_) => {}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1017,6 +1032,8 @@ impl NeedleMap {
|
||||
match self {
|
||||
NeedleMap::InMemory(nm) => nm.has_idx_writer(),
|
||||
NeedleMap::Redb(nm) => nm.has_idx_writer(),
|
||||
// Appends open the .idx on demand, so one is always available.
|
||||
NeedleMap::SortedFile(_) => true,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1025,6 +1042,7 @@ impl NeedleMap {
|
||||
match self {
|
||||
NeedleMap::InMemory(nm) => nm.content_size(),
|
||||
NeedleMap::Redb(nm) => nm.content_size(),
|
||||
NeedleMap::SortedFile(nm) => nm.content_size(),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1033,6 +1051,7 @@ impl NeedleMap {
|
||||
match self {
|
||||
NeedleMap::InMemory(nm) => nm.deleted_size(),
|
||||
NeedleMap::Redb(nm) => nm.deleted_size(),
|
||||
NeedleMap::SortedFile(nm) => nm.deleted_size(),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1041,6 +1060,7 @@ impl NeedleMap {
|
||||
match self {
|
||||
NeedleMap::InMemory(nm) => nm.file_count(),
|
||||
NeedleMap::Redb(nm) => nm.file_count(),
|
||||
NeedleMap::SortedFile(nm) => nm.file_count(),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1049,6 +1069,7 @@ impl NeedleMap {
|
||||
match self {
|
||||
NeedleMap::InMemory(nm) => nm.deleted_count(),
|
||||
NeedleMap::Redb(nm) => nm.deleted_count(),
|
||||
NeedleMap::SortedFile(nm) => nm.deleted_count(),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1057,6 +1078,7 @@ impl NeedleMap {
|
||||
match self {
|
||||
NeedleMap::InMemory(nm) => nm.max_file_key(),
|
||||
NeedleMap::Redb(nm) => nm.max_file_key(),
|
||||
NeedleMap::SortedFile(nm) => nm.max_file_key(),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1067,6 +1089,7 @@ impl NeedleMap {
|
||||
match self {
|
||||
NeedleMap::InMemory(nm) => nm.max_needle_end(),
|
||||
NeedleMap::Redb(nm) => nm.max_needle_end(),
|
||||
NeedleMap::SortedFile(nm) => nm.max_needle_end(),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1075,6 +1098,7 @@ impl NeedleMap {
|
||||
match self {
|
||||
NeedleMap::InMemory(nm) => nm.index_file_size(),
|
||||
NeedleMap::Redb(nm) => nm.index_file_size(),
|
||||
NeedleMap::SortedFile(nm) => nm.index_file_size(),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1083,6 +1107,7 @@ impl NeedleMap {
|
||||
match self {
|
||||
NeedleMap::InMemory(nm) => nm.sync(),
|
||||
NeedleMap::Redb(nm) => nm.sync(),
|
||||
NeedleMap::SortedFile(nm) => nm.sync(),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1091,6 +1116,7 @@ impl NeedleMap {
|
||||
match self {
|
||||
NeedleMap::InMemory(nm) => nm.close(),
|
||||
NeedleMap::Redb(nm) => nm.close(),
|
||||
NeedleMap::SortedFile(nm) => nm.close(),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1099,6 +1125,7 @@ impl NeedleMap {
|
||||
match self {
|
||||
NeedleMap::InMemory(nm) => nm.save_to_idx(path),
|
||||
NeedleMap::Redb(nm) => nm.save_to_idx(path),
|
||||
NeedleMap::SortedFile(nm) => nm.save_to_idx(path),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1110,22 +1137,28 @@ impl NeedleMap {
|
||||
match self {
|
||||
NeedleMap::InMemory(nm) => nm.ascending_visit(f),
|
||||
NeedleMap::Redb(nm) => nm.ascending_visit(f),
|
||||
NeedleMap::SortedFile(nm) => nm.ascending_visit(f),
|
||||
}
|
||||
}
|
||||
|
||||
/// Iterate all entries. Returns a Vec of (NeedleId, NeedleValue) pairs.
|
||||
/// For InMemory this collects via ascending visit; for Redb it reads from disk.
|
||||
pub fn iter_entries(&self) -> Vec<(NeedleId, NeedleValue)> {
|
||||
/// For InMemory this collects via ascending visit; the disk-backed maps read
|
||||
/// it back off disk, so a truncated .sdx or a redb read fault surfaces here
|
||||
/// as an error. Compaction treats the result as the complete live set, so a
|
||||
/// partial scan must never be mistaken for an empty tail.
|
||||
pub fn iter_entries(&self) -> io::Result<Vec<(NeedleId, NeedleValue)>> {
|
||||
match self {
|
||||
NeedleMap::InMemory(nm) => {
|
||||
let mut entries = Vec::new();
|
||||
// The visitor never fails, so neither can this.
|
||||
let _ = nm.ascending_visit(|id, nv| {
|
||||
entries.push((id, *nv));
|
||||
Ok(())
|
||||
});
|
||||
entries
|
||||
Ok(entries)
|
||||
}
|
||||
NeedleMap::Redb(nm) => nm.collect_entries(),
|
||||
NeedleMap::SortedFile(nm) => nm.iter_entries(),
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1280,13 +1313,13 @@ mod tests {
|
||||
nm.put(NeedleId(2), Offset::from_actual_offset(128), Size(200))
|
||||
.unwrap();
|
||||
|
||||
let v1 = nm.get(NeedleId(1)).unwrap();
|
||||
let v1 = nm.get(NeedleId(1)).unwrap().unwrap();
|
||||
assert_eq!(v1.size, Size(100));
|
||||
|
||||
let v2 = nm.get(NeedleId(2)).unwrap();
|
||||
let v2 = nm.get(NeedleId(2)).unwrap().unwrap();
|
||||
assert_eq!(v2.size, Size(200));
|
||||
|
||||
assert!(nm.get(NeedleId(99)).is_none());
|
||||
assert!(nm.get(NeedleId(99)).unwrap().is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -1311,7 +1344,7 @@ mod tests {
|
||||
assert_eq!(nm.deleted_size(), 100);
|
||||
|
||||
// Deleted entry should have negated size
|
||||
let nv = nm.get(NeedleId(1)).unwrap();
|
||||
let nv = nm.get(NeedleId(1)).unwrap().unwrap();
|
||||
assert_eq!(nv.size, Size(-100));
|
||||
}
|
||||
|
||||
@@ -1384,9 +1417,9 @@ mod tests {
|
||||
let mut cursor = Cursor::new(idx_data);
|
||||
let nm = RedbNeedleMap::load_from_idx(db_path.to_str().unwrap(), &mut cursor, Version::current()).unwrap();
|
||||
|
||||
assert!(nm.get(NeedleId(1)).is_some());
|
||||
assert!(nm.get(NeedleId(2)).is_none()); // deleted and removed
|
||||
assert!(nm.get(NeedleId(3)).is_some());
|
||||
assert!(nm.get(NeedleId(1)).unwrap().is_some());
|
||||
assert!(nm.get(NeedleId(2)).unwrap().is_none()); // deleted and removed
|
||||
assert!(nm.get(NeedleId(3)).unwrap().is_some());
|
||||
assert_eq!(nm.file_count(), 2);
|
||||
}
|
||||
|
||||
@@ -1503,7 +1536,7 @@ mod tests {
|
||||
let mut nm = NeedleMap::InMemory(CompactNeedleMap::new());
|
||||
nm.put(NeedleId(1), Offset::from_actual_offset(0), Size(100))
|
||||
.unwrap();
|
||||
assert_eq!(nm.get(NeedleId(1)).unwrap().size, Size(100));
|
||||
assert_eq!(nm.get(NeedleId(1)).unwrap().unwrap().size, Size(100));
|
||||
assert_eq!(nm.file_count(), 1);
|
||||
}
|
||||
|
||||
@@ -1514,7 +1547,7 @@ mod tests {
|
||||
let mut nm = NeedleMap::Redb(RedbNeedleMap::new(db_path.to_str().unwrap()).unwrap());
|
||||
nm.put(NeedleId(1), Offset::from_actual_offset(0), Size(100))
|
||||
.unwrap();
|
||||
assert_eq!(nm.get(NeedleId(1)).unwrap().size, Size(100));
|
||||
assert_eq!(nm.get(NeedleId(1)).unwrap().unwrap().size, Size(100));
|
||||
assert_eq!(nm.file_count(), 1);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,247 @@
|
||||
//! Bounded pool of open index-file descriptors.
|
||||
//!
|
||||
//! Read-only volumes — cloud-tiered ones above all — outnumber writable ones by
|
||||
//! orders of magnitude on a large server, and a volume that pins its `.idx` and
|
||||
//! `.sdx` for the life of the process costs two descriptors whether or not
|
||||
//! anybody reads it. At ~600K volumes per server that alone exhausts any fd
|
||||
//! limit. Neither file is needed except while a lookup is in flight, so
|
||||
//! [`SortedFileNeedleMap`](super::sorted_file::SortedFileNeedleMap) borrows them
|
||||
//! from this pool: an idle volume holds nothing, a busy one keeps its handles
|
||||
//! hot rather than paying an `open()` per needle.
|
||||
//!
|
||||
//! Mirrors Go's `weed/storage/needle_map_file_pool.go`. Handles are handed out
|
||||
//! as `Arc<File>`, so an eviction cannot close a descriptor a reader still
|
||||
//! holds — the file closes when the last borrower drops its `Arc`.
|
||||
|
||||
use std::collections::{BTreeMap, HashMap};
|
||||
use std::fs::{File, OpenOptions};
|
||||
use std::io;
|
||||
use std::sync::{Arc, Mutex, OnceLock};
|
||||
|
||||
/// Descriptors the pool keeps open. Matches Go's `maxPooledIndexFiles`.
|
||||
pub const MAX_POOLED_INDEX_FILES: usize = 1024;
|
||||
|
||||
struct Entry {
|
||||
file: Arc<File>,
|
||||
tick: u64,
|
||||
}
|
||||
|
||||
#[derive(Default)]
|
||||
struct Inner {
|
||||
entries: HashMap<String, Entry>,
|
||||
/// Recency order, oldest tick first, so eviction is a `pop_first`.
|
||||
order: BTreeMap<u64, String>,
|
||||
next_tick: u64,
|
||||
}
|
||||
|
||||
pub struct IndexFilePool {
|
||||
capacity: usize,
|
||||
inner: Mutex<Inner>,
|
||||
}
|
||||
|
||||
/// Writable and read-only handles for the same path are pooled separately so a
|
||||
/// read never depends on the file being openable for write — a volume served
|
||||
/// off a read-only mount still answers lookups.
|
||||
fn pool_key(path: &str, writable: bool) -> String {
|
||||
if writable {
|
||||
format!("{path}\0rw")
|
||||
} else {
|
||||
path.to_string()
|
||||
}
|
||||
}
|
||||
|
||||
impl IndexFilePool {
|
||||
pub fn new(capacity: usize) -> Self {
|
||||
IndexFilePool {
|
||||
capacity: capacity.max(1),
|
||||
inner: Mutex::new(Inner::default()),
|
||||
}
|
||||
}
|
||||
|
||||
/// Hand out an open handle for `path`, reusing the pooled one when there is
|
||||
/// one. The descriptor lives as long as the returned `Arc`.
|
||||
pub fn borrow(&self, path: &str, writable: bool) -> io::Result<Arc<File>> {
|
||||
let key = pool_key(path, writable);
|
||||
if let Some(file) = self.touch(&key) {
|
||||
return Ok(file);
|
||||
}
|
||||
|
||||
// Opened outside the lock: a cold open blocks on disk, and holding a
|
||||
// process-wide mutex across it would serialize every volume's lookups.
|
||||
let file = Arc::new(OpenOptions::new().read(true).write(writable).open(path)?);
|
||||
Ok(self.insert(key, file))
|
||||
}
|
||||
|
||||
/// Forget the pooled handles for `path`, so a later rename or delete of that
|
||||
/// path cannot be served from a descriptor on the old inode.
|
||||
pub fn discard(&self, path: &str) {
|
||||
let mut inner = self.inner.lock().unwrap();
|
||||
for key in [pool_key(path, false), pool_key(path, true)] {
|
||||
if let Some(entry) = inner.entries.remove(&key) {
|
||||
inner.order.remove(&entry.tick);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Descriptors currently pooled. Test-only visibility into the bound.
|
||||
#[cfg(test)]
|
||||
pub fn pooled_count(&self) -> usize {
|
||||
self.inner.lock().unwrap().entries.len()
|
||||
}
|
||||
|
||||
fn touch(&self, key: &str) -> Option<Arc<File>> {
|
||||
let mut inner = self.inner.lock().unwrap();
|
||||
let tick = inner.next_tick;
|
||||
let entry = inner.entries.get_mut(key)?;
|
||||
let file = entry.file.clone();
|
||||
let old_tick = std::mem::replace(&mut entry.tick, tick);
|
||||
inner.order.remove(&old_tick);
|
||||
inner.order.insert(tick, key.to_string());
|
||||
inner.next_tick += 1;
|
||||
Some(file)
|
||||
}
|
||||
|
||||
fn insert(&self, key: String, file: Arc<File>) -> Arc<File> {
|
||||
let mut inner = self.inner.lock().unwrap();
|
||||
if let Some(entry) = inner.entries.get(&key) {
|
||||
// Another borrower opened the same path first; keep one descriptor.
|
||||
return entry.file.clone();
|
||||
}
|
||||
let tick = inner.next_tick;
|
||||
inner.next_tick += 1;
|
||||
inner.order.insert(tick, key.clone());
|
||||
inner.entries.insert(
|
||||
key,
|
||||
Entry {
|
||||
file: file.clone(),
|
||||
tick,
|
||||
},
|
||||
);
|
||||
while inner.entries.len() > self.capacity {
|
||||
let Some((_, oldest)) = inner.order.pop_first() else {
|
||||
break;
|
||||
};
|
||||
inner.entries.remove(&oldest);
|
||||
}
|
||||
file
|
||||
}
|
||||
}
|
||||
|
||||
/// Process-wide pool shared by every read-only volume on this server.
|
||||
pub fn pooled_index_files() -> &'static IndexFilePool {
|
||||
static POOL: OnceLock<IndexFilePool> = OnceLock::new();
|
||||
POOL.get_or_init(|| IndexFilePool::new(MAX_POOLED_INDEX_FILES))
|
||||
}
|
||||
|
||||
/// Descriptors this process holds on `.idx`/`.sdx` files under `dir`, read from
|
||||
/// `/proc/self/fd` where it exists and from `lsof` otherwise. `None` when
|
||||
/// neither is available, so a caller can skip rather than assert vacuously.
|
||||
#[cfg(test)]
|
||||
pub(crate) fn open_index_fds(dir: &std::path::Path) -> Option<usize> {
|
||||
let prefix = std::fs::canonicalize(dir).unwrap_or_else(|_| dir.to_path_buf());
|
||||
let is_index = |target: &std::path::Path| {
|
||||
target.starts_with(&prefix)
|
||||
&& matches!(
|
||||
target.extension().and_then(|e| e.to_str()),
|
||||
Some("idx") | Some("sdx")
|
||||
)
|
||||
};
|
||||
|
||||
if let Ok(entries) = std::fs::read_dir("/proc/self/fd") {
|
||||
return Some(
|
||||
entries
|
||||
.filter_map(|e| std::fs::read_link(e.ok()?.path()).ok())
|
||||
.filter(|target| is_index(target))
|
||||
.count(),
|
||||
);
|
||||
}
|
||||
|
||||
let out = std::process::Command::new("lsof")
|
||||
.args(["-p", &std::process::id().to_string(), "-F", "n"])
|
||||
.output()
|
||||
.ok()?;
|
||||
Some(
|
||||
String::from_utf8_lossy(&out.stdout)
|
||||
.lines()
|
||||
.filter_map(|line| line.strip_prefix('n'))
|
||||
.filter(|line| is_index(std::path::Path::new(line)))
|
||||
.count(),
|
||||
)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use std::io::Write;
|
||||
|
||||
fn write_file(dir: &std::path::Path, name: &str, contents: &[u8]) -> String {
|
||||
let path = dir.join(name);
|
||||
let mut f = File::create(&path).unwrap();
|
||||
f.write_all(contents).unwrap();
|
||||
path.to_str().unwrap().to_string()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn evicted_handle_stays_usable_for_its_borrower() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let first = write_file(dir.path(), "first", b"first");
|
||||
let second = write_file(dir.path(), "second", b"second");
|
||||
|
||||
let pool = IndexFilePool::new(1);
|
||||
let borrowed = pool.borrow(&first, false).unwrap();
|
||||
|
||||
// Pushes the single slot over, evicting the entry still in use.
|
||||
let _other = pool.borrow(&second, false).unwrap();
|
||||
assert_eq!(pool.pooled_count(), 1);
|
||||
|
||||
let mut buf = [0u8; 5];
|
||||
#[cfg(unix)]
|
||||
{
|
||||
use std::os::unix::fs::FileExt;
|
||||
borrowed.read_exact_at(&mut buf, 0).unwrap();
|
||||
}
|
||||
assert_eq!(&buf, b"first");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn borrow_reuses_the_pooled_handle() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let path = write_file(dir.path(), "idx", b"x");
|
||||
|
||||
let pool = IndexFilePool::new(4);
|
||||
let a = pool.borrow(&path, false).unwrap();
|
||||
let b = pool.borrow(&path, false).unwrap();
|
||||
assert!(Arc::ptr_eq(&a, &b));
|
||||
assert_eq!(pool.pooled_count(), 1);
|
||||
|
||||
// A writable handle is pooled separately from the read-only one.
|
||||
let w = pool.borrow(&path, true).unwrap();
|
||||
assert!(!Arc::ptr_eq(&a, &w));
|
||||
assert_eq!(pool.pooled_count(), 2);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn discard_drops_both_handles() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let path = write_file(dir.path(), "idx", b"x");
|
||||
|
||||
let pool = IndexFilePool::new(4);
|
||||
let _r = pool.borrow(&path, false).unwrap();
|
||||
let _w = pool.borrow(&path, true).unwrap();
|
||||
assert_eq!(pool.pooled_count(), 2);
|
||||
|
||||
pool.discard(&path);
|
||||
assert_eq!(pool.pooled_count(), 0);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn pool_stays_within_capacity() {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let pool = IndexFilePool::new(3);
|
||||
for i in 0..10 {
|
||||
let path = write_file(dir.path(), &format!("f{i}"), b"x");
|
||||
let _ = pool.borrow(&path, false).unwrap();
|
||||
}
|
||||
assert_eq!(pool.pooled_count(), 3);
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -259,11 +259,25 @@ impl Store {
|
||||
/// Returns the index of the disk that should receive a new EC
|
||||
/// shard / index file for `(collection, vid)`. Selection order:
|
||||
///
|
||||
/// 0. a disk that already owns one of `shard_ids` (in-memory claim),
|
||||
/// 1. a disk that already has the EC volume mounted (in-memory state),
|
||||
/// 2. a disk that owns the `.ecx` file on disk (volume not yet mounted),
|
||||
/// 3. any HDD with free space,
|
||||
/// 4. any disk with free space.
|
||||
///
|
||||
/// Step 0 keeps the per-server invariant that a shard id is owned by
|
||||
/// at most one disk. Steps 1-4 only know the volume, and a multi-disk
|
||||
/// server legitimately mounts the same vid on several disks, so the
|
||||
/// free-count tie-break alone can send a re-copy of a shard the server
|
||||
/// already holds (a retried `ec.balance` / `ec.rebuild` move) to a
|
||||
/// sibling disk. Both disks then claim the same (vid, shard) and
|
||||
/// report it to the master from two disk ids, and which claimant
|
||||
/// serves reads or survives a later unmount/delete of the shard id
|
||||
/// becomes an accident of location order. Overwriting in place is what
|
||||
/// the caller meant, so an owning disk wins ahead of the space filters
|
||||
/// too — a re-copy needs no new shard slot, and a genuinely full disk
|
||||
/// fails the write rather than silently splitting the claim.
|
||||
///
|
||||
/// Step 2 is the missing primitive that pinned subsequent shards to
|
||||
/// the first-shard disk during `ec.rebuild`: rebuild only sets
|
||||
/// `CopyEcxFile=true` on the first shard, then relies on auto-select
|
||||
@@ -286,20 +300,26 @@ impl Store {
|
||||
collection: &str,
|
||||
vid: VolumeId,
|
||||
data_shard_count: u32,
|
||||
shard_ids: &[u32],
|
||||
) -> Option<usize> {
|
||||
const TIER_ANY_DISK: u8 = 1;
|
||||
const TIER_HDD: u8 = 2;
|
||||
const TIER_ECX_ON_DISK: u8 = 3;
|
||||
const TIER_MOUNTED: u8 = 4;
|
||||
const TIER_OWNS_SHARD: u8 = 5;
|
||||
|
||||
let mut best: Option<(usize, u8, i64)> = None;
|
||||
// (index, tier, owned shard count, free shard slots)
|
||||
let mut best: Option<(usize, u8, usize, i64)> = None;
|
||||
for (i, loc) in self.locations.iter().enumerate() {
|
||||
if loc.is_disk_space_low.load(Ordering::Relaxed) {
|
||||
continue;
|
||||
}
|
||||
let owned = owned_ec_shard_count(loc, vid, shard_ids);
|
||||
let free = ec_free_shard_count(loc, data_shard_count);
|
||||
if free <= 0 {
|
||||
continue;
|
||||
if owned == 0 {
|
||||
if loc.is_disk_space_low.load(Ordering::Relaxed) {
|
||||
continue;
|
||||
}
|
||||
if free <= 0 {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
let mut tier = TIER_ANY_DISK;
|
||||
if loc.disk_type == DiskType::HardDrive {
|
||||
@@ -311,15 +331,41 @@ impl Store {
|
||||
if loc.has_ec_volume(vid) {
|
||||
tier = TIER_MOUNTED;
|
||||
}
|
||||
if owned > 0 {
|
||||
tier = TIER_OWNS_SHARD;
|
||||
}
|
||||
let better = match best {
|
||||
None => true,
|
||||
Some((_, b_tier, b_free)) => tier > b_tier || (tier == b_tier && free > b_free),
|
||||
// owned only separates disks inside TIER_OWNS_SHARD; it is 0
|
||||
// everywhere else, so this falls through to the free-count
|
||||
// tie-break for the other tiers.
|
||||
Some((_, b_tier, b_owned, b_free)) => {
|
||||
tier > b_tier
|
||||
|| (tier == b_tier
|
||||
&& (owned > b_owned || (owned == b_owned && free > b_free)))
|
||||
}
|
||||
};
|
||||
if better {
|
||||
best = Some((i, tier, free));
|
||||
best = Some((i, tier, owned, free));
|
||||
}
|
||||
}
|
||||
best.map(|(i, _, _)| i)
|
||||
best.map(|(i, _, _, _)| i)
|
||||
}
|
||||
|
||||
/// Returns the distinct disk indexes that already own one of `shard_ids`
|
||||
/// for `vid`, in location order. More than one owner means the batch has
|
||||
/// no single correct destination — whichever disk receives it would
|
||||
/// duplicate a sibling disk's claim — so batch callers must split by
|
||||
/// owner (`volume_ec_shards_copy` refuses such a batch instead of
|
||||
/// guessing). Mirrors `Store.EcShardOwnerDisks` in
|
||||
/// `weed/storage/store_ec.go`.
|
||||
pub fn ec_shard_owner_disks(&self, vid: VolumeId, shard_ids: &[u32]) -> Vec<usize> {
|
||||
self.locations
|
||||
.iter()
|
||||
.enumerate()
|
||||
.filter(|(_, loc)| owned_ec_shard_count(loc, vid, shard_ids) > 0)
|
||||
.map(|(i, _)| i)
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Create a new volume, placing it on the location with the most free space.
|
||||
@@ -804,6 +850,37 @@ impl Store {
|
||||
}
|
||||
|
||||
/// Find an EC volume across all locations (mutable).
|
||||
/// Every per-disk `EcVolume` this store maps for `vid`. A vid can mount on
|
||||
/// N disks as N distinct runtimes, and the first-match `find_ec_volume_mut`
|
||||
/// hides the siblings — so anything that has to reach the whole volume,
|
||||
/// rather than any one runtime of it, iterates this instead.
|
||||
/// Every directory on this server that could hold an EC volume's metadata —
|
||||
/// each disk's data and index directory. Startup mirroring gives each
|
||||
/// shard-bearing disk its own .ecx/.ecj/.vif, but the checksum sidecar is
|
||||
/// not mirrored and a repair delivers exactly one copy, so a runtime looking
|
||||
/// only at its own two directories cannot see it. Handing this list to the
|
||||
/// sidecar resolution keeps one authoritative copy reachable from every
|
||||
/// runtime rather than duplicating a file that is rewritten as shards are
|
||||
/// repaired and generations published.
|
||||
pub fn ec_metadata_dirs(&self) -> Vec<String> {
|
||||
let mut dirs: Vec<String> = Vec::with_capacity(self.locations.len() * 2);
|
||||
for loc in &self.locations {
|
||||
for dir in [&loc.directory, &loc.idx_directory] {
|
||||
if !dir.is_empty() && !dirs.iter().any(|d| d == dir) {
|
||||
dirs.push(dir.clone());
|
||||
}
|
||||
}
|
||||
}
|
||||
dirs
|
||||
}
|
||||
|
||||
pub fn find_all_ec_volumes_mut(&mut self, vid: VolumeId) -> Vec<&mut EcVolume> {
|
||||
self.locations
|
||||
.iter_mut()
|
||||
.filter_map(|loc| loc.find_ec_volume_mut(vid))
|
||||
.collect()
|
||||
}
|
||||
|
||||
pub fn find_ec_volume_mut(&mut self, vid: VolumeId) -> Option<&mut EcVolume> {
|
||||
for loc in &mut self.locations {
|
||||
if let Some(ecv) = loc.find_ec_volume_mut(vid) {
|
||||
@@ -1302,6 +1379,20 @@ fn ec_free_shard_count(loc: &DiskLocation, data_shard_count: u32) -> i64 {
|
||||
free
|
||||
}
|
||||
|
||||
/// Reports how many of `shard_ids` this disk already claims for `vid`, per
|
||||
/// the in-memory registration the read path and heartbeats use.
|
||||
///
|
||||
/// Mirrors `ownedEcShardCount` in `weed/storage/store_ec.go`.
|
||||
fn owned_ec_shard_count(loc: &DiskLocation, vid: VolumeId, shard_ids: &[u32]) -> usize {
|
||||
let Some(ecv) = loc.find_ec_volume(vid) else {
|
||||
return 0;
|
||||
};
|
||||
shard_ids
|
||||
.iter()
|
||||
.filter(|&&shard_id| ecv.has_shard(shard_id as u8))
|
||||
.count()
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// Tests
|
||||
// ============================================================================
|
||||
@@ -1840,7 +1931,7 @@ mod tests {
|
||||
let base = volume_file_name(&store.locations[2].idx_directory, collection, vid);
|
||||
std::fs::write(format!("{}.ecx", base), vec![0u8; 20]).unwrap();
|
||||
|
||||
let got = store.find_ec_shard_target_location(collection, vid, 10);
|
||||
let got = store.find_ec_shard_target_location(collection, vid, 10, &[]);
|
||||
assert_eq!(
|
||||
got,
|
||||
Some(2),
|
||||
@@ -1982,7 +2073,7 @@ mod tests {
|
||||
let base = volume_file_name(&store.locations[2].idx_directory, collection, vid);
|
||||
std::fs::write(format!("{}.ecx", base), vec![0u8; 20]).unwrap();
|
||||
|
||||
let got = store.find_ec_shard_target_location(collection, vid, 10);
|
||||
let got = store.find_ec_shard_target_location(collection, vid, 10, &[]);
|
||||
assert_eq!(got, Some(1), "expected the mounted disk to win; got {:?}", got);
|
||||
}
|
||||
|
||||
@@ -1991,7 +2082,7 @@ mod tests {
|
||||
#[test]
|
||||
fn test_find_ec_shard_target_location_falls_through_to_hdd_when_nothing_matches() {
|
||||
let (store, _tmp) = make_ec_target_test_store(2);
|
||||
let got = store.find_ec_shard_target_location("grafana-loki", VolumeId(3333), 10);
|
||||
let got = store.find_ec_shard_target_location("grafana-loki", VolumeId(3333), 10, &[]);
|
||||
assert!(got.is_some(), "expected an HDD fallback");
|
||||
assert_eq!(store.locations[got.unwrap()].disk_type, DiskType::HardDrive);
|
||||
}
|
||||
@@ -2007,7 +2098,7 @@ mod tests {
|
||||
.max_volume_count
|
||||
.store(0, Ordering::Relaxed);
|
||||
|
||||
let got = store.find_ec_shard_target_location("grafana-loki", VolumeId(4444), 10);
|
||||
let got = store.find_ec_shard_target_location("grafana-loki", VolumeId(4444), 10, &[]);
|
||||
assert_eq!(
|
||||
got,
|
||||
Some(0),
|
||||
@@ -2045,7 +2136,7 @@ mod tests {
|
||||
.mount_ec_shards(vid, collection, &[0], "")
|
||||
.unwrap();
|
||||
|
||||
let got = store.find_ec_shard_target_location(collection, vid, 10);
|
||||
let got = store.find_ec_shard_target_location(collection, vid, 10, &[]);
|
||||
assert_eq!(
|
||||
got,
|
||||
Some(1),
|
||||
@@ -2053,4 +2144,130 @@ mod tests {
|
||||
got,
|
||||
);
|
||||
}
|
||||
|
||||
/// Per-server invariant: a shard id is owned by at most one disk.
|
||||
///
|
||||
/// A multi-disk server legitimately mounts one vid on several disks, each
|
||||
/// holding a disjoint subset of the shards, so the mounted tier ties and
|
||||
/// the free-count tie-break decides — and it points at whichever disk
|
||||
/// happens to be emptier, not at the disk that already has this shard. A
|
||||
/// re-copy of a shard the server already holds (a retried `ec.balance` /
|
||||
/// `ec.rebuild` move) then lands a second copy on the sibling disk, and
|
||||
/// both disks register the same shard id.
|
||||
#[test]
|
||||
fn test_find_ec_shard_target_location_pins_to_the_disk_owning_the_shard() {
|
||||
let (mut store, _tmp) = make_ec_target_test_store(2);
|
||||
let collection = "grafana-loki";
|
||||
let vid = VolumeId(8888);
|
||||
|
||||
// Disk 0 owns shards 0 and 1, disk 1 owns shard 2 — so disk 1 is the
|
||||
// emptier of the two and wins the free-count tie-break.
|
||||
let base0 = volume_file_name(&store.locations[0].directory, collection, vid);
|
||||
std::fs::write(format!("{}.ec00", base0), b"x").unwrap();
|
||||
std::fs::write(format!("{}.ec01", base0), b"x").unwrap();
|
||||
store.locations[0]
|
||||
.mount_ec_shards(vid, collection, &[0, 1], "")
|
||||
.unwrap();
|
||||
|
||||
let base1 = volume_file_name(&store.locations[1].directory, collection, vid);
|
||||
std::fs::write(format!("{}.ec02", base1), b"x").unwrap();
|
||||
store.locations[1]
|
||||
.mount_ec_shards(vid, collection, &[2], "")
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(
|
||||
store.find_ec_shard_target_location(collection, vid, 10, &[0]),
|
||||
Some(0),
|
||||
"a copy of shard 0 left its owning disk",
|
||||
);
|
||||
assert_eq!(
|
||||
store.find_ec_shard_target_location(collection, vid, 10, &[2]),
|
||||
Some(1),
|
||||
"a copy of shard 2 left its owning disk",
|
||||
);
|
||||
// A shard no disk owns yet is placed by the unchanged waterfall: both
|
||||
// disks have it mounted, so the emptier one wins.
|
||||
assert_eq!(
|
||||
store.find_ec_shard_target_location(collection, vid, 10, &[7]),
|
||||
Some(1),
|
||||
"placement of an unclaimed shard changed",
|
||||
);
|
||||
}
|
||||
|
||||
/// The second half of the invariant: a disk with no free shard slots
|
||||
/// still wins for a shard it already owns. Re-copying that shard
|
||||
/// overwrites bytes the disk is already accounted for, while routing to a
|
||||
/// sibling splits the claim across two disks.
|
||||
#[test]
|
||||
fn test_find_ec_shard_target_location_owning_disk_wins_when_full() {
|
||||
let (mut store, _tmp) = make_ec_target_test_store(2);
|
||||
store.locations[0]
|
||||
.max_volume_count
|
||||
.store(1, Ordering::Relaxed);
|
||||
|
||||
let collection = "grafana-loki";
|
||||
let vid = VolumeId(9999);
|
||||
|
||||
let base = volume_file_name(&store.locations[0].directory, collection, vid);
|
||||
std::fs::write(format!("{}.ec00", base), b"x").unwrap();
|
||||
store.locations[0]
|
||||
.mount_ec_shards(vid, collection, &[0], "")
|
||||
.unwrap();
|
||||
|
||||
// Fill disk 0 past its shard-slot budget so ec_free_shard_count is 0.
|
||||
let filler = VolumeId(10000);
|
||||
let filler_base = volume_file_name(&store.locations[0].directory, collection, filler);
|
||||
let filler_shards: Vec<u32> = (0..10).collect();
|
||||
for shard_id in &filler_shards {
|
||||
std::fs::write(format!("{}.ec{:02}", filler_base, shard_id), b"x").unwrap();
|
||||
}
|
||||
store.locations[0]
|
||||
.mount_ec_shards(filler, collection, &filler_shards, "")
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(
|
||||
store.find_ec_shard_target_location(collection, vid, 10, &[0]),
|
||||
Some(0),
|
||||
"a copy of shard 0 left its owning disk when the disk was full",
|
||||
);
|
||||
// A shard it does not own still respects the space filter.
|
||||
assert_eq!(
|
||||
store.find_ec_shard_target_location(collection, vid, 10, &[7]),
|
||||
Some(1),
|
||||
"an unclaimed shard should go to the disk with free slots",
|
||||
);
|
||||
}
|
||||
|
||||
/// Mixed-owner batch contract: a batch whose requested shards are
|
||||
/// already owned by different disks reports every owner, so
|
||||
/// `volume_ec_shards_copy` can refuse it rather than rank the owners
|
||||
/// into one destination and duplicate the loser's claim.
|
||||
#[test]
|
||||
fn test_ec_shard_owner_disks() {
|
||||
let (mut store, _tmp) = make_ec_target_test_store(3);
|
||||
let collection = "grafana-loki";
|
||||
let vid = VolumeId(11111);
|
||||
|
||||
let base0 = volume_file_name(&store.locations[0].directory, collection, vid);
|
||||
std::fs::write(format!("{}.ec00", base0), b"x").unwrap();
|
||||
std::fs::write(format!("{}.ec01", base0), b"x").unwrap();
|
||||
store.locations[0]
|
||||
.mount_ec_shards(vid, collection, &[0, 1], "")
|
||||
.unwrap();
|
||||
let base1 = volume_file_name(&store.locations[1].directory, collection, vid);
|
||||
std::fs::write(format!("{}.ec02", base1), b"x").unwrap();
|
||||
store.locations[1]
|
||||
.mount_ec_shards(vid, collection, &[2], "")
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(
|
||||
store.ec_shard_owner_disks(vid, &[0, 2]),
|
||||
vec![0, 1],
|
||||
"a batch owned by two disks must report both owners",
|
||||
);
|
||||
// A batch on one disk, with or without unowned extras, has one owner.
|
||||
assert_eq!(store.ec_shard_owner_disks(vid, &[0, 1, 7]), vec![0]);
|
||||
// A wholly unowned batch reports none — fresh placement stays allowed.
|
||||
assert!(store.ec_shard_owner_disks(vid, &[7, 8]).is_empty());
|
||||
}
|
||||
}
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
Generated
+65
@@ -4666,6 +4666,70 @@ dependencies = [
|
||||
"prost 0.14.4",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "protoc-bin-vendored"
|
||||
version = "3.2.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "d1c381df33c98266b5f08186583660090a4ffa0889e76c7e9a5e175f645a67fa"
|
||||
dependencies = [
|
||||
"protoc-bin-vendored-linux-aarch_64",
|
||||
"protoc-bin-vendored-linux-ppcle_64",
|
||||
"protoc-bin-vendored-linux-s390_64",
|
||||
"protoc-bin-vendored-linux-x86_32",
|
||||
"protoc-bin-vendored-linux-x86_64",
|
||||
"protoc-bin-vendored-macos-aarch_64",
|
||||
"protoc-bin-vendored-macos-x86_64",
|
||||
"protoc-bin-vendored-win32",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "protoc-bin-vendored-linux-aarch_64"
|
||||
version = "3.2.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "c350df4d49b5b9e3ca79f7e646fde2377b199e13cfa87320308397e1f37e1a4c"
|
||||
|
||||
[[package]]
|
||||
name = "protoc-bin-vendored-linux-ppcle_64"
|
||||
version = "3.2.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "a55a63e6c7244f19b5c6393f025017eb5d793fd5467823a099740a7a4222440c"
|
||||
|
||||
[[package]]
|
||||
name = "protoc-bin-vendored-linux-s390_64"
|
||||
version = "3.2.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "1dba5565db4288e935d5330a07c264a4ee8e4a5b4a4e6f4e83fad824cc32f3b0"
|
||||
|
||||
[[package]]
|
||||
name = "protoc-bin-vendored-linux-x86_32"
|
||||
version = "3.2.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "8854774b24ee28b7868cd71dccaae8e02a2365e67a4a87a6cd11ee6cdbdf9cf5"
|
||||
|
||||
[[package]]
|
||||
name = "protoc-bin-vendored-linux-x86_64"
|
||||
version = "3.2.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b38b07546580df720fa464ce124c4b03630a6fb83e05c336fea2a241df7e5d78"
|
||||
|
||||
[[package]]
|
||||
name = "protoc-bin-vendored-macos-aarch_64"
|
||||
version = "3.2.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "89278a9926ce312e51f1d999fee8825d324d603213344a9a706daa009f1d8092"
|
||||
|
||||
[[package]]
|
||||
name = "protoc-bin-vendored-macos-x86_64"
|
||||
version = "3.2.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "81745feda7ccfb9471d7a4de888f0652e806d5795b61480605d4943176299756"
|
||||
|
||||
[[package]]
|
||||
name = "protoc-bin-vendored-win32"
|
||||
version = "3.2.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "95067976aca6421a523e491fce939a3e65249bac4b977adee0ee9771568e8aa3"
|
||||
|
||||
[[package]]
|
||||
name = "quick-xml"
|
||||
version = "0.39.4"
|
||||
@@ -5379,6 +5443,7 @@ dependencies = [
|
||||
"prometheus",
|
||||
"prost 0.13.5",
|
||||
"prost-types 0.13.5",
|
||||
"protoc-bin-vendored",
|
||||
"tokio",
|
||||
"tokio-stream",
|
||||
"tonic",
|
||||
|
||||
@@ -12,6 +12,17 @@ one.
|
||||
`core` knows nothing about any job. A second worker is a new crate beside
|
||||
`lance` that depends on it, not a fork of the protocol.
|
||||
|
||||
## Building
|
||||
|
||||
`core` compiles `plugin.proto` with the protoc that protoc-bin-vendored ships,
|
||||
the way seaweed-volume does, so it needs no system install.
|
||||
|
||||
The lance crates compile protos of their own, in their own build-script
|
||||
processes, which nothing our build script sets can reach. They need a protoc of
|
||||
their own: either one on PATH — `brew install protobuf`, `apt install
|
||||
protobuf-compiler` — or `PROTOC` naming one. CI points it at the vendored
|
||||
binary for the runner's platform, resolved from the version in `Cargo.lock`.
|
||||
|
||||
## Running
|
||||
|
||||
cargo run -p weed-lance-worker -- --admin 127.0.0.1:23646
|
||||
@@ -20,6 +31,22 @@ The admin's *HTTP* address is what an operator has; the gRPC port is derived
|
||||
from it the way the Go side does. Dialling the HTTP port fails as "frame with
|
||||
invalid size", which reads like a protocol bug rather than a wrong port.
|
||||
|
||||
The binary is `weed-worker`, not `weed-lance-worker`: it is the Rust side of
|
||||
`weed worker`, and lance is the first family of jobs it carries rather than the
|
||||
only one it ever will.
|
||||
|
||||
Released builds do not need a toolchain. The worker ships inside the SeaweedFS
|
||||
image, beside the Rust volume server, under the verb that mirrors
|
||||
`volume-rust`:
|
||||
|
||||
docker run chrislusf/seaweedfs worker-rust --admin admin:23646
|
||||
|
||||
and as `weed-worker_linux_{amd64,arm64}.tar.gz` on each GitHub release. Both are
|
||||
linux amd64/arm64 only — lance, arrow and datafusion make every extra target an
|
||||
expensive build, and the worker runs beside the cluster it maintains. On an
|
||||
architecture without a build the image carries an empty placeholder and the
|
||||
entrypoint says so rather than failing as "not found".
|
||||
|
||||
## Metrics
|
||||
|
||||
cargo run -p weed-lance-worker -- --admin 127.0.0.1:23646 --metrics-port 9328
|
||||
|
||||
@@ -21,3 +21,7 @@ tracing.workspace = true
|
||||
|
||||
[build-dependencies]
|
||||
tonic-build.workspace = true
|
||||
# Ships protoc with the build so neither CI nor a developer needs a system
|
||||
# install, and so the version is pinned rather than whatever the platform's
|
||||
# package manager happens to carry. The same crate seaweed-volume uses.
|
||||
protoc-bin-vendored = "3"
|
||||
|
||||
@@ -1,4 +1,12 @@
|
||||
fn main() -> Result<(), Box<dyn std::error::Error>> {
|
||||
// Use the protoc that ships with protoc-bin-vendored rather than a system
|
||||
// one, so the build needs no package manager and always sees the same
|
||||
// version. An explicit PROTOC still wins, for packagers supplying their own
|
||||
// and for the lance crates, whose own build scripts read the same variable.
|
||||
if std::env::var_os("PROTOC").is_none() {
|
||||
std::env::set_var("PROTOC", protoc_bin_vendored::protoc_bin_path()?);
|
||||
}
|
||||
|
||||
// Compiled straight out of the Go tree, the way seaweed-volume already reads
|
||||
// filer.proto, so the contract cannot drift from a vendored copy.
|
||||
tonic_build::configure()
|
||||
|
||||
@@ -7,8 +7,11 @@ description = "SeaweedFS maintenance worker for Lance tables"
|
||||
[lib]
|
||||
name = "weed_lance_worker"
|
||||
|
||||
# The binary is not named for lance: it is the Rust side of `weed worker`, and
|
||||
# the job families it registers will outgrow this crate. When a second one
|
||||
# arrives the bin target moves to a crate of its own under the same name.
|
||||
[[bin]]
|
||||
name = "weed-lance-worker"
|
||||
name = "weed-worker"
|
||||
path = "src/main.rs"
|
||||
|
||||
[dependencies]
|
||||
|
||||
@@ -13,8 +13,8 @@ use weed_lance_worker::metrics::LanceMetrics;
|
||||
/// language and an operator should not have to learn a second set of names.
|
||||
#[derive(Parser, Debug)]
|
||||
#[command(
|
||||
name = "weed-lance-worker",
|
||||
about = "SeaweedFS maintenance worker for Lance tables"
|
||||
name = "weed-worker",
|
||||
about = "SeaweedFS maintenance worker"
|
||||
)]
|
||||
struct Args {
|
||||
/// Admin server gRPC address.
|
||||
|
||||
+1
-1
@@ -139,7 +139,7 @@ The telemetry server exposes these Prometheus metrics:
|
||||
### Cluster Metrics
|
||||
- `seaweedfs_telemetry_total_clusters`: Total unique clusters (30 days)
|
||||
- `seaweedfs_telemetry_active_clusters`: Active clusters (7 days)
|
||||
- `seaweedfs_telemetry_confirmed_clusters`: Active clusters seen on 2+ distinct days — one-shot reports don't count, and the version/OS distributions in `/api/stats` are computed over these
|
||||
- `seaweedfs_telemetry_confirmed_clusters`: Active clusters seen on 7+ distinct days — one-shot reports don't count, and the version/OS distributions in `/api/stats` are computed over these
|
||||
|
||||
### Per-Cluster Metrics
|
||||
- `seaweedfs_telemetry_volume_servers{cluster_id}`: Volume servers per cluster
|
||||
|
||||
@@ -2,6 +2,7 @@ package api
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/json"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"testing"
|
||||
@@ -11,6 +12,9 @@ import (
|
||||
protobuf "google.golang.org/protobuf/proto"
|
||||
)
|
||||
|
||||
// promauto registers on the global registry: one storage per test binary.
|
||||
var testHandler = NewHandler(storage.NewPrometheusStorage())
|
||||
|
||||
func validReport() *proto.TelemetryData {
|
||||
return &proto.TelemetryData{
|
||||
TopologyId: "38422678-6a0d-4482-aa33-65b90010ac47",
|
||||
@@ -43,8 +47,7 @@ func marshalReport(t *testing.T, data *proto.TelemetryData) []byte {
|
||||
}
|
||||
|
||||
func TestCollectTelemetryValidation(t *testing.T) {
|
||||
// promauto registers on the global registry: one storage per test binary.
|
||||
h := NewHandler(storage.NewPrometheusStorage())
|
||||
h := testHandler
|
||||
|
||||
t.Run("valid report accepted", func(t *testing.T) {
|
||||
if w := postCollect(t, h, marshalReport(t, validReport()), "application/x-protobuf"); w.Code != http.StatusOK {
|
||||
@@ -96,3 +99,42 @@ func TestCollectTelemetryValidation(t *testing.T) {
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// The dashboard looks confirmation windows up by the select's string value,
|
||||
// so the stats JSON must key confirmed_by_days by decimal strings and carry
|
||||
// unmet thresholds as zeros.
|
||||
func TestStatsSerializedThresholds(t *testing.T) {
|
||||
data := validReport()
|
||||
data.TopologyId = "49533789-7b1e-4593-bb44-76ca1121bd58"
|
||||
if w := postCollect(t, testHandler, marshalReport(t, data), "application/x-protobuf"); w.Code != http.StatusOK {
|
||||
t.Fatalf("collect: got %d: %s", w.Code, w.Body.String())
|
||||
}
|
||||
|
||||
req := httptest.NewRequest(http.MethodGet, "/api/stats", nil)
|
||||
w := httptest.NewRecorder()
|
||||
testHandler.GetStats(w, req)
|
||||
if w.Code != http.StatusOK {
|
||||
t.Fatalf("stats: got %d", w.Code)
|
||||
}
|
||||
|
||||
var stats struct {
|
||||
ConfirmedByDays map[string]int `json:"confirmed_by_days"`
|
||||
}
|
||||
if err := json.Unmarshal(w.Body.Bytes(), &stats); err != nil {
|
||||
t.Fatalf("decode: %v", err)
|
||||
}
|
||||
if len(stats.ConfirmedByDays) != 5 {
|
||||
t.Fatalf("confirmed_by_days = %v, want the 5 thresholds", stats.ConfirmedByDays)
|
||||
}
|
||||
for _, key := range []string{"1", "3", "7", "14", "30"} {
|
||||
if _, ok := stats.ConfirmedByDays[key]; !ok {
|
||||
t.Errorf("confirmed_by_days missing %q: %v", key, stats.ConfirmedByDays)
|
||||
}
|
||||
}
|
||||
if stats.ConfirmedByDays["1"] < 1 {
|
||||
t.Errorf("fresh cluster missing from the 1-day count: %v", stats.ConfirmedByDays)
|
||||
}
|
||||
if stats.ConfirmedByDays["30"] != 0 {
|
||||
t.Errorf("unmet threshold not zero: %v", stats.ConfirmedByDays)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -57,6 +57,15 @@ func (h *Handler) ServeIndex(w http.ResponseWriter, r *http.Request) {
|
||||
color: #666;
|
||||
margin-top: 5px;
|
||||
}
|
||||
.stat-label select {
|
||||
border: none;
|
||||
background: none;
|
||||
color: inherit;
|
||||
font: inherit;
|
||||
padding: 0;
|
||||
cursor: pointer;
|
||||
text-decoration: underline dotted;
|
||||
}
|
||||
.chart-container {
|
||||
background: white;
|
||||
padding: 20px;
|
||||
@@ -129,7 +138,14 @@ func (h *Handler) ServeIndex(w http.ResponseWriter, r *http.Request) {
|
||||
</div>
|
||||
<div class="stat-card">
|
||||
<div class="stat-value" id="confirmedInstances">-</div>
|
||||
<div class="stat-label">Confirmed Clusters (2+ days)</div>
|
||||
<div class="stat-label">Confirmed Clusters
|
||||
(<select id="confirmDays" onchange="updateConfirmed()">
|
||||
<option value="1">1+</option>
|
||||
<option value="3">3+</option>
|
||||
<option value="7" selected>7+</option>
|
||||
<option value="14">14+</option>
|
||||
<option value="30">30+</option>
|
||||
</select> days)</div>
|
||||
</div>
|
||||
<div class="stat-card">
|
||||
<div class="stat-value" id="totalVersions">-</div>
|
||||
@@ -229,14 +245,26 @@ func (h *Handler) ServeIndex(w http.ResponseWriter, r *http.Request) {
|
||||
}
|
||||
}
|
||||
|
||||
let latestStats = {};
|
||||
|
||||
function updateStats(stats) {
|
||||
latestStats = stats;
|
||||
document.getElementById('totalInstances').textContent = stats.total_instances || 0;
|
||||
document.getElementById('activeInstances').textContent = stats.active_instances || 0;
|
||||
document.getElementById('confirmedInstances').textContent = stats.confirmed_instances || 0;
|
||||
updateConfirmed();
|
||||
document.getElementById('totalVersions').textContent = Object.keys(stats.versions || {}).length;
|
||||
document.getElementById('totalOS').textContent = Object.keys(stats.os_distribution || {}).length;
|
||||
}
|
||||
|
||||
// Servers from before confirmed_by_days fall back to the fixed
|
||||
// 7-day count.
|
||||
function updateConfirmed() {
|
||||
const days = document.getElementById('confirmDays').value;
|
||||
const byDays = latestStats.confirmed_by_days || {};
|
||||
const count = byDays[days] !== undefined ? byDays[days] : latestStats.confirmed_instances;
|
||||
document.getElementById('confirmedInstances').textContent = count || 0;
|
||||
}
|
||||
|
||||
function updateCharts(stats) {
|
||||
createPieChart('versionChart', 'Version Distribution', stats.versions || {});
|
||||
createPieChart('osChart', 'Operating System Distribution', stats.os_distribution || {});
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
module github.com/seaweedfs/seaweedfs/telemetry/server
|
||||
|
||||
go 1.25.8
|
||||
go 1.26
|
||||
|
||||
require (
|
||||
github.com/prometheus/client_golang v1.24.1
|
||||
|
||||
@@ -44,13 +44,18 @@ func TestConfirmedClusters(t *testing.T) {
|
||||
t.Fatalf("fallback distribution missing active cluster: %v", v)
|
||||
}
|
||||
|
||||
// Give cluster A a sample from yesterday: now seen on 2 distinct days.
|
||||
// Give cluster A samples from the six previous days: now seen on 7
|
||||
// distinct days.
|
||||
s.mu.Lock()
|
||||
id := "aaaaaaaa-0000-0000-0000-000000000001"
|
||||
s.histories[id] = append([]HistorySample{{
|
||||
Ts: time.Now().AddDate(0, 0, -1).Unix(),
|
||||
TotalDiskBytes: 50,
|
||||
}}, s.histories[id]...)
|
||||
var older []HistorySample
|
||||
for offset := -6; offset < 0; offset++ {
|
||||
older = append(older, HistorySample{
|
||||
Ts: time.Now().AddDate(0, 0, offset).Unix(),
|
||||
TotalDiskBytes: 50,
|
||||
})
|
||||
}
|
||||
s.histories[id] = append(older, s.histories[id]...)
|
||||
s.mu.Unlock()
|
||||
|
||||
// A one-shot cluster B arrives (like an injected report): it counts as
|
||||
@@ -60,7 +65,7 @@ func TestConfirmedClusters(t *testing.T) {
|
||||
}
|
||||
stats = statsOf(t, s)
|
||||
if stats["active_instances"] != 2 || stats["confirmed_instances"] != 1 {
|
||||
t.Fatalf("day two: active=%v confirmed=%v, want 2/1", stats["active_instances"], stats["confirmed_instances"])
|
||||
t.Fatalf("day seven: active=%v confirmed=%v, want 2/1", stats["active_instances"], stats["confirmed_instances"])
|
||||
}
|
||||
v := stats["versions"].(map[string]int)
|
||||
if v["4.40"] != 1 {
|
||||
@@ -69,4 +74,14 @@ func TestConfirmedClusters(t *testing.T) {
|
||||
if _, ok := v["9.99"]; ok {
|
||||
t.Errorf("one-shot cluster polluted the distribution: %v", v)
|
||||
}
|
||||
|
||||
// Cluster A meets the 1/3/7-day thresholds, B only the 1-day one, and
|
||||
// unmet thresholds are present as zero so the dashboard can show them.
|
||||
byDays := stats["confirmed_by_days"].(map[int]int)
|
||||
want := map[int]int{1: 2, 3: 1, 7: 1, 14: 0, 30: 0}
|
||||
for threshold, expected := range want {
|
||||
if got, ok := byDays[threshold]; !ok || got != expected {
|
||||
t.Errorf("confirmed_by_days[%d] = %v (present=%v), want %d", threshold, got, ok, expected)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -8,7 +8,11 @@ import (
|
||||
|
||||
// confirmDays is how many distinct UTC days a cluster must have reported
|
||||
// on before it counts as confirmed in the aggregated stats.
|
||||
const confirmDays = 2
|
||||
const confirmDays = 7
|
||||
|
||||
// confirmThresholds are the confirmation windows the dashboard lets the
|
||||
// viewer pick between; confirmDays is the one everything else is built on.
|
||||
var confirmThresholds = []int{1, 3, 7, 14, 30}
|
||||
|
||||
// activeDays is how recently a cluster must have reported to count as active.
|
||||
const activeDays = 7
|
||||
@@ -43,9 +47,9 @@ func (s *PrometheusStorage) appendHistory(data *proto.TelemetryData, receivedAt
|
||||
}
|
||||
|
||||
// seriesHistories picks the clusters the fleet-wide series are built from: the
|
||||
// confirmed ones. A cluster that only ever reported on one day is usually a CI
|
||||
// or test cluster that lived for a minute, and those arrive faster than they
|
||||
// age out, so counting them makes every fleet total climb forever. Falls back to
|
||||
// confirmed ones. A cluster that reported for less than a week is usually a CI
|
||||
// or test cluster, and those arrive faster than they age out, so counting them
|
||||
// makes every fleet total climb forever. Falls back to
|
||||
// all clusters while none is confirmed yet, so a fresh server still draws its
|
||||
// charts. Callers must hold s.mu.
|
||||
func (s *PrometheusStorage) seriesHistories() map[string][]HistorySample {
|
||||
|
||||
@@ -15,7 +15,8 @@ func TestGetMetricsSumsEachDay(t *testing.T) {
|
||||
seedSamples(s, "daily", HistorySample{TotalDiskBytes: 300, VolumeServerCount: 3},
|
||||
-9, -8, -7, -6, -5, -4, -3, -2, -1, 0)
|
||||
// Stopped reporting past the active window: counts on its own days only.
|
||||
seedSamples(s, "gone", HistorySample{TotalDiskBytes: 900, VolumeServerCount: 9}, -9, -8)
|
||||
seedSamples(s, "gone", HistorySample{TotalDiskBytes: 900, VolumeServerCount: 9},
|
||||
-14, -13, -12, -11, -10, -9, -8)
|
||||
|
||||
metrics, err := s.GetMetrics(10)
|
||||
if err != nil {
|
||||
@@ -40,7 +41,8 @@ func TestGetMetricsSumsEachDay(t *testing.T) {
|
||||
func TestGetMetricsCarriesSkippedDaysForward(t *testing.T) {
|
||||
s := newPrometheusStorage(prometheus.NewRegistry())
|
||||
|
||||
seedSamples(s, "gappy", HistorySample{TotalDiskBytes: 500, VolumeServerCount: 5}, -3, -1)
|
||||
seedSamples(s, "gappy", HistorySample{TotalDiskBytes: 500, VolumeServerCount: 5},
|
||||
-8, -7, -6, -5, -4, -3, -1)
|
||||
|
||||
metrics, err := s.GetMetrics(4)
|
||||
if err != nil {
|
||||
@@ -56,21 +58,23 @@ func TestGetMetricsCarriesSkippedDaysForward(t *testing.T) {
|
||||
// out of nothing on its first day of data.
|
||||
func TestGetMetricsWindowStartsAtOldestSample(t *testing.T) {
|
||||
s := newPrometheusStorage(prometheus.NewRegistry())
|
||||
seedSamples(s, "recent", HistorySample{TotalDiskBytes: 100, VolumeServerCount: 1}, -2, -1, 0)
|
||||
seedSamples(s, "recent", HistorySample{TotalDiskBytes: 100, VolumeServerCount: 1},
|
||||
-6, -5, -4, -3, -2, -1, 0)
|
||||
|
||||
metrics, err := s.GetMetrics(30)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if got := metrics["dates"].([]string); len(got) != 3 {
|
||||
t.Errorf("dates = %v, want the 3 days with history, not 30", got)
|
||||
if got := metrics["dates"].([]string); len(got) != 7 {
|
||||
t.Errorf("dates = %v, want the 7 days with history, not 30", got)
|
||||
}
|
||||
if got := metrics["disk_usage"].([]uint64); !equal(got, []uint64{100, 100, 100}) {
|
||||
if got := metrics["disk_usage"].([]uint64); !equal(got, []uint64{100, 100, 100, 100, 100, 100, 100}) {
|
||||
t.Errorf("disk_usage = %v, want no leading zero days", got)
|
||||
}
|
||||
|
||||
// History reaching past the requested window still clips to the window.
|
||||
seedSamples(s, "old", HistorySample{TotalDiskBytes: 50, VolumeServerCount: 1}, -40, -39)
|
||||
seedSamples(s, "old", HistorySample{TotalDiskBytes: 50, VolumeServerCount: 1},
|
||||
-45, -44, -43, -42, -41, -40, -39)
|
||||
metrics, err = s.GetMetrics(10)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
@@ -86,8 +90,10 @@ func TestGetMetricsAgreesWithClusterSizes(t *testing.T) {
|
||||
s := newPrometheusStorage(prometheus.NewRegistry())
|
||||
seedSamples(s, "daily", HistorySample{TotalDiskBytes: 300, VolumeServerCount: 3},
|
||||
-9, -8, -7, -6, -5, -4, -3, -2, -1, 0)
|
||||
seedSamples(s, "lagging", HistorySample{TotalDiskBytes: 200, VolumeServerCount: 2}, -3, -2)
|
||||
seedSamples(s, "gone", HistorySample{TotalDiskBytes: 900, VolumeServerCount: 9}, -9, -8)
|
||||
seedSamples(s, "lagging", HistorySample{TotalDiskBytes: 200, VolumeServerCount: 2},
|
||||
-8, -7, -6, -5, -4, -3, -2)
|
||||
seedSamples(s, "gone", HistorySample{TotalDiskBytes: 900, VolumeServerCount: 9},
|
||||
-14, -13, -12, -11, -10, -9, -8)
|
||||
|
||||
metrics, err := s.GetMetrics(10)
|
||||
if err != nil {
|
||||
@@ -111,7 +117,8 @@ func TestGetMetricsAgreesWithClusterSizes(t *testing.T) {
|
||||
func TestGetMetricsExcludesUnconfirmedClusters(t *testing.T) {
|
||||
s := newPrometheusStorage(prometheus.NewRegistry())
|
||||
|
||||
seedSamples(s, "real", HistorySample{TotalDiskBytes: 300, VolumeServerCount: 3}, -3, -2, -1, 0)
|
||||
seedSamples(s, "real", HistorySample{TotalDiskBytes: 300, VolumeServerCount: 3},
|
||||
-6, -5, -4, -3, -2, -1, 0)
|
||||
for _, id := range []string{"ci-1", "ci-2", "ci-3"} {
|
||||
seedSamples(s, id, HistorySample{TotalDiskBytes: 5, VolumeServerCount: 14}, -1)
|
||||
}
|
||||
@@ -128,7 +135,7 @@ func TestGetMetricsExcludesUnconfirmedClusters(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// Until any cluster has two days of history the charts fall back to every
|
||||
// Until any cluster has a week of history the charts fall back to every
|
||||
// cluster, so a fresh server doesn't serve empty series.
|
||||
func TestGetMetricsFallsBackWhenNoneConfirmed(t *testing.T) {
|
||||
s := newPrometheusStorage(prometheus.NewRegistry())
|
||||
|
||||
@@ -53,7 +53,7 @@ func newPrometheusStorage(reg prometheus.Registerer) *PrometheusStorage {
|
||||
}),
|
||||
confirmedClusters: promauto.NewGauge(prometheus.GaugeOpts{
|
||||
Name: "seaweedfs_telemetry_confirmed_clusters",
|
||||
Help: "Active clusters seen on at least 2 distinct days (last 7 days)",
|
||||
Help: "Active clusters seen on at least 7 distinct days (last 7 days)",
|
||||
}),
|
||||
volumeServerCount: promauto.NewGaugeVec(prometheus.GaugeOpts{
|
||||
Name: "seaweedfs_telemetry_volume_servers",
|
||||
@@ -207,6 +207,10 @@ func (s *PrometheusStorage) updateStats() {
|
||||
totalInstances := 0
|
||||
activeInstances := 0
|
||||
confirmedInstances := 0
|
||||
confirmedByDays := make(map[int]int, len(confirmThresholds))
|
||||
for _, threshold := range confirmThresholds {
|
||||
confirmedByDays[threshold] = 0
|
||||
}
|
||||
versionsAll := make(map[string]int)
|
||||
osAll := make(map[string]int)
|
||||
versionsConfirmed := make(map[string]int)
|
||||
@@ -220,10 +224,16 @@ func (s *PrometheusStorage) updateStats() {
|
||||
activeInstances++
|
||||
versionsAll[instance.TelemetryData.Version]++
|
||||
osAll[instance.TelemetryData.Os]++
|
||||
// A cluster is confirmed once seen on >=2 distinct UTC days
|
||||
// (histories hold one sample per day), so one-shot reports
|
||||
// can't skew the distributions below.
|
||||
if len(s.histories[instance.TelemetryData.TopologyId]) >= confirmDays {
|
||||
// A cluster is confirmed once seen on confirmDays distinct UTC
|
||||
// days (histories hold one sample per day), so short-lived
|
||||
// clusters can't skew the distributions below.
|
||||
daysSeen := len(s.histories[instance.TelemetryData.TopologyId])
|
||||
for _, threshold := range confirmThresholds {
|
||||
if daysSeen >= threshold {
|
||||
confirmedByDays[threshold]++
|
||||
}
|
||||
}
|
||||
if daysSeen >= confirmDays {
|
||||
confirmedInstances++
|
||||
versionsConfirmed[instance.TelemetryData.Version]++
|
||||
osConfirmed[instance.TelemetryData.Os]++
|
||||
@@ -231,7 +241,7 @@ func (s *PrometheusStorage) updateStats() {
|
||||
}
|
||||
}
|
||||
|
||||
// Before any cluster has two days of history (fresh server with no
|
||||
// Before any cluster has a week of history (fresh server with no
|
||||
// prior state), fall back to all active clusters so the dashboard
|
||||
// distributions aren't empty.
|
||||
versions, osDistribution := versionsConfirmed, osConfirmed
|
||||
@@ -249,6 +259,7 @@ func (s *PrometheusStorage) updateStats() {
|
||||
"total_instances": totalInstances,
|
||||
"active_instances": activeInstances,
|
||||
"confirmed_instances": confirmedInstances,
|
||||
"confirmed_by_days": confirmedByDays,
|
||||
"versions": versions,
|
||||
"os_distribution": osDistribution,
|
||||
}
|
||||
|
||||
@@ -25,11 +25,13 @@ func TestClusterSizeSeries(t *testing.T) {
|
||||
// Reported every day of the window.
|
||||
seedSamples(s, "daily", HistorySample{TotalDiskBytes: 300, VolumeServerCount: 3},
|
||||
-9, -8, -7, -6, -5, -4, -3, -2, -1, 0)
|
||||
// Reported two days ago and not since: still active, so its size is held
|
||||
// to the right edge instead of dropping out of the stack.
|
||||
seedSamples(s, "lagging", HistorySample{TotalDiskBytes: 200, VolumeServerCount: 2}, -3, -2)
|
||||
// Stopped reporting two days ago: still active, so its size is held to
|
||||
// the right edge instead of dropping out of the stack.
|
||||
seedSamples(s, "lagging", HistorySample{TotalDiskBytes: 200, VolumeServerCount: 2},
|
||||
-8, -7, -6, -5, -4, -3, -2)
|
||||
// Stopped reporting past the active window: its own days only.
|
||||
seedSamples(s, "gone", HistorySample{TotalDiskBytes: 900, VolumeServerCount: 9}, -9, -8)
|
||||
seedSamples(s, "gone", HistorySample{TotalDiskBytes: 900, VolumeServerCount: 9},
|
||||
-14, -13, -12, -11, -10, -9, -8)
|
||||
// One day of history only: unconfirmed, so it stays out of the stack.
|
||||
seedSamples(s, "oneshot", HistorySample{TotalDiskBytes: 400, VolumeServerCount: 4}, -1)
|
||||
|
||||
@@ -48,7 +50,7 @@ func TestClusterSizeSeries(t *testing.T) {
|
||||
if got := byId["daily"].Disk; !equal(got, []uint64{300, 300, 300, 300, 300, 300, 300, 300, 300, 300}) {
|
||||
t.Errorf("daily = %v, want 300 every day", got)
|
||||
}
|
||||
if got := byId["lagging"].Disk; !equal(got, []uint64{0, 0, 0, 0, 0, 0, 200, 200, 200, 200}) {
|
||||
if got := byId["lagging"].Disk; !equal(got, []uint64{0, 200, 200, 200, 200, 200, 200, 200, 200, 200}) {
|
||||
t.Errorf("lagging = %v, want its size carried to the right edge", got)
|
||||
}
|
||||
if got := byId["gone"].Disk; !equal(got, []uint64{900, 900, 0, 0, 0, 0, 0, 0, 0, 0}) {
|
||||
@@ -62,7 +64,7 @@ func TestClusterSizeSeries(t *testing.T) {
|
||||
if got := byId["daily"].Servers; !equal(got, []uint64{3, 3, 3, 3, 3, 3, 3, 3, 3, 3}) {
|
||||
t.Errorf("daily servers = %v, want 3 every day", got)
|
||||
}
|
||||
if got := byId["lagging"].Servers; !equal(got, []uint64{0, 0, 0, 0, 0, 0, 2, 2, 2, 2}) {
|
||||
if got := byId["lagging"].Servers; !equal(got, []uint64{0, 2, 2, 2, 2, 2, 2, 2, 2, 2}) {
|
||||
t.Errorf("lagging servers = %v, want carried to the right edge", got)
|
||||
}
|
||||
if got := byId["gone"].Servers; !equal(got, []uint64{9, 9, 0, 0, 0, 0, 0, 0, 0, 0}) {
|
||||
@@ -89,10 +91,10 @@ func TestClusterSizeSeries(t *testing.T) {
|
||||
if series.Other == nil || series.Other.Count != 2 {
|
||||
t.Fatalf("other = %+v, want 2 clusters", series.Other)
|
||||
}
|
||||
if !equal(series.Other.Disk, []uint64{900, 900, 0, 0, 0, 0, 200, 200, 200, 200}) {
|
||||
if !equal(series.Other.Disk, []uint64{900, 1100, 200, 200, 200, 200, 200, 200, 200, 200}) {
|
||||
t.Errorf("other = %v, want lagging+gone summed per day", series.Other.Disk)
|
||||
}
|
||||
if !equal(series.Other.Servers, []uint64{9, 9, 0, 0, 0, 0, 2, 2, 2, 2}) {
|
||||
if !equal(series.Other.Servers, []uint64{9, 11, 2, 2, 2, 2, 2, 2, 2, 2}) {
|
||||
t.Errorf("other servers = %v, want lagging+gone summed per day", series.Other.Servers)
|
||||
}
|
||||
if series.ClusterCount != 3 || series.TotalDisk != 500 || series.TotalServers != 5 {
|
||||
|
||||
@@ -10,31 +10,31 @@ func TestVersionSeries(t *testing.T) {
|
||||
s := newPrometheusStorage(prometheus.NewRegistry())
|
||||
|
||||
// Upgraded mid-window: its band leaves the old version for the new one.
|
||||
seedSamples(s, "upgraded", HistorySample{Version: "4.39"}, -4, -3)
|
||||
seedSamples(s, "upgraded", HistorySample{Version: "4.39"}, -6, -5, -4, -3)
|
||||
seedSamples(s, "upgraded", HistorySample{Version: "4.40"}, -2, -1, 0)
|
||||
// Reported every day on the same version.
|
||||
seedSamples(s, "steady", HistorySample{Version: "4.40"}, -4, -3, -2, -1, 0)
|
||||
seedSamples(s, "steady", HistorySample{Version: "4.40"}, -6, -5, -4, -3, -2, -1, 0)
|
||||
// One day of history only: unconfirmed, so it stays out of the stack.
|
||||
seedSamples(s, "oneshot", HistorySample{Version: "4.40"}, -1)
|
||||
// Confirmed, but its samples predate versions being recorded.
|
||||
seedSamples(s, "versionless", HistorySample{}, -9, -8)
|
||||
seedSamples(s, "versionless", HistorySample{}, -13, -12, -11, -10, -9, -8, -7)
|
||||
|
||||
series := s.GetVersionSeries(10, 0)
|
||||
|
||||
// The versionless days are dropped before the axis is built, so the chart
|
||||
// spans the days a version is known for instead of climbing out of blanks.
|
||||
if len(series.Dates) != 5 {
|
||||
t.Fatalf("dates = %v, want the 5 days with versions", series.Dates)
|
||||
if len(series.Dates) != 7 {
|
||||
t.Fatalf("dates = %v, want the 7 days with versions", series.Dates)
|
||||
}
|
||||
if len(series.Versions) != 2 ||
|
||||
series.Versions[0].Version != "4.39" || series.Versions[1].Version != "4.40" {
|
||||
t.Fatalf("versions = %+v, want 4.39 then 4.40", series.Versions)
|
||||
}
|
||||
if got := series.Versions[0].Clusters; !equal(got, []uint64{1, 1, 0, 0, 0}) {
|
||||
t.Errorf("4.39 = %v, want the upgraded cluster's first two days", got)
|
||||
if got := series.Versions[0].Clusters; !equal(got, []uint64{1, 1, 1, 1, 0, 0, 0}) {
|
||||
t.Errorf("4.39 = %v, want the upgraded cluster's first four days", got)
|
||||
}
|
||||
if got := series.Versions[1].Clusters; !equal(got, []uint64{1, 1, 2, 2, 2}) {
|
||||
t.Errorf("4.40 = %v, want steady plus upgraded from day 3", got)
|
||||
if got := series.Versions[1].Clusters; !equal(got, []uint64{1, 1, 1, 1, 2, 2, 2}) {
|
||||
t.Errorf("4.40 = %v, want steady plus upgraded from day 5", got)
|
||||
}
|
||||
if series.TotalClusters != 2 {
|
||||
t.Errorf("total_clusters = %d, want 2", series.TotalClusters)
|
||||
@@ -48,12 +48,12 @@ func TestVersionSeriesHoldsForwardAndLimits(t *testing.T) {
|
||||
s := newPrometheusStorage(prometheus.NewRegistry())
|
||||
|
||||
// Stopped reporting past the active window: its own days only.
|
||||
seedSamples(s, "gone", HistorySample{Version: "3.97"}, -9, -8)
|
||||
seedSamples(s, "gone", HistorySample{Version: "3.97"}, -14, -13, -12, -11, -10, -9, -8)
|
||||
// Reported every day of the window.
|
||||
seedSamples(s, "daily", HistorySample{Version: "4.40"}, -9, -8, -7, -6, -5, -4, -3, -2, -1, 0)
|
||||
// Reported three days ago and not since: still active, so it holds its
|
||||
// version to the right edge instead of dropping out of the stack.
|
||||
seedSamples(s, "lagging", HistorySample{Version: "4.30"}, -3, -2)
|
||||
// Stopped reporting two days ago: still active, so it holds its version
|
||||
// to the right edge instead of dropping out of the stack.
|
||||
seedSamples(s, "lagging", HistorySample{Version: "4.30"}, -8, -7, -6, -5, -4, -3, -2)
|
||||
|
||||
series := s.GetVersionSeries(10, 0)
|
||||
if len(series.Dates) != 10 {
|
||||
@@ -66,7 +66,7 @@ func TestVersionSeriesHoldsForwardAndLimits(t *testing.T) {
|
||||
if got := byVersion["3.97"]; !equal(got, []uint64{1, 1, 0, 0, 0, 0, 0, 0, 0, 0}) {
|
||||
t.Errorf("3.97 = %v, want nothing after its last report", got)
|
||||
}
|
||||
if got := byVersion["4.30"]; !equal(got, []uint64{0, 0, 0, 0, 0, 0, 1, 1, 1, 1}) {
|
||||
if got := byVersion["4.30"]; !equal(got, []uint64{0, 1, 1, 1, 1, 1, 1, 1, 1, 1}) {
|
||||
t.Errorf("4.30 = %v, want carried to the right edge", got)
|
||||
}
|
||||
if got := byVersion["4.40"]; !equal(got, []uint64{1, 1, 1, 1, 1, 1, 1, 1, 1, 1}) {
|
||||
@@ -85,7 +85,7 @@ func TestVersionSeriesHoldsForwardAndLimits(t *testing.T) {
|
||||
if series.Other == nil || series.Other.Count != 2 {
|
||||
t.Fatalf("other = %+v, want 2 versions", series.Other)
|
||||
}
|
||||
if !equal(series.Other.Clusters, []uint64{1, 1, 0, 0, 0, 0, 1, 1, 1, 1}) {
|
||||
if !equal(series.Other.Clusters, []uint64{1, 2, 1, 1, 1, 1, 1, 1, 1, 1}) {
|
||||
t.Errorf("other = %v, want 3.97+4.30 summed per day", series.Other.Clusters)
|
||||
}
|
||||
if series.TotalClusters != 2 {
|
||||
|
||||
@@ -933,7 +933,7 @@ func copyFileContents(src, dst string) error {
|
||||
// chaosDataNodes lists the data nodes from a fresh master topology snapshot.
|
||||
func chaosDataNodes(commandEnv *shell.CommandEnv) []*master_pb.DataNodeInfo {
|
||||
var resp *master_pb.VolumeListResponse
|
||||
err := commandEnv.MasterClient.WithClient(false, func(client master_pb.SeaweedClient) error {
|
||||
err := commandEnv.MasterClient.WithClient(context.Background(), false, func(client master_pb.SeaweedClient) error {
|
||||
var e error
|
||||
resp, e = client.VolumeList(context.Background(), &master_pb.VolumeListRequest{})
|
||||
return e
|
||||
@@ -955,7 +955,7 @@ func chaosDataNodes(commandEnv *shell.CommandEnv) []*master_pb.DataNodeInfo {
|
||||
func masterEcGenerations(commandEnv *shell.CommandEnv, volumeId uint32) map[int64]bool {
|
||||
generations := map[int64]bool{}
|
||||
var resp *master_pb.VolumeListResponse
|
||||
err := commandEnv.MasterClient.WithClient(false, func(client master_pb.SeaweedClient) error {
|
||||
err := commandEnv.MasterClient.WithClient(context.Background(), false, func(client master_pb.SeaweedClient) error {
|
||||
var e error
|
||||
resp, e = client.VolumeList(context.Background(), &master_pb.VolumeListRequest{})
|
||||
return e
|
||||
|
||||
@@ -200,7 +200,7 @@ func disksWithShards(testDir string, volumeId uint32) int {
|
||||
func nodeVolumeDiskCounts(t *testing.T, commandEnv *shell.CommandEnv) map[string]int {
|
||||
t.Helper()
|
||||
var resp *master_pb.VolumeListResponse
|
||||
err := commandEnv.MasterClient.WithClient(false, func(client master_pb.SeaweedClient) error {
|
||||
err := commandEnv.MasterClient.WithClient(context.Background(), false, func(client master_pb.SeaweedClient) error {
|
||||
var e error
|
||||
resp, e = client.VolumeList(context.Background(), &master_pb.VolumeListRequest{})
|
||||
return e
|
||||
|
||||
@@ -370,7 +370,7 @@ func removeTwoShardFiles(t *testing.T, testDir string, volumeId uint32) []int {
|
||||
func masterEcShardIds(commandEnv *shell.CommandEnv, volumeId uint32) map[int]bool {
|
||||
ids := map[int]bool{}
|
||||
var resp *master_pb.VolumeListResponse
|
||||
err := commandEnv.MasterClient.WithClient(false, func(client master_pb.SeaweedClient) error {
|
||||
err := commandEnv.MasterClient.WithClient(context.Background(), false, func(client master_pb.SeaweedClient) error {
|
||||
var e error
|
||||
resp, e = client.VolumeList(context.Background(), &master_pb.VolumeListRequest{})
|
||||
return e
|
||||
|
||||
@@ -8,6 +8,7 @@ import (
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"strings"
|
||||
"sync"
|
||||
"syscall"
|
||||
"testing"
|
||||
@@ -90,6 +91,8 @@ func startDLMTestCluster(t testing.TB) *dlmTestCluster {
|
||||
require.NoError(t, c.startVolume(configDir))
|
||||
require.NoError(t, c.waitForTCP(fmt.Sprintf("127.0.0.1:%d", c.volumePort), 30*time.Second),
|
||||
"volume not ready\n%s", c.tailLog("volume"))
|
||||
require.NoError(t, c.waitForVolumeRegistered(30*time.Second),
|
||||
"volume server registration\n%s", c.tailLog("master"))
|
||||
|
||||
// Start 2 filers
|
||||
for i := 0; i < 2; i++ {
|
||||
@@ -261,7 +264,10 @@ func (c *dlmTestCluster) tailLog(name string) string {
|
||||
}
|
||||
|
||||
func (c *dlmTestCluster) copyLogsForCI() {
|
||||
ciLogDir := "/tmp/seaweedfs-fuse-dlm-logs"
|
||||
// One subdirectory per test: a flat layout lets every teardown overwrite
|
||||
// the previous test's logs, so the CI artifact only ever shows the last
|
||||
// cluster, never the failing one.
|
||||
ciLogDir := filepath.Join("/tmp/seaweedfs-fuse-dlm-logs", strings.ReplaceAll(c.t.Name(), "/", "_"))
|
||||
os.MkdirAll(ciLogDir, 0755)
|
||||
logsDir := filepath.Join(c.baseDir, "logs")
|
||||
entries, err := os.ReadDir(logsDir)
|
||||
@@ -348,6 +354,44 @@ func (c *dlmTestCluster) waitForFilerCount(expected int, timeout time.Duration)
|
||||
return fmt.Errorf("timed out waiting for %d filers in group %q", expected, filerGroup)
|
||||
}
|
||||
|
||||
// waitForVolumeRegistered waits until the volume server shows up in the master
|
||||
// topology with its slots reported. An open volume port only means the process
|
||||
// is listening: until the master has elected itself and accepted a heartbeat,
|
||||
// an assign fails and the first write on a fresh mount surfaces it as ENOSPC.
|
||||
func (c *dlmTestCluster) waitForVolumeRegistered(timeout time.Duration) error {
|
||||
addr := fmt.Sprintf("127.0.0.1:%d", c.masterGrpcPort)
|
||||
conn, err := grpc.NewClient(addr, grpc.WithTransportCredentials(insecure.NewCredentials()))
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer conn.Close()
|
||||
|
||||
client := master_pb.NewSeaweedClient(conn)
|
||||
deadline := time.Now().Add(timeout)
|
||||
for time.Now().Before(deadline) {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), time.Second)
|
||||
resp, err := client.VolumeList(ctx, &master_pb.VolumeListRequest{})
|
||||
cancel()
|
||||
if err == nil && resp.TopologyInfo != nil {
|
||||
var slots int64
|
||||
for _, dc := range resp.TopologyInfo.DataCenterInfos {
|
||||
for _, rack := range dc.RackInfos {
|
||||
for _, dn := range rack.DataNodeInfos {
|
||||
for _, disk := range dn.DiskInfos {
|
||||
slots += disk.MaxVolumeCount
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if slots > 0 {
|
||||
return nil
|
||||
}
|
||||
}
|
||||
time.Sleep(200 * time.Millisecond)
|
||||
}
|
||||
return fmt.Errorf("volume server not registered with the master within %v", timeout)
|
||||
}
|
||||
|
||||
// waitForLockRingConverged verifies that both filers have a consistent view of
|
||||
// the lock ring by acquiring the same lock through each filer and checking
|
||||
// mutual exclusion. Adapted from test/s3/distributed_lock/.
|
||||
|
||||
@@ -3,12 +3,14 @@
|
||||
package fuse_p2p
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"net"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"strings"
|
||||
"sync"
|
||||
"syscall"
|
||||
"testing"
|
||||
@@ -16,7 +18,10 @@ import (
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/test/testutil"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
|
||||
"github.com/stretchr/testify/require"
|
||||
"google.golang.org/grpc"
|
||||
"google.golang.org/grpc/credentials/insecure"
|
||||
)
|
||||
|
||||
// p2pTestCluster manages a minimal SeaweedFS cluster exercising the peer
|
||||
@@ -100,6 +105,8 @@ func startP2PTestCluster(t testing.TB, numMounts int) *p2pTestCluster {
|
||||
require.NoError(t, c.startVolume(configDir))
|
||||
require.NoError(t, c.waitForTCP(c.volumeCmd, "volume",
|
||||
fmt.Sprintf("127.0.0.1:%d", c.volumePort), 30*time.Second))
|
||||
require.NoError(t, c.waitForVolumeRegistered(30*time.Second),
|
||||
"volume server registration\n%s", c.tailLog("master"))
|
||||
|
||||
require.NoError(t, c.startFiler(configDir))
|
||||
require.NoError(t, c.waitForTCP(c.filerCmd, "filer",
|
||||
@@ -272,7 +279,10 @@ func (c *p2pTestCluster) tailLogFull(name string) string {
|
||||
}
|
||||
|
||||
func (c *p2pTestCluster) copyLogsForCI() {
|
||||
ciLogDir := "/tmp/seaweedfs-fuse-p2p-logs"
|
||||
// One subdirectory per test: a flat layout lets every teardown overwrite
|
||||
// the previous test's logs, so the CI artifact only ever shows the last
|
||||
// cluster, never the failing one.
|
||||
ciLogDir := filepath.Join("/tmp/seaweedfs-fuse-p2p-logs", strings.ReplaceAll(c.t.Name(), "/", "_"))
|
||||
os.MkdirAll(ciLogDir, 0755)
|
||||
logsDir := filepath.Join(c.baseDir, "logs")
|
||||
entries, err := os.ReadDir(logsDir)
|
||||
@@ -288,6 +298,44 @@ func (c *p2pTestCluster) copyLogsForCI() {
|
||||
}
|
||||
}
|
||||
|
||||
// waitForVolumeRegistered waits until the volume server shows up in the master
|
||||
// topology with its slots reported. An open volume port only means the process
|
||||
// is listening: until the master has elected itself and accepted a heartbeat,
|
||||
// an assign fails and the first write on a fresh mount surfaces it as ENOSPC.
|
||||
func (c *p2pTestCluster) waitForVolumeRegistered(timeout time.Duration) error {
|
||||
addr := fmt.Sprintf("127.0.0.1:%d", c.masterGrpcPort)
|
||||
conn, err := grpc.NewClient(addr, grpc.WithTransportCredentials(insecure.NewCredentials()))
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer conn.Close()
|
||||
|
||||
client := master_pb.NewSeaweedClient(conn)
|
||||
deadline := time.Now().Add(timeout)
|
||||
for time.Now().Before(deadline) {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), time.Second)
|
||||
resp, err := client.VolumeList(ctx, &master_pb.VolumeListRequest{})
|
||||
cancel()
|
||||
if err == nil && resp.TopologyInfo != nil {
|
||||
var slots int64
|
||||
for _, dc := range resp.TopologyInfo.DataCenterInfos {
|
||||
for _, rack := range dc.RackInfos {
|
||||
for _, dn := range rack.DataNodeInfos {
|
||||
for _, disk := range dn.DiskInfos {
|
||||
slots += disk.MaxVolumeCount
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if slots > 0 {
|
||||
return nil
|
||||
}
|
||||
}
|
||||
time.Sleep(200 * time.Millisecond)
|
||||
}
|
||||
return fmt.Errorf("volume server not registered with the master within %v", timeout)
|
||||
}
|
||||
|
||||
// waitForTCP polls addr until it accepts a connection, OR the supplied
|
||||
// subprocess exits — whichever comes first. Short-circuiting on child
|
||||
// exit turns a 30 s-spin-on-dead-process into an immediate failure with
|
||||
|
||||
@@ -10,6 +10,7 @@ import (
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"regexp"
|
||||
"runtime"
|
||||
"strconv"
|
||||
"strings"
|
||||
@@ -36,6 +37,9 @@ type masterNode struct {
|
||||
// peersStr overrides the cluster-wide peer list for this node, so a test
|
||||
// can start a master that only knows about a subset of the cluster.
|
||||
peersStr string
|
||||
// raftBootstrap starts the master with -raftBootstrap, the way the helm
|
||||
// chart renders master.raftBootstrap on every master, every restart.
|
||||
raftBootstrap bool
|
||||
}
|
||||
|
||||
// MasterCluster manages a 3-node master raft cluster for integration tests.
|
||||
@@ -153,6 +157,13 @@ func (mc *MasterCluster) SetNodePeers(i int, peers string) {
|
||||
mc.nodes[i].peersStr = peers
|
||||
}
|
||||
|
||||
// SetRaftBootstrap makes node i start with -raftBootstrap.
|
||||
func (mc *MasterCluster) SetRaftBootstrap(i int) {
|
||||
mc.mu.Lock()
|
||||
defer mc.mu.Unlock()
|
||||
mc.nodes[i].raftBootstrap = true
|
||||
}
|
||||
|
||||
// StartNode starts the master process at the given index (0–2).
|
||||
func (mc *MasterCluster) StartNode(i int) {
|
||||
mc.t.Helper()
|
||||
@@ -187,6 +198,9 @@ func (mc *MasterCluster) StartNode(i int) {
|
||||
if mc.raftHashicorp {
|
||||
args = append(args, "-raftHashicorp")
|
||||
}
|
||||
if n.raftBootstrap {
|
||||
args = append(args, "-raftBootstrap")
|
||||
}
|
||||
|
||||
n.cmd = exec.Command(mc.weedBinary, args...)
|
||||
n.cmd.Dir = mc.baseDir
|
||||
@@ -386,6 +400,57 @@ func (mc *MasterCluster) WaitForNodeReady(i int, timeout time.Duration) error {
|
||||
return fmt.Errorf("node %d not ready within %v", i, timeout)
|
||||
}
|
||||
|
||||
// LogContains reports whether node i's log holds the given text.
|
||||
func (mc *MasterCluster) LogContains(i int, text string) bool {
|
||||
b, err := os.ReadFile(mc.nodes[i].logFile)
|
||||
if err != nil {
|
||||
return false
|
||||
}
|
||||
return strings.Contains(string(b), text)
|
||||
}
|
||||
|
||||
// topologyIdLine matches every line a master logs when it learns a TopologyId,
|
||||
// whichever raft implementation applied it.
|
||||
var topologyIdLine = regexp.MustCompile(`TopologyId[^:]*: ([0-9a-f-]{36})`)
|
||||
|
||||
// NodeTopologyIds returns the TopologyIds node i has logged. /dir/status is
|
||||
// proxied to the leader, so a master's own view of the cluster identity is only
|
||||
// visible in its log, and that is where a fork shows up.
|
||||
func (mc *MasterCluster) NodeTopologyIds(i int) []string {
|
||||
b, err := os.ReadFile(mc.nodes[i].logFile)
|
||||
if err != nil {
|
||||
return nil
|
||||
}
|
||||
var ids []string
|
||||
for _, m := range topologyIdLine.FindAllStringSubmatch(string(b), -1) {
|
||||
ids = append(ids, m[1])
|
||||
}
|
||||
return ids
|
||||
}
|
||||
|
||||
// WaitForNodeTopologyIds waits until every master has logged a TopologyId and
|
||||
// returns what each one saw.
|
||||
func (mc *MasterCluster) WaitForNodeTopologyIds(timeout time.Duration) ([3][]string, error) {
|
||||
var ids [3][]string
|
||||
deadline := time.Now().Add(timeout)
|
||||
for {
|
||||
missing := -1
|
||||
for i := range 3 {
|
||||
ids[i] = mc.NodeTopologyIds(i)
|
||||
if len(ids[i]) == 0 {
|
||||
missing = i
|
||||
}
|
||||
}
|
||||
if missing < 0 {
|
||||
return ids, nil
|
||||
}
|
||||
if !time.Now().Before(deadline) {
|
||||
return ids, fmt.Errorf("master %d logged no TopologyId within %v", missing, timeout)
|
||||
}
|
||||
time.Sleep(waitTick)
|
||||
}
|
||||
}
|
||||
|
||||
// DumpLogs prints the tail of all master logs.
|
||||
func (mc *MasterCluster) DumpLogs() {
|
||||
for i := range 3 {
|
||||
|
||||
@@ -151,3 +151,62 @@ func peerCountExcludingSelf(peers []string, self string) int {
|
||||
}
|
||||
return count
|
||||
}
|
||||
|
||||
// TestRaftBootstrapKeepsExistingCluster covers a master restarting under
|
||||
// -raftBootstrap, the way the helm chart renders it on every master on every
|
||||
// roll. Bootstrapping is genesis: seeding a second cluster over committed raft
|
||||
// state mints a rival TopologyId, and the split-brain guard then Fatals every
|
||||
// master that still holds the first one.
|
||||
func TestRaftBootstrapKeepsExistingCluster(t *testing.T) {
|
||||
for _, impl := range raftImplementations {
|
||||
t.Run(impl.name, func(t *testing.T) {
|
||||
mc := NewMasterCluster(t, impl.raftHashicorp)
|
||||
for i := range 3 {
|
||||
mc.SetRaftBootstrap(i)
|
||||
mc.StartNode(i)
|
||||
}
|
||||
before, err := mc.WaitForTopologyId(waitTimeout)
|
||||
if err != nil {
|
||||
mc.DumpLogs()
|
||||
t.Fatalf("cluster did not mint a TopologyId: %v", err)
|
||||
}
|
||||
|
||||
for i := range 3 {
|
||||
mc.StopNode(i)
|
||||
}
|
||||
for i := range 3 {
|
||||
mc.StartNode(i)
|
||||
}
|
||||
|
||||
after, err := mc.WaitForTopologyId(waitTimeout)
|
||||
if err != nil {
|
||||
mc.DumpLogs()
|
||||
t.Fatalf("cluster did not come back after a restart: %v", err)
|
||||
}
|
||||
if after != before {
|
||||
mc.DumpLogs()
|
||||
t.Fatalf("-raftBootstrap re-seeded the cluster: TopologyId %s became %s", before, after)
|
||||
}
|
||||
|
||||
// The leader answers for the whole cluster, so a follower that
|
||||
// forked is only visible in its own log.
|
||||
seen, err := mc.WaitForNodeTopologyIds(waitTimeout)
|
||||
if err != nil {
|
||||
mc.DumpLogs()
|
||||
t.Fatal(err)
|
||||
}
|
||||
for i, ids := range seen {
|
||||
for _, id := range ids {
|
||||
if id != before {
|
||||
mc.DumpLogs()
|
||||
t.Fatalf("master %d saw TopologyId %s, want %s", i, id, before)
|
||||
}
|
||||
}
|
||||
if mc.LogContains(i, "Split-brain detected") {
|
||||
mc.DumpLogs()
|
||||
t.Fatalf("master %d hit the split-brain guard", i)
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
@@ -15,6 +15,7 @@ import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/operation"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/volume_server_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/volume_info"
|
||||
"google.golang.org/grpc"
|
||||
"google.golang.org/grpc/credentials/insecure"
|
||||
)
|
||||
@@ -339,7 +340,14 @@ func (v *VolumeServer) VolumeEcShardsInfo(ctx context.Context, req *volume_serve
|
||||
}
|
||||
}
|
||||
|
||||
resp := &volume_server_pb.VolumeEcShardsInfoResponse{}
|
||||
// Answer with the layout out of the .vif that was actually delivered here,
|
||||
// the way a real holder answers from the context it mounted the shards
|
||||
// with. A coordinator uses this to tell a server that understands the
|
||||
// shard block layout from one that never knew the field, so a fake that
|
||||
// always reported "unset" would look like a pre-upgrade server.
|
||||
resp := &volume_server_pb.VolumeEcShardsInfoResponse{
|
||||
EcShardConfig: v.ecShardConfigFromVif(req.VolumeId),
|
||||
}
|
||||
prefix := fmt.Sprintf("%d.ec", req.VolumeId)
|
||||
entries, _ := os.ReadDir(v.baseDir)
|
||||
for _, entry := range entries {
|
||||
@@ -372,6 +380,16 @@ func (v *VolumeServer) VolumeEcShardsInfo(ctx context.Context, req *volume_serve
|
||||
return resp, nil
|
||||
}
|
||||
|
||||
// ecShardConfigFromVif reads the EC layout out of the .vif this server was
|
||||
// given, which distribution ships to every holder alongside its shards.
|
||||
func (v *VolumeServer) ecShardConfigFromVif(volumeID uint32) *volume_server_pb.EcShardConfig {
|
||||
vi, _, found, err := volume_info.MaybeLoadVolumeInfo(v.filePath(volumeID, ".vif"))
|
||||
if err != nil || !found {
|
||||
return nil
|
||||
}
|
||||
return vi.GetEcShardConfig()
|
||||
}
|
||||
|
||||
func (v *VolumeServer) VolumeDelete(ctx context.Context, req *volume_server_pb.VolumeDeleteRequest) (*volume_server_pb.VolumeDeleteResponse, error) {
|
||||
v.mu.Lock()
|
||||
v.deleteRequests = append(v.deleteRequests, req)
|
||||
|
||||
@@ -156,8 +156,9 @@ func TestRenameObjectSourceIfMatch(t *testing.T) {
|
||||
assert.True(t, objectExists(t, client, bucketName, "target.txt"))
|
||||
}
|
||||
|
||||
// TestRenameObjectOntoDirectory: a key that already holds other objects is a
|
||||
// directory, and an object must not be allowed to replace one.
|
||||
// TestRenameObjectOntoDirectory: S3 keys are flat, so a key that other keys are
|
||||
// nested under is still a key of its own. The rename writes it without disturbing
|
||||
// them - it does not replace the directory, it stores the object on it.
|
||||
func TestRenameObjectOntoDirectory(t *testing.T) {
|
||||
client := getS3Client(t)
|
||||
bucketName := getNewBucketName()
|
||||
@@ -172,9 +173,10 @@ func TestRenameObjectOntoDirectory(t *testing.T) {
|
||||
Key: aws.String("target"),
|
||||
RenameSource: aws.String(createRenameSource("source.txt")),
|
||||
})
|
||||
requireRenameStatus(t, err, 409)
|
||||
assert.True(t, objectExists(t, client, bucketName, "source.txt"))
|
||||
assert.True(t, objectExists(t, client, bucketName, "target/child.txt"))
|
||||
require.NoError(t, err)
|
||||
assert.False(t, objectExists(t, client, bucketName, "source.txt"))
|
||||
assert.Equal(t, "content", getObjectBody(t, getObject(t, client, bucketName, "target")))
|
||||
assert.Equal(t, "child", getObjectBody(t, getObject(t, client, bucketName, "target/child.txt")))
|
||||
}
|
||||
|
||||
// TestRenameObjectDirectorySource: a directory can be named without a trailing
|
||||
|
||||
@@ -0,0 +1,546 @@
|
||||
package example
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"io"
|
||||
"net/http"
|
||||
"sort"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/aws/aws-sdk-go/aws"
|
||||
"github.com/aws/aws-sdk-go/aws/awserr"
|
||||
v1credentials "github.com/aws/aws-sdk-go/aws/credentials"
|
||||
v1signer "github.com/aws/aws-sdk-go/aws/signer/v4"
|
||||
"github.com/aws/aws-sdk-go/service/s3"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// TestS3PrefixObjectKeys covers keys that are a strict prefix of other keys: S3's
|
||||
// namespace is flat, so "collision/foo" and "collision/foo/bar" are independent
|
||||
// objects that coexist in either write order.
|
||||
func TestS3PrefixObjectKeys(t *testing.T) {
|
||||
if testing.Short() {
|
||||
t.Skip("Skipping integration test in short mode")
|
||||
}
|
||||
|
||||
cluster, err := startMiniCluster(t)
|
||||
require.NoError(t, err)
|
||||
defer cluster.Stop()
|
||||
|
||||
put := func(t *testing.T, bucket, key string, body []byte) {
|
||||
t.Helper()
|
||||
_, err := cluster.s3Client.PutObject(&s3.PutObjectInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String(key),
|
||||
Body: bytes.NewReader(body),
|
||||
})
|
||||
require.NoError(t, err, "put %s", key)
|
||||
}
|
||||
// read checks both paths a client reaches an object by, since a directory entry
|
||||
// carrying an object is served by neither the directory nor the plain object path
|
||||
// alone.
|
||||
read := func(t *testing.T, bucket, key string, want []byte) {
|
||||
t.Helper()
|
||||
head, err := cluster.s3Client.HeadObject(&s3.HeadObjectInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String(key),
|
||||
})
|
||||
require.NoError(t, err, "head %s", key)
|
||||
assert.Equal(t, int64(len(want)), aws.Int64Value(head.ContentLength), "head %s", key)
|
||||
|
||||
get, err := cluster.s3Client.GetObject(&s3.GetObjectInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String(key),
|
||||
})
|
||||
require.NoError(t, err, "get %s", key)
|
||||
defer get.Body.Close()
|
||||
got, err := io.ReadAll(get.Body)
|
||||
require.NoError(t, err, "read %s", key)
|
||||
assert.Equal(t, want, got, "get %s", key)
|
||||
}
|
||||
// gone checks the key answers as absent rather than lingering on a directory
|
||||
// entry that outlived the object.
|
||||
gone := func(t *testing.T, bucket, key string) {
|
||||
t.Helper()
|
||||
_, err := cluster.s3Client.GetObject(&s3.GetObjectInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String(key),
|
||||
})
|
||||
var missing awserr.RequestFailure
|
||||
require.ErrorAs(t, err, &missing, "get %s", key)
|
||||
assert.Equal(t, http.StatusNotFound, missing.StatusCode(), "get %s", key)
|
||||
|
||||
_, err = cluster.s3Client.HeadObject(&s3.HeadObjectInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String(key),
|
||||
})
|
||||
require.ErrorAs(t, err, &missing, "head %s", key)
|
||||
assert.Equal(t, http.StatusNotFound, missing.StatusCode(), "head %s", key)
|
||||
}
|
||||
listKeys := func(t *testing.T, bucket string) []string {
|
||||
t.Helper()
|
||||
resp, err := cluster.s3Client.ListObjectsV2(&s3.ListObjectsV2Input{Bucket: aws.String(bucket)})
|
||||
require.NoError(t, err)
|
||||
keys := collectKeys(resp.Contents)
|
||||
sort.Strings(keys)
|
||||
return keys
|
||||
}
|
||||
|
||||
body := []byte("prefix object")
|
||||
// Distinct bodies, so a read that resolves to the wrong entry cannot pass.
|
||||
nested := []byte("nested under the prefix object")
|
||||
|
||||
// The reported order: the nested key is written first, so the prefix key has to
|
||||
// land on a path the filer already holds a directory at.
|
||||
t.Run("ChildFirst", func(t *testing.T) {
|
||||
bucket := createTestBucket(t, cluster, "test-prefix-child-first-")
|
||||
put(t, bucket, "collision/foo/bar", nested)
|
||||
put(t, bucket, "collision/foo", body)
|
||||
|
||||
assert.Equal(t, []string{"collision/foo", "collision/foo/bar"}, listKeys(t, bucket))
|
||||
read(t, bucket, "collision/foo", body)
|
||||
read(t, bucket, "collision/foo/bar", nested)
|
||||
})
|
||||
|
||||
// The opposite order used to keep the prefix key's data but hide the key.
|
||||
t.Run("PrefixFirst", func(t *testing.T) {
|
||||
bucket := createTestBucket(t, cluster, "test-prefix-first-")
|
||||
put(t, bucket, "collision/foo", body)
|
||||
put(t, bucket, "collision/foo/bar", nested)
|
||||
|
||||
assert.Equal(t, []string{"collision/foo", "collision/foo/bar"}, listKeys(t, bucket))
|
||||
read(t, bucket, "collision/foo", body)
|
||||
read(t, bucket, "collision/foo/bar", nested)
|
||||
})
|
||||
|
||||
// An empty object leaves no chunks, content or mime behind, so it is the case a
|
||||
// promoted directory carries no other trace of.
|
||||
t.Run("EmptyObject", func(t *testing.T) {
|
||||
bucket := createTestBucket(t, cluster, "test-prefix-empty-")
|
||||
put(t, bucket, "a/foo/bar", nil)
|
||||
put(t, bucket, "a/foo", nil)
|
||||
put(t, bucket, "b/foo", nil)
|
||||
put(t, bucket, "b/foo/bar", nil)
|
||||
|
||||
assert.Equal(t, []string{"a/foo", "a/foo/bar", "b/foo", "b/foo/bar"}, listKeys(t, bucket))
|
||||
read(t, bucket, "a/foo", []byte{})
|
||||
read(t, bucket, "b/foo", []byte{})
|
||||
})
|
||||
|
||||
// The key has no trailing slash, and the keys nested under it still roll up into
|
||||
// their own CommonPrefix.
|
||||
t.Run("Delimiter", func(t *testing.T) {
|
||||
bucket := createTestBucket(t, cluster, "test-prefix-delimiter-")
|
||||
put(t, bucket, "collision/foo/bar", nested)
|
||||
put(t, bucket, "collision/foo", body)
|
||||
put(t, bucket, "collision/other", body)
|
||||
|
||||
resp, err := cluster.s3Client.ListObjectsV2(&s3.ListObjectsV2Input{
|
||||
Bucket: aws.String(bucket),
|
||||
Prefix: aws.String("collision/"),
|
||||
Delimiter: aws.String("/"),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
|
||||
keys := collectKeys(resp.Contents)
|
||||
sort.Strings(keys)
|
||||
assert.Equal(t, []string{"collision/foo", "collision/other"}, keys)
|
||||
assert.Equal(t, []string{"collision/foo/"}, collectPrefixes(resp.CommonPrefixes))
|
||||
|
||||
// Listing the prefix itself names only what is under it.
|
||||
resp, err = cluster.s3Client.ListObjectsV2(&s3.ListObjectsV2Input{
|
||||
Bucket: aws.String(bucket),
|
||||
Prefix: aws.String("collision/foo/"),
|
||||
Delimiter: aws.String("/"),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, []string{"collision/foo/bar"}, collectKeys(resp.Contents))
|
||||
assert.Empty(t, collectPrefixes(resp.CommonPrefixes))
|
||||
})
|
||||
|
||||
// The key and the CommonPrefix its nested keys fold into come off one filer
|
||||
// entry, so a page boundary must not drop either of them.
|
||||
t.Run("Paged", func(t *testing.T) {
|
||||
bucket := createTestBucket(t, cluster, "test-prefix-paged-")
|
||||
for _, key := range []string{"foo", "foo/bar", "foobar", "other", "zed", "zed/a"} {
|
||||
put(t, bucket, key, body)
|
||||
}
|
||||
|
||||
for _, maxKeys := range []int64{1, 2, 3, 4, 5} {
|
||||
var keys, prefixes []string
|
||||
var token *string
|
||||
for page := 0; page < 12; page++ {
|
||||
resp, err := cluster.s3Client.ListObjectsV2(&s3.ListObjectsV2Input{
|
||||
Bucket: aws.String(bucket),
|
||||
Delimiter: aws.String("/"),
|
||||
MaxKeys: aws.Int64(maxKeys),
|
||||
ContinuationToken: token,
|
||||
})
|
||||
require.NoError(t, err)
|
||||
// One entry over the budget is the documented same-entry exception;
|
||||
// anything more means the unsigned budget wrapped.
|
||||
assert.LessOrEqual(t, int64(len(resp.Contents)+len(resp.CommonPrefixes)), maxKeys+1,
|
||||
"maxKeys=%d page %d", maxKeys, page)
|
||||
keys = append(keys, collectKeys(resp.Contents)...)
|
||||
prefixes = append(prefixes, collectPrefixes(resp.CommonPrefixes)...)
|
||||
if !aws.BoolValue(resp.IsTruncated) {
|
||||
token = nil
|
||||
break
|
||||
}
|
||||
token = resp.NextContinuationToken
|
||||
require.NotNil(t, token, "a truncated page must name where to resume")
|
||||
}
|
||||
require.Nil(t, token, "maxKeys=%d did not finish", maxKeys)
|
||||
assert.Equal(t, []string{"foo", "foobar", "other", "zed"}, keys, "maxKeys=%d", maxKeys)
|
||||
assert.Equal(t, []string{"foo/", "zed/"}, prefixes, "maxKeys=%d", maxKeys)
|
||||
}
|
||||
})
|
||||
|
||||
// Versioning reaches a prefix object from two directions: a suspended bucket
|
||||
// writes the null version at the key's own path, and a bucket versioned later
|
||||
// finds one already sitting there. Both leave a key that is a directory with
|
||||
// version history beside it.
|
||||
t.Run("Versioned", func(t *testing.T) {
|
||||
setVersioning := func(t *testing.T, bucket, status string) {
|
||||
t.Helper()
|
||||
_, err := cluster.s3Client.PutBucketVersioning(&s3.PutBucketVersioningInput{
|
||||
Bucket: aws.String(bucket),
|
||||
VersioningConfiguration: &s3.VersioningConfiguration{Status: aws.String(status)},
|
||||
})
|
||||
require.NoError(t, err)
|
||||
}
|
||||
|
||||
// Both write orders, in a bucket that is versioned and in one where versioning
|
||||
// was suspended - the suspended one is the case that writes at the key's path.
|
||||
for _, state := range []string{"Enabled", "Suspended"} {
|
||||
bucket := createTestBucket(t, cluster, "test-prefix-"+strings.ToLower(state)+"-")
|
||||
setVersioning(t, bucket, "Enabled")
|
||||
if state == "Suspended" {
|
||||
setVersioning(t, bucket, "Suspended")
|
||||
}
|
||||
put(t, bucket, "child/foo/bar", nested)
|
||||
put(t, bucket, "child/foo", body)
|
||||
put(t, bucket, "prefix/foo", body)
|
||||
put(t, bucket, "prefix/foo/bar", nested)
|
||||
|
||||
assert.Equal(t, []string{"child/foo", "child/foo/bar", "prefix/foo", "prefix/foo/bar"},
|
||||
listKeys(t, bucket), state)
|
||||
for _, key := range []string{"child/foo", "prefix/foo"} {
|
||||
read(t, bucket, key, body)
|
||||
}
|
||||
for _, key := range []string{"child/foo/bar", "prefix/foo/bar"} {
|
||||
read(t, bucket, key, nested)
|
||||
}
|
||||
}
|
||||
|
||||
// A prefix object written before versioning is the key's null version. Removing
|
||||
// that version by id must not take the keys nested under it with it.
|
||||
bucket := createTestBucket(t, cluster, "test-prefix-nullversion-")
|
||||
put(t, bucket, "collision/foo/bar", nested)
|
||||
put(t, bucket, "collision/foo", body)
|
||||
setVersioning(t, bucket, "Enabled")
|
||||
newer := []byte("written after versioning was enabled")
|
||||
versioned, err := cluster.s3Client.PutObject(&s3.PutObjectInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String("collision/foo"),
|
||||
Body: bytes.NewReader(newer),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
|
||||
for _, v := range []struct {
|
||||
id string
|
||||
want []byte
|
||||
}{{"null", body}, {aws.StringValue(versioned.VersionId), newer}} {
|
||||
got, err := cluster.s3Client.GetObject(&s3.GetObjectInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String("collision/foo"),
|
||||
VersionId: aws.String(v.id),
|
||||
})
|
||||
require.NoError(t, err, "get version %s", v.id)
|
||||
body, err := io.ReadAll(got.Body)
|
||||
require.NoError(t, err)
|
||||
got.Body.Close()
|
||||
assert.Equal(t, v.want, body, "get version %s", v.id)
|
||||
}
|
||||
|
||||
_, err = cluster.s3Client.DeleteObject(&s3.DeleteObjectInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String("collision/foo"),
|
||||
VersionId: aws.String("null"),
|
||||
})
|
||||
require.NoError(t, err, "the null version sits on a directory other keys live in")
|
||||
|
||||
read(t, bucket, "collision/foo", newer)
|
||||
read(t, bucket, "collision/foo/bar", nested)
|
||||
remaining, err := cluster.s3Client.ListObjectVersions(&s3.ListObjectVersionsInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Prefix: aws.String("collision/foo"),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
for _, v := range remaining.Versions {
|
||||
if aws.StringValue(v.Key) != "collision/foo" {
|
||||
// collision/foo/bar predates versioning too, and keeps its null version.
|
||||
continue
|
||||
}
|
||||
assert.NotEqual(t, "null", aws.StringValue(v.VersionId), "the null version was deleted")
|
||||
}
|
||||
})
|
||||
|
||||
// The two listings walk the tree differently, and a prefix object is the entry
|
||||
// they disagree about: it is a directory the version listing descends through and
|
||||
// a key at the same time. They have to name the same keys and the same prefixes.
|
||||
t.Run("VersionListingMatchesObjectListing", func(t *testing.T) {
|
||||
bucket := createTestBucket(t, cluster, "test-prefix-parity-")
|
||||
for _, key := range []string{"foo", "foo/bar", "other", "a/foo", "a/foo/bar", "a/z"} {
|
||||
put(t, bucket, key, body)
|
||||
}
|
||||
|
||||
for _, q := range []struct{ prefix, delimiter string }{
|
||||
{"", ""},
|
||||
{"foo/", ""},
|
||||
{"a/", ""},
|
||||
{"a/foo/", ""},
|
||||
{"", "/"},
|
||||
{"a/", "/"},
|
||||
} {
|
||||
name := "prefix=" + q.prefix + " delimiter=" + q.delimiter
|
||||
|
||||
objects, err := cluster.s3Client.ListObjectsV2(&s3.ListObjectsV2Input{
|
||||
Bucket: aws.String(bucket),
|
||||
Prefix: aws.String(q.prefix),
|
||||
Delimiter: aws.String(q.delimiter),
|
||||
})
|
||||
require.NoError(t, err, name)
|
||||
|
||||
versions, err := cluster.s3Client.ListObjectVersions(&s3.ListObjectVersionsInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Prefix: aws.String(q.prefix),
|
||||
Delimiter: aws.String(q.delimiter),
|
||||
})
|
||||
require.NoError(t, err, name)
|
||||
|
||||
versionKeys := make([]string, 0, len(versions.Versions))
|
||||
for _, v := range versions.Versions {
|
||||
versionKeys = append(versionKeys, aws.StringValue(v.Key))
|
||||
}
|
||||
versionPrefixes := make([]string, 0, len(versions.CommonPrefixes))
|
||||
for _, p := range versions.CommonPrefixes {
|
||||
versionPrefixes = append(versionPrefixes, aws.StringValue(p.Prefix))
|
||||
}
|
||||
sort.Strings(versionKeys)
|
||||
sort.Strings(versionPrefixes)
|
||||
|
||||
objectKeys := collectKeys(objects.Contents)
|
||||
objectPrefixes := collectPrefixes(objects.CommonPrefixes)
|
||||
sort.Strings(objectKeys)
|
||||
sort.Strings(objectPrefixes)
|
||||
|
||||
assert.Equal(t, objectKeys, versionKeys, "keys, %s", name)
|
||||
assert.Equal(t, objectPrefixes, versionPrefixes, "prefixes, %s", name)
|
||||
}
|
||||
})
|
||||
|
||||
// A prefix object written before versioning is the key's null version, so the
|
||||
// version written after it has to take the latest flag off it.
|
||||
t.Run("VersionedAfterPrefixObject", func(t *testing.T) {
|
||||
bucket := createTestBucket(t, cluster, "test-prefix-versioned-")
|
||||
put(t, bucket, "collision/foo/bar", nested)
|
||||
put(t, bucket, "collision/foo", body)
|
||||
_, err := cluster.s3Client.PutBucketVersioning(&s3.PutBucketVersioningInput{
|
||||
Bucket: aws.String(bucket),
|
||||
VersioningConfiguration: &s3.VersioningConfiguration{Status: aws.String("Enabled")},
|
||||
})
|
||||
require.NoError(t, err)
|
||||
newer := []byte("written after versioning was enabled")
|
||||
put(t, bucket, "collision/foo", newer)
|
||||
|
||||
resp, err := cluster.s3Client.ListObjectVersions(&s3.ListObjectVersionsInput{Bucket: aws.String(bucket)})
|
||||
require.NoError(t, err)
|
||||
|
||||
latest := map[string]int{}
|
||||
var nullSize int64 = -1
|
||||
for _, v := range resp.Versions {
|
||||
if aws.BoolValue(v.IsLatest) {
|
||||
latest[aws.StringValue(v.Key)]++
|
||||
}
|
||||
if aws.StringValue(v.Key) == "collision/foo" && aws.StringValue(v.VersionId) == "null" {
|
||||
nullSize = aws.Int64Value(v.Size)
|
||||
assert.False(t, aws.BoolValue(v.IsLatest), "the newer version is the latest one")
|
||||
}
|
||||
}
|
||||
assert.Equal(t, 1, latest["collision/foo"], "exactly one version of a key is the latest")
|
||||
assert.Equal(t, int64(len(body)), nullSize, "the null version keeps the prefix object's size")
|
||||
read(t, bucket, "collision/foo", newer)
|
||||
read(t, bucket, "collision/foo/bar", nested)
|
||||
})
|
||||
|
||||
// A key that other keys are nested under is a copy source and a copy destination
|
||||
// like any other. The keys nested under either end are not part of the copy.
|
||||
t.Run("Copy", func(t *testing.T) {
|
||||
bucket := createTestBucket(t, cluster, "test-prefix-copy-")
|
||||
copyObject := func(t *testing.T, src, dst string) {
|
||||
t.Helper()
|
||||
_, err := cluster.s3Client.CopyObject(&s3.CopyObjectInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String(dst),
|
||||
CopySource: aws.String(bucket + "/" + src),
|
||||
})
|
||||
require.NoError(t, err, "copy %s to %s", src, dst)
|
||||
}
|
||||
|
||||
put(t, bucket, "collision/foo", body)
|
||||
put(t, bucket, "collision/foo/bar", nested)
|
||||
|
||||
// Out of a prefix object, into a key of its own.
|
||||
copyObject(t, "collision/foo", "plain")
|
||||
read(t, bucket, "plain", body)
|
||||
read(t, bucket, "collision/foo", body)
|
||||
read(t, bucket, "collision/foo/bar", nested)
|
||||
|
||||
// Into a key that other keys are nested under.
|
||||
put(t, bucket, "target/child", nested)
|
||||
copyObject(t, "plain", "target")
|
||||
read(t, bucket, "target", body)
|
||||
read(t, bucket, "target/child", nested)
|
||||
|
||||
// And between two of them.
|
||||
put(t, bucket, "other", []byte("copied between prefix keys"))
|
||||
copyObject(t, "other", "collision/foo")
|
||||
read(t, bucket, "collision/foo", []byte("copied between prefix keys"))
|
||||
read(t, bucket, "collision/foo/bar", nested)
|
||||
|
||||
assert.Equal(t, []string{"collision/foo", "collision/foo/bar", "other", "plain", "target", "target/child"},
|
||||
listKeys(t, bucket))
|
||||
})
|
||||
|
||||
// Rename moves the object off the key without moving the keys nested under it,
|
||||
// which is not what the filer's atomic rename of a directory would do.
|
||||
t.Run("Rename", func(t *testing.T) {
|
||||
bucket := createTestBucket(t, cluster, "test-prefix-rename-")
|
||||
renameObject := func(t *testing.T, src, dst string) {
|
||||
t.Helper()
|
||||
req, _ := http.NewRequest(http.MethodPut, cluster.s3Endpoint+"/"+bucket+"/"+dst+"?renameObject=", nil)
|
||||
req.Header.Set("x-amz-rename-source", "/"+bucket+"/"+src)
|
||||
signer := v1signer.NewSigner(v1credentials.NewStaticCredentials(testAccessKey, testSecretKey, ""))
|
||||
_, err := signer.Sign(req, nil, "s3", testRegion, time.Now())
|
||||
require.NoError(t, err)
|
||||
|
||||
resp, err := (&http.Client{Timeout: 20 * time.Second}).Do(req)
|
||||
require.NoError(t, err, "rename %s to %s", src, dst)
|
||||
defer resp.Body.Close()
|
||||
io.Copy(io.Discard, resp.Body)
|
||||
require.Equal(t, http.StatusOK, resp.StatusCode, "rename %s to %s", src, dst)
|
||||
}
|
||||
|
||||
put(t, bucket, "collision/foo", body)
|
||||
put(t, bucket, "collision/foo/bar", nested)
|
||||
|
||||
// Off a prefix object: the key goes, the keys under it stay.
|
||||
renameObject(t, "collision/foo", "moved")
|
||||
read(t, bucket, "moved", body)
|
||||
read(t, bucket, "collision/foo/bar", nested)
|
||||
gone(t, bucket, "collision/foo")
|
||||
|
||||
// Onto a key other keys are nested under.
|
||||
put(t, bucket, "target/child", nested)
|
||||
renameObject(t, "moved", "target")
|
||||
read(t, bucket, "target", body)
|
||||
read(t, bucket, "target/child", nested)
|
||||
gone(t, bucket, "moved")
|
||||
|
||||
assert.Equal(t, []string{"collision/foo/bar", "target", "target/child"}, listKeys(t, bucket))
|
||||
})
|
||||
|
||||
// A directory SeaweedFS keeps its own state in is not a prefix a key can be
|
||||
// stored on: the object would replace that state with its own.
|
||||
t.Run("ReservedDirectory", func(t *testing.T) {
|
||||
bucket := createTestBucket(t, cluster, "test-prefix-reserved-")
|
||||
_, err := cluster.s3Client.PutBucketVersioning(&s3.PutBucketVersioningInput{
|
||||
Bucket: aws.String(bucket),
|
||||
VersioningConfiguration: &s3.VersioningConfiguration{Status: aws.String("Enabled")},
|
||||
})
|
||||
require.NoError(t, err)
|
||||
put(t, bucket, "foo", body)
|
||||
_, err = cluster.s3Client.PutBucketVersioning(&s3.PutBucketVersioningInput{
|
||||
Bucket: aws.String(bucket),
|
||||
VersioningConfiguration: &s3.VersioningConfiguration{Status: aws.String("Suspended")},
|
||||
})
|
||||
require.NoError(t, err)
|
||||
|
||||
_, err = cluster.s3Client.PutObject(&s3.PutObjectInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String("foo.versions"),
|
||||
Body: bytes.NewReader(body),
|
||||
})
|
||||
// Not just any error: a permanently impossible write must not come back as a
|
||||
// 500 the SDK retries.
|
||||
var refused awserr.RequestFailure
|
||||
require.ErrorAs(t, err, &refused, "the version history of foo is not a prefix of foo.versions")
|
||||
assert.Equal(t, http.StatusConflict, refused.StatusCode())
|
||||
assert.Equal(t, "ExistingObjectIsDirectory", refused.Code())
|
||||
|
||||
versions, err := cluster.s3Client.ListObjectVersions(&s3.ListObjectVersionsInput{Bucket: aws.String(bucket)})
|
||||
require.NoError(t, err)
|
||||
require.Len(t, versions.Versions, 1)
|
||||
assert.Equal(t, "foo", aws.StringValue(versions.Versions[0].Key))
|
||||
read(t, bucket, "foo", body)
|
||||
|
||||
// The multipart staging folder is the other one, and an in-flight upload has
|
||||
// to survive the attempt.
|
||||
staging := createTestBucket(t, cluster, "test-prefix-uploads-")
|
||||
created, err := cluster.s3Client.CreateMultipartUpload(&s3.CreateMultipartUploadInput{
|
||||
Bucket: aws.String(staging),
|
||||
Key: aws.String("mp.bin"),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
|
||||
_, err = cluster.s3Client.PutObject(&s3.PutObjectInput{
|
||||
Bucket: aws.String(staging),
|
||||
Key: aws.String(".uploads"),
|
||||
Body: bytes.NewReader(body),
|
||||
})
|
||||
require.ErrorAs(t, err, &refused, "the multipart staging folder is not a prefix of .uploads")
|
||||
assert.Equal(t, http.StatusConflict, refused.StatusCode())
|
||||
assert.Equal(t, "ExistingObjectIsDirectory", refused.Code())
|
||||
|
||||
uploads, err := cluster.s3Client.ListMultipartUploads(&s3.ListMultipartUploadsInput{Bucket: aws.String(staging)})
|
||||
require.NoError(t, err)
|
||||
require.Len(t, uploads.Uploads, 1)
|
||||
assert.Equal(t, "mp.bin", aws.StringValue(uploads.Uploads[0].Key))
|
||||
_, err = cluster.s3Client.AbortMultipartUpload(&s3.AbortMultipartUploadInput{
|
||||
Bucket: aws.String(staging),
|
||||
Key: aws.String("mp.bin"),
|
||||
UploadId: created.UploadId,
|
||||
})
|
||||
require.NoError(t, err)
|
||||
})
|
||||
|
||||
// Either key can be deleted without touching the other.
|
||||
t.Run("Delete", func(t *testing.T) {
|
||||
bucket := createTestBucket(t, cluster, "test-prefix-delete-")
|
||||
put(t, bucket, "collision/foo/bar", nested)
|
||||
put(t, bucket, "collision/foo", body)
|
||||
|
||||
_, err := cluster.s3Client.DeleteObject(&s3.DeleteObjectInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String("collision/foo"),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, []string{"collision/foo/bar"}, listKeys(t, bucket))
|
||||
read(t, bucket, "collision/foo/bar", nested)
|
||||
gone(t, bucket, "collision/foo")
|
||||
|
||||
put(t, bucket, "collision/foo", body)
|
||||
_, err = cluster.s3Client.DeleteObject(&s3.DeleteObjectInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String("collision/foo/bar"),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, []string{"collision/foo"}, listKeys(t, bucket))
|
||||
read(t, bucket, "collision/foo", body)
|
||||
gone(t, bucket, "collision/foo/bar")
|
||||
})
|
||||
}
|
||||
@@ -26,16 +26,16 @@ nginx (:9000)
|
||||
v
|
||||
SeaweedFS S3 (:8333, -s3.externalUrl=http://localhost:9000)
|
||||
| externalHost = "localhost:9000" (parsed at startup)
|
||||
| extractHostHeader() returns "localhost:9000"
|
||||
| extractHostHeaderCandidates() tries "localhost:9000" first
|
||||
| Matches what AWS CLI signed with
|
||||
v
|
||||
Signature verification succeeds
|
||||
```
|
||||
|
||||
**Note:** When `-s3.externalUrl` is configured, direct access to the backend
|
||||
port (8333) will fail signature verification because the client signs with a
|
||||
different Host header than what `externalUrl` specifies. This is expected —
|
||||
all S3 traffic should go through the proxy.
|
||||
**Note:** `-s3.externalUrl` is tried first, not exclusively. A client that
|
||||
dials the backend port (8333) directly still verifies against the host it
|
||||
actually signed, so a mixed topology of proxied and in-cluster clients works
|
||||
with the flag set.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
|
||||
@@ -0,0 +1,231 @@
|
||||
package retention
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/aws/aws-sdk-go-v2/aws"
|
||||
"github.com/aws/aws-sdk-go-v2/service/s3"
|
||||
"github.com/aws/aws-sdk-go-v2/service/s3/types"
|
||||
"github.com/aws/smithy-go"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
func requireAccessDenied(t *testing.T, err error, msg string) {
|
||||
t.Helper()
|
||||
require.Error(t, err, msg)
|
||||
var apiErr smithy.APIError
|
||||
require.True(t, errors.As(err, &apiErr), "expected an API error, got %T", err)
|
||||
assert.Equal(t, "AccessDenied", apiErr.ErrorCode(), msg)
|
||||
}
|
||||
|
||||
// A key ending in "/" is stored as the filer directory rather than as an object
|
||||
// beside it, and is deleted the unversioned way. Object Lock still covers it: the
|
||||
// gateway lists it as an object and serves retention set on it.
|
||||
func TestObjectLockDirectoryMarker(t *testing.T) {
|
||||
client := getS3Client(t)
|
||||
bucketName := getNewBucketName()
|
||||
|
||||
createBucketWithObjectLock(t, client, bucketName)
|
||||
defer deleteBucket(t, client, bucketName)
|
||||
|
||||
retainUntil := time.Now().Add(24 * time.Hour)
|
||||
|
||||
t.Run("retention headers are honored, not dropped", func(t *testing.T) {
|
||||
key := "records/evidence/"
|
||||
_, err := client.PutObject(context.TODO(), &s3.PutObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String(key),
|
||||
Body: strings.NewReader("marker"),
|
||||
ObjectLockMode: types.ObjectLockModeCompliance,
|
||||
ObjectLockRetainUntilDate: aws.Time(retainUntil),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
|
||||
_, err = client.DeleteObject(context.TODO(), &s3.DeleteObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String(key),
|
||||
})
|
||||
requireAccessDenied(t, err, "a retained marker must not be deletable")
|
||||
|
||||
_, err = client.HeadObject(context.TODO(), &s3.HeadObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String(key),
|
||||
})
|
||||
assert.NoError(t, err, "the marker must still be there after the refused delete")
|
||||
})
|
||||
|
||||
t.Run("retention set through PutObjectRetention is honored", func(t *testing.T) {
|
||||
key := "records/ledger/"
|
||||
_, err := client.PutObject(context.TODO(), &s3.PutObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String(key),
|
||||
Body: strings.NewReader("marker"),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
|
||||
_, err = client.PutObjectRetention(context.TODO(), &s3.PutObjectRetentionInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String(key),
|
||||
Retention: &types.ObjectLockRetention{
|
||||
Mode: types.ObjectLockRetentionModeCompliance,
|
||||
RetainUntilDate: aws.Time(retainUntil),
|
||||
},
|
||||
})
|
||||
require.NoError(t, err)
|
||||
|
||||
_, err = client.DeleteObject(context.TODO(), &s3.DeleteObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String(key),
|
||||
})
|
||||
requireAccessDenied(t, err, "retention the gateway serves back must also block the delete")
|
||||
|
||||
_, err = client.HeadObject(context.TODO(), &s3.HeadObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String(key),
|
||||
})
|
||||
assert.NoError(t, err, "the marker must still be there after the refused delete")
|
||||
})
|
||||
|
||||
t.Run("multi-object delete is refused too", func(t *testing.T) {
|
||||
key := "records/batch/"
|
||||
_, err := client.PutObject(context.TODO(), &s3.PutObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String(key),
|
||||
Body: strings.NewReader("marker"),
|
||||
ObjectLockMode: types.ObjectLockModeCompliance,
|
||||
ObjectLockRetainUntilDate: aws.Time(retainUntil),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
|
||||
resp, err := client.DeleteObjects(context.TODO(), &s3.DeleteObjectsInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Delete: &types.Delete{Objects: []types.ObjectIdentifier{{Key: aws.String(key)}}},
|
||||
})
|
||||
require.NoError(t, err)
|
||||
assert.Empty(t, resp.Deleted, "a retained marker must not be reported deleted")
|
||||
require.Len(t, resp.Errors, 1)
|
||||
assert.Equal(t, key, aws.ToString(resp.Errors[0].Key))
|
||||
assert.Equal(t, "AccessDenied", aws.ToString(resp.Errors[0].Code))
|
||||
|
||||
_, err = client.HeadObject(context.TODO(), &s3.HeadObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String(key),
|
||||
})
|
||||
assert.NoError(t, err, "the marker must survive the batch delete")
|
||||
})
|
||||
|
||||
t.Run("invalid lock headers are rejected, as on a regular key", func(t *testing.T) {
|
||||
_, err := client.PutObject(context.TODO(), &s3.PutObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String("records/bad-mode/"),
|
||||
Body: strings.NewReader("marker"),
|
||||
ObjectLockMode: "INVALID_MODE",
|
||||
ObjectLockRetainUntilDate: aws.Time(retainUntil),
|
||||
})
|
||||
require.Error(t, err)
|
||||
|
||||
_, err = client.PutObject(context.TODO(), &s3.PutObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String("records/no-date/"),
|
||||
Body: strings.NewReader("marker"),
|
||||
ObjectLockMode: types.ObjectLockModeGovernance,
|
||||
})
|
||||
require.Error(t, err)
|
||||
})
|
||||
|
||||
// Past 1KiB a trailing-slash key is a real versioned object, and the delete
|
||||
// takes the whole history at once, so an older retained version has to block
|
||||
// it even when the version on top carries no retention of its own.
|
||||
t.Run("a retained version under an unretained one still blocks", func(t *testing.T) {
|
||||
key := "records/history/"
|
||||
body := strings.Repeat("x", 2048)
|
||||
|
||||
first, err := client.PutObject(context.TODO(), &s3.PutObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String(key),
|
||||
Body: strings.NewReader(body),
|
||||
ObjectLockMode: types.ObjectLockModeCompliance,
|
||||
ObjectLockRetainUntilDate: aws.Time(retainUntil),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
require.NotNil(t, first.VersionId)
|
||||
|
||||
_, err = client.PutObject(context.TODO(), &s3.PutObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String(key),
|
||||
Body: strings.NewReader(body),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
|
||||
_, err = client.DeleteObject(context.TODO(), &s3.DeleteObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String(key),
|
||||
})
|
||||
requireAccessDenied(t, err, "the retained version underneath must block the delete")
|
||||
|
||||
_, err = client.HeadObject(context.TODO(), &s3.HeadObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String(key),
|
||||
VersionId: first.VersionId,
|
||||
})
|
||||
assert.NoError(t, err, "the retained version must survive")
|
||||
})
|
||||
|
||||
// mkdir replaces the marker entry outright, taking its lock metadata with it,
|
||||
// so a plain PUT over a retained marker has to be refused - including once the
|
||||
// key has grown a version history that the latest-version lookup would find
|
||||
// instead of the marker.
|
||||
t.Run("a plain PUT cannot replace a retained marker", func(t *testing.T) {
|
||||
key := "records/overwrite/"
|
||||
_, err := client.PutObject(context.TODO(), &s3.PutObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String(key),
|
||||
Body: strings.NewReader("marker"),
|
||||
ObjectLockMode: types.ObjectLockModeCompliance,
|
||||
ObjectLockRetainUntilDate: aws.Time(retainUntil),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
|
||||
_, err = client.PutObject(context.TODO(), &s3.PutObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String(key),
|
||||
Body: strings.NewReader("replacement"),
|
||||
})
|
||||
requireAccessDenied(t, err, "a retained marker must not be replaceable")
|
||||
|
||||
_, err = client.PutObject(context.TODO(), &s3.PutObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String(key),
|
||||
Body: strings.NewReader(strings.Repeat("x", 2048)),
|
||||
})
|
||||
require.NoError(t, err, "a versioned write of the same key adds a version")
|
||||
|
||||
_, err = client.PutObject(context.TODO(), &s3.PutObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String(key),
|
||||
Body: strings.NewReader("replacement"),
|
||||
})
|
||||
requireAccessDenied(t, err, "the history must not hide the marker's own lock")
|
||||
})
|
||||
|
||||
t.Run("an unretained marker still deletes", func(t *testing.T) {
|
||||
key := "records/plain/"
|
||||
_, err := client.PutObject(context.TODO(), &s3.PutObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String(key),
|
||||
Body: strings.NewReader("marker"),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
|
||||
_, err = client.DeleteObject(context.TODO(), &s3.DeleteObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String(key),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
})
|
||||
}
|
||||
@@ -0,0 +1,130 @@
|
||||
package s3api
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/aws/aws-sdk-go-v2/aws"
|
||||
"github.com/aws/aws-sdk-go-v2/service/s3"
|
||||
"github.com/aws/smithy-go"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
func requireAPIErrorCode(t *testing.T, err error, expected string) {
|
||||
t.Helper()
|
||||
require.Error(t, err)
|
||||
var apiErr smithy.APIError
|
||||
require.True(t, errors.As(err, &apiErr), "expected a smithy.APIError, got %T: %v", err, err)
|
||||
assert.Equal(t, expected, apiErr.ErrorCode())
|
||||
}
|
||||
|
||||
// TestConditionalReadsOfMissingObject verifies that a missing key stays a missing key
|
||||
// under If-Match and If-Unmodified-Since instead of surfacing as 412.
|
||||
// reproduces issue #10984
|
||||
func TestConditionalReadsOfMissingObject(t *testing.T) {
|
||||
client := getS3Client(t)
|
||||
bucketName := getNewBucketName()
|
||||
|
||||
createBucket(t, client, bucketName)
|
||||
defer deleteBucket(t, client, bucketName)
|
||||
|
||||
existing := putObject(t, client, bucketName, "etag-source", "content")
|
||||
require.NotNil(t, existing.ETag)
|
||||
future := aws.Time(time.Now().Add(24 * time.Hour))
|
||||
missing := aws.String("conditional-missing")
|
||||
|
||||
t.Run("HeadObject If-Match", func(t *testing.T) {
|
||||
_, err := client.HeadObject(context.TODO(), &s3.HeadObjectInput{
|
||||
Bucket: aws.String(bucketName), Key: missing, IfMatch: existing.ETag,
|
||||
})
|
||||
requireAPIErrorCode(t, err, "NotFound")
|
||||
})
|
||||
|
||||
t.Run("HeadObject If-Unmodified-Since", func(t *testing.T) {
|
||||
_, err := client.HeadObject(context.TODO(), &s3.HeadObjectInput{
|
||||
Bucket: aws.String(bucketName), Key: missing, IfUnmodifiedSince: future,
|
||||
})
|
||||
requireAPIErrorCode(t, err, "NotFound")
|
||||
})
|
||||
|
||||
t.Run("GetObject If-Match", func(t *testing.T) {
|
||||
_, err := client.GetObject(context.TODO(), &s3.GetObjectInput{
|
||||
Bucket: aws.String(bucketName), Key: missing, IfMatch: existing.ETag,
|
||||
})
|
||||
requireAPIErrorCode(t, err, "NoSuchKey")
|
||||
})
|
||||
|
||||
t.Run("GetObject If-Unmodified-Since", func(t *testing.T) {
|
||||
_, err := client.GetObject(context.TODO(), &s3.GetObjectInput{
|
||||
Bucket: aws.String(bucketName), Key: missing, IfUnmodifiedSince: future,
|
||||
})
|
||||
requireAPIErrorCode(t, err, "NoSuchKey")
|
||||
})
|
||||
|
||||
t.Run("GetObject stale If-Match on a live object stays 412", func(t *testing.T) {
|
||||
_, err := client.GetObject(context.TODO(), &s3.GetObjectInput{
|
||||
Bucket: aws.String(bucketName), Key: aws.String("etag-source"),
|
||||
IfMatch: aws.String(`"0000000000000000000000000000dead"`),
|
||||
})
|
||||
requireAPIErrorCode(t, err, "PreconditionFailed")
|
||||
})
|
||||
}
|
||||
|
||||
// TestConditionalReadsOfNamedVersion verifies that a conditional GET or HEAD of an
|
||||
// explicit versionId is evaluated against that version rather than the latest one,
|
||||
// including when the latest version is a delete marker.
|
||||
func TestConditionalReadsOfNamedVersion(t *testing.T) {
|
||||
client := getS3Client(t)
|
||||
bucketName := getNewBucketName()
|
||||
|
||||
createBucket(t, client, bucketName)
|
||||
defer deleteBucket(t, client, bucketName)
|
||||
enableVersioning(t, client, bucketName)
|
||||
|
||||
key := "conditional-read-version"
|
||||
v1 := putObject(t, client, bucketName, key, "content-v1")
|
||||
require.NotNil(t, v1.ETag)
|
||||
require.NotNil(t, v1.VersionId)
|
||||
v2 := putObject(t, client, bucketName, key, "content-v2")
|
||||
require.NotNil(t, v2.ETag)
|
||||
require.NotEqual(t, *v1.ETag, *v2.ETag)
|
||||
|
||||
t.Run("If-Match matches the named version, not the latest", func(t *testing.T) {
|
||||
_, err := client.GetObject(context.TODO(), &s3.GetObjectInput{
|
||||
Bucket: aws.String(bucketName), Key: aws.String(key),
|
||||
VersionId: v1.VersionId, IfMatch: v1.ETag,
|
||||
})
|
||||
require.NoError(t, err)
|
||||
})
|
||||
|
||||
t.Run("If-Match against the latest ETag fails on the named version", func(t *testing.T) {
|
||||
_, err := client.GetObject(context.TODO(), &s3.GetObjectInput{
|
||||
Bucket: aws.String(bucketName), Key: aws.String(key),
|
||||
VersionId: v1.VersionId, IfMatch: v2.ETag,
|
||||
})
|
||||
requireAPIErrorCode(t, err, "PreconditionFailed")
|
||||
})
|
||||
|
||||
_, err := client.DeleteObject(context.TODO(), &s3.DeleteObjectInput{
|
||||
Bucket: aws.String(bucketName), Key: aws.String(key),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
|
||||
t.Run("named version survives a delete marker on the latest", func(t *testing.T) {
|
||||
_, err := client.HeadObject(context.TODO(), &s3.HeadObjectInput{
|
||||
Bucket: aws.String(bucketName), Key: aws.String(key),
|
||||
VersionId: v1.VersionId, IfMatch: v1.ETag,
|
||||
})
|
||||
require.NoError(t, err)
|
||||
})
|
||||
|
||||
t.Run("delete marker latest is a missing object", func(t *testing.T) {
|
||||
_, err := client.GetObject(context.TODO(), &s3.GetObjectInput{
|
||||
Bucket: aws.String(bucketName), Key: aws.String(key), IfMatch: v1.ETag,
|
||||
})
|
||||
requireAPIErrorCode(t, err, "NoSuchKey")
|
||||
})
|
||||
}
|
||||
@@ -1,6 +1,7 @@
|
||||
package tus
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/base64"
|
||||
@@ -996,3 +997,109 @@ func TestTusAbortedPatchKeepsStoredChunks(t *testing.T) {
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, testData, body, "the resumed upload should read back whole")
|
||||
}
|
||||
|
||||
// TestTusConcurrentPatchRefused checks that a PATCH sent while another PATCH
|
||||
// on the same session is still consuming its body is refused with 423 Locked.
|
||||
// Before the per-session claim, both were accepted at the same offset, recorded
|
||||
// the range twice, and completion failed on the duplicate while HEAD reported
|
||||
// the upload fully received - the file was never created and every stored byte
|
||||
// became garbage when the session expired.
|
||||
func TestTusConcurrentPatchRefused(t *testing.T) {
|
||||
if testing.Short() {
|
||||
t.Skip("Skipping integration test in short mode")
|
||||
}
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 180*time.Second)
|
||||
defer cancel()
|
||||
|
||||
cluster, err := startTestCluster(t, ctx)
|
||||
require.NoError(t, err)
|
||||
defer func() {
|
||||
cluster.Stop()
|
||||
os.RemoveAll(cluster.dataDir)
|
||||
}()
|
||||
|
||||
const subChunkSize = 4 * 1024 * 1024
|
||||
testData := make([]byte, 3*subChunkSize)
|
||||
for i := range testData {
|
||||
testData[i] = byte(i % 251)
|
||||
}
|
||||
targetPath := "/raced/video.bin"
|
||||
client := &http.Client{}
|
||||
|
||||
createReq, err := http.NewRequest(http.MethodPost, cluster.TusURL()+targetPath, nil)
|
||||
require.NoError(t, err)
|
||||
createReq.Header.Set("Tus-Resumable", TusVersion)
|
||||
createReq.Header.Set("Upload-Length", strconv.Itoa(len(testData)))
|
||||
|
||||
createResp, err := client.Do(createReq)
|
||||
require.NoError(t, err)
|
||||
createResp.Body.Close()
|
||||
require.Equal(t, http.StatusCreated, createResp.StatusCode)
|
||||
uploadLocation := createResp.Header.Get("Location")
|
||||
|
||||
// PATCH A promises one sub-chunk and stalls mid-body, holding the session.
|
||||
conn, err := net.Dial("tcp", "127.0.0.1:"+testFilerPort)
|
||||
require.NoError(t, err)
|
||||
defer conn.Close()
|
||||
// bound the raw reads below so a filer that never answers fails here
|
||||
require.NoError(t, conn.SetDeadline(time.Now().Add(60*time.Second)))
|
||||
_, err = fmt.Fprintf(conn, "PATCH %s HTTP/1.1\r\nHost: 127.0.0.1:%s\r\nTus-Resumable: %s\r\nContent-Type: application/offset+octet-stream\r\nUpload-Offset: 0\r\nContent-Length: %d\r\n\r\n",
|
||||
uploadLocation, testFilerPort, TusVersion, subChunkSize)
|
||||
require.NoError(t, err)
|
||||
_, err = conn.Write(testData[:1024*1024])
|
||||
require.NoError(t, err)
|
||||
time.Sleep(2 * time.Second)
|
||||
|
||||
// PATCH B is the client's retry of the same range while A is in flight.
|
||||
retryReq, err := http.NewRequest(http.MethodPatch, cluster.FullURL(uploadLocation), bytes.NewReader(testData[:subChunkSize]))
|
||||
require.NoError(t, err)
|
||||
retryReq.Header.Set("Tus-Resumable", TusVersion)
|
||||
retryReq.Header.Set("Upload-Offset", "0")
|
||||
retryReq.Header.Set("Content-Type", "application/offset+octet-stream")
|
||||
|
||||
retryResp, err := client.Do(retryReq)
|
||||
require.NoError(t, err)
|
||||
retryResp.Body.Close()
|
||||
require.Equal(t, http.StatusLocked, retryResp.StatusCode, "a concurrent PATCH must be refused, not recorded twice")
|
||||
|
||||
// Finish PATCH A and read its response.
|
||||
_, err = conn.Write(testData[1024*1024 : subChunkSize])
|
||||
require.NoError(t, err)
|
||||
respReader := bufio.NewReader(conn)
|
||||
respA, err := http.ReadResponse(respReader, nil)
|
||||
require.NoError(t, err)
|
||||
respA.Body.Close()
|
||||
require.Equal(t, http.StatusNoContent, respA.StatusCode)
|
||||
|
||||
headReq, err := http.NewRequest(http.MethodHead, cluster.FullURL(uploadLocation), nil)
|
||||
require.NoError(t, err)
|
||||
headReq.Header.Set("Tus-Resumable", TusVersion)
|
||||
|
||||
headResp, err := client.Do(headReq)
|
||||
require.NoError(t, err)
|
||||
headResp.Body.Close()
|
||||
require.Equal(t, http.StatusOK, headResp.StatusCode)
|
||||
currentOffset, err := strconv.Atoi(headResp.Header.Get("Upload-Offset"))
|
||||
require.NoError(t, err)
|
||||
require.Equal(t, subChunkSize, currentOffset, "only PATCH A's sub-chunk should be recorded")
|
||||
|
||||
patchReq, err := http.NewRequest(http.MethodPatch, cluster.FullURL(uploadLocation), bytes.NewReader(testData[currentOffset:]))
|
||||
require.NoError(t, err)
|
||||
patchReq.Header.Set("Tus-Resumable", TusVersion)
|
||||
patchReq.Header.Set("Upload-Offset", strconv.Itoa(currentOffset))
|
||||
patchReq.Header.Set("Content-Type", "application/offset+octet-stream")
|
||||
|
||||
patchResp, err := client.Do(patchReq)
|
||||
require.NoError(t, err)
|
||||
patchResp.Body.Close()
|
||||
require.Equal(t, http.StatusNoContent, patchResp.StatusCode)
|
||||
|
||||
getResp, err := client.Get(cluster.FilerURL() + targetPath)
|
||||
require.NoError(t, err)
|
||||
defer getResp.Body.Close()
|
||||
require.Equal(t, http.StatusOK, getResp.StatusCode)
|
||||
body, err := io.ReadAll(getResp.Body)
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, testData, body, "the raced upload should complete with intact content")
|
||||
}
|
||||
|
||||
@@ -3,6 +3,7 @@ package volume_server_grpc_test
|
||||
import (
|
||||
"context"
|
||||
"net/http"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
@@ -414,3 +415,87 @@ func TestScrubEcVolumeIndexCorruptEcx(t *testing.T) {
|
||||
t.Fatalf("expected broken volume after ECX corruption")
|
||||
}
|
||||
}
|
||||
|
||||
// scrubEcUntilLocated runs an EC scrub, retrying while the server is still waiting on
|
||||
// the master for the shard locations a distributed scrub needs.
|
||||
func scrubEcUntilLocated(t *testing.T, ctx context.Context, grpcClient volume_server_pb.VolumeServerClient, volumeID uint32, mode volume_server_pb.VolumeScrubMode) *volume_server_pb.ScrubEcVolumeResponse {
|
||||
t.Helper()
|
||||
deadline := time.Now().Add(60 * time.Second)
|
||||
for {
|
||||
resp, err := grpcClient.ScrubEcVolume(ctx, &volume_server_pb.ScrubEcVolumeRequest{
|
||||
VolumeIds: []uint32{volumeID},
|
||||
Mode: mode,
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("ScrubEcVolume %s failed: %v", mode, err)
|
||||
}
|
||||
waiting := false
|
||||
for _, d := range resp.GetDetails() {
|
||||
if strings.Contains(d, "failed to locate shard via master grpc") {
|
||||
waiting = true
|
||||
break
|
||||
}
|
||||
}
|
||||
if !waiting {
|
||||
return resp
|
||||
}
|
||||
if time.Now().After(deadline) {
|
||||
t.Fatalf("master never reported EC shard locations for volume %d: %v", volumeID, resp.GetDetails())
|
||||
}
|
||||
time.Sleep(time.Second)
|
||||
}
|
||||
}
|
||||
|
||||
// With a shard gone, FULL cannot read the intervals that lived on it, while READS
|
||||
// rebuilds them from parity. Either way the missing shard has to be reported: a
|
||||
// volume that scrubs clean is a volume nobody repairs.
|
||||
func TestScrubEcVolumeReadsRecoversMissingShard(t *testing.T) {
|
||||
if testing.Short() {
|
||||
t.Skip("skipping integration test in short mode")
|
||||
}
|
||||
|
||||
clusterHarness := framework.StartVolumeCluster(t, matrix.P1())
|
||||
conn, grpcClient := framework.DialVolumeServer(t, clusterHarness.VolumeGRPCAddress())
|
||||
defer conn.Close()
|
||||
|
||||
const volumeID = uint32(216)
|
||||
const missingShard = uint32(0)
|
||||
httpClient := framework.NewHTTPClient()
|
||||
ecSetup(t, grpcClient, httpClient, clusterHarness.VolumeAdminURL(), volumeID)
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 3*time.Minute)
|
||||
defer cancel()
|
||||
|
||||
if _, err := grpcClient.VolumeEcShardsUnmount(ctx, &volume_server_pb.VolumeEcShardsUnmountRequest{
|
||||
VolumeId: volumeID,
|
||||
ShardIds: []uint32{missingShard},
|
||||
}); err != nil {
|
||||
t.Fatalf("VolumeEcShardsUnmount shard %d failed: %v", missingShard, err)
|
||||
}
|
||||
if _, err := grpcClient.VolumeEcShardsDelete(ctx, &volume_server_pb.VolumeEcShardsDeleteRequest{
|
||||
VolumeId: volumeID,
|
||||
ShardIds: []uint32{missingShard},
|
||||
}); err != nil {
|
||||
t.Fatalf("VolumeEcShardsDelete shard %d failed: %v", missingShard, err)
|
||||
}
|
||||
|
||||
fullResp := scrubEcUntilLocated(t, ctx, grpcClient, volumeID, volume_server_pb.VolumeScrubMode_FULL)
|
||||
assertOnlyBrokenShard(t, "FULL", fullResp, missingShard)
|
||||
if len(fullResp.GetDetails()) == 0 {
|
||||
t.Fatalf("FULL should report the needles it could not read")
|
||||
}
|
||||
|
||||
readsResp := scrubEcUntilLocated(t, ctx, grpcClient, volumeID, volume_server_pb.VolumeScrubMode_READS)
|
||||
assertOnlyBrokenShard(t, "READS", readsResp, missingShard)
|
||||
if len(readsResp.GetDetails()) != 0 {
|
||||
t.Fatalf("READS should rebuild every needle from parity, got: %v", readsResp.GetDetails())
|
||||
}
|
||||
}
|
||||
|
||||
func assertOnlyBrokenShard(t *testing.T, mode string, resp *volume_server_pb.ScrubEcVolumeResponse, shardID uint32) {
|
||||
t.Helper()
|
||||
infos := resp.GetBrokenShardInfos()
|
||||
if len(infos) != 1 || infos[0].GetShardId() != shardID {
|
||||
t.Fatalf("%s reported broken shards %v, want only shard %d (details: %v)", mode, infos, shardID, resp.GetDetails())
|
||||
}
|
||||
}
|
||||
|
||||
@@ -180,11 +180,31 @@ func TestRenameOverExisting(t *testing.T) {
|
||||
// they indict different layers; a second look says whether it persists.
|
||||
time.Sleep(200 * time.Millisecond)
|
||||
fi2, err2 := os.Stat(src)
|
||||
t.Fatalf("stat of the renamed-away source did not return not-exist: stat=%v err=%v; 200ms later stat=%v err=%v",
|
||||
describeFileInfo(fi), err, describeFileInfo(fi2), err2)
|
||||
// A listing reads no per-path cache and the mount's own forgets
|
||||
// within a second, so a name that survives both is back on the filer.
|
||||
listed := dirNames(t, dir)
|
||||
time.Sleep(2 * time.Second)
|
||||
_, errLater := os.Stat(src)
|
||||
t.Fatalf("stat of the renamed-away source did not return not-exist: stat=%v err=%v; 200ms later stat=%v err=%v; past the path cache err=%v; %s lists %v",
|
||||
describeFileInfo(fi), err, describeFileInfo(fi2), err2, errLater, dir, listed)
|
||||
}
|
||||
}
|
||||
|
||||
// dirNames lists a directory for a failure message, reporting the error in
|
||||
// place of the names rather than failing a test that is already failing.
|
||||
func dirNames(t *testing.T, dir string) []string {
|
||||
t.Helper()
|
||||
entries, err := os.ReadDir(dir)
|
||||
if err != nil {
|
||||
return []string{"readdir: " + err.Error()}
|
||||
}
|
||||
names := make([]string, 0, len(entries))
|
||||
for _, entry := range entries {
|
||||
names = append(names, entry.Name())
|
||||
}
|
||||
return names
|
||||
}
|
||||
|
||||
func TestRenameAcrossDirectories(t *testing.T) {
|
||||
dir := testRoot(t)
|
||||
from := filepath.Join(dir, "from")
|
||||
|
||||
+147
-37
@@ -2,10 +2,13 @@ package dash
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"net"
|
||||
"net/http"
|
||||
"net/url"
|
||||
"sort"
|
||||
"strings"
|
||||
"sync"
|
||||
@@ -34,6 +37,7 @@ import (
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api"
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api/lifecycle_xml"
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api/policy_engine"
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api/s3_constants"
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api/s3lifecycle"
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api/s3lifecycle/scheduler"
|
||||
@@ -157,11 +161,15 @@ type AdminServer struct {
|
||||
s3TablesManager *s3tables.Manager
|
||||
icebergPort int
|
||||
lancePort int
|
||||
|
||||
// s3PublicEndpoint is the client-facing S3 address from admin.toml
|
||||
// (s3.public_endpoint); discovered S3 servers are the fallback.
|
||||
s3PublicEndpoint string
|
||||
}
|
||||
|
||||
// Type definitions moved to types.go
|
||||
|
||||
func NewAdminServer(masters string, filerGroup string, templateFS http.FileSystem, dataDir string, icebergPort, lancePort int) *AdminServer {
|
||||
func NewAdminServer(masters string, filerGroup string, templateFS http.FileSystem, dataDir string, icebergPort, lancePort int, s3PublicEndpoint string) *AdminServer {
|
||||
grpcDialOption := security.LoadClientTLS(util.GetViper(), "grpc.admin")
|
||||
|
||||
// Create master client with multiple master support
|
||||
@@ -198,6 +206,7 @@ func NewAdminServer(masters string, filerGroup string, templateFS http.FileSyste
|
||||
s3TablesManager: newS3TablesManager(),
|
||||
icebergPort: icebergPort,
|
||||
lancePort: lancePort,
|
||||
s3PublicEndpoint: normalizeS3PublicEndpoint(s3PublicEndpoint),
|
||||
pluginLock: lockManager,
|
||||
adminPresenceLock: presenceLock,
|
||||
bgCancel: bgCancel,
|
||||
@@ -259,33 +268,25 @@ func NewAdminServer(masters string, filerGroup string, templateFS http.FileSyste
|
||||
maintenanceConfig = maintenance.DefaultMaintenanceConfig()
|
||||
}
|
||||
|
||||
// Apply new defaults to handle schema changes (like enabling by default)
|
||||
schema := maintenance.GetMaintenanceConfigSchema()
|
||||
if err := schema.ApplyDefaultsToProtobuf(maintenanceConfig); err != nil {
|
||||
glog.Warningf("Failed to apply schema defaults to loaded config: %v", err)
|
||||
}
|
||||
|
||||
// Force enable maintenance system for new default behavior
|
||||
// This handles the case where old configs had Enabled=false as default
|
||||
if !maintenanceConfig.Enabled {
|
||||
glog.V(1).Infof("Enabling maintenance system (new default behavior)")
|
||||
maintenanceConfig.Enabled = true
|
||||
}
|
||||
|
||||
glog.V(1).Infof("Maintenance system initialized with persistent configuration (enabled: %v)", maintenanceConfig.Enabled)
|
||||
glog.V(1).Infof("Maintenance system initialized with persistent configuration (enabled: %v)", maintenanceConfig.GetEnabled())
|
||||
} else {
|
||||
maintenanceConfig = maintenance.DefaultMaintenanceConfig()
|
||||
glog.V(1).Infof("No data directory configured, maintenance system will run in memory-only mode (enabled: %v)", maintenanceConfig.Enabled)
|
||||
glog.V(1).Infof("No data directory configured, maintenance system will run in memory-only mode (enabled: %v)", maintenanceConfig.GetEnabled())
|
||||
}
|
||||
|
||||
// Load saved task configurations from persistence. This has to run before the maintenance
|
||||
// manager is created: creating it applies the maintenance policy to the registered
|
||||
// detectors and schedulers, while this call replaces each task's whole config object, so
|
||||
// running it afterwards would discard what the policy just applied. Both read the same
|
||||
// persisted task config files, so the policy ends up as the last writer and stays
|
||||
// authoritative for the task types it covers.
|
||||
server.loadTaskConfigurationsFromPersistence()
|
||||
|
||||
// Always initialize maintenance manager
|
||||
server.InitMaintenanceManager(maintenanceConfig)
|
||||
|
||||
// Load saved task configurations from persistence
|
||||
server.loadTaskConfigurationsFromPersistence()
|
||||
|
||||
// Start maintenance manager if enabled
|
||||
if maintenanceConfig.Enabled {
|
||||
if maintenanceConfig.GetEnabled() {
|
||||
go func() {
|
||||
// Give master client a bit of time to connect before starting scans
|
||||
time.Sleep(2 * time.Second)
|
||||
@@ -293,6 +294,8 @@ func NewAdminServer(masters string, filerGroup string, templateFS http.FileSyste
|
||||
glog.Errorf("Failed to start maintenance manager: %v", err)
|
||||
}
|
||||
}()
|
||||
} else {
|
||||
glog.V(0).Infof("Maintenance system is disabled by configuration, not starting the maintenance manager")
|
||||
}
|
||||
|
||||
pluginOpts := adminplugin.Options{
|
||||
@@ -432,7 +435,72 @@ func (s *AdminServer) publishMaintenanceMetrics(ctx context.Context) {
|
||||
}
|
||||
}
|
||||
|
||||
// workerFleetTotals aggregates connected workers and their task slots across
|
||||
// BOTH worker registries the admin server keeps: the legacy maintenance-worker
|
||||
// registry (workers that register over the worker gRPC stream) and the plugin
|
||||
// worker registry (workers started as `weed worker`). Reading only the legacy
|
||||
// one reported zero workers on clusters that run the admin and the workers as
|
||||
// separate components, where no legacy worker ever registers.
|
||||
//
|
||||
// A worker can appear in both registries: `weed mini` starts both runtimes from
|
||||
// one working directory, so they share the persisted worker ID. Merging by ID
|
||||
// keeps such a worker counted once, and its slots are taken from the legacy
|
||||
// registry, which is where they were accounted for before.
|
||||
func (s *AdminServer) workerFleetTotals() (workers, usedSlots, maxSlots int) {
|
||||
var legacySlots map[string]maintenance.WorkerSlots
|
||||
if s.maintenanceManager != nil {
|
||||
legacySlots = s.maintenanceManager.GetWorkerSlots()
|
||||
}
|
||||
return mergeWorkerFleetTotals(legacySlots, s.GetPluginWorkers())
|
||||
}
|
||||
|
||||
// mergeWorkerFleetTotals unions the legacy and plugin worker registries by
|
||||
// worker ID. Plugin workers report their slots in the heartbeat, so one that
|
||||
// has connected but not yet sent a heartbeat adds to the worker count with zero
|
||||
// slots until its first heartbeat lands.
|
||||
func mergeWorkerFleetTotals(legacySlots map[string]maintenance.WorkerSlots, pluginWorkers []*adminplugin.WorkerSession) (workers, usedSlots, maxSlots int) {
|
||||
for _, slots := range legacySlots {
|
||||
workers++
|
||||
usedSlots += slots.Used
|
||||
maxSlots += slots.Max
|
||||
}
|
||||
|
||||
for _, session := range pluginWorkers {
|
||||
if session == nil {
|
||||
continue
|
||||
}
|
||||
if _, counted := legacySlots[session.WorkerID]; counted {
|
||||
continue
|
||||
}
|
||||
workers++
|
||||
if heartbeat := session.Heartbeat; heartbeat != nil {
|
||||
used := int(heartbeat.DetectionSlotsUsed) + int(heartbeat.ExecutionSlotsUsed)
|
||||
max := int(heartbeat.DetectionSlotsTotal) + int(heartbeat.ExecutionSlotsTotal)
|
||||
// A worker's self-reported slots are untrusted input; a stale or
|
||||
// misbehaving one should not be able to drive the aggregate gauge
|
||||
// negative, matching the same defensiveness as registry.go's own
|
||||
// slot arithmetic.
|
||||
if used < 0 {
|
||||
used = 0
|
||||
}
|
||||
if max < 0 {
|
||||
max = 0
|
||||
}
|
||||
usedSlots += used
|
||||
maxSlots += max
|
||||
}
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
func (s *AdminServer) collectMaintenanceMetrics() {
|
||||
// Published before the maintenanceManager guard below: plugin workers are
|
||||
// tracked independently of the maintenance manager.
|
||||
workers, usedSlots, maxSlots := s.workerFleetTotals()
|
||||
stats_collect.AdminWorkersConnected.Set(float64(workers))
|
||||
stats_collect.AdminWorkerSlots.WithLabelValues("used").Set(float64(usedSlots))
|
||||
stats_collect.AdminWorkerSlots.WithLabelValues("max").Set(float64(maxSlots))
|
||||
|
||||
if s.maintenanceManager == nil {
|
||||
return
|
||||
}
|
||||
@@ -456,11 +524,6 @@ func (s *AdminServer) collectMaintenanceMetrics() {
|
||||
} else {
|
||||
stats_collect.AdminMaintenanceNextScanTimestampSeconds.Set(0)
|
||||
}
|
||||
|
||||
workers, usedSlots, maxSlots := s.maintenanceManager.GetWorkerSlotTotals()
|
||||
stats_collect.AdminWorkersConnected.Set(float64(workers))
|
||||
stats_collect.AdminWorkerSlots.WithLabelValues("used").Set(float64(usedSlots))
|
||||
stats_collect.AdminWorkerSlots.WithLabelValues("max").Set(float64(maxSlots))
|
||||
}
|
||||
|
||||
// loadTaskConfigurationsFromPersistence loads saved task configurations from protobuf files
|
||||
@@ -823,6 +886,7 @@ func (s *AdminServer) GetS3Buckets() ([]S3Bucket, error) {
|
||||
Owner: owner,
|
||||
LifecycleRuleCount: lifecycleRuleCount,
|
||||
LifecycleEnabledCount: lifecycleEnabledCount,
|
||||
PolicyStatementCount: extractPolicyStatementCountFromEntry(resp.Entry),
|
||||
}
|
||||
buckets = append(buckets, bucket)
|
||||
}
|
||||
@@ -928,6 +992,7 @@ func (s *AdminServer) GetBucketDetails(bucketName string) (*BucketDetails, error
|
||||
details.Bucket.ObjectLockDuration = objectLockDuration
|
||||
details.Bucket.Owner = owner
|
||||
details.Bucket.LifecycleRuleCount, details.Bucket.LifecycleEnabledCount = extractLifecycleCountsFromEntry(bucketResp.Entry)
|
||||
details.Bucket.PolicyStatementCount = extractPolicyStatementCountFromEntry(bucketResp.Entry)
|
||||
|
||||
return nil
|
||||
})
|
||||
@@ -1168,14 +1233,8 @@ func (s *AdminServer) DeleteS3Bucket(bucketName string) error {
|
||||
// Then delete bucket directory recursively from filer
|
||||
// Use same parameters as s3.bucket.delete shell command and S3 API
|
||||
return s.WithFilerClient(func(client filer_pb.SeaweedFilerClient) error {
|
||||
_, err := client.DeleteEntry(ctx, &filer_pb.DeleteEntryRequest{
|
||||
Directory: filerConfig.BucketsPath,
|
||||
Name: bucketName,
|
||||
IsDeleteData: false, // Collection already deleted, just remove metadata
|
||||
IsRecursive: true,
|
||||
IgnoreRecursiveError: true, // Same as S3 API and shell command
|
||||
})
|
||||
if err != nil {
|
||||
// The collection is already gone, so this only has to drop the metadata.
|
||||
if err := filer_pb.DoRemove(ctx, client, filerConfig.BucketsPath, bucketName, false, true, true, false, nil); err != nil {
|
||||
return fmt.Errorf("failed to delete bucket: %w", err)
|
||||
}
|
||||
|
||||
@@ -1488,6 +1547,32 @@ func (s *AdminServer) GetClusterS3Servers() (*ClusterS3ServersData, error) {
|
||||
}, nil
|
||||
}
|
||||
|
||||
// GetS3Endpoint returns the configured address clients reach the S3 gateway
|
||||
// at, or "". S3 servers register only their gRPC address with the master, so
|
||||
// the client-facing address cannot be discovered and must be configured.
|
||||
func (s *AdminServer) GetS3Endpoint() string {
|
||||
return s.s3PublicEndpoint
|
||||
}
|
||||
|
||||
// normalizeS3PublicEndpoint trims a trailing slash and drops, with a warning,
|
||||
// a value that is not a plain absolute http or https URL, so the file browser
|
||||
// hides its URL actions instead of copying broken links. Checking the string
|
||||
// for "?" and "#" rather than the parsed query and fragment also catches
|
||||
// delimiters with nothing after them, which url.Parse stores as empty.
|
||||
func normalizeS3PublicEndpoint(endpoint string) string {
|
||||
endpoint = strings.TrimRight(endpoint, "/")
|
||||
if endpoint == "" {
|
||||
return ""
|
||||
}
|
||||
u, err := url.Parse(endpoint)
|
||||
if err != nil || (u.Scheme != "http" && u.Scheme != "https") || u.Host == "" || u.User != nil || strings.ContainsAny(endpoint, "?#") {
|
||||
// the value is not echoed: it may hold credentials in userinfo or a query
|
||||
glog.Warningf("ignoring s3.public_endpoint: expecting an http:// or https:// URL with a host and no credentials, query, or fragment")
|
||||
return ""
|
||||
}
|
||||
return endpoint
|
||||
}
|
||||
|
||||
// GetAllFilers method moved to client_management.go
|
||||
|
||||
// GetVolumeDetails method moved to volume_management.go
|
||||
@@ -1514,6 +1599,7 @@ func (as *AdminServer) GetConfigInfo(w http.ResponseWriter, r *http.Request) {
|
||||
configInfo["master_address"] = string(currentMaster)
|
||||
configInfo["cache_expiration"] = as.cacheExpiration.String()
|
||||
configInfo["filer_cache_expiration"] = as.filerCacheExpiration.String()
|
||||
configInfo["s3_public_endpoint"] = as.s3PublicEndpoint
|
||||
|
||||
// Add maintenance system info
|
||||
if as.maintenanceManager != nil {
|
||||
@@ -1531,13 +1617,13 @@ func (as *AdminServer) GetConfigInfo(w http.ResponseWriter, r *http.Request) {
|
||||
}
|
||||
|
||||
// StartWorkerGrpcServer starts the worker gRPC server
|
||||
func (s *AdminServer) StartWorkerGrpcServer(grpcPort int) error {
|
||||
func (s *AdminServer) StartWorkerGrpcServer(grpcPort int, listener net.Listener) error {
|
||||
if s.workerGrpcServer != nil {
|
||||
return fmt.Errorf("worker gRPC server is already running")
|
||||
}
|
||||
|
||||
s.workerGrpcServer = NewWorkerGrpcServer(s)
|
||||
return s.workerGrpcServer.StartWithTLS(grpcPort)
|
||||
return s.workerGrpcServer.StartWithTLS(grpcPort, listener)
|
||||
}
|
||||
|
||||
// StopWorkerGrpcServer stops the worker gRPC server
|
||||
@@ -1779,7 +1865,16 @@ func (s *AdminServer) ListPluginSchedulerStates() ([]adminplugin.SchedulerJobTyp
|
||||
|
||||
// InitMaintenanceManager initializes the maintenance manager
|
||||
func (s *AdminServer) InitMaintenanceManager(config *maintenance.MaintenanceConfig) {
|
||||
s.maintenanceManager = maintenance.NewMaintenanceManager(s, config)
|
||||
// Hand the real config store to the manager so that, if it has to build the maintenance policy
|
||||
// itself, it reads the persisted task configs instead of compiled-in defaults. Only pass it when
|
||||
// a data directory is actually configured: an unconfigured store has nothing to read, and a typed
|
||||
// nil pointer would satisfy the loaders' type assertion and then panic on use.
|
||||
var configPersistence interface{}
|
||||
if s.configPersistence != nil && s.configPersistence.IsConfigured() {
|
||||
configPersistence = s.configPersistence
|
||||
}
|
||||
|
||||
s.maintenanceManager = maintenance.NewMaintenanceManager(s, config, configPersistence)
|
||||
|
||||
// Set up task persistence if config persistence is available
|
||||
if s.configPersistence != nil {
|
||||
@@ -1794,7 +1889,7 @@ func (s *AdminServer) InitMaintenanceManager(config *maintenance.MaintenanceConf
|
||||
}
|
||||
}
|
||||
|
||||
glog.V(1).Infof("Maintenance manager initialized (enabled: %v)", config.Enabled)
|
||||
glog.V(1).Infof("Maintenance manager initialized (enabled: %v)", config.GetEnabled())
|
||||
}
|
||||
|
||||
// GetMaintenanceManager returns the maintenance manager
|
||||
@@ -2032,6 +2127,21 @@ func extractLifecycleCountsFromEntry(entry *filer_pb.Entry) (ruleCount, enabledC
|
||||
return
|
||||
}
|
||||
|
||||
// extractPolicyStatementCountFromEntry returns the number of statements in
|
||||
// the bucket's policy, or 0 if it has none or the stored JSON can't be
|
||||
// parsed. Forgiving on parse failure, same as extractLifecycleCountsFromEntry.
|
||||
func extractPolicyStatementCountFromEntry(entry *filer_pb.Entry) int {
|
||||
policyJSON := entry.Extended[s3api.BUCKET_POLICY_METADATA_KEY]
|
||||
if len(policyJSON) == 0 {
|
||||
return 0
|
||||
}
|
||||
var doc policy_engine.PolicyDocument
|
||||
if err := json.Unmarshal(policyJSON, &doc); err != nil {
|
||||
return 0
|
||||
}
|
||||
return len(doc.Statement)
|
||||
}
|
||||
|
||||
// GetConfigPersistence returns the config persistence manager
|
||||
func (as *AdminServer) GetConfigPersistence() *ConfigPersistence {
|
||||
return as.configPersistence
|
||||
|
||||
@@ -13,6 +13,7 @@ import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/filer"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api"
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api/policy_engine"
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api/s3_constants"
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api/s3lifecycle"
|
||||
)
|
||||
@@ -257,6 +258,106 @@ func validateBucketLifecycleRules(rules []BucketLifecycleRule) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// ShowBucketPolicy returns the policy document for a specific bucket, or
|
||||
// {"bucket": ..., "policy": null} if the bucket has none.
|
||||
func (s *AdminServer) ShowBucketPolicy(w http.ResponseWriter, r *http.Request) {
|
||||
bucketName := mux.Vars(r)["bucket"]
|
||||
if bucketName == "" {
|
||||
writeJSONError(w, http.StatusBadRequest, "Bucket name is required")
|
||||
return
|
||||
}
|
||||
|
||||
policy, raw, err := s.GetBucketPolicy(bucketName)
|
||||
if err != nil {
|
||||
writeJSONError(w, bucketPolicyErrorStatus(err), "Failed to get bucket policy: "+err.Error())
|
||||
return
|
||||
}
|
||||
|
||||
resp := map[string]interface{}{
|
||||
"bucket": bucketName,
|
||||
"policy": policy,
|
||||
}
|
||||
if policy == nil && len(raw) > 0 {
|
||||
// Stored bytes the decoder rejects: hand them to the JSON tab so
|
||||
// the operator can fix or delete the document.
|
||||
resp["policy_text"] = string(raw)
|
||||
}
|
||||
writeJSON(w, http.StatusOK, resp)
|
||||
}
|
||||
|
||||
// UpdateBucketPolicy replaces the bucket policy for a bucket.
|
||||
func (s *AdminServer) UpdateBucketPolicy(w http.ResponseWriter, r *http.Request) {
|
||||
if !requireSessionCSRFToken(w, r) {
|
||||
return
|
||||
}
|
||||
|
||||
bucketName := mux.Vars(r)["bucket"]
|
||||
if bucketName == "" {
|
||||
writeJSONError(w, http.StatusBadRequest, "Bucket name is required")
|
||||
return
|
||||
}
|
||||
|
||||
var req struct {
|
||||
Policy *policy_engine.PolicyDocument `json:"policy"`
|
||||
}
|
||||
if err := decodeJSONBody(newJSONMaxReader(w, r), &req); err != nil {
|
||||
writeJSONError(w, http.StatusBadRequest, "Invalid request: "+err.Error())
|
||||
return
|
||||
}
|
||||
if req.Policy == nil {
|
||||
writeJSONError(w, http.StatusBadRequest, "policy is required; use DELETE to clear a bucket policy")
|
||||
return
|
||||
}
|
||||
|
||||
if err := s.SetBucketPolicy(bucketName, req.Policy); err != nil {
|
||||
writeJSONError(w, bucketPolicyErrorStatus(err), "Failed to update bucket policy: "+err.Error())
|
||||
return
|
||||
}
|
||||
|
||||
writeJSON(w, http.StatusOK, map[string]interface{}{
|
||||
"message": "Bucket policy updated successfully",
|
||||
"bucket": bucketName,
|
||||
})
|
||||
}
|
||||
|
||||
// RemoveBucketPolicy clears the bucket policy for a bucket. Named
|
||||
// "Remove", not "Delete", because (*AdminServer).DeleteBucketPolicy is the
|
||||
// data-layer method this handler calls.
|
||||
func (s *AdminServer) RemoveBucketPolicy(w http.ResponseWriter, r *http.Request) {
|
||||
if !requireSessionCSRFToken(w, r) {
|
||||
return
|
||||
}
|
||||
|
||||
bucketName := mux.Vars(r)["bucket"]
|
||||
if bucketName == "" {
|
||||
writeJSONError(w, http.StatusBadRequest, "Bucket name is required")
|
||||
return
|
||||
}
|
||||
|
||||
if err := s.DeleteBucketPolicy(bucketName); err != nil {
|
||||
writeJSONError(w, bucketPolicyErrorStatus(err), "Failed to delete bucket policy: "+err.Error())
|
||||
return
|
||||
}
|
||||
|
||||
writeJSON(w, http.StatusOK, map[string]interface{}{
|
||||
"message": "Bucket policy deleted successfully",
|
||||
"bucket": bucketName,
|
||||
})
|
||||
}
|
||||
|
||||
// bucketPolicyErrorStatus keeps a request for a bucket that does not exist,
|
||||
// or an invalid policy document, out of the 5xx bucket where a client
|
||||
// would retry it. Mirrors bucketLifecycleErrorStatus.
|
||||
func bucketPolicyErrorStatus(err error) int {
|
||||
if errors.Is(err, ErrBucketNotFound) {
|
||||
return http.StatusNotFound
|
||||
}
|
||||
if errors.Is(err, ErrInvalidBucketPolicy) {
|
||||
return http.StatusBadRequest
|
||||
}
|
||||
return http.StatusInternalServerError
|
||||
}
|
||||
|
||||
// CreateBucket creates a new S3 bucket
|
||||
func (s *AdminServer) CreateBucket(w http.ResponseWriter, r *http.Request) {
|
||||
var req CreateBucketRequest
|
||||
|
||||
@@ -0,0 +1,161 @@
|
||||
package dash
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api"
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api/policy_engine"
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api/s3_constants"
|
||||
)
|
||||
|
||||
// ErrInvalidBucketPolicy wraps a validation failure from SetBucketPolicy so
|
||||
// callers (the HTTP handler) can map it to 400 instead of 500 without
|
||||
// resorting to matching on the error string.
|
||||
var ErrInvalidBucketPolicy = errors.New("invalid bucket policy")
|
||||
|
||||
// GetBucketPolicy returns the policy document stored on a bucket's filer
|
||||
// entry, or (nil, nil, nil) if the bucket has no policy — that is not an
|
||||
// error, it just means the caller (e.g. the admin UI) should show an empty
|
||||
// editor instead of special-casing a 404. Stored bytes the current decoder
|
||||
// rejects come back as (nil, raw, nil): a 500 here would leave the UI
|
||||
// unable to show, fix, or even delete the one policy an operator most
|
||||
// needs to remove — and DeleteBucketPolicy never reads the document.
|
||||
func (s *AdminServer) GetBucketPolicy(bucketName string) (*policy_engine.PolicyDocument, []byte, error) {
|
||||
filerConfig, err := s.getFilerConfig()
|
||||
if err != nil {
|
||||
return nil, nil, fmt.Errorf("get filer configuration: %w", err)
|
||||
}
|
||||
|
||||
var doc *policy_engine.PolicyDocument
|
||||
var raw []byte
|
||||
err = s.WithFilerClient(func(client filer_pb.SeaweedFilerClient) error {
|
||||
resp, err := filer_pb.LookupEntry(context.Background(), client, &filer_pb.LookupDirectoryEntryRequest{
|
||||
Directory: filerConfig.BucketsPath,
|
||||
Name: bucketName,
|
||||
})
|
||||
if err != nil {
|
||||
if errors.Is(err, filer_pb.ErrNotFound) {
|
||||
return fmt.Errorf("%w: %s", ErrBucketNotFound, bucketName)
|
||||
}
|
||||
return fmt.Errorf("look up bucket %s: %w", bucketName, err)
|
||||
}
|
||||
|
||||
policyJSON := resp.Entry.Extended[s3api.BUCKET_POLICY_METADATA_KEY]
|
||||
if len(policyJSON) == 0 {
|
||||
return nil
|
||||
}
|
||||
raw = policyJSON
|
||||
|
||||
var parsed policy_engine.PolicyDocument
|
||||
if err := json.Unmarshal(policyJSON, &parsed); err == nil {
|
||||
doc = &parsed
|
||||
}
|
||||
return nil
|
||||
})
|
||||
if err != nil {
|
||||
return nil, nil, err
|
||||
}
|
||||
|
||||
return doc, raw, nil
|
||||
}
|
||||
|
||||
// SetBucketPolicy validates and stores a bucket policy, applying the exact
|
||||
// same validation the S3 gateway's PutBucketPolicy enforces
|
||||
// (policy_engine.ValidatePolicy + policy_engine.ValidateBucketPolicy), so
|
||||
// the admin UI and the S3 API never disagree about what's a valid policy.
|
||||
//
|
||||
// Propagation to every S3 gateway is automatic: writing the
|
||||
// s3-bucket-policy extended attribute drives the filer metadata log, which
|
||||
// each gateway's onBucketMetadataChange subscription watches to rebuild its
|
||||
// bucket policy cache and to maintain the advanced-IAM
|
||||
// "bucket-policy:<bucket>" mirror (mirrorBucketPolicyToIAM in
|
||||
// weed/s3api/s3api_bucket_policy_handlers.go). No separate notify step is
|
||||
// needed here.
|
||||
func (s *AdminServer) SetBucketPolicy(bucketName string, doc *policy_engine.PolicyDocument) error {
|
||||
if err := policy_engine.ValidatePolicy(doc); err != nil {
|
||||
return fmt.Errorf("%w: %w", ErrInvalidBucketPolicy, err)
|
||||
}
|
||||
if err := policy_engine.ValidateBucketPolicy(doc, bucketName); err != nil {
|
||||
return fmt.Errorf("%w: %w", ErrInvalidBucketPolicy, err)
|
||||
}
|
||||
|
||||
policyJSON, err := json.Marshal(doc)
|
||||
if err != nil {
|
||||
return fmt.Errorf("marshal policy document: %w", err)
|
||||
}
|
||||
if len(policyJSON) > policy_engine.MaxBucketPolicySize {
|
||||
return fmt.Errorf("%w: bucket policy is %d bytes, which exceeds the %d byte limit", ErrInvalidBucketPolicy, len(policyJSON), policy_engine.MaxBucketPolicySize)
|
||||
}
|
||||
|
||||
return s.writeBucketPolicy(bucketName, policyJSON)
|
||||
}
|
||||
|
||||
// DeleteBucketPolicy clears the bucket policy stored on a bucket's filer
|
||||
// entry. Deleting a policy that doesn't exist is a success, matching
|
||||
// DeleteBucketLifecycle's idempotent behavior. This cannot go through
|
||||
// SetBucketPolicy: validation there rejects a nil document.
|
||||
func (s *AdminServer) DeleteBucketPolicy(bucketName string) error {
|
||||
return s.writeBucketPolicy(bucketName, nil)
|
||||
}
|
||||
|
||||
// writeBucketPolicy patches the policy key on the bucket's filer entry; a
|
||||
// nil policyJSON clears it.
|
||||
func (s *AdminServer) writeBucketPolicy(bucketName string, policyJSON []byte) error {
|
||||
filerConfig, err := s.getFilerConfig()
|
||||
if err != nil {
|
||||
return fmt.Errorf("get filer configuration: %w", err)
|
||||
}
|
||||
|
||||
return s.WithFilerClient(func(client filer_pb.SeaweedFilerClient) error {
|
||||
// PATCH_EXTENDED is a no-op on a missing entry, so the existence
|
||||
// check has to happen here rather than fall out of the write.
|
||||
if _, err := filer_pb.LookupEntry(context.Background(), client, &filer_pb.LookupDirectoryEntryRequest{
|
||||
Directory: filerConfig.BucketsPath,
|
||||
Name: bucketName,
|
||||
}); err != nil {
|
||||
if errors.Is(err, filer_pb.ErrNotFound) {
|
||||
return fmt.Errorf("%w: %s", ErrBucketNotFound, bucketName)
|
||||
}
|
||||
return fmt.Errorf("look up bucket %s: %w", bucketName, err)
|
||||
}
|
||||
|
||||
bucketPath := filerConfig.BucketsPath + "/" + bucketName
|
||||
resp, err := client.ObjectTransaction(context.Background(), &filer_pb.ObjectTransactionRequest{
|
||||
LockKey: bucketPath,
|
||||
RouteKey: s3_constants.ObjectWriteRouteKeyPrefix + bucketPath,
|
||||
Mutations: []*filer_pb.ObjectMutation{bucketPolicyMutation(filerConfig.BucketsPath, bucketName, policyJSON)},
|
||||
})
|
||||
if err != nil {
|
||||
return fmt.Errorf("write bucket policy: %w", err)
|
||||
}
|
||||
if resp.Error != "" {
|
||||
return fmt.Errorf("write bucket policy: %s", resp.Error)
|
||||
}
|
||||
return nil
|
||||
})
|
||||
}
|
||||
|
||||
// bucketPolicyMutation patches only the policy key rather than writing the
|
||||
// whole entry back: the filer re-reads and merges under the bucket path
|
||||
// lock, so a concurrent owner/quota/versioning/lifecycle change is
|
||||
// preserved instead of being reverted by a stale snapshot. Same pattern as
|
||||
// bucketLifecycleMutation. A nil/empty policyJSON clears the key.
|
||||
func bucketPolicyMutation(bucketsPath, bucketName string, policyJSON []byte) *filer_pb.ObjectMutation {
|
||||
mutation := &filer_pb.ObjectMutation{
|
||||
Type: filer_pb.ObjectMutation_PATCH_EXTENDED,
|
||||
Directory: bucketsPath,
|
||||
Name: bucketName,
|
||||
}
|
||||
if len(policyJSON) > 0 {
|
||||
mutation.SetExtended = map[string][]byte{
|
||||
s3api.BUCKET_POLICY_METADATA_KEY: policyJSON,
|
||||
}
|
||||
return mutation
|
||||
}
|
||||
mutation.DeleteExtended = []string{s3api.BUCKET_POLICY_METADATA_KEY}
|
||||
return mutation
|
||||
}
|
||||
@@ -0,0 +1,168 @@
|
||||
package dash
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api"
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api/policy_engine"
|
||||
)
|
||||
|
||||
func validBucketPolicyJSON(bucket string) []byte {
|
||||
return []byte(fmt.Sprintf(`{"Version":"2012-10-17","Statement":[{"Effect":"Allow","Principal":"*","Action":"s3:GetObject","Resource":"arn:aws:s3:::%s/*"}]}`, bucket))
|
||||
}
|
||||
|
||||
func validBucketPolicyDoc(bucket string) *policy_engine.PolicyDocument {
|
||||
var doc policy_engine.PolicyDocument
|
||||
if err := json.Unmarshal(validBucketPolicyJSON(bucket), &doc); err != nil {
|
||||
panic(err)
|
||||
}
|
||||
return &doc
|
||||
}
|
||||
|
||||
func TestBucketPolicyMutation_SetsPolicy(t *testing.T) {
|
||||
policyJSON := validBucketPolicyJSON("mybucket")
|
||||
m := bucketPolicyMutation("/buckets", "mybucket", policyJSON)
|
||||
|
||||
if m.Type != filer_pb.ObjectMutation_PATCH_EXTENDED {
|
||||
t.Fatalf("expected a PATCH_EXTENDED mutation, got %v", m.Type)
|
||||
}
|
||||
if m.Directory != "/buckets" || m.Name != "mybucket" {
|
||||
t.Fatalf("expected the mutation to target /buckets/mybucket, got %s/%s", m.Directory, m.Name)
|
||||
}
|
||||
if got := m.SetExtended[s3api.BUCKET_POLICY_METADATA_KEY]; string(got) != string(policyJSON) {
|
||||
t.Fatalf("expected the policy key to carry the marshaled document, got %q", got)
|
||||
}
|
||||
if len(m.DeleteExtended) != 0 {
|
||||
t.Fatalf("expected no key deletions when saving a policy, got %v", m.DeleteExtended)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBucketPolicyMutation_ClearsKey(t *testing.T) {
|
||||
m := bucketPolicyMutation("/buckets", "mybucket", nil)
|
||||
|
||||
if len(m.SetExtended) != 0 {
|
||||
t.Fatalf("expected no key writes when clearing, got %v", m.SetExtended)
|
||||
}
|
||||
if len(m.DeleteExtended) != 1 || m.DeleteExtended[0] != s3api.BUCKET_POLICY_METADATA_KEY {
|
||||
t.Fatalf("expected only the policy key to be cleared, got %v", m.DeleteExtended)
|
||||
}
|
||||
}
|
||||
|
||||
// A whole-entry write would have carried the rest of the bucket entry with
|
||||
// it; the patch must name only the key it owns, so a concurrent owner,
|
||||
// quota, or lifecycle change survives.
|
||||
func TestBucketPolicyMutation_TouchesOnlyPolicyKey(t *testing.T) {
|
||||
for _, m := range []*filer_pb.ObjectMutation{
|
||||
bucketPolicyMutation("/buckets", "mybucket", validBucketPolicyJSON("mybucket")),
|
||||
bucketPolicyMutation("/buckets", "mybucket", nil),
|
||||
} {
|
||||
if m.Entry != nil {
|
||||
t.Fatal("expected the mutation to carry no entry snapshot")
|
||||
}
|
||||
if m.SetContent {
|
||||
t.Fatal("expected the mutation to leave entry content alone")
|
||||
}
|
||||
for k := range m.SetExtended {
|
||||
if k != s3api.BUCKET_POLICY_METADATA_KEY {
|
||||
t.Fatalf("unexpected key written: %s", k)
|
||||
}
|
||||
}
|
||||
for _, k := range m.DeleteExtended {
|
||||
if k != s3api.BUCKET_POLICY_METADATA_KEY {
|
||||
t.Fatalf("unexpected key deleted: %s", k)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestExtractPolicyStatementCountFromEntry(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
entry *filer_pb.Entry
|
||||
want int
|
||||
}{
|
||||
{"no extended attrs", &filer_pb.Entry{}, 0},
|
||||
{"absent key", &filer_pb.Entry{Extended: map[string][]byte{"other": []byte("x")}}, 0},
|
||||
{"one statement", &filer_pb.Entry{Extended: map[string][]byte{
|
||||
s3api.BUCKET_POLICY_METADATA_KEY: validBucketPolicyJSON("b"),
|
||||
}}, 1},
|
||||
{"three statements", &filer_pb.Entry{Extended: map[string][]byte{
|
||||
s3api.BUCKET_POLICY_METADATA_KEY: []byte(`{"Version":"2012-10-17","Statement":[
|
||||
{"Effect":"Allow","Principal":"*","Action":"s3:GetObject","Resource":"arn:aws:s3:::b/*"},
|
||||
{"Effect":"Allow","Principal":"*","Action":"s3:PutObject","Resource":"arn:aws:s3:::b/*"},
|
||||
{"Effect":"Deny","Principal":"*","Action":"s3:DeleteObject","Resource":"arn:aws:s3:::b/*"}
|
||||
]}`),
|
||||
}}, 3},
|
||||
{"garbage bytes", &filer_pb.Entry{Extended: map[string][]byte{
|
||||
s3api.BUCKET_POLICY_METADATA_KEY: []byte("not json"),
|
||||
}}, 0},
|
||||
}
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
if got := extractPolicyStatementCountFromEntry(tt.entry); got != tt.want {
|
||||
t.Errorf("extractPolicyStatementCountFromEntry() = %d, want %d", got, tt.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestBucketPolicyErrorStatus(t *testing.T) {
|
||||
if got := bucketPolicyErrorStatus(fmt.Errorf("%w: mybucket", ErrBucketNotFound)); got != http.StatusNotFound {
|
||||
t.Fatalf("expected a missing bucket to map to 404, got %d", got)
|
||||
}
|
||||
if got := bucketPolicyErrorStatus(fmt.Errorf("%w: bad statement", ErrInvalidBucketPolicy)); got != http.StatusBadRequest {
|
||||
t.Fatalf("expected an invalid policy to map to 400, got %d", got)
|
||||
}
|
||||
if got := bucketPolicyErrorStatus(errors.New("filer unreachable")); got != http.StatusInternalServerError {
|
||||
t.Fatalf("expected an unrelated failure to stay 500, got %d", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSetBucketPolicy_RejectsOversized(t *testing.T) {
|
||||
// A resource list long enough to blow the cap: this must fail before
|
||||
// any filer call, which is what makes it testable without one.
|
||||
doc := validBucketPolicyDoc("mybucket")
|
||||
doc.Statement[0].Sid = strings.Repeat("x", policy_engine.MaxBucketPolicySize+1)
|
||||
|
||||
err := (&AdminServer{}).SetBucketPolicy("mybucket", doc)
|
||||
if err == nil {
|
||||
t.Fatal("expected an oversized bucket policy to be rejected")
|
||||
}
|
||||
if !errors.Is(err, ErrInvalidBucketPolicy) {
|
||||
t.Fatalf("expected an ErrInvalidBucketPolicy, got: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSetBucketPolicy_RejectsForeignResource(t *testing.T) {
|
||||
// Proves the shared policy_engine.ValidateBucketPolicy validator is
|
||||
// actually wired in: this is exactly the check the S3 gateway applies.
|
||||
doc := validBucketPolicyDoc("mybucket")
|
||||
doc.Statement[0].Resource = policy_engine.NewStringOrStringSlicePtr("arn:aws:s3:::other-bucket/*")
|
||||
|
||||
err := (&AdminServer{}).SetBucketPolicy("mybucket", doc)
|
||||
if err == nil {
|
||||
t.Fatal("expected a policy referencing a different bucket to be rejected")
|
||||
}
|
||||
if !errors.Is(err, ErrInvalidBucketPolicy) {
|
||||
t.Fatalf("expected an ErrInvalidBucketPolicy, got: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSetBucketPolicy_RejectsMissingPrincipal(t *testing.T) {
|
||||
doc := validBucketPolicyDoc("mybucket")
|
||||
doc.Statement[0].Principal = nil
|
||||
|
||||
err := (&AdminServer{}).SetBucketPolicy("mybucket", doc)
|
||||
if err == nil {
|
||||
t.Fatal("expected a policy with no Principal to be rejected")
|
||||
}
|
||||
if !errors.Is(err, ErrInvalidBucketPolicy) {
|
||||
t.Fatalf("expected an ErrInvalidBucketPolicy, got: %v", err)
|
||||
}
|
||||
}
|
||||
@@ -20,7 +20,7 @@ import (
|
||||
|
||||
// WithMasterClient executes a function with a master client connection
|
||||
func (s *AdminServer) WithMasterClient(f func(client master_pb.SeaweedClient) error) error {
|
||||
return s.masterClient.WithClient(false, f)
|
||||
return s.masterClient.WithClient(context.Background(), false, f)
|
||||
}
|
||||
|
||||
// WithFilerClient executes a function with a filer client connection
|
||||
|
||||
@@ -14,6 +14,7 @@ import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/worker_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/worker/tasks/balance"
|
||||
"github.com/seaweedfs/seaweedfs/weed/worker/tasks/ec_balance"
|
||||
"github.com/seaweedfs/seaweedfs/weed/worker/tasks/erasure_coding"
|
||||
"github.com/seaweedfs/seaweedfs/weed/worker/tasks/vacuum"
|
||||
"google.golang.org/protobuf/encoding/protojson"
|
||||
@@ -29,6 +30,7 @@ const (
|
||||
VacuumTaskConfigFile = "task_vacuum.pb"
|
||||
ECTaskConfigFile = "task_erasure_coding.pb"
|
||||
BalanceTaskConfigFile = "task_balance.pb"
|
||||
EcBalanceTaskConfigFile = "task_ec_balance.pb"
|
||||
ReplicationTaskConfigFile = "task_replication.pb"
|
||||
|
||||
// JSON reference files
|
||||
@@ -36,6 +38,7 @@ const (
|
||||
VacuumTaskConfigJSONFile = "task_vacuum.json"
|
||||
ECTaskConfigJSONFile = "task_erasure_coding.json"
|
||||
BalanceTaskConfigJSONFile = "task_balance.json"
|
||||
EcBalanceTaskConfigJSONFile = "task_ec_balance.json"
|
||||
ReplicationTaskConfigJSONFile = "task_replication.json"
|
||||
|
||||
// Task persistence subdirectories and settings
|
||||
@@ -53,6 +56,7 @@ type (
|
||||
VacuumTaskConfig = worker_pb.VacuumTaskConfig
|
||||
ErasureCodingTaskConfig = worker_pb.ErasureCodingTaskConfig
|
||||
BalanceTaskConfig = worker_pb.BalanceTaskConfig
|
||||
EcBalanceTaskConfig = worker_pb.EcBalanceTaskConfig
|
||||
ReplicationTaskConfig = worker_pb.ReplicationTaskConfig
|
||||
)
|
||||
|
||||
@@ -155,8 +159,19 @@ func (cp *ConfigPersistence) LoadMaintenanceConfig() (*MaintenanceConfig, error)
|
||||
if configData, err := os.ReadFile(configPath); err == nil {
|
||||
var config MaintenanceConfig
|
||||
if err := proto.Unmarshal(configData, &config); err == nil {
|
||||
// Fill in fields added to the schema after this file was written. The
|
||||
// enabled flag tracks presence, so an explicitly persisted false survives
|
||||
// this, while a file from before presence tracking (where an operator's
|
||||
// explicit false and the field's absence look the same on the wire) keeps
|
||||
// the enabled default rather than silently switching maintenance off.
|
||||
if err := maintenance.GetMaintenanceConfigSchema().ApplyDefaultsToProtobuf(&config); err != nil {
|
||||
glog.Warningf("Failed to apply schema defaults to loaded maintenance config: %v", err)
|
||||
}
|
||||
if config.Enabled == nil {
|
||||
config.Enabled = proto.Bool(true)
|
||||
}
|
||||
// Always populate policy from separate task configuration files
|
||||
config.Policy = buildPolicyFromTaskConfigs()
|
||||
config.Policy = cp.buildPolicyFromTaskConfigs()
|
||||
return &config, nil
|
||||
}
|
||||
}
|
||||
@@ -268,6 +283,28 @@ func (cp *ConfigPersistence) RestoreConfig(filename, backupName string) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// Default task policies. These derive from each task's own NewDefaultConfig() so that a
|
||||
// task type has exactly one definition of its defaults. They used to be hand-written copies
|
||||
// here, and had drifted from the values the tasks themselves and the admin UI schema use:
|
||||
// vacuum scanned every 24h instead of 2h, balance every 6h instead of 30m with a 0.1 instead
|
||||
// of 0.2 imbalance threshold, and erasure coding every 168h instead of 1h with a 0.90 instead
|
||||
// of 0.95 fullness ratio and a 1024MB instead of 30MB minimum volume size.
|
||||
func defaultVacuumTaskPolicy() *worker_pb.TaskPolicy {
|
||||
return vacuum.NewDefaultConfig().ToTaskPolicy()
|
||||
}
|
||||
|
||||
func defaultErasureCodingTaskPolicy() *worker_pb.TaskPolicy {
|
||||
return erasure_coding.NewDefaultConfig().ToTaskPolicy()
|
||||
}
|
||||
|
||||
func defaultBalanceTaskPolicy() *worker_pb.TaskPolicy {
|
||||
return balance.NewDefaultConfig().ToTaskPolicy()
|
||||
}
|
||||
|
||||
func defaultEcBalanceTaskPolicy() *worker_pb.TaskPolicy {
|
||||
return ec_balance.NewDefaultConfig().ToTaskPolicy()
|
||||
}
|
||||
|
||||
// SaveVacuumTaskConfig saves vacuum task configuration to protobuf file
|
||||
func (cp *ConfigPersistence) SaveVacuumTaskConfig(config *VacuumTaskConfig) error {
|
||||
return cp.saveTaskConfig(VacuumTaskConfigFile, config)
|
||||
@@ -288,28 +325,14 @@ func (cp *ConfigPersistence) LoadVacuumTaskConfig() (*VacuumTaskConfig, error) {
|
||||
}
|
||||
|
||||
// Return default config if no valid config found
|
||||
return &VacuumTaskConfig{
|
||||
GarbageThreshold: 0.3,
|
||||
MinVolumeAgeHours: 24,
|
||||
}, nil
|
||||
return defaultVacuumTaskPolicy().GetVacuumConfig(), nil
|
||||
}
|
||||
|
||||
// LoadVacuumTaskPolicy loads complete vacuum task policy from protobuf file
|
||||
func (cp *ConfigPersistence) LoadVacuumTaskPolicy() (*worker_pb.TaskPolicy, error) {
|
||||
if cp.dataDir == "" {
|
||||
// Return default policy if no data directory
|
||||
return &worker_pb.TaskPolicy{
|
||||
Enabled: true,
|
||||
MaxConcurrent: 2,
|
||||
RepeatIntervalSeconds: 24 * 3600, // 24 hours in seconds
|
||||
CheckIntervalSeconds: 6 * 3600, // 6 hours in seconds
|
||||
TaskConfig: &worker_pb.TaskPolicy_VacuumConfig{
|
||||
VacuumConfig: &worker_pb.VacuumTaskConfig{
|
||||
GarbageThreshold: 0.3,
|
||||
MinVolumeAgeHours: 24,
|
||||
},
|
||||
},
|
||||
}, nil
|
||||
return defaultVacuumTaskPolicy(), nil
|
||||
}
|
||||
|
||||
confDir := filepath.Join(cp.dataDir, ConfigSubdir)
|
||||
@@ -318,18 +341,7 @@ func (cp *ConfigPersistence) LoadVacuumTaskPolicy() (*worker_pb.TaskPolicy, erro
|
||||
// Check if file exists
|
||||
if _, err := os.Stat(configPath); os.IsNotExist(err) {
|
||||
// Return default policy if file doesn't exist
|
||||
return &worker_pb.TaskPolicy{
|
||||
Enabled: true,
|
||||
MaxConcurrent: 2,
|
||||
RepeatIntervalSeconds: 24 * 3600, // 24 hours in seconds
|
||||
CheckIntervalSeconds: 6 * 3600, // 6 hours in seconds
|
||||
TaskConfig: &worker_pb.TaskPolicy_VacuumConfig{
|
||||
VacuumConfig: &worker_pb.VacuumTaskConfig{
|
||||
GarbageThreshold: 0.3,
|
||||
MinVolumeAgeHours: 24,
|
||||
},
|
||||
},
|
||||
}, nil
|
||||
return defaultVacuumTaskPolicy(), nil
|
||||
}
|
||||
|
||||
// Read file
|
||||
@@ -371,32 +383,14 @@ func (cp *ConfigPersistence) LoadErasureCodingTaskConfig() (*ErasureCodingTaskCo
|
||||
}
|
||||
|
||||
// Return default config if no valid config found
|
||||
return &ErasureCodingTaskConfig{
|
||||
FullnessRatio: 0.9,
|
||||
QuietForSeconds: 3600,
|
||||
MinVolumeSizeMb: 1024,
|
||||
CollectionFilter: "",
|
||||
}, nil
|
||||
return defaultErasureCodingTaskPolicy().GetErasureCodingConfig(), nil
|
||||
}
|
||||
|
||||
// LoadErasureCodingTaskPolicy loads complete EC task policy from protobuf file
|
||||
func (cp *ConfigPersistence) LoadErasureCodingTaskPolicy() (*worker_pb.TaskPolicy, error) {
|
||||
if cp.dataDir == "" {
|
||||
// Return default policy if no data directory
|
||||
return &worker_pb.TaskPolicy{
|
||||
Enabled: true,
|
||||
MaxConcurrent: 1,
|
||||
RepeatIntervalSeconds: 168 * 3600, // 1 week in seconds
|
||||
CheckIntervalSeconds: 24 * 3600, // 24 hours in seconds
|
||||
TaskConfig: &worker_pb.TaskPolicy_ErasureCodingConfig{
|
||||
ErasureCodingConfig: &worker_pb.ErasureCodingTaskConfig{
|
||||
FullnessRatio: 0.9,
|
||||
QuietForSeconds: 3600,
|
||||
MinVolumeSizeMb: 1024,
|
||||
CollectionFilter: "",
|
||||
},
|
||||
},
|
||||
}, nil
|
||||
return defaultErasureCodingTaskPolicy(), nil
|
||||
}
|
||||
|
||||
confDir := filepath.Join(cp.dataDir, ConfigSubdir)
|
||||
@@ -405,20 +399,7 @@ func (cp *ConfigPersistence) LoadErasureCodingTaskPolicy() (*worker_pb.TaskPolic
|
||||
// Check if file exists
|
||||
if _, err := os.Stat(configPath); os.IsNotExist(err) {
|
||||
// Return default policy if file doesn't exist
|
||||
return &worker_pb.TaskPolicy{
|
||||
Enabled: true,
|
||||
MaxConcurrent: 1,
|
||||
RepeatIntervalSeconds: 168 * 3600, // 1 week in seconds
|
||||
CheckIntervalSeconds: 24 * 3600, // 24 hours in seconds
|
||||
TaskConfig: &worker_pb.TaskPolicy_ErasureCodingConfig{
|
||||
ErasureCodingConfig: &worker_pb.ErasureCodingTaskConfig{
|
||||
FullnessRatio: 0.9,
|
||||
QuietForSeconds: 3600,
|
||||
MinVolumeSizeMb: 1024,
|
||||
CollectionFilter: "",
|
||||
},
|
||||
},
|
||||
}, nil
|
||||
return defaultErasureCodingTaskPolicy(), nil
|
||||
}
|
||||
|
||||
// Read file
|
||||
@@ -460,28 +441,14 @@ func (cp *ConfigPersistence) LoadBalanceTaskConfig() (*BalanceTaskConfig, error)
|
||||
}
|
||||
|
||||
// Return default config if no valid config found
|
||||
return &BalanceTaskConfig{
|
||||
ImbalanceThreshold: 0.1,
|
||||
MinServerCount: 2,
|
||||
}, nil
|
||||
return defaultBalanceTaskPolicy().GetBalanceConfig(), nil
|
||||
}
|
||||
|
||||
// LoadBalanceTaskPolicy loads complete balance task policy from protobuf file
|
||||
func (cp *ConfigPersistence) LoadBalanceTaskPolicy() (*worker_pb.TaskPolicy, error) {
|
||||
if cp.dataDir == "" {
|
||||
// Return default policy if no data directory
|
||||
return &worker_pb.TaskPolicy{
|
||||
Enabled: true,
|
||||
MaxConcurrent: 1,
|
||||
RepeatIntervalSeconds: 6 * 3600, // 6 hours in seconds
|
||||
CheckIntervalSeconds: 12 * 3600, // 12 hours in seconds
|
||||
TaskConfig: &worker_pb.TaskPolicy_BalanceConfig{
|
||||
BalanceConfig: &worker_pb.BalanceTaskConfig{
|
||||
ImbalanceThreshold: 0.1,
|
||||
MinServerCount: 2,
|
||||
},
|
||||
},
|
||||
}, nil
|
||||
return defaultBalanceTaskPolicy(), nil
|
||||
}
|
||||
|
||||
confDir := filepath.Join(cp.dataDir, ConfigSubdir)
|
||||
@@ -490,18 +457,7 @@ func (cp *ConfigPersistence) LoadBalanceTaskPolicy() (*worker_pb.TaskPolicy, err
|
||||
// Check if file exists
|
||||
if _, err := os.Stat(configPath); os.IsNotExist(err) {
|
||||
// Return default policy if file doesn't exist
|
||||
return &worker_pb.TaskPolicy{
|
||||
Enabled: true,
|
||||
MaxConcurrent: 1,
|
||||
RepeatIntervalSeconds: 6 * 3600, // 6 hours in seconds
|
||||
CheckIntervalSeconds: 12 * 3600, // 12 hours in seconds
|
||||
TaskConfig: &worker_pb.TaskPolicy_BalanceConfig{
|
||||
BalanceConfig: &worker_pb.BalanceTaskConfig{
|
||||
ImbalanceThreshold: 0.1,
|
||||
MinServerCount: 2,
|
||||
},
|
||||
},
|
||||
}, nil
|
||||
return defaultBalanceTaskPolicy(), nil
|
||||
}
|
||||
|
||||
// Read file
|
||||
@@ -523,6 +479,60 @@ func (cp *ConfigPersistence) LoadBalanceTaskPolicy() (*worker_pb.TaskPolicy, err
|
||||
return nil, fmt.Errorf("failed to unmarshal balance task configuration")
|
||||
}
|
||||
|
||||
// SaveEcBalanceTaskPolicy saves complete EC balance task policy to protobuf file
|
||||
func (cp *ConfigPersistence) SaveEcBalanceTaskPolicy(policy *worker_pb.TaskPolicy) error {
|
||||
return cp.saveTaskConfig(EcBalanceTaskConfigFile, policy)
|
||||
}
|
||||
|
||||
// LoadEcBalanceTaskConfig loads EC balance task configuration from protobuf file
|
||||
func (cp *ConfigPersistence) LoadEcBalanceTaskConfig() (*EcBalanceTaskConfig, error) {
|
||||
if taskPolicy, err := cp.LoadEcBalanceTaskPolicy(); err == nil && taskPolicy != nil {
|
||||
if ecBalanceConfig := taskPolicy.GetEcBalanceConfig(); ecBalanceConfig != nil {
|
||||
return ecBalanceConfig, nil
|
||||
}
|
||||
}
|
||||
|
||||
// Return default config if no valid config found
|
||||
return defaultEcBalanceTaskPolicy().GetEcBalanceConfig(), nil
|
||||
}
|
||||
|
||||
// LoadEcBalanceTaskPolicy loads complete EC balance task policy from protobuf file.
|
||||
// ec_balance is registered like the other maintenance tasks and ec_balance.LoadConfigFromPersistence
|
||||
// asserts on this accessor, so without it the task could never be configured at all.
|
||||
func (cp *ConfigPersistence) LoadEcBalanceTaskPolicy() (*worker_pb.TaskPolicy, error) {
|
||||
if cp.dataDir == "" {
|
||||
// Return default policy if no data directory
|
||||
return defaultEcBalanceTaskPolicy(), nil
|
||||
}
|
||||
|
||||
confDir := filepath.Join(cp.dataDir, ConfigSubdir)
|
||||
configPath := filepath.Join(confDir, EcBalanceTaskConfigFile)
|
||||
|
||||
// Check if file exists
|
||||
if _, err := os.Stat(configPath); os.IsNotExist(err) {
|
||||
// Return default policy if file doesn't exist
|
||||
return defaultEcBalanceTaskPolicy(), nil
|
||||
}
|
||||
|
||||
// Read file
|
||||
configData, err := os.ReadFile(configPath)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("failed to read EC balance task config file: %w", err)
|
||||
}
|
||||
|
||||
// Try to unmarshal as TaskPolicy
|
||||
var policy worker_pb.TaskPolicy
|
||||
if err := proto.Unmarshal(configData, &policy); err == nil {
|
||||
// Validate that it's actually a TaskPolicy with EC balance config
|
||||
if policy.GetEcBalanceConfig() != nil {
|
||||
glog.V(1).Infof("Loaded EC balance task policy from %s", configPath)
|
||||
return &policy, nil
|
||||
}
|
||||
}
|
||||
|
||||
return nil, fmt.Errorf("failed to unmarshal EC balance task configuration")
|
||||
}
|
||||
|
||||
// SaveReplicationTaskConfig saves replication task configuration to protobuf file
|
||||
func (cp *ConfigPersistence) SaveReplicationTaskConfig(config *ReplicationTaskConfig) error {
|
||||
return cp.saveTaskConfig(ReplicationTaskConfigFile, config)
|
||||
@@ -632,6 +642,8 @@ func (cp *ConfigPersistence) SaveTaskPolicy(taskType string, policy *worker_pb.T
|
||||
return cp.SaveErasureCodingTaskPolicy(policy)
|
||||
case "balance":
|
||||
return cp.SaveBalanceTaskPolicy(policy)
|
||||
case "ec_balance":
|
||||
return cp.SaveEcBalanceTaskPolicy(policy)
|
||||
case "replication":
|
||||
return cp.SaveReplicationTaskPolicy(policy)
|
||||
}
|
||||
@@ -687,67 +699,13 @@ func (cp *ConfigPersistence) GetConfigInfo() map[string]interface{} {
|
||||
return info
|
||||
}
|
||||
|
||||
// buildPolicyFromTaskConfigs loads task configurations from separate files and builds a MaintenancePolicy
|
||||
func buildPolicyFromTaskConfigs() *worker_pb.MaintenancePolicy {
|
||||
policy := &worker_pb.MaintenancePolicy{
|
||||
GlobalMaxConcurrent: 4,
|
||||
DefaultRepeatIntervalSeconds: 6 * 3600, // 6 hours in seconds
|
||||
DefaultCheckIntervalSeconds: 12 * 3600, // 12 hours in seconds
|
||||
TaskPolicies: make(map[string]*worker_pb.TaskPolicy),
|
||||
}
|
||||
|
||||
// Load vacuum task configuration
|
||||
if vacuumConfig := vacuum.LoadConfigFromPersistence(nil); vacuumConfig != nil {
|
||||
policy.TaskPolicies["vacuum"] = &worker_pb.TaskPolicy{
|
||||
Enabled: vacuumConfig.Enabled,
|
||||
MaxConcurrent: int32(vacuumConfig.MaxConcurrent),
|
||||
RepeatIntervalSeconds: int32(vacuumConfig.ScanIntervalSeconds),
|
||||
CheckIntervalSeconds: int32(vacuumConfig.ScanIntervalSeconds),
|
||||
TaskConfig: &worker_pb.TaskPolicy_VacuumConfig{
|
||||
VacuumConfig: &worker_pb.VacuumTaskConfig{
|
||||
GarbageThreshold: float64(vacuumConfig.GarbageThreshold),
|
||||
MinVolumeAgeHours: int32((vacuumConfig.MinVolumeAgeSeconds + 3599) / 3600), // round up so sub-hour values don't become 0
|
||||
},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// Load erasure coding task configuration
|
||||
if ecConfig := erasure_coding.LoadConfigFromPersistence(nil); ecConfig != nil {
|
||||
policy.TaskPolicies["erasure_coding"] = &worker_pb.TaskPolicy{
|
||||
Enabled: ecConfig.Enabled,
|
||||
MaxConcurrent: int32(ecConfig.MaxConcurrent),
|
||||
RepeatIntervalSeconds: int32(ecConfig.ScanIntervalSeconds),
|
||||
CheckIntervalSeconds: int32(ecConfig.ScanIntervalSeconds),
|
||||
TaskConfig: &worker_pb.TaskPolicy_ErasureCodingConfig{
|
||||
ErasureCodingConfig: &worker_pb.ErasureCodingTaskConfig{
|
||||
FullnessRatio: float64(ecConfig.FullnessRatio),
|
||||
QuietForSeconds: int32(ecConfig.QuietForSeconds),
|
||||
MinVolumeSizeMb: int32(ecConfig.MinSizeMB),
|
||||
CollectionFilter: ecConfig.CollectionFilter,
|
||||
},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// Load balance task configuration
|
||||
if balanceConfig := balance.LoadConfigFromPersistence(nil); balanceConfig != nil {
|
||||
policy.TaskPolicies["balance"] = &worker_pb.TaskPolicy{
|
||||
Enabled: balanceConfig.Enabled,
|
||||
MaxConcurrent: int32(balanceConfig.MaxConcurrent),
|
||||
RepeatIntervalSeconds: int32(balanceConfig.ScanIntervalSeconds),
|
||||
CheckIntervalSeconds: int32(balanceConfig.ScanIntervalSeconds),
|
||||
TaskConfig: &worker_pb.TaskPolicy_BalanceConfig{
|
||||
BalanceConfig: &worker_pb.BalanceTaskConfig{
|
||||
ImbalanceThreshold: float64(balanceConfig.ImbalanceThreshold),
|
||||
MinServerCount: int32(balanceConfig.MinServerCount),
|
||||
},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
glog.V(1).Infof("Built maintenance policy from separate task configs - %d task policies loaded", len(policy.TaskPolicies))
|
||||
return policy
|
||||
// buildPolicyFromTaskConfigs builds the maintenance policy from the persisted task configs.
|
||||
//
|
||||
// The body lives in weed/admin/maintenance because the maintenance manager needs the same
|
||||
// policy when it has to build one itself, and this package already imports that one. Keeping
|
||||
// a second copy here is what let the two drift apart in the first place.
|
||||
func (cp *ConfigPersistence) buildPolicyFromTaskConfigs() *worker_pb.MaintenancePolicy {
|
||||
return maintenance.BuildPolicyFromTaskConfigs(cp)
|
||||
}
|
||||
|
||||
// SaveTaskDetail saves detailed task information to disk
|
||||
|
||||
@@ -12,6 +12,7 @@ import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/worker/tasks/base"
|
||||
"github.com/seaweedfs/seaweedfs/weed/worker/tasks/erasure_coding"
|
||||
"github.com/seaweedfs/seaweedfs/weed/worker/tasks/vacuum"
|
||||
"google.golang.org/protobuf/proto"
|
||||
"google.golang.org/protobuf/types/known/timestamppb"
|
||||
)
|
||||
|
||||
@@ -30,6 +31,20 @@ type TomlConfig interface {
|
||||
// directory loss and override admin UI edits on restart. Absent keys keep
|
||||
// their persisted values.
|
||||
func (cp *ConfigPersistence) ApplyMaintenanceConfigFromToml(v TomlConfig) error {
|
||||
var maintenanceConf *MaintenanceConfig
|
||||
maintenanceChanged := false
|
||||
if k := "maintenance.enabled"; v.IsSet(k) {
|
||||
conf, err := cp.LoadMaintenanceConfig()
|
||||
if err != nil {
|
||||
return fmt.Errorf("load maintenance config: %w", err)
|
||||
}
|
||||
conf.Enabled = proto.Bool(v.GetBool(k))
|
||||
// the policy lives in the per-task config files; don't snapshot it here
|
||||
conf.Policy = nil
|
||||
maintenanceConf = conf
|
||||
maintenanceChanged = true
|
||||
}
|
||||
|
||||
vacuumConf := vacuum.LoadConfigFromPersistence(cp)
|
||||
vacuumChanged := applyBaseConfigFromToml(v, "maintenance.vacuum.", &vacuumConf.BaseConfig)
|
||||
if k := "maintenance.vacuum.garbage_threshold"; v.IsSet(k) {
|
||||
@@ -84,13 +99,19 @@ func (cp *ConfigPersistence) ApplyMaintenanceConfigFromToml(v TomlConfig) error
|
||||
ecChanged = true
|
||||
}
|
||||
|
||||
if !vacuumChanged && !balanceChanged && !ecChanged {
|
||||
if !maintenanceChanged && !vacuumChanged && !balanceChanged && !ecChanged {
|
||||
return nil
|
||||
}
|
||||
if !cp.IsConfigured() {
|
||||
return fmt.Errorf("admin.toml maintenance settings require -dataDir to persist")
|
||||
}
|
||||
|
||||
if maintenanceChanged {
|
||||
if err := cp.SaveMaintenanceConfig(maintenanceConf); err != nil {
|
||||
return fmt.Errorf("save maintenance config: %w", err)
|
||||
}
|
||||
glog.V(0).Infof("Applied [maintenance] settings from admin.toml (enabled: %v)", maintenanceConf.GetEnabled())
|
||||
}
|
||||
if vacuumChanged {
|
||||
if err := cp.SaveVacuumTaskPolicy(vacuumConf.ToTaskPolicy()); err != nil {
|
||||
return fmt.Errorf("save vacuum task config: %w", err)
|
||||
|
||||
@@ -90,6 +90,36 @@ preferred_tags = "Fast, ssd"
|
||||
}
|
||||
}
|
||||
|
||||
func TestApplyMaintenanceConfigFromTomlEnabledToggle(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
cp := NewConfigPersistence(dir)
|
||||
|
||||
if err := cp.ApplyMaintenanceConfigFromToml(tomlConfig(t, "[maintenance]\nenabled = false\n")); err != nil {
|
||||
t.Fatalf("apply: %v", err)
|
||||
}
|
||||
conf, err := cp.LoadMaintenanceConfig()
|
||||
if err != nil {
|
||||
t.Fatalf("load maintenance config: %v", err)
|
||||
}
|
||||
if conf.GetEnabled() {
|
||||
t.Errorf("maintenance still enabled after [maintenance] enabled = false")
|
||||
}
|
||||
if conf.ScanIntervalSeconds != 30*60 {
|
||||
t.Errorf("scan interval = %d, want default 1800 kept alongside the toggle", conf.ScanIntervalSeconds)
|
||||
}
|
||||
|
||||
if err := cp.ApplyMaintenanceConfigFromToml(tomlConfig(t, "[maintenance]\nenabled = true\n")); err != nil {
|
||||
t.Fatalf("apply: %v", err)
|
||||
}
|
||||
conf, err = cp.LoadMaintenanceConfig()
|
||||
if err != nil {
|
||||
t.Fatalf("load maintenance config: %v", err)
|
||||
}
|
||||
if !conf.GetEnabled() {
|
||||
t.Errorf("maintenance still disabled after [maintenance] enabled = true")
|
||||
}
|
||||
}
|
||||
|
||||
func TestApplyMaintenanceConfigFromTomlNoKeys(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
cp := NewConfigPersistence(dir)
|
||||
|
||||
@@ -73,9 +73,12 @@ func (s *AdminServer) recordDashboardSample() {
|
||||
}
|
||||
}
|
||||
sample.tasks = float64(active)
|
||||
sample.workers = float64(stats.ActiveWorkers)
|
||||
}
|
||||
}
|
||||
// Counted across both worker registries, same as the Prometheus gauge, so
|
||||
// the card isn't stuck at 0 on clusters that only run plugin workers.
|
||||
workers, _, _ := s.workerFleetTotals()
|
||||
sample.workers = float64(workers)
|
||||
|
||||
s.dashSamplesMu.Lock()
|
||||
s.dashSamples = append(s.dashSamples, sample)
|
||||
|
||||
@@ -561,7 +561,7 @@ func (s *AdminServer) GetEcVolumeDetails(volumeID uint32, sortBy string, sortOrd
|
||||
|
||||
// Get detailed EC shard information for the specific volume via gRPC
|
||||
err := s.WithMasterClient(func(client master_pb.SeaweedClient) error {
|
||||
resp, err := client.VolumeList(context.Background(), &master_pb.VolumeListRequest{VolumeId: volumeID})
|
||||
resp, err := client.VolumeList(context.Background(), &master_pb.VolumeListRequest{VolumeIds: []uint32{volumeID}})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
@@ -2,6 +2,7 @@ package dash
|
||||
|
||||
import (
|
||||
"context"
|
||||
"net/url"
|
||||
"path"
|
||||
"strings"
|
||||
"time"
|
||||
@@ -46,6 +47,7 @@ type FileBrowserData struct {
|
||||
BucketName string `json:"bucket_name"`
|
||||
IsTableBucketPath bool `json:"is_table_bucket_path"`
|
||||
TableBucketName string `json:"table_bucket_name"`
|
||||
S3Endpoint string `json:"s3_endpoint,omitempty"`
|
||||
// Pagination fields
|
||||
PageSize int `json:"page_size"`
|
||||
HasNextPage bool `json:"has_next_page"`
|
||||
@@ -53,8 +55,11 @@ type FileBrowserData struct {
|
||||
CurrentLastFileName string `json:"current_last_file_name"` // Cursor from current request (for page size changes)
|
||||
}
|
||||
|
||||
// GetFileBrowser retrieves file browser data for a given path with cursor-based pagination
|
||||
func (s *AdminServer) GetFileBrowser(dir string, lastFileName string, pageSize int) (*FileBrowserData, error) {
|
||||
// GetFileBrowser retrieves file browser data for a given path with cursor-based
|
||||
// pagination. A non-empty prefix limits the listing to entries whose name starts
|
||||
// with it, so callers that only want part of a large directory don't have to page
|
||||
// through all of it.
|
||||
func (s *AdminServer) GetFileBrowser(dir string, prefix string, lastFileName string, pageSize int) (*FileBrowserData, error) {
|
||||
if dir == "" {
|
||||
dir = "/"
|
||||
}
|
||||
@@ -76,7 +81,7 @@ func (s *AdminServer) GetFileBrowser(dir string, lastFileName string, pageSize i
|
||||
// Fetch entries starting from the cursor (lastFileName)
|
||||
stream, err := client.ListEntries(context.Background(), &filer_pb.ListEntriesRequest{
|
||||
Directory: dir,
|
||||
Prefix: "",
|
||||
Prefix: prefix,
|
||||
Limit: uint32(fetchLimit),
|
||||
StartFromFileName: lastFileName,
|
||||
InclusiveStartFrom: false, // Don't include the cursor file itself
|
||||
@@ -185,6 +190,7 @@ func (s *AdminServer) GetFileBrowser(dir string, lastFileName string, pageSize i
|
||||
bucketName := ""
|
||||
isTableBucketPath := false
|
||||
tableBucketName := ""
|
||||
isRegularBucket := false
|
||||
if strings.HasPrefix(dir, "/buckets/") {
|
||||
isBucketPath = true
|
||||
pathParts := strings.Split(strings.Trim(dir, "/"), "/")
|
||||
@@ -202,6 +208,8 @@ func (s *AdminServer) GetFileBrowser(dir string, lastFileName string, pageSize i
|
||||
if s3tables.IsTableBucketEntry(resp.Entry) {
|
||||
isTableBucketPath = true
|
||||
tableBucketName = bucketName
|
||||
} else {
|
||||
isRegularBucket = true
|
||||
}
|
||||
return nil
|
||||
}); err != nil {
|
||||
@@ -210,6 +218,11 @@ func (s *AdminServer) GetFileBrowser(dir string, lastFileName string, pageSize i
|
||||
}
|
||||
}
|
||||
|
||||
s3Endpoint := ""
|
||||
if isRegularBucket {
|
||||
s3Endpoint = s.GetS3Endpoint()
|
||||
}
|
||||
|
||||
return &FileBrowserData{
|
||||
CurrentPath: dir,
|
||||
ParentPath: parentPath,
|
||||
@@ -221,6 +234,7 @@ func (s *AdminServer) GetFileBrowser(dir string, lastFileName string, pageSize i
|
||||
BucketName: bucketName,
|
||||
IsTableBucketPath: isTableBucketPath,
|
||||
TableBucketName: tableBucketName,
|
||||
S3Endpoint: s3Endpoint,
|
||||
// Pagination metadata
|
||||
PageSize: pageSize,
|
||||
HasNextPage: hasNextPage,
|
||||
@@ -269,3 +283,52 @@ func (s *AdminServer) generateBreadcrumbs(dir string) []BreadcrumbItem {
|
||||
|
||||
return breadcrumbs
|
||||
}
|
||||
|
||||
// S3ObjectURL builds the path-style S3 URL for a filer path under /buckets/,
|
||||
// percent-encoding each path segment. Returns "" for other paths.
|
||||
func S3ObjectURL(endpoint, fullPath string) string {
|
||||
rel, ok := strings.CutPrefix(fullPath, "/buckets/")
|
||||
if !ok || rel == "" || endpoint == "" {
|
||||
return ""
|
||||
}
|
||||
segments := strings.Split(rel, "/")
|
||||
for i, segment := range segments {
|
||||
segments[i] = url.PathEscape(segment)
|
||||
}
|
||||
return strings.TrimRight(endpoint, "/") + "/" + strings.Join(segments, "/")
|
||||
}
|
||||
|
||||
// GetS3ObjectURL returns the S3 URL an object under a regular bucket is served
|
||||
// at, or "" when the path is not a bucket object, the bucket is an S3 Tables
|
||||
// bucket, or no S3 endpoint is known.
|
||||
func (s *AdminServer) GetS3ObjectURL(fullPath string) string {
|
||||
rel, ok := strings.CutPrefix(fullPath, "/buckets/")
|
||||
if !ok {
|
||||
return ""
|
||||
}
|
||||
bucketName, key, found := strings.Cut(rel, "/")
|
||||
if !found || key == "" {
|
||||
return ""
|
||||
}
|
||||
endpoint := s.GetS3Endpoint()
|
||||
if endpoint == "" {
|
||||
return ""
|
||||
}
|
||||
if err := s.WithFilerClient(func(client filer_pb.SeaweedFilerClient) error {
|
||||
resp, err := filer_pb.LookupEntry(context.Background(), client, &filer_pb.LookupDirectoryEntryRequest{
|
||||
Directory: "/buckets",
|
||||
Name: bucketName,
|
||||
})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if s3tables.IsTableBucketEntry(resp.Entry) {
|
||||
endpoint = ""
|
||||
}
|
||||
return nil
|
||||
}); err != nil {
|
||||
glog.V(1).Infof("object url bucket lookup failed for %s: %v", bucketName, err)
|
||||
return ""
|
||||
}
|
||||
return S3ObjectURL(endpoint, fullPath)
|
||||
}
|
||||
|
||||
@@ -93,6 +93,98 @@ func TestGenerateBreadcrumbs(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestS3ObjectURL verifies path-style S3 URL construction with per-segment
|
||||
// percent-encoding
|
||||
func TestS3ObjectURL(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
endpoint string
|
||||
fullPath string
|
||||
expected string
|
||||
}{
|
||||
{
|
||||
name: "simple object",
|
||||
endpoint: "https://s3.example.com",
|
||||
fullPath: "/buckets/public-images/homarr/example.png",
|
||||
expected: "https://s3.example.com/public-images/homarr/example.png",
|
||||
},
|
||||
{
|
||||
name: "space hash and question mark in key",
|
||||
endpoint: "https://s3.example.com",
|
||||
fullPath: "/buckets/b/a b#c?d.txt",
|
||||
expected: "https://s3.example.com/b/a%20b%23c%3Fd.txt",
|
||||
},
|
||||
{
|
||||
name: "non-ascii key",
|
||||
endpoint: "https://s3.example.com",
|
||||
fullPath: "/buckets/b/图片.png",
|
||||
expected: "https://s3.example.com/b/%E5%9B%BE%E7%89%87.png",
|
||||
},
|
||||
{
|
||||
name: "percent in key",
|
||||
endpoint: "https://s3.example.com",
|
||||
fullPath: "/buckets/b/100%.txt",
|
||||
expected: "https://s3.example.com/b/100%25.txt",
|
||||
},
|
||||
{
|
||||
name: "trailing slash on endpoint",
|
||||
endpoint: "https://s3.example.com/",
|
||||
fullPath: "/buckets/b/k",
|
||||
expected: "https://s3.example.com/b/k",
|
||||
},
|
||||
{
|
||||
name: "not a bucket path",
|
||||
endpoint: "https://s3.example.com",
|
||||
fullPath: "/topics/t/k",
|
||||
expected: "",
|
||||
},
|
||||
{
|
||||
name: "empty endpoint",
|
||||
endpoint: "",
|
||||
fullPath: "/buckets/b/k",
|
||||
expected: "",
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
if got := S3ObjectURL(tt.endpoint, tt.fullPath); got != tt.expected {
|
||||
t.Errorf("S3ObjectURL(%q, %q) = %q, expected %q", tt.endpoint, tt.fullPath, got, tt.expected)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestNormalizeS3PublicEndpoint verifies unusable configured endpoints are
|
||||
// dropped rather than producing broken links
|
||||
func TestNormalizeS3PublicEndpoint(t *testing.T) {
|
||||
tests := []struct {
|
||||
endpoint string
|
||||
expected string
|
||||
}{
|
||||
{"https://s3.example.com", "https://s3.example.com"},
|
||||
{"http://10.0.0.1:8333/", "http://10.0.0.1:8333"},
|
||||
{"http://[::1]:8333", "http://[::1]:8333"},
|
||||
{"https://proxy.example.com/s3", "https://proxy.example.com/s3"},
|
||||
{"", ""},
|
||||
{"/", ""},
|
||||
{"s3.example.com", ""},
|
||||
{"ftp://s3.example.com", ""},
|
||||
{"https://", ""},
|
||||
{"https://s3.example.com?x=1", ""},
|
||||
{"https://s3.example.com/?", ""},
|
||||
{"https://s3.example.com#frag", ""},
|
||||
{"https://s3.example.com/#", ""},
|
||||
{"http://user:pass@s3.example.com", ""},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
if got := normalizeS3PublicEndpoint(tt.endpoint); got != tt.expected {
|
||||
t.Errorf("normalizeS3PublicEndpoint(%q) = %q, expected %q", tt.endpoint, got, tt.expected)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestPathHandlingWithForwardSlashes verifies that the production code
|
||||
// correctly handles paths with forward slashes (not OS-specific backslashes)
|
||||
func TestPathHandlingWithForwardSlashes(t *testing.T) {
|
||||
|
||||
@@ -0,0 +1,73 @@
|
||||
package dash
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/worker/tasks/balance"
|
||||
"github.com/seaweedfs/seaweedfs/weed/worker/tasks/vacuum"
|
||||
)
|
||||
|
||||
// TestLoadMaintenanceConfigHonoursPersistedTaskConfigs guards against a regression: buildPolicyFromTaskConfigs used to call
|
||||
// LoadConfigFromPersistence(nil), which can never satisfy the loaders' type assertion, so every
|
||||
// task silently fell back to its compiled-in defaults (Enabled: true) and a task disabled in the
|
||||
// admin UI kept being scheduled.
|
||||
func TestLoadMaintenanceConfigHonoursPersistedTaskConfigs(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
cp := NewConfigPersistence(dir)
|
||||
|
||||
// A maintenance.pb must exist, otherwise LoadMaintenanceConfig returns early with defaults.
|
||||
if err := cp.SaveMaintenanceConfig(DefaultMaintenanceConfig()); err != nil {
|
||||
t.Fatalf("save maintenance config: %v", err)
|
||||
}
|
||||
|
||||
// Disable balance and vacuum the way the admin UI does, and change a value that is not a bool
|
||||
// so a fallback to defaults cannot pass by coincidence.
|
||||
disabledBalance := balance.NewDefaultConfig()
|
||||
disabledBalance.Enabled = false
|
||||
disabledBalance.MinServerCount = 7
|
||||
if err := cp.SaveBalanceTaskPolicy(disabledBalance.ToTaskPolicy()); err != nil {
|
||||
t.Fatalf("save balance policy: %v", err)
|
||||
}
|
||||
|
||||
disabledVacuum := vacuum.NewDefaultConfig()
|
||||
disabledVacuum.Enabled = false
|
||||
if err := cp.SaveVacuumTaskPolicy(disabledVacuum.ToTaskPolicy()); err != nil {
|
||||
t.Fatalf("save vacuum policy: %v", err)
|
||||
}
|
||||
|
||||
config, err := cp.LoadMaintenanceConfig()
|
||||
if err != nil {
|
||||
t.Fatalf("load maintenance config: %v", err)
|
||||
}
|
||||
if config.Policy == nil {
|
||||
t.Fatal("policy is nil, want it populated from the persisted task configs")
|
||||
}
|
||||
|
||||
balancePolicy := config.Policy.TaskPolicies["balance"]
|
||||
if balancePolicy == nil {
|
||||
t.Fatal("no balance task policy in the built maintenance policy")
|
||||
}
|
||||
if balancePolicy.Enabled {
|
||||
t.Error("balance enabled = true, want false from the persisted config")
|
||||
}
|
||||
if got := balancePolicy.GetBalanceConfig().GetMinServerCount(); got != 7 {
|
||||
t.Errorf("balance min server count = %d, want persisted 7", got)
|
||||
}
|
||||
|
||||
vacuumPolicy := config.Policy.TaskPolicies["vacuum"]
|
||||
if vacuumPolicy == nil {
|
||||
t.Fatal("no vacuum task policy in the built maintenance policy")
|
||||
}
|
||||
if vacuumPolicy.Enabled {
|
||||
t.Error("vacuum enabled = true, want false from the persisted config")
|
||||
}
|
||||
|
||||
// erasure_coding was never saved, so it keeps the loader's default of enabled.
|
||||
ecPolicy := config.Policy.TaskPolicies["erasure_coding"]
|
||||
if ecPolicy == nil {
|
||||
t.Fatal("no erasure_coding task policy in the built maintenance policy")
|
||||
}
|
||||
if !ecPolicy.Enabled {
|
||||
t.Error("erasure_coding enabled = false, want the default true for a config never saved")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,187 @@
|
||||
package dash
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/admin/maintenance"
|
||||
"github.com/seaweedfs/seaweedfs/weed/worker/tasks"
|
||||
"github.com/seaweedfs/seaweedfs/weed/worker/tasks/balance"
|
||||
"github.com/seaweedfs/seaweedfs/weed/worker/tasks/vacuum"
|
||||
"github.com/seaweedfs/seaweedfs/weed/worker/types"
|
||||
"google.golang.org/protobuf/proto"
|
||||
)
|
||||
|
||||
// The task definitions these tests configure are process-global, so put them back the way a
|
||||
// fresh process would have them. Passing no config store makes every task fall back to its
|
||||
// own NewDefaultConfig, which is exactly the state package init left them in.
|
||||
func restoreGlobalTaskState(t *testing.T) {
|
||||
t.Helper()
|
||||
|
||||
t.Cleanup(func() {
|
||||
tasks.GetGlobalConfigUpdateRegistry().UpdateAllConfigs(nil)
|
||||
})
|
||||
}
|
||||
|
||||
// TestDisabledTaskIsNotScannedAfterStartup walks the admin server's startup sequence over a
|
||||
// data directory that has a disabled balance task saved in it, and checks the end state that
|
||||
// actually matters: the balance detector reports disabled, so ScanWithTaskDetectors skips it.
|
||||
//
|
||||
// This is the whole reported bug in one test. The reporter disabled balance, and the
|
||||
// scanner kept detecting balance tasks, cancelling them and re-detecting them. Two separate
|
||||
// defects had to line up for the disabled flag to survive to here: the policy had to be built
|
||||
// from the persisted configs rather than from a nil store, and the policy had to reach
|
||||
// detector.IsEnabled() rather than dying in a failed type assertion.
|
||||
func TestDisabledTaskIsNotScannedAfterStartup(t *testing.T) {
|
||||
restoreGlobalTaskState(t)
|
||||
|
||||
dir := t.TempDir()
|
||||
cp := NewConfigPersistence(dir)
|
||||
|
||||
// What the admin writes when a user turns balance off, and leaves vacuum on.
|
||||
disabledBalance := balance.NewDefaultConfig()
|
||||
disabledBalance.Enabled = false
|
||||
if err := cp.SaveBalanceTaskPolicy(disabledBalance.ToTaskPolicy()); err != nil {
|
||||
t.Fatalf("save balance policy: %v", err)
|
||||
}
|
||||
|
||||
enabledVacuum := vacuum.NewDefaultConfig()
|
||||
enabledVacuum.Enabled = true
|
||||
if err := cp.SaveVacuumTaskPolicy(enabledVacuum.ToTaskPolicy()); err != nil {
|
||||
t.Fatalf("save vacuum policy: %v", err)
|
||||
}
|
||||
|
||||
// The admin server's startup sequence, in order:
|
||||
// loadTaskConfigurationsFromPersistence, then InitMaintenanceManager.
|
||||
tasks.GetGlobalConfigUpdateRegistry().UpdateAllConfigs(cp)
|
||||
|
||||
maintenanceConfig, err := cp.LoadMaintenanceConfig()
|
||||
if err != nil {
|
||||
t.Fatalf("load maintenance config: %v", err)
|
||||
}
|
||||
manager := maintenance.NewMaintenanceManager(nil, maintenanceConfig, cp)
|
||||
if manager == nil {
|
||||
t.Fatal("NewMaintenanceManager returned nil")
|
||||
}
|
||||
|
||||
registry := tasks.GetGlobalTypesRegistry()
|
||||
|
||||
balanceDetector := registry.GetDetector(types.TaskTypeBalance)
|
||||
if balanceDetector == nil {
|
||||
t.Fatal("no balance detector registered")
|
||||
}
|
||||
if balanceDetector.IsEnabled() {
|
||||
t.Error("balance detector reports enabled after startup over a data directory where " +
|
||||
"balance is saved as disabled; the scanner will keep detecting and cancelling balance tasks")
|
||||
}
|
||||
|
||||
vacuumDetector := registry.GetDetector(types.TaskTypeVacuum)
|
||||
if vacuumDetector == nil {
|
||||
t.Fatal("no vacuum detector registered")
|
||||
}
|
||||
if !vacuumDetector.IsEnabled() {
|
||||
t.Error("vacuum detector reports disabled although vacuum is saved as enabled; " +
|
||||
"the fix must not switch off tasks the user left on")
|
||||
}
|
||||
|
||||
// Tasks the user never touched keep their compiled-in default of enabled rather than
|
||||
// being switched off by a policy entry built from a config that was never saved.
|
||||
for _, taskType := range []types.TaskType{types.TaskTypeErasureCoding, types.TaskTypeECBalance} {
|
||||
detector := registry.GetDetector(taskType)
|
||||
if detector == nil {
|
||||
t.Fatalf("no %s detector registered", taskType)
|
||||
}
|
||||
if !detector.IsEnabled() {
|
||||
t.Errorf("%s detector reports disabled although its config was never saved", taskType)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestLoadMaintenanceConfigKeepsPersistedEnabledFalse covers the top of issue-shaped startup:
|
||||
// an operator persisted enabled=false, and the load path used to lose it twice over — once to
|
||||
// ApplyDefaultsToProtobuf treating the bool zero value as unset, and once to an explicit
|
||||
// force-enable "migration" block. The persisted flag must come back as saved, while fields the
|
||||
// old file never carried still pick up their schema defaults.
|
||||
func TestLoadMaintenanceConfigKeepsPersistedEnabledFalse(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
cp := NewConfigPersistence(dir)
|
||||
|
||||
// A file that only knows about the enabled flag: everything else zero.
|
||||
if err := cp.SaveMaintenanceConfig(&MaintenanceConfig{Enabled: proto.Bool(false)}); err != nil {
|
||||
t.Fatalf("save maintenance config: %v", err)
|
||||
}
|
||||
|
||||
loaded, err := cp.LoadMaintenanceConfig()
|
||||
if err != nil {
|
||||
t.Fatalf("load maintenance config: %v", err)
|
||||
}
|
||||
if loaded.GetEnabled() {
|
||||
t.Error("persisted enabled=false came back true; the maintenance system cannot be disabled")
|
||||
}
|
||||
if loaded.ScanIntervalSeconds != 30*60 {
|
||||
t.Errorf("scan interval = %d, want schema default 1800 filled in", loaded.ScanIntervalSeconds)
|
||||
}
|
||||
|
||||
// And with no file at all, the default is enabled.
|
||||
fresh, err := NewConfigPersistence(t.TempDir()).LoadMaintenanceConfig()
|
||||
if err != nil {
|
||||
t.Fatalf("load maintenance config: %v", err)
|
||||
}
|
||||
if !fresh.GetEnabled() {
|
||||
t.Error("maintenance not enabled by default when nothing is persisted")
|
||||
}
|
||||
}
|
||||
|
||||
// TestLoadMaintenanceConfigTreatsLegacyAbsentEnabledAsOn: a maintenance.pb written before
|
||||
// enabled tracked presence carries no enabled field on the wire whether the old default left
|
||||
// it false or an operator unchecked it — the two are indistinguishable. Such files must keep
|
||||
// the enabled default rather than silently switching maintenance off on upgrade; only a file
|
||||
// that explicitly persists the toggle may disable it.
|
||||
func TestLoadMaintenanceConfigTreatsLegacyAbsentEnabledAsOn(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
cp := NewConfigPersistence(dir)
|
||||
|
||||
// What an old writer produced for enabled=false plus a tuned scan interval: the bool is
|
||||
// simply absent from the wire.
|
||||
if err := cp.SaveMaintenanceConfig(&MaintenanceConfig{ScanIntervalSeconds: 15 * 60}); err != nil {
|
||||
t.Fatalf("save maintenance config: %v", err)
|
||||
}
|
||||
|
||||
loaded, err := cp.LoadMaintenanceConfig()
|
||||
if err != nil {
|
||||
t.Fatalf("load maintenance config: %v", err)
|
||||
}
|
||||
if !loaded.GetEnabled() {
|
||||
t.Error("legacy config without an enabled field loads as disabled; " +
|
||||
"upgrading would silently switch the maintenance system off")
|
||||
}
|
||||
if loaded.ScanIntervalSeconds != 15*60 {
|
||||
t.Errorf("scan interval = %d, want persisted 900 kept", loaded.ScanIntervalSeconds)
|
||||
}
|
||||
}
|
||||
|
||||
// TestPolicyMirrorsWhatTheDetectorsReport checks that the maintenance policy the queue and
|
||||
// the scanner run on agrees with the detectors. A disagreement means one of the two paths
|
||||
// into the task configs has gone stale again.
|
||||
func TestPolicyMirrorsWhatTheDetectorsReport(t *testing.T) {
|
||||
restoreGlobalTaskState(t)
|
||||
|
||||
dir := t.TempDir()
|
||||
cp := NewConfigPersistence(dir)
|
||||
|
||||
disabledBalance := balance.NewDefaultConfig()
|
||||
disabledBalance.Enabled = false
|
||||
if err := cp.SaveBalanceTaskPolicy(disabledBalance.ToTaskPolicy()); err != nil {
|
||||
t.Fatalf("save balance policy: %v", err)
|
||||
}
|
||||
|
||||
tasks.GetGlobalConfigUpdateRegistry().UpdateAllConfigs(cp)
|
||||
policy := cp.buildPolicyFromTaskConfigs()
|
||||
|
||||
for taskType, detector := range tasks.GetGlobalTypesRegistry().GetAllDetectors() {
|
||||
policyEnabled := maintenance.IsTaskEnabled(policy, maintenance.MaintenanceTaskType(taskType))
|
||||
if policyEnabled != detector.IsEnabled() {
|
||||
t.Errorf("%s: policy says enabled=%v but the detector says enabled=%v",
|
||||
taskType, policyEnabled, detector.IsEnabled())
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,57 @@
|
||||
package dash
|
||||
|
||||
import (
|
||||
"context"
|
||||
|
||||
weediam "github.com/seaweedfs/seaweedfs/weed/iam"
|
||||
"github.com/seaweedfs/seaweedfs/weed/iam/integration"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
)
|
||||
|
||||
// principalRoleArn builds the ARN SeaweedFS assigns a role by default (when
|
||||
// its RoleDefinition.RoleArn isn't explicitly set) - see
|
||||
// weed/iam/integration/iam_manager.go's CreateRole. ListRoles only returns
|
||||
// role names, so this reconstructs the well-known default rather than
|
||||
// fetching every role's stored definition just to populate a suggestion list.
|
||||
func principalRoleArn(roleName string) string {
|
||||
return "arn:aws:iam::role/" + roleName
|
||||
}
|
||||
|
||||
// GetPrincipalSuggestions returns candidate ARNs for the policy editor's
|
||||
// Principal/NotPrincipal autocomplete: one per S3 user, plus one per IAM
|
||||
// role. Service accounts are deliberately not listed separately - a service
|
||||
// account is just an additional credential for its parent user, so its ARN
|
||||
// is identical to the one already suggested for that user.
|
||||
//
|
||||
// Role listing is best-effort: if the filer or role store is unavailable,
|
||||
// the error is logged and suggestions fall back to users only, since an
|
||||
// incomplete autocomplete list is far less disruptive than blocking policy
|
||||
// editing over a suggestions-only feature.
|
||||
func (s *AdminServer) GetPrincipalSuggestions(ctx context.Context) ([]string, error) {
|
||||
var suggestions []string
|
||||
|
||||
users, err := s.GetObjectStoreUsers(ctx)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
for _, u := range users {
|
||||
suggestions = append(suggestions, weediam.UserArn(u.Username))
|
||||
}
|
||||
|
||||
roleStore, err := integration.NewFilerRoleStore(nil, func() string { return s.GetFilerAddress() })
|
||||
if err != nil {
|
||||
glog.Warningf("GetPrincipalSuggestions: failed to create role store: %v", err)
|
||||
return suggestions, nil
|
||||
}
|
||||
roleNames, err := roleStore.ListRoles(ctx, s.GetFilerAddress())
|
||||
if err != nil {
|
||||
glog.Warningf("GetPrincipalSuggestions: failed to list roles: %v", err)
|
||||
return suggestions, nil
|
||||
}
|
||||
for _, roleName := range roleNames {
|
||||
suggestions = append(suggestions, principalRoleArn(roleName))
|
||||
}
|
||||
|
||||
return suggestions, nil
|
||||
}
|
||||
@@ -0,0 +1,11 @@
|
||||
package dash
|
||||
|
||||
import "testing"
|
||||
|
||||
func TestPrincipalRoleArn(t *testing.T) {
|
||||
got := principalRoleArn("S3ReadOnlyRole")
|
||||
want := "arn:aws:iam::role/S3ReadOnlyRole"
|
||||
if got != want {
|
||||
t.Fatalf("principalRoleArn() = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
@@ -37,6 +37,12 @@ type S3TablesBucketSummary struct {
|
||||
// Format is empty for a bucket created before formats were declared. Such a
|
||||
// bucket takes tables of either format, which is what it always did.
|
||||
Format string `json:"format,omitempty"`
|
||||
// PolicyStatementCount is the number of statements in the table bucket's
|
||||
// resource policy, or 0 if it has none. Unrelated to the S3 bucket
|
||||
// policy mechanism (policy_engine.PolicyDocument / s3-bucket-policy):
|
||||
// S3 Tables stores its own s3tables.PolicyDocument under the
|
||||
// s3tables.policy extended attribute.
|
||||
PolicyStatementCount int `json:"policy_statement_count"`
|
||||
}
|
||||
|
||||
type S3TablesNamespacesData struct {
|
||||
@@ -144,11 +150,12 @@ func (s *AdminServer) GetS3TablesBucketsData(ctx context.Context) (S3TablesBucke
|
||||
continue
|
||||
}
|
||||
buckets = append(buckets, S3TablesBucketSummary{
|
||||
ARN: arn,
|
||||
Name: entry.Entry.Name,
|
||||
OwnerAccountID: metadata.OwnerAccountID,
|
||||
CreatedAt: metadata.CreatedAt,
|
||||
Format: metadata.Format,
|
||||
ARN: arn,
|
||||
Name: entry.Entry.Name,
|
||||
OwnerAccountID: metadata.OwnerAccountID,
|
||||
CreatedAt: metadata.CreatedAt,
|
||||
Format: metadata.Format,
|
||||
PolicyStatementCount: extractS3TablesPolicyStatementCountFromEntry(entry.Entry),
|
||||
})
|
||||
}
|
||||
return nil
|
||||
@@ -165,6 +172,23 @@ func (s *AdminServer) GetS3TablesBucketsData(ctx context.Context) (S3TablesBucke
|
||||
}, nil
|
||||
}
|
||||
|
||||
// extractS3TablesPolicyStatementCountFromEntry returns the number of
|
||||
// statements in the table bucket's resource policy, or 0 if it has none or
|
||||
// the stored JSON can't be parsed. Forgiving on parse failure, matching
|
||||
// extractPolicyStatementCountFromEntry (the S3 bucket policy equivalent in
|
||||
// admin_server.go, which is a different, unrelated policy mechanism).
|
||||
func extractS3TablesPolicyStatementCountFromEntry(entry *filer_pb.Entry) int {
|
||||
policyJSON := entry.Extended[s3tables.ExtendedKeyPolicy]
|
||||
if len(policyJSON) == 0 {
|
||||
return 0
|
||||
}
|
||||
var doc s3tables.PolicyDocument
|
||||
if err := json.Unmarshal(policyJSON, &doc); err != nil {
|
||||
return 0
|
||||
}
|
||||
return len(doc.Statement)
|
||||
}
|
||||
|
||||
// observedRowCounts collects what workers last reported for these tables. For a
|
||||
// format admin cannot read, this is the only row count that exists.
|
||||
func (s *AdminServer) observedRowCounts(bucketArn string, namespaceParts []string, tables []s3tables.TableSummary) map[string]string {
|
||||
@@ -1019,8 +1043,11 @@ func (s *AdminServer) GetS3TablesBucketPolicy(w http.ResponseWriter, r *http.Req
|
||||
getReq := &s3tables.GetTableBucketPolicyRequest{TableBucketARN: bucketArn}
|
||||
var resp s3tables.GetTableBucketPolicyResponse
|
||||
if err := s.executeS3TablesOperation(r.Context(), "GetTableBucketPolicy", getReq, &resp); err != nil {
|
||||
writeS3TablesError(w, err)
|
||||
return
|
||||
// No policy is a normal state for the UI (empty editor), not an error.
|
||||
if !isS3TablesNoSuchPolicy(err) {
|
||||
writeS3TablesError(w, err)
|
||||
return
|
||||
}
|
||||
}
|
||||
writeJSON(w, http.StatusOK, map[string]interface{}{"policy": resp.ResourcePolicy})
|
||||
}
|
||||
@@ -1087,8 +1114,10 @@ func (s *AdminServer) GetS3TablesTablePolicy(w http.ResponseWriter, r *http.Requ
|
||||
getReq := &s3tables.GetTablePolicyRequest{TableBucketARN: bucketArn, Namespace: namespaceParts, Name: name}
|
||||
var resp s3tables.GetTablePolicyResponse
|
||||
if err := s.executeS3TablesOperation(r.Context(), "GetTablePolicy", getReq, &resp); err != nil {
|
||||
writeS3TablesError(w, err)
|
||||
return
|
||||
if !isS3TablesNoSuchPolicy(err) {
|
||||
writeS3TablesError(w, err)
|
||||
return
|
||||
}
|
||||
}
|
||||
writeJSON(w, http.StatusOK, map[string]interface{}{"policy": resp.ResourcePolicy})
|
||||
}
|
||||
@@ -1201,6 +1230,11 @@ func writeS3TablesError(w http.ResponseWriter, err error) {
|
||||
writeJSONError(w, s3TablesErrorStatus(err), parseS3TablesErrorMessage(err))
|
||||
}
|
||||
|
||||
func isS3TablesNoSuchPolicy(err error) bool {
|
||||
var s3Err *s3tables.S3TablesError
|
||||
return errors.As(err, &s3Err) && s3Err.Type == s3tables.ErrCodeNoSuchPolicy
|
||||
}
|
||||
|
||||
func s3TablesErrorStatus(err error) int {
|
||||
var s3Err *s3tables.S3TablesError
|
||||
if errors.As(err, &s3Err) {
|
||||
|
||||
@@ -0,0 +1,184 @@
|
||||
package dash
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/worker_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/worker/tasks/balance"
|
||||
"github.com/seaweedfs/seaweedfs/weed/worker/tasks/ec_balance"
|
||||
"github.com/seaweedfs/seaweedfs/weed/worker/tasks/erasure_coding"
|
||||
"github.com/seaweedfs/seaweedfs/weed/worker/tasks/vacuum"
|
||||
"google.golang.org/protobuf/proto"
|
||||
)
|
||||
|
||||
// TestLoadTaskPolicyDefaultsMatchTaskDefaults pins the persistence layer's "nothing saved
|
||||
// yet" defaults to each task's own NewDefaultConfig(). They used to be a second, hand-written
|
||||
// copy and had drifted: with a data directory but no config file on disk, vacuum ran on a 24h
|
||||
// scan interval instead of 2h, balance on 6h with a 0.1 imbalance threshold instead of 30m
|
||||
// with 0.2, and erasure coding on 168h with a 0.90 fullness ratio and a 1024MB minimum volume
|
||||
// size instead of 1h with 0.95 and 30MB - none of which is what the admin UI shows as the
|
||||
// default for those fields.
|
||||
func TestLoadTaskPolicyDefaultsMatchTaskDefaults(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
want *worker_pb.TaskPolicy
|
||||
load func(cp *ConfigPersistence) (*worker_pb.TaskPolicy, error)
|
||||
}{
|
||||
{
|
||||
name: "vacuum",
|
||||
want: vacuum.NewDefaultConfig().ToTaskPolicy(),
|
||||
load: func(cp *ConfigPersistence) (*worker_pb.TaskPolicy, error) { return cp.LoadVacuumTaskPolicy() },
|
||||
},
|
||||
{
|
||||
name: "erasure_coding",
|
||||
want: erasure_coding.NewDefaultConfig().ToTaskPolicy(),
|
||||
load: func(cp *ConfigPersistence) (*worker_pb.TaskPolicy, error) {
|
||||
return cp.LoadErasureCodingTaskPolicy()
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "balance",
|
||||
want: balance.NewDefaultConfig().ToTaskPolicy(),
|
||||
load: func(cp *ConfigPersistence) (*worker_pb.TaskPolicy, error) { return cp.LoadBalanceTaskPolicy() },
|
||||
},
|
||||
{
|
||||
name: "ec_balance",
|
||||
want: ec_balance.NewDefaultConfig().ToTaskPolicy(),
|
||||
load: func(cp *ConfigPersistence) (*worker_pb.TaskPolicy, error) { return cp.LoadEcBalanceTaskPolicy() },
|
||||
},
|
||||
}
|
||||
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
// Both no-file branches have to agree with the task's own defaults: no data
|
||||
// directory at all, and a data directory that has never been written to.
|
||||
for _, cp := range []*ConfigPersistence{NewConfigPersistence(""), NewConfigPersistence(t.TempDir())} {
|
||||
got, err := tc.load(cp)
|
||||
if err != nil {
|
||||
t.Fatalf("load %s policy: %v", tc.name, err)
|
||||
}
|
||||
if !proto.Equal(got, tc.want) {
|
||||
t.Errorf("%s default policy (dataDir=%q) =\n %v\nwant NewDefaultConfig().ToTaskPolicy() =\n %v",
|
||||
tc.name, cp.GetDataDir(), got, tc.want)
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestLoadTaskConfigDefaultsMatchTaskDefaults covers the narrower Load*TaskConfig accessors,
|
||||
// which carried a third copy of the same defaults.
|
||||
func TestLoadTaskConfigDefaultsMatchTaskDefaults(t *testing.T) {
|
||||
cp := NewConfigPersistence(t.TempDir())
|
||||
|
||||
vacuumConfig, err := cp.LoadVacuumTaskConfig()
|
||||
if err != nil {
|
||||
t.Fatalf("load vacuum config: %v", err)
|
||||
}
|
||||
if want := vacuum.NewDefaultConfig().ToTaskPolicy().GetVacuumConfig(); !proto.Equal(vacuumConfig, want) {
|
||||
t.Errorf("vacuum default config = %v, want %v", vacuumConfig, want)
|
||||
}
|
||||
|
||||
ecConfig, err := cp.LoadErasureCodingTaskConfig()
|
||||
if err != nil {
|
||||
t.Fatalf("load erasure coding config: %v", err)
|
||||
}
|
||||
if want := erasure_coding.NewDefaultConfig().ToTaskPolicy().GetErasureCodingConfig(); !proto.Equal(ecConfig, want) {
|
||||
t.Errorf("erasure coding default config = %v, want %v", ecConfig, want)
|
||||
}
|
||||
|
||||
balanceConfig, err := cp.LoadBalanceTaskConfig()
|
||||
if err != nil {
|
||||
t.Fatalf("load balance config: %v", err)
|
||||
}
|
||||
if want := balance.NewDefaultConfig().ToTaskPolicy().GetBalanceConfig(); !proto.Equal(balanceConfig, want) {
|
||||
t.Errorf("balance default config = %v, want %v", balanceConfig, want)
|
||||
}
|
||||
}
|
||||
|
||||
// TestEcBalanceTaskPolicyRoundTrip checks the accessor ec_balance.LoadConfigFromPersistence
|
||||
// asserts on. Before it existed, ec_balance was the one registered maintenance task whose
|
||||
// configuration could not be persisted at all.
|
||||
func TestEcBalanceTaskPolicyRoundTrip(t *testing.T) {
|
||||
cp := NewConfigPersistence(t.TempDir())
|
||||
|
||||
saved := ec_balance.NewDefaultConfig()
|
||||
saved.Enabled = false
|
||||
saved.MinServerCount = 9
|
||||
saved.ImbalanceThreshold = 0.42
|
||||
saved.CollectionFilter = "pictures"
|
||||
|
||||
if err := cp.SaveEcBalanceTaskPolicy(saved.ToTaskPolicy()); err != nil {
|
||||
t.Fatalf("save ec_balance policy: %v", err)
|
||||
}
|
||||
|
||||
loaded := ec_balance.LoadConfigFromPersistence(cp)
|
||||
if loaded == nil {
|
||||
t.Fatal("ec_balance.LoadConfigFromPersistence returned nil")
|
||||
}
|
||||
if loaded.Enabled {
|
||||
t.Error("ec_balance enabled = true, want the persisted false")
|
||||
}
|
||||
if loaded.MinServerCount != 9 {
|
||||
t.Errorf("ec_balance min server count = %d, want the persisted 9", loaded.MinServerCount)
|
||||
}
|
||||
if loaded.ImbalanceThreshold != 0.42 {
|
||||
t.Errorf("ec_balance imbalance threshold = %v, want the persisted 0.42", loaded.ImbalanceThreshold)
|
||||
}
|
||||
if loaded.CollectionFilter != "pictures" {
|
||||
t.Errorf("ec_balance collection filter = %q, want the persisted %q", loaded.CollectionFilter, "pictures")
|
||||
}
|
||||
|
||||
// The generic dispatcher the maintenance manager uses has to know the type too, and
|
||||
// has to write the same file the dedicated loader reads.
|
||||
dispatched := NewConfigPersistence(t.TempDir())
|
||||
if err := dispatched.SaveTaskPolicy("ec_balance", saved.ToTaskPolicy()); err != nil {
|
||||
t.Fatalf("SaveTaskPolicy(ec_balance): %v", err)
|
||||
}
|
||||
roundTripped := ec_balance.LoadConfigFromPersistence(dispatched)
|
||||
if roundTripped == nil || roundTripped.MinServerCount != 9 {
|
||||
t.Errorf("SaveTaskPolicy(ec_balance) did not round-trip through the ec_balance store: %+v", roundTripped)
|
||||
}
|
||||
}
|
||||
|
||||
// TestBuildPolicyKeepsTaskSpecificFields guards the fields the hand-written policy builder
|
||||
// used to drop on the floor: the erasure coding preferred tags and replica placement, and
|
||||
// the balance IO rate limit. Building each entry from the task's own ToTaskPolicy() keeps
|
||||
// them, so a value set in admin.toml survives into the maintenance policy.
|
||||
func TestBuildPolicyKeepsTaskSpecificFields(t *testing.T) {
|
||||
cp := NewConfigPersistence(t.TempDir())
|
||||
|
||||
ecConfig := erasure_coding.NewDefaultConfig()
|
||||
ecConfig.PreferredTags = []string{"ssd", "archive"}
|
||||
ecConfig.ReplicaPlacement = "020"
|
||||
if err := cp.SaveErasureCodingTaskPolicy(ecConfig.ToTaskPolicy()); err != nil {
|
||||
t.Fatalf("save erasure coding policy: %v", err)
|
||||
}
|
||||
|
||||
balanceConfig := balance.NewDefaultConfig()
|
||||
balanceConfig.IoBytePerSecond = 5 << 20
|
||||
if err := cp.SaveBalanceTaskPolicy(balanceConfig.ToTaskPolicy()); err != nil {
|
||||
t.Fatalf("save balance policy: %v", err)
|
||||
}
|
||||
|
||||
policy := cp.buildPolicyFromTaskConfigs()
|
||||
|
||||
ecPolicy := policy.TaskPolicies["erasure_coding"].GetErasureCodingConfig()
|
||||
if ecPolicy == nil {
|
||||
t.Fatal("no erasure coding config in the built policy")
|
||||
}
|
||||
if got := ecPolicy.GetReplicaPlacement(); got != "020" {
|
||||
t.Errorf("erasure coding replica placement = %q, want the persisted %q", got, "020")
|
||||
}
|
||||
if got := ecPolicy.GetPreferredTags(); len(got) != 2 || got[0] != "ssd" || got[1] != "archive" {
|
||||
t.Errorf("erasure coding preferred tags = %v, want the persisted [ssd archive]", got)
|
||||
}
|
||||
|
||||
balancePolicy := policy.TaskPolicies["balance"].GetBalanceConfig()
|
||||
if balancePolicy == nil {
|
||||
t.Fatal("no balance config in the built policy")
|
||||
}
|
||||
if got := balancePolicy.GetIoBytePerSecond(); got != 5<<20 {
|
||||
t.Errorf("balance IO limit = %d, want the persisted %d", got, 5<<20)
|
||||
}
|
||||
}
|
||||
@@ -277,11 +277,7 @@ func (p *TopicRetentionPurger) deleteDirectoryRecursively(client filer_pb.Seawee
|
||||
}
|
||||
} else {
|
||||
// Delete file
|
||||
_, err = client.DeleteEntry(context.Background(), &filer_pb.DeleteEntryRequest{
|
||||
Directory: dirPath,
|
||||
Name: resp.Entry.Name,
|
||||
})
|
||||
if err != nil {
|
||||
if err := filer_pb.DoRemove(context.Background(), client, dirPath, resp.Entry.Name, false, false, false, false, nil); err != nil {
|
||||
return fmt.Errorf("failed to delete file %s: %v", entryPath, err)
|
||||
}
|
||||
}
|
||||
@@ -291,11 +287,7 @@ func (p *TopicRetentionPurger) deleteDirectoryRecursively(client filer_pb.Seawee
|
||||
parentDir := path.Dir(dirPath)
|
||||
dirName := path.Base(dirPath)
|
||||
|
||||
_, err = client.DeleteEntry(context.Background(), &filer_pb.DeleteEntryRequest{
|
||||
Directory: parentDir,
|
||||
Name: dirName,
|
||||
})
|
||||
if err != nil {
|
||||
if err := filer_pb.DoRemove(context.Background(), client, parentDir, dirName, false, false, false, false, nil); err != nil {
|
||||
return fmt.Errorf("failed to delete directory %s: %v", dirPath, err)
|
||||
}
|
||||
|
||||
|
||||
@@ -98,6 +98,12 @@ type S3Bucket struct {
|
||||
|
||||
LifecycleRuleCount int `json:"lifecycle_rule_count"`
|
||||
LifecycleEnabledCount int `json:"lifecycle_enabled_count"`
|
||||
|
||||
// PolicyStatementCount is the number of statements in the bucket policy,
|
||||
// or 0 if the bucket has none. A policy document can't have zero
|
||||
// statements (see policy_engine.ValidatePolicy), so >0 is a faithful
|
||||
// "has a policy" flag.
|
||||
PolicyStatementCount int `json:"policy_statement_count"`
|
||||
}
|
||||
|
||||
type S3Object struct {
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user