mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-10-10 08:35:50 +00:00
Compare commits
125
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
4d0afa3286 | ||
|
|
09f835842c | ||
|
|
37bf1cd91d | ||
|
|
a6d72bc272 | ||
|
|
934b9b4daf | ||
|
|
bd41ce39f7 | ||
|
|
0d6024e2e0 | ||
|
|
0ae7874ed9 | ||
|
|
81778defa1 | ||
|
|
39a8d3253c | ||
|
|
47f323bbb3 | ||
|
|
c72eda50a8 | ||
|
|
87ee3b63a2 | ||
|
|
0ca1c19821 | ||
|
|
f40687b34e | ||
|
|
15520f601f | ||
|
|
bdc37a1e86 | ||
|
|
2d2619f0b4 | ||
|
|
ce1e0dc30a | ||
|
|
d4e11a471d | ||
|
|
08d5daf0c1 | ||
|
|
8d34433308 | ||
|
|
4fd67001d9 | ||
|
|
c6a3280595 | ||
|
|
799c495226 | ||
|
|
4ec564469a | ||
|
|
66f1754896 | ||
|
|
994e1f7d64 | ||
|
|
74eeac6b66 | ||
|
|
0eb638f503 | ||
|
|
1a285c1334 | ||
|
|
caf3d157e6 | ||
|
|
3ebc05930d | ||
|
|
0c7beec697 | ||
|
|
a859f0a019 | ||
|
|
def25ca84d | ||
|
|
701e397337 | ||
|
|
4fc9ada2ec | ||
|
|
a73ba3adbb | ||
|
|
71f8128d75 | ||
|
|
01545fc4ff | ||
|
|
1f037e48f9 | ||
|
|
563c729e70 | ||
|
|
55367afded | ||
|
|
a9ecfeef45 | ||
|
|
beaf96a51d | ||
|
|
93d4a6aefd | ||
|
|
166af06a2b | ||
|
|
517f60e875 | ||
|
|
49a680dd64 | ||
|
|
87332eb60b | ||
|
|
e735c12869 | ||
|
|
e4ca0d09e7 | ||
|
|
01433e801d | ||
|
|
1f61097d4d | ||
|
|
0e82b4e351 | ||
|
|
0bd048b76f | ||
|
|
c997e54096 | ||
|
|
2aa6af033d | ||
|
|
02749c1192 | ||
|
|
ac03d3fd78 | ||
|
|
adaf3534fa | ||
|
|
49f20489e4 | ||
|
|
bdec508da9 | ||
|
|
fd33c07843 | ||
|
|
cf38c01978 | ||
|
|
f4bad510c9 | ||
|
|
15d9f6c6fe | ||
|
|
ea179963c0 | ||
|
|
c507336000 | ||
|
|
3c492b5ab1 | ||
|
|
38c14d3c13 | ||
|
|
92c379e5b4 | ||
|
|
5d8a463b3e | ||
|
|
bea10e269f | ||
|
|
10c0857476 | ||
|
|
c462fffce6 | ||
|
|
99d2479528 | ||
|
|
8db41d0217 | ||
|
|
bd6bcd47e3 | ||
|
|
eb6a7e93ca | ||
|
|
5b2fe374fc | ||
|
|
3b4a681e53 | ||
|
|
c46f82d29a | ||
|
|
5a0e017457 | ||
|
|
210afacd12 | ||
|
|
9f6feef299 | ||
|
|
79994b69af | ||
|
|
42b0ca7850 | ||
|
|
9b902a7662 | ||
|
|
d8aa7ecf04 | ||
|
|
2ebfeabfce | ||
|
|
a3638e479e | ||
|
|
80dae68dbf | ||
|
|
5ff49909a0 | ||
|
|
bc0efa4d10 | ||
|
|
3ae9e332ec | ||
|
|
0de9c1f231 | ||
|
|
b3aace2a08 | ||
|
|
4f9bbd51cb | ||
|
|
e919bec9d1 | ||
|
|
2cd6c36c54 | ||
|
|
7fa2f75f30 | ||
|
|
5061a16b12 | ||
|
|
3c9a4bbdda | ||
|
|
13bf056a15 | ||
|
|
c968084b34 | ||
|
|
516e251f9e | ||
|
|
01fc31cb71 | ||
|
|
966692fa23 | ||
|
|
168b9c39f8 | ||
|
|
f1f6886d0e | ||
|
|
2ffa696809 | ||
|
|
0ce5ca42ea | ||
|
|
9b12d13934 | ||
|
|
723f473f02 | ||
|
|
5f77a0b67e | ||
|
|
fd4fa72289 | ||
|
|
cb9fcd39d2 | ||
|
|
8782749f26 | ||
|
|
b88156fe6b | ||
|
|
557fffa350 | ||
|
|
4a1d65939f | ||
|
|
c6b330be2b | ||
|
|
213f4c5d5c |
@@ -27,7 +27,7 @@ jobs:
|
||||
|
||||
# Initializes the CodeQL tools for scanning.
|
||||
- name: Initialize CodeQL
|
||||
uses: github/codeql-action/init@v4.37.9
|
||||
uses: github/codeql-action/init@v4.38.0
|
||||
# Override language selection by uncommenting this and choosing your languages
|
||||
with:
|
||||
languages: go
|
||||
@@ -35,7 +35,7 @@ jobs:
|
||||
# Autobuild attempts to build any compiled languages (C/C++, C#, or Java).
|
||||
# If this step fails, then you should remove it and run the build manually (see below).
|
||||
- name: Autobuild
|
||||
uses: github/codeql-action/autobuild@v4.37.9
|
||||
uses: github/codeql-action/autobuild@v4.38.0
|
||||
|
||||
# ℹ️ Command-line programs to run using the OS shell.
|
||||
# 📚 See https://docs.github.com/en/actions/using-workflows/workflow-syntax-for-github-actions#jobsjob_idstepsrun
|
||||
@@ -49,4 +49,4 @@ jobs:
|
||||
# make release
|
||||
|
||||
- name: Perform CodeQL Analysis
|
||||
uses: github/codeql-action/analyze@v4.37.9
|
||||
uses: github/codeql-action/analyze@v4.38.0
|
||||
|
||||
@@ -405,7 +405,7 @@ jobs:
|
||||
output: trivy-results.sarif
|
||||
exit-code: '0'
|
||||
- name: Upload Trivy scan results to GitHub Security
|
||||
uses: github/codeql-action/upload-sarif@v4.37.9
|
||||
uses: github/codeql-action/upload-sarif@v4.38.0
|
||||
if: always()
|
||||
with:
|
||||
sarif_file: trivy-results.sarif
|
||||
|
||||
@@ -456,7 +456,7 @@ jobs:
|
||||
|
||||
- name: Upload Trivy scan results to GitHub Security
|
||||
if: always()
|
||||
uses: github/codeql-action/upload-sarif@v4.37.9
|
||||
uses: github/codeql-action/upload-sarif@v4.38.0
|
||||
with:
|
||||
sarif_file: trivy-results.sarif
|
||||
category: trivy-${{ matrix.variant }}
|
||||
|
||||
@@ -35,11 +35,48 @@ jobs:
|
||||
cd telemetry/server
|
||||
go mod tidy
|
||||
echo "Building telemetry server..."
|
||||
GOOS=linux GOARCH=amd64 go build -o ../../telemetry-server .
|
||||
CGO_ENABLED=0 GOOS=linux GOARCH=amd64 go build -o ../../telemetry-server .
|
||||
cd ../..
|
||||
ls -la telemetry-server
|
||||
echo "Build completed successfully"
|
||||
|
||||
- name: Generate Service Configuration
|
||||
if: github.event_name == 'workflow_dispatch' && (inputs.setup || inputs.deploy)
|
||||
env:
|
||||
REMOTE_USER: ${{ secrets.TELEMETRY_USER }}
|
||||
run: |
|
||||
# Create systemd service file
|
||||
echo "
|
||||
[Unit]
|
||||
Description=SeaweedFS Telemetry Server
|
||||
After=network.target
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
User=$REMOTE_USER
|
||||
WorkingDirectory=/home/$REMOTE_USER/seaweedfs-telemetry
|
||||
ExecStart=/bin/sh -c 'exec /home/$REMOTE_USER/seaweedfs-telemetry/bin/telemetry-server -port=8353 >>/home/$REMOTE_USER/seaweedfs-telemetry/logs/telemetry.log 2>>/home/$REMOTE_USER/seaweedfs-telemetry/logs/telemetry.error.log'
|
||||
Restart=always
|
||||
RestartSec=5
|
||||
|
||||
[Install]
|
||||
WantedBy=multi-user.target" > telemetry.service
|
||||
|
||||
# Setup logrotate configuration
|
||||
echo "# SeaweedFS Telemetry service log rotation
|
||||
/home/$REMOTE_USER/seaweedfs-telemetry/logs/*.log {
|
||||
daily
|
||||
rotate 30
|
||||
compress
|
||||
delaycompress
|
||||
missingok
|
||||
notifempty
|
||||
create 644 $REMOTE_USER $REMOTE_USER
|
||||
postrotate
|
||||
systemctl restart telemetry.service
|
||||
endscript
|
||||
}" > telemetry_logrotate
|
||||
|
||||
- name: First-time Server Setup
|
||||
if: github.event_name == 'workflow_dispatch' && inputs.setup
|
||||
env:
|
||||
@@ -61,40 +98,6 @@ jobs:
|
||||
touch ~/seaweedfs-telemetry/logs/telemetry.log ~/seaweedfs-telemetry/logs/telemetry.error.log && \
|
||||
chmod 644 ~/seaweedfs-telemetry/logs/*.log"
|
||||
|
||||
# Create systemd service file
|
||||
echo "
|
||||
[Unit]
|
||||
Description=SeaweedFS Telemetry Server
|
||||
After=network.target
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
User=$REMOTE_USER
|
||||
WorkingDirectory=/home/$REMOTE_USER/seaweedfs-telemetry
|
||||
ExecStart=/home/$REMOTE_USER/seaweedfs-telemetry/bin/telemetry-server -port=8353
|
||||
Restart=always
|
||||
RestartSec=5
|
||||
StandardOutput=append:/home/$REMOTE_USER/seaweedfs-telemetry/logs/telemetry.log
|
||||
StandardError=append:/home/$REMOTE_USER/seaweedfs-telemetry/logs/telemetry.error.log
|
||||
|
||||
[Install]
|
||||
WantedBy=multi-user.target" > telemetry.service
|
||||
|
||||
# Setup logrotate configuration
|
||||
echo "# SeaweedFS Telemetry service log rotation
|
||||
/home/$REMOTE_USER/seaweedfs-telemetry/logs/*.log {
|
||||
daily
|
||||
rotate 30
|
||||
compress
|
||||
delaycompress
|
||||
missingok
|
||||
notifempty
|
||||
create 644 $REMOTE_USER $REMOTE_USER
|
||||
postrotate
|
||||
systemctl restart telemetry.service
|
||||
endscript
|
||||
}" > telemetry_logrotate
|
||||
|
||||
# Copy configuration files
|
||||
scp -i ~/.ssh/deploy_key telemetry/grafana-dashboard.json $REMOTE_USER@$REMOTE_HOST:~/seaweedfs-telemetry/
|
||||
scp -i ~/.ssh/deploy_key telemetry/prometheus.yml $REMOTE_USER@$REMOTE_HOST:~/seaweedfs-telemetry/
|
||||
@@ -137,11 +140,18 @@ jobs:
|
||||
scp -i ~/.ssh/deploy_key telemetry/grafana-dashboard.json $REMOTE_USER@$REMOTE_HOST:~/seaweedfs-telemetry/
|
||||
scp -i ~/.ssh/deploy_key telemetry/prometheus.yml $REMOTE_USER@$REMOTE_HOST:~/seaweedfs-telemetry/
|
||||
|
||||
# Copy updated service and logrotate files
|
||||
scp -i ~/.ssh/deploy_key telemetry.service telemetry_logrotate $REMOTE_USER@$REMOTE_HOST:~/seaweedfs-telemetry/
|
||||
|
||||
# Check if service exists and deploy accordingly
|
||||
ssh -i ~/.ssh/deploy_key $REMOTE_USER@$REMOTE_HOST "
|
||||
if systemctl list-unit-files telemetry.service >/dev/null 2>&1; then
|
||||
echo 'Service exists, performing update...'
|
||||
set -e
|
||||
sudo systemctl stop telemetry.service
|
||||
sudo mv ~/seaweedfs-telemetry/telemetry.service /etc/systemd/system/
|
||||
sudo mv ~/seaweedfs-telemetry/telemetry_logrotate /etc/logrotate.d/seaweedfs-telemetry
|
||||
sudo systemctl daemon-reload
|
||||
mkdir -p ~/seaweedfs-telemetry/bin
|
||||
mv ~/seaweedfs-telemetry/tmp/telemetry-server ~/seaweedfs-telemetry/bin/
|
||||
chmod +x ~/seaweedfs-telemetry/bin/telemetry-server
|
||||
|
||||
@@ -227,6 +227,9 @@ jobs:
|
||||
out = render({
|
||||
"global.seaweedfs.securityConfig.jwtSigning.filerWrite": "true",
|
||||
"admin.enabled": "true",
|
||||
# admin.ip defaults to 0.0.0.0 (non-loopback), which weed admin 4.46
|
||||
# refuses to bind without authentication.
|
||||
"admin.secret.adminPassword": "ci-admin-password",
|
||||
})
|
||||
cm = configmap(out, "test-seaweedfs-security-config")
|
||||
if cm is None:
|
||||
@@ -1141,6 +1144,9 @@ jobs:
|
||||
"s3.enabled": "true",
|
||||
"sftp.enabled": "true",
|
||||
"admin.enabled": "true",
|
||||
# admin.ip defaults to 0.0.0.0 (non-loopback), which weed admin 4.46
|
||||
# refuses to bind without authentication.
|
||||
"admin.secret.adminPassword": "ci-admin-password",
|
||||
"worker.enabled": "true",
|
||||
"cosi.enabled": "true",
|
||||
"s3.createBuckets[0].name": "b",
|
||||
@@ -1337,7 +1343,7 @@ jobs:
|
||||
# Which means egress on its own must render for a release that runs
|
||||
# neither COSI nor a resize: no component of it reaches the API server,
|
||||
# so nothing may demand a CIDR for one.
|
||||
for label, values in {"defaults": {}, "admin": {"admin.enabled": "true"}}.items():
|
||||
for label, values in {"defaults": {}, "admin": {"admin.enabled": "true", "admin.secret.adminPassword": "ci-admin-password"}}.items():
|
||||
try:
|
||||
render(dict(values, **{"networkPolicy.enabled": "true",
|
||||
"networkPolicy.egress.enabled": "true"}))
|
||||
@@ -1398,6 +1404,9 @@ jobs:
|
||||
|
||||
ALL_ON = {
|
||||
"admin.enabled": "true",
|
||||
# admin.ip defaults to 0.0.0.0 (non-loopback), which weed admin 4.46
|
||||
# refuses to bind without authentication.
|
||||
"admin.secret.adminPassword": "ci-admin-password",
|
||||
"s3.enabled": "true",
|
||||
"sftp.enabled": "true",
|
||||
"worker.enabled": "true",
|
||||
|
||||
@@ -206,8 +206,9 @@ jobs:
|
||||
|
||||
# The dispatched workflow pins seaweedfs with `go get -u ...@latest`, so
|
||||
# wait until the proxy serves the release commit as the tip. Asking for
|
||||
# the commit by name is what makes the proxy fetch it.
|
||||
for _ in $(seq 30); do
|
||||
# the commit by name is what makes the proxy fetch it. The proxy can
|
||||
# take longer than five minutes to refresh @latest after a new tag.
|
||||
for _ in $(seq 120); do
|
||||
curl -sf "https://proxy.golang.org/${MODULE}/@v/${SHA}.info" >/dev/null || true
|
||||
TIP=$(curl -sf "https://proxy.golang.org/${MODULE}/@latest" | jq -r '.Origin.Hash // ""' || true)
|
||||
[ "$TIP" = "$SHA" ] && break
|
||||
|
||||
@@ -27,6 +27,29 @@ permissions:
|
||||
|
||||
|
||||
jobs:
|
||||
changes:
|
||||
name: Detect changed paths
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 5
|
||||
permissions:
|
||||
contents: read
|
||||
outputs:
|
||||
rust: ${{ steps.filter.outputs.rust }}
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v7
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Filter changed paths
|
||||
id: filter
|
||||
uses: dorny/paths-filter@v3
|
||||
with:
|
||||
filters: |
|
||||
rust:
|
||||
- 'seaweed-volume/**'
|
||||
- '.github/workflows/rust-volume-server-tests.yml'
|
||||
|
||||
rust-unit-tests:
|
||||
name: Rust Unit Tests
|
||||
runs-on: ubuntu-22.04
|
||||
@@ -59,6 +82,56 @@ jobs:
|
||||
- name: Build Rust volume server
|
||||
run: cd seaweed-volume && cargo build --release
|
||||
|
||||
# The crate is warning-free under clippy as of the sweep that added
|
||||
# this step. Uncomment to make that a gate; `[lints.clippy]` in
|
||||
# seaweed-volume/Cargo.toml is where crate-wide exceptions live.
|
||||
# - name: Clippy
|
||||
# run: cd seaweed-volume && cargo clippy --all-targets -- -D warnings
|
||||
|
||||
# The crate is rustfmt-clean as of the PR that added this step.
|
||||
# Uncomment to keep it that way.
|
||||
# - name: Check formatting
|
||||
# run: cd seaweed-volume && cargo fmt --check
|
||||
|
||||
- name: Run Rust unit tests
|
||||
run: cd seaweed-volume && cargo test
|
||||
|
||||
- name: Run Rust unit tests (redb experimental cursor)
|
||||
run: cd seaweed-volume && cargo test --features redb-experimental-cursor --lib storage::needle_map
|
||||
|
||||
rust-unit-tests-windows:
|
||||
name: Rust Unit Tests (Windows)
|
||||
runs-on: windows-latest
|
||||
timeout-minutes: 30
|
||||
needs: [changes]
|
||||
if: needs.changes.outputs.rust == 'true'
|
||||
defaults:
|
||||
run:
|
||||
shell: bash
|
||||
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@v7
|
||||
|
||||
- name: Install Rust toolchain
|
||||
uses: dtolnay/rust-toolchain@stable
|
||||
|
||||
# No glibc on Windows: key the cache on the toolchain and OS only.
|
||||
- name: Fingerprint build toolchain
|
||||
id: toolchain
|
||||
run: echo "fingerprint=windows-rustc-$(rustc -V | awk '{print $2}')" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Cache cargo registry and target
|
||||
uses: actions/cache@v6
|
||||
with:
|
||||
path: |
|
||||
~/.cargo/registry
|
||||
~/.cargo/git
|
||||
seaweed-volume/target
|
||||
key: rust-windows-${{ steps.toolchain.outputs.fingerprint }}-${{ hashFiles('seaweed-volume/Cargo.lock') }}
|
||||
restore-keys: |
|
||||
rust-windows-${{ steps.toolchain.outputs.fingerprint }}-
|
||||
|
||||
- name: Run Rust unit tests
|
||||
run: cd seaweed-volume && cargo test
|
||||
|
||||
|
||||
@@ -73,6 +73,17 @@ jobs:
|
||||
- name: Build the plugin workers
|
||||
run: cd seaweed-worker && cargo build --release
|
||||
|
||||
# The workspace is warning-free under clippy as of the sweep that added
|
||||
# this step. Uncomment to make that a gate; `[workspace.lints.clippy]`
|
||||
# in seaweed-worker/Cargo.toml is where crate-wide exceptions live.
|
||||
# - name: Clippy
|
||||
# run: cd seaweed-worker && cargo clippy --workspace --all-targets -- -D warnings
|
||||
|
||||
# The workspace is rustfmt-clean as of the PR that added this step.
|
||||
# Uncomment to keep it that way.
|
||||
# - name: Check formatting
|
||||
# run: cd seaweed-worker && cargo fmt --all --check
|
||||
|
||||
# The tests that need a live gateway skip themselves without one, the way
|
||||
# the Go integration tests skip without Docker; the lifecycle suite in
|
||||
# test/s3tables/lifecycle is what runs them against a real cluster.
|
||||
|
||||
@@ -9,6 +9,14 @@ on:
|
||||
permissions:
|
||||
contents: write
|
||||
|
||||
concurrency:
|
||||
# Only one chart regeneration per branch at a time; a newer run on the same
|
||||
# branch cancels an in-flight one so overlapping runs never conflict on
|
||||
# note/star_history.svg during rebase. Scoped by ref so a manual run on
|
||||
# another branch can't cancel the daily master update.
|
||||
group: star-history-${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
render:
|
||||
name: Regenerate star history chart
|
||||
@@ -17,6 +25,9 @@ jobs:
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v7
|
||||
with:
|
||||
# Full history so the chart commit can rebase onto a moved master.
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v7
|
||||
@@ -43,4 +54,21 @@ jobs:
|
||||
fi
|
||||
git add note/star_history.svg
|
||||
git commit -m "docs: regenerate star history chart"
|
||||
git push
|
||||
# Rebase and retry so a concurrent push to master doesn't lose the chart.
|
||||
for attempt in 1 2 3 4 5; do
|
||||
if [ "$attempt" -gt 1 ]; then
|
||||
# Guard the rebase: a transient fetch error or conflict must not
|
||||
# abort the fail-fast shell before the remaining attempts run.
|
||||
if ! git pull --rebase origin "$GITHUB_REF_NAME"; then
|
||||
echo "rebase failed (attempt ${attempt}); aborting and retrying"
|
||||
git rebase --abort || true
|
||||
continue
|
||||
fi
|
||||
fi
|
||||
if git push origin HEAD:"$GITHUB_REF_NAME"; then
|
||||
exit 0
|
||||
fi
|
||||
echo "push rejected (attempt ${attempt}); will rebase and retry"
|
||||
done
|
||||
echo "::error::could not push star history chart after retries"
|
||||
exit 1
|
||||
|
||||
@@ -17,7 +17,7 @@ SeaweedFS is a simple and highly scalable distributed file system. There are two
|
||||
1. to store billions of files!
|
||||
2. to serve the files fast!
|
||||
|
||||
One `weed` binary serves an S3 object store, a POSIX file system, and a lakehouse with S3 Tables, all over the same data. Each blob is one disk read away, capacity grows by starting another volume server, and cloud storage can be cached or tiered transparently.
|
||||
One `weed` binary serves an S3 object store, a POSIX file system, and a lakehouse with S3 Tables, all over the same data. Each blob is one disk read away, capacity grows by starting another volume server, and cloud storage can be cached or tiered transparently. Both read and write operations have O(1) complexity and can run at the full speed supported by the underlying hardware.
|
||||
|
||||
- [Download Binaries for different platforms](https://github.com/seaweedfs/seaweedfs/releases/latest)
|
||||
- [Wiki Documentation](https://github.com/seaweedfs/seaweedfs/wiki)
|
||||
@@ -400,8 +400,6 @@ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
|
||||
The text of this page is available for modification and reuse under the terms of the Creative Commons Attribution-Sharealike 3.0 Unported License and the GNU Free Documentation License (unversioned, with no invariant sections, front-cover texts, or back-cover texts).
|
||||
|
||||
[Back to TOC](#table-of-contents)
|
||||
|
||||
# Sponsors #
|
||||
|
||||
@@ -0,0 +1,418 @@
|
||||
# SeaweedFS as an Apache CloudStack Object Storage Provider
|
||||
|
||||
A CloudStack ObjectStore plugin that makes SeaweedFS a first-class object storage
|
||||
backend inside Apache CloudStack, alongside the existing MinIO and Ceph RGW
|
||||
providers. This is a collaboration with proIO (Swen), who builds private clouds on
|
||||
CloudStack and wants SeaweedFS as a storage option.
|
||||
|
||||
## The request
|
||||
|
||||
> We can only add MinIO and Ceph as object storage [in CloudStack] today. I want
|
||||
> to get SeaweedFS into this project... What we need is to build a provider which
|
||||
> does the communication between Cloudstack and SeaweedFS.
|
||||
|
||||
This is **not** a SeaweedFS-side feature. The work lives in the Apache CloudStack
|
||||
repo (Java): a new plugin under `plugins/storage/object/seaweedfs/` that implements
|
||||
CloudStack's ObjectStore plugin framework and talks to SeaweedFS over its S3 and
|
||||
IAM APIs. SeaweedFS itself needs no changes for the core to work — its S3 API
|
||||
already covers every bucket operation CloudStack requires, and its IAM API covers
|
||||
user/credential management.
|
||||
|
||||
## How the CloudStack ObjectStore framework works
|
||||
|
||||
CloudStack 4.18+ introduced an Object Storage framework. An admin registers an
|
||||
object storage pool via `addObjectStoragePool` (URL + provider + credentials);
|
||||
tenants then create and manage buckets on it through CloudStack APIs. CloudStack
|
||||
manages pool and bucket lifecycle; the underlying provider handles the actual
|
||||
object protocol.
|
||||
|
||||
A provider is a plugin module implementing three interfaces:
|
||||
|
||||
### 1. `ObjectStoreProvider` — registration
|
||||
|
||||
`MinIOObjectStoreProviderImpl` is the reference. It is a Spring `@Component` that:
|
||||
- Returns a provider name (`"MinIO"`)
|
||||
- Returns `DataStoreProviderType.OBJECT`
|
||||
- In `configure()`, injects the lifecycle and driver implementations and calls
|
||||
`storeMgr.registerDriver(name, driver)`
|
||||
|
||||
### 2. `ObjectStoreLifeCycle` — pool add/remove
|
||||
|
||||
`MinIOObjectStoreLifeCycleImpl.initialize()` reads the URL, name, and
|
||||
`accesskey`/`secretkey` details from the `addObjectStoragePool` call, tests the
|
||||
connection by listing buckets, and persists an `ObjectStoreVO` via
|
||||
`ObjectStoreHelper`. The other methods (attachCluster/Host/Zone, maintain,
|
||||
deleteDataStore) are no-ops for object storage.
|
||||
|
||||
### 3. `ObjectStoreDriver` — bucket + user operations
|
||||
|
||||
`ObjectStoreDriver` (in `engine/storage/.../object/ObjectStoreDriver.java`) extends
|
||||
`DataStoreDriver` and defines the bucket/user contract. Every provider must
|
||||
implement:
|
||||
|
||||
| Method | Purpose |
|
||||
| --- | --- |
|
||||
| `createBucket(Bucket, boolean objectLock)` | Create a bucket |
|
||||
| `listBuckets(long storeId)` | List all buckets |
|
||||
| `deleteBucket(BucketTO, long storeId)` | Delete a bucket |
|
||||
| `createUser(long accountId, long storeId)` | Provision a user + credentials for a CloudStack account |
|
||||
| `setBucketPolicy` / `getBucketPolicy` / `deleteBucketPolicy` | Bucket policy CRUD |
|
||||
| `setBucketEncryption` / `deleteBucketEncryption` | SSE config |
|
||||
| `setBucketVersioning` / `deleteBucketVersioning` | Versioning enable/suspend |
|
||||
| `setBucketQuota(BucketTO, long storeId, long size)` | Per-bucket quota |
|
||||
| `getAllBucketsUsage(long storeId)` | Usage map for billing/accounting |
|
||||
| `getBucketAcl` / `setBucketAcl` | ACLs (MinIO/Ceph return null / no-op) |
|
||||
|
||||
`BaseObjectStoreDriverImpl` provides no-op defaults for the `DataStoreDriver`
|
||||
methods (`createAsync`, `deleteAsync`, `copyAsync`, `canCopy`, `resize`,
|
||||
`getTO`, `getStoreTO`), so object-store providers only implement the bucket/user
|
||||
methods above.
|
||||
|
||||
## How the four existing providers differ (and where SeaweedFS lands)
|
||||
|
||||
CloudStack ships four object-store providers. Three are relevant; the simulator
|
||||
is a test stub.
|
||||
|
||||
| Concern | MinIO | Ceph RGW | Cloudian HyperStore | SeaweedFS |
|
||||
| --- | --- | --- | --- | --- |
|
||||
| Bucket CRUD | `MinioClient` (S3) | `AmazonS3` (AWS SDK v1) | `AmazonS3` (AWS SDK v1) | `AmazonS3` (AWS SDK v1) |
|
||||
| Bucket policy | `MinioClient` | `AmazonS3` | `AmazonS3` | `AmazonS3` |
|
||||
| Versioning | `MinioClient` | `AmazonS3` | `AmazonS3` | `AmazonS3` |
|
||||
| Encryption | `MinioClient` | not implemented | `AmazonS3` | `AmazonS3` |
|
||||
| **User creation** | `MinioAdminClient` | `RgwAdmin` | **`AmazonIdentityManagement`** | **`AmazonIdentityManagement`** |
|
||||
| **Per-bucket quota** | `MinioAdminClient` | `RgwAdmin` | **not supported** (throws) | **S3 extension** (`PUT /{bucket}?seaweedfs-quota`, SigV4, `s3:PutBucketQuota`) |
|
||||
| **Usage reporting** | `MinioAdminClient` | `RgwAdmin` | Cloudian admin API | S3 `ListObjectsV2` (MVP); Prometheus / SOSAPI `capacity.xml` (recommended) |
|
||||
|
||||
**Cloudian HyperStore is the direct precedent.** It is an S3-compatible store
|
||||
that, like SeaweedFS, manages users via the **standard AWS IAM API** using the
|
||||
AWS IAM Java SDK (`com.amazonaws.services.identitymanagement`). Its driver
|
||||
(`CloudianHyperStoreObjectStoreDriverImpl`) and util
|
||||
(`CloudianHyperStoreUtil`) are the template this design follows almost line for
|
||||
line. Cloudian even validates the quota limitation the same way this design
|
||||
proposes for the MVP: `setBucketQuota` throws for any non-zero size and only
|
||||
accepts `0` (no quota).
|
||||
|
||||
The SeaweedFS plugin is therefore a **simpler Cloudian** — same AWS S3 + IAM SDK
|
||||
clients, same store-details keys (`s3Url`, `iamUrl`, `accesskey`, `secretkey`),
|
||||
same IAM-user-with-restricted-policy pattern, but with no proprietary admin
|
||||
client at all (Cloudian has its own `CloudianClient` for its admin API; SeaweedFS
|
||||
needs only S3 + IAM). For quota, the plugin uses a narrow SeaweedFS S3 extension
|
||||
(see below); for usage reporting, it falls back to S3 `ListObjectsV2` in the MVP
|
||||
and recommends Prometheus or SOSAPI `capacity.xml` for production scale.
|
||||
|
||||
### Quota via the S3 `?seaweedfs-quota` extension
|
||||
|
||||
SeaweedFS supports bucket quota natively (server-side enforcement via a
|
||||
read-only flag when usage exceeds the limit). Rather than exposing the broad
|
||||
admin REST API (which would require a global bearer token and grant cluster-wide
|
||||
admin access), the integration uses a **narrow, scoped S3 subresource**:
|
||||
|
||||
- `PUT /{bucket}?seaweedfs-quota` — set bucket quota (IAM permission `s3:PutBucketQuota`)
|
||||
- `GET /{bucket}?seaweedfs-quota` — get bucket quota (IAM permission `s3:GetBucketQuota`)
|
||||
|
||||
**PUT request body** (JSON):
|
||||
```json
|
||||
{"quota_size": 100, "quota_unit": "GB", "quota_enabled": true}
|
||||
```
|
||||
|
||||
**GET response body** (JSON):
|
||||
```json
|
||||
{"quota_size": 107374182400, "quota_unit": "B", "quota_enabled": true}
|
||||
```
|
||||
|
||||
Note: GET always returns `quota_unit: "B"` and the absolute byte count, not
|
||||
the original unit. A disabled-but-retained quota returns a positive
|
||||
`quota_size` with `quota_enabled: false`.
|
||||
|
||||
Quota is stored on the bucket's filer entry (positive = enabled, negative =
|
||||
disabled but retained, zero = no quota), matching the existing admin REST API
|
||||
behavior. When quota is cleared, the bucket's read-only flag is also lifted.
|
||||
|
||||
**Authentication** uses the existing S3 SigV4 flow — no new global secret is
|
||||
needed. The CloudStack service credential (the `accesskey`/`secretkey` on the
|
||||
object store) is the admin credential used for all driver operations: bucket
|
||||
CRUD, IAM user provisioning, and quota management. It must have broad S3 and
|
||||
IAM permissions. The per-account IAM users created by `createUser` are the
|
||||
ones with restricted permissions (full S3 access except bucket
|
||||
creation/deletion). A future hardening could split quota management onto a
|
||||
separate credential scoped to only `s3:PutBucketQuota`/`s3:GetBucketQuota`,
|
||||
but the MVP uses the single admin credential for simplicity, matching how
|
||||
the MinIO and Ceph providers work.
|
||||
|
||||
The plugin's `setBucketQuota` signs and sends the `PUT /{bucket}?seaweedfs-quota`
|
||||
request using the AWS SDK v1 `AWSS3V4Signer` for SigV4 signing, then sends the
|
||||
signed request via `java.net.http.HttpClient` (the AWS S3 SDK doesn't natively
|
||||
support custom subresources, so we sign manually and send the request
|
||||
ourselves). The `seaweedfs-quota` query parameter is included in the signed
|
||||
canonical query string.
|
||||
|
||||
### Usage reporting
|
||||
|
||||
`getAllBucketsUsage` must return a `Map<String, Long>` of bucket name → size.
|
||||
MinIO uses `MinioAdminClient.getDataUsageInfo`; Ceph uses
|
||||
`RgwAdmin.listBucketInfo`. SeaweedFS has no admin rollup endpoint, so the MVP
|
||||
plugin computes it by listing buckets and summing object sizes via S3
|
||||
`ListObjectsV2` — expensive for large stores.
|
||||
|
||||
For production scale, SeaweedFS already exposes per-bucket size in:
|
||||
- **Prometheus metrics** (`bucket_size_bytes` gauge, refreshed every minute)
|
||||
- **SOSAPI `capacity.xml`** (reports capacity, available space, and usage
|
||||
through the S3 endpoint)
|
||||
|
||||
Operators should consume one of those instead of S3 list-based aggregation for
|
||||
large deployments. The MVP's list-based approach is correct but slow; flag it as
|
||||
a known limitation.
|
||||
|
||||
## SeaweedFS API surface (what the plugin relies on)
|
||||
|
||||
SeaweedFS exposes two relevant APIs, both AWS-compatible:
|
||||
|
||||
### S3 API (`weed s3`)
|
||||
Full S3-compatible surface. Confirmed against the SeaweedFS S3 wiki and code:
|
||||
- `CreateBucket`, `HeadBucket`, `ListBuckets`, `DeleteBucket`
|
||||
- `PutBucketPolicy`, `GetBucketPolicy`, `DeleteBucketPolicy`
|
||||
- `PutBucketVersioning` (Enabled / Suspended), `GetBucketVersioning`
|
||||
- `PutBucketEncryption`, `GetBucketEncryption`, `DeleteBucketEncryption`
|
||||
- `PutBucketAcl`, `GetBucketAcl`
|
||||
- `ListObjectsV2`, `HeadObject`, `GetObject`, `PutObject`, `DeleteObject`
|
||||
- Bucket quota via extended attributes / `s3.bucket.quota` (enforced server-side,
|
||||
surfaced as a read-only state when exceeded — see PR #10224)
|
||||
|
||||
### IAM API (`weed iam` / `iamapi`)
|
||||
AWS IAM-compatible REST endpoints, implemented in `weed/iamapi/`. Confirmed by
|
||||
the test suite which uses the **AWS IAM SDK** (`aws-sdk-go/service/iam`) against
|
||||
the same handlers CloudStack would call:
|
||||
- `CreateUser`, `DeleteUser`, `ListUsers`, `GetUser`
|
||||
- `CreateAccessKey`, `DeleteAccessKey`, `ListAccessKeys`
|
||||
- `PutUserPolicy`, `GetUserPolicy`, `DeleteUserPolicy`
|
||||
- `AttachUserPolicy`, `ListAttachedUserPolicies`
|
||||
|
||||
This means the CloudStack plugin can manage SeaweedFS users with the **AWS IAM
|
||||
Java SDK** (`com.amazonaws.services.identitymanagement.AmazonIdentityManagement`),
|
||||
exactly the way the AWS IAM Go SDK is used in SeaweedFS's own tests. No proprietary
|
||||
admin client is needed. **Cloudian HyperStore already does exactly this** in the
|
||||
CloudStack tree — the SeaweedFS plugin follows the same pattern.
|
||||
|
||||
## Design
|
||||
|
||||
### Module layout
|
||||
|
||||
New CloudStack plugin module, mirroring `plugins/storage/object/cloudian/`
|
||||
(the closest precedent — same AWS S3 + IAM SDK approach):
|
||||
|
||||
```
|
||||
plugins/storage/object/seaweedfs/
|
||||
pom.xml
|
||||
src/main/java/org/apache/cloudstack/storage/datastore/
|
||||
driver/SeaweedFSObjectStoreDriverImpl.java
|
||||
lifecycle/SeaweedFSObjectStoreLifeCycleImpl.java
|
||||
provider/SeaweedFSObjectStoreProviderImpl.java
|
||||
util/SeaweedFSObjectStoreUtil.java
|
||||
src/test/java/org/apache/cloudstack/storage/datastore/
|
||||
driver/SeaweedFSObjectStoreDriverImplTest.java
|
||||
provider/SeaweedFSObjectStoreProviderImplTest.java
|
||||
src/main/resources/META-INF/cloudstack/storage-object-seaweedfs/
|
||||
module.properties
|
||||
spring-storage-object-seaweedfs-context.xml
|
||||
```
|
||||
|
||||
### `SeaweedFSObjectStoreProviderImpl`
|
||||
|
||||
Direct copy of `MinIOObjectStoreProviderImpl` with `providerName = "SeaweedFS"`,
|
||||
injecting the SeaweedFS lifecycle and driver. Registers via
|
||||
`storeMgr.registerDriver`.
|
||||
|
||||
### `SeaweedFSObjectStoreLifeCycleImpl`
|
||||
|
||||
Copy of `MinIOObjectStoreLifeCycleImpl`. `initialize()` reads `url`, `name`,
|
||||
`accesskey`, `secretkey` from the `addObjectStoragePool` details map, tests the
|
||||
connection by calling `AmazonS3.listBuckets()` against the SeaweedFS S3 endpoint,
|
||||
and persists the `ObjectStoreVO`. No proprietary client needed — the AWS S3 SDK
|
||||
is enough for the health check.
|
||||
|
||||
### `SeaweedFSObjectStoreDriverImpl`
|
||||
|
||||
The substantive class. Uses two AWS SDK v1 clients (same dependency Ceph already
|
||||
pulls in, so no new CloudStack dependency):
|
||||
|
||||
- `AmazonS3` for bucket operations (path-style, endpoint-pinned, `us-east-1`
|
||||
region placeholder — same as Ceph's `getS3Client`)
|
||||
- `AmazonIdentityManagement` for user/credential operations, pointed at the
|
||||
SeaweedFS IAM endpoint
|
||||
|
||||
#### Bucket operations — straightforward S3
|
||||
|
||||
| Interface method | Implementation |
|
||||
| --- | --- |
|
||||
| `createBucket` | `s3.createBucket(name)`; reject if `doesBucketExistV2`; persist access/secret key + URL on `BucketVO` (same as Ceph) |
|
||||
| `listBuckets` | `s3.listBuckets()` → wrap as `BucketObject` (same as Ceph) |
|
||||
| `deleteBucket` | `s3.deleteBucket(name)` (same as Ceph) |
|
||||
| `setBucketPolicy` | `s3.setBucketPolicy(...)` with the same public/private JSON the MinIO/Ceph drivers build |
|
||||
| `getBucketPolicy` / `deleteBucketPolicy` | `s3.getBucketPolicy` / `s3.deleteBucketPolicy` |
|
||||
| `setBucketVersioning` | `s3.setBucketVersioningConfiguration(Enabled)` |
|
||||
| `deleteBucketVersioning` | `s3.setBucketVersioningConfiguration(Suspended)` |
|
||||
| `setBucketEncryption` | `s3.setBucketEncryptionConfiguration(SSE-S3 rule)` |
|
||||
| `deleteBucketEncryption` | `s3.deleteBucketEncryptionConfiguration` |
|
||||
| `getBucketAcl` / `setBucketAcl` | no-op / null (same as MinIO and Ceph) |
|
||||
|
||||
#### User creation — the key difference
|
||||
|
||||
MinIO calls `MinioAdminClient.addUser`; Ceph calls `RgwAdmin.createUser`. SeaweedFS
|
||||
exposes the standard AWS IAM API, so the plugin calls:
|
||||
|
||||
```java
|
||||
AmazonIdentityManagement iam = getIamClient(storeId);
|
||||
String userName = "acs-" + account.getUuid();
|
||||
|
||||
// CreateUser (idempotent — check GetUser first, like Ceph does)
|
||||
iam.createUser(new CreateUserRequest(userName));
|
||||
|
||||
// CreateAccessKey → returns the access key + secret key to persist
|
||||
CreateAccessKeyResult result = iam.createAccessKey(
|
||||
new CreateAccessKeyRequest().withUserName(userName));
|
||||
AccessKey key = result.getAccessKey();
|
||||
|
||||
// Persist per-account, same pattern as Ceph's CEPH_ACCESS_KEY/CEPH_SECRET_KEY
|
||||
details.put(SEAWEEDFS_ACCESS_KEY, key.getAccessKeyId());
|
||||
details.put(SEAWEEDFS_SECRET_KEY, key.getSecretAccessKey());
|
||||
_accountDetailsDao.persist(accountId, details);
|
||||
```
|
||||
|
||||
This is the cleanest mapping of the three providers: no proprietary admin client,
|
||||
just the AWS IAM SDK that CloudStack already has access to. The IAM endpoint URL
|
||||
is provided as `iamUrl` in the store details. If `iamUrl` is omitted, the driver
|
||||
defaults it to `s3Url` — SeaweedFS registers its embedded IAM API at `POST /` on
|
||||
the same S3 endpoint (`UnifiedPostHandler` in `s3api_server.go`), so the IAM
|
||||
endpoint is the same as the S3 endpoint unless the deployment runs a separate
|
||||
`weed iam` server.
|
||||
|
||||
#### Bucket quota — S3 `?seaweedfs-quota` extension
|
||||
|
||||
This is the one genuine gap. MinIO and Ceph both have an admin API to set a
|
||||
per-bucket quota that the backend enforces. SeaweedFS enforces bucket quota
|
||||
server-side, but the configuration path was **not exposed over a standard S3 or
|
||||
IAM API** — it was only set via the admin REST API or shell commands.
|
||||
|
||||
The integration adds a **narrow S3 subresource** to SeaweedFS:
|
||||
- `PUT /{bucket}?seaweedfs-quota` — set bucket quota (IAM permission `s3:PutBucketQuota`)
|
||||
- `GET /{bucket}?seaweedfs-quota` — get bucket quota (IAM permission `s3:GetBucketQuota`)
|
||||
|
||||
This is implemented in SeaweedFS PR #11279. It uses SigV4 authentication and
|
||||
dedicated IAM permissions, so the CloudStack service credential can be scoped
|
||||
to quota management only — no global admin token, no cluster-wide admin access.
|
||||
The enforcement already exists (PR #10224); this PR only adds the HTTP
|
||||
configuration surface.
|
||||
|
||||
An earlier approach (PR #11278, closed) added bearer-token auth to the broad
|
||||
admin REST API. After review, that was unnecessary for this integration —
|
||||
static S3 config plus standard S3 APIs plus one scoped quota mutation API is
|
||||
sufficient and far safer.
|
||||
|
||||
> **Note on AWS tools compatibility.** `?seaweedfs-quota` is a SeaweedFS-specific
|
||||
> S3 subresource, not part of the AWS S3 API. Standard AWS tools (`aws s3api`,
|
||||
> `s3cmd`, `rclone`) cannot call it directly. This is the same limitation MinIO
|
||||
> and Ceph have — MinIO quota lives behind a separate admin API (`mc admin
|
||||
> bucket quota`), and Ceph quota lives behind the Admin Ops API
|
||||
> (`radosgw-admin quota set`). Neither is callable via `aws s3api` either.
|
||||
> SeaweedFS's approach is the closest to standard S3 because it uses the same
|
||||
> endpoint and same SigV4 credentials, just with a custom query parameter.
|
||||
> Interactive quota management remains available via `weed shell`; the S3
|
||||
> extension exists for programmatic integration (CloudStack) where the
|
||||
> integrator can sign SigV4 requests but cannot run shell commands.
|
||||
|
||||
#### Usage reporting
|
||||
|
||||
`getAllBucketsUsage` must return a `Map<String, Long>` of bucket name → size.
|
||||
MinIO uses `MinioAdminClient.getDataUsageInfo`; Ceph uses
|
||||
`RgwAdmin.listBucketInfo`. SeaweedFS has no admin rollup endpoint, so the MVP
|
||||
plugin computes it by listing buckets and summing object sizes via S3
|
||||
`ListObjectsV2` — expensive for large stores. Better options exist in
|
||||
SeaweedFS already:
|
||||
- **Prometheus metrics** (`bucket_size_bytes` gauge, refreshed every minute)
|
||||
- **SOSAPI `capacity.xml`** (reports capacity, available space, and usage
|
||||
through the S3 endpoint — note: the current "return zero on backend error"
|
||||
behavior should be validated before using it for billing)
|
||||
|
||||
For the MVP, `listBuckets` + per-bucket size via the S3 API is correct but slow;
|
||||
flag it as a known limitation. Operators should consume Prometheus or SOSAPI
|
||||
for production-scale usage reporting.
|
||||
|
||||
### Spring wiring
|
||||
|
||||
`spring-storage-object-seaweedfs-context.xml` registers the provider bean,
|
||||
identical to the MinIO one. `module.properties` sets
|
||||
`name=storage-object-seaweedfs`, `parent=storage`.
|
||||
|
||||
### `pom.xml`
|
||||
|
||||
Depends on `aws-java-sdk-s3` and `aws-java-sdk-iam` — both already in the
|
||||
CloudStack dependency tree (Ceph uses the S3 SDK; the IAM SDK is the standard AWS
|
||||
bundle). No new third-party dependency, unlike MinIO which pulls in the MinIO
|
||||
Java client.
|
||||
|
||||
## What changes on the SeaweedFS side
|
||||
|
||||
**One narrow S3 extension is required for quota management.** SeaweedFS PR #11279
|
||||
adds the `?seaweedfs-quota` S3 subresource:
|
||||
|
||||
- `PUT /{bucket}?seaweedfs-quota` — set bucket quota (IAM permission `s3:PutBucketQuota`)
|
||||
- `GET /{bucket}?seaweedfs-quota` — get bucket quota (IAM permission `s3:GetBucketQuota`)
|
||||
|
||||
This is authenticated via standard S3 SigV4 and authorized via dedicated IAM
|
||||
permissions, so no global admin token is needed. The enforcement already exists
|
||||
(PR #10224); this PR only adds the HTTP configuration surface.
|
||||
|
||||
One follow-up improvement on the SeaweedFS side would close the usage reporting
|
||||
gap:
|
||||
|
||||
1. **Validate SOSAPI `capacity.xml` usage calculation** — the current "return
|
||||
zero on backend error" behavior should be validated before using it for
|
||||
billing. If reliable, CloudStack can consume it directly instead of
|
||||
list-based aggregation.
|
||||
|
||||
## Open questions for proIO / Swen
|
||||
|
||||
1. **IAM endpoint path.** ~~Where does `weed iam` listen relative to the S3
|
||||
endpoint in a typical proIO deployment?~~ **Resolved.** SeaweedFS registers
|
||||
its embedded IAM API at `POST /` on the same S3 endpoint
|
||||
(`UnifiedPostHandler`), so the driver defaults `iamUrl` to `s3Url`. A
|
||||
separate `iamUrl` is only needed if the deployment runs a standalone
|
||||
`weed iam` server on a different host/port.
|
||||
2. **Quota requirements.** Do proIO's customers need server-enforced per-bucket
|
||||
quotas, or is CloudStack-side accounting sufficient for the first release?
|
||||
The `?seaweedfs-quota` S3 extension (PR #11279) provides server-enforced
|
||||
quotas via a scoped credential; this is the recommended path.
|
||||
3. **Object Lock.** `createBucket` takes an `objectLock` boolean. MinIO supports
|
||||
it; Ceph ignores it. SeaweedFS has Object Lock support. Should the plugin pass
|
||||
it through?
|
||||
4. **Contribution model.** Does proIO want to submit the PR to
|
||||
`apache/cloudstack` themselves (with SeaweedFS maintainers as reviewers), or
|
||||
the reverse? Apache CloudStack requires an ICLA for non-trivial contributions.
|
||||
|
||||
## Files
|
||||
|
||||
All in the `apache/cloudstack` repo (new module):
|
||||
|
||||
| File | Purpose |
|
||||
| --- | --- |
|
||||
| `plugins/storage/object/seaweedfs/pom.xml` | Maven module |
|
||||
| `.../datastore/util/SeaweedFSObjectStoreUtil.java` | S3 + IAM client builders, constants, URL validators |
|
||||
| `.../datastore/provider/SeaweedFSObjectStoreProviderImpl.java` | Spring provider registration |
|
||||
| `.../datastore/lifecycle/SeaweedFSObjectStoreLifeCycleImpl.java` | Pool add/health-check |
|
||||
| `.../datastore/driver/SeaweedFSObjectStoreDriverImpl.java` | Bucket + user ops via S3 + IAM SDK |
|
||||
| `.../resources/META-INF/cloudstack/storage-object-seaweedfs/module.properties` | Module name |
|
||||
| `.../resources/META-INF/cloudstack/storage-object-seaweedfs/spring-storage-object-seaweedfs-context.xml` | Spring bean |
|
||||
| `plugins/pom.xml` | Register `storage/object/seaweedfs` module |
|
||||
|
||||
No files in `seaweedfs/seaweedfs` for the MVP.
|
||||
|
||||
### SeaweedFS-side changes (PR #11279)
|
||||
|
||||
| File | Purpose |
|
||||
| --- | --- |
|
||||
| `weed/s3api/s3_constants/s3_action_strings.go` | Add `S3_ACTION_PUT_BUCKET_QUOTA` and `S3_ACTION_GET_BUCKET_QUOTA` |
|
||||
| `weed/s3api/s3_constants/s3_actions.go` | Add coarse-grained `ACTION_PUT_BUCKET_QUOTA` and `ACTION_GET_BUCKET_QUOTA` |
|
||||
| `weed/s3api/s3_action_resolver.go` | Map `seaweedfs-quota` query param to fine-grained s3: actions |
|
||||
| `weed/s3api/s3api_bucket_quota_handlers.go` | New — `PutBucketQuotaHandler` and `GetBucketQuotaHandler` |
|
||||
| `weed/s3api/s3api_bucket_quota_handlers_test.go` | New — tests for unit conversion, validation, and error paths |
|
||||
| `weed/s3api/s3api_server.go` | Register the two routes in the bucket subrouter |
|
||||
@@ -1,11 +1,14 @@
|
||||
FROM alpine:latest
|
||||
|
||||
# Install required packages
|
||||
RUN apk add --no-cache \
|
||||
RUN apk upgrade --no-cache && \
|
||||
apk add --no-cache \
|
||||
ca-certificates \
|
||||
fuse \
|
||||
curl \
|
||||
jq
|
||||
jq \
|
||||
libcrypto3 \
|
||||
libssl3
|
||||
|
||||
# Copy our locally built binary
|
||||
COPY weed-local /usr/bin/weed
|
||||
|
||||
@@ -30,3 +30,5 @@ sleep_minutes = 17 # sleep minutes between each script execution
|
||||
bucket = "volume_bucket" # an existing bucket
|
||||
endpoint = "http://server2:8333"
|
||||
storage_class = "STANDARD_IA"
|
||||
# upload_concurrency = 5 # concurrent multipart part uploads per volume (volume.tier.upload -concurrent overrides)
|
||||
# download_concurrency = 5 # concurrent multipart part downloads per volume (volume.tier.download -concurrent overrides)
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
module github.com/seaweedfs/seaweedfs
|
||||
|
||||
go 1.26
|
||||
go 1.26.6
|
||||
|
||||
require (
|
||||
cloud.google.com/go v0.123.0 // indirect
|
||||
@@ -25,7 +25,7 @@ require (
|
||||
github.com/facebookgo/subset v0.0.0-20200203212716-c811ad88dec4 // indirect
|
||||
github.com/fsnotify/fsnotify v1.9.0 // indirect
|
||||
github.com/go-redsync/redsync/v4 v4.17.0
|
||||
github.com/go-sql-driver/mysql v1.10.0
|
||||
github.com/go-sql-driver/mysql v1.10.1
|
||||
github.com/go-zookeeper/zk v1.0.4 // indirect
|
||||
github.com/golang/protobuf v1.5.4
|
||||
github.com/golang/snappy v1.0.0
|
||||
@@ -65,7 +65,7 @@ require (
|
||||
github.com/rcrowley/go-metrics v0.0.0-20201227073835-cf1acfcdf475 // indirect
|
||||
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec // indirect
|
||||
github.com/seaweedfs/goexif v2.0.0+incompatible
|
||||
github.com/seaweedfs/raft v1.2.0
|
||||
github.com/seaweedfs/raft v1.2.1
|
||||
github.com/sirupsen/logrus v1.9.4 // indirect
|
||||
github.com/spf13/afero v1.15.0 // indirect
|
||||
github.com/spf13/cast v1.10.0 // indirect
|
||||
@@ -90,18 +90,18 @@ require (
|
||||
gocloud.dev v0.46.0
|
||||
gocloud.dev/pubsub/natspubsub v0.46.0
|
||||
gocloud.dev/pubsub/rabbitpubsub v0.46.0
|
||||
golang.org/x/crypto v0.55.0
|
||||
golang.org/x/crypto v0.56.0
|
||||
golang.org/x/exp v0.0.0-20260709172345-9ea1abe57597
|
||||
golang.org/x/image v0.45.0
|
||||
golang.org/x/image v0.46.0
|
||||
golang.org/x/net v0.58.0
|
||||
golang.org/x/oauth2 v0.36.0
|
||||
golang.org/x/sys v0.47.0
|
||||
golang.org/x/text v0.41.0 // indirect
|
||||
golang.org/x/tools v0.48.0 // indirect
|
||||
golang.org/x/sys v0.48.0
|
||||
golang.org/x/text v0.42.0 // indirect
|
||||
golang.org/x/tools v0.49.0 // indirect
|
||||
golang.org/x/xerrors v0.0.0-20240903120638-7835f813f4da // indirect
|
||||
google.golang.org/api v0.296.0
|
||||
google.golang.org/api v0.297.0
|
||||
google.golang.org/genproto v0.0.0-20260715232425-e75dac1f907d // indirect
|
||||
google.golang.org/grpc v1.85.0-dev
|
||||
google.golang.org/grpc v1.85.0-dev.0.20260915183914-4e49413dcab7
|
||||
google.golang.org/protobuf v1.36.12
|
||||
gopkg.in/inf.v0 v0.9.1 // indirect
|
||||
modernc.org/b v1.0.0 // indirect
|
||||
@@ -122,10 +122,10 @@ require (
|
||||
github.com/apple/foundationdb/bindings/go v0.0.0-20250911184653-27f7192f47c3
|
||||
github.com/arangodb/go-driver v1.6.9
|
||||
github.com/armon/go-metrics v0.4.1
|
||||
github.com/aws/aws-sdk-go-v2 v1.45.1
|
||||
github.com/aws/aws-sdk-go-v2 v1.47.0
|
||||
github.com/aws/aws-sdk-go-v2/config v1.32.35
|
||||
github.com/aws/aws-sdk-go-v2/credentials v1.20.1
|
||||
github.com/aws/aws-sdk-go-v2/service/s3 v1.107.0
|
||||
github.com/aws/aws-sdk-go-v2/credentials v1.20.4
|
||||
github.com/aws/aws-sdk-go-v2/service/s3 v1.107.3
|
||||
github.com/cespare/xxhash/v2 v2.3.0
|
||||
github.com/cognusion/imaging v1.0.4
|
||||
github.com/fluent/fluent-logger-golang v1.10.1
|
||||
@@ -144,9 +144,9 @@ require (
|
||||
github.com/parquet-go/parquet-go v0.32.0
|
||||
github.com/pkg/sftp v1.13.11
|
||||
github.com/rabbitmq/amqp091-go v1.14.0
|
||||
github.com/rclone/rclone v1.75.0
|
||||
github.com/rclone/rclone v1.75.1
|
||||
github.com/rdleal/intervalst v1.5.0
|
||||
github.com/redis/go-redis/v9 v9.21.0
|
||||
github.com/redis/go-redis/v9 v9.22.0
|
||||
github.com/schollz/progressbar/v3 v3.19.1
|
||||
github.com/seaweedfs/go-fuse/v2 v2.9.4
|
||||
github.com/shirou/gopsutil/v4 v4.26.7
|
||||
@@ -160,7 +160,7 @@ require (
|
||||
github.com/ydb-platform/ydb-go-sdk/v3 v3.151.1
|
||||
go.etcd.io/etcd/client/pkg/v3 v3.7.1
|
||||
go.uber.org/atomic v1.11.0
|
||||
golang.org/x/sync v0.22.0
|
||||
golang.org/x/sync v0.23.0
|
||||
golang.org/x/tools/godoc v0.1.0-deprecated
|
||||
google.golang.org/grpc/security/advancedtls v1.0.0
|
||||
)
|
||||
@@ -185,7 +185,7 @@ require (
|
||||
github.com/antlr4-go/antlr/v4 v4.13.1 // indirect
|
||||
github.com/apache/arrow-go/v18 v18.7.0 // indirect
|
||||
github.com/apache/thrift v0.24.0 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/signin v1.7.1 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/signin v1.10.0 // indirect
|
||||
github.com/bahlo/generic-list-go v0.2.0 // indirect
|
||||
github.com/bazelbuild/rules_go v0.46.0 // indirect
|
||||
github.com/biogo/store v0.0.0-20201120204734-aad293a2328f // indirect
|
||||
@@ -262,8 +262,8 @@ require (
|
||||
github.com/pquerna/otp v1.5.0 // indirect
|
||||
github.com/pterm/pterm v0.12.83 // indirect
|
||||
github.com/quic-go/qpack v0.6.0 // indirect
|
||||
github.com/rclone/Proton-API-Bridge v1.0.4 // indirect
|
||||
github.com/rclone/go-proton-api v1.0.3 // indirect
|
||||
github.com/rclone/Proton-API-Bridge v1.0.5 // indirect
|
||||
github.com/rclone/go-proton-api v1.0.4 // indirect
|
||||
github.com/rogpeppe/go-internal v1.15.0 // indirect
|
||||
github.com/rwcarlsen/goexif v0.0.0-20190401172101-9e8deecbddbd // indirect
|
||||
github.com/ryanuber/go-glob v1.0.0 // indirect
|
||||
@@ -283,19 +283,19 @@ require (
|
||||
github.com/xeipuuv/gojsonreference v0.0.0-20180127040603-bd5ef7bd5415 // indirect
|
||||
github.com/xo/terminfo v0.0.0-20220910002029-abceb7e1c41e // indirect
|
||||
github.com/zeebo/xxh3 v1.1.0 // indirect
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace v1.44.0 // indirect
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracegrpc v1.44.0 // indirect
|
||||
go.opentelemetry.io/otel/exporters/zipkin v1.36.0 // indirect
|
||||
go.opentelemetry.io/proto/otlp v1.10.0 // indirect
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace v1.45.0 // indirect
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracegrpc v1.45.0 // indirect
|
||||
go.opentelemetry.io/otel/exporters/zipkin v1.45.0 // indirect
|
||||
go.opentelemetry.io/proto/otlp v1.11.0 // indirect
|
||||
go.uber.org/mock v0.5.2 // indirect
|
||||
go.yaml.in/yaml/v2 v2.4.4 // indirect
|
||||
go.yaml.in/yaml/v3 v3.0.4 // indirect
|
||||
golang.org/x/mod v0.38.0 // indirect
|
||||
golang.org/x/mod v0.41.0 // indirect
|
||||
gonum.org/v1/gonum v0.17.0 // indirect
|
||||
)
|
||||
|
||||
require (
|
||||
cel.dev/expr v0.25.2 // indirect
|
||||
cel.dev/expr v0.25.3 // indirect
|
||||
cloud.google.com/go/auth v0.23.2 // indirect
|
||||
cloud.google.com/go/auth/oauth2adapt v0.2.8 // indirect
|
||||
cloud.google.com/go/compute/metadata v0.9.0 // indirect
|
||||
@@ -310,7 +310,7 @@ require (
|
||||
github.com/Azure/go-ntlmssp v0.1.1 // indirect
|
||||
github.com/AzureAD/microsoft-authentication-library-for-go v1.7.2 // indirect
|
||||
github.com/Files-com/files-sdk-go/v3 v3.3.194 // indirect
|
||||
github.com/GoogleCloudPlatform/opentelemetry-operations-go/detectors/gcp v1.34.0 // indirect
|
||||
github.com/GoogleCloudPlatform/opentelemetry-operations-go/detectors/gcp v1.35.0 // indirect
|
||||
github.com/GoogleCloudPlatform/opentelemetry-operations-go/exporter/metric v0.57.0 // indirect
|
||||
github.com/GoogleCloudPlatform/opentelemetry-operations-go/internal/resourcemapping v0.57.0 // indirect
|
||||
github.com/IBM/go-sdk-core/v5 v5.23.1 // indirect
|
||||
@@ -326,21 +326,21 @@ require (
|
||||
github.com/andybalholm/cascadia v1.3.4 // indirect
|
||||
github.com/appscode/go-querystring v0.0.0-20170504095604-0126cfb3f1dc // indirect
|
||||
github.com/arangodb/go-velocypack v0.0.0-20200318135517-5af53c29c67e // indirect
|
||||
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.16 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.19.1 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.18 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.20.0 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/feature/s3/manager v1.22.34 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.1 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.1 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.1 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.3 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.3 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.3 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.19 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.28 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.1 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.36 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.31 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.3 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.39 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/sns v1.39.14 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/sqs v1.42.24 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/sso v1.35.1 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/ssooidc v1.40.1 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/sts v1.47.1
|
||||
github.com/aws/aws-sdk-go-v2/service/sso v1.38.0 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/ssooidc v1.43.0 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/sts v1.50.0
|
||||
github.com/aws/smithy-go v1.28.1
|
||||
github.com/boltdb/bolt v1.3.1 // indirect
|
||||
github.com/bradenaw/juniper v0.15.3 // indirect
|
||||
@@ -363,7 +363,7 @@ require (
|
||||
github.com/elastic/gosigar v0.14.3 // indirect
|
||||
github.com/emersion/go-message v0.18.2 // indirect
|
||||
github.com/emersion/go-vcard v0.0.0-20260618161152-d854b7e0e2d3 // indirect
|
||||
github.com/envoyproxy/go-control-plane/envoy v1.37.0 // indirect
|
||||
github.com/envoyproxy/go-control-plane/envoy v1.39.1-0.20260819172001-e6e3fd93e4be // indirect
|
||||
github.com/envoyproxy/protoc-gen-validate v1.3.3 // indirect
|
||||
github.com/fatih/color v1.18.0 // indirect
|
||||
github.com/felixge/httpsnoop v1.1.0 // indirect
|
||||
@@ -388,7 +388,7 @@ require (
|
||||
github.com/gogo/protobuf v1.3.2 // indirect
|
||||
github.com/golang-jwt/jwt/v4 v4.5.2 // indirect
|
||||
github.com/google/s2a-go v0.1.9 // indirect
|
||||
github.com/googleapis/enterprise-certificate-proxy v0.3.20 // indirect
|
||||
github.com/googleapis/enterprise-certificate-proxy v0.3.21 // indirect
|
||||
github.com/gorilla/schema v1.4.1 // indirect
|
||||
github.com/gorilla/securecookie v1.1.2 // indirect
|
||||
github.com/gorilla/sessions v1.4.0
|
||||
@@ -438,7 +438,7 @@ require (
|
||||
github.com/oracle/oci-go-sdk/v65 v65.121.0 // indirect
|
||||
github.com/panjf2000/ants/v2 v2.12.1 // indirect
|
||||
github.com/patrickmn/go-cache v2.1.0+incompatible // indirect
|
||||
github.com/pelletier/go-toml/v2 v2.4.1 // indirect
|
||||
github.com/pelletier/go-toml/v2 v2.4.3 // indirect
|
||||
github.com/pengsrc/go-shared v0.2.1-0.20190131101655-1999055a4a14 // indirect
|
||||
github.com/philhofer/fwd v1.2.0 // indirect
|
||||
github.com/pierrec/lz4/v4 v4.1.29
|
||||
@@ -489,9 +489,9 @@ require (
|
||||
go.etcd.io/bbolt v1.5.0 // indirect
|
||||
go.etcd.io/etcd/api/v3 v3.7.1 // indirect
|
||||
go.opentelemetry.io/auto/sdk v1.2.1 // indirect
|
||||
go.opentelemetry.io/contrib/detectors/gcp v1.44.0 // indirect
|
||||
go.opentelemetry.io/contrib/detectors/gcp v1.45.0 // indirect
|
||||
go.opentelemetry.io/contrib/instrumentation/google.golang.org/grpc/otelgrpc v0.69.0 // indirect
|
||||
go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.69.0 // indirect
|
||||
go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.70.0 // indirect
|
||||
go.opentelemetry.io/otel v1.45.0 // indirect
|
||||
go.opentelemetry.io/otel/metric v1.45.0 // indirect
|
||||
go.opentelemetry.io/otel/sdk v1.45.0 // indirect
|
||||
@@ -501,7 +501,7 @@ require (
|
||||
go.uber.org/zap v1.27.1 // indirect
|
||||
golang.org/x/term v0.45.0
|
||||
golang.org/x/time v0.15.0
|
||||
google.golang.org/genproto/googleapis/api v0.0.0-20260715232425-e75dac1f907d // indirect
|
||||
google.golang.org/genproto/googleapis/api v0.0.0-20260817212433-ac3dfec99bb1 // indirect
|
||||
google.golang.org/genproto/googleapis/rpc v0.0.0-20260819154853-08b0e4226688 // indirect
|
||||
gopkg.in/natefinch/lumberjack.v2 v2.2.1 // indirect
|
||||
gopkg.in/validator.v2 v2.0.1 // indirect
|
||||
|
||||
@@ -6,8 +6,8 @@ atomicgo.dev/keyboard v0.2.9 h1:tOsIid3nlPLZ3lwgG8KZMp/SFmr7P0ssEN5JUsm78K8=
|
||||
atomicgo.dev/keyboard v0.2.9/go.mod h1:BC4w9g00XkxH/f1HXhW2sXmJFOCWbKn9xrOunSFtExQ=
|
||||
atomicgo.dev/schedule v0.1.0 h1:nTthAbhZS5YZmgYbb2+DH8uQIZcTlIrd4eYr3UQxEjs=
|
||||
atomicgo.dev/schedule v0.1.0/go.mod h1:xeUa3oAkiuHYh8bKiQBRojqAMq3PXXbJujjb0hw8pEU=
|
||||
cel.dev/expr v0.25.2 h1:K6j46C81hXtZQfuX60cVWQFBJahKSE2gfRbNuvr5bFs=
|
||||
cel.dev/expr v0.25.2/go.mod h1:hrXvqGP6G6gyx8UAHSHJ5RGk//1Oj5nXQ2NI02Nrsg4=
|
||||
cel.dev/expr v0.25.3 h1:A2jO8jwOugrrovveCWfj0KEZOfqiLgAcwjpHPhzIGw0=
|
||||
cel.dev/expr v0.25.3/go.mod h1:hrXvqGP6G6gyx8UAHSHJ5RGk//1Oj5nXQ2NI02Nrsg4=
|
||||
cloud.google.com/go v0.26.0/go.mod h1:aQUYkXzVsufM+DwF1aE+0xfcU+56JwCaLick0ClmMTw=
|
||||
cloud.google.com/go v0.34.0/go.mod h1:aQUYkXzVsufM+DwF1aE+0xfcU+56JwCaLick0ClmMTw=
|
||||
cloud.google.com/go v0.38.0/go.mod h1:990N+gfupTy94rShfmMCWGDn0LpTmnzTp2qbd1dvSRU=
|
||||
@@ -593,8 +593,8 @@ github.com/FilenCloudDienste/filen-sdk-go v0.0.39 h1:tgV5jYL6dsXop9TpDTIQU6UwJjw
|
||||
github.com/FilenCloudDienste/filen-sdk-go v0.0.39/go.mod h1:0cBhKXQg49XbKZZfk5TCDa3sVLP+xMxZTWL+7KY0XR0=
|
||||
github.com/Files-com/files-sdk-go/v3 v3.3.194 h1:dtOFxSTWWRpkmvXa6ycNiw8dVDu1wkgzcXyVV1VafNc=
|
||||
github.com/Files-com/files-sdk-go/v3 v3.3.194/go.mod h1:rl0WumSN9gSo775DgvQv+wMQ8rlb0ES/1hU5jkMtLXg=
|
||||
github.com/GoogleCloudPlatform/opentelemetry-operations-go/detectors/gcp v1.34.0 h1:yzIYdwuro811Z27D3T80Wkd3rqZzb0K43nner7Eh1yE=
|
||||
github.com/GoogleCloudPlatform/opentelemetry-operations-go/detectors/gcp v1.34.0/go.mod h1:pJTkW8hEUIIi3Pf65lPZOnn4Y81yCllX6IWk2jNXdkM=
|
||||
github.com/GoogleCloudPlatform/opentelemetry-operations-go/detectors/gcp v1.35.0 h1:bN1gA3of5bXtbnLsRPrwfmbbe7A5UWFlcTHseujLnpc=
|
||||
github.com/GoogleCloudPlatform/opentelemetry-operations-go/detectors/gcp v1.35.0/go.mod h1:Yj5vHEz/aAepZGliRJsA6uvHAVAQyEwajq9ORCHPxzM=
|
||||
github.com/GoogleCloudPlatform/opentelemetry-operations-go/exporter/metric v0.57.0 h1:jLdiS1vO+XJFyDSWRHBx56r4s/NNtcl5J6KyCcWUX/w=
|
||||
github.com/GoogleCloudPlatform/opentelemetry-operations-go/exporter/metric v0.57.0/go.mod h1:8lmpHY+1VRoteiOwyrQMDt1YGXOrFKCz+1wJW7n3ODY=
|
||||
github.com/GoogleCloudPlatform/opentelemetry-operations-go/internal/cloudmock v0.57.0 h1:cSjUzZ7KU8hicTgzaSv9NmSyM9fTVK3y5lsBUl3wOis=
|
||||
@@ -710,48 +710,48 @@ github.com/armon/go-metrics v0.4.1/go.mod h1:E6amYzXo6aW1tqzoZGT755KkbgrJsSdpwZ+
|
||||
github.com/atomicgo/cursor v0.0.1/go.mod h1:cBON2QmmrysudxNBFthvMtN32r3jxVRIvzkUiF/RuIk=
|
||||
github.com/aws/aws-sdk-go v1.55.8 h1:JRmEUbU52aJQZ2AjX4q4Wu7t4uZjOu71uyNmaWlUkJQ=
|
||||
github.com/aws/aws-sdk-go v1.55.8/go.mod h1:ZkViS9AqA6otK+JBBNH2++sx1sgxrPKcSzPPvQkUtXk=
|
||||
github.com/aws/aws-sdk-go-v2 v1.45.1 h1:iIoG3NaLhV6UZpPXyPXlDj2I9oS8tV/nMcMnITCC6Ks=
|
||||
github.com/aws/aws-sdk-go-v2 v1.45.1/go.mod h1:bttEH6JqnUL8LepvDVfdrds/fZ5bCIxzpe3abyUrhDU=
|
||||
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.16 h1:aiuaKlDweRC5qExJondpWjOgyzMHpofpwspGXUtwn4c=
|
||||
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.16/go.mod h1:nG/LOlmox9BDe9HvQnXWzgcK8uKbgBMZ/Hp5pVt/21I=
|
||||
github.com/aws/aws-sdk-go-v2 v1.47.0 h1:0jsHallhJCeaU0Ko48c/3FK1ctOQ7NpzggxriJOQ8MQ=
|
||||
github.com/aws/aws-sdk-go-v2 v1.47.0/go.mod h1:bttEH6JqnUL8LepvDVfdrds/fZ5bCIxzpe3abyUrhDU=
|
||||
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.18 h1:LAfOuhAH331fmOjTQpAaOlH+Ftn7RzSDJ2VFwjdMMy4=
|
||||
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.18/go.mod h1:4e5xhuXHx1e4U9EthvbPP1r/DIMp5c2823OL8karzcM=
|
||||
github.com/aws/aws-sdk-go-v2/config v1.32.35 h1:UEzXuET8E42lxBPijuACu/tEK7v5lFPlk0Q+GT5WD9E=
|
||||
github.com/aws/aws-sdk-go-v2/config v1.32.35/go.mod h1:KaMtJpFa2JlL2BStjjHQVwQpzZEmw+ND/EgVrfFoo2g=
|
||||
github.com/aws/aws-sdk-go-v2/credentials v1.20.1 h1:Z8GRNEx0u9sDkZOq4PUnN8mjGwbUQGRzMSXpvt3d8xQ=
|
||||
github.com/aws/aws-sdk-go-v2/credentials v1.20.1/go.mod h1:uBIK00kFo95dnemqfFMTWx0X8YRqsh6ecIoCjjOkZqM=
|
||||
github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.19.1 h1:YIEBqcqRnpi4Pfv0YHImtgi6czGCwKHANC7SwmUAVD0=
|
||||
github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.19.1/go.mod h1:imEf0oufgAo8KAkCHhrOdqGEC0YWx1PPBQH82shSxGw=
|
||||
github.com/aws/aws-sdk-go-v2/credentials v1.20.4 h1:hTvrJJseKbvw32kmiE0G+u/9ZqpqscjDrTigHIXP2qs=
|
||||
github.com/aws/aws-sdk-go-v2/credentials v1.20.4/go.mod h1:gWp9O1ZBWwpcIrgV+mVHk4gZUurAEDkgypu/OXOlIaw=
|
||||
github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.20.0 h1:AM4hHjww+PSFtt6E+UrBrPlZkWsePCLEt9AjkfQX+yM=
|
||||
github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.20.0/go.mod h1:3x/yXezeQjpOvBb4jEMxrS8SXvpdvJ5abv6l5c1gWM8=
|
||||
github.com/aws/aws-sdk-go-v2/feature/s3/manager v1.22.34 h1:Pn7OsMwBLbkZ6OnCxWHAjf0L/22H8cnhxZC0uPwtMtg=
|
||||
github.com/aws/aws-sdk-go-v2/feature/s3/manager v1.22.34/go.mod h1:eToXR/Gk1uqpn04eSmdgVXwfS0WvH8aG4eBFr8ygbpU=
|
||||
github.com/aws/aws-sdk-go-v2/feature/s3/transfermanager v0.3.11 h1:eBXB8KZgzQ8A9QB4iJS4aw/u6+4OY3i2hQXPABeAIOg=
|
||||
github.com/aws/aws-sdk-go-v2/feature/s3/transfermanager v0.3.11/go.mod h1:N9+5pG27Fy61GUL5YXVLXDTLmUudMrgwsuDbgBMNLxQ=
|
||||
github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.1 h1:pc138gM1CW+XPc60rEwUlwwuwWFQK16CI1T7v1F9Oec=
|
||||
github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.1/go.mod h1:1+koxpPIbfBdfzP6vojm5/zTpTQ/micYwlxIiNB3TxI=
|
||||
github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.1 h1:K0JsbZQj+1h208Ro1zHeA4l7bMp0NvRffHQ91q8Ol1s=
|
||||
github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.1/go.mod h1:W3/vL6EtCIatICGy9ab29QhMuae+cOKPWcMxv02CO+Q=
|
||||
github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.1 h1:yhw5KD1phVyP9vijxOUzDfEtJx+bt+L63k+VfuiYFAA=
|
||||
github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.1/go.mod h1:ZW2e0d7DYlRxlS9hEiMXE47gTdX5KRN4byUiNbUpG+Q=
|
||||
github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.3 h1:Hp/VgjP0BysR3OgLlR057Vz2LcbbVnoWeJ+3qWiS/fY=
|
||||
github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.3/go.mod h1:nwGV5qw7F1IZPgxCvA/ph8N2TAuz+BkRG/bXn808qMA=
|
||||
github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.3 h1:MUaM4f+kj1ZIBPZfUS8cxP1GKXXZtHJjAthy93AN7SM=
|
||||
github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.3/go.mod h1:6YmVmEVRI5ZZzRjCSsb9SryKH0hAlMRdgA7kG9aDvBU=
|
||||
github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.3 h1:fuSCw4Z2qfRCztMPO3GXJNSiEp6Wee+WOLwrHHUMy9c=
|
||||
github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.3/go.mod h1:6SxcHheD1pPR5+kWm1wGvjlL/YqUsh267sAfEmN4K7A=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.19 h1:bAdDl/HkGCcGPoe25ToSHEw23VIxt6CT5fLcg111BKg=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.19/go.mod h1:KaUzbLxv4CeSxh6ZCl9B4m7CuFenS8kUEaDs+f/DQr4=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.28 h1:Q1TF1J9jVD+vFo0LzNnmNdQ9EAt52TS+MQlq9Ir+Yxo=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.28/go.mod h1:4KqXXC/p1hrotmouDFbrRoWaLy962b9PMUReCG6+uWo=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.1 h1:RmmWQPREQdk9U+PfqeHW3MqZaBaNK7TpV9W3RY+b+7g=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.1/go.mod h1:0A3W4F+68ZnNk5XcNL/e9HFMwnP8RlEicFfy6eOEDyw=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.36 h1:EUIwBoN+q7UmhAejxgD27APiRjh1vwCFo53gSqdT0BM=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.36/go.mod h1:6u00gmlTGR6W0b2k9NBrld7MnOEmf1Spqx0VVt6AqyE=
|
||||
github.com/aws/aws-sdk-go-v2/service/s3 v1.107.0 h1:OkYV+1171za+ab9otU1tGxMXhx6uZvwVEtVddjLuYTg=
|
||||
github.com/aws/aws-sdk-go-v2/service/s3 v1.107.0/go.mod h1:5FTZoQxhmLEiCAtYVk6V+t0iS/B5yGZVLZ3Wq5FDJZI=
|
||||
github.com/aws/aws-sdk-go-v2/service/signin v1.7.1 h1:mdMtSVKdQ3+mzBh+l0ogrFYZVQUCg6pJZOirA2ARsYE=
|
||||
github.com/aws/aws-sdk-go-v2/service/signin v1.7.1/go.mod h1:9IqUlsJDbUPcg6cgx3WEzXdjrbWzLDQrak0aaSqlTcI=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.31 h1:uZOinZb+h7lZw8IYzP1z1IuEnueB76/EFkcf/fEW4Ag=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.31/go.mod h1:NRtwAM/p5VRt03TlEUs0pH3TeWamWdf4YyJpSrzPYLc=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.3 h1:bON1rJf67TSTDCKg816AAIE4xSTtoo9tl0XRkO72R+I=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.3/go.mod h1:c5BBpjJcQXpfeq9iASyVKA3T6vX6B6LEXY4mL/gklDY=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.39 h1:HLPAVrlLDaN2boN0xJx7MgaQDNEO3Q+c9L6kl/8m47Q=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.39/go.mod h1:Pg/dVfsNkm1hsIDK/gMvCKtmyNfNTV12mrgHqVE/6Oo=
|
||||
github.com/aws/aws-sdk-go-v2/service/s3 v1.107.3 h1:IKoCZqfWfZzSBi16QFQ+QcbQ3LRQ7QgB1S5tDAyPBQQ=
|
||||
github.com/aws/aws-sdk-go-v2/service/s3 v1.107.3/go.mod h1:RBpRcXiM4s2pOInVs32GsBonnje+fiAj4mcrStRmlCA=
|
||||
github.com/aws/aws-sdk-go-v2/service/signin v1.10.0 h1:ZD5qFpWcaOKdTuhBi431pIDkCgrMkMlMT6jlpSPoIRI=
|
||||
github.com/aws/aws-sdk-go-v2/service/signin v1.10.0/go.mod h1:8Nuuf+tR346PjJ3MvZPh9pekbLiLQFWJhzMXfwy7alA=
|
||||
github.com/aws/aws-sdk-go-v2/service/sns v1.39.14 h1:p8WdWDh5AwSZdp19Haa3XMyPCICi9Z375a/Nu3IIEZY=
|
||||
github.com/aws/aws-sdk-go-v2/service/sns v1.39.14/go.mod h1:NKVY7DER6VXHkt2I/ycmHakALNboi3Rqwt4eEf/1Cnk=
|
||||
github.com/aws/aws-sdk-go-v2/service/sqs v1.42.24 h1:JP2wjWGmUp8lTCZb13Dv0Eciyc1jbO8pd0HZVMHFlrc=
|
||||
github.com/aws/aws-sdk-go-v2/service/sqs v1.42.24/go.mod h1:Ql9ziDutk8ERAN9HMaYANCW3lop451ppebkxEJMLCTM=
|
||||
github.com/aws/aws-sdk-go-v2/service/sso v1.35.1 h1:B6WFn91tobD6gG4724ONHaqrpKsoETGnv98LHe/yIGM=
|
||||
github.com/aws/aws-sdk-go-v2/service/sso v1.35.1/go.mod h1:tWuiVBUtPBr8/rgRiYS8Uf85sHcAN+G7XS3D3CEoUh8=
|
||||
github.com/aws/aws-sdk-go-v2/service/ssooidc v1.40.1 h1:6yeYCWFvgbI2TI3K6jr9LtBNhXgJ7g4xqD+DEiaDDmM=
|
||||
github.com/aws/aws-sdk-go-v2/service/ssooidc v1.40.1/go.mod h1:naFe83jSMuYkH+QjQPX8n1MLhBkeCFM5Lsnh5m5wz3c=
|
||||
github.com/aws/aws-sdk-go-v2/service/sts v1.47.1 h1:Sv2xPnRHlThSUtVujYuUBPI/Il8si6UPHXL8DMiB/F0=
|
||||
github.com/aws/aws-sdk-go-v2/service/sts v1.47.1/go.mod h1:mKo/CzaCz8qytGW70NG4vIIGAx1HXTlb5lHNkC5k3lk=
|
||||
github.com/aws/aws-sdk-go-v2/service/sso v1.38.0 h1:JGeeBcMlhg1xtOXYpeCaTQBZObtXMPQCUqBcmr65NRA=
|
||||
github.com/aws/aws-sdk-go-v2/service/sso v1.38.0/go.mod h1:XwteswG9EOMRFm73UT0t+MbTwyLxMrEXkU6e+v92Lzo=
|
||||
github.com/aws/aws-sdk-go-v2/service/ssooidc v1.43.0 h1:obhahQXDEdVEv8y5bTKXR30LVaxYe1kyYM0L7l2Iq+k=
|
||||
github.com/aws/aws-sdk-go-v2/service/ssooidc v1.43.0/go.mod h1:6twZZ/aXHNy1vXUO8koUbp++MYzMASkOgEBdkbJYmO0=
|
||||
github.com/aws/aws-sdk-go-v2/service/sts v1.50.0 h1:khXV3+K5D3f4e8xtplaRdSFn1bEg3gj5EBHQvbCOZbQ=
|
||||
github.com/aws/aws-sdk-go-v2/service/sts v1.50.0/go.mod h1:/8JRcdTt//hG0Q4BTmGbuOplT7ABe+5rdtqUHqXvYIM=
|
||||
github.com/aws/smithy-go v1.28.1 h1:R/nXH00c8qcfCzQVELtRw+eLQWtzv+VAIEFJ1/xxXlQ=
|
||||
github.com/aws/smithy-go v1.28.1/go.mod h1:YE2RhdIuDbA5E5bTdciG9KrW3+TiEONeUWCqxX9i1Fc=
|
||||
github.com/bahlo/generic-list-go v0.2.0 h1:5sz/EEAK+ls5wF+NeqDpk5+iNdMDXrh3z3nPnH1Wvgk=
|
||||
@@ -987,8 +987,8 @@ github.com/envoyproxy/go-control-plane v0.10.3/go.mod h1:fJJn/j26vwOu972OllsvAgJ
|
||||
github.com/envoyproxy/go-control-plane v0.11.0/go.mod h1:VnHyVMpzcLvCFt9yUz1UnCwHLhwx1WguiVDV7pTG/tI=
|
||||
github.com/envoyproxy/go-control-plane v0.14.0 h1:hbG2kr4RuFj222B6+7T83thSPqLjwBIfQawTkC++2HA=
|
||||
github.com/envoyproxy/go-control-plane v0.14.0/go.mod h1:NcS5X47pLl/hfqxU70yPwL9ZMkUlwlKxtAohpi2wBEU=
|
||||
github.com/envoyproxy/go-control-plane/envoy v1.37.0 h1:u3riX6BoYRfF4Dr7dwSOroNfdSbEPe9Yyl09/B6wBrQ=
|
||||
github.com/envoyproxy/go-control-plane/envoy v1.37.0/go.mod h1:DReE9MMrmecPy+YvQOAOHNYMALuowAnbjjEMkkWOi6A=
|
||||
github.com/envoyproxy/go-control-plane/envoy v1.39.1-0.20260819172001-e6e3fd93e4be h1:SWe0x6yfglnxuvOiYgTTnNq7QD/yvqthOh1RBI8Bj8w=
|
||||
github.com/envoyproxy/go-control-plane/envoy v1.39.1-0.20260819172001-e6e3fd93e4be/go.mod h1:PYEOlng9XcrulfyWpm49jECTPV0LT4q8cO7fLW/xwgk=
|
||||
github.com/envoyproxy/go-control-plane/ratelimit v0.1.0 h1:/G9QYbddjL25KvtKTv3an9lx6VBE2cnb8wp1vEGNYGI=
|
||||
github.com/envoyproxy/go-control-plane/ratelimit v0.1.0/go.mod h1:Wk+tMFAFbCXaJPzVVHnPgRKdUdwW/KdbRt94AzgRee4=
|
||||
github.com/envoyproxy/protoc-gen-validate v0.1.0/go.mod h1:iSmxcyjqTsJpI2R4NaDN7+kN2VEUnK/pcBlmesArF7c=
|
||||
@@ -1109,8 +1109,8 @@ github.com/go-redsync/redsync/v4 v4.17.0 h1:FFJ+uxZs44y4Sq10//IFKic9T94AYl+u3Sog
|
||||
github.com/go-redsync/redsync/v4 v4.17.0/go.mod h1:CKVA6qwT07S/916i+Yd9h1/8YFQhCCpPYTQhvvYytJo=
|
||||
github.com/go-resty/resty/v2 v2.17.2 h1:FQW5oHYcIlkCNrMD2lloGScxcHJ0gkjshV3qcQAyHQk=
|
||||
github.com/go-resty/resty/v2 v2.17.2/go.mod h1:kCKZ3wWmwJaNc7S29BRtUhJwy7iqmn+2mLtQrOyQlVA=
|
||||
github.com/go-sql-driver/mysql v1.10.0 h1:Q+1LV8DkHJvSYAdR83XzuhDaTykuDx0l6fkXxoWCWfw=
|
||||
github.com/go-sql-driver/mysql v1.10.0/go.mod h1:M+cqaI7+xxXGG9swrdeUIoPG3Y3KCkF0pZej+SK+nWk=
|
||||
github.com/go-sql-driver/mysql v1.10.1 h1:arlSnNLq6a5yxGxV7qg9lF4j0C+KwD6NbQyKr9QL6ME=
|
||||
github.com/go-sql-driver/mysql v1.10.1/go.mod h1:M+cqaI7+xxXGG9swrdeUIoPG3Y3KCkF0pZej+SK+nWk=
|
||||
github.com/go-stack/stack v1.8.0/go.mod h1:v0f6uXyyMGvRgIKkXu+yp6POWl0qKG85gN/melR3HDY=
|
||||
github.com/go-task/slim-sprig v0.0.0-20230315185526-52ccab3ef572 h1:tfuBGBXKqDEevZMzYi5KSi8KkcZtzBcTgAUUtapy0OI=
|
||||
github.com/go-task/slim-sprig/v3 v3.0.0 h1:sUs3vkvUymDpBKi3qH1YSqBQk9+9D/8M2mN1vB6EwHI=
|
||||
@@ -1262,8 +1262,8 @@ github.com/googleapis/enterprise-certificate-proxy v0.1.0/go.mod h1:17drOmN3MwGY
|
||||
github.com/googleapis/enterprise-certificate-proxy v0.2.0/go.mod h1:8C0jb7/mgJe/9KK8Lm7X9ctZC2t60YyIpYEI16jx0Qg=
|
||||
github.com/googleapis/enterprise-certificate-proxy v0.2.1/go.mod h1:AwSRAtLfXpU5Nm3pW+v7rGDHp09LsPtGY9MduiEsR9k=
|
||||
github.com/googleapis/enterprise-certificate-proxy v0.2.3/go.mod h1:AwSRAtLfXpU5Nm3pW+v7rGDHp09LsPtGY9MduiEsR9k=
|
||||
github.com/googleapis/enterprise-certificate-proxy v0.3.20 h1:t/xL64VUoN69MuMRQuJETqYGOw4Z9mSRJK9epIEtwFk=
|
||||
github.com/googleapis/enterprise-certificate-proxy v0.3.20/go.mod h1:L3D/IQExI6LqEjBdXcZQ1WluSgigQmSwBboFstVPM4w=
|
||||
github.com/googleapis/enterprise-certificate-proxy v0.3.21 h1:OFdQ3tnCX/zaQ0Cedur3D3z7kI6HiLX9g3TiAN4/DFU=
|
||||
github.com/googleapis/enterprise-certificate-proxy v0.3.21/go.mod h1:L3D/IQExI6LqEjBdXcZQ1WluSgigQmSwBboFstVPM4w=
|
||||
github.com/googleapis/gax-go/v2 v2.0.4/go.mod h1:0Wqv26UfaUD9n4G6kQubkQ+KchISgw+vpHVxEJEs9eg=
|
||||
github.com/googleapis/gax-go/v2 v2.0.5/go.mod h1:DWXyrwAJ9X0FpwwEdw+IPEYBICEFu5mhpdKc/us6bOk=
|
||||
github.com/googleapis/gax-go/v2 v2.1.0/go.mod h1:Q3nei7sK6ybPYH7twZdmQpAd1MKb7pfu6SK+H1/DsU0=
|
||||
@@ -1652,8 +1652,8 @@ github.com/pascaldekloe/goe v0.1.0/go.mod h1:lzWF7FIEvWOWxwDKqyGYQf6ZUaNfKdP144T
|
||||
github.com/patrickmn/go-cache v2.1.0+incompatible h1:HRMgzkcYKYpi3C8ajMPV8OFXaaRUnok+kx1WdO15EQc=
|
||||
github.com/patrickmn/go-cache v2.1.0+incompatible/go.mod h1:3Qf8kWWT7OJRJbdiICTKqZju1ZixQ/KpMGzzAfe6+WQ=
|
||||
github.com/pborman/getopt v0.0.0-20170112200414-7148bc3a4c30/go.mod h1:85jBQOZwpVEaDAr341tbn15RS4fCAsIst0qp7i8ex1o=
|
||||
github.com/pelletier/go-toml/v2 v2.4.1 h1:j5OMOImsH+j2k7GJ5YO+RxfWwohNiH6t5zB/+h3bagc=
|
||||
github.com/pelletier/go-toml/v2 v2.4.1/go.mod h1:2gIqNv+qfxSVS7cM2xJQKtLSTLUE9V8t9Stt+h56mCY=
|
||||
github.com/pelletier/go-toml/v2 v2.4.3 h1:GTRvJQutkOSftxIFD5xw9aepkYNuPWmVJpffdDPYVpY=
|
||||
github.com/pelletier/go-toml/v2 v2.4.3/go.mod h1:2gIqNv+qfxSVS7cM2xJQKtLSTLUE9V8t9Stt+h56mCY=
|
||||
github.com/pengsrc/go-shared v0.2.1-0.20190131101655-1999055a4a14 h1:XeOYlK9W1uCmhjJSsY78Mcuh7MVkNjTzmHx1yBzizSU=
|
||||
github.com/pengsrc/go-shared v0.2.1-0.20190131101655-1999055a4a14/go.mod h1:jVblp62SafmidSkvWrXyxAme3gaTfEtWwRPGz5cpvHg=
|
||||
github.com/peterh/liner v1.2.2 h1:aJ4AOodmL+JxOZZEL2u9iJf8omNRpqHc/EbrK+3mAXw=
|
||||
@@ -1760,18 +1760,18 @@ github.com/quic-go/quic-go v0.59.0 h1:OLJkp1Mlm/aS7dpKgTc6cnpynnD2Xg7C1pwL6vy/SA
|
||||
github.com/quic-go/quic-go v0.59.0/go.mod h1:upnsH4Ju1YkqpLXC305eW3yDZ4NfnNbmQRCMWS58IKU=
|
||||
github.com/rabbitmq/amqp091-go v1.14.0 h1:RSaT7aOKt/OrkVUyswPDW29lnRz9psuGmfZFBmLqLek=
|
||||
github.com/rabbitmq/amqp091-go v1.14.0/go.mod h1:Hy4jKW5kQART1u+JkDTF9YYOQUHXqMuhrgxOEeS7G4o=
|
||||
github.com/rclone/Proton-API-Bridge v1.0.4 h1:uGQJRjQC1hVLd5kqLsXc6CWO6oqrVeLoKQYoHapEZDg=
|
||||
github.com/rclone/Proton-API-Bridge v1.0.4/go.mod h1:VTPBYZotKAeDLlAzxU2O/s14NXk9FxUt9hn1jhH2iY8=
|
||||
github.com/rclone/go-proton-api v1.0.3 h1:3gBTzR+j0dYiTwtj9yKIdN/aV3W2a8KIPKp0GArojyQ=
|
||||
github.com/rclone/go-proton-api v1.0.3/go.mod h1:QAlkFfswzrBuxvCORWV8rZdddg52hahMN98CFWoFW1E=
|
||||
github.com/rclone/rclone v1.75.0 h1:3ARHem4jXWltvl+b0PvDAG8s6J/inHd5BRzfwMRb3W8=
|
||||
github.com/rclone/rclone v1.75.0/go.mod h1:PGLJUW/WSIJCysALqUcxmaCFyfMXUevf8CbuoOwsAdU=
|
||||
github.com/rclone/Proton-API-Bridge v1.0.5 h1:K1++Qtk3PvgkiCCiv6Pahju1TMOzKY6VSwiwT7XLAVc=
|
||||
github.com/rclone/Proton-API-Bridge v1.0.5/go.mod h1:vCeOPhlXzevN0AFojgh1zsjhetiShy/ArvJ/xkFUDWk=
|
||||
github.com/rclone/go-proton-api v1.0.4 h1:AJW0e9pB4j0hVK4WqyGErFwaI+5MUQWPCtj5FYYxtPg=
|
||||
github.com/rclone/go-proton-api v1.0.4/go.mod h1:QAlkFfswzrBuxvCORWV8rZdddg52hahMN98CFWoFW1E=
|
||||
github.com/rclone/rclone v1.75.1 h1:kIxQcoDLj2Gke/gMSHK7OnxhX1Gu1cJBLP1kJZoaFp0=
|
||||
github.com/rclone/rclone v1.75.1/go.mod h1:4zmMjGatCkSJPRZDpo+7y3xOl8S29EMUyKvZop5mHr4=
|
||||
github.com/rcrowley/go-metrics v0.0.0-20201227073835-cf1acfcdf475 h1:N/ElC8H3+5XpJzTSTfLsJV/mx9Q9g7kxmchpfZyxgzM=
|
||||
github.com/rcrowley/go-metrics v0.0.0-20201227073835-cf1acfcdf475/go.mod h1:bCqnVzQkZxMG4s8nGwiZ5l3QUCyqpo9Y+/ZMZ9VjZe4=
|
||||
github.com/rdleal/intervalst v1.5.0 h1:SEB9bCFz5IqD1yhfH1Wv8IBnY/JQxDplwkxHjT6hamU=
|
||||
github.com/rdleal/intervalst v1.5.0/go.mod h1:xO89Z6BC+LQDH+IPQQw/OESt5UADgFD41tYMUINGpxQ=
|
||||
github.com/redis/go-redis/v9 v9.21.0 h1:FPBE4hhbAke+TLmcY3WkpbDffJEomdqPn3HYiqAtL9E=
|
||||
github.com/redis/go-redis/v9 v9.21.0/go.mod h1:v/M13XI1PVCDcm01VtPFOADfZtHf8YW3baQf57KlIkA=
|
||||
github.com/redis/go-redis/v9 v9.22.0 h1:laDvpYXTJtZLloinw1fA5Kqd6HAEH2XKxOkG/PDq2F0=
|
||||
github.com/redis/go-redis/v9 v9.22.0/go.mod h1:y2g0Wj8rQvuK0ELM+oxSudcLtC09JScs98I/X9gRWY4=
|
||||
github.com/redis/rueidis v1.0.76 h1:RdDWuvlYBSp+bTrBvaXqJnNEL3VVzsnjo+0psPFgLc4=
|
||||
github.com/redis/rueidis v1.0.76/go.mod h1:UsfHPSbomB6QAVMk4iiFkzRy0nh9o7scDGa+SitvBY4=
|
||||
github.com/redis/rueidis/rueidiscompat v1.0.76 h1:7LikbiqCQqCsZXeZ+akgZMnjIV/J0VHih9PIX4gGZC4=
|
||||
@@ -1821,8 +1821,8 @@ github.com/seaweedfs/go-fuse/v2 v2.9.4 h1:ACyloiuopdhRSjdLLeSWbsVaemMPskORaRF01T
|
||||
github.com/seaweedfs/go-fuse/v2 v2.9.4/go.mod h1:zABdmWEa6A0bwaBeEOBUeUkGIZlxUhcdv+V1Dcc/U/I=
|
||||
github.com/seaweedfs/goexif v2.0.0+incompatible h1:x8pckiT12QQhifwhDQpeISgDfsqmQ6VR4LFPQ64JRps=
|
||||
github.com/seaweedfs/goexif v2.0.0+incompatible/go.mod h1:Oni780Z236sXpIQzk1XoJlTwqrJ02smEin9zQeff7Fk=
|
||||
github.com/seaweedfs/raft v1.2.0 h1:Ez4Hw9ifBbTT7wg54DvGHBjw1vRlTb4roH0TKl0Oj9Y=
|
||||
github.com/seaweedfs/raft v1.2.0/go.mod h1:fgs/rAVEzjQ7e04XMzG3eJhwZZRmBW+2uRtjakeCGeU=
|
||||
github.com/seaweedfs/raft v1.2.1 h1:QgFl/aaPnagpUxYB6Bx+fFss1NyetVVcJmraMmnaQ5Q=
|
||||
github.com/seaweedfs/raft v1.2.1/go.mod h1:fgs/rAVEzjQ7e04XMzG3eJhwZZRmBW+2uRtjakeCGeU=
|
||||
github.com/secure-systems-lab/go-securesystemslib v0.11.0 h1:iuCR9kcMFD4QurdKrGvPLoKZLv9YvwPYVr0473BdtFs=
|
||||
github.com/secure-systems-lab/go-securesystemslib v0.11.0/go.mod h1:+PMOTjUGwHj2vcZ+TFKlb1tXRbrdWE1LYDT5i9JC80Q=
|
||||
github.com/sergi/go-diff v1.0.0/go.mod h1:0CfEIISq7TuYL3j771MWULgwwjU+GofnZX9QAmXWZgo=
|
||||
@@ -2103,30 +2103,30 @@ go.opencensus.io v0.24.0 h1:y73uSU6J157QMP2kn2r30vwW1A2W2WFwSCGnAVxeaD0=
|
||||
go.opencensus.io v0.24.0/go.mod h1:vNK8G9p7aAivkbmorf4v+7Hgx+Zs0yY+0fOtgBfjQKo=
|
||||
go.opentelemetry.io/auto/sdk v1.2.1 h1:jXsnJ4Lmnqd11kwkBV2LgLoFMZKizbCi5fNZ/ipaZ64=
|
||||
go.opentelemetry.io/auto/sdk v1.2.1/go.mod h1:KRTj+aOaElaLi+wW1kO/DZRXwkF4C5xPbEe3ZiIhN7Y=
|
||||
go.opentelemetry.io/contrib/detectors/gcp v1.44.0 h1:NmLfL734pJhM0JKaYd2Y28+nY9dPRWYAAbxhRCrKXPw=
|
||||
go.opentelemetry.io/contrib/detectors/gcp v1.44.0/go.mod h1:tNAsgd8avTGke1+MndXlU5Cru4PQ9Ai/cCNWQv/ZJ/s=
|
||||
go.opentelemetry.io/contrib/detectors/gcp v1.45.0 h1:9jR0ZPRok9ryaOQ2Wx8rg5F7Aon59mxrqbVI60/vlBk=
|
||||
go.opentelemetry.io/contrib/detectors/gcp v1.45.0/go.mod h1:VSme3o2fvSg5bVg0dRzyHaj4Z5EVhG+g2Fde6LKzmQA=
|
||||
go.opentelemetry.io/contrib/instrumentation/google.golang.org/grpc/otelgrpc v0.69.0 h1:2yEATaop1/a1I4psnSLgWVPLWwCzkqWakgJy7xTDVy0=
|
||||
go.opentelemetry.io/contrib/instrumentation/google.golang.org/grpc/otelgrpc v0.69.0/go.mod h1:D7J12YRapIekYyPWgGPlA/23pRmpSEZC5xJC/TTLI9U=
|
||||
go.opentelemetry.io/contrib/instrumentation/net/http/httptrace/otelhttptrace v0.69.0 h1:MCcYL7J6Vt/X0kjqbMZkekCmwsurbQRbL69vkiye2lk=
|
||||
go.opentelemetry.io/contrib/instrumentation/net/http/httptrace/otelhttptrace v0.69.0/go.mod h1:3jnStNwSufK+f5ktjL4EPcwtig4rtd81NS70lqHuXl8=
|
||||
go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.69.0 h1:8tvICD4vSTOOsNrsI4Ljf6C+6UKvpTEH5XY3JMoyPoo=
|
||||
go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.69.0/go.mod h1:z9+yiacE0IHRqM4qFfkbt/JYlmYXgss8GY/jXoNuPJI=
|
||||
go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.70.0 h1:LMuyCAyfalSjDyjdC65nK6N0zoTT63+E/u95X0JovZI=
|
||||
go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.70.0/go.mod h1:085m8qbm4hgc8rZWGDEa4vmyyo2c3nPxUslYUKUIU04=
|
||||
go.opentelemetry.io/otel v1.45.0 h1:pdrWmLHofpubmArBv1LgFSv1Z0Ie/ppdZzu+kUN5EeU=
|
||||
go.opentelemetry.io/otel v1.45.0/go.mod h1:XZxIqPapzEYnhNSScF5DIqXhm/rYi0FzCe2XddAwZfQ=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlpmetric/otlpmetricgrpc v1.44.0 h1:SUplec5dp06reu1zaXmOXdvqH398taqrDXqUl99jxSc=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlpmetric/otlpmetricgrpc v1.44.0/go.mod h1:ho2g4N+ane+swq5I/VBkKWnRDY4kUINH3FuqyZqX/Ug=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlpmetric/otlpmetrichttp v1.44.0 h1:RuynHbfU8JUEw7DyONgkVYg2SVtsoF28y0LGIr69jgA=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlpmetric/otlpmetrichttp v1.44.0/go.mod h1:qZF+/lBs71APw8mlnEZcqZHMzqrYrsFiJOv83lX1OGo=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace v1.44.0 h1:4YsVu3B8+3qtWYYrsUYgn0OG78pN0rnNPRGX4SbokQI=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace v1.44.0/go.mod h1:+wnlSn0mD1ADVMe3v9Z/WIaiz6q6gL2J/ejaAmdmv80=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracegrpc v1.44.0 h1:qazEJlUOQzhCpzQpFETGby7EdqjI1wsd0W+6Gg1SCTU=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracegrpc v1.44.0/go.mod h1:fOD2Yefuxixkx3ahVNf0O/PERb6r4OlbxfATVnYvzCo=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace v1.45.0 h1:QRefszxJmfPdjXUUm3j6iDzY03mTPXMjqErFqQ67vUg=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace v1.45.0/go.mod h1:Tiz03lTBVBrm7eWZBOidzEaYaJa8tjwGUGv6d8mlTyk=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracegrpc v1.45.0 h1:fG5MCxGz8+2VtrN/WgqSpJFctVz24gpxj8CxkKmc8Ww=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracegrpc v1.45.0/go.mod h1:BmAYTn+3ysbRe+IU2msxmf5Rx3g6DHvex+tWI3LdhYI=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracehttp v1.44.0 h1:lgh3PiVrRUWMLOVSkQicxzZll5NjF1r+AtsX1XRIHw0=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracehttp v1.44.0/go.mod h1:5Cnhth3m/AgOeTgE3ex12pPmiu/gGtZit03kSzx9X7s=
|
||||
go.opentelemetry.io/otel/exporters/stdout/stdoutmetric v1.44.0 h1:hqxVTu/GtBF+vJ8d1fzW7fRxZFvgoDjWcxwwCaFDYpU=
|
||||
go.opentelemetry.io/otel/exporters/stdout/stdoutmetric v1.44.0/go.mod h1:z5fVEF4X5v0ESvlJqBrrFlBVoj5EQuefZpzsu7R+x5Q=
|
||||
go.opentelemetry.io/otel/exporters/zipkin v1.36.0 h1:s0n95ya5tOG03exJ5JySOdJFtwGo4ZQ+KeY7Zro4CLI=
|
||||
go.opentelemetry.io/otel/exporters/zipkin v1.36.0/go.mod h1:m9wRxtKA2MZ1HcnNC4BKI+9aYe434qRZTCvI7QGUN7Y=
|
||||
go.opentelemetry.io/otel/exporters/zipkin v1.45.0 h1:KN3btaILMTxR4QDHVGAO87lq5ButzK7l+kIfLuxQ1oA=
|
||||
go.opentelemetry.io/otel/exporters/zipkin v1.45.0/go.mod h1:yNcodmUclM4InyWoOwX/YW4Jri0Gj5FWAlM+NqCrtqY=
|
||||
go.opentelemetry.io/otel/metric v1.45.0 h1:7Eg1uH7CJ5cXv9is6tnBe1FI6rj1nwUdbFypRm3br/M=
|
||||
go.opentelemetry.io/otel/metric v1.45.0/go.mod h1:HAPbm1nd3p1PmFH7v2dR+6BjXxw+Lq4a2+pndMAm08s=
|
||||
go.opentelemetry.io/otel/metric/x v0.67.0 h1:PcicCNZFkZ4bXfSooXdo3WN7RBOVOtjVdo1wD358Uns=
|
||||
@@ -2140,8 +2140,8 @@ go.opentelemetry.io/otel/trace v1.45.0/go.mod h1:qoJJA2xNMnxRrdISU/kLtfUH2wNeQbi
|
||||
go.opentelemetry.io/proto/otlp v0.7.0/go.mod h1:PqfVotwruBrMGOCsRd/89rSnXhoiJIqeYNgFYFoEGnI=
|
||||
go.opentelemetry.io/proto/otlp v0.15.0/go.mod h1:H7XAot3MsfNsj7EXtrA2q5xSNQ10UqI405h3+duxN4U=
|
||||
go.opentelemetry.io/proto/otlp v0.19.0/go.mod h1:H7XAot3MsfNsj7EXtrA2q5xSNQ10UqI405h3+duxN4U=
|
||||
go.opentelemetry.io/proto/otlp v1.10.0 h1:IQRWgT5srOCYfiWnpqUYz9CVmbO8bFmKcwYxpuCSL2g=
|
||||
go.opentelemetry.io/proto/otlp v1.10.0/go.mod h1:/CV4QoCR/S9yaPj8utp3lvQPoqMtxXdzn7ozvvozVqk=
|
||||
go.opentelemetry.io/proto/otlp v1.11.0 h1:5rrYs0Ykyj50sdU/JU0x8etU+LubXWb+gED6TbEdMIk=
|
||||
go.opentelemetry.io/proto/otlp v1.11.0/go.mod h1:SmVizdCOAm3XBtG1g1NnOdhW6jtddT72hLMhv8VwA8E=
|
||||
go.uber.org/atomic v1.6.0/go.mod h1:sABNBOSYdrvTF6hTgEIbc7YasKWGhgEQZyfxyTvoXHQ=
|
||||
go.uber.org/atomic v1.7.0/go.mod h1:fEN4uk6kAWBTFdckzkM89CLk9XfWZrxpCo0nPH17wJc=
|
||||
go.uber.org/atomic v1.9.0/go.mod h1:fEN4uk6kAWBTFdckzkM89CLk9XfWZrxpCo0nPH17wJc=
|
||||
@@ -2193,8 +2193,8 @@ golang.org/x/crypto v0.6.0/go.mod h1:OFC/31mSvZgRz0V1QTNCzfAI1aIRzbiufJtkMIlEp58
|
||||
golang.org/x/crypto v0.7.0/go.mod h1:pYwdfH91IfpZVANVyUOhSIPZaFoJGxTFbZhFTx+dXZU=
|
||||
golang.org/x/crypto v0.13.0/go.mod h1:y6Z2r+Rw4iayiXXAIxJIDAJ1zMW4yaTpebo8fPOliYc=
|
||||
golang.org/x/crypto v0.14.0/go.mod h1:MVFd36DqK4CsrnJYDkBA3VC4m2GkXAM0PvzMCn4JQf4=
|
||||
golang.org/x/crypto v0.55.0 h1:+KWHjbgOaAQ66dh/YlkZKHlz9ZUlq61AFirAR9ntP8M=
|
||||
golang.org/x/crypto v0.55.0/go.mod h1:uq0V9dE/fzQuJtbnL+2EhWOE63vo164FY8xqEnV9xis=
|
||||
golang.org/x/crypto v0.56.0 h1:GUh5Ii4J5jtcseSMiRqr1jXCNHoxjeV9Fmekc2oLy6Y=
|
||||
golang.org/x/crypto v0.56.0/go.mod h1:OMW5y6CY9l38uPLmxU6l6pwcXp1obtLo3e6gT7gQR2I=
|
||||
golang.org/x/exp v0.0.0-20180321215751-8460e604b9de/go.mod h1:CJ0aWSM057203Lf6IL+f9T1iT9GByDxfZKAQTCR3kQA=
|
||||
golang.org/x/exp v0.0.0-20180807140117-3d87b88a115f/go.mod h1:CJ0aWSM057203Lf6IL+f9T1iT9GByDxfZKAQTCR3kQA=
|
||||
golang.org/x/exp v0.0.0-20190121172915-509febef88a4/go.mod h1:CJ0aWSM057203Lf6IL+f9T1iT9GByDxfZKAQTCR3kQA=
|
||||
@@ -2225,8 +2225,8 @@ golang.org/x/image v0.0.0-20210607152325-775e3b0c77b9/go.mod h1:023OzeP/+EPmXeap
|
||||
golang.org/x/image v0.0.0-20210628002857-a66eb6448b8d/go.mod h1:023OzeP/+EPmXeapQh35lcL3II3LrY8Ic+EFFKVhULM=
|
||||
golang.org/x/image v0.0.0-20211028202545-6944b10bf410/go.mod h1:023OzeP/+EPmXeapQh35lcL3II3LrY8Ic+EFFKVhULM=
|
||||
golang.org/x/image v0.0.0-20220302094943-723b81ca9867/go.mod h1:023OzeP/+EPmXeapQh35lcL3II3LrY8Ic+EFFKVhULM=
|
||||
golang.org/x/image v0.45.0 h1:FMb1nTbH5H9vF55SriQHgFw5GnNL9Jg6L25BwXKzhB0=
|
||||
golang.org/x/image v0.45.0/go.mod h1:n62x/7RqlwXDvGsSU4u6IUTUf6KghUZ9Bt7cG/T9Fx4=
|
||||
golang.org/x/image v0.46.0 h1:b1+oYj0Jbp6K5MDT4i4/eZpYlk3V8SJhhDKh6LBHAyQ=
|
||||
golang.org/x/image v0.46.0/go.mod h1:3B3W05VGVQyuXucLINLjXKrqISASfi4Xj+iCVkLMwew=
|
||||
golang.org/x/lint v0.0.0-20181026193005-c67002cb31c3/go.mod h1:UVdnD1Gm6xHRNCYTkRU2/jEulfH38KcIWyp/GAMgvoE=
|
||||
golang.org/x/lint v0.0.0-20190227174305-5b3e6a55c961/go.mod h1:wehouNa3lNwaWXcvxsM5YxQ5yQlVC4a0KAMCusXpPoU=
|
||||
golang.org/x/lint v0.0.0-20190301231843-5614ed5bae6f/go.mod h1:UVdnD1Gm6xHRNCYTkRU2/jEulfH38KcIWyp/GAMgvoE=
|
||||
@@ -2258,8 +2258,8 @@ golang.org/x/mod v0.8.0/go.mod h1:iBbtSCu2XBx23ZKBPSOrRkjjQPZFPuis4dIYUhu/chs=
|
||||
golang.org/x/mod v0.9.0/go.mod h1:iBbtSCu2XBx23ZKBPSOrRkjjQPZFPuis4dIYUhu/chs=
|
||||
golang.org/x/mod v0.12.0/go.mod h1:iBbtSCu2XBx23ZKBPSOrRkjjQPZFPuis4dIYUhu/chs=
|
||||
golang.org/x/mod v0.13.0/go.mod h1:hTbmBsO62+eylJbnUtE2MGJUyE7QWk4xUqPFrRgJ+7c=
|
||||
golang.org/x/mod v0.38.0 h1:MECBjubtXD7yj4HrhIUcywNaGeNVUdfVnxmPajOk4yk=
|
||||
golang.org/x/mod v0.38.0/go.mod h1:V6Xz0pq8TQ3dGqVQ1FVHuelZpAL0uNhSkk9ogYP3c40=
|
||||
golang.org/x/mod v0.41.0 h1:qJmnOUb4YB+FsEuM3HcWucdZASCPGhsX6uljO6pog0c=
|
||||
golang.org/x/mod v0.41.0/go.mod h1:Ek9pY8RKWXwsWvd3rQiHYtMqkjSUV+s1Rj7j4H5Ur6o=
|
||||
golang.org/x/net v0.0.0-20180724234803-3673e40ba225/go.mod h1:mL1N/T3taQHkDXs73rZJwtUhF3w3ftmwwsq0BUmARs4=
|
||||
golang.org/x/net v0.0.0-20180826012351-8a410e7b638d/go.mod h1:mL1N/T3taQHkDXs73rZJwtUhF3w3ftmwwsq0BUmARs4=
|
||||
golang.org/x/net v0.0.0-20180906233101-161cd47e91fd/go.mod h1:mL1N/T3taQHkDXs73rZJwtUhF3w3ftmwwsq0BUmARs4=
|
||||
@@ -2374,8 +2374,8 @@ golang.org/x/sync v0.0.0-20220929204114-8fcdb60fdcc0/go.mod h1:RxMgew5VJxzue5/jJ
|
||||
golang.org/x/sync v0.1.0/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
|
||||
golang.org/x/sync v0.3.0/go.mod h1:FU7BRWz2tNW+3quACPkgCx/L+uEAv1htQ0V83Z9Rj+Y=
|
||||
golang.org/x/sync v0.4.0/go.mod h1:FU7BRWz2tNW+3quACPkgCx/L+uEAv1htQ0V83Z9Rj+Y=
|
||||
golang.org/x/sync v0.22.0 h1:SZjpbeLmrCk4xhRSZFNZW5gFUeCeFgjekvI/+gfScek=
|
||||
golang.org/x/sync v0.22.0/go.mod h1:9xrNwdLfx4jkKbNva9FpL6vEN7evnE43NNNJQ2LF3+0=
|
||||
golang.org/x/sync v0.23.0 h1:KameEIfc1IkluZyXWLn39Wd4tURc6GbCiISGiZm2bQk=
|
||||
golang.org/x/sync v0.23.0/go.mod h1:sUUOizhqBxiL6pEWpqNLUiaJn1ShEbZ6BBqskPbjZm0=
|
||||
golang.org/x/sys v0.0.0-20180810173357-98c5dad5d1a0/go.mod h1:STP8DvDyc/dI5b8T5hshtkjS+E42TnysNCUPdjciGhY=
|
||||
golang.org/x/sys v0.0.0-20180830151530-49385e6e1522/go.mod h1:STP8DvDyc/dI5b8T5hshtkjS+E42TnysNCUPdjciGhY=
|
||||
golang.org/x/sys v0.0.0-20180905080454-ebe1bf3edb33/go.mod h1:STP8DvDyc/dI5b8T5hshtkjS+E42TnysNCUPdjciGhY=
|
||||
@@ -2477,8 +2477,8 @@ golang.org/x/sys v0.6.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.8.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.12.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.13.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.47.0 h1:o7XGOvZQCADBQQ4Y7VNq2dRWQR7JmOUW8Kxx4ZsNgWs=
|
||||
golang.org/x/sys v0.47.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw=
|
||||
golang.org/x/sys v0.48.0 h1:bbX/i/6MgT9BVLM9RT1thmxL04yeTAhbEz4SyadbXoo=
|
||||
golang.org/x/sys v0.48.0/go.mod h1:hNLxWAXmnKAxqDtdwIYC4bM9oQPEecfsnNMuSxOs3og=
|
||||
golang.org/x/term v0.0.0-20201126162022-7de9c90e9dd1/go.mod h1:bj7SfCRtBDWHUb9snDiAeCFNEtKQo2Wmx5Cou7ajbmo=
|
||||
golang.org/x/term v0.0.0-20210220032956-6a3ed077a48d/go.mod h1:bj7SfCRtBDWHUb9snDiAeCFNEtKQo2Wmx5Cou7ajbmo=
|
||||
golang.org/x/term v0.0.0-20210615171337-6886f2dfbf5b/go.mod h1:jbD1KX2456YbFQfuXm/mYQcufACuNUgVhRMnK/tPxf8=
|
||||
@@ -2511,8 +2511,8 @@ golang.org/x/text v0.8.0/go.mod h1:e1OnstbJyHTd6l/uOt8jFFHp6TRDWZR/bV3emEE/zU8=
|
||||
golang.org/x/text v0.9.0/go.mod h1:e1OnstbJyHTd6l/uOt8jFFHp6TRDWZR/bV3emEE/zU8=
|
||||
golang.org/x/text v0.13.0/go.mod h1:TvPlkZtksWOMsz7fbANvkp4WM8x/WCo/om8BMLbz+aE=
|
||||
golang.org/x/text v0.14.0/go.mod h1:18ZOQIKpY8NJVqYksKHtTdi31H5itFRjB5/qKTNYzSU=
|
||||
golang.org/x/text v0.41.0 h1:vz/seA0lnX87Othu2f/0L24RcgrXD9/YFTSuGjj3rH8=
|
||||
golang.org/x/text v0.41.0/go.mod h1:jvf1O8ajNzZqhSrQBPbutR/EB83Cc0CFrezNQIwbb5M=
|
||||
golang.org/x/text v0.42.0 h1:JbOZXgfeCPU9gacVtYliJqOhD+zhrEqK4LfdpmlUZqI=
|
||||
golang.org/x/text v0.42.0/go.mod h1:ojzP1Z+2QtioaF8DTtO8K5q7JWVVYwZKenzujK0Zd0E=
|
||||
golang.org/x/time v0.0.0-20181108054448-85acf8d2951c/go.mod h1:tRJNPiyCQ0inRvYxbN9jk5I+vvW/OXSQhTDSoE431IQ=
|
||||
golang.org/x/time v0.0.0-20190308202827-9d24e82272b4/go.mod h1:tRJNPiyCQ0inRvYxbN9jk5I+vvW/OXSQhTDSoE431IQ=
|
||||
golang.org/x/time v0.0.0-20191024005414-555d28b269f0/go.mod h1:tRJNPiyCQ0inRvYxbN9jk5I+vvW/OXSQhTDSoE431IQ=
|
||||
@@ -2589,8 +2589,8 @@ golang.org/x/tools v0.6.0/go.mod h1:Xwgl3UAJ/d3gWutnCtw505GrjyAbvKui8lOU390QaIU=
|
||||
golang.org/x/tools v0.7.0/go.mod h1:4pg6aUX35JBAogB10C9AtvVL+qowtN4pT3CGSQex14s=
|
||||
golang.org/x/tools v0.13.0/go.mod h1:HvlwmtVNQAhOuCjW7xxvovg8wbNq7LwfXh/k7wXUl58=
|
||||
golang.org/x/tools v0.14.0/go.mod h1:uYBEerGOWcJyEORxN+Ek8+TT266gXkNlHdJBwexUsBg=
|
||||
golang.org/x/tools v0.48.0 h1:3+hClM1aLL5mjMKm5ovokw9epgRXPuu2tILgismM6RE=
|
||||
golang.org/x/tools v0.48.0/go.mod h1:08xX0orndb/F7jJxGDicx061tyd5pcMto75YMAXr6lk=
|
||||
golang.org/x/tools v0.49.0 h1:3NI7VXzL9+1WZD52Dx2ttoPwD5DWrFGpl9mFZDlmisI=
|
||||
golang.org/x/tools v0.49.0/go.mod h1:SJNXV9DBKT0UbdttsQjbfJlAE/q+y36++zo3uL3N0Oo=
|
||||
golang.org/x/tools/godoc v0.1.0-deprecated h1:o+aZ1BOj6Hsx/GBdJO/s815sqftjSnrZZwyYTHODvtk=
|
||||
golang.org/x/tools/godoc v0.1.0-deprecated/go.mod h1:qM63CriJ961IHWmnWa9CjZnBndniPt4a3CK0PVB9bIg=
|
||||
golang.org/x/xerrors v0.0.0-20190717185122-a985d3407aa7/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0=
|
||||
@@ -2668,8 +2668,8 @@ google.golang.org/api v0.106.0/go.mod h1:2Ts0XTHNVWxypznxWOYUeI4g3WdP9Pk2Qk58+a/
|
||||
google.golang.org/api v0.107.0/go.mod h1:2Ts0XTHNVWxypznxWOYUeI4g3WdP9Pk2Qk58+a/O9MY=
|
||||
google.golang.org/api v0.108.0/go.mod h1:2Ts0XTHNVWxypznxWOYUeI4g3WdP9Pk2Qk58+a/O9MY=
|
||||
google.golang.org/api v0.110.0/go.mod h1:7FC4Vvx1Mooxh8C5HWjzZHcavuS2f6pmJpZx60ca7iI=
|
||||
google.golang.org/api v0.296.0 h1:Nn5EHeKdGx70MFClaV/II0gsWUm6xhEjb0xYLylVvaA=
|
||||
google.golang.org/api v0.296.0/go.mod h1:02qB8+Ox1ZFzcaKFMguy1nQLJmSIyvV6Ff4txJEXtl4=
|
||||
google.golang.org/api v0.297.0 h1:WktxTsnnx0yZNnsR6j0q6hR21RnnK81FHTOPy/ux4OE=
|
||||
google.golang.org/api v0.297.0/go.mod h1:S4m8x0M6OkQpkOzGk1y9JG2sm4fFQrMh6dxzjCTszhE=
|
||||
google.golang.org/appengine v1.1.0/go.mod h1:EbEs0AVv82hx2wNQdGPgUI5lhzA/G0D9YwlJXL52JkM=
|
||||
google.golang.org/appengine v1.4.0/go.mod h1:xpcJRLb0r/rnEns0DIKYYv+WjYCduHsrkT7/EB5XEv4=
|
||||
google.golang.org/appengine v1.5.0/go.mod h1:xpcJRLb0r/rnEns0DIKYYv+WjYCduHsrkT7/EB5XEv4=
|
||||
@@ -2805,8 +2805,8 @@ google.golang.org/genproto v0.0.0-20230222225845-10f96fb3dbec/go.mod h1:3Dl5ZL0q
|
||||
google.golang.org/genproto v0.0.0-20230306155012-7f2fa6fef1f4/go.mod h1:NWraEVixdDnqcqQ30jipen1STv2r/n24Wb7twVTGR4s=
|
||||
google.golang.org/genproto v0.0.0-20260715232425-e75dac1f907d h1:C9v1o0/4quuhOAfmRXA2j+we0PqZIp8traLdeogF3Ms=
|
||||
google.golang.org/genproto v0.0.0-20260715232425-e75dac1f907d/go.mod h1:Wz2wFJntZFmLGo7pLDXZ3wYk5hyc0Mb+SkHhDDXT+lU=
|
||||
google.golang.org/genproto/googleapis/api v0.0.0-20260715232425-e75dac1f907d h1:QwnJwPte4XXAkhPu26LTDIahnsMSUV0kK8HkxbC+Pc4=
|
||||
google.golang.org/genproto/googleapis/api v0.0.0-20260715232425-e75dac1f907d/go.mod h1:WRrQ7/7N19PypuT0fxLOL5Lq0waoiRri4FbtHDEKrGE=
|
||||
google.golang.org/genproto/googleapis/api v0.0.0-20260817212433-ac3dfec99bb1 h1:lrupDmKL3p5kEX1M92oan027eCKcouzjuPbH6YBK+Rs=
|
||||
google.golang.org/genproto/googleapis/api v0.0.0-20260817212433-ac3dfec99bb1/go.mod h1:q/3oV3jAi5vwelxsVAprMBC8BcM2zmNe+IjRGd+9/ks=
|
||||
google.golang.org/genproto/googleapis/rpc v0.0.0-20260819154853-08b0e4226688 h1:cYNAzI2sUwhmCcoj9TxvihSrqsxt6uIkj3rDRhSDmW4=
|
||||
google.golang.org/genproto/googleapis/rpc v0.0.0-20260819154853-08b0e4226688/go.mod h1:DjtHYE8FKJLivXcBEjGwndXfIC23G0VpXiXKqG179uA=
|
||||
google.golang.org/grpc v1.19.0/go.mod h1:mqu4LbDTu4XGKhr4mRzUsmM4RtVoemTSY81AxZiDr8c=
|
||||
@@ -2849,8 +2849,8 @@ google.golang.org/grpc v1.51.0/go.mod h1:wgNDFcnuBGmxLKI/qn4T+m5BtEBYXJPvibbUPsA
|
||||
google.golang.org/grpc v1.52.0/go.mod h1:pu6fVzoFb+NBYNAvQL08ic+lvB2IojljRYuun5vorUY=
|
||||
google.golang.org/grpc v1.53.0/go.mod h1:OnIrk0ipVdj4N5d9IUoFUx72/VlD7+jUsHwZgwSMQpw=
|
||||
google.golang.org/grpc v1.55.0/go.mod h1:iYEXKGkEBhg1PjZQvoYEVPTDkHo1/bjTnfwTeGONTY8=
|
||||
google.golang.org/grpc v1.85.0-dev h1:HxkDyKIIZPpFnroC56tQv5gNuKTmVvi0t7TzOf5zt7g=
|
||||
google.golang.org/grpc v1.85.0-dev/go.mod h1:ljCht0DrxQrXBDRTZp52Qxh3Ffk8CdYm2sj4O2QN2C0=
|
||||
google.golang.org/grpc v1.85.0-dev.0.20260915183914-4e49413dcab7 h1:5+EEM1fC0yjOZID0NUZVrE2+8M/+1TclNrSz/l1xMYs=
|
||||
google.golang.org/grpc v1.85.0-dev.0.20260915183914-4e49413dcab7/go.mod h1:Ovl0ECo4xx5r4kn/6d4BPSNB7OIFuu6EAjOzjtVAKaM=
|
||||
google.golang.org/grpc/cmd/protoc-gen-go-grpc v1.1.0/go.mod h1:6Kw0yEErY5E/yWrBtf03jp27GLLJujG4z/JK95pnjjw=
|
||||
google.golang.org/grpc/examples v0.0.0-20250407062114-b368379ef8f6 h1:ExN12ndbJ608cboPYflpTny6mXSzPrDLh0iTaVrRrds=
|
||||
google.golang.org/grpc/examples v0.0.0-20250407062114-b368379ef8f6/go.mod h1:6ytKWczdvnpnO+m+JiG9NjEDzR1FJfsnmJdG7B8QVZ8=
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
apiVersion: v1
|
||||
description: SeaweedFS
|
||||
name: seaweedfs
|
||||
appVersion: "4.46"
|
||||
appVersion: "4.47"
|
||||
# Dev note: Trigger a helm chart release by `git tag -a helm-<version>`
|
||||
version: 4.46.0
|
||||
version: 4.47.1
|
||||
|
||||
@@ -286,7 +286,7 @@ metadata:
|
||||
app.kubernetes.io/component: s3
|
||||
stringData:
|
||||
# this key must be an inline json config file
|
||||
seaweedfs_s3_config: '{"identities":[{"name":"anvAdmin","credentials":[{"accessKey":"snu8yoP6QAlY0ne4","secretKey":"PNzBcmeLNEdR0oviwm04NQAicOrDH1Km"}],"actions":["Admin","Read","Write"]},{"name":"anvReadOnly","credentials":[{"accessKey":"SCigFee6c5lbi04A","secretKey":"kgFhbT38R8WUYVtiFQ1OiSVOrYr3NKku"}],"actions":["Read"]}]}'
|
||||
seaweedfs_s3_config: '{"identities":[{"name":"anvAdmin","credentials":[{"accessKey":"snu8yoP6QAlY0ne4","secretKey":"PNzBcmeLNEdR0oviwm04NQAicOrDH1Km"}],"actions":["Admin","Read","Write"]},{"name":"anvReadOnly","credentials":[{"accessKey":"SCigFee6c5lbi04A","secretKey":"kgFhbT38R8WUYVtiFQ1OiSVOrYr3NKku"}],"actions":["Read","List"]}]}'
|
||||
```
|
||||
|
||||
#### Source S3 credentials from an existing Secret
|
||||
@@ -363,6 +363,27 @@ If `adminPassword` is empty or not set, the admin interface runs without authent
|
||||
|
||||
As an alternative, a kubernetes Secret can be used (`admin.secret.existingSecret`).
|
||||
|
||||
### Admin listen address
|
||||
|
||||
Since SeaweedFS 4.46, `weed admin` defaults to listening on loopback (`127.0.0.1`).
|
||||
The chart's httpGet readiness/liveness probes dial the pod IP, so the admin
|
||||
server must bind a non-loopback address for the probes to succeed. The chart
|
||||
therefore passes `-ip={{ .Values.admin.ip }}`, defaulting `admin.ip` to `0.0.0.0`
|
||||
(the pre-4.46 behaviour of listening on all interfaces).
|
||||
|
||||
Binding a non-loopback address requires authentication: `weed admin` refuses to
|
||||
start on a non-loopback address without `-adminPassword`, so the chart fails at
|
||||
render time if `admin.ip` is non-loopback and authentication is not configured via
|
||||
`admin.secret.adminPassword`, `admin.secret.existingSecret`, or
|
||||
`WEED_ADMIN_PASSWORD` supplied through `admin.extraEnvironmentVars` /
|
||||
`admin.secretExtraEnvironmentVars`. The whole `127.0.0.0/8` range and `::1` are
|
||||
treated as loopback (matching `weed admin`); `localhost` is treated as
|
||||
non-loopback. Set `admin.ip` to a loopback address only if you also replace the
|
||||
httpGet probes (e.g. with an `exec` probe that checks `127.0.0.1`).
|
||||
|
||||
The `-ip` flag requires SeaweedFS 4.46 or newer; pinning `admin.imageOverride`
|
||||
to an older image is not supported with this chart version.
|
||||
|
||||
### Admin Data Persistence
|
||||
|
||||
The admin component can store configuration and maintenance data. You can configure storage in several ways:
|
||||
|
||||
@@ -0,0 +1,7 @@
|
||||
# Admin install: exercises the admin StatefulSet, which passes -ip (default
|
||||
# 0.0.0.0) and therefore requires authentication to bind a non-loopback address.
|
||||
admin:
|
||||
enabled: true
|
||||
secret:
|
||||
adminUser: "admin"
|
||||
adminPassword: "ci-admin-password"
|
||||
@@ -6,6 +6,11 @@
|
||||
{{- if and (not .Values.admin.masters) (not .Values.global.seaweedfs.masterServer) (not .Values.master.enabled) }}
|
||||
{{- fail "admin.masters or global.seaweedfs.masterServer must be set if master.enabled is false" -}}
|
||||
{{- end }}
|
||||
{{- $adminAuthEnabled := include "seaweedfs.admin.authEnabled" . }}
|
||||
{{- $adminIp := .Values.admin.ip | default "0.0.0.0" }}
|
||||
{{- if and (not (include "seaweedfs.admin.isLoopbackIp" $adminIp)) (ne $adminAuthEnabled "true") }}
|
||||
{{- fail (printf "admin.ip is set to %q (non-loopback) but admin authentication is not configured. Since `weed admin` 4.46 refuses to bind a non-loopback address without authentication, the admin container would exit on startup. Set admin.secret.adminPassword or admin.secret.existingSecret, or supply WEED_ADMIN_PASSWORD via admin.extraEnvironmentVars / admin.secretExtraEnvironmentVars, or set admin.ip to a loopback address such as 127.0.0.1 (note: a loopback bind makes the chart's httpGet readiness/liveness probes fail)." $adminIp) -}}
|
||||
{{- end }}
|
||||
apiVersion: apps/v1
|
||||
kind: StatefulSet
|
||||
metadata:
|
||||
@@ -162,6 +167,7 @@ spec:
|
||||
-v={{ .Values.global.seaweedfs.loggingLevel }} \
|
||||
{{- end }}
|
||||
admin \
|
||||
-ip={{ .Values.admin.ip | default "0.0.0.0" }} \
|
||||
-port={{ .Values.admin.port }} \
|
||||
-port.grpc={{ .Values.admin.grpcPort }} \
|
||||
{{- if or (eq .Values.admin.data.type "hostPath") (eq .Values.admin.data.type "persistentVolumeClaim") (eq .Values.admin.data.type "emptyDir") (eq .Values.admin.data.type "existingClaim") }}
|
||||
|
||||
@@ -37,13 +37,17 @@ spec:
|
||||
{{- with .Values.allInOne.podLabels }}
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
{{- $podAnnotations := mergeOverwrite (deepCopy (.Values.podAnnotations | default dict)) (.Values.allInOne.podAnnotations | default dict) }}
|
||||
{{- $existingS3ConfigSecret := or .Values.allInOne.s3.existingConfigSecret .Values.s3.existingConfigSecret .Values.filer.s3.existingConfigSecret }}
|
||||
{{- if $existingS3ConfigSecret }}
|
||||
{{- $configSecret := (lookup "v1" "Secret" .Release.Namespace $existingS3ConfigSecret) | default dict }}
|
||||
{{- $_ := set $podAnnotations "checksum/s3config" ($configSecret | toYaml | sha256sum) }}
|
||||
{{- else }}
|
||||
{{- $_ := set $podAnnotations "checksum/s3config" (include (print .Template.BasePath "/s3/s3-secret.yaml") . | sha256sum) }}
|
||||
{{- end }}
|
||||
{{- $_ := set $podAnnotations "checksum/master-config" (include (print .Template.BasePath "/master/master-configmap.yaml") . | sha256sum) }}
|
||||
annotations:
|
||||
{{- with .Values.podAnnotations }}
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
{{- with .Values.allInOne.podAnnotations }}
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
{{- toYaml $podAnnotations | nindent 8 }}
|
||||
spec:
|
||||
restartPolicy: {{ default .Values.global.seaweedfs.restartPolicy .Values.allInOne.restartPolicy }}
|
||||
{{- if .Values.allInOne.affinity }}
|
||||
|
||||
@@ -43,19 +43,15 @@ spec:
|
||||
{{- with .Values.filer.podLabels }}
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
annotations:
|
||||
{{- with .Values.podAnnotations }}
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
{{- with .Values.filer.podAnnotations }}
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
{{- $podAnnotations := mergeOverwrite (deepCopy (.Values.podAnnotations | default dict)) (.Values.filer.podAnnotations | default dict) }}
|
||||
{{- if .Values.filer.s3.existingConfigSecret }}
|
||||
{{- $configSecret := (lookup "v1" "Secret" .Release.Namespace .Values.filer.s3.existingConfigSecret) | default dict }}
|
||||
checksum/s3config: {{ $configSecret | toYaml | sha256sum }}
|
||||
{{- $configSecret := (lookup "v1" "Secret" .Release.Namespace .Values.filer.s3.existingConfigSecret) | default dict }}
|
||||
{{- $_ := set $podAnnotations "checksum/s3config" ($configSecret | toYaml | sha256sum) }}
|
||||
{{- else }}
|
||||
checksum/s3config: {{ include (print .Template.BasePath "/s3/s3-secret.yaml") . | sha256sum }}
|
||||
{{- $_ := set $podAnnotations "checksum/s3config" (include (print .Template.BasePath "/s3/s3-secret.yaml") . | sha256sum) }}
|
||||
{{- end }}
|
||||
annotations:
|
||||
{{- toYaml $podAnnotations | nindent 8 }}
|
||||
spec:
|
||||
restartPolicy: {{ default .Values.global.seaweedfs.restartPolicy .Values.filer.restartPolicy }}
|
||||
{{- if .Values.filer.affinity }}
|
||||
|
||||
@@ -43,13 +43,10 @@ spec:
|
||||
{{- with .Values.master.podLabels }}
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
{{- $podAnnotations := mergeOverwrite (deepCopy (.Values.podAnnotations | default dict)) (.Values.master.podAnnotations | default dict) }}
|
||||
{{- $_ := set $podAnnotations "checksum/master-config" (include (print .Template.BasePath "/master/master-configmap.yaml") . | sha256sum) }}
|
||||
annotations:
|
||||
{{ with .Values.podAnnotations }}
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
{{- with .Values.master.podAnnotations }}
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
{{- toYaml $podAnnotations | nindent 8 }}
|
||||
spec:
|
||||
restartPolicy: {{ default .Values.global.seaweedfs.restartPolicy .Values.master.restartPolicy }}
|
||||
{{- if .Values.master.affinity }}
|
||||
|
||||
@@ -35,13 +35,15 @@ spec:
|
||||
{{- with .Values.s3.podLabels }}
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
{{- $podAnnotations := mergeOverwrite (deepCopy (.Values.podAnnotations | default dict)) (.Values.s3.podAnnotations | default dict) }}
|
||||
{{- if .Values.s3.existingConfigSecret }}
|
||||
{{- $configSecret := (lookup "v1" "Secret" .Release.Namespace .Values.s3.existingConfigSecret) | default dict }}
|
||||
{{- $_ := set $podAnnotations "checksum/s3config" ($configSecret | toYaml | sha256sum) }}
|
||||
{{- else }}
|
||||
{{- $_ := set $podAnnotations "checksum/s3config" (include (print .Template.BasePath "/s3/s3-secret.yaml") . | sha256sum) }}
|
||||
{{- end }}
|
||||
annotations:
|
||||
{{ with .Values.podAnnotations }}
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
{{- with .Values.s3.podAnnotations }}
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
{{- toYaml $podAnnotations | nindent 8 }}
|
||||
spec:
|
||||
restartPolicy: {{ default .Values.global.seaweedfs.restartPolicy .Values.s3.restartPolicy }}
|
||||
{{- if .Values.s3.affinity }}
|
||||
|
||||
@@ -60,7 +60,7 @@ stringData:
|
||||
read_access_key_id: {{ $access_key_read }}
|
||||
read_secret_access_key: {{ $secret_key_read }}
|
||||
{{- end }}
|
||||
seaweedfs_s3_config: '{"identities":[{"name":"anvAdmin","credentials":[{"accessKey":"{{ $access_key_admin }}","secretKey":"{{ $secret_key_admin }}"}],"actions":["Admin","Read","Write"]},{"name":"anvReadOnly","credentials":[{"accessKey":"{{ $access_key_read }}","secretKey":"{{ $secret_key_read }}"}],"actions":["Read"]}]}'
|
||||
seaweedfs_s3_config: '{"identities":[{"name":"anvAdmin","credentials":[{"accessKey":"{{ $access_key_admin }}","secretKey":"{{ $secret_key_admin }}"}],"actions":["Admin","Read","Write"]},{"name":"anvReadOnly","credentials":[{"accessKey":"{{ $access_key_read }}","secretKey":"{{ $secret_key_read }}"}],"actions":["Read","List"]}]}'
|
||||
{{- if .Values.filer.s3.auditLogConfig }}
|
||||
filer_s3_auditLogConfig.json: |
|
||||
{{ toJson .Values.filer.s3.auditLogConfig | nindent 4 }}
|
||||
|
||||
@@ -88,6 +88,43 @@ true
|
||||
{{- end -}}
|
||||
{{- end -}}
|
||||
|
||||
{{/* Classify an admin bind address as loopback, mirroring weed admin's
|
||||
isLoopbackIp (net.ParseIP + IsLoopback). Helm templates cannot call
|
||||
net.ParseIP, so we approximate: valid IPv4 addresses in 127.0.0.0/8
|
||||
(validated via regex to reject malformed values like "127.not-an-ip")
|
||||
and the IPv6 loopback "::1" / its expanded form "0:0:0:0:0:0:0:1" are
|
||||
loopback. Hostnames (e.g. "localhost") and wildcard addresses
|
||||
("0.0.0.0", "::") are non-loopback, matching the binary, which
|
||||
treats unparseable hostnames as non-loopback to be safe. Other IPv6
|
||||
loopback representations are not matched; the binary's own runtime
|
||||
validation is the authoritative guard. */}}
|
||||
{{- define "seaweedfs.admin.isLoopbackIp" -}}
|
||||
{{- $ip := toString . -}}
|
||||
{{- if or (regexMatch "^127\\.[0-9]{1,3}\\.[0-9]{1,3}\\.[0-9]{1,3}$" $ip) (eq $ip "::1") (eq $ip "0:0:0:0:0:0:0:1") -}}
|
||||
true
|
||||
{{- end -}}
|
||||
{{- end -}}
|
||||
|
||||
{{/* Whether admin authentication is enabled from any supported source:
|
||||
admin.secret (adminPassword or existingSecret), or WEED_ADMIN_PASSWORD
|
||||
supplied via extraEnvironmentVars / secretExtraEnvironmentVars (which
|
||||
weed admin picks up through viper's AutomaticEnv). A secret-backed
|
||||
entry counts as enabled even though the chart cannot read its value. */}}
|
||||
{{- define "seaweedfs.admin.authEnabled" -}}
|
||||
{{- if or .Values.admin.secret.existingSecret .Values.admin.secret.adminPassword -}}
|
||||
true
|
||||
{{- else -}}
|
||||
{{- $merged := dict -}}
|
||||
{{- $_ := include "seaweedfs.mergeExtraEnvironmentVars" (dict "global" .Values.global.seaweedfs "component" .Values.admin "target" $merged) -}}
|
||||
{{- $envPassword := index $merged "WEED_ADMIN_PASSWORD" -}}
|
||||
{{- if or (kindIs "map" $envPassword) (hasKey (.Values.admin.secretExtraEnvironmentVars | default dict) "WEED_ADMIN_PASSWORD") -}}
|
||||
true
|
||||
{{- else if and $envPassword (ne (toString $envPassword) "") -}}
|
||||
true
|
||||
{{- end -}}
|
||||
{{- end -}}
|
||||
{{- end -}}
|
||||
|
||||
{{/* Return the proper filer image */}}
|
||||
{{- define "seaweedfs.filer.image" -}}
|
||||
{{- if .Values.filer.imageOverride -}}
|
||||
|
||||
@@ -190,6 +190,8 @@ master:
|
||||
podLabels: {}
|
||||
|
||||
# Annotations to be added to the master pods
|
||||
# The chart sets checksum/master-config on master pods; other checksum/* keys
|
||||
# can be used for custom rollouts.
|
||||
podAnnotations: {}
|
||||
|
||||
# Annotations to be added to the master resources
|
||||
@@ -773,6 +775,8 @@ filer:
|
||||
podLabels: {}
|
||||
|
||||
# Annotations to be added to the filer pods
|
||||
# The chart sets checksum/s3config on filer pods; other checksum/* keys can be
|
||||
# used for custom rollouts.
|
||||
podAnnotations: {}
|
||||
|
||||
# Annotations to be added to the filer resource
|
||||
@@ -1078,6 +1082,8 @@ s3:
|
||||
podLabels: {}
|
||||
|
||||
# Annotations to be added to the s3 pods
|
||||
# The chart sets checksum/s3config on s3 pods; other checksum/* keys can be
|
||||
# used for custom rollouts.
|
||||
podAnnotations: {}
|
||||
|
||||
# Annotations to be added to the s3 resources
|
||||
@@ -1328,6 +1334,20 @@ admin:
|
||||
replicas: 1
|
||||
port: 23646 # Default admin port
|
||||
grpcPort: 33646 # Default gRPC port for worker connections
|
||||
# IP address the admin server listens on. Since `weed admin` 4.46 defaults to
|
||||
# loopback (127.0.0.1), the chart must bind a non-loopback address for the
|
||||
# kubelet's httpGet readiness/liveness probes (which dial the pod IP) to ever
|
||||
# succeed. "0.0.0.0" restores the pre-4.46 behaviour of listening on all
|
||||
# interfaces. A non-loopback address requires authentication: set
|
||||
# admin.secret.adminPassword or admin.secret.existingSecret, or supply
|
||||
# WEED_ADMIN_PASSWORD via admin.extraEnvironmentVars /
|
||||
# admin.secretExtraEnvironmentVars; otherwise the admin container will exit
|
||||
# with a clear error rather than silently staying unready. The whole
|
||||
# 127.0.0.0/8 range and ::1 are treated as loopback (matching weed admin).
|
||||
# Set to a loopback address only if you also replace the httpGet probes.
|
||||
# Note: the -ip flag requires SeaweedFS 4.46 or newer; pinning
|
||||
# admin.imageOverride to an older image is not supported with this chart.
|
||||
ip: "0.0.0.0"
|
||||
loggingOverrideLevel: null
|
||||
|
||||
# Admin authentication
|
||||
@@ -1774,6 +1794,8 @@ allInOne:
|
||||
initContainers: "" # Init containers
|
||||
sidecars: "" # Sidecar containers
|
||||
annotations: {} # Annotations for the deployment
|
||||
# The chart sets checksum/master-config and checksum/s3config on all-in-one
|
||||
# pods; other checksum/* keys can be used for custom rollouts.
|
||||
podAnnotations: {} # Annotations for the pods
|
||||
podLabels: {} # Labels for the pods
|
||||
|
||||
@@ -1908,6 +1930,9 @@ certificates:
|
||||
# Labels to be added to all the created pods
|
||||
podLabels: {}
|
||||
# Annotations to be added to all the created pods
|
||||
# The chart sets checksum/master-config and checksum/s3config on pods whose
|
||||
# rendered ConfigMaps or Secrets should trigger rollouts. Other checksum/* keys
|
||||
# can be used for custom rollout annotations.
|
||||
podAnnotations: {}
|
||||
|
||||
networkPolicy:
|
||||
|
||||
+1027
-1059
File diff suppressed because it is too large
Load Diff
|
Before Width: | Height: | Size: 54 KiB After Width: | Height: | Size: 53 KiB |
Generated
+112
-127
@@ -503,7 +503,7 @@ dependencies = [
|
||||
"rustls-pki-types",
|
||||
"tokio",
|
||||
"tokio-rustls",
|
||||
"tower 0.5.3",
|
||||
"tower",
|
||||
"tracing",
|
||||
]
|
||||
|
||||
@@ -628,13 +628,13 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "axum"
|
||||
version = "0.7.9"
|
||||
version = "0.8.9"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "edca88bc138befd0323b20752846e6587272d3b03b0343c8ea28a6f819e6e71f"
|
||||
checksum = "31b698c5f9a010f6573133b09e0de5408834d0c82f8d7475a89fc1867a71cd90"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"axum-core",
|
||||
"bytes",
|
||||
"form_urlencoded",
|
||||
"futures-util",
|
||||
"http 1.4.0",
|
||||
"http-body 1.0.1",
|
||||
@@ -648,14 +648,13 @@ dependencies = [
|
||||
"multer",
|
||||
"percent-encoding",
|
||||
"pin-project-lite",
|
||||
"rustversion",
|
||||
"serde",
|
||||
"serde_core",
|
||||
"serde_json",
|
||||
"serde_path_to_error",
|
||||
"serde_urlencoded",
|
||||
"sync_wrapper",
|
||||
"tokio",
|
||||
"tower 0.5.3",
|
||||
"tower",
|
||||
"tower-layer",
|
||||
"tower-service",
|
||||
"tracing",
|
||||
@@ -663,19 +662,17 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "axum-core"
|
||||
version = "0.4.5"
|
||||
version = "0.5.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "09f2bd6146b97ae3359fa0cc6d6b376d9539582c7b4220f041a33ec24c226199"
|
||||
checksum = "08c78f31d7b1291f7ee735c1c6780ccde7785daae9a9206026862dab7d8792d1"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"bytes",
|
||||
"futures-util",
|
||||
"futures-core",
|
||||
"http 1.4.0",
|
||||
"http-body 1.0.1",
|
||||
"http-body-util",
|
||||
"mime",
|
||||
"pin-project-lite",
|
||||
"rustversion",
|
||||
"sync_wrapper",
|
||||
"tower-layer",
|
||||
"tower-service",
|
||||
@@ -1654,19 +1651,13 @@ dependencies = [
|
||||
"futures-core",
|
||||
"futures-sink",
|
||||
"http 1.4.0",
|
||||
"indexmap 2.13.1",
|
||||
"indexmap",
|
||||
"slab",
|
||||
"tokio",
|
||||
"tokio-util",
|
||||
"tracing",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "hashbrown"
|
||||
version = "0.12.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "8a9ee70c43aaf417c914396645a0fa852624801b24ebb7ae78fe8272889ac888"
|
||||
|
||||
[[package]]
|
||||
name = "hashbrown"
|
||||
version = "0.14.5"
|
||||
@@ -1860,7 +1851,7 @@ dependencies = [
|
||||
"libc",
|
||||
"percent-encoding",
|
||||
"pin-project-lite",
|
||||
"socket2 0.6.3",
|
||||
"socket2",
|
||||
"tokio",
|
||||
"tower-service",
|
||||
"tracing",
|
||||
@@ -2027,16 +2018,6 @@ dependencies = [
|
||||
"quick-error",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "indexmap"
|
||||
version = "1.9.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "bd070e393353796e801d209ad339e89596eb4c8d430d18ede6a1cced8fafbd99"
|
||||
dependencies = [
|
||||
"autocfg",
|
||||
"hashbrown 0.12.3",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "indexmap"
|
||||
version = "2.13.1"
|
||||
@@ -2250,9 +2231,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "matchit"
|
||||
version = "0.7.3"
|
||||
version = "0.8.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0e7465ac9959cc2b1404e8e2367b43684a6d13790fe23056cc8c6c5a6b7bcb94"
|
||||
checksum = "47e1ffaa40ddd1f3ed91f717a33c8c0ee23fff369e3aa8772b9605cc1d22f4c3"
|
||||
|
||||
[[package]]
|
||||
name = "md-5"
|
||||
@@ -2619,17 +2600,18 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b4c5cc86750666a3ed20bdaf5ca2a0344f9c67674cae0515bec2da16fbaa47db"
|
||||
dependencies = [
|
||||
"fixedbitset 0.4.2",
|
||||
"indexmap 2.13.1",
|
||||
"indexmap",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "petgraph"
|
||||
version = "0.7.1"
|
||||
version = "0.8.3"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3672b37090dbd86368a4145bc067582552b29c27377cad4e0a306c97f9bd7772"
|
||||
checksum = "8701b58ea97060d5e5b155d383a69952a60943f0e6dfe30b04c287beb0b27455"
|
||||
dependencies = [
|
||||
"fixedbitset 0.5.7",
|
||||
"indexmap 2.13.1",
|
||||
"hashbrown 0.15.5",
|
||||
"indexmap",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -2836,12 +2818,12 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "prost"
|
||||
version = "0.13.5"
|
||||
version = "0.14.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "2796faa41db3ec313a31f7624d9286acf277b52de526150b7e69f3debf891ee5"
|
||||
checksum = "528ac67416ff8646872a3c02cad9cc4ee5dc9f9540c9b10771855c95cb2e5ae1"
|
||||
dependencies = [
|
||||
"bytes",
|
||||
"prost-derive 0.13.5",
|
||||
"prost-derive 0.14.4",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -2867,19 +2849,20 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "prost-build"
|
||||
version = "0.13.5"
|
||||
version = "0.14.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "be769465445e8c1474e9c5dac2018218498557af32d9ed057325ec9a41ae81bf"
|
||||
checksum = "03da047801ff44bb6a4d407d4860c05fd70bb81714e6b2f3812603d5b145b042"
|
||||
dependencies = [
|
||||
"heck",
|
||||
"itertools 0.14.0",
|
||||
"log",
|
||||
"multimap",
|
||||
"once_cell",
|
||||
"petgraph 0.7.1",
|
||||
"petgraph 0.8.3",
|
||||
"prettyplease",
|
||||
"prost 0.13.5",
|
||||
"prost-types 0.13.5",
|
||||
"prost 0.14.4",
|
||||
"prost-types 0.14.4",
|
||||
"pulldown-cmark",
|
||||
"pulldown-cmark-to-cmark",
|
||||
"regex",
|
||||
"syn",
|
||||
"tempfile",
|
||||
@@ -2900,9 +2883,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "prost-derive"
|
||||
version = "0.13.5"
|
||||
version = "0.14.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "8a56d757972c98b346a9b766e3f02746cde6dd1cd1d1d563472929fdd74bec4d"
|
||||
checksum = "b570b25f7617e43d59005d0990ccb79e950a423952cea19671b7a876da390adf"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"itertools 0.14.0",
|
||||
@@ -2922,11 +2905,11 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "prost-types"
|
||||
version = "0.13.5"
|
||||
version = "0.14.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "52c2c1bf36ddb1a1c396b3601a3cec27c2462e45f07c386894ec3ccf5332bd16"
|
||||
checksum = "f94967dc7688f3054c7fac87473ffae4cc4c3904800e2d9f5b857246d8963b0a"
|
||||
dependencies = [
|
||||
"prost 0.13.5",
|
||||
"prost 0.14.4",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -2993,6 +2976,26 @@ version = "3.2.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "95067976aca6421a523e491fce939a3e65249bac4b977adee0ee9771568e8aa3"
|
||||
|
||||
[[package]]
|
||||
name = "pulldown-cmark"
|
||||
version = "0.13.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e9f068eba8e7071c5f9511831b44f32c740d5adf574e990f946ddb53db2f314e"
|
||||
dependencies = [
|
||||
"bitflags 2.11.0",
|
||||
"memchr",
|
||||
"unicase",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "pulldown-cmark-to-cmark"
|
||||
version = "22.0.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ab1ad36992cead65f02aa399a373a42730922f1525d988172634fdefdecb8a60"
|
||||
dependencies = [
|
||||
"pulldown-cmark",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "pxfm"
|
||||
version = "0.1.28"
|
||||
@@ -3018,7 +3021,7 @@ dependencies = [
|
||||
"quinn-udp",
|
||||
"rustc-hash",
|
||||
"rustls",
|
||||
"socket2 0.6.3",
|
||||
"socket2",
|
||||
"thiserror 2.0.18",
|
||||
"tokio",
|
||||
"tracing",
|
||||
@@ -3056,7 +3059,7 @@ dependencies = [
|
||||
"cfg_aliases",
|
||||
"libc",
|
||||
"once_cell",
|
||||
"socket2 0.6.3",
|
||||
"socket2",
|
||||
"tracing",
|
||||
"windows-sys 0.60.2",
|
||||
]
|
||||
@@ -3262,8 +3265,8 @@ dependencies = [
|
||||
"tokio",
|
||||
"tokio-rustls",
|
||||
"tokio-util",
|
||||
"tower 0.5.3",
|
||||
"tower-http 0.6.8",
|
||||
"tower",
|
||||
"tower-http",
|
||||
"tower-service",
|
||||
"url",
|
||||
"wasm-bindgen",
|
||||
@@ -3719,16 +3722,6 @@ version = "1.1.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "1b6b67fb9a61334225b5b790716f609cd58395f895b3fe8b328786812a40bc3b"
|
||||
|
||||
[[package]]
|
||||
name = "socket2"
|
||||
version = "0.5.10"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e22376abed350d73dd1cd119b57ffccad95b4e585a7cda43e286245ce23c0678"
|
||||
dependencies = [
|
||||
"libc",
|
||||
"windows-sys 0.52.0",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "socket2"
|
||||
version = "0.6.3"
|
||||
@@ -3990,7 +3983,7 @@ dependencies = [
|
||||
"parking_lot 0.12.5",
|
||||
"pin-project-lite",
|
||||
"signal-hook-registry",
|
||||
"socket2 0.6.3",
|
||||
"socket2",
|
||||
"tokio-macros",
|
||||
"windows-sys 0.61.2",
|
||||
]
|
||||
@@ -4077,7 +4070,7 @@ version = "0.22.27"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "41fe8c660ae4257887cf66394862d21dbca4a6ddd26f04a3560410406a2f819a"
|
||||
dependencies = [
|
||||
"indexmap 2.13.1",
|
||||
"indexmap",
|
||||
"serde",
|
||||
"serde_spanned",
|
||||
"toml_datetime",
|
||||
@@ -4093,11 +4086,10 @@ checksum = "5d99f8c9a7727884afe522e9bd5edbfc91a3312b36a77b5fb8926e4c31a41801"
|
||||
|
||||
[[package]]
|
||||
name = "tonic"
|
||||
version = "0.12.3"
|
||||
version = "0.14.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "877c5b330756d856ffcc4553ab34a5684481ade925ecc54bcd1bf02b1d0d4d52"
|
||||
checksum = "ac2a5518c70fa84342385732db33fb3f44bc4cc748936eb5833d2df34d6445ef"
|
||||
dependencies = [
|
||||
"async-stream",
|
||||
"async-trait",
|
||||
"axum",
|
||||
"base64",
|
||||
@@ -4111,13 +4103,12 @@ dependencies = [
|
||||
"hyper-util",
|
||||
"percent-encoding",
|
||||
"pin-project",
|
||||
"prost 0.13.5",
|
||||
"rustls-pemfile",
|
||||
"socket2 0.5.10",
|
||||
"socket2",
|
||||
"sync_wrapper",
|
||||
"tokio",
|
||||
"tokio-rustls",
|
||||
"tokio-stream",
|
||||
"tower 0.4.13",
|
||||
"tower",
|
||||
"tower-layer",
|
||||
"tower-service",
|
||||
"tracing",
|
||||
@@ -4125,49 +4116,55 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "tonic-build"
|
||||
version = "0.12.3"
|
||||
version = "0.14.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9557ce109ea773b399c9b9e5dca39294110b74f1f342cb347a80d1fce8c26a11"
|
||||
checksum = "c68f61875ac5293cf72e6c8cf0158086428c82c37229e98c840878f1706b0322"
|
||||
dependencies = [
|
||||
"prettyplease",
|
||||
"proc-macro2",
|
||||
"prost-build 0.13.5",
|
||||
"prost-types 0.13.5",
|
||||
"quote",
|
||||
"syn",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "tonic-reflection"
|
||||
version = "0.12.3"
|
||||
name = "tonic-prost"
|
||||
version = "0.14.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "878d81f52e7fcfd80026b7fdb6a9b578b3c3653ba987f87f0dce4b64043cba27"
|
||||
checksum = "50849f68853be452acf590cde0b146665b8d507b3b8af17261df47e02c209ea0"
|
||||
dependencies = [
|
||||
"prost 0.13.5",
|
||||
"prost-types 0.13.5",
|
||||
"tokio",
|
||||
"tokio-stream",
|
||||
"bytes",
|
||||
"prost 0.14.4",
|
||||
"tonic",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "tower"
|
||||
version = "0.4.13"
|
||||
name = "tonic-prost-build"
|
||||
version = "0.14.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b8fa9be0de6cf49e536ce1851f987bd21a43b771b09473c3549a6c853db37c1c"
|
||||
checksum = "654e5643eff75d7f8c99197ce1440ed19a3474eada74c12bbac488b2cafdae27"
|
||||
dependencies = [
|
||||
"futures-core",
|
||||
"futures-util",
|
||||
"indexmap 1.9.3",
|
||||
"pin-project",
|
||||
"pin-project-lite",
|
||||
"rand 0.8.7",
|
||||
"slab",
|
||||
"prettyplease",
|
||||
"proc-macro2",
|
||||
"prost-build 0.14.4",
|
||||
"prost-types 0.14.4",
|
||||
"quote",
|
||||
"syn",
|
||||
"tempfile",
|
||||
"tonic-build",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "tonic-reflection"
|
||||
version = "0.14.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "acccd136a4bf19810a1fde9c74edc6129b42a66b44d0c1c8aaa67aeb49a146a7"
|
||||
dependencies = [
|
||||
"prost 0.14.4",
|
||||
"prost-types 0.14.4",
|
||||
"tokio",
|
||||
"tokio-util",
|
||||
"tower-layer",
|
||||
"tower-service",
|
||||
"tracing",
|
||||
"tokio-stream",
|
||||
"tonic",
|
||||
"tonic-prost",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -4178,26 +4175,12 @@ checksum = "ebe5ef63511595f1344e2d5cfa636d973292adc0eec1f0ad45fae9f0851ab1d4"
|
||||
dependencies = [
|
||||
"futures-core",
|
||||
"futures-util",
|
||||
"indexmap",
|
||||
"pin-project-lite",
|
||||
"slab",
|
||||
"sync_wrapper",
|
||||
"tokio",
|
||||
"tower-layer",
|
||||
"tower-service",
|
||||
"tracing",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "tower-http"
|
||||
version = "0.5.2"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "1e9cd434a998747dd2c4276bc96ee2e0c7a2eadf3cae88e52be55a05fa9053f5"
|
||||
dependencies = [
|
||||
"bitflags 2.11.0",
|
||||
"bytes",
|
||||
"http 1.4.0",
|
||||
"http-body 1.0.1",
|
||||
"http-body-util",
|
||||
"pin-project-lite",
|
||||
"tokio-util",
|
||||
"tower-layer",
|
||||
"tower-service",
|
||||
"tracing",
|
||||
@@ -4216,9 +4199,10 @@ dependencies = [
|
||||
"http-body 1.0.1",
|
||||
"iri-string",
|
||||
"pin-project-lite",
|
||||
"tower 0.5.3",
|
||||
"tower",
|
||||
"tower-layer",
|
||||
"tower-service",
|
||||
"tracing",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -4495,7 +4479,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "bb0e353e6a2fbdc176932bbaab493762eb1255a7900fe0fea1a2f96c296cc909"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"indexmap 2.13.1",
|
||||
"indexmap",
|
||||
"wasm-encoder",
|
||||
"wasmparser",
|
||||
]
|
||||
@@ -4521,7 +4505,7 @@ checksum = "47b807c72e1bac69382b3a6fb3dbe8ea4c0ed87ff5629b8685ae6b9a611028fe"
|
||||
dependencies = [
|
||||
"bitflags 2.11.0",
|
||||
"hashbrown 0.15.5",
|
||||
"indexmap 2.13.1",
|
||||
"indexmap",
|
||||
"semver",
|
||||
]
|
||||
|
||||
@@ -4582,7 +4566,6 @@ dependencies = [
|
||||
"image",
|
||||
"jsonwebtoken",
|
||||
"kamadak-exif",
|
||||
"lazy_static",
|
||||
"libc",
|
||||
"md-5",
|
||||
"memmap2",
|
||||
@@ -4591,8 +4574,8 @@ dependencies = [
|
||||
"parking_lot 0.12.5",
|
||||
"pprof",
|
||||
"prometheus",
|
||||
"prost 0.13.5",
|
||||
"prost-types 0.13.5",
|
||||
"prost 0.14.4",
|
||||
"prost-types 0.14.4",
|
||||
"protoc-bin-vendored",
|
||||
"rand 0.10.2",
|
||||
"redb",
|
||||
@@ -4613,13 +4596,15 @@ dependencies = [
|
||||
"tokio-stream",
|
||||
"toml",
|
||||
"tonic",
|
||||
"tonic-build",
|
||||
"tonic-prost",
|
||||
"tonic-prost-build",
|
||||
"tonic-reflection",
|
||||
"tower 0.4.13",
|
||||
"tower-http 0.5.2",
|
||||
"tower",
|
||||
"tower-http",
|
||||
"tracing",
|
||||
"tracing-subscriber",
|
||||
"uuid",
|
||||
"windows-sys 0.61.2",
|
||||
"x509-parser",
|
||||
"xxhash-rust",
|
||||
]
|
||||
@@ -4966,7 +4951,7 @@ checksum = "b7c566e0f4b284dd6561c786d9cb0142da491f46a9fbed79ea69cdad5db17f21"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"heck",
|
||||
"indexmap 2.13.1",
|
||||
"indexmap",
|
||||
"prettyplease",
|
||||
"syn",
|
||||
"wasm-metadata",
|
||||
@@ -4997,7 +4982,7 @@ checksum = "9d66ea20e9553b30172b5e831994e35fbde2d165325bec84fc43dbf6f4eb9cb2"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"bitflags 2.11.0",
|
||||
"indexmap 2.13.1",
|
||||
"indexmap",
|
||||
"log",
|
||||
"serde",
|
||||
"serde_derive",
|
||||
@@ -5016,7 +5001,7 @@ checksum = "ecc8ac4bc1dc3381b7f59c34f00b67e18f910c2c0f50015669dde7def656a736"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"id-arena",
|
||||
"indexmap 2.13.1",
|
||||
"indexmap",
|
||||
"log",
|
||||
"semver",
|
||||
"serde",
|
||||
|
||||
+23
-10
@@ -1,7 +1,10 @@
|
||||
[package]
|
||||
name = "weed-volume"
|
||||
version = "0.1.0"
|
||||
edition = "2021"
|
||||
edition = "2024"
|
||||
# The edition needs 1.85; the dependency tree needs more. Verified with
|
||||
# `cargo +1.91.1 check --all-targets` (1.90 fails on the AWS SDK).
|
||||
rust-version = "1.91.1"
|
||||
description = "SeaweedFS Volume Server — Rust implementation"
|
||||
|
||||
[lib]
|
||||
@@ -20,6 +23,11 @@ default = ["5bytes"]
|
||||
# Pulls redb's experimental_cursor (and therefore experimental-api-5).
|
||||
redb-experimental-cursor = ["redb/experimental_cursor"]
|
||||
|
||||
[lints.clippy]
|
||||
# Protobuf message literals keep `..Default::default()` on purpose: it is
|
||||
# what lets a proto gain a field without touching every constructor.
|
||||
needless_update = "allow"
|
||||
|
||||
[dependencies]
|
||||
# Async runtime
|
||||
tokio = { version = "1", features = ["full"] }
|
||||
@@ -27,25 +35,25 @@ tokio-stream = { version = "0.1", features = ["net"] }
|
||||
tokio-io-timeout = "1"
|
||||
|
||||
# gRPC + protobuf
|
||||
tonic = { version = "0.12", features = ["tls"] }
|
||||
tonic-reflection = "0.12"
|
||||
prost = "0.13"
|
||||
prost-types = "0.13"
|
||||
tonic = { version = "0.14", features = ["tls-aws-lc"] }
|
||||
tonic-prost = "0.14"
|
||||
tonic-reflection = "0.14"
|
||||
prost = "0.14"
|
||||
prost-types = "0.14"
|
||||
|
||||
# HTTP server
|
||||
axum = { version = "0.7", features = ["multipart"] }
|
||||
axum = { version = "0.8", features = ["multipart"] }
|
||||
http-body = "1"
|
||||
hyper = { version = "1", features = ["full"] }
|
||||
hyper-util = { version = "0.1", features = ["tokio", "service", "server-auto", "http1", "http2"] }
|
||||
tower = "0.4"
|
||||
tower-http = { version = "0.5", features = ["cors", "trace"] }
|
||||
tower = { version = "0.5", features = ["util"] }
|
||||
tower-http = { version = "0.6", features = ["cors", "trace"] }
|
||||
|
||||
# CLI
|
||||
clap = { version = "4", features = ["derive"] }
|
||||
|
||||
# Metrics
|
||||
prometheus = { version = "0.13", default-features = false, features = ["process"] }
|
||||
lazy_static = "1"
|
||||
|
||||
# JWT
|
||||
jsonwebtoken = { version = "10", features = ["rust_crypto"] }
|
||||
@@ -135,11 +143,16 @@ aws-types = "1"
|
||||
[target.'cfg(unix)'.dependencies]
|
||||
pprof = { version = "0.15", features = ["prost-codec"] }
|
||||
|
||||
# GetDiskFreeSpaceExW for per-path disk capacity on Windows (0.61.2 already
|
||||
# in the tree via tempfile/mio, so this unifies rather than adding a version).
|
||||
[target.'cfg(windows)'.dependencies]
|
||||
windows-sys = { version = "0.61", features = ["Win32_Storage_FileSystem"] }
|
||||
|
||||
[dev-dependencies]
|
||||
tempfile = "3"
|
||||
|
||||
[build-dependencies]
|
||||
tonic-build = "0.12"
|
||||
tonic-prost-build = "0.14"
|
||||
# Ships protoc with the build so neither CI nor a developer needs a system
|
||||
# install, and so the version is pinned rather than whatever the platform's
|
||||
# package manager happens to carry.
|
||||
|
||||
@@ -4,7 +4,10 @@ A drop-in replacement for the [SeaweedFS](https://github.com/seaweedfs/seaweedfs
|
||||
|
||||
## Building
|
||||
|
||||
Requires Rust 1.75+ (2021 edition).
|
||||
Requires Rust 1.91.1+ (2024 edition), matching `rust-version` in `Cargo.toml`.
|
||||
The patch release matters: 1.91.0 does not build. The edition itself only needs
|
||||
1.85; the higher floor comes from the dependency tree — chiefly the AWS SDK — so
|
||||
it moves with those crates. CI builds on the latest stable.
|
||||
|
||||
```bash
|
||||
cd seaweed-volume
|
||||
|
||||
@@ -3,11 +3,16 @@ fn main() -> Result<(), Box<dyn std::error::Error>> {
|
||||
// one, so the build needs no package manager and always sees the same
|
||||
// version. An explicit PROTOC still wins, for packagers supplying their own.
|
||||
if std::env::var_os("PROTOC").is_none() {
|
||||
std::env::set_var("PROTOC", protoc_bin_vendored::protoc_bin_path()?);
|
||||
// SAFETY: a build script's main runs single-threaded before anything
|
||||
// else in this process, so no other thread can be reading the
|
||||
// environment concurrently.
|
||||
unsafe {
|
||||
std::env::set_var("PROTOC", protoc_bin_vendored::protoc_bin_path()?);
|
||||
}
|
||||
}
|
||||
|
||||
let out_dir = std::path::PathBuf::from(std::env::var("OUT_DIR")?);
|
||||
tonic_build::configure()
|
||||
tonic_prost_build::configure()
|
||||
.build_server(true)
|
||||
.build_client(true)
|
||||
// filer.proto uses proto3 optional, which protoc rejects without this
|
||||
|
||||
@@ -236,6 +236,7 @@ message VolumeIncrementalCopyResponse {
|
||||
|
||||
message VolumeMountRequest {
|
||||
uint32 volume_id = 1;
|
||||
optional string collection = 2;
|
||||
}
|
||||
message VolumeMountResponse {
|
||||
}
|
||||
|
||||
+109
-82
@@ -371,17 +371,18 @@ fn merge_options_file(args: Vec<String>) -> Vec<String> {
|
||||
if arg == "--" {
|
||||
break;
|
||||
}
|
||||
if arg.starts_with("--") {
|
||||
let key = if let Some(eq) = arg.find('=') {
|
||||
arg[2..eq].to_string()
|
||||
if let Some(long) = arg.strip_prefix("--") {
|
||||
let key = if let Some(eq) = long.find('=') {
|
||||
long[..eq].to_string()
|
||||
} else {
|
||||
arg[2..].to_string()
|
||||
long.to_string()
|
||||
};
|
||||
cli_flags.insert(key);
|
||||
} else if arg.starts_with('-') && arg.len() > 2 {
|
||||
} else if arg.len() > 2
|
||||
&& let Some(without_dash) = arg.strip_prefix('-')
|
||||
{
|
||||
// Single-dash long option (already normalized to -- at this point,
|
||||
// but handle both for safety)
|
||||
let without_dash = &arg[1..];
|
||||
let key = if let Some(eq) = without_dash.find('=') {
|
||||
without_dash[..eq].to_string()
|
||||
} else {
|
||||
@@ -401,15 +402,14 @@ fn merge_options_file(args: Vec<String>) -> Vec<String> {
|
||||
}
|
||||
|
||||
// Split on first `=`, ` `, or `:`
|
||||
let (name, value) =
|
||||
if let Some(pos) = trimmed.find(|c: char| c == '=' || c == ' ' || c == ':') {
|
||||
(
|
||||
trimmed[..pos].trim().to_string(),
|
||||
trimmed[pos + 1..].trim().to_string(),
|
||||
)
|
||||
} else {
|
||||
(trimmed.to_string(), String::new())
|
||||
};
|
||||
let (name, value) = if let Some(pos) = trimmed.find(['=', ' ', ':']) {
|
||||
(
|
||||
trimmed[..pos].trim().to_string(),
|
||||
trimmed[pos + 1..].trim().to_string(),
|
||||
)
|
||||
} else {
|
||||
(trimmed.to_string(), String::new())
|
||||
};
|
||||
|
||||
// Strip leading dashes from name
|
||||
let name = name.trim_start_matches('-').to_string();
|
||||
@@ -436,10 +436,8 @@ fn merge_options_file(args: Vec<String>) -> Vec<String> {
|
||||
/// Extract the options file path from args (looks for --options or -options).
|
||||
fn find_options_arg(args: &[String]) -> String {
|
||||
for i in 1..args.len() {
|
||||
if args[i] == "--options" || args[i] == "-options" {
|
||||
if i + 1 < args.len() {
|
||||
return args[i + 1].clone();
|
||||
}
|
||||
if (args[i] == "--options" || args[i] == "-options") && i + 1 < args.len() {
|
||||
return args[i + 1].clone();
|
||||
}
|
||||
if let Some(rest) = args[i].strip_prefix("--options=") {
|
||||
return rest.to_string();
|
||||
@@ -457,20 +455,22 @@ fn parse_duration(s: &str) -> std::time::Duration {
|
||||
if s.is_empty() {
|
||||
return std::time::Duration::from_secs(60);
|
||||
}
|
||||
if let Some(secs) = s.strip_suffix('s') {
|
||||
if let Ok(v) = secs.parse::<u64>() {
|
||||
return std::time::Duration::from_secs(v);
|
||||
}
|
||||
if let Some(secs) = s.strip_suffix('s')
|
||||
&& let Ok(v) = secs.parse::<u64>()
|
||||
{
|
||||
return std::time::Duration::from_secs(v);
|
||||
}
|
||||
if let Some(mins) = s.strip_suffix('m') {
|
||||
if let Ok(v) = mins.parse::<u64>() {
|
||||
return std::time::Duration::from_secs(v * 60);
|
||||
}
|
||||
if let Some(mins) = s.strip_suffix('m')
|
||||
&& let Ok(v) = mins.parse::<u64>()
|
||||
&& let Some(seconds) = v.checked_mul(60)
|
||||
{
|
||||
return std::time::Duration::from_secs(seconds);
|
||||
}
|
||||
if let Some(hours) = s.strip_suffix('h') {
|
||||
if let Ok(v) = hours.parse::<u64>() {
|
||||
return std::time::Duration::from_secs(v * 3600);
|
||||
}
|
||||
if let Some(hours) = s.strip_suffix('h')
|
||||
&& let Ok(v) = hours.parse::<u64>()
|
||||
&& let Some(seconds) = v.checked_mul(3600)
|
||||
{
|
||||
return std::time::Duration::from_secs(seconds);
|
||||
}
|
||||
// Fallback: try parsing as raw seconds
|
||||
if let Ok(v) = s.parse::<u64>() {
|
||||
@@ -503,40 +503,40 @@ fn parse_min_free_spaces(min_free_space: &str, min_free_space_percent: &str) ->
|
||||
}
|
||||
// Try parsing human-readable bytes: e.g. "10GiB", "500MiB", "1TiB"
|
||||
let s_upper = s.to_uppercase();
|
||||
if let Some(rest) = s_upper.strip_suffix("TIB") {
|
||||
if let Ok(v) = rest.trim().parse::<f64>() {
|
||||
return MinFreeSpace::Bytes((v * 1024.0 * 1024.0 * 1024.0 * 1024.0) as u64);
|
||||
}
|
||||
if let Some(rest) = s_upper.strip_suffix("TIB")
|
||||
&& let Ok(v) = rest.trim().parse::<f64>()
|
||||
{
|
||||
return MinFreeSpace::Bytes((v * 1024.0 * 1024.0 * 1024.0 * 1024.0) as u64);
|
||||
}
|
||||
if let Some(rest) = s_upper.strip_suffix("GIB") {
|
||||
if let Ok(v) = rest.trim().parse::<f64>() {
|
||||
return MinFreeSpace::Bytes((v * 1024.0 * 1024.0 * 1024.0) as u64);
|
||||
}
|
||||
if let Some(rest) = s_upper.strip_suffix("GIB")
|
||||
&& let Ok(v) = rest.trim().parse::<f64>()
|
||||
{
|
||||
return MinFreeSpace::Bytes((v * 1024.0 * 1024.0 * 1024.0) as u64);
|
||||
}
|
||||
if let Some(rest) = s_upper.strip_suffix("MIB") {
|
||||
if let Ok(v) = rest.trim().parse::<f64>() {
|
||||
return MinFreeSpace::Bytes((v * 1024.0 * 1024.0) as u64);
|
||||
}
|
||||
if let Some(rest) = s_upper.strip_suffix("MIB")
|
||||
&& let Ok(v) = rest.trim().parse::<f64>()
|
||||
{
|
||||
return MinFreeSpace::Bytes((v * 1024.0 * 1024.0) as u64);
|
||||
}
|
||||
if let Some(rest) = s_upper.strip_suffix("KIB") {
|
||||
if let Ok(v) = rest.trim().parse::<f64>() {
|
||||
return MinFreeSpace::Bytes((v * 1024.0) as u64);
|
||||
}
|
||||
if let Some(rest) = s_upper.strip_suffix("KIB")
|
||||
&& let Ok(v) = rest.trim().parse::<f64>()
|
||||
{
|
||||
return MinFreeSpace::Bytes((v * 1024.0) as u64);
|
||||
}
|
||||
if let Some(rest) = s_upper.strip_suffix("TB") {
|
||||
if let Ok(v) = rest.trim().parse::<f64>() {
|
||||
return MinFreeSpace::Bytes((v * 1_000_000_000_000.0) as u64);
|
||||
}
|
||||
if let Some(rest) = s_upper.strip_suffix("TB")
|
||||
&& let Ok(v) = rest.trim().parse::<f64>()
|
||||
{
|
||||
return MinFreeSpace::Bytes((v * 1_000_000_000_000.0) as u64);
|
||||
}
|
||||
if let Some(rest) = s_upper.strip_suffix("GB") {
|
||||
if let Ok(v) = rest.trim().parse::<f64>() {
|
||||
return MinFreeSpace::Bytes((v * 1_000_000_000.0) as u64);
|
||||
}
|
||||
if let Some(rest) = s_upper.strip_suffix("GB")
|
||||
&& let Ok(v) = rest.trim().parse::<f64>()
|
||||
{
|
||||
return MinFreeSpace::Bytes((v * 1_000_000_000.0) as u64);
|
||||
}
|
||||
if let Some(rest) = s_upper.strip_suffix("MB") {
|
||||
if let Ok(v) = rest.trim().parse::<f64>() {
|
||||
return MinFreeSpace::Bytes((v * 1_000_000.0) as u64);
|
||||
}
|
||||
if let Some(rest) = s_upper.strip_suffix("MB")
|
||||
&& let Ok(v) = rest.trim().parse::<f64>()
|
||||
{
|
||||
return MinFreeSpace::Bytes((v * 1_000_000.0) as u64);
|
||||
}
|
||||
// Default: 1%
|
||||
MinFreeSpace::Percent(1.0)
|
||||
@@ -987,7 +987,9 @@ pub fn parse_security_config(path: &str) -> SecurityConfig {
|
||||
},
|
||||
Section::JwtSigning => match key {
|
||||
"key" => cfg.jwt_signing_key = value.as_bytes().to_vec(),
|
||||
"expires_after_seconds" => cfg.jwt_signing_expires = value.parse().unwrap_or(10),
|
||||
"expires_after_seconds" => {
|
||||
cfg.jwt_signing_expires = value.parse().unwrap_or(10)
|
||||
}
|
||||
_ => {}
|
||||
},
|
||||
Section::HttpsClient => match key {
|
||||
@@ -1028,20 +1030,20 @@ pub fn parse_security_config(path: &str) -> SecurityConfig {
|
||||
"cipher_suites" => cfg.tls_policy.cipher_suites = value.to_string(),
|
||||
_ => {}
|
||||
},
|
||||
Section::Guard => match key {
|
||||
"white_list" => {
|
||||
Section::Guard => {
|
||||
if key == "white_list" {
|
||||
cfg.guard_white_list = value
|
||||
.split(',')
|
||||
.map(|s| s.trim().to_string())
|
||||
.filter(|s| !s.is_empty())
|
||||
.collect();
|
||||
}
|
||||
_ => {}
|
||||
},
|
||||
Section::Access => match key {
|
||||
"ui" => cfg.access_ui = value.parse().unwrap_or(false),
|
||||
_ => {}
|
||||
},
|
||||
}
|
||||
Section::Access => {
|
||||
if key == "ui" {
|
||||
cfg.access_ui = value.parse().unwrap_or(false)
|
||||
}
|
||||
}
|
||||
Section::None => {}
|
||||
}
|
||||
}
|
||||
@@ -1188,12 +1190,11 @@ fn apply_env_overrides(cfg: &mut SecurityConfig) {
|
||||
/// Mirrors Go's `util.DetectedHostAddress()`.
|
||||
fn detect_host_address() -> String {
|
||||
// Connect to a remote address to determine the local outbound IP
|
||||
if let Ok(socket) = UdpSocket::bind("0.0.0.0:0") {
|
||||
if socket.connect("8.8.8.8:80").is_ok() {
|
||||
if let Ok(addr) = socket.local_addr() {
|
||||
return addr.ip().to_string();
|
||||
}
|
||||
}
|
||||
if let Ok(socket) = UdpSocket::bind("0.0.0.0:0")
|
||||
&& socket.connect("8.8.8.8:80").is_ok()
|
||||
&& let Ok(addr) = socket.local_addr()
|
||||
{
|
||||
return addr.ip().to_string();
|
||||
}
|
||||
"localhost".to_string()
|
||||
}
|
||||
@@ -1209,21 +1210,30 @@ mod tests {
|
||||
LOCK.get_or_init(|| Mutex::new(())).lock().unwrap()
|
||||
}
|
||||
|
||||
// SAFETY (all env mutation in this module): `set_var`/`remove_var` are
|
||||
// unsafe as of Rust 2024 because they race with concurrent readers in
|
||||
// other threads. Every test that reaches these helpers holds
|
||||
// `process_state_lock()` for the duration, so only one test at a time
|
||||
// touches the environment and none observes another's edit.
|
||||
fn with_temp_env_var<F: FnOnce()>(key: &str, value: Option<&str>, f: F) {
|
||||
let previous = std::env::var_os(key);
|
||||
match value {
|
||||
Some(v) => std::env::set_var(key, v),
|
||||
None => std::env::remove_var(key),
|
||||
unsafe {
|
||||
match value {
|
||||
Some(v) => std::env::set_var(key, v),
|
||||
None => std::env::remove_var(key),
|
||||
}
|
||||
}
|
||||
f();
|
||||
restore_env_var(key, previous);
|
||||
}
|
||||
|
||||
fn restore_env_var(key: &str, value: Option<OsString>) {
|
||||
if let Some(value) = value {
|
||||
std::env::set_var(key, value);
|
||||
} else {
|
||||
std::env::remove_var(key);
|
||||
unsafe {
|
||||
if let Some(value) = value {
|
||||
std::env::set_var(key, value);
|
||||
} else {
|
||||
std::env::remove_var(key);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1268,7 +1278,10 @@ mod tests {
|
||||
.collect();
|
||||
|
||||
for key in KEYS {
|
||||
std::env::remove_var(key);
|
||||
// SAFETY: as above — the caller holds `process_state_lock()`.
|
||||
unsafe {
|
||||
std::env::remove_var(key);
|
||||
}
|
||||
}
|
||||
|
||||
f();
|
||||
@@ -1285,6 +1298,14 @@ mod tests {
|
||||
assert_eq!(parse_duration("1h"), std::time::Duration::from_secs(3600));
|
||||
assert_eq!(parse_duration("30"), std::time::Duration::from_secs(30));
|
||||
assert_eq!(parse_duration(""), std::time::Duration::from_secs(60));
|
||||
assert_eq!(
|
||||
parse_duration("307445734561825861m"),
|
||||
std::time::Duration::from_secs(60)
|
||||
);
|
||||
assert_eq!(
|
||||
parse_duration("5124095576030432h"),
|
||||
std::time::Duration::from_secs(60)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -1404,12 +1425,18 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn test_resolve_config_defaults_dir_to_platform_temp_dir() {
|
||||
// resolve_config reads HOME/USERPROFILE and the WEED_* set, so it has to
|
||||
// hold the same lock the mutation helpers take — a concurrent set_var
|
||||
// during this read is exactly what makes those calls unsafe.
|
||||
let _guard = process_state_lock();
|
||||
let cfg = resolve_config(Cli::parse_from(["bin"]));
|
||||
assert_eq!(cfg.folders, vec![default_volume_dir()]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_resolve_config_index_accepts_redb_and_leveldb_aliases() {
|
||||
// As above: resolve_config reads the environment.
|
||||
let _guard = process_state_lock();
|
||||
let pairs = [
|
||||
("memory", NeedleMapKind::InMemory),
|
||||
("redb", NeedleMapKind::Redb),
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
pub mod config;
|
||||
pub mod images;
|
||||
pub mod malloc_tuning;
|
||||
pub mod metrics;
|
||||
pub mod remote_storage;
|
||||
pub mod security;
|
||||
|
||||
+48
-27
@@ -6,8 +6,8 @@ use seaweed_volume::config::{self, VolumeServerConfig};
|
||||
use seaweed_volume::metrics;
|
||||
use seaweed_volume::pb::volume_server_pb::volume_server_server::VolumeServerServer;
|
||||
use seaweed_volume::security::tls::{
|
||||
build_rustls_server_config, build_rustls_server_config_with_grpc_client_auth,
|
||||
install_default_crypto_provider, GrpcClientAuthPolicy, TlsPolicy,
|
||||
GrpcClientAuthPolicy, TlsPolicy, build_rustls_server_config,
|
||||
build_rustls_server_config_with_grpc_client_auth, install_default_crypto_provider,
|
||||
};
|
||||
use seaweed_volume::security::{Guard, SigningKey};
|
||||
#[cfg(unix)]
|
||||
@@ -18,7 +18,7 @@ use seaweed_volume::server::grpc_server::VolumeGrpcService;
|
||||
use seaweed_volume::server::profiling::CpuProfileSession;
|
||||
use seaweed_volume::server::request_id::GrpcRequestIdLayer;
|
||||
use seaweed_volume::server::volume_server::{
|
||||
build_metrics_router, RuntimeMetricsConfig, VolumeServerState,
|
||||
RuntimeMetricsConfig, VolumeServerState, build_metrics_router,
|
||||
};
|
||||
use seaweed_volume::server::write_queue::WriteQueue;
|
||||
use seaweed_volume::storage::store::Store;
|
||||
@@ -39,6 +39,11 @@ const GRPC_MAX_HEADER_LIST_SIZE: u32 = 8 * 1024 * 1024;
|
||||
const GRPC_MAX_CONCURRENT_STREAMS: u32 = 1000;
|
||||
|
||||
fn main() {
|
||||
// Before anything allocates: stop glibc from training its mmap threshold
|
||||
// upward on our large EC buffers and turning them into heap it never
|
||||
// returns. See seaweed_volume::malloc_tuning for the measurements.
|
||||
let malloc_tuning = seaweed_volume::malloc_tuning::pin_mmap_threshold();
|
||||
|
||||
install_default_crypto_provider();
|
||||
|
||||
// Initialize tracing
|
||||
@@ -65,6 +70,19 @@ fn main() {
|
||||
"SeaweedFS Volume Server (Rust) v{}",
|
||||
seaweed_volume::version::full_version()
|
||||
);
|
||||
match malloc_tuning {
|
||||
seaweed_volume::malloc_tuning::MallocTuning::Pinned(bytes) => {
|
||||
info!("pinned glibc M_MMAP_THRESHOLD to {} bytes", bytes)
|
||||
}
|
||||
seaweed_volume::malloc_tuning::MallocTuning::DeferredToEnv => info!(
|
||||
"an allocator mmap-threshold override ({}) is set; leaving glibc's mmap threshold to the environment",
|
||||
seaweed_volume::malloc_tuning::MMAP_THRESHOLD_ENV
|
||||
),
|
||||
seaweed_volume::malloc_tuning::MallocTuning::Failed => {
|
||||
warn!("mallopt(M_MMAP_THRESHOLD) failed; large freed buffers may stay resident")
|
||||
}
|
||||
seaweed_volume::malloc_tuning::MallocTuning::NotApplicable => {}
|
||||
}
|
||||
|
||||
// Register Prometheus metrics
|
||||
metrics::register_metrics();
|
||||
@@ -653,8 +671,7 @@ async fn run(
|
||||
})
|
||||
.await
|
||||
} else {
|
||||
let incoming =
|
||||
tokio_stream::wrappers::TcpListenerStream::new(grpc_listener);
|
||||
let incoming = tokio_stream::wrappers::TcpListenerStream::new(grpc_listener);
|
||||
info!("gRPC server listening on {}", grpc_local_addr);
|
||||
build_grpc_server_builder()
|
||||
.layer(GrpcRequestIdLayer)
|
||||
@@ -1040,15 +1057,17 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn test_grpc_server_tls_returns_none_when_files_are_missing() {
|
||||
assert!(build_grpc_server_tls_acceptor(
|
||||
"/missing/server.crt",
|
||||
"/missing/server.key",
|
||||
"/missing/ca.crt",
|
||||
&TlsPolicy::default(),
|
||||
"",
|
||||
&[],
|
||||
)
|
||||
.is_none());
|
||||
assert!(
|
||||
build_grpc_server_tls_acceptor(
|
||||
"/missing/server.crt",
|
||||
"/missing/server.key",
|
||||
"/missing/ca.crt",
|
||||
&TlsPolicy::default(),
|
||||
"",
|
||||
&[],
|
||||
)
|
||||
.is_none()
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -1070,19 +1089,21 @@ mod tests {
|
||||
"-----BEGIN CERTIFICATE-----\nZmFrZQ==\n-----END CERTIFICATE-----\n",
|
||||
);
|
||||
|
||||
assert!(build_grpc_server_tls_acceptor(
|
||||
&cert,
|
||||
&key,
|
||||
&ca,
|
||||
&TlsPolicy {
|
||||
min_version: "TLS 1.0".to_string(),
|
||||
max_version: "TLS 1.1".to_string(),
|
||||
cipher_suites: String::new(),
|
||||
},
|
||||
"",
|
||||
&[],
|
||||
)
|
||||
.is_none());
|
||||
assert!(
|
||||
build_grpc_server_tls_acceptor(
|
||||
&cert,
|
||||
&key,
|
||||
&ca,
|
||||
&TlsPolicy {
|
||||
min_version: "TLS 1.0".to_string(),
|
||||
max_version: "TLS 1.1".to_string(),
|
||||
cipher_suites: String::new(),
|
||||
},
|
||||
"",
|
||||
&[],
|
||||
)
|
||||
.is_none()
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
|
||||
@@ -0,0 +1,467 @@
|
||||
//! Keep glibc from silently converting large short-lived buffers into heap the
|
||||
//! process never gives back.
|
||||
//!
|
||||
//! glibc serves an allocation with `mmap` when it is at least
|
||||
//! `M_MMAP_THRESHOLD` (128 KiB by default), and `munmap`s it on free, so the
|
||||
//! pages go straight back to the OS. That threshold is **adaptive**: whenever a
|
||||
//! block that came from `mmap` is freed, glibc raises the threshold to that
|
||||
//! block's size — up to 32 MiB — on the theory that a workload repeatedly
|
||||
//! allocating buffers of that size is better served from the heap.
|
||||
//!
|
||||
//! For a volume server that theory is wrong in a specific, expensive way. EC
|
||||
//! reconstruction and needle reassembly allocate large, short-lived buffers.
|
||||
//! The first few are mmap'd and freed, which trains the threshold upward; every
|
||||
//! later buffer of that size is then carved out of the heap instead. Heap
|
||||
//! memory is only returned to the OS from the top of the arena, so those pages
|
||||
//! stay resident as anonymous memory for the life of the process. They are
|
||||
//! still *reusable* — this is not a leak, and a repeat workload does not grow
|
||||
//! the footprint further — but under a hard cgroup `MemoryMax` they are
|
||||
//! indistinguishable from a leak, because anonymous pages cannot be reclaimed
|
||||
//! under pressure the way page cache can. The retained footprint eats exactly
|
||||
//! the headroom that a burst of maintenance work needs, and the process is
|
||||
//! OOM-killed while most of its resident memory is free-but-unreturned.
|
||||
//!
|
||||
//! Measured on a 17-node cluster (EC 10+4, `--index=redb`), one node, two
|
||||
//! identical `ec.scrub -mode full` rounds over 10912 EC files each, comparing
|
||||
//! the same unit restarted with and without a pinned threshold:
|
||||
//!
|
||||
//! | | baseline | round 1 | round 2 | 60s idle |
|
||||
//! |---|---|---|---|---|
|
||||
//! | default (adaptive) | 10 MB | 84 MB | 88 MB | **88 MB** |
|
||||
//! | pinned threshold | 10 MB | 13 MB | 14 MB | **14 MB** |
|
||||
//!
|
||||
//! 78 MB retained versus 4 MB for identical work. On that cluster's heavier
|
||||
//! mixed scrub workloads the same effect reached ~600 MB of retained anonymous
|
||||
//! memory per volume server, against a 3 GiB cap.
|
||||
//!
|
||||
//! Calling `mallopt(M_MMAP_THRESHOLD, ...)` sets the threshold *and* disables
|
||||
//! the dynamic adjustment, which is the documented behaviour of setting it
|
||||
//! explicitly. We pin it to glibc's own default rather than inventing a value:
|
||||
//! the goal is to stop the adaptation, not to second-guess the default.
|
||||
|
||||
/// glibc's own default `M_MMAP_THRESHOLD`. Pinning to this value changes
|
||||
/// nothing about which allocations use `mmap` on a freshly started process; it
|
||||
/// only prevents the threshold from drifting upward later.
|
||||
#[cfg(all(target_os = "linux", target_env = "gnu"))]
|
||||
const DEFAULT_MMAP_THRESHOLD: libc::c_int = 128 * 1024;
|
||||
|
||||
/// Legacy environment variable glibc reads for the same setting. If an operator
|
||||
/// has set it, honour their value and do not override it.
|
||||
pub const MMAP_THRESHOLD_ENV: &str = "MALLOC_MMAP_THRESHOLD_";
|
||||
|
||||
/// Modern glibc tunables environment variable. Operators may set the threshold
|
||||
/// via `GLIBC_TUNABLES=glibc.malloc.mmap_threshold=...` instead of the legacy
|
||||
/// variable; that override is honoured too.
|
||||
pub const GLIBC_TUNABLES_ENV: &str = "GLIBC_TUNABLES";
|
||||
|
||||
/// The tunable name within `GLIBC_TUNABLES` that maps to `M_MMAP_THRESHOLD`.
|
||||
#[cfg(all(target_os = "linux", target_env = "gnu"))]
|
||||
const MMAP_THRESHOLD_TUNABLE: &str = "glibc.malloc.mmap_threshold";
|
||||
|
||||
/// Outcome of the tuning attempt, so the caller can log it and tests can assert
|
||||
/// on it without inspecting global allocator state.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum MallocTuning {
|
||||
/// Threshold pinned to `DEFAULT_MMAP_THRESHOLD`; dynamic adjustment is off.
|
||||
Pinned(i32),
|
||||
/// An allocator override (`MALLOC_MMAP_THRESHOLD_` or
|
||||
/// `GLIBC_TUNABLES=glibc.malloc.mmap_threshold=...`) was set, so the
|
||||
/// operator's value wins.
|
||||
DeferredToEnv,
|
||||
/// `mallopt` reported failure. Not fatal — the server runs, it just keeps
|
||||
/// glibc's adaptive behaviour.
|
||||
Failed,
|
||||
/// Not glibc, so there is no adaptive threshold to pin.
|
||||
NotApplicable,
|
||||
}
|
||||
|
||||
/// Pin glibc's mmap threshold unless the operator has set an allocator override.
|
||||
/// Safe to call more than once; call it before serving traffic, since the point
|
||||
/// is to prevent the threshold from being trained upward by early allocations.
|
||||
pub fn pin_mmap_threshold() -> MallocTuning {
|
||||
pin_mmap_threshold_inner()
|
||||
}
|
||||
|
||||
#[cfg(all(target_os = "linux", target_env = "gnu"))]
|
||||
fn pin_mmap_threshold_inner() -> MallocTuning {
|
||||
if operator_mmap_threshold_override_active() {
|
||||
return MallocTuning::DeferredToEnv;
|
||||
}
|
||||
// SAFETY: `mallopt` is a libc entry point that takes two ints and mutates
|
||||
// only allocator-internal tunables. It has no preconditions and no effect
|
||||
// on memory this process already owns.
|
||||
let rc = unsafe { libc::mallopt(libc::M_MMAP_THRESHOLD, DEFAULT_MMAP_THRESHOLD) };
|
||||
if rc == 1 {
|
||||
MallocTuning::Pinned(DEFAULT_MMAP_THRESHOLD)
|
||||
} else {
|
||||
MallocTuning::Failed
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(not(all(target_os = "linux", target_env = "gnu")))]
|
||||
fn pin_mmap_threshold_inner() -> MallocTuning {
|
||||
MallocTuning::NotApplicable
|
||||
}
|
||||
|
||||
#[cfg(all(target_os = "linux", target_env = "gnu"))]
|
||||
fn operator_mmap_threshold_override_active() -> bool {
|
||||
// MALLOC_MMAP_THRESHOLD_: glibc calls atoi(value) then mallopt, which
|
||||
// always sets the threshold and disables dynamic adjustment — even for
|
||||
// empty, negative, or non-numeric values (atoi returns 0). So any presence
|
||||
// of the variable means the operator's override is in effect.
|
||||
std::env::var_os(MMAP_THRESHOLD_ENV).is_some()
|
||||
|| usable_glibc_tunable_threshold(std::env::var_os(GLIBC_TUNABLES_ENV))
|
||||
}
|
||||
|
||||
/// Look for `glibc.malloc.mmap_threshold=<value>` among the colon-separated
|
||||
/// tunables in `GLIBC_TUNABLES`. glibc's `parse_tunables_string` (elf/dl-tunables.c)
|
||||
/// rejects the **entire** string (returns -1) if it reaches `\0` before finding
|
||||
/// `=` in a name (last entry has no `=`), or if any entry's value contains a
|
||||
/// duplicate `=`. When `parse_tunables_string` returns -1, `parse_tunables`
|
||||
/// prints a warning and returns immediately without applying ANY tunable —
|
||||
/// including ones already parsed into the tunables array. We match that by
|
||||
/// returning `false` for the entire string on any of those conditions.
|
||||
///
|
||||
/// glibc parses tunable values with `_dl_strtoul`, which accepts decimal,
|
||||
/// `0x` hex, `0` octal, an optional sign (negatives wrap to `unsigned long`),
|
||||
/// and requires the entire value to be consumed; we match that with
|
||||
/// `dl_strtoul_consumes_all`.
|
||||
#[cfg(all(target_os = "linux", target_env = "gnu"))]
|
||||
fn usable_glibc_tunable_threshold(tunables: Option<std::ffi::OsString>) -> bool {
|
||||
let s = match tunables.and_then(|v| v.into_string().ok()) {
|
||||
Some(s) => s,
|
||||
None => return false,
|
||||
};
|
||||
if s.is_empty() {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Parse the string character-by-character, matching glibc's
|
||||
// parse_tunables_string logic exactly. Using split(':') would lose the
|
||||
// distinction between an entry terminated by ':' (skip) and one terminated
|
||||
// by '\0' with no '=' (reject entire string).
|
||||
let bytes = s.as_bytes();
|
||||
let mut pos = 0;
|
||||
let mut found_threshold = false;
|
||||
|
||||
loop {
|
||||
// Find where the name ends ('=', ':', or end of string).
|
||||
let name_start = pos;
|
||||
while pos < bytes.len() && bytes[pos] != b'=' && bytes[pos] != b':' {
|
||||
pos += 1;
|
||||
}
|
||||
|
||||
// End of string before '=' → glibc returns -1 (reject entire string).
|
||||
if pos >= bytes.len() {
|
||||
return false;
|
||||
}
|
||||
|
||||
// ':' before '=' → glibc skips this entry and continues.
|
||||
if bytes[pos] == b':' {
|
||||
pos += 1;
|
||||
continue;
|
||||
}
|
||||
|
||||
// Skip the '='.
|
||||
let name_end = pos;
|
||||
pos += 1;
|
||||
|
||||
// Find where the value ends ('=', ':', or end of string).
|
||||
let val_start = pos;
|
||||
while pos < bytes.len() && bytes[pos] != b'=' && bytes[pos] != b':' {
|
||||
pos += 1;
|
||||
}
|
||||
|
||||
// '=' in value → glibc returns -1 (reject entire string).
|
||||
if pos < bytes.len() && bytes[pos] == b'=' {
|
||||
return false;
|
||||
}
|
||||
|
||||
let key = &s[name_start..name_end];
|
||||
let val = &s[val_start..pos];
|
||||
|
||||
if key == MMAP_THRESHOLD_TUNABLE && dl_strtoul_consumes_all(val) {
|
||||
found_threshold = true;
|
||||
}
|
||||
|
||||
// End of string → done.
|
||||
if pos >= bytes.len() {
|
||||
break;
|
||||
}
|
||||
|
||||
// Skip the ':'.
|
||||
pos += 1;
|
||||
}
|
||||
|
||||
found_threshold
|
||||
}
|
||||
|
||||
/// Replicate glibc's `_dl_strtoul` (elf/dl-misc.c) just enough to determine
|
||||
/// whether it would consume the entire string — which is what
|
||||
/// `tunable_parse_num` checks (`endptr == strval + len`). Returns `true` if
|
||||
/// glibc would accept the value and apply it.
|
||||
///
|
||||
/// `_dl_strtoul` skips leading spaces/tabs, accepts an optional `+`/`-` sign,
|
||||
/// and parses `0x`-prefixed hex, `0`-prefixed octal, or plain decimal. A
|
||||
/// negative result wraps to `unsigned long` (`-1` → `SIZE_MAX`). If no digit is
|
||||
/// found after the sign, the end pointer stays at the current position — which
|
||||
/// still counts as "consumed" when the string is empty or whitespace-only
|
||||
/// (value 0). On overflow, `_dl_strtoul` stops at the overflowing digit (endptr
|
||||
/// does not reach the end), so `tunable_parse_num` rejects the value.
|
||||
#[cfg(all(target_os = "linux", target_env = "gnu"))]
|
||||
fn dl_strtoul_consumes_all(s: &str) -> bool {
|
||||
let bytes = s.as_bytes();
|
||||
let mut pos = 0;
|
||||
|
||||
// Skip leading whitespace (spaces and tabs, matching _dl_strtoul).
|
||||
while pos < bytes.len() && (bytes[pos] == b' ' || bytes[pos] == b'\t') {
|
||||
pos += 1;
|
||||
}
|
||||
|
||||
// Optional sign.
|
||||
if pos < bytes.len() && (bytes[pos] == b'-' || bytes[pos] == b'+') {
|
||||
pos += 1;
|
||||
}
|
||||
|
||||
// Must have at least one digit (0-9) to start parsing, unless we're already
|
||||
// at the end (empty / whitespace-only / sign-only → value 0, consumed).
|
||||
if pos >= bytes.len() {
|
||||
return true;
|
||||
}
|
||||
if bytes[pos] < b'0' || bytes[pos] > b'9' {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Determine base: 0x → hex, 0 → octal, else decimal. _dl_strtoul unconditionally
|
||||
// advances past "0x"/"0X" when the first char is '0' and the next is 'x'/'X',
|
||||
// even if no hex digit follows — in that case the digit loop breaks immediately,
|
||||
// endptr reaches the end, and the value is 0.
|
||||
let base: u32 = if bytes[pos] == b'0'
|
||||
&& pos + 1 < bytes.len()
|
||||
&& (bytes[pos + 1] == b'x' || bytes[pos + 1] == b'X')
|
||||
{
|
||||
pos += 2; // skip "0x"
|
||||
16
|
||||
} else if bytes[pos] == b'0' {
|
||||
8
|
||||
} else {
|
||||
10
|
||||
};
|
||||
|
||||
// Parse digits with overflow detection, matching _dl_strtoul's cutoff/cutlim
|
||||
// logic. On overflow, _dl_strtoul sets endptr to the overflowing digit and
|
||||
// returns UINT64_MAX — so the value is NOT fully consumed and
|
||||
// tunable_parse_num rejects it.
|
||||
let mut result: u64 = 0;
|
||||
let cutoff = u64::MAX / base as u64;
|
||||
let cutlim = u64::MAX % base as u64;
|
||||
|
||||
while pos < bytes.len() {
|
||||
let b = bytes[pos];
|
||||
let digval: u32 = match digit_value(b, base) {
|
||||
Some(v) => v,
|
||||
None => break,
|
||||
};
|
||||
if result > cutoff || (result == cutoff && digval as u64 > cutlim) {
|
||||
// Overflow: _dl_strtoul stops here, endptr points at this digit.
|
||||
return false;
|
||||
}
|
||||
result *= base as u64;
|
||||
result += digval as u64;
|
||||
pos += 1;
|
||||
}
|
||||
|
||||
// The entire string must be consumed (matching tunable_parse_num's check).
|
||||
pos == bytes.len()
|
||||
}
|
||||
|
||||
/// Returns the numeric value of a digit byte in the given base, or `None` if
|
||||
/// the byte is not a valid digit in that base.
|
||||
#[cfg(all(target_os = "linux", target_env = "gnu"))]
|
||||
fn digit_value(b: u8, base: u32) -> Option<u32> {
|
||||
if (b'0'..=b'0' + (base - 1).min(9) as u8).contains(&b) {
|
||||
return Some((b - b'0') as u32);
|
||||
}
|
||||
if base == 16 {
|
||||
if (b'a'..=b'f').contains(&b) {
|
||||
return Some((b - b'a' + 10) as u32);
|
||||
}
|
||||
if (b'A'..=b'F').contains(&b) {
|
||||
return Some((b - b'A' + 10) as u32);
|
||||
}
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn env_override_constants_match_glibc_names() {
|
||||
// Verified against the real accessor rather than a copy of the name, so
|
||||
// renaming the constant cannot silently break the override contract.
|
||||
assert_eq!(MMAP_THRESHOLD_ENV, "MALLOC_MMAP_THRESHOLD_");
|
||||
assert_eq!(GLIBC_TUNABLES_ENV, "GLIBC_TUNABLES");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn calling_twice_is_stable() {
|
||||
// Startup paths get re-entered in tests and in `weed mini`; the second
|
||||
// call must not report a different outcome from the first.
|
||||
let first = pin_mmap_threshold();
|
||||
let second = pin_mmap_threshold();
|
||||
assert_eq!(first, second);
|
||||
}
|
||||
|
||||
#[cfg(all(target_os = "linux", target_env = "gnu"))]
|
||||
#[test]
|
||||
fn pins_threshold_on_glibc_when_no_override_is_set() {
|
||||
// The env override is not set in the test process, so this exercises the
|
||||
// mallopt path. If an override happens to be present, defer to it.
|
||||
if operator_mmap_threshold_override_active() {
|
||||
assert_eq!(pin_mmap_threshold(), MallocTuning::DeferredToEnv);
|
||||
return;
|
||||
}
|
||||
assert_eq!(
|
||||
pin_mmap_threshold(),
|
||||
MallocTuning::Pinned(DEFAULT_MMAP_THRESHOLD),
|
||||
"mallopt(M_MMAP_THRESHOLD) should succeed on glibc"
|
||||
);
|
||||
}
|
||||
|
||||
#[cfg(not(all(target_os = "linux", target_env = "gnu")))]
|
||||
#[test]
|
||||
fn is_a_noop_off_glibc() {
|
||||
// No glibc adaptive threshold exists off glibc, so there is nothing to
|
||||
// pin regardless of any environment variables that happen to be set.
|
||||
assert_eq!(pin_mmap_threshold(), MallocTuning::NotApplicable);
|
||||
}
|
||||
|
||||
#[cfg(all(target_os = "linux", target_env = "gnu"))]
|
||||
#[test]
|
||||
fn dl_strtoul_consumes_all_matches_glibc_parser() {
|
||||
// Decimal — any non-empty decimal integer is accepted, including
|
||||
// negative (wraps to unsigned) and zero.
|
||||
assert!(dl_strtoul_consumes_all("131072"));
|
||||
assert!(dl_strtoul_consumes_all("0"));
|
||||
assert!(dl_strtoul_consumes_all("-1"));
|
||||
assert!(dl_strtoul_consumes_all("-131072"));
|
||||
// Values above i64::MAX are valid for glibc's unsigned parser.
|
||||
assert!(dl_strtoul_consumes_all("9223372036854775808"));
|
||||
// Hex with 0x prefix.
|
||||
assert!(dl_strtoul_consumes_all("0x20000"));
|
||||
assert!(dl_strtoul_consumes_all("0X20000"));
|
||||
assert!(dl_strtoul_consumes_all("0x0"));
|
||||
// Octal with leading 0.
|
||||
assert!(dl_strtoul_consumes_all("010"));
|
||||
// Leading whitespace (spaces and tabs) is skipped.
|
||||
assert!(dl_strtoul_consumes_all(" 131072"));
|
||||
assert!(dl_strtoul_consumes_all("\t0x20000"));
|
||||
// Empty and whitespace-only strings are accepted (value 0).
|
||||
assert!(dl_strtoul_consumes_all(""));
|
||||
assert!(dl_strtoul_consumes_all(" "));
|
||||
assert!(dl_strtoul_consumes_all("\t"));
|
||||
// Sign-only strings are accepted: _dl_strtoul skips the sign, finds no
|
||||
// digit, sets endptr to the position after the sign (== end of string),
|
||||
// and returns 0. tunable_parse_num sees endptr == strval + len → true.
|
||||
assert!(dl_strtoul_consumes_all("-"));
|
||||
assert!(dl_strtoul_consumes_all("+"));
|
||||
|
||||
// Trailing garbage is rejected — _dl_strtoul stops at the first
|
||||
// non-digit and tunable_parse_num requires the entire string consumed.
|
||||
assert!(!dl_strtoul_consumes_all("131072abc"));
|
||||
// In hex mode, a-f are digits, so "0x20000abc" is a valid hex number.
|
||||
// Use a non-hex character like 'g' to test trailing garbage in hex.
|
||||
assert!(!dl_strtoul_consumes_all("0x20000g"));
|
||||
assert!(!dl_strtoul_consumes_all("128K"));
|
||||
// Non-numeric strings are rejected.
|
||||
assert!(!dl_strtoul_consumes_all("abc"));
|
||||
// "0x" with no hex digits: _dl_strtoul advances past "0x", the digit loop
|
||||
// breaks immediately (no hex digit), endptr reaches the end, value is 0.
|
||||
// tunable_parse_num accepts it.
|
||||
assert!(dl_strtoul_consumes_all("0x"));
|
||||
assert!(dl_strtoul_consumes_all("0X"));
|
||||
|
||||
// Overflow: _dl_strtoul stops at the overflowing digit (endptr points
|
||||
// there, not at the end), so tunable_parse_num rejects the value.
|
||||
assert!(!dl_strtoul_consumes_all("18446744073709551616")); // u64::MAX + 1
|
||||
assert!(!dl_strtoul_consumes_all("99999999999999999999")); // 20 nines
|
||||
assert!(!dl_strtoul_consumes_all("0x10000000000000000")); // 2^64
|
||||
// u64::MAX itself is accepted: the last digit (5) equals cutlim (=5),
|
||||
// so the overflow check (digval > cutlim) is false.
|
||||
assert!(dl_strtoul_consumes_all("18446744073709551615")); // u64::MAX
|
||||
}
|
||||
|
||||
#[cfg(all(target_os = "linux", target_env = "gnu"))]
|
||||
#[test]
|
||||
fn usable_glibc_tunable_threshold_detects_mmap_threshold() {
|
||||
// Decimal, hex, octal, negative, and zero values are all accepted by
|
||||
// glibc's _dl_strtoul and cause the threshold to be pinned.
|
||||
assert!(usable_glibc_tunable_threshold(Some(
|
||||
"glibc.malloc.mmap_threshold=131072".into()
|
||||
)));
|
||||
assert!(usable_glibc_tunable_threshold(Some(
|
||||
"glibc.malloc.mmap_threshold=0x20000".into()
|
||||
)));
|
||||
assert!(usable_glibc_tunable_threshold(Some(
|
||||
"glibc.malloc.mmap_threshold=0".into()
|
||||
)));
|
||||
assert!(usable_glibc_tunable_threshold(Some(
|
||||
"glibc.malloc.mmap_threshold=-1".into()
|
||||
)));
|
||||
assert!(usable_glibc_tunable_threshold(Some(
|
||||
"glibc.malloc.mmap_threshold=9223372036854775808".into()
|
||||
)));
|
||||
// u64::MAX is accepted by _dl_strtoul (last digit == cutlim, no overflow).
|
||||
assert!(usable_glibc_tunable_threshold(Some(
|
||||
"glibc.malloc.mmap_threshold=18446744073709551615".into()
|
||||
)));
|
||||
// Appears alongside other tunables.
|
||||
assert!(usable_glibc_tunable_threshold(Some(
|
||||
"glibc.cpu.x=1:glibc.malloc.mmap_threshold=131072".into()
|
||||
)));
|
||||
// Leading ':' is accepted — glibc skips the empty entry and continues.
|
||||
assert!(usable_glibc_tunable_threshold(Some(
|
||||
":glibc.malloc.mmap_threshold=131072".into()
|
||||
)));
|
||||
// Empty value is accepted by _dl_strtoul (value 0).
|
||||
assert!(usable_glibc_tunable_threshold(Some(
|
||||
"glibc.malloc.mmap_threshold=".into()
|
||||
)));
|
||||
|
||||
// Non-numeric values are rejected by _dl_strtoul.
|
||||
assert!(!usable_glibc_tunable_threshold(Some(
|
||||
"glibc.malloc.mmap_threshold=abc".into()
|
||||
)));
|
||||
assert!(!usable_glibc_tunable_threshold(Some(
|
||||
"glibc.malloc.mmap_threshold=128K".into()
|
||||
)));
|
||||
// A malformed sibling entry (duplicate '=') makes glibc reject the
|
||||
// entire string, so we must not accept the threshold entry either.
|
||||
// This applies regardless of whether the threshold is before or after
|
||||
// the malformed entry — parse_tunables_string returns -1, and
|
||||
// parse_tunables discards all tunables without applying any.
|
||||
assert!(!usable_glibc_tunable_threshold(Some(
|
||||
"glibc.malloc.check=2=2:glibc.malloc.mmap_threshold=131072".into()
|
||||
)));
|
||||
assert!(!usable_glibc_tunable_threshold(Some(
|
||||
"glibc.malloc.mmap_threshold=262144:glibc.malloc.check=2=2".into()
|
||||
)));
|
||||
// A trailing entry with no '=' makes glibc reject the entire string
|
||||
// (parse_tunables_string hits '\0' before '=' and returns -1).
|
||||
assert!(!usable_glibc_tunable_threshold(Some(
|
||||
"glibc.malloc.mmap_threshold=262144:glibc.cpu.x".into()
|
||||
)));
|
||||
// A trailing ':' makes glibc reject the entire string (the empty entry
|
||||
// after ':' hits '\0' before '=' and returns -1).
|
||||
assert!(!usable_glibc_tunable_threshold(Some(
|
||||
"glibc.malloc.mmap_threshold=262144:".into()
|
||||
)));
|
||||
// Unrelated tunables do not count.
|
||||
assert!(!usable_glibc_tunable_threshold(Some(
|
||||
"glibc.cpu.x=1".into()
|
||||
)));
|
||||
assert!(!usable_glibc_tunable_threshold(None));
|
||||
}
|
||||
}
|
||||
+238
-118
@@ -3,10 +3,10 @@
|
||||
//! Mirrors the Go SeaweedFS volume server metrics.
|
||||
|
||||
use prometheus::{
|
||||
self, Encoder, GaugeVec, HistogramOpts, HistogramVec, IntCounterVec, IntGauge, IntGaugeVec,
|
||||
Opts, Registry, TextEncoder,
|
||||
self, Encoder, GaugeVec, HistogramOpts, HistogramVec, IntCounter, IntCounterVec, IntGauge,
|
||||
IntGaugeVec, Opts, Registry, TextEncoder,
|
||||
};
|
||||
use std::sync::Once;
|
||||
use std::sync::{LazyLock, Once};
|
||||
|
||||
use crate::version;
|
||||
|
||||
@@ -16,203 +16,320 @@ pub struct PushGatewayConfig {
|
||||
pub interval_seconds: u32,
|
||||
}
|
||||
|
||||
lazy_static::lazy_static! {
|
||||
pub static ref REGISTRY: Registry = Registry::new();
|
||||
pub static REGISTRY: LazyLock<Registry> = LazyLock::new(Registry::new);
|
||||
|
||||
// ---- Request metrics (Go: VolumeServerRequestCounter, VolumeServerRequestHistogram) ----
|
||||
// ---- Request metrics (Go: VolumeServerRequestCounter, VolumeServerRequestHistogram) ----
|
||||
|
||||
/// Request counter with labels `type` (HTTP method) and `code` (HTTP status).
|
||||
pub static ref REQUEST_COUNTER: IntCounterVec = IntCounterVec::new(
|
||||
Opts::new("SeaweedFS_volumeServer_request_total", "Volume server requests"),
|
||||
/// Request counter with labels `type` (HTTP method) and `code` (HTTP status).
|
||||
pub static REQUEST_COUNTER: LazyLock<IntCounterVec> = LazyLock::new(|| {
|
||||
IntCounterVec::new(
|
||||
Opts::new(
|
||||
"SeaweedFS_volumeServer_request_total",
|
||||
"Volume server requests",
|
||||
),
|
||||
&["type", "code"],
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Request duration histogram with label `type` (HTTP method).
|
||||
pub static ref REQUEST_DURATION: HistogramVec = HistogramVec::new(
|
||||
/// Request duration histogram with label `type` (HTTP method).
|
||||
pub static REQUEST_DURATION: LazyLock<HistogramVec> = LazyLock::new(|| {
|
||||
HistogramVec::new(
|
||||
HistogramOpts::new(
|
||||
"SeaweedFS_volumeServer_request_seconds",
|
||||
"Volume server request duration in seconds",
|
||||
).buckets(exponential_buckets(0.0001, 2.0, 24)),
|
||||
)
|
||||
.buckets(exponential_buckets(0.0001, 2.0, 24)),
|
||||
&["type"],
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
// ---- Handler counters (Go: VolumeServerHandlerCounter) ----
|
||||
// ---- Handler counters (Go: VolumeServerHandlerCounter) ----
|
||||
|
||||
/// Handler-level operation counter with label `type`.
|
||||
pub static ref HANDLER_COUNTER: IntCounterVec = IntCounterVec::new(
|
||||
Opts::new("SeaweedFS_volumeServer_handler_total", "Volume server handler counters"),
|
||||
/// Handler-level operation counter with label `type`.
|
||||
pub static HANDLER_COUNTER: LazyLock<IntCounterVec> = LazyLock::new(|| {
|
||||
IntCounterVec::new(
|
||||
Opts::new(
|
||||
"SeaweedFS_volumeServer_handler_total",
|
||||
"Volume server handler counters",
|
||||
),
|
||||
&["type"],
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
// ---- Vacuuming metrics (Go: VolumeServerVacuuming*) ----
|
||||
// ---- Vacuuming metrics (Go: VolumeServerVacuuming*) ----
|
||||
|
||||
/// Vacuuming compact counter with label `success` (true/false).
|
||||
pub static ref VACUUMING_COMPACT_COUNTER: IntCounterVec = IntCounterVec::new(
|
||||
Opts::new("SeaweedFS_volumeServer_vacuuming_compact_count", "Counter of volume vacuuming Compact counter"),
|
||||
/// Vacuuming compact counter with label `success` (true/false).
|
||||
pub static VACUUMING_COMPACT_COUNTER: LazyLock<IntCounterVec> = LazyLock::new(|| {
|
||||
IntCounterVec::new(
|
||||
Opts::new(
|
||||
"SeaweedFS_volumeServer_vacuuming_compact_count",
|
||||
"Counter of volume vacuuming Compact counter",
|
||||
),
|
||||
&["success"],
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Vacuuming commit counter with label `success` (true/false).
|
||||
pub static ref VACUUMING_COMMIT_COUNTER: IntCounterVec = IntCounterVec::new(
|
||||
Opts::new("SeaweedFS_volumeServer_vacuuming_commit_count", "Counter of volume vacuuming commit counter"),
|
||||
/// Vacuuming commit counter with label `success` (true/false).
|
||||
pub static VACUUMING_COMMIT_COUNTER: LazyLock<IntCounterVec> = LazyLock::new(|| {
|
||||
IntCounterVec::new(
|
||||
Opts::new(
|
||||
"SeaweedFS_volumeServer_vacuuming_commit_count",
|
||||
"Counter of volume vacuuming commit counter",
|
||||
),
|
||||
&["success"],
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Vacuuming duration histogram with label `type` (compact/commit).
|
||||
pub static ref VACUUMING_HISTOGRAM: HistogramVec = HistogramVec::new(
|
||||
/// Vacuuming duration histogram with label `type` (compact/commit).
|
||||
pub static VACUUMING_HISTOGRAM: LazyLock<HistogramVec> = LazyLock::new(|| {
|
||||
HistogramVec::new(
|
||||
HistogramOpts::new(
|
||||
"SeaweedFS_volumeServer_vacuuming_seconds",
|
||||
"Volume vacuuming duration in seconds",
|
||||
).buckets(exponential_buckets(0.0001, 2.0, 24)),
|
||||
)
|
||||
.buckets(exponential_buckets(0.0001, 2.0, 24)),
|
||||
&["type"],
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
// ---- Volume gauges (Go: VolumeServerVolumeGauge, VolumeServerReadOnlyVolumeGauge) ----
|
||||
// ---- Volume gauges (Go: VolumeServerVolumeGauge, VolumeServerReadOnlyVolumeGauge) ----
|
||||
|
||||
/// Volumes per collection and type (volume/ec_shards).
|
||||
pub static ref VOLUME_GAUGE: GaugeVec = GaugeVec::new(
|
||||
/// Volumes per collection and type (volume/ec_shards).
|
||||
pub static VOLUME_GAUGE: LazyLock<GaugeVec> = LazyLock::new(|| {
|
||||
GaugeVec::new(
|
||||
Opts::new("SeaweedFS_volumeServer_volumes", "Number of volumes"),
|
||||
&["collection", "type"],
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Read-only volumes per collection and type.
|
||||
pub static ref READ_ONLY_VOLUME_GAUGE: GaugeVec = GaugeVec::new(
|
||||
Opts::new("SeaweedFS_volumeServer_read_only_volumes", "Number of read-only volumes."),
|
||||
/// Read-only volumes per collection and type.
|
||||
pub static READ_ONLY_VOLUME_GAUGE: LazyLock<GaugeVec> = LazyLock::new(|| {
|
||||
GaugeVec::new(
|
||||
Opts::new(
|
||||
"SeaweedFS_volumeServer_read_only_volumes",
|
||||
"Number of read-only volumes.",
|
||||
),
|
||||
&["collection", "type"],
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Maximum number of volumes this server can hold.
|
||||
pub static ref MAX_VOLUMES: IntGauge = IntGauge::new(
|
||||
/// Maximum number of volumes this server can hold.
|
||||
pub static MAX_VOLUMES: LazyLock<IntGauge> = LazyLock::new(|| {
|
||||
IntGauge::new(
|
||||
"SeaweedFS_volumeServer_max_volumes",
|
||||
"Maximum number of volumes",
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
// ---- Disk size gauges (Go: VolumeServerDiskSizeGauge) ----
|
||||
// ---- Disk size gauges (Go: VolumeServerDiskSizeGauge) ----
|
||||
|
||||
/// Actual disk size used by volumes per collection and type (normal/deleted_bytes/ec).
|
||||
pub static ref DISK_SIZE_GAUGE: GaugeVec = GaugeVec::new(
|
||||
Opts::new("SeaweedFS_volumeServer_total_disk_size", "Actual disk size used by volumes"),
|
||||
/// Actual disk size used by volumes per collection and type (normal/deleted_bytes/ec).
|
||||
pub static DISK_SIZE_GAUGE: LazyLock<GaugeVec> = LazyLock::new(|| {
|
||||
GaugeVec::new(
|
||||
Opts::new(
|
||||
"SeaweedFS_volumeServer_total_disk_size",
|
||||
"Actual disk size used by volumes",
|
||||
),
|
||||
&["collection", "type"],
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
// ---- Resource gauges (Go: VolumeServerResourceGauge) ----
|
||||
// ---- Resource gauges (Go: VolumeServerResourceGauge) ----
|
||||
|
||||
/// Disk resource usage per directory and type (all/used/free/avail).
|
||||
pub static ref RESOURCE_GAUGE: GaugeVec = GaugeVec::new(
|
||||
/// Disk resource usage per directory and type (all/used/free/avail).
|
||||
pub static RESOURCE_GAUGE: LazyLock<GaugeVec> = LazyLock::new(|| {
|
||||
GaugeVec::new(
|
||||
Opts::new("SeaweedFS_volumeServer_resource", "Server resource usage"),
|
||||
&["name", "type"],
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
// ---- In-flight gauges (Go: VolumeServerInFlightRequestsGauge, InFlightDownload/UploadSize) ----
|
||||
// ---- In-flight gauges (Go: VolumeServerInFlightRequestsGauge, InFlightDownload/UploadSize) ----
|
||||
|
||||
/// In-flight requests per HTTP method.
|
||||
pub static ref INFLIGHT_REQUESTS_GAUGE: IntGaugeVec = IntGaugeVec::new(
|
||||
Opts::new("SeaweedFS_volumeServer_in_flight_requests", "Current number of in-flight requests being handled by volume server."),
|
||||
/// In-flight requests per HTTP method.
|
||||
pub static INFLIGHT_REQUESTS_GAUGE: LazyLock<IntGaugeVec> = LazyLock::new(|| {
|
||||
IntGaugeVec::new(
|
||||
Opts::new(
|
||||
"SeaweedFS_volumeServer_in_flight_requests",
|
||||
"Current number of in-flight requests being handled by volume server.",
|
||||
),
|
||||
&["type"],
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Concurrent download limit in bytes.
|
||||
pub static ref CONCURRENT_DOWNLOAD_LIMIT: IntGauge = IntGauge::new(
|
||||
/// Concurrent download limit in bytes.
|
||||
pub static CONCURRENT_DOWNLOAD_LIMIT: LazyLock<IntGauge> = LazyLock::new(|| {
|
||||
IntGauge::new(
|
||||
"SeaweedFS_volumeServer_concurrent_download_limit",
|
||||
"Limit for total concurrent download size in bytes",
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Concurrent upload limit in bytes.
|
||||
pub static ref CONCURRENT_UPLOAD_LIMIT: IntGauge = IntGauge::new(
|
||||
/// Concurrent upload limit in bytes.
|
||||
pub static CONCURRENT_UPLOAD_LIMIT: LazyLock<IntGauge> = LazyLock::new(|| {
|
||||
IntGauge::new(
|
||||
"SeaweedFS_volumeServer_concurrent_upload_limit",
|
||||
"Limit for total concurrent upload size in bytes",
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Current in-flight download bytes.
|
||||
pub static ref INFLIGHT_DOWNLOAD_SIZE: IntGauge = IntGauge::new(
|
||||
/// Current in-flight download bytes.
|
||||
pub static INFLIGHT_DOWNLOAD_SIZE: LazyLock<IntGauge> = LazyLock::new(|| {
|
||||
IntGauge::new(
|
||||
"SeaweedFS_volumeServer_in_flight_download_size",
|
||||
"In flight total download size.",
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Current in-flight upload bytes.
|
||||
pub static ref INFLIGHT_UPLOAD_SIZE: IntGauge = IntGauge::new(
|
||||
/// Current in-flight upload bytes.
|
||||
pub static INFLIGHT_UPLOAD_SIZE: LazyLock<IntGauge> = LazyLock::new(|| {
|
||||
IntGauge::new(
|
||||
"SeaweedFS_volumeServer_in_flight_upload_size",
|
||||
"In flight total upload size.",
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Upload error counter by HTTP status code. Code "0" = transport error (no response).
|
||||
pub static ref UPLOAD_ERROR_COUNTER: IntCounterVec = IntCounterVec::new(
|
||||
Opts::new("SeaweedFS_upload_error_total",
|
||||
"Counter of upload errors by HTTP status code. Code 0 means transport error (no response received)."),
|
||||
&["code"],
|
||||
).expect("metric can be created");
|
||||
/// Upload error counter by HTTP status code. Code "0" = transport error (no response).
|
||||
pub static UPLOAD_ERROR_COUNTER: LazyLock<IntCounterVec> = LazyLock::new(|| {
|
||||
IntCounterVec::new(
|
||||
Opts::new("SeaweedFS_upload_error_total",
|
||||
"Counter of upload errors by HTTP status code. Code 0 means transport error (no response received)."),
|
||||
&["code"],
|
||||
).expect("metric can be created")
|
||||
});
|
||||
|
||||
// ---- Scrubbing metrics (Go: VolumeServerScrub*) ----
|
||||
// ---- Scrubbing metrics (Go: VolumeServerScrub*) ----
|
||||
|
||||
/// Last scrub execution time, as seconds since UNIX epoch, with label `mode`.
|
||||
pub static ref SCRUB_LAST_TIME_SECONDS: GaugeVec = GaugeVec::new(
|
||||
/// Last scrub execution time, as seconds since UNIX epoch, with label `mode`.
|
||||
pub static SCRUB_LAST_TIME_SECONDS: LazyLock<GaugeVec> = LazyLock::new(|| {
|
||||
GaugeVec::new(
|
||||
Opts::new(
|
||||
"SeaweedFS_volumeServer_scrub_last_time_seconds",
|
||||
"Last scrub execution time, as seconds since UNIX epoch.",
|
||||
),
|
||||
&["mode"],
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Counter of overall volumes with issues detected during scrubbing, with label `mode`.
|
||||
pub static ref SCRUB_VOLUME_FAILURES: IntCounterVec = IntCounterVec::new(
|
||||
/// Counter of overall volumes with issues detected during scrubbing, with label `mode`.
|
||||
pub static SCRUB_VOLUME_FAILURES: LazyLock<IntCounterVec> = LazyLock::new(|| {
|
||||
IntCounterVec::new(
|
||||
Opts::new(
|
||||
"SeaweedFS_volumeServer_scrub_volume_failures",
|
||||
"Counter of overall volumes with issues detected during scrubbing.",
|
||||
),
|
||||
&["mode"],
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Counter of overall EC shards with issues detected during scrubbing, with label `mode`.
|
||||
pub static ref SCRUB_SHARD_FAILURES: IntCounterVec = IntCounterVec::new(
|
||||
/// Counter of overall EC shards with issues detected during scrubbing, with label `mode`.
|
||||
pub static SCRUB_SHARD_FAILURES: LazyLock<IntCounterVec> = LazyLock::new(|| {
|
||||
IntCounterVec::new(
|
||||
Opts::new(
|
||||
"SeaweedFS_volumeServer_scrub_shard_failures",
|
||||
"Counter of overall EC shards with issues detected during scrubbing.",
|
||||
),
|
||||
&["mode"],
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
// ---- Legacy aliases for backward compat with existing code ----
|
||||
/// Counter of storage read/write EIO errors on volumes and EC shards.
|
||||
/// Mirrors Go's VolumeServerStorageIoErrorCounter.
|
||||
pub static STORAGE_IO_ERROR_COUNTER: LazyLock<IntCounter> = LazyLock::new(|| {
|
||||
IntCounter::new(
|
||||
"SeaweedFS_volumeServer_storage_io_error_total",
|
||||
"Counter of storage read/write EIO errors on volumes and EC shards.",
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Total number of volumes on this server (flat gauge).
|
||||
pub static ref VOLUMES_TOTAL: IntGauge = IntGauge::new(
|
||||
"volume_server_volumes_total",
|
||||
"Total number of volumes",
|
||||
).expect("metric can be created");
|
||||
/// Number of volumes quarantined due to storage IO errors.
|
||||
/// Mirrors Go's VolumeServerIoQuarantineGauge.
|
||||
pub static IO_QUARANTINE_GAUGE: LazyLock<IntGaugeVec> = LazyLock::new(|| {
|
||||
IntGaugeVec::new(
|
||||
Opts::new(
|
||||
"SeaweedFS_volumeServer_io_quarantine",
|
||||
"Number of volumes or EC shards quarantined due to storage IO errors.",
|
||||
),
|
||||
&["kind"],
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Disk size in bytes per directory.
|
||||
pub static ref DISK_SIZE_BYTES: IntGaugeVec = IntGaugeVec::new(
|
||||
// ---- Legacy aliases for backward compat with existing code ----
|
||||
|
||||
/// Total number of volumes on this server (flat gauge).
|
||||
pub static VOLUMES_TOTAL: LazyLock<IntGauge> = LazyLock::new(|| {
|
||||
IntGauge::new("volume_server_volumes_total", "Total number of volumes")
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Disk size in bytes per directory.
|
||||
pub static DISK_SIZE_BYTES: LazyLock<IntGaugeVec> = LazyLock::new(|| {
|
||||
IntGaugeVec::new(
|
||||
Opts::new("volume_server_disk_size_bytes", "Disk size in bytes"),
|
||||
&["dir"],
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Disk free bytes per directory.
|
||||
pub static ref DISK_FREE_BYTES: IntGaugeVec = IntGaugeVec::new(
|
||||
/// Disk free bytes per directory.
|
||||
pub static DISK_FREE_BYTES: LazyLock<IntGaugeVec> = LazyLock::new(|| {
|
||||
IntGaugeVec::new(
|
||||
Opts::new("volume_server_disk_free_bytes", "Disk free space in bytes"),
|
||||
&["dir"],
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Current number of in-flight requests (flat gauge).
|
||||
pub static ref INFLIGHT_REQUESTS: IntGauge = IntGauge::new(
|
||||
/// Current number of in-flight requests (flat gauge).
|
||||
pub static INFLIGHT_REQUESTS: LazyLock<IntGauge> = LazyLock::new(|| {
|
||||
IntGauge::new(
|
||||
"volume_server_inflight_requests",
|
||||
"Current number of in-flight requests",
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Total number of files stored across all volumes.
|
||||
pub static ref VOLUME_FILE_COUNT: IntGauge = IntGauge::new(
|
||||
/// Total number of files stored across all volumes.
|
||||
pub static VOLUME_FILE_COUNT: LazyLock<IntGauge> = LazyLock::new(|| {
|
||||
IntGauge::new(
|
||||
"volume_server_volume_file_count",
|
||||
"Total number of files stored across all volumes",
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
// ---- Build info (Go: BuildInfo) ----
|
||||
// ---- Build info (Go: BuildInfo) ----
|
||||
|
||||
/// Build information gauge, always set to 1. Matches Go:
|
||||
/// Namespace="SeaweedFS", Subsystem="build", Name="info",
|
||||
/// labels: version, commit, sizelimit, goos, goarch.
|
||||
pub static ref BUILD_INFO: GaugeVec = GaugeVec::new(
|
||||
Opts::new("SeaweedFS_build_info", "A metric with a constant '1' value labeled by version, commit, sizelimit, goos, and goarch from which SeaweedFS was built."),
|
||||
&["version", "commit", "sizelimit", "goos", "goarch"],
|
||||
).expect("metric can be created");
|
||||
}
|
||||
/// Build information gauge, always set to 1. Matches Go:
|
||||
/// Namespace="SeaweedFS", Subsystem="build", Name="info",
|
||||
/// labels: version, commit, sizelimit, goos, goarch.
|
||||
pub static BUILD_INFO: LazyLock<GaugeVec> = LazyLock::new(|| {
|
||||
GaugeVec::new(
|
||||
Opts::new("SeaweedFS_build_info", "A metric with a constant '1' value labeled by version, commit, sizelimit, goos, and goarch from which SeaweedFS was built."),
|
||||
&["version", "commit", "sizelimit", "goos", "goarch"],
|
||||
).expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Generate exponential bucket boundaries for histograms.
|
||||
fn exponential_buckets(start: f64, factor: f64, count: usize) -> Vec<f64> {
|
||||
@@ -232,6 +349,7 @@ pub const DOWNLOAD_LIMIT_COND: &str = "downloadLimitCondition";
|
||||
pub const UPLOAD_LIMIT_COND: &str = "uploadLimitCondition";
|
||||
pub const READ_PROXY_REQ: &str = "readProxyRequest";
|
||||
pub const READ_REDIRECT_REQ: &str = "readRedirectRequest";
|
||||
pub const READ_DELETED_NEEDLE: &str = "readDeletedNeedle";
|
||||
pub const EMPTY_READ_PROXY_LOC: &str = "emptyReadProxyLocaction";
|
||||
pub const FAILED_READ_PROXY_REQ: &str = "failedReadProxyRequest";
|
||||
|
||||
@@ -283,6 +401,8 @@ pub fn register_metrics() {
|
||||
Box::new(SCRUB_LAST_TIME_SECONDS.clone()),
|
||||
Box::new(SCRUB_VOLUME_FAILURES.clone()),
|
||||
Box::new(SCRUB_SHARD_FAILURES.clone()),
|
||||
Box::new(STORAGE_IO_ERROR_COUNTER.clone()),
|
||||
Box::new(IO_QUARANTINE_GAUGE.clone()),
|
||||
// Legacy metrics
|
||||
Box::new(VOLUMES_TOTAL.clone()),
|
||||
Box::new(DISK_SIZE_BYTES.clone()),
|
||||
@@ -358,10 +478,8 @@ fn delete_partial_match_collection(gauge: &GaugeVec, collection: &str) {
|
||||
type_value = Some(label.get_value().to_string());
|
||||
}
|
||||
}
|
||||
if matches_collection {
|
||||
if let Some(ref tv) = type_value {
|
||||
let _ = gauge.remove_label_values(&[collection, tv]);
|
||||
}
|
||||
if matches_collection && let Some(ref tv) = type_value {
|
||||
let _ = gauge.remove_label_values(&[collection, tv]);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -408,7 +526,7 @@ pub async fn push_metrics_once(
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use axum::{routing::put, Router};
|
||||
use axum::{Router, routing::put};
|
||||
use std::sync::{Arc, Mutex};
|
||||
|
||||
#[test]
|
||||
@@ -480,7 +598,9 @@ mod tests {
|
||||
register_metrics();
|
||||
|
||||
VOLUME_GAUGE.with_label_values(&["pics", "volume"]).set(2.0);
|
||||
VOLUME_GAUGE.with_label_values(&["pics", "ec_shards"]).set(3.0);
|
||||
VOLUME_GAUGE
|
||||
.with_label_values(&["pics", "ec_shards"])
|
||||
.set(3.0);
|
||||
READ_ONLY_VOLUME_GAUGE
|
||||
.with_label_values(&["pics", "volume"])
|
||||
.set(1.0);
|
||||
|
||||
@@ -119,7 +119,11 @@ pub fn check_blocked_ip(endpoint: &str, ip: IpAddr) -> Result<(), String> {
|
||||
/// reachable for callers whose target legitimately sits on an internal network
|
||||
/// (peer volume servers), while still blocking loopback, link-local (IMDS) and
|
||||
/// unspecified. Mirrors Go's `checkBlockedIPPolicy`.
|
||||
pub fn check_blocked_ip_policy(endpoint: &str, ip: IpAddr, allow_private: bool) -> Result<(), String> {
|
||||
pub fn check_blocked_ip_policy(
|
||||
endpoint: &str,
|
||||
ip: IpAddr,
|
||||
allow_private: bool,
|
||||
) -> Result<(), String> {
|
||||
// Normalize IPv4-mapped IPv6 (`::ffff:a.b.c.d`) to its IPv4 form so the
|
||||
// IPv4 deny rules apply. The OS routes these to the embedded IPv4 address,
|
||||
// so without this `::ffff:127.0.0.1` / `::ffff:169.254.169.254` would slip
|
||||
@@ -173,10 +177,10 @@ pub fn check_blocked_ip_policy(endpoint: &str, ip: IpAddr, allow_private: bool)
|
||||
// same host wherever the matching relay exists (common in IPv6-only cloud).
|
||||
// to_ipv4_mapped above only covers ::ffff: mapped addresses, so pull the
|
||||
// embedded IPv4 out of the other forms and re-check it against the rules.
|
||||
if let IpAddr::V6(v6) = ip {
|
||||
if let Some(v4) = embedded_transition_ipv4(v6) {
|
||||
return check_blocked_ip_policy(endpoint, IpAddr::V4(v4), allow_private);
|
||||
}
|
||||
if let IpAddr::V6(v6) = ip
|
||||
&& let Some(v4) = embedded_transition_ipv4(v6)
|
||||
{
|
||||
return check_blocked_ip_policy(endpoint, IpAddr::V4(v4), allow_private);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@@ -214,9 +218,7 @@ fn precheck_endpoint(endpoint: &str) -> Result<HostCheck, String> {
|
||||
|
||||
// Authority is everything up to the first '/', '?', or '#'.
|
||||
let after = &trimmed[scheme_end + 3..];
|
||||
let authority_end = after
|
||||
.find(|c| c == '/' || c == '?' || c == '#')
|
||||
.unwrap_or(after.len());
|
||||
let authority_end = after.find(['/', '?', '#']).unwrap_or(after.len());
|
||||
let authority = &after[..authority_end];
|
||||
|
||||
// Strip optional userinfo ("user:pass@").
|
||||
@@ -233,7 +235,7 @@ fn precheck_endpoint(endpoint: &str) -> Result<HostCheck, String> {
|
||||
return Err(format!(
|
||||
"remote endpoint {:?} has a malformed IPv6 host",
|
||||
endpoint
|
||||
))
|
||||
));
|
||||
}
|
||||
}
|
||||
} else {
|
||||
@@ -309,7 +311,10 @@ pub async fn validate_replica_target(target: &str) -> Result<(), String> {
|
||||
return Err("replica target is empty".to_string());
|
||||
}
|
||||
if trimmed.contains("://") || trimmed.contains(['/', '?', '#', '@', '\\']) {
|
||||
return Err(format!("replica target {:?} must be a bare host:port", target));
|
||||
return Err(format!(
|
||||
"replica target {:?} must be a bare host:port",
|
||||
target
|
||||
));
|
||||
}
|
||||
|
||||
// Require an explicit host:port, handling `[IPv6]:port`. A bracketless IPv6
|
||||
@@ -318,12 +323,22 @@ pub async fn validate_replica_target(target: &str) -> Result<(), String> {
|
||||
let host = if let Some(rest) = trimmed.strip_prefix('[') {
|
||||
match rest.split_once(']') {
|
||||
Some((h, port)) if port.starts_with(':') && port.len() > 1 => h,
|
||||
_ => return Err(format!("replica target {:?} must be a bare host:port", target)),
|
||||
_ => {
|
||||
return Err(format!(
|
||||
"replica target {:?} must be a bare host:port",
|
||||
target
|
||||
));
|
||||
}
|
||||
}
|
||||
} else {
|
||||
match trimmed.rsplit_once(':') {
|
||||
Some((h, port)) if !port.is_empty() && !h.contains(':') => h,
|
||||
_ => return Err(format!("replica target {:?} must be a bare host:port", target)),
|
||||
_ => {
|
||||
return Err(format!(
|
||||
"replica target {:?} must be a bare host:port",
|
||||
target
|
||||
));
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
@@ -342,7 +357,10 @@ pub async fn validate_replica_target(target: &str) -> Result<(), String> {
|
||||
|
||||
let addrs = resolve_host(host).await?;
|
||||
if addrs.is_empty() {
|
||||
return Err(format!("resolve replica target host {:?}: no addresses", host));
|
||||
return Err(format!(
|
||||
"resolve replica target host {:?}: no addresses",
|
||||
host
|
||||
));
|
||||
}
|
||||
for ip in addrs {
|
||||
check_blocked_ip_policy(target, ip, true)?;
|
||||
@@ -350,6 +368,56 @@ pub async fn validate_replica_target(target: &str) -> Result<(), String> {
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Resolve `host`, re-apply the replica deny list (private peers allowed) to
|
||||
/// every resolved address, and connect to the first one that passes -- the
|
||||
/// connect-time twin of [`validate_replica_target`], so a hostname whose DNS
|
||||
/// answer flips to a blocked address after the up-front check is still refused.
|
||||
/// Mirrors Go's `guardedDialerPolicy` with allowPrivate=true.
|
||||
pub async fn guarded_tcp_connect(
|
||||
host: &str,
|
||||
port: u16,
|
||||
endpoint: &str,
|
||||
) -> std::io::Result<tokio::net::TcpStream> {
|
||||
use std::io::{Error, ErrorKind};
|
||||
|
||||
let denied = |e: String| Error::new(ErrorKind::PermissionDenied, e);
|
||||
|
||||
if is_blocked_imds_host(&host.to_ascii_lowercase()) {
|
||||
return Err(denied(format!(
|
||||
"remote endpoint {:?} targets instance metadata service",
|
||||
endpoint
|
||||
)));
|
||||
}
|
||||
if let Ok(ip) = host.parse::<IpAddr>() {
|
||||
check_blocked_ip_policy(endpoint, ip, true).map_err(denied)?;
|
||||
return tokio::net::TcpStream::connect((ip, port)).await;
|
||||
}
|
||||
|
||||
let lookup = tokio::net::lookup_host((host.to_string(), port));
|
||||
let addrs = tokio::time::timeout(std::time::Duration::from_secs(2), lookup)
|
||||
.await
|
||||
.map_err(|_| {
|
||||
Error::new(
|
||||
ErrorKind::TimedOut,
|
||||
format!("resolve remote endpoint host {:?}: timed out", host),
|
||||
)
|
||||
})??;
|
||||
|
||||
let mut first_block_err: Option<String> = None;
|
||||
for addr in addrs {
|
||||
if let Err(e) = check_blocked_ip_policy(endpoint, addr.ip(), true) {
|
||||
if first_block_err.is_none() {
|
||||
first_block_err = Some(e);
|
||||
}
|
||||
continue;
|
||||
}
|
||||
return tokio::net::TcpStream::connect(addr).await;
|
||||
}
|
||||
Err(denied(first_block_err.unwrap_or_else(|| {
|
||||
format!("resolve remote endpoint host {:?}: no addresses", host)
|
||||
})))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
@@ -380,22 +448,30 @@ mod tests {
|
||||
#[test]
|
||||
fn rejects_empty_and_bad_scheme() {
|
||||
assert!(precheck_endpoint("").unwrap_err().contains("empty"));
|
||||
assert!(precheck_endpoint("ftp://example.com/")
|
||||
.unwrap_err()
|
||||
.contains("http or https"));
|
||||
assert!(precheck_endpoint("example.com/")
|
||||
.unwrap_err()
|
||||
.contains("http or https"));
|
||||
assert!(
|
||||
precheck_endpoint("ftp://example.com/")
|
||||
.unwrap_err()
|
||||
.contains("http or https")
|
||||
);
|
||||
assert!(
|
||||
precheck_endpoint("example.com/")
|
||||
.unwrap_err()
|
||||
.contains("http or https")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rejects_imds_hostnames() {
|
||||
assert!(precheck_endpoint("http://metadata.google.internal/")
|
||||
.unwrap_err()
|
||||
.contains("metadata service"));
|
||||
assert!(precheck_endpoint("http://metadata/")
|
||||
.unwrap_err()
|
||||
.contains("metadata service"));
|
||||
assert!(
|
||||
precheck_endpoint("http://metadata.google.internal/")
|
||||
.unwrap_err()
|
||||
.contains("metadata service")
|
||||
);
|
||||
assert!(
|
||||
precheck_endpoint("http://metadata/")
|
||||
.unwrap_err()
|
||||
.contains("metadata service")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -415,27 +491,41 @@ mod tests {
|
||||
#[test]
|
||||
fn check_blocked_ip_matches_resolved_categories() {
|
||||
// Mirror Go's "host resolves to X" cases at the address level.
|
||||
assert!(check_blocked_ip("e", ip("127.0.0.1"))
|
||||
.unwrap_err()
|
||||
.contains("loopback"));
|
||||
assert!(check_blocked_ip("e", ip("169.254.10.20"))
|
||||
.unwrap_err()
|
||||
.contains("link-local"));
|
||||
assert!(check_blocked_ip("e", ip("10.1.2.3"))
|
||||
.unwrap_err()
|
||||
.contains("private"));
|
||||
assert!(check_blocked_ip("e", ip("172.20.0.5"))
|
||||
.unwrap_err()
|
||||
.contains("private"));
|
||||
assert!(check_blocked_ip("e", ip("192.168.1.1"))
|
||||
.unwrap_err()
|
||||
.contains("private"));
|
||||
assert!(check_blocked_ip("e", ip("100.64.0.42"))
|
||||
.unwrap_err()
|
||||
.contains("CGNAT"));
|
||||
assert!(check_blocked_ip("e", ip("fc00::1"))
|
||||
.unwrap_err()
|
||||
.contains("private"));
|
||||
assert!(
|
||||
check_blocked_ip("e", ip("127.0.0.1"))
|
||||
.unwrap_err()
|
||||
.contains("loopback")
|
||||
);
|
||||
assert!(
|
||||
check_blocked_ip("e", ip("169.254.10.20"))
|
||||
.unwrap_err()
|
||||
.contains("link-local")
|
||||
);
|
||||
assert!(
|
||||
check_blocked_ip("e", ip("10.1.2.3"))
|
||||
.unwrap_err()
|
||||
.contains("private")
|
||||
);
|
||||
assert!(
|
||||
check_blocked_ip("e", ip("172.20.0.5"))
|
||||
.unwrap_err()
|
||||
.contains("private")
|
||||
);
|
||||
assert!(
|
||||
check_blocked_ip("e", ip("192.168.1.1"))
|
||||
.unwrap_err()
|
||||
.contains("private")
|
||||
);
|
||||
assert!(
|
||||
check_blocked_ip("e", ip("100.64.0.42"))
|
||||
.unwrap_err()
|
||||
.contains("CGNAT")
|
||||
);
|
||||
assert!(
|
||||
check_blocked_ip("e", ip("fc00::1"))
|
||||
.unwrap_err()
|
||||
.contains("private")
|
||||
);
|
||||
assert!(check_blocked_ip("e", ip("52.216.10.10")).is_ok());
|
||||
assert!(check_blocked_ip("e", ip("2606:4700:4700::1111")).is_ok());
|
||||
}
|
||||
@@ -478,33 +568,45 @@ mod tests {
|
||||
assert!(check_blocked_ip("e", ip("2001::f7f7:f7f7")).is_ok());
|
||||
assert!(check_blocked_ip("e", ip("::808:808")).is_ok());
|
||||
// Bracketed transition literal via the full endpoint path.
|
||||
assert!(precheck_endpoint("http://[64:ff9b::a9fe:a9fe]/")
|
||||
.unwrap_err()
|
||||
.contains("metadata"));
|
||||
assert!(
|
||||
precheck_endpoint("http://[64:ff9b::a9fe:a9fe]/")
|
||||
.unwrap_err()
|
||||
.contains("metadata")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rejects_ipv4_mapped_ipv6() {
|
||||
// IPv4-mapped IPv6 must be unmapped so the IPv4 rules catch it.
|
||||
assert!(check_blocked_ip("e", ip("::ffff:127.0.0.1"))
|
||||
.unwrap_err()
|
||||
.contains("loopback"));
|
||||
assert!(check_blocked_ip("e", ip("::ffff:169.254.169.254"))
|
||||
.unwrap_err()
|
||||
.contains("metadata"));
|
||||
assert!(check_blocked_ip("e", ip("::ffff:10.0.0.1"))
|
||||
.unwrap_err()
|
||||
.contains("private"));
|
||||
assert!(
|
||||
check_blocked_ip("e", ip("::ffff:127.0.0.1"))
|
||||
.unwrap_err()
|
||||
.contains("loopback")
|
||||
);
|
||||
assert!(
|
||||
check_blocked_ip("e", ip("::ffff:169.254.169.254"))
|
||||
.unwrap_err()
|
||||
.contains("metadata")
|
||||
);
|
||||
assert!(
|
||||
check_blocked_ip("e", ip("::ffff:10.0.0.1"))
|
||||
.unwrap_err()
|
||||
.contains("private")
|
||||
);
|
||||
// A mapped public address still passes, and genuine IPv6 loopback is
|
||||
// still caught by the V6 path.
|
||||
assert!(check_blocked_ip("e", ip("::ffff:52.216.10.10")).is_ok());
|
||||
assert!(check_blocked_ip("e", ip("::1"))
|
||||
.unwrap_err()
|
||||
.contains("loopback"));
|
||||
assert!(
|
||||
check_blocked_ip("e", ip("::1"))
|
||||
.unwrap_err()
|
||||
.contains("loopback")
|
||||
);
|
||||
// Bracketed mapped literal via the full endpoint path.
|
||||
assert!(precheck_endpoint("http://[::ffff:127.0.0.1]/")
|
||||
.unwrap_err()
|
||||
.contains("loopback"));
|
||||
assert!(
|
||||
precheck_endpoint("http://[::ffff:127.0.0.1]/")
|
||||
.unwrap_err()
|
||||
.contains("loopback")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -514,60 +616,86 @@ mod tests {
|
||||
assert!(check_blocked_ip_policy("e", ip("192.168.1.5"), true).is_ok());
|
||||
assert!(check_blocked_ip_policy("e", ip("100.64.0.42"), true).is_ok());
|
||||
// Loopback / IMDS / unspecified stay blocked even when private is allowed.
|
||||
assert!(check_blocked_ip_policy("e", ip("127.0.0.1"), true)
|
||||
.unwrap_err()
|
||||
.contains("loopback"));
|
||||
assert!(check_blocked_ip_policy("e", ip("169.254.169.254"), true)
|
||||
.unwrap_err()
|
||||
.contains("metadata"));
|
||||
assert!(check_blocked_ip_policy("e", ip("0.0.0.0"), true)
|
||||
.unwrap_err()
|
||||
.contains("unspecified"));
|
||||
assert!(
|
||||
check_blocked_ip_policy("e", ip("127.0.0.1"), true)
|
||||
.unwrap_err()
|
||||
.contains("loopback")
|
||||
);
|
||||
assert!(
|
||||
check_blocked_ip_policy("e", ip("169.254.169.254"), true)
|
||||
.unwrap_err()
|
||||
.contains("metadata")
|
||||
);
|
||||
assert!(
|
||||
check_blocked_ip_policy("e", ip("0.0.0.0"), true)
|
||||
.unwrap_err()
|
||||
.contains("unspecified")
|
||||
);
|
||||
}
|
||||
|
||||
#[tokio::test]
|
||||
async fn validate_replica_target_rejects_and_allows() {
|
||||
// A path plus a trailing ?a= would otherwise swallow ?type=replicate.
|
||||
assert!(validate_replica_target("127.0.0.1:7000/status/x/?a=")
|
||||
.await
|
||||
.unwrap_err()
|
||||
.contains("bare host:port"));
|
||||
assert!(validate_replica_target("http://10.0.0.7:8080")
|
||||
.await
|
||||
.unwrap_err()
|
||||
.contains("bare host:port"));
|
||||
assert!(validate_replica_target("user@10.0.0.7:8080")
|
||||
.await
|
||||
.unwrap_err()
|
||||
.contains("bare host:port"));
|
||||
assert!(validate_replica_target("10.0.0.7")
|
||||
.await
|
||||
.unwrap_err()
|
||||
.contains("bare host:port"));
|
||||
assert!(validate_replica_target("peer.example.com")
|
||||
.await
|
||||
.unwrap_err()
|
||||
.contains("bare host:port"));
|
||||
assert!(validate_replica_target("127.0.0.1:8080")
|
||||
.await
|
||||
.unwrap_err()
|
||||
.contains("loopback"));
|
||||
assert!(validate_replica_target("[::1]:8080")
|
||||
.await
|
||||
.unwrap_err()
|
||||
.contains("loopback"));
|
||||
assert!(validate_replica_target("169.254.169.254:80")
|
||||
.await
|
||||
.unwrap_err()
|
||||
.contains("metadata"));
|
||||
assert!(validate_replica_target("metadata:80")
|
||||
.await
|
||||
.unwrap_err()
|
||||
.contains("metadata"));
|
||||
assert!(validate_replica_target("")
|
||||
.await
|
||||
.unwrap_err()
|
||||
.contains("empty"));
|
||||
assert!(
|
||||
validate_replica_target("127.0.0.1:7000/status/x/?a=")
|
||||
.await
|
||||
.unwrap_err()
|
||||
.contains("bare host:port")
|
||||
);
|
||||
assert!(
|
||||
validate_replica_target("http://10.0.0.7:8080")
|
||||
.await
|
||||
.unwrap_err()
|
||||
.contains("bare host:port")
|
||||
);
|
||||
assert!(
|
||||
validate_replica_target("user@10.0.0.7:8080")
|
||||
.await
|
||||
.unwrap_err()
|
||||
.contains("bare host:port")
|
||||
);
|
||||
assert!(
|
||||
validate_replica_target("10.0.0.7")
|
||||
.await
|
||||
.unwrap_err()
|
||||
.contains("bare host:port")
|
||||
);
|
||||
assert!(
|
||||
validate_replica_target("peer.example.com")
|
||||
.await
|
||||
.unwrap_err()
|
||||
.contains("bare host:port")
|
||||
);
|
||||
assert!(
|
||||
validate_replica_target("127.0.0.1:8080")
|
||||
.await
|
||||
.unwrap_err()
|
||||
.contains("loopback")
|
||||
);
|
||||
assert!(
|
||||
validate_replica_target("[::1]:8080")
|
||||
.await
|
||||
.unwrap_err()
|
||||
.contains("loopback")
|
||||
);
|
||||
assert!(
|
||||
validate_replica_target("169.254.169.254:80")
|
||||
.await
|
||||
.unwrap_err()
|
||||
.contains("metadata")
|
||||
);
|
||||
assert!(
|
||||
validate_replica_target("metadata:80")
|
||||
.await
|
||||
.unwrap_err()
|
||||
.contains("metadata")
|
||||
);
|
||||
assert!(
|
||||
validate_replica_target("")
|
||||
.await
|
||||
.unwrap_err()
|
||||
.contains("empty")
|
||||
);
|
||||
// Legitimate peer volume servers on private networks pass.
|
||||
assert!(validate_replica_target("10.0.0.7:8080").await.is_ok());
|
||||
assert!(validate_replica_target("192.168.1.5:8080").await.is_ok());
|
||||
|
||||
@@ -7,7 +7,9 @@ pub mod endpoint_guard;
|
||||
pub mod s3;
|
||||
pub mod s3_tier;
|
||||
|
||||
pub use endpoint_guard::{validate_remote_endpoint, validate_replica_target};
|
||||
pub use endpoint_guard::{
|
||||
guarded_tcp_connect, validate_remote_endpoint, validate_replica_target,
|
||||
};
|
||||
|
||||
use crate::pb::remote_pb::{RemoteConf, RemoteStorageLocation};
|
||||
|
||||
|
||||
@@ -2,9 +2,9 @@
|
||||
//!
|
||||
//! Works with AWS S3, MinIO, SeaweedFS S3, and all S3-compatible providers.
|
||||
|
||||
use aws_sdk_s3::Client;
|
||||
use aws_sdk_s3::config::{BehaviorVersion, Credentials, Region};
|
||||
use aws_sdk_s3::primitives::ByteStream;
|
||||
use aws_sdk_s3::Client;
|
||||
|
||||
use super::{RemoteEntry, RemoteStorageClient, RemoteStorageError};
|
||||
use crate::pb::remote_pb::{RemoteConf, RemoteStorageLocation};
|
||||
|
||||
@@ -7,9 +7,9 @@ use std::collections::HashMap;
|
||||
use std::future::Future;
|
||||
use std::sync::{Arc, OnceLock, RwLock};
|
||||
|
||||
use aws_sdk_s3::Client;
|
||||
use aws_sdk_s3::config::{BehaviorVersion, Credentials, Region};
|
||||
use aws_sdk_s3::types::{CompletedMultipartUpload, CompletedPart};
|
||||
use aws_sdk_s3::Client;
|
||||
use tokio::io::{AsyncReadExt, AsyncSeekExt, AsyncWriteExt};
|
||||
use tokio::sync::Semaphore;
|
||||
|
||||
|
||||
@@ -10,7 +10,7 @@ use std::collections::HashSet;
|
||||
use std::net::IpAddr;
|
||||
use std::time::{SystemTime, UNIX_EPOCH};
|
||||
|
||||
use jsonwebtoken::{decode, encode, Algorithm, DecodingKey, EncodingKey, Header, Validation};
|
||||
use jsonwebtoken::{Algorithm, DecodingKey, EncodingKey, Header, Validation, decode, encode};
|
||||
use serde::{Deserialize, Serialize};
|
||||
|
||||
// ============================================================================
|
||||
@@ -297,10 +297,10 @@ impl Guard {
|
||||
/// Extract host from "host:port" or "[::1]:port" format.
|
||||
fn extract_host(addr: &str) -> String {
|
||||
// Handle IPv6 with brackets
|
||||
if addr.starts_with('[') {
|
||||
if let Some(end) = addr.find(']') {
|
||||
return addr[1..end].to_string();
|
||||
}
|
||||
if addr.starts_with('[')
|
||||
&& let Some(end) = addr.find(']')
|
||||
{
|
||||
return addr[1..end].to_string();
|
||||
}
|
||||
// Handle host:port
|
||||
if let Some(pos) = addr.rfind(':') {
|
||||
@@ -481,9 +481,11 @@ mod tests {
|
||||
let token = gen_jwt(&key, 3600, "3,01637037d6").unwrap();
|
||||
|
||||
// Correct file ID
|
||||
assert!(guard
|
||||
.check_jwt_for_file(Some(&token), "3,01637037d6", true)
|
||||
.is_ok());
|
||||
assert!(
|
||||
guard
|
||||
.check_jwt_for_file(Some(&token), "3,01637037d6", true)
|
||||
.is_ok()
|
||||
);
|
||||
|
||||
// Wrong file ID
|
||||
let err = guard.check_jwt_for_file(Some(&token), "4,deadbeef", true);
|
||||
|
||||
@@ -3,12 +3,12 @@ use std::fmt;
|
||||
use std::sync::Arc;
|
||||
|
||||
use rustls::client::danger::HandshakeSignatureValid;
|
||||
use rustls::crypto::aws_lc_rs;
|
||||
use rustls::crypto::CryptoProvider;
|
||||
use rustls::crypto::aws_lc_rs;
|
||||
use rustls::pki_types::UnixTime;
|
||||
use rustls::pki_types::{CertificateDer, PrivateKeyDer};
|
||||
use rustls::server::danger::{ClientCertVerified, ClientCertVerifier};
|
||||
use rustls::server::WebPkiClientVerifier;
|
||||
use rustls::server::danger::{ClientCertVerified, ClientCertVerifier};
|
||||
use rustls::{
|
||||
CipherSuite, DigitallySignedStruct, DistinguishedName, RootCertStore, ServerConfig,
|
||||
SignatureScheme, SupportedCipherSuite, SupportedProtocolVersion,
|
||||
@@ -376,7 +376,7 @@ fn go_tls_version_for_supported(version: &SupportedProtocolVersion) -> GoTlsVers
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::{build_supported_versions, common_name_is_allowed, parse_cipher_suites, TlsPolicy};
|
||||
use super::{TlsPolicy, build_supported_versions, common_name_is_allowed, parse_cipher_suites};
|
||||
use rustls::crypto::aws_lc_rs;
|
||||
use std::collections::HashSet;
|
||||
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
use axum::Router;
|
||||
use axum::body::Body;
|
||||
use axum::extract::Query;
|
||||
use axum::http::{header, StatusCode};
|
||||
use axum::http::{StatusCode, header};
|
||||
use axum::response::{IntoResponse, Response};
|
||||
use axum::routing::{any, get};
|
||||
use axum::Router;
|
||||
use pprof::protos::Message;
|
||||
use serde::Deserialize;
|
||||
|
||||
|
||||
@@ -40,7 +40,9 @@ pub fn load_outgoing_grpc_tls(
|
||||
(&config.grpc_client_cert_file, &config.grpc_client_key_file)
|
||||
} else {
|
||||
if !config.grpc_client_cert_file.is_empty() || !config.grpc_client_key_file.is_empty() {
|
||||
tracing::warn!("grpc.volume.client_cert and grpc.volume.client_key must both be set, falling back to grpc.volume.cert and grpc.volume.key");
|
||||
tracing::warn!(
|
||||
"grpc.volume.client_cert and grpc.volume.client_key must both be set, falling back to grpc.volume.cert and grpc.volume.key"
|
||||
);
|
||||
}
|
||||
(&config.grpc_cert_file, &config.grpc_key_file)
|
||||
};
|
||||
@@ -115,6 +117,38 @@ pub fn build_grpc_endpoint(
|
||||
Ok(endpoint)
|
||||
}
|
||||
|
||||
/// Connect `endpoint` through a connector that re-validates every resolved
|
||||
/// address at connect time (Go's `guardedDialerPolicy` mirror), pinning a
|
||||
/// validated copy/tail source against DNS rebinding. `allow_untrusted`
|
||||
/// preserves the plain connect for operators that opted out.
|
||||
pub async fn connect_guarded(
|
||||
endpoint: Endpoint,
|
||||
target: &str,
|
||||
allow_untrusted: bool,
|
||||
) -> Result<Channel, GrpcClientError> {
|
||||
if allow_untrusted {
|
||||
return endpoint
|
||||
.connect()
|
||||
.await
|
||||
.map_err(|e| GrpcClientError(format!("connect {} failed: {}", target, e)));
|
||||
}
|
||||
let target_owned = target.to_string();
|
||||
let connector = tower::service_fn(move |uri: Uri| {
|
||||
let target = target_owned.clone();
|
||||
async move {
|
||||
let host = uri.host().unwrap_or_default().to_string();
|
||||
let port = uri.port_u16().unwrap_or(80);
|
||||
crate::remote_storage::guarded_tcp_connect(&host, port, &target)
|
||||
.await
|
||||
.map(hyper_util::rt::TokioIo::new)
|
||||
}
|
||||
});
|
||||
endpoint
|
||||
.connect_with_connector(connector)
|
||||
.await
|
||||
.map_err(|e| GrpcClientError(format!("connect {} failed: {}", target, e)))
|
||||
}
|
||||
|
||||
/// Parse a SeaweedFS server address (`"ip:port.grpcPort"` or
|
||||
/// `"ip:port"`) into the `host:grpcPort` form `build_grpc_endpoint`
|
||||
/// expects. With the trailing `.grpcPort` segment, that segment IS
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -5,25 +5,26 @@
|
||||
|
||||
use std::collections::HashMap;
|
||||
use std::path::Path;
|
||||
use std::sync::atomic::Ordering;
|
||||
use std::sync::Arc;
|
||||
use std::sync::atomic::Ordering;
|
||||
use std::time::{Duration, SystemTime, UNIX_EPOCH};
|
||||
|
||||
use tokio::sync::broadcast;
|
||||
use tracing::{error, info, warn};
|
||||
|
||||
use super::grpc_client::{build_grpc_endpoint, GRPC_MAX_MESSAGE_SIZE};
|
||||
use super::grpc_client::{GRPC_MAX_MESSAGE_SIZE, build_grpc_endpoint};
|
||||
use super::volume_server::VolumeServerState;
|
||||
use crate::pb::master_pb;
|
||||
use crate::pb::master_pb::seaweed_client::SeaweedClient;
|
||||
use crate::pb::volume_server_pb;
|
||||
use crate::remote_storage::s3_tier::{S3TierBackend, S3TierConfig};
|
||||
use crate::storage::store::Store;
|
||||
use crate::storage::types::NeedleId;
|
||||
use crate::storage::types::{NeedleId, VolumeId};
|
||||
use crate::storage::volume_report::VolumeReportKey;
|
||||
use crate::storage::volume_report_hash::report_hash;
|
||||
|
||||
const DUPLICATE_UUID_RETRY_MESSAGE: &str = "duplicate UUIDs detected, retrying connection";
|
||||
const VOLUME_IO_ERROR_TOLERANCE: i32 = 3;
|
||||
const MAX_DUPLICATE_UUID_RETRIES: u32 = 3;
|
||||
|
||||
/// Configuration for the heartbeat client.
|
||||
@@ -118,7 +119,9 @@ pub async fn run_heartbeat_with_state(
|
||||
|
||||
if err_msg.contains(DUPLICATE_UUID_RETRY_MESSAGE) {
|
||||
if duplicate_retry_count >= MAX_DUPLICATE_UUID_RETRIES {
|
||||
error!("Shut down Volume Server due to persistent duplicate volume directories after 3 retries");
|
||||
error!(
|
||||
"Shut down Volume Server due to persistent duplicate volume directories after 3 retries"
|
||||
);
|
||||
error!(
|
||||
"Please check if another volume server is using the same directory"
|
||||
);
|
||||
@@ -188,10 +191,10 @@ pub async fn run_heartbeat_with_state(
|
||||
pub fn to_grpc_address(master_addr: &str) -> String {
|
||||
if let Some((host, port_str)) = master_addr.rsplit_once(':') {
|
||||
// "host:port.grpcPort" — the part after the last '.' is the gRPC port.
|
||||
if let Some((_, grpc_port)) = port_str.rsplit_once('.') {
|
||||
if grpc_port.parse::<u16>().is_ok() {
|
||||
return format!("{}:{}", host, grpc_port);
|
||||
}
|
||||
if let Some((_, grpc_port)) = port_str.rsplit_once('.')
|
||||
&& grpc_port.parse::<u16>().is_ok()
|
||||
{
|
||||
return format!("{}:{}", host, grpc_port);
|
||||
}
|
||||
if let Ok(port) = port_str.parse::<u16>() {
|
||||
let grpc_port = port + 10000;
|
||||
@@ -315,6 +318,10 @@ fn collect_ec_shard_delta_messages(
|
||||
|
||||
for (disk_id, loc) in store.locations.iter().enumerate() {
|
||||
for (_, ec_vol) in loc.ec_volumes() {
|
||||
let (_, _, quarantined) = ec_vol.get_io_error_state();
|
||||
if quarantined {
|
||||
continue;
|
||||
}
|
||||
for shard in ec_vol.shards.iter().flatten() {
|
||||
messages.insert(
|
||||
(
|
||||
@@ -895,6 +902,7 @@ fn build_heartbeat_with_ec_status(
|
||||
// master can tell whether applying what it was sent leaves it current.
|
||||
// Volumes skipped below -- quarantined, phantom, expired -- are in neither.
|
||||
let mut volume_digest: u64 = 0;
|
||||
let mut quarantined_volumes: u32 = 0;
|
||||
let (send_full_list, report_generation, report_pass) = store.volume_report.begin();
|
||||
let mut changed_volumes = Vec::new();
|
||||
let mut max_file_key = NeedleId(0);
|
||||
@@ -916,10 +924,9 @@ fn build_heartbeat_with_ec_status(
|
||||
let mut effective_max_count = loc.max_volume_count.load(Ordering::Relaxed);
|
||||
if loc.is_disk_space_low.load(Ordering::Relaxed) {
|
||||
let used_slots = loc.volumes_len() as i32
|
||||
+ ((loc.ec_shard_count()
|
||||
+ crate::storage::erasure_coding::ec_shard::DATA_SHARDS_COUNT
|
||||
- 1)
|
||||
/ crate::storage::erasure_coding::ec_shard::DATA_SHARDS_COUNT)
|
||||
+ loc
|
||||
.ec_shard_count()
|
||||
.div_ceil(crate::storage::erasure_coding::ec_shard::DATA_SHARDS_COUNT)
|
||||
as i32;
|
||||
effective_max_count = used_slots;
|
||||
}
|
||||
@@ -937,6 +944,7 @@ fn build_heartbeat_with_ec_status(
|
||||
loc.disk_free_bytes.load(Ordering::Relaxed);
|
||||
|
||||
let mut delete_vids = Vec::new();
|
||||
let mut quarantine_vids: Vec<VolumeId> = Vec::new();
|
||||
for (_, vol) in loc.iter_volumes() {
|
||||
let cur_max = vol.max_file_key();
|
||||
if cur_max > max_file_key {
|
||||
@@ -946,9 +954,18 @@ fn build_heartbeat_with_ec_status(
|
||||
let volume_size = vol.dat_file_size().unwrap_or(0);
|
||||
let mut should_delete_volume = false;
|
||||
|
||||
if vol.last_io_error().is_some() {
|
||||
delete_vids.push(vol.id);
|
||||
should_delete_volume = true;
|
||||
let (_, io_count, io_quarantined) = vol.get_io_error_state();
|
||||
if io_quarantined || io_count >= VOLUME_IO_ERROR_TOLERANCE {
|
||||
if !io_quarantined {
|
||||
vol.mark_io_quarantined();
|
||||
warn!(
|
||||
"Volume {} quarantined after {} consecutive IO errors",
|
||||
vol.id.0, io_count
|
||||
);
|
||||
}
|
||||
quarantined_volumes += 1;
|
||||
quarantine_vids.push(vol.id);
|
||||
continue;
|
||||
} else if !vol.is_expired(volume_size, volume_size_limit) {
|
||||
// Detect phantom volumes: the .dat was unlinked from disk but is still
|
||||
// held open as a deleted FD, so the volume keeps serving and heartbeating
|
||||
@@ -966,7 +983,11 @@ fn build_heartbeat_with_ec_status(
|
||||
> DISK_CHECK_INTERVAL_NS
|
||||
{
|
||||
if !Path::new(&vol.file_name(".dat")).exists() {
|
||||
warn!("Volume {}: data file {} missing (held open as deleted FD) - not reporting to master", vol.id.0, vol.file_name(".dat"));
|
||||
warn!(
|
||||
"Volume {}: data file {} missing (held open as deleted FD) - not reporting to master",
|
||||
vol.id.0,
|
||||
vol.file_name(".dat")
|
||||
);
|
||||
continue;
|
||||
}
|
||||
vol.last_disk_check_ns.store(now_ns, Ordering::Relaxed);
|
||||
@@ -1046,6 +1067,12 @@ fn build_heartbeat_with_ec_status(
|
||||
for vid in delete_vids {
|
||||
let _ = loc.delete_volume(vid, false, false);
|
||||
}
|
||||
|
||||
for vid in quarantine_vids {
|
||||
if let Some(vol) = loc.find_volume_mut(vid) {
|
||||
vol.set_no_write_or_delete(true);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Update disk size and read-only gauges
|
||||
@@ -1102,6 +1129,23 @@ fn build_heartbeat_with_ec_status(
|
||||
};
|
||||
let (location_uuids, disk_tags) = collect_location_metadata(store, &disk_max_by_id);
|
||||
|
||||
let mut quarantined_ec_shards: u32 = 0;
|
||||
for loc in &store.locations {
|
||||
for (_, ec_vol) in loc.ec_volumes() {
|
||||
let (_, _, quarantined) = ec_vol.get_io_error_state();
|
||||
if quarantined {
|
||||
quarantined_ec_shards += ec_vol.shard_count() as u32;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
crate::metrics::IO_QUARANTINE_GAUGE
|
||||
.with_label_values(&["volume"])
|
||||
.set(quarantined_volumes as i64);
|
||||
crate::metrics::IO_QUARANTINE_GAUGE
|
||||
.with_label_values(&["ec_shard"])
|
||||
.set(quarantined_ec_shards as i64);
|
||||
|
||||
let heartbeat = master_pb::Heartbeat {
|
||||
id: store.id.clone(),
|
||||
ip: config.ip.clone(),
|
||||
@@ -1137,6 +1181,10 @@ fn collect_live_ec_shards(
|
||||
|
||||
for (disk_id, loc) in store.locations.iter().enumerate() {
|
||||
for (_, ec_vol) in loc.ec_volumes() {
|
||||
let (_, _, quarantined) = ec_vol.get_io_error_state();
|
||||
if quarantined {
|
||||
continue;
|
||||
}
|
||||
for message in ec_vol.to_volume_ec_shard_information_messages(disk_id as u32) {
|
||||
if update_metrics {
|
||||
let total_size: u64 = message
|
||||
@@ -1205,9 +1253,10 @@ mod tests {
|
||||
use crate::remote_storage::s3_tier::S3TierRegistry;
|
||||
use crate::security::{Guard, SigningKey};
|
||||
use crate::storage::needle_map::NeedleMapKind;
|
||||
use crate::storage::types::{DiskType, Version, VolumeId};
|
||||
use std::sync::atomic::Ordering;
|
||||
use crate::storage::types::{DiskType, VolumeId};
|
||||
use crate::storage::volume::VolumeSpec;
|
||||
use std::sync::RwLock;
|
||||
use std::sync::atomic::Ordering;
|
||||
use std::time::{SystemTime, UNIX_EPOCH};
|
||||
|
||||
fn test_config() -> HeartbeatConfig {
|
||||
@@ -1318,12 +1367,11 @@ mod tests {
|
||||
store
|
||||
.add_volume(
|
||||
VolumeId(7),
|
||||
"pics",
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
DiskType::HardDrive,
|
||||
Version::current(),
|
||||
&VolumeSpec {
|
||||
collection: "pics",
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
@@ -1368,12 +1416,11 @@ mod tests {
|
||||
store
|
||||
.add_volume(
|
||||
VolumeId(7),
|
||||
"pics",
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
DiskType::HardDrive,
|
||||
Version::current(),
|
||||
&VolumeSpec {
|
||||
collection: "pics",
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
@@ -1434,12 +1481,11 @@ mod tests {
|
||||
store
|
||||
.add_volume(
|
||||
VolumeId(id),
|
||||
"pics",
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
DiskType::HardDrive,
|
||||
Version::current(),
|
||||
&VolumeSpec {
|
||||
collection: "pics",
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
@@ -1488,12 +1534,11 @@ mod tests {
|
||||
store
|
||||
.add_volume(
|
||||
VolumeId(3),
|
||||
"pics",
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
DiskType::HardDrive,
|
||||
Version::current(),
|
||||
&VolumeSpec {
|
||||
collection: "pics",
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
@@ -1549,12 +1594,11 @@ mod tests {
|
||||
store
|
||||
.add_volume(
|
||||
VolumeId(3),
|
||||
"pics",
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
DiskType::HardDrive,
|
||||
Version::current(),
|
||||
&VolumeSpec {
|
||||
collection: "pics",
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
@@ -1592,12 +1636,11 @@ mod tests {
|
||||
store
|
||||
.add_volume(
|
||||
vid,
|
||||
"pics",
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
DiskType::HardDrive,
|
||||
Version::current(),
|
||||
&VolumeSpec {
|
||||
collection: "pics",
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
@@ -1631,12 +1674,11 @@ mod tests {
|
||||
store
|
||||
.add_volume(
|
||||
VolumeId(17),
|
||||
"heartbeat_metrics_case",
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
DiskType::HardDrive,
|
||||
Version::current(),
|
||||
&VolumeSpec {
|
||||
collection: "heartbeat_metrics_case",
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
store.locations[0]
|
||||
@@ -1734,12 +1776,11 @@ mod tests {
|
||||
store
|
||||
.add_volume(
|
||||
VolumeId(21),
|
||||
collection,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
DiskType::HardDrive,
|
||||
Version::current(),
|
||||
&VolumeSpec {
|
||||
collection,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
{
|
||||
@@ -1907,12 +1948,13 @@ mod tests {
|
||||
store
|
||||
.add_volume(
|
||||
VolumeId(41),
|
||||
"expired_volume_case",
|
||||
None,
|
||||
Some(crate::storage::needle::ttl::TTL::read("20m").unwrap()),
|
||||
1024,
|
||||
DiskType::HardDrive,
|
||||
Version::current(),
|
||||
&VolumeSpec {
|
||||
collection: "expired_volume_case",
|
||||
ttl: Some(crate::storage::needle::ttl::TTL::read("20m").unwrap()),
|
||||
preallocate: 1024,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
let dat_path = {
|
||||
@@ -1963,12 +2005,11 @@ mod tests {
|
||||
store
|
||||
.add_volume(
|
||||
VolumeId(51),
|
||||
"io_error_case",
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
DiskType::HardDrive,
|
||||
Version::current(),
|
||||
&VolumeSpec {
|
||||
collection: "io_error_case",
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
let (_, volume) = store.find_volume_mut(VolumeId(51)).unwrap();
|
||||
@@ -1976,8 +2017,13 @@ mod tests {
|
||||
|
||||
let heartbeat = build_heartbeat(&test_config(), &mut store);
|
||||
|
||||
// A sustained IO error quarantines the volume: it stays mounted
|
||||
// (so healthz can observe the quarantine state) but is not
|
||||
// advertised to the master.
|
||||
assert!(heartbeat.volumes.is_empty());
|
||||
assert!(!store.has_volume(VolumeId(51)));
|
||||
assert!(store.has_volume(VolumeId(51)));
|
||||
let (_, volume) = store.find_volume_mut(VolumeId(51)).unwrap();
|
||||
assert!(volume.is_no_write_or_delete());
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -1999,12 +2045,11 @@ mod tests {
|
||||
store
|
||||
.add_volume(
|
||||
VolumeId(71),
|
||||
"remote_volume_case",
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
DiskType::HardDrive,
|
||||
Version::current(),
|
||||
&VolumeSpec {
|
||||
collection: "remote_volume_case",
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
let (_, volume) = store.find_volume_mut(VolumeId(71)).unwrap();
|
||||
|
||||
@@ -29,11 +29,11 @@ impl<S> Layer<S> for GrpcRequestIdLayer {
|
||||
|
||||
impl<S, B> Service<http::Request<B>> for GrpcRequestIdService<S>
|
||||
where
|
||||
S: Service<http::Request<B>, Response = http::Response<tonic::body::BoxBody>> + Send + 'static,
|
||||
S: Service<http::Request<B>, Response = http::Response<tonic::body::Body>> + Send + 'static,
|
||||
S::Future: Send + 'static,
|
||||
B: Send + 'static,
|
||||
{
|
||||
type Response = http::Response<tonic::body::BoxBody>;
|
||||
type Response = http::Response<tonic::body::Body>;
|
||||
type Error = S::Error;
|
||||
type Future = Pin<Box<dyn Future<Output = Result<Self::Response, Self::Error>> + Send>>;
|
||||
|
||||
@@ -57,7 +57,7 @@ where
|
||||
let future = self.inner.call(request);
|
||||
|
||||
Box::pin(async move {
|
||||
let mut response: http::Response<tonic::body::BoxBody> =
|
||||
let mut response: http::Response<tonic::body::Body> =
|
||||
scope_request_id(request_id.clone(), future).await?;
|
||||
if let Ok(value) = HeaderValue::from_str(&request_id) {
|
||||
response.headers_mut().insert("x-amz-request-id", value);
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -423,13 +423,12 @@ fn collect_ui_data(
|
||||
shard_id: shard.shard_id,
|
||||
size: shard_size,
|
||||
});
|
||||
if created_at == "-" {
|
||||
if let Ok(metadata) = std::fs::metadata(shard.file_name()) {
|
||||
if let Ok(modified) = metadata.modified() {
|
||||
let ts: chrono::DateTime<chrono::Local> = modified.into();
|
||||
created_at = ts.format("%Y-%m-%d %H:%M").to_string();
|
||||
}
|
||||
}
|
||||
if created_at == "-"
|
||||
&& let Ok(metadata) = std::fs::metadata(shard.file_name())
|
||||
&& let Ok(modified) = metadata.modified()
|
||||
{
|
||||
let ts: chrono::DateTime<chrono::Local> = modified.into();
|
||||
created_at = ts.format("%Y-%m-%d %H:%M").to_string();
|
||||
}
|
||||
}
|
||||
let preferred_size = ec_volume.dat_file_size.max(0) as u64;
|
||||
|
||||
@@ -14,12 +14,12 @@ use std::sync::atomic::{AtomicBool, AtomicI64, AtomicU32, Ordering};
|
||||
use std::sync::{Arc, RwLock};
|
||||
|
||||
use axum::{
|
||||
extract::{connect_info::ConnectInfo, Request, State},
|
||||
http::{header, HeaderValue, Method, StatusCode},
|
||||
Router,
|
||||
extract::{Request, State, connect_info::ConnectInfo},
|
||||
http::{HeaderValue, Method, StatusCode, header},
|
||||
middleware::{self, Next},
|
||||
response::{IntoResponse, Response},
|
||||
routing::{any, get},
|
||||
Router,
|
||||
};
|
||||
|
||||
use crate::config::ReadMode;
|
||||
@@ -200,9 +200,7 @@ pub fn to_http_address(addr: &str) -> std::borrow::Cow<'_, str> {
|
||||
// rather than being silently rewritten. Mirrors the validation already
|
||||
// done in `to_grpc_address` for the inverse direction.
|
||||
if let (Ok(_), Ok(_)) = (http_port.parse::<u16>(), grpc_port.parse::<u16>()) {
|
||||
return std::borrow::Cow::Owned(
|
||||
addr[..ports_sep_index + 1 + dot_idx].to_string(),
|
||||
);
|
||||
return std::borrow::Cow::Owned(addr[..ports_sep_index + 1 + dot_idx].to_string());
|
||||
}
|
||||
}
|
||||
std::borrow::Cow::Borrowed(addr)
|
||||
@@ -312,16 +310,15 @@ async fn admin_store_handler(state: State<Arc<VolumeServerState>>, request: Requ
|
||||
)
|
||||
}
|
||||
};
|
||||
if method == Method::GET {
|
||||
if let Some(response_bytes) = response
|
||||
if method == Method::GET
|
||||
&& let Some(response_bytes) = response
|
||||
.headers()
|
||||
.get(header::CONTENT_LENGTH)
|
||||
.and_then(|value| value.to_str().ok())
|
||||
.and_then(|value| value.parse::<i64>().ok())
|
||||
.filter(|value| *value > 0)
|
||||
{
|
||||
super::server_stats::record_bytes_out(response_bytes);
|
||||
}
|
||||
{
|
||||
super::server_stats::record_bytes_out(response_bytes);
|
||||
}
|
||||
super::server_stats::record_request_close();
|
||||
crate::metrics::INFLIGHT_REQUESTS_GAUGE
|
||||
@@ -358,16 +355,15 @@ async fn public_store_handler(state: State<Arc<VolumeServerState>>, request: Req
|
||||
}
|
||||
_ => StatusCode::OK.into_response(),
|
||||
};
|
||||
if method == Method::GET {
|
||||
if let Some(response_bytes) = response
|
||||
if method == Method::GET
|
||||
&& let Some(response_bytes) = response
|
||||
.headers()
|
||||
.get(header::CONTENT_LENGTH)
|
||||
.and_then(|value| value.to_str().ok())
|
||||
.and_then(|value| value.parse::<i64>().ok())
|
||||
.filter(|value| *value > 0)
|
||||
{
|
||||
super::server_stats::record_bytes_out(response_bytes);
|
||||
}
|
||||
{
|
||||
super::server_stats::record_bytes_out(response_bytes);
|
||||
}
|
||||
super::server_stats::record_request_close();
|
||||
crate::metrics::INFLIGHT_REQUESTS_GAUGE
|
||||
@@ -432,13 +428,13 @@ pub fn build_admin_router_with_ui(state: Arc<VolumeServerState>, ui_enabled: boo
|
||||
.route("/healthz", get(handlers::healthz_handler))
|
||||
.route("/favicon.ico", get(handlers::favicon_handler))
|
||||
.route(
|
||||
"/seaweedfsstatic/*path",
|
||||
"/seaweedfsstatic/{*path}",
|
||||
get(handlers::static_asset_handler),
|
||||
)
|
||||
.route("/", any(admin_store_handler))
|
||||
.route("/:path", any(admin_store_handler))
|
||||
.route("/:vid/:fid", any(admin_store_handler))
|
||||
.route("/:vid/:fid/:filename", any(admin_store_handler))
|
||||
.route("/{path}", any(admin_store_handler))
|
||||
.route("/{vid}/{fid}", any(admin_store_handler))
|
||||
.route("/{vid}/{fid}/{filename}", any(admin_store_handler))
|
||||
.fallback(admin_store_handler);
|
||||
if ui_enabled {
|
||||
// Note: /stats/* endpoints are commented out in Go's volume_server.go (L130-134).
|
||||
@@ -455,13 +451,13 @@ pub fn build_public_router(state: Arc<VolumeServerState>) -> Router {
|
||||
Router::new()
|
||||
.route("/favicon.ico", get(handlers::favicon_handler))
|
||||
.route(
|
||||
"/seaweedfsstatic/*path",
|
||||
"/seaweedfsstatic/{*path}",
|
||||
get(handlers::static_asset_handler),
|
||||
)
|
||||
.route("/", any(public_store_handler))
|
||||
.route("/:path", any(public_store_handler))
|
||||
.route("/:vid/:fid", any(public_store_handler))
|
||||
.route("/:vid/:fid/:filename", any(public_store_handler))
|
||||
.route("/{path}", any(public_store_handler))
|
||||
.route("/{vid}/{fid}", any(public_store_handler))
|
||||
.route("/{vid}/{fid}/{filename}", any(public_store_handler))
|
||||
.fallback(public_store_handler)
|
||||
.layer(middleware::from_fn(common_headers_middleware))
|
||||
.with_state(state)
|
||||
@@ -516,7 +512,10 @@ mod tests {
|
||||
// "host:abc.def"), and silently rewriting it would just hide the bug.
|
||||
assert_eq!(to_http_address("host:abc.def"), "host:abc.def");
|
||||
assert_eq!(to_http_address("host:9333.notaport"), "host:9333.notaport");
|
||||
assert_eq!(to_http_address("host:notaport.19333"), "host:notaport.19333");
|
||||
assert_eq!(
|
||||
to_http_address("host:notaport.19333"),
|
||||
"host:notaport.19333"
|
||||
);
|
||||
// Out-of-range ports must not be silently truncated either.
|
||||
assert_eq!(to_http_address("host:99999.19333"), "host:99999.19333");
|
||||
}
|
||||
|
||||
@@ -178,8 +178,8 @@ mod tests {
|
||||
use crate::server::volume_server::RuntimeMetricsConfig;
|
||||
use crate::storage::needle_map::NeedleMapKind;
|
||||
use crate::storage::store::Store;
|
||||
use std::sync::atomic::{AtomicBool, AtomicI64, AtomicU32};
|
||||
use std::sync::RwLock;
|
||||
use std::sync::atomic::{AtomicBool, AtomicI64, AtomicU32};
|
||||
|
||||
let store = Store::new(NeedleMapKind::InMemory);
|
||||
let guard = Guard::new(&[], SigningKey(vec![]), 0, SigningKey(vec![]), 0);
|
||||
|
||||
@@ -15,15 +15,15 @@ use tracing::warn;
|
||||
use crate::config::MinFreeSpace;
|
||||
use crate::storage::erasure_coding::ec_bitrot::remove_bitrot_sidecars;
|
||||
use crate::storage::erasure_coding::ec_shard::{
|
||||
EcVolumeShard, DATA_SHARDS_COUNT, ERASURE_CODING_LARGE_BLOCK_SIZE,
|
||||
ERASURE_CODING_SMALL_BLOCK_SIZE,
|
||||
DATA_SHARDS_COUNT, ERASURE_CODING_LARGE_BLOCK_SIZE, ERASURE_CODING_SMALL_BLOCK_SIZE,
|
||||
EcVolumeShard, ShardId,
|
||||
};
|
||||
use crate::storage::erasure_coding::ec_volume::EcVolume;
|
||||
use crate::storage::needle_map::NeedleMapKind;
|
||||
use crate::storage::super_block::{ReplicaPlacement, SUPER_BLOCK_SIZE};
|
||||
use crate::storage::super_block::SUPER_BLOCK_SIZE;
|
||||
use crate::storage::types::*;
|
||||
use crate::storage::volume::{
|
||||
remove_volume_files, volume_file_name, VifVolumeInfo, Volume, VolumeError,
|
||||
VifVolumeInfo, Volume, VolumeError, VolumeSpec, remove_volume_files, volume_file_name,
|
||||
};
|
||||
|
||||
/// A single disk location managing volumes in one directory.
|
||||
@@ -131,10 +131,10 @@ impl DiskLocation {
|
||||
for entry in entries {
|
||||
let entry = entry?;
|
||||
let name = entry.file_name().into_string().unwrap_or_default();
|
||||
if let Some((collection, vid)) = parse_volume_filename(&name) {
|
||||
if seen.insert((collection.clone(), vid)) {
|
||||
dat_files.push((collection, vid));
|
||||
}
|
||||
if let Some((collection, vid)) = parse_volume_filename(&name)
|
||||
&& seen.insert((collection.clone(), vid))
|
||||
{
|
||||
dat_files.push((collection, vid));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -205,7 +205,6 @@ impl DiskLocation {
|
||||
continue;
|
||||
}
|
||||
|
||||
|
||||
// Load existing data only; never create a phantom `.dat`. A lone
|
||||
// `.vif`/`.idx` (e.g. an EC sidecar whose `.ecx` is on a sibling
|
||||
// disk) would otherwise have Volume::new write an 8-byte stub that
|
||||
@@ -280,30 +279,33 @@ impl DiskLocation {
|
||||
let opened = Mutex::new(Vec::with_capacity(to_load.len()));
|
||||
std::thread::scope(|scope| {
|
||||
for _ in 0..workers {
|
||||
scope.spawn(|| loop {
|
||||
let i = next.fetch_add(1, Ordering::Relaxed);
|
||||
let Some((vid, collections)) = to_load.get(i) else {
|
||||
return;
|
||||
};
|
||||
for collection in collections {
|
||||
match Volume::new(
|
||||
&self.directory,
|
||||
&self.idx_directory,
|
||||
collection,
|
||||
*vid,
|
||||
needle_map_kind,
|
||||
None, // replica placement read from superblock
|
||||
None, // TTL read from superblock
|
||||
0, // no preallocate on load
|
||||
Version::current(),
|
||||
) {
|
||||
Ok(mut v) => {
|
||||
v.location_disk_space_low = self.is_disk_space_low.clone();
|
||||
opened.lock().unwrap().push((collection.clone(), *vid, v));
|
||||
break;
|
||||
}
|
||||
Err(e) => {
|
||||
warn!(volume_id = vid.0, error = %e, "failed to load volume");
|
||||
scope.spawn(|| {
|
||||
loop {
|
||||
let i = next.fetch_add(1, Ordering::Relaxed);
|
||||
let Some((vid, collections)) = to_load.get(i) else {
|
||||
return;
|
||||
};
|
||||
for collection in collections {
|
||||
// Replica placement and TTL are read back from the
|
||||
// superblock, and a load never preallocates.
|
||||
match Volume::new(
|
||||
&self.directory,
|
||||
&self.idx_directory,
|
||||
*vid,
|
||||
needle_map_kind,
|
||||
&VolumeSpec {
|
||||
collection,
|
||||
..Default::default()
|
||||
},
|
||||
) {
|
||||
Ok(mut v) => {
|
||||
v.location_disk_space_low = self.is_disk_space_low.clone();
|
||||
opened.lock().unwrap().push((collection.clone(), *vid, v));
|
||||
break;
|
||||
}
|
||||
Err(e) => {
|
||||
warn!(volume_id = vid.0, error = %e, "failed to load volume");
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -327,10 +329,10 @@ impl DiskLocation {
|
||||
.strip_suffix(".cpc")
|
||||
.or_else(|| name.strip_suffix(".cpd"))
|
||||
.or_else(|| name.strip_suffix(".cpx"));
|
||||
if let Some(stem) = stem {
|
||||
if let Some(key) = parse_collection_volume_id(stem) {
|
||||
pending.insert(key);
|
||||
}
|
||||
if let Some(stem) = stem
|
||||
&& let Some(key) = parse_collection_volume_id(stem)
|
||||
{
|
||||
pending.insert(key);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -374,8 +376,10 @@ impl DiskLocation {
|
||||
let mut expected_shard_size: Option<i64> = None;
|
||||
let dat_exists = match fs::metadata(&dat_path) {
|
||||
Ok(meta) if meta.len() > SUPER_BLOCK_SIZE as u64 => {
|
||||
expected_shard_size =
|
||||
Some(calculate_expected_shard_size(meta.len() as i64, data_shards));
|
||||
expected_shard_size = Some(calculate_expected_shard_size(
|
||||
meta.len() as i64,
|
||||
data_shards,
|
||||
));
|
||||
true
|
||||
}
|
||||
Ok(_) => false,
|
||||
@@ -399,7 +403,13 @@ impl DiskLocation {
|
||||
if size != prev {
|
||||
// Inconsistent sizes signal corruption or mixed
|
||||
// generations; not trusted for deletion -> keep.
|
||||
warn!(volume_id = vid.0, shard = i, size, expected = prev, "EC shard size mismatch; keeping shards");
|
||||
warn!(
|
||||
volume_id = vid.0,
|
||||
shard = i,
|
||||
size,
|
||||
expected = prev,
|
||||
"EC shard size mismatch; keeping shards"
|
||||
);
|
||||
return true;
|
||||
}
|
||||
} else {
|
||||
@@ -426,11 +436,16 @@ impl DiskLocation {
|
||||
if shard_count == 0 {
|
||||
return false;
|
||||
}
|
||||
if let (Some(actual), Some(expected)) = (actual_shard_size, expected_shard_size) {
|
||||
if actual < expected {
|
||||
warn!(volume_id = vid.0, actual, expected, "shards smaller than the .dat's full encode; reclaiming the complete .dat");
|
||||
return false;
|
||||
}
|
||||
if let (Some(actual), Some(expected)) = (actual_shard_size, expected_shard_size)
|
||||
&& actual < expected
|
||||
{
|
||||
warn!(
|
||||
volume_id = vid.0,
|
||||
actual,
|
||||
expected,
|
||||
"shards smaller than the .dat's full encode; reclaiming the complete .dat"
|
||||
);
|
||||
return false;
|
||||
}
|
||||
true
|
||||
}
|
||||
@@ -510,10 +525,10 @@ impl DiskLocation {
|
||||
pub(crate) fn ec_generation_ts_ns(&self, collection: &str, vid: VolumeId) -> Option<i64> {
|
||||
for dir in [&self.directory, &self.idx_directory] {
|
||||
let vif = format!("{}.vif", volume_file_name(dir, collection, vid));
|
||||
if let Ok(s) = fs::read_to_string(&vif) {
|
||||
if let Ok(vi) = serde_json::from_str::<VifVolumeInfo>(&s) {
|
||||
return Some(vi.ec_shard_config.map(|c| c.encode_ts_ns).unwrap_or(0));
|
||||
}
|
||||
if let Ok(s) = fs::read_to_string(&vif)
|
||||
&& let Ok(vi) = serde_json::from_str::<VifVolumeInfo>(&s)
|
||||
{
|
||||
return Some(vi.ec_shard_config.map(|c| c.encode_ts_ns).unwrap_or(0));
|
||||
}
|
||||
if self.directory == self.idx_directory {
|
||||
break;
|
||||
@@ -545,27 +560,19 @@ impl DiskLocation {
|
||||
pub fn create_volume(
|
||||
&mut self,
|
||||
vid: VolumeId,
|
||||
collection: &str,
|
||||
needle_map_kind: NeedleMapKind,
|
||||
replica_placement: Option<ReplicaPlacement>,
|
||||
ttl: Option<crate::storage::needle::ttl::TTL>,
|
||||
preallocate: u64,
|
||||
version: Version,
|
||||
spec: &VolumeSpec<'_>,
|
||||
) -> Result<(), VolumeError> {
|
||||
let mut v = Volume::new(
|
||||
&self.directory,
|
||||
&self.idx_directory,
|
||||
collection,
|
||||
vid,
|
||||
needle_map_kind,
|
||||
replica_placement,
|
||||
ttl,
|
||||
preallocate,
|
||||
version,
|
||||
spec,
|
||||
)?;
|
||||
v.location_disk_space_low = self.is_disk_space_low.clone();
|
||||
crate::metrics::VOLUME_GAUGE
|
||||
.with_label_values(&[collection, "volume"])
|
||||
.with_label_values(&[spec.collection, "volume"])
|
||||
.inc();
|
||||
self.volumes.insert(vid, v);
|
||||
Ok(())
|
||||
@@ -669,8 +676,7 @@ impl DiskLocation {
|
||||
pub fn free_volume_count(&self) -> i32 {
|
||||
use crate::storage::erasure_coding::ec_shard::DATA_SHARDS_COUNT;
|
||||
let max = self.max_volume_count.load(Ordering::Relaxed);
|
||||
let free_count = (max as i64 - self.volumes.len() as i64)
|
||||
* DATA_SHARDS_COUNT as i64
|
||||
let free_count = (max as i64 - self.volumes.len() as i64) * DATA_SHARDS_COUNT as i64
|
||||
- self.ec_shard_count() as i64;
|
||||
let effective_free = free_count / DATA_SHARDS_COUNT as i64;
|
||||
if effective_free > 0 {
|
||||
@@ -777,18 +783,18 @@ impl DiskLocation {
|
||||
pub fn has_ecx_file_on_disk(&self, collection: &str, vid: VolumeId) -> bool {
|
||||
let idx_base = volume_file_name(&self.idx_directory, collection, vid);
|
||||
let idx_path = format!("{}.ecx", idx_base);
|
||||
if let Ok(meta) = fs::metadata(&idx_path) {
|
||||
if !meta.is_dir() {
|
||||
return true;
|
||||
}
|
||||
if let Ok(meta) = fs::metadata(&idx_path)
|
||||
&& !meta.is_dir()
|
||||
{
|
||||
return true;
|
||||
}
|
||||
if self.idx_directory != self.directory {
|
||||
let data_base = volume_file_name(&self.directory, collection, vid);
|
||||
let data_path = format!("{}.ecx", data_base);
|
||||
if let Ok(meta) = fs::metadata(&data_path) {
|
||||
if !meta.is_dir() {
|
||||
return true;
|
||||
}
|
||||
if let Ok(meta) = fs::metadata(&data_path)
|
||||
&& !meta.is_dir()
|
||||
{
|
||||
return true;
|
||||
}
|
||||
}
|
||||
false
|
||||
@@ -811,7 +817,7 @@ impl DiskLocation {
|
||||
&mut self,
|
||||
vid: VolumeId,
|
||||
collection: &str,
|
||||
shard_ids: &[u32],
|
||||
shard_ids: &[ShardId],
|
||||
source_disk_type: &str,
|
||||
) -> Result<(), VolumeError> {
|
||||
let idx_dir = self.idx_directory.clone();
|
||||
@@ -833,7 +839,7 @@ impl DiskLocation {
|
||||
&mut self,
|
||||
vid: VolumeId,
|
||||
collection: &str,
|
||||
shard_ids: &[u32],
|
||||
shard_ids: &[ShardId],
|
||||
idx_dir: &str,
|
||||
source_disk_type: &str,
|
||||
) -> Result<(), VolumeError> {
|
||||
@@ -845,14 +851,10 @@ impl DiskLocation {
|
||||
// propagate the error to the caller.
|
||||
let created = !self.ec_volumes.contains_key(&vid);
|
||||
if created {
|
||||
let ec_vol = EcVolume::new(&dir, idx_dir, collection, vid)
|
||||
.map_err(VolumeError::Io)?;
|
||||
let ec_vol = EcVolume::new(&dir, idx_dir, collection, vid).map_err(VolumeError::Io)?;
|
||||
self.ec_volumes.insert(vid, ec_vol);
|
||||
}
|
||||
let ec_vol = self
|
||||
.ec_volumes
|
||||
.get_mut(&vid)
|
||||
.expect("just inserted above");
|
||||
let ec_vol = self.ec_volumes.get_mut(&vid).expect("just inserted above");
|
||||
// When the orchestrator supplied a source disk type on the Mount
|
||||
// RPC, override the EC volume's disk type so heartbeats report
|
||||
// under the source volume's disk type (#9423). When the caller
|
||||
@@ -871,10 +873,10 @@ impl DiskLocation {
|
||||
// keep the existing registration (mirrors Go's AddEcVolumeShard
|
||||
// added=false) — re-adding would replace a serving fd and bump
|
||||
// the ec_shards gauge without growing the mounted count.
|
||||
if ec_vol.has_shard(shard_id as u8) {
|
||||
if ec_vol.has_shard(shard_id) {
|
||||
continue;
|
||||
}
|
||||
let mut shard = EcVolumeShard::new(&dir, collection, vid, shard_id as u8);
|
||||
let mut shard = EcVolumeShard::new(&dir, collection, vid, shard_id);
|
||||
shard.disk_type = ec_vol.disk_type.clone();
|
||||
if let Err(e) = ec_vol.add_shard(shard) {
|
||||
// The shard was dropped (its descriptors closed) inside the
|
||||
@@ -902,14 +904,14 @@ impl DiskLocation {
|
||||
/// caller passes a shard that lives on a sibling disk
|
||||
/// (cross-disk reconcile makes that the common case for the same
|
||||
/// `vid` after reconciliation).
|
||||
pub fn unmount_ec_shards(&mut self, vid: VolumeId, shard_ids: &[u32]) {
|
||||
pub fn unmount_ec_shards(&mut self, vid: VolumeId, shard_ids: &[ShardId]) {
|
||||
if let Some(ec_vol) = self.ec_volumes.get_mut(&vid) {
|
||||
let collection = ec_vol.collection.clone();
|
||||
for &shard_id in shard_ids {
|
||||
if !ec_vol.has_shard(shard_id as u8) {
|
||||
if !ec_vol.has_shard(shard_id) {
|
||||
continue;
|
||||
}
|
||||
ec_vol.remove_shard(shard_id as u8);
|
||||
let _ = ec_vol.remove_shard(shard_id);
|
||||
crate::metrics::VOLUME_GAUGE
|
||||
.with_label_values(&[&collection, "ec_shards"])
|
||||
.dec();
|
||||
@@ -969,7 +971,7 @@ impl DiskLocation {
|
||||
}
|
||||
entries.sort();
|
||||
|
||||
let mut same_volume_shards: Vec<(String, u32)> = Vec::new(); // (filename, shard_id)
|
||||
let mut same_volume_shards: Vec<(String, ShardId)> = Vec::new(); // (filename, shard_id)
|
||||
let mut prev_vid: Option<VolumeId> = None;
|
||||
let mut prev_collection: String = String::new();
|
||||
|
||||
@@ -1034,7 +1036,12 @@ impl DiskLocation {
|
||||
/// Validate + mount a (collection, vid) group when its `.ecx` is
|
||||
/// found. Mirrors `handleFoundEcxFile` in
|
||||
/// `weed/storage/disk_location_ec.go`.
|
||||
fn handle_found_ecx_file(&mut self, shards: &[(String, u32)], collection: &str, vid: VolumeId) {
|
||||
fn handle_found_ecx_file(
|
||||
&mut self,
|
||||
shards: &[(String, ShardId)],
|
||||
collection: &str,
|
||||
vid: VolumeId,
|
||||
) {
|
||||
let base = volume_file_name(&self.directory, collection, vid);
|
||||
let dat_path = format!("{}.dat", base);
|
||||
let dat_exists = check_dat_file_exists(&dat_path);
|
||||
@@ -1048,7 +1055,7 @@ impl DiskLocation {
|
||||
return;
|
||||
}
|
||||
|
||||
let shard_ids: Vec<u32> = shards.iter().map(|(_, sid)| *sid).collect();
|
||||
let shard_ids: Vec<ShardId> = shards.iter().map(|(_, sid)| *sid).collect();
|
||||
if let Err(e) = self.mount_ec_shards(vid, collection, &shard_ids, "") {
|
||||
// A mount failure (corrupt/locked .ecx, EMFILE, transient I/O) is
|
||||
// not proof the shards are disposable -- validate_ec_volume already
|
||||
@@ -1057,8 +1064,7 @@ impl DiskLocation {
|
||||
// delete on a load error.
|
||||
warn!(
|
||||
volume_id = vid.0,
|
||||
"Failed to load EC shards: {}; keeping files for retry",
|
||||
e,
|
||||
"Failed to load EC shards: {}; keeping files for retry", e,
|
||||
);
|
||||
self.unmount_ec_shards(vid, &shard_ids);
|
||||
}
|
||||
@@ -1071,7 +1077,7 @@ impl DiskLocation {
|
||||
/// distributed-EC shards waiting for cross-disk reconciliation.
|
||||
fn check_orphaned_shards(
|
||||
&self,
|
||||
shards: &[(String, u32)],
|
||||
shards: &[(String, ShardId)],
|
||||
collection: &str,
|
||||
vid: VolumeId,
|
||||
) -> bool {
|
||||
@@ -1107,7 +1113,7 @@ impl DiskLocation {
|
||||
|
||||
/// Close all volumes.
|
||||
pub fn close(&mut self) {
|
||||
for (_, v) in self.volumes.iter_mut() {
|
||||
for v in self.volumes.values_mut() {
|
||||
v.close();
|
||||
}
|
||||
self.volumes.clear();
|
||||
@@ -1137,10 +1143,45 @@ pub fn get_disk_stats(path: &str) -> (u64, u64) {
|
||||
}
|
||||
(0, 0)
|
||||
}
|
||||
#[cfg(not(unix))]
|
||||
#[cfg(windows)]
|
||||
{
|
||||
let _ = path;
|
||||
(0, 0)
|
||||
use std::os::windows::ffi::OsStrExt;
|
||||
|
||||
// Canonicalize so symlinks, `.`/`..` segments, and relative paths
|
||||
// resolve to the real location before querying. `\\?\`-prefixed
|
||||
// extended-length paths and UNC (`\\?\UNC\...`) are passed through
|
||||
// untouched: GetDiskFreeSpaceExW accepts them as-is.
|
||||
let canonical = match std::fs::canonicalize(path) {
|
||||
Ok(p) => p,
|
||||
Err(_) => return (0, 0),
|
||||
};
|
||||
// UTF-16 with trailing NUL for the Win32 wide-string call.
|
||||
let mut wide: Vec<u16> = canonical.as_os_str().encode_wide().collect();
|
||||
// UNC directory names must end in a backslash for GetDiskFreeSpaceExW.
|
||||
if !wide.ends_with(&[0x5C]) {
|
||||
wide.push(0x5C);
|
||||
}
|
||||
wide.push(0);
|
||||
// SAFETY: `wide` is NUL-terminated; the out-params are valid u64
|
||||
// writes; the call has no other preconditions.
|
||||
unsafe {
|
||||
let mut free_available: u64 = 0;
|
||||
let mut total: u64 = 0;
|
||||
let ok = windows_sys::Win32::Storage::FileSystem::GetDiskFreeSpaceExW(
|
||||
wide.as_ptr(),
|
||||
&mut free_available,
|
||||
&mut total,
|
||||
std::ptr::null_mut(),
|
||||
);
|
||||
if ok == 0 {
|
||||
return (0, 0);
|
||||
}
|
||||
return (total, free_available);
|
||||
}
|
||||
}
|
||||
#[cfg(not(any(unix, windows)))]
|
||||
{
|
||||
compile_error!("get_disk_stats is implemented for unix and windows only");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1176,7 +1217,12 @@ fn rm_if_present(path: String) -> io::Result<()> {
|
||||
}
|
||||
}
|
||||
|
||||
fn ec_data_shards_from_vif(directory: &str, idx_directory: &str, collection: &str, vid: VolumeId) -> usize {
|
||||
fn ec_data_shards_from_vif(
|
||||
directory: &str,
|
||||
idx_directory: &str,
|
||||
collection: &str,
|
||||
vid: VolumeId,
|
||||
) -> usize {
|
||||
for dir in [directory, idx_directory] {
|
||||
let vif = format!("{}.vif", volume_file_name(dir, collection, vid));
|
||||
if let Some(ds) = fs::read_to_string(&vif)
|
||||
@@ -1184,10 +1230,9 @@ fn ec_data_shards_from_vif(directory: &str, idx_directory: &str, collection: &st
|
||||
.and_then(|s| serde_json::from_str::<VifVolumeInfo>(&s).ok())
|
||||
.and_then(|vi| vi.ec_shard_config)
|
||||
.map(|c| c.data_shards as usize)
|
||||
&& ds > 0
|
||||
{
|
||||
if ds > 0 {
|
||||
return ds;
|
||||
}
|
||||
return ds;
|
||||
}
|
||||
if directory == idx_directory {
|
||||
break;
|
||||
@@ -1223,7 +1268,7 @@ fn parse_collection_volume_id(base: &str) -> Option<(String, VolumeId)> {
|
||||
|
||||
/// `pub(crate)` re-export of [`parse_ec_shard_extension`] for the
|
||||
/// cross-disk reconcile in `store_ec_reconcile.rs`.
|
||||
pub(crate) fn is_ec_shard_extension(ext: &str) -> Option<u32> {
|
||||
pub(crate) fn is_ec_shard_extension(ext: &str) -> Option<ShardId> {
|
||||
parse_ec_shard_extension(ext)
|
||||
}
|
||||
|
||||
@@ -1237,7 +1282,7 @@ pub(crate) fn is_ec_shard_extension(ext: &str) -> Option<u32> {
|
||||
/// shardId > 255` guard. The 3-digit form (`.ec100`–`.ec255`) is
|
||||
/// retained so the parser can still recognise shards from custom
|
||||
/// 32+ ratios that fit in a u8 even though OSS only ships 10+4.
|
||||
fn parse_ec_shard_extension(ext: &str) -> Option<u32> {
|
||||
fn parse_ec_shard_extension(ext: &str) -> Option<ShardId> {
|
||||
let rest = ext.strip_prefix(".ec")?;
|
||||
if rest.len() < 2 || rest.len() > 3 {
|
||||
return None;
|
||||
@@ -1246,7 +1291,7 @@ fn parse_ec_shard_extension(ext: &str) -> Option<u32> {
|
||||
if id > 255 {
|
||||
return None;
|
||||
}
|
||||
Some(id)
|
||||
ShardId::try_from(id).ok()
|
||||
}
|
||||
|
||||
/// Robust check that a `.dat` with actual data exists. An empty `.dat`
|
||||
@@ -1265,7 +1310,7 @@ fn check_dat_file_exists(path: &str) -> bool {
|
||||
/// True when a `.vif` references remote-tier files: a remote-only volume
|
||||
/// that has no local `.dat` but must still load via the remote path,
|
||||
/// rather than be skipped as a lone EC sidecar.
|
||||
fn vif_references_remote_file(vif_path: &str) -> bool {
|
||||
pub(crate) fn vif_references_remote_file(vif_path: &str) -> bool {
|
||||
fs::read_to_string(vif_path)
|
||||
.ok()
|
||||
.and_then(|s| serde_json::from_str::<VifVolumeInfo>(&s).ok())
|
||||
@@ -1308,7 +1353,10 @@ fn remove_empty_ec_dat_stub(volume_name: &str, idx_name: &str, vid: VolumeId) ->
|
||||
return false;
|
||||
}
|
||||
|
||||
warn!(volume_id = vid.0, "removing leftover empty .dat stub for EC volume");
|
||||
warn!(
|
||||
volume_id = vid.0,
|
||||
"removing leftover empty .dat stub for EC volume"
|
||||
);
|
||||
let _ = fs::remove_file(&dat_path);
|
||||
let _ = fs::remove_file(format!("{}.idx", idx_name));
|
||||
true
|
||||
@@ -1331,6 +1379,17 @@ mod tests {
|
||||
use super::*;
|
||||
use tempfile::TempDir;
|
||||
|
||||
/// get_disk_stats must report real capacity for a real path on every
|
||||
/// platform (Windows included) — consumers treat total==0 as "unknown"
|
||||
/// and leave available_space at 0, which breaks volume assignment.
|
||||
#[test]
|
||||
fn test_get_disk_stats_reports_capacity_for_real_path() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let (total, free) = get_disk_stats(tmp.path().to_str().unwrap());
|
||||
assert!(total > 0, "expected total>0, got {total}");
|
||||
assert!(free > 0, "expected free>0, got {free}");
|
||||
}
|
||||
|
||||
/// When `-dir.idx` is configured the EC `.vif` may live in the idx
|
||||
/// directory; the sweep must look there too, not only the data dir.
|
||||
#[test]
|
||||
@@ -1353,7 +1412,11 @@ mod tests {
|
||||
}),
|
||||
..Default::default()
|
||||
};
|
||||
std::fs::write(format!("{}.vif", ibase), serde_json::to_string(&vif).unwrap()).unwrap();
|
||||
std::fs::write(
|
||||
format!("{}.vif", ibase),
|
||||
serde_json::to_string(&vif).unwrap(),
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
assert!(
|
||||
remove_empty_ec_dat_stub(&vbase, &ibase, VolumeId(42)),
|
||||
@@ -1369,16 +1432,30 @@ mod tests {
|
||||
fn test_validate_ec_volume_partial_dat_next_to_full_shards_keeps() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
let loc = DiskLocation::new(dir, dir, 10, DiskType::HardDrive, MinFreeSpace::Percent(1.0), Vec::new()).unwrap();
|
||||
let loc = DiskLocation::new(
|
||||
dir,
|
||||
dir,
|
||||
10,
|
||||
DiskType::HardDrive,
|
||||
MinFreeSpace::Percent(1.0),
|
||||
Vec::new(),
|
||||
)
|
||||
.unwrap();
|
||||
let base = volume_file_name(dir, "", VolumeId(70));
|
||||
let ds = crate::storage::erasure_coding::ec_shard::DATA_SHARDS_COUNT;
|
||||
let full = calculate_expected_shard_size(30 * 1024 * 1024, ds);
|
||||
for i in 0..ds {
|
||||
std::fs::File::create(format!("{}.ec{:02}", base, i)).unwrap().set_len(full as u64).unwrap();
|
||||
std::fs::File::create(format!("{}.ec{:02}", base, i))
|
||||
.unwrap()
|
||||
.set_len(full as u64)
|
||||
.unwrap();
|
||||
}
|
||||
// Partial .dat: bigger than a superblock so it is not swept as a stub,
|
||||
// but smaller than what these shards encode.
|
||||
std::fs::File::create(format!("{}.dat", base)).unwrap().set_len(5 * 1024 * 1024).unwrap();
|
||||
std::fs::File::create(format!("{}.dat", base))
|
||||
.unwrap()
|
||||
.set_len(5 * 1024 * 1024)
|
||||
.unwrap();
|
||||
assert!(
|
||||
loc.validate_ec_volume("", VolumeId(70)),
|
||||
"full-size shards beside a smaller (stale/partial) .dat must be kept",
|
||||
@@ -1392,15 +1469,29 @@ mod tests {
|
||||
fn test_validate_ec_volume_interrupted_encode_reclaims() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
let loc = DiskLocation::new(dir, dir, 10, DiskType::HardDrive, MinFreeSpace::Percent(1.0), Vec::new()).unwrap();
|
||||
let loc = DiskLocation::new(
|
||||
dir,
|
||||
dir,
|
||||
10,
|
||||
DiskType::HardDrive,
|
||||
MinFreeSpace::Percent(1.0),
|
||||
Vec::new(),
|
||||
)
|
||||
.unwrap();
|
||||
let base = volume_file_name(dir, "", VolumeId(71));
|
||||
let ds = crate::storage::erasure_coding::ec_shard::DATA_SHARDS_COUNT;
|
||||
let dat_size = 30 * 1024 * 1024i64;
|
||||
std::fs::File::create(format!("{}.dat", base)).unwrap().set_len(dat_size as u64).unwrap();
|
||||
std::fs::File::create(format!("{}.dat", base))
|
||||
.unwrap()
|
||||
.set_len(dat_size as u64)
|
||||
.unwrap();
|
||||
let partial = calculate_expected_shard_size(dat_size, ds) / 3;
|
||||
assert!(partial > 0);
|
||||
for i in 0..ds {
|
||||
std::fs::File::create(format!("{}.ec{:02}", base, i)).unwrap().set_len(partial as u64).unwrap();
|
||||
std::fs::File::create(format!("{}.ec{:02}", base, i))
|
||||
.unwrap()
|
||||
.set_len(partial as u64)
|
||||
.unwrap();
|
||||
}
|
||||
assert!(
|
||||
!loc.validate_ec_volume("", VolumeId(71)),
|
||||
@@ -1445,7 +1536,11 @@ mod tests {
|
||||
}),
|
||||
..Default::default()
|
||||
};
|
||||
std::fs::write(format!("{}.vif", dbase), serde_json::to_string(&with_gen).unwrap()).unwrap();
|
||||
std::fs::write(
|
||||
format!("{}.vif", dbase),
|
||||
serde_json::to_string(&with_gen).unwrap(),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(loc.ec_generation_ts_ns("", vid), Some(4242));
|
||||
|
||||
// A .vif with no EC config reads as generation 0 (recovered/pre-upgrade live volume).
|
||||
@@ -1454,12 +1549,20 @@ mod tests {
|
||||
version: 3,
|
||||
..Default::default()
|
||||
};
|
||||
std::fs::write(format!("{}.vif", dbase), serde_json::to_string(&no_cfg).unwrap()).unwrap();
|
||||
std::fs::write(
|
||||
format!("{}.vif", dbase),
|
||||
serde_json::to_string(&no_cfg).unwrap(),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(loc.ec_generation_ts_ns("", vid), Some(0));
|
||||
|
||||
// idx-dir fallback: only the idx dir holds the .vif.
|
||||
std::fs::remove_file(format!("{}.vif", dbase)).unwrap();
|
||||
std::fs::write(format!("{}.vif", ibase), serde_json::to_string(&with_gen).unwrap()).unwrap();
|
||||
std::fs::write(
|
||||
format!("{}.vif", ibase),
|
||||
serde_json::to_string(&with_gen).unwrap(),
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(loc.ec_generation_ts_ns("", vid), Some(4242));
|
||||
}
|
||||
|
||||
@@ -1499,16 +1602,8 @@ mod tests {
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
loc.create_volume(
|
||||
VolumeId(1),
|
||||
"",
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
)
|
||||
.unwrap();
|
||||
loc.create_volume(VolumeId(1), NeedleMapKind::InMemory, &VolumeSpec::default())
|
||||
.unwrap();
|
||||
|
||||
assert_eq!(loc.volumes_len(), 1);
|
||||
assert!(loc.find_volume(VolumeId(1)).is_some());
|
||||
@@ -1532,24 +1627,15 @@ mod tests {
|
||||
Vec::new(),
|
||||
)
|
||||
.unwrap();
|
||||
loc.create_volume(
|
||||
VolumeId(1),
|
||||
"",
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
)
|
||||
.unwrap();
|
||||
loc.create_volume(VolumeId(1), NeedleMapKind::InMemory, &VolumeSpec::default())
|
||||
.unwrap();
|
||||
loc.create_volume(
|
||||
VolumeId(2),
|
||||
"test",
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
&VolumeSpec {
|
||||
collection: "test",
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
loc.close();
|
||||
@@ -1592,12 +1678,11 @@ mod tests {
|
||||
.unwrap();
|
||||
loc.create_volume(
|
||||
VolumeId(9),
|
||||
"good",
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
&VolumeSpec {
|
||||
collection: "good",
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
loc.close();
|
||||
@@ -1641,26 +1726,10 @@ mod tests {
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
loc.create_volume(
|
||||
VolumeId(1),
|
||||
"",
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
)
|
||||
.unwrap();
|
||||
loc.create_volume(
|
||||
VolumeId(2),
|
||||
"",
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
)
|
||||
.unwrap();
|
||||
loc.create_volume(VolumeId(1), NeedleMapKind::InMemory, &VolumeSpec::default())
|
||||
.unwrap();
|
||||
loc.create_volume(VolumeId(2), NeedleMapKind::InMemory, &VolumeSpec::default())
|
||||
.unwrap();
|
||||
assert_eq!(loc.volumes_len(), 2);
|
||||
|
||||
loc.delete_volume(VolumeId(1), false, false).unwrap();
|
||||
@@ -1684,32 +1753,29 @@ mod tests {
|
||||
|
||||
loc.create_volume(
|
||||
VolumeId(1),
|
||||
"pics",
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
&VolumeSpec {
|
||||
collection: "pics",
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
loc.create_volume(
|
||||
VolumeId(2),
|
||||
"pics",
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
&VolumeSpec {
|
||||
collection: "pics",
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
loc.create_volume(
|
||||
VolumeId(3),
|
||||
"docs",
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
&VolumeSpec {
|
||||
collection: "docs",
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
assert_eq!(loc.volumes_len(), 3);
|
||||
@@ -1772,7 +1838,8 @@ mod tests {
|
||||
// mount_ec_shards with source_disk_type="ssd" — simulating the
|
||||
// VolumeEcShardsMount RPC path.
|
||||
std::fs::write(format!("{}/pics_7.ec00", dir), b"ec-shard").unwrap();
|
||||
loc.mount_ec_shards(VolumeId(7), "pics", &[0], "ssd").unwrap();
|
||||
loc.mount_ec_shards(VolumeId(7), "pics", &[0], "ssd")
|
||||
.unwrap();
|
||||
{
|
||||
let ec_vol = loc.find_ec_volume(VolumeId(7)).expect("ec volume mounted");
|
||||
assert_eq!(
|
||||
@@ -1789,7 +1856,9 @@ mod tests {
|
||||
std::fs::write(format!("{}/pics_7.ec01", dir), b"ec-shard").unwrap();
|
||||
loc.mount_ec_shards(VolumeId(7), "pics", &[1], "").unwrap();
|
||||
{
|
||||
let ec_vol = loc.find_ec_volume(VolumeId(7)).expect("ec volume still mounted");
|
||||
let ec_vol = loc
|
||||
.find_ec_volume(VolumeId(7))
|
||||
.expect("ec volume still mounted");
|
||||
assert_eq!(
|
||||
ec_vol.disk_type,
|
||||
DiskType::Ssd,
|
||||
@@ -1871,7 +1940,8 @@ mod tests {
|
||||
let gauge = crate::metrics::VOLUME_GAUGE.with_label_values(&["dupmount", "ec_shards"]);
|
||||
let before = gauge.get();
|
||||
|
||||
loc.mount_ec_shards(VolumeId(11), "dupmount", &[0], "").unwrap();
|
||||
loc.mount_ec_shards(VolumeId(11), "dupmount", &[0], "")
|
||||
.unwrap();
|
||||
loc.mount_ec_shards(VolumeId(11), "dupmount", &[0], "")
|
||||
.expect("a duplicate mount must succeed as a no-op");
|
||||
|
||||
@@ -1947,8 +2017,11 @@ mod tests {
|
||||
let path = format!("{}/{}_{}.ec{:02}", dir, collection, vid.0, sid);
|
||||
std::fs::write(&path, b"shard data nonempty").unwrap();
|
||||
}
|
||||
std::fs::write(format!("{}/{}_{}.ecx", dir, collection, vid.0), vec![0u8; 20])
|
||||
.unwrap();
|
||||
std::fs::write(
|
||||
format!("{}/{}_{}.ecx", dir, collection, vid.0),
|
||||
vec![0u8; 20],
|
||||
)
|
||||
.unwrap();
|
||||
std::fs::write(format!("{}/{}_{}.ecj", dir, collection, vid.0), b"").unwrap();
|
||||
std::fs::write(
|
||||
format!("{}/{}_{}.vif", dir, collection, vid.0),
|
||||
|
||||
@@ -30,6 +30,7 @@ use crate::pb::volume_server_pb::{
|
||||
ChecksumAlgorithm, EcBitrotProtection, EcShardChecksums, EcShardConfig,
|
||||
};
|
||||
use crate::storage::erasure_coding::ec_shard::MAX_SHARD_COUNT;
|
||||
use crate::storage::io::read_exact_at;
|
||||
use crate::storage::needle::crc::CRC;
|
||||
|
||||
/// Canonical extension for the checksum sidecar. Generation 0 (legacy/fresh
|
||||
@@ -164,16 +165,16 @@ pub fn remove_bitrot_sidecars(base: &str) -> io::Result<()> {
|
||||
};
|
||||
let mut first_err: Option<io::Error> = None;
|
||||
let mut record = |res: io::Result<()>| {
|
||||
if let Err(e) = res {
|
||||
if first_err.is_none() {
|
||||
first_err = Some(e);
|
||||
}
|
||||
if let Err(e) = res
|
||||
&& first_err.is_none()
|
||||
{
|
||||
first_err = Some(e);
|
||||
}
|
||||
};
|
||||
record(rm(format!("{}{}", base, BITROT_SIDECAR_EXT).into()));
|
||||
let path = Path::new(base);
|
||||
if let (Some(parent), Some(fname)) = (path.parent(), path.file_name()) {
|
||||
let prefix = format!("{}{}.v", fname.to_string_lossy(), BITROT_SIDECAR_EXT);
|
||||
let prefix = format!("{}{}.v", fname.display(), BITROT_SIDECAR_EXT);
|
||||
match fs::read_dir(parent) {
|
||||
Ok(entries) => {
|
||||
for entry in entries.flatten() {
|
||||
@@ -203,7 +204,7 @@ pub fn new_encode_uuid() -> Vec<u8> {
|
||||
|
||||
/// Reports whether `block_size` is a power of two in [1 MiB, MAX_BITROT_BLOCK_SIZE].
|
||||
pub fn is_pow2_multiple_of_1mib(block_size: u32) -> bool {
|
||||
block_size >= (1 << 20) && block_size <= MAX_BITROT_BLOCK_SIZE && block_size.count_ones() == 1
|
||||
((1 << 20)..=MAX_BITROT_BLOCK_SIZE).contains(&block_size) && block_size.count_ones() == 1
|
||||
}
|
||||
|
||||
/// Returns ceil(covered_size / block_size).
|
||||
@@ -402,7 +403,7 @@ pub fn validate_manifest(
|
||||
total
|
||||
));
|
||||
}
|
||||
let mut seen = vec![false; MAX_SHARD_COUNT];
|
||||
let mut seen = [false; MAX_SHARD_COUNT];
|
||||
for s in &prot.shards {
|
||||
if s.shard_id >= total as u32 {
|
||||
return Err(format!(
|
||||
@@ -505,7 +506,21 @@ pub fn verify_shard_file_blocks(
|
||||
entry: &EcShardChecksums,
|
||||
block_size: i64,
|
||||
) -> io::Result<Vec<usize>> {
|
||||
let f = File::open(path)?;
|
||||
verify_shard_blocks(&File::open(path)?, entry, block_size)
|
||||
}
|
||||
|
||||
/// Same verification against an ALREADY-OPEN shard handle.
|
||||
///
|
||||
/// Go's `ChecksumScrub` reads through `shard.ReadAt`, i.e. the handle the
|
||||
/// EcVolumeShard already holds, so a concurrent teardown that unlinks the shard
|
||||
/// cannot turn an intentional removal into a scrub read error. A scrub that
|
||||
/// runs with the store lock released has to read the same way — see
|
||||
/// `EcChecksumScrubPlan`.
|
||||
pub fn verify_shard_blocks(
|
||||
f: &File,
|
||||
entry: &EcShardChecksums,
|
||||
block_size: i64,
|
||||
) -> io::Result<Vec<usize>> {
|
||||
let file_size = f.metadata()?.len() as i64;
|
||||
let want = unpack_u32_le(&entry.block_crc32c);
|
||||
|
||||
@@ -523,7 +538,7 @@ pub fn verify_shard_file_blocks(
|
||||
break;
|
||||
}
|
||||
let to_read = to_read as usize;
|
||||
read_full_at(&f, &mut buf[..to_read], offset as u64)?;
|
||||
read_exact_at(f, &mut buf[..to_read], offset as u64)?;
|
||||
if CRC::new(&buf[..to_read]).0 != *want_crc {
|
||||
mismatched.push(i);
|
||||
}
|
||||
@@ -532,33 +547,6 @@ pub fn verify_shard_file_blocks(
|
||||
Ok(mismatched)
|
||||
}
|
||||
|
||||
/// Reads exactly `buf.len()` bytes from `f` at `offset`, erroring on early EOF.
|
||||
fn read_full_at(f: &File, buf: &mut [u8], offset: u64) -> io::Result<()> {
|
||||
let mut total = 0usize;
|
||||
while total < buf.len() {
|
||||
#[cfg(unix)]
|
||||
let n = {
|
||||
use std::os::unix::fs::FileExt;
|
||||
f.read_at(&mut buf[total..], offset + total as u64)?
|
||||
};
|
||||
#[cfg(not(unix))]
|
||||
let n = {
|
||||
use std::io::{Read, Seek, SeekFrom};
|
||||
let mut fc = f.try_clone()?;
|
||||
fc.seek(SeekFrom::Start(offset + total as u64))?;
|
||||
fc.read(&mut buf[total..])?
|
||||
};
|
||||
if n == 0 {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::UnexpectedEof,
|
||||
"short read on shard block",
|
||||
));
|
||||
}
|
||||
total += n;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Builds the `EcShardConfig` proto for the given layout. The bitrot sidecar
|
||||
/// carries its own top-level encode_uuid, so the nested config leaves it empty.
|
||||
pub fn ec_shard_config(data_shards: u32, parity_shards: u32, block_size: i64) -> EcShardConfig {
|
||||
@@ -613,7 +601,10 @@ mod tests {
|
||||
save_bitrot_sidecar(path, &prot).unwrap();
|
||||
let bytes = std::fs::read(path).unwrap();
|
||||
let hex: String = bytes.iter().map(|b| format!("{:02x}", b)).collect();
|
||||
assert_eq!(hex, CANONICAL_HEX, "Rust .ecsum bytes drifted from the Go canonical form");
|
||||
assert_eq!(
|
||||
hex, CANONICAL_HEX,
|
||||
"Rust .ecsum bytes drifted from the Go canonical form"
|
||||
);
|
||||
let _ = std::fs::remove_file(path);
|
||||
}
|
||||
|
||||
@@ -648,7 +639,11 @@ mod tests {
|
||||
format!("{}.ecsum.v1", base),
|
||||
format!("{}.ecsum.v7", base),
|
||||
] {
|
||||
assert!(!std::path::Path::new(&p).exists(), "{} should be removed", p);
|
||||
assert!(
|
||||
!std::path::Path::new(&p).exists(),
|
||||
"{} should be removed",
|
||||
p
|
||||
);
|
||||
}
|
||||
assert!(std::path::Path::new(&keep_shard).exists());
|
||||
assert!(std::path::Path::new(&keep_other_vid).exists());
|
||||
@@ -667,7 +662,9 @@ mod tests {
|
||||
assert!(!is_pow2_multiple_of_1mib(1 << 19)); // 512 KiB, too small
|
||||
assert!(!is_pow2_multiple_of_1mib(3 << 20)); // 3 MiB, not pow2
|
||||
assert!(!is_pow2_multiple_of_1mib(128 * 1024 * 1024)); // pow2 but > MAX_BITROT_BLOCK_SIZE
|
||||
assert!(!is_pow2_multiple_of_1mib(DEFAULT_BITROT_BLOCK_SIZE as u32 + 1));
|
||||
assert!(!is_pow2_multiple_of_1mib(
|
||||
DEFAULT_BITROT_BLOCK_SIZE as u32 + 1
|
||||
));
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -721,12 +718,7 @@ mod tests {
|
||||
#[test]
|
||||
fn test_save_load_roundtrip() {
|
||||
let tmp = tempfile::TempDir::new().unwrap();
|
||||
let path = tmp
|
||||
.path()
|
||||
.join("vol.ecsum")
|
||||
.to_str()
|
||||
.unwrap()
|
||||
.to_string();
|
||||
let path = tmp.path().join("vol.ecsum").to_str().unwrap().to_string();
|
||||
|
||||
let mut builder = ShardChecksumBuilder::new(DEFAULT_BITROT_BLOCK_SIZE as i64);
|
||||
builder.write(b"hello world");
|
||||
@@ -887,8 +879,7 @@ mod tests {
|
||||
assert_eq!(resolve_status(¬found, 0, 10, 4), BitrotStatus::Off);
|
||||
|
||||
// Integrity failure => Invalid.
|
||||
let bad: Result<EcBitrotProtection, BitrotLoadError> =
|
||||
Err(BitrotLoadError::BadMagic(0));
|
||||
let bad: Result<EcBitrotProtection, BitrotLoadError> = Err(BitrotLoadError::BadMagic(0));
|
||||
assert_eq!(resolve_status(&bad, 0, 10, 4), BitrotStatus::Invalid);
|
||||
|
||||
// Generation mismatch => Off.
|
||||
|
||||
@@ -67,72 +67,45 @@ pub fn find_dat_file_size_with_dirs(
|
||||
Ok(dat_size)
|
||||
}
|
||||
|
||||
/// Reconstruct a .dat file from EC data shards.
|
||||
///
|
||||
/// Reads from .ec00-.ec09 and writes a new .dat file. All data shards
|
||||
/// must live in `dir`. For the cross-disk reconciled layout where
|
||||
/// shards are split across multiple data dirs of the same node, use
|
||||
/// [`write_dat_file_from_shards_with_dirs`] instead.
|
||||
pub fn write_dat_file_from_shards(
|
||||
dir: &str,
|
||||
collection: &str,
|
||||
volume_id: VolumeId,
|
||||
dat_file_size: i64,
|
||||
encoded_dat_file_size: i64,
|
||||
data_shards: usize,
|
||||
large_block_size: usize,
|
||||
small_block_size: usize,
|
||||
) -> io::Result<()> {
|
||||
let dirs: Vec<String> = (0..data_shards).map(|_| dir.to_string()).collect();
|
||||
write_dat_file_from_shards_with_dirs(
|
||||
dir,
|
||||
collection,
|
||||
volume_id,
|
||||
dat_file_size,
|
||||
encoded_dat_file_size,
|
||||
data_shards,
|
||||
&dirs,
|
||||
large_block_size,
|
||||
small_block_size,
|
||||
)
|
||||
}
|
||||
|
||||
/// Reconstruct a .dat file from EC data shards, taking the source
|
||||
/// directory for each shard separately.
|
||||
///
|
||||
/// `dat_dir` is where the produced `.dat` is written. `shard_dirs[i]`
|
||||
/// is the directory holding shard `i`. For the simple "all shards in
|
||||
/// one dir" case both can be the same value.
|
||||
/// What it takes to rebuild a volume's .dat from its EC data shards.
|
||||
///
|
||||
/// Mirrors Go's `WriteDatFile(baseFileName, datFileSize,
|
||||
/// encodedDatFileSize, shardFileNames)` shape — Go passes per-shard
|
||||
/// paths so a reconciled volume with shards split across disks of the
|
||||
/// same volume server can still be decoded back to a regular .dat
|
||||
/// (seaweedfs/seaweedfs#9252).
|
||||
#[derive(Clone, Copy, Debug)]
|
||||
pub struct DatRebuild<'a> {
|
||||
/// Where the produced `.dat` is written.
|
||||
pub dat_dir: &'a str,
|
||||
pub collection: &'a str,
|
||||
pub volume_id: VolumeId,
|
||||
/// The number of bytes to write, i.e. the live data extent from
|
||||
/// [`find_dat_file_size`].
|
||||
pub dat_file_size: i64,
|
||||
/// The .dat size at encode time, which fixed the shard block layout:
|
||||
/// deletions can move the live extent below the large-block row
|
||||
/// boundary, and deriving the layout from the shrunk extent would read
|
||||
/// the shards in the wrong block order. Zero when the .vif does not
|
||||
/// record the encode-time size; the layout is then inferred from the
|
||||
/// shard size.
|
||||
pub encoded_dat_file_size: i64,
|
||||
pub data_shards: usize,
|
||||
/// `shard_dirs[i]` is the directory holding shard `i`. `None` means every
|
||||
/// data shard sits in `dat_dir`.
|
||||
pub shard_dirs: Option<&'a [String]>,
|
||||
/// The volume's shard block layout, e.g. `EcVolume::large_block_size()`
|
||||
/// / `small_block_size()` from its .vif EC config.
|
||||
pub large_block_size: usize,
|
||||
pub small_block_size: usize,
|
||||
}
|
||||
|
||||
/// Reconstruct a .dat file from EC data shards.
|
||||
///
|
||||
/// `dat_file_size` is the number of bytes to write, i.e. the live data
|
||||
/// extent from [`find_dat_file_size`]. `encoded_dat_file_size` is the
|
||||
/// .dat size at encode time, which fixed the shard block layout:
|
||||
/// deletions can move the live extent below the large-block row
|
||||
/// boundary, and deriving the layout from the shrunk extent would read
|
||||
/// the shards in the wrong block order. Pass zero when the .vif does
|
||||
/// not record the encode-time size to infer the layout from the shard
|
||||
/// size. `large_block_size`/`small_block_size` are the volume's shard
|
||||
/// block layout, e.g. `EcVolume::large_block_size()` /
|
||||
/// `small_block_size()` from its .vif EC config.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
pub fn write_dat_file_from_shards_with_dirs(
|
||||
dat_dir: &str,
|
||||
collection: &str,
|
||||
volume_id: VolumeId,
|
||||
dat_file_size: i64,
|
||||
encoded_dat_file_size: i64,
|
||||
data_shards: usize,
|
||||
shard_dirs: &[String],
|
||||
large_block_size: usize,
|
||||
small_block_size: usize,
|
||||
) -> io::Result<()> {
|
||||
write_dat_file(
|
||||
/// Reads from .ec00-.ec09 and writes a new .dat file, from one directory or
|
||||
/// from the per-shard directories of a cross-disk reconciled volume.
|
||||
pub fn write_dat_file_from_shards(spec: &DatRebuild<'_>) -> io::Result<()> {
|
||||
let DatRebuild {
|
||||
dat_dir,
|
||||
collection,
|
||||
volume_id,
|
||||
@@ -142,21 +115,15 @@ pub fn write_dat_file_from_shards_with_dirs(
|
||||
shard_dirs,
|
||||
large_block_size,
|
||||
small_block_size,
|
||||
)
|
||||
}
|
||||
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
fn write_dat_file(
|
||||
dat_dir: &str,
|
||||
collection: &str,
|
||||
volume_id: VolumeId,
|
||||
dat_file_size: i64,
|
||||
encoded_dat_file_size: i64,
|
||||
data_shards: usize,
|
||||
shard_dirs: &[String],
|
||||
large_block_size: usize,
|
||||
small_block_size: usize,
|
||||
) -> io::Result<()> {
|
||||
} = *spec;
|
||||
let same_dir: Vec<String>;
|
||||
let shard_dirs: &[String] = match shard_dirs {
|
||||
Some(dirs) => dirs,
|
||||
None => {
|
||||
same_dir = vec![dat_dir.to_string(); data_shards];
|
||||
&same_dir
|
||||
}
|
||||
};
|
||||
if data_shards == 0 {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::InvalidInput,
|
||||
@@ -233,10 +200,10 @@ fn write_dat_file(
|
||||
|
||||
// Read large blocks
|
||||
while encoded_remaining >= large_row_size && remaining > 0 {
|
||||
for i in 0..data_shards {
|
||||
for (i, shard) in shards[..data_shards].iter().enumerate() {
|
||||
let to_write = large_block_size.min(remaining as usize);
|
||||
let mut buf = vec![0u8; to_write];
|
||||
let n = shards[i].read_at(&mut buf, shard_offset)?;
|
||||
let n = shard.read_at(&mut buf, shard_offset)?;
|
||||
if n != to_write {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::UnexpectedEof,
|
||||
@@ -255,10 +222,10 @@ fn write_dat_file(
|
||||
|
||||
// Read small blocks
|
||||
while remaining > 0 {
|
||||
for i in 0..data_shards {
|
||||
for (i, shard) in shards[..data_shards].iter().enumerate() {
|
||||
let to_write = small_block_size.min(remaining as usize);
|
||||
let mut buf = vec![0u8; to_write];
|
||||
let n = shards[i].read_at(&mut buf, shard_offset)?;
|
||||
let n = shard.read_at(&mut buf, shard_offset)?;
|
||||
if n != to_write {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::UnexpectedEof,
|
||||
@@ -324,10 +291,7 @@ pub fn write_idx_file_from_ec_index(
|
||||
// and treat only NotFound as "no journal": Path::exists would also
|
||||
// swallow a permission/IO error and silently skip deletions, which
|
||||
// would resurrect deleted needles as live.
|
||||
let mut idx_file = std::fs::OpenOptions::new()
|
||||
.write(true)
|
||||
.append(true)
|
||||
.open(&tmp_path)?;
|
||||
let mut idx_file = std::fs::OpenOptions::new().append(true).open(&tmp_path)?;
|
||||
match std::fs::read(&ecj_path) {
|
||||
Ok(ecj_data) => {
|
||||
let count = ecj_data.len() / NEEDLE_ID_SIZE;
|
||||
@@ -372,7 +336,7 @@ mod tests {
|
||||
use crate::storage::erasure_coding::ec_encoder;
|
||||
use crate::storage::needle::needle::Needle;
|
||||
use crate::storage::needle_map::NeedleMapKind;
|
||||
use crate::storage::volume::Volume;
|
||||
use crate::storage::volume::{Volume, VolumeSpec};
|
||||
use tempfile::TempDir;
|
||||
|
||||
#[test]
|
||||
@@ -384,13 +348,9 @@ mod tests {
|
||||
let mut v = Volume::new(
|
||||
dir,
|
||||
dir,
|
||||
"",
|
||||
VolumeId(1),
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
&VolumeSpec::default(),
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
@@ -430,16 +390,17 @@ mod tests {
|
||||
std::fs::remove_file(format!("{}/1.idx", dir)).unwrap();
|
||||
|
||||
// Reconstruct from EC shards
|
||||
write_dat_file_from_shards(
|
||||
dir,
|
||||
"",
|
||||
VolumeId(1),
|
||||
original_dat_size as i64,
|
||||
original_dat_size as i64,
|
||||
write_dat_file_from_shards(&DatRebuild {
|
||||
dat_dir: dir,
|
||||
collection: "",
|
||||
volume_id: VolumeId(1),
|
||||
dat_file_size: original_dat_size as i64,
|
||||
encoded_dat_file_size: original_dat_size as i64,
|
||||
data_shards,
|
||||
block_size as usize,
|
||||
block_size as usize,
|
||||
)
|
||||
shard_dirs: None,
|
||||
large_block_size: block_size as usize,
|
||||
small_block_size: block_size as usize,
|
||||
})
|
||||
.unwrap();
|
||||
write_idx_file_from_ec_index(dir, "", VolumeId(1)).unwrap();
|
||||
|
||||
@@ -459,13 +420,9 @@ mod tests {
|
||||
let v2 = Volume::new(
|
||||
dir,
|
||||
dir,
|
||||
"",
|
||||
VolumeId(1),
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
&VolumeSpec::default(),
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
@@ -485,29 +442,29 @@ mod tests {
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
// No shard files exist, so de-striping must fail and publish nothing:
|
||||
// neither the final .dat nor a partial .dat.tmp may remain.
|
||||
let res = write_dat_file_from_shards(
|
||||
dir,
|
||||
"",
|
||||
VolumeId(7),
|
||||
100,
|
||||
100,
|
||||
10,
|
||||
ERASURE_CODING_LARGE_BLOCK_SIZE,
|
||||
ERASURE_CODING_SMALL_BLOCK_SIZE,
|
||||
);
|
||||
let res = write_dat_file_from_shards(&DatRebuild {
|
||||
dat_dir: dir,
|
||||
collection: "",
|
||||
volume_id: VolumeId(7),
|
||||
dat_file_size: 100,
|
||||
encoded_dat_file_size: 100,
|
||||
data_shards: 10,
|
||||
shard_dirs: None,
|
||||
large_block_size: ERASURE_CODING_LARGE_BLOCK_SIZE,
|
||||
small_block_size: ERASURE_CODING_SMALL_BLOCK_SIZE,
|
||||
});
|
||||
assert!(res.is_err());
|
||||
assert!(!std::path::Path::new(&format!("{}/7.dat", dir)).exists());
|
||||
assert!(!std::path::Path::new(&format!("{}/7.dat.tmp", dir)).exists());
|
||||
}
|
||||
|
||||
|
||||
// Decoding when .vif does not record the encode-time size: the layout is
|
||||
// inferred from the shard size, except when that is an exact large-block
|
||||
// multiple and the live extent reaches the ambiguous region.
|
||||
#[test]
|
||||
fn test_write_dat_file_fallback_layout() {
|
||||
use crate::storage::erasure_coding::ec_bitrot::{
|
||||
ShardChecksumBuilder, DEFAULT_BITROT_BLOCK_SIZE,
|
||||
DEFAULT_BITROT_BLOCK_SIZE, ShardChecksumBuilder,
|
||||
};
|
||||
use reed_solomon_erasure::galois_8::ReedSolomon;
|
||||
|
||||
@@ -545,11 +502,13 @@ mod tests {
|
||||
&rs,
|
||||
&mut shards,
|
||||
&mut builders,
|
||||
data_shards,
|
||||
parity_shards,
|
||||
SMALL,
|
||||
LARGE,
|
||||
SMALL,
|
||||
ec_encoder::EcEncodeLayout {
|
||||
data_shards,
|
||||
parity_shards,
|
||||
buffer_size: SMALL,
|
||||
large_block_size: LARGE,
|
||||
small_block_size: SMALL,
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
for shard in &mut shards {
|
||||
@@ -567,7 +526,17 @@ mod tests {
|
||||
-> io::Result<Vec<u8>> {
|
||||
let out = format!("{}/{}", dir, sub);
|
||||
std::fs::create_dir_all(&out).unwrap();
|
||||
write_dat_file(&out, "", VolumeId(1), live, encoded, 10, shard_dirs, LARGE, SMALL)?;
|
||||
write_dat_file_from_shards(&DatRebuild {
|
||||
dat_dir: &out,
|
||||
collection: "",
|
||||
volume_id: VolumeId(1),
|
||||
dat_file_size: live,
|
||||
encoded_dat_file_size: encoded,
|
||||
data_shards: 10,
|
||||
shard_dirs: Some(shard_dirs),
|
||||
large_block_size: LARGE,
|
||||
small_block_size: SMALL,
|
||||
})?;
|
||||
Ok(std::fs::read(format!("{}/1.dat", out)).unwrap())
|
||||
};
|
||||
|
||||
@@ -581,14 +550,20 @@ mod tests {
|
||||
// each shard exactly one large block, indistinguishable from one large row
|
||||
let (dir, shard_dirs, _) = encode("ambig1", large_row_size - 1);
|
||||
let err = decode_to(&dir, "out", large_row_size / 2, 0, &shard_dirs).unwrap_err();
|
||||
assert!(err.to_string().contains("does not identify the block layout"));
|
||||
assert!(
|
||||
err.to_string()
|
||||
.contains("does not identify the block layout")
|
||||
);
|
||||
|
||||
// two-row equivalent: decoding within the agreed prefix still works
|
||||
let (dir, shard_dirs, original) = encode("ambig2", 2 * large_row_size - 1);
|
||||
let decoded = decode_to(&dir, "outa", large_row_size, 0, &shard_dirs).unwrap();
|
||||
assert_eq!(&original[..large_row_size as usize], &decoded[..]);
|
||||
let err = decode_to(&dir, "outb", large_row_size + 1, 0, &shard_dirs).unwrap_err();
|
||||
assert!(err.to_string().contains("does not identify the block layout"));
|
||||
assert!(
|
||||
err.to_string()
|
||||
.contains("does not identify the block layout")
|
||||
);
|
||||
}
|
||||
|
||||
// Decoding after deletions moved the live extent below the large-block row
|
||||
@@ -597,7 +572,7 @@ mod tests {
|
||||
#[test]
|
||||
fn test_write_dat_file_after_tail_deletion() {
|
||||
use crate::storage::erasure_coding::ec_bitrot::{
|
||||
ShardChecksumBuilder, DEFAULT_BITROT_BLOCK_SIZE,
|
||||
DEFAULT_BITROT_BLOCK_SIZE, ShardChecksumBuilder,
|
||||
};
|
||||
use reed_solomon_erasure::galois_8::ReedSolomon;
|
||||
|
||||
@@ -637,11 +612,13 @@ mod tests {
|
||||
&rs,
|
||||
&mut shards,
|
||||
&mut builders,
|
||||
data_shards,
|
||||
parity_shards,
|
||||
SMALL,
|
||||
LARGE,
|
||||
SMALL,
|
||||
ec_encoder::EcEncodeLayout {
|
||||
data_shards,
|
||||
parity_shards,
|
||||
buffer_size: SMALL,
|
||||
large_block_size: LARGE,
|
||||
small_block_size: SMALL,
|
||||
},
|
||||
)
|
||||
.unwrap();
|
||||
for shard in &mut shards {
|
||||
@@ -657,17 +634,17 @@ mod tests {
|
||||
std::fs::create_dir(&out_dir).unwrap();
|
||||
let out = out_dir.to_str().unwrap();
|
||||
let decode = |live_size: i64, encoded_size: i64| -> Vec<u8> {
|
||||
write_dat_file(
|
||||
out,
|
||||
"",
|
||||
VolumeId(1),
|
||||
live_size,
|
||||
encoded_size,
|
||||
write_dat_file_from_shards(&DatRebuild {
|
||||
dat_dir: out,
|
||||
collection: "",
|
||||
volume_id: VolumeId(1),
|
||||
dat_file_size: live_size,
|
||||
encoded_dat_file_size: encoded_size,
|
||||
data_shards,
|
||||
&shard_dirs,
|
||||
LARGE,
|
||||
SMALL,
|
||||
)
|
||||
shard_dirs: Some(&shard_dirs),
|
||||
large_block_size: LARGE,
|
||||
small_block_size: SMALL,
|
||||
})
|
||||
.unwrap();
|
||||
let path = format!("{}/1.dat", out);
|
||||
let decoded = std::fs::read(&path).unwrap();
|
||||
@@ -702,17 +679,19 @@ mod tests {
|
||||
assert_ne!(&original[..(large_row_size / 2) as usize], &control[..]);
|
||||
|
||||
// the live extent can never exceed the encode-time size
|
||||
assert!(write_dat_file(
|
||||
out,
|
||||
"",
|
||||
VolumeId(1),
|
||||
dat_size + 1,
|
||||
dat_size,
|
||||
data_shards,
|
||||
&shard_dirs,
|
||||
LARGE,
|
||||
SMALL,
|
||||
)
|
||||
.is_err());
|
||||
assert!(
|
||||
write_dat_file_from_shards(&DatRebuild {
|
||||
dat_dir: out,
|
||||
collection: "",
|
||||
volume_id: VolumeId(1),
|
||||
dat_file_size: dat_size + 1,
|
||||
encoded_dat_file_size: dat_size,
|
||||
data_shards,
|
||||
shard_dirs: Some(&shard_dirs),
|
||||
large_block_size: LARGE,
|
||||
small_block_size: SMALL,
|
||||
})
|
||||
.is_err()
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -5,16 +5,12 @@
|
||||
|
||||
use std::fs::File;
|
||||
use std::io;
|
||||
#[cfg(not(unix))]
|
||||
use std::io::{Read, Seek, SeekFrom};
|
||||
|
||||
use reed_solomon_erasure::galois_8::ReedSolomon;
|
||||
|
||||
use crate::pb::volume_server_pb::{
|
||||
ChecksumAlgorithm, EcBitrotProtection, EcShardChecksums,
|
||||
};
|
||||
use crate::pb::volume_server_pb::{ChecksumAlgorithm, EcBitrotProtection, EcShardChecksums};
|
||||
use crate::storage::erasure_coding::ec_bitrot::{
|
||||
self, ShardChecksumBuilder, DEFAULT_BITROT_BLOCK_SIZE,
|
||||
self, DEFAULT_BITROT_BLOCK_SIZE, ShardChecksumBuilder,
|
||||
};
|
||||
use crate::storage::erasure_coding::ec_shard::*;
|
||||
use crate::storage::idx;
|
||||
@@ -50,7 +46,7 @@ pub fn write_ec_files(
|
||||
let dat_size = dat_file.metadata()?.len() as i64;
|
||||
|
||||
let rs = ReedSolomon::new(data_shards, parity_shards)
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("reed-solomon init: {:?}", e)))?;
|
||||
.map_err(|e| io::Error::other(format!("reed-solomon init: {:?}", e)))?;
|
||||
|
||||
// Create shard files
|
||||
let total_shards = data_shards + parity_shards;
|
||||
@@ -77,11 +73,13 @@ pub fn write_ec_files(
|
||||
&rs,
|
||||
&mut shards,
|
||||
&mut builders,
|
||||
data_shards,
|
||||
parity_shards,
|
||||
ENCODE_BUFFER_SIZE,
|
||||
block_size as usize,
|
||||
block_size as usize,
|
||||
EcEncodeLayout {
|
||||
data_shards,
|
||||
parity_shards,
|
||||
buffer_size: ENCODE_BUFFER_SIZE,
|
||||
large_block_size: block_size as usize,
|
||||
small_block_size: block_size as usize,
|
||||
},
|
||||
)?;
|
||||
|
||||
// Close all shards
|
||||
@@ -162,7 +160,7 @@ pub fn rebuild_ec_files(
|
||||
}
|
||||
|
||||
let rs = ReedSolomon::new(data_shards, parity_shards)
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("reed-solomon init: {:?}", e)))?;
|
||||
.map_err(|e| io::Error::other(format!("reed-solomon init: {:?}", e)))?;
|
||||
|
||||
let total_shards = data_shards + parity_shards;
|
||||
let mut shards: Vec<EcVolumeShard> = (0..total_shards as u8)
|
||||
@@ -175,7 +173,7 @@ pub fn rebuild_ec_files(
|
||||
let mut shard_size = 0;
|
||||
for (i, shard) in shards.iter_mut().enumerate() {
|
||||
if !missing_shard_ids.contains(&(i as u32)) {
|
||||
if let Ok(_) = shard.open() {
|
||||
if shard.open().is_ok() {
|
||||
let size = shard.file_size();
|
||||
if size > shard_size {
|
||||
shard_size = size;
|
||||
@@ -185,7 +183,7 @@ pub fn rebuild_ec_files(
|
||||
let mut found = false;
|
||||
for &other_dir in additional_dirs {
|
||||
let mut alt = EcVolumeShard::new(other_dir, collection, volume_id, i as u8);
|
||||
if let Ok(_) = alt.open() {
|
||||
if alt.open().is_ok() {
|
||||
let size = alt.file_size();
|
||||
if size > shard_size {
|
||||
shard_size = size;
|
||||
@@ -251,12 +249,8 @@ pub fn rebuild_ec_files(
|
||||
}
|
||||
|
||||
// Reconstruct missing shards
|
||||
rs.reconstruct(&mut buffers).map_err(|e| {
|
||||
io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!("reed-solomon reconstruct: {:?}", e),
|
||||
)
|
||||
})?;
|
||||
rs.reconstruct(&mut buffers)
|
||||
.map_err(|e| io::Error::other(format!("reed-solomon reconstruct: {:?}", e)))?;
|
||||
|
||||
// Write recovered data into the missing shards
|
||||
for i in missing_shard_ids {
|
||||
@@ -284,40 +278,63 @@ pub fn rebuild_ec_files(
|
||||
/// FULL walk only reads live data-shard intervals, so on its own it can't catch
|
||||
/// bitrot in a parity shard or an unwalked region. Move to mode 4 (CHECKSUM) and
|
||||
/// drop it from mode 2 once the `.ecsum` subsystem lands.
|
||||
///
|
||||
/// `dirs` is indexed BY SHARD ID: each entry is the directory holding that
|
||||
/// shard, or `None` when no disk mounts it. A reconciled volume's shards can be
|
||||
/// split across disks, so a single directory cannot address them all.
|
||||
pub fn verify_ec_shards(
|
||||
dir: &str,
|
||||
dirs: &[Option<String>],
|
||||
collection: &str,
|
||||
volume_id: VolumeId,
|
||||
data_shards: usize,
|
||||
parity_shards: usize,
|
||||
) -> io::Result<(Vec<u32>, Vec<String>)> {
|
||||
let rs = ReedSolomon::new(data_shards, parity_shards)
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("reed-solomon init: {:?}", e)))?;
|
||||
.map_err(|e| io::Error::other(format!("reed-solomon init: {:?}", e)))?;
|
||||
|
||||
let total_shards = data_shards + parity_shards;
|
||||
let mut shards: Vec<EcVolumeShard> = (0..total_shards as u8)
|
||||
.map(|i| EcVolumeShard::new(dir, collection, volume_id, i))
|
||||
let mut shards: Vec<Option<EcVolumeShard>> = (0..total_shards)
|
||||
.map(|i| {
|
||||
dirs.get(i)
|
||||
.and_then(|d| d.as_ref())
|
||||
.map(|d| EcVolumeShard::new(d, collection, volume_id, i as u8))
|
||||
})
|
||||
.collect();
|
||||
|
||||
let mut shard_size = 0;
|
||||
let mut broken_shards = std::collections::HashSet::new();
|
||||
let mut details = Vec::new();
|
||||
|
||||
for (i, shard) in shards.iter_mut().enumerate() {
|
||||
if let Ok(_) = shard.open() {
|
||||
let size = shard.file_size();
|
||||
if size > shard_size {
|
||||
shard_size = size;
|
||||
for (i, slot) in shards.iter_mut().enumerate() {
|
||||
match slot.as_mut() {
|
||||
// Not a match guard: a binding is immutable until the guard ends,
|
||||
// and `open()` needs `&mut self`.
|
||||
Some(shard) => {
|
||||
if shard.open().is_ok() {
|
||||
let size = shard.file_size();
|
||||
if size > shard_size {
|
||||
shard_size = size;
|
||||
}
|
||||
} else {
|
||||
broken_shards.insert(i as u32);
|
||||
details.push(format!("failed to open or missing shard {}", i));
|
||||
}
|
||||
}
|
||||
None => {
|
||||
broken_shards.insert(i as u32);
|
||||
details.push(format!("shard {} is not mounted on any disk", i));
|
||||
}
|
||||
} else {
|
||||
broken_shards.insert(i as u32);
|
||||
details.push(format!("failed to open or missing shard {}", i));
|
||||
}
|
||||
}
|
||||
|
||||
if shard_size == 0 || broken_shards.len() >= parity_shards {
|
||||
// Can't do much if we don't know the size or have too many missing
|
||||
return Ok((broken_shards.into_iter().collect(), details));
|
||||
// Can't do much if we don't know the size or have too many missing.
|
||||
// Sort like the normal path below: a `HashSet` iteration order would
|
||||
// make this return shard ids in an arbitrary order, and enough `None`
|
||||
// entries in `dirs` now reach this branch for a caller to notice.
|
||||
let mut broken_vec: Vec<u32> = broken_shards.into_iter().collect();
|
||||
broken_vec.sort_unstable();
|
||||
return Ok((broken_vec, details));
|
||||
}
|
||||
|
||||
let block_size = ERASURE_CODING_SMALL_BLOCK_SIZE;
|
||||
@@ -331,7 +348,17 @@ pub fn verify_ec_shards(
|
||||
let mut read_failed = false;
|
||||
for i in 0..total_shards {
|
||||
if !broken_shards.contains(&(i as u32)) {
|
||||
if let Err(e) = shards[i].read_at(&mut buffers[i], offset) {
|
||||
// The `None` arm is defensive and unreachable: the open loop
|
||||
// put every unmounted slot in `broken_shards`, which this
|
||||
// branch already skipped. Kept because the `Option` forces
|
||||
// some handling here, and an error is the only shape that
|
||||
// cannot quietly feed an unread buffer into the parity
|
||||
// comparison below. Nothing needs to cover it.
|
||||
let read = match shards[i].as_mut() {
|
||||
Some(shard) => shard.read_at(&mut buffers[i], offset),
|
||||
None => Err(io::Error::new(io::ErrorKind::NotFound, "shard not mounted")),
|
||||
};
|
||||
if let Err(e) = read {
|
||||
broken_shards.insert(i as u32);
|
||||
details.push(format!("read error shard {}: {}", i, e));
|
||||
read_failed = true;
|
||||
@@ -345,27 +372,27 @@ pub fn verify_ec_shards(
|
||||
if !read_failed {
|
||||
// Need to convert Vec<Vec<u8>> to &[&[u8]] for rs.verify
|
||||
let slice_ptrs: Vec<&[u8]> = buffers.iter().map(|v| v.as_slice()).collect();
|
||||
if let Ok(is_valid) = rs.verify(&slice_ptrs) {
|
||||
if !is_valid {
|
||||
// Reed-Solomon verification failed. We cannot easily pinpoint which shard
|
||||
// is corrupted without recalculating parities or syndromes, so we just
|
||||
// log that this batch has corruption. Wait, we can test each parity shard!
|
||||
// Let's re-encode from the first `data_shards` and compare to the actual `parity_shards`.
|
||||
if let Ok(is_valid) = rs.verify(&slice_ptrs)
|
||||
&& !is_valid
|
||||
{
|
||||
// Reed-Solomon verification failed. We cannot easily pinpoint which shard
|
||||
// is corrupted without recalculating parities or syndromes, so we just
|
||||
// log that this batch has corruption. Wait, we can test each parity shard!
|
||||
// Let's re-encode from the first `data_shards` and compare to the actual `parity_shards`.
|
||||
|
||||
let mut verify_buffers = buffers.clone();
|
||||
// Clear the parity parts
|
||||
for i in data_shards..total_shards {
|
||||
verify_buffers[i].fill(0);
|
||||
}
|
||||
if rs.encode(&mut verify_buffers).is_ok() {
|
||||
for i in 0..total_shards {
|
||||
if buffers[i] != verify_buffers[i] {
|
||||
broken_shards.insert(i as u32);
|
||||
details.push(format!(
|
||||
"parity mismatch on shard {} at offset {}",
|
||||
i, offset
|
||||
));
|
||||
}
|
||||
let mut verify_buffers = buffers.clone();
|
||||
// Clear the parity parts
|
||||
for buf in &mut verify_buffers[data_shards..total_shards] {
|
||||
buf.fill(0);
|
||||
}
|
||||
if rs.encode(&mut verify_buffers).is_ok() {
|
||||
for i in 0..total_shards {
|
||||
if buffers[i] != verify_buffers[i] {
|
||||
broken_shards.insert(i as u32);
|
||||
details.push(format!(
|
||||
"parity mismatch on shard {} at offset {}",
|
||||
i, offset
|
||||
));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -377,7 +404,7 @@ pub fn verify_ec_shards(
|
||||
}
|
||||
|
||||
// Close all shards
|
||||
for shard in &mut shards {
|
||||
for shard in shards.iter_mut().flatten() {
|
||||
shard.close();
|
||||
}
|
||||
|
||||
@@ -398,22 +425,23 @@ pub(crate) fn write_sorted_ecx_from_idx(idx_path: &str, ecx_path: &str) -> io::R
|
||||
|
||||
// Read all idx entries
|
||||
let mut idx_file = File::open(idx_path)?;
|
||||
let mut entries: Vec<(NeedleId, Offset, Size)> = Vec::new();
|
||||
|
||||
let mut last: std::collections::HashMap<NeedleId, (Offset, Size)> =
|
||||
std::collections::HashMap::new();
|
||||
idx::walk_index_file(&mut idx_file, 0, |key, offset, size| {
|
||||
entries.push((key, offset, size));
|
||||
last.insert(key, (offset, size));
|
||||
Ok(())
|
||||
})?;
|
||||
|
||||
// Sort by NeedleId, then by actual offset so later entries come last
|
||||
entries.sort_by_key(|&(key, offset, _)| (key, offset.to_actual_offset()));
|
||||
|
||||
// Remove duplicates (keep last/latest entry for each key).
|
||||
// dedup_by_key keeps the first in each run, so we reverse first,
|
||||
// dedup, then reverse back.
|
||||
entries.reverse();
|
||||
entries.dedup_by_key(|entry| entry.0);
|
||||
entries.reverse();
|
||||
let mut entries: Vec<(NeedleId, Offset, Size)> = last
|
||||
.into_iter()
|
||||
.filter_map(|(key, (offset, size))| {
|
||||
if size.is_deleted() || offset.is_zero() {
|
||||
None
|
||||
} else {
|
||||
Some((key, offset, size))
|
||||
}
|
||||
})
|
||||
.collect();
|
||||
entries.sort_by_key(|&(key, _o, _s)| key);
|
||||
|
||||
// Write sorted entries to .ecx
|
||||
let mut ecx_file = File::create(ecx_path)?;
|
||||
@@ -457,7 +485,7 @@ pub fn rebuild_ecx_file(
|
||||
.collect();
|
||||
|
||||
for (i, shard) in shards.iter_mut().enumerate() {
|
||||
if let Err(_) = shard.open() {
|
||||
if shard.open().is_err() {
|
||||
let mut found = false;
|
||||
for &other_dir in additional_dirs {
|
||||
let mut alt = EcVolumeShard::new(other_dir, collection, volume_id, i as u8);
|
||||
@@ -474,7 +502,7 @@ pub fn rebuild_ecx_file(
|
||||
}
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::NotFound,
|
||||
format!("cannot open data shard for ecx rebuild"),
|
||||
"cannot open data shard for ecx rebuild".to_string(),
|
||||
));
|
||||
}
|
||||
}
|
||||
@@ -482,7 +510,7 @@ pub fn rebuild_ecx_file(
|
||||
|
||||
// Determine total logical data size from shard sizes
|
||||
let shard_size = shards.iter().map(|s| s.file_size()).max().unwrap_or(0);
|
||||
let total_data_size = shard_size as i64 * data_shards as i64;
|
||||
let total_data_size = shard_size * data_shards as i64;
|
||||
// The volume's shard block layout: the .vif-recorded uniform block size,
|
||||
// or the legacy two-tier sizes when 0. The row count comes from the shard
|
||||
// length; -1 disambiguates a legacy shard that is an exact large-block
|
||||
@@ -505,7 +533,7 @@ pub fn rebuild_ecx_file(
|
||||
let locate_shard_size = if dat_file_size > 0 {
|
||||
dat_file_size / data_shards as i64
|
||||
} else {
|
||||
(shard_size as i64 - 1).max(0)
|
||||
(shard_size - 1).max(0)
|
||||
};
|
||||
|
||||
// Read version from superblock (first byte of logical data)
|
||||
@@ -554,7 +582,8 @@ pub fn rebuild_ecx_file(
|
||||
}
|
||||
|
||||
let cookie = Cookie::from_bytes(&header_buf[..COOKIE_SIZE]);
|
||||
let needle_id = NeedleId::from_bytes(&header_buf[COOKIE_SIZE..COOKIE_SIZE + NEEDLE_ID_SIZE]);
|
||||
let needle_id =
|
||||
NeedleId::from_bytes(&header_buf[COOKIE_SIZE..COOKIE_SIZE + NEEDLE_ID_SIZE]);
|
||||
let size = Size::from_bytes(&header_buf[COOKIE_SIZE + NEEDLE_ID_SIZE..header_size]);
|
||||
|
||||
// Validate: stop if we hit zero cookie+id (end of data)
|
||||
@@ -607,7 +636,6 @@ pub fn rebuild_ecx_file(
|
||||
/// Read bytes from EC data shards at a logical offset in the .dat file,
|
||||
/// resolving the shard/offset through the volume's block layout via
|
||||
/// locate_data — the same mapping the read path uses.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
fn read_from_data_shards(
|
||||
shards: &[EcVolumeShard],
|
||||
buf: &mut [u8],
|
||||
@@ -674,30 +702,50 @@ fn read_from_data_shards(
|
||||
/// the uniform block is.
|
||||
const ENCODE_BUFFER_SIZE: usize = 256 * 1024;
|
||||
|
||||
/// Shape of one encode run: the Reed-Solomon split and the block sizes that
|
||||
/// fix where every byte of the .dat lands in the shards. Mirrors Go's
|
||||
/// `ECContext`. `buffer_size` must divide both block sizes.
|
||||
#[derive(Clone, Copy, Debug)]
|
||||
pub(crate) struct EcEncodeLayout {
|
||||
pub(crate) data_shards: usize,
|
||||
pub(crate) parity_shards: usize,
|
||||
/// Bytes of each shard's block handled per sub-batch; bounds memory at
|
||||
/// `total_shards * buffer_size` however large the blocks are.
|
||||
pub(crate) buffer_size: usize,
|
||||
pub(crate) large_block_size: usize,
|
||||
pub(crate) small_block_size: usize,
|
||||
}
|
||||
|
||||
/// Encode the .dat file data into shard files.
|
||||
///
|
||||
/// Uses a two-phase approach matching Go's ec_encoder.go:
|
||||
/// 1. Process as many large blocks as possible
|
||||
/// 2. Process remaining data with small blocks
|
||||
///
|
||||
/// `buffer_size` must divide both block sizes.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
pub(crate) fn encode_dat_file(
|
||||
dat_file: &File,
|
||||
dat_size: i64,
|
||||
rs: &ReedSolomon,
|
||||
shards: &mut [EcVolumeShard],
|
||||
builders: &mut [ShardChecksumBuilder],
|
||||
data_shards: usize,
|
||||
parity_shards: usize,
|
||||
buffer_size: usize,
|
||||
large_block_size: usize,
|
||||
small_block_size: usize,
|
||||
layout: EcEncodeLayout,
|
||||
) -> io::Result<()> {
|
||||
let EcEncodeLayout {
|
||||
data_shards,
|
||||
parity_shards,
|
||||
buffer_size,
|
||||
large_block_size,
|
||||
small_block_size,
|
||||
} = layout;
|
||||
let total_shards = data_shards + parity_shards;
|
||||
let mut buffers: Vec<Vec<u8>> = (0..total_shards)
|
||||
.map(|_| vec![0u8; buffer_size])
|
||||
.collect();
|
||||
let mut buffers: Vec<Vec<u8>> = (0..total_shards).map(|_| vec![0u8; buffer_size]).collect();
|
||||
let mut run = EncodeRun {
|
||||
dat_file,
|
||||
rs,
|
||||
buffers: &mut buffers,
|
||||
shards,
|
||||
builders,
|
||||
data_shards,
|
||||
};
|
||||
|
||||
let mut remaining = dat_size;
|
||||
let mut offset: u64 = 0;
|
||||
@@ -706,16 +754,7 @@ pub(crate) fn encode_dat_file(
|
||||
let large_row_size = large_block_size * data_shards;
|
||||
|
||||
while remaining >= large_row_size as i64 {
|
||||
encode_data(
|
||||
dat_file,
|
||||
offset,
|
||||
large_block_size,
|
||||
rs,
|
||||
&mut buffers,
|
||||
shards,
|
||||
builders,
|
||||
data_shards,
|
||||
)?;
|
||||
run.encode_row(offset, large_block_size)?;
|
||||
offset += large_row_size as u64;
|
||||
remaining -= large_row_size as i64;
|
||||
}
|
||||
@@ -725,16 +764,7 @@ pub(crate) fn encode_dat_file(
|
||||
|
||||
while remaining > 0 {
|
||||
let to_process = remaining.min(small_row_size as i64);
|
||||
encode_data(
|
||||
dat_file,
|
||||
offset,
|
||||
small_block_size,
|
||||
rs,
|
||||
&mut buffers,
|
||||
shards,
|
||||
builders,
|
||||
data_shards,
|
||||
)?;
|
||||
run.encode_row(offset, small_block_size)?;
|
||||
offset += to_process as u64;
|
||||
remaining -= to_process;
|
||||
}
|
||||
@@ -742,102 +772,73 @@ pub(crate) fn encode_dat_file(
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Encode one row of blocks, streaming it in ENCODE_BUFFER_SIZE sub-batches so
|
||||
/// arbitrarily large blocks never require block-sized allocations. Mirrors
|
||||
/// Go's encodeData.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
fn encode_data(
|
||||
dat_file: &File,
|
||||
row_offset: u64,
|
||||
block_size: usize,
|
||||
rs: &ReedSolomon,
|
||||
buffers: &mut [Vec<u8>],
|
||||
shards: &mut [EcVolumeShard],
|
||||
builders: &mut [ShardChecksumBuilder],
|
||||
/// Everything one encode run streams through: the source .dat, the codec, a
|
||||
/// buffer per shard, and the per-shard file and checksum sinks.
|
||||
struct EncodeRun<'a> {
|
||||
dat_file: &'a File,
|
||||
rs: &'a ReedSolomon,
|
||||
buffers: &'a mut [Vec<u8>],
|
||||
shards: &'a mut [EcVolumeShard],
|
||||
builders: &'a mut [ShardChecksumBuilder],
|
||||
data_shards: usize,
|
||||
) -> io::Result<()> {
|
||||
let buffer_size = buffers[0].len();
|
||||
if block_size % buffer_size != 0 {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::InvalidInput,
|
||||
format!(
|
||||
"unexpected block size {} buffer size {}",
|
||||
block_size, buffer_size
|
||||
),
|
||||
));
|
||||
}
|
||||
let batch_count = block_size / buffer_size;
|
||||
for b in 0..batch_count {
|
||||
encode_one_batch(
|
||||
dat_file,
|
||||
row_offset + (b * buffer_size) as u64,
|
||||
block_size,
|
||||
rs,
|
||||
buffers,
|
||||
shards,
|
||||
builders,
|
||||
data_shards,
|
||||
)?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Encode one sub-batch: the same buffer-sized slice of every shard's block in
|
||||
/// this row. Mirrors Go's encodeDataOneBatch.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
fn encode_one_batch(
|
||||
dat_file: &File,
|
||||
offset: u64,
|
||||
block_size: usize,
|
||||
rs: &ReedSolomon,
|
||||
buffers: &mut [Vec<u8>],
|
||||
shards: &mut [EcVolumeShard],
|
||||
builders: &mut [ShardChecksumBuilder],
|
||||
data_shards: usize,
|
||||
) -> io::Result<()> {
|
||||
// Read data shards from the .dat file, zero-filling past EOF — the buffers
|
||||
// are reused across batches, so the tail must be cleared explicitly.
|
||||
for i in 0..data_shards {
|
||||
let read_offset = offset + (i * block_size) as u64;
|
||||
let n = read_at_most(dat_file, &mut buffers[i], read_offset)?;
|
||||
for b in buffers[i][n..].iter_mut() {
|
||||
*b = 0;
|
||||
impl EncodeRun<'_> {
|
||||
/// Encode one row of blocks, streaming it in ENCODE_BUFFER_SIZE sub-batches
|
||||
/// so arbitrarily large blocks never require block-sized allocations.
|
||||
/// Mirrors Go's encodeData.
|
||||
fn encode_row(&mut self, row_offset: u64, block_size: usize) -> io::Result<()> {
|
||||
let buffer_size = self.buffers[0].len();
|
||||
if !block_size.is_multiple_of(buffer_size) {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::InvalidInput,
|
||||
format!(
|
||||
"unexpected block size {} buffer size {}",
|
||||
block_size, buffer_size
|
||||
),
|
||||
));
|
||||
}
|
||||
let batch_count = block_size / buffer_size;
|
||||
for b in 0..batch_count {
|
||||
self.encode_one_batch(row_offset + (b * buffer_size) as u64, block_size)?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
// Encode parity shards
|
||||
rs.encode(&mut *buffers).map_err(|e| {
|
||||
io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!("reed-solomon encode: {:?}", e),
|
||||
)
|
||||
})?;
|
||||
/// Encode one sub-batch: the same buffer-sized slice of every shard's block
|
||||
/// in this row. Mirrors Go's encodeDataOneBatch.
|
||||
fn encode_one_batch(&mut self, offset: u64, block_size: usize) -> io::Result<()> {
|
||||
// Read data shards from the .dat file, zero-filling past EOF — the
|
||||
// buffers are reused across batches, so the tail must be cleared
|
||||
// explicitly.
|
||||
for (i, buf) in self.buffers[..self.data_shards].iter_mut().enumerate() {
|
||||
let read_offset = offset + (i * block_size) as u64;
|
||||
let n = read_at_most(self.dat_file, buf, read_offset)?;
|
||||
buf[n..].fill(0);
|
||||
}
|
||||
|
||||
// Write all shard buffers to files and feed the same bytes to each
|
||||
// shard's bitrot checksum builder, keeping covered_size == on-disk length.
|
||||
for (i, buf) in buffers.iter().enumerate() {
|
||||
shards[i].write_all(buf)?;
|
||||
builders[i].write(buf);
|
||||
// Encode parity shards
|
||||
self.rs
|
||||
.encode(&mut *self.buffers)
|
||||
.map_err(|e| io::Error::other(format!("reed-solomon encode: {:?}", e)))?;
|
||||
|
||||
// Write all shard buffers to files and feed the same bytes to each
|
||||
// shard's bitrot checksum builder, keeping covered_size == on-disk
|
||||
// length.
|
||||
for (i, buf) in self.buffers.iter().enumerate() {
|
||||
self.shards[i].write_all(buf)?;
|
||||
self.builders[i].write(buf);
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Read into `buf` at `offset` until it is full or EOF; returns bytes read.
|
||||
fn read_at_most(dat_file: &File, buf: &mut [u8], offset: u64) -> io::Result<usize> {
|
||||
let mut n = 0;
|
||||
while n < buf.len() {
|
||||
#[cfg(unix)]
|
||||
let r = {
|
||||
use std::os::unix::fs::FileExt;
|
||||
dat_file.read_at(&mut buf[n..], offset + n as u64)?
|
||||
};
|
||||
#[cfg(not(unix))]
|
||||
let r = {
|
||||
let mut f = dat_file.try_clone()?;
|
||||
f.seek(SeekFrom::Start(offset + n as u64))?;
|
||||
f.read(&mut buf[n..])?
|
||||
};
|
||||
let r = crate::storage::io::read_at(dat_file, &mut buf[n..], offset + n as u64)?;
|
||||
if r == 0 {
|
||||
break;
|
||||
}
|
||||
@@ -851,7 +852,7 @@ mod tests {
|
||||
use super::*;
|
||||
use crate::storage::needle::needle::Needle;
|
||||
use crate::storage::needle_map::NeedleMapKind;
|
||||
use crate::storage::volume::Volume;
|
||||
use crate::storage::volume::{Volume, VolumeSpec};
|
||||
use tempfile::TempDir;
|
||||
|
||||
#[test]
|
||||
@@ -863,13 +864,9 @@ mod tests {
|
||||
let mut v = Volume::new(
|
||||
dir,
|
||||
dir,
|
||||
"",
|
||||
VolumeId(1),
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
&VolumeSpec::default(),
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
@@ -914,13 +911,9 @@ mod tests {
|
||||
let mut v = Volume::new(
|
||||
dir,
|
||||
dir,
|
||||
"",
|
||||
VolumeId(1),
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
&VolumeSpec::default(),
|
||||
)
|
||||
.unwrap();
|
||||
for i in 1..=n {
|
||||
@@ -993,13 +986,9 @@ mod tests {
|
||||
let mut v = Volume::new(
|
||||
&dir,
|
||||
&dir,
|
||||
"",
|
||||
VolumeId(1),
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
&VolumeSpec::default(),
|
||||
)
|
||||
.unwrap();
|
||||
for i in 1..=20 {
|
||||
@@ -1031,7 +1020,10 @@ mod tests {
|
||||
let victim = format!("{}/1.ec03", dir);
|
||||
let full = std::fs::metadata(&victim).unwrap().len();
|
||||
assert!(full > 0, "encoded shard should be non-empty");
|
||||
let f = std::fs::OpenOptions::new().write(true).open(&victim).unwrap();
|
||||
let f = std::fs::OpenOptions::new()
|
||||
.write(true)
|
||||
.open(&victim)
|
||||
.unwrap();
|
||||
f.set_len(full / 2).unwrap();
|
||||
drop(f);
|
||||
|
||||
@@ -1185,19 +1177,15 @@ mod tests {
|
||||
#[test]
|
||||
fn test_rebuild_ecx_file_uniform_layout() {
|
||||
use crate::storage::needle_map::NeedleMapKind;
|
||||
use crate::storage::volume::Volume;
|
||||
use crate::storage::volume::{Volume, VolumeSpec};
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap().to_string();
|
||||
let mut v = Volume::new(
|
||||
&dir,
|
||||
&dir,
|
||||
"",
|
||||
VolumeId(2),
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
&VolumeSpec::default(),
|
||||
)
|
||||
.unwrap();
|
||||
for i in 1u64..=12 {
|
||||
@@ -1227,7 +1215,10 @@ mod tests {
|
||||
|
||||
rebuild_ecx_file(&dir, "", VolumeId(2), 10, block_size, 0, &[]).unwrap();
|
||||
let rebuilt = std::fs::read(&ecx_path).unwrap();
|
||||
assert_eq!(canonical, rebuilt, "rebuilt .ecx must match the encode-time .ecx");
|
||||
assert_eq!(
|
||||
canonical, rebuilt,
|
||||
"rebuilt .ecx must match the encode-time .ecx"
|
||||
);
|
||||
}
|
||||
|
||||
// A truncated data shard must FAIL the .ecx rebuild, not publish the
|
||||
@@ -1235,19 +1226,15 @@ mod tests {
|
||||
#[test]
|
||||
fn test_rebuild_ecx_file_fails_on_truncated_shard() {
|
||||
use crate::storage::needle_map::NeedleMapKind;
|
||||
use crate::storage::volume::Volume;
|
||||
use crate::storage::volume::{Volume, VolumeSpec};
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap().to_string();
|
||||
let mut v = Volume::new(
|
||||
&dir,
|
||||
&dir,
|
||||
"",
|
||||
VolumeId(3),
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
&VolumeSpec::default(),
|
||||
)
|
||||
.unwrap();
|
||||
for i in 1u64..=12 {
|
||||
@@ -1345,13 +1332,9 @@ mod tests {
|
||||
let mut v = Volume::new(
|
||||
dat_dir,
|
||||
idx_dir,
|
||||
"",
|
||||
VolumeId(1),
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
&VolumeSpec::default(),
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
@@ -1429,13 +1412,9 @@ mod tests {
|
||||
let mut v = Volume::new(
|
||||
dat_dir,
|
||||
idx_dir,
|
||||
"",
|
||||
VolumeId(1),
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
&VolumeSpec::default(),
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
@@ -1457,4 +1436,189 @@ mod tests {
|
||||
"should fail when idx_dir doesn't contain .idx"
|
||||
);
|
||||
}
|
||||
|
||||
/// Write a real 10+4 encoded volume into `dir`.
|
||||
///
|
||||
/// Unlike `make_volume_with_needles` and `encode_sample_volume` this seeds
|
||||
/// a caller-chosen directory, which is what a split-disk test needs: the
|
||||
/// shards have to be scattered out of the directory they were encoded into.
|
||||
fn seed_encoded_volume(dir: &str, vid: VolumeId) {
|
||||
let mut v = Volume::new(
|
||||
dir,
|
||||
dir,
|
||||
vid,
|
||||
NeedleMapKind::InMemory,
|
||||
&VolumeSpec::default(),
|
||||
)
|
||||
.unwrap();
|
||||
for i in 1..=8 {
|
||||
let data = format!("test data for needle {} with a bit more length", i);
|
||||
let mut n = Needle {
|
||||
id: NeedleId(i),
|
||||
cookie: Cookie(i as u32),
|
||||
data: data.as_bytes().to_vec(),
|
||||
data_size: data.len() as u32,
|
||||
..Needle::default()
|
||||
};
|
||||
v.write_needle(&mut n, true, false).unwrap();
|
||||
}
|
||||
v.sync_to_disk().unwrap();
|
||||
v.close();
|
||||
write_ec_files(dir, dir, "", vid, 10, 4).unwrap();
|
||||
}
|
||||
|
||||
/// Shards split across two directories must all be found. Passing one dir
|
||||
/// per shard is what lets a reconciled volume's parity be checked at all.
|
||||
#[test]
|
||||
fn test_verify_ec_shards_reads_shards_from_multiple_dirs() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let src = tmp.path().join("src");
|
||||
let d0 = tmp.path().join("d0");
|
||||
let d1 = tmp.path().join("d1");
|
||||
for d in [&src, &d0, &d1] {
|
||||
std::fs::create_dir_all(d).unwrap();
|
||||
}
|
||||
let src_s = src.to_str().unwrap();
|
||||
seed_encoded_volume(src_s, VolumeId(1));
|
||||
|
||||
// Move shards 0..=6 to d0 and 7..=13 to d1.
|
||||
let mut dirs: Vec<Option<String>> = Vec::new();
|
||||
for id in 0..14u8 {
|
||||
let target = if id < 7 { &d0 } else { &d1 };
|
||||
std::fs::rename(
|
||||
format!("{}/1.ec{:02}", src_s, id),
|
||||
format!("{}/1.ec{:02}", target.to_str().unwrap(), id),
|
||||
)
|
||||
.unwrap();
|
||||
dirs.push(Some(target.to_str().unwrap().to_string()));
|
||||
}
|
||||
|
||||
let (broken, details) = verify_ec_shards(&dirs, "", VolumeId(1), 10, 4).unwrap();
|
||||
assert!(
|
||||
broken.is_empty(),
|
||||
"split-dir shards reported broken: {:?}",
|
||||
details
|
||||
);
|
||||
}
|
||||
|
||||
/// A shard no disk holds is a missing shard, not a panic and not a silent
|
||||
/// pass: it is REPORTED, by id, with a message that distinguishes "no disk
|
||||
/// holds this shard" from "the disk holds it but it won't open".
|
||||
///
|
||||
/// Read the scope literally. This does NOT show that the mounted shards
|
||||
/// verify clean. `dirs[5] = None` puts shard 5 in `broken_shards` before
|
||||
/// the block loop starts, so every iteration takes the
|
||||
/// `else { read_failed = true; }` arm and the Reed-Solomon comparison never
|
||||
/// runs at all. `broken == vec![5]` therefore holds because the other 13
|
||||
/// were never verified, not because they verified clean -- a parity check
|
||||
/// over intact shards is what
|
||||
/// `test_verify_ec_shards_reads_shards_from_multiple_dirs` and the
|
||||
/// end-to-end split-disk FULL scrub establish.
|
||||
#[test]
|
||||
fn test_verify_ec_shards_treats_a_none_dir_as_missing() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
seed_encoded_volume(dir, VolumeId(1));
|
||||
|
||||
let mut dirs: Vec<Option<String>> = (0..14).map(|_| Some(dir.to_string())).collect();
|
||||
dirs[5] = None;
|
||||
|
||||
let (broken, details) = verify_ec_shards(&dirs, "", VolumeId(1), 10, 4).unwrap();
|
||||
assert_eq!(
|
||||
broken,
|
||||
vec![5],
|
||||
"an unmounted shard must be reported, and only it: {:?}",
|
||||
details
|
||||
);
|
||||
// "no disk holds this shard" and "the disk holds it but it won't open"
|
||||
// are different operator problems, which is why they carry different
|
||||
// messages. Asserting only the id would let one masquerade as the other.
|
||||
assert!(
|
||||
details.iter().any(|d| d.contains("not mounted")),
|
||||
"an unmounted shard must be distinguished from an unopenable one, got {:?}",
|
||||
details
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_encode_drops_tombstone_last_wins() {
|
||||
use crate::storage::idx;
|
||||
use crate::storage::types::{NeedleId, Offset, Size, TOMBSTONE_FILE_SIZE};
|
||||
let tmp = tempfile::TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
let idx_path = format!("{}/t.idx", dir);
|
||||
let ecx_path = format!("{}/t.ecx", dir);
|
||||
let key = NeedleId(12345);
|
||||
{
|
||||
let mut f = std::fs::File::create(&idx_path).unwrap();
|
||||
idx::write_index_entry(&mut f, key, Offset::from_actual_offset(1024), Size(100))
|
||||
.unwrap();
|
||||
idx::write_index_entry(&mut f, key, Offset::default(), TOMBSTONE_FILE_SIZE).unwrap();
|
||||
}
|
||||
super::write_sorted_ecx_from_idx(&idx_path, &ecx_path).unwrap();
|
||||
let mut found = false;
|
||||
{
|
||||
let mut f = std::fs::File::open(&ecx_path).unwrap();
|
||||
idx::walk_index_file(&mut f, 0, |k, _o, _s| {
|
||||
if k == key {
|
||||
found = true;
|
||||
}
|
||||
Ok(())
|
||||
})
|
||||
.unwrap();
|
||||
}
|
||||
assert!(!found, "tombstoned key must not appear in .ecx");
|
||||
let idx2 = format!("{}/t2.idx", dir);
|
||||
let ecx2 = format!("{}/t2.ecx", dir);
|
||||
{
|
||||
let mut f = std::fs::File::create(&idx2).unwrap();
|
||||
idx::write_index_entry(&mut f, key, Offset::default(), TOMBSTONE_FILE_SIZE).unwrap();
|
||||
idx::write_index_entry(&mut f, key, Offset::from_actual_offset(2048), Size(200))
|
||||
.unwrap();
|
||||
}
|
||||
super::write_sorted_ecx_from_idx(&idx2, &ecx2).unwrap();
|
||||
let mut found2 = false;
|
||||
{
|
||||
let mut f = std::fs::File::open(&ecx2).unwrap();
|
||||
idx::walk_index_file(&mut f, 0, |k, o, s| {
|
||||
if k == key {
|
||||
found2 = true;
|
||||
assert_eq!(o.to_actual_offset(), 2048);
|
||||
assert_eq!(s, Size(200));
|
||||
}
|
||||
Ok(())
|
||||
})
|
||||
.unwrap();
|
||||
}
|
||||
assert!(found2, "re-created key must appear live");
|
||||
// Zero offset with non-negative size is also a deletion: Go
|
||||
// readNeedleMap (`if !offset.IsZero() && !size.IsDeleted() { Set }
|
||||
// else { Delete }`) and CompactNeedleMap::load_from_idx both treat
|
||||
// it as deleted. Encode must drop it too, or the .ecx live-map
|
||||
// mismatches replay.
|
||||
let idx3 = format!("{}/t3.idx", dir);
|
||||
let ecx3 = format!("{}/t3.ecx", dir);
|
||||
{
|
||||
let mut f = std::fs::File::create(&idx3).unwrap();
|
||||
idx::write_index_entry(&mut f, key, Offset::from_actual_offset(1024), Size(100))
|
||||
.unwrap();
|
||||
idx::write_index_entry(&mut f, key, Offset::default(), Size(0)).unwrap();
|
||||
}
|
||||
super::write_sorted_ecx_from_idx(&idx3, &ecx3).unwrap();
|
||||
let mut found3 = false;
|
||||
{
|
||||
let mut f = std::fs::File::open(&ecx3).unwrap();
|
||||
idx::walk_index_file(&mut f, 0, |k, _o, _s| {
|
||||
if k == key {
|
||||
found3 = true;
|
||||
}
|
||||
Ok(())
|
||||
})
|
||||
.unwrap();
|
||||
}
|
||||
assert!(
|
||||
!found3,
|
||||
"zero-offset row must not appear in .ecx even with non-negative size"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -16,6 +16,20 @@ pub const ERASURE_CODING_SMALL_BLOCK_SIZE: usize = 1024 * 1024; // 1MB
|
||||
|
||||
pub type ShardId = u8;
|
||||
|
||||
/// Validate a wire shard id. `ShardId` is `u8` but only 0..MAX_SHARD_COUNT are valid.
|
||||
/// Rejects 256 (would truncate to 0 and delete .ec00) and 270 (would alias 14).
|
||||
pub fn shard_id_try_from(v: u32) -> Result<ShardId, String> {
|
||||
if v < MAX_SHARD_COUNT as u32 {
|
||||
Ok(v as ShardId)
|
||||
} else {
|
||||
Err(format!(
|
||||
"invalid shard id {} (max {})",
|
||||
v,
|
||||
MAX_SHARD_COUNT - 1
|
||||
))
|
||||
}
|
||||
}
|
||||
|
||||
/// A single erasure-coded shard file.
|
||||
pub struct EcVolumeShard {
|
||||
pub volume_id: VolumeId,
|
||||
@@ -78,23 +92,9 @@ impl EcVolumeShard {
|
||||
let file = self
|
||||
.ecd_file
|
||||
.as_ref()
|
||||
.ok_or_else(|| io::Error::new(io::ErrorKind::Other, "shard file not open"))?;
|
||||
.ok_or_else(|| io::Error::other("shard file not open"))?;
|
||||
|
||||
#[cfg(unix)]
|
||||
{
|
||||
use std::os::unix::fs::FileExt;
|
||||
file.read_at(buf, offset)
|
||||
}
|
||||
|
||||
#[cfg(not(unix))]
|
||||
{
|
||||
use std::io::{Read, Seek, SeekFrom};
|
||||
// File::read_at is unix-only; fall back to seek + read.
|
||||
// We need a mutable reference for seek/read, so clone the handle.
|
||||
let mut f = file.try_clone()?;
|
||||
f.seek(SeekFrom::Start(offset))?;
|
||||
f.read(buf)
|
||||
}
|
||||
crate::storage::io::read_at(file, buf, offset)
|
||||
}
|
||||
|
||||
/// Write data to the shard file (appends).
|
||||
@@ -102,7 +102,7 @@ impl EcVolumeShard {
|
||||
let file = self
|
||||
.ecd_file
|
||||
.as_mut()
|
||||
.ok_or_else(|| io::Error::new(io::ErrorKind::Other, "shard file not open"))?;
|
||||
.ok_or_else(|| io::Error::other("shard file not open"))?;
|
||||
file.write_all(data)?;
|
||||
self.ecd_file_size += data.len() as i64;
|
||||
Ok(())
|
||||
@@ -112,6 +112,21 @@ impl EcVolumeShard {
|
||||
self.ecd_file_size
|
||||
}
|
||||
|
||||
/// A duplicate of the mounted shard handle, for a reader that has to
|
||||
/// outlive the store guard.
|
||||
///
|
||||
/// This is the same descriptor `read_at` serves from, so it carries the
|
||||
/// `O_NOATIME` from `open_volume_file` and keeps pointing at the shard
|
||||
/// that was mounted, whatever later happens to the path. `dup` shares the
|
||||
/// kernel file offset, which is why every read through it must be
|
||||
/// positional (`read_at`), never seek-based.
|
||||
pub fn try_clone_file(&self) -> io::Result<File> {
|
||||
self.ecd_file
|
||||
.as_ref()
|
||||
.ok_or_else(|| io::Error::other("shard file not open"))?
|
||||
.try_clone()
|
||||
}
|
||||
|
||||
/// Protobuf descriptor for this shard. Mirrors Go's ToEcShardInfo.
|
||||
pub fn to_ec_shard_info(&self) -> crate::pb::volume_server_pb::EcShardInfo {
|
||||
crate::pb::volume_server_pb::EcShardInfo {
|
||||
@@ -236,4 +251,42 @@ mod tests {
|
||||
let shard = EcVolumeShard::new("/data", "", VolumeId(7), 13);
|
||||
assert_eq!(shard.file_name(), "/data/7.ec13");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_shard_id_try_from_u32_rejects_overflow() {
|
||||
use super::{MAX_SHARD_COUNT, shard_id_try_from};
|
||||
assert_eq!(shard_id_try_from(0).unwrap(), 0u8);
|
||||
assert_eq!(shard_id_try_from(14).unwrap(), 14u8);
|
||||
assert_eq!(shard_id_try_from(31).unwrap(), 31u8);
|
||||
assert!(shard_id_try_from(32).is_err());
|
||||
assert!(shard_id_try_from(256).is_err());
|
||||
assert!(shard_id_try_from(270).is_err());
|
||||
assert!(shard_id_try_from(u32::MAX).is_err());
|
||||
assert_eq!(MAX_SHARD_COUNT, 32);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_shard_batch_validation_is_atomic_rejects_without_partial_prefix() {
|
||||
use super::shard_id_try_from;
|
||||
// The mount/unmount handlers pre-validate the ENTIRE req.shard_ids into
|
||||
// a Vec<ShardId> BEFORE acquiring the write lock or mutating any EC
|
||||
// state. This test pins the validation half of that contract at the
|
||||
// unit level: a batch like [0, 32] must fail as a whole, so by
|
||||
// construction no validated prefix (e.g. shard 0) is ever applied.
|
||||
// The handler-level tests below assert the no-state-change half.
|
||||
let batch = vec![0u32, 32u32];
|
||||
let validated: Result<Vec<_>, _> =
|
||||
batch.iter().map(|&sid| shard_id_try_from(sid)).collect();
|
||||
assert!(
|
||||
validated.is_err(),
|
||||
"batch {:?} must be rejected as a whole",
|
||||
batch
|
||||
);
|
||||
// A fully-valid batch still validates cleanly.
|
||||
let ok: Result<Vec<_>, _> = [0u32, 1u32, 13u32]
|
||||
.iter()
|
||||
.map(|&sid| shard_id_try_from(sid))
|
||||
.collect();
|
||||
assert_eq!(ok.unwrap(), vec![0u8, 1u8, 13u8]);
|
||||
}
|
||||
}
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -11,7 +11,7 @@ pub mod ec_shard;
|
||||
pub mod ec_volume;
|
||||
|
||||
pub use ec_shard::{
|
||||
EcVolumeShard, ShardId, DATA_SHARDS_COUNT, MAX_SHARD_COUNT, MIN_TOTAL_DISKS,
|
||||
PARITY_SHARDS_COUNT, TOTAL_SHARDS_COUNT,
|
||||
DATA_SHARDS_COUNT, EcVolumeShard, MAX_SHARD_COUNT, MIN_TOTAL_DISKS, PARITY_SHARDS_COUNT,
|
||||
ShardId, TOTAL_SHARDS_COUNT,
|
||||
};
|
||||
pub use ec_volume::EcVolume;
|
||||
|
||||
@@ -57,7 +57,7 @@ pub fn check_index_file<R: Read + Seek>(
|
||||
errs.push(format!("walk index file: {}", e));
|
||||
}
|
||||
|
||||
entries.sort_by(|a, b| a.2.cmp(&b.2).then(a.3 .0.cmp(&b.3 .0)));
|
||||
entries.sort_by(|a, b| a.2.cmp(&b.2).then(a.3.0.cmp(&b.3.0)));
|
||||
|
||||
// Offset-0 logical tombstones (remote-tier deletes) occupy no physical extent,
|
||||
// so they cannot overlap anything — exclude them from the overlap check. They
|
||||
@@ -213,7 +213,11 @@ mod tests {
|
||||
let size = data.len() as i64;
|
||||
let (count, errs) = check_index_file(&mut Cursor::new(data), size, Version(3));
|
||||
assert_eq!(count, 2, "tombstone row is still counted: {:?}", errs);
|
||||
assert!(errs.is_empty(), "offset-0 tombstone must not overlap: {:?}", errs);
|
||||
assert!(
|
||||
errs.is_empty(),
|
||||
"offset-0 tombstone must not overlap: {:?}",
|
||||
errs
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
|
||||
@@ -0,0 +1,142 @@
|
||||
//! Positional file reads.
|
||||
//!
|
||||
//! Every read here is "these bytes at this offset", never "the next bytes".
|
||||
//! The handles are shared — `.dat` and `.idx` descriptors are borrowed from
|
||||
//! [`file_pool`](super::needle_map::file_pool), a mounted EC shard's handle is
|
||||
//! duplicated into a scrub plan — so no caller may rely on a file position.
|
||||
//!
|
||||
//! On unix that is `pread(2)` through `std::os::unix::fs::FileExt`. On Windows
|
||||
//! it is `seek_read`, which passes the offset through `OVERLAPPED`, so the read
|
||||
//! itself is independent of the current cursor.
|
||||
//!
|
||||
//! What these helpers replace is `try_clone()` + `seek()` + `read()`. A
|
||||
//! duplicated handle shares one kernel file offset with the original, so that
|
||||
//! sequence is two syscalls against state another thread can move in between:
|
||||
//! the seek positions the offset, a concurrent reader or an append moves it,
|
||||
//! and the read returns bytes from somewhere else entirely. `seek_read` carries
|
||||
//! its own offset in a single call, so there is no window.
|
||||
//!
|
||||
//! `seek_read` does still advance the cursor as a side effect — Windows updates
|
||||
//! the file pointer even for an `OVERLAPPED` read — which nothing here relies
|
||||
//! on. A caller that genuinely needs a private position must open the file
|
||||
//! again rather than duplicate a handle; see `Volume::dat_scan_plan` in
|
||||
//! [`storage::volume`](super::volume).
|
||||
|
||||
use std::fs::File;
|
||||
use std::io;
|
||||
|
||||
/// Reads exactly `buf.len()` bytes from `file` starting at `offset`.
|
||||
///
|
||||
/// Fails with [`io::ErrorKind::UnexpectedEof`] if the file ends first.
|
||||
pub(crate) fn read_exact_at(file: &File, buf: &mut [u8], offset: u64) -> io::Result<()> {
|
||||
#[cfg(unix)]
|
||||
{
|
||||
use std::os::unix::fs::FileExt;
|
||||
file.read_exact_at(buf, offset)?;
|
||||
}
|
||||
#[cfg(windows)]
|
||||
{
|
||||
use std::os::windows::fs::FileExt;
|
||||
let mut filled = 0;
|
||||
let mut at = offset;
|
||||
while filled < buf.len() {
|
||||
let n = match file.seek_read(&mut buf[filled..], at) {
|
||||
Ok(n) => n,
|
||||
Err(err) if err.kind() == io::ErrorKind::Interrupted => continue,
|
||||
Err(err) => return Err(err),
|
||||
};
|
||||
if n == 0 {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::UnexpectedEof,
|
||||
"unexpected EOF in seek_read",
|
||||
));
|
||||
}
|
||||
filled += n;
|
||||
at += n as u64;
|
||||
}
|
||||
}
|
||||
#[cfg(not(any(unix, windows)))]
|
||||
{
|
||||
compile_error!("Platform not supported: only unix and windows are supported");
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Reads up to `buf.len()` bytes from `file` starting at `offset`, returning
|
||||
/// how many were read.
|
||||
///
|
||||
/// A short read — including `0` at or past end of file — is not an error; use
|
||||
/// [`read_exact_at`] when the whole buffer must be filled.
|
||||
pub(crate) fn read_at(file: &File, buf: &mut [u8], offset: u64) -> io::Result<usize> {
|
||||
#[cfg(unix)]
|
||||
{
|
||||
use std::os::unix::fs::FileExt;
|
||||
file.read_at(buf, offset)
|
||||
}
|
||||
#[cfg(windows)]
|
||||
{
|
||||
use std::os::windows::fs::FileExt;
|
||||
file.seek_read(buf, offset)
|
||||
}
|
||||
#[cfg(not(any(unix, windows)))]
|
||||
{
|
||||
compile_error!("Platform not supported: only unix and windows are supported");
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::{read_at, read_exact_at};
|
||||
use std::io::{ErrorKind, Write};
|
||||
|
||||
fn temp_file(bytes: &[u8]) -> tempfile::NamedTempFile {
|
||||
let mut f = tempfile::NamedTempFile::new().expect("temp file");
|
||||
f.write_all(bytes).expect("write");
|
||||
f.flush().expect("flush");
|
||||
f
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn read_exact_at_fills_the_whole_buffer() {
|
||||
let f = temp_file(b"0123456789");
|
||||
let mut buf = [0u8; 10];
|
||||
read_exact_at(f.as_file(), &mut buf, 0).expect("read");
|
||||
assert_eq!(&buf, b"0123456789");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn read_exact_at_reads_from_the_offset() {
|
||||
let f = temp_file(b"0123456789");
|
||||
let mut buf = [0u8; 4];
|
||||
read_exact_at(f.as_file(), &mut buf, 3).expect("read");
|
||||
assert_eq!(&buf, b"3456");
|
||||
|
||||
// The helper is positional: a second read at a lower offset sees the
|
||||
// bytes at that offset, not wherever the first read left a cursor.
|
||||
let mut again = [0u8; 4];
|
||||
read_exact_at(f.as_file(), &mut again, 1).expect("read");
|
||||
assert_eq!(&again, b"1234");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn read_exact_at_short_file_is_unexpected_eof() {
|
||||
let f = temp_file(b"0123");
|
||||
let mut buf = [0u8; 8];
|
||||
let err = read_exact_at(f.as_file(), &mut buf, 0).expect_err("short file");
|
||||
assert_eq!(err.kind(), ErrorKind::UnexpectedEof);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn read_at_allows_a_short_read_at_eof() {
|
||||
let f = temp_file(b"0123456789");
|
||||
let mut buf = [0u8; 8];
|
||||
|
||||
let n = read_at(f.as_file(), &mut buf, 6).expect("read");
|
||||
assert_eq!(n, 4);
|
||||
assert_eq!(&buf[..n], b"6789");
|
||||
|
||||
// Entirely past the end is zero bytes, not an error.
|
||||
let n = read_at(f.as_file(), &mut buf, 10).expect("read");
|
||||
assert_eq!(n, 0);
|
||||
}
|
||||
}
|
||||
@@ -1,6 +1,7 @@
|
||||
pub mod disk_location;
|
||||
pub mod erasure_coding;
|
||||
pub mod idx;
|
||||
pub(crate) mod io;
|
||||
pub mod needle;
|
||||
pub mod needle_map;
|
||||
pub mod store;
|
||||
|
||||
@@ -21,7 +21,7 @@ impl CRC {
|
||||
/// Legacy `.Value()` function — deprecated in Go but needed for backward compat check.
|
||||
/// Formula: (crc >> 15 | crc << 17) + 0xa282ead8
|
||||
pub fn legacy_value(&self) -> u32 {
|
||||
(self.0 >> 15 | self.0 << 17).wrapping_add(0xa282ead8)
|
||||
self.0.rotate_right(15).wrapping_add(0xa282ead8)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -67,7 +67,9 @@ mod tests {
|
||||
fn test_crc_legacy_value() {
|
||||
let crc = CRC(0x12345678);
|
||||
let v = crc.legacy_value();
|
||||
let expected = (0x12345678u32 >> 15 | 0x12345678u32 << 17).wrapping_add(0xa282ead8);
|
||||
// (0x12345678 >> 15 | 0x12345678 << 17) + 0xa282ead8, worked out by hand so
|
||||
// the test checks the rotate rather than restating it.
|
||||
let expected = 0x4f730f40_u32;
|
||||
assert_eq!(v, expected);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,4 +1,8 @@
|
||||
pub mod crc;
|
||||
#[expect(
|
||||
clippy::module_inception,
|
||||
reason = "needle/needle.rs mirrors the Go package layout"
|
||||
)]
|
||||
pub mod needle;
|
||||
pub mod ttl;
|
||||
|
||||
|
||||
@@ -198,8 +198,8 @@ impl Needle {
|
||||
/// the data payload from disk at all, matching Go's `ReadNeedleMeta`.
|
||||
pub fn read_paged_meta(
|
||||
&mut self,
|
||||
header_bytes: &[u8], // first 20 bytes: NEEDLE_HEADER_SIZE + DATA_SIZE_SIZE
|
||||
meta_bytes: &[u8], // tail: non-data body metadata + checksum + timestamp + padding
|
||||
header_bytes: &[u8], // first 20 bytes: NEEDLE_HEADER_SIZE + DATA_SIZE_SIZE
|
||||
meta_bytes: &[u8], // tail: non-data body metadata + checksum + timestamp + padding
|
||||
offset: i64,
|
||||
expected_size: Size,
|
||||
version: Version,
|
||||
@@ -560,7 +560,7 @@ impl Needle {
|
||||
|
||||
// Padding to 8-byte alignment
|
||||
let padding = padding_length(self.size, version).0 as usize;
|
||||
buf.extend(std::iter::repeat(0u8).take(padding));
|
||||
buf.extend(std::iter::repeat_n(0u8, padding));
|
||||
|
||||
buf
|
||||
}
|
||||
@@ -581,23 +581,19 @@ impl Needle {
|
||||
// ============================================================================
|
||||
|
||||
/// Compute padding to align needle to NEEDLE_PADDING_SIZE (8 bytes).
|
||||
///
|
||||
/// The sum is formed in i64: a size read from a corrupt header can sit near
|
||||
/// `i32::MAX`, and adding the header, checksum and timestamp widths to it in
|
||||
/// i32 would overflow (a panic with overflow checks, a wrapped padding
|
||||
/// without). The result is at most NEEDLE_PADDING_SIZE, so it fits `Size`.
|
||||
pub fn padding_length(needle_size: Size, version: Version) -> Size {
|
||||
if version == VERSION_3 {
|
||||
Size(
|
||||
NEEDLE_PADDING_SIZE as i32
|
||||
- ((NEEDLE_HEADER_SIZE as i32
|
||||
+ needle_size.0
|
||||
+ NEEDLE_CHECKSUM_SIZE as i32
|
||||
+ TIMESTAMP_SIZE as i32)
|
||||
% NEEDLE_PADDING_SIZE as i32),
|
||||
)
|
||||
let fixed = if version == VERSION_3 {
|
||||
NEEDLE_HEADER_SIZE + NEEDLE_CHECKSUM_SIZE + TIMESTAMP_SIZE
|
||||
} else {
|
||||
Size(
|
||||
NEEDLE_PADDING_SIZE as i32
|
||||
- ((NEEDLE_HEADER_SIZE as i32 + needle_size.0 + NEEDLE_CHECKSUM_SIZE as i32)
|
||||
% NEEDLE_PADDING_SIZE as i32),
|
||||
)
|
||||
}
|
||||
NEEDLE_HEADER_SIZE + NEEDLE_CHECKSUM_SIZE
|
||||
};
|
||||
let unpadded = fixed as i64 + needle_size.0 as i64;
|
||||
Size((NEEDLE_PADDING_SIZE as i64 - unpadded % NEEDLE_PADDING_SIZE as i64) as i32)
|
||||
}
|
||||
|
||||
/// Body length = Size + Checksum + [Timestamp] + Padding.
|
||||
@@ -619,6 +615,30 @@ pub fn get_actual_size(size: Size, version: Version) -> i64 {
|
||||
NEEDLE_HEADER_SIZE as i64 + needle_body_length(size, version)
|
||||
}
|
||||
|
||||
/// Validate a wire-supplied needle body size before any `as usize` cast.
|
||||
/// Rejects negative/deleted sizes and bodies larger than the gRPC max message.
|
||||
/// Size(0) is allowed: empty/anomalous entries and tombstones read as size 0
|
||||
/// (actual_size = header+checksum+pad > 0, safe alloc, no wrap).
|
||||
/// Transport cap only: storage paths must NOT use this cap — see volume.rs
|
||||
/// guards (a >1GiB stored needle from a high-limit cluster must remain
|
||||
/// readable/compaction-safe). Keep `get_actual_size` unchanged (it
|
||||
/// intentionally returns negative for deleted index entries).
|
||||
pub fn validate_wire_size(size: Size) -> Result<(), String> {
|
||||
if size.0 < 0 {
|
||||
return Err(format!("invalid needle size {}", size.0));
|
||||
}
|
||||
// Keep in sync with canonical `GRPC_MAX_MESSAGE_SIZE` in server/grpc_client.rs:10
|
||||
// (duplicated here to avoid a storage->server import and prevent drift).
|
||||
const WIRE_MAX_NEEDLE_SIZE: i32 = 1 << 30;
|
||||
if size.0 > WIRE_MAX_NEEDLE_SIZE {
|
||||
return Err(format!(
|
||||
"needle size {} exceeds max {}",
|
||||
size.0, WIRE_MAX_NEEDLE_SIZE
|
||||
));
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Read 5 bytes as a u64 (big-endian, zero-padded high bytes).
|
||||
fn bytes_to_u64_5(bytes: &[u8]) -> u64 {
|
||||
assert!(bytes.len() >= 5);
|
||||
@@ -770,7 +790,9 @@ pub fn parse_needle_id_cookie(s: &str) -> Result<(NeedleId, Cookie), String> {
|
||||
|
||||
#[derive(Debug, thiserror::Error)]
|
||||
pub enum NeedleError {
|
||||
#[error("size mismatch at offset {offset}: found id={id} size={found:?}, expected size={expected:?}")]
|
||||
#[error(
|
||||
"size mismatch at offset {offset}: found id={id} size={found:?}, expected size={expected:?}"
|
||||
)]
|
||||
SizeMismatch {
|
||||
offset: i64,
|
||||
id: NeedleId,
|
||||
@@ -824,11 +846,13 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn test_needle_write_read_round_trip_v3() {
|
||||
let mut n = Needle::default();
|
||||
n.cookie = Cookie(42);
|
||||
n.id = NeedleId(100);
|
||||
n.data = b"hello world".to_vec();
|
||||
n.flags = 0;
|
||||
let mut n = Needle {
|
||||
cookie: Cookie(42),
|
||||
id: NeedleId(100),
|
||||
data: b"hello world".to_vec(),
|
||||
flags: 0,
|
||||
..Needle::default()
|
||||
};
|
||||
n.set_has_name();
|
||||
n.name = b"test.txt".to_vec();
|
||||
n.name_size = 8;
|
||||
@@ -867,11 +891,13 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn test_needle_write_read_round_trip_v2() {
|
||||
let mut n = Needle::default();
|
||||
n.cookie = Cookie(77);
|
||||
n.id = NeedleId(200);
|
||||
n.data = b"data v2".to_vec();
|
||||
n.flags = 0;
|
||||
let mut n = Needle {
|
||||
cookie: Cookie(77),
|
||||
id: NeedleId(200),
|
||||
data: b"data v2".to_vec(),
|
||||
flags: 0,
|
||||
..Needle::default()
|
||||
};
|
||||
|
||||
let bytes = n.write_bytes(VERSION_2);
|
||||
let expected_size = get_actual_size(n.size, VERSION_2);
|
||||
@@ -886,10 +912,12 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn test_read_bytes_meta_only_handles_tombstone_v3() {
|
||||
let mut tombstone = Needle::default();
|
||||
tombstone.cookie = Cookie(0x1234abcd);
|
||||
tombstone.id = NeedleId(300);
|
||||
tombstone.append_at_ns = 999_999;
|
||||
let mut tombstone = Needle {
|
||||
cookie: Cookie(0x1234abcd),
|
||||
id: NeedleId(300),
|
||||
append_at_ns: 999_999,
|
||||
..Needle::default()
|
||||
};
|
||||
|
||||
let bytes = tombstone.write_bytes(VERSION_3);
|
||||
|
||||
@@ -917,6 +945,21 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn padding_length_does_not_overflow_on_a_corrupt_size() {
|
||||
// A header read from a corrupt or truncated file can carry any i32
|
||||
// size. The scanners bound it against the bytes left before sizing a
|
||||
// buffer, but on a volume with more than 2 GiB left a size near
|
||||
// i32::MAX passes that bound, so the padding arithmetic itself must
|
||||
// not overflow. Overflow checks are on in test builds, so an i32 sum
|
||||
// here would panic rather than wrap.
|
||||
for version in [VERSION_2, VERSION_3] {
|
||||
let padding = padding_length(Size(i32::MAX), version).0 as i64;
|
||||
assert!((1..=NEEDLE_PADDING_SIZE as i64).contains(&padding));
|
||||
assert_eq!(get_actual_size(Size(i32::MAX), version) % 8, 0);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_file_id_parse() {
|
||||
let fid = FileId::parse("3,01637037d6").unwrap();
|
||||
@@ -961,4 +1004,16 @@ mod tests {
|
||||
assert_eq!(fid.key, NeedleId(0x123));
|
||||
assert_eq!(fid.cookie, Cookie(0));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_validate_wire_size_boundaries() {
|
||||
assert!(validate_wire_size(Size(-100)).is_err());
|
||||
assert!(validate_wire_size(Size(-1)).is_err());
|
||||
assert!(validate_wire_size(Size(0)).is_ok());
|
||||
assert!(validate_wire_size(Size(1024)).is_ok());
|
||||
assert!(validate_wire_size(Size(1)).is_ok());
|
||||
assert!(validate_wire_size(Size(1 << 30)).is_ok());
|
||||
assert!(validate_wire_size(Size((1 << 30) + 1)).is_err());
|
||||
assert!(validate_wire_size(Size(i32::MAX)).is_err());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -81,7 +81,7 @@ impl TTL {
|
||||
return Ok(TTL::EMPTY);
|
||||
}
|
||||
let last_byte = s.as_bytes()[s.len() - 1];
|
||||
let (num_str, unit_byte) = if last_byte >= b'0' && last_byte <= b'9' {
|
||||
let (num_str, unit_byte) = if last_byte.is_ascii_digit() {
|
||||
// All digits — default to minutes (matching Go)
|
||||
(s, b'm')
|
||||
} else {
|
||||
@@ -144,40 +144,73 @@ fn fit_ttl_count(count: u32, unit: u8) -> TTL {
|
||||
const MINUTE_SECS: u64 = 60;
|
||||
|
||||
// First pass: try exact fits from largest to smallest
|
||||
if seconds % YEAR_SECS == 0 && seconds / YEAR_SECS < 256 {
|
||||
return TTL { count: (seconds / YEAR_SECS) as u8, unit: TTL_UNIT_YEAR };
|
||||
if seconds.is_multiple_of(YEAR_SECS) && seconds / YEAR_SECS < 256 {
|
||||
return TTL {
|
||||
count: (seconds / YEAR_SECS) as u8,
|
||||
unit: TTL_UNIT_YEAR,
|
||||
};
|
||||
}
|
||||
if seconds % MONTH_SECS == 0 && seconds / MONTH_SECS < 256 {
|
||||
return TTL { count: (seconds / MONTH_SECS) as u8, unit: TTL_UNIT_MONTH };
|
||||
if seconds.is_multiple_of(MONTH_SECS) && seconds / MONTH_SECS < 256 {
|
||||
return TTL {
|
||||
count: (seconds / MONTH_SECS) as u8,
|
||||
unit: TTL_UNIT_MONTH,
|
||||
};
|
||||
}
|
||||
if seconds % WEEK_SECS == 0 && seconds / WEEK_SECS < 256 {
|
||||
return TTL { count: (seconds / WEEK_SECS) as u8, unit: TTL_UNIT_WEEK };
|
||||
if seconds.is_multiple_of(WEEK_SECS) && seconds / WEEK_SECS < 256 {
|
||||
return TTL {
|
||||
count: (seconds / WEEK_SECS) as u8,
|
||||
unit: TTL_UNIT_WEEK,
|
||||
};
|
||||
}
|
||||
if seconds % DAY_SECS == 0 && seconds / DAY_SECS < 256 {
|
||||
return TTL { count: (seconds / DAY_SECS) as u8, unit: TTL_UNIT_DAY };
|
||||
if seconds.is_multiple_of(DAY_SECS) && seconds / DAY_SECS < 256 {
|
||||
return TTL {
|
||||
count: (seconds / DAY_SECS) as u8,
|
||||
unit: TTL_UNIT_DAY,
|
||||
};
|
||||
}
|
||||
if seconds % HOUR_SECS == 0 && seconds / HOUR_SECS < 256 {
|
||||
return TTL { count: (seconds / HOUR_SECS) as u8, unit: TTL_UNIT_HOUR };
|
||||
if seconds.is_multiple_of(HOUR_SECS) && seconds / HOUR_SECS < 256 {
|
||||
return TTL {
|
||||
count: (seconds / HOUR_SECS) as u8,
|
||||
unit: TTL_UNIT_HOUR,
|
||||
};
|
||||
}
|
||||
// Minutes: truncating division
|
||||
if seconds / MINUTE_SECS < 256 {
|
||||
return TTL { count: (seconds / MINUTE_SECS) as u8, unit: TTL_UNIT_MINUTE };
|
||||
return TTL {
|
||||
count: (seconds / MINUTE_SECS) as u8,
|
||||
unit: TTL_UNIT_MINUTE,
|
||||
};
|
||||
}
|
||||
// Second pass: truncating division from smallest to largest
|
||||
if seconds / HOUR_SECS < 256 {
|
||||
return TTL { count: (seconds / HOUR_SECS) as u8, unit: TTL_UNIT_HOUR };
|
||||
return TTL {
|
||||
count: (seconds / HOUR_SECS) as u8,
|
||||
unit: TTL_UNIT_HOUR,
|
||||
};
|
||||
}
|
||||
if seconds / DAY_SECS < 256 {
|
||||
return TTL { count: (seconds / DAY_SECS) as u8, unit: TTL_UNIT_DAY };
|
||||
return TTL {
|
||||
count: (seconds / DAY_SECS) as u8,
|
||||
unit: TTL_UNIT_DAY,
|
||||
};
|
||||
}
|
||||
if seconds / WEEK_SECS < 256 {
|
||||
return TTL { count: (seconds / WEEK_SECS) as u8, unit: TTL_UNIT_WEEK };
|
||||
return TTL {
|
||||
count: (seconds / WEEK_SECS) as u8,
|
||||
unit: TTL_UNIT_WEEK,
|
||||
};
|
||||
}
|
||||
if seconds / MONTH_SECS < 256 {
|
||||
return TTL { count: (seconds / MONTH_SECS) as u8, unit: TTL_UNIT_MONTH };
|
||||
return TTL {
|
||||
count: (seconds / MONTH_SECS) as u8,
|
||||
unit: TTL_UNIT_MONTH,
|
||||
};
|
||||
}
|
||||
if seconds / YEAR_SECS < 256 {
|
||||
return TTL { count: (seconds / YEAR_SECS) as u8, unit: TTL_UNIT_YEAR };
|
||||
return TTL {
|
||||
count: (seconds / YEAR_SECS) as u8,
|
||||
unit: TTL_UNIT_YEAR,
|
||||
};
|
||||
}
|
||||
TTL::EMPTY
|
||||
}
|
||||
@@ -225,7 +258,13 @@ mod tests {
|
||||
// 24h normalizes to 1d via fitTtlCount
|
||||
let ttl = TTL::read("24h").unwrap();
|
||||
assert_eq!(ttl.to_seconds(), 86400);
|
||||
assert_eq!(ttl, TTL { count: 1, unit: TTL_UNIT_DAY });
|
||||
assert_eq!(
|
||||
ttl,
|
||||
TTL {
|
||||
count: 1,
|
||||
unit: TTL_UNIT_DAY
|
||||
}
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -271,12 +310,24 @@ mod tests {
|
||||
fn test_ttl_overflow_normalizes() {
|
||||
// Go's ReadTTL calls fitTtlCount: 300m = 18000s = 5h (exact fit)
|
||||
let ttl = TTL::read("300m").unwrap();
|
||||
assert_eq!(ttl, TTL { count: 5, unit: TTL_UNIT_HOUR });
|
||||
assert_eq!(
|
||||
ttl,
|
||||
TTL {
|
||||
count: 5,
|
||||
unit: TTL_UNIT_HOUR
|
||||
}
|
||||
);
|
||||
|
||||
// 256h = 921600s. Doesn't fit in hours (256 >= 256), doesn't fit exact in days.
|
||||
// Second pass: 921600/86400 = 10 (truncated) < 256 -> 10d
|
||||
let ttl = TTL::read("256h").unwrap();
|
||||
assert_eq!(ttl, TTL { count: 10, unit: TTL_UNIT_DAY });
|
||||
assert_eq!(
|
||||
ttl,
|
||||
TTL {
|
||||
count: 10,
|
||||
unit: TTL_UNIT_DAY
|
||||
}
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -284,19 +335,49 @@ mod tests {
|
||||
// Go's ReadTTL calls fitTtlCount which normalizes to coarsest unit.
|
||||
// 120m -> 2h, 7d -> 1w, 24h -> 1d.
|
||||
let ttl = TTL::read("120m").unwrap();
|
||||
assert_eq!(ttl, TTL { count: 2, unit: TTL_UNIT_HOUR });
|
||||
assert_eq!(
|
||||
ttl,
|
||||
TTL {
|
||||
count: 2,
|
||||
unit: TTL_UNIT_HOUR
|
||||
}
|
||||
);
|
||||
|
||||
let ttl = TTL::read("7d").unwrap();
|
||||
assert_eq!(ttl, TTL { count: 1, unit: TTL_UNIT_WEEK });
|
||||
assert_eq!(
|
||||
ttl,
|
||||
TTL {
|
||||
count: 1,
|
||||
unit: TTL_UNIT_WEEK
|
||||
}
|
||||
);
|
||||
|
||||
let ttl = TTL::read("24h").unwrap();
|
||||
assert_eq!(ttl, TTL { count: 1, unit: TTL_UNIT_DAY });
|
||||
assert_eq!(
|
||||
ttl,
|
||||
TTL {
|
||||
count: 1,
|
||||
unit: TTL_UNIT_DAY
|
||||
}
|
||||
);
|
||||
|
||||
// Values that don't simplify stay as-is
|
||||
let ttl = TTL::read("5d").unwrap();
|
||||
assert_eq!(ttl, TTL { count: 5, unit: TTL_UNIT_DAY });
|
||||
assert_eq!(
|
||||
ttl,
|
||||
TTL {
|
||||
count: 5,
|
||||
unit: TTL_UNIT_DAY
|
||||
}
|
||||
);
|
||||
|
||||
let ttl = TTL::read("3m").unwrap();
|
||||
assert_eq!(ttl, TTL { count: 3, unit: TTL_UNIT_MINUTE });
|
||||
assert_eq!(
|
||||
ttl,
|
||||
TTL {
|
||||
count: 3,
|
||||
unit: TTL_UNIT_MINUTE
|
||||
}
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -97,12 +97,13 @@ impl NeedleMapMetric {
|
||||
self.file_byte_count
|
||||
.fetch_add(new_size.0 as u64, Ordering::Relaxed);
|
||||
// Go: if oldSize > 0 && oldSize.IsValid() { LogDeletionCounter(oldSize) }
|
||||
if let Some(old_val) = old {
|
||||
if old_val.size.0 > 0 && old_val.size.is_valid() {
|
||||
self.deletion_count.fetch_add(1, Ordering::Relaxed);
|
||||
self.deletion_byte_count
|
||||
.fetch_add(old_val.size.0 as u64, Ordering::Relaxed);
|
||||
}
|
||||
if let Some(old_val) = old
|
||||
&& old_val.size.0 > 0
|
||||
&& old_val.size.is_valid()
|
||||
{
|
||||
self.deletion_count.fetch_add(1, Ordering::Relaxed);
|
||||
self.deletion_byte_count
|
||||
.fetch_add(old_val.size.0 as u64, Ordering::Relaxed);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -225,6 +226,12 @@ pub struct CompactNeedleMap {
|
||||
idx_file_offset: u64,
|
||||
}
|
||||
|
||||
impl Default for CompactNeedleMap {
|
||||
fn default() -> Self {
|
||||
Self::new()
|
||||
}
|
||||
}
|
||||
|
||||
impl CompactNeedleMap {
|
||||
/// Create a new empty in-memory map.
|
||||
pub fn new() -> Self {
|
||||
@@ -465,9 +472,9 @@ impl RedbNeedleMap {
|
||||
/// loses at most the writes since the last checkpoint from redb, and
|
||||
/// the next load replays them from .idx.
|
||||
fn begin_write_no_fsync(db: &Database) -> io::Result<redb::WriteTransaction> {
|
||||
let mut txn = db.begin_write().map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb begin_write: {}", e))
|
||||
})?;
|
||||
let mut txn = db
|
||||
.begin_write()
|
||||
.map_err(|e| io::Error::other(format!("redb begin_write: {}", e)))?;
|
||||
let _ = txn.set_durability(Durability::None);
|
||||
Ok(txn)
|
||||
}
|
||||
@@ -501,7 +508,7 @@ impl RedbNeedleMap {
|
||||
pub fn checkpoint(&mut self, sync_idx: bool) -> io::Result<()> {
|
||||
let txn = self.begin_checkpoint(sync_idx)?;
|
||||
txn.commit()
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("redb commit: {}", e)))?;
|
||||
.map_err(|e| io::Error::other(format!("redb commit: {}", e)))?;
|
||||
self.writes_since_checkpoint = 0;
|
||||
Ok(())
|
||||
}
|
||||
@@ -516,17 +523,17 @@ impl RedbNeedleMap {
|
||||
if sync_idx {
|
||||
self.sync()?;
|
||||
}
|
||||
let mut txn = self.db_or_err()?.begin_write().map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb begin_write: {}", e))
|
||||
})?;
|
||||
let mut txn = self
|
||||
.db_or_err()?
|
||||
.begin_write()
|
||||
.map_err(|e| io::Error::other(format!("redb begin_write: {}", e)))?;
|
||||
txn.set_quick_repair(true);
|
||||
if self.idx_file.is_some() {
|
||||
let mut meta = txn.open_table(META_TABLE).map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb open meta: {}", e))
|
||||
})?;
|
||||
meta.insert(META_IDX_SIZE, self.idx_file_offset).map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb insert meta: {}", e))
|
||||
})?;
|
||||
let mut meta = txn
|
||||
.open_table(META_TABLE)
|
||||
.map_err(|e| io::Error::other(format!("redb open meta: {}", e)))?;
|
||||
meta.insert(META_IDX_SIZE, self.idx_file_offset)
|
||||
.map_err(|e| io::Error::other(format!("redb insert meta: {}", e)))?;
|
||||
}
|
||||
Ok(txn)
|
||||
}
|
||||
@@ -538,22 +545,20 @@ impl RedbNeedleMap {
|
||||
let db = Database::builder()
|
||||
.set_cache_size(cache_bytes)
|
||||
.create(db_path)
|
||||
.map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb create error: {}", e))
|
||||
})?;
|
||||
.map_err(|e| io::Error::other(format!("redb create error: {}", e)))?;
|
||||
|
||||
// Ensure tables exist
|
||||
let txn = Self::begin_write_no_fsync(&db)?;
|
||||
{
|
||||
let _table = txn.open_table(NEEDLE_TABLE).map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb open_table: {}", e))
|
||||
})?;
|
||||
let _meta = txn.open_table(META_TABLE).map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb open_table meta: {}", e))
|
||||
})?;
|
||||
let _table = txn
|
||||
.open_table(NEEDLE_TABLE)
|
||||
.map_err(|e| io::Error::other(format!("redb open_table: {}", e)))?;
|
||||
let _meta = txn
|
||||
.open_table(META_TABLE)
|
||||
.map_err(|e| io::Error::other(format!("redb open_table meta: {}", e)))?;
|
||||
}
|
||||
txn.commit()
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("redb commit: {}", e)))?;
|
||||
.map_err(|e| io::Error::other(format!("redb commit: {}", e)))?;
|
||||
|
||||
Ok(RedbNeedleMap {
|
||||
db: Some(db),
|
||||
@@ -572,16 +577,14 @@ impl RedbNeedleMap {
|
||||
fn save_idx_size_meta(&self, idx_size: u64) -> io::Result<()> {
|
||||
let txn = Self::begin_write_no_fsync(self.db_or_err()?)?;
|
||||
{
|
||||
let mut meta = txn.open_table(META_TABLE).map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb open meta: {}", e))
|
||||
})?;
|
||||
meta.insert(META_IDX_SIZE, idx_size).map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb insert meta: {}", e))
|
||||
})?;
|
||||
let mut meta = txn
|
||||
.open_table(META_TABLE)
|
||||
.map_err(|e| io::Error::other(format!("redb open meta: {}", e)))?;
|
||||
meta.insert(META_IDX_SIZE, idx_size)
|
||||
.map_err(|e| io::Error::other(format!("redb insert meta: {}", e)))?;
|
||||
}
|
||||
txn.commit().map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb commit meta: {}", e))
|
||||
})?;
|
||||
txn.commit()
|
||||
.map_err(|e| io::Error::other(format!("redb commit meta: {}", e)))?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -590,22 +593,26 @@ impl RedbNeedleMap {
|
||||
let txn = self
|
||||
.db_or_err()?
|
||||
.begin_read()
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("redb begin_read: {}", e)))?;
|
||||
.map_err(|e| io::Error::other(format!("redb begin_read: {}", e)))?;
|
||||
let meta = txn
|
||||
.open_table(META_TABLE)
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("redb open meta: {}", e)))?;
|
||||
.map_err(|e| io::Error::other(format!("redb open meta: {}", e)))?;
|
||||
// experimental-api-5 drops inherent ReadOnlyTable::get ('static guard).
|
||||
// ReadableTable::get guard borrows `meta`; bind the match so the
|
||||
// temporary Result is dropped before `meta`.
|
||||
let result = match meta.get(META_IDX_SIZE) {
|
||||
// ReadableTable::get guard borrows `meta`; edition 2024 drops the tail
|
||||
// expression's temporaries before `meta`, so no extra binding is needed.
|
||||
match meta.get(META_IDX_SIZE) {
|
||||
Ok(Some(guard)) => Ok(Some(guard.value())),
|
||||
Ok(None) => Ok(None),
|
||||
Err(e) => Err(io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!("redb get meta: {}", e),
|
||||
)),
|
||||
};
|
||||
result
|
||||
Err(e) => Err(io::Error::other(format!("redb get meta: {}", e))),
|
||||
}
|
||||
}
|
||||
|
||||
/// Test-only read of META `idx_size` through the live handle. See
|
||||
/// [`test_support::live_meta_idx_size`] for why durability tests use
|
||||
/// this instead of copying the open `.rdb`.
|
||||
#[cfg(test)]
|
||||
pub(crate) fn live_meta_idx_size(&self) -> Option<u64> {
|
||||
self.read_idx_size_meta().unwrap()
|
||||
}
|
||||
|
||||
/// Load from an .idx file, reusing an existing .rdb if it is consistent.
|
||||
@@ -648,7 +655,7 @@ impl RedbNeedleMap {
|
||||
let db = Database::builder()
|
||||
.set_cache_size(cache_bytes)
|
||||
.open(db_path)
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("redb open: {}", e)))?;
|
||||
.map_err(|e| io::Error::other(format!("redb open: {}", e)))?;
|
||||
|
||||
let mut nm = RedbNeedleMap {
|
||||
db: Some(db),
|
||||
@@ -663,14 +670,11 @@ impl RedbNeedleMap {
|
||||
|
||||
let stored_idx_size = nm
|
||||
.read_idx_size_meta()?
|
||||
.ok_or_else(|| io::Error::new(io::ErrorKind::Other, "no idx_size in redb meta"))?;
|
||||
.ok_or_else(|| io::Error::other("no idx_size in redb meta"))?;
|
||||
|
||||
if stored_idx_size > idx_size {
|
||||
// .idx shrank — corrupted or truncated, need full rebuild
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
"idx file smaller than stored size",
|
||||
));
|
||||
return Err(io::Error::other("idx file smaller than stored size"));
|
||||
}
|
||||
|
||||
// Counters come from the whole .idx history, never from the table,
|
||||
@@ -683,40 +687,37 @@ impl RedbNeedleMap {
|
||||
let start_entry = stored_idx_size / NEEDLE_MAP_ENTRY_SIZE as u64;
|
||||
let txn = Self::begin_write_no_fsync(nm.db.as_ref().unwrap())?;
|
||||
{
|
||||
let mut table = txn.open_table(NEEDLE_TABLE).map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb open_table: {}", e))
|
||||
})?;
|
||||
let mut table = txn
|
||||
.open_table(NEEDLE_TABLE)
|
||||
.map_err(|e| io::Error::other(format!("redb open_table: {}", e)))?;
|
||||
idx::walk_index_file(reader, start_entry, |key, offset, size| {
|
||||
let key_u64: u64 = key.into();
|
||||
if offset.is_zero() || size.is_deleted() {
|
||||
// Delete: store a tombstone (negative size, original
|
||||
// offset) over a live value; already deleted is a no-op.
|
||||
if let Ok(Some(old)) = nm.get_via_table(&table, key_u64) {
|
||||
if old.size.is_valid() {
|
||||
let deleted_nv = NeedleValue {
|
||||
offset: old.offset,
|
||||
size: Size(-(old.size.0)),
|
||||
};
|
||||
let packed = pack_needle_value(&deleted_nv);
|
||||
table.insert(key_u64, packed.as_slice()).map_err(|e| {
|
||||
io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!("redb insert: {}", e),
|
||||
)
|
||||
})?;
|
||||
}
|
||||
if let Ok(Some(old)) = nm.get_via_table(&table, key_u64)
|
||||
&& old.size.is_valid()
|
||||
{
|
||||
let deleted_nv = NeedleValue {
|
||||
offset: old.offset,
|
||||
size: Size(-(old.size.0)),
|
||||
};
|
||||
let packed = pack_needle_value(&deleted_nv);
|
||||
table
|
||||
.insert(key_u64, packed.as_slice())
|
||||
.map_err(|e| io::Error::other(format!("redb insert: {}", e)))?;
|
||||
}
|
||||
} else {
|
||||
let packed = pack_needle_value(&NeedleValue { offset, size });
|
||||
table.insert(key_u64, packed.as_slice()).map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb insert: {}", e))
|
||||
})?;
|
||||
table
|
||||
.insert(key_u64, packed.as_slice())
|
||||
.map_err(|e| io::Error::other(format!("redb insert: {}", e)))?;
|
||||
}
|
||||
Ok(())
|
||||
})?;
|
||||
}
|
||||
txn.commit()
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("redb commit: {}", e)))?;
|
||||
.map_err(|e| io::Error::other(format!("redb commit: {}", e)))?;
|
||||
|
||||
nm.save_idx_size_meta(idx_size)?;
|
||||
}
|
||||
@@ -734,10 +735,7 @@ impl RedbNeedleMap {
|
||||
match table.get(key_u64) {
|
||||
Ok(Some(guard)) => Ok(packed_to_needle_value(guard.value())),
|
||||
Ok(None) => Ok(None),
|
||||
Err(e) => Err(io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!("redb get: {}", e),
|
||||
)),
|
||||
Err(e) => Err(io::Error::other(format!("redb get: {}", e))),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -790,13 +788,13 @@ impl RedbNeedleMap {
|
||||
|
||||
let txn = Self::begin_write_no_fsync(nm.db.as_ref().unwrap())?;
|
||||
{
|
||||
let mut table = txn.open_table(NEEDLE_TABLE).map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb open_table: {}", e))
|
||||
})?;
|
||||
let mut table = txn
|
||||
.open_table(NEEDLE_TABLE)
|
||||
.map_err(|e| io::Error::other(format!("redb open_table: {}", e)))?;
|
||||
if !unlinked {
|
||||
table.retain(|_, _| false).map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb retain: {}", e))
|
||||
})?;
|
||||
table
|
||||
.retain(|_, _| false)
|
||||
.map_err(|e| io::Error::other(format!("redb retain: {}", e)))?;
|
||||
}
|
||||
|
||||
#[cfg(not(feature = "redb-experimental-cursor"))]
|
||||
@@ -804,30 +802,33 @@ impl RedbNeedleMap {
|
||||
for (key, nv) in &entries {
|
||||
let key_u64: u64 = (*key).into();
|
||||
let packed = pack_needle_value(nv);
|
||||
table.insert(key_u64, packed.as_slice()).map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb insert: {}", e))
|
||||
})?;
|
||||
table
|
||||
.insert(key_u64, packed.as_slice())
|
||||
.map_err(|e| io::Error::other(format!("redb insert: {}", e)))?;
|
||||
}
|
||||
}
|
||||
#[cfg(feature = "redb-experimental-cursor")]
|
||||
{
|
||||
let mut cursor = table
|
||||
.upper_bound_mut(Bound::<u64>::Unbounded)
|
||||
.map_err(|e| {
|
||||
io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!("redb upper_bound_mut: {}", e),
|
||||
)
|
||||
})?;
|
||||
let mut cursor =
|
||||
table
|
||||
.upper_bound_mut(Bound::<u64>::Unbounded)
|
||||
.map_err(|e| {
|
||||
io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!("redb upper_bound_mut: {}", e),
|
||||
)
|
||||
})?;
|
||||
for (key, nv) in &entries {
|
||||
let key_u64: u64 = (*key).into();
|
||||
let packed = pack_needle_value(nv);
|
||||
cursor.insert_before(key_u64, packed.as_slice()).map_err(|e| {
|
||||
io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!("redb insert_before: {}", e),
|
||||
)
|
||||
})?;
|
||||
cursor
|
||||
.insert_before(key_u64, packed.as_slice())
|
||||
.map_err(|e| {
|
||||
io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!("redb insert_before: {}", e),
|
||||
)
|
||||
})?;
|
||||
}
|
||||
cursor.close().map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb cursor close: {}", e))
|
||||
@@ -835,7 +836,7 @@ impl RedbNeedleMap {
|
||||
}
|
||||
}
|
||||
txn.commit()
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("redb commit: {}", e)))?;
|
||||
.map_err(|e| io::Error::other(format!("redb commit: {}", e)))?;
|
||||
|
||||
nm.save_idx_size_meta(idx_size)?;
|
||||
Ok(())
|
||||
@@ -901,23 +902,16 @@ impl RedbNeedleMap {
|
||||
Ok(t) => t,
|
||||
Err(e) => {
|
||||
self.truncate_idx_to_offset();
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!("redb open_table: {}", e),
|
||||
));
|
||||
return Err(io::Error::other(format!("redb open_table: {}", e)));
|
||||
}
|
||||
};
|
||||
let result = match table.insert(key_u64, packed.as_slice()) {
|
||||
match table.insert(key_u64, packed.as_slice()) {
|
||||
Ok(prev) => prev.and_then(|g| packed_to_needle_value(g.value())),
|
||||
Err(e) => {
|
||||
self.truncate_idx_to_offset();
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!("redb insert: {}", e),
|
||||
));
|
||||
return Err(io::Error::other(format!("redb insert: {}", e)));
|
||||
}
|
||||
};
|
||||
result
|
||||
}
|
||||
};
|
||||
match txn.commit() {
|
||||
Ok(()) => old,
|
||||
@@ -925,8 +919,7 @@ impl RedbNeedleMap {
|
||||
// Transaction rolled back, database still usable:
|
||||
// truncate the orphan .idx row.
|
||||
self.truncate_idx_to_offset();
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
return Err(io::Error::other(
|
||||
"redb commit: Transaction was poisoned by a panic",
|
||||
));
|
||||
}
|
||||
@@ -935,12 +928,9 @@ impl RedbNeedleMap {
|
||||
// visible and redb refuses further writes. Keep
|
||||
// the .idx row (do NOT truncate) and reopen from
|
||||
// .idx to repair redb's internal state.
|
||||
let err = io::Error::new(io::ErrorKind::Other, format!("redb commit: {}", e));
|
||||
let err = io::Error::other(format!("redb commit: {}", e));
|
||||
if let Err(reopen_err) = self.reopen_from_idx() {
|
||||
tracing::warn!(
|
||||
"redb reopen after put commit error failed: {}",
|
||||
reopen_err
|
||||
);
|
||||
tracing::warn!("redb reopen after put commit error failed: {}", reopen_err);
|
||||
}
|
||||
return Err(err);
|
||||
}
|
||||
@@ -968,39 +958,32 @@ impl RedbNeedleMap {
|
||||
let txn = self
|
||||
.db_or_err()?
|
||||
.begin_read()
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("redb begin_read: {}", e)))?;
|
||||
.map_err(|e| io::Error::other(format!("redb begin_read: {}", e)))?;
|
||||
let table = txn
|
||||
.open_table(NEEDLE_TABLE)
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("redb open_table: {}", e)))?;
|
||||
.map_err(|e| io::Error::other(format!("redb open_table: {}", e)))?;
|
||||
// experimental-api-5 drops inherent ReadOnlyTable::get ('static guard).
|
||||
// ReadableTable::get guard borrows `table`; bind the match so the
|
||||
// temporary Result is dropped before `table`.
|
||||
let result = match table.get(key_u64) {
|
||||
// ReadableTable::get guard borrows `table`; edition 2024 drops the tail
|
||||
// expression's temporaries before `table`, so no extra binding is needed.
|
||||
match table.get(key_u64) {
|
||||
Ok(Some(guard)) => Ok(packed_to_needle_value(guard.value())),
|
||||
Ok(None) => Ok(None),
|
||||
Err(e) => Err(io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!("redb get: {}", e),
|
||||
)),
|
||||
};
|
||||
result
|
||||
Err(e) => Err(io::Error::other(format!("redb get: {}", e))),
|
||||
}
|
||||
}
|
||||
|
||||
/// Mark a needle as deleted. Appends tombstone to .idx file, negates size in redb.
|
||||
pub fn delete(&mut self, key: NeedleId, offset: Offset) -> io::Result<Option<Size>> {
|
||||
let key_u64: u64 = key.into();
|
||||
let txn = Self::begin_write_no_fsync(self.db_or_err()?)?;
|
||||
let mut table = txn.open_table(NEEDLE_TABLE).map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb open_table: {}", e))
|
||||
})?;
|
||||
let mut table = txn
|
||||
.open_table(NEEDLE_TABLE)
|
||||
.map_err(|e| io::Error::other(format!("redb open_table: {}", e)))?;
|
||||
let old = match table.get(key_u64) {
|
||||
Ok(Some(guard)) => packed_to_needle_value(guard.value()),
|
||||
Ok(None) => None,
|
||||
Err(e) => {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!("redb get: {}", e),
|
||||
));
|
||||
return Err(io::Error::other(format!("redb get: {}", e)));
|
||||
}
|
||||
};
|
||||
let Some(old) = old.filter(|nv| nv.size.is_valid()) else {
|
||||
@@ -1021,10 +1004,7 @@ impl RedbNeedleMap {
|
||||
drop(table);
|
||||
if let Err(e) = insert_res {
|
||||
self.truncate_idx_to_offset();
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!("redb insert: {}", e),
|
||||
));
|
||||
return Err(io::Error::other(format!("redb insert: {}", e)));
|
||||
}
|
||||
match txn.commit() {
|
||||
Ok(()) => {}
|
||||
@@ -1032,8 +1012,7 @@ impl RedbNeedleMap {
|
||||
// Transaction rolled back, database still usable:
|
||||
// truncate the orphan .idx row.
|
||||
self.truncate_idx_to_offset();
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
return Err(io::Error::other(
|
||||
"redb commit: Transaction was poisoned by a panic",
|
||||
));
|
||||
}
|
||||
@@ -1042,7 +1021,7 @@ impl RedbNeedleMap {
|
||||
// and redb refuses further writes. Keep the .idx row
|
||||
// (do NOT truncate) and reopen from .idx to repair
|
||||
// redb's internal state.
|
||||
let err = io::Error::new(io::ErrorKind::Other, format!("redb commit: {}", e));
|
||||
let err = io::Error::other(format!("redb commit: {}", e));
|
||||
if let Err(reopen_err) = self.reopen_from_idx() {
|
||||
tracing::warn!(
|
||||
"redb reopen after delete commit error failed: {}",
|
||||
@@ -1105,10 +1084,10 @@ impl RedbNeedleMap {
|
||||
/// after the orphan, `idx_file_offset` advances past it, and a later
|
||||
/// checkpoint records an offset that makes the reload skip the orphan.
|
||||
fn truncate_idx_to_offset(&mut self) {
|
||||
if let Some(ref mut idx_file) = self.idx_file {
|
||||
if let Err(e) = idx_file.truncate_to(self.idx_file_offset) {
|
||||
tracing::warn!("failed to truncate orphan .idx row: {}", e);
|
||||
}
|
||||
if let Some(ref mut idx_file) = self.idx_file
|
||||
&& let Err(e) = idx_file.truncate_to(self.idx_file_offset)
|
||||
{
|
||||
tracing::warn!("failed to truncate orphan .idx row: {}", e);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1146,18 +1125,12 @@ impl RedbNeedleMap {
|
||||
let read_file = std::fs::OpenOptions::new()
|
||||
.read(true)
|
||||
.open(&idx_path)
|
||||
.map_err(|e| {
|
||||
io::Error::other(format!("reopen: open .idx {}: {}", idx_path, e))
|
||||
})?;
|
||||
.map_err(|e| io::Error::other(format!("reopen: open .idx {}: {}", idx_path, e)))?;
|
||||
let actual_idx_size = read_file.metadata()?.len();
|
||||
let mut reader = io::BufReader::new(read_file);
|
||||
|
||||
let reopened = Self::load_from_idx(
|
||||
&self.rdb_path,
|
||||
&mut reader,
|
||||
self.version,
|
||||
self.cache_bytes,
|
||||
)?;
|
||||
let reopened =
|
||||
Self::load_from_idx(&self.rdb_path, &mut reader, self.version, self.cache_bytes)?;
|
||||
|
||||
// Preserve the append writer and the paths/version/cache; adopt the
|
||||
// repaired database, metrics, and idx_file_offset from the reload.
|
||||
@@ -1198,10 +1171,10 @@ impl RedbNeedleMap {
|
||||
let txn = self
|
||||
.db_or_err()?
|
||||
.begin_read()
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("redb begin_read: {}", e)))?;
|
||||
.map_err(|e| io::Error::other(format!("redb begin_read: {}", e)))?;
|
||||
let table = txn
|
||||
.open_table(NEEDLE_TABLE)
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("redb open_table: {}", e)))?;
|
||||
.map_err(|e| io::Error::other(format!("redb open_table: {}", e)))?;
|
||||
|
||||
let mut file = std::fs::OpenOptions::new()
|
||||
.write(true)
|
||||
@@ -1212,18 +1185,17 @@ impl RedbNeedleMap {
|
||||
// redb iterates in key order (u64 ascending)
|
||||
let iter = table
|
||||
.iter()
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("redb iter: {}", e)))?;
|
||||
.map_err(|e| io::Error::other(format!("redb iter: {}", e)))?;
|
||||
|
||||
for entry in iter {
|
||||
let (key_guard, val_guard) = entry.map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb iter next: {}", e))
|
||||
})?;
|
||||
let (key_guard, val_guard) =
|
||||
entry.map_err(|e| io::Error::other(format!("redb iter next: {}", e)))?;
|
||||
let key_u64: u64 = key_guard.value();
|
||||
let bytes: &[u8] = val_guard.value();
|
||||
if let Some(nv) = packed_to_needle_value(bytes) {
|
||||
if nv.size.is_valid() {
|
||||
idx::write_index_entry(&mut file, NeedleId(key_u64), nv.offset, nv.size)?;
|
||||
}
|
||||
if let Some(nv) = packed_to_needle_value(bytes)
|
||||
&& nv.size.is_valid()
|
||||
{
|
||||
idx::write_index_entry(&mut file, NeedleId(key_u64), nv.offset, nv.size)?;
|
||||
}
|
||||
}
|
||||
file.sync_all()?;
|
||||
@@ -1512,22 +1484,30 @@ impl NeedleMap {
|
||||
pub(crate) mod test_support {
|
||||
use super::*;
|
||||
|
||||
/// The `.idx` size recorded in the durable state of the `.rdb` at
|
||||
/// `rdb_path`, read from a copy taken while the map may still be open:
|
||||
/// exactly what a crash would leave behind. `None` when nothing durable
|
||||
/// has been recorded yet.
|
||||
pub(crate) fn durable_idx_size(rdb_path: &Path) -> Option<u64> {
|
||||
let copy = rdb_path.with_extension("crash-copy.rdb");
|
||||
std::fs::copy(rdb_path, ©).unwrap();
|
||||
let db = Database::open(©).unwrap();
|
||||
let txn = db.begin_read().unwrap();
|
||||
let meta = txn.open_table(META_TABLE).ok()?;
|
||||
let size = meta.get(META_IDX_SIZE).unwrap().map(|g| g.value());
|
||||
drop(meta);
|
||||
drop(txn);
|
||||
drop(db);
|
||||
let _ = std::fs::remove_file(©);
|
||||
size
|
||||
/// The `.idx` size in the map's META table, read through the live
|
||||
/// handle.
|
||||
///
|
||||
/// The load path records the `.idx` size with `Durability::None`, and
|
||||
/// every `put`/`delete` also commits non-durably, so before the first
|
||||
/// checkpoint this is the load-time value (`Some(0)` for a fresh map) —
|
||||
/// NOT the crash-durable `None` a copy of the open `.rdb` would show.
|
||||
/// A live read is the only portable observation: redb 4.2.0 takes an
|
||||
/// exclusive whole-file lock, which is advisory on Unix but mandatory
|
||||
/// on Windows, so copying the open `.rdb` fails there with OS error 33.
|
||||
///
|
||||
/// It still pins the property under test: the only *durable* META
|
||||
/// writer is `checkpoint`, so any value other than the load-time one
|
||||
/// proves a checkpoint recorded progress — and the post-checkpoint
|
||||
/// value equals the durable one, because checkpoints commit with
|
||||
/// `Durability::Immediate`. What is lost vs the old copy: strict crash
|
||||
/// fidelity — a hard crash pre-checkpoint would leave META absent
|
||||
/// rather than `Some(0)` (loader-equivalent outcomes: full rebuild vs
|
||||
/// replay-from-0, both correct). A clean close+reopen cannot recover
|
||||
/// that distinction either: dropping the `Database` flushes pending
|
||||
/// non-durable commits, so a reopened handle reads `Some(0)` just like
|
||||
/// the live one.
|
||||
pub(crate) fn live_meta_idx_size(nm: &RedbNeedleMap) -> Option<u64> {
|
||||
nm.live_meta_idx_size()
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1682,6 +1662,7 @@ mod tests {
|
||||
.read(true)
|
||||
.write(true)
|
||||
.create(true)
|
||||
.truncate(false)
|
||||
.open(&idx_path)
|
||||
.unwrap();
|
||||
let idx_size = idx_file.metadata().unwrap().len();
|
||||
@@ -2168,8 +2149,14 @@ mod tests {
|
||||
// server opens one redb database per volume, so the process-wide
|
||||
// ceiling is roughly (volumes x budget).
|
||||
assert_eq!(NeedleMapKind::Redb.redb_cache_bytes(), 4 * 1024 * 1024);
|
||||
assert_eq!(NeedleMapKind::RedbMedium.redb_cache_bytes(), 8 * 1024 * 1024);
|
||||
assert_eq!(NeedleMapKind::RedbLarge.redb_cache_bytes(), 16 * 1024 * 1024);
|
||||
assert_eq!(
|
||||
NeedleMapKind::RedbMedium.redb_cache_bytes(),
|
||||
8 * 1024 * 1024
|
||||
);
|
||||
assert_eq!(
|
||||
NeedleMapKind::RedbLarge.redb_cache_bytes(),
|
||||
16 * 1024 * 1024
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -2198,7 +2185,7 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn test_redb_checkpoint_is_explicit_and_due_every_interval() {
|
||||
use test_support::durable_idx_size;
|
||||
use test_support::live_meta_idx_size;
|
||||
|
||||
// Every non-durable redb commit leaves bookkeeping behind until a
|
||||
// durable one clears it, so a writable map asks for a checkpoint on
|
||||
@@ -2208,8 +2195,12 @@ mod tests {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let (mut nm, db_path, _idx_path) = open_writable_redb(dir.path());
|
||||
for i in 1..EXPECTED_INTERVAL {
|
||||
nm.put(NeedleId(i), Offset::from_actual_offset((i * 8) as i64), Size(1))
|
||||
.unwrap();
|
||||
nm.put(
|
||||
NeedleId(i),
|
||||
Offset::from_actual_offset((i * 8) as i64),
|
||||
Size(1),
|
||||
)
|
||||
.unwrap();
|
||||
assert!(!nm.checkpoint_due(), "due after only {i} writes");
|
||||
}
|
||||
nm.put(
|
||||
@@ -2219,20 +2210,29 @@ mod tests {
|
||||
)
|
||||
.unwrap();
|
||||
assert!(nm.checkpoint_due());
|
||||
assert_eq!(durable_idx_size(&db_path), None, "put() must not commit durably");
|
||||
// No checkpoint taken yet: META still holds the load-time .idx size.
|
||||
// put() only commits non-durably, so the live value is unchanged.
|
||||
assert_eq!(
|
||||
live_meta_idx_size(&nm),
|
||||
Some(0),
|
||||
"put() must not record checkpoint progress"
|
||||
);
|
||||
|
||||
nm.checkpoint(true).unwrap();
|
||||
assert!(!nm.checkpoint_due());
|
||||
assert_eq!(
|
||||
durable_idx_size(&db_path),
|
||||
live_meta_idx_size(&nm),
|
||||
Some(EXPECTED_INTERVAL * NEEDLE_MAP_ENTRY_SIZE as u64),
|
||||
"checkpoint records how much of the .idx the table reflects"
|
||||
);
|
||||
|
||||
// Snapshot the .rdb while the map is still open: what a crash leaves.
|
||||
// Everything is durable after the checkpoint, so the map is closed
|
||||
// first and the snapshot sees the same bytes on every platform.
|
||||
// (Copying while open fails on Windows, where redb's file lock is
|
||||
// mandatory: what a crash leaves.)
|
||||
drop(nm);
|
||||
let crash_copy = dir.path().join("crash.rdb");
|
||||
std::fs::copy(&db_path, &crash_copy).unwrap();
|
||||
drop(nm);
|
||||
let db = Database::open(&crash_copy).unwrap();
|
||||
let txn = db.begin_read().unwrap();
|
||||
let table = txn.open_table(NEEDLE_TABLE).unwrap();
|
||||
@@ -2247,8 +2247,12 @@ mod tests {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let (mut nm, db_path, idx_path) = open_writable_redb(dir.path());
|
||||
for i in 1..=5u64 {
|
||||
nm.put(NeedleId(i), Offset::from_actual_offset((i * 8) as i64), Size(1))
|
||||
.unwrap();
|
||||
nm.put(
|
||||
NeedleId(i),
|
||||
Offset::from_actual_offset((i * 8) as i64),
|
||||
Size(1),
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
nm.close();
|
||||
drop(nm);
|
||||
@@ -2272,8 +2276,12 @@ mod tests {
|
||||
let dir = tempfile::tempdir().unwrap();
|
||||
let (mut nm, db_path, idx_path) = open_writable_redb(dir.path());
|
||||
for i in 1..=5u64 {
|
||||
nm.put(NeedleId(i), Offset::from_actual_offset((i * 8) as i64), Size(1))
|
||||
.unwrap();
|
||||
nm.put(
|
||||
NeedleId(i),
|
||||
Offset::from_actual_offset((i * 8) as i64),
|
||||
Size(1),
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
// Drop without close(): redb makes the table durable on drop, but the
|
||||
// recorded .idx size stays at its load-time value (0), so the reload
|
||||
@@ -2344,12 +2352,14 @@ mod tests {
|
||||
reloaded.deleted_count(),
|
||||
reloaded.deleted_size(),
|
||||
);
|
||||
assert_eq!(
|
||||
after, live,
|
||||
"close_first={close_first} rebuild={rebuild}"
|
||||
);
|
||||
assert_eq!(after, live, "close_first={close_first} rebuild={rebuild}");
|
||||
assert_eq!(reloaded.get(NeedleId(1)).unwrap().unwrap().size, Size(200));
|
||||
assert!(reloaded.get(NeedleId(2)).unwrap().map_or(true, |v| v.size.is_deleted()));
|
||||
assert!(
|
||||
reloaded
|
||||
.get(NeedleId(2))
|
||||
.unwrap()
|
||||
.is_none_or(|v| v.size.is_deleted())
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -31,7 +31,7 @@ struct CompactEntry {
|
||||
}
|
||||
|
||||
impl CompactEntry {
|
||||
fn to_needle_value(&self) -> NeedleValue {
|
||||
fn to_needle_value(self) -> NeedleValue {
|
||||
NeedleValue {
|
||||
offset: Offset::from_bytes(&self.offset),
|
||||
size: self.size,
|
||||
|
||||
@@ -205,6 +205,18 @@ mod tests {
|
||||
use std::os::unix::fs::FileExt;
|
||||
borrowed.read_exact_at(&mut buf, 0).unwrap();
|
||||
}
|
||||
#[cfg(windows)]
|
||||
{
|
||||
use std::os::windows::fs::FileExt;
|
||||
let mut filled = 0;
|
||||
let mut at = 0;
|
||||
while filled < buf.len() {
|
||||
let n = borrowed.seek_read(&mut buf[filled..], at).unwrap();
|
||||
assert!(n != 0, "unexpected EOF in seek_read");
|
||||
filled += n;
|
||||
at += n as u64;
|
||||
}
|
||||
}
|
||||
assert_eq!(&buf, b"first");
|
||||
}
|
||||
|
||||
|
||||
@@ -132,9 +132,7 @@ mod tests {
|
||||
let mut seen = SeenKeys::new(10_000, FALSE_POSITIVE_RATE);
|
||||
// Fresh keys may occasionally collide (that is the false-positive
|
||||
// rate), but only rarely.
|
||||
let fresh_reported_seen = (0..10_000u64)
|
||||
.filter(|&key| seen.test_and_add(key))
|
||||
.count();
|
||||
let fresh_reported_seen = (0..10_000u64).filter(|&key| seen.test_and_add(key)).count();
|
||||
assert!(
|
||||
fresh_reported_seen < 50,
|
||||
"fresh keys reported seen: {fresh_reported_seen}"
|
||||
|
||||
@@ -18,6 +18,7 @@ use std::sync::{Mutex, RwLock};
|
||||
|
||||
use super::file_pool::pooled_index_files;
|
||||
use crate::storage::idx;
|
||||
use crate::storage::io::read_exact_at;
|
||||
use crate::storage::needle_map::{CompactNeedleMap, NeedleMapMetric, NeedleValue};
|
||||
use crate::storage::types::*;
|
||||
|
||||
@@ -133,9 +134,7 @@ impl SortedFileNeedleMap {
|
||||
}
|
||||
let file = pooled_index_files()
|
||||
.borrow(&self.db_file_name, false)
|
||||
.map_err(|e| {
|
||||
io::Error::new(e.kind(), format!("open {}: {}", self.db_file_name, e))
|
||||
})?;
|
||||
.map_err(|e| io::Error::new(e.kind(), format!("open {}: {}", self.db_file_name, e)))?;
|
||||
match search_sorted_index(&file, self.db_file_size, key)? {
|
||||
Some((_, offset, size)) => Ok(Some(NeedleValue { offset, size })),
|
||||
None => Ok(None),
|
||||
@@ -226,10 +225,7 @@ impl SortedFileNeedleMap {
|
||||
.fail_sdx_mark
|
||||
.load(std::sync::atomic::Ordering::Relaxed)
|
||||
{
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
"injected .sdx mark failure",
|
||||
));
|
||||
return Err(io::Error::other("injected .sdx mark failure"));
|
||||
}
|
||||
let mut buf = [0u8; SIZE_SIZE];
|
||||
TOMBSTONE_FILE_SIZE.to_bytes(&mut buf);
|
||||
@@ -309,7 +305,7 @@ impl SortedFileNeedleMap {
|
||||
let rows = rows_per_read.min(entry_count - done) as usize;
|
||||
let bytes = &mut block[..rows * NEEDLE_MAP_ENTRY_SIZE];
|
||||
read_exact_at(&file, bytes, done * NEEDLE_MAP_ENTRY_SIZE as u64)?;
|
||||
for entry in bytes.chunks_exact(NEEDLE_MAP_ENTRY_SIZE) {
|
||||
for entry in bytes.as_chunks::<NEEDLE_MAP_ENTRY_SIZE>().0 {
|
||||
let (key, offset, size) = idx_entry_from_bytes(entry);
|
||||
if !size.is_valid() || pending.contains_key(&key) {
|
||||
continue; // deleted in place, or still awaiting that mark
|
||||
@@ -525,32 +521,6 @@ fn search_sorted_index(
|
||||
Ok(None)
|
||||
}
|
||||
|
||||
fn read_exact_at(file: &File, buf: &mut [u8], offset: u64) -> io::Result<()> {
|
||||
#[cfg(unix)]
|
||||
{
|
||||
use std::os::unix::fs::FileExt;
|
||||
file.read_exact_at(buf, offset)
|
||||
}
|
||||
#[cfg(windows)]
|
||||
{
|
||||
use std::os::windows::fs::FileExt;
|
||||
let mut filled = 0;
|
||||
let mut at = offset;
|
||||
while filled < buf.len() {
|
||||
let n = file.seek_read(&mut buf[filled..], at)?;
|
||||
if n == 0 {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::UnexpectedEof,
|
||||
"unexpected EOF in seek_read",
|
||||
));
|
||||
}
|
||||
filled += n;
|
||||
at += n as u64;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
}
|
||||
|
||||
fn write_at(file: &File, buf: &[u8], offset: u64) -> io::Result<()> {
|
||||
#[cfg(unix)]
|
||||
{
|
||||
@@ -703,10 +673,11 @@ mod tests {
|
||||
// without a reload — the same contract Go's Get has, where callers
|
||||
// check size.is_deleted().
|
||||
assert!(m.get(NeedleId(2)).unwrap().unwrap().size.is_deleted());
|
||||
assert!(m
|
||||
.delete(NeedleId(2), Offset::from_actual_offset(16))
|
||||
.unwrap()
|
||||
.is_none());
|
||||
assert!(
|
||||
m.delete(NeedleId(2), Offset::from_actual_offset(16))
|
||||
.unwrap()
|
||||
.is_none()
|
||||
);
|
||||
assert!(!m.get(NeedleId(1)).unwrap().unwrap().size.is_deleted());
|
||||
}
|
||||
|
||||
@@ -1002,7 +973,8 @@ mod tests {
|
||||
|
||||
// The retry is a no-op: no second tombstone, no double counting.
|
||||
assert_eq!(
|
||||
m.delete(NeedleId(1), Offset::from_actual_offset(8)).unwrap(),
|
||||
m.delete(NeedleId(1), Offset::from_actual_offset(8))
|
||||
.unwrap(),
|
||||
None
|
||||
);
|
||||
assert_eq!(m.deleted_count(), deleted_before + 2);
|
||||
@@ -1032,7 +1004,8 @@ mod tests {
|
||||
);
|
||||
// And a retry must not append a second tombstone for it.
|
||||
assert_eq!(
|
||||
m.delete(NeedleId(1), Offset::from_actual_offset(8)).unwrap(),
|
||||
m.delete(NeedleId(1), Offset::from_actual_offset(8))
|
||||
.unwrap(),
|
||||
None
|
||||
);
|
||||
}
|
||||
|
||||
+627
-180
File diff suppressed because it is too large
Load Diff
@@ -8,7 +8,7 @@ use std::path::Path;
|
||||
|
||||
use tracing::{info, warn};
|
||||
|
||||
use crate::storage::disk_location::{parse_collection_volume_id_pub, DiskLocation};
|
||||
use crate::storage::disk_location::{DiskLocation, parse_collection_volume_id_pub};
|
||||
use crate::storage::store::Store;
|
||||
use crate::storage::types::VolumeId;
|
||||
|
||||
@@ -286,9 +286,7 @@ fn collect_shard_disk_volumes(loc: &DiskLocation) -> HashMap<EcKey, Vec<String>>
|
||||
let Some((collection, vid)) = parse_collection_volume_id_pub(base) else {
|
||||
continue;
|
||||
};
|
||||
out.entry(EcKey { collection, vid })
|
||||
.or_default()
|
||||
.push(name);
|
||||
out.entry(EcKey { collection, vid }).or_default().push(name);
|
||||
}
|
||||
out
|
||||
}
|
||||
|
||||
@@ -21,7 +21,7 @@ use std::fs;
|
||||
use tracing::{error, info, warn};
|
||||
|
||||
use crate::storage::disk_location::{is_ec_shard_extension, parse_collection_volume_id_pub};
|
||||
use crate::storage::erasure_coding::ec_shard::DATA_SHARDS_COUNT;
|
||||
use crate::storage::erasure_coding::ec_shard::{DATA_SHARDS_COUNT, ShardId};
|
||||
use crate::storage::store::Store;
|
||||
use crate::storage::types::VolumeId;
|
||||
|
||||
@@ -80,6 +80,11 @@ struct EcxOwnerInfo {
|
||||
idx_dir: String,
|
||||
}
|
||||
|
||||
/// One unit of reconcile work: the disk holding orphan shards, the volume
|
||||
/// they belong to, the shard files, the `.ecx` owner, and whether the
|
||||
/// mirror already installed sidecars locally (`use_local_idx`).
|
||||
type OrphanShardLoad = (usize, EcKey, Vec<(String, ShardId)>, EcxOwnerInfo, bool);
|
||||
|
||||
impl Store {
|
||||
/// Run cross-disk orphan-shard reconciliation. Should be called
|
||||
/// after every DiskLocation has finished its per-disk EC scan.
|
||||
@@ -98,7 +103,7 @@ impl Store {
|
||||
// `use_local_idx` is the post-mirror fast path: when the
|
||||
// mirror already installed sidecars locally, mount against
|
||||
// loc.idx_directory instead of the owner disk.
|
||||
let mut to_load: Vec<(usize, EcKey, Vec<(String, u32)>, EcxOwnerInfo, bool)> = Vec::new();
|
||||
let mut to_load: Vec<OrphanShardLoad> = Vec::new();
|
||||
for (loc_idx, loc) in self.locations.iter().enumerate() {
|
||||
let orphans = collect_orphan_ec_shards(loc, loc_idx);
|
||||
for (key, shards) in orphans {
|
||||
@@ -117,9 +122,7 @@ impl Store {
|
||||
let use_local_idx = std::path::Path::new(&local_ecx).exists()
|
||||
|| std::path::Path::new(&local_ecx_in_data).exists();
|
||||
|
||||
if !use_local_idx
|
||||
&& owner.location == loc_idx
|
||||
&& owner.idx_dir == loc.idx_directory
|
||||
if !use_local_idx && owner.location == loc_idx && owner.idx_dir == loc.idx_directory
|
||||
{
|
||||
// Same-disk no-op: load_all_ec_shards already
|
||||
// tried and logged the failure.
|
||||
@@ -132,7 +135,7 @@ impl Store {
|
||||
for (loc_idx, key, shards, owner, use_local_idx) in to_load {
|
||||
let shard_names: Vec<&str> = shards.iter().map(|(n, _)| n.as_str()).collect();
|
||||
let loc_dir = self.locations[loc_idx].directory.clone();
|
||||
let shard_ids: Vec<u32> = shards.iter().map(|(_, sid)| *sid).collect();
|
||||
let shard_ids: Vec<ShardId> = shards.iter().map(|(_, sid)| *sid).collect();
|
||||
|
||||
if use_local_idx {
|
||||
info!(
|
||||
@@ -293,10 +296,10 @@ impl Store {
|
||||
// may be sole copies of a distributed volume.
|
||||
let mut node_wide_bits = ev.shard_bits().0;
|
||||
for other in &self.locations {
|
||||
if let Some(other_ev) = other.find_ec_volume(*vid) {
|
||||
if other_ev.collection == ev.collection {
|
||||
node_wide_bits |= other_ev.shard_bits().0;
|
||||
}
|
||||
if let Some(other_ev) = other.find_ec_volume(*vid)
|
||||
&& other_ev.collection == ev.collection
|
||||
{
|
||||
node_wide_bits |= other_ev.shard_bits().0;
|
||||
}
|
||||
}
|
||||
let node_wide = node_wide_bits.count_ones() as usize;
|
||||
@@ -474,13 +477,13 @@ impl Store {
|
||||
/// Unlike `reconcile_ec_shards_across_disks` it needs no sibling disk, so a
|
||||
/// single-disk store recovers once its index has been fetched from a peer.
|
||||
fn load_orphan_ec_shards_with_local_index(&mut self) {
|
||||
let mut work: Vec<(usize, EcKey, Vec<u32>)> = Vec::new();
|
||||
let mut work: Vec<(usize, EcKey, Vec<ShardId>)> = Vec::new();
|
||||
for (loc_idx, loc) in self.locations.iter().enumerate() {
|
||||
for (key, shards) in collect_orphan_ec_shards(loc, loc_idx) {
|
||||
if !loc.has_ecx_file_on_disk(&key.collection, key.vid) {
|
||||
continue;
|
||||
}
|
||||
let ids: Vec<u32> = shards.iter().map(|(_, sid)| *sid).collect();
|
||||
let ids: Vec<ShardId> = shards.iter().map(|(_, sid)| *sid).collect();
|
||||
work.push((loc_idx, key, ids));
|
||||
}
|
||||
}
|
||||
@@ -499,6 +502,53 @@ impl Store {
|
||||
}
|
||||
}
|
||||
|
||||
/// Walk a disk's data directory and return the `.ec??` shard files
|
||||
/// that are present on disk but not yet registered in the location's
|
||||
/// `ec_volumes` map. Keyed by (collection, vid) so callers can match
|
||||
/// each group against its `.ecx`-owning disk in one lookup. Zero-byte
|
||||
/// shard files are ignored — same shape as `load_all_ec_shards`.
|
||||
fn collect_orphan_ec_shards(
|
||||
loc: &crate::storage::disk_location::DiskLocation,
|
||||
_loc_idx: usize,
|
||||
) -> HashMap<EcKey, Vec<(String, ShardId)>> {
|
||||
let mut orphans: HashMap<EcKey, Vec<(String, ShardId)>> = HashMap::new();
|
||||
let Ok(read) = fs::read_dir(&loc.directory) else {
|
||||
return orphans;
|
||||
};
|
||||
for ent in read.flatten() {
|
||||
if ent.file_type().map(|ft| ft.is_dir()).unwrap_or(false) {
|
||||
continue;
|
||||
}
|
||||
let name = ent.file_name().to_string_lossy().into_owned();
|
||||
let Some(dot) = name.rfind('.') else {
|
||||
continue;
|
||||
};
|
||||
let (base, ext) = name.split_at(dot);
|
||||
let Some(shard_id) = is_ec_shard_extension(ext) else {
|
||||
continue;
|
||||
};
|
||||
// Ignore zero-byte shards. Use the DirEntry's metadata so we
|
||||
// don't pay a second stat syscall per file beyond what
|
||||
// read_dir already returned.
|
||||
match ent.metadata() {
|
||||
Ok(meta) if meta.len() > 0 => {}
|
||||
_ => continue,
|
||||
}
|
||||
let Some((collection, vid)) = parse_collection_volume_id_pub(base) else {
|
||||
continue;
|
||||
};
|
||||
// Skip shards that are already registered to an EcVolume.
|
||||
if let Some(ecv) = loc.find_ec_volume(vid)
|
||||
&& ecv.has_shard(shard_id)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
let key = EcKey { collection, vid };
|
||||
orphans.entry(key).or_default().push((name, shard_id));
|
||||
}
|
||||
orphans
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
@@ -541,7 +591,13 @@ mod tests {
|
||||
std::fs::write(&p, b"shard data nonempty").unwrap();
|
||||
}
|
||||
|
||||
fn write_index_files(idx_dir: &str, collection: &str, vid: u32, data_shards: u32, parity_shards: u32) {
|
||||
fn write_index_files(
|
||||
idx_dir: &str,
|
||||
collection: &str,
|
||||
vid: u32,
|
||||
data_shards: u32,
|
||||
parity_shards: u32,
|
||||
) {
|
||||
// Minimal sealed .ecx (the loader only opens the file; it
|
||||
// doesn't parse it during placement).
|
||||
std::fs::write(
|
||||
@@ -895,15 +951,24 @@ mod tests {
|
||||
// dir1 owns the .ecx and so already has shard 1 mounted via
|
||||
// its own load_all_ec_shards.
|
||||
let ev1 = store.locations[1].find_ec_volume(VolumeId(vid));
|
||||
assert!(ev1.is_some(), "baseline broken: dir1 should have mounted shard 1");
|
||||
assert!(
|
||||
ev1.is_some(),
|
||||
"baseline broken: dir1 should have mounted shard 1"
|
||||
);
|
||||
|
||||
// dir0's shards must be reconciled across to its own
|
||||
// ec_volumes map, pointing at dir1's idx dir.
|
||||
let ev0 = store.locations[0]
|
||||
.find_ec_volume(VolumeId(vid))
|
||||
.expect("dir0 should now have an EcVolume after reconcile");
|
||||
assert!(ev0.has_shard(0), "shard 0 missing from dir0 after reconcile");
|
||||
assert!(ev0.has_shard(12), "shard 12 missing from dir0 after reconcile");
|
||||
assert!(
|
||||
ev0.has_shard(0),
|
||||
"shard 0 missing from dir0 after reconcile"
|
||||
);
|
||||
assert!(
|
||||
ev0.has_shard(12),
|
||||
"shard 12 missing from dir0 after reconcile"
|
||||
);
|
||||
}
|
||||
|
||||
/// PR 9244 review case: idx_directory is configured but the
|
||||
@@ -1012,7 +1077,13 @@ mod tests {
|
||||
assert!(store.locations[0].find_ec_volume(VolumeId(vid)).is_none());
|
||||
// Shard files must still exist on disk for operator recovery.
|
||||
for sid in [0u8, 12u8] {
|
||||
let p = format!("{}/{}_{}.ec{:02}", dir0.to_str().unwrap(), collection, vid, sid);
|
||||
let p = format!(
|
||||
"{}/{}_{}.ec{:02}",
|
||||
dir0.to_str().unwrap(),
|
||||
collection,
|
||||
vid,
|
||||
sid
|
||||
);
|
||||
assert!(
|
||||
std::path::Path::new(&p).exists(),
|
||||
"orphan shard {} was destroyed",
|
||||
@@ -1077,10 +1148,12 @@ mod tests {
|
||||
assert!(ev1.has_shard(6), "dir1 shard missing");
|
||||
|
||||
// Nothing left to recover.
|
||||
assert!(store
|
||||
.collect_ec_volumes_missing_index()
|
||||
.iter()
|
||||
.all(|m| m.vid != VolumeId(vid)));
|
||||
assert!(
|
||||
store
|
||||
.collect_ec_volumes_missing_index()
|
||||
.iter()
|
||||
.all(|m| m.vid != VolumeId(vid))
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -1112,10 +1185,12 @@ mod tests {
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
assert!(store
|
||||
.collect_ec_volumes_missing_index()
|
||||
.iter()
|
||||
.all(|m| m.vid != VolumeId(vid)));
|
||||
assert!(
|
||||
store
|
||||
.collect_ec_volumes_missing_index()
|
||||
.iter()
|
||||
.all(|m| m.vid != VolumeId(vid))
|
||||
);
|
||||
}
|
||||
|
||||
/// Helper: build a 2-disk store where reconcile produces the
|
||||
@@ -1184,6 +1259,87 @@ mod tests {
|
||||
assert!(!std::ptr::eq(ev0, ev1));
|
||||
}
|
||||
|
||||
/// `find_ec_volume` returns only disk 0's runtime, which is what hides
|
||||
/// sibling-disk shards from every scrub mode. The plural lookup must
|
||||
/// return one runtime per disk holding the vid, in location order.
|
||||
#[test]
|
||||
fn test_find_all_ec_volumes_returns_every_disk() {
|
||||
let (store, _tmp) = build_split_disk_store(7010);
|
||||
let vid = VolumeId(7010);
|
||||
|
||||
let all = store.find_all_ec_volumes(vid);
|
||||
assert_eq!(
|
||||
all.len(),
|
||||
2,
|
||||
"expected one EcVolume per disk holding the vid"
|
||||
);
|
||||
|
||||
// Disk 0 carries shards 0 and 12; disk 1 carries shard 1.
|
||||
assert!(all[0].has_shard(0));
|
||||
assert!(all[0].has_shard(12));
|
||||
assert!(all[1].has_shard(1));
|
||||
|
||||
// The singular lookup sees only the first — the bug being fixed.
|
||||
let first = store.find_ec_volume(vid).unwrap();
|
||||
assert!(std::ptr::eq(first, all[0]));
|
||||
|
||||
// A vid nobody mounts yields an empty vec, not a panic.
|
||||
assert!(store.find_all_ec_volumes(VolumeId(9999)).is_empty());
|
||||
}
|
||||
|
||||
/// End-to-end: with the vid mounted on two disks, a scrub driven through
|
||||
/// the Store must reach BOTH disks' shards. Before the aggregation fix
|
||||
/// `find_ec_volume` returned disk 0 and disk 1's shard 1 was never read.
|
||||
#[test]
|
||||
fn test_scrub_plans_reach_every_disk_through_the_store() {
|
||||
use crate::storage::erasure_coding::ec_volume::{
|
||||
EcChecksumScrubPlan, EcLocalScrubPlan, merge_ec_runtimes,
|
||||
};
|
||||
|
||||
let (store, _tmp) = build_split_disk_store(7030);
|
||||
let vid = VolumeId(7030);
|
||||
|
||||
let runtimes = store.find_all_ec_volumes(vid);
|
||||
assert_eq!(runtimes.len(), 2);
|
||||
|
||||
// Reachability is the invariant, so assert on the resolved slots rather
|
||||
// than on scrub message text: shards 0 and 12 live on disk 0, shard 1 on
|
||||
// disk 1. The old first-match lookup could never see shard 1.
|
||||
let merged = merge_ec_runtimes(&runtimes).expect("two runtimes merge");
|
||||
assert!(merged.slots[0].is_some(), "disk 0's shard 0 unreachable");
|
||||
assert!(merged.slots[12].is_some(), "disk 0's shard 12 unreachable");
|
||||
assert!(
|
||||
merged.slots[1].is_some(),
|
||||
"disk 1's shard 1 unreachable — the bug"
|
||||
);
|
||||
assert!(
|
||||
merged.skipped.is_empty(),
|
||||
"same generation: {:?}",
|
||||
merged.skipped
|
||||
);
|
||||
|
||||
// Shard 1 is owned by the sibling runtime, not the anchor.
|
||||
let (owner, _) = merged.slots[1].unwrap();
|
||||
assert!(std::ptr::eq(owner, runtimes[1]));
|
||||
|
||||
// Both plans build over the union rather than over disk 0 alone.
|
||||
assert!(EcChecksumScrubPlan::for_volumes(&runtimes).is_some());
|
||||
assert!(EcLocalScrubPlan::for_volumes(&runtimes).is_some());
|
||||
// ...and `is_some()` is a real question: `for_volumes` has exactly one
|
||||
// `None` (the vanished-volume case), so without this the two lines above
|
||||
// would hold for any input at all.
|
||||
assert!(EcChecksumScrubPlan::for_volumes(&[]).is_none());
|
||||
assert!(EcLocalScrubPlan::for_volumes(&[]).is_none());
|
||||
|
||||
// Regression guard: a single-runtime view still sees only its own disk,
|
||||
// which is exactly what made aggregation necessary.
|
||||
let disk0 = merge_ec_runtimes(&[runtimes[0]]).unwrap();
|
||||
assert!(
|
||||
disk0.slots.get(1).copied().flatten().is_none(),
|
||||
"disk 0's runtime must not see the sibling's shard"
|
||||
);
|
||||
}
|
||||
|
||||
/// `Store::unmount_ec_shards` used to return after the first
|
||||
/// location with the vid, so a request to unmount a shard that
|
||||
/// lives on a sibling disk became a silent no-op. After the fix,
|
||||
@@ -1267,17 +1423,22 @@ mod tests {
|
||||
let (_ev, dirs) = store.collect_ec_shard_dirs(vid, max_shards).unwrap();
|
||||
|
||||
// Shards 0 and 12 → disk 0's directory.
|
||||
assert_eq!(dirs[0].as_deref(), Some(store.locations[0].directory.as_str()));
|
||||
assert_eq!(dirs[12].as_deref(), Some(store.locations[0].directory.as_str()));
|
||||
assert_eq!(
|
||||
dirs[0].as_deref(),
|
||||
Some(store.locations[0].directory.as_str())
|
||||
);
|
||||
assert_eq!(
|
||||
dirs[12].as_deref(),
|
||||
Some(store.locations[0].directory.as_str())
|
||||
);
|
||||
// Shard 1 → disk 1's directory.
|
||||
assert_eq!(dirs[1].as_deref(), Some(store.locations[1].directory.as_str()));
|
||||
assert_eq!(
|
||||
dirs[1].as_deref(),
|
||||
Some(store.locations[1].directory.as_str())
|
||||
);
|
||||
// Unmounted shards → None.
|
||||
for sid in [2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 13] {
|
||||
assert_eq!(
|
||||
dirs[sid], None,
|
||||
"shard {} unexpectedly reported a dir",
|
||||
sid,
|
||||
);
|
||||
assert_eq!(dirs[sid], None, "shard {} unexpectedly reported a dir", sid,);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1586,11 +1747,7 @@ mod tests {
|
||||
vec![0u8; 20],
|
||||
)
|
||||
.unwrap();
|
||||
std::fs::write(
|
||||
ec_dir.join(format!("{}_{}.ecj", collection, vid)),
|
||||
b"",
|
||||
)
|
||||
.unwrap();
|
||||
std::fs::write(ec_dir.join(format!("{}_{}.ecj", collection, vid)), b"").unwrap();
|
||||
|
||||
let mut store = Store::new(NeedleMapKind::InMemory);
|
||||
store
|
||||
@@ -1689,50 +1846,3 @@ mod tests {
|
||||
assert!(std::path::Path::new(&format!("{}.ecx", ec_base)).exists());
|
||||
}
|
||||
}
|
||||
|
||||
/// Walk a disk's data directory and return the `.ec??` shard files
|
||||
/// that are present on disk but not yet registered in the location's
|
||||
/// `ec_volumes` map. Keyed by (collection, vid) so callers can match
|
||||
/// each group against its `.ecx`-owning disk in one lookup. Zero-byte
|
||||
/// shard files are ignored — same shape as `load_all_ec_shards`.
|
||||
fn collect_orphan_ec_shards(
|
||||
loc: &crate::storage::disk_location::DiskLocation,
|
||||
_loc_idx: usize,
|
||||
) -> HashMap<EcKey, Vec<(String, u32)>> {
|
||||
let mut orphans: HashMap<EcKey, Vec<(String, u32)>> = HashMap::new();
|
||||
let Ok(read) = fs::read_dir(&loc.directory) else {
|
||||
return orphans;
|
||||
};
|
||||
for ent in read.flatten() {
|
||||
if ent.file_type().map(|ft| ft.is_dir()).unwrap_or(false) {
|
||||
continue;
|
||||
}
|
||||
let name = ent.file_name().to_string_lossy().into_owned();
|
||||
let Some(dot) = name.rfind('.') else {
|
||||
continue;
|
||||
};
|
||||
let (base, ext) = name.split_at(dot);
|
||||
let Some(shard_id) = is_ec_shard_extension(ext) else {
|
||||
continue;
|
||||
};
|
||||
// Ignore zero-byte shards. Use the DirEntry's metadata so we
|
||||
// don't pay a second stat syscall per file beyond what
|
||||
// read_dir already returned.
|
||||
match ent.metadata() {
|
||||
Ok(meta) if meta.len() > 0 => {}
|
||||
_ => continue,
|
||||
}
|
||||
let Some((collection, vid)) = parse_collection_volume_id_pub(base) else {
|
||||
continue;
|
||||
};
|
||||
// Skip shards that are already registered to an EcVolume.
|
||||
if let Some(ecv) = loc.find_ec_volume(vid) {
|
||||
if ecv.has_shard(shard_id as u8) {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
let key = EcKey { collection, vid };
|
||||
orphans.entry(key).or_default().push((name, shard_id));
|
||||
}
|
||||
orphans
|
||||
}
|
||||
|
||||
@@ -155,7 +155,7 @@ impl Size {
|
||||
return 0;
|
||||
}
|
||||
if self.0 < 0 {
|
||||
return (self.0 * -1) as u32;
|
||||
return -self.0 as u32;
|
||||
}
|
||||
self.0 as u32
|
||||
}
|
||||
@@ -284,8 +284,9 @@ impl fmt::Display for Offset {
|
||||
// DiskType
|
||||
// ============================================================================
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq, Hash)]
|
||||
#[derive(Debug, Clone, PartialEq, Eq, Hash, Default)]
|
||||
pub enum DiskType {
|
||||
#[default]
|
||||
HardDrive,
|
||||
Ssd,
|
||||
Custom(String),
|
||||
@@ -319,12 +320,6 @@ impl fmt::Display for DiskType {
|
||||
}
|
||||
}
|
||||
|
||||
impl Default for DiskType {
|
||||
fn default() -> Self {
|
||||
DiskType::HardDrive
|
||||
}
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// VolumeId
|
||||
// ============================================================================
|
||||
@@ -397,7 +392,7 @@ impl From<u8> for Version {
|
||||
///
|
||||
/// Fields are split into request-side options (set by the caller) and response-side
|
||||
/// flags (set during the read to communicate status back).
|
||||
#[derive(Debug, Clone)]
|
||||
#[derive(Debug, Clone, Default)]
|
||||
pub struct ReadOption {
|
||||
// -- request --
|
||||
/// If true, allow reading needles that have been soft-deleted.
|
||||
@@ -423,21 +418,6 @@ pub struct ReadOption {
|
||||
pub read_buffer_size: i32,
|
||||
}
|
||||
|
||||
impl Default for ReadOption {
|
||||
fn default() -> Self {
|
||||
ReadOption {
|
||||
read_deleted: false,
|
||||
attempt_meta_only: false,
|
||||
must_meta_only: false,
|
||||
is_meta_only: false,
|
||||
volume_revision: 0,
|
||||
is_out_of_range: false,
|
||||
has_slow_read: false,
|
||||
read_buffer_size: 0,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// NeedleMapEntry helpers (for .idx file)
|
||||
// ============================================================================
|
||||
|
||||
+1318
-640
File diff suppressed because it is too large
Load Diff
@@ -10,7 +10,7 @@ use crate::storage::needle::Needle;
|
||||
use crate::storage::super_block::SuperBlock;
|
||||
use crate::storage::types::*;
|
||||
use crate::storage::volume::{
|
||||
fsync_dir, needle_disk_end, scan_volume_file, Volume, VolumeError, VolumeFileVisitor,
|
||||
Volume, VolumeError, VolumeFileVisitor, fsync_dir, needle_disk_end, scan_volume_file,
|
||||
};
|
||||
|
||||
/// Writes one .idx row per .dat record, in .dat append order, which is the
|
||||
@@ -105,11 +105,11 @@ impl Volume {
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use crate::storage::needle::crc::CRC;
|
||||
use crate::storage::needle::Needle;
|
||||
use crate::storage::needle::crc::CRC;
|
||||
use crate::storage::needle_map::NeedleMapKind;
|
||||
use crate::storage::types::*;
|
||||
use crate::storage::volume::Volume;
|
||||
use crate::storage::volume::{Volume, VolumeSpec};
|
||||
use std::fs;
|
||||
use std::path::Path;
|
||||
use tempfile::TempDir;
|
||||
@@ -145,13 +145,9 @@ mod tests {
|
||||
let mut v = Volume::new(
|
||||
data,
|
||||
old_idx,
|
||||
"",
|
||||
VolumeId(1),
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
&VolumeSpec::default(),
|
||||
)
|
||||
.unwrap();
|
||||
for id in 1..=3 {
|
||||
@@ -167,13 +163,9 @@ mod tests {
|
||||
let reopened = Volume::new(
|
||||
data,
|
||||
new_idx,
|
||||
"",
|
||||
VolumeId(1),
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
&VolumeSpec::default(),
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
@@ -207,13 +199,9 @@ mod tests {
|
||||
let mut v = Volume::new(
|
||||
dir,
|
||||
dir,
|
||||
"",
|
||||
VolumeId(1),
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
&VolumeSpec::default(),
|
||||
)
|
||||
.unwrap();
|
||||
v.write_needle(&mut needle(1), true, false).unwrap();
|
||||
@@ -233,13 +221,9 @@ mod tests {
|
||||
let reopened = Volume::new(
|
||||
dir,
|
||||
dir,
|
||||
"",
|
||||
VolumeId(1),
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
&VolumeSpec::default(),
|
||||
)
|
||||
.unwrap();
|
||||
drop(reopened);
|
||||
@@ -261,13 +245,9 @@ mod tests {
|
||||
let mut v = Volume::new(
|
||||
dir,
|
||||
dir,
|
||||
"",
|
||||
VolumeId(1),
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
&VolumeSpec::default(),
|
||||
)
|
||||
.unwrap();
|
||||
v.write_needle(&mut needle(1), true, false).unwrap();
|
||||
@@ -290,13 +270,9 @@ mod tests {
|
||||
let reopened = Volume::new(
|
||||
dir,
|
||||
dir,
|
||||
"",
|
||||
VolumeId(1),
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
&VolumeSpec::default(),
|
||||
)
|
||||
.unwrap();
|
||||
drop(reopened);
|
||||
@@ -318,13 +294,9 @@ mod tests {
|
||||
let mut v = Volume::new(
|
||||
dir,
|
||||
dir,
|
||||
"",
|
||||
VolumeId(1),
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
&VolumeSpec::default(),
|
||||
)
|
||||
.unwrap();
|
||||
v.write_needle(&mut needle(1), true, false).unwrap();
|
||||
@@ -356,13 +328,9 @@ mod tests {
|
||||
let reopened = Volume::new(
|
||||
dir,
|
||||
dir,
|
||||
"",
|
||||
VolumeId(1),
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
&VolumeSpec::default(),
|
||||
)
|
||||
.unwrap();
|
||||
drop(reopened);
|
||||
|
||||
@@ -9,10 +9,10 @@ use std::path::Path;
|
||||
use tracing::info;
|
||||
|
||||
use crate::storage::idx;
|
||||
use crate::storage::needle::needle::needle_body_length;
|
||||
use crate::storage::needle::Needle;
|
||||
use crate::storage::needle::needle::needle_body_length;
|
||||
use crate::storage::types::*;
|
||||
use crate::storage::volume::{fsync_dir, Volume, VolumeError};
|
||||
use crate::storage::volume::{Volume, VolumeError, fsync_dir};
|
||||
|
||||
/// Needles found in the head of .dat, keyed by id, plus the ids in .dat order.
|
||||
type DatHeadNeedles = (HashMap<NeedleId, (Offset, Size)>, Vec<NeedleId>);
|
||||
@@ -215,20 +215,19 @@ mod tests {
|
||||
use super::*;
|
||||
use crate::storage::needle::crc::CRC;
|
||||
use crate::storage::needle_map::NeedleMapKind;
|
||||
use std::os::unix::fs::{FileExt, PermissionsExt};
|
||||
use crate::storage::volume::VolumeSpec;
|
||||
use std::io::{Seek, SeekFrom};
|
||||
#[cfg(unix)]
|
||||
use std::os::unix::fs::PermissionsExt;
|
||||
use tempfile::TempDir;
|
||||
|
||||
fn open_volume(dir: &str) -> Volume {
|
||||
Volume::new(
|
||||
dir,
|
||||
dir,
|
||||
"",
|
||||
VolumeId(1),
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
&VolumeSpec::default(),
|
||||
)
|
||||
.unwrap()
|
||||
}
|
||||
@@ -256,7 +255,7 @@ mod tests {
|
||||
/// writes (key, offset 0, tombstone) rows over the front of .idx instead of
|
||||
/// appending them.
|
||||
fn clobber_idx_head(idx_path: &str, keys: &[u64]) {
|
||||
let file = OpenOptions::new().write(true).open(idx_path).unwrap();
|
||||
let mut file = OpenOptions::new().write(true).open(idx_path).unwrap();
|
||||
for (i, key) in keys.iter().enumerate() {
|
||||
let mut row = Vec::new();
|
||||
idx::write_index_entry(
|
||||
@@ -266,8 +265,13 @@ mod tests {
|
||||
TOMBSTONE_FILE_SIZE,
|
||||
)
|
||||
.unwrap();
|
||||
file.write_at(&row, (i * NEEDLE_MAP_ENTRY_SIZE) as u64)
|
||||
// Positional write without Unix-only `FileExt::write_at`, so this
|
||||
// helper (and the tests using it) also builds on Windows.
|
||||
// Single-threaded test helper: no concurrent reader can move the
|
||||
// offset between seek and write.
|
||||
file.seek(SeekFrom::Start((i * NEEDLE_MAP_ENTRY_SIZE) as u64))
|
||||
.unwrap();
|
||||
file.write_all(&row).unwrap();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -305,6 +309,7 @@ mod tests {
|
||||
let size_before = idx_size(&idx_path);
|
||||
|
||||
// The rewrite replaces .idx wholesale, so it must not widen the mode.
|
||||
#[cfg(unix)]
|
||||
fs::set_permissions(&idx_path, fs::Permissions::from_mode(0o600)).unwrap();
|
||||
|
||||
// Deletes against needles 9..12 land on the front of .idx and take the
|
||||
@@ -324,11 +329,14 @@ mod tests {
|
||||
let want = size_before + 4 * NEEDLE_MAP_ENTRY_SIZE as u64;
|
||||
assert_eq!(idx_size(&idx_path), want, "idx size after recovery");
|
||||
|
||||
assert_eq!(
|
||||
fs::metadata(&idx_path).unwrap().permissions().mode() & 0o777,
|
||||
0o600,
|
||||
"idx mode after recovery"
|
||||
);
|
||||
#[cfg(unix)]
|
||||
{
|
||||
assert_eq!(
|
||||
fs::metadata(&idx_path).unwrap().permissions().mode() & 0o777,
|
||||
0o600,
|
||||
"idx mode after recovery"
|
||||
);
|
||||
}
|
||||
|
||||
// The recovered rows go back in front, so .idx is in .dat append order
|
||||
// again: the fingerprint is gone and the last row is still the .dat tail.
|
||||
|
||||
@@ -73,8 +73,10 @@ mod tests {
|
||||
let empty = master_pb::VolumeInformationMessage::default();
|
||||
assert_eq!(report_hash(&empty), 10988706248825469653);
|
||||
|
||||
let mut one = master_pb::VolumeInformationMessage::default();
|
||||
one.id = 1;
|
||||
let one = master_pb::VolumeInformationMessage {
|
||||
id: 1,
|
||||
..Default::default()
|
||||
};
|
||||
assert_eq!(report_hash(&one), 2035849960016744285);
|
||||
|
||||
let full = master_pb::VolumeInformationMessage {
|
||||
|
||||
@@ -66,7 +66,7 @@ fn parse_go_version_number() -> Option<String> {
|
||||
}
|
||||
}
|
||||
match (major, minor) {
|
||||
(Some(maj), Some(min)) => Some(format!("{}.{}", maj, format!("{:02}", min))),
|
||||
(Some(maj), Some(min)) => Some(format!("{}.{:02}", maj, min)),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -20,26 +20,71 @@ use std::collections::{HashMap, HashSet};
|
||||
fn ungated_handlers() -> HashMap<&'static str, &'static str> {
|
||||
[
|
||||
// Cluster-internal: issued volume server -> volume server.
|
||||
("copy_file", "replica sync and EC task pull whole files from a peer"),
|
||||
("read_needle_blob", "replica sync, vacuum and EC rebuild read needles from a peer"),
|
||||
("read_needle_meta", "replica sync compares needle metadata across peers"),
|
||||
(
|
||||
"copy_file",
|
||||
"replica sync and EC task pull whole files from a peer",
|
||||
),
|
||||
(
|
||||
"read_needle_blob",
|
||||
"replica sync, vacuum and EC rebuild read needles from a peer",
|
||||
),
|
||||
(
|
||||
"read_needle_meta",
|
||||
"replica sync compares needle metadata across peers",
|
||||
),
|
||||
("write_needle_blob", "replica sync repairs a peer's needle"),
|
||||
("receive_file", "EC shard distribution pushes shards to a peer"),
|
||||
("read_volume_file_status", "the copy path queries the source volume server"),
|
||||
("volume_ec_shard_read", "a volume server reads EC shards held by a peer"),
|
||||
("volume_ec_blob_delete", "EC delete is fanned out to the shard holders"),
|
||||
("volume_ec_shards_info", "EC verification polls shard holders"),
|
||||
("volume_ec_shards_mount", "EC shard distribution mounts on the receiving peer"),
|
||||
("volume_incremental_copy", "volume backup pulls increments from a peer"),
|
||||
("volume_sync_status", "sync compares volume state across peers"),
|
||||
("volume_tail_sender", "the tail source streams to the receiving peer"),
|
||||
("volume_status", "replica sync and the master's vacuum loop poll volume status"),
|
||||
(
|
||||
"receive_file",
|
||||
"EC shard distribution pushes shards to a peer",
|
||||
),
|
||||
(
|
||||
"read_volume_file_status",
|
||||
"the copy path queries the source volume server",
|
||||
),
|
||||
(
|
||||
"volume_ec_shard_read",
|
||||
"a volume server reads EC shards held by a peer",
|
||||
),
|
||||
(
|
||||
"volume_ec_blob_delete",
|
||||
"EC delete is fanned out to the shard holders",
|
||||
),
|
||||
(
|
||||
"volume_ec_shards_info",
|
||||
"EC verification polls shard holders",
|
||||
),
|
||||
(
|
||||
"volume_ec_shards_mount",
|
||||
"EC shard distribution mounts on the receiving peer",
|
||||
),
|
||||
(
|
||||
"volume_incremental_copy",
|
||||
"volume backup pulls increments from a peer",
|
||||
),
|
||||
(
|
||||
"volume_sync_status",
|
||||
"sync compares volume state across peers",
|
||||
),
|
||||
(
|
||||
"volume_tail_sender",
|
||||
"the tail source streams to the receiving peer",
|
||||
),
|
||||
(
|
||||
"volume_status",
|
||||
"replica sync and the master's vacuum loop poll volume status",
|
||||
),
|
||||
// Read-only or liveness: no state change.
|
||||
("ping", "liveness probe"),
|
||||
("get_state", "read-only volume server state"),
|
||||
("query", "read-only data query"),
|
||||
("vacuum_volume_check", "read-only garbage ratio; the vacuum steps that act on it are gated"),
|
||||
("volume_server_status", "read-only status, the gRPC counterpart of the /status page"),
|
||||
(
|
||||
"vacuum_volume_check",
|
||||
"read-only garbage ratio; the vacuum steps that act on it are gated",
|
||||
),
|
||||
(
|
||||
"volume_server_status",
|
||||
"read-only status, the gRPC counterpart of the /status page",
|
||||
),
|
||||
]
|
||||
.into_iter()
|
||||
.collect()
|
||||
@@ -129,5 +174,9 @@ fn volume_server_admin_auth_coverage() {
|
||||
}
|
||||
}
|
||||
|
||||
assert!(problems.is_empty(), "admin-auth coverage gaps:\n{}", problems.join("\n"));
|
||||
assert!(
|
||||
problems.is_empty(),
|
||||
"admin-auth coverage gaps:\n{}",
|
||||
problems.join("\n")
|
||||
);
|
||||
}
|
||||
|
||||
@@ -12,12 +12,13 @@ use tower::ServiceExt; // for `oneshot`
|
||||
|
||||
use seaweed_volume::security::{Guard, SigningKey};
|
||||
use seaweed_volume::server::volume_server::{
|
||||
build_admin_router, build_admin_router_with_ui, build_metrics_router, build_public_router,
|
||||
VolumeServerState,
|
||||
VolumeServerState, build_admin_router, build_admin_router_with_ui, build_metrics_router,
|
||||
build_public_router,
|
||||
};
|
||||
use seaweed_volume::storage::needle_map::NeedleMapKind;
|
||||
use seaweed_volume::storage::store::Store;
|
||||
use seaweed_volume::storage::types::{DiskType, Version, VolumeId};
|
||||
use seaweed_volume::storage::types::{DiskType, VolumeId};
|
||||
use seaweed_volume::storage::volume::VolumeSpec;
|
||||
|
||||
use tempfile::TempDir;
|
||||
|
||||
@@ -73,12 +74,11 @@ fn build_test_state(
|
||||
store
|
||||
.add_volume(
|
||||
VolumeId(1),
|
||||
"",
|
||||
replica_placement,
|
||||
None,
|
||||
0,
|
||||
DiskType::HardDrive,
|
||||
Version::current(),
|
||||
&VolumeSpec {
|
||||
replica_placement,
|
||||
..Default::default()
|
||||
},
|
||||
)
|
||||
.expect("failed to create volume");
|
||||
|
||||
@@ -957,9 +957,10 @@ async fn replicate_write_does_not_re_replicate() {
|
||||
#[tokio::test]
|
||||
async fn chunk_manifest_expands_chunk_stored_on_ec_volume() {
|
||||
use seaweed_volume::storage::erasure_coding::ec_encoder::write_ec_files;
|
||||
use seaweed_volume::storage::erasure_coding::ec_shard::ShardId;
|
||||
use seaweed_volume::storage::needle::needle::{FileId, Needle};
|
||||
use seaweed_volume::storage::types::{Cookie, NeedleId};
|
||||
use seaweed_volume::storage::volume::Volume;
|
||||
use seaweed_volume::storage::volume::{Volume, VolumeSpec};
|
||||
|
||||
let (state, tmp) = test_state();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
@@ -976,13 +977,9 @@ async fn chunk_manifest_expands_chunk_stored_on_ec_volume() {
|
||||
let mut v = Volume::new(
|
||||
dir,
|
||||
dir,
|
||||
"",
|
||||
VolumeId(2),
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
&VolumeSpec::default(),
|
||||
)
|
||||
.unwrap();
|
||||
let mut n = Needle {
|
||||
@@ -1002,7 +999,7 @@ async fn chunk_manifest_expands_chunk_stored_on_ec_volume() {
|
||||
// after ec.encode retired the regular volume.
|
||||
{
|
||||
let mut store = state.store.write().unwrap();
|
||||
let shard_ids: Vec<u32> = (0..14).collect();
|
||||
let shard_ids: Vec<ShardId> = (0..14).collect();
|
||||
store.mount_ec_shards(VolumeId(2), "", &shard_ids).unwrap();
|
||||
}
|
||||
|
||||
|
||||
Generated
+102
-209
@@ -394,28 +394,6 @@ dependencies = [
|
||||
"syn 2.0.119",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "async-stream"
|
||||
version = "0.3.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0b5a71a6f37880a80d1d7f19efd781e4b5de42c88f0722cc13bcb6cc2cfe8476"
|
||||
dependencies = [
|
||||
"async-stream-impl",
|
||||
"futures-core",
|
||||
"pin-project-lite",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "async-stream-impl"
|
||||
version = "0.3.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "c7c24de15d275a1ecfd47a380fb4d5ec9bfe0933f309ed5e705b775596a3574d"
|
||||
dependencies = [
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"syn 2.0.119",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "async-trait"
|
||||
version = "0.1.92"
|
||||
@@ -701,7 +679,7 @@ dependencies = [
|
||||
"rustls-pki-types",
|
||||
"tokio",
|
||||
"tokio-rustls",
|
||||
"tower 0.5.3",
|
||||
"tower",
|
||||
"tracing",
|
||||
]
|
||||
|
||||
@@ -856,13 +834,13 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "axum"
|
||||
version = "0.7.9"
|
||||
version = "0.8.9"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "edca88bc138befd0323b20752846e6587272d3b03b0343c8ea28a6f819e6e71f"
|
||||
checksum = "31b698c5f9a010f6573133b09e0de5408834d0c82f8d7475a89fc1867a71cd90"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"axum-core",
|
||||
"bytes",
|
||||
"form_urlencoded",
|
||||
"futures-util",
|
||||
"http 1.5.0",
|
||||
"http-body 1.1.0",
|
||||
@@ -875,14 +853,13 @@ dependencies = [
|
||||
"mime",
|
||||
"percent-encoding",
|
||||
"pin-project-lite",
|
||||
"rustversion",
|
||||
"serde",
|
||||
"serde_core",
|
||||
"serde_json",
|
||||
"serde_path_to_error",
|
||||
"serde_urlencoded",
|
||||
"sync_wrapper",
|
||||
"tokio",
|
||||
"tower 0.5.3",
|
||||
"tower",
|
||||
"tower-layer",
|
||||
"tower-service",
|
||||
"tracing",
|
||||
@@ -890,19 +867,17 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "axum-core"
|
||||
version = "0.4.5"
|
||||
version = "0.5.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "09f2bd6146b97ae3359fa0cc6d6b376d9539582c7b4220f041a33ec24c226199"
|
||||
checksum = "08c78f31d7b1291f7ee735c1c6780ccde7785daae9a9206026862dab7d8792d1"
|
||||
dependencies = [
|
||||
"async-trait",
|
||||
"bytes",
|
||||
"futures-util",
|
||||
"futures-core",
|
||||
"http 1.5.0",
|
||||
"http-body 1.1.0",
|
||||
"http-body-util",
|
||||
"mime",
|
||||
"pin-project-lite",
|
||||
"rustversion",
|
||||
"sync_wrapper",
|
||||
"tower-layer",
|
||||
"tower-service",
|
||||
@@ -1965,7 +1940,7 @@ dependencies = [
|
||||
"indexmap 2.14.0",
|
||||
"itertools",
|
||||
"parking_lot",
|
||||
"petgraph 0.8.3",
|
||||
"petgraph",
|
||||
"tokio",
|
||||
]
|
||||
|
||||
@@ -2793,7 +2768,7 @@ dependencies = [
|
||||
"libc",
|
||||
"percent-encoding",
|
||||
"pin-project-lite",
|
||||
"socket2 0.6.5",
|
||||
"socket2",
|
||||
"tokio",
|
||||
"tower-service",
|
||||
"tracing",
|
||||
@@ -3249,9 +3224,9 @@ dependencies = [
|
||||
"object_store",
|
||||
"permutation",
|
||||
"pin-project",
|
||||
"prost 0.14.4",
|
||||
"prost-build 0.14.4",
|
||||
"prost-types 0.14.4",
|
||||
"prost",
|
||||
"prost-build",
|
||||
"prost-types",
|
||||
"rand 0.9.5",
|
||||
"rayon",
|
||||
"roaring",
|
||||
@@ -3357,7 +3332,7 @@ dependencies = [
|
||||
"num_cpus",
|
||||
"object_store",
|
||||
"pin-project",
|
||||
"prost 0.14.4",
|
||||
"prost",
|
||||
"quick_cache",
|
||||
"rand 0.9.5",
|
||||
"roaring",
|
||||
@@ -3398,8 +3373,8 @@ dependencies = [
|
||||
"lance-datagen",
|
||||
"log",
|
||||
"pin-project",
|
||||
"prost 0.14.4",
|
||||
"prost-build 0.14.4",
|
||||
"prost",
|
||||
"prost-build",
|
||||
"tokio",
|
||||
"tracing",
|
||||
]
|
||||
@@ -3461,8 +3436,8 @@ dependencies = [
|
||||
"log",
|
||||
"lz4",
|
||||
"num-traits",
|
||||
"prost 0.14.4",
|
||||
"prost-build 0.14.4",
|
||||
"prost",
|
||||
"prost-build",
|
||||
"rand 0.9.5",
|
||||
"strum",
|
||||
"tokio",
|
||||
@@ -3497,9 +3472,9 @@ dependencies = [
|
||||
"log",
|
||||
"num-traits",
|
||||
"object_store",
|
||||
"prost 0.14.4",
|
||||
"prost-build 0.14.4",
|
||||
"prost-types 0.14.4",
|
||||
"prost",
|
||||
"prost-build",
|
||||
"prost-types",
|
||||
"tokio",
|
||||
"tracing",
|
||||
]
|
||||
@@ -3554,9 +3529,9 @@ dependencies = [
|
||||
"ndarray",
|
||||
"num-traits",
|
||||
"object_store",
|
||||
"prost 0.14.4",
|
||||
"prost-build 0.14.4",
|
||||
"prost-types 0.14.4",
|
||||
"prost",
|
||||
"prost-build",
|
||||
"prost-types",
|
||||
"rand 0.9.5",
|
||||
"rand_distr",
|
||||
"rangemap",
|
||||
@@ -3590,7 +3565,7 @@ dependencies = [
|
||||
"lance-core",
|
||||
"lance-io",
|
||||
"lance-select",
|
||||
"prost-types 0.14.4",
|
||||
"prost-types",
|
||||
"roaring",
|
||||
"serde",
|
||||
"serde_json",
|
||||
@@ -3624,7 +3599,7 @@ dependencies = [
|
||||
"opendal",
|
||||
"path_abs",
|
||||
"pin-project",
|
||||
"prost 0.14.4",
|
||||
"prost",
|
||||
"rand 0.9.5",
|
||||
"serde",
|
||||
"tempfile",
|
||||
@@ -3719,9 +3694,9 @@ dependencies = [
|
||||
"lance-select",
|
||||
"log",
|
||||
"object_store",
|
||||
"prost 0.14.4",
|
||||
"prost-build 0.14.4",
|
||||
"prost-types 0.14.4",
|
||||
"prost",
|
||||
"prost-build",
|
||||
"prost-types",
|
||||
"rand 0.9.5",
|
||||
"rangemap",
|
||||
"roaring",
|
||||
@@ -3938,9 +3913,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "matchit"
|
||||
version = "0.7.3"
|
||||
version = "0.8.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0e7465ac9959cc2b1404e8e2367b43684a6d13790fe23056cc8c6c5a6b7bcb94"
|
||||
checksum = "47e1ffaa40ddd1f3ed91f717a33c8c0ee23fff369e3aa8772b9605cc1d22f4c3"
|
||||
|
||||
[[package]]
|
||||
name = "matrixmultiply"
|
||||
@@ -4411,16 +4386,6 @@ version = "0.4.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "df202b0b0f5b8e389955afd5f27b007b00fb948162953f1db9c70d2c7e3157d7"
|
||||
|
||||
[[package]]
|
||||
name = "petgraph"
|
||||
version = "0.7.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "3672b37090dbd86368a4145bc067582552b29c27377cad4e0a306c97f9bd7772"
|
||||
dependencies = [
|
||||
"fixedbitset",
|
||||
"indexmap 2.14.0",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "petgraph"
|
||||
version = "0.8.3"
|
||||
@@ -4563,16 +4528,6 @@ dependencies = [
|
||||
"thiserror 1.0.69",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "prost"
|
||||
version = "0.13.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "2796faa41db3ec313a31f7624d9286acf277b52de526150b7e69f3debf891ee5"
|
||||
dependencies = [
|
||||
"bytes",
|
||||
"prost-derive 0.13.5",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "prost"
|
||||
version = "0.14.4"
|
||||
@@ -4580,27 +4535,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "528ac67416ff8646872a3c02cad9cc4ee5dc9f9540c9b10771855c95cb2e5ae1"
|
||||
dependencies = [
|
||||
"bytes",
|
||||
"prost-derive 0.14.4",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "prost-build"
|
||||
version = "0.13.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "be769465445e8c1474e9c5dac2018218498557af32d9ed057325ec9a41ae81bf"
|
||||
dependencies = [
|
||||
"heck",
|
||||
"itertools",
|
||||
"log",
|
||||
"multimap",
|
||||
"once_cell",
|
||||
"petgraph 0.7.1",
|
||||
"prettyplease",
|
||||
"prost 0.13.5",
|
||||
"prost-types 0.13.5",
|
||||
"regex",
|
||||
"syn 2.0.119",
|
||||
"tempfile",
|
||||
"prost-derive",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -4613,28 +4548,17 @@ dependencies = [
|
||||
"itertools",
|
||||
"log",
|
||||
"multimap",
|
||||
"petgraph 0.8.3",
|
||||
"petgraph",
|
||||
"prettyplease",
|
||||
"prost 0.14.4",
|
||||
"prost-types 0.14.4",
|
||||
"prost",
|
||||
"prost-types",
|
||||
"pulldown-cmark",
|
||||
"pulldown-cmark-to-cmark",
|
||||
"regex",
|
||||
"syn 2.0.119",
|
||||
"tempfile",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "prost-derive"
|
||||
version = "0.13.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "8a56d757972c98b346a9b766e3f02746cde6dd1cd1d1d563472929fdd74bec4d"
|
||||
dependencies = [
|
||||
"anyhow",
|
||||
"itertools",
|
||||
"proc-macro2",
|
||||
"quote",
|
||||
"syn 2.0.119",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "prost-derive"
|
||||
version = "0.14.4"
|
||||
@@ -4648,22 +4572,13 @@ dependencies = [
|
||||
"syn 2.0.119",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "prost-types"
|
||||
version = "0.13.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "52c2c1bf36ddb1a1c396b3601a3cec27c2462e45f07c386894ec3ccf5332bd16"
|
||||
dependencies = [
|
||||
"prost 0.13.5",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "prost-types"
|
||||
version = "0.14.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "f94967dc7688f3054c7fac87473ffae4cc4c3904800e2d9f5b857246d8963b0a"
|
||||
dependencies = [
|
||||
"prost 0.14.4",
|
||||
"prost",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -4730,6 +4645,26 @@ version = "3.2.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "95067976aca6421a523e491fce939a3e65249bac4b977adee0ee9771568e8aa3"
|
||||
|
||||
[[package]]
|
||||
name = "pulldown-cmark"
|
||||
version = "0.13.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e9f068eba8e7071c5f9511831b44f32c740d5adf574e990f946ddb53db2f314e"
|
||||
dependencies = [
|
||||
"bitflags 2.13.1",
|
||||
"memchr",
|
||||
"unicase",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "pulldown-cmark-to-cmark"
|
||||
version = "22.0.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ab1ad36992cead65f02aa399a373a42730922f1525d988172634fdefdecb8a60"
|
||||
dependencies = [
|
||||
"pulldown-cmark",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "quick-xml"
|
||||
version = "0.39.4"
|
||||
@@ -4775,7 +4710,7 @@ dependencies = [
|
||||
"quinn-udp",
|
||||
"rustc-hash",
|
||||
"rustls",
|
||||
"socket2 0.6.5",
|
||||
"socket2",
|
||||
"thiserror 2.0.20",
|
||||
"tokio",
|
||||
"tracing",
|
||||
@@ -4814,7 +4749,7 @@ dependencies = [
|
||||
"cfg_aliases",
|
||||
"libc",
|
||||
"once_cell",
|
||||
"socket2 0.6.5",
|
||||
"socket2",
|
||||
"tracing",
|
||||
"windows-sys 0.61.2",
|
||||
]
|
||||
@@ -4846,24 +4781,13 @@ version = "0.7.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "dc33ff2d4973d518d823d61aa239014831e521c75da58e3df4840d3f47749d09"
|
||||
|
||||
[[package]]
|
||||
name = "rand"
|
||||
version = "0.8.7"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "22f6172bdec972074665ed81ed53b71da00bfc44b65a753cfde883ec4c702a1a"
|
||||
dependencies = [
|
||||
"libc",
|
||||
"rand_chacha 0.3.1",
|
||||
"rand_core 0.6.4",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rand"
|
||||
version = "0.9.5"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b9ef1d0d795eb7d84685bca4f72f3649f064e6641543d3a8c415898726a57b41"
|
||||
dependencies = [
|
||||
"rand_chacha 0.9.0",
|
||||
"rand_chacha",
|
||||
"rand_core 0.9.5",
|
||||
]
|
||||
|
||||
@@ -4878,16 +4802,6 @@ dependencies = [
|
||||
"rand_core 0.10.1",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rand_chacha"
|
||||
version = "0.3.1"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e6c10a63a0fa32252be49d21e7709d4d4baf8d231c2dbce1eaa8141b9b127d88"
|
||||
dependencies = [
|
||||
"ppv-lite86",
|
||||
"rand_core 0.6.4",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rand_chacha"
|
||||
version = "0.9.0"
|
||||
@@ -4898,15 +4812,6 @@ dependencies = [
|
||||
"rand_core 0.9.5",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rand_core"
|
||||
version = "0.6.4"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "ec0be4795e2f6a28069bec0b5ff3e2ac9bafc99e6a9a7dc3547996c5c816922c"
|
||||
dependencies = [
|
||||
"getrandom 0.2.17",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rand_core"
|
||||
version = "0.9.5"
|
||||
@@ -5160,7 +5065,7 @@ dependencies = [
|
||||
"tokio",
|
||||
"tokio-rustls",
|
||||
"tokio-util",
|
||||
"tower 0.5.3",
|
||||
"tower",
|
||||
"tower-http",
|
||||
"tower-service",
|
||||
"url",
|
||||
@@ -5199,7 +5104,7 @@ dependencies = [
|
||||
"tokio",
|
||||
"tokio-rustls",
|
||||
"tokio-util",
|
||||
"tower 0.5.3",
|
||||
"tower",
|
||||
"tower-http",
|
||||
"tower-service",
|
||||
"url",
|
||||
@@ -5309,15 +5214,6 @@ dependencies = [
|
||||
"security-framework",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rustls-pemfile"
|
||||
version = "2.2.0"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "dce314e5fee3f39953d46bb63bb8a46d40c2f8fb7cc5a3b6cab2bde9721d6e50"
|
||||
dependencies = [
|
||||
"rustls-pki-types",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "rustls-pki-types"
|
||||
version = "1.15.1"
|
||||
@@ -5441,13 +5337,14 @@ dependencies = [
|
||||
"async-trait",
|
||||
"axum",
|
||||
"prometheus",
|
||||
"prost 0.13.5",
|
||||
"prost-types 0.13.5",
|
||||
"prost",
|
||||
"prost-types",
|
||||
"protoc-bin-vendored",
|
||||
"tokio",
|
||||
"tokio-stream",
|
||||
"tonic",
|
||||
"tonic-build",
|
||||
"tonic-prost",
|
||||
"tonic-prost-build",
|
||||
"tracing",
|
||||
]
|
||||
|
||||
@@ -5723,16 +5620,6 @@ dependencies = [
|
||||
"syn 2.0.119",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "socket2"
|
||||
version = "0.5.10"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "e22376abed350d73dd1cd119b57ffccad95b4e585a7cda43e286245ce23c0678"
|
||||
dependencies = [
|
||||
"libc",
|
||||
"windows-sys 0.52.0",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "socket2"
|
||||
version = "0.6.5"
|
||||
@@ -6027,7 +5914,7 @@ dependencies = [
|
||||
"parking_lot",
|
||||
"pin-project-lite",
|
||||
"signal-hook-registry",
|
||||
"socket2 0.6.5",
|
||||
"socket2",
|
||||
"tokio-macros",
|
||||
"windows-sys 0.61.2",
|
||||
]
|
||||
@@ -6081,11 +5968,10 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "tonic"
|
||||
version = "0.12.3"
|
||||
version = "0.14.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "877c5b330756d856ffcc4553ab34a5684481ade925ecc54bcd1bf02b1d0d4d52"
|
||||
checksum = "ac2a5518c70fa84342385732db33fb3f44bc4cc748936eb5833d2df34d6445ef"
|
||||
dependencies = [
|
||||
"async-stream",
|
||||
"async-trait",
|
||||
"axum",
|
||||
"base64 0.22.1",
|
||||
@@ -6099,13 +5985,12 @@ dependencies = [
|
||||
"hyper-util",
|
||||
"percent-encoding",
|
||||
"pin-project",
|
||||
"prost 0.13.5",
|
||||
"rustls-pemfile",
|
||||
"socket2 0.5.10",
|
||||
"socket2",
|
||||
"sync_wrapper",
|
||||
"tokio",
|
||||
"tokio-rustls",
|
||||
"tokio-stream",
|
||||
"tower 0.4.13",
|
||||
"tower",
|
||||
"tower-layer",
|
||||
"tower-service",
|
||||
"tracing",
|
||||
@@ -6113,36 +5998,41 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "tonic-build"
|
||||
version = "0.12.3"
|
||||
version = "0.14.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "9557ce109ea773b399c9b9e5dca39294110b74f1f342cb347a80d1fce8c26a11"
|
||||
checksum = "c68f61875ac5293cf72e6c8cf0158086428c82c37229e98c840878f1706b0322"
|
||||
dependencies = [
|
||||
"prettyplease",
|
||||
"proc-macro2",
|
||||
"prost-build 0.13.5",
|
||||
"prost-types 0.13.5",
|
||||
"quote",
|
||||
"syn 2.0.119",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "tower"
|
||||
version = "0.4.13"
|
||||
name = "tonic-prost"
|
||||
version = "0.14.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "b8fa9be0de6cf49e536ce1851f987bd21a43b771b09473c3549a6c853db37c1c"
|
||||
checksum = "50849f68853be452acf590cde0b146665b8d507b3b8af17261df47e02c209ea0"
|
||||
dependencies = [
|
||||
"futures-core",
|
||||
"futures-util",
|
||||
"indexmap 1.9.3",
|
||||
"pin-project",
|
||||
"pin-project-lite",
|
||||
"rand 0.8.7",
|
||||
"slab",
|
||||
"tokio",
|
||||
"tokio-util",
|
||||
"tower-layer",
|
||||
"tower-service",
|
||||
"tracing",
|
||||
"bytes",
|
||||
"prost",
|
||||
"tonic",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "tonic-prost-build"
|
||||
version = "0.14.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "654e5643eff75d7f8c99197ce1440ed19a3474eada74c12bbac488b2cafdae27"
|
||||
dependencies = [
|
||||
"prettyplease",
|
||||
"proc-macro2",
|
||||
"prost-build",
|
||||
"prost-types",
|
||||
"quote",
|
||||
"syn 2.0.119",
|
||||
"tempfile",
|
||||
"tonic-build",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -6153,9 +6043,12 @@ checksum = "ebe5ef63511595f1344e2d5cfa636d973292adc0eec1f0ad45fae9f0851ab1d4"
|
||||
dependencies = [
|
||||
"futures-core",
|
||||
"futures-util",
|
||||
"indexmap 2.14.0",
|
||||
"pin-project-lite",
|
||||
"slab",
|
||||
"sync_wrapper",
|
||||
"tokio",
|
||||
"tokio-util",
|
||||
"tower-layer",
|
||||
"tower-service",
|
||||
"tracing",
|
||||
@@ -6178,7 +6071,7 @@ dependencies = [
|
||||
"pin-project-lite",
|
||||
"tokio",
|
||||
"tokio-util",
|
||||
"tower 0.5.3",
|
||||
"tower",
|
||||
"tower-layer",
|
||||
"tower-service",
|
||||
"url",
|
||||
|
||||
@@ -10,19 +10,29 @@ members = ["crates/core", "crates/lance", "crates/sort"]
|
||||
|
||||
[workspace.package]
|
||||
version = "0.1.0"
|
||||
edition = "2021"
|
||||
edition = "2024"
|
||||
# The edition needs 1.85; the dependency tree needs more. Verified with
|
||||
# `cargo +1.94.1 check --all-targets` (1.94.0 fails on the AWS SDK that
|
||||
# lance's `aws` feature pulls in).
|
||||
rust-version = "1.94.1"
|
||||
|
||||
[workspace.lints.clippy]
|
||||
# Protobuf message literals keep `..Default::default()` on purpose: it is
|
||||
# what lets a proto gain a field without touching every constructor.
|
||||
needless_update = "allow"
|
||||
|
||||
[workspace.dependencies]
|
||||
anyhow = "1"
|
||||
async-trait = "0.1"
|
||||
prost = "0.13"
|
||||
prost-types = "0.13"
|
||||
prost = "0.14"
|
||||
prost-types = "0.14"
|
||||
tokio = { version = "1", features = ["full"] }
|
||||
tokio-stream = "0.1"
|
||||
tonic = { version = "0.12", features = ["tls"] }
|
||||
tonic = { version = "0.14", features = ["tls-aws-lc"] }
|
||||
tonic-prost = "0.14"
|
||||
# Already in the tree via tonic; named here so the metrics server can use them.
|
||||
axum = "0.7"
|
||||
axum = "0.8"
|
||||
prometheus = { version = "0.13", default-features = false }
|
||||
tonic-build = "0.12"
|
||||
tonic-prost-build = "0.14"
|
||||
tracing = "0.1"
|
||||
tracing-subscriber = { version = "0.3", features = ["env-filter"] }
|
||||
|
||||
@@ -14,6 +14,12 @@ one.
|
||||
|
||||
## Building
|
||||
|
||||
Requires Rust 1.94.1+ (2024 edition), matching `rust-version` in `Cargo.toml`.
|
||||
The patch release matters: 1.94.0 does not build. The edition itself only needs
|
||||
1.85; the higher floor comes from the dependency tree — lance's `aws` feature
|
||||
pulls in the AWS SDK — so it moves with those crates. CI builds on the latest
|
||||
stable.
|
||||
|
||||
`core` compiles `plugin.proto` with the protoc that protoc-bin-vendored ships,
|
||||
the way seaweed-volume does, so it needs no system install.
|
||||
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
name = "seaweed-worker-core"
|
||||
version.workspace = true
|
||||
edition.workspace = true
|
||||
rust-version.workspace = true
|
||||
description = "SeaweedFS plugin.proto worker contract"
|
||||
|
||||
[lib]
|
||||
@@ -15,13 +16,17 @@ prost-types.workspace = true
|
||||
tokio.workspace = true
|
||||
tokio-stream.workspace = true
|
||||
tonic.workspace = true
|
||||
tonic-prost.workspace = true
|
||||
axum.workspace = true
|
||||
prometheus.workspace = true
|
||||
tracing.workspace = true
|
||||
|
||||
[build-dependencies]
|
||||
tonic-build.workspace = true
|
||||
tonic-prost-build.workspace = true
|
||||
# Ships protoc with the build so neither CI nor a developer needs a system
|
||||
# install, and so the version is pinned rather than whatever the platform's
|
||||
# package manager happens to carry. The same crate seaweed-volume uses.
|
||||
protoc-bin-vendored = "3"
|
||||
|
||||
[lints]
|
||||
workspace = true
|
||||
|
||||
@@ -4,12 +4,17 @@ fn main() -> Result<(), Box<dyn std::error::Error>> {
|
||||
// version. An explicit PROTOC still wins, for packagers supplying their own
|
||||
// and for the lance crates, whose own build scripts read the same variable.
|
||||
if std::env::var_os("PROTOC").is_none() {
|
||||
std::env::set_var("PROTOC", protoc_bin_vendored::protoc_bin_path()?);
|
||||
// SAFETY: a build script's main runs single-threaded before anything
|
||||
// else in this process, so no other thread can be reading the
|
||||
// environment concurrently.
|
||||
unsafe {
|
||||
std::env::set_var("PROTOC", protoc_bin_vendored::protoc_bin_path()?);
|
||||
}
|
||||
}
|
||||
|
||||
// Compiled straight out of the Go tree, the way seaweed-volume already reads
|
||||
// filer.proto, so the contract cannot drift from a vendored copy.
|
||||
tonic_build::configure()
|
||||
tonic_prost_build::configure()
|
||||
// The server half is only for tests, which stand up a fake admin.
|
||||
.build_server(true)
|
||||
.build_client(true)
|
||||
|
||||
@@ -14,10 +14,10 @@ pub fn server_to_grpc_address(server: &str) -> Option<String> {
|
||||
let (host, port_part) = server.rsplit_once(':')?;
|
||||
|
||||
// "port.grpcPort" states the gRPC port outright.
|
||||
if let Some((_, grpc_port)) = port_part.split_once('.') {
|
||||
if let Ok(port) = grpc_port.parse::<u16>() {
|
||||
return Some(join_host_port(host, port));
|
||||
}
|
||||
if let Some((_, grpc_port)) = port_part.split_once('.')
|
||||
&& let Ok(port) = grpc_port.parse::<u16>()
|
||||
{
|
||||
return Some(join_host_port(host, port));
|
||||
}
|
||||
|
||||
let port: u16 = port_part.parse().ok()?;
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
use std::collections::HashMap;
|
||||
|
||||
use crate::pb::{
|
||||
config_value::Kind, ConfigField, ConfigFieldType, ConfigForm, ConfigSection, ConfigValue,
|
||||
ConfigField, ConfigFieldType, ConfigForm, ConfigSection, ConfigValue, config_value::Kind,
|
||||
};
|
||||
|
||||
pub fn int_value(value: i64) -> ConfigValue {
|
||||
|
||||
@@ -16,6 +16,9 @@ pub mod stream;
|
||||
|
||||
/// Generated plugin.proto types.
|
||||
pub mod pb {
|
||||
// prost gives every oneof its own enum; the variant sizes are the
|
||||
// messages' own, not a choice made here.
|
||||
#![allow(clippy::large_enum_variant)]
|
||||
tonic::include_proto!("plugin");
|
||||
}
|
||||
|
||||
|
||||
@@ -270,10 +270,10 @@ impl Metrics {
|
||||
/// logged rather than fatal: a worker that cannot publish metrics should still
|
||||
/// do its work.
|
||||
pub async fn serve(metrics: Metrics, addr: SocketAddr) -> Result<()> {
|
||||
use axum::Router;
|
||||
use axum::extract::State;
|
||||
use axum::http::StatusCode;
|
||||
use axum::routing::get;
|
||||
use axum::Router;
|
||||
|
||||
let app = Router::new()
|
||||
.route("/health", get(|| async { StatusCode::OK }))
|
||||
@@ -355,28 +355,37 @@ mod tests {
|
||||
|
||||
metrics.stream_connected();
|
||||
assert!(metrics.is_ready());
|
||||
assert!(metrics
|
||||
.gather()
|
||||
.unwrap()
|
||||
.contains("SeaweedFS_worker_connected 1"));
|
||||
assert!(
|
||||
metrics
|
||||
.gather()
|
||||
.unwrap()
|
||||
.contains("SeaweedFS_worker_connected 1")
|
||||
);
|
||||
|
||||
metrics.stream_ended("closed");
|
||||
assert!(!metrics.is_ready());
|
||||
assert!(metrics
|
||||
.gather()
|
||||
.unwrap()
|
||||
.contains("SeaweedFS_worker_connected 0"));
|
||||
assert!(metrics
|
||||
.gather()
|
||||
.unwrap()
|
||||
.contains("SeaweedFS_worker_stream_events_total{event=\"closed\"} 1"));
|
||||
assert!(
|
||||
metrics
|
||||
.gather()
|
||||
.unwrap()
|
||||
.contains("SeaweedFS_worker_connected 0")
|
||||
);
|
||||
assert!(
|
||||
metrics
|
||||
.gather()
|
||||
.unwrap()
|
||||
.contains("SeaweedFS_worker_stream_events_total{event=\"closed\"} 1")
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn build_info_names_the_worker() {
|
||||
let text = metrics().gather().expect("gather");
|
||||
assert!(text
|
||||
.contains("SeaweedFS_worker_build_info{version=\"0.1.0\",worker_id=\"worker-1\"} 1"));
|
||||
assert!(
|
||||
text.contains(
|
||||
"SeaweedFS_worker_build_info{version=\"0.1.0\",worker_id=\"worker-1\"} 1"
|
||||
)
|
||||
);
|
||||
}
|
||||
|
||||
// A format's own numbers land on the same registry, so one endpoint serves
|
||||
|
||||
@@ -2,8 +2,8 @@ use anyhow::Result;
|
||||
use tokio::sync::mpsc;
|
||||
|
||||
use crate::pb::{
|
||||
worker_to_admin_message::Body, ActivityEvent, DetectionComplete, DetectionProposals,
|
||||
JobCompleted, JobProgressUpdate, WorkerObservations, WorkerToAdminMessage,
|
||||
ActivityEvent, DetectionComplete, DetectionProposals, JobCompleted, JobProgressUpdate,
|
||||
WorkerObservations, WorkerToAdminMessage, worker_to_admin_message::Body,
|
||||
};
|
||||
|
||||
/// Replies to one detection request.
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
use std::sync::Arc;
|
||||
use std::time::{Duration, Instant};
|
||||
|
||||
use anyhow::{anyhow, Context, Result};
|
||||
use tokio::sync::{mpsc, Semaphore};
|
||||
use anyhow::{Context, Result, anyhow};
|
||||
use tokio::sync::{Semaphore, mpsc};
|
||||
use tokio_stream::wrappers::UnboundedReceiverStream;
|
||||
use tonic::transport::{Certificate, Channel, ClientTlsConfig, Identity};
|
||||
use tracing::{info, warn};
|
||||
@@ -10,11 +10,11 @@ use tracing::{info, warn};
|
||||
use crate::config::WorkerOptions;
|
||||
use crate::metrics::Metrics;
|
||||
use crate::pb::{
|
||||
ConfigSchemaResponse, ExecuteJobRequest, JobCompleted, ObjectPreviewResponse, PreviewRow,
|
||||
RequestObjectPreview, RunDetectionRequest, RunningWork, WorkerHeartbeat, WorkerHello,
|
||||
admin_to_worker_message::Body as AdminBody,
|
||||
plugin_control_service_client::PluginControlServiceClient,
|
||||
worker_to_admin_message::Body as WorkerBody, ConfigSchemaResponse, ExecuteJobRequest,
|
||||
JobCompleted, ObjectPreviewResponse, PreviewRow, RequestObjectPreview, RunDetectionRequest,
|
||||
RunningWork, WorkerHeartbeat, WorkerHello,
|
||||
worker_to_admin_message::Body as WorkerBody,
|
||||
};
|
||||
use crate::registry::Registry;
|
||||
use crate::senders::{MeteredSender, StreamSender};
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
name = "weed-lance-worker"
|
||||
version.workspace = true
|
||||
edition.workspace = true
|
||||
rust-version.workspace = true
|
||||
description = "SeaweedFS maintenance worker for Lance tables"
|
||||
|
||||
[lib]
|
||||
@@ -49,3 +50,6 @@ arrow-array = "58"
|
||||
arrow-schema = "58"
|
||||
arrow-cast = "58"
|
||||
lance-linalg = "10"
|
||||
|
||||
[lints]
|
||||
workspace = true
|
||||
|
||||
@@ -7,4 +7,4 @@
|
||||
|
||||
pub mod namespace;
|
||||
|
||||
pub use namespace::{parse_id, NamespaceClient, TableDescription};
|
||||
pub use namespace::{NamespaceClient, TableDescription, parse_id};
|
||||
|
||||
@@ -8,8 +8,8 @@
|
||||
use std::collections::HashMap;
|
||||
|
||||
use anyhow::{Context, Result};
|
||||
use lance::dataset::builder::DatasetBuilder;
|
||||
use lance::dataset::Dataset;
|
||||
use lance::dataset::builder::DatasetBuilder;
|
||||
|
||||
use crate::catalog::{NamespaceClient, TableDescription};
|
||||
|
||||
|
||||
@@ -1,9 +1,9 @@
|
||||
use std::collections::HashMap;
|
||||
|
||||
use anyhow::{anyhow, Context, Result};
|
||||
use anyhow::{Context, Result, anyhow};
|
||||
use async_trait::async_trait;
|
||||
use chrono::{Duration, Utc};
|
||||
use lance::dataset::cleanup::{cleanup_old_versions, CleanupPolicy};
|
||||
use lance::dataset::cleanup::{CleanupPolicy, cleanup_old_versions};
|
||||
use seaweed_worker_core::config_form::{form, int_or, int_value, number_field};
|
||||
use seaweed_worker_core::pb::{
|
||||
ConfigValue, DetectionComplete, DetectionProposals, ExecuteJobRequest, JobCompleted,
|
||||
@@ -13,7 +13,7 @@ use seaweed_worker_core::pb::{
|
||||
use seaweed_worker_core::{DetectionSender, ExecutionSender, JobHandler};
|
||||
use tracing::warn;
|
||||
|
||||
use crate::catalog::{parse_id, NamespaceClient};
|
||||
use crate::catalog::{NamespaceClient, parse_id};
|
||||
use crate::dataset;
|
||||
use crate::jobs::{clamp, string_list, table_id};
|
||||
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
use std::collections::HashMap;
|
||||
|
||||
use anyhow::{anyhow, Context, Result};
|
||||
use anyhow::{Context, Result, anyhow};
|
||||
use async_trait::async_trait;
|
||||
use lance::dataset::optimize::{compact_files, CompactionOptions};
|
||||
use lance::dataset::optimize::{CompactionOptions, compact_files};
|
||||
use seaweed_worker_core::config_form::{form, int_or, int_value, number_field, string_value};
|
||||
use seaweed_worker_core::pb::{
|
||||
ConfigValue, DetectionComplete, DetectionProposals, ExecuteJobRequest, JobCompleted,
|
||||
@@ -12,9 +12,9 @@ use seaweed_worker_core::pb::{
|
||||
use seaweed_worker_core::{DetectionSender, ExecutionSender, JobHandler};
|
||||
use tracing::warn;
|
||||
|
||||
use crate::catalog::{parse_id, NamespaceClient};
|
||||
use crate::catalog::{NamespaceClient, parse_id};
|
||||
use crate::dataset;
|
||||
use crate::jobs::{clamp, observation, string_list, table_id, FORMAT};
|
||||
use crate::jobs::{FORMAT, clamp, observation, string_list, table_id};
|
||||
|
||||
pub const JOB_TYPE: &str = "lance_compact";
|
||||
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
use std::collections::HashMap;
|
||||
|
||||
use anyhow::{anyhow, Context, Result};
|
||||
use anyhow::{Context, Result, anyhow};
|
||||
use async_trait::async_trait;
|
||||
use lance::index::DatasetIndexExt;
|
||||
use lance_index::optimize::OptimizeOptions;
|
||||
@@ -13,7 +13,7 @@ use seaweed_worker_core::pb::{
|
||||
use seaweed_worker_core::{DetectionSender, ExecutionSender, JobHandler};
|
||||
use tracing::warn;
|
||||
|
||||
use crate::catalog::{parse_id, NamespaceClient};
|
||||
use crate::catalog::{NamespaceClient, parse_id};
|
||||
use crate::dataset::{self, OpenTable};
|
||||
use crate::jobs::{clamp, string_list, table_id};
|
||||
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user