mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-10-09 08:05:51 +00:00
Compare commits
83
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
a6e07181b2 | ||
|
|
51831d6850 | ||
|
|
d235dd280b | ||
|
|
7b9332953d | ||
|
|
85147522a9 | ||
|
|
207bc0b75a | ||
|
|
2f26d5779b | ||
|
|
14bc6e5e4f | ||
|
|
dfde24f3ee | ||
|
|
e735c12869 | ||
|
|
e4ca0d09e7 | ||
|
|
01433e801d | ||
|
|
1f61097d4d | ||
|
|
0e82b4e351 | ||
|
|
0bd048b76f | ||
|
|
c997e54096 | ||
|
|
2aa6af033d | ||
|
|
02749c1192 | ||
|
|
ac03d3fd78 | ||
|
|
adaf3534fa | ||
|
|
49f20489e4 | ||
|
|
bdec508da9 | ||
|
|
fd33c07843 | ||
|
|
cf38c01978 | ||
|
|
f4bad510c9 | ||
|
|
15d9f6c6fe | ||
|
|
ea179963c0 | ||
|
|
c507336000 | ||
|
|
3c492b5ab1 | ||
|
|
38c14d3c13 | ||
|
|
92c379e5b4 | ||
|
|
5d8a463b3e | ||
|
|
bea10e269f | ||
|
|
10c0857476 | ||
|
|
c462fffce6 | ||
|
|
99d2479528 | ||
|
|
8db41d0217 | ||
|
|
bd6bcd47e3 | ||
|
|
eb6a7e93ca | ||
|
|
5b2fe374fc | ||
|
|
3b4a681e53 | ||
|
|
c46f82d29a | ||
|
|
5a0e017457 | ||
|
|
210afacd12 | ||
|
|
9f6feef299 | ||
|
|
79994b69af | ||
|
|
42b0ca7850 | ||
|
|
9b902a7662 | ||
|
|
d8aa7ecf04 | ||
|
|
2ebfeabfce | ||
|
|
a3638e479e | ||
|
|
80dae68dbf | ||
|
|
5ff49909a0 | ||
|
|
bc0efa4d10 | ||
|
|
3ae9e332ec | ||
|
|
0de9c1f231 | ||
|
|
b3aace2a08 | ||
|
|
4f9bbd51cb | ||
|
|
e919bec9d1 | ||
|
|
2cd6c36c54 | ||
|
|
7fa2f75f30 | ||
|
|
5061a16b12 | ||
|
|
3c9a4bbdda | ||
|
|
13bf056a15 | ||
|
|
c968084b34 | ||
|
|
516e251f9e | ||
|
|
01fc31cb71 | ||
|
|
966692fa23 | ||
|
|
168b9c39f8 | ||
|
|
f1f6886d0e | ||
|
|
2ffa696809 | ||
|
|
0ce5ca42ea | ||
|
|
9b12d13934 | ||
|
|
723f473f02 | ||
|
|
5f77a0b67e | ||
|
|
fd4fa72289 | ||
|
|
cb9fcd39d2 | ||
|
|
8782749f26 | ||
|
|
b88156fe6b | ||
|
|
557fffa350 | ||
|
|
4a1d65939f | ||
|
|
c6b330be2b | ||
|
|
213f4c5d5c |
@@ -27,7 +27,7 @@ jobs:
|
||||
|
||||
# Initializes the CodeQL tools for scanning.
|
||||
- name: Initialize CodeQL
|
||||
uses: github/codeql-action/init@v4.37.9
|
||||
uses: github/codeql-action/init@v4.38.0
|
||||
# Override language selection by uncommenting this and choosing your languages
|
||||
with:
|
||||
languages: go
|
||||
@@ -35,7 +35,7 @@ jobs:
|
||||
# Autobuild attempts to build any compiled languages (C/C++, C#, or Java).
|
||||
# If this step fails, then you should remove it and run the build manually (see below).
|
||||
- name: Autobuild
|
||||
uses: github/codeql-action/autobuild@v4.37.9
|
||||
uses: github/codeql-action/autobuild@v4.38.0
|
||||
|
||||
# ℹ️ Command-line programs to run using the OS shell.
|
||||
# 📚 See https://docs.github.com/en/actions/using-workflows/workflow-syntax-for-github-actions#jobsjob_idstepsrun
|
||||
@@ -49,4 +49,4 @@ jobs:
|
||||
# make release
|
||||
|
||||
- name: Perform CodeQL Analysis
|
||||
uses: github/codeql-action/analyze@v4.37.9
|
||||
uses: github/codeql-action/analyze@v4.38.0
|
||||
|
||||
@@ -405,7 +405,7 @@ jobs:
|
||||
output: trivy-results.sarif
|
||||
exit-code: '0'
|
||||
- name: Upload Trivy scan results to GitHub Security
|
||||
uses: github/codeql-action/upload-sarif@v4.37.9
|
||||
uses: github/codeql-action/upload-sarif@v4.38.0
|
||||
if: always()
|
||||
with:
|
||||
sarif_file: trivy-results.sarif
|
||||
|
||||
@@ -456,7 +456,7 @@ jobs:
|
||||
|
||||
- name: Upload Trivy scan results to GitHub Security
|
||||
if: always()
|
||||
uses: github/codeql-action/upload-sarif@v4.37.9
|
||||
uses: github/codeql-action/upload-sarif@v4.38.0
|
||||
with:
|
||||
sarif_file: trivy-results.sarif
|
||||
category: trivy-${{ matrix.variant }}
|
||||
|
||||
@@ -227,6 +227,9 @@ jobs:
|
||||
out = render({
|
||||
"global.seaweedfs.securityConfig.jwtSigning.filerWrite": "true",
|
||||
"admin.enabled": "true",
|
||||
# admin.ip defaults to 0.0.0.0 (non-loopback), which weed admin 4.46
|
||||
# refuses to bind without authentication.
|
||||
"admin.secret.adminPassword": "ci-admin-password",
|
||||
})
|
||||
cm = configmap(out, "test-seaweedfs-security-config")
|
||||
if cm is None:
|
||||
@@ -1141,6 +1144,9 @@ jobs:
|
||||
"s3.enabled": "true",
|
||||
"sftp.enabled": "true",
|
||||
"admin.enabled": "true",
|
||||
# admin.ip defaults to 0.0.0.0 (non-loopback), which weed admin 4.46
|
||||
# refuses to bind without authentication.
|
||||
"admin.secret.adminPassword": "ci-admin-password",
|
||||
"worker.enabled": "true",
|
||||
"cosi.enabled": "true",
|
||||
"s3.createBuckets[0].name": "b",
|
||||
@@ -1337,7 +1343,7 @@ jobs:
|
||||
# Which means egress on its own must render for a release that runs
|
||||
# neither COSI nor a resize: no component of it reaches the API server,
|
||||
# so nothing may demand a CIDR for one.
|
||||
for label, values in {"defaults": {}, "admin": {"admin.enabled": "true"}}.items():
|
||||
for label, values in {"defaults": {}, "admin": {"admin.enabled": "true", "admin.secret.adminPassword": "ci-admin-password"}}.items():
|
||||
try:
|
||||
render(dict(values, **{"networkPolicy.enabled": "true",
|
||||
"networkPolicy.egress.enabled": "true"}))
|
||||
@@ -1398,6 +1404,9 @@ jobs:
|
||||
|
||||
ALL_ON = {
|
||||
"admin.enabled": "true",
|
||||
# admin.ip defaults to 0.0.0.0 (non-loopback), which weed admin 4.46
|
||||
# refuses to bind without authentication.
|
||||
"admin.secret.adminPassword": "ci-admin-password",
|
||||
"s3.enabled": "true",
|
||||
"sftp.enabled": "true",
|
||||
"worker.enabled": "true",
|
||||
|
||||
@@ -206,8 +206,9 @@ jobs:
|
||||
|
||||
# The dispatched workflow pins seaweedfs with `go get -u ...@latest`, so
|
||||
# wait until the proxy serves the release commit as the tip. Asking for
|
||||
# the commit by name is what makes the proxy fetch it.
|
||||
for _ in $(seq 30); do
|
||||
# the commit by name is what makes the proxy fetch it. The proxy can
|
||||
# take longer than five minutes to refresh @latest after a new tag.
|
||||
for _ in $(seq 120); do
|
||||
curl -sf "https://proxy.golang.org/${MODULE}/@v/${SHA}.info" >/dev/null || true
|
||||
TIP=$(curl -sf "https://proxy.golang.org/${MODULE}/@latest" | jq -r '.Origin.Hash // ""' || true)
|
||||
[ "$TIP" = "$SHA" ] && break
|
||||
|
||||
@@ -59,6 +59,12 @@ jobs:
|
||||
- name: Build Rust volume server
|
||||
run: cd seaweed-volume && cargo build --release
|
||||
|
||||
# The crate is warning-free under clippy as of the sweep that added
|
||||
# this step. Uncomment to make that a gate; `[lints.clippy]` in
|
||||
# seaweed-volume/Cargo.toml is where crate-wide exceptions live.
|
||||
# - name: Clippy
|
||||
# run: cd seaweed-volume && cargo clippy --all-targets -- -D warnings
|
||||
|
||||
- name: Run Rust unit tests
|
||||
run: cd seaweed-volume && cargo test
|
||||
|
||||
|
||||
@@ -73,6 +73,12 @@ jobs:
|
||||
- name: Build the plugin workers
|
||||
run: cd seaweed-worker && cargo build --release
|
||||
|
||||
# The workspace is warning-free under clippy as of the sweep that added
|
||||
# this step. Uncomment to make that a gate; `[workspace.lints.clippy]`
|
||||
# in seaweed-worker/Cargo.toml is where crate-wide exceptions live.
|
||||
# - name: Clippy
|
||||
# run: cd seaweed-worker && cargo clippy --workspace --all-targets -- -D warnings
|
||||
|
||||
# The tests that need a live gateway skip themselves without one, the way
|
||||
# the Go integration tests skip without Docker; the lifecycle suite in
|
||||
# test/s3tables/lifecycle is what runs them against a real cluster.
|
||||
|
||||
@@ -9,6 +9,14 @@ on:
|
||||
permissions:
|
||||
contents: write
|
||||
|
||||
concurrency:
|
||||
# Only one chart regeneration per branch at a time; a newer run on the same
|
||||
# branch cancels an in-flight one so overlapping runs never conflict on
|
||||
# note/star_history.svg during rebase. Scoped by ref so a manual run on
|
||||
# another branch can't cancel the daily master update.
|
||||
group: star-history-${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
render:
|
||||
name: Regenerate star history chart
|
||||
@@ -17,6 +25,9 @@ jobs:
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v7
|
||||
with:
|
||||
# Full history so the chart commit can rebase onto a moved master.
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v7
|
||||
@@ -43,4 +54,21 @@ jobs:
|
||||
fi
|
||||
git add note/star_history.svg
|
||||
git commit -m "docs: regenerate star history chart"
|
||||
git push
|
||||
# Rebase and retry so a concurrent push to master doesn't lose the chart.
|
||||
for attempt in 1 2 3 4 5; do
|
||||
if [ "$attempt" -gt 1 ]; then
|
||||
# Guard the rebase: a transient fetch error or conflict must not
|
||||
# abort the fail-fast shell before the remaining attempts run.
|
||||
if ! git pull --rebase origin "$GITHUB_REF_NAME"; then
|
||||
echo "rebase failed (attempt ${attempt}); aborting and retrying"
|
||||
git rebase --abort || true
|
||||
continue
|
||||
fi
|
||||
fi
|
||||
if git push origin HEAD:"$GITHUB_REF_NAME"; then
|
||||
exit 0
|
||||
fi
|
||||
echo "push rejected (attempt ${attempt}); will rebase and retry"
|
||||
done
|
||||
echo "::error::could not push star history chart after retries"
|
||||
exit 1
|
||||
|
||||
@@ -0,0 +1,418 @@
|
||||
# SeaweedFS as an Apache CloudStack Object Storage Provider
|
||||
|
||||
A CloudStack ObjectStore plugin that makes SeaweedFS a first-class object storage
|
||||
backend inside Apache CloudStack, alongside the existing MinIO and Ceph RGW
|
||||
providers. This is a collaboration with proIO (Swen), who builds private clouds on
|
||||
CloudStack and wants SeaweedFS as a storage option.
|
||||
|
||||
## The request
|
||||
|
||||
> We can only add MinIO and Ceph as object storage [in CloudStack] today. I want
|
||||
> to get SeaweedFS into this project... What we need is to build a provider which
|
||||
> does the communication between Cloudstack and SeaweedFS.
|
||||
|
||||
This is **not** a SeaweedFS-side feature. The work lives in the Apache CloudStack
|
||||
repo (Java): a new plugin under `plugins/storage/object/seaweedfs/` that implements
|
||||
CloudStack's ObjectStore plugin framework and talks to SeaweedFS over its S3 and
|
||||
IAM APIs. SeaweedFS itself needs no changes for the core to work — its S3 API
|
||||
already covers every bucket operation CloudStack requires, and its IAM API covers
|
||||
user/credential management.
|
||||
|
||||
## How the CloudStack ObjectStore framework works
|
||||
|
||||
CloudStack 4.18+ introduced an Object Storage framework. An admin registers an
|
||||
object storage pool via `addObjectStoragePool` (URL + provider + credentials);
|
||||
tenants then create and manage buckets on it through CloudStack APIs. CloudStack
|
||||
manages pool and bucket lifecycle; the underlying provider handles the actual
|
||||
object protocol.
|
||||
|
||||
A provider is a plugin module implementing three interfaces:
|
||||
|
||||
### 1. `ObjectStoreProvider` — registration
|
||||
|
||||
`MinIOObjectStoreProviderImpl` is the reference. It is a Spring `@Component` that:
|
||||
- Returns a provider name (`"MinIO"`)
|
||||
- Returns `DataStoreProviderType.OBJECT`
|
||||
- In `configure()`, injects the lifecycle and driver implementations and calls
|
||||
`storeMgr.registerDriver(name, driver)`
|
||||
|
||||
### 2. `ObjectStoreLifeCycle` — pool add/remove
|
||||
|
||||
`MinIOObjectStoreLifeCycleImpl.initialize()` reads the URL, name, and
|
||||
`accesskey`/`secretkey` details from the `addObjectStoragePool` call, tests the
|
||||
connection by listing buckets, and persists an `ObjectStoreVO` via
|
||||
`ObjectStoreHelper`. The other methods (attachCluster/Host/Zone, maintain,
|
||||
deleteDataStore) are no-ops for object storage.
|
||||
|
||||
### 3. `ObjectStoreDriver` — bucket + user operations
|
||||
|
||||
`ObjectStoreDriver` (in `engine/storage/.../object/ObjectStoreDriver.java`) extends
|
||||
`DataStoreDriver` and defines the bucket/user contract. Every provider must
|
||||
implement:
|
||||
|
||||
| Method | Purpose |
|
||||
| --- | --- |
|
||||
| `createBucket(Bucket, boolean objectLock)` | Create a bucket |
|
||||
| `listBuckets(long storeId)` | List all buckets |
|
||||
| `deleteBucket(BucketTO, long storeId)` | Delete a bucket |
|
||||
| `createUser(long accountId, long storeId)` | Provision a user + credentials for a CloudStack account |
|
||||
| `setBucketPolicy` / `getBucketPolicy` / `deleteBucketPolicy` | Bucket policy CRUD |
|
||||
| `setBucketEncryption` / `deleteBucketEncryption` | SSE config |
|
||||
| `setBucketVersioning` / `deleteBucketVersioning` | Versioning enable/suspend |
|
||||
| `setBucketQuota(BucketTO, long storeId, long size)` | Per-bucket quota |
|
||||
| `getAllBucketsUsage(long storeId)` | Usage map for billing/accounting |
|
||||
| `getBucketAcl` / `setBucketAcl` | ACLs (MinIO/Ceph return null / no-op) |
|
||||
|
||||
`BaseObjectStoreDriverImpl` provides no-op defaults for the `DataStoreDriver`
|
||||
methods (`createAsync`, `deleteAsync`, `copyAsync`, `canCopy`, `resize`,
|
||||
`getTO`, `getStoreTO`), so object-store providers only implement the bucket/user
|
||||
methods above.
|
||||
|
||||
## How the four existing providers differ (and where SeaweedFS lands)
|
||||
|
||||
CloudStack ships four object-store providers. Three are relevant; the simulator
|
||||
is a test stub.
|
||||
|
||||
| Concern | MinIO | Ceph RGW | Cloudian HyperStore | SeaweedFS |
|
||||
| --- | --- | --- | --- | --- |
|
||||
| Bucket CRUD | `MinioClient` (S3) | `AmazonS3` (AWS SDK v1) | `AmazonS3` (AWS SDK v1) | `AmazonS3` (AWS SDK v1) |
|
||||
| Bucket policy | `MinioClient` | `AmazonS3` | `AmazonS3` | `AmazonS3` |
|
||||
| Versioning | `MinioClient` | `AmazonS3` | `AmazonS3` | `AmazonS3` |
|
||||
| Encryption | `MinioClient` | not implemented | `AmazonS3` | `AmazonS3` |
|
||||
| **User creation** | `MinioAdminClient` | `RgwAdmin` | **`AmazonIdentityManagement`** | **`AmazonIdentityManagement`** |
|
||||
| **Per-bucket quota** | `MinioAdminClient` | `RgwAdmin` | **not supported** (throws) | **S3 extension** (`PUT /{bucket}?seaweedfs-quota`, SigV4, `s3:PutBucketQuota`) |
|
||||
| **Usage reporting** | `MinioAdminClient` | `RgwAdmin` | Cloudian admin API | S3 `ListObjectsV2` (MVP); Prometheus / SOSAPI `capacity.xml` (recommended) |
|
||||
|
||||
**Cloudian HyperStore is the direct precedent.** It is an S3-compatible store
|
||||
that, like SeaweedFS, manages users via the **standard AWS IAM API** using the
|
||||
AWS IAM Java SDK (`com.amazonaws.services.identitymanagement`). Its driver
|
||||
(`CloudianHyperStoreObjectStoreDriverImpl`) and util
|
||||
(`CloudianHyperStoreUtil`) are the template this design follows almost line for
|
||||
line. Cloudian even validates the quota limitation the same way this design
|
||||
proposes for the MVP: `setBucketQuota` throws for any non-zero size and only
|
||||
accepts `0` (no quota).
|
||||
|
||||
The SeaweedFS plugin is therefore a **simpler Cloudian** — same AWS S3 + IAM SDK
|
||||
clients, same store-details keys (`s3Url`, `iamUrl`, `accesskey`, `secretkey`),
|
||||
same IAM-user-with-restricted-policy pattern, but with no proprietary admin
|
||||
client at all (Cloudian has its own `CloudianClient` for its admin API; SeaweedFS
|
||||
needs only S3 + IAM). For quota, the plugin uses a narrow SeaweedFS S3 extension
|
||||
(see below); for usage reporting, it falls back to S3 `ListObjectsV2` in the MVP
|
||||
and recommends Prometheus or SOSAPI `capacity.xml` for production scale.
|
||||
|
||||
### Quota via the S3 `?seaweedfs-quota` extension
|
||||
|
||||
SeaweedFS supports bucket quota natively (server-side enforcement via a
|
||||
read-only flag when usage exceeds the limit). Rather than exposing the broad
|
||||
admin REST API (which would require a global bearer token and grant cluster-wide
|
||||
admin access), the integration uses a **narrow, scoped S3 subresource**:
|
||||
|
||||
- `PUT /{bucket}?seaweedfs-quota` — set bucket quota (IAM permission `s3:PutBucketQuota`)
|
||||
- `GET /{bucket}?seaweedfs-quota` — get bucket quota (IAM permission `s3:GetBucketQuota`)
|
||||
|
||||
**PUT request body** (JSON):
|
||||
```json
|
||||
{"quota_size": 100, "quota_unit": "GB", "quota_enabled": true}
|
||||
```
|
||||
|
||||
**GET response body** (JSON):
|
||||
```json
|
||||
{"quota_size": 107374182400, "quota_unit": "B", "quota_enabled": true}
|
||||
```
|
||||
|
||||
Note: GET always returns `quota_unit: "B"` and the absolute byte count, not
|
||||
the original unit. A disabled-but-retained quota returns a positive
|
||||
`quota_size` with `quota_enabled: false`.
|
||||
|
||||
Quota is stored on the bucket's filer entry (positive = enabled, negative =
|
||||
disabled but retained, zero = no quota), matching the existing admin REST API
|
||||
behavior. When quota is cleared, the bucket's read-only flag is also lifted.
|
||||
|
||||
**Authentication** uses the existing S3 SigV4 flow — no new global secret is
|
||||
needed. The CloudStack service credential (the `accesskey`/`secretkey` on the
|
||||
object store) is the admin credential used for all driver operations: bucket
|
||||
CRUD, IAM user provisioning, and quota management. It must have broad S3 and
|
||||
IAM permissions. The per-account IAM users created by `createUser` are the
|
||||
ones with restricted permissions (full S3 access except bucket
|
||||
creation/deletion). A future hardening could split quota management onto a
|
||||
separate credential scoped to only `s3:PutBucketQuota`/`s3:GetBucketQuota`,
|
||||
but the MVP uses the single admin credential for simplicity, matching how
|
||||
the MinIO and Ceph providers work.
|
||||
|
||||
The plugin's `setBucketQuota` signs and sends the `PUT /{bucket}?seaweedfs-quota`
|
||||
request using the AWS SDK v1 `AWSS3V4Signer` for SigV4 signing, then sends the
|
||||
signed request via `java.net.http.HttpClient` (the AWS S3 SDK doesn't natively
|
||||
support custom subresources, so we sign manually and send the request
|
||||
ourselves). The `seaweedfs-quota` query parameter is included in the signed
|
||||
canonical query string.
|
||||
|
||||
### Usage reporting
|
||||
|
||||
`getAllBucketsUsage` must return a `Map<String, Long>` of bucket name → size.
|
||||
MinIO uses `MinioAdminClient.getDataUsageInfo`; Ceph uses
|
||||
`RgwAdmin.listBucketInfo`. SeaweedFS has no admin rollup endpoint, so the MVP
|
||||
plugin computes it by listing buckets and summing object sizes via S3
|
||||
`ListObjectsV2` — expensive for large stores.
|
||||
|
||||
For production scale, SeaweedFS already exposes per-bucket size in:
|
||||
- **Prometheus metrics** (`bucket_size_bytes` gauge, refreshed every minute)
|
||||
- **SOSAPI `capacity.xml`** (reports capacity, available space, and usage
|
||||
through the S3 endpoint)
|
||||
|
||||
Operators should consume one of those instead of S3 list-based aggregation for
|
||||
large deployments. The MVP's list-based approach is correct but slow; flag it as
|
||||
a known limitation.
|
||||
|
||||
## SeaweedFS API surface (what the plugin relies on)
|
||||
|
||||
SeaweedFS exposes two relevant APIs, both AWS-compatible:
|
||||
|
||||
### S3 API (`weed s3`)
|
||||
Full S3-compatible surface. Confirmed against the SeaweedFS S3 wiki and code:
|
||||
- `CreateBucket`, `HeadBucket`, `ListBuckets`, `DeleteBucket`
|
||||
- `PutBucketPolicy`, `GetBucketPolicy`, `DeleteBucketPolicy`
|
||||
- `PutBucketVersioning` (Enabled / Suspended), `GetBucketVersioning`
|
||||
- `PutBucketEncryption`, `GetBucketEncryption`, `DeleteBucketEncryption`
|
||||
- `PutBucketAcl`, `GetBucketAcl`
|
||||
- `ListObjectsV2`, `HeadObject`, `GetObject`, `PutObject`, `DeleteObject`
|
||||
- Bucket quota via extended attributes / `s3.bucket.quota` (enforced server-side,
|
||||
surfaced as a read-only state when exceeded — see PR #10224)
|
||||
|
||||
### IAM API (`weed iam` / `iamapi`)
|
||||
AWS IAM-compatible REST endpoints, implemented in `weed/iamapi/`. Confirmed by
|
||||
the test suite which uses the **AWS IAM SDK** (`aws-sdk-go/service/iam`) against
|
||||
the same handlers CloudStack would call:
|
||||
- `CreateUser`, `DeleteUser`, `ListUsers`, `GetUser`
|
||||
- `CreateAccessKey`, `DeleteAccessKey`, `ListAccessKeys`
|
||||
- `PutUserPolicy`, `GetUserPolicy`, `DeleteUserPolicy`
|
||||
- `AttachUserPolicy`, `ListAttachedUserPolicies`
|
||||
|
||||
This means the CloudStack plugin can manage SeaweedFS users with the **AWS IAM
|
||||
Java SDK** (`com.amazonaws.services.identitymanagement.AmazonIdentityManagement`),
|
||||
exactly the way the AWS IAM Go SDK is used in SeaweedFS's own tests. No proprietary
|
||||
admin client is needed. **Cloudian HyperStore already does exactly this** in the
|
||||
CloudStack tree — the SeaweedFS plugin follows the same pattern.
|
||||
|
||||
## Design
|
||||
|
||||
### Module layout
|
||||
|
||||
New CloudStack plugin module, mirroring `plugins/storage/object/cloudian/`
|
||||
(the closest precedent — same AWS S3 + IAM SDK approach):
|
||||
|
||||
```
|
||||
plugins/storage/object/seaweedfs/
|
||||
pom.xml
|
||||
src/main/java/org/apache/cloudstack/storage/datastore/
|
||||
driver/SeaweedFSObjectStoreDriverImpl.java
|
||||
lifecycle/SeaweedFSObjectStoreLifeCycleImpl.java
|
||||
provider/SeaweedFSObjectStoreProviderImpl.java
|
||||
util/SeaweedFSObjectStoreUtil.java
|
||||
src/test/java/org/apache/cloudstack/storage/datastore/
|
||||
driver/SeaweedFSObjectStoreDriverImplTest.java
|
||||
provider/SeaweedFSObjectStoreProviderImplTest.java
|
||||
src/main/resources/META-INF/cloudstack/storage-object-seaweedfs/
|
||||
module.properties
|
||||
spring-storage-object-seaweedfs-context.xml
|
||||
```
|
||||
|
||||
### `SeaweedFSObjectStoreProviderImpl`
|
||||
|
||||
Direct copy of `MinIOObjectStoreProviderImpl` with `providerName = "SeaweedFS"`,
|
||||
injecting the SeaweedFS lifecycle and driver. Registers via
|
||||
`storeMgr.registerDriver`.
|
||||
|
||||
### `SeaweedFSObjectStoreLifeCycleImpl`
|
||||
|
||||
Copy of `MinIOObjectStoreLifeCycleImpl`. `initialize()` reads `url`, `name`,
|
||||
`accesskey`, `secretkey` from the `addObjectStoragePool` details map, tests the
|
||||
connection by calling `AmazonS3.listBuckets()` against the SeaweedFS S3 endpoint,
|
||||
and persists the `ObjectStoreVO`. No proprietary client needed — the AWS S3 SDK
|
||||
is enough for the health check.
|
||||
|
||||
### `SeaweedFSObjectStoreDriverImpl`
|
||||
|
||||
The substantive class. Uses two AWS SDK v1 clients (same dependency Ceph already
|
||||
pulls in, so no new CloudStack dependency):
|
||||
|
||||
- `AmazonS3` for bucket operations (path-style, endpoint-pinned, `us-east-1`
|
||||
region placeholder — same as Ceph's `getS3Client`)
|
||||
- `AmazonIdentityManagement` for user/credential operations, pointed at the
|
||||
SeaweedFS IAM endpoint
|
||||
|
||||
#### Bucket operations — straightforward S3
|
||||
|
||||
| Interface method | Implementation |
|
||||
| --- | --- |
|
||||
| `createBucket` | `s3.createBucket(name)`; reject if `doesBucketExistV2`; persist access/secret key + URL on `BucketVO` (same as Ceph) |
|
||||
| `listBuckets` | `s3.listBuckets()` → wrap as `BucketObject` (same as Ceph) |
|
||||
| `deleteBucket` | `s3.deleteBucket(name)` (same as Ceph) |
|
||||
| `setBucketPolicy` | `s3.setBucketPolicy(...)` with the same public/private JSON the MinIO/Ceph drivers build |
|
||||
| `getBucketPolicy` / `deleteBucketPolicy` | `s3.getBucketPolicy` / `s3.deleteBucketPolicy` |
|
||||
| `setBucketVersioning` | `s3.setBucketVersioningConfiguration(Enabled)` |
|
||||
| `deleteBucketVersioning` | `s3.setBucketVersioningConfiguration(Suspended)` |
|
||||
| `setBucketEncryption` | `s3.setBucketEncryptionConfiguration(SSE-S3 rule)` |
|
||||
| `deleteBucketEncryption` | `s3.deleteBucketEncryptionConfiguration` |
|
||||
| `getBucketAcl` / `setBucketAcl` | no-op / null (same as MinIO and Ceph) |
|
||||
|
||||
#### User creation — the key difference
|
||||
|
||||
MinIO calls `MinioAdminClient.addUser`; Ceph calls `RgwAdmin.createUser`. SeaweedFS
|
||||
exposes the standard AWS IAM API, so the plugin calls:
|
||||
|
||||
```java
|
||||
AmazonIdentityManagement iam = getIamClient(storeId);
|
||||
String userName = "acs-" + account.getUuid();
|
||||
|
||||
// CreateUser (idempotent — check GetUser first, like Ceph does)
|
||||
iam.createUser(new CreateUserRequest(userName));
|
||||
|
||||
// CreateAccessKey → returns the access key + secret key to persist
|
||||
CreateAccessKeyResult result = iam.createAccessKey(
|
||||
new CreateAccessKeyRequest().withUserName(userName));
|
||||
AccessKey key = result.getAccessKey();
|
||||
|
||||
// Persist per-account, same pattern as Ceph's CEPH_ACCESS_KEY/CEPH_SECRET_KEY
|
||||
details.put(SEAWEEDFS_ACCESS_KEY, key.getAccessKeyId());
|
||||
details.put(SEAWEEDFS_SECRET_KEY, key.getSecretAccessKey());
|
||||
_accountDetailsDao.persist(accountId, details);
|
||||
```
|
||||
|
||||
This is the cleanest mapping of the three providers: no proprietary admin client,
|
||||
just the AWS IAM SDK that CloudStack already has access to. The IAM endpoint URL
|
||||
is provided as `iamUrl` in the store details. If `iamUrl` is omitted, the driver
|
||||
defaults it to `s3Url` — SeaweedFS registers its embedded IAM API at `POST /` on
|
||||
the same S3 endpoint (`UnifiedPostHandler` in `s3api_server.go`), so the IAM
|
||||
endpoint is the same as the S3 endpoint unless the deployment runs a separate
|
||||
`weed iam` server.
|
||||
|
||||
#### Bucket quota — S3 `?seaweedfs-quota` extension
|
||||
|
||||
This is the one genuine gap. MinIO and Ceph both have an admin API to set a
|
||||
per-bucket quota that the backend enforces. SeaweedFS enforces bucket quota
|
||||
server-side, but the configuration path was **not exposed over a standard S3 or
|
||||
IAM API** — it was only set via the admin REST API or shell commands.
|
||||
|
||||
The integration adds a **narrow S3 subresource** to SeaweedFS:
|
||||
- `PUT /{bucket}?seaweedfs-quota` — set bucket quota (IAM permission `s3:PutBucketQuota`)
|
||||
- `GET /{bucket}?seaweedfs-quota` — get bucket quota (IAM permission `s3:GetBucketQuota`)
|
||||
|
||||
This is implemented in SeaweedFS PR #11279. It uses SigV4 authentication and
|
||||
dedicated IAM permissions, so the CloudStack service credential can be scoped
|
||||
to quota management only — no global admin token, no cluster-wide admin access.
|
||||
The enforcement already exists (PR #10224); this PR only adds the HTTP
|
||||
configuration surface.
|
||||
|
||||
An earlier approach (PR #11278, closed) added bearer-token auth to the broad
|
||||
admin REST API. After review, that was unnecessary for this integration —
|
||||
static S3 config plus standard S3 APIs plus one scoped quota mutation API is
|
||||
sufficient and far safer.
|
||||
|
||||
> **Note on AWS tools compatibility.** `?seaweedfs-quota` is a SeaweedFS-specific
|
||||
> S3 subresource, not part of the AWS S3 API. Standard AWS tools (`aws s3api`,
|
||||
> `s3cmd`, `rclone`) cannot call it directly. This is the same limitation MinIO
|
||||
> and Ceph have — MinIO quota lives behind a separate admin API (`mc admin
|
||||
> bucket quota`), and Ceph quota lives behind the Admin Ops API
|
||||
> (`radosgw-admin quota set`). Neither is callable via `aws s3api` either.
|
||||
> SeaweedFS's approach is the closest to standard S3 because it uses the same
|
||||
> endpoint and same SigV4 credentials, just with a custom query parameter.
|
||||
> Interactive quota management remains available via `weed shell`; the S3
|
||||
> extension exists for programmatic integration (CloudStack) where the
|
||||
> integrator can sign SigV4 requests but cannot run shell commands.
|
||||
|
||||
#### Usage reporting
|
||||
|
||||
`getAllBucketsUsage` must return a `Map<String, Long>` of bucket name → size.
|
||||
MinIO uses `MinioAdminClient.getDataUsageInfo`; Ceph uses
|
||||
`RgwAdmin.listBucketInfo`. SeaweedFS has no admin rollup endpoint, so the MVP
|
||||
plugin computes it by listing buckets and summing object sizes via S3
|
||||
`ListObjectsV2` — expensive for large stores. Better options exist in
|
||||
SeaweedFS already:
|
||||
- **Prometheus metrics** (`bucket_size_bytes` gauge, refreshed every minute)
|
||||
- **SOSAPI `capacity.xml`** (reports capacity, available space, and usage
|
||||
through the S3 endpoint — note: the current "return zero on backend error"
|
||||
behavior should be validated before using it for billing)
|
||||
|
||||
For the MVP, `listBuckets` + per-bucket size via the S3 API is correct but slow;
|
||||
flag it as a known limitation. Operators should consume Prometheus or SOSAPI
|
||||
for production-scale usage reporting.
|
||||
|
||||
### Spring wiring
|
||||
|
||||
`spring-storage-object-seaweedfs-context.xml` registers the provider bean,
|
||||
identical to the MinIO one. `module.properties` sets
|
||||
`name=storage-object-seaweedfs`, `parent=storage`.
|
||||
|
||||
### `pom.xml`
|
||||
|
||||
Depends on `aws-java-sdk-s3` and `aws-java-sdk-iam` — both already in the
|
||||
CloudStack dependency tree (Ceph uses the S3 SDK; the IAM SDK is the standard AWS
|
||||
bundle). No new third-party dependency, unlike MinIO which pulls in the MinIO
|
||||
Java client.
|
||||
|
||||
## What changes on the SeaweedFS side
|
||||
|
||||
**One narrow S3 extension is required for quota management.** SeaweedFS PR #11279
|
||||
adds the `?seaweedfs-quota` S3 subresource:
|
||||
|
||||
- `PUT /{bucket}?seaweedfs-quota` — set bucket quota (IAM permission `s3:PutBucketQuota`)
|
||||
- `GET /{bucket}?seaweedfs-quota` — get bucket quota (IAM permission `s3:GetBucketQuota`)
|
||||
|
||||
This is authenticated via standard S3 SigV4 and authorized via dedicated IAM
|
||||
permissions, so no global admin token is needed. The enforcement already exists
|
||||
(PR #10224); this PR only adds the HTTP configuration surface.
|
||||
|
||||
One follow-up improvement on the SeaweedFS side would close the usage reporting
|
||||
gap:
|
||||
|
||||
1. **Validate SOSAPI `capacity.xml` usage calculation** — the current "return
|
||||
zero on backend error" behavior should be validated before using it for
|
||||
billing. If reliable, CloudStack can consume it directly instead of
|
||||
list-based aggregation.
|
||||
|
||||
## Open questions for proIO / Swen
|
||||
|
||||
1. **IAM endpoint path.** ~~Where does `weed iam` listen relative to the S3
|
||||
endpoint in a typical proIO deployment?~~ **Resolved.** SeaweedFS registers
|
||||
its embedded IAM API at `POST /` on the same S3 endpoint
|
||||
(`UnifiedPostHandler`), so the driver defaults `iamUrl` to `s3Url`. A
|
||||
separate `iamUrl` is only needed if the deployment runs a standalone
|
||||
`weed iam` server on a different host/port.
|
||||
2. **Quota requirements.** Do proIO's customers need server-enforced per-bucket
|
||||
quotas, or is CloudStack-side accounting sufficient for the first release?
|
||||
The `?seaweedfs-quota` S3 extension (PR #11279) provides server-enforced
|
||||
quotas via a scoped credential; this is the recommended path.
|
||||
3. **Object Lock.** `createBucket` takes an `objectLock` boolean. MinIO supports
|
||||
it; Ceph ignores it. SeaweedFS has Object Lock support. Should the plugin pass
|
||||
it through?
|
||||
4. **Contribution model.** Does proIO want to submit the PR to
|
||||
`apache/cloudstack` themselves (with SeaweedFS maintainers as reviewers), or
|
||||
the reverse? Apache CloudStack requires an ICLA for non-trivial contributions.
|
||||
|
||||
## Files
|
||||
|
||||
All in the `apache/cloudstack` repo (new module):
|
||||
|
||||
| File | Purpose |
|
||||
| --- | --- |
|
||||
| `plugins/storage/object/seaweedfs/pom.xml` | Maven module |
|
||||
| `.../datastore/util/SeaweedFSObjectStoreUtil.java` | S3 + IAM client builders, constants, URL validators |
|
||||
| `.../datastore/provider/SeaweedFSObjectStoreProviderImpl.java` | Spring provider registration |
|
||||
| `.../datastore/lifecycle/SeaweedFSObjectStoreLifeCycleImpl.java` | Pool add/health-check |
|
||||
| `.../datastore/driver/SeaweedFSObjectStoreDriverImpl.java` | Bucket + user ops via S3 + IAM SDK |
|
||||
| `.../resources/META-INF/cloudstack/storage-object-seaweedfs/module.properties` | Module name |
|
||||
| `.../resources/META-INF/cloudstack/storage-object-seaweedfs/spring-storage-object-seaweedfs-context.xml` | Spring bean |
|
||||
| `plugins/pom.xml` | Register `storage/object/seaweedfs` module |
|
||||
|
||||
No files in `seaweedfs/seaweedfs` for the MVP.
|
||||
|
||||
### SeaweedFS-side changes (PR #11279)
|
||||
|
||||
| File | Purpose |
|
||||
| --- | --- |
|
||||
| `weed/s3api/s3_constants/s3_action_strings.go` | Add `S3_ACTION_PUT_BUCKET_QUOTA` and `S3_ACTION_GET_BUCKET_QUOTA` |
|
||||
| `weed/s3api/s3_constants/s3_actions.go` | Add coarse-grained `ACTION_PUT_BUCKET_QUOTA` and `ACTION_GET_BUCKET_QUOTA` |
|
||||
| `weed/s3api/s3_action_resolver.go` | Map `seaweedfs-quota` query param to fine-grained s3: actions |
|
||||
| `weed/s3api/s3api_bucket_quota_handlers.go` | New — `PutBucketQuotaHandler` and `GetBucketQuotaHandler` |
|
||||
| `weed/s3api/s3api_bucket_quota_handlers_test.go` | New — tests for unit conversion, validation, and error paths |
|
||||
| `weed/s3api/s3api_server.go` | Register the two routes in the bucket subrouter |
|
||||
@@ -1,6 +1,6 @@
|
||||
module github.com/seaweedfs/seaweedfs
|
||||
|
||||
go 1.26
|
||||
go 1.26.0
|
||||
|
||||
require (
|
||||
cloud.google.com/go v0.123.0 // indirect
|
||||
@@ -25,7 +25,7 @@ require (
|
||||
github.com/facebookgo/subset v0.0.0-20200203212716-c811ad88dec4 // indirect
|
||||
github.com/fsnotify/fsnotify v1.9.0 // indirect
|
||||
github.com/go-redsync/redsync/v4 v4.17.0
|
||||
github.com/go-sql-driver/mysql v1.10.0
|
||||
github.com/go-sql-driver/mysql v1.10.1
|
||||
github.com/go-zookeeper/zk v1.0.4 // indirect
|
||||
github.com/golang/protobuf v1.5.4
|
||||
github.com/golang/snappy v1.0.0
|
||||
@@ -60,12 +60,12 @@ require (
|
||||
github.com/pquerna/cachecontrol v0.2.0
|
||||
github.com/prometheus/client_golang v1.24.1
|
||||
github.com/prometheus/client_model v0.6.3
|
||||
github.com/prometheus/common v0.70.1 // indirect
|
||||
github.com/prometheus/common v0.70.1
|
||||
github.com/prometheus/procfs v0.22.0
|
||||
github.com/rcrowley/go-metrics v0.0.0-20201227073835-cf1acfcdf475 // indirect
|
||||
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec // indirect
|
||||
github.com/seaweedfs/goexif v2.0.0+incompatible
|
||||
github.com/seaweedfs/raft v1.2.0
|
||||
github.com/seaweedfs/raft v1.2.1
|
||||
github.com/sirupsen/logrus v1.9.4 // indirect
|
||||
github.com/spf13/afero v1.15.0 // indirect
|
||||
github.com/spf13/cast v1.10.0 // indirect
|
||||
@@ -90,16 +90,16 @@ require (
|
||||
gocloud.dev v0.46.0
|
||||
gocloud.dev/pubsub/natspubsub v0.46.0
|
||||
gocloud.dev/pubsub/rabbitpubsub v0.46.0
|
||||
golang.org/x/crypto v0.55.0
|
||||
golang.org/x/crypto v0.56.0
|
||||
golang.org/x/exp v0.0.0-20260709172345-9ea1abe57597
|
||||
golang.org/x/image v0.45.0
|
||||
golang.org/x/image v0.46.0
|
||||
golang.org/x/net v0.58.0
|
||||
golang.org/x/oauth2 v0.36.0
|
||||
golang.org/x/sys v0.47.0
|
||||
golang.org/x/text v0.41.0 // indirect
|
||||
golang.org/x/tools v0.48.0 // indirect
|
||||
golang.org/x/sys v0.48.0
|
||||
golang.org/x/text v0.42.0 // indirect
|
||||
golang.org/x/tools v0.49.0 // indirect
|
||||
golang.org/x/xerrors v0.0.0-20240903120638-7835f813f4da // indirect
|
||||
google.golang.org/api v0.296.0
|
||||
google.golang.org/api v0.297.0
|
||||
google.golang.org/genproto v0.0.0-20260715232425-e75dac1f907d // indirect
|
||||
google.golang.org/grpc v1.85.0-dev
|
||||
google.golang.org/protobuf v1.36.12
|
||||
@@ -122,10 +122,10 @@ require (
|
||||
github.com/apple/foundationdb/bindings/go v0.0.0-20250911184653-27f7192f47c3
|
||||
github.com/arangodb/go-driver v1.6.9
|
||||
github.com/armon/go-metrics v0.4.1
|
||||
github.com/aws/aws-sdk-go-v2 v1.45.1
|
||||
github.com/aws/aws-sdk-go-v2 v1.47.0
|
||||
github.com/aws/aws-sdk-go-v2/config v1.32.35
|
||||
github.com/aws/aws-sdk-go-v2/credentials v1.20.1
|
||||
github.com/aws/aws-sdk-go-v2/service/s3 v1.107.0
|
||||
github.com/aws/aws-sdk-go-v2/credentials v1.20.4
|
||||
github.com/aws/aws-sdk-go-v2/service/s3 v1.107.3
|
||||
github.com/cespare/xxhash/v2 v2.3.0
|
||||
github.com/cognusion/imaging v1.0.4
|
||||
github.com/fluent/fluent-logger-golang v1.10.1
|
||||
@@ -144,9 +144,9 @@ require (
|
||||
github.com/parquet-go/parquet-go v0.32.0
|
||||
github.com/pkg/sftp v1.13.11
|
||||
github.com/rabbitmq/amqp091-go v1.14.0
|
||||
github.com/rclone/rclone v1.75.0
|
||||
github.com/rclone/rclone v1.75.1
|
||||
github.com/rdleal/intervalst v1.5.0
|
||||
github.com/redis/go-redis/v9 v9.21.0
|
||||
github.com/redis/go-redis/v9 v9.22.0
|
||||
github.com/schollz/progressbar/v3 v3.19.1
|
||||
github.com/seaweedfs/go-fuse/v2 v2.9.4
|
||||
github.com/shirou/gopsutil/v4 v4.26.7
|
||||
@@ -160,7 +160,7 @@ require (
|
||||
github.com/ydb-platform/ydb-go-sdk/v3 v3.151.1
|
||||
go.etcd.io/etcd/client/pkg/v3 v3.7.1
|
||||
go.uber.org/atomic v1.11.0
|
||||
golang.org/x/sync v0.22.0
|
||||
golang.org/x/sync v0.23.0
|
||||
golang.org/x/tools/godoc v0.1.0-deprecated
|
||||
google.golang.org/grpc/security/advancedtls v1.0.0
|
||||
)
|
||||
@@ -185,7 +185,7 @@ require (
|
||||
github.com/antlr4-go/antlr/v4 v4.13.1 // indirect
|
||||
github.com/apache/arrow-go/v18 v18.7.0 // indirect
|
||||
github.com/apache/thrift v0.24.0 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/signin v1.7.1 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/signin v1.10.0 // indirect
|
||||
github.com/bahlo/generic-list-go v0.2.0 // indirect
|
||||
github.com/bazelbuild/rules_go v0.46.0 // indirect
|
||||
github.com/biogo/store v0.0.0-20201120204734-aad293a2328f // indirect
|
||||
@@ -262,8 +262,8 @@ require (
|
||||
github.com/pquerna/otp v1.5.0 // indirect
|
||||
github.com/pterm/pterm v0.12.83 // indirect
|
||||
github.com/quic-go/qpack v0.6.0 // indirect
|
||||
github.com/rclone/Proton-API-Bridge v1.0.4 // indirect
|
||||
github.com/rclone/go-proton-api v1.0.3 // indirect
|
||||
github.com/rclone/Proton-API-Bridge v1.0.5 // indirect
|
||||
github.com/rclone/go-proton-api v1.0.4 // indirect
|
||||
github.com/rogpeppe/go-internal v1.15.0 // indirect
|
||||
github.com/rwcarlsen/goexif v0.0.0-20190401172101-9e8deecbddbd // indirect
|
||||
github.com/ryanuber/go-glob v1.0.0 // indirect
|
||||
@@ -290,7 +290,7 @@ require (
|
||||
go.uber.org/mock v0.5.2 // indirect
|
||||
go.yaml.in/yaml/v2 v2.4.4 // indirect
|
||||
go.yaml.in/yaml/v3 v3.0.4 // indirect
|
||||
golang.org/x/mod v0.38.0 // indirect
|
||||
golang.org/x/mod v0.41.0 // indirect
|
||||
gonum.org/v1/gonum v0.17.0 // indirect
|
||||
)
|
||||
|
||||
@@ -326,21 +326,21 @@ require (
|
||||
github.com/andybalholm/cascadia v1.3.4 // indirect
|
||||
github.com/appscode/go-querystring v0.0.0-20170504095604-0126cfb3f1dc // indirect
|
||||
github.com/arangodb/go-velocypack v0.0.0-20200318135517-5af53c29c67e // indirect
|
||||
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.16 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.19.1 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.18 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.20.0 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/feature/s3/manager v1.22.34 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.1 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.1 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.1 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.3 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.3 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.3 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.19 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.28 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.1 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.36 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.31 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.3 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.39 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/sns v1.39.14 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/sqs v1.42.24 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/sso v1.35.1 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/ssooidc v1.40.1 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/sts v1.47.1
|
||||
github.com/aws/aws-sdk-go-v2/service/sso v1.38.0 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/ssooidc v1.43.0 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/sts v1.50.0
|
||||
github.com/aws/smithy-go v1.28.1
|
||||
github.com/boltdb/bolt v1.3.1 // indirect
|
||||
github.com/bradenaw/juniper v0.15.3 // indirect
|
||||
|
||||
@@ -710,48 +710,48 @@ github.com/armon/go-metrics v0.4.1/go.mod h1:E6amYzXo6aW1tqzoZGT755KkbgrJsSdpwZ+
|
||||
github.com/atomicgo/cursor v0.0.1/go.mod h1:cBON2QmmrysudxNBFthvMtN32r3jxVRIvzkUiF/RuIk=
|
||||
github.com/aws/aws-sdk-go v1.55.8 h1:JRmEUbU52aJQZ2AjX4q4Wu7t4uZjOu71uyNmaWlUkJQ=
|
||||
github.com/aws/aws-sdk-go v1.55.8/go.mod h1:ZkViS9AqA6otK+JBBNH2++sx1sgxrPKcSzPPvQkUtXk=
|
||||
github.com/aws/aws-sdk-go-v2 v1.45.1 h1:iIoG3NaLhV6UZpPXyPXlDj2I9oS8tV/nMcMnITCC6Ks=
|
||||
github.com/aws/aws-sdk-go-v2 v1.45.1/go.mod h1:bttEH6JqnUL8LepvDVfdrds/fZ5bCIxzpe3abyUrhDU=
|
||||
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.16 h1:aiuaKlDweRC5qExJondpWjOgyzMHpofpwspGXUtwn4c=
|
||||
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.16/go.mod h1:nG/LOlmox9BDe9HvQnXWzgcK8uKbgBMZ/Hp5pVt/21I=
|
||||
github.com/aws/aws-sdk-go-v2 v1.47.0 h1:0jsHallhJCeaU0Ko48c/3FK1ctOQ7NpzggxriJOQ8MQ=
|
||||
github.com/aws/aws-sdk-go-v2 v1.47.0/go.mod h1:bttEH6JqnUL8LepvDVfdrds/fZ5bCIxzpe3abyUrhDU=
|
||||
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.18 h1:LAfOuhAH331fmOjTQpAaOlH+Ftn7RzSDJ2VFwjdMMy4=
|
||||
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.18/go.mod h1:4e5xhuXHx1e4U9EthvbPP1r/DIMp5c2823OL8karzcM=
|
||||
github.com/aws/aws-sdk-go-v2/config v1.32.35 h1:UEzXuET8E42lxBPijuACu/tEK7v5lFPlk0Q+GT5WD9E=
|
||||
github.com/aws/aws-sdk-go-v2/config v1.32.35/go.mod h1:KaMtJpFa2JlL2BStjjHQVwQpzZEmw+ND/EgVrfFoo2g=
|
||||
github.com/aws/aws-sdk-go-v2/credentials v1.20.1 h1:Z8GRNEx0u9sDkZOq4PUnN8mjGwbUQGRzMSXpvt3d8xQ=
|
||||
github.com/aws/aws-sdk-go-v2/credentials v1.20.1/go.mod h1:uBIK00kFo95dnemqfFMTWx0X8YRqsh6ecIoCjjOkZqM=
|
||||
github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.19.1 h1:YIEBqcqRnpi4Pfv0YHImtgi6czGCwKHANC7SwmUAVD0=
|
||||
github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.19.1/go.mod h1:imEf0oufgAo8KAkCHhrOdqGEC0YWx1PPBQH82shSxGw=
|
||||
github.com/aws/aws-sdk-go-v2/credentials v1.20.4 h1:hTvrJJseKbvw32kmiE0G+u/9ZqpqscjDrTigHIXP2qs=
|
||||
github.com/aws/aws-sdk-go-v2/credentials v1.20.4/go.mod h1:gWp9O1ZBWwpcIrgV+mVHk4gZUurAEDkgypu/OXOlIaw=
|
||||
github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.20.0 h1:AM4hHjww+PSFtt6E+UrBrPlZkWsePCLEt9AjkfQX+yM=
|
||||
github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.20.0/go.mod h1:3x/yXezeQjpOvBb4jEMxrS8SXvpdvJ5abv6l5c1gWM8=
|
||||
github.com/aws/aws-sdk-go-v2/feature/s3/manager v1.22.34 h1:Pn7OsMwBLbkZ6OnCxWHAjf0L/22H8cnhxZC0uPwtMtg=
|
||||
github.com/aws/aws-sdk-go-v2/feature/s3/manager v1.22.34/go.mod h1:eToXR/Gk1uqpn04eSmdgVXwfS0WvH8aG4eBFr8ygbpU=
|
||||
github.com/aws/aws-sdk-go-v2/feature/s3/transfermanager v0.3.11 h1:eBXB8KZgzQ8A9QB4iJS4aw/u6+4OY3i2hQXPABeAIOg=
|
||||
github.com/aws/aws-sdk-go-v2/feature/s3/transfermanager v0.3.11/go.mod h1:N9+5pG27Fy61GUL5YXVLXDTLmUudMrgwsuDbgBMNLxQ=
|
||||
github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.1 h1:pc138gM1CW+XPc60rEwUlwwuwWFQK16CI1T7v1F9Oec=
|
||||
github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.1/go.mod h1:1+koxpPIbfBdfzP6vojm5/zTpTQ/micYwlxIiNB3TxI=
|
||||
github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.1 h1:K0JsbZQj+1h208Ro1zHeA4l7bMp0NvRffHQ91q8Ol1s=
|
||||
github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.1/go.mod h1:W3/vL6EtCIatICGy9ab29QhMuae+cOKPWcMxv02CO+Q=
|
||||
github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.1 h1:yhw5KD1phVyP9vijxOUzDfEtJx+bt+L63k+VfuiYFAA=
|
||||
github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.1/go.mod h1:ZW2e0d7DYlRxlS9hEiMXE47gTdX5KRN4byUiNbUpG+Q=
|
||||
github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.3 h1:Hp/VgjP0BysR3OgLlR057Vz2LcbbVnoWeJ+3qWiS/fY=
|
||||
github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.3/go.mod h1:nwGV5qw7F1IZPgxCvA/ph8N2TAuz+BkRG/bXn808qMA=
|
||||
github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.3 h1:MUaM4f+kj1ZIBPZfUS8cxP1GKXXZtHJjAthy93AN7SM=
|
||||
github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.3/go.mod h1:6YmVmEVRI5ZZzRjCSsb9SryKH0hAlMRdgA7kG9aDvBU=
|
||||
github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.3 h1:fuSCw4Z2qfRCztMPO3GXJNSiEp6Wee+WOLwrHHUMy9c=
|
||||
github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.3/go.mod h1:6SxcHheD1pPR5+kWm1wGvjlL/YqUsh267sAfEmN4K7A=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.19 h1:bAdDl/HkGCcGPoe25ToSHEw23VIxt6CT5fLcg111BKg=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.19/go.mod h1:KaUzbLxv4CeSxh6ZCl9B4m7CuFenS8kUEaDs+f/DQr4=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.28 h1:Q1TF1J9jVD+vFo0LzNnmNdQ9EAt52TS+MQlq9Ir+Yxo=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.28/go.mod h1:4KqXXC/p1hrotmouDFbrRoWaLy962b9PMUReCG6+uWo=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.1 h1:RmmWQPREQdk9U+PfqeHW3MqZaBaNK7TpV9W3RY+b+7g=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.1/go.mod h1:0A3W4F+68ZnNk5XcNL/e9HFMwnP8RlEicFfy6eOEDyw=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.36 h1:EUIwBoN+q7UmhAejxgD27APiRjh1vwCFo53gSqdT0BM=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.36/go.mod h1:6u00gmlTGR6W0b2k9NBrld7MnOEmf1Spqx0VVt6AqyE=
|
||||
github.com/aws/aws-sdk-go-v2/service/s3 v1.107.0 h1:OkYV+1171za+ab9otU1tGxMXhx6uZvwVEtVddjLuYTg=
|
||||
github.com/aws/aws-sdk-go-v2/service/s3 v1.107.0/go.mod h1:5FTZoQxhmLEiCAtYVk6V+t0iS/B5yGZVLZ3Wq5FDJZI=
|
||||
github.com/aws/aws-sdk-go-v2/service/signin v1.7.1 h1:mdMtSVKdQ3+mzBh+l0ogrFYZVQUCg6pJZOirA2ARsYE=
|
||||
github.com/aws/aws-sdk-go-v2/service/signin v1.7.1/go.mod h1:9IqUlsJDbUPcg6cgx3WEzXdjrbWzLDQrak0aaSqlTcI=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.31 h1:uZOinZb+h7lZw8IYzP1z1IuEnueB76/EFkcf/fEW4Ag=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.31/go.mod h1:NRtwAM/p5VRt03TlEUs0pH3TeWamWdf4YyJpSrzPYLc=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.3 h1:bON1rJf67TSTDCKg816AAIE4xSTtoo9tl0XRkO72R+I=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.3/go.mod h1:c5BBpjJcQXpfeq9iASyVKA3T6vX6B6LEXY4mL/gklDY=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.39 h1:HLPAVrlLDaN2boN0xJx7MgaQDNEO3Q+c9L6kl/8m47Q=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.39/go.mod h1:Pg/dVfsNkm1hsIDK/gMvCKtmyNfNTV12mrgHqVE/6Oo=
|
||||
github.com/aws/aws-sdk-go-v2/service/s3 v1.107.3 h1:IKoCZqfWfZzSBi16QFQ+QcbQ3LRQ7QgB1S5tDAyPBQQ=
|
||||
github.com/aws/aws-sdk-go-v2/service/s3 v1.107.3/go.mod h1:RBpRcXiM4s2pOInVs32GsBonnje+fiAj4mcrStRmlCA=
|
||||
github.com/aws/aws-sdk-go-v2/service/signin v1.10.0 h1:ZD5qFpWcaOKdTuhBi431pIDkCgrMkMlMT6jlpSPoIRI=
|
||||
github.com/aws/aws-sdk-go-v2/service/signin v1.10.0/go.mod h1:8Nuuf+tR346PjJ3MvZPh9pekbLiLQFWJhzMXfwy7alA=
|
||||
github.com/aws/aws-sdk-go-v2/service/sns v1.39.14 h1:p8WdWDh5AwSZdp19Haa3XMyPCICi9Z375a/Nu3IIEZY=
|
||||
github.com/aws/aws-sdk-go-v2/service/sns v1.39.14/go.mod h1:NKVY7DER6VXHkt2I/ycmHakALNboi3Rqwt4eEf/1Cnk=
|
||||
github.com/aws/aws-sdk-go-v2/service/sqs v1.42.24 h1:JP2wjWGmUp8lTCZb13Dv0Eciyc1jbO8pd0HZVMHFlrc=
|
||||
github.com/aws/aws-sdk-go-v2/service/sqs v1.42.24/go.mod h1:Ql9ziDutk8ERAN9HMaYANCW3lop451ppebkxEJMLCTM=
|
||||
github.com/aws/aws-sdk-go-v2/service/sso v1.35.1 h1:B6WFn91tobD6gG4724ONHaqrpKsoETGnv98LHe/yIGM=
|
||||
github.com/aws/aws-sdk-go-v2/service/sso v1.35.1/go.mod h1:tWuiVBUtPBr8/rgRiYS8Uf85sHcAN+G7XS3D3CEoUh8=
|
||||
github.com/aws/aws-sdk-go-v2/service/ssooidc v1.40.1 h1:6yeYCWFvgbI2TI3K6jr9LtBNhXgJ7g4xqD+DEiaDDmM=
|
||||
github.com/aws/aws-sdk-go-v2/service/ssooidc v1.40.1/go.mod h1:naFe83jSMuYkH+QjQPX8n1MLhBkeCFM5Lsnh5m5wz3c=
|
||||
github.com/aws/aws-sdk-go-v2/service/sts v1.47.1 h1:Sv2xPnRHlThSUtVujYuUBPI/Il8si6UPHXL8DMiB/F0=
|
||||
github.com/aws/aws-sdk-go-v2/service/sts v1.47.1/go.mod h1:mKo/CzaCz8qytGW70NG4vIIGAx1HXTlb5lHNkC5k3lk=
|
||||
github.com/aws/aws-sdk-go-v2/service/sso v1.38.0 h1:JGeeBcMlhg1xtOXYpeCaTQBZObtXMPQCUqBcmr65NRA=
|
||||
github.com/aws/aws-sdk-go-v2/service/sso v1.38.0/go.mod h1:XwteswG9EOMRFm73UT0t+MbTwyLxMrEXkU6e+v92Lzo=
|
||||
github.com/aws/aws-sdk-go-v2/service/ssooidc v1.43.0 h1:obhahQXDEdVEv8y5bTKXR30LVaxYe1kyYM0L7l2Iq+k=
|
||||
github.com/aws/aws-sdk-go-v2/service/ssooidc v1.43.0/go.mod h1:6twZZ/aXHNy1vXUO8koUbp++MYzMASkOgEBdkbJYmO0=
|
||||
github.com/aws/aws-sdk-go-v2/service/sts v1.50.0 h1:khXV3+K5D3f4e8xtplaRdSFn1bEg3gj5EBHQvbCOZbQ=
|
||||
github.com/aws/aws-sdk-go-v2/service/sts v1.50.0/go.mod h1:/8JRcdTt//hG0Q4BTmGbuOplT7ABe+5rdtqUHqXvYIM=
|
||||
github.com/aws/smithy-go v1.28.1 h1:R/nXH00c8qcfCzQVELtRw+eLQWtzv+VAIEFJ1/xxXlQ=
|
||||
github.com/aws/smithy-go v1.28.1/go.mod h1:YE2RhdIuDbA5E5bTdciG9KrW3+TiEONeUWCqxX9i1Fc=
|
||||
github.com/bahlo/generic-list-go v0.2.0 h1:5sz/EEAK+ls5wF+NeqDpk5+iNdMDXrh3z3nPnH1Wvgk=
|
||||
@@ -1109,8 +1109,8 @@ github.com/go-redsync/redsync/v4 v4.17.0 h1:FFJ+uxZs44y4Sq10//IFKic9T94AYl+u3Sog
|
||||
github.com/go-redsync/redsync/v4 v4.17.0/go.mod h1:CKVA6qwT07S/916i+Yd9h1/8YFQhCCpPYTQhvvYytJo=
|
||||
github.com/go-resty/resty/v2 v2.17.2 h1:FQW5oHYcIlkCNrMD2lloGScxcHJ0gkjshV3qcQAyHQk=
|
||||
github.com/go-resty/resty/v2 v2.17.2/go.mod h1:kCKZ3wWmwJaNc7S29BRtUhJwy7iqmn+2mLtQrOyQlVA=
|
||||
github.com/go-sql-driver/mysql v1.10.0 h1:Q+1LV8DkHJvSYAdR83XzuhDaTykuDx0l6fkXxoWCWfw=
|
||||
github.com/go-sql-driver/mysql v1.10.0/go.mod h1:M+cqaI7+xxXGG9swrdeUIoPG3Y3KCkF0pZej+SK+nWk=
|
||||
github.com/go-sql-driver/mysql v1.10.1 h1:arlSnNLq6a5yxGxV7qg9lF4j0C+KwD6NbQyKr9QL6ME=
|
||||
github.com/go-sql-driver/mysql v1.10.1/go.mod h1:M+cqaI7+xxXGG9swrdeUIoPG3Y3KCkF0pZej+SK+nWk=
|
||||
github.com/go-stack/stack v1.8.0/go.mod h1:v0f6uXyyMGvRgIKkXu+yp6POWl0qKG85gN/melR3HDY=
|
||||
github.com/go-task/slim-sprig v0.0.0-20230315185526-52ccab3ef572 h1:tfuBGBXKqDEevZMzYi5KSi8KkcZtzBcTgAUUtapy0OI=
|
||||
github.com/go-task/slim-sprig/v3 v3.0.0 h1:sUs3vkvUymDpBKi3qH1YSqBQk9+9D/8M2mN1vB6EwHI=
|
||||
@@ -1760,18 +1760,18 @@ github.com/quic-go/quic-go v0.59.0 h1:OLJkp1Mlm/aS7dpKgTc6cnpynnD2Xg7C1pwL6vy/SA
|
||||
github.com/quic-go/quic-go v0.59.0/go.mod h1:upnsH4Ju1YkqpLXC305eW3yDZ4NfnNbmQRCMWS58IKU=
|
||||
github.com/rabbitmq/amqp091-go v1.14.0 h1:RSaT7aOKt/OrkVUyswPDW29lnRz9psuGmfZFBmLqLek=
|
||||
github.com/rabbitmq/amqp091-go v1.14.0/go.mod h1:Hy4jKW5kQART1u+JkDTF9YYOQUHXqMuhrgxOEeS7G4o=
|
||||
github.com/rclone/Proton-API-Bridge v1.0.4 h1:uGQJRjQC1hVLd5kqLsXc6CWO6oqrVeLoKQYoHapEZDg=
|
||||
github.com/rclone/Proton-API-Bridge v1.0.4/go.mod h1:VTPBYZotKAeDLlAzxU2O/s14NXk9FxUt9hn1jhH2iY8=
|
||||
github.com/rclone/go-proton-api v1.0.3 h1:3gBTzR+j0dYiTwtj9yKIdN/aV3W2a8KIPKp0GArojyQ=
|
||||
github.com/rclone/go-proton-api v1.0.3/go.mod h1:QAlkFfswzrBuxvCORWV8rZdddg52hahMN98CFWoFW1E=
|
||||
github.com/rclone/rclone v1.75.0 h1:3ARHem4jXWltvl+b0PvDAG8s6J/inHd5BRzfwMRb3W8=
|
||||
github.com/rclone/rclone v1.75.0/go.mod h1:PGLJUW/WSIJCysALqUcxmaCFyfMXUevf8CbuoOwsAdU=
|
||||
github.com/rclone/Proton-API-Bridge v1.0.5 h1:K1++Qtk3PvgkiCCiv6Pahju1TMOzKY6VSwiwT7XLAVc=
|
||||
github.com/rclone/Proton-API-Bridge v1.0.5/go.mod h1:vCeOPhlXzevN0AFojgh1zsjhetiShy/ArvJ/xkFUDWk=
|
||||
github.com/rclone/go-proton-api v1.0.4 h1:AJW0e9pB4j0hVK4WqyGErFwaI+5MUQWPCtj5FYYxtPg=
|
||||
github.com/rclone/go-proton-api v1.0.4/go.mod h1:QAlkFfswzrBuxvCORWV8rZdddg52hahMN98CFWoFW1E=
|
||||
github.com/rclone/rclone v1.75.1 h1:kIxQcoDLj2Gke/gMSHK7OnxhX1Gu1cJBLP1kJZoaFp0=
|
||||
github.com/rclone/rclone v1.75.1/go.mod h1:4zmMjGatCkSJPRZDpo+7y3xOl8S29EMUyKvZop5mHr4=
|
||||
github.com/rcrowley/go-metrics v0.0.0-20201227073835-cf1acfcdf475 h1:N/ElC8H3+5XpJzTSTfLsJV/mx9Q9g7kxmchpfZyxgzM=
|
||||
github.com/rcrowley/go-metrics v0.0.0-20201227073835-cf1acfcdf475/go.mod h1:bCqnVzQkZxMG4s8nGwiZ5l3QUCyqpo9Y+/ZMZ9VjZe4=
|
||||
github.com/rdleal/intervalst v1.5.0 h1:SEB9bCFz5IqD1yhfH1Wv8IBnY/JQxDplwkxHjT6hamU=
|
||||
github.com/rdleal/intervalst v1.5.0/go.mod h1:xO89Z6BC+LQDH+IPQQw/OESt5UADgFD41tYMUINGpxQ=
|
||||
github.com/redis/go-redis/v9 v9.21.0 h1:FPBE4hhbAke+TLmcY3WkpbDffJEomdqPn3HYiqAtL9E=
|
||||
github.com/redis/go-redis/v9 v9.21.0/go.mod h1:v/M13XI1PVCDcm01VtPFOADfZtHf8YW3baQf57KlIkA=
|
||||
github.com/redis/go-redis/v9 v9.22.0 h1:laDvpYXTJtZLloinw1fA5Kqd6HAEH2XKxOkG/PDq2F0=
|
||||
github.com/redis/go-redis/v9 v9.22.0/go.mod h1:y2g0Wj8rQvuK0ELM+oxSudcLtC09JScs98I/X9gRWY4=
|
||||
github.com/redis/rueidis v1.0.76 h1:RdDWuvlYBSp+bTrBvaXqJnNEL3VVzsnjo+0psPFgLc4=
|
||||
github.com/redis/rueidis v1.0.76/go.mod h1:UsfHPSbomB6QAVMk4iiFkzRy0nh9o7scDGa+SitvBY4=
|
||||
github.com/redis/rueidis/rueidiscompat v1.0.76 h1:7LikbiqCQqCsZXeZ+akgZMnjIV/J0VHih9PIX4gGZC4=
|
||||
@@ -1821,8 +1821,8 @@ github.com/seaweedfs/go-fuse/v2 v2.9.4 h1:ACyloiuopdhRSjdLLeSWbsVaemMPskORaRF01T
|
||||
github.com/seaweedfs/go-fuse/v2 v2.9.4/go.mod h1:zABdmWEa6A0bwaBeEOBUeUkGIZlxUhcdv+V1Dcc/U/I=
|
||||
github.com/seaweedfs/goexif v2.0.0+incompatible h1:x8pckiT12QQhifwhDQpeISgDfsqmQ6VR4LFPQ64JRps=
|
||||
github.com/seaweedfs/goexif v2.0.0+incompatible/go.mod h1:Oni780Z236sXpIQzk1XoJlTwqrJ02smEin9zQeff7Fk=
|
||||
github.com/seaweedfs/raft v1.2.0 h1:Ez4Hw9ifBbTT7wg54DvGHBjw1vRlTb4roH0TKl0Oj9Y=
|
||||
github.com/seaweedfs/raft v1.2.0/go.mod h1:fgs/rAVEzjQ7e04XMzG3eJhwZZRmBW+2uRtjakeCGeU=
|
||||
github.com/seaweedfs/raft v1.2.1 h1:QgFl/aaPnagpUxYB6Bx+fFss1NyetVVcJmraMmnaQ5Q=
|
||||
github.com/seaweedfs/raft v1.2.1/go.mod h1:fgs/rAVEzjQ7e04XMzG3eJhwZZRmBW+2uRtjakeCGeU=
|
||||
github.com/secure-systems-lab/go-securesystemslib v0.11.0 h1:iuCR9kcMFD4QurdKrGvPLoKZLv9YvwPYVr0473BdtFs=
|
||||
github.com/secure-systems-lab/go-securesystemslib v0.11.0/go.mod h1:+PMOTjUGwHj2vcZ+TFKlb1tXRbrdWE1LYDT5i9JC80Q=
|
||||
github.com/sergi/go-diff v1.0.0/go.mod h1:0CfEIISq7TuYL3j771MWULgwwjU+GofnZX9QAmXWZgo=
|
||||
@@ -2193,8 +2193,8 @@ golang.org/x/crypto v0.6.0/go.mod h1:OFC/31mSvZgRz0V1QTNCzfAI1aIRzbiufJtkMIlEp58
|
||||
golang.org/x/crypto v0.7.0/go.mod h1:pYwdfH91IfpZVANVyUOhSIPZaFoJGxTFbZhFTx+dXZU=
|
||||
golang.org/x/crypto v0.13.0/go.mod h1:y6Z2r+Rw4iayiXXAIxJIDAJ1zMW4yaTpebo8fPOliYc=
|
||||
golang.org/x/crypto v0.14.0/go.mod h1:MVFd36DqK4CsrnJYDkBA3VC4m2GkXAM0PvzMCn4JQf4=
|
||||
golang.org/x/crypto v0.55.0 h1:+KWHjbgOaAQ66dh/YlkZKHlz9ZUlq61AFirAR9ntP8M=
|
||||
golang.org/x/crypto v0.55.0/go.mod h1:uq0V9dE/fzQuJtbnL+2EhWOE63vo164FY8xqEnV9xis=
|
||||
golang.org/x/crypto v0.56.0 h1:GUh5Ii4J5jtcseSMiRqr1jXCNHoxjeV9Fmekc2oLy6Y=
|
||||
golang.org/x/crypto v0.56.0/go.mod h1:OMW5y6CY9l38uPLmxU6l6pwcXp1obtLo3e6gT7gQR2I=
|
||||
golang.org/x/exp v0.0.0-20180321215751-8460e604b9de/go.mod h1:CJ0aWSM057203Lf6IL+f9T1iT9GByDxfZKAQTCR3kQA=
|
||||
golang.org/x/exp v0.0.0-20180807140117-3d87b88a115f/go.mod h1:CJ0aWSM057203Lf6IL+f9T1iT9GByDxfZKAQTCR3kQA=
|
||||
golang.org/x/exp v0.0.0-20190121172915-509febef88a4/go.mod h1:CJ0aWSM057203Lf6IL+f9T1iT9GByDxfZKAQTCR3kQA=
|
||||
@@ -2225,8 +2225,8 @@ golang.org/x/image v0.0.0-20210607152325-775e3b0c77b9/go.mod h1:023OzeP/+EPmXeap
|
||||
golang.org/x/image v0.0.0-20210628002857-a66eb6448b8d/go.mod h1:023OzeP/+EPmXeapQh35lcL3II3LrY8Ic+EFFKVhULM=
|
||||
golang.org/x/image v0.0.0-20211028202545-6944b10bf410/go.mod h1:023OzeP/+EPmXeapQh35lcL3II3LrY8Ic+EFFKVhULM=
|
||||
golang.org/x/image v0.0.0-20220302094943-723b81ca9867/go.mod h1:023OzeP/+EPmXeapQh35lcL3II3LrY8Ic+EFFKVhULM=
|
||||
golang.org/x/image v0.45.0 h1:FMb1nTbH5H9vF55SriQHgFw5GnNL9Jg6L25BwXKzhB0=
|
||||
golang.org/x/image v0.45.0/go.mod h1:n62x/7RqlwXDvGsSU4u6IUTUf6KghUZ9Bt7cG/T9Fx4=
|
||||
golang.org/x/image v0.46.0 h1:b1+oYj0Jbp6K5MDT4i4/eZpYlk3V8SJhhDKh6LBHAyQ=
|
||||
golang.org/x/image v0.46.0/go.mod h1:3B3W05VGVQyuXucLINLjXKrqISASfi4Xj+iCVkLMwew=
|
||||
golang.org/x/lint v0.0.0-20181026193005-c67002cb31c3/go.mod h1:UVdnD1Gm6xHRNCYTkRU2/jEulfH38KcIWyp/GAMgvoE=
|
||||
golang.org/x/lint v0.0.0-20190227174305-5b3e6a55c961/go.mod h1:wehouNa3lNwaWXcvxsM5YxQ5yQlVC4a0KAMCusXpPoU=
|
||||
golang.org/x/lint v0.0.0-20190301231843-5614ed5bae6f/go.mod h1:UVdnD1Gm6xHRNCYTkRU2/jEulfH38KcIWyp/GAMgvoE=
|
||||
@@ -2258,8 +2258,8 @@ golang.org/x/mod v0.8.0/go.mod h1:iBbtSCu2XBx23ZKBPSOrRkjjQPZFPuis4dIYUhu/chs=
|
||||
golang.org/x/mod v0.9.0/go.mod h1:iBbtSCu2XBx23ZKBPSOrRkjjQPZFPuis4dIYUhu/chs=
|
||||
golang.org/x/mod v0.12.0/go.mod h1:iBbtSCu2XBx23ZKBPSOrRkjjQPZFPuis4dIYUhu/chs=
|
||||
golang.org/x/mod v0.13.0/go.mod h1:hTbmBsO62+eylJbnUtE2MGJUyE7QWk4xUqPFrRgJ+7c=
|
||||
golang.org/x/mod v0.38.0 h1:MECBjubtXD7yj4HrhIUcywNaGeNVUdfVnxmPajOk4yk=
|
||||
golang.org/x/mod v0.38.0/go.mod h1:V6Xz0pq8TQ3dGqVQ1FVHuelZpAL0uNhSkk9ogYP3c40=
|
||||
golang.org/x/mod v0.41.0 h1:qJmnOUb4YB+FsEuM3HcWucdZASCPGhsX6uljO6pog0c=
|
||||
golang.org/x/mod v0.41.0/go.mod h1:Ek9pY8RKWXwsWvd3rQiHYtMqkjSUV+s1Rj7j4H5Ur6o=
|
||||
golang.org/x/net v0.0.0-20180724234803-3673e40ba225/go.mod h1:mL1N/T3taQHkDXs73rZJwtUhF3w3ftmwwsq0BUmARs4=
|
||||
golang.org/x/net v0.0.0-20180826012351-8a410e7b638d/go.mod h1:mL1N/T3taQHkDXs73rZJwtUhF3w3ftmwwsq0BUmARs4=
|
||||
golang.org/x/net v0.0.0-20180906233101-161cd47e91fd/go.mod h1:mL1N/T3taQHkDXs73rZJwtUhF3w3ftmwwsq0BUmARs4=
|
||||
@@ -2374,8 +2374,8 @@ golang.org/x/sync v0.0.0-20220929204114-8fcdb60fdcc0/go.mod h1:RxMgew5VJxzue5/jJ
|
||||
golang.org/x/sync v0.1.0/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
|
||||
golang.org/x/sync v0.3.0/go.mod h1:FU7BRWz2tNW+3quACPkgCx/L+uEAv1htQ0V83Z9Rj+Y=
|
||||
golang.org/x/sync v0.4.0/go.mod h1:FU7BRWz2tNW+3quACPkgCx/L+uEAv1htQ0V83Z9Rj+Y=
|
||||
golang.org/x/sync v0.22.0 h1:SZjpbeLmrCk4xhRSZFNZW5gFUeCeFgjekvI/+gfScek=
|
||||
golang.org/x/sync v0.22.0/go.mod h1:9xrNwdLfx4jkKbNva9FpL6vEN7evnE43NNNJQ2LF3+0=
|
||||
golang.org/x/sync v0.23.0 h1:KameEIfc1IkluZyXWLn39Wd4tURc6GbCiISGiZm2bQk=
|
||||
golang.org/x/sync v0.23.0/go.mod h1:sUUOizhqBxiL6pEWpqNLUiaJn1ShEbZ6BBqskPbjZm0=
|
||||
golang.org/x/sys v0.0.0-20180810173357-98c5dad5d1a0/go.mod h1:STP8DvDyc/dI5b8T5hshtkjS+E42TnysNCUPdjciGhY=
|
||||
golang.org/x/sys v0.0.0-20180830151530-49385e6e1522/go.mod h1:STP8DvDyc/dI5b8T5hshtkjS+E42TnysNCUPdjciGhY=
|
||||
golang.org/x/sys v0.0.0-20180905080454-ebe1bf3edb33/go.mod h1:STP8DvDyc/dI5b8T5hshtkjS+E42TnysNCUPdjciGhY=
|
||||
@@ -2477,8 +2477,8 @@ golang.org/x/sys v0.6.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.8.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.12.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.13.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.47.0 h1:o7XGOvZQCADBQQ4Y7VNq2dRWQR7JmOUW8Kxx4ZsNgWs=
|
||||
golang.org/x/sys v0.47.0/go.mod h1:4GL1E5IUh+htKOUEOaiffhrAeqysfVGipDYzABqnCmw=
|
||||
golang.org/x/sys v0.48.0 h1:bbX/i/6MgT9BVLM9RT1thmxL04yeTAhbEz4SyadbXoo=
|
||||
golang.org/x/sys v0.48.0/go.mod h1:hNLxWAXmnKAxqDtdwIYC4bM9oQPEecfsnNMuSxOs3og=
|
||||
golang.org/x/term v0.0.0-20201126162022-7de9c90e9dd1/go.mod h1:bj7SfCRtBDWHUb9snDiAeCFNEtKQo2Wmx5Cou7ajbmo=
|
||||
golang.org/x/term v0.0.0-20210220032956-6a3ed077a48d/go.mod h1:bj7SfCRtBDWHUb9snDiAeCFNEtKQo2Wmx5Cou7ajbmo=
|
||||
golang.org/x/term v0.0.0-20210615171337-6886f2dfbf5b/go.mod h1:jbD1KX2456YbFQfuXm/mYQcufACuNUgVhRMnK/tPxf8=
|
||||
@@ -2511,8 +2511,8 @@ golang.org/x/text v0.8.0/go.mod h1:e1OnstbJyHTd6l/uOt8jFFHp6TRDWZR/bV3emEE/zU8=
|
||||
golang.org/x/text v0.9.0/go.mod h1:e1OnstbJyHTd6l/uOt8jFFHp6TRDWZR/bV3emEE/zU8=
|
||||
golang.org/x/text v0.13.0/go.mod h1:TvPlkZtksWOMsz7fbANvkp4WM8x/WCo/om8BMLbz+aE=
|
||||
golang.org/x/text v0.14.0/go.mod h1:18ZOQIKpY8NJVqYksKHtTdi31H5itFRjB5/qKTNYzSU=
|
||||
golang.org/x/text v0.41.0 h1:vz/seA0lnX87Othu2f/0L24RcgrXD9/YFTSuGjj3rH8=
|
||||
golang.org/x/text v0.41.0/go.mod h1:jvf1O8ajNzZqhSrQBPbutR/EB83Cc0CFrezNQIwbb5M=
|
||||
golang.org/x/text v0.42.0 h1:JbOZXgfeCPU9gacVtYliJqOhD+zhrEqK4LfdpmlUZqI=
|
||||
golang.org/x/text v0.42.0/go.mod h1:ojzP1Z+2QtioaF8DTtO8K5q7JWVVYwZKenzujK0Zd0E=
|
||||
golang.org/x/time v0.0.0-20181108054448-85acf8d2951c/go.mod h1:tRJNPiyCQ0inRvYxbN9jk5I+vvW/OXSQhTDSoE431IQ=
|
||||
golang.org/x/time v0.0.0-20190308202827-9d24e82272b4/go.mod h1:tRJNPiyCQ0inRvYxbN9jk5I+vvW/OXSQhTDSoE431IQ=
|
||||
golang.org/x/time v0.0.0-20191024005414-555d28b269f0/go.mod h1:tRJNPiyCQ0inRvYxbN9jk5I+vvW/OXSQhTDSoE431IQ=
|
||||
@@ -2589,8 +2589,8 @@ golang.org/x/tools v0.6.0/go.mod h1:Xwgl3UAJ/d3gWutnCtw505GrjyAbvKui8lOU390QaIU=
|
||||
golang.org/x/tools v0.7.0/go.mod h1:4pg6aUX35JBAogB10C9AtvVL+qowtN4pT3CGSQex14s=
|
||||
golang.org/x/tools v0.13.0/go.mod h1:HvlwmtVNQAhOuCjW7xxvovg8wbNq7LwfXh/k7wXUl58=
|
||||
golang.org/x/tools v0.14.0/go.mod h1:uYBEerGOWcJyEORxN+Ek8+TT266gXkNlHdJBwexUsBg=
|
||||
golang.org/x/tools v0.48.0 h1:3+hClM1aLL5mjMKm5ovokw9epgRXPuu2tILgismM6RE=
|
||||
golang.org/x/tools v0.48.0/go.mod h1:08xX0orndb/F7jJxGDicx061tyd5pcMto75YMAXr6lk=
|
||||
golang.org/x/tools v0.49.0 h1:3NI7VXzL9+1WZD52Dx2ttoPwD5DWrFGpl9mFZDlmisI=
|
||||
golang.org/x/tools v0.49.0/go.mod h1:SJNXV9DBKT0UbdttsQjbfJlAE/q+y36++zo3uL3N0Oo=
|
||||
golang.org/x/tools/godoc v0.1.0-deprecated h1:o+aZ1BOj6Hsx/GBdJO/s815sqftjSnrZZwyYTHODvtk=
|
||||
golang.org/x/tools/godoc v0.1.0-deprecated/go.mod h1:qM63CriJ961IHWmnWa9CjZnBndniPt4a3CK0PVB9bIg=
|
||||
golang.org/x/xerrors v0.0.0-20190717185122-a985d3407aa7/go.mod h1:I/5z698sn9Ka8TeJc9MKroUUfqBBauWjQqLJ2OPfmY0=
|
||||
@@ -2668,8 +2668,8 @@ google.golang.org/api v0.106.0/go.mod h1:2Ts0XTHNVWxypznxWOYUeI4g3WdP9Pk2Qk58+a/
|
||||
google.golang.org/api v0.107.0/go.mod h1:2Ts0XTHNVWxypznxWOYUeI4g3WdP9Pk2Qk58+a/O9MY=
|
||||
google.golang.org/api v0.108.0/go.mod h1:2Ts0XTHNVWxypznxWOYUeI4g3WdP9Pk2Qk58+a/O9MY=
|
||||
google.golang.org/api v0.110.0/go.mod h1:7FC4Vvx1Mooxh8C5HWjzZHcavuS2f6pmJpZx60ca7iI=
|
||||
google.golang.org/api v0.296.0 h1:Nn5EHeKdGx70MFClaV/II0gsWUm6xhEjb0xYLylVvaA=
|
||||
google.golang.org/api v0.296.0/go.mod h1:02qB8+Ox1ZFzcaKFMguy1nQLJmSIyvV6Ff4txJEXtl4=
|
||||
google.golang.org/api v0.297.0 h1:WktxTsnnx0yZNnsR6j0q6hR21RnnK81FHTOPy/ux4OE=
|
||||
google.golang.org/api v0.297.0/go.mod h1:S4m8x0M6OkQpkOzGk1y9JG2sm4fFQrMh6dxzjCTszhE=
|
||||
google.golang.org/appengine v1.1.0/go.mod h1:EbEs0AVv82hx2wNQdGPgUI5lhzA/G0D9YwlJXL52JkM=
|
||||
google.golang.org/appengine v1.4.0/go.mod h1:xpcJRLb0r/rnEns0DIKYYv+WjYCduHsrkT7/EB5XEv4=
|
||||
google.golang.org/appengine v1.5.0/go.mod h1:xpcJRLb0r/rnEns0DIKYYv+WjYCduHsrkT7/EB5XEv4=
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
apiVersion: v1
|
||||
description: SeaweedFS
|
||||
name: seaweedfs
|
||||
appVersion: "4.46"
|
||||
appVersion: "4.47"
|
||||
# Dev note: Trigger a helm chart release by `git tag -a helm-<version>`
|
||||
version: 4.46.0
|
||||
version: 4.47.1
|
||||
|
||||
@@ -286,7 +286,7 @@ metadata:
|
||||
app.kubernetes.io/component: s3
|
||||
stringData:
|
||||
# this key must be an inline json config file
|
||||
seaweedfs_s3_config: '{"identities":[{"name":"anvAdmin","credentials":[{"accessKey":"snu8yoP6QAlY0ne4","secretKey":"PNzBcmeLNEdR0oviwm04NQAicOrDH1Km"}],"actions":["Admin","Read","Write"]},{"name":"anvReadOnly","credentials":[{"accessKey":"SCigFee6c5lbi04A","secretKey":"kgFhbT38R8WUYVtiFQ1OiSVOrYr3NKku"}],"actions":["Read"]}]}'
|
||||
seaweedfs_s3_config: '{"identities":[{"name":"anvAdmin","credentials":[{"accessKey":"snu8yoP6QAlY0ne4","secretKey":"PNzBcmeLNEdR0oviwm04NQAicOrDH1Km"}],"actions":["Admin","Read","Write"]},{"name":"anvReadOnly","credentials":[{"accessKey":"SCigFee6c5lbi04A","secretKey":"kgFhbT38R8WUYVtiFQ1OiSVOrYr3NKku"}],"actions":["Read","List"]}]}'
|
||||
```
|
||||
|
||||
#### Source S3 credentials from an existing Secret
|
||||
@@ -363,6 +363,27 @@ If `adminPassword` is empty or not set, the admin interface runs without authent
|
||||
|
||||
As an alternative, a kubernetes Secret can be used (`admin.secret.existingSecret`).
|
||||
|
||||
### Admin listen address
|
||||
|
||||
Since SeaweedFS 4.46, `weed admin` defaults to listening on loopback (`127.0.0.1`).
|
||||
The chart's httpGet readiness/liveness probes dial the pod IP, so the admin
|
||||
server must bind a non-loopback address for the probes to succeed. The chart
|
||||
therefore passes `-ip={{ .Values.admin.ip }}`, defaulting `admin.ip` to `0.0.0.0`
|
||||
(the pre-4.46 behaviour of listening on all interfaces).
|
||||
|
||||
Binding a non-loopback address requires authentication: `weed admin` refuses to
|
||||
start on a non-loopback address without `-adminPassword`, so the chart fails at
|
||||
render time if `admin.ip` is non-loopback and authentication is not configured via
|
||||
`admin.secret.adminPassword`, `admin.secret.existingSecret`, or
|
||||
`WEED_ADMIN_PASSWORD` supplied through `admin.extraEnvironmentVars` /
|
||||
`admin.secretExtraEnvironmentVars`. The whole `127.0.0.0/8` range and `::1` are
|
||||
treated as loopback (matching `weed admin`); `localhost` is treated as
|
||||
non-loopback. Set `admin.ip` to a loopback address only if you also replace the
|
||||
httpGet probes (e.g. with an `exec` probe that checks `127.0.0.1`).
|
||||
|
||||
The `-ip` flag requires SeaweedFS 4.46 or newer; pinning `admin.imageOverride`
|
||||
to an older image is not supported with this chart version.
|
||||
|
||||
### Admin Data Persistence
|
||||
|
||||
The admin component can store configuration and maintenance data. You can configure storage in several ways:
|
||||
|
||||
@@ -0,0 +1,7 @@
|
||||
# Admin install: exercises the admin StatefulSet, which passes -ip (default
|
||||
# 0.0.0.0) and therefore requires authentication to bind a non-loopback address.
|
||||
admin:
|
||||
enabled: true
|
||||
secret:
|
||||
adminUser: "admin"
|
||||
adminPassword: "ci-admin-password"
|
||||
@@ -6,6 +6,11 @@
|
||||
{{- if and (not .Values.admin.masters) (not .Values.global.seaweedfs.masterServer) (not .Values.master.enabled) }}
|
||||
{{- fail "admin.masters or global.seaweedfs.masterServer must be set if master.enabled is false" -}}
|
||||
{{- end }}
|
||||
{{- $adminAuthEnabled := include "seaweedfs.admin.authEnabled" . }}
|
||||
{{- $adminIp := .Values.admin.ip | default "0.0.0.0" }}
|
||||
{{- if and (not (include "seaweedfs.admin.isLoopbackIp" $adminIp)) (ne $adminAuthEnabled "true") }}
|
||||
{{- fail (printf "admin.ip is set to %q (non-loopback) but admin authentication is not configured. Since `weed admin` 4.46 refuses to bind a non-loopback address without authentication, the admin container would exit on startup. Set admin.secret.adminPassword or admin.secret.existingSecret, or supply WEED_ADMIN_PASSWORD via admin.extraEnvironmentVars / admin.secretExtraEnvironmentVars, or set admin.ip to a loopback address such as 127.0.0.1 (note: a loopback bind makes the chart's httpGet readiness/liveness probes fail)." $adminIp) -}}
|
||||
{{- end }}
|
||||
apiVersion: apps/v1
|
||||
kind: StatefulSet
|
||||
metadata:
|
||||
@@ -162,6 +167,7 @@ spec:
|
||||
-v={{ .Values.global.seaweedfs.loggingLevel }} \
|
||||
{{- end }}
|
||||
admin \
|
||||
-ip={{ .Values.admin.ip | default "0.0.0.0" }} \
|
||||
-port={{ .Values.admin.port }} \
|
||||
-port.grpc={{ .Values.admin.grpcPort }} \
|
||||
{{- if or (eq .Values.admin.data.type "hostPath") (eq .Values.admin.data.type "persistentVolumeClaim") (eq .Values.admin.data.type "emptyDir") (eq .Values.admin.data.type "existingClaim") }}
|
||||
|
||||
@@ -44,6 +44,13 @@ spec:
|
||||
{{- with .Values.allInOne.podAnnotations }}
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
{{- $existingS3ConfigSecret := or .Values.allInOne.s3.existingConfigSecret .Values.s3.existingConfigSecret .Values.filer.s3.existingConfigSecret }}
|
||||
{{- if $existingS3ConfigSecret }}
|
||||
{{- $configSecret := (lookup "v1" "Secret" .Release.Namespace $existingS3ConfigSecret) | default dict }}
|
||||
checksum/s3config: {{ $configSecret | toYaml | sha256sum }}
|
||||
{{- else }}
|
||||
checksum/s3config: {{ include (print .Template.BasePath "/s3/s3-secret.yaml") . | sha256sum }}
|
||||
{{- end }}
|
||||
spec:
|
||||
restartPolicy: {{ default .Values.global.seaweedfs.restartPolicy .Values.allInOne.restartPolicy }}
|
||||
{{- if .Values.allInOne.affinity }}
|
||||
|
||||
@@ -42,6 +42,12 @@ spec:
|
||||
{{- with .Values.s3.podAnnotations }}
|
||||
{{- toYaml . | nindent 8 }}
|
||||
{{- end }}
|
||||
{{- if .Values.s3.existingConfigSecret }}
|
||||
{{- $configSecret := (lookup "v1" "Secret" .Release.Namespace .Values.s3.existingConfigSecret) | default dict }}
|
||||
checksum/s3config: {{ $configSecret | toYaml | sha256sum }}
|
||||
{{- else }}
|
||||
checksum/s3config: {{ include (print .Template.BasePath "/s3/s3-secret.yaml") . | sha256sum }}
|
||||
{{- end }}
|
||||
spec:
|
||||
restartPolicy: {{ default .Values.global.seaweedfs.restartPolicy .Values.s3.restartPolicy }}
|
||||
{{- if .Values.s3.affinity }}
|
||||
|
||||
@@ -60,7 +60,7 @@ stringData:
|
||||
read_access_key_id: {{ $access_key_read }}
|
||||
read_secret_access_key: {{ $secret_key_read }}
|
||||
{{- end }}
|
||||
seaweedfs_s3_config: '{"identities":[{"name":"anvAdmin","credentials":[{"accessKey":"{{ $access_key_admin }}","secretKey":"{{ $secret_key_admin }}"}],"actions":["Admin","Read","Write"]},{"name":"anvReadOnly","credentials":[{"accessKey":"{{ $access_key_read }}","secretKey":"{{ $secret_key_read }}"}],"actions":["Read"]}]}'
|
||||
seaweedfs_s3_config: '{"identities":[{"name":"anvAdmin","credentials":[{"accessKey":"{{ $access_key_admin }}","secretKey":"{{ $secret_key_admin }}"}],"actions":["Admin","Read","Write"]},{"name":"anvReadOnly","credentials":[{"accessKey":"{{ $access_key_read }}","secretKey":"{{ $secret_key_read }}"}],"actions":["Read","List"]}]}'
|
||||
{{- if .Values.filer.s3.auditLogConfig }}
|
||||
filer_s3_auditLogConfig.json: |
|
||||
{{ toJson .Values.filer.s3.auditLogConfig | nindent 4 }}
|
||||
|
||||
@@ -88,6 +88,43 @@ true
|
||||
{{- end -}}
|
||||
{{- end -}}
|
||||
|
||||
{{/* Classify an admin bind address as loopback, mirroring weed admin's
|
||||
isLoopbackIp (net.ParseIP + IsLoopback). Helm templates cannot call
|
||||
net.ParseIP, so we approximate: valid IPv4 addresses in 127.0.0.0/8
|
||||
(validated via regex to reject malformed values like "127.not-an-ip")
|
||||
and the IPv6 loopback "::1" / its expanded form "0:0:0:0:0:0:0:1" are
|
||||
loopback. Hostnames (e.g. "localhost") and wildcard addresses
|
||||
("0.0.0.0", "::") are non-loopback, matching the binary, which
|
||||
treats unparseable hostnames as non-loopback to be safe. Other IPv6
|
||||
loopback representations are not matched; the binary's own runtime
|
||||
validation is the authoritative guard. */}}
|
||||
{{- define "seaweedfs.admin.isLoopbackIp" -}}
|
||||
{{- $ip := toString . -}}
|
||||
{{- if or (regexMatch "^127\\.[0-9]{1,3}\\.[0-9]{1,3}\\.[0-9]{1,3}$" $ip) (eq $ip "::1") (eq $ip "0:0:0:0:0:0:0:1") -}}
|
||||
true
|
||||
{{- end -}}
|
||||
{{- end -}}
|
||||
|
||||
{{/* Whether admin authentication is enabled from any supported source:
|
||||
admin.secret (adminPassword or existingSecret), or WEED_ADMIN_PASSWORD
|
||||
supplied via extraEnvironmentVars / secretExtraEnvironmentVars (which
|
||||
weed admin picks up through viper's AutomaticEnv). A secret-backed
|
||||
entry counts as enabled even though the chart cannot read its value. */}}
|
||||
{{- define "seaweedfs.admin.authEnabled" -}}
|
||||
{{- if or .Values.admin.secret.existingSecret .Values.admin.secret.adminPassword -}}
|
||||
true
|
||||
{{- else -}}
|
||||
{{- $merged := dict -}}
|
||||
{{- $_ := include "seaweedfs.mergeExtraEnvironmentVars" (dict "global" .Values.global.seaweedfs "component" .Values.admin "target" $merged) -}}
|
||||
{{- $envPassword := index $merged "WEED_ADMIN_PASSWORD" -}}
|
||||
{{- if or (kindIs "map" $envPassword) (hasKey (.Values.admin.secretExtraEnvironmentVars | default dict) "WEED_ADMIN_PASSWORD") -}}
|
||||
true
|
||||
{{- else if and $envPassword (ne (toString $envPassword) "") -}}
|
||||
true
|
||||
{{- end -}}
|
||||
{{- end -}}
|
||||
{{- end -}}
|
||||
|
||||
{{/* Return the proper filer image */}}
|
||||
{{- define "seaweedfs.filer.image" -}}
|
||||
{{- if .Values.filer.imageOverride -}}
|
||||
|
||||
@@ -1328,6 +1328,20 @@ admin:
|
||||
replicas: 1
|
||||
port: 23646 # Default admin port
|
||||
grpcPort: 33646 # Default gRPC port for worker connections
|
||||
# IP address the admin server listens on. Since `weed admin` 4.46 defaults to
|
||||
# loopback (127.0.0.1), the chart must bind a non-loopback address for the
|
||||
# kubelet's httpGet readiness/liveness probes (which dial the pod IP) to ever
|
||||
# succeed. "0.0.0.0" restores the pre-4.46 behaviour of listening on all
|
||||
# interfaces. A non-loopback address requires authentication: set
|
||||
# admin.secret.adminPassword or admin.secret.existingSecret, or supply
|
||||
# WEED_ADMIN_PASSWORD via admin.extraEnvironmentVars /
|
||||
# admin.secretExtraEnvironmentVars; otherwise the admin container will exit
|
||||
# with a clear error rather than silently staying unready. The whole
|
||||
# 127.0.0.0/8 range and ::1 are treated as loopback (matching weed admin).
|
||||
# Set to a loopback address only if you also replace the httpGet probes.
|
||||
# Note: the -ip flag requires SeaweedFS 4.46 or newer; pinning
|
||||
# admin.imageOverride to an older image is not supported with this chart.
|
||||
ip: "0.0.0.0"
|
||||
loggingOverrideLevel: null
|
||||
|
||||
# Admin authentication
|
||||
|
||||
+1028
-1059
File diff suppressed because it is too large
Load Diff
|
Before Width: | Height: | Size: 54 KiB After Width: | Height: | Size: 53 KiB |
Generated
-1
@@ -4582,7 +4582,6 @@ dependencies = [
|
||||
"image",
|
||||
"jsonwebtoken",
|
||||
"kamadak-exif",
|
||||
"lazy_static",
|
||||
"libc",
|
||||
"md-5",
|
||||
"memmap2",
|
||||
|
||||
@@ -1,7 +1,10 @@
|
||||
[package]
|
||||
name = "weed-volume"
|
||||
version = "0.1.0"
|
||||
edition = "2021"
|
||||
edition = "2024"
|
||||
# The edition needs 1.85; the dependency tree needs more. Verified with
|
||||
# `cargo +1.91.1 check --all-targets` (1.90 fails on the AWS SDK).
|
||||
rust-version = "1.91.1"
|
||||
description = "SeaweedFS Volume Server — Rust implementation"
|
||||
|
||||
[lib]
|
||||
@@ -20,6 +23,14 @@ default = ["5bytes"]
|
||||
# Pulls redb's experimental_cursor (and therefore experimental-api-5).
|
||||
redb-experimental-cursor = ["redb/experimental_cursor"]
|
||||
|
||||
[lints.clippy]
|
||||
# Every RPC path returns tonic::Status (176 bytes). Boxing it would change
|
||||
# every handler signature for no gain, so the large-Err lint is off.
|
||||
result_large_err = "allow"
|
||||
# Protobuf message literals keep `..Default::default()` on purpose: it is
|
||||
# what lets a proto gain a field without touching every constructor.
|
||||
needless_update = "allow"
|
||||
|
||||
[dependencies]
|
||||
# Async runtime
|
||||
tokio = { version = "1", features = ["full"] }
|
||||
@@ -45,7 +56,6 @@ clap = { version = "4", features = ["derive"] }
|
||||
|
||||
# Metrics
|
||||
prometheus = { version = "0.13", default-features = false, features = ["process"] }
|
||||
lazy_static = "1"
|
||||
|
||||
# JWT
|
||||
jsonwebtoken = { version = "10", features = ["rust_crypto"] }
|
||||
|
||||
@@ -4,7 +4,10 @@ A drop-in replacement for the [SeaweedFS](https://github.com/seaweedfs/seaweedfs
|
||||
|
||||
## Building
|
||||
|
||||
Requires Rust 1.75+ (2021 edition).
|
||||
Requires Rust 1.91.1+ (2024 edition), matching `rust-version` in `Cargo.toml`.
|
||||
The patch release matters: 1.91.0 does not build. The edition itself only needs
|
||||
1.85; the higher floor comes from the dependency tree — chiefly the AWS SDK — so
|
||||
it moves with those crates. CI builds on the latest stable.
|
||||
|
||||
```bash
|
||||
cd seaweed-volume
|
||||
|
||||
@@ -3,7 +3,12 @@ fn main() -> Result<(), Box<dyn std::error::Error>> {
|
||||
// one, so the build needs no package manager and always sees the same
|
||||
// version. An explicit PROTOC still wins, for packagers supplying their own.
|
||||
if std::env::var_os("PROTOC").is_none() {
|
||||
std::env::set_var("PROTOC", protoc_bin_vendored::protoc_bin_path()?);
|
||||
// SAFETY: a build script's main runs single-threaded before anything
|
||||
// else in this process, so no other thread can be reading the
|
||||
// environment concurrently.
|
||||
unsafe {
|
||||
std::env::set_var("PROTOC", protoc_bin_vendored::protoc_bin_path()?);
|
||||
}
|
||||
}
|
||||
|
||||
let out_dir = std::path::PathBuf::from(std::env::var("OUT_DIR")?);
|
||||
|
||||
@@ -236,6 +236,7 @@ message VolumeIncrementalCopyResponse {
|
||||
|
||||
message VolumeMountRequest {
|
||||
uint32 volume_id = 1;
|
||||
optional string collection = 2;
|
||||
}
|
||||
message VolumeMountResponse {
|
||||
}
|
||||
|
||||
+106
-81
@@ -371,17 +371,18 @@ fn merge_options_file(args: Vec<String>) -> Vec<String> {
|
||||
if arg == "--" {
|
||||
break;
|
||||
}
|
||||
if arg.starts_with("--") {
|
||||
let key = if let Some(eq) = arg.find('=') {
|
||||
arg[2..eq].to_string()
|
||||
if let Some(long) = arg.strip_prefix("--") {
|
||||
let key = if let Some(eq) = long.find('=') {
|
||||
long[..eq].to_string()
|
||||
} else {
|
||||
arg[2..].to_string()
|
||||
long.to_string()
|
||||
};
|
||||
cli_flags.insert(key);
|
||||
} else if arg.starts_with('-') && arg.len() > 2 {
|
||||
} else if arg.len() > 2
|
||||
&& let Some(without_dash) = arg.strip_prefix('-')
|
||||
{
|
||||
// Single-dash long option (already normalized to -- at this point,
|
||||
// but handle both for safety)
|
||||
let without_dash = &arg[1..];
|
||||
let key = if let Some(eq) = without_dash.find('=') {
|
||||
without_dash[..eq].to_string()
|
||||
} else {
|
||||
@@ -401,15 +402,14 @@ fn merge_options_file(args: Vec<String>) -> Vec<String> {
|
||||
}
|
||||
|
||||
// Split on first `=`, ` `, or `:`
|
||||
let (name, value) =
|
||||
if let Some(pos) = trimmed.find(|c: char| c == '=' || c == ' ' || c == ':') {
|
||||
(
|
||||
trimmed[..pos].trim().to_string(),
|
||||
trimmed[pos + 1..].trim().to_string(),
|
||||
)
|
||||
} else {
|
||||
(trimmed.to_string(), String::new())
|
||||
};
|
||||
let (name, value) = if let Some(pos) = trimmed.find(['=', ' ', ':']) {
|
||||
(
|
||||
trimmed[..pos].trim().to_string(),
|
||||
trimmed[pos + 1..].trim().to_string(),
|
||||
)
|
||||
} else {
|
||||
(trimmed.to_string(), String::new())
|
||||
};
|
||||
|
||||
// Strip leading dashes from name
|
||||
let name = name.trim_start_matches('-').to_string();
|
||||
@@ -436,10 +436,8 @@ fn merge_options_file(args: Vec<String>) -> Vec<String> {
|
||||
/// Extract the options file path from args (looks for --options or -options).
|
||||
fn find_options_arg(args: &[String]) -> String {
|
||||
for i in 1..args.len() {
|
||||
if args[i] == "--options" || args[i] == "-options" {
|
||||
if i + 1 < args.len() {
|
||||
return args[i + 1].clone();
|
||||
}
|
||||
if (args[i] == "--options" || args[i] == "-options") && i + 1 < args.len() {
|
||||
return args[i + 1].clone();
|
||||
}
|
||||
if let Some(rest) = args[i].strip_prefix("--options=") {
|
||||
return rest.to_string();
|
||||
@@ -457,20 +455,22 @@ fn parse_duration(s: &str) -> std::time::Duration {
|
||||
if s.is_empty() {
|
||||
return std::time::Duration::from_secs(60);
|
||||
}
|
||||
if let Some(secs) = s.strip_suffix('s') {
|
||||
if let Ok(v) = secs.parse::<u64>() {
|
||||
return std::time::Duration::from_secs(v);
|
||||
}
|
||||
if let Some(secs) = s.strip_suffix('s')
|
||||
&& let Ok(v) = secs.parse::<u64>()
|
||||
{
|
||||
return std::time::Duration::from_secs(v);
|
||||
}
|
||||
if let Some(mins) = s.strip_suffix('m') {
|
||||
if let Ok(v) = mins.parse::<u64>() {
|
||||
return std::time::Duration::from_secs(v * 60);
|
||||
}
|
||||
if let Some(mins) = s.strip_suffix('m')
|
||||
&& let Ok(v) = mins.parse::<u64>()
|
||||
&& let Some(seconds) = v.checked_mul(60)
|
||||
{
|
||||
return std::time::Duration::from_secs(seconds);
|
||||
}
|
||||
if let Some(hours) = s.strip_suffix('h') {
|
||||
if let Ok(v) = hours.parse::<u64>() {
|
||||
return std::time::Duration::from_secs(v * 3600);
|
||||
}
|
||||
if let Some(hours) = s.strip_suffix('h')
|
||||
&& let Ok(v) = hours.parse::<u64>()
|
||||
&& let Some(seconds) = v.checked_mul(3600)
|
||||
{
|
||||
return std::time::Duration::from_secs(seconds);
|
||||
}
|
||||
// Fallback: try parsing as raw seconds
|
||||
if let Ok(v) = s.parse::<u64>() {
|
||||
@@ -503,40 +503,40 @@ fn parse_min_free_spaces(min_free_space: &str, min_free_space_percent: &str) ->
|
||||
}
|
||||
// Try parsing human-readable bytes: e.g. "10GiB", "500MiB", "1TiB"
|
||||
let s_upper = s.to_uppercase();
|
||||
if let Some(rest) = s_upper.strip_suffix("TIB") {
|
||||
if let Ok(v) = rest.trim().parse::<f64>() {
|
||||
return MinFreeSpace::Bytes((v * 1024.0 * 1024.0 * 1024.0 * 1024.0) as u64);
|
||||
}
|
||||
if let Some(rest) = s_upper.strip_suffix("TIB")
|
||||
&& let Ok(v) = rest.trim().parse::<f64>()
|
||||
{
|
||||
return MinFreeSpace::Bytes((v * 1024.0 * 1024.0 * 1024.0 * 1024.0) as u64);
|
||||
}
|
||||
if let Some(rest) = s_upper.strip_suffix("GIB") {
|
||||
if let Ok(v) = rest.trim().parse::<f64>() {
|
||||
return MinFreeSpace::Bytes((v * 1024.0 * 1024.0 * 1024.0) as u64);
|
||||
}
|
||||
if let Some(rest) = s_upper.strip_suffix("GIB")
|
||||
&& let Ok(v) = rest.trim().parse::<f64>()
|
||||
{
|
||||
return MinFreeSpace::Bytes((v * 1024.0 * 1024.0 * 1024.0) as u64);
|
||||
}
|
||||
if let Some(rest) = s_upper.strip_suffix("MIB") {
|
||||
if let Ok(v) = rest.trim().parse::<f64>() {
|
||||
return MinFreeSpace::Bytes((v * 1024.0 * 1024.0) as u64);
|
||||
}
|
||||
if let Some(rest) = s_upper.strip_suffix("MIB")
|
||||
&& let Ok(v) = rest.trim().parse::<f64>()
|
||||
{
|
||||
return MinFreeSpace::Bytes((v * 1024.0 * 1024.0) as u64);
|
||||
}
|
||||
if let Some(rest) = s_upper.strip_suffix("KIB") {
|
||||
if let Ok(v) = rest.trim().parse::<f64>() {
|
||||
return MinFreeSpace::Bytes((v * 1024.0) as u64);
|
||||
}
|
||||
if let Some(rest) = s_upper.strip_suffix("KIB")
|
||||
&& let Ok(v) = rest.trim().parse::<f64>()
|
||||
{
|
||||
return MinFreeSpace::Bytes((v * 1024.0) as u64);
|
||||
}
|
||||
if let Some(rest) = s_upper.strip_suffix("TB") {
|
||||
if let Ok(v) = rest.trim().parse::<f64>() {
|
||||
return MinFreeSpace::Bytes((v * 1_000_000_000_000.0) as u64);
|
||||
}
|
||||
if let Some(rest) = s_upper.strip_suffix("TB")
|
||||
&& let Ok(v) = rest.trim().parse::<f64>()
|
||||
{
|
||||
return MinFreeSpace::Bytes((v * 1_000_000_000_000.0) as u64);
|
||||
}
|
||||
if let Some(rest) = s_upper.strip_suffix("GB") {
|
||||
if let Ok(v) = rest.trim().parse::<f64>() {
|
||||
return MinFreeSpace::Bytes((v * 1_000_000_000.0) as u64);
|
||||
}
|
||||
if let Some(rest) = s_upper.strip_suffix("GB")
|
||||
&& let Ok(v) = rest.trim().parse::<f64>()
|
||||
{
|
||||
return MinFreeSpace::Bytes((v * 1_000_000_000.0) as u64);
|
||||
}
|
||||
if let Some(rest) = s_upper.strip_suffix("MB") {
|
||||
if let Ok(v) = rest.trim().parse::<f64>() {
|
||||
return MinFreeSpace::Bytes((v * 1_000_000.0) as u64);
|
||||
}
|
||||
if let Some(rest) = s_upper.strip_suffix("MB")
|
||||
&& let Ok(v) = rest.trim().parse::<f64>()
|
||||
{
|
||||
return MinFreeSpace::Bytes((v * 1_000_000.0) as u64);
|
||||
}
|
||||
// Default: 1%
|
||||
MinFreeSpace::Percent(1.0)
|
||||
@@ -1028,20 +1028,20 @@ pub fn parse_security_config(path: &str) -> SecurityConfig {
|
||||
"cipher_suites" => cfg.tls_policy.cipher_suites = value.to_string(),
|
||||
_ => {}
|
||||
},
|
||||
Section::Guard => match key {
|
||||
"white_list" => {
|
||||
Section::Guard => {
|
||||
if key == "white_list" {
|
||||
cfg.guard_white_list = value
|
||||
.split(',')
|
||||
.map(|s| s.trim().to_string())
|
||||
.filter(|s| !s.is_empty())
|
||||
.collect();
|
||||
}
|
||||
_ => {}
|
||||
},
|
||||
Section::Access => match key {
|
||||
"ui" => cfg.access_ui = value.parse().unwrap_or(false),
|
||||
_ => {}
|
||||
},
|
||||
}
|
||||
Section::Access => {
|
||||
if key == "ui" {
|
||||
cfg.access_ui = value.parse().unwrap_or(false)
|
||||
}
|
||||
}
|
||||
Section::None => {}
|
||||
}
|
||||
}
|
||||
@@ -1188,12 +1188,11 @@ fn apply_env_overrides(cfg: &mut SecurityConfig) {
|
||||
/// Mirrors Go's `util.DetectedHostAddress()`.
|
||||
fn detect_host_address() -> String {
|
||||
// Connect to a remote address to determine the local outbound IP
|
||||
if let Ok(socket) = UdpSocket::bind("0.0.0.0:0") {
|
||||
if socket.connect("8.8.8.8:80").is_ok() {
|
||||
if let Ok(addr) = socket.local_addr() {
|
||||
return addr.ip().to_string();
|
||||
}
|
||||
}
|
||||
if let Ok(socket) = UdpSocket::bind("0.0.0.0:0")
|
||||
&& socket.connect("8.8.8.8:80").is_ok()
|
||||
&& let Ok(addr) = socket.local_addr()
|
||||
{
|
||||
return addr.ip().to_string();
|
||||
}
|
||||
"localhost".to_string()
|
||||
}
|
||||
@@ -1209,21 +1208,30 @@ mod tests {
|
||||
LOCK.get_or_init(|| Mutex::new(())).lock().unwrap()
|
||||
}
|
||||
|
||||
// SAFETY (all env mutation in this module): `set_var`/`remove_var` are
|
||||
// unsafe as of Rust 2024 because they race with concurrent readers in
|
||||
// other threads. Every test that reaches these helpers holds
|
||||
// `process_state_lock()` for the duration, so only one test at a time
|
||||
// touches the environment and none observes another's edit.
|
||||
fn with_temp_env_var<F: FnOnce()>(key: &str, value: Option<&str>, f: F) {
|
||||
let previous = std::env::var_os(key);
|
||||
match value {
|
||||
Some(v) => std::env::set_var(key, v),
|
||||
None => std::env::remove_var(key),
|
||||
unsafe {
|
||||
match value {
|
||||
Some(v) => std::env::set_var(key, v),
|
||||
None => std::env::remove_var(key),
|
||||
}
|
||||
}
|
||||
f();
|
||||
restore_env_var(key, previous);
|
||||
}
|
||||
|
||||
fn restore_env_var(key: &str, value: Option<OsString>) {
|
||||
if let Some(value) = value {
|
||||
std::env::set_var(key, value);
|
||||
} else {
|
||||
std::env::remove_var(key);
|
||||
unsafe {
|
||||
if let Some(value) = value {
|
||||
std::env::set_var(key, value);
|
||||
} else {
|
||||
std::env::remove_var(key);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1268,7 +1276,10 @@ mod tests {
|
||||
.collect();
|
||||
|
||||
for key in KEYS {
|
||||
std::env::remove_var(key);
|
||||
// SAFETY: as above — the caller holds `process_state_lock()`.
|
||||
unsafe {
|
||||
std::env::remove_var(key);
|
||||
}
|
||||
}
|
||||
|
||||
f();
|
||||
@@ -1285,6 +1296,14 @@ mod tests {
|
||||
assert_eq!(parse_duration("1h"), std::time::Duration::from_secs(3600));
|
||||
assert_eq!(parse_duration("30"), std::time::Duration::from_secs(30));
|
||||
assert_eq!(parse_duration(""), std::time::Duration::from_secs(60));
|
||||
assert_eq!(
|
||||
parse_duration("307445734561825861m"),
|
||||
std::time::Duration::from_secs(60)
|
||||
);
|
||||
assert_eq!(
|
||||
parse_duration("5124095576030432h"),
|
||||
std::time::Duration::from_secs(60)
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
@@ -1404,12 +1423,18 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn test_resolve_config_defaults_dir_to_platform_temp_dir() {
|
||||
// resolve_config reads HOME/USERPROFILE and the WEED_* set, so it has to
|
||||
// hold the same lock the mutation helpers take — a concurrent set_var
|
||||
// during this read is exactly what makes those calls unsafe.
|
||||
let _guard = process_state_lock();
|
||||
let cfg = resolve_config(Cli::parse_from(["bin"]));
|
||||
assert_eq!(cfg.folders, vec![default_volume_dir()]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_resolve_config_index_accepts_redb_and_leveldb_aliases() {
|
||||
// As above: resolve_config reads the environment.
|
||||
let _guard = process_state_lock();
|
||||
let pairs = [
|
||||
("memory", NeedleMapKind::InMemory),
|
||||
("redb", NeedleMapKind::Redb),
|
||||
|
||||
@@ -1,5 +1,6 @@
|
||||
pub mod config;
|
||||
pub mod images;
|
||||
pub mod malloc_tuning;
|
||||
pub mod metrics;
|
||||
pub mod remote_storage;
|
||||
pub mod security;
|
||||
|
||||
@@ -39,6 +39,11 @@ const GRPC_MAX_HEADER_LIST_SIZE: u32 = 8 * 1024 * 1024;
|
||||
const GRPC_MAX_CONCURRENT_STREAMS: u32 = 1000;
|
||||
|
||||
fn main() {
|
||||
// Before anything allocates: stop glibc from training its mmap threshold
|
||||
// upward on our large EC buffers and turning them into heap it never
|
||||
// returns. See seaweed_volume::malloc_tuning for the measurements.
|
||||
let malloc_tuning = seaweed_volume::malloc_tuning::pin_mmap_threshold();
|
||||
|
||||
install_default_crypto_provider();
|
||||
|
||||
// Initialize tracing
|
||||
@@ -65,6 +70,19 @@ fn main() {
|
||||
"SeaweedFS Volume Server (Rust) v{}",
|
||||
seaweed_volume::version::full_version()
|
||||
);
|
||||
match malloc_tuning {
|
||||
seaweed_volume::malloc_tuning::MallocTuning::Pinned(bytes) => {
|
||||
info!("pinned glibc M_MMAP_THRESHOLD to {} bytes", bytes)
|
||||
}
|
||||
seaweed_volume::malloc_tuning::MallocTuning::DeferredToEnv => info!(
|
||||
"an allocator mmap-threshold override ({}) is set; leaving glibc's mmap threshold to the environment",
|
||||
seaweed_volume::malloc_tuning::MMAP_THRESHOLD_ENV
|
||||
),
|
||||
seaweed_volume::malloc_tuning::MallocTuning::Failed => {
|
||||
warn!("mallopt(M_MMAP_THRESHOLD) failed; large freed buffers may stay resident")
|
||||
}
|
||||
seaweed_volume::malloc_tuning::MallocTuning::NotApplicable => {}
|
||||
}
|
||||
|
||||
// Register Prometheus metrics
|
||||
metrics::register_metrics();
|
||||
|
||||
@@ -0,0 +1,467 @@
|
||||
//! Keep glibc from silently converting large short-lived buffers into heap the
|
||||
//! process never gives back.
|
||||
//!
|
||||
//! glibc serves an allocation with `mmap` when it is at least
|
||||
//! `M_MMAP_THRESHOLD` (128 KiB by default), and `munmap`s it on free, so the
|
||||
//! pages go straight back to the OS. That threshold is **adaptive**: whenever a
|
||||
//! block that came from `mmap` is freed, glibc raises the threshold to that
|
||||
//! block's size — up to 32 MiB — on the theory that a workload repeatedly
|
||||
//! allocating buffers of that size is better served from the heap.
|
||||
//!
|
||||
//! For a volume server that theory is wrong in a specific, expensive way. EC
|
||||
//! reconstruction and needle reassembly allocate large, short-lived buffers.
|
||||
//! The first few are mmap'd and freed, which trains the threshold upward; every
|
||||
//! later buffer of that size is then carved out of the heap instead. Heap
|
||||
//! memory is only returned to the OS from the top of the arena, so those pages
|
||||
//! stay resident as anonymous memory for the life of the process. They are
|
||||
//! still *reusable* — this is not a leak, and a repeat workload does not grow
|
||||
//! the footprint further — but under a hard cgroup `MemoryMax` they are
|
||||
//! indistinguishable from a leak, because anonymous pages cannot be reclaimed
|
||||
//! under pressure the way page cache can. The retained footprint eats exactly
|
||||
//! the headroom that a burst of maintenance work needs, and the process is
|
||||
//! OOM-killed while most of its resident memory is free-but-unreturned.
|
||||
//!
|
||||
//! Measured on a 17-node cluster (EC 10+4, `--index=redb`), one node, two
|
||||
//! identical `ec.scrub -mode full` rounds over 10912 EC files each, comparing
|
||||
//! the same unit restarted with and without a pinned threshold:
|
||||
//!
|
||||
//! | | baseline | round 1 | round 2 | 60s idle |
|
||||
//! |---|---|---|---|---|
|
||||
//! | default (adaptive) | 10 MB | 84 MB | 88 MB | **88 MB** |
|
||||
//! | pinned threshold | 10 MB | 13 MB | 14 MB | **14 MB** |
|
||||
//!
|
||||
//! 78 MB retained versus 4 MB for identical work. On that cluster's heavier
|
||||
//! mixed scrub workloads the same effect reached ~600 MB of retained anonymous
|
||||
//! memory per volume server, against a 3 GiB cap.
|
||||
//!
|
||||
//! Calling `mallopt(M_MMAP_THRESHOLD, ...)` sets the threshold *and* disables
|
||||
//! the dynamic adjustment, which is the documented behaviour of setting it
|
||||
//! explicitly. We pin it to glibc's own default rather than inventing a value:
|
||||
//! the goal is to stop the adaptation, not to second-guess the default.
|
||||
|
||||
/// glibc's own default `M_MMAP_THRESHOLD`. Pinning to this value changes
|
||||
/// nothing about which allocations use `mmap` on a freshly started process; it
|
||||
/// only prevents the threshold from drifting upward later.
|
||||
#[cfg(all(target_os = "linux", target_env = "gnu"))]
|
||||
const DEFAULT_MMAP_THRESHOLD: libc::c_int = 128 * 1024;
|
||||
|
||||
/// Legacy environment variable glibc reads for the same setting. If an operator
|
||||
/// has set it, honour their value and do not override it.
|
||||
pub const MMAP_THRESHOLD_ENV: &str = "MALLOC_MMAP_THRESHOLD_";
|
||||
|
||||
/// Modern glibc tunables environment variable. Operators may set the threshold
|
||||
/// via `GLIBC_TUNABLES=glibc.malloc.mmap_threshold=...` instead of the legacy
|
||||
/// variable; that override is honoured too.
|
||||
pub const GLIBC_TUNABLES_ENV: &str = "GLIBC_TUNABLES";
|
||||
|
||||
/// The tunable name within `GLIBC_TUNABLES` that maps to `M_MMAP_THRESHOLD`.
|
||||
#[cfg(all(target_os = "linux", target_env = "gnu"))]
|
||||
const MMAP_THRESHOLD_TUNABLE: &str = "glibc.malloc.mmap_threshold";
|
||||
|
||||
/// Outcome of the tuning attempt, so the caller can log it and tests can assert
|
||||
/// on it without inspecting global allocator state.
|
||||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||||
pub enum MallocTuning {
|
||||
/// Threshold pinned to `DEFAULT_MMAP_THRESHOLD`; dynamic adjustment is off.
|
||||
Pinned(i32),
|
||||
/// An allocator override (`MALLOC_MMAP_THRESHOLD_` or
|
||||
/// `GLIBC_TUNABLES=glibc.malloc.mmap_threshold=...`) was set, so the
|
||||
/// operator's value wins.
|
||||
DeferredToEnv,
|
||||
/// `mallopt` reported failure. Not fatal — the server runs, it just keeps
|
||||
/// glibc's adaptive behaviour.
|
||||
Failed,
|
||||
/// Not glibc, so there is no adaptive threshold to pin.
|
||||
NotApplicable,
|
||||
}
|
||||
|
||||
/// Pin glibc's mmap threshold unless the operator has set an allocator override.
|
||||
/// Safe to call more than once; call it before serving traffic, since the point
|
||||
/// is to prevent the threshold from being trained upward by early allocations.
|
||||
pub fn pin_mmap_threshold() -> MallocTuning {
|
||||
pin_mmap_threshold_inner()
|
||||
}
|
||||
|
||||
#[cfg(all(target_os = "linux", target_env = "gnu"))]
|
||||
fn pin_mmap_threshold_inner() -> MallocTuning {
|
||||
if operator_mmap_threshold_override_active() {
|
||||
return MallocTuning::DeferredToEnv;
|
||||
}
|
||||
// SAFETY: `mallopt` is a libc entry point that takes two ints and mutates
|
||||
// only allocator-internal tunables. It has no preconditions and no effect
|
||||
// on memory this process already owns.
|
||||
let rc = unsafe { libc::mallopt(libc::M_MMAP_THRESHOLD, DEFAULT_MMAP_THRESHOLD) };
|
||||
if rc == 1 {
|
||||
MallocTuning::Pinned(DEFAULT_MMAP_THRESHOLD)
|
||||
} else {
|
||||
MallocTuning::Failed
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(not(all(target_os = "linux", target_env = "gnu")))]
|
||||
fn pin_mmap_threshold_inner() -> MallocTuning {
|
||||
MallocTuning::NotApplicable
|
||||
}
|
||||
|
||||
#[cfg(all(target_os = "linux", target_env = "gnu"))]
|
||||
fn operator_mmap_threshold_override_active() -> bool {
|
||||
// MALLOC_MMAP_THRESHOLD_: glibc calls atoi(value) then mallopt, which
|
||||
// always sets the threshold and disables dynamic adjustment — even for
|
||||
// empty, negative, or non-numeric values (atoi returns 0). So any presence
|
||||
// of the variable means the operator's override is in effect.
|
||||
std::env::var_os(MMAP_THRESHOLD_ENV).is_some()
|
||||
|| usable_glibc_tunable_threshold(std::env::var_os(GLIBC_TUNABLES_ENV))
|
||||
}
|
||||
|
||||
/// Look for `glibc.malloc.mmap_threshold=<value>` among the colon-separated
|
||||
/// tunables in `GLIBC_TUNABLES`. glibc's `parse_tunables_string` (elf/dl-tunables.c)
|
||||
/// rejects the **entire** string (returns -1) if it reaches `\0` before finding
|
||||
/// `=` in a name (last entry has no `=`), or if any entry's value contains a
|
||||
/// duplicate `=`. When `parse_tunables_string` returns -1, `parse_tunables`
|
||||
/// prints a warning and returns immediately without applying ANY tunable —
|
||||
/// including ones already parsed into the tunables array. We match that by
|
||||
/// returning `false` for the entire string on any of those conditions.
|
||||
///
|
||||
/// glibc parses tunable values with `_dl_strtoul`, which accepts decimal,
|
||||
/// `0x` hex, `0` octal, an optional sign (negatives wrap to `unsigned long`),
|
||||
/// and requires the entire value to be consumed; we match that with
|
||||
/// `dl_strtoul_consumes_all`.
|
||||
#[cfg(all(target_os = "linux", target_env = "gnu"))]
|
||||
fn usable_glibc_tunable_threshold(tunables: Option<std::ffi::OsString>) -> bool {
|
||||
let s = match tunables.and_then(|v| v.into_string().ok()) {
|
||||
Some(s) => s,
|
||||
None => return false,
|
||||
};
|
||||
if s.is_empty() {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Parse the string character-by-character, matching glibc's
|
||||
// parse_tunables_string logic exactly. Using split(':') would lose the
|
||||
// distinction between an entry terminated by ':' (skip) and one terminated
|
||||
// by '\0' with no '=' (reject entire string).
|
||||
let bytes = s.as_bytes();
|
||||
let mut pos = 0;
|
||||
let mut found_threshold = false;
|
||||
|
||||
loop {
|
||||
// Find where the name ends ('=', ':', or end of string).
|
||||
let name_start = pos;
|
||||
while pos < bytes.len() && bytes[pos] != b'=' && bytes[pos] != b':' {
|
||||
pos += 1;
|
||||
}
|
||||
|
||||
// End of string before '=' → glibc returns -1 (reject entire string).
|
||||
if pos >= bytes.len() {
|
||||
return false;
|
||||
}
|
||||
|
||||
// ':' before '=' → glibc skips this entry and continues.
|
||||
if bytes[pos] == b':' {
|
||||
pos += 1;
|
||||
continue;
|
||||
}
|
||||
|
||||
// Skip the '='.
|
||||
let name_end = pos;
|
||||
pos += 1;
|
||||
|
||||
// Find where the value ends ('=', ':', or end of string).
|
||||
let val_start = pos;
|
||||
while pos < bytes.len() && bytes[pos] != b'=' && bytes[pos] != b':' {
|
||||
pos += 1;
|
||||
}
|
||||
|
||||
// '=' in value → glibc returns -1 (reject entire string).
|
||||
if pos < bytes.len() && bytes[pos] == b'=' {
|
||||
return false;
|
||||
}
|
||||
|
||||
let key = &s[name_start..name_end];
|
||||
let val = &s[val_start..pos];
|
||||
|
||||
if key == MMAP_THRESHOLD_TUNABLE && dl_strtoul_consumes_all(val) {
|
||||
found_threshold = true;
|
||||
}
|
||||
|
||||
// End of string → done.
|
||||
if pos >= bytes.len() {
|
||||
break;
|
||||
}
|
||||
|
||||
// Skip the ':'.
|
||||
pos += 1;
|
||||
}
|
||||
|
||||
found_threshold
|
||||
}
|
||||
|
||||
/// Replicate glibc's `_dl_strtoul` (elf/dl-misc.c) just enough to determine
|
||||
/// whether it would consume the entire string — which is what
|
||||
/// `tunable_parse_num` checks (`endptr == strval + len`). Returns `true` if
|
||||
/// glibc would accept the value and apply it.
|
||||
///
|
||||
/// `_dl_strtoul` skips leading spaces/tabs, accepts an optional `+`/`-` sign,
|
||||
/// and parses `0x`-prefixed hex, `0`-prefixed octal, or plain decimal. A
|
||||
/// negative result wraps to `unsigned long` (`-1` → `SIZE_MAX`). If no digit is
|
||||
/// found after the sign, the end pointer stays at the current position — which
|
||||
/// still counts as "consumed" when the string is empty or whitespace-only
|
||||
/// (value 0). On overflow, `_dl_strtoul` stops at the overflowing digit (endptr
|
||||
/// does not reach the end), so `tunable_parse_num` rejects the value.
|
||||
#[cfg(all(target_os = "linux", target_env = "gnu"))]
|
||||
fn dl_strtoul_consumes_all(s: &str) -> bool {
|
||||
let bytes = s.as_bytes();
|
||||
let mut pos = 0;
|
||||
|
||||
// Skip leading whitespace (spaces and tabs, matching _dl_strtoul).
|
||||
while pos < bytes.len() && (bytes[pos] == b' ' || bytes[pos] == b'\t') {
|
||||
pos += 1;
|
||||
}
|
||||
|
||||
// Optional sign.
|
||||
if pos < bytes.len() && (bytes[pos] == b'-' || bytes[pos] == b'+') {
|
||||
pos += 1;
|
||||
}
|
||||
|
||||
// Must have at least one digit (0-9) to start parsing, unless we're already
|
||||
// at the end (empty / whitespace-only / sign-only → value 0, consumed).
|
||||
if pos >= bytes.len() {
|
||||
return true;
|
||||
}
|
||||
if bytes[pos] < b'0' || bytes[pos] > b'9' {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Determine base: 0x → hex, 0 → octal, else decimal. _dl_strtoul unconditionally
|
||||
// advances past "0x"/"0X" when the first char is '0' and the next is 'x'/'X',
|
||||
// even if no hex digit follows — in that case the digit loop breaks immediately,
|
||||
// endptr reaches the end, and the value is 0.
|
||||
let base: u32 = if bytes[pos] == b'0'
|
||||
&& pos + 1 < bytes.len()
|
||||
&& (bytes[pos + 1] == b'x' || bytes[pos + 1] == b'X')
|
||||
{
|
||||
pos += 2; // skip "0x"
|
||||
16
|
||||
} else if bytes[pos] == b'0' {
|
||||
8
|
||||
} else {
|
||||
10
|
||||
};
|
||||
|
||||
// Parse digits with overflow detection, matching _dl_strtoul's cutoff/cutlim
|
||||
// logic. On overflow, _dl_strtoul sets endptr to the overflowing digit and
|
||||
// returns UINT64_MAX — so the value is NOT fully consumed and
|
||||
// tunable_parse_num rejects it.
|
||||
let mut result: u64 = 0;
|
||||
let cutoff = u64::MAX / base as u64;
|
||||
let cutlim = u64::MAX % base as u64;
|
||||
|
||||
while pos < bytes.len() {
|
||||
let b = bytes[pos];
|
||||
let digval: u32 = match digit_value(b, base) {
|
||||
Some(v) => v,
|
||||
None => break,
|
||||
};
|
||||
if result > cutoff || (result == cutoff && digval as u64 > cutlim) {
|
||||
// Overflow: _dl_strtoul stops here, endptr points at this digit.
|
||||
return false;
|
||||
}
|
||||
result *= base as u64;
|
||||
result += digval as u64;
|
||||
pos += 1;
|
||||
}
|
||||
|
||||
// The entire string must be consumed (matching tunable_parse_num's check).
|
||||
pos == bytes.len()
|
||||
}
|
||||
|
||||
/// Returns the numeric value of a digit byte in the given base, or `None` if
|
||||
/// the byte is not a valid digit in that base.
|
||||
#[cfg(all(target_os = "linux", target_env = "gnu"))]
|
||||
fn digit_value(b: u8, base: u32) -> Option<u32> {
|
||||
if (b'0'..=b'0' + (base - 1).min(9) as u8).contains(&b) {
|
||||
return Some((b - b'0') as u32);
|
||||
}
|
||||
if base == 16 {
|
||||
if (b'a'..=b'f').contains(&b) {
|
||||
return Some((b - b'a' + 10) as u32);
|
||||
}
|
||||
if (b'A'..=b'F').contains(&b) {
|
||||
return Some((b - b'A' + 10) as u32);
|
||||
}
|
||||
}
|
||||
None
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
#[test]
|
||||
fn env_override_constants_match_glibc_names() {
|
||||
// Verified against the real accessor rather than a copy of the name, so
|
||||
// renaming the constant cannot silently break the override contract.
|
||||
assert_eq!(MMAP_THRESHOLD_ENV, "MALLOC_MMAP_THRESHOLD_");
|
||||
assert_eq!(GLIBC_TUNABLES_ENV, "GLIBC_TUNABLES");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn calling_twice_is_stable() {
|
||||
// Startup paths get re-entered in tests and in `weed mini`; the second
|
||||
// call must not report a different outcome from the first.
|
||||
let first = pin_mmap_threshold();
|
||||
let second = pin_mmap_threshold();
|
||||
assert_eq!(first, second);
|
||||
}
|
||||
|
||||
#[cfg(all(target_os = "linux", target_env = "gnu"))]
|
||||
#[test]
|
||||
fn pins_threshold_on_glibc_when_no_override_is_set() {
|
||||
// The env override is not set in the test process, so this exercises the
|
||||
// mallopt path. If an override happens to be present, defer to it.
|
||||
if operator_mmap_threshold_override_active() {
|
||||
assert_eq!(pin_mmap_threshold(), MallocTuning::DeferredToEnv);
|
||||
return;
|
||||
}
|
||||
assert_eq!(
|
||||
pin_mmap_threshold(),
|
||||
MallocTuning::Pinned(DEFAULT_MMAP_THRESHOLD),
|
||||
"mallopt(M_MMAP_THRESHOLD) should succeed on glibc"
|
||||
);
|
||||
}
|
||||
|
||||
#[cfg(not(all(target_os = "linux", target_env = "gnu")))]
|
||||
#[test]
|
||||
fn is_a_noop_off_glibc() {
|
||||
// No glibc adaptive threshold exists off glibc, so there is nothing to
|
||||
// pin regardless of any environment variables that happen to be set.
|
||||
assert_eq!(pin_mmap_threshold(), MallocTuning::NotApplicable);
|
||||
}
|
||||
|
||||
#[cfg(all(target_os = "linux", target_env = "gnu"))]
|
||||
#[test]
|
||||
fn dl_strtoul_consumes_all_matches_glibc_parser() {
|
||||
// Decimal — any non-empty decimal integer is accepted, including
|
||||
// negative (wraps to unsigned) and zero.
|
||||
assert!(dl_strtoul_consumes_all("131072"));
|
||||
assert!(dl_strtoul_consumes_all("0"));
|
||||
assert!(dl_strtoul_consumes_all("-1"));
|
||||
assert!(dl_strtoul_consumes_all("-131072"));
|
||||
// Values above i64::MAX are valid for glibc's unsigned parser.
|
||||
assert!(dl_strtoul_consumes_all("9223372036854775808"));
|
||||
// Hex with 0x prefix.
|
||||
assert!(dl_strtoul_consumes_all("0x20000"));
|
||||
assert!(dl_strtoul_consumes_all("0X20000"));
|
||||
assert!(dl_strtoul_consumes_all("0x0"));
|
||||
// Octal with leading 0.
|
||||
assert!(dl_strtoul_consumes_all("010"));
|
||||
// Leading whitespace (spaces and tabs) is skipped.
|
||||
assert!(dl_strtoul_consumes_all(" 131072"));
|
||||
assert!(dl_strtoul_consumes_all("\t0x20000"));
|
||||
// Empty and whitespace-only strings are accepted (value 0).
|
||||
assert!(dl_strtoul_consumes_all(""));
|
||||
assert!(dl_strtoul_consumes_all(" "));
|
||||
assert!(dl_strtoul_consumes_all("\t"));
|
||||
// Sign-only strings are accepted: _dl_strtoul skips the sign, finds no
|
||||
// digit, sets endptr to the position after the sign (== end of string),
|
||||
// and returns 0. tunable_parse_num sees endptr == strval + len → true.
|
||||
assert!(dl_strtoul_consumes_all("-"));
|
||||
assert!(dl_strtoul_consumes_all("+"));
|
||||
|
||||
// Trailing garbage is rejected — _dl_strtoul stops at the first
|
||||
// non-digit and tunable_parse_num requires the entire string consumed.
|
||||
assert!(!dl_strtoul_consumes_all("131072abc"));
|
||||
// In hex mode, a-f are digits, so "0x20000abc" is a valid hex number.
|
||||
// Use a non-hex character like 'g' to test trailing garbage in hex.
|
||||
assert!(!dl_strtoul_consumes_all("0x20000g"));
|
||||
assert!(!dl_strtoul_consumes_all("128K"));
|
||||
// Non-numeric strings are rejected.
|
||||
assert!(!dl_strtoul_consumes_all("abc"));
|
||||
// "0x" with no hex digits: _dl_strtoul advances past "0x", the digit loop
|
||||
// breaks immediately (no hex digit), endptr reaches the end, value is 0.
|
||||
// tunable_parse_num accepts it.
|
||||
assert!(dl_strtoul_consumes_all("0x"));
|
||||
assert!(dl_strtoul_consumes_all("0X"));
|
||||
|
||||
// Overflow: _dl_strtoul stops at the overflowing digit (endptr points
|
||||
// there, not at the end), so tunable_parse_num rejects the value.
|
||||
assert!(!dl_strtoul_consumes_all("18446744073709551616")); // u64::MAX + 1
|
||||
assert!(!dl_strtoul_consumes_all("99999999999999999999")); // 20 nines
|
||||
assert!(!dl_strtoul_consumes_all("0x10000000000000000")); // 2^64
|
||||
// u64::MAX itself is accepted: the last digit (5) equals cutlim (=5),
|
||||
// so the overflow check (digval > cutlim) is false.
|
||||
assert!(dl_strtoul_consumes_all("18446744073709551615")); // u64::MAX
|
||||
}
|
||||
|
||||
#[cfg(all(target_os = "linux", target_env = "gnu"))]
|
||||
#[test]
|
||||
fn usable_glibc_tunable_threshold_detects_mmap_threshold() {
|
||||
// Decimal, hex, octal, negative, and zero values are all accepted by
|
||||
// glibc's _dl_strtoul and cause the threshold to be pinned.
|
||||
assert!(usable_glibc_tunable_threshold(Some(
|
||||
"glibc.malloc.mmap_threshold=131072".into()
|
||||
)));
|
||||
assert!(usable_glibc_tunable_threshold(Some(
|
||||
"glibc.malloc.mmap_threshold=0x20000".into()
|
||||
)));
|
||||
assert!(usable_glibc_tunable_threshold(Some(
|
||||
"glibc.malloc.mmap_threshold=0".into()
|
||||
)));
|
||||
assert!(usable_glibc_tunable_threshold(Some(
|
||||
"glibc.malloc.mmap_threshold=-1".into()
|
||||
)));
|
||||
assert!(usable_glibc_tunable_threshold(Some(
|
||||
"glibc.malloc.mmap_threshold=9223372036854775808".into()
|
||||
)));
|
||||
// u64::MAX is accepted by _dl_strtoul (last digit == cutlim, no overflow).
|
||||
assert!(usable_glibc_tunable_threshold(Some(
|
||||
"glibc.malloc.mmap_threshold=18446744073709551615".into()
|
||||
)));
|
||||
// Appears alongside other tunables.
|
||||
assert!(usable_glibc_tunable_threshold(Some(
|
||||
"glibc.cpu.x=1:glibc.malloc.mmap_threshold=131072".into()
|
||||
)));
|
||||
// Leading ':' is accepted — glibc skips the empty entry and continues.
|
||||
assert!(usable_glibc_tunable_threshold(Some(
|
||||
":glibc.malloc.mmap_threshold=131072".into()
|
||||
)));
|
||||
// Empty value is accepted by _dl_strtoul (value 0).
|
||||
assert!(usable_glibc_tunable_threshold(Some(
|
||||
"glibc.malloc.mmap_threshold=".into()
|
||||
)));
|
||||
|
||||
// Non-numeric values are rejected by _dl_strtoul.
|
||||
assert!(!usable_glibc_tunable_threshold(Some(
|
||||
"glibc.malloc.mmap_threshold=abc".into()
|
||||
)));
|
||||
assert!(!usable_glibc_tunable_threshold(Some(
|
||||
"glibc.malloc.mmap_threshold=128K".into()
|
||||
)));
|
||||
// A malformed sibling entry (duplicate '=') makes glibc reject the
|
||||
// entire string, so we must not accept the threshold entry either.
|
||||
// This applies regardless of whether the threshold is before or after
|
||||
// the malformed entry — parse_tunables_string returns -1, and
|
||||
// parse_tunables discards all tunables without applying any.
|
||||
assert!(!usable_glibc_tunable_threshold(Some(
|
||||
"glibc.malloc.check=2=2:glibc.malloc.mmap_threshold=131072".into()
|
||||
)));
|
||||
assert!(!usable_glibc_tunable_threshold(Some(
|
||||
"glibc.malloc.mmap_threshold=262144:glibc.malloc.check=2=2".into()
|
||||
)));
|
||||
// A trailing entry with no '=' makes glibc reject the entire string
|
||||
// (parse_tunables_string hits '\0' before '=' and returns -1).
|
||||
assert!(!usable_glibc_tunable_threshold(Some(
|
||||
"glibc.malloc.mmap_threshold=262144:glibc.cpu.x".into()
|
||||
)));
|
||||
// A trailing ':' makes glibc reject the entire string (the empty entry
|
||||
// after ':' hits '\0' before '=' and returns -1).
|
||||
assert!(!usable_glibc_tunable_threshold(Some(
|
||||
"glibc.malloc.mmap_threshold=262144:".into()
|
||||
)));
|
||||
// Unrelated tunables do not count.
|
||||
assert!(!usable_glibc_tunable_threshold(Some(
|
||||
"glibc.cpu.x=1".into()
|
||||
)));
|
||||
assert!(!usable_glibc_tunable_threshold(None));
|
||||
}
|
||||
}
|
||||
+233
-116
@@ -3,10 +3,10 @@
|
||||
//! Mirrors the Go SeaweedFS volume server metrics.
|
||||
|
||||
use prometheus::{
|
||||
self, Encoder, GaugeVec, HistogramOpts, HistogramVec, IntCounterVec, IntGauge, IntGaugeVec,
|
||||
Opts, Registry, TextEncoder,
|
||||
self, Encoder, GaugeVec, HistogramOpts, HistogramVec, IntCounter, IntCounterVec, IntGauge,
|
||||
IntGaugeVec, Opts, Registry, TextEncoder,
|
||||
};
|
||||
use std::sync::Once;
|
||||
use std::sync::{LazyLock, Once};
|
||||
|
||||
use crate::version;
|
||||
|
||||
@@ -16,203 +16,320 @@ pub struct PushGatewayConfig {
|
||||
pub interval_seconds: u32,
|
||||
}
|
||||
|
||||
lazy_static::lazy_static! {
|
||||
pub static ref REGISTRY: Registry = Registry::new();
|
||||
pub static REGISTRY: LazyLock<Registry> = LazyLock::new(Registry::new);
|
||||
|
||||
// ---- Request metrics (Go: VolumeServerRequestCounter, VolumeServerRequestHistogram) ----
|
||||
// ---- Request metrics (Go: VolumeServerRequestCounter, VolumeServerRequestHistogram) ----
|
||||
|
||||
/// Request counter with labels `type` (HTTP method) and `code` (HTTP status).
|
||||
pub static ref REQUEST_COUNTER: IntCounterVec = IntCounterVec::new(
|
||||
Opts::new("SeaweedFS_volumeServer_request_total", "Volume server requests"),
|
||||
/// Request counter with labels `type` (HTTP method) and `code` (HTTP status).
|
||||
pub static REQUEST_COUNTER: LazyLock<IntCounterVec> = LazyLock::new(|| {
|
||||
IntCounterVec::new(
|
||||
Opts::new(
|
||||
"SeaweedFS_volumeServer_request_total",
|
||||
"Volume server requests",
|
||||
),
|
||||
&["type", "code"],
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Request duration histogram with label `type` (HTTP method).
|
||||
pub static ref REQUEST_DURATION: HistogramVec = HistogramVec::new(
|
||||
/// Request duration histogram with label `type` (HTTP method).
|
||||
pub static REQUEST_DURATION: LazyLock<HistogramVec> = LazyLock::new(|| {
|
||||
HistogramVec::new(
|
||||
HistogramOpts::new(
|
||||
"SeaweedFS_volumeServer_request_seconds",
|
||||
"Volume server request duration in seconds",
|
||||
).buckets(exponential_buckets(0.0001, 2.0, 24)),
|
||||
)
|
||||
.buckets(exponential_buckets(0.0001, 2.0, 24)),
|
||||
&["type"],
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
// ---- Handler counters (Go: VolumeServerHandlerCounter) ----
|
||||
// ---- Handler counters (Go: VolumeServerHandlerCounter) ----
|
||||
|
||||
/// Handler-level operation counter with label `type`.
|
||||
pub static ref HANDLER_COUNTER: IntCounterVec = IntCounterVec::new(
|
||||
Opts::new("SeaweedFS_volumeServer_handler_total", "Volume server handler counters"),
|
||||
/// Handler-level operation counter with label `type`.
|
||||
pub static HANDLER_COUNTER: LazyLock<IntCounterVec> = LazyLock::new(|| {
|
||||
IntCounterVec::new(
|
||||
Opts::new(
|
||||
"SeaweedFS_volumeServer_handler_total",
|
||||
"Volume server handler counters",
|
||||
),
|
||||
&["type"],
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
// ---- Vacuuming metrics (Go: VolumeServerVacuuming*) ----
|
||||
// ---- Vacuuming metrics (Go: VolumeServerVacuuming*) ----
|
||||
|
||||
/// Vacuuming compact counter with label `success` (true/false).
|
||||
pub static ref VACUUMING_COMPACT_COUNTER: IntCounterVec = IntCounterVec::new(
|
||||
Opts::new("SeaweedFS_volumeServer_vacuuming_compact_count", "Counter of volume vacuuming Compact counter"),
|
||||
/// Vacuuming compact counter with label `success` (true/false).
|
||||
pub static VACUUMING_COMPACT_COUNTER: LazyLock<IntCounterVec> = LazyLock::new(|| {
|
||||
IntCounterVec::new(
|
||||
Opts::new(
|
||||
"SeaweedFS_volumeServer_vacuuming_compact_count",
|
||||
"Counter of volume vacuuming Compact counter",
|
||||
),
|
||||
&["success"],
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Vacuuming commit counter with label `success` (true/false).
|
||||
pub static ref VACUUMING_COMMIT_COUNTER: IntCounterVec = IntCounterVec::new(
|
||||
Opts::new("SeaweedFS_volumeServer_vacuuming_commit_count", "Counter of volume vacuuming commit counter"),
|
||||
/// Vacuuming commit counter with label `success` (true/false).
|
||||
pub static VACUUMING_COMMIT_COUNTER: LazyLock<IntCounterVec> = LazyLock::new(|| {
|
||||
IntCounterVec::new(
|
||||
Opts::new(
|
||||
"SeaweedFS_volumeServer_vacuuming_commit_count",
|
||||
"Counter of volume vacuuming commit counter",
|
||||
),
|
||||
&["success"],
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Vacuuming duration histogram with label `type` (compact/commit).
|
||||
pub static ref VACUUMING_HISTOGRAM: HistogramVec = HistogramVec::new(
|
||||
/// Vacuuming duration histogram with label `type` (compact/commit).
|
||||
pub static VACUUMING_HISTOGRAM: LazyLock<HistogramVec> = LazyLock::new(|| {
|
||||
HistogramVec::new(
|
||||
HistogramOpts::new(
|
||||
"SeaweedFS_volumeServer_vacuuming_seconds",
|
||||
"Volume vacuuming duration in seconds",
|
||||
).buckets(exponential_buckets(0.0001, 2.0, 24)),
|
||||
)
|
||||
.buckets(exponential_buckets(0.0001, 2.0, 24)),
|
||||
&["type"],
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
// ---- Volume gauges (Go: VolumeServerVolumeGauge, VolumeServerReadOnlyVolumeGauge) ----
|
||||
// ---- Volume gauges (Go: VolumeServerVolumeGauge, VolumeServerReadOnlyVolumeGauge) ----
|
||||
|
||||
/// Volumes per collection and type (volume/ec_shards).
|
||||
pub static ref VOLUME_GAUGE: GaugeVec = GaugeVec::new(
|
||||
/// Volumes per collection and type (volume/ec_shards).
|
||||
pub static VOLUME_GAUGE: LazyLock<GaugeVec> = LazyLock::new(|| {
|
||||
GaugeVec::new(
|
||||
Opts::new("SeaweedFS_volumeServer_volumes", "Number of volumes"),
|
||||
&["collection", "type"],
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Read-only volumes per collection and type.
|
||||
pub static ref READ_ONLY_VOLUME_GAUGE: GaugeVec = GaugeVec::new(
|
||||
Opts::new("SeaweedFS_volumeServer_read_only_volumes", "Number of read-only volumes."),
|
||||
/// Read-only volumes per collection and type.
|
||||
pub static READ_ONLY_VOLUME_GAUGE: LazyLock<GaugeVec> = LazyLock::new(|| {
|
||||
GaugeVec::new(
|
||||
Opts::new(
|
||||
"SeaweedFS_volumeServer_read_only_volumes",
|
||||
"Number of read-only volumes.",
|
||||
),
|
||||
&["collection", "type"],
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Maximum number of volumes this server can hold.
|
||||
pub static ref MAX_VOLUMES: IntGauge = IntGauge::new(
|
||||
/// Maximum number of volumes this server can hold.
|
||||
pub static MAX_VOLUMES: LazyLock<IntGauge> = LazyLock::new(|| {
|
||||
IntGauge::new(
|
||||
"SeaweedFS_volumeServer_max_volumes",
|
||||
"Maximum number of volumes",
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
// ---- Disk size gauges (Go: VolumeServerDiskSizeGauge) ----
|
||||
// ---- Disk size gauges (Go: VolumeServerDiskSizeGauge) ----
|
||||
|
||||
/// Actual disk size used by volumes per collection and type (normal/deleted_bytes/ec).
|
||||
pub static ref DISK_SIZE_GAUGE: GaugeVec = GaugeVec::new(
|
||||
Opts::new("SeaweedFS_volumeServer_total_disk_size", "Actual disk size used by volumes"),
|
||||
/// Actual disk size used by volumes per collection and type (normal/deleted_bytes/ec).
|
||||
pub static DISK_SIZE_GAUGE: LazyLock<GaugeVec> = LazyLock::new(|| {
|
||||
GaugeVec::new(
|
||||
Opts::new(
|
||||
"SeaweedFS_volumeServer_total_disk_size",
|
||||
"Actual disk size used by volumes",
|
||||
),
|
||||
&["collection", "type"],
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
// ---- Resource gauges (Go: VolumeServerResourceGauge) ----
|
||||
// ---- Resource gauges (Go: VolumeServerResourceGauge) ----
|
||||
|
||||
/// Disk resource usage per directory and type (all/used/free/avail).
|
||||
pub static ref RESOURCE_GAUGE: GaugeVec = GaugeVec::new(
|
||||
/// Disk resource usage per directory and type (all/used/free/avail).
|
||||
pub static RESOURCE_GAUGE: LazyLock<GaugeVec> = LazyLock::new(|| {
|
||||
GaugeVec::new(
|
||||
Opts::new("SeaweedFS_volumeServer_resource", "Server resource usage"),
|
||||
&["name", "type"],
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
// ---- In-flight gauges (Go: VolumeServerInFlightRequestsGauge, InFlightDownload/UploadSize) ----
|
||||
// ---- In-flight gauges (Go: VolumeServerInFlightRequestsGauge, InFlightDownload/UploadSize) ----
|
||||
|
||||
/// In-flight requests per HTTP method.
|
||||
pub static ref INFLIGHT_REQUESTS_GAUGE: IntGaugeVec = IntGaugeVec::new(
|
||||
Opts::new("SeaweedFS_volumeServer_in_flight_requests", "Current number of in-flight requests being handled by volume server."),
|
||||
/// In-flight requests per HTTP method.
|
||||
pub static INFLIGHT_REQUESTS_GAUGE: LazyLock<IntGaugeVec> = LazyLock::new(|| {
|
||||
IntGaugeVec::new(
|
||||
Opts::new(
|
||||
"SeaweedFS_volumeServer_in_flight_requests",
|
||||
"Current number of in-flight requests being handled by volume server.",
|
||||
),
|
||||
&["type"],
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Concurrent download limit in bytes.
|
||||
pub static ref CONCURRENT_DOWNLOAD_LIMIT: IntGauge = IntGauge::new(
|
||||
/// Concurrent download limit in bytes.
|
||||
pub static CONCURRENT_DOWNLOAD_LIMIT: LazyLock<IntGauge> = LazyLock::new(|| {
|
||||
IntGauge::new(
|
||||
"SeaweedFS_volumeServer_concurrent_download_limit",
|
||||
"Limit for total concurrent download size in bytes",
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Concurrent upload limit in bytes.
|
||||
pub static ref CONCURRENT_UPLOAD_LIMIT: IntGauge = IntGauge::new(
|
||||
/// Concurrent upload limit in bytes.
|
||||
pub static CONCURRENT_UPLOAD_LIMIT: LazyLock<IntGauge> = LazyLock::new(|| {
|
||||
IntGauge::new(
|
||||
"SeaweedFS_volumeServer_concurrent_upload_limit",
|
||||
"Limit for total concurrent upload size in bytes",
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Current in-flight download bytes.
|
||||
pub static ref INFLIGHT_DOWNLOAD_SIZE: IntGauge = IntGauge::new(
|
||||
/// Current in-flight download bytes.
|
||||
pub static INFLIGHT_DOWNLOAD_SIZE: LazyLock<IntGauge> = LazyLock::new(|| {
|
||||
IntGauge::new(
|
||||
"SeaweedFS_volumeServer_in_flight_download_size",
|
||||
"In flight total download size.",
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Current in-flight upload bytes.
|
||||
pub static ref INFLIGHT_UPLOAD_SIZE: IntGauge = IntGauge::new(
|
||||
/// Current in-flight upload bytes.
|
||||
pub static INFLIGHT_UPLOAD_SIZE: LazyLock<IntGauge> = LazyLock::new(|| {
|
||||
IntGauge::new(
|
||||
"SeaweedFS_volumeServer_in_flight_upload_size",
|
||||
"In flight total upload size.",
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Upload error counter by HTTP status code. Code "0" = transport error (no response).
|
||||
pub static ref UPLOAD_ERROR_COUNTER: IntCounterVec = IntCounterVec::new(
|
||||
Opts::new("SeaweedFS_upload_error_total",
|
||||
"Counter of upload errors by HTTP status code. Code 0 means transport error (no response received)."),
|
||||
&["code"],
|
||||
).expect("metric can be created");
|
||||
/// Upload error counter by HTTP status code. Code "0" = transport error (no response).
|
||||
pub static UPLOAD_ERROR_COUNTER: LazyLock<IntCounterVec> = LazyLock::new(|| {
|
||||
IntCounterVec::new(
|
||||
Opts::new("SeaweedFS_upload_error_total",
|
||||
"Counter of upload errors by HTTP status code. Code 0 means transport error (no response received)."),
|
||||
&["code"],
|
||||
).expect("metric can be created")
|
||||
});
|
||||
|
||||
// ---- Scrubbing metrics (Go: VolumeServerScrub*) ----
|
||||
// ---- Scrubbing metrics (Go: VolumeServerScrub*) ----
|
||||
|
||||
/// Last scrub execution time, as seconds since UNIX epoch, with label `mode`.
|
||||
pub static ref SCRUB_LAST_TIME_SECONDS: GaugeVec = GaugeVec::new(
|
||||
/// Last scrub execution time, as seconds since UNIX epoch, with label `mode`.
|
||||
pub static SCRUB_LAST_TIME_SECONDS: LazyLock<GaugeVec> = LazyLock::new(|| {
|
||||
GaugeVec::new(
|
||||
Opts::new(
|
||||
"SeaweedFS_volumeServer_scrub_last_time_seconds",
|
||||
"Last scrub execution time, as seconds since UNIX epoch.",
|
||||
),
|
||||
&["mode"],
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Counter of overall volumes with issues detected during scrubbing, with label `mode`.
|
||||
pub static ref SCRUB_VOLUME_FAILURES: IntCounterVec = IntCounterVec::new(
|
||||
/// Counter of overall volumes with issues detected during scrubbing, with label `mode`.
|
||||
pub static SCRUB_VOLUME_FAILURES: LazyLock<IntCounterVec> = LazyLock::new(|| {
|
||||
IntCounterVec::new(
|
||||
Opts::new(
|
||||
"SeaweedFS_volumeServer_scrub_volume_failures",
|
||||
"Counter of overall volumes with issues detected during scrubbing.",
|
||||
),
|
||||
&["mode"],
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Counter of overall EC shards with issues detected during scrubbing, with label `mode`.
|
||||
pub static ref SCRUB_SHARD_FAILURES: IntCounterVec = IntCounterVec::new(
|
||||
/// Counter of overall EC shards with issues detected during scrubbing, with label `mode`.
|
||||
pub static SCRUB_SHARD_FAILURES: LazyLock<IntCounterVec> = LazyLock::new(|| {
|
||||
IntCounterVec::new(
|
||||
Opts::new(
|
||||
"SeaweedFS_volumeServer_scrub_shard_failures",
|
||||
"Counter of overall EC shards with issues detected during scrubbing.",
|
||||
),
|
||||
&["mode"],
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
// ---- Legacy aliases for backward compat with existing code ----
|
||||
/// Counter of storage read/write EIO errors on volumes and EC shards.
|
||||
/// Mirrors Go's VolumeServerStorageIoErrorCounter.
|
||||
pub static STORAGE_IO_ERROR_COUNTER: LazyLock<IntCounter> = LazyLock::new(|| {
|
||||
IntCounter::new(
|
||||
"SeaweedFS_volumeServer_storage_io_error_total",
|
||||
"Counter of storage read/write EIO errors on volumes and EC shards.",
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Total number of volumes on this server (flat gauge).
|
||||
pub static ref VOLUMES_TOTAL: IntGauge = IntGauge::new(
|
||||
"volume_server_volumes_total",
|
||||
"Total number of volumes",
|
||||
).expect("metric can be created");
|
||||
/// Number of volumes quarantined due to storage IO errors.
|
||||
/// Mirrors Go's VolumeServerIoQuarantineGauge.
|
||||
pub static IO_QUARANTINE_GAUGE: LazyLock<IntGaugeVec> = LazyLock::new(|| {
|
||||
IntGaugeVec::new(
|
||||
Opts::new(
|
||||
"SeaweedFS_volumeServer_io_quarantine",
|
||||
"Number of volumes or EC shards quarantined due to storage IO errors.",
|
||||
),
|
||||
&["kind"],
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Disk size in bytes per directory.
|
||||
pub static ref DISK_SIZE_BYTES: IntGaugeVec = IntGaugeVec::new(
|
||||
// ---- Legacy aliases for backward compat with existing code ----
|
||||
|
||||
/// Total number of volumes on this server (flat gauge).
|
||||
pub static VOLUMES_TOTAL: LazyLock<IntGauge> = LazyLock::new(|| {
|
||||
IntGauge::new("volume_server_volumes_total", "Total number of volumes")
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Disk size in bytes per directory.
|
||||
pub static DISK_SIZE_BYTES: LazyLock<IntGaugeVec> = LazyLock::new(|| {
|
||||
IntGaugeVec::new(
|
||||
Opts::new("volume_server_disk_size_bytes", "Disk size in bytes"),
|
||||
&["dir"],
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Disk free bytes per directory.
|
||||
pub static ref DISK_FREE_BYTES: IntGaugeVec = IntGaugeVec::new(
|
||||
/// Disk free bytes per directory.
|
||||
pub static DISK_FREE_BYTES: LazyLock<IntGaugeVec> = LazyLock::new(|| {
|
||||
IntGaugeVec::new(
|
||||
Opts::new("volume_server_disk_free_bytes", "Disk free space in bytes"),
|
||||
&["dir"],
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Current number of in-flight requests (flat gauge).
|
||||
pub static ref INFLIGHT_REQUESTS: IntGauge = IntGauge::new(
|
||||
/// Current number of in-flight requests (flat gauge).
|
||||
pub static INFLIGHT_REQUESTS: LazyLock<IntGauge> = LazyLock::new(|| {
|
||||
IntGauge::new(
|
||||
"volume_server_inflight_requests",
|
||||
"Current number of in-flight requests",
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Total number of files stored across all volumes.
|
||||
pub static ref VOLUME_FILE_COUNT: IntGauge = IntGauge::new(
|
||||
/// Total number of files stored across all volumes.
|
||||
pub static VOLUME_FILE_COUNT: LazyLock<IntGauge> = LazyLock::new(|| {
|
||||
IntGauge::new(
|
||||
"volume_server_volume_file_count",
|
||||
"Total number of files stored across all volumes",
|
||||
).expect("metric can be created");
|
||||
)
|
||||
.expect("metric can be created")
|
||||
});
|
||||
|
||||
// ---- Build info (Go: BuildInfo) ----
|
||||
// ---- Build info (Go: BuildInfo) ----
|
||||
|
||||
/// Build information gauge, always set to 1. Matches Go:
|
||||
/// Namespace="SeaweedFS", Subsystem="build", Name="info",
|
||||
/// labels: version, commit, sizelimit, goos, goarch.
|
||||
pub static ref BUILD_INFO: GaugeVec = GaugeVec::new(
|
||||
Opts::new("SeaweedFS_build_info", "A metric with a constant '1' value labeled by version, commit, sizelimit, goos, and goarch from which SeaweedFS was built."),
|
||||
&["version", "commit", "sizelimit", "goos", "goarch"],
|
||||
).expect("metric can be created");
|
||||
}
|
||||
/// Build information gauge, always set to 1. Matches Go:
|
||||
/// Namespace="SeaweedFS", Subsystem="build", Name="info",
|
||||
/// labels: version, commit, sizelimit, goos, goarch.
|
||||
pub static BUILD_INFO: LazyLock<GaugeVec> = LazyLock::new(|| {
|
||||
GaugeVec::new(
|
||||
Opts::new("SeaweedFS_build_info", "A metric with a constant '1' value labeled by version, commit, sizelimit, goos, and goarch from which SeaweedFS was built."),
|
||||
&["version", "commit", "sizelimit", "goos", "goarch"],
|
||||
).expect("metric can be created")
|
||||
});
|
||||
|
||||
/// Generate exponential bucket boundaries for histograms.
|
||||
fn exponential_buckets(start: f64, factor: f64, count: usize) -> Vec<f64> {
|
||||
@@ -283,6 +400,8 @@ pub fn register_metrics() {
|
||||
Box::new(SCRUB_LAST_TIME_SECONDS.clone()),
|
||||
Box::new(SCRUB_VOLUME_FAILURES.clone()),
|
||||
Box::new(SCRUB_SHARD_FAILURES.clone()),
|
||||
Box::new(STORAGE_IO_ERROR_COUNTER.clone()),
|
||||
Box::new(IO_QUARANTINE_GAUGE.clone()),
|
||||
// Legacy metrics
|
||||
Box::new(VOLUMES_TOTAL.clone()),
|
||||
Box::new(DISK_SIZE_BYTES.clone()),
|
||||
@@ -358,10 +477,8 @@ fn delete_partial_match_collection(gauge: &GaugeVec, collection: &str) {
|
||||
type_value = Some(label.get_value().to_string());
|
||||
}
|
||||
}
|
||||
if matches_collection {
|
||||
if let Some(ref tv) = type_value {
|
||||
let _ = gauge.remove_label_values(&[collection, tv]);
|
||||
}
|
||||
if matches_collection && let Some(ref tv) = type_value {
|
||||
let _ = gauge.remove_label_values(&[collection, tv]);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -173,10 +173,10 @@ pub fn check_blocked_ip_policy(endpoint: &str, ip: IpAddr, allow_private: bool)
|
||||
// same host wherever the matching relay exists (common in IPv6-only cloud).
|
||||
// to_ipv4_mapped above only covers ::ffff: mapped addresses, so pull the
|
||||
// embedded IPv4 out of the other forms and re-check it against the rules.
|
||||
if let IpAddr::V6(v6) = ip {
|
||||
if let Some(v4) = embedded_transition_ipv4(v6) {
|
||||
return check_blocked_ip_policy(endpoint, IpAddr::V4(v4), allow_private);
|
||||
}
|
||||
if let IpAddr::V6(v6) = ip
|
||||
&& let Some(v4) = embedded_transition_ipv4(v6)
|
||||
{
|
||||
return check_blocked_ip_policy(endpoint, IpAddr::V4(v4), allow_private);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
@@ -214,9 +214,7 @@ fn precheck_endpoint(endpoint: &str) -> Result<HostCheck, String> {
|
||||
|
||||
// Authority is everything up to the first '/', '?', or '#'.
|
||||
let after = &trimmed[scheme_end + 3..];
|
||||
let authority_end = after
|
||||
.find(|c| c == '/' || c == '?' || c == '#')
|
||||
.unwrap_or(after.len());
|
||||
let authority_end = after.find(['/', '?', '#']).unwrap_or(after.len());
|
||||
let authority = &after[..authority_end];
|
||||
|
||||
// Strip optional userinfo ("user:pass@").
|
||||
|
||||
@@ -297,10 +297,10 @@ impl Guard {
|
||||
/// Extract host from "host:port" or "[::1]:port" format.
|
||||
fn extract_host(addr: &str) -> String {
|
||||
// Handle IPv6 with brackets
|
||||
if addr.starts_with('[') {
|
||||
if let Some(end) = addr.find(']') {
|
||||
return addr[1..end].to_string();
|
||||
}
|
||||
if addr.starts_with('[')
|
||||
&& let Some(end) = addr.find(']')
|
||||
{
|
||||
return addr[1..end].to_string();
|
||||
}
|
||||
// Handle host:port
|
||||
if let Some(pos) = addr.rfind(':') {
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -193,10 +193,7 @@ impl http_body::Body for StreamingBody {
|
||||
}
|
||||
Ok(Err(e)) => return std::task::Poll::Ready(Some(Err(e))),
|
||||
Err(e) => {
|
||||
return std::task::Poll::Ready(Some(Err(std::io::Error::new(
|
||||
std::io::ErrorKind::Other,
|
||||
e,
|
||||
))))
|
||||
return std::task::Poll::Ready(Some(Err(std::io::Error::other(e))));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -322,10 +319,9 @@ fn parse_url_path(path: &str) -> Option<(VolumeId, NeedleId, Cookie)> {
|
||||
// Try "vid,fid" or "vid/fid" or "vid/fid/filename" formats
|
||||
let (vid_str, fid_part) = if let Some(pos) = path.find(',') {
|
||||
(&path[..pos], &path[pos + 1..])
|
||||
} else if let Some(pos) = path.find('/') {
|
||||
(&path[..pos], &path[pos + 1..])
|
||||
} else {
|
||||
return None;
|
||||
let pos = path.find('/')?;
|
||||
(&path[..pos], &path[pos + 1..])
|
||||
};
|
||||
|
||||
// For fid part, strip extension from the fid (not from filename)
|
||||
@@ -409,10 +405,10 @@ async fn lookup_volume(
|
||||
.json()
|
||||
.await
|
||||
.map_err(|e| format!("lookup parse failed: {}", e))?;
|
||||
if let Some(err) = result.error {
|
||||
if !err.is_empty() {
|
||||
return Err(err);
|
||||
}
|
||||
if let Some(err) = result.error
|
||||
&& !err.is_empty()
|
||||
{
|
||||
return Err(err);
|
||||
}
|
||||
Ok(result.locations.unwrap_or_default())
|
||||
}
|
||||
@@ -688,7 +684,8 @@ fn build_proxy_request_info(
|
||||
raw_fid
|
||||
};
|
||||
(trimmed[..pos].to_string(), fid.to_string())
|
||||
} else if let Some(pos) = trimmed.find('/') {
|
||||
} else {
|
||||
let pos = trimmed.find('/')?;
|
||||
let after = &trimmed[pos + 1..];
|
||||
let fid_part = if let Some(slash) = after.find('/') {
|
||||
&after[..slash]
|
||||
@@ -696,8 +693,6 @@ fn build_proxy_request_info(
|
||||
after
|
||||
};
|
||||
(trimmed[..pos].to_string(), fid_part.to_string())
|
||||
} else {
|
||||
return None;
|
||||
};
|
||||
|
||||
Some(ProxyRequestInfo {
|
||||
@@ -837,12 +832,12 @@ fn redirect_request(info: &ProxyRequestInfo, target: &VolumeLocation, scheme: &s
|
||||
let mut query_params = Vec::new();
|
||||
if !info.original_query.is_empty() {
|
||||
for param in info.original_query.split('&') {
|
||||
if let Some((key, value)) = param.split_once('=') {
|
||||
if key == "collection" {
|
||||
query_params.push(format!("collection={}", value));
|
||||
}
|
||||
// Intentionally drop readDeleted and other params (Go parity)
|
||||
if let Some((key, value)) = param.split_once('=')
|
||||
&& key == "collection"
|
||||
{
|
||||
query_params.push(format!("collection={}", value));
|
||||
}
|
||||
// Intentionally drop readDeleted and other params (Go parity)
|
||||
}
|
||||
}
|
||||
query_params.push("proxied=true".to_string());
|
||||
@@ -852,7 +847,7 @@ fn redirect_request(info: &ProxyRequestInfo, target: &VolumeLocation, scheme: &s
|
||||
let target_http = to_http_address(&target.url);
|
||||
let raw_target = format!(
|
||||
"{}/{},{}?{}",
|
||||
target_http, &info.vid_str, &info.fid_str, query
|
||||
target_http, info.vid_str, info.fid_str, query
|
||||
);
|
||||
let location = match normalize_outgoing_http_url(scheme, &raw_target) {
|
||||
Ok(url) => url,
|
||||
@@ -954,12 +949,12 @@ async fn get_or_head_handler_inner(
|
||||
// so invalid paths with JWT enabled return 401, not 400.
|
||||
let file_id = extract_file_id(&path);
|
||||
let token = extract_jwt(&headers, request.uri());
|
||||
if let Err(_) =
|
||||
state
|
||||
.guard
|
||||
.read()
|
||||
.unwrap()
|
||||
.check_jwt_for_file(token.as_deref(), &file_id, false)
|
||||
if state
|
||||
.guard
|
||||
.read()
|
||||
.unwrap()
|
||||
.check_jwt_for_file(token.as_deref(), &file_id, false)
|
||||
.is_err()
|
||||
{
|
||||
let body = serde_json::json!({"error": "wrong jwt"});
|
||||
return Response::builder()
|
||||
@@ -1015,16 +1010,15 @@ async fn get_or_head_handler_inner(
|
||||
let should_try_replica =
|
||||
!query_string.contains("proxied=true") && !state.master_url.is_empty() && {
|
||||
let store = state.store.read().unwrap();
|
||||
store.find_volume(vid).map_or(false, |(_, vol)| {
|
||||
store.find_volume(vid).is_some_and(|(_, vol)| {
|
||||
vol.super_block.replica_placement.get_copy_count() > 1
|
||||
})
|
||||
};
|
||||
if should_try_replica {
|
||||
if let Some(info) =
|
||||
if should_try_replica
|
||||
&& let Some(info) =
|
||||
build_proxy_request_info(&path, request.headers(), &query_string)
|
||||
{
|
||||
return proxy_or_redirect_to_target(&state, info, vid, true).await;
|
||||
}
|
||||
{
|
||||
return proxy_or_redirect_to_target(&state, info, vid, true).await;
|
||||
}
|
||||
|
||||
// Blocking wait loop (Go's waitForDownloadSlot)
|
||||
@@ -1231,57 +1225,53 @@ async fn get_or_head_handler_inner(
|
||||
// Build Last-Modified header (RFC 1123 format) — must be done before conditional checks
|
||||
let last_modified_str = if n.last_modified > 0 {
|
||||
use chrono::{TimeZone, Utc};
|
||||
if let Some(dt) = Utc.timestamp_opt(n.last_modified as i64, 0).single() {
|
||||
Some(dt.format("%a, %d %b %Y %H:%M:%S GMT").to_string())
|
||||
} else {
|
||||
None
|
||||
}
|
||||
Utc.timestamp_opt(n.last_modified as i64, 0)
|
||||
.single()
|
||||
.map(|dt| dt.format("%a, %d %b %Y %H:%M:%S GMT").to_string())
|
||||
} else {
|
||||
None
|
||||
};
|
||||
|
||||
// Check If-Modified-Since FIRST (Go checks this before If-None-Match)
|
||||
if n.last_modified > 0 {
|
||||
if let Some(ims_header) = headers.get(header::IF_MODIFIED_SINCE) {
|
||||
if let Ok(ims_str) = ims_header.to_str() {
|
||||
// Parse HTTP date format: "Mon, 02 Jan 2006 15:04:05 GMT"
|
||||
if let Ok(ims_time) =
|
||||
chrono::NaiveDateTime::parse_from_str(ims_str, "%a, %d %b %Y %H:%M:%S GMT")
|
||||
{
|
||||
if (n.last_modified as i64) <= ims_time.and_utc().timestamp() {
|
||||
let mut resp = StatusCode::NOT_MODIFIED.into_response();
|
||||
if let Some(ref lm) = last_modified_str {
|
||||
resp.headers_mut()
|
||||
.insert(header::LAST_MODIFIED, lm.parse().unwrap());
|
||||
}
|
||||
// Go sets ETag AFTER the 304 return paths (L235), so 304 does NOT include ETag
|
||||
return resp;
|
||||
}
|
||||
}
|
||||
if n.last_modified > 0
|
||||
&& let Some(ims_header) = headers.get(header::IF_MODIFIED_SINCE)
|
||||
&& let Ok(ims_str) = ims_header.to_str()
|
||||
{
|
||||
// Parse HTTP date format: "Mon, 02 Jan 2006 15:04:05 GMT"
|
||||
if let Ok(ims_time) =
|
||||
chrono::NaiveDateTime::parse_from_str(ims_str, "%a, %d %b %Y %H:%M:%S GMT")
|
||||
&& (n.last_modified as i64) <= ims_time.and_utc().timestamp()
|
||||
{
|
||||
let mut resp = StatusCode::NOT_MODIFIED.into_response();
|
||||
if let Some(ref lm) = last_modified_str {
|
||||
resp.headers_mut()
|
||||
.insert(header::LAST_MODIFIED, lm.parse().unwrap());
|
||||
}
|
||||
// Go sets ETag AFTER the 304 return paths (L235), so 304 does NOT include ETag
|
||||
return resp;
|
||||
}
|
||||
}
|
||||
|
||||
// Check If-None-Match SECOND
|
||||
if let Some(if_none_match) = headers.get(header::IF_NONE_MATCH) {
|
||||
if let Ok(inm) = if_none_match.to_str() {
|
||||
if inm == etag {
|
||||
let mut resp = StatusCode::NOT_MODIFIED.into_response();
|
||||
if let Some(ref lm) = last_modified_str {
|
||||
resp.headers_mut()
|
||||
.insert(header::LAST_MODIFIED, lm.parse().unwrap());
|
||||
}
|
||||
// Go sets ETag AFTER the 304 return paths (L235), so 304 does NOT include ETag
|
||||
return resp;
|
||||
}
|
||||
if let Some(if_none_match) = headers.get(header::IF_NONE_MATCH)
|
||||
&& let Ok(inm) = if_none_match.to_str()
|
||||
&& inm == etag
|
||||
{
|
||||
let mut resp = StatusCode::NOT_MODIFIED.into_response();
|
||||
if let Some(ref lm) = last_modified_str {
|
||||
resp.headers_mut()
|
||||
.insert(header::LAST_MODIFIED, lm.parse().unwrap());
|
||||
}
|
||||
// Go sets ETag AFTER the 304 return paths (L235), so 304 does NOT include ETag
|
||||
return resp;
|
||||
}
|
||||
|
||||
// Chunk manifest expansion (needs full data) — after conditional checks, before response
|
||||
// Pass ETag so chunk manifest responses include it (matches Go: ETag is set on the
|
||||
// response writer before tryHandleChunkedFile runs).
|
||||
if n.is_chunk_manifest() && !bypass_cm {
|
||||
if let Some(resp) = try_expand_chunk_manifest(
|
||||
if n.is_chunk_manifest()
|
||||
&& !bypass_cm
|
||||
&& let Some(resp) = try_expand_chunk_manifest(
|
||||
&state,
|
||||
&n,
|
||||
&headers,
|
||||
@@ -1292,27 +1282,26 @@ async fn get_or_head_handler_inner(
|
||||
&last_modified_str,
|
||||
)
|
||||
.await
|
||||
{
|
||||
return resp;
|
||||
}
|
||||
// If manifest expansion fails (invalid JSON etc.), fall through to raw data
|
||||
{
|
||||
return resp;
|
||||
}
|
||||
// If manifest expansion fails (invalid JSON etc.), fall through to raw data
|
||||
|
||||
let mut response_headers = HeaderMap::new();
|
||||
response_headers.insert(header::ETAG, etag.parse().unwrap());
|
||||
|
||||
// H1: Emit pairs as response headers
|
||||
if n.has_pairs() && !n.pairs.is_empty() {
|
||||
if let Ok(pair_map) =
|
||||
if n.has_pairs()
|
||||
&& !n.pairs.is_empty()
|
||||
&& let Ok(pair_map) =
|
||||
serde_json::from_slice::<std::collections::HashMap<String, String>>(&n.pairs)
|
||||
{
|
||||
for (k, v) in &pair_map {
|
||||
if let (Ok(hname), Ok(hval)) = (
|
||||
axum::http::HeaderName::from_bytes(k.as_bytes()),
|
||||
axum::http::HeaderValue::from_str(v),
|
||||
) {
|
||||
response_headers.insert(hname, hval);
|
||||
}
|
||||
{
|
||||
for (k, v) in &pair_map {
|
||||
if let (Ok(hname), Ok(hval)) = (
|
||||
axum::http::HeaderName::from_bytes(k.as_bytes()),
|
||||
axum::http::HeaderValue::from_str(v),
|
||||
) {
|
||||
response_headers.insert(hname, hval);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1322,10 +1311,10 @@ async fn get_or_head_handler_inner(
|
||||
let mut ext = ext;
|
||||
if n.name_size > 0 && filename.is_empty() {
|
||||
filename = String::from_utf8_lossy(&n.name).to_string();
|
||||
if ext.is_empty() {
|
||||
if let Some(dot_pos) = filename.rfind('.') {
|
||||
ext = filename[dot_pos..].to_lowercase();
|
||||
}
|
||||
if ext.is_empty()
|
||||
&& let Some(dot_pos) = filename.rfind('.')
|
||||
{
|
||||
ext = filename[dot_pos..].to_lowercase();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1429,80 +1418,72 @@ async fn get_or_head_handler_inner(
|
||||
}
|
||||
|
||||
// ---- Streaming path: large uncompressed files ----
|
||||
if can_stream {
|
||||
if let Some(info) = stream_info {
|
||||
response_headers.insert(header::ACCEPT_RANGES, "bytes".parse().unwrap());
|
||||
response_headers.insert(
|
||||
header::CONTENT_LENGTH,
|
||||
info.data_size.to_string().parse().unwrap(),
|
||||
);
|
||||
if can_stream && let Some(info) = stream_info {
|
||||
response_headers.insert(header::ACCEPT_RANGES, "bytes".parse().unwrap());
|
||||
response_headers.insert(
|
||||
header::CONTENT_LENGTH,
|
||||
info.data_size.to_string().parse().unwrap(),
|
||||
);
|
||||
|
||||
let tracked_bytes = info.data_size as i64;
|
||||
let tracking_state = if download_guard.is_some() {
|
||||
let new_val = state
|
||||
.inflight_download_bytes
|
||||
.fetch_add(tracked_bytes, Ordering::Relaxed)
|
||||
+ tracked_bytes;
|
||||
metrics::INFLIGHT_DOWNLOAD_SIZE.set(new_val);
|
||||
Some(state.clone())
|
||||
} else {
|
||||
let tracked_bytes = info.data_size as i64;
|
||||
let tracking_state = if download_guard.is_some() {
|
||||
let new_val = state
|
||||
.inflight_download_bytes
|
||||
.fetch_add(tracked_bytes, Ordering::Relaxed)
|
||||
+ tracked_bytes;
|
||||
metrics::INFLIGHT_DOWNLOAD_SIZE.set(new_val);
|
||||
Some(state.clone())
|
||||
} else {
|
||||
None
|
||||
};
|
||||
|
||||
let streaming = StreamingBody {
|
||||
source: info.source,
|
||||
data_offset: info.data_file_offset,
|
||||
data_size: info.data_size,
|
||||
pos: 0,
|
||||
chunk_size: streaming_chunk_size(state.read_buffer_size_bytes, info.data_size as usize),
|
||||
_held_read_lease: if state.has_slow_read {
|
||||
None
|
||||
};
|
||||
} else {
|
||||
Some(info.data_file_access_control.read_lock())
|
||||
},
|
||||
data_file_access_control: info.data_file_access_control,
|
||||
hold_read_lock_for_stream: !state.has_slow_read,
|
||||
pending: None,
|
||||
state: tracking_state,
|
||||
tracked_bytes,
|
||||
server_state: state.clone(),
|
||||
volume_id: info.volume_id,
|
||||
needle_id: info.needle_id,
|
||||
compaction_revision: info.compaction_revision,
|
||||
};
|
||||
|
||||
let streaming = StreamingBody {
|
||||
source: info.source,
|
||||
data_offset: info.data_file_offset,
|
||||
data_size: info.data_size,
|
||||
pos: 0,
|
||||
chunk_size: streaming_chunk_size(
|
||||
state.read_buffer_size_bytes,
|
||||
info.data_size as usize,
|
||||
),
|
||||
_held_read_lease: if state.has_slow_read {
|
||||
None
|
||||
} else {
|
||||
Some(info.data_file_access_control.read_lock())
|
||||
},
|
||||
data_file_access_control: info.data_file_access_control,
|
||||
hold_read_lock_for_stream: !state.has_slow_read,
|
||||
pending: None,
|
||||
state: tracking_state,
|
||||
tracked_bytes,
|
||||
server_state: state.clone(),
|
||||
volume_id: info.volume_id,
|
||||
needle_id: info.needle_id,
|
||||
compaction_revision: info.compaction_revision,
|
||||
};
|
||||
|
||||
let body = Body::new(streaming);
|
||||
let mut resp = Response::new(body);
|
||||
*resp.status_mut() = StatusCode::OK;
|
||||
*resp.headers_mut() = response_headers;
|
||||
return resp;
|
||||
}
|
||||
let body = Body::new(streaming);
|
||||
let mut resp = Response::new(body);
|
||||
*resp.status_mut() = StatusCode::OK;
|
||||
*resp.headers_mut() = response_headers;
|
||||
return resp;
|
||||
}
|
||||
|
||||
if can_handle_head_from_meta {
|
||||
if let Some(info) = stream_info {
|
||||
response_headers.insert(
|
||||
header::CONTENT_LENGTH,
|
||||
info.data_size.to_string().parse().unwrap(),
|
||||
);
|
||||
return (StatusCode::OK, response_headers).into_response();
|
||||
}
|
||||
if can_handle_head_from_meta && let Some(info) = stream_info {
|
||||
response_headers.insert(
|
||||
header::CONTENT_LENGTH,
|
||||
info.data_size.to_string().parse().unwrap(),
|
||||
);
|
||||
return (StatusCode::OK, response_headers).into_response();
|
||||
}
|
||||
|
||||
if can_handle_range_from_source {
|
||||
if let (Some(range_header), Some(info)) = (headers.get(header::RANGE), stream_info) {
|
||||
if let Ok(range_str) = range_header.to_str() {
|
||||
return handle_range_request_from_source(
|
||||
range_str,
|
||||
info,
|
||||
response_headers,
|
||||
track_download.then(|| state.clone()),
|
||||
);
|
||||
}
|
||||
}
|
||||
if can_handle_range_from_source
|
||||
&& let (Some(range_header), Some(info)) = (headers.get(header::RANGE), stream_info)
|
||||
&& let Ok(range_str) = range_header.to_str()
|
||||
{
|
||||
return handle_range_request_from_source(
|
||||
range_str,
|
||||
info,
|
||||
response_headers,
|
||||
track_download.then(|| state.clone()),
|
||||
);
|
||||
}
|
||||
|
||||
// ---- Buffered path: small files, compressed, images, range requests ----
|
||||
@@ -1572,15 +1553,15 @@ async fn get_or_head_handler_inner(
|
||||
response_headers.insert(header::ACCEPT_RANGES, "bytes".parse().unwrap());
|
||||
|
||||
// Check Range header
|
||||
if let Some(range_header) = headers.get(header::RANGE) {
|
||||
if let Ok(range_str) = range_header.to_str() {
|
||||
return handle_range_request(
|
||||
range_str,
|
||||
&data,
|
||||
response_headers,
|
||||
track_download.then(|| state.clone()),
|
||||
);
|
||||
}
|
||||
if let Some(range_header) = headers.get(header::RANGE)
|
||||
&& let Ok(range_str) = range_header.to_str()
|
||||
{
|
||||
return handle_range_request(
|
||||
range_str,
|
||||
&data,
|
||||
response_headers,
|
||||
track_download.then(|| state.clone()),
|
||||
);
|
||||
}
|
||||
|
||||
if method == Method::HEAD {
|
||||
@@ -2006,7 +1987,7 @@ fn extract_extension_from_path(path: &str) -> String {
|
||||
if let Some(dot_pos) = filename.rfind('.') {
|
||||
return filename[dot_pos..].to_lowercase();
|
||||
}
|
||||
} else if parts.len() >= 1 {
|
||||
} else if !parts.is_empty() {
|
||||
// 2-segment path: /vid,fid.ext or /vid/fid.ext
|
||||
// Go's parseURLPath extracts ext from the full path for all formats
|
||||
let last = parts[parts.len() - 1];
|
||||
@@ -2129,7 +2110,7 @@ pub async fn post_handler(
|
||||
// Go's r.ParseForm() returns 400 on malformed query strings
|
||||
return json_error_with_query(
|
||||
StatusCode::BAD_REQUEST,
|
||||
&format!("form parse error: {}", e),
|
||||
format!("form parse error: {}", e),
|
||||
Some(&query),
|
||||
);
|
||||
}
|
||||
@@ -2145,11 +2126,12 @@ pub async fn post_handler(
|
||||
// JWT check for writes
|
||||
let file_id = extract_file_id(&path);
|
||||
let token = extract_jwt(&headers, request.uri());
|
||||
if let Err(_) = state
|
||||
if state
|
||||
.guard
|
||||
.read()
|
||||
.unwrap()
|
||||
.check_jwt_for_file(token.as_deref(), &file_id, true)
|
||||
.is_err()
|
||||
{
|
||||
return json_error_with_query(StatusCode::UNAUTHORIZED, "wrong jwt", Some(&query));
|
||||
}
|
||||
@@ -2279,11 +2261,8 @@ pub async fn post_handler(
|
||||
.split(';')
|
||||
.find_map(|part| {
|
||||
let part = part.trim();
|
||||
if let Some(val) = part.strip_prefix("boundary=") {
|
||||
Some(val.trim_matches('"').to_string())
|
||||
} else {
|
||||
None
|
||||
}
|
||||
part.strip_prefix("boundary=")
|
||||
.map(|val| val.trim_matches('"').to_string())
|
||||
})
|
||||
.unwrap_or_default();
|
||||
|
||||
@@ -2425,17 +2404,17 @@ pub async fn post_handler(
|
||||
} else {
|
||||
None
|
||||
};
|
||||
if let (Some(ref expected_md5), Some(ref actual_md5)) = (&content_md5, &original_content_md5) {
|
||||
if expected_md5 != actual_md5 {
|
||||
return json_error_with_query(
|
||||
StatusCode::BAD_REQUEST,
|
||||
format!(
|
||||
"Content-MD5 did not match md5 of file data expected [{}] received [{}] size {}",
|
||||
expected_md5, actual_md5, original_data_size
|
||||
),
|
||||
Some(&query),
|
||||
);
|
||||
}
|
||||
if let (Some(expected_md5), Some(actual_md5)) = (&content_md5, &original_content_md5)
|
||||
&& expected_md5 != actual_md5
|
||||
{
|
||||
return json_error_with_query(
|
||||
StatusCode::BAD_REQUEST,
|
||||
format!(
|
||||
"Content-MD5 did not match md5 of file data expected [{}] received [{}] size {}",
|
||||
expected_md5, actual_md5, original_data_size
|
||||
),
|
||||
Some(&query),
|
||||
);
|
||||
}
|
||||
|
||||
let now = std::time::SystemTime::now()
|
||||
@@ -2577,7 +2556,7 @@ pub async fn post_handler(
|
||||
cookie,
|
||||
data_size: final_data.len() as u32,
|
||||
data: final_data,
|
||||
last_modified: last_modified,
|
||||
last_modified,
|
||||
..Needle::default()
|
||||
};
|
||||
n.set_has_last_modified_date();
|
||||
@@ -2595,22 +2574,21 @@ pub async fn post_handler(
|
||||
}
|
||||
|
||||
// Set TTL on needle
|
||||
if let Some(ref t) = ttl {
|
||||
if !t.is_empty() {
|
||||
n.ttl = Some(*t);
|
||||
n.set_has_ttl();
|
||||
}
|
||||
if let Some(ref t) = ttl
|
||||
&& !t.is_empty()
|
||||
{
|
||||
n.ttl = Some(*t);
|
||||
n.set_has_ttl();
|
||||
}
|
||||
|
||||
// Set pairs on needle
|
||||
if !pair_map.is_empty() {
|
||||
if let Ok(pairs_json) = serde_json::to_vec(&pair_map) {
|
||||
if pairs_json.len() < 65536 {
|
||||
n.pairs_size = pairs_json.len() as u16;
|
||||
n.pairs = pairs_json;
|
||||
n.set_has_pairs();
|
||||
}
|
||||
}
|
||||
if !pair_map.is_empty()
|
||||
&& let Ok(pairs_json) = serde_json::to_vec(&pair_map)
|
||||
&& pairs_json.len() < 65536
|
||||
{
|
||||
n.pairs_size = pairs_json.len() as u16;
|
||||
n.pairs = pairs_json;
|
||||
n.set_has_pairs();
|
||||
}
|
||||
|
||||
// Set filename on needle (matches Go: if len(pu.FileName) < 256)
|
||||
@@ -2639,9 +2617,9 @@ pub async fn post_handler(
|
||||
if !is_replicate && write_result.is_ok() && !state.master_url.is_empty() {
|
||||
let needs_replication = {
|
||||
let store = state.store.read().unwrap();
|
||||
store.find_volume(vid).map_or(false, |(_, v)| {
|
||||
v.super_block.replica_placement.get_copy_count() > 1
|
||||
})
|
||||
store
|
||||
.find_volume(vid)
|
||||
.is_some_and(|(_, v)| v.super_block.replica_placement.get_copy_count() > 1)
|
||||
};
|
||||
if needs_replication {
|
||||
let state_clone = state.clone();
|
||||
@@ -2664,7 +2642,7 @@ pub async fn post_handler(
|
||||
let replication_result = replication
|
||||
.await
|
||||
.map_err(|e| format!("replication task failed: {}", e))
|
||||
.and_then(|result| result);
|
||||
.flatten();
|
||||
if let Err(e) = replication_result {
|
||||
tracing::error!("replicated write failed: {}", e);
|
||||
return json_error_with_query(
|
||||
@@ -2758,11 +2736,12 @@ pub async fn delete_handler(
|
||||
// JWT check for writes (deletes use write key)
|
||||
let file_id = extract_file_id(&path);
|
||||
let token = extract_jwt(&headers, request.uri());
|
||||
if let Err(_) = state
|
||||
if state
|
||||
.guard
|
||||
.read()
|
||||
.unwrap()
|
||||
.check_jwt_for_file(token.as_deref(), &file_id, true)
|
||||
.is_err()
|
||||
{
|
||||
return json_error_with_query(StatusCode::UNAUTHORIZED, "wrong jwt", Some(&del_query));
|
||||
}
|
||||
@@ -2793,14 +2772,14 @@ pub async fn delete_handler(
|
||||
let count = ec_needle.data_size as i64;
|
||||
// Step 3: Journal the delete
|
||||
let mut store = state.store.write().unwrap();
|
||||
if let Some(ecv) = store.find_ec_volume_mut(vid) {
|
||||
if let Err(e) = ecv.journal_delete(needle_id) {
|
||||
return json_error_with_query(
|
||||
StatusCode::INTERNAL_SERVER_ERROR,
|
||||
format!("Deletion Failed: {}", e),
|
||||
Some(&del_query),
|
||||
);
|
||||
}
|
||||
if let Some(ecv) = store.find_ec_volume_mut(vid)
|
||||
&& let Err(e) = ecv.journal_delete(needle_id)
|
||||
{
|
||||
return json_error_with_query(
|
||||
StatusCode::INTERNAL_SERVER_ERROR,
|
||||
format!("Deletion Failed: {}", e),
|
||||
Some(&del_query),
|
||||
);
|
||||
}
|
||||
let result = DeleteResult { size: count };
|
||||
return json_response_with_params(
|
||||
@@ -2946,12 +2925,12 @@ pub async fn delete_handler(
|
||||
if !is_replicate && delete_result.is_ok() && !state.master_url.is_empty() {
|
||||
let needs_replication = {
|
||||
let store = state.store.read().unwrap();
|
||||
store.find_volume(vid).map_or(false, |(_, v)| {
|
||||
v.super_block.replica_placement.get_copy_count() > 1
|
||||
})
|
||||
store
|
||||
.find_volume(vid)
|
||||
.is_some_and(|(_, v)| v.super_block.replica_placement.get_copy_count() > 1)
|
||||
};
|
||||
if needs_replication {
|
||||
if let Err(e) = do_replicated_request(
|
||||
if needs_replication
|
||||
&& let Err(e) = do_replicated_request(
|
||||
&state,
|
||||
vid.0,
|
||||
Method::DELETE,
|
||||
@@ -2961,14 +2940,13 @@ pub async fn delete_handler(
|
||||
None,
|
||||
)
|
||||
.await
|
||||
{
|
||||
tracing::error!("replicated delete failed: {}", e);
|
||||
return json_error_with_query(
|
||||
StatusCode::INTERNAL_SERVER_ERROR,
|
||||
format!("replication failed: {}", e),
|
||||
Some(&del_query),
|
||||
);
|
||||
}
|
||||
{
|
||||
tracing::error!("replicated delete failed: {}", e);
|
||||
return json_error_with_query(
|
||||
StatusCode::INTERNAL_SERVER_ERROR,
|
||||
format!("replication failed: {}", e),
|
||||
Some(&del_query),
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -3127,6 +3105,11 @@ pub async fn healthz_handler(State(state): State<Arc<VolumeServerState>>) -> Res
|
||||
if !state.is_heartbeating.load(Ordering::Relaxed) {
|
||||
return StatusCode::SERVICE_UNAVAILABLE.into_response();
|
||||
}
|
||||
// A server with quarantined local replicas has faulty storage media;
|
||||
// report degraded so a load balancer can drain it.
|
||||
if state.store.read().unwrap().has_io_quarantine() {
|
||||
return StatusCode::SERVICE_UNAVAILABLE.into_response();
|
||||
}
|
||||
StatusCode::OK.into_response()
|
||||
}
|
||||
|
||||
@@ -3229,7 +3212,6 @@ pub async fn ui_handler(State(state): State<Arc<VolumeServerState>>) -> Response
|
||||
// ============================================================================
|
||||
|
||||
#[derive(Deserialize)]
|
||||
#[allow(dead_code)]
|
||||
struct ChunkManifest {
|
||||
#[serde(default)]
|
||||
name: String,
|
||||
@@ -3249,6 +3231,7 @@ struct ChunkInfo {
|
||||
}
|
||||
|
||||
/// Try to expand a chunk manifest needle. Returns None if manifest can't be parsed.
|
||||
#[expect(clippy::too_many_arguments)]
|
||||
async fn try_expand_chunk_manifest(
|
||||
state: &Arc<VolumeServerState>,
|
||||
n: &Needle,
|
||||
@@ -3377,53 +3360,53 @@ async fn try_expand_chunk_manifest(
|
||||
response_headers.insert(header::ACCEPT_RANGES, "bytes".parse().unwrap());
|
||||
|
||||
// Last-Modified — Go sets this on the response writer before tryHandleChunkedFile
|
||||
if let Some(ref lm) = last_modified_str {
|
||||
if let Ok(hval) = lm.parse() {
|
||||
response_headers.insert(header::LAST_MODIFIED, hval);
|
||||
}
|
||||
if let Some(lm) = last_modified_str
|
||||
&& let Ok(hval) = lm.parse()
|
||||
{
|
||||
response_headers.insert(header::LAST_MODIFIED, hval);
|
||||
}
|
||||
|
||||
// Pairs — Go sets needle pairs on the response writer before tryHandleChunkedFile
|
||||
if n.has_pairs() && !n.pairs.is_empty() {
|
||||
if let Ok(pair_map) =
|
||||
if n.has_pairs()
|
||||
&& !n.pairs.is_empty()
|
||||
&& let Ok(pair_map) =
|
||||
serde_json::from_slice::<std::collections::HashMap<String, String>>(&n.pairs)
|
||||
{
|
||||
for (k, v) in &pair_map {
|
||||
if let (Ok(hname), Ok(hval)) = (
|
||||
axum::http::HeaderName::from_bytes(k.as_bytes()),
|
||||
axum::http::HeaderValue::from_str(v),
|
||||
) {
|
||||
response_headers.insert(hname, hval);
|
||||
}
|
||||
{
|
||||
for (k, v) in &pair_map {
|
||||
if let (Ok(hname), Ok(hval)) = (
|
||||
axum::http::HeaderName::from_bytes(k.as_bytes()),
|
||||
axum::http::HeaderValue::from_str(v),
|
||||
) {
|
||||
response_headers.insert(hname, hval);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// S3 response passthrough headers — Go sets these via AdjustPassthroughHeaders
|
||||
if let Some(ref cc) = query.response_cache_control {
|
||||
if let Ok(hval) = cc.parse() {
|
||||
response_headers.insert(header::CACHE_CONTROL, hval);
|
||||
}
|
||||
if let Some(ref cc) = query.response_cache_control
|
||||
&& let Ok(hval) = cc.parse()
|
||||
{
|
||||
response_headers.insert(header::CACHE_CONTROL, hval);
|
||||
}
|
||||
if let Some(ref ce) = query.response_content_encoding {
|
||||
if let Ok(hval) = ce.parse() {
|
||||
response_headers.insert(header::CONTENT_ENCODING, hval);
|
||||
}
|
||||
if let Some(ref ce) = query.response_content_encoding
|
||||
&& let Ok(hval) = ce.parse()
|
||||
{
|
||||
response_headers.insert(header::CONTENT_ENCODING, hval);
|
||||
}
|
||||
if let Some(ref exp) = query.response_expires {
|
||||
if let Ok(hval) = exp.parse() {
|
||||
response_headers.insert(header::EXPIRES, hval);
|
||||
}
|
||||
if let Some(ref exp) = query.response_expires
|
||||
&& let Ok(hval) = exp.parse()
|
||||
{
|
||||
response_headers.insert(header::EXPIRES, hval);
|
||||
}
|
||||
if let Some(ref cl) = query.response_content_language {
|
||||
if let Ok(hval) = cl.parse() {
|
||||
response_headers.insert("Content-Language", hval);
|
||||
}
|
||||
if let Some(ref cl) = query.response_content_language
|
||||
&& let Ok(hval) = cl.parse()
|
||||
{
|
||||
response_headers.insert("Content-Language", hval);
|
||||
}
|
||||
if let Some(ref cd) = query.response_content_disposition {
|
||||
if let Ok(hval) = cd.parse() {
|
||||
response_headers.insert(header::CONTENT_DISPOSITION, hval);
|
||||
}
|
||||
if let Some(ref cd) = query.response_content_disposition
|
||||
&& let Ok(hval) = cd.parse()
|
||||
{
|
||||
response_headers.insert(header::CONTENT_DISPOSITION, hval);
|
||||
}
|
||||
|
||||
// Content-Disposition
|
||||
@@ -3454,7 +3437,6 @@ async fn try_expand_chunk_manifest(
|
||||
} else {
|
||||
String::new()
|
||||
};
|
||||
let mut result = result;
|
||||
if is_image_crop_ext(&cm_ext) {
|
||||
result = maybe_crop_image(&result, &cm_ext, query);
|
||||
}
|
||||
@@ -3730,33 +3712,33 @@ fn extract_jwt(headers: &HeaderMap, uri: &axum::http::Uri) -> Option<String> {
|
||||
// 1. Check ?jwt= query parameter
|
||||
if let Some(query) = uri.query() {
|
||||
for pair in query.split('&') {
|
||||
if let Some(value) = pair.strip_prefix("jwt=") {
|
||||
if !value.is_empty() {
|
||||
return Some(value.to_string());
|
||||
}
|
||||
if let Some(value) = pair.strip_prefix("jwt=")
|
||||
&& !value.is_empty()
|
||||
{
|
||||
return Some(value.to_string());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// 2. Check Authorization: Bearer <token> (case-insensitive prefix)
|
||||
if let Some(auth) = headers.get(header::AUTHORIZATION) {
|
||||
if let Ok(auth_str) = auth.to_str() {
|
||||
if auth_str.len() > 7 && auth_str[..7].eq_ignore_ascii_case("bearer ") {
|
||||
return Some(auth_str[7..].to_string());
|
||||
}
|
||||
}
|
||||
if let Some(auth) = headers.get(header::AUTHORIZATION)
|
||||
&& let Ok(auth_str) = auth.to_str()
|
||||
&& auth_str.len() > 7
|
||||
&& auth_str[..7].eq_ignore_ascii_case("bearer ")
|
||||
{
|
||||
return Some(auth_str[7..].to_string());
|
||||
}
|
||||
|
||||
// 3. Check Cookie
|
||||
if let Some(cookie_header) = headers.get(header::COOKIE) {
|
||||
if let Ok(cookie_str) = cookie_header.to_str() {
|
||||
for cookie in cookie_str.split(';') {
|
||||
let cookie = cookie.trim();
|
||||
if let Some(value) = cookie.strip_prefix("AT=") {
|
||||
if !value.is_empty() {
|
||||
return Some(value.to_string());
|
||||
}
|
||||
}
|
||||
if let Some(cookie_header) = headers.get(header::COOKIE)
|
||||
&& let Ok(cookie_str) = cookie_header.to_str()
|
||||
{
|
||||
for cookie in cookie_str.split(';') {
|
||||
let cookie = cookie.trim();
|
||||
if let Some(value) = cookie.strip_prefix("AT=")
|
||||
&& !value.is_empty()
|
||||
{
|
||||
return Some(value.to_string());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -19,11 +19,12 @@ use crate::pb::master_pb::seaweed_client::SeaweedClient;
|
||||
use crate::pb::volume_server_pb;
|
||||
use crate::remote_storage::s3_tier::{S3TierBackend, S3TierConfig};
|
||||
use crate::storage::store::Store;
|
||||
use crate::storage::types::NeedleId;
|
||||
use crate::storage::types::{NeedleId, VolumeId};
|
||||
use crate::storage::volume_report::VolumeReportKey;
|
||||
use crate::storage::volume_report_hash::report_hash;
|
||||
|
||||
const DUPLICATE_UUID_RETRY_MESSAGE: &str = "duplicate UUIDs detected, retrying connection";
|
||||
const VOLUME_IO_ERROR_TOLERANCE: i32 = 3;
|
||||
const MAX_DUPLICATE_UUID_RETRIES: u32 = 3;
|
||||
|
||||
/// Configuration for the heartbeat client.
|
||||
@@ -188,10 +189,10 @@ pub async fn run_heartbeat_with_state(
|
||||
pub fn to_grpc_address(master_addr: &str) -> String {
|
||||
if let Some((host, port_str)) = master_addr.rsplit_once(':') {
|
||||
// "host:port.grpcPort" — the part after the last '.' is the gRPC port.
|
||||
if let Some((_, grpc_port)) = port_str.rsplit_once('.') {
|
||||
if grpc_port.parse::<u16>().is_ok() {
|
||||
return format!("{}:{}", host, grpc_port);
|
||||
}
|
||||
if let Some((_, grpc_port)) = port_str.rsplit_once('.')
|
||||
&& grpc_port.parse::<u16>().is_ok()
|
||||
{
|
||||
return format!("{}:{}", host, grpc_port);
|
||||
}
|
||||
if let Ok(port) = port_str.parse::<u16>() {
|
||||
let grpc_port = port + 10000;
|
||||
@@ -315,6 +316,10 @@ fn collect_ec_shard_delta_messages(
|
||||
|
||||
for (disk_id, loc) in store.locations.iter().enumerate() {
|
||||
for (_, ec_vol) in loc.ec_volumes() {
|
||||
let (_, _, quarantined) = ec_vol.get_io_error_state();
|
||||
if quarantined {
|
||||
continue;
|
||||
}
|
||||
for shard in ec_vol.shards.iter().flatten() {
|
||||
messages.insert(
|
||||
(
|
||||
@@ -895,6 +900,7 @@ fn build_heartbeat_with_ec_status(
|
||||
// master can tell whether applying what it was sent leaves it current.
|
||||
// Volumes skipped below -- quarantined, phantom, expired -- are in neither.
|
||||
let mut volume_digest: u64 = 0;
|
||||
let mut quarantined_volumes: u32 = 0;
|
||||
let (send_full_list, report_generation, report_pass) = store.volume_report.begin();
|
||||
let mut changed_volumes = Vec::new();
|
||||
let mut max_file_key = NeedleId(0);
|
||||
@@ -916,10 +922,9 @@ fn build_heartbeat_with_ec_status(
|
||||
let mut effective_max_count = loc.max_volume_count.load(Ordering::Relaxed);
|
||||
if loc.is_disk_space_low.load(Ordering::Relaxed) {
|
||||
let used_slots = loc.volumes_len() as i32
|
||||
+ ((loc.ec_shard_count()
|
||||
+ crate::storage::erasure_coding::ec_shard::DATA_SHARDS_COUNT
|
||||
- 1)
|
||||
/ crate::storage::erasure_coding::ec_shard::DATA_SHARDS_COUNT)
|
||||
+ loc
|
||||
.ec_shard_count()
|
||||
.div_ceil(crate::storage::erasure_coding::ec_shard::DATA_SHARDS_COUNT)
|
||||
as i32;
|
||||
effective_max_count = used_slots;
|
||||
}
|
||||
@@ -937,6 +942,7 @@ fn build_heartbeat_with_ec_status(
|
||||
loc.disk_free_bytes.load(Ordering::Relaxed);
|
||||
|
||||
let mut delete_vids = Vec::new();
|
||||
let mut quarantine_vids: Vec<VolumeId> = Vec::new();
|
||||
for (_, vol) in loc.iter_volumes() {
|
||||
let cur_max = vol.max_file_key();
|
||||
if cur_max > max_file_key {
|
||||
@@ -946,9 +952,18 @@ fn build_heartbeat_with_ec_status(
|
||||
let volume_size = vol.dat_file_size().unwrap_or(0);
|
||||
let mut should_delete_volume = false;
|
||||
|
||||
if vol.last_io_error().is_some() {
|
||||
delete_vids.push(vol.id);
|
||||
should_delete_volume = true;
|
||||
let (_, io_count, io_quarantined) = vol.get_io_error_state();
|
||||
if io_quarantined || io_count >= VOLUME_IO_ERROR_TOLERANCE {
|
||||
if !io_quarantined {
|
||||
vol.mark_io_quarantined();
|
||||
warn!(
|
||||
"Volume {} quarantined after {} consecutive IO errors",
|
||||
vol.id.0, io_count
|
||||
);
|
||||
}
|
||||
quarantined_volumes += 1;
|
||||
quarantine_vids.push(vol.id);
|
||||
continue;
|
||||
} else if !vol.is_expired(volume_size, volume_size_limit) {
|
||||
// Detect phantom volumes: the .dat was unlinked from disk but is still
|
||||
// held open as a deleted FD, so the volume keeps serving and heartbeating
|
||||
@@ -1046,6 +1061,12 @@ fn build_heartbeat_with_ec_status(
|
||||
for vid in delete_vids {
|
||||
let _ = loc.delete_volume(vid, false, false);
|
||||
}
|
||||
|
||||
for vid in quarantine_vids {
|
||||
if let Some(vol) = loc.find_volume_mut(vid) {
|
||||
vol.set_no_write_or_delete(true);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Update disk size and read-only gauges
|
||||
@@ -1102,6 +1123,23 @@ fn build_heartbeat_with_ec_status(
|
||||
};
|
||||
let (location_uuids, disk_tags) = collect_location_metadata(store, &disk_max_by_id);
|
||||
|
||||
let mut quarantined_ec_shards: u32 = 0;
|
||||
for loc in &store.locations {
|
||||
for (_, ec_vol) in loc.ec_volumes() {
|
||||
let (_, _, quarantined) = ec_vol.get_io_error_state();
|
||||
if quarantined {
|
||||
quarantined_ec_shards += ec_vol.shard_count() as u32;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
crate::metrics::IO_QUARANTINE_GAUGE
|
||||
.with_label_values(&["volume"])
|
||||
.set(quarantined_volumes as i64);
|
||||
crate::metrics::IO_QUARANTINE_GAUGE
|
||||
.with_label_values(&["ec_shard"])
|
||||
.set(quarantined_ec_shards as i64);
|
||||
|
||||
let heartbeat = master_pb::Heartbeat {
|
||||
id: store.id.clone(),
|
||||
ip: config.ip.clone(),
|
||||
@@ -1137,6 +1175,10 @@ fn collect_live_ec_shards(
|
||||
|
||||
for (disk_id, loc) in store.locations.iter().enumerate() {
|
||||
for (_, ec_vol) in loc.ec_volumes() {
|
||||
let (_, _, quarantined) = ec_vol.get_io_error_state();
|
||||
if quarantined {
|
||||
continue;
|
||||
}
|
||||
for message in ec_vol.to_volume_ec_shard_information_messages(disk_id as u32) {
|
||||
if update_metrics {
|
||||
let total_size: u64 = message
|
||||
@@ -1976,8 +2018,13 @@ mod tests {
|
||||
|
||||
let heartbeat = build_heartbeat(&test_config(), &mut store);
|
||||
|
||||
// A sustained IO error quarantines the volume: it stays mounted
|
||||
// (so healthz can observe the quarantine state) but is not
|
||||
// advertised to the master.
|
||||
assert!(heartbeat.volumes.is_empty());
|
||||
assert!(!store.has_volume(VolumeId(51)));
|
||||
assert!(store.has_volume(VolumeId(51)));
|
||||
let (_, volume) = store.find_volume_mut(VolumeId(51)).unwrap();
|
||||
assert!(volume.is_no_write_or_delete());
|
||||
}
|
||||
|
||||
#[test]
|
||||
|
||||
@@ -134,8 +134,13 @@ pub async fn read_ec_shard_needle_distributed(
|
||||
Ok(fresh) => {
|
||||
// A complete reply merges into the cache; an incomplete one
|
||||
// (< data_shards) is left unwritten — keep the prior cache.
|
||||
match write_back_shard_locations(state, vid, fresh, snapshot.data_shards as usize)
|
||||
{
|
||||
match write_back_shard_locations(
|
||||
state,
|
||||
vid,
|
||||
fresh,
|
||||
snapshot.data_shards as usize,
|
||||
snapshot.encode_ts_ns,
|
||||
) {
|
||||
Some(merged) => shard_locations = merged,
|
||||
// An incomplete reply leaves the cache unwritten and its refresh
|
||||
// time unadvanced, so the mark this refresh consumed goes back.
|
||||
@@ -168,36 +173,37 @@ pub async fn read_ec_shard_needle_distributed(
|
||||
let parity_shards = snapshot.parity_shards as usize;
|
||||
let encode_ts_ns = snapshot.encode_ts_ns;
|
||||
let intervals = std::mem::take(&mut snapshot.intervals);
|
||||
let fetched: Vec<io::Result<(Vec<u8>, bool)>> = stream::iter(intervals.into_iter().map(|res| {
|
||||
let shard_locations = &shard_locations;
|
||||
async move {
|
||||
match res {
|
||||
IntervalResult::Local(buf) => Ok((buf, false)),
|
||||
IntervalResult::NeedRemote {
|
||||
shard_id,
|
||||
shard_offset,
|
||||
size,
|
||||
} => {
|
||||
fetch_one_interval(
|
||||
state,
|
||||
vid,
|
||||
needle_id,
|
||||
let fetched: Vec<io::Result<(Vec<u8>, bool)>> =
|
||||
stream::iter(intervals.into_iter().map(|res| {
|
||||
let shard_locations = &shard_locations;
|
||||
async move {
|
||||
match res {
|
||||
IntervalResult::Local(buf) => Ok((buf, false)),
|
||||
IntervalResult::NeedRemote {
|
||||
shard_id,
|
||||
shard_offset,
|
||||
size,
|
||||
shard_locations,
|
||||
data_shards,
|
||||
parity_shards,
|
||||
encode_ts_ns,
|
||||
)
|
||||
.await
|
||||
} => {
|
||||
fetch_one_interval(
|
||||
state,
|
||||
vid,
|
||||
needle_id,
|
||||
shard_id,
|
||||
shard_offset,
|
||||
size,
|
||||
shard_locations,
|
||||
data_shards,
|
||||
parity_shards,
|
||||
encode_ts_ns,
|
||||
)
|
||||
.await
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}))
|
||||
.buffered(INTERVAL_READ_CONCURRENCY)
|
||||
.collect()
|
||||
.await;
|
||||
}))
|
||||
.buffered(INTERVAL_READ_CONCURRENCY)
|
||||
.collect()
|
||||
.await;
|
||||
|
||||
let mut assembled: Vec<Vec<u8>> = Vec::with_capacity(fetched.len());
|
||||
for res in fetched {
|
||||
@@ -230,8 +236,10 @@ pub async fn read_ec_shard_needle_distributed(
|
||||
));
|
||||
}
|
||||
|
||||
let mut n = Needle::default();
|
||||
n.id = needle_id;
|
||||
let mut n = Needle {
|
||||
id: needle_id,
|
||||
..Needle::default()
|
||||
};
|
||||
n.read_bytes(
|
||||
&bytes,
|
||||
snapshot.offset.to_actual_offset(),
|
||||
@@ -256,22 +264,32 @@ pub async fn read_ec_shard_needle_distributed(
|
||||
pub async fn scrub_ec_volume_distributed(
|
||||
state: &Arc<VolumeServerState>,
|
||||
vid: VolumeId,
|
||||
expected_encode_ts_ns: i64,
|
||||
force_deleted_needles_check: bool,
|
||||
recover_unreadable: bool,
|
||||
) -> (i64, Vec<crate::pb::volume_server_pb::EcShardInfo>, Vec<String>) {
|
||||
// Phase A — under the Store read lock, run the index scrub and grab the
|
||||
) -> (
|
||||
i64,
|
||||
Vec<crate::pb::volume_server_pb::EcShardInfo>,
|
||||
Vec<String>,
|
||||
) {
|
||||
// Phase A — under the Store read lock, snapshot the index scrub and grab the
|
||||
// paths/scalars + shard-location staleness; release the lock before any await.
|
||||
let (
|
||||
ecx_path,
|
||||
collection,
|
||||
seed_errs,
|
||||
index_plan,
|
||||
ecx_walk,
|
||||
encode_ts_ns,
|
||||
cached_locations,
|
||||
cache_refreshed_at,
|
||||
data_shards,
|
||||
total_shards,
|
||||
) = {
|
||||
let store = state.store.read().unwrap();
|
||||
let ecv = match store.find_ec_volume(vid) {
|
||||
// Resolve the runtime matching the anchor's encode generation, not the
|
||||
// first-match find_ec_volume — otherwise the needle walk can scan an
|
||||
// older run while the parity half scans the newest.
|
||||
let ecv = match find_ec_volume_for_scrub(&store, vid, expected_encode_ts_ns) {
|
||||
Some(v) => v,
|
||||
None => {
|
||||
return (
|
||||
@@ -282,7 +300,26 @@ pub async fn scrub_ec_volume_distributed(
|
||||
}
|
||||
};
|
||||
// full scan means verifying the index as well
|
||||
let (_, errs) = ecv.scrub_index();
|
||||
let index_plan = ecv.scrub_index_plan();
|
||||
// A SECOND .ecx descriptor, opened under the guard for the needle walk
|
||||
// below. The index plan's handle is consumed by its own structural walk,
|
||||
// and both seek, so a `dup` would race the cursor. Reopening by PATH
|
||||
// after the guard is dropped would let a teardown that legitimately
|
||||
// unlinks or replaces the .ecx (the heartbeat's
|
||||
// delete_expired_ec_volumes, volume_ec_shards_delete) surface an
|
||||
// intentional removal as a scrub error or mix index generations within
|
||||
// one scrub — the same race the vanished-volume policy exists to hide.
|
||||
// The descriptor outlives the name, the same way the checksum plan's
|
||||
// shard handles do.
|
||||
let ecx_walk = fs::File::open(ecv.ecx_file_name());
|
||||
// Encode-run identity of the volume this scrub started against. The
|
||||
// per-needle `scrub_snapshot_under_lock` re-resolves the volume by id
|
||||
// under a fresh guard, so a teardown-and-remount of the same vid between
|
||||
// two rows would otherwise apply the captured .ecx's offsets to a
|
||||
// replacement volume's shards. Bind the walk to this generation: if the
|
||||
// mounted volume's encode_ts_ns no longer matches, abort like a
|
||||
// mid-scan unmount rather than mixing generations.
|
||||
let encode_ts_ns = ecv.encode_ts_ns;
|
||||
// Bind to locals so the inner RwLock/Mutex guards drop before the block ends.
|
||||
let cached_locations = ecv.shard_locations.read().unwrap().clone();
|
||||
let cache_refreshed_at = *ecv.shard_locations_refresh_time.lock().unwrap();
|
||||
@@ -291,13 +328,47 @@ pub async fn scrub_ec_volume_distributed(
|
||||
(
|
||||
ecv.ecx_file_name(),
|
||||
ecv.collection.clone(),
|
||||
errs,
|
||||
index_plan,
|
||||
ecx_walk,
|
||||
encode_ts_ns,
|
||||
cached_locations,
|
||||
cache_refreshed_at,
|
||||
data_shards,
|
||||
total_shards,
|
||||
)
|
||||
};
|
||||
// Lock released: walk the index now, before anything else appends to errs,
|
||||
// so the seeded errors keep their position in the reported details.
|
||||
//
|
||||
// `index_plan.run()` reads the whole .ecx synchronously, so run it in the
|
||||
// blocking pool rather than on this async worker — a large index scan would
|
||||
// otherwise block unrelated RPC work handled on the same executor. Same
|
||||
// treatment as the CHECKSUM/LOCAL plans in the gRPC handler.
|
||||
let (_, seed_errs) = match tokio::task::spawn_blocking(move || index_plan.run()).await {
|
||||
Ok(v) => v,
|
||||
Err(e) => {
|
||||
// A panic is evidence about the volume and counts as broken; a
|
||||
// cancellation is not — spawn_blocking only reports it when the
|
||||
// runtime is going down, the volume was never scanned, and the
|
||||
// caller (FULL/READS) would put a false corruption into
|
||||
// broken_volume_ids if it reached the errs path. Match the
|
||||
// record_scrub_join_failure distinction used by the handler arms.
|
||||
if e.is_panic() {
|
||||
return (
|
||||
0,
|
||||
Vec::new(),
|
||||
vec![format!(
|
||||
"EC volume {} index scrub task panicked: {}",
|
||||
vid.0, e
|
||||
)],
|
||||
);
|
||||
}
|
||||
// Cancellation: the runtime is shutting down, so this response is
|
||||
// unlikely to reach anyone. Return clean rather than inventing a
|
||||
// corruption for a volume that was never scanned.
|
||||
return (0, Vec::new(), Vec::new());
|
||||
}
|
||||
};
|
||||
let mut errs = seed_errs;
|
||||
|
||||
// Refresh the shard-location cache once up front (mirrors Go's
|
||||
@@ -314,7 +385,9 @@ pub async fn scrub_ec_volume_distributed(
|
||||
) {
|
||||
match cached_lookup_ec_shard_locations(state, vid).await {
|
||||
Ok(fresh) => {
|
||||
if write_back_shard_locations(state, vid, fresh, data_shards).is_none() {
|
||||
if write_back_shard_locations(state, vid, fresh, data_shards, expected_encode_ts_ns)
|
||||
.is_none()
|
||||
{
|
||||
mark_shard_locations_stale(state, vid);
|
||||
return (
|
||||
0,
|
||||
@@ -341,53 +414,88 @@ pub async fn scrub_ec_volume_distributed(
|
||||
// walk, so per-needle snapshots no longer clone it.
|
||||
let locations: HashMap<ShardId, Vec<String>> = {
|
||||
let store = state.store.read().unwrap();
|
||||
let ecv = match store.find_ec_volume(vid) {
|
||||
let ecv = match find_ec_volume_for_scrub(&store, vid, expected_encode_ts_ns) {
|
||||
Some(v) => v,
|
||||
None => {
|
||||
return (
|
||||
0,
|
||||
Vec::new(),
|
||||
vec![format!("EC volume id {} not found", vid.0)],
|
||||
)
|
||||
);
|
||||
}
|
||||
};
|
||||
let map = ecv.shard_locations.read().unwrap().clone();
|
||||
map
|
||||
ecv.shard_locations.read().unwrap().clone()
|
||||
};
|
||||
|
||||
// Walk the .ecx (private fd, no lock) for the row count + live (id, offset, size).
|
||||
let mut count: i64 = 0;
|
||||
let mut needles: Vec<(NeedleId, Offset, Size)> = Vec::new();
|
||||
match fs::File::open(&ecx_path) {
|
||||
Ok(mut f) => {
|
||||
if let Err(e) = crate::storage::idx::walk_index_file(&mut f, 0, |id, offset, size| {
|
||||
count += 1;
|
||||
// Skip ALL deleted entries: -1 tombstones (runtime delete folded
|
||||
// into .ecx) and -originalSize entries (a needle deleted on the
|
||||
// regular volume before EC encode). get_actual_size uses the raw
|
||||
// signed size, so a negative would yield empty intervals
|
||||
// (false-positive) or an under-16-byte buffer (parse panic).
|
||||
if !size.is_deleted() {
|
||||
needles.push((id, offset, size));
|
||||
// Walk the .ecx (private fd captured under the lock, no lock held) for the
|
||||
// row count + live (id, offset, size). Reading through the captured
|
||||
// descriptor — not a pathname reopen — keeps a concurrent teardown from
|
||||
// surfacing an intentional removal as a scrub error or mixing index
|
||||
// generations, the same invariant the vanished-volume policy enforces.
|
||||
//
|
||||
// `walk_index_file` reads the full .ecx synchronously, so run it in the
|
||||
// blocking pool rather than on this async worker — same reason as
|
||||
// `index_plan.run()` above.
|
||||
let (count, needles, walk_errs) = match tokio::task::spawn_blocking(
|
||||
move || -> (i64, Vec<(NeedleId, Offset, Size)>, Vec<String>) {
|
||||
let mut count: i64 = 0;
|
||||
let mut needles: Vec<(NeedleId, Offset, Size)> = Vec::new();
|
||||
let mut walk_errs: Vec<String> = Vec::new();
|
||||
match ecx_walk {
|
||||
Ok(mut f) => {
|
||||
if let Err(e) =
|
||||
crate::storage::idx::walk_index_file(&mut f, 0, |id, offset, size| {
|
||||
count += 1;
|
||||
// Skip ALL deleted entries: -1 tombstones (runtime delete folded
|
||||
// into .ecx) and -originalSize entries (a needle deleted on the
|
||||
// regular volume before EC encode). get_actual_size uses the raw
|
||||
// signed size, so a negative would yield empty intervals
|
||||
// (false-positive) or an under-16-byte buffer (parse panic).
|
||||
if !size.is_deleted() {
|
||||
needles.push((id, offset, size));
|
||||
}
|
||||
Ok(())
|
||||
})
|
||||
{
|
||||
walk_errs.push(format!("walk ECX file {}: {}", ecx_path, e));
|
||||
}
|
||||
}
|
||||
Ok(())
|
||||
}) {
|
||||
errs.push(format!("walk ECX file {}: {}", ecx_path, e));
|
||||
Err(e) => walk_errs.push(format!("open ECX file {}: {}", ecx_path, e)),
|
||||
}
|
||||
(count, needles, walk_errs)
|
||||
},
|
||||
)
|
||||
.await
|
||||
{
|
||||
Ok(v) => v,
|
||||
Err(e) => {
|
||||
// A panic is evidence about the volume and counts as broken; a
|
||||
// cancellation is not — see the index_plan join above for the
|
||||
// same reasoning.
|
||||
if e.is_panic() {
|
||||
return (
|
||||
0,
|
||||
Vec::new(),
|
||||
vec![format!("EC volume {} ecx walk task panicked: {}", vid.0, e)],
|
||||
);
|
||||
}
|
||||
return (0, Vec::new(), Vec::new());
|
||||
}
|
||||
Err(e) => errs.push(format!("open ECX file {}: {}", ecx_path, e)),
|
||||
}
|
||||
};
|
||||
errs.extend(walk_errs);
|
||||
|
||||
// reads for EC chunks can hit the same shard repeatedly, so dedupe broken shards
|
||||
let mut broken_shards: HashMap<ShardId, crate::pb::volume_server_pb::EcShardInfo> = HashMap::new();
|
||||
let mut broken_shards: HashMap<ShardId, crate::pb::volume_server_pb::EcShardInfo> =
|
||||
HashMap::new();
|
||||
|
||||
for (id, offset, size) in needles {
|
||||
// Per-needle snapshot under the lock from the RAW .ecx (offset, size) so
|
||||
// logically-deleted needles are still verified; lock dropped before await.
|
||||
let snapshot = match scrub_snapshot_under_lock(state, vid, offset, size) {
|
||||
let snapshot = match scrub_snapshot_under_lock(state, vid, offset, size, encode_ts_ns) {
|
||||
Ok(s) => s,
|
||||
// Volume unmounted mid-scan: abort with an error rather than skipping
|
||||
// every remaining needle, which would report a false-CLEAN result.
|
||||
// Volume unmounted (or remounted as a different encode run) mid-scan:
|
||||
// abort with an error rather than skipping every remaining needle,
|
||||
// which would report a false-CLEAN result.
|
||||
Err(e) if e.kind() == io::ErrorKind::NotFound => {
|
||||
errs.push(format!("EC volume {} unmounted during scrub: {}", vid.0, e));
|
||||
break;
|
||||
@@ -513,7 +621,11 @@ pub async fn scrub_ec_volume_distributed(
|
||||
// Mirror Go CmpEcShardInfo: sort by (volume_id, shard_id).
|
||||
let mut broken: Vec<crate::pb::volume_server_pb::EcShardInfo> =
|
||||
broken_shards.into_values().collect();
|
||||
broken.sort_by(|a, b| a.volume_id.cmp(&b.volume_id).then(a.shard_id.cmp(&b.shard_id)));
|
||||
broken.sort_by(|a, b| {
|
||||
a.volume_id
|
||||
.cmp(&b.volume_id)
|
||||
.then(a.shard_id.cmp(&b.shard_id))
|
||||
});
|
||||
|
||||
(count, broken, errs)
|
||||
}
|
||||
@@ -551,9 +663,10 @@ fn scrub_snapshot_under_lock(
|
||||
vid: VolumeId,
|
||||
offset: Offset,
|
||||
size: Size,
|
||||
expected_encode_ts: i64,
|
||||
) -> io::Result<ScrubSnapshot> {
|
||||
let store = state.store.read().unwrap();
|
||||
let ecv = match store.find_ec_volume(vid) {
|
||||
let ecv = match find_ec_volume_for_scrub(&store, vid, expected_encode_ts) {
|
||||
Some(v) => v,
|
||||
// Volume unmounted mid-scan: a distinct NotFound so the caller aborts
|
||||
// with an error rather than silently skipping (which would false-CLEAN).
|
||||
@@ -564,6 +677,28 @@ fn scrub_snapshot_under_lock(
|
||||
))
|
||||
}
|
||||
};
|
||||
// The volume was torn down and remounted as a DIFFERENT encode run between
|
||||
// two rows. The .ecx offsets captured at the start of the walk belong to
|
||||
// the old generation; applying them to the replacement's shards would
|
||||
// falsely report corruption. Abort like a mid-scan unmount instead of
|
||||
// mixing generations within one scrub.
|
||||
//
|
||||
// `encode_ts_ns == 0` means the .vif carried no encode-run identity (a
|
||||
// legacy or pre-feature volume). Two such volumes are NOT the same mount
|
||||
// by this check alone — 0 == 0 would accept a teardown-and-remount and
|
||||
// apply the old .ecx's offsets to the replacement's shards. Only treat a
|
||||
// match as verified when the identity is non-zero; when it is zero, fall
|
||||
// back to the pre-check behavior (no generation binding) rather than
|
||||
// aborting a scrub that was already running without the guard.
|
||||
if expected_encode_ts != 0 && ecv.encode_ts_ns != expected_encode_ts {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::NotFound,
|
||||
format!(
|
||||
"EC volume {} remounted as a different encode run during scrub (was {}, now {})",
|
||||
vid.0, expected_encode_ts, ecv.encode_ts_ns
|
||||
),
|
||||
));
|
||||
}
|
||||
let intervals = ecv.locate_ec_shard_needle_interval(offset.to_actual_offset(), size);
|
||||
if intervals.is_empty() {
|
||||
return Err(io::Error::new(
|
||||
@@ -678,9 +813,9 @@ fn needs_refresh(
|
||||
let ttl = if stale || shard_count < data_shards {
|
||||
Duration::from_secs(11)
|
||||
} else if shard_count == total_shards {
|
||||
Duration::from_secs(37 * 60)
|
||||
Duration::from_mins(37)
|
||||
} else {
|
||||
Duration::from_secs(7 * 60)
|
||||
Duration::from_mins(7)
|
||||
};
|
||||
age >= ttl
|
||||
}
|
||||
@@ -734,22 +869,19 @@ async fn cached_lookup_ec_shard_locations(
|
||||
}
|
||||
};
|
||||
if master.is_empty() {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
"no master configured for ec shard lookup",
|
||||
));
|
||||
return Err(io::Error::other("no master configured for ec shard lookup"));
|
||||
}
|
||||
|
||||
let grpc_addr = parse_grpc_address(&master)
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::InvalidInput, e))?;
|
||||
let grpc_addr =
|
||||
parse_grpc_address(&master).map_err(|e| io::Error::new(io::ErrorKind::InvalidInput, e))?;
|
||||
let endpoint = build_grpc_endpoint(&grpc_addr, state.outgoing_grpc_tls.as_ref())
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, e.to_string()))?;
|
||||
.map_err(|e| io::Error::other(e.to_string()))?;
|
||||
let channel = endpoint
|
||||
.connect_timeout(Duration::from_secs(5))
|
||||
.timeout(Duration::from_secs(10))
|
||||
.connect()
|
||||
.await
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("master connect: {}", e)))?;
|
||||
.map_err(|e| io::Error::other(format!("master connect: {}", e)))?;
|
||||
|
||||
let mut client = SeaweedClient::with_interceptor(channel, outgoing_request_id_interceptor)
|
||||
.max_decoding_message_size(GRPC_MAX_MESSAGE_SIZE)
|
||||
@@ -758,7 +890,7 @@ async fn cached_lookup_ec_shard_locations(
|
||||
let resp = client
|
||||
.lookup_ec_volume(Request::new(LookupEcVolumeRequest { volume_id: vid.0 }))
|
||||
.await
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("lookup_ec_volume: {}", e)))?;
|
||||
.map_err(|e| io::Error::other(format!("lookup_ec_volume: {}", e)))?;
|
||||
let resp = resp.into_inner();
|
||||
|
||||
let mut out = HashMap::new();
|
||||
@@ -789,15 +921,34 @@ fn write_back_shard_locations(
|
||||
vid: VolumeId,
|
||||
locations: HashMap<ShardId, Vec<String>>,
|
||||
data_shards: usize,
|
||||
expected_encode_ts_ns: i64,
|
||||
) -> Option<HashMap<ShardId, Vec<String>>> {
|
||||
if locations.len() < data_shards {
|
||||
return None;
|
||||
}
|
||||
let store = state.store.read().unwrap();
|
||||
let ecv = store.find_ec_volume(vid)?;
|
||||
let ecv = find_ec_volume_for_scrub(&store, vid, expected_encode_ts_ns)?;
|
||||
Some(ecv.merge_shard_locations(locations))
|
||||
}
|
||||
|
||||
/// Resolve the runtime matching the scrub's anchor encode generation, not the
|
||||
/// first-match `find_ec_volume`. When `expected_encode_ts_ns` is 0 (legacy or
|
||||
/// pre-feature), falls back to first-match so existing behavior is preserved.
|
||||
fn find_ec_volume_for_scrub(
|
||||
store: &crate::storage::store::Store,
|
||||
vid: VolumeId,
|
||||
expected_encode_ts_ns: i64,
|
||||
) -> Option<&crate::storage::erasure_coding::EcVolume> {
|
||||
if expected_encode_ts_ns != 0 {
|
||||
store
|
||||
.find_all_ec_volumes(vid)
|
||||
.into_iter()
|
||||
.find(|v| v.encode_ts_ns == expected_encode_ts_ns)
|
||||
} else {
|
||||
store.find_ec_volume(vid)
|
||||
}
|
||||
}
|
||||
|
||||
/// Build a SeaweedFS-style `host:httpPort.grpcPort` address from a
|
||||
/// master `Location` so the result is what `parse_grpc_address` (and
|
||||
/// the heartbeat path) already understand.
|
||||
@@ -806,16 +957,17 @@ fn format_location_as_server_address(loc: &master_pb::Location) -> String {
|
||||
.url
|
||||
.trim_start_matches("http://")
|
||||
.trim_start_matches("https://");
|
||||
if loc.grpc_port > 0 {
|
||||
if let Some((host, http_port)) = raw.rsplit_once(':') {
|
||||
return format!("{}:{}.{}", host, http_port, loc.grpc_port);
|
||||
}
|
||||
if loc.grpc_port > 0
|
||||
&& let Some((host, http_port)) = raw.rsplit_once(':')
|
||||
{
|
||||
return format!("{}:{}.{}", host, http_port, loc.grpc_port);
|
||||
}
|
||||
raw.to_string()
|
||||
}
|
||||
|
||||
/// Try direct peer read; on failure, reconstruct via Reed-Solomon
|
||||
/// from the other shards. Mirrors `readOneEcShardInterval`'s tail.
|
||||
#[expect(clippy::too_many_arguments)]
|
||||
async fn fetch_one_interval(
|
||||
state: &Arc<VolumeServerState>,
|
||||
vid: VolumeId,
|
||||
@@ -829,35 +981,35 @@ async fn fetch_one_interval(
|
||||
expected_encode_ts_ns: i64,
|
||||
) -> io::Result<(Vec<u8>, bool)> {
|
||||
// Direct peer read against the cached locations for this shard.
|
||||
if let Some(sources) = shard_locations.get(&shard_id) {
|
||||
if !sources.is_empty() {
|
||||
match read_remote_ec_shard_interval(
|
||||
state,
|
||||
sources,
|
||||
vid,
|
||||
needle_id,
|
||||
shard_id,
|
||||
shard_offset,
|
||||
size,
|
||||
expected_encode_ts_ns,
|
||||
)
|
||||
.await
|
||||
{
|
||||
// A deleted needle short-circuits: don't reconstruct (every shard
|
||||
// would report deleted), let the caller return "deleted".
|
||||
Ok((buf, is_deleted)) => return Ok((buf, is_deleted)),
|
||||
Err(e) => {
|
||||
tracing::debug!(
|
||||
"direct read ec shard {}.{} from {:?} failed: {} — will reconstruct",
|
||||
vid.0,
|
||||
shard_id,
|
||||
sources,
|
||||
e
|
||||
);
|
||||
// Reconstruction below skips this very shard, so nothing else
|
||||
// invalidates the location that just failed.
|
||||
mark_shard_locations_stale(state, vid);
|
||||
}
|
||||
if let Some(sources) = shard_locations.get(&shard_id)
|
||||
&& !sources.is_empty()
|
||||
{
|
||||
match read_remote_ec_shard_interval(
|
||||
state,
|
||||
sources,
|
||||
vid,
|
||||
needle_id,
|
||||
shard_id,
|
||||
shard_offset,
|
||||
size,
|
||||
expected_encode_ts_ns,
|
||||
)
|
||||
.await
|
||||
{
|
||||
// A deleted needle short-circuits: don't reconstruct (every shard
|
||||
// would report deleted), let the caller return "deleted".
|
||||
Ok((buf, is_deleted)) => return Ok((buf, is_deleted)),
|
||||
Err(e) => {
|
||||
tracing::debug!(
|
||||
"direct read ec shard {}.{} from {:?} failed: {} — will reconstruct",
|
||||
vid.0,
|
||||
shard_id,
|
||||
sources,
|
||||
e
|
||||
);
|
||||
// Reconstruction below skips this very shard, so nothing else
|
||||
// invalidates the location that just failed.
|
||||
mark_shard_locations_stale(state, vid);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -879,6 +1031,7 @@ async fn fetch_one_interval(
|
||||
.await
|
||||
}
|
||||
|
||||
#[expect(clippy::too_many_arguments)]
|
||||
async fn read_remote_ec_shard_interval(
|
||||
state: &Arc<VolumeServerState>,
|
||||
sources: &[String],
|
||||
@@ -915,6 +1068,7 @@ async fn read_remote_ec_shard_interval(
|
||||
}))
|
||||
}
|
||||
|
||||
#[expect(clippy::too_many_arguments)]
|
||||
async fn do_read_remote_ec_shard_interval(
|
||||
state: &Arc<VolumeServerState>,
|
||||
source: &str,
|
||||
@@ -928,18 +1082,13 @@ async fn do_read_remote_ec_shard_interval(
|
||||
let grpc_addr =
|
||||
parse_grpc_address(source).map_err(|e| io::Error::new(io::ErrorKind::InvalidInput, e))?;
|
||||
let endpoint = build_grpc_endpoint(&grpc_addr, state.outgoing_grpc_tls.as_ref())
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, e.to_string()))?;
|
||||
.map_err(|e| io::Error::other(e.to_string()))?;
|
||||
let channel = endpoint
|
||||
.connect_timeout(Duration::from_secs(5))
|
||||
.timeout(Duration::from_secs(30))
|
||||
.connect()
|
||||
.await
|
||||
.map_err(|e| {
|
||||
io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!("connect to {}: {}", source, e),
|
||||
)
|
||||
})?;
|
||||
.map_err(|e| io::Error::other(format!("connect to {}: {}", source, e)))?;
|
||||
|
||||
// TODO(grpc-jwt): clusters with `jwt.signing.key` configured will
|
||||
// reject peer-to-peer VolumeEcShardRead calls until the Rust
|
||||
@@ -965,10 +1114,10 @@ async fn do_read_remote_ec_shard_interval(
|
||||
.volume_ec_shard_read(Request::new(req))
|
||||
.await
|
||||
.map_err(|e| {
|
||||
io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!("volume_ec_shard_read {}.{} from {}: {}", vid.0, shard_id, source, e),
|
||||
)
|
||||
io::Error::other(format!(
|
||||
"volume_ec_shard_read {}.{} from {}: {}",
|
||||
vid.0, shard_id, source, e
|
||||
))
|
||||
})?;
|
||||
let mut stream = resp.into_inner();
|
||||
|
||||
@@ -977,19 +1126,16 @@ async fn do_read_remote_ec_shard_interval(
|
||||
while let Some(msg) = stream
|
||||
.message()
|
||||
.await
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("recv: {}", e)))?
|
||||
.map_err(|e| io::Error::other(format!("recv: {}", e)))?
|
||||
{
|
||||
// Validate the served shard's identity client-side, so the guard holds even
|
||||
// against a pre-upgrade server that ignored the request field (returns 0).
|
||||
// A mismatch fails the read; the caller recovers from parity.
|
||||
if expected_encode_ts_ns != 0 && msg.encode_ts_ns != expected_encode_ts_ns {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!(
|
||||
"ec shard {}.{} from {} belongs to a different encode run (want {} got {})",
|
||||
vid.0, shard_id, source, expected_encode_ts_ns, msg.encode_ts_ns
|
||||
),
|
||||
));
|
||||
return Err(io::Error::other(format!(
|
||||
"ec shard {}.{} from {} belongs to a different encode run (want {} got {})",
|
||||
vid.0, shard_id, source, expected_encode_ts_ns, msg.encode_ts_ns
|
||||
)));
|
||||
}
|
||||
if msg.is_deleted {
|
||||
is_deleted = true;
|
||||
@@ -1025,6 +1171,7 @@ async fn do_read_remote_ec_shard_interval(
|
||||
Ok((out, false))
|
||||
}
|
||||
|
||||
#[expect(clippy::too_many_arguments)]
|
||||
async fn recover_one_remote_ec_shard_interval(
|
||||
state: &Arc<VolumeServerState>,
|
||||
vid: VolumeId,
|
||||
@@ -1038,12 +1185,8 @@ async fn recover_one_remote_ec_shard_interval(
|
||||
expected_encode_ts_ns: i64,
|
||||
) -> io::Result<(Vec<u8>, bool)> {
|
||||
let total_shards = data_shards + parity_shards;
|
||||
let rs = ReedSolomon::new(data_shards, parity_shards).map_err(|e| {
|
||||
io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!("reed-solomon init: {:?}", e),
|
||||
)
|
||||
})?;
|
||||
let rs = ReedSolomon::new(data_shards, parity_shards)
|
||||
.map_err(|e| io::Error::other(format!("reed-solomon init: {:?}", e)))?;
|
||||
|
||||
// Charge the buffers this recovery is about to hold against the budget, so a
|
||||
// burst of them queues here rather than on the heap. An interval whose
|
||||
@@ -1053,13 +1196,10 @@ async fn recover_one_remote_ec_shard_interval(
|
||||
.acquire_many((size * data_shards).min(EC_RECOVER_BUDGET) as u32)
|
||||
.await
|
||||
.map_err(|e| {
|
||||
io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!(
|
||||
"ec recover budget for shard {}.{}: {}",
|
||||
vid.0, shard_id_to_recover, e
|
||||
),
|
||||
)
|
||||
io::Error::other(format!(
|
||||
"ec recover budget for shard {}.{}: {}",
|
||||
vid.0, shard_id_to_recover, e
|
||||
))
|
||||
})?;
|
||||
|
||||
let mut bufs: Vec<Option<Vec<u8>>> = vec![None; total_shards];
|
||||
@@ -1072,7 +1212,7 @@ async fn recover_one_remote_ec_shard_interval(
|
||||
let mut available = 0usize;
|
||||
{
|
||||
let store = state.store.read().unwrap();
|
||||
for sid in 0..total_shards {
|
||||
for (sid, slot) in bufs.iter_mut().enumerate() {
|
||||
if available >= data_shards {
|
||||
break;
|
||||
}
|
||||
@@ -1085,13 +1225,21 @@ async fn recover_one_remote_ec_shard_interval(
|
||||
// lenient only when the caller carries no identity (pre-upgrade).
|
||||
// Mirrors Go's `readLocalEcShardInterval`.
|
||||
let owner = match store.find_ec_volume_with_shard(vid, sid as u32) {
|
||||
Some(ecv) if expected_encode_ts_ns == 0 || ecv.encode_ts_ns == expected_encode_ts_ns => ecv,
|
||||
Some(ecv)
|
||||
if expected_encode_ts_ns == 0 || ecv.encode_ts_ns == expected_encode_ts_ns =>
|
||||
{
|
||||
ecv
|
||||
}
|
||||
_ => continue,
|
||||
};
|
||||
if let Some(Some(shard)) = owner.shards.get(sid) {
|
||||
let mut buf = vec![0u8; size];
|
||||
if shard.read_at(&mut buf, shard_offset as u64).map(|n| n == size).unwrap_or(false) {
|
||||
bufs[sid] = Some(buf);
|
||||
if shard
|
||||
.read_at(&mut buf, shard_offset as u64)
|
||||
.map(|n| n == size)
|
||||
.unwrap_or(false)
|
||||
{
|
||||
*slot = Some(buf);
|
||||
available += 1;
|
||||
}
|
||||
}
|
||||
@@ -1175,34 +1323,25 @@ async fn recover_one_remote_ec_shard_interval(
|
||||
if any_deleted {
|
||||
return Ok((Vec::new(), true));
|
||||
}
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!(
|
||||
"cannot recover ec shard {}.{}: only {} shards available, need at least {}",
|
||||
vid.0, shard_id_to_recover, available, data_shards
|
||||
),
|
||||
));
|
||||
return Err(io::Error::other(format!(
|
||||
"cannot recover ec shard {}.{}: only {} shards available, need at least {}",
|
||||
vid.0, shard_id_to_recover, available, data_shards
|
||||
)));
|
||||
}
|
||||
|
||||
rs.reconstruct(&mut bufs).map_err(|e| {
|
||||
io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!(
|
||||
"reed-solomon reconstruct ec shard {}.{}: {:?}",
|
||||
vid.0, shard_id_to_recover, e
|
||||
),
|
||||
)
|
||||
io::Error::other(format!(
|
||||
"reed-solomon reconstruct ec shard {}.{}: {:?}",
|
||||
vid.0, shard_id_to_recover, e
|
||||
))
|
||||
})?;
|
||||
|
||||
match bufs.into_iter().nth(shard_id_to_recover as usize).flatten() {
|
||||
Some(buf) => Ok((buf, any_deleted)),
|
||||
None => Err(io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!(
|
||||
"reconstructed buffer for shard {}.{} missing after RS reconstruct",
|
||||
vid.0, shard_id_to_recover
|
||||
),
|
||||
)),
|
||||
None => Err(io::Error::other(format!(
|
||||
"reconstructed buffer for shard {}.{} missing after RS reconstruct",
|
||||
vid.0, shard_id_to_recover
|
||||
))),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1344,12 +1483,12 @@ async fn fetch_ec_index_from_one_peer(
|
||||
let grpc_addr =
|
||||
parse_grpc_address(peer).map_err(|e| io::Error::new(io::ErrorKind::InvalidInput, e))?;
|
||||
let channel = build_grpc_endpoint(&grpc_addr, state.outgoing_grpc_tls.as_ref())
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, e.to_string()))?
|
||||
.map_err(|e| io::Error::other(e.to_string()))?
|
||||
.connect_timeout(Duration::from_secs(5))
|
||||
.timeout(Duration::from_secs(30))
|
||||
.connect()
|
||||
.await
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("connect {}: {}", peer, e)))?;
|
||||
.map_err(|e| io::Error::other(format!("connect {}: {}", peer, e)))?;
|
||||
let mut client = VolumeServerClient::with_interceptor(channel, outgoing_request_id_interceptor)
|
||||
.max_decoding_message_size(GRPC_MAX_MESSAGE_SIZE)
|
||||
.max_encoding_message_size(GRPC_MAX_MESSAGE_SIZE);
|
||||
@@ -1370,18 +1509,19 @@ async fn fetch_ec_index_from_one_peer(
|
||||
let stream = client
|
||||
.copy_file(copy_req(".ecx", false))
|
||||
.await
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("copy .ecx: {}", e)))?
|
||||
.map_err(|e| io::Error::other(format!("copy .ecx: {}", e)))?
|
||||
.into_inner();
|
||||
drain_copy_stream(stream, ecx_path, false).await?;
|
||||
|
||||
let meta = fs::metadata(ecx_path)
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("stat copied .ecx: {}", e)))?;
|
||||
let meta =
|
||||
fs::metadata(ecx_path).map_err(|e| io::Error::other(format!("stat copied .ecx: {}", e)))?;
|
||||
if meta.is_dir() || meta.len() == 0 {
|
||||
let _ = fs::remove_file(ecx_path);
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!("peer {} served an unusable .ecx (size {})", peer, meta.len()),
|
||||
));
|
||||
return Err(io::Error::other(format!(
|
||||
"peer {} served an unusable .ecx (size {})",
|
||||
peer,
|
||||
meta.len()
|
||||
)));
|
||||
}
|
||||
|
||||
// .ecj is the source peer's deletion journal (appended); .vif carries EC
|
||||
@@ -1418,18 +1558,21 @@ async fn drain_copy_stream(
|
||||
) -> io::Result<()> {
|
||||
use std::io::Write;
|
||||
let mut file = if append {
|
||||
fs::OpenOptions::new().create(true).append(true).open(dest_path)
|
||||
fs::OpenOptions::new()
|
||||
.create(true)
|
||||
.append(true)
|
||||
.open(dest_path)
|
||||
} else {
|
||||
fs::File::create(dest_path)
|
||||
}
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("create {}: {}", dest_path, e)))?;
|
||||
.map_err(|e| io::Error::other(format!("create {}: {}", dest_path, e)))?;
|
||||
while let Some(chunk) = stream
|
||||
.message()
|
||||
.await
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("recv {}: {}", dest_path, e)))?
|
||||
.map_err(|e| io::Error::other(format!("recv {}: {}", dest_path, e)))?
|
||||
{
|
||||
file.write_all(&chunk.file_content)
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("write {}: {}", dest_path, e)))?;
|
||||
.map_err(|e| io::Error::other(format!("write {}: {}", dest_path, e)))?;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -423,13 +423,12 @@ fn collect_ui_data(
|
||||
shard_id: shard.shard_id,
|
||||
size: shard_size,
|
||||
});
|
||||
if created_at == "-" {
|
||||
if let Ok(metadata) = std::fs::metadata(shard.file_name()) {
|
||||
if let Ok(modified) = metadata.modified() {
|
||||
let ts: chrono::DateTime<chrono::Local> = modified.into();
|
||||
created_at = ts.format("%Y-%m-%d %H:%M").to_string();
|
||||
}
|
||||
}
|
||||
if created_at == "-"
|
||||
&& let Ok(metadata) = std::fs::metadata(shard.file_name())
|
||||
&& let Ok(modified) = metadata.modified()
|
||||
{
|
||||
let ts: chrono::DateTime<chrono::Local> = modified.into();
|
||||
created_at = ts.format("%Y-%m-%d %H:%M").to_string();
|
||||
}
|
||||
}
|
||||
let preferred_size = ec_volume.dat_file_size.max(0) as u64;
|
||||
|
||||
@@ -312,16 +312,15 @@ async fn admin_store_handler(state: State<Arc<VolumeServerState>>, request: Requ
|
||||
)
|
||||
}
|
||||
};
|
||||
if method == Method::GET {
|
||||
if let Some(response_bytes) = response
|
||||
if method == Method::GET
|
||||
&& let Some(response_bytes) = response
|
||||
.headers()
|
||||
.get(header::CONTENT_LENGTH)
|
||||
.and_then(|value| value.to_str().ok())
|
||||
.and_then(|value| value.parse::<i64>().ok())
|
||||
.filter(|value| *value > 0)
|
||||
{
|
||||
super::server_stats::record_bytes_out(response_bytes);
|
||||
}
|
||||
{
|
||||
super::server_stats::record_bytes_out(response_bytes);
|
||||
}
|
||||
super::server_stats::record_request_close();
|
||||
crate::metrics::INFLIGHT_REQUESTS_GAUGE
|
||||
@@ -358,16 +357,15 @@ async fn public_store_handler(state: State<Arc<VolumeServerState>>, request: Req
|
||||
}
|
||||
_ => StatusCode::OK.into_response(),
|
||||
};
|
||||
if method == Method::GET {
|
||||
if let Some(response_bytes) = response
|
||||
if method == Method::GET
|
||||
&& let Some(response_bytes) = response
|
||||
.headers()
|
||||
.get(header::CONTENT_LENGTH)
|
||||
.and_then(|value| value.to_str().ok())
|
||||
.and_then(|value| value.parse::<i64>().ok())
|
||||
.filter(|value| *value > 0)
|
||||
{
|
||||
super::server_stats::record_bytes_out(response_bytes);
|
||||
}
|
||||
{
|
||||
super::server_stats::record_bytes_out(response_bytes);
|
||||
}
|
||||
super::server_stats::record_request_close();
|
||||
crate::metrics::INFLIGHT_REQUESTS_GAUGE
|
||||
|
||||
@@ -131,10 +131,10 @@ impl DiskLocation {
|
||||
for entry in entries {
|
||||
let entry = entry?;
|
||||
let name = entry.file_name().into_string().unwrap_or_default();
|
||||
if let Some((collection, vid)) = parse_volume_filename(&name) {
|
||||
if seen.insert((collection.clone(), vid)) {
|
||||
dat_files.push((collection, vid));
|
||||
}
|
||||
if let Some((collection, vid)) = parse_volume_filename(&name)
|
||||
&& seen.insert((collection.clone(), vid))
|
||||
{
|
||||
dat_files.push((collection, vid));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -327,10 +327,10 @@ impl DiskLocation {
|
||||
.strip_suffix(".cpc")
|
||||
.or_else(|| name.strip_suffix(".cpd"))
|
||||
.or_else(|| name.strip_suffix(".cpx"));
|
||||
if let Some(stem) = stem {
|
||||
if let Some(key) = parse_collection_volume_id(stem) {
|
||||
pending.insert(key);
|
||||
}
|
||||
if let Some(stem) = stem
|
||||
&& let Some(key) = parse_collection_volume_id(stem)
|
||||
{
|
||||
pending.insert(key);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -426,11 +426,16 @@ impl DiskLocation {
|
||||
if shard_count == 0 {
|
||||
return false;
|
||||
}
|
||||
if let (Some(actual), Some(expected)) = (actual_shard_size, expected_shard_size) {
|
||||
if actual < expected {
|
||||
warn!(volume_id = vid.0, actual, expected, "shards smaller than the .dat's full encode; reclaiming the complete .dat");
|
||||
return false;
|
||||
}
|
||||
if let (Some(actual), Some(expected)) = (actual_shard_size, expected_shard_size)
|
||||
&& actual < expected
|
||||
{
|
||||
warn!(
|
||||
volume_id = vid.0,
|
||||
actual,
|
||||
expected,
|
||||
"shards smaller than the .dat's full encode; reclaiming the complete .dat"
|
||||
);
|
||||
return false;
|
||||
}
|
||||
true
|
||||
}
|
||||
@@ -510,10 +515,10 @@ impl DiskLocation {
|
||||
pub(crate) fn ec_generation_ts_ns(&self, collection: &str, vid: VolumeId) -> Option<i64> {
|
||||
for dir in [&self.directory, &self.idx_directory] {
|
||||
let vif = format!("{}.vif", volume_file_name(dir, collection, vid));
|
||||
if let Ok(s) = fs::read_to_string(&vif) {
|
||||
if let Ok(vi) = serde_json::from_str::<VifVolumeInfo>(&s) {
|
||||
return Some(vi.ec_shard_config.map(|c| c.encode_ts_ns).unwrap_or(0));
|
||||
}
|
||||
if let Ok(s) = fs::read_to_string(&vif)
|
||||
&& let Ok(vi) = serde_json::from_str::<VifVolumeInfo>(&s)
|
||||
{
|
||||
return Some(vi.ec_shard_config.map(|c| c.encode_ts_ns).unwrap_or(0));
|
||||
}
|
||||
if self.directory == self.idx_directory {
|
||||
break;
|
||||
@@ -542,6 +547,7 @@ impl DiskLocation {
|
||||
}
|
||||
|
||||
/// Create a new volume in this location.
|
||||
#[expect(clippy::too_many_arguments)]
|
||||
pub fn create_volume(
|
||||
&mut self,
|
||||
vid: VolumeId,
|
||||
@@ -777,18 +783,18 @@ impl DiskLocation {
|
||||
pub fn has_ecx_file_on_disk(&self, collection: &str, vid: VolumeId) -> bool {
|
||||
let idx_base = volume_file_name(&self.idx_directory, collection, vid);
|
||||
let idx_path = format!("{}.ecx", idx_base);
|
||||
if let Ok(meta) = fs::metadata(&idx_path) {
|
||||
if !meta.is_dir() {
|
||||
return true;
|
||||
}
|
||||
if let Ok(meta) = fs::metadata(&idx_path)
|
||||
&& !meta.is_dir()
|
||||
{
|
||||
return true;
|
||||
}
|
||||
if self.idx_directory != self.directory {
|
||||
let data_base = volume_file_name(&self.directory, collection, vid);
|
||||
let data_path = format!("{}.ecx", data_base);
|
||||
if let Ok(meta) = fs::metadata(&data_path) {
|
||||
if !meta.is_dir() {
|
||||
return true;
|
||||
}
|
||||
if let Ok(meta) = fs::metadata(&data_path)
|
||||
&& !meta.is_dir()
|
||||
{
|
||||
return true;
|
||||
}
|
||||
}
|
||||
false
|
||||
@@ -1107,7 +1113,7 @@ impl DiskLocation {
|
||||
|
||||
/// Close all volumes.
|
||||
pub fn close(&mut self) {
|
||||
for (_, v) in self.volumes.iter_mut() {
|
||||
for v in self.volumes.values_mut() {
|
||||
v.close();
|
||||
}
|
||||
self.volumes.clear();
|
||||
@@ -1184,10 +1190,9 @@ fn ec_data_shards_from_vif(directory: &str, idx_directory: &str, collection: &st
|
||||
.and_then(|s| serde_json::from_str::<VifVolumeInfo>(&s).ok())
|
||||
.and_then(|vi| vi.ec_shard_config)
|
||||
.map(|c| c.data_shards as usize)
|
||||
&& ds > 0
|
||||
{
|
||||
if ds > 0 {
|
||||
return ds;
|
||||
}
|
||||
return ds;
|
||||
}
|
||||
if directory == idx_directory {
|
||||
break;
|
||||
@@ -1265,7 +1270,7 @@ fn check_dat_file_exists(path: &str) -> bool {
|
||||
/// True when a `.vif` references remote-tier files: a remote-only volume
|
||||
/// that has no local `.dat` but must still load via the remote path,
|
||||
/// rather than be skipped as a lone EC sidecar.
|
||||
fn vif_references_remote_file(vif_path: &str) -> bool {
|
||||
pub(crate) fn vif_references_remote_file(vif_path: &str) -> bool {
|
||||
fs::read_to_string(vif_path)
|
||||
.ok()
|
||||
.and_then(|s| serde_json::from_str::<VifVolumeInfo>(&s).ok())
|
||||
|
||||
@@ -164,16 +164,16 @@ pub fn remove_bitrot_sidecars(base: &str) -> io::Result<()> {
|
||||
};
|
||||
let mut first_err: Option<io::Error> = None;
|
||||
let mut record = |res: io::Result<()>| {
|
||||
if let Err(e) = res {
|
||||
if first_err.is_none() {
|
||||
first_err = Some(e);
|
||||
}
|
||||
if let Err(e) = res
|
||||
&& first_err.is_none()
|
||||
{
|
||||
first_err = Some(e);
|
||||
}
|
||||
};
|
||||
record(rm(format!("{}{}", base, BITROT_SIDECAR_EXT).into()));
|
||||
let path = Path::new(base);
|
||||
if let (Some(parent), Some(fname)) = (path.parent(), path.file_name()) {
|
||||
let prefix = format!("{}{}.v", fname.to_string_lossy(), BITROT_SIDECAR_EXT);
|
||||
let prefix = format!("{}{}.v", fname.display(), BITROT_SIDECAR_EXT);
|
||||
match fs::read_dir(parent) {
|
||||
Ok(entries) => {
|
||||
for entry in entries.flatten() {
|
||||
@@ -203,7 +203,7 @@ pub fn new_encode_uuid() -> Vec<u8> {
|
||||
|
||||
/// Reports whether `block_size` is a power of two in [1 MiB, MAX_BITROT_BLOCK_SIZE].
|
||||
pub fn is_pow2_multiple_of_1mib(block_size: u32) -> bool {
|
||||
block_size >= (1 << 20) && block_size <= MAX_BITROT_BLOCK_SIZE && block_size.count_ones() == 1
|
||||
((1 << 20)..=MAX_BITROT_BLOCK_SIZE).contains(&block_size) && block_size.count_ones() == 1
|
||||
}
|
||||
|
||||
/// Returns ceil(covered_size / block_size).
|
||||
@@ -402,7 +402,7 @@ pub fn validate_manifest(
|
||||
total
|
||||
));
|
||||
}
|
||||
let mut seen = vec![false; MAX_SHARD_COUNT];
|
||||
let mut seen = [false; MAX_SHARD_COUNT];
|
||||
for s in &prot.shards {
|
||||
if s.shard_id >= total as u32 {
|
||||
return Err(format!(
|
||||
@@ -505,7 +505,21 @@ pub fn verify_shard_file_blocks(
|
||||
entry: &EcShardChecksums,
|
||||
block_size: i64,
|
||||
) -> io::Result<Vec<usize>> {
|
||||
let f = File::open(path)?;
|
||||
verify_shard_blocks(&File::open(path)?, entry, block_size)
|
||||
}
|
||||
|
||||
/// Same verification against an ALREADY-OPEN shard handle.
|
||||
///
|
||||
/// Go's `ChecksumScrub` reads through `shard.ReadAt`, i.e. the handle the
|
||||
/// EcVolumeShard already holds, so a concurrent teardown that unlinks the shard
|
||||
/// cannot turn an intentional removal into a scrub read error. A scrub that
|
||||
/// runs with the store lock released has to read the same way — see
|
||||
/// `EcChecksumScrubPlan`.
|
||||
pub fn verify_shard_blocks(
|
||||
f: &File,
|
||||
entry: &EcShardChecksums,
|
||||
block_size: i64,
|
||||
) -> io::Result<Vec<usize>> {
|
||||
let file_size = f.metadata()?.len() as i64;
|
||||
let want = unpack_u32_le(&entry.block_crc32c);
|
||||
|
||||
@@ -523,7 +537,7 @@ pub fn verify_shard_file_blocks(
|
||||
break;
|
||||
}
|
||||
let to_read = to_read as usize;
|
||||
read_full_at(&f, &mut buf[..to_read], offset as u64)?;
|
||||
read_full_at(f, &mut buf[..to_read], offset as u64)?;
|
||||
if CRC::new(&buf[..to_read]).0 != *want_crc {
|
||||
mismatched.push(i);
|
||||
}
|
||||
|
||||
@@ -73,6 +73,7 @@ pub fn find_dat_file_size_with_dirs(
|
||||
/// must live in `dir`. For the cross-disk reconciled layout where
|
||||
/// shards are split across multiple data dirs of the same node, use
|
||||
/// [`write_dat_file_from_shards_with_dirs`] instead.
|
||||
#[expect(clippy::too_many_arguments)]
|
||||
pub fn write_dat_file_from_shards(
|
||||
dir: &str,
|
||||
collection: &str,
|
||||
@@ -120,7 +121,7 @@ pub fn write_dat_file_from_shards(
|
||||
/// size. `large_block_size`/`small_block_size` are the volume's shard
|
||||
/// block layout, e.g. `EcVolume::large_block_size()` /
|
||||
/// `small_block_size()` from its .vif EC config.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
#[expect(clippy::too_many_arguments)]
|
||||
pub fn write_dat_file_from_shards_with_dirs(
|
||||
dat_dir: &str,
|
||||
collection: &str,
|
||||
@@ -145,7 +146,7 @@ pub fn write_dat_file_from_shards_with_dirs(
|
||||
)
|
||||
}
|
||||
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
#[expect(clippy::too_many_arguments)]
|
||||
fn write_dat_file(
|
||||
dat_dir: &str,
|
||||
collection: &str,
|
||||
@@ -233,10 +234,10 @@ fn write_dat_file(
|
||||
|
||||
// Read large blocks
|
||||
while encoded_remaining >= large_row_size && remaining > 0 {
|
||||
for i in 0..data_shards {
|
||||
for (i, shard) in shards[..data_shards].iter().enumerate() {
|
||||
let to_write = large_block_size.min(remaining as usize);
|
||||
let mut buf = vec![0u8; to_write];
|
||||
let n = shards[i].read_at(&mut buf, shard_offset)?;
|
||||
let n = shard.read_at(&mut buf, shard_offset)?;
|
||||
if n != to_write {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::UnexpectedEof,
|
||||
@@ -255,10 +256,10 @@ fn write_dat_file(
|
||||
|
||||
// Read small blocks
|
||||
while remaining > 0 {
|
||||
for i in 0..data_shards {
|
||||
for (i, shard) in shards[..data_shards].iter().enumerate() {
|
||||
let to_write = small_block_size.min(remaining as usize);
|
||||
let mut buf = vec![0u8; to_write];
|
||||
let n = shards[i].read_at(&mut buf, shard_offset)?;
|
||||
let n = shard.read_at(&mut buf, shard_offset)?;
|
||||
if n != to_write {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::UnexpectedEof,
|
||||
@@ -324,10 +325,7 @@ pub fn write_idx_file_from_ec_index(
|
||||
// and treat only NotFound as "no journal": Path::exists would also
|
||||
// swallow a permission/IO error and silently skip deletions, which
|
||||
// would resurrect deleted needles as live.
|
||||
let mut idx_file = std::fs::OpenOptions::new()
|
||||
.write(true)
|
||||
.append(true)
|
||||
.open(&tmp_path)?;
|
||||
let mut idx_file = std::fs::OpenOptions::new().append(true).open(&tmp_path)?;
|
||||
match std::fs::read(&ecj_path) {
|
||||
Ok(ecj_data) => {
|
||||
let count = ecj_data.len() / NEEDLE_ID_SIZE;
|
||||
|
||||
@@ -50,7 +50,7 @@ pub fn write_ec_files(
|
||||
let dat_size = dat_file.metadata()?.len() as i64;
|
||||
|
||||
let rs = ReedSolomon::new(data_shards, parity_shards)
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("reed-solomon init: {:?}", e)))?;
|
||||
.map_err(|e| io::Error::other(format!("reed-solomon init: {:?}", e)))?;
|
||||
|
||||
// Create shard files
|
||||
let total_shards = data_shards + parity_shards;
|
||||
@@ -162,7 +162,7 @@ pub fn rebuild_ec_files(
|
||||
}
|
||||
|
||||
let rs = ReedSolomon::new(data_shards, parity_shards)
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("reed-solomon init: {:?}", e)))?;
|
||||
.map_err(|e| io::Error::other(format!("reed-solomon init: {:?}", e)))?;
|
||||
|
||||
let total_shards = data_shards + parity_shards;
|
||||
let mut shards: Vec<EcVolumeShard> = (0..total_shards as u8)
|
||||
@@ -175,7 +175,7 @@ pub fn rebuild_ec_files(
|
||||
let mut shard_size = 0;
|
||||
for (i, shard) in shards.iter_mut().enumerate() {
|
||||
if !missing_shard_ids.contains(&(i as u32)) {
|
||||
if let Ok(_) = shard.open() {
|
||||
if shard.open().is_ok() {
|
||||
let size = shard.file_size();
|
||||
if size > shard_size {
|
||||
shard_size = size;
|
||||
@@ -185,7 +185,7 @@ pub fn rebuild_ec_files(
|
||||
let mut found = false;
|
||||
for &other_dir in additional_dirs {
|
||||
let mut alt = EcVolumeShard::new(other_dir, collection, volume_id, i as u8);
|
||||
if let Ok(_) = alt.open() {
|
||||
if alt.open().is_ok() {
|
||||
let size = alt.file_size();
|
||||
if size > shard_size {
|
||||
shard_size = size;
|
||||
@@ -251,12 +251,8 @@ pub fn rebuild_ec_files(
|
||||
}
|
||||
|
||||
// Reconstruct missing shards
|
||||
rs.reconstruct(&mut buffers).map_err(|e| {
|
||||
io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!("reed-solomon reconstruct: {:?}", e),
|
||||
)
|
||||
})?;
|
||||
rs.reconstruct(&mut buffers)
|
||||
.map_err(|e| io::Error::other(format!("reed-solomon reconstruct: {:?}", e)))?;
|
||||
|
||||
// Write recovered data into the missing shards
|
||||
for i in missing_shard_ids {
|
||||
@@ -284,40 +280,63 @@ pub fn rebuild_ec_files(
|
||||
/// FULL walk only reads live data-shard intervals, so on its own it can't catch
|
||||
/// bitrot in a parity shard or an unwalked region. Move to mode 4 (CHECKSUM) and
|
||||
/// drop it from mode 2 once the `.ecsum` subsystem lands.
|
||||
///
|
||||
/// `dirs` is indexed BY SHARD ID: each entry is the directory holding that
|
||||
/// shard, or `None` when no disk mounts it. A reconciled volume's shards can be
|
||||
/// split across disks, so a single directory cannot address them all.
|
||||
pub fn verify_ec_shards(
|
||||
dir: &str,
|
||||
dirs: &[Option<String>],
|
||||
collection: &str,
|
||||
volume_id: VolumeId,
|
||||
data_shards: usize,
|
||||
parity_shards: usize,
|
||||
) -> io::Result<(Vec<u32>, Vec<String>)> {
|
||||
let rs = ReedSolomon::new(data_shards, parity_shards)
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("reed-solomon init: {:?}", e)))?;
|
||||
.map_err(|e| io::Error::other(format!("reed-solomon init: {:?}", e)))?;
|
||||
|
||||
let total_shards = data_shards + parity_shards;
|
||||
let mut shards: Vec<EcVolumeShard> = (0..total_shards as u8)
|
||||
.map(|i| EcVolumeShard::new(dir, collection, volume_id, i))
|
||||
let mut shards: Vec<Option<EcVolumeShard>> = (0..total_shards)
|
||||
.map(|i| {
|
||||
dirs.get(i)
|
||||
.and_then(|d| d.as_ref())
|
||||
.map(|d| EcVolumeShard::new(d, collection, volume_id, i as u8))
|
||||
})
|
||||
.collect();
|
||||
|
||||
let mut shard_size = 0;
|
||||
let mut broken_shards = std::collections::HashSet::new();
|
||||
let mut details = Vec::new();
|
||||
|
||||
for (i, shard) in shards.iter_mut().enumerate() {
|
||||
if let Ok(_) = shard.open() {
|
||||
let size = shard.file_size();
|
||||
if size > shard_size {
|
||||
shard_size = size;
|
||||
for (i, slot) in shards.iter_mut().enumerate() {
|
||||
match slot.as_mut() {
|
||||
// Not a match guard: a binding is immutable until the guard ends,
|
||||
// and `open()` needs `&mut self`.
|
||||
Some(shard) => {
|
||||
if shard.open().is_ok() {
|
||||
let size = shard.file_size();
|
||||
if size > shard_size {
|
||||
shard_size = size;
|
||||
}
|
||||
} else {
|
||||
broken_shards.insert(i as u32);
|
||||
details.push(format!("failed to open or missing shard {}", i));
|
||||
}
|
||||
}
|
||||
None => {
|
||||
broken_shards.insert(i as u32);
|
||||
details.push(format!("shard {} is not mounted on any disk", i));
|
||||
}
|
||||
} else {
|
||||
broken_shards.insert(i as u32);
|
||||
details.push(format!("failed to open or missing shard {}", i));
|
||||
}
|
||||
}
|
||||
|
||||
if shard_size == 0 || broken_shards.len() >= parity_shards {
|
||||
// Can't do much if we don't know the size or have too many missing
|
||||
return Ok((broken_shards.into_iter().collect(), details));
|
||||
// Can't do much if we don't know the size or have too many missing.
|
||||
// Sort like the normal path below: a `HashSet` iteration order would
|
||||
// make this return shard ids in an arbitrary order, and enough `None`
|
||||
// entries in `dirs` now reach this branch for a caller to notice.
|
||||
let mut broken_vec: Vec<u32> = broken_shards.into_iter().collect();
|
||||
broken_vec.sort_unstable();
|
||||
return Ok((broken_vec, details));
|
||||
}
|
||||
|
||||
let block_size = ERASURE_CODING_SMALL_BLOCK_SIZE;
|
||||
@@ -331,7 +350,17 @@ pub fn verify_ec_shards(
|
||||
let mut read_failed = false;
|
||||
for i in 0..total_shards {
|
||||
if !broken_shards.contains(&(i as u32)) {
|
||||
if let Err(e) = shards[i].read_at(&mut buffers[i], offset) {
|
||||
// The `None` arm is defensive and unreachable: the open loop
|
||||
// put every unmounted slot in `broken_shards`, which this
|
||||
// branch already skipped. Kept because the `Option` forces
|
||||
// some handling here, and an error is the only shape that
|
||||
// cannot quietly feed an unread buffer into the parity
|
||||
// comparison below. Nothing needs to cover it.
|
||||
let read = match shards[i].as_mut() {
|
||||
Some(shard) => shard.read_at(&mut buffers[i], offset),
|
||||
None => Err(io::Error::new(io::ErrorKind::NotFound, "shard not mounted")),
|
||||
};
|
||||
if let Err(e) = read {
|
||||
broken_shards.insert(i as u32);
|
||||
details.push(format!("read error shard {}: {}", i, e));
|
||||
read_failed = true;
|
||||
@@ -345,27 +374,27 @@ pub fn verify_ec_shards(
|
||||
if !read_failed {
|
||||
// Need to convert Vec<Vec<u8>> to &[&[u8]] for rs.verify
|
||||
let slice_ptrs: Vec<&[u8]> = buffers.iter().map(|v| v.as_slice()).collect();
|
||||
if let Ok(is_valid) = rs.verify(&slice_ptrs) {
|
||||
if !is_valid {
|
||||
// Reed-Solomon verification failed. We cannot easily pinpoint which shard
|
||||
// is corrupted without recalculating parities or syndromes, so we just
|
||||
// log that this batch has corruption. Wait, we can test each parity shard!
|
||||
// Let's re-encode from the first `data_shards` and compare to the actual `parity_shards`.
|
||||
if let Ok(is_valid) = rs.verify(&slice_ptrs)
|
||||
&& !is_valid
|
||||
{
|
||||
// Reed-Solomon verification failed. We cannot easily pinpoint which shard
|
||||
// is corrupted without recalculating parities or syndromes, so we just
|
||||
// log that this batch has corruption. Wait, we can test each parity shard!
|
||||
// Let's re-encode from the first `data_shards` and compare to the actual `parity_shards`.
|
||||
|
||||
let mut verify_buffers = buffers.clone();
|
||||
// Clear the parity parts
|
||||
for i in data_shards..total_shards {
|
||||
verify_buffers[i].fill(0);
|
||||
}
|
||||
if rs.encode(&mut verify_buffers).is_ok() {
|
||||
for i in 0..total_shards {
|
||||
if buffers[i] != verify_buffers[i] {
|
||||
broken_shards.insert(i as u32);
|
||||
details.push(format!(
|
||||
"parity mismatch on shard {} at offset {}",
|
||||
i, offset
|
||||
));
|
||||
}
|
||||
let mut verify_buffers = buffers.clone();
|
||||
// Clear the parity parts
|
||||
for buf in &mut verify_buffers[data_shards..total_shards] {
|
||||
buf.fill(0);
|
||||
}
|
||||
if rs.encode(&mut verify_buffers).is_ok() {
|
||||
for i in 0..total_shards {
|
||||
if buffers[i] != verify_buffers[i] {
|
||||
broken_shards.insert(i as u32);
|
||||
details.push(format!(
|
||||
"parity mismatch on shard {} at offset {}",
|
||||
i, offset
|
||||
));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -377,7 +406,7 @@ pub fn verify_ec_shards(
|
||||
}
|
||||
|
||||
// Close all shards
|
||||
for shard in &mut shards {
|
||||
for shard in shards.iter_mut().flatten() {
|
||||
shard.close();
|
||||
}
|
||||
|
||||
@@ -457,7 +486,7 @@ pub fn rebuild_ecx_file(
|
||||
.collect();
|
||||
|
||||
for (i, shard) in shards.iter_mut().enumerate() {
|
||||
if let Err(_) = shard.open() {
|
||||
if shard.open().is_err() {
|
||||
let mut found = false;
|
||||
for &other_dir in additional_dirs {
|
||||
let mut alt = EcVolumeShard::new(other_dir, collection, volume_id, i as u8);
|
||||
@@ -474,7 +503,7 @@ pub fn rebuild_ecx_file(
|
||||
}
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::NotFound,
|
||||
format!("cannot open data shard for ecx rebuild"),
|
||||
"cannot open data shard for ecx rebuild".to_string(),
|
||||
));
|
||||
}
|
||||
}
|
||||
@@ -482,7 +511,7 @@ pub fn rebuild_ecx_file(
|
||||
|
||||
// Determine total logical data size from shard sizes
|
||||
let shard_size = shards.iter().map(|s| s.file_size()).max().unwrap_or(0);
|
||||
let total_data_size = shard_size as i64 * data_shards as i64;
|
||||
let total_data_size = shard_size * data_shards as i64;
|
||||
// The volume's shard block layout: the .vif-recorded uniform block size,
|
||||
// or the legacy two-tier sizes when 0. The row count comes from the shard
|
||||
// length; -1 disambiguates a legacy shard that is an exact large-block
|
||||
@@ -505,7 +534,7 @@ pub fn rebuild_ecx_file(
|
||||
let locate_shard_size = if dat_file_size > 0 {
|
||||
dat_file_size / data_shards as i64
|
||||
} else {
|
||||
(shard_size as i64 - 1).max(0)
|
||||
(shard_size - 1).max(0)
|
||||
};
|
||||
|
||||
// Read version from superblock (first byte of logical data)
|
||||
@@ -607,7 +636,6 @@ pub fn rebuild_ecx_file(
|
||||
/// Read bytes from EC data shards at a logical offset in the .dat file,
|
||||
/// resolving the shard/offset through the volume's block layout via
|
||||
/// locate_data — the same mapping the read path uses.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
fn read_from_data_shards(
|
||||
shards: &[EcVolumeShard],
|
||||
buf: &mut [u8],
|
||||
@@ -681,7 +709,7 @@ const ENCODE_BUFFER_SIZE: usize = 256 * 1024;
|
||||
/// 2. Process remaining data with small blocks
|
||||
///
|
||||
/// `buffer_size` must divide both block sizes.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
#[expect(clippy::too_many_arguments)]
|
||||
pub(crate) fn encode_dat_file(
|
||||
dat_file: &File,
|
||||
dat_size: i64,
|
||||
@@ -745,7 +773,7 @@ pub(crate) fn encode_dat_file(
|
||||
/// Encode one row of blocks, streaming it in ENCODE_BUFFER_SIZE sub-batches so
|
||||
/// arbitrarily large blocks never require block-sized allocations. Mirrors
|
||||
/// Go's encodeData.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
#[expect(clippy::too_many_arguments)]
|
||||
fn encode_data(
|
||||
dat_file: &File,
|
||||
row_offset: u64,
|
||||
@@ -757,7 +785,7 @@ fn encode_data(
|
||||
data_shards: usize,
|
||||
) -> io::Result<()> {
|
||||
let buffer_size = buffers[0].len();
|
||||
if block_size % buffer_size != 0 {
|
||||
if !block_size.is_multiple_of(buffer_size) {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::InvalidInput,
|
||||
format!(
|
||||
@@ -784,7 +812,7 @@ fn encode_data(
|
||||
|
||||
/// Encode one sub-batch: the same buffer-sized slice of every shard's block in
|
||||
/// this row. Mirrors Go's encodeDataOneBatch.
|
||||
#[allow(clippy::too_many_arguments)]
|
||||
#[expect(clippy::too_many_arguments)]
|
||||
fn encode_one_batch(
|
||||
dat_file: &File,
|
||||
offset: u64,
|
||||
@@ -797,21 +825,15 @@ fn encode_one_batch(
|
||||
) -> io::Result<()> {
|
||||
// Read data shards from the .dat file, zero-filling past EOF — the buffers
|
||||
// are reused across batches, so the tail must be cleared explicitly.
|
||||
for i in 0..data_shards {
|
||||
for (i, buf) in buffers[..data_shards].iter_mut().enumerate() {
|
||||
let read_offset = offset + (i * block_size) as u64;
|
||||
let n = read_at_most(dat_file, &mut buffers[i], read_offset)?;
|
||||
for b in buffers[i][n..].iter_mut() {
|
||||
*b = 0;
|
||||
}
|
||||
let n = read_at_most(dat_file, buf, read_offset)?;
|
||||
buf[n..].fill(0);
|
||||
}
|
||||
|
||||
// Encode parity shards
|
||||
rs.encode(&mut *buffers).map_err(|e| {
|
||||
io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!("reed-solomon encode: {:?}", e),
|
||||
)
|
||||
})?;
|
||||
rs.encode(&mut *buffers)
|
||||
.map_err(|e| io::Error::other(format!("reed-solomon encode: {:?}", e)))?;
|
||||
|
||||
// Write all shard buffers to files and feed the same bytes to each
|
||||
// shard's bitrot checksum builder, keeping covered_size == on-disk length.
|
||||
@@ -1457,4 +1479,111 @@ mod tests {
|
||||
"should fail when idx_dir doesn't contain .idx"
|
||||
);
|
||||
}
|
||||
|
||||
/// Write a real 10+4 encoded volume into `dir`.
|
||||
///
|
||||
/// Unlike `make_volume_with_needles` and `encode_sample_volume` this seeds
|
||||
/// a caller-chosen directory, which is what a split-disk test needs: the
|
||||
/// shards have to be scattered out of the directory they were encoded into.
|
||||
fn seed_encoded_volume(dir: &str, vid: VolumeId) {
|
||||
let mut v = Volume::new(
|
||||
dir,
|
||||
dir,
|
||||
"",
|
||||
vid,
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
)
|
||||
.unwrap();
|
||||
for i in 1..=8 {
|
||||
let data = format!("test data for needle {} with a bit more length", i);
|
||||
let mut n = Needle {
|
||||
id: NeedleId(i),
|
||||
cookie: Cookie(i as u32),
|
||||
data: data.as_bytes().to_vec(),
|
||||
data_size: data.len() as u32,
|
||||
..Needle::default()
|
||||
};
|
||||
v.write_needle(&mut n, true, false).unwrap();
|
||||
}
|
||||
v.sync_to_disk().unwrap();
|
||||
v.close();
|
||||
write_ec_files(dir, dir, "", vid, 10, 4).unwrap();
|
||||
}
|
||||
|
||||
/// Shards split across two directories must all be found. Passing one dir
|
||||
/// per shard is what lets a reconciled volume's parity be checked at all.
|
||||
#[test]
|
||||
fn test_verify_ec_shards_reads_shards_from_multiple_dirs() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let src = tmp.path().join("src");
|
||||
let d0 = tmp.path().join("d0");
|
||||
let d1 = tmp.path().join("d1");
|
||||
for d in [&src, &d0, &d1] {
|
||||
std::fs::create_dir_all(d).unwrap();
|
||||
}
|
||||
let src_s = src.to_str().unwrap();
|
||||
seed_encoded_volume(src_s, VolumeId(1));
|
||||
|
||||
// Move shards 0..=6 to d0 and 7..=13 to d1.
|
||||
let mut dirs: Vec<Option<String>> = Vec::new();
|
||||
for id in 0..14u8 {
|
||||
let target = if id < 7 { &d0 } else { &d1 };
|
||||
std::fs::rename(
|
||||
format!("{}/1.ec{:02}", src_s, id),
|
||||
format!("{}/1.ec{:02}", target.to_str().unwrap(), id),
|
||||
)
|
||||
.unwrap();
|
||||
dirs.push(Some(target.to_str().unwrap().to_string()));
|
||||
}
|
||||
|
||||
let (broken, details) = verify_ec_shards(&dirs, "", VolumeId(1), 10, 4).unwrap();
|
||||
assert!(
|
||||
broken.is_empty(),
|
||||
"split-dir shards reported broken: {:?}",
|
||||
details
|
||||
);
|
||||
}
|
||||
|
||||
/// A shard no disk holds is a missing shard, not a panic and not a silent
|
||||
/// pass: it is REPORTED, by id, with a message that distinguishes "no disk
|
||||
/// holds this shard" from "the disk holds it but it won't open".
|
||||
///
|
||||
/// Read the scope literally. This does NOT show that the mounted shards
|
||||
/// verify clean. `dirs[5] = None` puts shard 5 in `broken_shards` before
|
||||
/// the block loop starts, so every iteration takes the
|
||||
/// `else { read_failed = true; }` arm and the Reed-Solomon comparison never
|
||||
/// runs at all. `broken == vec![5]` therefore holds because the other 13
|
||||
/// were never verified, not because they verified clean -- a parity check
|
||||
/// over intact shards is what
|
||||
/// `test_verify_ec_shards_reads_shards_from_multiple_dirs` and the
|
||||
/// end-to-end split-disk FULL scrub establish.
|
||||
#[test]
|
||||
fn test_verify_ec_shards_treats_a_none_dir_as_missing() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
seed_encoded_volume(dir, VolumeId(1));
|
||||
|
||||
let mut dirs: Vec<Option<String>> = (0..14).map(|_| Some(dir.to_string())).collect();
|
||||
dirs[5] = None;
|
||||
|
||||
let (broken, details) = verify_ec_shards(&dirs, "", VolumeId(1), 10, 4).unwrap();
|
||||
assert_eq!(
|
||||
broken,
|
||||
vec![5],
|
||||
"an unmounted shard must be reported, and only it: {:?}",
|
||||
details
|
||||
);
|
||||
// "no disk holds this shard" and "the disk holds it but it won't open"
|
||||
// are different operator problems, which is why they carry different
|
||||
// messages. Asserting only the id would let one masquerade as the other.
|
||||
assert!(
|
||||
details.iter().any(|d| d.contains("not mounted")),
|
||||
"an unmounted shard must be distinguished from an unopenable one, got {:?}",
|
||||
details
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -78,7 +78,7 @@ impl EcVolumeShard {
|
||||
let file = self
|
||||
.ecd_file
|
||||
.as_ref()
|
||||
.ok_or_else(|| io::Error::new(io::ErrorKind::Other, "shard file not open"))?;
|
||||
.ok_or_else(|| io::Error::other("shard file not open"))?;
|
||||
|
||||
#[cfg(unix)]
|
||||
{
|
||||
@@ -102,7 +102,7 @@ impl EcVolumeShard {
|
||||
let file = self
|
||||
.ecd_file
|
||||
.as_mut()
|
||||
.ok_or_else(|| io::Error::new(io::ErrorKind::Other, "shard file not open"))?;
|
||||
.ok_or_else(|| io::Error::other("shard file not open"))?;
|
||||
file.write_all(data)?;
|
||||
self.ecd_file_size += data.len() as i64;
|
||||
Ok(())
|
||||
@@ -112,6 +112,21 @@ impl EcVolumeShard {
|
||||
self.ecd_file_size
|
||||
}
|
||||
|
||||
/// A duplicate of the mounted shard handle, for a reader that has to
|
||||
/// outlive the store guard.
|
||||
///
|
||||
/// This is the same descriptor `read_at` serves from, so it carries the
|
||||
/// `O_NOATIME` from `open_volume_file` and keeps pointing at the shard
|
||||
/// that was mounted, whatever later happens to the path. `dup` shares the
|
||||
/// kernel file offset, which is why every read through it must be
|
||||
/// positional (`read_at`), never seek-based.
|
||||
pub fn try_clone_file(&self) -> io::Result<File> {
|
||||
self.ecd_file
|
||||
.as_ref()
|
||||
.ok_or_else(|| io::Error::other("shard file not open"))?
|
||||
.try_clone()
|
||||
}
|
||||
|
||||
/// Protobuf descriptor for this shard. Mirrors Go's ToEcShardInfo.
|
||||
pub fn to_ec_shard_info(&self) -> crate::pb::volume_server_pb::EcShardInfo {
|
||||
crate::pb::volume_server_pb::EcShardInfo {
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -21,7 +21,7 @@ impl CRC {
|
||||
/// Legacy `.Value()` function — deprecated in Go but needed for backward compat check.
|
||||
/// Formula: (crc >> 15 | crc << 17) + 0xa282ead8
|
||||
pub fn legacy_value(&self) -> u32 {
|
||||
(self.0 >> 15 | self.0 << 17).wrapping_add(0xa282ead8)
|
||||
self.0.rotate_right(15).wrapping_add(0xa282ead8)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -67,7 +67,9 @@ mod tests {
|
||||
fn test_crc_legacy_value() {
|
||||
let crc = CRC(0x12345678);
|
||||
let v = crc.legacy_value();
|
||||
let expected = (0x12345678u32 >> 15 | 0x12345678u32 << 17).wrapping_add(0xa282ead8);
|
||||
// (0x12345678 >> 15 | 0x12345678 << 17) + 0xa282ead8, worked out by hand so
|
||||
// the test checks the rotate rather than restating it.
|
||||
let expected = 0x4f730f40_u32;
|
||||
assert_eq!(v, expected);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
pub mod crc;
|
||||
#[expect(clippy::module_inception, reason = "needle/needle.rs mirrors the Go package layout")]
|
||||
pub mod needle;
|
||||
pub mod ttl;
|
||||
|
||||
|
||||
@@ -560,7 +560,7 @@ impl Needle {
|
||||
|
||||
// Padding to 8-byte alignment
|
||||
let padding = padding_length(self.size, version).0 as usize;
|
||||
buf.extend(std::iter::repeat(0u8).take(padding));
|
||||
buf.extend(std::iter::repeat_n(0u8, padding));
|
||||
|
||||
buf
|
||||
}
|
||||
@@ -824,11 +824,13 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn test_needle_write_read_round_trip_v3() {
|
||||
let mut n = Needle::default();
|
||||
n.cookie = Cookie(42);
|
||||
n.id = NeedleId(100);
|
||||
n.data = b"hello world".to_vec();
|
||||
n.flags = 0;
|
||||
let mut n = Needle {
|
||||
cookie: Cookie(42),
|
||||
id: NeedleId(100),
|
||||
data: b"hello world".to_vec(),
|
||||
flags: 0,
|
||||
..Needle::default()
|
||||
};
|
||||
n.set_has_name();
|
||||
n.name = b"test.txt".to_vec();
|
||||
n.name_size = 8;
|
||||
@@ -867,11 +869,13 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn test_needle_write_read_round_trip_v2() {
|
||||
let mut n = Needle::default();
|
||||
n.cookie = Cookie(77);
|
||||
n.id = NeedleId(200);
|
||||
n.data = b"data v2".to_vec();
|
||||
n.flags = 0;
|
||||
let mut n = Needle {
|
||||
cookie: Cookie(77),
|
||||
id: NeedleId(200),
|
||||
data: b"data v2".to_vec(),
|
||||
flags: 0,
|
||||
..Needle::default()
|
||||
};
|
||||
|
||||
let bytes = n.write_bytes(VERSION_2);
|
||||
let expected_size = get_actual_size(n.size, VERSION_2);
|
||||
@@ -886,10 +890,12 @@ mod tests {
|
||||
|
||||
#[test]
|
||||
fn test_read_bytes_meta_only_handles_tombstone_v3() {
|
||||
let mut tombstone = Needle::default();
|
||||
tombstone.cookie = Cookie(0x1234abcd);
|
||||
tombstone.id = NeedleId(300);
|
||||
tombstone.append_at_ns = 999_999;
|
||||
let mut tombstone = Needle {
|
||||
cookie: Cookie(0x1234abcd),
|
||||
id: NeedleId(300),
|
||||
append_at_ns: 999_999,
|
||||
..Needle::default()
|
||||
};
|
||||
|
||||
let bytes = tombstone.write_bytes(VERSION_3);
|
||||
|
||||
|
||||
@@ -81,7 +81,7 @@ impl TTL {
|
||||
return Ok(TTL::EMPTY);
|
||||
}
|
||||
let last_byte = s.as_bytes()[s.len() - 1];
|
||||
let (num_str, unit_byte) = if last_byte >= b'0' && last_byte <= b'9' {
|
||||
let (num_str, unit_byte) = if last_byte.is_ascii_digit() {
|
||||
// All digits — default to minutes (matching Go)
|
||||
(s, b'm')
|
||||
} else {
|
||||
@@ -144,40 +144,73 @@ fn fit_ttl_count(count: u32, unit: u8) -> TTL {
|
||||
const MINUTE_SECS: u64 = 60;
|
||||
|
||||
// First pass: try exact fits from largest to smallest
|
||||
if seconds % YEAR_SECS == 0 && seconds / YEAR_SECS < 256 {
|
||||
return TTL { count: (seconds / YEAR_SECS) as u8, unit: TTL_UNIT_YEAR };
|
||||
if seconds.is_multiple_of(YEAR_SECS) && seconds / YEAR_SECS < 256 {
|
||||
return TTL {
|
||||
count: (seconds / YEAR_SECS) as u8,
|
||||
unit: TTL_UNIT_YEAR,
|
||||
};
|
||||
}
|
||||
if seconds % MONTH_SECS == 0 && seconds / MONTH_SECS < 256 {
|
||||
return TTL { count: (seconds / MONTH_SECS) as u8, unit: TTL_UNIT_MONTH };
|
||||
if seconds.is_multiple_of(MONTH_SECS) && seconds / MONTH_SECS < 256 {
|
||||
return TTL {
|
||||
count: (seconds / MONTH_SECS) as u8,
|
||||
unit: TTL_UNIT_MONTH,
|
||||
};
|
||||
}
|
||||
if seconds % WEEK_SECS == 0 && seconds / WEEK_SECS < 256 {
|
||||
return TTL { count: (seconds / WEEK_SECS) as u8, unit: TTL_UNIT_WEEK };
|
||||
if seconds.is_multiple_of(WEEK_SECS) && seconds / WEEK_SECS < 256 {
|
||||
return TTL {
|
||||
count: (seconds / WEEK_SECS) as u8,
|
||||
unit: TTL_UNIT_WEEK,
|
||||
};
|
||||
}
|
||||
if seconds % DAY_SECS == 0 && seconds / DAY_SECS < 256 {
|
||||
return TTL { count: (seconds / DAY_SECS) as u8, unit: TTL_UNIT_DAY };
|
||||
if seconds.is_multiple_of(DAY_SECS) && seconds / DAY_SECS < 256 {
|
||||
return TTL {
|
||||
count: (seconds / DAY_SECS) as u8,
|
||||
unit: TTL_UNIT_DAY,
|
||||
};
|
||||
}
|
||||
if seconds % HOUR_SECS == 0 && seconds / HOUR_SECS < 256 {
|
||||
return TTL { count: (seconds / HOUR_SECS) as u8, unit: TTL_UNIT_HOUR };
|
||||
if seconds.is_multiple_of(HOUR_SECS) && seconds / HOUR_SECS < 256 {
|
||||
return TTL {
|
||||
count: (seconds / HOUR_SECS) as u8,
|
||||
unit: TTL_UNIT_HOUR,
|
||||
};
|
||||
}
|
||||
// Minutes: truncating division
|
||||
if seconds / MINUTE_SECS < 256 {
|
||||
return TTL { count: (seconds / MINUTE_SECS) as u8, unit: TTL_UNIT_MINUTE };
|
||||
return TTL {
|
||||
count: (seconds / MINUTE_SECS) as u8,
|
||||
unit: TTL_UNIT_MINUTE,
|
||||
};
|
||||
}
|
||||
// Second pass: truncating division from smallest to largest
|
||||
if seconds / HOUR_SECS < 256 {
|
||||
return TTL { count: (seconds / HOUR_SECS) as u8, unit: TTL_UNIT_HOUR };
|
||||
return TTL {
|
||||
count: (seconds / HOUR_SECS) as u8,
|
||||
unit: TTL_UNIT_HOUR,
|
||||
};
|
||||
}
|
||||
if seconds / DAY_SECS < 256 {
|
||||
return TTL { count: (seconds / DAY_SECS) as u8, unit: TTL_UNIT_DAY };
|
||||
return TTL {
|
||||
count: (seconds / DAY_SECS) as u8,
|
||||
unit: TTL_UNIT_DAY,
|
||||
};
|
||||
}
|
||||
if seconds / WEEK_SECS < 256 {
|
||||
return TTL { count: (seconds / WEEK_SECS) as u8, unit: TTL_UNIT_WEEK };
|
||||
return TTL {
|
||||
count: (seconds / WEEK_SECS) as u8,
|
||||
unit: TTL_UNIT_WEEK,
|
||||
};
|
||||
}
|
||||
if seconds / MONTH_SECS < 256 {
|
||||
return TTL { count: (seconds / MONTH_SECS) as u8, unit: TTL_UNIT_MONTH };
|
||||
return TTL {
|
||||
count: (seconds / MONTH_SECS) as u8,
|
||||
unit: TTL_UNIT_MONTH,
|
||||
};
|
||||
}
|
||||
if seconds / YEAR_SECS < 256 {
|
||||
return TTL { count: (seconds / YEAR_SECS) as u8, unit: TTL_UNIT_YEAR };
|
||||
return TTL {
|
||||
count: (seconds / YEAR_SECS) as u8,
|
||||
unit: TTL_UNIT_YEAR,
|
||||
};
|
||||
}
|
||||
TTL::EMPTY
|
||||
}
|
||||
|
||||
@@ -97,12 +97,13 @@ impl NeedleMapMetric {
|
||||
self.file_byte_count
|
||||
.fetch_add(new_size.0 as u64, Ordering::Relaxed);
|
||||
// Go: if oldSize > 0 && oldSize.IsValid() { LogDeletionCounter(oldSize) }
|
||||
if let Some(old_val) = old {
|
||||
if old_val.size.0 > 0 && old_val.size.is_valid() {
|
||||
self.deletion_count.fetch_add(1, Ordering::Relaxed);
|
||||
self.deletion_byte_count
|
||||
.fetch_add(old_val.size.0 as u64, Ordering::Relaxed);
|
||||
}
|
||||
if let Some(old_val) = old
|
||||
&& old_val.size.0 > 0
|
||||
&& old_val.size.is_valid()
|
||||
{
|
||||
self.deletion_count.fetch_add(1, Ordering::Relaxed);
|
||||
self.deletion_byte_count
|
||||
.fetch_add(old_val.size.0 as u64, Ordering::Relaxed);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -225,6 +226,12 @@ pub struct CompactNeedleMap {
|
||||
idx_file_offset: u64,
|
||||
}
|
||||
|
||||
impl Default for CompactNeedleMap {
|
||||
fn default() -> Self {
|
||||
Self::new()
|
||||
}
|
||||
}
|
||||
|
||||
impl CompactNeedleMap {
|
||||
/// Create a new empty in-memory map.
|
||||
pub fn new() -> Self {
|
||||
@@ -465,9 +472,9 @@ impl RedbNeedleMap {
|
||||
/// loses at most the writes since the last checkpoint from redb, and
|
||||
/// the next load replays them from .idx.
|
||||
fn begin_write_no_fsync(db: &Database) -> io::Result<redb::WriteTransaction> {
|
||||
let mut txn = db.begin_write().map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb begin_write: {}", e))
|
||||
})?;
|
||||
let mut txn = db
|
||||
.begin_write()
|
||||
.map_err(|e| io::Error::other(format!("redb begin_write: {}", e)))?;
|
||||
let _ = txn.set_durability(Durability::None);
|
||||
Ok(txn)
|
||||
}
|
||||
@@ -501,7 +508,7 @@ impl RedbNeedleMap {
|
||||
pub fn checkpoint(&mut self, sync_idx: bool) -> io::Result<()> {
|
||||
let txn = self.begin_checkpoint(sync_idx)?;
|
||||
txn.commit()
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("redb commit: {}", e)))?;
|
||||
.map_err(|e| io::Error::other(format!("redb commit: {}", e)))?;
|
||||
self.writes_since_checkpoint = 0;
|
||||
Ok(())
|
||||
}
|
||||
@@ -516,17 +523,17 @@ impl RedbNeedleMap {
|
||||
if sync_idx {
|
||||
self.sync()?;
|
||||
}
|
||||
let mut txn = self.db_or_err()?.begin_write().map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb begin_write: {}", e))
|
||||
})?;
|
||||
let mut txn = self
|
||||
.db_or_err()?
|
||||
.begin_write()
|
||||
.map_err(|e| io::Error::other(format!("redb begin_write: {}", e)))?;
|
||||
txn.set_quick_repair(true);
|
||||
if self.idx_file.is_some() {
|
||||
let mut meta = txn.open_table(META_TABLE).map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb open meta: {}", e))
|
||||
})?;
|
||||
meta.insert(META_IDX_SIZE, self.idx_file_offset).map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb insert meta: {}", e))
|
||||
})?;
|
||||
let mut meta = txn
|
||||
.open_table(META_TABLE)
|
||||
.map_err(|e| io::Error::other(format!("redb open meta: {}", e)))?;
|
||||
meta.insert(META_IDX_SIZE, self.idx_file_offset)
|
||||
.map_err(|e| io::Error::other(format!("redb insert meta: {}", e)))?;
|
||||
}
|
||||
Ok(txn)
|
||||
}
|
||||
@@ -538,22 +545,20 @@ impl RedbNeedleMap {
|
||||
let db = Database::builder()
|
||||
.set_cache_size(cache_bytes)
|
||||
.create(db_path)
|
||||
.map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb create error: {}", e))
|
||||
})?;
|
||||
.map_err(|e| io::Error::other(format!("redb create error: {}", e)))?;
|
||||
|
||||
// Ensure tables exist
|
||||
let txn = Self::begin_write_no_fsync(&db)?;
|
||||
{
|
||||
let _table = txn.open_table(NEEDLE_TABLE).map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb open_table: {}", e))
|
||||
})?;
|
||||
let _meta = txn.open_table(META_TABLE).map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb open_table meta: {}", e))
|
||||
})?;
|
||||
let _table = txn
|
||||
.open_table(NEEDLE_TABLE)
|
||||
.map_err(|e| io::Error::other(format!("redb open_table: {}", e)))?;
|
||||
let _meta = txn
|
||||
.open_table(META_TABLE)
|
||||
.map_err(|e| io::Error::other(format!("redb open_table meta: {}", e)))?;
|
||||
}
|
||||
txn.commit()
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("redb commit: {}", e)))?;
|
||||
.map_err(|e| io::Error::other(format!("redb commit: {}", e)))?;
|
||||
|
||||
Ok(RedbNeedleMap {
|
||||
db: Some(db),
|
||||
@@ -572,16 +577,14 @@ impl RedbNeedleMap {
|
||||
fn save_idx_size_meta(&self, idx_size: u64) -> io::Result<()> {
|
||||
let txn = Self::begin_write_no_fsync(self.db_or_err()?)?;
|
||||
{
|
||||
let mut meta = txn.open_table(META_TABLE).map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb open meta: {}", e))
|
||||
})?;
|
||||
meta.insert(META_IDX_SIZE, idx_size).map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb insert meta: {}", e))
|
||||
})?;
|
||||
let mut meta = txn
|
||||
.open_table(META_TABLE)
|
||||
.map_err(|e| io::Error::other(format!("redb open meta: {}", e)))?;
|
||||
meta.insert(META_IDX_SIZE, idx_size)
|
||||
.map_err(|e| io::Error::other(format!("redb insert meta: {}", e)))?;
|
||||
}
|
||||
txn.commit().map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb commit meta: {}", e))
|
||||
})?;
|
||||
txn.commit()
|
||||
.map_err(|e| io::Error::other(format!("redb commit meta: {}", e)))?;
|
||||
Ok(())
|
||||
}
|
||||
|
||||
@@ -590,22 +593,18 @@ impl RedbNeedleMap {
|
||||
let txn = self
|
||||
.db_or_err()?
|
||||
.begin_read()
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("redb begin_read: {}", e)))?;
|
||||
.map_err(|e| io::Error::other(format!("redb begin_read: {}", e)))?;
|
||||
let meta = txn
|
||||
.open_table(META_TABLE)
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("redb open meta: {}", e)))?;
|
||||
.map_err(|e| io::Error::other(format!("redb open meta: {}", e)))?;
|
||||
// experimental-api-5 drops inherent ReadOnlyTable::get ('static guard).
|
||||
// ReadableTable::get guard borrows `meta`; bind the match so the
|
||||
// temporary Result is dropped before `meta`.
|
||||
let result = match meta.get(META_IDX_SIZE) {
|
||||
// ReadableTable::get guard borrows `meta`; edition 2024 drops the tail
|
||||
// expression's temporaries before `meta`, so no extra binding is needed.
|
||||
match meta.get(META_IDX_SIZE) {
|
||||
Ok(Some(guard)) => Ok(Some(guard.value())),
|
||||
Ok(None) => Ok(None),
|
||||
Err(e) => Err(io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!("redb get meta: {}", e),
|
||||
)),
|
||||
};
|
||||
result
|
||||
Err(e) => Err(io::Error::other(format!("redb get meta: {}", e))),
|
||||
}
|
||||
}
|
||||
|
||||
/// Load from an .idx file, reusing an existing .rdb if it is consistent.
|
||||
@@ -648,7 +647,7 @@ impl RedbNeedleMap {
|
||||
let db = Database::builder()
|
||||
.set_cache_size(cache_bytes)
|
||||
.open(db_path)
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("redb open: {}", e)))?;
|
||||
.map_err(|e| io::Error::other(format!("redb open: {}", e)))?;
|
||||
|
||||
let mut nm = RedbNeedleMap {
|
||||
db: Some(db),
|
||||
@@ -663,14 +662,11 @@ impl RedbNeedleMap {
|
||||
|
||||
let stored_idx_size = nm
|
||||
.read_idx_size_meta()?
|
||||
.ok_or_else(|| io::Error::new(io::ErrorKind::Other, "no idx_size in redb meta"))?;
|
||||
.ok_or_else(|| io::Error::other("no idx_size in redb meta"))?;
|
||||
|
||||
if stored_idx_size > idx_size {
|
||||
// .idx shrank — corrupted or truncated, need full rebuild
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
"idx file smaller than stored size",
|
||||
));
|
||||
return Err(io::Error::other("idx file smaller than stored size"));
|
||||
}
|
||||
|
||||
// Counters come from the whole .idx history, never from the table,
|
||||
@@ -683,40 +679,37 @@ impl RedbNeedleMap {
|
||||
let start_entry = stored_idx_size / NEEDLE_MAP_ENTRY_SIZE as u64;
|
||||
let txn = Self::begin_write_no_fsync(nm.db.as_ref().unwrap())?;
|
||||
{
|
||||
let mut table = txn.open_table(NEEDLE_TABLE).map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb open_table: {}", e))
|
||||
})?;
|
||||
let mut table = txn
|
||||
.open_table(NEEDLE_TABLE)
|
||||
.map_err(|e| io::Error::other(format!("redb open_table: {}", e)))?;
|
||||
idx::walk_index_file(reader, start_entry, |key, offset, size| {
|
||||
let key_u64: u64 = key.into();
|
||||
if offset.is_zero() || size.is_deleted() {
|
||||
// Delete: store a tombstone (negative size, original
|
||||
// offset) over a live value; already deleted is a no-op.
|
||||
if let Ok(Some(old)) = nm.get_via_table(&table, key_u64) {
|
||||
if old.size.is_valid() {
|
||||
let deleted_nv = NeedleValue {
|
||||
offset: old.offset,
|
||||
size: Size(-(old.size.0)),
|
||||
};
|
||||
let packed = pack_needle_value(&deleted_nv);
|
||||
table.insert(key_u64, packed.as_slice()).map_err(|e| {
|
||||
io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!("redb insert: {}", e),
|
||||
)
|
||||
})?;
|
||||
}
|
||||
if let Ok(Some(old)) = nm.get_via_table(&table, key_u64)
|
||||
&& old.size.is_valid()
|
||||
{
|
||||
let deleted_nv = NeedleValue {
|
||||
offset: old.offset,
|
||||
size: Size(-(old.size.0)),
|
||||
};
|
||||
let packed = pack_needle_value(&deleted_nv);
|
||||
table
|
||||
.insert(key_u64, packed.as_slice())
|
||||
.map_err(|e| io::Error::other(format!("redb insert: {}", e)))?;
|
||||
}
|
||||
} else {
|
||||
let packed = pack_needle_value(&NeedleValue { offset, size });
|
||||
table.insert(key_u64, packed.as_slice()).map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb insert: {}", e))
|
||||
})?;
|
||||
table
|
||||
.insert(key_u64, packed.as_slice())
|
||||
.map_err(|e| io::Error::other(format!("redb insert: {}", e)))?;
|
||||
}
|
||||
Ok(())
|
||||
})?;
|
||||
}
|
||||
txn.commit()
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("redb commit: {}", e)))?;
|
||||
.map_err(|e| io::Error::other(format!("redb commit: {}", e)))?;
|
||||
|
||||
nm.save_idx_size_meta(idx_size)?;
|
||||
}
|
||||
@@ -734,10 +727,7 @@ impl RedbNeedleMap {
|
||||
match table.get(key_u64) {
|
||||
Ok(Some(guard)) => Ok(packed_to_needle_value(guard.value())),
|
||||
Ok(None) => Ok(None),
|
||||
Err(e) => Err(io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!("redb get: {}", e),
|
||||
)),
|
||||
Err(e) => Err(io::Error::other(format!("redb get: {}", e))),
|
||||
}
|
||||
}
|
||||
|
||||
@@ -790,13 +780,13 @@ impl RedbNeedleMap {
|
||||
|
||||
let txn = Self::begin_write_no_fsync(nm.db.as_ref().unwrap())?;
|
||||
{
|
||||
let mut table = txn.open_table(NEEDLE_TABLE).map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb open_table: {}", e))
|
||||
})?;
|
||||
let mut table = txn
|
||||
.open_table(NEEDLE_TABLE)
|
||||
.map_err(|e| io::Error::other(format!("redb open_table: {}", e)))?;
|
||||
if !unlinked {
|
||||
table.retain(|_, _| false).map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb retain: {}", e))
|
||||
})?;
|
||||
table
|
||||
.retain(|_, _| false)
|
||||
.map_err(|e| io::Error::other(format!("redb retain: {}", e)))?;
|
||||
}
|
||||
|
||||
#[cfg(not(feature = "redb-experimental-cursor"))]
|
||||
@@ -804,9 +794,9 @@ impl RedbNeedleMap {
|
||||
for (key, nv) in &entries {
|
||||
let key_u64: u64 = (*key).into();
|
||||
let packed = pack_needle_value(nv);
|
||||
table.insert(key_u64, packed.as_slice()).map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb insert: {}", e))
|
||||
})?;
|
||||
table
|
||||
.insert(key_u64, packed.as_slice())
|
||||
.map_err(|e| io::Error::other(format!("redb insert: {}", e)))?;
|
||||
}
|
||||
}
|
||||
#[cfg(feature = "redb-experimental-cursor")]
|
||||
@@ -835,7 +825,7 @@ impl RedbNeedleMap {
|
||||
}
|
||||
}
|
||||
txn.commit()
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("redb commit: {}", e)))?;
|
||||
.map_err(|e| io::Error::other(format!("redb commit: {}", e)))?;
|
||||
|
||||
nm.save_idx_size_meta(idx_size)?;
|
||||
Ok(())
|
||||
@@ -901,23 +891,16 @@ impl RedbNeedleMap {
|
||||
Ok(t) => t,
|
||||
Err(e) => {
|
||||
self.truncate_idx_to_offset();
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!("redb open_table: {}", e),
|
||||
));
|
||||
return Err(io::Error::other(format!("redb open_table: {}", e)));
|
||||
}
|
||||
};
|
||||
let result = match table.insert(key_u64, packed.as_slice()) {
|
||||
match table.insert(key_u64, packed.as_slice()) {
|
||||
Ok(prev) => prev.and_then(|g| packed_to_needle_value(g.value())),
|
||||
Err(e) => {
|
||||
self.truncate_idx_to_offset();
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!("redb insert: {}", e),
|
||||
));
|
||||
return Err(io::Error::other(format!("redb insert: {}", e)));
|
||||
}
|
||||
};
|
||||
result
|
||||
}
|
||||
};
|
||||
match txn.commit() {
|
||||
Ok(()) => old,
|
||||
@@ -925,8 +908,7 @@ impl RedbNeedleMap {
|
||||
// Transaction rolled back, database still usable:
|
||||
// truncate the orphan .idx row.
|
||||
self.truncate_idx_to_offset();
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
return Err(io::Error::other(
|
||||
"redb commit: Transaction was poisoned by a panic",
|
||||
));
|
||||
}
|
||||
@@ -935,12 +917,9 @@ impl RedbNeedleMap {
|
||||
// visible and redb refuses further writes. Keep
|
||||
// the .idx row (do NOT truncate) and reopen from
|
||||
// .idx to repair redb's internal state.
|
||||
let err = io::Error::new(io::ErrorKind::Other, format!("redb commit: {}", e));
|
||||
let err = io::Error::other(format!("redb commit: {}", e));
|
||||
if let Err(reopen_err) = self.reopen_from_idx() {
|
||||
tracing::warn!(
|
||||
"redb reopen after put commit error failed: {}",
|
||||
reopen_err
|
||||
);
|
||||
tracing::warn!("redb reopen after put commit error failed: {}", reopen_err);
|
||||
}
|
||||
return Err(err);
|
||||
}
|
||||
@@ -968,39 +947,32 @@ impl RedbNeedleMap {
|
||||
let txn = self
|
||||
.db_or_err()?
|
||||
.begin_read()
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("redb begin_read: {}", e)))?;
|
||||
.map_err(|e| io::Error::other(format!("redb begin_read: {}", e)))?;
|
||||
let table = txn
|
||||
.open_table(NEEDLE_TABLE)
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("redb open_table: {}", e)))?;
|
||||
.map_err(|e| io::Error::other(format!("redb open_table: {}", e)))?;
|
||||
// experimental-api-5 drops inherent ReadOnlyTable::get ('static guard).
|
||||
// ReadableTable::get guard borrows `table`; bind the match so the
|
||||
// temporary Result is dropped before `table`.
|
||||
let result = match table.get(key_u64) {
|
||||
// ReadableTable::get guard borrows `table`; edition 2024 drops the tail
|
||||
// expression's temporaries before `table`, so no extra binding is needed.
|
||||
match table.get(key_u64) {
|
||||
Ok(Some(guard)) => Ok(packed_to_needle_value(guard.value())),
|
||||
Ok(None) => Ok(None),
|
||||
Err(e) => Err(io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!("redb get: {}", e),
|
||||
)),
|
||||
};
|
||||
result
|
||||
Err(e) => Err(io::Error::other(format!("redb get: {}", e))),
|
||||
}
|
||||
}
|
||||
|
||||
/// Mark a needle as deleted. Appends tombstone to .idx file, negates size in redb.
|
||||
pub fn delete(&mut self, key: NeedleId, offset: Offset) -> io::Result<Option<Size>> {
|
||||
let key_u64: u64 = key.into();
|
||||
let txn = Self::begin_write_no_fsync(self.db_or_err()?)?;
|
||||
let mut table = txn.open_table(NEEDLE_TABLE).map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb open_table: {}", e))
|
||||
})?;
|
||||
let mut table = txn
|
||||
.open_table(NEEDLE_TABLE)
|
||||
.map_err(|e| io::Error::other(format!("redb open_table: {}", e)))?;
|
||||
let old = match table.get(key_u64) {
|
||||
Ok(Some(guard)) => packed_to_needle_value(guard.value()),
|
||||
Ok(None) => None,
|
||||
Err(e) => {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!("redb get: {}", e),
|
||||
));
|
||||
return Err(io::Error::other(format!("redb get: {}", e)));
|
||||
}
|
||||
};
|
||||
let Some(old) = old.filter(|nv| nv.size.is_valid()) else {
|
||||
@@ -1021,10 +993,7 @@ impl RedbNeedleMap {
|
||||
drop(table);
|
||||
if let Err(e) = insert_res {
|
||||
self.truncate_idx_to_offset();
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!("redb insert: {}", e),
|
||||
));
|
||||
return Err(io::Error::other(format!("redb insert: {}", e)));
|
||||
}
|
||||
match txn.commit() {
|
||||
Ok(()) => {}
|
||||
@@ -1032,8 +1001,7 @@ impl RedbNeedleMap {
|
||||
// Transaction rolled back, database still usable:
|
||||
// truncate the orphan .idx row.
|
||||
self.truncate_idx_to_offset();
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
return Err(io::Error::other(
|
||||
"redb commit: Transaction was poisoned by a panic",
|
||||
));
|
||||
}
|
||||
@@ -1042,7 +1010,7 @@ impl RedbNeedleMap {
|
||||
// and redb refuses further writes. Keep the .idx row
|
||||
// (do NOT truncate) and reopen from .idx to repair
|
||||
// redb's internal state.
|
||||
let err = io::Error::new(io::ErrorKind::Other, format!("redb commit: {}", e));
|
||||
let err = io::Error::other(format!("redb commit: {}", e));
|
||||
if let Err(reopen_err) = self.reopen_from_idx() {
|
||||
tracing::warn!(
|
||||
"redb reopen after delete commit error failed: {}",
|
||||
@@ -1105,10 +1073,10 @@ impl RedbNeedleMap {
|
||||
/// after the orphan, `idx_file_offset` advances past it, and a later
|
||||
/// checkpoint records an offset that makes the reload skip the orphan.
|
||||
fn truncate_idx_to_offset(&mut self) {
|
||||
if let Some(ref mut idx_file) = self.idx_file {
|
||||
if let Err(e) = idx_file.truncate_to(self.idx_file_offset) {
|
||||
tracing::warn!("failed to truncate orphan .idx row: {}", e);
|
||||
}
|
||||
if let Some(ref mut idx_file) = self.idx_file
|
||||
&& let Err(e) = idx_file.truncate_to(self.idx_file_offset)
|
||||
{
|
||||
tracing::warn!("failed to truncate orphan .idx row: {}", e);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1198,10 +1166,10 @@ impl RedbNeedleMap {
|
||||
let txn = self
|
||||
.db_or_err()?
|
||||
.begin_read()
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("redb begin_read: {}", e)))?;
|
||||
.map_err(|e| io::Error::other(format!("redb begin_read: {}", e)))?;
|
||||
let table = txn
|
||||
.open_table(NEEDLE_TABLE)
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("redb open_table: {}", e)))?;
|
||||
.map_err(|e| io::Error::other(format!("redb open_table: {}", e)))?;
|
||||
|
||||
let mut file = std::fs::OpenOptions::new()
|
||||
.write(true)
|
||||
@@ -1212,18 +1180,17 @@ impl RedbNeedleMap {
|
||||
// redb iterates in key order (u64 ascending)
|
||||
let iter = table
|
||||
.iter()
|
||||
.map_err(|e| io::Error::new(io::ErrorKind::Other, format!("redb iter: {}", e)))?;
|
||||
.map_err(|e| io::Error::other(format!("redb iter: {}", e)))?;
|
||||
|
||||
for entry in iter {
|
||||
let (key_guard, val_guard) = entry.map_err(|e| {
|
||||
io::Error::new(io::ErrorKind::Other, format!("redb iter next: {}", e))
|
||||
})?;
|
||||
let (key_guard, val_guard) =
|
||||
entry.map_err(|e| io::Error::other(format!("redb iter next: {}", e)))?;
|
||||
let key_u64: u64 = key_guard.value();
|
||||
let bytes: &[u8] = val_guard.value();
|
||||
if let Some(nv) = packed_to_needle_value(bytes) {
|
||||
if nv.size.is_valid() {
|
||||
idx::write_index_entry(&mut file, NeedleId(key_u64), nv.offset, nv.size)?;
|
||||
}
|
||||
if let Some(nv) = packed_to_needle_value(bytes)
|
||||
&& nv.size.is_valid()
|
||||
{
|
||||
idx::write_index_entry(&mut file, NeedleId(key_u64), nv.offset, nv.size)?;
|
||||
}
|
||||
}
|
||||
file.sync_all()?;
|
||||
@@ -1682,6 +1649,7 @@ mod tests {
|
||||
.read(true)
|
||||
.write(true)
|
||||
.create(true)
|
||||
.truncate(false)
|
||||
.open(&idx_path)
|
||||
.unwrap();
|
||||
let idx_size = idx_file.metadata().unwrap().len();
|
||||
@@ -2344,12 +2312,14 @@ mod tests {
|
||||
reloaded.deleted_count(),
|
||||
reloaded.deleted_size(),
|
||||
);
|
||||
assert_eq!(
|
||||
after, live,
|
||||
"close_first={close_first} rebuild={rebuild}"
|
||||
);
|
||||
assert_eq!(after, live, "close_first={close_first} rebuild={rebuild}");
|
||||
assert_eq!(reloaded.get(NeedleId(1)).unwrap().unwrap().size, Size(200));
|
||||
assert!(reloaded.get(NeedleId(2)).unwrap().map_or(true, |v| v.size.is_deleted()));
|
||||
assert!(
|
||||
reloaded
|
||||
.get(NeedleId(2))
|
||||
.unwrap()
|
||||
.is_none_or(|v| v.size.is_deleted())
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -31,7 +31,7 @@ struct CompactEntry {
|
||||
}
|
||||
|
||||
impl CompactEntry {
|
||||
fn to_needle_value(&self) -> NeedleValue {
|
||||
fn to_needle_value(self) -> NeedleValue {
|
||||
NeedleValue {
|
||||
offset: Offset::from_bytes(&self.offset),
|
||||
size: self.size,
|
||||
|
||||
@@ -226,10 +226,7 @@ impl SortedFileNeedleMap {
|
||||
.fail_sdx_mark
|
||||
.load(std::sync::atomic::Ordering::Relaxed)
|
||||
{
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
"injected .sdx mark failure",
|
||||
));
|
||||
return Err(io::Error::other("injected .sdx mark failure"));
|
||||
}
|
||||
let mut buf = [0u8; SIZE_SIZE];
|
||||
TOMBSTONE_FILE_SIZE.to_bytes(&mut buf);
|
||||
@@ -309,7 +306,7 @@ impl SortedFileNeedleMap {
|
||||
let rows = rows_per_read.min(entry_count - done) as usize;
|
||||
let bytes = &mut block[..rows * NEEDLE_MAP_ENTRY_SIZE];
|
||||
read_exact_at(&file, bytes, done * NEEDLE_MAP_ENTRY_SIZE as u64)?;
|
||||
for entry in bytes.chunks_exact(NEEDLE_MAP_ENTRY_SIZE) {
|
||||
for entry in bytes.as_chunks::<NEEDLE_MAP_ENTRY_SIZE>().0 {
|
||||
let (key, offset, size) = idx_entry_from_bytes(entry);
|
||||
if !size.is_valid() || pending.contains_key(&key) {
|
||||
continue; // deleted in place, or still awaiting that mark
|
||||
|
||||
@@ -369,6 +369,7 @@ impl Store {
|
||||
}
|
||||
|
||||
/// Create a new volume, placing it on the location with the most free space.
|
||||
#[expect(clippy::too_many_arguments)]
|
||||
pub fn add_volume(
|
||||
&mut self,
|
||||
vid: VolumeId,
|
||||
@@ -383,10 +384,10 @@ impl Store {
|
||||
return Err(VolumeError::AlreadyExists);
|
||||
}
|
||||
let loc_idx = self.find_free_location(&disk_type).ok_or_else(|| {
|
||||
VolumeError::Io(io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!("no free location for disk type {:?}", disk_type),
|
||||
))
|
||||
VolumeError::Io(io::Error::other(format!(
|
||||
"no free location for disk type {:?}",
|
||||
disk_type
|
||||
)))
|
||||
})?;
|
||||
|
||||
self.locations[loc_idx].create_volume(
|
||||
@@ -427,6 +428,26 @@ impl Store {
|
||||
false
|
||||
}
|
||||
|
||||
/// Reports whether any local volume or EC shard is currently quarantined
|
||||
/// due to sustained storage-media EIO. Mirrors Go's Store.HasIoQuarantine.
|
||||
pub fn has_io_quarantine(&self) -> bool {
|
||||
for loc in &self.locations {
|
||||
for (_, vol) in loc.iter_volumes() {
|
||||
let (_, _, quarantined) = vol.get_io_error_state();
|
||||
if quarantined {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
for (_, ec_vol) in loc.ec_volumes() {
|
||||
let (_, _, quarantined) = ec_vol.get_io_error_state();
|
||||
if quarantined {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
false
|
||||
}
|
||||
|
||||
/// Mount a volume from an existing .dat file.
|
||||
pub fn mount_volume(
|
||||
&mut self,
|
||||
@@ -439,7 +460,7 @@ impl Store {
|
||||
}
|
||||
// Find the location where the .dat file exists
|
||||
for loc in &mut self.locations {
|
||||
if &loc.disk_type != &disk_type {
|
||||
if loc.disk_type != disk_type {
|
||||
continue;
|
||||
}
|
||||
let base = crate::storage::volume::volume_file_name(&loc.directory, collection, vid);
|
||||
@@ -452,10 +473,10 @@ impl Store {
|
||||
// Fail the mount so the caller (VolumeCopy) treats it as an error.
|
||||
let note_path = format!("{}.note", base);
|
||||
if std::path::Path::new(¬e_path).exists() {
|
||||
return Err(VolumeError::Io(io::Error::new(
|
||||
io::ErrorKind::Other,
|
||||
format!("volume {} copy incomplete: .note still present", vid),
|
||||
)));
|
||||
return Err(VolumeError::Io(io::Error::other(format!(
|
||||
"volume {} copy incomplete: .note still present",
|
||||
vid
|
||||
))));
|
||||
}
|
||||
return loc.create_volume(
|
||||
vid,
|
||||
@@ -476,45 +497,185 @@ impl Store {
|
||||
|
||||
/// Mount a volume by id only (Go's MountVolume behavior).
|
||||
/// Scans all locations for a matching .dat file and loads with its collection prefix.
|
||||
pub fn mount_volume_by_id(&mut self, vid: VolumeId) -> Result<(), VolumeError> {
|
||||
/// When a collection hint is given, the expected <collection>_<vid>.vif/.idx
|
||||
/// path is probed directly before falling back to the directory scan.
|
||||
pub fn mount_volume_by_id(
|
||||
&mut self,
|
||||
vid: VolumeId,
|
||||
collection: Option<&str>,
|
||||
) -> Result<(), VolumeError> {
|
||||
if self.find_volume(vid).is_some() {
|
||||
return Err(VolumeError::AlreadyExists);
|
||||
}
|
||||
if let Some((loc_idx, _base_path, collection)) = self.find_volume_file_base(vid) {
|
||||
let loc = &mut self.locations[loc_idx];
|
||||
return loc.create_volume(
|
||||
vid,
|
||||
&collection,
|
||||
self.needle_map_kind,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
);
|
||||
// Remember the last non-NotFound error so a caller gets a useful
|
||||
// message when every candidate fails to open, instead of a generic
|
||||
// NotFound. A successful mount returns immediately.
|
||||
let mut last_err: Option<VolumeError> = None;
|
||||
if let Some(collection) = collection {
|
||||
// The hint is only an optimization: a collection carrying a path
|
||||
// separator could route file creation outside the storage
|
||||
// directory, so skip the shortcut and let the safe directory scan
|
||||
// resolve the volume instead. Rejecting the exact ".." name is
|
||||
// sufficient — a collection like "foo..bar" is a valid name and
|
||||
// stays inside the directory (the ".." is part of the filename, not
|
||||
// a parent reference, because volume_file_name joins with "_").
|
||||
let hint_safe = !collection.is_empty()
|
||||
&& !collection.contains('/')
|
||||
&& !collection.contains('\\')
|
||||
&& collection != "..";
|
||||
if hint_safe {
|
||||
for loc in &mut self.locations {
|
||||
let base = crate::storage::volume::volume_file_name(
|
||||
&loc.directory,
|
||||
collection,
|
||||
vid,
|
||||
);
|
||||
// Confirm a collection-named sidecar exists before using the
|
||||
// hint. A lone .vif/.idx (e.g. an EC sidecar whose .ecx is on
|
||||
// a sibling disk) must NOT mount here: create_volume would
|
||||
// write an empty .dat and register a phantom normal volume
|
||||
// that shadows the real EC volume. Match the guard in
|
||||
// load_existing_volumes: only mount when a real .dat is
|
||||
// present, or the .vif points at a remote-tiered file.
|
||||
let sidecar_present = [".vif", ".idx"].iter().any(|ext| {
|
||||
std::fs::metadata(format!("{}{}", base, ext))
|
||||
.map(|m| !m.is_dir())
|
||||
.unwrap_or(false)
|
||||
});
|
||||
if !sidecar_present {
|
||||
continue;
|
||||
}
|
||||
let dat_path = format!("{}.dat", base);
|
||||
let dat_exists = std::fs::metadata(&dat_path)
|
||||
.map(|m| !m.is_dir())
|
||||
.unwrap_or(false);
|
||||
let idx_base = crate::storage::volume::volume_file_name(
|
||||
&loc.idx_directory,
|
||||
collection,
|
||||
vid,
|
||||
);
|
||||
let has_remote = crate::storage::disk_location::vif_references_remote_file(
|
||||
&format!("{}.vif", base),
|
||||
) || crate::storage::disk_location::vif_references_remote_file(
|
||||
&format!("{}.vif", idx_base),
|
||||
);
|
||||
if dat_exists || has_remote {
|
||||
// A persisting .note means the copy that produced these
|
||||
// files never completed; mounting it would expose a
|
||||
// truncated volume. Skip this candidate and keep
|
||||
// searching (matches load_existing_volumes).
|
||||
let note_path = format!("{}.note", base);
|
||||
if std::path::Path::new(¬e_path).exists() {
|
||||
continue;
|
||||
}
|
||||
// An open failure on one candidate must not block a
|
||||
// valid volume on a later disk — remember the error and
|
||||
// keep scanning (matches open_volumes / Go mountVolume).
|
||||
match loc.create_volume(
|
||||
vid,
|
||||
collection,
|
||||
self.needle_map_kind,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
) {
|
||||
Ok(()) => return Ok(()),
|
||||
Err(e) => {
|
||||
last_err = Some(e);
|
||||
continue;
|
||||
}
|
||||
}
|
||||
}
|
||||
// Lone sidecar: leave it for the directory scan below.
|
||||
}
|
||||
}
|
||||
}
|
||||
Err(VolumeError::Io(io::Error::new(
|
||||
// Iterate every matching candidate, not just the first. A lone
|
||||
// sidecar on an earlier disk must not hide a real .dat on a later
|
||||
// disk (the split-disk EC layout the guard above protects against).
|
||||
for (loc_idx, base_path, collection) in self.find_volume_file_bases(vid) {
|
||||
// The scan matches any volume file (.dat/.vif/.idx). A lone .vif/.idx
|
||||
// sidecar (e.g. an EC sidecar whose .ecx is on a sibling disk) must
|
||||
// NOT mount here: create_volume would write an empty .dat and
|
||||
// register a phantom normal volume that shadows the real EC volume.
|
||||
// Match the guard in load_existing_volumes: only mount when a real
|
||||
// .dat is present, or the .vif points at a remote-tiered file.
|
||||
let dat_exists = std::fs::metadata(format!("{}.dat", base_path))
|
||||
.map(|m| !m.is_dir())
|
||||
.unwrap_or(false);
|
||||
let idx_base = crate::storage::volume::volume_file_name(
|
||||
&self.locations[loc_idx].idx_directory,
|
||||
&collection,
|
||||
vid,
|
||||
);
|
||||
let has_remote = crate::storage::disk_location::vif_references_remote_file(
|
||||
&format!("{}.vif", base_path),
|
||||
) || crate::storage::disk_location::vif_references_remote_file(
|
||||
&format!("{}.vif", idx_base),
|
||||
);
|
||||
if dat_exists || has_remote {
|
||||
// A persisting .note means the copy that produced these files
|
||||
// never completed; mounting it would expose a truncated volume.
|
||||
// Skip this candidate and keep searching (matches
|
||||
// load_existing_volumes).
|
||||
let note_path = format!("{}.note", base_path);
|
||||
if std::path::Path::new(¬e_path).exists() {
|
||||
continue;
|
||||
}
|
||||
// An open failure on one candidate must not block a valid
|
||||
// volume on a later disk — remember the error and keep
|
||||
// scanning (matches open_volumes / Go mountVolume).
|
||||
let loc = &mut self.locations[loc_idx];
|
||||
match loc.create_volume(
|
||||
vid,
|
||||
&collection,
|
||||
self.needle_map_kind,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
) {
|
||||
Ok(()) => return Ok(()),
|
||||
Err(e) => {
|
||||
last_err = Some(e);
|
||||
continue;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
Err(last_err.unwrap_or_else(|| VolumeError::Io(io::Error::new(
|
||||
io::ErrorKind::NotFound,
|
||||
format!("volume {} not found on disk", vid),
|
||||
)))
|
||||
))))
|
||||
}
|
||||
|
||||
fn find_volume_file_base(&self, vid: VolumeId) -> Option<(usize, String, String)> {
|
||||
self.find_volume_file_bases(vid).into_iter().next()
|
||||
}
|
||||
|
||||
/// Collect every location/collection whose directory holds a volume file
|
||||
/// (.dat/.vif/.idx) for `vid`, in scan order. Callers that need to skip
|
||||
/// lone sidecars (mount_volume_by_id) must see all candidates so a sidecar
|
||||
/// on an earlier disk does not hide a real .dat on a later one.
|
||||
fn find_volume_file_bases(&self, vid: VolumeId) -> Vec<(usize, String, String)> {
|
||||
let mut results = Vec::new();
|
||||
for (loc_idx, loc) in self.locations.iter().enumerate() {
|
||||
if let Ok(entries) = std::fs::read_dir(&loc.directory) {
|
||||
for entry in entries.flatten() {
|
||||
let name = entry.file_name();
|
||||
let name = name.to_string_lossy();
|
||||
if let Some((collection, file_vid)) = parse_volume_filename(&name) {
|
||||
if file_vid == vid {
|
||||
let base = strip_volume_suffix(&name)?;
|
||||
let base_path = format!("{}/{}", loc.directory, base);
|
||||
return Some((loc_idx, base_path, collection));
|
||||
}
|
||||
if let Some((collection, file_vid)) = parse_volume_filename(&name)
|
||||
&& file_vid == vid
|
||||
&& let Some(base) = strip_volume_suffix(&name)
|
||||
{
|
||||
let base_path = format!("{}/{}", loc.directory, base);
|
||||
results.push((loc_idx, base_path, collection));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
None
|
||||
results
|
||||
}
|
||||
|
||||
/// Configure a volume's replica placement on disk.
|
||||
@@ -681,10 +842,8 @@ impl Store {
|
||||
|
||||
let vol_count = loc.volumes_len() as i32;
|
||||
let loc_ec_shards = loc.ec_shard_count();
|
||||
let ec_equivalent = ((loc_ec_shards
|
||||
+ crate::storage::erasure_coding::ec_shard::DATA_SHARDS_COUNT
|
||||
- 1)
|
||||
/ crate::storage::erasure_coding::ec_shard::DATA_SHARDS_COUNT)
|
||||
let ec_equivalent = loc_ec_shards
|
||||
.div_ceil(crate::storage::erasure_coding::ec_shard::DATA_SHARDS_COUNT)
|
||||
as i32;
|
||||
let mut max_count = vol_count + ec_equivalent;
|
||||
|
||||
@@ -874,6 +1033,20 @@ impl Store {
|
||||
dirs
|
||||
}
|
||||
|
||||
/// Every per-disk `EcVolume` this store maps for `vid`, in location order.
|
||||
/// Immutable twin of [`Self::find_all_ec_volumes_mut`].
|
||||
///
|
||||
/// Reconciliation can mount one vid as N runtimes holding disjoint shard
|
||||
/// subsets, and the first-match `find_ec_volume` hides the siblings. Anything
|
||||
/// that has to reach the whole volume, rather than any one runtime of it,
|
||||
/// uses this.
|
||||
pub fn find_all_ec_volumes(&self, vid: VolumeId) -> Vec<&EcVolume> {
|
||||
self.locations
|
||||
.iter()
|
||||
.filter_map(|loc| loc.find_ec_volume(vid))
|
||||
.collect()
|
||||
}
|
||||
|
||||
pub fn find_all_ec_volumes_mut(&mut self, vid: VolumeId) -> Vec<&mut EcVolume> {
|
||||
self.locations
|
||||
.iter_mut()
|
||||
@@ -919,10 +1092,10 @@ impl Store {
|
||||
/// first disk and miss shards that live on a sibling.
|
||||
pub fn find_ec_shard_location(&self, vid: VolumeId, shard_id: u32) -> Option<usize> {
|
||||
for (i, loc) in self.locations.iter().enumerate() {
|
||||
if let Some(ecv) = loc.find_ec_volume(vid) {
|
||||
if ecv.has_shard(shard_id as u8) {
|
||||
return Some(i);
|
||||
}
|
||||
if let Some(ecv) = loc.find_ec_volume(vid)
|
||||
&& ecv.has_shard(shard_id as u8)
|
||||
{
|
||||
return Some(i);
|
||||
}
|
||||
}
|
||||
None
|
||||
@@ -931,16 +1104,12 @@ impl Store {
|
||||
/// Like [`Self::find_ec_shard_location`] but returns the EcVolume
|
||||
/// reference directly. Borrows the store immutably for the
|
||||
/// EcVolume's lifetime.
|
||||
pub fn find_ec_volume_with_shard(
|
||||
&self,
|
||||
vid: VolumeId,
|
||||
shard_id: u32,
|
||||
) -> Option<&EcVolume> {
|
||||
pub fn find_ec_volume_with_shard(&self, vid: VolumeId, shard_id: u32) -> Option<&EcVolume> {
|
||||
for loc in &self.locations {
|
||||
if let Some(ecv) = loc.find_ec_volume(vid) {
|
||||
if ecv.has_shard(shard_id as u8) {
|
||||
return Some(ecv);
|
||||
}
|
||||
if let Some(ecv) = loc.find_ec_volume(vid)
|
||||
&& ecv.has_shard(shard_id as u8)
|
||||
{
|
||||
return Some(ecv);
|
||||
}
|
||||
}
|
||||
None
|
||||
@@ -969,9 +1138,9 @@ impl Store {
|
||||
if found_vol.is_none() {
|
||||
found_vol = Some(ecv);
|
||||
}
|
||||
for shard_id in 0..max_shard_count {
|
||||
if dirs[shard_id].is_none() && ecv.has_shard(shard_id as u8) {
|
||||
dirs[shard_id] = Some(loc.directory.clone());
|
||||
for (shard_id, dir) in dirs.iter_mut().enumerate() {
|
||||
if dir.is_none() && ecv.has_shard(shard_id as u8) {
|
||||
*dir = Some(loc.directory.clone());
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -990,12 +1159,19 @@ impl Store {
|
||||
|
||||
for (disk_id, loc) in self.locations.iter_mut().enumerate() {
|
||||
let mut expired_vids = Vec::new();
|
||||
let mut io_quarantined_vids = Vec::new();
|
||||
for (vid, ec_vol) in loc.ec_volumes() {
|
||||
if ec_vol.is_time_to_destroy() {
|
||||
expired_vids.push(*vid);
|
||||
} else {
|
||||
ec_shards
|
||||
.extend(ec_vol.to_volume_ec_shard_information_messages(disk_id as u32));
|
||||
let (_, io_count, quarantined) = ec_vol.get_io_error_state();
|
||||
if quarantined || io_count >= crate::storage::erasure_coding::ec_volume::IO_ERROR_TOLERANCE
|
||||
{
|
||||
io_quarantined_vids.push(*vid);
|
||||
} else {
|
||||
ec_shards
|
||||
.extend(ec_vol.to_volume_ec_shard_information_messages(disk_id as u32));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1016,6 +1192,20 @@ impl Store {
|
||||
ec_shards.extend(messages);
|
||||
}
|
||||
}
|
||||
|
||||
for vid in io_quarantined_vids {
|
||||
if let Some(ec_vol) = loc.find_ec_volume(vid) {
|
||||
let (_, io_count, quarantined) = ec_vol.get_io_error_state();
|
||||
if !quarantined {
|
||||
ec_vol.mark_io_quarantined();
|
||||
tracing::warn!(
|
||||
volume_id = vid.0,
|
||||
io_count,
|
||||
"ec volume quarantined after consecutive IO errors"
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
(ec_shards, deleted)
|
||||
@@ -1320,9 +1510,10 @@ fn load_vif_volume_info(path: &str) -> Result<VifVolumeInfo, VolumeError> {
|
||||
read_only: bool,
|
||||
}
|
||||
if let Ok(legacy) = serde_json::from_str::<LegacyVolumeInfo>(&content) {
|
||||
let mut vif = VifVolumeInfo::default();
|
||||
vif.read_only = legacy.read_only;
|
||||
return Ok(vif);
|
||||
return Ok(VifVolumeInfo {
|
||||
read_only: legacy.read_only,
|
||||
..VifVolumeInfo::default()
|
||||
});
|
||||
}
|
||||
Err(VolumeError::Io(io::Error::new(
|
||||
io::ErrorKind::InvalidData,
|
||||
@@ -1332,7 +1523,7 @@ fn load_vif_volume_info(path: &str) -> Result<VifVolumeInfo, VolumeError> {
|
||||
|
||||
fn save_vif_volume_info(path: &str, info: &VifVolumeInfo) -> Result<(), VolumeError> {
|
||||
let content = serde_json::to_string_pretty(info)
|
||||
.map_err(|e| VolumeError::Io(io::Error::new(io::ErrorKind::Other, e.to_string())))?;
|
||||
.map_err(|e| VolumeError::Io(io::Error::other(e.to_string())))?;
|
||||
std::fs::write(path, content)?;
|
||||
Ok(())
|
||||
}
|
||||
@@ -1463,6 +1654,311 @@ mod tests {
|
||||
assert_eq!(store.total_volume_count(), 1);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_mount_volume_by_id_hint_skips_lone_sidecar() {
|
||||
// A lone .vif/.idx sidecar (e.g. an EC sidecar whose .ecx is on a
|
||||
// sibling disk) must NOT make the collection hint create a phantom
|
||||
// empty .dat. The hint is skipped and the directory scan finds nothing.
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
let mut store = make_test_store(&[dir]);
|
||||
|
||||
let base = volume_file_name(dir, "coll", VolumeId(5));
|
||||
std::fs::write(format!("{}.vif", base), "{}").unwrap();
|
||||
std::fs::write(format!("{}.idx", base), b"").unwrap();
|
||||
// No .dat, and the .vif does not reference a remote file.
|
||||
|
||||
let err = store
|
||||
.mount_volume_by_id(VolumeId(5), Some("coll"))
|
||||
.unwrap_err();
|
||||
assert!(matches!(err, VolumeError::Io(ref e)
|
||||
if e.kind() == std::io::ErrorKind::NotFound));
|
||||
// No phantom .dat was created.
|
||||
assert!(!std::path::Path::new(&format!("{}.dat", base)).exists());
|
||||
assert!(store.find_volume(VolumeId(5)).is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_mount_volume_by_id_hint_rejects_traversal_collection() {
|
||||
// A collection carrying a parent reference must not route file
|
||||
// creation outside the storage directory; the hint is dropped.
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
let mut store = make_test_store(&[dir]);
|
||||
|
||||
let escaped = format!(
|
||||
"{}/../evil_5.dat",
|
||||
dir
|
||||
);
|
||||
let err = store
|
||||
.mount_volume_by_id(VolumeId(5), Some("../evil"))
|
||||
.unwrap_err();
|
||||
assert!(matches!(err, VolumeError::Io(ref e)
|
||||
if e.kind() == std::io::ErrorKind::NotFound));
|
||||
assert!(!std::path::Path::new(&escaped).exists());
|
||||
assert!(store.find_volume(VolumeId(5)).is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_mount_volume_by_id_hint_loads_real_volume() {
|
||||
// With a real .dat on disk, the collection hint mounts the volume
|
||||
// directly without scanning the directory.
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
let mut store = make_test_store(&[dir]);
|
||||
store
|
||||
.add_volume(
|
||||
VolumeId(7),
|
||||
"coll",
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
DiskType::HardDrive,
|
||||
Version::current(),
|
||||
)
|
||||
.unwrap();
|
||||
// Write a needle so the volume has real data, then unmount it so the
|
||||
// .dat stays on disk but is no longer registered.
|
||||
let mut n = Needle {
|
||||
id: NeedleId(1),
|
||||
cookie: Cookie(0xaa),
|
||||
data: b"hint me".to_vec(),
|
||||
data_size: 7,
|
||||
..Needle::default()
|
||||
};
|
||||
store
|
||||
.write_volume_needle(VolumeId(7), &mut n, false)
|
||||
.unwrap();
|
||||
assert!(store.unmount_volume(VolumeId(7)));
|
||||
|
||||
store
|
||||
.mount_volume_by_id(VolumeId(7), Some("coll"))
|
||||
.unwrap();
|
||||
assert!(store.find_volume(VolumeId(7)).is_some());
|
||||
|
||||
let mut got = Needle {
|
||||
id: NeedleId(1),
|
||||
..Needle::default()
|
||||
};
|
||||
let count = store.read_volume_needle(VolumeId(7), &mut got).unwrap();
|
||||
assert_eq!(count, 7);
|
||||
assert_eq!(got.data, b"hint me");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_mount_volume_by_id_hint_accepts_double_dot_collection() {
|
||||
// A valid collection like "foo..bar" contains ".." but is not a parent
|
||||
// reference — volume_file_name joins with "_" so it stays in the dir.
|
||||
// The hint must NOT be rejected for it.
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
let mut store = make_test_store(&[dir]);
|
||||
store
|
||||
.add_volume(
|
||||
VolumeId(9),
|
||||
"foo..bar",
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
DiskType::HardDrive,
|
||||
Version::current(),
|
||||
)
|
||||
.unwrap();
|
||||
let mut n = Needle {
|
||||
id: NeedleId(1),
|
||||
cookie: Cookie(0xaa),
|
||||
data: b"dots".to_vec(),
|
||||
data_size: 4,
|
||||
..Needle::default()
|
||||
};
|
||||
store.write_volume_needle(VolumeId(9), &mut n, false).unwrap();
|
||||
assert!(store.unmount_volume(VolumeId(9)));
|
||||
|
||||
// The hint is accepted and mounts the volume.
|
||||
store
|
||||
.mount_volume_by_id(VolumeId(9), Some("foo..bar"))
|
||||
.unwrap();
|
||||
assert!(store.find_volume(VolumeId(9)).is_some());
|
||||
|
||||
let mut got = Needle {
|
||||
id: NeedleId(1),
|
||||
..Needle::default()
|
||||
};
|
||||
let count = store.read_volume_needle(VolumeId(9), &mut got).unwrap();
|
||||
assert_eq!(count, 4);
|
||||
assert_eq!(got.data, b"dots");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_mount_volume_by_id_skips_incomplete_note() {
|
||||
// A persisting .note means a VolumeCopy was interrupted; the volume
|
||||
// must not mount as live (would expose truncated data). The candidate
|
||||
// is skipped and mount returns NotFound.
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
let mut store = make_test_store(&[dir]);
|
||||
store
|
||||
.add_volume(
|
||||
VolumeId(11),
|
||||
"coll",
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
DiskType::HardDrive,
|
||||
Version::current(),
|
||||
)
|
||||
.unwrap();
|
||||
let mut n = Needle {
|
||||
id: NeedleId(1),
|
||||
cookie: Cookie(0xaa),
|
||||
data: b"partial".to_vec(),
|
||||
data_size: 7,
|
||||
..Needle::default()
|
||||
};
|
||||
store.write_volume_needle(VolumeId(11), &mut n, false).unwrap();
|
||||
assert!(store.unmount_volume(VolumeId(11)));
|
||||
|
||||
// Simulate an interrupted copy: drop a .note marker.
|
||||
let base = volume_file_name(dir, "coll", VolumeId(11));
|
||||
std::fs::write(format!("{}.note", base), "interrupted").unwrap();
|
||||
|
||||
// Hint path: skipped because of .note.
|
||||
let err = store
|
||||
.mount_volume_by_id(VolumeId(11), Some("coll"))
|
||||
.unwrap_err();
|
||||
assert!(matches!(err, VolumeError::Io(ref e)
|
||||
if e.kind() == std::io::ErrorKind::NotFound));
|
||||
assert!(store.find_volume(VolumeId(11)).is_none());
|
||||
|
||||
// Fallback path (no hint): also skipped because of .note.
|
||||
let err = store.mount_volume_by_id(VolumeId(11), None).unwrap_err();
|
||||
assert!(matches!(err, VolumeError::Io(ref e)
|
||||
if e.kind() == std::io::ErrorKind::NotFound));
|
||||
assert!(store.find_volume(VolumeId(11)).is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_mount_volume_by_id_fallback_skips_sidecar_finds_real_dat() {
|
||||
// A lone sidecar on disk 0 must not hide a real .dat on disk 1.
|
||||
// The fallback now iterates all candidates instead of stopping at
|
||||
// the first (sidecar-only) match.
|
||||
let tmp1 = TempDir::new().unwrap();
|
||||
let tmp2 = TempDir::new().unwrap();
|
||||
let dir0 = tmp1.path().to_str().unwrap();
|
||||
let dir1 = tmp2.path().to_str().unwrap();
|
||||
|
||||
// disk 0: lone .vif sidecar, no .dat, no remote.
|
||||
let base0 = volume_file_name(dir0, "coll", VolumeId(13));
|
||||
std::fs::write(format!("{}.vif", base0), "{}").unwrap();
|
||||
|
||||
// disk 1: real volume with data.
|
||||
let mut store = make_test_store(&[dir0, dir1]);
|
||||
store
|
||||
.add_volume(
|
||||
VolumeId(13),
|
||||
"coll",
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
DiskType::HardDrive,
|
||||
Version::current(),
|
||||
)
|
||||
.unwrap();
|
||||
let mut n = Needle {
|
||||
id: NeedleId(1),
|
||||
cookie: Cookie(0xaa),
|
||||
data: b"real".to_vec(),
|
||||
data_size: 4,
|
||||
..Needle::default()
|
||||
};
|
||||
store.write_volume_needle(VolumeId(13), &mut n, false).unwrap();
|
||||
assert!(store.unmount_volume(VolumeId(13)));
|
||||
|
||||
// No hint: the fallback scan finds the sidecar on disk 0 first (skip,
|
||||
// no .dat), then the real .dat on disk 1 (mount).
|
||||
store.mount_volume_by_id(VolumeId(13), None).unwrap();
|
||||
assert!(store.find_volume(VolumeId(13)).is_some());
|
||||
|
||||
let mut got = Needle {
|
||||
id: NeedleId(1),
|
||||
..Needle::default()
|
||||
};
|
||||
let count = store.read_volume_needle(VolumeId(13), &mut got).unwrap();
|
||||
assert_eq!(count, 4);
|
||||
assert_eq!(got.data, b"real");
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
#[test]
|
||||
fn test_mount_volume_by_id_continues_past_open_failure() {
|
||||
// A create_volume failure on an earlier candidate must not block a
|
||||
// valid volume on a later disk. The scan remembers the error and
|
||||
// keeps going (matches open_volumes / Go mountVolume).
|
||||
use std::os::unix::fs::PermissionsExt;
|
||||
use std::sync::atomic::Ordering;
|
||||
let tmp1 = TempDir::new().unwrap();
|
||||
let tmp2 = TempDir::new().unwrap();
|
||||
let dir0 = tmp1.path().to_str().unwrap();
|
||||
let dir1 = tmp2.path().to_str().unwrap();
|
||||
|
||||
let mut store = make_test_store(&[dir0, dir1]);
|
||||
|
||||
// Force add_volume onto disk 1 by marking disk 0 as low on space.
|
||||
store.locations[0].is_disk_space_low.store(true, Ordering::Relaxed);
|
||||
store
|
||||
.add_volume(
|
||||
VolumeId(15),
|
||||
"coll",
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
DiskType::HardDrive,
|
||||
Version::current(),
|
||||
)
|
||||
.unwrap();
|
||||
let mut n = Needle {
|
||||
id: NeedleId(1),
|
||||
cookie: Cookie(0xaa),
|
||||
data: b"later".to_vec(),
|
||||
data_size: 5,
|
||||
..Needle::default()
|
||||
};
|
||||
store.write_volume_needle(VolumeId(15), &mut n, false).unwrap();
|
||||
assert!(store.unmount_volume(VolumeId(15)));
|
||||
// Clear the low-space flag so mount_volume_by_id considers disk 0.
|
||||
store.locations[0].is_disk_space_low.store(false, Ordering::Relaxed);
|
||||
|
||||
// disk 0: a .dat that exists but is unreadable (chmod 000). The guard
|
||||
// sees dat_exists=true (metadata succeeds, not a dir), but
|
||||
// create_volume -> Volume::new -> load fails opening it.
|
||||
let base0 = volume_file_name(dir0, "coll", VolumeId(15));
|
||||
std::fs::write(format!("{}.dat", base0), b"").unwrap();
|
||||
std::fs::set_permissions(
|
||||
format!("{}.dat", base0),
|
||||
std::fs::Permissions::from_mode(0o000),
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
// The fallback scan hits disk 0 first (create_volume fails on the
|
||||
// unreadable .dat), then disk 1 (succeeds). Mount succeeds from disk 1.
|
||||
store.mount_volume_by_id(VolumeId(15), None).unwrap();
|
||||
assert!(store.find_volume(VolumeId(15)).is_some());
|
||||
|
||||
let mut got = Needle {
|
||||
id: NeedleId(1),
|
||||
..Needle::default()
|
||||
};
|
||||
let count = store.read_volume_needle(VolumeId(15), &mut got).unwrap();
|
||||
assert_eq!(count, 5);
|
||||
assert_eq!(got.data, b"later");
|
||||
|
||||
// Restore permissions so TempDir cleanup can remove the file.
|
||||
let _ = std::fs::set_permissions(
|
||||
format!("{}.dat", base0),
|
||||
std::fs::Permissions::from_mode(0o644),
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_store_read_write_delete() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
|
||||
@@ -80,6 +80,11 @@ struct EcxOwnerInfo {
|
||||
idx_dir: String,
|
||||
}
|
||||
|
||||
/// One unit of reconcile work: the disk holding orphan shards, the volume
|
||||
/// they belong to, the shard files, the `.ecx` owner, and whether the
|
||||
/// mirror already installed sidecars locally (`use_local_idx`).
|
||||
type OrphanShardLoad = (usize, EcKey, Vec<(String, u32)>, EcxOwnerInfo, bool);
|
||||
|
||||
impl Store {
|
||||
/// Run cross-disk orphan-shard reconciliation. Should be called
|
||||
/// after every DiskLocation has finished its per-disk EC scan.
|
||||
@@ -98,7 +103,7 @@ impl Store {
|
||||
// `use_local_idx` is the post-mirror fast path: when the
|
||||
// mirror already installed sidecars locally, mount against
|
||||
// loc.idx_directory instead of the owner disk.
|
||||
let mut to_load: Vec<(usize, EcKey, Vec<(String, u32)>, EcxOwnerInfo, bool)> = Vec::new();
|
||||
let mut to_load: Vec<OrphanShardLoad> = Vec::new();
|
||||
for (loc_idx, loc) in self.locations.iter().enumerate() {
|
||||
let orphans = collect_orphan_ec_shards(loc, loc_idx);
|
||||
for (key, shards) in orphans {
|
||||
@@ -293,10 +298,10 @@ impl Store {
|
||||
// may be sole copies of a distributed volume.
|
||||
let mut node_wide_bits = ev.shard_bits().0;
|
||||
for other in &self.locations {
|
||||
if let Some(other_ev) = other.find_ec_volume(*vid) {
|
||||
if other_ev.collection == ev.collection {
|
||||
node_wide_bits |= other_ev.shard_bits().0;
|
||||
}
|
||||
if let Some(other_ev) = other.find_ec_volume(*vid)
|
||||
&& other_ev.collection == ev.collection
|
||||
{
|
||||
node_wide_bits |= other_ev.shard_bits().0;
|
||||
}
|
||||
}
|
||||
let node_wide = node_wide_bits.count_ones() as usize;
|
||||
@@ -499,6 +504,53 @@ impl Store {
|
||||
}
|
||||
}
|
||||
|
||||
/// Walk a disk's data directory and return the `.ec??` shard files
|
||||
/// that are present on disk but not yet registered in the location's
|
||||
/// `ec_volumes` map. Keyed by (collection, vid) so callers can match
|
||||
/// each group against its `.ecx`-owning disk in one lookup. Zero-byte
|
||||
/// shard files are ignored — same shape as `load_all_ec_shards`.
|
||||
fn collect_orphan_ec_shards(
|
||||
loc: &crate::storage::disk_location::DiskLocation,
|
||||
_loc_idx: usize,
|
||||
) -> HashMap<EcKey, Vec<(String, u32)>> {
|
||||
let mut orphans: HashMap<EcKey, Vec<(String, u32)>> = HashMap::new();
|
||||
let Ok(read) = fs::read_dir(&loc.directory) else {
|
||||
return orphans;
|
||||
};
|
||||
for ent in read.flatten() {
|
||||
if ent.file_type().map(|ft| ft.is_dir()).unwrap_or(false) {
|
||||
continue;
|
||||
}
|
||||
let name = ent.file_name().to_string_lossy().into_owned();
|
||||
let Some(dot) = name.rfind('.') else {
|
||||
continue;
|
||||
};
|
||||
let (base, ext) = name.split_at(dot);
|
||||
let Some(shard_id) = is_ec_shard_extension(ext) else {
|
||||
continue;
|
||||
};
|
||||
// Ignore zero-byte shards. Use the DirEntry's metadata so we
|
||||
// don't pay a second stat syscall per file beyond what
|
||||
// read_dir already returned.
|
||||
match ent.metadata() {
|
||||
Ok(meta) if meta.len() > 0 => {}
|
||||
_ => continue,
|
||||
}
|
||||
let Some((collection, vid)) = parse_collection_volume_id_pub(base) else {
|
||||
continue;
|
||||
};
|
||||
// Skip shards that are already registered to an EcVolume.
|
||||
if let Some(ecv) = loc.find_ec_volume(vid)
|
||||
&& ecv.has_shard(shard_id as u8)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
let key = EcKey { collection, vid };
|
||||
orphans.entry(key).or_default().push((name, shard_id));
|
||||
}
|
||||
orphans
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
@@ -1184,6 +1236,76 @@ mod tests {
|
||||
assert!(!std::ptr::eq(ev0, ev1));
|
||||
}
|
||||
|
||||
/// `find_ec_volume` returns only disk 0's runtime, which is what hides
|
||||
/// sibling-disk shards from every scrub mode. The plural lookup must
|
||||
/// return one runtime per disk holding the vid, in location order.
|
||||
#[test]
|
||||
fn test_find_all_ec_volumes_returns_every_disk() {
|
||||
let (store, _tmp) = build_split_disk_store(7010);
|
||||
let vid = VolumeId(7010);
|
||||
|
||||
let all = store.find_all_ec_volumes(vid);
|
||||
assert_eq!(all.len(), 2, "expected one EcVolume per disk holding the vid");
|
||||
|
||||
// Disk 0 carries shards 0 and 12; disk 1 carries shard 1.
|
||||
assert!(all[0].has_shard(0));
|
||||
assert!(all[0].has_shard(12));
|
||||
assert!(all[1].has_shard(1));
|
||||
|
||||
// The singular lookup sees only the first — the bug being fixed.
|
||||
let first = store.find_ec_volume(vid).unwrap();
|
||||
assert!(std::ptr::eq(first, all[0]));
|
||||
|
||||
// A vid nobody mounts yields an empty vec, not a panic.
|
||||
assert!(store.find_all_ec_volumes(VolumeId(9999)).is_empty());
|
||||
}
|
||||
|
||||
/// End-to-end: with the vid mounted on two disks, a scrub driven through
|
||||
/// the Store must reach BOTH disks' shards. Before the aggregation fix
|
||||
/// `find_ec_volume` returned disk 0 and disk 1's shard 1 was never read.
|
||||
#[test]
|
||||
fn test_scrub_plans_reach_every_disk_through_the_store() {
|
||||
use crate::storage::erasure_coding::ec_volume::{
|
||||
merge_ec_runtimes, EcChecksumScrubPlan, EcLocalScrubPlan,
|
||||
};
|
||||
|
||||
let (store, _tmp) = build_split_disk_store(7030);
|
||||
let vid = VolumeId(7030);
|
||||
|
||||
let runtimes = store.find_all_ec_volumes(vid);
|
||||
assert_eq!(runtimes.len(), 2);
|
||||
|
||||
// Reachability is the invariant, so assert on the resolved slots rather
|
||||
// than on scrub message text: shards 0 and 12 live on disk 0, shard 1 on
|
||||
// disk 1. The old first-match lookup could never see shard 1.
|
||||
let merged = merge_ec_runtimes(&runtimes).expect("two runtimes merge");
|
||||
assert!(merged.slots[0].is_some(), "disk 0's shard 0 unreachable");
|
||||
assert!(merged.slots[12].is_some(), "disk 0's shard 12 unreachable");
|
||||
assert!(merged.slots[1].is_some(), "disk 1's shard 1 unreachable — the bug");
|
||||
assert!(merged.skipped.is_empty(), "same generation: {:?}", merged.skipped);
|
||||
|
||||
// Shard 1 is owned by the sibling runtime, not the anchor.
|
||||
let (owner, _) = merged.slots[1].unwrap();
|
||||
assert!(std::ptr::eq(owner, runtimes[1]));
|
||||
|
||||
// Both plans build over the union rather than over disk 0 alone.
|
||||
assert!(EcChecksumScrubPlan::for_volumes(&runtimes).is_some());
|
||||
assert!(EcLocalScrubPlan::for_volumes(&runtimes).is_some());
|
||||
// ...and `is_some()` is a real question: `for_volumes` has exactly one
|
||||
// `None` (the vanished-volume case), so without this the two lines above
|
||||
// would hold for any input at all.
|
||||
assert!(EcChecksumScrubPlan::for_volumes(&[]).is_none());
|
||||
assert!(EcLocalScrubPlan::for_volumes(&[]).is_none());
|
||||
|
||||
// Regression guard: a single-runtime view still sees only its own disk,
|
||||
// which is exactly what made aggregation necessary.
|
||||
let disk0 = merge_ec_runtimes(&[runtimes[0]]).unwrap();
|
||||
assert!(
|
||||
disk0.slots.get(1).copied().flatten().is_none(),
|
||||
"disk 0's runtime must not see the sibling's shard"
|
||||
);
|
||||
}
|
||||
|
||||
/// `Store::unmount_ec_shards` used to return after the first
|
||||
/// location with the vid, so a request to unmount a shard that
|
||||
/// lives on a sibling disk became a silent no-op. After the fix,
|
||||
@@ -1689,50 +1811,3 @@ mod tests {
|
||||
assert!(std::path::Path::new(&format!("{}.ecx", ec_base)).exists());
|
||||
}
|
||||
}
|
||||
|
||||
/// Walk a disk's data directory and return the `.ec??` shard files
|
||||
/// that are present on disk but not yet registered in the location's
|
||||
/// `ec_volumes` map. Keyed by (collection, vid) so callers can match
|
||||
/// each group against its `.ecx`-owning disk in one lookup. Zero-byte
|
||||
/// shard files are ignored — same shape as `load_all_ec_shards`.
|
||||
fn collect_orphan_ec_shards(
|
||||
loc: &crate::storage::disk_location::DiskLocation,
|
||||
_loc_idx: usize,
|
||||
) -> HashMap<EcKey, Vec<(String, u32)>> {
|
||||
let mut orphans: HashMap<EcKey, Vec<(String, u32)>> = HashMap::new();
|
||||
let Ok(read) = fs::read_dir(&loc.directory) else {
|
||||
return orphans;
|
||||
};
|
||||
for ent in read.flatten() {
|
||||
if ent.file_type().map(|ft| ft.is_dir()).unwrap_or(false) {
|
||||
continue;
|
||||
}
|
||||
let name = ent.file_name().to_string_lossy().into_owned();
|
||||
let Some(dot) = name.rfind('.') else {
|
||||
continue;
|
||||
};
|
||||
let (base, ext) = name.split_at(dot);
|
||||
let Some(shard_id) = is_ec_shard_extension(ext) else {
|
||||
continue;
|
||||
};
|
||||
// Ignore zero-byte shards. Use the DirEntry's metadata so we
|
||||
// don't pay a second stat syscall per file beyond what
|
||||
// read_dir already returned.
|
||||
match ent.metadata() {
|
||||
Ok(meta) if meta.len() > 0 => {}
|
||||
_ => continue,
|
||||
}
|
||||
let Some((collection, vid)) = parse_collection_volume_id_pub(base) else {
|
||||
continue;
|
||||
};
|
||||
// Skip shards that are already registered to an EcVolume.
|
||||
if let Some(ecv) = loc.find_ec_volume(vid) {
|
||||
if ecv.has_shard(shard_id as u8) {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
let key = EcKey { collection, vid };
|
||||
orphans.entry(key).or_default().push((name, shard_id));
|
||||
}
|
||||
orphans
|
||||
}
|
||||
|
||||
@@ -155,7 +155,7 @@ impl Size {
|
||||
return 0;
|
||||
}
|
||||
if self.0 < 0 {
|
||||
return (self.0 * -1) as u32;
|
||||
return -self.0 as u32;
|
||||
}
|
||||
self.0 as u32
|
||||
}
|
||||
@@ -284,8 +284,9 @@ impl fmt::Display for Offset {
|
||||
// DiskType
|
||||
// ============================================================================
|
||||
|
||||
#[derive(Debug, Clone, PartialEq, Eq, Hash)]
|
||||
#[derive(Debug, Clone, PartialEq, Eq, Hash, Default)]
|
||||
pub enum DiskType {
|
||||
#[default]
|
||||
HardDrive,
|
||||
Ssd,
|
||||
Custom(String),
|
||||
@@ -319,12 +320,6 @@ impl fmt::Display for DiskType {
|
||||
}
|
||||
}
|
||||
|
||||
impl Default for DiskType {
|
||||
fn default() -> Self {
|
||||
DiskType::HardDrive
|
||||
}
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// VolumeId
|
||||
// ============================================================================
|
||||
@@ -397,7 +392,7 @@ impl From<u8> for Version {
|
||||
///
|
||||
/// Fields are split into request-side options (set by the caller) and response-side
|
||||
/// flags (set during the read to communicate status back).
|
||||
#[derive(Debug, Clone)]
|
||||
#[derive(Debug, Clone, Default)]
|
||||
pub struct ReadOption {
|
||||
// -- request --
|
||||
/// If true, allow reading needles that have been soft-deleted.
|
||||
@@ -423,21 +418,6 @@ pub struct ReadOption {
|
||||
pub read_buffer_size: i32,
|
||||
}
|
||||
|
||||
impl Default for ReadOption {
|
||||
fn default() -> Self {
|
||||
ReadOption {
|
||||
read_deleted: false,
|
||||
attempt_meta_only: false,
|
||||
must_meta_only: false,
|
||||
is_meta_only: false,
|
||||
volume_revision: 0,
|
||||
is_out_of_range: false,
|
||||
has_slow_read: false,
|
||||
read_buffer_size: 0,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// NeedleMapEntry helpers (for .idx file)
|
||||
// ============================================================================
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -73,8 +73,10 @@ mod tests {
|
||||
let empty = master_pb::VolumeInformationMessage::default();
|
||||
assert_eq!(report_hash(&empty), 10988706248825469653);
|
||||
|
||||
let mut one = master_pb::VolumeInformationMessage::default();
|
||||
one.id = 1;
|
||||
let one = master_pb::VolumeInformationMessage {
|
||||
id: 1,
|
||||
..Default::default()
|
||||
};
|
||||
assert_eq!(report_hash(&one), 2035849960016744285);
|
||||
|
||||
let full = master_pb::VolumeInformationMessage {
|
||||
|
||||
@@ -66,7 +66,7 @@ fn parse_go_version_number() -> Option<String> {
|
||||
}
|
||||
}
|
||||
match (major, minor) {
|
||||
(Some(maj), Some(min)) => Some(format!("{}.{}", maj, format!("{:02}", min))),
|
||||
(Some(maj), Some(min)) => Some(format!("{}.{:02}", maj, min)),
|
||||
_ => None,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -10,7 +10,19 @@ members = ["crates/core", "crates/lance", "crates/sort"]
|
||||
|
||||
[workspace.package]
|
||||
version = "0.1.0"
|
||||
edition = "2021"
|
||||
edition = "2024"
|
||||
# The edition needs 1.85; the dependency tree needs more. Verified with
|
||||
# `cargo +1.94.1 check --all-targets` (1.94.0 fails on the AWS SDK that
|
||||
# lance's `aws` feature pulls in).
|
||||
rust-version = "1.94.1"
|
||||
|
||||
[workspace.lints.clippy]
|
||||
# Every RPC path returns tonic::Status (176 bytes). Boxing it would change
|
||||
# every handler signature for no gain, so the large-Err lint is off.
|
||||
result_large_err = "allow"
|
||||
# Protobuf message literals keep `..Default::default()` on purpose: it is
|
||||
# what lets a proto gain a field without touching every constructor.
|
||||
needless_update = "allow"
|
||||
|
||||
[workspace.dependencies]
|
||||
anyhow = "1"
|
||||
|
||||
@@ -14,6 +14,12 @@ one.
|
||||
|
||||
## Building
|
||||
|
||||
Requires Rust 1.94.1+ (2024 edition), matching `rust-version` in `Cargo.toml`.
|
||||
The patch release matters: 1.94.0 does not build. The edition itself only needs
|
||||
1.85; the higher floor comes from the dependency tree — lance's `aws` feature
|
||||
pulls in the AWS SDK — so it moves with those crates. CI builds on the latest
|
||||
stable.
|
||||
|
||||
`core` compiles `plugin.proto` with the protoc that protoc-bin-vendored ships,
|
||||
the way seaweed-volume does, so it needs no system install.
|
||||
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
name = "seaweed-worker-core"
|
||||
version.workspace = true
|
||||
edition.workspace = true
|
||||
rust-version.workspace = true
|
||||
description = "SeaweedFS plugin.proto worker contract"
|
||||
|
||||
[lib]
|
||||
@@ -25,3 +26,6 @@ tonic-build.workspace = true
|
||||
# install, and so the version is pinned rather than whatever the platform's
|
||||
# package manager happens to carry. The same crate seaweed-volume uses.
|
||||
protoc-bin-vendored = "3"
|
||||
|
||||
[lints]
|
||||
workspace = true
|
||||
|
||||
@@ -4,7 +4,12 @@ fn main() -> Result<(), Box<dyn std::error::Error>> {
|
||||
// version. An explicit PROTOC still wins, for packagers supplying their own
|
||||
// and for the lance crates, whose own build scripts read the same variable.
|
||||
if std::env::var_os("PROTOC").is_none() {
|
||||
std::env::set_var("PROTOC", protoc_bin_vendored::protoc_bin_path()?);
|
||||
// SAFETY: a build script's main runs single-threaded before anything
|
||||
// else in this process, so no other thread can be reading the
|
||||
// environment concurrently.
|
||||
unsafe {
|
||||
std::env::set_var("PROTOC", protoc_bin_vendored::protoc_bin_path()?);
|
||||
}
|
||||
}
|
||||
|
||||
// Compiled straight out of the Go tree, the way seaweed-volume already reads
|
||||
|
||||
@@ -14,10 +14,10 @@ pub fn server_to_grpc_address(server: &str) -> Option<String> {
|
||||
let (host, port_part) = server.rsplit_once(':')?;
|
||||
|
||||
// "port.grpcPort" states the gRPC port outright.
|
||||
if let Some((_, grpc_port)) = port_part.split_once('.') {
|
||||
if let Ok(port) = grpc_port.parse::<u16>() {
|
||||
return Some(join_host_port(host, port));
|
||||
}
|
||||
if let Some((_, grpc_port)) = port_part.split_once('.')
|
||||
&& let Ok(port) = grpc_port.parse::<u16>()
|
||||
{
|
||||
return Some(join_host_port(host, port));
|
||||
}
|
||||
|
||||
let port: u16 = port_part.parse().ok()?;
|
||||
|
||||
@@ -16,6 +16,9 @@ pub mod stream;
|
||||
|
||||
/// Generated plugin.proto types.
|
||||
pub mod pb {
|
||||
// prost gives every oneof its own enum; the variant sizes are the
|
||||
// messages' own, not a choice made here.
|
||||
#![allow(clippy::large_enum_variant)]
|
||||
tonic::include_proto!("plugin");
|
||||
}
|
||||
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
name = "weed-lance-worker"
|
||||
version.workspace = true
|
||||
edition.workspace = true
|
||||
rust-version.workspace = true
|
||||
description = "SeaweedFS maintenance worker for Lance tables"
|
||||
|
||||
[lib]
|
||||
@@ -49,3 +50,6 @@ arrow-array = "58"
|
||||
arrow-schema = "58"
|
||||
arrow-cast = "58"
|
||||
lance-linalg = "10"
|
||||
|
||||
[lints]
|
||||
workspace = true
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
name = "seaweed-worker-sort"
|
||||
version.workspace = true
|
||||
edition.workspace = true
|
||||
rust-version.workspace = true
|
||||
description = "The sort specification shared by SeaweedFS sorting jobs"
|
||||
|
||||
[lib]
|
||||
@@ -10,3 +11,6 @@ name = "seaweed_worker_sort"
|
||||
[dependencies]
|
||||
seaweed-worker-core = { path = "../core" }
|
||||
anyhow.workspace = true
|
||||
|
||||
[lints]
|
||||
workspace = true
|
||||
|
||||
@@ -152,10 +152,10 @@ fn parse_field(entry: &str) -> Result<SortField> {
|
||||
/// back: sorting by the worker's default order instead of the one the table
|
||||
/// asked for would silently rewrite the table the wrong way.
|
||||
pub fn resolve(declared: Option<&str>, configured: &str) -> Result<Option<SortSpec>> {
|
||||
if let Some(declared) = declared {
|
||||
if let Some(spec) = SortSpec::parse(declared).context("read the table's declared order")? {
|
||||
return Ok(Some(spec));
|
||||
}
|
||||
if let Some(declared) = declared
|
||||
&& let Some(spec) = SortSpec::parse(declared).context("read the table's declared order")?
|
||||
{
|
||||
return Ok(Some(spec));
|
||||
}
|
||||
SortSpec::parse(configured).context("read the configured sort order")
|
||||
}
|
||||
|
||||
@@ -726,6 +726,20 @@ func (r *chaosRun) seedAndSpread() {
|
||||
require.GreaterOrEqual(r.t, len(r.volumes), 2, "seeding should produce at least two volumes")
|
||||
time.Sleep(3 * time.Second)
|
||||
|
||||
// Cap the grows per server so heartbeat lag cannot run away: the spread
|
||||
// check reads the master topology, which lags the volume.grow writes, so
|
||||
// an uncapped loop re-fires every tick and fills every disk to capacity
|
||||
// before the master registers the prior grow. A full disk has no free EC
|
||||
// shard slots and the source disk drops below the encode's FreeVolumeCount
|
||||
// >= 2 health check, failing ec.encode with "no healthy replicas" or "no
|
||||
// free ec shard slots".
|
||||
//
|
||||
// Use -count 1 so each grow lands exactly one volume on the volume
|
||||
// server's least-loaded disk (deterministic spreading), and cap the
|
||||
// total grows per server well below the per-disk max so the source disk
|
||||
// always retains FreeVolumeCount >= 2 for ec.encode's shard generation.
|
||||
growsPerServer := make(map[string]int)
|
||||
const maxGrowsPerServer = 4
|
||||
require.Eventually(r.t, func() bool {
|
||||
spread := nodeVolumeDiskCounts(r.t, r.env)
|
||||
if len(spread) == chaosServerCount && allAtLeast(spread, 2) {
|
||||
@@ -733,9 +747,26 @@ func (r *chaosRun) seedAndSpread() {
|
||||
}
|
||||
for i := 0; i < chaosServerCount; i++ {
|
||||
server := "127.0.0.1:" + chaosVolumePort(i)
|
||||
if spread[server] < 2 {
|
||||
if spread[server] < 2 && growsPerServer[server] < maxGrowsPerServer {
|
||||
// Pin the data center and rack as well as the data node. The
|
||||
// master's grow picks the rack by weighted random when -rack is
|
||||
// unset, and only one of the three racks holds the requested
|
||||
// data node, so an unpinned grow lands on the wrong rack two
|
||||
// times out of three and the VolumeGrow RPC swallows the
|
||||
// "No matching data node" error (non-cache grows ignore the
|
||||
// internal failure). Those silent no-ops exhaust the per-server
|
||||
// cap before the volumes ever spread, so seedAndSpread times
|
||||
// out. Pinning the rack makes every grow reach the target node.
|
||||
out, gerr := captureCommandOutput(r.t, shell.Commands[findCommandIndex("volume.grow")],
|
||||
[]string{"-collection", chaosCollection, "-dataNode", server, "-count", "4"}, r.env)
|
||||
[]string{"-collection", chaosCollection, "-dataCenter", "dc1",
|
||||
"-rack", fmt.Sprintf("rack%d", i), "-dataNode", server, "-count", "1"}, r.env)
|
||||
// Only count successful grows toward the cap: a transient
|
||||
// collectTopologyInfo or VolumeGrow RPC error would otherwise
|
||||
// exhaust the retry budget without creating any volume, and
|
||||
// the loop would then only poll until the Eventually timeout.
|
||||
if gerr == nil {
|
||||
growsPerServer[server]++
|
||||
}
|
||||
r.t.Logf("volume.grow on %s: err=%v output:\n%s", server, gerr, out)
|
||||
}
|
||||
}
|
||||
@@ -1159,7 +1190,13 @@ func (c *chaosCluster) startVolumeServer(ctx context.Context, i int, logName str
|
||||
return nil, err
|
||||
}
|
||||
diskDirs = append(diskDirs, dir)
|
||||
maxVolumes = append(maxVolumes, "4")
|
||||
// ec.encode generates 14 shards on the source disk (2 volume-slot
|
||||
// equivalents) and refuses a source whose disk has FreeVolumeCount < 2,
|
||||
// and the cluster-wide capacity check needs at least one free EC shard
|
||||
// slot. seedAndSpread's spread loop grows 4 volumes per tick and can
|
||||
// over-grow before the master's topology catches up, so leave enough
|
||||
// headroom that a near-full disk never starves the encode.
|
||||
maxVolumes = append(maxVolumes, "8")
|
||||
if d == chaosDisksPerNode-1 {
|
||||
diskTypes = append(diskTypes, "ssd")
|
||||
} else {
|
||||
@@ -1176,6 +1213,7 @@ func (c *chaosCluster) startVolumeServer(ctx context.Context, i int, logName str
|
||||
"-dir.idx", idxDir,
|
||||
"-disk", strings.Join(diskTypes, ","),
|
||||
"-max", strings.Join(maxVolumes, ","),
|
||||
"-minFreeSpace", "0",
|
||||
"-master", chaosMasterAddr,
|
||||
"-ip", "127.0.0.1",
|
||||
"-dataCenter", "dc1",
|
||||
|
||||
@@ -309,6 +309,57 @@ func (c *failoverCluster) FileVolumeIds(path string) ([]uint32, error) {
|
||||
return vids, nil
|
||||
}
|
||||
|
||||
// FileChunkList returns the filer's chunk list for a file: each chunk's file
|
||||
// id, logical offset, size, and the volume id it lives on. Used to pinpoint
|
||||
// which chunk covers a corrupted byte range and which volume server holds it.
|
||||
func (c *failoverCluster) FileChunkList(path string) ([]struct {
|
||||
FileId string
|
||||
Offset int64
|
||||
Size uint64
|
||||
VolumeId uint32
|
||||
}, error) {
|
||||
body, err := c.FilerGet(path + "?metadata=true&resolveManifest=true")
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var entry struct {
|
||||
Chunks []struct {
|
||||
FileId string `json:"file_id"`
|
||||
Offset int64 `json:"offset"`
|
||||
Size uint64 `json:"size"`
|
||||
Fid struct {
|
||||
VolumeId uint32 `json:"volume_id"`
|
||||
} `json:"fid"`
|
||||
} `json:"chunks"`
|
||||
}
|
||||
if err = json.Unmarshal(body, &entry); err != nil {
|
||||
return nil, fmt.Errorf("decode entry %s: %w", path, err)
|
||||
}
|
||||
out := make([]struct {
|
||||
FileId string
|
||||
Offset int64
|
||||
Size uint64
|
||||
VolumeId uint32
|
||||
}, len(entry.Chunks))
|
||||
for i, ch := range entry.Chunks {
|
||||
vid := ch.Fid.VolumeId
|
||||
if vid == 0 && ch.FileId != "" {
|
||||
parsed, parseErr := strconv.ParseUint(strings.SplitN(ch.FileId, ",", 2)[0], 10, 32)
|
||||
if parseErr != nil {
|
||||
return nil, fmt.Errorf("parse file id %s: %w", ch.FileId, parseErr)
|
||||
}
|
||||
vid = uint32(parsed)
|
||||
}
|
||||
out[i] = struct {
|
||||
FileId string
|
||||
Offset int64
|
||||
Size uint64
|
||||
VolumeId uint32
|
||||
}{ch.FileId, ch.Offset, ch.Size, vid}
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// VolumeHolders returns the volume server addresses the master currently lists
|
||||
// for a volume id.
|
||||
func (c *failoverCluster) VolumeHolders(vid uint32) ([]string, error) {
|
||||
|
||||
@@ -9,6 +9,7 @@ import (
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
@@ -396,13 +397,91 @@ func runChaosAppend(t *testing.T, c *failoverCluster, name string, chaos func())
|
||||
// own mount and the filer make of the same file, which says whether the
|
||||
// data was lost on the way in or is only invisible from this side.
|
||||
d := firstDiff(want, got)
|
||||
fromWriter, _ := os.ReadFile(writePath)
|
||||
fromWriter, writerReadErr := os.ReadFile(writePath)
|
||||
viaFiler, filerErr := c.FilerGet("/" + name)
|
||||
require.Failf(t, "final content mismatch",
|
||||
"%s: first difference at offset %d (want %d bytes, got %d)\nwant %q\ngot %q\nmount0 matches=%v filer matches=%v (err %v)\n%s",
|
||||
"%s: first difference at offset %d (want %d bytes, got %d)\nwant %q\ngot %q\nmount0 matches=%v (writer read err %v) filer matches=%v (err %v)\n%s\n%s\n%s",
|
||||
name, d, len(want), len(got), window(want, d), window(got, d),
|
||||
string(fromWriter) == want, string(viaFiler) == want, filerErr,
|
||||
c.tailLog("mount0"))
|
||||
string(fromWriter) == want, writerReadErr, string(viaFiler) == want, filerErr,
|
||||
c.tailLog("mount0"),
|
||||
dumpChunkList(c, "/"+name, d),
|
||||
dumpHexAround(fromWriter, d, "writer mount"),
|
||||
)
|
||||
}
|
||||
|
||||
// dumpChunkList returns a human-readable summary of the filer's chunk list
|
||||
// for path, annotated with the volume id and the master's current holders
|
||||
// for each. When diffAt >= 0, the chunk covering that logical offset is
|
||||
// flagged so a corruption can be tied to a specific chunk and volume server.
|
||||
func dumpChunkList(c *failoverCluster, path string, diffAt int) string {
|
||||
chunks, err := c.FileChunkList(path)
|
||||
if err != nil {
|
||||
return fmt.Sprintf("chunk list for %s: %v", path, err)
|
||||
}
|
||||
var b strings.Builder
|
||||
fmt.Fprintf(&b, "chunk list for %s (%d chunks):", path, len(chunks))
|
||||
for i, ch := range chunks {
|
||||
holders, holdersErr := c.VolumeHolders(ch.VolumeId)
|
||||
holdersStr := "unknown"
|
||||
if holdersErr != nil {
|
||||
holdersStr = fmt.Sprintf("lookup failed: %v", holdersErr)
|
||||
} else if len(holders) > 0 {
|
||||
holdersStr = strings.Join(holders, ",")
|
||||
}
|
||||
marker := ""
|
||||
if diffAt >= 0 && int64(diffAt) >= ch.Offset && int64(diffAt) < ch.Offset+int64(ch.Size) {
|
||||
marker = " <-- covers diff"
|
||||
}
|
||||
fmt.Fprintf(&b, "\n [%d] fid=%s offset=%d size=%d vid=%d holders=[%s]%s",
|
||||
i, ch.FileId, ch.Offset, ch.Size, ch.VolumeId, holdersStr, marker)
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
// dumpHexAround returns a hex+ASCII dump of a 32-byte window around off in
|
||||
// label's copy of the file, so the exact zero-filled region is visible in
|
||||
// the failure message instead of only a quoted string window.
|
||||
func dumpHexAround(data []byte, off int, label string) string {
|
||||
if off < 0 {
|
||||
return fmt.Sprintf("%s hex dump: no divergence offset", label)
|
||||
}
|
||||
start := max(0, off-16)
|
||||
end := min(len(data), off+16)
|
||||
if start >= end {
|
||||
return fmt.Sprintf("%s hex dump: offset %d out of range (len %d)", label, off, len(data))
|
||||
}
|
||||
var b strings.Builder
|
||||
fmt.Fprintf(&b, "%s hex dump around offset %d:", label, off)
|
||||
for i := start; i < end; i += 16 {
|
||||
lineEnd := min(i+16, end)
|
||||
hexPart := hexDump(data[i:lineEnd])
|
||||
asciiPart := asciiDump(data[i:lineEnd])
|
||||
fmt.Fprintf(&b, "\n %06x %-48s %s", i, hexPart, asciiPart)
|
||||
}
|
||||
return b.String()
|
||||
}
|
||||
|
||||
func hexDump(b []byte) string {
|
||||
var sb strings.Builder
|
||||
for i, x := range b {
|
||||
if i > 0 {
|
||||
sb.WriteByte(' ')
|
||||
}
|
||||
fmt.Fprintf(&sb, "%02x", x)
|
||||
}
|
||||
return sb.String()
|
||||
}
|
||||
|
||||
func asciiDump(b []byte) string {
|
||||
var sb strings.Builder
|
||||
for _, x := range b {
|
||||
if x >= 0x20 && x < 0x7f {
|
||||
sb.WriteByte(x)
|
||||
} else {
|
||||
sb.WriteByte('.')
|
||||
}
|
||||
}
|
||||
return sb.String()
|
||||
}
|
||||
|
||||
// firstDiff returns the offset of the first differing byte, or -1 when equal.
|
||||
|
||||
@@ -0,0 +1,66 @@
|
||||
package example
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/aws/aws-sdk-go/aws"
|
||||
"github.com/aws/aws-sdk-go/service/s3"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// TestListMultipartUploadsInitiated verifies that ListMultipartUploads
|
||||
// returns the Initiated timestamp for each in-progress upload, and that
|
||||
// the timestamp is preserved across repeated listings rather than
|
||||
// reflecting the time of the listing.
|
||||
func TestListMultipartUploadsInitiated(t *testing.T) {
|
||||
if testing.Short() {
|
||||
t.Skip("Skipping integration test in short mode")
|
||||
}
|
||||
|
||||
cluster, err := startMiniCluster(t)
|
||||
require.NoError(t, err)
|
||||
defer cluster.Stop()
|
||||
|
||||
bucket := createTestBucket(t, cluster, "test-list-mpu-initiated-")
|
||||
|
||||
beforeInit := time.Now().UTC()
|
||||
createOut, err := cluster.s3Client.CreateMultipartUpload(&s3.CreateMultipartUploadInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String("unfinished.bin"),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
afterInit := time.Now().UTC()
|
||||
|
||||
listOut, err := cluster.s3Client.ListMultipartUploads(&s3.ListMultipartUploadsInput{
|
||||
Bucket: aws.String(bucket),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
require.Len(t, listOut.Uploads, 1)
|
||||
|
||||
upload := listOut.Uploads[0]
|
||||
assert.Equal(t, "unfinished.bin", aws.StringValue(upload.Key))
|
||||
assert.Equal(t, aws.StringValue(createOut.UploadId), aws.StringValue(upload.UploadId))
|
||||
require.NotNil(t, upload.Initiated, "Initiated timestamp must be populated")
|
||||
|
||||
initiated := upload.Initiated.UTC()
|
||||
assert.False(t, initiated.Before(beforeInit.Add(-time.Second)), "Initiated %v is before upload creation %v", initiated, beforeInit)
|
||||
assert.False(t, initiated.After(afterInit.Add(time.Second)), "Initiated %v is after upload creation %v", initiated, afterInit)
|
||||
|
||||
time.Sleep(2 * time.Second)
|
||||
listOut2, err := cluster.s3Client.ListMultipartUploads(&s3.ListMultipartUploadsInput{
|
||||
Bucket: aws.String(bucket),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
require.Len(t, listOut2.Uploads, 1)
|
||||
require.NotNil(t, listOut2.Uploads[0].Initiated, "Initiated timestamp must be populated on repeated listing")
|
||||
assert.True(t, listOut2.Uploads[0].Initiated.Equal(*upload.Initiated), "Initiated must be preserved across listings, got %v then %v", *upload.Initiated, *listOut2.Uploads[0].Initiated)
|
||||
|
||||
_, err = cluster.s3Client.AbortMultipartUpload(&s3.AbortMultipartUploadInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String("unfinished.bin"),
|
||||
UploadId: createOut.UploadId,
|
||||
})
|
||||
require.NoError(t, err)
|
||||
}
|
||||
@@ -0,0 +1,285 @@
|
||||
package s3api
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"fmt"
|
||||
"io"
|
||||
"mime/multipart"
|
||||
"net/http"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/aws/aws-sdk-go-v2/aws"
|
||||
"github.com/aws/aws-sdk-go-v2/service/s3"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
type postPolicyUploadResult struct {
|
||||
versionID string
|
||||
etag string
|
||||
}
|
||||
|
||||
func postPolicyUpload(ctx context.Context, client *s3.Client, bucket, key string, body []byte) (postPolicyUploadResult, error) {
|
||||
return postPolicyUploadWithFields(ctx, client, bucket, key, body, nil)
|
||||
}
|
||||
|
||||
func postPolicyUploadWithFields(ctx context.Context, client *s3.Client, bucket, key string, body []byte, extraFields map[string]string) (postPolicyUploadResult, error) {
|
||||
presigner := s3.NewPresignClient(client, s3.WithPresignClientFromClientOptions(func(options *s3.Options) {
|
||||
options.BaseEndpoint = aws.String(defaultConfig.Endpoint)
|
||||
options.EndpointResolver = nil
|
||||
options.UsePathStyle = true
|
||||
}))
|
||||
presigned, err := presigner.PresignPostObject(ctx, &s3.PutObjectInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String(key),
|
||||
}, func(options *s3.PresignPostOptions) {
|
||||
options.Expires = time.Hour
|
||||
for name, value := range extraFields {
|
||||
options.Conditions = append(options.Conditions, map[string]string{name: value})
|
||||
}
|
||||
})
|
||||
if err != nil {
|
||||
return postPolicyUploadResult{}, err
|
||||
}
|
||||
|
||||
var requestBody bytes.Buffer
|
||||
writer := multipart.NewWriter(&requestBody)
|
||||
for name, value := range presigned.Values {
|
||||
if err := writer.WriteField(name, value); err != nil {
|
||||
return postPolicyUploadResult{}, err
|
||||
}
|
||||
}
|
||||
for name, value := range extraFields {
|
||||
if err := writer.WriteField(name, value); err != nil {
|
||||
return postPolicyUploadResult{}, err
|
||||
}
|
||||
}
|
||||
file, err := writer.CreateFormFile("file", "payload.bin")
|
||||
if err != nil {
|
||||
return postPolicyUploadResult{}, err
|
||||
}
|
||||
if _, err := file.Write(body); err != nil {
|
||||
return postPolicyUploadResult{}, err
|
||||
}
|
||||
if err := writer.Close(); err != nil {
|
||||
return postPolicyUploadResult{}, err
|
||||
}
|
||||
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodPost, presigned.URL, &requestBody)
|
||||
if err != nil {
|
||||
return postPolicyUploadResult{}, err
|
||||
}
|
||||
req.Header.Set("Content-Type", writer.FormDataContentType())
|
||||
resp, err := (&http.Client{Timeout: 30 * time.Second}).Do(req)
|
||||
if err != nil {
|
||||
return postPolicyUploadResult{}, err
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
responseBody, readErr := io.ReadAll(resp.Body)
|
||||
if readErr != nil {
|
||||
return postPolicyUploadResult{}, readErr
|
||||
}
|
||||
if resp.StatusCode != http.StatusNoContent {
|
||||
return postPolicyUploadResult{}, fmt.Errorf("POST Object returned %s: %s", resp.Status, responseBody)
|
||||
}
|
||||
return postPolicyUploadResult{
|
||||
versionID: resp.Header.Get("x-amz-version-id"),
|
||||
etag: resp.Header.Get("ETag"),
|
||||
}, nil
|
||||
}
|
||||
|
||||
func TestPostPolicyPreservesVersionHistoryAcrossPutAndDelete(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
client := getS3Client(t)
|
||||
bucket := getNewBucketName()
|
||||
key := "post-policy-history.bin"
|
||||
createBucket(t, client, bucket)
|
||||
defer deleteBucket(t, client, bucket)
|
||||
|
||||
putObject(t, client, bucket, key, "legacy-null")
|
||||
enableVersioning(t, client, bucket)
|
||||
|
||||
postOne, err := postPolicyUpload(ctx, client, bucket, key, []byte("post-one"))
|
||||
require.NoError(t, err)
|
||||
require.NotEmpty(t, postOne.versionID)
|
||||
require.NotEmpty(t, postOne.etag)
|
||||
|
||||
putTwo := putObject(t, client, bucket, key, "put-two")
|
||||
require.NotNil(t, putTwo.VersionId)
|
||||
require.NotEmpty(t, *putTwo.VersionId)
|
||||
|
||||
postThree, err := postPolicyUpload(ctx, client, bucket, key, []byte("post-three"))
|
||||
require.NoError(t, err)
|
||||
require.NotEmpty(t, postThree.versionID)
|
||||
|
||||
deleted, err := client.DeleteObject(ctx, &s3.DeleteObjectInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String(key),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
require.True(t, aws.ToBool(deleted.DeleteMarker))
|
||||
require.NotEmpty(t, aws.ToString(deleted.VersionId))
|
||||
|
||||
postFour, err := postPolicyUpload(ctx, client, bucket, key, []byte("post-four"))
|
||||
require.NoError(t, err)
|
||||
require.NotEmpty(t, postFour.versionID)
|
||||
|
||||
requireVersionBody(t, client, bucket, key, "null", []byte("legacy-null"), "legacy null version")
|
||||
requireVersionBody(t, client, bucket, key, postOne.versionID, []byte("post-one"), "first POST version")
|
||||
requireVersionBody(t, client, bucket, key, aws.ToString(putTwo.VersionId), []byte("put-two"), "interleaved PUT version")
|
||||
requireVersionBody(t, client, bucket, key, postThree.versionID, []byte("post-three"), "second POST version")
|
||||
requireVersionBody(t, client, bucket, key, postFour.versionID, []byte("post-four"), "POST after delete marker")
|
||||
|
||||
listed, err := client.ListObjectVersions(ctx, &s3.ListObjectVersionsInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Prefix: aws.String(key),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
require.Len(t, listed.Versions, 5)
|
||||
require.Len(t, listed.DeleteMarkers, 1)
|
||||
latest := 0
|
||||
latestVersionID := ""
|
||||
for _, version := range listed.Versions {
|
||||
if aws.ToBool(version.IsLatest) {
|
||||
latest++
|
||||
latestVersionID = aws.ToString(version.VersionId)
|
||||
}
|
||||
}
|
||||
require.Equal(t, 1, latest)
|
||||
require.Equal(t, postFour.versionID, latestVersionID)
|
||||
require.False(t, aws.ToBool(listed.DeleteMarkers[0].IsLatest))
|
||||
}
|
||||
|
||||
func TestPostPolicyRejectsIncompleteObjectLockHeaders(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
client := getS3Client(t)
|
||||
bucket := getNewBucketName()
|
||||
key := "post-policy-invalid-object-lock.bin"
|
||||
createBucketWithObjectLock(t, client, bucket)
|
||||
defer deleteBucket(t, client, bucket)
|
||||
|
||||
_, err := postPolicyUploadWithFields(ctx, client, bucket, key, []byte("rejected"), map[string]string{
|
||||
"x-amz-object-lock-mode": "GOVERNANCE",
|
||||
})
|
||||
require.ErrorContains(t, err, "400 Bad Request")
|
||||
require.ErrorContains(t, err, "<Code>InvalidRequest</Code>")
|
||||
|
||||
listed, listErr := client.ListObjectVersions(ctx, &s3.ListObjectVersionsInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Prefix: aws.String(key),
|
||||
})
|
||||
require.NoError(t, listErr)
|
||||
require.Empty(t, listed.Versions)
|
||||
}
|
||||
|
||||
func TestPostPolicyVersioningStateCompatibility(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
client := getS3Client(t)
|
||||
bucket := getNewBucketName()
|
||||
key := "post-policy-state.bin"
|
||||
createBucket(t, client, bucket)
|
||||
defer deleteBucket(t, client, bucket)
|
||||
|
||||
unconfigured, err := postPolicyUpload(ctx, client, bucket, key, []byte("unconfigured"))
|
||||
require.NoError(t, err)
|
||||
require.Empty(t, unconfigured.versionID)
|
||||
unconfigured, err = postPolicyUpload(ctx, client, bucket, key, []byte("unconfigured-two"))
|
||||
require.NoError(t, err)
|
||||
require.Empty(t, unconfigured.versionID)
|
||||
requireVersionBody(t, client, bucket, key, "null", []byte("unconfigured-two"), "unconfigured POST replaces null version")
|
||||
|
||||
enableVersioning(t, client, bucket)
|
||||
versioned := putObject(t, client, bucket, key, "numbered")
|
||||
require.NotEmpty(t, aws.ToString(versioned.VersionId))
|
||||
suspendVersioning(t, client, bucket)
|
||||
|
||||
suspended, err := postPolicyUpload(ctx, client, bucket, key, []byte("suspended-one"))
|
||||
require.NoError(t, err)
|
||||
require.Empty(t, suspended.versionID)
|
||||
suspended, err = postPolicyUpload(ctx, client, bucket, key, []byte("suspended-two"))
|
||||
require.NoError(t, err)
|
||||
require.Empty(t, suspended.versionID)
|
||||
requireVersionBody(t, client, bucket, key, "null", []byte("suspended-two"), "suspended POST replaces null version")
|
||||
requireVersionBody(t, client, bucket, key, aws.ToString(versioned.VersionId), []byte("numbered"), "suspended POST preserves numbered version")
|
||||
|
||||
listed, err := client.ListObjectVersions(ctx, &s3.ListObjectVersionsInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Prefix: aws.String(key),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
require.Len(t, listed.Versions, 2)
|
||||
}
|
||||
|
||||
func TestPostPolicyConcurrentWritesKeepEveryVersion(t *testing.T) {
|
||||
const writeCount = 8
|
||||
|
||||
ctx := context.Background()
|
||||
client := getS3Client(t)
|
||||
bucket := getNewBucketName()
|
||||
key := "post-policy-concurrent.bin"
|
||||
createBucket(t, client, bucket)
|
||||
defer deleteBucket(t, client, bucket)
|
||||
enableVersioning(t, client, bucket)
|
||||
|
||||
type writeResult struct {
|
||||
versionID string
|
||||
body []byte
|
||||
err error
|
||||
}
|
||||
start := make(chan struct{})
|
||||
results := make(chan writeResult, writeCount)
|
||||
var wg sync.WaitGroup
|
||||
for i := 0; i < writeCount; i++ {
|
||||
i := i
|
||||
wg.Add(1)
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
<-start
|
||||
body := []byte(fmt.Sprintf("concurrent-%d", i))
|
||||
if i%2 == 0 {
|
||||
post, err := postPolicyUpload(ctx, client, bucket, key, body)
|
||||
results <- writeResult{versionID: post.versionID, body: body, err: err}
|
||||
return
|
||||
}
|
||||
put, err := client.PutObject(ctx, &s3.PutObjectInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Key: aws.String(key),
|
||||
Body: bytes.NewReader(body),
|
||||
})
|
||||
versionID := ""
|
||||
if put != nil {
|
||||
versionID = aws.ToString(put.VersionId)
|
||||
}
|
||||
results <- writeResult{versionID: versionID, body: body, err: err}
|
||||
}()
|
||||
}
|
||||
close(start)
|
||||
wg.Wait()
|
||||
close(results)
|
||||
|
||||
seen := make(map[string]struct{}, writeCount)
|
||||
for result := range results {
|
||||
require.NoError(t, result.err)
|
||||
require.NotEmpty(t, result.versionID)
|
||||
_, duplicate := seen[result.versionID]
|
||||
require.False(t, duplicate, "each successful write must receive a unique version ID")
|
||||
seen[result.versionID] = struct{}{}
|
||||
requireVersionBody(t, client, bucket, key, result.versionID, result.body, "concurrent version body")
|
||||
}
|
||||
|
||||
listed, err := client.ListObjectVersions(ctx, &s3.ListObjectVersionsInput{
|
||||
Bucket: aws.String(bucket),
|
||||
Prefix: aws.String(key),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
require.Len(t, listed.Versions, writeCount)
|
||||
latest := 0
|
||||
for _, version := range listed.Versions {
|
||||
if aws.ToBool(version.IsLatest) {
|
||||
latest++
|
||||
}
|
||||
}
|
||||
require.Equal(t, 1, latest)
|
||||
}
|
||||
@@ -201,6 +201,7 @@ def main():
|
||||
"uri": args.catalog_url,
|
||||
"warehouse": args.warehouse,
|
||||
"prefix": args.prefix,
|
||||
"auth": {"type": "noop"},
|
||||
"s3.anonymous": "true", # Disable AWS request signing for unauthenticated access
|
||||
}
|
||||
)
|
||||
|
||||
@@ -173,6 +173,7 @@ def main():
|
||||
"uri": args.catalog_url,
|
||||
"warehouse": args.warehouse,
|
||||
"prefix": args.prefix,
|
||||
"auth": {"type": "noop"},
|
||||
"s3.access-key-id": args.access_key,
|
||||
"s3.secret-access-key": args.secret_key,
|
||||
}
|
||||
|
||||
@@ -99,6 +99,7 @@ def connect(args):
|
||||
"uri": args.catalog_url,
|
||||
"warehouse": f"s3://{args.bucket}/",
|
||||
"prefix": args.bucket,
|
||||
"auth": {"type": "noop"},
|
||||
"s3.endpoint": args.s3_endpoint,
|
||||
"s3.access-key-id": args.access_key,
|
||||
"s3.secret-access-key": args.secret_key,
|
||||
|
||||
@@ -146,6 +146,8 @@ type FilerNode struct {
|
||||
DataCenter string `json:"datacenter"`
|
||||
Rack string `json:"rack"`
|
||||
LastUpdated time.Time `json:"last_updated"`
|
||||
// MetricsPort is the node's advertised Prometheus port, 0 when disabled.
|
||||
MetricsPort uint32 `json:"metrics_port"`
|
||||
}
|
||||
|
||||
type MessageBrokerNode struct {
|
||||
@@ -159,6 +161,8 @@ type S3Node struct {
|
||||
Address string `json:"address"`
|
||||
DataCenter string `json:"datacenter"`
|
||||
LastUpdated time.Time `json:"last_updated"`
|
||||
// MetricsPort is the node's advertised Prometheus port, 0 when disabled.
|
||||
MetricsPort uint32 `json:"metrics_port"`
|
||||
}
|
||||
|
||||
// GetAdminData retrieves admin data as a struct (for reuse by both JSON and HTML handlers)
|
||||
@@ -356,6 +360,7 @@ func (s *AdminServer) getFilerNodesStatus() []FilerNode {
|
||||
DataCenter: node.DataCenter,
|
||||
Rack: node.Rack,
|
||||
LastUpdated: time.Now(),
|
||||
MetricsPort: node.MetricsPort,
|
||||
})
|
||||
}
|
||||
|
||||
@@ -433,6 +438,7 @@ func (s *AdminServer) getS3NodesStatus() []S3Node {
|
||||
Address: pb.ServerAddress(node.Address).ToHttpAddress(),
|
||||
DataCenter: node.DataCenter,
|
||||
LastUpdated: time.Now(),
|
||||
MetricsPort: node.MetricsPort,
|
||||
})
|
||||
}
|
||||
|
||||
|
||||
@@ -126,6 +126,13 @@ type AdminServer struct {
|
||||
dashSamples []dashSample
|
||||
dashSamplesMu sync.Mutex
|
||||
|
||||
// metricsStore holds scraped per-server Prometheus series for the
|
||||
// monitoring pages. Filled by the scrape loop in startMetricsScraper.
|
||||
// metricsDeriver turns raw counters and histograms into interval rates
|
||||
// and latency quantiles before they are stored.
|
||||
metricsStore *metricsStore
|
||||
metricsDeriver *metricsDeriver
|
||||
|
||||
// Filer discovery and caching
|
||||
cachedFilers []string
|
||||
lastFilerUpdate time.Time
|
||||
@@ -210,6 +217,8 @@ func NewAdminServer(masters string, filerGroup string, templateFS http.FileSyste
|
||||
pluginLock: lockManager,
|
||||
adminPresenceLock: presenceLock,
|
||||
bgCancel: bgCancel,
|
||||
metricsStore: newMetricsStore(),
|
||||
metricsDeriver: newMetricsDeriver(),
|
||||
}
|
||||
|
||||
// Initialize topic retention purger
|
||||
@@ -321,6 +330,7 @@ func NewAdminServer(masters string, filerGroup string, templateFS http.FileSyste
|
||||
}
|
||||
|
||||
go server.publishMaintenanceMetrics(bgCtx)
|
||||
go server.startMetricsScraper(bgCtx)
|
||||
|
||||
return server
|
||||
}
|
||||
@@ -1616,14 +1626,21 @@ func (as *AdminServer) GetConfigInfo(w http.ResponseWriter, r *http.Request) {
|
||||
})
|
||||
}
|
||||
|
||||
// StartWorkerGrpcServer starts the worker gRPC server
|
||||
func (s *AdminServer) StartWorkerGrpcServer(grpcPort int, listener net.Listener) error {
|
||||
// StartWorkerGrpcServer starts the worker gRPC server. bindIp is honored when no
|
||||
// listener is supplied so the worker gRPC does not wildcard-bind past -ip.
|
||||
func (s *AdminServer) StartWorkerGrpcServer(bindIp string, grpcPort int, listener net.Listener) error {
|
||||
if s.workerGrpcServer != nil {
|
||||
return fmt.Errorf("worker gRPC server is already running")
|
||||
}
|
||||
|
||||
s.workerGrpcServer = NewWorkerGrpcServer(s)
|
||||
return s.workerGrpcServer.StartWithTLS(grpcPort, listener)
|
||||
return s.workerGrpcServer.StartWithTLS(bindIp, grpcPort, listener)
|
||||
}
|
||||
|
||||
// WorkerGrpcMTLSEnabled reports whether the worker gRPC server actually loaded
|
||||
// grpc.admin mTLS credentials, not just whether they were configured.
|
||||
func (s *AdminServer) WorkerGrpcMTLSEnabled() bool {
|
||||
return s.workerGrpcServer != nil && s.workerGrpcServer.mtlsEnabled
|
||||
}
|
||||
|
||||
// StopWorkerGrpcServer stops the worker gRPC server
|
||||
|
||||
@@ -0,0 +1,174 @@
|
||||
package dash
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
)
|
||||
|
||||
type chartSeries struct {
|
||||
name string
|
||||
color string
|
||||
data []float64
|
||||
area bool
|
||||
}
|
||||
|
||||
type chartOptions struct {
|
||||
unit string
|
||||
threshold *float64
|
||||
labels []string
|
||||
}
|
||||
|
||||
func renderChartSVG(series []chartSeries, opts chartOptions) string {
|
||||
const w, h = 560.0, 190.0
|
||||
const l, r, t, b = 46.0, 8.0, 10.0, 22.0
|
||||
|
||||
if len(series) == 0 || len(series[0].data) < 2 {
|
||||
return flatChart(w, h, l, r, t, b)
|
||||
}
|
||||
n := len(series[0].data)
|
||||
min, max := computeRange(series, opts.threshold)
|
||||
if min > 0 {
|
||||
min = 0
|
||||
}
|
||||
if max == min {
|
||||
max = min + 1
|
||||
}
|
||||
span := max - min
|
||||
|
||||
px := func(i int) float64 { return l + float64(i)*(w-l-r)/float64(n-1) }
|
||||
py := func(v float64) float64 { return t + (h-t-b)*(1-(v-min)/span) }
|
||||
|
||||
var s strings.Builder
|
||||
for g := 0; g <= 4; g++ {
|
||||
v := min + span*float64(g)/4
|
||||
y := py(v)
|
||||
fmt.Fprintf(&s, `<line x1="%g" y1="%g" x2="%g" y2="%g" stroke="#e3e6f0" stroke-width="1"/>`, l, y, w-r, y)
|
||||
fmt.Fprintf(&s, `<text x="%g" y="%g" text-anchor="end" font-size="9" fill="#858796">%s</text>`, l-6, y+3, formatChartValue(v, opts.unit))
|
||||
}
|
||||
for _, i := range []int{0, n / 2, n - 1} {
|
||||
fmt.Fprintf(&s, `<text x="%g" y="%g" text-anchor="middle" font-size="9" fill="#858796">%s</text>`, px(i), h-6, xLabel(opts, i, n))
|
||||
}
|
||||
if opts.threshold != nil {
|
||||
y := py(*opts.threshold)
|
||||
fmt.Fprintf(&s, `<line x1="%g" y1="%g" x2="%g" y2="%g" stroke="#a5615c" stroke-width="1" stroke-dasharray="4 3"/>`, l, y, w-r, y)
|
||||
}
|
||||
for _, se := range series {
|
||||
var d strings.Builder
|
||||
for i, v := range se.data {
|
||||
if i == 0 {
|
||||
fmt.Fprintf(&d, "M%.1f %.1f", px(i), py(v))
|
||||
} else {
|
||||
fmt.Fprintf(&d, " L%.1f %.1f", px(i), py(v))
|
||||
}
|
||||
}
|
||||
if se.area {
|
||||
fmt.Fprintf(&s, `<path d="%s L%.1f %.1f L%.1f %.1f Z" fill="%s" opacity="0.12"/>`, d.String(), px(n-1), py(0), px(0), py(0), se.color)
|
||||
}
|
||||
fmt.Fprintf(&s, `<path d="%s" fill="none" stroke="%s" stroke-width="1.8" stroke-linejoin="round"/>`, d.String(), se.color)
|
||||
}
|
||||
return fmt.Sprintf(`<svg viewBox="0 0 %g %g" preserveAspectRatio="xMidYMid meet" style="width:100%%;height:auto">%s</svg>`, w, h, s.String())
|
||||
}
|
||||
|
||||
func flatChart(w, h, l, r, t, b float64) string {
|
||||
return fmt.Sprintf(`<svg viewBox="0 0 %g %g" preserveAspectRatio="xMidYMid meet" style="width:100%%;height:auto"><line x1="%g" y1="%g" x2="%g" y2="%g" stroke="#e3e6f0" stroke-width="1"/></svg>`, w, h, l, (t + (h-t-b)/2), w-r, (t + (h-t-b)/2))
|
||||
}
|
||||
|
||||
func computeRange(series []chartSeries, threshold *float64) (float64, float64) {
|
||||
min, max := 1e9, -1e9
|
||||
for _, se := range series {
|
||||
for _, v := range se.data {
|
||||
if v < min {
|
||||
min = v
|
||||
}
|
||||
if v > max {
|
||||
max = v
|
||||
}
|
||||
}
|
||||
}
|
||||
if threshold != nil && *threshold > max {
|
||||
max = *threshold * 1.15
|
||||
}
|
||||
return min, max
|
||||
}
|
||||
|
||||
func xLabel(opts chartOptions, i, n int) string {
|
||||
if i < len(opts.labels) {
|
||||
return opts.labels[i]
|
||||
}
|
||||
return fmt.Sprintf("-%dm", n-1-i)
|
||||
}
|
||||
|
||||
func formatChartValue(v float64, unit string) string {
|
||||
switch unit {
|
||||
case "bytes":
|
||||
return chartFormatBytes(int64(v))
|
||||
case "bps":
|
||||
return chartFormatBytes(int64(v)) + "/s"
|
||||
case "ms":
|
||||
return fmt.Sprintf("%.0f ms", v)
|
||||
case "pct":
|
||||
return fmt.Sprintf("%.0f%%", v)
|
||||
default:
|
||||
if v >= 1000 {
|
||||
return fmt.Sprintf("%.1fk", v/1000)
|
||||
}
|
||||
if v == float64(int64(v)) {
|
||||
return fmt.Sprintf("%d", int64(v))
|
||||
}
|
||||
return fmt.Sprintf("%.1f", v)
|
||||
}
|
||||
}
|
||||
|
||||
func chartFormatBytes(b int64) string {
|
||||
const u = 1024
|
||||
if b < u {
|
||||
return fmt.Sprintf("%d B", b)
|
||||
}
|
||||
div, exp := int64(u), 0
|
||||
for n := b / u; n >= u; n /= u {
|
||||
div *= u
|
||||
exp++
|
||||
}
|
||||
return fmt.Sprintf("%.1f %cB", float64(b)/float64(div), "KMGTPE"[exp])
|
||||
}
|
||||
|
||||
func renderLegend(series []chartSeries) string {
|
||||
var b strings.Builder
|
||||
b.WriteString(`<div class="legend">`)
|
||||
for _, se := range series {
|
||||
fmt.Fprintf(&b, `<span class="key"><span class="dot" style="background:%s"></span>%s</span>`, se.color, se.name)
|
||||
}
|
||||
b.WriteString(`</div>`)
|
||||
return b.String()
|
||||
}
|
||||
|
||||
func sampleTimes(n int) []string {
|
||||
out := make([]string, n)
|
||||
for i := 0; i < n; i++ {
|
||||
m := -(n - 1 - i)
|
||||
if m == 0 {
|
||||
out[i] = "now"
|
||||
} else {
|
||||
out[i] = fmt.Sprintf("%dm", m)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func seriesFromSamples(samples []metricsSample) []float64 {
|
||||
out := make([]float64, len(samples))
|
||||
for i, s := range samples {
|
||||
if v, ok := s.values[""]; ok {
|
||||
out[i] = v
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func timeLabels(samples []metricsSample) []string {
|
||||
out := make([]string, len(samples))
|
||||
for i, s := range samples {
|
||||
out[i] = s.t.Format("15:04")
|
||||
}
|
||||
return out
|
||||
}
|
||||
@@ -207,6 +207,7 @@ func (s *AdminServer) getTopologyViaGRPC(topology *ClusterTopology) error {
|
||||
DiskCapacity: diskCapacity,
|
||||
LastHeartbeat: time.Now(),
|
||||
RemoteSize: remoteSize,
|
||||
MetricsPort: node.MetricsPort,
|
||||
}
|
||||
|
||||
rackObj.Nodes = append(rackObj.Nodes, vs)
|
||||
|
||||
@@ -0,0 +1,123 @@
|
||||
package dash
|
||||
|
||||
import (
|
||||
"math"
|
||||
"sync"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Suffixes for series derived from raw scrapes. Counters become per-second
|
||||
// rates and histograms become latency quantiles, both computed from the delta
|
||||
// against the previous scrape so the values reflect the last interval rather
|
||||
// than process lifetime totals.
|
||||
const (
|
||||
suffixRate = ":rate"
|
||||
suffixP50 = ":p50"
|
||||
suffixP95 = ":p95"
|
||||
suffixP99 = ":p99"
|
||||
)
|
||||
|
||||
type prevScrape struct {
|
||||
t time.Time
|
||||
value float64
|
||||
buckets []histogramBucket
|
||||
}
|
||||
|
||||
type metricsDeriver struct {
|
||||
mu sync.Mutex
|
||||
prev map[string]prevScrape
|
||||
}
|
||||
|
||||
func newMetricsDeriver() *metricsDeriver {
|
||||
return &metricsDeriver{prev: make(map[string]prevScrape)}
|
||||
}
|
||||
|
||||
// record stores the raw value and, for counters and histograms, the derived
|
||||
// rate/quantile series for this interval.
|
||||
func (d *metricsDeriver) record(store *metricsStore, source string, m scrapedMetric, now time.Time) {
|
||||
key := source + "/" + m.name + "/" + labelKey(m.labels)
|
||||
|
||||
d.mu.Lock()
|
||||
prev, hadPrev := d.prev[key]
|
||||
d.prev[key] = prevScrape{t: now, value: m.value, buckets: m.buckets}
|
||||
d.mu.Unlock()
|
||||
|
||||
switch m.kind {
|
||||
case kindGauge:
|
||||
store.recordLabeled(source, m.name, m.labels, m.value, now)
|
||||
case kindCounter:
|
||||
if !hadPrev {
|
||||
return
|
||||
}
|
||||
dt := now.Sub(prev.t).Seconds()
|
||||
if dt <= 0 {
|
||||
return
|
||||
}
|
||||
delta := m.value - prev.value
|
||||
if delta < 0 {
|
||||
// Counter reset (process restart); skip this interval.
|
||||
return
|
||||
}
|
||||
store.recordLabeled(source, m.name+suffixRate, m.labels, delta/dt, now)
|
||||
case kindHistogram:
|
||||
if !hadPrev {
|
||||
return
|
||||
}
|
||||
delta := bucketDelta(prev.buckets, m.buckets)
|
||||
if len(delta) == 0 {
|
||||
return
|
||||
}
|
||||
for suffix, q := range map[string]float64{suffixP50: 0.5, suffixP95: 0.95, suffixP99: 0.99} {
|
||||
store.recordLabeled(source, m.name+suffix, m.labels, histogramQuantile(delta, q), now)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// bucketDelta subtracts cumulative bucket counts, yielding the distribution
|
||||
// observed during the interval. Returns nil on a reset or bucket mismatch.
|
||||
func bucketDelta(prev, cur []histogramBucket) []histogramBucket {
|
||||
if len(prev) != len(cur) {
|
||||
return nil
|
||||
}
|
||||
out := make([]histogramBucket, len(cur))
|
||||
for i := range cur {
|
||||
if cur[i].upperBound != prev[i].upperBound {
|
||||
return nil
|
||||
}
|
||||
c := cur[i].count - prev[i].count
|
||||
if c < 0 {
|
||||
return nil
|
||||
}
|
||||
out[i] = histogramBucket{upperBound: cur[i].upperBound, count: c}
|
||||
}
|
||||
if out[len(out)-1].count == 0 {
|
||||
return nil
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// histogramQuantile estimates a quantile from cumulative buckets by linear
|
||||
// interpolation within the matching bucket, matching Prometheus' approach.
|
||||
func histogramQuantile(buckets []histogramBucket, q float64) float64 {
|
||||
total := buckets[len(buckets)-1].count
|
||||
if total == 0 {
|
||||
return 0
|
||||
}
|
||||
rank := q * total
|
||||
prevCount, prevBound := 0.0, 0.0
|
||||
for _, b := range buckets {
|
||||
if b.count < rank {
|
||||
prevCount, prevBound = b.count, b.upperBound
|
||||
continue
|
||||
}
|
||||
if math.IsInf(b.upperBound, 1) {
|
||||
return prevBound
|
||||
}
|
||||
span := b.count - prevCount
|
||||
if span <= 0 {
|
||||
return b.upperBound
|
||||
}
|
||||
return prevBound + (b.upperBound-prevBound)*(rank-prevCount)/span
|
||||
}
|
||||
return buckets[len(buckets)-1].upperBound
|
||||
}
|
||||
@@ -0,0 +1,124 @@
|
||||
package dash
|
||||
|
||||
import (
|
||||
"math"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
func TestHistogramQuantile(t *testing.T) {
|
||||
// 100 observations spread evenly across 0-1s in 10 buckets.
|
||||
buckets := []histogramBucket{
|
||||
{0.1, 10}, {0.2, 20}, {0.3, 30}, {0.4, 40}, {0.5, 50},
|
||||
{0.6, 60}, {0.7, 70}, {0.8, 80}, {0.9, 90}, {1.0, 100},
|
||||
{math.Inf(1), 100},
|
||||
}
|
||||
for _, tc := range []struct {
|
||||
q float64
|
||||
want float64
|
||||
}{
|
||||
{0.5, 0.5},
|
||||
{0.95, 0.95},
|
||||
{0.99, 0.99},
|
||||
} {
|
||||
got := histogramQuantile(buckets, tc.q)
|
||||
if math.Abs(got-tc.want) > 1e-9 {
|
||||
t.Errorf("quantile(%v) = %v, want %v", tc.q, got, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestHistogramQuantileEmpty(t *testing.T) {
|
||||
if got := histogramQuantile([]histogramBucket{{math.Inf(1), 0}}, 0.99); got != 0 {
|
||||
t.Errorf("empty histogram quantile = %v, want 0", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestBucketDeltaResetAndMismatch(t *testing.T) {
|
||||
prev := []histogramBucket{{0.1, 5}, {math.Inf(1), 10}}
|
||||
if got := bucketDelta(prev, []histogramBucket{{0.1, 1}, {math.Inf(1), 2}}); got != nil {
|
||||
t.Errorf("counter reset should yield nil, got %v", got)
|
||||
}
|
||||
if got := bucketDelta(prev, []histogramBucket{{0.2, 5}, {math.Inf(1), 10}}); got != nil {
|
||||
t.Errorf("bound mismatch should yield nil, got %v", got)
|
||||
}
|
||||
if got := bucketDelta(prev, prev); got != nil {
|
||||
t.Errorf("no new observations should yield nil, got %v", got)
|
||||
}
|
||||
got := bucketDelta(prev, []histogramBucket{{0.1, 7}, {math.Inf(1), 14}})
|
||||
if len(got) != 2 || got[0].count != 2 || got[1].count != 4 {
|
||||
t.Errorf("unexpected delta %v", got)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDeriveCounterRate(t *testing.T) {
|
||||
store := newMetricsStore()
|
||||
d := newMetricsDeriver()
|
||||
t0 := time.Now()
|
||||
m := scrapedMetric{name: "reqs", kind: kindCounter, value: 100}
|
||||
|
||||
d.record(store, "volume/a", m, t0)
|
||||
if got := store.match("volume", "reqs"+suffixRate); len(got) != 0 {
|
||||
t.Fatalf("first scrape should not emit a rate, got %d series", len(got))
|
||||
}
|
||||
|
||||
m.value = 250
|
||||
d.record(store, "volume/a", m, t0.Add(15*time.Second))
|
||||
series := store.match("volume", "reqs"+suffixRate)
|
||||
if len(series) != 1 {
|
||||
t.Fatalf("expected 1 rate series, got %d", len(series))
|
||||
}
|
||||
samples := series[0].snapshot()
|
||||
if len(samples) != 1 {
|
||||
t.Fatalf("expected 1 sample, got %d", len(samples))
|
||||
}
|
||||
if want := 10.0; math.Abs(samples[0].values[""]-want) > 1e-9 {
|
||||
t.Errorf("rate = %v, want %v", samples[0].values[""], want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDeriveCounterResetSkipped(t *testing.T) {
|
||||
store := newMetricsStore()
|
||||
d := newMetricsDeriver()
|
||||
t0 := time.Now()
|
||||
m := scrapedMetric{name: "reqs", kind: kindCounter, value: 100}
|
||||
d.record(store, "volume/a", m, t0)
|
||||
m.value = 5
|
||||
d.record(store, "volume/a", m, t0.Add(15*time.Second))
|
||||
if got := store.match("volume", "reqs"+suffixRate); len(got) != 0 {
|
||||
t.Errorf("counter reset should emit no rate, got %d series", len(got))
|
||||
}
|
||||
}
|
||||
|
||||
func TestMetricsEndpoint(t *testing.T) {
|
||||
for _, tc := range []struct {
|
||||
node string
|
||||
port uint32
|
||||
want string
|
||||
}{
|
||||
// The advertised port replaces the node's service port.
|
||||
{"127.0.0.1:8080", 9327, "127.0.0.1:9327"},
|
||||
{"127.0.0.1:8888.18888", 9327, "127.0.0.1:9327"},
|
||||
{"[::1]:8080", 9327, "[::1]:9327"},
|
||||
// A node without -metricsPort must never be scraped.
|
||||
{"127.0.0.1:8080", 0, ""},
|
||||
} {
|
||||
if got := metricsEndpoint(tc.node, tc.port); got != tc.want {
|
||||
t.Errorf("metricsEndpoint(%q, %d) = %q, want %q", tc.node, tc.port, got, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestStoreRingIsBounded(t *testing.T) {
|
||||
s := newMetricsSeries()
|
||||
for i := 0; i < metricsMaxSamples+50; i++ {
|
||||
s.record(time.Now(), map[string]float64{"": float64(i)})
|
||||
}
|
||||
got := s.snapshot()
|
||||
if len(got) != metricsMaxSamples {
|
||||
t.Fatalf("len = %d, want %d", len(got), metricsMaxSamples)
|
||||
}
|
||||
if got[len(got)-1].values[""] != float64(metricsMaxSamples+49) {
|
||||
t.Errorf("newest sample not retained: %v", got[len(got)-1].values[""])
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,267 @@
|
||||
package dash
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"io"
|
||||
"net"
|
||||
"net/http"
|
||||
"sort"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
dto "github.com/prometheus/client_model/go"
|
||||
"github.com/prometheus/common/expfmt"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
|
||||
stats_collect "github.com/seaweedfs/seaweedfs/weed/stats"
|
||||
util_http "github.com/seaweedfs/seaweedfs/weed/util/http"
|
||||
)
|
||||
|
||||
type metricKind int
|
||||
|
||||
const (
|
||||
kindGauge metricKind = iota
|
||||
kindCounter
|
||||
kindHistogram
|
||||
)
|
||||
|
||||
type histogramBucket struct {
|
||||
upperBound float64
|
||||
count float64
|
||||
}
|
||||
|
||||
type scrapedMetric struct {
|
||||
name string
|
||||
labels map[string]string
|
||||
kind metricKind
|
||||
value float64
|
||||
buckets []histogramBucket
|
||||
}
|
||||
|
||||
func scrapeMetrics(ctx context.Context, target string) ([]scrapedMetric, error) {
|
||||
if !strings.HasPrefix(target, "http://") && !strings.HasPrefix(target, "https://") {
|
||||
target = "http://" + target
|
||||
}
|
||||
target = strings.TrimRight(target, "/") + "/metrics"
|
||||
|
||||
req, err := http.NewRequestWithContext(ctx, http.MethodGet, target, nil)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
req.Header.Set("Accept", string(expfmt.TextVersion))
|
||||
|
||||
client := util_http.GetGlobalHttpClient()
|
||||
resp, err := client.Do(req)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
return nil, fmt.Errorf("scrape %s: status %d", target, resp.StatusCode)
|
||||
}
|
||||
|
||||
return parsePrometheusText(resp.Body)
|
||||
}
|
||||
|
||||
func parsePrometheusText(r io.Reader) ([]scrapedMetric, error) {
|
||||
dec := expfmt.NewDecoder(r, expfmt.NewFormat(expfmt.TypeTextPlain))
|
||||
var out []scrapedMetric
|
||||
for {
|
||||
var fam dto.MetricFamily
|
||||
if err := dec.Decode(&fam); err != nil {
|
||||
if err == io.EOF {
|
||||
break
|
||||
}
|
||||
return nil, err
|
||||
}
|
||||
for _, m := range fam.Metric {
|
||||
out = append(out, toScrapedMetric(fam.GetName(), m))
|
||||
}
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func toScrapedMetric(name string, m *dto.Metric) scrapedMetric {
|
||||
labels := map[string]string{}
|
||||
for _, l := range m.Label {
|
||||
labels[l.GetName()] = l.GetValue()
|
||||
}
|
||||
sm := scrapedMetric{name: name, labels: labels}
|
||||
switch {
|
||||
case m.Counter != nil:
|
||||
sm.kind, sm.value = kindCounter, m.Counter.GetValue()
|
||||
case m.Histogram != nil:
|
||||
sm.kind = kindHistogram
|
||||
for _, b := range m.Histogram.Bucket {
|
||||
sm.buckets = append(sm.buckets, histogramBucket{upperBound: b.GetUpperBound(), count: float64(b.GetCumulativeCount())})
|
||||
}
|
||||
case m.Summary != nil:
|
||||
sm.kind, sm.value = kindCounter, m.Summary.GetSampleSum()
|
||||
case m.Gauge != nil:
|
||||
sm.value = m.Gauge.GetValue()
|
||||
case m.Untyped != nil:
|
||||
sm.value = m.Untyped.GetValue()
|
||||
}
|
||||
return sm
|
||||
}
|
||||
|
||||
// gatherLocalMetrics records the admin's own registry (maintenance tasks,
|
||||
// worker slots) without a network round trip.
|
||||
func (s *AdminServer) gatherLocalMetrics(now time.Time) {
|
||||
families, err := stats_collect.Gather.Gather()
|
||||
if err != nil {
|
||||
glog.V(1).Infof("gather admin metrics: %v", err)
|
||||
return
|
||||
}
|
||||
for _, fam := range families {
|
||||
for _, m := range fam.Metric {
|
||||
s.metricsDeriver.record(s.metricsStore, "admin/local", toScrapedMetric(fam.GetName(), m), now)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func (s *AdminServer) scrapeAllServers(ctx context.Context) {
|
||||
now := time.Now()
|
||||
s.gatherLocalMetrics(now)
|
||||
|
||||
targets := s.scrapeTargets()
|
||||
if len(targets) == 0 {
|
||||
return
|
||||
}
|
||||
type result struct {
|
||||
source string
|
||||
metrics []scrapedMetric
|
||||
err error
|
||||
}
|
||||
results := make(chan result, len(targets))
|
||||
scrapeCtx, cancel := context.WithTimeout(ctx, 5*time.Second)
|
||||
defer cancel()
|
||||
for _, t := range targets {
|
||||
go func(t scrapeTarget) {
|
||||
ms, err := scrapeMetrics(scrapeCtx, t.address)
|
||||
results <- result{source: t.source, metrics: ms, err: err}
|
||||
}(t)
|
||||
}
|
||||
for i := 0; i < len(targets); i++ {
|
||||
r := <-results
|
||||
if r.err != nil {
|
||||
glog.V(1).Infof("metrics scrape %s: %v", r.source, r.err)
|
||||
continue
|
||||
}
|
||||
for _, m := range r.metrics {
|
||||
s.metricsDeriver.record(s.metricsStore, r.source, m, now)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// scrapeTarget is one Prometheus endpoint. source is the endpoint address, not
|
||||
// a component name: a combined "weed server" advertises one listener for
|
||||
// master, volume, filer and S3 alike, and metric names already identify the
|
||||
// component. nodes records which cluster members advertised this endpoint, so
|
||||
// the UI can label it.
|
||||
type scrapeTarget struct {
|
||||
source string
|
||||
address string
|
||||
nodes []string
|
||||
}
|
||||
|
||||
// scrapeTargets lists the distinct metrics endpoints advertised by the cluster.
|
||||
// Nodes started without -metricsPort advertise 0 and are skipped, so nothing is
|
||||
// scraped from a client-facing service port.
|
||||
func (s *AdminServer) scrapeTargets() []scrapeTarget {
|
||||
byAddress := map[string][]string{}
|
||||
add := func(nodeAddress string, metricsPort uint32) {
|
||||
endpoint := metricsEndpoint(nodeAddress, metricsPort)
|
||||
if endpoint == "" {
|
||||
return
|
||||
}
|
||||
byAddress[endpoint] = append(byAddress[endpoint], nodeAddress)
|
||||
}
|
||||
|
||||
for _, m := range s.mastersWithMetricsPort() {
|
||||
add(m.address, m.metricsPort)
|
||||
}
|
||||
if topo, err := s.GetClusterTopology(); err == nil && topo != nil {
|
||||
for _, vs := range topo.VolumeServers {
|
||||
add(vs.Address, vs.MetricsPort)
|
||||
}
|
||||
}
|
||||
for _, f := range s.getFilerNodesStatus() {
|
||||
add(f.Address, f.MetricsPort)
|
||||
}
|
||||
for _, n := range s.getS3NodesStatus() {
|
||||
add(n.Address, n.MetricsPort)
|
||||
}
|
||||
|
||||
out := make([]scrapeTarget, 0, len(byAddress))
|
||||
for endpoint, nodes := range byAddress {
|
||||
sort.Strings(nodes)
|
||||
out = append(out, scrapeTarget{source: endpoint, address: endpoint, nodes: nodes})
|
||||
}
|
||||
sort.Slice(out, func(i, j int) bool { return out[i].source < out[j].source })
|
||||
return out
|
||||
}
|
||||
|
||||
// metricsEndpoint combines a node's host with its advertised metrics port.
|
||||
// Returns "" when the node does not run a metrics listener.
|
||||
func metricsEndpoint(nodeAddress string, metricsPort uint32) string {
|
||||
if metricsPort == 0 {
|
||||
return ""
|
||||
}
|
||||
host, _, err := net.SplitHostPort(nodeAddress)
|
||||
if err != nil {
|
||||
host = nodeAddress
|
||||
}
|
||||
return net.JoinHostPort(host, strconv.Itoa(int(metricsPort)))
|
||||
}
|
||||
|
||||
type masterMetricsTarget struct {
|
||||
address string
|
||||
metricsPort uint32
|
||||
}
|
||||
|
||||
// mastersWithMetricsPort asks each master for its own metrics port.
|
||||
// GetMasterConfiguration reports the configuration of the master that answers,
|
||||
// so it is called per address rather than once via the leader.
|
||||
func (s *AdminServer) mastersWithMetricsPort() []masterMetricsTarget {
|
||||
md, err := s.GetClusterMasters()
|
||||
if err != nil || md == nil {
|
||||
return nil
|
||||
}
|
||||
var out []masterMetricsTarget
|
||||
for _, m := range md.Masters {
|
||||
address := m.Address
|
||||
err := pb.WithMasterClient(context.Background(), false, pb.ServerAddress(address), s.grpcDialOption, false,
|
||||
func(client master_pb.SeaweedClient) error {
|
||||
resp, err := client.GetMasterConfiguration(context.Background(), &master_pb.GetMasterConfigurationRequest{})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
out = append(out, masterMetricsTarget{address: address, metricsPort: resp.MetricsPort})
|
||||
return nil
|
||||
})
|
||||
if err != nil {
|
||||
glog.V(1).Infof("master %s configuration: %v", address, err)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func (s *AdminServer) startMetricsScraper(ctx context.Context) {
|
||||
const interval = 15 * time.Second
|
||||
ticker := time.NewTicker(interval)
|
||||
defer ticker.Stop()
|
||||
s.scrapeAllServers(ctx)
|
||||
for {
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return
|
||||
case <-ticker.C:
|
||||
s.scrapeAllServers(ctx)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,148 @@
|
||||
package dash
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
)
|
||||
|
||||
const metricsMaxSamples = 240
|
||||
|
||||
type metricsSample struct {
|
||||
t time.Time
|
||||
values map[string]float64
|
||||
}
|
||||
|
||||
type metricsSeries struct {
|
||||
mu sync.Mutex
|
||||
samples []metricsSample
|
||||
}
|
||||
|
||||
func newMetricsSeries() *metricsSeries {
|
||||
return &metricsSeries{samples: make([]metricsSample, 0, metricsMaxSamples)}
|
||||
}
|
||||
|
||||
func (s *metricsSeries) record(t time.Time, values map[string]float64) {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
s.samples = append(s.samples, metricsSample{t: t, values: values})
|
||||
if len(s.samples) > metricsMaxSamples {
|
||||
s.samples = s.samples[len(s.samples)-metricsMaxSamples:]
|
||||
}
|
||||
}
|
||||
|
||||
func (s *metricsSeries) snapshot() []metricsSample {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
out := make([]metricsSample, len(s.samples))
|
||||
copy(out, s.samples)
|
||||
return out
|
||||
}
|
||||
|
||||
type metricsStore struct {
|
||||
mu sync.Mutex
|
||||
series map[string]*metricsSeries
|
||||
}
|
||||
|
||||
func newMetricsStore() *metricsStore {
|
||||
return &metricsStore{series: make(map[string]*metricsSeries)}
|
||||
}
|
||||
|
||||
func (s *metricsStore) record(source, name string, value float64, t time.Time) {
|
||||
s.mu.Lock()
|
||||
key := source + "/" + name
|
||||
ser, ok := s.series[key]
|
||||
if !ok {
|
||||
ser = newMetricsSeries()
|
||||
s.series[key] = ser
|
||||
}
|
||||
s.mu.Unlock()
|
||||
ser.record(t, map[string]float64{"": value})
|
||||
}
|
||||
|
||||
func (s *metricsStore) recordLabeled(source, name string, labels map[string]string, value float64, t time.Time) {
|
||||
s.mu.Lock()
|
||||
key := source + "/" + name + "/" + labelKey(labels)
|
||||
ser, ok := s.series[key]
|
||||
if !ok {
|
||||
ser = newMetricsSeries()
|
||||
s.series[key] = ser
|
||||
}
|
||||
s.mu.Unlock()
|
||||
ser.record(t, map[string]float64{"": value})
|
||||
}
|
||||
|
||||
func (s *metricsStore) get(source, name string) []metricsSample {
|
||||
s.mu.Lock()
|
||||
key := source + "/" + name
|
||||
ser, ok := s.series[key]
|
||||
s.mu.Unlock()
|
||||
if !ok {
|
||||
return nil
|
||||
}
|
||||
return ser.snapshot()
|
||||
}
|
||||
|
||||
func (s *metricsStore) getLabeled(source, name string, labels map[string]string) []metricsSample {
|
||||
s.mu.Lock()
|
||||
key := source + "/" + name + "/" + labelKey(labels)
|
||||
ser, ok := s.series[key]
|
||||
s.mu.Unlock()
|
||||
if !ok {
|
||||
return nil
|
||||
}
|
||||
return ser.snapshot()
|
||||
}
|
||||
|
||||
// match returns every series whose source has the given prefix and whose
|
||||
// metric name matches exactly.
|
||||
func (s *metricsStore) match(sourcePrefix, name string) []*metricsSeries {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
var out []*metricsSeries
|
||||
for k, ser := range s.series {
|
||||
if !strings.HasPrefix(k, sourcePrefix) {
|
||||
continue
|
||||
}
|
||||
rest := k[len(sourcePrefix):]
|
||||
if !strings.HasPrefix(rest, "/") {
|
||||
continue
|
||||
}
|
||||
rest = rest[1:]
|
||||
// rest is either "<addr>/<metric>[/<labels>]" or "<metric>[/<labels>]".
|
||||
if rest == name || strings.HasPrefix(rest, name+"/") {
|
||||
out = append(out, ser)
|
||||
continue
|
||||
}
|
||||
if i := strings.Index(rest, "/"); i >= 0 {
|
||||
tail := rest[i+1:]
|
||||
if tail == name || strings.HasPrefix(tail, name+"/") {
|
||||
out = append(out, ser)
|
||||
}
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func labelKey(labels map[string]string) string {
|
||||
if len(labels) == 0 {
|
||||
return ""
|
||||
}
|
||||
keys := make([]string, 0, len(labels))
|
||||
for k := range labels {
|
||||
keys = append(keys, k)
|
||||
}
|
||||
for i := 1; i < len(keys); i++ {
|
||||
for j := i; j > 0 && keys[j] < keys[j-1]; j-- {
|
||||
keys[j], keys[j-1] = keys[j-1], keys[j]
|
||||
}
|
||||
}
|
||||
out := ""
|
||||
for i, k := range keys {
|
||||
if i > 0 {
|
||||
out += ","
|
||||
}
|
||||
out += k + "=" + labels[k]
|
||||
}
|
||||
return out
|
||||
}
|
||||
@@ -2,7 +2,9 @@ package dash
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/credential"
|
||||
@@ -10,6 +12,10 @@ import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api/policy_engine"
|
||||
)
|
||||
|
||||
// ErrPolicyStillAttached is returned when deleting a managed policy that is
|
||||
// still attached to one or more users or groups.
|
||||
var ErrPolicyStillAttached = errors.New("policy is still attached")
|
||||
|
||||
type IAMPolicy struct {
|
||||
Name string `json:"name"`
|
||||
Document policy_engine.PolicyDocument `json:"document"`
|
||||
@@ -146,7 +152,10 @@ func (s *AdminServer) UpdatePolicy(name string, document policy_engine.PolicyDoc
|
||||
return policyManager.UpdatePolicy(ctx, name, document)
|
||||
}
|
||||
|
||||
// DeletePolicy deletes an IAM policy
|
||||
// DeletePolicy deletes an IAM policy. Deletion is rejected while the policy is
|
||||
// still attached to any user or group, matching AWS IAM behavior and the IAM
|
||||
// API handler, so a deleted policy name never lingers in an attached policy
|
||||
// list.
|
||||
func (s *AdminServer) DeletePolicy(name string) error {
|
||||
policyManager := s.GetPolicyManager()
|
||||
if policyManager == nil {
|
||||
@@ -154,9 +163,67 @@ func (s *AdminServer) DeletePolicy(name string) error {
|
||||
}
|
||||
|
||||
ctx := context.Background()
|
||||
attached, err := s.IsPolicyAttached(ctx, name)
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to check policy attachments: %w", err)
|
||||
}
|
||||
if len(attached) > 0 {
|
||||
return fmt.Errorf("policy %s is still attached to: %s: %w", name, strings.Join(attached, ", "), ErrPolicyStillAttached)
|
||||
}
|
||||
|
||||
return policyManager.DeletePolicy(ctx, name)
|
||||
}
|
||||
|
||||
// IsPolicyAttached returns the names of users and groups that still have the
|
||||
// given managed policy attached. The returned entries are prefixed with
|
||||
// "user:" or "group:". Returns nil when the policy is not attached anywhere.
|
||||
func (s *AdminServer) IsPolicyAttached(ctx context.Context, policyName string) ([]string, error) {
|
||||
if s.credentialManager == nil {
|
||||
return nil, fmt.Errorf("credential manager not available")
|
||||
}
|
||||
|
||||
var attached []string
|
||||
|
||||
usernames, err := s.credentialManager.ListUsers(ctx)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("failed to list users: %w", err)
|
||||
}
|
||||
for _, username := range usernames {
|
||||
policies, err := s.credentialManager.ListAttachedUserPolicies(ctx, username)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("failed to list policies for user %s: %w", username, err)
|
||||
}
|
||||
for _, p := range policies {
|
||||
if p == policyName {
|
||||
attached = append(attached, "user:"+username)
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
groupNames, err := s.credentialManager.ListGroups(ctx)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("failed to list groups: %w", err)
|
||||
}
|
||||
for _, groupName := range groupNames {
|
||||
group, err := s.credentialManager.GetGroup(ctx, groupName)
|
||||
if errors.Is(err, credential.ErrGroupNotFound) {
|
||||
continue
|
||||
}
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("failed to get group %s: %w", groupName, err)
|
||||
}
|
||||
for _, p := range group.PolicyNames {
|
||||
if p == policyName {
|
||||
attached = append(attached, "group:"+groupName)
|
||||
break
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return attached, nil
|
||||
}
|
||||
|
||||
// GetPolicy retrieves a specific IAM policy
|
||||
func (s *AdminServer) GetPolicy(name string) (*IAMPolicy, error) {
|
||||
policyManager := s.GetPolicyManager()
|
||||
|
||||
@@ -0,0 +1,160 @@
|
||||
package dash
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"testing"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/credential"
|
||||
_ "github.com/seaweedfs/seaweedfs/weed/credential/memory" // register memory store
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/iam_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api/policy_engine"
|
||||
)
|
||||
|
||||
func newAdminServerWithMemoryStore(t *testing.T) *AdminServer {
|
||||
t.Helper()
|
||||
cm, err := credential.NewCredentialManagerWithDefaults(credential.StoreTypeMemory)
|
||||
if err != nil {
|
||||
t.Fatalf("failed to create credential manager: %v", err)
|
||||
}
|
||||
return &AdminServer{credentialManager: cm}
|
||||
}
|
||||
|
||||
func samplePolicyDocument() policy_engine.PolicyDocument {
|
||||
return policy_engine.PolicyDocument{
|
||||
Version: "2012-10-17",
|
||||
Statement: []policy_engine.PolicyStatement{{
|
||||
Effect: policy_engine.PolicyEffectAllow,
|
||||
Action: policy_engine.NewStringOrStringSlice("s3:GetObject"),
|
||||
Resource: policy_engine.NewStringOrStringSlicePtr("arn:aws:s3:::test/*"),
|
||||
}},
|
||||
}
|
||||
}
|
||||
|
||||
func TestIsPolicyAttached(t *testing.T) {
|
||||
server := newAdminServerWithMemoryStore(t)
|
||||
ctx := context.Background()
|
||||
const policyName = "policy_a"
|
||||
|
||||
if err := server.CreatePolicy(policyName, samplePolicyDocument()); err != nil {
|
||||
t.Fatalf("CreatePolicy: %v", err)
|
||||
}
|
||||
|
||||
if attached, err := server.IsPolicyAttached(ctx, policyName); err != nil {
|
||||
t.Fatalf("IsPolicyAttached: %v", err)
|
||||
} else if len(attached) != 0 {
|
||||
t.Fatalf("expected no attachments, got %v", attached)
|
||||
}
|
||||
|
||||
if err := server.credentialManager.CreateUser(ctx, &iam_pb.Identity{Name: "alice"}); err != nil {
|
||||
t.Fatalf("CreateUser: %v", err)
|
||||
}
|
||||
if err := server.credentialManager.AttachUserPolicy(ctx, "alice", policyName); err != nil {
|
||||
t.Fatalf("AttachUserPolicy: %v", err)
|
||||
}
|
||||
|
||||
attached, err := server.IsPolicyAttached(ctx, policyName)
|
||||
if err != nil {
|
||||
t.Fatalf("IsPolicyAttached: %v", err)
|
||||
}
|
||||
if len(attached) != 1 || attached[0] != "user:alice" {
|
||||
t.Fatalf("expected [user:alice], got %v", attached)
|
||||
}
|
||||
|
||||
if err := server.credentialManager.CreateGroup(ctx, &iam_pb.Group{Name: "devs", PolicyNames: []string{policyName}}); err != nil {
|
||||
t.Fatalf("CreateGroup: %v", err)
|
||||
}
|
||||
attached, err = server.IsPolicyAttached(ctx, policyName)
|
||||
if err != nil {
|
||||
t.Fatalf("IsPolicyAttached: %v", err)
|
||||
}
|
||||
if len(attached) != 2 {
|
||||
t.Fatalf("expected 2 attachments, got %v", attached)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDeletePolicyRejectsWhenAttachedToUser(t *testing.T) {
|
||||
server := newAdminServerWithMemoryStore(t)
|
||||
ctx := context.Background()
|
||||
const policyName = "policy_u"
|
||||
|
||||
if err := server.CreatePolicy(policyName, samplePolicyDocument()); err != nil {
|
||||
t.Fatalf("CreatePolicy: %v", err)
|
||||
}
|
||||
if err := server.credentialManager.CreateUser(ctx, &iam_pb.Identity{Name: "bob"}); err != nil {
|
||||
t.Fatalf("CreateUser: %v", err)
|
||||
}
|
||||
if err := server.credentialManager.AttachUserPolicy(ctx, "bob", policyName); err != nil {
|
||||
t.Fatalf("AttachUserPolicy: %v", err)
|
||||
}
|
||||
|
||||
if err := server.DeletePolicy(policyName); !errors.Is(err, ErrPolicyStillAttached) {
|
||||
t.Fatalf("expected ErrPolicyStillAttached, got %v", err)
|
||||
}
|
||||
|
||||
if p, err := server.GetPolicy(policyName); err != nil {
|
||||
t.Fatalf("policy should still exist after rejected deletion: %v", err)
|
||||
} else if p == nil {
|
||||
t.Fatal("policy should still exist after rejected deletion, got nil")
|
||||
}
|
||||
|
||||
attached, err := server.credentialManager.ListAttachedUserPolicies(ctx, "bob")
|
||||
if err != nil {
|
||||
t.Fatalf("ListAttachedUserPolicies: %v", err)
|
||||
}
|
||||
if len(attached) != 1 || attached[0] != policyName {
|
||||
t.Fatalf("expected policy %q to remain attached, got %v", policyName, attached)
|
||||
}
|
||||
|
||||
if err := server.credentialManager.DetachUserPolicy(ctx, "bob", policyName); err != nil {
|
||||
t.Fatalf("DetachUserPolicy: %v", err)
|
||||
}
|
||||
if err := server.DeletePolicy(policyName); err != nil {
|
||||
t.Fatalf("DeletePolicy after detach failed: %v", err)
|
||||
}
|
||||
if p, err := server.GetPolicy(policyName); err != nil {
|
||||
t.Fatalf("GetPolicy after detach-delete: %v", err)
|
||||
} else if p != nil {
|
||||
t.Fatalf("policy should be gone after deletion, got %v", p)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDeletePolicyRejectsWhenAttachedToGroup(t *testing.T) {
|
||||
server := newAdminServerWithMemoryStore(t)
|
||||
ctx := context.Background()
|
||||
const policyName = "policy_g"
|
||||
|
||||
if err := server.CreatePolicy(policyName, samplePolicyDocument()); err != nil {
|
||||
t.Fatalf("CreatePolicy: %v", err)
|
||||
}
|
||||
if err := server.credentialManager.CreateGroup(ctx, &iam_pb.Group{Name: "team_g", PolicyNames: []string{policyName}}); err != nil {
|
||||
t.Fatalf("CreateGroup: %v", err)
|
||||
}
|
||||
|
||||
if err := server.DeletePolicy(policyName); !errors.Is(err, ErrPolicyStillAttached) {
|
||||
t.Fatalf("expected ErrPolicyStillAttached, got %v", err)
|
||||
}
|
||||
|
||||
if p, err := server.GetPolicy(policyName); err != nil {
|
||||
t.Fatalf("policy should still exist after rejected deletion: %v", err)
|
||||
} else if p == nil {
|
||||
t.Fatal("policy should still exist after rejected deletion, got nil")
|
||||
}
|
||||
}
|
||||
|
||||
func TestDeletePolicySucceedsWhenNotAttached(t *testing.T) {
|
||||
server := newAdminServerWithMemoryStore(t)
|
||||
const policyName = "policy_free"
|
||||
|
||||
if err := server.CreatePolicy(policyName, samplePolicyDocument()); err != nil {
|
||||
t.Fatalf("CreatePolicy: %v", err)
|
||||
}
|
||||
if err := server.DeletePolicy(policyName); err != nil {
|
||||
t.Fatalf("DeletePolicy for unattached policy failed: %v", err)
|
||||
}
|
||||
if p, err := server.GetPolicy(policyName); err != nil {
|
||||
t.Fatalf("GetPolicy after delete: %v", err)
|
||||
} else if p != nil {
|
||||
t.Fatalf("policy should be gone after deletion, got %v", p)
|
||||
}
|
||||
}
|
||||
@@ -50,6 +50,8 @@ type VolumeServer struct {
|
||||
DiskUsage int64 `json:"disk_usage"`
|
||||
DiskCapacity int64 `json:"disk_capacity"`
|
||||
LastHeartbeat time.Time `json:"last_heartbeat"`
|
||||
// MetricsPort is the node's advertised Prometheus port, 0 when disabled.
|
||||
MetricsPort uint32 `json:"metrics_port"`
|
||||
|
||||
// EC shard information
|
||||
EcVolumes int `json:"ec_volumes"` // Number of EC volumes this server has shards for
|
||||
|
||||
@@ -47,10 +47,11 @@ type WorkerGrpcServer struct {
|
||||
logRequestsMutex sync.RWMutex
|
||||
|
||||
// gRPC server
|
||||
grpcServer *grpc.Server
|
||||
listener net.Listener
|
||||
running bool
|
||||
stopChan chan struct{}
|
||||
grpcServer *grpc.Server
|
||||
listener net.Listener
|
||||
running bool
|
||||
stopChan chan struct{}
|
||||
mtlsEnabled bool
|
||||
}
|
||||
|
||||
// LogRequestContext tracks pending log requests
|
||||
@@ -84,22 +85,25 @@ func NewWorkerGrpcServer(adminServer *AdminServer) *WorkerGrpcServer {
|
||||
}
|
||||
|
||||
// StartWithTLS starts the gRPC server on the specified port with optional TLS.
|
||||
// A caller that already holds the port passes its listener instead.
|
||||
func (s *WorkerGrpcServer) StartWithTLS(port int, listener net.Listener) error {
|
||||
// A caller that already holds the port passes its listener instead. When no
|
||||
// listener is supplied the server binds to bindIp to honor the operator's -ip.
|
||||
func (s *WorkerGrpcServer) StartWithTLS(bindIp string, port int, listener net.Listener) error {
|
||||
if s.running {
|
||||
return fmt.Errorf("worker gRPC server is already running")
|
||||
}
|
||||
|
||||
if listener == nil {
|
||||
var err error
|
||||
listener, err = net.Listen("tcp", fmt.Sprintf(":%d", port))
|
||||
listener, err = net.Listen("tcp", util.JoinHostPort(bindIp, port))
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to listen on port %d: %v", port, err)
|
||||
}
|
||||
}
|
||||
|
||||
// Create gRPC server with optional TLS
|
||||
grpcServer := pb.NewGrpcServer(security.LoadServerTLS(util.GetViper(), "grpc.admin"))
|
||||
tlsOption, _ := security.LoadServerTLS(util.GetViper(), "grpc.admin")
|
||||
s.mtlsEnabled = tlsOption != nil
|
||||
grpcServer := pb.NewGrpcServer(tlsOption)
|
||||
|
||||
worker_pb.RegisterWorkerServiceServer(grpcServer, s)
|
||||
if plugin := s.adminServer.GetPlugin(); plugin != nil {
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
package handlers
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"net/http"
|
||||
"time"
|
||||
@@ -186,8 +187,13 @@ func (h *PolicyHandlers) DeletePolicy(w http.ResponseWriter, r *http.Request) {
|
||||
// Delete the policy
|
||||
err = h.adminServer.DeletePolicy(policyName)
|
||||
if err != nil {
|
||||
glog.Errorf("Failed to delete policy %s: %v", policyName, err)
|
||||
writeJSONError(w, http.StatusInternalServerError, "Failed to delete policy: "+err.Error())
|
||||
status := http.StatusInternalServerError
|
||||
if errors.Is(err, dash.ErrPolicyStillAttached) {
|
||||
status = http.StatusConflict
|
||||
} else {
|
||||
glog.Errorf("Failed to delete policy %s: %v", policyName, err)
|
||||
}
|
||||
writeJSONError(w, status, "Failed to delete policy: "+err.Error())
|
||||
return
|
||||
}
|
||||
|
||||
|
||||
@@ -27,6 +27,9 @@ type ClusterNode struct {
|
||||
CreatedTs time.Time
|
||||
DataCenter DataCenter
|
||||
Rack Rack
|
||||
// MetricsPort is the node's Prometheus /metrics port, or 0 when the node
|
||||
// does not run a metrics listener.
|
||||
MetricsPort uint32
|
||||
}
|
||||
|
||||
type ClusterNodeGroups struct {
|
||||
@@ -53,11 +56,11 @@ func (g *ClusterNodeGroups) getGroupMembers(filerGroup FilerGroupName, createIfN
|
||||
return members
|
||||
}
|
||||
|
||||
func (g *ClusterNodeGroups) AddClusterNode(filerGroup FilerGroupName, nodeType string, dataCenter DataCenter, rack Rack, address pb.ServerAddress, version string) []*master_pb.KeepConnectedResponse {
|
||||
func (g *ClusterNodeGroups) AddClusterNode(filerGroup FilerGroupName, nodeType string, dataCenter DataCenter, rack Rack, address pb.ServerAddress, version string, metricsPort uint32) []*master_pb.KeepConnectedResponse {
|
||||
g.Lock()
|
||||
defer g.Unlock()
|
||||
m := g.getGroupMembers(filerGroup, true)
|
||||
if t := m.addMember(dataCenter, rack, address, version); t != nil {
|
||||
if t := m.addMember(dataCenter, rack, address, version, metricsPort); t != nil {
|
||||
return buildClusterNodeUpdateMessage(true, filerGroup, nodeType, address)
|
||||
}
|
||||
return nil
|
||||
@@ -95,15 +98,15 @@ func NewCluster() *Cluster {
|
||||
}
|
||||
}
|
||||
|
||||
func (cluster *Cluster) AddClusterNode(ns, nodeType string, dataCenter DataCenter, rack Rack, address pb.ServerAddress, version string) []*master_pb.KeepConnectedResponse {
|
||||
func (cluster *Cluster) AddClusterNode(ns, nodeType string, dataCenter DataCenter, rack Rack, address pb.ServerAddress, version string, metricsPort uint32) []*master_pb.KeepConnectedResponse {
|
||||
filerGroup := FilerGroupName(ns)
|
||||
switch nodeType {
|
||||
case FilerType:
|
||||
return cluster.filerGroups.AddClusterNode(filerGroup, nodeType, dataCenter, rack, address, version)
|
||||
return cluster.filerGroups.AddClusterNode(filerGroup, nodeType, dataCenter, rack, address, version, metricsPort)
|
||||
case BrokerType:
|
||||
return cluster.brokerGroups.AddClusterNode(filerGroup, nodeType, dataCenter, rack, address, version)
|
||||
return cluster.brokerGroups.AddClusterNode(filerGroup, nodeType, dataCenter, rack, address, version, metricsPort)
|
||||
case S3Type:
|
||||
return cluster.s3Groups.AddClusterNode(filerGroup, nodeType, dataCenter, rack, address, version)
|
||||
return cluster.s3Groups.AddClusterNode(filerGroup, nodeType, dataCenter, rack, address, version, metricsPort)
|
||||
case MasterType:
|
||||
return buildClusterNodeUpdateMessage(true, filerGroup, nodeType, address)
|
||||
}
|
||||
|
||||
@@ -16,7 +16,7 @@ func TestConcurrentAddRemoveNodes(t *testing.T) {
|
||||
go func(i int) {
|
||||
defer wg.Done()
|
||||
address := strconv.Itoa(i)
|
||||
c.AddClusterNode("", "filer", "", "", pb.ServerAddress(address), "23.45")
|
||||
c.AddClusterNode("", "filer", "", "", pb.ServerAddress(address), "23.45", 0)
|
||||
}(i)
|
||||
}
|
||||
wg.Wait()
|
||||
@@ -43,8 +43,8 @@ func TestConcurrentAddRemoveNodes(t *testing.T) {
|
||||
func TestListClusterNodeUpdates(t *testing.T) {
|
||||
c := NewCluster()
|
||||
filer := pb.ServerAddress("10.0.0.20:8888")
|
||||
c.AddClusterNode("group", FilerType, "dc1", "rack1", filer, "test")
|
||||
c.AddClusterNode("group", BrokerType, "dc1", "rack1", pb.ServerAddress("10.0.0.20:17777"), "test")
|
||||
c.AddClusterNode("group", FilerType, "dc1", "rack1", filer, "test", 0)
|
||||
c.AddClusterNode("group", BrokerType, "dc1", "rack1", pb.ServerAddress("10.0.0.20:17777"), "test", 0)
|
||||
|
||||
updates := c.ListClusterNodeUpdates("group", FilerType)
|
||||
if len(updates) != 1 {
|
||||
@@ -64,7 +64,7 @@ func TestListClusterNodeUpdates(t *testing.T) {
|
||||
func TestIsKnownNode(t *testing.T) {
|
||||
c := NewCluster()
|
||||
filer := pb.ServerAddress("10.0.0.20:8888")
|
||||
c.AddClusterNode("", FilerType, "dc1", "rack1", filer, "test")
|
||||
c.AddClusterNode("", FilerType, "dc1", "rack1", filer, "test", 0)
|
||||
|
||||
if !c.IsKnownNode(FilerType, filer) {
|
||||
t.Fatalf("registered filer %s should be known", filer)
|
||||
|
||||
@@ -16,18 +16,21 @@ func newGroupMembers() *GroupMembers {
|
||||
}
|
||||
}
|
||||
|
||||
func (m *GroupMembers) addMember(dataCenter DataCenter, rack Rack, address pb.ServerAddress, version string) *ClusterNode {
|
||||
func (m *GroupMembers) addMember(dataCenter DataCenter, rack Rack, address pb.ServerAddress, version string, metricsPort uint32) *ClusterNode {
|
||||
if existingNode, found := m.members[address]; found {
|
||||
existingNode.counter++
|
||||
// A restarted node may have gained or lost its metrics listener.
|
||||
existingNode.MetricsPort = metricsPort
|
||||
return nil
|
||||
}
|
||||
t := &ClusterNode{
|
||||
Address: address,
|
||||
Version: version,
|
||||
counter: 1,
|
||||
CreatedTs: time.Now(),
|
||||
DataCenter: dataCenter,
|
||||
Rack: rack,
|
||||
Address: address,
|
||||
Version: version,
|
||||
counter: 1,
|
||||
CreatedTs: time.Now(),
|
||||
DataCenter: dataCenter,
|
||||
Rack: rack,
|
||||
MetricsPort: metricsPort,
|
||||
}
|
||||
m.members[address] = t
|
||||
return t
|
||||
|
||||
+112
-37
@@ -50,22 +50,31 @@ type AdminOptions struct {
|
||||
adminPassword *string
|
||||
readOnlyUser *string
|
||||
readOnlyPassword *string
|
||||
dataDir *string
|
||||
icebergPort *int
|
||||
lancePort *int
|
||||
urlPrefix *string
|
||||
metricsHttpPort *int
|
||||
metricsHttpIp *string
|
||||
debug *bool
|
||||
debugPort *int
|
||||
cpuProfile *string
|
||||
memProfile *string
|
||||
// nil for callers other than runAdmin (e.g. `weed mini`)
|
||||
allowInsecureBind *bool
|
||||
dataDir *string
|
||||
icebergPort *int
|
||||
lancePort *int
|
||||
urlPrefix *string
|
||||
metricsHttpPort *int
|
||||
metricsHttpIp *string
|
||||
debug *bool
|
||||
debugPort *int
|
||||
cpuProfile *string
|
||||
memProfile *string
|
||||
|
||||
// workerGrpcListener, when set, is a listener already bound to grpcPort by
|
||||
// the caller. `weed mini` reserves the port this way because the admin
|
||||
// binds it only after every other service is up.
|
||||
workerGrpcListener net.Listener
|
||||
|
||||
// workerGrpcBindIp, when non-empty, is the address the worker gRPC
|
||||
// listener binds to. It is separate from ip because the worker gRPC has
|
||||
// no password auth (its mTLS comes from grpc.admin, not https.admin), so
|
||||
// it must not follow ip's auto-upgrade to 0.0.0.0 based on adminPassword.
|
||||
// `weed mini` leaves it empty to fall back to ip.
|
||||
workerGrpcBindIp string
|
||||
|
||||
// defaultS3PublicEndpoint, when set, is used for object URLs when
|
||||
// s3.public_endpoint is not configured. `weed mini` sets it to its own
|
||||
// S3 address.
|
||||
@@ -76,7 +85,7 @@ func init() {
|
||||
cmdAdmin.Run = runAdmin // break init cycle
|
||||
a.port = cmdAdmin.Flag.Int("port", 23646, "admin server port")
|
||||
a.grpcPort = cmdAdmin.Flag.Int("port.grpc", 0, "gRPC server port for worker connections (default: http port + 10000)")
|
||||
a.ip = cmdAdmin.Flag.String("ip", "127.0.0.1", "ip address to listen on. Default is loopback; set to 0.0.0.0 to listen on all interfaces (requires -adminPassword or [https.admin] mTLS in security.toml).")
|
||||
a.ip = cmdAdmin.Flag.String("ip", "127.0.0.1", "ip address to listen on. Defaults to loopback when auth is disabled, or 0.0.0.0 when -adminPassword or [https.admin] mTLS is configured. Set explicitly to override.")
|
||||
a.master = cmdAdmin.Flag.String("master", "localhost:9333", "comma-separated master servers")
|
||||
a.masters = cmdAdmin.Flag.String("masters", "", "comma-separated master servers (deprecated, use -master instead)")
|
||||
a.filerGroup = cmdAdmin.Flag.String("filerGroup", "", "filerGroup for the filers, brokers, and S3 servers")
|
||||
@@ -86,6 +95,7 @@ func init() {
|
||||
a.adminPassword = cmdAdmin.Flag.String("adminPassword", "", "admin interface password (if empty, auth is disabled)")
|
||||
a.readOnlyUser = cmdAdmin.Flag.String("readOnlyUser", "", "read-only user username (optional, for view-only access)")
|
||||
a.readOnlyPassword = cmdAdmin.Flag.String("readOnlyPassword", "", "read-only user password (optional, for view-only access; requires adminPassword to be set)")
|
||||
a.allowInsecureBind = cmdAdmin.Flag.Bool("allowInsecureBind", false, "INSECURE: allow binding a non-loopback ip without adminPassword or mTLS, exposing the admin API unauthenticated on the network")
|
||||
a.icebergPort = cmdAdmin.Flag.Int("iceberg.port", 8181, "Iceberg REST Catalog port (0 to hide in UI)")
|
||||
a.lancePort = cmdAdmin.Flag.Int("lance.port", 9101, "Lance Namespace port (0 to hide in UI)")
|
||||
a.urlPrefix = cmdAdmin.Flag.String("urlPrefix", "", "URL path prefix when running behind a reverse proxy under a subdirectory (e.g. /seaweedfs)")
|
||||
@@ -140,11 +150,15 @@ var cmdAdmin = &Command{
|
||||
- Precedence: CLI flag > env var / security.toml > default value
|
||||
|
||||
Network Binding:
|
||||
- By default the admin server binds to 127.0.0.1 (loopback only).
|
||||
- Use -ip=0.0.0.0 to listen on all interfaces.
|
||||
- When binding to a non-loopback address, authentication MUST be enabled
|
||||
(-adminPassword) or mTLS configured ([https.admin] key and ca in security.toml).
|
||||
Otherwise the server refuses to start.
|
||||
- When authentication is disabled, the admin server binds to 127.0.0.1
|
||||
(loopback only) so the unauthenticated API is never exposed on the network.
|
||||
- When -adminPassword or [https.admin] mTLS is configured, the default
|
||||
upgrades to 0.0.0.0 (all interfaces) so authenticated deployments stay
|
||||
reachable from the network without an explicit -ip flag.
|
||||
- Set -ip explicitly to override either default.
|
||||
- Binding a non-loopback address with authentication disabled (no
|
||||
-adminPassword and no mTLS) is refused unless -allowInsecureBind is set
|
||||
(INSECURE; only for trusted isolated networks).
|
||||
|
||||
Security Configuration:
|
||||
- The admin server reads TLS configuration from security.toml
|
||||
@@ -282,22 +296,49 @@ func runAdmin(cmd *Command, args []string) bool {
|
||||
*a.grpcPort = *a.port + 10000
|
||||
}
|
||||
|
||||
hasMTLS := viper.GetString("https.admin.key") != "" && viper.GetString("https.admin.ca") != ""
|
||||
|
||||
// The worker gRPC control plane has no password auth (its mTLS comes from
|
||||
// grpc.admin, separate from https.admin), so its bind address must not
|
||||
// follow the HTTP auto-upgrade below. Capture the raw -ip value first;
|
||||
// the worker gRPC stays on loopback unless the operator sets -ip
|
||||
// explicitly, matching the pre-existing behavior.
|
||||
a.workerGrpcBindIp = *a.ip
|
||||
|
||||
// -ip defaults to loopback so an unauthenticated admin API is never
|
||||
// exposed on the network by accident. An authenticated deployment
|
||||
// (adminPassword or mTLS) is safe to reach from the network, so upgrade
|
||||
// the default to 0.0.0.0 and keep existing deployments reachable after
|
||||
// upgrade without forcing a -ip=0.0.0.0 config change. An operator who
|
||||
// explicitly set -ip is left alone.
|
||||
if !isFlagExplicitlySet(cmd, "ip") && (*a.adminPassword != "" || hasMTLS) {
|
||||
*a.ip = "0.0.0.0"
|
||||
}
|
||||
|
||||
// Security validation: refuse to bind a non-loopback address without
|
||||
// authentication or mTLS. This prevents accidental exposure of the
|
||||
// unauthenticated admin REST API on the network. Server-only TLS
|
||||
// (https.admin.key without ca) encrypts transport but does not authenticate
|
||||
// clients, so it is not sufficient — the operator must also set a password
|
||||
// or configure mTLS (both key and ca).
|
||||
hasMTLS := viper.GetString("https.admin.key") != "" && viper.GetString("https.admin.ca") != ""
|
||||
// -allowInsecureBind opts out of this check for operators who knowingly
|
||||
// keep the pre-existing unauthenticated setup.
|
||||
insecureAllowed := a.allowInsecureBind != nil && *a.allowInsecureBind
|
||||
if !isLoopbackIp(*a.ip) && *a.adminPassword == "" && !hasMTLS {
|
||||
fmt.Printf("Error: the admin server is configured to bind to %s (non-loopback) with\n", *a.ip)
|
||||
fmt.Printf(" authentication disabled. This would expose the admin API unauthenticated\n")
|
||||
fmt.Printf(" on the network.\n")
|
||||
fmt.Printf(" To fix this, either:\n")
|
||||
fmt.Printf(" - set -adminPassword to enable authentication, or\n")
|
||||
fmt.Printf(" - configure [https.admin] key and ca in security.toml for mTLS, or\n")
|
||||
fmt.Printf(" - set -ip=127.0.0.1 to bind to loopback only.\n")
|
||||
return false
|
||||
if !insecureAllowed {
|
||||
fmt.Printf("Error: the admin server is configured to bind to %s (non-loopback) with\n", *a.ip)
|
||||
fmt.Printf(" authentication disabled. This would expose the admin API unauthenticated\n")
|
||||
fmt.Printf(" on the network.\n")
|
||||
fmt.Printf(" To fix this, either:\n")
|
||||
fmt.Printf(" - set -adminPassword to enable authentication, or\n")
|
||||
fmt.Printf(" - configure [https.admin] key and ca in security.toml for mTLS, or\n")
|
||||
fmt.Printf(" - set -ip=127.0.0.1 to bind to loopback only, or\n")
|
||||
fmt.Printf(" - set -allowInsecureBind to start anyway (INSECURE).\n")
|
||||
return false
|
||||
}
|
||||
fmt.Printf("WARNING: -allowInsecureBind is set: the admin API is exposed on %s without\n", *a.ip)
|
||||
fmt.Printf(" authentication. Anyone who can reach this address has full control\n")
|
||||
fmt.Printf(" of the cluster. Set -adminPassword or configure mTLS instead.\n")
|
||||
}
|
||||
|
||||
// Security warnings
|
||||
@@ -306,6 +347,9 @@ func runAdmin(cmd *Command, args []string) bool {
|
||||
fmt.Println(" Set -adminPassword for production use")
|
||||
}
|
||||
fmt.Printf("Starting SeaweedFS Admin Interface on %s\n", util.JoinHostPort(*a.ip, *a.port))
|
||||
if isLoopbackIp(*a.ip) {
|
||||
fmt.Printf(" (loopback only; not reachable from other hosts. Set -ip=0.0.0.0 with -adminPassword or mTLS to expose.)\n")
|
||||
}
|
||||
fmt.Printf("Worker gRPC server will run on port %d\n", *a.grpcPort)
|
||||
fmt.Printf("Masters: %s\n", *a.master)
|
||||
fmt.Printf("Filers will be discovered automatically from masters\n")
|
||||
@@ -446,11 +490,19 @@ func startAdminServer(ctx context.Context, options AdminOptions, enableUI bool,
|
||||
glog.Infof("No filers discovered from masters")
|
||||
}
|
||||
|
||||
// Start worker gRPC server for worker connections
|
||||
err = adminServer.StartWorkerGrpcServer(*options.grpcPort, options.workerGrpcListener)
|
||||
// Start worker gRPC server for worker connections. The worker gRPC binds
|
||||
// to its own address (workerGrpcBindIp) which, unlike the HTTP ip, does
|
||||
// not auto-upgrade to 0.0.0.0 based on adminPassword, since the worker
|
||||
// gRPC has no password auth.
|
||||
workerGrpcIp := options.workerGrpcBindIp
|
||||
if workerGrpcIp == "" {
|
||||
workerGrpcIp = *options.ip
|
||||
}
|
||||
err = adminServer.StartWorkerGrpcServer(workerGrpcIp, *options.grpcPort, options.workerGrpcListener)
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to start worker gRPC server: %w", err)
|
||||
}
|
||||
warnInsecureWorkerGrpcBind(workerGrpcIp, *options.grpcPort, adminServer.WorkerGrpcMTLSEnabled())
|
||||
|
||||
// Set up cleanup for gRPC server
|
||||
defer func() {
|
||||
@@ -714,19 +766,26 @@ func loadOrGenerateSessionKeys(dataDir string) ([]byte, []byte, error) {
|
||||
return key[:keyLen], key[keyLen:], nil
|
||||
}
|
||||
|
||||
// isFlagExplicitlySet reports whether the named flag was passed on the
|
||||
// command line (as opposed to left at its default).
|
||||
func isFlagExplicitlySet(cmd *Command, flagName string) bool {
|
||||
set := false
|
||||
cmd.Flag.Visit(func(f *flag.Flag) {
|
||||
if f.Name == flagName {
|
||||
set = true
|
||||
}
|
||||
})
|
||||
return set
|
||||
}
|
||||
|
||||
// applyViperFallback sets a flag's value from viper (security.toml / env var)
|
||||
// when the flag was not explicitly set on the command line.
|
||||
func applyViperFallback(cmd *Command, flagPtr *string, flagName, viperKey string) {
|
||||
explicitlySet := false
|
||||
cmd.Flag.Visit(func(f *flag.Flag) {
|
||||
if f.Name == flagName {
|
||||
explicitlySet = true
|
||||
}
|
||||
})
|
||||
if !explicitlySet {
|
||||
if v := util.GetViper().GetString(viperKey); v != "" {
|
||||
*flagPtr = v
|
||||
}
|
||||
if isFlagExplicitlySet(cmd, flagName) {
|
||||
return
|
||||
}
|
||||
if v := util.GetViper().GetString(viperKey); v != "" {
|
||||
*flagPtr = v
|
||||
}
|
||||
}
|
||||
|
||||
@@ -744,3 +803,19 @@ func isLoopbackIp(ip string) bool {
|
||||
}
|
||||
return parsed.IsLoopback()
|
||||
}
|
||||
|
||||
// warnInsecureWorkerGrpcBind warns when the worker gRPC control plane is
|
||||
// reachable off loopback without grpc.admin mTLS, its only auth once exposed.
|
||||
// mtlsEnabled reflects whether the worker gRPC actually loaded mTLS, so a
|
||||
// misconfigured cert/key that fails to load still triggers the warning.
|
||||
func warnInsecureWorkerGrpcBind(ip string, grpcPort int, mtlsEnabled bool) {
|
||||
if isLoopbackIp(ip) {
|
||||
return
|
||||
}
|
||||
if mtlsEnabled {
|
||||
return
|
||||
}
|
||||
glog.Warningf("Worker gRPC control plane is bound to %s (non-loopback) without grpc.admin mTLS.", ip)
|
||||
glog.Warningf("Anyone who can reach port %d can register a maintenance worker unauthenticated.", grpcPort)
|
||||
glog.Warningf("Enable [grpc.admin] cert/key and [grpc.ca] in security.toml, or bind to loopback.")
|
||||
}
|
||||
|
||||
@@ -380,6 +380,7 @@ func (fo *FilerOptions) startFiler() {
|
||||
|
||||
fs, nfs_err := weed_server.NewFilerServer(defaultMux, publicVolumeMux, &weed_server.FilerOption{
|
||||
Masters: fo.masters,
|
||||
MetricsPort: uint32(*fo.metricsHttpPort),
|
||||
FilerGroup: *fo.filerGroup,
|
||||
Collection: *fo.collection,
|
||||
DefaultReplication: *fo.defaultReplicaPlacement,
|
||||
|
||||
@@ -213,54 +213,7 @@ func (option *RemoteSyncOptions) makeEventProcessor(remoteStorage *remote_pb.Rem
|
||||
return client.DeleteFile(dest)
|
||||
}
|
||||
if message.OldEntry != nil && message.NewEntry != nil {
|
||||
if isMultipartUploadFile(message.NewParentPath, message.NewEntry.Name) {
|
||||
return nil
|
||||
}
|
||||
// Skip updates to internal version paths
|
||||
if isVersionedPath(message.NewParentPath, message.NewEntry.Name, message.NewEntry.IsDirectory) {
|
||||
glog.V(2).Infof("skipping update of internal version path: %s/%s", message.NewParentPath, message.NewEntry.Name)
|
||||
return nil
|
||||
}
|
||||
oldDest := toRemoteStorageLocation(util.FullPath(mountedDir), util.NewFullPath(resp.Directory, message.OldEntry.Name), remoteStorageMountLocation)
|
||||
dest := toRemoteStorageLocation(util.FullPath(mountedDir), util.NewFullPath(message.NewParentPath, message.NewEntry.Name), remoteStorageMountLocation)
|
||||
if !shouldSendToRemote(message.NewEntry) {
|
||||
glog.V(2).Infof("skipping updating: %+v", resp)
|
||||
return nil
|
||||
}
|
||||
if message.NewEntry.IsDirectory {
|
||||
return client.WriteDirectory(dest, message.NewEntry)
|
||||
}
|
||||
if isMetadataOnlyUpdate(resp.Directory, message) {
|
||||
remoteEntry, err := liveRemoteEntry(option, message.NewParentPath, message.NewEntry)
|
||||
if errors.Is(err, filer_pb.ErrNotFound) {
|
||||
glog.V(2).Infof("skipping updating deleted entry: %+v", resp)
|
||||
return nil
|
||||
}
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if remoteEntry != nil {
|
||||
glog.V(2).Infof("update meta: %+v", resp)
|
||||
return client.UpdateFileMetadata(dest, message.OldEntry, message.NewEntry)
|
||||
}
|
||||
glog.V(0).Infof("never replicated, uploading %s", remote_storage.FormatLocation(dest))
|
||||
}
|
||||
glog.V(2).Infof("update: %+v", resp)
|
||||
if !proto.Equal(oldDest, dest) {
|
||||
glog.V(0).Infof("delete %s", remote_storage.FormatLocation(oldDest))
|
||||
if err := client.DeleteFile(oldDest); err != nil && isMultipartUploadFile(resp.Directory, message.OldEntry.Name) {
|
||||
return nil
|
||||
}
|
||||
}
|
||||
remoteEntry, writeErr := retriedWriteFile(client, filerSource, message.NewParentPath, message.NewEntry, dest)
|
||||
if errors.Is(writeErr, errSuperseded) {
|
||||
glog.Errorf("skipping %s: %v", remote_storage.FormatLocation(dest), writeErr)
|
||||
return nil
|
||||
}
|
||||
if writeErr != nil {
|
||||
return writeErr
|
||||
}
|
||||
return updateLocalEntry(option, message.NewParentPath, message.NewEntry, remoteEntry)
|
||||
return processUpdateEvent(option, filerSource, client, mountedDir, remoteStorageMountLocation, resp)
|
||||
}
|
||||
|
||||
return nil
|
||||
@@ -268,6 +221,73 @@ func (option *RemoteSyncOptions) makeEventProcessor(remoteStorage *remote_pb.Rem
|
||||
return eachEntryFunc, nil
|
||||
}
|
||||
|
||||
func processUpdateEvent(
|
||||
filerClient filer_pb.FilerClient,
|
||||
filerSource filer_pb.FilerClient,
|
||||
client remote_storage.RemoteStorageClient,
|
||||
mountedDir string,
|
||||
remoteStorageMountLocation *remote_pb.RemoteStorageLocation,
|
||||
resp *filer_pb.SubscribeMetadataResponse,
|
||||
) error {
|
||||
message := resp.EventNotification
|
||||
if isMultipartUploadFile(message.NewParentPath, message.NewEntry.Name) {
|
||||
return nil
|
||||
}
|
||||
if isVersionedPath(message.NewParentPath, message.NewEntry.Name, message.NewEntry.IsDirectory) {
|
||||
glog.V(2).Infof("skipping update of internal version path: %s/%s", message.NewParentPath, message.NewEntry.Name)
|
||||
return nil
|
||||
}
|
||||
oldDest := toRemoteStorageLocation(util.FullPath(mountedDir), util.NewFullPath(resp.Directory, message.OldEntry.Name), remoteStorageMountLocation)
|
||||
dest := toRemoteStorageLocation(util.FullPath(mountedDir), util.NewFullPath(message.NewParentPath, message.NewEntry.Name), remoteStorageMountLocation)
|
||||
if proto.Equal(oldDest, dest) && !shouldSendToRemote(message.NewEntry) {
|
||||
glog.V(2).Infof("skipping updating: %+v", resp)
|
||||
return nil
|
||||
}
|
||||
if message.NewEntry.IsDirectory {
|
||||
return client.WriteDirectory(dest, message.NewEntry)
|
||||
}
|
||||
if isMetadataOnlyUpdate(resp.Directory, message) {
|
||||
remoteEntry, err := liveRemoteEntry(filerClient, message.NewParentPath, message.NewEntry)
|
||||
if errors.Is(err, filer_pb.ErrNotFound) {
|
||||
glog.V(2).Infof("skipping updating deleted entry: %+v", resp)
|
||||
return nil
|
||||
}
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if remoteEntry != nil {
|
||||
glog.V(2).Infof("update meta: %+v", resp)
|
||||
return client.UpdateFileMetadata(dest, message.OldEntry, message.NewEntry)
|
||||
}
|
||||
glog.V(0).Infof("never replicated, uploading %s", remote_storage.FormatLocation(dest))
|
||||
}
|
||||
if !proto.Equal(oldDest, dest) && !filer.HasData(message.NewEntry) && message.NewEntry.IsInRemoteOnly() {
|
||||
glog.V(0).Infof("skip uploading renamed remote-only entry %s: content is only on the deleted remote object", remote_storage.FormatLocation(dest))
|
||||
return nil
|
||||
}
|
||||
glog.V(2).Infof("update: %+v", resp)
|
||||
if !proto.Equal(oldDest, dest) {
|
||||
glog.V(0).Infof("delete %s", remote_storage.FormatLocation(oldDest))
|
||||
if err := client.DeleteFile(oldDest); err != nil {
|
||||
if isMultipartUploadFile(resp.Directory, message.OldEntry.Name) {
|
||||
return nil
|
||||
}
|
||||
if !errors.Is(err, remote_storage.ErrRemoteObjectNotFound) {
|
||||
return err
|
||||
}
|
||||
}
|
||||
}
|
||||
remoteEntry, writeErr := retriedWriteFile(client, filerSource, message.NewParentPath, message.NewEntry, dest)
|
||||
if errors.Is(writeErr, errSuperseded) {
|
||||
glog.Errorf("skipping %s: %v", remote_storage.FormatLocation(dest), writeErr)
|
||||
return nil
|
||||
}
|
||||
if writeErr != nil {
|
||||
return writeErr
|
||||
}
|
||||
return updateLocalEntry(filerClient, message.NewParentPath, message.NewEntry, remoteEntry)
|
||||
}
|
||||
|
||||
// isSuperseded reports whether the filer has moved past the entry an event
|
||||
// described: it is deleted, or it no longer references every chunk the event
|
||||
// named. Those are the chunks the filer deletes when an entry is updated, so
|
||||
|
||||
@@ -16,6 +16,7 @@ import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api/s3_constants"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
"google.golang.org/grpc"
|
||||
"google.golang.org/protobuf/proto"
|
||||
)
|
||||
|
||||
// TestVersionedFilePathRewrittenForRemote verifies that the fix for
|
||||
@@ -425,6 +426,10 @@ func (c *stubFilerClient) LookupDirectoryEntry(context.Context, *filer_pb.Lookup
|
||||
return &filer_pb.LookupDirectoryEntryResponse{Entry: c.entry}, nil
|
||||
}
|
||||
|
||||
func (c *stubFilerClient) UpdateEntry(context.Context, *filer_pb.UpdateEntryRequest, ...grpc.CallOption) (*filer_pb.UpdateEntryResponse, error) {
|
||||
return &filer_pb.UpdateEntryResponse{}, nil
|
||||
}
|
||||
|
||||
func (c *stubFilerClient) WithFilerClient(_ bool, fn func(filer_pb.SeaweedFilerClient) error) error {
|
||||
return fn(c)
|
||||
}
|
||||
@@ -629,3 +634,177 @@ func TestRetriedWriteFileStopsWhenSuperseded(t *testing.T) {
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
type recordingRemote struct {
|
||||
remote_storage.RemoteStorageClient
|
||||
deletes []*remote_pb.RemoteStorageLocation
|
||||
writes []*remote_pb.RemoteStorageLocation
|
||||
deleteErr error
|
||||
}
|
||||
|
||||
func (r *recordingRemote) WriteFile(loc *remote_pb.RemoteStorageLocation, entry *filer_pb.Entry, _ io.Reader) (*filer_pb.RemoteEntry, error) {
|
||||
r.writes = append(r.writes, loc)
|
||||
return &filer_pb.RemoteEntry{StorageName: loc.Name, RemoteETag: "etag", RemoteSize: int64(len(entry.Content)), RemoteMtime: entry.Attributes.GetMtime()}, nil
|
||||
}
|
||||
|
||||
func (r *recordingRemote) DeleteFile(loc *remote_pb.RemoteStorageLocation) error {
|
||||
r.deletes = append(r.deletes, loc)
|
||||
return r.deleteErr
|
||||
}
|
||||
|
||||
// TestRenameWithInheritedRemoteEntryWritesNewKey reproduces #11261: a rename
|
||||
// arrives as an update whose NewEntry carries the RemoteEntry inherited from
|
||||
// the source. shouldSendToRemote returns false for it (RemoteMtime >= Mtime),
|
||||
// so the old code skipped the event and never wrote the new key, while the
|
||||
// filer had already deleted the old object. A path change must always write.
|
||||
func TestRenameWithInheritedRemoteEntryWritesNewKey(t *testing.T) {
|
||||
const mountedDir = "/buckets"
|
||||
mountLoc := &remote_pb.RemoteStorageLocation{Name: "b2", Bucket: "bucket", Path: "/"}
|
||||
|
||||
replicated := &filer_pb.RemoteEntry{StorageName: "b2", RemoteETag: "abc", RemoteSize: 2048, RemoteMtime: 1786096669}
|
||||
oldEntry := &filer_pb.Entry{Name: "probe.bin", Attributes: &filer_pb.FuseAttributes{Mtime: 1786096669}, RemoteEntry: replicated}
|
||||
newEntry := &filer_pb.Entry{
|
||||
Name: "probe.bin",
|
||||
Content: []byte("payload"),
|
||||
Attributes: &filer_pb.FuseAttributes{Mtime: 1786096669},
|
||||
RemoteEntry: replicated,
|
||||
}
|
||||
resp := &filer_pb.SubscribeMetadataResponse{
|
||||
Directory: "/buckets/b/src",
|
||||
EventNotification: &filer_pb.EventNotification{
|
||||
OldEntry: oldEntry,
|
||||
NewParentPath: "/buckets/b/dst",
|
||||
NewEntry: newEntry,
|
||||
},
|
||||
}
|
||||
|
||||
if shouldSendToRemote(newEntry) {
|
||||
t.Fatal("precondition: inherited RemoteEntry should make shouldSendToRemote false, the bug's trigger")
|
||||
}
|
||||
|
||||
remote := &recordingRemote{}
|
||||
filerClient := &stubFilerClient{}
|
||||
if err := processUpdateEvent(filerClient, filerClient, remote, mountedDir, mountLoc, resp); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
wantDelete := &remote_pb.RemoteStorageLocation{Name: "b2", Bucket: "bucket", Path: "/b/src/probe.bin"}
|
||||
if len(remote.deletes) != 1 || !proto.Equal(remote.deletes[0], wantDelete) {
|
||||
t.Errorf("deletes = %+v, want the old key %s deleted", remote.deletes, remote_storage.FormatLocation(wantDelete))
|
||||
}
|
||||
wantWrite := &remote_pb.RemoteStorageLocation{Name: "b2", Bucket: "bucket", Path: "/b/dst/probe.bin"}
|
||||
if len(remote.writes) != 1 || !proto.Equal(remote.writes[0], wantWrite) {
|
||||
t.Errorf("writes = %+v, want the new key %s written", remote.writes, remote_storage.FormatLocation(wantWrite))
|
||||
}
|
||||
}
|
||||
|
||||
// TestRenameRemoteOnlyEntrySkipsEmptyUpload guards the edge case from the
|
||||
// review of #11261: a remote-only entry (no local chunks, content lives only
|
||||
// on the remote object the filer already deleted) must not be re-uploaded,
|
||||
// since NewFileReader would supply EOF and create a zero-byte object.
|
||||
func TestRenameRemoteOnlyEntrySkipsEmptyUpload(t *testing.T) {
|
||||
const mountedDir = "/buckets"
|
||||
mountLoc := &remote_pb.RemoteStorageLocation{Name: "b2", Bucket: "bucket", Path: "/"}
|
||||
|
||||
remoteOnly := &filer_pb.RemoteEntry{StorageName: "b2", RemoteETag: "abc", RemoteSize: 20971520, RemoteMtime: 1786096669}
|
||||
oldEntry := &filer_pb.Entry{Name: "video.mp4", Attributes: &filer_pb.FuseAttributes{Mtime: 1786096669}, RemoteEntry: remoteOnly}
|
||||
newEntry := &filer_pb.Entry{
|
||||
Name: "video.mp4",
|
||||
Attributes: &filer_pb.FuseAttributes{Mtime: 1786096669},
|
||||
RemoteEntry: remoteOnly,
|
||||
}
|
||||
if !newEntry.IsInRemoteOnly() {
|
||||
t.Fatal("precondition: entry should be remote-only")
|
||||
}
|
||||
if filer.HasData(newEntry) {
|
||||
t.Fatal("precondition: remote-only entry should have no local data")
|
||||
}
|
||||
|
||||
resp := &filer_pb.SubscribeMetadataResponse{
|
||||
Directory: "/buckets/b/src",
|
||||
EventNotification: &filer_pb.EventNotification{
|
||||
OldEntry: oldEntry,
|
||||
NewParentPath: "/buckets/b/dst",
|
||||
NewEntry: newEntry,
|
||||
},
|
||||
}
|
||||
|
||||
remote := &recordingRemote{}
|
||||
filerClient := &stubFilerClient{}
|
||||
if err := processUpdateEvent(filerClient, filerClient, remote, mountedDir, mountLoc, resp); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(remote.writes) != 0 {
|
||||
t.Errorf("writes = %+v, want none: a remote-only rename must not upload an empty object", remote.writes)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRenameDeleteOldKeyFailureReturnsError checks that a failed delete of the
|
||||
// old key on a rename is returned so MetadataProcessor retries the event,
|
||||
// instead of silently continuing and leaving both keys on the remote.
|
||||
func TestRenameDeleteOldKeyFailureReturnsError(t *testing.T) {
|
||||
const mountedDir = "/buckets"
|
||||
mountLoc := &remote_pb.RemoteStorageLocation{Name: "b2", Bucket: "bucket", Path: "/"}
|
||||
|
||||
newEntry := &filer_pb.Entry{
|
||||
Name: "probe.bin",
|
||||
Content: []byte("payload"),
|
||||
Attributes: &filer_pb.FuseAttributes{Mtime: 1786096669},
|
||||
}
|
||||
oldEntry := &filer_pb.Entry{Name: "probe.bin", Attributes: &filer_pb.FuseAttributes{Mtime: 1786096669}}
|
||||
resp := &filer_pb.SubscribeMetadataResponse{
|
||||
Directory: "/buckets/b/src",
|
||||
EventNotification: &filer_pb.EventNotification{
|
||||
OldEntry: oldEntry,
|
||||
NewParentPath: "/buckets/b/dst",
|
||||
NewEntry: newEntry,
|
||||
},
|
||||
}
|
||||
|
||||
deleteErr := errors.New("AccessDenied: Access Denied")
|
||||
remote := &recordingRemote{deleteErr: deleteErr}
|
||||
filerClient := &stubFilerClient{}
|
||||
err := processUpdateEvent(filerClient, filerClient, remote, mountedDir, mountLoc, resp)
|
||||
if !errors.Is(err, deleteErr) {
|
||||
t.Errorf("err = %v, want the delete failure returned so the event is retried", err)
|
||||
}
|
||||
if len(remote.writes) != 0 {
|
||||
t.Errorf("writes = %+v, want none: must not write the new key when the old key delete failed", remote.writes)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRenameDeleteOldKeyNotFoundStillWrites checks that an already-deleted old
|
||||
// key (the filer deletes the source object synchronously during rename) does
|
||||
// not block the destination write. GCS reports a missing object as
|
||||
// ErrRemoteObjectNotFound, unlike S3/Azure whose deletes are idempotent, so
|
||||
// treating it as a real error would pin the sync offset and never write the
|
||||
// new key.
|
||||
func TestRenameDeleteOldKeyNotFoundStillWrites(t *testing.T) {
|
||||
const mountedDir = "/buckets"
|
||||
mountLoc := &remote_pb.RemoteStorageLocation{Name: "gcs", Bucket: "bucket", Path: "/"}
|
||||
|
||||
newEntry := &filer_pb.Entry{
|
||||
Name: "probe.bin",
|
||||
Content: []byte("payload"),
|
||||
Attributes: &filer_pb.FuseAttributes{Mtime: 1786096669},
|
||||
}
|
||||
oldEntry := &filer_pb.Entry{Name: "probe.bin", Attributes: &filer_pb.FuseAttributes{Mtime: 1786096669}}
|
||||
resp := &filer_pb.SubscribeMetadataResponse{
|
||||
Directory: "/buckets/b/src",
|
||||
EventNotification: &filer_pb.EventNotification{
|
||||
OldEntry: oldEntry,
|
||||
NewParentPath: "/buckets/b/dst",
|
||||
NewEntry: newEntry,
|
||||
},
|
||||
}
|
||||
|
||||
remote := &recordingRemote{deleteErr: remote_storage.ErrRemoteObjectNotFound}
|
||||
filerClient := &stubFilerClient{}
|
||||
if err := processUpdateEvent(filerClient, filerClient, remote, mountedDir, mountLoc, resp); err != nil {
|
||||
t.Fatalf("err = %v, want nil: an already-deleted old key must not block the write", err)
|
||||
}
|
||||
wantWrite := &remote_pb.RemoteStorageLocation{Name: "gcs", Bucket: "bucket", Path: "/b/dst/probe.bin"}
|
||||
if len(remote.writes) != 1 || !proto.Equal(remote.writes[0], wantWrite) {
|
||||
t.Errorf("writes = %+v, want the new key %s written", remote.writes, remote_storage.FormatLocation(wantWrite))
|
||||
}
|
||||
}
|
||||
|
||||
@@ -465,6 +465,12 @@ func peerIndex(self pb.ServerAddress, peers []pb.ServerAddress) int {
|
||||
}
|
||||
|
||||
func (m *MasterOptions) toMasterOption(whiteList []string) *weed_server.MasterOption {
|
||||
// Not every caller wires up every flag, so treat an unset metrics port as
|
||||
// disabled rather than dereferencing nil.
|
||||
metricsPort := 0
|
||||
if m.metricsHttpPort != nil {
|
||||
metricsPort = *m.metricsHttpPort
|
||||
}
|
||||
masterAddress := pb.NewServerAddress(*m.ip, *m.port, *m.portGrpc)
|
||||
return &weed_server.MasterOption{
|
||||
Master: masterAddress,
|
||||
@@ -479,6 +485,7 @@ func (m *MasterOptions) toMasterOption(whiteList []string) *weed_server.MasterOp
|
||||
WhiteList: whiteList,
|
||||
DisableHttp: *m.disableHttp,
|
||||
MetricsAddress: *m.metricsAddress,
|
||||
MetricsPort: metricsPort,
|
||||
MetricsIntervalSec: *m.metricsIntervalSec,
|
||||
TelemetryUrl: *m.telemetryUrl,
|
||||
TelemetryEnabled: *m.telemetryEnabled,
|
||||
|
||||
@@ -953,7 +953,7 @@ func ensureAllPortsAvailableOnIP(bindIp string) error {
|
||||
// first: an in-process rerun would otherwise inherit the closed listener
|
||||
// of the previous run and only find out inside Serve.
|
||||
miniAdminOptions.workerGrpcListener = nil
|
||||
if listener, err := net.Listen("tcp", fmt.Sprintf(":%d", *miniAdminOptions.grpcPort)); err != nil {
|
||||
if listener, err := net.Listen("tcp", util.JoinHostPort(bindIp, *miniAdminOptions.grpcPort)); err != nil {
|
||||
glog.Warningf("Could not reserve Admin gRPC port %d: %v", *miniAdminOptions.grpcPort, err)
|
||||
} else {
|
||||
miniAdminOptions.workerGrpcListener = listener
|
||||
@@ -1308,6 +1308,13 @@ func runMini(cmd *Command, args []string) bool {
|
||||
}
|
||||
pb.RegisterLocalGrpcSocket(*miniIp, *miniAdminOptions.grpcPort, fmt.Sprintf("/tmp/seaweedfs-admin-grpc-%d.sock", *miniAdminOptions.grpcPort))
|
||||
|
||||
// One process, one shared Prometheus registry, so this single listener
|
||||
// serves every component's series. Point each component at it so they all
|
||||
// advertise the same port to the master.
|
||||
miniMasterOptions.metricsHttpPort = miniMetricsHttpPort
|
||||
miniOptions.v.metricsHttpPort = miniMetricsHttpPort
|
||||
miniFilerOptions.metricsHttpPort = miniMetricsHttpPort
|
||||
|
||||
go stats_collect.StartMetricsServer(*miniMetricsHttpIp, *miniMetricsHttpPort)
|
||||
|
||||
if *miniMasterOptions.volumeSizeLimitMB > util.MaxVolumeSizeLimitMB {
|
||||
|
||||
@@ -20,6 +20,8 @@ type MountOptions struct {
|
||||
chunkSizeLimitMB *int
|
||||
concurrentWriters *int
|
||||
concurrentReaders *int
|
||||
readerCacheSizeMB *int64
|
||||
memoryLimitMB *int64
|
||||
cacheMetaTtlSec *int
|
||||
cacheDirMaxEntries *int
|
||||
cacheDirForRead *string
|
||||
@@ -104,6 +106,8 @@ func init() {
|
||||
mountOptions.chunkSizeLimitMB = cmdMount.Flag.Int("chunkSizeLimitMB", 2, "local write buffer size, also chunk large files")
|
||||
mountOptions.concurrentWriters = cmdMount.Flag.Int("concurrentWriters", 128, "limit concurrent goroutine writers")
|
||||
mountOptions.concurrentReaders = cmdMount.Flag.Int("concurrentReaders", 128, "limit concurrent chunk fetches for read operations")
|
||||
mountOptions.memoryLimitMB = cmdMount.Flag.Int64("memoryLimitMB", 0, "soft Go runtime memory limit in MiB; 0 preserves GOMEMLIMIT; leave headroom below the container limit")
|
||||
mountOptions.readerCacheSizeMB = cmdMount.Flag.Int64("readerCacheSizeMB", 256, "memory budget in MiB for downloaded and in-flight reader buffers across all files; must fit the largest pooled chunk buffer")
|
||||
mountOptions.cacheDirForRead = cmdMount.Flag.String("cacheDir", os.TempDir(), "local cache directory for file chunks and meta data")
|
||||
mountOptions.cacheSizeMBForRead = cmdMount.Flag.Int64("cacheCapacityMB", 128, "file chunk read cache capacity in MB")
|
||||
mountOptions.cacheDirForWrite = cmdMount.Flag.String("cacheDirWrite", "", "buffer writes mostly for large files")
|
||||
|
||||
@@ -5,11 +5,13 @@ package command
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"math"
|
||||
"net"
|
||||
"net/http"
|
||||
"os"
|
||||
"path"
|
||||
"runtime"
|
||||
"runtime/debug"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
@@ -184,6 +186,10 @@ type fileSystemParams struct {
|
||||
}
|
||||
|
||||
func buildSeaweedFileSystem(option *MountOptions, p fileSystemParams) *mount.WFS {
|
||||
readerCacheSizeMB := int64(256)
|
||||
if option.readerCacheSizeMB != nil {
|
||||
readerCacheSizeMB = *option.readerCacheSizeMB
|
||||
}
|
||||
return mount.NewSeaweedFileSystem(&mount.Option{
|
||||
MountDirectory: p.dir,
|
||||
FilerAddresses: p.filerAddresses,
|
||||
@@ -198,6 +204,7 @@ func buildSeaweedFileSystem(option *MountOptions, p fileSystemParams) *mount.WFS
|
||||
ChunkSizeLimit: int64(p.chunkSizeLimitMB) * 1024 * 1024,
|
||||
ConcurrentWriters: *option.concurrentWriters,
|
||||
ConcurrentReaders: *option.concurrentReaders,
|
||||
ReaderCacheSizeMB: readerCacheSizeMB,
|
||||
CacheDirForRead: p.cacheDirForRead,
|
||||
CacheSizeMBForRead: *option.cacheSizeMBForRead,
|
||||
CacheDirForWrite: p.cacheDirForWrite,
|
||||
@@ -304,3 +311,18 @@ func lastSegment(p string) string {
|
||||
}
|
||||
return name
|
||||
}
|
||||
|
||||
func configureMountMemory(option *MountOptions) error {
|
||||
if option.readerCacheSizeMB != nil && (*option.readerCacheSizeMB <= 0 || *option.readerCacheSizeMB > math.MaxInt64>>20) {
|
||||
return fmt.Errorf("readerCacheSizeMB must be positive and fit in an int64 byte budget")
|
||||
}
|
||||
if option.memoryLimitMB != nil {
|
||||
if *option.memoryLimitMB < 0 || *option.memoryLimitMB > math.MaxInt64>>20 {
|
||||
return fmt.Errorf("memoryLimitMB must be non-negative and fit in an int64 byte limit")
|
||||
}
|
||||
if *option.memoryLimitMB > 0 {
|
||||
debug.SetMemoryLimit(*option.memoryLimitMB << 20)
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -0,0 +1,39 @@
|
||||
//go:build linux || darwin || freebsd || windows
|
||||
|
||||
package command
|
||||
|
||||
import (
|
||||
"math"
|
||||
"runtime/debug"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestConfigureMountMemory(t *testing.T) {
|
||||
for _, size := range []int64{-1, 0, 1, 256, math.MaxInt64 >> 20, math.MaxInt64} {
|
||||
err := configureMountMemory(&MountOptions{readerCacheSizeMB: &size})
|
||||
valid := size > 0 && size <= math.MaxInt64>>20
|
||||
if (err == nil) != valid {
|
||||
t.Errorf("size=%d: err=%v", size, err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestConfigureMountMemoryRuntimeLimit(t *testing.T) {
|
||||
previous := debug.SetMemoryLimit(512 << 20)
|
||||
defer debug.SetMemoryLimit(previous)
|
||||
for _, size := range []int64{0, -1, math.MaxInt64, 768, 0} {
|
||||
before := debug.SetMemoryLimit(-1)
|
||||
err := configureMountMemory(&MountOptions{memoryLimitMB: &size})
|
||||
valid := size >= 0 && size <= math.MaxInt64>>20
|
||||
if (err == nil) != valid {
|
||||
t.Errorf("size=%d: err=%v", size, err)
|
||||
}
|
||||
want := before
|
||||
if valid && size > 0 {
|
||||
want = size << 20
|
||||
}
|
||||
if got := debug.SetMemoryLimit(-1); got != want {
|
||||
t.Errorf("size=%d: runtime limit=%d, want %d", size, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user