Compare commits

..
Author SHA1 Message Date
Chris Lu a6e07181b2 admin: scrape the metrics ports the cluster advertises
Replaces scraping each node's service port with the dedicated Prometheus
listener each node now advertises. Nodes started without -metricsPort
advertise 0 and are skipped, so nothing is ever fetched from a
client-facing port.

Endpoints are deduplicated by address because a combined "weed server"
advertises one listener for all of its components; series are attributed
by metric name, which already identifies the component.
2026-09-14 23:59:38 -07:00
Chris Lu 51831d6850 admin: derive interval rates and latency quantiles from scrapes
Raw counters and histograms are process-lifetime cumulative, so
charting them directly is meaningless. Counters now become per-second
rates and histograms become p50/p95/p99, both computed from the delta
against the previous scrape, with counter resets skipped. Quantiles use
linear interpolation within the matching bucket, as Prometheus does.
2026-09-14 23:59:38 -07:00
Chris Lu d235dd280b admin: gather the admin's own registry into the metrics store
The admin's maintenance and worker metrics live in the local
stats.Gather registry, so record them directly under the admin/local
source instead of scraping over HTTP. Adds metricsStore.match for
prefix/metric lookups.
2026-09-14 23:59:38 -07:00
Chris Lu 7b9332953d admin: add server-side SVG chart renderer
Extends the existing sparklineSVG approach into a full chart helper
with axes, gridlines, multi-series lines/areas, legends, threshold
lines, and unit formatters (bytes, bps, ms, pct). No JS chart library;
safe to inline in templ pages.
2026-09-14 23:59:38 -07:00
Chris Lu 85147522a9 admin: scrape per-server /metrics into the store
Adds a 15s scrape loop that fetches /metrics from every discovered
master, volume, filer, and S3 server (addresses come from the
existing topology + ListClusterNodes helpers) and records each
series into the in-memory store. Parses Prometheus text exposition
via prometheus/common/expfmt.
2026-09-14 23:59:38 -07:00
Chris Lu 207bc0b75a admin: add in-memory metrics series store
Bounded ring buffer (240 samples) keyed by source/metric[/labels],
reusing the dashSample ring pattern. No persistence; powers the
upcoming monitoring charts.
2026-09-14 23:59:38 -07:00
Chris Lu 2f26d5779b mini: wire the shared metrics port, and tolerate an unset one
weed mini builds MasterOptions directly and never set metricsHttpPort, so
reading it in toMasterOption dereferenced nil and crashed startup. Point
mini's master, volume and filer at its single -metricsPort listener, the
same way weed server does, and treat an unset port as disabled so a
partially initialised MasterOptions cannot panic again.

Reproduced with 'weed mini -dir=... -s3.port=...', which is what the S3
filer-group and delete-regression suites start.
2026-09-14 23:59:32 -07:00
Chris Lu 14bc6e5e4f servers: advertise the configured metricsPort to the master
Filer, S3 and broker report it on the KeepConnected registration, volume
servers in their heartbeat, and masters return their own in
GetMasterConfiguration. MasterClient gains SetMetricsPort so the eight
callers that have no metrics listener are untouched, and the volume
server takes it as a constructor argument because its heartbeat goroutine
starts there.

In combined "weed server" one metrics listener serves the whole shared
registry, so every component advertises the same port.
2026-09-14 23:32:32 -07:00
Chris Lu dfde24f3ee pb: add metrics_port so nodes can advertise their metrics listener
Each server already has a -metricsPort Prometheus listener, but nothing
advertises it, so a central scraper cannot find it. Adds metrics_port to
KeepConnectedRequest and ListClusterNodes (filer, S3, broker), Heartbeat
and DataNodeInfo (volume servers), and GetMasterConfigurationResponse
(masters).

Note the existing metrics_address fields are unrelated: they carry the
Prometheus push gateway that servers push to, not a scrape target.
2026-09-14 23:23:25 -07:00
709 changed files with 13398 additions and 68944 deletions
-240
View File
@@ -1,240 +0,0 @@
#!/usr/bin/env python3
"""Check JWT extraction and, with --context, real Helm upgrade persistence."""
import argparse
import base64
import json
from pathlib import Path
import shutil
import subprocess
import sys
import tempfile
import uuid
try:
import tomllib
except ModuleNotFoundError: # CI also exercises Python 3.10.
import tomli as tomllib
ROOT = Path(__file__).resolve().parents[2]
CHART = ROOT / "k8s/charts/seaweedfs"
SECTIONS = ("jwt.signing", "jwt.signing.read", "jwt.filer_signing", "jwt.filer_signing.read")
KEYS = {section: f"active-{index}" for index, section in enumerate(SECTIONS)}
ESCAPED_HEADER = '["jw\\u0074".signing]\nkey = "existing"'
def run(*args):
return subprocess.run(args, check=True, text=True, capture_output=True).stdout
def keys(raw):
document = tomllib.loads(raw)
result = {}
for section in SECTIONS:
table = document
for part in section.split("."):
table = table.get(part, {})
if "key" in table:
result[section] = table["key"]
return result
def fixtures():
canonical = "\n".join(f'[{section}]\nkey = "{value}"' for section, value in KEYS.items())
yield "canonical / final line without newline", canonical
yield "commented stale keys", canonical.replace("key =", '# key = "stale"\nkey =')
yield "commented headers / mixed line endings", "\n".join(
f'# [{section}]\r\n# key = "stale"\r\n[{section}]\nkey = "{value}"'
for section, value in KEYS.items()
)
yield "indented headers and keys / trailing comments", "\n".join(
f' \t[ {section} ] \t# header [note]\n \tkey \t= \t"{value}" # key comment'
for section, value in KEYS.items()
)
yield "CRLF", canonical.replace("\n", "\r\n")
yield "quoted and spaced section names", "\n".join(
f'["{section.split(".")[0]}" . \'{section.split(".")[1]}\''
+ (f' . "{section.split(".")[2]}"' if section.count(".") == 2 else "")
+ f'] # original table\nkey = "{value}"'
for section, value in KEYS.items()
)
yield "unrelated quoted header before JWT keys", '["custom section"]\nkey = "other"\n' + canonical
yield "brackets in comments", canonical.replace("key =", "# consult [notes]\nkey =")
yield "quoted values and quoted key names", "\n".join((
'[jwt.signing]\n"key" = "brackets[inside]#value"',
"[jwt.signing.read]\n'key' = 'literal\\path[#value]'",
r'[jwt.filer_signing]' + '\n' + r'key = "escaped\"quote\\slash\u0041"',
'[jwt.filer_signing.read]\nkey = ""',
))
yield "absent keys and sections / unrelated key", '\n'.join((
'[jwt.signing]\nexpires_after_seconds = 10',
'[jwt.signing.read]\nkey = "read-only"',
'[unrelated]\nkey = "not-a-jwt-key"',
'# [jwt.filer_signing]\n# key = "not-active"',
))
yield "no existing security.toml", ""
def check_generated(value):
decoded = base64.b64decode(value, validate=True).decode("ascii")
assert len(decoded) == 10 and decoded.isascii() and decoded.isalnum(), "invalid generated JWT key"
def check_helpers(helm, reference_helm):
# Exercise the real helper, not a second implementation of its matching rules.
with tempfile.TemporaryDirectory(prefix="helm-jwt-helper-") as directory:
chart = Path(directory)
(chart / "templates").mkdir()
(chart / "Chart.yaml").write_text("apiVersion: v2\nname: jwt-regression\nversion: 0.0.0\n")
shutil.copyfile(CHART / "templates/shared/_helpers.tpl", chart / "templates/_helpers.tpl")
entries = [
json.dumps(section) + ': {{ include "seaweedfs.existingTomlKey" (list '
+ json.dumps(section) + ' .Values.raw) | toJson }}'
for section in SECTIONS
]
template = chart / "templates/keys.yaml"
prefix = '{"apiVersion":"v1","kind":"ConfigMap","metadata":{"name":"keys"},"data":{'
helper_template = prefix + ",".join(entries) + "}}"
reference_entries = [
json.dumps(section) + ': {{ dig '
+ " ".join(json.dumps(part) for part in section.split("."))
+ ' "key" "__ABSENT__" (fromToml .Values.raw) | toJson }}'
for section in SECTIONS
]
reference_template = prefix + ",".join(reference_entries) + "}}"
def render(binary, source, raw):
template.write_text(source)
values = chart / "input.json"
values.write_text(json.dumps({"raw": raw}))
output = run(binary, "template", "keys", str(chart), "-f", str(values))
return json.loads(output[output.index("{"):])["data"]
for name, raw in fixtures():
expected = keys(raw)
tokens = render(helm, helper_template, raw)
actual = {section: tomllib.loads("key = " + token)["key"]
for section, token in tokens.items() if token != ""}
assert actual == expected, f"{name}: extracted keys differ from stored TOML"
if reference_helm:
reference = render(reference_helm, reference_template, raw)
reference = {section: value for section, value in reference.items() if value != "__ABSENT__"}
assert actual == reference, f"{name}: keys differ from fromToml/dig"
print(f"PASS helper: {name}")
# A present but unsupported value must not silently become a fresh key.
try:
render(helm, helper_template, '[jwt.signing]\nkey = """multi\nline"""')
except subprocess.CalledProcessError as error:
assert "refusing to replace an existing key" in error.stderr, error.stderr
else:
raise AssertionError("multiline existing key was silently accepted or replaced")
print("PASS helper: unsupported existing value fails without rotation")
try:
render(helm, helper_template, ESCAPED_HEADER)
except subprocess.CalledProcessError as error:
assert "unsupported quoted section header" in error.stderr, error.stderr
else:
raise AssertionError("unsupported quoted header silently rotated its key")
print("PASS helper: unsupported quoted header fails without rotation")
def check_upgrades(helm, context):
namespace = "jwt-key-persist-" + uuid.uuid4().hex[:8]
current = "jk-seaweedfs-security-config"
legacy = "seaweedfs-security-config"
kubectl = ["kubectl", "--context", context, "-n", namespace]
release_args = ["jk", str(CHART), "--kube-context", context, "-n", namespace]
# No workload is needed to exercise Helm's real ConfigMap lookup and update.
for setting in (
"master.enabled=false", "volume.enabled=false", "filer.enabled=false",
"global.seaweedfs.createClusterRole=false",
"global.seaweedfs.securityConfig.jwtSigning.volumeWrite=true",
"global.seaweedfs.securityConfig.jwtSigning.volumeRead=true",
"global.seaweedfs.securityConfig.jwtSigning.filerWrite=true",
"global.seaweedfs.securityConfig.jwtSigning.filerRead=true",
):
release_args += ["--set", setting]
def stored():
cm = json.loads(run(*kubectl, "get", "configmap", current, "-o", "json"))
return keys(cm["data"]["security.toml"])
def upgrade():
run(helm, "upgrade", *release_args)
return stored()
def patch(raw):
# Seed previous-release content without taking Helm 4's SSA ownership.
run(*kubectl, "patch", "configmap", current, "--type=merge", "--field-manager=helm", "-p",
json.dumps({"data": {"security.toml": raw}}))
run(*kubectl, "create", "namespace", namespace)
try:
run(helm, "install", *release_args)
initial = stored()
assert set(initial) == set(SECTIONS), "install omitted a JWT section"
for value in initial.values():
check_generated(value)
assert upgrade() == initial, "no-op upgrade changed an existing key"
print("PASS upgrade: all four generated keys persist")
for name, raw in fixtures():
patch(raw)
actual = upgrade()
expected = keys(raw)
assert set(actual) == set(SECTIONS), f"{name}: upgrade omitted a JWT section"
for section in SECTIONS:
if section in expected:
assert actual[section] == expected[section], f"{name}: changed {section}"
else:
check_generated(actual[section])
assert upgrade() == actual, f"{name}: subsequent upgrade changed a key"
print(f"PASS upgrade: {name}")
# Migration from the old chart name, followed by precedence of the current name.
legacy_raw = "\n".join(f'[{section}]\nkey = "legacy-{index}"'
for index, section in enumerate(SECTIONS))
run(*kubectl, "create", "configmap", legacy, "--from-literal=security.toml=" + legacy_raw)
run(*kubectl, "delete", "configmap", current)
assert upgrade() == keys(legacy_raw), "legacy ConfigMap keys were not preserved"
print("PASS upgrade: legacy ConfigMap migration")
current_raw = next(fixtures())[1]
patch(current_raw)
assert upgrade() == keys(current_raw), "legacy ConfigMap overrode current ConfigMap"
print("PASS upgrade: current ConfigMap takes precedence")
patch(ESCAPED_HEADER)
try:
upgrade()
except subprocess.CalledProcessError as error:
assert "unsupported quoted section header" in error.stderr, error.stderr
else:
raise AssertionError("unsupported quoted header silently rotated its key")
assert keys(run(*kubectl, "get", "configmap", current, "-o",
"jsonpath={.data.security\\.toml}"))["jwt.signing"] == "existing"
print("PASS upgrade: unsupported quoted header leaves stored key untouched")
finally:
run(*kubectl, "delete", "namespace", namespace, "--wait=false")
def main():
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--helm", default="helm")
parser.add_argument("--reference-helm", help="Helm >=3.17 binary for differential checks")
parser.add_argument("--context", help="explicit disposable Kubernetes context for live upgrade checks")
args = parser.parse_args()
print(run(args.helm, "version", "--short").strip())
check_helpers(args.helm, args.reference_helm)
if args.context:
check_upgrades(args.helm, args.context)
if __name__ == "__main__":
try:
main()
except subprocess.CalledProcessError as error:
print(error.stderr, file=sys.stderr)
sys.exit(error.returncode)
+3 -3
View File
@@ -27,7 +27,7 @@ jobs:
# Initializes the CodeQL tools for scanning.
- name: Initialize CodeQL
uses: github/codeql-action/init@v4.38.2
uses: github/codeql-action/init@v4.38.0
# Override language selection by uncommenting this and choosing your languages
with:
languages: go
@@ -35,7 +35,7 @@ jobs:
# Autobuild attempts to build any compiled languages (C/C++, C#, or Java).
# If this step fails, then you should remove it and run the build manually (see below).
- name: Autobuild
uses: github/codeql-action/autobuild@v4.38.2
uses: github/codeql-action/autobuild@v4.38.0
# ℹ️ Command-line programs to run using the OS shell.
# 📚 See https://docs.github.com/en/actions/using-workflows/workflow-syntax-for-github-actions#jobsjob_idstepsrun
@@ -49,4 +49,4 @@ jobs:
# make release
- name: Perform CodeQL Analysis
uses: github/codeql-action/analyze@v4.38.2
uses: github/codeql-action/analyze@v4.38.0
+1 -2
View File
@@ -6,7 +6,6 @@ on:
paths:
- 'weed/**'
- 'seaweed-volume/**'
- 'seaweed-common/**'
- 'seaweed-worker/**'
- 'docker/**'
- 'go.mod'
@@ -153,7 +152,7 @@ jobs:
org.opencontainers.image.vendor=Chris Lu
- name: Set up QEMU
uses: docker/setup-qemu-action@v4.4.0
uses: docker/setup-qemu-action@v4.3.0
- name: Create BuildKit config
run: |
@@ -129,7 +129,7 @@ jobs:
echo "seaweedfs_ref=$seaweed" >> "$GITHUB_OUTPUT"
- name: Set up QEMU
uses: docker/setup-qemu-action@v4.4.0
uses: docker/setup-qemu-action@v4.3.0
- name: Set up Docker Buildx
uses: docker/setup-buildx-action@v4
+6 -10
View File
@@ -156,7 +156,7 @@ jobs:
runs-on: ubuntu-latest
strategy:
matrix:
platform: [amd64, arm64, arm, 386, ppc64le, s390x]
platform: [amd64, arm64, arm, 386]
variant: ${{ fromJSON(needs.setup.outputs.variants) }}
steps:
@@ -236,7 +236,7 @@ jobs:
org.opencontainers.image.vendor=Chris Lu
- name: Set up QEMU
if: matrix.platform != 'amd64'
uses: docker/setup-qemu-action@v4.4.0
uses: docker/setup-qemu-action@v4.3.0
- name: Create BuildKit config
run: |
cat > /tmp/buildkitd.toml <<EOF
@@ -405,7 +405,7 @@ jobs:
output: trivy-results.sarif
exit-code: '0'
- name: Upload Trivy scan results to GitHub Security
uses: github/codeql-action/upload-sarif@v4.38.2
uses: github/codeql-action/upload-sarif@v4.38.0
if: always()
with:
sarif_file: trivy-results.sarif
@@ -505,9 +505,7 @@ jobs:
ghcr.io/chrislusf/seaweedfs:${BASE_TAG}${SUFFIX}-amd64 \
ghcr.io/chrislusf/seaweedfs:${BASE_TAG}${SUFFIX}-arm64 \
ghcr.io/chrislusf/seaweedfs:${BASE_TAG}${SUFFIX}-arm \
ghcr.io/chrislusf/seaweedfs:${BASE_TAG}${SUFFIX}-386 \
ghcr.io/chrislusf/seaweedfs:${BASE_TAG}${SUFFIX}-ppc64le \
ghcr.io/chrislusf/seaweedfs:${BASE_TAG}${SUFFIX}-s390x
ghcr.io/chrislusf/seaweedfs:${BASE_TAG}${SUFFIX}-386
# The copy and the signature below use the digest this run pushed, not whatever the tag points at by then.
DIGEST=$(jq -er '."containerimage.descriptor".digest' /tmp/manifest.json)
echo "digest=${DIGEST}" >> "$GITHUB_OUTPUT"
@@ -551,15 +549,13 @@ jobs:
echo "Using skopeo to copy..."
retry_with_backoff skopeo copy --all docker://ghcr.io/chrislusf/seaweedfs@${DIGEST} docker://chrislusf/seaweedfs:${BASE_TAG}${SUFFIX}
else
echo "Using docker buildx imagetools (pulling 6 images from Docker Hub)..."
echo "Using docker buildx imagetools (pulling 4 images from Docker Hub)..."
# Fallback: create manifest directly on Docker Hub (pulls from Docker Hub - rate limited)
retry_with_backoff docker buildx imagetools create -t chrislusf/seaweedfs:${BASE_TAG}${SUFFIX} \
ghcr.io/chrislusf/seaweedfs:${BASE_TAG}${SUFFIX}-amd64 \
ghcr.io/chrislusf/seaweedfs:${BASE_TAG}${SUFFIX}-arm64 \
ghcr.io/chrislusf/seaweedfs:${BASE_TAG}${SUFFIX}-arm \
ghcr.io/chrislusf/seaweedfs:${BASE_TAG}${SUFFIX}-386 \
ghcr.io/chrislusf/seaweedfs:${BASE_TAG}${SUFFIX}-ppc64le \
ghcr.io/chrislusf/seaweedfs:${BASE_TAG}${SUFFIX}-s390x
ghcr.io/chrislusf/seaweedfs:${BASE_TAG}${SUFFIX}-386
fi
- name: Sign
@@ -46,7 +46,7 @@ jobs:
org.opencontainers.image.vendor=Chris Lu
-
name: Set up QEMU
uses: docker/setup-qemu-action@v4.4.0
uses: docker/setup-qemu-action@v4.3.0
-
name: Set up Docker Buildx
uses: docker/setup-buildx-action@v4
@@ -149,16 +149,12 @@ jobs:
- { variant: normal, tag_suffix: "", dockerfile: ./docker/Dockerfile.go_build, build_args: "", rust_variant: normal, platform: linux/arm64, arch: arm64, runner: ubuntu-24.04-arm, qemu: false }
- { variant: normal, tag_suffix: "", dockerfile: ./docker/Dockerfile.go_build, build_args: "", rust_variant: normal, platform: linux/arm/v7, arch: armv7, runner: ubuntu-latest, qemu: true }
- { variant: normal, tag_suffix: "", dockerfile: ./docker/Dockerfile.go_build, build_args: "", rust_variant: normal, platform: linux/386, arch: i386, runner: ubuntu-latest, qemu: false }
- { variant: normal, tag_suffix: "", dockerfile: ./docker/Dockerfile.go_build, build_args: "", rust_variant: normal, platform: linux/ppc64le, arch: ppc64le, runner: ubuntu-latest, qemu: true }
- { variant: normal, tag_suffix: "", dockerfile: ./docker/Dockerfile.go_build, build_args: "", rust_variant: normal, platform: linux/s390x, arch: s390x, runner: ubuntu-latest, qemu: true }
# Large disk - multi-arch
- { variant: large_disk, tag_suffix: _large_disk, dockerfile: ./docker/Dockerfile.go_build, build_args: TAGS=5BytesOffset, rust_variant: large-disk, platform: linux/amd64, arch: amd64, runner: ubuntu-latest, qemu: false }
- { variant: large_disk, tag_suffix: _large_disk, dockerfile: ./docker/Dockerfile.go_build, build_args: TAGS=5BytesOffset, rust_variant: large-disk, platform: linux/arm64, arch: arm64, runner: ubuntu-24.04-arm, qemu: false }
- { variant: large_disk, tag_suffix: _large_disk, dockerfile: ./docker/Dockerfile.go_build, build_args: TAGS=5BytesOffset, rust_variant: large-disk, platform: linux/arm/v7, arch: armv7, runner: ubuntu-latest, qemu: true }
- { variant: large_disk, tag_suffix: _large_disk, dockerfile: ./docker/Dockerfile.go_build, build_args: TAGS=5BytesOffset, rust_variant: large-disk, platform: linux/386, arch: i386, runner: ubuntu-latest, qemu: false }
- { variant: large_disk, tag_suffix: _large_disk, dockerfile: ./docker/Dockerfile.go_build, build_args: TAGS=5BytesOffset, rust_variant: large-disk, platform: linux/ppc64le, arch: ppc64le, runner: ubuntu-latest, qemu: true }
- { variant: large_disk, tag_suffix: _large_disk, dockerfile: ./docker/Dockerfile.go_build, build_args: TAGS=5BytesOffset, rust_variant: large-disk, platform: linux/s390x, arch: s390x, runner: ubuntu-latest, qemu: true }
# Full tags - multi-arch
- { variant: full, tag_suffix: _full, dockerfile: ./docker/Dockerfile.go_build, build_args: "TAGS=elastic,gocdk,rclone,sqlite,tarantool,tikv,ydb", rust_variant: normal, platform: linux/amd64, arch: amd64, runner: ubuntu-latest, qemu: false }
@@ -235,7 +231,7 @@ jobs:
- name: Set up QEMU
if: (github.event_name != 'workflow_dispatch' || github.event.inputs.variant == 'all' || github.event.inputs.variant == matrix.variant) && matrix.qemu
uses: docker/setup-qemu-action@v4.4.0
uses: docker/setup-qemu-action@v4.3.0
- name: Create BuildKit config
if: github.event_name != 'workflow_dispatch' || github.event.inputs.variant == 'all' || github.event.inputs.variant == matrix.variant
@@ -460,7 +456,7 @@ jobs:
- name: Upload Trivy scan results to GitHub Security
if: always()
uses: github/codeql-action/upload-sarif@v4.38.2
uses: github/codeql-action/upload-sarif@v4.38.0
with:
sarif_file: trivy-results.sarif
category: trivy-${{ matrix.variant }}
@@ -85,7 +85,7 @@ jobs:
echo "seaweedfs_ref=$seaweed" >> "$GITHUB_OUTPUT"
- name: Set up QEMU
uses: docker/setup-qemu-action@99012661954931238ded8c8b007157a8430204e1 # v1
uses: docker/setup-qemu-action@1f40c72289eff860ee54a304f1438e3cff362e0a # v1
- name: Set up Docker Buildx
uses: docker/setup-buildx-action@4d04d5d9486b7bd6fa91e7baf45bbb4f8b9deedd # v1
+35 -45
View File
@@ -35,48 +35,11 @@ jobs:
cd telemetry/server
go mod tidy
echo "Building telemetry server..."
CGO_ENABLED=0 GOOS=linux GOARCH=amd64 go build -o ../../telemetry-server .
GOOS=linux GOARCH=amd64 go build -o ../../telemetry-server .
cd ../..
ls -la telemetry-server
echo "Build completed successfully"
- name: Generate Service Configuration
if: github.event_name == 'workflow_dispatch' && (inputs.setup || inputs.deploy)
env:
REMOTE_USER: ${{ secrets.TELEMETRY_USER }}
run: |
# Create systemd service file
echo "
[Unit]
Description=SeaweedFS Telemetry Server
After=network.target
[Service]
Type=simple
User=$REMOTE_USER
WorkingDirectory=/home/$REMOTE_USER/seaweedfs-telemetry
ExecStart=/bin/sh -c 'exec /home/$REMOTE_USER/seaweedfs-telemetry/bin/telemetry-server -port=8353 >>/home/$REMOTE_USER/seaweedfs-telemetry/logs/telemetry.log 2>>/home/$REMOTE_USER/seaweedfs-telemetry/logs/telemetry.error.log'
Restart=always
RestartSec=5
[Install]
WantedBy=multi-user.target" > telemetry.service
# Setup logrotate configuration
echo "# SeaweedFS Telemetry service log rotation
/home/$REMOTE_USER/seaweedfs-telemetry/logs/*.log {
daily
rotate 30
compress
delaycompress
missingok
notifempty
create 644 $REMOTE_USER $REMOTE_USER
postrotate
systemctl restart telemetry.service
endscript
}" > telemetry_logrotate
- name: First-time Server Setup
if: github.event_name == 'workflow_dispatch' && inputs.setup
env:
@@ -98,6 +61,40 @@ jobs:
touch ~/seaweedfs-telemetry/logs/telemetry.log ~/seaweedfs-telemetry/logs/telemetry.error.log && \
chmod 644 ~/seaweedfs-telemetry/logs/*.log"
# Create systemd service file
echo "
[Unit]
Description=SeaweedFS Telemetry Server
After=network.target
[Service]
Type=simple
User=$REMOTE_USER
WorkingDirectory=/home/$REMOTE_USER/seaweedfs-telemetry
ExecStart=/home/$REMOTE_USER/seaweedfs-telemetry/bin/telemetry-server -port=8353
Restart=always
RestartSec=5
StandardOutput=append:/home/$REMOTE_USER/seaweedfs-telemetry/logs/telemetry.log
StandardError=append:/home/$REMOTE_USER/seaweedfs-telemetry/logs/telemetry.error.log
[Install]
WantedBy=multi-user.target" > telemetry.service
# Setup logrotate configuration
echo "# SeaweedFS Telemetry service log rotation
/home/$REMOTE_USER/seaweedfs-telemetry/logs/*.log {
daily
rotate 30
compress
delaycompress
missingok
notifempty
create 644 $REMOTE_USER $REMOTE_USER
postrotate
systemctl restart telemetry.service
endscript
}" > telemetry_logrotate
# Copy configuration files
scp -i ~/.ssh/deploy_key telemetry/grafana-dashboard.json $REMOTE_USER@$REMOTE_HOST:~/seaweedfs-telemetry/
scp -i ~/.ssh/deploy_key telemetry/prometheus.yml $REMOTE_USER@$REMOTE_HOST:~/seaweedfs-telemetry/
@@ -140,18 +137,11 @@ jobs:
scp -i ~/.ssh/deploy_key telemetry/grafana-dashboard.json $REMOTE_USER@$REMOTE_HOST:~/seaweedfs-telemetry/
scp -i ~/.ssh/deploy_key telemetry/prometheus.yml $REMOTE_USER@$REMOTE_HOST:~/seaweedfs-telemetry/
# Copy updated service and logrotate files
scp -i ~/.ssh/deploy_key telemetry.service telemetry_logrotate $REMOTE_USER@$REMOTE_HOST:~/seaweedfs-telemetry/
# Check if service exists and deploy accordingly
ssh -i ~/.ssh/deploy_key $REMOTE_USER@$REMOTE_HOST "
if systemctl list-unit-files telemetry.service >/dev/null 2>&1; then
echo 'Service exists, performing update...'
set -e
sudo systemctl stop telemetry.service
sudo mv ~/seaweedfs-telemetry/telemetry.service /etc/systemd/system/
sudo mv ~/seaweedfs-telemetry/telemetry_logrotate /etc/logrotate.d/seaweedfs-telemetry
sudo systemctl daemon-reload
mkdir -p ~/seaweedfs-telemetry/bin
mv ~/seaweedfs-telemetry/tmp/telemetry-server ~/seaweedfs-telemetry/bin/
chmod +x ~/seaweedfs-telemetry/bin/telemetry-server
+3 -313
View File
@@ -3,10 +3,10 @@ name: "helm: lint and test charts"
on:
push:
branches: [ master ]
paths: ['k8s/**', '.github/workflows/helm_ci.yml', '.github/scripts/helm_jwt_keys.py']
paths: ['k8s/**', '.github/workflows/helm_ci.yml']
pull_request:
branches: [ master ]
paths: ['k8s/**', '.github/workflows/helm_ci.yml', '.github/scripts/helm_jwt_keys.py']
paths: ['k8s/**', '.github/workflows/helm_ci.yml']
permissions:
contents: read
@@ -20,14 +20,6 @@ jobs:
with:
fetch-depth: 0
- name: Set up Helm before fromToml was available
uses: azure/setup-helm@v5
with:
version: v3.16.3
- name: Record legacy Helm binary
run: echo "HELM_LEGACY=$(command -v helm)" >> "$GITHUB_ENV"
- name: Set up Helm
uses: azure/setup-helm@v5
with:
@@ -52,17 +44,6 @@ jobs:
- name: Run chart-testing (lint)
run: ct lint --target-branch ${{ github.event.repository.default_branch }} --all --validate-maintainers=false --chart-dirs k8s/charts
- name: Verify legacy Helm rendering
run: |
"$HELM_LEGACY" lint k8s/charts/seaweedfs
"$HELM_LEGACY" template test k8s/charts/seaweedfs > "$RUNNER_TEMP/legacy-default.yaml"
"$HELM_LEGACY" template test k8s/charts/seaweedfs \
--set global.seaweedfs.securityConfig.jwtSigning.volumeRead=true \
--set global.seaweedfs.securityConfig.jwtSigning.filerWrite=true \
--set global.seaweedfs.securityConfig.jwtSigning.filerRead=true \
> "$RUNNER_TEMP/legacy-jwt.yaml"
- name: Verify template rendering
run: |
set -e
@@ -76,25 +57,7 @@ jobs:
helm template test $CHART_DIR --set s3.enabled=true > /tmp/s3.yaml
grep -q "kind: Deployment" /tmp/s3.yaml && grep -q "seaweedfs-s3" /tmp/s3.yaml
echo "S3 deployment renders correctly"
echo "=== Testing S3 rollout settings ==="
helm template test $CHART_DIR --show-only templates/s3/s3-deployment.yaml \
--set s3.enabled=true > /tmp/s3-rollout-defaults.yaml
grep -q "terminationGracePeriodSeconds: 10$" /tmp/s3-rollout-defaults.yaml
test "$(grep -cE '^ strategy:|^ +lifecycle:' /tmp/s3-rollout-defaults.yaml)" -eq 0
helm template test $CHART_DIR --show-only templates/s3/s3-deployment.yaml \
--set s3.enabled=true \
--set s3.terminationGracePeriodSeconds=30 \
--set s3.updateStrategy.type=RollingUpdate \
--set s3.updateStrategy.rollingUpdate.maxUnavailable=0 \
--set-string 's3.lifecycle.preStop.exec.command={sleep,5}' \
> /tmp/s3-rollout.yaml
grep -q "terminationGracePeriodSeconds: 30$" /tmp/s3-rollout.yaml
grep -A 4 "^ strategy:" /tmp/s3-rollout.yaml | grep -q "maxUnavailable: 0"
grep -A 4 "^ strategy:" /tmp/s3-rollout.yaml | grep -q "type: RollingUpdate"
grep -A 5 "^ lifecycle:" /tmp/s3-rollout.yaml | grep -q -- "- sleep"
echo "S3 rollout settings render correctly"
echo "=== Testing S3 credentials from an existing secret ==="
credential_args=(
--set s3.credentials.admin.existingSecret=minio-root
@@ -153,35 +116,6 @@ jobs:
grep -q "security-config" /tmp/security.yaml
echo "Security configuration renders correctly"
echo "=== Testing secure bucket-creation hook certificate mounts ==="
helm template test $CHART_DIR \
--show-only templates/shared/post-install-bucket-hook.yaml \
--set s3.enabled=true \
--set 's3.createBuckets[0].name=data' \
--set global.seaweedfs.enableSecurity=true \
> /tmp/security-bucket-hook.yaml
test "$(grep -cE '^[[:space:]]*- name: ca-cert$' /tmp/security-bucket-hook.yaml)" -eq 2
test "$(grep -cE '^[[:space:]]*- name: client-cert$' /tmp/security-bucket-hook.yaml)" -eq 2
grep -q 'mountPath: /usr/local/share/ca-certificates/ca/' /tmp/security-bucket-hook.yaml
grep -q 'mountPath: /usr/local/share/ca-certificates/client/' /tmp/security-bucket-hook.yaml
grep -q 'secretName: test-seaweedfs-ca-cert' /tmp/security-bucket-hook.yaml
grep -q 'secretName: test-seaweedfs-client-cert' /tmp/security-bucket-hook.yaml
echo "Secure bucket-creation hook mounts its CA and client certificate"
echo ""
echo "=== Testing admin.allowInsecureBind satisfies the admin auth render guard ==="
helm template test $CHART_DIR --set admin.enabled=true --set admin.allowInsecureBind=true \
> /tmp/admin-allow-insecure-bind.yaml
grep -q -- "-allowInsecureBind" /tmp/admin-allow-insecure-bind.yaml
echo "admin.allowInsecureBind renders -allowInsecureBind and passes the render guard"
if helm template test $CHART_DIR --set admin.enabled=true > /tmp/admin-no-auth.yaml 2>/tmp/admin-no-auth.err; then
echo "FAIL: admin.enabled=true with no auth configured should fail to render"
exit 1
fi
grep -q "admin.allowInsecureBind" /tmp/admin-no-auth.err
echo "admin with no auth configured still fails the render guard, and the guard mentions admin.allowInsecureBind"
echo ""
echo "=== Testing JWT expiration overrides ==="
helm template test $CHART_DIR \
@@ -770,168 +704,6 @@ jobs:
helm template test $CHART_DIR --set cosi.enabled=true > /tmp/cosi.yaml
grep -q "seaweedfs-cosi" /tmp/cosi.yaml
echo "COSI driver renders correctly"
echo ""
echo "=== Testing configurable pod and container security contexts ==="
helm template test $CHART_DIR > /tmp/security-context-defaults.yaml
security_context_args=()
for component in master volume filer s3 sftp admin worker allInOne cosi; do
security_context_args+=(
--set "$component.podSecurityContext.enabled=true"
--set "$component.podSecurityContext.seccompProfile.type=RuntimeDefault"
--set "$component.containerSecurityContext.enabled=true"
--set "$component.containerSecurityContext.privileged=false"
--set "$component.containerSecurityContext.allowPrivilegeEscalation=false"
--set "$component.containerSecurityContext.readOnlyRootFilesystem=true"
--set "$component.containerSecurityContext.capabilities.drop[0]=ALL"
--set "$component.containerSecurityContext.seccompProfile.type=RuntimeDefault"
)
done
helm template test $CHART_DIR \
"${security_context_args[@]}" \
--set s3.enabled=true \
--set s3.createBuckets[0].name=test \
--set sftp.enabled=true \
--set admin.enabled=true \
--set admin.secret.adminPassword=ci-admin-password \
--set worker.enabled=true \
--set volume.idx.type=hostPath \
--set volume.idx.hostPathPrefix=/tmp \
--set cosi.enabled=true \
--set global.seaweedfs.tmpDir.sizeLimit=64Mi > /tmp/security-contexts.yaml
helm template test $CHART_DIR \
"${security_context_args[@]}" \
--set allInOne.enabled=true \
--set master.enabled=false \
--set volume.enabled=false \
--set filer.enabled=false \
--set global.seaweedfs.tmpDir.sizeLimit=64Mi > /tmp/security-contexts-aio.yaml
python3 - /tmp/security-context-defaults.yaml /tmp/security-contexts.yaml /tmp/security-contexts-aio.yaml <<'PYEOF'
import sys
import yaml
errors = []
workloads = 0
containers = 0
chart_managed_init_containers = 0
components = set()
def validate_container(workload_name, container):
context = container.get("securityContext", {})
prefix = f"{workload_name}/{container['name']}"
if "enabled" in context:
errors.append(f"{prefix}: internal enabled flag leaked into container securityContext")
if context.get("privileged") is not False:
errors.append(f"{prefix}: privileged is not false")
if context.get("allowPrivilegeEscalation") is not False:
errors.append(f"{prefix}: allowPrivilegeEscalation is not false")
if context.get("readOnlyRootFilesystem") is not True:
errors.append(f"{prefix}: readOnlyRootFilesystem is not true")
if context.get("capabilities", {}).get("drop") != ["ALL"]:
errors.append(f"{prefix}: capabilities.drop is not exactly [ALL]")
if context.get("seccompProfile", {}).get("type") != "RuntimeDefault":
errors.append(f"{prefix}: seccompProfile is not RuntimeDefault")
mounts = {mount["name"]: mount for mount in container.get("volumeMounts", [])}
if mounts.get("seaweedfs-tmp", {}).get("mountPath") != "/tmp":
errors.append(f"{prefix}: writable temporary volume is not mounted at /tmp")
with open(sys.argv[1]) as stream:
default_documents = [document for document in yaml.safe_load_all(stream) if document]
for document in default_documents:
if document.get("kind") not in ("Deployment", "StatefulSet", "Job"):
continue
name = document["metadata"]["name"]
pod = document["spec"]["template"]["spec"]
if "securityContext" in pod:
errors.append(f"{name}: pod securityContext should be absent by default")
if any(volume["name"] == "seaweedfs-tmp" for volume in pod.get("volumes", [])):
errors.append(f"{name}: writable /tmp volume should be absent by default")
for container in pod.get("containers", []):
if "securityContext" in container:
errors.append(
f"{name}/{container['name']}: container securityContext "
f"should be absent by default"
)
if any(
mount["name"] == "seaweedfs-tmp"
for mount in container.get("volumeMounts", [])
):
errors.append(
f"{name}/{container['name']}: writable /tmp mount "
f"should be absent by default"
)
for path in sys.argv[2:]:
with open(path) as stream:
documents = [document for document in yaml.safe_load_all(stream) if document]
for document in documents:
if document.get("kind") not in ("Deployment", "StatefulSet", "Job"):
continue
workloads += 1
name = document["metadata"]["name"]
pod = document["spec"]["template"]["spec"]
volumes = {volume["name"]: volume for volume in pod.get("volumes", [])}
temporary = volumes.get("seaweedfs-tmp", {}).get("emptyDir")
if temporary is None:
errors.append(f"{name}: writable /tmp emptyDir is missing")
elif temporary.get("sizeLimit") != "64Mi":
errors.append(f"{name}: temporary volume sizeLimit is not 64Mi")
component = document["spec"]["template"]["metadata"]["labels"].get("app.kubernetes.io/component")
if component:
components.add(component)
else:
errors.append(f"{name}: app.kubernetes.io/component label is missing")
pod_context = pod.get("securityContext", {})
if "enabled" in pod_context:
errors.append(f"{name}: internal enabled flag leaked into pod securityContext")
if pod_context.get("seccompProfile", {}).get("type") != "RuntimeDefault":
errors.append(f"{name}: pod seccompProfile is not RuntimeDefault")
for container in pod.get("containers", []):
containers += 1
validate_container(name, container)
for container in pod.get("initContainers", []):
if container["name"] != "seaweedfs-vol-move-idx":
continue
chart_managed_init_containers += 1
validate_container(name, container)
expected_components = {
"master", "volume", "filer", "s3", "sftp", "admin", "worker",
"objectstorage-provisioner", "bucket-hook", "seaweedfs-all-in-one",
}
if components != expected_components:
errors.append(
f"security context workload coverage is incomplete: "
f"expected {sorted(expected_components)}, got {sorted(components)}"
)
if workloads != 10 or containers != 12:
errors.append(
f"expected 10 workloads and 12 containers, got "
f"{workloads} workloads and {containers} containers"
)
if chart_managed_init_containers != 1:
errors.append(
f"expected one chart-managed seaweedfs-vol-move-idx init container, "
f"got {chart_managed_init_containers}"
)
if errors:
print("\n".join(f"FAIL: {error}" for error in errors), file=sys.stderr)
sys.exit(1)
print(f"Validated security contexts on {workloads} workloads and {containers} containers")
PYEOF
# The resize hook depends on lookup finding a live StatefulSet and
# therefore cannot render during helm template. Keep static coverage
# for both security-context blocks.
grep -Fqx ' securityContext: {{- omit .Values.volume.podSecurityContext "enabled" | toYaml | nindent 8 }}' \
"$CHART_DIR/templates/volume/volume-resize-hook.yaml"
grep -Fqx ' securityContext: {{- omit .Values.volume.containerSecurityContext "enabled" | toYaml | nindent 12 }}' \
"$CHART_DIR/templates/volume/volume-resize-hook.yaml"
echo "Volume resize hook security contexts are covered"
echo ""
echo "=== Testing long release name: service names match DNS references ==="
@@ -1819,78 +1591,6 @@ jobs:
- name: Create kind cluster
uses: helm/kind-action@v1.15.0
- name: Verify volume resize hook security contexts
run: |
set -e
CHART_DIR="k8s/charts/seaweedfs"
NS="resize-hook-security"
kubectl create namespace "$NS"
kubectl apply -n "$NS" -f - <<'EOF'
apiVersion: v1
kind: PersistentVolumeClaim
metadata:
name: data1-resize-seaweedfs-volume-0
spec:
accessModes:
- ReadWriteOnce
resources:
requests:
storage: 1Gi
EOF
helm template resize "$CHART_DIR" -n "$NS" --dry-run=server \
--set volume.dataDirs[0].name=data1 \
--set volume.dataDirs[0].type=persistentVolumeClaim \
--set volume.dataDirs[0].size=2Gi \
--set volume.podSecurityContext.enabled=true \
--set volume.podSecurityContext.seccompProfile.type=RuntimeDefault \
--set volume.containerSecurityContext.enabled=true \
--set volume.containerSecurityContext.privileged=false \
--set volume.containerSecurityContext.allowPrivilegeEscalation=false \
--set volume.containerSecurityContext.readOnlyRootFilesystem=true \
--set volume.containerSecurityContext.capabilities.drop[0]=ALL \
--set volume.containerSecurityContext.seccompProfile.type=RuntimeDefault \
--set global.seaweedfs.tmpDir.sizeLimit=64Mi \
> /tmp/security-context-resize-hook.yaml
python3 - /tmp/security-context-resize-hook.yaml <<'PYEOF'
import sys
import yaml
with open(sys.argv[1]) as stream:
documents = [document for document in yaml.safe_load_all(stream) if document]
jobs = [
document for document in documents
if document.get("kind") == "Job"
and document["metadata"]["name"].endswith("-volume-resize-hook")
]
if len(jobs) != 1:
raise AssertionError(f"expected one volume resize hook Job, got {len(jobs)}")
pod = jobs[0]["spec"]["template"]["spec"]
assert pod["securityContext"] == {
"seccompProfile": {"type": "RuntimeDefault"},
}
assert len(pod["containers"]) == 1
assert pod["containers"][0]["securityContext"] == {
"allowPrivilegeEscalation": False,
"capabilities": {"drop": ["ALL"]},
"privileged": False,
"readOnlyRootFilesystem": True,
"seccompProfile": {"type": "RuntimeDefault"},
}
assert pod["containers"][0]["volumeMounts"] == [
{"mountPath": "/tmp", "name": "seaweedfs-tmp"},
]
assert pod["volumes"] == [
{"emptyDir": {"sizeLimit": "64Mi"}, "name": "seaweedfs-tmp"},
]
PYEOF
kubectl delete namespace "$NS"
echo "Volume resize hook security contexts render correctly"
- name: Run chart-testing (install)
run: |
ct install --target-branch ${{ github.event.repository.default_branch }} --all --chart-dirs k8s/charts \
@@ -1941,16 +1641,6 @@ jobs:
kubectl delete namespace "$NS"
echo "SFTP host key lifecycle tests passed"
- name: Verify JWT signing key persistence across upgrades
run: |
# chart-testing puts its pip-less venv first on PATH; use setup-python.
PYTHON="$pythonLocation/bin/python3"
"$PYTHON" -m pip install tomli==2.2.1
CONTEXT=$(kubectl config current-context)
"$PYTHON" .github/scripts/helm_jwt_keys.py --context "$CONTEXT"
"$PYTHON" .github/scripts/helm_jwt_keys.py --helm "$HELM_LEGACY" \
--reference-helm helm --context "$CONTEXT"
- name: Verify install into a default-deny namespace
run: |
set -e
-1
View File
@@ -8,7 +8,6 @@ on:
- 'go.mod'
- 'go.sum'
- 'seaweed-volume/**'
- 'seaweed-common/**'
- 'test/perf/**'
- '.github/workflows/performance.yml'
workflow_dispatch:
+3 -82
View File
@@ -5,7 +5,6 @@ on:
branches: [ master ]
paths:
- 'seaweed-volume/**'
- 'seaweed-common/**'
- 'test/volume_server/**'
- 'weed/pb/volume_server.proto'
- 'weed/pb/volume_server_pb/**'
@@ -14,7 +13,6 @@ on:
branches: [ master, main ]
paths:
- 'seaweed-volume/**'
- 'seaweed-common/**'
- 'test/volume_server/**'
- 'weed/pb/volume_server.proto'
- 'weed/pb/volume_server_pb/**'
@@ -29,29 +27,6 @@ permissions:
jobs:
changes:
name: Detect changed paths
runs-on: ubuntu-latest
timeout-minutes: 5
permissions:
contents: read
outputs:
rust: ${{ steps.filter.outputs.rust }}
steps:
- name: Checkout code
uses: actions/checkout@v7
with:
fetch-depth: 0
- name: Filter changed paths
id: filter
uses: dorny/paths-filter@v4
with:
filters: |
rust:
- 'seaweed-volume/**'
- '.github/workflows/rust-volume-server-tests.yml'
rust-unit-tests:
name: Rust Unit Tests
runs-on: ubuntu-22.04
@@ -77,7 +52,7 @@ jobs:
~/.cargo/registry
~/.cargo/git
seaweed-volume/target
key: rust-${{ steps.toolchain.outputs.fingerprint }}-${{ hashFiles('seaweed-volume/Cargo.lock', 'seaweed-common/Cargo.lock') }}
key: rust-${{ steps.toolchain.outputs.fingerprint }}-${{ hashFiles('seaweed-volume/Cargo.lock') }}
restore-keys: |
rust-${{ steps.toolchain.outputs.fingerprint }}-
@@ -90,60 +65,6 @@ jobs:
# - name: Clippy
# run: cd seaweed-volume && cargo clippy --all-targets -- -D warnings
# The crate is rustfmt-clean as of the PR that added this step.
# Uncomment to keep it that way.
# - name: Check formatting
# run: cd seaweed-volume && cargo fmt --check
# seaweed-common is a path dependency of this crate, not a member of its
# workspace, so the run below does not reach its own tests. It builds into
# this job's cached target directory, and the cache key above covers the
# shared crate's lock, so the aws-lc-sys that rustls pulls in is restored
# with the cache instead of compiled from scratch on every run.
- name: Run shared-crate unit tests
env:
CARGO_TARGET_DIR: ${{ github.workspace }}/seaweed-volume/target
run: cd seaweed-common && cargo test
- name: Run Rust unit tests
run: cd seaweed-volume && cargo test
- name: Run Rust unit tests (redb experimental cursor)
run: cd seaweed-volume && cargo test --features redb-experimental-cursor --lib storage::needle_map
rust-unit-tests-windows:
name: Rust Unit Tests (Windows)
runs-on: windows-latest
timeout-minutes: 30
needs: [changes]
if: needs.changes.outputs.rust == 'true'
defaults:
run:
shell: bash
steps:
- name: Checkout code
uses: actions/checkout@v7
- name: Install Rust toolchain
uses: dtolnay/rust-toolchain@stable
# No glibc on Windows: key the cache on the toolchain and OS only.
- name: Fingerprint build toolchain
id: toolchain
run: echo "fingerprint=windows-rustc-$(rustc -V | awk '{print $2}')" >> "$GITHUB_OUTPUT"
- name: Cache cargo registry and target
uses: actions/cache@v6
with:
path: |
~/.cargo/registry
~/.cargo/git
seaweed-volume/target
key: rust-windows-${{ steps.toolchain.outputs.fingerprint }}-${{ hashFiles('seaweed-volume/Cargo.lock') }}
restore-keys: |
rust-windows-${{ steps.toolchain.outputs.fingerprint }}-
- name: Run Rust unit tests
run: cd seaweed-volume && cargo test
@@ -180,7 +101,7 @@ jobs:
~/.cargo/registry
~/.cargo/git
seaweed-volume/target
key: rust-${{ steps.toolchain.outputs.fingerprint }}-${{ hashFiles('seaweed-volume/Cargo.lock', 'seaweed-common/Cargo.lock') }}
key: rust-${{ steps.toolchain.outputs.fingerprint }}-${{ hashFiles('seaweed-volume/Cargo.lock') }}
restore-keys: |
rust-${{ steps.toolchain.outputs.fingerprint }}-
@@ -262,7 +183,7 @@ jobs:
~/.cargo/registry
~/.cargo/git
seaweed-volume/target
key: rust-${{ steps.toolchain.outputs.fingerprint }}-${{ hashFiles('seaweed-volume/Cargo.lock', 'seaweed-common/Cargo.lock') }}
key: rust-${{ steps.toolchain.outputs.fingerprint }}-${{ hashFiles('seaweed-volume/Cargo.lock') }}
restore-keys: |
rust-${{ steps.toolchain.outputs.fingerprint }}-
+1 -19
View File
@@ -5,14 +5,12 @@ on:
branches: [ master ]
paths:
- 'seaweed-worker/**'
- 'seaweed-common/**'
- 'weed/pb/plugin.proto'
- '.github/workflows/rust-worker-tests.yml'
push:
branches: [ master, main ]
paths:
- 'seaweed-worker/**'
- 'seaweed-common/**'
- 'weed/pb/plugin.proto'
- '.github/workflows/rust-worker-tests.yml'
@@ -51,7 +49,7 @@ jobs:
~/.cargo/registry
~/.cargo/git
seaweed-worker/target/release
key: rust-worker-${{ steps.toolchain.outputs.fingerprint }}-${{ hashFiles('seaweed-worker/Cargo.lock', 'seaweed-common/Cargo.lock') }}
key: rust-worker-${{ steps.toolchain.outputs.fingerprint }}-${{ hashFiles('seaweed-worker/Cargo.lock') }}
restore-keys: |
rust-worker-${{ steps.toolchain.outputs.fingerprint }}-
@@ -81,22 +79,6 @@ jobs:
# - name: Clippy
# run: cd seaweed-worker && cargo clippy --workspace --all-targets -- -D warnings
# The workspace is rustfmt-clean as of the PR that added this step.
# Uncomment to keep it that way.
# - name: Check formatting
# run: cd seaweed-worker && cargo fmt --all --check
# seaweed-common is a path dependency of core and lance, not a member of
# this workspace, so `--workspace` below does not reach its own tests.
# Release and this job's cached target directory, and the cache key above
# covers the shared crate's lock. That lock pins the same rustls and
# aws-lc-sys this workspace resolves, so the release build above has
# already paid for them.
- name: Run shared-crate unit tests
env:
CARGO_TARGET_DIR: ${{ github.workspace }}/seaweed-worker/target
run: cd seaweed-common && cargo test --release
# The tests that need a live gateway skip themselves without one, the way
# the Go integration tests skip without Docker; the lifecycle suite in
# test/s3tables/lifecycle is what runs them against a real cluster.
-1
View File
@@ -5,7 +5,6 @@ on:
branches: [ master ]
paths:
- 'seaweed-volume/**'
- 'seaweed-common/**'
- '.github/workflows/rust_binaries_dev.yml'
permissions:
+2 -3
View File
@@ -139,8 +139,7 @@ jobs:
unit-tests:
name: Go Unit Tests (Implicit Directory)
runs-on: ubuntu-latest
# Leave time for setup and the focused run before the nine-minute suite.
timeout-minutes: 20
timeout-minutes: 10
steps:
- name: Checkout code
@@ -160,5 +159,5 @@ jobs:
- name: Run all S3 API tests
run: |
cd weed/s3api
go test -v -timeout 9m
go test -v -timeout 5m
-106
View File
@@ -1,106 +0,0 @@
name: "Snowflake S3Compat API tests"
on:
push:
branches: [ master ]
paths:
- 'weed/s3api/**'
- 'weed/filer/**'
- 'weed/server/**'
- 'weed/iam/**'
- 'weed/command/**'
- 'weed/storage/**'
- 'weed/operation/**'
- 'weed/wdclient/**'
- 'weed/cluster/**'
- 'weed/pb/**'
- 'test/s3/snowflake/**'
- 'go.mod'
- 'go.sum'
- '.github/workflows/s3-snowflake-tests.yml'
pull_request:
branches: [ master ]
paths:
- 'weed/s3api/**'
- 'weed/filer/**'
- 'weed/server/**'
- 'weed/iam/**'
- 'weed/command/**'
- 'weed/storage/**'
- 'weed/operation/**'
- 'weed/wdclient/**'
- 'weed/cluster/**'
- 'weed/pb/**'
- 'test/s3/snowflake/**'
- 'go.mod'
- 'go.sum'
- '.github/workflows/s3-snowflake-tests.yml'
concurrency:
group: ${{ github.event.pull_request.number || github.ref }}/s3-snowflake-tests
cancel-in-progress: true
permissions:
contents: read
jobs:
snowflake-s3compat-tests:
name: Snowflake S3Compat API tests
runs-on: ubuntu-22.04
timeout-minutes: 30
env:
WORK_DIR: /tmp/seaweedfs-snowflake-tests
steps:
- name: Check out code
uses: actions/checkout@v7
with:
persist-credentials: false
- name: Set up Go
uses: actions/setup-go@v7
with:
go-version-file: 'go.mod'
id: go
- name: Set up Java
uses: actions/setup-java@v6
with:
java-version: '17'
distribution: 'temurin'
cache: 'maven'
- name: Install SeaweedFS
run: |
cd weed
go install -buildvcs=false
weed version
- name: Run Snowflake S3Compat API tests
timeout-minutes: 20
run: |
# Starts weed server, creates the buckets/objects the suite needs,
# clones the upstream suite, and runs mvn -Dtest=S3CompatApiTest.
bash test/s3/snowflake/run.sh
- name: Show logs on failure
if: failure()
run: |
echo "=== SeaweedFS Server Log ==="
tail -200 "$WORK_DIR/weed.log" || echo "No server log"
echo ""
echo "=== Surefire results ==="
cat "$WORK_DIR"/snowflake-s3compat-api-test-suite/s3compatapi/target/surefire-reports/*.txt 2>/dev/null || echo "No surefire reports"
- name: Upload test results
if: always()
uses: actions/upload-artifact@v7
with:
name: snowflake-s3compat-surefire-reports
path: /tmp/seaweedfs-snowflake-tests/snowflake-s3compat-api-test-suite/s3compatapi/target/surefire-reports/
retention-days: 14
- name: Cleanup
if: always()
run: |
pkill -9 -f "weed server" || true
rm -rf "$WORK_DIR" || true
-111
View File
@@ -439,117 +439,6 @@ jobs:
path: test/s3tables/catalog_clickhouse/test-output.log
retention-days: 3
olake-iceberg-catalog-tests:
name: OLake Iceberg Catalog Integration Tests (${{ matrix.tag }})
runs-on: ubuntu-22.04
timeout-minutes: 30
strategy:
fail-fast: false
matrix:
include:
# Pinned baseline, and latest so new OLake releases are exercised
# without a code change. OLake's Iceberg writer is a Java sidecar
# whose Iceberg version moves independently of the Go release, so
# the latest leg is the one that catches library drift.
- olake-image: olakego/source-postgres:v0.10.1
tag: "v0.10.1"
- olake-image: olakego/source-postgres:latest
tag: latest
steps:
- name: Check out code
uses: actions/checkout@v7
- name: Set up Go
uses: actions/setup-go@v7
with:
go-version-file: 'go.mod'
id: go
- name: Configure Docker Hub mirror
run: |
echo '{"registry-mirrors": ["https://mirror.gcr.io"]}' | sudo tee /etc/docker/daemon.json
sudo systemctl restart docker
- name: Pre-pull images
run: |
pull() { for i in 1 2 3; do docker pull "$1" && return 0; sleep 15; done; return 1; }
pull ${{ matrix.olake-image }}
pull postgres:16
pull python:3.11-slim
- name: Run go mod tidy
run: go mod tidy
- name: Install SeaweedFS
run: |
go install -buildvcs=false ./weed
- name: Run OLake Iceberg Catalog Integration Tests
timeout-minutes: 25
working-directory: test/s3tables/catalog_olake
env:
OLAKE_IMAGE: ${{ matrix.olake-image }}
run: |
set -x
set -o pipefail
echo "=== System Information ==="
uname -a
free -h
df -h
docker info
echo "=== Starting OLake Iceberg Catalog Tests ==="
go test -v -timeout 20m . 2>&1 | tee test-output.log || {
echo "OLake Iceberg catalog integration tests failed"
exit 1
}
# The suite skips itself when Docker is unavailable, so a green job is not
# by itself evidence that anything ran. Assert execution explicitly.
- name: Assert the suite actually ran
working-directory: test/s3tables/catalog_olake
run: |
log=test-output.log
if [ ! -f "$log" ]; then
echo "::error::no test-output.log; the suite did not run"
exit 1
fi
passes=$(grep -c '^--- PASS' "$log" || true)
skips=$(grep -c '^--- SKIP' "$log" || true)
echo "top-level PASS=$passes SKIP=$skips"
if [ "$skips" -gt 0 ]; then
echo "::error::the OLake suite skipped $skips top-level test(s); the environment it needs was not provisioned, so this job proves nothing"
grep '^--- SKIP' "$log" | head -20
exit 1
fi
if [ "$passes" -lt 1 ]; then
echo "::error::the OLake suite recorded no passing top-level test"
exit 1
fi
- name: Show test output on failure
if: failure()
working-directory: test/s3tables/catalog_olake
run: |
echo "=== Test Output ==="
if [ -f test-output.log ]; then
tail -200 test-output.log
fi
echo "=== Process information ==="
ps aux | grep -E "(weed|test|docker|olake|postgres)" || true
echo "=== Containers ==="
docker ps -a | head -30 || true
- name: Upload test logs on failure
if: failure()
uses: actions/upload-artifact@v7
with:
name: olake-iceberg-catalog-test-logs-${{ matrix.tag }}
path: test/s3tables/catalog_olake/test-output.log
retention-days: 3
polaris-integration-tests:
name: Polaris Integration Tests
runs-on: ubuntu-22.04
-4
View File
@@ -289,7 +289,6 @@ jobs:
s3tests/functional/test_s3.py::test_object_write_check_etag \
s3tests/functional/test_s3.py::test_object_write_cache_control \
s3tests/functional/test_s3.py::test_object_write_expires \
s3tests/functional/test_s3.py::test_object_content_encoding_aws_chunked \
s3tests/functional/test_s3.py::test_object_write_read_update_read_delete \
s3tests/functional/test_s3.py::test_object_metadata_replaced_on_put \
s3tests/functional/test_s3.py::test_object_write_file \
@@ -312,7 +311,6 @@ jobs:
s3tests/functional/test_s3.py::test_get_object_ifmodifiedsince_good \
s3tests/functional/test_s3.py::test_get_object_ifmodifiedsince_failed \
s3tests/functional/test_s3.py::test_get_object_ifunmodifiedsince_failed \
s3tests/functional/test_s3.py::test_get_checksum_object_attributes \
s3tests/functional/test_s3.py::test_bucket_head \
s3tests/functional/test_s3.py::test_bucket_head_notexist \
s3tests/functional/test_s3.py::test_object_raw_authenticated \
@@ -1151,7 +1149,6 @@ jobs:
s3tests/functional/test_s3.py::test_object_write_check_etag \
s3tests/functional/test_s3.py::test_object_write_cache_control \
s3tests/functional/test_s3.py::test_object_write_expires \
s3tests/functional/test_s3.py::test_object_content_encoding_aws_chunked \
s3tests/functional/test_s3.py::test_object_write_read_update_read_delete \
s3tests/functional/test_s3.py::test_object_metadata_replaced_on_put \
s3tests/functional/test_s3.py::test_object_write_file \
@@ -1174,7 +1171,6 @@ jobs:
s3tests/functional/test_s3.py::test_get_object_ifmodifiedsince_good \
s3tests/functional/test_s3.py::test_get_object_ifmodifiedsince_failed \
s3tests/functional/test_s3.py::test_get_object_ifunmodifiedsince_failed \
s3tests/functional/test_s3.py::test_get_checksum_object_attributes \
s3tests/functional/test_s3.py::test_bucket_head \
s3tests/functional/test_s3.py::test_bucket_head_notexist \
s3tests/functional/test_s3.py::test_object_raw_authenticated \
+3 -4
View File
@@ -17,11 +17,10 @@ SeaweedFS is a simple and highly scalable distributed file system. There are two
1. to store billions of files!
2. to serve the files fast!
One `weed` binary serves an S3 object store, a POSIX file system, and a lakehouse with S3 Tables, all over the same data. Each blob is one disk read away, capacity grows by starting another volume server, and cloud storage can be cached or tiered transparently. Both read and write operations have O(1) complexity and can run at the full speed supported by the underlying hardware.
One `weed` binary serves an S3 object store, a POSIX file system, and a lakehouse with S3 Tables, all over the same data. Each blob is one disk read away, capacity grows by starting another volume server, and cloud storage can be cached or tiered transparently.
- [Download Binaries for different platforms](https://github.com/seaweedfs/seaweedfs/releases/latest)
- [Wiki Documentation](https://github.com/seaweedfs/seaweedfs/wiki)
- [HTTP REST API](REST_API.md) for the filer, master, and volume servers
- Community: [Slack](https://join.slack.com/t/seaweedfs/shared_invite/enQtMzI4MTMwMjU2MzA3LTEyYzZmZWYzOGQ3MDJlZWMzYmI0OTE4OTJiZjJjODBmMzUxNmYwODg0YjY3MTNlMjBmZDQ1NzQ5NDJhZWI2ZmY), [Twitter](https://twitter.com/SeaweedFS), [Telegram](https://t.me/Seaweedfs), [Reddit](https://www.reddit.com/r/SeaweedFS/), [Mailing List](https://groups.google.com/d/forum/seaweedfs)
- [SeaweedFS White Paper](https://github.com/seaweedfs/seaweedfs/wiki/SeaweedFS_Architecture.pdf) and introduction slides: [2025.5](https://docs.google.com/presentation/d/1tdkp45J01oRV68dIm4yoTXKJDof-EhainlA0LMXexQE/edit?usp=sharing), [2021.5](https://docs.google.com/presentation/d/1DcxKWlINc-HNCjhYeERkpGXXm6nTCES8mi2W5G0Z4Ts/edit?usp=sharing), [2019.3](https://www.slideshare.net/chrislusf/seaweedfs-introduction)
@@ -82,8 +81,6 @@ AWS_ACCESS_KEY_ID=admin AWS_SECRET_ACCESS_KEY=secret \
The same process also runs the master, a volume server, the filer, WebDAV, the Iceberg REST catalog, and the Admin UI. Add `S3_TABLE_BUCKET=warehouse` to also create an Iceberg table bucket, or `warehouse:LANCE` for a Lance one. Drop the AWS keys to run without authentication for development.
Without Admin authentication or mTLS, `weed mini` binds the Admin UI/API and its worker gRPC control plane to loopback rather than `-ip.bind`. Set `WEED_ADMIN_PASSWORD` or configure `https.admin` mTLS to keep the network bind. Remote workers must opt in with `-admin.worker.ip=<address>` and should configure `grpc.admin` mTLS. `-admin.allowInsecureBind` restores the legacy unauthenticated network bind and should only be used on an isolated network.
> macOS: if the binary is quarantined, run `xattr -d com.apple.quarantine ./weed` first.
`weed mini` is auto-tuned for one node and is fine for single-node production, such as an S3 gateway that issues presigned URLs. See [Quick Start with weed mini][WeedMini].
@@ -403,6 +400,8 @@ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
The text of this page is available for modification and reuse under the terms of the Creative Commons Attribution-Sharealike 3.0 Unported License and the GNU Free Documentation License (unversioned, with no invariant sections, front-cover texts, or back-cover texts).
[Back to TOC](#table-of-contents)
# Sponsors #
-344
View File
@@ -1,344 +0,0 @@
# SeaweedFS HTTP REST API
SeaweedFS exposes three HTTP surfaces:
| Service | Default port | Addressing |
|---------|--------------|------------|
| Filer | 8888 | File system paths (`/dir/name`) |
| Master | 9333 | File id assignment and cluster topology |
| Volume server | 8080 | File content by file id (`vid,fid`) |
Most clients only need the filer API (paths) or the S3 API. The master and
volume APIs are the lower-level blob store interface.
Conventions applying to all three:
- Responses are JSON unless noted otherwise. Append `&pretty=y` to pretty-print.
- A file id (`fid`) has the form `volumeId,fileKeyCookie`, e.g. `3,01637037d6`.
An optional suffix selects a reserved id from a `count` assignment
(`3,01637037d6_1`, `_2`, ...), and an optional extension
(`3,01637037d6.jpg`) sets the content type on reads.
- `replication` is a 3-digit replica placement `xyz`: `x` copies in other
data centers, `y` on other racks in the same data center, `z` on other
volume servers on the same rack. `000` = no replication, `001` = one copy
on the same rack, `010` = one copy on a different rack, `100` = one copy in
another data center, `200` = two copies in two other data centers, `110` =
one copy in another data center plus one on another rack.
- `ttl` units: `m` minute, `h` hour, `d` day, `w` week, `M` month, `y` year.
## Filer API (port 8888)
The filer presents a POSIX-like namespace over the volume servers.
### Upload a file
```bash
# PUT the raw body to the target path
curl -T /home/chris/myphoto.jpg "http://localhost:8888/dir/myphoto.jpg"
# or POST as multipart form (the part filename becomes the entry name)
curl -F file=@/home/chris/myphoto.jpg "http://localhost:8888/dir/"
```
Response `201 Created`:
```json
{"name":"myphoto.jpg","size":43234,"eTag":"0x6c656...","mtime":"...","chunks":[...]}
```
Query parameters:
| Parameter | Description | Default |
|-----------|-------------|---------|
| `collection` | collection name | empty |
| `replication` | replica placement code | filer default |
| `ttl` | file expiration, e.g. `3d` | never |
| `disk` | disk type to store on | filer default |
| `fsync` | `true` fsyncs on the volume server | false |
| `dataCenter` | preferred data center | empty |
| `rack` | preferred rack | empty |
| `dataNode` | preferred volume server | empty |
| `saveInside` | store small content inside the metadata instead of a volume | false |
| `maxMB` | split the upload into chunks of this many MB | filer `-maxMB` |
| `mode` | unix permission bits, e.g. `0644` | `0660` |
| `op` | `append` appends to an existing file | overwrite |
| `skipCheckParentDir` | `true` skips the parent-directory existence check | false |
### Create a directory
```bash
curl -X POST "http://localhost:8888/dir/newdir/"
```
A POST to a path ending in `/` with no content creates the directory,
including missing parents.
### Read a file
```bash
curl "http://localhost:8888/dir/myphoto.jpg"
```
Supports `Range` requests (`Accept-Ranges: bytes`), `ETag`, and the
`If-None-Match` / `If-Modified-Since` conditional headers. `HEAD` returns
headers only. Entry headers stored as extended attributes are echoed back,
minus internal `Seaweed-` and `xattr-` keys.
Entry metadata instead of content:
```bash
curl "http://localhost:8888/dir/myphoto.jpg?metadata=true"
```
`metadata=true&resolveManifest=true` additionally resolves chunked-manifest
entries into their real chunk list.
### List a directory
```bash
curl -H "Accept: application/json" "http://localhost:8888/dir/?limit=10&lastFileName=a.jpg"
```
| Parameter | Description | Default |
|-----------|-------------|---------|
| `limit` | max entries per page | filer `-dirListLimit` |
| `lastFileName` | resume listing after this entry name | empty |
| `namePattern` | include only names matching the wildcard | empty |
| `namePatternExclude` | exclude names matching the wildcard | empty |
The JSON response carries `Path`, `Entries`, `Limit`, `LastFileName`,
`ShouldDisplayLoadMore`, and `EmptyFolder`. Without the `Accept` header the
filer renders its HTML browser.
### Move and copy
```bash
curl -X POST "http://localhost:8888/dir/newname.jpg?mv.from=/dir/myphoto.jpg"
curl -X POST "http://localhost:8888/dir/copy.jpg?cp.from=/dir/myphoto.jpg"
```
`mv.from` renames or moves the source to the request path (`204 No Content`).
`cp.from` copies it.
### Append
```bash
curl -T chunk2.bin "http://localhost:8888/dir/file.bin?op=append"
```
### Delete
```bash
curl -X DELETE "http://localhost:8888/dir/myphoto.jpg"
curl -X DELETE "http://localhost:8888/dir/?recursive=true"
```
| Parameter | Description | Default |
|-----------|-------------|---------|
| `recursive` | delete a non-empty directory tree | false; when the filer runs with `filer.options.recursive_delete=true`, deletes are recursive unless `recursive=false` |
| `ignoreRecursiveError` | keep deleting remaining entries after an error | false |
| `skipChunkDeletion` | remove only the metadata, keep volume data | false |
### Tagging
Tags are carried as `Seaweed-`-prefixed request headers, not query
parameters; `?tagging` selects the tagging handler and `?tagging=K1,K2`
lists the keys to remove. Header names are canonicalized on write
(`Seaweed-k1` is stored as `Seaweed-K1`), and the delete list is matched
case-sensitively against the stored names.
```bash
curl -X PUT -H "Seaweed-k1: v1" -H "Seaweed-k2: v2" "http://localhost:8888/dir/file.jpg?tagging"
curl -X DELETE "http://localhost:8888/dir/file.jpg?tagging=K1,K2"
```
### Read by file id
```bash
curl "http://localhost:8888/?proxyChunkId=3,01637037d6"
```
The filer proxies the chunk read to the right volume server, so only the
filer port needs to be exposed.
### Resumable uploads
The filer serves the [TUS protocol](https://tus.io/) for resumable uploads
(`POST`, `PATCH`, `HEAD` on upload URLs). It is enabled by default at
`/.tus`; `-tusBasePath` changes the endpoint base path.
### Health
`GET /healthz` and `GET /readyz` return `200 OK`.
## Master API (port 9333)
Write-affecting endpoints are automatically proxied to the current leader, so
any master in the quorum can serve them.
### Assign a file id
```bash
curl "http://localhost:9333/dir/assign?count=1&replication=001&collection=turbo&dataCenter=dc1&ttl=3d&disk=ssd"
{"count":1,"fid":"3,01637037d6","url":"127.0.0.1:8080","publicUrl":"localhost:8080"}
```
Upload the file content to `http://<url>/<fid>` afterwards. With `count>1`,
use `<fid>_1`, `<fid>_2`, ... for the additional ids.
| Parameter | Description | Default |
|-----------|-------------|---------|
| `count` | file ids to reserve | 1 |
| `collection` | collection name | empty |
| `dataCenter` | preferred data center | empty |
| `rack` | preferred rack | empty |
| `dataNode` | preferred volume server | empty |
| `replication` | replica placement | master `-defaultReplication` |
| `ttl` | file expiration, e.g. `3d` | never |
| `disk` | disk type | empty |
| `dataSize` | expected file size in bytes | 0 |
| `preallocate` | bytes to preallocate for new volumes | master `-volumePreallocate` |
| `writableVolumeCount` | grow this many volumes when none are writable | master default |
| `memoryMapMaxSizeMb` | memory-mapped file size (Windows) | 0 |
### Look up a volume or file id
```bash
curl "http://localhost:9333/dir/lookup?volumeId=3"
{"locations":[{"url":"localhost:8080","publicUrl":"localhost:8080"}]}
```
| Parameter | Description | Default |
|-----------|-------------|---------|
| `volumeId` | volume id; a full `vid,fid` is accepted too | required |
| `fileId` | like `volumeId`, but also returns a write JWT when security is on | empty |
| `collection` | speeds up the lookup | empty |
| `read` | `yes` generates a read JWT instead of a write JWT | empty |
### Store a file in one call
```bash
curl -F file=@/home/chris/report.pdf "http://localhost:9333/submit?collection=turbo&replication=001"
{"fileName":"report.pdf","fid":"3,01637037d6","fileUrl":"localhost:8080/3,01637037d6","size":43234,"eTag":"0x6c656..."}
```
`POST /submit` accepts multipart file data plus the `dir/assign` placement
parameters (`count`, `collection`, `dataCenter`, `rack`, `replication`,
`ttl`, `disk`), assigns a file id, uploads to the volume server, and returns
the result.
### Redirect to a file
```bash
curl -v "http://localhost:9333/3,01637037d6"
```
`GET /{fileId}` answers `308 Permanent Redirect` to a volume server holding
the file, preserving the query string (e.g. image-resize parameters).
### Cluster status
```bash
curl "http://localhost:9333/dir/status?pretty=y" # full topology tree
curl "http://localhost:9333/vol/status?pretty=y" # every volume on every node
curl "http://localhost:9333/collection/info?collection=turbo"
curl "http://localhost:9333/collection/info?collection=turbo&detail=true"
```
`collection/info` returns aggregated `TotalSize`, `FileCount`, `UsedSize`,
`VolumeCount`; `detail=true` splits them per volume layout.
### Grow volumes
```bash
curl "http://localhost:9333/vol/grow?count=4&replication=001&collection=turbo&ttl=5d&disk=ssd&dataCenter=dc1&rack=rack1"
{"count":4}
```
`count` is required; the placement parameters match `dir/assign`. One volume
serves one write at a time, so pre-allocated volumes raise write concurrency.
### Vacuum deleted space
```bash
curl "http://localhost:9333/vol/vacuum?garbageThreshold=0.4"
```
| Parameter | Description | Default |
|-----------|-------------|---------|
| `garbageThreshold` | minimum deleted-bytes ratio before a volume is compacted | master `-garbageThreshold` (0.3) |
Vacuuming makes a volume read-only, copies live needles to a new volume, and
swaps it in.
### Delete a collection
```bash
curl "http://localhost:9333/col/delete?collection=benchmark"
```
Deletes all volumes of the collection, including erasure-coded shards.
`204 No Content` on success.
### Health
```bash
curl -I "http://localhost:9333/healthz" # liveness
curl -I "http://localhost:9333/readyz" # readiness
curl "http://localhost:9333/" # web UI
```
## Volume server API (port 8080)
The volume server stores file content by file id. Clients normally get the
volume URL from `dir/assign` or `dir/lookup`.
### Upload
```bash
curl -F file=@/home/chris/myphoto.jpg "http://127.0.0.1:8080/3,01637037d6"
{"name":"myphoto.jpg","size":43234,"eTag":"0x6c656...","mime":"image/jpeg","contentMd5":"..."}
```
PUT or POST the body (or a multipart `file` part) to `/{vid},{fid}`.
`204 No Content` is returned when the content is unchanged. `?ts=<unix>`
sets the stored modification time.
### Read
```bash
curl "http://127.0.0.1:8080/3,01637037d6"
curl "http://127.0.0.1:8080/3,01637037d6.jpg" # sets Content-Type from the extension
```
Supports `Range` and `HEAD`. Image files can be resized server-side:
| Parameter | Description |
|-----------|-------------|
| `width`, `height` | resize bounds in pixels |
| `mode` | `fit` (contain) or `fill` (cover); omitted resizes to `width`/`height` |
| `crop_x1`, `crop_y1`, `crop_x2`, `crop_y2` | explicit crop rectangle |
| `cm` | `false` returns the chunk-manifest blob instead of resolving it |
| `readDeleted` | `true` reads soft-deleted needles |
| `collection` | passed through redirects for the right volume |
### Delete
```bash
curl -X DELETE "http://127.0.0.1:8080/3,01637037d6"
{"size":43234}
```
`?ts=<unix>` sets the deletion timestamp. Replicated volumes propagate the
delete to every replica.
### Status
```bash
curl "http://localhost:8080/status?pretty=y" # disk and volume inventory
curl -I "http://localhost:8080/healthz" # liveness/readiness
```
`OPTIONS` preflights answer CORS headers. When `-port.public` differs from
`-port`, the volume server opens a separate read-only public listener on
that port; `-publicUrl` sets the address it advertises to clients.
-6
View File
@@ -14,9 +14,6 @@ RUN cd /go/src/github.com/seaweedfs/seaweedfs && \
git checkout $BRANCH) || \
(echo "ERROR: Branch/commit $BRANCH not found in repository" && \
echo "Available branches:" && git branch -a && exit 1))
# seaweed-common only exists on revisions that have it; a BRANCH predating it
# still needs the directory so the COPY into rust_builder below never fails.
RUN mkdir -p /go/src/github.com/seaweedfs/seaweedfs/seaweed-common
ARG TARGETOS TARGETARCH TARGETVARIANT
RUN cd /go/src/github.com/seaweedfs/seaweedfs/weed \
&& export LDFLAGS="-X github.com/seaweedfs/seaweedfs/weed/util/version.COMMIT=$(git rev-parse --short HEAD)" \
@@ -34,9 +31,6 @@ ARG TAGS
COPY weed-volume-prebuilt/ /prebuilt/
COPY weed-worker-prebuilt/ /prebuilt-worker/
COPY --from=builder /go/src/github.com/seaweedfs/seaweedfs/seaweed-volume /build/seaweed-volume
# seaweed-common is a path dependency of seaweed-volume that lives beside it,
# so the source build below needs it in the same relative position.
COPY --from=builder /go/src/github.com/seaweedfs/seaweedfs/seaweed-common /build/seaweed-common
COPY --from=builder /go/src/github.com/seaweedfs/seaweedfs/weed /build/weed
WORKDIR /build/seaweed-volume
RUN if [ -f "/prebuilt/weed-volume-${TARGETARCH}" ]; then \
+2 -5
View File
@@ -1,14 +1,11 @@
FROM alpine:latest
# Install required packages
RUN apk upgrade --no-cache && \
apk add --no-cache \
RUN apk add --no-cache \
ca-certificates \
fuse \
curl \
jq \
libcrypto3 \
libssl3
jq
# Copy our locally built binary
COPY weed-local /usr/bin/weed
-2
View File
@@ -30,5 +30,3 @@ sleep_minutes = 17 # sleep minutes between each script execution
bucket = "volume_bucket" # an existing bucket
endpoint = "http://server2:8333"
storage_class = "STANDARD_IA"
# upload_concurrency = 5 # concurrent multipart part uploads per volume (volume.tier.upload -concurrent overrides)
# download_concurrency = 5 # concurrent multipart part downloads per volume (volume.tier.download -concurrent overrides)
+60 -52
View File
@@ -1,6 +1,6 @@
module github.com/seaweedfs/seaweedfs
go 1.26.6
go 1.26.0
require (
cloud.google.com/go v0.123.0 // indirect
@@ -14,7 +14,7 @@ require (
github.com/coreos/go-semver v0.3.1 // indirect
github.com/coreos/go-systemd/v22 v22.7.0 // indirect
github.com/davecgh/go-spew v1.1.2-0.20180830191138-d8f796af33cc // indirect
github.com/dustin/go-humanize v1.1.0
github.com/dustin/go-humanize v1.0.1
github.com/eapache/go-resiliency v1.6.0 // indirect
github.com/eapache/go-xerial-snappy v0.0.0-20230731223053-c322873962e3 // indirect
github.com/eapache/queue v1.1.0 // indirect
@@ -60,7 +60,7 @@ require (
github.com/pquerna/cachecontrol v0.2.0
github.com/prometheus/client_golang v1.24.1
github.com/prometheus/client_model v0.6.3
github.com/prometheus/common v0.71.0 // indirect
github.com/prometheus/common v0.70.1
github.com/prometheus/procfs v0.22.0
github.com/rcrowley/go-metrics v0.0.0-20201227073835-cf1acfcdf475 // indirect
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec // indirect
@@ -70,10 +70,10 @@ require (
github.com/spf13/afero v1.15.0 // indirect
github.com/spf13/cast v1.10.0 // indirect
github.com/spf13/viper v1.21.0
github.com/stretchr/testify v1.12.1
github.com/stretchr/testify v1.11.1
github.com/stvp/tempredis v0.0.0-20181119212430-b82af8480203
github.com/syndtr/goleveldb v1.0.1-0.20190318030020-c3a204f8e965
github.com/tidwall/gjson v1.19.0
github.com/tidwall/gjson v1.18.0
github.com/tidwall/match v1.2.0
github.com/tidwall/pretty v1.2.0 // indirect
github.com/tsuna/gohbase v0.0.0-20201125011725-348991136365
@@ -90,18 +90,18 @@ require (
gocloud.dev v0.46.0
gocloud.dev/pubsub/natspubsub v0.46.0
gocloud.dev/pubsub/rabbitpubsub v0.46.0
golang.org/x/crypto v0.57.0
golang.org/x/crypto v0.56.0
golang.org/x/exp v0.0.0-20260709172345-9ea1abe57597
golang.org/x/image v0.46.0
golang.org/x/net v0.58.0
golang.org/x/oauth2 v0.37.0
golang.org/x/oauth2 v0.36.0
golang.org/x/sys v0.48.0
golang.org/x/text v0.42.0 // indirect
golang.org/x/tools v0.49.0 // indirect
golang.org/x/xerrors v0.0.0-20240903120638-7835f813f4da // indirect
google.golang.org/api v0.297.0
google.golang.org/genproto v0.0.0-20260715232425-e75dac1f907d // indirect
google.golang.org/grpc v1.85.0-dev.0.20260915183914-4e49413dcab7
google.golang.org/grpc v1.85.0-dev
google.golang.org/protobuf v1.36.12
gopkg.in/inf.v0 v0.9.1 // indirect
modernc.org/b v1.0.0 // indirect
@@ -111,21 +111,21 @@ require (
)
require (
cloud.google.com/go/kms v1.35.0
cloud.google.com/go/kms v1.33.0
github.com/Azure/azure-sdk-for-go/sdk/keyvault/azkeys v0.10.0
github.com/DATA-DOG/go-sqlmock v1.5.2
github.com/Jille/raft-grpc-transport v1.6.1
github.com/ThreeDotsLabs/watermill v1.5.2
github.com/a-h/templ v0.3.1020
github.com/apache/cassandra-gocql-driver/v2 v2.1.2
github.com/apache/iceberg-go v0.7.0
github.com/apache/iceberg-go v0.6.1-0.20260817192109-c2105090c9e2
github.com/apple/foundationdb/bindings/go v0.0.0-20250911184653-27f7192f47c3
github.com/arangodb/go-driver v1.6.9
github.com/armon/go-metrics v0.4.1
github.com/aws/aws-sdk-go-v2 v1.47.0
github.com/aws/aws-sdk-go-v2/config v1.33.1
github.com/aws/aws-sdk-go-v2/config v1.32.35
github.com/aws/aws-sdk-go-v2/credentials v1.20.4
github.com/aws/aws-sdk-go-v2/service/s3 v1.109.1
github.com/aws/aws-sdk-go-v2/service/s3 v1.107.3
github.com/cespare/xxhash/v2 v2.3.0
github.com/cognusion/imaging v1.0.4
github.com/fluent/fluent-logger-golang v1.10.1
@@ -134,8 +134,8 @@ require (
github.com/golang-jwt/jwt/v5 v5.3.1
github.com/google/flatbuffers/go v0.0.0-20230108230133-3b8644d32c50
github.com/hashicorp/golang-lru/v2 v2.0.7
github.com/hashicorp/raft v1.8.0
github.com/hashicorp/raft-boltdb/v2 v2.4.2
github.com/hashicorp/raft v1.7.3
github.com/hashicorp/raft-boltdb/v2 v2.3.1
github.com/hashicorp/vault/api v1.23.0
github.com/jhump/protoreflect v1.18.0
github.com/linkedin/goavro/v2 v2.15.0
@@ -143,7 +143,7 @@ require (
github.com/orcaman/concurrent-map/v2 v2.0.1
github.com/parquet-go/parquet-go v0.32.0
github.com/pkg/sftp v1.13.11
github.com/rabbitmq/amqp091-go v1.15.0
github.com/rabbitmq/amqp091-go v1.14.0
github.com/rclone/rclone v1.75.1
github.com/rdleal/intervalst v1.5.0
github.com/redis/go-redis/v9 v9.22.0
@@ -151,15 +151,15 @@ require (
github.com/seaweedfs/go-fuse/v2 v2.9.4
github.com/shirou/gopsutil/v4 v4.26.7
github.com/tarantool/go-option v1.1.0
github.com/tarantool/go-tarantool/v3 v3.0.2
github.com/tarantool/go-tarantool/v3 v3.0.1
github.com/testcontainers/testcontainers-go v0.44.0
github.com/tikv/client-go/v2 v2.0.7
github.com/twmb/avro v1.9.0
github.com/twmb/avro v1.8.0
github.com/xeipuuv/gojsonschema v1.2.0
github.com/ydb-platform/ydb-go-sdk-auth-environ v0.5.2
github.com/ydb-platform/ydb-go-sdk/v3 v3.151.1
go.etcd.io/etcd/client/pkg/v3 v3.7.1
go.uber.org/atomic v1.12.0
go.uber.org/atomic v1.11.0
golang.org/x/sync v0.23.0
golang.org/x/tools/godoc v0.1.0-deprecated
google.golang.org/grpc/security/advancedtls v1.0.0
@@ -168,6 +168,9 @@ require (
require github.com/k0kubun/colorstring v0.0.0-20150214042306-9440f1994b88 // indirect
require (
atomicgo.dev/cursor v0.2.0 // indirect
atomicgo.dev/keyboard v0.2.9 // indirect
atomicgo.dev/schedule v0.1.0 // indirect
cloud.google.com/go/longrunning v1.2.0 // indirect
cloud.google.com/go/pubsub/v2 v2.6.0 // indirect
dario.cat/mergo v1.0.2 // indirect
@@ -175,12 +178,12 @@ require (
github.com/Azure/go-ansiterm v0.0.0-20250102033503-faa5f7b0171c // indirect
github.com/FilenCloudDienste/filen-sdk-go v0.0.39 // indirect
github.com/ProtonMail/gopenpgp/v3 v3.4.1 // indirect
github.com/RoaringBitmap/roaring/v2 v2.26.0 // indirect
github.com/RoaringBitmap/roaring/v2 v2.24.0 // indirect
github.com/a1ex3/zstd-seekable-format-go/pkg v0.10.0 // indirect
github.com/adrg/xdg v0.5.3 // indirect
github.com/anchore/go-lzo v0.1.1 // indirect
github.com/antlr4-go/antlr/v4 v4.13.1 // indirect
github.com/apache/arrow-go/v18 v18.8.0 // indirect
github.com/apache/arrow-go/v18 v18.7.0 // indirect
github.com/apache/thrift v0.24.0 // indirect
github.com/aws/aws-sdk-go-v2/service/signin v1.10.0 // indirect
github.com/bahlo/generic-list-go v0.2.0 // indirect
@@ -197,6 +200,7 @@ require (
github.com/cockroachdb/logtags v0.0.0-20241215232642-bb51bb14a506 // indirect
github.com/cockroachdb/redact v1.1.5 // indirect
github.com/cockroachdb/version v0.0.0-20250314144055-3860cd14adf2 // indirect
github.com/containerd/console v1.0.5 // indirect
github.com/containerd/errdefs v1.0.0 // indirect
github.com/containerd/errdefs/pkg v0.3.0 // indirect
github.com/containerd/log v0.1.0 // indirect
@@ -215,6 +219,7 @@ require (
github.com/goccy/go-yaml v1.18.0 // indirect
github.com/golang/geo v0.0.0-20210211234256-740aa86cb551 // indirect
github.com/google/go-cmp v0.7.0 // indirect
github.com/gookit/color v1.6.0 // indirect
github.com/gopherjs/gopherjs v1.17.2 // indirect
github.com/grpc-ecosystem/grpc-gateway v1.16.0 // indirect
github.com/hashicorp/go-rootcerts v1.0.2 // indirect
@@ -232,6 +237,7 @@ require (
github.com/kr/pretty v0.3.1 // indirect
github.com/kr/text v0.2.0 // indirect
github.com/lib/pq v1.12.0 // indirect
github.com/lithammer/fuzzysearch v1.1.8 // indirect
github.com/lithammer/shortuuid/v3 v3.0.7 // indirect
github.com/lpar/calendar v0.2.0 // indirect
github.com/magiconair/properties v1.8.10 // indirect
@@ -254,6 +260,7 @@ require (
github.com/petermattis/goid v0.0.0-20260113132338-7c7de50cc741 // indirect
github.com/pierrre/geohash v1.0.0 // indirect
github.com/pquerna/otp v1.5.0 // indirect
github.com/pterm/pterm v0.12.83 // indirect
github.com/quic-go/qpack v0.6.0 // indirect
github.com/rclone/Proton-API-Bridge v1.0.5 // indirect
github.com/rclone/go-proton-api v1.0.4 // indirect
@@ -274,35 +281,36 @@ require (
github.com/wk8/go-ordered-map/v2 v2.1.8 // indirect
github.com/xeipuuv/gojsonpointer v0.0.0-20190905194746-02993c407bfb // indirect
github.com/xeipuuv/gojsonreference v0.0.0-20180127040603-bd5ef7bd5415 // indirect
github.com/xo/terminfo v0.0.0-20220910002029-abceb7e1c41e // indirect
github.com/zeebo/xxh3 v1.1.0 // indirect
go.opentelemetry.io/otel/exporters/otlp/otlptrace v1.45.0 // indirect
go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracegrpc v1.45.0 // indirect
go.opentelemetry.io/otel/exporters/zipkin v1.45.0 // indirect
go.opentelemetry.io/proto/otlp v1.11.0 // indirect
go.opentelemetry.io/otel/exporters/otlp/otlptrace v1.44.0 // indirect
go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracegrpc v1.44.0 // indirect
go.opentelemetry.io/otel/exporters/zipkin v1.36.0 // indirect
go.opentelemetry.io/proto/otlp v1.10.0 // indirect
go.uber.org/mock v0.5.2 // indirect
go.yaml.in/yaml/v2 v2.4.4 // indirect
go.yaml.in/yaml/v3 v3.0.5 // indirect
go.yaml.in/yaml/v3 v3.0.4 // indirect
golang.org/x/mod v0.41.0 // indirect
gonum.org/v1/gonum v0.17.0 // indirect
)
require (
cel.dev/expr v0.25.3 // indirect
cel.dev/expr v0.25.2 // indirect
cloud.google.com/go/auth v0.23.2 // indirect
cloud.google.com/go/auth/oauth2adapt v0.2.8 // indirect
cloud.google.com/go/compute/metadata v0.9.0 // indirect
cloud.google.com/go/iam v1.12.0 // indirect
cloud.google.com/go/monitoring v1.30.0 // indirect
filippo.io/edwards25519 v1.2.0 // indirect
github.com/Azure/azure-sdk-for-go/sdk/azcore v1.23.1
github.com/Azure/azure-sdk-for-go/sdk/azidentity v1.14.1
github.com/Azure/azure-sdk-for-go/sdk/azcore v1.22.0
github.com/Azure/azure-sdk-for-go/sdk/azidentity v1.14.0
github.com/Azure/azure-sdk-for-go/sdk/internal v1.12.0 // indirect
github.com/Azure/azure-sdk-for-go/sdk/storage/azblob v1.8.0
github.com/Azure/azure-sdk-for-go/sdk/storage/azfile v1.7.0 // indirect
github.com/Azure/go-ntlmssp v0.1.1 // indirect
github.com/AzureAD/microsoft-authentication-library-for-go v1.8.0 // indirect
github.com/AzureAD/microsoft-authentication-library-for-go v1.7.2 // indirect
github.com/Files-com/files-sdk-go/v3 v3.3.194 // indirect
github.com/GoogleCloudPlatform/opentelemetry-operations-go/detectors/gcp v1.35.0 // indirect
github.com/GoogleCloudPlatform/opentelemetry-operations-go/detectors/gcp v1.34.0 // indirect
github.com/GoogleCloudPlatform/opentelemetry-operations-go/exporter/metric v0.57.0 // indirect
github.com/GoogleCloudPlatform/opentelemetry-operations-go/internal/resourcemapping v0.57.0 // indirect
github.com/IBM/go-sdk-core/v5 v5.23.1 // indirect
@@ -314,25 +322,25 @@ require (
github.com/ProtonMail/go-srp v0.0.7 // indirect
github.com/PuerkitoBio/goquery v1.12.0 // indirect
github.com/abbot/go-http-auth v0.4.0 // indirect
github.com/andybalholm/brotli v1.2.3 // indirect
github.com/andybalholm/brotli v1.2.2 // indirect
github.com/andybalholm/cascadia v1.3.4 // indirect
github.com/appscode/go-querystring v0.0.0-20170504095604-0126cfb3f1dc // indirect
github.com/arangodb/go-velocypack v0.0.0-20200318135517-5af53c29c67e // indirect
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.20 // indirect
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.18 // indirect
github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.20.0 // indirect
github.com/aws/aws-sdk-go-v2/feature/s3/manager v1.22.34 // indirect
github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.3 // indirect
github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.3 // indirect
github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.3 // indirect
github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.19 // indirect
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.11.1 // indirect
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.31 // indirect
github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.3 // indirect
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.20.1 // indirect
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.39 // indirect
github.com/aws/aws-sdk-go-v2/service/sns v1.39.14 // indirect
github.com/aws/aws-sdk-go-v2/service/sqs v1.42.24 // indirect
github.com/aws/aws-sdk-go-v2/service/sso v1.38.0 // indirect
github.com/aws/aws-sdk-go-v2/service/ssooidc v1.43.0 // indirect
github.com/aws/aws-sdk-go-v2/service/sts v1.51.0
github.com/aws/aws-sdk-go-v2/service/sts v1.50.0
github.com/aws/smithy-go v1.28.1
github.com/boltdb/bolt v1.3.1 // indirect
github.com/bradenaw/juniper v0.15.3 // indirect
@@ -355,9 +363,9 @@ require (
github.com/elastic/gosigar v0.14.3 // indirect
github.com/emersion/go-message v0.18.2 // indirect
github.com/emersion/go-vcard v0.0.0-20260618161152-d854b7e0e2d3 // indirect
github.com/envoyproxy/go-control-plane/envoy v1.39.1-0.20260819172001-e6e3fd93e4be // indirect
github.com/envoyproxy/go-control-plane/envoy v1.37.0 // indirect
github.com/envoyproxy/protoc-gen-validate v1.3.3 // indirect
github.com/fatih/color v1.19.0 // indirect
github.com/fatih/color v1.18.0 // indirect
github.com/felixge/httpsnoop v1.1.0 // indirect
github.com/flynn/noise v1.1.0 // indirect
github.com/gabriel-vasile/mimetype v1.4.13 // indirect
@@ -380,7 +388,7 @@ require (
github.com/gogo/protobuf v1.3.2 // indirect
github.com/golang-jwt/jwt/v4 v4.5.2 // indirect
github.com/google/s2a-go v0.1.9 // indirect
github.com/googleapis/enterprise-certificate-proxy v0.3.21 // indirect
github.com/googleapis/enterprise-certificate-proxy v0.3.20 // indirect
github.com/gorilla/schema v1.4.1 // indirect
github.com/gorilla/securecookie v1.1.2 // indirect
github.com/gorilla/sessions v1.4.0
@@ -389,10 +397,10 @@ require (
github.com/hashicorp/go-cleanhttp v0.5.2 // indirect
github.com/hashicorp/go-hclog v1.6.3 // indirect
github.com/hashicorp/go-immutable-radix v1.3.1 // indirect
github.com/hashicorp/go-metrics v0.7.0 // indirect
github.com/hashicorp/go-msgpack/v2 v2.1.5 // indirect
github.com/hashicorp/go-metrics v0.5.4 // indirect
github.com/hashicorp/go-msgpack/v2 v2.1.2 // indirect
github.com/hashicorp/go-retryablehttp v0.7.8 // indirect
github.com/hashicorp/golang-lru v1.0.2 // indirect
github.com/hashicorp/golang-lru v0.6.0 // indirect
github.com/jcmturner/aescts/v2 v2.0.0 // indirect
github.com/jcmturner/dnsutils/v2 v2.0.0 // indirect
github.com/jcmturner/goidentity/v6 v6.0.1 // indirect
@@ -430,7 +438,7 @@ require (
github.com/oracle/oci-go-sdk/v65 v65.121.0 // indirect
github.com/panjf2000/ants/v2 v2.12.1 // indirect
github.com/patrickmn/go-cache v2.1.0+incompatible // indirect
github.com/pelletier/go-toml/v2 v2.4.3 // indirect
github.com/pelletier/go-toml/v2 v2.4.1 // indirect
github.com/pengsrc/go-shared v0.2.1-0.20190131101655-1999055a4a14 // indirect
github.com/philhofer/fwd v1.2.0 // indirect
github.com/pierrec/lz4/v4 v4.1.29
@@ -481,19 +489,19 @@ require (
go.etcd.io/bbolt v1.5.0 // indirect
go.etcd.io/etcd/api/v3 v3.7.1 // indirect
go.opentelemetry.io/auto/sdk v1.2.1 // indirect
go.opentelemetry.io/contrib/detectors/gcp v1.45.0 // indirect
go.opentelemetry.io/contrib/detectors/gcp v1.44.0 // indirect
go.opentelemetry.io/contrib/instrumentation/google.golang.org/grpc/otelgrpc v0.69.0 // indirect
go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.70.0 // indirect
go.opentelemetry.io/otel v1.46.0 // indirect
go.opentelemetry.io/otel/metric v1.46.0 // indirect
go.opentelemetry.io/otel/sdk v1.46.0 // indirect
go.opentelemetry.io/otel/sdk/metric v1.46.0 // indirect
go.opentelemetry.io/otel/trace v1.46.0 // indirect
go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.69.0 // indirect
go.opentelemetry.io/otel v1.45.0 // indirect
go.opentelemetry.io/otel/metric v1.45.0 // indirect
go.opentelemetry.io/otel/sdk v1.45.0 // indirect
go.opentelemetry.io/otel/sdk/metric v1.45.0 // indirect
go.opentelemetry.io/otel/trace v1.45.0 // indirect
go.uber.org/multierr v1.11.0 // indirect
go.uber.org/zap v1.27.1 // indirect
golang.org/x/term v0.46.0
golang.org/x/time v0.16.0
google.golang.org/genproto/googleapis/api v0.0.0-20260817212433-ac3dfec99bb1 // indirect
golang.org/x/term v0.45.0
golang.org/x/time v0.15.0
google.golang.org/genproto/googleapis/api v0.0.0-20260715232425-e75dac1f907d // indirect
google.golang.org/genproto/googleapis/rpc v0.0.0-20260819154853-08b0e4226688 // indirect
gopkg.in/natefinch/lumberjack.v2 v2.2.1 // indirect
gopkg.in/validator.v2 v2.0.1 // indirect
+179 -112
View File
@@ -1,5 +1,13 @@
cel.dev/expr v0.25.3 h1:A2jO8jwOugrrovveCWfj0KEZOfqiLgAcwjpHPhzIGw0=
cel.dev/expr v0.25.3/go.mod h1:hrXvqGP6G6gyx8UAHSHJ5RGk//1Oj5nXQ2NI02Nrsg4=
atomicgo.dev/assert v0.0.2 h1:FiKeMiZSgRrZsPo9qn/7vmr7mCsh5SZyXY4YGYiYwrg=
atomicgo.dev/assert v0.0.2/go.mod h1:ut4NcI3QDdJtlmAxQULOmA13Gz6e2DWbSAS8RUOmNYQ=
atomicgo.dev/cursor v0.2.0 h1:H6XN5alUJ52FZZUkI7AlJbUc1aW38GWZalpYRPpoPOw=
atomicgo.dev/cursor v0.2.0/go.mod h1:Lr4ZJB3U7DfPPOkbH7/6TOtJ4vFGHlgj1nc+n900IpU=
atomicgo.dev/keyboard v0.2.9 h1:tOsIid3nlPLZ3lwgG8KZMp/SFmr7P0ssEN5JUsm78K8=
atomicgo.dev/keyboard v0.2.9/go.mod h1:BC4w9g00XkxH/f1HXhW2sXmJFOCWbKn9xrOunSFtExQ=
atomicgo.dev/schedule v0.1.0 h1:nTthAbhZS5YZmgYbb2+DH8uQIZcTlIrd4eYr3UQxEjs=
atomicgo.dev/schedule v0.1.0/go.mod h1:xeUa3oAkiuHYh8bKiQBRojqAMq3PXXbJujjb0hw8pEU=
cel.dev/expr v0.25.2 h1:K6j46C81hXtZQfuX60cVWQFBJahKSE2gfRbNuvr5bFs=
cel.dev/expr v0.25.2/go.mod h1:hrXvqGP6G6gyx8UAHSHJ5RGk//1Oj5nXQ2NI02Nrsg4=
cloud.google.com/go v0.26.0/go.mod h1:aQUYkXzVsufM+DwF1aE+0xfcU+56JwCaLick0ClmMTw=
cloud.google.com/go v0.34.0/go.mod h1:aQUYkXzVsufM+DwF1aE+0xfcU+56JwCaLick0ClmMTw=
cloud.google.com/go v0.38.0/go.mod h1:990N+gfupTy94rShfmMCWGDn0LpTmnzTp2qbd1dvSRU=
@@ -290,8 +298,8 @@ cloud.google.com/go/kms v1.4.0/go.mod h1:fajBHndQ+6ubNw6Ss2sSd+SWvjL26RNo/dr7uxs
cloud.google.com/go/kms v1.5.0/go.mod h1:QJS2YY0eJGBg3mnDfuaCyLauWwBJiHRboYxJ++1xJNg=
cloud.google.com/go/kms v1.6.0/go.mod h1:Jjy850yySiasBUDi6KFUwUv2n1+o7QZFyuUJg6OgjA0=
cloud.google.com/go/kms v1.9.0/go.mod h1:qb1tPTgfF9RQP8e1wq4cLFErVuTJv7UsSC915J8dh3w=
cloud.google.com/go/kms v1.35.0 h1:nJ/ktaqspx1nPM9vIcO0SHbhqCAm8nvAxL1siuVgKm0=
cloud.google.com/go/kms v1.35.0/go.mod h1:0++71pIHvJL+GmMa8K4jOWFq7gNOX3jm2PRMSJwTKJw=
cloud.google.com/go/kms v1.33.0 h1:pG0X78m212b2pv9N4fdMoUO69LuZGQ9kSvn8sHBOFAo=
cloud.google.com/go/kms v1.33.0/go.mod h1:CSGvW6GnMQbY+1nOHcIzhMtHSbExXlOmCKjWtYVjcpA=
cloud.google.com/go/language v1.4.0/go.mod h1:F9dRpNFQmJbkaop6g0JhSBXCNlO90e1KWx5iDdxbWic=
cloud.google.com/go/language v1.6.0/go.mod h1:6dJ8t3B+lUYfStgls25GusK04NLh3eDLQnWM3mdEbhI=
cloud.google.com/go/language v1.7.0/go.mod h1:DJ6dYN/W+SQOjF8e1hLQXMF21AkH2w9wiPzPCJa2MIE=
@@ -545,10 +553,10 @@ gioui.org v0.0.0-20210308172011-57750fc8a0a6/go.mod h1:RSH6KIUZ0p2xy5zHDxgAM4zum
git.sr.ht/~sbinet/gg v0.3.1/go.mod h1:KGYtlADtqsqANL9ueOFkWymvzUvLMQllU5Ixo+8v3pc=
github.com/AdaLogics/go-fuzz-headers v0.0.0-20240806141605-e8a1dd7889d6 h1:He8afgbRMd7mFxO99hRNu+6tazq8nFF9lIwo9JFroBk=
github.com/AdaLogics/go-fuzz-headers v0.0.0-20240806141605-e8a1dd7889d6/go.mod h1:8o94RPi1/7XTJvwPpRSzSUedZrtlirdB3r9Z20bi2f8=
github.com/Azure/azure-sdk-for-go/sdk/azcore v1.23.1 h1:zvXfGJCWvywnCA814d8ZiVyt+fm9nnTE8xSb99zRyfo=
github.com/Azure/azure-sdk-for-go/sdk/azcore v1.23.1/go.mod h1:iptorS+VYKFL2N6PnebpS91dubG35eAOEERnT4PJbQU=
github.com/Azure/azure-sdk-for-go/sdk/azidentity v1.14.1 h1:u93s+zU2JD62im61Bm5CZIc1ZrOJaIAWEg0WOrMVkEo=
github.com/Azure/azure-sdk-for-go/sdk/azidentity v1.14.1/go.mod h1:oXtinPO4OLj9d1DOTrqrL1oRwGhcqadvAmrl6wTeGlk=
github.com/Azure/azure-sdk-for-go/sdk/azcore v1.22.0 h1:aokoqcHvaGjiM3VpjKDfMMnF/8epJ+Q1HLJ7CudztqE=
github.com/Azure/azure-sdk-for-go/sdk/azcore v1.22.0/go.mod h1:/WYEx9pcM9Y+Dd/APJaNlSvVSvzl54rrMdZT5+Oi2LM=
github.com/Azure/azure-sdk-for-go/sdk/azidentity v1.14.0 h1:CU4+EJeJi3TKYWEcYuSdWsjzw0nVsK/H0MSQOiPcymU=
github.com/Azure/azure-sdk-for-go/sdk/azidentity v1.14.0/go.mod h1:q0+UTSRvShwUCrR/s5HtyInYphN7Wvxb7snFM3u+SLA=
github.com/Azure/azure-sdk-for-go/sdk/azidentity/cache v0.4.0 h1:xFaZZ+IubdftrDHnGGwZ6QvQ3KHTtWl2MCK+GMt2vxs=
github.com/Azure/azure-sdk-for-go/sdk/azidentity/cache v0.4.0/go.mod h1:mCBhUhlMjLLJKr5aqw2TNS/VqJOie8MzWq3DAMJeKso=
github.com/Azure/azure-sdk-for-go/sdk/internal v1.12.0 h1:fhqpLE3UEXi9lPaBRpQ6XuRW0nU7hgg4zlmZZa+a9q4=
@@ -569,8 +577,8 @@ github.com/Azure/go-ntlmssp v0.1.1 h1:l+FM/EEMb0U9QZE7mKNEDw5Mu3mFiaa2GKOoTSsNDP
github.com/Azure/go-ntlmssp v0.1.1/go.mod h1:NYqdhxd/8aAct/s4qSYZEerdPuH1liG2/X9DiVTbhpk=
github.com/AzureAD/microsoft-authentication-extensions-for-go/cache v0.1.1 h1:WJTmL004Abzc5wDB5VtZG2PJk5ndYDgVacGqfirKxjM=
github.com/AzureAD/microsoft-authentication-extensions-for-go/cache v0.1.1/go.mod h1:tCcJZ0uHAmvjsVYzEFivsRTN00oz5BEsRgQHu5JZ9WE=
github.com/AzureAD/microsoft-authentication-library-for-go v1.8.0 h1:Nljr4q1GRA/5vCrMONS+g4u4LRHNgOXVSh3O43J2CnI=
github.com/AzureAD/microsoft-authentication-library-for-go v1.8.0/go.mod h1:Y33QHnf0FfdVewFFISOGe20mkZbxX4H839o955/PoeI=
github.com/AzureAD/microsoft-authentication-library-for-go v1.7.2 h1:RHK7bS+HQMslb1sZpAokUt+zTVmue0hKSs2C791hhzU=
github.com/AzureAD/microsoft-authentication-library-for-go v1.7.2/go.mod h1:HKpQxkWaGLJ+D/5H8QRpyQXA1eKjxkFlOMwck5+33Jk=
github.com/BurntSushi/toml v0.3.1/go.mod h1:xHWCNGjB5oqiDr8zfno3MHue2Ht5sIBksp03qcyfWMU=
github.com/BurntSushi/xgb v0.0.0-20160522181843-27f122750802/go.mod h1:IVnqGOEym/WlBOVXweHU+Q+/VP0lqqI8lqeDx9IjBqo=
github.com/Codefor/geohash v0.0.0-20140723084247-1b41c28e3a9d h1:iG9B49Q218F/XxXNRM7k/vWf7MKmLIS8AcJV9cGN4nA=
@@ -585,8 +593,8 @@ github.com/FilenCloudDienste/filen-sdk-go v0.0.39 h1:tgV5jYL6dsXop9TpDTIQU6UwJjw
github.com/FilenCloudDienste/filen-sdk-go v0.0.39/go.mod h1:0cBhKXQg49XbKZZfk5TCDa3sVLP+xMxZTWL+7KY0XR0=
github.com/Files-com/files-sdk-go/v3 v3.3.194 h1:dtOFxSTWWRpkmvXa6ycNiw8dVDu1wkgzcXyVV1VafNc=
github.com/Files-com/files-sdk-go/v3 v3.3.194/go.mod h1:rl0WumSN9gSo775DgvQv+wMQ8rlb0ES/1hU5jkMtLXg=
github.com/GoogleCloudPlatform/opentelemetry-operations-go/detectors/gcp v1.35.0 h1:bN1gA3of5bXtbnLsRPrwfmbbe7A5UWFlcTHseujLnpc=
github.com/GoogleCloudPlatform/opentelemetry-operations-go/detectors/gcp v1.35.0/go.mod h1:Yj5vHEz/aAepZGliRJsA6uvHAVAQyEwajq9ORCHPxzM=
github.com/GoogleCloudPlatform/opentelemetry-operations-go/detectors/gcp v1.34.0 h1:yzIYdwuro811Z27D3T80Wkd3rqZzb0K43nner7Eh1yE=
github.com/GoogleCloudPlatform/opentelemetry-operations-go/detectors/gcp v1.34.0/go.mod h1:pJTkW8hEUIIi3Pf65lPZOnn4Y81yCllX6IWk2jNXdkM=
github.com/GoogleCloudPlatform/opentelemetry-operations-go/exporter/metric v0.57.0 h1:jLdiS1vO+XJFyDSWRHBx56r4s/NNtcl5J6KyCcWUX/w=
github.com/GoogleCloudPlatform/opentelemetry-operations-go/exporter/metric v0.57.0/go.mod h1:8lmpHY+1VRoteiOwyrQMDt1YGXOrFKCz+1wJW7n3ODY=
github.com/GoogleCloudPlatform/opentelemetry-operations-go/internal/cloudmock v0.57.0 h1:cSjUzZ7KU8hicTgzaSv9NmSyM9fTVK3y5lsBUl3wOis=
@@ -598,6 +606,15 @@ github.com/IBM/go-sdk-core/v5 v5.23.1/go.mod h1:yO+OQpByKDLTvpEcsFFexgzpeR8eRfCF
github.com/Jille/raft-grpc-transport v1.6.1 h1:gN3sjapb+fVbiebS7AfQQgbV2ecTOI7ur7NPPC7Mhoc=
github.com/Jille/raft-grpc-transport v1.6.1/go.mod h1:HbOjEdu/yzCJ/mjTF6wEOJNbAUpHfU2UOA2hVD4CNFg=
github.com/JohnCGriffin/overflow v0.0.0-20211019200055-46fa312c352c/go.mod h1:X0CRv0ky0k6m906ixxpzmDRLvX58TFUKS2eePweuyxk=
github.com/MarvinJWendt/testza v0.1.0/go.mod h1:7AxNvlfeHP7Z/hDQ5JtE3OKYT3XFUeLCDE2DQninSqs=
github.com/MarvinJWendt/testza v0.2.1/go.mod h1:God7bhG8n6uQxwdScay+gjm9/LnO4D3kkcZX4hv9Rp8=
github.com/MarvinJWendt/testza v0.2.8/go.mod h1:nwIcjmr0Zz+Rcwfh3/4UhBp7ePKVhuBExvZqnKYWlII=
github.com/MarvinJWendt/testza v0.2.10/go.mod h1:pd+VWsoGUiFtq+hRKSU1Bktnn+DMCSrDrXDpX2bG66k=
github.com/MarvinJWendt/testza v0.2.12/go.mod h1:JOIegYyV7rX+7VZ9r77L/eH6CfJHHzXjB69adAhzZkI=
github.com/MarvinJWendt/testza v0.3.0/go.mod h1:eFcL4I0idjtIx8P9C6KkAuLgATNKpX4/2oUqKc6bF2c=
github.com/MarvinJWendt/testza v0.4.2/go.mod h1:mSdhXiKH8sg/gQehJ63bINcCKp7RtYewEjXsvsVUPbE=
github.com/MarvinJWendt/testza v0.5.2 h1:53KDo64C1z/h/d/stCYCPY69bt/OSwjq5KpFNwi+zB4=
github.com/MarvinJWendt/testza v0.5.2/go.mod h1:xu53QFE5sCdjtMCKk8YMQ2MnymimEctc4n3EjyIYvEY=
github.com/Masterminds/semver/v3 v3.2.0 h1:3MEsd0SM6jqZojhjLWWeBY+Kcjy9i6MQAeY7YgDP83g=
github.com/Masterminds/semver/v3 v3.2.0/go.mod h1:qvl/7zhW3nngYb5+80sSMF+FG2BjYrf8m9wsX0PNOMQ=
github.com/Max-Sum/base32768 v0.0.0-20230304063302-18e6ce5945fd h1:nzE1YQBdx1bq9IlZinHa+HVffy+NmVRoKr+wHN8fpLE=
@@ -620,8 +637,8 @@ github.com/ProtonMail/gopenpgp/v3 v3.4.1 h1:K7uUhSHSJxORZ+RuHpilTT6S4MA2whCRlXNw
github.com/ProtonMail/gopenpgp/v3 v3.4.1/go.mod h1:bGdV9f6edhmd581wzXsQCTKdH8bXBbyhkgDKPjwPc6U=
github.com/PuerkitoBio/goquery v1.12.0 h1:pAcL4g3WRXekcB9AU/y1mbKez2dbY2AajVhtkO8RIBo=
github.com/PuerkitoBio/goquery v1.12.0/go.mod h1:802ej+gV2y7bbIhOIoPY5sT183ZW0YFofScC4q/hIpQ=
github.com/RoaringBitmap/roaring/v2 v2.26.0 h1:K30ZxF4vZcIKvJsbmgfiep2K64f+dILJqkYGoj4xnwU=
github.com/RoaringBitmap/roaring/v2 v2.26.0/go.mod h1:BZufmFbox589n3j5eOmyTaLSGXbRLc2LmQvjKjzSEGU=
github.com/RoaringBitmap/roaring/v2 v2.24.0 h1:zQkkBZtG3WRP4j+P3A5DO221SvL1Br88TJkhyqEQRZo=
github.com/RoaringBitmap/roaring/v2 v2.24.0/go.mod h1:SfT3of9nYh3vis1dIbCj4Yw6KQGujTN+f345nrN/0JA=
github.com/Sereal/Sereal/Go/sereal v0.0.0-20231009093132-b9187f1a92c6/go.mod h1:JwrycNnC8+sZPDyzM3MQ86LvaGzSpfxg885KOOwFRW4=
github.com/Shopify/sarama v1.38.1 h1:lqqPUPQZ7zPqYlWpTh+LQ9bhYNu2xJL6k1SJN4WVe2A=
github.com/Shopify/sarama v1.38.1/go.mod h1:iwv9a67Ha8VNa+TifujYoWGxWnu2kNVAQdSdZ4X2o5g=
@@ -655,13 +672,14 @@ github.com/alecthomas/template v0.0.0-20160405071501-a0175ee3bccc/go.mod h1:LOuy
github.com/alecthomas/template v0.0.0-20190718012654-fb15b899a751/go.mod h1:LOuyumcjzFXgccqObfd/Ljyb9UuFJ6TxHnclSeseNhc=
github.com/alecthomas/units v0.0.0-20151022065526-2efee857e7cf/go.mod h1:ybxpYRFXyAe+OPACYpWeL0wqObRcbAqCMya13uyzqw0=
github.com/alecthomas/units v0.0.0-20190717042225-c3de453c63f4/go.mod h1:ybxpYRFXyAe+OPACYpWeL0wqObRcbAqCMya13uyzqw0=
github.com/alecthomas/units v0.0.0-20190924025748-f65c72e2690d/go.mod h1:rBZYJk541a8SKzHPHnH3zbiI+7dagKZ0cgpgrD7Fyho=
github.com/alexbrainman/sspi v0.0.0-20250919150558-7d374ff0d59e h1:4dAU9FXIyQktpoUAgOJK3OTFc/xug0PCXYCqU0FgDKI=
github.com/alexbrainman/sspi v0.0.0-20250919150558-7d374ff0d59e/go.mod h1:cEWa1LVoE5KvSD9ONXsZrj0z6KqySlCCNKHlLzbqAt4=
github.com/anchore/go-lzo v0.1.1 h1:IwL/fvkdtlIrYIXck6WxZ3nb8WjjHziYYmGxlooyOnM=
github.com/anchore/go-lzo v0.1.1/go.mod h1:3kLx0bve2oN1iDwgM1U5zGku1Tfbdb0No5qp1eL1fIk=
github.com/andybalholm/brotli v1.0.4/go.mod h1:fO7iG3H7G2nSZ7m0zPUDn85XEX2GTukHGRSepvi9Eig=
github.com/andybalholm/brotli v1.2.3 h1:8H1qwOkl2LPfjf3YezB90JnCliZb6SInJ/OJkEbA5NQ=
github.com/andybalholm/brotli v1.2.3/go.mod h1:rzTDkvFWvIrjDXZHkuS16NPggd91W3kUSvPlQ1pLaKY=
github.com/andybalholm/brotli v1.2.2 h1:HzTuoo2ErYQqf5qvcJInB8uvqSVxRttzkFexPWtnceM=
github.com/andybalholm/brotli v1.2.2/go.mod h1:rzTDkvFWvIrjDXZHkuS16NPggd91W3kUSvPlQ1pLaKY=
github.com/andybalholm/cascadia v1.3.4 h1:vM2lgh0Vru9Vwyfm4cQqWP2HHMW0u0+2PAW7Q38Qufg=
github.com/andybalholm/cascadia v1.3.4/go.mod h1:BLRmbRjpEtNKieZOCCvYj4RqN+KRA41GBe/5O+G93kM=
github.com/antihax/optional v1.0.0/go.mod h1:uupD/76wgC+ih3iEmQUL+0Ugr19nfwCT1kdvxnR2qWY=
@@ -669,13 +687,13 @@ github.com/antithesishq/antithesis-sdk-go v0.6.0-default-no-op h1:kpBdlEPbRvff0m
github.com/antithesishq/antithesis-sdk-go v0.6.0-default-no-op/go.mod h1:IUpT2DPAKh6i/YhSbt6Gl3v2yvUZjmKncl7U91fup7E=
github.com/antlr4-go/antlr/v4 v4.13.1 h1:SqQKkuVZ+zWkMMNkjy5FZe5mr5WURWnlpmOuzYWrPrQ=
github.com/antlr4-go/antlr/v4 v4.13.1/go.mod h1:GKmUxMtwp6ZgGwZSva4eWPC5mS6vUAmOABFgjdkM7Nw=
github.com/apache/arrow-go/v18 v18.8.0 h1:BLOzbPv7bxMPgXPacAg6HQjnxupYsZzC4tf+FkqPU/M=
github.com/apache/arrow-go/v18 v18.8.0/go.mod h1:uJCFfCwq0KsxCmsCfQg4ft+LsW+iHYzAXiSDh5ug/8U=
github.com/apache/arrow-go/v18 v18.7.0 h1:Vw/i+cJyebUofT7JlqFpe65LrmwxULn166jjwStM4HY=
github.com/apache/arrow-go/v18 v18.7.0/go.mod h1:PM6IigLJkdMwIpeHXnymo+xZ52f42a9EYiLtRel4p/A=
github.com/apache/arrow/go/v10 v10.0.1/go.mod h1:YvhnlEePVnBS4+0z3fhPfUy7W1Ikj0Ih0vcRo/gZ1M0=
github.com/apache/cassandra-gocql-driver/v2 v2.1.2 h1:lu/p0Db2av18enHJvWJQoChLssI0P+AR06STq4VdvCc=
github.com/apache/cassandra-gocql-driver/v2 v2.1.2/go.mod h1:QH/asJjB3mHvY6Dot6ZKMMpTcOrWJ8i9GhsvG1g0PK4=
github.com/apache/iceberg-go v0.7.0 h1:bTD6Pb4uM4sWcMfIx0/cV1haHMc84r+CvPieUqYzD4I=
github.com/apache/iceberg-go v0.7.0/go.mod h1:oGz5MX3/m3GDc6acsrwbdHChj28jZH8MVDWqUrW6WQQ=
github.com/apache/iceberg-go v0.6.1-0.20260817192109-c2105090c9e2 h1:xRULj4L2wrlAAPAYNruyuP17lz2zq2DGLgIFDgqiBjo=
github.com/apache/iceberg-go v0.6.1-0.20260817192109-c2105090c9e2/go.mod h1:u6gs2aFRl7QOL0FqfHX6DOzvxOyev6mmq3XT+3VUS9M=
github.com/apache/thrift v0.16.0/go.mod h1:PHK3hniurgQaNMZYaCLEqXKsYK8upmhPbmdP2FXSqgU=
github.com/apache/thrift v0.24.0 h1:zy31L1a49QTNB2bG1BBfMXol3yJrTH975G3pPubQVLQ=
github.com/apache/thrift v0.24.0/go.mod h1:zPt6WxgvTOM6hF92y8C+MkEM5LMxZuk4JcQOiU4Esvs=
@@ -689,22 +707,23 @@ github.com/arangodb/go-velocypack v0.0.0-20200318135517-5af53c29c67e h1:Xg+hGrY2
github.com/arangodb/go-velocypack v0.0.0-20200318135517-5af53c29c67e/go.mod h1:mq7Shfa/CaixoDxiyAAc5jZ6CVBAyPaNQCGS7mkj4Ho=
github.com/armon/go-metrics v0.4.1 h1:hR91U9KYmb6bLBYLQjyM+3j+rcd/UhE+G78SFnF8gJA=
github.com/armon/go-metrics v0.4.1/go.mod h1:E6amYzXo6aW1tqzoZGT755KkbgrJsSdpwZ+3JqfkOG4=
github.com/atomicgo/cursor v0.0.1/go.mod h1:cBON2QmmrysudxNBFthvMtN32r3jxVRIvzkUiF/RuIk=
github.com/aws/aws-sdk-go v1.55.8 h1:JRmEUbU52aJQZ2AjX4q4Wu7t4uZjOu71uyNmaWlUkJQ=
github.com/aws/aws-sdk-go v1.55.8/go.mod h1:ZkViS9AqA6otK+JBBNH2++sx1sgxrPKcSzPPvQkUtXk=
github.com/aws/aws-sdk-go-v2 v1.47.0 h1:0jsHallhJCeaU0Ko48c/3FK1ctOQ7NpzggxriJOQ8MQ=
github.com/aws/aws-sdk-go-v2 v1.47.0/go.mod h1:bttEH6JqnUL8LepvDVfdrds/fZ5bCIxzpe3abyUrhDU=
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.20 h1:GPRlPwz40I2B2VrBEASOA3Bi77NyeqejNLkifosX0rs=
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.20/go.mod h1:g7PNzKcsOKWb4fkSRBA7BZVAS6Y8IcxzN+nRohhQ1Q8=
github.com/aws/aws-sdk-go-v2/config v1.33.1 h1:bq9jze1hQ5YTCLoVxNnbp0T7rglrlOE7N9YsHqjGkEw=
github.com/aws/aws-sdk-go-v2/config v1.33.1/go.mod h1:2A3HQwG4zaL5Tm80rc6RZj8LmWWv4WYT5v8raSz/L7A=
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.18 h1:LAfOuhAH331fmOjTQpAaOlH+Ftn7RzSDJ2VFwjdMMy4=
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.18/go.mod h1:4e5xhuXHx1e4U9EthvbPP1r/DIMp5c2823OL8karzcM=
github.com/aws/aws-sdk-go-v2/config v1.32.35 h1:UEzXuET8E42lxBPijuACu/tEK7v5lFPlk0Q+GT5WD9E=
github.com/aws/aws-sdk-go-v2/config v1.32.35/go.mod h1:KaMtJpFa2JlL2BStjjHQVwQpzZEmw+ND/EgVrfFoo2g=
github.com/aws/aws-sdk-go-v2/credentials v1.20.4 h1:hTvrJJseKbvw32kmiE0G+u/9ZqpqscjDrTigHIXP2qs=
github.com/aws/aws-sdk-go-v2/credentials v1.20.4/go.mod h1:gWp9O1ZBWwpcIrgV+mVHk4gZUurAEDkgypu/OXOlIaw=
github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.20.0 h1:AM4hHjww+PSFtt6E+UrBrPlZkWsePCLEt9AjkfQX+yM=
github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.20.0/go.mod h1:3x/yXezeQjpOvBb4jEMxrS8SXvpdvJ5abv6l5c1gWM8=
github.com/aws/aws-sdk-go-v2/feature/s3/manager v1.22.34 h1:Pn7OsMwBLbkZ6OnCxWHAjf0L/22H8cnhxZC0uPwtMtg=
github.com/aws/aws-sdk-go-v2/feature/s3/manager v1.22.34/go.mod h1:eToXR/Gk1uqpn04eSmdgVXwfS0WvH8aG4eBFr8ygbpU=
github.com/aws/aws-sdk-go-v2/feature/s3/transfermanager v0.4.1 h1:I3mWvASaICc5c8vJ3ftYjroh7LT3jG0q/KtzSu5wW/s=
github.com/aws/aws-sdk-go-v2/feature/s3/transfermanager v0.4.1/go.mod h1:VwSN8piv62OyyxYWtJzj6j7gECyLc0WwMo30L5NQZlE=
github.com/aws/aws-sdk-go-v2/feature/s3/transfermanager v0.3.11 h1:eBXB8KZgzQ8A9QB4iJS4aw/u6+4OY3i2hQXPABeAIOg=
github.com/aws/aws-sdk-go-v2/feature/s3/transfermanager v0.3.11/go.mod h1:N9+5pG27Fy61GUL5YXVLXDTLmUudMrgwsuDbgBMNLxQ=
github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.3 h1:Hp/VgjP0BysR3OgLlR057Vz2LcbbVnoWeJ+3qWiS/fY=
github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.3/go.mod h1:nwGV5qw7F1IZPgxCvA/ph8N2TAuz+BkRG/bXn808qMA=
github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.3 h1:MUaM4f+kj1ZIBPZfUS8cxP1GKXXZtHJjAthy93AN7SM=
@@ -713,14 +732,14 @@ github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.3 h1:fuSCw4Z2qfRCztMPO3GXJNSiEp6W
github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.3/go.mod h1:6SxcHheD1pPR5+kWm1wGvjlL/YqUsh267sAfEmN4K7A=
github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.19 h1:bAdDl/HkGCcGPoe25ToSHEw23VIxt6CT5fLcg111BKg=
github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.19/go.mod h1:KaUzbLxv4CeSxh6ZCl9B4m7CuFenS8kUEaDs+f/DQr4=
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.11.1 h1:s67hBfG5t9rn1NCvDuB4E3QIep3UFhHPtaIqFDjV3N8=
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.11.1/go.mod h1:FpvjBMXtSNMLPmDJsWwcY5cRnqJlpS2y1R6n4pvzs4k=
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.31 h1:uZOinZb+h7lZw8IYzP1z1IuEnueB76/EFkcf/fEW4Ag=
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.31/go.mod h1:NRtwAM/p5VRt03TlEUs0pH3TeWamWdf4YyJpSrzPYLc=
github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.3 h1:bON1rJf67TSTDCKg816AAIE4xSTtoo9tl0XRkO72R+I=
github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.3/go.mod h1:c5BBpjJcQXpfeq9iASyVKA3T6vX6B6LEXY4mL/gklDY=
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.20.1 h1:ZMbtPZZQRca+3+XYQne9PBvRiYpHZlNJJOZfE9WNfT0=
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.20.1/go.mod h1:YAGWQdCYlVCoqrzvfv3RLxO6zKwti7gsAULOGWPLYv4=
github.com/aws/aws-sdk-go-v2/service/s3 v1.109.1 h1:kVpzaDBzOdRtOftmiSpTdQbWVqRg0kONLXijktiwXnk=
github.com/aws/aws-sdk-go-v2/service/s3 v1.109.1/go.mod h1:CUr46sCpGAg/rHaclRyhJX0LJAmH73uWSJPPSaMUrSk=
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.39 h1:HLPAVrlLDaN2boN0xJx7MgaQDNEO3Q+c9L6kl/8m47Q=
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.39/go.mod h1:Pg/dVfsNkm1hsIDK/gMvCKtmyNfNTV12mrgHqVE/6Oo=
github.com/aws/aws-sdk-go-v2/service/s3 v1.107.3 h1:IKoCZqfWfZzSBi16QFQ+QcbQ3LRQ7QgB1S5tDAyPBQQ=
github.com/aws/aws-sdk-go-v2/service/s3 v1.107.3/go.mod h1:RBpRcXiM4s2pOInVs32GsBonnje+fiAj4mcrStRmlCA=
github.com/aws/aws-sdk-go-v2/service/signin v1.10.0 h1:ZD5qFpWcaOKdTuhBi431pIDkCgrMkMlMT6jlpSPoIRI=
github.com/aws/aws-sdk-go-v2/service/signin v1.10.0/go.mod h1:8Nuuf+tR346PjJ3MvZPh9pekbLiLQFWJhzMXfwy7alA=
github.com/aws/aws-sdk-go-v2/service/sns v1.39.14 h1:p8WdWDh5AwSZdp19Haa3XMyPCICi9Z375a/Nu3IIEZY=
@@ -731,8 +750,8 @@ github.com/aws/aws-sdk-go-v2/service/sso v1.38.0 h1:JGeeBcMlhg1xtOXYpeCaTQBZObtX
github.com/aws/aws-sdk-go-v2/service/sso v1.38.0/go.mod h1:XwteswG9EOMRFm73UT0t+MbTwyLxMrEXkU6e+v92Lzo=
github.com/aws/aws-sdk-go-v2/service/ssooidc v1.43.0 h1:obhahQXDEdVEv8y5bTKXR30LVaxYe1kyYM0L7l2Iq+k=
github.com/aws/aws-sdk-go-v2/service/ssooidc v1.43.0/go.mod h1:6twZZ/aXHNy1vXUO8koUbp++MYzMASkOgEBdkbJYmO0=
github.com/aws/aws-sdk-go-v2/service/sts v1.51.0 h1:Zpnqa6XtrNzXZnwbdCqHOXpXhMsa01ql/pcRQ1sb4hk=
github.com/aws/aws-sdk-go-v2/service/sts v1.51.0/go.mod h1:/8JRcdTt//hG0Q4BTmGbuOplT7ABe+5rdtqUHqXvYIM=
github.com/aws/aws-sdk-go-v2/service/sts v1.50.0 h1:khXV3+K5D3f4e8xtplaRdSFn1bEg3gj5EBHQvbCOZbQ=
github.com/aws/aws-sdk-go-v2/service/sts v1.50.0/go.mod h1:/8JRcdTt//hG0Q4BTmGbuOplT7ABe+5rdtqUHqXvYIM=
github.com/aws/smithy-go v1.28.1 h1:R/nXH00c8qcfCzQVELtRw+eLQWtzv+VAIEFJ1/xxXlQ=
github.com/aws/smithy-go v1.28.1/go.mod h1:YE2RhdIuDbA5E5bTdciG9KrW3+TiEONeUWCqxX9i1Fc=
github.com/bahlo/generic-list-go v0.2.0 h1:5sz/EEAK+ls5wF+NeqDpk5+iNdMDXrh3z3nPnH1Wvgk=
@@ -848,12 +867,13 @@ github.com/colinmarc/hdfs/v2 v2.4.0 h1:v6R8oBx/Wu9fHpdPoJJjpGSUxo8NhHIwrwsfhFvU9
github.com/colinmarc/hdfs/v2 v2.4.0/go.mod h1:0NAO+/3knbMx6+5pCv+Hcbaz4xn/Zzbn9+WIib2rKVI=
github.com/compose-spec/compose-go/v2 v2.12.1 h1:+xBZNxcgSus4atQJwXPEdhHRgCEyZmj/BuqN5m33Ou0=
github.com/compose-spec/compose-go/v2 v2.12.1/go.mod h1:ZU6zlcweCZKyiB7BVfCizQT9XmkEIMFE+PRZydVcsZg=
github.com/containerd/console v1.0.3/go.mod h1:7LqA/THxQ86k76b8c/EMSiaJ3h1eZkMkXar0TQ1gf3U=
github.com/containerd/console v1.0.5 h1:R0ymNeydRqH2DmakFNdmjR2k0t7UPuiOV/N/27/qqsc=
github.com/containerd/console v1.0.5/go.mod h1:YynlIjWYF8myEu6sdkwKIvGQq+cOckRm6So2avqoYAk=
github.com/containerd/containerd/api v1.11.1 h1:h8nfoDW9+fNsC/9TwiAHj8B1GzXKtR4eFtkhi/X5RLU=
github.com/containerd/containerd/api v1.11.1/go.mod h1:CaQFRu+N1MtbgL6JDOJLUB1hCKESU1lD6MuTJhgtdlw=
github.com/containerd/containerd/v2 v2.2.8 h1:8nnNE5FqBmofd3lccku8GbWi6d1TO4rrcB2E/0o+HU0=
github.com/containerd/containerd/v2 v2.2.8/go.mod h1:lTw+wrjREio28N9+3umHS73C6Cs1mxrhczBcAliInuI=
github.com/containerd/containerd/v2 v2.2.5 h1:KTFzB02LviYmmfRmz8r9UFd+n6YlddVFK+5lbgQXUTU=
github.com/containerd/containerd/v2 v2.2.5/go.mod h1:5t2+xFv2dGd/iDYp9Z8DXB4cmWrWQi1XqxGJPS2gBzU=
github.com/containerd/continuity v0.5.0 h1:7a85HZpCSs+1Zps0Ee3DPSuAWY+0SJM1JNM51nlEVDg=
github.com/containerd/continuity v0.5.0/go.mod h1:/lNJvtJKUQStBzpVQ1+rasXO1LAWtUQssk28EZvJ3nE=
github.com/containerd/errdefs v1.0.0 h1:tg5yIfIlQIrxYtu9ajqY42W3lpS19XqdxRQeEwYG8PI=
@@ -933,8 +953,8 @@ github.com/dropbox/dropbox-sdk-go-unofficial/v6 v6.4.0/go.mod h1:gDXhl0OElhzYoDs
github.com/dsnet/try v0.0.3 h1:ptR59SsrcFUYbT/FhAbKTV6iLkeD6O18qfIWRml2fqI=
github.com/dsnet/try v0.0.3/go.mod h1:WBM8tRpUmnXXhY1U6/S8dt6UWdHTQ7y8A5YSkRCkq40=
github.com/dustin/go-humanize v1.0.0/go.mod h1:HtrtbFcZ19U5GC7JDqmcUSB87Iq5E25KnS6fMYU6eOk=
github.com/dustin/go-humanize v1.1.0 h1:dbKTrvD0klcbBV/h4AWJdMuZogJACoMlvWIWZ5b2xWg=
github.com/dustin/go-humanize v1.1.0/go.mod h1:hc1CvRkJMsgxqjmjMQF3QNRAZBwY8AXBAzKYoSX9sFI=
github.com/dustin/go-humanize v1.0.1 h1:GzkhY7T5VNhEkwH0PVJgjz+fX1rhBrR7pRT3mDkpeCY=
github.com/dustin/go-humanize v1.0.1/go.mod h1:Mu1zIs6XwVuF/gI1OepvI0qD18qycQx+mFykh5fBlto=
github.com/eapache/go-resiliency v1.6.0 h1:CqGDTLtpwuWKn6Nj3uNUdflaq+/kIPsg0gfNzHton30=
github.com/eapache/go-resiliency v1.6.0/go.mod h1:5yPzW0MIvSe0JDsv0v+DvcjEv2FyD6iZYSs1ZI+iQho=
github.com/eapache/go-xerial-snappy v0.0.0-20230731223053-c322873962e3 h1:Oy0F4ALJ04o5Qqpdz8XLIpNA3WM/iSIXqxtqo7UGVws=
@@ -967,8 +987,8 @@ github.com/envoyproxy/go-control-plane v0.10.3/go.mod h1:fJJn/j26vwOu972OllsvAgJ
github.com/envoyproxy/go-control-plane v0.11.0/go.mod h1:VnHyVMpzcLvCFt9yUz1UnCwHLhwx1WguiVDV7pTG/tI=
github.com/envoyproxy/go-control-plane v0.14.0 h1:hbG2kr4RuFj222B6+7T83thSPqLjwBIfQawTkC++2HA=
github.com/envoyproxy/go-control-plane v0.14.0/go.mod h1:NcS5X47pLl/hfqxU70yPwL9ZMkUlwlKxtAohpi2wBEU=
github.com/envoyproxy/go-control-plane/envoy v1.39.1-0.20260819172001-e6e3fd93e4be h1:SWe0x6yfglnxuvOiYgTTnNq7QD/yvqthOh1RBI8Bj8w=
github.com/envoyproxy/go-control-plane/envoy v1.39.1-0.20260819172001-e6e3fd93e4be/go.mod h1:PYEOlng9XcrulfyWpm49jECTPV0LT4q8cO7fLW/xwgk=
github.com/envoyproxy/go-control-plane/envoy v1.37.0 h1:u3riX6BoYRfF4Dr7dwSOroNfdSbEPe9Yyl09/B6wBrQ=
github.com/envoyproxy/go-control-plane/envoy v1.37.0/go.mod h1:DReE9MMrmecPy+YvQOAOHNYMALuowAnbjjEMkkWOi6A=
github.com/envoyproxy/go-control-plane/ratelimit v0.1.0 h1:/G9QYbddjL25KvtKTv3an9lx6VBE2cnb8wp1vEGNYGI=
github.com/envoyproxy/go-control-plane/ratelimit v0.1.0/go.mod h1:Wk+tMFAFbCXaJPzVVHnPgRKdUdwW/KdbRt94AzgRee4=
github.com/envoyproxy/protoc-gen-validate v0.1.0/go.mod h1:iSmxcyjqTsJpI2R4NaDN7+kN2VEUnK/pcBlmesArF7c=
@@ -990,8 +1010,8 @@ github.com/facebookgo/subset v0.0.0-20200203212716-c811ad88dec4/go.mod h1:5tD+ne
github.com/fanixk/geohash v0.0.0-20150324002647-c1f9b5fa157a h1:Fyfh/dsHFrC6nkX7H7+nFdTd1wROlX/FxEIWVpKYf1U=
github.com/fanixk/geohash v0.0.0-20150324002647-c1f9b5fa157a/go.mod h1:UgNw+PTmmGN8rV7RvjvnBMsoTU8ZXXnaT3hYsDTBlgQ=
github.com/fatih/color v1.13.0/go.mod h1:kLAiJbzzSOZDVNGyDpeOxJ47H46qBXwg5ILebYFFOfk=
github.com/fatih/color v1.19.0 h1:Zp3PiM21/9Ld6FzSKyL5c/BULoe/ONr9KlbYVOfG8+w=
github.com/fatih/color v1.19.0/go.mod h1:zNk67I0ZUT1bEGsSGyCZYZNrHuTkJJB+r6Q9VuMi0LE=
github.com/fatih/color v1.18.0 h1:S8gINlzdQ840/4pfAwic/ZE0djQEH3wM94VfqLTZcOM=
github.com/fatih/color v1.18.0/go.mod h1:4FelSpRwEGDpQ12mAdzqdOukCy4u8WUtOY6lkT/6HfU=
github.com/felixge/httpsnoop v1.1.0 h1:3YtUj32ZZkqZtt3sZZsClsymw/QDuVfpNhoA31zeORc=
github.com/felixge/httpsnoop v1.1.0/go.mod h1:Zqxgdd+1Rkcz8euOqdr7lqgCRJztwr5hp9vDSi5UZCE=
github.com/fluent/fluent-logger-golang v1.10.1 h1:wu54iN1O2afll5oQrtTjhgZRwWcfOeFFzwRsEkABfFQ=
@@ -1242,8 +1262,8 @@ github.com/googleapis/enterprise-certificate-proxy v0.1.0/go.mod h1:17drOmN3MwGY
github.com/googleapis/enterprise-certificate-proxy v0.2.0/go.mod h1:8C0jb7/mgJe/9KK8Lm7X9ctZC2t60YyIpYEI16jx0Qg=
github.com/googleapis/enterprise-certificate-proxy v0.2.1/go.mod h1:AwSRAtLfXpU5Nm3pW+v7rGDHp09LsPtGY9MduiEsR9k=
github.com/googleapis/enterprise-certificate-proxy v0.2.3/go.mod h1:AwSRAtLfXpU5Nm3pW+v7rGDHp09LsPtGY9MduiEsR9k=
github.com/googleapis/enterprise-certificate-proxy v0.3.21 h1:OFdQ3tnCX/zaQ0Cedur3D3z7kI6HiLX9g3TiAN4/DFU=
github.com/googleapis/enterprise-certificate-proxy v0.3.21/go.mod h1:L3D/IQExI6LqEjBdXcZQ1WluSgigQmSwBboFstVPM4w=
github.com/googleapis/enterprise-certificate-proxy v0.3.20 h1:t/xL64VUoN69MuMRQuJETqYGOw4Z9mSRJK9epIEtwFk=
github.com/googleapis/enterprise-certificate-proxy v0.3.20/go.mod h1:L3D/IQExI6LqEjBdXcZQ1WluSgigQmSwBboFstVPM4w=
github.com/googleapis/gax-go/v2 v2.0.4/go.mod h1:0Wqv26UfaUD9n4G6kQubkQ+KchISgw+vpHVxEJEs9eg=
github.com/googleapis/gax-go/v2 v2.0.5/go.mod h1:DWXyrwAJ9X0FpwwEdw+IPEYBICEFu5mhpdKc/us6bOk=
github.com/googleapis/gax-go/v2 v2.1.0/go.mod h1:Q3nei7sK6ybPYH7twZdmQpAd1MKb7pfu6SK+H1/DsU0=
@@ -1258,6 +1278,12 @@ github.com/googleapis/gax-go/v2 v2.24.0 h1:myMaPYyF9MecEmvQqMqomIwn9t/4KCZN9qnws
github.com/googleapis/gax-go/v2 v2.24.0/go.mod h1:IaTHBDd7NHxSCiu0vEs8pQZu4dGZrWwuSoxCnk16OFM=
github.com/googleapis/go-type-adapters v1.0.0/go.mod h1:zHW75FOG2aur7gAO2B+MLby+cLsWGBF62rFAi7WjWO4=
github.com/googleapis/google-cloud-go-testing v0.0.0-20200911160855-bcd43fbb19e8/go.mod h1:dvDLG8qkwmyD9a/MJJN3XJcT3xFxOKAvTZGvuZmac9g=
github.com/gookit/assert v0.1.1 h1:lh3GcawXe/p+cU7ESTZ5Ui3Sm/x8JWpIis4/1aF0mY0=
github.com/gookit/assert v0.1.1/go.mod h1:jS5bmIVQZTIwk42uXl4lyj4iaaxx32tqH16CFj0VX2E=
github.com/gookit/color v1.4.2/go.mod h1:fqRyamkC1W8uxl+lxCQxOT09l/vYfZ+QeiX3rKQHCoQ=
github.com/gookit/color v1.5.0/go.mod h1:43aQb+Zerm/BWh2GnrgOQm7ffz7tvQXEKV6BFMl7wAo=
github.com/gookit/color v1.6.0 h1:JjJXBTk1ETNyqyilJhkTXJYYigHG24TM9Xa2M1xAhRA=
github.com/gookit/color v1.6.0/go.mod h1:9ACFc7/1IpHGBW8RwuDm/0YEnhg3dwwXpoMsmtyHfjs=
github.com/gopherjs/gopherjs v1.17.2 h1:fQnZVsXk8uxXIStYb0N4bGk7jeyTalG/wsZjQ25dO0g=
github.com/gopherjs/gopherjs v1.17.2/go.mod h1:pRRIvn/QzFLrKfvEz3qUuEhtE/zLCWfreZ6J5gM2i+k=
github.com/gorilla/mux v1.8.1 h1:TuBL49tXwgrFYWhqrNgrUNEY92u81SPhu7sTdzQEiWY=
@@ -1290,13 +1316,13 @@ github.com/hashicorp/go-hclog v1.6.3/go.mod h1:W4Qnvbt70Wk/zYJryRzDRU/4r0kIg0PVH
github.com/hashicorp/go-immutable-radix v1.0.0/go.mod h1:0y9vanUI8NX6FsYoO3zeMjhV/C5i9g4Q3DwcSNZ4P60=
github.com/hashicorp/go-immutable-radix v1.3.1 h1:DKHmCUm2hRBK510BaiZlwvpD40f8bJFeZnpfm2KLowc=
github.com/hashicorp/go-immutable-radix v1.3.1/go.mod h1:0y9vanUI8NX6FsYoO3zeMjhV/C5i9g4Q3DwcSNZ4P60=
github.com/hashicorp/go-metrics v0.7.0 h1:lLWieZTcbzZT+rY0zrqKbyryXG8RIajdUjmM0+R79eg=
github.com/hashicorp/go-metrics v0.7.0/go.mod h1:8T/Es8FPTfQvY7azBPGyrwXwwg7mbA9/TmQ1/lWfxb4=
github.com/hashicorp/go-metrics v0.5.4 h1:8mmPiIJkTPPEbAiV97IxdAGNdRdaWwVap1BU6elejKY=
github.com/hashicorp/go-metrics v0.5.4/go.mod h1:CG5yz4NZ/AI/aQt9Ucm/vdBnbh7fvmv4lxZ350i+QQI=
github.com/hashicorp/go-msgpack v0.5.5 h1:i9R9JSrqIz0QVLz3sz+i3YJdT7TTSLcfLLzJi9aZTuI=
github.com/hashicorp/go-msgpack v0.5.5/go.mod h1:ahLV/dePpqEmjfWmKiqvPkv/twdG7iPBM1vqhUKIvfM=
github.com/hashicorp/go-msgpack/v2 v2.1.1/go.mod h1:upybraOAblm4S7rx0+jeNy+CWWhzywQsSRV5033mMu4=
github.com/hashicorp/go-msgpack/v2 v2.1.5 h1:Ue879bPnutj/hXfmUk6s/jtIK90XxgiUIcXRl656T44=
github.com/hashicorp/go-msgpack/v2 v2.1.5/go.mod h1:bjCsRXpZ7NsJdk45PoCQnzRGDaK8TKm5ZnDI/9y3J4M=
github.com/hashicorp/go-msgpack/v2 v2.1.2 h1:4Ee8FTp834e+ewB71RDrQ0VKpyFdrKOjvYtnQ/ltVj0=
github.com/hashicorp/go-msgpack/v2 v2.1.2/go.mod h1:upybraOAblm4S7rx0+jeNy+CWWhzywQsSRV5033mMu4=
github.com/hashicorp/go-multierror v1.0.0/go.mod h1:dHtQlpGsu+cZNNAkkCN/P3hoUDHhCYQXV3UM06sGGrk=
github.com/hashicorp/go-multierror v1.1.1 h1:H5DkEtf6CXdFp0N0Em5UCwQpXMWke8IA0+lD48awMYo=
github.com/hashicorp/go-multierror v1.1.1/go.mod h1:iw975J/qwKPdAO1clOe2L8331t/9/fmwbPZ6JB6eMoM=
@@ -1320,19 +1346,19 @@ github.com/hashicorp/go-version v1.9.0/go.mod h1:fltr4n8CU8Ke44wwGCBoEymUuxUHl09
github.com/hashicorp/golang-lru v0.5.0/go.mod h1:/m3WP610KZHVQ1SGc6re/UDhFvYD7pJ4Ao+sR/qLZy8=
github.com/hashicorp/golang-lru v0.5.1/go.mod h1:/m3WP610KZHVQ1SGc6re/UDhFvYD7pJ4Ao+sR/qLZy8=
github.com/hashicorp/golang-lru v0.5.4/go.mod h1:iADmTwqILo4mZ8BN3D2Q6+9jd8WM5uGBxy+E8yxSoD4=
github.com/hashicorp/golang-lru v1.0.2 h1:dV3g9Z/unq5DpblPpw+Oqcv4dU/1omnb4Ok8iPY6p1c=
github.com/hashicorp/golang-lru v1.0.2/go.mod h1:iADmTwqILo4mZ8BN3D2Q6+9jd8WM5uGBxy+E8yxSoD4=
github.com/hashicorp/golang-lru v0.6.0 h1:uL2shRDx7RTrOrTCUZEGP/wJUFiUI8QT6E7z5o8jga4=
github.com/hashicorp/golang-lru v0.6.0/go.mod h1:iADmTwqILo4mZ8BN3D2Q6+9jd8WM5uGBxy+E8yxSoD4=
github.com/hashicorp/golang-lru/v2 v2.0.7 h1:a+bsQ5rvGLjzHuww6tVxozPZFVghXaHOwFs4luLUK2k=
github.com/hashicorp/golang-lru/v2 v2.0.7/go.mod h1:QeFd9opnmA6QUJc5vARoKUSoFhyfM2/ZepoAG6RGpeM=
github.com/hashicorp/hcl v1.0.1-vault-7 h1:ag5OxFVy3QYTFTJODRzTKVZ6xvdfLLCA1cy/Y6xGI0I=
github.com/hashicorp/hcl v1.0.1-vault-7/go.mod h1:XYhtn6ijBSAj6n4YqAaf7RBPS4I06AItNorpy+MoQNM=
github.com/hashicorp/raft v1.7.0/go.mod h1:N1sKh6Vn47mrWvEArQgILTyng8GoDRNYlgKyK7PMjs0=
github.com/hashicorp/raft v1.8.0 h1:YbfecBcuTar/LNFEDfVTpqu9Aw+MczTk7MYczvy+62k=
github.com/hashicorp/raft v1.8.0/go.mod h1:agL5fncrpEsbxr5P5KOd2srskDwPY18opjXN5x0661s=
github.com/hashicorp/raft v1.7.3 h1:DxpEqZJysHN0wK+fviai5mFcSYsCkNpFUl1xpAW8Rbo=
github.com/hashicorp/raft v1.7.3/go.mod h1:DfvCGFxpAUPE0L4Uc8JLlTPtc3GzSbdH0MTJCLgnmJQ=
github.com/hashicorp/raft-boltdb v0.0.0-20230125174641-2a8082862702 h1:RLKEcCuKcZ+qp2VlaaZsYZfLOmIiuJNpEi48Rl8u9cQ=
github.com/hashicorp/raft-boltdb v0.0.0-20230125174641-2a8082862702/go.mod h1:nTakvJ4XYq45UXtn0DbwR4aU9ZdjlnIenpbs6Cd+FM0=
github.com/hashicorp/raft-boltdb/v2 v2.4.2 h1:r2RRgZ6ajT+VH8/9yC5mfoKAlt3KrYiUDhfbUSmaoTs=
github.com/hashicorp/raft-boltdb/v2 v2.4.2/go.mod h1:+wvKK1thWEqM7amY1MXjtQpkm+b9KNUXks0fcus5ZGU=
github.com/hashicorp/raft-boltdb/v2 v2.3.1 h1:ackhdCNPKblmOhjEU9+4lHSJYFkJd6Jqyvj6eW9pwkc=
github.com/hashicorp/raft-boltdb/v2 v2.3.1/go.mod h1:n4S+g43dXF1tqDT+yzcXHhXM6y7MrlUd3TTwGRcUvQE=
github.com/hashicorp/vault/api v1.23.0 h1:gXgluBsSECfRWTSW9niY2jwg2e9mMJc4WoHNv4g3h6A=
github.com/hashicorp/vault/api v1.23.0/go.mod h1:zransKiB9ftp+kgY8ydjnvCU7Wk8i9L0DYWpXeMj9ko=
github.com/hexops/gotextdiff v1.0.3 h1:gitA9+qJrrTCsiCl7+kh75nPqQt1cx4ZkudSTLoUqJM=
@@ -1392,8 +1418,11 @@ github.com/jonboulle/clockwork v0.5.0 h1:Hyh9A8u51kptdkR+cqRpT1EebBwTn1oK9YfGYbd
github.com/jonboulle/clockwork v0.5.0/go.mod h1:3mZlmanh0g2NDKO5TWZVJAfofYk64M7XN3SzBPjZF60=
github.com/josharian/intern v1.0.0 h1:vlS4z54oSdjm0bgjRigI+G1HpF+tI+9rE5LLzOg8HmY=
github.com/josharian/intern v1.0.0/go.mod h1:5DoeVV0s6jJacbCEi61lwdGj/aVlrQvzHFFd8Hwg//Y=
github.com/jpillora/backoff v1.0.0/go.mod h1:J/6gKK9jxlEcS3zixgDgUAsiuZ7yrSoa/FX5e0EB2j4=
github.com/json-iterator/go v1.1.6/go.mod h1:+SdeFBvtyEkXs7REEP0seUULqWtbJapLOCVDaaPEHmU=
github.com/json-iterator/go v1.1.9/go.mod h1:KdQUCv79m/52Kvf8AW2vK1V8akMuk1QjK/uOdHXbAo4=
github.com/json-iterator/go v1.1.10/go.mod h1:KdQUCv79m/52Kvf8AW2vK1V8akMuk1QjK/uOdHXbAo4=
github.com/json-iterator/go v1.1.11/go.mod h1:KdQUCv79m/52Kvf8AW2vK1V8akMuk1QjK/uOdHXbAo4=
github.com/json-iterator/go v1.1.12 h1:PV8peI4a0ysnczrg+LtxykD8LfKY9ML6u2jnxaEnrnM=
github.com/json-iterator/go v1.1.12/go.mod h1:e30LSqwooZae/UwlEbR2852Gd8hjQvJoHmT4TnhNGBo=
github.com/jstemmer/go-junit-report v0.0.0-20190106144839-af01ea7f8024/go.mod h1:6v2b51hI/fHJwM22ozAgKL4VKDeJcHhJFhtBdhmNjmU=
@@ -1403,6 +1432,7 @@ github.com/jtolds/gls v4.20.0+incompatible/go.mod h1:QJZ7F/aHp+rZTRtaJ1ow/lLfFfV
github.com/jtolio/noiseconn v0.0.0-20231127013910-f6d9ecbf1de7 h1:JcltaO1HXM5S2KYOYcKgAV7slU0xPy1OcvrVgn98sRQ=
github.com/jtolio/noiseconn v0.0.0-20231127013910-f6d9ecbf1de7/go.mod h1:MEkhEPFwP3yudWO0lj6vfYpLIB+3eIcuIW+e0AZzUQk=
github.com/julienschmidt/httprouter v1.2.0/go.mod h1:SYymIcj16QtmaHHD7aYtjjsJG7VTCxuUUipMqKk8s4w=
github.com/julienschmidt/httprouter v1.3.0/go.mod h1:JR6WtHb+2LUe8TCKY3cZOxFyyO8IZAc4RVcycCCAKdM=
github.com/jung-kurt/gofpdf v1.0.0/go.mod h1:7Id9E/uU8ce6rXgefFLlgrJj/GYY22cpxn+r32jIOes=
github.com/jung-kurt/gofpdf v1.0.3-0.20190309125859-24315acbbda5/go.mod h1:7Id9E/uU8ce6rXgefFLlgrJj/GYY22cpxn+r32jIOes=
github.com/jzelinskie/whirlpool v0.0.0-20201016144138-0675e54bb004 h1:G+9t9cEtnC9jFiTxyptEKuNIAbiN5ZCQzX2a74lj3xg=
@@ -1426,11 +1456,14 @@ github.com/klauspost/compress v1.15.9/go.mod h1:PhcZ0MbTNciWF3rruxRgKxI5NkcHHrHU
github.com/klauspost/compress v1.19.2 h1:hMRETovs/pu/dVWN7zIT1PGG8t509MwT6bO7XSi26R8=
github.com/klauspost/compress v1.19.2/go.mod h1:cwPg85FWrGar70rWktvGQj8/hthj3wpl0PGDogxkrSQ=
github.com/klauspost/cpuid/v2 v2.0.9/go.mod h1:FInQzS24/EEf25PyTYn52gqo7WaD8xa0213Md/qVLRg=
github.com/klauspost/cpuid/v2 v2.0.10/go.mod h1:g2LTdtYhdyuGPqyWyv7qRAmj1WBqxuObKfj5c0PQa7c=
github.com/klauspost/cpuid/v2 v2.0.12/go.mod h1:g2LTdtYhdyuGPqyWyv7qRAmj1WBqxuObKfj5c0PQa7c=
github.com/klauspost/cpuid/v2 v2.4.0 h1:S6Hrbc7+ywsr0r+RLapfGBHfyefhCTwEh3A0tV913Dw=
github.com/klauspost/cpuid/v2 v2.4.0/go.mod h1:19jmZ9mjzoF//ddRSUsv0zfBTJWh3QJh9FNxZTMrGxU=
github.com/klauspost/reedsolomon v1.14.2 h1:SafJYwpBBQBI6amHUygcjxZjXeN2HpiENHQDwuPWCCQ=
github.com/klauspost/reedsolomon v1.14.2/go.mod h1:yjqqjgMTQkBUHSG97/rm4zipffCNbCiZcB3kTqr++sQ=
github.com/konsorten/go-windows-terminal-sequences v1.0.1/go.mod h1:T0+1ngSBFLxvqU3pZ+m/2kptfBszLMUkC4ZK/EgS/cQ=
github.com/konsorten/go-windows-terminal-sequences v1.0.3/go.mod h1:T0+1ngSBFLxvqU3pZ+m/2kptfBszLMUkC4ZK/EgS/cQ=
github.com/koofr/go-httpclient v0.0.0-20240520111329-e20f8f203988 h1:CjEMN21Xkr9+zwPmZPaJJw+apzVbjGL5uK/6g9Q2jGU=
github.com/koofr/go-httpclient v0.0.0-20240520111329-e20f8f203988/go.mod h1:/agobYum3uo/8V6yPVnq+R82pyVGCeuWW5arT4Txn8A=
github.com/koofr/go-koofrclient v0.0.0-20221207135200-cbd7fc9ad6a6 h1:FHVoZMOVRA+6/y4yRlbiR3WvsrOcKBd/f64H7YiWR2U=
@@ -1462,6 +1495,8 @@ github.com/linkedin/goavro/v2 v2.15.0 h1:pDj1UrjUOO62iXhgBiE7jQkpNIc5/tA5eZsgolM
github.com/linkedin/goavro/v2 v2.15.0/go.mod h1:KXx+erlq+RPlGSPmLF7xGo6SAbh8sCQ53x064+ioxhk=
github.com/linxGnu/grocksdb v1.10.8 h1:Nau01Hhm/0kaVTR6d4viwD6npYbnDvZAfzwJCLzKRYo=
github.com/linxGnu/grocksdb v1.10.8/go.mod h1:OLQKZwiKwaJiAVCsOzWKvwiLwfZ5Vz8Md5TYR7t7pM8=
github.com/lithammer/fuzzysearch v1.1.8 h1:/HIuJnjHuXS8bKaiTMeeDlW2/AyIWk2brx1V8LFgLN4=
github.com/lithammer/fuzzysearch v1.1.8/go.mod h1:IdqeyBClc3FFqSzYq/MXESsS4S0FsZ5ajtkr5xPLts4=
github.com/lithammer/shortuuid/v3 v3.0.7 h1:trX0KTHy4Pbwo/6ia8fscyHoGA+mf1jWbPJVuvyJQQ8=
github.com/lithammer/shortuuid/v3 v3.0.7/go.mod h1:vMk8ke37EmiewwolSO1NLW8vP4ZaKlRuDIi8tWWmAts=
github.com/lpar/calendar v0.2.0 h1:A1kxv6sbvBHFUkd2XotanIRqEXQGreQOeuGhkJqIaRA=
@@ -1486,6 +1521,7 @@ github.com/mattn/go-isatty v0.0.16/go.mod h1:kYGgaQfpe5nmfYZH+SKPsOc2e4SrIfOl2e/
github.com/mattn/go-isatty v0.0.24 h1:tGZZoVgT/KiqK1c8ocVLeDS8BSWMRd47J3Lbz7vsReI=
github.com/mattn/go-isatty v0.0.24/go.mod h1:nMCL3Zebbrt45jsMDgnfIwz6ydEQApk5oEI3HqDio6A=
github.com/mattn/go-runewidth v0.0.3/go.mod h1:LwmH8dsx7+W8Uxz3IHJYH5QSwggIsqBzpuz5H//U1FU=
github.com/mattn/go-runewidth v0.0.13/go.mod h1:Jdepj2loyihRzMpdS35Xk/zdY8IAYHsh153qUoGf23w=
github.com/mattn/go-runewidth v0.0.24 h1:cpokDiIn0MGnhdHwuWnJBITySJ20QyNGnY2kR/ay2DU=
github.com/mattn/go-runewidth v0.0.24/go.mod h1:XBkDxAl56ILZc9knddidhrOlY5R/pDhgLpndooCuJAs=
github.com/mattn/go-shellwords v1.0.13 h1:DC0OMEpGjm6LfNFU4ckYcvbQKyp2vE8atyFGXNtDcf4=
@@ -1510,8 +1546,8 @@ github.com/mitchellh/mapstructure v1.5.1-0.20220423185008-bf980b35cac4 h1:BpfhmL
github.com/mitchellh/mapstructure v1.5.1-0.20220423185008-bf980b35cac4/go.mod h1:bFUtVrKA4DC2yAKiSyO/QUcy7e+RRV2QTWOzhPopBRo=
github.com/mmcloughlin/geohash v0.9.0 h1:FihR004p/aE1Sju6gcVq5OLDqGcMnpBY+8moBqIsVOs=
github.com/mmcloughlin/geohash v0.9.0/go.mod h1:oNZxQo5yWJh0eMQEP/8hwQuVx9Z9tjwFUqcTB1SmG0c=
github.com/moby/buildkit v0.31.1 h1:j3p55abBl4kiXXPZgYX+6zWgB2aefqHXoPown12fIzU=
github.com/moby/buildkit v0.31.1/go.mod h1:YM5iNEbNCc6L1Zt3YWFB/aXNLufvf4Rcu0DPlc9HwQg=
github.com/moby/buildkit v0.31.0 h1:hMUAbQGgjtzJDDOZ6o7MQk5XBZkBTyzLWEvnjguHHQI=
github.com/moby/buildkit v0.31.0/go.mod h1:YM5iNEbNCc6L1Zt3YWFB/aXNLufvf4Rcu0DPlc9HwQg=
github.com/moby/docker-image-spec v1.3.1 h1:jMKff3w6PgbfSa69GfNg+zN/XLhfXJGnEx3Nl2EsFP0=
github.com/moby/docker-image-spec v1.3.1/go.mod h1:eKmb5VW8vQEh/BAr2yvVNvuiJuY6UIocYsFu/DxxRpo=
github.com/moby/go-archive v0.3.0 h1:nos4BtzzUIqB406BgQnWGMI4qib9BZ8XUHU+ucv/n1c=
@@ -1558,6 +1594,7 @@ github.com/mschoch/smat v0.2.0/go.mod h1:kc9mz7DoBKqDyiRL7VZN8KvXQMWeTaVnttLRXOl
github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822 h1:C3w9PqII01/Oq1c1nUAm88MOHcQC9l5mIlSMApZMrHA=
github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822/go.mod h1:+n7T8mK8HuQTcFwEeznm/DIxMOiR9yIdICNftLE1DvQ=
github.com/mwitkow/go-conntrack v0.0.0-20161129095857-cc309e4a2223/go.mod h1:qRWi+5nqEBWmkhHvq77mSJWrCKwh8bxhgT7d/eI7P4U=
github.com/mwitkow/go-conntrack v0.0.0-20190716064945-2f068394615f/go.mod h1:qRWi+5nqEBWmkhHvq77mSJWrCKwh8bxhgT7d/eI7P4U=
github.com/nats-io/jwt/v2 v2.8.1 h1:V0xpGuD/N8Mi+fQNDynXohVvp7ZztevW5io8CUWlPmU=
github.com/nats-io/jwt/v2 v2.8.1/go.mod h1:nWnOEEiVMiKHQpnAy4eXlizVEtSfzacZ1Q43LIRavZg=
github.com/nats-io/nats-server/v2 v2.11.15 h1:StSf9TINInaZtr4oww2+kXmfwa9SkN//g/LwS19/UJ0=
@@ -1615,8 +1652,8 @@ github.com/pascaldekloe/goe v0.1.0/go.mod h1:lzWF7FIEvWOWxwDKqyGYQf6ZUaNfKdP144T
github.com/patrickmn/go-cache v2.1.0+incompatible h1:HRMgzkcYKYpi3C8ajMPV8OFXaaRUnok+kx1WdO15EQc=
github.com/patrickmn/go-cache v2.1.0+incompatible/go.mod h1:3Qf8kWWT7OJRJbdiICTKqZju1ZixQ/KpMGzzAfe6+WQ=
github.com/pborman/getopt v0.0.0-20170112200414-7148bc3a4c30/go.mod h1:85jBQOZwpVEaDAr341tbn15RS4fCAsIst0qp7i8ex1o=
github.com/pelletier/go-toml/v2 v2.4.3 h1:GTRvJQutkOSftxIFD5xw9aepkYNuPWmVJpffdDPYVpY=
github.com/pelletier/go-toml/v2 v2.4.3/go.mod h1:2gIqNv+qfxSVS7cM2xJQKtLSTLUE9V8t9Stt+h56mCY=
github.com/pelletier/go-toml/v2 v2.4.1 h1:j5OMOImsH+j2k7GJ5YO+RxfWwohNiH6t5zB/+h3bagc=
github.com/pelletier/go-toml/v2 v2.4.1/go.mod h1:2gIqNv+qfxSVS7cM2xJQKtLSTLUE9V8t9Stt+h56mCY=
github.com/pengsrc/go-shared v0.2.1-0.20190131101655-1999055a4a14 h1:XeOYlK9W1uCmhjJSsY78Mcuh7MVkNjTzmHx1yBzizSU=
github.com/pengsrc/go-shared v0.2.1-0.20190131101655-1999055a4a14/go.mod h1:jVblp62SafmidSkvWrXyxAme3gaTfEtWwRPGz5cpvHg=
github.com/peterh/liner v1.2.2 h1:aJ4AOodmL+JxOZZEL2u9iJf8omNRpqHc/EbrK+3mAXw=
@@ -1680,6 +1717,8 @@ github.com/pquerna/otp v1.5.0/go.mod h1:dkJfzwRKNiegxyNb54X/3fLwhCynbMspSyWKnvi1
github.com/prometheus/client_golang v0.9.1/go.mod h1:7SWBe2y4D6OKWSNQJUaRYU/AaXPKyh/dDVn+NZz0KFw=
github.com/prometheus/client_golang v1.0.0/go.mod h1:db9x61etRT2tGnBNRi70OPL5FsnadC4Ky3P0J6CfImo=
github.com/prometheus/client_golang v1.4.0/go.mod h1:e9GMxYsXl05ICDXkRhurwBS4Q3OK1iX/F2sw+iXX5zU=
github.com/prometheus/client_golang v1.7.1/go.mod h1:PY5Wy2awLA44sXw4AOSfFBetzPP4j5+D6mVACh+pe2M=
github.com/prometheus/client_golang v1.11.1/go.mod h1:Z6t4BnS23TR94PD6BsDNk8yVqroYurpAkEiz0P2BEV0=
github.com/prometheus/client_golang v1.24.1 h1:JnJkREXzWxUdCuPFpIWZiPispT9xVV59uiuyR2bPlnU=
github.com/prometheus/client_golang v1.24.1/go.mod h1:F+oSRECHg4sse5ucfYpYDeIv/hu68Zo0uoHKetWnzcE=
github.com/prometheus/client_model v0.0.0-20180712105110-5c3871d89910/go.mod h1:MbSGuTsp3dbXC40dX6PRTWyKYBIrTGTE9sqQNg2J8bo=
@@ -1691,13 +1730,26 @@ github.com/prometheus/client_model v0.6.3 h1:O0jaTVAYNxTHYInEPFJt5I3+sN8zqBtVMPT
github.com/prometheus/client_model v0.6.3/go.mod h1:gpN5P9S7Rr6Yr92PiQ+Ixvhf6JZEkF1dnxsYL2aPBEM=
github.com/prometheus/common v0.4.1/go.mod h1:TNfzLD0ON7rHzMJeJkieUDPYmFC7Snx/y86RQel1bk4=
github.com/prometheus/common v0.9.1/go.mod h1:yhUN8i9wzaXS3w1O07YhxHEBxD+W35wd8bs7vj7HSQ4=
github.com/prometheus/common v0.71.0 h1:9KDAKb7Mj3HEVKyFCK6Dc/HIwlBzZIN2l7/lrHl3KK8=
github.com/prometheus/common v0.71.0/go.mod h1:CLJ5H8TEsGX8bl31BdMkfhIZ+QmZ9tBPPotUxUbfcmk=
github.com/prometheus/common v0.10.0/go.mod h1:Tlit/dnDKsSWFlCLTWaA1cyBgKHSMdTB80sz/V91rCo=
github.com/prometheus/common v0.26.0/go.mod h1:M7rCNAaPfAosfx8veZJCuw84e35h3Cfd9VFqTh1DIvc=
github.com/prometheus/common v0.70.1 h1:1HvjP4D5oL3t8RsPlwxA9onvvStjtIHYE5XuuwOi/PY=
github.com/prometheus/common v0.70.1/go.mod h1:VdFUQDMZK3VLkurFUVhia6uys/0suUp86TJz5qbJRhc=
github.com/prometheus/procfs v0.0.0-20181005140218-185b4288413d/go.mod h1:c3At6R/oaqEKCNdg8wHV1ftS6bRYblBhIjjI8uT2IGk=
github.com/prometheus/procfs v0.0.2/go.mod h1:TjEm7ze935MbeOT/UhFTIMYKhuLP4wbCsTZCD3I8kEA=
github.com/prometheus/procfs v0.0.8/go.mod h1:7Qr8sr6344vo1JqZ6HhLceV9o3AJ1Ff+GxbHq6oeK9A=
github.com/prometheus/procfs v0.1.3/go.mod h1:lV6e/gmhEcM9IjHGsFOCxxuZ+z1YqCvr4OA4YeYWdaU=
github.com/prometheus/procfs v0.6.0/go.mod h1:cz+aTbrPOrUb4q7XlbU9ygM+/jj0fzG6c1xBZuNvfVA=
github.com/prometheus/procfs v0.22.0 h1:6q9+/JL9IKAPbCmBrv9n5O5Ty3NKnciV5X7YGw0oics=
github.com/prometheus/procfs v0.22.0/go.mod h1:CvmFr/GVhIjIvWJZW3tgkODBQMRIf0EyWMQLHCHab58=
github.com/pterm/pterm v0.12.27/go.mod h1:PhQ89w4i95rhgE+xedAoqous6K9X+r6aSOI2eFF7DZI=
github.com/pterm/pterm v0.12.29/go.mod h1:WI3qxgvoQFFGKGjGnJR849gU0TsEOvKn5Q8LlY1U7lg=
github.com/pterm/pterm v0.12.30/go.mod h1:MOqLIyMOgmTDz9yorcYbcw+HsgoZo3BQfg2wtl3HEFE=
github.com/pterm/pterm v0.12.31/go.mod h1:32ZAWZVXD7ZfG0s8qqHXePte42kdz8ECtRyEejaWgXU=
github.com/pterm/pterm v0.12.33/go.mod h1:x+h2uL+n7CP/rel9+bImHD5lF3nM9vJj80k9ybiiTTE=
github.com/pterm/pterm v0.12.36/go.mod h1:NjiL09hFhT/vWjQHSj1athJpx6H8cjpHXNAK5bUw8T8=
github.com/pterm/pterm v0.12.40/go.mod h1:ffwPLwlbXxP+rxT0GsgDTzS3y3rmpAO1NMjUkGTYf8s=
github.com/pterm/pterm v0.12.83 h1:ie+YmGmA727VuhxBlyGr74Ks+7McV6kT99IB8EU80aA=
github.com/pterm/pterm v0.12.83/go.mod h1:xlgc6bFWyJIMtmLJvGim+L7jhSReilOlOnodeIYe4Tk=
github.com/putdotio/go-putio/putio v0.0.0-20200123120452-16d982cac2b8 h1:Y258uzXU/potCYnQd1r6wlAnoMB68BiCkCcCnKx1SH8=
github.com/putdotio/go-putio/putio v0.0.0-20200123120452-16d982cac2b8/go.mod h1:bSJjRokAHHOhA+XFxplld8w2R/dXLH7Z3BZ532vhFwU=
github.com/puzpuzpuz/xsync/v3 v3.5.1 h1:GJYJZwO6IdxN/IKbneznS6yPkVC+c3zyY/j19c++5Fg=
@@ -1706,8 +1758,8 @@ github.com/quic-go/qpack v0.6.0 h1:g7W+BMYynC1LbYLSqRt8PBg5Tgwxn214ZZR34VIOjz8=
github.com/quic-go/qpack v0.6.0/go.mod h1:lUpLKChi8njB4ty2bFLX2x4gzDqXwUpaO1DP9qMDZII=
github.com/quic-go/quic-go v0.59.0 h1:OLJkp1Mlm/aS7dpKgTc6cnpynnD2Xg7C1pwL6vy/SAw=
github.com/quic-go/quic-go v0.59.0/go.mod h1:upnsH4Ju1YkqpLXC305eW3yDZ4NfnNbmQRCMWS58IKU=
github.com/rabbitmq/amqp091-go v1.15.0 h1:LEQL4/yp48/Wigt6A6XOu18RQRo8ZHtB5I/KZJn+gkw=
github.com/rabbitmq/amqp091-go v1.15.0/go.mod h1:Hy4jKW5kQART1u+JkDTF9YYOQUHXqMuhrgxOEeS7G4o=
github.com/rabbitmq/amqp091-go v1.14.0 h1:RSaT7aOKt/OrkVUyswPDW29lnRz9psuGmfZFBmLqLek=
github.com/rabbitmq/amqp091-go v1.14.0/go.mod h1:Hy4jKW5kQART1u+JkDTF9YYOQUHXqMuhrgxOEeS7G4o=
github.com/rclone/Proton-API-Bridge v1.0.5 h1:K1++Qtk3PvgkiCCiv6Pahju1TMOzKY6VSwiwT7XLAVc=
github.com/rclone/Proton-API-Bridge v1.0.5/go.mod h1:vCeOPhlXzevN0AFojgh1zsjhetiShy/ArvJ/xkFUDWk=
github.com/rclone/go-proton-api v1.0.4 h1:AJW0e9pB4j0hVK4WqyGErFwaI+5MUQWPCtj5FYYxtPg=
@@ -1734,6 +1786,7 @@ github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec h1:W09IVJc94
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec/go.mod h1:qqbHyh8v60DhA7CoWK5oRCqLrMHRGoxYCSS9EjAz6Eo=
github.com/rfjakob/eme v1.2.0 h1:8dAHL+WVAw06+7DkRKnRiFp1JL3QjcJEZFqDnndUaSI=
github.com/rfjakob/eme v1.2.0/go.mod h1:cVvpasglm/G3ngEfcfT/Wt0GwhkuO32pf/poW6Nyk1k=
github.com/rivo/uniseg v0.2.0/go.mod h1:J6wj4VEh+S6ZtnVlnTBMWIodfgj8LQOQFoIToxlJtxc=
github.com/rivo/uniseg v0.4.7 h1:WUdvkW8uEhrYfLC4ZzdpI2ztxP1I582+49Oc5Mq64VQ=
github.com/rivo/uniseg v0.4.7/go.mod h1:FN3SvrM+Zdj16jyLfmOkMNblXMcoc8DfTHruCPUcx88=
github.com/rogpeppe/fastuuid v1.2.0/go.mod h1:jVj6XXZzXRy/MSR5jhDC/2q6DgLz+nrA6LYCDYWNEvQ=
@@ -1787,6 +1840,7 @@ github.com/sigstore/sigstore-go v1.2.1/go.mod h1:I8BqVwAb/SaQJ5pBu5IDFY+ksq8O/1/
github.com/sirupsen/logrus v1.2.0/go.mod h1:LxeOpSwHxABJmUn/MG1IvRgCAasNZTLOkJPxbbu5VWo=
github.com/sirupsen/logrus v1.4.2/go.mod h1:tLMulIdttU9McNUspp0xgXVQah82FyeX6MwdIuYE2rE=
github.com/sirupsen/logrus v1.5.0/go.mod h1:+F7Ogzej0PZc/94MaYx/nvG9jOFMD2osvC3s+Squfpo=
github.com/sirupsen/logrus v1.6.0/go.mod h1:7uNnSEd1DgxDLC74fIahvMZmmYsHGZGEOFrfsX/uA88=
github.com/sirupsen/logrus v1.7.0/go.mod h1:yWOB1SBYBC5VeMP7gHvWumXLIWorT60ONWic61uBYv0=
github.com/sirupsen/logrus v1.9.4 h1:TsZE7l11zFCLZnZ+teH4Umoq5BhEIfIzfRDZ1Uzql2w=
github.com/sirupsen/logrus v1.9.4/go.mod h1:ftWc9WdOfJ0a92nsE2jF5u5ZwH8Bv2zdeOC42RjbV2g=
@@ -1843,8 +1897,8 @@ github.com/stretchr/testify v1.8.1/go.mod h1:w2LPCIKwWwSfY2zedu0+kehJoqGctiVI29o
github.com/stretchr/testify v1.8.2/go.mod h1:w2LPCIKwWwSfY2zedu0+kehJoqGctiVI29o6fzry7u4=
github.com/stretchr/testify v1.8.3/go.mod h1:sz/lmYIOXD/1dqDmKjjqLyZ2RngseejIcXlSw2iwfAo=
github.com/stretchr/testify v1.8.4/go.mod h1:sz/lmYIOXD/1dqDmKjjqLyZ2RngseejIcXlSw2iwfAo=
github.com/stretchr/testify v1.12.1 h1:EuwCh5fleGS7H32xRwO3wRGT7DxrDhLAT6FF8MpWDWE=
github.com/stretchr/testify v1.12.1/go.mod h1:MDEgiDPPsNp5cuIrHPPCyornHKgEVbtFUmoNlxoYthg=
github.com/stretchr/testify v1.11.1 h1:7s2iGBzp5EwR7/aIZr8ao5+dra3wiQyKjjFuvgVKu7U=
github.com/stretchr/testify v1.11.1/go.mod h1:wZwfW3scLgRK+23gO65QZefKpKQRnfz6sD981Nm4B6U=
github.com/stvp/tempredis v0.0.0-20181119212430-b82af8480203 h1:QVqDTf3h2WHt08YuiTGPZLls0Wq99X9bWd0Q5ZSBesM=
github.com/stvp/tempredis v0.0.0-20181119212430-b82af8480203/go.mod h1:oqN97ltKNihBbwlX8dLpwxCl3+HnXKV/R0e+sRLd9C8=
github.com/subosito/gotenv v1.6.0 h1:9NlTDc1FTs4qu0DDq7AEtTPNw6SVm7uBMsUCUjABIf8=
@@ -1864,8 +1918,8 @@ github.com/tarantool/go-iproto v1.1.0 h1:HULVOIHsiehI+FnHfM7wMDntuzUddO09DKqu2Wn
github.com/tarantool/go-iproto v1.1.0/go.mod h1:LNCtdyZxojUed8SbOiYHoc3v9NvaZTB7p96hUySMlIo=
github.com/tarantool/go-option v1.1.0 h1:ShoOhNsdL41sRpm4hXCRDjV8H0WzPkd4UnKhLKbW//w=
github.com/tarantool/go-option v1.1.0/go.mod h1:hMr9z2JXOWlgdCBpCPSL2nwp8718GKYvNBJ+ZuzJbCo=
github.com/tarantool/go-tarantool/v3 v3.0.2 h1:9ZtHllun80QX7KS9tqd3dXa+QAu2BfqFtX1ZWcNnaW8=
github.com/tarantool/go-tarantool/v3 v3.0.2/go.mod h1:TXxLWhUCgdxXFfelnTSkq+goKRTRj660zxq4/WXPe8k=
github.com/tarantool/go-tarantool/v3 v3.0.1 h1:vaUX4xmVmXh2dIJ/LqlX1MXK3iYqAqV6YiE54Wwl/qg=
github.com/tarantool/go-tarantool/v3 v3.0.1/go.mod h1:TXxLWhUCgdxXFfelnTSkq+goKRTRj660zxq4/WXPe8k=
github.com/testcontainers/testcontainers-go v0.44.0 h1:/Fwh6HY1mIikhnm9e7HwoxGycx0lzRAE0f5VQpjFxzI=
github.com/testcontainers/testcontainers-go v0.44.0/go.mod h1:IcnwQrYTO86xHXu5bvMaBH7ATlbS3Qn1M1QWW3c66rE=
github.com/testcontainers/testcontainers-go/modules/compose v0.44.0 h1:8YcW51jhgpkkiRVe10Wj9TCBthJmoNpU2fK5WSf7TQ8=
@@ -1874,8 +1928,9 @@ github.com/the42/cartconvert v0.0.0-20131203171324-aae784c392b8 h1:I4DY8wLxJXCrM
github.com/the42/cartconvert v0.0.0-20131203171324-aae784c392b8/go.mod h1:fWO/msnJVhHqN1yX6OBoxSyfj7TEj1hHiL8bJSQsK30=
github.com/tiancaiamao/gp v0.0.0-20221230034425-4025bc8a4d4a h1:J/YdBZ46WKpXsxsW93SG+q0F8KI+yFrcIDT4c/RNoc4=
github.com/tiancaiamao/gp v0.0.0-20221230034425-4025bc8a4d4a/go.mod h1:h4xBhSNtOeEosLJ4P7JyKXX7Cabg7AVkWCK5gV2vOrM=
github.com/tidwall/gjson v1.19.0 h1:xwxm7n691Uf3u5OFjzngavjGTh55KX5q/9w9xHW88JU=
github.com/tidwall/gjson v1.19.0/go.mod h1:V37/opeE/JbLUOfH0QTXiNez2l0RUjYUhpT4szFQAfc=
github.com/tidwall/gjson v1.18.0 h1:FIDeeyB800efLX89e5a8Y0BNH+LOngJyGrIWxG2FKQY=
github.com/tidwall/gjson v1.18.0/go.mod h1:/wbyibRr2FHMks5tjHJ5F8dMZh3AcwJEMf5vlfC0lxk=
github.com/tidwall/match v1.1.1/go.mod h1:eRSPERbgtNPcGhD8UCthc6PmLEQXEWd3PRB5JTxsfmM=
github.com/tidwall/match v1.2.0 h1:0pt8FlkOwjN2fPt4bIl4BoNxb98gGHN2ObFEDkrfZnM=
github.com/tidwall/match v1.2.0/go.mod h1:eRSPERbgtNPcGhD8UCthc6PmLEQXEWd3PRB5JTxsfmM=
github.com/tidwall/pretty v1.2.0 h1:RWIZEg2iJ8/g6fDDYzMpobmaoGh5OLl4AXtGUGPcqCs=
@@ -1910,8 +1965,8 @@ github.com/tsuna/gohbase v0.0.0-20201125011725-348991136365/go.mod h1:zj0GJHGvyf
github.com/tv42/httpunix v0.0.0-20150427012821-b75d8614f926/go.mod h1:9ESjWnEqriFuLhtthL60Sar/7RFoluCcXsuvEwTV5KM=
github.com/twitchyliquid64/golang-asm v0.15.1 h1:SU5vSMR7hnwNxj24w34ZyCi/FmDZTkS4MhqMhdFk5YI=
github.com/twitchyliquid64/golang-asm v0.15.1/go.mod h1:a1lVb/DtPvCB8fslRZhAngC2+aY1QWCk3Cedj/Gdt08=
github.com/twmb/avro v1.9.0 h1:JSiqewo3AANj7vlVCQXBsIZTA3LiiBble7Xp8T4lbtA=
github.com/twmb/avro v1.9.0/go.mod h1:X0fT1dY2xcbV4YuCE4mYro+qljHl4kUF5uA/2z1rgSk=
github.com/twmb/avro v1.8.0 h1:UMWLg+nH4P3yad5Om7yFSohYLy2RG1s7BcFFiOvmK9Q=
github.com/twmb/avro v1.8.0/go.mod h1:X0fT1dY2xcbV4YuCE4mYro+qljHl4kUF5uA/2z1rgSk=
github.com/twmb/murmur3 v1.1.8 h1:8Yt9taO/WN3l08xErzjeschgZU2QSrwm1kclYq+0aRg=
github.com/twmb/murmur3 v1.1.8/go.mod h1:Qq/R7NUyOfr65zD+6Q5IHKsJLwP7exErjN6lyyq3OSQ=
github.com/twpayne/go-geom v1.6.1 h1:iLE+Opv0Ihm/ABIcvQFGIiFBXd76oBIar9drAwHFhR4=
@@ -1977,6 +2032,9 @@ github.com/xeipuuv/gojsonschema v1.2.0 h1:LhYJRs+L4fBtjZUfuSZIKGeVu0QRy8e5Xi7D17
github.com/xeipuuv/gojsonschema v1.2.0/go.mod h1:anYRn/JVcOK2ZgGU+IjEV4nwlhoK5sQluxsYJ78Id3Y=
github.com/xhit/go-str2duration/v2 v2.1.0 h1:lxklc02Drh6ynqX+DdPyp5pCKLUQpRT8bp8Ydu2Bstc=
github.com/xhit/go-str2duration/v2 v2.1.0/go.mod h1:ohY8p+0f07DiV6Em5LKB0s2YpLtXVyJfNt1+BlmyAsU=
github.com/xo/terminfo v0.0.0-20210125001918-ca9a967f8778/go.mod h1:2MuV+tbUrU1zIOPMxZ5EncGwgmMJsa+9ucAQZXxsObs=
github.com/xo/terminfo v0.0.0-20220910002029-abceb7e1c41e h1:JVG44RsyaB9T2KIHavMF/ppJZNG9ZpyihvCd0w101no=
github.com/xo/terminfo v0.0.0-20220910002029-abceb7e1c41e/go.mod h1:RbqR21r5mrJuqunuUZ/Dhy/avygyECGrLceyNeo4LiM=
github.com/xyproto/randomstring v1.0.5 h1:YtlWPoRdgMu3NZtP45drfy1GKoojuR7hmRcnhZqKjWU=
github.com/xyproto/randomstring v1.0.5/go.mod h1:rgmS5DeNXLivK7YprL0pY+lTuhNQW3iGxZ18UQApw/E=
github.com/yandex-cloud/go-genproto v0.0.0-20211115083454-9ca41db5ed9e h1:9LPdmD1vqadsDQUva6t2O9MbnyvoOgo8nFNPaOIH5U8=
@@ -2045,50 +2103,50 @@ go.opencensus.io v0.24.0 h1:y73uSU6J157QMP2kn2r30vwW1A2W2WFwSCGnAVxeaD0=
go.opencensus.io v0.24.0/go.mod h1:vNK8G9p7aAivkbmorf4v+7Hgx+Zs0yY+0fOtgBfjQKo=
go.opentelemetry.io/auto/sdk v1.2.1 h1:jXsnJ4Lmnqd11kwkBV2LgLoFMZKizbCi5fNZ/ipaZ64=
go.opentelemetry.io/auto/sdk v1.2.1/go.mod h1:KRTj+aOaElaLi+wW1kO/DZRXwkF4C5xPbEe3ZiIhN7Y=
go.opentelemetry.io/contrib/detectors/gcp v1.45.0 h1:9jR0ZPRok9ryaOQ2Wx8rg5F7Aon59mxrqbVI60/vlBk=
go.opentelemetry.io/contrib/detectors/gcp v1.45.0/go.mod h1:VSme3o2fvSg5bVg0dRzyHaj4Z5EVhG+g2Fde6LKzmQA=
go.opentelemetry.io/contrib/detectors/gcp v1.44.0 h1:NmLfL734pJhM0JKaYd2Y28+nY9dPRWYAAbxhRCrKXPw=
go.opentelemetry.io/contrib/detectors/gcp v1.44.0/go.mod h1:tNAsgd8avTGke1+MndXlU5Cru4PQ9Ai/cCNWQv/ZJ/s=
go.opentelemetry.io/contrib/instrumentation/google.golang.org/grpc/otelgrpc v0.69.0 h1:2yEATaop1/a1I4psnSLgWVPLWwCzkqWakgJy7xTDVy0=
go.opentelemetry.io/contrib/instrumentation/google.golang.org/grpc/otelgrpc v0.69.0/go.mod h1:D7J12YRapIekYyPWgGPlA/23pRmpSEZC5xJC/TTLI9U=
go.opentelemetry.io/contrib/instrumentation/net/http/httptrace/otelhttptrace v0.69.0 h1:MCcYL7J6Vt/X0kjqbMZkekCmwsurbQRbL69vkiye2lk=
go.opentelemetry.io/contrib/instrumentation/net/http/httptrace/otelhttptrace v0.69.0/go.mod h1:3jnStNwSufK+f5ktjL4EPcwtig4rtd81NS70lqHuXl8=
go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.70.0 h1:LMuyCAyfalSjDyjdC65nK6N0zoTT63+E/u95X0JovZI=
go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.70.0/go.mod h1:085m8qbm4hgc8rZWGDEa4vmyyo2c3nPxUslYUKUIU04=
go.opentelemetry.io/otel v1.46.0 h1:FHt5/CDyVxi/8IM1CH7VE/rRgq3kLHa2mSTVMO8AWyc=
go.opentelemetry.io/otel v1.46.0/go.mod h1:Gj3SEScelsNC45tp4nSxRYlS+f5iez7W8XPMCt905kE=
go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.69.0 h1:8tvICD4vSTOOsNrsI4Ljf6C+6UKvpTEH5XY3JMoyPoo=
go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.69.0/go.mod h1:z9+yiacE0IHRqM4qFfkbt/JYlmYXgss8GY/jXoNuPJI=
go.opentelemetry.io/otel v1.45.0 h1:pdrWmLHofpubmArBv1LgFSv1Z0Ie/ppdZzu+kUN5EeU=
go.opentelemetry.io/otel v1.45.0/go.mod h1:XZxIqPapzEYnhNSScF5DIqXhm/rYi0FzCe2XddAwZfQ=
go.opentelemetry.io/otel/exporters/otlp/otlpmetric/otlpmetricgrpc v1.44.0 h1:SUplec5dp06reu1zaXmOXdvqH398taqrDXqUl99jxSc=
go.opentelemetry.io/otel/exporters/otlp/otlpmetric/otlpmetricgrpc v1.44.0/go.mod h1:ho2g4N+ane+swq5I/VBkKWnRDY4kUINH3FuqyZqX/Ug=
go.opentelemetry.io/otel/exporters/otlp/otlpmetric/otlpmetrichttp v1.44.0 h1:RuynHbfU8JUEw7DyONgkVYg2SVtsoF28y0LGIr69jgA=
go.opentelemetry.io/otel/exporters/otlp/otlpmetric/otlpmetrichttp v1.44.0/go.mod h1:qZF+/lBs71APw8mlnEZcqZHMzqrYrsFiJOv83lX1OGo=
go.opentelemetry.io/otel/exporters/otlp/otlptrace v1.45.0 h1:QRefszxJmfPdjXUUm3j6iDzY03mTPXMjqErFqQ67vUg=
go.opentelemetry.io/otel/exporters/otlp/otlptrace v1.45.0/go.mod h1:Tiz03lTBVBrm7eWZBOidzEaYaJa8tjwGUGv6d8mlTyk=
go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracegrpc v1.45.0 h1:fG5MCxGz8+2VtrN/WgqSpJFctVz24gpxj8CxkKmc8Ww=
go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracegrpc v1.45.0/go.mod h1:BmAYTn+3ysbRe+IU2msxmf5Rx3g6DHvex+tWI3LdhYI=
go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracehttp v1.45.0 h1:QBajQ2SrwQijzHyZbQlPsuIzpl/ll8DY6wPWsajeGcI=
go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracehttp v1.45.0/go.mod h1:08ZQLjrPLQ6R4kAXvuOvODEer5Yh4CoFvll5qB2BCI8=
go.opentelemetry.io/otel/exporters/otlp/otlptrace v1.44.0 h1:4YsVu3B8+3qtWYYrsUYgn0OG78pN0rnNPRGX4SbokQI=
go.opentelemetry.io/otel/exporters/otlp/otlptrace v1.44.0/go.mod h1:+wnlSn0mD1ADVMe3v9Z/WIaiz6q6gL2J/ejaAmdmv80=
go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracegrpc v1.44.0 h1:qazEJlUOQzhCpzQpFETGby7EdqjI1wsd0W+6Gg1SCTU=
go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracegrpc v1.44.0/go.mod h1:fOD2Yefuxixkx3ahVNf0O/PERb6r4OlbxfATVnYvzCo=
go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracehttp v1.44.0 h1:lgh3PiVrRUWMLOVSkQicxzZll5NjF1r+AtsX1XRIHw0=
go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracehttp v1.44.0/go.mod h1:5Cnhth3m/AgOeTgE3ex12pPmiu/gGtZit03kSzx9X7s=
go.opentelemetry.io/otel/exporters/stdout/stdoutmetric v1.44.0 h1:hqxVTu/GtBF+vJ8d1fzW7fRxZFvgoDjWcxwwCaFDYpU=
go.opentelemetry.io/otel/exporters/stdout/stdoutmetric v1.44.0/go.mod h1:z5fVEF4X5v0ESvlJqBrrFlBVoj5EQuefZpzsu7R+x5Q=
go.opentelemetry.io/otel/exporters/zipkin v1.45.0 h1:KN3btaILMTxR4QDHVGAO87lq5ButzK7l+kIfLuxQ1oA=
go.opentelemetry.io/otel/exporters/zipkin v1.45.0/go.mod h1:yNcodmUclM4InyWoOwX/YW4Jri0Gj5FWAlM+NqCrtqY=
go.opentelemetry.io/otel/metric v1.46.0 h1:yBnkXvgV7AXFILZc5K6IZe/CBFF3OS7BJ8ov6/lj0K8=
go.opentelemetry.io/otel/metric v1.46.0/go.mod h1:iPmdWqifKUdzziPkvvzIJXITl56fQx2mGM/DHLB3/2o=
go.opentelemetry.io/otel/metric/x v0.68.0 h1:TA/cBT23D3MnxYPwHL7YFOdYGdx0A0v+s7Mzotpd1dU=
go.opentelemetry.io/otel/metric/x v0.68.0/go.mod h1:agudOmvWhwUTjgibWDzxD2PoWYnpw5Ht5jISYOD2Hd4=
go.opentelemetry.io/otel/sdk v1.46.0 h1:h5CNQQjEbuQXY/JfZtgt3i7HVFV3aHPO2OAwO2eTYPI=
go.opentelemetry.io/otel/sdk v1.46.0/go.mod h1:GAERFXFt5SYCEB+YiKUbMBeza6UaDH7GmGOZEfh2gSM=
go.opentelemetry.io/otel/sdk/metric v1.46.0 h1:0piZ26EG4RBfebb2jhDH6ERCYHoVWduc3kLgPCwSnSE=
go.opentelemetry.io/otel/sdk/metric v1.46.0/go.mod h1:I1PbKrdVc8Qu8HYVDNtqVIwLwjNrhsV/uFuxfwg8mO4=
go.opentelemetry.io/otel/trace v1.46.0 h1:OULy7ccdJnZtJ0UDYFOIGaCmiWzJ8Vi2G/Rsu60qs1c=
go.opentelemetry.io/otel/trace v1.46.0/go.mod h1:J7GAXweO77XSFkB/rmAqk9D6ihszhFjLU+d9WuUxDLI=
go.opentelemetry.io/otel/exporters/zipkin v1.36.0 h1:s0n95ya5tOG03exJ5JySOdJFtwGo4ZQ+KeY7Zro4CLI=
go.opentelemetry.io/otel/exporters/zipkin v1.36.0/go.mod h1:m9wRxtKA2MZ1HcnNC4BKI+9aYe434qRZTCvI7QGUN7Y=
go.opentelemetry.io/otel/metric v1.45.0 h1:7Eg1uH7CJ5cXv9is6tnBe1FI6rj1nwUdbFypRm3br/M=
go.opentelemetry.io/otel/metric v1.45.0/go.mod h1:HAPbm1nd3p1PmFH7v2dR+6BjXxw+Lq4a2+pndMAm08s=
go.opentelemetry.io/otel/metric/x v0.67.0 h1:PcicCNZFkZ4bXfSooXdo3WN7RBOVOtjVdo1wD358Uns=
go.opentelemetry.io/otel/metric/x v0.67.0/go.mod h1:FBjCWZe6wgcqxcMtjdGiClDKXb2YxxXii0CXftE4QtI=
go.opentelemetry.io/otel/sdk v1.45.0 h1:4VVSMgQ83dUgW2aoX5f6JgLvHwIvzcuLnF9lUdCSpCw=
go.opentelemetry.io/otel/sdk v1.45.0/go.mod h1:Sr40LgXV7DsKMMJMKOhUWOgMWTfAaqvm2kF0g7ilwuA=
go.opentelemetry.io/otel/sdk/metric v1.45.0 h1:oVFszMfyj1Am6s24Vtc7wBb8BKLcwepJjNEYILuiE3o=
go.opentelemetry.io/otel/sdk/metric v1.45.0/go.mod h1:vUWUxDZvu1WVRj8JA8S0AdhsPrZoDpA2DdZauIh4mDA=
go.opentelemetry.io/otel/trace v1.45.0 h1:l/mP6Uv7oNO7/TblbhpbgMidxhq1uO/rPsikOyVhxag=
go.opentelemetry.io/otel/trace v1.45.0/go.mod h1:qoJJA2xNMnxRrdISU/kLtfUH2wNeQbiv+jhs/CxI8bc=
go.opentelemetry.io/proto/otlp v0.7.0/go.mod h1:PqfVotwruBrMGOCsRd/89rSnXhoiJIqeYNgFYFoEGnI=
go.opentelemetry.io/proto/otlp v0.15.0/go.mod h1:H7XAot3MsfNsj7EXtrA2q5xSNQ10UqI405h3+duxN4U=
go.opentelemetry.io/proto/otlp v0.19.0/go.mod h1:H7XAot3MsfNsj7EXtrA2q5xSNQ10UqI405h3+duxN4U=
go.opentelemetry.io/proto/otlp v1.11.0 h1:5rrYs0Ykyj50sdU/JU0x8etU+LubXWb+gED6TbEdMIk=
go.opentelemetry.io/proto/otlp v1.11.0/go.mod h1:SmVizdCOAm3XBtG1g1NnOdhW6jtddT72hLMhv8VwA8E=
go.opentelemetry.io/proto/otlp v1.10.0 h1:IQRWgT5srOCYfiWnpqUYz9CVmbO8bFmKcwYxpuCSL2g=
go.opentelemetry.io/proto/otlp v1.10.0/go.mod h1:/CV4QoCR/S9yaPj8utp3lvQPoqMtxXdzn7ozvvozVqk=
go.uber.org/atomic v1.6.0/go.mod h1:sABNBOSYdrvTF6hTgEIbc7YasKWGhgEQZyfxyTvoXHQ=
go.uber.org/atomic v1.7.0/go.mod h1:fEN4uk6kAWBTFdckzkM89CLk9XfWZrxpCo0nPH17wJc=
go.uber.org/atomic v1.9.0/go.mod h1:fEN4uk6kAWBTFdckzkM89CLk9XfWZrxpCo0nPH17wJc=
go.uber.org/atomic v1.12.0 h1:BvcXdFKuviU4fTL/f+SxdQ5qJX/Jix8pAkgdUcb3XOE=
go.uber.org/atomic v1.12.0/go.mod h1:I6c4cg+6HCxRjfjSsYtApoFILnpc0CGUdGkXVqbYVNk=
go.uber.org/atomic v1.11.0 h1:ZvwS0R+56ePWxUNi+Atn9dWONBPp/AUETXlHW0DxSjE=
go.uber.org/atomic v1.11.0/go.mod h1:LUxbIzbOniOlMKjJjyPfpl4v+PKK2cNJn91OQbhoJI0=
go.uber.org/goleak v1.1.10/go.mod h1:8a7PlsEVH3e/a/GLqe5IIrQx6GzcnRmZEufDUTk4A7A=
go.uber.org/goleak v1.1.12/go.mod h1:cwTWslyiVhfpKIDGSZEM2HlOvcqm+tG4zioyIeLoqMQ=
go.uber.org/goleak v1.3.0 h1:2K3zAYmnTNqV73imy9J1T3WC+gmCePx2hEGkimedGto=
@@ -2105,8 +2163,8 @@ go.uber.org/zap v1.27.1 h1:08RqriUEv8+ArZRYSTXy1LeBScaMpVSTBhCeaZYfMYc=
go.uber.org/zap v1.27.1/go.mod h1:GB2qFLM7cTU87MWRP2mPIjqfIDnGu+VIO4V/SdhGo2E=
go.yaml.in/yaml/v2 v2.4.4 h1:tuyd0P+2Ont/d6e2rl3be67goVK4R6deVxCUX5vyPaQ=
go.yaml.in/yaml/v2 v2.4.4/go.mod h1:gMZqIpDtDqOfM0uNfy0SkpRhvUryYH0Z6wdMYcacYXQ=
go.yaml.in/yaml/v3 v3.0.5 h1:N6y/pJk8buWs9NY5ERU2HSMfm+IuD/OtfdAnq6kESPw=
go.yaml.in/yaml/v3 v3.0.5/go.mod h1:HVTZu1O7/Vkt2N+BFy8Zza+lnLsABggaTM2ZpNIGuKg=
go.yaml.in/yaml/v3 v3.0.4 h1:tfq32ie2Jv2UxXFdLJdh3jXuOzWiL1fo0bu/FbuKpbc=
go.yaml.in/yaml/v3 v3.0.4/go.mod h1:DhzuOOF2ATzADvBadXxruRBLzYTpT36CKvDb3+aBEFg=
go.yaml.in/yaml/v4 v4.0.0-rc.6 h1:1h7H1ohdUh93/FyE4YaDa1Zh64K6VVbjF4K6WUxMtH4=
go.yaml.in/yaml/v4 v4.0.0-rc.6/go.mod h1:aZqd9kCMsGL7AuUv/m/PvWLdg5sjJsZ4oHDEnfPPfY0=
gocloud.dev v0.46.0 h1:niIuZwSjMtBx8K+ITB2s5kZullB13PGOS2ZoQPZxQ4Q=
@@ -2135,8 +2193,8 @@ golang.org/x/crypto v0.6.0/go.mod h1:OFC/31mSvZgRz0V1QTNCzfAI1aIRzbiufJtkMIlEp58
golang.org/x/crypto v0.7.0/go.mod h1:pYwdfH91IfpZVANVyUOhSIPZaFoJGxTFbZhFTx+dXZU=
golang.org/x/crypto v0.13.0/go.mod h1:y6Z2r+Rw4iayiXXAIxJIDAJ1zMW4yaTpebo8fPOliYc=
golang.org/x/crypto v0.14.0/go.mod h1:MVFd36DqK4CsrnJYDkBA3VC4m2GkXAM0PvzMCn4JQf4=
golang.org/x/crypto v0.57.0 h1:3ZVCjf8Ggz7zneR/EHRVx68Ctf+2pmIMP2UFhh9cC6M=
golang.org/x/crypto v0.57.0/go.mod h1:Fdz0i5U6CoizGwLda9DttjSk6qlZo25zYNtR+ycvuZA=
golang.org/x/crypto v0.56.0 h1:GUh5Ii4J5jtcseSMiRqr1jXCNHoxjeV9Fmekc2oLy6Y=
golang.org/x/crypto v0.56.0/go.mod h1:OMW5y6CY9l38uPLmxU6l6pwcXp1obtLo3e6gT7gQR2I=
golang.org/x/exp v0.0.0-20180321215751-8460e604b9de/go.mod h1:CJ0aWSM057203Lf6IL+f9T1iT9GByDxfZKAQTCR3kQA=
golang.org/x/exp v0.0.0-20180807140117-3d87b88a115f/go.mod h1:CJ0aWSM057203Lf6IL+f9T1iT9GByDxfZKAQTCR3kQA=
golang.org/x/exp v0.0.0-20190121172915-509febef88a4/go.mod h1:CJ0aWSM057203Lf6IL+f9T1iT9GByDxfZKAQTCR3kQA=
@@ -2296,8 +2354,8 @@ golang.org/x/oauth2 v0.0.0-20221014153046-6fdb5e3db783/go.mod h1:h4gKUeWbJ4rQPri
golang.org/x/oauth2 v0.4.0/go.mod h1:RznEsdpjGAINPTOF0UH/t+xJ75L18YO3Ho6Pyn+uRec=
golang.org/x/oauth2 v0.5.0/go.mod h1:9/XBHVqLaWO3/BRHs5jbpYCnOZVjj5V0ndyaAM7KB4I=
golang.org/x/oauth2 v0.6.0/go.mod h1:ycmewcwgD4Rpr3eZJLSB4Kyyljb3qDh40vJ8STE5HKw=
golang.org/x/oauth2 v0.37.0 h1:JUlcxA8oAtauLfiH8FX2/FkAWHAdi0QtGCGc+hofE98=
golang.org/x/oauth2 v0.37.0/go.mod h1:IxwZNxUULJmpBFf9K/9NTMSIfZZuvuTy1gGxhigP/58=
golang.org/x/oauth2 v0.36.0 h1:peZ/1z27fi9hUOFCAZaHyrpWG5lwe0RJEEEeH0ThlIs=
golang.org/x/oauth2 v0.36.0/go.mod h1:YDBUJMTkDnJS+A4BP4eZBjCqtokkg1hODuPjwiGPO7Q=
golang.org/x/sync v0.0.0-20180314180146-1d60e4601c6f/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
golang.org/x/sync v0.0.0-20181108010431-42b317875d0f/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
golang.org/x/sync v0.0.0-20181221193216-37e7f081c4d4/go.mod h1:RxMgew5VJxzue5/jJTE5uejpjVlOe/izrB70Jof72aM=
@@ -2337,6 +2395,7 @@ golang.org/x/sys v0.0.0-20191001151750-bb3f8db39f24/go.mod h1:h1NjWce9XRLGQEsW7w
golang.org/x/sys v0.0.0-20191026070338-33540a1f6037/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
golang.org/x/sys v0.0.0-20191204072324-ce4227a45e2e/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
golang.org/x/sys v0.0.0-20191228213918-04cbcbbfeed8/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
golang.org/x/sys v0.0.0-20200106162015-b016eb3dc98e/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
golang.org/x/sys v0.0.0-20200113162924-86b910548bc1/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
golang.org/x/sys v0.0.0-20200116001909-b77594299b42/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
golang.org/x/sys v0.0.0-20200122134326-e047566fdf82/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
@@ -2350,6 +2409,8 @@ golang.org/x/sys v0.0.0-20200501052902-10377860bb8e/go.mod h1:h1NjWce9XRLGQEsW7w
golang.org/x/sys v0.0.0-20200511232937-7e40ca221e25/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
golang.org/x/sys v0.0.0-20200515095857-1151b9dac4a9/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
golang.org/x/sys v0.0.0-20200523222454-059865788121/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
golang.org/x/sys v0.0.0-20200615200032-f1bc736245b1/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
golang.org/x/sys v0.0.0-20200625212154-ddb9806d33ae/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
golang.org/x/sys v0.0.0-20200803210538-64077c9b5642/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
golang.org/x/sys v0.0.0-20200905004654-be1d3432aa8f/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
golang.org/x/sys v0.0.0-20200930185726-fdedc70b468f/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
@@ -2370,6 +2431,7 @@ golang.org/x/sys v0.0.0-20210423082822-04245dca01da/go.mod h1:h1NjWce9XRLGQEsW7w
golang.org/x/sys v0.0.0-20210423185535-09eb48e85fd7/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
golang.org/x/sys v0.0.0-20210510120138-977fb7262007/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
golang.org/x/sys v0.0.0-20210514084401-e8d321eab015/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
golang.org/x/sys v0.0.0-20210603081109-ebe580a85c40/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
golang.org/x/sys v0.0.0-20210603125802-9665404d3644/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
golang.org/x/sys v0.0.0-20210615035016-665e8c7367d1/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
golang.org/x/sys v0.0.0-20210616094352-59db8d763f22/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
@@ -2380,6 +2442,7 @@ golang.org/x/sys v0.0.0-20210823070655-63515b42dcdf/go.mod h1:oPkhp1MJrh7nUepCBc
golang.org/x/sys v0.0.0-20210908233432-aa78b53d3365/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
golang.org/x/sys v0.0.0-20210927094055-39ccf1dd6fa6/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
golang.org/x/sys v0.0.0-20211007075335-d3039528d8ac/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
golang.org/x/sys v0.0.0-20211013075003-97ac67df715c/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
golang.org/x/sys v0.0.0-20211019181941-9d821ace8654/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
golang.org/x/sys v0.0.0-20211025201205-69cdffdb9359/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
golang.org/x/sys v0.0.0-20211117180635-dee7805ff2e1/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
@@ -2389,6 +2452,7 @@ golang.org/x/sys v0.0.0-20211216021012-1d35b9e2eb4e/go.mod h1:oPkhp1MJrh7nUepCBc
golang.org/x/sys v0.0.0-20220128215802-99c3d69c2c27/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
golang.org/x/sys v0.0.0-20220209214540-3681064d5158/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
golang.org/x/sys v0.0.0-20220227234510-4e6760a101f9/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
golang.org/x/sys v0.0.0-20220319134239-a9b59b0215f8/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
golang.org/x/sys v0.0.0-20220328115105-d36c6a25d886/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
golang.org/x/sys v0.0.0-20220408201424-a24fb2fb8a0f/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
golang.org/x/sys v0.0.0-20220412211240-33da011f77ad/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
@@ -2416,6 +2480,8 @@ golang.org/x/sys v0.13.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
golang.org/x/sys v0.48.0 h1:bbX/i/6MgT9BVLM9RT1thmxL04yeTAhbEz4SyadbXoo=
golang.org/x/sys v0.48.0/go.mod h1:hNLxWAXmnKAxqDtdwIYC4bM9oQPEecfsnNMuSxOs3og=
golang.org/x/term v0.0.0-20201126162022-7de9c90e9dd1/go.mod h1:bj7SfCRtBDWHUb9snDiAeCFNEtKQo2Wmx5Cou7ajbmo=
golang.org/x/term v0.0.0-20210220032956-6a3ed077a48d/go.mod h1:bj7SfCRtBDWHUb9snDiAeCFNEtKQo2Wmx5Cou7ajbmo=
golang.org/x/term v0.0.0-20210615171337-6886f2dfbf5b/go.mod h1:jbD1KX2456YbFQfuXm/mYQcufACuNUgVhRMnK/tPxf8=
golang.org/x/term v0.0.0-20210927222741-03fcf44c2211/go.mod h1:jbD1KX2456YbFQfuXm/mYQcufACuNUgVhRMnK/tPxf8=
golang.org/x/term v0.2.0/go.mod h1:TVmDHMZPmdnySmBfhjOoOdhjzdE1h4u1VwSiw2l1Nuc=
golang.org/x/term v0.3.0/go.mod h1:q750SLmJuPmVoN1blW3UFBPREJfb1KmY3vwxfr+nFDA=
@@ -2425,8 +2491,8 @@ golang.org/x/term v0.6.0/go.mod h1:m6U89DPEgQRMq3DNkDClhWw02AUbt2daBVO4cn4Hv9U=
golang.org/x/term v0.8.0/go.mod h1:xPskH00ivmX89bAKVGSKKtLOWNx2+17Eiy94tnKShWo=
golang.org/x/term v0.12.0/go.mod h1:owVbMEjm3cBLCHdkQu9b1opXd4ETQWc3BhuQGKgXgvU=
golang.org/x/term v0.13.0/go.mod h1:LTmsnFJwVN6bCy1rVCoS+qHT1HhALEFxKncY3WNNh4U=
golang.org/x/term v0.46.0 h1:3+OXuTbaKDgwk8jTi3aSLHRlmWqHEUDUtxnbFigO4YE=
golang.org/x/term v0.46.0/go.mod h1:+K02xbkittuwc0Am4abfA3Fc+XRGXkvBXNO88NCXPoc=
golang.org/x/term v0.45.0 h1:NwWyBmoJCbfTHpxrWoZ9C6/VxOf7ic219I8xZZFdrf0=
golang.org/x/term v0.45.0/go.mod h1:9aqxs0blBcrm/n0L9QW0aRVD+ktan8ssZromtqJC43w=
golang.org/x/text v0.0.0-20170915032832-14c0d48ead0c/go.mod h1:NqM8EUOU14njkJ3fqMW+pc6Ldnwhi/IjpwHt7yyuwOQ=
golang.org/x/text v0.3.0/go.mod h1:NqM8EUOU14njkJ3fqMW+pc6Ldnwhi/IjpwHt7yyuwOQ=
golang.org/x/text v0.3.1-0.20180807135948-17ff2d5776d2/go.mod h1:NqM8EUOU14njkJ3fqMW+pc6Ldnwhi/IjpwHt7yyuwOQ=
@@ -2452,8 +2518,8 @@ golang.org/x/time v0.0.0-20190308202827-9d24e82272b4/go.mod h1:tRJNPiyCQ0inRvYxb
golang.org/x/time v0.0.0-20191024005414-555d28b269f0/go.mod h1:tRJNPiyCQ0inRvYxbN9jk5I+vvW/OXSQhTDSoE431IQ=
golang.org/x/time v0.0.0-20220922220347-f3bd1da661af/go.mod h1:tRJNPiyCQ0inRvYxbN9jk5I+vvW/OXSQhTDSoE431IQ=
golang.org/x/time v0.1.0/go.mod h1:tRJNPiyCQ0inRvYxbN9jk5I+vvW/OXSQhTDSoE431IQ=
golang.org/x/time v0.16.0 h1:vMb6ptszcQMkcwiRTAuNNU50gom6++Q/6gY2hDM6VDE=
golang.org/x/time v0.16.0/go.mod h1:rVKOqvZeKvrDKTQiAHJ7wmwP0RzleSphoEA9RcdLA0s=
golang.org/x/time v0.15.0 h1:bbrp8t3bGUeFOx08pvsMYRTCVSMk89u4tKbNOZbp88U=
golang.org/x/time v0.15.0/go.mod h1:Y4YMaQmXwGQZoFaVFk4YpCt4FLQMYKZe9oeV/f4MSno=
golang.org/x/tools v0.0.0-20180525024113-a5b4c53f6e8b/go.mod h1:n7NCudcB/nEzxVGmLbDWY5pfWTLqBcC2KZ6jyYvM4mQ=
golang.org/x/tools v0.0.0-20180917221912-90fa682c2a6e/go.mod h1:n7NCudcB/nEzxVGmLbDWY5pfWTLqBcC2KZ6jyYvM4mQ=
golang.org/x/tools v0.0.0-20190114222345-bf090417da8b/go.mod h1:n7NCudcB/nEzxVGmLbDWY5pfWTLqBcC2KZ6jyYvM4mQ=
@@ -2739,8 +2805,8 @@ google.golang.org/genproto v0.0.0-20230222225845-10f96fb3dbec/go.mod h1:3Dl5ZL0q
google.golang.org/genproto v0.0.0-20230306155012-7f2fa6fef1f4/go.mod h1:NWraEVixdDnqcqQ30jipen1STv2r/n24Wb7twVTGR4s=
google.golang.org/genproto v0.0.0-20260715232425-e75dac1f907d h1:C9v1o0/4quuhOAfmRXA2j+we0PqZIp8traLdeogF3Ms=
google.golang.org/genproto v0.0.0-20260715232425-e75dac1f907d/go.mod h1:Wz2wFJntZFmLGo7pLDXZ3wYk5hyc0Mb+SkHhDDXT+lU=
google.golang.org/genproto/googleapis/api v0.0.0-20260817212433-ac3dfec99bb1 h1:lrupDmKL3p5kEX1M92oan027eCKcouzjuPbH6YBK+Rs=
google.golang.org/genproto/googleapis/api v0.0.0-20260817212433-ac3dfec99bb1/go.mod h1:q/3oV3jAi5vwelxsVAprMBC8BcM2zmNe+IjRGd+9/ks=
google.golang.org/genproto/googleapis/api v0.0.0-20260715232425-e75dac1f907d h1:QwnJwPte4XXAkhPu26LTDIahnsMSUV0kK8HkxbC+Pc4=
google.golang.org/genproto/googleapis/api v0.0.0-20260715232425-e75dac1f907d/go.mod h1:WRrQ7/7N19PypuT0fxLOL5Lq0waoiRri4FbtHDEKrGE=
google.golang.org/genproto/googleapis/rpc v0.0.0-20260819154853-08b0e4226688 h1:cYNAzI2sUwhmCcoj9TxvihSrqsxt6uIkj3rDRhSDmW4=
google.golang.org/genproto/googleapis/rpc v0.0.0-20260819154853-08b0e4226688/go.mod h1:DjtHYE8FKJLivXcBEjGwndXfIC23G0VpXiXKqG179uA=
google.golang.org/grpc v1.19.0/go.mod h1:mqu4LbDTu4XGKhr4mRzUsmM4RtVoemTSY81AxZiDr8c=
@@ -2783,8 +2849,8 @@ google.golang.org/grpc v1.51.0/go.mod h1:wgNDFcnuBGmxLKI/qn4T+m5BtEBYXJPvibbUPsA
google.golang.org/grpc v1.52.0/go.mod h1:pu6fVzoFb+NBYNAvQL08ic+lvB2IojljRYuun5vorUY=
google.golang.org/grpc v1.53.0/go.mod h1:OnIrk0ipVdj4N5d9IUoFUx72/VlD7+jUsHwZgwSMQpw=
google.golang.org/grpc v1.55.0/go.mod h1:iYEXKGkEBhg1PjZQvoYEVPTDkHo1/bjTnfwTeGONTY8=
google.golang.org/grpc v1.85.0-dev.0.20260915183914-4e49413dcab7 h1:5+EEM1fC0yjOZID0NUZVrE2+8M/+1TclNrSz/l1xMYs=
google.golang.org/grpc v1.85.0-dev.0.20260915183914-4e49413dcab7/go.mod h1:Ovl0ECo4xx5r4kn/6d4BPSNB7OIFuu6EAjOzjtVAKaM=
google.golang.org/grpc v1.85.0-dev h1:HxkDyKIIZPpFnroC56tQv5gNuKTmVvi0t7TzOf5zt7g=
google.golang.org/grpc v1.85.0-dev/go.mod h1:ljCht0DrxQrXBDRTZp52Qxh3Ffk8CdYm2sj4O2QN2C0=
google.golang.org/grpc/cmd/protoc-gen-go-grpc v1.1.0/go.mod h1:6Kw0yEErY5E/yWrBtf03jp27GLLJujG4z/JK95pnjjw=
google.golang.org/grpc/examples v0.0.0-20250407062114-b368379ef8f6 h1:ExN12ndbJ608cboPYflpTny6mXSzPrDLh0iTaVrRrds=
google.golang.org/grpc/examples v0.0.0-20250407062114-b368379ef8f6/go.mod h1:6ytKWczdvnpnO+m+JiG9NjEDzR1FJfsnmJdG7B8QVZ8=
@@ -2835,6 +2901,7 @@ gopkg.in/yaml.v2 v2.2.3/go.mod h1:hI93XBmqTisBFMUTm0b8Fm+jr3Dg1NNxqwp+5A1VGuI=
gopkg.in/yaml.v2 v2.2.4/go.mod h1:hI93XBmqTisBFMUTm0b8Fm+jr3Dg1NNxqwp+5A1VGuI=
gopkg.in/yaml.v2 v2.2.5/go.mod h1:hI93XBmqTisBFMUTm0b8Fm+jr3Dg1NNxqwp+5A1VGuI=
gopkg.in/yaml.v2 v2.2.8/go.mod h1:hI93XBmqTisBFMUTm0b8Fm+jr3Dg1NNxqwp+5A1VGuI=
gopkg.in/yaml.v2 v2.3.0/go.mod h1:hI93XBmqTisBFMUTm0b8Fm+jr3Dg1NNxqwp+5A1VGuI=
gopkg.in/yaml.v2 v2.4.0 h1:D8xgwECY7CYvx+Y2n4sBz93Jn9JRvxdiyyo8CTfuKaY=
gopkg.in/yaml.v2 v2.4.0/go.mod h1:RDklbk79AGWmwhnvt/jBztapEOGDOx6ZbXqjP6csGnQ=
gopkg.in/yaml.v3 v3.0.0-20200313102051-9f266ea9e77c/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM=
+2 -2
View File
@@ -1,6 +1,6 @@
apiVersion: v1
description: SeaweedFS
name: seaweedfs
appVersion: "4.48"
appVersion: "4.47"
# Dev note: Trigger a helm chart release by `git tag -a helm-<version>`
version: 4.48.2
version: 4.47.1
+1 -35
View File
@@ -27,19 +27,6 @@ so your deployment will be spread/HA.
* cert config exists and can be enabled, but not been tested, requires cert-manager to be installed.
## Prerequisites
The chart's templates render with Helm 3.16.3 and newer versions tested in CI.
Earlier Helm versions are not covered by this chart's compatibility checks.
When JWT signing is enabled, a live upgrade reuses keys from the existing
security ConfigMap (including the legacy ConfigMap name); a key absent from
that ConfigMap is generated. Plain `helm template` does not read cluster state;
use an install and upgrade against a cluster to check key persistence.
The key reader supports single-line quoted TOML strings in JWT sections,
including indented assignments, quoted key names, and bare or simply quoted
section-name segments. An unsupported value or quoted header fails the upgrade
rather than silently rotating a key.
### Database
leveldb is the default database, this supports multiple filer replicas that will [sync automatically](https://github.com/seaweedfs/seaweedfs/wiki/Filer-Store-Replication), with some [limitations](https://github.com/seaweedfs/seaweedfs/wiki/Filer-Store-Replication#limitation).
@@ -389,10 +376,7 @@ start on a non-loopback address without `-adminPassword`, so the chart fails at
render time if `admin.ip` is non-loopback and authentication is not configured via
`admin.secret.adminPassword`, `admin.secret.existingSecret`, or
`WEED_ADMIN_PASSWORD` supplied through `admin.extraEnvironmentVars` /
`admin.secretExtraEnvironmentVars`. Setting `admin.allowInsecureBind` renders
`-allowInsecureBind` and bypasses this guard; it leaves the admin API
unauthenticated on the network, so use it only when access is otherwise
restricted (e.g. network policies). The whole `127.0.0.0/8` range and `::1` are
`admin.secretExtraEnvironmentVars`. The whole `127.0.0.0/8` range and `::1` are
treated as loopback (matching `weed admin`); `localhost` is treated as
non-loopback. Set `admin.ip` to a loopback address only if you also replace the
httpGet probes (e.g. with an `exec` probe that checks `127.0.0.1`).
@@ -583,23 +567,6 @@ Two things worth knowing before you turn this on:
The DNS selectors default to CoreDNS as kubeadm, kind and the managed offerings from AWS, Google and Azure install it. On OpenShift, override `egress.dnsNamespaceSelector` and `egress.dnsPodSelector` to match `openshift-dns`; see the comment in `values.yaml`.
## Pod and container security contexts
Pod and container security contexts are configurable independently for every built-in workload and remain empty by default for backwards compatibility. The examples in `values.yaml` show how to enable a `RuntimeDefault` seccomp profile, disable privilege escalation and privileged mode, drop all Linux capabilities, and use a read-only root filesystem.
SeaweedFS uses `/tmp` for Unix sockets, temporary uploads, worker task files, and other runtime data. When `readOnlyRootFilesystem` is enabled for a built-in component, the chart mounts a writable `emptyDir` at `/tmp` for its chart-managed containers. Its optional size limit can be configured globally:
```yaml
global:
seaweedfs:
tmpDir:
sizeLimit: 1Gi
```
The chart does not enable `runAsNonRoot` by default because its default `hostPath` storage may be owned by root. To enforce the Kubernetes `restricted` Pod Security Standard, use storage that is writable by a non-root user and configure `runAsNonRoot` or use the OpenShift overrides below.
Security contexts configured for a component also apply to the chart-managed helper containers for that component. User-provided init containers and sidecars must define their own container security context and writable mounts.
## OpenShift Support
SeaweedFS can be deployed on OpenShift or any cluster enforcing the Kubernetes "restricted" Pod Security Standard. By default, OpenShift blocks containers that run as root or use `hostPath` volumes.
@@ -608,7 +575,6 @@ To deploy on OpenShift, use the provided `openshift-values.yaml` which overrides
1. Use `PersistentVolumeClaims` instead of `hostPath`.
2. Enable `runAsNonRoot` and omit hardcoded UIDs to allow OpenShift to assign valid UIDs automatically.
3. Apply appropriate `seccompProfile` and drop capabilities.
4. Use a read-only root filesystem with writable temporary storage at `/tmp`.
Usage:
```bash
@@ -15,7 +15,6 @@
# automatically assign a valid UID from the namespace's allocated range.
# 3. Dropping all Linux capabilities and setting allowPrivilegeEscalation: false
# 4. Enabling RuntimeDefault seccompProfile
# 5. Using a read-only root filesystem with writable temporary storage
#
# Usage:
# helm install seaweedfs seaweedfs/seaweedfs \
@@ -50,7 +49,6 @@ master:
containerSecurityContext:
enabled: true
allowPrivilegeEscalation: false
readOnlyRootFilesystem: true
capabilities:
drop: ["ALL"]
runAsNonRoot: true
@@ -78,7 +76,6 @@ volume:
containerSecurityContext:
enabled: true
allowPrivilegeEscalation: false
readOnlyRootFilesystem: true
capabilities:
drop: ["ALL"]
runAsNonRoot: true
@@ -104,7 +101,6 @@ filer:
containerSecurityContext:
enabled: true
allowPrivilegeEscalation: false
readOnlyRootFilesystem: true
capabilities:
drop: ["ALL"]
runAsNonRoot: true
@@ -129,7 +125,6 @@ s3:
containerSecurityContext:
enabled: true
allowPrivilegeEscalation: false
readOnlyRootFilesystem: true
capabilities:
drop: ["ALL"]
runAsNonRoot: true
@@ -9,7 +9,7 @@
{{- $adminAuthEnabled := include "seaweedfs.admin.authEnabled" . }}
{{- $adminIp := .Values.admin.ip | default "0.0.0.0" }}
{{- if and (not (include "seaweedfs.admin.isLoopbackIp" $adminIp)) (ne $adminAuthEnabled "true") }}
{{- fail (printf "admin.ip is set to %q (non-loopback) but admin authentication is not configured. Since `weed admin` 4.46 refuses to bind a non-loopback address without authentication, the admin container would exit on startup. Set admin.secret.adminPassword or admin.secret.existingSecret, or supply WEED_ADMIN_PASSWORD via admin.extraEnvironmentVars / admin.secretExtraEnvironmentVars, or set admin.ip to a loopback address such as 127.0.0.1 (note: a loopback bind makes the chart's httpGet readiness/liveness probes fail), or set admin.allowInsecureBind to true to opt out via -allowInsecureBind (INSECURE: exposes the admin API unauthenticated on the network)." $adminIp) -}}
{{- fail (printf "admin.ip is set to %q (non-loopback) but admin authentication is not configured. Since `weed admin` 4.46 refuses to bind a non-loopback address without authentication, the admin container would exit on startup. Set admin.secret.adminPassword or admin.secret.existingSecret, or supply WEED_ADMIN_PASSWORD via admin.extraEnvironmentVars / admin.secretExtraEnvironmentVars, or set admin.ip to a loopback address such as 127.0.0.1 (note: a loopback bind makes the chart's httpGet readiness/liveness probes fail)." $adminIp) -}}
{{- end }}
apiVersion: apps/v1
kind: StatefulSet
@@ -176,23 +176,19 @@ spec:
-dataDir={{ .Values.admin.dataDir }} \
{{- end }}
{{- if .Values.admin.masters }}
-masters={{ .Values.admin.masters }} \
-masters={{ .Values.admin.masters }}{{- if or $urlPrefix .Values.admin.extraArgs }} \{{ end }}
{{- else if .Values.global.seaweedfs.masterServer }}
-masters={{ .Values.global.seaweedfs.masterServer }} \
-masters={{ .Values.global.seaweedfs.masterServer }}{{- if or $urlPrefix .Values.admin.extraArgs }} \{{ end }}
{{- else }}
-masters={{ range $index := until (.Values.master.replicas | int) }}${SEAWEEDFS_FULLNAME}-master-{{ $index }}.${SEAWEEDFS_FULLNAME}-master.{{ $.Release.Namespace }}:{{ $.Values.master.port }}{{ if lt $index (sub ($.Values.master.replicas | int) 1) }},{{ end }}{{ end }} \
-masters={{ range $index := until (.Values.master.replicas | int) }}${SEAWEEDFS_FULLNAME}-master-{{ $index }}.${SEAWEEDFS_FULLNAME}-master.{{ $.Release.Namespace }}:{{ $.Values.master.port }}{{ if lt $index (sub ($.Values.master.replicas | int) 1) }},{{ end }}{{ end }}{{- if or $urlPrefix .Values.admin.extraArgs }} \{{ end }}
{{- end }}
{{- if $urlPrefix }}
-urlPrefix={{ $urlPrefix }} \
{{- end }}
{{- if .Values.admin.allowInsecureBind }}
-allowInsecureBind \
-urlPrefix={{ $urlPrefix }}{{- if .Values.admin.extraArgs }} \{{ end }}
{{- end }}
{{- range $index, $arg := .Values.admin.extraArgs }}
{{ $arg }}{{- if lt $index (sub (len $.Values.admin.extraArgs) 1) }} \{{ end }}
{{- end }}
volumeMounts:
{{- include "seaweedfs.tmpDirVolumeMount" (list . .Values.admin.containerSecurityContext (tpl (.Values.admin.extraVolumeMounts | default "") .) (tpl (.Values.admin.extraVolumes | default "") .)) | nindent 12 }}
{{- if or (eq .Values.admin.data.type "hostPath") (eq .Values.admin.data.type "persistentVolumeClaim") (eq .Values.admin.data.type "emptyDir") (eq .Values.admin.data.type "existingClaim") }}
- name: admin-data
mountPath: /data
@@ -268,7 +264,6 @@ spec:
{{- include "seaweedfs.tplvalues.render" (dict "value" .Values.admin.sidecars "context" $) | nindent 8 }}
{{- end }}
volumes:
{{- include "seaweedfs.tmpDirVolume" (list . .Values.admin.containerSecurityContext (tpl (.Values.admin.extraVolumeMounts | default "") .) (tpl (.Values.admin.extraVolumes | default "") .) false) | nindent 8 }}
{{- if eq .Values.admin.data.type "hostPath" }}
- name: admin-data
hostPath:
@@ -37,17 +37,20 @@ spec:
{{- with .Values.allInOne.podLabels }}
{{- toYaml . | nindent 8 }}
{{- end }}
{{- $podAnnotations := mergeOverwrite (deepCopy (.Values.podAnnotations | default dict)) (.Values.allInOne.podAnnotations | default dict) }}
{{- $existingS3ConfigSecret := or .Values.allInOne.s3.existingConfigSecret .Values.s3.existingConfigSecret .Values.filer.s3.existingConfigSecret }}
{{- if $existingS3ConfigSecret }}
{{- $configSecret := (lookup "v1" "Secret" .Release.Namespace $existingS3ConfigSecret) | default dict }}
{{- $_ := set $podAnnotations "checksum/s3config" ($configSecret | toYaml | sha256sum) }}
{{- else }}
{{- $_ := set $podAnnotations "checksum/s3config" (include (print .Template.BasePath "/s3/s3-secret.yaml") . | sha256sum) }}
{{- end }}
{{- $_ := set $podAnnotations "checksum/master-config" (include (print .Template.BasePath "/master/master-configmap.yaml") . | sha256sum) }}
annotations:
{{- toYaml $podAnnotations | nindent 8 }}
{{- with .Values.podAnnotations }}
{{- toYaml . | nindent 8 }}
{{- end }}
{{- with .Values.allInOne.podAnnotations }}
{{- toYaml . | nindent 8 }}
{{- end }}
{{- $existingS3ConfigSecret := or .Values.allInOne.s3.existingConfigSecret .Values.s3.existingConfigSecret .Values.filer.s3.existingConfigSecret }}
{{- if $existingS3ConfigSecret }}
{{- $configSecret := (lookup "v1" "Secret" .Release.Namespace $existingS3ConfigSecret) | default dict }}
checksum/s3config: {{ $configSecret | toYaml | sha256sum }}
{{- else }}
checksum/s3config: {{ include (print .Template.BasePath "/s3/s3-secret.yaml") . | sha256sum }}
{{- end }}
spec:
restartPolicy: {{ default .Values.global.seaweedfs.restartPolicy .Values.allInOne.restartPolicy }}
{{- if .Values.allInOne.affinity }}
@@ -306,7 +309,6 @@ spec:
{{- end }}
volumeMounts:
{{- include "seaweedfs.tmpDirVolumeMount" (list . .Values.allInOne.containerSecurityContext (tpl (.Values.allInOne.extraVolumeMounts | default "") .) (tpl (.Values.allInOne.extraVolumes | default "") .)) | nindent 12 }}
- name: data
mountPath: /data
{{- if and .Values.allInOne.s3.enabled (or .Values.allInOne.s3.enableAuth .Values.s3.enableAuth .Values.filer.s3.enableAuth) }}
@@ -434,7 +436,6 @@ spec:
{{- include "seaweedfs.tplvalues.render" (dict "value" .Values.allInOne.sidecars "context" $) | nindent 8 }}
{{- end }}
volumes:
{{- include "seaweedfs.tmpDirVolume" (list . .Values.allInOne.containerSecurityContext (tpl (.Values.allInOne.extraVolumeMounts | default "") .) (tpl (.Values.allInOne.extraVolumes | default "") .) false) | nindent 8 }}
{{- include "seaweedfs.licenseVolume" . | nindent 8 }}
- name: data
{{- if eq .Values.allInOne.data.type "hostPath" }}
@@ -25,10 +25,6 @@ spec:
organizations:
- "SeaweedFS CA"
dnsNames:
- '{{ include "seaweedfs.fullname" . }}-admin'
- '{{ include "seaweedfs.fullname" . }}-admin.{{ .Release.Namespace }}'
- '{{ include "seaweedfs.fullname" . }}-admin.{{ .Release.Namespace }}.svc'
- '{{ include "seaweedfs.fullname" . }}-admin.{{ .Release.Namespace }}.svc.cluster.local'
- '*.{{ include "seaweedfs.fullname" . }}-admin'
- '*.{{ include "seaweedfs.fullname" . }}-admin.{{ .Release.Namespace }}'
- '*.{{ include "seaweedfs.fullname" . }}-admin.{{ .Release.Namespace }}.svc'
@@ -28,13 +28,6 @@ rules:
- "update"
- "create"
- "delete"
- apiGroups: ["objectstorage.k8s.io"]
resources:
- "bucketclasses"
verbs:
- "get"
- "list"
- "watch"
- apiGroups: ["coordination.k8s.io"]
resources: ["leases"]
verbs:
@@ -68,7 +61,7 @@ metadata:
app.kubernetes.io/instance: {{ .Release.Name }}
subjects:
- kind: ServiceAccount
name: {{ include "seaweedfs.serviceAccountName" . }}-objectstorage-provisioner
name: {{ .Values.global.seaweedfs.serviceAccountName }}-objectstorage-provisioner
namespace: {{ .Release.Namespace }}
roleRef:
kind: ClusterRole
@@ -61,7 +61,7 @@ spec:
schedulerName: {{ .Values.cosi.schedulerName | quote }}
{{- end }}
enableServiceLinks: false
serviceAccountName: {{ include "seaweedfs.serviceAccountName" . }}-objectstorage-provisioner
serviceAccountName: {{ include "seaweedfs.componentName" (list . "objectstorage-provisioner") }}
{{- if .Values.cosi.initContainers }}
initContainers:
{{ tpl .Values.cosi.initContainers . | nindent 8 | trim }}
@@ -113,7 +113,6 @@ spec:
{{- end -}}
{{- end }}
volumeMounts:
{{- include "seaweedfs.tmpDirVolumeMount" (list . .Values.cosi.containerSecurityContext (tpl (.Values.cosi.extraVolumeMounts | default "") .) (tpl (.Values.cosi.extraVolumes | default "") .)) | nindent 12 }}
- mountPath: /var/lib/cosi
name: socket
{{- if .Values.cosi.enableAuth }}
@@ -149,9 +148,6 @@ spec:
resources:
{{- toYaml . | nindent 12 }}
{{- end }}
{{- if .Values.cosi.containerSecurityContext.enabled }}
securityContext: {{- omit .Values.cosi.containerSecurityContext "enabled" | toYaml | nindent 12 }}
{{- end }}
- name: seaweedfs-cosi-sidecar
image: "{{ .Values.cosi.sidecar.image }}"
imagePullPolicy: {{ default "IfNotPresent" .Values.global.seaweedfs.imagePullPolicy }}
@@ -163,7 +159,6 @@ spec:
fieldRef:
fieldPath: metadata.namespace
volumeMounts:
{{- include "seaweedfs.tmpDirVolumeMount" (list . .Values.cosi.containerSecurityContext "" "") | nindent 12 }}
- mountPath: /var/lib/cosi
name: socket
{{- with .Values.cosi.sidecar.resources }}
@@ -177,7 +172,6 @@ spec:
{{- include "seaweedfs.tplvalues.render" (dict "value" .Values.cosi.sidecars "context" $) | nindent 8 }}
{{- end }}
volumes:
{{- include "seaweedfs.tmpDirVolume" (list . .Values.cosi.containerSecurityContext (tpl (.Values.cosi.extraVolumeMounts | default "") .) (tpl (.Values.cosi.extraVolumes | default "") .) true) | nindent 8 }}
- name: socket
emptyDir: {}
{{- if .Values.cosi.enableAuth }}
@@ -3,7 +3,7 @@
apiVersion: v1
kind: ServiceAccount
metadata:
name: {{ include "seaweedfs.serviceAccountName" . }}-objectstorage-provisioner
name: {{ .Values.global.seaweedfs.serviceAccountName }}-objectstorage-provisioner
namespace: {{ .Release.Namespace }}
labels:
app.kubernetes.io/name: {{ template "seaweedfs.name" . }}
@@ -43,15 +43,19 @@ spec:
{{- with .Values.filer.podLabels }}
{{- toYaml . | nindent 8 }}
{{- end }}
{{- $podAnnotations := mergeOverwrite (deepCopy (.Values.podAnnotations | default dict)) (.Values.filer.podAnnotations | default dict) }}
{{- if .Values.filer.s3.existingConfigSecret }}
{{- $configSecret := (lookup "v1" "Secret" .Release.Namespace .Values.filer.s3.existingConfigSecret) | default dict }}
{{- $_ := set $podAnnotations "checksum/s3config" ($configSecret | toYaml | sha256sum) }}
{{- else }}
{{- $_ := set $podAnnotations "checksum/s3config" (include (print .Template.BasePath "/s3/s3-secret.yaml") . | sha256sum) }}
{{- end }}
annotations:
{{- toYaml $podAnnotations | nindent 8 }}
{{- with .Values.podAnnotations }}
{{- toYaml . | nindent 8 }}
{{- end }}
{{- with .Values.filer.podAnnotations }}
{{- toYaml . | nindent 8 }}
{{- end }}
{{- if .Values.filer.s3.existingConfigSecret }}
{{- $configSecret := (lookup "v1" "Secret" .Release.Namespace .Values.filer.s3.existingConfigSecret) | default dict }}
checksum/s3config: {{ $configSecret | toYaml | sha256sum }}
{{- else }}
checksum/s3config: {{ include (print .Template.BasePath "/s3/s3-secret.yaml") . | sha256sum }}
{{- end }}
spec:
restartPolicy: {{ default .Values.global.seaweedfs.restartPolicy .Values.filer.restartPolicy }}
{{- if .Values.filer.affinity }}
@@ -229,7 +233,6 @@ spec:
{{ . }} \
{{- end }}
volumeMounts:
{{- include "seaweedfs.tmpDirVolumeMount" (list . .Values.filer.containerSecurityContext (tpl (.Values.filer.extraVolumeMounts | default "") .) (tpl (.Values.filer.extraVolumes | default "") .)) | nindent 12 }}
{{- if (or (eq .Values.filer.logs.type "hostPath") (eq .Values.filer.logs.type "persistentVolumeClaim") (eq .Values.filer.logs.type "emptyDir")) }}
- name: seaweedfs-filer-log-volume
mountPath: "/logs/"
@@ -338,7 +341,6 @@ spec:
{{- include "seaweedfs.tplvalues.render" (dict "value" .Values.filer.sidecars "context" $) | nindent 8 }}
{{- end }}
volumes:
{{- include "seaweedfs.tmpDirVolume" (list . .Values.filer.containerSecurityContext (tpl (.Values.filer.extraVolumeMounts | default "") .) (tpl (.Values.filer.extraVolumes | default "") .) false) | nindent 8 }}
{{- if eq .Values.filer.logs.type "hostPath" }}
- name: seaweedfs-filer-log-volume
hostPath:
@@ -43,10 +43,13 @@ spec:
{{- with .Values.master.podLabels }}
{{- toYaml . | nindent 8 }}
{{- end }}
{{- $podAnnotations := mergeOverwrite (deepCopy (.Values.podAnnotations | default dict)) (.Values.master.podAnnotations | default dict) }}
{{- $_ := set $podAnnotations "checksum/master-config" (include (print .Template.BasePath "/master/master-configmap.yaml") . | sha256sum) }}
annotations:
{{- toYaml $podAnnotations | nindent 8 }}
{{ with .Values.podAnnotations }}
{{- toYaml . | nindent 8 }}
{{- end }}
{{- with .Values.master.podAnnotations }}
{{- toYaml . | nindent 8 }}
{{- end }}
spec:
restartPolicy: {{ default .Values.global.seaweedfs.restartPolicy .Values.master.restartPolicy }}
{{- if .Values.master.affinity }}
@@ -179,7 +182,6 @@ spec:
{{ . }} \
{{- end }}
volumeMounts:
{{- include "seaweedfs.tmpDirVolumeMount" (list . .Values.master.containerSecurityContext (tpl (.Values.master.extraVolumeMounts | default "") .) (tpl (.Values.master.extraVolumes | default "") .)) | nindent 12 }}
- name : data-{{ .Release.Namespace }}
mountPath: /data
{{- if or (eq .Values.master.logs.type "hostPath") (eq .Values.master.logs.type "persistentVolumeClaim") (eq .Values.master.logs.type "emptyDir") }}
@@ -259,7 +261,6 @@ spec:
{{- include "seaweedfs.tplvalues.render" (dict "value" .Values.master.sidecars "context" $) | nindent 8 }}
{{- end }}
volumes:
{{- include "seaweedfs.tmpDirVolume" (list . .Values.master.containerSecurityContext (tpl (.Values.master.extraVolumeMounts | default "") .) (tpl (.Values.master.extraVolumes | default "") .) false) | nindent 8 }}
{{- include "seaweedfs.licenseVolume" . | nindent 8 }}
{{- if eq .Values.master.logs.type "hostPath" }}
- name: seaweedfs-master-log-volume
@@ -17,10 +17,6 @@ metadata:
{{- end }}
spec:
replicas: {{ .Values.s3.replicas }}
{{- with .Values.s3.updateStrategy }}
strategy:
{{- toYaml . | nindent 4 }}
{{- end }}
selector:
matchLabels:
app.kubernetes.io/name: {{ template "seaweedfs.name" . }}
@@ -39,15 +35,19 @@ spec:
{{- with .Values.s3.podLabels }}
{{- toYaml . | nindent 8 }}
{{- end }}
{{- $podAnnotations := mergeOverwrite (deepCopy (.Values.podAnnotations | default dict)) (.Values.s3.podAnnotations | default dict) }}
{{- if .Values.s3.existingConfigSecret }}
{{- $configSecret := (lookup "v1" "Secret" .Release.Namespace .Values.s3.existingConfigSecret) | default dict }}
{{- $_ := set $podAnnotations "checksum/s3config" ($configSecret | toYaml | sha256sum) }}
{{- else }}
{{- $_ := set $podAnnotations "checksum/s3config" (include (print .Template.BasePath "/s3/s3-secret.yaml") . | sha256sum) }}
{{- end }}
annotations:
{{- toYaml $podAnnotations | nindent 8 }}
{{ with .Values.podAnnotations }}
{{- toYaml . | nindent 8 }}
{{- end }}
{{- with .Values.s3.podAnnotations }}
{{- toYaml . | nindent 8 }}
{{- end }}
{{- if .Values.s3.existingConfigSecret }}
{{- $configSecret := (lookup "v1" "Secret" .Release.Namespace .Values.s3.existingConfigSecret) | default dict }}
checksum/s3config: {{ $configSecret | toYaml | sha256sum }}
{{- else }}
checksum/s3config: {{ include (print .Template.BasePath "/s3/s3-secret.yaml") . | sha256sum }}
{{- end }}
spec:
restartPolicy: {{ default .Values.global.seaweedfs.restartPolicy .Values.s3.restartPolicy }}
{{- if .Values.s3.affinity }}
@@ -63,7 +63,7 @@ spec:
{{ tpl .Values.s3.tolerations . | nindent 8 | trim }}
{{- end }}
{{- include "seaweedfs.imagePullSecrets" . | nindent 6 }}
terminationGracePeriodSeconds: {{ .Values.s3.terminationGracePeriodSeconds }}
terminationGracePeriodSeconds: 10
{{- if .Values.s3.priorityClassName }}
priorityClassName: {{ .Values.s3.priorityClassName | quote }}
{{- end }}
@@ -161,7 +161,6 @@ spec:
{{ . }} \
{{- end }}
volumeMounts:
{{- include "seaweedfs.tmpDirVolumeMount" (list . .Values.s3.containerSecurityContext (tpl (.Values.s3.extraVolumeMounts | default "") .) (tpl (.Values.s3.extraVolumes | default "") .)) | nindent 12 }}
{{- if or (eq .Values.s3.logs.type "hostPath") (eq .Values.s3.logs.type "emptyDir") }}
- name: logs
mountPath: "/logs/"
@@ -239,10 +238,6 @@ spec:
failureThreshold: {{ .Values.s3.livenessProbe.failureThreshold }}
timeoutSeconds: {{ .Values.s3.livenessProbe.timeoutSeconds }}
{{- end }}
{{- with .Values.s3.lifecycle }}
lifecycle:
{{- toYaml . | nindent 12 }}
{{- end }}
{{- with .Values.s3.resources }}
resources:
{{- toYaml . | nindent 12 }}
@@ -254,7 +249,6 @@ spec:
{{- include "seaweedfs.tplvalues.render" (dict "value" .Values.s3.sidecars "context" $) | nindent 8 }}
{{- end }}
volumes:
{{- include "seaweedfs.tmpDirVolume" (list . .Values.s3.containerSecurityContext (tpl (.Values.s3.extraVolumeMounts | default "") .) (tpl (.Values.s3.extraVolumes | default "") .) false) | nindent 8 }}
{{- if .Values.s3.enableAuth }}
- name: config-users
secret:
@@ -170,7 +170,6 @@ spec:
-userStoreFile=/etc/sw/seaweedfs_sftp_config \
-filer={{ include "seaweedfs.componentName" (list . "filer-client") }}.{{ .Release.Namespace }}:{{ .Values.filer.port }}
volumeMounts:
{{- include "seaweedfs.tmpDirVolumeMount" (list . .Values.sftp.containerSecurityContext (tpl (.Values.sftp.extraVolumeMounts | default "") .) (tpl (.Values.sftp.extraVolumes | default "") .)) | nindent 12 }}
{{- if or (eq .Values.sftp.logs.type "hostPath") (eq .Values.sftp.logs.type "emptyDir") }}
- name: logs
mountPath: "/logs/"
@@ -250,7 +249,6 @@ spec:
{{- include "seaweedfs.tplvalues.render" (dict "value" .Values.sftp.sidecars "context" $) | nindent 8 }}
{{- end }}
volumes:
{{- include "seaweedfs.tmpDirVolume" (list . .Values.sftp.containerSecurityContext (tpl (.Values.sftp.extraVolumeMounts | default "") .) (tpl (.Values.sftp.extraVolumes | default "") .) false) | nindent 8 }}
{{- if .Values.sftp.enableAuth }}
- name: config-users
secret:
@@ -76,43 +76,6 @@ Inject extra environment vars in the format key:value, if populated
{{- end }}
{{- end -}}
{{/*
Writable temporary directory for containers using a read-only root filesystem.
Input: list of the root context, the component container security context, the
rendered extraVolumeMounts and extraVolumes, and whether the pod has secondary
chart-managed containers that mount seaweedfs-tmp. A user-supplied /tmp mount
only covers the main container, so the volume is still emitted for secondaries;
a user-supplied seaweedfs-tmp volume is reused rather than duplicated.
*/}}
{{- define "seaweedfs.tmpDirCovered" -}}
{{- regexMatch `(?m)^\s*-?\s*mountPath:\s*['"]?/tmp/?['"]?\s*(#.*)?$` (index . 2) -}}
{{- end -}}
{{- define "seaweedfs.tmpDirVolume" -}}
{{- $root := index . 0 -}}
{{- $securityContext := index . 1 -}}
{{- if and $securityContext.enabled $securityContext.readOnlyRootFilesystem
(or (index . 4) (ne (include "seaweedfs.tmpDirCovered" .) "true"))
(not (regexMatch `(?m)^\s*-?\s*name:\s*['"]?seaweedfs-tmp['"]?\s*(#.*)?$` (index . 3))) }}
- name: seaweedfs-tmp
{{- with $root.Values.global.seaweedfs.tmpDir.sizeLimit }}
emptyDir:
sizeLimit: {{ . | quote }}
{{- else }}
emptyDir: {}
{{- end }}
{{- end }}
{{- end -}}
{{- define "seaweedfs.tmpDirVolumeMount" -}}
{{- $securityContext := index . 1 -}}
{{- if and $securityContext.enabled $securityContext.readOnlyRootFilesystem
(ne (include "seaweedfs.tmpDirCovered" .) "true") }}
- name: seaweedfs-tmp
mountPath: /tmp
{{- end }}
{{- end -}}
{{/* Whether the mysql filer store is selected; a flag the chart cannot read counts as selected. */}}
{{- define "seaweedfs.filer.mysqlEnabled" -}}
{{- $merged := dict -}}
@@ -142,13 +105,13 @@ true
{{- end -}}
{{- end -}}
{{/* Whether the admin non-loopback bind guard is satisfied: admin.secret
(adminPassword or existingSecret), WEED_ADMIN_PASSWORD via
extraEnvironmentVars / secretExtraEnvironmentVars, or
admin.allowInsecureBind. A secret-backed entry counts as enabled even
though the chart cannot read its value. */}}
{{/* Whether admin authentication is enabled from any supported source:
admin.secret (adminPassword or existingSecret), or WEED_ADMIN_PASSWORD
supplied via extraEnvironmentVars / secretExtraEnvironmentVars (which
weed admin picks up through viper's AutomaticEnv). A secret-backed
entry counts as enabled even though the chart cannot read its value. */}}
{{- define "seaweedfs.admin.authEnabled" -}}
{{- if or .Values.admin.secret.existingSecret .Values.admin.secret.adminPassword .Values.admin.allowInsecureBind -}}
{{- if or .Values.admin.secret.existingSecret .Values.admin.secret.adminPassword -}}
true
{{- else -}}
{{- $merged := dict -}}
@@ -444,46 +407,6 @@ true
{{- end -}}
{{- end -}}
{{/* Read a JWT key from the chart's existing security.toml without fromToml
(which requires Helm >=3.17). Args: (list "<section>" $raw).
Return the single-line TOML string token, including its quotes, so escapes
and explicitly empty values survive unchanged. No output means absent.
Accept bare or simply quoted dotted header segments. Fail on other quoted
headers rather than risk rotating a key hidden by unsupported syntax.
This reads the chart's section/key layout, not arbitrary TOML syntax. */}}
{{- define "seaweedfs.existingTomlKey" -}}
{{- $section := index . 0 -}}
{{- $raw := index . 1 -}}
{{- $parts := list -}}
{{- range $part := splitList "." $section -}}
{{- $escaped := regexQuoteMeta $part -}}
{{- $parts = append $parts (printf `(?:%s|"%s"|'%s')` $escaped $escaped $escaped) -}}
{{- end -}}
{{- $header := printf `^\[[ \t]*%s[ \t]*\][ \t]*(#.*)?$` (join `[ \t]*\.[ \t]*` $parts) -}}
{{- $segment := `(?:[A-Za-z0-9_-]+|"[^"\\]*"|'[^']*')` -}}
{{- $simpleHeader := printf `^\[[ \t]*%s(?:[ \t]*\.[ \t]*%s)*[ \t]*\][ \t]*(#.*)?$` $segment $segment -}}
{{- $assignment := `^(key|"key"|'key')[ \t]*=[ \t]*` -}}
{{- $string := `"([^"\\]|\\.)*"|'[^']*'` -}}
{{- $active := false -}}
{{- $key := "" -}}
{{- range $rawLine := splitList "\n" $raw -}}
{{- $line := trim $rawLine -}}
{{- if hasPrefix "[" $line -}}
{{- if and (regexMatch `^\[[^]]*["']` $line) (not (regexMatch $simpleHeader $line)) (eq $key "") -}}
{{- fail (printf "security.toml has an unsupported quoted section header; refusing to replace [%s].key" $section) -}}
{{- end -}}
{{- $active = regexMatch $header $line -}}
{{- else if and $active (eq $key "") (regexMatch $assignment $line) -}}
{{- $value := regexReplaceAll $assignment $line "" -}}
{{- if not (regexMatch (printf `^(%s)[ \t]*(#.*)?$` $string) $value) -}}
{{- fail (printf "security.toml [%s].key must be a single-line quoted TOML string; refusing to replace an existing key" $section) -}}
{{- end -}}
{{- $key = regexFind $string $value -}}
{{- end -}}
{{- end -}}
{{- $key -}}
{{- end -}}
{{/* True when the post-install bucket hook Job renders: an S3 endpoint, plus
buckets to create on it. Read by the Job itself and by its NetworkPolicy,
which has to appear exactly when the Job does - a Job without its policy
@@ -544,9 +467,8 @@ true
{{- $pvcName := printf "%s-%s-%s-%d" $dir.name $seaweedfsName $volumeName $e }}
{{- $currentPVC := (lookup "v1" "PersistentVolumeClaim" $.Release.Namespace $pvcName) }}
{{- if $currentPVC }}
{{- /* include returns a string such as "6.442450944e+10"; convert back to a number, or gt compares lexically */}}
{{- $oldSize := include "seaweedfs.resource-quantity" $currentPVC.spec.resources.requests.storage | float64 }}
{{- $newSize := include "seaweedfs.resource-quantity" $desiredSize | float64 }}
{{- $oldSize := include "seaweedfs.resource-quantity" $currentPVC.spec.resources.requests.storage }}
{{- $newSize := include "seaweedfs.resource-quantity" $desiredSize }}
{{- if gt $newSize $oldSize }}
{{- $commands = append $commands (printf "kubectl patch pvc %s-%s-%s-%d -p '{\"spec\":{\"resources\":{\"requests\":{\"storage\":\"%s\"}}}}'" $dir.name $seaweedfsName $volumeName $e $desiredSize) }}
{{- end }}
@@ -16,12 +16,6 @@
{{- $clusterUpper := upper $clusterAlias }}
{{- $clusterMasterKey := printf "WEED_CLUSTER_%s_MASTER" $clusterUpper }}
{{- $clusterFilerKey := printf "WEED_CLUSTER_%s_FILER" $clusterUpper }}
{{- $podSecurityContext := .Values.filer.podSecurityContext }}
{{- $containerSecurityContext := .Values.filer.containerSecurityContext }}
{{- if .Values.allInOne.enabled }}
{{- $podSecurityContext = .Values.allInOne.podSecurityContext }}
{{- $containerSecurityContext = .Values.allInOne.containerSecurityContext }}
{{- end }}
{{- /* Check allInOne mode first */}}
{{- if .Values.allInOne.enabled }}
@@ -77,8 +71,8 @@ spec:
app.kubernetes.io/component: bucket-hook
spec:
restartPolicy: Never
{{- if $podSecurityContext.enabled }}
securityContext: {{- omit $podSecurityContext "enabled" | toYaml | nindent 8 }}
{{- if .Values.filer.podSecurityContext.enabled }}
securityContext: {{- omit .Values.filer.podSecurityContext "enabled" | toYaml | nindent 8 }}
{{- end }}
{{- include "seaweedfs.imagePullSecrets" $ | nindent 6 }}
containers:
@@ -208,9 +202,8 @@ spec:
/usr/bin/weed shell
{{- end }}
{{- end }}
{{- if or (and $containerSecurityContext.enabled $containerSecurityContext.readOnlyRootFilesystem) $enableAuth (include "seaweedfs.securityConfigEnabled" .) }}
{{- if or $enableAuth (include "seaweedfs.securityConfigEnabled" .) }}
volumeMounts:
{{- include "seaweedfs.tmpDirVolumeMount" (list . $containerSecurityContext "" "") | nindent 10 }}
{{- if $enableAuth }}
- name: config-users
mountPath: /etc/sw
@@ -222,14 +215,6 @@ spec:
mountPath: /etc/seaweedfs/security.toml
subPath: security.toml
{{- end }}
{{- if .Values.global.seaweedfs.enableSecurity }}
- name: ca-cert
readOnly: true
mountPath: /usr/local/share/ca-certificates/ca/
- name: client-cert
readOnly: true
mountPath: /usr/local/share/ca-certificates/client/
{{- end }}
{{- end }}
ports:
- containerPort: {{ .Values.master.port }}
@@ -244,12 +229,11 @@ spec:
resources:
{{- toYaml . | nindent 10 }}
{{- end }}
{{- if $containerSecurityContext.enabled }}
securityContext: {{- omit $containerSecurityContext "enabled" | toYaml | nindent 12 }}
{{- if .Values.filer.containerSecurityContext.enabled }}
securityContext: {{- omit .Values.filer.containerSecurityContext "enabled" | toYaml | nindent 12 }}
{{- end }}
{{- if or (and $containerSecurityContext.enabled $containerSecurityContext.readOnlyRootFilesystem) $enableAuth (include "seaweedfs.securityConfigEnabled" .) }}
{{- if or $enableAuth (include "seaweedfs.securityConfigEnabled" .) }}
volumes:
{{- include "seaweedfs.tmpDirVolume" (list . $containerSecurityContext "" "" false) | nindent 8 }}
{{- if $enableAuth }}
- name: config-users
secret:
@@ -265,13 +249,5 @@ spec:
configMap:
name: {{ include "seaweedfs.fullname" . }}-security-config
{{- end }}
{{- if .Values.global.seaweedfs.enableSecurity }}
- name: ca-cert
secret:
secretName: {{ include "seaweedfs.fullname" . }}-ca-cert
- name: client-cert
secret:
secretName: {{ include "seaweedfs.fullname" . }}-client-cert
{{- end }}
{{- end }}
{{- end }}
@@ -18,7 +18,7 @@ data:
{{- $legacyName := printf "%s-%s" (include "seaweedfs.name" .) "security-config" }}
{{- $existing = lookup "v1" "ConfigMap" .Release.Namespace $legacyName }}
{{- end }}
{{- $existingToml := dig "data" "security.toml" "" $existing }}
{{- $securityConfig := fromToml (dig "data" "security.toml" "" $existing) }}
{{- $securityConfigValues := .Values.global.seaweedfs.securityConfig | default dict }}
{{- $jwtSigning := $securityConfigValues.jwtSigning | default dict }}
{{- $expiresAfterSeconds := $jwtSigning.expiresAfterSeconds | default dict }}
@@ -29,7 +29,7 @@ data:
# the jwt signing key is read by master and volume server
# the jwt defaults to expire after 10 seconds
[jwt.signing]
key = {{ include "seaweedfs.existingTomlKey" (list "jwt.signing" $existingToml) | default (randAlphaNum 10 | b64enc | quote) }}
key = "{{ dig "jwt" "signing" "key" (randAlphaNum 10 | b64enc) $securityConfig }}"
{{- if gt (int $expiresAfterSeconds.volumeWrite) 0 }}
expires_after_seconds = {{ int $expiresAfterSeconds.volumeWrite }}
{{- end }}
@@ -41,7 +41,7 @@ data:
# - the Volume server validates the JWT on reading
# the jwt defaults to expire after 60 seconds
[jwt.signing.read]
key = {{ include "seaweedfs.existingTomlKey" (list "jwt.signing.read" $existingToml) | default (randAlphaNum 10 | b64enc | quote) }}
key = "{{ dig "jwt" "signing" "read" "key" (randAlphaNum 10 | b64enc) $securityConfig }}"
{{- if gt (int $expiresAfterSeconds.volumeRead) 0 }}
expires_after_seconds = {{ int $expiresAfterSeconds.volumeRead }}
{{- end }}
@@ -53,7 +53,7 @@ data:
# - the Filer server validates the JWT on writing
# the jwt defaults to expire after 10 seconds
[jwt.filer_signing]
key = {{ include "seaweedfs.existingTomlKey" (list "jwt.filer_signing" $existingToml) | default (randAlphaNum 10 | b64enc | quote) }}
key = "{{ dig "jwt" "filer_signing" "key" (randAlphaNum 10 | b64enc) $securityConfig }}"
{{- if gt (int $expiresAfterSeconds.filerWrite) 0 }}
expires_after_seconds = {{ int $expiresAfterSeconds.filerWrite }}
{{- end }}
@@ -65,7 +65,7 @@ data:
# - the Filer server validates the JWT on reading
# the jwt defaults to expire after 60 seconds
[jwt.filer_signing.read]
key = {{ include "seaweedfs.existingTomlKey" (list "jwt.filer_signing.read" $existingToml) | default (randAlphaNum 10 | b64enc | quote) }}
key = "{{ dig "jwt" "filer_signing" "read" "key" (randAlphaNum 10 | b64enc) $securityConfig }}"
{{- if gt (int $expiresAfterSeconds.filerRead) 0 }}
expires_after_seconds = {{ int $expiresAfterSeconds.filerRead }}
{{- end }}
@@ -32,32 +32,13 @@ spec:
spec:
serviceAccountName: {{ $seaweedfsName }}-volume-resize-hook
restartPolicy: Never
{{- if .Values.volume.podSecurityContext.enabled }}
securityContext: {{- omit .Values.volume.podSecurityContext "enabled" | toYaml | nindent 8 }}
{{- end }}
containers:
- name: resize
image: {{ .Values.volume.resizeHook.image }}
{{- if and .Values.volume.containerSecurityContext.enabled .Values.volume.containerSecurityContext.readOnlyRootFilesystem }}
env:
- name: HOME
value: /tmp
{{- end }}
command: ["sh", "-xec"]
args:
- |
{{ $commands | indent 14 }}
{{- if and .Values.volume.containerSecurityContext.enabled .Values.volume.containerSecurityContext.readOnlyRootFilesystem }}
volumeMounts:
{{- include "seaweedfs.tmpDirVolumeMount" (list . .Values.volume.containerSecurityContext "" "") | nindent 12 }}
{{- end }}
{{- if .Values.volume.containerSecurityContext.enabled }}
securityContext: {{- omit .Values.volume.containerSecurityContext "enabled" | toYaml | nindent 12 }}
{{- end }}
{{- if and .Values.volume.containerSecurityContext.enabled .Values.volume.containerSecurityContext.readOnlyRootFilesystem }}
volumes:
{{- include "seaweedfs.tmpDirVolume" (list . .Values.volume.containerSecurityContext "" "" false) | nindent 8 }}
{{- end }}
---
apiVersion: v1
kind: ServiceAccount
@@ -104,7 +104,6 @@ spec:
fi
done
volumeMounts:
{{- include "seaweedfs.tmpDirVolumeMount" (list $ $volume.containerSecurityContext "" "") | nindent 12 }}
- name: idx
mountPath: /idx
{{- range $dir := $volume.dataDirs }}
@@ -225,7 +224,6 @@ spec:
{{ . }} \
{{- end }}
volumeMounts:
{{- include "seaweedfs.tmpDirVolumeMount" (list $ $volume.containerSecurityContext (tpl (printf "{{ $volumeName := \"%s\" }}%s" $volumeName ($volume.extraVolumeMounts | default "")) $) (tpl ($volume.extraVolumes | default "") $)) | nindent 12 }}
{{- range $dir := $volume.dataDirs }}
{{- if not ( eq $dir.type "custom" ) }}
- name: {{ $dir.name }}
@@ -308,7 +306,6 @@ spec:
{{- include "seaweedfs.tplvalues.render" (dict "value" (printf "{{ $volumeName := \"%s\" }}%s" $volumeName $volume.sidecars) "context" $) | nindent 8 }}
{{- end }}
volumes:
{{- include "seaweedfs.tmpDirVolume" (list $ $volume.containerSecurityContext (tpl (printf "{{ $volumeName := \"%s\" }}%s" $volumeName ($volume.extraVolumeMounts | default "")) $) (tpl ($volume.extraVolumes | default "") $) (and $initContainers_exists $volume.idx)) | nindent 8 }}
{{- range $dir := $volume.dataDirs }}
@@ -144,7 +144,6 @@ spec:
{{ $arg }}{{- if lt $index (sub (len $.Values.worker.extraArgs) 1) }} \{{ end }}
{{- end }}
volumeMounts:
{{- include "seaweedfs.tmpDirVolumeMount" (list . .Values.worker.containerSecurityContext (tpl (.Values.worker.extraVolumeMounts | default "") .) (tpl (.Values.worker.extraVolumes | default "") .)) | nindent 12 }}
{{- if or (eq .Values.worker.data.type "hostPath") (eq .Values.worker.data.type "emptyDir") (eq .Values.worker.data.type "existingClaim") }}
- name: worker-data
mountPath: {{ .Values.worker.workingDir }}
@@ -263,16 +262,15 @@ spec:
--metrics-ip=0.0.0.0 \
{{- end }}
--max-concurrency={{ .Values.worker.maxExecute }}
{{- if .Values.global.seaweedfs.enableSecurity }}
volumeMounts:
{{- include "seaweedfs.tmpDirVolumeMount" (list . .Values.worker.containerSecurityContext "" "") | nindent 12 }}
{{- if .Values.global.seaweedfs.enableSecurity }}
- name: ca-cert
readOnly: true
mountPath: /usr/local/share/ca-certificates/ca/
- name: worker-cert
readOnly: true
mountPath: /usr/local/share/ca-certificates/worker/
{{- end }}
{{- end }}
{{- if .Values.worker.lanceMetricsPort }}
ports:
- containerPort: {{ .Values.worker.lanceMetricsPort }}
@@ -304,7 +302,6 @@ spec:
{{- include "seaweedfs.tplvalues.render" (dict "value" .Values.worker.sidecars "context" $) | nindent 8 }}
{{- end }}
volumes:
{{- include "seaweedfs.tmpDirVolume" (list . .Values.worker.containerSecurityContext (tpl (.Values.worker.extraVolumeMounts | default "") .) (tpl (.Values.worker.extraVolumes | default "") .) (ne (include "seaweedfs.worker.lanceNamespaceUrl" .) "")) | nindent 8 }}
{{- if eq .Values.worker.data.type "hostPath" }}
- name: worker-data
hostPath:
+3 -202
View File
@@ -25,9 +25,6 @@ global:
imagePullPolicy: IfNotPresent
restartPolicy: Always
loggingLevel: 1
# Writable temporary storage used when containers run with a read-only root filesystem.
tmpDir:
sizeLimit: ""
enableSecurity: false
masterServer: null
# filerWrite: true mounts security.toml on filer + admin without needing
@@ -193,8 +190,6 @@ master:
podLabels: {}
# Annotations to be added to the master pods
# The chart sets checksum/master-config on master pods; other checksum/* keys
# can be used for custom rollouts.
podAnnotations: {}
# Annotations to be added to the master resources
@@ -263,8 +258,6 @@ master:
# runAsUser: 1000
# runAsGroup: 3000
# fsGroup: 2000
# seccompProfile:
# type: RuntimeDefault
podSecurityContext: {}
# Configure security context for Container
@@ -273,16 +266,7 @@ master:
# containerSecurityContext:
# enabled: true
# runAsUser: 2000
# runAsGroup: 3000
# runAsNonRoot: true
# privileged: false
# allowPrivilegeEscalation: false
# readOnlyRootFilesystem: true
# capabilities:
# drop:
# - ALL
# seccompProfile:
# type: RuntimeDefault
containerSecurityContext: {}
ingress:
@@ -574,8 +558,6 @@ volume:
# runAsUser: 1000
# runAsGroup: 3000
# fsGroup: 2000
# seccompProfile:
# type: RuntimeDefault
podSecurityContext: {}
# Configure security context for Container
@@ -584,16 +566,7 @@ volume:
# containerSecurityContext:
# enabled: true
# runAsUser: 2000
# runAsGroup: 3000
# runAsNonRoot: true
# privileged: false
# allowPrivilegeEscalation: false
# readOnlyRootFilesystem: true
# capabilities:
# drop:
# - ALL
# seccompProfile:
# type: RuntimeDefault
containerSecurityContext: {}
# used to configure livenessProbe on volume-server containers
@@ -800,8 +773,6 @@ filer:
podLabels: {}
# Annotations to be added to the filer pods
# The chart sets checksum/s3config on filer pods; other checksum/* keys can be
# used for custom rollouts.
podAnnotations: {}
# Annotations to be added to the filer resource
@@ -870,8 +841,6 @@ filer:
# runAsUser: 1000
# runAsGroup: 3000
# fsGroup: 2000
# seccompProfile:
# type: RuntimeDefault
podSecurityContext: {}
# Configure security context for Container
@@ -880,16 +849,7 @@ filer:
# containerSecurityContext:
# enabled: true
# runAsUser: 2000
# runAsGroup: 3000
# runAsNonRoot: true
# privileged: false
# allowPrivilegeEscalation: false
# readOnlyRootFilesystem: true
# capabilities:
# drop:
# - ALL
# seccompProfile:
# type: RuntimeDefault
containerSecurityContext: {}
ingresses:
@@ -1035,17 +995,6 @@ s3:
imageOverride: null
restartPolicy: null
replicas: 1
# Deployment update strategy, rendered as the Deployment's strategy
# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy
# Example:
# updateStrategy:
# type: RollingUpdate
# rollingUpdate:
# maxUnavailable: 0
# maxSurge: 25%
updateStrategy: {}
# Time the s3 pod is given to shut down, including any preStop hook
terminationGracePeriodSeconds: 10
bindAddress: 0.0.0.0
port: 8333
# add additional https port
@@ -1129,8 +1078,6 @@ s3:
podLabels: {}
# Annotations to be added to the s3 pods
# The chart sets checksum/s3config on s3 pods; other checksum/* keys can be
# used for custom rollouts.
podAnnotations: {}
# Annotations to be added to the s3 resources
@@ -1174,8 +1121,6 @@ s3:
# runAsUser: 1000
# runAsGroup: 3000
# fsGroup: 2000
# seccompProfile:
# type: RuntimeDefault
podSecurityContext: {}
# Configure security context for Container
@@ -1184,16 +1129,7 @@ s3:
# containerSecurityContext:
# enabled: true
# runAsUser: 2000
# runAsGroup: 3000
# runAsNonRoot: true
# privileged: false
# allowPrivilegeEscalation: false
# readOnlyRootFilesystem: true
# capabilities:
# drop:
# - ALL
# seccompProfile:
# type: RuntimeDefault
containerSecurityContext: {}
# You can also use emptyDir storage:
@@ -1238,17 +1174,6 @@ s3:
failureThreshold: 100
timeoutSeconds: 10
# Lifecycle hooks for the s3 container. A preStop sleep lets the pod leave
# the Service endpoints before it receives SIGTERM; keep it shorter than
# terminationGracePeriodSeconds.
# ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/
# Example:
# lifecycle:
# preStop:
# exec:
# command: ["sleep", "5"]
lifecycle: {}
createBucketsHook:
resources: {}
@@ -1356,33 +1281,7 @@ sftp:
priorityClassName: ""
schedulerName: ""
serviceAccountName: ""
# Configure security context for Pod
# ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/
# Example:
# podSecurityContext:
# enabled: true
# runAsUser: 1000
# runAsGroup: 3000
# fsGroup: 2000
# seccompProfile:
# type: RuntimeDefault
podSecurityContext: {}
# Configure security context for Container
# ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/
# Example:
# containerSecurityContext:
# enabled: true
# runAsUser: 2000
# runAsGroup: 3000
# runAsNonRoot: true
# privileged: false
# allowPrivilegeEscalation: false
# readOnlyRootFilesystem: true
# capabilities:
# drop:
# - ALL
# seccompProfile:
# type: RuntimeDefault
containerSecurityContext: {}
logs:
@@ -1434,11 +1333,10 @@ admin:
# kubelet's httpGet readiness/liveness probes (which dial the pod IP) to ever
# succeed. "0.0.0.0" restores the pre-4.46 behaviour of listening on all
# interfaces. A non-loopback address requires authentication: set
# admin.secret.adminPassword or admin.secret.existingSecret, supply
# admin.secret.adminPassword or admin.secret.existingSecret, or supply
# WEED_ADMIN_PASSWORD via admin.extraEnvironmentVars /
# admin.secretExtraEnvironmentVars, or opt out with admin.allowInsecureBind;
# otherwise the admin container will exit with a clear error rather than
# silently staying unready. The whole
# admin.secretExtraEnvironmentVars; otherwise the admin container will exit
# with a clear error rather than silently staying unready. The whole
# 127.0.0.0/8 range and ::1 are treated as loopback (matching weed admin).
# Set to a loopback address only if you also replace the httpGet probes.
# Note: the -ip flag requires SeaweedFS 4.46 or newer; pinning
@@ -1446,9 +1344,6 @@ admin:
ip: "0.0.0.0"
loggingOverrideLevel: null
# INSECURE: allow binding a non-loopback ip without authentication.
allowInsecureBind: false
# Admin authentication
secret:
# Name of an existing secret containing admin credentials. If set, adminUser and adminPassword below are ignored.
@@ -1529,33 +1424,7 @@ admin:
priorityClassName: ""
schedulerName: ""
serviceAccountName: ""
# Configure security context for Pod
# ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/
# Example:
# podSecurityContext:
# enabled: true
# runAsUser: 1000
# runAsGroup: 3000
# fsGroup: 2000
# seccompProfile:
# type: RuntimeDefault
podSecurityContext: {}
# Configure security context for Container
# ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/
# Example:
# containerSecurityContext:
# enabled: true
# runAsUser: 2000
# runAsGroup: 3000
# runAsNonRoot: true
# privileged: false
# allowPrivilegeEscalation: false
# readOnlyRootFilesystem: true
# capabilities:
# drop:
# - ALL
# seccompProfile:
# type: RuntimeDefault
containerSecurityContext: {}
extraEnvironmentVars: {}
@@ -1712,33 +1581,7 @@ worker:
priorityClassName: ""
schedulerName: ""
serviceAccountName: ""
# Configure security context for Pod
# ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/
# Example:
# podSecurityContext:
# enabled: true
# runAsUser: 1000
# runAsGroup: 3000
# fsGroup: 2000
# seccompProfile:
# type: RuntimeDefault
podSecurityContext: {}
# Configure security context for Container
# ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/
# Example:
# containerSecurityContext:
# enabled: true
# runAsUser: 2000
# runAsGroup: 3000
# runAsNonRoot: true
# privileged: false
# allowPrivilegeEscalation: false
# readOnlyRootFilesystem: true
# capabilities:
# drop:
# - ALL
# seccompProfile:
# type: RuntimeDefault
containerSecurityContext: {}
extraEnvironmentVars: {}
@@ -1945,8 +1788,6 @@ allInOne:
initContainers: "" # Init containers
sidecars: "" # Sidecar containers
annotations: {} # Annotations for the deployment
# The chart sets checksum/master-config and checksum/s3config on all-in-one
# pods; other checksum/* keys can be used for custom rollouts.
podAnnotations: {} # Annotations for the pods
podLabels: {} # Labels for the pods
@@ -1998,8 +1839,6 @@ allInOne:
# runAsUser: 1000
# runAsGroup: 3000
# fsGroup: 2000
# seccompProfile:
# type: RuntimeDefault
podSecurityContext: {}
# Configure security context for Container
@@ -2008,16 +1847,7 @@ allInOne:
# containerSecurityContext:
# enabled: true
# runAsUser: 2000
# runAsGroup: 3000
# runAsNonRoot: true
# privileged: false
# allowPrivilegeEscalation: false
# readOnlyRootFilesystem: true
# capabilities:
# drop:
# - ALL
# seccompProfile:
# type: RuntimeDefault
containerSecurityContext: {}
# Resource management
@@ -2056,33 +1886,7 @@ cosi:
# should have a secret key called seaweedfs_s3_config with an inline json configure
existingConfigSecret: null
# Configure security context for Pod
# ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/
# Example:
# podSecurityContext:
# enabled: true
# runAsUser: 1000
# runAsGroup: 3000
# fsGroup: 2000
# seccompProfile:
# type: RuntimeDefault
podSecurityContext: {}
# Configure security context for Container
# ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/
# Example:
# containerSecurityContext:
# enabled: true
# runAsUser: 2000
# runAsGroup: 3000
# runAsNonRoot: true
# privileged: false
# allowPrivilegeEscalation: false
# readOnlyRootFilesystem: true
# capabilities:
# drop:
# - ALL
# seccompProfile:
# type: RuntimeDefault
containerSecurityContext: {}
# used to assign a custom scheduler to cosi pods
@@ -2118,9 +1922,6 @@ certificates:
# Labels to be added to all the created pods
podLabels: {}
# Annotations to be added to all the created pods
# The chart sets checksum/master-config and checksum/s3config on pods whose
# rendered ConfigMaps or Secrets should trigger rollouts. Other checksum/* keys
# can be used for custom rollout annotations.
podAnnotations: {}
networkPolicy:
+1027 -1059
View File
File diff suppressed because it is too large Load Diff

Before

Width:  |  Height:  |  Size: 54 KiB

After

Width:  |  Height:  |  Size: 53 KiB

-293
View File
@@ -1,293 +0,0 @@
# This file is automatically @generated by Cargo.
# It is not intended for manual editing.
version = 4
[[package]]
name = "aws-lc-rs"
version = "1.18.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ce2b2dcc879c3bae0d371e77c99f2238400ef24ec001394befa67b6e543add9e"
dependencies = [
"aws-lc-sys",
"zeroize",
]
[[package]]
name = "aws-lc-sys"
version = "0.44.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f09fae7be8bb3174e05c6afdb34199e6dc0c7c04ba9fa237b1967adfbde27483"
dependencies = [
"cc",
"cmake",
"dunce",
"fs_extra",
"pkg-config",
]
[[package]]
name = "cc"
version = "1.4.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "509591b7bcd67f4ef775afad7662703b4935daaa6ec0e5605cfb1090b32a2b6d"
dependencies = [
"find-msvc-tools",
"jobserver",
"libc",
"shlex",
]
[[package]]
name = "cfg-if"
version = "1.0.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9330f8b2ff13f34540b44e946ef35111825727b38d33286ef986142615121801"
[[package]]
name = "cmake"
version = "0.1.58"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c0f78a02292a74a88ac736019ab962ece0bc380e3f977bf72e376c5d78ff0678"
dependencies = [
"cc",
]
[[package]]
name = "dunce"
version = "1.0.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "92773504d58c093f6de2459af4af33faa518c13451eb8f2b5698ed3d36e7c813"
[[package]]
name = "find-msvc-tools"
version = "0.1.11"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "d45db016d36b838f563236e9193d0ee6ce38f3f68b6c94e914b4929c96bbb890"
[[package]]
name = "fs_extra"
version = "1.3.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "42703706b716c37f96a77aea830392ad231f44c9e9a67872fa5548707e11b11c"
[[package]]
name = "getrandom"
version = "0.2.17"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ff2abc00be7fca6ebc474524697ae276ad847ad0a6b3faa4bcb027e9a4614ad0"
dependencies = [
"cfg-if",
"libc",
"wasi",
]
[[package]]
name = "getrandom"
version = "0.4.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "300e883d756b2e4ec94e02791f39b04b522276138852cfc41d9fb7e904106099"
dependencies = [
"cfg-if",
"libc",
"r-efi",
]
[[package]]
name = "jobserver"
version = "0.1.35"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1c00acbd29eabad4a2392fa0e921c874934dbbf4194312ad20f04a0ed67a3cb3"
dependencies = [
"getrandom 0.4.3",
"libc",
]
[[package]]
name = "libc"
version = "0.2.189"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "3eaf3ede3fee6db1a4c2ee091bf8a8b4dccdc6d17f656fb07896ee72867612f2"
[[package]]
name = "log"
version = "0.4.33"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "0ceec5bc11778974d1bcb055b18002eba7f4b3518b6a0081b3af5f21666da9ad"
[[package]]
name = "once_cell"
version = "1.21.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9f7c3e4beb33f85d45ae3e3a1792185706c8e16d043238c593331cc7cd313b50"
[[package]]
name = "pkg-config"
version = "0.3.34"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f6b464fbc74e149a392436b17d523f769e057cb6877f6a5c4618bc6f11800548"
[[package]]
name = "r-efi"
version = "6.0.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f8dcc9c7d52a811697d2151c701e0d08956f92b0e24136cf4cf27b57a6a0d9bf"
[[package]]
name = "ring"
version = "0.17.14"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "a4689e6c2294d81e88dc6261c768b63bc4fcdb852be6d1352498b114f61383b7"
dependencies = [
"cc",
"cfg-if",
"getrandom 0.2.17",
"libc",
"untrusted",
"windows-sys",
]
[[package]]
name = "rustls"
version = "0.23.43"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "0283386ce02abc0151e1761d08802dfe86c173b0b494af5cbc086574e453da06"
dependencies = [
"aws-lc-rs",
"log",
"once_cell",
"rustls-pki-types",
"rustls-webpki",
"subtle",
"zeroize",
]
[[package]]
name = "rustls-pki-types"
version = "1.15.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "2f4925028c7eb5d1fcdaf196971378ed9d2c1c4efc7dc5d011256f76c99c0a96"
dependencies = [
"zeroize",
]
[[package]]
name = "rustls-webpki"
version = "0.103.14"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "0527518605e68109d875e248ea259b6758801cf165e4b2c2733ae3b51f12535a"
dependencies = [
"aws-lc-rs",
"ring",
"rustls-pki-types",
"untrusted",
]
[[package]]
name = "seaweed-common"
version = "0.1.0"
dependencies = [
"rustls",
]
[[package]]
name = "shlex"
version = "2.0.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f8fadd59c855ef2080decdef8ff161eb6661b86933c9d82e5ba29dc602a55aba"
[[package]]
name = "subtle"
version = "2.6.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "13c2bddecc57b384dee18652358fb23172facb8a2c51ccc10d74c157bdea3292"
[[package]]
name = "untrusted"
version = "0.9.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8ecb6da28b8a351d773b68d5825ac39017e680750f980f3a1a85cd8dd28a47c1"
[[package]]
name = "wasi"
version = "0.11.1+wasi-snapshot-preview1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ccf3ec651a847eb01de73ccad15eb7d99f80485de043efb2f370cd654f4ea44b"
[[package]]
name = "windows-sys"
version = "0.52.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "282be5f36a8ce781fad8c8ae18fa3f9beff57ec1b52cb3de0789201425d9a33d"
dependencies = [
"windows-targets",
]
[[package]]
name = "windows-targets"
version = "0.52.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "9b724f72796e036ab90c1021d4780d4d3d648aca59e491e6b98e725b84e99973"
dependencies = [
"windows_aarch64_gnullvm",
"windows_aarch64_msvc",
"windows_i686_gnu",
"windows_i686_gnullvm",
"windows_i686_msvc",
"windows_x86_64_gnu",
"windows_x86_64_gnullvm",
"windows_x86_64_msvc",
]
[[package]]
name = "windows_aarch64_gnullvm"
version = "0.52.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "32a4622180e7a0ec044bb555404c800bc9fd9ec262ec147edd5989ccd0c02cd3"
[[package]]
name = "windows_aarch64_msvc"
version = "0.52.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "09ec2a7bb152e2252b53fa7803150007879548bc709c039df7627cabbd05d469"
[[package]]
name = "windows_i686_gnu"
version = "0.52.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8e9b5ad5ab802e97eb8e295ac6720e509ee4c243f69d781394014ebfe8bbfa0b"
[[package]]
name = "windows_i686_gnullvm"
version = "0.52.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "0eee52d38c090b3caa76c563b86c3a4bd71ef1a819287c19d586d7334ae8ed66"
[[package]]
name = "windows_i686_msvc"
version = "0.52.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "240948bc05c5e7c6dabba28bf89d89ffce3e303022809e73deaefe4f6ec56c66"
[[package]]
name = "windows_x86_64_gnu"
version = "0.52.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "147a5c80aabfbf0c7d901cb5895d1de30ef2907eb21fbbab29ca94c5b08b1a78"
[[package]]
name = "windows_x86_64_gnullvm"
version = "0.52.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "24d5b23dc417412679681396f2b49f3de8c1473deb516bd34410872eff51ed0d"
[[package]]
name = "windows_x86_64_msvc"
version = "0.52.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "589f6da84c646204747d1270a2a5661ea66ed1cced2631d546fdfb155959f9ec"
[[package]]
name = "zeroize"
version = "1.9.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e13c156562582aa81c60cb29407084cdb54c4164760106ab78e6c5b0858cf64e"
-27
View File
@@ -1,27 +0,0 @@
[package]
name = "seaweed-common"
version = "0.1.0"
edition = "2024"
# The lower of the two consumers' floors (seaweed-volume 1.91.1,
# seaweed-worker 1.94.1), so depending on this crate cannot raise either
# tree's MSRV. Verified with `cargo +1.91.1 check --all-targets`.
rust-version = "1.91.1"
description = "Helpers shared by the SeaweedFS Rust volume server and the Rust plugin workers"
# There is no root manifest: seaweed-volume and seaweed-worker are separate
# cargo trees with their own lockfiles, and this crate is a path dependency of
# both rather than a member of either. Keeping the lint policy identical in all
# three manifests is what stops them drifting.
[lints.clippy]
# Protobuf message literals keep `..Default::default()` on purpose: it is
# what lets a proto gain a field without touching every constructor.
needless_update = "allow"
[dependencies]
# The same requirement both consumers already write. Cargo unifies all
# semver-compatible `rustls = "0.23"` requirements into one crate per binary,
# which is what makes `install_default_crypto_provider` write the same
# process-wide static the consuming crate reads. rustls is already in both
# trees (the volume server directly, seaweed-worker-core through tonic's
# `tls-aws-lc`), so this adds no crate to either graph.
rustls = "0.23"
-307
View File
@@ -1,307 +0,0 @@
//! SeaweedFS server addresses, the way the Go tree does them.
//!
//! An operator gives a SeaweedFS process an HTTP address and the gRPC port is
//! derived from it rather than asked for separately: `host:port` means gRPC on
//! `port + 10000`, and the explicit `host:port.grpcPort` form names it outright.
//! Dialling the HTTP port by mistake fails as "frame with invalid size", which
//! reads like a protocol bug rather than a wrong port, so the rule is worth its
//! own module. Mirrors `pb.ServerToGrpcAddress` in
//! `weed/pb/grpc_client_server.go`.
//!
//! The volume server and the workers each had their own copy of this and the
//! copies had drifted: the worker's bracketed IPv6 literals and the volume
//! server's did not, so `::1:19333` produced `::1:29333`, which the HTTP
//! authority parser rejects. One implementation, two thin wrappers.
use std::fmt;
use std::num::ParseIntError;
/// SeaweedFS's HTTP↔gRPC port-offset convention.
pub const GRPC_PORT_OFFSET: u16 = 10000;
/// Why an address could not be turned into a gRPC address.
///
/// The `Display` text is the volume server's original wording, because its
/// `parse_grpc_address` wrapper hands it straight to callers that put it in a
/// `Status` or an `io::Error`.
#[derive(Debug, Clone, PartialEq, Eq)]
#[non_exhaustive]
pub enum AddressError {
/// No `:` at all, so there is no port to translate.
MissingPort(String),
/// The HTTP port of the `host:port.grpcPort` form is not a `u16`. It is
/// validated even though it is then discarded, so that a malformed address
/// is rejected here instead of failing later as an opaque connect error.
InvalidHttpPort { port: String, source: ParseIntError },
/// The gRPC port of the `host:port.grpcPort` form is not a `u16`.
InvalidGrpcPort { port: String, source: ParseIntError },
/// The port of the `host:port` form is not a `u16`.
InvalidPort { port: String, source: ParseIntError },
/// `port + GRPC_PORT_OFFSET` leaves the TCP port range, e.g. `host:60000`.
/// Without the check the cast would wrap silently.
ImplicitGrpcPortOutOfRange(u16),
}
impl fmt::Display for AddressError {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
match self {
Self::MissingPort(address) => write!(f, "cannot parse address: {address}"),
Self::InvalidHttpPort { port, source } => {
write!(f, "invalid http port {port:?}: {source}")
}
Self::InvalidGrpcPort { port, source } => {
write!(f, "invalid grpc port {port:?}: {source}")
}
Self::InvalidPort { port, source } => write!(f, "invalid port {port:?}: {source}"),
Self::ImplicitGrpcPortOutOfRange(port) => write!(
f,
"implicit grpc port out of range: {port} + {GRPC_PORT_OFFSET} = {}",
u32::from(*port) + u32::from(GRPC_PORT_OFFSET)
),
}
}
}
impl std::error::Error for AddressError {
fn source(&self) -> Option<&(dyn std::error::Error + 'static)> {
match self {
Self::InvalidHttpPort { source, .. }
| Self::InvalidGrpcPort { source, .. }
| Self::InvalidPort { source, .. } => Some(source),
Self::MissingPort(_) | Self::ImplicitGrpcPortOutOfRange(_) => None,
}
}
}
/// Turn a SeaweedFS server address (`"host:port.grpcPort"` or `"host:port"`)
/// into the `host:grpcPort` form the endpoint builders expect.
///
/// With the trailing `.grpcPort` segment that segment *is* the gRPC port;
/// without it the gRPC port is `port + GRPC_PORT_OFFSET`. An unbracketed IPv6
/// literal comes back bracketed, because otherwise the port reads as part of
/// the address.
pub fn to_grpc_address(server: &str) -> Result<String, AddressError> {
// rfind, not find: an IPv6 literal is full of colons and the port is after
// the last one.
let colon_idx = server
.rfind(':')
.ok_or_else(|| AddressError::MissingPort(server.to_string()))?;
let host = &server[..colon_idx];
let port_part = &server[colon_idx + 1..];
// rfind again rather than split_once: the host may be an IPv4 address, and
// only the part after the last colon is being split here anyway.
if let Some(dot_idx) = port_part.rfind('.') {
let http_port = &port_part[..dot_idx];
let grpc_port = &port_part[dot_idx + 1..];
http_port
.parse::<u16>()
.map_err(|source| AddressError::InvalidHttpPort {
port: http_port.to_string(),
source,
})?;
let grpc_port =
grpc_port
.parse::<u16>()
.map_err(|source| AddressError::InvalidGrpcPort {
port: grpc_port.to_string(),
source,
})?;
return Ok(join_host_port(host, grpc_port));
}
let port: u16 = port_part
.parse()
.map_err(|source| AddressError::InvalidPort {
port: port_part.to_string(),
source,
})?;
let grpc_port = port
.checked_add(GRPC_PORT_OFFSET)
.ok_or(AddressError::ImplicitGrpcPortOutOfRange(port))?;
Ok(join_host_port(host, grpc_port))
}
/// Join a host and a port, bracketing an IPv6 literal that is not bracketed
/// already. Public because the address rule is not the only place that has to
/// put a host and a port back together.
pub fn join_host_port(host: &str, port: u16) -> String {
// An IPv6 literal has to keep its brackets or the port reads as part of it.
if host.contains(':') && !host.starts_with('[') {
format!("[{host}]:{port}")
} else {
format!("{host}:{port}")
}
}
#[cfg(test)]
mod tests {
use super::{AddressError, GRPC_PORT_OFFSET, join_host_port, to_grpc_address};
// ---- the volume server's cases -------------------------------------
#[test]
fn dotted_form_states_the_grpc_port() {
assert_eq!(
to_grpc_address("127.0.0.1:8080.18080").unwrap(),
"127.0.0.1:18080"
);
assert_eq!(
to_grpc_address("192.168.1.66:8080.18080").unwrap(),
"192.168.1.66:18080"
);
}
#[test]
fn implicit_form_adds_the_offset() {
assert_eq!(
to_grpc_address("127.0.0.1:8080").unwrap(),
"127.0.0.1:18080"
);
assert_eq!(
to_grpc_address("192.168.1.66:8080").unwrap(),
"192.168.1.66:18080"
);
assert_eq!(
to_grpc_address("localhost:9333").unwrap(),
"localhost:19333"
);
}
#[test]
fn the_dotted_grpc_port_comes_back_normalised() {
// The volume server's copy validated this segment as a u16 and then
// emitted the original text, so a padded or signed port produced an
// authority the URI parser rejects. The parsed value is emitted now.
assert_eq!(to_grpc_address("host:8080.018080").unwrap(), "host:18080");
assert_eq!(to_grpc_address("host:8080.+18080").unwrap(), "host:18080");
}
#[test]
fn an_ipv4_host_is_not_confused_with_the_dotted_port() {
// Regression: a naive split on '.' breaks on IP addresses.
assert_eq!(
to_grpc_address("10.0.0.1:8080.18080").unwrap(),
"10.0.0.1:18080"
);
assert_eq!(to_grpc_address("10.0.0.1:8080").unwrap(), "10.0.0.1:18080");
}
#[test]
fn rejects_a_non_numeric_http_port_in_the_dotted_form() {
let err = to_grpc_address("host:abc.18080").unwrap_err();
assert!(
matches!(err, AddressError::InvalidHttpPort { .. }),
"{err:?}"
);
assert!(err.to_string().contains("invalid http port"), "{err}");
}
#[test]
fn rejects_a_non_numeric_grpc_port_in_the_dotted_form() {
let err = to_grpc_address("host:8080.xyz").unwrap_err();
assert!(
matches!(err, AddressError::InvalidGrpcPort { .. }),
"{err:?}"
);
assert!(err.to_string().contains("invalid grpc port"), "{err}");
}
#[test]
fn rejects_an_implicit_port_that_leaves_the_tcp_range() {
let err = to_grpc_address("127.0.0.1:60000").unwrap_err();
assert!(
matches!(err, AddressError::ImplicitGrpcPortOutOfRange(60000)),
"{err:?}"
);
assert!(err.to_string().contains("out of range"), "{err}");
}
#[test]
fn the_messages_are_the_volume_servers_wording_verbatim() {
// parse_grpc_address hands these straight to callers that put them in a
// Status or an io::Error, so the whole string is the contract, not just
// the substring the older tests match on. Only the two variants whose
// text is entirely ours are pinned exactly; the other three end in a
// std ParseIntError message, which is std's to reword.
assert_eq!(
to_grpc_address("127.0.0.1:60000").unwrap_err().to_string(),
"implicit grpc port out of range: 60000 + 10000 = 70000"
);
assert_eq!(
to_grpc_address("hostname").unwrap_err().to_string(),
"cannot parse address: hostname"
);
}
#[test]
fn rejects_an_address_without_a_port() {
for source in ["hostname", "no-colon", "localhost"] {
let err = to_grpc_address(source).unwrap_err();
assert!(matches!(err, AddressError::MissingPort(_)), "{err:?}");
assert!(err.to_string().contains("cannot parse"), "{err}");
}
}
// ---- the worker's cases --------------------------------------------
#[test]
fn derives_the_grpc_port() {
assert_eq!(
to_grpc_address("localhost:23646").unwrap(),
"localhost:33646"
);
assert_eq!(
to_grpc_address("127.0.0.1:9333").unwrap(),
"127.0.0.1:19333"
);
}
#[test]
fn honours_an_explicit_grpc_port() {
assert_eq!(
to_grpc_address("localhost:23646.33999").unwrap(),
"localhost:33999"
);
}
#[test]
fn rejects_what_it_cannot_parse() {
let err = to_grpc_address("localhost:notaport").unwrap_err();
assert!(matches!(err, AddressError::InvalidPort { .. }), "{err:?}");
assert!(err.to_string().contains("invalid port"), "{err}");
}
// ---- IPv6, which only the worker's copy handled --------------------
#[test]
fn brackets_ipv6_literals() {
assert_eq!(to_grpc_address("::1:23646").unwrap(), "[::1]:33646");
assert_eq!(to_grpc_address("::1:9333").unwrap(), "[::1]:19333");
assert_eq!(
to_grpc_address("fe80::1:9333.19333").unwrap(),
"[fe80::1]:19333"
);
}
#[test]
fn leaves_an_already_bracketed_literal_alone() {
assert_eq!(to_grpc_address("[::1]:9333").unwrap(), "[::1]:19333");
assert_eq!(to_grpc_address("[::1]:9333.19333").unwrap(), "[::1]:19333");
}
#[test]
fn join_host_port_brackets_only_unbracketed_literals() {
assert_eq!(join_host_port("127.0.0.1", 19333), "127.0.0.1:19333");
assert_eq!(join_host_port("localhost", 19333), "localhost:19333");
assert_eq!(join_host_port("::1", 19333), "[::1]:19333");
assert_eq!(join_host_port("[::1]", 19333), "[::1]:19333");
}
// ---- the offset itself ---------------------------------------------
#[test]
fn the_offset_is_the_seaweedfs_convention() {
assert_eq!(GRPC_PORT_OFFSET, 10000);
}
}
-11
View File
@@ -1,11 +0,0 @@
//! Helpers the SeaweedFS Rust volume server and the Rust plugin workers both need.
//!
//! `seaweed-volume` and `seaweed-worker` are separate cargo trees with separate
//! lockfiles and no root manifest, so anything both of them need was, until this
//! crate existed, written twice. The two things in here are the ones where a
//! second copy is a correctness risk rather than a typing cost: the HTTP↔gRPC
//! address rule, which two copies had already drifted on, and the process-wide
//! rustls provider, which only works if every binary installs the same one.
pub mod address;
pub mod tls;
-28
View File
@@ -1,28 +0,0 @@
//! The process-wide rustls crypto provider.
//!
//! Both binaries link aws-lc-rs and ring transitively — in the volume server
//! through the AWS SDK and reqwest, in the lance worker through lance's `aws`
//! backend and reqwest — so rustls cannot auto-select a provider and tonic's
//! client TLS panics on first use. Each binary has to pin one, and it has to be
//! the same one, which is why the choice lives here rather than in either tree.
use rustls::crypto::aws_lc_rs;
/// Pin rustls's process-wide default provider to aws-lc-rs, matching the
/// volume server's TLS config. Idempotent: the first call wins and every
/// later one is a no-op, so callers do not have to coordinate.
pub fn install_default_crypto_provider() {
let _ = aws_lc_rs::default_provider().install_default();
}
#[cfg(test)]
mod tests {
use super::install_default_crypto_provider;
#[test]
fn installing_is_idempotent_and_leaves_a_default_behind() {
install_default_crypto_provider();
install_default_crypto_provider();
assert!(rustls::crypto::CryptoProvider::get_default().is_some());
}
}
+129 -124
View File
@@ -503,7 +503,7 @@ dependencies = [
"rustls-pki-types",
"tokio",
"tokio-rustls",
"tower",
"tower 0.5.3",
"tracing",
]
@@ -628,13 +628,13 @@ dependencies = [
[[package]]
name = "axum"
version = "0.8.9"
version = "0.7.9"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "31b698c5f9a010f6573133b09e0de5408834d0c82f8d7475a89fc1867a71cd90"
checksum = "edca88bc138befd0323b20752846e6587272d3b03b0343c8ea28a6f819e6e71f"
dependencies = [
"async-trait",
"axum-core",
"bytes",
"form_urlencoded",
"futures-util",
"http 1.4.0",
"http-body 1.0.1",
@@ -648,13 +648,14 @@ dependencies = [
"multer",
"percent-encoding",
"pin-project-lite",
"serde_core",
"rustversion",
"serde",
"serde_json",
"serde_path_to_error",
"serde_urlencoded",
"sync_wrapper",
"tokio",
"tower",
"tower 0.5.3",
"tower-layer",
"tower-service",
"tracing",
@@ -662,17 +663,19 @@ dependencies = [
[[package]]
name = "axum-core"
version = "0.5.6"
version = "0.4.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "08c78f31d7b1291f7ee735c1c6780ccde7785daae9a9206026862dab7d8792d1"
checksum = "09f2bd6146b97ae3359fa0cc6d6b376d9539582c7b4220f041a33ec24c226199"
dependencies = [
"async-trait",
"bytes",
"futures-core",
"futures-util",
"http 1.4.0",
"http-body 1.0.1",
"http-body-util",
"mime",
"pin-project-lite",
"rustversion",
"sync_wrapper",
"tower-layer",
"tower-service",
@@ -1651,13 +1654,19 @@ dependencies = [
"futures-core",
"futures-sink",
"http 1.4.0",
"indexmap",
"indexmap 2.13.1",
"slab",
"tokio",
"tokio-util",
"tracing",
]
[[package]]
name = "hashbrown"
version = "0.12.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8a9ee70c43aaf417c914396645a0fa852624801b24ebb7ae78fe8272889ac888"
[[package]]
name = "hashbrown"
version = "0.14.5"
@@ -1851,7 +1860,7 @@ dependencies = [
"libc",
"percent-encoding",
"pin-project-lite",
"socket2",
"socket2 0.6.3",
"tokio",
"tower-service",
"tracing",
@@ -2018,6 +2027,16 @@ dependencies = [
"quick-error",
]
[[package]]
name = "indexmap"
version = "1.9.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "bd070e393353796e801d209ad339e89596eb4c8d430d18ede6a1cced8fafbd99"
dependencies = [
"autocfg",
"hashbrown 0.12.3",
]
[[package]]
name = "indexmap"
version = "2.13.1"
@@ -2231,9 +2250,9 @@ dependencies = [
[[package]]
name = "matchit"
version = "0.8.4"
version = "0.7.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "47e1ffaa40ddd1f3ed91f717a33c8c0ee23fff369e3aa8772b9605cc1d22f4c3"
checksum = "0e7465ac9959cc2b1404e8e2367b43684a6d13790fe23056cc8c6c5a6b7bcb94"
[[package]]
name = "md-5"
@@ -2600,18 +2619,17 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b4c5cc86750666a3ed20bdaf5ca2a0344f9c67674cae0515bec2da16fbaa47db"
dependencies = [
"fixedbitset 0.4.2",
"indexmap",
"indexmap 2.13.1",
]
[[package]]
name = "petgraph"
version = "0.8.3"
version = "0.7.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "8701b58ea97060d5e5b155d383a69952a60943f0e6dfe30b04c287beb0b27455"
checksum = "3672b37090dbd86368a4145bc067582552b29c27377cad4e0a306c97f9bd7772"
dependencies = [
"fixedbitset 0.5.7",
"hashbrown 0.15.5",
"indexmap",
"indexmap 2.13.1",
]
[[package]]
@@ -2818,12 +2836,12 @@ dependencies = [
[[package]]
name = "prost"
version = "0.14.4"
version = "0.13.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "528ac67416ff8646872a3c02cad9cc4ee5dc9f9540c9b10771855c95cb2e5ae1"
checksum = "2796faa41db3ec313a31f7624d9286acf277b52de526150b7e69f3debf891ee5"
dependencies = [
"bytes",
"prost-derive 0.14.4",
"prost-derive 0.13.5",
]
[[package]]
@@ -2849,20 +2867,19 @@ dependencies = [
[[package]]
name = "prost-build"
version = "0.14.4"
version = "0.13.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "03da047801ff44bb6a4d407d4860c05fd70bb81714e6b2f3812603d5b145b042"
checksum = "be769465445e8c1474e9c5dac2018218498557af32d9ed057325ec9a41ae81bf"
dependencies = [
"heck",
"itertools 0.14.0",
"log",
"multimap",
"petgraph 0.8.3",
"once_cell",
"petgraph 0.7.1",
"prettyplease",
"prost 0.14.4",
"prost-types 0.14.4",
"pulldown-cmark",
"pulldown-cmark-to-cmark",
"prost 0.13.5",
"prost-types 0.13.5",
"regex",
"syn",
"tempfile",
@@ -2883,9 +2900,9 @@ dependencies = [
[[package]]
name = "prost-derive"
version = "0.14.4"
version = "0.13.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b570b25f7617e43d59005d0990ccb79e950a423952cea19671b7a876da390adf"
checksum = "8a56d757972c98b346a9b766e3f02746cde6dd1cd1d1d563472929fdd74bec4d"
dependencies = [
"anyhow",
"itertools 0.14.0",
@@ -2905,11 +2922,11 @@ dependencies = [
[[package]]
name = "prost-types"
version = "0.14.4"
version = "0.13.5"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "f94967dc7688f3054c7fac87473ffae4cc4c3904800e2d9f5b857246d8963b0a"
checksum = "52c2c1bf36ddb1a1c396b3601a3cec27c2462e45f07c386894ec3ccf5332bd16"
dependencies = [
"prost 0.14.4",
"prost 0.13.5",
]
[[package]]
@@ -2976,26 +2993,6 @@ version = "3.2.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "95067976aca6421a523e491fce939a3e65249bac4b977adee0ee9771568e8aa3"
[[package]]
name = "pulldown-cmark"
version = "0.13.4"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e9f068eba8e7071c5f9511831b44f32c740d5adf574e990f946ddb53db2f314e"
dependencies = [
"bitflags 2.11.0",
"memchr",
"unicase",
]
[[package]]
name = "pulldown-cmark-to-cmark"
version = "22.0.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ab1ad36992cead65f02aa399a373a42730922f1525d988172634fdefdecb8a60"
dependencies = [
"pulldown-cmark",
]
[[package]]
name = "pxfm"
version = "0.1.28"
@@ -3021,7 +3018,7 @@ dependencies = [
"quinn-udp",
"rustc-hash",
"rustls",
"socket2",
"socket2 0.6.3",
"thiserror 2.0.18",
"tokio",
"tracing",
@@ -3059,7 +3056,7 @@ dependencies = [
"cfg_aliases",
"libc",
"once_cell",
"socket2",
"socket2 0.6.3",
"tracing",
"windows-sys 0.60.2",
]
@@ -3265,8 +3262,8 @@ dependencies = [
"tokio",
"tokio-rustls",
"tokio-util",
"tower",
"tower-http",
"tower 0.5.3",
"tower-http 0.6.8",
"tower-service",
"url",
"wasm-bindgen",
@@ -3487,13 +3484,6 @@ version = "1.2.0"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "94143f37725109f92c262ed2cf5e59bce7498c01bcc1502d7b9afe439a4e9f49"
[[package]]
name = "seaweed-common"
version = "0.1.0"
dependencies = [
"rustls",
]
[[package]]
name = "sec1"
version = "0.3.0"
@@ -3729,6 +3719,16 @@ version = "1.1.1"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1b6b67fb9a61334225b5b790716f609cd58395f895b3fe8b328786812a40bc3b"
[[package]]
name = "socket2"
version = "0.5.10"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "e22376abed350d73dd1cd119b57ffccad95b4e585a7cda43e286245ce23c0678"
dependencies = [
"libc",
"windows-sys 0.52.0",
]
[[package]]
name = "socket2"
version = "0.6.3"
@@ -3990,7 +3990,7 @@ dependencies = [
"parking_lot 0.12.5",
"pin-project-lite",
"signal-hook-registry",
"socket2",
"socket2 0.6.3",
"tokio-macros",
"windows-sys 0.61.2",
]
@@ -4077,7 +4077,7 @@ version = "0.22.27"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "41fe8c660ae4257887cf66394862d21dbca4a6ddd26f04a3560410406a2f819a"
dependencies = [
"indexmap",
"indexmap 2.13.1",
"serde",
"serde_spanned",
"toml_datetime",
@@ -4093,10 +4093,11 @@ checksum = "5d99f8c9a7727884afe522e9bd5edbfc91a3312b36a77b5fb8926e4c31a41801"
[[package]]
name = "tonic"
version = "0.14.6"
version = "0.12.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "ac2a5518c70fa84342385732db33fb3f44bc4cc748936eb5833d2df34d6445ef"
checksum = "877c5b330756d856ffcc4553ab34a5684481ade925ecc54bcd1bf02b1d0d4d52"
dependencies = [
"async-stream",
"async-trait",
"axum",
"base64",
@@ -4110,12 +4111,13 @@ dependencies = [
"hyper-util",
"percent-encoding",
"pin-project",
"socket2",
"sync_wrapper",
"prost 0.13.5",
"rustls-pemfile",
"socket2 0.5.10",
"tokio",
"tokio-rustls",
"tokio-stream",
"tower",
"tower 0.4.13",
"tower-layer",
"tower-service",
"tracing",
@@ -4123,55 +4125,49 @@ dependencies = [
[[package]]
name = "tonic-build"
version = "0.14.6"
version = "0.12.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "c68f61875ac5293cf72e6c8cf0158086428c82c37229e98c840878f1706b0322"
checksum = "9557ce109ea773b399c9b9e5dca39294110b74f1f342cb347a80d1fce8c26a11"
dependencies = [
"prettyplease",
"proc-macro2",
"prost-build 0.13.5",
"prost-types 0.13.5",
"quote",
"syn",
]
[[package]]
name = "tonic-prost"
version = "0.14.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "50849f68853be452acf590cde0b146665b8d507b3b8af17261df47e02c209ea0"
dependencies = [
"bytes",
"prost 0.14.4",
"tonic",
]
[[package]]
name = "tonic-prost-build"
version = "0.14.6"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "654e5643eff75d7f8c99197ce1440ed19a3474eada74c12bbac488b2cafdae27"
dependencies = [
"prettyplease",
"proc-macro2",
"prost-build 0.14.4",
"prost-types 0.14.4",
"quote",
"syn",
"tempfile",
"tonic-build",
]
[[package]]
name = "tonic-reflection"
version = "0.14.6"
version = "0.12.3"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "acccd136a4bf19810a1fde9c74edc6129b42a66b44d0c1c8aaa67aeb49a146a7"
checksum = "878d81f52e7fcfd80026b7fdb6a9b578b3c3653ba987f87f0dce4b64043cba27"
dependencies = [
"prost 0.14.4",
"prost-types 0.14.4",
"prost 0.13.5",
"prost-types 0.13.5",
"tokio",
"tokio-stream",
"tonic",
"tonic-prost",
]
[[package]]
name = "tower"
version = "0.4.13"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "b8fa9be0de6cf49e536ce1851f987bd21a43b771b09473c3549a6c853db37c1c"
dependencies = [
"futures-core",
"futures-util",
"indexmap 1.9.3",
"pin-project",
"pin-project-lite",
"rand 0.8.7",
"slab",
"tokio",
"tokio-util",
"tower-layer",
"tower-service",
"tracing",
]
[[package]]
@@ -4182,12 +4178,26 @@ checksum = "ebe5ef63511595f1344e2d5cfa636d973292adc0eec1f0ad45fae9f0851ab1d4"
dependencies = [
"futures-core",
"futures-util",
"indexmap",
"pin-project-lite",
"slab",
"sync_wrapper",
"tokio",
"tokio-util",
"tower-layer",
"tower-service",
"tracing",
]
[[package]]
name = "tower-http"
version = "0.5.2"
source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "1e9cd434a998747dd2c4276bc96ee2e0c7a2eadf3cae88e52be55a05fa9053f5"
dependencies = [
"bitflags 2.11.0",
"bytes",
"http 1.4.0",
"http-body 1.0.1",
"http-body-util",
"pin-project-lite",
"tower-layer",
"tower-service",
"tracing",
@@ -4206,10 +4216,9 @@ dependencies = [
"http-body 1.0.1",
"iri-string",
"pin-project-lite",
"tower",
"tower 0.5.3",
"tower-layer",
"tower-service",
"tracing",
]
[[package]]
@@ -4486,7 +4495,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
checksum = "bb0e353e6a2fbdc176932bbaab493762eb1255a7900fe0fea1a2f96c296cc909"
dependencies = [
"anyhow",
"indexmap",
"indexmap 2.13.1",
"wasm-encoder",
"wasmparser",
]
@@ -4512,7 +4521,7 @@ checksum = "47b807c72e1bac69382b3a6fb3dbe8ea4c0ed87ff5629b8685ae6b9a611028fe"
dependencies = [
"bitflags 2.11.0",
"hashbrown 0.15.5",
"indexmap",
"indexmap 2.13.1",
"semver",
]
@@ -4555,7 +4564,6 @@ dependencies = [
"aws-config",
"aws-credential-types",
"aws-sdk-s3",
"aws-smithy-runtime-api",
"aws-types",
"axum",
"base64",
@@ -4582,8 +4590,8 @@ dependencies = [
"parking_lot 0.12.5",
"pprof",
"prometheus",
"prost 0.14.4",
"prost-types 0.14.4",
"prost 0.13.5",
"prost-types 0.13.5",
"protoc-bin-vendored",
"rand 0.10.2",
"redb",
@@ -4592,7 +4600,6 @@ dependencies = [
"rustls",
"rustls-pemfile",
"rusty-leveldb",
"seaweed-common",
"serde",
"serde_json",
"serde_urlencoded",
@@ -4605,15 +4612,13 @@ dependencies = [
"tokio-stream",
"toml",
"tonic",
"tonic-prost",
"tonic-prost-build",
"tonic-build",
"tonic-reflection",
"tower",
"tower-http",
"tower 0.4.13",
"tower-http 0.5.2",
"tracing",
"tracing-subscriber",
"uuid",
"windows-sys 0.61.2",
"x509-parser",
"xxhash-rust",
]
@@ -4960,7 +4965,7 @@ checksum = "b7c566e0f4b284dd6561c786d9cb0142da491f46a9fbed79ea69cdad5db17f21"
dependencies = [
"anyhow",
"heck",
"indexmap",
"indexmap 2.13.1",
"prettyplease",
"syn",
"wasm-metadata",
@@ -4991,7 +4996,7 @@ checksum = "9d66ea20e9553b30172b5e831994e35fbde2d165325bec84fc43dbf6f4eb9cb2"
dependencies = [
"anyhow",
"bitflags 2.11.0",
"indexmap",
"indexmap 2.13.1",
"log",
"serde",
"serde_derive",
@@ -5010,7 +5015,7 @@ checksum = "ecc8ac4bc1dc3381b7f59c34f00b67e18f910c2c0f50015669dde7def656a736"
dependencies = [
"anyhow",
"id-arena",
"indexmap",
"indexmap 2.13.1",
"log",
"semver",
"serde",
+11 -24
View File
@@ -24,37 +24,32 @@ default = ["5bytes"]
redb-experimental-cursor = ["redb/experimental_cursor"]
[lints.clippy]
# Every RPC path returns tonic::Status (176 bytes). Boxing it would change
# every handler signature for no gain, so the large-Err lint is off.
result_large_err = "allow"
# Protobuf message literals keep `..Default::default()` on purpose: it is
# what lets a proto gain a field without touching every constructor.
needless_update = "allow"
# Every `unsafe` block states its precondition, right above the block.
undocumented_unsafe_blocks = "warn"
[dependencies]
# Helpers the Rust plugin workers (seaweed-worker) need as well. A path
# dependency because the two trees are separate cargo workspaces with no
# common root manifest.
seaweed-common = { path = "../seaweed-common" }
# Async runtime
tokio = { version = "1", features = ["full"] }
tokio-stream = { version = "0.1", features = ["net"] }
tokio-io-timeout = "1"
# gRPC + protobuf
tonic = { version = "0.14", features = ["tls-aws-lc"] }
tonic-prost = "0.14"
tonic-reflection = "0.14"
prost = "0.14"
prost-types = "0.14"
tonic = { version = "0.12", features = ["tls"] }
tonic-reflection = "0.12"
prost = "0.13"
prost-types = "0.13"
# HTTP server
axum = { version = "0.8", features = ["multipart"] }
axum = { version = "0.7", features = ["multipart"] }
http-body = "1"
hyper = { version = "1", features = ["full"] }
hyper-util = { version = "0.1", features = ["tokio", "service", "server-auto", "http1", "http2"] }
tower = { version = "0.5", features = ["util"] }
tower-http = { version = "0.6", features = ["cors", "trace"] }
tower = "0.4"
tower-http = { version = "0.5", features = ["cors", "trace"] }
# CLI
clap = { version = "4", features = ["derive"] }
@@ -150,19 +145,11 @@ aws-types = "1"
[target.'cfg(unix)'.dependencies]
pprof = { version = "0.15", features = ["prost-codec"] }
# GetDiskFreeSpaceExW for per-path disk capacity on Windows (0.61.2 already
# in the tree via tempfile/mio, so this unifies rather than adding a version).
[target.'cfg(windows)'.dependencies]
windows-sys = { version = "0.61", features = ["Win32_Storage_FileSystem"] }
[dev-dependencies]
tempfile = "3"
# Already a transitive dependency of aws-sdk-s3 at a single locked version;
# needed directly only for the canned HttpClient in remote_storage::s3 tests.
aws-smithy-runtime-api = "1"
[build-dependencies]
tonic-prost-build = "0.14"
tonic-build = "0.12"
# Ships protoc with the build so neither CI nor a developer needs a system
# install, and so the version is pinned rather than whatever the platform's
# package manager happens to carry.
+1 -1
View File
@@ -12,7 +12,7 @@ fn main() -> Result<(), Box<dyn std::error::Error>> {
}
let out_dir = std::path::PathBuf::from(std::env::var("OUT_DIR")?);
tonic_prost_build::configure()
tonic_build::configure()
.build_server(true)
.build_client(true)
// filer.proto uses proto3 optional, which protoc rejects without this
-3
View File
@@ -213,9 +213,6 @@ message KeepConnectedRequest {
string filer_group = 5;
string data_center = 6;
string rack = 7;
// A draining filer leaves the lock ring but stays a cluster member, so peers
// keep following its metadata until the stream closes.
bool leave_lock_ring = 8;
}
message VolumeLocation {
-6
View File
@@ -168,7 +168,6 @@ message VacuumVolumeCheckRequest {
}
message VacuumVolumeCheckResponse {
double garbage_ratio = 1;
bool disk_space_low = 4; // the volume is read-only solely because its disk is low on space — a cause compaction itself reclaims
}
message VacuumVolumeCompactRequest {
@@ -260,10 +259,6 @@ message VolumeDeleteRequest {
// when true, do not remove the cloud-tier object backing the volume.
// used for moves where another server is taking over the same .vif.
bool keep_remote_data = 3;
// when true, delete only if every needle is deleted: the volume held
// data once but nothing is live anymore. Passing either check,
// only_empty or this one, is enough to delete.
bool only_garbage = 4;
}
message VolumeDeleteResponse {
}
@@ -476,7 +471,6 @@ message VolumeEcShardsDeleteRequest {
repeated uint32 shard_ids = 3;
bool full_teardown = 4; // pre-encode cleanup: wipe every EC artifact + generation for this volume, not just shard_ids
int64 encode_ts_ns = 5; // full_teardown generation fence: delete only a disk whose .vif generation is strictly OLDER than this; preserve same-or-newer, generation 0, and an unreadable .vif. 0 => wipe-all (shell pre-encode / pre-upgrade)
uint32 delete_generations_older_than = 6; // post-commit cleanup: delete only staged <base>.*.v<N> artifacts with N strictly below this; 0 disables
}
message VolumeEcShardsDeleteResponse {
bool full_teardown_done = 1; // set by a new server that performed full_teardown; absent from an old server lets the caller detect the silent no-op
File diff suppressed because it is too large Load Diff
+35 -41
View File
@@ -6,22 +6,19 @@ use seaweed_volume::config::{self, VolumeServerConfig};
use seaweed_volume::metrics;
use seaweed_volume::pb::volume_server_pb::volume_server_server::VolumeServerServer;
use seaweed_volume::security::tls::{
GrpcClientAuthPolicy, TlsPolicy, build_rustls_server_config,
build_rustls_server_config_with_grpc_client_auth, install_default_crypto_provider,
build_rustls_server_config, build_rustls_server_config_with_grpc_client_auth,
install_default_crypto_provider, GrpcClientAuthPolicy, TlsPolicy,
};
use seaweed_volume::security::{Guard, SigningKey};
#[cfg(unix)]
use seaweed_volume::server::debug::build_debug_router;
use seaweed_volume::server::grpc_client::{
GRPC_INITIAL_WINDOW_SIZE, GRPC_KEEPALIVE_INTERVAL, GRPC_KEEPALIVE_TIMEOUT,
GRPC_MAX_MESSAGE_SIZE, load_outgoing_grpc_tls,
};
use seaweed_volume::server::grpc_client::load_outgoing_grpc_tls;
use seaweed_volume::server::grpc_server::VolumeGrpcService;
#[cfg(unix)]
use seaweed_volume::server::profiling::CpuProfileSession;
use seaweed_volume::server::request_id::GrpcRequestIdLayer;
use seaweed_volume::server::volume_server::{
RuntimeMetricsConfig, VolumeServerState, build_metrics_router,
build_metrics_router, RuntimeMetricsConfig, VolumeServerState,
};
use seaweed_volume::server::write_queue::WriteQueue;
use seaweed_volume::storage::store::Store;
@@ -34,10 +31,10 @@ type CpuProfileParam = Option<CpuProfileSession>;
#[cfg(not(unix))]
type CpuProfileParam = Option<()>;
// The two settings that only make sense for the inbound server. The rest of
// this server's HTTP/2 tuning — keepalive, window sizes, message size — is
// imported from `server::grpc_client` above, which is also what the outgoing
// clients dial with, so the two directions cannot drift apart.
const GRPC_MAX_MESSAGE_SIZE: usize = 1 << 30;
const GRPC_KEEPALIVE_INTERVAL: std::time::Duration = std::time::Duration::from_secs(60);
const GRPC_KEEPALIVE_TIMEOUT: std::time::Duration = std::time::Duration::from_secs(20);
const GRPC_INITIAL_WINDOW_SIZE: u32 = 16 * 1024 * 1024;
const GRPC_MAX_HEADER_LIST_SIZE: u32 = 8 * 1024 * 1024;
const GRPC_MAX_CONCURRENT_STREAMS: u32 = 1000;
@@ -346,6 +343,9 @@ async fn run(
pre_stop_seconds: config.pre_stop_seconds,
volume_state_notify: tokio::sync::Notify::new(),
write_queue: std::sync::OnceLock::new(),
s3_tier_registry: std::sync::RwLock::new(
seaweed_volume::remote_storage::s3_tier::S3TierRegistry::new(),
),
read_mode: config.read_mode,
allow_untrusted_remote_endpoints: config.allow_untrusted_remote_endpoints,
master_url,
@@ -371,9 +371,6 @@ async fn run(
.to_string_lossy()
.into_owned()
},
ec_decodes_in_flight: std::sync::Mutex::new(std::collections::HashSet::new()),
ec_decode_tail: std::sync::Mutex::new(std::collections::HashSet::new()),
ec_decode_tail_notify: tokio::sync::Notify::new(),
});
// Load persisted state from disk if it exists (matches Go's State.Load on startup)
@@ -674,7 +671,8 @@ async fn run(
})
.await
} else {
let incoming = tokio_stream::wrappers::TcpListenerStream::new(grpc_listener);
let incoming =
tokio_stream::wrappers::TcpListenerStream::new(grpc_listener);
info!("gRPC server listening on {}", grpc_local_addr);
build_grpc_server_builder()
.layer(GrpcRequestIdLayer)
@@ -1060,17 +1058,15 @@ mod tests {
#[test]
fn test_grpc_server_tls_returns_none_when_files_are_missing() {
assert!(
build_grpc_server_tls_acceptor(
"/missing/server.crt",
"/missing/server.key",
"/missing/ca.crt",
&TlsPolicy::default(),
"",
&[],
)
.is_none()
);
assert!(build_grpc_server_tls_acceptor(
"/missing/server.crt",
"/missing/server.key",
"/missing/ca.crt",
&TlsPolicy::default(),
"",
&[],
)
.is_none());
}
#[test]
@@ -1092,21 +1088,19 @@ mod tests {
"-----BEGIN CERTIFICATE-----\nZmFrZQ==\n-----END CERTIFICATE-----\n",
);
assert!(
build_grpc_server_tls_acceptor(
&cert,
&key,
&ca,
&TlsPolicy {
min_version: "TLS 1.0".to_string(),
max_version: "TLS 1.1".to_string(),
cipher_suites: String::new(),
},
"",
&[],
)
.is_none()
);
assert!(build_grpc_server_tls_acceptor(
&cert,
&key,
&ca,
&TlsPolicy {
min_version: "TLS 1.0".to_string(),
max_version: "TLS 1.1".to_string(),
cipher_suites: String::new(),
},
"",
&[],
)
.is_none());
}
#[test]
+2 -5
View File
@@ -349,7 +349,6 @@ pub const DOWNLOAD_LIMIT_COND: &str = "downloadLimitCondition";
pub const UPLOAD_LIMIT_COND: &str = "uploadLimitCondition";
pub const READ_PROXY_REQ: &str = "readProxyRequest";
pub const READ_REDIRECT_REQ: &str = "readRedirectRequest";
pub const READ_DELETED_NEEDLE: &str = "readDeletedNeedle";
pub const EMPTY_READ_PROXY_LOC: &str = "emptyReadProxyLocaction";
pub const FAILED_READ_PROXY_REQ: &str = "failedReadProxyRequest";
@@ -526,7 +525,7 @@ pub async fn push_metrics_once(
#[cfg(test)]
mod tests {
use super::*;
use axum::{Router, routing::put};
use axum::{routing::put, Router};
use std::sync::{Arc, Mutex};
#[test]
@@ -598,9 +597,7 @@ mod tests {
register_metrics();
VOLUME_GAUGE.with_label_values(&["pics", "volume"]).set(2.0);
VOLUME_GAUGE
.with_label_values(&["pics", "ec_shards"])
.set(3.0);
VOLUME_GAUGE.with_label_values(&["pics", "ec_shards"]).set(3.0);
READ_ONLY_VOLUME_GAUGE
.with_label_values(&["pics", "volume"])
.set(1.0);
@@ -119,11 +119,7 @@ pub fn check_blocked_ip(endpoint: &str, ip: IpAddr) -> Result<(), String> {
/// reachable for callers whose target legitimately sits on an internal network
/// (peer volume servers), while still blocking loopback, link-local (IMDS) and
/// unspecified. Mirrors Go's `checkBlockedIPPolicy`.
pub fn check_blocked_ip_policy(
endpoint: &str,
ip: IpAddr,
allow_private: bool,
) -> Result<(), String> {
pub fn check_blocked_ip_policy(endpoint: &str, ip: IpAddr, allow_private: bool) -> Result<(), String> {
// Normalize IPv4-mapped IPv6 (`::ffff:a.b.c.d`) to its IPv4 form so the
// IPv4 deny rules apply. The OS routes these to the embedded IPv4 address,
// so without this `::ffff:127.0.0.1` / `::ffff:169.254.169.254` would slip
@@ -235,7 +231,7 @@ fn precheck_endpoint(endpoint: &str) -> Result<HostCheck, String> {
return Err(format!(
"remote endpoint {:?} has a malformed IPv6 host",
endpoint
));
))
}
}
} else {
@@ -311,10 +307,7 @@ pub async fn validate_replica_target(target: &str) -> Result<(), String> {
return Err("replica target is empty".to_string());
}
if trimmed.contains("://") || trimmed.contains(['/', '?', '#', '@', '\\']) {
return Err(format!(
"replica target {:?} must be a bare host:port",
target
));
return Err(format!("replica target {:?} must be a bare host:port", target));
}
// Require an explicit host:port, handling `[IPv6]:port`. A bracketless IPv6
@@ -323,22 +316,12 @@ pub async fn validate_replica_target(target: &str) -> Result<(), String> {
let host = if let Some(rest) = trimmed.strip_prefix('[') {
match rest.split_once(']') {
Some((h, port)) if port.starts_with(':') && port.len() > 1 => h,
_ => {
return Err(format!(
"replica target {:?} must be a bare host:port",
target
));
}
_ => return Err(format!("replica target {:?} must be a bare host:port", target)),
}
} else {
match trimmed.rsplit_once(':') {
Some((h, port)) if !port.is_empty() && !h.contains(':') => h,
_ => {
return Err(format!(
"replica target {:?} must be a bare host:port",
target
));
}
_ => return Err(format!("replica target {:?} must be a bare host:port", target)),
}
};
@@ -357,10 +340,7 @@ pub async fn validate_replica_target(target: &str) -> Result<(), String> {
let addrs = resolve_host(host).await?;
if addrs.is_empty() {
return Err(format!(
"resolve replica target host {:?}: no addresses",
host
));
return Err(format!("resolve replica target host {:?}: no addresses", host));
}
for ip in addrs {
check_blocked_ip_policy(target, ip, true)?;
@@ -368,56 +348,6 @@ pub async fn validate_replica_target(target: &str) -> Result<(), String> {
Ok(())
}
/// Resolve `host`, re-apply the replica deny list (private peers allowed) to
/// every resolved address, and connect to the first one that passes -- the
/// connect-time twin of [`validate_replica_target`], so a hostname whose DNS
/// answer flips to a blocked address after the up-front check is still refused.
/// Mirrors Go's `guardedDialerPolicy` with allowPrivate=true.
pub async fn guarded_tcp_connect(
host: &str,
port: u16,
endpoint: &str,
) -> std::io::Result<tokio::net::TcpStream> {
use std::io::{Error, ErrorKind};
let denied = |e: String| Error::new(ErrorKind::PermissionDenied, e);
if is_blocked_imds_host(&host.to_ascii_lowercase()) {
return Err(denied(format!(
"remote endpoint {:?} targets instance metadata service",
endpoint
)));
}
if let Ok(ip) = host.parse::<IpAddr>() {
check_blocked_ip_policy(endpoint, ip, true).map_err(denied)?;
return tokio::net::TcpStream::connect((ip, port)).await;
}
let lookup = tokio::net::lookup_host((host.to_string(), port));
let addrs = tokio::time::timeout(std::time::Duration::from_secs(2), lookup)
.await
.map_err(|_| {
Error::new(
ErrorKind::TimedOut,
format!("resolve remote endpoint host {:?}: timed out", host),
)
})??;
let mut first_block_err: Option<String> = None;
for addr in addrs {
if let Err(e) = check_blocked_ip_policy(endpoint, addr.ip(), true) {
if first_block_err.is_none() {
first_block_err = Some(e);
}
continue;
}
return tokio::net::TcpStream::connect(addr).await;
}
Err(denied(first_block_err.unwrap_or_else(|| {
format!("resolve remote endpoint host {:?}: no addresses", host)
})))
}
#[cfg(test)]
mod tests {
use super::*;
@@ -448,30 +378,22 @@ mod tests {
#[test]
fn rejects_empty_and_bad_scheme() {
assert!(precheck_endpoint("").unwrap_err().contains("empty"));
assert!(
precheck_endpoint("ftp://example.com/")
.unwrap_err()
.contains("http or https")
);
assert!(
precheck_endpoint("example.com/")
.unwrap_err()
.contains("http or https")
);
assert!(precheck_endpoint("ftp://example.com/")
.unwrap_err()
.contains("http or https"));
assert!(precheck_endpoint("example.com/")
.unwrap_err()
.contains("http or https"));
}
#[test]
fn rejects_imds_hostnames() {
assert!(
precheck_endpoint("http://metadata.google.internal/")
.unwrap_err()
.contains("metadata service")
);
assert!(
precheck_endpoint("http://metadata/")
.unwrap_err()
.contains("metadata service")
);
assert!(precheck_endpoint("http://metadata.google.internal/")
.unwrap_err()
.contains("metadata service"));
assert!(precheck_endpoint("http://metadata/")
.unwrap_err()
.contains("metadata service"));
}
#[test]
@@ -491,41 +413,27 @@ mod tests {
#[test]
fn check_blocked_ip_matches_resolved_categories() {
// Mirror Go's "host resolves to X" cases at the address level.
assert!(
check_blocked_ip("e", ip("127.0.0.1"))
.unwrap_err()
.contains("loopback")
);
assert!(
check_blocked_ip("e", ip("169.254.10.20"))
.unwrap_err()
.contains("link-local")
);
assert!(
check_blocked_ip("e", ip("10.1.2.3"))
.unwrap_err()
.contains("private")
);
assert!(
check_blocked_ip("e", ip("172.20.0.5"))
.unwrap_err()
.contains("private")
);
assert!(
check_blocked_ip("e", ip("192.168.1.1"))
.unwrap_err()
.contains("private")
);
assert!(
check_blocked_ip("e", ip("100.64.0.42"))
.unwrap_err()
.contains("CGNAT")
);
assert!(
check_blocked_ip("e", ip("fc00::1"))
.unwrap_err()
.contains("private")
);
assert!(check_blocked_ip("e", ip("127.0.0.1"))
.unwrap_err()
.contains("loopback"));
assert!(check_blocked_ip("e", ip("169.254.10.20"))
.unwrap_err()
.contains("link-local"));
assert!(check_blocked_ip("e", ip("10.1.2.3"))
.unwrap_err()
.contains("private"));
assert!(check_blocked_ip("e", ip("172.20.0.5"))
.unwrap_err()
.contains("private"));
assert!(check_blocked_ip("e", ip("192.168.1.1"))
.unwrap_err()
.contains("private"));
assert!(check_blocked_ip("e", ip("100.64.0.42"))
.unwrap_err()
.contains("CGNAT"));
assert!(check_blocked_ip("e", ip("fc00::1"))
.unwrap_err()
.contains("private"));
assert!(check_blocked_ip("e", ip("52.216.10.10")).is_ok());
assert!(check_blocked_ip("e", ip("2606:4700:4700::1111")).is_ok());
}
@@ -568,45 +476,33 @@ mod tests {
assert!(check_blocked_ip("e", ip("2001::f7f7:f7f7")).is_ok());
assert!(check_blocked_ip("e", ip("::808:808")).is_ok());
// Bracketed transition literal via the full endpoint path.
assert!(
precheck_endpoint("http://[64:ff9b::a9fe:a9fe]/")
.unwrap_err()
.contains("metadata")
);
assert!(precheck_endpoint("http://[64:ff9b::a9fe:a9fe]/")
.unwrap_err()
.contains("metadata"));
}
#[test]
fn rejects_ipv4_mapped_ipv6() {
// IPv4-mapped IPv6 must be unmapped so the IPv4 rules catch it.
assert!(
check_blocked_ip("e", ip("::ffff:127.0.0.1"))
.unwrap_err()
.contains("loopback")
);
assert!(
check_blocked_ip("e", ip("::ffff:169.254.169.254"))
.unwrap_err()
.contains("metadata")
);
assert!(
check_blocked_ip("e", ip("::ffff:10.0.0.1"))
.unwrap_err()
.contains("private")
);
assert!(check_blocked_ip("e", ip("::ffff:127.0.0.1"))
.unwrap_err()
.contains("loopback"));
assert!(check_blocked_ip("e", ip("::ffff:169.254.169.254"))
.unwrap_err()
.contains("metadata"));
assert!(check_blocked_ip("e", ip("::ffff:10.0.0.1"))
.unwrap_err()
.contains("private"));
// A mapped public address still passes, and genuine IPv6 loopback is
// still caught by the V6 path.
assert!(check_blocked_ip("e", ip("::ffff:52.216.10.10")).is_ok());
assert!(
check_blocked_ip("e", ip("::1"))
.unwrap_err()
.contains("loopback")
);
assert!(check_blocked_ip("e", ip("::1"))
.unwrap_err()
.contains("loopback"));
// Bracketed mapped literal via the full endpoint path.
assert!(
precheck_endpoint("http://[::ffff:127.0.0.1]/")
.unwrap_err()
.contains("loopback")
);
assert!(precheck_endpoint("http://[::ffff:127.0.0.1]/")
.unwrap_err()
.contains("loopback"));
}
#[test]
@@ -616,86 +512,60 @@ mod tests {
assert!(check_blocked_ip_policy("e", ip("192.168.1.5"), true).is_ok());
assert!(check_blocked_ip_policy("e", ip("100.64.0.42"), true).is_ok());
// Loopback / IMDS / unspecified stay blocked even when private is allowed.
assert!(
check_blocked_ip_policy("e", ip("127.0.0.1"), true)
.unwrap_err()
.contains("loopback")
);
assert!(
check_blocked_ip_policy("e", ip("169.254.169.254"), true)
.unwrap_err()
.contains("metadata")
);
assert!(
check_blocked_ip_policy("e", ip("0.0.0.0"), true)
.unwrap_err()
.contains("unspecified")
);
assert!(check_blocked_ip_policy("e", ip("127.0.0.1"), true)
.unwrap_err()
.contains("loopback"));
assert!(check_blocked_ip_policy("e", ip("169.254.169.254"), true)
.unwrap_err()
.contains("metadata"));
assert!(check_blocked_ip_policy("e", ip("0.0.0.0"), true)
.unwrap_err()
.contains("unspecified"));
}
#[tokio::test]
async fn validate_replica_target_rejects_and_allows() {
// A path plus a trailing ?a= would otherwise swallow ?type=replicate.
assert!(
validate_replica_target("127.0.0.1:7000/status/x/?a=")
.await
.unwrap_err()
.contains("bare host:port")
);
assert!(
validate_replica_target("http://10.0.0.7:8080")
.await
.unwrap_err()
.contains("bare host:port")
);
assert!(
validate_replica_target("user@10.0.0.7:8080")
.await
.unwrap_err()
.contains("bare host:port")
);
assert!(
validate_replica_target("10.0.0.7")
.await
.unwrap_err()
.contains("bare host:port")
);
assert!(
validate_replica_target("peer.example.com")
.await
.unwrap_err()
.contains("bare host:port")
);
assert!(
validate_replica_target("127.0.0.1:8080")
.await
.unwrap_err()
.contains("loopback")
);
assert!(
validate_replica_target("[::1]:8080")
.await
.unwrap_err()
.contains("loopback")
);
assert!(
validate_replica_target("169.254.169.254:80")
.await
.unwrap_err()
.contains("metadata")
);
assert!(
validate_replica_target("metadata:80")
.await
.unwrap_err()
.contains("metadata")
);
assert!(
validate_replica_target("")
.await
.unwrap_err()
.contains("empty")
);
assert!(validate_replica_target("127.0.0.1:7000/status/x/?a=")
.await
.unwrap_err()
.contains("bare host:port"));
assert!(validate_replica_target("http://10.0.0.7:8080")
.await
.unwrap_err()
.contains("bare host:port"));
assert!(validate_replica_target("user@10.0.0.7:8080")
.await
.unwrap_err()
.contains("bare host:port"));
assert!(validate_replica_target("10.0.0.7")
.await
.unwrap_err()
.contains("bare host:port"));
assert!(validate_replica_target("peer.example.com")
.await
.unwrap_err()
.contains("bare host:port"));
assert!(validate_replica_target("127.0.0.1:8080")
.await
.unwrap_err()
.contains("loopback"));
assert!(validate_replica_target("[::1]:8080")
.await
.unwrap_err()
.contains("loopback"));
assert!(validate_replica_target("169.254.169.254:80")
.await
.unwrap_err()
.contains("metadata"));
assert!(validate_replica_target("metadata:80")
.await
.unwrap_err()
.contains("metadata"));
assert!(validate_replica_target("")
.await
.unwrap_err()
.contains("empty"));
// Legitimate peer volume servers on private networks pass.
assert!(validate_replica_target("10.0.0.7:8080").await.is_ok());
assert!(validate_replica_target("192.168.1.5:8080").await.is_ok());
+1 -1
View File
@@ -7,7 +7,7 @@ pub mod endpoint_guard;
pub mod s3;
pub mod s3_tier;
pub use endpoint_guard::{guarded_tcp_connect, validate_remote_endpoint, validate_replica_target};
pub use endpoint_guard::{validate_remote_endpoint, validate_replica_target};
use crate::pb::remote_pb::{RemoteConf, RemoteStorageLocation};
+22 -220
View File
@@ -2,10 +2,9 @@
//!
//! Works with AWS S3, MinIO, SeaweedFS S3, and all S3-compatible providers.
use aws_sdk_s3::Client;
use aws_sdk_s3::config::{BehaviorVersion, Credentials, Region};
use aws_sdk_s3::error::{DisplayErrorContext, SdkError};
use aws_sdk_s3::primitives::ByteStream;
use aws_sdk_s3::Client;
use super::{RemoteEntry, RemoteStorageClient, RemoteStorageError};
use crate::pb::remote_pb::{RemoteConf, RemoteStorageLocation};
@@ -26,23 +25,6 @@ impl S3RemoteStorageClient {
endpoint: &str,
force_path_style: bool,
) -> Self {
let client = Client::from_conf(
Self::config_builder(access_key, secret_key, region, endpoint, force_path_style)
.build(),
);
S3RemoteStorageClient { client, conf }
}
/// Build the SDK config for the given credentials and endpoint. Split out so
/// tests can attach a canned HTTP client before building the [`Client`].
fn config_builder(
access_key: &str,
secret_key: &str,
region: &str,
endpoint: &str,
force_path_style: bool,
) -> aws_sdk_s3::config::Builder {
let region = if region.is_empty() {
"us-east-1"
} else {
@@ -67,7 +49,9 @@ impl S3RemoteStorageClient {
s3_config = s3_config.endpoint_url(endpoint);
}
s3_config
let client = Client::from_conf(s3_config.build());
S3RemoteStorageClient { client, conf }
}
}
@@ -91,14 +75,13 @@ impl RemoteStorageClient for S3RemoteStorageClient {
req = req.range(format!("bytes={}-", offset));
}
let resp = req.send().await.map_err(|e| match e {
// Go compares `aerr.Code()` to NoSuchKey on GET
// (s3_storage_client.go:436): a bare 404 maps to "NotFound"
// and stays a generic error, as it does here.
SdkError::ServiceError(ref se) if se.err().is_no_such_key() => {
let resp = req.send().await.map_err(|e| {
let msg = format!("{}", e);
if msg.contains("NoSuchKey") || msg.contains("404") {
RemoteStorageError::ObjectNotFound(format!("{}/{}", loc.bucket, key))
} else {
RemoteStorageError::Other(format!("s3 get object: {}", e))
}
e => RemoteStorageError::Other(format!("s3 get object: {}", DisplayErrorContext(&e))),
})?;
let data = resp
@@ -125,9 +108,7 @@ impl RemoteStorageClient for S3RemoteStorageClient {
.body(ByteStream::from(data.to_vec()))
.send()
.await
.map_err(|e| {
RemoteStorageError::Other(format!("s3 put object: {}", DisplayErrorContext(&e)))
})?;
.map_err(|e| RemoteStorageError::Other(format!("s3 put object: {}", e)))?;
Ok(RemoteEntry {
size: data.len() as i64,
@@ -153,18 +134,13 @@ impl RemoteStorageClient for S3RemoteStorageClient {
.key(key)
.send()
.await
.map_err(|e| match e {
// Go checks only the raw HTTP status on HEAD
// (s3_storage_client.go:373): a HEAD response carries no
// error body, so a 404 is not-found whatever code the SDK
// assigns, and a non-404 is not.
SdkError::ServiceError(ref se) if se.raw().status().as_u16() == 404 => {
.map_err(|e| {
let msg = format!("{}", e);
if msg.contains("404") || msg.contains("NotFound") {
RemoteStorageError::ObjectNotFound(format!("{}/{}", loc.bucket, key))
} else {
RemoteStorageError::Other(format!("s3 head object: {}", e))
}
e => RemoteStorageError::Other(format!(
"s3 head object: {}",
DisplayErrorContext(&e)
)),
})?;
Ok(RemoteEntry {
@@ -184,17 +160,18 @@ impl RemoteStorageClient for S3RemoteStorageClient {
.key(key)
.send()
.await
.map_err(|e| {
RemoteStorageError::Other(format!("s3 delete object: {}", DisplayErrorContext(&e)))
})?;
.map_err(|e| RemoteStorageError::Other(format!("s3 delete object: {}", e)))?;
Ok(())
}
async fn list_buckets(&self) -> Result<Vec<String>, RemoteStorageError> {
let resp = self.client.list_buckets().send().await.map_err(|e| {
RemoteStorageError::Other(format!("s3 list buckets: {}", DisplayErrorContext(&e)))
})?;
let resp = self
.client
.list_buckets()
.send()
.await
.map_err(|e| RemoteStorageError::Other(format!("s3 list buckets: {}", e)))?;
Ok(resp
.buckets()
@@ -207,178 +184,3 @@ impl RemoteStorageClient for S3RemoteStorageClient {
&self.conf
}
}
#[cfg(test)]
pub(crate) mod tests {
use super::*;
use aws_sdk_s3::config::http::{HttpRequest, HttpResponse};
use aws_sdk_s3::config::retry::RetryConfig;
use aws_sdk_s3::config::{HttpClient, RuntimeComponents};
use aws_sdk_s3::primitives::SdkBody;
use aws_smithy_runtime_api::client::http::{
HttpConnector, HttpConnectorFuture, HttpConnectorSettings, SharedHttpConnector,
};
use aws_smithy_runtime_api::http::StatusCode;
/// An SDK HTTP client that answers every request with one canned response,
/// so the error-mapping paths can be exercised without a network or a
/// running S3 server.
#[derive(Debug, Clone)]
pub(crate) struct CannedResponse {
pub(crate) status: u16,
pub(crate) body: &'static str,
}
impl HttpConnector for CannedResponse {
fn call(&self, _request: HttpRequest) -> HttpConnectorFuture {
let status = StatusCode::try_from(self.status).expect("valid HTTP status");
HttpConnectorFuture::ready(Ok(HttpResponse::new(status, SdkBody::from(self.body))))
}
}
impl HttpClient for CannedResponse {
fn http_connector(
&self,
_settings: &HttpConnectorSettings,
_components: &RuntimeComponents,
) -> SharedHttpConnector {
SharedHttpConnector::new(self.clone())
}
}
fn client_with(status: u16, body: &'static str) -> S3RemoteStorageClient {
let config = S3RemoteStorageClient::config_builder(
"AKIATEST",
"secret",
"us-east-1",
"http://127.0.0.1:1",
true,
)
.http_client(CannedResponse { status, body })
.retry_config(RetryConfig::disabled())
.build();
S3RemoteStorageClient {
client: Client::from_conf(config),
conf: RemoteConf::default(),
}
}
fn location() -> RemoteStorageLocation {
RemoteStorageLocation {
name: "remote".to_string(),
bucket: "bucket".to_string(),
path: "/dir/missing".to_string(),
..Default::default()
}
}
pub(crate) const NO_SUCH_KEY: &str = r#"<?xml version="1.0" encoding="UTF-8"?>
<Error><Code>NoSuchKey</Code><Message>The specified key does not exist.</Message><Key>dir/missing</Key></Error>"#;
const NOT_FOUND_BODY: &str = r#"<?xml version="1.0" encoding="UTF-8"?>
<Error><Code>NotFound</Code><Message>Not Found</Message></Error>"#;
const ACCESS_DENIED: &str = r#"<?xml version="1.0" encoding="UTF-8"?>
<Error><Code>AccessDenied</Code><Message>Access Denied</Message></Error>"#;
#[tokio::test]
async fn get_no_such_key_is_object_not_found() {
let err = client_with(404, NO_SUCH_KEY)
.read_file(&location(), 0, 0)
.await
.unwrap_err();
assert!(
matches!(&err, RemoteStorageError::ObjectNotFound(path) if path == "bucket/dir/missing"),
"expected ObjectNotFound, got {err:?}"
);
}
#[tokio::test]
async fn get_bare_404_is_not_object_not_found() {
// Go compares codes, not statuses, on GET: a body-less 404 stays generic.
let err = client_with(404, "")
.read_file(&location(), 0, 0)
.await
.unwrap_err();
assert!(
matches!(err, RemoteStorageError::Other(_)),
"expected Other, got {err:?}"
);
}
#[tokio::test]
async fn head_404_is_object_not_found() {
let err = client_with(404, "")
.stat_file(&location())
.await
.unwrap_err();
assert!(
matches!(&err, RemoteStorageError::ObjectNotFound(path) if path == "bucket/dir/missing"),
"expected ObjectNotFound, got {err:?}"
);
}
#[tokio::test]
async fn head_404_with_foreign_error_body_is_object_not_found() {
// The raw status check makes a 404 not-found whatever body it carries.
let err = client_with(404, NO_SUCH_KEY)
.stat_file(&location())
.await
.unwrap_err();
assert!(
matches!(err, RemoteStorageError::ObjectNotFound(_)),
"expected ObjectNotFound, got {err:?}"
);
}
#[tokio::test]
async fn head_not_found_code_on_a_non_404_status_is_not_object_not_found() {
// A NotFound body on a non-404 status stays an error, as in Go.
let err = client_with(400, NOT_FOUND_BODY)
.stat_file(&location())
.await
.unwrap_err();
assert!(
matches!(&err, RemoteStorageError::Other(msg) if msg.contains("NotFound")),
"expected Other naming the code, got {err:?}"
);
}
#[tokio::test]
async fn get_access_denied_keeps_service_error_code() {
let err = client_with(403, ACCESS_DENIED)
.read_file(&location(), 0, 0)
.await
.unwrap_err();
let msg = err.to_string();
assert!(
matches!(err, RemoteStorageError::Other(_)),
"expected Other, got {err:?}"
);
assert!(
msg.contains("AccessDenied"),
"message should carry the S3 error code, got: {msg}"
);
assert!(
!msg.ends_with("service error"),
"message should not be the bare SdkError Display, got: {msg}"
);
}
#[tokio::test]
async fn head_access_denied_keeps_service_error_code() {
let err = client_with(403, ACCESS_DENIED)
.stat_file(&location())
.await
.unwrap_err();
let msg = err.to_string();
assert!(
matches!(err, RemoteStorageError::Other(_)),
"expected Other, got {err:?}"
);
assert!(
msg.contains("AccessDenied"),
"message should carry the S3 error code, got: {msg}"
);
}
}
+57 -408
View File
@@ -7,66 +7,15 @@ use std::collections::HashMap;
use std::future::Future;
use std::sync::{Arc, OnceLock, RwLock};
use aws_sdk_s3::Client;
use aws_sdk_s3::config::http::HttpResponse;
use aws_sdk_s3::config::{BehaviorVersion, Credentials, Region};
use aws_sdk_s3::error::{DisplayErrorContext, SdkError};
use aws_sdk_s3::operation::get_object::GetObjectError;
use aws_sdk_s3::operation::head_object::HeadObjectError;
use aws_sdk_s3::types::{CompletedMultipartUpload, CompletedPart};
use aws_sdk_s3::Client;
use tokio::io::{AsyncReadExt, AsyncSeekExt, AsyncWriteExt};
use tokio::sync::Semaphore;
/// Concurrency limit for multipart upload/download (matches Go's s3manager).
const CONCURRENCY: usize = 5;
/// A tier transfer failure. The variant is what callers match on; the
/// message is the operator-facing text.
#[derive(Debug, thiserror::Error)]
pub enum TierError {
/// The remote object does not exist.
#[error("{0}")]
NotFound(String),
/// An S3 request or a local file operation failed.
#[error("{0}")]
Io(String),
/// The tier I/O runtime could not be built or dropped the task.
#[error("{0}")]
RuntimeUnavailable(String),
/// The progress callback asked to stop.
#[error("{0}")]
Aborted(String),
}
// Not-found rules as in remote_storage/s3.rs: HEAD by the raw 404 status,
// GET by the NoSuchKey code only.
fn head_object_error(key: &str, e: SdkError<HeadObjectError, HttpResponse>) -> TierError {
let message = format!("failed to head object {}: {}", key, DisplayErrorContext(&e));
match e {
SdkError::ServiceError(ref se) if se.raw().status().as_u16() == 404 => {
TierError::NotFound(message)
}
_ => TierError::Io(message),
}
}
fn get_object_error(
key: &str,
range: &str,
e: SdkError<GetObjectError, HttpResponse>,
) -> TierError {
let message = format!(
"failed to get object {} range {}: {}",
key,
range,
DisplayErrorContext(&e)
);
match e {
SdkError::ServiceError(ref se) if se.err().is_no_such_key() => TierError::NotFound(message),
_ => TierError::Io(message),
}
}
/// Configuration for an S3 tier backend.
#[derive(Debug, Clone)]
pub struct S3TierConfig {
@@ -140,7 +89,7 @@ impl S3TierBackend {
&self,
file_path: &str,
progress_fn: F,
) -> Result<(String, u64), TierError>
) -> Result<(String, u64), String>
where
F: FnMut(i64, f32) -> Result<(), String> + Send + Sync + 'static,
{
@@ -148,7 +97,7 @@ impl S3TierBackend {
let metadata = tokio::fs::metadata(file_path)
.await
.map_err(|e| TierError::Io(format!("failed to stat file {}: {}", file_path, e)))?;
.map_err(|e| format!("failed to stat file {}: {}", file_path, e))?;
let file_size = metadata.len();
// Calculate part size: start at 64MB, scale up for very large files (matches Go)
@@ -170,16 +119,11 @@ impl S3TierBackend {
)
.send()
.await
.map_err(|e| {
TierError::Io(format!(
"failed to create multipart upload: {}",
DisplayErrorContext(&e)
))
})?;
.map_err(|e| format!("failed to create multipart upload: {}", e))?;
let upload_id = create_resp
.upload_id()
.ok_or_else(|| TierError::Io("no upload_id in multipart upload response".to_string()))?
.ok_or_else(|| "no upload_id in multipart upload response".to_string())?
.to_string();
// Build list of (part_number, offset, size) for all parts
@@ -215,21 +159,19 @@ impl S3TierBackend {
let _permit = sem
.acquire()
.await
.map_err(|e| TierError::Io(format!("semaphore error: {}", e)))?;
.map_err(|e| format!("semaphore error: {}", e))?;
// Read this part's data from the file at the correct offset
let mut file = tokio::fs::File::open(&fp)
.await
.map_err(|e| TierError::Io(format!("failed to open file {}: {}", fp, e)))?;
.map_err(|e| format!("failed to open file {}: {}", fp, e))?;
file.seek(std::io::SeekFrom::Start(off))
.await
.map_err(|e| {
TierError::Io(format!("failed to seek to offset {}: {}", off, e))
})?;
.map_err(|e| format!("failed to seek to offset {}: {}", off, e))?;
let mut buf = vec![0u8; size];
file.read_exact(&mut buf).await.map_err(|e| {
TierError::Io(format!("failed to read file at offset {}: {}", off, e))
})?;
file.read_exact(&mut buf)
.await
.map_err(|e| format!("failed to read file at offset {}: {}", off, e))?;
let upload_part_resp = client
.upload_part()
@@ -241,12 +183,7 @@ impl S3TierBackend {
.send()
.await
.map_err(|e| {
TierError::Io(format!(
"failed to upload part {} at offset {}: {}",
pn,
off,
DisplayErrorContext(&e)
))
format!("failed to upload part {} at offset {}: {}", pn, off, e)
})?;
let e_tag = upload_part_resp.e_tag().unwrap_or_default().to_string();
@@ -265,9 +202,9 @@ impl S3TierBackend {
};
(guard.1)(uploaded as i64, pct)
};
progress_result.map_err(TierError::Aborted)?;
progress_result?;
Ok::<_, TierError>(
Ok::<_, String>(
CompletedPart::builder()
.e_tag(e_tag)
.part_number(pn)
@@ -282,7 +219,7 @@ impl S3TierBackend {
for handle in handles {
let part = handle
.await
.map_err(|e| TierError::Io(format!("upload task panicked: {}", e)))??;
.map_err(|e| format!("upload task panicked: {}", e))??;
completed_parts.push(part);
}
@@ -299,14 +236,9 @@ impl S3TierBackend {
.multipart_upload(completed_upload)
.send()
.await
.map_err(|e| {
TierError::Io(format!(
"failed to complete multipart upload: {}",
DisplayErrorContext(&e)
))
})?;
.map_err(|e| format!("failed to complete multipart upload: {}", e))?;
Ok::<(), TierError>(())
Ok::<(), String>(())
}
.await;
@@ -349,7 +281,7 @@ impl S3TierBackend {
dest_path: &str,
key: &str,
progress_fn: F,
) -> Result<u64, TierError>
) -> Result<u64, String>
where
F: FnMut(i64, f32) -> Result<(), String> + Send + Sync + 'static,
{
@@ -361,7 +293,7 @@ impl S3TierBackend {
.key(key)
.send()
.await
.map_err(|e| head_object_error(key, e))?;
.map_err(|e| format!("failed to head object {}: {}", key, e))?;
let file_size = head_resp.content_length().unwrap_or(0) as u64;
@@ -373,12 +305,10 @@ impl S3TierBackend {
.truncate(true)
.open(dest_path)
.await
.map_err(|e| {
TierError::Io(format!("failed to open dest file {}: {}", dest_path, e))
})?;
.map_err(|e| format!("failed to open dest file {}: {}", dest_path, e))?;
file.set_len(file_size)
.await
.map_err(|e| TierError::Io(format!("failed to set file length: {}", e)))?;
.map_err(|e| format!("failed to set file length: {}", e))?;
}
let part_size: u64 = 64 * 1024 * 1024;
@@ -414,7 +344,7 @@ impl S3TierBackend {
let _permit = sem
.acquire()
.await
.map_err(|e| TierError::Io(format!("semaphore error: {}", e)))?;
.map_err(|e| format!("semaphore error: {}", e))?;
let end = off + size - 1;
let range = format!("bytes={}-{}", off, end);
@@ -426,13 +356,13 @@ impl S3TierBackend {
.range(&range)
.send()
.await
.map_err(|e| get_object_error(&key, &range, e))?;
.map_err(|e| format!("failed to get object {} range {}: {}", key, range, e))?;
let body = get_resp
.body
.collect()
.await
.map_err(|e| TierError::Io(format!("failed to read body: {}", e)))?;
.map_err(|e| format!("failed to read body: {}", e))?;
let bytes = body.into_bytes();
// Write at the correct offset (like Go's WriteAt)
@@ -440,17 +370,13 @@ impl S3TierBackend {
.write(true)
.open(&dp)
.await
.map_err(|e| {
TierError::Io(format!("failed to open dest file {}: {}", dp, e))
})?;
.map_err(|e| format!("failed to open dest file {}: {}", dp, e))?;
file.seek(std::io::SeekFrom::Start(off))
.await
.map_err(|e| {
TierError::Io(format!("failed to seek to offset {}: {}", off, e))
})?;
.map_err(|e| format!("failed to seek to offset {}: {}", off, e))?;
file.write_all(&bytes)
.await
.map_err(|e| TierError::Io(format!("failed to write to {}: {}", dp, e)))?;
.map_err(|e| format!("failed to write to {}: {}", dp, e))?;
// Report progress. The lock is released before the result is
// propagated so an aborting callback cannot poison the mutex
@@ -466,9 +392,9 @@ impl S3TierBackend {
};
(guard.1)(downloaded as i64, pct)
};
progress_result.map_err(TierError::Aborted)?;
progress_result?;
Ok::<_, TierError>(())
Ok::<_, String>(())
}));
}
@@ -476,7 +402,7 @@ impl S3TierBackend {
for handle in handles {
handle
.await
.map_err(|e| TierError::Io(format!("download task panicked: {}", e)))??;
.map_err(|e| format!("download task panicked: {}", e))??;
}
// fsync the file so its content is durable before the caller trims the .vif
@@ -485,21 +411,16 @@ impl S3TierBackend {
.write(true)
.open(dest_path)
.await
.map_err(|e| TierError::Io(format!("failed to open {} for fsync: {}", dest_path, e)))?;
.map_err(|e| format!("failed to open {} for fsync: {}", dest_path, e))?;
synced
.sync_all()
.await
.map_err(|e| TierError::Io(format!("failed to fsync {}: {}", dest_path, e)))?;
.map_err(|e| format!("failed to fsync {}: {}", dest_path, e))?;
Ok(file_size)
}
pub async fn read_range(
&self,
key: &str,
offset: u64,
size: usize,
) -> Result<Vec<u8>, TierError> {
pub async fn read_range(&self, key: &str, offset: u64, size: usize) -> Result<Vec<u8>, String> {
let end = offset + (size as u64).saturating_sub(1);
let range = format!("bytes={}-{}", offset, end);
let resp = self
@@ -510,35 +431,29 @@ impl S3TierBackend {
.range(&range)
.send()
.await
.map_err(|e| get_object_error(key, &range, e))?;
.map_err(|e| format!("failed to get object {} range {}: {}", key, range, e))?;
let body = resp
.body
.collect()
.await
.map_err(|e| TierError::Io(format!("failed to read object {} body: {}", key, e)))?;
.map_err(|e| format!("failed to read object {} body: {}", key, e))?;
Ok(body.into_bytes().to_vec())
}
/// Delete a file from S3.
pub async fn delete_file(&self, key: &str) -> Result<(), TierError> {
pub async fn delete_file(&self, key: &str) -> Result<(), String> {
self.client
.delete_object()
.bucket(&self.bucket)
.key(key)
.send()
.await
.map_err(|e| {
TierError::Io(format!(
"failed to delete object {}: {}",
key,
DisplayErrorContext(&e)
))
})?;
.map_err(|e| format!("failed to delete object {}: {}", key, e))?;
Ok(())
}
pub fn delete_file_blocking(&self, key: &str) -> Result<(), TierError> {
pub fn delete_file_blocking(&self, key: &str) -> Result<(), String> {
let client = self.client.clone();
let bucket = self.bucket.clone();
let key = key.to_string();
@@ -549,13 +464,7 @@ impl S3TierBackend {
.key(&key)
.send()
.await
.map_err(|e| {
TierError::Io(format!(
"failed to delete object {}: {}",
key,
DisplayErrorContext(&e)
))
})?;
.map_err(|e| format!("failed to delete object {}: {}", key, e))?;
Ok(())
})
}
@@ -565,7 +474,7 @@ impl S3TierBackend {
key: &str,
offset: u64,
size: usize,
) -> Result<Vec<u8>, TierError> {
) -> Result<Vec<u8>, String> {
let client = self.client.clone();
let bucket = self.bucket.clone();
let key = key.to_string();
@@ -579,12 +488,13 @@ impl S3TierBackend {
.range(&range)
.send()
.await
.map_err(|e| get_object_error(&key, &range, e))?;
.map_err(|e| format!("failed to get object {} range {}: {}", key, range, e))?;
let body =
resp.body.collect().await.map_err(|e| {
TierError::Io(format!("failed to read object {} body: {}", key, e))
})?;
let body = resp
.body
.collect()
.await
.map_err(|e| format!("failed to read object {} body: {}", key, e))?;
Ok(body.into_bytes().to_vec())
})
}
@@ -645,279 +555,18 @@ pub fn global_s3_tier_registry() -> &'static RwLock<S3TierRegistry> {
GLOBAL_S3_TIER_REGISTRY.get_or_init(|| RwLock::new(S3TierRegistry::new()))
}
/// The one process-wide runtime for tiered-S3 I/O issued from synchronous
/// storage code. A per-call runtime tore down the SDK's pooled connections
/// after every 64 KiB chunk, re-dialing TLS per read; a long-lived runtime
/// keeps the pool warm.
///
/// Built on first use. A build failure is returned, not cached or panicked:
/// callers sit inside `Volume::destroy` and needle reads, whose own error
/// paths must run, and a later call may succeed.
static TIER_RUNTIME: std::sync::Mutex<Option<tokio::runtime::Runtime>> =
std::sync::Mutex::new(None);
fn tier_handle() -> Result<tokio::runtime::Handle, TierError> {
let mut slot = TIER_RUNTIME
.lock()
.unwrap_or_else(|poisoned| poisoned.into_inner());
if slot.is_none() {
let runtime = tokio::runtime::Builder::new_multi_thread()
.worker_threads(2)
.thread_name("tier-io")
.enable_all()
.build()
.map_err(|e| {
TierError::RuntimeUnavailable(format!(
"failed to build the tier I/O tokio runtime: {}",
e
))
})?;
*slot = Some(runtime);
}
Ok(slot.as_ref().expect("just initialised").handle().clone())
}
/// Run `future` on the tier runtime and block the calling thread until it
/// finishes. The caller may be a worker of *another* tokio runtime, so this
/// waits on a channel rather than `Handle::block_on`, which panics when
/// called from inside any runtime context.
fn block_on_tier_future<F, T>(future: F) -> Result<T, TierError>
fn block_on_tier_future<F, T>(future: F) -> Result<T, String>
where
F: Future<Output = Result<T, TierError>> + Send + 'static,
F: Future<Output = Result<T, String>> + Send + 'static,
T: Send + 'static,
{
let handle = tier_handle()?;
let task = handle.spawn(future);
let (tx, rx) = std::sync::mpsc::sync_channel(1);
handle.spawn(async move {
// The receiver only goes away if the caller was unwound; nothing to
// report then.
let _ = tx.send(task.await);
});
match rx.recv() {
Ok(Ok(result)) => result,
Ok(Err(join_error)) => Err(describe_join_error(join_error)),
Err(_) => Err(TierError::RuntimeUnavailable(
"tier I/O runtime dropped the task before it finished".to_string(),
)),
}
}
/// Turn a `JoinError` into a message that keeps the panic payload, so an
/// SDK panic surfaces as "boom" rather than a fixed "thread panicked".
fn describe_join_error(join_error: tokio::task::JoinError) -> TierError {
if join_error.is_panic() {
let payload = join_error.into_panic();
let message = if let Some(s) = payload.downcast_ref::<&str>() {
(*s).to_string()
} else if let Some(s) = payload.downcast_ref::<String>() {
s.clone()
} else {
"non-string panic payload".to_string()
};
TierError::Io(format!("tier I/O task panicked: {}", message))
} else {
TierError::RuntimeUnavailable(format!("tier I/O task failed: {}", join_error))
}
}
#[cfg(test)]
mod tests {
use super::*;
use crate::remote_storage::s3::tests::{CannedResponse, NO_SUCH_KEY};
use std::collections::HashSet;
use tokio::runtime::Handle;
fn probe() -> Result<(tokio::runtime::Id, Option<String>), TierError> {
block_on_tier_future(async {
Ok((
Handle::current().id(),
std::thread::current().name().map(str::to_string),
))
})
}
#[test]
fn block_on_tier_future_reuses_one_runtime() {
let (first_runtime, first_thread) = probe().expect("first call");
let (second_runtime, second_thread) = probe().expect("second call");
assert_eq!(
first_runtime, second_runtime,
"each call must run on the same long-lived tier runtime"
);
assert_eq!(first_thread.as_deref(), Some("tier-io"));
assert_eq!(second_thread.as_deref(), Some("tier-io"));
let mut runtimes = HashSet::new();
for _ in 0..20 {
let (id, _) = probe().expect("probe");
runtimes.insert(id);
}
assert_eq!(runtimes.len(), 1);
}
#[test]
fn block_on_tier_future_returns_the_value_and_the_error() {
assert_eq!(block_on_tier_future(async { Ok(7u32) }).unwrap(), 7);
let err = block_on_tier_future::<_, u32>(async { Err(TierError::NotFound("nope".into())) })
.unwrap_err();
assert!(
matches!(&err, TierError::NotFound(m) if m == "nope"),
"{err:?}"
);
}
#[test]
fn block_on_tier_future_works_from_a_std_thread() {
let (id, _) = std::thread::spawn(probe)
.join()
.expect("probe thread")
.expect("probe");
assert_eq!(id, tier_handle().expect("tier runtime").id());
}
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
async fn block_on_tier_future_works_from_spawn_blocking() {
let (id, _) = tokio::task::spawn_blocking(probe)
.await
.expect("spawn_blocking")
.expect("probe");
assert_eq!(id, tier_handle().expect("tier runtime").id());
assert_ne!(id, Handle::current().id());
}
// Called straight from another runtime's async context: the case that
// would panic with `Handle::block_on` ("Cannot start a runtime from
// within a runtime").
#[tokio::test]
async fn block_on_tier_future_works_from_a_current_thread_runtime() {
let (id, _) = probe().expect("probe");
assert_eq!(id, tier_handle().expect("tier runtime").id());
assert_ne!(id, Handle::current().id());
}
#[tokio::test(flavor = "multi_thread", worker_threads = 2)]
async fn block_on_tier_future_works_from_a_multi_thread_runtime_worker() {
let (id, _) = probe().expect("probe");
assert_eq!(id, tier_handle().expect("tier runtime").id());
assert_ne!(id, Handle::current().id());
}
#[test]
fn block_on_tier_future_reports_the_panic_payload() {
let err = block_on_tier_future::<_, ()>(async {
if std::hint::black_box(true) {
panic!("boom {}", 42);
}
Ok(())
})
.expect_err("a panicking future must be an error");
assert!(matches!(err, TierError::Io(_)), "got: {err:?}");
assert!(err.to_string().contains("boom 42"), "got: {err}");
assert!(err.to_string().contains("panicked"), "got: {err}");
}
#[test]
fn block_on_tier_future_reports_a_str_panic_payload() {
let err = block_on_tier_future::<_, ()>(async {
if std::hint::black_box(true) {
panic!("static boom");
}
Ok(())
})
.expect_err("a panicking future must be an error");
assert!(err.to_string().contains("static boom"), "got: {err}");
}
fn backend_answering(status: u16, body: &'static str) -> S3TierBackend {
let config = aws_sdk_s3::Config::builder()
.behavior_version(BehaviorVersion::latest())
.region(Region::new("us-east-1"))
.credentials_provider(Credentials::new("AKIATEST", "secret", None, None, "test"))
.endpoint_url("http://127.0.0.1:1")
.force_path_style(true)
.http_client(CannedResponse { status, body })
.retry_config(aws_sdk_s3::config::retry::RetryConfig::disabled())
.build();
S3TierBackend {
client: Client::from_conf(config),
bucket: "bucket".to_string(),
storage_class: "STANDARD".to_string(),
}
}
#[tokio::test]
async fn download_head_404_is_not_found() {
let tmp = tempfile::tempdir().unwrap();
let dest = tmp.path().join("1.dat");
let err = backend_answering(404, "")
.download_file(dest.to_str().unwrap(), "missing", |_, _| Ok(()))
.await
.unwrap_err();
assert!(matches!(err, TierError::NotFound(_)), "{err:?}");
assert!(
err.to_string()
.starts_with("failed to head object missing: "),
"{err}"
);
}
#[tokio::test]
async fn download_head_403_is_io() {
let tmp = tempfile::tempdir().unwrap();
let dest = tmp.path().join("1.dat");
let err = backend_answering(403, "")
.download_file(dest.to_str().unwrap(), "denied", |_, _| Ok(()))
.await
.unwrap_err();
assert!(matches!(err, TierError::Io(_)), "{err:?}");
}
#[tokio::test]
async fn read_range_no_such_key_is_not_found() {
let err = backend_answering(404, NO_SUCH_KEY)
.read_range("missing", 0, 8)
.await
.unwrap_err();
assert!(matches!(err, TierError::NotFound(_)), "{err:?}");
assert!(
err.to_string()
.starts_with("failed to get object missing range bytes=0-7: "),
"{err}"
);
}
#[tokio::test]
async fn read_range_bare_404_is_io() {
// As in Go, GET is not-found by the NoSuchKey code, not the status.
let err = backend_answering(404, "")
.read_range("missing", 0, 8)
.await
.unwrap_err();
assert!(matches!(err, TierError::Io(_)), "{err:?}");
}
#[test]
fn read_range_blocking_no_such_key_is_not_found() {
let err = backend_answering(404, NO_SUCH_KEY)
.read_range_blocking("missing", 0, 8)
.unwrap_err();
assert!(matches!(err, TierError::NotFound(_)), "{err:?}");
}
#[test]
fn backend_name_to_type_id_splits_on_dot() {
assert_eq!(
backend_name_to_type_id("s3"),
("s3".to_string(), "default".to_string())
);
assert_eq!(
backend_name_to_type_id("s3.eu"),
("s3".to_string(), "eu".to_string())
);
assert_eq!(
backend_name_to_type_id("s3.a.b"),
(String::new(), String::new())
);
}
std::thread::spawn(move || {
let runtime = tokio::runtime::Builder::new_current_thread()
.enable_all()
.build()
.map_err(|e| format!("failed to build tokio runtime: {}", e))?;
runtime.block_on(future)
})
.join()
.map_err(|_| "tier runtime thread panicked".to_string())?
}
+4 -6
View File
@@ -10,7 +10,7 @@ use std::collections::HashSet;
use std::net::IpAddr;
use std::time::{SystemTime, UNIX_EPOCH};
use jsonwebtoken::{Algorithm, DecodingKey, EncodingKey, Header, Validation, decode, encode};
use jsonwebtoken::{decode, encode, Algorithm, DecodingKey, EncodingKey, Header, Validation};
use serde::{Deserialize, Serialize};
// ============================================================================
@@ -481,11 +481,9 @@ mod tests {
let token = gen_jwt(&key, 3600, "3,01637037d6").unwrap();
// Correct file ID
assert!(
guard
.check_jwt_for_file(Some(&token), "3,01637037d6", true)
.is_ok()
);
assert!(guard
.check_jwt_for_file(Some(&token), "3,01637037d6", true)
.is_ok());
// Wrong file ID
let err = guard.check_jwt_for_file(Some(&token), "4,deadbeef", true);
+7 -8
View File
@@ -3,12 +3,12 @@ use std::fmt;
use std::sync::Arc;
use rustls::client::danger::HandshakeSignatureValid;
use rustls::crypto::CryptoProvider;
use rustls::crypto::aws_lc_rs;
use rustls::crypto::CryptoProvider;
use rustls::pki_types::UnixTime;
use rustls::pki_types::{CertificateDer, PrivateKeyDer};
use rustls::server::WebPkiClientVerifier;
use rustls::server::danger::{ClientCertVerified, ClientCertVerifier};
use rustls::server::WebPkiClientVerifier;
use rustls::{
CipherSuite, DigitallySignedStruct, DistinguishedName, RootCertStore, ServerConfig,
SignatureScheme, SupportedCipherSuite, SupportedProtocolVersion,
@@ -120,11 +120,10 @@ impl ClientCertVerifier for CommonNameVerifier {
// aws-lc-rs and ring both get linked transitively, so rustls can't auto-select
// a provider and tonic's client TLS panics on first use. Pin the default to
// aws-lc-rs, matching the server config. Idempotent. The body lives in
// seaweed-common so this binary and the Rust plugin workers cannot end up
// installing different providers; re-exported here so callers keep their
// import path.
pub use seaweed_common::tls::install_default_crypto_provider;
// aws-lc-rs, matching the server config. Idempotent.
pub fn install_default_crypto_provider() {
let _ = aws_lc_rs::default_provider().install_default();
}
pub fn build_rustls_server_config(
cert_path: &str,
@@ -377,7 +376,7 @@ fn go_tls_version_for_supported(version: &SupportedProtocolVersion) -> GoTlsVers
#[cfg(test)]
mod tests {
use super::{TlsPolicy, build_supported_versions, common_name_is_allowed, parse_cipher_suites};
use super::{build_supported_versions, common_name_is_allowed, parse_cipher_suites, TlsPolicy};
use rustls::crypto::aws_lc_rs;
use std::collections::HashSet;
+2 -2
View File
@@ -1,9 +1,9 @@
use axum::Router;
use axum::body::Body;
use axum::extract::Query;
use axum::http::{StatusCode, header};
use axum::http::{header, StatusCode};
use axum::response::{IntoResponse, Response};
use axum::routing::{any, get};
use axum::Router;
use pprof::protos::Message;
use serde::Deserialize;
+45 -352
View File
@@ -1,42 +1,16 @@
//! Construction of the volume server's *outgoing* gRPC clients: TLS material,
//! endpoint tuning, dial bounds, and the three client constructors every call
//! site goes through.
//!
//! The keepalive, window-size and message-size constants below are shared with
//! the *inbound* server built in `main.rs`, which imports them from here rather
//! than declaring its own. Changing one therefore changes both directions at
//! once, which is deliberate: a volume server talks to its peers with the same
//! HTTP/2 settings it offers them.
use std::error::Error;
use std::fmt;
use std::time::Duration;
use hyper::http::Uri;
use tonic::service::interceptor::InterceptedService;
use tonic::transport::{Certificate, Channel, ClientTlsConfig, Endpoint, Identity};
use tonic::{Request, Status};
use crate::config::VolumeServerConfig;
use crate::pb::filer_pb::seaweed_filer_client::SeaweedFilerClient;
use crate::pb::master_pb::seaweed_client::SeaweedClient;
use crate::pb::volume_server_pb::volume_server_client::VolumeServerClient;
use crate::server::request_id::outgoing_request_id_interceptor;
pub const GRPC_MAX_MESSAGE_SIZE: usize = 1 << 30;
pub const GRPC_KEEPALIVE_INTERVAL: Duration = Duration::from_secs(60);
pub const GRPC_KEEPALIVE_TIMEOUT: Duration = Duration::from_secs(20);
pub const GRPC_INITIAL_WINDOW_SIZE: u32 = 16 * 1024 * 1024;
/// Bound on the TCP connect of every outgoing dial. `build_grpc_endpoint` is
/// private and `connect_channel` is the only way out of this module, so every
/// call site picks this up whether it thinks about timeouts or not.
///
/// It bounds the TCP handshake only — tonic hands it to
/// `HttpConnector::set_connect_timeout`. A peer that completes the handshake
/// and then stalls in the TLS or HTTP/2 exchange is not covered; callers that
/// need that bound wrap the whole dial (see `connect_ping_target`).
const GRPC_CONNECT_TIMEOUT: Duration = Duration::from_secs(5);
const GRPC_KEEPALIVE_INTERVAL: Duration = Duration::from_secs(60);
const GRPC_KEEPALIVE_TIMEOUT: Duration = Duration::from_secs(20);
const GRPC_INITIAL_WINDOW_SIZE: u32 = 16 * 1024 * 1024;
#[derive(Clone, Debug)]
pub struct OutgoingGrpcTlsConfig {
@@ -66,9 +40,7 @@ pub fn load_outgoing_grpc_tls(
(&config.grpc_client_cert_file, &config.grpc_client_key_file)
} else {
if !config.grpc_client_cert_file.is_empty() || !config.grpc_client_key_file.is_empty() {
tracing::warn!(
"grpc.volume.client_cert and grpc.volume.client_key must both be set, falling back to grpc.volume.cert and grpc.volume.key"
);
tracing::warn!("grpc.volume.client_cert and grpc.volume.client_key must both be set, falling back to grpc.volume.cert and grpc.volume.key");
}
(&config.grpc_cert_file, &config.grpc_key_file)
};
@@ -107,7 +79,7 @@ pub fn grpc_endpoint_uri(grpc_host_port: &str, tls: Option<&OutgoingGrpcTlsConfi
format!("{}://{}", scheme, grpc_host_port)
}
fn build_grpc_endpoint(
pub fn build_grpc_endpoint(
grpc_host_port: &str,
tls: Option<&OutgoingGrpcTlsConfig>,
) -> Result<Endpoint, GrpcClientError> {
@@ -143,184 +115,6 @@ fn build_grpc_endpoint(
Ok(endpoint)
}
/// Connect `endpoint` through a connector that re-validates every resolved
/// address at connect time (Go's `guardedDialerPolicy` mirror), pinning a
/// validated copy/tail source against DNS rebinding. `allow_untrusted`
/// preserves the plain connect for operators that opted out.
pub async fn connect_guarded(
endpoint: Endpoint,
target: &str,
allow_untrusted: bool,
) -> Result<Channel, GrpcClientError> {
if allow_untrusted {
return endpoint
.connect()
.await
.map_err(|e| GrpcClientError(format!("connect {} failed: {}", target, e)));
}
let target_owned = target.to_string();
let connector = tower::service_fn(move |uri: Uri| {
let target = target_owned.clone();
async move {
let host = uri.host().unwrap_or_default().to_string();
let port = uri.port_u16().unwrap_or(80);
crate::remote_storage::guarded_tcp_connect(&host, port, &target)
.await
.map(hyper_util::rt::TokioIo::new)
}
});
endpoint
.connect_with_connector(connector)
.await
.map_err(|e| GrpcClientError(format!("connect {} failed: {}", target, e)))
}
/// How a dial is bounded.
///
/// `connect_timeout` is handed to the TCP connector. `request_timeout` becomes
/// [`Endpoint::timeout`], which tonic installs as a `GrpcTimeout` layer in
/// front of *every* request the resulting channel carries — it is not a
/// property of one call.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub struct GrpcDialOptions {
/// Bound on establishing the connection to the peer.
pub connect_timeout: Duration,
/// Deadline applied to each RPC on the channel, or `None` to leave them
/// unbounded.
pub request_timeout: Option<Duration>,
}
impl GrpcDialOptions {
/// A short request/response call: connect within 5 s, answer within 10 s.
pub fn unary() -> Self {
Self {
connect_timeout: GRPC_CONNECT_TIMEOUT,
request_timeout: Some(Duration::from_secs(10)),
}
}
/// A call the peer may take a while to answer: connect within 5 s, answer
/// within 30 s.
pub fn long() -> Self {
Self {
connect_timeout: GRPC_CONNECT_TIMEOUT,
request_timeout: Some(Duration::from_secs(30)),
}
}
/// A bounded connect with no deadline on the RPCs themselves.
///
/// `request_timeout` must stay `None` here. [`Endpoint::timeout`] is not a
/// transfer budget: tonic layers it as a `GrpcTimeout` around the
/// response future, which resolves when the server's *first response
/// headers* arrive, so it bounds how long the peer may take to start
/// answering — per request, for every request the channel carries. A 10 s
/// value picked to suit one short call would therefore also be the header
/// deadline for the `VolumeCopy` that shares the dial, and a busy source
/// that takes longer than that to open its file would lose the whole copy.
/// `VolumeCopy`, `VolumeTailSender` and `VolumeEcShardsCopy` have never
/// carried one.
pub fn stream() -> Self {
Self {
connect_timeout: GRPC_CONNECT_TIMEOUT,
request_timeout: None,
}
}
}
/// Dial a peer and return a connected channel.
///
/// The error carries only the transport failure: every caller already wraps it
/// with the address and the operation it was attempting.
pub async fn connect_channel(
grpc_host_port: &str,
tls: Option<&OutgoingGrpcTlsConfig>,
opts: GrpcDialOptions,
) -> Result<Channel, GrpcClientError> {
let mut endpoint =
build_grpc_endpoint(grpc_host_port, tls)?.connect_timeout(opts.connect_timeout);
if let Some(request_timeout) = opts.request_timeout {
endpoint = endpoint.timeout(request_timeout);
}
endpoint
.connect()
.await
.map_err(|e| GrpcClientError(e.to_string()))
}
/// Dial a copy/tail source and return a connected channel, re-validating every
/// resolved address at connect time.
///
/// The guarded equivalent of [`connect_channel`]: same `opts` bounds, but the
/// dial goes through [`connect_guarded`] so a source address that passed
/// validation cannot be re-pointed by DNS between the check and the connect.
/// The bounds are applied to the endpoint *before* delegating, so the
/// `allow_untrusted` opt-out is timed too.
///
/// `target` is the caller-facing source address (the unparsed
/// `"ip:port.grpcPort"` form), which is what the guard pins against; the error
/// carries only the transport failure, as every caller already wraps it with
/// the address and the operation it was attempting.
pub async fn connect_channel_guarded(
grpc_host_port: &str,
target: &str,
tls: Option<&OutgoingGrpcTlsConfig>,
opts: GrpcDialOptions,
allow_untrusted: bool,
) -> Result<Channel, GrpcClientError> {
let mut endpoint =
build_grpc_endpoint(grpc_host_port, tls)?.connect_timeout(opts.connect_timeout);
if let Some(request_timeout) = opts.request_timeout {
endpoint = endpoint.timeout(request_timeout);
}
connect_guarded(endpoint, target, allow_untrusted).await
}
/// The outgoing request-id interceptor as a concrete type, so the client
/// aliases below can name it.
pub type RequestIdInterceptor = fn(Request<()>) -> Result<Request<()>, Status>;
/// A volume-server client with the request-id interceptor attached.
pub type VolumeServerGrpcClient =
VolumeServerClient<InterceptedService<Channel, RequestIdInterceptor>>;
/// A master client with the request-id interceptor attached.
pub type MasterGrpcClient = SeaweedClient<InterceptedService<Channel, RequestIdInterceptor>>;
/// A filer client with the request-id interceptor attached.
pub type FilerGrpcClient = SeaweedFilerClient<InterceptedService<Channel, RequestIdInterceptor>>;
/// Wrap a connected channel in a volume-server client that forwards the
/// current request id and lifts both message-size limits.
pub fn volume_server_client(channel: Channel) -> VolumeServerGrpcClient {
VolumeServerClient::with_interceptor(
channel,
outgoing_request_id_interceptor as RequestIdInterceptor,
)
.max_decoding_message_size(GRPC_MAX_MESSAGE_SIZE)
.max_encoding_message_size(GRPC_MAX_MESSAGE_SIZE)
}
/// Wrap a connected channel in a master client that forwards the current
/// request id and lifts both message-size limits.
pub fn master_client(channel: Channel) -> MasterGrpcClient {
SeaweedClient::with_interceptor(
channel,
outgoing_request_id_interceptor as RequestIdInterceptor,
)
.max_decoding_message_size(GRPC_MAX_MESSAGE_SIZE)
.max_encoding_message_size(GRPC_MAX_MESSAGE_SIZE)
}
/// Wrap a connected channel in a filer client that forwards the current
/// request id and lifts both message-size limits.
pub fn filer_client(channel: Channel) -> FilerGrpcClient {
SeaweedFilerClient::with_interceptor(
channel,
outgoing_request_id_interceptor as RequestIdInterceptor,
)
.max_decoding_message_size(GRPC_MAX_MESSAGE_SIZE)
.max_encoding_message_size(GRPC_MAX_MESSAGE_SIZE)
}
/// Parse a SeaweedFS server address (`"ip:port.grpcPort"` or
/// `"ip:port"`) into the `host:grpcPort` form `build_grpc_endpoint`
/// expects. With the trailing `.grpcPort` segment, that segment IS
@@ -330,27 +124,53 @@ pub fn filer_client(channel: Channel) -> FilerGrpcClient {
/// Shared between `grpc_server.rs` and the distributed-EC-read path
/// in `store_ec.rs` — keep this as the single source of truth so the
/// HTTP↔gRPC port translation can't drift between callers.
///
/// The rule itself lives in `seaweed_common::address`, which the Rust
/// plugin workers share; this wrapper only flattens the typed error
/// back to the `String` its callers already handle. Unbracketed IPv6
/// literals come back bracketed, which this copy used to get wrong.
pub fn parse_grpc_address(source: &str) -> Result<String, String> {
seaweed_common::address::to_grpc_address(source).map_err(|e| e.to_string())
let colon_idx = source
.rfind(':')
.ok_or_else(|| format!("cannot parse address: {}", source))?;
let host = &source[..colon_idx];
let port_part = &source[colon_idx + 1..];
if let Some(dot_idx) = port_part.rfind('.') {
// Format: "ip:port.grpcPort". Validate BOTH ports as u16
// so a malformed HTTP port (e.g. `host:abc.18080`) is
// rejected here rather than tripping a downstream
// `build_grpc_endpoint` URI parse failure with a less
// useful error.
let http_port = &port_part[..dot_idx];
let grpc_port = &port_part[dot_idx + 1..];
http_port
.parse::<u16>()
.map_err(|e| format!("invalid http port {:?}: {}", http_port, e))?;
grpc_port
.parse::<u16>()
.map_err(|e| format!("invalid grpc port {:?}: {}", grpc_port, e))?;
return Ok(format!("{}:{}", host, grpc_port));
}
// Format: "ip:port" → grpc = port + 10000. Reject inputs whose
// implicit grpc port would overflow the TCP port range (e.g.
// `host:60000` produces 70000 — invalid). Without this check
// the cast silently wraps and the endpoint call later fails
// with an opaque connection error.
let port: u16 = port_part
.parse()
.map_err(|e| format!("invalid port {:?}: {}", port_part, e))?;
let grpc_port = port as u32 + 10000;
if grpc_port > u16::MAX as u32 {
return Err(format!(
"implicit grpc port out of range: {} + 10000 = {}",
port, grpc_port
));
}
Ok(format!("{}:{}", host, grpc_port))
}
#[cfg(test)]
mod tests {
use super::{
GrpcDialOptions, build_grpc_endpoint, connect_channel, grpc_endpoint_uri,
load_outgoing_grpc_tls, volume_server_client,
};
use super::{build_grpc_endpoint, grpc_endpoint_uri, load_outgoing_grpc_tls};
use crate::config::{NeedleMapKind, ReadMode, VolumeServerConfig};
use crate::pb::volume_server_pb;
use crate::security::tls::TlsPolicy;
use crate::server::request_id::scope_request_id;
use std::sync::{Arc, Mutex};
use std::time::Duration;
const TEST_CERT_PEM: &str = "-----BEGIN CERTIFICATE-----\nMIIBPDCB76ADAgECAhRuRPQgeAu43BT/M7EfAWSdapVdYDAFBgMrZXAwFDESMBAG\nA1UEAwwJbG9jYWxob3N0MB4XDTI2MDcwNTE2MTUyOVoXDTM2MDcwMjE2MTUyOVow\nFDESMBAGA1UEAwwJbG9jYWxob3N0MCowBQYDK2VwAyEAr/3bNIFI+8V32oCiY6y+\nXRFmZpdNQ2g//VtRkT+nQg+jUzBRMB0GA1UdDgQWBBTsy9tLf1zPiXCQfgci6zNi\ndEzRSjAfBgNVHSMEGDAWgBTsy9tLf1zPiXCQfgci6zNidEzRSjAPBgNVHRMBAf8E\nBTADAQH/MAUGAytlcANBAIvsdw0IbvOBBkb9cd7BfMJfIP9pQQrAL03pCRWJFnFh\nSysaLVgFXI4T078IiaM874oO+iB+5vNbWEpc7CkGow4=\n-----END CERTIFICATE-----\n";
const TEST_KEY_PEM: &str = "-----BEGIN PRIVATE KEY-----\nMC4CAQAwBQYDK2VwBCIEIHbyn71Kk+Y7KT3sBctit7uZpErpoH6qDbFj6P8qGaZH\n-----END PRIVATE KEY-----\n";
@@ -545,131 +365,4 @@ mod tests {
let err = parse_grpc_address("hostname").unwrap_err();
assert!(err.contains("cannot parse"), "{}", err);
}
#[test]
fn test_parse_grpc_address_brackets_ipv6_literals() {
use super::parse_grpc_address;
// This used to come back as `::1:29333`, which is not a valid
// authority: `build_grpc_endpoint` reads the last colon as the port
// separator and rejects the rest.
assert_eq!(parse_grpc_address("::1:19333").unwrap(), "[::1]:29333");
assert_eq!(parse_grpc_address("::1:9333.19333").unwrap(), "[::1]:19333");
// Already bracketed, so it is left alone.
assert_eq!(parse_grpc_address("[::1]:9333").unwrap(), "[::1]:19333");
}
#[test]
fn test_build_grpc_endpoint_accepts_an_ipv6_master_address() {
use super::parse_grpc_address;
let endpoint = build_grpc_endpoint(&parse_grpc_address("::1:9333").unwrap(), None).unwrap();
assert_eq!(endpoint.uri().port_u16(), Some(19333));
}
/// A minimal HTTP/2 server that records the gRPC request headers it is
/// sent and answers every call with a trailers-only `unimplemented`. It is
/// enough to prove what a helper-built client puts on the wire, without
/// standing up the whole `VolumeServer` service behind a tonic server.
async fn serve_header_capture() -> (u16, Arc<Mutex<Option<String>>>) {
use hyper::service::service_fn;
use hyper_util::rt::{TokioExecutor, TokioIo};
let listener = tokio::net::TcpListener::bind("127.0.0.1:0").await.unwrap();
let port = listener.local_addr().unwrap().port();
let seen: Arc<Mutex<Option<String>>> = Arc::new(Mutex::new(None));
let captured = Arc::clone(&seen);
tokio::spawn(async move {
while let Ok((stream, _)) = listener.accept().await {
let captured = Arc::clone(&captured);
tokio::spawn(async move {
let _ = hyper::server::conn::http2::Builder::new(TokioExecutor::new())
.serve_connection(
TokioIo::new(stream),
service_fn(move |req: hyper::Request<hyper::body::Incoming>| {
let captured = Arc::clone(&captured);
async move {
let value = req
.headers()
.get("x-amz-request-id")
.and_then(|v| v.to_str().ok())
.map(str::to_string);
*captured.lock().unwrap() = value;
Ok::<_, std::convert::Infallible>(
hyper::http::Response::builder()
.status(200)
.header("content-type", "application/grpc")
.header("grpc-status", "12")
.body(tonic::body::Body::empty())
.unwrap(),
)
}
}),
)
.await;
});
}
});
(port, seen)
}
#[tokio::test]
async fn test_helper_built_client_sends_the_scoped_request_id() {
let (port, seen) = serve_header_capture().await;
let channel = connect_channel(
&format!("127.0.0.1:{}", port),
None,
GrpcDialOptions::unary(),
)
.await
.expect("dial the header-capturing server");
let mut client = volume_server_client(channel);
// The interceptor has a request id to forward only inside a scope, so
// the call has to run inside one for this to test anything.
let _ = scope_request_id("REQUEST-ID-ON-THE-WIRE".to_string(), async move {
client
.ping(volume_server_pb::PingRequest {
target: String::new(),
target_type: String::new(),
})
.await
})
.await;
assert_eq!(
seen.lock().unwrap().as_deref(),
Some("REQUEST-ID-ON-THE-WIRE"),
"a client built by volume_server_client must carry the outgoing request id"
);
}
#[test]
fn test_dial_presets_match_the_call_sites_they_replace() {
assert_eq!(
GrpcDialOptions::unary().connect_timeout,
Duration::from_secs(5)
);
assert_eq!(
GrpcDialOptions::unary().request_timeout,
Some(Duration::from_secs(10))
);
assert_eq!(
GrpcDialOptions::long().connect_timeout,
Duration::from_secs(5)
);
assert_eq!(
GrpcDialOptions::long().request_timeout,
Some(Duration::from_secs(30))
);
assert_eq!(
GrpcDialOptions::stream().connect_timeout,
Duration::from_secs(5)
);
assert_eq!(
GrpcDialOptions::stream().request_timeout,
None,
"a streaming dial must not put a per-request deadline on the channel"
);
}
}
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
+9 -13
View File
@@ -36,20 +36,16 @@ pub fn collect_mem_status() -> volume_server_pb::MemStatus {
#[cfg(target_os = "linux")]
fn get_system_memory_linux() -> Option<(u64, u64)> {
// SAFETY: `libc::sysinfo` is plain data — integers and trailing padding,
// no pointers and no restricted niches — so the all-zero value is a valid
// one for the kernel to overwrite.
let mut info: libc::sysinfo = unsafe { std::mem::zeroed() };
// SAFETY: `&mut info` is a live, aligned, exclusive pointer to a
// `sysinfo` that the kernel only writes through, and its fields are read
// below only after the call reports success.
if unsafe { libc::sysinfo(&mut info) } != 0 {
return None;
unsafe {
let mut info: libc::sysinfo = std::mem::zeroed();
if libc::sysinfo(&mut info) == 0 {
let unit = info.mem_unit as u64;
let total = info.totalram as u64 * unit;
let free = info.freeram as u64 * unit;
return Some((total, free));
}
}
let unit = info.mem_unit as u64;
let total = info.totalram as u64 * unit;
let free = info.freeram as u64 * unit;
Some((total, free))
None
}
#[cfg(target_os = "linux")]
-113
View File
@@ -1,8 +1,3 @@
use tonic::Status;
use crate::remote_storage::s3_tier::TierError;
use crate::storage::volume::VolumeError;
#[cfg(unix)]
pub mod debug;
pub mod grpc_client;
@@ -18,111 +13,3 @@ pub mod store_ec;
pub mod ui;
pub mod volume_server;
pub mod write_queue;
/// Map a storage error onto the gRPC code that describes it.
impl From<VolumeError> for Status {
fn from(err: VolumeError) -> Self {
let message = err.to_string();
match err {
VolumeError::NotFound
| VolumeError::VolumeNotFound(_)
| VolumeError::Tier(TierError::NotFound(_)) => Status::not_found(message),
VolumeError::ReadOnly(_) | VolumeError::NotEmpty => {
Status::failed_precondition(message)
}
VolumeError::InsufficientSpace { .. } => Status::resource_exhausted(message),
VolumeError::AlreadyExists => Status::already_exists(message),
_ => Status::internal(message),
}
}
}
/// Same mapping, with the RPC's own context prefixed (`compact volume 7: ...`).
pub fn status_with_context(context: &str, err: VolumeError) -> Status {
let status = Status::from(err);
Status::new(status.code(), format!("{context}: {}", status.message()))
}
/// Render a configured disk directory as an absolute path for display, so the
/// status JSON and the UI show the same thing for a relative `-dir`. Falls
/// back to the configured spelling when the current directory cannot be read.
pub(crate) fn absolute_display_path(path: &str) -> String {
let p = std::path::Path::new(path);
if p.is_absolute() {
return path.to_string();
}
std::env::current_dir()
.map(|cwd| cwd.join(p).to_string_lossy().to_string())
.unwrap_or_else(|_| path.to_string())
}
#[cfg(test)]
mod tests {
use super::*;
use crate::storage::types::VolumeId;
#[test]
fn test_volume_error_maps_to_grpc_code() {
use tonic::Code;
let code = |e: VolumeError| Status::from(e).code();
assert_eq!(
code(VolumeError::VolumeNotFound(VolumeId(7))),
Code::NotFound
);
assert_eq!(code(VolumeError::NotFound), Code::NotFound);
assert_eq!(
code(VolumeError::ReadOnly(VolumeId(7))),
Code::FailedPrecondition
);
assert_eq!(
VolumeError::ReadOnly(VolumeId(7)).to_string(),
"volume 7 is read only"
);
assert_eq!(
code(VolumeError::InsufficientSpace {
vid: VolumeId(7),
required: 2,
free: 1,
}),
Code::ResourceExhausted
);
assert_eq!(code(VolumeError::AlreadyExists), Code::AlreadyExists);
assert_eq!(code(VolumeError::NotInitialized), Code::Internal);
assert_eq!(
code(TierError::NotFound("gone".into()).into()),
Code::NotFound
);
for tier in [
TierError::Io("io".into()),
TierError::RuntimeUnavailable("rt".into()),
TierError::Aborted("bye".into()),
] {
assert_eq!(code(tier.into()), Code::Internal);
}
let status = status_with_context(
"backend s3.default copy file /data/1.dat",
TierError::NotFound("failed to head object k: NotFound".into()).into(),
);
assert_eq!(status.code(), Code::NotFound);
assert_eq!(
status.message(),
"backend s3.default copy file /data/1.dat: failed to head object k: NotFound"
);
let status = status_with_context(
"compact volume 7",
VolumeError::InsufficientSpace {
vid: VolumeId(7),
required: 2,
free: 1,
},
);
assert_eq!(status.code(), Code::ResourceExhausted);
assert_eq!(
status.message(),
"compact volume 7: not enough free space: required 2, free 1"
);
}
}
+3 -3
View File
@@ -29,11 +29,11 @@ impl<S> Layer<S> for GrpcRequestIdLayer {
impl<S, B> Service<http::Request<B>> for GrpcRequestIdService<S>
where
S: Service<http::Request<B>, Response = http::Response<tonic::body::Body>> + Send + 'static,
S: Service<http::Request<B>, Response = http::Response<tonic::body::BoxBody>> + Send + 'static,
S::Future: Send + 'static,
B: Send + 'static,
{
type Response = http::Response<tonic::body::Body>;
type Response = http::Response<tonic::body::BoxBody>;
type Error = S::Error;
type Future = Pin<Box<dyn Future<Output = Result<Self::Response, Self::Error>> + Send>>;
@@ -57,7 +57,7 @@ where
let future = self.inner.call(request);
Box::pin(async move {
let mut response: http::Response<tonic::body::Body> =
let mut response: http::Response<tonic::body::BoxBody> =
scope_request_id(request_id.clone(), future).await?;
if let Ok(value) = HeaderValue::from_str(&request_id) {
response.headers_mut().insert("x-amz-request-id", value);
+209 -493
View File
@@ -26,7 +26,7 @@
//! cache write-back briefly reacquires the EcVolume's internal
//! `RwLock` so we do not contend with the Store-level lock at all.
use std::collections::{HashMap, HashSet};
use std::collections::HashMap;
use std::fs;
use std::io;
use std::sync::Arc;
@@ -38,16 +38,15 @@ use reed_solomon_erasure::galois_8::ReedSolomon;
use tokio::sync::Semaphore;
use tonic::Request;
use crate::pb::master_pb::{self, LookupEcVolumeRequest};
use crate::pb::master_pb::{self, seaweed_client::SeaweedClient, LookupEcVolumeRequest};
use crate::pb::volume_server_pb::{
CopyFileRequest, VolumeEcBlobDeleteRequest, VolumeEcShardReadRequest,
volume_server_client::VolumeServerClient, CopyFileRequest, VolumeEcShardReadRequest,
};
use crate::server::grpc_client::{
GrpcDialOptions, connect_channel, master_client, parse_grpc_address, volume_server_client,
};
use crate::server::volume_server::{VolumeServerState, to_http_address};
use crate::storage::erasure_coding::ec_shard::{ShardId, shard_id_try_from};
use crate::storage::needle::needle::{Needle, NeedleError, get_actual_size};
use crate::server::grpc_client::{build_grpc_endpoint, parse_grpc_address, GRPC_MAX_MESSAGE_SIZE};
use crate::server::request_id::outgoing_request_id_interceptor;
use crate::server::volume_server::{to_http_address, VolumeServerState};
use crate::storage::erasure_coding::ec_shard::ShardId;
use crate::storage::needle::needle::{get_actual_size, Needle, NeedleError};
use crate::storage::store_ec_reconcile::EcVolumeMissingIndex;
use crate::storage::types::*;
use crate::storage::volume::volume_file_name;
@@ -95,41 +94,20 @@ struct Snapshot {
encode_ts_ns: i64,
}
/// Why a distributed EC read has no needle to return.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum EcMiss {
NotFound,
/// Tombstoned in the local `.ecx`/`.ecj`, or reported deleted by a peer.
Deleted,
VolumeNotFound,
}
/// Top-level entry point. Returns `Ok(None)` for any miss — absent,
/// deleted, or volume gone; errors propagate as `io::Error`.
/// Top-level entry point. Returns `Ok(None)` for "not found" (matches
/// Go's `ReadEcShardNeedle`); errors propagate as `io::Error`.
pub async fn read_ec_shard_needle_distributed(
state: &Arc<VolumeServerState>,
vid: VolumeId,
needle_id: NeedleId,
) -> io::Result<Option<Needle>> {
Ok(read_ec_shard_needle_or_miss(state, vid, needle_id)
.await?
.ok())
}
/// Like `read_ec_shard_needle_distributed`, but says why there is no needle,
/// as Go's `ReadEcShardNeedle` tells `ErrorDeleted` from not-found.
pub async fn read_ec_shard_needle_or_miss(
state: &Arc<VolumeServerState>,
vid: VolumeId,
needle_id: NeedleId,
) -> io::Result<Result<Needle, EcMiss>> {
// Phase A — under the Store read lock, locate the needle, compute
// intervals, and read any locally-mounted shard intervals. We must
// not `.await` while holding this guard (std::sync::RwLockReadGuard
// is !Send).
let mut snapshot = match snapshot_under_lock(state, vid, needle_id)? {
Ok(s) => s,
Err(miss) => return Ok(Err(miss)),
Some(s) => s,
None => return Ok(None),
};
// Phase B — refresh the shard_locations cache from the master if
@@ -208,19 +186,15 @@ pub async fn read_ec_shard_needle_or_miss(
} => {
fetch_one_interval(
state,
EcInterval {
vid,
needle_id,
shard_id,
shard_offset,
size,
expected_encode_ts_ns: encode_ts_ns,
},
EcShardMap {
locations: shard_locations,
data_shards,
parity_shards,
},
vid,
needle_id,
shard_id,
shard_offset,
size,
shard_locations,
data_shards,
parity_shards,
encode_ts_ns,
)
.await
}
@@ -231,11 +205,17 @@ pub async fn read_ec_shard_needle_or_miss(
.collect()
.await;
// A peer reports the needle deleted (a cross-server window where the
// local index still shows it live): answer deleted rather than serving zeros.
let Some(assembled) = gather_intervals(fetched)? else {
return Ok(Err(EcMiss::Deleted));
};
let mut assembled: Vec<Vec<u8>> = Vec::with_capacity(fetched.len());
for res in fetched {
let (buf, is_deleted) = res?;
// A peer reports the needle deleted (a cross-server window where the
// local index still shows it live): treat as not-found rather than
// serving zeros, mirroring Go's ErrorDeleted.
if is_deleted {
return Ok(None);
}
assembled.push(buf);
}
// Phase D — assemble and parse the Needle. Mirrors the tail of
// `EcVolume::read_ec_shard_needle`.
@@ -267,271 +247,7 @@ pub async fn read_ec_shard_needle_or_miss(
snapshot.version,
)
.map_err(|e| io::Error::new(io::ErrorKind::InvalidData, format!("{}", e)))?;
Ok(Ok(n))
}
/// `None` when any holder reported the needle deleted. That outranks another
/// interval's error: deletes are never invented, so the needle is gone either way.
fn gather_intervals(fetched: Vec<io::Result<(Vec<u8>, bool)>>) -> io::Result<Option<Vec<Vec<u8>>>> {
if fetched.iter().any(|r| matches!(r, Ok((_, true)))) {
return Ok(None);
}
fetched
.into_iter()
.map(|r| r.map(|(buf, _)| buf))
.collect::<io::Result<_>>()
.map(Some)
}
/// What one EC delete RPC carries — `VolumeEcBlobDeleteRequest` minus tonic.
struct EcDeleteTarget<'a> {
vid: VolumeId,
collection: &'a str,
version: Version,
needle_id: NeedleId,
}
/// `Store.doDeleteNeedleFromAtLeastOneRemoteEcShards` in Go: journal the
/// tombstone on one holder of the needle's primary data shard, falling back
/// to any other shard holder when the primary has none. Exactly one node
/// journals — replicas of a shard hold identical .ecx copies, so journaling
/// on more than one would double the reported delete count.
///
/// `NotFound` means the volume or needle is gone; other errors mean every
/// reachable holder failed or no shard has a holder at all.
pub async fn delete_ec_shard_needle_distributed(
state: &Arc<VolumeServerState>,
vid: VolumeId,
needle_id: NeedleId,
) -> io::Result<()> {
let (
primary_shard_id,
collection,
version,
total_shards,
local_shards,
data_shards,
encode_ts_ns,
refreshed_at,
cached_locations,
) = {
let store = state.store.read().unwrap();
let ecv = store.find_ec_volume(vid).ok_or_else(|| {
io::Error::new(
io::ErrorKind::NotFound,
format!("ec volume {} not mounted", vid.0),
)
})?;
let (_, _, intervals) = ecv.locate_needle(needle_id)?.ok_or_else(|| {
io::Error::new(
io::ErrorKind::NotFound,
format!("needle {} not in ec volume {}", needle_id, vid.0),
)
})?;
let (shard_id, _) = intervals
.first()
.map(|i| ecv.interval_to_shard_id_and_offset(i))
.ok_or_else(|| io::Error::new(io::ErrorKind::NotFound, "no intervals for needle"))?;
let (cached_locations, refreshed_at) = ecv.shard_locations_snapshot();
(
shard_id,
ecv.collection.clone(),
ecv.version,
ecv.data_shards + ecv.parity_shards,
local_shard_ids(ecv),
ecv.data_shards as usize,
ecv.encode_ts_ns,
refreshed_at,
cached_locations,
)
};
let target = EcDeleteTarget {
vid,
collection: &collection,
version,
needle_id,
};
// Holder addresses come from the same staleness-gated cache as the read
// path: a master LookupEcVolume only when due, merged on a complete reply.
let mut locations = cached_locations;
if claim_shard_locations_refresh(
state,
vid,
&locations,
refreshed_at,
data_shards,
total_shards as usize,
) {
match cached_lookup_ec_shard_locations(state, vid).await {
Ok(fresh) => {
match write_back_shard_locations(state, vid, fresh, data_shards, encode_ts_ns) {
Some(merged) => locations = merged,
None => mark_shard_locations_stale(state, vid),
}
}
Err(_) => mark_shard_locations_stale(state, vid),
}
}
match delete_on_ec_shard_holders(state, &locations, &local_shards, primary_shard_id, &target)
.await
{
Ok(true) => return Ok(()),
Err(e) => return Err(e),
Ok(false) => {}
}
for shard_id in 0..total_shards {
let Ok(shard_id) = shard_id_try_from(shard_id) else {
continue;
};
if shard_id == primary_shard_id {
continue;
}
if let Ok(true) =
delete_on_ec_shard_holders(state, &locations, &local_shards, shard_id, &target).await
{
return Ok(());
}
}
Err(io::Error::other(format!(
"ec volume {}: no shard holder could journal the delete",
vid.0
)))
}
/// `doDeleteNeedleFromRemoteEcShardServers` in Go. `Ok(false)` is the
/// shard-missing signal — no live holder anywhere — that triggers the
/// caller's fallback walk over the remaining shards.
async fn delete_on_ec_shard_holders(
state: &Arc<VolumeServerState>,
locations: &HashMap<ShardId, Vec<String>>,
local_shards: &HashSet<ShardId>,
shard_id: ShardId,
target: &EcDeleteTarget<'_>,
) -> io::Result<bool> {
let addrs = locations.get(&shard_id);
if !local_shards.contains(&shard_id) && addrs.is_none_or(|a| a.is_empty()) {
return Ok(false);
}
let mut last_err = None;
if local_shards.contains(&shard_id) {
// A decode in its publishing tail must not miss this delete. The
// tail membership is verified again under the store write lock
// inside journal_delete_local — WouldBlock means the decode claimed
// it in the gap after this wait — so wait and retry.
loop {
crate::server::grpc_server::wait_ec_decode_tail(state, target.vid).await;
match journal_delete_local(state, target.vid, target.needle_id) {
Ok(()) => return Ok(true),
Err(e) if e.kind() == io::ErrorKind::WouldBlock => continue,
// Nothing was committed — the volume unmounted or remounted
// without the needle — so it is safe to fall back to other
// shard holders, unlike an RPC failure which may have landed.
Err(e) if e.kind() == io::ErrorKind::NotFound => return Ok(false),
Err(e) => {
last_err = Some(e);
break;
}
}
}
}
if let Some(addrs) = addrs {
let self_http = to_http_address(&state.self_url);
for addr in addrs {
// A stale self entry: the loopback RPC would journal on this same
// volume, which the local attempt above already covered.
if to_http_address(addr).as_ref() == self_http.as_ref() {
continue;
}
match delete_on_remote_ec_shard(state, addr, target).await {
Ok(()) => return Ok(true),
Err(e) => last_err = Some(e),
}
}
}
match last_err {
Some(e) => Err(e),
None => Ok(false),
}
}
/// `doDeleteNeedleFromRemoteEcShard` in Go — one `VolumeEcBlobDelete` RPC.
async fn delete_on_remote_ec_shard(
state: &Arc<VolumeServerState>,
addr: &str,
target: &EcDeleteTarget<'_>,
) -> io::Result<()> {
let grpc_addr =
parse_grpc_address(addr).map_err(|e| io::Error::new(io::ErrorKind::InvalidInput, e))?;
let channel = connect_channel(
&grpc_addr,
state.outgoing_grpc_tls.as_ref(),
GrpcDialOptions::unary(),
)
.await
.map_err(|e| io::Error::other(format!("connect to {}: {}", addr, e)))?;
let mut client = volume_server_client(channel);
client
.volume_ec_blob_delete(Request::new(VolumeEcBlobDeleteRequest {
volume_id: target.vid.0,
collection: target.collection.to_string(),
file_key: target.needle_id.0,
version: target.version.0 as u32,
}))
.await
.map_err(|e| io::Error::other(format!("volume_ec_blob_delete on {}: {}", addr, e)))?;
Ok(())
}
/// Journals on the local volume — what the `VolumeEcBlobDelete` handler runs
/// when this server is the shard holder. An absent needle is an error, not a
/// no-op: `journal_delete` would accept it silently, but here it means the
/// volume remounted as a different generation mid-delete and the tombstone
/// should go to a replica that still has the needle.
fn journal_delete_local(
state: &Arc<VolumeServerState>,
vid: VolumeId,
needle_id: NeedleId,
) -> io::Result<()> {
let mut store = state.store.write().unwrap();
// Membership is read under the write lock: a decode can claim the
// publishing tail while this call waited for the decoder's read lock,
// so a check taken earlier would be stale by commit time.
if crate::server::grpc_server::ec_decode_tail_contains(state, vid) {
return Err(io::Error::new(
io::ErrorKind::WouldBlock,
format!("ec volume {} is in decode publishing tail", vid.0),
));
}
let ecv = store.find_ec_volume_mut(vid).ok_or_else(|| {
io::Error::new(
io::ErrorKind::NotFound,
format!("ec volume {} unmounted", vid.0),
)
})?;
match ecv.find_needle_from_ecx(needle_id)? {
None => {
return Err(io::Error::new(
io::ErrorKind::NotFound,
format!("needle {} not in local ecx", needle_id),
));
}
Some((_, size)) if size.is_deleted() => return Ok(()),
Some(_) => {}
}
ecv.journal_delete(needle_id)
}
fn local_shard_ids(ecv: &crate::storage::erasure_coding::EcVolume) -> HashSet<ShardId> {
ecv.shards
.iter()
.enumerate()
.filter_map(|(i, s)| s.as_ref().map(|_| i as ShardId))
.collect()
Ok(Some(n))
}
/// FULL EC scrub: verify every needle's bytes across local AND remote shards,
@@ -580,7 +296,7 @@ pub async fn scrub_ec_volume_distributed(
0,
Vec::new(),
vec![format!("EC volume id {} not found", vid.0)],
);
)
}
};
// full scan means verifying the index as well
@@ -604,9 +320,9 @@ pub async fn scrub_ec_volume_distributed(
// mounted volume's encode_ts_ns no longer matches, abort like a
// mid-scan unmount rather than mixing generations.
let encode_ts_ns = ecv.encode_ts_ns;
// One read section, so the map and the refresh time it is aged against
// describe the same lookup.
let (cached_locations, cache_refreshed_at) = ecv.shard_locations_snapshot();
// Bind to locals so the inner RwLock/Mutex guards drop before the block ends.
let cached_locations = ecv.shard_locations.read().unwrap().clone();
let cache_refreshed_at = *ecv.shard_locations_refresh_time.lock().unwrap();
let data_shards = ecv.data_shards as usize;
let total_shards = (ecv.data_shards + ecv.parity_shards) as usize;
(
@@ -708,7 +424,7 @@ pub async fn scrub_ec_volume_distributed(
);
}
};
ecv.shard_locations_snapshot().0
ecv.shard_locations.read().unwrap().clone()
};
// Walk the .ecx (private fd captured under the lock, no lock held) for the
@@ -805,15 +521,18 @@ pub async fn scrub_ec_volume_distributed(
} => {
let sources: &[String] =
locations.get(shard_id).map(Vec::as_slice).unwrap_or(&[]);
let iv = EcInterval {
match read_remote_ec_shard_interval(
state,
sources,
vid,
needle_id: id,
shard_id: *shard_id,
shard_offset: *shard_offset,
size: *ssize,
expected_encode_ts_ns: snapshot.encode_ts_ns,
};
match read_remote_ec_shard_interval(state, sources, iv).await {
id,
*shard_id,
*shard_offset,
*ssize,
snapshot.encode_ts_ns,
)
.await
{
// A deleted shard yields no bytes; zero-fill the interval so
// the assembled needle reaches read_bytes -> SizeMismatch{0}
// -> the delete-state suppression (mirrors Go's pre-zeroed buffer).
@@ -841,12 +560,15 @@ pub async fn scrub_ec_volume_distributed(
}
match recover_one_remote_ec_shard_interval(
state,
iv,
EcShardMap {
locations: &locations,
data_shards,
parity_shards: total_shards - data_shards,
},
vid,
id,
*shard_id,
*shard_offset,
*ssize,
&locations,
data_shards,
total_shards - data_shards,
snapshot.encode_ts_ns,
)
.await
{
@@ -912,10 +634,11 @@ fn snapshot_under_lock(
state: &Arc<VolumeServerState>,
vid: VolumeId,
needle_id: NeedleId,
) -> io::Result<Result<Snapshot, EcMiss>> {
) -> io::Result<Option<Snapshot>> {
let store = state.store.read().unwrap();
let Some(ecv) = store.find_ec_volume(vid) else {
return Ok(Err(EcMiss::VolumeNotFound));
let ecv = match store.find_ec_volume(vid) {
Some(v) => v,
None => return Ok(None),
};
// Reuse EcVolume::locate_needle for offset/size resolution AND
@@ -923,17 +646,11 @@ fn snapshot_under_lock(
// local-only read path uses, so we stay byte-identical on the
// shard-size + interval boundaries. locate_needle applies the runtime
// delete mask, which is correct for serving reads.
let Some((offset, size, intervals)) = ecv.locate_needle(needle_id)? else {
// locate_needle folds a tombstone into not-found.
let deleted =
matches!(ecv.find_needle_from_ecx(needle_id)?, Some((_, s)) if s.is_deleted());
return Ok(Err(if deleted {
EcMiss::Deleted
} else {
EcMiss::NotFound
}));
let (offset, size, intervals) = match ecv.locate_needle(needle_id)? {
Some(v) => v,
None => return Ok(None),
};
build_snapshot(ecv, offset, size, &intervals).map(Ok)
build_snapshot(ecv, offset, size, &intervals).map(Some)
}
/// Like `snapshot_under_lock`, but locates intervals from the RAW .ecx
@@ -957,7 +674,7 @@ fn scrub_snapshot_under_lock(
return Err(io::Error::new(
io::ErrorKind::NotFound,
format!("EC volume {} not found (unmounted mid-scan)", vid.0),
));
))
}
};
// The volume was torn down and remounted as a DIFFERENT encode run between
@@ -1058,7 +775,8 @@ fn build_snapshot(
}
let actual = get_actual_size(size, ecv.version);
let interval_results = read_local_intervals(ecv, intervals);
let (cached_locations, cache_refreshed_at) = ecv.shard_locations_snapshot();
let cached_locations = ecv.shard_locations.read().unwrap().clone();
let cache_refreshed_at = *ecv.shard_locations_refresh_time.lock().unwrap();
Ok(Snapshot {
data_shards: ecv.data_shards,
@@ -1110,14 +828,14 @@ fn needs_refresh(
fn mark_shard_locations_stale(state: &Arc<VolumeServerState>, vid: VolumeId) {
let store = state.store.read().unwrap();
if let Some(ecv) = store.find_ec_volume(vid) {
ecv.mark_shard_locations_stale();
*ecv.shard_locations_stale.lock().unwrap() = true;
}
}
/// Decide whether the caller's snapshot is due a master lookup and, when it is,
/// consume the cache's stale mark in the same critical section. A mark raised
/// from here on belongs to the next refresh: the read that raised it has
/// disproved the map this lookup is about to install.
/// Decide whether the cached map is due a master lookup and, when it is, consume
/// its stale mark in the same critical section. A mark raised from here on
/// belongs to the next refresh: the read that raised it has disproved the map
/// this lookup is about to install.
fn claim_shard_locations_refresh(
state: &Arc<VolumeServerState>,
vid: VolumeId,
@@ -1130,9 +848,12 @@ fn claim_shard_locations_refresh(
let Some(ecv) = store.find_ec_volume(vid) else {
return needs_refresh(locations, refreshed_at, false, data_shards, total_shards);
};
ecv.claim_shard_locations_refresh(|stale| {
needs_refresh(locations, refreshed_at, stale, data_shards, total_shards)
})
let mut stale = ecv.shard_locations_stale.lock().unwrap();
let refresh = needs_refresh(locations, refreshed_at, *stale, data_shards, total_shards);
if refresh {
*stale = false;
}
refresh
}
async fn cached_lookup_ec_shard_locations(
@@ -1153,15 +874,18 @@ async fn cached_lookup_ec_shard_locations(
let grpc_addr =
parse_grpc_address(&master).map_err(|e| io::Error::new(io::ErrorKind::InvalidInput, e))?;
let channel = connect_channel(
&grpc_addr,
state.outgoing_grpc_tls.as_ref(),
GrpcDialOptions::unary(),
)
.await
.map_err(|e| io::Error::other(format!("master connect: {}", e)))?;
let endpoint = build_grpc_endpoint(&grpc_addr, state.outgoing_grpc_tls.as_ref())
.map_err(|e| io::Error::other(e.to_string()))?;
let channel = endpoint
.connect_timeout(Duration::from_secs(5))
.timeout(Duration::from_secs(10))
.connect()
.await
.map_err(|e| io::Error::other(format!("master connect: {}", e)))?;
let mut client = master_client(channel);
let mut client = SeaweedClient::with_interceptor(channel, outgoing_request_id_interceptor)
.max_decoding_message_size(GRPC_MAX_MESSAGE_SIZE)
.max_encoding_message_size(GRPC_MAX_MESSAGE_SIZE);
let resp = client
.lookup_ec_volume(Request::new(LookupEcVolumeRequest { volume_id: vid.0 }))
@@ -1176,12 +900,7 @@ async fn cached_lookup_ec_shard_locations(
.iter()
.map(format_location_as_server_address)
.collect();
// Defensive: skip out-of-range shard ids from the master instead of
// truncating (256 would alias 0). Valid replies are unaffected.
let Ok(sid) = shard_id_try_from(entry.shard_id) else {
continue;
};
out.insert(sid, addrs);
out.insert(entry.shard_id as ShardId, addrs);
}
Ok(out)
}
@@ -1246,44 +965,37 @@ fn format_location_as_server_address(loc: &master_pb::Location) -> String {
raw.to_string()
}
/// One shard-relative byte range of a needle on an EC volume, the unit the
/// peer-read and recovery paths work in. Mirrors the argument list of Go's
/// `readOneEcShardInterval`.
#[derive(Clone, Copy, Debug)]
struct EcInterval {
/// Try direct peer read; on failure, reconstruct via Reed-Solomon
/// from the other shards. Mirrors `readOneEcShardInterval`'s tail.
#[expect(clippy::too_many_arguments)]
async fn fetch_one_interval(
state: &Arc<VolumeServerState>,
vid: VolumeId,
needle_id: NeedleId,
/// The shard the bytes live on, or the one to rebuild when recovering.
shard_id: ShardId,
shard_offset: i64,
size: usize,
/// Encode run the caller expects the shard to belong to; 0 accepts any,
/// for peers that predate the identity check.
expected_encode_ts_ns: i64,
}
/// Where the shards of one EC volume can be fetched from, and the volume's
/// Reed-Solomon shape, as recovery needs both together.
#[derive(Clone, Copy)]
struct EcShardMap<'a> {
locations: &'a HashMap<ShardId, Vec<String>>,
shard_locations: &HashMap<ShardId, Vec<String>>,
data_shards: usize,
parity_shards: usize,
}
/// Try direct peer read; on failure, reconstruct via Reed-Solomon
/// from the other shards. Mirrors `readOneEcShardInterval`'s tail.
async fn fetch_one_interval(
state: &Arc<VolumeServerState>,
iv: EcInterval,
map: EcShardMap<'_>,
expected_encode_ts_ns: i64,
) -> io::Result<(Vec<u8>, bool)> {
let EcInterval { vid, shard_id, .. } = iv;
// Direct peer read against the cached locations for this shard.
if let Some(sources) = map.locations.get(&shard_id)
if let Some(sources) = shard_locations.get(&shard_id)
&& !sources.is_empty()
{
match read_remote_ec_shard_interval(state, sources, iv).await {
match read_remote_ec_shard_interval(
state,
sources,
vid,
needle_id,
shard_id,
shard_offset,
size,
expected_encode_ts_ns,
)
.await
{
// A deleted needle short-circuits: don't reconstruct (every shard
// would report deleted), let the caller return "deleted".
Ok((buf, is_deleted)) => return Ok((buf, is_deleted)),
@@ -1304,17 +1016,46 @@ async fn fetch_one_interval(
// Reconstruct: fan-out reads to every other shard at the same
// (shard_offset, size). Mirrors `recoverOneRemoteEcShardInterval`.
recover_one_remote_ec_shard_interval(state, iv, map).await
recover_one_remote_ec_shard_interval(
state,
vid,
needle_id,
shard_id,
shard_offset,
size,
shard_locations,
data_shards,
parity_shards,
expected_encode_ts_ns,
)
.await
}
#[expect(clippy::too_many_arguments)]
async fn read_remote_ec_shard_interval(
state: &Arc<VolumeServerState>,
sources: &[String],
iv: EcInterval,
vid: VolumeId,
needle_id: NeedleId,
shard_id: ShardId,
shard_offset: i64,
size: usize,
expected_encode_ts_ns: i64,
) -> io::Result<(Vec<u8>, bool)> {
let mut last_err: Option<io::Error> = None;
for src in sources {
match do_read_remote_ec_shard_interval(state, src, iv).await {
match do_read_remote_ec_shard_interval(
state,
src,
vid,
needle_id,
shard_id,
shard_offset,
size,
expected_encode_ts_ns,
)
.await
{
Ok(res) => return Ok(res),
Err(e) => last_err = Some(e),
}
@@ -1322,33 +1063,32 @@ async fn read_remote_ec_shard_interval(
Err(last_err.unwrap_or_else(|| {
io::Error::new(
io::ErrorKind::NotFound,
format!("no source for ec shard {}.{}", iv.vid.0, iv.shard_id),
format!("no source for ec shard {}.{}", vid.0, shard_id),
)
}))
}
#[expect(clippy::too_many_arguments)]
async fn do_read_remote_ec_shard_interval(
state: &Arc<VolumeServerState>,
source: &str,
iv: EcInterval,
vid: VolumeId,
needle_id: NeedleId,
shard_id: ShardId,
shard_offset: i64,
size: usize,
expected_encode_ts_ns: i64,
) -> io::Result<(Vec<u8>, bool)> {
let EcInterval {
vid,
needle_id,
shard_id,
shard_offset,
size,
expected_encode_ts_ns,
} = iv;
let grpc_addr =
parse_grpc_address(source).map_err(|e| io::Error::new(io::ErrorKind::InvalidInput, e))?;
let channel = connect_channel(
&grpc_addr,
state.outgoing_grpc_tls.as_ref(),
GrpcDialOptions::long(),
)
.await
.map_err(|e| io::Error::other(format!("connect to {}: {}", source, e)))?;
let endpoint = build_grpc_endpoint(&grpc_addr, state.outgoing_grpc_tls.as_ref())
.map_err(|e| io::Error::other(e.to_string()))?;
let channel = endpoint
.connect_timeout(Duration::from_secs(5))
.timeout(Duration::from_secs(30))
.connect()
.await
.map_err(|e| io::Error::other(format!("connect to {}: {}", source, e)))?;
// TODO(grpc-jwt): clusters with `jwt.signing.key` configured will
// reject peer-to-peer VolumeEcShardRead calls until the Rust
@@ -1358,7 +1098,9 @@ async fn do_read_remote_ec_shard_interval(
// here in isolation would split the credential plumbing across
// call sites. Re-visit when outgoing JWT signing lands as a
// server-wide helper.
let mut client = volume_server_client(channel);
let mut client = VolumeServerClient::with_interceptor(channel, outgoing_request_id_interceptor)
.max_decoding_message_size(GRPC_MAX_MESSAGE_SIZE)
.max_encoding_message_size(GRPC_MAX_MESSAGE_SIZE);
let req = VolumeEcShardReadRequest {
volume_id: vid.0,
@@ -1429,24 +1171,19 @@ async fn do_read_remote_ec_shard_interval(
Ok((out, false))
}
#[expect(clippy::too_many_arguments)]
async fn recover_one_remote_ec_shard_interval(
state: &Arc<VolumeServerState>,
iv: EcInterval,
map: EcShardMap<'_>,
vid: VolumeId,
needle_id: NeedleId,
shard_id_to_recover: ShardId,
shard_offset: i64,
size: usize,
shard_locations: &HashMap<ShardId, Vec<String>>,
data_shards: usize,
parity_shards: usize,
expected_encode_ts_ns: i64,
) -> io::Result<(Vec<u8>, bool)> {
let EcInterval {
vid,
needle_id,
shard_id: shard_id_to_recover,
shard_offset,
size,
expected_encode_ts_ns,
} = iv;
let EcShardMap {
locations: shard_locations,
data_shards,
parity_shards,
} = map;
let total_shards = data_shards + parity_shards;
let rs = ReedSolomon::new(data_shards, parity_shards)
.map_err(|e| io::Error::other(format!("reed-solomon init: {:?}", e)))?;
@@ -1487,10 +1224,7 @@ async fn recover_one_remote_ec_shard_interval(
// shard from a different encode run must not be fed to Reed-Solomon;
// lenient only when the caller carries no identity (pre-upgrade).
// Mirrors Go's `readLocalEcShardInterval`.
let Ok(sid_shard) = ShardId::try_from(sid) else {
continue;
};
let owner = match store.find_ec_volume_with_shard(vid, sid_shard) {
let owner = match store.find_ec_volume_with_shard(vid, sid as u32) {
Some(ecv)
if expected_encode_ts_ns == 0 || ecv.encode_ts_ns == expected_encode_ts_ns =>
{
@@ -1538,10 +1272,12 @@ async fn recover_one_remote_ec_shard_interval(
let res = read_remote_ec_shard_interval(
&state,
&locs,
EcInterval {
shard_id: sid,
..iv
},
vid,
needle_id,
sid,
shard_offset,
size,
expected_encode_ts_ns,
)
.await;
(sid, res)
@@ -1746,14 +1482,16 @@ async fn fetch_ec_index_from_one_peer(
) -> io::Result<()> {
let grpc_addr =
parse_grpc_address(peer).map_err(|e| io::Error::new(io::ErrorKind::InvalidInput, e))?;
let channel = connect_channel(
&grpc_addr,
state.outgoing_grpc_tls.as_ref(),
GrpcDialOptions::long(),
)
.await
.map_err(|e| io::Error::other(format!("connect {}: {}", peer, e)))?;
let mut client = volume_server_client(channel);
let channel = build_grpc_endpoint(&grpc_addr, state.outgoing_grpc_tls.as_ref())
.map_err(|e| io::Error::other(e.to_string()))?
.connect_timeout(Duration::from_secs(5))
.timeout(Duration::from_secs(30))
.connect()
.await
.map_err(|e| io::Error::other(format!("connect {}: {}", peer, e)))?;
let mut client = VolumeServerClient::with_interceptor(channel, outgoing_request_id_interceptor)
.max_decoding_message_size(GRPC_MAX_MESSAGE_SIZE)
.max_encoding_message_size(GRPC_MAX_MESSAGE_SIZE);
let copy_req = |ext: &str, ignore_not_found: bool| CopyFileRequest {
volume_id: m.vid.0,
@@ -1773,7 +1511,7 @@ async fn fetch_ec_index_from_one_peer(
.await
.map_err(|e| io::Error::other(format!("copy .ecx: {}", e)))?
.into_inner();
drain_copy_stream(stream, ecx_path).await?;
drain_copy_stream(stream, ecx_path, false).await?;
let meta =
fs::metadata(ecx_path).map_err(|e| io::Error::other(format!("stat copied .ecx: {}", e)))?;
@@ -1786,30 +1524,15 @@ async fn fetch_ec_index_from_one_peer(
)));
}
// .ecj is the source peer's deletion journal; .vif carries EC params. Both
// are best-effort: a missing .ecj is recreated at mount and a missing .vif
// falls back to default EC parameters. The journal is a *set*: merge the
// peer's ids into any local ones as a union instead of appending, so
// a volume bounced between servers cannot double its journal. The merge
// only appends whole records, so a failure leaves nothing to clean up.
// .ecj is the source peer's deletion journal (appended); .vif carries EC
// params. Both are best-effort: a missing .ecj is recreated at mount and a
// missing .vif falls back to default EC parameters. A failed .ecj append
// leaves a partial file, so drop it.
match client.copy_file(copy_req(".ecj", true)).await {
Ok(resp) => {
let mut stream = resp.into_inner();
let merged = match crate::server::grpc_server::receive_ecj_ids(&mut stream).await {
Ok((ids, true)) => crate::server::grpc_server::merge_ecj_ids(
state,
m.vid,
m.data_dir.clone(),
ecj_path.to_string(),
ids,
)
.await
.map(|_| ()),
Ok((_, false)) => Ok(()),
Err(e) => Err(e),
};
if let Err(e) = merged {
if let Err(e) = drain_copy_stream(resp.into_inner(), ecj_path, true).await {
tracing::warn!(volume_id = m.vid.0, peer = %peer, "copy .ecj: {}", e);
let _ = fs::remove_file(ecj_path);
}
}
Err(e) => tracing::warn!(volume_id = m.vid.0, peer = %peer, "copy .ecj: {}", e),
@@ -1817,7 +1540,7 @@ async fn fetch_ec_index_from_one_peer(
match client.copy_file(copy_req(".vif", true)).await {
Ok(resp) => {
if let Err(e) = drain_copy_stream(resp.into_inner(), vif_path).await {
if let Err(e) = drain_copy_stream(resp.into_inner(), vif_path, false).await {
tracing::warn!(volume_id = m.vid.0, peer = %peer, "copy .vif: {}", e);
}
}
@@ -1827,14 +1550,22 @@ async fn fetch_ec_index_from_one_peer(
Ok(())
}
/// Drain a CopyFile stream into a local file, truncating it first.
/// Drain a CopyFile stream into a local file, appending or truncating.
async fn drain_copy_stream(
mut stream: tonic::Streaming<crate::pb::volume_server_pb::CopyFileResponse>,
dest_path: &str,
append: bool,
) -> io::Result<()> {
use std::io::Write;
let mut file = fs::File::create(dest_path)
.map_err(|e| io::Error::other(format!("create {}: {}", dest_path, e)))?;
let mut file = if append {
fs::OpenOptions::new()
.create(true)
.append(true)
.open(dest_path)
} else {
fs::File::create(dest_path)
}
.map_err(|e| io::Error::other(format!("create {}: {}", dest_path, e)))?;
while let Some(chunk) = stream
.message()
.await
@@ -1850,21 +1581,6 @@ async fn drain_copy_stream(
mod tests {
use super::*;
#[test]
fn gather_intervals_puts_a_reported_deletion_ahead_of_errors() {
let failed = || Err(io::Error::other("shard unreachable"));
assert!(
gather_intervals(vec![failed(), Ok((Vec::new(), true))])
.unwrap()
.is_none()
);
assert!(gather_intervals(vec![failed(), Ok((vec![1], false))]).is_err());
assert_eq!(
gather_intervals(vec![Ok((vec![1], false)), Ok((vec![2], false))]).unwrap(),
Some(vec![vec![1], vec![2]])
);
}
fn locations(count: usize) -> HashMap<ShardId, Vec<String>> {
(0..count)
.map(|sid| (sid as ShardId, vec!["127.0.0.1:8080".to_string()]))
+10 -1
View File
@@ -1,6 +1,5 @@
use std::fmt::Write as _;
use crate::server::absolute_display_path;
use crate::server::server_stats;
use crate::server::volume_server::VolumeServerState;
use crate::storage::store::Store;
@@ -451,6 +450,16 @@ fn collect_ui_data(
(disk_rows, volumes, remote_volumes, ec_volumes)
}
fn absolute_display_path(path: &str) -> String {
let p = std::path::Path::new(path);
if p.is_absolute() {
return path.to_string();
}
std::env::current_dir()
.map(|cwd| cwd.join(p).to_string_lossy().to_string())
.unwrap_or_else(|_| path.to_string())
}
fn join_i64(values: &[i64]) -> String {
values
.iter()
+17 -31
View File
@@ -14,12 +14,12 @@ use std::sync::atomic::{AtomicBool, AtomicI64, AtomicU32, Ordering};
use std::sync::{Arc, RwLock};
use axum::{
Router,
extract::{Request, State, connect_info::ConnectInfo},
http::{HeaderValue, Method, StatusCode, header},
extract::{connect_info::ConnectInfo, Request, State},
http::{header, HeaderValue, Method, StatusCode},
middleware::{self, Next},
response::{IntoResponse, Response},
routing::{any, get},
Router,
};
use crate::config::ReadMode;
@@ -73,6 +73,8 @@ pub struct VolumeServerState {
pub volume_state_notify: tokio::sync::Notify,
/// Optional batched write queue for improved throughput under load.
pub write_queue: std::sync::OnceLock<WriteQueue>,
/// Registry of S3 tier backends for tiered storage operations.
pub s3_tier_registry: std::sync::RwLock<crate::remote_storage::s3_tier::S3TierRegistry>,
/// Read mode: local, proxy, or redirect for non-local volumes.
pub read_mode: ReadMode,
/// If true, FetchAndWriteNeedle skips remote S3 endpoint validation,
@@ -114,21 +116,6 @@ pub struct VolumeServerState {
pub cli_white_list: Vec<String>,
/// Path to state.pb file for persisting VolumeServerState across restarts.
pub state_file_path: String,
/// Volumes with an EC decode in flight. A dropped request leaves the
/// blocking job running; this keeps a retry from racing it on the
/// same volume files.
pub ec_decodes_in_flight:
std::sync::Mutex<std::collections::HashSet<crate::storage::types::VolumeId>>,
/// Volumes whose EC decode is in its publishing tail (journal catch-up,
/// .idx write, compaction). Local .ecj appenders wait on
/// `ec_decode_tail_notify` while their vid is listed, so no committed
/// delete falls between the last catch_up and the .cpd/.cpx swap —
/// the per-volume slice of Go's EcVolume.ecjFileAccessLock.
pub ec_decode_tail:
std::sync::Mutex<std::collections::HashSet<crate::storage::types::VolumeId>>,
/// Wakes .ecj appenders waiting on `ec_decode_tail` when a decode's
/// publishing tail ends.
pub ec_decode_tail_notify: tokio::sync::Notify,
}
impl VolumeServerState {
@@ -213,7 +200,9 @@ pub fn to_http_address(addr: &str) -> std::borrow::Cow<'_, str> {
// rather than being silently rewritten. Mirrors the validation already
// done in `to_grpc_address` for the inverse direction.
if let (Ok(_), Ok(_)) = (http_port.parse::<u16>(), grpc_port.parse::<u16>()) {
return std::borrow::Cow::Owned(addr[..ports_sep_index + 1 + dot_idx].to_string());
return std::borrow::Cow::Owned(
addr[..ports_sep_index + 1 + dot_idx].to_string(),
);
}
}
std::borrow::Cow::Borrowed(addr)
@@ -441,13 +430,13 @@ pub fn build_admin_router_with_ui(state: Arc<VolumeServerState>, ui_enabled: boo
.route("/healthz", get(handlers::healthz_handler))
.route("/favicon.ico", get(handlers::favicon_handler))
.route(
"/seaweedfsstatic/{*path}",
"/seaweedfsstatic/*path",
get(handlers::static_asset_handler),
)
.route("/", any(admin_store_handler))
.route("/{path}", any(admin_store_handler))
.route("/{vid}/{fid}", any(admin_store_handler))
.route("/{vid}/{fid}/{filename}", any(admin_store_handler))
.route("/:path", any(admin_store_handler))
.route("/:vid/:fid", any(admin_store_handler))
.route("/:vid/:fid/:filename", any(admin_store_handler))
.fallback(admin_store_handler);
if ui_enabled {
// Note: /stats/* endpoints are commented out in Go's volume_server.go (L130-134).
@@ -464,13 +453,13 @@ pub fn build_public_router(state: Arc<VolumeServerState>) -> Router {
Router::new()
.route("/favicon.ico", get(handlers::favicon_handler))
.route(
"/seaweedfsstatic/{*path}",
"/seaweedfsstatic/*path",
get(handlers::static_asset_handler),
)
.route("/", any(public_store_handler))
.route("/{path}", any(public_store_handler))
.route("/{vid}/{fid}", any(public_store_handler))
.route("/{vid}/{fid}/{filename}", any(public_store_handler))
.route("/:path", any(public_store_handler))
.route("/:vid/:fid", any(public_store_handler))
.route("/:vid/:fid/:filename", any(public_store_handler))
.fallback(public_store_handler)
.layer(middleware::from_fn(common_headers_middleware))
.with_state(state)
@@ -525,10 +514,7 @@ mod tests {
// "host:abc.def"), and silently rewriting it would just hide the bug.
assert_eq!(to_http_address("host:abc.def"), "host:abc.def");
assert_eq!(to_http_address("host:9333.notaport"), "host:9333.notaport");
assert_eq!(
to_http_address("host:notaport.19333"),
"host:notaport.19333"
);
assert_eq!(to_http_address("host:notaport.19333"), "host:notaport.19333");
// Out-of-range ports must not be silently truncated either.
assert_eq!(to_http_address("host:99999.19333"), "host:99999.19333");
}
+8 -69
View File
@@ -3,8 +3,8 @@
//! Instead of each upload handler directly calling `write_needle`, writes are
//! submitted to a queue. A background worker drains the queue in batches (up to
//! 128 entries), groups them by volume ID, and processes them together under a
//! single store lock. Durable writes to a volume share their .dat and .idx
//! flushes (see `Volume::write_needles_grouped`).
//! single store lock. Requests that asked for `fsync` are flushed by
//! `write_needle` itself, one flush per durable write.
use std::sync::Arc;
@@ -159,12 +159,8 @@ fn process_batch(state: Arc<VolumeServerState>, batch: Vec<WriteRequest>) {
let mut store = state.store.write().unwrap();
for (vid, entries) in groups {
let (mut writes, senders): (Vec<_>, Vec<_>) = entries
.into_iter()
.map(|(needle, fsync, response_tx)| ((needle, fsync), response_tx))
.unzip();
let results = store.write_volume_needles(vid, &mut writes);
for (response_tx, result) in senders.into_iter().zip(results) {
for (mut needle, fsync, response_tx) in entries {
let result = store.write_volume_needle(vid, &mut needle, fsync);
// Send result back; ignore error if receiver dropped.
let _ = response_tx.send(result);
}
@@ -182,8 +178,8 @@ mod tests {
use crate::server::volume_server::RuntimeMetricsConfig;
use crate::storage::needle_map::NeedleMapKind;
use crate::storage::store::Store;
use std::sync::RwLock;
use std::sync::atomic::{AtomicBool, AtomicI64, AtomicU32};
use std::sync::RwLock;
let store = Store::new(NeedleMapKind::InMemory);
let guard = Guard::new(&[], SigningKey(vec![]), 0, SigningKey(vec![]), 0);
@@ -211,6 +207,9 @@ mod tests {
pre_stop_seconds: 0,
volume_state_notify: tokio::sync::Notify::new(),
write_queue: std::sync::OnceLock::new(),
s3_tier_registry: std::sync::RwLock::new(
crate::remote_storage::s3_tier::S3TierRegistry::new(),
),
read_mode: crate::config::ReadMode::Local,
allow_untrusted_remote_endpoints: false,
master_url: String::new(),
@@ -229,9 +228,6 @@ mod tests {
security_file: String::new(),
cli_white_list: vec![],
state_file_path: String::new(),
ec_decodes_in_flight: std::sync::Mutex::new(std::collections::HashSet::new()),
ec_decode_tail: std::sync::Mutex::new(std::collections::HashSet::new()),
ec_decode_tail_notify: tokio::sync::Notify::new(),
})
}
@@ -319,63 +315,6 @@ mod tests {
}
}
/// The queue hands a volume's batch to the grouped path, so ten durable
/// writes cost one .dat sync and one .idx sync, not ten of each.
#[test]
fn test_process_batch_group_commits_fsync_writes() {
use crate::config::MinFreeSpace;
use crate::storage::types::{DiskType, NeedleId};
use crate::storage::volume::VolumeSpec;
let tmp = tempfile::TempDir::new().unwrap();
let dir = tmp.path().to_str().unwrap();
let state = make_test_state();
{
let mut store = state.store.write().unwrap();
store
.add_location(
dir,
dir,
10,
DiskType::HardDrive,
MinFreeSpace::Percent(1.0),
Vec::new(),
)
.unwrap();
store
.add_volume(VolumeId(1), DiskType::HardDrive, &VolumeSpec::default())
.unwrap();
}
let mut receivers = Vec::new();
let batch = (1..=10u64)
.map(|id| {
let (response_tx, response_rx) = oneshot::channel();
receivers.push(response_rx);
WriteRequest {
volume_id: VolumeId(1),
needle: Needle {
id: NeedleId(id),
cookie: 0x1111.into(),
data: vec![id as u8; 8],
data_size: 8,
..Needle::default()
},
fsync: true,
response_tx,
}
})
.collect();
process_batch(state.clone(), batch);
for mut rx in receivers {
assert!(matches!(rx.try_recv().unwrap(), Ok((_, _, false))));
}
let store = state.store.read().unwrap();
let (_, vol) = store.find_volume(VolumeId(1)).unwrap();
assert_eq!(vol.sync_counts_for_test(), (1, 1));
}
#[tokio::test]
async fn test_write_queue_dropped_sender() {
// When the queue is dropped, subsequent submits should fail gracefully.
+173 -307
View File
@@ -15,17 +15,15 @@ use tracing::warn;
use crate::config::MinFreeSpace;
use crate::storage::erasure_coding::ec_bitrot::remove_bitrot_sidecars;
use crate::storage::erasure_coding::ec_shard::{
DATA_SHARDS_COUNT, ERASURE_CODING_LARGE_BLOCK_SIZE, ERASURE_CODING_SMALL_BLOCK_SIZE,
EcVolumeShard, ShardId,
};
use crate::storage::erasure_coding::ec_volume::{
ECJ_COMPACT_TMP_EXT, EcVolume, is_usable_ecx_file,
EcVolumeShard, DATA_SHARDS_COUNT, ERASURE_CODING_LARGE_BLOCK_SIZE,
ERASURE_CODING_SMALL_BLOCK_SIZE,
};
use crate::storage::erasure_coding::ec_volume::EcVolume;
use crate::storage::needle_map::NeedleMapKind;
use crate::storage::super_block::SUPER_BLOCK_SIZE;
use crate::storage::super_block::{ReplicaPlacement, SUPER_BLOCK_SIZE};
use crate::storage::types::*;
use crate::storage::volume::{
VifVolumeInfo, Volume, VolumeError, VolumeSpec, remove_volume_files, volume_file_name,
remove_volume_files, volume_file_name, VifVolumeInfo, Volume, VolumeError,
};
/// A single disk location managing volumes in one directory.
@@ -207,6 +205,7 @@ impl DiskLocation {
continue;
}
// Load existing data only; never create a phantom `.dat`. A lone
// `.vif`/`.idx` (e.g. an EC sidecar whose `.ecx` is on a sibling
// disk) would otherwise have Volume::new write an 8-byte stub that
@@ -281,33 +280,30 @@ impl DiskLocation {
let opened = Mutex::new(Vec::with_capacity(to_load.len()));
std::thread::scope(|scope| {
for _ in 0..workers {
scope.spawn(|| {
loop {
let i = next.fetch_add(1, Ordering::Relaxed);
let Some((vid, collections)) = to_load.get(i) else {
return;
};
for collection in collections {
// Replica placement and TTL are read back from the
// superblock, and a load never preallocates.
match Volume::new(
&self.directory,
&self.idx_directory,
*vid,
needle_map_kind,
&VolumeSpec {
collection,
..Default::default()
},
) {
Ok(mut v) => {
v.location_disk_space_low = self.is_disk_space_low.clone();
opened.lock().unwrap().push((collection.clone(), *vid, v));
break;
}
Err(e) => {
warn!(volume_id = vid.0, error = %e, "failed to load volume");
}
scope.spawn(|| loop {
let i = next.fetch_add(1, Ordering::Relaxed);
let Some((vid, collections)) = to_load.get(i) else {
return;
};
for collection in collections {
match Volume::new(
&self.directory,
&self.idx_directory,
collection,
*vid,
needle_map_kind,
None, // replica placement read from superblock
None, // TTL read from superblock
0, // no preallocate on load
Version::current(),
) {
Ok(mut v) => {
v.location_disk_space_low = self.is_disk_space_low.clone();
opened.lock().unwrap().push((collection.clone(), *vid, v));
break;
}
Err(e) => {
warn!(volume_id = vid.0, error = %e, "failed to load volume");
}
}
}
@@ -378,10 +374,8 @@ impl DiskLocation {
let mut expected_shard_size: Option<i64> = None;
let dat_exists = match fs::metadata(&dat_path) {
Ok(meta) if meta.len() > SUPER_BLOCK_SIZE as u64 => {
expected_shard_size = Some(calculate_expected_shard_size(
meta.len() as i64,
data_shards,
));
expected_shard_size =
Some(calculate_expected_shard_size(meta.len() as i64, data_shards));
true
}
Ok(_) => false,
@@ -405,13 +399,7 @@ impl DiskLocation {
if size != prev {
// Inconsistent sizes signal corruption or mixed
// generations; not trusted for deletion -> keep.
warn!(
volume_id = vid.0,
shard = i,
size,
expected = prev,
"EC shard size mismatch; keeping shards"
);
warn!(volume_id = vid.0, shard = i, size, expected = prev, "EC shard size mismatch; keeping shards");
return true;
}
} else {
@@ -462,16 +450,13 @@ impl DiskLocation {
let idx_base = volume_file_name(&self.idx_directory, collection, vid);
const MAX_SHARD_COUNT: usize = 32;
// Remove index files from idx directory (.ecx, .ecj, and a compaction
// tmp a crash may have left beside the .ecj)
// Remove index files from idx directory (.ecx, .ecj)
rm_if_present(format!("{}.ecx", idx_base))?;
rm_if_present(format!("{}.ecj", idx_base))?;
rm_if_present(format!("{}{}", idx_base, ECJ_COMPACT_TMP_EXT))?;
// Also try data directory in case .ecx/.ecj were created before -dir.idx was configured
if self.idx_directory != self.directory {
rm_if_present(format!("{}.ecx", base))?;
rm_if_present(format!("{}.ecj", base))?;
rm_if_present(format!("{}{}", base, ECJ_COMPACT_TMP_EXT))?;
}
// Remove all EC shard files (.ec00 ~ .ec31)
@@ -484,13 +469,6 @@ impl DiskLocation {
if self.idx_directory != self.directory {
remove_bitrot_sidecars(&idx_base)?;
}
// Staged 2PC generations (<base>.ecNN.v<N>, versioned .ecx/.ecj/.vif)
// belong to this volume's EC state too; leaving them orphans the files.
crate::storage::erasure_coding::ec_shard::remove_ec_generation_files(&base, 0)?;
if self.idx_directory != self.directory {
crate::storage::erasure_coding::ec_shard::remove_ec_generation_files(&idx_base, 0)?;
}
Ok(())
}
@@ -569,22 +547,31 @@ impl DiskLocation {
}
/// Create a new volume in this location.
#[expect(clippy::too_many_arguments)]
pub fn create_volume(
&mut self,
vid: VolumeId,
collection: &str,
needle_map_kind: NeedleMapKind,
spec: &VolumeSpec<'_>,
replica_placement: Option<ReplicaPlacement>,
ttl: Option<crate::storage::needle::ttl::TTL>,
preallocate: u64,
version: Version,
) -> Result<(), VolumeError> {
let mut v = Volume::new(
&self.directory,
&self.idx_directory,
collection,
vid,
needle_map_kind,
spec,
replica_placement,
ttl,
preallocate,
version,
)?;
v.location_disk_space_low = self.is_disk_space_low.clone();
crate::metrics::VOLUME_GAUGE
.with_label_values(&[spec.collection, "volume"])
.with_label_values(&[collection, "volume"])
.inc();
self.volumes.insert(vid, v);
Ok(())
@@ -610,20 +597,13 @@ impl DiskLocation {
&mut self,
vid: VolumeId,
only_empty: bool,
only_garbage: bool,
keep_remote_data: bool,
) -> Result<(), VolumeError> {
// Refuse before removing: a refused destroy must leave it mounted.
if let Some(v) = self.volumes.get(&vid)
&& v.is_compacting()
{
return Err(v.compacting_error());
}
if let Some(mut v) = self.volumes.remove(&vid) {
crate::metrics::VOLUME_GAUGE
.with_label_values(&[&v.collection, "volume"])
.dec();
v.destroy(only_empty, only_garbage, keep_remote_data)?;
v.destroy(only_empty, keep_remote_data)?;
Ok(())
} else {
Err(VolumeError::NotFound)
@@ -644,7 +624,7 @@ impl DiskLocation {
crate::metrics::VOLUME_GAUGE
.with_label_values(&[&v.collection, "volume"])
.dec();
if let Err(e) = v.destroy(false, false, false) {
if let Err(e) = v.destroy(false, false) {
warn!(volume_id = vid.0, error = %e, "delete collection: failed to destroy volume");
}
}
@@ -695,7 +675,8 @@ impl DiskLocation {
pub fn free_volume_count(&self) -> i32 {
use crate::storage::erasure_coding::ec_shard::DATA_SHARDS_COUNT;
let max = self.max_volume_count.load(Ordering::Relaxed);
let free_count = (max as i64 - self.volumes.len() as i64) * DATA_SHARDS_COUNT as i64
let free_count = (max as i64 - self.volumes.len() as i64)
* DATA_SHARDS_COUNT as i64
- self.ec_shard_count() as i64;
let effective_free = free_count / DATA_SHARDS_COUNT as i64;
if effective_free > 0 {
@@ -798,17 +779,21 @@ impl DiskLocation {
/// Mirrors `DiskLocation.HasEcxFileOnDisk` in
/// `weed/storage/disk_location_ec.go`. Skips entries that are
/// directories so a stray dir named `<collection>_<vid>.ecx` doesn't
/// register as a present index file. A 0-byte `.ecx` is a corrupt stub
/// left by a failed EC distribute copy; it must not steer placement
/// toward this disk, so it counts as absent (Go requires `Size() > 0`).
/// register as a present index file.
pub fn has_ecx_file_on_disk(&self, collection: &str, vid: VolumeId) -> bool {
let idx_base = volume_file_name(&self.idx_directory, collection, vid);
if is_usable_ecx_file(&format!("{}.ecx", idx_base)) {
let idx_path = format!("{}.ecx", idx_base);
if let Ok(meta) = fs::metadata(&idx_path)
&& !meta.is_dir()
{
return true;
}
if self.idx_directory != self.directory {
let data_base = volume_file_name(&self.directory, collection, vid);
if is_usable_ecx_file(&format!("{}.ecx", data_base)) {
let data_path = format!("{}.ecx", data_base);
if let Ok(meta) = fs::metadata(&data_path)
&& !meta.is_dir()
{
return true;
}
}
@@ -820,20 +805,6 @@ impl DiskLocation {
self.ec_volumes.remove(&vid)
}
/// Drop the in-memory EC volume for vid and close its descriptors without
/// deleting files, so a following unlink frees the inodes instead of
/// leaving open fds serving the old bytes. Mirrors Go's unloadEcVolume.
pub fn unload_ec_volume(&mut self, vid: VolumeId) {
if let Some(mut ec_vol) = self.ec_volumes.remove(&vid) {
for _ in 0..ec_vol.shard_count() {
crate::metrics::VOLUME_GAUGE
.with_label_values(&[&ec_vol.collection, "ec_shards"])
.dec();
}
ec_vol.close();
}
}
/// Mount EC shards for a volume on this location.
///
/// `source_disk_type` is the source volume's disk type carried on the
@@ -846,7 +817,7 @@ impl DiskLocation {
&mut self,
vid: VolumeId,
collection: &str,
shard_ids: &[ShardId],
shard_ids: &[u32],
source_disk_type: &str,
) -> Result<(), VolumeError> {
let idx_dir = self.idx_directory.clone();
@@ -868,7 +839,7 @@ impl DiskLocation {
&mut self,
vid: VolumeId,
collection: &str,
shard_ids: &[ShardId],
shard_ids: &[u32],
idx_dir: &str,
source_disk_type: &str,
) -> Result<(), VolumeError> {
@@ -880,10 +851,14 @@ impl DiskLocation {
// propagate the error to the caller.
let created = !self.ec_volumes.contains_key(&vid);
if created {
let ec_vol = EcVolume::new(&dir, idx_dir, collection, vid).map_err(VolumeError::Io)?;
let ec_vol = EcVolume::new(&dir, idx_dir, collection, vid)
.map_err(VolumeError::Io)?;
self.ec_volumes.insert(vid, ec_vol);
}
let ec_vol = self.ec_volumes.get_mut(&vid).expect("just inserted above");
let ec_vol = self
.ec_volumes
.get_mut(&vid)
.expect("just inserted above");
// When the orchestrator supplied a source disk type on the Mount
// RPC, override the EC volume's disk type so heartbeats report
// under the source volume's disk type (#9423). When the caller
@@ -902,10 +877,10 @@ impl DiskLocation {
// keep the existing registration (mirrors Go's AddEcVolumeShard
// added=false) — re-adding would replace a serving fd and bump
// the ec_shards gauge without growing the mounted count.
if ec_vol.has_shard(shard_id) {
if ec_vol.has_shard(shard_id as u8) {
continue;
}
let mut shard = EcVolumeShard::new(&dir, collection, vid, shard_id);
let mut shard = EcVolumeShard::new(&dir, collection, vid, shard_id as u8);
shard.disk_type = ec_vol.disk_type.clone();
if let Err(e) = ec_vol.add_shard(shard) {
// The shard was dropped (its descriptors closed) inside the
@@ -933,14 +908,14 @@ impl DiskLocation {
/// caller passes a shard that lives on a sibling disk
/// (cross-disk reconcile makes that the common case for the same
/// `vid` after reconciliation).
pub fn unmount_ec_shards(&mut self, vid: VolumeId, shard_ids: &[ShardId]) {
pub fn unmount_ec_shards(&mut self, vid: VolumeId, shard_ids: &[u32]) {
if let Some(ec_vol) = self.ec_volumes.get_mut(&vid) {
let collection = ec_vol.collection.clone();
for &shard_id in shard_ids {
if !ec_vol.has_shard(shard_id) {
if !ec_vol.has_shard(shard_id as u8) {
continue;
}
let _ = ec_vol.remove_shard(shard_id);
ec_vol.remove_shard(shard_id as u8);
crate::metrics::VOLUME_GAUGE
.with_label_values(&[&collection, "ec_shards"])
.dec();
@@ -1000,7 +975,7 @@ impl DiskLocation {
}
entries.sort();
let mut same_volume_shards: Vec<(String, ShardId)> = Vec::new(); // (filename, shard_id)
let mut same_volume_shards: Vec<(String, u32)> = Vec::new(); // (filename, shard_id)
let mut prev_vid: Option<VolumeId> = None;
let mut prev_collection: String = String::new();
@@ -1065,12 +1040,7 @@ impl DiskLocation {
/// Validate + mount a (collection, vid) group when its `.ecx` is
/// found. Mirrors `handleFoundEcxFile` in
/// `weed/storage/disk_location_ec.go`.
fn handle_found_ecx_file(
&mut self,
shards: &[(String, ShardId)],
collection: &str,
vid: VolumeId,
) {
fn handle_found_ecx_file(&mut self, shards: &[(String, u32)], collection: &str, vid: VolumeId) {
let base = volume_file_name(&self.directory, collection, vid);
let dat_path = format!("{}.dat", base);
let dat_exists = check_dat_file_exists(&dat_path);
@@ -1084,7 +1054,7 @@ impl DiskLocation {
return;
}
let shard_ids: Vec<ShardId> = shards.iter().map(|(_, sid)| *sid).collect();
let shard_ids: Vec<u32> = shards.iter().map(|(_, sid)| *sid).collect();
if let Err(e) = self.mount_ec_shards(vid, collection, &shard_ids, "") {
// A mount failure (corrupt/locked .ecx, EMFILE, transient I/O) is
// not proof the shards are disposable -- validate_ec_volume already
@@ -1093,7 +1063,8 @@ impl DiskLocation {
// delete on a load error.
warn!(
volume_id = vid.0,
"Failed to load EC shards: {}; keeping files for retry", e,
"Failed to load EC shards: {}; keeping files for retry",
e,
);
self.unmount_ec_shards(vid, &shard_ids);
}
@@ -1106,7 +1077,7 @@ impl DiskLocation {
/// distributed-EC shards waiting for cross-disk reconciliation.
fn check_orphaned_shards(
&self,
shards: &[(String, ShardId)],
shards: &[(String, u32)],
collection: &str,
vid: VolumeId,
) -> bool {
@@ -1162,60 +1133,20 @@ pub fn get_disk_stats(path: &str) -> (u64, u64) {
Ok(p) => p,
Err(_) => return (0, 0),
};
// SAFETY: `libc::statvfs` is plain data — integers and reserved
// padding, no pointers and no restricted niches — so the all-zero
// value is a valid one for the call to overwrite.
let mut stat: libc::statvfs = unsafe { std::mem::zeroed() };
// SAFETY: `c_path` is a live NUL-terminated `CString` that outlives
// the call, and `&mut stat` is a live, aligned, exclusive pointer the
// kernel only writes through; the fields are read below only after
// the call reports success.
if unsafe { libc::statvfs(c_path.as_ptr(), &mut stat) } != 0 {
return (0, 0);
}
let all = stat.f_blocks as u64 * stat.f_frsize as u64;
let free = stat.f_bavail as u64 * stat.f_frsize as u64;
(all, free)
}
#[cfg(windows)]
{
use std::os::windows::ffi::OsStrExt;
// Canonicalize so symlinks, `.`/`..` segments, and relative paths
// resolve to the real location before querying. `\\?\`-prefixed
// extended-length paths and UNC (`\\?\UNC\...`) are passed through
// untouched: GetDiskFreeSpaceExW accepts them as-is.
let canonical = match std::fs::canonicalize(path) {
Ok(p) => p,
Err(_) => return (0, 0),
};
// UTF-16 with trailing NUL for the Win32 wide-string call.
let mut wide: Vec<u16> = canonical.as_os_str().encode_wide().collect();
// UNC directory names must end in a backslash for GetDiskFreeSpaceExW.
if !wide.ends_with(&[0x5C]) {
wide.push(0x5C);
}
wide.push(0);
// SAFETY: `wide` is NUL-terminated; the out-params are valid u64
// writes; the call has no other preconditions.
unsafe {
let mut free_available: u64 = 0;
let mut total: u64 = 0;
let ok = windows_sys::Win32::Storage::FileSystem::GetDiskFreeSpaceExW(
wide.as_ptr(),
&mut free_available,
&mut total,
std::ptr::null_mut(),
);
if ok == 0 {
return (0, 0);
let mut stat: libc::statvfs = std::mem::zeroed();
if libc::statvfs(c_path.as_ptr(), &mut stat) == 0 {
let all = stat.f_blocks as u64 * stat.f_frsize as u64;
let free = stat.f_bavail as u64 * stat.f_frsize as u64;
return (all, free);
}
return (total, free_available);
}
(0, 0)
}
#[cfg(not(any(unix, windows)))]
#[cfg(not(unix))]
{
compile_error!("get_disk_stats is implemented for unix and windows only");
let _ = path;
(0, 0)
}
}
@@ -1251,12 +1182,7 @@ fn rm_if_present(path: String) -> io::Result<()> {
}
}
fn ec_data_shards_from_vif(
directory: &str,
idx_directory: &str,
collection: &str,
vid: VolumeId,
) -> usize {
fn ec_data_shards_from_vif(directory: &str, idx_directory: &str, collection: &str, vid: VolumeId) -> usize {
for dir in [directory, idx_directory] {
let vif = format!("{}.vif", volume_file_name(dir, collection, vid));
if let Some(ds) = fs::read_to_string(&vif)
@@ -1302,7 +1228,7 @@ fn parse_collection_volume_id(base: &str) -> Option<(String, VolumeId)> {
/// `pub(crate)` re-export of [`parse_ec_shard_extension`] for the
/// cross-disk reconcile in `store_ec_reconcile.rs`.
pub(crate) fn is_ec_shard_extension(ext: &str) -> Option<ShardId> {
pub(crate) fn is_ec_shard_extension(ext: &str) -> Option<u32> {
parse_ec_shard_extension(ext)
}
@@ -1316,7 +1242,7 @@ pub(crate) fn is_ec_shard_extension(ext: &str) -> Option<ShardId> {
/// shardId > 255` guard. The 3-digit form (`.ec100`–`.ec255`) is
/// retained so the parser can still recognise shards from custom
/// 32+ ratios that fit in a u8 even though OSS only ships 10+4.
fn parse_ec_shard_extension(ext: &str) -> Option<ShardId> {
fn parse_ec_shard_extension(ext: &str) -> Option<u32> {
let rest = ext.strip_prefix(".ec")?;
if rest.len() < 2 || rest.len() > 3 {
return None;
@@ -1325,7 +1251,7 @@ fn parse_ec_shard_extension(ext: &str) -> Option<ShardId> {
if id > 255 {
return None;
}
ShardId::try_from(id).ok()
Some(id)
}
/// Robust check that a `.dat` with actual data exists. An empty `.dat`
@@ -1387,10 +1313,7 @@ fn remove_empty_ec_dat_stub(volume_name: &str, idx_name: &str, vid: VolumeId) ->
return false;
}
warn!(
volume_id = vid.0,
"removing leftover empty .dat stub for EC volume"
);
warn!(volume_id = vid.0, "removing leftover empty .dat stub for EC volume");
let _ = fs::remove_file(&dat_path);
let _ = fs::remove_file(format!("{}.idx", idx_name));
true
@@ -1413,17 +1336,6 @@ mod tests {
use super::*;
use tempfile::TempDir;
/// get_disk_stats must report real capacity for a real path on every
/// platform (Windows included) — consumers treat total==0 as "unknown"
/// and leave available_space at 0, which breaks volume assignment.
#[test]
fn test_get_disk_stats_reports_capacity_for_real_path() {
let tmp = TempDir::new().unwrap();
let (total, free) = get_disk_stats(tmp.path().to_str().unwrap());
assert!(total > 0, "expected total>0, got {total}");
assert!(free > 0, "expected free>0, got {free}");
}
/// When `-dir.idx` is configured the EC `.vif` may live in the idx
/// directory; the sweep must look there too, not only the data dir.
#[test]
@@ -1446,11 +1358,7 @@ mod tests {
}),
..Default::default()
};
std::fs::write(
format!("{}.vif", ibase),
serde_json::to_string(&vif).unwrap(),
)
.unwrap();
std::fs::write(format!("{}.vif", ibase), serde_json::to_string(&vif).unwrap()).unwrap();
assert!(
remove_empty_ec_dat_stub(&vbase, &ibase, VolumeId(42)),
@@ -1466,30 +1374,16 @@ mod tests {
fn test_validate_ec_volume_partial_dat_next_to_full_shards_keeps() {
let tmp = TempDir::new().unwrap();
let dir = tmp.path().to_str().unwrap();
let loc = DiskLocation::new(
dir,
dir,
10,
DiskType::HardDrive,
MinFreeSpace::Percent(1.0),
Vec::new(),
)
.unwrap();
let loc = DiskLocation::new(dir, dir, 10, DiskType::HardDrive, MinFreeSpace::Percent(1.0), Vec::new()).unwrap();
let base = volume_file_name(dir, "", VolumeId(70));
let ds = crate::storage::erasure_coding::ec_shard::DATA_SHARDS_COUNT;
let full = calculate_expected_shard_size(30 * 1024 * 1024, ds);
for i in 0..ds {
std::fs::File::create(format!("{}.ec{:02}", base, i))
.unwrap()
.set_len(full as u64)
.unwrap();
std::fs::File::create(format!("{}.ec{:02}", base, i)).unwrap().set_len(full as u64).unwrap();
}
// Partial .dat: bigger than a superblock so it is not swept as a stub,
// but smaller than what these shards encode.
std::fs::File::create(format!("{}.dat", base))
.unwrap()
.set_len(5 * 1024 * 1024)
.unwrap();
std::fs::File::create(format!("{}.dat", base)).unwrap().set_len(5 * 1024 * 1024).unwrap();
assert!(
loc.validate_ec_volume("", VolumeId(70)),
"full-size shards beside a smaller (stale/partial) .dat must be kept",
@@ -1503,29 +1397,15 @@ mod tests {
fn test_validate_ec_volume_interrupted_encode_reclaims() {
let tmp = TempDir::new().unwrap();
let dir = tmp.path().to_str().unwrap();
let loc = DiskLocation::new(
dir,
dir,
10,
DiskType::HardDrive,
MinFreeSpace::Percent(1.0),
Vec::new(),
)
.unwrap();
let loc = DiskLocation::new(dir, dir, 10, DiskType::HardDrive, MinFreeSpace::Percent(1.0), Vec::new()).unwrap();
let base = volume_file_name(dir, "", VolumeId(71));
let ds = crate::storage::erasure_coding::ec_shard::DATA_SHARDS_COUNT;
let dat_size = 30 * 1024 * 1024i64;
std::fs::File::create(format!("{}.dat", base))
.unwrap()
.set_len(dat_size as u64)
.unwrap();
std::fs::File::create(format!("{}.dat", base)).unwrap().set_len(dat_size as u64).unwrap();
let partial = calculate_expected_shard_size(dat_size, ds) / 3;
assert!(partial > 0);
for i in 0..ds {
std::fs::File::create(format!("{}.ec{:02}", base, i))
.unwrap()
.set_len(partial as u64)
.unwrap();
std::fs::File::create(format!("{}.ec{:02}", base, i)).unwrap().set_len(partial as u64).unwrap();
}
assert!(
!loc.validate_ec_volume("", VolumeId(71)),
@@ -1570,11 +1450,7 @@ mod tests {
}),
..Default::default()
};
std::fs::write(
format!("{}.vif", dbase),
serde_json::to_string(&with_gen).unwrap(),
)
.unwrap();
std::fs::write(format!("{}.vif", dbase), serde_json::to_string(&with_gen).unwrap()).unwrap();
assert_eq!(loc.ec_generation_ts_ns("", vid), Some(4242));
// A .vif with no EC config reads as generation 0 (recovered/pre-upgrade live volume).
@@ -1583,20 +1459,12 @@ mod tests {
version: 3,
..Default::default()
};
std::fs::write(
format!("{}.vif", dbase),
serde_json::to_string(&no_cfg).unwrap(),
)
.unwrap();
std::fs::write(format!("{}.vif", dbase), serde_json::to_string(&no_cfg).unwrap()).unwrap();
assert_eq!(loc.ec_generation_ts_ns("", vid), Some(0));
// idx-dir fallback: only the idx dir holds the .vif.
std::fs::remove_file(format!("{}.vif", dbase)).unwrap();
std::fs::write(
format!("{}.vif", ibase),
serde_json::to_string(&with_gen).unwrap(),
)
.unwrap();
std::fs::write(format!("{}.vif", ibase), serde_json::to_string(&with_gen).unwrap()).unwrap();
assert_eq!(loc.ec_generation_ts_ns("", vid), Some(4242));
}
@@ -1636,8 +1504,16 @@ mod tests {
)
.unwrap();
loc.create_volume(VolumeId(1), NeedleMapKind::InMemory, &VolumeSpec::default())
.unwrap();
loc.create_volume(
VolumeId(1),
"",
NeedleMapKind::InMemory,
None,
None,
0,
Version::current(),
)
.unwrap();
assert_eq!(loc.volumes_len(), 1);
assert!(loc.find_volume(VolumeId(1)).is_some());
@@ -1661,15 +1537,24 @@ mod tests {
Vec::new(),
)
.unwrap();
loc.create_volume(VolumeId(1), NeedleMapKind::InMemory, &VolumeSpec::default())
.unwrap();
loc.create_volume(
VolumeId(1),
"",
NeedleMapKind::InMemory,
None,
None,
0,
Version::current(),
)
.unwrap();
loc.create_volume(
VolumeId(2),
"test",
NeedleMapKind::InMemory,
&VolumeSpec {
collection: "test",
..Default::default()
},
None,
None,
0,
Version::current(),
)
.unwrap();
loc.close();
@@ -1712,11 +1597,12 @@ mod tests {
.unwrap();
loc.create_volume(
VolumeId(9),
"good",
NeedleMapKind::InMemory,
&VolumeSpec {
collection: "good",
..Default::default()
},
None,
None,
0,
Version::current(),
)
.unwrap();
loc.close();
@@ -1760,13 +1646,29 @@ mod tests {
)
.unwrap();
loc.create_volume(VolumeId(1), NeedleMapKind::InMemory, &VolumeSpec::default())
.unwrap();
loc.create_volume(VolumeId(2), NeedleMapKind::InMemory, &VolumeSpec::default())
.unwrap();
loc.create_volume(
VolumeId(1),
"",
NeedleMapKind::InMemory,
None,
None,
0,
Version::current(),
)
.unwrap();
loc.create_volume(
VolumeId(2),
"",
NeedleMapKind::InMemory,
None,
None,
0,
Version::current(),
)
.unwrap();
assert_eq!(loc.volumes_len(), 2);
loc.delete_volume(VolumeId(1), false, false, false).unwrap();
loc.delete_volume(VolumeId(1), false, false).unwrap();
assert_eq!(loc.volumes_len(), 1);
assert!(loc.find_volume(VolumeId(1)).is_none());
}
@@ -1787,29 +1689,32 @@ mod tests {
loc.create_volume(
VolumeId(1),
"pics",
NeedleMapKind::InMemory,
&VolumeSpec {
collection: "pics",
..Default::default()
},
None,
None,
0,
Version::current(),
)
.unwrap();
loc.create_volume(
VolumeId(2),
"pics",
NeedleMapKind::InMemory,
&VolumeSpec {
collection: "pics",
..Default::default()
},
None,
None,
0,
Version::current(),
)
.unwrap();
loc.create_volume(
VolumeId(3),
"docs",
NeedleMapKind::InMemory,
&VolumeSpec {
collection: "docs",
..Default::default()
},
None,
None,
0,
Version::current(),
)
.unwrap();
assert_eq!(loc.volumes_len(), 3);
@@ -1819,34 +1724,6 @@ mod tests {
assert!(loc.find_volume(VolumeId(3)).is_some());
}
/// A 0-byte `.ecx` is the stub a failed EC distribute copy leaves behind.
/// Go's HasEcxFileOnDisk requires Size() > 0 so the stub cannot pin
/// placement to a disk that has no usable index.
#[test]
fn test_has_ecx_file_on_disk_ignores_zero_byte_stub() {
let tmp = TempDir::new().unwrap();
let data = tmp.path().join("data");
let idx = tmp.path().join("idx");
fs::create_dir_all(&data).unwrap();
fs::create_dir_all(&idx).unwrap();
let loc = DiskLocation::new(
data.to_str().unwrap(),
idx.to_str().unwrap(),
10,
DiskType::HardDrive,
MinFreeSpace::Percent(1.0),
Vec::new(),
)
.unwrap();
fs::write(idx.join("pics_7.ecx"), b"").unwrap();
assert!(!loc.has_ecx_file_on_disk("pics", VolumeId(7)));
// A real index in the data dir still counts, stub or no stub.
fs::write(data.join("pics_7.ecx"), [0u8; 16]).unwrap();
assert!(loc.has_ecx_file_on_disk("pics", VolumeId(7)));
}
#[test]
fn test_disk_location_delete_collection_removes_ec_volumes() {
let tmp = TempDir::new().unwrap();
@@ -1863,8 +1740,6 @@ mod tests {
let shard_path = format!("{}/pics_7.ec00", dir);
std::fs::write(&shard_path, b"ec-shard").unwrap();
// An EC volume needs its .ecx to mount.
std::fs::write(format!("{}/pics_7.ecx", dir), [0u8; 16]).unwrap();
loc.mount_ec_shards(VolumeId(7), "pics", &[0], "").unwrap();
assert!(loc.has_ec_volume(VolumeId(7)));
@@ -1902,9 +1777,7 @@ mod tests {
// mount_ec_shards with source_disk_type="ssd" — simulating the
// VolumeEcShardsMount RPC path.
std::fs::write(format!("{}/pics_7.ec00", dir), b"ec-shard").unwrap();
std::fs::write(format!("{}/pics_7.ecx", dir), [0u8; 16]).unwrap();
loc.mount_ec_shards(VolumeId(7), "pics", &[0], "ssd")
.unwrap();
loc.mount_ec_shards(VolumeId(7), "pics", &[0], "ssd").unwrap();
{
let ec_vol = loc.find_ec_volume(VolumeId(7)).expect("ec volume mounted");
assert_eq!(
@@ -1921,9 +1794,7 @@ mod tests {
std::fs::write(format!("{}/pics_7.ec01", dir), b"ec-shard").unwrap();
loc.mount_ec_shards(VolumeId(7), "pics", &[1], "").unwrap();
{
let ec_vol = loc
.find_ec_volume(VolumeId(7))
.expect("ec volume still mounted");
let ec_vol = loc.find_ec_volume(VolumeId(7)).expect("ec volume still mounted");
assert_eq!(
ec_vol.disk_type,
DiskType::Ssd,
@@ -2002,12 +1873,10 @@ mod tests {
// A collection name unique to this test: the gauge is process-global
// and sibling tests running in parallel touch other labels.
std::fs::write(format!("{}/dupmount_11.ec00", dir), b"shard bytes").unwrap();
std::fs::write(format!("{}/dupmount_11.ecx", dir), [0u8; 16]).unwrap();
let gauge = crate::metrics::VOLUME_GAUGE.with_label_values(&["dupmount", "ec_shards"]);
let before = gauge.get();
loc.mount_ec_shards(VolumeId(11), "dupmount", &[0], "")
.unwrap();
loc.mount_ec_shards(VolumeId(11), "dupmount", &[0], "").unwrap();
loc.mount_ec_shards(VolumeId(11), "dupmount", &[0], "")
.expect("a duplicate mount must succeed as a no-op");
@@ -2083,11 +1952,8 @@ mod tests {
let path = format!("{}/{}_{}.ec{:02}", dir, collection, vid.0, sid);
std::fs::write(&path, b"shard data nonempty").unwrap();
}
std::fs::write(
format!("{}/{}_{}.ecx", dir, collection, vid.0),
vec![0u8; 20],
)
.unwrap();
std::fs::write(format!("{}/{}_{}.ecx", dir, collection, vid.0), vec![0u8; 20])
.unwrap();
std::fs::write(format!("{}/{}_{}.ecj", dir, collection, vid.0), b"").unwrap();
std::fs::write(
format!("{}/{}_{}.vif", dir, collection, vid.0),
@@ -30,7 +30,6 @@ use crate::pb::volume_server_pb::{
ChecksumAlgorithm, EcBitrotProtection, EcShardChecksums, EcShardConfig,
};
use crate::storage::erasure_coding::ec_shard::MAX_SHARD_COUNT;
use crate::storage::io::read_exact_at;
use crate::storage::needle::crc::CRC;
/// Canonical extension for the checksum sidecar. Generation 0 (legacy/fresh
@@ -538,7 +537,7 @@ pub fn verify_shard_blocks(
break;
}
let to_read = to_read as usize;
read_exact_at(f, &mut buf[..to_read], offset as u64)?;
read_full_at(f, &mut buf[..to_read], offset as u64)?;
if CRC::new(&buf[..to_read]).0 != *want_crc {
mismatched.push(i);
}
@@ -547,6 +546,33 @@ pub fn verify_shard_blocks(
Ok(mismatched)
}
/// Reads exactly `buf.len()` bytes from `f` at `offset`, erroring on early EOF.
fn read_full_at(f: &File, buf: &mut [u8], offset: u64) -> io::Result<()> {
let mut total = 0usize;
while total < buf.len() {
#[cfg(unix)]
let n = {
use std::os::unix::fs::FileExt;
f.read_at(&mut buf[total..], offset + total as u64)?
};
#[cfg(not(unix))]
let n = {
use std::io::{Read, Seek, SeekFrom};
let mut fc = f.try_clone()?;
fc.seek(SeekFrom::Start(offset + total as u64))?;
fc.read(&mut buf[total..])?
};
if n == 0 {
return Err(io::Error::new(
io::ErrorKind::UnexpectedEof,
"short read on shard block",
));
}
total += n;
}
Ok(())
}
/// Builds the `EcShardConfig` proto for the given layout. The bitrot sidecar
/// carries its own top-level encode_uuid, so the nested config leaves it empty.
pub fn ec_shard_config(data_shards: u32, parity_shards: u32, block_size: i64) -> EcShardConfig {
@@ -601,10 +627,7 @@ mod tests {
save_bitrot_sidecar(path, &prot).unwrap();
let bytes = std::fs::read(path).unwrap();
let hex: String = bytes.iter().map(|b| format!("{:02x}", b)).collect();
assert_eq!(
hex, CANONICAL_HEX,
"Rust .ecsum bytes drifted from the Go canonical form"
);
assert_eq!(hex, CANONICAL_HEX, "Rust .ecsum bytes drifted from the Go canonical form");
let _ = std::fs::remove_file(path);
}
@@ -639,11 +662,7 @@ mod tests {
format!("{}.ecsum.v1", base),
format!("{}.ecsum.v7", base),
] {
assert!(
!std::path::Path::new(&p).exists(),
"{} should be removed",
p
);
assert!(!std::path::Path::new(&p).exists(), "{} should be removed", p);
}
assert!(std::path::Path::new(&keep_shard).exists());
assert!(std::path::Path::new(&keep_other_vid).exists());
@@ -662,9 +681,7 @@ mod tests {
assert!(!is_pow2_multiple_of_1mib(1 << 19)); // 512 KiB, too small
assert!(!is_pow2_multiple_of_1mib(3 << 20)); // 3 MiB, not pow2
assert!(!is_pow2_multiple_of_1mib(128 * 1024 * 1024)); // pow2 but > MAX_BITROT_BLOCK_SIZE
assert!(!is_pow2_multiple_of_1mib(
DEFAULT_BITROT_BLOCK_SIZE as u32 + 1
));
assert!(!is_pow2_multiple_of_1mib(DEFAULT_BITROT_BLOCK_SIZE as u32 + 1));
}
#[test]
@@ -718,7 +735,12 @@ mod tests {
#[test]
fn test_save_load_roundtrip() {
let tmp = tempfile::TempDir::new().unwrap();
let path = tmp.path().join("vol.ecsum").to_str().unwrap().to_string();
let path = tmp
.path()
.join("vol.ecsum")
.to_str()
.unwrap()
.to_string();
let mut builder = ShardChecksumBuilder::new(DEFAULT_BITROT_BLOCK_SIZE as i64);
builder.write(b"hello world");
@@ -879,7 +901,8 @@ mod tests {
assert_eq!(resolve_status(&notfound, 0, 10, 4), BitrotStatus::Off);
// Integrity failure => Invalid.
let bad: Result<EcBitrotProtection, BitrotLoadError> = Err(BitrotLoadError::BadMagic(0));
let bad: Result<EcBitrotProtection, BitrotLoadError> =
Err(BitrotLoadError::BadMagic(0));
assert_eq!(resolve_status(&bad, 0, 10, 4), BitrotStatus::Invalid);
// Generation mismatch => Off.
@@ -3,12 +3,10 @@
//! Rebuilds the original .dat + .idx files from data shards (.ec00-.ec09)
//! and the sorted index (.ecx) + deletion journal (.ecj).
use std::collections::HashSet;
use std::fs::File;
use std::io::{self, Read, Write};
use crate::storage::erasure_coding::ec_shard::*;
use crate::storage::erasure_coding::ec_volume::read_ecj_ids;
use crate::storage::idx;
use crate::storage::needle::needle::get_actual_size;
use crate::storage::super_block::SUPER_BLOCK_SIZE;
@@ -22,21 +20,18 @@ use crate::storage::volume::{fsync_dir, volume_file_name};
/// `dir` is used both for reading `.ec00` and `.ecx`. For split-disk
/// reconciled volumes call [`find_dat_file_size_with_dirs`] instead.
pub fn find_dat_file_size(dir: &str, collection: &str, volume_id: VolumeId) -> io::Result<i64> {
let deleted = read_ecj_deletions(&[dir], collection, volume_id)?;
find_dat_file_size_with_dirs(dir, dir, collection, volume_id, &deleted)
find_dat_file_size_with_dirs(dir, dir, collection, volume_id)
}
/// Like [`find_dat_file_size`] but lets the caller pass separate dirs
/// for `.ec00` (the data shard) and `.ecx` (the sealed index). This
/// is the form needed when shards are split across data dirs and the
/// `.ecx` lives on a sibling disk's idx dir (#9252). Needles in `deleted`
/// count as deleted.
/// `.ecx` lives on a sibling disk's idx dir (#9252).
pub fn find_dat_file_size_with_dirs(
ec00_dir: &str,
ecx_dir: &str,
collection: &str,
volume_id: VolumeId,
deleted: &HashSet<NeedleId>,
) -> io::Result<i64> {
let ec00_base = volume_file_name(ec00_dir, collection, volume_id);
let ecx_base = volume_file_name(ecx_dir, collection, volume_id);
@@ -58,9 +53,9 @@ pub fn find_dat_file_size_with_dirs(
for i in 0..entry_count {
let start = i * NEEDLE_MAP_ENTRY_SIZE;
let (key, offset, size) =
let (_, offset, size) =
idx_entry_from_bytes(&ecx_data[start..start + NEEDLE_MAP_ENTRY_SIZE]);
if size.is_deleted() || deleted.contains(&key) {
if size.is_deleted() {
continue;
}
let entry_stop = offset.to_actual_offset() + get_actual_size(size, version);
@@ -72,163 +67,73 @@ pub fn find_dat_file_size_with_dirs(
Ok(dat_size)
}
/// Whether the `.ecx` in `ecx_dir` indexes a needle deleted neither there nor
/// in `deleted`.
pub fn has_live_needles(
ecx_dir: &str,
/// Reconstruct a .dat file from EC data shards.
///
/// Reads from .ec00-.ec09 and writes a new .dat file. All data shards
/// must live in `dir`. For the cross-disk reconciled layout where
/// shards are split across multiple data dirs of the same node, use
/// [`write_dat_file_from_shards_with_dirs`] instead.
#[expect(clippy::too_many_arguments)]
pub fn write_dat_file_from_shards(
dir: &str,
collection: &str,
volume_id: VolumeId,
deleted: &HashSet<NeedleId>,
) -> io::Result<bool> {
let ecx_base = volume_file_name(ecx_dir, collection, volume_id);
let ecx_data = std::fs::read(format!("{}.ecx", ecx_base))?;
let (entries, _) = ecx_data.as_chunks::<NEEDLE_MAP_ENTRY_SIZE>();
Ok(entries.iter().any(|entry| {
let (key, _, size) = idx_entry_from_bytes(entry);
!size.is_deleted() && !deleted.contains(&key)
}))
dat_file_size: i64,
encoded_dat_file_size: i64,
data_shards: usize,
large_block_size: usize,
small_block_size: usize,
) -> io::Result<()> {
let dirs: Vec<String> = (0..data_shards).map(|_| dir.to_string()).collect();
write_dat_file_from_shards_with_dirs(
dir,
collection,
volume_id,
dat_file_size,
encoded_dat_file_size,
data_shards,
&dirs,
large_block_size,
small_block_size,
)
}
/// Distinct needle ids journaled in the `.ecj` of any of `dirs`. Go folds the
/// journal into the `.ecx` (RebuildEcxFile) before a decode; reading it leaves
/// the sealed index untouched. Only NotFound means "no journal".
pub fn read_ecj_deletions(
dirs: &[&str],
collection: &str,
volume_id: VolumeId,
) -> io::Result<HashSet<NeedleId>> {
Ok(EcjDeletions::read(dirs, collection, volume_id)?.ids)
}
/// The ids [`read_ecj_deletions`] returns, plus how far each journal was read
/// so that ids journaled later can be added.
pub struct EcjDeletions {
pub ids: HashSet<NeedleId>,
/// Each distinct journal path and the whole-record length read so far.
journals: Vec<(String, u64)>,
}
impl EcjDeletions {
pub fn read(dirs: &[&str], collection: &str, volume_id: VolumeId) -> io::Result<Self> {
let mut journals: Vec<(String, u64)> = Vec::new();
for dir in dirs {
let path = format!("{}.ecj", volume_file_name(dir, collection, volume_id));
if !journals.iter().any(|(p, _)| *p == path) {
journals.push((path, 0));
}
}
let mut deletions = EcjDeletions {
ids: HashSet::new(),
journals,
};
deletions.catch_up()?;
Ok(deletions)
}
/// Adds the ids appended to each journal since the last read. A journal
/// that shrank is read again from the start — and the whole set rebuilt,
/// since ids already folded in from the truncated tail may have been a
/// rolled-back append. Non-regular journals (the FIFOs the tests stand
/// in for a blocking disk) stat empty and cannot be rolled back, so they
/// never count as shrunk.
pub fn catch_up(&mut self) -> io::Result<()> {
let mut shrank = false;
for (path, read_to) in &self.journals {
if *read_to == 0 {
continue;
}
match std::fs::metadata(path) {
Ok(m) => {
if m.is_file() && m.len() < *read_to {
shrank = true;
break;
}
}
// A journal that was read before and is now gone shrank to
// nothing (e.g. the volume was destroyed mid-scan) — the ids
// read from it no longer reflect committed content.
Err(e) if e.kind() == io::ErrorKind::NotFound => {
shrank = true;
break;
}
Err(e) => return Err(e),
}
}
if shrank {
self.ids.clear();
for (_, read_to) in &mut self.journals {
*read_to = 0;
}
}
for (path, read_to) in &mut self.journals {
let file = match File::open(&*path) {
Ok(file) => file,
Err(e) if e.kind() == io::ErrorKind::NotFound => continue,
Err(e) => return Err(e),
};
let len = file.metadata()?.len();
if len < *read_to {
*read_to = 0;
}
read_ecj_ids(&file, *read_to, len, &mut self.ids)?;
*read_to = len - len % NEEDLE_ID_SIZE as u64;
}
Ok(())
}
/// Clears the set and re-reads every journal from the start. Run under
/// the caller's store read lock (no append in flight): the result is
/// then exactly the committed content — an earlier unlocked read may
/// have folded in bytes a rolled-back append later truncated, or missed
/// a record re-appended to the very offset a rollback freed.
pub fn rescan(&mut self) -> io::Result<()> {
self.ids.clear();
for (_, read_to) in &mut self.journals {
*read_to = 0;
}
self.catch_up()
}
}
/// What it takes to rebuild a volume's .dat from its EC data shards.
/// Reconstruct a .dat file from EC data shards, taking the source
/// directory for each shard separately.
///
/// `dat_dir` is where the produced `.dat` is written. `shard_dirs[i]`
/// is the directory holding shard `i`. For the simple "all shards in
/// one dir" case both can be the same value.
///
/// Mirrors Go's `WriteDatFile(baseFileName, datFileSize,
/// encodedDatFileSize, shardFileNames)` shape — Go passes per-shard
/// paths so a reconciled volume with shards split across disks of the
/// same volume server can still be decoded back to a regular .dat
/// (seaweedfs/seaweedfs#9252).
#[derive(Clone, Copy, Debug)]
pub struct DatRebuild<'a> {
/// Where the produced `.dat` is written.
pub dat_dir: &'a str,
pub collection: &'a str,
pub volume_id: VolumeId,
/// The number of bytes to write, i.e. the live data extent from
/// [`find_dat_file_size`].
pub dat_file_size: i64,
/// The .dat size at encode time, which fixed the shard block layout:
/// deletions can move the live extent below the large-block row
/// boundary, and deriving the layout from the shrunk extent would read
/// the shards in the wrong block order. Zero when the .vif does not
/// record the encode-time size; the layout is then inferred from the
/// shard size.
pub encoded_dat_file_size: i64,
pub data_shards: usize,
/// `shard_dirs[i]` is the directory holding shard `i`. `None` means every
/// data shard sits in `dat_dir`.
pub shard_dirs: Option<&'a [String]>,
/// The volume's shard block layout, e.g. `EcVolume::large_block_size()`
/// / `small_block_size()` from its .vif EC config.
pub large_block_size: usize,
pub small_block_size: usize,
}
/// Reconstruct a .dat file from EC data shards.
///
/// Reads from .ec00-.ec09 and writes a new .dat file, from one directory or
/// from the per-shard directories of a cross-disk reconciled volume.
pub fn write_dat_file_from_shards(spec: &DatRebuild<'_>) -> io::Result<()> {
let DatRebuild {
/// `dat_file_size` is the number of bytes to write, i.e. the live data
/// extent from [`find_dat_file_size`]. `encoded_dat_file_size` is the
/// .dat size at encode time, which fixed the shard block layout:
/// deletions can move the live extent below the large-block row
/// boundary, and deriving the layout from the shrunk extent would read
/// the shards in the wrong block order. Pass zero when the .vif does
/// not record the encode-time size to infer the layout from the shard
/// size. `large_block_size`/`small_block_size` are the volume's shard
/// block layout, e.g. `EcVolume::large_block_size()` /
/// `small_block_size()` from its .vif EC config.
#[expect(clippy::too_many_arguments)]
pub fn write_dat_file_from_shards_with_dirs(
dat_dir: &str,
collection: &str,
volume_id: VolumeId,
dat_file_size: i64,
encoded_dat_file_size: i64,
data_shards: usize,
shard_dirs: &[String],
large_block_size: usize,
small_block_size: usize,
) -> io::Result<()> {
write_dat_file(
dat_dir,
collection,
volume_id,
@@ -238,15 +143,21 @@ pub fn write_dat_file_from_shards(spec: &DatRebuild<'_>) -> io::Result<()> {
shard_dirs,
large_block_size,
small_block_size,
} = *spec;
let same_dir: Vec<String>;
let shard_dirs: &[String] = match shard_dirs {
Some(dirs) => dirs,
None => {
same_dir = vec![dat_dir.to_string(); data_shards];
&same_dir
}
};
)
}
#[expect(clippy::too_many_arguments)]
fn write_dat_file(
dat_dir: &str,
collection: &str,
volume_id: VolumeId,
dat_file_size: i64,
encoded_dat_file_size: i64,
data_shards: usize,
shard_dirs: &[String],
large_block_size: usize,
small_block_size: usize,
) -> io::Result<()> {
if data_shards == 0 {
return Err(io::Error::new(
io::ErrorKind::InvalidInput,
@@ -388,87 +299,53 @@ pub fn write_dat_file_from_shards(spec: &DatRebuild<'_>) -> io::Result<()> {
write_result
}
/// Fails when the decoded `.dat` in `dat_dir` is shorter than the
/// `dat_file_size` bytes its EC index references: the caller deletes the
/// shards next, and they are the only other copy of the needles past the cut.
/// A longer file passes.
pub fn verify_decoded_dat_file(
dat_dir: &str,
collection: &str,
volume_id: VolumeId,
dat_file_size: i64,
) -> io::Result<()> {
let dat_path = format!("{}.dat", volume_file_name(dat_dir, collection, volume_id));
let size = std::fs::metadata(&dat_path)?.len();
if (size as i64) < dat_file_size {
return Err(io::Error::new(
io::ErrorKind::UnexpectedEof,
format!(
"decoded {} is {} bytes, short of the {} its ec index references",
dat_path, size, dat_file_size
),
));
}
Ok(())
}
/// Write .idx file from .ecx index + .ecj deletion journal.
///
/// See [`write_idx_file_from_ec_index_with_dirs`]; everything lives in `dir`.
/// Copies sorted .ecx entries to .idx, then appends tombstones for
/// deleted needles from .ecj.
pub fn write_idx_file_from_ec_index(
dir: &str,
collection: &str,
volume_id: VolumeId,
) -> io::Result<()> {
let deleted = read_ecj_deletions(&[dir], collection, volume_id)?;
let dat_file_size = find_dat_file_size_with_dirs(dir, dir, collection, volume_id, &deleted)?;
write_idx_file_from_ec_index_with_dirs(dir, dir, collection, volume_id, &deleted, dat_file_size)
}
/// Write the `.idx` for a `.dat` decoded to `dat_file_size` bytes, from the
/// `.ecx` in `ecx_dir`, into `idx_dir`.
///
/// Copies the `.ecx` rows, then appends one tombstone per row whose needle is
/// in `deleted`. A deleted needle at or past `dat_file_size` was cut from the
/// `.dat`, so its row is dropped: a row pointing past the end of the `.dat`
/// makes the volume load read-only.
pub fn write_idx_file_from_ec_index_with_dirs(
ecx_dir: &str,
idx_dir: &str,
collection: &str,
volume_id: VolumeId,
deleted: &HashSet<NeedleId>,
dat_file_size: i64,
) -> io::Result<()> {
let ecx_path = format!("{}.ecx", volume_file_name(ecx_dir, collection, volume_id));
let idx_path = format!("{}.idx", volume_file_name(idx_dir, collection, volume_id));
let base = volume_file_name(dir, collection, volume_id);
let ecx_path = format!("{}.ecx", base);
let ecj_path = format!("{}.ecj", base);
let idx_path = format!("{}.idx", base);
// Write to a temp file and atomically rename into place, so a crash
// mid-write never leaves a partial .idx at the final name beside the
// source shards.
let tmp_path = format!("{}.tmp", idx_path);
let write_result = (|| -> io::Result<()> {
let mut ecx_file = File::open(&ecx_path)?;
let mut idx_file = io::BufWriter::new(File::create(&tmp_path)?);
let mut tombstoned = Vec::new();
idx::walk_index_file(&mut ecx_file, 0, |key, offset, size| {
let is_deleted = size.is_deleted() || deleted.contains(&key);
if is_deleted && offset.to_actual_offset() >= dat_file_size {
return Ok(());
// Copy .ecx to the temp .idx
std::fs::copy(&ecx_path, &tmp_path)?;
// Append deletions from .ecj as tombstones. Read the journal directly
// and treat only NotFound as "no journal": Path::exists would also
// swallow a permission/IO error and silently skip deletions, which
// would resurrect deleted needles as live.
let mut idx_file = std::fs::OpenOptions::new().append(true).open(&tmp_path)?;
match std::fs::read(&ecj_path) {
Ok(ecj_data) => {
let count = ecj_data.len() / NEEDLE_ID_SIZE;
for i in 0..count {
let start = i * NEEDLE_ID_SIZE;
let needle_id = NeedleId::from_bytes(&ecj_data[start..start + NEEDLE_ID_SIZE]);
idx::write_index_entry(
&mut idx_file,
needle_id,
Offset::default(),
TOMBSTONE_FILE_SIZE,
)?;
}
}
idx::write_index_entry(&mut idx_file, key, offset, size)?;
if !size.is_deleted() && deleted.contains(&key) {
tombstoned.push(key);
}
Ok(())
})?;
for key in tombstoned {
idx::write_index_entry(&mut idx_file, key, Offset::default(), TOMBSTONE_FILE_SIZE)?;
Err(e) if e.kind() == io::ErrorKind::NotFound => {}
Err(e) => return Err(e),
}
// fsync, rename, then fsync the dir so the decoded .idx is durable and
// atomically published before the caller deletes the source shards.
let idx_file = idx_file.into_inner().map_err(|e| e.into_error())?;
idx_file.sync_all()?;
drop(idx_file);
// Windows rename does not replace an existing file on every version;
@@ -493,7 +370,7 @@ mod tests {
use crate::storage::erasure_coding::ec_encoder;
use crate::storage::needle::needle::Needle;
use crate::storage::needle_map::NeedleMapKind;
use crate::storage::volume::{Volume, VolumeSpec};
use crate::storage::volume::Volume;
use tempfile::TempDir;
#[test]
@@ -505,9 +382,13 @@ mod tests {
let mut v = Volume::new(
dir,
dir,
"",
VolumeId(1),
NeedleMapKind::InMemory,
&VolumeSpec::default(),
None,
None,
0,
Version::current(),
)
.unwrap();
@@ -547,17 +428,16 @@ mod tests {
std::fs::remove_file(format!("{}/1.idx", dir)).unwrap();
// Reconstruct from EC shards
write_dat_file_from_shards(&DatRebuild {
dat_dir: dir,
collection: "",
volume_id: VolumeId(1),
dat_file_size: original_dat_size as i64,
encoded_dat_file_size: original_dat_size as i64,
write_dat_file_from_shards(
dir,
"",
VolumeId(1),
original_dat_size as i64,
original_dat_size as i64,
data_shards,
shard_dirs: None,
large_block_size: block_size as usize,
small_block_size: block_size as usize,
})
block_size as usize,
block_size as usize,
)
.unwrap();
write_idx_file_from_ec_index(dir, "", VolumeId(1)).unwrap();
@@ -577,9 +457,13 @@ mod tests {
let v2 = Volume::new(
dir,
dir,
"",
VolumeId(1),
NeedleMapKind::InMemory,
&VolumeSpec::default(),
None,
None,
0,
Version::current(),
)
.unwrap();
@@ -599,29 +483,29 @@ mod tests {
let dir = tmp.path().to_str().unwrap();
// No shard files exist, so de-striping must fail and publish nothing:
// neither the final .dat nor a partial .dat.tmp may remain.
let res = write_dat_file_from_shards(&DatRebuild {
dat_dir: dir,
collection: "",
volume_id: VolumeId(7),
dat_file_size: 100,
encoded_dat_file_size: 100,
data_shards: 10,
shard_dirs: None,
large_block_size: ERASURE_CODING_LARGE_BLOCK_SIZE,
small_block_size: ERASURE_CODING_SMALL_BLOCK_SIZE,
});
let res = write_dat_file_from_shards(
dir,
"",
VolumeId(7),
100,
100,
10,
ERASURE_CODING_LARGE_BLOCK_SIZE,
ERASURE_CODING_SMALL_BLOCK_SIZE,
);
assert!(res.is_err());
assert!(!std::path::Path::new(&format!("{}/7.dat", dir)).exists());
assert!(!std::path::Path::new(&format!("{}/7.dat.tmp", dir)).exists());
}
// Decoding when .vif does not record the encode-time size: the layout is
// inferred from the shard size, except when that is an exact large-block
// multiple and the live extent reaches the ambiguous region.
#[test]
fn test_write_dat_file_fallback_layout() {
use crate::storage::erasure_coding::ec_bitrot::{
DEFAULT_BITROT_BLOCK_SIZE, ShardChecksumBuilder,
ShardChecksumBuilder, DEFAULT_BITROT_BLOCK_SIZE,
};
use reed_solomon_erasure::galois_8::ReedSolomon;
@@ -659,13 +543,11 @@ mod tests {
&rs,
&mut shards,
&mut builders,
ec_encoder::EcEncodeLayout {
data_shards,
parity_shards,
buffer_size: SMALL,
large_block_size: LARGE,
small_block_size: SMALL,
},
data_shards,
parity_shards,
SMALL,
LARGE,
SMALL,
)
.unwrap();
for shard in &mut shards {
@@ -683,17 +565,7 @@ mod tests {
-> io::Result<Vec<u8>> {
let out = format!("{}/{}", dir, sub);
std::fs::create_dir_all(&out).unwrap();
write_dat_file_from_shards(&DatRebuild {
dat_dir: &out,
collection: "",
volume_id: VolumeId(1),
dat_file_size: live,
encoded_dat_file_size: encoded,
data_shards: 10,
shard_dirs: Some(shard_dirs),
large_block_size: LARGE,
small_block_size: SMALL,
})?;
write_dat_file(&out, "", VolumeId(1), live, encoded, 10, shard_dirs, LARGE, SMALL)?;
Ok(std::fs::read(format!("{}/1.dat", out)).unwrap())
};
@@ -707,20 +579,14 @@ mod tests {
// each shard exactly one large block, indistinguishable from one large row
let (dir, shard_dirs, _) = encode("ambig1", large_row_size - 1);
let err = decode_to(&dir, "out", large_row_size / 2, 0, &shard_dirs).unwrap_err();
assert!(
err.to_string()
.contains("does not identify the block layout")
);
assert!(err.to_string().contains("does not identify the block layout"));
// two-row equivalent: decoding within the agreed prefix still works
let (dir, shard_dirs, original) = encode("ambig2", 2 * large_row_size - 1);
let decoded = decode_to(&dir, "outa", large_row_size, 0, &shard_dirs).unwrap();
assert_eq!(&original[..large_row_size as usize], &decoded[..]);
let err = decode_to(&dir, "outb", large_row_size + 1, 0, &shard_dirs).unwrap_err();
assert!(
err.to_string()
.contains("does not identify the block layout")
);
assert!(err.to_string().contains("does not identify the block layout"));
}
// Decoding after deletions moved the live extent below the large-block row
@@ -729,7 +595,7 @@ mod tests {
#[test]
fn test_write_dat_file_after_tail_deletion() {
use crate::storage::erasure_coding::ec_bitrot::{
DEFAULT_BITROT_BLOCK_SIZE, ShardChecksumBuilder,
ShardChecksumBuilder, DEFAULT_BITROT_BLOCK_SIZE,
};
use reed_solomon_erasure::galois_8::ReedSolomon;
@@ -769,13 +635,11 @@ mod tests {
&rs,
&mut shards,
&mut builders,
ec_encoder::EcEncodeLayout {
data_shards,
parity_shards,
buffer_size: SMALL,
large_block_size: LARGE,
small_block_size: SMALL,
},
data_shards,
parity_shards,
SMALL,
LARGE,
SMALL,
)
.unwrap();
for shard in &mut shards {
@@ -791,17 +655,17 @@ mod tests {
std::fs::create_dir(&out_dir).unwrap();
let out = out_dir.to_str().unwrap();
let decode = |live_size: i64, encoded_size: i64| -> Vec<u8> {
write_dat_file_from_shards(&DatRebuild {
dat_dir: out,
collection: "",
volume_id: VolumeId(1),
dat_file_size: live_size,
encoded_dat_file_size: encoded_size,
write_dat_file(
out,
"",
VolumeId(1),
live_size,
encoded_size,
data_shards,
shard_dirs: Some(&shard_dirs),
large_block_size: LARGE,
small_block_size: SMALL,
})
&shard_dirs,
LARGE,
SMALL,
)
.unwrap();
let path = format!("{}/1.dat", out);
let decoded = std::fs::read(&path).unwrap();
@@ -836,131 +700,17 @@ mod tests {
assert_ne!(&original[..(large_row_size / 2) as usize], &control[..]);
// the live extent can never exceed the encode-time size
assert!(
write_dat_file_from_shards(&DatRebuild {
dat_dir: out,
collection: "",
volume_id: VolumeId(1),
dat_file_size: dat_size + 1,
encoded_dat_file_size: dat_size,
data_shards,
shard_dirs: Some(&shard_dirs),
large_block_size: LARGE,
small_block_size: SMALL,
})
.is_err()
);
}
/// A journal many chunks long that repeats a few ids reads back as those
/// ids, from each dir once, whatever its length.
#[test]
fn test_read_ecj_deletions_collects_distinct_ids_across_dirs() {
let tmp = TempDir::new().unwrap();
let data = tmp.path().join("data");
let idx = tmp.path().join("idx");
let missing = tmp.path().join("missing");
std::fs::create_dir_all(&data).unwrap();
std::fs::create_dir_all(&idx).unwrap();
let (data, idx, missing) = (
data.to_str().unwrap(),
idx.to_str().unwrap(),
missing.to_str().unwrap(),
);
let entry = |id: u64| {
let mut buf = [0u8; NEEDLE_ID_SIZE];
NeedleId(id).to_bytes(&mut buf);
buf
};
// Past two load chunks of three repeating ids, then an id only in the
// last chunk and a torn trailing record.
let mut ecj = Vec::new();
while ecj.len() <= 2 * (1 << 20) {
for id in [1, 2, 3] {
ecj.extend_from_slice(&entry(id));
}
}
ecj.extend_from_slice(&entry(7));
ecj.extend_from_slice(&entry(8)[..3]);
std::fs::write(format!("{idx}/1.ecj"), &ecj).unwrap();
std::fs::write(format!("{data}/1.ecj"), entry(9)).unwrap();
let ids = read_ecj_deletions(&[data, idx, idx, missing], "", VolumeId(1)).unwrap();
let expected: HashSet<NeedleId> = [1, 2, 3, 7, 9].into_iter().map(NeedleId).collect();
assert_eq!(ids, expected);
}
#[test]
fn test_verify_decoded_dat_file_rejects_a_short_dat() {
let tmp = TempDir::new().unwrap();
let dir = tmp.path().to_str().unwrap();
let dat_path = format!("{dir}/1.dat");
let err = verify_decoded_dat_file(dir, "", VolumeId(1), 100).unwrap_err();
assert_eq!(err.kind(), io::ErrorKind::NotFound);
std::fs::write(&dat_path, vec![0u8; 99]).unwrap();
let err = verify_decoded_dat_file(dir, "", VolumeId(1), 100).unwrap_err();
assert!(err.to_string().contains("short of the 100"), "{err}");
std::fs::write(&dat_path, vec![0u8; 100]).unwrap();
verify_decoded_dat_file(dir, "", VolumeId(1), 100).unwrap();
std::fs::write(&dat_path, vec![0u8; 101]).unwrap();
verify_decoded_dat_file(dir, "", VolumeId(1), 100).unwrap();
}
/// Ids appended after the first read, including the rest of a record torn
/// at that point, are picked up; a journal that shrank is read again.
#[test]
fn test_ecj_deletions_catch_up_reads_appended_ids() {
let tmp = TempDir::new().unwrap();
let dir = tmp.path().to_str().unwrap();
let ecj_path = format!("{dir}/1.ecj");
let entry = |id: u64| {
let mut buf = [0u8; NEEDLE_ID_SIZE];
NeedleId(id).to_bytes(&mut buf);
buf
};
let append = |bytes: &[u8]| {
let mut f = std::fs::OpenOptions::new()
.create(true)
.append(true)
.open(&ecj_path)
.unwrap();
f.write_all(bytes).unwrap();
};
let ids = |d: &EcjDeletions| {
let mut ids: Vec<u64> = d.ids.iter().map(|id| id.0).collect();
ids.sort();
ids
};
// No journal yet.
let mut deletions = EcjDeletions::read(&[dir, dir], "", VolumeId(1)).unwrap();
assert!(deletions.ids.is_empty());
append(&entry(1));
append(&entry(2)[..3]);
deletions.catch_up().unwrap();
assert_eq!(ids(&deletions), [1]);
append(&entry(2)[3..]);
append(&entry(3));
deletions.catch_up().unwrap();
assert_eq!(ids(&deletions), [1, 2, 3]);
// A shrunk journal is rebuilt from its surviving content: ids folded
// in from the truncated tail may have been rolled back and must not
// linger as phantom tombstones.
std::fs::write(&ecj_path, entry(9)).unwrap();
deletions.catch_up().unwrap();
assert_eq!(ids(&deletions), [9]);
// A journal removed since it was read is the extreme shrink: its
// earlier ids must go with it, not linger.
std::fs::remove_file(&ecj_path).unwrap();
deletions.catch_up().unwrap();
assert!(deletions.ids.is_empty());
assert!(write_dat_file(
out,
"",
VolumeId(1),
dat_size + 1,
dat_size,
data_shards,
&shard_dirs,
LARGE,
SMALL,
)
.is_err());
}
}
@@ -5,12 +5,16 @@
use std::fs::File;
use std::io;
#[cfg(not(unix))]
use std::io::{Read, Seek, SeekFrom};
use reed_solomon_erasure::galois_8::ReedSolomon;
use crate::pb::volume_server_pb::{ChecksumAlgorithm, EcBitrotProtection, EcShardChecksums};
use crate::pb::volume_server_pb::{
ChecksumAlgorithm, EcBitrotProtection, EcShardChecksums,
};
use crate::storage::erasure_coding::ec_bitrot::{
self, DEFAULT_BITROT_BLOCK_SIZE, ShardChecksumBuilder,
self, ShardChecksumBuilder, DEFAULT_BITROT_BLOCK_SIZE,
};
use crate::storage::erasure_coding::ec_shard::*;
use crate::storage::idx;
@@ -73,13 +77,11 @@ pub fn write_ec_files(
&rs,
&mut shards,
&mut builders,
EcEncodeLayout {
data_shards,
parity_shards,
buffer_size: ENCODE_BUFFER_SIZE,
large_block_size: block_size as usize,
small_block_size: block_size as usize,
},
data_shards,
parity_shards,
ENCODE_BUFFER_SIZE,
block_size as usize,
block_size as usize,
)?;
// Close all shards
@@ -425,23 +427,22 @@ pub(crate) fn write_sorted_ecx_from_idx(idx_path: &str, ecx_path: &str) -> io::R
// Read all idx entries
let mut idx_file = File::open(idx_path)?;
let mut last: std::collections::HashMap<NeedleId, (Offset, Size)> =
std::collections::HashMap::new();
let mut entries: Vec<(NeedleId, Offset, Size)> = Vec::new();
idx::walk_index_file(&mut idx_file, 0, |key, offset, size| {
last.insert(key, (offset, size));
entries.push((key, offset, size));
Ok(())
})?;
let mut entries: Vec<(NeedleId, Offset, Size)> = last
.into_iter()
.filter_map(|(key, (offset, size))| {
if size.is_deleted() || offset.is_zero() {
None
} else {
Some((key, offset, size))
}
})
.collect();
entries.sort_by_key(|&(key, _o, _s)| key);
// Sort by NeedleId, then by actual offset so later entries come last
entries.sort_by_key(|&(key, offset, _)| (key, offset.to_actual_offset()));
// Remove duplicates (keep last/latest entry for each key).
// dedup_by_key keeps the first in each run, so we reverse first,
// dedup, then reverse back.
entries.reverse();
entries.dedup_by_key(|entry| entry.0);
entries.reverse();
// Write sorted entries to .ecx
let mut ecx_file = File::create(ecx_path)?;
@@ -582,8 +583,7 @@ pub fn rebuild_ecx_file(
}
let cookie = Cookie::from_bytes(&header_buf[..COOKIE_SIZE]);
let needle_id =
NeedleId::from_bytes(&header_buf[COOKIE_SIZE..COOKIE_SIZE + NEEDLE_ID_SIZE]);
let needle_id = NeedleId::from_bytes(&header_buf[COOKIE_SIZE..COOKIE_SIZE + NEEDLE_ID_SIZE]);
let size = Size::from_bytes(&header_buf[COOKIE_SIZE + NEEDLE_ID_SIZE..header_size]);
// Validate: stop if we hit zero cookie+id (end of data)
@@ -702,50 +702,30 @@ fn read_from_data_shards(
/// the uniform block is.
const ENCODE_BUFFER_SIZE: usize = 256 * 1024;
/// Shape of one encode run: the Reed-Solomon split and the block sizes that
/// fix where every byte of the .dat lands in the shards. Mirrors Go's
/// `ECContext`. `buffer_size` must divide both block sizes.
#[derive(Clone, Copy, Debug)]
pub(crate) struct EcEncodeLayout {
pub(crate) data_shards: usize,
pub(crate) parity_shards: usize,
/// Bytes of each shard's block handled per sub-batch; bounds memory at
/// `total_shards * buffer_size` however large the blocks are.
pub(crate) buffer_size: usize,
pub(crate) large_block_size: usize,
pub(crate) small_block_size: usize,
}
/// Encode the .dat file data into shard files.
///
/// Uses a two-phase approach matching Go's ec_encoder.go:
/// 1. Process as many large blocks as possible
/// 2. Process remaining data with small blocks
///
/// `buffer_size` must divide both block sizes.
#[expect(clippy::too_many_arguments)]
pub(crate) fn encode_dat_file(
dat_file: &File,
dat_size: i64,
rs: &ReedSolomon,
shards: &mut [EcVolumeShard],
builders: &mut [ShardChecksumBuilder],
layout: EcEncodeLayout,
data_shards: usize,
parity_shards: usize,
buffer_size: usize,
large_block_size: usize,
small_block_size: usize,
) -> io::Result<()> {
let EcEncodeLayout {
data_shards,
parity_shards,
buffer_size,
large_block_size,
small_block_size,
} = layout;
let total_shards = data_shards + parity_shards;
let mut buffers: Vec<Vec<u8>> = (0..total_shards).map(|_| vec![0u8; buffer_size]).collect();
let mut run = EncodeRun {
dat_file,
rs,
buffers: &mut buffers,
shards,
builders,
data_shards,
};
let mut buffers: Vec<Vec<u8>> = (0..total_shards)
.map(|_| vec![0u8; buffer_size])
.collect();
let mut remaining = dat_size;
let mut offset: u64 = 0;
@@ -754,7 +734,16 @@ pub(crate) fn encode_dat_file(
let large_row_size = large_block_size * data_shards;
while remaining >= large_row_size as i64 {
run.encode_row(offset, large_block_size)?;
encode_data(
dat_file,
offset,
large_block_size,
rs,
&mut buffers,
shards,
builders,
data_shards,
)?;
offset += large_row_size as u64;
remaining -= large_row_size as i64;
}
@@ -764,7 +753,16 @@ pub(crate) fn encode_dat_file(
while remaining > 0 {
let to_process = remaining.min(small_row_size as i64);
run.encode_row(offset, small_block_size)?;
encode_data(
dat_file,
offset,
small_block_size,
rs,
&mut buffers,
shards,
builders,
data_shards,
)?;
offset += to_process as u64;
remaining -= to_process;
}
@@ -772,66 +770,102 @@ pub(crate) fn encode_dat_file(
Ok(())
}
/// Everything one encode run streams through: the source .dat, the codec, a
/// buffer per shard, and the per-shard file and checksum sinks.
struct EncodeRun<'a> {
dat_file: &'a File,
rs: &'a ReedSolomon,
buffers: &'a mut [Vec<u8>],
shards: &'a mut [EcVolumeShard],
builders: &'a mut [ShardChecksumBuilder],
/// Encode one row of blocks, streaming it in ENCODE_BUFFER_SIZE sub-batches so
/// arbitrarily large blocks never require block-sized allocations. Mirrors
/// Go's encodeData.
#[expect(clippy::too_many_arguments)]
fn encode_data(
dat_file: &File,
row_offset: u64,
block_size: usize,
rs: &ReedSolomon,
buffers: &mut [Vec<u8>],
shards: &mut [EcVolumeShard],
builders: &mut [ShardChecksumBuilder],
data_shards: usize,
) -> io::Result<()> {
let buffer_size = buffers[0].len();
if !block_size.is_multiple_of(buffer_size) {
return Err(io::Error::new(
io::ErrorKind::InvalidInput,
format!(
"unexpected block size {} buffer size {}",
block_size, buffer_size
),
));
}
let batch_count = block_size / buffer_size;
for b in 0..batch_count {
encode_one_batch(
dat_file,
row_offset + (b * buffer_size) as u64,
block_size,
rs,
buffers,
shards,
builders,
data_shards,
)?;
}
Ok(())
}
impl EncodeRun<'_> {
/// Encode one row of blocks, streaming it in ENCODE_BUFFER_SIZE sub-batches
/// so arbitrarily large blocks never require block-sized allocations.
/// Mirrors Go's encodeData.
fn encode_row(&mut self, row_offset: u64, block_size: usize) -> io::Result<()> {
let buffer_size = self.buffers[0].len();
if !block_size.is_multiple_of(buffer_size) {
return Err(io::Error::new(
io::ErrorKind::InvalidInput,
format!(
"unexpected block size {} buffer size {}",
block_size, buffer_size
),
));
}
let batch_count = block_size / buffer_size;
for b in 0..batch_count {
self.encode_one_batch(row_offset + (b * buffer_size) as u64, block_size)?;
}
Ok(())
/// Encode one sub-batch: the same buffer-sized slice of every shard's block in
/// this row. Mirrors Go's encodeDataOneBatch.
#[expect(clippy::too_many_arguments)]
fn encode_one_batch(
dat_file: &File,
offset: u64,
block_size: usize,
rs: &ReedSolomon,
buffers: &mut [Vec<u8>],
shards: &mut [EcVolumeShard],
builders: &mut [ShardChecksumBuilder],
data_shards: usize,
) -> io::Result<()> {
// Read data shards from the .dat file, zero-filling past EOF — the buffers
// are reused across batches, so the tail must be cleared explicitly.
for (i, buf) in buffers[..data_shards].iter_mut().enumerate() {
let read_offset = offset + (i * block_size) as u64;
let n = read_at_most(dat_file, buf, read_offset)?;
buf[n..].fill(0);
}
/// Encode one sub-batch: the same buffer-sized slice of every shard's block
/// in this row. Mirrors Go's encodeDataOneBatch.
fn encode_one_batch(&mut self, offset: u64, block_size: usize) -> io::Result<()> {
// Read data shards from the .dat file, zero-filling past EOF — the
// buffers are reused across batches, so the tail must be cleared
// explicitly.
for (i, buf) in self.buffers[..self.data_shards].iter_mut().enumerate() {
let read_offset = offset + (i * block_size) as u64;
let n = crate::storage::io::read_full_at(self.dat_file, buf, read_offset)?;
buf[n..].fill(0);
}
// Encode parity shards
rs.encode(&mut *buffers)
.map_err(|e| io::Error::other(format!("reed-solomon encode: {:?}", e)))?;
// Encode parity shards
self.rs
.encode(&mut *self.buffers)
.map_err(|e| io::Error::other(format!("reed-solomon encode: {:?}", e)))?;
// Write all shard buffers to files and feed the same bytes to each
// shard's bitrot checksum builder, keeping covered_size == on-disk
// length.
for (i, buf) in self.buffers.iter().enumerate() {
self.shards[i].write_all(buf)?;
self.builders[i].write(buf);
}
Ok(())
// Write all shard buffers to files and feed the same bytes to each
// shard's bitrot checksum builder, keeping covered_size == on-disk length.
for (i, buf) in buffers.iter().enumerate() {
shards[i].write_all(buf)?;
builders[i].write(buf);
}
Ok(())
}
/// Read into `buf` at `offset` until it is full or EOF; returns bytes read.
fn read_at_most(dat_file: &File, buf: &mut [u8], offset: u64) -> io::Result<usize> {
let mut n = 0;
while n < buf.len() {
#[cfg(unix)]
let r = {
use std::os::unix::fs::FileExt;
dat_file.read_at(&mut buf[n..], offset + n as u64)?
};
#[cfg(not(unix))]
let r = {
let mut f = dat_file.try_clone()?;
f.seek(SeekFrom::Start(offset + n as u64))?;
f.read(&mut buf[n..])?
};
if r == 0 {
break;
}
n += r;
}
Ok(n)
}
#[cfg(test)]
@@ -839,7 +873,7 @@ mod tests {
use super::*;
use crate::storage::needle::needle::Needle;
use crate::storage::needle_map::NeedleMapKind;
use crate::storage::volume::{Volume, VolumeSpec};
use crate::storage::volume::Volume;
use tempfile::TempDir;
#[test]
@@ -851,9 +885,13 @@ mod tests {
let mut v = Volume::new(
dir,
dir,
"",
VolumeId(1),
NeedleMapKind::InMemory,
&VolumeSpec::default(),
None,
None,
0,
Version::current(),
)
.unwrap();
@@ -898,9 +936,13 @@ mod tests {
let mut v = Volume::new(
dir,
dir,
"",
VolumeId(1),
NeedleMapKind::InMemory,
&VolumeSpec::default(),
None,
None,
0,
Version::current(),
)
.unwrap();
for i in 1..=n {
@@ -973,9 +1015,13 @@ mod tests {
let mut v = Volume::new(
&dir,
&dir,
"",
VolumeId(1),
NeedleMapKind::InMemory,
&VolumeSpec::default(),
None,
None,
0,
Version::current(),
)
.unwrap();
for i in 1..=20 {
@@ -1007,10 +1053,7 @@ mod tests {
let victim = format!("{}/1.ec03", dir);
let full = std::fs::metadata(&victim).unwrap().len();
assert!(full > 0, "encoded shard should be non-empty");
let f = std::fs::OpenOptions::new()
.write(true)
.open(&victim)
.unwrap();
let f = std::fs::OpenOptions::new().write(true).open(&victim).unwrap();
f.set_len(full / 2).unwrap();
drop(f);
@@ -1164,15 +1207,19 @@ mod tests {
#[test]
fn test_rebuild_ecx_file_uniform_layout() {
use crate::storage::needle_map::NeedleMapKind;
use crate::storage::volume::{Volume, VolumeSpec};
use crate::storage::volume::Volume;
let tmp = TempDir::new().unwrap();
let dir = tmp.path().to_str().unwrap().to_string();
let mut v = Volume::new(
&dir,
&dir,
"",
VolumeId(2),
NeedleMapKind::InMemory,
&VolumeSpec::default(),
None,
None,
0,
Version::current(),
)
.unwrap();
for i in 1u64..=12 {
@@ -1202,10 +1249,7 @@ mod tests {
rebuild_ecx_file(&dir, "", VolumeId(2), 10, block_size, 0, &[]).unwrap();
let rebuilt = std::fs::read(&ecx_path).unwrap();
assert_eq!(
canonical, rebuilt,
"rebuilt .ecx must match the encode-time .ecx"
);
assert_eq!(canonical, rebuilt, "rebuilt .ecx must match the encode-time .ecx");
}
// A truncated data shard must FAIL the .ecx rebuild, not publish the
@@ -1213,15 +1257,19 @@ mod tests {
#[test]
fn test_rebuild_ecx_file_fails_on_truncated_shard() {
use crate::storage::needle_map::NeedleMapKind;
use crate::storage::volume::{Volume, VolumeSpec};
use crate::storage::volume::Volume;
let tmp = TempDir::new().unwrap();
let dir = tmp.path().to_str().unwrap().to_string();
let mut v = Volume::new(
&dir,
&dir,
"",
VolumeId(3),
NeedleMapKind::InMemory,
&VolumeSpec::default(),
None,
None,
0,
Version::current(),
)
.unwrap();
for i in 1u64..=12 {
@@ -1319,9 +1367,13 @@ mod tests {
let mut v = Volume::new(
dat_dir,
idx_dir,
"",
VolumeId(1),
NeedleMapKind::InMemory,
&VolumeSpec::default(),
None,
None,
0,
Version::current(),
)
.unwrap();
@@ -1399,9 +1451,13 @@ mod tests {
let mut v = Volume::new(
dat_dir,
idx_dir,
"",
VolumeId(1),
NeedleMapKind::InMemory,
&VolumeSpec::default(),
None,
None,
0,
Version::current(),
)
.unwrap();
@@ -1433,9 +1489,13 @@ mod tests {
let mut v = Volume::new(
dir,
dir,
"",
vid,
NeedleMapKind::InMemory,
&VolumeSpec::default(),
None,
None,
0,
Version::current(),
)
.unwrap();
for i in 1..=8 {
@@ -1526,86 +1586,4 @@ mod tests {
details
);
}
#[test]
fn test_encode_drops_tombstone_last_wins() {
use crate::storage::idx;
use crate::storage::types::{NeedleId, Offset, Size, TOMBSTONE_FILE_SIZE};
let tmp = tempfile::TempDir::new().unwrap();
let dir = tmp.path().to_str().unwrap();
let idx_path = format!("{}/t.idx", dir);
let ecx_path = format!("{}/t.ecx", dir);
let key = NeedleId(12345);
{
let mut f = std::fs::File::create(&idx_path).unwrap();
idx::write_index_entry(&mut f, key, Offset::from_actual_offset(1024), Size(100))
.unwrap();
idx::write_index_entry(&mut f, key, Offset::default(), TOMBSTONE_FILE_SIZE).unwrap();
}
super::write_sorted_ecx_from_idx(&idx_path, &ecx_path).unwrap();
let mut found = false;
{
let mut f = std::fs::File::open(&ecx_path).unwrap();
idx::walk_index_file(&mut f, 0, |k, _o, _s| {
if k == key {
found = true;
}
Ok(())
})
.unwrap();
}
assert!(!found, "tombstoned key must not appear in .ecx");
let idx2 = format!("{}/t2.idx", dir);
let ecx2 = format!("{}/t2.ecx", dir);
{
let mut f = std::fs::File::create(&idx2).unwrap();
idx::write_index_entry(&mut f, key, Offset::default(), TOMBSTONE_FILE_SIZE).unwrap();
idx::write_index_entry(&mut f, key, Offset::from_actual_offset(2048), Size(200))
.unwrap();
}
super::write_sorted_ecx_from_idx(&idx2, &ecx2).unwrap();
let mut found2 = false;
{
let mut f = std::fs::File::open(&ecx2).unwrap();
idx::walk_index_file(&mut f, 0, |k, o, s| {
if k == key {
found2 = true;
assert_eq!(o.to_actual_offset(), 2048);
assert_eq!(s, Size(200));
}
Ok(())
})
.unwrap();
}
assert!(found2, "re-created key must appear live");
// Zero offset with non-negative size is also a deletion: Go
// readNeedleMap (`if !offset.IsZero() && !size.IsDeleted() { Set }
// else { Delete }`) and CompactNeedleMap::load_from_idx both treat
// it as deleted. Encode must drop it too, or the .ecx live-map
// mismatches replay.
let idx3 = format!("{}/t3.idx", dir);
let ecx3 = format!("{}/t3.ecx", dir);
{
let mut f = std::fs::File::create(&idx3).unwrap();
idx::write_index_entry(&mut f, key, Offset::from_actual_offset(1024), Size(100))
.unwrap();
idx::write_index_entry(&mut f, key, Offset::default(), Size(0)).unwrap();
}
super::write_sorted_ecx_from_idx(&idx3, &ecx3).unwrap();
let mut found3 = false;
{
let mut f = std::fs::File::open(&ecx3).unwrap();
idx::walk_index_file(&mut f, 0, |k, _o, _s| {
if k == key {
found3 = true;
}
Ok(())
})
.unwrap();
}
assert!(
!found3,
"zero-offset row must not appear in .ecx even with non-negative size"
);
}
}
@@ -16,20 +16,6 @@ pub const ERASURE_CODING_SMALL_BLOCK_SIZE: usize = 1024 * 1024; // 1MB
pub type ShardId = u8;
/// Validate a wire shard id. `ShardId` is `u8` but only 0..MAX_SHARD_COUNT are valid.
/// Rejects 256 (would truncate to 0 and delete .ec00) and 270 (would alias 14).
pub fn shard_id_try_from(v: u32) -> Result<ShardId, String> {
if v < MAX_SHARD_COUNT as u32 {
Ok(v as ShardId)
} else {
Err(format!(
"invalid shard id {} (max {})",
v,
MAX_SHARD_COUNT - 1
))
}
}
/// A single erasure-coded shard file.
pub struct EcVolumeShard {
pub volume_id: VolumeId,
@@ -87,14 +73,28 @@ impl EcVolumeShard {
Ok(())
}
/// Read data at a specific offset, filling `buf` unless the shard ends first.
/// Read data at a specific offset.
pub fn read_at(&self, buf: &mut [u8], offset: u64) -> io::Result<usize> {
let file = self
.ecd_file
.as_ref()
.ok_or_else(|| io::Error::other("shard file not open"))?;
crate::storage::io::read_full_at(file, buf, offset)
#[cfg(unix)]
{
use std::os::unix::fs::FileExt;
file.read_at(buf, offset)
}
#[cfg(not(unix))]
{
use std::io::{Read, Seek, SeekFrom};
// File::read_at is unix-only; fall back to seek + read.
// We need a mutable reference for seek/read, so clone the handle.
let mut f = file.try_clone()?;
f.seek(SeekFrom::Start(offset))?;
f.read(buf)
}
}
/// Write data to the shard file (appends).
@@ -195,104 +195,6 @@ impl ShardBits {
}
}
/// Parses the generation of a 2PC-staged `<base>.v<N>` file: `None` means the
/// name is not a generation file of `base`.
pub fn ec_file_generation(name: &str, base: &str) -> Option<u32> {
let suffix = name.strip_prefix(&format!("{}.v", base))?;
match suffix.parse::<u32>() {
Ok(g) if g > 0 => Some(g),
_ => None,
}
}
/// Removes 2PC generation files staged under `base`:
/// `<base>.ecNN.v<N>`, `<base>.ecx.v<N>`, `<base>.ecj.v<N>`, `<base>.ecsum.v<N>`
/// and `<base>.vif.v<N>`. `generations_older_than == 0` removes every
/// generation; otherwise only generations strictly below it. Returns the
/// first real removal failure. Mirrors Go's `RemoveEcGenerationFiles`.
pub fn remove_ec_generation_files(base: &str, generations_older_than: u32) -> io::Result<()> {
let path = std::path::Path::new(base);
let (Some(parent), Some(fname)) = (path.parent(), path.file_name()) else {
return Ok(());
};
let ec_prefix = format!("{}.ec", fname.to_string_lossy());
let vif_name = format!("{}.vif", fname.to_string_lossy());
let mut first_err: Option<io::Error> = None;
let mut record = |res: io::Result<()>| {
if let Err(e) = res
&& first_err.is_none()
{
first_err = Some(e);
}
};
match fs::read_dir(parent) {
Ok(entries) => {
for entry in entries {
let entry = match entry {
Ok(entry) => entry,
Err(e) => {
// A skipped entry means an incomplete sweep; report it
// instead of pretending the cleanup finished.
record(Err(e));
continue;
}
};
let name = entry.file_name().to_string_lossy().into_owned();
let Some((artifact, _)) = name.rsplit_once(".v") else {
continue;
};
if artifact != vif_name && !artifact.starts_with(&ec_prefix) {
continue;
}
let Some(generation) = ec_file_generation(&name, artifact) else {
continue;
};
if generations_older_than > 0 && generation >= generations_older_than {
continue;
}
record(match fs::remove_file(entry.path()) {
Err(e) if e.kind() != io::ErrorKind::NotFound => Err(e),
_ => Ok(()),
});
}
}
Err(e) if e.kind() != io::ErrorKind::NotFound => record(Err(e)),
Err(_) => {}
}
match first_err {
Some(e) => Err(e),
None => Ok(()),
}
}
/// Removes every staged generation `<shard_file>.v<N>` of one shard file.
/// Returns true when at least one generation file was removed.
pub fn remove_ec_shard_generations(shard_file: &str) -> io::Result<bool> {
let path = std::path::Path::new(shard_file);
let (Some(parent), Some(fname)) = (path.parent(), path.file_name()) else {
return Ok(false);
};
let fname = fname.to_string_lossy().into_owned();
let mut removed = false;
match fs::read_dir(parent) {
Ok(entries) => {
for entry in entries {
let entry = entry?;
let name = entry.file_name().to_string_lossy().into_owned();
if ec_file_generation(&name, &fname).is_some() {
match fs::remove_file(entry.path()) {
Err(e) if e.kind() != io::ErrorKind::NotFound => return Err(e),
_ => removed = true,
}
}
}
}
Err(e) if e.kind() != io::ErrorKind::NotFound => return Err(e),
Err(_) => {}
}
Ok(removed)
}
#[cfg(test)]
mod tests {
use super::*;
@@ -349,42 +251,4 @@ mod tests {
let shard = EcVolumeShard::new("/data", "", VolumeId(7), 13);
assert_eq!(shard.file_name(), "/data/7.ec13");
}
#[test]
fn test_shard_id_try_from_u32_rejects_overflow() {
use super::{MAX_SHARD_COUNT, shard_id_try_from};
assert_eq!(shard_id_try_from(0).unwrap(), 0u8);
assert_eq!(shard_id_try_from(14).unwrap(), 14u8);
assert_eq!(shard_id_try_from(31).unwrap(), 31u8);
assert!(shard_id_try_from(32).is_err());
assert!(shard_id_try_from(256).is_err());
assert!(shard_id_try_from(270).is_err());
assert!(shard_id_try_from(u32::MAX).is_err());
assert_eq!(MAX_SHARD_COUNT, 32);
}
#[test]
fn test_shard_batch_validation_is_atomic_rejects_without_partial_prefix() {
use super::shard_id_try_from;
// The mount/unmount handlers pre-validate the ENTIRE req.shard_ids into
// a Vec<ShardId> BEFORE acquiring the write lock or mutating any EC
// state. This test pins the validation half of that contract at the
// unit level: a batch like [0, 32] must fail as a whole, so by
// construction no validated prefix (e.g. shard 0) is ever applied.
// The handler-level tests below assert the no-state-change half.
let batch = vec![0u32, 32u32];
let validated: Result<Vec<_>, _> =
batch.iter().map(|&sid| shard_id_try_from(sid)).collect();
assert!(
validated.is_err(),
"batch {:?} must be rejected as a whole",
batch
);
// A fully-valid batch still validates cleanly.
let ok: Result<Vec<_>, _> = [0u32, 1u32, 13u32]
.iter()
.map(|&sid| shard_id_try_from(sid))
.collect();
assert_eq!(ok.unwrap(), vec![0u8, 1u8, 13u8]);
}
}
File diff suppressed because it is too large Load Diff
@@ -1,290 +0,0 @@
//! Set-union merge of `.ecj` deletion journals for EC shard copy / index
//! recovery. Mirrors Go's `weed/storage/erasure_coding/ecj_merge.go`.
//!
//! An EC volume's deletion journal (`<vid>.ecj`) is a *set* of deleted needle
//! ids stored as 8-byte big-endian records. Shard copy and index recovery fold
//! a peer's journal into the local one; they must append only the ids the
//! local journal lacks, or every `ec_balance` round trip doubles the file.
//!
//! The journal is only ever appended to, never replaced: a mounted `EcVolume`
//! holds it open, and a rename would leave that handle writing to an unlinked
//! inode, losing every later delete at the next mount. A mounted volume merges
//! through [`EcVolume::merge_journal`](super::ec_volume::EcVolume::merge_journal);
//! [`append_ecj_ids`] is for a journal no volume has open.
use std::collections::HashSet;
use std::fs::{self, OpenOptions};
use std::io::{self, Read, Seek, SeekFrom, Write};
use std::path::Path;
use crate::storage::types::{NEEDLE_ID_SIZE, NeedleId};
use crate::storage::volume::fsync_dir;
use crate::storage::volume_open::open_volume_file;
/// Bytes per read when scanning a journal; a multiple of `NEEDLE_ID_SIZE`.
const ECJ_READ_CHUNK_BYTES: usize = 1 << 20;
/// Decodes `.ecj` records from a byte stream split at arbitrary boundaries (a
/// CopyFile stream, chunked reads), collecting the distinct ids. Memory
/// follows the number of distinct ids, not the journal's length. A trailing
/// partial record is never decoded.
#[derive(Default)]
pub(crate) struct EcjIdDecoder {
ids: HashSet<NeedleId>,
partial: [u8; NEEDLE_ID_SIZE],
pending: usize,
}
impl EcjIdDecoder {
pub(crate) fn push(&mut self, mut bytes: &[u8]) {
if self.pending > 0 {
let n = (NEEDLE_ID_SIZE - self.pending).min(bytes.len());
self.partial[self.pending..self.pending + n].copy_from_slice(&bytes[..n]);
self.pending += n;
bytes = &bytes[n..];
if self.pending < NEEDLE_ID_SIZE {
return;
}
self.ids.insert(NeedleId::from_bytes(&self.partial));
self.pending = 0;
}
let mut records = bytes.chunks_exact(NEEDLE_ID_SIZE);
for record in &mut records {
self.ids.insert(NeedleId::from_bytes(record));
}
let rest = records.remainder();
self.partial[..rest.len()].copy_from_slice(rest);
self.pending = rest.len();
}
pub(crate) fn into_ids(self) -> HashSet<NeedleId> {
self.ids
}
}
/// Read the distinct ids of the journal at `path` in bounded chunks. A missing
/// file reads as empty. Also returns the whole-record length read; a torn
/// trailing partial record is excluded from it.
pub(crate) fn read_ecj_ids(path: &str) -> io::Result<(HashSet<NeedleId>, u64)> {
let mut file = match fs::File::open(path) {
Ok(f) => f,
Err(e) if e.kind() == io::ErrorKind::NotFound => return Ok((HashSet::new(), 0)),
Err(e) => return Err(e),
};
let len = file.metadata()?.len();
let size = len - len % NEEDLE_ID_SIZE as u64;
let mut decoder = EcjIdDecoder::default();
let mut buf = vec![0u8; (ECJ_READ_CHUNK_BYTES as u64).min(size) as usize];
let mut off = 0u64;
while off < size {
let want = (buf.len() as u64).min(size - off) as usize;
file.read_exact(&mut buf[..want])?;
decoder.push(&buf[..want]);
off += want as u64;
}
Ok((decoder.into_ids(), size))
}
/// The ids of `incoming` that `has` does not report, ascending, so a merge
/// appends deterministic output.
pub(crate) fn ecj_delta(
incoming: &HashSet<NeedleId>,
has: impl Fn(&NeedleId) -> bool,
) -> Vec<NeedleId> {
let mut delta: Vec<NeedleId> = incoming.iter().copied().filter(|id| !has(id)).collect();
delta.sort_unstable();
delta
}
pub(crate) fn encode_ecj_ids(ids: &[NeedleId]) -> Vec<u8> {
let mut buf = vec![0u8; ids.len() * NEEDLE_ID_SIZE];
for (record, id) in buf.chunks_exact_mut(NEEDLE_ID_SIZE).zip(ids) {
id.to_bytes(record);
}
buf
}
/// Append to the journal at `path` the ids of `incoming` that `local` lacks,
/// in one write and one fsync, returning how many were added. `local` and
/// `size` come from [`read_ecj_ids`] on the same path: if the journal's
/// whole-record length is no longer `size`, returns `Ok(None)` so the caller
/// re-reads. A torn tail past `size` is truncated first so the new records
/// stay aligned.
pub(crate) fn append_ecj_ids(
path: &str,
local: &HashSet<NeedleId>,
incoming: &HashSet<NeedleId>,
size: u64,
) -> io::Result<Option<usize>> {
let delta = ecj_delta(incoming, |id| local.contains(id));
if delta.is_empty() {
return Ok(Some(0));
}
let created = !Path::new(path).exists();
let mut file = open_volume_file(OpenOptions::new().read(true).write(true).create(true), path)?;
let len = file.metadata()?.len();
if len - len % NEEDLE_ID_SIZE as u64 != size {
return Ok(None);
}
if len != size {
file.set_len(size)?;
}
let appended = file
.seek(SeekFrom::Start(size))
.and_then(|_| file.write_all(&encode_ecj_ids(&delta)))
.and_then(|_| file.sync_all());
if let Err(e) = appended {
let _ = file.set_len(size);
return Err(e);
}
if created {
fsync_dir(path)?;
}
Ok(Some(delta.len()))
}
#[cfg(test)]
mod tests {
use super::*;
fn ids(v: &[u64]) -> HashSet<NeedleId> {
v.iter().map(|&id| NeedleId(id)).collect()
}
fn bytes(v: &[u64]) -> Vec<u8> {
encode_ecj_ids(&v.iter().map(|&id| NeedleId(id)).collect::<Vec<_>>())
}
fn records(path: &str) -> Vec<u64> {
let data = fs::read(path).expect("read ecj");
assert_eq!(data.len() % NEEDLE_ID_SIZE, 0, "journal must stay aligned");
data.chunks_exact(NEEDLE_ID_SIZE)
.map(|c| NeedleId::from_bytes(c).0)
.collect()
}
/// Run the unmounted merge the way the server does: read, then append.
fn merge_file(path: &str, incoming: &HashSet<NeedleId>) -> usize {
let (local, size) = read_ecj_ids(path).expect("read");
append_ecj_ids(path, &local, incoming, size)
.expect("append")
.expect("journal unchanged")
}
#[test]
fn decoder_handles_records_split_across_chunks() {
let mut stream = bytes(&[1, 2, 3, 2, 0x0102030405060708]);
stream.extend_from_slice(&[9, 9, 9]);
for chunk in 1..=stream.len() {
let mut d = EcjIdDecoder::default();
for piece in stream.chunks(chunk) {
d.push(piece);
}
assert_eq!(
d.into_ids(),
ids(&[1, 2, 3, 0x0102030405060708]),
"chunk {}",
chunk
);
}
}
#[test]
fn read_ecj_ids_dedups_and_ignores_torn_tail() {
let dir = tempfile::tempdir().expect("tempdir");
let missing = dir.path().join("missing.ecj");
let (got, size) = read_ecj_ids(missing.to_str().unwrap()).expect("read");
assert!(got.is_empty());
assert_eq!(size, 0);
let torn = dir.path().join("torn.ecj");
let mut data = bytes(&[1, 2, 1]);
data.extend_from_slice(&[7, 7, 7]);
fs::write(&torn, data).unwrap();
let (got, size) = read_ecj_ids(torn.to_str().unwrap()).expect("read");
assert_eq!(got, ids(&[1, 2]));
assert_eq!(size, 3 * NEEDLE_ID_SIZE as u64);
// A bloated journal repeating a few ids across several read chunks
// keeps only the distinct ids.
let bloated = dir.path().join("bloated.ecj");
let data: Vec<u8> = (0..300_000u64).flat_map(|i| bytes(&[i % 3])).collect();
fs::write(&bloated, &data).unwrap();
let (got, size) = read_ecj_ids(bloated.to_str().unwrap()).expect("read");
assert_eq!(got, ids(&[0, 1, 2]));
assert_eq!(size, data.len() as u64);
}
#[test]
fn appends_only_missing_ids() {
let dir = tempfile::tempdir().expect("tempdir");
let path = dir.path().join("vol.ecj");
let path = path.to_str().unwrap();
fs::write(path, bytes(&[1, 2, 3])).unwrap();
assert_eq!(merge_file(path, &ids(&[3, 4])), 1);
assert_eq!(records(path), vec![1, 2, 3, 4]);
}
#[test]
fn round_trip_stays_constant() {
// A->B->A->B 20 times: the old append path doubled the journal each trip.
let dir = tempfile::tempdir().expect("tempdir");
let a = dir.path().join("a.ecj");
let b = dir.path().join("b.ecj");
let (a, b) = (a.to_str().unwrap(), b.to_str().unwrap());
fs::write(a, bytes(&[1, 2])).unwrap();
fs::write(b, bytes(&[2, 3])).unwrap();
for i in 0..20 {
let (src, dst) = if i % 2 == 0 { (a, b) } else { (b, a) };
let (incoming, _) = read_ecj_ids(src).expect("read");
merge_file(dst, &incoming);
}
for path in [a, b] {
let mut got = records(path);
got.sort_unstable();
assert_eq!(got, vec![1, 2, 3]);
}
}
#[test]
fn nothing_new_leaves_journal_alone() {
let dir = tempfile::tempdir().expect("tempdir");
let missing = dir.path().join("missing.ecj");
assert_eq!(merge_file(missing.to_str().unwrap(), &ids(&[])), 0);
assert!(
!missing.exists(),
"an empty merge must not create a journal"
);
let path = dir.path().join("vol.ecj");
let path = path.to_str().unwrap();
fs::write(path, bytes(&[1, 2])).unwrap();
assert_eq!(merge_file(path, &ids(&[2, 1])), 0);
assert_eq!(records(path), vec![1, 2]);
}
#[test]
fn repairs_torn_tail() {
let dir = tempfile::tempdir().expect("tempdir");
let path = dir.path().join("vol.ecj");
let path = path.to_str().unwrap();
let mut data = bytes(&[1, 2]);
data.extend_from_slice(&[9, 9, 9]);
fs::write(path, data).unwrap();
assert_eq!(merge_file(path, &ids(&[3])), 1);
assert_eq!(records(path), vec![1, 2, 3]);
}
#[test]
fn rejects_changed_journal() {
let dir = tempfile::tempdir().expect("tempdir");
let path = dir.path().join("vol.ecj");
let path = path.to_str().unwrap();
fs::write(path, bytes(&[1])).unwrap();
let (local, size) = read_ecj_ids(path).expect("read");
fs::write(path, bytes(&[1, 5])).unwrap();
let outcome = append_ecj_ids(path, &local, &ids(&[2]), size).expect("append");
assert_eq!(outcome, None);
assert_eq!(records(path), vec![1, 5]);
}
}
@@ -1,326 +0,0 @@
//! Process-wide coordination of everything that touches one `.ecj` path.
//!
//! Mount-time compaction replaces a deletion journal with a new inode. That is
//! only safe while nothing else in this process can write the old one:
//!
//! - **Holders** are mounted `EcVolume`s with an append handle on the path. A
//! store holds one `EcVolume` per disk location, and a shard mount or a
//! cross-disk reconcile can point one disk's volume at another disk's
//! `.ecj`, so several holders of one path are normal. A holder that keeps
//! appending to a replaced inode acknowledges deletes that are gone at the
//! next mount.
//! - **Writers** append to or replace the path by name without holding it
//! open across calls: `ReceiveFile` of an EC `.ecj`, and the unmounted
//! append in `merge_ec_journal`, which `VolumeEcShardsCopy` and EC index
//! recovery funnel a peer's journal through. Bytes they write after the
//! compactor sized the journal would be dropped by the rename.
//!
//! Compaction therefore runs only while its caller is the sole holder and no
//! writer is active, and while it runs no holder may open the path and no
//! writer may start. Both wait instead; a compaction rewrites only the distinct
//! id set, so the wait is short.
//!
//! No writer active at the reservation is not enough: one that ran while the
//! holder loaded the journal, or after, and has finished may have rewritten it
//! in place to the same length (`ReceiveFile` truncates and refills), which
//! the inode-and-size re-check cannot see. So each writer bumps the path's
//! write generation as it starts, and a holder may compact only if no writer
//! was active when it registered and the generation has not moved since.
//!
//! Paths are keyed by their canonical parent directory, so two disk locations
//! that spell one directory differently still meet here.
use std::collections::HashMap;
use std::path::{Path, PathBuf};
use std::sync::{Condvar, LazyLock, Mutex, MutexGuard};
#[derive(Default)]
struct PathState {
holders: usize,
writers: usize,
compacting: bool,
/// Writers that have started on the path. Lives as long as the entry,
/// which a registered holder keeps.
write_gen: u64,
}
impl PathState {
fn idle(&self) -> bool {
self.holders == 0 && self.writers == 0 && !self.compacting
}
}
struct Registry {
paths: Mutex<HashMap<PathBuf, PathState>>,
changed: Condvar,
}
static REGISTRY: LazyLock<Registry> = LazyLock::new(|| Registry {
paths: Mutex::new(HashMap::new()),
changed: Condvar::new(),
});
fn lock() -> MutexGuard<'static, HashMap<PathBuf, PathState>> {
// The critical sections only adjust counters and cannot panic midway, so
// a poisoned lock still guards consistent state.
REGISTRY.paths.lock().unwrap_or_else(|e| e.into_inner())
}
/// Canonical key for `path`: its resolved parent directory joined with the
/// file name. The file itself may not exist yet (a copy creates it), so only
/// the directory is resolved.
fn key_for(path: &str) -> PathBuf {
let p = Path::new(path);
let (Some(parent), Some(name)) = (p.parent(), p.file_name()) else {
return std::path::absolute(p).unwrap_or_else(|_| p.to_path_buf());
};
let parent = if parent.as_os_str().is_empty() {
Path::new(".")
} else {
parent
};
let dir = std::fs::canonicalize(parent)
.or_else(|_| std::path::absolute(parent))
.unwrap_or_else(|_| parent.to_path_buf());
dir.join(name)
}
/// Block until no compaction is running on `key`, then apply `f` to its state.
fn update_when_not_compacting<R>(key: &Path, f: impl FnOnce(&mut PathState) -> R) -> R {
let mut paths = lock();
while paths.get(key).is_some_and(|s| s.compacting) {
paths = REGISTRY
.changed
.wait(paths)
.unwrap_or_else(|e| e.into_inner());
}
f(paths.entry(key.to_path_buf()).or_default())
}
fn release(key: &Path, f: impl FnOnce(&mut PathState)) {
let mut paths = lock();
if let Some(state) = paths.get_mut(key) {
f(state);
if state.idle() {
paths.remove(key);
}
}
drop(paths);
REGISTRY.changed.notify_all();
}
/// A mounted `EcVolume`'s registration as a holder of its `.ecj`. Taken before
/// the journal is opened and released when dropped.
pub(crate) struct EcjHold {
key: PathBuf,
/// The path's write generation when the hold was taken, and whether a
/// writer was active then. Taken before the journal is opened and loaded,
/// so they cover every write the load might have missed.
write_gen: u64,
writer_at_start: bool,
}
impl EcjHold {
/// Register as a holder of `ecj_path`, first waiting out any compaction in
/// progress so the handle opened afterwards is on the final inode.
pub(crate) fn acquire(ecj_path: &str) -> Self {
let key = key_for(ecj_path);
let (write_gen, writer_at_start) = update_when_not_compacting(&key, |s| {
s.holders += 1;
(s.write_gen, s.writers > 0)
});
EcjHold {
key,
write_gen,
writer_at_start,
}
}
/// Reserve the path for a compaction, or `None` when another holder or an
/// active writer could still reach the current inode, or when a writer
/// has run on the path since the hold was taken, so the journal may no
/// longer be what the holder loaded.
pub(crate) fn try_begin_compaction(&self) -> Option<EcjCompaction> {
let mut paths = lock();
let state = paths.get_mut(&self.key)?;
if state.holders != 1 || state.writers != 0 || state.compacting {
return None;
}
if self.writer_at_start || state.write_gen != self.write_gen {
return None;
}
state.compacting = true;
Some(EcjCompaction {
key: self.key.clone(),
})
}
}
impl Drop for EcjHold {
fn drop(&mut self) {
release(&self.key, |s| s.holders = s.holders.saturating_sub(1));
}
}
/// An exclusive reservation of a `.ecj` path for compaction. Holders and
/// writers wait until it is dropped.
pub(crate) struct EcjCompaction {
key: PathBuf,
}
impl Drop for EcjCompaction {
fn drop(&mut self) {
release(&self.key, |s| s.compacting = false);
}
}
/// An out-of-band writer (shard copy, index recovery, `ReceiveFile`) on a
/// `.ecj` path.
/// Compaction does not start while one is alive.
pub(crate) struct EcjWrite {
key: PathBuf,
}
impl Drop for EcjWrite {
fn drop(&mut self) {
release(&self.key, |s| s.writers = s.writers.saturating_sub(1));
}
}
/// Register as a writer of `ecj_path`, waiting out any compaction in progress.
/// Blocks; async callers use [`begin_ecj_write_async`].
pub(crate) fn begin_ecj_write(ecj_path: &str) -> EcjWrite {
let key = key_for(ecj_path);
update_when_not_compacting(&key, |s| {
s.writers += 1;
s.write_gen += 1;
});
EcjWrite { key }
}
/// [`begin_ecj_write`] for async handlers: the wait runs on the blocking pool
/// so a compaction in progress never stalls a runtime worker.
pub(crate) async fn begin_ecj_write_async(ecj_path: &str) -> EcjWrite {
let path = ecj_path.to_string();
match tokio::task::spawn_blocking(move || begin_ecj_write(&path)).await {
Ok(write) => write,
// Only a panic inside the registry lands here, and it leaves no count
// behind; registering inline is still correct, merely blocking.
Err(_) => begin_ecj_write(ecj_path),
}
}
#[cfg(test)]
mod tests {
use super::*;
use std::sync::mpsc;
use std::time::Duration;
use tempfile::TempDir;
fn ecj(dir: &TempDir) -> String {
dir.path().join("1.ecj").to_str().unwrap().to_string()
}
#[test]
fn sole_holder_may_compact() {
let dir = TempDir::new().unwrap();
let hold = EcjHold::acquire(&ecj(&dir));
assert!(hold.try_begin_compaction().is_some());
}
#[test]
fn second_holder_blocks_compaction() {
let dir = TempDir::new().unwrap();
let a = EcjHold::acquire(&ecj(&dir));
let b = EcjHold::acquire(&ecj(&dir));
assert!(a.try_begin_compaction().is_none());
drop(b);
assert!(a.try_begin_compaction().is_some());
}
#[test]
fn active_writer_blocks_compaction() {
let dir = TempDir::new().unwrap();
let hold = EcjHold::acquire(&ecj(&dir));
let w = begin_ecj_write(&ecj(&dir));
assert!(hold.try_begin_compaction().is_none());
drop(w);
// This hold loaded before the write; a later one may compact.
drop(hold);
let hold = EcjHold::acquire(&ecj(&dir));
assert!(hold.try_begin_compaction().is_some());
}
/// A writer that ran after the hold was taken, or was already running
/// then, may have changed the journal the holder loaded, even though it
/// has finished by the time compaction asks.
#[test]
fn finished_writer_since_hold_blocks_compaction() {
let dir = TempDir::new().unwrap();
let path = ecj(&dir);
let hold = EcjHold::acquire(&path);
drop(begin_ecj_write(&path));
assert!(
hold.try_begin_compaction().is_none(),
"a writer that started after the hold",
);
drop(hold);
let w = begin_ecj_write(&path);
let hold = EcjHold::acquire(&path);
drop(w);
assert!(
hold.try_begin_compaction().is_none(),
"a writer active when the hold was taken",
);
drop(hold);
let hold = EcjHold::acquire(&path);
assert!(
hold.try_begin_compaction().is_some(),
"no writer since the hold",
);
}
#[test]
fn differently_spelled_paths_share_one_key() {
let dir = TempDir::new().unwrap();
std::fs::create_dir(dir.path().join("sub")).unwrap();
let plain = ecj(&dir);
let dotted = dir
.path()
.join("sub")
.join("..")
.join("1.ecj")
.to_str()
.unwrap()
.to_string();
let a = EcjHold::acquire(&plain);
let _b = EcjHold::acquire(&dotted);
assert!(a.try_begin_compaction().is_none());
}
#[test]
fn writer_waits_for_compaction_to_finish() {
let dir = TempDir::new().unwrap();
let path = ecj(&dir);
let hold = EcjHold::acquire(&path);
let compaction = hold.try_begin_compaction().unwrap();
let (tx, rx) = mpsc::channel();
let p = path.clone();
let t = std::thread::spawn(move || {
let _w = begin_ecj_write(&p);
tx.send(()).unwrap();
});
assert!(
rx.recv_timeout(Duration::from_millis(100)).is_err(),
"a writer must not start while a compaction holds the path",
);
drop(compaction);
rx.recv_timeout(Duration::from_secs(5))
.expect("writer must proceed once the compaction ends");
t.join().unwrap();
}
}
@@ -9,11 +9,9 @@ pub mod ec_encoder;
pub mod ec_locate;
pub mod ec_shard;
pub mod ec_volume;
pub mod ecj_merge;
pub(crate) mod ecj_registry;
pub use ec_shard::{
DATA_SHARDS_COUNT, EcVolumeShard, MAX_SHARD_COUNT, MIN_TOTAL_DISKS, PARITY_SHARDS_COUNT,
ShardId, TOTAL_SHARDS_COUNT,
EcVolumeShard, ShardId, DATA_SHARDS_COUNT, MAX_SHARD_COUNT, MIN_TOTAL_DISKS,
PARITY_SHARDS_COUNT, TOTAL_SHARDS_COUNT,
};
pub use ec_volume::EcVolume;
+8 -136
View File
@@ -21,26 +21,12 @@ where
let mut buf = vec![0u8; NEEDLE_MAP_ENTRY_SIZE * ROWS_TO_READ];
loop {
// Fill the batch before decoding: `read` may return a count that is
// not a multiple of the entry size, and a split entry would misalign
// every later row. Go is immune: `ReadAt` fills or errors.
let mut count = 0;
let mut eof = false;
while count < buf.len() {
match reader.read(&mut buf[count..]) {
Ok(0) => {
eof = true;
break;
}
Ok(n) => count += n,
Err(ref e) if e.kind() == io::ErrorKind::Interrupted => continue,
Err(ref e) if e.kind() == io::ErrorKind::UnexpectedEof => {
eof = true;
break;
}
Err(e) => return Err(e),
}
}
let count = match reader.read(&mut buf) {
Ok(0) => return Ok(()),
Ok(n) => n,
Err(ref e) if e.kind() == io::ErrorKind::UnexpectedEof => return Ok(()),
Err(e) => return Err(e),
};
let mut i = 0;
while i + NEEDLE_MAP_ENTRY_SIZE <= count {
@@ -48,11 +34,6 @@ where
f(key, offset, size)?;
i += NEEDLE_MAP_ENTRY_SIZE;
}
// A trailing partial entry at EOF is ignored, as Go does on `io.EOF`.
if eof {
return Ok(());
}
}
}
@@ -76,7 +57,7 @@ pub fn check_index_file<R: Read + Seek>(
errs.push(format!("walk index file: {}", e));
}
entries.sort_by(|a, b| a.2.cmp(&b.2).then(a.3.0.cmp(&b.3.0)));
entries.sort_by(|a, b| a.2.cmp(&b.2).then(a.3 .0.cmp(&b.3 .0)));
// Offset-0 logical tombstones (remote-tier deletes) occupy no physical extent,
// so they cannot overlap anything — exclude them from the overlap check. They
@@ -196,111 +177,6 @@ mod tests {
data
}
/// Reader that hands back at most `chunk` bytes per `read`. 7 is coprime
/// with the 17-byte entry size, so nearly every read ends mid-entry. With
/// `interrupts`, every other call fails with `ErrorKind::Interrupted`.
struct ShortReader {
inner: Cursor<Vec<u8>>,
chunk: usize,
interrupts: bool,
interrupt_next: bool,
}
impl ShortReader {
fn new(data: Vec<u8>, interrupts: bool) -> Self {
ShortReader {
inner: Cursor::new(data),
chunk: 7,
interrupts,
interrupt_next: false,
}
}
}
impl Read for ShortReader {
fn read(&mut self, buf: &mut [u8]) -> io::Result<usize> {
if self.interrupt_next {
self.interrupt_next = false;
return Err(io::Error::from(io::ErrorKind::Interrupted));
}
self.interrupt_next = self.interrupts;
let n = buf.len().min(self.chunk);
self.inner.read(&mut buf[..n])
}
}
impl Seek for ShortReader {
fn seek(&mut self, pos: SeekFrom) -> io::Result<u64> {
self.inner.seek(pos)
}
}
fn walk_all<R: Read + Seek>(reader: &mut R, start_from: u64) -> Vec<(NeedleId, i64, Size)> {
let mut collected = Vec::new();
walk_index_file(reader, start_from, |key, offset, size| {
collected.push((key, offset.to_actual_offset(), size));
Ok(())
})
.unwrap();
collected
}
/// More than one ROWS_TO_READ batch, so the walk crosses a buffer refill.
fn many_entries() -> Vec<(NeedleId, Offset, Size)> {
(0..(ROWS_TO_READ as u64 * 2 + 37))
.map(|i| {
(
NeedleId(i * 7 + 1),
Offset::from_actual_offset(i as i64 * 128),
Size(i as i32 + 1),
)
})
.collect()
}
#[test]
fn test_walk_index_file_short_reads_keep_alignment() {
let data = idx_bytes(&many_entries());
let expected = walk_all(&mut Cursor::new(data.clone()), 0);
assert_eq!(expected.len(), ROWS_TO_READ * 2 + 37);
let mut short = ShortReader::new(data, false);
assert_eq!(walk_all(&mut short, 0), expected);
}
#[test]
fn test_walk_index_file_retries_interrupted_reads() {
let data = idx_bytes(&many_entries());
let expected = walk_all(&mut Cursor::new(data.clone()), 0);
let mut short = ShortReader::new(data, true);
assert_eq!(walk_all(&mut short, 0), expected);
}
#[test]
fn test_walk_index_file_short_reads_start_from() {
let data = idx_bytes(&many_entries());
let expected = walk_all(&mut Cursor::new(data.clone()), 0);
let start = ROWS_TO_READ as u64 + 5;
let mut short = ShortReader::new(data, false);
assert_eq!(walk_all(&mut short, start), expected[start as usize..]);
}
#[test]
fn test_walk_index_file_ignores_trailing_partial_entry() {
// A torn final entry is dropped without an error, as Go does on io.EOF.
let entries = many_entries();
let mut data = idx_bytes(&entries);
data.extend_from_slice(&[0xAB; NEEDLE_MAP_ENTRY_SIZE - 1]);
let expected = walk_all(&mut Cursor::new(idx_bytes(&entries)), 0);
assert_eq!(walk_all(&mut Cursor::new(data.clone()), 0), expected);
let mut short = ShortReader::new(data, false);
assert_eq!(walk_all(&mut short, 0), expected);
}
#[test]
fn test_check_index_file_clean() {
let data = idx_bytes(&[
@@ -337,11 +213,7 @@ mod tests {
let size = data.len() as i64;
let (count, errs) = check_index_file(&mut Cursor::new(data), size, Version(3));
assert_eq!(count, 2, "tombstone row is still counted: {:?}", errs);
assert!(
errs.is_empty(),
"offset-0 tombstone must not overlap: {:?}",
errs
);
assert!(errs.is_empty(), "offset-0 tombstone must not overlap: {:?}", errs);
}
#[test]
-208
View File
@@ -1,208 +0,0 @@
//! Positional file reads.
//!
//! Every read here is "these bytes at this offset", never "the next bytes".
//! The handles are shared — `.dat` and `.idx` descriptors are borrowed from
//! [`file_pool`](super::needle_map::file_pool), a mounted EC shard's handle is
//! duplicated into a scrub plan — so no caller may rely on a file position.
//!
//! On unix that is `pread(2)` through `std::os::unix::fs::FileExt`. On Windows
//! it is `seek_read`, which passes the offset through `OVERLAPPED`, so the read
//! itself is independent of the current cursor.
//!
//! What these helpers replace is `try_clone()` + `seek()` + `read()`. A
//! duplicated handle shares one kernel file offset with the original, so that
//! sequence is two syscalls against state another thread can move in between:
//! the seek positions the offset, a concurrent reader or an append moves it,
//! and the read returns bytes from somewhere else entirely. `seek_read` carries
//! its own offset in a single call, so there is no window.
//!
//! `seek_read` does still advance the cursor as a side effect — Windows updates
//! the file pointer even for an `OVERLAPPED` read — which nothing here relies
//! on. A caller that genuinely needs a private position must open the file
//! again rather than duplicate a handle; see `Volume::dat_scan_plan` in
//! [`storage::volume`](super::volume).
use std::fs::File;
use std::io;
/// Reads exactly `buf.len()` bytes from `file` starting at `offset`.
///
/// Fails with [`io::ErrorKind::UnexpectedEof`] if the file ends first.
pub(crate) fn read_exact_at(file: &File, buf: &mut [u8], offset: u64) -> io::Result<()> {
#[cfg(unix)]
{
use std::os::unix::fs::FileExt;
file.read_exact_at(buf, offset)?;
}
#[cfg(windows)]
{
if read_full_at(file, buf, offset)? < buf.len() {
return Err(io::Error::new(
io::ErrorKind::UnexpectedEof,
"unexpected EOF in seek_read",
));
}
}
#[cfg(not(any(unix, windows)))]
{
compile_error!("Platform not supported: only unix and windows are supported");
}
Ok(())
}
/// Reads up to `buf.len()` bytes from `file` starting at `offset`, returning
/// how many were read.
///
/// A short read — including `0` at or past end of file — is not an error; use
/// [`read_exact_at`] when the whole buffer must be filled.
pub(crate) fn read_at(file: &File, buf: &mut [u8], offset: u64) -> io::Result<usize> {
#[cfg(unix)]
{
use std::os::unix::fs::FileExt;
file.read_at(buf, offset)
}
#[cfg(windows)]
{
use std::os::windows::fs::FileExt;
file.seek_read(buf, offset)
}
#[cfg(not(any(unix, windows)))]
{
compile_error!("Platform not supported: only unix and windows are supported");
}
}
/// Reads into `buf` at `offset` until it is full or the file ends, retrying
/// interrupted reads; returns how many bytes were read.
///
/// Unlike [`read_at`], a count below `buf.len()` always means end of file.
pub(crate) fn read_full_at(file: &File, buf: &mut [u8], offset: u64) -> io::Result<usize> {
fill_at(|b, at| read_at(file, b, at), buf, offset)
}
fn fill_at(
mut read: impl FnMut(&mut [u8], u64) -> io::Result<usize>,
buf: &mut [u8],
offset: u64,
) -> io::Result<usize> {
let mut filled = 0;
while filled < buf.len() {
match read(&mut buf[filled..], offset + filled as u64) {
Ok(0) => break,
Ok(n) => filled += n,
Err(err) if err.kind() == io::ErrorKind::Interrupted => {}
Err(err) => return Err(err),
}
}
Ok(filled)
}
#[cfg(test)]
mod tests {
use super::{fill_at, read_at, read_exact_at, read_full_at};
use std::io::{ErrorKind, Write};
fn temp_file(bytes: &[u8]) -> tempfile::NamedTempFile {
let mut f = tempfile::NamedTempFile::new().expect("temp file");
f.write_all(bytes).expect("write");
f.flush().expect("flush");
f
}
#[test]
fn read_exact_at_fills_the_whole_buffer() {
let f = temp_file(b"0123456789");
let mut buf = [0u8; 10];
read_exact_at(f.as_file(), &mut buf, 0).expect("read");
assert_eq!(&buf, b"0123456789");
}
#[test]
fn read_exact_at_reads_from_the_offset() {
let f = temp_file(b"0123456789");
let mut buf = [0u8; 4];
read_exact_at(f.as_file(), &mut buf, 3).expect("read");
assert_eq!(&buf, b"3456");
// The helper is positional: a second read at a lower offset sees the
// bytes at that offset, not wherever the first read left a cursor.
let mut again = [0u8; 4];
read_exact_at(f.as_file(), &mut again, 1).expect("read");
assert_eq!(&again, b"1234");
}
#[test]
fn read_exact_at_short_file_is_unexpected_eof() {
let f = temp_file(b"0123");
let mut buf = [0u8; 8];
let err = read_exact_at(f.as_file(), &mut buf, 0).expect_err("short file");
assert_eq!(err.kind(), ErrorKind::UnexpectedEof);
}
#[test]
fn read_at_allows_a_short_read_at_eof() {
let f = temp_file(b"0123456789");
let mut buf = [0u8; 8];
let n = read_at(f.as_file(), &mut buf, 6).expect("read");
assert_eq!(n, 4);
assert_eq!(&buf[..n], b"6789");
// Entirely past the end is zero bytes, not an error.
let n = read_at(f.as_file(), &mut buf, 10).expect("read");
assert_eq!(n, 0);
}
/// A source that returns at most `chunk` bytes per call and fails with
/// `Interrupted` on its first call, like a network mount under a signal.
fn chunked(src: &[u8], chunk: usize) -> impl FnMut(&mut [u8], u64) -> std::io::Result<usize> {
let mut interrupted = false;
move |buf, at| {
if !interrupted {
interrupted = true;
return Err(ErrorKind::Interrupted.into());
}
let at = (at as usize).min(src.len());
let n = buf.len().min(chunk).min(src.len() - at);
buf[..n].copy_from_slice(&src[at..at + n]);
Ok(n)
}
}
#[test]
fn fill_at_fills_across_short_and_interrupted_reads() {
let src: Vec<u8> = (0..=255).collect();
let mut buf = [0u8; 100];
let n = fill_at(chunked(&src, 7), &mut buf, 50).expect("read");
assert_eq!(n, buf.len());
assert_eq!(&buf[..], &src[50..150]);
}
#[test]
fn fill_at_stops_at_end_of_source() {
let src: Vec<u8> = (0..=255).collect();
let mut buf = [0u8; 100];
let n = fill_at(chunked(&src, 7), &mut buf, 200).expect("read");
assert_eq!(n, 56);
assert_eq!(&buf[..n], &src[200..]);
}
#[test]
fn fill_at_propagates_other_errors() {
let mut buf = [0u8; 8];
let err = fill_at(|_, _| Err(ErrorKind::PermissionDenied.into()), &mut buf, 0)
.expect_err("error");
assert_eq!(err.kind(), ErrorKind::PermissionDenied);
}
#[test]
fn read_full_at_returns_the_short_count_only_at_eof() {
let f = temp_file(b"0123456789");
let mut buf = [0u8; 8];
assert_eq!(read_full_at(f.as_file(), &mut buf, 0).expect("read"), 8);
assert_eq!(&buf, b"01234567");
assert_eq!(read_full_at(f.as_file(), &mut buf, 6).expect("read"), 4);
assert_eq!(&buf[..4], b"6789");
assert_eq!(read_full_at(f.as_file(), &mut buf, 10).expect("read"), 0);
}
}
-354
View File
@@ -1,354 +0,0 @@
//! Consecutive storage-media error tracking shared by `Volume` and
//! `EcVolume`. Mirrors Go's `weed/storage/io_error.go`.
use std::io;
use std::sync::Mutex;
use std::sync::atomic::{AtomicBool, AtomicU64, Ordering};
/// Consecutive storage-media errors allowed before the volume is quarantined.
pub(crate) const IO_ERROR_TOLERANCE: i32 = 3;
/// Returns true for I/O errors that indicate faulty storage media, not
/// transient/network failures. On Unix this is EIO; on Windows it covers
/// ERROR_CRC and ERROR_IO_DEVICE, which the kernel returns for failing disks.
pub(crate) fn is_storage_io_error(e: &io::Error) -> bool {
#[cfg(unix)]
{
e.raw_os_error() == Some(libc::EIO)
}
#[cfg(windows)]
{
const ERROR_CRC: i32 = 23;
const ERROR_IO_DEVICE: i32 = 1117;
return e.raw_os_error() == Some(ERROR_CRC) || e.raw_os_error() == Some(ERROR_IO_DEVICE);
}
#[cfg(not(any(unix, windows)))]
{
false
}
}
/// Consecutive storage-media error state for one volume. `quarantined` is
/// sticky: once set it survives later successful I/O and is lifted only by
/// `reset_io_error_state`.
#[derive(Default)]
pub(crate) struct IoErrorTracker {
last: Mutex<Option<String>>,
/// The consecutive error count in the low 32 bits and, in the high 32,
/// how many times it has been cleared. They share one word so that
/// `record_success_at` updates both in one step: reads record their
/// outcomes here without the volume's write lock.
streak: AtomicU64,
quarantined: AtomicBool,
}
const STREAK_COUNT_BITS: u64 = 0xffff_ffff;
fn streak_count(streak: u64) -> i32 {
(streak & STREAK_COUNT_BITS) as i32
}
/// `streak` with its count cleared and one more clear on record.
fn streak_cleared(streak: u64) -> u64 {
(streak >> 32).wrapping_add(1) << 32
}
/// A point in the error streak, taken where a write landed whose success
/// is only recorded later. See `IoErrorTracker::record_success_at`.
#[derive(Clone, Copy)]
pub(crate) struct StreakMark(u64);
impl IoErrorTracker {
/// `Some(e)` records a failure, `None` a success. Only storage-media
/// failures count; every other outcome clears the count and last error.
pub(crate) fn check_read_write_error(&self, err: Option<&io::Error>) {
if let Some(e) = err
&& is_storage_io_error(e)
{
self.streak.fetch_add(1, Ordering::Relaxed);
if let Ok(mut guard) = self.last.lock() {
*guard = Some(e.to_string());
}
crate::metrics::STORAGE_IO_ERROR_COUNTER.inc();
return;
}
self.clear_count();
self.clear_last();
}
fn clear_count(&self) {
self.update_streak(|streak| Some(streak_cleared(streak)));
}
fn clear_last(&self) {
if let Ok(mut guard) = self.last.lock()
&& guard.is_some()
{
*guard = None;
}
}
/// Apply `f` to the streak atomically; `None` leaves it as it is.
/// Returns the streak `f` produced, if any.
fn update_streak(&self, mut f: impl FnMut(u64) -> Option<u64>) -> Option<u64> {
let mut updated = None;
let _ = self
.streak
.fetch_update(Ordering::Relaxed, Ordering::Relaxed, |streak| {
updated = f(streak);
updated
});
updated
}
pub(crate) fn mark(&self) -> StreakMark {
StreakMark(self.streak.load(Ordering::Relaxed))
}
/// Record a success as if it had come at `mark`: the errors counted
/// before the mark are cleared and the ones counted since still stand,
/// as they would had each outcome been recorded in order. A streak
/// cleared since the mark is left as it is.
pub(crate) fn record_success_at(&self, mark: StreakMark) {
let before = streak_count(mark.0);
let updated = self.update_streak(|streak| {
if streak >> 32 != mark.0 >> 32 {
return None;
}
if streak_count(streak) <= before {
return Some(streak_cleared(streak));
}
Some(streak - before as u64)
});
// The last error stays when one counted since the mark is left.
if updated.is_some_and(|streak| streak_count(streak) == 0) {
self.clear_last();
}
}
/// The last recorded error, the consecutive count, and the quarantine flag.
pub(crate) fn get_io_error_state(&self) -> (Option<String>, i32, bool) {
let err = self.last.lock().ok().and_then(|g| g.clone());
let count = self.count();
let quarantined = self.quarantined.load(Ordering::Relaxed);
(err, count, quarantined)
}
fn count(&self) -> i32 {
streak_count(self.streak.load(Ordering::Relaxed))
}
pub(crate) fn should_quarantine(&self) -> bool {
self.quarantined.load(Ordering::Relaxed) || self.count() >= IO_ERROR_TOLERANCE
}
pub(crate) fn mark_io_quarantined(&self) {
self.quarantined.store(true, Ordering::Relaxed);
}
pub(crate) fn reset_io_error_state(&self) {
self.clear_count();
self.quarantined.store(false, Ordering::Relaxed);
if let Ok(mut guard) = self.last.lock() {
*guard = None;
}
}
#[cfg(test)]
pub(crate) fn set_last_io_error_for_test(&self, err: Option<&str>) {
if let Ok(mut guard) = self.last.lock() {
*guard = err.map(|value| value.to_string());
}
if err.is_some() {
self.update_streak(|streak| {
Some((streak & !STREAK_COUNT_BITS) | IO_ERROR_TOLERANCE as u64)
});
} else {
self.clear_count();
}
}
}
// The tracker only reacts to errors `is_storage_io_error` recognises, which is
// nothing at all on a platform that is neither Unix nor Windows.
#[cfg(all(test, any(unix, windows)))]
mod tests {
use super::*;
/// An OS error the platform reports for failing storage media.
#[cfg(unix)]
fn media_error() -> io::Error {
io::Error::from_raw_os_error(libc::EIO)
}
/// An OS error the platform reports for failing storage media.
#[cfg(windows)]
fn media_error() -> io::Error {
const ERROR_IO_DEVICE: i32 = 1117;
io::Error::from_raw_os_error(ERROR_IO_DEVICE)
}
#[test]
fn check_read_write_error_counts_consecutive_media_errors() {
let tracker = IoErrorTracker::default();
tracker.check_read_write_error(Some(&media_error()));
tracker.check_read_write_error(Some(&media_error()));
let (last, count, quarantined) = tracker.get_io_error_state();
assert_eq!(last, Some(media_error().to_string()));
assert_eq!(count, 2);
assert!(!quarantined);
}
#[test]
fn success_clears_the_count_and_the_last_error() {
let tracker = IoErrorTracker::default();
tracker.check_read_write_error(Some(&media_error()));
tracker.check_read_write_error(None);
assert_eq!(tracker.get_io_error_state(), (None, 0, false));
}
#[test]
fn non_media_error_clears_the_count() {
let tracker = IoErrorTracker::default();
tracker.check_read_write_error(Some(&media_error()));
tracker.check_read_write_error(Some(&io::Error::new(
io::ErrorKind::NotFound,
"no such file",
)));
assert_eq!(tracker.get_io_error_state(), (None, 0, false));
}
#[test]
fn should_quarantine_only_once_the_tolerance_is_reached() {
let tracker = IoErrorTracker::default();
for _ in 1..IO_ERROR_TOLERANCE {
tracker.check_read_write_error(Some(&media_error()));
assert!(!tracker.should_quarantine());
}
tracker.check_read_write_error(Some(&media_error()));
assert!(tracker.should_quarantine());
}
#[test]
fn quarantine_survives_later_successful_io() {
let tracker = IoErrorTracker::default();
tracker.mark_io_quarantined();
tracker.check_read_write_error(None);
assert_eq!(tracker.get_io_error_state(), (None, 0, true));
assert!(tracker.should_quarantine());
}
#[test]
fn success_at_a_mark_keeps_only_the_errors_after_it() {
let tracker = IoErrorTracker::default();
tracker.check_read_write_error(Some(&media_error()));
tracker.check_read_write_error(Some(&media_error()));
let mark = tracker.mark();
tracker.check_read_write_error(Some(&media_error()));
tracker.record_success_at(mark);
assert_eq!(
tracker.get_io_error_state(),
(Some(media_error().to_string()), 1, false)
);
}
#[test]
fn success_at_a_mark_with_nothing_after_it_clears_the_streak() {
let tracker = IoErrorTracker::default();
tracker.check_read_write_error(Some(&media_error()));
let mark = tracker.mark();
tracker.record_success_at(mark);
assert_eq!(tracker.get_io_error_state(), (None, 0, false));
}
#[test]
fn success_at_a_mark_leaves_a_streak_cleared_since() {
let tracker = IoErrorTracker::default();
tracker.check_read_write_error(Some(&media_error()));
tracker.check_read_write_error(Some(&media_error()));
let mark = tracker.mark();
tracker.check_read_write_error(None);
tracker.check_read_write_error(Some(&media_error()));
tracker.record_success_at(mark);
assert_eq!(
tracker.get_io_error_state(),
(Some(media_error().to_string()), 1, false)
);
}
/// Reads update the tracker without the volume's write lock, so a
/// success replayed at a mark must not lose the errors they record
/// while it runs.
#[test]
fn success_at_a_mark_keeps_concurrent_errors() {
use std::sync::{Arc, Barrier};
const READERS: i32 = 4;
const ERRORS: i32 = 200;
for _ in 0..500 {
let tracker = Arc::new(IoErrorTracker::default());
tracker.check_read_write_error(Some(&media_error()));
tracker.check_read_write_error(Some(&media_error()));
let mark = tracker.mark();
let start = Arc::new(Barrier::new(READERS as usize + 1));
let readers: Vec<_> = (0..READERS)
.map(|_| {
let (tracker, start) = (tracker.clone(), start.clone());
std::thread::spawn(move || {
start.wait();
for _ in 0..ERRORS {
tracker.check_read_write_error(Some(&media_error()));
}
})
})
.collect();
start.wait();
while tracker.get_io_error_state().1 < 2 + READERS * ERRORS / 2 {
std::hint::spin_loop();
}
tracker.record_success_at(mark);
for reader in readers {
reader.join().unwrap();
}
// The two errors before the mark are cleared; every error the
// readers recorded after it stands.
assert_eq!(tracker.get_io_error_state().1, READERS * ERRORS);
}
}
#[test]
fn reset_io_error_state_lifts_the_quarantine() {
let tracker = IoErrorTracker::default();
tracker.check_read_write_error(Some(&media_error()));
tracker.mark_io_quarantined();
tracker.reset_io_error_state();
assert_eq!(tracker.get_io_error_state(), (None, 0, false));
assert!(!tracker.should_quarantine());
}
#[test]
fn test_helper_arms_a_sustained_error() {
let tracker = IoErrorTracker::default();
tracker.set_last_io_error_for_test(Some("input/output error"));
assert!(tracker.should_quarantine());
assert_eq!(
tracker.get_io_error_state(),
(
Some("input/output error".to_string()),
IO_ERROR_TOLERANCE,
false
)
);
tracker.set_last_io_error_for_test(None);
assert_eq!(tracker.get_io_error_state(), (None, 0, false));
}
}
-3
View File
@@ -1,12 +1,9 @@
pub mod disk_location;
pub mod erasure_coding;
pub mod idx;
pub(crate) mod io;
pub(crate) mod io_error;
pub mod needle;
pub mod needle_map;
pub mod store;
pub mod store_ec_journal;
pub mod store_ec_mirror;
pub mod store_ec_reconcile;
pub mod super_block;
+1 -4
View File
@@ -1,8 +1,5 @@
pub mod crc;
#[expect(
clippy::module_inception,
reason = "needle/needle.rs mirrors the Go package layout"
)]
#[expect(clippy::module_inception, reason = "needle/needle.rs mirrors the Go package layout")]
pub mod needle;
pub mod ttl;
+18 -99
View File
@@ -198,8 +198,8 @@ impl Needle {
/// the data payload from disk at all, matching Go's `ReadNeedleMeta`.
pub fn read_paged_meta(
&mut self,
header_bytes: &[u8], // first 20 bytes: NEEDLE_HEADER_SIZE + DATA_SIZE_SIZE
meta_bytes: &[u8], // tail: non-data body metadata + checksum + timestamp + padding
header_bytes: &[u8], // first 20 bytes: NEEDLE_HEADER_SIZE + DATA_SIZE_SIZE
meta_bytes: &[u8], // tail: non-data body metadata + checksum + timestamp + padding
offset: i64,
expected_size: Size,
version: Version,
@@ -581,19 +581,23 @@ impl Needle {
// ============================================================================
/// Compute padding to align needle to NEEDLE_PADDING_SIZE (8 bytes).
///
/// The sum is formed in i64: a size read from a corrupt header can sit near
/// `i32::MAX`, and adding the header, checksum and timestamp widths to it in
/// i32 would overflow (a panic with overflow checks, a wrapped padding
/// without). The result is at most NEEDLE_PADDING_SIZE, so it fits `Size`.
pub fn padding_length(needle_size: Size, version: Version) -> Size {
let fixed = if version == VERSION_3 {
NEEDLE_HEADER_SIZE + NEEDLE_CHECKSUM_SIZE + TIMESTAMP_SIZE
if version == VERSION_3 {
Size(
NEEDLE_PADDING_SIZE as i32
- ((NEEDLE_HEADER_SIZE as i32
+ needle_size.0
+ NEEDLE_CHECKSUM_SIZE as i32
+ TIMESTAMP_SIZE as i32)
% NEEDLE_PADDING_SIZE as i32),
)
} else {
NEEDLE_HEADER_SIZE + NEEDLE_CHECKSUM_SIZE
};
let unpadded = fixed as i64 + needle_size.0 as i64;
Size((NEEDLE_PADDING_SIZE as i64 - unpadded % NEEDLE_PADDING_SIZE as i64) as i32)
Size(
NEEDLE_PADDING_SIZE as i32
- ((NEEDLE_HEADER_SIZE as i32 + needle_size.0 + NEEDLE_CHECKSUM_SIZE as i32)
% NEEDLE_PADDING_SIZE as i32),
)
}
}
/// Body length = Size + Checksum + [Timestamp] + Padding.
@@ -615,30 +619,6 @@ pub fn get_actual_size(size: Size, version: Version) -> i64 {
NEEDLE_HEADER_SIZE as i64 + needle_body_length(size, version)
}
/// Validate a wire-supplied needle body size before any `as usize` cast.
/// Rejects negative/deleted sizes and bodies larger than the gRPC max message.
/// Size(0) is allowed: empty/anomalous entries and tombstones read as size 0
/// (actual_size = header+checksum+pad > 0, safe alloc, no wrap).
/// Transport cap only: storage paths must NOT use this cap — see volume.rs
/// guards (a >1GiB stored needle from a high-limit cluster must remain
/// readable/compaction-safe). Keep `get_actual_size` unchanged (it
/// intentionally returns negative for deleted index entries).
pub fn validate_wire_size(size: Size) -> Result<(), String> {
if size.0 < 0 {
return Err(format!("invalid needle size {}", size.0));
}
// Keep in sync with canonical `GRPC_MAX_MESSAGE_SIZE` in server/grpc_client.rs:10
// (duplicated here to avoid a storage->server import and prevent drift).
const WIRE_MAX_NEEDLE_SIZE: i32 = 1 << 30;
if size.0 > WIRE_MAX_NEEDLE_SIZE {
return Err(format!(
"needle size {} exceeds max {}",
size.0, WIRE_MAX_NEEDLE_SIZE
));
}
Ok(())
}
/// Read 5 bytes as a u64 (big-endian, zero-padded high bytes).
fn bytes_to_u64_5(bytes: &[u8]) -> u64 {
assert!(bytes.len() >= 5);
@@ -749,14 +729,6 @@ pub fn parse_needle_id_cookie(s: &str) -> Result<(NeedleId, Cookie), String> {
(s, None)
};
// Every length check and the split below are in BYTES, so a multi-byte
// character would let `split` land inside one and panic the slice. Hex is
// ASCII by definition; reject anything else up front, as Go's ParseUint
// does a step later.
if !hex_part.is_ascii() {
return Err("KeyHash must be ASCII hex.".to_string());
}
// Go: len(key_hash_string) <= CookieSize*2 => error (must be > 8 hex chars)
if hex_part.len() <= COOKIE_SIZE * 2 {
return Err("KeyHash is too short.".to_string());
@@ -798,9 +770,7 @@ pub fn parse_needle_id_cookie(s: &str) -> Result<(NeedleId, Cookie), String> {
#[derive(Debug, thiserror::Error)]
pub enum NeedleError {
#[error(
"size mismatch at offset {offset}: found id={id} size={found:?}, expected size={expected:?}"
)]
#[error("size mismatch at offset {offset}: found id={id} size={found:?}, expected size={expected:?}")]
SizeMismatch {
offset: i64,
id: NeedleId,
@@ -836,30 +806,6 @@ pub enum NeedleError {
mod tests {
use super::*;
/// A fid whose hex part carries multi-byte UTF-8 must be rejected, not
/// panic. `split` is a byte offset into `hex_part`; before the ASCII guard
/// `&hex_part[..split]` could land inside a character. `GET /3,ééééa` is
/// nine bytes, so it passes the length checks and splits at byte 1 —
/// halfway through the first `é`. Go's `ParseUint` just errors.
#[test]
fn parse_needle_id_cookie_rejects_non_ascii_instead_of_panicking() {
for s in ["ééééa", "ééééaaaaa", "0123456é9abc", "ééééa_1"] {
assert!(
parse_needle_id_cookie(s).is_err(),
"non-ASCII fid {:?} must be an error",
s
);
}
}
/// The ASCII guard must not change any accepted input.
#[test]
fn parse_needle_id_cookie_still_accepts_ascii_hex() {
let (id, cookie) = parse_needle_id_cookie("01637037d6").unwrap();
assert_eq!(id, NeedleId(0x01));
assert_eq!(cookie, Cookie(0x637037d6));
}
#[test]
fn test_parse_header() {
let mut buf = [0u8; NEEDLE_HEADER_SIZE];
@@ -977,21 +923,6 @@ mod tests {
}
}
#[test]
fn padding_length_does_not_overflow_on_a_corrupt_size() {
// A header read from a corrupt or truncated file can carry any i32
// size. The scanners bound it against the bytes left before sizing a
// buffer, but on a volume with more than 2 GiB left a size near
// i32::MAX passes that bound, so the padding arithmetic itself must
// not overflow. Overflow checks are on in test builds, so an i32 sum
// here would panic rather than wrap.
for version in [VERSION_2, VERSION_3] {
let padding = padding_length(Size(i32::MAX), version).0 as i64;
assert!((1..=NEEDLE_PADDING_SIZE as i64).contains(&padding));
assert_eq!(get_actual_size(Size(i32::MAX), version) % 8, 0);
}
}
#[test]
fn test_file_id_parse() {
let fid = FileId::parse("3,01637037d6").unwrap();
@@ -1036,16 +967,4 @@ mod tests {
assert_eq!(fid.key, NeedleId(0x123));
assert_eq!(fid.cookie, Cookie(0));
}
#[test]
fn test_validate_wire_size_boundaries() {
assert!(validate_wire_size(Size(-100)).is_err());
assert!(validate_wire_size(Size(-1)).is_err());
assert!(validate_wire_size(Size(0)).is_ok());
assert!(validate_wire_size(Size(1024)).is_ok());
assert!(validate_wire_size(Size(1)).is_ok());
assert!(validate_wire_size(Size(1 << 30)).is_ok());
assert!(validate_wire_size(Size((1 << 30) + 1)).is_err());
assert!(validate_wire_size(Size(i32::MAX)).is_err());
}
}
+8 -72
View File
@@ -80,12 +80,6 @@ impl TTL {
if s.is_empty() {
return Ok(TTL::EMPTY);
}
// The unit is read as the last BYTE and the count as everything before
// it, so a trailing multi-byte character would split inside itself and
// panic. A TTL is digits plus a one-letter unit; reject the rest.
if !s.is_ascii() {
return Err(format!("invalid TTL {:?}: must be ASCII", s));
}
let last_byte = s.as_bytes()[s.len() - 1];
let (num_str, unit_byte) = if last_byte.is_ascii_digit() {
// All digits — default to minutes (matching Go)
@@ -246,16 +240,6 @@ impl fmt::Display for TTL {
mod tests {
use super::*;
/// `?ttl=5%C3%A9` must be an error, not a panic. The unit is taken as the
/// last *byte*, so a trailing multi-byte character made `&s[..s.len()-1]`
/// split inside it.
#[test]
fn ttl_read_rejects_non_ascii_instead_of_panicking() {
for s in ["5é", "é", "3🦀", "12é"] {
assert!(TTL::read(s).is_err(), "non-ASCII TTL {:?} must error", s);
}
}
#[test]
fn test_ttl_parse() {
let ttl = TTL::read("3m").unwrap();
@@ -274,13 +258,7 @@ mod tests {
// 24h normalizes to 1d via fitTtlCount
let ttl = TTL::read("24h").unwrap();
assert_eq!(ttl.to_seconds(), 86400);
assert_eq!(
ttl,
TTL {
count: 1,
unit: TTL_UNIT_DAY
}
);
assert_eq!(ttl, TTL { count: 1, unit: TTL_UNIT_DAY });
}
#[test]
@@ -326,24 +304,12 @@ mod tests {
fn test_ttl_overflow_normalizes() {
// Go's ReadTTL calls fitTtlCount: 300m = 18000s = 5h (exact fit)
let ttl = TTL::read("300m").unwrap();
assert_eq!(
ttl,
TTL {
count: 5,
unit: TTL_UNIT_HOUR
}
);
assert_eq!(ttl, TTL { count: 5, unit: TTL_UNIT_HOUR });
// 256h = 921600s. Doesn't fit in hours (256 >= 256), doesn't fit exact in days.
// Second pass: 921600/86400 = 10 (truncated) < 256 -> 10d
let ttl = TTL::read("256h").unwrap();
assert_eq!(
ttl,
TTL {
count: 10,
unit: TTL_UNIT_DAY
}
);
assert_eq!(ttl, TTL { count: 10, unit: TTL_UNIT_DAY });
}
#[test]
@@ -351,49 +317,19 @@ mod tests {
// Go's ReadTTL calls fitTtlCount which normalizes to coarsest unit.
// 120m -> 2h, 7d -> 1w, 24h -> 1d.
let ttl = TTL::read("120m").unwrap();
assert_eq!(
ttl,
TTL {
count: 2,
unit: TTL_UNIT_HOUR
}
);
assert_eq!(ttl, TTL { count: 2, unit: TTL_UNIT_HOUR });
let ttl = TTL::read("7d").unwrap();
assert_eq!(
ttl,
TTL {
count: 1,
unit: TTL_UNIT_WEEK
}
);
assert_eq!(ttl, TTL { count: 1, unit: TTL_UNIT_WEEK });
let ttl = TTL::read("24h").unwrap();
assert_eq!(
ttl,
TTL {
count: 1,
unit: TTL_UNIT_DAY
}
);
assert_eq!(ttl, TTL { count: 1, unit: TTL_UNIT_DAY });
// Values that don't simplify stay as-is
let ttl = TTL::read("5d").unwrap();
assert_eq!(
ttl,
TTL {
count: 5,
unit: TTL_UNIT_DAY
}
);
assert_eq!(ttl, TTL { count: 5, unit: TTL_UNIT_DAY });
let ttl = TTL::read("3m").unwrap();
assert_eq!(
ttl,
TTL {
count: 3,
unit: TTL_UNIT_MINUTE
}
);
assert_eq!(ttl, TTL { count: 3, unit: TTL_UNIT_MINUTE });
}
}
+73 -215
View File
@@ -195,12 +195,7 @@ impl NeedleMapKind {
// ============================================================================
/// Trait for appending to an index file.
///
/// The file is opened without append mode and each row is written at the
/// current `idx_file_offset` — the same positioned-write model the Go
/// server uses — because an append-mode handle cannot truncate on Windows,
/// where the std library keeps it strictly append-only.
pub trait IdxFileWriter: Write + Seek + Send + Sync {
pub trait IdxFileWriter: Write + Send + Sync {
fn sync_all(&self) -> io::Result<()>;
/// Truncate the file to `len` bytes. Used to remove an orphan .idx row
/// left by a failed redb commit so `idx_file_offset` stays a contiguous
@@ -229,9 +224,6 @@ pub struct CompactNeedleMap {
metric: NeedleMapMetric,
idx_file: Option<Box<dyn IdxFileWriter>>,
idx_file_offset: u64,
/// The file holds bytes past `idx_file_offset` that must be trimmed
/// before another row can land aligned.
idx_torn: bool,
}
impl Default for CompactNeedleMap {
@@ -248,7 +240,6 @@ impl CompactNeedleMap {
metric: NeedleMapMetric::default(),
idx_file: None,
idx_file_offset: 0,
idx_torn: false,
}
}
@@ -256,8 +247,6 @@ impl CompactNeedleMap {
pub fn load_from_idx<R: Read + Seek>(reader: &mut R, version: Version) -> io::Result<Self> {
let mut nm = CompactNeedleMap::new();
idx::walk_index_file(reader, 0, |key, offset, size| {
// A read-only load attaches no writer, so this is its only size.
nm.idx_file_offset += NEEDLE_MAP_ENTRY_SIZE as u64;
nm.metric.maybe_set_max_needle_end(offset, size, version);
if offset.is_zero() || size.is_deleted() {
nm.delete_from_map(key);
@@ -273,7 +262,6 @@ impl CompactNeedleMap {
pub fn set_idx_file(&mut self, file: Box<dyn IdxFileWriter>, offset: u64) {
self.idx_file = Some(file);
self.idx_file_offset = offset;
self.idx_torn = false;
}
/// True when an .idx file writer is attached. A read-only load leaves
@@ -288,8 +276,8 @@ impl CompactNeedleMap {
/// Insert or update an entry. Appends to .idx file if present.
pub fn put(&mut self, key: NeedleId, offset: Offset, size: Size) -> io::Result<()> {
// Persist to idx file BEFORE mutating in-memory state for crash consistency
self.append_to_index_file(key, offset, size)?;
if self.idx_file.is_some() {
if let Some(ref mut idx_file) = self.idx_file {
idx::write_index_entry(idx_file, key, offset, size)?;
self.idx_file_offset += NEEDLE_MAP_ENTRY_SIZE as u64;
}
@@ -299,41 +287,6 @@ impl CompactNeedleMap {
Ok(())
}
/// Write one row to the .idx file at `idx_file_offset`. A row left
/// half-written by a failed write is trimmed back to the offset so the
/// next row still lands aligned; while the trim keeps failing no row is
/// written at all, or it would sit off alignment and parse as garbage on
/// load. The offset itself is advanced by the caller once the row counts.
fn append_to_index_file(
&mut self,
key: NeedleId,
offset: Offset,
size: Size,
) -> io::Result<()> {
let Some(idx_file) = self.idx_file.as_mut() else {
return Ok(());
};
if self.idx_torn {
match idx_file.truncate_to(self.idx_file_offset) {
Ok(()) => self.idx_torn = false,
Err(e) => {
return Err(io::Error::other(format!(
"index file still holds a torn row: {e}"
)));
}
}
}
idx_file.seek(io::SeekFrom::Start(self.idx_file_offset))?;
if let Err(e) = idx::write_index_entry(idx_file, key, offset, size) {
if let Err(te) = idx_file.truncate_to(self.idx_file_offset) {
self.idx_torn = true;
tracing::warn!("failed to trim torn .idx row: {}", te);
}
return Err(e);
}
Ok(())
}
/// Look up a needle.
pub fn get(&self, key: NeedleId) -> Option<NeedleValue> {
self.map.get(key)
@@ -358,8 +311,8 @@ impl CompactNeedleMap {
}
// Always write tombstone to idx file (matching Go)
self.append_to_index_file(key, offset, TOMBSTONE_FILE_SIZE)?;
if self.idx_file.is_some() {
if let Some(ref mut idx_file) = self.idx_file {
idx::write_index_entry(idx_file, key, offset, TOMBSTONE_FILE_SIZE)?;
self.idx_file_offset += NEEDLE_MAP_ENTRY_SIZE as u64;
}
@@ -464,9 +417,9 @@ impl CompactNeedleMap {
}
/// Visit all entries in ascending order by needle ID.
pub fn ascending_visit<F, E>(&self, f: F) -> Result<(), E>
pub fn ascending_visit<F>(&self, f: F) -> Result<(), String>
where
F: FnMut(NeedleId, &NeedleValue) -> Result<(), E>,
F: FnMut(NeedleId, &NeedleValue) -> Result<(), String>,
{
self.map.ascending_visit(f)
}
@@ -509,9 +462,6 @@ pub struct RedbNeedleMap {
metric: NeedleMapMetric,
idx_file: Option<Box<dyn IdxFileWriter>>,
idx_file_offset: u64,
/// The file holds bytes past `idx_file_offset` that must be trimmed
/// before another row can land aligned.
idx_torn: bool,
/// Puts/deletes since the last durable checkpoint.
writes_since_checkpoint: u32,
}
@@ -618,7 +568,6 @@ impl RedbNeedleMap {
metric: NeedleMapMetric::default(),
idx_file: None,
idx_file_offset: 0,
idx_torn: false,
writes_since_checkpoint: 0,
})
}
@@ -658,14 +607,6 @@ impl RedbNeedleMap {
}
}
/// Test-only read of META `idx_size` through the live handle. See
/// [`test_support::live_meta_idx_size`] for why durability tests use
/// this instead of copying the open `.rdb`.
#[cfg(test)]
pub(crate) fn live_meta_idx_size(&self) -> Option<u64> {
self.read_idx_size_meta().unwrap()
}
/// Load from an .idx file, reusing an existing .rdb if it is consistent.
///
/// Strategy:
@@ -716,7 +657,6 @@ impl RedbNeedleMap {
metric: NeedleMapMetric::default(),
idx_file: None,
idx_file_offset: 0,
idx_torn: false,
writes_since_checkpoint: 0,
};
@@ -861,26 +801,23 @@ impl RedbNeedleMap {
}
#[cfg(feature = "redb-experimental-cursor")]
{
let mut cursor =
table
.upper_bound_mut(Bound::<u64>::Unbounded)
.map_err(|e| {
io::Error::new(
io::ErrorKind::Other,
format!("redb upper_bound_mut: {}", e),
)
})?;
let mut cursor = table
.upper_bound_mut(Bound::<u64>::Unbounded)
.map_err(|e| {
io::Error::new(
io::ErrorKind::Other,
format!("redb upper_bound_mut: {}", e),
)
})?;
for (key, nv) in &entries {
let key_u64: u64 = (*key).into();
let packed = pack_needle_value(nv);
cursor
.insert_before(key_u64, packed.as_slice())
.map_err(|e| {
io::Error::new(
io::ErrorKind::Other,
format!("redb insert_before: {}", e),
)
})?;
cursor.insert_before(key_u64, packed.as_slice()).map_err(|e| {
io::Error::new(
io::ErrorKind::Other,
format!("redb insert_before: {}", e),
)
})?;
}
cursor.close().map_err(|e| {
io::Error::new(io::ErrorKind::Other, format!("redb cursor close: {}", e))
@@ -908,7 +845,6 @@ impl RedbNeedleMap {
pub fn set_idx_file(&mut self, file: Box<dyn IdxFileWriter>, offset: u64) {
self.idx_file = Some(file);
self.idx_file_offset = offset;
self.idx_torn = false;
}
/// True when an .idx file writer is attached. See CompactNeedleMap.
@@ -925,7 +861,9 @@ impl RedbNeedleMap {
// commit leaves an orphan row in .idx that redb doesn't reflect, and
// advancing the offset here would let a later checkpoint record it as
// reflected, making the reload skip it permanently.
self.append_to_index_file(key, offset, size)?;
if let Some(ref mut idx_file) = self.idx_file {
idx::write_index_entry(idx_file, key, offset, size)?;
}
let key_u64: u64 = key.into();
let packed = pack_needle_value(&NeedleValue { offset, size });
@@ -996,41 +934,6 @@ impl RedbNeedleMap {
Ok(())
}
/// Write one row to the .idx file at `idx_file_offset`. A row left
/// half-written by a failed write is trimmed back to the offset so the
/// next row still lands aligned; while the trim keeps failing no row is
/// written at all, or it would sit off alignment and parse as garbage on
/// load. The offset itself is advanced by the caller once the row counts.
fn append_to_index_file(
&mut self,
key: NeedleId,
offset: Offset,
size: Size,
) -> io::Result<()> {
let Some(idx_file) = self.idx_file.as_mut() else {
return Ok(());
};
if self.idx_torn {
match idx_file.truncate_to(self.idx_file_offset) {
Ok(()) => self.idx_torn = false,
Err(e) => {
return Err(io::Error::other(format!(
"index file still holds a torn row: {e}"
)));
}
}
}
idx_file.seek(io::SeekFrom::Start(self.idx_file_offset))?;
if let Err(e) = idx::write_index_entry(idx_file, key, offset, size) {
if let Err(te) = idx_file.truncate_to(self.idx_file_offset) {
self.idx_torn = true;
tracing::warn!("failed to trim torn .idx row: {}", te);
}
return Err(e);
}
Ok(())
}
/// Look up a needle. A redb failure is an ERROR, not an absent needle:
/// answering "not found" would turn a database problem into a read miss
/// and let a delete report success without recording a tombstone.
@@ -1077,7 +980,9 @@ impl RedbNeedleMap {
return Ok(None);
};
self.append_to_index_file(key, offset, TOMBSTONE_FILE_SIZE)?;
if let Some(ref mut idx_file) = self.idx_file {
idx::write_index_entry(idx_file, key, offset, TOMBSTONE_FILE_SIZE)?;
}
let deleted_nv = NeedleValue {
offset: old.offset,
@@ -1167,13 +1072,10 @@ impl RedbNeedleMap {
/// a failed redb commit. Without this the next successful write appends
/// after the orphan, `idx_file_offset` advances past it, and a later
/// checkpoint records an offset that makes the reload skip the orphan.
/// When the trim fails the file is latched torn so no later row lands
/// after bytes the map does not reflect.
fn truncate_idx_to_offset(&mut self) {
if let Some(ref mut idx_file) = self.idx_file
&& let Err(e) = idx_file.truncate_to(self.idx_file_offset)
{
self.idx_torn = true;
tracing::warn!("failed to truncate orphan .idx row: {}", e);
}
}
@@ -1212,19 +1114,24 @@ impl RedbNeedleMap {
let read_file = std::fs::OpenOptions::new()
.read(true)
.open(&idx_path)
.map_err(|e| io::Error::other(format!("reopen: open .idx {}: {}", idx_path, e)))?;
.map_err(|e| {
io::Error::other(format!("reopen: open .idx {}: {}", idx_path, e))
})?;
let actual_idx_size = read_file.metadata()?.len();
let mut reader = io::BufReader::new(read_file);
let reopened =
Self::load_from_idx(&self.rdb_path, &mut reader, self.version, self.cache_bytes)?;
let reopened = Self::load_from_idx(
&self.rdb_path,
&mut reader,
self.version,
self.cache_bytes,
)?;
// Preserve the append writer and the paths/version/cache; adopt the
// repaired database, metrics, and idx_file_offset from the reload.
self.db = reopened.db;
self.metric = reopened.metric;
self.idx_file_offset = actual_idx_size;
self.idx_torn = reopened.idx_torn;
// The reopen replayed all rows since the last durable checkpoint
// non-durably; start the counter fresh.
self.writes_since_checkpoint = 0;
@@ -1291,10 +1198,9 @@ impl RedbNeedleMap {
}
/// Visit all entries in ascending order by needle ID.
pub fn ascending_visit<F, E>(&self, mut f: F) -> Result<(), E>
pub fn ascending_visit<F>(&self, mut f: F) -> Result<(), String>
where
F: FnMut(NeedleId, &NeedleValue) -> Result<(), E>,
E: From<String>,
F: FnMut(NeedleId, &NeedleValue) -> Result<(), String>,
{
let txn = self
.db_or_err()
@@ -1474,18 +1380,6 @@ impl NeedleMap {
}
}
/// Skew the live file count away from what the `.idx` holds, so tests
/// can build a volume whose reported count disagrees with a reload.
#[cfg(test)]
pub(crate) fn add_file_count_for_test(&self, delta: i64) {
let metric = match self {
NeedleMap::InMemory(nm) => &nm.metric,
NeedleMap::Redb(nm) => &nm.metric,
NeedleMap::SortedFile(_) => panic!("sorted-file needle maps are read-only"),
};
metric.file_count.fetch_add(delta, Ordering::Relaxed);
}
/// Largest (offset + actual size) seen during the load walk; 0 if the
/// map is empty. Used at volume load to detect .idx entries that
/// reference past the end of .dat (issue #8928) without a second scan.
@@ -1544,10 +1438,9 @@ impl NeedleMap {
}
/// Visit all entries in ascending order by needle ID.
pub fn ascending_visit<F, E>(&self, f: F) -> Result<(), E>
pub fn ascending_visit<F>(&self, f: F) -> Result<(), String>
where
F: FnMut(NeedleId, &NeedleValue) -> Result<(), E>,
E: From<String>,
F: FnMut(NeedleId, &NeedleValue) -> Result<(), String>,
{
match self {
NeedleMap::InMemory(nm) => nm.ascending_visit(f),
@@ -1568,7 +1461,7 @@ impl NeedleMap {
// The visitor never fails, so neither can this.
let _ = nm.ascending_visit(|id, nv| {
entries.push((id, *nv));
Ok::<(), std::convert::Infallible>(())
Ok(())
});
Ok(entries)
}
@@ -1586,30 +1479,22 @@ impl NeedleMap {
pub(crate) mod test_support {
use super::*;
/// The `.idx` size in the map's META table, read through the live
/// handle.
///
/// The load path records the `.idx` size with `Durability::None`, and
/// every `put`/`delete` also commits non-durably, so before the first
/// checkpoint this is the load-time value (`Some(0)` for a fresh map) —
/// NOT the crash-durable `None` a copy of the open `.rdb` would show.
/// A live read is the only portable observation: redb 4.2.0 takes an
/// exclusive whole-file lock, which is advisory on Unix but mandatory
/// on Windows, so copying the open `.rdb` fails there with OS error 33.
///
/// It still pins the property under test: the only *durable* META
/// writer is `checkpoint`, so any value other than the load-time one
/// proves a checkpoint recorded progress — and the post-checkpoint
/// value equals the durable one, because checkpoints commit with
/// `Durability::Immediate`. What is lost vs the old copy: strict crash
/// fidelity — a hard crash pre-checkpoint would leave META absent
/// rather than `Some(0)` (loader-equivalent outcomes: full rebuild vs
/// replay-from-0, both correct). A clean close+reopen cannot recover
/// that distinction either: dropping the `Database` flushes pending
/// non-durable commits, so a reopened handle reads `Some(0)` just like
/// the live one.
pub(crate) fn live_meta_idx_size(nm: &RedbNeedleMap) -> Option<u64> {
nm.live_meta_idx_size()
/// The `.idx` size recorded in the durable state of the `.rdb` at
/// `rdb_path`, read from a copy taken while the map may still be open:
/// exactly what a crash would leave behind. `None` when nothing durable
/// has been recorded yet.
pub(crate) fn durable_idx_size(rdb_path: &Path) -> Option<u64> {
let copy = rdb_path.with_extension("crash-copy.rdb");
std::fs::copy(rdb_path, &copy).unwrap();
let db = Database::open(&copy).unwrap();
let txn = db.begin_read().unwrap();
let meta = txn.open_table(META_TABLE).ok()?;
let size = meta.get(META_IDX_SIZE).unwrap().map(|g| g.value());
drop(meta);
drop(txn);
drop(db);
let _ = std::fs::remove_file(&copy);
size
}
}
@@ -1777,7 +1662,7 @@ mod tests {
)
.unwrap();
let writer = std::fs::OpenOptions::new()
.write(true)
.append(true)
.open(&idx_path)
.unwrap();
nm.set_idx_file(Box::new(writer), idx_size);
@@ -1940,7 +1825,7 @@ mod tests {
let mut live = 0u64;
nm.ascending_visit(|_, _| {
live += 1;
Ok::<(), String>(())
Ok(())
})
.unwrap();
assert_eq!(live, N - 1);
@@ -2153,7 +2038,7 @@ mod tests {
let mut visited = Vec::new();
nm.ascending_visit(|id, nv| {
visited.push((id, nv.size));
Ok::<(), String>(())
Ok(())
})
.unwrap();
@@ -2251,14 +2136,8 @@ mod tests {
// server opens one redb database per volume, so the process-wide
// ceiling is roughly (volumes x budget).
assert_eq!(NeedleMapKind::Redb.redb_cache_bytes(), 4 * 1024 * 1024);
assert_eq!(
NeedleMapKind::RedbMedium.redb_cache_bytes(),
8 * 1024 * 1024
);
assert_eq!(
NeedleMapKind::RedbLarge.redb_cache_bytes(),
16 * 1024 * 1024
);
assert_eq!(NeedleMapKind::RedbMedium.redb_cache_bytes(), 8 * 1024 * 1024);
assert_eq!(NeedleMapKind::RedbLarge.redb_cache_bytes(), 16 * 1024 * 1024);
}
#[test]
@@ -2287,7 +2166,7 @@ mod tests {
#[test]
fn test_redb_checkpoint_is_explicit_and_due_every_interval() {
use test_support::live_meta_idx_size;
use test_support::durable_idx_size;
// Every non-durable redb commit leaves bookkeeping behind until a
// durable one clears it, so a writable map asks for a checkpoint on
@@ -2297,12 +2176,8 @@ mod tests {
let dir = tempfile::tempdir().unwrap();
let (mut nm, db_path, _idx_path) = open_writable_redb(dir.path());
for i in 1..EXPECTED_INTERVAL {
nm.put(
NeedleId(i),
Offset::from_actual_offset((i * 8) as i64),
Size(1),
)
.unwrap();
nm.put(NeedleId(i), Offset::from_actual_offset((i * 8) as i64), Size(1))
.unwrap();
assert!(!nm.checkpoint_due(), "due after only {i} writes");
}
nm.put(
@@ -2312,29 +2187,20 @@ mod tests {
)
.unwrap();
assert!(nm.checkpoint_due());
// No checkpoint taken yet: META still holds the load-time .idx size.
// put() only commits non-durably, so the live value is unchanged.
assert_eq!(
live_meta_idx_size(&nm),
Some(0),
"put() must not record checkpoint progress"
);
assert_eq!(durable_idx_size(&db_path), None, "put() must not commit durably");
nm.checkpoint(true).unwrap();
assert!(!nm.checkpoint_due());
assert_eq!(
live_meta_idx_size(&nm),
durable_idx_size(&db_path),
Some(EXPECTED_INTERVAL * NEEDLE_MAP_ENTRY_SIZE as u64),
"checkpoint records how much of the .idx the table reflects"
);
// Everything is durable after the checkpoint, so the map is closed
// first and the snapshot sees the same bytes on every platform.
// (Copying while open fails on Windows, where redb's file lock is
// mandatory: what a crash leaves.)
drop(nm);
// Snapshot the .rdb while the map is still open: what a crash leaves.
let crash_copy = dir.path().join("crash.rdb");
std::fs::copy(&db_path, &crash_copy).unwrap();
drop(nm);
let db = Database::open(&crash_copy).unwrap();
let txn = db.begin_read().unwrap();
let table = txn.open_table(NEEDLE_TABLE).unwrap();
@@ -2349,12 +2215,8 @@ mod tests {
let dir = tempfile::tempdir().unwrap();
let (mut nm, db_path, idx_path) = open_writable_redb(dir.path());
for i in 1..=5u64 {
nm.put(
NeedleId(i),
Offset::from_actual_offset((i * 8) as i64),
Size(1),
)
.unwrap();
nm.put(NeedleId(i), Offset::from_actual_offset((i * 8) as i64), Size(1))
.unwrap();
}
nm.close();
drop(nm);
@@ -2378,12 +2240,8 @@ mod tests {
let dir = tempfile::tempdir().unwrap();
let (mut nm, db_path, idx_path) = open_writable_redb(dir.path());
for i in 1..=5u64 {
nm.put(
NeedleId(i),
Offset::from_actual_offset((i * 8) as i64),
Size(1),
)
.unwrap();
nm.put(NeedleId(i), Offset::from_actual_offset((i * 8) as i64), Size(1))
.unwrap();
}
// Drop without close(): redb makes the table durable on drop, but the
// recorded .idx size stays at its load-time value (0), so the reload
@@ -205,18 +205,6 @@ mod tests {
use std::os::unix::fs::FileExt;
borrowed.read_exact_at(&mut buf, 0).unwrap();
}
#[cfg(windows)]
{
use std::os::windows::fs::FileExt;
let mut filled = 0;
let mut at = 0;
while filled < buf.len() {
let n = borrowed.seek_read(&mut buf[filled..], at).unwrap();
assert!(n != 0, "unexpected EOF in seek_read");
filled += n;
at += n as u64;
}
}
assert_eq!(&buf, b"first");
}
@@ -132,7 +132,9 @@ mod tests {
let mut seen = SeenKeys::new(10_000, FALSE_POSITIVE_RATE);
// Fresh keys may occasionally collide (that is the false-positive
// rate), but only rarely.
let fresh_reported_seen = (0..10_000u64).filter(|&key| seen.test_and_add(key)).count();
let fresh_reported_seen = (0..10_000u64)
.filter(|&key| seen.test_and_add(key))
.count();
assert!(
fresh_reported_seen < 50,
"fresh keys reported seen: {fresh_reported_seen}"
@@ -18,7 +18,6 @@ use std::sync::{Mutex, RwLock};
use super::file_pool::pooled_index_files;
use crate::storage::idx;
use crate::storage::io::read_exact_at;
use crate::storage::needle_map::{CompactNeedleMap, NeedleMapMetric, NeedleValue};
use crate::storage::types::*;
@@ -134,7 +133,9 @@ impl SortedFileNeedleMap {
}
let file = pooled_index_files()
.borrow(&self.db_file_name, false)
.map_err(|e| io::Error::new(e.kind(), format!("open {}: {}", self.db_file_name, e)))?;
.map_err(|e| {
io::Error::new(e.kind(), format!("open {}: {}", self.db_file_name, e))
})?;
match search_sorted_index(&file, self.db_file_size, key)? {
Some((_, offset, size)) => Ok(Some(NeedleValue { offset, size })),
None => Ok(None),
@@ -317,11 +318,9 @@ impl SortedFileNeedleMap {
Ok(())
}
/// Visit all live entries in ascending order by needle ID.
pub fn ascending_visit<F, E>(&self, mut f: F) -> Result<(), E>
pub fn ascending_visit<F>(&self, mut f: F) -> Result<(), String>
where
F: FnMut(NeedleId, &NeedleValue) -> Result<(), E>,
E: From<String>,
F: FnMut(NeedleId, &NeedleValue) -> Result<(), String>,
{
let mut visit_error = None;
self.visit_live_entries(|id, nv| {
@@ -331,7 +330,7 @@ impl SortedFileNeedleMap {
}
Ok(())
})
.map_err(|e| visit_error.take().unwrap_or_else(|| E::from(e.to_string())))
.map_err(|e| visit_error.take().unwrap_or_else(|| e.to_string()))
}
pub fn iter_entries(&self) -> io::Result<Vec<(NeedleId, NeedleValue)>> {
@@ -523,6 +522,32 @@ fn search_sorted_index(
Ok(None)
}
fn read_exact_at(file: &File, buf: &mut [u8], offset: u64) -> io::Result<()> {
#[cfg(unix)]
{
use std::os::unix::fs::FileExt;
file.read_exact_at(buf, offset)
}
#[cfg(windows)]
{
use std::os::windows::fs::FileExt;
let mut filled = 0;
let mut at = offset;
while filled < buf.len() {
let n = file.seek_read(&mut buf[filled..], at)?;
if n == 0 {
return Err(io::Error::new(
io::ErrorKind::UnexpectedEof,
"unexpected EOF in seek_read",
));
}
filled += n;
at += n as u64;
}
Ok(())
}
}
fn write_at(file: &File, buf: &[u8], offset: u64) -> io::Result<()> {
#[cfg(unix)]
{
@@ -675,11 +700,10 @@ mod tests {
// without a reload — the same contract Go's Get has, where callers
// check size.is_deleted().
assert!(m.get(NeedleId(2)).unwrap().unwrap().size.is_deleted());
assert!(
m.delete(NeedleId(2), Offset::from_actual_offset(16))
.unwrap()
.is_none()
);
assert!(m
.delete(NeedleId(2), Offset::from_actual_offset(16))
.unwrap()
.is_none());
assert!(!m.get(NeedleId(1)).unwrap().unwrap().size.is_deleted());
}
@@ -975,8 +999,7 @@ mod tests {
// The retry is a no-op: no second tombstone, no double counting.
assert_eq!(
m.delete(NeedleId(1), Offset::from_actual_offset(8))
.unwrap(),
m.delete(NeedleId(1), Offset::from_actual_offset(8)).unwrap(),
None
);
assert_eq!(m.deleted_count(), deleted_before + 2);
@@ -1006,8 +1029,7 @@ mod tests {
);
// And a retry must not append a second tombstone for it.
assert_eq!(
m.delete(NeedleId(1), Offset::from_actual_offset(8))
.unwrap(),
m.delete(NeedleId(1), Offset::from_actual_offset(8)).unwrap(),
None
);
}
@@ -1037,7 +1059,7 @@ mod tests {
let mut visited = Vec::new();
m.ascending_visit(|id, _| {
visited.push(id);
Ok::<(), String>(())
Ok(())
})
.unwrap();
assert_eq!(visited, vec![NeedleId(2)]);
File diff suppressed because it is too large Load Diff
@@ -1,374 +0,0 @@
//! Merging a peer's `.ecj` deletion ids into a local EC journal. Mirrors Go's
//! `Store.MergeEcJournal` (`weed/storage/store_ec_journal.go`).
use std::collections::HashSet;
use std::io;
use std::path::Path;
use std::sync::RwLock;
use crate::storage::erasure_coding::ecj_merge::{append_ecj_ids, read_ecj_ids};
use crate::storage::store::Store;
use crate::storage::types::{NeedleId, VolumeId};
/// How often an unmounted merge re-reads a journal that changed under it. Only
/// a mount-delete-unmount or a concurrent merge between the read and the
/// append changes it, so one retry is nearly always enough.
const ECJ_MERGE_ATTEMPTS: usize = 5;
/// Fold a peer's deletion `ids` into the local journal of EC volume `vid` on
/// the receiving disk, the one whose data directory is `data_dir`; `ecj_path`
/// is that journal's path in the disk's index directory. Appends only the ids
/// the journal lacks and returns how many it added. Blocking: call it from
/// `spawn_blocking`.
///
/// A mounted volume owns its journal: the merge goes through its open handle
/// and in-memory set. That is the receiving disk's own runtime for `vid`,
/// wherever its journal lives (it may sit in the data dir rather than
/// `ecj_path`'s index dir), else a sibling runtime journaling into `ecj_path`
/// itself: disks sharing one index directory, or reconciliation mounting `vid`
/// on a disk that journals into another's (#9212). Otherwise `ecj_path` is
/// appended to under the store write lock, which mounts take, so no mount can
/// open it mid-append. The read that computes the delta runs outside the lock,
/// and a journal that changed in between — including by a concurrent merge —
/// is re-read.
pub fn merge_ec_journal(
store: &RwLock<Store>,
vid: VolumeId,
data_dir: &str,
ecj_path: &str,
ids: &HashSet<NeedleId>,
) -> io::Result<usize> {
merge_ec_journal_with(store, vid, data_dir, ecj_path, ids, read_ecj_ids)
}
/// `merge_ec_journal` with the unlocked journal read injected, so a test can
/// mount the volume between that read and the append.
fn merge_ec_journal_with(
store: &RwLock<Store>,
vid: VolumeId,
data_dir: &str,
ecj_path: &str,
ids: &HashSet<NeedleId>,
mut read: impl FnMut(&str) -> io::Result<(HashSet<NeedleId>, u64)>,
) -> io::Result<usize> {
for _ in 0..ECJ_MERGE_ATTEMPTS {
{
let mut store = store
.write()
.map_err(|_| io::Error::other("store lock poisoned"))?;
if let Some(primary) = mounted_ec_journal(&store, vid, data_dir, ecj_path)? {
let ecv = store.locations[primary]
.find_ec_volume_mut(vid)
.expect("mounted journal runtime");
let journal_path = ecv.ecj_file_name();
let added = ecv.merge_journal(ids)?;
// Publish to the holders of the file the merge wrote to — the
// picked runtime's journal may live outside ecj_path, and a
// holder of a different file must not claim ids it lacks.
publish_to_journal_siblings(&store, primary, vid, &journal_path, ids);
return Ok(added);
}
}
let (local, size) = read(ecj_path)?;
let mut store = store
.write()
.map_err(|_| io::Error::other("store lock poisoned"))?;
if let Some(primary) = mounted_ec_journal(&store, vid, data_dir, ecj_path)? {
// Mounted since the read: its handle owns the journal now.
let ecv = store.locations[primary]
.find_ec_volume_mut(vid)
.expect("mounted journal runtime");
let journal_path = ecv.ecj_file_name();
let added = ecv.merge_journal(ids)?;
publish_to_journal_siblings(&store, primary, vid, &journal_path, ids);
return Ok(added);
}
// The path append registers as a writer so a mount compacting this
// journal cannot swap its inode underneath it (the write itself is
// already serialized with mounts by the store lock).
let _ecj_write =
crate::storage::erasure_coding::ecj_registry::begin_ecj_write(ecj_path);
if let Some(added) = append_ecj_ids(ecj_path, &local, ids, size)? {
return Ok(added);
}
}
Err(io::Error::other(format!(
"ec volume {}: journal {} kept changing during merge",
vid.0, ecj_path
)))
}
/// The disk index of the runtime holding `ecj_path` open, if any: the disk at
/// `data_dir`'s own, else the first sibling journaling into it.
fn mounted_ec_journal(
store: &Store,
vid: VolumeId,
data_dir: &str,
ecj_path: &str,
) -> io::Result<Option<usize>> {
let owner = store
.locations
.iter()
.position(|loc| Path::new(&loc.directory) == Path::new(data_dir))
.ok_or_else(|| {
io::Error::other(format!(
"ec volume {}: no disk at {} owns journal {}",
vid.0, data_dir, ecj_path
))
})?;
let runtime = if store.locations[owner].has_ec_volume(vid) {
Some(owner)
} else {
store.locations.iter().position(|loc| {
loc.find_ec_volume(vid)
.is_some_and(|ecv| Path::new(&ecv.ecj_file_name()) == Path::new(ecj_path))
})
};
Ok(runtime)
}
/// Publishes merged ids into every other runtime journaling into `ecj_path`.
fn publish_to_journal_siblings(
store: &Store,
primary: usize,
vid: VolumeId,
ecj_path: &str,
ids: &HashSet<NeedleId>,
) {
for (i, loc) in store.locations.iter().enumerate() {
if i == primary {
continue;
}
if let Some(ecv) = loc.find_ec_volume(vid) {
if Path::new(&ecv.ecj_file_name()) == Path::new(ecj_path) {
ecv.publish_merged_ids(ids);
}
}
}
}
#[cfg(test)]
mod tests {
use super::*;
use crate::config::MinFreeSpace;
use crate::storage::needle_map::NeedleMapKind;
use crate::storage::types::DiskType;
use crate::storage::volume::{VifEcShardConfig, VifVolumeInfo};
use tempfile::TempDir;
const COLLECTION: &str = "c";
const VID: VolumeId = VolumeId(9);
/// A store with one disk per entry of `data`, all sharing `idx` when given,
/// else each indexing into its own data dir.
fn make_store(tmp: &TempDir, data: &[&str], idx: Option<&str>) -> RwLock<Store> {
let mut store = Store::new(NeedleMapKind::InMemory);
for d in data {
let dir = tmp.path().join(d).to_string_lossy().into_owned();
let idx_dir = idx
.map(|i| tmp.path().join(i).to_string_lossy().into_owned())
.unwrap_or_else(|| dir.clone());
std::fs::create_dir_all(&dir).unwrap();
std::fs::create_dir_all(&idx_dir).unwrap();
store
.add_location(
&dir,
&idx_dir,
100,
DiskType::HardDrive,
MinFreeSpace::Percent(0.0),
Vec::new(),
)
.unwrap();
}
RwLock::new(store)
}
fn dir(tmp: &TempDir, d: &str) -> String {
tmp.path().join(d).to_string_lossy().into_owned()
}
fn records(ids: &[u64]) -> Vec<u8> {
let ids: Vec<NeedleId> = ids.iter().copied().map(NeedleId).collect();
crate::storage::erasure_coding::ecj_merge::encode_ecj_ids(&ids)
}
fn id_set(ids: &[u64]) -> HashSet<NeedleId> {
ids.iter().copied().map(NeedleId).collect()
}
/// Shard 0 of `VID` and its `.vif` in `data_dir`.
fn write_shard0(data_dir: &str) {
let base = format!("{}/{}_{}", data_dir, COLLECTION, VID.0);
std::fs::write(format!("{}.ec00", base), b"shard data nonempty").unwrap();
let vif = VifVolumeInfo {
version: 3,
ec_shard_config: Some(VifEcShardConfig {
data_shards: 10,
parity_shards: 4,
..Default::default()
}),
..Default::default()
};
std::fs::write(
format!("{}.vif", base),
serde_json::to_string(&vif).unwrap(),
)
.unwrap();
}
/// `VID`'s `.ecx` and a `.ecj` holding `deleted` in `dir`; returns the
/// journal path.
fn write_index(dir: &str, deleted: &[u64]) -> String {
let base = format!("{}/{}_{}", dir, COLLECTION, VID.0);
std::fs::write(format!("{}.ecx", base), vec![0u8; 16]).unwrap();
let ecj = format!("{}.ecj", base);
std::fs::write(&ecj, records(deleted)).unwrap();
ecj
}
fn deleted_on(store: &RwLock<Store>, disk: usize, id: u64) -> bool {
store.read().unwrap().locations[disk]
.find_ec_volume(VID)
.expect("mounted")
.is_needle_deleted(NeedleId(id))
}
/// Disks sharing one index directory all hold the same journal path.
/// Copying shards onto a disk that has not mounted `vid` must still reach
/// the sibling runtime holding that journal open.
#[test]
fn shared_index_dir_reaches_sibling_mount() {
let tmp = TempDir::new().unwrap();
let store = make_store(&tmp, &["d0", "d1"], Some("idx"));
write_shard0(&dir(&tmp, "d0"));
let ecj = write_index(&dir(&tmp, "idx"), &[1]);
store.write().unwrap().locations[0]
.mount_ec_shards(VID, COLLECTION, &[0], "")
.unwrap();
assert_eq!(
store.read().unwrap().locations[0]
.find_ec_volume(VID)
.unwrap()
.ecj_file_name(),
ecj
);
let added =
merge_ec_journal(&store, VID, &dir(&tmp, "d1"), &ecj, &id_set(&[1, 2])).unwrap();
assert_eq!(added, 1);
assert!(
deleted_on(&store, 0, 2),
"the mounted sibling must see id 2"
);
assert_eq!(std::fs::read(&ecj).unwrap(), records(&[1, 2]));
}
/// Every runtime holding the journal open must see merged ids in memory.
#[test]
fn shared_journal_reaches_every_holder() {
let tmp = TempDir::new().unwrap();
let store = make_store(&tmp, &["d0", "d1"], Some("idx"));
write_shard0(&dir(&tmp, "d0"));
write_shard0(&dir(&tmp, "d1"));
let ecj = write_index(&dir(&tmp, "idx"), &[1]);
for i in 0..2 {
store.write().unwrap().locations[i]
.mount_ec_shards(VID, COLLECTION, &[0], "")
.unwrap();
}
let added =
merge_ec_journal(&store, VID, &dir(&tmp, "d1"), &ecj, &id_set(&[1, 2])).unwrap();
assert_eq!(added, 1);
assert!(
deleted_on(&store, 0, 2) && deleted_on(&store, 1, 2),
"every journal holder must see the merged id"
);
assert_eq!(std::fs::read(&ecj).unwrap(), records(&[1, 2]));
}
/// The picked runtime may journal to a different file than the copied
/// one — its index lives in its data directory while a sibling's lives
/// in the index directory. The ids must be published only to holders of
/// the file they were written to.
#[test]
fn publishes_to_actual_journal_holders() {
let tmp = TempDir::new().unwrap();
let store = make_store(&tmp, &["d0", "d1"], Some("idx"));
write_shard0(&dir(&tmp, "d0"));
write_shard0(&dir(&tmp, "d1"));
let data_ecj = write_index(&dir(&tmp, "d0"), &[1]);
let idx_ecj = write_index(&dir(&tmp, "idx"), &[1]);
for i in 0..2 {
store.write().unwrap().locations[i]
.mount_ec_shards(VID, COLLECTION, &[0], "")
.unwrap();
}
assert_eq!(
store.read().unwrap().locations[0]
.find_ec_volume(VID)
.unwrap()
.ecj_file_name(),
data_ecj
);
let added =
merge_ec_journal(&store, VID, &dir(&tmp, "d0"), &idx_ecj, &id_set(&[1, 2])).unwrap();
assert_eq!(added, 1);
assert_eq!(std::fs::read(&data_ecj).unwrap(), records(&[1, 2]));
assert_eq!(std::fs::read(&idx_ecj).unwrap(), records(&[1]));
assert!(deleted_on(&store, 0, 2));
assert!(
!deleted_on(&store, 1, 2),
"a different journal's holder must not claim the merged id"
);
}
/// A sibling disk can mount `vid` from the receiving disk's index (#9212)
/// while the merge reads the journal unlocked. The merge must go through
/// that mount and report what it added.
#[test]
fn mount_during_read_is_merged_through_and_counted() {
let tmp = TempDir::new().unwrap();
let store = make_store(&tmp, &["d0", "d1"], None);
let owner = dir(&tmp, "d0");
let ecj = write_index(&owner, &[1]);
write_shard0(&dir(&tmp, "d1"));
let mut mounted = false;
let read = |path: &str| {
let read = read_ecj_ids(path);
if !mounted {
store.write().unwrap().locations[1]
.mount_ec_shards_with_idx_dir(VID, COLLECTION, &[0], &owner, "")
.unwrap();
mounted = true;
}
read
};
let added =
merge_ec_journal_with(&store, VID, &owner, &ecj, &id_set(&[1, 2, 3]), read).unwrap();
assert_eq!(added, 2, "the ids merged through the new mount are counted");
assert!(deleted_on(&store, 1, 2) && deleted_on(&store, 1, 3));
assert_eq!(std::fs::read(&ecj).unwrap(), records(&[1, 2, 3]));
}
/// An unmounted journal is merged on disk and a repeat adds nothing; a
/// data dir that is no disk is refused.
#[test]
fn unmounted_journal_is_idempotent() {
let tmp = TempDir::new().unwrap();
let store = make_store(&tmp, &["d0"], Some("idx"));
let ecj = write_index(&dir(&tmp, "idx"), &[1, 2]);
let d0 = dir(&tmp, "d0");
for want in [2, 0, 0] {
let added = merge_ec_journal(&store, VID, &d0, &ecj, &id_set(&[2, 3, 4])).unwrap();
assert_eq!(added, want);
}
assert_eq!(std::fs::read(&ecj).unwrap(), records(&[1, 2, 3, 4]));
assert!(
merge_ec_journal(&store, VID, &dir(&tmp, "elsewhere"), &ecj, &id_set(&[1])).is_err()
);
}
}
+4 -37
View File
@@ -8,7 +8,7 @@ use std::path::Path;
use tracing::{info, warn};
use crate::storage::disk_location::{DiskLocation, parse_collection_volume_id_pub};
use crate::storage::disk_location::{parse_collection_volume_id_pub, DiskLocation};
use crate::storage::store::Store;
use crate::storage::types::VolumeId;
@@ -131,12 +131,6 @@ impl Store {
let Some(base) = name.strip_suffix(".ecx") else {
continue;
};
// A 0-byte .ecx is a corrupt stub from a failed copy, not a
// credible owner — skip it so the scan keeps looking for a
// real index on a sibling disk (Go's indexEcxOwners).
if !ent.metadata().is_ok_and(|m| m.len() > 0) {
continue;
}
let Some((collection, vid)) = parse_collection_volume_id_pub(base) else {
continue;
};
@@ -292,7 +286,9 @@ fn collect_shard_disk_volumes(loc: &DiskLocation) -> HashMap<EcKey, Vec<String>>
let Some((collection, vid)) = parse_collection_volume_id_pub(base) else {
continue;
};
out.entry(EcKey { collection, vid }).or_default().push(name);
out.entry(EcKey { collection, vid })
.or_default()
.push(name);
}
out
}
@@ -423,33 +419,4 @@ mod tests {
let post = fs::read(dir0.join(format!("{}_{}.ecx", collection, vid))).unwrap();
assert_eq!(post, ecx_local, "mirror overwrote dir0's existing .ecx");
}
/// The mirror shares Go's indexEcxOwners, which skips a 0-byte `.ecx`:
/// a stub must not be chosen as the source to mirror from.
#[test]
fn mirror_owner_index_skips_zero_byte_ecx() {
let tmp = TempDir::new().unwrap();
let dir0 = tmp.path().join("data0");
let dir1 = tmp.path().join("data1");
fs::create_dir_all(&dir0).unwrap();
fs::create_dir_all(&dir1).unwrap();
let collection = "video-recordings";
let vid = 4123u32;
plant_ecx(&dir0, collection, vid, b"");
plant_ecx(&dir1, collection, vid, &[0xA1u8; 20]);
let mut store = Store::new(NeedleMapKind::InMemory);
add_loc(&mut store, &dir0);
add_loc(&mut store, &dir1);
let owners = store.index_ecx_owners_for_mirror();
let owner = owners
.get(&EcKey {
collection: collection.to_string(),
vid: VolumeId(vid),
})
.expect("the valid .ecx on disk 1 must be indexed");
assert_eq!(owner.location, 1);
}
}

Some files were not shown because too many files have changed in this diff Show More