mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-10-07 15:15:52 +00:00
Compare commits
70
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
56fbc3deb5 | ||
|
|
079a7d8ba3 | ||
|
|
0bcebf708c | ||
|
|
576837f1b2 | ||
|
|
3288d90b21 | ||
|
|
199d78539a | ||
|
|
ec261c5fbc | ||
|
|
cac6cd4b16 | ||
|
|
d829275de6 | ||
|
|
23893eb378 | ||
|
|
32e77ff980 | ||
|
|
de75450655 | ||
|
|
b14dd1cee4 | ||
|
|
3a65365f6e | ||
|
|
cb053e601a | ||
|
|
8ebef03509 | ||
|
|
c739e5cf78 | ||
|
|
c0a751072d | ||
|
|
52c6df3bec | ||
|
|
aa5b337716 | ||
|
|
abdb95dc87 | ||
|
|
076fd24186 | ||
|
|
90eb4ec091 | ||
|
|
e710cfc4b0 | ||
|
|
cf144d5eb2 | ||
|
|
825c3dff8b | ||
|
|
a7a590474d | ||
|
|
8c67b75190 | ||
|
|
7a19961074 | ||
|
|
16e66b1bad | ||
|
|
7e809c9991 | ||
|
|
3c17c5146e | ||
|
|
483dd4b12e | ||
|
|
9d1c24d80d | ||
|
|
76ddde6a6d | ||
|
|
0d93dec145 | ||
|
|
d9b69a7f76 | ||
|
|
2b5fdc639f | ||
|
|
d7a02567e3 | ||
|
|
39bc9cd0ef | ||
|
|
f4ef37e752 | ||
|
|
10b0f2b8ad | ||
|
|
eafe79ebff | ||
|
|
562afa8ec9 | ||
|
|
8c1ebbee32 | ||
|
|
c9ade3f9fd | ||
|
|
6c07a5fdd0 | ||
|
|
07da302da0 | ||
|
|
68df7511f6 | ||
|
|
793ce06b10 | ||
|
|
0ca484c354 | ||
|
|
3e679e925e | ||
|
|
52fb9f93ff | ||
|
|
0ff7794c54 | ||
|
|
94edd0a6d4 | ||
|
|
30069f3e45 | ||
|
|
fa77cde7da | ||
|
|
35b090a4df | ||
|
|
1445960f8c | ||
|
|
43abe21ffa | ||
|
|
67034dee12 | ||
|
|
8d97284d0a | ||
|
|
d8f926cf46 | ||
|
|
8a9563e53d | ||
|
|
90f8c4378f | ||
|
|
eeec9ec09a | ||
|
|
7c7834c98e | ||
|
|
2a42d56437 | ||
|
|
164c3db606 | ||
|
|
6f9becaa37 |
Executable
+240
@@ -0,0 +1,240 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Check JWT extraction and, with --context, real Helm upgrade persistence."""
|
||||
|
||||
import argparse
|
||||
import base64
|
||||
import json
|
||||
from pathlib import Path
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
import uuid
|
||||
|
||||
try:
|
||||
import tomllib
|
||||
except ModuleNotFoundError: # CI also exercises Python 3.10.
|
||||
import tomli as tomllib
|
||||
|
||||
|
||||
ROOT = Path(__file__).resolve().parents[2]
|
||||
CHART = ROOT / "k8s/charts/seaweedfs"
|
||||
SECTIONS = ("jwt.signing", "jwt.signing.read", "jwt.filer_signing", "jwt.filer_signing.read")
|
||||
KEYS = {section: f"active-{index}" for index, section in enumerate(SECTIONS)}
|
||||
ESCAPED_HEADER = '["jw\\u0074".signing]\nkey = "existing"'
|
||||
|
||||
|
||||
def run(*args):
|
||||
return subprocess.run(args, check=True, text=True, capture_output=True).stdout
|
||||
|
||||
|
||||
def keys(raw):
|
||||
document = tomllib.loads(raw)
|
||||
result = {}
|
||||
for section in SECTIONS:
|
||||
table = document
|
||||
for part in section.split("."):
|
||||
table = table.get(part, {})
|
||||
if "key" in table:
|
||||
result[section] = table["key"]
|
||||
return result
|
||||
|
||||
|
||||
def fixtures():
|
||||
canonical = "\n".join(f'[{section}]\nkey = "{value}"' for section, value in KEYS.items())
|
||||
yield "canonical / final line without newline", canonical
|
||||
yield "commented stale keys", canonical.replace("key =", '# key = "stale"\nkey =')
|
||||
yield "commented headers / mixed line endings", "\n".join(
|
||||
f'# [{section}]\r\n# key = "stale"\r\n[{section}]\nkey = "{value}"'
|
||||
for section, value in KEYS.items()
|
||||
)
|
||||
yield "indented headers and keys / trailing comments", "\n".join(
|
||||
f' \t[ {section} ] \t# header [note]\n \tkey \t= \t"{value}" # key comment'
|
||||
for section, value in KEYS.items()
|
||||
)
|
||||
yield "CRLF", canonical.replace("\n", "\r\n")
|
||||
yield "quoted and spaced section names", "\n".join(
|
||||
f'["{section.split(".")[0]}" . \'{section.split(".")[1]}\''
|
||||
+ (f' . "{section.split(".")[2]}"' if section.count(".") == 2 else "")
|
||||
+ f'] # original table\nkey = "{value}"'
|
||||
for section, value in KEYS.items()
|
||||
)
|
||||
yield "unrelated quoted header before JWT keys", '["custom section"]\nkey = "other"\n' + canonical
|
||||
yield "brackets in comments", canonical.replace("key =", "# consult [notes]\nkey =")
|
||||
yield "quoted values and quoted key names", "\n".join((
|
||||
'[jwt.signing]\n"key" = "brackets[inside]#value"',
|
||||
"[jwt.signing.read]\n'key' = 'literal\\path[#value]'",
|
||||
r'[jwt.filer_signing]' + '\n' + r'key = "escaped\"quote\\slash\u0041"',
|
||||
'[jwt.filer_signing.read]\nkey = ""',
|
||||
))
|
||||
yield "absent keys and sections / unrelated key", '\n'.join((
|
||||
'[jwt.signing]\nexpires_after_seconds = 10',
|
||||
'[jwt.signing.read]\nkey = "read-only"',
|
||||
'[unrelated]\nkey = "not-a-jwt-key"',
|
||||
'# [jwt.filer_signing]\n# key = "not-active"',
|
||||
))
|
||||
yield "no existing security.toml", ""
|
||||
|
||||
|
||||
def check_generated(value):
|
||||
decoded = base64.b64decode(value, validate=True).decode("ascii")
|
||||
assert len(decoded) == 10 and decoded.isascii() and decoded.isalnum(), "invalid generated JWT key"
|
||||
|
||||
|
||||
def check_helpers(helm, reference_helm):
|
||||
# Exercise the real helper, not a second implementation of its matching rules.
|
||||
with tempfile.TemporaryDirectory(prefix="helm-jwt-helper-") as directory:
|
||||
chart = Path(directory)
|
||||
(chart / "templates").mkdir()
|
||||
(chart / "Chart.yaml").write_text("apiVersion: v2\nname: jwt-regression\nversion: 0.0.0\n")
|
||||
shutil.copyfile(CHART / "templates/shared/_helpers.tpl", chart / "templates/_helpers.tpl")
|
||||
entries = [
|
||||
json.dumps(section) + ': {{ include "seaweedfs.existingTomlKey" (list '
|
||||
+ json.dumps(section) + ' .Values.raw) | toJson }}'
|
||||
for section in SECTIONS
|
||||
]
|
||||
template = chart / "templates/keys.yaml"
|
||||
prefix = '{"apiVersion":"v1","kind":"ConfigMap","metadata":{"name":"keys"},"data":{'
|
||||
helper_template = prefix + ",".join(entries) + "}}"
|
||||
reference_entries = [
|
||||
json.dumps(section) + ': {{ dig '
|
||||
+ " ".join(json.dumps(part) for part in section.split("."))
|
||||
+ ' "key" "__ABSENT__" (fromToml .Values.raw) | toJson }}'
|
||||
for section in SECTIONS
|
||||
]
|
||||
reference_template = prefix + ",".join(reference_entries) + "}}"
|
||||
|
||||
def render(binary, source, raw):
|
||||
template.write_text(source)
|
||||
values = chart / "input.json"
|
||||
values.write_text(json.dumps({"raw": raw}))
|
||||
output = run(binary, "template", "keys", str(chart), "-f", str(values))
|
||||
return json.loads(output[output.index("{"):])["data"]
|
||||
|
||||
for name, raw in fixtures():
|
||||
expected = keys(raw)
|
||||
tokens = render(helm, helper_template, raw)
|
||||
actual = {section: tomllib.loads("key = " + token)["key"]
|
||||
for section, token in tokens.items() if token != ""}
|
||||
assert actual == expected, f"{name}: extracted keys differ from stored TOML"
|
||||
if reference_helm:
|
||||
reference = render(reference_helm, reference_template, raw)
|
||||
reference = {section: value for section, value in reference.items() if value != "__ABSENT__"}
|
||||
assert actual == reference, f"{name}: keys differ from fromToml/dig"
|
||||
print(f"PASS helper: {name}")
|
||||
|
||||
# A present but unsupported value must not silently become a fresh key.
|
||||
try:
|
||||
render(helm, helper_template, '[jwt.signing]\nkey = """multi\nline"""')
|
||||
except subprocess.CalledProcessError as error:
|
||||
assert "refusing to replace an existing key" in error.stderr, error.stderr
|
||||
else:
|
||||
raise AssertionError("multiline existing key was silently accepted or replaced")
|
||||
print("PASS helper: unsupported existing value fails without rotation")
|
||||
|
||||
try:
|
||||
render(helm, helper_template, ESCAPED_HEADER)
|
||||
except subprocess.CalledProcessError as error:
|
||||
assert "unsupported quoted section header" in error.stderr, error.stderr
|
||||
else:
|
||||
raise AssertionError("unsupported quoted header silently rotated its key")
|
||||
print("PASS helper: unsupported quoted header fails without rotation")
|
||||
|
||||
|
||||
def check_upgrades(helm, context):
|
||||
namespace = "jwt-key-persist-" + uuid.uuid4().hex[:8]
|
||||
current = "jk-seaweedfs-security-config"
|
||||
legacy = "seaweedfs-security-config"
|
||||
kubectl = ["kubectl", "--context", context, "-n", namespace]
|
||||
release_args = ["jk", str(CHART), "--kube-context", context, "-n", namespace]
|
||||
# No workload is needed to exercise Helm's real ConfigMap lookup and update.
|
||||
for setting in (
|
||||
"master.enabled=false", "volume.enabled=false", "filer.enabled=false",
|
||||
"global.seaweedfs.createClusterRole=false",
|
||||
"global.seaweedfs.securityConfig.jwtSigning.volumeWrite=true",
|
||||
"global.seaweedfs.securityConfig.jwtSigning.volumeRead=true",
|
||||
"global.seaweedfs.securityConfig.jwtSigning.filerWrite=true",
|
||||
"global.seaweedfs.securityConfig.jwtSigning.filerRead=true",
|
||||
):
|
||||
release_args += ["--set", setting]
|
||||
|
||||
def stored():
|
||||
cm = json.loads(run(*kubectl, "get", "configmap", current, "-o", "json"))
|
||||
return keys(cm["data"]["security.toml"])
|
||||
|
||||
def upgrade():
|
||||
run(helm, "upgrade", *release_args)
|
||||
return stored()
|
||||
|
||||
def patch(raw):
|
||||
# Seed previous-release content without taking Helm 4's SSA ownership.
|
||||
run(*kubectl, "patch", "configmap", current, "--type=merge", "--field-manager=helm", "-p",
|
||||
json.dumps({"data": {"security.toml": raw}}))
|
||||
|
||||
run(*kubectl, "create", "namespace", namespace)
|
||||
try:
|
||||
run(helm, "install", *release_args)
|
||||
initial = stored()
|
||||
assert set(initial) == set(SECTIONS), "install omitted a JWT section"
|
||||
for value in initial.values():
|
||||
check_generated(value)
|
||||
assert upgrade() == initial, "no-op upgrade changed an existing key"
|
||||
print("PASS upgrade: all four generated keys persist")
|
||||
|
||||
for name, raw in fixtures():
|
||||
patch(raw)
|
||||
actual = upgrade()
|
||||
expected = keys(raw)
|
||||
assert set(actual) == set(SECTIONS), f"{name}: upgrade omitted a JWT section"
|
||||
for section in SECTIONS:
|
||||
if section in expected:
|
||||
assert actual[section] == expected[section], f"{name}: changed {section}"
|
||||
else:
|
||||
check_generated(actual[section])
|
||||
assert upgrade() == actual, f"{name}: subsequent upgrade changed a key"
|
||||
print(f"PASS upgrade: {name}")
|
||||
|
||||
# Migration from the old chart name, followed by precedence of the current name.
|
||||
legacy_raw = "\n".join(f'[{section}]\nkey = "legacy-{index}"'
|
||||
for index, section in enumerate(SECTIONS))
|
||||
run(*kubectl, "create", "configmap", legacy, "--from-literal=security.toml=" + legacy_raw)
|
||||
run(*kubectl, "delete", "configmap", current)
|
||||
assert upgrade() == keys(legacy_raw), "legacy ConfigMap keys were not preserved"
|
||||
print("PASS upgrade: legacy ConfigMap migration")
|
||||
current_raw = next(fixtures())[1]
|
||||
patch(current_raw)
|
||||
assert upgrade() == keys(current_raw), "legacy ConfigMap overrode current ConfigMap"
|
||||
print("PASS upgrade: current ConfigMap takes precedence")
|
||||
|
||||
patch(ESCAPED_HEADER)
|
||||
try:
|
||||
upgrade()
|
||||
except subprocess.CalledProcessError as error:
|
||||
assert "unsupported quoted section header" in error.stderr, error.stderr
|
||||
else:
|
||||
raise AssertionError("unsupported quoted header silently rotated its key")
|
||||
assert keys(run(*kubectl, "get", "configmap", current, "-o",
|
||||
"jsonpath={.data.security\\.toml}"))["jwt.signing"] == "existing"
|
||||
print("PASS upgrade: unsupported quoted header leaves stored key untouched")
|
||||
finally:
|
||||
run(*kubectl, "delete", "namespace", namespace, "--wait=false")
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--helm", default="helm")
|
||||
parser.add_argument("--reference-helm", help="Helm >=3.17 binary for differential checks")
|
||||
parser.add_argument("--context", help="explicit disposable Kubernetes context for live upgrade checks")
|
||||
args = parser.parse_args()
|
||||
print(run(args.helm, "version", "--short").strip())
|
||||
check_helpers(args.helm, args.reference_helm)
|
||||
if args.context:
|
||||
check_upgrades(args.helm, args.context)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
try:
|
||||
main()
|
||||
except subprocess.CalledProcessError as error:
|
||||
print(error.stderr, file=sys.stderr)
|
||||
sys.exit(error.returncode)
|
||||
@@ -3,10 +3,10 @@ name: "helm: lint and test charts"
|
||||
on:
|
||||
push:
|
||||
branches: [ master ]
|
||||
paths: ['k8s/**', '.github/workflows/helm_ci.yml']
|
||||
paths: ['k8s/**', '.github/workflows/helm_ci.yml', '.github/scripts/helm_jwt_keys.py']
|
||||
pull_request:
|
||||
branches: [ master ]
|
||||
paths: ['k8s/**', '.github/workflows/helm_ci.yml']
|
||||
paths: ['k8s/**', '.github/workflows/helm_ci.yml', '.github/scripts/helm_jwt_keys.py']
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
@@ -20,6 +20,14 @@ jobs:
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Set up Helm before fromToml was available
|
||||
uses: azure/setup-helm@v5
|
||||
with:
|
||||
version: v3.16.3
|
||||
|
||||
- name: Record legacy Helm binary
|
||||
run: echo "HELM_LEGACY=$(command -v helm)" >> "$GITHUB_ENV"
|
||||
|
||||
- name: Set up Helm
|
||||
uses: azure/setup-helm@v5
|
||||
with:
|
||||
@@ -44,6 +52,17 @@ jobs:
|
||||
- name: Run chart-testing (lint)
|
||||
run: ct lint --target-branch ${{ github.event.repository.default_branch }} --all --validate-maintainers=false --chart-dirs k8s/charts
|
||||
|
||||
- name: Verify legacy Helm rendering
|
||||
run: |
|
||||
"$HELM_LEGACY" lint k8s/charts/seaweedfs
|
||||
"$HELM_LEGACY" template test k8s/charts/seaweedfs > "$RUNNER_TEMP/legacy-default.yaml"
|
||||
"$HELM_LEGACY" template test k8s/charts/seaweedfs \
|
||||
--set global.seaweedfs.securityConfig.jwtSigning.volumeRead=true \
|
||||
--set global.seaweedfs.securityConfig.jwtSigning.filerWrite=true \
|
||||
--set global.seaweedfs.securityConfig.jwtSigning.filerRead=true \
|
||||
> "$RUNNER_TEMP/legacy-jwt.yaml"
|
||||
|
||||
|
||||
- name: Verify template rendering
|
||||
run: |
|
||||
set -e
|
||||
@@ -57,7 +76,25 @@ jobs:
|
||||
helm template test $CHART_DIR --set s3.enabled=true > /tmp/s3.yaml
|
||||
grep -q "kind: Deployment" /tmp/s3.yaml && grep -q "seaweedfs-s3" /tmp/s3.yaml
|
||||
echo "S3 deployment renders correctly"
|
||||
|
||||
|
||||
echo "=== Testing S3 rollout settings ==="
|
||||
helm template test $CHART_DIR --show-only templates/s3/s3-deployment.yaml \
|
||||
--set s3.enabled=true > /tmp/s3-rollout-defaults.yaml
|
||||
grep -q "terminationGracePeriodSeconds: 10$" /tmp/s3-rollout-defaults.yaml
|
||||
test "$(grep -cE '^ strategy:|^ +lifecycle:' /tmp/s3-rollout-defaults.yaml)" -eq 0
|
||||
helm template test $CHART_DIR --show-only templates/s3/s3-deployment.yaml \
|
||||
--set s3.enabled=true \
|
||||
--set s3.terminationGracePeriodSeconds=30 \
|
||||
--set s3.updateStrategy.type=RollingUpdate \
|
||||
--set s3.updateStrategy.rollingUpdate.maxUnavailable=0 \
|
||||
--set-string 's3.lifecycle.preStop.exec.command={sleep,5}' \
|
||||
> /tmp/s3-rollout.yaml
|
||||
grep -q "terminationGracePeriodSeconds: 30$" /tmp/s3-rollout.yaml
|
||||
grep -A 4 "^ strategy:" /tmp/s3-rollout.yaml | grep -q "maxUnavailable: 0"
|
||||
grep -A 4 "^ strategy:" /tmp/s3-rollout.yaml | grep -q "type: RollingUpdate"
|
||||
grep -A 5 "^ lifecycle:" /tmp/s3-rollout.yaml | grep -q -- "- sleep"
|
||||
echo "S3 rollout settings render correctly"
|
||||
|
||||
echo "=== Testing S3 credentials from an existing secret ==="
|
||||
credential_args=(
|
||||
--set s3.credentials.admin.existingSecret=minio-root
|
||||
@@ -116,6 +153,21 @@ jobs:
|
||||
grep -q "security-config" /tmp/security.yaml
|
||||
echo "Security configuration renders correctly"
|
||||
|
||||
echo "=== Testing secure bucket-creation hook certificate mounts ==="
|
||||
helm template test $CHART_DIR \
|
||||
--show-only templates/shared/post-install-bucket-hook.yaml \
|
||||
--set s3.enabled=true \
|
||||
--set 's3.createBuckets[0].name=data' \
|
||||
--set global.seaweedfs.enableSecurity=true \
|
||||
> /tmp/security-bucket-hook.yaml
|
||||
test "$(grep -cE '^[[:space:]]*- name: ca-cert$' /tmp/security-bucket-hook.yaml)" -eq 2
|
||||
test "$(grep -cE '^[[:space:]]*- name: client-cert$' /tmp/security-bucket-hook.yaml)" -eq 2
|
||||
grep -q 'mountPath: /usr/local/share/ca-certificates/ca/' /tmp/security-bucket-hook.yaml
|
||||
grep -q 'mountPath: /usr/local/share/ca-certificates/client/' /tmp/security-bucket-hook.yaml
|
||||
grep -q 'secretName: test-seaweedfs-ca-cert' /tmp/security-bucket-hook.yaml
|
||||
grep -q 'secretName: test-seaweedfs-client-cert' /tmp/security-bucket-hook.yaml
|
||||
echo "Secure bucket-creation hook mounts its CA and client certificate"
|
||||
|
||||
echo ""
|
||||
echo "=== Testing admin.allowInsecureBind satisfies the admin auth render guard ==="
|
||||
helm template test $CHART_DIR --set admin.enabled=true --set admin.allowInsecureBind=true \
|
||||
@@ -718,6 +770,168 @@ jobs:
|
||||
helm template test $CHART_DIR --set cosi.enabled=true > /tmp/cosi.yaml
|
||||
grep -q "seaweedfs-cosi" /tmp/cosi.yaml
|
||||
echo "COSI driver renders correctly"
|
||||
|
||||
echo ""
|
||||
echo "=== Testing configurable pod and container security contexts ==="
|
||||
helm template test $CHART_DIR > /tmp/security-context-defaults.yaml
|
||||
|
||||
security_context_args=()
|
||||
for component in master volume filer s3 sftp admin worker allInOne cosi; do
|
||||
security_context_args+=(
|
||||
--set "$component.podSecurityContext.enabled=true"
|
||||
--set "$component.podSecurityContext.seccompProfile.type=RuntimeDefault"
|
||||
--set "$component.containerSecurityContext.enabled=true"
|
||||
--set "$component.containerSecurityContext.privileged=false"
|
||||
--set "$component.containerSecurityContext.allowPrivilegeEscalation=false"
|
||||
--set "$component.containerSecurityContext.readOnlyRootFilesystem=true"
|
||||
--set "$component.containerSecurityContext.capabilities.drop[0]=ALL"
|
||||
--set "$component.containerSecurityContext.seccompProfile.type=RuntimeDefault"
|
||||
)
|
||||
done
|
||||
|
||||
helm template test $CHART_DIR \
|
||||
"${security_context_args[@]}" \
|
||||
--set s3.enabled=true \
|
||||
--set s3.createBuckets[0].name=test \
|
||||
--set sftp.enabled=true \
|
||||
--set admin.enabled=true \
|
||||
--set admin.secret.adminPassword=ci-admin-password \
|
||||
--set worker.enabled=true \
|
||||
--set volume.idx.type=hostPath \
|
||||
--set volume.idx.hostPathPrefix=/tmp \
|
||||
--set cosi.enabled=true \
|
||||
--set global.seaweedfs.tmpDir.sizeLimit=64Mi > /tmp/security-contexts.yaml
|
||||
helm template test $CHART_DIR \
|
||||
"${security_context_args[@]}" \
|
||||
--set allInOne.enabled=true \
|
||||
--set master.enabled=false \
|
||||
--set volume.enabled=false \
|
||||
--set filer.enabled=false \
|
||||
--set global.seaweedfs.tmpDir.sizeLimit=64Mi > /tmp/security-contexts-aio.yaml
|
||||
python3 - /tmp/security-context-defaults.yaml /tmp/security-contexts.yaml /tmp/security-contexts-aio.yaml <<'PYEOF'
|
||||
import sys
|
||||
|
||||
import yaml
|
||||
|
||||
errors = []
|
||||
workloads = 0
|
||||
containers = 0
|
||||
chart_managed_init_containers = 0
|
||||
components = set()
|
||||
|
||||
def validate_container(workload_name, container):
|
||||
context = container.get("securityContext", {})
|
||||
prefix = f"{workload_name}/{container['name']}"
|
||||
if "enabled" in context:
|
||||
errors.append(f"{prefix}: internal enabled flag leaked into container securityContext")
|
||||
if context.get("privileged") is not False:
|
||||
errors.append(f"{prefix}: privileged is not false")
|
||||
if context.get("allowPrivilegeEscalation") is not False:
|
||||
errors.append(f"{prefix}: allowPrivilegeEscalation is not false")
|
||||
if context.get("readOnlyRootFilesystem") is not True:
|
||||
errors.append(f"{prefix}: readOnlyRootFilesystem is not true")
|
||||
if context.get("capabilities", {}).get("drop") != ["ALL"]:
|
||||
errors.append(f"{prefix}: capabilities.drop is not exactly [ALL]")
|
||||
if context.get("seccompProfile", {}).get("type") != "RuntimeDefault":
|
||||
errors.append(f"{prefix}: seccompProfile is not RuntimeDefault")
|
||||
mounts = {mount["name"]: mount for mount in container.get("volumeMounts", [])}
|
||||
if mounts.get("seaweedfs-tmp", {}).get("mountPath") != "/tmp":
|
||||
errors.append(f"{prefix}: writable temporary volume is not mounted at /tmp")
|
||||
|
||||
with open(sys.argv[1]) as stream:
|
||||
default_documents = [document for document in yaml.safe_load_all(stream) if document]
|
||||
for document in default_documents:
|
||||
if document.get("kind") not in ("Deployment", "StatefulSet", "Job"):
|
||||
continue
|
||||
name = document["metadata"]["name"]
|
||||
pod = document["spec"]["template"]["spec"]
|
||||
if "securityContext" in pod:
|
||||
errors.append(f"{name}: pod securityContext should be absent by default")
|
||||
if any(volume["name"] == "seaweedfs-tmp" for volume in pod.get("volumes", [])):
|
||||
errors.append(f"{name}: writable /tmp volume should be absent by default")
|
||||
for container in pod.get("containers", []):
|
||||
if "securityContext" in container:
|
||||
errors.append(
|
||||
f"{name}/{container['name']}: container securityContext "
|
||||
f"should be absent by default"
|
||||
)
|
||||
if any(
|
||||
mount["name"] == "seaweedfs-tmp"
|
||||
for mount in container.get("volumeMounts", [])
|
||||
):
|
||||
errors.append(
|
||||
f"{name}/{container['name']}: writable /tmp mount "
|
||||
f"should be absent by default"
|
||||
)
|
||||
|
||||
for path in sys.argv[2:]:
|
||||
with open(path) as stream:
|
||||
documents = [document for document in yaml.safe_load_all(stream) if document]
|
||||
for document in documents:
|
||||
if document.get("kind") not in ("Deployment", "StatefulSet", "Job"):
|
||||
continue
|
||||
workloads += 1
|
||||
name = document["metadata"]["name"]
|
||||
pod = document["spec"]["template"]["spec"]
|
||||
volumes = {volume["name"]: volume for volume in pod.get("volumes", [])}
|
||||
temporary = volumes.get("seaweedfs-tmp", {}).get("emptyDir")
|
||||
if temporary is None:
|
||||
errors.append(f"{name}: writable /tmp emptyDir is missing")
|
||||
elif temporary.get("sizeLimit") != "64Mi":
|
||||
errors.append(f"{name}: temporary volume sizeLimit is not 64Mi")
|
||||
component = document["spec"]["template"]["metadata"]["labels"].get("app.kubernetes.io/component")
|
||||
if component:
|
||||
components.add(component)
|
||||
else:
|
||||
errors.append(f"{name}: app.kubernetes.io/component label is missing")
|
||||
pod_context = pod.get("securityContext", {})
|
||||
if "enabled" in pod_context:
|
||||
errors.append(f"{name}: internal enabled flag leaked into pod securityContext")
|
||||
if pod_context.get("seccompProfile", {}).get("type") != "RuntimeDefault":
|
||||
errors.append(f"{name}: pod seccompProfile is not RuntimeDefault")
|
||||
for container in pod.get("containers", []):
|
||||
containers += 1
|
||||
validate_container(name, container)
|
||||
for container in pod.get("initContainers", []):
|
||||
if container["name"] != "seaweedfs-vol-move-idx":
|
||||
continue
|
||||
chart_managed_init_containers += 1
|
||||
validate_container(name, container)
|
||||
|
||||
expected_components = {
|
||||
"master", "volume", "filer", "s3", "sftp", "admin", "worker",
|
||||
"objectstorage-provisioner", "bucket-hook", "seaweedfs-all-in-one",
|
||||
}
|
||||
if components != expected_components:
|
||||
errors.append(
|
||||
f"security context workload coverage is incomplete: "
|
||||
f"expected {sorted(expected_components)}, got {sorted(components)}"
|
||||
)
|
||||
if workloads != 10 or containers != 12:
|
||||
errors.append(
|
||||
f"expected 10 workloads and 12 containers, got "
|
||||
f"{workloads} workloads and {containers} containers"
|
||||
)
|
||||
if chart_managed_init_containers != 1:
|
||||
errors.append(
|
||||
f"expected one chart-managed seaweedfs-vol-move-idx init container, "
|
||||
f"got {chart_managed_init_containers}"
|
||||
)
|
||||
|
||||
if errors:
|
||||
print("\n".join(f"FAIL: {error}" for error in errors), file=sys.stderr)
|
||||
sys.exit(1)
|
||||
print(f"Validated security contexts on {workloads} workloads and {containers} containers")
|
||||
PYEOF
|
||||
|
||||
# The resize hook depends on lookup finding a live StatefulSet and
|
||||
# therefore cannot render during helm template. Keep static coverage
|
||||
# for both security-context blocks.
|
||||
grep -Fqx ' securityContext: {{- omit .Values.volume.podSecurityContext "enabled" | toYaml | nindent 8 }}' \
|
||||
"$CHART_DIR/templates/volume/volume-resize-hook.yaml"
|
||||
grep -Fqx ' securityContext: {{- omit .Values.volume.containerSecurityContext "enabled" | toYaml | nindent 12 }}' \
|
||||
"$CHART_DIR/templates/volume/volume-resize-hook.yaml"
|
||||
echo "Volume resize hook security contexts are covered"
|
||||
|
||||
echo ""
|
||||
echo "=== Testing long release name: service names match DNS references ==="
|
||||
@@ -1605,6 +1819,78 @@ jobs:
|
||||
- name: Create kind cluster
|
||||
uses: helm/kind-action@v1.15.0
|
||||
|
||||
- name: Verify volume resize hook security contexts
|
||||
run: |
|
||||
set -e
|
||||
CHART_DIR="k8s/charts/seaweedfs"
|
||||
NS="resize-hook-security"
|
||||
kubectl create namespace "$NS"
|
||||
kubectl apply -n "$NS" -f - <<'EOF'
|
||||
apiVersion: v1
|
||||
kind: PersistentVolumeClaim
|
||||
metadata:
|
||||
name: data1-resize-seaweedfs-volume-0
|
||||
spec:
|
||||
accessModes:
|
||||
- ReadWriteOnce
|
||||
resources:
|
||||
requests:
|
||||
storage: 1Gi
|
||||
EOF
|
||||
|
||||
helm template resize "$CHART_DIR" -n "$NS" --dry-run=server \
|
||||
--set volume.dataDirs[0].name=data1 \
|
||||
--set volume.dataDirs[0].type=persistentVolumeClaim \
|
||||
--set volume.dataDirs[0].size=2Gi \
|
||||
--set volume.podSecurityContext.enabled=true \
|
||||
--set volume.podSecurityContext.seccompProfile.type=RuntimeDefault \
|
||||
--set volume.containerSecurityContext.enabled=true \
|
||||
--set volume.containerSecurityContext.privileged=false \
|
||||
--set volume.containerSecurityContext.allowPrivilegeEscalation=false \
|
||||
--set volume.containerSecurityContext.readOnlyRootFilesystem=true \
|
||||
--set volume.containerSecurityContext.capabilities.drop[0]=ALL \
|
||||
--set volume.containerSecurityContext.seccompProfile.type=RuntimeDefault \
|
||||
--set global.seaweedfs.tmpDir.sizeLimit=64Mi \
|
||||
> /tmp/security-context-resize-hook.yaml
|
||||
|
||||
python3 - /tmp/security-context-resize-hook.yaml <<'PYEOF'
|
||||
import sys
|
||||
|
||||
import yaml
|
||||
|
||||
with open(sys.argv[1]) as stream:
|
||||
documents = [document for document in yaml.safe_load_all(stream) if document]
|
||||
|
||||
jobs = [
|
||||
document for document in documents
|
||||
if document.get("kind") == "Job"
|
||||
and document["metadata"]["name"].endswith("-volume-resize-hook")
|
||||
]
|
||||
if len(jobs) != 1:
|
||||
raise AssertionError(f"expected one volume resize hook Job, got {len(jobs)}")
|
||||
|
||||
pod = jobs[0]["spec"]["template"]["spec"]
|
||||
assert pod["securityContext"] == {
|
||||
"seccompProfile": {"type": "RuntimeDefault"},
|
||||
}
|
||||
assert len(pod["containers"]) == 1
|
||||
assert pod["containers"][0]["securityContext"] == {
|
||||
"allowPrivilegeEscalation": False,
|
||||
"capabilities": {"drop": ["ALL"]},
|
||||
"privileged": False,
|
||||
"readOnlyRootFilesystem": True,
|
||||
"seccompProfile": {"type": "RuntimeDefault"},
|
||||
}
|
||||
assert pod["containers"][0]["volumeMounts"] == [
|
||||
{"mountPath": "/tmp", "name": "seaweedfs-tmp"},
|
||||
]
|
||||
assert pod["volumes"] == [
|
||||
{"emptyDir": {"sizeLimit": "64Mi"}, "name": "seaweedfs-tmp"},
|
||||
]
|
||||
PYEOF
|
||||
kubectl delete namespace "$NS"
|
||||
echo "Volume resize hook security contexts render correctly"
|
||||
|
||||
- name: Run chart-testing (install)
|
||||
run: |
|
||||
ct install --target-branch ${{ github.event.repository.default_branch }} --all --chart-dirs k8s/charts \
|
||||
@@ -1655,6 +1941,16 @@ jobs:
|
||||
kubectl delete namespace "$NS"
|
||||
echo "SFTP host key lifecycle tests passed"
|
||||
|
||||
- name: Verify JWT signing key persistence across upgrades
|
||||
run: |
|
||||
# chart-testing puts its pip-less venv first on PATH; use setup-python.
|
||||
PYTHON="$pythonLocation/bin/python3"
|
||||
"$PYTHON" -m pip install tomli==2.2.1
|
||||
CONTEXT=$(kubectl config current-context)
|
||||
"$PYTHON" .github/scripts/helm_jwt_keys.py --context "$CONTEXT"
|
||||
"$PYTHON" .github/scripts/helm_jwt_keys.py --helm "$HELM_LEGACY" \
|
||||
--reference-helm helm --context "$CONTEXT"
|
||||
|
||||
- name: Verify install into a default-deny namespace
|
||||
run: |
|
||||
set -e
|
||||
|
||||
@@ -139,7 +139,8 @@ jobs:
|
||||
unit-tests:
|
||||
name: Go Unit Tests (Implicit Directory)
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
# Leave time for setup and the focused run before the nine-minute suite.
|
||||
timeout-minutes: 20
|
||||
|
||||
steps:
|
||||
- name: Checkout code
|
||||
@@ -159,5 +160,5 @@ jobs:
|
||||
- name: Run all S3 API tests
|
||||
run: |
|
||||
cd weed/s3api
|
||||
go test -v -timeout 5m
|
||||
go test -v -timeout 9m
|
||||
|
||||
|
||||
@@ -82,6 +82,8 @@ AWS_ACCESS_KEY_ID=admin AWS_SECRET_ACCESS_KEY=secret \
|
||||
|
||||
The same process also runs the master, a volume server, the filer, WebDAV, the Iceberg REST catalog, and the Admin UI. Add `S3_TABLE_BUCKET=warehouse` to also create an Iceberg table bucket, or `warehouse:LANCE` for a Lance one. Drop the AWS keys to run without authentication for development.
|
||||
|
||||
Without Admin authentication or mTLS, `weed mini` binds the Admin UI/API and its worker gRPC control plane to loopback rather than `-ip.bind`. Set `WEED_ADMIN_PASSWORD` or configure `https.admin` mTLS to keep the network bind. Remote workers must opt in with `-admin.worker.ip=<address>` and should configure `grpc.admin` mTLS. `-admin.allowInsecureBind` restores the legacy unauthenticated network bind and should only be used on an isolated network.
|
||||
|
||||
> macOS: if the binary is quarantined, run `xattr -d com.apple.quarantine ./weed` first.
|
||||
|
||||
`weed mini` is auto-tuned for one node and is fine for single-node production, such as an S3 gateway that issues presigned URLs. See [Quick Start with weed mini][WeedMini].
|
||||
|
||||
@@ -60,7 +60,7 @@ require (
|
||||
github.com/pquerna/cachecontrol v0.2.0
|
||||
github.com/prometheus/client_golang v1.24.1
|
||||
github.com/prometheus/client_model v0.6.3
|
||||
github.com/prometheus/common v0.70.1 // indirect
|
||||
github.com/prometheus/common v0.71.0 // indirect
|
||||
github.com/prometheus/procfs v0.22.0
|
||||
github.com/rcrowley/go-metrics v0.0.0-20201227073835-cf1acfcdf475 // indirect
|
||||
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec // indirect
|
||||
@@ -73,7 +73,7 @@ require (
|
||||
github.com/stretchr/testify v1.12.1
|
||||
github.com/stvp/tempredis v0.0.0-20181119212430-b82af8480203
|
||||
github.com/syndtr/goleveldb v1.0.1-0.20190318030020-c3a204f8e965
|
||||
github.com/tidwall/gjson v1.18.0
|
||||
github.com/tidwall/gjson v1.19.0
|
||||
github.com/tidwall/match v1.2.0
|
||||
github.com/tidwall/pretty v1.2.0 // indirect
|
||||
github.com/tsuna/gohbase v0.0.0-20201125011725-348991136365
|
||||
@@ -118,14 +118,14 @@ require (
|
||||
github.com/ThreeDotsLabs/watermill v1.5.2
|
||||
github.com/a-h/templ v0.3.1020
|
||||
github.com/apache/cassandra-gocql-driver/v2 v2.1.2
|
||||
github.com/apache/iceberg-go v0.6.1-0.20260817192109-c2105090c9e2
|
||||
github.com/apache/iceberg-go v0.7.0
|
||||
github.com/apple/foundationdb/bindings/go v0.0.0-20250911184653-27f7192f47c3
|
||||
github.com/arangodb/go-driver v1.6.9
|
||||
github.com/armon/go-metrics v0.4.1
|
||||
github.com/aws/aws-sdk-go-v2 v1.47.0
|
||||
github.com/aws/aws-sdk-go-v2/config v1.32.35
|
||||
github.com/aws/aws-sdk-go-v2/config v1.33.1
|
||||
github.com/aws/aws-sdk-go-v2/credentials v1.20.4
|
||||
github.com/aws/aws-sdk-go-v2/service/s3 v1.107.3
|
||||
github.com/aws/aws-sdk-go-v2/service/s3 v1.109.1
|
||||
github.com/cespare/xxhash/v2 v2.3.0
|
||||
github.com/cognusion/imaging v1.0.4
|
||||
github.com/fluent/fluent-logger-golang v1.10.1
|
||||
@@ -134,8 +134,8 @@ require (
|
||||
github.com/golang-jwt/jwt/v5 v5.3.1
|
||||
github.com/google/flatbuffers/go v0.0.0-20230108230133-3b8644d32c50
|
||||
github.com/hashicorp/golang-lru/v2 v2.0.7
|
||||
github.com/hashicorp/raft v1.7.3
|
||||
github.com/hashicorp/raft-boltdb/v2 v2.3.1
|
||||
github.com/hashicorp/raft v1.8.0
|
||||
github.com/hashicorp/raft-boltdb/v2 v2.4.2
|
||||
github.com/hashicorp/vault/api v1.23.0
|
||||
github.com/jhump/protoreflect v1.18.0
|
||||
github.com/linkedin/goavro/v2 v2.15.0
|
||||
@@ -143,7 +143,7 @@ require (
|
||||
github.com/orcaman/concurrent-map/v2 v2.0.1
|
||||
github.com/parquet-go/parquet-go v0.32.0
|
||||
github.com/pkg/sftp v1.13.11
|
||||
github.com/rabbitmq/amqp091-go v1.14.0
|
||||
github.com/rabbitmq/amqp091-go v1.15.0
|
||||
github.com/rclone/rclone v1.75.1
|
||||
github.com/rdleal/intervalst v1.5.0
|
||||
github.com/redis/go-redis/v9 v9.22.0
|
||||
@@ -168,9 +168,6 @@ require (
|
||||
require github.com/k0kubun/colorstring v0.0.0-20150214042306-9440f1994b88 // indirect
|
||||
|
||||
require (
|
||||
atomicgo.dev/cursor v0.2.0 // indirect
|
||||
atomicgo.dev/keyboard v0.2.9 // indirect
|
||||
atomicgo.dev/schedule v0.1.0 // indirect
|
||||
cloud.google.com/go/longrunning v1.2.0 // indirect
|
||||
cloud.google.com/go/pubsub/v2 v2.6.0 // indirect
|
||||
dario.cat/mergo v1.0.2 // indirect
|
||||
@@ -178,12 +175,12 @@ require (
|
||||
github.com/Azure/go-ansiterm v0.0.0-20250102033503-faa5f7b0171c // indirect
|
||||
github.com/FilenCloudDienste/filen-sdk-go v0.0.39 // indirect
|
||||
github.com/ProtonMail/gopenpgp/v3 v3.4.1 // indirect
|
||||
github.com/RoaringBitmap/roaring/v2 v2.24.0 // indirect
|
||||
github.com/RoaringBitmap/roaring/v2 v2.26.0 // indirect
|
||||
github.com/a1ex3/zstd-seekable-format-go/pkg v0.10.0 // indirect
|
||||
github.com/adrg/xdg v0.5.3 // indirect
|
||||
github.com/anchore/go-lzo v0.1.1 // indirect
|
||||
github.com/antlr4-go/antlr/v4 v4.13.1 // indirect
|
||||
github.com/apache/arrow-go/v18 v18.7.0 // indirect
|
||||
github.com/apache/arrow-go/v18 v18.8.0 // indirect
|
||||
github.com/apache/thrift v0.24.0 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/signin v1.10.0 // indirect
|
||||
github.com/bahlo/generic-list-go v0.2.0 // indirect
|
||||
@@ -200,7 +197,6 @@ require (
|
||||
github.com/cockroachdb/logtags v0.0.0-20241215232642-bb51bb14a506 // indirect
|
||||
github.com/cockroachdb/redact v1.1.5 // indirect
|
||||
github.com/cockroachdb/version v0.0.0-20250314144055-3860cd14adf2 // indirect
|
||||
github.com/containerd/console v1.0.5 // indirect
|
||||
github.com/containerd/errdefs v1.0.0 // indirect
|
||||
github.com/containerd/errdefs/pkg v0.3.0 // indirect
|
||||
github.com/containerd/log v0.1.0 // indirect
|
||||
@@ -219,7 +215,6 @@ require (
|
||||
github.com/goccy/go-yaml v1.18.0 // indirect
|
||||
github.com/golang/geo v0.0.0-20210211234256-740aa86cb551 // indirect
|
||||
github.com/google/go-cmp v0.7.0 // indirect
|
||||
github.com/gookit/color v1.6.0 // indirect
|
||||
github.com/gopherjs/gopherjs v1.17.2 // indirect
|
||||
github.com/grpc-ecosystem/grpc-gateway v1.16.0 // indirect
|
||||
github.com/hashicorp/go-rootcerts v1.0.2 // indirect
|
||||
@@ -237,7 +232,6 @@ require (
|
||||
github.com/kr/pretty v0.3.1 // indirect
|
||||
github.com/kr/text v0.2.0 // indirect
|
||||
github.com/lib/pq v1.12.0 // indirect
|
||||
github.com/lithammer/fuzzysearch v1.1.8 // indirect
|
||||
github.com/lithammer/shortuuid/v3 v3.0.7 // indirect
|
||||
github.com/lpar/calendar v0.2.0 // indirect
|
||||
github.com/magiconair/properties v1.8.10 // indirect
|
||||
@@ -260,7 +254,6 @@ require (
|
||||
github.com/petermattis/goid v0.0.0-20260113132338-7c7de50cc741 // indirect
|
||||
github.com/pierrre/geohash v1.0.0 // indirect
|
||||
github.com/pquerna/otp v1.5.0 // indirect
|
||||
github.com/pterm/pterm v0.12.83 // indirect
|
||||
github.com/quic-go/qpack v0.6.0 // indirect
|
||||
github.com/rclone/Proton-API-Bridge v1.0.5 // indirect
|
||||
github.com/rclone/go-proton-api v1.0.4 // indirect
|
||||
@@ -281,7 +274,6 @@ require (
|
||||
github.com/wk8/go-ordered-map/v2 v2.1.8 // indirect
|
||||
github.com/xeipuuv/gojsonpointer v0.0.0-20190905194746-02993c407bfb // indirect
|
||||
github.com/xeipuuv/gojsonreference v0.0.0-20180127040603-bd5ef7bd5415 // indirect
|
||||
github.com/xo/terminfo v0.0.0-20220910002029-abceb7e1c41e // indirect
|
||||
github.com/zeebo/xxh3 v1.1.0 // indirect
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace v1.45.0 // indirect
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracegrpc v1.45.0 // indirect
|
||||
@@ -322,20 +314,20 @@ require (
|
||||
github.com/ProtonMail/go-srp v0.0.7 // indirect
|
||||
github.com/PuerkitoBio/goquery v1.12.0 // indirect
|
||||
github.com/abbot/go-http-auth v0.4.0 // indirect
|
||||
github.com/andybalholm/brotli v1.2.2 // indirect
|
||||
github.com/andybalholm/brotli v1.2.3 // indirect
|
||||
github.com/andybalholm/cascadia v1.3.4 // indirect
|
||||
github.com/appscode/go-querystring v0.0.0-20170504095604-0126cfb3f1dc // indirect
|
||||
github.com/arangodb/go-velocypack v0.0.0-20200318135517-5af53c29c67e // indirect
|
||||
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.18 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.20 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.20.0 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/feature/s3/manager v1.22.34 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.3 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.3 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.3 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.19 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.31 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.11.1 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.3 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.39 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.20.1 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/sns v1.39.14 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/sqs v1.42.24 // indirect
|
||||
github.com/aws/aws-sdk-go-v2/service/sso v1.38.0 // indirect
|
||||
@@ -365,7 +357,7 @@ require (
|
||||
github.com/emersion/go-vcard v0.0.0-20260618161152-d854b7e0e2d3 // indirect
|
||||
github.com/envoyproxy/go-control-plane/envoy v1.39.1-0.20260819172001-e6e3fd93e4be // indirect
|
||||
github.com/envoyproxy/protoc-gen-validate v1.3.3 // indirect
|
||||
github.com/fatih/color v1.18.0 // indirect
|
||||
github.com/fatih/color v1.19.0 // indirect
|
||||
github.com/felixge/httpsnoop v1.1.0 // indirect
|
||||
github.com/flynn/noise v1.1.0 // indirect
|
||||
github.com/gabriel-vasile/mimetype v1.4.13 // indirect
|
||||
@@ -397,10 +389,10 @@ require (
|
||||
github.com/hashicorp/go-cleanhttp v0.5.2 // indirect
|
||||
github.com/hashicorp/go-hclog v1.6.3 // indirect
|
||||
github.com/hashicorp/go-immutable-radix v1.3.1 // indirect
|
||||
github.com/hashicorp/go-metrics v0.5.4 // indirect
|
||||
github.com/hashicorp/go-msgpack/v2 v2.1.2 // indirect
|
||||
github.com/hashicorp/go-metrics v0.7.0 // indirect
|
||||
github.com/hashicorp/go-msgpack/v2 v2.1.5 // indirect
|
||||
github.com/hashicorp/go-retryablehttp v0.7.8 // indirect
|
||||
github.com/hashicorp/golang-lru v0.6.0 // indirect
|
||||
github.com/hashicorp/golang-lru v1.0.2 // indirect
|
||||
github.com/jcmturner/aescts/v2 v2.0.0 // indirect
|
||||
github.com/jcmturner/dnsutils/v2 v2.0.0 // indirect
|
||||
github.com/jcmturner/goidentity/v6 v6.0.1 // indirect
|
||||
@@ -492,15 +484,15 @@ require (
|
||||
go.opentelemetry.io/contrib/detectors/gcp v1.45.0 // indirect
|
||||
go.opentelemetry.io/contrib/instrumentation/google.golang.org/grpc/otelgrpc v0.69.0 // indirect
|
||||
go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.70.0 // indirect
|
||||
go.opentelemetry.io/otel v1.45.0 // indirect
|
||||
go.opentelemetry.io/otel/metric v1.45.0 // indirect
|
||||
go.opentelemetry.io/otel/sdk v1.45.0 // indirect
|
||||
go.opentelemetry.io/otel/sdk/metric v1.45.0 // indirect
|
||||
go.opentelemetry.io/otel/trace v1.45.0 // indirect
|
||||
go.opentelemetry.io/otel v1.46.0 // indirect
|
||||
go.opentelemetry.io/otel/metric v1.46.0 // indirect
|
||||
go.opentelemetry.io/otel/sdk v1.46.0 // indirect
|
||||
go.opentelemetry.io/otel/sdk/metric v1.46.0 // indirect
|
||||
go.opentelemetry.io/otel/trace v1.46.0 // indirect
|
||||
go.uber.org/multierr v1.11.0 // indirect
|
||||
go.uber.org/zap v1.27.1 // indirect
|
||||
golang.org/x/term v0.46.0
|
||||
golang.org/x/time v0.15.0
|
||||
golang.org/x/time v0.16.0
|
||||
google.golang.org/genproto/googleapis/api v0.0.0-20260817212433-ac3dfec99bb1 // indirect
|
||||
google.golang.org/genproto/googleapis/rpc v0.0.0-20260819154853-08b0e4226688 // indirect
|
||||
gopkg.in/natefinch/lumberjack.v2 v2.2.1 // indirect
|
||||
|
||||
@@ -1,11 +1,3 @@
|
||||
atomicgo.dev/assert v0.0.2 h1:FiKeMiZSgRrZsPo9qn/7vmr7mCsh5SZyXY4YGYiYwrg=
|
||||
atomicgo.dev/assert v0.0.2/go.mod h1:ut4NcI3QDdJtlmAxQULOmA13Gz6e2DWbSAS8RUOmNYQ=
|
||||
atomicgo.dev/cursor v0.2.0 h1:H6XN5alUJ52FZZUkI7AlJbUc1aW38GWZalpYRPpoPOw=
|
||||
atomicgo.dev/cursor v0.2.0/go.mod h1:Lr4ZJB3U7DfPPOkbH7/6TOtJ4vFGHlgj1nc+n900IpU=
|
||||
atomicgo.dev/keyboard v0.2.9 h1:tOsIid3nlPLZ3lwgG8KZMp/SFmr7P0ssEN5JUsm78K8=
|
||||
atomicgo.dev/keyboard v0.2.9/go.mod h1:BC4w9g00XkxH/f1HXhW2sXmJFOCWbKn9xrOunSFtExQ=
|
||||
atomicgo.dev/schedule v0.1.0 h1:nTthAbhZS5YZmgYbb2+DH8uQIZcTlIrd4eYr3UQxEjs=
|
||||
atomicgo.dev/schedule v0.1.0/go.mod h1:xeUa3oAkiuHYh8bKiQBRojqAMq3PXXbJujjb0hw8pEU=
|
||||
cel.dev/expr v0.25.3 h1:A2jO8jwOugrrovveCWfj0KEZOfqiLgAcwjpHPhzIGw0=
|
||||
cel.dev/expr v0.25.3/go.mod h1:hrXvqGP6G6gyx8UAHSHJ5RGk//1Oj5nXQ2NI02Nrsg4=
|
||||
cloud.google.com/go v0.26.0/go.mod h1:aQUYkXzVsufM+DwF1aE+0xfcU+56JwCaLick0ClmMTw=
|
||||
@@ -606,15 +598,6 @@ github.com/IBM/go-sdk-core/v5 v5.23.1/go.mod h1:yO+OQpByKDLTvpEcsFFexgzpeR8eRfCF
|
||||
github.com/Jille/raft-grpc-transport v1.6.1 h1:gN3sjapb+fVbiebS7AfQQgbV2ecTOI7ur7NPPC7Mhoc=
|
||||
github.com/Jille/raft-grpc-transport v1.6.1/go.mod h1:HbOjEdu/yzCJ/mjTF6wEOJNbAUpHfU2UOA2hVD4CNFg=
|
||||
github.com/JohnCGriffin/overflow v0.0.0-20211019200055-46fa312c352c/go.mod h1:X0CRv0ky0k6m906ixxpzmDRLvX58TFUKS2eePweuyxk=
|
||||
github.com/MarvinJWendt/testza v0.1.0/go.mod h1:7AxNvlfeHP7Z/hDQ5JtE3OKYT3XFUeLCDE2DQninSqs=
|
||||
github.com/MarvinJWendt/testza v0.2.1/go.mod h1:God7bhG8n6uQxwdScay+gjm9/LnO4D3kkcZX4hv9Rp8=
|
||||
github.com/MarvinJWendt/testza v0.2.8/go.mod h1:nwIcjmr0Zz+Rcwfh3/4UhBp7ePKVhuBExvZqnKYWlII=
|
||||
github.com/MarvinJWendt/testza v0.2.10/go.mod h1:pd+VWsoGUiFtq+hRKSU1Bktnn+DMCSrDrXDpX2bG66k=
|
||||
github.com/MarvinJWendt/testza v0.2.12/go.mod h1:JOIegYyV7rX+7VZ9r77L/eH6CfJHHzXjB69adAhzZkI=
|
||||
github.com/MarvinJWendt/testza v0.3.0/go.mod h1:eFcL4I0idjtIx8P9C6KkAuLgATNKpX4/2oUqKc6bF2c=
|
||||
github.com/MarvinJWendt/testza v0.4.2/go.mod h1:mSdhXiKH8sg/gQehJ63bINcCKp7RtYewEjXsvsVUPbE=
|
||||
github.com/MarvinJWendt/testza v0.5.2 h1:53KDo64C1z/h/d/stCYCPY69bt/OSwjq5KpFNwi+zB4=
|
||||
github.com/MarvinJWendt/testza v0.5.2/go.mod h1:xu53QFE5sCdjtMCKk8YMQ2MnymimEctc4n3EjyIYvEY=
|
||||
github.com/Masterminds/semver/v3 v3.2.0 h1:3MEsd0SM6jqZojhjLWWeBY+Kcjy9i6MQAeY7YgDP83g=
|
||||
github.com/Masterminds/semver/v3 v3.2.0/go.mod h1:qvl/7zhW3nngYb5+80sSMF+FG2BjYrf8m9wsX0PNOMQ=
|
||||
github.com/Max-Sum/base32768 v0.0.0-20230304063302-18e6ce5945fd h1:nzE1YQBdx1bq9IlZinHa+HVffy+NmVRoKr+wHN8fpLE=
|
||||
@@ -637,8 +620,8 @@ github.com/ProtonMail/gopenpgp/v3 v3.4.1 h1:K7uUhSHSJxORZ+RuHpilTT6S4MA2whCRlXNw
|
||||
github.com/ProtonMail/gopenpgp/v3 v3.4.1/go.mod h1:bGdV9f6edhmd581wzXsQCTKdH8bXBbyhkgDKPjwPc6U=
|
||||
github.com/PuerkitoBio/goquery v1.12.0 h1:pAcL4g3WRXekcB9AU/y1mbKez2dbY2AajVhtkO8RIBo=
|
||||
github.com/PuerkitoBio/goquery v1.12.0/go.mod h1:802ej+gV2y7bbIhOIoPY5sT183ZW0YFofScC4q/hIpQ=
|
||||
github.com/RoaringBitmap/roaring/v2 v2.24.0 h1:zQkkBZtG3WRP4j+P3A5DO221SvL1Br88TJkhyqEQRZo=
|
||||
github.com/RoaringBitmap/roaring/v2 v2.24.0/go.mod h1:SfT3of9nYh3vis1dIbCj4Yw6KQGujTN+f345nrN/0JA=
|
||||
github.com/RoaringBitmap/roaring/v2 v2.26.0 h1:K30ZxF4vZcIKvJsbmgfiep2K64f+dILJqkYGoj4xnwU=
|
||||
github.com/RoaringBitmap/roaring/v2 v2.26.0/go.mod h1:BZufmFbox589n3j5eOmyTaLSGXbRLc2LmQvjKjzSEGU=
|
||||
github.com/Sereal/Sereal/Go/sereal v0.0.0-20231009093132-b9187f1a92c6/go.mod h1:JwrycNnC8+sZPDyzM3MQ86LvaGzSpfxg885KOOwFRW4=
|
||||
github.com/Shopify/sarama v1.38.1 h1:lqqPUPQZ7zPqYlWpTh+LQ9bhYNu2xJL6k1SJN4WVe2A=
|
||||
github.com/Shopify/sarama v1.38.1/go.mod h1:iwv9a67Ha8VNa+TifujYoWGxWnu2kNVAQdSdZ4X2o5g=
|
||||
@@ -672,14 +655,13 @@ github.com/alecthomas/template v0.0.0-20160405071501-a0175ee3bccc/go.mod h1:LOuy
|
||||
github.com/alecthomas/template v0.0.0-20190718012654-fb15b899a751/go.mod h1:LOuyumcjzFXgccqObfd/Ljyb9UuFJ6TxHnclSeseNhc=
|
||||
github.com/alecthomas/units v0.0.0-20151022065526-2efee857e7cf/go.mod h1:ybxpYRFXyAe+OPACYpWeL0wqObRcbAqCMya13uyzqw0=
|
||||
github.com/alecthomas/units v0.0.0-20190717042225-c3de453c63f4/go.mod h1:ybxpYRFXyAe+OPACYpWeL0wqObRcbAqCMya13uyzqw0=
|
||||
github.com/alecthomas/units v0.0.0-20190924025748-f65c72e2690d/go.mod h1:rBZYJk541a8SKzHPHnH3zbiI+7dagKZ0cgpgrD7Fyho=
|
||||
github.com/alexbrainman/sspi v0.0.0-20250919150558-7d374ff0d59e h1:4dAU9FXIyQktpoUAgOJK3OTFc/xug0PCXYCqU0FgDKI=
|
||||
github.com/alexbrainman/sspi v0.0.0-20250919150558-7d374ff0d59e/go.mod h1:cEWa1LVoE5KvSD9ONXsZrj0z6KqySlCCNKHlLzbqAt4=
|
||||
github.com/anchore/go-lzo v0.1.1 h1:IwL/fvkdtlIrYIXck6WxZ3nb8WjjHziYYmGxlooyOnM=
|
||||
github.com/anchore/go-lzo v0.1.1/go.mod h1:3kLx0bve2oN1iDwgM1U5zGku1Tfbdb0No5qp1eL1fIk=
|
||||
github.com/andybalholm/brotli v1.0.4/go.mod h1:fO7iG3H7G2nSZ7m0zPUDn85XEX2GTukHGRSepvi9Eig=
|
||||
github.com/andybalholm/brotli v1.2.2 h1:HzTuoo2ErYQqf5qvcJInB8uvqSVxRttzkFexPWtnceM=
|
||||
github.com/andybalholm/brotli v1.2.2/go.mod h1:rzTDkvFWvIrjDXZHkuS16NPggd91W3kUSvPlQ1pLaKY=
|
||||
github.com/andybalholm/brotli v1.2.3 h1:8H1qwOkl2LPfjf3YezB90JnCliZb6SInJ/OJkEbA5NQ=
|
||||
github.com/andybalholm/brotli v1.2.3/go.mod h1:rzTDkvFWvIrjDXZHkuS16NPggd91W3kUSvPlQ1pLaKY=
|
||||
github.com/andybalholm/cascadia v1.3.4 h1:vM2lgh0Vru9Vwyfm4cQqWP2HHMW0u0+2PAW7Q38Qufg=
|
||||
github.com/andybalholm/cascadia v1.3.4/go.mod h1:BLRmbRjpEtNKieZOCCvYj4RqN+KRA41GBe/5O+G93kM=
|
||||
github.com/antihax/optional v1.0.0/go.mod h1:uupD/76wgC+ih3iEmQUL+0Ugr19nfwCT1kdvxnR2qWY=
|
||||
@@ -687,13 +669,13 @@ github.com/antithesishq/antithesis-sdk-go v0.6.0-default-no-op h1:kpBdlEPbRvff0m
|
||||
github.com/antithesishq/antithesis-sdk-go v0.6.0-default-no-op/go.mod h1:IUpT2DPAKh6i/YhSbt6Gl3v2yvUZjmKncl7U91fup7E=
|
||||
github.com/antlr4-go/antlr/v4 v4.13.1 h1:SqQKkuVZ+zWkMMNkjy5FZe5mr5WURWnlpmOuzYWrPrQ=
|
||||
github.com/antlr4-go/antlr/v4 v4.13.1/go.mod h1:GKmUxMtwp6ZgGwZSva4eWPC5mS6vUAmOABFgjdkM7Nw=
|
||||
github.com/apache/arrow-go/v18 v18.7.0 h1:Vw/i+cJyebUofT7JlqFpe65LrmwxULn166jjwStM4HY=
|
||||
github.com/apache/arrow-go/v18 v18.7.0/go.mod h1:PM6IigLJkdMwIpeHXnymo+xZ52f42a9EYiLtRel4p/A=
|
||||
github.com/apache/arrow-go/v18 v18.8.0 h1:BLOzbPv7bxMPgXPacAg6HQjnxupYsZzC4tf+FkqPU/M=
|
||||
github.com/apache/arrow-go/v18 v18.8.0/go.mod h1:uJCFfCwq0KsxCmsCfQg4ft+LsW+iHYzAXiSDh5ug/8U=
|
||||
github.com/apache/arrow/go/v10 v10.0.1/go.mod h1:YvhnlEePVnBS4+0z3fhPfUy7W1Ikj0Ih0vcRo/gZ1M0=
|
||||
github.com/apache/cassandra-gocql-driver/v2 v2.1.2 h1:lu/p0Db2av18enHJvWJQoChLssI0P+AR06STq4VdvCc=
|
||||
github.com/apache/cassandra-gocql-driver/v2 v2.1.2/go.mod h1:QH/asJjB3mHvY6Dot6ZKMMpTcOrWJ8i9GhsvG1g0PK4=
|
||||
github.com/apache/iceberg-go v0.6.1-0.20260817192109-c2105090c9e2 h1:xRULj4L2wrlAAPAYNruyuP17lz2zq2DGLgIFDgqiBjo=
|
||||
github.com/apache/iceberg-go v0.6.1-0.20260817192109-c2105090c9e2/go.mod h1:u6gs2aFRl7QOL0FqfHX6DOzvxOyev6mmq3XT+3VUS9M=
|
||||
github.com/apache/iceberg-go v0.7.0 h1:bTD6Pb4uM4sWcMfIx0/cV1haHMc84r+CvPieUqYzD4I=
|
||||
github.com/apache/iceberg-go v0.7.0/go.mod h1:oGz5MX3/m3GDc6acsrwbdHChj28jZH8MVDWqUrW6WQQ=
|
||||
github.com/apache/thrift v0.16.0/go.mod h1:PHK3hniurgQaNMZYaCLEqXKsYK8upmhPbmdP2FXSqgU=
|
||||
github.com/apache/thrift v0.24.0 h1:zy31L1a49QTNB2bG1BBfMXol3yJrTH975G3pPubQVLQ=
|
||||
github.com/apache/thrift v0.24.0/go.mod h1:zPt6WxgvTOM6hF92y8C+MkEM5LMxZuk4JcQOiU4Esvs=
|
||||
@@ -707,23 +689,22 @@ github.com/arangodb/go-velocypack v0.0.0-20200318135517-5af53c29c67e h1:Xg+hGrY2
|
||||
github.com/arangodb/go-velocypack v0.0.0-20200318135517-5af53c29c67e/go.mod h1:mq7Shfa/CaixoDxiyAAc5jZ6CVBAyPaNQCGS7mkj4Ho=
|
||||
github.com/armon/go-metrics v0.4.1 h1:hR91U9KYmb6bLBYLQjyM+3j+rcd/UhE+G78SFnF8gJA=
|
||||
github.com/armon/go-metrics v0.4.1/go.mod h1:E6amYzXo6aW1tqzoZGT755KkbgrJsSdpwZ+3JqfkOG4=
|
||||
github.com/atomicgo/cursor v0.0.1/go.mod h1:cBON2QmmrysudxNBFthvMtN32r3jxVRIvzkUiF/RuIk=
|
||||
github.com/aws/aws-sdk-go v1.55.8 h1:JRmEUbU52aJQZ2AjX4q4Wu7t4uZjOu71uyNmaWlUkJQ=
|
||||
github.com/aws/aws-sdk-go v1.55.8/go.mod h1:ZkViS9AqA6otK+JBBNH2++sx1sgxrPKcSzPPvQkUtXk=
|
||||
github.com/aws/aws-sdk-go-v2 v1.47.0 h1:0jsHallhJCeaU0Ko48c/3FK1ctOQ7NpzggxriJOQ8MQ=
|
||||
github.com/aws/aws-sdk-go-v2 v1.47.0/go.mod h1:bttEH6JqnUL8LepvDVfdrds/fZ5bCIxzpe3abyUrhDU=
|
||||
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.18 h1:LAfOuhAH331fmOjTQpAaOlH+Ftn7RzSDJ2VFwjdMMy4=
|
||||
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.18/go.mod h1:4e5xhuXHx1e4U9EthvbPP1r/DIMp5c2823OL8karzcM=
|
||||
github.com/aws/aws-sdk-go-v2/config v1.32.35 h1:UEzXuET8E42lxBPijuACu/tEK7v5lFPlk0Q+GT5WD9E=
|
||||
github.com/aws/aws-sdk-go-v2/config v1.32.35/go.mod h1:KaMtJpFa2JlL2BStjjHQVwQpzZEmw+ND/EgVrfFoo2g=
|
||||
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.20 h1:GPRlPwz40I2B2VrBEASOA3Bi77NyeqejNLkifosX0rs=
|
||||
github.com/aws/aws-sdk-go-v2/aws/protocol/eventstream v1.7.20/go.mod h1:g7PNzKcsOKWb4fkSRBA7BZVAS6Y8IcxzN+nRohhQ1Q8=
|
||||
github.com/aws/aws-sdk-go-v2/config v1.33.1 h1:bq9jze1hQ5YTCLoVxNnbp0T7rglrlOE7N9YsHqjGkEw=
|
||||
github.com/aws/aws-sdk-go-v2/config v1.33.1/go.mod h1:2A3HQwG4zaL5Tm80rc6RZj8LmWWv4WYT5v8raSz/L7A=
|
||||
github.com/aws/aws-sdk-go-v2/credentials v1.20.4 h1:hTvrJJseKbvw32kmiE0G+u/9ZqpqscjDrTigHIXP2qs=
|
||||
github.com/aws/aws-sdk-go-v2/credentials v1.20.4/go.mod h1:gWp9O1ZBWwpcIrgV+mVHk4gZUurAEDkgypu/OXOlIaw=
|
||||
github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.20.0 h1:AM4hHjww+PSFtt6E+UrBrPlZkWsePCLEt9AjkfQX+yM=
|
||||
github.com/aws/aws-sdk-go-v2/feature/ec2/imds v1.20.0/go.mod h1:3x/yXezeQjpOvBb4jEMxrS8SXvpdvJ5abv6l5c1gWM8=
|
||||
github.com/aws/aws-sdk-go-v2/feature/s3/manager v1.22.34 h1:Pn7OsMwBLbkZ6OnCxWHAjf0L/22H8cnhxZC0uPwtMtg=
|
||||
github.com/aws/aws-sdk-go-v2/feature/s3/manager v1.22.34/go.mod h1:eToXR/Gk1uqpn04eSmdgVXwfS0WvH8aG4eBFr8ygbpU=
|
||||
github.com/aws/aws-sdk-go-v2/feature/s3/transfermanager v0.3.11 h1:eBXB8KZgzQ8A9QB4iJS4aw/u6+4OY3i2hQXPABeAIOg=
|
||||
github.com/aws/aws-sdk-go-v2/feature/s3/transfermanager v0.3.11/go.mod h1:N9+5pG27Fy61GUL5YXVLXDTLmUudMrgwsuDbgBMNLxQ=
|
||||
github.com/aws/aws-sdk-go-v2/feature/s3/transfermanager v0.4.1 h1:I3mWvASaICc5c8vJ3ftYjroh7LT3jG0q/KtzSu5wW/s=
|
||||
github.com/aws/aws-sdk-go-v2/feature/s3/transfermanager v0.4.1/go.mod h1:VwSN8piv62OyyxYWtJzj6j7gECyLc0WwMo30L5NQZlE=
|
||||
github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.3 h1:Hp/VgjP0BysR3OgLlR057Vz2LcbbVnoWeJ+3qWiS/fY=
|
||||
github.com/aws/aws-sdk-go-v2/internal/configsources v1.5.3/go.mod h1:nwGV5qw7F1IZPgxCvA/ph8N2TAuz+BkRG/bXn808qMA=
|
||||
github.com/aws/aws-sdk-go-v2/internal/endpoints/v2 v2.8.3 h1:MUaM4f+kj1ZIBPZfUS8cxP1GKXXZtHJjAthy93AN7SM=
|
||||
@@ -732,14 +713,14 @@ github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.3 h1:fuSCw4Z2qfRCztMPO3GXJNSiEp6W
|
||||
github.com/aws/aws-sdk-go-v2/internal/v4a v1.5.3/go.mod h1:6SxcHheD1pPR5+kWm1wGvjlL/YqUsh267sAfEmN4K7A=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.19 h1:bAdDl/HkGCcGPoe25ToSHEw23VIxt6CT5fLcg111BKg=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/accept-encoding v1.13.19/go.mod h1:KaUzbLxv4CeSxh6ZCl9B4m7CuFenS8kUEaDs+f/DQr4=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.31 h1:uZOinZb+h7lZw8IYzP1z1IuEnueB76/EFkcf/fEW4Ag=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.9.31/go.mod h1:NRtwAM/p5VRt03TlEUs0pH3TeWamWdf4YyJpSrzPYLc=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.11.1 h1:s67hBfG5t9rn1NCvDuB4E3QIep3UFhHPtaIqFDjV3N8=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/checksum v1.11.1/go.mod h1:FpvjBMXtSNMLPmDJsWwcY5cRnqJlpS2y1R6n4pvzs4k=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.3 h1:bON1rJf67TSTDCKg816AAIE4xSTtoo9tl0XRkO72R+I=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/presigned-url v1.14.3/go.mod h1:c5BBpjJcQXpfeq9iASyVKA3T6vX6B6LEXY4mL/gklDY=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.39 h1:HLPAVrlLDaN2boN0xJx7MgaQDNEO3Q+c9L6kl/8m47Q=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.19.39/go.mod h1:Pg/dVfsNkm1hsIDK/gMvCKtmyNfNTV12mrgHqVE/6Oo=
|
||||
github.com/aws/aws-sdk-go-v2/service/s3 v1.107.3 h1:IKoCZqfWfZzSBi16QFQ+QcbQ3LRQ7QgB1S5tDAyPBQQ=
|
||||
github.com/aws/aws-sdk-go-v2/service/s3 v1.107.3/go.mod h1:RBpRcXiM4s2pOInVs32GsBonnje+fiAj4mcrStRmlCA=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.20.1 h1:ZMbtPZZQRca+3+XYQne9PBvRiYpHZlNJJOZfE9WNfT0=
|
||||
github.com/aws/aws-sdk-go-v2/service/internal/s3shared v1.20.1/go.mod h1:YAGWQdCYlVCoqrzvfv3RLxO6zKwti7gsAULOGWPLYv4=
|
||||
github.com/aws/aws-sdk-go-v2/service/s3 v1.109.1 h1:kVpzaDBzOdRtOftmiSpTdQbWVqRg0kONLXijktiwXnk=
|
||||
github.com/aws/aws-sdk-go-v2/service/s3 v1.109.1/go.mod h1:CUr46sCpGAg/rHaclRyhJX0LJAmH73uWSJPPSaMUrSk=
|
||||
github.com/aws/aws-sdk-go-v2/service/signin v1.10.0 h1:ZD5qFpWcaOKdTuhBi431pIDkCgrMkMlMT6jlpSPoIRI=
|
||||
github.com/aws/aws-sdk-go-v2/service/signin v1.10.0/go.mod h1:8Nuuf+tR346PjJ3MvZPh9pekbLiLQFWJhzMXfwy7alA=
|
||||
github.com/aws/aws-sdk-go-v2/service/sns v1.39.14 h1:p8WdWDh5AwSZdp19Haa3XMyPCICi9Z375a/Nu3IIEZY=
|
||||
@@ -867,13 +848,12 @@ github.com/colinmarc/hdfs/v2 v2.4.0 h1:v6R8oBx/Wu9fHpdPoJJjpGSUxo8NhHIwrwsfhFvU9
|
||||
github.com/colinmarc/hdfs/v2 v2.4.0/go.mod h1:0NAO+/3knbMx6+5pCv+Hcbaz4xn/Zzbn9+WIib2rKVI=
|
||||
github.com/compose-spec/compose-go/v2 v2.12.1 h1:+xBZNxcgSus4atQJwXPEdhHRgCEyZmj/BuqN5m33Ou0=
|
||||
github.com/compose-spec/compose-go/v2 v2.12.1/go.mod h1:ZU6zlcweCZKyiB7BVfCizQT9XmkEIMFE+PRZydVcsZg=
|
||||
github.com/containerd/console v1.0.3/go.mod h1:7LqA/THxQ86k76b8c/EMSiaJ3h1eZkMkXar0TQ1gf3U=
|
||||
github.com/containerd/console v1.0.5 h1:R0ymNeydRqH2DmakFNdmjR2k0t7UPuiOV/N/27/qqsc=
|
||||
github.com/containerd/console v1.0.5/go.mod h1:YynlIjWYF8myEu6sdkwKIvGQq+cOckRm6So2avqoYAk=
|
||||
github.com/containerd/containerd/api v1.11.1 h1:h8nfoDW9+fNsC/9TwiAHj8B1GzXKtR4eFtkhi/X5RLU=
|
||||
github.com/containerd/containerd/api v1.11.1/go.mod h1:CaQFRu+N1MtbgL6JDOJLUB1hCKESU1lD6MuTJhgtdlw=
|
||||
github.com/containerd/containerd/v2 v2.2.5 h1:KTFzB02LviYmmfRmz8r9UFd+n6YlddVFK+5lbgQXUTU=
|
||||
github.com/containerd/containerd/v2 v2.2.5/go.mod h1:5t2+xFv2dGd/iDYp9Z8DXB4cmWrWQi1XqxGJPS2gBzU=
|
||||
github.com/containerd/containerd/v2 v2.2.8 h1:8nnNE5FqBmofd3lccku8GbWi6d1TO4rrcB2E/0o+HU0=
|
||||
github.com/containerd/containerd/v2 v2.2.8/go.mod h1:lTw+wrjREio28N9+3umHS73C6Cs1mxrhczBcAliInuI=
|
||||
github.com/containerd/continuity v0.5.0 h1:7a85HZpCSs+1Zps0Ee3DPSuAWY+0SJM1JNM51nlEVDg=
|
||||
github.com/containerd/continuity v0.5.0/go.mod h1:/lNJvtJKUQStBzpVQ1+rasXO1LAWtUQssk28EZvJ3nE=
|
||||
github.com/containerd/errdefs v1.0.0 h1:tg5yIfIlQIrxYtu9ajqY42W3lpS19XqdxRQeEwYG8PI=
|
||||
@@ -1010,8 +990,8 @@ github.com/facebookgo/subset v0.0.0-20200203212716-c811ad88dec4/go.mod h1:5tD+ne
|
||||
github.com/fanixk/geohash v0.0.0-20150324002647-c1f9b5fa157a h1:Fyfh/dsHFrC6nkX7H7+nFdTd1wROlX/FxEIWVpKYf1U=
|
||||
github.com/fanixk/geohash v0.0.0-20150324002647-c1f9b5fa157a/go.mod h1:UgNw+PTmmGN8rV7RvjvnBMsoTU8ZXXnaT3hYsDTBlgQ=
|
||||
github.com/fatih/color v1.13.0/go.mod h1:kLAiJbzzSOZDVNGyDpeOxJ47H46qBXwg5ILebYFFOfk=
|
||||
github.com/fatih/color v1.18.0 h1:S8gINlzdQ840/4pfAwic/ZE0djQEH3wM94VfqLTZcOM=
|
||||
github.com/fatih/color v1.18.0/go.mod h1:4FelSpRwEGDpQ12mAdzqdOukCy4u8WUtOY6lkT/6HfU=
|
||||
github.com/fatih/color v1.19.0 h1:Zp3PiM21/9Ld6FzSKyL5c/BULoe/ONr9KlbYVOfG8+w=
|
||||
github.com/fatih/color v1.19.0/go.mod h1:zNk67I0ZUT1bEGsSGyCZYZNrHuTkJJB+r6Q9VuMi0LE=
|
||||
github.com/felixge/httpsnoop v1.1.0 h1:3YtUj32ZZkqZtt3sZZsClsymw/QDuVfpNhoA31zeORc=
|
||||
github.com/felixge/httpsnoop v1.1.0/go.mod h1:Zqxgdd+1Rkcz8euOqdr7lqgCRJztwr5hp9vDSi5UZCE=
|
||||
github.com/fluent/fluent-logger-golang v1.10.1 h1:wu54iN1O2afll5oQrtTjhgZRwWcfOeFFzwRsEkABfFQ=
|
||||
@@ -1278,12 +1258,6 @@ github.com/googleapis/gax-go/v2 v2.24.0 h1:myMaPYyF9MecEmvQqMqomIwn9t/4KCZN9qnws
|
||||
github.com/googleapis/gax-go/v2 v2.24.0/go.mod h1:IaTHBDd7NHxSCiu0vEs8pQZu4dGZrWwuSoxCnk16OFM=
|
||||
github.com/googleapis/go-type-adapters v1.0.0/go.mod h1:zHW75FOG2aur7gAO2B+MLby+cLsWGBF62rFAi7WjWO4=
|
||||
github.com/googleapis/google-cloud-go-testing v0.0.0-20200911160855-bcd43fbb19e8/go.mod h1:dvDLG8qkwmyD9a/MJJN3XJcT3xFxOKAvTZGvuZmac9g=
|
||||
github.com/gookit/assert v0.1.1 h1:lh3GcawXe/p+cU7ESTZ5Ui3Sm/x8JWpIis4/1aF0mY0=
|
||||
github.com/gookit/assert v0.1.1/go.mod h1:jS5bmIVQZTIwk42uXl4lyj4iaaxx32tqH16CFj0VX2E=
|
||||
github.com/gookit/color v1.4.2/go.mod h1:fqRyamkC1W8uxl+lxCQxOT09l/vYfZ+QeiX3rKQHCoQ=
|
||||
github.com/gookit/color v1.5.0/go.mod h1:43aQb+Zerm/BWh2GnrgOQm7ffz7tvQXEKV6BFMl7wAo=
|
||||
github.com/gookit/color v1.6.0 h1:JjJXBTk1ETNyqyilJhkTXJYYigHG24TM9Xa2M1xAhRA=
|
||||
github.com/gookit/color v1.6.0/go.mod h1:9ACFc7/1IpHGBW8RwuDm/0YEnhg3dwwXpoMsmtyHfjs=
|
||||
github.com/gopherjs/gopherjs v1.17.2 h1:fQnZVsXk8uxXIStYb0N4bGk7jeyTalG/wsZjQ25dO0g=
|
||||
github.com/gopherjs/gopherjs v1.17.2/go.mod h1:pRRIvn/QzFLrKfvEz3qUuEhtE/zLCWfreZ6J5gM2i+k=
|
||||
github.com/gorilla/mux v1.8.1 h1:TuBL49tXwgrFYWhqrNgrUNEY92u81SPhu7sTdzQEiWY=
|
||||
@@ -1316,13 +1290,13 @@ github.com/hashicorp/go-hclog v1.6.3/go.mod h1:W4Qnvbt70Wk/zYJryRzDRU/4r0kIg0PVH
|
||||
github.com/hashicorp/go-immutable-radix v1.0.0/go.mod h1:0y9vanUI8NX6FsYoO3zeMjhV/C5i9g4Q3DwcSNZ4P60=
|
||||
github.com/hashicorp/go-immutable-radix v1.3.1 h1:DKHmCUm2hRBK510BaiZlwvpD40f8bJFeZnpfm2KLowc=
|
||||
github.com/hashicorp/go-immutable-radix v1.3.1/go.mod h1:0y9vanUI8NX6FsYoO3zeMjhV/C5i9g4Q3DwcSNZ4P60=
|
||||
github.com/hashicorp/go-metrics v0.5.4 h1:8mmPiIJkTPPEbAiV97IxdAGNdRdaWwVap1BU6elejKY=
|
||||
github.com/hashicorp/go-metrics v0.5.4/go.mod h1:CG5yz4NZ/AI/aQt9Ucm/vdBnbh7fvmv4lxZ350i+QQI=
|
||||
github.com/hashicorp/go-metrics v0.7.0 h1:lLWieZTcbzZT+rY0zrqKbyryXG8RIajdUjmM0+R79eg=
|
||||
github.com/hashicorp/go-metrics v0.7.0/go.mod h1:8T/Es8FPTfQvY7azBPGyrwXwwg7mbA9/TmQ1/lWfxb4=
|
||||
github.com/hashicorp/go-msgpack v0.5.5 h1:i9R9JSrqIz0QVLz3sz+i3YJdT7TTSLcfLLzJi9aZTuI=
|
||||
github.com/hashicorp/go-msgpack v0.5.5/go.mod h1:ahLV/dePpqEmjfWmKiqvPkv/twdG7iPBM1vqhUKIvfM=
|
||||
github.com/hashicorp/go-msgpack/v2 v2.1.1/go.mod h1:upybraOAblm4S7rx0+jeNy+CWWhzywQsSRV5033mMu4=
|
||||
github.com/hashicorp/go-msgpack/v2 v2.1.2 h1:4Ee8FTp834e+ewB71RDrQ0VKpyFdrKOjvYtnQ/ltVj0=
|
||||
github.com/hashicorp/go-msgpack/v2 v2.1.2/go.mod h1:upybraOAblm4S7rx0+jeNy+CWWhzywQsSRV5033mMu4=
|
||||
github.com/hashicorp/go-msgpack/v2 v2.1.5 h1:Ue879bPnutj/hXfmUk6s/jtIK90XxgiUIcXRl656T44=
|
||||
github.com/hashicorp/go-msgpack/v2 v2.1.5/go.mod h1:bjCsRXpZ7NsJdk45PoCQnzRGDaK8TKm5ZnDI/9y3J4M=
|
||||
github.com/hashicorp/go-multierror v1.0.0/go.mod h1:dHtQlpGsu+cZNNAkkCN/P3hoUDHhCYQXV3UM06sGGrk=
|
||||
github.com/hashicorp/go-multierror v1.1.1 h1:H5DkEtf6CXdFp0N0Em5UCwQpXMWke8IA0+lD48awMYo=
|
||||
github.com/hashicorp/go-multierror v1.1.1/go.mod h1:iw975J/qwKPdAO1clOe2L8331t/9/fmwbPZ6JB6eMoM=
|
||||
@@ -1346,19 +1320,19 @@ github.com/hashicorp/go-version v1.9.0/go.mod h1:fltr4n8CU8Ke44wwGCBoEymUuxUHl09
|
||||
github.com/hashicorp/golang-lru v0.5.0/go.mod h1:/m3WP610KZHVQ1SGc6re/UDhFvYD7pJ4Ao+sR/qLZy8=
|
||||
github.com/hashicorp/golang-lru v0.5.1/go.mod h1:/m3WP610KZHVQ1SGc6re/UDhFvYD7pJ4Ao+sR/qLZy8=
|
||||
github.com/hashicorp/golang-lru v0.5.4/go.mod h1:iADmTwqILo4mZ8BN3D2Q6+9jd8WM5uGBxy+E8yxSoD4=
|
||||
github.com/hashicorp/golang-lru v0.6.0 h1:uL2shRDx7RTrOrTCUZEGP/wJUFiUI8QT6E7z5o8jga4=
|
||||
github.com/hashicorp/golang-lru v0.6.0/go.mod h1:iADmTwqILo4mZ8BN3D2Q6+9jd8WM5uGBxy+E8yxSoD4=
|
||||
github.com/hashicorp/golang-lru v1.0.2 h1:dV3g9Z/unq5DpblPpw+Oqcv4dU/1omnb4Ok8iPY6p1c=
|
||||
github.com/hashicorp/golang-lru v1.0.2/go.mod h1:iADmTwqILo4mZ8BN3D2Q6+9jd8WM5uGBxy+E8yxSoD4=
|
||||
github.com/hashicorp/golang-lru/v2 v2.0.7 h1:a+bsQ5rvGLjzHuww6tVxozPZFVghXaHOwFs4luLUK2k=
|
||||
github.com/hashicorp/golang-lru/v2 v2.0.7/go.mod h1:QeFd9opnmA6QUJc5vARoKUSoFhyfM2/ZepoAG6RGpeM=
|
||||
github.com/hashicorp/hcl v1.0.1-vault-7 h1:ag5OxFVy3QYTFTJODRzTKVZ6xvdfLLCA1cy/Y6xGI0I=
|
||||
github.com/hashicorp/hcl v1.0.1-vault-7/go.mod h1:XYhtn6ijBSAj6n4YqAaf7RBPS4I06AItNorpy+MoQNM=
|
||||
github.com/hashicorp/raft v1.7.0/go.mod h1:N1sKh6Vn47mrWvEArQgILTyng8GoDRNYlgKyK7PMjs0=
|
||||
github.com/hashicorp/raft v1.7.3 h1:DxpEqZJysHN0wK+fviai5mFcSYsCkNpFUl1xpAW8Rbo=
|
||||
github.com/hashicorp/raft v1.7.3/go.mod h1:DfvCGFxpAUPE0L4Uc8JLlTPtc3GzSbdH0MTJCLgnmJQ=
|
||||
github.com/hashicorp/raft v1.8.0 h1:YbfecBcuTar/LNFEDfVTpqu9Aw+MczTk7MYczvy+62k=
|
||||
github.com/hashicorp/raft v1.8.0/go.mod h1:agL5fncrpEsbxr5P5KOd2srskDwPY18opjXN5x0661s=
|
||||
github.com/hashicorp/raft-boltdb v0.0.0-20230125174641-2a8082862702 h1:RLKEcCuKcZ+qp2VlaaZsYZfLOmIiuJNpEi48Rl8u9cQ=
|
||||
github.com/hashicorp/raft-boltdb v0.0.0-20230125174641-2a8082862702/go.mod h1:nTakvJ4XYq45UXtn0DbwR4aU9ZdjlnIenpbs6Cd+FM0=
|
||||
github.com/hashicorp/raft-boltdb/v2 v2.3.1 h1:ackhdCNPKblmOhjEU9+4lHSJYFkJd6Jqyvj6eW9pwkc=
|
||||
github.com/hashicorp/raft-boltdb/v2 v2.3.1/go.mod h1:n4S+g43dXF1tqDT+yzcXHhXM6y7MrlUd3TTwGRcUvQE=
|
||||
github.com/hashicorp/raft-boltdb/v2 v2.4.2 h1:r2RRgZ6ajT+VH8/9yC5mfoKAlt3KrYiUDhfbUSmaoTs=
|
||||
github.com/hashicorp/raft-boltdb/v2 v2.4.2/go.mod h1:+wvKK1thWEqM7amY1MXjtQpkm+b9KNUXks0fcus5ZGU=
|
||||
github.com/hashicorp/vault/api v1.23.0 h1:gXgluBsSECfRWTSW9niY2jwg2e9mMJc4WoHNv4g3h6A=
|
||||
github.com/hashicorp/vault/api v1.23.0/go.mod h1:zransKiB9ftp+kgY8ydjnvCU7Wk8i9L0DYWpXeMj9ko=
|
||||
github.com/hexops/gotextdiff v1.0.3 h1:gitA9+qJrrTCsiCl7+kh75nPqQt1cx4ZkudSTLoUqJM=
|
||||
@@ -1418,11 +1392,8 @@ github.com/jonboulle/clockwork v0.5.0 h1:Hyh9A8u51kptdkR+cqRpT1EebBwTn1oK9YfGYbd
|
||||
github.com/jonboulle/clockwork v0.5.0/go.mod h1:3mZlmanh0g2NDKO5TWZVJAfofYk64M7XN3SzBPjZF60=
|
||||
github.com/josharian/intern v1.0.0 h1:vlS4z54oSdjm0bgjRigI+G1HpF+tI+9rE5LLzOg8HmY=
|
||||
github.com/josharian/intern v1.0.0/go.mod h1:5DoeVV0s6jJacbCEi61lwdGj/aVlrQvzHFFd8Hwg//Y=
|
||||
github.com/jpillora/backoff v1.0.0/go.mod h1:J/6gKK9jxlEcS3zixgDgUAsiuZ7yrSoa/FX5e0EB2j4=
|
||||
github.com/json-iterator/go v1.1.6/go.mod h1:+SdeFBvtyEkXs7REEP0seUULqWtbJapLOCVDaaPEHmU=
|
||||
github.com/json-iterator/go v1.1.9/go.mod h1:KdQUCv79m/52Kvf8AW2vK1V8akMuk1QjK/uOdHXbAo4=
|
||||
github.com/json-iterator/go v1.1.10/go.mod h1:KdQUCv79m/52Kvf8AW2vK1V8akMuk1QjK/uOdHXbAo4=
|
||||
github.com/json-iterator/go v1.1.11/go.mod h1:KdQUCv79m/52Kvf8AW2vK1V8akMuk1QjK/uOdHXbAo4=
|
||||
github.com/json-iterator/go v1.1.12 h1:PV8peI4a0ysnczrg+LtxykD8LfKY9ML6u2jnxaEnrnM=
|
||||
github.com/json-iterator/go v1.1.12/go.mod h1:e30LSqwooZae/UwlEbR2852Gd8hjQvJoHmT4TnhNGBo=
|
||||
github.com/jstemmer/go-junit-report v0.0.0-20190106144839-af01ea7f8024/go.mod h1:6v2b51hI/fHJwM22ozAgKL4VKDeJcHhJFhtBdhmNjmU=
|
||||
@@ -1432,7 +1403,6 @@ github.com/jtolds/gls v4.20.0+incompatible/go.mod h1:QJZ7F/aHp+rZTRtaJ1ow/lLfFfV
|
||||
github.com/jtolio/noiseconn v0.0.0-20231127013910-f6d9ecbf1de7 h1:JcltaO1HXM5S2KYOYcKgAV7slU0xPy1OcvrVgn98sRQ=
|
||||
github.com/jtolio/noiseconn v0.0.0-20231127013910-f6d9ecbf1de7/go.mod h1:MEkhEPFwP3yudWO0lj6vfYpLIB+3eIcuIW+e0AZzUQk=
|
||||
github.com/julienschmidt/httprouter v1.2.0/go.mod h1:SYymIcj16QtmaHHD7aYtjjsJG7VTCxuUUipMqKk8s4w=
|
||||
github.com/julienschmidt/httprouter v1.3.0/go.mod h1:JR6WtHb+2LUe8TCKY3cZOxFyyO8IZAc4RVcycCCAKdM=
|
||||
github.com/jung-kurt/gofpdf v1.0.0/go.mod h1:7Id9E/uU8ce6rXgefFLlgrJj/GYY22cpxn+r32jIOes=
|
||||
github.com/jung-kurt/gofpdf v1.0.3-0.20190309125859-24315acbbda5/go.mod h1:7Id9E/uU8ce6rXgefFLlgrJj/GYY22cpxn+r32jIOes=
|
||||
github.com/jzelinskie/whirlpool v0.0.0-20201016144138-0675e54bb004 h1:G+9t9cEtnC9jFiTxyptEKuNIAbiN5ZCQzX2a74lj3xg=
|
||||
@@ -1456,14 +1426,11 @@ github.com/klauspost/compress v1.15.9/go.mod h1:PhcZ0MbTNciWF3rruxRgKxI5NkcHHrHU
|
||||
github.com/klauspost/compress v1.19.2 h1:hMRETovs/pu/dVWN7zIT1PGG8t509MwT6bO7XSi26R8=
|
||||
github.com/klauspost/compress v1.19.2/go.mod h1:cwPg85FWrGar70rWktvGQj8/hthj3wpl0PGDogxkrSQ=
|
||||
github.com/klauspost/cpuid/v2 v2.0.9/go.mod h1:FInQzS24/EEf25PyTYn52gqo7WaD8xa0213Md/qVLRg=
|
||||
github.com/klauspost/cpuid/v2 v2.0.10/go.mod h1:g2LTdtYhdyuGPqyWyv7qRAmj1WBqxuObKfj5c0PQa7c=
|
||||
github.com/klauspost/cpuid/v2 v2.0.12/go.mod h1:g2LTdtYhdyuGPqyWyv7qRAmj1WBqxuObKfj5c0PQa7c=
|
||||
github.com/klauspost/cpuid/v2 v2.4.0 h1:S6Hrbc7+ywsr0r+RLapfGBHfyefhCTwEh3A0tV913Dw=
|
||||
github.com/klauspost/cpuid/v2 v2.4.0/go.mod h1:19jmZ9mjzoF//ddRSUsv0zfBTJWh3QJh9FNxZTMrGxU=
|
||||
github.com/klauspost/reedsolomon v1.14.2 h1:SafJYwpBBQBI6amHUygcjxZjXeN2HpiENHQDwuPWCCQ=
|
||||
github.com/klauspost/reedsolomon v1.14.2/go.mod h1:yjqqjgMTQkBUHSG97/rm4zipffCNbCiZcB3kTqr++sQ=
|
||||
github.com/konsorten/go-windows-terminal-sequences v1.0.1/go.mod h1:T0+1ngSBFLxvqU3pZ+m/2kptfBszLMUkC4ZK/EgS/cQ=
|
||||
github.com/konsorten/go-windows-terminal-sequences v1.0.3/go.mod h1:T0+1ngSBFLxvqU3pZ+m/2kptfBszLMUkC4ZK/EgS/cQ=
|
||||
github.com/koofr/go-httpclient v0.0.0-20240520111329-e20f8f203988 h1:CjEMN21Xkr9+zwPmZPaJJw+apzVbjGL5uK/6g9Q2jGU=
|
||||
github.com/koofr/go-httpclient v0.0.0-20240520111329-e20f8f203988/go.mod h1:/agobYum3uo/8V6yPVnq+R82pyVGCeuWW5arT4Txn8A=
|
||||
github.com/koofr/go-koofrclient v0.0.0-20221207135200-cbd7fc9ad6a6 h1:FHVoZMOVRA+6/y4yRlbiR3WvsrOcKBd/f64H7YiWR2U=
|
||||
@@ -1495,8 +1462,6 @@ github.com/linkedin/goavro/v2 v2.15.0 h1:pDj1UrjUOO62iXhgBiE7jQkpNIc5/tA5eZsgolM
|
||||
github.com/linkedin/goavro/v2 v2.15.0/go.mod h1:KXx+erlq+RPlGSPmLF7xGo6SAbh8sCQ53x064+ioxhk=
|
||||
github.com/linxGnu/grocksdb v1.10.8 h1:Nau01Hhm/0kaVTR6d4viwD6npYbnDvZAfzwJCLzKRYo=
|
||||
github.com/linxGnu/grocksdb v1.10.8/go.mod h1:OLQKZwiKwaJiAVCsOzWKvwiLwfZ5Vz8Md5TYR7t7pM8=
|
||||
github.com/lithammer/fuzzysearch v1.1.8 h1:/HIuJnjHuXS8bKaiTMeeDlW2/AyIWk2brx1V8LFgLN4=
|
||||
github.com/lithammer/fuzzysearch v1.1.8/go.mod h1:IdqeyBClc3FFqSzYq/MXESsS4S0FsZ5ajtkr5xPLts4=
|
||||
github.com/lithammer/shortuuid/v3 v3.0.7 h1:trX0KTHy4Pbwo/6ia8fscyHoGA+mf1jWbPJVuvyJQQ8=
|
||||
github.com/lithammer/shortuuid/v3 v3.0.7/go.mod h1:vMk8ke37EmiewwolSO1NLW8vP4ZaKlRuDIi8tWWmAts=
|
||||
github.com/lpar/calendar v0.2.0 h1:A1kxv6sbvBHFUkd2XotanIRqEXQGreQOeuGhkJqIaRA=
|
||||
@@ -1521,7 +1486,6 @@ github.com/mattn/go-isatty v0.0.16/go.mod h1:kYGgaQfpe5nmfYZH+SKPsOc2e4SrIfOl2e/
|
||||
github.com/mattn/go-isatty v0.0.24 h1:tGZZoVgT/KiqK1c8ocVLeDS8BSWMRd47J3Lbz7vsReI=
|
||||
github.com/mattn/go-isatty v0.0.24/go.mod h1:nMCL3Zebbrt45jsMDgnfIwz6ydEQApk5oEI3HqDio6A=
|
||||
github.com/mattn/go-runewidth v0.0.3/go.mod h1:LwmH8dsx7+W8Uxz3IHJYH5QSwggIsqBzpuz5H//U1FU=
|
||||
github.com/mattn/go-runewidth v0.0.13/go.mod h1:Jdepj2loyihRzMpdS35Xk/zdY8IAYHsh153qUoGf23w=
|
||||
github.com/mattn/go-runewidth v0.0.24 h1:cpokDiIn0MGnhdHwuWnJBITySJ20QyNGnY2kR/ay2DU=
|
||||
github.com/mattn/go-runewidth v0.0.24/go.mod h1:XBkDxAl56ILZc9knddidhrOlY5R/pDhgLpndooCuJAs=
|
||||
github.com/mattn/go-shellwords v1.0.13 h1:DC0OMEpGjm6LfNFU4ckYcvbQKyp2vE8atyFGXNtDcf4=
|
||||
@@ -1546,8 +1510,8 @@ github.com/mitchellh/mapstructure v1.5.1-0.20220423185008-bf980b35cac4 h1:BpfhmL
|
||||
github.com/mitchellh/mapstructure v1.5.1-0.20220423185008-bf980b35cac4/go.mod h1:bFUtVrKA4DC2yAKiSyO/QUcy7e+RRV2QTWOzhPopBRo=
|
||||
github.com/mmcloughlin/geohash v0.9.0 h1:FihR004p/aE1Sju6gcVq5OLDqGcMnpBY+8moBqIsVOs=
|
||||
github.com/mmcloughlin/geohash v0.9.0/go.mod h1:oNZxQo5yWJh0eMQEP/8hwQuVx9Z9tjwFUqcTB1SmG0c=
|
||||
github.com/moby/buildkit v0.31.0 h1:hMUAbQGgjtzJDDOZ6o7MQk5XBZkBTyzLWEvnjguHHQI=
|
||||
github.com/moby/buildkit v0.31.0/go.mod h1:YM5iNEbNCc6L1Zt3YWFB/aXNLufvf4Rcu0DPlc9HwQg=
|
||||
github.com/moby/buildkit v0.31.1 h1:j3p55abBl4kiXXPZgYX+6zWgB2aefqHXoPown12fIzU=
|
||||
github.com/moby/buildkit v0.31.1/go.mod h1:YM5iNEbNCc6L1Zt3YWFB/aXNLufvf4Rcu0DPlc9HwQg=
|
||||
github.com/moby/docker-image-spec v1.3.1 h1:jMKff3w6PgbfSa69GfNg+zN/XLhfXJGnEx3Nl2EsFP0=
|
||||
github.com/moby/docker-image-spec v1.3.1/go.mod h1:eKmb5VW8vQEh/BAr2yvVNvuiJuY6UIocYsFu/DxxRpo=
|
||||
github.com/moby/go-archive v0.3.0 h1:nos4BtzzUIqB406BgQnWGMI4qib9BZ8XUHU+ucv/n1c=
|
||||
@@ -1594,7 +1558,6 @@ github.com/mschoch/smat v0.2.0/go.mod h1:kc9mz7DoBKqDyiRL7VZN8KvXQMWeTaVnttLRXOl
|
||||
github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822 h1:C3w9PqII01/Oq1c1nUAm88MOHcQC9l5mIlSMApZMrHA=
|
||||
github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822/go.mod h1:+n7T8mK8HuQTcFwEeznm/DIxMOiR9yIdICNftLE1DvQ=
|
||||
github.com/mwitkow/go-conntrack v0.0.0-20161129095857-cc309e4a2223/go.mod h1:qRWi+5nqEBWmkhHvq77mSJWrCKwh8bxhgT7d/eI7P4U=
|
||||
github.com/mwitkow/go-conntrack v0.0.0-20190716064945-2f068394615f/go.mod h1:qRWi+5nqEBWmkhHvq77mSJWrCKwh8bxhgT7d/eI7P4U=
|
||||
github.com/nats-io/jwt/v2 v2.8.1 h1:V0xpGuD/N8Mi+fQNDynXohVvp7ZztevW5io8CUWlPmU=
|
||||
github.com/nats-io/jwt/v2 v2.8.1/go.mod h1:nWnOEEiVMiKHQpnAy4eXlizVEtSfzacZ1Q43LIRavZg=
|
||||
github.com/nats-io/nats-server/v2 v2.11.15 h1:StSf9TINInaZtr4oww2+kXmfwa9SkN//g/LwS19/UJ0=
|
||||
@@ -1717,8 +1680,6 @@ github.com/pquerna/otp v1.5.0/go.mod h1:dkJfzwRKNiegxyNb54X/3fLwhCynbMspSyWKnvi1
|
||||
github.com/prometheus/client_golang v0.9.1/go.mod h1:7SWBe2y4D6OKWSNQJUaRYU/AaXPKyh/dDVn+NZz0KFw=
|
||||
github.com/prometheus/client_golang v1.0.0/go.mod h1:db9x61etRT2tGnBNRi70OPL5FsnadC4Ky3P0J6CfImo=
|
||||
github.com/prometheus/client_golang v1.4.0/go.mod h1:e9GMxYsXl05ICDXkRhurwBS4Q3OK1iX/F2sw+iXX5zU=
|
||||
github.com/prometheus/client_golang v1.7.1/go.mod h1:PY5Wy2awLA44sXw4AOSfFBetzPP4j5+D6mVACh+pe2M=
|
||||
github.com/prometheus/client_golang v1.11.1/go.mod h1:Z6t4BnS23TR94PD6BsDNk8yVqroYurpAkEiz0P2BEV0=
|
||||
github.com/prometheus/client_golang v1.24.1 h1:JnJkREXzWxUdCuPFpIWZiPispT9xVV59uiuyR2bPlnU=
|
||||
github.com/prometheus/client_golang v1.24.1/go.mod h1:F+oSRECHg4sse5ucfYpYDeIv/hu68Zo0uoHKetWnzcE=
|
||||
github.com/prometheus/client_model v0.0.0-20180712105110-5c3871d89910/go.mod h1:MbSGuTsp3dbXC40dX6PRTWyKYBIrTGTE9sqQNg2J8bo=
|
||||
@@ -1730,26 +1691,13 @@ github.com/prometheus/client_model v0.6.3 h1:O0jaTVAYNxTHYInEPFJt5I3+sN8zqBtVMPT
|
||||
github.com/prometheus/client_model v0.6.3/go.mod h1:gpN5P9S7Rr6Yr92PiQ+Ixvhf6JZEkF1dnxsYL2aPBEM=
|
||||
github.com/prometheus/common v0.4.1/go.mod h1:TNfzLD0ON7rHzMJeJkieUDPYmFC7Snx/y86RQel1bk4=
|
||||
github.com/prometheus/common v0.9.1/go.mod h1:yhUN8i9wzaXS3w1O07YhxHEBxD+W35wd8bs7vj7HSQ4=
|
||||
github.com/prometheus/common v0.10.0/go.mod h1:Tlit/dnDKsSWFlCLTWaA1cyBgKHSMdTB80sz/V91rCo=
|
||||
github.com/prometheus/common v0.26.0/go.mod h1:M7rCNAaPfAosfx8veZJCuw84e35h3Cfd9VFqTh1DIvc=
|
||||
github.com/prometheus/common v0.70.1 h1:1HvjP4D5oL3t8RsPlwxA9onvvStjtIHYE5XuuwOi/PY=
|
||||
github.com/prometheus/common v0.70.1/go.mod h1:VdFUQDMZK3VLkurFUVhia6uys/0suUp86TJz5qbJRhc=
|
||||
github.com/prometheus/common v0.71.0 h1:9KDAKb7Mj3HEVKyFCK6Dc/HIwlBzZIN2l7/lrHl3KK8=
|
||||
github.com/prometheus/common v0.71.0/go.mod h1:CLJ5H8TEsGX8bl31BdMkfhIZ+QmZ9tBPPotUxUbfcmk=
|
||||
github.com/prometheus/procfs v0.0.0-20181005140218-185b4288413d/go.mod h1:c3At6R/oaqEKCNdg8wHV1ftS6bRYblBhIjjI8uT2IGk=
|
||||
github.com/prometheus/procfs v0.0.2/go.mod h1:TjEm7ze935MbeOT/UhFTIMYKhuLP4wbCsTZCD3I8kEA=
|
||||
github.com/prometheus/procfs v0.0.8/go.mod h1:7Qr8sr6344vo1JqZ6HhLceV9o3AJ1Ff+GxbHq6oeK9A=
|
||||
github.com/prometheus/procfs v0.1.3/go.mod h1:lV6e/gmhEcM9IjHGsFOCxxuZ+z1YqCvr4OA4YeYWdaU=
|
||||
github.com/prometheus/procfs v0.6.0/go.mod h1:cz+aTbrPOrUb4q7XlbU9ygM+/jj0fzG6c1xBZuNvfVA=
|
||||
github.com/prometheus/procfs v0.22.0 h1:6q9+/JL9IKAPbCmBrv9n5O5Ty3NKnciV5X7YGw0oics=
|
||||
github.com/prometheus/procfs v0.22.0/go.mod h1:CvmFr/GVhIjIvWJZW3tgkODBQMRIf0EyWMQLHCHab58=
|
||||
github.com/pterm/pterm v0.12.27/go.mod h1:PhQ89w4i95rhgE+xedAoqous6K9X+r6aSOI2eFF7DZI=
|
||||
github.com/pterm/pterm v0.12.29/go.mod h1:WI3qxgvoQFFGKGjGnJR849gU0TsEOvKn5Q8LlY1U7lg=
|
||||
github.com/pterm/pterm v0.12.30/go.mod h1:MOqLIyMOgmTDz9yorcYbcw+HsgoZo3BQfg2wtl3HEFE=
|
||||
github.com/pterm/pterm v0.12.31/go.mod h1:32ZAWZVXD7ZfG0s8qqHXePte42kdz8ECtRyEejaWgXU=
|
||||
github.com/pterm/pterm v0.12.33/go.mod h1:x+h2uL+n7CP/rel9+bImHD5lF3nM9vJj80k9ybiiTTE=
|
||||
github.com/pterm/pterm v0.12.36/go.mod h1:NjiL09hFhT/vWjQHSj1athJpx6H8cjpHXNAK5bUw8T8=
|
||||
github.com/pterm/pterm v0.12.40/go.mod h1:ffwPLwlbXxP+rxT0GsgDTzS3y3rmpAO1NMjUkGTYf8s=
|
||||
github.com/pterm/pterm v0.12.83 h1:ie+YmGmA727VuhxBlyGr74Ks+7McV6kT99IB8EU80aA=
|
||||
github.com/pterm/pterm v0.12.83/go.mod h1:xlgc6bFWyJIMtmLJvGim+L7jhSReilOlOnodeIYe4Tk=
|
||||
github.com/putdotio/go-putio/putio v0.0.0-20200123120452-16d982cac2b8 h1:Y258uzXU/potCYnQd1r6wlAnoMB68BiCkCcCnKx1SH8=
|
||||
github.com/putdotio/go-putio/putio v0.0.0-20200123120452-16d982cac2b8/go.mod h1:bSJjRokAHHOhA+XFxplld8w2R/dXLH7Z3BZ532vhFwU=
|
||||
github.com/puzpuzpuz/xsync/v3 v3.5.1 h1:GJYJZwO6IdxN/IKbneznS6yPkVC+c3zyY/j19c++5Fg=
|
||||
@@ -1758,8 +1706,8 @@ github.com/quic-go/qpack v0.6.0 h1:g7W+BMYynC1LbYLSqRt8PBg5Tgwxn214ZZR34VIOjz8=
|
||||
github.com/quic-go/qpack v0.6.0/go.mod h1:lUpLKChi8njB4ty2bFLX2x4gzDqXwUpaO1DP9qMDZII=
|
||||
github.com/quic-go/quic-go v0.59.0 h1:OLJkp1Mlm/aS7dpKgTc6cnpynnD2Xg7C1pwL6vy/SAw=
|
||||
github.com/quic-go/quic-go v0.59.0/go.mod h1:upnsH4Ju1YkqpLXC305eW3yDZ4NfnNbmQRCMWS58IKU=
|
||||
github.com/rabbitmq/amqp091-go v1.14.0 h1:RSaT7aOKt/OrkVUyswPDW29lnRz9psuGmfZFBmLqLek=
|
||||
github.com/rabbitmq/amqp091-go v1.14.0/go.mod h1:Hy4jKW5kQART1u+JkDTF9YYOQUHXqMuhrgxOEeS7G4o=
|
||||
github.com/rabbitmq/amqp091-go v1.15.0 h1:LEQL4/yp48/Wigt6A6XOu18RQRo8ZHtB5I/KZJn+gkw=
|
||||
github.com/rabbitmq/amqp091-go v1.15.0/go.mod h1:Hy4jKW5kQART1u+JkDTF9YYOQUHXqMuhrgxOEeS7G4o=
|
||||
github.com/rclone/Proton-API-Bridge v1.0.5 h1:K1++Qtk3PvgkiCCiv6Pahju1TMOzKY6VSwiwT7XLAVc=
|
||||
github.com/rclone/Proton-API-Bridge v1.0.5/go.mod h1:vCeOPhlXzevN0AFojgh1zsjhetiShy/ArvJ/xkFUDWk=
|
||||
github.com/rclone/go-proton-api v1.0.4 h1:AJW0e9pB4j0hVK4WqyGErFwaI+5MUQWPCtj5FYYxtPg=
|
||||
@@ -1786,7 +1734,6 @@ github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec h1:W09IVJc94
|
||||
github.com/remyoudompheng/bigfft v0.0.0-20230129092748-24d4a6f8daec/go.mod h1:qqbHyh8v60DhA7CoWK5oRCqLrMHRGoxYCSS9EjAz6Eo=
|
||||
github.com/rfjakob/eme v1.2.0 h1:8dAHL+WVAw06+7DkRKnRiFp1JL3QjcJEZFqDnndUaSI=
|
||||
github.com/rfjakob/eme v1.2.0/go.mod h1:cVvpasglm/G3ngEfcfT/Wt0GwhkuO32pf/poW6Nyk1k=
|
||||
github.com/rivo/uniseg v0.2.0/go.mod h1:J6wj4VEh+S6ZtnVlnTBMWIodfgj8LQOQFoIToxlJtxc=
|
||||
github.com/rivo/uniseg v0.4.7 h1:WUdvkW8uEhrYfLC4ZzdpI2ztxP1I582+49Oc5Mq64VQ=
|
||||
github.com/rivo/uniseg v0.4.7/go.mod h1:FN3SvrM+Zdj16jyLfmOkMNblXMcoc8DfTHruCPUcx88=
|
||||
github.com/rogpeppe/fastuuid v1.2.0/go.mod h1:jVj6XXZzXRy/MSR5jhDC/2q6DgLz+nrA6LYCDYWNEvQ=
|
||||
@@ -1840,7 +1787,6 @@ github.com/sigstore/sigstore-go v1.2.1/go.mod h1:I8BqVwAb/SaQJ5pBu5IDFY+ksq8O/1/
|
||||
github.com/sirupsen/logrus v1.2.0/go.mod h1:LxeOpSwHxABJmUn/MG1IvRgCAasNZTLOkJPxbbu5VWo=
|
||||
github.com/sirupsen/logrus v1.4.2/go.mod h1:tLMulIdttU9McNUspp0xgXVQah82FyeX6MwdIuYE2rE=
|
||||
github.com/sirupsen/logrus v1.5.0/go.mod h1:+F7Ogzej0PZc/94MaYx/nvG9jOFMD2osvC3s+Squfpo=
|
||||
github.com/sirupsen/logrus v1.6.0/go.mod h1:7uNnSEd1DgxDLC74fIahvMZmmYsHGZGEOFrfsX/uA88=
|
||||
github.com/sirupsen/logrus v1.7.0/go.mod h1:yWOB1SBYBC5VeMP7gHvWumXLIWorT60ONWic61uBYv0=
|
||||
github.com/sirupsen/logrus v1.9.4 h1:TsZE7l11zFCLZnZ+teH4Umoq5BhEIfIzfRDZ1Uzql2w=
|
||||
github.com/sirupsen/logrus v1.9.4/go.mod h1:ftWc9WdOfJ0a92nsE2jF5u5ZwH8Bv2zdeOC42RjbV2g=
|
||||
@@ -1928,9 +1874,8 @@ github.com/the42/cartconvert v0.0.0-20131203171324-aae784c392b8 h1:I4DY8wLxJXCrM
|
||||
github.com/the42/cartconvert v0.0.0-20131203171324-aae784c392b8/go.mod h1:fWO/msnJVhHqN1yX6OBoxSyfj7TEj1hHiL8bJSQsK30=
|
||||
github.com/tiancaiamao/gp v0.0.0-20221230034425-4025bc8a4d4a h1:J/YdBZ46WKpXsxsW93SG+q0F8KI+yFrcIDT4c/RNoc4=
|
||||
github.com/tiancaiamao/gp v0.0.0-20221230034425-4025bc8a4d4a/go.mod h1:h4xBhSNtOeEosLJ4P7JyKXX7Cabg7AVkWCK5gV2vOrM=
|
||||
github.com/tidwall/gjson v1.18.0 h1:FIDeeyB800efLX89e5a8Y0BNH+LOngJyGrIWxG2FKQY=
|
||||
github.com/tidwall/gjson v1.18.0/go.mod h1:/wbyibRr2FHMks5tjHJ5F8dMZh3AcwJEMf5vlfC0lxk=
|
||||
github.com/tidwall/match v1.1.1/go.mod h1:eRSPERbgtNPcGhD8UCthc6PmLEQXEWd3PRB5JTxsfmM=
|
||||
github.com/tidwall/gjson v1.19.0 h1:xwxm7n691Uf3u5OFjzngavjGTh55KX5q/9w9xHW88JU=
|
||||
github.com/tidwall/gjson v1.19.0/go.mod h1:V37/opeE/JbLUOfH0QTXiNez2l0RUjYUhpT4szFQAfc=
|
||||
github.com/tidwall/match v1.2.0 h1:0pt8FlkOwjN2fPt4bIl4BoNxb98gGHN2ObFEDkrfZnM=
|
||||
github.com/tidwall/match v1.2.0/go.mod h1:eRSPERbgtNPcGhD8UCthc6PmLEQXEWd3PRB5JTxsfmM=
|
||||
github.com/tidwall/pretty v1.2.0 h1:RWIZEg2iJ8/g6fDDYzMpobmaoGh5OLl4AXtGUGPcqCs=
|
||||
@@ -2032,9 +1977,6 @@ github.com/xeipuuv/gojsonschema v1.2.0 h1:LhYJRs+L4fBtjZUfuSZIKGeVu0QRy8e5Xi7D17
|
||||
github.com/xeipuuv/gojsonschema v1.2.0/go.mod h1:anYRn/JVcOK2ZgGU+IjEV4nwlhoK5sQluxsYJ78Id3Y=
|
||||
github.com/xhit/go-str2duration/v2 v2.1.0 h1:lxklc02Drh6ynqX+DdPyp5pCKLUQpRT8bp8Ydu2Bstc=
|
||||
github.com/xhit/go-str2duration/v2 v2.1.0/go.mod h1:ohY8p+0f07DiV6Em5LKB0s2YpLtXVyJfNt1+BlmyAsU=
|
||||
github.com/xo/terminfo v0.0.0-20210125001918-ca9a967f8778/go.mod h1:2MuV+tbUrU1zIOPMxZ5EncGwgmMJsa+9ucAQZXxsObs=
|
||||
github.com/xo/terminfo v0.0.0-20220910002029-abceb7e1c41e h1:JVG44RsyaB9T2KIHavMF/ppJZNG9ZpyihvCd0w101no=
|
||||
github.com/xo/terminfo v0.0.0-20220910002029-abceb7e1c41e/go.mod h1:RbqR21r5mrJuqunuUZ/Dhy/avygyECGrLceyNeo4LiM=
|
||||
github.com/xyproto/randomstring v1.0.5 h1:YtlWPoRdgMu3NZtP45drfy1GKoojuR7hmRcnhZqKjWU=
|
||||
github.com/xyproto/randomstring v1.0.5/go.mod h1:rgmS5DeNXLivK7YprL0pY+lTuhNQW3iGxZ18UQApw/E=
|
||||
github.com/yandex-cloud/go-genproto v0.0.0-20211115083454-9ca41db5ed9e h1:9LPdmD1vqadsDQUva6t2O9MbnyvoOgo8nFNPaOIH5U8=
|
||||
@@ -2111,8 +2053,8 @@ go.opentelemetry.io/contrib/instrumentation/net/http/httptrace/otelhttptrace v0.
|
||||
go.opentelemetry.io/contrib/instrumentation/net/http/httptrace/otelhttptrace v0.69.0/go.mod h1:3jnStNwSufK+f5ktjL4EPcwtig4rtd81NS70lqHuXl8=
|
||||
go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.70.0 h1:LMuyCAyfalSjDyjdC65nK6N0zoTT63+E/u95X0JovZI=
|
||||
go.opentelemetry.io/contrib/instrumentation/net/http/otelhttp v0.70.0/go.mod h1:085m8qbm4hgc8rZWGDEa4vmyyo2c3nPxUslYUKUIU04=
|
||||
go.opentelemetry.io/otel v1.45.0 h1:pdrWmLHofpubmArBv1LgFSv1Z0Ie/ppdZzu+kUN5EeU=
|
||||
go.opentelemetry.io/otel v1.45.0/go.mod h1:XZxIqPapzEYnhNSScF5DIqXhm/rYi0FzCe2XddAwZfQ=
|
||||
go.opentelemetry.io/otel v1.46.0 h1:FHt5/CDyVxi/8IM1CH7VE/rRgq3kLHa2mSTVMO8AWyc=
|
||||
go.opentelemetry.io/otel v1.46.0/go.mod h1:Gj3SEScelsNC45tp4nSxRYlS+f5iez7W8XPMCt905kE=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlpmetric/otlpmetricgrpc v1.44.0 h1:SUplec5dp06reu1zaXmOXdvqH398taqrDXqUl99jxSc=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlpmetric/otlpmetricgrpc v1.44.0/go.mod h1:ho2g4N+ane+swq5I/VBkKWnRDY4kUINH3FuqyZqX/Ug=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlpmetric/otlpmetrichttp v1.44.0 h1:RuynHbfU8JUEw7DyONgkVYg2SVtsoF28y0LGIr69jgA=
|
||||
@@ -2121,22 +2063,22 @@ go.opentelemetry.io/otel/exporters/otlp/otlptrace v1.45.0 h1:QRefszxJmfPdjXUUm3j
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace v1.45.0/go.mod h1:Tiz03lTBVBrm7eWZBOidzEaYaJa8tjwGUGv6d8mlTyk=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracegrpc v1.45.0 h1:fG5MCxGz8+2VtrN/WgqSpJFctVz24gpxj8CxkKmc8Ww=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracegrpc v1.45.0/go.mod h1:BmAYTn+3ysbRe+IU2msxmf5Rx3g6DHvex+tWI3LdhYI=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracehttp v1.44.0 h1:lgh3PiVrRUWMLOVSkQicxzZll5NjF1r+AtsX1XRIHw0=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracehttp v1.44.0/go.mod h1:5Cnhth3m/AgOeTgE3ex12pPmiu/gGtZit03kSzx9X7s=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracehttp v1.45.0 h1:QBajQ2SrwQijzHyZbQlPsuIzpl/ll8DY6wPWsajeGcI=
|
||||
go.opentelemetry.io/otel/exporters/otlp/otlptrace/otlptracehttp v1.45.0/go.mod h1:08ZQLjrPLQ6R4kAXvuOvODEer5Yh4CoFvll5qB2BCI8=
|
||||
go.opentelemetry.io/otel/exporters/stdout/stdoutmetric v1.44.0 h1:hqxVTu/GtBF+vJ8d1fzW7fRxZFvgoDjWcxwwCaFDYpU=
|
||||
go.opentelemetry.io/otel/exporters/stdout/stdoutmetric v1.44.0/go.mod h1:z5fVEF4X5v0ESvlJqBrrFlBVoj5EQuefZpzsu7R+x5Q=
|
||||
go.opentelemetry.io/otel/exporters/zipkin v1.45.0 h1:KN3btaILMTxR4QDHVGAO87lq5ButzK7l+kIfLuxQ1oA=
|
||||
go.opentelemetry.io/otel/exporters/zipkin v1.45.0/go.mod h1:yNcodmUclM4InyWoOwX/YW4Jri0Gj5FWAlM+NqCrtqY=
|
||||
go.opentelemetry.io/otel/metric v1.45.0 h1:7Eg1uH7CJ5cXv9is6tnBe1FI6rj1nwUdbFypRm3br/M=
|
||||
go.opentelemetry.io/otel/metric v1.45.0/go.mod h1:HAPbm1nd3p1PmFH7v2dR+6BjXxw+Lq4a2+pndMAm08s=
|
||||
go.opentelemetry.io/otel/metric/x v0.67.0 h1:PcicCNZFkZ4bXfSooXdo3WN7RBOVOtjVdo1wD358Uns=
|
||||
go.opentelemetry.io/otel/metric/x v0.67.0/go.mod h1:FBjCWZe6wgcqxcMtjdGiClDKXb2YxxXii0CXftE4QtI=
|
||||
go.opentelemetry.io/otel/sdk v1.45.0 h1:4VVSMgQ83dUgW2aoX5f6JgLvHwIvzcuLnF9lUdCSpCw=
|
||||
go.opentelemetry.io/otel/sdk v1.45.0/go.mod h1:Sr40LgXV7DsKMMJMKOhUWOgMWTfAaqvm2kF0g7ilwuA=
|
||||
go.opentelemetry.io/otel/sdk/metric v1.45.0 h1:oVFszMfyj1Am6s24Vtc7wBb8BKLcwepJjNEYILuiE3o=
|
||||
go.opentelemetry.io/otel/sdk/metric v1.45.0/go.mod h1:vUWUxDZvu1WVRj8JA8S0AdhsPrZoDpA2DdZauIh4mDA=
|
||||
go.opentelemetry.io/otel/trace v1.45.0 h1:l/mP6Uv7oNO7/TblbhpbgMidxhq1uO/rPsikOyVhxag=
|
||||
go.opentelemetry.io/otel/trace v1.45.0/go.mod h1:qoJJA2xNMnxRrdISU/kLtfUH2wNeQbiv+jhs/CxI8bc=
|
||||
go.opentelemetry.io/otel/metric v1.46.0 h1:yBnkXvgV7AXFILZc5K6IZe/CBFF3OS7BJ8ov6/lj0K8=
|
||||
go.opentelemetry.io/otel/metric v1.46.0/go.mod h1:iPmdWqifKUdzziPkvvzIJXITl56fQx2mGM/DHLB3/2o=
|
||||
go.opentelemetry.io/otel/metric/x v0.68.0 h1:TA/cBT23D3MnxYPwHL7YFOdYGdx0A0v+s7Mzotpd1dU=
|
||||
go.opentelemetry.io/otel/metric/x v0.68.0/go.mod h1:agudOmvWhwUTjgibWDzxD2PoWYnpw5Ht5jISYOD2Hd4=
|
||||
go.opentelemetry.io/otel/sdk v1.46.0 h1:h5CNQQjEbuQXY/JfZtgt3i7HVFV3aHPO2OAwO2eTYPI=
|
||||
go.opentelemetry.io/otel/sdk v1.46.0/go.mod h1:GAERFXFt5SYCEB+YiKUbMBeza6UaDH7GmGOZEfh2gSM=
|
||||
go.opentelemetry.io/otel/sdk/metric v1.46.0 h1:0piZ26EG4RBfebb2jhDH6ERCYHoVWduc3kLgPCwSnSE=
|
||||
go.opentelemetry.io/otel/sdk/metric v1.46.0/go.mod h1:I1PbKrdVc8Qu8HYVDNtqVIwLwjNrhsV/uFuxfwg8mO4=
|
||||
go.opentelemetry.io/otel/trace v1.46.0 h1:OULy7ccdJnZtJ0UDYFOIGaCmiWzJ8Vi2G/Rsu60qs1c=
|
||||
go.opentelemetry.io/otel/trace v1.46.0/go.mod h1:J7GAXweO77XSFkB/rmAqk9D6ihszhFjLU+d9WuUxDLI=
|
||||
go.opentelemetry.io/proto/otlp v0.7.0/go.mod h1:PqfVotwruBrMGOCsRd/89rSnXhoiJIqeYNgFYFoEGnI=
|
||||
go.opentelemetry.io/proto/otlp v0.15.0/go.mod h1:H7XAot3MsfNsj7EXtrA2q5xSNQ10UqI405h3+duxN4U=
|
||||
go.opentelemetry.io/proto/otlp v0.19.0/go.mod h1:H7XAot3MsfNsj7EXtrA2q5xSNQ10UqI405h3+duxN4U=
|
||||
@@ -2395,7 +2337,6 @@ golang.org/x/sys v0.0.0-20191001151750-bb3f8db39f24/go.mod h1:h1NjWce9XRLGQEsW7w
|
||||
golang.org/x/sys v0.0.0-20191026070338-33540a1f6037/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||
golang.org/x/sys v0.0.0-20191204072324-ce4227a45e2e/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||
golang.org/x/sys v0.0.0-20191228213918-04cbcbbfeed8/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||
golang.org/x/sys v0.0.0-20200106162015-b016eb3dc98e/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||
golang.org/x/sys v0.0.0-20200113162924-86b910548bc1/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||
golang.org/x/sys v0.0.0-20200116001909-b77594299b42/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||
golang.org/x/sys v0.0.0-20200122134326-e047566fdf82/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||
@@ -2409,8 +2350,6 @@ golang.org/x/sys v0.0.0-20200501052902-10377860bb8e/go.mod h1:h1NjWce9XRLGQEsW7w
|
||||
golang.org/x/sys v0.0.0-20200511232937-7e40ca221e25/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||
golang.org/x/sys v0.0.0-20200515095857-1151b9dac4a9/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||
golang.org/x/sys v0.0.0-20200523222454-059865788121/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||
golang.org/x/sys v0.0.0-20200615200032-f1bc736245b1/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||
golang.org/x/sys v0.0.0-20200625212154-ddb9806d33ae/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||
golang.org/x/sys v0.0.0-20200803210538-64077c9b5642/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||
golang.org/x/sys v0.0.0-20200905004654-be1d3432aa8f/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||
golang.org/x/sys v0.0.0-20200930185726-fdedc70b468f/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||
@@ -2431,7 +2370,6 @@ golang.org/x/sys v0.0.0-20210423082822-04245dca01da/go.mod h1:h1NjWce9XRLGQEsW7w
|
||||
golang.org/x/sys v0.0.0-20210423185535-09eb48e85fd7/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
|
||||
golang.org/x/sys v0.0.0-20210510120138-977fb7262007/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.0.0-20210514084401-e8d321eab015/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.0.0-20210603081109-ebe580a85c40/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.0.0-20210603125802-9665404d3644/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.0.0-20210615035016-665e8c7367d1/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.0.0-20210616094352-59db8d763f22/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
@@ -2442,7 +2380,6 @@ golang.org/x/sys v0.0.0-20210823070655-63515b42dcdf/go.mod h1:oPkhp1MJrh7nUepCBc
|
||||
golang.org/x/sys v0.0.0-20210908233432-aa78b53d3365/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.0.0-20210927094055-39ccf1dd6fa6/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.0.0-20211007075335-d3039528d8ac/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.0.0-20211013075003-97ac67df715c/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.0.0-20211019181941-9d821ace8654/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.0.0-20211025201205-69cdffdb9359/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.0.0-20211117180635-dee7805ff2e1/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
@@ -2452,7 +2389,6 @@ golang.org/x/sys v0.0.0-20211216021012-1d35b9e2eb4e/go.mod h1:oPkhp1MJrh7nUepCBc
|
||||
golang.org/x/sys v0.0.0-20220128215802-99c3d69c2c27/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.0.0-20220209214540-3681064d5158/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.0.0-20220227234510-4e6760a101f9/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.0.0-20220319134239-a9b59b0215f8/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.0.0-20220328115105-d36c6a25d886/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.0.0-20220408201424-a24fb2fb8a0f/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.0.0-20220412211240-33da011f77ad/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
@@ -2480,8 +2416,6 @@ golang.org/x/sys v0.13.0/go.mod h1:oPkhp1MJrh7nUepCBck5+mAzfO9JrbApNNgaTdGDITg=
|
||||
golang.org/x/sys v0.48.0 h1:bbX/i/6MgT9BVLM9RT1thmxL04yeTAhbEz4SyadbXoo=
|
||||
golang.org/x/sys v0.48.0/go.mod h1:hNLxWAXmnKAxqDtdwIYC4bM9oQPEecfsnNMuSxOs3og=
|
||||
golang.org/x/term v0.0.0-20201126162022-7de9c90e9dd1/go.mod h1:bj7SfCRtBDWHUb9snDiAeCFNEtKQo2Wmx5Cou7ajbmo=
|
||||
golang.org/x/term v0.0.0-20210220032956-6a3ed077a48d/go.mod h1:bj7SfCRtBDWHUb9snDiAeCFNEtKQo2Wmx5Cou7ajbmo=
|
||||
golang.org/x/term v0.0.0-20210615171337-6886f2dfbf5b/go.mod h1:jbD1KX2456YbFQfuXm/mYQcufACuNUgVhRMnK/tPxf8=
|
||||
golang.org/x/term v0.0.0-20210927222741-03fcf44c2211/go.mod h1:jbD1KX2456YbFQfuXm/mYQcufACuNUgVhRMnK/tPxf8=
|
||||
golang.org/x/term v0.2.0/go.mod h1:TVmDHMZPmdnySmBfhjOoOdhjzdE1h4u1VwSiw2l1Nuc=
|
||||
golang.org/x/term v0.3.0/go.mod h1:q750SLmJuPmVoN1blW3UFBPREJfb1KmY3vwxfr+nFDA=
|
||||
@@ -2518,8 +2452,8 @@ golang.org/x/time v0.0.0-20190308202827-9d24e82272b4/go.mod h1:tRJNPiyCQ0inRvYxb
|
||||
golang.org/x/time v0.0.0-20191024005414-555d28b269f0/go.mod h1:tRJNPiyCQ0inRvYxbN9jk5I+vvW/OXSQhTDSoE431IQ=
|
||||
golang.org/x/time v0.0.0-20220922220347-f3bd1da661af/go.mod h1:tRJNPiyCQ0inRvYxbN9jk5I+vvW/OXSQhTDSoE431IQ=
|
||||
golang.org/x/time v0.1.0/go.mod h1:tRJNPiyCQ0inRvYxbN9jk5I+vvW/OXSQhTDSoE431IQ=
|
||||
golang.org/x/time v0.15.0 h1:bbrp8t3bGUeFOx08pvsMYRTCVSMk89u4tKbNOZbp88U=
|
||||
golang.org/x/time v0.15.0/go.mod h1:Y4YMaQmXwGQZoFaVFk4YpCt4FLQMYKZe9oeV/f4MSno=
|
||||
golang.org/x/time v0.16.0 h1:vMb6ptszcQMkcwiRTAuNNU50gom6++Q/6gY2hDM6VDE=
|
||||
golang.org/x/time v0.16.0/go.mod h1:rVKOqvZeKvrDKTQiAHJ7wmwP0RzleSphoEA9RcdLA0s=
|
||||
golang.org/x/tools v0.0.0-20180525024113-a5b4c53f6e8b/go.mod h1:n7NCudcB/nEzxVGmLbDWY5pfWTLqBcC2KZ6jyYvM4mQ=
|
||||
golang.org/x/tools v0.0.0-20180917221912-90fa682c2a6e/go.mod h1:n7NCudcB/nEzxVGmLbDWY5pfWTLqBcC2KZ6jyYvM4mQ=
|
||||
golang.org/x/tools v0.0.0-20190114222345-bf090417da8b/go.mod h1:n7NCudcB/nEzxVGmLbDWY5pfWTLqBcC2KZ6jyYvM4mQ=
|
||||
@@ -2901,7 +2835,6 @@ gopkg.in/yaml.v2 v2.2.3/go.mod h1:hI93XBmqTisBFMUTm0b8Fm+jr3Dg1NNxqwp+5A1VGuI=
|
||||
gopkg.in/yaml.v2 v2.2.4/go.mod h1:hI93XBmqTisBFMUTm0b8Fm+jr3Dg1NNxqwp+5A1VGuI=
|
||||
gopkg.in/yaml.v2 v2.2.5/go.mod h1:hI93XBmqTisBFMUTm0b8Fm+jr3Dg1NNxqwp+5A1VGuI=
|
||||
gopkg.in/yaml.v2 v2.2.8/go.mod h1:hI93XBmqTisBFMUTm0b8Fm+jr3Dg1NNxqwp+5A1VGuI=
|
||||
gopkg.in/yaml.v2 v2.3.0/go.mod h1:hI93XBmqTisBFMUTm0b8Fm+jr3Dg1NNxqwp+5A1VGuI=
|
||||
gopkg.in/yaml.v2 v2.4.0 h1:D8xgwECY7CYvx+Y2n4sBz93Jn9JRvxdiyyo8CTfuKaY=
|
||||
gopkg.in/yaml.v2 v2.4.0/go.mod h1:RDklbk79AGWmwhnvt/jBztapEOGDOx6ZbXqjP6csGnQ=
|
||||
gopkg.in/yaml.v3 v3.0.0-20200313102051-9f266ea9e77c/go.mod h1:K4uyk7z7BCEPqu6E+C64Yfv1cQ7kz7rIZviUmN+EgEM=
|
||||
|
||||
@@ -3,4 +3,4 @@ description: SeaweedFS
|
||||
name: seaweedfs
|
||||
appVersion: "4.48"
|
||||
# Dev note: Trigger a helm chart release by `git tag -a helm-<version>`
|
||||
version: 4.48.0
|
||||
version: 4.48.2
|
||||
|
||||
@@ -27,6 +27,19 @@ so your deployment will be spread/HA.
|
||||
* cert config exists and can be enabled, but not been tested, requires cert-manager to be installed.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
The chart's templates render with Helm 3.16.3 and newer versions tested in CI.
|
||||
Earlier Helm versions are not covered by this chart's compatibility checks.
|
||||
When JWT signing is enabled, a live upgrade reuses keys from the existing
|
||||
security ConfigMap (including the legacy ConfigMap name); a key absent from
|
||||
that ConfigMap is generated. Plain `helm template` does not read cluster state;
|
||||
use an install and upgrade against a cluster to check key persistence.
|
||||
|
||||
The key reader supports single-line quoted TOML strings in JWT sections,
|
||||
including indented assignments, quoted key names, and bare or simply quoted
|
||||
section-name segments. An unsupported value or quoted header fails the upgrade
|
||||
rather than silently rotating a key.
|
||||
|
||||
### Database
|
||||
|
||||
leveldb is the default database, this supports multiple filer replicas that will [sync automatically](https://github.com/seaweedfs/seaweedfs/wiki/Filer-Store-Replication), with some [limitations](https://github.com/seaweedfs/seaweedfs/wiki/Filer-Store-Replication#limitation).
|
||||
@@ -570,6 +583,23 @@ Two things worth knowing before you turn this on:
|
||||
|
||||
The DNS selectors default to CoreDNS as kubeadm, kind and the managed offerings from AWS, Google and Azure install it. On OpenShift, override `egress.dnsNamespaceSelector` and `egress.dnsPodSelector` to match `openshift-dns`; see the comment in `values.yaml`.
|
||||
|
||||
## Pod and container security contexts
|
||||
|
||||
Pod and container security contexts are configurable independently for every built-in workload and remain empty by default for backwards compatibility. The examples in `values.yaml` show how to enable a `RuntimeDefault` seccomp profile, disable privilege escalation and privileged mode, drop all Linux capabilities, and use a read-only root filesystem.
|
||||
|
||||
SeaweedFS uses `/tmp` for Unix sockets, temporary uploads, worker task files, and other runtime data. When `readOnlyRootFilesystem` is enabled for a built-in component, the chart mounts a writable `emptyDir` at `/tmp` for its chart-managed containers. Its optional size limit can be configured globally:
|
||||
|
||||
```yaml
|
||||
global:
|
||||
seaweedfs:
|
||||
tmpDir:
|
||||
sizeLimit: 1Gi
|
||||
```
|
||||
|
||||
The chart does not enable `runAsNonRoot` by default because its default `hostPath` storage may be owned by root. To enforce the Kubernetes `restricted` Pod Security Standard, use storage that is writable by a non-root user and configure `runAsNonRoot` or use the OpenShift overrides below.
|
||||
|
||||
Security contexts configured for a component also apply to the chart-managed helper containers for that component. User-provided init containers and sidecars must define their own container security context and writable mounts.
|
||||
|
||||
## OpenShift Support
|
||||
|
||||
SeaweedFS can be deployed on OpenShift or any cluster enforcing the Kubernetes "restricted" Pod Security Standard. By default, OpenShift blocks containers that run as root or use `hostPath` volumes.
|
||||
@@ -578,6 +608,7 @@ To deploy on OpenShift, use the provided `openshift-values.yaml` which overrides
|
||||
1. Use `PersistentVolumeClaims` instead of `hostPath`.
|
||||
2. Enable `runAsNonRoot` and omit hardcoded UIDs to allow OpenShift to assign valid UIDs automatically.
|
||||
3. Apply appropriate `seccompProfile` and drop capabilities.
|
||||
4. Use a read-only root filesystem with writable temporary storage at `/tmp`.
|
||||
|
||||
Usage:
|
||||
```bash
|
||||
|
||||
@@ -15,6 +15,7 @@
|
||||
# automatically assign a valid UID from the namespace's allocated range.
|
||||
# 3. Dropping all Linux capabilities and setting allowPrivilegeEscalation: false
|
||||
# 4. Enabling RuntimeDefault seccompProfile
|
||||
# 5. Using a read-only root filesystem with writable temporary storage
|
||||
#
|
||||
# Usage:
|
||||
# helm install seaweedfs seaweedfs/seaweedfs \
|
||||
@@ -49,6 +50,7 @@ master:
|
||||
containerSecurityContext:
|
||||
enabled: true
|
||||
allowPrivilegeEscalation: false
|
||||
readOnlyRootFilesystem: true
|
||||
capabilities:
|
||||
drop: ["ALL"]
|
||||
runAsNonRoot: true
|
||||
@@ -76,6 +78,7 @@ volume:
|
||||
containerSecurityContext:
|
||||
enabled: true
|
||||
allowPrivilegeEscalation: false
|
||||
readOnlyRootFilesystem: true
|
||||
capabilities:
|
||||
drop: ["ALL"]
|
||||
runAsNonRoot: true
|
||||
@@ -101,6 +104,7 @@ filer:
|
||||
containerSecurityContext:
|
||||
enabled: true
|
||||
allowPrivilegeEscalation: false
|
||||
readOnlyRootFilesystem: true
|
||||
capabilities:
|
||||
drop: ["ALL"]
|
||||
runAsNonRoot: true
|
||||
@@ -125,6 +129,7 @@ s3:
|
||||
containerSecurityContext:
|
||||
enabled: true
|
||||
allowPrivilegeEscalation: false
|
||||
readOnlyRootFilesystem: true
|
||||
capabilities:
|
||||
drop: ["ALL"]
|
||||
runAsNonRoot: true
|
||||
|
||||
@@ -192,6 +192,7 @@ spec:
|
||||
{{ $arg }}{{- if lt $index (sub (len $.Values.admin.extraArgs) 1) }} \{{ end }}
|
||||
{{- end }}
|
||||
volumeMounts:
|
||||
{{- include "seaweedfs.tmpDirVolumeMount" (list . .Values.admin.containerSecurityContext (tpl (.Values.admin.extraVolumeMounts | default "") .) (tpl (.Values.admin.extraVolumes | default "") .)) | nindent 12 }}
|
||||
{{- if or (eq .Values.admin.data.type "hostPath") (eq .Values.admin.data.type "persistentVolumeClaim") (eq .Values.admin.data.type "emptyDir") (eq .Values.admin.data.type "existingClaim") }}
|
||||
- name: admin-data
|
||||
mountPath: /data
|
||||
@@ -267,6 +268,7 @@ spec:
|
||||
{{- include "seaweedfs.tplvalues.render" (dict "value" .Values.admin.sidecars "context" $) | nindent 8 }}
|
||||
{{- end }}
|
||||
volumes:
|
||||
{{- include "seaweedfs.tmpDirVolume" (list . .Values.admin.containerSecurityContext (tpl (.Values.admin.extraVolumeMounts | default "") .) (tpl (.Values.admin.extraVolumes | default "") .) false) | nindent 8 }}
|
||||
{{- if eq .Values.admin.data.type "hostPath" }}
|
||||
- name: admin-data
|
||||
hostPath:
|
||||
|
||||
@@ -306,6 +306,7 @@ spec:
|
||||
{{- end }}
|
||||
|
||||
volumeMounts:
|
||||
{{- include "seaweedfs.tmpDirVolumeMount" (list . .Values.allInOne.containerSecurityContext (tpl (.Values.allInOne.extraVolumeMounts | default "") .) (tpl (.Values.allInOne.extraVolumes | default "") .)) | nindent 12 }}
|
||||
- name: data
|
||||
mountPath: /data
|
||||
{{- if and .Values.allInOne.s3.enabled (or .Values.allInOne.s3.enableAuth .Values.s3.enableAuth .Values.filer.s3.enableAuth) }}
|
||||
@@ -433,6 +434,7 @@ spec:
|
||||
{{- include "seaweedfs.tplvalues.render" (dict "value" .Values.allInOne.sidecars "context" $) | nindent 8 }}
|
||||
{{- end }}
|
||||
volumes:
|
||||
{{- include "seaweedfs.tmpDirVolume" (list . .Values.allInOne.containerSecurityContext (tpl (.Values.allInOne.extraVolumeMounts | default "") .) (tpl (.Values.allInOne.extraVolumes | default "") .) false) | nindent 8 }}
|
||||
{{- include "seaweedfs.licenseVolume" . | nindent 8 }}
|
||||
- name: data
|
||||
{{- if eq .Values.allInOne.data.type "hostPath" }}
|
||||
|
||||
@@ -28,6 +28,13 @@ rules:
|
||||
- "update"
|
||||
- "create"
|
||||
- "delete"
|
||||
- apiGroups: ["objectstorage.k8s.io"]
|
||||
resources:
|
||||
- "bucketclasses"
|
||||
verbs:
|
||||
- "get"
|
||||
- "list"
|
||||
- "watch"
|
||||
- apiGroups: ["coordination.k8s.io"]
|
||||
resources: ["leases"]
|
||||
verbs:
|
||||
@@ -61,7 +68,7 @@ metadata:
|
||||
app.kubernetes.io/instance: {{ .Release.Name }}
|
||||
subjects:
|
||||
- kind: ServiceAccount
|
||||
name: {{ .Values.global.seaweedfs.serviceAccountName }}-objectstorage-provisioner
|
||||
name: {{ include "seaweedfs.serviceAccountName" . }}-objectstorage-provisioner
|
||||
namespace: {{ .Release.Namespace }}
|
||||
roleRef:
|
||||
kind: ClusterRole
|
||||
|
||||
@@ -61,7 +61,7 @@ spec:
|
||||
schedulerName: {{ .Values.cosi.schedulerName | quote }}
|
||||
{{- end }}
|
||||
enableServiceLinks: false
|
||||
serviceAccountName: {{ include "seaweedfs.componentName" (list . "objectstorage-provisioner") }}
|
||||
serviceAccountName: {{ include "seaweedfs.serviceAccountName" . }}-objectstorage-provisioner
|
||||
{{- if .Values.cosi.initContainers }}
|
||||
initContainers:
|
||||
{{ tpl .Values.cosi.initContainers . | nindent 8 | trim }}
|
||||
@@ -113,6 +113,7 @@ spec:
|
||||
{{- end -}}
|
||||
{{- end }}
|
||||
volumeMounts:
|
||||
{{- include "seaweedfs.tmpDirVolumeMount" (list . .Values.cosi.containerSecurityContext (tpl (.Values.cosi.extraVolumeMounts | default "") .) (tpl (.Values.cosi.extraVolumes | default "") .)) | nindent 12 }}
|
||||
- mountPath: /var/lib/cosi
|
||||
name: socket
|
||||
{{- if .Values.cosi.enableAuth }}
|
||||
@@ -148,6 +149,9 @@ spec:
|
||||
resources:
|
||||
{{- toYaml . | nindent 12 }}
|
||||
{{- end }}
|
||||
{{- if .Values.cosi.containerSecurityContext.enabled }}
|
||||
securityContext: {{- omit .Values.cosi.containerSecurityContext "enabled" | toYaml | nindent 12 }}
|
||||
{{- end }}
|
||||
- name: seaweedfs-cosi-sidecar
|
||||
image: "{{ .Values.cosi.sidecar.image }}"
|
||||
imagePullPolicy: {{ default "IfNotPresent" .Values.global.seaweedfs.imagePullPolicy }}
|
||||
@@ -159,6 +163,7 @@ spec:
|
||||
fieldRef:
|
||||
fieldPath: metadata.namespace
|
||||
volumeMounts:
|
||||
{{- include "seaweedfs.tmpDirVolumeMount" (list . .Values.cosi.containerSecurityContext "" "") | nindent 12 }}
|
||||
- mountPath: /var/lib/cosi
|
||||
name: socket
|
||||
{{- with .Values.cosi.sidecar.resources }}
|
||||
@@ -172,6 +177,7 @@ spec:
|
||||
{{- include "seaweedfs.tplvalues.render" (dict "value" .Values.cosi.sidecars "context" $) | nindent 8 }}
|
||||
{{- end }}
|
||||
volumes:
|
||||
{{- include "seaweedfs.tmpDirVolume" (list . .Values.cosi.containerSecurityContext (tpl (.Values.cosi.extraVolumeMounts | default "") .) (tpl (.Values.cosi.extraVolumes | default "") .) true) | nindent 8 }}
|
||||
- name: socket
|
||||
emptyDir: {}
|
||||
{{- if .Values.cosi.enableAuth }}
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
apiVersion: v1
|
||||
kind: ServiceAccount
|
||||
metadata:
|
||||
name: {{ .Values.global.seaweedfs.serviceAccountName }}-objectstorage-provisioner
|
||||
name: {{ include "seaweedfs.serviceAccountName" . }}-objectstorage-provisioner
|
||||
namespace: {{ .Release.Namespace }}
|
||||
labels:
|
||||
app.kubernetes.io/name: {{ template "seaweedfs.name" . }}
|
||||
|
||||
@@ -229,6 +229,7 @@ spec:
|
||||
{{ . }} \
|
||||
{{- end }}
|
||||
volumeMounts:
|
||||
{{- include "seaweedfs.tmpDirVolumeMount" (list . .Values.filer.containerSecurityContext (tpl (.Values.filer.extraVolumeMounts | default "") .) (tpl (.Values.filer.extraVolumes | default "") .)) | nindent 12 }}
|
||||
{{- if (or (eq .Values.filer.logs.type "hostPath") (eq .Values.filer.logs.type "persistentVolumeClaim") (eq .Values.filer.logs.type "emptyDir")) }}
|
||||
- name: seaweedfs-filer-log-volume
|
||||
mountPath: "/logs/"
|
||||
@@ -337,6 +338,7 @@ spec:
|
||||
{{- include "seaweedfs.tplvalues.render" (dict "value" .Values.filer.sidecars "context" $) | nindent 8 }}
|
||||
{{- end }}
|
||||
volumes:
|
||||
{{- include "seaweedfs.tmpDirVolume" (list . .Values.filer.containerSecurityContext (tpl (.Values.filer.extraVolumeMounts | default "") .) (tpl (.Values.filer.extraVolumes | default "") .) false) | nindent 8 }}
|
||||
{{- if eq .Values.filer.logs.type "hostPath" }}
|
||||
- name: seaweedfs-filer-log-volume
|
||||
hostPath:
|
||||
|
||||
@@ -179,6 +179,7 @@ spec:
|
||||
{{ . }} \
|
||||
{{- end }}
|
||||
volumeMounts:
|
||||
{{- include "seaweedfs.tmpDirVolumeMount" (list . .Values.master.containerSecurityContext (tpl (.Values.master.extraVolumeMounts | default "") .) (tpl (.Values.master.extraVolumes | default "") .)) | nindent 12 }}
|
||||
- name : data-{{ .Release.Namespace }}
|
||||
mountPath: /data
|
||||
{{- if or (eq .Values.master.logs.type "hostPath") (eq .Values.master.logs.type "persistentVolumeClaim") (eq .Values.master.logs.type "emptyDir") }}
|
||||
@@ -258,6 +259,7 @@ spec:
|
||||
{{- include "seaweedfs.tplvalues.render" (dict "value" .Values.master.sidecars "context" $) | nindent 8 }}
|
||||
{{- end }}
|
||||
volumes:
|
||||
{{- include "seaweedfs.tmpDirVolume" (list . .Values.master.containerSecurityContext (tpl (.Values.master.extraVolumeMounts | default "") .) (tpl (.Values.master.extraVolumes | default "") .) false) | nindent 8 }}
|
||||
{{- include "seaweedfs.licenseVolume" . | nindent 8 }}
|
||||
{{- if eq .Values.master.logs.type "hostPath" }}
|
||||
- name: seaweedfs-master-log-volume
|
||||
|
||||
@@ -17,6 +17,10 @@ metadata:
|
||||
{{- end }}
|
||||
spec:
|
||||
replicas: {{ .Values.s3.replicas }}
|
||||
{{- with .Values.s3.updateStrategy }}
|
||||
strategy:
|
||||
{{- toYaml . | nindent 4 }}
|
||||
{{- end }}
|
||||
selector:
|
||||
matchLabels:
|
||||
app.kubernetes.io/name: {{ template "seaweedfs.name" . }}
|
||||
@@ -59,7 +63,7 @@ spec:
|
||||
{{ tpl .Values.s3.tolerations . | nindent 8 | trim }}
|
||||
{{- end }}
|
||||
{{- include "seaweedfs.imagePullSecrets" . | nindent 6 }}
|
||||
terminationGracePeriodSeconds: 10
|
||||
terminationGracePeriodSeconds: {{ .Values.s3.terminationGracePeriodSeconds }}
|
||||
{{- if .Values.s3.priorityClassName }}
|
||||
priorityClassName: {{ .Values.s3.priorityClassName | quote }}
|
||||
{{- end }}
|
||||
@@ -157,6 +161,7 @@ spec:
|
||||
{{ . }} \
|
||||
{{- end }}
|
||||
volumeMounts:
|
||||
{{- include "seaweedfs.tmpDirVolumeMount" (list . .Values.s3.containerSecurityContext (tpl (.Values.s3.extraVolumeMounts | default "") .) (tpl (.Values.s3.extraVolumes | default "") .)) | nindent 12 }}
|
||||
{{- if or (eq .Values.s3.logs.type "hostPath") (eq .Values.s3.logs.type "emptyDir") }}
|
||||
- name: logs
|
||||
mountPath: "/logs/"
|
||||
@@ -234,6 +239,10 @@ spec:
|
||||
failureThreshold: {{ .Values.s3.livenessProbe.failureThreshold }}
|
||||
timeoutSeconds: {{ .Values.s3.livenessProbe.timeoutSeconds }}
|
||||
{{- end }}
|
||||
{{- with .Values.s3.lifecycle }}
|
||||
lifecycle:
|
||||
{{- toYaml . | nindent 12 }}
|
||||
{{- end }}
|
||||
{{- with .Values.s3.resources }}
|
||||
resources:
|
||||
{{- toYaml . | nindent 12 }}
|
||||
@@ -245,6 +254,7 @@ spec:
|
||||
{{- include "seaweedfs.tplvalues.render" (dict "value" .Values.s3.sidecars "context" $) | nindent 8 }}
|
||||
{{- end }}
|
||||
volumes:
|
||||
{{- include "seaweedfs.tmpDirVolume" (list . .Values.s3.containerSecurityContext (tpl (.Values.s3.extraVolumeMounts | default "") .) (tpl (.Values.s3.extraVolumes | default "") .) false) | nindent 8 }}
|
||||
{{- if .Values.s3.enableAuth }}
|
||||
- name: config-users
|
||||
secret:
|
||||
|
||||
@@ -170,6 +170,7 @@ spec:
|
||||
-userStoreFile=/etc/sw/seaweedfs_sftp_config \
|
||||
-filer={{ include "seaweedfs.componentName" (list . "filer-client") }}.{{ .Release.Namespace }}:{{ .Values.filer.port }}
|
||||
volumeMounts:
|
||||
{{- include "seaweedfs.tmpDirVolumeMount" (list . .Values.sftp.containerSecurityContext (tpl (.Values.sftp.extraVolumeMounts | default "") .) (tpl (.Values.sftp.extraVolumes | default "") .)) | nindent 12 }}
|
||||
{{- if or (eq .Values.sftp.logs.type "hostPath") (eq .Values.sftp.logs.type "emptyDir") }}
|
||||
- name: logs
|
||||
mountPath: "/logs/"
|
||||
@@ -249,6 +250,7 @@ spec:
|
||||
{{- include "seaweedfs.tplvalues.render" (dict "value" .Values.sftp.sidecars "context" $) | nindent 8 }}
|
||||
{{- end }}
|
||||
volumes:
|
||||
{{- include "seaweedfs.tmpDirVolume" (list . .Values.sftp.containerSecurityContext (tpl (.Values.sftp.extraVolumeMounts | default "") .) (tpl (.Values.sftp.extraVolumes | default "") .) false) | nindent 8 }}
|
||||
{{- if .Values.sftp.enableAuth }}
|
||||
- name: config-users
|
||||
secret:
|
||||
|
||||
@@ -76,6 +76,43 @@ Inject extra environment vars in the format key:value, if populated
|
||||
{{- end }}
|
||||
{{- end -}}
|
||||
|
||||
{{/*
|
||||
Writable temporary directory for containers using a read-only root filesystem.
|
||||
Input: list of the root context, the component container security context, the
|
||||
rendered extraVolumeMounts and extraVolumes, and whether the pod has secondary
|
||||
chart-managed containers that mount seaweedfs-tmp. A user-supplied /tmp mount
|
||||
only covers the main container, so the volume is still emitted for secondaries;
|
||||
a user-supplied seaweedfs-tmp volume is reused rather than duplicated.
|
||||
*/}}
|
||||
{{- define "seaweedfs.tmpDirCovered" -}}
|
||||
{{- regexMatch `(?m)^\s*-?\s*mountPath:\s*['"]?/tmp/?['"]?\s*(#.*)?$` (index . 2) -}}
|
||||
{{- end -}}
|
||||
|
||||
{{- define "seaweedfs.tmpDirVolume" -}}
|
||||
{{- $root := index . 0 -}}
|
||||
{{- $securityContext := index . 1 -}}
|
||||
{{- if and $securityContext.enabled $securityContext.readOnlyRootFilesystem
|
||||
(or (index . 4) (ne (include "seaweedfs.tmpDirCovered" .) "true"))
|
||||
(not (regexMatch `(?m)^\s*-?\s*name:\s*['"]?seaweedfs-tmp['"]?\s*(#.*)?$` (index . 3))) }}
|
||||
- name: seaweedfs-tmp
|
||||
{{- with $root.Values.global.seaweedfs.tmpDir.sizeLimit }}
|
||||
emptyDir:
|
||||
sizeLimit: {{ . | quote }}
|
||||
{{- else }}
|
||||
emptyDir: {}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
{{- end -}}
|
||||
|
||||
{{- define "seaweedfs.tmpDirVolumeMount" -}}
|
||||
{{- $securityContext := index . 1 -}}
|
||||
{{- if and $securityContext.enabled $securityContext.readOnlyRootFilesystem
|
||||
(ne (include "seaweedfs.tmpDirCovered" .) "true") }}
|
||||
- name: seaweedfs-tmp
|
||||
mountPath: /tmp
|
||||
{{- end }}
|
||||
{{- end -}}
|
||||
|
||||
{{/* Whether the mysql filer store is selected; a flag the chart cannot read counts as selected. */}}
|
||||
{{- define "seaweedfs.filer.mysqlEnabled" -}}
|
||||
{{- $merged := dict -}}
|
||||
@@ -407,6 +444,46 @@ true
|
||||
{{- end -}}
|
||||
{{- end -}}
|
||||
|
||||
{{/* Read a JWT key from the chart's existing security.toml without fromToml
|
||||
(which requires Helm >=3.17). Args: (list "<section>" $raw).
|
||||
Return the single-line TOML string token, including its quotes, so escapes
|
||||
and explicitly empty values survive unchanged. No output means absent.
|
||||
Accept bare or simply quoted dotted header segments. Fail on other quoted
|
||||
headers rather than risk rotating a key hidden by unsupported syntax.
|
||||
This reads the chart's section/key layout, not arbitrary TOML syntax. */}}
|
||||
{{- define "seaweedfs.existingTomlKey" -}}
|
||||
{{- $section := index . 0 -}}
|
||||
{{- $raw := index . 1 -}}
|
||||
{{- $parts := list -}}
|
||||
{{- range $part := splitList "." $section -}}
|
||||
{{- $escaped := regexQuoteMeta $part -}}
|
||||
{{- $parts = append $parts (printf `(?:%s|"%s"|'%s')` $escaped $escaped $escaped) -}}
|
||||
{{- end -}}
|
||||
{{- $header := printf `^\[[ \t]*%s[ \t]*\][ \t]*(#.*)?$` (join `[ \t]*\.[ \t]*` $parts) -}}
|
||||
{{- $segment := `(?:[A-Za-z0-9_-]+|"[^"\\]*"|'[^']*')` -}}
|
||||
{{- $simpleHeader := printf `^\[[ \t]*%s(?:[ \t]*\.[ \t]*%s)*[ \t]*\][ \t]*(#.*)?$` $segment $segment -}}
|
||||
{{- $assignment := `^(key|"key"|'key')[ \t]*=[ \t]*` -}}
|
||||
{{- $string := `"([^"\\]|\\.)*"|'[^']*'` -}}
|
||||
{{- $active := false -}}
|
||||
{{- $key := "" -}}
|
||||
{{- range $rawLine := splitList "\n" $raw -}}
|
||||
{{- $line := trim $rawLine -}}
|
||||
{{- if hasPrefix "[" $line -}}
|
||||
{{- if and (regexMatch `^\[[^]]*["']` $line) (not (regexMatch $simpleHeader $line)) (eq $key "") -}}
|
||||
{{- fail (printf "security.toml has an unsupported quoted section header; refusing to replace [%s].key" $section) -}}
|
||||
{{- end -}}
|
||||
{{- $active = regexMatch $header $line -}}
|
||||
{{- else if and $active (eq $key "") (regexMatch $assignment $line) -}}
|
||||
{{- $value := regexReplaceAll $assignment $line "" -}}
|
||||
{{- if not (regexMatch (printf `^(%s)[ \t]*(#.*)?$` $string) $value) -}}
|
||||
{{- fail (printf "security.toml [%s].key must be a single-line quoted TOML string; refusing to replace an existing key" $section) -}}
|
||||
{{- end -}}
|
||||
{{- $key = regexFind $string $value -}}
|
||||
{{- end -}}
|
||||
{{- end -}}
|
||||
{{- $key -}}
|
||||
{{- end -}}
|
||||
|
||||
{{/* True when the post-install bucket hook Job renders: an S3 endpoint, plus
|
||||
buckets to create on it. Read by the Job itself and by its NetworkPolicy,
|
||||
which has to appear exactly when the Job does - a Job without its policy
|
||||
@@ -467,8 +544,9 @@ true
|
||||
{{- $pvcName := printf "%s-%s-%s-%d" $dir.name $seaweedfsName $volumeName $e }}
|
||||
{{- $currentPVC := (lookup "v1" "PersistentVolumeClaim" $.Release.Namespace $pvcName) }}
|
||||
{{- if $currentPVC }}
|
||||
{{- $oldSize := include "seaweedfs.resource-quantity" $currentPVC.spec.resources.requests.storage }}
|
||||
{{- $newSize := include "seaweedfs.resource-quantity" $desiredSize }}
|
||||
{{- /* include returns a string such as "6.442450944e+10"; convert back to a number, or gt compares lexically */}}
|
||||
{{- $oldSize := include "seaweedfs.resource-quantity" $currentPVC.spec.resources.requests.storage | float64 }}
|
||||
{{- $newSize := include "seaweedfs.resource-quantity" $desiredSize | float64 }}
|
||||
{{- if gt $newSize $oldSize }}
|
||||
{{- $commands = append $commands (printf "kubectl patch pvc %s-%s-%s-%d -p '{\"spec\":{\"resources\":{\"requests\":{\"storage\":\"%s\"}}}}'" $dir.name $seaweedfsName $volumeName $e $desiredSize) }}
|
||||
{{- end }}
|
||||
|
||||
@@ -16,6 +16,12 @@
|
||||
{{- $clusterUpper := upper $clusterAlias }}
|
||||
{{- $clusterMasterKey := printf "WEED_CLUSTER_%s_MASTER" $clusterUpper }}
|
||||
{{- $clusterFilerKey := printf "WEED_CLUSTER_%s_FILER" $clusterUpper }}
|
||||
{{- $podSecurityContext := .Values.filer.podSecurityContext }}
|
||||
{{- $containerSecurityContext := .Values.filer.containerSecurityContext }}
|
||||
{{- if .Values.allInOne.enabled }}
|
||||
{{- $podSecurityContext = .Values.allInOne.podSecurityContext }}
|
||||
{{- $containerSecurityContext = .Values.allInOne.containerSecurityContext }}
|
||||
{{- end }}
|
||||
|
||||
{{- /* Check allInOne mode first */}}
|
||||
{{- if .Values.allInOne.enabled }}
|
||||
@@ -71,8 +77,8 @@ spec:
|
||||
app.kubernetes.io/component: bucket-hook
|
||||
spec:
|
||||
restartPolicy: Never
|
||||
{{- if .Values.filer.podSecurityContext.enabled }}
|
||||
securityContext: {{- omit .Values.filer.podSecurityContext "enabled" | toYaml | nindent 8 }}
|
||||
{{- if $podSecurityContext.enabled }}
|
||||
securityContext: {{- omit $podSecurityContext "enabled" | toYaml | nindent 8 }}
|
||||
{{- end }}
|
||||
{{- include "seaweedfs.imagePullSecrets" $ | nindent 6 }}
|
||||
containers:
|
||||
@@ -202,8 +208,9 @@ spec:
|
||||
/usr/bin/weed shell
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
{{- if or $enableAuth (include "seaweedfs.securityConfigEnabled" .) }}
|
||||
{{- if or (and $containerSecurityContext.enabled $containerSecurityContext.readOnlyRootFilesystem) $enableAuth (include "seaweedfs.securityConfigEnabled" .) }}
|
||||
volumeMounts:
|
||||
{{- include "seaweedfs.tmpDirVolumeMount" (list . $containerSecurityContext "" "") | nindent 10 }}
|
||||
{{- if $enableAuth }}
|
||||
- name: config-users
|
||||
mountPath: /etc/sw
|
||||
@@ -215,6 +222,14 @@ spec:
|
||||
mountPath: /etc/seaweedfs/security.toml
|
||||
subPath: security.toml
|
||||
{{- end }}
|
||||
{{- if .Values.global.seaweedfs.enableSecurity }}
|
||||
- name: ca-cert
|
||||
readOnly: true
|
||||
mountPath: /usr/local/share/ca-certificates/ca/
|
||||
- name: client-cert
|
||||
readOnly: true
|
||||
mountPath: /usr/local/share/ca-certificates/client/
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
ports:
|
||||
- containerPort: {{ .Values.master.port }}
|
||||
@@ -229,11 +244,12 @@ spec:
|
||||
resources:
|
||||
{{- toYaml . | nindent 10 }}
|
||||
{{- end }}
|
||||
{{- if .Values.filer.containerSecurityContext.enabled }}
|
||||
securityContext: {{- omit .Values.filer.containerSecurityContext "enabled" | toYaml | nindent 12 }}
|
||||
{{- if $containerSecurityContext.enabled }}
|
||||
securityContext: {{- omit $containerSecurityContext "enabled" | toYaml | nindent 12 }}
|
||||
{{- end }}
|
||||
{{- if or $enableAuth (include "seaweedfs.securityConfigEnabled" .) }}
|
||||
{{- if or (and $containerSecurityContext.enabled $containerSecurityContext.readOnlyRootFilesystem) $enableAuth (include "seaweedfs.securityConfigEnabled" .) }}
|
||||
volumes:
|
||||
{{- include "seaweedfs.tmpDirVolume" (list . $containerSecurityContext "" "" false) | nindent 8 }}
|
||||
{{- if $enableAuth }}
|
||||
- name: config-users
|
||||
secret:
|
||||
@@ -249,5 +265,13 @@ spec:
|
||||
configMap:
|
||||
name: {{ include "seaweedfs.fullname" . }}-security-config
|
||||
{{- end }}
|
||||
{{- if .Values.global.seaweedfs.enableSecurity }}
|
||||
- name: ca-cert
|
||||
secret:
|
||||
secretName: {{ include "seaweedfs.fullname" . }}-ca-cert
|
||||
- name: client-cert
|
||||
secret:
|
||||
secretName: {{ include "seaweedfs.fullname" . }}-client-cert
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
|
||||
@@ -18,7 +18,7 @@ data:
|
||||
{{- $legacyName := printf "%s-%s" (include "seaweedfs.name" .) "security-config" }}
|
||||
{{- $existing = lookup "v1" "ConfigMap" .Release.Namespace $legacyName }}
|
||||
{{- end }}
|
||||
{{- $securityConfig := fromToml (dig "data" "security.toml" "" $existing) }}
|
||||
{{- $existingToml := dig "data" "security.toml" "" $existing }}
|
||||
{{- $securityConfigValues := .Values.global.seaweedfs.securityConfig | default dict }}
|
||||
{{- $jwtSigning := $securityConfigValues.jwtSigning | default dict }}
|
||||
{{- $expiresAfterSeconds := $jwtSigning.expiresAfterSeconds | default dict }}
|
||||
@@ -29,7 +29,7 @@ data:
|
||||
# the jwt signing key is read by master and volume server
|
||||
# the jwt defaults to expire after 10 seconds
|
||||
[jwt.signing]
|
||||
key = "{{ dig "jwt" "signing" "key" (randAlphaNum 10 | b64enc) $securityConfig }}"
|
||||
key = {{ include "seaweedfs.existingTomlKey" (list "jwt.signing" $existingToml) | default (randAlphaNum 10 | b64enc | quote) }}
|
||||
{{- if gt (int $expiresAfterSeconds.volumeWrite) 0 }}
|
||||
expires_after_seconds = {{ int $expiresAfterSeconds.volumeWrite }}
|
||||
{{- end }}
|
||||
@@ -41,7 +41,7 @@ data:
|
||||
# - the Volume server validates the JWT on reading
|
||||
# the jwt defaults to expire after 60 seconds
|
||||
[jwt.signing.read]
|
||||
key = "{{ dig "jwt" "signing" "read" "key" (randAlphaNum 10 | b64enc) $securityConfig }}"
|
||||
key = {{ include "seaweedfs.existingTomlKey" (list "jwt.signing.read" $existingToml) | default (randAlphaNum 10 | b64enc | quote) }}
|
||||
{{- if gt (int $expiresAfterSeconds.volumeRead) 0 }}
|
||||
expires_after_seconds = {{ int $expiresAfterSeconds.volumeRead }}
|
||||
{{- end }}
|
||||
@@ -53,7 +53,7 @@ data:
|
||||
# - the Filer server validates the JWT on writing
|
||||
# the jwt defaults to expire after 10 seconds
|
||||
[jwt.filer_signing]
|
||||
key = "{{ dig "jwt" "filer_signing" "key" (randAlphaNum 10 | b64enc) $securityConfig }}"
|
||||
key = {{ include "seaweedfs.existingTomlKey" (list "jwt.filer_signing" $existingToml) | default (randAlphaNum 10 | b64enc | quote) }}
|
||||
{{- if gt (int $expiresAfterSeconds.filerWrite) 0 }}
|
||||
expires_after_seconds = {{ int $expiresAfterSeconds.filerWrite }}
|
||||
{{- end }}
|
||||
@@ -65,7 +65,7 @@ data:
|
||||
# - the Filer server validates the JWT on reading
|
||||
# the jwt defaults to expire after 60 seconds
|
||||
[jwt.filer_signing.read]
|
||||
key = "{{ dig "jwt" "filer_signing" "read" "key" (randAlphaNum 10 | b64enc) $securityConfig }}"
|
||||
key = {{ include "seaweedfs.existingTomlKey" (list "jwt.filer_signing.read" $existingToml) | default (randAlphaNum 10 | b64enc | quote) }}
|
||||
{{- if gt (int $expiresAfterSeconds.filerRead) 0 }}
|
||||
expires_after_seconds = {{ int $expiresAfterSeconds.filerRead }}
|
||||
{{- end }}
|
||||
|
||||
@@ -32,13 +32,32 @@ spec:
|
||||
spec:
|
||||
serviceAccountName: {{ $seaweedfsName }}-volume-resize-hook
|
||||
restartPolicy: Never
|
||||
{{- if .Values.volume.podSecurityContext.enabled }}
|
||||
securityContext: {{- omit .Values.volume.podSecurityContext "enabled" | toYaml | nindent 8 }}
|
||||
{{- end }}
|
||||
containers:
|
||||
- name: resize
|
||||
image: {{ .Values.volume.resizeHook.image }}
|
||||
{{- if and .Values.volume.containerSecurityContext.enabled .Values.volume.containerSecurityContext.readOnlyRootFilesystem }}
|
||||
env:
|
||||
- name: HOME
|
||||
value: /tmp
|
||||
{{- end }}
|
||||
command: ["sh", "-xec"]
|
||||
args:
|
||||
- |
|
||||
{{ $commands | indent 14 }}
|
||||
{{- if and .Values.volume.containerSecurityContext.enabled .Values.volume.containerSecurityContext.readOnlyRootFilesystem }}
|
||||
volumeMounts:
|
||||
{{- include "seaweedfs.tmpDirVolumeMount" (list . .Values.volume.containerSecurityContext "" "") | nindent 12 }}
|
||||
{{- end }}
|
||||
{{- if .Values.volume.containerSecurityContext.enabled }}
|
||||
securityContext: {{- omit .Values.volume.containerSecurityContext "enabled" | toYaml | nindent 12 }}
|
||||
{{- end }}
|
||||
{{- if and .Values.volume.containerSecurityContext.enabled .Values.volume.containerSecurityContext.readOnlyRootFilesystem }}
|
||||
volumes:
|
||||
{{- include "seaweedfs.tmpDirVolume" (list . .Values.volume.containerSecurityContext "" "" false) | nindent 8 }}
|
||||
{{- end }}
|
||||
---
|
||||
apiVersion: v1
|
||||
kind: ServiceAccount
|
||||
|
||||
@@ -104,6 +104,7 @@ spec:
|
||||
fi
|
||||
done
|
||||
volumeMounts:
|
||||
{{- include "seaweedfs.tmpDirVolumeMount" (list $ $volume.containerSecurityContext "" "") | nindent 12 }}
|
||||
- name: idx
|
||||
mountPath: /idx
|
||||
{{- range $dir := $volume.dataDirs }}
|
||||
@@ -224,6 +225,7 @@ spec:
|
||||
{{ . }} \
|
||||
{{- end }}
|
||||
volumeMounts:
|
||||
{{- include "seaweedfs.tmpDirVolumeMount" (list $ $volume.containerSecurityContext (tpl (printf "{{ $volumeName := \"%s\" }}%s" $volumeName ($volume.extraVolumeMounts | default "")) $) (tpl ($volume.extraVolumes | default "") $)) | nindent 12 }}
|
||||
{{- range $dir := $volume.dataDirs }}
|
||||
{{- if not ( eq $dir.type "custom" ) }}
|
||||
- name: {{ $dir.name }}
|
||||
@@ -306,6 +308,7 @@ spec:
|
||||
{{- include "seaweedfs.tplvalues.render" (dict "value" (printf "{{ $volumeName := \"%s\" }}%s" $volumeName $volume.sidecars) "context" $) | nindent 8 }}
|
||||
{{- end }}
|
||||
volumes:
|
||||
{{- include "seaweedfs.tmpDirVolume" (list $ $volume.containerSecurityContext (tpl (printf "{{ $volumeName := \"%s\" }}%s" $volumeName ($volume.extraVolumeMounts | default "")) $) (tpl ($volume.extraVolumes | default "") $) (and $initContainers_exists $volume.idx)) | nindent 8 }}
|
||||
|
||||
{{- range $dir := $volume.dataDirs }}
|
||||
|
||||
|
||||
@@ -144,6 +144,7 @@ spec:
|
||||
{{ $arg }}{{- if lt $index (sub (len $.Values.worker.extraArgs) 1) }} \{{ end }}
|
||||
{{- end }}
|
||||
volumeMounts:
|
||||
{{- include "seaweedfs.tmpDirVolumeMount" (list . .Values.worker.containerSecurityContext (tpl (.Values.worker.extraVolumeMounts | default "") .) (tpl (.Values.worker.extraVolumes | default "") .)) | nindent 12 }}
|
||||
{{- if or (eq .Values.worker.data.type "hostPath") (eq .Values.worker.data.type "emptyDir") (eq .Values.worker.data.type "existingClaim") }}
|
||||
- name: worker-data
|
||||
mountPath: {{ .Values.worker.workingDir }}
|
||||
@@ -262,15 +263,16 @@ spec:
|
||||
--metrics-ip=0.0.0.0 \
|
||||
{{- end }}
|
||||
--max-concurrency={{ .Values.worker.maxExecute }}
|
||||
{{- if .Values.global.seaweedfs.enableSecurity }}
|
||||
volumeMounts:
|
||||
{{- include "seaweedfs.tmpDirVolumeMount" (list . .Values.worker.containerSecurityContext "" "") | nindent 12 }}
|
||||
{{- if .Values.global.seaweedfs.enableSecurity }}
|
||||
- name: ca-cert
|
||||
readOnly: true
|
||||
mountPath: /usr/local/share/ca-certificates/ca/
|
||||
- name: worker-cert
|
||||
readOnly: true
|
||||
mountPath: /usr/local/share/ca-certificates/worker/
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
{{- if .Values.worker.lanceMetricsPort }}
|
||||
ports:
|
||||
- containerPort: {{ .Values.worker.lanceMetricsPort }}
|
||||
@@ -302,6 +304,7 @@ spec:
|
||||
{{- include "seaweedfs.tplvalues.render" (dict "value" .Values.worker.sidecars "context" $) | nindent 8 }}
|
||||
{{- end }}
|
||||
volumes:
|
||||
{{- include "seaweedfs.tmpDirVolume" (list . .Values.worker.containerSecurityContext (tpl (.Values.worker.extraVolumeMounts | default "") .) (tpl (.Values.worker.extraVolumes | default "") .) (ne (include "seaweedfs.worker.lanceNamespaceUrl" .) "")) | nindent 8 }}
|
||||
{{- if eq .Values.worker.data.type "hostPath" }}
|
||||
- name: worker-data
|
||||
hostPath:
|
||||
|
||||
@@ -25,6 +25,9 @@ global:
|
||||
imagePullPolicy: IfNotPresent
|
||||
restartPolicy: Always
|
||||
loggingLevel: 1
|
||||
# Writable temporary storage used when containers run with a read-only root filesystem.
|
||||
tmpDir:
|
||||
sizeLimit: ""
|
||||
enableSecurity: false
|
||||
masterServer: null
|
||||
# filerWrite: true mounts security.toml on filer + admin without needing
|
||||
@@ -260,6 +263,8 @@ master:
|
||||
# runAsUser: 1000
|
||||
# runAsGroup: 3000
|
||||
# fsGroup: 2000
|
||||
# seccompProfile:
|
||||
# type: RuntimeDefault
|
||||
podSecurityContext: {}
|
||||
|
||||
# Configure security context for Container
|
||||
@@ -268,7 +273,16 @@ master:
|
||||
# containerSecurityContext:
|
||||
# enabled: true
|
||||
# runAsUser: 2000
|
||||
# runAsGroup: 3000
|
||||
# runAsNonRoot: true
|
||||
# privileged: false
|
||||
# allowPrivilegeEscalation: false
|
||||
# readOnlyRootFilesystem: true
|
||||
# capabilities:
|
||||
# drop:
|
||||
# - ALL
|
||||
# seccompProfile:
|
||||
# type: RuntimeDefault
|
||||
containerSecurityContext: {}
|
||||
|
||||
ingress:
|
||||
@@ -560,6 +574,8 @@ volume:
|
||||
# runAsUser: 1000
|
||||
# runAsGroup: 3000
|
||||
# fsGroup: 2000
|
||||
# seccompProfile:
|
||||
# type: RuntimeDefault
|
||||
podSecurityContext: {}
|
||||
|
||||
# Configure security context for Container
|
||||
@@ -568,7 +584,16 @@ volume:
|
||||
# containerSecurityContext:
|
||||
# enabled: true
|
||||
# runAsUser: 2000
|
||||
# runAsGroup: 3000
|
||||
# runAsNonRoot: true
|
||||
# privileged: false
|
||||
# allowPrivilegeEscalation: false
|
||||
# readOnlyRootFilesystem: true
|
||||
# capabilities:
|
||||
# drop:
|
||||
# - ALL
|
||||
# seccompProfile:
|
||||
# type: RuntimeDefault
|
||||
containerSecurityContext: {}
|
||||
|
||||
# used to configure livenessProbe on volume-server containers
|
||||
@@ -845,6 +870,8 @@ filer:
|
||||
# runAsUser: 1000
|
||||
# runAsGroup: 3000
|
||||
# fsGroup: 2000
|
||||
# seccompProfile:
|
||||
# type: RuntimeDefault
|
||||
podSecurityContext: {}
|
||||
|
||||
# Configure security context for Container
|
||||
@@ -853,7 +880,16 @@ filer:
|
||||
# containerSecurityContext:
|
||||
# enabled: true
|
||||
# runAsUser: 2000
|
||||
# runAsGroup: 3000
|
||||
# runAsNonRoot: true
|
||||
# privileged: false
|
||||
# allowPrivilegeEscalation: false
|
||||
# readOnlyRootFilesystem: true
|
||||
# capabilities:
|
||||
# drop:
|
||||
# - ALL
|
||||
# seccompProfile:
|
||||
# type: RuntimeDefault
|
||||
containerSecurityContext: {}
|
||||
|
||||
ingresses:
|
||||
@@ -999,6 +1035,17 @@ s3:
|
||||
imageOverride: null
|
||||
restartPolicy: null
|
||||
replicas: 1
|
||||
# Deployment update strategy, rendered as the Deployment's strategy
|
||||
# ref: https://kubernetes.io/docs/concepts/workloads/controllers/deployment/#strategy
|
||||
# Example:
|
||||
# updateStrategy:
|
||||
# type: RollingUpdate
|
||||
# rollingUpdate:
|
||||
# maxUnavailable: 0
|
||||
# maxSurge: 25%
|
||||
updateStrategy: {}
|
||||
# Time the s3 pod is given to shut down, including any preStop hook
|
||||
terminationGracePeriodSeconds: 10
|
||||
bindAddress: 0.0.0.0
|
||||
port: 8333
|
||||
# add additional https port
|
||||
@@ -1127,6 +1174,8 @@ s3:
|
||||
# runAsUser: 1000
|
||||
# runAsGroup: 3000
|
||||
# fsGroup: 2000
|
||||
# seccompProfile:
|
||||
# type: RuntimeDefault
|
||||
podSecurityContext: {}
|
||||
|
||||
# Configure security context for Container
|
||||
@@ -1135,7 +1184,16 @@ s3:
|
||||
# containerSecurityContext:
|
||||
# enabled: true
|
||||
# runAsUser: 2000
|
||||
# runAsGroup: 3000
|
||||
# runAsNonRoot: true
|
||||
# privileged: false
|
||||
# allowPrivilegeEscalation: false
|
||||
# readOnlyRootFilesystem: true
|
||||
# capabilities:
|
||||
# drop:
|
||||
# - ALL
|
||||
# seccompProfile:
|
||||
# type: RuntimeDefault
|
||||
containerSecurityContext: {}
|
||||
|
||||
# You can also use emptyDir storage:
|
||||
@@ -1180,6 +1238,17 @@ s3:
|
||||
failureThreshold: 100
|
||||
timeoutSeconds: 10
|
||||
|
||||
# Lifecycle hooks for the s3 container. A preStop sleep lets the pod leave
|
||||
# the Service endpoints before it receives SIGTERM; keep it shorter than
|
||||
# terminationGracePeriodSeconds.
|
||||
# ref: https://kubernetes.io/docs/concepts/containers/container-lifecycle-hooks/
|
||||
# Example:
|
||||
# lifecycle:
|
||||
# preStop:
|
||||
# exec:
|
||||
# command: ["sleep", "5"]
|
||||
lifecycle: {}
|
||||
|
||||
createBucketsHook:
|
||||
resources: {}
|
||||
|
||||
@@ -1287,7 +1356,33 @@ sftp:
|
||||
priorityClassName: ""
|
||||
schedulerName: ""
|
||||
serviceAccountName: ""
|
||||
# Configure security context for Pod
|
||||
# ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/
|
||||
# Example:
|
||||
# podSecurityContext:
|
||||
# enabled: true
|
||||
# runAsUser: 1000
|
||||
# runAsGroup: 3000
|
||||
# fsGroup: 2000
|
||||
# seccompProfile:
|
||||
# type: RuntimeDefault
|
||||
podSecurityContext: {}
|
||||
# Configure security context for Container
|
||||
# ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/
|
||||
# Example:
|
||||
# containerSecurityContext:
|
||||
# enabled: true
|
||||
# runAsUser: 2000
|
||||
# runAsGroup: 3000
|
||||
# runAsNonRoot: true
|
||||
# privileged: false
|
||||
# allowPrivilegeEscalation: false
|
||||
# readOnlyRootFilesystem: true
|
||||
# capabilities:
|
||||
# drop:
|
||||
# - ALL
|
||||
# seccompProfile:
|
||||
# type: RuntimeDefault
|
||||
containerSecurityContext: {}
|
||||
|
||||
logs:
|
||||
@@ -1434,7 +1529,33 @@ admin:
|
||||
priorityClassName: ""
|
||||
schedulerName: ""
|
||||
serviceAccountName: ""
|
||||
# Configure security context for Pod
|
||||
# ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/
|
||||
# Example:
|
||||
# podSecurityContext:
|
||||
# enabled: true
|
||||
# runAsUser: 1000
|
||||
# runAsGroup: 3000
|
||||
# fsGroup: 2000
|
||||
# seccompProfile:
|
||||
# type: RuntimeDefault
|
||||
podSecurityContext: {}
|
||||
# Configure security context for Container
|
||||
# ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/
|
||||
# Example:
|
||||
# containerSecurityContext:
|
||||
# enabled: true
|
||||
# runAsUser: 2000
|
||||
# runAsGroup: 3000
|
||||
# runAsNonRoot: true
|
||||
# privileged: false
|
||||
# allowPrivilegeEscalation: false
|
||||
# readOnlyRootFilesystem: true
|
||||
# capabilities:
|
||||
# drop:
|
||||
# - ALL
|
||||
# seccompProfile:
|
||||
# type: RuntimeDefault
|
||||
containerSecurityContext: {}
|
||||
|
||||
extraEnvironmentVars: {}
|
||||
@@ -1591,7 +1712,33 @@ worker:
|
||||
priorityClassName: ""
|
||||
schedulerName: ""
|
||||
serviceAccountName: ""
|
||||
# Configure security context for Pod
|
||||
# ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/
|
||||
# Example:
|
||||
# podSecurityContext:
|
||||
# enabled: true
|
||||
# runAsUser: 1000
|
||||
# runAsGroup: 3000
|
||||
# fsGroup: 2000
|
||||
# seccompProfile:
|
||||
# type: RuntimeDefault
|
||||
podSecurityContext: {}
|
||||
# Configure security context for Container
|
||||
# ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/
|
||||
# Example:
|
||||
# containerSecurityContext:
|
||||
# enabled: true
|
||||
# runAsUser: 2000
|
||||
# runAsGroup: 3000
|
||||
# runAsNonRoot: true
|
||||
# privileged: false
|
||||
# allowPrivilegeEscalation: false
|
||||
# readOnlyRootFilesystem: true
|
||||
# capabilities:
|
||||
# drop:
|
||||
# - ALL
|
||||
# seccompProfile:
|
||||
# type: RuntimeDefault
|
||||
containerSecurityContext: {}
|
||||
|
||||
extraEnvironmentVars: {}
|
||||
@@ -1851,6 +1998,8 @@ allInOne:
|
||||
# runAsUser: 1000
|
||||
# runAsGroup: 3000
|
||||
# fsGroup: 2000
|
||||
# seccompProfile:
|
||||
# type: RuntimeDefault
|
||||
podSecurityContext: {}
|
||||
|
||||
# Configure security context for Container
|
||||
@@ -1859,7 +2008,16 @@ allInOne:
|
||||
# containerSecurityContext:
|
||||
# enabled: true
|
||||
# runAsUser: 2000
|
||||
# runAsGroup: 3000
|
||||
# runAsNonRoot: true
|
||||
# privileged: false
|
||||
# allowPrivilegeEscalation: false
|
||||
# readOnlyRootFilesystem: true
|
||||
# capabilities:
|
||||
# drop:
|
||||
# - ALL
|
||||
# seccompProfile:
|
||||
# type: RuntimeDefault
|
||||
containerSecurityContext: {}
|
||||
|
||||
# Resource management
|
||||
@@ -1898,7 +2056,33 @@ cosi:
|
||||
# should have a secret key called seaweedfs_s3_config with an inline json configure
|
||||
existingConfigSecret: null
|
||||
|
||||
# Configure security context for Pod
|
||||
# ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/
|
||||
# Example:
|
||||
# podSecurityContext:
|
||||
# enabled: true
|
||||
# runAsUser: 1000
|
||||
# runAsGroup: 3000
|
||||
# fsGroup: 2000
|
||||
# seccompProfile:
|
||||
# type: RuntimeDefault
|
||||
podSecurityContext: {}
|
||||
# Configure security context for Container
|
||||
# ref: https://kubernetes.io/docs/tasks/configure-pod-container/security-context/
|
||||
# Example:
|
||||
# containerSecurityContext:
|
||||
# enabled: true
|
||||
# runAsUser: 2000
|
||||
# runAsGroup: 3000
|
||||
# runAsNonRoot: true
|
||||
# privileged: false
|
||||
# allowPrivilegeEscalation: false
|
||||
# readOnlyRootFilesystem: true
|
||||
# capabilities:
|
||||
# drop:
|
||||
# - ALL
|
||||
# seccompProfile:
|
||||
# type: RuntimeDefault
|
||||
containerSecurityContext: {}
|
||||
|
||||
# used to assign a custom scheduler to cosi pods
|
||||
|
||||
+1049
-1005
File diff suppressed because it is too large
Load Diff
|
Before Width: | Height: | Size: 53 KiB After Width: | Height: | Size: 54 KiB |
@@ -213,6 +213,9 @@ message KeepConnectedRequest {
|
||||
string filer_group = 5;
|
||||
string data_center = 6;
|
||||
string rack = 7;
|
||||
// A draining filer leaves the lock ring but stays a cluster member, so peers
|
||||
// keep following its metadata until the stream closes.
|
||||
bool leave_lock_ring = 8;
|
||||
}
|
||||
|
||||
message VolumeLocation {
|
||||
|
||||
@@ -371,6 +371,9 @@ async fn run(
|
||||
.to_string_lossy()
|
||||
.into_owned()
|
||||
},
|
||||
ec_decodes_in_flight: std::sync::Mutex::new(std::collections::HashSet::new()),
|
||||
ec_decode_tail: std::sync::Mutex::new(std::collections::HashSet::new()),
|
||||
ec_decode_tail_notify: tokio::sync::Notify::new(),
|
||||
});
|
||||
|
||||
// Load persisted state from disk if it exists (matches Go's State.Load on startup)
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
+1317
-509
File diff suppressed because it is too large
Load Diff
@@ -1508,6 +1508,9 @@ mod tests {
|
||||
security_file: String::new(),
|
||||
cli_white_list: vec![],
|
||||
state_file_path: String::new(),
|
||||
ec_decodes_in_flight: std::sync::Mutex::new(std::collections::HashSet::new()),
|
||||
ec_decode_tail: std::sync::Mutex::new(std::collections::HashSet::new()),
|
||||
ec_decode_tail_notify: tokio::sync::Notify::new(),
|
||||
})
|
||||
}
|
||||
|
||||
|
||||
@@ -419,13 +419,24 @@ async fn delete_on_ec_shard_holders(
|
||||
|
||||
let mut last_err = None;
|
||||
if local_shards.contains(&shard_id) {
|
||||
match journal_delete_local(state, target.vid, target.needle_id) {
|
||||
Ok(()) => return Ok(true),
|
||||
// Nothing was committed — the volume unmounted or remounted
|
||||
// without the needle — so it is safe to fall back to other
|
||||
// shard holders, unlike an RPC failure which may have landed.
|
||||
Err(e) if e.kind() == io::ErrorKind::NotFound => return Ok(false),
|
||||
Err(e) => last_err = Some(e),
|
||||
// A decode in its publishing tail must not miss this delete. The
|
||||
// tail membership is verified again under the store write lock
|
||||
// inside journal_delete_local — WouldBlock means the decode claimed
|
||||
// it in the gap after this wait — so wait and retry.
|
||||
loop {
|
||||
crate::server::grpc_server::wait_ec_decode_tail(state, target.vid).await;
|
||||
match journal_delete_local(state, target.vid, target.needle_id) {
|
||||
Ok(()) => return Ok(true),
|
||||
Err(e) if e.kind() == io::ErrorKind::WouldBlock => continue,
|
||||
// Nothing was committed — the volume unmounted or remounted
|
||||
// without the needle — so it is safe to fall back to other
|
||||
// shard holders, unlike an RPC failure which may have landed.
|
||||
Err(e) if e.kind() == io::ErrorKind::NotFound => return Ok(false),
|
||||
Err(e) => {
|
||||
last_err = Some(e);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if let Some(addrs) = addrs {
|
||||
@@ -487,6 +498,15 @@ fn journal_delete_local(
|
||||
needle_id: NeedleId,
|
||||
) -> io::Result<()> {
|
||||
let mut store = state.store.write().unwrap();
|
||||
// Membership is read under the write lock: a decode can claim the
|
||||
// publishing tail while this call waited for the decoder's read lock,
|
||||
// so a check taken earlier would be stale by commit time.
|
||||
if crate::server::grpc_server::ec_decode_tail_contains(state, vid) {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::WouldBlock,
|
||||
format!("ec volume {} is in decode publishing tail", vid.0),
|
||||
));
|
||||
}
|
||||
let ecv = store.find_ec_volume_mut(vid).ok_or_else(|| {
|
||||
io::Error::new(
|
||||
io::ErrorKind::NotFound,
|
||||
@@ -1753,7 +1773,7 @@ async fn fetch_ec_index_from_one_peer(
|
||||
.await
|
||||
.map_err(|e| io::Error::other(format!("copy .ecx: {}", e)))?
|
||||
.into_inner();
|
||||
drain_copy_stream(stream, ecx_path, false).await?;
|
||||
drain_copy_stream(stream, ecx_path).await?;
|
||||
|
||||
let meta =
|
||||
fs::metadata(ecx_path).map_err(|e| io::Error::other(format!("stat copied .ecx: {}", e)))?;
|
||||
@@ -1766,15 +1786,30 @@ async fn fetch_ec_index_from_one_peer(
|
||||
)));
|
||||
}
|
||||
|
||||
// .ecj is the source peer's deletion journal (appended); .vif carries EC
|
||||
// params. Both are best-effort: a missing .ecj is recreated at mount and a
|
||||
// missing .vif falls back to default EC parameters. A failed .ecj append
|
||||
// leaves a partial file, so drop it.
|
||||
// .ecj is the source peer's deletion journal; .vif carries EC params. Both
|
||||
// are best-effort: a missing .ecj is recreated at mount and a missing .vif
|
||||
// falls back to default EC parameters. The journal is a *set*: merge the
|
||||
// peer's ids into any local ones as a union instead of appending, so
|
||||
// a volume bounced between servers cannot double its journal. The merge
|
||||
// only appends whole records, so a failure leaves nothing to clean up.
|
||||
match client.copy_file(copy_req(".ecj", true)).await {
|
||||
Ok(resp) => {
|
||||
if let Err(e) = drain_copy_stream(resp.into_inner(), ecj_path, true).await {
|
||||
let mut stream = resp.into_inner();
|
||||
let merged = match crate::server::grpc_server::receive_ecj_ids(&mut stream).await {
|
||||
Ok((ids, true)) => crate::server::grpc_server::merge_ecj_ids(
|
||||
state,
|
||||
m.vid,
|
||||
m.data_dir.clone(),
|
||||
ecj_path.to_string(),
|
||||
ids,
|
||||
)
|
||||
.await
|
||||
.map(|_| ()),
|
||||
Ok((_, false)) => Ok(()),
|
||||
Err(e) => Err(e),
|
||||
};
|
||||
if let Err(e) = merged {
|
||||
tracing::warn!(volume_id = m.vid.0, peer = %peer, "copy .ecj: {}", e);
|
||||
let _ = fs::remove_file(ecj_path);
|
||||
}
|
||||
}
|
||||
Err(e) => tracing::warn!(volume_id = m.vid.0, peer = %peer, "copy .ecj: {}", e),
|
||||
@@ -1782,7 +1817,7 @@ async fn fetch_ec_index_from_one_peer(
|
||||
|
||||
match client.copy_file(copy_req(".vif", true)).await {
|
||||
Ok(resp) => {
|
||||
if let Err(e) = drain_copy_stream(resp.into_inner(), vif_path, false).await {
|
||||
if let Err(e) = drain_copy_stream(resp.into_inner(), vif_path).await {
|
||||
tracing::warn!(volume_id = m.vid.0, peer = %peer, "copy .vif: {}", e);
|
||||
}
|
||||
}
|
||||
@@ -1792,22 +1827,14 @@ async fn fetch_ec_index_from_one_peer(
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Drain a CopyFile stream into a local file, appending or truncating.
|
||||
/// Drain a CopyFile stream into a local file, truncating it first.
|
||||
async fn drain_copy_stream(
|
||||
mut stream: tonic::Streaming<crate::pb::volume_server_pb::CopyFileResponse>,
|
||||
dest_path: &str,
|
||||
append: bool,
|
||||
) -> io::Result<()> {
|
||||
use std::io::Write;
|
||||
let mut file = if append {
|
||||
fs::OpenOptions::new()
|
||||
.create(true)
|
||||
.append(true)
|
||||
.open(dest_path)
|
||||
} else {
|
||||
fs::File::create(dest_path)
|
||||
}
|
||||
.map_err(|e| io::Error::other(format!("create {}: {}", dest_path, e)))?;
|
||||
let mut file = fs::File::create(dest_path)
|
||||
.map_err(|e| io::Error::other(format!("create {}: {}", dest_path, e)))?;
|
||||
while let Some(chunk) = stream
|
||||
.message()
|
||||
.await
|
||||
|
||||
@@ -114,6 +114,21 @@ pub struct VolumeServerState {
|
||||
pub cli_white_list: Vec<String>,
|
||||
/// Path to state.pb file for persisting VolumeServerState across restarts.
|
||||
pub state_file_path: String,
|
||||
/// Volumes with an EC decode in flight. A dropped request leaves the
|
||||
/// blocking job running; this keeps a retry from racing it on the
|
||||
/// same volume files.
|
||||
pub ec_decodes_in_flight:
|
||||
std::sync::Mutex<std::collections::HashSet<crate::storage::types::VolumeId>>,
|
||||
/// Volumes whose EC decode is in its publishing tail (journal catch-up,
|
||||
/// .idx write, compaction). Local .ecj appenders wait on
|
||||
/// `ec_decode_tail_notify` while their vid is listed, so no committed
|
||||
/// delete falls between the last catch_up and the .cpd/.cpx swap —
|
||||
/// the per-volume slice of Go's EcVolume.ecjFileAccessLock.
|
||||
pub ec_decode_tail:
|
||||
std::sync::Mutex<std::collections::HashSet<crate::storage::types::VolumeId>>,
|
||||
/// Wakes .ecj appenders waiting on `ec_decode_tail` when a decode's
|
||||
/// publishing tail ends.
|
||||
pub ec_decode_tail_notify: tokio::sync::Notify,
|
||||
}
|
||||
|
||||
impl VolumeServerState {
|
||||
|
||||
@@ -229,6 +229,9 @@ mod tests {
|
||||
security_file: String::new(),
|
||||
cli_white_list: vec![],
|
||||
state_file_path: String::new(),
|
||||
ec_decodes_in_flight: std::sync::Mutex::new(std::collections::HashSet::new()),
|
||||
ec_decode_tail: std::sync::Mutex::new(std::collections::HashSet::new()),
|
||||
ec_decode_tail_notify: tokio::sync::Notify::new(),
|
||||
})
|
||||
}
|
||||
|
||||
|
||||
@@ -18,7 +18,9 @@ use crate::storage::erasure_coding::ec_shard::{
|
||||
DATA_SHARDS_COUNT, ERASURE_CODING_LARGE_BLOCK_SIZE, ERASURE_CODING_SMALL_BLOCK_SIZE,
|
||||
EcVolumeShard, ShardId,
|
||||
};
|
||||
use crate::storage::erasure_coding::ec_volume::{EcVolume, is_usable_ecx_file};
|
||||
use crate::storage::erasure_coding::ec_volume::{
|
||||
ECJ_COMPACT_TMP_EXT, EcVolume, is_usable_ecx_file,
|
||||
};
|
||||
use crate::storage::needle_map::NeedleMapKind;
|
||||
use crate::storage::super_block::SUPER_BLOCK_SIZE;
|
||||
use crate::storage::types::*;
|
||||
@@ -460,13 +462,16 @@ impl DiskLocation {
|
||||
let idx_base = volume_file_name(&self.idx_directory, collection, vid);
|
||||
const MAX_SHARD_COUNT: usize = 32;
|
||||
|
||||
// Remove index files from idx directory (.ecx, .ecj)
|
||||
// Remove index files from idx directory (.ecx, .ecj, and a compaction
|
||||
// tmp a crash may have left beside the .ecj)
|
||||
rm_if_present(format!("{}.ecx", idx_base))?;
|
||||
rm_if_present(format!("{}.ecj", idx_base))?;
|
||||
rm_if_present(format!("{}{}", idx_base, ECJ_COMPACT_TMP_EXT))?;
|
||||
// Also try data directory in case .ecx/.ecj were created before -dir.idx was configured
|
||||
if self.idx_directory != self.directory {
|
||||
rm_if_present(format!("{}.ecx", base))?;
|
||||
rm_if_present(format!("{}.ecj", base))?;
|
||||
rm_if_present(format!("{}{}", base, ECJ_COMPACT_TMP_EXT))?;
|
||||
}
|
||||
|
||||
// Remove all EC shard files (.ec00 ~ .ec31)
|
||||
|
||||
@@ -97,21 +97,97 @@ pub fn read_ecj_deletions(
|
||||
collection: &str,
|
||||
volume_id: VolumeId,
|
||||
) -> io::Result<HashSet<NeedleId>> {
|
||||
let mut ids = HashSet::new();
|
||||
for (i, dir) in dirs.iter().enumerate() {
|
||||
if dirs[..i].contains(dir) {
|
||||
continue;
|
||||
Ok(EcjDeletions::read(dirs, collection, volume_id)?.ids)
|
||||
}
|
||||
|
||||
/// The ids [`read_ecj_deletions`] returns, plus how far each journal was read
|
||||
/// so that ids journaled later can be added.
|
||||
pub struct EcjDeletions {
|
||||
pub ids: HashSet<NeedleId>,
|
||||
/// Each distinct journal path and the whole-record length read so far.
|
||||
journals: Vec<(String, u64)>,
|
||||
}
|
||||
|
||||
impl EcjDeletions {
|
||||
pub fn read(dirs: &[&str], collection: &str, volume_id: VolumeId) -> io::Result<Self> {
|
||||
let mut journals: Vec<(String, u64)> = Vec::new();
|
||||
for dir in dirs {
|
||||
let path = format!("{}.ecj", volume_file_name(dir, collection, volume_id));
|
||||
if !journals.iter().any(|(p, _)| *p == path) {
|
||||
journals.push((path, 0));
|
||||
}
|
||||
}
|
||||
let path = format!("{}.ecj", volume_file_name(dir, collection, volume_id));
|
||||
let file = match File::open(&path) {
|
||||
Ok(file) => file,
|
||||
Err(e) if e.kind() == io::ErrorKind::NotFound => continue,
|
||||
Err(e) => return Err(e),
|
||||
let mut deletions = EcjDeletions {
|
||||
ids: HashSet::new(),
|
||||
journals,
|
||||
};
|
||||
let len = file.metadata()?.len();
|
||||
read_ecj_ids(&file, len, &mut ids)?;
|
||||
deletions.catch_up()?;
|
||||
Ok(deletions)
|
||||
}
|
||||
|
||||
/// Adds the ids appended to each journal since the last read. A journal
|
||||
/// that shrank is read again from the start — and the whole set rebuilt,
|
||||
/// since ids already folded in from the truncated tail may have been a
|
||||
/// rolled-back append. Non-regular journals (the FIFOs the tests stand
|
||||
/// in for a blocking disk) stat empty and cannot be rolled back, so they
|
||||
/// never count as shrunk.
|
||||
pub fn catch_up(&mut self) -> io::Result<()> {
|
||||
let mut shrank = false;
|
||||
for (path, read_to) in &self.journals {
|
||||
if *read_to == 0 {
|
||||
continue;
|
||||
}
|
||||
match std::fs::metadata(path) {
|
||||
Ok(m) => {
|
||||
if m.is_file() && m.len() < *read_to {
|
||||
shrank = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
// A journal that was read before and is now gone shrank to
|
||||
// nothing (e.g. the volume was destroyed mid-scan) — the ids
|
||||
// read from it no longer reflect committed content.
|
||||
Err(e) if e.kind() == io::ErrorKind::NotFound => {
|
||||
shrank = true;
|
||||
break;
|
||||
}
|
||||
Err(e) => return Err(e),
|
||||
}
|
||||
}
|
||||
if shrank {
|
||||
self.ids.clear();
|
||||
for (_, read_to) in &mut self.journals {
|
||||
*read_to = 0;
|
||||
}
|
||||
}
|
||||
for (path, read_to) in &mut self.journals {
|
||||
let file = match File::open(&*path) {
|
||||
Ok(file) => file,
|
||||
Err(e) if e.kind() == io::ErrorKind::NotFound => continue,
|
||||
Err(e) => return Err(e),
|
||||
};
|
||||
let len = file.metadata()?.len();
|
||||
if len < *read_to {
|
||||
*read_to = 0;
|
||||
}
|
||||
read_ecj_ids(&file, *read_to, len, &mut self.ids)?;
|
||||
*read_to = len - len % NEEDLE_ID_SIZE as u64;
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Clears the set and re-reads every journal from the start. Run under
|
||||
/// the caller's store read lock (no append in flight): the result is
|
||||
/// then exactly the committed content — an earlier unlocked read may
|
||||
/// have folded in bytes a rolled-back append later truncated, or missed
|
||||
/// a record re-appended to the very offset a rollback freed.
|
||||
pub fn rescan(&mut self) -> io::Result<()> {
|
||||
self.ids.clear();
|
||||
for (_, read_to) in &mut self.journals {
|
||||
*read_to = 0;
|
||||
}
|
||||
self.catch_up()
|
||||
}
|
||||
Ok(ids)
|
||||
}
|
||||
|
||||
/// What it takes to rebuild a volume's .dat from its EC data shards.
|
||||
@@ -312,6 +388,30 @@ pub fn write_dat_file_from_shards(spec: &DatRebuild<'_>) -> io::Result<()> {
|
||||
write_result
|
||||
}
|
||||
|
||||
/// Fails when the decoded `.dat` in `dat_dir` is shorter than the
|
||||
/// `dat_file_size` bytes its EC index references: the caller deletes the
|
||||
/// shards next, and they are the only other copy of the needles past the cut.
|
||||
/// A longer file passes.
|
||||
pub fn verify_decoded_dat_file(
|
||||
dat_dir: &str,
|
||||
collection: &str,
|
||||
volume_id: VolumeId,
|
||||
dat_file_size: i64,
|
||||
) -> io::Result<()> {
|
||||
let dat_path = format!("{}.dat", volume_file_name(dat_dir, collection, volume_id));
|
||||
let size = std::fs::metadata(&dat_path)?.len();
|
||||
if (size as i64) < dat_file_size {
|
||||
return Err(io::Error::new(
|
||||
io::ErrorKind::UnexpectedEof,
|
||||
format!(
|
||||
"decoded {} is {} bytes, short of the {} its ec index references",
|
||||
dat_path, size, dat_file_size
|
||||
),
|
||||
));
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Write .idx file from .ecx index + .ecj deletion journal.
|
||||
///
|
||||
/// See [`write_idx_file_from_ec_index_with_dirs`]; everything lives in `dir`.
|
||||
@@ -790,4 +890,77 @@ mod tests {
|
||||
let expected: HashSet<NeedleId> = [1, 2, 3, 7, 9].into_iter().map(NeedleId).collect();
|
||||
assert_eq!(ids, expected);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_verify_decoded_dat_file_rejects_a_short_dat() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
let dat_path = format!("{dir}/1.dat");
|
||||
|
||||
let err = verify_decoded_dat_file(dir, "", VolumeId(1), 100).unwrap_err();
|
||||
assert_eq!(err.kind(), io::ErrorKind::NotFound);
|
||||
|
||||
std::fs::write(&dat_path, vec![0u8; 99]).unwrap();
|
||||
let err = verify_decoded_dat_file(dir, "", VolumeId(1), 100).unwrap_err();
|
||||
assert!(err.to_string().contains("short of the 100"), "{err}");
|
||||
|
||||
std::fs::write(&dat_path, vec![0u8; 100]).unwrap();
|
||||
verify_decoded_dat_file(dir, "", VolumeId(1), 100).unwrap();
|
||||
std::fs::write(&dat_path, vec![0u8; 101]).unwrap();
|
||||
verify_decoded_dat_file(dir, "", VolumeId(1), 100).unwrap();
|
||||
}
|
||||
|
||||
/// Ids appended after the first read, including the rest of a record torn
|
||||
/// at that point, are picked up; a journal that shrank is read again.
|
||||
#[test]
|
||||
fn test_ecj_deletions_catch_up_reads_appended_ids() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
let ecj_path = format!("{dir}/1.ecj");
|
||||
let entry = |id: u64| {
|
||||
let mut buf = [0u8; NEEDLE_ID_SIZE];
|
||||
NeedleId(id).to_bytes(&mut buf);
|
||||
buf
|
||||
};
|
||||
let append = |bytes: &[u8]| {
|
||||
let mut f = std::fs::OpenOptions::new()
|
||||
.create(true)
|
||||
.append(true)
|
||||
.open(&ecj_path)
|
||||
.unwrap();
|
||||
f.write_all(bytes).unwrap();
|
||||
};
|
||||
let ids = |d: &EcjDeletions| {
|
||||
let mut ids: Vec<u64> = d.ids.iter().map(|id| id.0).collect();
|
||||
ids.sort();
|
||||
ids
|
||||
};
|
||||
|
||||
// No journal yet.
|
||||
let mut deletions = EcjDeletions::read(&[dir, dir], "", VolumeId(1)).unwrap();
|
||||
assert!(deletions.ids.is_empty());
|
||||
|
||||
append(&entry(1));
|
||||
append(&entry(2)[..3]);
|
||||
deletions.catch_up().unwrap();
|
||||
assert_eq!(ids(&deletions), [1]);
|
||||
|
||||
append(&entry(2)[3..]);
|
||||
append(&entry(3));
|
||||
deletions.catch_up().unwrap();
|
||||
assert_eq!(ids(&deletions), [1, 2, 3]);
|
||||
|
||||
// A shrunk journal is rebuilt from its surviving content: ids folded
|
||||
// in from the truncated tail may have been rolled back and must not
|
||||
// linger as phantom tombstones.
|
||||
std::fs::write(&ecj_path, entry(9)).unwrap();
|
||||
deletions.catch_up().unwrap();
|
||||
assert_eq!(ids(&deletions), [9]);
|
||||
|
||||
// A journal removed since it was read is the extreme shrink: its
|
||||
// earlier ids must go with it, not linger.
|
||||
std::fs::remove_file(&ecj_path).unwrap();
|
||||
deletions.catch_up().unwrap();
|
||||
assert!(deletions.ids.is_empty());
|
||||
}
|
||||
}
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,290 @@
|
||||
//! Set-union merge of `.ecj` deletion journals for EC shard copy / index
|
||||
//! recovery. Mirrors Go's `weed/storage/erasure_coding/ecj_merge.go`.
|
||||
//!
|
||||
//! An EC volume's deletion journal (`<vid>.ecj`) is a *set* of deleted needle
|
||||
//! ids stored as 8-byte big-endian records. Shard copy and index recovery fold
|
||||
//! a peer's journal into the local one; they must append only the ids the
|
||||
//! local journal lacks, or every `ec_balance` round trip doubles the file.
|
||||
//!
|
||||
//! The journal is only ever appended to, never replaced: a mounted `EcVolume`
|
||||
//! holds it open, and a rename would leave that handle writing to an unlinked
|
||||
//! inode, losing every later delete at the next mount. A mounted volume merges
|
||||
//! through [`EcVolume::merge_journal`](super::ec_volume::EcVolume::merge_journal);
|
||||
//! [`append_ecj_ids`] is for a journal no volume has open.
|
||||
|
||||
use std::collections::HashSet;
|
||||
use std::fs::{self, OpenOptions};
|
||||
use std::io::{self, Read, Seek, SeekFrom, Write};
|
||||
use std::path::Path;
|
||||
|
||||
use crate::storage::types::{NEEDLE_ID_SIZE, NeedleId};
|
||||
use crate::storage::volume::fsync_dir;
|
||||
use crate::storage::volume_open::open_volume_file;
|
||||
|
||||
/// Bytes per read when scanning a journal; a multiple of `NEEDLE_ID_SIZE`.
|
||||
const ECJ_READ_CHUNK_BYTES: usize = 1 << 20;
|
||||
|
||||
/// Decodes `.ecj` records from a byte stream split at arbitrary boundaries (a
|
||||
/// CopyFile stream, chunked reads), collecting the distinct ids. Memory
|
||||
/// follows the number of distinct ids, not the journal's length. A trailing
|
||||
/// partial record is never decoded.
|
||||
#[derive(Default)]
|
||||
pub(crate) struct EcjIdDecoder {
|
||||
ids: HashSet<NeedleId>,
|
||||
partial: [u8; NEEDLE_ID_SIZE],
|
||||
pending: usize,
|
||||
}
|
||||
|
||||
impl EcjIdDecoder {
|
||||
pub(crate) fn push(&mut self, mut bytes: &[u8]) {
|
||||
if self.pending > 0 {
|
||||
let n = (NEEDLE_ID_SIZE - self.pending).min(bytes.len());
|
||||
self.partial[self.pending..self.pending + n].copy_from_slice(&bytes[..n]);
|
||||
self.pending += n;
|
||||
bytes = &bytes[n..];
|
||||
if self.pending < NEEDLE_ID_SIZE {
|
||||
return;
|
||||
}
|
||||
self.ids.insert(NeedleId::from_bytes(&self.partial));
|
||||
self.pending = 0;
|
||||
}
|
||||
let mut records = bytes.chunks_exact(NEEDLE_ID_SIZE);
|
||||
for record in &mut records {
|
||||
self.ids.insert(NeedleId::from_bytes(record));
|
||||
}
|
||||
let rest = records.remainder();
|
||||
self.partial[..rest.len()].copy_from_slice(rest);
|
||||
self.pending = rest.len();
|
||||
}
|
||||
|
||||
pub(crate) fn into_ids(self) -> HashSet<NeedleId> {
|
||||
self.ids
|
||||
}
|
||||
}
|
||||
|
||||
/// Read the distinct ids of the journal at `path` in bounded chunks. A missing
|
||||
/// file reads as empty. Also returns the whole-record length read; a torn
|
||||
/// trailing partial record is excluded from it.
|
||||
pub(crate) fn read_ecj_ids(path: &str) -> io::Result<(HashSet<NeedleId>, u64)> {
|
||||
let mut file = match fs::File::open(path) {
|
||||
Ok(f) => f,
|
||||
Err(e) if e.kind() == io::ErrorKind::NotFound => return Ok((HashSet::new(), 0)),
|
||||
Err(e) => return Err(e),
|
||||
};
|
||||
let len = file.metadata()?.len();
|
||||
let size = len - len % NEEDLE_ID_SIZE as u64;
|
||||
let mut decoder = EcjIdDecoder::default();
|
||||
let mut buf = vec![0u8; (ECJ_READ_CHUNK_BYTES as u64).min(size) as usize];
|
||||
let mut off = 0u64;
|
||||
while off < size {
|
||||
let want = (buf.len() as u64).min(size - off) as usize;
|
||||
file.read_exact(&mut buf[..want])?;
|
||||
decoder.push(&buf[..want]);
|
||||
off += want as u64;
|
||||
}
|
||||
Ok((decoder.into_ids(), size))
|
||||
}
|
||||
|
||||
/// The ids of `incoming` that `has` does not report, ascending, so a merge
|
||||
/// appends deterministic output.
|
||||
pub(crate) fn ecj_delta(
|
||||
incoming: &HashSet<NeedleId>,
|
||||
has: impl Fn(&NeedleId) -> bool,
|
||||
) -> Vec<NeedleId> {
|
||||
let mut delta: Vec<NeedleId> = incoming.iter().copied().filter(|id| !has(id)).collect();
|
||||
delta.sort_unstable();
|
||||
delta
|
||||
}
|
||||
|
||||
pub(crate) fn encode_ecj_ids(ids: &[NeedleId]) -> Vec<u8> {
|
||||
let mut buf = vec![0u8; ids.len() * NEEDLE_ID_SIZE];
|
||||
for (record, id) in buf.chunks_exact_mut(NEEDLE_ID_SIZE).zip(ids) {
|
||||
id.to_bytes(record);
|
||||
}
|
||||
buf
|
||||
}
|
||||
|
||||
/// Append to the journal at `path` the ids of `incoming` that `local` lacks,
|
||||
/// in one write and one fsync, returning how many were added. `local` and
|
||||
/// `size` come from [`read_ecj_ids`] on the same path: if the journal's
|
||||
/// whole-record length is no longer `size`, returns `Ok(None)` so the caller
|
||||
/// re-reads. A torn tail past `size` is truncated first so the new records
|
||||
/// stay aligned.
|
||||
pub(crate) fn append_ecj_ids(
|
||||
path: &str,
|
||||
local: &HashSet<NeedleId>,
|
||||
incoming: &HashSet<NeedleId>,
|
||||
size: u64,
|
||||
) -> io::Result<Option<usize>> {
|
||||
let delta = ecj_delta(incoming, |id| local.contains(id));
|
||||
if delta.is_empty() {
|
||||
return Ok(Some(0));
|
||||
}
|
||||
let created = !Path::new(path).exists();
|
||||
let mut file = open_volume_file(OpenOptions::new().read(true).write(true).create(true), path)?;
|
||||
let len = file.metadata()?.len();
|
||||
if len - len % NEEDLE_ID_SIZE as u64 != size {
|
||||
return Ok(None);
|
||||
}
|
||||
if len != size {
|
||||
file.set_len(size)?;
|
||||
}
|
||||
let appended = file
|
||||
.seek(SeekFrom::Start(size))
|
||||
.and_then(|_| file.write_all(&encode_ecj_ids(&delta)))
|
||||
.and_then(|_| file.sync_all());
|
||||
if let Err(e) = appended {
|
||||
let _ = file.set_len(size);
|
||||
return Err(e);
|
||||
}
|
||||
if created {
|
||||
fsync_dir(path)?;
|
||||
}
|
||||
Ok(Some(delta.len()))
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
fn ids(v: &[u64]) -> HashSet<NeedleId> {
|
||||
v.iter().map(|&id| NeedleId(id)).collect()
|
||||
}
|
||||
|
||||
fn bytes(v: &[u64]) -> Vec<u8> {
|
||||
encode_ecj_ids(&v.iter().map(|&id| NeedleId(id)).collect::<Vec<_>>())
|
||||
}
|
||||
|
||||
fn records(path: &str) -> Vec<u64> {
|
||||
let data = fs::read(path).expect("read ecj");
|
||||
assert_eq!(data.len() % NEEDLE_ID_SIZE, 0, "journal must stay aligned");
|
||||
data.chunks_exact(NEEDLE_ID_SIZE)
|
||||
.map(|c| NeedleId::from_bytes(c).0)
|
||||
.collect()
|
||||
}
|
||||
|
||||
/// Run the unmounted merge the way the server does: read, then append.
|
||||
fn merge_file(path: &str, incoming: &HashSet<NeedleId>) -> usize {
|
||||
let (local, size) = read_ecj_ids(path).expect("read");
|
||||
append_ecj_ids(path, &local, incoming, size)
|
||||
.expect("append")
|
||||
.expect("journal unchanged")
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn decoder_handles_records_split_across_chunks() {
|
||||
let mut stream = bytes(&[1, 2, 3, 2, 0x0102030405060708]);
|
||||
stream.extend_from_slice(&[9, 9, 9]);
|
||||
for chunk in 1..=stream.len() {
|
||||
let mut d = EcjIdDecoder::default();
|
||||
for piece in stream.chunks(chunk) {
|
||||
d.push(piece);
|
||||
}
|
||||
assert_eq!(
|
||||
d.into_ids(),
|
||||
ids(&[1, 2, 3, 0x0102030405060708]),
|
||||
"chunk {}",
|
||||
chunk
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn read_ecj_ids_dedups_and_ignores_torn_tail() {
|
||||
let dir = tempfile::tempdir().expect("tempdir");
|
||||
let missing = dir.path().join("missing.ecj");
|
||||
let (got, size) = read_ecj_ids(missing.to_str().unwrap()).expect("read");
|
||||
assert!(got.is_empty());
|
||||
assert_eq!(size, 0);
|
||||
|
||||
let torn = dir.path().join("torn.ecj");
|
||||
let mut data = bytes(&[1, 2, 1]);
|
||||
data.extend_from_slice(&[7, 7, 7]);
|
||||
fs::write(&torn, data).unwrap();
|
||||
let (got, size) = read_ecj_ids(torn.to_str().unwrap()).expect("read");
|
||||
assert_eq!(got, ids(&[1, 2]));
|
||||
assert_eq!(size, 3 * NEEDLE_ID_SIZE as u64);
|
||||
|
||||
// A bloated journal repeating a few ids across several read chunks
|
||||
// keeps only the distinct ids.
|
||||
let bloated = dir.path().join("bloated.ecj");
|
||||
let data: Vec<u8> = (0..300_000u64).flat_map(|i| bytes(&[i % 3])).collect();
|
||||
fs::write(&bloated, &data).unwrap();
|
||||
let (got, size) = read_ecj_ids(bloated.to_str().unwrap()).expect("read");
|
||||
assert_eq!(got, ids(&[0, 1, 2]));
|
||||
assert_eq!(size, data.len() as u64);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn appends_only_missing_ids() {
|
||||
let dir = tempfile::tempdir().expect("tempdir");
|
||||
let path = dir.path().join("vol.ecj");
|
||||
let path = path.to_str().unwrap();
|
||||
fs::write(path, bytes(&[1, 2, 3])).unwrap();
|
||||
assert_eq!(merge_file(path, &ids(&[3, 4])), 1);
|
||||
assert_eq!(records(path), vec![1, 2, 3, 4]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn round_trip_stays_constant() {
|
||||
// A->B->A->B 20 times: the old append path doubled the journal each trip.
|
||||
let dir = tempfile::tempdir().expect("tempdir");
|
||||
let a = dir.path().join("a.ecj");
|
||||
let b = dir.path().join("b.ecj");
|
||||
let (a, b) = (a.to_str().unwrap(), b.to_str().unwrap());
|
||||
fs::write(a, bytes(&[1, 2])).unwrap();
|
||||
fs::write(b, bytes(&[2, 3])).unwrap();
|
||||
for i in 0..20 {
|
||||
let (src, dst) = if i % 2 == 0 { (a, b) } else { (b, a) };
|
||||
let (incoming, _) = read_ecj_ids(src).expect("read");
|
||||
merge_file(dst, &incoming);
|
||||
}
|
||||
for path in [a, b] {
|
||||
let mut got = records(path);
|
||||
got.sort_unstable();
|
||||
assert_eq!(got, vec![1, 2, 3]);
|
||||
}
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn nothing_new_leaves_journal_alone() {
|
||||
let dir = tempfile::tempdir().expect("tempdir");
|
||||
let missing = dir.path().join("missing.ecj");
|
||||
assert_eq!(merge_file(missing.to_str().unwrap(), &ids(&[])), 0);
|
||||
assert!(
|
||||
!missing.exists(),
|
||||
"an empty merge must not create a journal"
|
||||
);
|
||||
|
||||
let path = dir.path().join("vol.ecj");
|
||||
let path = path.to_str().unwrap();
|
||||
fs::write(path, bytes(&[1, 2])).unwrap();
|
||||
assert_eq!(merge_file(path, &ids(&[2, 1])), 0);
|
||||
assert_eq!(records(path), vec![1, 2]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn repairs_torn_tail() {
|
||||
let dir = tempfile::tempdir().expect("tempdir");
|
||||
let path = dir.path().join("vol.ecj");
|
||||
let path = path.to_str().unwrap();
|
||||
let mut data = bytes(&[1, 2]);
|
||||
data.extend_from_slice(&[9, 9, 9]);
|
||||
fs::write(path, data).unwrap();
|
||||
assert_eq!(merge_file(path, &ids(&[3])), 1);
|
||||
assert_eq!(records(path), vec![1, 2, 3]);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn rejects_changed_journal() {
|
||||
let dir = tempfile::tempdir().expect("tempdir");
|
||||
let path = dir.path().join("vol.ecj");
|
||||
let path = path.to_str().unwrap();
|
||||
fs::write(path, bytes(&[1])).unwrap();
|
||||
let (local, size) = read_ecj_ids(path).expect("read");
|
||||
fs::write(path, bytes(&[1, 5])).unwrap();
|
||||
let outcome = append_ecj_ids(path, &local, &ids(&[2]), size).expect("append");
|
||||
assert_eq!(outcome, None);
|
||||
assert_eq!(records(path), vec![1, 5]);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,326 @@
|
||||
//! Process-wide coordination of everything that touches one `.ecj` path.
|
||||
//!
|
||||
//! Mount-time compaction replaces a deletion journal with a new inode. That is
|
||||
//! only safe while nothing else in this process can write the old one:
|
||||
//!
|
||||
//! - **Holders** are mounted `EcVolume`s with an append handle on the path. A
|
||||
//! store holds one `EcVolume` per disk location, and a shard mount or a
|
||||
//! cross-disk reconcile can point one disk's volume at another disk's
|
||||
//! `.ecj`, so several holders of one path are normal. A holder that keeps
|
||||
//! appending to a replaced inode acknowledges deletes that are gone at the
|
||||
//! next mount.
|
||||
//! - **Writers** append to or replace the path by name without holding it
|
||||
//! open across calls: `ReceiveFile` of an EC `.ecj`, and the unmounted
|
||||
//! append in `merge_ec_journal`, which `VolumeEcShardsCopy` and EC index
|
||||
//! recovery funnel a peer's journal through. Bytes they write after the
|
||||
//! compactor sized the journal would be dropped by the rename.
|
||||
//!
|
||||
//! Compaction therefore runs only while its caller is the sole holder and no
|
||||
//! writer is active, and while it runs no holder may open the path and no
|
||||
//! writer may start. Both wait instead; a compaction rewrites only the distinct
|
||||
//! id set, so the wait is short.
|
||||
//!
|
||||
//! No writer active at the reservation is not enough: one that ran while the
|
||||
//! holder loaded the journal, or after, and has finished may have rewritten it
|
||||
//! in place to the same length (`ReceiveFile` truncates and refills), which
|
||||
//! the inode-and-size re-check cannot see. So each writer bumps the path's
|
||||
//! write generation as it starts, and a holder may compact only if no writer
|
||||
//! was active when it registered and the generation has not moved since.
|
||||
//!
|
||||
//! Paths are keyed by their canonical parent directory, so two disk locations
|
||||
//! that spell one directory differently still meet here.
|
||||
|
||||
use std::collections::HashMap;
|
||||
use std::path::{Path, PathBuf};
|
||||
use std::sync::{Condvar, LazyLock, Mutex, MutexGuard};
|
||||
|
||||
#[derive(Default)]
|
||||
struct PathState {
|
||||
holders: usize,
|
||||
writers: usize,
|
||||
compacting: bool,
|
||||
/// Writers that have started on the path. Lives as long as the entry,
|
||||
/// which a registered holder keeps.
|
||||
write_gen: u64,
|
||||
}
|
||||
|
||||
impl PathState {
|
||||
fn idle(&self) -> bool {
|
||||
self.holders == 0 && self.writers == 0 && !self.compacting
|
||||
}
|
||||
}
|
||||
|
||||
struct Registry {
|
||||
paths: Mutex<HashMap<PathBuf, PathState>>,
|
||||
changed: Condvar,
|
||||
}
|
||||
|
||||
static REGISTRY: LazyLock<Registry> = LazyLock::new(|| Registry {
|
||||
paths: Mutex::new(HashMap::new()),
|
||||
changed: Condvar::new(),
|
||||
});
|
||||
|
||||
fn lock() -> MutexGuard<'static, HashMap<PathBuf, PathState>> {
|
||||
// The critical sections only adjust counters and cannot panic midway, so
|
||||
// a poisoned lock still guards consistent state.
|
||||
REGISTRY.paths.lock().unwrap_or_else(|e| e.into_inner())
|
||||
}
|
||||
|
||||
/// Canonical key for `path`: its resolved parent directory joined with the
|
||||
/// file name. The file itself may not exist yet (a copy creates it), so only
|
||||
/// the directory is resolved.
|
||||
fn key_for(path: &str) -> PathBuf {
|
||||
let p = Path::new(path);
|
||||
let (Some(parent), Some(name)) = (p.parent(), p.file_name()) else {
|
||||
return std::path::absolute(p).unwrap_or_else(|_| p.to_path_buf());
|
||||
};
|
||||
let parent = if parent.as_os_str().is_empty() {
|
||||
Path::new(".")
|
||||
} else {
|
||||
parent
|
||||
};
|
||||
let dir = std::fs::canonicalize(parent)
|
||||
.or_else(|_| std::path::absolute(parent))
|
||||
.unwrap_or_else(|_| parent.to_path_buf());
|
||||
dir.join(name)
|
||||
}
|
||||
|
||||
/// Block until no compaction is running on `key`, then apply `f` to its state.
|
||||
fn update_when_not_compacting<R>(key: &Path, f: impl FnOnce(&mut PathState) -> R) -> R {
|
||||
let mut paths = lock();
|
||||
while paths.get(key).is_some_and(|s| s.compacting) {
|
||||
paths = REGISTRY
|
||||
.changed
|
||||
.wait(paths)
|
||||
.unwrap_or_else(|e| e.into_inner());
|
||||
}
|
||||
f(paths.entry(key.to_path_buf()).or_default())
|
||||
}
|
||||
|
||||
fn release(key: &Path, f: impl FnOnce(&mut PathState)) {
|
||||
let mut paths = lock();
|
||||
if let Some(state) = paths.get_mut(key) {
|
||||
f(state);
|
||||
if state.idle() {
|
||||
paths.remove(key);
|
||||
}
|
||||
}
|
||||
drop(paths);
|
||||
REGISTRY.changed.notify_all();
|
||||
}
|
||||
|
||||
/// A mounted `EcVolume`'s registration as a holder of its `.ecj`. Taken before
|
||||
/// the journal is opened and released when dropped.
|
||||
pub(crate) struct EcjHold {
|
||||
key: PathBuf,
|
||||
/// The path's write generation when the hold was taken, and whether a
|
||||
/// writer was active then. Taken before the journal is opened and loaded,
|
||||
/// so they cover every write the load might have missed.
|
||||
write_gen: u64,
|
||||
writer_at_start: bool,
|
||||
}
|
||||
|
||||
impl EcjHold {
|
||||
/// Register as a holder of `ecj_path`, first waiting out any compaction in
|
||||
/// progress so the handle opened afterwards is on the final inode.
|
||||
pub(crate) fn acquire(ecj_path: &str) -> Self {
|
||||
let key = key_for(ecj_path);
|
||||
let (write_gen, writer_at_start) = update_when_not_compacting(&key, |s| {
|
||||
s.holders += 1;
|
||||
(s.write_gen, s.writers > 0)
|
||||
});
|
||||
EcjHold {
|
||||
key,
|
||||
write_gen,
|
||||
writer_at_start,
|
||||
}
|
||||
}
|
||||
|
||||
/// Reserve the path for a compaction, or `None` when another holder or an
|
||||
/// active writer could still reach the current inode, or when a writer
|
||||
/// has run on the path since the hold was taken, so the journal may no
|
||||
/// longer be what the holder loaded.
|
||||
pub(crate) fn try_begin_compaction(&self) -> Option<EcjCompaction> {
|
||||
let mut paths = lock();
|
||||
let state = paths.get_mut(&self.key)?;
|
||||
if state.holders != 1 || state.writers != 0 || state.compacting {
|
||||
return None;
|
||||
}
|
||||
if self.writer_at_start || state.write_gen != self.write_gen {
|
||||
return None;
|
||||
}
|
||||
state.compacting = true;
|
||||
Some(EcjCompaction {
|
||||
key: self.key.clone(),
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
impl Drop for EcjHold {
|
||||
fn drop(&mut self) {
|
||||
release(&self.key, |s| s.holders = s.holders.saturating_sub(1));
|
||||
}
|
||||
}
|
||||
|
||||
/// An exclusive reservation of a `.ecj` path for compaction. Holders and
|
||||
/// writers wait until it is dropped.
|
||||
pub(crate) struct EcjCompaction {
|
||||
key: PathBuf,
|
||||
}
|
||||
|
||||
impl Drop for EcjCompaction {
|
||||
fn drop(&mut self) {
|
||||
release(&self.key, |s| s.compacting = false);
|
||||
}
|
||||
}
|
||||
|
||||
/// An out-of-band writer (shard copy, index recovery, `ReceiveFile`) on a
|
||||
/// `.ecj` path.
|
||||
/// Compaction does not start while one is alive.
|
||||
pub(crate) struct EcjWrite {
|
||||
key: PathBuf,
|
||||
}
|
||||
|
||||
impl Drop for EcjWrite {
|
||||
fn drop(&mut self) {
|
||||
release(&self.key, |s| s.writers = s.writers.saturating_sub(1));
|
||||
}
|
||||
}
|
||||
|
||||
/// Register as a writer of `ecj_path`, waiting out any compaction in progress.
|
||||
/// Blocks; async callers use [`begin_ecj_write_async`].
|
||||
pub(crate) fn begin_ecj_write(ecj_path: &str) -> EcjWrite {
|
||||
let key = key_for(ecj_path);
|
||||
update_when_not_compacting(&key, |s| {
|
||||
s.writers += 1;
|
||||
s.write_gen += 1;
|
||||
});
|
||||
EcjWrite { key }
|
||||
}
|
||||
|
||||
/// [`begin_ecj_write`] for async handlers: the wait runs on the blocking pool
|
||||
/// so a compaction in progress never stalls a runtime worker.
|
||||
pub(crate) async fn begin_ecj_write_async(ecj_path: &str) -> EcjWrite {
|
||||
let path = ecj_path.to_string();
|
||||
match tokio::task::spawn_blocking(move || begin_ecj_write(&path)).await {
|
||||
Ok(write) => write,
|
||||
// Only a panic inside the registry lands here, and it leaves no count
|
||||
// behind; registering inline is still correct, merely blocking.
|
||||
Err(_) => begin_ecj_write(ecj_path),
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use std::sync::mpsc;
|
||||
use std::time::Duration;
|
||||
use tempfile::TempDir;
|
||||
|
||||
fn ecj(dir: &TempDir) -> String {
|
||||
dir.path().join("1.ecj").to_str().unwrap().to_string()
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn sole_holder_may_compact() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let hold = EcjHold::acquire(&ecj(&dir));
|
||||
assert!(hold.try_begin_compaction().is_some());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn second_holder_blocks_compaction() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let a = EcjHold::acquire(&ecj(&dir));
|
||||
let b = EcjHold::acquire(&ecj(&dir));
|
||||
assert!(a.try_begin_compaction().is_none());
|
||||
drop(b);
|
||||
assert!(a.try_begin_compaction().is_some());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn active_writer_blocks_compaction() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let hold = EcjHold::acquire(&ecj(&dir));
|
||||
let w = begin_ecj_write(&ecj(&dir));
|
||||
assert!(hold.try_begin_compaction().is_none());
|
||||
drop(w);
|
||||
// This hold loaded before the write; a later one may compact.
|
||||
drop(hold);
|
||||
let hold = EcjHold::acquire(&ecj(&dir));
|
||||
assert!(hold.try_begin_compaction().is_some());
|
||||
}
|
||||
|
||||
/// A writer that ran after the hold was taken, or was already running
|
||||
/// then, may have changed the journal the holder loaded, even though it
|
||||
/// has finished by the time compaction asks.
|
||||
#[test]
|
||||
fn finished_writer_since_hold_blocks_compaction() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let path = ecj(&dir);
|
||||
|
||||
let hold = EcjHold::acquire(&path);
|
||||
drop(begin_ecj_write(&path));
|
||||
assert!(
|
||||
hold.try_begin_compaction().is_none(),
|
||||
"a writer that started after the hold",
|
||||
);
|
||||
drop(hold);
|
||||
|
||||
let w = begin_ecj_write(&path);
|
||||
let hold = EcjHold::acquire(&path);
|
||||
drop(w);
|
||||
assert!(
|
||||
hold.try_begin_compaction().is_none(),
|
||||
"a writer active when the hold was taken",
|
||||
);
|
||||
drop(hold);
|
||||
|
||||
let hold = EcjHold::acquire(&path);
|
||||
assert!(
|
||||
hold.try_begin_compaction().is_some(),
|
||||
"no writer since the hold",
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn differently_spelled_paths_share_one_key() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
std::fs::create_dir(dir.path().join("sub")).unwrap();
|
||||
let plain = ecj(&dir);
|
||||
let dotted = dir
|
||||
.path()
|
||||
.join("sub")
|
||||
.join("..")
|
||||
.join("1.ecj")
|
||||
.to_str()
|
||||
.unwrap()
|
||||
.to_string();
|
||||
let a = EcjHold::acquire(&plain);
|
||||
let _b = EcjHold::acquire(&dotted);
|
||||
assert!(a.try_begin_compaction().is_none());
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn writer_waits_for_compaction_to_finish() {
|
||||
let dir = TempDir::new().unwrap();
|
||||
let path = ecj(&dir);
|
||||
let hold = EcjHold::acquire(&path);
|
||||
let compaction = hold.try_begin_compaction().unwrap();
|
||||
|
||||
let (tx, rx) = mpsc::channel();
|
||||
let p = path.clone();
|
||||
let t = std::thread::spawn(move || {
|
||||
let _w = begin_ecj_write(&p);
|
||||
tx.send(()).unwrap();
|
||||
});
|
||||
assert!(
|
||||
rx.recv_timeout(Duration::from_millis(100)).is_err(),
|
||||
"a writer must not start while a compaction holds the path",
|
||||
);
|
||||
drop(compaction);
|
||||
rx.recv_timeout(Duration::from_secs(5))
|
||||
.expect("writer must proceed once the compaction ends");
|
||||
t.join().unwrap();
|
||||
}
|
||||
}
|
||||
@@ -9,6 +9,8 @@ pub mod ec_encoder;
|
||||
pub mod ec_locate;
|
||||
pub mod ec_shard;
|
||||
pub mod ec_volume;
|
||||
pub mod ecj_merge;
|
||||
pub(crate) mod ecj_registry;
|
||||
|
||||
pub use ec_shard::{
|
||||
DATA_SHARDS_COUNT, EcVolumeShard, MAX_SHARD_COUNT, MIN_TOTAL_DISKS, PARITY_SHARDS_COUNT,
|
||||
|
||||
@@ -6,6 +6,7 @@ pub(crate) mod io_error;
|
||||
pub mod needle;
|
||||
pub mod needle_map;
|
||||
pub mod store;
|
||||
pub mod store_ec_journal;
|
||||
pub mod store_ec_mirror;
|
||||
pub mod store_ec_reconcile;
|
||||
pub mod super_block;
|
||||
|
||||
@@ -195,7 +195,12 @@ impl NeedleMapKind {
|
||||
// ============================================================================
|
||||
|
||||
/// Trait for appending to an index file.
|
||||
pub trait IdxFileWriter: Write + Send + Sync {
|
||||
///
|
||||
/// The file is opened without append mode and each row is written at the
|
||||
/// current `idx_file_offset` — the same positioned-write model the Go
|
||||
/// server uses — because an append-mode handle cannot truncate on Windows,
|
||||
/// where the std library keeps it strictly append-only.
|
||||
pub trait IdxFileWriter: Write + Seek + Send + Sync {
|
||||
fn sync_all(&self) -> io::Result<()>;
|
||||
/// Truncate the file to `len` bytes. Used to remove an orphan .idx row
|
||||
/// left by a failed redb commit so `idx_file_offset` stays a contiguous
|
||||
@@ -224,6 +229,9 @@ pub struct CompactNeedleMap {
|
||||
metric: NeedleMapMetric,
|
||||
idx_file: Option<Box<dyn IdxFileWriter>>,
|
||||
idx_file_offset: u64,
|
||||
/// The file holds bytes past `idx_file_offset` that must be trimmed
|
||||
/// before another row can land aligned.
|
||||
idx_torn: bool,
|
||||
}
|
||||
|
||||
impl Default for CompactNeedleMap {
|
||||
@@ -240,6 +248,7 @@ impl CompactNeedleMap {
|
||||
metric: NeedleMapMetric::default(),
|
||||
idx_file: None,
|
||||
idx_file_offset: 0,
|
||||
idx_torn: false,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -264,6 +273,7 @@ impl CompactNeedleMap {
|
||||
pub fn set_idx_file(&mut self, file: Box<dyn IdxFileWriter>, offset: u64) {
|
||||
self.idx_file = Some(file);
|
||||
self.idx_file_offset = offset;
|
||||
self.idx_torn = false;
|
||||
}
|
||||
|
||||
/// True when an .idx file writer is attached. A read-only load leaves
|
||||
@@ -278,8 +288,8 @@ impl CompactNeedleMap {
|
||||
/// Insert or update an entry. Appends to .idx file if present.
|
||||
pub fn put(&mut self, key: NeedleId, offset: Offset, size: Size) -> io::Result<()> {
|
||||
// Persist to idx file BEFORE mutating in-memory state for crash consistency
|
||||
if let Some(ref mut idx_file) = self.idx_file {
|
||||
idx::write_index_entry(idx_file, key, offset, size)?;
|
||||
self.append_to_index_file(key, offset, size)?;
|
||||
if self.idx_file.is_some() {
|
||||
self.idx_file_offset += NEEDLE_MAP_ENTRY_SIZE as u64;
|
||||
}
|
||||
|
||||
@@ -289,6 +299,41 @@ impl CompactNeedleMap {
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Write one row to the .idx file at `idx_file_offset`. A row left
|
||||
/// half-written by a failed write is trimmed back to the offset so the
|
||||
/// next row still lands aligned; while the trim keeps failing no row is
|
||||
/// written at all, or it would sit off alignment and parse as garbage on
|
||||
/// load. The offset itself is advanced by the caller once the row counts.
|
||||
fn append_to_index_file(
|
||||
&mut self,
|
||||
key: NeedleId,
|
||||
offset: Offset,
|
||||
size: Size,
|
||||
) -> io::Result<()> {
|
||||
let Some(idx_file) = self.idx_file.as_mut() else {
|
||||
return Ok(());
|
||||
};
|
||||
if self.idx_torn {
|
||||
match idx_file.truncate_to(self.idx_file_offset) {
|
||||
Ok(()) => self.idx_torn = false,
|
||||
Err(e) => {
|
||||
return Err(io::Error::other(format!(
|
||||
"index file still holds a torn row: {e}"
|
||||
)));
|
||||
}
|
||||
}
|
||||
}
|
||||
idx_file.seek(io::SeekFrom::Start(self.idx_file_offset))?;
|
||||
if let Err(e) = idx::write_index_entry(idx_file, key, offset, size) {
|
||||
if let Err(te) = idx_file.truncate_to(self.idx_file_offset) {
|
||||
self.idx_torn = true;
|
||||
tracing::warn!("failed to trim torn .idx row: {}", te);
|
||||
}
|
||||
return Err(e);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Look up a needle.
|
||||
pub fn get(&self, key: NeedleId) -> Option<NeedleValue> {
|
||||
self.map.get(key)
|
||||
@@ -313,8 +358,8 @@ impl CompactNeedleMap {
|
||||
}
|
||||
|
||||
// Always write tombstone to idx file (matching Go)
|
||||
if let Some(ref mut idx_file) = self.idx_file {
|
||||
idx::write_index_entry(idx_file, key, offset, TOMBSTONE_FILE_SIZE)?;
|
||||
self.append_to_index_file(key, offset, TOMBSTONE_FILE_SIZE)?;
|
||||
if self.idx_file.is_some() {
|
||||
self.idx_file_offset += NEEDLE_MAP_ENTRY_SIZE as u64;
|
||||
}
|
||||
|
||||
@@ -464,6 +509,9 @@ pub struct RedbNeedleMap {
|
||||
metric: NeedleMapMetric,
|
||||
idx_file: Option<Box<dyn IdxFileWriter>>,
|
||||
idx_file_offset: u64,
|
||||
/// The file holds bytes past `idx_file_offset` that must be trimmed
|
||||
/// before another row can land aligned.
|
||||
idx_torn: bool,
|
||||
/// Puts/deletes since the last durable checkpoint.
|
||||
writes_since_checkpoint: u32,
|
||||
}
|
||||
@@ -570,6 +618,7 @@ impl RedbNeedleMap {
|
||||
metric: NeedleMapMetric::default(),
|
||||
idx_file: None,
|
||||
idx_file_offset: 0,
|
||||
idx_torn: false,
|
||||
writes_since_checkpoint: 0,
|
||||
})
|
||||
}
|
||||
@@ -667,6 +716,7 @@ impl RedbNeedleMap {
|
||||
metric: NeedleMapMetric::default(),
|
||||
idx_file: None,
|
||||
idx_file_offset: 0,
|
||||
idx_torn: false,
|
||||
writes_since_checkpoint: 0,
|
||||
};
|
||||
|
||||
@@ -858,6 +908,7 @@ impl RedbNeedleMap {
|
||||
pub fn set_idx_file(&mut self, file: Box<dyn IdxFileWriter>, offset: u64) {
|
||||
self.idx_file = Some(file);
|
||||
self.idx_file_offset = offset;
|
||||
self.idx_torn = false;
|
||||
}
|
||||
|
||||
/// True when an .idx file writer is attached. See CompactNeedleMap.
|
||||
@@ -874,9 +925,7 @@ impl RedbNeedleMap {
|
||||
// commit leaves an orphan row in .idx that redb doesn't reflect, and
|
||||
// advancing the offset here would let a later checkpoint record it as
|
||||
// reflected, making the reload skip it permanently.
|
||||
if let Some(ref mut idx_file) = self.idx_file {
|
||||
idx::write_index_entry(idx_file, key, offset, size)?;
|
||||
}
|
||||
self.append_to_index_file(key, offset, size)?;
|
||||
|
||||
let key_u64: u64 = key.into();
|
||||
let packed = pack_needle_value(&NeedleValue { offset, size });
|
||||
@@ -947,6 +996,41 @@ impl RedbNeedleMap {
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Write one row to the .idx file at `idx_file_offset`. A row left
|
||||
/// half-written by a failed write is trimmed back to the offset so the
|
||||
/// next row still lands aligned; while the trim keeps failing no row is
|
||||
/// written at all, or it would sit off alignment and parse as garbage on
|
||||
/// load. The offset itself is advanced by the caller once the row counts.
|
||||
fn append_to_index_file(
|
||||
&mut self,
|
||||
key: NeedleId,
|
||||
offset: Offset,
|
||||
size: Size,
|
||||
) -> io::Result<()> {
|
||||
let Some(idx_file) = self.idx_file.as_mut() else {
|
||||
return Ok(());
|
||||
};
|
||||
if self.idx_torn {
|
||||
match idx_file.truncate_to(self.idx_file_offset) {
|
||||
Ok(()) => self.idx_torn = false,
|
||||
Err(e) => {
|
||||
return Err(io::Error::other(format!(
|
||||
"index file still holds a torn row: {e}"
|
||||
)));
|
||||
}
|
||||
}
|
||||
}
|
||||
idx_file.seek(io::SeekFrom::Start(self.idx_file_offset))?;
|
||||
if let Err(e) = idx::write_index_entry(idx_file, key, offset, size) {
|
||||
if let Err(te) = idx_file.truncate_to(self.idx_file_offset) {
|
||||
self.idx_torn = true;
|
||||
tracing::warn!("failed to trim torn .idx row: {}", te);
|
||||
}
|
||||
return Err(e);
|
||||
}
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Look up a needle. A redb failure is an ERROR, not an absent needle:
|
||||
/// answering "not found" would turn a database problem into a read miss
|
||||
/// and let a delete report success without recording a tombstone.
|
||||
@@ -993,9 +1077,7 @@ impl RedbNeedleMap {
|
||||
return Ok(None);
|
||||
};
|
||||
|
||||
if let Some(ref mut idx_file) = self.idx_file {
|
||||
idx::write_index_entry(idx_file, key, offset, TOMBSTONE_FILE_SIZE)?;
|
||||
}
|
||||
self.append_to_index_file(key, offset, TOMBSTONE_FILE_SIZE)?;
|
||||
|
||||
let deleted_nv = NeedleValue {
|
||||
offset: old.offset,
|
||||
@@ -1085,10 +1167,13 @@ impl RedbNeedleMap {
|
||||
/// a failed redb commit. Without this the next successful write appends
|
||||
/// after the orphan, `idx_file_offset` advances past it, and a later
|
||||
/// checkpoint records an offset that makes the reload skip the orphan.
|
||||
/// When the trim fails the file is latched torn so no later row lands
|
||||
/// after bytes the map does not reflect.
|
||||
fn truncate_idx_to_offset(&mut self) {
|
||||
if let Some(ref mut idx_file) = self.idx_file
|
||||
&& let Err(e) = idx_file.truncate_to(self.idx_file_offset)
|
||||
{
|
||||
self.idx_torn = true;
|
||||
tracing::warn!("failed to truncate orphan .idx row: {}", e);
|
||||
}
|
||||
}
|
||||
@@ -1139,6 +1224,7 @@ impl RedbNeedleMap {
|
||||
self.db = reopened.db;
|
||||
self.metric = reopened.metric;
|
||||
self.idx_file_offset = actual_idx_size;
|
||||
self.idx_torn = reopened.idx_torn;
|
||||
// The reopen replayed all rows since the last durable checkpoint
|
||||
// non-durably; start the counter fresh.
|
||||
self.writes_since_checkpoint = 0;
|
||||
@@ -1691,7 +1777,7 @@ mod tests {
|
||||
)
|
||||
.unwrap();
|
||||
let writer = std::fs::OpenOptions::new()
|
||||
.append(true)
|
||||
.write(true)
|
||||
.open(&idx_path)
|
||||
.unwrap();
|
||||
nm.set_idx_file(Box::new(writer), idx_size);
|
||||
|
||||
@@ -13,13 +13,45 @@ use crate::config::MinFreeSpace;
|
||||
use crate::pb::master_pb;
|
||||
use crate::storage::disk_location::DiskLocation;
|
||||
use crate::storage::erasure_coding::ec_shard::{EcVolumeShard, MAX_SHARD_COUNT, ShardId};
|
||||
use crate::storage::erasure_coding::ec_volume::{EcVolume, is_usable_ecx_file};
|
||||
use crate::storage::erasure_coding::ec_volume::{
|
||||
ECJ_COMPACT_TMP_EXT, EcVolume, is_usable_ecx_file,
|
||||
};
|
||||
use crate::storage::needle::needle::{Needle, get_actual_size};
|
||||
use crate::storage::needle_map::NeedleMapKind;
|
||||
use crate::storage::super_block::{ReplicaPlacement, SUPER_BLOCK_SIZE};
|
||||
use crate::storage::types::*;
|
||||
use crate::storage::volume::{CompactionJob, VifVolumeInfo, Volume, VolumeError, VolumeSpec};
|
||||
|
||||
/// Mirrors Go's ensureCompactVolumeSpace, per filesystem: the new .dat lands
|
||||
/// next to the old one and the new .idx next to the old index, so when the
|
||||
/// index directory is on another filesystem each disk answers for its own
|
||||
/// share, while two directories on one filesystem must cover the sum.
|
||||
fn ensure_compact_volume_space(v: &Volume, preallocate: u64) -> Result<(), VolumeError> {
|
||||
let (data_bytes, index_bytes) = compaction_space_needed(v, preallocate);
|
||||
let (dir, dir_idx) = (v.dir(), v.dir_idx());
|
||||
let check = |dir: &str, needed: u64| -> Result<(), VolumeError> {
|
||||
let (_, free) = crate::storage::disk_location::get_disk_stats(dir);
|
||||
if free < needed {
|
||||
return Err(VolumeError::InsufficientSpace {
|
||||
vid: v.id,
|
||||
required: needed,
|
||||
free,
|
||||
});
|
||||
}
|
||||
Ok(())
|
||||
};
|
||||
|
||||
if dir_idx.is_empty() || dir_idx == dir {
|
||||
check(dir, data_bytes + index_bytes)
|
||||
} else if !same_filesystem(dir, dir_idx) {
|
||||
check(dir, data_bytes)?;
|
||||
check(dir_idx, index_bytes)
|
||||
} else {
|
||||
check(dir, data_bytes + index_bytes)?;
|
||||
check(dir_idx, index_bytes)
|
||||
}
|
||||
}
|
||||
|
||||
/// Top-level storage manager containing all disk locations and their volumes.
|
||||
pub struct Store {
|
||||
pub locations: Vec<DiskLocation>,
|
||||
@@ -726,17 +758,6 @@ impl Store {
|
||||
vol.needle_read_plan(id, read_deleted)
|
||||
}
|
||||
|
||||
/// Re-lookup a needle's data-file offset after compaction may have moved it.
|
||||
/// Returns `(new_data_file_offset, current_compaction_revision)`.
|
||||
pub fn re_lookup_needle_data_offset(
|
||||
&self,
|
||||
vid: VolumeId,
|
||||
needle_id: NeedleId,
|
||||
) -> Result<(u64, u16), VolumeError> {
|
||||
let (_, vol) = self.find_volume(vid).ok_or(VolumeError::NotFound)?;
|
||||
vol.re_lookup_needle_data_offset(needle_id)
|
||||
}
|
||||
|
||||
/// Write a needle to a volume. With `fsync` the volume flushes its .dat
|
||||
/// before returning, so the caller can ack a durable write.
|
||||
pub fn write_volume_needle(
|
||||
@@ -1445,10 +1466,12 @@ impl Store {
|
||||
crate::storage::volume::volume_file_name(&loc.directory, collection, vid);
|
||||
let _ = std::fs::remove_file(format!("{}.ecx", idx_base));
|
||||
let _ = std::fs::remove_file(format!("{}.ecj", idx_base));
|
||||
let _ = std::fs::remove_file(format!("{}{}", idx_base, ECJ_COMPACT_TMP_EXT));
|
||||
// Also try data directory in case .ecx/.ecj were created before -dir.idx
|
||||
if loc.idx_directory != loc.directory {
|
||||
let _ = std::fs::remove_file(format!("{}.ecx", data_base));
|
||||
let _ = std::fs::remove_file(format!("{}.ecj", data_base));
|
||||
let _ = std::fs::remove_file(format!("{}{}", data_base, ECJ_COMPACT_TMP_EXT));
|
||||
}
|
||||
// A shard-only disk also drops its stale .vif (Go
|
||||
// removeEcSharedIndexFiles): a live .idx means this disk still
|
||||
@@ -1564,33 +1587,10 @@ impl Store {
|
||||
vid: VolumeId,
|
||||
preallocate: u64,
|
||||
) -> Result<Option<CompactionJob>, VolumeError> {
|
||||
// Required space matches Go's CompactVolume check: the larger of the
|
||||
// requested preallocation and the estimated compacted size — the live
|
||||
// needles, not the .dat the garbage already occupies, so a full disk
|
||||
// can still be reclaimed.
|
||||
let (loc_idx, space_needed) = {
|
||||
let (loc_idx, v) = self
|
||||
.find_volume(vid)
|
||||
.ok_or(VolumeError::VolumeNotFound(vid))?;
|
||||
let live_count = (v.file_count() - v.deleted_count()).max(0) as u64;
|
||||
let live_bytes = v.content_size().saturating_sub(v.deleted_size());
|
||||
let per_needle = (get_actual_size(Size(0), v.version())
|
||||
+ NEEDLE_PADDING_SIZE as i64
|
||||
+ NEEDLE_MAP_ENTRY_SIZE as i64) as u64;
|
||||
let estimated = SUPER_BLOCK_SIZE as u64 + live_count * per_needle + live_bytes;
|
||||
let space_needed = std::cmp::max(preallocate, estimated);
|
||||
(loc_idx, space_needed + space_needed / 10)
|
||||
};
|
||||
|
||||
let dir = self.locations[loc_idx].directory.clone();
|
||||
let (_, free) = crate::storage::disk_location::get_disk_stats(&dir);
|
||||
if free < space_needed {
|
||||
return Err(VolumeError::InsufficientSpace {
|
||||
vid,
|
||||
required: space_needed,
|
||||
free,
|
||||
});
|
||||
}
|
||||
let (_, v) = self
|
||||
.find_volume(vid)
|
||||
.ok_or(VolumeError::VolumeNotFound(vid))?;
|
||||
ensure_compact_volume_space(v, preallocate)?;
|
||||
|
||||
let (_, v) = self
|
||||
.find_volume_mut(vid)
|
||||
@@ -1598,6 +1598,37 @@ impl Store {
|
||||
v.begin_compact_by_index()
|
||||
}
|
||||
|
||||
/// Rewrite the volume in `dir`/`dir_idx`, which is not mounted, with its
|
||||
/// live needles only. Go's `Store.CompactVolumeFiles`.
|
||||
pub fn compact_volume_files(
|
||||
dir: &str,
|
||||
dir_idx: &str,
|
||||
collection: &str,
|
||||
vid: VolumeId,
|
||||
needle_map_kind: NeedleMapKind,
|
||||
) -> Result<(), VolumeError> {
|
||||
let spec = VolumeSpec {
|
||||
collection,
|
||||
..VolumeSpec::default()
|
||||
};
|
||||
let mut v = Volume::new(dir, dir_idx, vid, needle_map_kind, &spec)?;
|
||||
let mut compact = || -> Result<(), VolumeError> {
|
||||
ensure_compact_volume_space(&v, 0)?;
|
||||
v.compact_by_index(0, 0, |_| true)?;
|
||||
v.commit_compact()
|
||||
};
|
||||
let result = compact();
|
||||
if result.is_err() {
|
||||
// A failed commit may have swapped only one of .dat/.idx;
|
||||
// reconcile rolls a decided swap forward or removes orphan
|
||||
// temp files before this volume can mount a mismatched pair.
|
||||
let _ = v.reconcile_compact_state();
|
||||
let _ = v.cleanup_compact();
|
||||
}
|
||||
v.close();
|
||||
result
|
||||
}
|
||||
|
||||
/// Commit a completed compaction: swap files and reload.
|
||||
pub fn commit_compact_volume(&mut self, vid: VolumeId) -> Result<(bool, u64), VolumeError> {
|
||||
let (_, v) = self
|
||||
@@ -1762,6 +1793,65 @@ fn owned_ec_shard_count(loc: &DiskLocation, vid: VolumeId, shard_ids: &[ShardId]
|
||||
.count()
|
||||
}
|
||||
|
||||
/// Mirrors Go's compactionSpaceNeeded: what compaction will write, split into
|
||||
/// the new .dat and rebuilt .idx shares, each capped at its current file.
|
||||
fn compaction_space_needed(v: &Volume, preallocate: u64) -> (u64, u64) {
|
||||
let mut data_bytes = v.current_dat_file_size().unwrap_or(0);
|
||||
let mut index_bytes = v.idx_file_size();
|
||||
|
||||
let live_count = v.file_count() - v.deleted_count();
|
||||
let live_content = v.content_size() as i64 - v.deleted_size() as i64;
|
||||
// Unknown or inconsistent deleted sizes: the whole volume stays the
|
||||
// estimate.
|
||||
let deleted_size_known = v.deleted_count() == 0 || v.deleted_size() > 0;
|
||||
if deleted_size_known && live_count >= 0 && live_content >= 0 {
|
||||
// Empty-needle framing plus a padding unit covers the worst case.
|
||||
let per_needle =
|
||||
(get_actual_size(Size(0), v.version()) + NEEDLE_PADDING_SIZE as i64) as u64;
|
||||
let estimate = with_headroom(
|
||||
SUPER_BLOCK_SIZE as u64 + live_content as u64 + live_count as u64 * per_needle,
|
||||
);
|
||||
if estimate < data_bytes {
|
||||
data_bytes = estimate;
|
||||
}
|
||||
let estimate = with_headroom(live_count as u64 * NEEDLE_MAP_ENTRY_SIZE as u64);
|
||||
if estimate < index_bytes {
|
||||
index_bytes = estimate;
|
||||
}
|
||||
}
|
||||
if preallocate > data_bytes {
|
||||
data_bytes = preallocate;
|
||||
}
|
||||
(data_bytes, index_bytes)
|
||||
}
|
||||
|
||||
/// Headroom for Bloom-filter false positives in the live/deleted counters.
|
||||
fn with_headroom(estimate: u64) -> u64 {
|
||||
estimate + estimate / 16
|
||||
}
|
||||
|
||||
/// Whether two directories draw on the same free-space pool; in doubt, yes.
|
||||
fn same_filesystem(a: &str, b: &str) -> bool {
|
||||
if a == b {
|
||||
return true;
|
||||
}
|
||||
same_filesystem_impl(a, b)
|
||||
}
|
||||
|
||||
#[cfg(unix)]
|
||||
fn same_filesystem_impl(a: &str, b: &str) -> bool {
|
||||
use std::os::unix::fs::MetadataExt;
|
||||
match (std::fs::metadata(a), std::fs::metadata(b)) {
|
||||
(Ok(ma), Ok(mb)) => ma.dev() == mb.dev(),
|
||||
_ => true,
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(not(unix))]
|
||||
fn same_filesystem_impl(_a: &str, _b: &str) -> bool {
|
||||
true
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// Tests
|
||||
// ============================================================================
|
||||
@@ -2483,6 +2573,52 @@ mod tests {
|
||||
assert_eq!(selected, Some(0));
|
||||
}
|
||||
|
||||
// VolumeCopy picks a disk before deleting the replica it replaces, so only
|
||||
// the location actually holding that replica may count its slot as free.
|
||||
#[test]
|
||||
fn test_find_free_location_predicate_credits_only_the_holding_location() {
|
||||
let tmp1 = TempDir::new().unwrap();
|
||||
let dir1 = tmp1.path().to_str().unwrap();
|
||||
let tmp2 = TempDir::new().unwrap();
|
||||
let dir2 = tmp2.path().to_str().unwrap();
|
||||
|
||||
let mut store = Store::new(NeedleMapKind::InMemory);
|
||||
for dir in [dir1, dir2] {
|
||||
store
|
||||
.add_location(
|
||||
dir,
|
||||
dir,
|
||||
1,
|
||||
DiskType::HardDrive,
|
||||
MinFreeSpace::Percent(0.0),
|
||||
Vec::new(),
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
for vid in [81, 82] {
|
||||
store
|
||||
.add_volume(VolumeId(vid), DiskType::HardDrive, &VolumeSpec::default())
|
||||
.unwrap();
|
||||
}
|
||||
let loc_of = |vid| store.find_volume(VolumeId(vid)).unwrap().0;
|
||||
assert_ne!(loc_of(81), loc_of(82), "fixture must fill both locations");
|
||||
|
||||
let hdd = |loc: &DiskLocation| loc.disk_type == DiskType::HardDrive;
|
||||
assert_eq!(store.find_free_location_predicate(hdd, None), None);
|
||||
assert_eq!(
|
||||
store.find_free_location_predicate(hdd, Some(VolumeId(99))),
|
||||
None
|
||||
);
|
||||
assert_eq!(
|
||||
store.find_free_location_predicate(hdd, Some(VolumeId(81))),
|
||||
Some(loc_of(81))
|
||||
);
|
||||
assert_eq!(
|
||||
store.find_free_location_predicate(hdd, Some(VolumeId(82))),
|
||||
Some(loc_of(82))
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_delete_expired_ec_volumes_removes_expired_entries() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
|
||||
@@ -0,0 +1,374 @@
|
||||
//! Merging a peer's `.ecj` deletion ids into a local EC journal. Mirrors Go's
|
||||
//! `Store.MergeEcJournal` (`weed/storage/store_ec_journal.go`).
|
||||
|
||||
use std::collections::HashSet;
|
||||
use std::io;
|
||||
use std::path::Path;
|
||||
use std::sync::RwLock;
|
||||
|
||||
use crate::storage::erasure_coding::ecj_merge::{append_ecj_ids, read_ecj_ids};
|
||||
use crate::storage::store::Store;
|
||||
use crate::storage::types::{NeedleId, VolumeId};
|
||||
|
||||
/// How often an unmounted merge re-reads a journal that changed under it. Only
|
||||
/// a mount-delete-unmount or a concurrent merge between the read and the
|
||||
/// append changes it, so one retry is nearly always enough.
|
||||
const ECJ_MERGE_ATTEMPTS: usize = 5;
|
||||
|
||||
/// Fold a peer's deletion `ids` into the local journal of EC volume `vid` on
|
||||
/// the receiving disk, the one whose data directory is `data_dir`; `ecj_path`
|
||||
/// is that journal's path in the disk's index directory. Appends only the ids
|
||||
/// the journal lacks and returns how many it added. Blocking: call it from
|
||||
/// `spawn_blocking`.
|
||||
///
|
||||
/// A mounted volume owns its journal: the merge goes through its open handle
|
||||
/// and in-memory set. That is the receiving disk's own runtime for `vid`,
|
||||
/// wherever its journal lives (it may sit in the data dir rather than
|
||||
/// `ecj_path`'s index dir), else a sibling runtime journaling into `ecj_path`
|
||||
/// itself: disks sharing one index directory, or reconciliation mounting `vid`
|
||||
/// on a disk that journals into another's (#9212). Otherwise `ecj_path` is
|
||||
/// appended to under the store write lock, which mounts take, so no mount can
|
||||
/// open it mid-append. The read that computes the delta runs outside the lock,
|
||||
/// and a journal that changed in between — including by a concurrent merge —
|
||||
/// is re-read.
|
||||
pub fn merge_ec_journal(
|
||||
store: &RwLock<Store>,
|
||||
vid: VolumeId,
|
||||
data_dir: &str,
|
||||
ecj_path: &str,
|
||||
ids: &HashSet<NeedleId>,
|
||||
) -> io::Result<usize> {
|
||||
merge_ec_journal_with(store, vid, data_dir, ecj_path, ids, read_ecj_ids)
|
||||
}
|
||||
|
||||
/// `merge_ec_journal` with the unlocked journal read injected, so a test can
|
||||
/// mount the volume between that read and the append.
|
||||
fn merge_ec_journal_with(
|
||||
store: &RwLock<Store>,
|
||||
vid: VolumeId,
|
||||
data_dir: &str,
|
||||
ecj_path: &str,
|
||||
ids: &HashSet<NeedleId>,
|
||||
mut read: impl FnMut(&str) -> io::Result<(HashSet<NeedleId>, u64)>,
|
||||
) -> io::Result<usize> {
|
||||
for _ in 0..ECJ_MERGE_ATTEMPTS {
|
||||
{
|
||||
let mut store = store
|
||||
.write()
|
||||
.map_err(|_| io::Error::other("store lock poisoned"))?;
|
||||
if let Some(primary) = mounted_ec_journal(&store, vid, data_dir, ecj_path)? {
|
||||
let ecv = store.locations[primary]
|
||||
.find_ec_volume_mut(vid)
|
||||
.expect("mounted journal runtime");
|
||||
let journal_path = ecv.ecj_file_name();
|
||||
let added = ecv.merge_journal(ids)?;
|
||||
// Publish to the holders of the file the merge wrote to — the
|
||||
// picked runtime's journal may live outside ecj_path, and a
|
||||
// holder of a different file must not claim ids it lacks.
|
||||
publish_to_journal_siblings(&store, primary, vid, &journal_path, ids);
|
||||
return Ok(added);
|
||||
}
|
||||
}
|
||||
let (local, size) = read(ecj_path)?;
|
||||
let mut store = store
|
||||
.write()
|
||||
.map_err(|_| io::Error::other("store lock poisoned"))?;
|
||||
if let Some(primary) = mounted_ec_journal(&store, vid, data_dir, ecj_path)? {
|
||||
// Mounted since the read: its handle owns the journal now.
|
||||
let ecv = store.locations[primary]
|
||||
.find_ec_volume_mut(vid)
|
||||
.expect("mounted journal runtime");
|
||||
let journal_path = ecv.ecj_file_name();
|
||||
let added = ecv.merge_journal(ids)?;
|
||||
publish_to_journal_siblings(&store, primary, vid, &journal_path, ids);
|
||||
return Ok(added);
|
||||
}
|
||||
// The path append registers as a writer so a mount compacting this
|
||||
// journal cannot swap its inode underneath it (the write itself is
|
||||
// already serialized with mounts by the store lock).
|
||||
let _ecj_write =
|
||||
crate::storage::erasure_coding::ecj_registry::begin_ecj_write(ecj_path);
|
||||
if let Some(added) = append_ecj_ids(ecj_path, &local, ids, size)? {
|
||||
return Ok(added);
|
||||
}
|
||||
}
|
||||
Err(io::Error::other(format!(
|
||||
"ec volume {}: journal {} kept changing during merge",
|
||||
vid.0, ecj_path
|
||||
)))
|
||||
}
|
||||
|
||||
/// The disk index of the runtime holding `ecj_path` open, if any: the disk at
|
||||
/// `data_dir`'s own, else the first sibling journaling into it.
|
||||
fn mounted_ec_journal(
|
||||
store: &Store,
|
||||
vid: VolumeId,
|
||||
data_dir: &str,
|
||||
ecj_path: &str,
|
||||
) -> io::Result<Option<usize>> {
|
||||
let owner = store
|
||||
.locations
|
||||
.iter()
|
||||
.position(|loc| Path::new(&loc.directory) == Path::new(data_dir))
|
||||
.ok_or_else(|| {
|
||||
io::Error::other(format!(
|
||||
"ec volume {}: no disk at {} owns journal {}",
|
||||
vid.0, data_dir, ecj_path
|
||||
))
|
||||
})?;
|
||||
let runtime = if store.locations[owner].has_ec_volume(vid) {
|
||||
Some(owner)
|
||||
} else {
|
||||
store.locations.iter().position(|loc| {
|
||||
loc.find_ec_volume(vid)
|
||||
.is_some_and(|ecv| Path::new(&ecv.ecj_file_name()) == Path::new(ecj_path))
|
||||
})
|
||||
};
|
||||
Ok(runtime)
|
||||
}
|
||||
|
||||
/// Publishes merged ids into every other runtime journaling into `ecj_path`.
|
||||
fn publish_to_journal_siblings(
|
||||
store: &Store,
|
||||
primary: usize,
|
||||
vid: VolumeId,
|
||||
ecj_path: &str,
|
||||
ids: &HashSet<NeedleId>,
|
||||
) {
|
||||
for (i, loc) in store.locations.iter().enumerate() {
|
||||
if i == primary {
|
||||
continue;
|
||||
}
|
||||
if let Some(ecv) = loc.find_ec_volume(vid) {
|
||||
if Path::new(&ecv.ecj_file_name()) == Path::new(ecj_path) {
|
||||
ecv.publish_merged_ids(ids);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
use crate::config::MinFreeSpace;
|
||||
use crate::storage::needle_map::NeedleMapKind;
|
||||
use crate::storage::types::DiskType;
|
||||
use crate::storage::volume::{VifEcShardConfig, VifVolumeInfo};
|
||||
use tempfile::TempDir;
|
||||
|
||||
const COLLECTION: &str = "c";
|
||||
const VID: VolumeId = VolumeId(9);
|
||||
|
||||
/// A store with one disk per entry of `data`, all sharing `idx` when given,
|
||||
/// else each indexing into its own data dir.
|
||||
fn make_store(tmp: &TempDir, data: &[&str], idx: Option<&str>) -> RwLock<Store> {
|
||||
let mut store = Store::new(NeedleMapKind::InMemory);
|
||||
for d in data {
|
||||
let dir = tmp.path().join(d).to_string_lossy().into_owned();
|
||||
let idx_dir = idx
|
||||
.map(|i| tmp.path().join(i).to_string_lossy().into_owned())
|
||||
.unwrap_or_else(|| dir.clone());
|
||||
std::fs::create_dir_all(&dir).unwrap();
|
||||
std::fs::create_dir_all(&idx_dir).unwrap();
|
||||
store
|
||||
.add_location(
|
||||
&dir,
|
||||
&idx_dir,
|
||||
100,
|
||||
DiskType::HardDrive,
|
||||
MinFreeSpace::Percent(0.0),
|
||||
Vec::new(),
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
RwLock::new(store)
|
||||
}
|
||||
|
||||
fn dir(tmp: &TempDir, d: &str) -> String {
|
||||
tmp.path().join(d).to_string_lossy().into_owned()
|
||||
}
|
||||
|
||||
fn records(ids: &[u64]) -> Vec<u8> {
|
||||
let ids: Vec<NeedleId> = ids.iter().copied().map(NeedleId).collect();
|
||||
crate::storage::erasure_coding::ecj_merge::encode_ecj_ids(&ids)
|
||||
}
|
||||
|
||||
fn id_set(ids: &[u64]) -> HashSet<NeedleId> {
|
||||
ids.iter().copied().map(NeedleId).collect()
|
||||
}
|
||||
|
||||
/// Shard 0 of `VID` and its `.vif` in `data_dir`.
|
||||
fn write_shard0(data_dir: &str) {
|
||||
let base = format!("{}/{}_{}", data_dir, COLLECTION, VID.0);
|
||||
std::fs::write(format!("{}.ec00", base), b"shard data nonempty").unwrap();
|
||||
let vif = VifVolumeInfo {
|
||||
version: 3,
|
||||
ec_shard_config: Some(VifEcShardConfig {
|
||||
data_shards: 10,
|
||||
parity_shards: 4,
|
||||
..Default::default()
|
||||
}),
|
||||
..Default::default()
|
||||
};
|
||||
std::fs::write(
|
||||
format!("{}.vif", base),
|
||||
serde_json::to_string(&vif).unwrap(),
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
/// `VID`'s `.ecx` and a `.ecj` holding `deleted` in `dir`; returns the
|
||||
/// journal path.
|
||||
fn write_index(dir: &str, deleted: &[u64]) -> String {
|
||||
let base = format!("{}/{}_{}", dir, COLLECTION, VID.0);
|
||||
std::fs::write(format!("{}.ecx", base), vec![0u8; 16]).unwrap();
|
||||
let ecj = format!("{}.ecj", base);
|
||||
std::fs::write(&ecj, records(deleted)).unwrap();
|
||||
ecj
|
||||
}
|
||||
|
||||
fn deleted_on(store: &RwLock<Store>, disk: usize, id: u64) -> bool {
|
||||
store.read().unwrap().locations[disk]
|
||||
.find_ec_volume(VID)
|
||||
.expect("mounted")
|
||||
.is_needle_deleted(NeedleId(id))
|
||||
}
|
||||
|
||||
/// Disks sharing one index directory all hold the same journal path.
|
||||
/// Copying shards onto a disk that has not mounted `vid` must still reach
|
||||
/// the sibling runtime holding that journal open.
|
||||
#[test]
|
||||
fn shared_index_dir_reaches_sibling_mount() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let store = make_store(&tmp, &["d0", "d1"], Some("idx"));
|
||||
write_shard0(&dir(&tmp, "d0"));
|
||||
let ecj = write_index(&dir(&tmp, "idx"), &[1]);
|
||||
store.write().unwrap().locations[0]
|
||||
.mount_ec_shards(VID, COLLECTION, &[0], "")
|
||||
.unwrap();
|
||||
assert_eq!(
|
||||
store.read().unwrap().locations[0]
|
||||
.find_ec_volume(VID)
|
||||
.unwrap()
|
||||
.ecj_file_name(),
|
||||
ecj
|
||||
);
|
||||
|
||||
let added =
|
||||
merge_ec_journal(&store, VID, &dir(&tmp, "d1"), &ecj, &id_set(&[1, 2])).unwrap();
|
||||
assert_eq!(added, 1);
|
||||
assert!(
|
||||
deleted_on(&store, 0, 2),
|
||||
"the mounted sibling must see id 2"
|
||||
);
|
||||
assert_eq!(std::fs::read(&ecj).unwrap(), records(&[1, 2]));
|
||||
}
|
||||
|
||||
/// Every runtime holding the journal open must see merged ids in memory.
|
||||
#[test]
|
||||
fn shared_journal_reaches_every_holder() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let store = make_store(&tmp, &["d0", "d1"], Some("idx"));
|
||||
write_shard0(&dir(&tmp, "d0"));
|
||||
write_shard0(&dir(&tmp, "d1"));
|
||||
let ecj = write_index(&dir(&tmp, "idx"), &[1]);
|
||||
for i in 0..2 {
|
||||
store.write().unwrap().locations[i]
|
||||
.mount_ec_shards(VID, COLLECTION, &[0], "")
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
let added =
|
||||
merge_ec_journal(&store, VID, &dir(&tmp, "d1"), &ecj, &id_set(&[1, 2])).unwrap();
|
||||
assert_eq!(added, 1);
|
||||
assert!(
|
||||
deleted_on(&store, 0, 2) && deleted_on(&store, 1, 2),
|
||||
"every journal holder must see the merged id"
|
||||
);
|
||||
assert_eq!(std::fs::read(&ecj).unwrap(), records(&[1, 2]));
|
||||
}
|
||||
|
||||
/// The picked runtime may journal to a different file than the copied
|
||||
/// one — its index lives in its data directory while a sibling's lives
|
||||
/// in the index directory. The ids must be published only to holders of
|
||||
/// the file they were written to.
|
||||
#[test]
|
||||
fn publishes_to_actual_journal_holders() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let store = make_store(&tmp, &["d0", "d1"], Some("idx"));
|
||||
write_shard0(&dir(&tmp, "d0"));
|
||||
write_shard0(&dir(&tmp, "d1"));
|
||||
let data_ecj = write_index(&dir(&tmp, "d0"), &[1]);
|
||||
let idx_ecj = write_index(&dir(&tmp, "idx"), &[1]);
|
||||
for i in 0..2 {
|
||||
store.write().unwrap().locations[i]
|
||||
.mount_ec_shards(VID, COLLECTION, &[0], "")
|
||||
.unwrap();
|
||||
}
|
||||
assert_eq!(
|
||||
store.read().unwrap().locations[0]
|
||||
.find_ec_volume(VID)
|
||||
.unwrap()
|
||||
.ecj_file_name(),
|
||||
data_ecj
|
||||
);
|
||||
|
||||
let added =
|
||||
merge_ec_journal(&store, VID, &dir(&tmp, "d0"), &idx_ecj, &id_set(&[1, 2])).unwrap();
|
||||
assert_eq!(added, 1);
|
||||
assert_eq!(std::fs::read(&data_ecj).unwrap(), records(&[1, 2]));
|
||||
assert_eq!(std::fs::read(&idx_ecj).unwrap(), records(&[1]));
|
||||
assert!(deleted_on(&store, 0, 2));
|
||||
assert!(
|
||||
!deleted_on(&store, 1, 2),
|
||||
"a different journal's holder must not claim the merged id"
|
||||
);
|
||||
}
|
||||
|
||||
/// A sibling disk can mount `vid` from the receiving disk's index (#9212)
|
||||
/// while the merge reads the journal unlocked. The merge must go through
|
||||
/// that mount and report what it added.
|
||||
#[test]
|
||||
fn mount_during_read_is_merged_through_and_counted() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let store = make_store(&tmp, &["d0", "d1"], None);
|
||||
let owner = dir(&tmp, "d0");
|
||||
let ecj = write_index(&owner, &[1]);
|
||||
write_shard0(&dir(&tmp, "d1"));
|
||||
|
||||
let mut mounted = false;
|
||||
let read = |path: &str| {
|
||||
let read = read_ecj_ids(path);
|
||||
if !mounted {
|
||||
store.write().unwrap().locations[1]
|
||||
.mount_ec_shards_with_idx_dir(VID, COLLECTION, &[0], &owner, "")
|
||||
.unwrap();
|
||||
mounted = true;
|
||||
}
|
||||
read
|
||||
};
|
||||
let added =
|
||||
merge_ec_journal_with(&store, VID, &owner, &ecj, &id_set(&[1, 2, 3]), read).unwrap();
|
||||
assert_eq!(added, 2, "the ids merged through the new mount are counted");
|
||||
assert!(deleted_on(&store, 1, 2) && deleted_on(&store, 1, 3));
|
||||
assert_eq!(std::fs::read(&ecj).unwrap(), records(&[1, 2, 3]));
|
||||
}
|
||||
|
||||
/// An unmounted journal is merged on disk and a repeat adds nothing; a
|
||||
/// data dir that is no disk is refused.
|
||||
#[test]
|
||||
fn unmounted_journal_is_idempotent() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let store = make_store(&tmp, &["d0"], Some("idx"));
|
||||
let ecj = write_index(&dir(&tmp, "idx"), &[1, 2]);
|
||||
let d0 = dir(&tmp, "d0");
|
||||
for want in [2, 0, 0] {
|
||||
let added = merge_ec_journal(&store, VID, &d0, &ecj, &id_set(&[2, 3, 4])).unwrap();
|
||||
assert_eq!(added, want);
|
||||
}
|
||||
assert_eq!(std::fs::read(&ecj).unwrap(), records(&[1, 2, 3, 4]));
|
||||
assert!(
|
||||
merge_ec_journal(&store, VID, &dir(&tmp, "elsewhere"), &ecj, &id_set(&[1])).is_err()
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -536,13 +536,6 @@ pub(crate) enum NeedleStreamSource {
|
||||
}
|
||||
|
||||
impl NeedleStreamSource {
|
||||
pub(crate) fn clone_for_read(&self) -> io::Result<Self> {
|
||||
match self {
|
||||
NeedleStreamSource::Local(file) => Ok(NeedleStreamSource::Local(file.try_clone()?)),
|
||||
NeedleStreamSource::Remote(remote) => Ok(NeedleStreamSource::Remote(remote.clone())),
|
||||
}
|
||||
}
|
||||
|
||||
pub(crate) fn read_exact_at(&self, buf: &mut [u8], offset: u64) -> io::Result<()> {
|
||||
match self {
|
||||
NeedleStreamSource::Local(file) => read_exact_at(file, buf, offset),
|
||||
@@ -719,6 +712,29 @@ impl DatScanPlan {
|
||||
}
|
||||
}
|
||||
|
||||
/// Why a volume refuses all I/O, shared with its in-flight needle streams so
|
||||
/// they see a mark made after they left the store lock. A leaf lock.
|
||||
#[derive(Debug, Default)]
|
||||
pub(crate) struct IoUnavailable(Mutex<Option<String>>);
|
||||
|
||||
impl IoUnavailable {
|
||||
fn set(&self, reason: String) {
|
||||
*self.0.lock().unwrap() = Some(reason);
|
||||
}
|
||||
|
||||
fn is_set(&self) -> bool {
|
||||
self.0.lock().unwrap().is_some()
|
||||
}
|
||||
|
||||
pub(crate) fn error(&self) -> Option<VolumeError> {
|
||||
self.0
|
||||
.lock()
|
||||
.unwrap()
|
||||
.as_ref()
|
||||
.map(|reason| VolumeError::Unavailable(reason.clone()))
|
||||
}
|
||||
}
|
||||
|
||||
/// A needle read resolved under a store guard and run after it is released;
|
||||
/// the handle pins the inode, as for `DatScanPlan`. It takes no data-file
|
||||
/// lease: writers wait for one while holding the store write lock.
|
||||
@@ -727,11 +743,10 @@ pub(crate) struct NeedleReadPlan {
|
||||
offset: i64,
|
||||
size: Size,
|
||||
version: Version,
|
||||
volume_id: VolumeId,
|
||||
needle_id: NeedleId,
|
||||
compaction_revision: u16,
|
||||
data_file_access_control: Arc<DataFileAccessControl>,
|
||||
io_errors: Arc<IoErrorTracker>,
|
||||
io_unavailable: Arc<IoUnavailable>,
|
||||
}
|
||||
|
||||
impl NeedleReadPlan {
|
||||
@@ -794,9 +809,8 @@ impl NeedleReadPlan {
|
||||
data_file_offset,
|
||||
data_size: n.data_size,
|
||||
data_file_access_control: self.data_file_access_control,
|
||||
volume_id: self.volume_id,
|
||||
io_unavailable: self.io_unavailable,
|
||||
needle_id: self.needle_id,
|
||||
compaction_revision: self.compaction_revision,
|
||||
checksum: n.checksum.0,
|
||||
}
|
||||
}
|
||||
@@ -1070,14 +1084,9 @@ pub struct NeedleStreamInfo {
|
||||
pub data_size: u32,
|
||||
/// Per-volume file access lock used to match Go's slow-read behavior.
|
||||
pub data_file_access_control: Arc<DataFileAccessControl>,
|
||||
/// Volume ID — used to re-lookup needle offset if compaction occurs during streaming.
|
||||
pub volume_id: VolumeId,
|
||||
/// Needle ID — used to re-lookup needle offset if compaction occurs during streaming.
|
||||
/// Checked before each chunk: the volume can become unavailable mid-stream.
|
||||
pub(crate) io_unavailable: Arc<IoUnavailable>,
|
||||
pub needle_id: NeedleId,
|
||||
/// Compaction revision at the time of the initial read. If this changes during
|
||||
/// streaming, the needle's disk offset must be re-read from the needle map because
|
||||
/// compaction may have moved the needle to a different location.
|
||||
pub compaction_revision: u16,
|
||||
/// Checksum stored in the needle tail, verified once the last chunk has
|
||||
/// been read — before that frame is emitted.
|
||||
pub checksum: u32,
|
||||
@@ -1165,7 +1174,7 @@ pub struct Volume {
|
||||
/// Set when a failed recovery leaves the .dat/index pair unverified: all
|
||||
/// I/O is refused and a `.unavailable` marker keeps the volume quarantined
|
||||
/// across restarts. Mirrors Go's ioUnavailable.
|
||||
io_unavailable: Option<String>,
|
||||
io_unavailable: Arc<IoUnavailable>,
|
||||
|
||||
/// Shared flag from the parent DiskLocation indicating low disk space.
|
||||
/// Matches Go's `v.location.isDiskSpaceLow` checked in `IsReadOnly()`.
|
||||
@@ -1268,7 +1277,7 @@ impl Volume {
|
||||
},
|
||||
no_write_or_delete: false,
|
||||
no_write_can_delete: false,
|
||||
io_unavailable: None,
|
||||
io_unavailable: Arc::default(),
|
||||
location_disk_space_low: Arc::new(AtomicBool::new(false)),
|
||||
last_modified_ts_seconds: 0,
|
||||
last_append_at_ns: 0,
|
||||
@@ -1314,7 +1323,7 @@ impl Volume {
|
||||
super_block: SuperBlock::default(),
|
||||
no_write_or_delete: false,
|
||||
no_write_can_delete: false,
|
||||
io_unavailable: None,
|
||||
io_unavailable: Arc::default(),
|
||||
location_disk_space_low: Arc::new(AtomicBool::new(false)),
|
||||
last_modified_ts_seconds: 0,
|
||||
last_append_at_ns: 0,
|
||||
@@ -1721,8 +1730,10 @@ impl Volume {
|
||||
let mut idx_reader = io::BufReader::new(&idx_file);
|
||||
let mut nm = CompactNeedleMap::load_from_idx(&mut idx_reader, self.version())?;
|
||||
|
||||
// Re-open for append-only writes
|
||||
let write_file = OpenOptions::new().append(true).open(idx_path)?;
|
||||
// Re-open for positioned writes: rows are written at
|
||||
// idx_file_offset so a torn tail can be trimmed (an append-mode
|
||||
// handle cannot set_len on Windows).
|
||||
let write_file = OpenOptions::new().write(true).open(idx_path)?;
|
||||
nm.set_idx_file(Box::new(write_file), idx_size);
|
||||
self.nm = Some(NeedleMap::InMemory(nm));
|
||||
}
|
||||
@@ -1779,8 +1790,10 @@ impl Volume {
|
||||
cache_bytes,
|
||||
)?;
|
||||
|
||||
// Re-open for append-only writes
|
||||
let write_file = OpenOptions::new().append(true).open(idx_path)?;
|
||||
// Re-open for positioned writes: rows are written at
|
||||
// idx_file_offset so a torn tail can be trimmed (an append-mode
|
||||
// handle cannot set_len on Windows).
|
||||
let write_file = OpenOptions::new().write(true).open(idx_path)?;
|
||||
nm.set_idx_file(Box::new(write_file), idx_size);
|
||||
self.nm = Some(NeedleMap::Redb(nm));
|
||||
}
|
||||
@@ -1851,7 +1864,7 @@ impl Volume {
|
||||
self.dat_file.is_some() || self.remote_dat_file.is_some()
|
||||
}
|
||||
|
||||
fn current_dat_file_size(&self) -> io::Result<u64> {
|
||||
pub(crate) fn current_dat_file_size(&self) -> io::Result<u64> {
|
||||
if let Some(ref f) = self.dat_file {
|
||||
Ok(f.metadata()?.len())
|
||||
} else if let Some(ref remote_dat_file) = self.remote_dat_file {
|
||||
@@ -2210,11 +2223,10 @@ impl Volume {
|
||||
offset: nv.offset.to_actual_offset(),
|
||||
size,
|
||||
version: self.version(),
|
||||
volume_id: self.id,
|
||||
needle_id: id,
|
||||
compaction_revision: self.super_block.compaction_revision,
|
||||
data_file_access_control: self.data_file_access_control.clone(),
|
||||
io_errors: self.io_errors.clone(),
|
||||
io_unavailable: self.io_unavailable.clone(),
|
||||
})
|
||||
}
|
||||
|
||||
@@ -2237,39 +2249,6 @@ impl Volume {
|
||||
}
|
||||
}
|
||||
|
||||
/// Re-lookup a needle's data-file offset after compaction may have moved it.
|
||||
///
|
||||
/// Returns `(new_data_file_offset, current_compaction_revision)` or an error
|
||||
/// if the needle is no longer present / has been deleted.
|
||||
///
|
||||
/// This matches Go's `readNeedleDataInto` behaviour: when the volume's
|
||||
/// `CompactionRevision` changes between streaming chunks, the needle offset
|
||||
/// is re-read from the needle map because compaction may have relocated it.
|
||||
pub fn re_lookup_needle_data_offset(
|
||||
&self,
|
||||
needle_id: NeedleId,
|
||||
) -> Result<(u64, u16), VolumeError> {
|
||||
let nm = self.nm_or_not_found()?;
|
||||
let nv = nm.get(needle_id)?.ok_or(VolumeError::NotFound)?;
|
||||
if nv.offset.is_zero() {
|
||||
return Err(VolumeError::NotFound);
|
||||
}
|
||||
if nv.size.is_deleted() {
|
||||
return Err(VolumeError::Deleted);
|
||||
}
|
||||
|
||||
let offset = nv.offset.to_actual_offset();
|
||||
let version = self.version();
|
||||
|
||||
let data_file_offset = if version == VERSION_1 {
|
||||
offset as u64 + NEEDLE_HEADER_SIZE as u64
|
||||
} else {
|
||||
offset as u64 + NEEDLE_HEADER_SIZE as u64 + 4 // skip DataSize (4 bytes)
|
||||
};
|
||||
|
||||
Ok((data_file_offset, self.super_block.compaction_revision))
|
||||
}
|
||||
|
||||
// ---- Write ----
|
||||
|
||||
/// Write a needle to the volume (synchronous path).
|
||||
@@ -2295,8 +2274,10 @@ impl Volume {
|
||||
/// sync and one .idx sync per run of distinct needle ids that holds a
|
||||
/// durable write, instead of two per durable needle. Nothing in such a
|
||||
/// run is published before its sync, so a failed sync takes the whole
|
||||
/// run back off the .dat and fails every entry. A repeated id starts a
|
||||
/// new run, so its dedup and cookie checks see the earlier write.
|
||||
/// run back off the .dat and fails every entry. A durable entry that
|
||||
/// fails to index stops the volume taking writes, and the entries after
|
||||
/// it in the run are refused read only. A repeated id starts a new run,
|
||||
/// so its dedup and cookie checks see the earlier write.
|
||||
pub fn write_needles_grouped(
|
||||
&mut self,
|
||||
writes: &mut [(Needle, bool)],
|
||||
@@ -2359,11 +2340,21 @@ impl Volume {
|
||||
}
|
||||
self.last_append_at_ns = last_append_at_ns;
|
||||
|
||||
// A durable entry that fails to publish stops the volume taking
|
||||
// writes, so the entries after it are refused the way a lone
|
||||
// write would be.
|
||||
let mut refused = false;
|
||||
for ((n, fsync), r) in run.iter().zip(staged.iter_mut()) {
|
||||
if let Ok(Some(offset)) = *r
|
||||
if refused {
|
||||
*r = Err(self
|
||||
.check_writable()
|
||||
.err()
|
||||
.unwrap_or(VolumeError::ReadOnly(self.id)));
|
||||
} else if let Ok(Some(offset)) = *r
|
||||
&& let Err(e) = self.publish_write(n, offset, *fsync)
|
||||
{
|
||||
*r = Err(e);
|
||||
refused = *fsync;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2884,14 +2875,14 @@ impl Volume {
|
||||
pub fn is_read_only(&self) -> bool {
|
||||
self.no_write_or_delete
|
||||
|| self.no_write_can_delete
|
||||
|| self.io_unavailable.is_some()
|
||||
|| self.io_unavailable.is_set()
|
||||
|| self.location_disk_space_low.load(Ordering::Relaxed)
|
||||
}
|
||||
|
||||
/// Mirrors Go's ReadOnlyReasons: `no_write_or_delete` already covers the
|
||||
/// io_unavailable quarantine.
|
||||
pub fn read_only_reasons(&self) -> (bool, bool, bool, bool) {
|
||||
let no_write_or_delete = self.no_write_or_delete || self.io_unavailable.is_some();
|
||||
let no_write_or_delete = self.no_write_or_delete || self.io_unavailable.is_set();
|
||||
let disk_space_low = self.location_disk_space_low.load(Ordering::Relaxed);
|
||||
(
|
||||
no_write_or_delete || self.no_write_can_delete || disk_space_low,
|
||||
@@ -2904,9 +2895,7 @@ impl Volume {
|
||||
/// The reason the volume refuses all I/O, when a failed recovery left the
|
||||
/// .dat/index pair unverified. Mirrors Go's unavailableError.
|
||||
pub fn unavailable_error(&self) -> Option<VolumeError> {
|
||||
self.io_unavailable
|
||||
.as_ref()
|
||||
.map(|reason| VolumeError::Unavailable(reason.clone()))
|
||||
self.io_unavailable.error()
|
||||
}
|
||||
|
||||
/// Fail closed after a recovery could not return the volume to a verified
|
||||
@@ -2914,7 +2903,7 @@ impl Volume {
|
||||
/// reload stays unavailable until an operator verifies the volume.
|
||||
fn mark_io_unavailable(&mut self, reason: String) {
|
||||
self.no_write_or_delete = true;
|
||||
self.io_unavailable = Some(reason.clone());
|
||||
self.io_unavailable.set(reason.clone());
|
||||
self.mark_io_quarantined();
|
||||
if let Err(e) = self.persist_unavailable(&reason) {
|
||||
warn!(
|
||||
@@ -2956,7 +2945,7 @@ impl Volume {
|
||||
return;
|
||||
};
|
||||
self.no_write_or_delete = true;
|
||||
self.io_unavailable = Some(reason.trim().to_string());
|
||||
self.io_unavailable.set(reason.trim().to_string());
|
||||
self.mark_io_quarantined();
|
||||
warn!(
|
||||
volume_id = self.id.0,
|
||||
@@ -3662,9 +3651,12 @@ impl Volume {
|
||||
.unwrap_or(false);
|
||||
if needs_idx_writer {
|
||||
let idx_path = self.file_name(".idx");
|
||||
// Positioned writes: an append-mode handle cannot set_len on
|
||||
// Windows.
|
||||
let write_file = OpenOptions::new()
|
||||
.append(true)
|
||||
.write(true)
|
||||
.create(true)
|
||||
.truncate(false)
|
||||
.open(&idx_path)?;
|
||||
let idx_size = trim_torn_idx_tail(&write_file, &idx_path)?;
|
||||
if let Some(ref mut nm) = self.nm {
|
||||
@@ -3977,6 +3969,11 @@ impl Volume {
|
||||
&self.dir
|
||||
}
|
||||
|
||||
/// Get the directory this volume's index is stored in.
|
||||
pub fn dir_idx(&self) -> &str {
|
||||
&self.dir_idx
|
||||
}
|
||||
|
||||
/// Throttle IO during compaction to avoid saturating disk.
|
||||
pub fn maybe_throttle_compaction(&self, bytes_written: u64) {
|
||||
if self.compaction_byte_per_second <= 0 || !self.is_compacting() {
|
||||
@@ -4408,6 +4405,13 @@ impl Volume {
|
||||
return Err(e);
|
||||
}
|
||||
let idx_size = nm.index_file_size();
|
||||
// The copy would stream the .dat from remote storage for a commit that refuses it.
|
||||
if self.has_remote_file() {
|
||||
return Err(VolumeError::Io(io::Error::other(format!(
|
||||
"volume {} is tiered to remote storage, cannot compact",
|
||||
self.id
|
||||
))));
|
||||
}
|
||||
|
||||
// Fresh opens, not `try_clone`: see `dat_scan_plan`.
|
||||
let src_dat = if self.dat_file.is_some() {
|
||||
@@ -6507,6 +6511,215 @@ mod tests {
|
||||
assert!(matches!(results[0], Err(VolumeError::Unavailable(_))));
|
||||
}
|
||||
|
||||
/// An .idx writer that tears its `tear_at`-th row: half of the row
|
||||
/// reaches the file, then the write fails.
|
||||
struct TornIdxWriter {
|
||||
file: File,
|
||||
writes: usize,
|
||||
tear_at: usize,
|
||||
fail_truncates: bool,
|
||||
}
|
||||
|
||||
impl Write for TornIdxWriter {
|
||||
fn write(&mut self, buf: &[u8]) -> io::Result<usize> {
|
||||
self.writes += 1;
|
||||
if self.writes == self.tear_at {
|
||||
self.file.write_all(&buf[..buf.len() / 2])?;
|
||||
return Err(io::Error::other("injected torn .idx write"));
|
||||
}
|
||||
self.file.write(buf)
|
||||
}
|
||||
|
||||
fn flush(&mut self) -> io::Result<()> {
|
||||
self.file.flush()
|
||||
}
|
||||
}
|
||||
|
||||
impl Seek for TornIdxWriter {
|
||||
fn seek(&mut self, pos: SeekFrom) -> io::Result<u64> {
|
||||
self.file.seek(pos)
|
||||
}
|
||||
}
|
||||
|
||||
impl crate::storage::needle_map::IdxFileWriter for TornIdxWriter {
|
||||
fn sync_all(&self) -> io::Result<()> {
|
||||
self.file.sync_all()
|
||||
}
|
||||
|
||||
fn truncate_to(&mut self, len: u64) -> io::Result<()> {
|
||||
if self.fail_truncates {
|
||||
return Err(io::Error::other("injected trim failure"));
|
||||
}
|
||||
self.file.set_len(len)
|
||||
}
|
||||
}
|
||||
|
||||
/// A durable entry whose index update fails stops the volume taking
|
||||
/// writes, as it does when sent on its own, so the entries after it in
|
||||
/// the run are refused instead of indexed behind a row that may be
|
||||
/// torn. Entries before it stay acked.
|
||||
#[test]
|
||||
fn test_grouped_failed_durable_index_refuses_rest_of_run() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
let mut v = make_test_volume(dir);
|
||||
let mut kept = batch_needle(4, 0xdd, b"kept");
|
||||
v.write_needle(&mut kept, true, true).unwrap();
|
||||
let mut other = batch_needle(5, 0xee, b"other");
|
||||
v.write_needle(&mut other, true, true).unwrap();
|
||||
|
||||
let nm = v.nm.as_mut().unwrap();
|
||||
let idx_len = nm.index_file_size();
|
||||
let file = OpenOptions::new()
|
||||
.write(true)
|
||||
.open(format!("{dir}/1.idx"))
|
||||
.unwrap();
|
||||
nm.set_idx_file(
|
||||
Box::new(TornIdxWriter {
|
||||
file,
|
||||
writes: 0,
|
||||
tear_at: 2,
|
||||
fail_truncates: false,
|
||||
}),
|
||||
idx_len,
|
||||
);
|
||||
|
||||
let mut writes = vec![
|
||||
(batch_needle(1, 0xaa, b"before"), true),
|
||||
(batch_needle(2, 0xbb, b"torn"), true),
|
||||
(batch_needle(3, 0xcc, b"after"), true),
|
||||
(batch_needle(4, 0xdd, b"kept"), true),
|
||||
(batch_needle(5, 0xef, b"wrong cookie"), true),
|
||||
];
|
||||
let results = v.write_needles_grouped(&mut writes);
|
||||
// Two setup writes, then the run's one sync of each file.
|
||||
assert_eq!(v.sync_counts_for_test(), (3, 3));
|
||||
assert!(v.is_read_only());
|
||||
drop(v);
|
||||
|
||||
let reopened = match Volume::new(
|
||||
dir,
|
||||
dir,
|
||||
VolumeId(1),
|
||||
NeedleMapKind::InMemory,
|
||||
&VolumeSpec::default(),
|
||||
) {
|
||||
Ok(v) => v,
|
||||
Err(e) => panic!("the volume does not reload: {e}"),
|
||||
};
|
||||
for ((n, _), r) in writes.iter().zip(&results) {
|
||||
if r.is_ok() {
|
||||
let mut got = Needle {
|
||||
id: n.id,
|
||||
..Needle::default()
|
||||
};
|
||||
let read = reopened.read_needle(&mut got);
|
||||
assert!(
|
||||
read.is_ok() && got.data == n.data,
|
||||
"acked write {} lost on reload: {read:?}",
|
||||
n.id.0
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
assert!(matches!(results[0], Ok((_, _, false))), "{results:?}");
|
||||
assert!(matches!(results[1], Err(VolumeError::Io(_))), "{results:?}");
|
||||
for r in &results[2..] {
|
||||
assert!(
|
||||
matches!(r, Err(VolumeError::ReadOnly(VolumeId(1)))),
|
||||
"refused as it would be sent on its own: {results:?}"
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
/// A torn .idx row is trimmed back, so the row a later write appends
|
||||
/// still lands aligned and survives a reload.
|
||||
#[test]
|
||||
fn test_failed_index_write_is_trimmed() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
let mut v = make_test_volume(dir);
|
||||
|
||||
let nm = v.nm.as_mut().unwrap();
|
||||
let idx_len = nm.index_file_size();
|
||||
let file = OpenOptions::new()
|
||||
.write(true)
|
||||
.open(format!("{dir}/1.idx"))
|
||||
.unwrap();
|
||||
nm.set_idx_file(
|
||||
Box::new(TornIdxWriter {
|
||||
file,
|
||||
writes: 0,
|
||||
tear_at: 1,
|
||||
fail_truncates: false,
|
||||
}),
|
||||
idx_len,
|
||||
);
|
||||
|
||||
let mut writes = vec![
|
||||
(batch_needle(1, 0xaa, b"torn"), false),
|
||||
(batch_needle(2, 0xbb, b"after"), true),
|
||||
];
|
||||
let results = v.write_needles_grouped(&mut writes);
|
||||
assert!(matches!(results[0], Err(VolumeError::Io(_))), "{results:?}");
|
||||
assert!(matches!(results[1], Ok((_, _, false))), "{results:?}");
|
||||
drop(v);
|
||||
|
||||
let reopened = match Volume::new(
|
||||
dir,
|
||||
dir,
|
||||
VolumeId(1),
|
||||
NeedleMapKind::InMemory,
|
||||
&VolumeSpec::default(),
|
||||
) {
|
||||
Ok(v) => v,
|
||||
Err(e) => panic!("the volume does not reload: {e}"),
|
||||
};
|
||||
let mut got = Needle {
|
||||
id: NeedleId(2),
|
||||
..Needle::default()
|
||||
};
|
||||
reopened.read_needle(&mut got).unwrap();
|
||||
assert_eq!(got.data, b"after");
|
||||
}
|
||||
|
||||
/// When the trim of a torn .idx row itself fails, no later row is
|
||||
/// appended after the torn bytes.
|
||||
#[test]
|
||||
fn test_untrimmed_torn_row_refuses_later_appends() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
let mut v = make_test_volume(dir);
|
||||
|
||||
let nm = v.nm.as_mut().unwrap();
|
||||
let idx_len = nm.index_file_size();
|
||||
let file = OpenOptions::new()
|
||||
.write(true)
|
||||
.open(format!("{dir}/1.idx"))
|
||||
.unwrap();
|
||||
nm.set_idx_file(
|
||||
Box::new(TornIdxWriter {
|
||||
file,
|
||||
writes: 0,
|
||||
tear_at: 1,
|
||||
fail_truncates: true,
|
||||
}),
|
||||
idx_len,
|
||||
);
|
||||
|
||||
let mut writes = vec![
|
||||
(batch_needle(1, 0xaa, b"torn"), false),
|
||||
(batch_needle(2, 0xbb, b"after"), false),
|
||||
];
|
||||
let results = v.write_needles_grouped(&mut writes);
|
||||
assert!(matches!(results[0], Err(VolumeError::Io(_))), "{results:?}");
|
||||
assert!(results[1].is_err(), "{results:?}");
|
||||
|
||||
// The second row was never appended behind the torn bytes.
|
||||
let size = std::fs::metadata(format!("{dir}/1.idx")).unwrap().len();
|
||||
assert_eq!(size, idx_len + 8);
|
||||
}
|
||||
|
||||
/// The I/O error streak after writing four durable needles, the ones in
|
||||
/// `failing` with a media error on their append, first one at a time and
|
||||
/// then as one grouped run. Also returns which grouped writes landed.
|
||||
@@ -8200,83 +8413,7 @@ mod tests {
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_compaction_revision_relookup() {
|
||||
// Verifies that re_lookup_needle_data_offset returns the correct data offset
|
||||
// and compaction revision, and that after compaction the offset changes.
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
let mut v = make_test_volume(dir);
|
||||
|
||||
// Write two needles
|
||||
let mut n1 = Needle {
|
||||
id: NeedleId(1),
|
||||
cookie: Cookie(0xAABBCCDD),
|
||||
data: b"first-needle-data".to_vec(),
|
||||
data_size: 17,
|
||||
..Needle::default()
|
||||
};
|
||||
v.write_needle(&mut n1, true, false).unwrap();
|
||||
|
||||
let mut n2 = Needle {
|
||||
id: NeedleId(2),
|
||||
cookie: Cookie(0x11223344),
|
||||
data: b"second-needle-data".to_vec(),
|
||||
data_size: 18,
|
||||
..Needle::default()
|
||||
};
|
||||
v.write_needle(&mut n2, true, false).unwrap();
|
||||
|
||||
// Get initial revision and offset for needle 1
|
||||
let initial_rev = v.super_block.compaction_revision;
|
||||
let (initial_offset, rev) = v.re_lookup_needle_data_offset(NeedleId(1)).unwrap();
|
||||
assert_eq!(rev, initial_rev);
|
||||
assert!(initial_offset > 0, "data offset should be positive");
|
||||
|
||||
// Delete needle 2 so compaction removes it
|
||||
let mut del_n2 = Needle {
|
||||
id: NeedleId(2),
|
||||
cookie: Cookie(0x11223344),
|
||||
..Needle::default()
|
||||
};
|
||||
v.delete_needle(&mut del_n2).unwrap();
|
||||
|
||||
// Compact the volume — this increments compaction_revision and may move needles
|
||||
v.compact_by_index(0, 0, |_| true).unwrap();
|
||||
v.commit_compact().unwrap();
|
||||
|
||||
// After compaction, the revision should have changed
|
||||
let new_rev = v.super_block.compaction_revision;
|
||||
assert_eq!(
|
||||
new_rev,
|
||||
initial_rev + 1,
|
||||
"compaction should increment revision"
|
||||
);
|
||||
|
||||
// Re-lookup needle 1 — should still be found with the new revision
|
||||
let (new_offset, relookup_rev) = v.re_lookup_needle_data_offset(NeedleId(1)).unwrap();
|
||||
assert_eq!(relookup_rev, new_rev);
|
||||
assert!(new_offset > 0, "data offset should still be positive");
|
||||
|
||||
// The data should still be readable correctly after compaction
|
||||
let mut read_n1 = Needle {
|
||||
id: NeedleId(1),
|
||||
..Needle::default()
|
||||
};
|
||||
v.read_needle(&mut read_n1).unwrap();
|
||||
assert_eq!(read_n1.data, b"first-needle-data");
|
||||
|
||||
// Deleted needle should not be found
|
||||
let result = v.re_lookup_needle_data_offset(NeedleId(2));
|
||||
assert!(
|
||||
result.is_err(),
|
||||
"deleted needle should not be found after compaction"
|
||||
);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_stream_info_includes_compaction_revision() {
|
||||
// Verifies that NeedleStreamInfo carries the volume's compaction revision
|
||||
// so that StreamingBody can detect when compaction has occurred.
|
||||
fn test_stream_info_locates_needle_data() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
let mut v = make_test_volume(dir);
|
||||
@@ -8302,9 +8439,7 @@ mod tests {
|
||||
plan.read_meta(&mut read_n).unwrap();
|
||||
let info = plan.into_stream_info(&read_n);
|
||||
|
||||
assert_eq!(info.volume_id, VolumeId(1));
|
||||
assert_eq!(info.needle_id, NeedleId(42));
|
||||
assert_eq!(info.compaction_revision, v.super_block.compaction_revision);
|
||||
assert_eq!(info.data_size, data.len() as u32);
|
||||
assert!(info.data_file_offset > 0);
|
||||
}
|
||||
@@ -8650,9 +8785,17 @@ mod tests {
|
||||
fn test_compaction_aborts_on_truncated_index() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
let mut v = reload_as_tiered(dir, "vif_compact_test", 4);
|
||||
{
|
||||
let mut v = make_test_volume(dir);
|
||||
for i in 1..=4u64 {
|
||||
write_test_needle(&mut v, i, format!("needle-{i}").as_bytes());
|
||||
}
|
||||
v.set_read_only_persist(false, true).unwrap();
|
||||
v.sync_to_disk().unwrap();
|
||||
}
|
||||
let mut v = make_test_volume(dir);
|
||||
let Some(NeedleMap::SortedFile(_)) = v.nm else {
|
||||
panic!("tiered volume should search the on-disk .sdx");
|
||||
panic!("read-only volume should search the on-disk .sdx");
|
||||
};
|
||||
|
||||
let idx = OpenOptions::new()
|
||||
@@ -8669,11 +8812,6 @@ mod tests {
|
||||
matches!(err, VolumeError::Io(ref e) if e.kind() == io::ErrorKind::UnexpectedEof),
|
||||
"unexpected error: {err:?}"
|
||||
);
|
||||
|
||||
crate::remote_storage::s3_tier::global_s3_tier_registry()
|
||||
.write()
|
||||
.unwrap()
|
||||
.remove("s3.vif_compact_test");
|
||||
}
|
||||
|
||||
// Building .sdx writes to the index directory, which a read-only volume's
|
||||
@@ -9549,6 +9687,31 @@ mod tests {
|
||||
}
|
||||
}
|
||||
|
||||
// A directory occupying the .dat path fails open even as root; load must
|
||||
// return the error. Mirrors Go's TestLoad_DatOpenFail_NoNilPanic.
|
||||
#[test]
|
||||
fn test_load_dat_open_fail_returns_error() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
|
||||
{
|
||||
let _v = make_test_volume(dir);
|
||||
}
|
||||
|
||||
let dat_path = format!("{dir}/1.dat");
|
||||
std::fs::remove_file(&dat_path).unwrap();
|
||||
std::fs::create_dir(&dat_path).unwrap();
|
||||
|
||||
let result = Volume::new(
|
||||
dir,
|
||||
dir,
|
||||
VolumeId(1),
|
||||
NeedleMapKind::InMemory,
|
||||
&VolumeSpec::default(),
|
||||
);
|
||||
assert!(result.is_err(), "unopenable .dat must fail load");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_remote_only_volume_load_reads_from_tier_backend() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
|
||||
@@ -108,6 +108,9 @@ fn build_test_state(
|
||||
file_size_limit_bytes: 0,
|
||||
maintenance_byte_per_second: 0,
|
||||
is_heartbeating: std::sync::atomic::AtomicBool::new(true),
|
||||
ec_decodes_in_flight: std::sync::Mutex::new(std::collections::HashSet::new()),
|
||||
ec_decode_tail: std::sync::Mutex::new(std::collections::HashSet::new()),
|
||||
ec_decode_tail_notify: tokio::sync::Notify::new(),
|
||||
has_master: false,
|
||||
pre_stop_seconds: 0,
|
||||
volume_state_notify: tokio::sync::Notify::new(),
|
||||
|
||||
Generated
+3
-3
@@ -5188,9 +5188,9 @@ dependencies = [
|
||||
|
||||
[[package]]
|
||||
name = "rustls"
|
||||
version = "0.23.43"
|
||||
version = "0.23.45"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "0283386ce02abc0151e1761d08802dfe86c173b0b494af5cbc086574e453da06"
|
||||
checksum = "0d41d731c7d2f962d1ccc364cec258de3c0e93b38c2fb3ba97ac74513048d634"
|
||||
dependencies = [
|
||||
"aws-lc-rs",
|
||||
"log",
|
||||
@@ -5781,7 +5781,7 @@ source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "32497e9a4c7b38532efcdebeef879707aa9f794296a4f0244f6f69e9bc8574bd"
|
||||
dependencies = [
|
||||
"fastrand",
|
||||
"getrandom 0.4.3",
|
||||
"getrandom 0.3.4",
|
||||
"once_cell",
|
||||
"rustix",
|
||||
"windows-sys 0.61.2",
|
||||
|
||||
@@ -16,7 +16,7 @@
|
||||
<project.build.sourceEncoding>UTF-8</project.build.sourceEncoding>
|
||||
<maven.compiler.source>11</maven.compiler.source>
|
||||
<maven.compiler.target>11</maven.compiler.target>
|
||||
<spark.version>3.5.7</spark.version>
|
||||
<spark.version>3.5.8</spark.version>
|
||||
<hadoop.version>3.3.6</hadoop.version>
|
||||
<scala.binary.version>2.12</scala.binary.version>
|
||||
<junit.version>4.13.2</junit.version>
|
||||
|
||||
@@ -0,0 +1,101 @@
|
||||
# S3/IAM error response compatibility test harness
|
||||
|
||||
.PHONY: all test help build-weed check-deps start-server stop-server test-errors test-with-server clean logs
|
||||
|
||||
WEED_BINARY := ../../../weed/weed_binary
|
||||
S3_PORT := 8334
|
||||
TEST_TIMEOUT := 10m
|
||||
SERVER_DIR := ./test-volume-data
|
||||
S3_CONFIG := ./test_s3.json
|
||||
|
||||
.DEFAULT_GOAL := help
|
||||
|
||||
all: test
|
||||
|
||||
test: test-with-server
|
||||
|
||||
help:
|
||||
@echo "S3/IAM error response compatibility test harness"
|
||||
@echo ""
|
||||
@echo "Available targets:"
|
||||
@echo " build-weed - Build the SeaweedFS binary"
|
||||
@echo " start-server - Start SeaweedFS mini with S3 and embedded IAM enabled"
|
||||
@echo " stop-server - Stop the test server"
|
||||
@echo " test-errors - Run the error compatibility checks"
|
||||
@echo " test-with-server - Start server, run checks, stop server"
|
||||
@echo " logs - Show server logs"
|
||||
@echo " clean - Remove test artifacts"
|
||||
|
||||
build-weed:
|
||||
@echo "Building SeaweedFS binary..."
|
||||
@cd ../../../weed && go build -o weed_binary .
|
||||
@chmod +x $(WEED_BINARY)
|
||||
|
||||
check-deps: build-weed
|
||||
@command -v go >/dev/null 2>&1 || (echo "Go is required but not installed" && exit 1)
|
||||
@command -v python3 >/dev/null 2>&1 || (echo "python3 is required but not installed" && exit 1)
|
||||
@python3 -c "import botocore" 2>/dev/null || (echo "botocore is required: pip install botocore" && exit 1)
|
||||
@test -f $(WEED_BINARY) || (echo "SeaweedFS binary not found at $(WEED_BINARY)" && exit 1)
|
||||
@test -f $(S3_CONFIG) || (echo "S3 config not found at $(S3_CONFIG)" && exit 1)
|
||||
|
||||
start-server: check-deps
|
||||
@echo "Starting SeaweedFS mini for error compatibility tests..."
|
||||
@rm -f weed-server.pid
|
||||
@mkdir -p $(SERVER_DIR)
|
||||
@if curl -sS -o /dev/null http://127.0.0.1:$(S3_PORT)/ >/dev/null 2>&1; then \
|
||||
echo "port $(S3_PORT) is already served by another process; refusing to test against it"; \
|
||||
exit 1; \
|
||||
fi
|
||||
@$(WEED_BINARY) mini \
|
||||
-dir=$(SERVER_DIR) \
|
||||
-s3 \
|
||||
-s3.port=$(S3_PORT) \
|
||||
-s3.config=$(S3_CONFIG) \
|
||||
-s3.iam.readOnly=false \
|
||||
-master.peers=none \
|
||||
> weed-test.log 2>&1 & echo $$! > weed-server.pid
|
||||
@for i in $$(seq 1 60); do \
|
||||
kill -0 $$(cat weed-server.pid) 2>/dev/null || { \
|
||||
echo "SeaweedFS server exited during startup"; \
|
||||
echo "=== Server logs ==="; \
|
||||
test -f weed-test.log && cat weed-test.log || true; \
|
||||
exit 1; \
|
||||
}; \
|
||||
if curl -sS -o /dev/null http://127.0.0.1:$(S3_PORT)/ >/dev/null 2>&1; then \
|
||||
echo "SeaweedFS S3 server is ready on port $(S3_PORT)"; \
|
||||
sleep 3; \
|
||||
exit 0; \
|
||||
fi; \
|
||||
sleep 1; \
|
||||
done; \
|
||||
echo "SeaweedFS S3 server failed to start"; \
|
||||
echo "=== Server logs ==="; \
|
||||
test -f weed-test.log && cat weed-test.log || true; \
|
||||
$(MAKE) --no-print-directory stop-server; \
|
||||
exit 1
|
||||
|
||||
stop-server:
|
||||
@if [ -f weed-server.pid ]; then \
|
||||
echo "Stopping SeaweedFS server..."; \
|
||||
kill $$(cat weed-server.pid) 2>/dev/null || true; \
|
||||
sleep 2; \
|
||||
rm -f weed-server.pid; \
|
||||
fi
|
||||
|
||||
test-errors:
|
||||
@echo "Running S3/IAM error compatibility checks..."
|
||||
python3 s3_error_compat_test.py \
|
||||
--endpoint http://127.0.0.1:$(S3_PORT) \
|
||||
--access-key admin \
|
||||
--secret-key admin
|
||||
|
||||
test-with-server: start-server
|
||||
@$(MAKE) test-errors; TEST_RESULT=$$?; \
|
||||
$(MAKE) stop-server; exit $$TEST_RESULT
|
||||
|
||||
logs:
|
||||
@test -f weed-test.log && tail -100 weed-test.log || echo "No log file found"
|
||||
|
||||
clean: stop-server
|
||||
@rm -rf $(SERVER_DIR) weed-test.log weed-server.pid
|
||||
@echo "Cleaned up test artifacts"
|
||||
@@ -0,0 +1,216 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Verify S3/IAM error responses match AWS for common client mistakes.
|
||||
|
||||
Needs only botocore (pip install botocore). Sends raw SigV4 requests so the exact
|
||||
status and response body are visible, compares each with what AWS returns, then
|
||||
cleans up everything it created (one bucket, one IAM user).
|
||||
|
||||
Usage:
|
||||
python3 s3_error_compat_test.py --endpoint http://127.0.0.1:8333 \
|
||||
--access-key <admin key> --secret-key <admin secret>
|
||||
|
||||
The key needs Admin rights, and IAM writes must be enabled (-s3.iam.readOnly=false).
|
||||
"""
|
||||
import argparse, atexit, base64, datetime, hashlib, http.client, re, sys, time, urllib.parse, uuid
|
||||
|
||||
from botocore.auth import S3SigV4Auth, SigV4Auth
|
||||
from botocore.awsrequest import AWSRequest
|
||||
from botocore.credentials import Credentials
|
||||
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument('--endpoint', default='http://127.0.0.1:8333')
|
||||
ap.add_argument('--access-key', required=True)
|
||||
ap.add_argument('--secret-key', required=True)
|
||||
ap.add_argument('--region', default='us-east-1')
|
||||
args = ap.parse_args()
|
||||
|
||||
EP = args.endpoint.rstrip('/')
|
||||
U = urllib.parse.urlsplit(EP)
|
||||
ADMIN = Credentials(args.access_key, args.secret_key)
|
||||
BAD_KEY = Credentials('AKIAREPRONOSUCHKEY00', 'x' * 40)
|
||||
BAD_SECRET = Credentials(args.access_key, 'y' * 40)
|
||||
B = 'errorcompat-' + uuid.uuid4().hex[:8]
|
||||
S3DOC = 'https://docs.aws.amazon.com/AmazonS3/latest/API/'
|
||||
IAMDOC = 'https://docs.aws.amazon.com/IAM/latest/APIReference/'
|
||||
rows = []
|
||||
|
||||
|
||||
def send(method, path, body=b'', headers=None, creds=ADMIN, service='s3', mutate=None, sign=True,
|
||||
keep_sha256_header=False):
|
||||
headers = dict(headers or {})
|
||||
url = EP + path
|
||||
if sign:
|
||||
req = AWSRequest(method=method, url=url, data=body, headers=headers)
|
||||
if service == 's3':
|
||||
if 'X-Amz-Content-SHA256' not in req.headers:
|
||||
req.headers['X-Amz-Content-SHA256'] = hashlib.sha256(body).hexdigest()
|
||||
# S3SigV4Auth always recomputes X-Amz-Content-SHA256 from the body; the
|
||||
# generic SigV4Auth signs a caller-supplied value instead, which is how
|
||||
# a mismatched declared hash can be produced on the wire.
|
||||
signer = SigV4Auth if keep_sha256_header else S3SigV4Auth
|
||||
signer(creds, 's3', args.region).add_auth(req)
|
||||
else:
|
||||
SigV4Auth(creds, service, args.region).add_auth(req)
|
||||
headers = dict(req.headers.items())
|
||||
if mutate:
|
||||
headers = mutate(headers)
|
||||
conn = http.client.HTTPConnection(U.hostname, U.port, timeout=30)
|
||||
conn.request(method, path, body=body, headers=headers)
|
||||
r = conn.getresponse()
|
||||
data = r.read().decode('utf-8', 'replace')
|
||||
conn.close()
|
||||
return r.status, data
|
||||
|
||||
|
||||
def shape(data):
|
||||
"""Root element and <Code> of an XML error body."""
|
||||
body = re.sub(r'<\?xml[^>]*\?>', '', data).strip()
|
||||
root = re.match(r'<(\w+)', body)
|
||||
code = re.search(r'<Code>([^<]*)</Code>', data)
|
||||
return (root.group(1) if root else ('-' if not body else 'non-xml')), (code.group(1) if code else '')
|
||||
|
||||
|
||||
def iam(params, creds=ADMIN):
|
||||
p = dict(params, Version='2010-05-08')
|
||||
return dict(method='POST', path='/', body=urllib.parse.urlencode(p).encode(), creds=creds, service='iam',
|
||||
headers={'Content-Type': 'application/x-www-form-urlencoded; charset=utf-8'})
|
||||
|
||||
|
||||
def case(group, name, expect, ref, req, check=None):
|
||||
"""expect = (status, root element or None, code or None); check() adds a side-effect note."""
|
||||
st, data = send(**req)
|
||||
root, code = shape(data)
|
||||
note = check() if check else ''
|
||||
ok = st == expect[0] and expect[1] in (None, root) and expect[2] in (None, code) and not note
|
||||
rows.append((group, name, '%s %s %s' % (st, root, code), '%s %s %s' % (expect[0], expect[1] or '', expect[2] or ''), note, ok, ref))
|
||||
|
||||
|
||||
def exists(key):
|
||||
return lambda: 'object %r now exists' % key if send('HEAD', '/%s/%s' % (B, key))[0] == 200 else ''
|
||||
|
||||
|
||||
def S3(method, path, **kw):
|
||||
return dict(method=method, path=path, **kw)
|
||||
|
||||
|
||||
# ---------------- cleanup (atexit so failed runs still clean up) ----------------
|
||||
user = USER = upload_id = None
|
||||
|
||||
|
||||
def cleanup():
|
||||
try:
|
||||
if user and USER:
|
||||
send(**iam({'Action': 'DeleteAccessKey', 'UserName': user, 'AccessKeyId': USER.access_key}))
|
||||
if user:
|
||||
send(**iam({'Action': 'DeleteUser', 'UserName': user}))
|
||||
if upload_id:
|
||||
send('DELETE', '/%s/mp?uploadId=%s' % (B, upload_id))
|
||||
st, d = send('GET', '/%s?list-type=2' % B)
|
||||
for k in re.findall(r'<Key>([^<]+)</Key>', d):
|
||||
send('DELETE', '/%s/%s' % (B, urllib.parse.quote(k)))
|
||||
send('DELETE', '/' + B)
|
||||
except Exception as e:
|
||||
print('cleanup failed: %s' % e)
|
||||
|
||||
|
||||
atexit.register(cleanup)
|
||||
|
||||
# ---------------- setup ----------------
|
||||
st, d = send('PUT', '/' + B)
|
||||
if st != 200:
|
||||
sys.exit('cannot create bucket %s: %s %s' % (B, st, d[:200]))
|
||||
send('PUT', '/%s/obj' % B, body=b'hello')
|
||||
st, d = send('POST', '/%s/mp?uploads' % B)
|
||||
upload_id = re.search(r'<UploadId>([^<]+)', d).group(1)
|
||||
user = 'errorcompat-' + uuid.uuid4().hex[:6]
|
||||
send(**iam({'Action': 'CreateUser', 'UserName': user}))
|
||||
st, d = send(**iam({'Action': 'CreateAccessKey', 'UserName': user}))
|
||||
USER = Credentials(re.search(r'<AccessKeyId>([^<]+)', d).group(1), re.search(r'<SecretAccessKey>([^<]+)', d).group(1))
|
||||
for _ in range(15): # wait until the new key is accepted
|
||||
if 'InvalidAccessKeyId' not in send('GET', '/', creds=USER)[1]:
|
||||
break
|
||||
time.sleep(1)
|
||||
|
||||
# ---------------- 1. requests that run a different operation ----------------
|
||||
for sub in ('logging', 'notification', 'accelerate', 'website', 'replication',
|
||||
'analytics&id=a', 'inventory&id=a', 'metrics&id=a', 'intelligent-tiering&id=a'):
|
||||
case(1, 'PUT ?%s' % sub, (501, 'Error', 'NotImplemented'), S3DOC + 'API_Error.html#:~:text=Code%3A%20NotImplemented',
|
||||
S3('PUT', '/%s?%s' % (B, sub), body=b'<X/>'))
|
||||
case(1, 'DELETE ?logging', (501, 'Error', 'NotImplemented'), S3DOC + 'API_Error.html#:~:text=Code%3A%20NotImplemented',
|
||||
S3('DELETE', '/%s?logging' % B),
|
||||
lambda: '' if send('HEAD', '/' + B)[0] == 200 else 'BUCKET DELETED')
|
||||
case(1, 'CopyObject, x-amz-copy-source without "/"', (400, 'Error', 'InvalidArgument'),
|
||||
S3DOC + 'API_CopyObject.html#AmazonS3-CopyObject-request-header-CopySource',
|
||||
S3('PUT', '/%s/copy-dst' % B, headers={'x-amz-copy-source': 'nobucketonly'}), exists('copy-dst'))
|
||||
case(1, 'UploadPart partNumber=abc', (400, 'Error', 'InvalidArgument'),
|
||||
S3DOC + 'API_UploadPart.html#AmazonS3-UploadPart-request-uri-querystring-PartNumber',
|
||||
S3('PUT', '/%s/mp?partNumber=abc&uploadId=%s' % (B, upload_id), body=b'part-body'), exists('mp'))
|
||||
case(1, 'PutObject, x-amz-content-sha256 != body', (400, 'Error', 'XAmzContentSHA256Mismatch'),
|
||||
'undocumented code; ' + S3DOC + 'API_UploadPart.html#API_UploadPart_RequestSyntax',
|
||||
S3('PUT', '/%s/sha-mismatch' % B, body=b'abc',
|
||||
headers={'X-Amz-Content-SHA256': hashlib.sha256(b'zzz').hexdigest()}, keep_sha256_header=True),
|
||||
exists('sha-mismatch'))
|
||||
case(1, 'PutObject, x-amz-content-sha256 not hex', (400, 'Error', None), 'undocumented',
|
||||
S3('PUT', '/%s/sha-nothex' % B, body=b'abc',
|
||||
headers={'X-Amz-Content-SHA256': 'nothex'}, keep_sha256_header=True), exists('sha-nothex'))
|
||||
|
||||
# ---------------- 2. IAM failures in the S3 <Error> envelope ----------------
|
||||
case(2, 'IAM ListUsers, unknown access key', (403, 'ErrorResponse', 'InvalidClientTokenId'),
|
||||
'checked against iam.amazonaws.com (docs list UnrecognizedClientException); ' + IAMDOC + 'CommonErrors.html#CommonErrors-UnrecognizedClientException', iam({'Action': 'ListUsers'}, creds=BAD_KEY))
|
||||
case(2, 'IAM ListUsers, wrong secret', (403, 'ErrorResponse', None), 'undocumented for IAM',
|
||||
iam({'Action': 'ListUsers'}, creds=BAD_SECRET))
|
||||
case(2, 'IAM ListUsers, non-admin user', (403, 'ErrorResponse', 'AccessDenied'),
|
||||
IAMDOC + 'CommonErrors.html#CommonErrors-AccessDeniedException', iam({'Action': 'ListUsers'}, creds=USER))
|
||||
case(2, 'IAM CreateUser, non-admin user', (403, 'ErrorResponse', 'AccessDenied'),
|
||||
IAMDOC + 'CommonErrors.html#CommonErrors-AccessDeniedException', iam({'Action': 'CreateUser', 'UserName': user + 'x'}, creds=USER))
|
||||
|
||||
# ---------------- 3. wrong status or code ----------------
|
||||
tags = [('Action', 'TagUser'), ('UserName', user)] + [('Tags.member.%d.%s' % (i, k), 'k%d' % i if k == 'Key' else 'v')
|
||||
for i in range(1, 52) for k in ('Key', 'Value')]
|
||||
case(3, 'IAM TagUser with 51 tags', (409, 'ErrorResponse', 'LimitExceeded'), IAMDOC + 'API_TagUser.html#API_TagUser_Errors',
|
||||
iam(dict(tags)))
|
||||
case(3, 'IAM unknown Action', (404, 'ErrorResponse', 'InvalidAction'), 'checked against iam.amazonaws.com',
|
||||
iam({'Action': 'NoSuchActionZZ'}))
|
||||
|
||||
|
||||
def no_date(h):
|
||||
return {k: v for k, v in h.items() if k.lower() != 'x-amz-date'}
|
||||
|
||||
|
||||
def drop(field):
|
||||
return lambda h: dict(h, Authorization=re.sub(r',? *%s=[^,]*' % field, '', h['Authorization']))
|
||||
|
||||
|
||||
case(3, 'S3 request without x-amz-date', (403, 'Error', 'AccessDenied'), 'checked against s3.amazonaws.com; ' + S3DOC + 'API_Error.html#:~:text=Code%3A%20AccessDenied',
|
||||
S3('GET', '/' + B, mutate=no_date))
|
||||
case(3, 'Authorization without Signature=', (400, 'Error', 'AuthorizationHeaderMalformed'),
|
||||
'checked against s3.amazonaws.com; ' + S3DOC + 'API_Error.html#:~:text=Code%3A%20AuthorizationHeaderMalformed', S3('GET', '/' + B, mutate=drop('Signature')))
|
||||
case(3, 'Authorization without Credential=', (400, 'Error', 'InvalidArgument'),
|
||||
'checked against s3.amazonaws.com ("Unsupported Authorization Type")', S3('GET', '/' + B, mutate=drop('Credential')))
|
||||
for m, exp, chk in (('GET', (400, 'Error', 'InvalidArgument'), 'checked against s3.amazonaws.com ("Invalid version id specified"); '),
|
||||
('HEAD', (400, None, None), 'checked against s3.amazonaws.com; '),
|
||||
('DELETE', (400, None, None), 'unverified (needs write access); ')):
|
||||
case(3, '%s ?versionId=bogus, unversioned bucket' % m, exp,
|
||||
chk + S3DOC + 'API_GetObject.html#AmazonS3-GetObject-request-uri-querystring-VersionId',
|
||||
S3(m, '/%s/obj?versionId=bogus' % B))
|
||||
for pn in ('0', '10001'):
|
||||
case(3, 'UploadPart partNumber=%s' % pn, (400, 'Error', 'InvalidArgument'),
|
||||
S3DOC + 'API_UploadPart.html#AmazonS3-UploadPart-request-uri-querystring-PartNumber',
|
||||
S3('PUT', '/%s/mp?partNumber=%s&uploadId=%s' % (B, pn, upload_id), body=b'x'))
|
||||
case(3, 'DeleteObjects with 1001 keys', (400, 'Error', 'MalformedXML'), S3DOC + 'API_DeleteObjects.html#API_DeleteObjects_RequestBody',
|
||||
S3('POST', '/%s?delete' % B, body=('<Delete>' + ''.join('<Object><Key>k%d</Key></Object>' % i for i in range(1001)) + '</Delete>').encode()))
|
||||
# last: this one deletes the user when it should be refused
|
||||
case(3, 'IAM DeleteUser while it has an access key', (409, 'ErrorResponse', 'DeleteConflict'),
|
||||
IAMDOC + 'API_DeleteUser.html#API_DeleteUser_Errors', iam({'Action': 'DeleteUser', 'UserName': user}))
|
||||
|
||||
# ---------------- report ----------------
|
||||
w = max(len(r[1]) for r in rows)
|
||||
print('%-4s %-*s %-46s %-40s %s' % ('', w, 'case', 'got (status root code)', 'expected (AWS)', 'side effect'))
|
||||
for g, name, got, exp, note, ok, ref in rows:
|
||||
print('%-4s %-*s %-46s %-40s %s' % ('ok' if ok else 'DIFF', w, name, got, exp, note))
|
||||
print('\nreferences:')
|
||||
for g, name, got, exp, note, ok, ref in rows:
|
||||
print(' %-*s %s' % (w, name, ref))
|
||||
diff = sum(not r[5] for r in rows)
|
||||
print('\n%d cases, %d differ from AWS' % (len(rows), diff))
|
||||
sys.exit(1 if diff else 0)
|
||||
@@ -0,0 +1,20 @@
|
||||
{
|
||||
"identities": [
|
||||
{
|
||||
"name": "admin",
|
||||
"credentials": [
|
||||
{
|
||||
"accessKey": "admin",
|
||||
"secretKey": "admin"
|
||||
}
|
||||
],
|
||||
"actions": [
|
||||
"Admin",
|
||||
"Read",
|
||||
"List",
|
||||
"Tagging",
|
||||
"Write"
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -895,7 +895,7 @@ func (cp *ConfigPersistence) ListTaskDetails() ([]string, error) {
|
||||
return taskIDs, nil
|
||||
}
|
||||
|
||||
// CleanupCompletedTasks removes old completed tasks beyond the retention limit
|
||||
// CleanupCompletedTasks removes old terminal task files beyond the retention limit
|
||||
func (cp *ConfigPersistence) CleanupCompletedTasks() error {
|
||||
cp.tasksMu.Lock()
|
||||
defer cp.tasksMu.Unlock()
|
||||
@@ -914,10 +914,10 @@ func (cp *ConfigPersistence) CleanupCompletedTasks() error {
|
||||
return fmt.Errorf("failed to load tasks for cleanup: %w", err)
|
||||
}
|
||||
|
||||
// Filter completed and failed tasks, sort by completion time
|
||||
var completedTasks []*maintenance.MaintenanceTask
|
||||
for _, task := range allTasks {
|
||||
if (task.Status == maintenance.TaskStatusCompleted || task.Status == maintenance.TaskStatusFailed) && task.CompletedAt != nil {
|
||||
switch task.Status {
|
||||
case maintenance.TaskStatusCompleted, maintenance.TaskStatusFailed, maintenance.TaskStatusCancelled:
|
||||
completedTasks = append(completedTasks, task)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,108 @@
|
||||
package dash
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/admin/maintenance"
|
||||
)
|
||||
|
||||
func countTaskStateFiles(t *testing.T, dir string) int {
|
||||
t.Helper()
|
||||
entries, err := os.ReadDir(filepath.Join(dir, TasksSubdir))
|
||||
if os.IsNotExist(err) {
|
||||
return 0
|
||||
}
|
||||
if err != nil {
|
||||
t.Fatalf("read tasks dir: %v", err)
|
||||
}
|
||||
count := 0
|
||||
for _, entry := range entries {
|
||||
if !entry.IsDir() && filepath.Ext(entry.Name()) == ".pb" {
|
||||
count++
|
||||
}
|
||||
}
|
||||
return count
|
||||
}
|
||||
|
||||
// TestCancelledTaskFilesDoNotAccumulate reproduces issue #11595: every scan
|
||||
// cycle cancels pending tasks of each detected type and re-detects them, so a
|
||||
// cancelled task file per candidate volume accumulates on disk forever.
|
||||
func TestCancelledTaskFilesDoNotAccumulate(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
cp := NewConfigPersistence(dir)
|
||||
|
||||
queue := maintenance.NewMaintenanceQueue(nil)
|
||||
queue.SetPersistence(cp)
|
||||
|
||||
for cycle := 0; cycle < 3; cycle++ {
|
||||
queue.AddTask(&maintenance.MaintenanceTask{
|
||||
ID: fmt.Sprintf("ec_vol_%d_cycle_%d", cycle, cycle),
|
||||
Type: "erasure_coding",
|
||||
VolumeID: uint32(cycle + 1),
|
||||
Server: "server1",
|
||||
})
|
||||
if cancelled := queue.CancelPendingTasksByType("erasure_coding"); cancelled != 1 {
|
||||
t.Fatalf("cycle %d: cancelled %d tasks, want 1", cycle, cancelled)
|
||||
}
|
||||
}
|
||||
|
||||
if n := countTaskStateFiles(t, dir); n != 0 {
|
||||
t.Errorf("%d task files on disk after %d cancel cycles, want 0", n, 3)
|
||||
}
|
||||
}
|
||||
|
||||
// TestManuallyCancelledTaskFileIsRemoved covers the CancelTask path used by
|
||||
// the UI: a cancelled pending task must not leave its file behind, where a
|
||||
// restart would resurrect it as pending.
|
||||
func TestManuallyCancelledTaskFileIsRemoved(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
cp := NewConfigPersistence(dir)
|
||||
|
||||
manager := maintenance.NewMaintenanceManager(nil, nil, cp)
|
||||
queue := manager.GetQueue()
|
||||
queue.SetPersistence(cp)
|
||||
|
||||
queue.AddTask(&maintenance.MaintenanceTask{
|
||||
ID: "manual_1",
|
||||
Type: "vacuum",
|
||||
VolumeID: 7,
|
||||
Server: "server1",
|
||||
})
|
||||
if n := countTaskStateFiles(t, dir); n != 1 {
|
||||
t.Fatalf("%d task files after AddTask, want 1", n)
|
||||
}
|
||||
|
||||
if err := manager.CancelTask("manual_1"); err != nil {
|
||||
t.Fatalf("CancelTask: %v", err)
|
||||
}
|
||||
if n := countTaskStateFiles(t, dir); n != 0 {
|
||||
t.Errorf("%d task files on disk after CancelTask, want 0", n)
|
||||
}
|
||||
}
|
||||
|
||||
// TestCleanupCompletedTasksBoundsCancelledFiles checks retention covers
|
||||
// cancelled files, e.g. ones written by older versions.
|
||||
func TestCleanupCompletedTasksBoundsCancelledFiles(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
cp := NewConfigPersistence(dir)
|
||||
|
||||
for i := 0; i < MaxCompletedTasks+5; i++ {
|
||||
if err := cp.SaveTaskState(&maintenance.MaintenanceTask{
|
||||
ID: fmt.Sprintf("old_cancelled_%02d", i),
|
||||
Type: "erasure_coding",
|
||||
Status: maintenance.TaskStatusCancelled,
|
||||
}); err != nil {
|
||||
t.Fatalf("save task state: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
if err := cp.CleanupCompletedTasks(); err != nil {
|
||||
t.Fatalf("CleanupCompletedTasks: %v", err)
|
||||
}
|
||||
if n := countTaskStateFiles(t, dir); n > MaxCompletedTasks {
|
||||
t.Errorf("%d task files after cleanup, want at most %d", n, MaxCompletedTasks)
|
||||
}
|
||||
}
|
||||
@@ -452,6 +452,7 @@ func (mm *MaintenanceManager) performCleanup() {
|
||||
|
||||
removedTasks := mm.queue.CleanupOldTasks(taskRetention)
|
||||
removedWorkers := mm.queue.RemoveStaleWorkers(workerTimeout)
|
||||
mm.queue.cleanupCompletedTasks()
|
||||
|
||||
// Clean up stale pending operations (operations running for more than 4 hours)
|
||||
staleOperationTimeout := 4 * time.Hour
|
||||
@@ -616,37 +617,48 @@ func (mm *MaintenanceManager) saveTaskConfigsFromPolicy(policy *worker_pb.Mainte
|
||||
// CancelTask cancels a pending task
|
||||
func (mm *MaintenanceManager) CancelTask(taskID string) error {
|
||||
mm.queue.mutex.Lock()
|
||||
defer mm.queue.mutex.Unlock()
|
||||
|
||||
task, exists := mm.queue.tasks[taskID]
|
||||
if !exists {
|
||||
mm.queue.mutex.Unlock()
|
||||
return fmt.Errorf("task %s not found", taskID)
|
||||
}
|
||||
|
||||
if task.Status == TaskStatusPending {
|
||||
task.Status = TaskStatusCancelled
|
||||
task.CompletedAt = &[]time.Time{time.Now()}[0]
|
||||
|
||||
// Remove from pending tasks
|
||||
for i, pendingTask := range mm.queue.pendingTasks {
|
||||
if pendingTask.ID == taskID {
|
||||
mm.queue.pendingTasks = append(mm.queue.pendingTasks[:i], mm.queue.pendingTasks[i+1:]...)
|
||||
break
|
||||
}
|
||||
}
|
||||
|
||||
// Notify ActiveTopology to release capacity
|
||||
if mm.scanner != nil && mm.scanner.integration != nil {
|
||||
if at := mm.scanner.integration.GetActiveTopology(); at != nil {
|
||||
_ = at.CompleteTask(taskID)
|
||||
}
|
||||
}
|
||||
|
||||
glog.V(2).Infof("Cancelled task %s", taskID)
|
||||
return nil
|
||||
if task.Status != TaskStatusPending {
|
||||
status := task.Status
|
||||
mm.queue.mutex.Unlock()
|
||||
return fmt.Errorf("task %s cannot be cancelled (status: %s)", taskID, status)
|
||||
}
|
||||
|
||||
return fmt.Errorf("task %s cannot be cancelled (status: %s)", taskID, task.Status)
|
||||
task.Status = TaskStatusCancelled
|
||||
completedTime := time.Now()
|
||||
task.CompletedAt = &completedTime
|
||||
cancelledSnapshot := snapshotTask(task)
|
||||
|
||||
// Remove from pending tasks
|
||||
for i, pendingTask := range mm.queue.pendingTasks {
|
||||
if pendingTask.ID == taskID {
|
||||
mm.queue.pendingTasks = append(mm.queue.pendingTasks[:i], mm.queue.pendingTasks[i+1:]...)
|
||||
break
|
||||
}
|
||||
}
|
||||
|
||||
// Notify ActiveTopology to release capacity
|
||||
if mm.scanner != nil && mm.scanner.integration != nil {
|
||||
if at := mm.scanner.integration.GetActiveTopology(); at != nil {
|
||||
_ = at.CompleteTask(taskID)
|
||||
}
|
||||
}
|
||||
mm.queue.mutex.Unlock()
|
||||
|
||||
if mm.queue.persistence != nil {
|
||||
mm.queue.persistMu.Lock()
|
||||
if mm.queue.deleteTaskStateLocked(taskID) != nil {
|
||||
mm.queue.saveTaskStateLocked(cancelledSnapshot)
|
||||
}
|
||||
mm.queue.persistMu.Unlock()
|
||||
}
|
||||
glog.V(2).Infof("Cancelled task %s", taskID)
|
||||
return nil
|
||||
}
|
||||
|
||||
// RegisterWorker registers a new worker
|
||||
|
||||
@@ -97,21 +97,53 @@ func (mq *MaintenanceQueue) LoadTasksFromPersistence() error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// saveTaskState saves a task to persistent storage
|
||||
// isTerminalStatus reports whether the status is a terminal task state.
|
||||
func isTerminalStatus(status MaintenanceTaskStatus) bool {
|
||||
return status == TaskStatusCompleted || status == TaskStatusFailed || status == TaskStatusCancelled
|
||||
}
|
||||
|
||||
// saveTaskState saves a task to persistent storage. persistMu orders the
|
||||
// status check and the write against the cancel paths' deletes, so a stale
|
||||
// non-terminal file cannot resurrect the task on the next restart.
|
||||
func (mq *MaintenanceQueue) saveTaskState(task *MaintenanceTask) {
|
||||
if mq.persistence != nil {
|
||||
if err := mq.persistence.SaveTaskState(task); err != nil {
|
||||
glog.Errorf("Failed to save task state for %s: %v", task.ID, err)
|
||||
}
|
||||
if mq.persistence == nil {
|
||||
return
|
||||
}
|
||||
mq.persistMu.Lock()
|
||||
defer mq.persistMu.Unlock()
|
||||
mq.saveTaskStateLocked(task)
|
||||
}
|
||||
|
||||
// saveTaskStateLocked must be called with persistMu held.
|
||||
func (mq *MaintenanceQueue) saveTaskStateLocked(task *MaintenanceTask) {
|
||||
mq.mutex.RLock()
|
||||
live, ok := mq.tasks[task.ID]
|
||||
stale := !isTerminalStatus(task.Status) && (!ok || isTerminalStatus(live.Status))
|
||||
mq.mutex.RUnlock()
|
||||
if stale {
|
||||
return
|
||||
}
|
||||
if err := mq.persistence.SaveTaskState(task); err != nil {
|
||||
glog.Errorf("Failed to save task state for %s: %v", task.ID, err)
|
||||
}
|
||||
}
|
||||
|
||||
func (mq *MaintenanceQueue) deleteTaskState(taskID string) {
|
||||
if mq.persistence != nil {
|
||||
if err := mq.persistence.DeleteTaskState(taskID); err != nil {
|
||||
glog.V(2).Infof("Failed to delete task state for %s: %v", taskID, err)
|
||||
}
|
||||
func (mq *MaintenanceQueue) deleteTaskState(taskID string) error {
|
||||
if mq.persistence == nil {
|
||||
return nil
|
||||
}
|
||||
mq.persistMu.Lock()
|
||||
defer mq.persistMu.Unlock()
|
||||
return mq.deleteTaskStateLocked(taskID)
|
||||
}
|
||||
|
||||
// deleteTaskStateLocked must be called with persistMu held.
|
||||
func (mq *MaintenanceQueue) deleteTaskStateLocked(taskID string) error {
|
||||
if err := mq.persistence.DeleteTaskState(taskID); err != nil {
|
||||
glog.Warningf("Failed to delete task state for %s: %v", taskID, err)
|
||||
return err
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// cleanupCompletedTasks removes old completed tasks beyond the retention limit
|
||||
@@ -297,9 +329,18 @@ func (mq *MaintenanceQueue) CancelPendingTasksByType(taskType MaintenanceTaskTyp
|
||||
mq.pendingTasks = remaining
|
||||
mq.mutex.Unlock()
|
||||
|
||||
// Persist cancelled state outside the lock to avoid blocking
|
||||
for _, snapshot := range cancelledSnapshots {
|
||||
mq.saveTaskState(snapshot)
|
||||
// Cancelled is terminal: drop the file like CompleteTask does instead of
|
||||
// leaving one orphaned .pb per cancelled task per scan cycle. If removal
|
||||
// fails, persist the cancelled state so a restart drops it instead of
|
||||
// re-queueing it.
|
||||
if mq.persistence != nil {
|
||||
mq.persistMu.Lock()
|
||||
for _, snapshot := range cancelledSnapshots {
|
||||
if mq.deleteTaskStateLocked(snapshot.ID) != nil {
|
||||
mq.saveTaskStateLocked(snapshot)
|
||||
}
|
||||
}
|
||||
mq.persistMu.Unlock()
|
||||
}
|
||||
return cancelled
|
||||
}
|
||||
@@ -922,7 +963,7 @@ func generateTaskID() string {
|
||||
return fmt.Sprintf("%s-%04d", string(b), timestamp)
|
||||
}
|
||||
|
||||
// CleanupOldTasks removes old completed and failed tasks
|
||||
// CleanupOldTasks removes old terminal tasks from memory
|
||||
func (mq *MaintenanceQueue) CleanupOldTasks(retention time.Duration) int {
|
||||
mq.mutex.Lock()
|
||||
defer mq.mutex.Unlock()
|
||||
@@ -931,7 +972,7 @@ func (mq *MaintenanceQueue) CleanupOldTasks(retention time.Duration) int {
|
||||
removed := 0
|
||||
|
||||
for id, task := range mq.tasks {
|
||||
if (task.Status == TaskStatusCompleted || task.Status == TaskStatusFailed) &&
|
||||
if (task.Status == TaskStatusCompleted || task.Status == TaskStatusFailed || task.Status == TaskStatusCancelled) &&
|
||||
task.CompletedAt != nil &&
|
||||
task.CompletedAt.Before(cutoff) {
|
||||
delete(mq.tasks, id)
|
||||
|
||||
@@ -218,6 +218,10 @@ type MaintenanceQueue struct {
|
||||
policy *MaintenancePolicy
|
||||
integration *MaintenanceIntegration
|
||||
persistence TaskPersistence // Interface for task persistence
|
||||
// persistMu serializes the check+write in saveTaskState against the
|
||||
// deletes in the cancel paths, so a stale pending save cannot land on
|
||||
// disk after the task's file was removed.
|
||||
persistMu sync.Mutex
|
||||
}
|
||||
|
||||
// MaintenanceScanner analyzes the cluster and generates maintenance tasks
|
||||
|
||||
@@ -152,6 +152,17 @@ func (r *LockRing) GetSnapshot() (servers []pb.ServerAddress) {
|
||||
return r.snapshots[0].servers
|
||||
}
|
||||
|
||||
// PriorOwnerWindowEnd is when the latest ring change stops routing moved keys
|
||||
// to their prior owner.
|
||||
func (r *LockRing) PriorOwnerWindowEnd() time.Time {
|
||||
r.RLock()
|
||||
defer r.RUnlock()
|
||||
if len(r.snapshots) == 0 {
|
||||
return time.Time{}
|
||||
}
|
||||
return r.snapshots[0].ts.Add(r.snapshotInterval)
|
||||
}
|
||||
|
||||
// WaitForCleanup waits for all pending cleanup operations to complete
|
||||
func (r *LockRing) WaitForCleanup() {
|
||||
r.cleanupWg.Wait()
|
||||
|
||||
@@ -32,6 +32,7 @@ var Commands = []*Command{
|
||||
cmdFix,
|
||||
cmdFuse,
|
||||
cmdIam,
|
||||
cmdImage,
|
||||
cmdMaster,
|
||||
cmdMasterFollower,
|
||||
cmdMini,
|
||||
|
||||
+20
-38
@@ -23,6 +23,7 @@ import (
|
||||
_ "github.com/seaweedfs/seaweedfs/weed/credential/postgres"
|
||||
"github.com/seaweedfs/seaweedfs/weed/filer"
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/iam/integration"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/iam_pb"
|
||||
@@ -475,9 +476,18 @@ func (fo *FilerOptions) startFiler() {
|
||||
if credentialManager != nil {
|
||||
adminSigningKey := security.SigningKey(util.GetViper().GetString("jwt.filer_signing.key"))
|
||||
iamGrpcServer := weed_server.NewIamGrpcServer(credentialManager, adminSigningKey)
|
||||
// The OIDC provider and role RPCs write where S3 servers configured with
|
||||
// filer-typed "oidcProviderStore" and "roleStore" read: this filer, at
|
||||
// the stores' default base paths.
|
||||
selfAddress := func() string { return string(filerAddress) }
|
||||
if roleStore, err := integration.NewFilerRoleStore(nil, selfAddress); err != nil {
|
||||
glog.Warningf("IAM gRPC: role RPCs disabled: %v", err)
|
||||
} else {
|
||||
iamGrpcServer.SetSTSStores(integration.NewFilerOIDCProviderStore(nil, selfAddress), roleStore)
|
||||
}
|
||||
iam_pb.RegisterSeaweedIdentityAccessManagementServer(grpcS, iamGrpcServer)
|
||||
if len(adminSigningKey) == 0 {
|
||||
glog.V(0).Info("Registered IAM gRPC service on filer (unauthenticated; set jwt.filer_signing.key in security.toml to require admin Bearer token)")
|
||||
glog.Warning("IAM gRPC service on filer is UNAUTHENTICATED: anyone who can reach this port can create users and policies, and its OIDC provider and role RPCs are refused; set jwt.filer_signing.key in security.toml to require an admin Bearer token")
|
||||
} else {
|
||||
glog.V(0).Info("Registered IAM gRPC service on filer (admin Bearer token required)")
|
||||
}
|
||||
@@ -495,21 +505,7 @@ func (fo *FilerOptions) startFiler() {
|
||||
if gracefulTimeout <= 0 {
|
||||
gracefulTimeout = 15 * time.Second
|
||||
}
|
||||
stopGrpcServer := func() {
|
||||
glog.V(0).Infof("Gracefully stopping gRPC server")
|
||||
stopped := make(chan struct{})
|
||||
go func() {
|
||||
grpcS.GracefulStop()
|
||||
close(stopped)
|
||||
}()
|
||||
select {
|
||||
case <-stopped:
|
||||
glog.V(0).Infof("gRPC server stopped gracefully")
|
||||
case <-time.After(gracefulTimeout):
|
||||
glog.V(0).Infof("gRPC server graceful stop timed out after %s, forcing stop", gracefulTimeout)
|
||||
grpcS.Stop()
|
||||
}
|
||||
}
|
||||
stopGrpcServer := func() { gracefulStopGrpc(grpcS, gracefulTimeout) }
|
||||
|
||||
var socketServer *http.Server
|
||||
if runtime.GOOS != "windows" {
|
||||
@@ -578,7 +574,7 @@ func (fo *FilerOptions) startFiler() {
|
||||
}
|
||||
httpS := newHttpServer(defaultHandler, tlsConfig)
|
||||
httpServers = append(httpServers, httpS)
|
||||
shutdown := newFilerShutdown(stopGrpcServer, fs.Shutdown, httpServers...)
|
||||
shutdown := newFilerShutdown(fs.LeaveLockRing, stopGrpcServer, fs.Shutdown, httpServers...)
|
||||
|
||||
grace.OnInterrupt(shutdown)
|
||||
|
||||
@@ -607,7 +603,7 @@ func (fo *FilerOptions) startFiler() {
|
||||
}
|
||||
httpS := newHttpServer(defaultHandler, nil)
|
||||
httpServers = append(httpServers, httpS)
|
||||
shutdown := newFilerShutdown(stopGrpcServer, fs.Shutdown, httpServers...)
|
||||
shutdown := newFilerShutdown(fs.LeaveLockRing, stopGrpcServer, fs.Shutdown, httpServers...)
|
||||
|
||||
grace.OnInterrupt(shutdown)
|
||||
|
||||
@@ -627,26 +623,12 @@ func (fo *FilerOptions) startFiler() {
|
||||
}
|
||||
|
||||
// newFilerShutdown joins shutdown callers while gRPC and HTTP drain concurrently.
|
||||
func newFilerShutdown(stopGrpc, shutdownFiler func(), httpServers ...*http.Server) func() {
|
||||
// The filer leaves the lock ring first: peers and S3 gateways route keys to it
|
||||
// until the ring changes, and would hit refused connections once gRPC stops.
|
||||
func newFilerShutdown(leaveLockRing, stopGrpc, shutdownFiler func(), httpServers ...*http.Server) func() {
|
||||
drain := newGracefulShutdown(stopGrpc, shutdownFiler, httpServers...)
|
||||
return sync.OnceFunc(func() {
|
||||
shutdownCtx, cancel := context.WithTimeout(context.Background(), 15*time.Second)
|
||||
defer cancel()
|
||||
var drained sync.WaitGroup
|
||||
drained.Add(1)
|
||||
go func() {
|
||||
defer drained.Done()
|
||||
stopGrpc()
|
||||
}()
|
||||
for _, server := range httpServers {
|
||||
drained.Add(1)
|
||||
go func() {
|
||||
defer drained.Done()
|
||||
if err := server.Shutdown(shutdownCtx); err != nil {
|
||||
glog.Warningf("filer HTTP shutdown: %v", err)
|
||||
}
|
||||
}()
|
||||
}
|
||||
drained.Wait()
|
||||
shutdownFiler()
|
||||
leaveLockRing()
|
||||
drain()
|
||||
})
|
||||
}
|
||||
|
||||
@@ -64,6 +64,10 @@ func (option *RemoteGatewayOptions) followBucketUpdatesAndUploadToRemote(filerSo
|
||||
StartTsNs: lastOffsetTs.UnixNano(),
|
||||
StopTsNs: 0,
|
||||
EventErrorType: pb.RetryForeverOnError,
|
||||
GetResumeTsNs: func() int64 {
|
||||
return processor.processedTsWatermark.Load()
|
||||
},
|
||||
Resubscribe: processor.ResubscribeCh(),
|
||||
}
|
||||
|
||||
return pb.FollowMetadata(pb.ServerAddress(*option.filerAddress), option.grpcDialOption, metadataFollowOption, processEventFnWithOffset)
|
||||
|
||||
@@ -5,6 +5,7 @@ import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"strings"
|
||||
"time"
|
||||
@@ -75,6 +76,10 @@ func followUpdatesAndUploadToRemote(option *RemoteSyncOptions, filerSource *sour
|
||||
StartTsNs: lastOffsetTs.UnixNano(),
|
||||
StopTsNs: 0,
|
||||
EventErrorType: pb.RetryForeverOnError,
|
||||
GetResumeTsNs: func() int64 {
|
||||
return processor.processedTsWatermark.Load()
|
||||
},
|
||||
Resubscribe: processor.ResubscribeCh(),
|
||||
}
|
||||
|
||||
return pb.FollowMetadata(pb.ServerAddress(*option.filerAddress), option.grpcDialOption, metadataFollowOption, processEventFnWithOffset)
|
||||
@@ -171,7 +176,6 @@ func (option *RemoteSyncOptions) makeEventProcessor(remoteStorage *remote_pb.Rem
|
||||
glog.V(0).Infof("mkdir %s", remote_storage.FormatLocation(dest))
|
||||
return client.WriteDirectory(dest, remoteWriteEntry(message.NewEntry, *option.storageClass))
|
||||
}
|
||||
glog.V(0).Infof("create %s", remote_storage.FormatLocation(dest))
|
||||
remoteEntry, writeErr := retriedWriteFile(client, filerSource, message.NewParentPath, remoteWriteEntry(message.NewEntry, *option.storageClass), dest)
|
||||
if errors.Is(writeErr, errSuperseded) {
|
||||
glog.Errorf("skipping %s: %v", remote_storage.FormatLocation(dest), writeErr)
|
||||
@@ -191,20 +195,7 @@ func (option *RemoteSyncOptions) makeEventProcessor(remoteStorage *remote_pb.Rem
|
||||
return updateLocalEntry(option, message.NewParentPath, message.NewEntry, remoteEntry)
|
||||
}
|
||||
if filer_pb.IsDelete(resp) {
|
||||
// Skip deletion of internal version files; individual version
|
||||
// deletes should not propagate to the remote object
|
||||
if isVersionedPath(resp.Directory, message.OldEntry.Name, message.OldEntry.IsDirectory) {
|
||||
glog.V(2).Infof("skipping delete of internal version path: %s/%s", resp.Directory, message.OldEntry.Name)
|
||||
return nil
|
||||
}
|
||||
glog.V(2).Infof("delete: %+v", resp)
|
||||
dest := toRemoteStorageLocation(util.FullPath(mountedDir), util.NewFullPath(resp.Directory, message.OldEntry.Name), remoteStorageMountLocation)
|
||||
if message.OldEntry.IsDirectory {
|
||||
glog.V(0).Infof("rmdir %s", remote_storage.FormatLocation(dest))
|
||||
return client.RemoveDirectory(dest)
|
||||
}
|
||||
glog.V(0).Infof("delete %s", remote_storage.FormatLocation(dest))
|
||||
return client.DeleteFile(dest)
|
||||
return processDeleteEvent(client, mountedDir, remoteStorageMountLocation, resp)
|
||||
}
|
||||
if message.OldEntry != nil && message.NewEntry != nil {
|
||||
return processUpdateEvent(option, filerSource, *option.storageClass, client, mountedDir, remoteStorageMountLocation, resp)
|
||||
@@ -215,6 +206,45 @@ func (option *RemoteSyncOptions) makeEventProcessor(remoteStorage *remote_pb.Rem
|
||||
return eachEntryFunc, nil
|
||||
}
|
||||
|
||||
// processDeleteEvent removes the remote object an entry mapped to. A remote
|
||||
// object that is already absent counts as deleted: the entry was never
|
||||
// uploaded (created and deleted faster than the sync ran, or a replay of an
|
||||
// event the inline delete already handled), and GCS reports that case as
|
||||
// ErrRemoteObjectNotFound where S3 and Azure answer an idempotent success.
|
||||
// Returning the error would pin the sync offset on an event that has nothing
|
||||
// left to do.
|
||||
func processDeleteEvent(
|
||||
client remote_storage.RemoteStorageClient,
|
||||
mountedDir string,
|
||||
remoteStorageMountLocation *remote_pb.RemoteStorageLocation,
|
||||
resp *filer_pb.SubscribeMetadataResponse,
|
||||
) error {
|
||||
message := resp.EventNotification
|
||||
// Skip deletion of internal version files; individual version
|
||||
// deletes should not propagate to the remote object
|
||||
if isVersionedPath(resp.Directory, message.OldEntry.Name, message.OldEntry.IsDirectory) {
|
||||
glog.V(2).Infof("skipping delete of internal version path: %s/%s", resp.Directory, message.OldEntry.Name)
|
||||
return nil
|
||||
}
|
||||
glog.V(2).Infof("delete: %+v", resp)
|
||||
dest := toRemoteStorageLocation(util.FullPath(mountedDir), util.NewFullPath(resp.Directory, message.OldEntry.Name), remoteStorageMountLocation)
|
||||
if message.OldEntry.IsDirectory {
|
||||
glog.V(0).Infof("rmdir %s", remote_storage.FormatLocation(dest))
|
||||
return client.RemoveDirectory(dest)
|
||||
}
|
||||
glog.V(0).Infof("delete %s", remote_storage.FormatLocation(dest))
|
||||
return deleteRemoteFile(client, dest)
|
||||
}
|
||||
|
||||
// deleteRemoteFile deletes the object and treats an already-absent object as
|
||||
// deleted.
|
||||
func deleteRemoteFile(client remote_storage.RemoteStorageClient, dest *remote_pb.RemoteStorageLocation) error {
|
||||
if err := client.DeleteFile(dest); err != nil && !errors.Is(err, remote_storage.ErrRemoteObjectNotFound) {
|
||||
return err
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
func processUpdateEvent(
|
||||
filerClient filer_pb.FilerClient,
|
||||
filerSource filer_pb.FilerClient,
|
||||
@@ -256,12 +286,22 @@ func processUpdateEvent(
|
||||
}
|
||||
glog.V(0).Infof("never replicated, uploading %s", remote_storage.FormatLocation(dest))
|
||||
}
|
||||
if !proto.Equal(oldDest, dest) && !filer.HasData(message.NewEntry) && message.NewEntry.IsInRemoteOnly() {
|
||||
glog.V(0).Infof("skip uploading renamed remote-only entry %s: content is only on the deleted remote object", remote_storage.FormatLocation(dest))
|
||||
return nil
|
||||
}
|
||||
glog.V(2).Infof("update: %+v", resp)
|
||||
if !proto.Equal(oldDest, dest) {
|
||||
// A renamed entry that holds no local data now (the snapshot was
|
||||
// remote-only, or remote.uncache ran since) has its content only as a
|
||||
// remote object. Remote-only reads resolve by the entry's own path, so
|
||||
// the rename completes only once the destination object holds it. The
|
||||
// filer's current state decides, not the snapshot: an entry rewritten
|
||||
// since the event has local data again, and the rewrite's bytes are
|
||||
// what the destination must hold.
|
||||
current, err := currentEntry(filerSource, message.NewParentPath, message.NewEntry.Name)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if isRemoteOnly(current) {
|
||||
return completeRemoteOnlyRename(filerClient, client, message.NewParentPath, current, oldDest, dest, storageClass)
|
||||
}
|
||||
glog.V(0).Infof("delete %s", remote_storage.FormatLocation(oldDest))
|
||||
if err := client.DeleteFile(oldDest); err != nil {
|
||||
if isMultipartUploadFile(resp.Directory, message.OldEntry.Name) {
|
||||
@@ -275,6 +315,9 @@ func processUpdateEvent(
|
||||
remoteEntry, writeErr := retriedWriteFile(client, filerSource, message.NewParentPath, remoteWriteEntry(message.NewEntry, storageClass), dest)
|
||||
if errors.Is(writeErr, errSuperseded) {
|
||||
glog.Errorf("skipping %s: %v", remote_storage.FormatLocation(dest), writeErr)
|
||||
if !proto.Equal(oldDest, dest) {
|
||||
return uploadCurrentEntry(filerClient, filerSource, client, message.NewParentPath, message.NewEntry.Name, oldDest, dest, storageClass)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
if writeErr != nil {
|
||||
@@ -283,6 +326,128 @@ func processUpdateEvent(
|
||||
return updateLocalEntry(filerClient, message.NewParentPath, message.NewEntry, remoteEntry)
|
||||
}
|
||||
|
||||
// currentEntry returns what the filer holds at dir/name now, nil when the
|
||||
// entry is gone.
|
||||
func currentEntry(filerSource filer_pb.FilerClient, dir, name string) (*filer_pb.Entry, error) {
|
||||
current, _, _, err := filer_pb.GetEntry(context.Background(), filerSource, util.NewFullPath(dir, name))
|
||||
if errors.Is(err, filer_pb.ErrNotFound) {
|
||||
return nil, nil
|
||||
}
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return current, nil
|
||||
}
|
||||
|
||||
// isRemoteOnly reports an entry whose content exists only as its remote
|
||||
// object: no local data, a RemoteEntry with a size.
|
||||
func isRemoteOnly(entry *filer_pb.Entry) bool {
|
||||
return entry != nil && !filer.HasData(entry) && entry.IsInRemoteOnly()
|
||||
}
|
||||
|
||||
// completeRemoteOnlyRename finishes a rename whose content exists only on the
|
||||
// remote. When the destination object already holds what the entry's stamp
|
||||
// describes (the rename ran before, its offset was not persisted, and
|
||||
// remote.uncache followed), the old key goes. Otherwise the old object is
|
||||
// copied over the destination, the entry is stamped with the copy, and the
|
||||
// old key goes. With neither object present the content is lost: the event
|
||||
// fails and holds the offset for recovery.
|
||||
func completeRemoteOnlyRename(filerClient filer_pb.FilerClient, client remote_storage.RemoteStorageClient, dir string, current *filer_pb.Entry, oldDest, dest *remote_pb.RemoteStorageLocation, storageClass string) error {
|
||||
if existing, err := client.StatFile(dest); err == nil {
|
||||
if describes(current.RemoteEntry, existing) {
|
||||
glog.V(0).Infof("%s already holds the renamed content", remote_storage.FormatLocation(dest))
|
||||
return deleteRemoteFile(client, oldDest)
|
||||
}
|
||||
glog.V(0).Infof("%s holds an object the entry does not describe (size %d, want %d); replacing it from %s", remote_storage.FormatLocation(dest), existing.RemoteSize, current.RemoteEntry.RemoteSize, remote_storage.FormatLocation(oldDest))
|
||||
} else if !errors.Is(err, remote_storage.ErrRemoteObjectNotFound) {
|
||||
return err
|
||||
}
|
||||
stat, err := client.StatFile(oldDest)
|
||||
if errors.Is(err, remote_storage.ErrRemoteObjectNotFound) {
|
||||
return fmt.Errorf("%s: content is on neither %s nor %s", util.NewFullPath(dir, current.Name), remote_storage.FormatLocation(oldDest), remote_storage.FormatLocation(dest))
|
||||
}
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
reader, err := openRemoteObject(client, oldDest, stat.RemoteSize)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
defer reader.Close()
|
||||
glog.V(0).Infof("copy %s -> %s", remote_storage.FormatLocation(oldDest), remote_storage.FormatLocation(dest))
|
||||
remoteEntry, err := client.WriteFile(dest, remoteWriteEntry(current, storageClass), reader)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := updateLocalEntry(filerClient, dir, current, remoteEntry); err != nil {
|
||||
return err
|
||||
}
|
||||
return deleteRemoteFile(client, oldDest)
|
||||
}
|
||||
|
||||
// describes reports whether the object a stat returned is the one the entry's
|
||||
// stamp describes: same size, and the same ETag when both sides carry one (a
|
||||
// copy written as one stream can legitimately carry a different ETag from a
|
||||
// multipart original).
|
||||
func describes(stamp, object *filer_pb.RemoteEntry) bool {
|
||||
if stamp == nil || object == nil || stamp.RemoteSize != object.RemoteSize {
|
||||
return false
|
||||
}
|
||||
return stamp.RemoteETag == "" || object.RemoteETag == "" || stamp.RemoteETag == object.RemoteETag
|
||||
}
|
||||
|
||||
// openRemoteObject streams the object when the client can, and reads it whole
|
||||
// otherwise.
|
||||
func openRemoteObject(client remote_storage.RemoteStorageClient, loc *remote_pb.RemoteStorageLocation, size int64) (io.ReadCloser, error) {
|
||||
if streamer, ok := client.(remote_storage.RemoteStorageStreamReader); ok {
|
||||
return streamer.ReadFileAsStream(context.Background(), loc, 0, size)
|
||||
}
|
||||
data, err := client.ReadFile(loc, 0, size)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return io.NopCloser(bytes.NewReader(data)), nil
|
||||
}
|
||||
|
||||
// uploadCurrentEntry uploads what the filer holds at dir/name now, when the
|
||||
// event that superseded this rename will not. A rename whose snapshot is
|
||||
// superseded has already deleted the old key; the rewrite behind it in the
|
||||
// log uploads the destination itself unless shouldSendToRemote skips it on
|
||||
// the inherited RemoteEntry, whose RemoteMtime can equal the rewrite's mtime
|
||||
// within the same second. Only that case uploads here, so the content goes
|
||||
// up once. A remote-only entry is finished the way completeRemoteOnlyRename
|
||||
// finishes a rename: its stamp can already describe the destination (a sync
|
||||
// plus remote.uncache in the meantime), and only then is it complete.
|
||||
func uploadCurrentEntry(filerClient filer_pb.FilerClient, filerSource filer_pb.FilerClient, client remote_storage.RemoteStorageClient, dir, name string, oldDest, dest *remote_pb.RemoteStorageLocation, storageClass string) error {
|
||||
current, _, _, err := filer_pb.GetEntry(context.Background(), filerSource, util.NewFullPath(dir, name))
|
||||
if errors.Is(err, filer_pb.ErrNotFound) {
|
||||
return nil
|
||||
}
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if current.IsDirectory {
|
||||
return nil
|
||||
}
|
||||
if isRemoteOnly(current) {
|
||||
return completeRemoteOnlyRename(filerClient, client, dir, current, oldDest, dest, storageClass)
|
||||
}
|
||||
if shouldSendToRemote(current) {
|
||||
glog.V(0).Infof("leaving %s to the rewrite that superseded the rename", remote_storage.FormatLocation(dest))
|
||||
return nil
|
||||
}
|
||||
glog.V(0).Infof("uploading the current %s in place of the superseded rename", remote_storage.FormatLocation(dest))
|
||||
remoteEntry, writeErr := retriedWriteFile(client, filerSource, dir, remoteWriteEntry(current, storageClass), dest)
|
||||
if errors.Is(writeErr, errSuperseded) {
|
||||
glog.Errorf("skipping %s: %v", remote_storage.FormatLocation(dest), writeErr)
|
||||
return nil
|
||||
}
|
||||
if writeErr != nil {
|
||||
return writeErr
|
||||
}
|
||||
return updateLocalEntry(filerClient, dir, current, remoteEntry)
|
||||
}
|
||||
|
||||
// isSuperseded reports whether the filer has moved past the entry an event
|
||||
// described: it is deleted, or it no longer references every chunk the event
|
||||
// named. Those are the chunks the filer deletes when an entry is updated, so
|
||||
@@ -304,6 +469,12 @@ func isSuperseded(filerClient filer_pb.FilerClient, dir string, entry *filer_pb.
|
||||
if err != nil {
|
||||
return false
|
||||
}
|
||||
if !filer.HasData(entry) && filer.HasData(current) {
|
||||
// The event described an entry without data (remote-only, or empty);
|
||||
// the filer has written to it since. Uploading the snapshot would put
|
||||
// an empty object where the write belongs.
|
||||
return true
|
||||
}
|
||||
if len(entry.Content) > 0 || len(current.Content) > 0 {
|
||||
return !bytes.Equal(entry.Content, current.Content)
|
||||
}
|
||||
@@ -314,11 +485,18 @@ func isSuperseded(filerClient filer_pb.FilerClient, dir string, entry *filer_pb.
|
||||
// moved past (isSuperseded). The caller skips the event instead of failing it.
|
||||
var errSuperseded = errors.New("deleted or rewritten since the event was logged")
|
||||
|
||||
// retriedWriteFile uploads the entry, retrying transient failures. Every failed
|
||||
// attempt first asks the filer whether the entry is superseded, and stops at
|
||||
// once when it is: a dead chunk reads as a transient "RequestError" from the
|
||||
// SDK, and waiting out the backoff on it buys nothing.
|
||||
// retriedWriteFile uploads the entry, retrying transient failures. The filer is
|
||||
// asked whether the entry is superseded before the first attempt and after
|
||||
// every failed one, and the upload stops at once when it is. Before: a backlog
|
||||
// (a restart resuming from an old offset) carries every intermediate version
|
||||
// of a hot file, and uploading each one in turn is wasted bandwidth that can
|
||||
// trip the remote's per-object mutation rate limit; only the version the filer
|
||||
// still holds is worth sending. After: a dead chunk reads as a transient
|
||||
// "RequestError" from the SDK, and waiting out the backoff on it buys nothing.
|
||||
func retriedWriteFile(client remote_storage.RemoteStorageClient, filerSource filer_pb.FilerClient, dir string, newEntry *filer_pb.Entry, dest *remote_pb.RemoteStorageLocation) (remoteEntry *filer_pb.RemoteEntry, err error) {
|
||||
if isSuperseded(filerSource, dir, newEntry) {
|
||||
return nil, fmt.Errorf("%s %w", util.NewFullPath(dir, newEntry.Name), errSuperseded)
|
||||
}
|
||||
err = util.RetryOnError("writeFile", func(err error) bool {
|
||||
return !errors.Is(err, errSuperseded) && util.IsTransientError(err)
|
||||
}, func() error {
|
||||
@@ -362,6 +540,11 @@ func collectLastSyncOffset(filerClient filer_pb.FilerClient, grpcDialOption grpc
|
||||
}
|
||||
} else {
|
||||
lastOffsetTs = time.Now().Add(-timeAgo)
|
||||
if lastOffsetTsNs, err := remote_storage.GetSyncOffset(grpcDialOption, filerAddress, mountedDir); err == nil && lastOffsetTsNs > 0 {
|
||||
if savedOffsetTs := time.Unix(0, lastOffsetTsNs); savedOffsetTs.Before(lastOffsetTs) {
|
||||
lastOffsetTs = savedOffsetTs
|
||||
}
|
||||
}
|
||||
}
|
||||
return lastOffsetTs
|
||||
}
|
||||
@@ -453,9 +636,29 @@ func updateLocalEntry(filerClient filer_pb.FilerClient, dir string, entry *filer
|
||||
glog.Errorf("skipping stale stamp of %s: %v", util.NewFullPath(dir, entry.Name), err)
|
||||
return nil
|
||||
}
|
||||
if isEntryGone(err) {
|
||||
glog.Errorf("skipping stamp of %s: deleted since the event was logged: %v", util.NewFullPath(dir, entry.Name), err)
|
||||
return nil
|
||||
}
|
||||
return err
|
||||
}
|
||||
|
||||
// isEntryGone reports an UpdateEntry the filer refused because the entry no
|
||||
// longer exists: a delete that followed the event superseded the stamp, and
|
||||
// the delete's own event follows in the log. A filer with the typed answer
|
||||
// returns codes.NotFound; an older filer returns a plain error of the form
|
||||
// "not found <path>: <cause>", so the cause (the message's tail, never the
|
||||
// path) is matched against filer_pb.ErrNotFound's text.
|
||||
func isEntryGone(err error) bool {
|
||||
if err == nil {
|
||||
return false
|
||||
}
|
||||
if st, ok := status.FromError(err); ok && st.Code() == codes.NotFound {
|
||||
return true
|
||||
}
|
||||
return strings.HasSuffix(strings.TrimSpace(err.Error()), filer_pb.ErrNotFound.Error())
|
||||
}
|
||||
|
||||
// ifEntryEqual builds the precondition that the stored entry still equals the
|
||||
// one the event described: chunk fids, inline content, and metadata alike.
|
||||
func ifEntryEqual(entry *filer_pb.Entry) *filer_pb.WriteCondition {
|
||||
@@ -501,7 +704,7 @@ func syncDeleteMarker(
|
||||
dest *remote_pb.RemoteStorageLocation,
|
||||
) error {
|
||||
glog.V(0).Infof("delete (marker) %s", remote_storage.FormatLocation(dest))
|
||||
if err := client.DeleteFile(dest); err != nil {
|
||||
if err := deleteRemoteFile(client, dest); err != nil {
|
||||
return err
|
||||
}
|
||||
return updateLocalEntry(filerClient, message.NewParentPath, message.NewEntry, &filer_pb.RemoteEntry{
|
||||
|
||||
@@ -415,9 +415,11 @@ func TestIsMetadataOnlyUpdate(t *testing.T) {
|
||||
// every lookup instead.
|
||||
type stubFilerClient struct {
|
||||
filer_pb.SeaweedFilerClient
|
||||
entry *filer_pb.Entry
|
||||
err error
|
||||
lookups int
|
||||
entry *filer_pb.Entry
|
||||
err error
|
||||
updateErr error
|
||||
lookups int
|
||||
updates int
|
||||
}
|
||||
|
||||
func (c *stubFilerClient) LookupDirectoryEntry(context.Context, *filer_pb.LookupDirectoryEntryRequest, ...grpc.CallOption) (*filer_pb.LookupDirectoryEntryResponse, error) {
|
||||
@@ -429,6 +431,10 @@ func (c *stubFilerClient) LookupDirectoryEntry(context.Context, *filer_pb.Lookup
|
||||
}
|
||||
|
||||
func (c *stubFilerClient) UpdateEntry(context.Context, *filer_pb.UpdateEntryRequest, ...grpc.CallOption) (*filer_pb.UpdateEntryResponse, error) {
|
||||
c.updates++
|
||||
if c.updateErr != nil {
|
||||
return &filer_pb.UpdateEntryResponse{}, c.updateErr
|
||||
}
|
||||
return &filer_pb.UpdateEntryResponse{}, nil
|
||||
}
|
||||
|
||||
@@ -549,6 +555,23 @@ func TestIsSuperseded(t *testing.T) {
|
||||
})
|
||||
}
|
||||
|
||||
t.Run("written since: the event described an entry without data", func(t *testing.T) {
|
||||
remoteOnly := entryWith("video.mp4", &filer_pb.RemoteEntry{StorageName: "b2", RemoteSize: 20971520, RemoteMtime: 1786096669})
|
||||
if !isSuperseded(&stubFilerClient{entry: entryWith("video.mp4", remoteOnly.RemoteEntry, chunk("3,09", "e9"))}, dir, remoteOnly) {
|
||||
t.Error("isSuperseded = false, want true: the filer wrote local data to a remote-only entry")
|
||||
}
|
||||
if isSuperseded(&stubFilerClient{entry: entryWith("video.mp4", remoteOnly.RemoteEntry)}, dir, remoteOnly) {
|
||||
t.Error("isSuperseded = true, want false: the entry is still remote-only")
|
||||
}
|
||||
empty := &filer_pb.Entry{Name: "touch.txt", Attributes: &filer_pb.FuseAttributes{Mtime: 1786096669}}
|
||||
if !isSuperseded(&stubFilerClient{entry: &filer_pb.Entry{Name: "touch.txt", Content: []byte("now")}}, dir, empty) {
|
||||
t.Error("isSuperseded = false, want true: the empty file was written to")
|
||||
}
|
||||
if isSuperseded(&stubFilerClient{entry: &filer_pb.Entry{Name: "touch.txt"}}, dir, empty) {
|
||||
t.Error("isSuperseded = true, want false: the file is still empty")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("inline content rewritten since", func(t *testing.T) {
|
||||
event := &filer_pb.Entry{Name: "note.txt", Content: []byte("v1")}
|
||||
if !isSuperseded(&stubFilerClient{entry: &filer_pb.Entry{Name: "note.txt", Content: []byte("v2")}}, dir, event) {
|
||||
@@ -584,7 +607,7 @@ func TestRetriedWriteFileStopsWhenSuperseded(t *testing.T) {
|
||||
// "requesterror" makes IsTransientError retry it to exhaustion.
|
||||
deadChunk := errors.New("RequestError: send request failed\ncaused by: Put \"https://s3.example.com/tier/x\": http://volume:8444/3,01?readDeleted=true: 404 Not Found: not found")
|
||||
|
||||
t.Run("superseded, one attempt", func(t *testing.T) {
|
||||
t.Run("superseded before the first attempt, no write", func(t *testing.T) {
|
||||
remote := &failingRemote{err: deadChunk}
|
||||
filerClient := &stubFilerClient{}
|
||||
start := time.Now()
|
||||
@@ -592,17 +615,32 @@ func TestRetriedWriteFileStopsWhenSuperseded(t *testing.T) {
|
||||
if !errors.Is(err, errSuperseded) {
|
||||
t.Errorf("err = %v, want errSuperseded", err)
|
||||
}
|
||||
if remote.writes != 0 {
|
||||
t.Errorf("wrote %d times, want 0: a superseded version is not worth uploading", remote.writes)
|
||||
}
|
||||
if filerClient.lookups != 1 {
|
||||
t.Errorf("looked up the filer %d times, want 1", filerClient.lookups)
|
||||
}
|
||||
if elapsed := time.Since(start); elapsed > 500*time.Millisecond {
|
||||
t.Errorf("took %v, want no backoff", elapsed)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("superseded during the attempt, one write", func(t *testing.T) {
|
||||
remote := &failingRemote{err: deadChunk}
|
||||
filerClient := &supersedingFilerClient{live: entryWith(event.Name, nil, chunk("3,01", "e1"))}
|
||||
_, err := retriedWriteFile(remote, filerClient, dir, event, dest)
|
||||
if !errors.Is(err, errSuperseded) {
|
||||
t.Errorf("err = %v, want errSuperseded", err)
|
||||
}
|
||||
if !strings.Contains(err.Error(), "404 Not Found") {
|
||||
t.Errorf("err = %v, want it to keep the write failure", err)
|
||||
}
|
||||
if remote.writes != 1 {
|
||||
t.Errorf("wrote %d times, want 1: retrying a dead chunk cannot succeed", remote.writes)
|
||||
}
|
||||
if filerClient.lookups != 1 {
|
||||
t.Errorf("looked up the filer %d times, want 1", filerClient.lookups)
|
||||
}
|
||||
if elapsed := time.Since(start); elapsed > 500*time.Millisecond {
|
||||
t.Errorf("took %v, want no backoff", elapsed)
|
||||
if filerClient.lookups != 2 {
|
||||
t.Errorf("looked up the filer %d times, want one before the attempt and one after it failed", filerClient.lookups)
|
||||
}
|
||||
})
|
||||
|
||||
@@ -619,8 +657,8 @@ func TestRetriedWriteFileStopsWhenSuperseded(t *testing.T) {
|
||||
if remote.writes != 2 {
|
||||
t.Errorf("wrote %d times, want 2: a transient failure on a live entry is still retried", remote.writes)
|
||||
}
|
||||
if filerClient.lookups != 2 {
|
||||
t.Errorf("looked up the filer %d times, want one per failed attempt", filerClient.lookups)
|
||||
if filerClient.lookups != 3 {
|
||||
t.Errorf("looked up the filer %d times, want one before the first attempt and one per failed attempt", filerClient.lookups)
|
||||
}
|
||||
})
|
||||
|
||||
@@ -639,16 +677,51 @@ func TestRetriedWriteFileStopsWhenSuperseded(t *testing.T) {
|
||||
|
||||
type recordingRemote struct {
|
||||
remote_storage.RemoteStorageClient
|
||||
deletes []*remote_pb.RemoteStorageLocation
|
||||
writes []*remote_pb.RemoteStorageLocation
|
||||
deleteErr error
|
||||
deletes []*remote_pb.RemoteStorageLocation
|
||||
writes []*remote_pb.RemoteStorageLocation
|
||||
written [][]byte
|
||||
// chunk ids of each written entry: which version of the file went up
|
||||
writtenChunks [][]string
|
||||
deleteErr error
|
||||
// objects present on the remote, by path: StatFile and ReadFile answer
|
||||
// from it, everything else is ErrRemoteObjectNotFound.
|
||||
objects map[string][]byte
|
||||
}
|
||||
|
||||
func (r *recordingRemote) WriteFile(loc *remote_pb.RemoteStorageLocation, entry *filer_pb.Entry, _ io.Reader) (*filer_pb.RemoteEntry, error) {
|
||||
func (r *recordingRemote) WriteFile(loc *remote_pb.RemoteStorageLocation, entry *filer_pb.Entry, reader io.Reader) (*filer_pb.RemoteEntry, error) {
|
||||
r.writes = append(r.writes, loc)
|
||||
var ids []string
|
||||
for _, c := range entry.GetChunks() {
|
||||
ids = append(ids, c.GetFileIdString())
|
||||
}
|
||||
r.writtenChunks = append(r.writtenChunks, ids)
|
||||
// A chunked upload's reader walks volume servers the stub filer cannot
|
||||
// name, so it is not read; the copy path hands over the old object's
|
||||
// stream for a chunkless entry, which is drained.
|
||||
var body []byte
|
||||
if rc, ok := reader.(io.ReadCloser); ok && len(entry.GetChunks()) == 0 && r.objects != nil {
|
||||
body, _ = io.ReadAll(rc)
|
||||
}
|
||||
r.written = append(r.written, body)
|
||||
return &filer_pb.RemoteEntry{StorageName: loc.Name, RemoteETag: "etag", RemoteSize: int64(len(entry.Content)), RemoteMtime: entry.Attributes.GetMtime()}, nil
|
||||
}
|
||||
|
||||
func (r *recordingRemote) StatFile(loc *remote_pb.RemoteStorageLocation) (*filer_pb.RemoteEntry, error) {
|
||||
body, ok := r.objects[loc.Path]
|
||||
if !ok {
|
||||
return nil, remote_storage.ErrRemoteObjectNotFound
|
||||
}
|
||||
return &filer_pb.RemoteEntry{StorageName: loc.Name, RemoteSize: int64(len(body))}, nil
|
||||
}
|
||||
|
||||
func (r *recordingRemote) ReadFile(loc *remote_pb.RemoteStorageLocation, offset int64, size int64) ([]byte, error) {
|
||||
body, ok := r.objects[loc.Path]
|
||||
if !ok {
|
||||
return nil, remote_storage.ErrRemoteObjectNotFound
|
||||
}
|
||||
return body[offset : offset+size], nil
|
||||
}
|
||||
|
||||
func (r *recordingRemote) DeleteFile(loc *remote_pb.RemoteStorageLocation) error {
|
||||
r.deletes = append(r.deletes, loc)
|
||||
return r.deleteErr
|
||||
@@ -685,7 +758,8 @@ func TestRenameWithInheritedRemoteEntryWritesNewKey(t *testing.T) {
|
||||
}
|
||||
|
||||
remote := &recordingRemote{}
|
||||
filerClient := &stubFilerClient{}
|
||||
// the filer still holds the renamed entry as the event described it
|
||||
filerClient := &stubFilerClient{entry: newEntry}
|
||||
if err := processUpdateEvent(filerClient, filerClient, "", remote, mountedDir, mountLoc, resp); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
@@ -801,7 +875,8 @@ func TestRenameDeleteOldKeyNotFoundStillWrites(t *testing.T) {
|
||||
}
|
||||
|
||||
remote := &recordingRemote{deleteErr: remote_storage.ErrRemoteObjectNotFound}
|
||||
filerClient := &stubFilerClient{}
|
||||
// the filer still holds the renamed entry as the event described it
|
||||
filerClient := &stubFilerClient{entry: newEntry}
|
||||
if err := processUpdateEvent(filerClient, filerClient, "", remote, mountedDir, mountLoc, resp); err != nil {
|
||||
t.Fatalf("err = %v, want nil: an already-deleted old key must not block the write", err)
|
||||
}
|
||||
@@ -880,3 +955,336 @@ func TestIsFailedPrecondition(t *testing.T) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// supersedingFilerClient answers the first lookup with the live entry and
|
||||
// every later one with "not found": the entry is deleted while the upload is
|
||||
// in flight.
|
||||
type supersedingFilerClient struct {
|
||||
filer_pb.SeaweedFilerClient
|
||||
live *filer_pb.Entry
|
||||
lookups int
|
||||
}
|
||||
|
||||
func (c *supersedingFilerClient) LookupDirectoryEntry(context.Context, *filer_pb.LookupDirectoryEntryRequest, ...grpc.CallOption) (*filer_pb.LookupDirectoryEntryResponse, error) {
|
||||
c.lookups++
|
||||
if c.lookups == 1 {
|
||||
return &filer_pb.LookupDirectoryEntryResponse{Entry: c.live}, nil
|
||||
}
|
||||
return nil, filer_pb.ErrNotFound
|
||||
}
|
||||
|
||||
func (c *supersedingFilerClient) WithFilerClient(_ bool, fn func(filer_pb.SeaweedFilerClient) error) error {
|
||||
return fn(c)
|
||||
}
|
||||
|
||||
func (c *supersedingFilerClient) AdjustedUrl(location *filer_pb.Location) string { return location.Url }
|
||||
|
||||
func (c *supersedingFilerClient) GetDataCenter() string { return "" }
|
||||
|
||||
// TestDeleteEventAbsentRemoteObjectIsSuccess: an entry created and deleted
|
||||
// before the sync uploaded it has no remote object. GCS reports the delete as
|
||||
// ErrRemoteObjectNotFound; the event has nothing left to do, so it must not
|
||||
// fail (a failed event pins the sync offset and is replayed on every restart).
|
||||
func TestDeleteEventAbsentRemoteObjectIsSuccess(t *testing.T) {
|
||||
const mountedDir = "/buckets"
|
||||
mountLoc := &remote_pb.RemoteStorageLocation{Name: "gcs", Bucket: "bucket", Path: "/"}
|
||||
resp := &filer_pb.SubscribeMetadataResponse{
|
||||
Directory: "/buckets/b/vol",
|
||||
EventNotification: &filer_pb.EventNotification{
|
||||
OldEntry: &filer_pb.Entry{Name: "state.db.tmp", Attributes: &filer_pb.FuseAttributes{Mtime: 1786096669}},
|
||||
DeleteChunks: true,
|
||||
},
|
||||
}
|
||||
wantDelete := &remote_pb.RemoteStorageLocation{Name: "gcs", Bucket: "bucket", Path: "/b/vol/state.db.tmp"}
|
||||
|
||||
t.Run("absent object", func(t *testing.T) {
|
||||
remote := &recordingRemote{deleteErr: remote_storage.ErrRemoteObjectNotFound}
|
||||
if err := processDeleteEvent(remote, mountedDir, mountLoc, resp); err != nil {
|
||||
t.Fatalf("err = %v, want nil: deleting an absent object is complete", err)
|
||||
}
|
||||
if len(remote.deletes) != 1 || !proto.Equal(remote.deletes[0], wantDelete) {
|
||||
t.Errorf("deletes = %+v, want %s", remote.deletes, remote_storage.FormatLocation(wantDelete))
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("other failure still fails", func(t *testing.T) {
|
||||
remote := &recordingRemote{deleteErr: errors.New("AccessDenied: Access Denied")}
|
||||
if err := processDeleteEvent(remote, mountedDir, mountLoc, resp); err == nil {
|
||||
t.Fatal("err = nil, want the delete failure")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("delete marker, absent object", func(t *testing.T) {
|
||||
remote := &recordingRemote{deleteErr: remote_storage.ErrRemoteObjectNotFound}
|
||||
filerClient := &stubFilerClient{}
|
||||
markerDelete := &remote_pb.RemoteStorageLocation{Name: "gcs", Bucket: "bucket", Path: "/b/vol/state.db"}
|
||||
message := &filer_pb.EventNotification{
|
||||
NewParentPath: "/buckets/b/vol",
|
||||
NewEntry: &filer_pb.Entry{Name: "state.db", Attributes: &filer_pb.FuseAttributes{Mtime: 1786096669}},
|
||||
}
|
||||
if err := syncDeleteMarker(remote, filerClient, message, markerDelete); err != nil {
|
||||
t.Fatalf("err = %v, want nil", err)
|
||||
}
|
||||
if len(remote.deletes) != 1 || !proto.Equal(remote.deletes[0], markerDelete) {
|
||||
t.Errorf("deletes = %+v, want %s", remote.deletes, remote_storage.FormatLocation(markerDelete))
|
||||
}
|
||||
if filerClient.updates != 1 {
|
||||
t.Errorf("stamped %d times, want 1: the marker is recorded locally so a replay is a no-op", filerClient.updates)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// TestUpdateLocalEntrySkipsDeletedEntry: the stamp that follows an upload lands
|
||||
// on an entry a later event already deleted. The filer answers "not found"; the
|
||||
// delete's own event follows in the log, so the stamp is skipped rather than
|
||||
// failed.
|
||||
func TestUpdateLocalEntrySkipsDeletedEntry(t *testing.T) {
|
||||
const dir = "/buckets/b/vol"
|
||||
remoteEntry := &filer_pb.RemoteEntry{StorageName: "gcs", RemoteSize: 7}
|
||||
|
||||
t.Run("plain not found from the filer", func(t *testing.T) {
|
||||
filerClient := &stubFilerClient{updateErr: fmt.Errorf("not found %s: %w", dir+"/state.db", filer_pb.ErrNotFound)}
|
||||
if err := updateLocalEntry(filerClient, dir, entryWith("state.db", nil, chunk("3,01", "e1")), remoteEntry); err != nil {
|
||||
t.Errorf("err = %v, want nil", err)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("not found over grpc", func(t *testing.T) {
|
||||
filerClient := &stubFilerClient{updateErr: status.Error(codes.Unknown, "not found "+dir+"/state.db: "+filer_pb.ErrNotFound.Error())}
|
||||
if err := updateLocalEntry(filerClient, dir, entryWith("state.db", nil, chunk("3,01", "e1")), remoteEntry); err != nil {
|
||||
t.Errorf("err = %v, want nil", err)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("typed not found from the filer", func(t *testing.T) {
|
||||
filerClient := &stubFilerClient{updateErr: status.Errorf(codes.NotFound, "not found %s/state.db: %v", dir, filer_pb.ErrNotFound)}
|
||||
if err := updateLocalEntry(filerClient, dir, entryWith("state.db", nil, chunk("3,01", "e1")), remoteEntry); err != nil {
|
||||
t.Errorf("err = %v, want nil", err)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("the sentinel inside the path is not a missing entry", func(t *testing.T) {
|
||||
name := filer_pb.ErrNotFound.Error()
|
||||
filerClient := &stubFilerClient{updateErr: status.Error(codes.Unknown, "not found "+dir+"/"+name+": database unavailable")}
|
||||
if err := updateLocalEntry(filerClient, dir, entryWith(name, nil, chunk("3,01", "e1")), remoteEntry); err == nil {
|
||||
t.Error("err = nil, want the store failure: the stamp must be retried, not dropped")
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("other failure still fails", func(t *testing.T) {
|
||||
filerClient := &stubFilerClient{updateErr: status.Error(codes.Unavailable, "filer is shutting down")}
|
||||
if err := updateLocalEntry(filerClient, dir, entryWith("state.db", nil, chunk("3,01", "e1")), remoteEntry); err == nil {
|
||||
t.Error("err = nil, want the update failure")
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// TestSupersededRenameUploadsCurrentEntry: a rename A -> B queued behind a
|
||||
// rewrite of B. The rename's snapshot is superseded (B's chunks changed) and
|
||||
// the old key A is already deleted. The rewrite's own event uploads B unless
|
||||
// shouldSendToRemote skips it on the inherited RemoteEntry (same-second
|
||||
// mtimes); only then does the rename upload what the filer holds now, so the
|
||||
// content goes up exactly once.
|
||||
func TestSupersededRenameUploadsCurrentEntry(t *testing.T) {
|
||||
const mountedDir = "/buckets"
|
||||
mountLoc := &remote_pb.RemoteStorageLocation{Name: "gcs", Bucket: "bucket", Path: "/"}
|
||||
eventEntry := entryWith("b.txt", nil, chunk("3,01", "e1"))
|
||||
resp := &filer_pb.SubscribeMetadataResponse{
|
||||
Directory: "/buckets/b/dir",
|
||||
EventNotification: &filer_pb.EventNotification{
|
||||
OldEntry: entryWith("a.txt", nil, chunk("3,01", "e1")),
|
||||
NewParentPath: "/buckets/b/dir",
|
||||
NewEntry: eventEntry,
|
||||
},
|
||||
}
|
||||
wantDelete := &remote_pb.RemoteStorageLocation{Name: "gcs", Bucket: "bucket", Path: "/b/dir/a.txt"}
|
||||
wantWrite := &remote_pb.RemoteStorageLocation{Name: "gcs", Bucket: "bucket", Path: "/b/dir/b.txt"}
|
||||
|
||||
t.Run("rewrite hidden by the inherited stamp: upload once here", func(t *testing.T) {
|
||||
// RemoteMtime inherited from a.txt equals the rewrite's mtime.
|
||||
current := entryWith("b.txt", &filer_pb.RemoteEntry{StorageName: "gcs", RemoteMtime: 1786096669, RemoteSize: 1024}, chunk("3,02", "e2"))
|
||||
remote := &recordingRemote{}
|
||||
filerClient := &stubFilerClient{entry: current}
|
||||
if err := processUpdateEvent(filerClient, filerClient, "", remote, mountedDir, mountLoc, resp); err != nil {
|
||||
t.Fatalf("err = %v, want nil", err)
|
||||
}
|
||||
if len(remote.deletes) != 1 || !proto.Equal(remote.deletes[0], wantDelete) {
|
||||
t.Errorf("deletes = %+v, want the old key %s deleted", remote.deletes, remote_storage.FormatLocation(wantDelete))
|
||||
}
|
||||
if len(remote.writes) != 1 || !proto.Equal(remote.writes[0], wantWrite) {
|
||||
t.Fatalf("writes = %+v, want the new key %s written once with the current entry", remote.writes, remote_storage.FormatLocation(wantWrite))
|
||||
}
|
||||
if filerClient.updates != 1 {
|
||||
t.Errorf("stamped %d times, want 1: the current entry is stamped", filerClient.updates)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("rewrite newer than the stamp: its own event uploads, not this one", func(t *testing.T) {
|
||||
current := entryWith("b.txt", &filer_pb.RemoteEntry{StorageName: "gcs", RemoteMtime: 1786096000, RemoteSize: 1024}, chunk("3,02", "e2"))
|
||||
remote := &recordingRemote{}
|
||||
filerClient := &stubFilerClient{entry: current}
|
||||
if err := processUpdateEvent(filerClient, filerClient, "", remote, mountedDir, mountLoc, resp); err != nil {
|
||||
t.Fatalf("err = %v, want nil", err)
|
||||
}
|
||||
if len(remote.deletes) != 1 {
|
||||
t.Errorf("deletes = %+v, want the old key deleted", remote.deletes)
|
||||
}
|
||||
if len(remote.writes) != 0 {
|
||||
t.Errorf("writes = %+v, want none: the rewrite event behind this one uploads b.txt", remote.writes)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("deleted meanwhile: nothing to upload", func(t *testing.T) {
|
||||
remote := &recordingRemote{}
|
||||
filerClient := &stubFilerClient{}
|
||||
if err := processUpdateEvent(filerClient, filerClient, "", remote, mountedDir, mountLoc, resp); err != nil {
|
||||
t.Fatalf("err = %v, want nil", err)
|
||||
}
|
||||
if len(remote.writes) != 0 {
|
||||
t.Errorf("writes = %+v, want none: the delete event follows", remote.writes)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("uncached meanwhile, old object present: copied to the destination, stamped, old key deleted", func(t *testing.T) {
|
||||
current := entryWith("b.txt", &filer_pb.RemoteEntry{StorageName: "gcs", RemoteMtime: 1786096669, RemoteSize: 7})
|
||||
remote := &recordingRemote{objects: map[string][]byte{"/b/dir/a.txt": []byte("payload")}}
|
||||
filerClient := &stubFilerClient{entry: current}
|
||||
if err := processUpdateEvent(filerClient, filerClient, "", remote, mountedDir, mountLoc, resp); err != nil {
|
||||
t.Fatalf("err = %v, want nil", err)
|
||||
}
|
||||
if len(remote.writes) != 1 || !proto.Equal(remote.writes[0], wantWrite) || string(remote.written[0]) != "payload" {
|
||||
t.Fatalf("writes = %+v (%q), want the old object's bytes written at %s", remote.writes, remote.written, remote_storage.FormatLocation(wantWrite))
|
||||
}
|
||||
if filerClient.updates != 1 {
|
||||
t.Errorf("stamped %d times, want 1: the entry now points at the destination object", filerClient.updates)
|
||||
}
|
||||
if len(remote.deletes) != 1 || !proto.Equal(remote.deletes[0], wantDelete) {
|
||||
t.Errorf("deletes = %+v, want the old key deleted after the copy", remote.deletes)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("uncached after an earlier run completed the rename: destination present, old key deleted, nothing written", func(t *testing.T) {
|
||||
current := entryWith("b.txt", &filer_pb.RemoteEntry{StorageName: "gcs", RemoteMtime: 1786096669, RemoteSize: 7})
|
||||
remote := &recordingRemote{objects: map[string][]byte{"/b/dir/a.txt": []byte("payload"), "/b/dir/b.txt": []byte("payload")}}
|
||||
filerClient := &stubFilerClient{entry: current}
|
||||
if err := processUpdateEvent(filerClient, filerClient, "", remote, mountedDir, mountLoc, resp); err != nil {
|
||||
t.Fatalf("err = %v, want nil: the replayed rename is already complete", err)
|
||||
}
|
||||
if len(remote.writes) != 0 {
|
||||
t.Errorf("writes = %+v, want none", remote.writes)
|
||||
}
|
||||
if len(remote.deletes) != 1 || !proto.Equal(remote.deletes[0], wantDelete) {
|
||||
t.Errorf("deletes = %+v, want the old key deleted", remote.deletes)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("uncached and neither object exists: the event fails and holds the offset", func(t *testing.T) {
|
||||
current := entryWith("b.txt", &filer_pb.RemoteEntry{StorageName: "gcs", RemoteMtime: 1786096669, RemoteSize: 7})
|
||||
remote := &recordingRemote{}
|
||||
filerClient := &stubFilerClient{entry: current}
|
||||
err := processUpdateEvent(filerClient, filerClient, "", remote, mountedDir, mountLoc, resp)
|
||||
if err == nil || !strings.Contains(err.Error(), "neither") {
|
||||
t.Fatalf("err = %v, want the lost-content failure", err)
|
||||
}
|
||||
if len(remote.writes) != 0 || len(remote.deletes) != 0 {
|
||||
t.Errorf("writes = %+v deletes = %+v, want none", remote.writes, remote.deletes)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("destination holds an object of another size: replaced from the old key", func(t *testing.T) {
|
||||
current := entryWith("b.txt", &filer_pb.RemoteEntry{StorageName: "gcs", RemoteMtime: 1786096669, RemoteSize: 7})
|
||||
remote := &recordingRemote{objects: map[string][]byte{"/b/dir/a.txt": []byte("payload"), "/b/dir/b.txt": []byte("something else")}}
|
||||
filerClient := &stubFilerClient{entry: current}
|
||||
if err := processUpdateEvent(filerClient, filerClient, "", remote, mountedDir, mountLoc, resp); err != nil {
|
||||
t.Fatalf("err = %v, want nil", err)
|
||||
}
|
||||
if len(remote.writes) != 1 || string(remote.written[0]) != "payload" {
|
||||
t.Fatalf("writes = %+v (%q), want the old object copied over the destination", remote.writes, remote.written)
|
||||
}
|
||||
if len(remote.deletes) != 1 || !proto.Equal(remote.deletes[0], wantDelete) {
|
||||
t.Errorf("deletes = %+v, want the old key deleted after the copy", remote.deletes)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("snapshot remote-only but rewritten since: the rewrite's bytes go up, nothing is copied", func(t *testing.T) {
|
||||
remoteOnly := &filer_pb.RemoteEntry{StorageName: "gcs", RemoteMtime: 1786096669, RemoteSize: 7}
|
||||
snapshot := &filer_pb.SubscribeMetadataResponse{
|
||||
Directory: "/buckets/b/dir",
|
||||
EventNotification: &filer_pb.EventNotification{
|
||||
OldEntry: entryWith("a.txt", remoteOnly),
|
||||
NewParentPath: "/buckets/b/dir",
|
||||
NewEntry: entryWith("b.txt", remoteOnly),
|
||||
},
|
||||
}
|
||||
// rewritten with the inherited stamp still hiding it from shouldSendToRemote
|
||||
current := entryWith("b.txt", remoteOnly, chunk("3,09", "e9"))
|
||||
remote := &recordingRemote{objects: map[string][]byte{"/b/dir/a.txt": []byte("payload")}}
|
||||
filerClient := &stubFilerClient{entry: current}
|
||||
if err := processUpdateEvent(filerClient, filerClient, "", remote, mountedDir, mountLoc, snapshot); err != nil {
|
||||
t.Fatalf("err = %v, want nil", err)
|
||||
}
|
||||
if len(remote.writes) != 1 || !proto.Equal(remote.writes[0], wantWrite) {
|
||||
t.Fatalf("writes = %+v, want one upload of the rewritten b.txt", remote.writes)
|
||||
}
|
||||
if len(remote.writtenChunks[0]) != 1 || remote.writtenChunks[0][0] != "3,09" {
|
||||
t.Errorf("uploaded chunks = %v, want the rewrite's chunk 3,09, not the chunkless snapshot", remote.writtenChunks[0])
|
||||
}
|
||||
if string(remote.written[0]) == "payload" {
|
||||
t.Error("the old object's bytes were copied over a rewritten entry")
|
||||
}
|
||||
if len(remote.deletes) != 1 {
|
||||
t.Errorf("deletes = %+v, want the old key deleted", remote.deletes)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("snapshot already remote-only: completed the same way", func(t *testing.T) {
|
||||
remoteOnly := &filer_pb.RemoteEntry{StorageName: "gcs", RemoteMtime: 1786096669, RemoteSize: 7}
|
||||
snapshot := &filer_pb.SubscribeMetadataResponse{
|
||||
Directory: "/buckets/b/dir",
|
||||
EventNotification: &filer_pb.EventNotification{
|
||||
OldEntry: entryWith("a.txt", remoteOnly),
|
||||
NewParentPath: "/buckets/b/dir",
|
||||
NewEntry: entryWith("b.txt", remoteOnly),
|
||||
},
|
||||
}
|
||||
remote := &recordingRemote{objects: map[string][]byte{"/b/dir/a.txt": []byte("payload")}}
|
||||
filerClient := &stubFilerClient{entry: entryWith("b.txt", remoteOnly)}
|
||||
if err := processUpdateEvent(filerClient, filerClient, "", remote, mountedDir, mountLoc, snapshot); err != nil {
|
||||
t.Fatalf("err = %v, want nil", err)
|
||||
}
|
||||
if len(remote.writes) != 1 || string(remote.written[0]) != "payload" {
|
||||
t.Errorf("writes = %+v (%q), want the old object copied to the destination", remote.writes, remote.written)
|
||||
}
|
||||
if len(remote.deletes) != 1 {
|
||||
t.Errorf("deletes = %+v, want the old key deleted after the copy", remote.deletes)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("uncached after the old key was deleted: the event fails and holds the offset", func(t *testing.T) {
|
||||
current := entryWith("b.txt", &filer_pb.RemoteEntry{StorageName: "gcs", RemoteMtime: 1786096669, RemoteSize: 1024})
|
||||
remote := &recordingRemote{}
|
||||
filerClient := &stubFilerClient{entry: current}
|
||||
err := uploadCurrentEntry(filerClient, filerClient, remote, "/buckets/b/dir", "b.txt", wantDelete, wantWrite, "")
|
||||
if err == nil || !strings.Contains(err.Error(), "neither") {
|
||||
t.Errorf("err = %v, want the lost-content failure", err)
|
||||
}
|
||||
if len(remote.writes) != 0 {
|
||||
t.Errorf("writes = %+v, want none", remote.writes)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("uncached with the destination already synced: complete without a write", func(t *testing.T) {
|
||||
current := entryWith("b.txt", &filer_pb.RemoteEntry{StorageName: "gcs", RemoteMtime: 1786096669, RemoteSize: 7})
|
||||
remote := &recordingRemote{objects: map[string][]byte{"/b/dir/b.txt": []byte("payload")}}
|
||||
filerClient := &stubFilerClient{entry: current}
|
||||
if err := uploadCurrentEntry(filerClient, filerClient, remote, "/buckets/b/dir", "b.txt", wantDelete, wantWrite, ""); err != nil {
|
||||
t.Fatalf("err = %v, want nil: the destination object matches the entry's stamp", err)
|
||||
}
|
||||
if len(remote.writes) != 0 {
|
||||
t.Errorf("writes = %+v, want none", remote.writes)
|
||||
}
|
||||
if len(remote.deletes) != 1 || !proto.Equal(remote.deletes[0], wantDelete) {
|
||||
t.Errorf("deletes = %+v, want the old key deleted", remote.deletes)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
@@ -56,7 +56,7 @@ func TestFilerShutdownJoinsServeDuringParallelDrain(t *testing.T) {
|
||||
grpcStopped := make(chan struct{})
|
||||
filerClosed := make(chan struct{})
|
||||
var closes atomic.Int32
|
||||
shutdown := newFilerShutdown(func() {
|
||||
shutdown := newFilerShutdown(func() {}, func() {
|
||||
close(grpcStarted)
|
||||
<-grpcRelease
|
||||
close(grpcStopped)
|
||||
@@ -138,7 +138,7 @@ func TestFilerShutdownWaitsForHTTPAfterGrpcStops(t *testing.T) {
|
||||
server.Config.RegisterOnShutdown(func() { close(httpClosing) })
|
||||
grpcStopped := make(chan struct{})
|
||||
filerClosed := make(chan struct{})
|
||||
shutdown := newFilerShutdown(func() { close(grpcStopped) }, func() { close(filerClosed) }, server.Config)
|
||||
shutdown := newFilerShutdown(func() {}, func() { close(grpcStopped) }, func() { close(filerClosed) }, server.Config)
|
||||
joined := make(chan struct{})
|
||||
go func() { shutdown(); close(joined) }()
|
||||
<-httpClosing
|
||||
@@ -157,3 +157,42 @@ func TestFilerShutdownWaitsForHTTPAfterGrpcStops(t *testing.T) {
|
||||
t.Error("filer was not closed after HTTP request completed")
|
||||
}
|
||||
}
|
||||
|
||||
func TestFilerShutdownLeavesLockRingBeforeDraining(t *testing.T) {
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {}))
|
||||
t.Cleanup(server.Close)
|
||||
httpClosing := make(chan struct{})
|
||||
server.Config.RegisterOnShutdown(func() { close(httpClosing) })
|
||||
|
||||
leaving := make(chan struct{})
|
||||
releaseLeave := make(chan struct{})
|
||||
finishLeave := sync.OnceFunc(func() { close(releaseLeave) })
|
||||
t.Cleanup(finishLeave)
|
||||
grpcStopping := make(chan struct{})
|
||||
shutdown := newFilerShutdown(func() {
|
||||
close(leaving)
|
||||
<-releaseLeave
|
||||
}, func() { close(grpcStopping) }, func() {}, server.Config)
|
||||
joined := make(chan struct{})
|
||||
go func() { shutdown(); close(joined) }()
|
||||
|
||||
select {
|
||||
case <-leaving:
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatal("shutdown did not leave the lock ring")
|
||||
}
|
||||
select {
|
||||
case <-grpcStopping:
|
||||
t.Fatal("gRPC stopped accepting while the filer was still leaving the lock ring")
|
||||
case <-httpClosing:
|
||||
t.Fatal("HTTP stopped accepting while the filer was still leaving the lock ring")
|
||||
case <-time.After(100 * time.Millisecond):
|
||||
}
|
||||
finishLeave()
|
||||
<-joined
|
||||
select {
|
||||
case <-grpcStopping:
|
||||
default:
|
||||
t.Error("gRPC was not stopped after leaving the lock ring")
|
||||
}
|
||||
}
|
||||
|
||||
@@ -475,6 +475,10 @@ func doSubscribeFilerMetaChanges(clientId int32, clientEpoch int32, sourceGrpcDi
|
||||
StartTsNs: sourceFilerOffsetTsNs,
|
||||
StopTsNs: 0,
|
||||
EventErrorType: pb.RetryForeverOnError,
|
||||
GetResumeTsNs: func() int64 {
|
||||
return processor.processedTsWatermark.Load()
|
||||
},
|
||||
Resubscribe: processor.ResubscribeCh(),
|
||||
// While the source has only read activity it emits no metadata events, so
|
||||
// the watermark above never advances and sync_offset would look stuck.
|
||||
// The idle heartbeat moves the gauge to the source's current time once we
|
||||
|
||||
+160
-23
@@ -16,6 +16,11 @@ import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
)
|
||||
|
||||
// maxFailedSyncEvents bounds the failedTs ledger. A destination rejecting
|
||||
// every event would otherwise add an entry per source event for the life of
|
||||
// the processor.
|
||||
var maxFailedSyncEvents = 1 << 16
|
||||
|
||||
// tsMinHeap implements heap.Interface for int64 timestamps.
|
||||
type tsMinHeap []int64
|
||||
|
||||
@@ -60,6 +65,16 @@ type syncJobPaths struct {
|
||||
dataSize int64
|
||||
}
|
||||
|
||||
// failedEventKey identifies an event for the failure ledger. A timestamp alone
|
||||
// is not unique across events, so a success for one event must not clear an
|
||||
// unresolved failure recorded for a different event at the same TsNs.
|
||||
type failedEventKey struct {
|
||||
tsNs int64
|
||||
path util.FullPath
|
||||
newPath util.FullPath
|
||||
kind jobKind
|
||||
}
|
||||
|
||||
// syncStreamMetrics holds the metric children for one sync stream, curried
|
||||
// once so per-event updates skip the label lookup.
|
||||
type syncStreamMetrics struct {
|
||||
@@ -74,12 +89,18 @@ type syncStreamMetrics struct {
|
||||
}
|
||||
|
||||
type MetadataProcessor struct {
|
||||
activeJobs map[int64]*syncJobPaths
|
||||
// activeJobCount is the number of in-flight jobs and activeJobTs counts
|
||||
// them per event timestamp. Several events can share a TsNs — batched
|
||||
// writes log together — so a single slot per TsNs would drain early and
|
||||
// let the resubscribe or the watermark outrun a sibling still running.
|
||||
activeJobCount int
|
||||
activeJobTs map[int64]int
|
||||
activeJobsLock sync.Mutex
|
||||
activeJobsCond *sync.Cond
|
||||
concurrencyLimit int
|
||||
fn pb.ProcessMetadataFunc
|
||||
processedTsWatermark atomic.Int64
|
||||
filteredTsNs int64
|
||||
|
||||
// Indexes for O(depth) conflict detection, replacing O(n) linear scan.
|
||||
// activeFilePaths counts active file jobs at each exact path.
|
||||
@@ -105,12 +126,35 @@ type MetadataProcessor struct {
|
||||
// used for O(log n) amortized watermark tracking.
|
||||
tsHeap tsMinHeap
|
||||
|
||||
// oldestFailedTsNs is the timestamp of the oldest event whose job returned
|
||||
// an error, or 0 when none has. The watermark is never advanced to it or
|
||||
// past it, so the persisted sync offset stays behind the failure and a
|
||||
// restart replays the event instead of skipping it forever.
|
||||
// failedTs records every event whose job returned an error and has not
|
||||
// since completed, and oldestFailedTsNs caches its minimum (0 when empty).
|
||||
// The watermark is never advanced to it or past it, so the persisted sync
|
||||
// offset stays behind the failure and a restart replays the event instead
|
||||
// of skipping it forever. Past maxFailedSyncEvents the set collapses to a
|
||||
// sticky pin at the smallest failure seen: replay from the oldest failure
|
||||
// still works, but individual recoveries no longer unpin until a restart.
|
||||
failedTs map[failedEventKey]struct{}
|
||||
failedSticky bool
|
||||
oldestFailedTsNs int64
|
||||
|
||||
// resubscribeCh closes once a failure has stopped the processor and all
|
||||
// in-flight jobs have drained, asking the metadata follower to drop the
|
||||
// stream so the caller's reconnect replays the pinned events in order —
|
||||
// and never races the replay against work still running in this abandoned
|
||||
// processor.
|
||||
resubscribeCh chan struct{}
|
||||
resubscribeOnce sync.Once
|
||||
|
||||
// stopped latches when a job failure pins the watermark. Admission then
|
||||
// drops new events instead of queueing them into a processor that is about
|
||||
// to be abandoned: every skipped event replays from the pinned watermark
|
||||
// after the reconnect, and on a stream that never goes quiet the drain —
|
||||
// and with it the replay — would otherwise never come. The cond broadcast
|
||||
// releases a blocked AddSyncJob. A redelivery of an event still in
|
||||
// failedTs is the one exception: it may run so its success shrinks the
|
||||
// replay.
|
||||
stopped bool
|
||||
|
||||
// metrics is nil for callers that do not report per-event metrics.
|
||||
metrics *syncStreamMetrics
|
||||
}
|
||||
@@ -118,12 +162,14 @@ type MetadataProcessor struct {
|
||||
func NewMetadataProcessor(fn pb.ProcessMetadataFunc, concurrency int, offsetTsNs int64) *MetadataProcessor {
|
||||
t := &MetadataProcessor{
|
||||
fn: fn,
|
||||
activeJobs: make(map[int64]*syncJobPaths),
|
||||
activeJobTs: make(map[int64]int),
|
||||
concurrencyLimit: concurrency,
|
||||
activeFilePaths: make(map[util.FullPath]int),
|
||||
activeBarrierDirPaths: make(map[util.FullPath]int),
|
||||
activeNonBarrierDirPaths: make(map[util.FullPath]int),
|
||||
descendantCount: make(map[util.FullPath]int),
|
||||
failedTs: make(map[failedEventKey]struct{}),
|
||||
resubscribeCh: make(chan struct{}),
|
||||
}
|
||||
t.processedTsWatermark.Store(offsetTsNs)
|
||||
t.activeJobsCond = sync.NewCond(&t.activeJobsLock)
|
||||
@@ -153,6 +199,13 @@ func (t *MetadataProcessor) OldestFailedTsNs() int64 {
|
||||
return t.oldestFailedTsNs
|
||||
}
|
||||
|
||||
// ResubscribeCh closes once a job failure has stopped the processor and its
|
||||
// in-flight jobs have drained, signaling the metadata follower to drop the
|
||||
// stream so a reconnect replays what the watermark still covers.
|
||||
func (t *MetadataProcessor) ResubscribeCh() <-chan struct{} {
|
||||
return t.resubscribeCh
|
||||
}
|
||||
|
||||
// pathAncestors returns all proper ancestor directories of p.
|
||||
// For "/a/b/c", returns ["/a/b", "/a", "/"].
|
||||
func pathAncestors(p util.FullPath) []util.FullPath {
|
||||
@@ -281,29 +334,65 @@ func (t *MetadataProcessor) conflictsWith(resp *filer_pb.SubscribeMetadataRespon
|
||||
|
||||
func (t *MetadataProcessor) AddSyncJob(resp *filer_pb.SubscribeMetadataResponse) {
|
||||
if filer_pb.IsEmpty(resp) {
|
||||
// A filtered-progress marker means the source skipped everything below
|
||||
// it for us; once all earlier work has finished, the watermark can move
|
||||
// to it so idle stretches still advance the resume point.
|
||||
t.activeJobsLock.Lock()
|
||||
defer t.activeJobsLock.Unlock()
|
||||
if resp.TsNs > t.filteredTsNs {
|
||||
t.filteredTsNs = resp.TsNs
|
||||
}
|
||||
if t.activeJobCount == 0 && resp.TsNs > t.processedTsWatermark.Load() &&
|
||||
(t.oldestFailedTsNs == 0 || resp.TsNs < t.oldestFailedTsNs) {
|
||||
t.processedTsWatermark.Store(resp.TsNs)
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
dataSize := eventDataSize(resp)
|
||||
|
||||
// counted before the admission wait: received-processed-failed is the
|
||||
// number of events read off the stream but not yet done
|
||||
t.activeJobsLock.Lock()
|
||||
defer t.activeJobsLock.Unlock()
|
||||
|
||||
p, newPath, kind := extractJobInfo(resp)
|
||||
eventKey := failedEventKey{tsNs: resp.TsNs, path: p, newPath: newPath, kind: kind}
|
||||
|
||||
for t.activeJobCount >= t.concurrencyLimit || t.conflictsWith(resp) {
|
||||
// A stopped processor never queues: an event that cannot start drops
|
||||
// and replays in order after the resubscribe.
|
||||
if t.stopped {
|
||||
return
|
||||
}
|
||||
t.activeJobsCond.Wait()
|
||||
}
|
||||
if t.stopped {
|
||||
select {
|
||||
case <-t.resubscribeCh:
|
||||
// Already drained and signaled: nothing may start now, or it would
|
||||
// race the replay the signal just asked for.
|
||||
return
|
||||
default:
|
||||
}
|
||||
// The one event still worth running is a failure still pinning the
|
||||
// watermark, redelivered on this stream: its success clears the ledger
|
||||
// entry and shrinks the replay. Everything else replays anyway.
|
||||
if _, pinned := t.failedTs[eventKey]; !pinned {
|
||||
return
|
||||
}
|
||||
}
|
||||
|
||||
// counted once admitted: received-processed-failed is the number of events
|
||||
// this processor read but has not finished. A dropped event is not counted
|
||||
// here — the replay's own admission counts it.
|
||||
if t.metrics != nil {
|
||||
t.metrics.received.Inc()
|
||||
t.metrics.receivedBytes.Add(float64(dataSize))
|
||||
}
|
||||
|
||||
t.activeJobsLock.Lock()
|
||||
defer t.activeJobsLock.Unlock()
|
||||
|
||||
for len(t.activeJobs) >= t.concurrencyLimit || t.conflictsWith(resp) {
|
||||
t.activeJobsCond.Wait()
|
||||
}
|
||||
|
||||
p, newPath, kind := extractJobInfo(resp)
|
||||
jobPaths := &syncJobPaths{path: p, newPath: newPath, kind: kind, dataSize: dataSize}
|
||||
|
||||
t.activeJobs[resp.TsNs] = jobPaths
|
||||
t.activeJobCount++
|
||||
t.activeJobTs[resp.TsNs]++
|
||||
t.addPathToIndex(p, kind)
|
||||
if newPath != "" {
|
||||
t.addPathToIndex(newPath, kind)
|
||||
@@ -327,16 +416,55 @@ func (t *MetadataProcessor) AddSyncJob(resp *filer_pb.SubscribeMetadataResponse)
|
||||
t.activeJobsLock.Lock()
|
||||
defer t.activeJobsLock.Unlock()
|
||||
|
||||
failedKey := failedEventKey{tsNs: resp.TsNs, path: jobPaths.path, newPath: jobPaths.newPath, kind: jobPaths.kind}
|
||||
if jobErr != nil {
|
||||
if t.oldestFailedTsNs == 0 || resp.TsNs < t.oldestFailedTsNs {
|
||||
t.oldestFailedTsNs = resp.TsNs
|
||||
glog.Errorf("process %v: %v; holding sync offset at %v so this event is replayed on restart", resp, jobErr, time.Unix(0, resp.TsNs))
|
||||
} else {
|
||||
// Latch the stop the moment a failure lands: events behind the pin
|
||||
// replay after the resubscribe anyway, and a stream that never goes
|
||||
// quiet would otherwise keep the active count above zero so the
|
||||
// failure stays pinned until an unrelated reconnect — the wait this
|
||||
// whole mechanism exists to remove.
|
||||
t.stopped = true
|
||||
t.activeJobsCond.Broadcast()
|
||||
if t.failedSticky {
|
||||
if resp.TsNs < t.oldestFailedTsNs {
|
||||
t.oldestFailedTsNs = resp.TsNs
|
||||
}
|
||||
glog.Errorf("process %v: %v", resp, jobErr)
|
||||
} else if _, recorded := t.failedTs[failedKey]; !recorded {
|
||||
if len(t.failedTs) >= maxFailedSyncEvents {
|
||||
t.failedSticky = true
|
||||
t.failedTs = nil
|
||||
if resp.TsNs < t.oldestFailedTsNs {
|
||||
t.oldestFailedTsNs = resp.TsNs
|
||||
}
|
||||
glog.Warningf("process %v: %v; over %d unresolved failures, pinning sync offset at %v until restart", resp, jobErr, maxFailedSyncEvents, time.Unix(0, t.oldestFailedTsNs))
|
||||
} else {
|
||||
t.failedTs[failedKey] = struct{}{}
|
||||
if t.oldestFailedTsNs == 0 || resp.TsNs < t.oldestFailedTsNs {
|
||||
t.oldestFailedTsNs = resp.TsNs
|
||||
glog.Errorf("process %v: %v; holding sync offset at %v so this event is replayed on restart", resp, jobErr, time.Unix(0, resp.TsNs))
|
||||
} else {
|
||||
glog.Errorf("process %v: %v", resp, jobErr)
|
||||
}
|
||||
}
|
||||
}
|
||||
} else if _, recorded := t.failedTs[failedKey]; recorded {
|
||||
delete(t.failedTs, failedKey)
|
||||
if resp.TsNs == t.oldestFailedTsNs {
|
||||
t.oldestFailedTsNs = 0
|
||||
for k := range t.failedTs {
|
||||
if t.oldestFailedTsNs == 0 || k.tsNs < t.oldestFailedTsNs {
|
||||
t.oldestFailedTsNs = k.tsNs
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
delete(t.activeJobs, resp.TsNs)
|
||||
t.activeJobCount--
|
||||
t.activeJobTs[resp.TsNs]--
|
||||
if t.activeJobTs[resp.TsNs] == 0 {
|
||||
delete(t.activeJobTs, resp.TsNs)
|
||||
}
|
||||
t.removePathFromIndex(jobPaths.path, jobPaths.kind)
|
||||
if jobPaths.newPath != "" {
|
||||
t.removePathFromIndex(jobPaths.newPath, jobPaths.kind)
|
||||
@@ -356,7 +484,7 @@ func (t *MetadataProcessor) AddSyncJob(resp *filer_pb.SubscribeMetadataResponse)
|
||||
// Lazy-clean stale entries from heap top (already-completed jobs).
|
||||
// Each entry is pushed once and popped once: O(log n) amortized.
|
||||
for t.tsHeap.Len() > 0 {
|
||||
if _, active := t.activeJobs[t.tsHeap[0]]; active {
|
||||
if t.activeJobTs[t.tsHeap[0]] > 0 {
|
||||
break
|
||||
}
|
||||
heap.Pop(&t.tsHeap)
|
||||
@@ -369,6 +497,15 @@ func (t *MetadataProcessor) AddSyncJob(resp *filer_pb.SubscribeMetadataResponse)
|
||||
t.processedTsWatermark.Store(resp.TsNs)
|
||||
}
|
||||
}
|
||||
if t.activeJobCount == 0 && t.filteredTsNs > t.processedTsWatermark.Load() &&
|
||||
(t.oldestFailedTsNs == 0 || t.filteredTsNs < t.oldestFailedTsNs) {
|
||||
t.processedTsWatermark.Store(t.filteredTsNs)
|
||||
}
|
||||
// Signal once the stop has drained: even if a redelivery cleared the
|
||||
// pin, events dropped while stopped still have to replay.
|
||||
if t.stopped && t.activeJobCount == 0 {
|
||||
t.resubscribeOnce.Do(func() { close(t.resubscribeCh) })
|
||||
}
|
||||
t.activeJobsCond.Signal()
|
||||
}()
|
||||
}
|
||||
|
||||
@@ -94,8 +94,8 @@ func TestFileVsFileConflict(t *testing.T) {
|
||||
|
||||
// Add a file job
|
||||
active := makeResp("/dir1", "file.txt", false, 1, true)
|
||||
path, newPath, kind := extractJobInfo(active)
|
||||
p.activeJobs[active.TsNs] = &syncJobPaths{path: path, newPath: newPath, kind: kind}
|
||||
path, _, kind := extractJobInfo(active)
|
||||
p.activeJobTs[active.TsNs] = 1
|
||||
p.addPathToIndex(path, kind)
|
||||
|
||||
// Same file should conflict
|
||||
@@ -125,8 +125,8 @@ func TestFileUnderActiveDirConflict(t *testing.T) {
|
||||
|
||||
// Add a directory job at /dir1
|
||||
active := makeResp("/", "dir1", true, 1, true)
|
||||
path, newPath, kind := extractJobInfo(active)
|
||||
p.activeJobs[active.TsNs] = &syncJobPaths{path: path, newPath: newPath, kind: kind}
|
||||
path, _, kind := extractJobInfo(active)
|
||||
p.activeJobTs[active.TsNs] = 1
|
||||
p.addPathToIndex(path, kind)
|
||||
|
||||
// File under /dir1 should conflict
|
||||
@@ -163,8 +163,8 @@ func TestDirWithActiveFileUnder(t *testing.T) {
|
||||
|
||||
// Add file jobs under /dir1
|
||||
f1 := makeResp("/dir1/sub", "file.txt", false, 1, true)
|
||||
path, newPath, kind := extractJobInfo(f1)
|
||||
p.activeJobs[f1.TsNs] = &syncJobPaths{path: path, newPath: newPath, kind: kind}
|
||||
path, _, kind := extractJobInfo(f1)
|
||||
p.activeJobTs[f1.TsNs] = 1
|
||||
p.addPathToIndex(path, kind)
|
||||
|
||||
// Directory /dir1 should conflict (has active file under it)
|
||||
@@ -187,8 +187,8 @@ func TestDirVsDirConflict(t *testing.T) {
|
||||
|
||||
// Add directory job at /a/b
|
||||
active := makeResp("/a", "b", true, 1, true)
|
||||
path, newPath, kind := extractJobInfo(active)
|
||||
p.activeJobs[active.TsNs] = &syncJobPaths{path: path, newPath: newPath, kind: kind}
|
||||
path, _, kind := extractJobInfo(active)
|
||||
p.activeJobTs[active.TsNs] = 1
|
||||
p.addPathToIndex(path, kind)
|
||||
|
||||
// /a/b/c (descendant) should conflict
|
||||
@@ -224,8 +224,8 @@ func TestRenameConflict(t *testing.T) {
|
||||
|
||||
// Add file job at /dir1/file.txt
|
||||
f1 := makeResp("/dir1", "file.txt", false, 1, true)
|
||||
path, newPath, kind := extractJobInfo(f1)
|
||||
p.activeJobs[f1.TsNs] = &syncJobPaths{path: path, newPath: newPath, kind: kind}
|
||||
path, _, kind := extractJobInfo(f1)
|
||||
p.activeJobTs[f1.TsNs] = 1
|
||||
p.addPathToIndex(path, kind)
|
||||
|
||||
// Rename from /dir2/a.txt to /dir1/file.txt should conflict (newPath matches)
|
||||
@@ -255,7 +255,7 @@ func TestActiveRenameConflict(t *testing.T) {
|
||||
// Add active rename job: /dir1/old.txt -> /dir2/new.txt
|
||||
rename := makeRenameResp("/dir1", "old.txt", "/dir2", "new.txt", false, 1)
|
||||
path, newPath, kind := extractJobInfo(rename)
|
||||
p.activeJobs[rename.TsNs] = &syncJobPaths{path: path, newPath: newPath, kind: kind}
|
||||
p.activeJobTs[rename.TsNs] = 1
|
||||
p.addPathToIndex(path, kind)
|
||||
if newPath != "" {
|
||||
p.addPathToIndex(newPath, kind)
|
||||
@@ -289,8 +289,8 @@ func TestRootDirConflict(t *testing.T) {
|
||||
// Note: a dir entry at "/" would be created as FullPath("/").Child("somedir")
|
||||
// But let's test what happens with an active dir at /some/path and check root
|
||||
active := makeResp("/some", "dir", true, 1, true)
|
||||
path, newPath, kind := extractJobInfo(active)
|
||||
p.activeJobs[active.TsNs] = &syncJobPaths{path: path, newPath: newPath, kind: kind}
|
||||
path, _, kind := extractJobInfo(active)
|
||||
p.activeJobTs[active.TsNs] = 1
|
||||
p.addPathToIndex(path, kind)
|
||||
|
||||
// Root dir should conflict because active dir /some/dir is under /
|
||||
@@ -334,7 +334,7 @@ func TestWatermarkWithHeap(t *testing.T) {
|
||||
// Simulate adding jobs in order
|
||||
for _, ts := range []int64{10, 20, 30} {
|
||||
jobPath := util.FullPath("/file" + string(rune('0'+ts/10)))
|
||||
p.activeJobs[ts] = &syncJobPaths{path: jobPath, kind: kindFile}
|
||||
p.activeJobTs[ts] = 1
|
||||
p.addPathToIndex(jobPath, kindFile)
|
||||
heap.Push(&p.tsHeap, ts)
|
||||
}
|
||||
@@ -344,11 +344,11 @@ func TestWatermarkWithHeap(t *testing.T) {
|
||||
}
|
||||
|
||||
// Remove non-oldest (ts=20) — heap top should stay 10
|
||||
delete(p.activeJobs, 20)
|
||||
delete(p.activeJobTs, 20)
|
||||
p.removePathFromIndex("/file2", kindFile)
|
||||
// Lazy clean: top is 10 which is still active, so no pop
|
||||
for p.tsHeap.Len() > 0 {
|
||||
if _, active := p.activeJobs[p.tsHeap[0]]; active {
|
||||
if p.activeJobTs[p.tsHeap[0]] > 0 {
|
||||
break
|
||||
}
|
||||
heap.Pop(&p.tsHeap)
|
||||
@@ -358,10 +358,10 @@ func TestWatermarkWithHeap(t *testing.T) {
|
||||
}
|
||||
|
||||
// Remove oldest (ts=10) — lazy clean should find 30
|
||||
delete(p.activeJobs, 10)
|
||||
delete(p.activeJobTs, 10)
|
||||
p.removePathFromIndex("/file1", kindFile)
|
||||
for p.tsHeap.Len() > 0 {
|
||||
if _, active := p.activeJobs[p.tsHeap[0]]; active {
|
||||
if p.activeJobTs[p.tsHeap[0]] > 0 {
|
||||
break
|
||||
}
|
||||
heap.Pop(&p.tsHeap)
|
||||
@@ -383,11 +383,11 @@ func TestNonBarrierDirUpdateDoesNotBlockDescendants(t *testing.T) {
|
||||
|
||||
// Active non-barrier: attribute update on /dir1.
|
||||
active := makeDirUpdateResp("/", "dir1", 1)
|
||||
path, newPath, kind := extractJobInfo(active)
|
||||
path, _, kind := extractJobInfo(active)
|
||||
if kind != kindNonBarrierDir {
|
||||
t.Fatalf("expected kindNonBarrierDir for dir attribute update, got %v", kind)
|
||||
}
|
||||
p.activeJobs[active.TsNs] = &syncJobPaths{path: path, newPath: newPath, kind: kind}
|
||||
p.activeJobTs[active.TsNs] = 1
|
||||
p.addPathToIndex(path, kind)
|
||||
|
||||
// File under /dir1 should NOT conflict with the attribute update.
|
||||
@@ -407,11 +407,11 @@ func TestNonBarrierDirUpdateDoesNotBlockDescendants(t *testing.T) {
|
||||
p := NewMetadataProcessor(noop, 100, 0)
|
||||
|
||||
active := makeResp("/", "dir1", true, 1, true) // create
|
||||
path, newPath, kind := extractJobInfo(active)
|
||||
path, _, kind := extractJobInfo(active)
|
||||
if kind != kindBarrierDir {
|
||||
t.Fatalf("expected kindBarrierDir for dir create, got %v", kind)
|
||||
}
|
||||
p.activeJobs[active.TsNs] = &syncJobPaths{path: path, newPath: newPath, kind: kind}
|
||||
p.activeJobTs[active.TsNs] = 1
|
||||
p.addPathToIndex(path, kind)
|
||||
|
||||
under := makeResp("/dir1", "file.txt", false, 2, true)
|
||||
@@ -425,8 +425,8 @@ func TestNonBarrierDirUpdateDoesNotBlockDescendants(t *testing.T) {
|
||||
|
||||
// Active file under /dir1.
|
||||
f := makeResp("/dir1", "file.txt", false, 1, true)
|
||||
path, newPath, kind := extractJobInfo(f)
|
||||
p.activeJobs[f.TsNs] = &syncJobPaths{path: path, newPath: newPath, kind: kind}
|
||||
path, _, kind := extractJobInfo(f)
|
||||
p.activeJobTs[f.TsNs] = 1
|
||||
p.addPathToIndex(path, kind)
|
||||
|
||||
// Incoming barrier delete on /dir1 should still wait for the
|
||||
@@ -442,8 +442,8 @@ func TestNonBarrierDirUpdateDoesNotBlockDescendants(t *testing.T) {
|
||||
|
||||
// Active non-barrier dir update at /a/b.
|
||||
upd := makeDirUpdateResp("/a", "b", 1)
|
||||
path, newPath, kind := extractJobInfo(upd)
|
||||
p.activeJobs[upd.TsNs] = &syncJobPaths{path: path, newPath: newPath, kind: kind}
|
||||
path, _, kind := extractJobInfo(upd)
|
||||
p.activeJobTs[upd.TsNs] = 1
|
||||
p.addPathToIndex(path, kind)
|
||||
|
||||
// A barrier delete on /a (the ancestor) should wait for it.
|
||||
@@ -464,8 +464,8 @@ func TestSamePathBarrierSerialization(t *testing.T) {
|
||||
t.Run("barrier dir at p blocks same-path file", func(t *testing.T) {
|
||||
p := NewMetadataProcessor(noop, 100, 0)
|
||||
active := makeResp("/", "dir1", true, 1, true) // dir create
|
||||
path, newPath, kind := extractJobInfo(active)
|
||||
p.activeJobs[active.TsNs] = &syncJobPaths{path: path, newPath: newPath, kind: kind}
|
||||
path, _, kind := extractJobInfo(active)
|
||||
p.activeJobTs[active.TsNs] = 1
|
||||
p.addPathToIndex(path, kind)
|
||||
|
||||
file := makeResp("/", "dir1", false, 2, true)
|
||||
@@ -477,8 +477,8 @@ func TestSamePathBarrierSerialization(t *testing.T) {
|
||||
t.Run("barrier dir at p blocks another same-path barrier dir", func(t *testing.T) {
|
||||
p := NewMetadataProcessor(noop, 100, 0)
|
||||
active := makeResp("/", "dir1", true, 1, true) // dir create
|
||||
path, newPath, kind := extractJobInfo(active)
|
||||
p.activeJobs[active.TsNs] = &syncJobPaths{path: path, newPath: newPath, kind: kind}
|
||||
path, _, kind := extractJobInfo(active)
|
||||
p.activeJobTs[active.TsNs] = 1
|
||||
p.addPathToIndex(path, kind)
|
||||
|
||||
del := makeResp("/", "dir1", true, 2, false) // dir delete, same path
|
||||
@@ -490,8 +490,8 @@ func TestSamePathBarrierSerialization(t *testing.T) {
|
||||
t.Run("barrier dir at p blocks non-barrier update at same path", func(t *testing.T) {
|
||||
p := NewMetadataProcessor(noop, 100, 0)
|
||||
active := makeResp("/", "dir1", true, 1, true) // dir create
|
||||
path, newPath, kind := extractJobInfo(active)
|
||||
p.activeJobs[active.TsNs] = &syncJobPaths{path: path, newPath: newPath, kind: kind}
|
||||
path, _, kind := extractJobInfo(active)
|
||||
p.activeJobTs[active.TsNs] = 1
|
||||
p.addPathToIndex(path, kind)
|
||||
|
||||
upd := makeDirUpdateResp("/", "dir1", 2)
|
||||
@@ -503,8 +503,8 @@ func TestSamePathBarrierSerialization(t *testing.T) {
|
||||
t.Run("file at p blocks same-path barrier dir", func(t *testing.T) {
|
||||
p := NewMetadataProcessor(noop, 100, 0)
|
||||
active := makeResp("/", "thing", false, 1, true) // file create at /thing
|
||||
path, newPath, kind := extractJobInfo(active)
|
||||
p.activeJobs[active.TsNs] = &syncJobPaths{path: path, newPath: newPath, kind: kind}
|
||||
path, _, kind := extractJobInfo(active)
|
||||
p.activeJobTs[active.TsNs] = 1
|
||||
p.addPathToIndex(path, kind)
|
||||
|
||||
// Barrier dir at /thing (e.g. a file→dir promotion) must wait.
|
||||
@@ -520,11 +520,11 @@ func TestSamePathBarrierSerialization(t *testing.T) {
|
||||
// delete/rename/create on /dir1.
|
||||
p := NewMetadataProcessor(noop, 100, 0)
|
||||
active := makeDirUpdateResp("/", "dir1", 1)
|
||||
path, newPath, kind := extractJobInfo(active)
|
||||
path, _, kind := extractJobInfo(active)
|
||||
if kind != kindNonBarrierDir {
|
||||
t.Fatalf("expected kindNonBarrierDir, got %v", kind)
|
||||
}
|
||||
p.activeJobs[active.TsNs] = &syncJobPaths{path: path, newPath: newPath, kind: kind}
|
||||
p.activeJobTs[active.TsNs] = 1
|
||||
p.addPathToIndex(path, kind)
|
||||
|
||||
del := makeResp("/", "dir1", true, 2, false) // dir delete
|
||||
@@ -541,8 +541,8 @@ func TestSamePathBarrierSerialization(t *testing.T) {
|
||||
t.Run("non-barrier update at p does NOT block same-path non-barrier update", func(t *testing.T) {
|
||||
p := NewMetadataProcessor(noop, 100, 0)
|
||||
active := makeDirUpdateResp("/", "dir1", 1)
|
||||
path, newPath, kind := extractJobInfo(active)
|
||||
p.activeJobs[active.TsNs] = &syncJobPaths{path: path, newPath: newPath, kind: kind}
|
||||
path, _, kind := extractJobInfo(active)
|
||||
p.activeJobTs[active.TsNs] = 1
|
||||
p.addPathToIndex(path, kind)
|
||||
|
||||
// Concurrent attribute bumps are allowed: last writer wins.
|
||||
@@ -569,8 +569,8 @@ func BenchmarkConflictCheck(b *testing.B) {
|
||||
dir := fmt.Sprintf("/dir%d/sub%d", i/100, i%100)
|
||||
name := fmt.Sprintf("file%d.txt", i)
|
||||
resp := makeResp(dir, name, false, int64(i+1), true)
|
||||
path, newPath, kind := extractJobInfo(resp)
|
||||
p.activeJobs[resp.TsNs] = &syncJobPaths{path: path, newPath: newPath, kind: kind}
|
||||
path, _, kind := extractJobInfo(resp)
|
||||
p.activeJobTs[resp.TsNs] = 1
|
||||
p.addPathToIndex(path, kind)
|
||||
}
|
||||
|
||||
@@ -586,40 +586,13 @@ func BenchmarkConflictCheck(b *testing.B) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestMetadataProcessorEmptyMarkerKeepsWatermarkStale: the MaxUnsyncedEvents
|
||||
// marker (empty EventNotification, fresh timestamp) is dropped by AddSyncJob and
|
||||
// does NOT advance processedTsWatermark, so offsetFunc keeps publishing the stale
|
||||
// offset. This is why the client must not drive sync_offset off the watermark
|
||||
// for these markers.
|
||||
func TestMetadataProcessorEmptyMarkerKeepsWatermarkStale(t *testing.T) {
|
||||
const staleOffset = int64(1_000_000_000)
|
||||
freshTs := staleOffset + int64(time.Hour) // a "now"-ish source timestamp
|
||||
|
||||
p := NewMetadataProcessor(func(*filer_pb.SubscribeMetadataResponse) error { return nil }, 4, staleOffset)
|
||||
|
||||
marker := &filer_pb.SubscribeMetadataResponse{
|
||||
TsNs: freshTs,
|
||||
EventNotification: &filer_pb.EventNotification{},
|
||||
}
|
||||
if !filer_pb.IsEmpty(marker) {
|
||||
t.Fatal("marker should be IsEmpty")
|
||||
}
|
||||
|
||||
p.AddSyncJob(marker)
|
||||
|
||||
if got := p.processedTsWatermark.Load(); got != staleOffset {
|
||||
t.Fatalf("empty marker advanced watermark to %d; want it to stay stale at %d", got, staleOffset)
|
||||
}
|
||||
t.Logf("marker carried fresh ts %d but watermark stayed stale at %d", freshTs, staleOffset)
|
||||
}
|
||||
|
||||
// waitForJobsToDrain blocks until every job goroutine has finished bookkeeping.
|
||||
func waitForJobsToDrain(t *testing.T, p *MetadataProcessor) {
|
||||
t.Helper()
|
||||
deadline := time.Now().Add(10 * time.Second)
|
||||
for time.Now().Before(deadline) {
|
||||
p.activeJobsLock.Lock()
|
||||
remaining := len(p.activeJobs)
|
||||
remaining := p.activeJobCount
|
||||
p.activeJobsLock.Unlock()
|
||||
if remaining == 0 {
|
||||
return
|
||||
@@ -651,35 +624,54 @@ func TestFailedJobHoldsWatermark(t *testing.T) {
|
||||
t.Fatalf("watermark = %d after a successful job, want 100", got)
|
||||
}
|
||||
|
||||
p.AddSyncJob(makeResp("/dir", "b.txt", false, failedTsNs, true))
|
||||
waitForJobsToDrain(t, p)
|
||||
if got := p.processedTsWatermark.Load(); got != 100 {
|
||||
t.Fatalf("watermark = %d after a failed job, want it held at 100", got)
|
||||
// a later event still in flight when the failure lands finishes fine, but
|
||||
// the offset stays behind the failure
|
||||
release := make(chan struct{})
|
||||
slowFn := func(resp *filer_pb.SubscribeMetadataResponse) error {
|
||||
if resp.TsNs == 300 {
|
||||
<-release
|
||||
}
|
||||
return fn(resp)
|
||||
}
|
||||
|
||||
// later events keep flowing, but the offset stays behind the failure
|
||||
p.AddSyncJob(makeResp("/dir", "c.txt", false, 300, true))
|
||||
waitForJobsToDrain(t, p)
|
||||
if got := p.processedTsWatermark.Load(); got != 100 {
|
||||
p2 := NewMetadataProcessor(slowFn, 10, 0)
|
||||
// admit the slow job first so it is in flight when the failure lands
|
||||
p2.AddSyncJob(makeResp("/dir", "c.txt", false, 300, true))
|
||||
p2.AddSyncJob(makeResp("/dir", "a.txt", false, 100, true))
|
||||
p2.AddSyncJob(makeResp("/dir", "b.txt", false, failedTsNs, true))
|
||||
deadline := time.Now().Add(10 * time.Second)
|
||||
for p2.OldestFailedTsNs() == 0 && time.Now().Before(deadline) {
|
||||
time.Sleep(time.Millisecond)
|
||||
}
|
||||
if p2.OldestFailedTsNs() != failedTsNs {
|
||||
t.Fatalf("oldest failed = %d, want the pin at %d", p2.OldestFailedTsNs(), failedTsNs)
|
||||
}
|
||||
close(release)
|
||||
waitForJobsToDrain(t, p2)
|
||||
if got := p2.processedTsWatermark.Load(); got != 100 {
|
||||
t.Fatalf("watermark = %d after a later success, want it held at 100", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestFailedJobHoldsWatermarkAtOldestFailure verifies that the watermark is
|
||||
// pinned by the oldest failure, not the most recent one.
|
||||
// pinned by the oldest failure, not the most recent one. Once the processor
|
||||
// drains it stops accepting events for the resubscribe, so both failures have
|
||||
// to be in the same drained batch.
|
||||
func TestFailedJobHoldsWatermarkAtOldestFailure(t *testing.T) {
|
||||
release := make(chan struct{})
|
||||
fn := func(resp *filer_pb.SubscribeMetadataResponse) error {
|
||||
<-release
|
||||
if resp.TsNs == 200 || resp.TsNs == 400 {
|
||||
return errors.New("AccessDenied: Access Denied")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
p := NewMetadataProcessor(fn, 1, 0)
|
||||
p := NewMetadataProcessor(fn, 10, 0)
|
||||
|
||||
for _, ts := range []int64{100, 200, 300, 400, 500} {
|
||||
p.AddSyncJob(makeResp("/dir", fmt.Sprintf("f%d.txt", ts), false, ts, true))
|
||||
waitForJobsToDrain(t, p)
|
||||
}
|
||||
close(release)
|
||||
waitForJobsToDrain(t, p)
|
||||
|
||||
if got := p.processedTsWatermark.Load(); got != 100 {
|
||||
t.Fatalf("watermark = %d, want it held at 100 by the failure at 200", got)
|
||||
@@ -692,14 +684,18 @@ func TestFailedJobHoldsWatermarkAtOldestFailure(t *testing.T) {
|
||||
// update's new 60-byte chunk but not its shared one, and nothing for the
|
||||
// delete despite its chunk.
|
||||
func TestSyncStreamMetrics(t *testing.T) {
|
||||
release := make(chan struct{})
|
||||
fn := func(resp *filer_pb.SubscribeMetadataResponse) error {
|
||||
<-release
|
||||
if resp.TsNs == 2 {
|
||||
return errors.New("AccessDenied: Access Denied")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
p := NewMetadataProcessor(fn, 100, 0)
|
||||
p.SetMetrics("srcFiler", "dstFiler", "TestSyncStreamMetrics", "/")
|
||||
// the counters are process-global: a unique client name keeps repeated
|
||||
// runs (-count>1) from accumulating into each other
|
||||
p.SetMetrics("srcFiler", "dstFiler", fmt.Sprintf("TestSyncStreamMetrics-%d", time.Now().UnixNano()), "/")
|
||||
|
||||
create := makeResp("/dir1", "a.txt", false, 1, true)
|
||||
create.EventNotification.NewEntry.Chunks = []*filer_pb.FileChunk{{FileId: "1,a0", Size: 100}}
|
||||
@@ -721,6 +717,7 @@ func TestSyncStreamMetrics(t *testing.T) {
|
||||
for _, resp := range []*filer_pb.SubscribeMetadataResponse{create, failing, update, del} {
|
||||
p.AddSyncJob(resp)
|
||||
}
|
||||
close(release)
|
||||
waitForJobsToDrain(t, p)
|
||||
|
||||
for _, tc := range []struct {
|
||||
@@ -742,3 +739,337 @@ func TestSyncStreamMetrics(t *testing.T) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestFailedJobReplaySuccessClearsPin verifies that when the failed event is
|
||||
// redelivered while the processor is still alive and succeeds this time, the
|
||||
// failure pin clears and the watermark can move again. It is the one event a
|
||||
// stopped processor still runs. The resubscribe still signals once the jobs
|
||||
// drain: anything dropped after the stop has to replay too.
|
||||
func TestFailedJobReplaySuccessClearsPin(t *testing.T) {
|
||||
failed := true
|
||||
release := make(chan struct{})
|
||||
fn := func(resp *filer_pb.SubscribeMetadataResponse) error {
|
||||
if resp.TsNs == 300 {
|
||||
<-release
|
||||
}
|
||||
if resp.TsNs == 200 && failed {
|
||||
failed = false
|
||||
return errors.New("AccessDenied: Access Denied")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
p := NewMetadataProcessor(fn, 10, 0)
|
||||
|
||||
// the slow job is admitted first so the processor has in-flight work when
|
||||
// the failure lands, keeping the drain — and the resubscribe — open
|
||||
p.AddSyncJob(makeResp("/dir", "c.txt", false, 300, true))
|
||||
p.AddSyncJob(makeResp("/dir", "b.txt", false, 200, true))
|
||||
|
||||
deadline := time.Now().Add(10 * time.Second)
|
||||
for p.OldestFailedTsNs() != 200 && time.Now().Before(deadline) {
|
||||
time.Sleep(time.Millisecond)
|
||||
}
|
||||
if got := p.OldestFailedTsNs(); got != 200 {
|
||||
t.Fatalf("oldest failed = %d, want 200", got)
|
||||
}
|
||||
|
||||
// once stopped, a new event drops instead of queueing into a processor
|
||||
// that is about to be abandoned — it replays after the resubscribe
|
||||
p.AddSyncJob(makeResp("/dir", "d.txt", false, 400, true))
|
||||
p.activeJobsLock.Lock()
|
||||
dropped := p.activeJobTs[400] == 0
|
||||
p.activeJobsLock.Unlock()
|
||||
if !dropped {
|
||||
t.Fatal("new event admitted after the failure stopped the processor")
|
||||
}
|
||||
|
||||
// the redelivery is the exception: it runs and its success clears the pin
|
||||
p.AddSyncJob(makeResp("/dir", "b.txt", false, 200, true))
|
||||
deadline = time.Now().Add(10 * time.Second)
|
||||
for p.OldestFailedTsNs() != 0 && time.Now().Before(deadline) {
|
||||
time.Sleep(time.Millisecond)
|
||||
}
|
||||
if got := p.OldestFailedTsNs(); got != 0 {
|
||||
t.Fatalf("oldest failed = %d after a successful replay, want 0", got)
|
||||
}
|
||||
|
||||
close(release)
|
||||
waitForJobsToDrain(t, p)
|
||||
select {
|
||||
case <-p.ResubscribeCh():
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("resubscribe never signaled after the stopped processor drained")
|
||||
}
|
||||
if got := p.processedTsWatermark.Load(); got != 300 {
|
||||
t.Fatalf("watermark = %d after recovery, want 300", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestFilteredMarkerAdvancesWatermark verifies that a filtered-progress marker
|
||||
// (empty event with a timestamp) moves the watermark once all earlier work has
|
||||
// finished, but never past an in-flight job or an unresolved failure.
|
||||
func TestFilteredMarkerAdvancesWatermark(t *testing.T) {
|
||||
marker := func(ts int64) *filer_pb.SubscribeMetadataResponse {
|
||||
return &filer_pb.SubscribeMetadataResponse{TsNs: ts, EventNotification: &filer_pb.EventNotification{}}
|
||||
}
|
||||
|
||||
t.Run("idle", func(t *testing.T) {
|
||||
p := NewMetadataProcessor(func(resp *filer_pb.SubscribeMetadataResponse) error { return nil }, 100, 50)
|
||||
p.AddSyncJob(marker(90))
|
||||
if got := p.processedTsWatermark.Load(); got != 90 {
|
||||
t.Fatalf("watermark = %d after marker, want 90", got)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("behind in-flight job", func(t *testing.T) {
|
||||
release := make(chan struct{})
|
||||
p := NewMetadataProcessor(func(resp *filer_pb.SubscribeMetadataResponse) error {
|
||||
<-release
|
||||
return nil
|
||||
}, 100, 50)
|
||||
p.AddSyncJob(makeResp("/dir", "f.txt", false, 60, true))
|
||||
p.AddSyncJob(marker(80))
|
||||
if got := p.processedTsWatermark.Load(); got != 50 {
|
||||
t.Fatalf("watermark = %d with a job in flight, want 50", got)
|
||||
}
|
||||
close(release)
|
||||
waitForJobsToDrain(t, p)
|
||||
if got := p.processedTsWatermark.Load(); got != 80 {
|
||||
t.Fatalf("watermark = %d after drain, want the retained marker at 80", got)
|
||||
}
|
||||
p.AddSyncJob(marker(90))
|
||||
if got := p.processedTsWatermark.Load(); got != 90 {
|
||||
t.Fatalf("watermark = %d after drain and marker, want 90", got)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("behind a failure", func(t *testing.T) {
|
||||
p := NewMetadataProcessor(func(resp *filer_pb.SubscribeMetadataResponse) error {
|
||||
return errors.New("AccessDenied: Access Denied")
|
||||
}, 100, 50)
|
||||
p.AddSyncJob(makeResp("/dir", "f.txt", false, 100, true))
|
||||
waitForJobsToDrain(t, p)
|
||||
p.AddSyncJob(marker(200))
|
||||
if got := p.processedTsWatermark.Load(); got != 50 {
|
||||
t.Fatalf("watermark = %d past a failure pin, want 50", got)
|
||||
}
|
||||
p.AddSyncJob(marker(70))
|
||||
if got := p.processedTsWatermark.Load(); got != 70 {
|
||||
t.Fatalf("watermark = %d behind the pin, want 70", got)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// TestFailedLedgerCapsAndStaysPinned verifies that a sustained run of distinct
|
||||
// failures cannot grow failedTs without bound: past maxFailedSyncEvents the
|
||||
// ledger collapses to a sticky pin at the oldest failure, so the watermark
|
||||
// still replays from it while memory stays bounded.
|
||||
func TestFailedLedgerCapsAndStaysPinned(t *testing.T) {
|
||||
defer func(old int) { maxFailedSyncEvents = old }(maxFailedSyncEvents)
|
||||
maxFailedSyncEvents = 4
|
||||
|
||||
fail := true
|
||||
release := make(chan struct{})
|
||||
p := NewMetadataProcessor(func(resp *filer_pb.SubscribeMetadataResponse) error {
|
||||
<-release
|
||||
if fail {
|
||||
return errors.New("AccessDenied: Access Denied")
|
||||
}
|
||||
return nil
|
||||
}, 100, 0)
|
||||
for i := int64(1); i <= 10; i++ {
|
||||
p.AddSyncJob(makeResp("/dir", fmt.Sprintf("f%d.txt", i), false, i*100, true))
|
||||
}
|
||||
close(release)
|
||||
waitForJobsToDrain(t, p)
|
||||
|
||||
if !p.failedSticky {
|
||||
t.Fatal("ledger did not collapse past the cap")
|
||||
}
|
||||
if got := p.OldestFailedTsNs(); got != 100 {
|
||||
t.Fatalf("oldest failed = %d, want the pin at the oldest failure 100", got)
|
||||
}
|
||||
if got := p.processedTsWatermark.Load(); got != 0 {
|
||||
t.Fatalf("watermark = %d, want it pinned at 0", got)
|
||||
}
|
||||
|
||||
fail = false
|
||||
p.AddSyncJob(makeResp("/dir", "f1.txt", false, 100, true))
|
||||
waitForJobsToDrain(t, p)
|
||||
if got := p.OldestFailedTsNs(); got != 100 {
|
||||
t.Fatalf("oldest failed = %d after a collapsed replay, want the pin held at 100", got)
|
||||
}
|
||||
if got := p.processedTsWatermark.Load(); got != 0 {
|
||||
t.Fatalf("watermark = %d after a collapsed replay, want it still pinned at 0", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestFailedLedgerDistinguishesEventsAtSameTs verifies that a success for one
|
||||
// event does not clear the pin recorded for a different event that happened to
|
||||
// share its timestamp — the ledger keys on event identity, not just TsNs. A
|
||||
// slow job holds the drain open so the redelivery still lands on this
|
||||
// processor generation.
|
||||
func TestFailedLedgerDistinguishesEventsAtSameTs(t *testing.T) {
|
||||
fail := true
|
||||
release := make(chan struct{})
|
||||
hold := make(chan struct{})
|
||||
p := NewMetadataProcessor(func(resp *filer_pb.SubscribeMetadataResponse) error {
|
||||
if resp.EventNotification.NewEntry.GetName() == "slow.txt" {
|
||||
<-hold
|
||||
} else {
|
||||
<-release
|
||||
}
|
||||
if resp.EventNotification.NewEntry.GetName() == "bad.txt" && fail {
|
||||
return errors.New("AccessDenied: Access Denied")
|
||||
}
|
||||
return nil
|
||||
}, 100, 0)
|
||||
|
||||
// both same-ts events admit before either resolves, so the success lands
|
||||
// while the failure is already pinned
|
||||
p.AddSyncJob(makeResp("/dir", "slow.txt", false, 900, true))
|
||||
p.AddSyncJob(makeResp("/dir", "bad.txt", false, 200, true))
|
||||
p.AddSyncJob(makeResp("/dir", "good.txt", false, 200, true))
|
||||
close(release)
|
||||
|
||||
deadline := time.Now().Add(10 * time.Second)
|
||||
for p.OldestFailedTsNs() != 200 && time.Now().Before(deadline) {
|
||||
time.Sleep(time.Millisecond)
|
||||
}
|
||||
if got := p.OldestFailedTsNs(); got != 200 {
|
||||
t.Fatalf("oldest failed = %d, want the other event's pin held at 200", got)
|
||||
}
|
||||
if got := p.processedTsWatermark.Load(); got != 0 {
|
||||
t.Fatalf("watermark = %d, want it still pinned at 0", got)
|
||||
}
|
||||
|
||||
// redelivering the failed event itself is what clears the pin — and it is
|
||||
// the one event a stopped processor still admits
|
||||
fail = false
|
||||
p.AddSyncJob(makeResp("/dir", "bad.txt", false, 200, true))
|
||||
deadline = time.Now().Add(10 * time.Second)
|
||||
for p.OldestFailedTsNs() != 0 && time.Now().Before(deadline) {
|
||||
time.Sleep(time.Millisecond)
|
||||
}
|
||||
close(hold)
|
||||
waitForJobsToDrain(t, p)
|
||||
if got := p.OldestFailedTsNs(); got != 0 {
|
||||
t.Fatalf("oldest failed = %d after the failed event itself recovered, want 0", got)
|
||||
}
|
||||
if got := p.processedTsWatermark.Load(); got != 900 {
|
||||
t.Fatalf("watermark = %d after the pin cleared and the rest drained, want 900", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestFailedJobSignalsResubscribe verifies that a job exhausting its retries
|
||||
// closes ResubscribeCh so the follower drops the stream and the reconnect
|
||||
// replays the pinned event — the path that used to wait for a restart.
|
||||
func TestFailedJobSignalsResubscribe(t *testing.T) {
|
||||
p := NewMetadataProcessor(func(resp *filer_pb.SubscribeMetadataResponse) error {
|
||||
return errors.New("AccessDenied: Access Denied")
|
||||
}, 100, 0)
|
||||
|
||||
p.AddSyncJob(makeResp("/dir", "a.txt", false, 100, true))
|
||||
waitForJobsToDrain(t, p)
|
||||
|
||||
select {
|
||||
case <-p.ResubscribeCh():
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("resubscribe channel never closed after the failure pinned the watermark")
|
||||
}
|
||||
if got := p.OldestFailedTsNs(); got != 100 {
|
||||
t.Fatalf("oldest failed = %d, want 100", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestResubscribeWaitsForInFlightJobs verifies the signal stays open while
|
||||
// jobs admitted before the failure are still running — replaying behind them
|
||||
// could restore older state over their writes — and closes once they drain.
|
||||
// An event arriving after the stop drops instead of keeping the drain open,
|
||||
// so a busy stream cannot starve the replay.
|
||||
func TestResubscribeWaitsForInFlightJobs(t *testing.T) {
|
||||
release := make(chan struct{})
|
||||
p := NewMetadataProcessor(func(resp *filer_pb.SubscribeMetadataResponse) error {
|
||||
if resp.TsNs == 100 {
|
||||
return errors.New("AccessDenied: Access Denied")
|
||||
}
|
||||
<-release
|
||||
return nil
|
||||
}, 100, 0)
|
||||
|
||||
// the slow job admits first so it is in flight when the failure lands
|
||||
p.AddSyncJob(makeResp("/dir", "b.txt", false, 200, true))
|
||||
p.AddSyncJob(makeResp("/dir", "a.txt", false, 100, true))
|
||||
|
||||
deadline := time.Now().Add(10 * time.Second)
|
||||
for p.OldestFailedTsNs() == 0 && time.Now().Before(deadline) {
|
||||
time.Sleep(time.Millisecond)
|
||||
}
|
||||
if p.OldestFailedTsNs() != 100 {
|
||||
t.Fatalf("oldest failed = %d, want the pin at 100", p.OldestFailedTsNs())
|
||||
}
|
||||
select {
|
||||
case <-p.ResubscribeCh():
|
||||
t.Fatal("resubscribe signaled while an in-flight job could still race the replay")
|
||||
case <-time.After(50 * time.Millisecond):
|
||||
}
|
||||
|
||||
// the processor stopped on the failure, so this event drops — it replays
|
||||
// after the resubscribe — instead of starving the drain
|
||||
p.AddSyncJob(makeResp("/dir", "c.txt", false, 300, true))
|
||||
p.activeJobsLock.Lock()
|
||||
dropped := p.activeJobTs[300] == 0
|
||||
p.activeJobsLock.Unlock()
|
||||
if !dropped {
|
||||
t.Fatal("event admitted after the processor stopped")
|
||||
}
|
||||
|
||||
close(release)
|
||||
waitForJobsToDrain(t, p)
|
||||
select {
|
||||
case <-p.ResubscribeCh():
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("resubscribe channel never closed after the in-flight jobs drained")
|
||||
}
|
||||
}
|
||||
|
||||
// TestResubscribeWaitsForSameTsSibling guards the per-timestamp job
|
||||
// counting: events in one batch can share a TsNs, and the failed job must
|
||||
// not free the bookkeeping of a sibling still running at that timestamp.
|
||||
// Otherwise its completion could report the processor drained and the
|
||||
// resubscribe would replay over the sibling's writes.
|
||||
func TestResubscribeWaitsForSameTsSibling(t *testing.T) {
|
||||
release := make(chan struct{})
|
||||
p := NewMetadataProcessor(func(resp *filer_pb.SubscribeMetadataResponse) error {
|
||||
if resp.EventNotification.NewEntry.GetName() == "bad.txt" {
|
||||
return errors.New("AccessDenied: Access Denied")
|
||||
}
|
||||
<-release
|
||||
return nil
|
||||
}, 100, 0)
|
||||
|
||||
// the slow job admits first so it is in flight when the same-ts failure lands
|
||||
p.AddSyncJob(makeResp("/dir", "slow.txt", false, 200, true))
|
||||
p.AddSyncJob(makeResp("/dir", "bad.txt", false, 200, true))
|
||||
|
||||
deadline := time.Now().Add(10 * time.Second)
|
||||
for p.OldestFailedTsNs() == 0 && time.Now().Before(deadline) {
|
||||
time.Sleep(time.Millisecond)
|
||||
}
|
||||
if p.OldestFailedTsNs() != 200 {
|
||||
t.Fatalf("oldest failed = %d, want the pin at 200", p.OldestFailedTsNs())
|
||||
}
|
||||
select {
|
||||
case <-p.ResubscribeCh():
|
||||
t.Fatal("resubscribe signaled while a same-ts job was still running")
|
||||
case <-time.After(50 * time.Millisecond):
|
||||
}
|
||||
|
||||
close(release)
|
||||
waitForJobsToDrain(t, p)
|
||||
select {
|
||||
case <-p.ResubscribeCh():
|
||||
case <-time.After(time.Second):
|
||||
t.Fatal("resubscribe channel never closed after the same-ts jobs drained")
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,85 @@
|
||||
package command
|
||||
|
||||
import (
|
||||
"net"
|
||||
"net/http"
|
||||
"os"
|
||||
"strconv"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/images/gateway"
|
||||
)
|
||||
|
||||
var cmdImage = &Command{
|
||||
UsageLine: "image -source=http://localhost:8333/public-bucket -imgproxy=http://localhost:8080",
|
||||
Short: "Start an on-demand gateway for public images",
|
||||
Long: `Run an optional image gateway in front of public S3 objects. x-oss-process
|
||||
supports aspect-preserving downscaling, absolute quality,Q_85, and JPEG/PNG/WebP.
|
||||
Source access is checked before every cache read. Processed images remain in a
|
||||
bounded memory cache and are never written to S3 or the Filer.
|
||||
Encoding runs in a separate imgproxy service; configure its pixel, file size,
|
||||
and process resource limits. This public endpoint does not accept S3 signatures
|
||||
or read private objects.`,
|
||||
}
|
||||
|
||||
var imageOptions struct {
|
||||
source, imgproxy, bind *string
|
||||
port, concurrency, maxDimension *int
|
||||
cacheMB, sourceMB, resultMB *int64
|
||||
timeout *time.Duration
|
||||
}
|
||||
|
||||
// init registers the optional endpoint without changing S3, Filer, or Volume services.
|
||||
func init() {
|
||||
cmdImage.Run = runImage
|
||||
imageOptions.source = cmdImage.Flag.String("source", "", "Fixed anonymous S3 HTTP(S) source URL, optionally with a bucket path")
|
||||
imageOptions.imgproxy = cmdImage.Flag.String("imgproxy", "", "Separate imgproxy HTTP(S) service URL")
|
||||
imageOptions.bind = cmdImage.Flag.String("ip.bind", "127.0.0.1", "Listen address")
|
||||
imageOptions.port = cmdImage.Flag.Int("port", 8334, "HTTP port")
|
||||
imageOptions.concurrency = cmdImage.Flag.Int("concurrency", 8, "Maximum concurrent requests and encoding jobs; excess requests return 429")
|
||||
imageOptions.maxDimension = cmdImage.Flag.Int("maxDimension", 4096, "Maximum output dimension")
|
||||
imageOptions.cacheMB = cmdImage.Flag.Int64("cacheCapacityMB", 64, "Processed image memory cache in MiB; 0 disables caching")
|
||||
imageOptions.sourceMB = cmdImage.Flag.Int64("maxSourceMB", 25, "Maximum source image size in MiB")
|
||||
imageOptions.resultMB = cmdImage.Flag.Int64("maxResultMB", 10, "Maximum processed image size in MiB")
|
||||
imageOptions.timeout = cmdImage.Flag.Duration("timeout", 15*time.Second, "Separate timeout budgets for source metadata, shared encoding, and client writes")
|
||||
}
|
||||
|
||||
// runImage starts the gateway, reading signing material from the environment rather than process arguments.
|
||||
func runImage(cmd *Command, args []string) bool {
|
||||
if *imageOptions.cacheMB < 0 || *imageOptions.cacheMB > 1<<20 ||
|
||||
*imageOptions.sourceMB < 1 || *imageOptions.sourceMB > 1024 ||
|
||||
*imageOptions.resultMB < 1 || *imageOptions.resultMB > 1024 {
|
||||
glog.Errorf("Invalid image gateway capacity options")
|
||||
return false
|
||||
}
|
||||
handler, err := gateway.New(gateway.Config{
|
||||
Source: *imageOptions.source, Imgproxy: *imageOptions.imgproxy,
|
||||
Key: os.Getenv("IMGPROXY_KEY"), Salt: os.Getenv("IMGPROXY_SALT"),
|
||||
Concurrency: *imageOptions.concurrency, MaxDimension: *imageOptions.maxDimension,
|
||||
CacheBytes: *imageOptions.cacheMB << 20, MaxSourceBytes: *imageOptions.sourceMB << 20,
|
||||
MaxResultBytes: *imageOptions.resultMB << 20, Timeout: *imageOptions.timeout,
|
||||
})
|
||||
if err != nil {
|
||||
glog.Errorf("Invalid image gateway configuration: %v", err)
|
||||
return false
|
||||
}
|
||||
if *imageOptions.port < 1 || *imageOptions.port > 65535 {
|
||||
glog.Errorf("Invalid image gateway port")
|
||||
return false
|
||||
}
|
||||
// Bound the initial source check, shared encoding, and response write phases.
|
||||
// ReadTimeout also limits draining of ignored request bodies after the handler returns.
|
||||
server := &http.Server{
|
||||
Addr: net.JoinHostPort(*imageOptions.bind, strconv.Itoa(*imageOptions.port)), Handler: handler,
|
||||
ReadHeaderTimeout: 5 * time.Second, ReadTimeout: 10 * time.Second,
|
||||
WriteTimeout: 3 * *imageOptions.timeout,
|
||||
IdleTimeout: 60 * time.Second, MaxHeaderBytes: 16 << 10,
|
||||
}
|
||||
glog.V(0).Infof("Image gateway listening on %s", server.Addr)
|
||||
if err = server.ListenAndServe(); err != nil && err != http.ErrServerClosed {
|
||||
glog.Errorf("Image gateway failed: %v", err)
|
||||
return false
|
||||
}
|
||||
return true
|
||||
}
|
||||
+107
-16
@@ -69,6 +69,7 @@ var (
|
||||
miniEnableWebDAV *bool
|
||||
miniEnableS3 *bool
|
||||
miniEnableAdminUI *bool
|
||||
miniAdminWorkerBindIP *string
|
||||
miniS3IamReadOnly *bool
|
||||
miniVolumeMaxDataVolumeCounts *string
|
||||
// MiniClusterCtx is the context for the mini cluster. If set, the mini cluster will stop when the context is cancelled.
|
||||
@@ -567,6 +568,8 @@ func initMiniAdminFlags() {
|
||||
miniAdminOptions.adminPassword = cmdMini.Flag.String("admin.password", "", "admin interface password (if empty, auth is disabled)")
|
||||
miniAdminOptions.readOnlyUser = cmdMini.Flag.String("admin.readOnlyUser", "", "read-only user username (optional, for view-only access)")
|
||||
miniAdminOptions.readOnlyPassword = cmdMini.Flag.String("admin.readOnlyPassword", "", "read-only user password (optional, for view-only access; requires admin.password to be set)")
|
||||
miniAdminOptions.allowInsecureBind = cmdMini.Flag.Bool("admin.allowInsecureBind", false, "INSECURE: allow the unauthenticated admin interface to bind to -ip.bind instead of loopback")
|
||||
miniAdminWorkerBindIP = cmdMini.Flag.String("admin.worker.ip", "127.0.0.1", "worker gRPC bind address; expose only with grpc.admin mTLS configured")
|
||||
miniAdminOptions.urlPrefix = cmdMini.Flag.String("admin.urlPrefix", "", "URL path prefix when running the admin UI behind a reverse proxy under a subdirectory (e.g. /seaweedfs)")
|
||||
}
|
||||
|
||||
@@ -957,7 +960,11 @@ func ensureAllPortsAvailableOnIP(bindIp string) error {
|
||||
// first: an in-process rerun would otherwise inherit the closed listener
|
||||
// of the previous run and only find out inside Serve.
|
||||
miniAdminOptions.workerGrpcListener = nil
|
||||
if listener, err := net.Listen("tcp", util.JoinHostPort(bindIp, *miniAdminOptions.grpcPort)); err != nil {
|
||||
workerGrpcBindIP := miniAdminOptions.workerGrpcBindIp
|
||||
if workerGrpcBindIP == "" {
|
||||
workerGrpcBindIP = "127.0.0.1"
|
||||
}
|
||||
if listener, err := listenMiniAdminWorker(workerGrpcBindIP, *miniAdminOptions.grpcPort); err != nil {
|
||||
glog.Warningf("Could not reserve Admin gRPC port %d: %v", *miniAdminOptions.grpcPort, err)
|
||||
} else {
|
||||
miniAdminOptions.workerGrpcListener = listener
|
||||
@@ -984,6 +991,12 @@ func ensureAllPortsAvailableOnIP(bindIp string) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// listenMiniAdminWorker reserves the exact worker gRPC address that the Admin
|
||||
// server will later use, preventing another connection from taking the port.
|
||||
func listenMiniAdminWorker(bindIP string, port int) (net.Listener, error) {
|
||||
return net.Listen("tcp", util.JoinHostPort(bindIP, port))
|
||||
}
|
||||
|
||||
// initializeGrpcPortsOnIP initializes all gRPC ports based on their HTTP ports on a specific IP
|
||||
// If a gRPC port is 0, it will be set to httpPort + GrpcPortOffset
|
||||
// This must be called after HTTP ports are finalized and before services start
|
||||
@@ -1239,6 +1252,10 @@ func runMini(cmd *Command, args []string) bool {
|
||||
|
||||
// Determine bind IP
|
||||
bindIp := getBindIp()
|
||||
miniAdminOptions.workerGrpcBindIp = *miniAdminWorkerBindIP
|
||||
if miniAdminOptions.workerGrpcBindIp == "" {
|
||||
miniAdminOptions.workerGrpcBindIp = "127.0.0.1"
|
||||
}
|
||||
|
||||
// Ensure all ports are available, find alternatives if needed
|
||||
if err := ensureAllPortsAvailableOnIP(bindIp); err != nil {
|
||||
@@ -1583,6 +1600,46 @@ func applyMiniAdminCredentialFallback(options *AdminOptions) {
|
||||
applyViperFallback(cmdMini, options.readOnlyPassword, "admin.readOnlyPassword", "admin.readonly.password")
|
||||
}
|
||||
|
||||
// miniAdminBindIP selects a loopback HTTP bind unless Admin authentication,
|
||||
// mTLS, or an explicit insecure opt-out permits the requested address.
|
||||
func miniAdminBindIP(requestedIP string, passwordConfigured, mtlsConfigured, allowInsecure bool) string {
|
||||
if isLoopbackIp(requestedIP) || passwordConfigured || mtlsConfigured || allowInsecure {
|
||||
return requestedIP
|
||||
}
|
||||
return "127.0.0.1"
|
||||
}
|
||||
|
||||
// miniAdminWorkerAddress encodes the finalized HTTP and gRPC ports in the
|
||||
// server-address format understood by pb.ServerToGrpcAddress.
|
||||
func miniAdminWorkerAddress(bindIP string, httpPort, grpcPort int) string {
|
||||
return fmt.Sprintf("%s:%d.%d", miniAdminWorkerDialIP(bindIP), httpPort, grpcPort)
|
||||
}
|
||||
|
||||
// miniAdminWorkerDialIP maps unspecified listener addresses to same-family
|
||||
// loopback destinations while preserving specific addresses for local dials.
|
||||
func miniAdminWorkerDialIP(bindIP string) string {
|
||||
ip := net.ParseIP(strings.TrimSuffix(strings.TrimPrefix(bindIP, "["), "]"))
|
||||
if ip == nil || !ip.IsUnspecified() {
|
||||
return bindIP
|
||||
}
|
||||
if ip.To4() != nil {
|
||||
return "127.0.0.1"
|
||||
}
|
||||
return "::1"
|
||||
}
|
||||
|
||||
// miniAdminAdvertisedIP returns a reachable address for URLs in the welcome
|
||||
// message while preserving a specifically selected Admin HTTP bind address.
|
||||
func miniAdminAdvertisedIP() string {
|
||||
if miniAdminOptions.ip == nil || *miniAdminOptions.ip == "" {
|
||||
return *miniIp
|
||||
}
|
||||
if *miniAdminOptions.ip == "0.0.0.0" || *miniAdminOptions.ip == "::" {
|
||||
return *miniIp
|
||||
}
|
||||
return *miniAdminOptions.ip
|
||||
}
|
||||
|
||||
// startMiniAdminWithWorker starts the admin server with one worker
|
||||
func startMiniAdminWithWorker(allServicesReady chan struct{}) {
|
||||
defer close(allServicesReady) // Ensure channel is always closed on all paths
|
||||
@@ -1590,9 +1647,6 @@ func startMiniAdminWithWorker(allServicesReady chan struct{}) {
|
||||
// Admin shuts down when mini clients shutdown is triggered.
|
||||
ctx := miniClientsCtx()
|
||||
|
||||
// Determine bind IP for health checks
|
||||
bindIp := getBindIp()
|
||||
|
||||
// Prepare master address with gRPC port
|
||||
masterAddr := string(pb.NewServerAddress(*miniIp, *miniMasterOptions.port, *miniMasterOptions.portGrpc))
|
||||
|
||||
@@ -1611,6 +1665,34 @@ func startMiniAdminWithWorker(allServicesReady chan struct{}) {
|
||||
// vars, matching the standalone `weed admin` command.
|
||||
applyMiniAdminCredentialFallback(&miniAdminOptions)
|
||||
|
||||
requestedBindIP := getBindIp()
|
||||
hasMTLS := util.GetViper().GetString("https.admin.key") != "" &&
|
||||
util.GetViper().GetString("https.admin.ca") != ""
|
||||
allowInsecure := miniAdminOptions.allowInsecureBind != nil &&
|
||||
*miniAdminOptions.allowInsecureBind
|
||||
bindIP := miniAdminBindIP(
|
||||
requestedBindIP,
|
||||
*miniAdminOptions.adminPassword != "",
|
||||
hasMTLS,
|
||||
allowInsecure,
|
||||
)
|
||||
miniAdminOptions.ip = &bindIP
|
||||
if bindIP != requestedBindIP {
|
||||
glog.Warningf(
|
||||
"Admin authentication is disabled; binding the admin interface to %s instead of %s. "+
|
||||
"Set -admin.password, configure https.admin mTLS, or use "+
|
||||
"-admin.allowInsecureBind to retain the insecure network bind.",
|
||||
bindIP,
|
||||
requestedBindIP,
|
||||
)
|
||||
} else if allowInsecure && !isLoopbackIp(bindIP) &&
|
||||
*miniAdminOptions.adminPassword == "" && !hasMTLS {
|
||||
glog.Warningf(
|
||||
"-admin.allowInsecureBind exposes the unauthenticated admin interface on %s.",
|
||||
bindIP,
|
||||
)
|
||||
}
|
||||
|
||||
// Security validation: prevent empty username when password is set
|
||||
if *miniAdminOptions.adminPassword != "" && *miniAdminOptions.adminUser == "" {
|
||||
glog.Fatalf("Error: -admin.user cannot be empty when -admin.password is set")
|
||||
@@ -1681,7 +1763,7 @@ func startMiniAdminWithWorker(allServicesReady chan struct{}) {
|
||||
}()
|
||||
|
||||
// Wait for admin server's HTTP port to be ready before launching worker
|
||||
adminAddr := "http://" + util.JoinHostPort(bindIp, *miniAdminOptions.port)
|
||||
adminAddr := "http://" + util.JoinHostPort(bindIP, *miniAdminOptions.port)
|
||||
if err := waitForAdminServerReady(ctx, adminAddr); err != nil {
|
||||
// If the parent context was cancelled (e.g. a previous in-process
|
||||
// mini run is being torn down), bail out gracefully instead of
|
||||
@@ -1705,7 +1787,10 @@ func startMiniAdminWithWorker(allServicesReady chan struct{}) {
|
||||
startMiniPluginWorker(ctx, workerDir)
|
||||
|
||||
// Wait for worker to be ready by polling its gRPC port
|
||||
workerGrpcAddr := fmt.Sprintf("%s:%d", bindIp, *miniAdminOptions.grpcPort)
|
||||
workerGrpcAddr := util.JoinHostPort(
|
||||
miniAdminWorkerDialIP(miniAdminOptions.workerGrpcBindIp),
|
||||
*miniAdminOptions.grpcPort,
|
||||
)
|
||||
waitForWorkerReady(workerGrpcAddr)
|
||||
if miniProgressBoard != nil {
|
||||
miniProgressBoard.ready("Admin")
|
||||
@@ -1783,7 +1868,11 @@ func waitForWorkerReady(workerGrpcAddr string) {
|
||||
func startMiniWorker(workerDir string) {
|
||||
glog.V(1).Infof("Initializing standard worker runtime")
|
||||
|
||||
adminAddr := fmt.Sprintf("%s:%d", *miniIp, *miniAdminOptions.port)
|
||||
adminAddr := miniAdminWorkerAddress(
|
||||
miniAdminOptions.workerGrpcBindIp,
|
||||
*miniAdminOptions.port,
|
||||
*miniAdminOptions.grpcPort,
|
||||
)
|
||||
capabilities := "vacuum,ec,balance"
|
||||
|
||||
// Use common worker directory
|
||||
@@ -1858,11 +1947,11 @@ func startMiniWorker(workerDir string) {
|
||||
func startMiniPluginWorker(ctx context.Context, workerDir string) {
|
||||
glog.V(1).Infof("Starting plugin worker for admin server")
|
||||
|
||||
adminAddr := fmt.Sprintf("%s:%d", *miniIp, *miniAdminOptions.port)
|
||||
resolvedAdminAddr := resolvePluginWorkerAdminServer(adminAddr)
|
||||
if resolvedAdminAddr != adminAddr {
|
||||
glog.V(1).Infof("Resolved mini plugin worker admin endpoint: %s -> %s", adminAddr, resolvedAdminAddr)
|
||||
}
|
||||
adminAddr := miniAdminWorkerAddress(
|
||||
miniAdminOptions.workerGrpcBindIp,
|
||||
*miniAdminOptions.port,
|
||||
*miniAdminOptions.grpcPort,
|
||||
)
|
||||
|
||||
// Use common worker directory
|
||||
|
||||
@@ -1880,7 +1969,7 @@ func startMiniPluginWorker(ctx context.Context, workerDir string) {
|
||||
}
|
||||
|
||||
pluginRuntime, err := pluginworker.NewWorker(pluginworker.WorkerOptions{
|
||||
AdminServer: resolvedAdminAddr,
|
||||
AdminServer: adminAddr,
|
||||
WorkerID: workerID,
|
||||
WorkerVersion: version.Version(),
|
||||
WorkerAddress: *miniIp,
|
||||
@@ -1919,13 +2008,15 @@ const credentialsInstructionTemplate = `
|
||||
Creates initial credentials for the 'mini' user and pre-creates the bucket.
|
||||
|
||||
Option 2: Use the Admin UI
|
||||
Open: http://%s:%d
|
||||
Open: http://%s
|
||||
Add a new identity to create S3 credentials.
|
||||
`
|
||||
|
||||
// printWelcomeMessage prints the welcome message after all services are running
|
||||
func printWelcomeMessage() {
|
||||
var sb strings.Builder
|
||||
adminIP := miniAdminAdvertisedIP()
|
||||
adminHTTPAddress := util.JoinHostPort(adminIP, *miniAdminOptions.port)
|
||||
|
||||
sb.WriteString("╔═══════════════════════════════════════════════════════════════════════════════╗\n")
|
||||
sb.WriteString("║ SeaweedFS Mini - All-in-One Mode ║\n")
|
||||
@@ -1947,7 +2038,7 @@ func printWelcomeMessage() {
|
||||
}
|
||||
}
|
||||
if *miniEnableAdminUI {
|
||||
fmt.Fprintf(&sb, " Admin UI: http://%s:%d\n", *miniIp, *miniAdminOptions.port)
|
||||
fmt.Fprintf(&sb, " Admin UI: http://%s\n", adminHTTPAddress)
|
||||
}
|
||||
|
||||
fmt.Fprintf(&sb, "\n Data Directory: %s\n", *miniDataFolders)
|
||||
@@ -1970,7 +2061,7 @@ func printWelcomeMessage() {
|
||||
// run, configured via env vars, static config file, etc.) — no need
|
||||
// to show setup hints.
|
||||
case *miniEnableAdminUI:
|
||||
fmt.Fprintf(&sb, credentialsInstructionTemplate, *miniIp, *miniAdminOptions.port)
|
||||
fmt.Fprintf(&sb, credentialsInstructionTemplate, adminHTTPAddress)
|
||||
default:
|
||||
sb.WriteString("\n To create S3 credentials, use environment variables:\n\n")
|
||||
sb.WriteString(" export AWS_ACCESS_KEY_ID=your-access-key\n")
|
||||
|
||||
@@ -1,12 +1,19 @@
|
||||
package command
|
||||
|
||||
import "testing"
|
||||
import (
|
||||
"net"
|
||||
"testing"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
)
|
||||
|
||||
// weed mini must resolve admin credentials from security.toml [admin] /
|
||||
// WEED_ADMIN_* env vars the same way the standalone `weed admin` command does.
|
||||
// This exercises the production fallback so the flag-name -> viper-key mapping
|
||||
// stays correct, in particular the read-only keys where the mini flag
|
||||
// (admin.readOnlyUser) and viper key (admin.readonly.user) differ.
|
||||
// TestApplyMiniAdminCredentialFallbackFromEnv verifies those environment
|
||||
// fallbacks without starting the full mini cluster.
|
||||
func TestApplyMiniAdminCredentialFallbackFromEnv(t *testing.T) {
|
||||
adminUser, adminPassword, readOnlyUser, readOnlyPassword := "admin", "", "", ""
|
||||
options := &AdminOptions{
|
||||
@@ -39,3 +46,211 @@ func TestApplyMiniAdminCredentialFallbackFromEnv(t *testing.T) {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestMiniAdminBindIP covers the authentication-dependent HTTP bind policy.
|
||||
func TestMiniAdminBindIP(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
requestedIP string
|
||||
passwordConfigured bool
|
||||
mtlsConfigured bool
|
||||
allowInsecure bool
|
||||
want string
|
||||
}{
|
||||
{
|
||||
name: "unauthenticated wildcard binds to loopback",
|
||||
requestedIP: "0.0.0.0",
|
||||
want: "127.0.0.1",
|
||||
},
|
||||
{
|
||||
name: "unauthenticated IPv6 wildcard binds to loopback",
|
||||
requestedIP: "::",
|
||||
want: "127.0.0.1",
|
||||
},
|
||||
{
|
||||
name: "existing loopback bind is preserved",
|
||||
requestedIP: "127.0.0.1",
|
||||
want: "127.0.0.1",
|
||||
},
|
||||
{
|
||||
name: "password permits requested bind",
|
||||
requestedIP: "0.0.0.0",
|
||||
passwordConfigured: true,
|
||||
want: "0.0.0.0",
|
||||
},
|
||||
{
|
||||
name: "mTLS permits requested bind",
|
||||
requestedIP: "0.0.0.0",
|
||||
mtlsConfigured: true,
|
||||
want: "0.0.0.0",
|
||||
},
|
||||
{
|
||||
name: "explicit insecure opt-out permits requested bind",
|
||||
requestedIP: "0.0.0.0",
|
||||
allowInsecure: true,
|
||||
want: "0.0.0.0",
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
got := miniAdminBindIP(
|
||||
tt.requestedIP,
|
||||
tt.passwordConfigured,
|
||||
tt.mtlsConfigured,
|
||||
tt.allowInsecure,
|
||||
)
|
||||
if got != tt.want {
|
||||
t.Fatalf("miniAdminBindIP() = %q, want %q", got, tt.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestMiniAdminWorkerBindDefaultsToLoopback protects the worker control-plane default.
|
||||
func TestMiniAdminWorkerBindDefaultsToLoopback(t *testing.T) {
|
||||
if miniAdminWorkerBindIP == nil {
|
||||
t.Fatal("mini Admin worker bind flag is not initialized")
|
||||
}
|
||||
if got, want := *miniAdminWorkerBindIP, "127.0.0.1"; got != want {
|
||||
t.Fatalf("default mini Admin worker bind IP = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
// TestListenMiniAdminWorkerUsesRequestedAddress verifies the production
|
||||
// listener is actually restricted to the configured loopback address.
|
||||
func TestListenMiniAdminWorkerUsesRequestedAddress(t *testing.T) {
|
||||
listener, err := listenMiniAdminWorker("127.0.0.1", 0)
|
||||
if err != nil {
|
||||
t.Fatalf("listen for mini Admin worker: %v", err)
|
||||
}
|
||||
t.Cleanup(func() {
|
||||
_ = listener.Close()
|
||||
})
|
||||
|
||||
address, ok := listener.Addr().(*net.TCPAddr)
|
||||
if !ok {
|
||||
t.Fatalf("listener address type = %T, want *net.TCPAddr", listener.Addr())
|
||||
}
|
||||
if !address.IP.IsLoopback() {
|
||||
t.Fatalf("listener IP = %s, want loopback", address.IP)
|
||||
}
|
||||
}
|
||||
|
||||
// TestMiniAdminWorkerAddressUsesFinalGrpcPort verifies that local workers do
|
||||
// not have to discover or infer a custom Admin worker gRPC port.
|
||||
func TestMiniAdminWorkerAddressUsesFinalGrpcPort(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
ip string
|
||||
want string
|
||||
wantGrpc string
|
||||
}{
|
||||
{
|
||||
name: "IPv4",
|
||||
ip: "127.0.0.1",
|
||||
want: "127.0.0.1:23646.34567",
|
||||
wantGrpc: "127.0.0.1:34567",
|
||||
},
|
||||
{
|
||||
name: "IPv6",
|
||||
ip: "::1",
|
||||
want: "::1:23646.34567",
|
||||
wantGrpc: "[::1]:34567",
|
||||
},
|
||||
{
|
||||
name: "IPv4 wildcard dials loopback",
|
||||
ip: "0.0.0.0",
|
||||
want: "127.0.0.1:23646.34567",
|
||||
wantGrpc: "127.0.0.1:34567",
|
||||
},
|
||||
{
|
||||
name: "IPv6 wildcard dials loopback",
|
||||
ip: "::",
|
||||
want: "::1:23646.34567",
|
||||
wantGrpc: "[::1]:34567",
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
address := miniAdminWorkerAddress(tt.ip, 23646, 34567)
|
||||
if address != tt.want {
|
||||
t.Fatalf("miniAdminWorkerAddress() = %q, want %q", address, tt.want)
|
||||
}
|
||||
if got := pb.ServerToGrpcAddress(address); got != tt.wantGrpc {
|
||||
t.Fatalf("pb.ServerToGrpcAddress() = %q, want %q", got, tt.wantGrpc)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestMiniAdminWorkerDialIP verifies wildcard listener addresses are converted
|
||||
// into valid same-family destinations without rewriting specific addresses.
|
||||
func TestMiniAdminWorkerDialIP(t *testing.T) {
|
||||
tests := []struct {
|
||||
bindIP string
|
||||
want string
|
||||
}{
|
||||
{bindIP: "0.0.0.0", want: "127.0.0.1"},
|
||||
{bindIP: "::", want: "::1"},
|
||||
{bindIP: "[::]", want: "::1"},
|
||||
{bindIP: "192.0.2.10", want: "192.0.2.10"},
|
||||
{bindIP: "2001:db8::10", want: "2001:db8::10"},
|
||||
{bindIP: "worker.internal", want: "worker.internal"},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.bindIP, func(t *testing.T) {
|
||||
if got := miniAdminWorkerDialIP(tt.bindIP); got != tt.want {
|
||||
t.Fatalf("miniAdminWorkerDialIP(%q) = %q, want %q", tt.bindIP, got, tt.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestMiniAdminAdvertisedIP verifies that the welcome message uses the
|
||||
// selected loopback address but replaces wildcard binds with a reachable host.
|
||||
func TestMiniAdminAdvertisedIP(t *testing.T) {
|
||||
oldMiniIP := miniIp
|
||||
oldAdminIP := miniAdminOptions.ip
|
||||
t.Cleanup(func() {
|
||||
miniIp = oldMiniIP
|
||||
miniAdminOptions.ip = oldAdminIP
|
||||
})
|
||||
|
||||
detectedIP := "192.0.2.10"
|
||||
miniIp = &detectedIP
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
adminIP string
|
||||
expected string
|
||||
}{
|
||||
{
|
||||
name: "selected loopback",
|
||||
adminIP: "127.0.0.1",
|
||||
expected: "127.0.0.1",
|
||||
},
|
||||
{
|
||||
name: "IPv4 wildcard",
|
||||
adminIP: "0.0.0.0",
|
||||
expected: detectedIP,
|
||||
},
|
||||
{
|
||||
name: "IPv6 wildcard",
|
||||
adminIP: "::",
|
||||
expected: detectedIP,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
adminIP := tt.adminIP
|
||||
miniAdminOptions.ip = &adminIP
|
||||
if got := miniAdminAdvertisedIP(); got != tt.expected {
|
||||
t.Fatalf("miniAdminAdvertisedIP() = %q, want %q", got, tt.expected)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
+61
-51
@@ -83,10 +83,9 @@ type S3Options struct {
|
||||
readerCacheSizeMB *int64
|
||||
|
||||
allowUntrustedRemoteEndpoints *bool
|
||||
// shutdownCtx, when non-nil, tells startS3Server/startIcebergServer to
|
||||
// gracefully shut down their HTTP/gRPC servers once the ctx is cancelled.
|
||||
// Used by weed mini to orchestrate an ordered shutdown; nil for standalone
|
||||
// weed s3.
|
||||
// shutdownCtx, when non-nil, tells startS3Server to gracefully shut down
|
||||
// its HTTP/gRPC servers once the ctx is cancelled, in addition to on
|
||||
// interrupt. Used by weed mini to orchestrate an ordered shutdown.
|
||||
shutdownCtx context.Context
|
||||
}
|
||||
|
||||
@@ -395,16 +394,17 @@ func (s3opt *S3Options) startS3Server() bool {
|
||||
if s3ApiServer_err != nil {
|
||||
glog.Fatalf("S3 API Server startup error: %v", s3ApiServer_err)
|
||||
}
|
||||
defer s3ApiServer.Shutdown()
|
||||
|
||||
var httpServers []*http.Server
|
||||
|
||||
// Start Iceberg REST Catalog server if enabled
|
||||
if *s3opt.portIceberg > 0 {
|
||||
go s3opt.startIcebergServer(s3ApiServer)
|
||||
httpServers = append(httpServers, s3opt.startIcebergServer(s3ApiServer))
|
||||
}
|
||||
|
||||
// Start Lance Namespace server if enabled
|
||||
if s3opt.portLance != nil && *s3opt.portLance > 0 {
|
||||
go s3opt.startLanceServer(s3ApiServer)
|
||||
httpServers = append(httpServers, s3opt.startLanceServer(s3ApiServer))
|
||||
}
|
||||
|
||||
if runtime.GOOS != "windows" {
|
||||
@@ -415,13 +415,14 @@ func (s3opt *S3Options) startS3Server() bool {
|
||||
if err := os.Remove(localSocket); err != nil && !os.IsNotExist(err) {
|
||||
glog.Fatalf("Failed to remove %s, error: %s", localSocket, err.Error())
|
||||
}
|
||||
s3SocketListener, err := net.Listen("unix", localSocket)
|
||||
if err != nil {
|
||||
glog.Fatalf("Failed to listen on %s: %v", localSocket, err)
|
||||
}
|
||||
socketServer := newHttpServer(router, nil)
|
||||
httpServers = append(httpServers, socketServer)
|
||||
go func() {
|
||||
// start on local unix socket
|
||||
s3SocketListener, err := net.Listen("unix", localSocket)
|
||||
if err != nil {
|
||||
glog.Fatalf("Failed to listen on %s: %v", localSocket, err)
|
||||
}
|
||||
if err := newHttpServer(router, nil).Serve(s3SocketListener); err != nil && err != http.ErrServerClosed {
|
||||
if err := socketServer.Serve(s3SocketListener); err != nil && err != http.ErrServerClosed {
|
||||
glog.Fatalf("Failed to start S3 http server: %v", err)
|
||||
}
|
||||
}()
|
||||
@@ -456,6 +457,7 @@ func (s3opt *S3Options) startS3Server() bool {
|
||||
}
|
||||
go grpcS.Serve(grpcL)
|
||||
pb.ServeGrpcOnLocalSocket(grpcS, grpcPort)
|
||||
stopGrpcServer := func() { gracefulStopGrpc(grpcS, 15*time.Second) }
|
||||
|
||||
if *s3opt.tlsPrivateKey != "" {
|
||||
// Check for port conflict when both HTTP and HTTPS are enabled on the same port
|
||||
@@ -496,23 +498,22 @@ func (s3opt *S3Options) startS3Server() bool {
|
||||
if *s3opt.portHttps == 0 {
|
||||
glog.V(0).Infof("Start Seaweed S3 API Server %s at https port %d", version.Version(), *s3opt.port)
|
||||
if s3ApiLocalListener != nil {
|
||||
localServer := newHttpServer(router, tlsConfig)
|
||||
httpServers = append(httpServers, localServer)
|
||||
go func() {
|
||||
if err = newHttpServer(router, tlsConfig).ServeTLS(s3ApiLocalListener, "", ""); err != nil {
|
||||
if err := localServer.ServeTLS(s3ApiLocalListener, "", ""); err != nil && err != http.ErrServerClosed {
|
||||
glog.Fatalf("S3 API Server Fail to serve: %v", err)
|
||||
}
|
||||
}()
|
||||
}
|
||||
httpS := newHttpServer(router, tlsConfig)
|
||||
if s3opt.shutdownCtx != nil {
|
||||
go func() {
|
||||
<-s3opt.shutdownCtx.Done()
|
||||
httpS.Shutdown(context.Background())
|
||||
grpcS.Stop()
|
||||
}()
|
||||
}
|
||||
httpServers = append(httpServers, httpS)
|
||||
shutdown := s3opt.newShutdown(stopGrpcServer, s3ApiServer, httpServers)
|
||||
if err = httpS.ServeTLS(s3ApiListener, "", ""); err != nil && err != http.ErrServerClosed {
|
||||
glog.Fatalf("S3 API Server Fail to serve: %v", err)
|
||||
}
|
||||
// Serve returns when listeners close, before active requests finish.
|
||||
shutdown()
|
||||
} else {
|
||||
glog.V(0).Infof("Start Seaweed S3 API Server %s at https port %d", version.Version(), *s3opt.portHttps)
|
||||
s3ApiListenerHttps, s3ApiLocalListenerHttps, err := util.NewIpAndLocalListeners(
|
||||
@@ -521,14 +522,18 @@ func (s3opt *S3Options) startS3Server() bool {
|
||||
glog.Fatalf("S3 API HTTPS listener on %s:%d error: %v", *s3opt.bindIp, *s3opt.portHttps, err)
|
||||
}
|
||||
if s3ApiLocalListenerHttps != nil {
|
||||
localHttpsServer := newHttpServer(router, tlsConfig)
|
||||
httpServers = append(httpServers, localHttpsServer)
|
||||
go func() {
|
||||
if err = newHttpServer(router, tlsConfig).ServeTLS(s3ApiLocalListenerHttps, "", ""); err != nil {
|
||||
if err := localHttpsServer.ServeTLS(s3ApiLocalListenerHttps, "", ""); err != nil && err != http.ErrServerClosed {
|
||||
glog.Fatalf("S3 API Server Fail to serve: %v", err)
|
||||
}
|
||||
}()
|
||||
}
|
||||
httpsServer := newHttpServer(router, tlsConfig)
|
||||
httpServers = append(httpServers, httpsServer)
|
||||
go func() {
|
||||
if err = newHttpServer(router, tlsConfig).ServeTLS(s3ApiListenerHttps, "", ""); err != nil {
|
||||
if err := httpsServer.ServeTLS(s3ApiListenerHttps, "", ""); err != nil && err != http.ErrServerClosed {
|
||||
glog.Fatalf("S3 API Server Fail to serve: %v", err)
|
||||
}
|
||||
}()
|
||||
@@ -537,31 +542,42 @@ func (s3opt *S3Options) startS3Server() bool {
|
||||
if *s3opt.tlsPrivateKey == "" || *s3opt.portHttps > 0 {
|
||||
glog.V(0).Infof("Start Seaweed S3 API Server %s at http port %d", version.Version(), *s3opt.port)
|
||||
if s3ApiLocalListener != nil {
|
||||
localServer := newHttpServer(router, nil)
|
||||
httpServers = append(httpServers, localServer)
|
||||
go func() {
|
||||
if err = newHttpServer(router, nil).Serve(s3ApiLocalListener); err != nil {
|
||||
if err := localServer.Serve(s3ApiLocalListener); err != nil && err != http.ErrServerClosed {
|
||||
glog.Fatalf("S3 API Server Fail to serve: %v", err)
|
||||
}
|
||||
}()
|
||||
}
|
||||
httpS := newHttpServer(router, nil)
|
||||
if s3opt.shutdownCtx != nil {
|
||||
go func() {
|
||||
<-s3opt.shutdownCtx.Done()
|
||||
httpS.Shutdown(context.Background())
|
||||
grpcS.Stop()
|
||||
}()
|
||||
}
|
||||
httpServers = append(httpServers, httpS)
|
||||
shutdown := s3opt.newShutdown(stopGrpcServer, s3ApiServer, httpServers)
|
||||
if err = httpS.Serve(s3ApiListener); err != nil && err != http.ErrServerClosed {
|
||||
glog.Fatalf("S3 API Server Fail to serve: %v", err)
|
||||
}
|
||||
// Serve returns when listeners close, before active requests finish.
|
||||
shutdown()
|
||||
}
|
||||
|
||||
return true
|
||||
|
||||
}
|
||||
|
||||
func (s3opt *S3Options) newShutdown(stopGrpc func(), s3ApiServer *s3api.S3ApiServer, httpServers []*http.Server) func() {
|
||||
shutdown := newGracefulShutdown(stopGrpc, s3ApiServer.Shutdown, httpServers...)
|
||||
grace.OnInterrupt(shutdown)
|
||||
if s3opt.shutdownCtx != nil {
|
||||
go func() {
|
||||
<-s3opt.shutdownCtx.Done()
|
||||
shutdown()
|
||||
}()
|
||||
}
|
||||
return shutdown
|
||||
}
|
||||
|
||||
// startIcebergServer starts the Iceberg REST Catalog server on a separate port.
|
||||
func (s3opt *S3Options) startIcebergServer(s3ApiServer *s3api.S3ApiServer) {
|
||||
func (s3opt *S3Options) startIcebergServer(s3ApiServer *s3api.S3ApiServer) *http.Server {
|
||||
icebergRouter := mux.NewRouter().SkipClean(true)
|
||||
// warehouse/parent query values may legally contain ';', which Go's
|
||||
// url.ParseQuery would otherwise drop
|
||||
@@ -591,12 +607,6 @@ func (s3opt *S3Options) startIcebergServer(s3ApiServer *s3api.S3ApiServer) {
|
||||
glog.V(0).Infof("Start Iceberg REST Catalog Server at http://%s", listenAddress)
|
||||
|
||||
httpS := newHttpServer(icebergRouter, nil)
|
||||
if s3opt.shutdownCtx != nil {
|
||||
go func() {
|
||||
<-s3opt.shutdownCtx.Done()
|
||||
httpS.Shutdown(context.Background())
|
||||
}()
|
||||
}
|
||||
// Serve on localhost as well if we're bound to a different interface
|
||||
if icebergLocalListener != nil {
|
||||
go func() {
|
||||
@@ -605,15 +615,18 @@ func (s3opt *S3Options) startIcebergServer(s3ApiServer *s3api.S3ApiServer) {
|
||||
}
|
||||
}()
|
||||
}
|
||||
if err = httpS.Serve(icebergListener); err != nil && err != http.ErrServerClosed {
|
||||
glog.Fatalf("Iceberg REST Catalog Server Fail to serve: %v", err)
|
||||
}
|
||||
go func() {
|
||||
if err := httpS.Serve(icebergListener); err != nil && err != http.ErrServerClosed {
|
||||
glog.Fatalf("Iceberg REST Catalog Server Fail to serve: %v", err)
|
||||
}
|
||||
}()
|
||||
return httpS
|
||||
}
|
||||
|
||||
// startLanceServer starts the Lance Namespace server on a separate port. It
|
||||
// shares the Iceberg catalog's credential role: one deployment vends table
|
||||
// credentials one way, whichever catalog the client speaks to.
|
||||
func (s3opt *S3Options) startLanceServer(s3ApiServer *s3api.S3ApiServer) {
|
||||
func (s3opt *S3Options) startLanceServer(s3ApiServer *s3api.S3ApiServer) *http.Server {
|
||||
lanceRouter := mux.NewRouter().SkipClean(true)
|
||||
lanceRouter.Use(util_http.EscapeSemicolonsInQuery)
|
||||
|
||||
@@ -636,12 +649,6 @@ func (s3opt *S3Options) startLanceServer(s3ApiServer *s3api.S3ApiServer) {
|
||||
glog.V(0).Infof("Start Lance Namespace Server at http://%s", listenAddress)
|
||||
|
||||
httpS := newHttpServer(lanceRouter, nil)
|
||||
if s3opt.shutdownCtx != nil {
|
||||
go func() {
|
||||
<-s3opt.shutdownCtx.Done()
|
||||
httpS.Shutdown(context.Background())
|
||||
}()
|
||||
}
|
||||
if lanceLocalListener != nil {
|
||||
go func() {
|
||||
if err := httpS.Serve(lanceLocalListener); err != nil && err != http.ErrServerClosed {
|
||||
@@ -649,9 +656,12 @@ func (s3opt *S3Options) startLanceServer(s3ApiServer *s3api.S3ApiServer) {
|
||||
}
|
||||
}()
|
||||
}
|
||||
if err = httpS.Serve(lanceListener); err != nil && err != http.ErrServerClosed {
|
||||
glog.Fatalf("Lance Namespace Server Fail to serve: %v", err)
|
||||
}
|
||||
go func() {
|
||||
if err := httpS.Serve(lanceListener); err != nil && err != http.ErrServerClosed {
|
||||
glog.Fatalf("Lance Namespace Server Fail to serve: %v", err)
|
||||
}
|
||||
}()
|
||||
return httpS
|
||||
}
|
||||
|
||||
// deriveLanceStorageEndpoint picks the endpoint the Lance namespace puts in
|
||||
|
||||
@@ -0,0 +1,81 @@
|
||||
package command
|
||||
|
||||
import (
|
||||
"context"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api"
|
||||
)
|
||||
|
||||
func TestS3ShutdownDrainsActiveRequestBeforeServeReturns(t *testing.T) {
|
||||
entered := make(chan struct{})
|
||||
release := make(chan struct{})
|
||||
finish := sync.OnceFunc(func() { close(release) })
|
||||
server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) {
|
||||
close(entered)
|
||||
<-release
|
||||
w.WriteHeader(http.StatusNoContent)
|
||||
}))
|
||||
t.Cleanup(server.Close)
|
||||
t.Cleanup(finish)
|
||||
httpClosing := make(chan struct{})
|
||||
server.Config.RegisterOnShutdown(func() { close(httpClosing) })
|
||||
|
||||
requestDone := make(chan struct{})
|
||||
go func() {
|
||||
defer close(requestDone)
|
||||
response, err := server.Client().Get(server.URL)
|
||||
if err != nil {
|
||||
t.Error(err)
|
||||
return
|
||||
}
|
||||
response.Body.Close()
|
||||
if response.StatusCode != http.StatusNoContent {
|
||||
t.Errorf("active request returned %s", response.Status)
|
||||
}
|
||||
}()
|
||||
select {
|
||||
case <-entered:
|
||||
case <-requestDone:
|
||||
t.Fatal("request ended before reaching the handler")
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatal("request did not reach the handler")
|
||||
}
|
||||
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
defer cancel()
|
||||
grpcStopped := make(chan struct{})
|
||||
s3opt := &S3Options{shutdownCtx: ctx}
|
||||
shutdown := s3opt.newShutdown(func() { close(grpcStopped) }, &s3api.S3ApiServer{}, []*http.Server{server.Config})
|
||||
|
||||
cancel()
|
||||
select {
|
||||
case <-httpClosing:
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatal("cancelling shutdownCtx did not start the HTTP drain")
|
||||
}
|
||||
<-grpcStopped
|
||||
|
||||
// Model the main Serve path joining shutdown once its listener closes.
|
||||
serveReturned := make(chan struct{})
|
||||
go func() { shutdown(); close(serveReturned) }()
|
||||
select {
|
||||
case <-serveReturned:
|
||||
t.Fatal("Serve exit path returned while a request was still active")
|
||||
case <-requestDone:
|
||||
t.Fatal("active request ended before it was released")
|
||||
case <-time.After(100 * time.Millisecond):
|
||||
}
|
||||
|
||||
finish()
|
||||
<-requestDone
|
||||
select {
|
||||
case <-serveReturned:
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatal("Serve exit path did not return after the request drained")
|
||||
}
|
||||
}
|
||||
@@ -16,6 +16,13 @@ recursive_delete = false
|
||||
# for S3: how long to wait before deleting an empty folder.
|
||||
# increase this if using tools like Spark that create temporary directories.
|
||||
#s3.empty_folder_cleanup_delay = "2m"
|
||||
# where the filer's internal system metadata log chunks are stored.
|
||||
# By default the log follows the filer's -collection flag; set this to keep
|
||||
# the internal log in its own collection, out of the default one.
|
||||
# Keep the target stable: log chunks written under an older value are not
|
||||
# moved, and old log entries keep referencing that collection's volumes.
|
||||
#metaLog.collection = "filer-meta"
|
||||
#metaLog.replication = ""
|
||||
|
||||
####################################################
|
||||
# The following are filer store options
|
||||
|
||||
@@ -20,6 +20,7 @@ hosts = [
|
||||
"localhost:9092"
|
||||
]
|
||||
topic = "seaweedfs_filer"
|
||||
# event_types = ["create", "update", "delete", "rename"] # optional: filter by event types (default: all)
|
||||
offsetFile = "./last.offset"
|
||||
offsetSaveIntervalSeconds = 10
|
||||
# SASL Authentication
|
||||
|
||||
@@ -0,0 +1,53 @@
|
||||
package command
|
||||
|
||||
import (
|
||||
"context"
|
||||
"net/http"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"google.golang.org/grpc"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
)
|
||||
|
||||
func gracefulStopGrpc(grpcS *grpc.Server, timeout time.Duration) {
|
||||
glog.V(0).Infof("Gracefully stopping gRPC server")
|
||||
stopped := make(chan struct{})
|
||||
go func() {
|
||||
grpcS.GracefulStop()
|
||||
close(stopped)
|
||||
}()
|
||||
select {
|
||||
case <-stopped:
|
||||
glog.V(0).Infof("gRPC server stopped gracefully")
|
||||
case <-time.After(timeout):
|
||||
glog.V(0).Infof("gRPC server graceful stop timed out after %s, forcing stop", timeout)
|
||||
grpcS.Stop()
|
||||
}
|
||||
}
|
||||
|
||||
// newGracefulShutdown joins shutdown callers while gRPC and HTTP drain concurrently.
|
||||
func newGracefulShutdown(stopGrpc, closeServer func(), httpServers ...*http.Server) func() {
|
||||
return sync.OnceFunc(func() {
|
||||
shutdownCtx, cancel := context.WithTimeout(context.Background(), 15*time.Second)
|
||||
defer cancel()
|
||||
var drained sync.WaitGroup
|
||||
drained.Add(1)
|
||||
go func() {
|
||||
defer drained.Done()
|
||||
stopGrpc()
|
||||
}()
|
||||
for _, server := range httpServers {
|
||||
drained.Add(1)
|
||||
go func() {
|
||||
defer drained.Done()
|
||||
if err := server.Shutdown(shutdownCtx); err != nil {
|
||||
glog.Warningf("HTTP shutdown: %v", err)
|
||||
}
|
||||
}()
|
||||
}
|
||||
drained.Wait()
|
||||
closeServer()
|
||||
})
|
||||
}
|
||||
@@ -0,0 +1,96 @@
|
||||
package command
|
||||
|
||||
import (
|
||||
"context"
|
||||
"net"
|
||||
"sync"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"google.golang.org/grpc"
|
||||
"google.golang.org/grpc/credentials/insecure"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/iam_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/s3_pb"
|
||||
)
|
||||
|
||||
type blockingIamCache struct {
|
||||
s3_pb.UnimplementedSeaweedS3IamCacheServer
|
||||
entered chan struct{}
|
||||
release chan struct{}
|
||||
}
|
||||
|
||||
func (b *blockingIamCache) PutIdentity(ctx context.Context, _ *iam_pb.PutIdentityRequest) (*iam_pb.PutIdentityResponse, error) {
|
||||
close(b.entered)
|
||||
// Stop only cancels handler contexts; it cannot interrupt a handler that ignores them.
|
||||
select {
|
||||
case <-b.release:
|
||||
return &iam_pb.PutIdentityResponse{}, nil
|
||||
case <-ctx.Done():
|
||||
return nil, ctx.Err()
|
||||
}
|
||||
}
|
||||
|
||||
func startBlockingGrpc(t *testing.T) (grpcS *grpc.Server, release func(), rpcErr chan error) {
|
||||
t.Helper()
|
||||
listener, err := net.Listen("tcp", "127.0.0.1:0")
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
handler := &blockingIamCache{entered: make(chan struct{}), release: make(chan struct{})}
|
||||
grpcS = grpc.NewServer()
|
||||
s3_pb.RegisterSeaweedS3IamCacheServer(grpcS, handler)
|
||||
go grpcS.Serve(listener)
|
||||
t.Cleanup(grpcS.Stop)
|
||||
release = sync.OnceFunc(func() { close(handler.release) })
|
||||
t.Cleanup(release)
|
||||
|
||||
conn, err := grpc.NewClient(listener.Addr().String(), grpc.WithTransportCredentials(insecure.NewCredentials()))
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
t.Cleanup(func() { conn.Close() })
|
||||
rpcErr = make(chan error, 1)
|
||||
go func() {
|
||||
_, err := s3_pb.NewSeaweedS3IamCacheClient(conn).PutIdentity(context.Background(), &iam_pb.PutIdentityRequest{})
|
||||
rpcErr <- err
|
||||
}()
|
||||
<-handler.entered
|
||||
return grpcS, release, rpcErr
|
||||
}
|
||||
|
||||
func TestGracefulStopGrpcWaitsForInFlightRPC(t *testing.T) {
|
||||
grpcS, release, rpcErr := startBlockingGrpc(t)
|
||||
|
||||
stopped := make(chan struct{})
|
||||
go func() { gracefulStopGrpc(grpcS, 5*time.Second); close(stopped) }()
|
||||
select {
|
||||
case <-stopped:
|
||||
t.Fatal("gRPC stopped while an RPC was in flight")
|
||||
case <-time.After(100 * time.Millisecond):
|
||||
}
|
||||
release()
|
||||
if err := <-rpcErr; err != nil {
|
||||
t.Errorf("in-flight RPC failed: %v", err)
|
||||
}
|
||||
select {
|
||||
case <-stopped:
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatal("gRPC graceful stop did not return after the RPC completed")
|
||||
}
|
||||
}
|
||||
|
||||
func TestGracefulStopGrpcForcesStopAfterTimeout(t *testing.T) {
|
||||
grpcS, _, rpcErr := startBlockingGrpc(t)
|
||||
|
||||
stopped := make(chan struct{})
|
||||
go func() { gracefulStopGrpc(grpcS, 50*time.Millisecond); close(stopped) }()
|
||||
select {
|
||||
case <-stopped:
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatal("gRPC graceful stop did not honour its timeout")
|
||||
}
|
||||
if err := <-rpcErr; err == nil {
|
||||
t.Error("in-flight RPC succeeded although the server was force-stopped")
|
||||
}
|
||||
}
|
||||
@@ -883,6 +883,9 @@ func (ecb *ecBalancer) balance(collections []string) error {
|
||||
// nodes fill proportionally (matching the worker). This is identical to raw
|
||||
// shard count when capacities are uniform.
|
||||
GlobalUtilizationBased: true,
|
||||
RackTotalCapRaised: func(collection string, vid uint32, rackCap, evenCap int) {
|
||||
fmt.Printf("ec volume %d (collection %q): rack-total-cap raised to %d shards per rack, above the even %d: the racks lack room for an even spread\n", vid, collection, rackCap, evenCap)
|
||||
},
|
||||
})
|
||||
if len(ecb.volumeIds) > 0 {
|
||||
var deletions int
|
||||
|
||||
@@ -380,15 +380,6 @@ func fetchWholeChunk(ctx context.Context, bytesBuffer *bytes.Buffer, lookupFileI
|
||||
})
|
||||
}
|
||||
|
||||
func fetchChunkRange(ctx context.Context, buffer []byte, lookupFileIdFn wdclient.LookupFileIdFunctionType, fileId string, cipherKey []byte, isGzipped bool, offset int64, refreshUrls util_http.RefreshUrlsFunc) (int, error) {
|
||||
urlStrings, err := lookupFileIdFn(ctx, fileId)
|
||||
if err != nil {
|
||||
glog.ErrorfCtx(ctx, "operation LookupFileId %s failed, err: %v", fileId, err)
|
||||
return 0, err
|
||||
}
|
||||
return util_http.RetriedFetchChunkData(ctx, buffer, urlStrings, cipherKey, isGzipped, false, offset, fileId, refreshUrls)
|
||||
}
|
||||
|
||||
// retriedStreamFetchChunkData streams a chunk from the first location that
|
||||
// answers. refreshUrls may be nil; when a location failed and a later one
|
||||
// answered, it is called so the reads that follow start from a fresh list.
|
||||
|
||||
@@ -185,6 +185,15 @@ func (cv *ChunkView) IsFullChunk() bool {
|
||||
return cv.OffsetInChunk == 0 && cv.ViewSize == cv.ChunkSize
|
||||
}
|
||||
|
||||
// CanRangeFetch reports whether fetching just the view's byte range avoids
|
||||
// reading more than the view needs. Ciphered and compressed chunks are
|
||||
// stored and served whole — a range on either still costs a full read plus
|
||||
// decrypt or decompress on the volume server — so partial views of them
|
||||
// take the shared whole-chunk path instead.
|
||||
func (cv *ChunkView) CanRangeFetch() bool {
|
||||
return cv.CipherKey == nil && !cv.IsGzipped
|
||||
}
|
||||
|
||||
func ViewFromChunks(ctx context.Context, lookupFileIdFn wdclient.LookupFileIdFunctionType, chunks []*filer_pb.FileChunk, offset int64, size int64) (chunkViews *IntervalList[*ChunkView]) {
|
||||
|
||||
visibles, _ := NonOverlappingVisibleIntervals(ctx, lookupFileIdFn, chunks, offset, offset+size)
|
||||
|
||||
+82
-12
@@ -7,6 +7,7 @@ import (
|
||||
"os"
|
||||
"sort"
|
||||
"strings"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
"time"
|
||||
|
||||
@@ -46,17 +47,22 @@ var (
|
||||
)
|
||||
|
||||
type Filer struct {
|
||||
UniqueFilerId int32
|
||||
UniqueFilerEpoch int32
|
||||
Store VirtualFilerStore
|
||||
MasterClient *wdclient.MasterClient
|
||||
FileIdDeletionQueue *util.UnboundedQueue
|
||||
GrpcDialOption grpc.DialOption
|
||||
DirBucketsPath string
|
||||
Cipher bool
|
||||
LocalMetaLogBuffer *log_buffer.LogBuffer
|
||||
metaLogCollection string
|
||||
metaLogReplication string
|
||||
UniqueFilerId int32
|
||||
UniqueFilerEpoch int32
|
||||
Store VirtualFilerStore
|
||||
MasterClient *wdclient.MasterClient
|
||||
FileIdDeletionQueue *util.UnboundedQueue
|
||||
GrpcDialOption grpc.DialOption
|
||||
DirBucketsPath string
|
||||
Cipher bool
|
||||
LocalMetaLogBuffer *log_buffer.LogBuffer
|
||||
metaLogCollection string
|
||||
metaLogReplication string
|
||||
// Override where system metadata-log chunks are assigned, keeping the
|
||||
// internal log out of the default collection; empty keeps today's
|
||||
// behaviour. Set via viper: filer.options.metaLog.collection / .replication.
|
||||
metaLogTargetCollection string
|
||||
metaLogTargetReplication string
|
||||
DefaultDiskType string
|
||||
MetaAggregator *MetaAggregator
|
||||
Signature int32
|
||||
@@ -80,6 +86,36 @@ type Filer struct {
|
||||
// rebuild finishes; lazy remote reads wait on it so a pending delete
|
||||
// cannot resurrect in the gap.
|
||||
remoteTombstonesDone atomic.Pointer[chan struct{}]
|
||||
|
||||
// Durable deletion ledger (see filer_deletion_persist.go). The set of
|
||||
// fileIds that still need deleting but are not yet confirmed gone, mirrored
|
||||
// to the store so a restart does not leak chunks. Guarded by
|
||||
// deletionLedgerLock; nil-safe for Filer literals in tests.
|
||||
// pendingDeletions maps each id to its enqueue epoch so an expiry-forget
|
||||
// cannot erase a re-queued id.
|
||||
deletionLedgerLock sync.Mutex
|
||||
pendingDeletions map[string]uint64
|
||||
deletionSeq uint64
|
||||
deletionLedgerDirty bool
|
||||
deletionLedgerParts int
|
||||
deletionLedgerGen int
|
||||
// deletionLedgerStale holds orphan part keys from abandoned multipart
|
||||
// writes, retried on the next snapshot.
|
||||
deletionLedgerStale []string
|
||||
// deletionSnapshotLock serializes ledger writes in copy order so an
|
||||
// in-flight timer snapshot cannot overwrite a newer shutdown snapshot.
|
||||
deletionSnapshotLock sync.Mutex
|
||||
// deletionLedgerBlocked is set when the startup ledger read fails: the
|
||||
// persisted set is then unknown, so snapshots retry the read instead of
|
||||
// overwriting the unread ledger with a partial set.
|
||||
deletionLedgerBlocked atomic.Bool
|
||||
deletionLedgerFlush chan struct{}
|
||||
// flushCtx is the context metadata-log flushes append under. It stays
|
||||
// live for the process lifetime and is cancelled once during Shutdown,
|
||||
// bounding every retry - in-flight and queued alike - to one deadline.
|
||||
// Nil for Filer literals built without NewFiler.
|
||||
flushCtx context.Context
|
||||
flushCancel context.CancelFunc
|
||||
}
|
||||
|
||||
func NewFiler(masters pb.ServerDiscovery, grpcDialOption grpc.DialOption, filerHost pb.ServerAddress, filerGroup string, collection string, replication string, dataCenter string, maxFilenameLength uint32, notifyFn func()) *Filer {
|
||||
@@ -93,10 +129,12 @@ func NewFiler(masters pb.ServerDiscovery, grpcDialOption grpc.DialOption, filerH
|
||||
Dlm: lock_manager.NewDistributedLockManager(filerHost),
|
||||
MaxFilenameLength: maxFilenameLength,
|
||||
deletionQuit: make(chan struct{}),
|
||||
deletionLedgerFlush: make(chan struct{}, 1),
|
||||
DeletionRetryQueue: NewDeletionRetryQueue(),
|
||||
persistedLogCache: newPersistedLogCache(persistedLogCacheMaxBytes),
|
||||
remoteTombstones: newRemoteDeletionTombstones(),
|
||||
}
|
||||
f.flushCtx, f.flushCancel = context.WithCancel(context.Background())
|
||||
if f.UniqueFilerId < 0 {
|
||||
f.UniqueFilerId = -f.UniqueFilerId
|
||||
}
|
||||
@@ -111,6 +149,19 @@ func NewFiler(masters pb.ServerDiscovery, grpcDialOption grpc.DialOption, filerH
|
||||
f.metaLogCollection = collection
|
||||
f.metaLogReplication = replication
|
||||
|
||||
// Optional override for where the system metadata-log chunks land, so
|
||||
// operators can keep internal log volumes out of the default collection.
|
||||
// Unset (""), this changes nothing: meta logs keep following the filer
|
||||
// default exactly as before.
|
||||
v := util.GetViper()
|
||||
v.SetDefault("filer.options.metaLog.collection", "")
|
||||
v.SetDefault("filer.options.metaLog.replication", "")
|
||||
f.metaLogTargetCollection = v.GetString("filer.options.metaLog.collection")
|
||||
f.metaLogTargetReplication = v.GetString("filer.options.metaLog.replication")
|
||||
if f.metaLogTargetCollection != "" {
|
||||
glog.V(0).Infof("system metadata logs will be stored in collection %q", f.metaLogTargetCollection)
|
||||
}
|
||||
|
||||
if newPlacementOverlay != nil {
|
||||
f.placementOverlay = newPlacementOverlay(f)
|
||||
}
|
||||
@@ -210,7 +261,16 @@ func (f *Filer) ListExistingPeerUpdates(ctx context.Context) (existingNodes []*m
|
||||
func (f *Filer) SetStore(store FilerStore) (isFresh bool) {
|
||||
f.Store = NewFilerStoreWrapper(store)
|
||||
|
||||
return f.setOrLoadFilerStoreSignature(store)
|
||||
isFresh = f.setOrLoadFilerStoreSignature(store)
|
||||
|
||||
// Recover deletions that were pending when a previous process died, and keep
|
||||
// the durable ledger snapshotted while running (see filer_deletion_persist.go).
|
||||
// A failed ledger read leaves writes blocked until a snapshot retries and
|
||||
// the read succeeds, so the unread ledger is never overwritten.
|
||||
f.reloadDeletionLedger()
|
||||
f.startDeletionLedgerSnapshotter()
|
||||
|
||||
return isFresh
|
||||
}
|
||||
|
||||
func (f *Filer) setOrLoadFilerStoreSignature(store FilerStore) (isFresh bool) {
|
||||
@@ -755,9 +815,19 @@ func (f *Filer) Shutdown() {
|
||||
if f.EmptyFolderCleaner != nil {
|
||||
f.EmptyFolderCleaner.Stop()
|
||||
}
|
||||
// Bound the remaining flush retries: with the cluster already gone each
|
||||
// append keeps failing, and an unbounded retry would hold the shutdown
|
||||
// drain open indefinitely. One shared deadline covers every in-flight and
|
||||
// queued window; the flush loop drops what it cannot write.
|
||||
if f.flushCancel != nil {
|
||||
time.AfterFunc(shutdownMetadataLogFlushBudget, f.flushCancel)
|
||||
}
|
||||
f.LocalMetaLogBuffer.ShutdownLogBuffer()
|
||||
// The final metadata-log flush still needs the store to append its entry.
|
||||
f.LocalMetaLogBuffer.WaitForShutdown()
|
||||
// Persist the deletion ledger one last time before the store closes, so a
|
||||
// clean shutdown leaves the recovery set exactly consistent with reality.
|
||||
f.snapshotDeletionLedger()
|
||||
f.Store.Shutdown()
|
||||
}
|
||||
|
||||
|
||||
@@ -262,8 +262,9 @@ func TestClearReadOnly(t *testing.T) {
|
||||
type fakeFilerConfClient struct {
|
||||
filer_pb.SeaweedFilerClient
|
||||
|
||||
mu sync.Mutex
|
||||
entries map[string]*filer_pb.Entry // key: dir+"/"+name
|
||||
mu sync.Mutex
|
||||
entries map[string]*filer_pb.Entry // key: dir+"/"+name
|
||||
updateCalls int
|
||||
}
|
||||
|
||||
func newFakeFilerConfClient() *fakeFilerConfClient {
|
||||
@@ -310,6 +311,7 @@ func (c *fakeFilerConfClient) CreateEntry(_ context.Context, in *filer_pb.Create
|
||||
func (c *fakeFilerConfClient) UpdateEntry(_ context.Context, in *filer_pb.UpdateEntryRequest, _ ...grpc.CallOption) (*filer_pb.UpdateEntryResponse, error) {
|
||||
c.mu.Lock()
|
||||
defer c.mu.Unlock()
|
||||
c.updateCalls++
|
||||
key := c.key(in.Directory, in.Entry.Name)
|
||||
if !conditionHoldsLocked(in.Condition, c.entries[key]) {
|
||||
return nil, errors.New("precondition failed")
|
||||
|
||||
@@ -72,9 +72,12 @@ func (f *Filer) DeleteEntryMetaAndData(ctx context.Context, p util.FullPath, isR
|
||||
if isDeleteCollection {
|
||||
collectionName = f.bucketCollection(ctx, entry.Name())
|
||||
}
|
||||
// A preserved collection outlives the bucket, so its chunks are collected
|
||||
// per entry rather than dropped wholesale with it.
|
||||
dropsCollection := isDeleteCollection && collectionName != ""
|
||||
if entry.IsDirectory() {
|
||||
// delete the folder children, not including the folder itself
|
||||
err = f.doBatchDeleteFolderMetaAndData(ctx, entry, isRecursive, ignoreRecursiveError, shouldDeleteChunks && !isDeleteCollection, isDeleteCollection, isFromOtherCluster, signatures, func(hardLinkIds []HardLinkId) error {
|
||||
err = f.doBatchDeleteFolderMetaAndData(ctx, entry, isRecursive, ignoreRecursiveError, shouldDeleteChunks && !dropsCollection, isDeleteCollection && (dropsCollection || !shouldDeleteChunks), isFromOtherCluster, signatures, func(hardLinkIds []HardLinkId) error {
|
||||
// A case not handled:
|
||||
// what if the chunk is in a different collection?
|
||||
if shouldDeleteChunks {
|
||||
@@ -97,7 +100,7 @@ func (f *Filer) DeleteEntryMetaAndData(ctx context.Context, p util.FullPath, isR
|
||||
return fmt.Errorf("delete file %s: %v", p, err)
|
||||
}
|
||||
|
||||
if shouldDeleteChunks && !isDeleteCollection {
|
||||
if shouldDeleteChunks && !dropsCollection {
|
||||
if len(entry.HardLinkId) != 0 && entry.HardLinkCounter > 1 {
|
||||
// if the file is a hard link and there are other hard links, do not delete the chunks
|
||||
} else {
|
||||
@@ -261,10 +264,17 @@ func (f *Filer) bucketCollection(ctx context.Context, bucket string) (collection
|
||||
collection = resolve(bucketDir, bucket)
|
||||
|
||||
// Rule-less writes outside buckets fall back to the filer's default
|
||||
// collection, so a bucket resolving there shares it with them.
|
||||
// collection, so a bucket resolving there shares it with them. The
|
||||
// system metadata-log collection (when explicitly redirected via
|
||||
// filer.options.metaLog.collection) is in the same boat: it backs internal
|
||||
// log volumes, so a bucket that resolves there must never drop it
|
||||
// either.
|
||||
if collection == f.metaLogCollection {
|
||||
return ""
|
||||
}
|
||||
if f.metaLogTargetCollection != "" && collection == f.metaLogTargetCollection {
|
||||
return ""
|
||||
}
|
||||
|
||||
// A rule whose prefix escapes the bucket can route other paths into the
|
||||
// same collection, including prefixes nested under surviving buckets.
|
||||
|
||||
@@ -66,8 +66,9 @@ type DeletionRetryItem struct {
|
||||
RetryCount int
|
||||
NextRetryAt time.Time
|
||||
LastError string
|
||||
heapIndex int // index in the heap (for heap.Interface)
|
||||
inFlight bool // true when item is being processed, prevents duplicate additions
|
||||
ledgerEpoch uint64 // pendingDeletions epoch when the item was queued; expiry forgets only a matching epoch
|
||||
heapIndex int // index in the heap (for heap.Interface)
|
||||
inFlight bool // true when item is being processed, prevents duplicate additions
|
||||
}
|
||||
|
||||
// retryHeap implements heap.Interface for DeletionRetryItem
|
||||
@@ -163,7 +164,7 @@ func calculateBackoff(retryCount int) time.Duration {
|
||||
|
||||
// AddOrUpdate adds a new failed deletion or updates an existing one
|
||||
// Time complexity: O(log N) for insertion/update
|
||||
func (q *DeletionRetryQueue) AddOrUpdate(fileId string, errorMsg string) {
|
||||
func (q *DeletionRetryQueue) AddOrUpdate(fileId string, errorMsg string, ledgerEpoch uint64) {
|
||||
q.lock.Lock()
|
||||
defer q.lock.Unlock()
|
||||
|
||||
@@ -173,6 +174,13 @@ func (q *DeletionRetryQueue) AddOrUpdate(fileId string, errorMsg string) {
|
||||
// The existing retry schedule should proceed.
|
||||
// RetryCount is only incremented in RequeueForRetry when an actual retry is performed.
|
||||
item.LastError = errorMsg
|
||||
// Keep the recorded epoch while the item is in flight: an expiry or
|
||||
// permanent outcome from that attempt must forget only the record it
|
||||
// started with, not a newer enqueue for the same id. This also keeps
|
||||
// the field immutable once the worker can read it without the lock.
|
||||
if !item.inFlight {
|
||||
item.ledgerEpoch = ledgerEpoch
|
||||
}
|
||||
if item.inFlight {
|
||||
glog.V(2).Infof("retry for %s in-flight: attempt %d, will preserve retry state", fileId, item.RetryCount)
|
||||
} else {
|
||||
@@ -188,6 +196,7 @@ func (q *DeletionRetryQueue) AddOrUpdate(fileId string, errorMsg string) {
|
||||
RetryCount: 1,
|
||||
NextRetryAt: time.Now().Add(delay),
|
||||
LastError: errorMsg,
|
||||
ledgerEpoch: ledgerEpoch,
|
||||
inFlight: false,
|
||||
}
|
||||
heap.Push(&q.heap, item)
|
||||
@@ -216,18 +225,19 @@ func (q *DeletionRetryQueue) RequeueForRetry(item *DeletionRetryItem, errorMsg s
|
||||
heap.Push(&q.heap, item)
|
||||
}
|
||||
|
||||
// GetReadyItems returns items that are ready to be retried and marks them as in-flight
|
||||
// GetReadyItems returns items that are ready to be retried and marks them as in-flight.
|
||||
// Expired reports fileIds dropped for exceeding MaxRetryAttempts — permanent
|
||||
// dropouts the caller must also forget in the durable deletion ledger.
|
||||
// Time complexity: O(K log N) where K is the number of ready items
|
||||
// Items are processed in order of NextRetryAt (earliest first)
|
||||
func (q *DeletionRetryQueue) GetReadyItems(maxItems int) []*DeletionRetryItem {
|
||||
func (q *DeletionRetryQueue) GetReadyItems(maxItems int) (ready []*DeletionRetryItem, expired []*DeletionRetryItem) {
|
||||
q.lock.Lock()
|
||||
defer q.lock.Unlock()
|
||||
|
||||
now := time.Now()
|
||||
var readyItems []*DeletionRetryItem
|
||||
|
||||
// Peek at items from the top of the heap (earliest NextRetryAt)
|
||||
for len(q.heap) > 0 && len(readyItems) < maxItems {
|
||||
for len(q.heap) > 0 && len(ready) < maxItems {
|
||||
item := q.heap[0]
|
||||
|
||||
// If the earliest item is not ready yet, no other items are ready either
|
||||
@@ -240,15 +250,16 @@ func (q *DeletionRetryQueue) GetReadyItems(maxItems int) []*DeletionRetryItem {
|
||||
|
||||
if item.RetryCount <= MaxRetryAttempts {
|
||||
item.inFlight = true // Mark as being processed
|
||||
readyItems = append(readyItems, item)
|
||||
ready = append(ready, item)
|
||||
} else {
|
||||
// Max attempts reached, log and discard completely
|
||||
delete(q.itemIndex, item.FileId)
|
||||
expired = append(expired, item)
|
||||
glog.Warningf("max retry attempts (%d) reached for %s, last error: %s", MaxRetryAttempts, item.FileId, item.LastError)
|
||||
}
|
||||
}
|
||||
|
||||
return readyItems
|
||||
return ready, expired
|
||||
}
|
||||
|
||||
// Remove removes an item from the queue (called when deletion succeeds or fails permanently)
|
||||
@@ -343,6 +354,13 @@ func (f *Filer) processDeletionBatch(ctx context.Context, toDeleteFileIds []stri
|
||||
return
|
||||
}
|
||||
|
||||
// Remember each id's ledger epoch before the remote deletes run: a
|
||||
// permanent outcome below must not erase a record re-queued mid-flight.
|
||||
epochs := make(map[string]uint64, len(uniqueFileIdsSlice))
|
||||
for _, fileId := range uniqueFileIdsSlice {
|
||||
epochs[fileId] = f.deletionEpoch(fileId)
|
||||
}
|
||||
|
||||
// Delete files and classify outcomes
|
||||
outcomes := deleteFilesAndClassify(ctx, f.GrpcDialOption, uniqueFileIdsSlice, lookupFunc)
|
||||
|
||||
@@ -356,16 +374,21 @@ func (f *Filer) processDeletionBatch(ctx context.Context, toDeleteFileIds []stri
|
||||
switch outcome.status {
|
||||
case deletionOutcomeSuccess:
|
||||
successCount++
|
||||
f.forgetDeletion(fileId) // confirmed gone: drop from durable ledger
|
||||
case deletionOutcomeNotFound:
|
||||
notFoundCount++
|
||||
f.forgetDeletion(fileId) // already gone: drop from durable ledger
|
||||
case deletionOutcomeRetryable, deletionOutcomeNoResult:
|
||||
retryableErrorCount++
|
||||
f.DeletionRetryQueue.AddOrUpdate(fileId, outcome.errorMsg)
|
||||
// Keep in the durable ledger: not confirmed, must survive restarts.
|
||||
f.DeletionRetryQueue.AddOrUpdate(fileId, outcome.errorMsg, f.deletionEpoch(fileId))
|
||||
if len(errorDetails) < MaxLoggedErrorDetails {
|
||||
errorDetails = append(errorDetails, fileId+": "+outcome.errorMsg+" (will retry)")
|
||||
}
|
||||
case deletionOutcomePermanent:
|
||||
permanentErrorCount++
|
||||
// gave up: drop the record, unless the id was re-queued meanwhile
|
||||
f.forgetDeletionEpoch(fileId, epochs[fileId])
|
||||
if len(errorDetails) < MaxLoggedErrorDetails {
|
||||
errorDetails = append(errorDetails, fileId+": "+outcome.errorMsg+" (permanent)")
|
||||
}
|
||||
@@ -523,7 +546,16 @@ func (f *Filer) loopProcessingDeletionRetry(lookupFunc func([]string) (map[strin
|
||||
// Process all ready items in batches until queue is empty
|
||||
totalProcessed := 0
|
||||
for {
|
||||
readyItems := f.DeletionRetryQueue.GetReadyItems(DeletionRetryBatchSize)
|
||||
readyItems, expired := f.DeletionRetryQueue.GetReadyItems(DeletionRetryBatchSize)
|
||||
for _, item := range expired {
|
||||
// Permanently discarded — drop the ledger record, but only
|
||||
// if the id was not re-queued after this retry item was
|
||||
// recorded. A surviving newer record still needs work, so
|
||||
// it goes back through the hot queue.
|
||||
if !f.forgetDeletionEpoch(item.FileId, item.ledgerEpoch) {
|
||||
f.queueDeletions(item.FileId)
|
||||
}
|
||||
}
|
||||
if len(readyItems) == 0 {
|
||||
break
|
||||
}
|
||||
@@ -561,19 +593,27 @@ func (f *Filer) processRetryBatch(readyItems []*DeletionRetryItem, lookupFunc fu
|
||||
case deletionOutcomeSuccess:
|
||||
successCount++
|
||||
f.DeletionRetryQueue.Remove(item) // Remove from queue (success)
|
||||
f.forgetDeletion(item.FileId) // confirmed gone: drop from durable ledger
|
||||
glog.V(2).Infof("retry successful for %s after %d attempts", item.FileId, item.RetryCount)
|
||||
case deletionOutcomeNotFound:
|
||||
notFoundCount++
|
||||
f.DeletionRetryQueue.Remove(item) // Remove from queue (already deleted)
|
||||
f.forgetDeletion(item.FileId) // already gone: drop from durable ledger
|
||||
case deletionOutcomeRetryable, deletionOutcomeNoResult:
|
||||
retryCount++
|
||||
if outcome.status == deletionOutcomeNoResult {
|
||||
glog.Warningf("no deletion result for retried file %s, re-queuing to avoid loss", item.FileId)
|
||||
}
|
||||
// Keep in the durable ledger: still not confirmed.
|
||||
f.DeletionRetryQueue.RequeueForRetry(item, outcome.errorMsg)
|
||||
case deletionOutcomePermanent:
|
||||
permanentErrorCount++
|
||||
f.DeletionRetryQueue.Remove(item) // Remove from queue (permanent failure)
|
||||
// gave up: drop the record, unless the id was re-queued meanwhile
|
||||
// — a surviving newer record goes back through the hot queue.
|
||||
if !f.forgetDeletionEpoch(item.FileId, item.ledgerEpoch) {
|
||||
f.queueDeletions(item.FileId)
|
||||
}
|
||||
glog.Warningf("permanent error on retry for %s after %d attempts: %s", item.FileId, item.RetryCount, outcome.errorMsg)
|
||||
}
|
||||
}
|
||||
@@ -604,7 +644,7 @@ func (f *Filer) DeleteChunks(ctx context.Context, fullpath util.FullPath, chunks
|
||||
func (f *Filer) doDeleteChunks(ctx context.Context, chunks []*filer_pb.FileChunk) {
|
||||
for _, chunk := range chunks {
|
||||
if !chunk.IsChunkManifest {
|
||||
f.FileIdDeletionQueue.EnQueue(chunk.GetFileIdString())
|
||||
f.queueDeletions(chunk.GetFileIdString())
|
||||
continue
|
||||
}
|
||||
dataChunks, manifestResolveErr := ResolveOneChunkManifest(ctx, f.MasterClient.LookupFileId, chunk, f.MasterClient)
|
||||
@@ -612,15 +652,15 @@ func (f *Filer) doDeleteChunks(ctx context.Context, chunks []*filer_pb.FileChunk
|
||||
glog.V(0).InfofCtx(ctx, "failed to resolve manifest %s: %v", chunk.FileId, manifestResolveErr)
|
||||
}
|
||||
for _, dChunk := range dataChunks {
|
||||
f.FileIdDeletionQueue.EnQueue(dChunk.GetFileIdString())
|
||||
f.queueDeletions(dChunk.GetFileIdString())
|
||||
}
|
||||
f.FileIdDeletionQueue.EnQueue(chunk.GetFileIdString())
|
||||
f.queueDeletions(chunk.GetFileIdString())
|
||||
}
|
||||
}
|
||||
|
||||
func (f *Filer) DeleteChunksNotRecursive(chunks []*filer_pb.FileChunk) {
|
||||
for _, chunk := range chunks {
|
||||
f.FileIdDeletionQueue.EnQueue(chunk.GetFileIdString())
|
||||
f.queueDeletions(chunk.GetFileIdString())
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,758 @@
|
||||
package filer
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/glog"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
)
|
||||
|
||||
// Persistent deletion ledger.
|
||||
//
|
||||
// Background: the in-memory FileIdDeletionQueue and DeletionRetryQueue lose every
|
||||
// queued-but-unconfirmed deletion when the filer process restarts (the upstream
|
||||
// TODO in filer_deletion.go notes this and proposes exactly the "periodic snapshot
|
||||
// with recovery on startup" strategy implemented here). Because a delete only ever
|
||||
// enters the pipeline through the queue, a crash between enqueue and the volume
|
||||
// actually confirming the delete leaks the chunk: nothing remembers it, so it
|
||||
// becomes an orphan the next fsck sees as 100% orphaned and a later
|
||||
// meta-replay from a lagging peer can "resurrect" as if it were live data.
|
||||
//
|
||||
// The ledger is a KV entry per filer (KvKeyDeletionLedger suffixed with this
|
||||
// filer's address, so filers sharing one store never overwrite each other's
|
||||
// pending sets) holding the fileIds that still need to be deleted but have not
|
||||
// yet been confirmed gone. It is:
|
||||
// - additive: every EnQueue also records the fileId here,
|
||||
// - subtractive: only terminal outcomes (success / not-found / permanent)
|
||||
// remove it; retryable failures keep it,
|
||||
// - snapshotted when it changes, on a timer, and on Shutdown, so a crash
|
||||
// loses only ids queued in the last in-flight write — recovery re-enqueues
|
||||
// the whole pending set and the idempotent volume delete absorbs anything
|
||||
// that actually completed before the crash.
|
||||
//
|
||||
// Stores that cap value size (FoundationDB at 100KB) get the set split into
|
||||
// part keys when one value would exceed deletionLedgerPartSize. Multipart
|
||||
// snapshots write each generation under generation-scoped part keys and
|
||||
// publish only the manifest last, so a crash mid-write leaves the previous
|
||||
// generation intact; orphaned parts from abandoned generations are tracked in
|
||||
// deletionLedgerStale and retried on the next write.
|
||||
//
|
||||
// Safety on a lagging peer (the resurrection case): recovered entries are replayed
|
||||
// into the queue only after a grace window (DeletionRecoveryGrace), long enough for
|
||||
// the initial peer meta-aggregation to settle so we don't purge a chunk that a
|
||||
// peer is about to re-reference as live. New live deletes are never gated.
|
||||
const (
|
||||
// KvKeyDeletionLedger is the reserved store key prefix for the persisted
|
||||
// ledger, namespaced next to FilerStoreId so it never collides with user
|
||||
// data. The full key is this prefix plus the filer's own address.
|
||||
KvKeyDeletionLedger = "filer.deleteQueue.ledger.v1"
|
||||
|
||||
// KvKeyDeletionLedgerIndex lists the scoped ledger keys in the store so a
|
||||
// filer that restarts under a different advertised address can still find
|
||||
// and claim the ledger it left behind.
|
||||
KvKeyDeletionLedgerIndex = KvKeyDeletionLedger + ".index"
|
||||
|
||||
// deletionLedgerPartSize bounds one ledger value; stores with a value cap
|
||||
// reject anything larger, which would strand a big backlog in memory only.
|
||||
deletionLedgerPartSize = 64 * 1024
|
||||
|
||||
// Defaults; all three overridable via viper (filer.deleteQueue.*).
|
||||
defaultDeletionPersistInterval = 10 * time.Second
|
||||
defaultDeletionRecoveryGrace = 30 * time.Second
|
||||
)
|
||||
|
||||
// deletionPersistEnabled reports whether ledger persistence is on.
|
||||
// Read from viper with a default of true so a plain `weed filer` gets the
|
||||
// durability guarantee without extra flags.
|
||||
func deletionPersistEnabled() bool {
|
||||
v := util.GetViper()
|
||||
v.SetDefault("filer.deleteQueue.persist", true)
|
||||
return v.GetBool("filer.deleteQueue.persist")
|
||||
}
|
||||
|
||||
func deletionPersistInterval() time.Duration {
|
||||
v := util.GetViper()
|
||||
v.SetDefault("filer.deleteQueue.persistInterval", defaultDeletionPersistInterval)
|
||||
d := v.GetDuration("filer.deleteQueue.persistInterval")
|
||||
if d <= 0 {
|
||||
return defaultDeletionPersistInterval
|
||||
}
|
||||
return d
|
||||
}
|
||||
|
||||
func deletionRecoveryGrace() time.Duration {
|
||||
v := util.GetViper()
|
||||
v.SetDefault("filer.deleteQueue.recoveryGrace", defaultDeletionRecoveryGrace)
|
||||
d := v.GetDuration("filer.deleteQueue.recoveryGrace")
|
||||
if d < 0 {
|
||||
return 0
|
||||
}
|
||||
return d
|
||||
}
|
||||
|
||||
// deletionLedgerKey scopes the ledger to this filer so several filers sharing
|
||||
// one metadata store do not overwrite each other's pending sets. A filer
|
||||
// literal without an advertised address (tests) falls back to the base key.
|
||||
func (f *Filer) deletionLedgerKey() string {
|
||||
if f.Dlm != nil && f.Dlm.Host != "" {
|
||||
return KvKeyDeletionLedger + "." + f.Dlm.Host.ToHttpAddress()
|
||||
}
|
||||
return KvKeyDeletionLedger
|
||||
}
|
||||
|
||||
func deletionLedgerPartKey(key string, part int) []byte {
|
||||
return []byte(fmt.Sprintf("%s.part.%05d", key, part))
|
||||
}
|
||||
|
||||
// deletionLedgerGenPartKey returns the part key for one generation. Generation
|
||||
// 0 keeps the original format so ledgers written before generation publishing
|
||||
// still decode.
|
||||
func deletionLedgerGenPartKey(key string, gen, part int) []byte {
|
||||
if gen == 0 {
|
||||
return deletionLedgerPartKey(key, part)
|
||||
}
|
||||
return []byte(fmt.Sprintf("%s.g%06d.part.%05d", key, gen, part))
|
||||
}
|
||||
|
||||
// startDeletionLedgerSnapshotter periodically flushes the pending deletion set
|
||||
// to durable storage. It also wakes on deletionLedgerFlush so a queued id is
|
||||
// persisted within milliseconds instead of a full interval. Started from
|
||||
// SetStore only after the ledger read succeeded.
|
||||
func (f *Filer) startDeletionLedgerSnapshotter() {
|
||||
go func() {
|
||||
ticker := time.NewTicker(deletionPersistInterval())
|
||||
defer ticker.Stop()
|
||||
for {
|
||||
select {
|
||||
case <-f.deletionQuit:
|
||||
return
|
||||
case <-ticker.C:
|
||||
case <-f.deletionLedgerFlush:
|
||||
}
|
||||
// A close of deletionQuit may be concurrent with this wake; the
|
||||
// final snapshot in Shutdown covers whatever the loop left dirty.
|
||||
select {
|
||||
case <-f.deletionQuit:
|
||||
return
|
||||
default:
|
||||
}
|
||||
f.snapshotDeletionLedger()
|
||||
}
|
||||
}()
|
||||
}
|
||||
|
||||
// signalLedgerFlush wakes the snapshotter without blocking the caller.
|
||||
func (f *Filer) signalLedgerFlush() {
|
||||
f.deletionLedgerLock.Lock()
|
||||
if f.deletionLedgerFlush == nil {
|
||||
f.deletionLedgerFlush = make(chan struct{}, 1)
|
||||
}
|
||||
ch := f.deletionLedgerFlush
|
||||
f.deletionLedgerLock.Unlock()
|
||||
select {
|
||||
case ch <- struct{}{}:
|
||||
default:
|
||||
}
|
||||
}
|
||||
|
||||
// queueDeletions is the single entry point for adding fileIds to the deletion
|
||||
// pipeline. It keeps the in-memory hot queue AND the durable ledger in sync.
|
||||
// Safe on a zero-value Filer (tests build struct literals without NewFiler):
|
||||
// the map is lazily created and the mutex is zero-value friendly.
|
||||
func (f *Filer) queueDeletions(fileIds ...string) {
|
||||
if len(fileIds) == 0 {
|
||||
return
|
||||
}
|
||||
// Hot path: existing in-memory queue, unchanged.
|
||||
f.FileIdDeletionQueue.EnQueue(fileIds...)
|
||||
|
||||
// Durable ledger: record what still needs deleting. Every enqueue bumps the
|
||||
// id's epoch so an expiry-forget carrying an older epoch cannot erase the
|
||||
// re-queued intent.
|
||||
f.deletionLedgerLock.Lock()
|
||||
if f.pendingDeletions == nil {
|
||||
f.pendingDeletions = make(map[string]uint64, len(fileIds))
|
||||
}
|
||||
for _, id := range fileIds {
|
||||
if id == "" {
|
||||
continue
|
||||
}
|
||||
_, exists := f.pendingDeletions[id]
|
||||
f.deletionSeq++
|
||||
f.pendingDeletions[id] = f.deletionSeq
|
||||
if !exists {
|
||||
f.deletionLedgerDirty = true
|
||||
}
|
||||
}
|
||||
f.deletionLedgerLock.Unlock()
|
||||
f.signalLedgerFlush()
|
||||
}
|
||||
|
||||
// pendingDeletionCount returns the number of fileIds still tracked as needing
|
||||
// deletion in the durable ledger. Handy for diagnostics and tests; 0 when the
|
||||
// map was never initialised.
|
||||
func (f *Filer) pendingDeletionCount() int {
|
||||
f.deletionLedgerLock.Lock()
|
||||
defer f.deletionLedgerLock.Unlock()
|
||||
return len(f.pendingDeletions)
|
||||
}
|
||||
|
||||
// deletionEpoch returns the pending record's current epoch for fileId.
|
||||
func (f *Filer) deletionEpoch(fileId string) uint64 {
|
||||
f.deletionLedgerLock.Lock()
|
||||
defer f.deletionLedgerLock.Unlock()
|
||||
return f.pendingDeletions[fileId]
|
||||
}
|
||||
|
||||
// forgetDeletion removes a fileId from the ledger once its deletion is terminal
|
||||
// (deleted or already absent: the chunks are gone, so any pending record — even
|
||||
// a re-queued one — refers to work that no longer exists). Non-terminal
|
||||
// outcomes must NOT call this so the entry survives until confirmed.
|
||||
func (f *Filer) forgetDeletion(fileId string) {
|
||||
if fileId == "" {
|
||||
return
|
||||
}
|
||||
f.deletionLedgerLock.Lock()
|
||||
if _, exists := f.pendingDeletions[fileId]; exists {
|
||||
delete(f.pendingDeletions, fileId)
|
||||
f.deletionLedgerDirty = true
|
||||
}
|
||||
f.deletionLedgerLock.Unlock()
|
||||
f.signalLedgerFlush()
|
||||
}
|
||||
|
||||
// forgetDeletionEpoch removes the ledger record only when its epoch still
|
||||
// matches the one captured when the deletion attempt began. A mismatch means
|
||||
// the id was re-queued meanwhile, so the newer record wins. Reports whether
|
||||
// the record was removed.
|
||||
func (f *Filer) forgetDeletionEpoch(fileId string, epoch uint64) bool {
|
||||
if fileId == "" {
|
||||
return false
|
||||
}
|
||||
removed := false
|
||||
f.deletionLedgerLock.Lock()
|
||||
if cur, exists := f.pendingDeletions[fileId]; exists && cur == epoch {
|
||||
delete(f.pendingDeletions, fileId)
|
||||
f.deletionLedgerDirty = true
|
||||
removed = true
|
||||
}
|
||||
f.deletionLedgerLock.Unlock()
|
||||
if removed {
|
||||
f.signalLedgerFlush()
|
||||
}
|
||||
return removed
|
||||
}
|
||||
|
||||
// snapshotDeletionLedger serialises the current pending set to the store.
|
||||
// Only writes when something changed since the last snapshot to keep KV churn low.
|
||||
// Writes serialize on deletionSnapshotLock so a snapshot in flight when Shutdown
|
||||
// starts cannot overwrite the final one with an older copy.
|
||||
func (f *Filer) snapshotDeletionLedger() {
|
||||
if !deletionPersistEnabled() || f.Store == nil {
|
||||
return
|
||||
}
|
||||
f.deletionSnapshotLock.Lock()
|
||||
defer f.deletionSnapshotLock.Unlock()
|
||||
|
||||
if f.deletionLedgerBlocked.Load() && !f.tryUnblockDeletionLedger() {
|
||||
return
|
||||
}
|
||||
|
||||
f.deletionLedgerLock.Lock()
|
||||
if !f.deletionLedgerDirty {
|
||||
hasStale := len(f.deletionLedgerStale) > 0
|
||||
f.deletionLedgerLock.Unlock()
|
||||
if !hasStale {
|
||||
return
|
||||
}
|
||||
// Nothing to republish, but orphaned part keys still need cleanup.
|
||||
ctx := context.Background()
|
||||
f.flushStaleLedgerParts(ctx)
|
||||
f.persistStaleLedgerParts(ctx, f.deletionLedgerKey())
|
||||
return
|
||||
}
|
||||
// Take a stable copy under lock; marshal + KV write happen outside it.
|
||||
ids := make([]string, 0, len(f.pendingDeletions))
|
||||
for id := range f.pendingDeletions {
|
||||
ids = append(ids, id)
|
||||
}
|
||||
f.deletionLedgerDirty = false
|
||||
f.deletionLedgerLock.Unlock()
|
||||
|
||||
if err := f.writeDeletionLedger(ids); err != nil {
|
||||
glog.Warningf("failed to persist deletion ledger (%d ids): %v", len(ids), err)
|
||||
f.deletionLedgerLock.Lock()
|
||||
f.deletionLedgerDirty = true
|
||||
f.deletionLedgerLock.Unlock()
|
||||
return
|
||||
}
|
||||
glog.V(3).Infof("persisted deletion ledger: %d pending deletions", len(ids))
|
||||
}
|
||||
|
||||
// tryUnblockDeletionLedger retries the ledger reload that originally failed.
|
||||
// Once the read succeeds the persisted ids merge back into the pending set and
|
||||
// snapshots resume; until then nothing is written so the unread ledger cannot
|
||||
// be overwritten.
|
||||
func (f *Filer) tryUnblockDeletionLedger() bool {
|
||||
f.reloadDeletionLedger()
|
||||
return !f.deletionLedgerBlocked.Load()
|
||||
}
|
||||
|
||||
// writeDeletionLedger persists the id set. A single value is published
|
||||
// atomically; a multipart set writes a new generation's part keys first and
|
||||
// only then the manifest, so a crash or failed write leaves the previously
|
||||
// published generation readable. Orphaned parts are retried on the next write.
|
||||
func (f *Filer) writeDeletionLedger(ids []string) error {
|
||||
ctx := context.Background()
|
||||
key := f.deletionLedgerKey()
|
||||
|
||||
f.deletionLedgerLock.Lock()
|
||||
prevGen, prevParts := f.deletionLedgerGen, f.deletionLedgerParts
|
||||
f.deletionLedgerLock.Unlock()
|
||||
|
||||
f.flushStaleLedgerParts(ctx)
|
||||
|
||||
var parts [][]byte
|
||||
var batch []string
|
||||
size := 2 // "[]"
|
||||
flush := func() error {
|
||||
if batch == nil {
|
||||
batch = []string{}
|
||||
}
|
||||
payload, err := json.Marshal(batch)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
parts = append(parts, payload)
|
||||
batch = nil
|
||||
size = 2
|
||||
return nil
|
||||
}
|
||||
for _, id := range ids {
|
||||
if need := len(id) + 3; len(batch) > 0 && size+need > deletionLedgerPartSize {
|
||||
if err := flush(); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
batch = append(batch, id)
|
||||
size += len(id) + 3
|
||||
}
|
||||
if len(batch) > 0 || len(parts) == 0 {
|
||||
if err := flush(); err != nil {
|
||||
return err
|
||||
}
|
||||
}
|
||||
|
||||
if len(parts) == 1 {
|
||||
if err := f.Store.KvPut(ctx, []byte(key), parts[0]); err != nil {
|
||||
return err
|
||||
}
|
||||
f.deleteLedgerParts(ctx, key, prevGen, prevParts)
|
||||
f.deletionLedgerLock.Lock()
|
||||
f.deletionLedgerGen, f.deletionLedgerParts = 0, 0
|
||||
f.deletionLedgerLock.Unlock()
|
||||
} else {
|
||||
gen := prevGen + 1
|
||||
wrote := 0
|
||||
for i, payload := range parts {
|
||||
if err := f.Store.KvPut(ctx, deletionLedgerGenPartKey(key, gen, i), payload); err != nil {
|
||||
f.markStaleLedgerParts(key, gen, wrote)
|
||||
return err
|
||||
}
|
||||
wrote++
|
||||
}
|
||||
manifest, _ := json.Marshal(struct {
|
||||
Parts int `json:"parts"`
|
||||
Gen int `json:"gen,omitempty"`
|
||||
}{len(parts), gen})
|
||||
if err := f.Store.KvPut(ctx, []byte(key), manifest); err != nil {
|
||||
f.markStaleLedgerParts(key, gen, wrote)
|
||||
return err
|
||||
}
|
||||
f.deleteLedgerParts(ctx, key, prevGen, prevParts)
|
||||
f.deletionLedgerLock.Lock()
|
||||
f.deletionLedgerGen, f.deletionLedgerParts = gen, len(parts)
|
||||
f.deletionLedgerLock.Unlock()
|
||||
}
|
||||
f.persistStaleLedgerParts(ctx, key)
|
||||
f.touchLedgerIndex(ctx, key)
|
||||
return nil
|
||||
}
|
||||
|
||||
// deleteLedgerParts removes a published generation's part keys; failures are
|
||||
// tracked as stale so the next snapshot retries them.
|
||||
func (f *Filer) deleteLedgerParts(ctx context.Context, key string, gen, parts int) {
|
||||
for i := 0; i < parts; i++ {
|
||||
if err := f.Store.KvDelete(ctx, deletionLedgerGenPartKey(key, gen, i)); err != nil {
|
||||
f.deletionLedgerLock.Lock()
|
||||
f.deletionLedgerStale = append(f.deletionLedgerStale, string(deletionLedgerGenPartKey(key, gen, i)))
|
||||
f.deletionLedgerLock.Unlock()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// markStaleLedgerParts remembers parts of an abandoned generation so they are
|
||||
// deleted by a later snapshot instead of leaking store space.
|
||||
func (f *Filer) markStaleLedgerParts(key string, gen, wrote int) {
|
||||
if wrote == 0 {
|
||||
return
|
||||
}
|
||||
f.deletionLedgerLock.Lock()
|
||||
for i := 0; i < wrote; i++ {
|
||||
f.deletionLedgerStale = append(f.deletionLedgerStale, string(deletionLedgerGenPartKey(key, gen, i)))
|
||||
}
|
||||
f.deletionLedgerLock.Unlock()
|
||||
}
|
||||
|
||||
func (f *Filer) flushStaleLedgerParts(ctx context.Context) {
|
||||
f.deletionLedgerLock.Lock()
|
||||
stale := f.deletionLedgerStale
|
||||
f.deletionLedgerStale = nil
|
||||
f.deletionLedgerLock.Unlock()
|
||||
|
||||
var keep []string
|
||||
for _, k := range stale {
|
||||
if err := f.Store.KvDelete(ctx, []byte(k)); err != nil {
|
||||
keep = append(keep, k)
|
||||
}
|
||||
}
|
||||
if len(keep) > 0 {
|
||||
f.deletionLedgerLock.Lock()
|
||||
f.deletionLedgerStale = append(keep, f.deletionLedgerStale...)
|
||||
f.deletionLedgerLock.Unlock()
|
||||
}
|
||||
}
|
||||
|
||||
// persistStaleLedgerParts mirrors the stale-part list under a sidecar key so a
|
||||
// crash between manifest publication and part cleanup can retry after restart
|
||||
// instead of orphaning the part keys.
|
||||
func (f *Filer) persistStaleLedgerParts(ctx context.Context, key string) {
|
||||
f.deletionLedgerLock.Lock()
|
||||
stale := append([]string(nil), f.deletionLedgerStale...)
|
||||
f.deletionLedgerLock.Unlock()
|
||||
staleKey := []byte(key + ".stale")
|
||||
if len(stale) == 0 {
|
||||
_ = f.Store.KvDelete(ctx, staleKey)
|
||||
return
|
||||
}
|
||||
payload, _ := json.Marshal(stale)
|
||||
_ = f.Store.KvPut(ctx, staleKey, payload)
|
||||
}
|
||||
|
||||
// loadStaleLedgerParts seeds the stale-part list left by a previous run.
|
||||
func (f *Filer) loadStaleLedgerParts(ctx context.Context, key string) {
|
||||
payload, err := f.Store.KvGet(ctx, []byte(key+".stale"))
|
||||
if err != nil {
|
||||
return
|
||||
}
|
||||
var stale []string
|
||||
if json.Unmarshal(payload, &stale) != nil {
|
||||
return
|
||||
}
|
||||
f.deletionLedgerLock.Lock()
|
||||
f.deletionLedgerStale = append(f.deletionLedgerStale, stale...)
|
||||
f.deletionLedgerLock.Unlock()
|
||||
}
|
||||
|
||||
// touchLedgerIndex records this filer's scoped key in the shared index so the
|
||||
// ledger remains discoverable if the filer restarts under a new address. The
|
||||
// index is a read-modify-write list with no CAS, so the write is verified and
|
||||
// retried: a peer's concurrent update must not drop this key.
|
||||
func (f *Filer) touchLedgerIndex(ctx context.Context, key string) {
|
||||
if key == KvKeyDeletionLedger {
|
||||
return
|
||||
}
|
||||
for attempt := 0; attempt < 3; attempt++ {
|
||||
keys, err := f.readLedgerIndex(ctx)
|
||||
if err != nil {
|
||||
// Only a genuinely absent index means "empty"; a store error must
|
||||
// not let us write a one-key list that drops every peer entry.
|
||||
if err != ErrKvNotFound {
|
||||
glog.V(1).Infof("deletion ledger index unreadable, skipping update: %v", err)
|
||||
return
|
||||
}
|
||||
keys = nil
|
||||
}
|
||||
found := false
|
||||
for _, k := range keys {
|
||||
if k == key {
|
||||
found = true
|
||||
break
|
||||
}
|
||||
}
|
||||
if found {
|
||||
return
|
||||
}
|
||||
payload, _ := json.Marshal(append(keys, key))
|
||||
if err := f.Store.KvPut(ctx, []byte(KvKeyDeletionLedgerIndex), payload); err != nil {
|
||||
glog.V(1).Infof("failed to update deletion ledger index: %v", err)
|
||||
return
|
||||
}
|
||||
}
|
||||
glog.V(1).Infof("deletion ledger index lost %q to a concurrent update; retrying on the next snapshot", key)
|
||||
}
|
||||
|
||||
func (f *Filer) readLedgerIndex(ctx context.Context) ([]string, error) {
|
||||
payload, err := f.Store.KvGet(ctx, []byte(KvKeyDeletionLedgerIndex))
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
var keys []string
|
||||
if err := json.Unmarshal(payload, &keys); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return keys, nil
|
||||
}
|
||||
|
||||
func (f *Filer) pruneLedgerIndex(ctx context.Context, key string) {
|
||||
keys, err := f.readLedgerIndex(ctx)
|
||||
if err != nil {
|
||||
return
|
||||
}
|
||||
kept := keys[:0]
|
||||
for _, k := range keys {
|
||||
if k != key {
|
||||
kept = append(kept, k)
|
||||
}
|
||||
}
|
||||
payload, _ := json.Marshal(kept)
|
||||
_ = f.Store.KvPut(ctx, []byte(KvKeyDeletionLedgerIndex), payload)
|
||||
}
|
||||
|
||||
// readDeletionLedger reads the manifest key: a JSON array is the whole set; a
|
||||
// {"parts":N,"gen":G} manifest points at that generation's part keys. A part
|
||||
// the manifest references but the store lacks is corruption, not absence, so
|
||||
// it surfaces as a wrapped error rather than ErrKvNotFound.
|
||||
func (f *Filer) readDeletionLedger(key string) (ids []string, gen, parts int, err error) {
|
||||
ctx := context.Background()
|
||||
|
||||
payload, err := f.Store.KvGet(ctx, []byte(key))
|
||||
if err != nil {
|
||||
return nil, 0, 0, err
|
||||
}
|
||||
if !bytes.HasPrefix(bytes.TrimSpace(payload), []byte("{")) {
|
||||
if err := json.Unmarshal(payload, &ids); err != nil {
|
||||
return nil, 0, 0, err
|
||||
}
|
||||
return ids, 0, 0, nil
|
||||
}
|
||||
var manifest struct {
|
||||
Parts int `json:"parts"`
|
||||
Gen int `json:"gen,omitempty"`
|
||||
}
|
||||
if err := json.Unmarshal(payload, &manifest); err != nil {
|
||||
return nil, 0, 0, err
|
||||
}
|
||||
for i := 0; i < manifest.Parts; i++ {
|
||||
partPayload, err := f.Store.KvGet(ctx, deletionLedgerGenPartKey(key, manifest.Gen, i))
|
||||
if err != nil {
|
||||
return nil, 0, 0, fmt.Errorf("deletion ledger part %d of %s unreadable: %w", i, key, err)
|
||||
}
|
||||
var part []string
|
||||
if err := json.Unmarshal(partPayload, &part); err != nil {
|
||||
return nil, 0, 0, fmt.Errorf("deletion ledger part %d of %s corrupt: %w", i, key, err)
|
||||
}
|
||||
ids = append(ids, part...)
|
||||
}
|
||||
return ids, manifest.Gen, manifest.Parts, nil
|
||||
}
|
||||
|
||||
// recoverForeignDeletionLedgers runs when the scoped key is absent: the ledger
|
||||
// may sit under the pre-scoping base key, or under a scoped key belonging to an
|
||||
// earlier incarnation of this filer whose advertised address changed. Every
|
||||
// discovered set is published under this filer's key first and the source keys
|
||||
// deleted only after that write succeeds. A live peer's ledger claimed here is
|
||||
// rewritten by the peer's next snapshot, so the ids only get processed twice —
|
||||
// deletions are not owner-specific.
|
||||
func (f *Filer) recoverForeignDeletionLedgers(key string) (ids []string, err error) {
|
||||
ctx := context.Background()
|
||||
|
||||
claimed := map[string]ledgerClaim{} // source key -> what was read
|
||||
lIds, lGen, lParts, lErr := f.readDeletionLedger(KvKeyDeletionLedger)
|
||||
switch {
|
||||
case lErr == nil:
|
||||
ids = append(ids, lIds...)
|
||||
claimed[KvKeyDeletionLedger] = ledgerClaim{lIds, lGen, lParts}
|
||||
f.loadStaleLedgerParts(ctx, KvKeyDeletionLedger)
|
||||
case lErr != ErrKvNotFound:
|
||||
return nil, lErr
|
||||
}
|
||||
|
||||
// Any read failure aborts the whole claim so reload retries: skipping an
|
||||
// unreadable source would strand it, since a successful claim here means
|
||||
// the index is never searched again.
|
||||
indexKeys, idxErr := f.readLedgerIndex(ctx)
|
||||
if idxErr != nil && idxErr != ErrKvNotFound {
|
||||
return nil, idxErr
|
||||
}
|
||||
for _, other := range indexKeys {
|
||||
if other == key {
|
||||
continue
|
||||
}
|
||||
oIds, oGen, oParts, oErr := f.readDeletionLedger(other)
|
||||
if oErr == ErrKvNotFound {
|
||||
f.pruneLedgerIndex(ctx, other)
|
||||
continue
|
||||
}
|
||||
if oErr != nil {
|
||||
return nil, fmt.Errorf("deletion ledger %s unreadable: %w", other, oErr)
|
||||
}
|
||||
ids = append(ids, oIds...)
|
||||
claimed[other] = ledgerClaim{oIds, oGen, oParts}
|
||||
f.loadStaleLedgerParts(ctx, other)
|
||||
}
|
||||
|
||||
if len(claimed) == 0 {
|
||||
return nil, ErrKvNotFound
|
||||
}
|
||||
if err := f.writeDeletionLedger(ids); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
mergedMore := false
|
||||
for src, claim := range claimed {
|
||||
// A source republished since we read it belongs to a live filer
|
||||
// (or a racing claimer): merge the newer ids and leave it in place.
|
||||
curIds, curGen, curParts, rerr := f.readDeletionLedger(src)
|
||||
if rerr != nil || curGen != claim.gen || curParts != claim.parts || !sameStringSet(curIds, claim.ids) {
|
||||
if rerr == nil {
|
||||
for _, id := range curIds {
|
||||
if !containsId(ids, id) {
|
||||
ids = append(ids, id)
|
||||
mergedMore = true
|
||||
}
|
||||
}
|
||||
}
|
||||
glog.V(0).Infof("deletion ledger %s changed while claiming; leaving it for its owner", src)
|
||||
continue
|
||||
}
|
||||
f.deleteLedgerParts(ctx, src, claim.gen, claim.parts)
|
||||
_ = f.Store.KvDelete(ctx, []byte(src))
|
||||
_ = f.Store.KvDelete(ctx, []byte(src+".stale"))
|
||||
f.pruneLedgerIndex(ctx, src)
|
||||
}
|
||||
if mergedMore {
|
||||
// Ids from a changed source are durable only under that source's key;
|
||||
// rewrite our ledger so they survive under ours too.
|
||||
if err := f.writeDeletionLedger(ids); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
}
|
||||
return ids, nil
|
||||
}
|
||||
|
||||
func containsId(ids []string, id string) bool {
|
||||
for _, s := range ids {
|
||||
if s == id {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
type ledgerClaim struct {
|
||||
ids []string
|
||||
gen int
|
||||
parts int
|
||||
}
|
||||
|
||||
func sameStringSet(a, b []string) bool {
|
||||
if len(a) != len(b) {
|
||||
return false
|
||||
}
|
||||
seen := make(map[string]struct{}, len(a))
|
||||
for _, s := range a {
|
||||
seen[s] = struct{}{}
|
||||
}
|
||||
for _, s := range b {
|
||||
if _, ok := seen[s]; !ok {
|
||||
return false
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// reloadDeletionLedger re-enqueues any pending deletions found in the store after
|
||||
// a restart, so a crash that killed the in-memory queues does not leak chunks.
|
||||
// Recovered ids join the pending set immediately so snapshots rewrite the full
|
||||
// ledger; only the re-queue waits out DeletionRecoveryGrace.
|
||||
//
|
||||
// It reports whether the ledger is usable. A read error or an unparseable
|
||||
// payload leaves the persisted set unknown, so snapshots retry the read rather
|
||||
// than overwrite the unread ledger with a partial set.
|
||||
//
|
||||
// Safe to call on a Filer with a nil Store (no-op). Idempotent for the volume
|
||||
// side: a chunk that was actually deleted before the crash re-deletes as not-found.
|
||||
func (f *Filer) reloadDeletionLedger() bool {
|
||||
if !deletionPersistEnabled() || f.Store == nil {
|
||||
return false
|
||||
}
|
||||
|
||||
key := f.deletionLedgerKey()
|
||||
ids, gen, parts, err := f.readDeletionLedger(key)
|
||||
stateWritten := false
|
||||
if err == ErrKvNotFound && key != KvKeyDeletionLedger {
|
||||
ids, err = f.recoverForeignDeletionLedgers(key)
|
||||
stateWritten = err == nil
|
||||
}
|
||||
if err != nil {
|
||||
if err == ErrKvNotFound {
|
||||
f.deletionLedgerBlocked.Store(false)
|
||||
return true
|
||||
}
|
||||
f.deletionLedgerBlocked.Store(true)
|
||||
glog.Warningf("failed to read persisted deletion ledger; persistence disabled until it reads: %v", err)
|
||||
return false
|
||||
}
|
||||
f.deletionLedgerBlocked.Store(false)
|
||||
f.loadStaleLedgerParts(context.Background(), key)
|
||||
|
||||
// Merge into the pending set now so an early snapshot rewrites the
|
||||
// recovered ids instead of overwriting the ledger with only new ones.
|
||||
f.deletionLedgerLock.Lock()
|
||||
if len(ids) > 0 {
|
||||
if f.pendingDeletions == nil {
|
||||
f.pendingDeletions = make(map[string]uint64, len(ids))
|
||||
}
|
||||
for _, id := range ids {
|
||||
if id != "" {
|
||||
if _, exists := f.pendingDeletions[id]; !exists {
|
||||
f.deletionSeq++
|
||||
f.pendingDeletions[id] = f.deletionSeq
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if !stateWritten {
|
||||
f.deletionLedgerGen = gen
|
||||
f.deletionLedgerParts = parts
|
||||
}
|
||||
f.deletionLedgerLock.Unlock()
|
||||
|
||||
if len(ids) == 0 {
|
||||
return true
|
||||
}
|
||||
|
||||
grace := deletionRecoveryGrace()
|
||||
glog.V(0).Infof("recovered %d pending deletions from ledger, applying in %v", len(ids), grace)
|
||||
|
||||
go func() {
|
||||
timer := time.NewTimer(grace)
|
||||
defer timer.Stop()
|
||||
select {
|
||||
case <-f.deletionQuit:
|
||||
return
|
||||
case <-timer.C:
|
||||
}
|
||||
// The ids are already in the pending set; only the queue push waits
|
||||
// for peer meta-aggregation to settle. The ledger is deliberately NOT
|
||||
// cleared here: it shrinks only as the delete pipeline confirms each
|
||||
// id terminal, so a second crash mid-recovery replays everything.
|
||||
f.queueDeletions(ids...)
|
||||
glog.V(0).Infof("re-queued %d recovered pending deletions", len(ids))
|
||||
}()
|
||||
return true
|
||||
}
|
||||
@@ -0,0 +1,584 @@
|
||||
package filer
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"errors"
|
||||
"fmt"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/cluster/lock_manager"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
)
|
||||
|
||||
// readPersistedLedger pulls the persisted ledger payload straight from the store
|
||||
// the filer is wired to, the same way a restarting filer would.
|
||||
func readPersistedLedger(t *testing.T, f *Filer) ([]string, bool) {
|
||||
t.Helper()
|
||||
raw, err := f.Store.KvGet(context.Background(), []byte(f.deletionLedgerKey()))
|
||||
if err != nil {
|
||||
if err == ErrKvNotFound {
|
||||
return nil, false
|
||||
}
|
||||
t.Fatalf("unexpected error reading ledger: %v", err)
|
||||
}
|
||||
var ids []string
|
||||
if err := json.Unmarshal(raw, &ids); err != nil {
|
||||
t.Fatalf("persisted payload not valid JSON: %v", err)
|
||||
}
|
||||
return ids, true
|
||||
}
|
||||
|
||||
// withPersistedConfig pins the ledger knobs at viper's override tier so the test
|
||||
// sees deterministic semantics regardless of global SetDefault ordering (the
|
||||
// production getters SetDefault(true), which would otherwise win the tier race).
|
||||
// The overrides are restored on cleanup so later tests see production defaults.
|
||||
func withPersistedConfig(t *testing.T, enabled bool) {
|
||||
t.Helper()
|
||||
v := util.GetViper()
|
||||
v.Set("filer.deleteQueue.persist", enabled)
|
||||
v.Set("filer.deleteQueue.recoveryGrace", defaultDeletionRecoveryGrace)
|
||||
t.Cleanup(func() {
|
||||
v.Set("filer.deleteQueue.persist", true)
|
||||
v.Set("filer.deleteQueue.recoveryGrace", defaultDeletionRecoveryGrace)
|
||||
})
|
||||
}
|
||||
|
||||
// newLedgerTestFiler builds a Filer with the pieces the ledger touches, backed by
|
||||
// a real (stub) store, bypassing NewFiler's master/aggregator wiring.
|
||||
func newLedgerTestFiler(store FilerStore) *Filer {
|
||||
return &Filer{
|
||||
FileIdDeletionQueue: util.NewUnboundedQueue(),
|
||||
DeletionRetryQueue: NewDeletionRetryQueue(),
|
||||
deletionQuit: make(chan struct{}),
|
||||
Store: NewFilerStoreWrapper(store),
|
||||
}
|
||||
}
|
||||
|
||||
// TestDeletionLedgerSnapshotAndRecover is the core guarantee: enqueued-but-
|
||||
// unconfirmed deletions survive a simulated process restart and land back in the
|
||||
// hot queue (minus whatever was confirmed terminal in the meantime).
|
||||
func TestDeletionLedgerSnapshotAndRecover(t *testing.T) {
|
||||
withPersistedConfig(t, true)
|
||||
store := newStubFilerStore()
|
||||
f := newLedgerTestFiler(store)
|
||||
|
||||
f.queueDeletions("1,01", "1,02", "1,03")
|
||||
f.snapshotDeletionLedger()
|
||||
|
||||
persisted, ok := readPersistedLedger(t, f)
|
||||
if !ok {
|
||||
t.Fatalf("expected persisted ledger under %q", KvKeyDeletionLedger)
|
||||
}
|
||||
if len(persisted) != 3 {
|
||||
t.Fatalf("expected 3 persisted ids, got %d: %v", len(persisted), persisted)
|
||||
}
|
||||
for _, want := range []string{"1,01", "1,02", "1,03"} {
|
||||
if !contains(persisted, want) {
|
||||
t.Errorf("expected %q in persisted ledger, got %v", want, persisted)
|
||||
}
|
||||
}
|
||||
|
||||
// Confirmed-gone for one id, then snapshot again.
|
||||
f.forgetDeletion("1,02")
|
||||
f.snapshotDeletionLedger()
|
||||
persisted, _ = readPersistedLedger(t, f)
|
||||
if len(persisted) != 2 {
|
||||
t.Fatalf("expected 2 persisted ids after forget, got %d: %v", len(persisted), persisted)
|
||||
}
|
||||
if contains(persisted, "1,02") {
|
||||
t.Errorf("forgotten id still in persisted ledger: %v", persisted)
|
||||
}
|
||||
|
||||
// --- simulated restart: a fresh Filer with NO ledger in memory must
|
||||
// recover the still-pending ids from the store and re-queue them. ---
|
||||
f2 := newLedgerTestFiler(store)
|
||||
// recoveryGrace 0 so the background re-enqueue is near-instant.
|
||||
util.GetViper().Set("filer.deleteQueue.recoveryGrace", time.Duration(0))
|
||||
f2.reloadDeletionLedger()
|
||||
|
||||
// Wait for the async recovery goroutine to drain into the hot queue.
|
||||
seen := make(map[string]bool)
|
||||
deadline := time.Now().Add(2 * time.Second)
|
||||
for time.Now().Before(deadline) {
|
||||
f2.FileIdDeletionQueue.Consume(func(ids []string) {
|
||||
for _, id := range ids {
|
||||
seen[id] = true
|
||||
}
|
||||
})
|
||||
if len(seen) >= 2 {
|
||||
break
|
||||
}
|
||||
time.Sleep(10 * time.Millisecond)
|
||||
}
|
||||
|
||||
if !seen["1,01"] || !seen["1,03"] {
|
||||
t.Fatalf("expected recovered ids 1,01 and 1,03 back in hot queue, got %v", seen)
|
||||
}
|
||||
if seen["1,02"] {
|
||||
t.Errorf("forgotten id 1,02 must NOT come back: %v", seen)
|
||||
}
|
||||
}
|
||||
|
||||
// TestDeletionLedgerRetryKeepsEntry pins the key semantic: a non-terminal
|
||||
// (retryable) failure must keep the entry persisted, so a crash mid-retry does
|
||||
// not orphan the chunk. This is the whole reason forgetDeletion is only called on
|
||||
// success / not-found / permanent.
|
||||
func TestDeletionLedgerRetryKeepsEntry(t *testing.T) {
|
||||
withPersistedConfig(t, true)
|
||||
store := newStubFilerStore()
|
||||
f := newLedgerTestFiler(store)
|
||||
|
||||
f.queueDeletions("9,01")
|
||||
f.snapshotDeletionLedger()
|
||||
|
||||
// A retryable outcome deliberately does NOT call forgetDeletion, so the id
|
||||
// must still be in the durable ledger.
|
||||
persisted, ok := readPersistedLedger(t, f)
|
||||
if !ok {
|
||||
t.Fatalf("ledger missing — retryable ids must remain persisted")
|
||||
}
|
||||
if len(persisted) != 1 || persisted[0] != "9,01" {
|
||||
t.Fatalf("retryable id must stay in ledger, got %v", persisted)
|
||||
}
|
||||
}
|
||||
|
||||
// TestDeletionLedgerDisabled verifies the kill switch: with persist off, nothing
|
||||
// reaches KV, and reload refuses to recover even if a ledger exists in the store
|
||||
// (so an operator turning the feature off never resurrects a stale ledger).
|
||||
func TestDeletionLedgerDisabled(t *testing.T) {
|
||||
// First, with persistence ON, put a real ledger into the store.
|
||||
withPersistedConfig(t, true)
|
||||
store := newStubFilerStore()
|
||||
fOn := newLedgerTestFiler(store)
|
||||
fOn.queueDeletions("5,01")
|
||||
fOn.snapshotDeletionLedger()
|
||||
if _, ok := readPersistedLedger(t, fOn); !ok {
|
||||
t.Fatalf("setup: expected a persisted ledger with persist enabled")
|
||||
}
|
||||
|
||||
// Now flip the switch OFF and check the contract.
|
||||
withPersistedConfig(t, false)
|
||||
|
||||
// 1. Snapshot does not touch KV (no new writes, no clear).
|
||||
f := newLedgerTestFiler(store)
|
||||
f.queueDeletions("6,01") // in-memory only
|
||||
f.snapshotDeletionLedger()
|
||||
persisted, ok := readPersistedLedger(t, f)
|
||||
if !ok {
|
||||
t.Fatalf("setup: ledger should still exist from the enabled phase")
|
||||
}
|
||||
// The enabled-phase ledger had "5,01"; disabled phase must not have added "6,01".
|
||||
if contains(persisted, "6,01") {
|
||||
t.Errorf("persist disabled but snapshot wrote 6,01: %v", persisted)
|
||||
}
|
||||
|
||||
// 2. Reload with the switch off must NOT recover, even though a ledger exists.
|
||||
f2 := newLedgerTestFiler(store)
|
||||
f2.reloadDeletionLedger()
|
||||
if f2.pendingDeletionCount() != 0 {
|
||||
t.Fatalf("reload must be a no-op when disabled, got %d pending", f2.pendingDeletionCount())
|
||||
}
|
||||
}
|
||||
|
||||
// Filers sharing one metadata store must not overwrite each other's ledgers:
|
||||
// each filer keys its ledger by its own address.
|
||||
func TestDeletionLedgerScopedPerFiler(t *testing.T) {
|
||||
withPersistedConfig(t, true)
|
||||
store := newStubFilerStore()
|
||||
fA := newLedgerTestFiler(store)
|
||||
fA.Dlm = lock_manager.NewDistributedLockManager("filer-a:8888")
|
||||
fB := newLedgerTestFiler(store)
|
||||
fB.Dlm = lock_manager.NewDistributedLockManager("filer-b:8888")
|
||||
|
||||
fA.queueDeletions("1,01")
|
||||
fB.queueDeletions("2,02")
|
||||
fA.snapshotDeletionLedger()
|
||||
fB.snapshotDeletionLedger()
|
||||
|
||||
idsA, okA := readPersistedLedger(t, fA)
|
||||
idsB, okB := readPersistedLedger(t, fB)
|
||||
if !okA || !okB {
|
||||
t.Fatalf("both filers must have their own ledger, got %v %v", idsA, idsB)
|
||||
}
|
||||
if !contains(idsA, "1,01") || contains(idsA, "2,02") {
|
||||
t.Fatalf("filer A ledger wrong: %v", idsA)
|
||||
}
|
||||
if !contains(idsB, "2,02") || contains(idsB, "1,01") {
|
||||
t.Fatalf("filer B ledger wrong: %v", idsB)
|
||||
}
|
||||
|
||||
// A restarted filer on B's address recovers only B's pending deletions.
|
||||
fB2 := newLedgerTestFiler(store)
|
||||
fB2.Dlm = lock_manager.NewDistributedLockManager("filer-b:8888")
|
||||
if !fB2.reloadDeletionLedger() {
|
||||
t.Fatalf("reload should succeed")
|
||||
}
|
||||
if fB2.pendingDeletionCount() != 1 {
|
||||
t.Fatalf("B's restart should recover exactly its own id, got %d", fB2.pendingDeletionCount())
|
||||
}
|
||||
}
|
||||
|
||||
// A ledger bigger than one store value must still persist: it splits into part
|
||||
// keys under a manifest, and shrinking below the part size removes the parts.
|
||||
func TestDeletionLedgerChunked(t *testing.T) {
|
||||
withPersistedConfig(t, true)
|
||||
store := newStubFilerStore()
|
||||
f := newLedgerTestFiler(store)
|
||||
|
||||
var ids []string
|
||||
for i := 0; i < 4000; i++ {
|
||||
ids = append(ids, fmt.Sprintf("7,%08xbeefcafe", i))
|
||||
}
|
||||
f.queueDeletions(ids...)
|
||||
f.snapshotDeletionLedger()
|
||||
|
||||
raw, err := f.Store.KvGet(context.Background(), []byte(f.deletionLedgerKey()))
|
||||
if err != nil || len(raw) == 0 || raw[0] != '{' {
|
||||
t.Fatalf("expected a chunked manifest, got %v %q", err, raw)
|
||||
}
|
||||
|
||||
f2 := newLedgerTestFiler(store)
|
||||
if !f2.reloadDeletionLedger() {
|
||||
t.Fatalf("chunked ledger should reload")
|
||||
}
|
||||
if got := f2.pendingDeletionCount(); got != len(ids) {
|
||||
t.Fatalf("recovered %d ids, want %d", got, len(ids))
|
||||
}
|
||||
|
||||
// Forget everything; the next snapshot is a single value again and the
|
||||
// part keys are removed.
|
||||
for _, id := range ids {
|
||||
f2.forgetDeletion(id)
|
||||
}
|
||||
f2.snapshotDeletionLedger()
|
||||
raw, err = f.Store.KvGet(context.Background(), []byte(f2.deletionLedgerKey()))
|
||||
if err != nil || len(raw) == 0 || raw[0] != '[' {
|
||||
t.Fatalf("expected single-value ledger after shrink, got %v %q", err, raw)
|
||||
}
|
||||
if _, err := f.Store.KvGet(context.Background(), deletionLedgerPartKey(f2.deletionLedgerKey(), 0)); err != ErrKvNotFound {
|
||||
t.Fatalf("stale part key must be removed, got %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// A startup ledger read that fails for a reason other than not-found must not
|
||||
// let snapshots overwrite the unread ledger with a partial set.
|
||||
func TestDeletionLedgerReadFailureBlocksPersistence(t *testing.T) {
|
||||
withPersistedConfig(t, true)
|
||||
store := newStubFilerStore()
|
||||
f := newLedgerTestFiler(store)
|
||||
store.kvGetErr = errors.New("kv read down")
|
||||
|
||||
if f.reloadDeletionLedger() {
|
||||
t.Fatalf("reload should report the ledger as unusable")
|
||||
}
|
||||
|
||||
f.queueDeletions("1,01")
|
||||
f.snapshotDeletionLedger()
|
||||
if len(store.kv) != 0 {
|
||||
t.Fatalf("snapshot must not overwrite a ledger that was never read, wrote %v", store.kv)
|
||||
}
|
||||
|
||||
// The block is not permanent: once the store reads again, persistence
|
||||
// resumes — a transient failure must not disable the ledger for the life
|
||||
// of the process.
|
||||
store.kvGetErr = nil
|
||||
f.snapshotDeletionLedger()
|
||||
persisted, ok := readPersistedLedger(t, f)
|
||||
if !ok || len(persisted) != 1 || persisted[0] != "1,01" {
|
||||
t.Fatalf("persistence should resume once the ledger reads, got %v", persisted)
|
||||
}
|
||||
}
|
||||
|
||||
// Recovered ids join the pending set immediately — before the grace delay — so
|
||||
// an early snapshot rewrites the recovered ids instead of dropping them.
|
||||
func TestDeletionLedgerRecoveryKeepsIdsPending(t *testing.T) {
|
||||
withPersistedConfig(t, true)
|
||||
store := newStubFilerStore()
|
||||
f := newLedgerTestFiler(store)
|
||||
f.queueDeletions("1,01", "1,02")
|
||||
f.snapshotDeletionLedger()
|
||||
|
||||
util.GetViper().Set("filer.deleteQueue.recoveryGrace", time.Hour)
|
||||
f2 := newLedgerTestFiler(store)
|
||||
if !f2.reloadDeletionLedger() {
|
||||
t.Fatalf("reload should succeed")
|
||||
}
|
||||
// Still inside the grace window, but the ids are already pending.
|
||||
if f2.pendingDeletionCount() != 2 {
|
||||
t.Fatalf("recovered ids must be pending immediately, got %d", f2.pendingDeletionCount())
|
||||
}
|
||||
// And an early shutdown snapshot keeps them.
|
||||
f2.snapshotDeletionLedger()
|
||||
persisted, ok := readPersistedLedger(t, f2)
|
||||
if !ok || len(persisted) != 2 {
|
||||
t.Fatalf("snapshot during grace must carry recovered ids, got %v", persisted)
|
||||
}
|
||||
}
|
||||
|
||||
// TestDeletionLedgerZeroValueFiler proves nil-map safety for the struct literals
|
||||
// used across the test suite (they never call NewFiler).
|
||||
func TestDeletionLedgerZeroValueFiler(t *testing.T) {
|
||||
withPersistedConfig(t, true)
|
||||
f := &Filer{
|
||||
FileIdDeletionQueue: util.NewUnboundedQueue(),
|
||||
DeletionRetryQueue: NewDeletionRetryQueue(),
|
||||
deletionQuit: make(chan struct{}),
|
||||
}
|
||||
f.queueDeletions("0,01")
|
||||
if f.pendingDeletionCount() != 1 {
|
||||
t.Fatalf("zero-value filer should lazily init the ledger, got %d", f.pendingDeletionCount())
|
||||
}
|
||||
f.forgetDeletion("0,01")
|
||||
if f.pendingDeletionCount() != 0 {
|
||||
t.Fatalf("forget on zero-value filer failed, got %d", f.pendingDeletionCount())
|
||||
}
|
||||
}
|
||||
|
||||
// A manifest that references a part the store cannot return is corruption, not
|
||||
// absence: reload must report failure and nothing may overwrite the surviving
|
||||
// manifest with only the in-memory set.
|
||||
func TestDeletionLedgerMissingPartBlocks(t *testing.T) {
|
||||
withPersistedConfig(t, true)
|
||||
store := newStubFilerStore()
|
||||
f := newLedgerTestFiler(store)
|
||||
|
||||
key := f.deletionLedgerKey()
|
||||
store.kv[key] = []byte(`{"parts":2,"gen":4}`)
|
||||
store.kv[string(deletionLedgerGenPartKey(key, 4, 0))] = []byte(`["1,01"]`)
|
||||
// Part 1 of generation 4 is missing.
|
||||
|
||||
if f.reloadDeletionLedger() {
|
||||
t.Fatalf("a manifest referencing a missing part is corrupt, not absent")
|
||||
}
|
||||
f.queueDeletions("9,99")
|
||||
f.snapshotDeletionLedger()
|
||||
if got := string(store.kv[key]); got != `{"parts":2,"gen":4}` {
|
||||
t.Fatalf("unread ledger must not be overwritten, got %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
// A filer restarting under a new advertised address must still find its
|
||||
// previous ledger: the index lists scoped keys, the ids are published under
|
||||
// the new key first, and only then is the stranded key removed.
|
||||
func TestDeletionLedgerAddressChangeClaim(t *testing.T) {
|
||||
withPersistedConfig(t, true)
|
||||
store := newStubFilerStore()
|
||||
fOld := newLedgerTestFiler(store)
|
||||
fOld.Dlm = lock_manager.NewDistributedLockManager("filer-old:8888")
|
||||
fOld.queueDeletions("1,01", "1,02")
|
||||
fOld.snapshotDeletionLedger()
|
||||
|
||||
fNew := newLedgerTestFiler(store)
|
||||
fNew.Dlm = lock_manager.NewDistributedLockManager("filer-new:9999")
|
||||
if !fNew.reloadDeletionLedger() {
|
||||
t.Fatalf("reload should claim the stranded ledger")
|
||||
}
|
||||
if fNew.pendingDeletionCount() != 2 {
|
||||
t.Fatalf("expected 2 claimed ids, got %d", fNew.pendingDeletionCount())
|
||||
}
|
||||
if _, err := store.KvGet(context.Background(), []byte(fOld.deletionLedgerKey())); err != ErrKvNotFound {
|
||||
t.Fatalf("stranded ledger must be removed after claim, got %v", err)
|
||||
}
|
||||
persisted, ok := readPersistedLedger(t, fNew)
|
||||
if !ok || len(persisted) != 2 {
|
||||
t.Fatalf("claimed ids must be durably stored under the new key, got %v", persisted)
|
||||
}
|
||||
indexKeys, _ := fNew.readLedgerIndex(context.Background())
|
||||
if contains(indexKeys, fOld.deletionLedgerKey()) {
|
||||
t.Fatalf("claimed key must leave the index, got %v", indexKeys)
|
||||
}
|
||||
}
|
||||
|
||||
// Every multipart snapshot writes a fresh generation's part keys and publishes
|
||||
// the manifest only after all parts land, so the committed generation is never
|
||||
// overwritten. The previous generation is removed after publication.
|
||||
func TestDeletionLedgerGenerationPublish(t *testing.T) {
|
||||
withPersistedConfig(t, true)
|
||||
store := newStubFilerStore()
|
||||
f := newLedgerTestFiler(store)
|
||||
|
||||
var ids []string
|
||||
for i := 0; i < 4000; i++ {
|
||||
ids = append(ids, fmt.Sprintf("7,%08xbeefcafe", i))
|
||||
}
|
||||
f.queueDeletions(ids...)
|
||||
f.snapshotDeletionLedger()
|
||||
key := f.deletionLedgerKey()
|
||||
if _, err := store.KvGet(context.Background(), deletionLedgerGenPartKey(key, 1, 0)); err != nil {
|
||||
t.Fatalf("generation 1 part must exist, got %v", err)
|
||||
}
|
||||
|
||||
f.queueDeletions("8,08")
|
||||
f.snapshotDeletionLedger()
|
||||
if _, err := store.KvGet(context.Background(), deletionLedgerGenPartKey(key, 2, 0)); err != nil {
|
||||
t.Fatalf("generation 2 part must exist, got %v", err)
|
||||
}
|
||||
if _, err := store.KvGet(context.Background(), deletionLedgerGenPartKey(key, 1, 0)); err != ErrKvNotFound {
|
||||
t.Fatalf("generation 1 part must be cleaned after publication, got %v", err)
|
||||
}
|
||||
var manifest struct {
|
||||
Parts int `json:"parts"`
|
||||
Gen int `json:"gen"`
|
||||
}
|
||||
raw, _ := store.KvGet(context.Background(), []byte(key))
|
||||
if err := json.Unmarshal(raw, &manifest); err != nil || manifest.Gen != 2 {
|
||||
t.Fatalf("manifest must publish generation 2, got %q", raw)
|
||||
}
|
||||
}
|
||||
|
||||
// A failed cleanup of a superseded generation is retried by the next snapshot
|
||||
// instead of leaking the part keys.
|
||||
func TestDeletionLedgerStalePartsRetried(t *testing.T) {
|
||||
withPersistedConfig(t, true)
|
||||
store := newStubFilerStore()
|
||||
f := newLedgerTestFiler(store)
|
||||
|
||||
var ids []string
|
||||
for i := 0; i < 4000; i++ {
|
||||
ids = append(ids, fmt.Sprintf("7,%08xbeefcafe", i))
|
||||
}
|
||||
f.queueDeletions(ids...)
|
||||
f.snapshotDeletionLedger()
|
||||
key := f.deletionLedgerKey()
|
||||
|
||||
store.kvDeleteErr = errors.New("delete down")
|
||||
f.queueDeletions("8,08")
|
||||
f.snapshotDeletionLedger()
|
||||
if len(f.deletionLedgerStale) == 0 {
|
||||
t.Fatalf("failed part deletes must be tracked for retry")
|
||||
}
|
||||
store.kvDeleteErr = nil
|
||||
f.queueDeletions("8,09")
|
||||
f.snapshotDeletionLedger()
|
||||
if len(f.deletionLedgerStale) != 0 {
|
||||
t.Fatalf("stale parts must flush once deletes work, got %v", f.deletionLedgerStale)
|
||||
}
|
||||
if _, err := store.KvGet(context.Background(), deletionLedgerGenPartKey(key, 1, 0)); err != ErrKvNotFound {
|
||||
t.Fatalf("stale generation part must be deleted, got %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// A pre-scoping ledger under the base key must migrate only after the scoped
|
||||
// copy is durable: the recovered ids are written under the scoped key first,
|
||||
// then the base key is removed.
|
||||
func TestDeletionLedgerLegacyMigrationDurable(t *testing.T) {
|
||||
withPersistedConfig(t, true)
|
||||
store := newStubFilerStore()
|
||||
store.kv[KvKeyDeletionLedger] = []byte(`["1,01","1,02"]`)
|
||||
|
||||
f := newLedgerTestFiler(store)
|
||||
f.Dlm = lock_manager.NewDistributedLockManager("filer-new:9999")
|
||||
if !f.reloadDeletionLedger() {
|
||||
t.Fatalf("reload should migrate the legacy ledger")
|
||||
}
|
||||
persisted, ok := readPersistedLedger(t, f)
|
||||
if !ok || len(persisted) != 2 {
|
||||
t.Fatalf("legacy ids must land under the scoped key first, got %v", persisted)
|
||||
}
|
||||
if _, err := store.KvGet(context.Background(), []byte(KvKeyDeletionLedger)); err != ErrKvNotFound {
|
||||
t.Fatalf("legacy key must be removed once the scoped write is durable, got %v", err)
|
||||
}
|
||||
if f.pendingDeletionCount() != 2 {
|
||||
t.Fatalf("migrated ids must join the pending set, got %d", f.pendingDeletionCount())
|
||||
}
|
||||
}
|
||||
|
||||
// An indexed ledger that cannot be read must not be skipped: claiming the
|
||||
// readable ones and leaving the unreadable one behind would strand it, since
|
||||
// the index is only searched while the filer's own key is absent. The claim
|
||||
// aborts so the reload is retried.
|
||||
func TestDeletionLedgerUnreadableForeignBlocks(t *testing.T) {
|
||||
withPersistedConfig(t, true)
|
||||
store := newStubFilerStore()
|
||||
stranded := KvKeyDeletionLedger + ".filer-old:8888"
|
||||
store.kv[stranded] = []byte(`{"parts":2,"gen":1}`) // manifest without parts = corrupt
|
||||
index, _ := json.Marshal([]string{stranded})
|
||||
store.kv[KvKeyDeletionLedgerIndex] = index
|
||||
|
||||
f := newLedgerTestFiler(store)
|
||||
f.Dlm = lock_manager.NewDistributedLockManager("filer-new:9999")
|
||||
if f.reloadDeletionLedger() {
|
||||
t.Fatalf("an unreadable indexed ledger must fail the reload")
|
||||
}
|
||||
if _, err := store.KvGet(context.Background(), []byte(stranded)); err != nil {
|
||||
t.Fatalf("unreadable ledger must be left untouched, got %v", err)
|
||||
}
|
||||
if _, ok := readPersistedLedger(t, f); ok {
|
||||
t.Fatalf("no scoped ledger may be written while a claim source is unreadable")
|
||||
}
|
||||
}
|
||||
|
||||
// A live filer's ledger is not deleted when it republished between the claim's
|
||||
// read and cleanup: the source stays and its newer ids merge into the claim.
|
||||
// The hook swaps the source content on its second read — the claim's first
|
||||
// read sees the old set, the verification read sees the republished one,
|
||||
// exactly like a peer snapshot landing mid-claim.
|
||||
func TestDeletionLedgerClaimKeepsChangedSource(t *testing.T) {
|
||||
withPersistedConfig(t, true)
|
||||
store := newStubFilerStore()
|
||||
srcKey := KvKeyDeletionLedger + ".filer-old:8888"
|
||||
store.kv[srcKey] = []byte(`["1,01","1,02"]`)
|
||||
index, _ := json.Marshal([]string{srcKey})
|
||||
store.kv[KvKeyDeletionLedgerIndex] = index
|
||||
|
||||
srcReads := 0
|
||||
store.kvGetHook = func(key []byte) {
|
||||
if string(key) == srcKey {
|
||||
srcReads++
|
||||
if srcReads == 2 {
|
||||
store.kv[srcKey] = []byte(`["1,01","1,02","1,03"]`)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
f := newLedgerTestFiler(store)
|
||||
f.Dlm = lock_manager.NewDistributedLockManager("filer-new:9999")
|
||||
if !f.reloadDeletionLedger() {
|
||||
t.Fatalf("claim should succeed")
|
||||
}
|
||||
if got := string(store.kv[srcKey]); got != `["1,01","1,02","1,03"]` {
|
||||
t.Fatalf("republished source must be left intact, got %q", got)
|
||||
}
|
||||
if f.pendingDeletionCount() != 3 {
|
||||
t.Fatalf("republished ids must merge into the pending set, got %d", f.pendingDeletionCount())
|
||||
}
|
||||
persisted, ok := readPersistedLedger(t, f)
|
||||
if !ok || len(persisted) != 3 {
|
||||
t.Fatalf("merged ids must be durable under our key, got %v", persisted)
|
||||
}
|
||||
}
|
||||
|
||||
// An expired retry item must not erase a newer enqueue for the same file id:
|
||||
// expiry forgets only the epoch the retry item recorded.
|
||||
func TestForgetDeletionEpochSkipsNewer(t *testing.T) {
|
||||
withPersistedConfig(t, true)
|
||||
f := newLedgerTestFiler(newStubFilerStore())
|
||||
|
||||
f.queueDeletions("1,01")
|
||||
oldEpoch := f.deletionEpoch("1,01")
|
||||
f.queueDeletions("1,01") // re-enqueue bumps the epoch
|
||||
newEpoch := f.deletionEpoch("1,01")
|
||||
if oldEpoch == newEpoch {
|
||||
t.Fatalf("re-enqueue must bump the epoch")
|
||||
}
|
||||
|
||||
f.forgetDeletionEpoch("1,01", oldEpoch)
|
||||
if f.pendingDeletionCount() != 1 {
|
||||
t.Fatalf("stale expiry must not erase a newer enqueue")
|
||||
}
|
||||
f.forgetDeletionEpoch("1,01", newEpoch)
|
||||
if f.pendingDeletionCount() != 0 {
|
||||
t.Fatalf("matching expiry must forget the id")
|
||||
}
|
||||
}
|
||||
|
||||
func contains(haystack []string, needle string) bool {
|
||||
for _, s := range haystack {
|
||||
if s == needle {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
@@ -10,15 +10,15 @@ func TestDeletionRetryQueue_AddAndRetrieve(t *testing.T) {
|
||||
queue := NewDeletionRetryQueue()
|
||||
|
||||
// Add items
|
||||
queue.AddOrUpdate("file1", "is read only")
|
||||
queue.AddOrUpdate("file2", "connection reset")
|
||||
queue.AddOrUpdate("file1", "is read only", 0)
|
||||
queue.AddOrUpdate("file2", "connection reset", 0)
|
||||
|
||||
if queue.Size() != 2 {
|
||||
t.Errorf("Expected queue size 2, got %d", queue.Size())
|
||||
}
|
||||
|
||||
// Items not ready yet (initial delay is 5 minutes)
|
||||
readyItems := queue.GetReadyItems(10)
|
||||
readyItems, _ := queue.GetReadyItems(10)
|
||||
if len(readyItems) != 0 {
|
||||
t.Errorf("Expected 0 ready items, got %d", len(readyItems))
|
||||
}
|
||||
@@ -104,7 +104,7 @@ func TestDeletionRetryQueue_MaxAttemptsReached(t *testing.T) {
|
||||
queue := NewDeletionRetryQueue()
|
||||
|
||||
// Add item
|
||||
queue.AddOrUpdate("file1", "error")
|
||||
queue.AddOrUpdate("file1", "error", 0)
|
||||
|
||||
// Manually set retry count to max
|
||||
queue.lock.Lock()
|
||||
@@ -119,7 +119,7 @@ func TestDeletionRetryQueue_MaxAttemptsReached(t *testing.T) {
|
||||
queue.lock.Unlock()
|
||||
|
||||
// Try to get ready items - should be returned for the last retry (attempt #10)
|
||||
readyItems := queue.GetReadyItems(10)
|
||||
readyItems, _ := queue.GetReadyItems(10)
|
||||
if len(readyItems) != 1 {
|
||||
t.Fatalf("Expected 1 item for last retry, got %d", len(readyItems))
|
||||
}
|
||||
@@ -139,7 +139,7 @@ func TestDeletionRetryQueue_MaxAttemptsReached(t *testing.T) {
|
||||
queue.lock.Unlock()
|
||||
|
||||
// Now it should be discarded (retry count is 11, exceeds max of 10)
|
||||
readyItems = queue.GetReadyItems(10)
|
||||
readyItems, _ = queue.GetReadyItems(10)
|
||||
if len(readyItems) != 0 {
|
||||
t.Errorf("Expected 0 items (max attempts exceeded), got %d", len(readyItems))
|
||||
}
|
||||
@@ -248,7 +248,7 @@ func TestDeletionRetryQueue_HeapOrdering(t *testing.T) {
|
||||
queue.lock.Unlock()
|
||||
|
||||
// GetReadyItems should return in NextRetryAt order
|
||||
readyItems := queue.GetReadyItems(10)
|
||||
readyItems, _ := queue.GetReadyItems(10)
|
||||
expectedOrder := []string{"file1", "file2", "file3"}
|
||||
|
||||
if len(readyItems) != 3 {
|
||||
@@ -266,7 +266,7 @@ func TestDeletionRetryQueue_DuplicateFileIds(t *testing.T) {
|
||||
queue := NewDeletionRetryQueue()
|
||||
|
||||
// Add same file ID twice with retryable error - simulates duplicate in batch
|
||||
queue.AddOrUpdate("file1", "timeout error")
|
||||
queue.AddOrUpdate("file1", "timeout error", 0)
|
||||
|
||||
// Verify only one item exists in queue
|
||||
if queue.Size() != 1 {
|
||||
@@ -284,7 +284,7 @@ func TestDeletionRetryQueue_DuplicateFileIds(t *testing.T) {
|
||||
queue.lock.Unlock()
|
||||
|
||||
// Add same file ID again - should NOT increment retry count (just update error)
|
||||
queue.AddOrUpdate("file1", "timeout error again")
|
||||
queue.AddOrUpdate("file1", "timeout error again", 0)
|
||||
|
||||
// Verify still only one item exists in queue (not duplicated)
|
||||
if queue.Size() != 1 {
|
||||
@@ -306,3 +306,27 @@ func TestDeletionRetryQueue_DuplicateFileIds(t *testing.T) {
|
||||
t.Errorf("Expected LastError to be updated to 'timeout error again', got %q", item2.LastError)
|
||||
}
|
||||
}
|
||||
|
||||
// AddOrUpdate must not overwrite the ledger epoch on an in-flight item: the
|
||||
// worker that popped it reads that field without the queue lock, and its
|
||||
// expiry/permanent-forget must only match the record the attempt started with.
|
||||
func TestDeletionRetryQueue_InFlightKeepsEpoch(t *testing.T) {
|
||||
queue := NewDeletionRetryQueue()
|
||||
queue.AddOrUpdate("file1", "timeout", 7)
|
||||
|
||||
queue.lock.Lock()
|
||||
item := queue.itemIndex["file1"]
|
||||
item.NextRetryAt = time.Now().Add(-time.Second)
|
||||
heap.Init(&queue.heap)
|
||||
queue.lock.Unlock()
|
||||
|
||||
ready, _ := queue.GetReadyItems(1)
|
||||
if len(ready) != 1 {
|
||||
t.Fatalf("expected the item ready, got %d", len(ready))
|
||||
}
|
||||
|
||||
queue.AddOrUpdate("file1", "newer error", 42)
|
||||
if got := ready[0].ledgerEpoch; got != 7 {
|
||||
t.Fatalf("in-flight epoch must stay 7, got %d", got)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -19,7 +19,7 @@ func (f *Filer) ensureEntryInode(entry *Entry) {
|
||||
entry.Attr.Crtime = time.Now()
|
||||
}
|
||||
if len(entry.HardLinkId) > 0 {
|
||||
entry.Attr.Inode = uint64(util.HashStringToLong(string(entry.HardLinkId)))
|
||||
entry.Attr.Inode = util.NormalizeInode(uint64(util.HashStringToLong(string(entry.HardLinkId))))
|
||||
return
|
||||
}
|
||||
entry.Attr.Inode = entry.FullPath.AsInode(entry.Attr.Crtime.Unix())
|
||||
|
||||
@@ -3,6 +3,8 @@ package filer
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"math"
|
||||
"os"
|
||||
"testing"
|
||||
"time"
|
||||
@@ -51,10 +53,56 @@ func TestEnsureEntryInodeSharesAcrossHardLinks(t *testing.T) {
|
||||
|
||||
// Every link to the same target resolves to one inode, independent of path
|
||||
// or creation time.
|
||||
assert.Equal(t, uint64(util.HashStringToLong(string(hardLinkId))), a.Attr.Inode)
|
||||
assert.Equal(t, util.NormalizeInode(uint64(util.HashStringToLong(string(hardLinkId)))), a.Attr.Inode)
|
||||
assert.Equal(t, a.Attr.Inode, b.Attr.Inode)
|
||||
}
|
||||
|
||||
// TestEnsureEntryInodeFitsSignedLong pins the invariant that every generated
|
||||
// inode is storable: the filer hands Attr.Inode to the backing store verbatim,
|
||||
// and the Elasticsearch store indexes it as a signed `long`. Roughly half of
|
||||
// the unsigned hash space sits above math.MaxInt64, so a store that rejects
|
||||
// those values used to fail the metadata write for half of all entries.
|
||||
func TestEnsureEntryInodeFitsSignedLong(t *testing.T) {
|
||||
f := &Filer{}
|
||||
crtime := time.Unix(1700000000, 0)
|
||||
|
||||
seen := make(map[uint64]string)
|
||||
for i := 0; i < 5000; i++ {
|
||||
fullPath := util.FullPath(fmt.Sprintf("/topics/.system/log/2026-10-02/entry-%d", i))
|
||||
entry := &Entry{FullPath: fullPath, Attr: Attr{Crtime: crtime}}
|
||||
f.ensureEntryInode(entry)
|
||||
|
||||
if entry.Attr.Inode > math.MaxInt64 {
|
||||
t.Fatalf("ensureEntryInode(%q) = %d, above math.MaxInt64", fullPath, entry.Attr.Inode)
|
||||
}
|
||||
// Folding must not collapse distinct paths onto one inode.
|
||||
if other, ok := seen[entry.Attr.Inode]; ok {
|
||||
t.Fatalf("ensureEntryInode(%q) collided with %q on inode %d", fullPath, other, entry.Attr.Inode)
|
||||
}
|
||||
seen[entry.Attr.Inode] = string(fullPath)
|
||||
}
|
||||
}
|
||||
|
||||
// TestEnsureEntryInodeHardLinkFitsSignedLong covers the hard-link branch, which
|
||||
// hashes HardLinkId instead of the path and so has no path-derived crtime term.
|
||||
func TestEnsureEntryInodeHardLinkFitsSignedLong(t *testing.T) {
|
||||
f := &Filer{}
|
||||
crtime := time.Unix(1700000000, 0)
|
||||
|
||||
for i := 0; i < 5000; i++ {
|
||||
entry := &Entry{
|
||||
FullPath: util.FullPath(fmt.Sprintf("/links/target-%d.txt", i)),
|
||||
Attr: Attr{Crtime: crtime},
|
||||
HardLinkId: NewHardLinkId(),
|
||||
}
|
||||
f.ensureEntryInode(entry)
|
||||
|
||||
if entry.Attr.Inode > math.MaxInt64 {
|
||||
t.Fatalf("ensureEntryInode(%q) = %d, above math.MaxInt64", entry.FullPath, entry.Attr.Inode)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func newTestFilerWithStubStore() (*Filer, *stubFilerStore) {
|
||||
store := newStubFilerStore()
|
||||
f := NewFiler(pb.ServerDiscovery{}, nil, "", "", "", "", "", 255, nil)
|
||||
|
||||
@@ -35,6 +35,9 @@ type stubFilerStore struct {
|
||||
kv map[string][]byte
|
||||
insertErr error
|
||||
findErr error
|
||||
kvGetErr error
|
||||
kvDeleteErr error
|
||||
kvGetHook func(key []byte)
|
||||
deleteErrByPath map[string]error
|
||||
}
|
||||
|
||||
@@ -63,6 +66,12 @@ func (s *stubFilerStore) KvPut(_ context.Context, key []byte, value []byte) erro
|
||||
func (s *stubFilerStore) KvGet(_ context.Context, key []byte) ([]byte, error) {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
if s.kvGetHook != nil {
|
||||
s.kvGetHook(key)
|
||||
}
|
||||
if s.kvGetErr != nil {
|
||||
return nil, s.kvGetErr
|
||||
}
|
||||
value, found := s.kv[string(key)]
|
||||
if !found {
|
||||
return nil, ErrKvNotFound
|
||||
@@ -72,6 +81,9 @@ func (s *stubFilerStore) KvGet(_ context.Context, key []byte) ([]byte, error) {
|
||||
func (s *stubFilerStore) KvDelete(_ context.Context, key []byte) error {
|
||||
s.mu.Lock()
|
||||
defer s.mu.Unlock()
|
||||
if s.kvDeleteErr != nil {
|
||||
return s.kvDeleteErr
|
||||
}
|
||||
delete(s.kv, string(key))
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -247,6 +247,11 @@ func (f *Filer) logMetaEvent(ctx context.Context, event *filer_pb.SubscribeMetad
|
||||
// in the rejection and volumeFileSizeLimit picks the real limit up from there.
|
||||
const metadataLogUploadLimit = log_buffer.BufferSize
|
||||
|
||||
// shutdownMetadataLogFlushBudget bounds how long metadata-log flushes may keep
|
||||
// retrying once the filer is shutting down; Shutdown cancels the shared flush
|
||||
// context when it expires.
|
||||
const shutdownMetadataLogFlushBudget = 15 * time.Second
|
||||
|
||||
var fileSizeLimitPattern = regexp.MustCompile(`file over the limited (\d+) bytes`)
|
||||
|
||||
// volumeFileSizeLimit reads the byte limit back out of a volume server's size
|
||||
@@ -278,17 +283,30 @@ func (f *Filer) logFlushFunc(logBuffer *log_buffer.LogBuffer, startTime, stopTim
|
||||
|
||||
// One piece at a time, each retried on its own so a partial success is not
|
||||
// replayed, and the piece size follows the limit the volume servers report.
|
||||
// While the filer keeps running the retry is unbounded so no metadata is
|
||||
// dropped; once Shutdown arms the flush deadline the shared context cancels
|
||||
// and the flush drops what is left instead of holding shutdown open.
|
||||
limit := metadataLogUploadLimit
|
||||
ctx := f.flushCtx
|
||||
if ctx == nil {
|
||||
ctx = context.Background()
|
||||
}
|
||||
for len(buf) > 0 {
|
||||
piece := nextLogPiece(buf, limit)
|
||||
if err := f.appendToFile(targetFile, piece); err != nil {
|
||||
if err := f.appendToFile(ctx, targetFile, piece); err != nil {
|
||||
glog.V(0).Infof("metadata log write failed %s: %v", targetFile, err)
|
||||
if reported := volumeFileSizeLimit(err); reported > 0 && reported < limit {
|
||||
glog.V(0).Infof("metadata log upload limit lowered to %d bytes", reported)
|
||||
limit = reported
|
||||
continue
|
||||
}
|
||||
time.Sleep(737 * time.Millisecond)
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
logBuffer.NoteFlushDropped(len(buf))
|
||||
glog.V(0).Infof("metadata log flush abandoned during shutdown, %d bytes left for %s: %v", len(buf), targetFile, ctx.Err())
|
||||
return
|
||||
case <-time.After(737 * time.Millisecond):
|
||||
}
|
||||
continue
|
||||
}
|
||||
buf = buf[len(piece):]
|
||||
|
||||
@@ -11,16 +11,20 @@ import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
)
|
||||
|
||||
func (f *Filer) appendToFile(targetFile string, data []byte) error {
|
||||
func (f *Filer) appendToFile(ctx context.Context, targetFile string, data []byte) error {
|
||||
|
||||
assignResult, uploadResult, err2 := f.assignAndUpload(targetFile, data)
|
||||
assignResult, uploadResult, err2 := f.assignAndUpload(ctx, targetFile, data)
|
||||
if err2 != nil {
|
||||
return err2
|
||||
}
|
||||
|
||||
// The piece is already uploaded; commit it on a detached context so an
|
||||
// expired shutdown deadline does not strand the chunk.
|
||||
ctx = context.WithoutCancel(ctx)
|
||||
|
||||
// find out existing entry
|
||||
fullpath := util.FullPath(targetFile)
|
||||
entry, err := f.FindEntry(context.Background(), fullpath)
|
||||
entry, err := f.FindEntry(ctx, fullpath)
|
||||
var offset int64 = 0
|
||||
if err == filer_pb.ErrNotFound {
|
||||
entry = &Entry{
|
||||
@@ -43,7 +47,7 @@ func (f *Filer) appendToFile(targetFile string, data []byte) error {
|
||||
entry.Chunks = append(entry.GetChunks(), uploadResult.ToPbFileChunk(assignResult.Fid, offset, time.Now().UnixNano()))
|
||||
|
||||
// update the entry
|
||||
err = f.CreateEntry(context.Background(), entry, nil, false, false, nil, false, f.MaxFilenameLength)
|
||||
err = f.CreateEntry(ctx, entry, nil, false, false, nil, false, f.MaxFilenameLength)
|
||||
|
||||
return err
|
||||
}
|
||||
@@ -59,19 +63,30 @@ func (f *Filer) resolveMetadataLogAssignDiskType(targetFile string) (string, *fi
|
||||
return util.Nvl(rule.DiskType, f.DefaultDiskType), rule
|
||||
}
|
||||
|
||||
func (f *Filer) assignAndUpload(targetFile string, data []byte) (*operation.AssignResult, *operation.UploadResult, error) {
|
||||
// metaLogCollectionFor resolves the system metadata log's collection: the
|
||||
// filer.options.metaLog.collection override, then the filer default, then the
|
||||
// matched storage rule.
|
||||
func (f *Filer) metaLogCollectionFor(ruleCollection string) string {
|
||||
return util.Nvl(f.metaLogTargetCollection, f.metaLogCollection, ruleCollection)
|
||||
}
|
||||
|
||||
func (f *Filer) metaLogReplicationFor(ruleReplication string) string {
|
||||
return util.Nvl(f.metaLogTargetReplication, f.metaLogReplication, ruleReplication)
|
||||
}
|
||||
|
||||
func (f *Filer) assignAndUpload(ctx context.Context, targetFile string, data []byte) (*operation.AssignResult, *operation.UploadResult, error) {
|
||||
// assign a volume location
|
||||
diskType, rule := f.resolveMetadataLogAssignDiskType(targetFile)
|
||||
assignRequest := &operation.VolumeAssignRequest{
|
||||
Count: 1,
|
||||
Collection: util.Nvl(f.metaLogCollection, rule.Collection),
|
||||
Replication: util.Nvl(f.metaLogReplication, rule.Replication),
|
||||
Collection: f.metaLogCollectionFor(rule.Collection),
|
||||
Replication: f.metaLogReplicationFor(rule.Replication),
|
||||
DiskType: diskType,
|
||||
WritableVolumeCount: rule.VolumeGrowthCount,
|
||||
ExpectedDataSize: uint64(len(data)),
|
||||
}
|
||||
|
||||
assignResult, err := operation.Assign(context.Background(), f.GetMaster, f.GrpcDialOption, assignRequest)
|
||||
assignResult, err := operation.Assign(ctx, f.GetMaster, f.GrpcDialOption, assignRequest)
|
||||
if err != nil {
|
||||
return nil, nil, fmt.Errorf("AssignVolume: %w", err)
|
||||
}
|
||||
@@ -96,7 +111,7 @@ func (f *Filer) assignAndUpload(targetFile string, data []byte) (*operation.Assi
|
||||
return nil, nil, fmt.Errorf("upload data %s: %v", targetUrl, err)
|
||||
}
|
||||
|
||||
uploadResult, err := uploader.UploadData(context.Background(), data, uploadOption)
|
||||
uploadResult, err := uploader.UploadData(ctx, data, uploadOption)
|
||||
if err != nil {
|
||||
return nil, nil, fmt.Errorf("upload data %s: %v", targetUrl, err)
|
||||
}
|
||||
|
||||
@@ -1,6 +1,8 @@
|
||||
package filer
|
||||
|
||||
import (
|
||||
"context"
|
||||
"sync/atomic"
|
||||
"testing"
|
||||
"testing/synctest"
|
||||
"time"
|
||||
@@ -62,3 +64,44 @@ func TestShutdownKeepsStoreOpenUntilMetadataIsFlushed(t *testing.T) {
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestShutdownBoundsBlockedMetadataFlush(t *testing.T) {
|
||||
synctest.Test(t, func(t *testing.T) {
|
||||
store := &shutdownStore{closed: make(chan struct{})}
|
||||
f := &Filer{Store: store, deletionQuit: make(chan struct{})}
|
||||
f.flushCtx, f.flushCancel = context.WithCancel(context.Background())
|
||||
|
||||
var flushStarted atomic.Bool
|
||||
lb := log_buffer.NewLogBuffer("blocked flush", time.Hour,
|
||||
func(lb *log_buffer.LogBuffer, _, _ time.Time, buf []byte, _, _ int64) {
|
||||
flushStarted.Store(true)
|
||||
// A dead cluster makes append retries hang here; the shutdown
|
||||
// deadline cancels the shared flush context to unblock it.
|
||||
<-f.flushCtx.Done()
|
||||
lb.NoteFlushDropped(len(buf))
|
||||
}, nil, nil)
|
||||
f.LocalMetaLogBuffer = lb
|
||||
if err := lb.AddDataToBuffer(nil, []byte("last metadata event"), 0); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
done := make(chan struct{})
|
||||
go func() {
|
||||
f.Shutdown()
|
||||
close(done)
|
||||
}()
|
||||
|
||||
<-done
|
||||
if !flushStarted.Load() {
|
||||
t.Error("shutdown finished without running the pending flush")
|
||||
}
|
||||
if ts := lb.GetLastFlushTsNs(); ts != 0 {
|
||||
t.Errorf("dropped flush advanced the flushed watermark to %d", ts)
|
||||
}
|
||||
select {
|
||||
case <-store.closed:
|
||||
default:
|
||||
t.Error("metadata store was not closed after the bounded wait")
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
@@ -0,0 +1,137 @@
|
||||
package filer
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
)
|
||||
|
||||
// The metadata log's collection resolution must let operators redirect the
|
||||
// internal /topics/.system/log chunks into their own collection without
|
||||
// touching where user data goes, and must be strictly backward compatible
|
||||
// when the override is unset.
|
||||
|
||||
func TestMetaLogCollectionResolution(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
target string // filer.options.metaLog.collection override
|
||||
filerDefault string // -collection flag value
|
||||
ruleCollection string // storage rule matched on the log path
|
||||
want string
|
||||
}{
|
||||
{
|
||||
name: "no override: filer default wins (today's behaviour)",
|
||||
target: "",
|
||||
filerDefault: "mydata",
|
||||
ruleCollection: "",
|
||||
want: "mydata",
|
||||
},
|
||||
{
|
||||
name: "no override anywhere: still empty (default collection)",
|
||||
target: "",
|
||||
filerDefault: "",
|
||||
ruleCollection: "",
|
||||
want: "",
|
||||
},
|
||||
{
|
||||
name: "override wins over filer default",
|
||||
target: "filer-meta",
|
||||
filerDefault: "mydata",
|
||||
ruleCollection: "",
|
||||
want: "filer-meta",
|
||||
},
|
||||
{
|
||||
name: "override wins over matched rule",
|
||||
target: "filer-meta",
|
||||
filerDefault: "",
|
||||
ruleCollection: "rulecol",
|
||||
want: "filer-meta",
|
||||
},
|
||||
{
|
||||
name: "no override but rule set: rule used (unchanged fallback chain)",
|
||||
target: "",
|
||||
filerDefault: "",
|
||||
ruleCollection: "rulecol",
|
||||
want: "rulecol",
|
||||
},
|
||||
}
|
||||
|
||||
for _, tc := range tests {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
// Drive the same setter a running filer would use post-construction.
|
||||
f := &Filer{}
|
||||
f.metaLogTargetCollection = tc.target
|
||||
f.metaLogCollection = tc.filerDefault
|
||||
if got := f.metaLogCollectionFor(tc.ruleCollection); got != tc.want {
|
||||
t.Errorf("metaLogCollectionFor(%q) = %q, want %q", tc.ruleCollection, got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestMetaLogReplicationResolution(t *testing.T) {
|
||||
f := &Filer{}
|
||||
f.metaLogTargetReplication = "110"
|
||||
f.metaLogReplication = "010"
|
||||
if got := f.metaLogReplicationFor("001"); got != "110" {
|
||||
t.Errorf("override must win, got %q", got)
|
||||
}
|
||||
f.metaLogTargetReplication = ""
|
||||
if got := f.metaLogReplicationFor("001"); got != "010" {
|
||||
t.Errorf("filer replication must win when no override, got %q", got)
|
||||
}
|
||||
f.metaLogReplication = ""
|
||||
if got := f.metaLogReplicationFor("001"); got != "001" {
|
||||
t.Errorf("rule replication must be used last, got %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestViperReadsMetaLogOverrides proves the exact viper keys the docs promise
|
||||
// are read into the fields a running filer uses.
|
||||
func TestViperReadsMetaLogOverrides(t *testing.T) {
|
||||
v := util.GetViper()
|
||||
v.Set("filer.options.metaLog.collection", "filer-meta")
|
||||
v.Set("filer.options.metaLog.replication", "100")
|
||||
defer func() {
|
||||
// Reset so other tests see the unset default.
|
||||
v.Set("filer.options.metaLog.collection", "")
|
||||
v.Set("filer.options.metaLog.replication", "")
|
||||
}()
|
||||
|
||||
f := NewFiler(pb.ServerDiscovery{}, nil, "", "", "", "", "", 255, nil)
|
||||
if got := f.metaLogTargetCollection; got != "filer-meta" {
|
||||
t.Fatalf("NewFiler did not read the collection override: %q", got)
|
||||
}
|
||||
if got := f.metaLogTargetReplication; got != "100" {
|
||||
t.Fatalf("NewFiler did not read the replication override: %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestBucketCollectionKeepsMetaLogTargetCollection mirrors the guarantee that
|
||||
// protects the filer's default collection: when a bucket resolves to the
|
||||
// collection the system metadata log was redirected to, deleting the bucket
|
||||
// must not drop that collection (it backs internal log volumes).
|
||||
func TestBucketCollectionKeepsMetaLogTargetCollection(t *testing.T) {
|
||||
f, store, master := newFilerWithFakeMaster(t)
|
||||
// Operator redirected the meta log to its own collection...
|
||||
f.metaLogTargetCollection = "filer-meta"
|
||||
// ...and a bucket happens to resolve to that very collection.
|
||||
f.FilerConf.SetLocationConf(&filer_pb.FilerConf_PathConf{
|
||||
LocationPrefix: "/buckets/a",
|
||||
Collection: "filer-meta",
|
||||
})
|
||||
seedBucket(t, store, util.FullPath("/buckets/a"))
|
||||
|
||||
if err := f.DeleteEntryMetaAndData(context.Background(), "/buckets/a", true, false, true, false, nil, 0); err != nil {
|
||||
t.Fatalf("DeleteEntryMetaAndData: %v", err)
|
||||
}
|
||||
|
||||
select {
|
||||
case call := <-master.calls:
|
||||
t.Fatalf("the meta-log target collection was deleted by a bucket delete: %q", call.name)
|
||||
default:
|
||||
}
|
||||
}
|
||||
@@ -75,6 +75,9 @@ func SaveInsideFiler(ctx context.Context, client filer_pb.SeaweedFilerClient, di
|
||||
})
|
||||
} else if err == nil {
|
||||
entry := resp.Entry
|
||||
if bytes.Equal(entry.Content, content) && bytes.Equal(entry.GetAttributes().GetMd5(), contentMd5[:]) {
|
||||
return nil
|
||||
}
|
||||
entry.Content = content
|
||||
entry.Attributes.Mtime = time.Now().Unix()
|
||||
entry.Attributes.FileSize = uint64(len(content))
|
||||
|
||||
@@ -0,0 +1,51 @@
|
||||
package filer
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// Rewriting identical inline content must not issue UpdateEntry: each one is a
|
||||
// metadata event the local meta log persists to /topics/.system/log, and a
|
||||
// client that rewrites the same config on a timer (e.g. the operator's 5-minute
|
||||
// resync) keeps the volumes growing on an otherwise idle cluster.
|
||||
func TestSaveInsideFilerSkipsIdenticalContent(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
client := newFakeFilerConfClient()
|
||||
|
||||
require.NoError(t, SaveInsideFiler(ctx, client, DirectoryEtcSeaweedFS, "a.json", []byte("v1")))
|
||||
require.Equal(t, 0, client.updateCalls)
|
||||
|
||||
require.NoError(t, SaveInsideFiler(ctx, client, DirectoryEtcSeaweedFS, "a.json", []byte("v1")))
|
||||
assert.Equal(t, 0, client.updateCalls)
|
||||
|
||||
require.NoError(t, SaveInsideFiler(ctx, client, DirectoryEtcSeaweedFS, "a.json", []byte("v2")))
|
||||
assert.Equal(t, 1, client.updateCalls)
|
||||
|
||||
content, err := ReadInsideFiler(ctx, client, DirectoryEtcSeaweedFS, "a.json")
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, []byte("v2"), content)
|
||||
}
|
||||
|
||||
// A legacy entry holding identical bytes but no Md5 still gets one write to
|
||||
// stamp the hash that IF_ETAG_MATCH conditional writes key off; later
|
||||
// identical writes skip.
|
||||
func TestSaveInsideFilerStampsMd5OnLegacyEntry(t *testing.T) {
|
||||
ctx := context.Background()
|
||||
client := newFakeFilerConfClient()
|
||||
client.entries[client.key(DirectoryEtcSeaweedFS, "b.json")] = &filer_pb.Entry{
|
||||
Name: "b.json",
|
||||
Content: []byte("v1"),
|
||||
Attributes: &filer_pb.FuseAttributes{},
|
||||
}
|
||||
|
||||
require.NoError(t, SaveInsideFiler(ctx, client, DirectoryEtcSeaweedFS, "b.json", []byte("v1")))
|
||||
assert.Equal(t, 1, client.updateCalls)
|
||||
|
||||
require.NoError(t, SaveInsideFiler(ctx, client, DirectoryEtcSeaweedFS, "b.json", []byte("v1")))
|
||||
assert.Equal(t, 1, client.updateCalls)
|
||||
}
|
||||
+19
-3
@@ -349,14 +349,23 @@ func (c *ChunkReadAt) doReadAt(ctx context.Context, p []byte, offset int64) (n i
|
||||
|
||||
func (c *ChunkReadAt) readChunkSliceAt(ctx context.Context, buffer []byte, chunkView *ChunkView, nextChunkViews *Interval[*ChunkView], offset uint64) (n int, err error) {
|
||||
|
||||
if c.readerPattern.IsRandomMode() {
|
||||
// A view clipped to part of its chunk (e.g. the edge of a ranged GET,
|
||||
// whose views ViewFromVisibleIntervals clips to the request) only ever
|
||||
// needs that part: fetch it as a range no matter the detected pattern.
|
||||
// Fetching the chunk whole would multiply volume-server reads. Ciphered
|
||||
// and compressed chunks are the exception: the volume server reads the
|
||||
// whole blob to serve a range, so they take the shared whole-chunk path
|
||||
// where one download serves every buffer — unless the whole chunk cannot
|
||||
// even fit the reader budget, in which case a range fetch is the only
|
||||
// way to serve the request.
|
||||
rangeFetch := chunkView.CanRangeFetch() || !c.readerCache.budget.canFit(int(chunkView.ChunkSize))
|
||||
if rangeFetch && (!chunkView.IsFullChunk() || c.readerPattern.IsRandomMode()) {
|
||||
c.readerCache.releaseStream(&c.stream)
|
||||
n, err := c.readerCache.chunkCache.ReadChunkAt(buffer, chunkView.FileId, offset)
|
||||
if n > 0 {
|
||||
return n, err
|
||||
}
|
||||
return fetchChunkRange(ctx, buffer, c.readerCache.lookupFileIdFn, chunkView.FileId, chunkView.CipherKey, chunkView.IsGzipped, int64(offset),
|
||||
refreshUrls(ctx, c.readerCache.cacheInvalidator, c.readerCache.lookupFileIdFn, chunkView.FileId))
|
||||
return c.readerCache.fetchChunkRange(ctx, buffer, chunkView, int64(offset))
|
||||
}
|
||||
|
||||
shouldCache := (uint64(chunkView.ViewOffset) + chunkView.ChunkSize) <= c.readerCache.chunkCache.GetMaxFilePartSizeInCache()
|
||||
@@ -380,6 +389,13 @@ func (c *ChunkReadAt) readChunkSliceAt(ctx context.Context, buffer []byte, chunk
|
||||
// readChunkSliceAtForParallel is a simplified version for parallel chunk fetching
|
||||
// It doesn't update lastChunkFid or trigger prefetch (handled by the caller)
|
||||
func (c *ChunkReadAt) readChunkSliceAtForParallel(ctx context.Context, buffer []byte, chunkView *ChunkView, offset uint64) (n int, err error) {
|
||||
if (chunkView.CanRangeFetch() || !c.readerCache.budget.canFit(int(chunkView.ChunkSize))) && !chunkView.IsFullChunk() {
|
||||
n, err = c.readerCache.chunkCache.ReadChunkAt(buffer, chunkView.FileId, offset)
|
||||
if n > 0 {
|
||||
return n, err
|
||||
}
|
||||
return c.readerCache.fetchChunkRange(ctx, buffer, chunkView, int64(offset))
|
||||
}
|
||||
shouldCache := (uint64(chunkView.ViewOffset) + chunkView.ChunkSize) <= c.readerCache.chunkCache.GetMaxFilePartSizeInCache()
|
||||
return c.readerCache.ReadChunkAt(ctx, buffer, chunkView.FileId, chunkView.CipherKey, chunkView.IsGzipped, int64(offset), int(chunkView.ChunkSize), shouldCache)
|
||||
}
|
||||
|
||||
@@ -208,6 +208,201 @@ func TestChunkStreamConcurrentReadsOnOneReader(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
type recordedFetch struct {
|
||||
fileId string
|
||||
isFullChunk bool
|
||||
offset int64
|
||||
size int
|
||||
}
|
||||
|
||||
// fetchRecorder stubs the volume fetch and records how each chunk was
|
||||
// requested: isFullChunk=false is a range fetch of just the view's slice,
|
||||
// isFullChunk=true is a whole-chunk download into the shared cache.
|
||||
func fetchRecorder(rc *ReaderCache) (fetches *[]recordedFetch) {
|
||||
var mu sync.Mutex
|
||||
recorded := &[]recordedFetch{}
|
||||
rc.fetchChunkDataFn = func(_ context.Context, buffer []byte, _ []string, _ []byte, _ bool, isFullChunk bool, offset int64, fileId string, _ util_http.RefreshUrlsFunc) (int, error) {
|
||||
mu.Lock()
|
||||
*recorded = append(*recorded, recordedFetch{fileId, isFullChunk, offset, len(buffer)})
|
||||
mu.Unlock()
|
||||
for i := range buffer {
|
||||
buffer[i] = fileId[len(fileId)-1]
|
||||
}
|
||||
return len(buffer), nil
|
||||
}
|
||||
return recorded
|
||||
}
|
||||
|
||||
// A reader whose views are clipped to a request window — how the S3 gateway
|
||||
// builds a ranged GET — must fetch only the covered part of each chunk:
|
||||
// clipped edge views take range fetches, a fully covered chunk keeps the
|
||||
// shared whole-chunk path. This is what keeps a ranged GET larger than a
|
||||
// buffer from multiplying volume-server reads (issue #11564), without giving
|
||||
// up whole-chunk caching where the whole chunk is actually wanted.
|
||||
func TestChunkReadAtClippedViewsFetchOnlyCoveredParts(t *testing.T) {
|
||||
const chunkSize = 64 << 10
|
||||
|
||||
rc := NewReaderCache(64, (*chunk_cache.TieredChunkCache)(nil), func(context.Context, string) ([]string, error) {
|
||||
return []string{"unused"}, nil
|
||||
}, nil)
|
||||
defer rc.destroy()
|
||||
fetches := fetchRecorder(rc)
|
||||
|
||||
// Window [56KiB, 152KiB): tail of chunk0, all of chunk1, head of
|
||||
// chunk2, head of ciphered chunk3, head of compressed chunk4 (file
|
||||
// chunks need not be aligned).
|
||||
views := NewIntervalList[*ChunkView]()
|
||||
views.AppendInterval(&Interval[*ChunkView]{
|
||||
StartOffset: chunkSize - 8<<10,
|
||||
StopOffset: chunkSize,
|
||||
Value: &ChunkView{FileId: "chunk0", OffsetInChunk: chunkSize - 8<<10, ViewSize: 8 << 10, ViewOffset: chunkSize - 8<<10, ChunkSize: chunkSize},
|
||||
})
|
||||
views.AppendInterval(&Interval[*ChunkView]{
|
||||
StartOffset: chunkSize,
|
||||
StopOffset: 2 * chunkSize,
|
||||
Value: &ChunkView{FileId: "chunk1", ViewSize: chunkSize, ViewOffset: chunkSize, ChunkSize: chunkSize},
|
||||
})
|
||||
views.AppendInterval(&Interval[*ChunkView]{
|
||||
StartOffset: 2 * chunkSize,
|
||||
StopOffset: 2*chunkSize + 8<<10,
|
||||
Value: &ChunkView{FileId: "chunk2", ViewSize: 8 << 10, ViewOffset: 2 * chunkSize, ChunkSize: chunkSize},
|
||||
})
|
||||
views.AppendInterval(&Interval[*ChunkView]{
|
||||
StartOffset: 2*chunkSize + 8<<10,
|
||||
StopOffset: 2*chunkSize + 16<<10,
|
||||
Value: &ChunkView{FileId: "chunk3", ViewSize: 8 << 10, ViewOffset: 2*chunkSize + 8<<10, ChunkSize: chunkSize, CipherKey: []byte("key")},
|
||||
})
|
||||
views.AppendInterval(&Interval[*ChunkView]{
|
||||
StartOffset: 2*chunkSize + 16<<10,
|
||||
StopOffset: 2*chunkSize + 24<<10,
|
||||
Value: &ChunkView{FileId: "chunk4", ViewSize: 8 << 10, ViewOffset: 2*chunkSize + 16<<10, ChunkSize: chunkSize, IsGzipped: true},
|
||||
})
|
||||
|
||||
reader := NewChunkReaderAtFromClient(context.Background(), rc, views, 4*chunkSize, 0)
|
||||
buf := make([]byte, chunkSize+32<<10)
|
||||
if n, err := reader.ReadAt(buf, chunkSize-8<<10); err != nil || n != len(buf) {
|
||||
t.Fatalf("window read: n=%d err=%v", n, err)
|
||||
}
|
||||
// buf holds [56KiB, 152KiB): chunk0's tail, chunk1, and the heads of
|
||||
// chunk2, chunk3 and chunk4.
|
||||
for i, b := range buf {
|
||||
want := byte('1')
|
||||
if i < 8<<10 {
|
||||
want = '0'
|
||||
} else if i >= 24<<10+chunkSize {
|
||||
want = '4'
|
||||
} else if i >= 16<<10+chunkSize {
|
||||
want = '3'
|
||||
} else if i >= 8<<10+chunkSize {
|
||||
want = '2'
|
||||
}
|
||||
if b != want {
|
||||
t.Fatalf("buf[%d]=%q, want %q", i, b, want)
|
||||
}
|
||||
}
|
||||
|
||||
want := []recordedFetch{
|
||||
{fileId: "chunk0", isFullChunk: false, offset: chunkSize - 8<<10, size: 8 << 10},
|
||||
{fileId: "chunk1", isFullChunk: true, offset: 0, size: chunkSize},
|
||||
{fileId: "chunk2", isFullChunk: false, offset: 0, size: 8 << 10},
|
||||
// partial views, but ciphered and compressed chunks download whole
|
||||
// either way and the shared path decrypts/decompresses once for
|
||||
// every buffer
|
||||
{fileId: "chunk3", isFullChunk: true, offset: 0, size: chunkSize},
|
||||
{fileId: "chunk4", isFullChunk: true, offset: 0, size: chunkSize},
|
||||
}
|
||||
got := map[string]recordedFetch{}
|
||||
for _, f := range *fetches {
|
||||
if _, dup := got[f.fileId]; dup {
|
||||
t.Fatalf("chunk %s fetched more than once: %+v", f.fileId, *fetches)
|
||||
}
|
||||
got[f.fileId] = f
|
||||
}
|
||||
for _, w := range want {
|
||||
if g, ok := got[w.fileId]; !ok {
|
||||
t.Fatalf("chunk %s never fetched: %+v", w.fileId, *fetches)
|
||||
} else if g != w {
|
||||
t.Fatalf("chunk %s fetched as %+v, want %+v", w.fileId, g, w)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The regression from issue #11564: a ranged GET sitting inside one big
|
||||
// chunk. Every buffer of the request must stay a range fetch — none may
|
||||
// escalate into a whole-chunk download once the reads look sequential.
|
||||
func TestChunkReadAtRangeInsideOneChunkStaysRangeFetch(t *testing.T) {
|
||||
const chunkSize = 1 << 20
|
||||
const sliceSize = 16 << 10
|
||||
|
||||
rc := NewReaderCache(64, (*chunk_cache.TieredChunkCache)(nil), func(context.Context, string) ([]string, error) {
|
||||
return []string{"unused"}, nil
|
||||
}, nil)
|
||||
defer rc.destroy()
|
||||
fetches := fetchRecorder(rc)
|
||||
|
||||
// Range [32KiB, 96KiB) inside one 1MiB chunk: a single clipped view.
|
||||
views := NewIntervalList[*ChunkView]()
|
||||
views.AppendInterval(&Interval[*ChunkView]{
|
||||
StartOffset: 32 << 10,
|
||||
StopOffset: 96 << 10,
|
||||
Value: &ChunkView{FileId: "chunk0", OffsetInChunk: 32 << 10, ViewSize: 64 << 10, ViewOffset: 32 << 10, ChunkSize: chunkSize},
|
||||
})
|
||||
|
||||
reader := NewChunkReaderAtFromClient(context.Background(), rc, views, chunkSize, 0)
|
||||
for offset := int64(32 << 10); offset < 96<<10; offset += sliceSize {
|
||||
buf := make([]byte, sliceSize)
|
||||
if n, err := reader.ReadAt(buf, offset); err != nil || n != sliceSize {
|
||||
t.Fatalf("read at %d: n=%d err=%v", offset, n, err)
|
||||
}
|
||||
}
|
||||
|
||||
if len(*fetches) != 4 {
|
||||
t.Fatalf("got %d fetches, want 4 range fetches: %+v", len(*fetches), *fetches)
|
||||
}
|
||||
for i, f := range *fetches {
|
||||
wantOffset := int64(32<<10) + int64(i)*sliceSize
|
||||
if f.isFullChunk || f.offset != wantOffset || f.size != sliceSize {
|
||||
t.Fatalf("fetch %d = %+v, want range fetch offset=%d size=%d", i, f, wantOffset, sliceSize)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// A compressed chunk larger than the reader cache budget can never be
|
||||
// downloaded whole — the budget rejects the buffer — so its partial view
|
||||
// must fall back to a range fetch even though each range costs a full
|
||||
// decompress server-side. The alternative is a failed GET.
|
||||
func TestChunkReadAtOversizedCompressedChunkFallsBackToRange(t *testing.T) {
|
||||
const chunkSize = 1 << 20
|
||||
const sliceSize = 16 << 10
|
||||
|
||||
budget := NewReaderCacheBudget(64 << 10) // smaller than the chunk
|
||||
rc := NewReaderCache(64, (*chunk_cache.TieredChunkCache)(nil), func(context.Context, string) ([]string, error) {
|
||||
return []string{"unused"}, nil
|
||||
}, nil, budget)
|
||||
defer rc.destroy()
|
||||
fetches := fetchRecorder(rc)
|
||||
|
||||
views := NewIntervalList[*ChunkView]()
|
||||
views.AppendInterval(&Interval[*ChunkView]{
|
||||
StartOffset: 32 << 10,
|
||||
StopOffset: 64 << 10,
|
||||
Value: &ChunkView{FileId: "chunk0", OffsetInChunk: 32 << 10, ViewSize: 32 << 10, ViewOffset: 32 << 10, ChunkSize: chunkSize, IsGzipped: true},
|
||||
})
|
||||
|
||||
reader := NewChunkReaderAtFromClient(context.Background(), rc, views, chunkSize, 0)
|
||||
buf := make([]byte, 32<<10)
|
||||
if n, err := reader.ReadAt(buf, 32<<10); err != nil || n != len(buf) {
|
||||
t.Fatalf("read: n=%d err=%v", n, err)
|
||||
}
|
||||
|
||||
if len(*fetches) != 1 {
|
||||
t.Fatalf("got %d fetches, want 1 range fetch: %+v", len(*fetches), *fetches)
|
||||
}
|
||||
if f := (*fetches)[0]; f.isFullChunk || f.offset != 32<<10 || f.size != 32<<10 {
|
||||
t.Fatalf("fetch = %+v, want range fetch offset=%d size=%d", f, 32<<10, 32<<10)
|
||||
}
|
||||
}
|
||||
|
||||
// A chunk a stream is positioned in must outlast downloader-limit eviction:
|
||||
// otherwise a busy cache drops the buffer mid-stream and forces a refetch.
|
||||
func TestChunkReadAtPinnedChunkSurvivesEviction(t *testing.T) {
|
||||
|
||||
@@ -100,6 +100,14 @@ func (rc *ReaderCache) MaybeCache(chunkViews *Interval[*ChunkView], count int) {
|
||||
// abort when slots are filled
|
||||
return
|
||||
}
|
||||
if (chunkView.CanRangeFetch() || !rc.budget.canFit(int(chunkView.ChunkSize))) && !chunkView.IsFullChunk() {
|
||||
// the view is clipped to part of the chunk and will be
|
||||
// range-fetched, so prefetching it whole would download bytes
|
||||
// nobody needs; a ciphered or compressed partial view needs
|
||||
// the whole blob anyway and is worth prefetching, but not when
|
||||
// it cannot fit the budget at all
|
||||
continue
|
||||
}
|
||||
|
||||
// glog.V(4).Infof("prefetch %s offset %d", chunkView.FileId, chunkView.ViewOffset)
|
||||
// cache this chunk if not yet
|
||||
@@ -114,6 +122,20 @@ func (rc *ReaderCache) MaybeCache(chunkViews *Interval[*ChunkView], count int) {
|
||||
return
|
||||
}
|
||||
|
||||
// fetchChunkRange downloads only [offset, offset+len(buffer)) of a chunk,
|
||||
// for views clipped to part of their chunk and for random-mode reads. It
|
||||
// goes through fetchChunkDataFn so tests observe range fetches the same way
|
||||
// they observe whole-chunk downloads.
|
||||
func (rc *ReaderCache) fetchChunkRange(ctx context.Context, buffer []byte, chunkView *ChunkView, offset int64) (int, error) {
|
||||
urlStrings, err := rc.lookupFileIdFn(ctx, chunkView.FileId)
|
||||
if err != nil {
|
||||
glog.ErrorfCtx(ctx, "operation LookupFileId %s failed, err: %v", chunkView.FileId, err)
|
||||
return 0, err
|
||||
}
|
||||
return rc.fetchChunkDataFn(ctx, buffer, urlStrings, chunkView.CipherKey, chunkView.IsGzipped, false, offset, chunkView.FileId,
|
||||
refreshUrls(ctx, rc.cacheInvalidator, rc.lookupFileIdFn, chunkView.FileId))
|
||||
}
|
||||
|
||||
// chunkStream is one sequential reader's position in a shared ReaderCache.
|
||||
// The chunk it is reading stays pinned until the stream reads it to the end or
|
||||
// moves elsewhere, so another stream finishing or leaving the same chunk does
|
||||
|
||||
@@ -93,6 +93,13 @@ func (b *ReaderCacheBudget) reserve(s *SingleChunkCacher) error {
|
||||
}
|
||||
}
|
||||
|
||||
// canFit reports whether a whole-chunk buffer of this size can ever be
|
||||
// reserved. A chunk bigger than the budget cannot be read through the
|
||||
// whole-chunk path at all, so callers must fall back to range fetches.
|
||||
func (b *ReaderCacheBudget) canFit(size int) bool {
|
||||
return b == nil || int64(mem.AllocationSize(size)) <= b.limit
|
||||
}
|
||||
|
||||
func (b *ReaderCacheBudget) complete(s *SingleChunkCacher) {
|
||||
if b == nil {
|
||||
return
|
||||
|
||||
@@ -72,7 +72,7 @@ func TestReaderCacheBudgetInFlight(t *testing.T) {
|
||||
return len(buffer), nil
|
||||
}
|
||||
if prefetch {
|
||||
rc.MaybeCache(&Interval[*ChunkView]{Value: &ChunkView{FileId: "chunk", ChunkSize: 3 << 10}}, 1)
|
||||
rc.MaybeCache(&Interval[*ChunkView]{Value: &ChunkView{FileId: "chunk", ViewSize: 3 << 10, ChunkSize: 3 << 10}}, 1)
|
||||
} else {
|
||||
readers.Add(1)
|
||||
go func() {
|
||||
@@ -180,7 +180,7 @@ func TestReaderCacheFailedPrefetchReleasesBudget(t *testing.T) {
|
||||
rc.fetchChunkDataFn = func(_ context.Context, _ []byte, _ []string, _ []byte, _ bool, _ bool, _ int64, _ string, _ util_http.RefreshUrlsFunc) (int, error) {
|
||||
return 0, fmt.Errorf("fetch failed")
|
||||
}
|
||||
rc.MaybeCache(&Interval[*ChunkView]{Value: &ChunkView{FileId: "failed", ChunkSize: 1024}}, 1)
|
||||
rc.MaybeCache(&Interval[*ChunkView]{Value: &ChunkView{FileId: "failed", ViewSize: 1024, ChunkSize: 1024}}, 1)
|
||||
deadline := time.Now().Add(5 * time.Second)
|
||||
for {
|
||||
rc.Lock()
|
||||
@@ -278,7 +278,7 @@ func TestReaderCachePrefetchBufferDroppedAfterRead(t *testing.T) {
|
||||
buffer[0] = 42
|
||||
return len(buffer), nil
|
||||
}
|
||||
rc.MaybeCache(&Interval[*ChunkView]{Value: &ChunkView{FileId: "chunk", ChunkSize: 4 << 10}}, 1)
|
||||
rc.MaybeCache(&Interval[*ChunkView]{Value: &ChunkView{FileId: "chunk", ViewSize: 4 << 10, ChunkSize: 4 << 10}}, 1)
|
||||
|
||||
buf := make([]byte, 4<<10)
|
||||
n, err := rc.ReadChunkAt(context.Background(), buf, "chunk", nil, false, 0, 4<<10, false)
|
||||
|
||||
@@ -54,10 +54,13 @@ func (rp *ReaderPattern) MonitorReadAt(offset int64, size int) {
|
||||
if counter < ModeChangeLimit {
|
||||
atomic.AddInt64(&rp.isSequentialCounter, 1)
|
||||
}
|
||||
} else if counter <= 0 {
|
||||
// Entering random mode is a strong verdict: drop to the bottom of
|
||||
// the window so the contiguous tail of one ranged request cannot
|
||||
// flip it back on the next buffer read and pay a whole-chunk fetch.
|
||||
atomic.StoreInt64(&rp.isSequentialCounter, -ModeChangeLimit)
|
||||
} else {
|
||||
if counter > -ModeChangeLimit {
|
||||
atomic.AddInt64(&rp.isSequentialCounter, -1)
|
||||
}
|
||||
atomic.AddInt64(&rp.isSequentialCounter, -1)
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user