mirror of
https://github.com/seaweedfs/seaweedfs.git
synced 2026-10-03 21:21:57 +00:00
Compare commits
74
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
0b5d1c0c64 | ||
|
|
56468c83e4 | ||
|
|
bfe4b810bf | ||
|
|
618febdba5 | ||
|
|
ed9d1eec64 | ||
|
|
f07aabb39f | ||
|
|
17fe96e620 | ||
|
|
35e9f84334 | ||
|
|
10b64686ba | ||
|
|
045c834dcf | ||
|
|
318e1c64d6 | ||
|
|
2c84bb1161 | ||
|
|
61348b147b | ||
|
|
a7fec8004e | ||
|
|
b65bcd4afa | ||
|
|
79297b549e | ||
|
|
4c7e5afbfe | ||
|
|
9cb7dc7204 | ||
|
|
77a2b1b378 | ||
|
|
7387866fd6 | ||
|
|
4bf126944f | ||
|
|
1737211ffa | ||
|
|
4ee57f214c | ||
|
|
211bf4d2fe | ||
|
|
deb8b9bef1 | ||
|
|
53fe128511 | ||
|
|
6863412f4e | ||
|
|
567052bfb6 | ||
|
|
38db7e1493 | ||
|
|
923d0bd20c | ||
|
|
2d9ea0285c | ||
|
|
67b0cc0706 | ||
|
|
2dc59c9b51 | ||
|
|
0dd33bd7ec | ||
|
|
a2ffc7aadf | ||
|
|
25d7f62749 | ||
|
|
3a61debaa5 | ||
|
|
37f3dff677 | ||
|
|
9d11278d95 | ||
|
|
344ac7684e | ||
|
|
e9cde3e4b1 | ||
|
|
5b9236c76d | ||
|
|
ce7d388639 | ||
|
|
213eb4c23a | ||
|
|
cab666fca1 | ||
|
|
457277ec9a | ||
|
|
08f0ba5564 | ||
|
|
4527947afc | ||
|
|
0b78381513 | ||
|
|
5ec813b4f1 | ||
|
|
75ae33ade8 | ||
|
|
dd73fee077 | ||
|
|
506ce0850b | ||
|
|
6d08b08f37 | ||
|
|
5532a316c5 | ||
|
|
553bc5ab90 | ||
|
|
12627d376d | ||
|
|
a49cf11e16 | ||
|
|
b46946ece5 | ||
|
|
af7cf6ab8a | ||
|
|
cce3bab0e2 | ||
|
|
228e850da1 | ||
|
|
9f1e21e73f | ||
|
|
0cfca436f1 | ||
|
|
8aa57bef78 | ||
|
|
ee54fd6c08 | ||
|
|
33c36fc7a3 | ||
|
|
4f0322af86 | ||
|
|
1d8d9570eb | ||
|
|
2ec899bdee | ||
|
|
3fce1a938d | ||
|
|
2ff3dda7cd | ||
|
|
fb92d46e2d | ||
|
|
a8c8372b99 |
@@ -1296,6 +1296,180 @@ jobs:
|
||||
PYEOF
|
||||
echo "NetworkPolicy rendering tests passed"
|
||||
|
||||
echo ""
|
||||
echo "=== Testing enterprise license mount + master persistence ==="
|
||||
python3 - "$CHART_DIR" <<'PYEOF'
|
||||
import subprocess, sys, yaml
|
||||
chart = sys.argv[1]
|
||||
|
||||
def render(values):
|
||||
args = ["helm", "template", "test", chart]
|
||||
for k, v in values.items():
|
||||
args += ["--set", f"{k}={v}"]
|
||||
return subprocess.check_output(args, text=True)
|
||||
|
||||
def workloads(manifest):
|
||||
for d in yaml.safe_load_all(manifest):
|
||||
if d and d.get("kind") in ("Deployment", "StatefulSet"):
|
||||
yield d
|
||||
|
||||
ALL_ON = {
|
||||
"admin.enabled": "true",
|
||||
"s3.enabled": "true",
|
||||
"sftp.enabled": "true",
|
||||
"worker.enabled": "true",
|
||||
}
|
||||
failed = []
|
||||
|
||||
# Case 1: unconfigured renders nothing, so existing installs are unaffected.
|
||||
out = render(ALL_ON)
|
||||
if "SEAWEED_LICENSE" in out or "seaweedfs-license" in out:
|
||||
failed.append("defaults: license plumbing should not render")
|
||||
else:
|
||||
print("defaults: no license volume/env (unchanged)")
|
||||
|
||||
# Case 2: only the workloads that run a master carry the license.
|
||||
LICENSE_PATH = "/etc/seaweedfs/license/seaweed-license.json"
|
||||
out = render(dict(ALL_ON, **{
|
||||
"global.seaweedfs.license.existingSecret": "lic",
|
||||
}))
|
||||
carrying = set()
|
||||
for d in workloads(out):
|
||||
name = d["metadata"]["name"]
|
||||
pod = d["spec"]["template"]["spec"]
|
||||
vols = [v for v in pod.get("volumes", []) if v.get("name") == "seaweedfs-license"]
|
||||
if not vols:
|
||||
continue
|
||||
carrying.add(name)
|
||||
secret = vols[0].get("secret", {})
|
||||
if secret.get("secretName") != "lic":
|
||||
failed.append(f"{name}: license volume points at {secret.get('secretName')!r}, not the configured Secret")
|
||||
items = [i.get("key") for i in secret.get("items", [])]
|
||||
if items != ["seaweed-license.json"]:
|
||||
failed.append(f"{name}: license volume projects {items}, expected only the license key")
|
||||
for c in pod.get("containers", []):
|
||||
mounts = [m for m in c.get("volumeMounts", []) if m["name"] == "seaweedfs-license"]
|
||||
if not mounts:
|
||||
failed.append(f"{name}/{c['name']}: license volume not mounted")
|
||||
continue
|
||||
# subPath would freeze the file at container start.
|
||||
if any("subPath" in m for m in mounts):
|
||||
failed.append(f"{name}/{c['name']}: license mount uses subPath (renewals would not propagate)")
|
||||
if not all(m.get("readOnly") for m in mounts):
|
||||
failed.append(f"{name}/{c['name']}: license mount is not readOnly")
|
||||
env = {e["name"]: e.get("value") for e in c.get("env", [])}
|
||||
if env.get("SEAWEED_LICENSE") != LICENSE_PATH:
|
||||
failed.append(f"{name}/{c['name']}: SEAWEED_LICENSE not set to the mounted file")
|
||||
if carrying != {"test-seaweedfs-master"}:
|
||||
failed.append(f"license: expected only the master to carry it, got {sorted(carrying)}")
|
||||
else:
|
||||
print("license: master only, key-scoped, readOnly, no subPath, SEAWEED_LICENSE set")
|
||||
|
||||
# all-in-one runs `weed server -master`, so it needs the license too.
|
||||
out = render({
|
||||
"allInOne.enabled": "true",
|
||||
"master.enabled": "false",
|
||||
"volume.enabled": "false",
|
||||
"filer.enabled": "false",
|
||||
"global.seaweedfs.license.existingSecret": "lic",
|
||||
})
|
||||
aio = [d for d in workloads(out) if d["metadata"]["name"].endswith("-all-in-one")]
|
||||
if len(aio) != 1:
|
||||
failed.append(f"all-in-one: expected 1 workload, got {len(aio)}")
|
||||
else:
|
||||
pod = aio[0]["spec"]["template"]["spec"]
|
||||
if not any(v.get("name") == "seaweedfs-license" for v in pod.get("volumes", [])):
|
||||
failed.append("all-in-one: missing the license volume")
|
||||
else:
|
||||
print("all-in-one: carries the license")
|
||||
|
||||
# SEAWEED_LICENSE is reserved on both env paths.
|
||||
out = render({
|
||||
"global.seaweedfs.license.existingSecret": "lic",
|
||||
"global.seaweedfs.extraEnvironmentVars.SEAWEED_LICENSE": "/tmp/bogus.json",
|
||||
})
|
||||
for d in workloads(out):
|
||||
if not d["metadata"]["name"].endswith("-master"):
|
||||
continue
|
||||
for c in d["spec"]["template"]["spec"].get("containers", []):
|
||||
names = [e["name"] for e in c.get("env", [])]
|
||||
if names.count("SEAWEED_LICENSE") != 1:
|
||||
failed.append(f"SEAWEED_LICENSE rendered {names.count('SEAWEED_LICENSE')} times")
|
||||
else:
|
||||
value = next(e.get("value") for e in c["env"] if e["name"] == "SEAWEED_LICENSE")
|
||||
if value != LICENSE_PATH:
|
||||
failed.append(f"SEAWEED_LICENSE overridden to {value!r} by extraEnvironmentVars")
|
||||
else:
|
||||
print("SEAWEED_LICENSE: reserved, rendered once, points at the mount")
|
||||
|
||||
# secretExtraEnvironmentVars is rendered outside the merge helper.
|
||||
out = render({
|
||||
"allInOne.enabled": "true",
|
||||
"master.enabled": "false",
|
||||
"volume.enabled": "false",
|
||||
"filer.enabled": "false",
|
||||
"global.seaweedfs.license.existingSecret": "lic",
|
||||
"allInOne.secretExtraEnvironmentVars.SEAWEED_LICENSE.secretKeyRef.name": "x",
|
||||
"allInOne.secretExtraEnvironmentVars.SEAWEED_LICENSE.secretKeyRef.key": "y",
|
||||
})
|
||||
for d in workloads(out):
|
||||
if not d["metadata"]["name"].endswith("-all-in-one"):
|
||||
continue
|
||||
for c in d["spec"]["template"]["spec"].get("containers", []):
|
||||
names = [e["name"] for e in c.get("env", [])]
|
||||
if names.count("SEAWEED_LICENSE") != 1:
|
||||
failed.append(f"all-in-one: SEAWEED_LICENSE rendered {names.count('SEAWEED_LICENSE')} times via secretExtraEnvironmentVars")
|
||||
elif next(e.get("value") for e in c["env"] if e["name"] == "SEAWEED_LICENSE") != LICENSE_PATH:
|
||||
failed.append("all-in-one: secretExtraEnvironmentVars overrode SEAWEED_LICENSE")
|
||||
else:
|
||||
print("all-in-one: SEAWEED_LICENSE reserved on the secret env path too")
|
||||
|
||||
# Case 3: the default stays hostPath. volumeClaimTemplates is immutable,
|
||||
# so growing one here would break helm upgrade on existing releases.
|
||||
masters = [d for d in workloads(render({})) if d["metadata"]["name"].endswith("-master")]
|
||||
if len(masters) != 1:
|
||||
failed.append(f"master: expected 1 StatefulSet by default, got {len(masters)}")
|
||||
else:
|
||||
d = masters[0]
|
||||
claims = [c["metadata"]["name"] for c in d["spec"].get("volumeClaimTemplates", [])]
|
||||
if any(c.startswith("data-") for c in claims):
|
||||
failed.append(f"master: default grew a data volumeClaimTemplate ({claims}); "
|
||||
"that breaks helm upgrade on existing releases")
|
||||
data_host = [v["name"] for v in d["spec"]["template"]["spec"].get("volumes", [])
|
||||
if v.get("hostPath") and v["name"].startswith("data-")]
|
||||
if not data_host:
|
||||
failed.append("master: default no longer renders a hostPath data volume")
|
||||
else:
|
||||
print("master: default still hostPath (upgrade-compatible)")
|
||||
|
||||
# Case 4: opting into a claim renders one, with the requested size.
|
||||
masters = [d for d in workloads(render({"master.data.type": "persistentVolumeClaim"}))
|
||||
if d["metadata"]["name"].endswith("-master")]
|
||||
if len(masters) != 1:
|
||||
failed.append(f"master.data.type=persistentVolumeClaim: expected 1 StatefulSet, got {len(masters)}")
|
||||
else:
|
||||
d = masters[0]
|
||||
claims = {c["metadata"]["name"]: c["spec"]["resources"]["requests"]["storage"]
|
||||
for c in d["spec"].get("volumeClaimTemplates", [])}
|
||||
data = {k: v for k, v in claims.items() if k.startswith("data-")}
|
||||
if not data:
|
||||
failed.append(f"master.data.type=persistentVolumeClaim: no data claim rendered ({claims})")
|
||||
elif set(data.values()) != {"1Gi"}:
|
||||
failed.append(f"master.data.type=persistentVolumeClaim: unexpected size {data}")
|
||||
else:
|
||||
print(f"master.data.type=persistentVolumeClaim: renders {sorted(data)} at 1Gi")
|
||||
if any(v.get("hostPath") and v["name"].startswith("data-")
|
||||
for v in d["spec"]["template"]["spec"].get("volumes", [])):
|
||||
failed.append("master.data.type=persistentVolumeClaim: data still on a hostPath")
|
||||
|
||||
if failed:
|
||||
print("\nFAIL:", file=sys.stderr)
|
||||
for f in failed:
|
||||
print(f" - {f}", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
PYEOF
|
||||
echo "License + master persistence tests passed"
|
||||
|
||||
echo "All template rendering tests passed!"
|
||||
|
||||
- name: Create kind cluster
|
||||
|
||||
@@ -8,6 +8,7 @@ on:
|
||||
- 'weed/pb/filer_pb/**'
|
||||
- 'weed/util/log_buffer/**'
|
||||
- 'weed/server/filer_grpc_server_sub_meta.go'
|
||||
- 'weed/server/master_grpc_server.go'
|
||||
- 'weed/command/filer_backup.go'
|
||||
- 'test/metadata_subscribe/**'
|
||||
- '.github/workflows/metadata-subscribe-tests.yml'
|
||||
@@ -18,6 +19,7 @@ on:
|
||||
- 'weed/pb/filer_pb/**'
|
||||
- 'weed/util/log_buffer/**'
|
||||
- 'weed/server/filer_grpc_server_sub_meta.go'
|
||||
- 'weed/server/master_grpc_server.go'
|
||||
- 'weed/command/filer_backup.go'
|
||||
- 'test/metadata_subscribe/**'
|
||||
- '.github/workflows/metadata-subscribe-tests.yml'
|
||||
|
||||
@@ -353,6 +353,79 @@ jobs:
|
||||
path: test/s3tables/catalog_doris/test-output.log
|
||||
retention-days: 3
|
||||
|
||||
clickhouse-iceberg-catalog-tests:
|
||||
name: ClickHouse Iceberg Catalog Integration Tests
|
||||
runs-on: ubuntu-22.04
|
||||
timeout-minutes: 30
|
||||
|
||||
steps:
|
||||
- name: Check out code
|
||||
uses: actions/checkout@v7
|
||||
|
||||
- name: Set up Go
|
||||
uses: actions/setup-go@v7
|
||||
with:
|
||||
go-version-file: 'go.mod'
|
||||
id: go
|
||||
|
||||
- name: Configure Docker Hub mirror
|
||||
run: |
|
||||
echo '{"registry-mirrors": ["https://mirror.gcr.io"]}' | sudo tee /etc/docker/daemon.json
|
||||
sudo systemctl restart docker
|
||||
|
||||
- name: Pre-pull images
|
||||
run: |
|
||||
pull() { for i in 1 2 3; do docker pull "$1" && return 0; sleep 15; done; return 1; }
|
||||
pull clickhouse/clickhouse-server:25.8
|
||||
pull python:3.11-slim
|
||||
|
||||
- name: Run go mod tidy
|
||||
run: go mod tidy
|
||||
|
||||
- name: Install SeaweedFS
|
||||
run: |
|
||||
go install -buildvcs=false ./weed
|
||||
|
||||
- name: Run ClickHouse Iceberg Catalog Integration Tests
|
||||
timeout-minutes: 25
|
||||
working-directory: test/s3tables/catalog_clickhouse
|
||||
run: |
|
||||
set -x
|
||||
set -o pipefail
|
||||
echo "=== System Information ==="
|
||||
uname -a
|
||||
free -h
|
||||
df -h
|
||||
docker info
|
||||
echo "=== Starting ClickHouse Iceberg Catalog Tests ==="
|
||||
|
||||
go test -v -timeout 20m . 2>&1 | tee test-output.log || {
|
||||
echo "ClickHouse Iceberg catalog integration tests failed"
|
||||
exit 1
|
||||
}
|
||||
|
||||
- name: Show test output on failure
|
||||
if: failure()
|
||||
working-directory: test/s3tables/catalog_clickhouse
|
||||
run: |
|
||||
echo "=== Test Output ==="
|
||||
if [ -f test-output.log ]; then
|
||||
tail -200 test-output.log
|
||||
fi
|
||||
|
||||
echo "=== Process information ==="
|
||||
ps aux | grep -E "(weed|test|docker|clickhouse)" || true
|
||||
echo "=== ClickHouse containers ==="
|
||||
docker ps -a --filter "name=seaweed-clickhouse" || true
|
||||
|
||||
- name: Upload test logs on failure
|
||||
if: failure()
|
||||
uses: actions/upload-artifact@v7
|
||||
with:
|
||||
name: clickhouse-iceberg-catalog-test-logs
|
||||
path: test/s3tables/catalog_clickhouse/test-output.log
|
||||
retention-days: 3
|
||||
|
||||
polaris-integration-tests:
|
||||
name: Polaris Integration Tests
|
||||
runs-on: ubuntu-22.04
|
||||
|
||||
@@ -11,7 +11,6 @@ require (
|
||||
github.com/beorn7/perks v1.0.1 // indirect
|
||||
github.com/bwmarrin/snowflake v0.3.0
|
||||
github.com/cenkalti/backoff/v4 v4.3.0
|
||||
github.com/cespare/xxhash/v2 v2.3.0 // indirect
|
||||
github.com/coreos/go-semver v0.3.1 // indirect
|
||||
github.com/coreos/go-systemd/v22 v22.6.0 // indirect
|
||||
github.com/davecgh/go-spew v1.1.2-0.20180830191138-d8f796af33cc // indirect
|
||||
@@ -127,6 +126,7 @@ require (
|
||||
github.com/aws/aws-sdk-go-v2/config v1.32.33
|
||||
github.com/aws/aws-sdk-go-v2/credentials v1.19.32
|
||||
github.com/aws/aws-sdk-go-v2/service/s3 v1.105.2
|
||||
github.com/cespare/xxhash/v2 v2.3.0
|
||||
github.com/cognusion/imaging v1.0.4
|
||||
github.com/fluent/fluent-logger-golang v1.10.1
|
||||
github.com/getsentry/sentry-go v0.44.1
|
||||
|
||||
@@ -107,6 +107,120 @@ https://github.com/rancher/local-path-provisioner
|
||||
you can use ANY storage class you like, just update the correct storage-class
|
||||
for your deployment.
|
||||
|
||||
### Master data: hostPath vs a claim
|
||||
|
||||
The master's `-mdir` holds its Raft log and snapshots, and with them the
|
||||
cluster's identity (its topology UUID). `master.data.type` defaults to
|
||||
`hostPath`, which does not follow a pod to another node: a master that is
|
||||
rescheduled comes back with an empty data directory and a brand new cluster
|
||||
UUID. With the chart's default of a single master replica there is no peer to
|
||||
recover the identity from either.
|
||||
|
||||
Putting the master's data on a claim avoids that:
|
||||
|
||||
```yaml
|
||||
master:
|
||||
data:
|
||||
type: "persistentVolumeClaim"
|
||||
size: "1Gi"
|
||||
storageClass: "" # empty uses the cluster's default StorageClass
|
||||
```
|
||||
|
||||
Raft state is small, so a modest claim is enough — sizing matters far more for
|
||||
volume and filer.
|
||||
|
||||
The default is left at `hostPath` for backward compatibility:
|
||||
`volumeClaimTemplates` is immutable on a StatefulSet, so flipping the type on a
|
||||
release that already exists fails, whether the chart changes the default or you
|
||||
change it yourself:
|
||||
|
||||
```text
|
||||
StatefulSet.apps "<release>-seaweedfs-master" is invalid: spec: Forbidden:
|
||||
updates to statefulset spec for fields other than 'replicas', ... are forbidden
|
||||
```
|
||||
|
||||
New installs can set the claim from the start. To move an **existing** release
|
||||
onto a claim without losing the cluster UUID, use the migration below. The
|
||||
claim has to be seeded while the master is stopped: a running master rewrites
|
||||
its Raft state, so copying into a live pod is silently undone by the next
|
||||
restart.
|
||||
|
||||
The steps below are for the chart's default of a single master
|
||||
(`master.replicas: 1`). With several master replicas, repeat steps 1, 3 and 4
|
||||
for every ordinal, or migrate one at a time and let the remaining quorum
|
||||
re-replicate.
|
||||
|
||||
Take the names from the cluster rather than assembling them — the release
|
||||
name, `nameOverride` and `fullnameOverride` all feed the chart's fullname
|
||||
helper, so `<release>-seaweedfs` is not always right:
|
||||
|
||||
```bash
|
||||
NS=<namespace>; REL=<release>
|
||||
# scope by instance as well as component: several releases can share a namespace
|
||||
STS=$(kubectl -n $NS get sts \
|
||||
-l app.kubernetes.io/instance=$REL,app.kubernetes.io/component=master \
|
||||
-o jsonpath='{.items[0].metadata.name}')
|
||||
POD=$STS-0
|
||||
# a StatefulSet names its claims <template>-<statefulset>-<ordinal>, and this
|
||||
# chart's template is data-<namespace>
|
||||
PVC=data-$NS-$STS-0
|
||||
|
||||
# 1. back up the master data directory
|
||||
kubectl -n $NS cp $POD:/data ./master-backup
|
||||
|
||||
# 2. stop the master, leaving the rest of the release running
|
||||
kubectl -n $NS delete sts $STS --cascade=orphan
|
||||
kubectl -n $NS delete pod $POD
|
||||
|
||||
# 3. create the claim the new StatefulSet will adopt, and seed it through a
|
||||
# pod that actually mounts it
|
||||
kubectl -n $NS apply -f - <<EOF
|
||||
apiVersion: v1
|
||||
kind: PersistentVolumeClaim
|
||||
metadata:
|
||||
name: $PVC
|
||||
spec:
|
||||
accessModes: ["ReadWriteOnce"]
|
||||
resources:
|
||||
requests:
|
||||
storage: 1Gi
|
||||
---
|
||||
apiVersion: v1
|
||||
kind: Pod
|
||||
metadata:
|
||||
name: seed
|
||||
spec:
|
||||
containers:
|
||||
- name: seed
|
||||
image: alpine:3.20
|
||||
command: ["sleep", "600"]
|
||||
volumeMounts:
|
||||
- name: d
|
||||
mountPath: /data
|
||||
volumes:
|
||||
- name: d
|
||||
persistentVolumeClaim:
|
||||
claimName: $PVC
|
||||
EOF
|
||||
kubectl -n $NS wait --for=condition=Ready pod/seed --timeout=120s
|
||||
kubectl -n $NS cp ./master-backup/m9333 seed:/data/
|
||||
kubectl -n $NS exec seed -- ls /data/m9333 # conf, log, snapshot, state
|
||||
kubectl -n $NS delete pod seed
|
||||
|
||||
# 4. upgrade; the StatefulSet adopts the claim you created
|
||||
helm upgrade $REL seaweedfs/seaweedfs -n $NS -f values.yaml
|
||||
```
|
||||
|
||||
Confirm the UUID survived — it must match what the cluster reported before the
|
||||
migration:
|
||||
|
||||
```bash
|
||||
kubectl -n $NS exec $POD -- curl -s localhost:9333/license/status
|
||||
```
|
||||
|
||||
**Or accept a new cluster UUID** and, if you run the enterprise edition, have
|
||||
the license re-issued against it.
|
||||
|
||||
## current instances config (AIO):
|
||||
|
||||
1 instance for each type (master/filer+s3/volume)
|
||||
@@ -426,3 +540,38 @@ helm install seaweedfs seaweedfs/seaweedfs \
|
||||
|
||||
For enterprise users, please visit [seaweedfs.com](https://seaweedfs.com) for the SeaweedFS Enterprise Edition,
|
||||
which has advanced features, including data recovery, self-healing storage, customizable erasure coding, EC vacuum and repair, etc.
|
||||
|
||||
To run it, set the image and point the chart at a Secret holding the license
|
||||
file:
|
||||
|
||||
```bash
|
||||
kubectl create secret generic seaweedfs-license -n <namespace> \
|
||||
--from-file=seaweed-license.json=/path/to/seaweed-license.json
|
||||
```
|
||||
|
||||
```yaml
|
||||
global:
|
||||
seaweedfs:
|
||||
image:
|
||||
name: chrislusf/seaweedfs-enterprise
|
||||
license:
|
||||
existingSecret: seaweedfs-license
|
||||
# secretKey: seaweed-license.json # key within the Secret
|
||||
# mountPath: /etc/seaweedfs/license # directory it is mounted at
|
||||
```
|
||||
|
||||
Set the image globally rather than per component: a per-component
|
||||
`imageOverride` wins, and a cluster that mixes editions comes up looking
|
||||
healthy with enterprise features quietly off.
|
||||
|
||||
Only the master reads the license, so the Secret is mounted read-only there
|
||||
and on all-in-one (which runs `weed server -master`). It is mounted as a
|
||||
directory, not a `subPath`, so a renewed Secret reaches the running master —
|
||||
which re-reads the file periodically — without a restart.
|
||||
|
||||
The license is tied to the cluster UUID kept in the master's Raft state, so put
|
||||
`master.data` on a claim — a master that restarts onto an empty data directory
|
||||
generates a new UUID and the license stops matching. See
|
||||
[Master data](#master-data-hostpath-vs-a-claim). Check the binding with
|
||||
`kubectl exec <master-pod> -- curl -s localhost:9333/license/status`
|
||||
(`cluster_uuid` must equal `license_uuid`).
|
||||
|
||||
@@ -80,6 +80,7 @@ spec:
|
||||
image: {{ template "seaweedfs.master.image" . }}
|
||||
imagePullPolicy: {{ default "IfNotPresent" .Values.global.seaweedfs.imagePullPolicy }}
|
||||
env:
|
||||
{{- include "seaweedfs.licenseEnv" . | nindent 12 }}
|
||||
{{- /* Determine default cluster alias and the corresponding env var keys to avoid conflicts */}}
|
||||
{{- $mergedExtraEnvironmentVars := dict }}
|
||||
{{- include "seaweedfs.mergeExtraEnvironmentVars" (dict "global" .Values.global.seaweedfs "component" .Values.allInOne "target" $mergedExtraEnvironmentVars) }}
|
||||
@@ -120,11 +121,13 @@ spec:
|
||||
value: {{ include "seaweedfs.cluster.filerAddress" . | quote }}
|
||||
{{- if .Values.allInOne.secretExtraEnvironmentVars }}
|
||||
{{- range $key, $value := .Values.allInOne.secretExtraEnvironmentVars }}
|
||||
{{- if not (and $key (eq $key "SEAWEED_LICENSE") (include "seaweedfs.licenseEnabled" $)) }}
|
||||
- name: {{ $key }}
|
||||
valueFrom:
|
||||
{{ toYaml $value | nindent 16 }}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
command:
|
||||
- "/bin/sh"
|
||||
- "-ec"
|
||||
@@ -352,6 +355,7 @@ spec:
|
||||
{{- include "seaweedfs.s3.tlsVolumeMount" . | nindent 12 }}
|
||||
{{- end }}
|
||||
{{- end }}
|
||||
{{- include "seaweedfs.licenseVolumeMount" . | nindent 12 }}
|
||||
{{ tpl .Values.allInOne.extraVolumeMounts . | nindent 12 }}
|
||||
ports:
|
||||
- containerPort: {{ .Values.master.port }}
|
||||
@@ -419,6 +423,7 @@ spec:
|
||||
{{- include "seaweedfs.tplvalues.render" (dict "value" .Values.allInOne.sidecars "context" $) | nindent 8 }}
|
||||
{{- end }}
|
||||
volumes:
|
||||
{{- include "seaweedfs.licenseVolume" . | nindent 8 }}
|
||||
- name: data
|
||||
{{- if eq .Values.allInOne.data.type "hostPath" }}
|
||||
hostPath:
|
||||
|
||||
@@ -83,6 +83,7 @@ spec:
|
||||
image: {{ template "seaweedfs.master.image" . }}
|
||||
imagePullPolicy: {{ default "IfNotPresent" .Values.global.seaweedfs.imagePullPolicy }}
|
||||
env:
|
||||
{{- include "seaweedfs.licenseEnv" . | nindent 12 }}
|
||||
- name: POD_IP
|
||||
valueFrom:
|
||||
fieldRef:
|
||||
@@ -211,6 +212,7 @@ spec:
|
||||
readOnly: true
|
||||
mountPath: /usr/local/share/ca-certificates/client/
|
||||
{{- end }}
|
||||
{{- include "seaweedfs.licenseVolumeMount" . | nindent 12 }}
|
||||
{{ tpl .Values.master.extraVolumeMounts . | nindent 12 | trim }}
|
||||
ports:
|
||||
- containerPort: {{ .Values.master.port }}
|
||||
@@ -256,6 +258,7 @@ spec:
|
||||
{{- include "seaweedfs.tplvalues.render" (dict "value" .Values.master.sidecars "context" $) | nindent 8 }}
|
||||
{{- end }}
|
||||
volumes:
|
||||
{{- include "seaweedfs.licenseVolume" . | nindent 8 }}
|
||||
{{- if eq .Values.master.logs.type "hostPath" }}
|
||||
- name: seaweedfs-master-log-volume
|
||||
hostPath:
|
||||
|
||||
@@ -69,6 +69,11 @@ Inject extra environment vars in the format key:value, if populated
|
||||
{{- range $key, $value := $component }}
|
||||
{{- $_ := set $target $key $value }}
|
||||
{{- end }}
|
||||
{{/* the license block owns SEAWEED_LICENSE; letting one through here too would
|
||||
render the key twice in one container */}}
|
||||
{{- if ((.global | default dict).license | default dict).existingSecret }}
|
||||
{{- $_ := unset $target "SEAWEED_LICENSE" }}
|
||||
{{- end }}
|
||||
{{- end -}}
|
||||
|
||||
{{/* Return the proper filer image */}}
|
||||
@@ -450,6 +455,47 @@ true
|
||||
{{- end }}
|
||||
{{- end -}}
|
||||
|
||||
{{/* True when an enterprise license Secret is configured. */}}
|
||||
{{- define "seaweedfs.licenseEnabled" -}}
|
||||
{{- if ((.Values.global.seaweedfs).license).existingSecret -}}
|
||||
true
|
||||
{{- end -}}
|
||||
{{- end -}}
|
||||
|
||||
{{/* Enterprise license volume. Projects just the license key. */}}
|
||||
{{- define "seaweedfs.licenseVolume" -}}
|
||||
{{- if include "seaweedfs.licenseEnabled" . -}}
|
||||
- name: seaweedfs-license
|
||||
secret:
|
||||
secretName: {{ .Values.global.seaweedfs.license.existingSecret }}
|
||||
defaultMode: 0444
|
||||
items:
|
||||
- key: {{ .Values.global.seaweedfs.license.secretKey | default "seaweed-license.json" | quote }}
|
||||
path: {{ .Values.global.seaweedfs.license.secretKey | default "seaweed-license.json" | quote }}
|
||||
{{- end }}
|
||||
{{- end -}}
|
||||
|
||||
{{/* Enterprise license volume mount. Never a subPath: that is resolved once at
|
||||
container start, so a renewed Secret would not reach a running master. */}}
|
||||
{{- define "seaweedfs.licenseVolumeMount" -}}
|
||||
{{- if include "seaweedfs.licenseEnabled" . -}}
|
||||
- name: seaweedfs-license
|
||||
readOnly: true
|
||||
mountPath: {{ .Values.global.seaweedfs.license.mountPath | default "/etc/seaweedfs/license" | quote }}
|
||||
{{- end }}
|
||||
{{- end -}}
|
||||
|
||||
{{/* SEAWEED_LICENSE, set explicitly rather than relying on the binary's search
|
||||
paths, which depend on the working directory. */}}
|
||||
{{- define "seaweedfs.licenseEnv" -}}
|
||||
{{- if include "seaweedfs.licenseEnabled" . -}}
|
||||
- name: SEAWEED_LICENSE
|
||||
value: {{ printf "%s/%s"
|
||||
(.Values.global.seaweedfs.license.mountPath | default "/etc/seaweedfs/license")
|
||||
(.Values.global.seaweedfs.license.secretKey | default "seaweed-license.json") | quote }}
|
||||
{{- end }}
|
||||
{{- end -}}
|
||||
|
||||
{{/* Generate a compatible trafficDistribution value due to "PreferClose" fast deprecation in k8s v1.35.
|
||||
Accepts a dict with "value" (the trafficDistribution string) and "Capabilities". */}}
|
||||
{{- define "seaweedfs.trafficDistribution" -}}
|
||||
|
||||
@@ -12,7 +12,16 @@ global:
|
||||
image:
|
||||
# if repository is set, it overrides the namespace part of image.name
|
||||
repository: ""
|
||||
# chrislusf/seaweedfs-enterprise for the enterprise edition
|
||||
name: chrislusf/seaweedfs
|
||||
# Enterprise license file, held in a Secret in the release namespace and
|
||||
# mounted read-only on the components that run a master. A renewed Secret
|
||||
# reaches the running master without a restart. SEAWEED_LICENSE is reserved
|
||||
# while this is set: an extraEnvironmentVars entry of that name is dropped.
|
||||
license:
|
||||
existingSecret: ""
|
||||
secretKey: seaweed-license.json
|
||||
mountPath: /etc/seaweedfs/license
|
||||
imagePullPolicy: IfNotPresent
|
||||
restartPolicy: Always
|
||||
loggingLevel: 1
|
||||
@@ -133,8 +142,13 @@ master:
|
||||
# You can also use emptyDir storage:
|
||||
# data:
|
||||
# type: "emptyDir"
|
||||
# -mdir holds the Raft log and snapshots, and with them the cluster's
|
||||
# identity (its topology UUID). A hostPath does not follow a rescheduled pod,
|
||||
# so prefer type "persistentVolumeClaim" for new installs. The default stays
|
||||
# hostPath so existing releases keep upgrading; see the README to switch one.
|
||||
data:
|
||||
type: "hostPath"
|
||||
size: "1Gi"
|
||||
storageClass: ""
|
||||
hostPathPrefix: /ssd
|
||||
|
||||
|
||||
Executable
BIN
Binary file not shown.
@@ -136,6 +136,10 @@ message ListEntriesRequest {
|
||||
bool inclusiveStartFrom = 4;
|
||||
uint32 limit = 5;
|
||||
int64 snapshot_ts_ns = 6;
|
||||
// Leave the chunk list out of every entry in the response. The attributes
|
||||
// still carry the file size, so a listing that only reads attributes can
|
||||
// ask for this and skip the largest part of the payload.
|
||||
bool omit_chunks = 7;
|
||||
}
|
||||
|
||||
message ListEntriesResponse {
|
||||
|
||||
Generated
+7
@@ -4556,6 +4556,7 @@ dependencies = [
|
||||
"tracing-subscriber",
|
||||
"uuid",
|
||||
"x509-parser",
|
||||
"xxhash-rust",
|
||||
]
|
||||
|
||||
[[package]]
|
||||
@@ -4989,6 +4990,12 @@ version = "0.13.6"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "66fee0b777b0f5ac1c69bb06d361268faafa61cd4682ae064a171c16c433e9e4"
|
||||
|
||||
[[package]]
|
||||
name = "xxhash-rust"
|
||||
version = "0.8.18"
|
||||
source = "registry+https://github.com/rust-lang/crates.io-index"
|
||||
checksum = "aee1b19627c7c60102ab80d3a9cbe18de90bfe03bfa6c3715447681f0e8c8af6"
|
||||
|
||||
[[package]]
|
||||
name = "yoke"
|
||||
version = "0.8.2"
|
||||
|
||||
@@ -75,6 +75,9 @@ serde_urlencoded = "0.7"
|
||||
crc32c = "0.6"
|
||||
crc32fast = "1"
|
||||
|
||||
# xxhash64 - must match Go's cespare/xxhash for the heartbeat volume digest
|
||||
xxhash-rust = { version = "0.8", features = ["xxh64"] }
|
||||
|
||||
# Memory-mapped files
|
||||
memmap2 = "0.9"
|
||||
|
||||
|
||||
@@ -59,6 +59,8 @@ service Seaweed {
|
||||
}
|
||||
rpc VolumeGrow (VolumeGrowRequest) returns (VolumeGrowResponse) {
|
||||
}
|
||||
rpc CollectionStatistics (CollectionStatisticsRequest) returns (CollectionStatisticsResponse) {
|
||||
}
|
||||
}
|
||||
|
||||
//////////////////////////////////////////////////
|
||||
@@ -105,6 +107,16 @@ message Heartbeat {
|
||||
// physical disk capacity per disk type, in bytes, from the underlying filesystem
|
||||
map<string, uint64> disk_total_bytes = 25;
|
||||
map<string, uint64> disk_free_bytes = 26;
|
||||
|
||||
// Digest of every volume in this heartbeat's view of the server, letting the
|
||||
// master check its copy is current without being sent the whole list. Absent
|
||||
// from servers that do not compute it, and distinct from a digest of 0, which
|
||||
// is what a server holding no volumes reports.
|
||||
optional uint64 volume_digest = 27;
|
||||
// Volumes whose reported state changed since the last heartbeat, sent in
|
||||
// place of `volumes`. A master that does not understand this never sets
|
||||
// volume_digest_supported, so it keeps being sent the whole list.
|
||||
repeated VolumeInformationMessage changed_volumes = 28;
|
||||
}
|
||||
|
||||
message HeartbeatResponse {
|
||||
@@ -115,6 +127,12 @@ message HeartbeatResponse {
|
||||
repeated StorageBackend storage_backends = 5;
|
||||
repeated string duplicated_uuids = 6;
|
||||
bool preallocate = 7;
|
||||
// The master's view of this server's volumes disagrees with the reported
|
||||
// digest, so it needs the full volume list rather than changes alone.
|
||||
bool resend_full_volume_list = 8;
|
||||
// The master compares volume digests, so a server that reports one may send
|
||||
// changed_volumes in place of its whole list.
|
||||
bool volume_digest_supported = 9;
|
||||
}
|
||||
|
||||
message VolumeInformationMessage {
|
||||
@@ -294,6 +312,11 @@ message StatisticsResponse {
|
||||
uint64 total_size = 4;
|
||||
uint64 used_size = 5;
|
||||
uint64 file_count = 6;
|
||||
// sizes counting one copy of the data: a single replica of a regular volume,
|
||||
// the data shards of an ec volume. logical_total_size scales the free space
|
||||
// by the copies the requested replication makes.
|
||||
uint64 logical_total_size = 7;
|
||||
uint64 logical_used_size = 8;
|
||||
}
|
||||
|
||||
//
|
||||
@@ -310,6 +333,26 @@ message CollectionListResponse {
|
||||
repeated Collection collections = 1;
|
||||
}
|
||||
|
||||
// Summarises what each collection holds, so a caller tracking usage does not
|
||||
// have to be sent every volume in the cluster to add it up itself.
|
||||
message CollectionStatisticsRequest {
|
||||
}
|
||||
message CollectionStatisticsResponse {
|
||||
repeated CollectionStatistics collections = 1;
|
||||
}
|
||||
message CollectionStatistics {
|
||||
string collection = 1;
|
||||
uint64 file_count = 2;
|
||||
uint64 delete_count = 3;
|
||||
uint64 deleted_byte_count = 4;
|
||||
// one copy of the data: a single replica of a regular volume, the data
|
||||
// shards of an ec volume
|
||||
uint64 size = 5;
|
||||
// what is on disk: every replica, and parity shards
|
||||
uint64 physical_size = 6;
|
||||
uint64 volume_count = 7;
|
||||
}
|
||||
|
||||
message CollectionDeleteRequest {
|
||||
string name = 1;
|
||||
}
|
||||
|
||||
@@ -660,10 +660,17 @@ impl VolumeServer for VolumeGrpcService {
|
||||
) -> Result<Response<volume_server_pb::DeleteCollectionResponse>, Status> {
|
||||
self.check_grpc_admin_auth(&request)?;
|
||||
let collection = &request.into_inner().collection;
|
||||
let mut store = self.state.store.write().unwrap();
|
||||
store
|
||||
.delete_collection(collection)
|
||||
.map_err(|e| Status::internal(e))?;
|
||||
{
|
||||
let mut store = self.state.store.write().unwrap();
|
||||
store
|
||||
.delete_collection(collection)
|
||||
.map_err(|e| Status::internal(e))?;
|
||||
}
|
||||
// The delta the notify path derives is the only thing that tells the
|
||||
// master these slots came free: a heartbeat carries the whole list only
|
||||
// when the master asks, and a volume grown and destroyed between two of
|
||||
// them was never in one at all.
|
||||
self.state.volume_state_notify.notify_one();
|
||||
Ok(Response::new(volume_server_pb::DeleteCollectionResponse {}))
|
||||
}
|
||||
|
||||
|
||||
@@ -19,6 +19,8 @@ use crate::pb::master_pb::seaweed_client::SeaweedClient;
|
||||
use crate::pb::volume_server_pb;
|
||||
use crate::remote_storage::s3_tier::{S3TierBackend, S3TierConfig};
|
||||
use crate::storage::store::Store;
|
||||
use crate::storage::volume_report::VolumeReportKey;
|
||||
use crate::storage::volume_report_hash::report_hash;
|
||||
use crate::storage::types::NeedleId;
|
||||
|
||||
const DUPLICATE_UUID_RETRY_MESSAGE: &str = "duplicate UUIDs detected, retrying connection";
|
||||
@@ -273,6 +275,13 @@ fn duplicate_directories(store: &Store, duplicated_uuids: &[String]) -> Vec<Stri
|
||||
}
|
||||
|
||||
fn apply_master_volume_options(store: &Store, hb_resp: &master_pb::HeartbeatResponse) -> bool {
|
||||
if hb_resp.volume_digest_supported {
|
||||
store.volume_report.accept_deltas();
|
||||
}
|
||||
if hb_resp.resend_full_volume_list {
|
||||
info!("master asked for the full volume list");
|
||||
store.volume_report.request_full_list();
|
||||
}
|
||||
let mut volume_opts_changed = false;
|
||||
if store.get_preallocate() != hb_resp.preallocate {
|
||||
store.set_preallocate(hb_resp.preallocate);
|
||||
@@ -383,13 +392,14 @@ async fn do_heartbeat(
|
||||
|
||||
let (tx, rx) = tokio::sync::mpsc::channel::<master_pb::Heartbeat>(32);
|
||||
|
||||
// This master may know nothing about this server, and has not yet said
|
||||
// whether it understands digests, so start from the whole list.
|
||||
state.store.read().unwrap().volume_report.reset();
|
||||
|
||||
// Keep track of what we sent, to generate delta updates
|
||||
let initial_hb = collect_heartbeat(config, state);
|
||||
let mut last_volumes: HashMap<u32, master_pb::VolumeInformationMessage> = initial_hb
|
||||
.volumes
|
||||
.iter()
|
||||
.map(|v| (v.id, v.clone()))
|
||||
.collect();
|
||||
let (initial_hb, initial_volumes) = collect_heartbeat_with_snapshot(config, state);
|
||||
let mut last_volumes: HashMap<u32, master_pb::VolumeInformationMessage> =
|
||||
initial_volumes.iter().map(|v| (v.id, v.clone())).collect();
|
||||
let mut last_ec_shards = {
|
||||
let store = state.store.read().unwrap();
|
||||
collect_ec_shard_delta_messages(&store)
|
||||
@@ -454,9 +464,10 @@ async fn do_heartbeat(
|
||||
apply_master_volume_options(&s, &hb_resp)
|
||||
};
|
||||
if changed {
|
||||
let adjusted_hb = collect_heartbeat(config, state);
|
||||
let (adjusted_hb, adjusted_volumes) =
|
||||
collect_heartbeat_with_snapshot(config, state);
|
||||
last_volumes =
|
||||
adjusted_hb.volumes.iter().map(|v| (v.id, v.clone())).collect();
|
||||
adjusted_volumes.iter().map(|v| (v.id, v.clone())).collect();
|
||||
last_ec_shards = {
|
||||
let store = state.store.read().unwrap();
|
||||
collect_ec_shard_delta_messages(&store)
|
||||
@@ -489,8 +500,8 @@ async fn do_heartbeat(
|
||||
let s = state.store.read().unwrap();
|
||||
s.maybe_adjust_volume_max();
|
||||
}
|
||||
let current_hb = collect_heartbeat(config, state);
|
||||
last_volumes = current_hb.volumes.iter().map(|v| (v.id, v.clone())).collect();
|
||||
let (current_hb, current_volumes) = collect_heartbeat_with_snapshot(config, state);
|
||||
last_volumes = current_volumes.iter().map(|v| (v.id, v.clone())).collect();
|
||||
last_ec_shards = {
|
||||
let store = state.store.read().unwrap();
|
||||
collect_ec_shard_delta_messages(&store)
|
||||
@@ -518,8 +529,8 @@ async fn do_heartbeat(
|
||||
info!("Heartbeat stopping");
|
||||
return Ok(None);
|
||||
}
|
||||
let current_hb = collect_heartbeat(config, state);
|
||||
let current_volumes: HashMap<u32, _> = current_hb.volumes.iter().map(|v| (v.id, v.clone())).collect();
|
||||
let held_volumes = collect_volume_snapshot(config, state);
|
||||
let current_volumes: HashMap<u32, _> = held_volumes.iter().map(|v| (v.id, v.clone())).collect();
|
||||
let current_ec_shards = {
|
||||
let store = state.store.read().unwrap();
|
||||
collect_ec_shard_delta_messages(&store)
|
||||
@@ -734,10 +745,10 @@ fn parse_bool_property(value: Option<&String>) -> bool {
|
||||
}
|
||||
|
||||
/// Collect volume information into a Heartbeat message.
|
||||
fn collect_heartbeat(
|
||||
fn collect_heartbeat_with_snapshot(
|
||||
config: &HeartbeatConfig,
|
||||
state: &Arc<VolumeServerState>,
|
||||
) -> master_pb::Heartbeat {
|
||||
) -> (master_pb::Heartbeat, Vec<master_pb::VolumeInformationMessage>) {
|
||||
let mut store = state.store.write().unwrap();
|
||||
let (ec_shards, deleted_ec_shards) = store.delete_expired_ec_volumes();
|
||||
build_heartbeat_with_ec_status(
|
||||
@@ -745,9 +756,28 @@ fn collect_heartbeat(
|
||||
&mut store,
|
||||
deleted_ec_shards,
|
||||
ec_shards.is_empty(),
|
||||
true,
|
||||
)
|
||||
}
|
||||
|
||||
/// Lists the volumes the server holds without touching reporting state or
|
||||
/// expiring anything, for callers that only need to diff against a previous
|
||||
/// snapshot and send a message of their own.
|
||||
fn collect_volume_snapshot(
|
||||
config: &HeartbeatConfig,
|
||||
state: &Arc<VolumeServerState>,
|
||||
) -> Vec<master_pb::VolumeInformationMessage> {
|
||||
let mut store = state.store.write().unwrap();
|
||||
build_heartbeat_with_ec_status(config, &mut store, Vec::new(), true, false).1
|
||||
}
|
||||
|
||||
fn collect_heartbeat(
|
||||
config: &HeartbeatConfig,
|
||||
state: &Arc<VolumeServerState>,
|
||||
) -> master_pb::Heartbeat {
|
||||
collect_heartbeat_with_snapshot(config, state).0
|
||||
}
|
||||
|
||||
fn collect_location_metadata(
|
||||
store: &Store,
|
||||
disk_max_by_id: &[i32],
|
||||
@@ -780,15 +810,19 @@ fn collect_location_metadata(
|
||||
#[cfg(test)]
|
||||
fn build_heartbeat(config: &HeartbeatConfig, store: &mut Store) -> master_pb::Heartbeat {
|
||||
let has_no_ec_shards = collect_live_ec_shards(store, false).is_empty();
|
||||
build_heartbeat_with_ec_status(config, store, Vec::new(), has_no_ec_shards)
|
||||
build_heartbeat_with_ec_status(config, store, Vec::new(), has_no_ec_shards, true).0
|
||||
}
|
||||
|
||||
/// Returns the heartbeat to send and, separately, every volume held. The
|
||||
/// caller derives mount and unmount deltas by diffing successive snapshots, so
|
||||
/// it must not be handed the partial list a heartbeat may carry.
|
||||
fn build_heartbeat_with_ec_status(
|
||||
config: &HeartbeatConfig,
|
||||
store: &mut Store,
|
||||
deleted_ec_shards: Vec<master_pb::VolumeEcShardInformationMessage>,
|
||||
has_no_ec_shards: bool,
|
||||
) -> master_pb::Heartbeat {
|
||||
commit_report: bool,
|
||||
) -> (master_pb::Heartbeat, Vec<master_pb::VolumeInformationMessage>) {
|
||||
const MAX_TTL_VOLUME_REMOVAL_DELAY: u32 = 10;
|
||||
|
||||
#[derive(Default)]
|
||||
@@ -800,6 +834,13 @@ fn build_heartbeat_with_ec_status(
|
||||
}
|
||||
|
||||
let mut volumes = Vec::new();
|
||||
// Covers every volume held, whether or not this heartbeat names it, so the
|
||||
// master can tell whether applying what it was sent leaves it current.
|
||||
// Volumes skipped below -- quarantined, phantom, expired -- are in neither.
|
||||
let mut volume_digest: u64 = 0;
|
||||
let (send_full_list, report_generation) = store.volume_report.begin();
|
||||
let mut reported_hashes: HashMap<VolumeReportKey, u64> = HashMap::new();
|
||||
let mut changed_volumes = Vec::new();
|
||||
let mut max_file_key = NeedleId(0);
|
||||
let mut max_volume_counts: HashMap<String, u32> = HashMap::new();
|
||||
let mut disk_total_bytes: HashMap<String, u64> = HashMap::new();
|
||||
@@ -875,7 +916,7 @@ fn build_heartbeat_with_ec_status(
|
||||
}
|
||||
|
||||
let (remote_storage_name, remote_storage_key) = vol.remote_storage_name_key();
|
||||
volumes.push(master_pb::VolumeInformationMessage {
|
||||
let volume_message = master_pb::VolumeInformationMessage {
|
||||
id: vol.id.0,
|
||||
size: volume_size,
|
||||
collection: vol.collection.clone(),
|
||||
@@ -892,8 +933,15 @@ fn build_heartbeat_with_ec_status(
|
||||
disk_id: disk_id as u32,
|
||||
remote_storage_name,
|
||||
remote_storage_key,
|
||||
..Default::default()
|
||||
});
|
||||
};
|
||||
let hash = report_hash(&volume_message);
|
||||
volume_digest ^= hash;
|
||||
let key: VolumeReportKey = (volume_message.disk_id, volume_message.id);
|
||||
reported_hashes.insert(key, hash);
|
||||
if send_full_list || store.volume_report.changed(key, hash) {
|
||||
changed_volumes.push(volume_message.clone());
|
||||
}
|
||||
volumes.push(volume_message);
|
||||
} else if vol.is_expired_long_enough(MAX_TTL_VOLUME_REMOVAL_DELAY) {
|
||||
delete_vids.push(vol.id);
|
||||
should_delete_volume = true;
|
||||
@@ -954,10 +1002,23 @@ fn build_heartbeat_with_ec_status(
|
||||
let total_max: i64 = max_volume_counts.values().map(|v| *v as i64).sum();
|
||||
crate::metrics::MAX_VOLUMES.set(total_max);
|
||||
|
||||
let has_no_volumes = volumes.is_empty();
|
||||
// Only when this heartbeat is going to be sent: marking volumes reported
|
||||
// and then discarding the message would leave the master never told.
|
||||
if commit_report {
|
||||
store.volume_report.commit(reported_hashes, report_generation);
|
||||
}
|
||||
|
||||
// has_no_volumes says the server holds nothing, so it may only be derived
|
||||
// from a full list. Deriving it from a changed-only heartbeat would make a
|
||||
// quiet one read as an empty server and drop every volume on it.
|
||||
let (heartbeat_volumes, changed_volumes, has_no_volumes) = if send_full_list {
|
||||
(changed_volumes, Vec::new(), volumes.is_empty())
|
||||
} else {
|
||||
(Vec::new(), changed_volumes, false)
|
||||
};
|
||||
let (location_uuids, disk_tags) = collect_location_metadata(store, &disk_max_by_id);
|
||||
|
||||
master_pb::Heartbeat {
|
||||
let heartbeat = master_pb::Heartbeat {
|
||||
id: store.id.clone(),
|
||||
ip: config.ip.clone(),
|
||||
port: config.port as u32,
|
||||
@@ -966,7 +1027,9 @@ fn build_heartbeat_with_ec_status(
|
||||
data_center: config.data_center.clone(),
|
||||
rack: config.rack.clone(),
|
||||
admin_port: config.port as u32,
|
||||
volumes,
|
||||
volumes: heartbeat_volumes,
|
||||
changed_volumes,
|
||||
volume_digest: Some(volume_digest),
|
||||
deleted_ec_shards,
|
||||
has_no_volumes,
|
||||
has_no_ec_shards,
|
||||
@@ -977,7 +1040,8 @@ fn build_heartbeat_with_ec_status(
|
||||
location_uuids,
|
||||
disk_tags,
|
||||
..Default::default()
|
||||
}
|
||||
};
|
||||
(heartbeat, volumes)
|
||||
}
|
||||
|
||||
fn collect_live_ec_shards(
|
||||
@@ -1246,6 +1310,205 @@ mod tests {
|
||||
assert!(heartbeat.has_no_volumes);
|
||||
}
|
||||
|
||||
// The digest must cover exactly the volumes the heartbeat carries. A volume
|
||||
// reported but left out of the digest, or the reverse, makes the master's
|
||||
// comparison disagree forever. An empty store still reports a digest, so
|
||||
// the master can tell it from a server that computes none.
|
||||
fn reporting_store(dir: &str, count: u32) -> Store {
|
||||
let mut store = Store::new(NeedleMapKind::InMemory);
|
||||
store
|
||||
.add_location(
|
||||
dir,
|
||||
dir,
|
||||
16,
|
||||
DiskType::HardDrive,
|
||||
MinFreeSpace::Percent(1.0),
|
||||
Vec::new(),
|
||||
)
|
||||
.unwrap();
|
||||
for id in 1..=count {
|
||||
store
|
||||
.add_volume(
|
||||
VolumeId(id),
|
||||
"pics",
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
DiskType::HardDrive,
|
||||
Version::current(),
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
store.volume_report.reset();
|
||||
store
|
||||
}
|
||||
|
||||
// Until the master says it compares digests it may be one that reads a
|
||||
// partial list as the whole truth, so it keeps getting the whole list.
|
||||
#[test]
|
||||
fn test_heartbeat_sends_full_list_until_the_master_accepts() {
|
||||
let temp_dir = tempfile::tempdir().unwrap();
|
||||
let mut store = reporting_store(temp_dir.path().to_str().unwrap(), 2);
|
||||
|
||||
for _ in 0..3 {
|
||||
let heartbeat = build_heartbeat(&test_config(), &mut store);
|
||||
assert_eq!(heartbeat.volumes.len(), 2);
|
||||
assert!(heartbeat.changed_volumes.is_empty());
|
||||
}
|
||||
}
|
||||
|
||||
// The one that would be catastrophic: a heartbeat with nothing to report
|
||||
// must not look like a server that has lost every volume.
|
||||
#[test]
|
||||
fn test_quiet_heartbeat_does_not_look_like_an_empty_server() {
|
||||
let temp_dir = tempfile::tempdir().unwrap();
|
||||
let mut store = reporting_store(temp_dir.path().to_str().unwrap(), 2);
|
||||
store.volume_report.accept_deltas();
|
||||
let full = build_heartbeat(&test_config(), &mut store);
|
||||
|
||||
let quiet = build_heartbeat(&test_config(), &mut store);
|
||||
assert!(quiet.volumes.is_empty());
|
||||
assert!(quiet.changed_volumes.is_empty());
|
||||
assert!(!quiet.has_no_volumes);
|
||||
// The digest still covers everything held, not just what was sent.
|
||||
assert_eq!(quiet.volume_digest, full.volume_digest);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_heartbeat_reports_only_what_changed() {
|
||||
let temp_dir = tempfile::tempdir().unwrap();
|
||||
let mut store = reporting_store(temp_dir.path().to_str().unwrap(), 2);
|
||||
store.volume_report.accept_deltas();
|
||||
build_heartbeat(&test_config(), &mut store);
|
||||
|
||||
store
|
||||
.add_volume(
|
||||
VolumeId(3),
|
||||
"pics",
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
DiskType::HardDrive,
|
||||
Version::current(),
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
let heartbeat = build_heartbeat(&test_config(), &mut store);
|
||||
assert!(heartbeat.volumes.is_empty());
|
||||
assert_eq!(heartbeat.changed_volumes.len(), 1);
|
||||
assert_eq!(heartbeat.changed_volumes[0].id, 3);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_heartbeat_returns_to_the_full_list_on_request() {
|
||||
let temp_dir = tempfile::tempdir().unwrap();
|
||||
let mut store = reporting_store(temp_dir.path().to_str().unwrap(), 2);
|
||||
store.volume_report.accept_deltas();
|
||||
build_heartbeat(&test_config(), &mut store);
|
||||
|
||||
store.volume_report.request_full_list();
|
||||
let resent = build_heartbeat(&test_config(), &mut store);
|
||||
assert_eq!(resent.volumes.len(), 2);
|
||||
assert!(resent.changed_volumes.is_empty());
|
||||
|
||||
let next = build_heartbeat(&test_config(), &mut store);
|
||||
assert!(next.volumes.is_empty());
|
||||
}
|
||||
|
||||
// A request that lands while a heartbeat is being built asked about a later
|
||||
// state than that heartbeat carries, so it must survive being committed over.
|
||||
#[test]
|
||||
fn test_full_list_request_during_collection_survives() {
|
||||
let temp_dir = tempfile::tempdir().unwrap();
|
||||
let mut store = reporting_store(temp_dir.path().to_str().unwrap(), 2);
|
||||
store.volume_report.accept_deltas();
|
||||
build_heartbeat(&test_config(), &mut store);
|
||||
|
||||
let (full, generation) = store.volume_report.begin();
|
||||
assert!(!full);
|
||||
store.volume_report.request_full_list();
|
||||
store.volume_report.commit(HashMap::new(), generation);
|
||||
|
||||
let heartbeat = build_heartbeat(&test_config(), &mut store);
|
||||
assert_eq!(heartbeat.volumes.len(), 2);
|
||||
}
|
||||
|
||||
// Taking a snapshot must not mark volumes as told to a master that is
|
||||
// getting a different message.
|
||||
#[test]
|
||||
fn test_snapshot_does_not_mark_volumes_reported() {
|
||||
let temp_dir = tempfile::tempdir().unwrap();
|
||||
let mut store = reporting_store(temp_dir.path().to_str().unwrap(), 2);
|
||||
store.volume_report.accept_deltas();
|
||||
build_heartbeat(&test_config(), &mut store);
|
||||
|
||||
store
|
||||
.add_volume(
|
||||
VolumeId(3),
|
||||
"pics",
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
DiskType::HardDrive,
|
||||
Version::current(),
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
// What the notify path does: collect a snapshot, send a message of its own.
|
||||
let snapshot =
|
||||
build_heartbeat_with_ec_status(&test_config(), &mut store, Vec::new(), true, false).1;
|
||||
assert_eq!(snapshot.len(), 3);
|
||||
|
||||
let heartbeat = build_heartbeat(&test_config(), &mut store);
|
||||
assert_eq!(heartbeat.changed_volumes.len(), 1);
|
||||
assert_eq!(heartbeat.changed_volumes[0].id, 3);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_build_heartbeat_digests_exactly_what_it_reports() {
|
||||
let temp_dir = tempfile::tempdir().unwrap();
|
||||
let dir = temp_dir.path().to_str().unwrap();
|
||||
|
||||
let mut store = Store::new(NeedleMapKind::InMemory);
|
||||
store
|
||||
.add_location(
|
||||
dir,
|
||||
dir,
|
||||
8,
|
||||
DiskType::HardDrive,
|
||||
MinFreeSpace::Percent(1.0),
|
||||
Vec::new(),
|
||||
)
|
||||
.unwrap();
|
||||
|
||||
let empty = build_heartbeat(&test_config(), &mut store);
|
||||
assert_eq!(empty.volume_digest, Some(0));
|
||||
|
||||
for vid in [VolumeId(1), VolumeId(2)] {
|
||||
store
|
||||
.add_volume(
|
||||
vid,
|
||||
"pics",
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
DiskType::HardDrive,
|
||||
Version::current(),
|
||||
)
|
||||
.unwrap();
|
||||
}
|
||||
|
||||
let heartbeat = build_heartbeat(&test_config(), &mut store);
|
||||
assert_eq!(heartbeat.volumes.len(), 2);
|
||||
|
||||
let expected = heartbeat
|
||||
.volumes
|
||||
.iter()
|
||||
.fold(0u64, |acc, m| acc ^ crate::storage::volume_report_hash::report_hash(m));
|
||||
assert_eq!(heartbeat.volume_digest, Some(expected));
|
||||
assert_ne!(heartbeat.volume_digest, Some(0));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_build_heartbeat_tracks_go_read_only_labels_and_disk_id() {
|
||||
let temp_dir = tempfile::tempdir().unwrap();
|
||||
|
||||
@@ -7,8 +7,8 @@
|
||||
use std::collections::{HashMap, HashSet};
|
||||
use std::fs;
|
||||
use std::io;
|
||||
use std::sync::atomic::{AtomicBool, AtomicI32, AtomicU64, Ordering};
|
||||
use std::sync::Arc;
|
||||
use std::sync::atomic::{AtomicBool, AtomicI32, AtomicU64, AtomicUsize, Ordering};
|
||||
use std::sync::{Arc, Mutex};
|
||||
|
||||
use tracing::warn;
|
||||
|
||||
@@ -122,6 +122,10 @@ impl DiskLocation {
|
||||
// Scan for .dat files
|
||||
let entries = fs::read_dir(&self.directory)?;
|
||||
let mut dat_files: Vec<(String, VolumeId)> = Vec::new();
|
||||
// Every collection claiming an id, in scan order; open_volumes keeps
|
||||
// the first that opens.
|
||||
let mut to_load: Vec<(VolumeId, Vec<String>)> = Vec::new();
|
||||
let mut queued: HashMap<VolumeId, usize> = HashMap::new();
|
||||
let mut seen = HashSet::new();
|
||||
|
||||
for entry in entries {
|
||||
@@ -201,6 +205,7 @@ impl DiskLocation {
|
||||
continue;
|
||||
}
|
||||
|
||||
|
||||
// Load existing data only; never create a phantom `.dat`. A lone
|
||||
// `.vif`/`.idx` (e.g. an EC sidecar whose `.ecx` is on a sibling
|
||||
// disk) would otherwise have Volume::new write an 8-byte stub that
|
||||
@@ -221,30 +226,22 @@ impl DiskLocation {
|
||||
continue;
|
||||
}
|
||||
|
||||
match Volume::new(
|
||||
&self.directory,
|
||||
&self.idx_directory,
|
||||
&collection,
|
||||
vid,
|
||||
needle_map_kind,
|
||||
None, // replica placement read from superblock
|
||||
None, // TTL read from superblock
|
||||
0, // no preallocate on load
|
||||
Version::current(),
|
||||
) {
|
||||
Ok(mut v) => {
|
||||
v.location_disk_space_low = self.is_disk_space_low.clone();
|
||||
crate::metrics::VOLUME_GAUGE
|
||||
.with_label_values(&[&collection, "volume"])
|
||||
.inc();
|
||||
self.volumes.insert(vid, v);
|
||||
}
|
||||
Err(e) => {
|
||||
warn!(volume_id = vid.0, error = %e, "failed to load volume");
|
||||
match queued.get(&vid) {
|
||||
Some(&i) => to_load[i].1.push(collection),
|
||||
None => {
|
||||
queued.insert(vid, to_load.len());
|
||||
to_load.push((vid, vec![collection]));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (collection, vid, v) in self.open_volumes(to_load, needle_map_kind) {
|
||||
crate::metrics::VOLUME_GAUGE
|
||||
.with_label_values(&[&collection, "volume"])
|
||||
.inc();
|
||||
self.volumes.insert(vid, v);
|
||||
}
|
||||
|
||||
// After regular volumes, auto-discover EC shards on disk so a
|
||||
// fresh restart picks up shards without an explicit
|
||||
// VolumeEcShardsMount RPC. Mirrors Go's loadExistingVolumes
|
||||
@@ -257,6 +254,65 @@ impl DiskLocation {
|
||||
Ok(())
|
||||
}
|
||||
|
||||
/// Open the volumes the directory scan selected. Opening one is dominated
|
||||
/// by reading its .idx into the needle map, so a disk holding thousands
|
||||
/// takes thousands of serial index reads to come up; mirrors Go's
|
||||
/// concurrentLoadingVolumes down to the max(cores, 10) worker count, whose
|
||||
/// floor keeps a small-core box off one-at-a-time on IO-bound work.
|
||||
///
|
||||
/// An id is only spoken for once a volume actually loads, so a corrupt
|
||||
/// `colA_5.dat` still leaves `colB_5.dat` a chance.
|
||||
fn open_volumes(
|
||||
&self,
|
||||
to_load: Vec<(VolumeId, Vec<String>)>,
|
||||
needle_map_kind: NeedleMapKind,
|
||||
) -> Vec<(String, VolumeId, Volume)> {
|
||||
if to_load.is_empty() {
|
||||
return Vec::new();
|
||||
}
|
||||
let workers = std::thread::available_parallelism()
|
||||
.map(|n| n.get())
|
||||
.unwrap_or(1)
|
||||
.max(10)
|
||||
.min(to_load.len());
|
||||
|
||||
let next = AtomicUsize::new(0);
|
||||
let opened = Mutex::new(Vec::with_capacity(to_load.len()));
|
||||
std::thread::scope(|scope| {
|
||||
for _ in 0..workers {
|
||||
scope.spawn(|| loop {
|
||||
let i = next.fetch_add(1, Ordering::Relaxed);
|
||||
let Some((vid, collections)) = to_load.get(i) else {
|
||||
return;
|
||||
};
|
||||
for collection in collections {
|
||||
match Volume::new(
|
||||
&self.directory,
|
||||
&self.idx_directory,
|
||||
collection,
|
||||
*vid,
|
||||
needle_map_kind,
|
||||
None, // replica placement read from superblock
|
||||
None, // TTL read from superblock
|
||||
0, // no preallocate on load
|
||||
Version::current(),
|
||||
) {
|
||||
Ok(mut v) => {
|
||||
v.location_disk_space_low = self.is_disk_space_low.clone();
|
||||
opened.lock().unwrap().push((collection.clone(), *vid, v));
|
||||
break;
|
||||
}
|
||||
Err(e) => {
|
||||
warn!(volume_id = vid.0, error = %e, "failed to load volume");
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
});
|
||||
opened.into_inner().unwrap_or_else(|e| e.into_inner())
|
||||
}
|
||||
|
||||
/// Directory pre-pass that recovers interrupted compaction commits. Collects
|
||||
/// every volume id that still has a .cpc commit marker or a leftover
|
||||
/// .cpd/.cpx temp file across the data and idx directories, then rolls the
|
||||
@@ -1493,6 +1549,60 @@ mod tests {
|
||||
assert!(ids.contains(&VolumeId(2)));
|
||||
}
|
||||
|
||||
// Two collections can name the same volume id on one disk; a candidate
|
||||
// that fails to open must not shadow a good one behind it.
|
||||
#[test]
|
||||
fn test_open_volumes_falls_back_past_a_corrupt_candidate() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
|
||||
{
|
||||
let mut loc = DiskLocation::new(
|
||||
dir,
|
||||
dir,
|
||||
10,
|
||||
DiskType::HardDrive,
|
||||
MinFreeSpace::Percent(1.0),
|
||||
Vec::new(),
|
||||
)
|
||||
.unwrap();
|
||||
loc.create_volume(
|
||||
VolumeId(9),
|
||||
"good",
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
Version::current(),
|
||||
)
|
||||
.unwrap();
|
||||
loc.close();
|
||||
}
|
||||
|
||||
// Same id under another collection, unopenable.
|
||||
let mut bad = vec![0u8; 16];
|
||||
bad[0] = 9; // unsupported version
|
||||
std::fs::write(format!("{}/bad_9.dat", dir), &bad).unwrap();
|
||||
|
||||
let loc = DiskLocation::new(
|
||||
dir,
|
||||
dir,
|
||||
10,
|
||||
DiskType::HardDrive,
|
||||
MinFreeSpace::Percent(1.0),
|
||||
Vec::new(),
|
||||
)
|
||||
.unwrap();
|
||||
let opened = loc.open_volumes(
|
||||
vec![(VolumeId(9), vec!["bad".to_string(), "good".to_string()])],
|
||||
NeedleMapKind::InMemory,
|
||||
);
|
||||
|
||||
assert_eq!(opened.len(), 1, "the good candidate should still open");
|
||||
assert_eq!(opened[0].0, "good");
|
||||
assert_eq!(opened[0].1, VolumeId(9));
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn test_disk_location_delete_volume() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
|
||||
@@ -10,3 +10,5 @@ pub mod super_block;
|
||||
pub mod types;
|
||||
pub mod volume;
|
||||
pub mod volume_idx_repair;
|
||||
pub mod volume_report;
|
||||
pub mod volume_report_hash;
|
||||
|
||||
@@ -31,6 +31,7 @@ pub struct Store {
|
||||
pub public_url: String,
|
||||
pub data_center: String,
|
||||
pub rack: String,
|
||||
pub volume_report: crate::storage::volume_report::VolumeReportState,
|
||||
}
|
||||
|
||||
impl Store {
|
||||
@@ -45,6 +46,7 @@ impl Store {
|
||||
port: 0,
|
||||
grpc_port: 0,
|
||||
public_url: String::new(),
|
||||
volume_report: Default::default(),
|
||||
data_center: String::new(),
|
||||
rack: String::new(),
|
||||
}
|
||||
|
||||
@@ -1970,11 +1970,13 @@ impl Volume {
|
||||
}
|
||||
|
||||
/// Verify the live needle at the .dat tail: its header must match the
|
||||
/// .idx entry and the .dat must end exactly at its on-disk end. Extra
|
||||
/// bytes mean an unindexed trailing record (a torn append); appending
|
||||
/// after one would place the next needle at an offset the 8-byte .idx
|
||||
/// encoding may not represent, so the volume is quarantined read-only
|
||||
/// instead. Mirrors Go's verifyNeedleIntegrity.
|
||||
/// .idx entry and, on v3, the .dat must end exactly at its on-disk end.
|
||||
/// Extra bytes mean an unindexed trailing record (a torn append);
|
||||
/// appending after one would place the next needle at an offset the
|
||||
/// 8-byte .idx encoding may not represent, so the volume is quarantined
|
||||
/// read-only instead. Mirrors Go's verifyNeedleIntegrity, whose tail
|
||||
/// check is v3-only -- a v1/v2 volume Go serves read-write must not go
|
||||
/// read-only here.
|
||||
fn verify_needle_integrity(
|
||||
&mut self,
|
||||
actual_offset: i64,
|
||||
@@ -2013,15 +2015,16 @@ impl Volume {
|
||||
)));
|
||||
}
|
||||
|
||||
if version == VERSION_3 {
|
||||
let ts_offset =
|
||||
checked_offset as u64 + NEEDLE_HEADER_SIZE as u64 + size.0 as u64 + 4; // skip checksum
|
||||
let mut ts_buf = [0u8; 8];
|
||||
self.read_exact_at_backend(&mut ts_buf, ts_offset)?;
|
||||
let ts = u64::from_be_bytes(ts_buf);
|
||||
if ts > 0 {
|
||||
self.last_append_at_ns = ts;
|
||||
}
|
||||
if version != VERSION_3 {
|
||||
return Ok(());
|
||||
}
|
||||
|
||||
let ts_offset = checked_offset as u64 + NEEDLE_HEADER_SIZE as u64 + size.0 as u64 + 4; // skip checksum
|
||||
let mut ts_buf = [0u8; 8];
|
||||
self.read_exact_at_backend(&mut ts_buf, ts_offset)?;
|
||||
let ts = u64::from_be_bytes(ts_buf);
|
||||
if ts > 0 {
|
||||
self.last_append_at_ns = ts;
|
||||
}
|
||||
|
||||
self.verify_dat_ends_at(checked_offset + get_actual_size(size, version))
|
||||
@@ -4115,6 +4118,54 @@ mod tests {
|
||||
);
|
||||
}
|
||||
|
||||
// Go's tail check is v3-only, so it serves a v1/v2 volume with an
|
||||
// unindexed tail read-write. Checking it here flipped whole legacy disks
|
||||
// read-only on their first boot under this server.
|
||||
#[test]
|
||||
fn test_integrity_skips_dat_tail_check_before_v3() {
|
||||
let tmp = TempDir::new().unwrap();
|
||||
let dir = tmp.path().to_str().unwrap();
|
||||
{
|
||||
let mut v = Volume::new(
|
||||
dir,
|
||||
dir,
|
||||
"",
|
||||
VolumeId(1),
|
||||
NeedleMapKind::InMemory,
|
||||
None,
|
||||
None,
|
||||
0,
|
||||
VERSION_2,
|
||||
)
|
||||
.unwrap();
|
||||
for i in 1..=3u64 {
|
||||
let data = format!("data {}", i);
|
||||
let mut n = Needle {
|
||||
id: NeedleId(i),
|
||||
cookie: Cookie(i as u32),
|
||||
data: data.as_bytes().to_vec(),
|
||||
data_size: data.len() as u32,
|
||||
..Needle::default()
|
||||
};
|
||||
v.write_needle(&mut n, true).unwrap();
|
||||
}
|
||||
v.sync_to_disk().unwrap();
|
||||
}
|
||||
|
||||
let dat_path = volume_file_name(dir, "", VolumeId(1)) + ".dat";
|
||||
let mut f = OpenOptions::new().append(true).open(&dat_path).unwrap();
|
||||
f.write_all(&[0xFFu8; 96]).unwrap();
|
||||
f.sync_all().unwrap();
|
||||
drop(f);
|
||||
|
||||
let v = reload_volume(dir);
|
||||
assert_eq!(v.version(), VERSION_2, "volume should reload as v2");
|
||||
assert!(
|
||||
!v.is_no_write_or_delete(),
|
||||
"v2 volume with unindexed .dat tail must stay writable, as in Go"
|
||||
);
|
||||
}
|
||||
|
||||
// Same torn tail, but with a deletion tombstone as the last indexed
|
||||
// record — the tombstone path must also verify the .dat ends at it.
|
||||
#[test]
|
||||
|
||||
@@ -0,0 +1,73 @@
|
||||
//! Mirror of `weed/storage/store_volume_report.go`.
|
||||
|
||||
use std::collections::HashMap;
|
||||
use std::sync::atomic::{AtomicBool, AtomicU64, Ordering};
|
||||
use std::sync::Mutex;
|
||||
|
||||
/// Identifies one reported copy. Keyed by disk as well as id because a volume
|
||||
/// id can be mounted on two disks, and reporting one of them would leave the
|
||||
/// other's changes untold.
|
||||
pub type VolumeReportKey = (u32, u32);
|
||||
|
||||
/// Remembers what the master was last told about each volume, so a heartbeat
|
||||
/// can carry only what moved since.
|
||||
///
|
||||
/// Per-connection: a server that reconnects, or reaches a different master,
|
||||
/// knows nothing about what that master holds and starts again from the full
|
||||
/// list. The default has told no master anything, so it sends the whole list
|
||||
/// until one accepts changes.
|
||||
#[derive(Default)]
|
||||
pub struct VolumeReportState {
|
||||
/// Set once the master says it compares digests. Until then the whole list
|
||||
/// goes every time, which is what an older master needs.
|
||||
deltas_accepted: AtomicBool,
|
||||
full_list_needed: AtomicBool,
|
||||
/// Counts requests for the whole list, so one arriving while a heartbeat is
|
||||
/// being built is not marked satisfied by it.
|
||||
full_list_generation: AtomicU64,
|
||||
last_reported: Mutex<HashMap<VolumeReportKey, u64>>,
|
||||
}
|
||||
|
||||
impl VolumeReportState {
|
||||
/// Drops everything known about the master's view.
|
||||
pub fn reset(&self) {
|
||||
self.deltas_accepted.store(false, Ordering::Relaxed);
|
||||
self.full_list_needed.store(true, Ordering::Relaxed);
|
||||
self.full_list_generation.fetch_add(1, Ordering::Relaxed);
|
||||
self.last_reported.lock().unwrap().clear();
|
||||
}
|
||||
|
||||
pub fn accept_deltas(&self) {
|
||||
self.deltas_accepted.store(true, Ordering::Relaxed);
|
||||
}
|
||||
|
||||
pub fn request_full_list(&self) {
|
||||
self.full_list_needed.store(true, Ordering::Relaxed);
|
||||
self.full_list_generation.fetch_add(1, Ordering::Relaxed);
|
||||
}
|
||||
|
||||
/// Reports whether this heartbeat must carry the whole list, and the
|
||||
/// request it answers.
|
||||
pub fn begin(&self) -> (bool, u64) {
|
||||
let full = self.full_list_needed.load(Ordering::Relaxed)
|
||||
|| !self.deltas_accepted.load(Ordering::Relaxed);
|
||||
(full, self.full_list_generation.load(Ordering::Relaxed))
|
||||
}
|
||||
|
||||
/// Reports whether the master needs telling about this volume, given what
|
||||
/// it was last told.
|
||||
pub fn changed(&self, key: VolumeReportKey, hash: u64) -> bool {
|
||||
self.last_reported.lock().unwrap().get(&key) != Some(&hash)
|
||||
}
|
||||
|
||||
/// Records what this heartbeat told the master. Volumes absent from
|
||||
/// `reported` are forgotten, so one that comes back is reported again.
|
||||
pub fn commit(&self, reported: HashMap<VolumeReportKey, u64>, generation: u64) {
|
||||
*self.last_reported.lock().unwrap() = reported;
|
||||
// A request that arrived while this heartbeat was being built asked
|
||||
// about a later state than it carries, so it stands.
|
||||
if self.full_list_generation.load(Ordering::Relaxed) == generation {
|
||||
self.full_list_needed.store(false, Ordering::Relaxed);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,104 @@
|
||||
//! Mirror of `weed/storage/volume_report_hash.go`.
|
||||
//!
|
||||
//! The master compares the digest a volume server reports against one it
|
||||
//! computes itself, so this has to agree with the Go implementation
|
||||
//! byte-for-byte. `report_hash_vectors` pins that against values produced by
|
||||
//! the Go side; do not change the layout without regenerating them there.
|
||||
|
||||
use xxhash_rust::xxh64::xxh64;
|
||||
|
||||
use crate::pb::master_pb;
|
||||
|
||||
/// Digests everything a volume server reports about a volume.
|
||||
///
|
||||
/// It must cover every field of `VolumeInformationMessage`: a change the hash
|
||||
/// misses is a change the master would never be told about.
|
||||
pub fn report_hash(m: &master_pb::VolumeInformationMessage) -> u64 {
|
||||
let mut buf = [0u8; 57];
|
||||
buf[0..4].copy_from_slice(&m.id.to_le_bytes());
|
||||
buf[4..12].copy_from_slice(&m.size.to_le_bytes());
|
||||
buf[12..20].copy_from_slice(&m.file_count.to_le_bytes());
|
||||
buf[20..28].copy_from_slice(&m.delete_count.to_le_bytes());
|
||||
buf[28..36].copy_from_slice(&m.deleted_byte_count.to_le_bytes());
|
||||
// The master stores these narrowed, so hash what it will hold, not what the
|
||||
// wire type could carry.
|
||||
buf[36..40].copy_from_slice(&((m.replica_placement as u8) as u32).to_le_bytes());
|
||||
buf[40..44].copy_from_slice(&((m.version as u8) as u32).to_le_bytes());
|
||||
buf[44..48].copy_from_slice(&normalize_ttl(m.ttl).to_le_bytes());
|
||||
buf[48..52].copy_from_slice(&m.compact_revision.to_le_bytes());
|
||||
buf[52..56].copy_from_slice(&m.disk_id.to_le_bytes());
|
||||
if m.read_only {
|
||||
buf[56] = 1;
|
||||
}
|
||||
let mut h = xxh64(&buf, 0);
|
||||
|
||||
h = fold(h, xxh64(&(m.modified_at_second as u64).to_le_bytes(), 0));
|
||||
h = fold(h, xxh64(m.collection.as_bytes(), 0));
|
||||
h = fold(h, xxh64(m.disk_type.as_bytes(), 0));
|
||||
h = fold(h, xxh64(m.remote_storage_name.as_bytes(), 0));
|
||||
h = fold(h, xxh64(m.remote_storage_key.as_bytes(), 0));
|
||||
h
|
||||
}
|
||||
|
||||
/// A ttl whose count is zero encodes as zero however the unit is set, matching
|
||||
/// what the master stores after decoding it.
|
||||
fn normalize_ttl(ttl: u32) -> u32 {
|
||||
let count = (ttl >> 8) & 0xff;
|
||||
if count == 0 {
|
||||
return 0;
|
||||
}
|
||||
(count << 8) | (ttl & 0xff)
|
||||
}
|
||||
|
||||
/// Combines two hashes order-dependently, so swapping two string fields is not
|
||||
/// invisible.
|
||||
fn fold(h: u64, x: u64) -> u64 {
|
||||
let h = (h ^ x).wrapping_mul(0x9E37_79B9_7F4A_7C15);
|
||||
h ^ (h >> 29)
|
||||
}
|
||||
|
||||
#[cfg(test)]
|
||||
mod tests {
|
||||
use super::*;
|
||||
|
||||
// Produced by the Go implementation. If these drift, every volume server
|
||||
// running this build reports a digest the master can never match, and falls
|
||||
// back to sending its whole volume list forever.
|
||||
#[test]
|
||||
fn report_hash_vectors() {
|
||||
let empty = master_pb::VolumeInformationMessage::default();
|
||||
assert_eq!(report_hash(&empty), 17122085700329870549);
|
||||
|
||||
let mut one = master_pb::VolumeInformationMessage::default();
|
||||
one.id = 1;
|
||||
assert_eq!(report_hash(&one), 12867601919960834066);
|
||||
|
||||
let full = master_pb::VolumeInformationMessage {
|
||||
id: 42,
|
||||
size: 1 << 30,
|
||||
collection: "c".to_string(),
|
||||
file_count: 7,
|
||||
delete_count: 2,
|
||||
deleted_byte_count: 99,
|
||||
read_only: true,
|
||||
replica_placement: 10,
|
||||
version: 3,
|
||||
ttl: 3 << 8,
|
||||
compact_revision: 5,
|
||||
modified_at_second: 1700000000,
|
||||
remote_storage_name: "s3".to_string(),
|
||||
remote_storage_key: "k/1.dat".to_string(),
|
||||
disk_type: "ssd".to_string(),
|
||||
disk_id: 2,
|
||||
};
|
||||
assert_eq!(report_hash(&full), 12500327696413250175);
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn ttl_with_no_count_is_dropped() {
|
||||
assert_eq!(normalize_ttl(0), 0);
|
||||
assert_eq!(normalize_ttl(3), 0);
|
||||
assert_eq!(normalize_ttl(3 << 8), 3 << 8);
|
||||
assert_eq!(normalize_ttl((3 << 8) | 4), (3 << 8) | 4);
|
||||
}
|
||||
}
|
||||
@@ -184,6 +184,10 @@ GET /api/history?cluster_id=<uuid>&days=90
|
||||
# Get per-cluster disk usage and volume servers over time, largest first,
|
||||
# the rest summed as "other"
|
||||
GET /api/cluster-sizes?days=30&limit=20
|
||||
|
||||
# Get how many clusters ran each version over time, oldest version first,
|
||||
# the rest summed as "other"
|
||||
GET /api/versions?days=30&limit=8
|
||||
```
|
||||
|
||||
### Monitoring
|
||||
|
||||
@@ -174,6 +174,30 @@ func (h *Handler) GetClusterSizes(w http.ResponseWriter, r *http.Request) {
|
||||
json.NewEncoder(w).Encode(h.storage.GetClusterSizeSeries(days, limit))
|
||||
}
|
||||
|
||||
func (h *Handler) GetVersions(w http.ResponseWriter, r *http.Request) {
|
||||
if r.Method != http.MethodGet {
|
||||
http.Error(w, "Method not allowed", http.StatusMethodNotAllowed)
|
||||
return
|
||||
}
|
||||
|
||||
days := 30 // default
|
||||
if daysStr := r.URL.Query().Get("days"); daysStr != "" {
|
||||
if d, err := strconv.Atoi(daysStr); err == nil && d > 0 && d <= 365 {
|
||||
days = d
|
||||
}
|
||||
}
|
||||
|
||||
limit := 8 // default, the number of colours the version stack has
|
||||
if limitStr := r.URL.Query().Get("limit"); limitStr != "" {
|
||||
if l, err := strconv.Atoi(limitStr); err == nil && l > 0 && l <= 1000 {
|
||||
limit = l
|
||||
}
|
||||
}
|
||||
|
||||
w.Header().Set("Content-Type", "application/json")
|
||||
json.NewEncoder(w).Encode(h.storage.GetVersionSeries(days, limit))
|
||||
}
|
||||
|
||||
func (h *Handler) GetHistory(w http.ResponseWriter, r *http.Request) {
|
||||
if r.Method != http.MethodGet {
|
||||
http.Error(w, "Method not allowed", http.StatusMethodNotAllowed)
|
||||
|
||||
@@ -143,12 +143,24 @@ func (h *Handler) ServeIndex(w http.ResponseWriter, r *http.Request) {
|
||||
|
||||
<div class="chart-container">
|
||||
<div class="chart-title">Version Distribution</div>
|
||||
<canvas id="versionChart" width="400" height="200"></canvas>
|
||||
<div style="position: relative; height: 320px;">
|
||||
<canvas id="versionChart"></canvas>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="chart-container">
|
||||
<div class="chart-title">Versions Over Time</div>
|
||||
<div class="chart-subtitle" id="versionSeriesTotal"></div>
|
||||
<div style="position: relative; height: 420px;">
|
||||
<canvas id="versionSeriesChart"></canvas>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="chart-container">
|
||||
<div class="chart-title">Operating System Distribution</div>
|
||||
<canvas id="osChart" width="400" height="200"></canvas>
|
||||
<div style="position: relative; height: 320px;">
|
||||
<canvas id="osChart"></canvas>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="chart-container">
|
||||
@@ -197,12 +209,20 @@ func (h *Handler) ServeIndex(w http.ResponseWriter, r *http.Request) {
|
||||
const sizesResponse = await fetch('/api/cluster-sizes?days=30&limit=20');
|
||||
const sizes = await sizesResponse.json();
|
||||
|
||||
updateStats(stats);
|
||||
updateCharts(stats);
|
||||
updateClusterSizes(sizes);
|
||||
|
||||
// Load the fleet's version make-up over time
|
||||
const versionsResponse = await fetch('/api/versions?days=30&limit=8');
|
||||
const versions = await versionsResponse.json();
|
||||
|
||||
// Show the dashboard before drawing into it: a canvas in a
|
||||
// display:none container measures zero, and a pie sized from
|
||||
// that never grows back.
|
||||
document.getElementById('loading').style.display = 'none';
|
||||
document.getElementById('dashboard').style.display = 'block';
|
||||
|
||||
updateStats(stats);
|
||||
updateCharts(stats);
|
||||
updateVersions(versions);
|
||||
updateClusterSizes(sizes);
|
||||
} catch (error) {
|
||||
console.error('Error loading dashboard:', error);
|
||||
showError('Failed to load telemetry data: ' + error.message);
|
||||
@@ -246,6 +266,9 @@ func (h *Handler) ServeIndex(w http.ResponseWriter, r *http.Request) {
|
||||
},
|
||||
options: {
|
||||
responsive: true,
|
||||
// Without this the canvas keeps its 2:1 attribute ratio at
|
||||
// the card's full width, drawing a pie taller than the card.
|
||||
maintainAspectRatio: false,
|
||||
plugins: {
|
||||
legend: {
|
||||
position: 'bottom'
|
||||
@@ -298,6 +321,76 @@ func (h *Handler) ServeIndex(w http.ResponseWriter, r *http.Request) {
|
||||
return (unit === 0 ? value : value.toFixed(value >= 100 ? 0 : 1)) + ' ' + units[unit];
|
||||
}
|
||||
|
||||
// Fixed hues rather than the evenly spaced ones the cluster stacks use:
|
||||
// a handful of versions is a set you read, and evenly spaced hues put
|
||||
// pairs next to each other that colour-blind readers can't separate.
|
||||
const versionColors = ['#2a78d6', '#eb6834', '#1baf7a', '#eda100',
|
||||
'#e87ba4', '#008300', '#4a3aa7', '#e34948'];
|
||||
|
||||
// One stacked band per version: the band is how many clusters ran that
|
||||
// version that day, the top of the stack is the confirmed fleet, so the
|
||||
// chart shows both how it grows and what it upgrades to. The pie above
|
||||
// it is the same fleet on the last of these days.
|
||||
function updateVersions(series) {
|
||||
const versions = series.versions || [];
|
||||
const dates = series.dates || [];
|
||||
const total = series.total_clusters || 0;
|
||||
document.getElementById('versionSeriesTotal').textContent =
|
||||
total + ' cluster' + (total === 1 ? '' : 's') + ' on ' + (dates[dates.length - 1] || 'no data');
|
||||
|
||||
// Newest release on the floor, oldest on top: the current release
|
||||
// is the band being read, and one anchored to the baseline reads
|
||||
// straight off the axis instead of riding on everything below it.
|
||||
// Colours follow the same order, so a release keeps its colour as
|
||||
// older ones age out from the top.
|
||||
const datasets = versions.slice().reverse().map((v, i) =>
|
||||
band(v.version, v.clusters, '#ffffff', versionColors[i % versionColors.length], 2));
|
||||
if (series.other) {
|
||||
datasets.push(band('other (' + series.other.count + ' versions)', series.other.clusters,
|
||||
'#ffffff', '#9a9a94', 2));
|
||||
}
|
||||
|
||||
stackedArea('versionSeriesChart', dates, datasets, value => value, { plugins: [bandLabels] });
|
||||
}
|
||||
|
||||
// Writes each version into its own band, so the chart reads without
|
||||
// matching colours against the legend. Bands too thin to hold the text
|
||||
// keep it, and the halo carries it over whatever it crosses.
|
||||
const bandLabels = {
|
||||
id: 'bandLabels',
|
||||
afterDatasetsDraw(chart) {
|
||||
const ctx = chart.ctx;
|
||||
ctx.save();
|
||||
ctx.font = '600 12px -apple-system, BlinkMacSystemFont, sans-serif';
|
||||
ctx.textAlign = 'center';
|
||||
ctx.textBaseline = 'middle';
|
||||
ctx.lineJoin = 'round';
|
||||
chart.data.datasets.forEach((dataset, d) => {
|
||||
const meta = chart.getDatasetMeta(d);
|
||||
if (meta.hidden) return;
|
||||
// Label where the band is thickest, so the text has room.
|
||||
let at = -1, thickest = 0;
|
||||
dataset.data.forEach((value, i) => {
|
||||
if (value > thickest) {
|
||||
thickest = value;
|
||||
at = i;
|
||||
}
|
||||
});
|
||||
const point = at >= 0 && meta.data[at];
|
||||
if (!point) return;
|
||||
const below = d === 0 ? chart.scales.y.getPixelForValue(0)
|
||||
: chart.getDatasetMeta(d - 1).data[at].y;
|
||||
if (below - point.y < 18) return;
|
||||
ctx.lineWidth = 3;
|
||||
ctx.strokeStyle = 'rgba(255, 255, 255, 0.85)';
|
||||
ctx.strokeText(dataset.label, point.x, (point.y + below) / 2);
|
||||
ctx.fillStyle = '#1a1a1a';
|
||||
ctx.fillText(dataset.label, point.x, (point.y + below) / 2);
|
||||
});
|
||||
ctx.restore();
|
||||
}
|
||||
};
|
||||
|
||||
// Disk usage and volume servers both drawn as one stacked band per
|
||||
// cluster over time: the band is that cluster's share, the top of the
|
||||
// stack is the fleet total. Clusters beyond the requested limit are
|
||||
@@ -337,6 +430,24 @@ func (h *Handler) ServeIndex(w http.ResponseWriter, r *http.Request) {
|
||||
'#9E9E9E', 'rgba(158, 158, 158, 0.6)'));
|
||||
}
|
||||
|
||||
stackedArea(canvasId, dates, datasets, format, {
|
||||
onClick: (event, elements) => {
|
||||
const id = elements.length && clusterSizeIds[elements[0].datasetIndex];
|
||||
if (id) {
|
||||
document.getElementById('clusterIdInput').value = id;
|
||||
loadClusterHistory();
|
||||
}
|
||||
},
|
||||
// The legend shows shortened ids; the tooltip has room for the
|
||||
// full one to paste into the lookup box.
|
||||
label: item => (clusterSizeIds[item.datasetIndex] || item.dataset.label) + ': ' + format(item.raw)
|
||||
});
|
||||
}
|
||||
|
||||
// Draws the bands as one stack, so their heights add up to the day's
|
||||
// total. hooks.onClick, hooks.label and hooks.plugins are optional.
|
||||
function stackedArea(canvasId, dates, datasets, format, hooks) {
|
||||
hooks = hooks || {};
|
||||
const ctx = document.getElementById(canvasId).getContext('2d');
|
||||
if (charts[canvasId]) {
|
||||
charts[canvasId].destroy();
|
||||
@@ -344,25 +455,17 @@ func (h *Handler) ServeIndex(w http.ResponseWriter, r *http.Request) {
|
||||
charts[canvasId] = new Chart(ctx, {
|
||||
type: 'line',
|
||||
data: { labels: dates, datasets: datasets },
|
||||
plugins: hooks.plugins,
|
||||
options: {
|
||||
responsive: true,
|
||||
maintainAspectRatio: false,
|
||||
interaction: { mode: 'band', intersect: false },
|
||||
onClick: (event, elements) => {
|
||||
const id = elements.length && clusterSizeIds[elements[0].datasetIndex];
|
||||
if (id) {
|
||||
document.getElementById('clusterIdInput').value = id;
|
||||
loadClusterHistory();
|
||||
}
|
||||
},
|
||||
onClick: hooks.onClick,
|
||||
plugins: {
|
||||
legend: { position: 'bottom', labels: { boxWidth: 12, font: { size: 11 } } },
|
||||
tooltip: {
|
||||
callbacks: {
|
||||
label: item => {
|
||||
const id = clusterSizeIds[item.datasetIndex];
|
||||
return (id || item.dataset.label) + ': ' + format(item.raw);
|
||||
}
|
||||
label: hooks.label || (item => item.dataset.label + ': ' + format(item.raw))
|
||||
}
|
||||
}
|
||||
},
|
||||
@@ -396,13 +499,13 @@ func (h *Handler) ServeIndex(w http.ResponseWriter, r *http.Request) {
|
||||
return [];
|
||||
};
|
||||
|
||||
function band(label, data, borderColor, backgroundColor) {
|
||||
function band(label, data, borderColor, backgroundColor, borderWidth) {
|
||||
return {
|
||||
label: label,
|
||||
data: data,
|
||||
borderColor: borderColor,
|
||||
backgroundColor: backgroundColor,
|
||||
borderWidth: 1,
|
||||
borderWidth: borderWidth || 1,
|
||||
pointRadius: 0,
|
||||
pointHitRadius: 8,
|
||||
fill: true,
|
||||
|
||||
@@ -77,6 +77,7 @@ func main() {
|
||||
mux.HandleFunc("/api/metrics", corsMiddleware(logMiddleware(apiHandler.GetMetrics)))
|
||||
mux.HandleFunc("/api/history", corsMiddleware(logMiddleware(apiHandler.GetHistory)))
|
||||
mux.HandleFunc("/api/cluster-sizes", corsMiddleware(logMiddleware(apiHandler.GetClusterSizes)))
|
||||
mux.HandleFunc("/api/versions", corsMiddleware(logMiddleware(apiHandler.GetVersions)))
|
||||
|
||||
// Dashboard (optional)
|
||||
if *enableDashboard {
|
||||
|
||||
@@ -20,6 +20,7 @@ type HistorySample struct {
|
||||
TotalDiskBytes uint64 `json:"disk"`
|
||||
TotalVolumeCount int32 `json:"volumes"`
|
||||
VolumeServerCount int32 `json:"servers"`
|
||||
Version string `json:"ver,omitempty"` // empty in samples written before this was recorded
|
||||
}
|
||||
|
||||
// appendHistory records the report as the cluster's sample for the day,
|
||||
@@ -30,6 +31,7 @@ func (s *PrometheusStorage) appendHistory(data *proto.TelemetryData, receivedAt
|
||||
TotalDiskBytes: data.TotalDiskBytes,
|
||||
TotalVolumeCount: data.TotalVolumeCount,
|
||||
VolumeServerCount: data.VolumeServerCount,
|
||||
Version: data.Version,
|
||||
}
|
||||
h := s.histories[data.TopologyId]
|
||||
if n := len(h); n > 0 && sameUTCDay(h[n-1].Ts, sample.Ts) {
|
||||
@@ -110,16 +112,17 @@ func utcDay(ts int64) time.Time {
|
||||
return time.Date(t.Year(), t.Month(), t.Day(), 0, 0, 0, 0, time.UTC)
|
||||
}
|
||||
|
||||
func diskBytes(s HistorySample) uint64 { return s.TotalDiskBytes }
|
||||
func serverCount(s HistorySample) uint64 { return uint64(s.VolumeServerCount) }
|
||||
func diskBytes(s HistorySample) uint64 { return s.TotalDiskBytes }
|
||||
func serverCount(s HistorySample) uint64 { return uint64(s.VolumeServerCount) }
|
||||
func sampleVersion(s HistorySample) string { return s.Version }
|
||||
|
||||
// align lays one cluster's history onto the axis, picking `value` out of each
|
||||
// sample. Clusters report roughly once a day at no fixed hour, so a day without
|
||||
// a report carries the previous value forward rather than dropping to zero; a
|
||||
// cluster that stopped reporting altogether ends at its last sample instead of
|
||||
// holding capacity forever. Reports false when the cluster has nothing in range.
|
||||
func (d dailySeries) align(history []HistorySample, activeSince int64, value func(HistorySample) uint64) ([]uint64, bool) {
|
||||
out := make([]uint64, len(d.dates))
|
||||
func align[T any](d dailySeries, history []HistorySample, activeSince int64, value func(HistorySample) T) ([]T, bool) {
|
||||
out := make([]T, len(d.dates))
|
||||
reported := make([]bool, len(d.dates))
|
||||
first, last := -1, -1
|
||||
for _, sample := range history {
|
||||
|
||||
@@ -44,9 +44,18 @@ func (s *PrometheusStorage) LoadState(path string) (int, error) {
|
||||
loaded++
|
||||
}
|
||||
for id, h := range state.Histories {
|
||||
if _, ok := s.instances[id]; ok && len(h) > 0 {
|
||||
s.histories[id] = h
|
||||
instance, ok := s.instances[id]
|
||||
if !ok || len(h) == 0 {
|
||||
continue
|
||||
}
|
||||
// State written before versions were recorded carries none on its
|
||||
// samples. The newest sample is the report the instance record itself
|
||||
// came from, so that day's version is known and the version series can
|
||||
// start there rather than a day after the upgrade.
|
||||
if newest := len(h) - 1; h[newest].Version == "" {
|
||||
h[newest].Version = instance.TelemetryData.Version
|
||||
}
|
||||
s.histories[id] = h
|
||||
}
|
||||
s.updateStats()
|
||||
return loaded, nil
|
||||
|
||||
@@ -72,3 +72,38 @@ func TestStateRoundTrip(t *testing.T) {
|
||||
t.Error("dirty flag set after load")
|
||||
}
|
||||
}
|
||||
|
||||
// State written before versions were recorded still knows the version of the
|
||||
// report its newest sample came from: the instance record's.
|
||||
func TestLoadStateFillsNewestSampleVersion(t *testing.T) {
|
||||
path := filepath.Join(t.TempDir(), "telemetry-state.json")
|
||||
|
||||
s := newPrometheusStorage(prometheus.NewRegistry())
|
||||
report := &proto.TelemetryData{TopologyId: "test-cluster-1", Version: "4.40", Os: "linux/amd64"}
|
||||
if err := s.StoreTelemetry(report); err != nil {
|
||||
t.Fatalf("store: %v", err)
|
||||
}
|
||||
yesterday := time.Now().AddDate(0, 0, -1).Unix()
|
||||
s.histories[report.TopologyId] = []HistorySample{
|
||||
{Ts: yesterday, TotalDiskBytes: 10},
|
||||
{Ts: time.Now().Unix(), TotalDiskBytes: 20},
|
||||
}
|
||||
if err := s.SaveStateIfDirty(path); err != nil {
|
||||
t.Fatalf("save: %v", err)
|
||||
}
|
||||
|
||||
s = newPrometheusStorage(prometheus.NewRegistry())
|
||||
if _, err := s.LoadState(path); err != nil {
|
||||
t.Fatalf("load: %v", err)
|
||||
}
|
||||
loaded := s.histories[report.TopologyId]
|
||||
if len(loaded) != 2 {
|
||||
t.Fatalf("history = %+v, want 2 samples", loaded)
|
||||
}
|
||||
if loaded[1].Version != report.Version {
|
||||
t.Errorf("newest sample version = %q, want %q", loaded[1].Version, report.Version)
|
||||
}
|
||||
if loaded[0].Version != "" {
|
||||
t.Errorf("older sample version = %q, want it left unknown", loaded[0].Version)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -180,12 +180,12 @@ func (s *PrometheusStorage) GetMetrics(days int) (map[string]interface{}, error)
|
||||
diskUsage := make([]uint64, len(axis.dates))
|
||||
serverCounts := make([]int64, len(axis.dates))
|
||||
for _, history := range histories {
|
||||
if disk, ok := axis.align(history, activeSince, diskBytes); ok {
|
||||
if disk, ok := align(axis, history, activeSince, diskBytes); ok {
|
||||
for i, v := range disk {
|
||||
diskUsage[i] += v
|
||||
}
|
||||
}
|
||||
if servers, ok := axis.align(history, activeSince, serverCount); ok {
|
||||
if servers, ok := align(axis, history, activeSince, serverCount); ok {
|
||||
for i, v := range servers {
|
||||
serverCounts[i] += int64(v)
|
||||
}
|
||||
|
||||
@@ -48,11 +48,11 @@ func (s *PrometheusStorage) GetClusterSizeSeries(days, limit int) ClusterSizeSer
|
||||
series := ClusterSizeSeries{Dates: axis.dates}
|
||||
|
||||
for id, history := range histories {
|
||||
disk, ok := axis.align(history, activeSince, diskBytes)
|
||||
disk, ok := align(axis, history, activeSince, diskBytes)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
servers, _ := axis.align(history, activeSince, serverCount)
|
||||
servers, _ := align(axis, history, activeSince, serverCount)
|
||||
series.Clusters = append(series.Clusters, ClusterSeries{ClusterId: id, Disk: disk, Servers: servers})
|
||||
series.TotalDisk += disk[last]
|
||||
series.TotalServers += servers[last]
|
||||
|
||||
@@ -0,0 +1,163 @@
|
||||
package storage
|
||||
|
||||
import (
|
||||
"sort"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
// VersionCounts is how many clusters ran one version per day, aligned to the
|
||||
// shared date axis of the enclosing VersionSeries.
|
||||
type VersionCounts struct {
|
||||
Version string `json:"version"`
|
||||
Clusters []uint64 `json:"clusters"`
|
||||
}
|
||||
|
||||
// OtherVersions is the versions beyond the caller's limit, summed per day so a
|
||||
// stacked chart still adds up to the cluster total.
|
||||
type OtherVersions struct {
|
||||
Count int `json:"count"`
|
||||
Clusters []uint64 `json:"clusters"`
|
||||
}
|
||||
|
||||
// VersionSeries is the version make-up of the fleet over time: one cluster
|
||||
// count per version per day, oldest release first.
|
||||
type VersionSeries struct {
|
||||
Dates []string `json:"dates"`
|
||||
Versions []VersionCounts `json:"versions"`
|
||||
Other *OtherVersions `json:"other,omitempty"`
|
||||
TotalClusters uint64 `json:"total_clusters"` // across all versions on the last day
|
||||
}
|
||||
|
||||
// GetVersionSeries returns the last `days` days of per-version cluster counts
|
||||
// across confirmed clusters. Versions beyond `limit` are folded into Other,
|
||||
// keeping the ones with the most clusters on the last day.
|
||||
func (s *PrometheusStorage) GetVersionSeries(days, limit int) VersionSeries {
|
||||
s.mu.RLock()
|
||||
defer s.mu.RUnlock()
|
||||
|
||||
histories := versionedHistories(s.seriesHistories())
|
||||
axis := newDailySeries(days, histories)
|
||||
activeSince := time.Now().UTC().AddDate(0, 0, -activeDays).Unix()
|
||||
last := len(axis.dates) - 1
|
||||
series := VersionSeries{Dates: axis.dates}
|
||||
|
||||
daily := make(map[string][]uint64)
|
||||
for _, history := range histories {
|
||||
versions, ok := align(axis, history, activeSince, sampleVersion)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
for i, v := range versions {
|
||||
if v == "" {
|
||||
continue // before the cluster's first day in the window
|
||||
}
|
||||
counts, ok := daily[v]
|
||||
if !ok {
|
||||
counts = make([]uint64, len(axis.dates))
|
||||
daily[v] = counts
|
||||
}
|
||||
counts[i]++
|
||||
}
|
||||
}
|
||||
|
||||
for v, counts := range daily {
|
||||
series.Versions = append(series.Versions, VersionCounts{Version: v, Clusters: counts})
|
||||
series.TotalClusters += counts[last]
|
||||
}
|
||||
// Ordered by release rather than by size: the caller stacks them, and a
|
||||
// stack whose order changes with the counts is unreadable over time.
|
||||
sort.Slice(series.Versions, func(i, j int) bool {
|
||||
return versionLess(series.Versions[i].Version, series.Versions[j].Version)
|
||||
})
|
||||
|
||||
if limit > 0 && len(series.Versions) > limit {
|
||||
// Old releases are a long tail of one-cluster bands; keep the versions
|
||||
// most of the fleet is on now and sum the rest into one band.
|
||||
ranked := append([]VersionCounts(nil), series.Versions...)
|
||||
sort.Slice(ranked, func(i, j int) bool {
|
||||
if ranked[i].Clusters[last] != ranked[j].Clusters[last] {
|
||||
return ranked[i].Clusters[last] > ranked[j].Clusters[last]
|
||||
}
|
||||
return versionLess(ranked[j].Version, ranked[i].Version)
|
||||
})
|
||||
other := OtherVersions{Count: len(ranked) - limit, Clusters: make([]uint64, len(axis.dates))}
|
||||
for _, v := range ranked[limit:] {
|
||||
for i, c := range v.Clusters {
|
||||
other.Clusters[i] += c
|
||||
}
|
||||
}
|
||||
kept := make(map[string]bool, limit)
|
||||
for _, v := range ranked[:limit] {
|
||||
kept[v.Version] = true
|
||||
}
|
||||
versions := series.Versions[:0]
|
||||
for _, v := range series.Versions {
|
||||
if kept[v.Version] {
|
||||
versions = append(versions, v)
|
||||
}
|
||||
}
|
||||
series.Versions = versions
|
||||
series.Other = &other
|
||||
}
|
||||
return series
|
||||
}
|
||||
|
||||
// versionedHistories drops the samples recorded before the reported version was
|
||||
// kept in history, so the version series spans the days it actually knows a
|
||||
// version for instead of climbing out of a run of blank days.
|
||||
func versionedHistories(histories map[string][]HistorySample) map[string][]HistorySample {
|
||||
out := make(map[string][]HistorySample, len(histories))
|
||||
for id, history := range histories {
|
||||
var kept []HistorySample
|
||||
for _, sample := range history {
|
||||
if sample.Version != "" {
|
||||
kept = append(kept, sample)
|
||||
}
|
||||
}
|
||||
if len(kept) > 0 {
|
||||
out[id] = kept
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// versionLess orders release strings like "3.97" and "4.40" by their numeric
|
||||
// components rather than lexically, which would sort "10.02" before "9.99". A
|
||||
// build suffix sits next to the release it was built from. Anything that
|
||||
// doesn't parse, such as the "unknown" a client with no version compiled in
|
||||
// reports, sorts first.
|
||||
func versionLess(a, b string) bool {
|
||||
an, aSuffix, aok := versionParts(a)
|
||||
bn, bSuffix, bok := versionParts(b)
|
||||
if aok != bok {
|
||||
return !aok
|
||||
}
|
||||
if !aok {
|
||||
return a < b
|
||||
}
|
||||
for i := 0; i < len(an) && i < len(bn); i++ {
|
||||
if an[i] != bn[i] {
|
||||
return an[i] < bn[i]
|
||||
}
|
||||
}
|
||||
if len(an) != len(bn) {
|
||||
return len(an) < len(bn)
|
||||
}
|
||||
return aSuffix < bSuffix
|
||||
}
|
||||
|
||||
func versionParts(v string) ([]int, string, bool) {
|
||||
number, suffix, _ := strings.Cut(v, "-")
|
||||
fields := strings.Split(number, ".")
|
||||
parts := make([]int, 0, len(fields))
|
||||
for _, field := range fields {
|
||||
n, err := strconv.Atoi(field)
|
||||
if err != nil {
|
||||
return nil, "", false
|
||||
}
|
||||
parts = append(parts, n)
|
||||
}
|
||||
return parts, suffix, len(parts) > 0
|
||||
}
|
||||
@@ -0,0 +1,109 @@
|
||||
package storage
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"github.com/prometheus/client_golang/prometheus"
|
||||
)
|
||||
|
||||
func TestVersionSeries(t *testing.T) {
|
||||
s := newPrometheusStorage(prometheus.NewRegistry())
|
||||
|
||||
// Upgraded mid-window: its band leaves the old version for the new one.
|
||||
seedSamples(s, "upgraded", HistorySample{Version: "4.39"}, -4, -3)
|
||||
seedSamples(s, "upgraded", HistorySample{Version: "4.40"}, -2, -1, 0)
|
||||
// Reported every day on the same version.
|
||||
seedSamples(s, "steady", HistorySample{Version: "4.40"}, -4, -3, -2, -1, 0)
|
||||
// One day of history only: unconfirmed, so it stays out of the stack.
|
||||
seedSamples(s, "oneshot", HistorySample{Version: "4.40"}, -1)
|
||||
// Confirmed, but its samples predate versions being recorded.
|
||||
seedSamples(s, "versionless", HistorySample{}, -9, -8)
|
||||
|
||||
series := s.GetVersionSeries(10, 0)
|
||||
|
||||
// The versionless days are dropped before the axis is built, so the chart
|
||||
// spans the days a version is known for instead of climbing out of blanks.
|
||||
if len(series.Dates) != 5 {
|
||||
t.Fatalf("dates = %v, want the 5 days with versions", series.Dates)
|
||||
}
|
||||
if len(series.Versions) != 2 ||
|
||||
series.Versions[0].Version != "4.39" || series.Versions[1].Version != "4.40" {
|
||||
t.Fatalf("versions = %+v, want 4.39 then 4.40", series.Versions)
|
||||
}
|
||||
if got := series.Versions[0].Clusters; !equal(got, []uint64{1, 1, 0, 0, 0}) {
|
||||
t.Errorf("4.39 = %v, want the upgraded cluster's first two days", got)
|
||||
}
|
||||
if got := series.Versions[1].Clusters; !equal(got, []uint64{1, 1, 2, 2, 2}) {
|
||||
t.Errorf("4.40 = %v, want steady plus upgraded from day 3", got)
|
||||
}
|
||||
if series.TotalClusters != 2 {
|
||||
t.Errorf("total_clusters = %d, want 2", series.TotalClusters)
|
||||
}
|
||||
if series.Other != nil {
|
||||
t.Errorf("other = %+v, want none without a limit", series.Other)
|
||||
}
|
||||
}
|
||||
|
||||
func TestVersionSeriesHoldsForwardAndLimits(t *testing.T) {
|
||||
s := newPrometheusStorage(prometheus.NewRegistry())
|
||||
|
||||
// Stopped reporting past the active window: its own days only.
|
||||
seedSamples(s, "gone", HistorySample{Version: "3.97"}, -9, -8)
|
||||
// Reported every day of the window.
|
||||
seedSamples(s, "daily", HistorySample{Version: "4.40"}, -9, -8, -7, -6, -5, -4, -3, -2, -1, 0)
|
||||
// Reported three days ago and not since: still active, so it holds its
|
||||
// version to the right edge instead of dropping out of the stack.
|
||||
seedSamples(s, "lagging", HistorySample{Version: "4.30"}, -3, -2)
|
||||
|
||||
series := s.GetVersionSeries(10, 0)
|
||||
if len(series.Dates) != 10 {
|
||||
t.Fatalf("dates = %v, want 10 days", series.Dates)
|
||||
}
|
||||
byVersion := map[string][]uint64{}
|
||||
for _, v := range series.Versions {
|
||||
byVersion[v.Version] = v.Clusters
|
||||
}
|
||||
if got := byVersion["3.97"]; !equal(got, []uint64{1, 1, 0, 0, 0, 0, 0, 0, 0, 0}) {
|
||||
t.Errorf("3.97 = %v, want nothing after its last report", got)
|
||||
}
|
||||
if got := byVersion["4.30"]; !equal(got, []uint64{0, 0, 0, 0, 0, 0, 1, 1, 1, 1}) {
|
||||
t.Errorf("4.30 = %v, want carried to the right edge", got)
|
||||
}
|
||||
if got := byVersion["4.40"]; !equal(got, []uint64{1, 1, 1, 1, 1, 1, 1, 1, 1, 1}) {
|
||||
t.Errorf("4.40 = %v, want 1 every day", got)
|
||||
}
|
||||
if series.TotalClusters != 2 {
|
||||
t.Errorf("total_clusters = %d, want the 2 clusters still reporting", series.TotalClusters)
|
||||
}
|
||||
|
||||
// Past the limit, the versions the fewest clusters are on are summed into
|
||||
// one band; ties on the last day keep the newer version.
|
||||
series = s.GetVersionSeries(10, 1)
|
||||
if len(series.Versions) != 1 || series.Versions[0].Version != "4.40" {
|
||||
t.Fatalf("limited versions = %+v, want just 4.40", series.Versions)
|
||||
}
|
||||
if series.Other == nil || series.Other.Count != 2 {
|
||||
t.Fatalf("other = %+v, want 2 versions", series.Other)
|
||||
}
|
||||
if !equal(series.Other.Clusters, []uint64{1, 1, 0, 0, 0, 0, 1, 1, 1, 1}) {
|
||||
t.Errorf("other = %v, want 3.97+4.30 summed per day", series.Other.Clusters)
|
||||
}
|
||||
if series.TotalClusters != 2 {
|
||||
t.Errorf("limit changed total_clusters: %d, want 2", series.TotalClusters)
|
||||
}
|
||||
}
|
||||
|
||||
func TestVersionLess(t *testing.T) {
|
||||
// Ordered oldest to newest; every pair must compare in this order.
|
||||
ordered := []string{"unknown", "3.97", "4.02", "4.30", "4.30-enterprise", "9.99", "10.02"}
|
||||
for i := 0; i < len(ordered); i++ {
|
||||
for j := i + 1; j < len(ordered); j++ {
|
||||
if !versionLess(ordered[i], ordered[j]) {
|
||||
t.Errorf("versionLess(%q, %q) = false, want true", ordered[i], ordered[j])
|
||||
}
|
||||
if versionLess(ordered[j], ordered[i]) {
|
||||
t.Errorf("versionLess(%q, %q) = true, want false", ordered[j], ordered[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,248 @@
|
||||
//go:build !windows
|
||||
|
||||
package metadata_subscribe
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strconv"
|
||||
"strings"
|
||||
"syscall"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/stretchr/testify/require"
|
||||
"google.golang.org/grpc"
|
||||
"google.golang.org/grpc/credentials/insecure"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/test/testutil"
|
||||
"github.com/seaweedfs/seaweedfs/weed/cluster"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
|
||||
)
|
||||
|
||||
// A filer drops the metadata subscription to a peer that leaves, and only an
|
||||
// add from the master brings it back. Those updates are broadcast to the
|
||||
// clients connected at that moment, so a filer whose master stream broke while
|
||||
// the peer came back used to stay unsubscribed for good, and metadata written
|
||||
// on the peer never reached it again.
|
||||
func TestFilerResubscribesToPeerAfterMasterReconnect(t *testing.T) {
|
||||
if testing.Short() {
|
||||
t.Skip("skipping integration test in short mode")
|
||||
}
|
||||
|
||||
weedBinary := findWeedBinary()
|
||||
require.NotEmpty(t, weedBinary, "weed binary not found")
|
||||
|
||||
testDir, err := os.MkdirTemp("", "seaweedfs_peer_resubscribe_")
|
||||
require.NoError(t, err)
|
||||
t.Cleanup(func() {
|
||||
if t.Failed() {
|
||||
t.Logf("logs kept at %s", testDir)
|
||||
return
|
||||
}
|
||||
os.RemoveAll(testDir)
|
||||
})
|
||||
|
||||
ports, err := testutil.AllocateMiniPorts(3)
|
||||
require.NoError(t, err)
|
||||
masterPort, filer1Port, filer2Port := ports[0], ports[1], ports[2]
|
||||
|
||||
master := pb.ServerAddress(fmt.Sprintf("127.0.0.1:%d", masterPort))
|
||||
filer1Address := fmt.Sprintf("127.0.0.1:%d", filer1Port)
|
||||
filer2Address := fmt.Sprintf("127.0.0.1:%d", filer2Port)
|
||||
filer1Log := filepath.Join(testDir, "filer1.log")
|
||||
|
||||
masterArgs := []string{"master",
|
||||
"-ip=127.0.0.1",
|
||||
"-port=" + strconv.Itoa(masterPort),
|
||||
"-mdir=" + mkdir(t, testDir, "master"),
|
||||
"-peers=none"}
|
||||
masterProcess := startProcess(t, weedBinary, filepath.Join(testDir, "master.log"), masterArgs...)
|
||||
require.NoError(t, waitForLeader(masterPort, 60*time.Second))
|
||||
|
||||
// one at a time: a filer bootstraps from the peers the master already knows,
|
||||
// and gives up if one of them is registered but not yet listening
|
||||
filer1 := startFiler(t, weedBinary, testDir, "filer1", filer1Port, masterPort)
|
||||
require.NoError(t, waitForHTTPServer(fmt.Sprintf("http://127.0.0.1:%d/", filer1Port), 30*time.Second))
|
||||
filer2 := startFiler(t, weedBinary, testDir, "filer2", filer2Port, masterPort)
|
||||
require.NoError(t, waitForHTTPServer(fmt.Sprintf("http://127.0.0.1:%d/", filer2Port), 30*time.Second))
|
||||
|
||||
// the peer subscription works to begin with
|
||||
createPeerEntry(t, filer2Address, "baseline")
|
||||
require.NoError(t, waitForPeerEntry(filer1Address, "baseline", 60*time.Second),
|
||||
"filer1 never replicated the baseline entry from filer2")
|
||||
|
||||
// filer2 leaves, and filer1 drops the subscription
|
||||
stopProcess(filer2)
|
||||
require.NoError(t, waitForLog(filer1Log, "stop subscribing peer "+filer2Address, 60*time.Second),
|
||||
"filer1 never dropped the subscription to filer2")
|
||||
|
||||
// filer1 stops reading its master stream, and restarting the master breaks
|
||||
// it, so filer1 hears nothing until it reconnects
|
||||
require.NoError(t, filer1.Process.Signal(syscall.SIGSTOP))
|
||||
stopProcess(masterProcess)
|
||||
startProcess(t, weedBinary, filepath.Join(testDir, "master.log"), masterArgs...)
|
||||
require.NoError(t, waitForLeader(masterPort, 60*time.Second))
|
||||
|
||||
// filer2 comes back and registers while filer1 cannot hear about it
|
||||
filer2 = startFiler(t, weedBinary, testDir, "filer2", filer2Port, masterPort)
|
||||
require.NoError(t, waitForHTTPServer(fmt.Sprintf("http://127.0.0.1:%d/", filer2Port), 30*time.Second))
|
||||
require.NoError(t, waitForClusterNode(master, filer2Address, 60*time.Second),
|
||||
"the master never registered filer2 again")
|
||||
|
||||
require.NoError(t, filer1.Process.Signal(syscall.SIGCONT))
|
||||
require.NoError(t, waitForClusterNode(master, filer1Address, 60*time.Second),
|
||||
"filer1 never reconnected to the master")
|
||||
|
||||
createPeerEntry(t, filer2Address, "after-reconnect")
|
||||
require.NoError(t, waitForPeerEntry(filer1Address, "after-reconnect", 90*time.Second),
|
||||
"filer1 reconnected to the master but never resubscribed to filer2")
|
||||
}
|
||||
|
||||
const peerEntryDir = "/peer-resubscribe"
|
||||
|
||||
func createPeerEntry(t *testing.T, filerAddress, name string) {
|
||||
t.Helper()
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 30*time.Second)
|
||||
defer cancel()
|
||||
err := pb.WithFilerClient(false, 0, pb.ServerAddress(filerAddress), grpc.WithTransportCredentials(insecure.NewCredentials()), func(client filer_pb.SeaweedFilerClient) error {
|
||||
_, err := client.CreateEntry(ctx, &filer_pb.CreateEntryRequest{
|
||||
Directory: peerEntryDir,
|
||||
Entry: &filer_pb.Entry{
|
||||
Name: name,
|
||||
Attributes: &filer_pb.FuseAttributes{
|
||||
Mtime: time.Now().Unix(),
|
||||
FileMode: 0644,
|
||||
},
|
||||
},
|
||||
})
|
||||
return err
|
||||
})
|
||||
require.NoError(t, err, "create %s/%s on %s", peerEntryDir, name, filerAddress)
|
||||
}
|
||||
|
||||
func waitForPeerEntry(filerAddress, name string, timeout time.Duration) error {
|
||||
deadline := time.Now().Add(timeout)
|
||||
var lastErr error
|
||||
for time.Now().Before(deadline) {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second)
|
||||
lastErr = pb.WithFilerClient(false, 0, pb.ServerAddress(filerAddress), grpc.WithTransportCredentials(insecure.NewCredentials()), func(client filer_pb.SeaweedFilerClient) error {
|
||||
_, err := client.LookupDirectoryEntry(ctx, &filer_pb.LookupDirectoryEntryRequest{
|
||||
Directory: peerEntryDir,
|
||||
Name: name,
|
||||
})
|
||||
return err
|
||||
})
|
||||
cancel()
|
||||
if lastErr == nil {
|
||||
return nil
|
||||
}
|
||||
time.Sleep(time.Second)
|
||||
}
|
||||
return fmt.Errorf("%s/%s not on %s within %v: %w", peerEntryDir, name, filerAddress, timeout, lastErr)
|
||||
}
|
||||
|
||||
func waitForClusterNode(master pb.ServerAddress, address string, timeout time.Duration) error {
|
||||
deadline := time.Now().Add(timeout)
|
||||
for time.Now().Before(deadline) {
|
||||
found := false
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 5*time.Second)
|
||||
err := pb.WithMasterClient(ctx, false, master, grpc.WithTransportCredentials(insecure.NewCredentials()), false, func(client master_pb.SeaweedClient) error {
|
||||
resp, err := client.ListClusterNodes(ctx, &master_pb.ListClusterNodesRequest{ClientType: cluster.FilerType})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
for _, node := range resp.ClusterNodes {
|
||||
// the master reports the grpc port too, as "host:port.grpcPort"
|
||||
if pb.ServerAddress(node.Address).Equals(pb.ServerAddress(address)) {
|
||||
found = true
|
||||
}
|
||||
}
|
||||
return nil
|
||||
})
|
||||
cancel()
|
||||
if err == nil && found {
|
||||
return nil
|
||||
}
|
||||
time.Sleep(time.Second)
|
||||
}
|
||||
return fmt.Errorf("%s not registered within %v", address, timeout)
|
||||
}
|
||||
|
||||
func waitForLeader(masterPort int, timeout time.Duration) error {
|
||||
deadline := time.Now().Add(timeout)
|
||||
url := fmt.Sprintf("http://127.0.0.1:%d/cluster/status", masterPort)
|
||||
client := &http.Client{Timeout: 2 * time.Second}
|
||||
for time.Now().Before(deadline) {
|
||||
if resp, err := client.Get(url); err == nil {
|
||||
body, _ := io.ReadAll(resp.Body)
|
||||
resp.Body.Close()
|
||||
if strings.Contains(string(body), `"IsLeader":true`) {
|
||||
return nil
|
||||
}
|
||||
}
|
||||
time.Sleep(time.Second)
|
||||
}
|
||||
return fmt.Errorf("master on %d has no leader within %v", masterPort, timeout)
|
||||
}
|
||||
|
||||
func waitForLog(logFile, message string, timeout time.Duration) error {
|
||||
deadline := time.Now().Add(timeout)
|
||||
for time.Now().Before(deadline) {
|
||||
content, err := os.ReadFile(logFile)
|
||||
if err == nil && strings.Contains(string(content), message) {
|
||||
return nil
|
||||
}
|
||||
time.Sleep(time.Second)
|
||||
}
|
||||
return fmt.Errorf("%q not in %s within %v", message, logFile, timeout)
|
||||
}
|
||||
|
||||
func mkdir(t *testing.T, dir, name string) string {
|
||||
t.Helper()
|
||||
path := filepath.Join(dir, name)
|
||||
require.NoError(t, os.MkdirAll(path, 0755))
|
||||
return path
|
||||
}
|
||||
|
||||
func startFiler(t *testing.T, weedBinary, testDir, name string, port, masterPort int) *exec.Cmd {
|
||||
t.Helper()
|
||||
return startProcess(t, weedBinary, filepath.Join(testDir, name+".log"), "filer",
|
||||
"-ip=127.0.0.1",
|
||||
"-port="+strconv.Itoa(port),
|
||||
"-master=127.0.0.1:"+strconv.Itoa(masterPort),
|
||||
"-defaultStoreDir="+mkdir(t, testDir, name))
|
||||
}
|
||||
|
||||
func startProcess(t *testing.T, weedBinary, logFile string, args ...string) *exec.Cmd {
|
||||
t.Helper()
|
||||
log, err := os.OpenFile(logFile, os.O_CREATE|os.O_WRONLY|os.O_APPEND, 0644)
|
||||
require.NoError(t, err)
|
||||
|
||||
cmd := exec.Command(weedBinary, args...)
|
||||
cmd.Stdout = log
|
||||
cmd.Stderr = log
|
||||
require.NoError(t, cmd.Start())
|
||||
t.Cleanup(func() {
|
||||
cmd.Process.Signal(syscall.SIGCONT)
|
||||
stopProcess(cmd)
|
||||
})
|
||||
return cmd
|
||||
}
|
||||
|
||||
// stopProcess kills the process outright: a filer takes seconds to shut down
|
||||
// gracefully, long enough to register with the master again on the way out.
|
||||
func stopProcess(cmd *exec.Cmd) {
|
||||
if cmd == nil || cmd.Process == nil {
|
||||
return
|
||||
}
|
||||
cmd.Process.Kill()
|
||||
cmd.Wait()
|
||||
}
|
||||
@@ -0,0 +1,103 @@
|
||||
package multi_master
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"google.golang.org/grpc"
|
||||
"google.golang.org/grpc/credentials/insecure"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/cluster"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
|
||||
)
|
||||
|
||||
// A filer only learns about its peers from the cluster node updates on its
|
||||
// KeepConnected stream, and those are broadcast to whoever is connected at that
|
||||
// moment. A filer that reconnects has to be told the membership again, or it
|
||||
// never subscribes to the peers that registered while it was away.
|
||||
func TestKeepConnectedSendsExistingFilers(t *testing.T) {
|
||||
mc := StartMasterCluster(t)
|
||||
|
||||
leaderIdx, leaderAddr := mc.FindLeader()
|
||||
if leaderIdx < 0 {
|
||||
t.Fatal("no leader")
|
||||
}
|
||||
master := pb.ServerAddress(leaderAddr)
|
||||
dialOption := grpc.WithTransportCredentials(insecure.NewCredentials())
|
||||
|
||||
const existingFiler = "127.0.0.1:18888"
|
||||
const joiningFiler = "127.0.0.1:18889"
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), waitTimeout)
|
||||
defer cancel()
|
||||
|
||||
err := pb.WithMasterClient(ctx, true, master, dialOption, false, func(client master_pb.SeaweedClient) error {
|
||||
stream, err := client.KeepConnected(ctx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := stream.Send(&master_pb.KeepConnectedRequest{
|
||||
ClientType: cluster.FilerType,
|
||||
ClientAddress: existingFiler,
|
||||
}); err != nil {
|
||||
return err
|
||||
}
|
||||
if err := waitForClusterNode(ctx, client, existingFiler); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
return pb.WithMasterClient(ctx, true, master, dialOption, false, func(joining master_pb.SeaweedClient) error {
|
||||
joiningCtx, cancelJoining := context.WithTimeout(ctx, waitTimeout)
|
||||
defer cancelJoining()
|
||||
|
||||
joiningStream, err := joining.KeepConnected(joiningCtx)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if err := joiningStream.Send(&master_pb.KeepConnectedRequest{
|
||||
ClientType: cluster.FilerType,
|
||||
ClientAddress: joiningFiler,
|
||||
}); err != nil {
|
||||
return err
|
||||
}
|
||||
for i := 0; ; i++ {
|
||||
resp, err := joiningStream.Recv()
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
// a client only reads the volume locations out of the first
|
||||
// message, an update sent ahead of them would be dropped
|
||||
if i == 0 && resp.VolumeLocation == nil {
|
||||
return fmt.Errorf("first message is not a volume location: %+v", resp)
|
||||
}
|
||||
if update := resp.ClusterNodeUpdate; update != nil && update.IsAdd && update.Address == existingFiler {
|
||||
return nil
|
||||
}
|
||||
}
|
||||
})
|
||||
})
|
||||
if err != nil {
|
||||
mc.DumpLogs()
|
||||
t.Fatalf("a joining filer was not told about %s: %v", existingFiler, err)
|
||||
}
|
||||
}
|
||||
|
||||
func waitForClusterNode(ctx context.Context, client master_pb.SeaweedClient, address string) error {
|
||||
deadline := time.Now().Add(waitTimeout)
|
||||
for time.Now().Before(deadline) {
|
||||
resp, err := client.ListClusterNodes(ctx, &master_pb.ListClusterNodesRequest{ClientType: cluster.FilerType})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
for _, node := range resp.ClusterNodes {
|
||||
if node.Address == address {
|
||||
return nil
|
||||
}
|
||||
}
|
||||
time.Sleep(waitTick)
|
||||
}
|
||||
return context.DeadlineExceeded
|
||||
}
|
||||
@@ -0,0 +1,331 @@
|
||||
package copying_test
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"net/url"
|
||||
"testing"
|
||||
|
||||
"github.com/aws/aws-sdk-go-v2/aws"
|
||||
"github.com/aws/aws-sdk-go-v2/service/s3"
|
||||
"github.com/aws/aws-sdk-go-v2/service/s3/types"
|
||||
smithyhttp "github.com/aws/smithy-go/transport/http"
|
||||
"github.com/stretchr/testify/assert"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// createRenameSource builds the x-amz-rename-source value the way the AWS CLI,
|
||||
// Java and Rust examples do: the bare source key, URL encoded.
|
||||
func createRenameSource(key string) string {
|
||||
return url.PathEscape(key)
|
||||
}
|
||||
|
||||
// createQualifiedRenameSource builds the alternate bucket/key form the second
|
||||
// AWS CLI example and the boto3 conditional example use.
|
||||
func createQualifiedRenameSource(bucketName, key string) string {
|
||||
return fmt.Sprintf("%s/%s", bucketName, url.PathEscape(key))
|
||||
}
|
||||
|
||||
func renameObject(t *testing.T, client *s3.Client, bucketName, srcKey, dstKey string) {
|
||||
t.Helper()
|
||||
_, err := client.RenameObject(context.TODO(), &s3.RenameObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String(dstKey),
|
||||
RenameSource: aws.String(createRenameSource(srcKey)),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
}
|
||||
|
||||
// requireRenameStatus asserts err carries the given HTTP status code.
|
||||
func requireRenameStatus(t *testing.T, err error, status int) {
|
||||
t.Helper()
|
||||
require.Error(t, err)
|
||||
var respErr *smithyhttp.ResponseError
|
||||
require.True(t, errors.As(err, &respErr), "expected an HTTP response error, got %v", err)
|
||||
assert.Equal(t, status, respErr.HTTPStatusCode(), "unexpected error: %v", err)
|
||||
}
|
||||
|
||||
func objectExists(t *testing.T, client *s3.Client, bucketName, key string) bool {
|
||||
t.Helper()
|
||||
_, err := client.HeadObject(context.TODO(), &s3.HeadObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String(key),
|
||||
})
|
||||
return err == nil
|
||||
}
|
||||
|
||||
// TestRenameObject renames an object and checks the bytes, metadata and ETag
|
||||
// arrive under the new key while the old key disappears.
|
||||
func TestRenameObject(t *testing.T) {
|
||||
client := getS3Client(t)
|
||||
bucketName := getNewBucketName()
|
||||
createBucket(t, client, bucketName)
|
||||
defer deleteBucket(t, client, bucketName)
|
||||
|
||||
content := "rename me"
|
||||
put := putObjectWithMetadata(t, client, bucketName, "source.txt", content,
|
||||
map[string]string{"origin": "source"}, "text/plain")
|
||||
|
||||
renameObject(t, client, bucketName, "source.txt", "renamed/target.txt")
|
||||
|
||||
assert.False(t, objectExists(t, client, bucketName, "source.txt"), "source should be gone")
|
||||
|
||||
resp := getObject(t, client, bucketName, "renamed/target.txt")
|
||||
assert.Equal(t, content, getObjectBody(t, resp))
|
||||
assert.Equal(t, "text/plain", aws.ToString(resp.ContentType))
|
||||
assert.Equal(t, "source", resp.Metadata["origin"])
|
||||
assert.Equal(t, aws.ToString(put.ETag), aws.ToString(resp.ETag), "ETag must survive the rename")
|
||||
}
|
||||
|
||||
// TestRenameObjectOverwritesDestination: without a conditional header a rename
|
||||
// replaces whatever the destination key held.
|
||||
func TestRenameObjectOverwritesDestination(t *testing.T) {
|
||||
client := getS3Client(t)
|
||||
bucketName := getNewBucketName()
|
||||
createBucket(t, client, bucketName)
|
||||
defer deleteBucket(t, client, bucketName)
|
||||
|
||||
putObject(t, client, bucketName, "source.txt", "new content")
|
||||
putObject(t, client, bucketName, "target.txt", "old content")
|
||||
|
||||
renameObject(t, client, bucketName, "source.txt", "target.txt")
|
||||
|
||||
resp := getObject(t, client, bucketName, "target.txt")
|
||||
assert.Equal(t, "new content", getObjectBody(t, resp))
|
||||
}
|
||||
|
||||
// TestRenameObjectIfNoneMatch: If-None-Match: * protects an existing destination.
|
||||
func TestRenameObjectIfNoneMatch(t *testing.T) {
|
||||
client := getS3Client(t)
|
||||
bucketName := getNewBucketName()
|
||||
createBucket(t, client, bucketName)
|
||||
defer deleteBucket(t, client, bucketName)
|
||||
|
||||
putObject(t, client, bucketName, "source.txt", "new content")
|
||||
putObject(t, client, bucketName, "target.txt", "old content")
|
||||
|
||||
_, err := client.RenameObject(context.TODO(), &s3.RenameObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String("target.txt"),
|
||||
RenameSource: aws.String(createRenameSource("source.txt")),
|
||||
DestinationIfNoneMatch: aws.String("*"),
|
||||
})
|
||||
requireRenameStatus(t, err, 412)
|
||||
|
||||
resp := getObject(t, client, bucketName, "target.txt")
|
||||
assert.Equal(t, "old content", getObjectBody(t, resp))
|
||||
assert.True(t, objectExists(t, client, bucketName, "source.txt"), "a failed rename must leave the source alone")
|
||||
|
||||
// The same rename onto a free key succeeds.
|
||||
_, err = client.RenameObject(context.TODO(), &s3.RenameObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String("free.txt"),
|
||||
RenameSource: aws.String(createRenameSource("source.txt")),
|
||||
DestinationIfNoneMatch: aws.String("*"),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
}
|
||||
|
||||
// TestRenameObjectSourceIfMatch gates the rename on the source's ETag.
|
||||
func TestRenameObjectSourceIfMatch(t *testing.T) {
|
||||
client := getS3Client(t)
|
||||
bucketName := getNewBucketName()
|
||||
createBucket(t, client, bucketName)
|
||||
defer deleteBucket(t, client, bucketName)
|
||||
|
||||
put := putObject(t, client, bucketName, "source.txt", "content")
|
||||
|
||||
_, err := client.RenameObject(context.TODO(), &s3.RenameObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String("target.txt"),
|
||||
RenameSource: aws.String(createRenameSource("source.txt")),
|
||||
SourceIfMatch: aws.String("\"00000000000000000000000000000000\""),
|
||||
})
|
||||
requireRenameStatus(t, err, 412)
|
||||
assert.True(t, objectExists(t, client, bucketName, "source.txt"))
|
||||
|
||||
_, err = client.RenameObject(context.TODO(), &s3.RenameObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String("target.txt"),
|
||||
RenameSource: aws.String(createRenameSource("source.txt")),
|
||||
SourceIfMatch: put.ETag,
|
||||
})
|
||||
require.NoError(t, err)
|
||||
assert.False(t, objectExists(t, client, bucketName, "source.txt"))
|
||||
assert.True(t, objectExists(t, client, bucketName, "target.txt"))
|
||||
}
|
||||
|
||||
// TestRenameObjectOntoDirectory: a key that already holds other objects is a
|
||||
// directory, and an object must not be allowed to replace one.
|
||||
func TestRenameObjectOntoDirectory(t *testing.T) {
|
||||
client := getS3Client(t)
|
||||
bucketName := getNewBucketName()
|
||||
createBucket(t, client, bucketName)
|
||||
defer deleteBucket(t, client, bucketName)
|
||||
|
||||
putObject(t, client, bucketName, "source.txt", "content")
|
||||
putObject(t, client, bucketName, "target/child.txt", "child")
|
||||
|
||||
_, err := client.RenameObject(context.TODO(), &s3.RenameObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String("target"),
|
||||
RenameSource: aws.String(createRenameSource("source.txt")),
|
||||
})
|
||||
requireRenameStatus(t, err, 409)
|
||||
assert.True(t, objectExists(t, client, bucketName, "source.txt"))
|
||||
assert.True(t, objectExists(t, client, bucketName, "target/child.txt"))
|
||||
}
|
||||
|
||||
// TestRenameObjectDirectorySource: a directory can be named without a trailing
|
||||
// slash, and renaming one would move a whole subtree. It is not an object, so it
|
||||
// is a missing key.
|
||||
func TestRenameObjectDirectorySource(t *testing.T) {
|
||||
client := getS3Client(t)
|
||||
bucketName := getNewBucketName()
|
||||
createBucket(t, client, bucketName)
|
||||
defer deleteBucket(t, client, bucketName)
|
||||
|
||||
putObject(t, client, bucketName, "source/child.txt", "child")
|
||||
|
||||
for _, src := range []string{"source", "source/"} {
|
||||
_, err := client.RenameObject(context.TODO(), &s3.RenameObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String("target.txt"),
|
||||
RenameSource: aws.String(createRenameSource(src)),
|
||||
})
|
||||
requireRenameStatus(t, err, 404)
|
||||
}
|
||||
assert.True(t, objectExists(t, client, bucketName, "source/child.txt"))
|
||||
assert.False(t, objectExists(t, client, bucketName, "target.txt"))
|
||||
}
|
||||
|
||||
// TestRenameObjectMissingSource reports a missing source as NoSuchKey.
|
||||
func TestRenameObjectMissingSource(t *testing.T) {
|
||||
client := getS3Client(t)
|
||||
bucketName := getNewBucketName()
|
||||
createBucket(t, client, bucketName)
|
||||
defer deleteBucket(t, client, bucketName)
|
||||
|
||||
_, err := client.RenameObject(context.TODO(), &s3.RenameObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String("target.txt"),
|
||||
RenameSource: aws.String(createRenameSource("absent.txt")),
|
||||
})
|
||||
requireRenameStatus(t, err, 404)
|
||||
}
|
||||
|
||||
// TestRenameObjectQualifiedSource: the bucket/key form AWS's second CLI example
|
||||
// and the boto3 conditional example use resolves to the same object as the bare
|
||||
// key.
|
||||
func TestRenameObjectQualifiedSource(t *testing.T) {
|
||||
client := getS3Client(t)
|
||||
bucketName := getNewBucketName()
|
||||
createBucket(t, client, bucketName)
|
||||
defer deleteBucket(t, client, bucketName)
|
||||
|
||||
putObject(t, client, bucketName, "dir/source.txt", "content")
|
||||
|
||||
_, err := client.RenameObject(context.TODO(), &s3.RenameObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String("target.txt"),
|
||||
RenameSource: aws.String(createQualifiedRenameSource(bucketName, "dir/source.txt")),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
assert.False(t, objectExists(t, client, bucketName, "dir/source.txt"))
|
||||
assert.Equal(t, "content", getObjectBody(t, getObject(t, client, bucketName, "target.txt")))
|
||||
}
|
||||
|
||||
// TestRenameObjectSourceShadowingTheBucketName: a key whose own first segment is
|
||||
// the bucket name is a real key, and must win over reading the same value as a
|
||||
// bucket-qualified source.
|
||||
func TestRenameObjectSourceShadowingTheBucketName(t *testing.T) {
|
||||
client := getS3Client(t)
|
||||
bucketName := getNewBucketName()
|
||||
createBucket(t, client, bucketName)
|
||||
defer deleteBucket(t, client, bucketName)
|
||||
|
||||
shadowed := bucketName + "/source.txt"
|
||||
putObject(t, client, bucketName, shadowed, "shadowed")
|
||||
putObject(t, client, bucketName, "source.txt", "bare")
|
||||
|
||||
_, err := client.RenameObject(context.TODO(), &s3.RenameObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String("target.txt"),
|
||||
RenameSource: aws.String(createQualifiedRenameSource(bucketName, "source.txt")),
|
||||
})
|
||||
require.NoError(t, err)
|
||||
assert.Equal(t, "shadowed", getObjectBody(t, getObject(t, client, bucketName, "target.txt")))
|
||||
assert.False(t, objectExists(t, client, bucketName, shadowed))
|
||||
assert.True(t, objectExists(t, client, bucketName, "source.txt"), "the bare key must be left alone")
|
||||
}
|
||||
|
||||
// TestRenameObjectSourceShadowedByADirectory: the literal reading of the source
|
||||
// names a directory here, and the bucket-qualified reading names a live object.
|
||||
// Naming a directory is an error about that directory — falling through to the
|
||||
// other reading would rename a different object than the one asked for.
|
||||
func TestRenameObjectSourceShadowedByADirectory(t *testing.T) {
|
||||
client := getS3Client(t)
|
||||
bucketName := getNewBucketName()
|
||||
createBucket(t, client, bucketName)
|
||||
defer deleteBucket(t, client, bucketName)
|
||||
|
||||
putObject(t, client, bucketName, bucketName+"/source.txt/child.txt", "child")
|
||||
putObject(t, client, bucketName, "source.txt", "bare")
|
||||
|
||||
_, err := client.RenameObject(context.TODO(), &s3.RenameObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String("target.txt"),
|
||||
RenameSource: aws.String(createQualifiedRenameSource(bucketName, "source.txt")),
|
||||
})
|
||||
requireRenameStatus(t, err, 404)
|
||||
assert.True(t, objectExists(t, client, bucketName, "source.txt"), "the bare key must be left alone")
|
||||
assert.False(t, objectExists(t, client, bucketName, "target.txt"))
|
||||
}
|
||||
|
||||
// TestRenameObjectCrossBucket: RenameObject moves within one bucket, so another
|
||||
// bucket's name in the source is just part of a key this bucket does not hold.
|
||||
func TestRenameObjectCrossBucket(t *testing.T) {
|
||||
client := getS3Client(t)
|
||||
srcBucket := getNewBucketName()
|
||||
createBucket(t, client, srcBucket)
|
||||
defer deleteBucket(t, client, srcBucket)
|
||||
dstBucket := getNewBucketName()
|
||||
createBucket(t, client, dstBucket)
|
||||
defer deleteBucket(t, client, dstBucket)
|
||||
|
||||
putObject(t, client, srcBucket, "source.txt", "content")
|
||||
|
||||
_, err := client.RenameObject(context.TODO(), &s3.RenameObjectInput{
|
||||
Bucket: aws.String(dstBucket),
|
||||
Key: aws.String("target.txt"),
|
||||
RenameSource: aws.String(createQualifiedRenameSource(srcBucket, "source.txt")),
|
||||
})
|
||||
requireRenameStatus(t, err, 404)
|
||||
assert.True(t, objectExists(t, client, srcBucket, "source.txt"))
|
||||
}
|
||||
|
||||
// TestRenameObjectVersionedBucket: versioned buckets are not supported yet, and
|
||||
// must say so rather than silently dropping versions.
|
||||
func TestRenameObjectVersionedBucket(t *testing.T) {
|
||||
client := getS3Client(t)
|
||||
bucketName := getNewBucketName()
|
||||
createBucket(t, client, bucketName)
|
||||
defer deleteBucket(t, client, bucketName)
|
||||
|
||||
_, err := client.PutBucketVersioning(context.TODO(), &s3.PutBucketVersioningInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
VersioningConfiguration: &types.VersioningConfiguration{Status: types.BucketVersioningStatusEnabled},
|
||||
})
|
||||
require.NoError(t, err)
|
||||
|
||||
putObject(t, client, bucketName, "source.txt", "content")
|
||||
|
||||
_, err = client.RenameObject(context.TODO(), &s3.RenameObjectInput{
|
||||
Bucket: aws.String(bucketName),
|
||||
Key: aws.String("target.txt"),
|
||||
RenameSource: aws.String(createRenameSource("source.txt")),
|
||||
})
|
||||
requireRenameStatus(t, err, 501)
|
||||
assert.True(t, objectExists(t, client, bucketName, "source.txt"))
|
||||
}
|
||||
@@ -0,0 +1,12 @@
|
||||
# Iceberg writer used by the ClickHouse integration test to populate a table
|
||||
# with data files via PyIceberg, so the downstream ClickHouse SELECT verifies
|
||||
# the read path against non-empty results rather than just count() = 0.
|
||||
FROM python:3.11-slim
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
RUN pip install --no-cache-dir "pyiceberg[s3fs]==0.11.1" "pyarrow==25.0.0"
|
||||
|
||||
COPY append_rows.py read_rows.py /app/
|
||||
|
||||
ENTRYPOINT ["python3", "/app/append_rows.py"]
|
||||
@@ -0,0 +1,82 @@
|
||||
# ClickHouse Iceberg Catalog Integration Test
|
||||
|
||||
This directory contains a ClickHouse integration smoke test for SeaweedFS's
|
||||
Iceberg REST Catalog implementation, using ClickHouse's `DataLakeCatalog`
|
||||
database engine.
|
||||
|
||||
## What It Tests
|
||||
|
||||
`TestClickHouseIcebergCatalog` verifies the ClickHouse path end to end:
|
||||
|
||||
1. Starts a local SeaweedFS mini cluster with S3 Tables and Iceberg REST enabled.
|
||||
2. Creates a SeaweedFS table bucket.
|
||||
3. Creates an Iceberg namespace and an empty table through the SeaweedFS REST
|
||||
catalog OAuth flow.
|
||||
4. Creates a second table and populates it with three rows by running a
|
||||
PyIceberg writer container (`Dockerfile.writer` + `append_rows.py`) before
|
||||
ClickHouse connects, so the snapshot is part of the catalog's first scan.
|
||||
5. Starts `clickhouse/clickhouse-server:25.8` and waits for the HTTP interface.
|
||||
6. Attaches the catalog with `CREATE DATABASE ... ENGINE = DataLakeCatalog`
|
||||
(`catalog_type = 'rest'`), authenticating to the catalog via the OAuth2
|
||||
client-credentials flow (`catalog_credential` + `oauth_server_uri`) and to
|
||||
S3 via the engine's access/secret key arguments and `storage_endpoint`.
|
||||
7. Runs subtests against the SeaweedFS-backed Iceberg tables:
|
||||
- `BasicSelect`: ClickHouse is alive and answering SQL.
|
||||
- `DatabaseVisible`: the DataLakeCatalog database exists.
|
||||
- `TableVisible`: seeded tables appear as `namespace.table` entries in
|
||||
`SHOW TABLES` (ClickHouse flattens Iceberg namespaces into table names).
|
||||
- `DescribeTable`: the Iceberg schema mapped to `id Int64` and
|
||||
`label Nullable(String)`. Failure here means ClickHouse could not parse
|
||||
the schema returned by the SeaweedFS catalog.
|
||||
- `CountEmptyTable`: catalog-to-table resolution and a scan of an empty table.
|
||||
- `ReadWrittenDataCount` and `ReadWrittenDataValues`: ClickHouse reads back
|
||||
the three PyIceberg-appended rows and the values match. This exercises the
|
||||
actual data path (parquet reads via S3), not just metadata.
|
||||
- `WriteReadBack`: ClickHouse inserts rows with its experimental Iceberg
|
||||
write support using default settings, which produces manifests without
|
||||
avro field-ids, bucket-relative paths, and parquet without field ids. The
|
||||
SeaweedFS catalog repairs the manifests at commit time and stamps a
|
||||
default name mapping on the table, so PyIceberg (`read_rows.py`, a strict
|
||||
reader) must return the rows ClickHouse wrote.
|
||||
|
||||
Queries go through ClickHouse's HTTP interface (port 8123, mapped to a
|
||||
dynamically allocated host port), so the test needs no ClickHouse client
|
||||
driver. Tables are referenced as ``iceberg_catalog.`namespace.table` ``.
|
||||
|
||||
## Running Locally
|
||||
|
||||
Build or install `weed`, then run:
|
||||
|
||||
```bash
|
||||
cd test/s3tables/catalog_clickhouse
|
||||
go test -v -timeout 20m .
|
||||
```
|
||||
|
||||
The test requires Docker. The GitHub Actions job runs on `ubuntu-22.04` and
|
||||
executes the test for pull requests.
|
||||
|
||||
## Configuration
|
||||
|
||||
The test uses these fixed credentials for the local SeaweedFS IAM config:
|
||||
|
||||
- S3 access key: `AKIAIOSFODNN7EXAMPLE`
|
||||
- S3 secret key: `wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY`
|
||||
- Region: `us-west-2`
|
||||
- Warehouse bucket: `iceberg-tables`
|
||||
|
||||
ClickHouse ports:
|
||||
|
||||
- Only `8123` (HTTP interface) is mapped to a host port (allocated
|
||||
dynamically) so the Go test can issue queries from the test process.
|
||||
- The Iceberg REST endpoint and the S3 endpoint are reached from inside the
|
||||
ClickHouse container via `host.docker.internal`, matching the Doris, Trino,
|
||||
and Dremio test paths.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- Ensure Docker is running: `docker version`
|
||||
- Ensure `weed` is built or available on `PATH`
|
||||
- `DataLakeCatalog` requires `allow_experimental_database_iceberg = 1`; the
|
||||
test passes it as a URL setting on the CREATE DATABASE request.
|
||||
- Container logs are printed in the failure message; you can also check
|
||||
`docker logs <seaweed-clickhouse-...>` while the test is running.
|
||||
@@ -0,0 +1,88 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Append rows to an existing Iceberg table via the SeaweedFS REST catalog.
|
||||
|
||||
Used by the ClickHouse integration test to materialize data files so a
|
||||
downstream SELECT from ClickHouse verifies the read path against non-empty
|
||||
results.
|
||||
|
||||
Usage:
|
||||
python3 append_rows.py \\
|
||||
--catalog-url http://localhost:8181 \\
|
||||
--warehouse s3://my-bucket \\
|
||||
--prefix my-bucket \\
|
||||
--s3-endpoint http://localhost:8333 \\
|
||||
--access-key AKIA... --secret-key wJalr... \\
|
||||
--namespace foo --namespace bar \\
|
||||
--table events
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import sys
|
||||
|
||||
import pyarrow as pa
|
||||
from pyiceberg.catalog import load_catalog
|
||||
|
||||
|
||||
def main() -> int:
|
||||
p = argparse.ArgumentParser()
|
||||
p.add_argument("--catalog-url", required=True)
|
||||
p.add_argument("--warehouse", required=True, help="s3://<bucket-name>")
|
||||
p.add_argument("--prefix", required=True, help="REST catalog prefix (table bucket name)")
|
||||
p.add_argument("--s3-endpoint", required=True, help="http://host:port")
|
||||
p.add_argument("--access-key", required=True)
|
||||
p.add_argument("--secret-key", required=True)
|
||||
p.add_argument("--region", default="us-east-1")
|
||||
p.add_argument(
|
||||
"--namespace",
|
||||
action="append",
|
||||
required=True,
|
||||
help="One per level (e.g. --namespace foo --namespace bar for foo.bar).",
|
||||
)
|
||||
p.add_argument("--table", required=True)
|
||||
args = p.parse_args()
|
||||
|
||||
# `credential` triggers OAuth2 client_credentials against
|
||||
# <catalog_uri>/v1/oauth/tokens, matching the helper Go test uses for
|
||||
# REST-API table creation. The s3.* keys are needed for parquet writes.
|
||||
catalog = load_catalog(
|
||||
"rest",
|
||||
**{
|
||||
"type": "rest",
|
||||
"uri": args.catalog_url,
|
||||
"warehouse": args.warehouse,
|
||||
"prefix": args.prefix,
|
||||
"credential": f"{args.access_key}:{args.secret_key}",
|
||||
"s3.access-key-id": args.access_key,
|
||||
"s3.secret-access-key": args.secret_key,
|
||||
"s3.endpoint": args.s3_endpoint,
|
||||
"s3.region": args.region,
|
||||
"s3.path-style-access": "true",
|
||||
},
|
||||
)
|
||||
|
||||
table_id = tuple(args.namespace) + (args.table,)
|
||||
table = catalog.load_table(table_id)
|
||||
|
||||
# Match the Iceberg table schema: id is `required long`, label is
|
||||
# `optional string`. Default pyarrow columns are nullable, which fails
|
||||
# PyIceberg's required-field compatibility check.
|
||||
arrow_schema = pa.schema(
|
||||
[
|
||||
pa.field("id", pa.int64(), nullable=False),
|
||||
pa.field("label", pa.string(), nullable=True),
|
||||
]
|
||||
)
|
||||
arrow_table = pa.Table.from_pydict(
|
||||
{
|
||||
"id": [1, 2, 3],
|
||||
"label": ["one", "two", "three"],
|
||||
},
|
||||
schema=arrow_schema,
|
||||
)
|
||||
table.append(arrow_table)
|
||||
print(f"appended {arrow_table.num_rows} rows to {'.'.join(table_id)}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,780 @@
|
||||
// Package catalog_clickhouse provides integration tests for ClickHouse with
|
||||
// the SeaweedFS Iceberg REST Catalog.
|
||||
package catalog_clickhouse
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"crypto/rand"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"net/url"
|
||||
"os"
|
||||
"os/exec"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/test/testutil"
|
||||
)
|
||||
|
||||
const (
|
||||
clickhouseImage = "clickhouse/clickhouse-server:25.8"
|
||||
clickhouseDatabase = "iceberg_catalog"
|
||||
clickhouseHTTPPort = 8123
|
||||
clickhouseStartTimeout = 2 * time.Minute
|
||||
|
||||
// The image's passwordless `default` user only accepts connections from
|
||||
// localhost, and our queries arrive via the mapped port from the Docker
|
||||
// gateway. The entrypoint grants a CLICKHOUSE_USER/CLICKHOUSE_PASSWORD
|
||||
// user access from any host.
|
||||
clickhouseUser = "seaweed"
|
||||
clickhousePassword = "seaweedtest"
|
||||
)
|
||||
|
||||
type TestEnvironment struct {
|
||||
seaweedDir string
|
||||
weedBinary string
|
||||
dataDir string
|
||||
bindIP string
|
||||
s3Port int
|
||||
s3GrpcPort int
|
||||
icebergPort int
|
||||
masterPort int
|
||||
masterGrpcPort int
|
||||
filerPort int
|
||||
filerGrpcPort int
|
||||
volumePort int
|
||||
volumeGrpcPort int
|
||||
weedProcess *exec.Cmd
|
||||
weedCancel context.CancelFunc
|
||||
clickhouseContainer string
|
||||
clickhouseHostPort int
|
||||
accessKey string
|
||||
secretKey string
|
||||
}
|
||||
|
||||
// TestClickHouseIcebergCatalog brings up SeaweedFS + ClickHouse and validates
|
||||
// that ClickHouse's DataLakeCatalog database engine can discover catalog
|
||||
// metadata served by SeaweedFS's Iceberg REST API and read both empty and
|
||||
// populated tables through the standard S3 data path.
|
||||
//
|
||||
// Subtests:
|
||||
// - BasicSelect: ClickHouse is alive and answering SQL.
|
||||
// - DatabaseVisible: the DataLakeCatalog database exists.
|
||||
// - TableVisible: seeded tables appear as `namespace.table` entries.
|
||||
// - DescribeTable: ClickHouse mapped the Iceberg schema (long -> Int64,
|
||||
// optional string -> Nullable(String)).
|
||||
// - CountEmptyTable: ClickHouse resolves the table and scans an empty
|
||||
// Iceberg snapshot.
|
||||
// - ReadWrittenDataCount / ReadWrittenDataValues: a separate table is
|
||||
// populated by a PyIceberg writer container before ClickHouse connects;
|
||||
// ClickHouse then reads the three rows back, exercising the actual data
|
||||
// path (parquet over S3), not just metadata.
|
||||
func TestClickHouseIcebergCatalog(t *testing.T) {
|
||||
requireClickHouseRuntime(t)
|
||||
|
||||
env := NewTestEnvironment(t)
|
||||
defer env.Cleanup(t)
|
||||
|
||||
fmt.Printf(">>> Starting SeaweedFS...\n")
|
||||
env.StartSeaweedFS(t)
|
||||
fmt.Printf(">>> SeaweedFS started.\n")
|
||||
|
||||
tableBucket := "iceberg-tables"
|
||||
fmt.Printf(">>> Creating table bucket: %s\n", tableBucket)
|
||||
createTableBucket(t, env, tableBucket)
|
||||
fmt.Printf(">>> Table bucket created.\n")
|
||||
|
||||
testIcebergRestAPI(t, env)
|
||||
|
||||
namespace := "clickhouse_" + randomString(6)
|
||||
tableName := "smoke_" + randomString(6)
|
||||
icebergToken := requestIcebergOAuthToken(t, env)
|
||||
createIcebergNamespace(t, env, icebergToken, tableBucket, namespace)
|
||||
createIcebergTable(t, env, icebergToken, tableBucket, namespace, tableName)
|
||||
|
||||
// Seed a populated table by creating an empty one through the REST API
|
||||
// and then appending three rows via PyIceberg, so the snapshot exists
|
||||
// before ClickHouse's first scan.
|
||||
populatedTable := "populated_" + randomString(6)
|
||||
createIcebergTable(t, env, icebergToken, tableBucket, namespace, populatedTable)
|
||||
buildClickHouseWriterImage(t)
|
||||
writeIcebergRows(t, env, tableBucket, []string{namespace}, populatedTable)
|
||||
|
||||
// Empty table that ClickHouse writes into during the WriteReadBack subtest.
|
||||
writeTable := "chwrite_" + randomString(6)
|
||||
createIcebergTable(t, env, icebergToken, tableBucket, namespace, writeTable)
|
||||
|
||||
env.startClickHouseContainer(t)
|
||||
env.waitForClickHouse(t, clickhouseStartTimeout)
|
||||
|
||||
env.createClickHouseCatalogDatabase(t, tableBucket)
|
||||
|
||||
t.Run("BasicSelect", func(t *testing.T) {
|
||||
out := env.mustQuery(t, "SELECT 1")
|
||||
if out != "1" {
|
||||
t.Fatalf("SELECT 1 = %q, want 1", out)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("DatabaseVisible", func(t *testing.T) {
|
||||
out := env.mustQuery(t, "SHOW DATABASES")
|
||||
if !containsLine(out, clickhouseDatabase) {
|
||||
t.Fatalf("SHOW DATABASES did not list %s:\n%s", clickhouseDatabase, out)
|
||||
}
|
||||
})
|
||||
|
||||
// DataLakeCatalog flattens the Iceberg namespace hierarchy into table
|
||||
// names of the form "namespace.table", queried with backtick quoting.
|
||||
emptyRef := fmt.Sprintf("%s.`%s.%s`", clickhouseDatabase, namespace, tableName)
|
||||
populatedRef := fmt.Sprintf("%s.`%s.%s`", clickhouseDatabase, namespace, populatedTable)
|
||||
|
||||
t.Run("TableVisible", func(t *testing.T) {
|
||||
out := env.mustQuery(t, fmt.Sprintf("SHOW TABLES FROM %s", clickhouseDatabase))
|
||||
for _, want := range []string{namespace + "." + tableName, namespace + "." + populatedTable} {
|
||||
if !containsLine(out, want) {
|
||||
t.Fatalf("SHOW TABLES FROM %s did not list %s:\n%s", clickhouseDatabase, want, out)
|
||||
}
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("DescribeTable", func(t *testing.T) {
|
||||
out := env.mustQuery(t, fmt.Sprintf("DESCRIBE TABLE %s", emptyRef))
|
||||
if !strings.Contains(out, "id\tInt64") {
|
||||
t.Fatalf("DESCRIBE %s missing id Int64:\n%s", emptyRef, out)
|
||||
}
|
||||
if !strings.Contains(out, "label\tNullable(String)") {
|
||||
t.Fatalf("DESCRIBE %s missing label Nullable(String):\n%s", emptyRef, out)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("CountEmptyTable", func(t *testing.T) {
|
||||
out := env.mustQuery(t, fmt.Sprintf("SELECT count() FROM %s", emptyRef))
|
||||
if out != "0" {
|
||||
t.Fatalf("count(%s) = %q, want 0", emptyRef, out)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("ReadWrittenDataCount", func(t *testing.T) {
|
||||
out := env.mustQuery(t, fmt.Sprintf("SELECT count() FROM %s", populatedRef))
|
||||
if out != "3" {
|
||||
t.Fatalf("count(%s) = %q, want 3", populatedRef, out)
|
||||
}
|
||||
})
|
||||
|
||||
t.Run("ReadWrittenDataValues", func(t *testing.T) {
|
||||
out := env.mustQuery(t, fmt.Sprintf("SELECT id, label FROM %s ORDER BY id", populatedRef))
|
||||
want := "1\tone\n2\ttwo\n3\tthree"
|
||||
if out != want {
|
||||
t.Fatalf("SELECT id, label FROM %s = %q, want %q", populatedRef, out, want)
|
||||
}
|
||||
})
|
||||
|
||||
// ClickHouse's experimental Iceberg writes produce manifests without avro
|
||||
// field-ids, bucket-relative paths, and parquet without field ids. The
|
||||
// catalog repairs the manifests at commit and stamps a name mapping on the
|
||||
// table, so a strict reader (PyIceberg) must see ClickHouse's rows.
|
||||
t.Run("WriteReadBack", func(t *testing.T) {
|
||||
writeRef := fmt.Sprintf("%s.`%s.%s`", clickhouseDatabase, namespace, writeTable)
|
||||
insert := fmt.Sprintf("INSERT INTO %s (id, label) VALUES (1, 'alpha'), (2, 'beta')", writeRef)
|
||||
if _, err := env.query(insert, map[string]string{"allow_experimental_insert_into_iceberg": "1"}); err != nil {
|
||||
t.Fatalf("%s: %v\nContainer logs:\n%s", insert, err, clickhouseContainerLogs(env.clickhouseContainer))
|
||||
}
|
||||
|
||||
out := env.mustQuery(t, fmt.Sprintf("SELECT id, label FROM %s ORDER BY id", writeRef))
|
||||
if want := "1\talpha\n2\tbeta"; out != want {
|
||||
t.Fatalf("ClickHouse read-back = %q, want %q", out, want)
|
||||
}
|
||||
|
||||
rows := readIcebergRows(t, env, tableBucket, []string{namespace}, writeTable)
|
||||
if want := "1,alpha\n2,beta"; rows != want {
|
||||
t.Fatalf("PyIceberg read of ClickHouse-written table = %q, want %q", rows, want)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// NewTestEnvironment allocates ports and returns an environment for the test.
|
||||
func NewTestEnvironment(t *testing.T) *TestEnvironment {
|
||||
t.Helper()
|
||||
|
||||
wd, err := os.Getwd()
|
||||
if err != nil {
|
||||
t.Fatalf("Failed to get working directory: %v", err)
|
||||
}
|
||||
|
||||
seaweedDir := wd
|
||||
for i := 0; i < 6; i++ {
|
||||
if _, err := os.Stat(filepath.Join(seaweedDir, "go.mod")); err == nil {
|
||||
break
|
||||
}
|
||||
seaweedDir = filepath.Dir(seaweedDir)
|
||||
}
|
||||
|
||||
weedBinary := filepath.Join(seaweedDir, "weed", "weed")
|
||||
info, err := os.Stat(weedBinary)
|
||||
if err != nil || info.IsDir() {
|
||||
weedBinary = filepath.Join(seaweedDir, "weed", "weed", "weed")
|
||||
info, err = os.Stat(weedBinary)
|
||||
if err != nil || info.IsDir() {
|
||||
weedBinary = "weed"
|
||||
if _, err := exec.LookPath(weedBinary); err != nil {
|
||||
t.Skip("weed binary not found, skipping integration test")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
dataDir, err := os.MkdirTemp("", "seaweed-clickhouse-test-*")
|
||||
if err != nil {
|
||||
t.Fatalf("Failed to create temp dir: %v", err)
|
||||
}
|
||||
|
||||
bindIP := testutil.FindBindIP()
|
||||
// 9 ports for the seaweed mini cluster, plus one for the ClickHouse HTTP
|
||||
// interface mapped on the host.
|
||||
ports := testutil.MustAllocatePorts(t, 10)
|
||||
|
||||
env := &TestEnvironment{
|
||||
seaweedDir: seaweedDir,
|
||||
weedBinary: weedBinary,
|
||||
dataDir: dataDir,
|
||||
bindIP: bindIP,
|
||||
masterPort: ports[0],
|
||||
masterGrpcPort: ports[1],
|
||||
volumePort: ports[2],
|
||||
volumeGrpcPort: ports[3],
|
||||
filerPort: ports[4],
|
||||
filerGrpcPort: ports[5],
|
||||
s3Port: ports[6],
|
||||
s3GrpcPort: ports[7],
|
||||
icebergPort: ports[8],
|
||||
clickhouseHostPort: ports[9],
|
||||
}
|
||||
|
||||
env.accessKey = "AKIAIOSFODNN7EXAMPLE"
|
||||
env.secretKey = "wJalrXUtnFEMI/K7MDENG/bPxRfiCYEXAMPLEKEY"
|
||||
|
||||
return env
|
||||
}
|
||||
|
||||
// StartSeaweedFS starts a SeaweedFS mini instance with the Iceberg REST API.
|
||||
func (env *TestEnvironment) StartSeaweedFS(t *testing.T) {
|
||||
t.Helper()
|
||||
|
||||
iamConfigPath, err := testutil.WriteIAMConfig(env.dataDir, env.accessKey, env.secretKey)
|
||||
if err != nil {
|
||||
t.Fatalf("Failed to create IAM config: %v", err)
|
||||
}
|
||||
|
||||
securityToml := filepath.Join(env.dataDir, "security.toml")
|
||||
if err := os.WriteFile(securityToml, []byte("# Empty security config for testing\n"), 0644); err != nil {
|
||||
t.Fatalf("Failed to create security.toml: %v", err)
|
||||
}
|
||||
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
env.weedCancel = cancel
|
||||
|
||||
cmd := exec.CommandContext(ctx, env.weedBinary, "mini",
|
||||
"-master.port", fmt.Sprintf("%d", env.masterPort),
|
||||
"-master.port.grpc", fmt.Sprintf("%d", env.masterGrpcPort),
|
||||
"-volume.port", fmt.Sprintf("%d", env.volumePort),
|
||||
"-volume.port.grpc", fmt.Sprintf("%d", env.volumeGrpcPort),
|
||||
"-filer.port", fmt.Sprintf("%d", env.filerPort),
|
||||
"-filer.port.grpc", fmt.Sprintf("%d", env.filerGrpcPort),
|
||||
"-s3.port", fmt.Sprintf("%d", env.s3Port),
|
||||
"-s3.port.grpc", fmt.Sprintf("%d", env.s3GrpcPort),
|
||||
"-s3.port.iceberg", fmt.Sprintf("%d", env.icebergPort),
|
||||
"-s3.config", iamConfigPath,
|
||||
"-ip", env.bindIP,
|
||||
"-ip.bind", "0.0.0.0",
|
||||
"-dir", env.dataDir,
|
||||
)
|
||||
cmd.Dir = env.dataDir
|
||||
cmd.Stdout = os.Stdout
|
||||
cmd.Stderr = os.Stderr
|
||||
|
||||
cmd.Env = append(os.Environ(),
|
||||
"AWS_ACCESS_KEY_ID="+env.accessKey,
|
||||
"AWS_SECRET_ACCESS_KEY="+env.secretKey,
|
||||
"ICEBERG_WAREHOUSE=s3://iceberg-tables",
|
||||
"S3TABLES_DEFAULT_BUCKET=iceberg-tables",
|
||||
)
|
||||
|
||||
if err := cmd.Start(); err != nil {
|
||||
t.Fatalf("Failed to start SeaweedFS: %v", err)
|
||||
}
|
||||
env.weedProcess = cmd
|
||||
|
||||
icebergURL := fmt.Sprintf("http://%s:%d/v1/config", env.bindIP, env.icebergPort)
|
||||
if !env.waitForService(icebergURL, 30*time.Second) {
|
||||
t.Fatalf("Iceberg REST API did not become ready at %s", icebergURL)
|
||||
}
|
||||
}
|
||||
|
||||
// Cleanup stops ClickHouse, SeaweedFS, and removes temporary state.
|
||||
func (env *TestEnvironment) Cleanup(t *testing.T) {
|
||||
t.Helper()
|
||||
|
||||
if env.clickhouseContainer != "" {
|
||||
_ = exec.Command("docker", "rm", "-f", env.clickhouseContainer).Run()
|
||||
}
|
||||
|
||||
if env.weedCancel != nil {
|
||||
env.weedCancel()
|
||||
}
|
||||
|
||||
if env.weedProcess != nil {
|
||||
time.Sleep(2 * time.Second)
|
||||
_ = env.weedProcess.Wait()
|
||||
}
|
||||
|
||||
if env.dataDir != "" {
|
||||
_ = os.RemoveAll(env.dataDir)
|
||||
}
|
||||
}
|
||||
|
||||
// waitForService polls a URL until it returns a 2xx/401/403 status or timeout.
|
||||
func (env *TestEnvironment) waitForService(url string, timeout time.Duration) bool {
|
||||
client := &http.Client{Timeout: 2 * time.Second}
|
||||
deadline := time.Now().Add(timeout)
|
||||
for time.Now().Before(deadline) {
|
||||
resp, err := client.Get(url)
|
||||
if err != nil {
|
||||
time.Sleep(500 * time.Millisecond)
|
||||
continue
|
||||
}
|
||||
statusCode := resp.StatusCode
|
||||
resp.Body.Close()
|
||||
if statusCode >= 200 && statusCode < 300 {
|
||||
return true
|
||||
}
|
||||
if statusCode == http.StatusUnauthorized || statusCode == http.StatusForbidden {
|
||||
return true
|
||||
}
|
||||
time.Sleep(500 * time.Millisecond)
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// testIcebergRestAPI verifies the Iceberg REST endpoint is reachable.
|
||||
func testIcebergRestAPI(t *testing.T, env *TestEnvironment) {
|
||||
t.Helper()
|
||||
|
||||
url := fmt.Sprintf("http://%s:%d/v1/config", env.bindIP, env.icebergPort)
|
||||
client := &http.Client{Timeout: 10 * time.Second}
|
||||
resp, err := client.Get(url)
|
||||
if err != nil {
|
||||
t.Fatalf("Failed to connect to Iceberg REST API at %s: %v", url, err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
body, _ := io.ReadAll(resp.Body)
|
||||
t.Fatalf("Expected 200 OK from /v1/config, got %d, body: %s", resp.StatusCode, body)
|
||||
}
|
||||
}
|
||||
|
||||
// startClickHouseContainer launches the ClickHouse server image and exposes
|
||||
// only the HTTP interface to the host. The Iceberg REST and S3 endpoints are
|
||||
// reached via host.docker.internal, matching the Doris/Trino paths.
|
||||
func (env *TestEnvironment) startClickHouseContainer(t *testing.T) {
|
||||
t.Helper()
|
||||
|
||||
containerName := "seaweed-clickhouse-" + randomString(8)
|
||||
env.clickhouseContainer = containerName
|
||||
|
||||
cmd := exec.Command("docker", "run", "-d",
|
||||
"--name", containerName,
|
||||
"--add-host", "host.docker.internal:host-gateway",
|
||||
"-p", fmt.Sprintf("%d:%d", env.clickhouseHostPort, clickhouseHTTPPort),
|
||||
"--ulimit", "nofile=262144:262144",
|
||||
"-e", "CLICKHOUSE_USER="+clickhouseUser,
|
||||
"-e", "CLICKHOUSE_PASSWORD="+clickhousePassword,
|
||||
clickhouseImage,
|
||||
)
|
||||
if output, err := cmd.CombinedOutput(); err != nil {
|
||||
t.Fatalf("Failed to start ClickHouse container: %v\n%s", err, string(output))
|
||||
}
|
||||
}
|
||||
|
||||
// clickhouseContainerLogs returns the tail of the container logs for diagnostics.
|
||||
func clickhouseContainerLogs(containerName string) string {
|
||||
cmd := exec.Command("docker", "logs", "--tail", "200", containerName)
|
||||
output, err := cmd.CombinedOutput()
|
||||
if err != nil {
|
||||
return fmt.Sprintf("(failed to fetch docker logs: %v)\n%s", err, string(output))
|
||||
}
|
||||
return string(output)
|
||||
}
|
||||
|
||||
// containerRunning returns true if the named container is in `running` state.
|
||||
func containerRunning(containerName string) bool {
|
||||
cmd := exec.Command("docker", "inspect", "--format", "{{.State.Running}}", containerName)
|
||||
out, err := cmd.Output()
|
||||
if err != nil {
|
||||
return false
|
||||
}
|
||||
return strings.TrimSpace(string(out)) == "true"
|
||||
}
|
||||
|
||||
// waitForClickHouse polls the HTTP /ping endpoint until it answers.
|
||||
func (env *TestEnvironment) waitForClickHouse(t *testing.T, timeout time.Duration) {
|
||||
t.Helper()
|
||||
|
||||
pingURL := fmt.Sprintf("http://127.0.0.1:%d/ping", env.clickhouseHostPort)
|
||||
client := &http.Client{Timeout: 2 * time.Second}
|
||||
deadline := time.Now().Add(timeout)
|
||||
for time.Now().Before(deadline) {
|
||||
if !containerRunning(env.clickhouseContainer) {
|
||||
t.Fatalf("ClickHouse container exited before becoming ready\nContainer logs:\n%s",
|
||||
clickhouseContainerLogs(env.clickhouseContainer))
|
||||
}
|
||||
resp, err := client.Get(pingURL)
|
||||
if err == nil {
|
||||
statusCode := resp.StatusCode
|
||||
resp.Body.Close()
|
||||
if statusCode == http.StatusOK {
|
||||
return
|
||||
}
|
||||
}
|
||||
time.Sleep(time.Second)
|
||||
}
|
||||
t.Fatalf("Timed out waiting for ClickHouse to be ready\nContainer logs:\n%s",
|
||||
clickhouseContainerLogs(env.clickhouseContainer))
|
||||
}
|
||||
|
||||
// query sends one SQL statement to ClickHouse's HTTP interface as the default
|
||||
// user and returns the TabSeparated response body with trailing whitespace
|
||||
// trimmed. Settings ride along as URL parameters because SET does not persist
|
||||
// across stateless HTTP requests.
|
||||
func (env *TestEnvironment) query(sqlText string, settings map[string]string) (string, error) {
|
||||
params := url.Values{}
|
||||
params.Set("user", clickhouseUser)
|
||||
params.Set("password", clickhousePassword)
|
||||
params.Set("default_format", "TabSeparated")
|
||||
for k, v := range settings {
|
||||
params.Set(k, v)
|
||||
}
|
||||
queryURL := fmt.Sprintf("http://127.0.0.1:%d/?%s", env.clickhouseHostPort, params.Encode())
|
||||
|
||||
client := &http.Client{Timeout: 120 * time.Second}
|
||||
resp, err := client.Post(queryURL, "text/plain", strings.NewReader(sqlText))
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("POST query: %v", err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
|
||||
body, _ := io.ReadAll(resp.Body)
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
return "", fmt.Errorf("query returned status %d: %s", resp.StatusCode, strings.TrimSpace(string(body)))
|
||||
}
|
||||
return strings.TrimRight(string(body), "\n"), nil
|
||||
}
|
||||
|
||||
// mustQuery runs a query and fails the test with container logs on error.
|
||||
func (env *TestEnvironment) mustQuery(t *testing.T, sqlText string) string {
|
||||
t.Helper()
|
||||
|
||||
out, err := env.query(sqlText, nil)
|
||||
if err != nil {
|
||||
t.Fatalf("%s: %v\nContainer logs:\n%s", sqlText, err, clickhouseContainerLogs(env.clickhouseContainer))
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// createClickHouseCatalogDatabase attaches the SeaweedFS Iceberg REST catalog
|
||||
// as a DataLakeCatalog database. The engine arguments carry the S3 storage
|
||||
// credentials; catalog authentication uses the OAuth2 client_credentials flow
|
||||
// against the SeaweedFS token endpoint, matching the PyIceberg writer.
|
||||
func (env *TestEnvironment) createClickHouseCatalogDatabase(t *testing.T, warehouseBucket string) {
|
||||
t.Helper()
|
||||
|
||||
icebergURI := fmt.Sprintf("http://host.docker.internal:%d/v1", env.icebergPort)
|
||||
storageEndpoint := fmt.Sprintf("http://host.docker.internal:%d/%s", env.s3Port, warehouseBucket)
|
||||
oauthURI := icebergURI + "/oauth/tokens"
|
||||
|
||||
createSQL := fmt.Sprintf(`CREATE DATABASE %s
|
||||
ENGINE = DataLakeCatalog('%s', '%s', '%s')
|
||||
SETTINGS catalog_type = 'rest',
|
||||
warehouse = 's3://%s',
|
||||
storage_endpoint = '%s',
|
||||
catalog_credential = '%s:%s',
|
||||
oauth_server_uri = '%s'`,
|
||||
clickhouseDatabase,
|
||||
icebergURI, env.accessKey, env.secretKey,
|
||||
warehouseBucket,
|
||||
storageEndpoint,
|
||||
env.accessKey, env.secretKey,
|
||||
oauthURI,
|
||||
)
|
||||
|
||||
if _, err := env.query(createSQL, map[string]string{"allow_experimental_database_iceberg": "1"}); err != nil {
|
||||
t.Fatalf("CREATE DATABASE %s failed: %v\nContainer logs:\n%s",
|
||||
clickhouseDatabase, err, clickhouseContainerLogs(env.clickhouseContainer))
|
||||
}
|
||||
}
|
||||
|
||||
// containsLine reports whether any line of a TabSeparated result equals want.
|
||||
func containsLine(out, want string) bool {
|
||||
for _, line := range strings.Split(out, "\n") {
|
||||
if strings.TrimSpace(line) == want {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// requestIcebergOAuthToken requests an OAuth2 client_credentials token from
|
||||
// the SeaweedFS Iceberg REST catalog. Used to seed the catalog with a
|
||||
// namespace and tables directly through the REST API before ClickHouse connects.
|
||||
func requestIcebergOAuthToken(t *testing.T, env *TestEnvironment) string {
|
||||
t.Helper()
|
||||
|
||||
client := &http.Client{Timeout: 10 * time.Second}
|
||||
resp, err := client.PostForm(fmt.Sprintf("http://%s:%d/v1/oauth/tokens", env.bindIP, env.icebergPort), url.Values{
|
||||
"grant_type": {"client_credentials"},
|
||||
"client_id": {env.accessKey},
|
||||
"client_secret": {env.secretKey},
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("POST /v1/oauth/tokens: %v", err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
|
||||
body, _ := io.ReadAll(resp.Body)
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
t.Fatalf("OAuth token request failed: status=%d body=%s", resp.StatusCode, body)
|
||||
}
|
||||
|
||||
var tokenResp struct {
|
||||
AccessToken string `json:"access_token"`
|
||||
TokenType string `json:"token_type"`
|
||||
}
|
||||
if err := json.Unmarshal(body, &tokenResp); err != nil {
|
||||
t.Fatalf("decode token response: %v", err)
|
||||
}
|
||||
if tokenResp.AccessToken == "" {
|
||||
t.Fatal("got empty access_token")
|
||||
}
|
||||
return tokenResp.AccessToken
|
||||
}
|
||||
|
||||
// createIcebergNamespace creates a single-level Iceberg namespace through
|
||||
// the REST catalog.
|
||||
func createIcebergNamespace(t *testing.T, env *TestEnvironment, token, bucketName, namespace string) {
|
||||
t.Helper()
|
||||
|
||||
doIcebergJSONRequest(t, env, token, http.MethodPost, fmt.Sprintf("/v1/%s/namespaces", url.PathEscape(bucketName)), map[string]any{
|
||||
"namespace": []string{namespace},
|
||||
}, http.StatusOK, http.StatusConflict)
|
||||
}
|
||||
|
||||
// createIcebergTable creates a table inside a single-level namespace through
|
||||
// the REST catalog. The table is created with the canonical
|
||||
// (id long not null, label string nullable) schema used by all subtests.
|
||||
func createIcebergTable(t *testing.T, env *TestEnvironment, token, bucketName, namespace, tableName string) {
|
||||
t.Helper()
|
||||
|
||||
doIcebergJSONRequest(t, env, token, http.MethodPost,
|
||||
fmt.Sprintf("/v1/%s/namespaces/%s/tables", url.PathEscape(bucketName), url.PathEscape(namespace)),
|
||||
map[string]any{
|
||||
"name": tableName,
|
||||
"schema": map[string]any{
|
||||
"type": "struct",
|
||||
"schema-id": 0,
|
||||
"fields": []map[string]any{
|
||||
{"id": 1, "name": "id", "required": true, "type": "long"},
|
||||
{"id": 2, "name": "label", "required": false, "type": "string"},
|
||||
},
|
||||
},
|
||||
}, http.StatusOK)
|
||||
}
|
||||
|
||||
const clickhouseWriterImage = "seaweedfs-clickhouse-writer"
|
||||
|
||||
// buildClickHouseWriterImage builds the local PyIceberg writer image. Layer
|
||||
// caching makes repeat invocations cheap; the first build pulls
|
||||
// python:3.11-slim and pip-installs pyiceberg+pyarrow (~1-2 min in CI).
|
||||
func buildClickHouseWriterImage(t *testing.T) {
|
||||
t.Helper()
|
||||
|
||||
wd, err := os.Getwd()
|
||||
if err != nil {
|
||||
t.Fatalf("Failed to get working directory: %v", err)
|
||||
}
|
||||
|
||||
cmd := exec.Command("docker", "build",
|
||||
"-t", clickhouseWriterImage,
|
||||
"-f", filepath.Join(wd, "Dockerfile.writer"),
|
||||
wd,
|
||||
)
|
||||
if out, err := cmd.CombinedOutput(); err != nil {
|
||||
t.Fatalf("Failed to build %s image: %v\n%s", clickhouseWriterImage, err, out)
|
||||
}
|
||||
}
|
||||
|
||||
// writeIcebergRows runs the PyIceberg writer container, which loads the
|
||||
// already-created table and appends three rows.
|
||||
func writeIcebergRows(t *testing.T, env *TestEnvironment, bucketName string, namespace []string, tableName string) {
|
||||
t.Helper()
|
||||
|
||||
args := []string{
|
||||
"run", "--rm",
|
||||
"--add-host", "host.docker.internal:host-gateway",
|
||||
clickhouseWriterImage,
|
||||
"--catalog-url", fmt.Sprintf("http://host.docker.internal:%d", env.icebergPort),
|
||||
"--warehouse", "s3://" + bucketName,
|
||||
"--prefix", bucketName,
|
||||
"--s3-endpoint", fmt.Sprintf("http://host.docker.internal:%d", env.s3Port),
|
||||
"--access-key", env.accessKey,
|
||||
"--secret-key", env.secretKey,
|
||||
"--region", "us-west-2",
|
||||
"--table", tableName,
|
||||
}
|
||||
for _, level := range namespace {
|
||||
args = append(args, "--namespace", level)
|
||||
}
|
||||
|
||||
cmd := exec.Command("docker", args...)
|
||||
out, err := cmd.CombinedOutput()
|
||||
if err != nil {
|
||||
t.Fatalf("PyIceberg writer failed: %v\n%s", err, out)
|
||||
}
|
||||
t.Logf("PyIceberg writer output: %s", strings.TrimSpace(string(out)))
|
||||
}
|
||||
|
||||
// readIcebergRows scans a table with PyIceberg through the REST catalog and
|
||||
// returns its "id,label" lines, ordered by id.
|
||||
func readIcebergRows(t *testing.T, env *TestEnvironment, bucketName string, namespace []string, tableName string) string {
|
||||
t.Helper()
|
||||
|
||||
args := []string{
|
||||
"run", "--rm",
|
||||
"--add-host", "host.docker.internal:host-gateway",
|
||||
"--entrypoint", "python3",
|
||||
clickhouseWriterImage,
|
||||
"/app/read_rows.py",
|
||||
"--catalog-url", fmt.Sprintf("http://host.docker.internal:%d", env.icebergPort),
|
||||
"--warehouse", "s3://" + bucketName,
|
||||
"--prefix", bucketName,
|
||||
"--s3-endpoint", fmt.Sprintf("http://host.docker.internal:%d", env.s3Port),
|
||||
"--access-key", env.accessKey,
|
||||
"--secret-key", env.secretKey,
|
||||
"--region", "us-west-2",
|
||||
"--table", tableName,
|
||||
}
|
||||
for _, level := range namespace {
|
||||
args = append(args, "--namespace", level)
|
||||
}
|
||||
|
||||
// Keep stdout separate: the caller compares it exactly, and warnings on
|
||||
// stderr from the python stack must not pollute the row data.
|
||||
cmd := exec.Command("docker", args...)
|
||||
var stdout, stderr bytes.Buffer
|
||||
cmd.Stdout = &stdout
|
||||
cmd.Stderr = &stderr
|
||||
if err := cmd.Run(); err != nil {
|
||||
t.Fatalf("PyIceberg reader failed: %v\nstdout:\n%s\nstderr:\n%s", err, stdout.String(), stderr.String())
|
||||
}
|
||||
return strings.TrimSpace(stdout.String())
|
||||
}
|
||||
|
||||
// doIcebergJSONRequest issues an authenticated JSON request to the Iceberg
|
||||
// REST endpoint and returns the response body. It fails the test unless the
|
||||
// response status matches one of expectedStatuses.
|
||||
func doIcebergJSONRequest(t *testing.T, env *TestEnvironment, token, method, path string, payload any, expectedStatuses ...int) string {
|
||||
t.Helper()
|
||||
|
||||
var body io.Reader
|
||||
if payload != nil {
|
||||
payloadBytes, err := json.Marshal(payload)
|
||||
if err != nil {
|
||||
t.Fatalf("marshal Iceberg request: %v", err)
|
||||
}
|
||||
body = bytes.NewReader(payloadBytes)
|
||||
}
|
||||
|
||||
req, err := http.NewRequest(method, fmt.Sprintf("http://%s:%d%s", env.bindIP, env.icebergPort, path), body)
|
||||
if err != nil {
|
||||
t.Fatalf("create Iceberg request: %v", err)
|
||||
}
|
||||
req.Header.Set("Authorization", "Bearer "+token)
|
||||
if payload != nil {
|
||||
req.Header.Set("Content-Type", "application/json")
|
||||
}
|
||||
|
||||
client := &http.Client{Timeout: 30 * time.Second}
|
||||
resp, err := client.Do(req)
|
||||
if err != nil {
|
||||
t.Fatalf("Iceberg request failed: %v", err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
|
||||
respBody, _ := io.ReadAll(resp.Body)
|
||||
for _, expectedStatus := range expectedStatuses {
|
||||
if resp.StatusCode == expectedStatus {
|
||||
return string(respBody)
|
||||
}
|
||||
}
|
||||
t.Fatalf("Iceberg request returned unexpected status %d, want %v\nPath: %s\nBody: %s",
|
||||
resp.StatusCode, expectedStatuses, path, respBody)
|
||||
return ""
|
||||
}
|
||||
|
||||
// createTableBucket creates an S3 table bucket using `weed shell`, which
|
||||
// talks to the master over gRPC and bypasses the S3 SigV4 path. The `-master`
|
||||
// flag uses SeaweedFS's canonical `host:port.grpcPort` ServerAddress format
|
||||
// produced by pb.NewServerAddress — the dot separates the HTTP port from the
|
||||
// gRPC port and is required, not a typo.
|
||||
func createTableBucket(t *testing.T, env *TestEnvironment, bucketName string) {
|
||||
t.Helper()
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), 60*time.Second)
|
||||
defer cancel()
|
||||
|
||||
cmd := exec.CommandContext(ctx, env.weedBinary, "shell",
|
||||
fmt.Sprintf("-master=%s:%d.%d", env.bindIP, env.masterPort, env.masterGrpcPort),
|
||||
)
|
||||
cmd.Stdin = strings.NewReader(fmt.Sprintf("s3tables.bucket -create -name %s -account 000000000000\nexit\n", bucketName))
|
||||
output, err := cmd.CombinedOutput()
|
||||
if err != nil {
|
||||
t.Fatalf("Failed to create table bucket %s via weed shell: %v\nOutput: %s", bucketName, err, string(output))
|
||||
}
|
||||
t.Logf("Created table bucket: %s", bucketName)
|
||||
}
|
||||
|
||||
// requireClickHouseRuntime skips the test in `-short` mode or when Docker
|
||||
// isn't available, since the test cannot run without the ClickHouse container.
|
||||
func requireClickHouseRuntime(t *testing.T) {
|
||||
t.Helper()
|
||||
|
||||
if testing.Short() {
|
||||
t.Skip("Skipping integration test in short mode")
|
||||
}
|
||||
if !hasDocker() {
|
||||
t.Skip("Docker not available, skipping ClickHouse integration test")
|
||||
}
|
||||
}
|
||||
|
||||
// hasDocker reports whether `docker version` can run, which we treat as a
|
||||
// sufficient signal that a Docker daemon is reachable from this process.
|
||||
func hasDocker() bool {
|
||||
cmd := exec.Command("docker", "version")
|
||||
return cmd.Run() == nil
|
||||
}
|
||||
|
||||
// randomString returns a lowercase alphanumeric string of the given length.
|
||||
func randomString(length int) string {
|
||||
const charset = "abcdefghijklmnopqrstuvwxyz0123456789"
|
||||
b := make([]byte, length)
|
||||
if _, err := rand.Read(b); err != nil {
|
||||
panic("failed to generate random string: " + err.Error())
|
||||
}
|
||||
for i := range b {
|
||||
b[i] = charset[int(b[i])%len(charset)]
|
||||
}
|
||||
return string(b)
|
||||
}
|
||||
@@ -0,0 +1,54 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Read all rows of an Iceberg table via the SeaweedFS REST catalog.
|
||||
|
||||
Used by the ClickHouse integration test to prove that rows written by
|
||||
ClickHouse are readable by another engine: PyIceberg is a strict reader that
|
||||
requires spec-compliant manifests and either parquet field ids or a name
|
||||
mapping. Prints one "id,label" line per row, ordered by id.
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import sys
|
||||
|
||||
from pyiceberg.catalog import load_catalog
|
||||
|
||||
|
||||
def main() -> int:
|
||||
p = argparse.ArgumentParser()
|
||||
p.add_argument("--catalog-url", required=True)
|
||||
p.add_argument("--warehouse", required=True)
|
||||
p.add_argument("--prefix", required=True)
|
||||
p.add_argument("--s3-endpoint", required=True)
|
||||
p.add_argument("--access-key", required=True)
|
||||
p.add_argument("--secret-key", required=True)
|
||||
p.add_argument("--region", default="us-east-1")
|
||||
p.add_argument("--namespace", action="append", required=True)
|
||||
p.add_argument("--table", required=True)
|
||||
args = p.parse_args()
|
||||
|
||||
catalog = load_catalog(
|
||||
"rest",
|
||||
**{
|
||||
"type": "rest",
|
||||
"uri": args.catalog_url,
|
||||
"warehouse": args.warehouse,
|
||||
"prefix": args.prefix,
|
||||
"credential": f"{args.access_key}:{args.secret_key}",
|
||||
"s3.access-key-id": args.access_key,
|
||||
"s3.secret-access-key": args.secret_key,
|
||||
"s3.endpoint": args.s3_endpoint,
|
||||
"s3.region": args.region,
|
||||
"s3.path-style-access": "true",
|
||||
},
|
||||
)
|
||||
|
||||
table = catalog.load_table(tuple(args.namespace) + (args.table,))
|
||||
data = table.scan().to_arrow().to_pydict()
|
||||
rows = sorted(zip(data["id"], data["label"]))
|
||||
for row_id, label in rows:
|
||||
print(f"{row_id},{label}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -90,7 +90,7 @@ func TestCopyFileIgnoreNotFoundAndStopOffsetZeroPaths(t *testing.T) {
|
||||
|
||||
missingNoIgnore, err := grpcClient.CopyFile(ctx, &volume_server_pb.CopyFileRequest{
|
||||
VolumeId: volumeID,
|
||||
Ext: ".definitely-missing",
|
||||
Ext: ".missing",
|
||||
CompactionRevision: math.MaxUint32,
|
||||
StopOffset: 1,
|
||||
IgnoreSourceFileNotFound: false,
|
||||
@@ -104,7 +104,7 @@ func TestCopyFileIgnoreNotFoundAndStopOffsetZeroPaths(t *testing.T) {
|
||||
|
||||
missingIgnored, err := grpcClient.CopyFile(ctx, &volume_server_pb.CopyFileRequest{
|
||||
VolumeId: volumeID,
|
||||
Ext: ".definitely-missing",
|
||||
Ext: ".missing",
|
||||
CompactionRevision: math.MaxUint32,
|
||||
StopOffset: 1,
|
||||
IgnoreSourceFileNotFound: true,
|
||||
@@ -119,7 +119,7 @@ func TestCopyFileIgnoreNotFoundAndStopOffsetZeroPaths(t *testing.T) {
|
||||
|
||||
stopZeroStream, err := grpcClient.CopyFile(ctx, &volume_server_pb.CopyFileRequest{
|
||||
VolumeId: volumeID,
|
||||
Ext: ".definitely-missing",
|
||||
Ext: ".missing",
|
||||
CompactionRevision: math.MaxUint32,
|
||||
StopOffset: 0,
|
||||
IgnoreSourceFileNotFound: false,
|
||||
|
||||
@@ -352,6 +352,21 @@ func TestConcurrentReaders(t *testing.T) {
|
||||
t.Fatalf("write: %v", err)
|
||||
}
|
||||
|
||||
// WriteFile's own existence probe ran while the file did not exist, and
|
||||
// WinFsp may serve that answer from its metadata cache for up to the
|
||||
// mount's FileInfoTimeout. Establish visibility once before racing the
|
||||
// readers, so the race exercises concurrent reading rather than the
|
||||
// cache window.
|
||||
deadline := time.Now().Add(3 * time.Second)
|
||||
for {
|
||||
if _, err := os.Stat(path); err == nil {
|
||||
break
|
||||
} else if time.Now().After(deadline) {
|
||||
t.Fatalf("file never became visible: %v", err)
|
||||
}
|
||||
time.Sleep(50 * time.Millisecond)
|
||||
}
|
||||
|
||||
var wg sync.WaitGroup
|
||||
errs := make(chan error, 8)
|
||||
for r := 0; r < 8; r++ {
|
||||
|
||||
@@ -138,6 +138,16 @@ func (cluster *Cluster) ListClusterNode(filerGroup FilerGroupName, nodeType stri
|
||||
return
|
||||
}
|
||||
|
||||
// ListClusterNodeUpdates reports the current members as add updates, so a
|
||||
// client that just connected can rebuild the membership it missed while it was
|
||||
// away.
|
||||
func (cluster *Cluster) ListClusterNodeUpdates(filerGroup FilerGroupName, nodeType string) (updates []*master_pb.KeepConnectedResponse) {
|
||||
for _, node := range cluster.ListClusterNode(filerGroup, nodeType) {
|
||||
updates = append(updates, buildClusterNodeUpdateMessage(true, filerGroup, nodeType, node.Address)...)
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
// IsKnownNode reports whether address is currently registered under nodeType
|
||||
// in any filer group. The lookup is intentionally group-agnostic because callers
|
||||
// (e.g. Ping admission) only know the target address, not the group it joined.
|
||||
|
||||
@@ -40,6 +40,27 @@ func TestConcurrentAddRemoveNodes(t *testing.T) {
|
||||
wg.Wait()
|
||||
}
|
||||
|
||||
func TestListClusterNodeUpdates(t *testing.T) {
|
||||
c := NewCluster()
|
||||
filer := pb.ServerAddress("10.0.0.20:8888")
|
||||
c.AddClusterNode("group", FilerType, "dc1", "rack1", filer, "test")
|
||||
c.AddClusterNode("group", BrokerType, "dc1", "rack1", pb.ServerAddress("10.0.0.20:17777"), "test")
|
||||
|
||||
updates := c.ListClusterNodeUpdates("group", FilerType)
|
||||
if len(updates) != 1 {
|
||||
t.Fatalf("expecting one filer update, got %d", len(updates))
|
||||
}
|
||||
update := updates[0].ClusterNodeUpdate
|
||||
if update.Address != string(filer) || !update.IsAdd || update.FilerGroup != "group" {
|
||||
t.Fatalf("unexpected update %+v", update)
|
||||
}
|
||||
|
||||
c.RemoveClusterNode("group", FilerType, filer)
|
||||
if updates := c.ListClusterNodeUpdates("group", FilerType); len(updates) != 0 {
|
||||
t.Fatalf("expecting no update for a removed filer, got %d", len(updates))
|
||||
}
|
||||
}
|
||||
|
||||
func TestIsKnownNode(t *testing.T) {
|
||||
c := NewCluster()
|
||||
filer := pb.ServerAddress("10.0.0.20:8888")
|
||||
|
||||
@@ -22,6 +22,7 @@ var Commands = []*Command{
|
||||
cmdFilerCat,
|
||||
cmdFilerCopy,
|
||||
cmdFilerMetaBackup,
|
||||
cmdFilerMetaScan,
|
||||
cmdFilerMetaTail,
|
||||
cmdFilerRemoteGateway,
|
||||
cmdFilerRemoteSynchronize,
|
||||
|
||||
@@ -80,6 +80,8 @@ type FilerOptions struct {
|
||||
allowedOrigins *string
|
||||
exposeDirectoryData *bool
|
||||
tusBasePath *string
|
||||
tusMaxSizeMB *int
|
||||
tusSessionExpiry *time.Duration
|
||||
s3ConfigFile *string // optional path to static S3 identity config
|
||||
// shutdownCtx, when non-nil, tells startFiler to gracefully shut down its
|
||||
// HTTP/gRPC servers once the ctx is cancelled. Used by integration tests
|
||||
@@ -123,6 +125,8 @@ func init() {
|
||||
f.allowedOrigins = cmdFiler.Flag.String("allowedOrigins", "*", "comma separated list of allowed origins")
|
||||
f.exposeDirectoryData = cmdFiler.Flag.Bool("exposeDirectoryData", true, "whether to return directory metadata and content in Filer UI")
|
||||
f.tusBasePath = cmdFiler.Flag.String("tusBasePath", "/.tus", "TUS resumable upload endpoint base path (e.g., /.tus)")
|
||||
f.tusMaxSizeMB = cmdFiler.Flag.Int("tusMaxSizeMB", 5*1024, "maximum TUS upload size in MB")
|
||||
f.tusSessionExpiry = cmdFiler.Flag.Duration("tusSessionExpiry", 24*time.Hour, "incomplete TUS upload sessions are cleaned up after this duration, e.g. \"48h\", \"7h30m\"")
|
||||
|
||||
// start s3 on filer
|
||||
filerStartS3 = cmdFiler.Flag.Bool("s3", false, "whether to start S3 gateway")
|
||||
@@ -386,6 +390,8 @@ func (fo *FilerOptions) startFiler() {
|
||||
DiskType: *fo.diskType,
|
||||
AllowedOrigins: strings.Split(*fo.allowedOrigins, ","),
|
||||
TusBasePath: *fo.tusBasePath,
|
||||
TusMaxSize: int64(*fo.tusMaxSizeMB) * 1024 * 1024,
|
||||
TusSessionExpiry: *fo.tusSessionExpiry,
|
||||
CredentialManager: credentialManager,
|
||||
})
|
||||
if nfs_err != nil {
|
||||
|
||||
@@ -0,0 +1,352 @@
|
||||
package command
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"io"
|
||||
"os"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"google.golang.org/grpc"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/filer"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api/s3_constants"
|
||||
"github.com/seaweedfs/seaweedfs/weed/security"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
)
|
||||
|
||||
func init() {
|
||||
cmdFilerMetaScan.Run = runFilerMetaScan // break init cycle
|
||||
}
|
||||
|
||||
var cmdFilerMetaScan = &Command{
|
||||
UsageLine: "filer.meta.scan -pathPrefix=/some/dir [-since=...] [-until=...]",
|
||||
Short: "audit what happened under a directory, one line per change",
|
||||
Long: `Replay the filer's metadata log for one directory and print one line per change.
|
||||
|
||||
Unlike filer.meta.tail this stops when it reaches the end of the requested
|
||||
range instead of following, so it can be piped straight into grep or awk.
|
||||
|
||||
Each line is:
|
||||
|
||||
<time> <OP> <path> <details>
|
||||
|
||||
OP is CREATE, DELETE, UPDATE or RENAME. Times are printed with the offset of
|
||||
the machine running the command.
|
||||
|
||||
Ranges may be given as absolute timestamps or as durations before now:
|
||||
|
||||
weed filer.meta.scan -pathPrefix=/buckets/b -since="2026-08-03 00:00:00" -until="2026-08-03 06:00:00"
|
||||
weed filer.meta.scan -pathPrefix=/buckets/b -timeAgo=48h -untilTimeAgo=24h
|
||||
weed filer.meta.scan -pathPrefix=/buckets/b -timeAgo=1h -follow
|
||||
|
||||
-since and -until are read in the machine's timezone unless -tz is given:
|
||||
|
||||
weed filer.meta.scan -pathPrefix=/buckets/b -since="2026-08-03 01:00:00" -tz=America/Sao_Paulo
|
||||
|
||||
On a versioned bucket an object is stored as <key>.versions/v_<versionId>.
|
||||
Those are reported against <key> with the version id in the details, so
|
||||
-name matches the key a client would ask for rather than the internal path:
|
||||
|
||||
weed filer.meta.scan -pathPrefix=/buckets/b -timeAgo=720h -name=summary.xml
|
||||
weed filer.meta.scan -pathPrefix=/buckets/b -timeAgo=720h -op=DELETE
|
||||
|
||||
Pass -raw to report the stored paths instead.
|
||||
|
||||
Persisted ranges are read straight from the volume servers, so the filer
|
||||
hands out log chunk ids instead of decoding and filtering the whole range
|
||||
itself — worth having on a busy cluster, where that decode is the expensive
|
||||
part and is charged to the filer no matter how narrow the prefix is. It
|
||||
needs a route to the volume servers; without one the scan falls back to
|
||||
reading through the filer, and -directRead=false forces that path.
|
||||
|
||||
`,
|
||||
}
|
||||
|
||||
var (
|
||||
scanFiler = cmdFilerMetaScan.Flag.String("filer", "localhost:8888", "filer hostname:port")
|
||||
scanPathPrefix = cmdFilerMetaScan.Flag.String("pathPrefix", "/", "directory to audit; filtered on the filer, so keep it as tight as possible")
|
||||
scanSince = cmdFilerMetaScan.Flag.String("since", "", "start time, \"2006-01-02 15:04:05\" or RFC3339")
|
||||
scanUntil = cmdFilerMetaScan.Flag.String("until", "", "stop time, same formats as -since; defaults to now")
|
||||
scanTimeAgo = cmdFilerMetaScan.Flag.Duration("timeAgo", 0, "start this long before now, e.g. \"48h\"; ignored when -since is set")
|
||||
scanUntilTimeAgo = cmdFilerMetaScan.Flag.Duration("untilTimeAgo", 0, "stop this long before now; ignored when -until is set")
|
||||
scanTz = cmdFilerMetaScan.Flag.String("tz", "", "timezone for -since/-until, e.g. \"America/Sao_Paulo\" or \"UTC\"; defaults to this machine's")
|
||||
scanName = cmdFilerMetaScan.Flag.String("name", "", "only report paths containing this text, case-insensitive")
|
||||
scanOp = cmdFilerMetaScan.Flag.String("op", "", "only report these operations, comma-separated: CREATE,DELETE,UPDATE,RENAME")
|
||||
scanRaw = cmdFilerMetaScan.Flag.Bool("raw", false, "report stored paths instead of resolving .versions/v_<id> back to the object key")
|
||||
scanFollow = cmdFilerMetaScan.Flag.Bool("follow", false, "keep following after reaching the end of the range")
|
||||
scanDirectRead = cmdFilerMetaScan.Flag.Bool("directRead", true, "read log chunks straight from the volume servers, so the filer does not decode the range; needs volume server access, falls back automatically")
|
||||
)
|
||||
|
||||
// scanFilerClient adapts a filer address to filer_pb.FilerClient so the chunk
|
||||
// reader can resolve a log chunk's volume locations.
|
||||
type scanFilerClient struct {
|
||||
address pb.ServerAddress
|
||||
grpcDialOption grpc.DialOption
|
||||
signature int32
|
||||
}
|
||||
|
||||
func (c *scanFilerClient) WithFilerClient(streamingMode bool, fn func(filer_pb.SeaweedFilerClient) error) error {
|
||||
return pb.WithFilerClient(streamingMode, c.signature, c.address, c.grpcDialOption, fn)
|
||||
}
|
||||
|
||||
func (c *scanFilerClient) AdjustedUrl(location *filer_pb.Location) string { return location.Url }
|
||||
|
||||
func (c *scanFilerClient) GetDataCenter() string { return "" }
|
||||
|
||||
const scanTimeLayout = "2006-01-02 15:04:05"
|
||||
|
||||
// parseScanTime accepts either the human layout or RFC3339. Without an explicit
|
||||
// zone in the string, loc decides — which is the difference between finding the
|
||||
// window and missing it by the UTC offset.
|
||||
func parseScanTime(value string, loc *time.Location) (time.Time, error) {
|
||||
if t, err := time.Parse(time.RFC3339, value); err == nil {
|
||||
return t, nil
|
||||
}
|
||||
t, err := time.ParseInLocation(scanTimeLayout, value, loc)
|
||||
if err != nil {
|
||||
return time.Time{}, fmt.Errorf("parse %q: want %q or RFC3339", value, scanTimeLayout)
|
||||
}
|
||||
return t, nil
|
||||
}
|
||||
|
||||
// scanEvent is one metadata change reduced to what a path audit needs.
|
||||
type scanEvent struct {
|
||||
tsNs int64
|
||||
op string
|
||||
path string
|
||||
details string
|
||||
}
|
||||
|
||||
const versionsSuffix = ".versions"
|
||||
|
||||
// logicalPath maps a stored path back to the object key a client would use. A
|
||||
// versioned object lives at <key>.versions/v_<id>, so both the container and the
|
||||
// version files are reported against <key>; the version id moves to the details.
|
||||
// kind distinguishes the container from the versions inside it, since a change to
|
||||
// the container is a pointer flip rather than a write of object data.
|
||||
func logicalPath(dir, name string) (path, version, kind string) {
|
||||
full := dir + "/" + name
|
||||
if *scanRaw {
|
||||
return full, "", ""
|
||||
}
|
||||
if strings.HasPrefix(name, "v_") && strings.HasSuffix(dir, versionsSuffix) {
|
||||
return strings.TrimSuffix(dir, versionsSuffix), strings.TrimPrefix(name, "v_"), ""
|
||||
}
|
||||
if strings.HasSuffix(name, versionsSuffix) {
|
||||
return strings.TrimSuffix(full, versionsSuffix), "", "versions-container"
|
||||
}
|
||||
return full, "", ""
|
||||
}
|
||||
|
||||
func describeEntry(entry *filer_pb.Entry, version, kind string) string {
|
||||
var parts []string
|
||||
if version != "" {
|
||||
parts = append(parts, "version="+version)
|
||||
}
|
||||
if kind != "" {
|
||||
parts = append(parts, kind)
|
||||
}
|
||||
if entry != nil {
|
||||
// A retraction is stored as a zero-length version carrying this flag.
|
||||
// Reporting it as an ordinary write would read as "the object was
|
||||
// written" when it means the opposite.
|
||||
if marker, ok := entry.Extended[s3_constants.ExtDeleteMarkerKey]; ok && string(marker) == "true" {
|
||||
parts = append(parts, "delete-marker")
|
||||
} else if entry.IsDirectory {
|
||||
if kind == "" {
|
||||
parts = append(parts, "dir")
|
||||
}
|
||||
} else {
|
||||
parts = append(parts, fmt.Sprintf("size=%d", filer.FileSize(entry)))
|
||||
if n := len(entry.GetChunks()); n > 0 {
|
||||
parts = append(parts, fmt.Sprintf("chunks=%d", n))
|
||||
}
|
||||
}
|
||||
}
|
||||
if len(parts) == 0 {
|
||||
return "-"
|
||||
}
|
||||
return strings.Join(parts, " ")
|
||||
}
|
||||
|
||||
// toScanEvent classifies one notification. Which of old/new is present is what
|
||||
// distinguishes the operations; when both are, the path tells a rename from an
|
||||
// in-place update.
|
||||
func toScanEvent(resp *filer_pb.SubscribeMetadataResponse) *scanEvent {
|
||||
n := resp.EventNotification
|
||||
switch {
|
||||
case n.OldEntry == nil && n.NewEntry == nil:
|
||||
return nil
|
||||
|
||||
case n.OldEntry == nil:
|
||||
path, version, kind := logicalPath(n.NewParentPath, n.NewEntry.Name)
|
||||
return &scanEvent{resp.TsNs, "CREATE", path, describeEntry(n.NewEntry, version, kind)}
|
||||
|
||||
case n.NewEntry == nil:
|
||||
path, version, kind := logicalPath(resp.Directory, n.OldEntry.Name)
|
||||
return &scanEvent{resp.TsNs, "DELETE", path, describeEntry(n.OldEntry, version, kind)}
|
||||
|
||||
default:
|
||||
oldPath, oldVersion, _ := logicalPath(resp.Directory, n.OldEntry.Name)
|
||||
newPath, newVersion, newKind := logicalPath(n.NewParentPath, n.NewEntry.Name)
|
||||
if oldPath == newPath && oldVersion == newVersion {
|
||||
return &scanEvent{resp.TsNs, "UPDATE", newPath, describeEntry(n.NewEntry, newVersion, newKind)}
|
||||
}
|
||||
details := describeEntry(n.NewEntry, newVersion, newKind) + " from=" + oldPath
|
||||
if oldVersion != "" {
|
||||
details += "#" + oldVersion
|
||||
}
|
||||
return &scanEvent{resp.TsNs, "RENAME", newPath, details}
|
||||
}
|
||||
}
|
||||
|
||||
func runFilerMetaScan(cmd *Command, args []string) bool {
|
||||
|
||||
loc := time.Local
|
||||
if *scanTz != "" {
|
||||
parsed, err := time.LoadLocation(*scanTz)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "-tz: %v\n", err)
|
||||
return false
|
||||
}
|
||||
loc = parsed
|
||||
}
|
||||
|
||||
now := time.Now()
|
||||
|
||||
var startTs time.Time
|
||||
switch {
|
||||
case *scanSince != "":
|
||||
parsed, err := parseScanTime(*scanSince, loc)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "-since: %v\n", err)
|
||||
return false
|
||||
}
|
||||
startTs = parsed
|
||||
case *scanTimeAgo > 0:
|
||||
startTs = now.Add(-*scanTimeAgo)
|
||||
default:
|
||||
fmt.Fprintln(os.Stderr, "need a start: pass -since or -timeAgo")
|
||||
return false
|
||||
}
|
||||
|
||||
// Default the end to now so the command terminates; following is opt-in.
|
||||
stopTs := now
|
||||
switch {
|
||||
case *scanUntil != "":
|
||||
parsed, err := parseScanTime(*scanUntil, loc)
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "-until: %v\n", err)
|
||||
return false
|
||||
}
|
||||
stopTs = parsed
|
||||
case *scanUntilTimeAgo > 0:
|
||||
stopTs = now.Add(-*scanUntilTimeAgo)
|
||||
}
|
||||
|
||||
if !stopTs.After(startTs) && !*scanFollow {
|
||||
fmt.Fprintf(os.Stderr, "empty range: start %s is not before stop %s\n",
|
||||
startTs.Format(time.RFC3339), stopTs.Format(time.RFC3339))
|
||||
return false
|
||||
}
|
||||
|
||||
wantOps := map[string]bool{}
|
||||
for _, op := range strings.Split(*scanOp, ",") {
|
||||
if op = strings.ToUpper(strings.TrimSpace(op)); op != "" {
|
||||
wantOps[op] = true
|
||||
}
|
||||
}
|
||||
nameFilter := strings.ToLower(*scanName)
|
||||
|
||||
var stopTsNs int64
|
||||
if !*scanFollow {
|
||||
stopTsNs = stopTs.UnixNano()
|
||||
}
|
||||
|
||||
fmt.Fprintf(os.Stderr, "scanning %s from %s to %s\n", *scanPathPrefix,
|
||||
startTs.Format(time.RFC3339), stopTs.Format(time.RFC3339))
|
||||
|
||||
util.LoadSecurityConfiguration()
|
||||
grpcDialOption := security.LoadClientTLS(util.GetViper(), "grpc.client")
|
||||
|
||||
filerAddress := pb.ServerAddress(*scanFiler)
|
||||
matched := 0
|
||||
|
||||
emit := func(resp *filer_pb.SubscribeMetadataResponse) error {
|
||||
if filer_pb.IsEmpty(resp) {
|
||||
return nil
|
||||
}
|
||||
event := toScanEvent(resp)
|
||||
if event == nil {
|
||||
return nil
|
||||
}
|
||||
if len(wantOps) > 0 && !wantOps[event.op] {
|
||||
return nil
|
||||
}
|
||||
if nameFilter != "" && !strings.Contains(strings.ToLower(event.path), nameFilter) {
|
||||
return nil
|
||||
}
|
||||
matched++
|
||||
fmt.Printf("%s\t%s\t%s\t%s\n",
|
||||
time.Unix(0, event.tsNs).Format(time.RFC3339), event.op, event.path, event.details)
|
||||
return nil
|
||||
}
|
||||
|
||||
scan := func(directRead bool) error {
|
||||
option := &pb.MetadataFollowOption{
|
||||
ClientName: "scan",
|
||||
ClientId: util.RandomInt32(),
|
||||
ClientEpoch: 0,
|
||||
SelfSignature: 0,
|
||||
PathPrefix: *scanPathPrefix,
|
||||
StartTsNs: startTs.UnixNano(),
|
||||
StopTsNs: stopTsNs,
|
||||
EventErrorType: pb.TrivialOnError,
|
||||
}
|
||||
if directRead {
|
||||
// The server hands out log chunk fids and this reads them straight
|
||||
// from the volume servers, so the filer never decodes the range.
|
||||
// The prefix filter is re-applied client-side by ReadLogFileRefs,
|
||||
// so the result is the same either way.
|
||||
client := &scanFilerClient{address: filerAddress, grpcDialOption: grpcDialOption, signature: option.SelfSignature}
|
||||
lookupFn := filer.LookupFn(client)
|
||||
option.LogFileReaderFn = func(chunks []*filer_pb.FileChunk) (io.ReadCloser, error) {
|
||||
return filer.NewChunkStreamReaderFromLookup(context.Background(), lookupFn, chunks), nil
|
||||
}
|
||||
}
|
||||
return pb.FollowMetadata(filerAddress, grpcDialOption, option, emit)
|
||||
}
|
||||
|
||||
followErr := scan(*scanDirectRead)
|
||||
if *scanDirectRead && matched == 0 {
|
||||
// Two ways direct read can come back with nothing: it failed outright
|
||||
// (reading chunks needs a route to the volume servers that the filer
|
||||
// does not), or it succeeded and found none. The second is the
|
||||
// dangerous one — for an audit an empty answer reads as "nothing
|
||||
// happened here", so it has to be confirmed rather than trusted.
|
||||
// Re-running is safe precisely because nothing was printed; after
|
||||
// partial output a replay would duplicate lines instead.
|
||||
if followErr != nil {
|
||||
fmt.Fprintf(os.Stderr, "direct read failed (%v); retrying through the filer\n", followErr)
|
||||
} else {
|
||||
fmt.Fprintln(os.Stderr, "direct read found nothing; confirming through the filer")
|
||||
}
|
||||
followErr = scan(false)
|
||||
if followErr == nil && matched > 0 {
|
||||
fmt.Fprintf(os.Stderr, "warning: direct read missed %d change(s) the filer returned; "+
|
||||
"please report this, and pass -directRead=false meanwhile\n", matched)
|
||||
}
|
||||
}
|
||||
|
||||
if followErr != nil {
|
||||
fmt.Fprintf(os.Stderr, "scan %s: %v\n", *scanFiler, followErr)
|
||||
return false
|
||||
}
|
||||
|
||||
fmt.Fprintf(os.Stderr, "%d matching change(s)\n", matched)
|
||||
return true
|
||||
}
|
||||
@@ -0,0 +1,177 @@
|
||||
package command
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/s3api/s3_constants"
|
||||
)
|
||||
|
||||
func TestLogicalPathResolvesVersionedLayout(t *testing.T) {
|
||||
raw := false
|
||||
scanRaw = &raw
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
dir, entry string
|
||||
wantPath string
|
||||
wantVersion string
|
||||
wantKind string
|
||||
}{
|
||||
{
|
||||
// The reason this exists: a client asks for summary.xml, but every
|
||||
// write lands on a v_<id> inside a sibling directory, so a filter on
|
||||
// the stored name never matches what the client is looking for.
|
||||
name: "version file reports the object key",
|
||||
dir: "/buckets/b/rp/summary.xml.versions", entry: "v_abc123",
|
||||
wantPath: "/buckets/b/rp/summary.xml", wantVersion: "abc123",
|
||||
},
|
||||
{
|
||||
name: "versions directory reports the object key",
|
||||
dir: "/buckets/b/rp", entry: "summary.xml.versions",
|
||||
wantPath: "/buckets/b/rp/summary.xml", wantKind: "versions-container",
|
||||
},
|
||||
{
|
||||
name: "unversioned object is unchanged",
|
||||
dir: "/buckets/b/rp", entry: "summary.xml",
|
||||
wantPath: "/buckets/b/rp/summary.xml",
|
||||
},
|
||||
{
|
||||
// A v_ prefix only means a version inside a .versions directory.
|
||||
name: "v_ prefix outside a versions directory is a normal name",
|
||||
dir: "/buckets/b/rp", entry: "v_notaversion",
|
||||
wantPath: "/buckets/b/rp/v_notaversion",
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
path, version, kind := logicalPath(tt.dir, tt.entry)
|
||||
if path != tt.wantPath || version != tt.wantVersion || kind != tt.wantKind {
|
||||
t.Errorf("logicalPath(%q,%q) = (%q,%q,%q), want (%q,%q,%q)",
|
||||
tt.dir, tt.entry, path, version, kind, tt.wantPath, tt.wantVersion, tt.wantKind)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestLogicalPathRawKeepsStoredPath(t *testing.T) {
|
||||
raw := true
|
||||
scanRaw = &raw
|
||||
|
||||
path, version, kind := logicalPath("/buckets/b/rp/summary.xml.versions", "v_abc123")
|
||||
if path != "/buckets/b/rp/summary.xml.versions/v_abc123" || version != "" || kind != "" {
|
||||
t.Errorf("raw mode must not rewrite the path, got (%q,%q,%q)", path, version, kind)
|
||||
}
|
||||
}
|
||||
|
||||
func TestToScanEventClassifiesOperations(t *testing.T) {
|
||||
raw := false
|
||||
scanRaw = &raw
|
||||
|
||||
entry := func(name string) *filer_pb.Entry {
|
||||
return &filer_pb.Entry{Name: name, Attributes: &filer_pb.FuseAttributes{FileSize: 11}}
|
||||
}
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
resp *filer_pb.SubscribeMetadataResponse
|
||||
wantOp string
|
||||
wantPath string
|
||||
}{
|
||||
{
|
||||
name: "create",
|
||||
resp: &filer_pb.SubscribeMetadataResponse{
|
||||
EventNotification: &filer_pb.EventNotification{
|
||||
NewParentPath: "/b/rp", NewEntry: entry("obj"),
|
||||
},
|
||||
},
|
||||
wantOp: "CREATE", wantPath: "/b/rp/obj",
|
||||
},
|
||||
{
|
||||
name: "delete",
|
||||
resp: &filer_pb.SubscribeMetadataResponse{
|
||||
Directory: "/b/rp",
|
||||
EventNotification: &filer_pb.EventNotification{OldEntry: entry("obj")},
|
||||
},
|
||||
wantOp: "DELETE", wantPath: "/b/rp/obj",
|
||||
},
|
||||
{
|
||||
name: "update in place",
|
||||
resp: &filer_pb.SubscribeMetadataResponse{
|
||||
Directory: "/b/rp",
|
||||
EventNotification: &filer_pb.EventNotification{
|
||||
OldEntry: entry("obj"), NewParentPath: "/b/rp", NewEntry: entry("obj"),
|
||||
},
|
||||
},
|
||||
wantOp: "UPDATE", wantPath: "/b/rp/obj",
|
||||
},
|
||||
{
|
||||
name: "rename",
|
||||
resp: &filer_pb.SubscribeMetadataResponse{
|
||||
Directory: "/b/rp",
|
||||
EventNotification: &filer_pb.EventNotification{
|
||||
OldEntry: entry("obj"), NewParentPath: "/b/rp2", NewEntry: entry("obj2"),
|
||||
},
|
||||
},
|
||||
wantOp: "RENAME", wantPath: "/b/rp2/obj2",
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
got := toScanEvent(tt.resp)
|
||||
if got == nil {
|
||||
t.Fatal("expected an event")
|
||||
}
|
||||
if got.op != tt.wantOp || got.path != tt.wantPath {
|
||||
t.Errorf("got (%s,%s), want (%s,%s)", got.op, got.path, tt.wantOp, tt.wantPath)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// A retraction is a zero-length version carrying a flag; reporting it as a plain
|
||||
// write would read as the opposite of what happened.
|
||||
func TestDescribeEntryMarksDeleteMarkers(t *testing.T) {
|
||||
marker := &filer_pb.Entry{
|
||||
Name: "v_abc",
|
||||
Attributes: &filer_pb.FuseAttributes{},
|
||||
Extended: map[string][]byte{s3_constants.ExtDeleteMarkerKey: []byte("true")},
|
||||
}
|
||||
got := describeEntry(marker, "abc", "")
|
||||
if want := "version=abc delete-marker"; got != want {
|
||||
t.Errorf("describeEntry = %q, want %q", got, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestParseScanTimeHonoursTimezone(t *testing.T) {
|
||||
saoPaulo, err := time.LoadLocation("America/Sao_Paulo")
|
||||
if err != nil {
|
||||
t.Skipf("tzdata unavailable: %v", err)
|
||||
}
|
||||
|
||||
// The same wall-clock string is a different instant per zone, which is the
|
||||
// difference between finding a window and missing it by the UTC offset.
|
||||
inSP, err := parseScanTime("2026-08-03 01:16:42", saoPaulo)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
inUTC, err := parseScanTime("2026-08-03 01:16:42", time.UTC)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if delta := inSP.Sub(inUTC); delta != 3*time.Hour {
|
||||
t.Errorf("Sao Paulo should be 3h behind UTC, got %v", delta)
|
||||
}
|
||||
|
||||
// An explicit zone in the string wins over loc.
|
||||
explicit, err := parseScanTime("2026-08-03T01:16:42Z", saoPaulo)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !explicit.Equal(inUTC) {
|
||||
t.Errorf("RFC3339 zone must win, got %s want %s", explicit, inUTC)
|
||||
}
|
||||
}
|
||||
@@ -458,6 +458,8 @@ func initMiniFilerFlags() {
|
||||
miniFilerOptions.allowedOrigins = cmdMini.Flag.String("filer.allowedOrigins", "*", "comma separated list of allowed origins")
|
||||
miniFilerOptions.exposeDirectoryData = cmdMini.Flag.Bool("filer.exposeDirectoryData", true, "whether to return directory metadata and content in Filer UI")
|
||||
miniFilerOptions.tusBasePath = cmdMini.Flag.String("filer.tusBasePath", "/.tus", "TUS resumable upload endpoint base path")
|
||||
miniFilerOptions.tusMaxSizeMB = cmdMini.Flag.Int("filer.tusMaxSizeMB", 5*1024, "maximum TUS upload size in MB")
|
||||
miniFilerOptions.tusSessionExpiry = cmdMini.Flag.Duration("filer.tusSessionExpiry", 24*time.Hour, "incomplete TUS upload sessions are cleaned up after this duration")
|
||||
}
|
||||
|
||||
// initMiniVolumeFlags initializes Volume server flag options
|
||||
|
||||
@@ -20,6 +20,7 @@ type MountOptions struct {
|
||||
concurrentWriters *int
|
||||
concurrentReaders *int
|
||||
cacheMetaTtlSec *int
|
||||
cacheDirMaxEntries *int
|
||||
cacheDirForRead *string
|
||||
cacheDirForWrite *string
|
||||
cacheSizeMBForRead *int64
|
||||
@@ -114,6 +115,7 @@ func init() {
|
||||
mountOptions.cacheDirForWrite = cmdMount.Flag.String("cacheDirWrite", "", "buffer writes mostly for large files")
|
||||
mountOptions.writeBufferSizeMB = cmdMount.Flag.Int64("writeBufferSizeMB", 0, "global cap on the per-mount write buffer (memory + swap) in MB, 0 means unlimited. Bounds /tmp growth when volume uploads stall")
|
||||
mountOptions.cacheMetaTtlSec = cmdMount.Flag.Int("cacheMetaTtlSec", 60, "metadata cache validity seconds")
|
||||
mountOptions.cacheDirMaxEntries = cmdMount.Flag.Int("cacheDirMaxEntries", 10000, "a directory with more children than this is not cached locally but read directly from the filer; 0 caches everything")
|
||||
mountOptions.dataCenter = cmdMount.Flag.String("dataCenter", "", "prefer to write to the data center")
|
||||
mountOptions.allowOthers = cmdMount.Flag.Bool("allowOthers", true, "allows other users to access the file system")
|
||||
mountOptions.defaultPermissions = cmdMount.Flag.Bool("defaultPermissions", true, "enforce permissions by the operating system")
|
||||
|
||||
@@ -197,6 +197,7 @@ func buildSeaweedFileSystem(option *MountOptions, p fileSystemParams) *mount.WFS
|
||||
CacheDirForWrite: p.cacheDirForWrite,
|
||||
WriteBufferSizeMB: *option.writeBufferSizeMB,
|
||||
CacheMetaTTlSec: *option.cacheMetaTtlSec,
|
||||
CacheDirMaxEntries: *option.cacheDirMaxEntries,
|
||||
DataCenter: *option.dataCenter,
|
||||
Quota: int64(*option.collectionQuota) * 1024 * 1024,
|
||||
LogicalDiskUsage: *option.logicalDiskUsage,
|
||||
|
||||
@@ -130,6 +130,8 @@ func init() {
|
||||
filerOptions.diskType = cmdServer.Flag.String("filer.disk", "", "[hdd|ssd|<tag>] hard drive or solid state drive or any tag")
|
||||
filerOptions.exposeDirectoryData = cmdServer.Flag.Bool("filer.exposeDirectoryData", true, "expose directory data via filer. If false, filer UI will be inaccessible.")
|
||||
filerOptions.tusBasePath = cmdServer.Flag.String("filer.tusBasePath", "/.tus", "TUS resumable upload endpoint base path (e.g., /.tus)")
|
||||
filerOptions.tusMaxSizeMB = cmdServer.Flag.Int("filer.tusMaxSizeMB", 5*1024, "maximum TUS upload size in MB")
|
||||
filerOptions.tusSessionExpiry = cmdServer.Flag.Duration("filer.tusSessionExpiry", 24*time.Hour, "incomplete TUS upload sessions are cleaned up after this duration, e.g. \"48h\", \"7h30m\"")
|
||||
|
||||
serverOptions.v.port = cmdServer.Flag.Int("volume.port", 8080, "volume server http listen port")
|
||||
serverOptions.v.portGrpc = cmdServer.Flag.Int("volume.port.grpc", 0, "volume server grpc listen port")
|
||||
|
||||
@@ -0,0 +1,213 @@
|
||||
package filer
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"sync"
|
||||
"unicode/utf8"
|
||||
|
||||
"google.golang.org/protobuf/encoding/protowire"
|
||||
"google.golang.org/protobuf/proto"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
)
|
||||
|
||||
// Field numbers from filer.proto. Only the ones this decoder has to recognise
|
||||
// are named; every other Entry field is handed to the generated unmarshaller
|
||||
// as-is.
|
||||
const (
|
||||
entryChunksField = 3 // Entry.chunks
|
||||
fileChunkOffsetField = 2 // FileChunk.offset
|
||||
fileChunkSizeField = 3 // FileChunk.size
|
||||
)
|
||||
|
||||
// The chunk bytes are the one part of the blob the generated unmarshaller never
|
||||
// sees, so the checks it would have made are made here instead: a submessage
|
||||
// has to parse, and a proto3 string has to be valid UTF-8. Anything else in a
|
||||
// FileChunk is a scalar, which walking it already validates.
|
||||
// TestChunkValidationCoversEveryField fails if FileChunk gains a field of
|
||||
// either kind that is missing from these.
|
||||
var (
|
||||
fileChunkMessageFields = map[protowire.Number]bool{
|
||||
7: true, // fid, a FileId of scalars only, so walking it is a full check
|
||||
8: true, // source_fid
|
||||
}
|
||||
fileChunkStringFields = map[protowire.Number]bool{
|
||||
1: true, // file_id
|
||||
5: true, // e_tag
|
||||
6: true, // source_file_id
|
||||
}
|
||||
)
|
||||
|
||||
// attributesScratchPool holds the re-encoded entry, which is the blob minus its
|
||||
// chunks and so much smaller than what came in.
|
||||
var attributesScratchPool = sync.Pool{
|
||||
New: func() any {
|
||||
b := make([]byte, 0, 256)
|
||||
return &b
|
||||
},
|
||||
}
|
||||
|
||||
// DecodeListedEntry decodes one listed entry, dropping the chunk list when the
|
||||
// listing asked for attributes only.
|
||||
func DecodeListedEntry(ctx context.Context, entry *Entry, blob []byte) error {
|
||||
if filer_pb.ChunksOmitted(ctx) {
|
||||
return entry.DecodeAttributesOnly(blob)
|
||||
}
|
||||
return entry.DecodeAttributesAndChunks(blob)
|
||||
}
|
||||
|
||||
// DecodeAttributesOnly fills entry from blob without building its chunk list,
|
||||
// which is the bulk of the work for anything but a tiny file. Chunks are still
|
||||
// measured, because an entry whose stored FileSize is zero takes its size from
|
||||
// them, but no FileChunk is allocated.
|
||||
//
|
||||
// entry.Chunks is left nil. The one exception is a hard link, whose attributes
|
||||
// and chunks are replaced wholesale by a full decode of its own record in
|
||||
// FilerStoreWrapper.maybeReadHardLink straight after the store listing. Only a
|
||||
// caller that reads attributes and nothing else may use this.
|
||||
func (entry *Entry) DecodeAttributesOnly(blob []byte) error {
|
||||
// The scratch buffer is only taken once a chunk is actually found, so an
|
||||
// entry with none — every directory, for one — neither re-encodes nor
|
||||
// touches the pool, and is unmarshalled where it lies.
|
||||
var scratchPtr *[]byte
|
||||
var scratch []byte
|
||||
defer func() {
|
||||
if scratchPtr != nil {
|
||||
*scratchPtr = scratch
|
||||
attributesScratchPool.Put(scratchPtr)
|
||||
}
|
||||
}()
|
||||
|
||||
var chunkExtent uint64
|
||||
attributes := blob
|
||||
for pos := 0; pos < len(blob); {
|
||||
rest := blob[pos:]
|
||||
num, typ, tagLen := protowire.ConsumeTag(rest)
|
||||
if tagLen < 0 {
|
||||
return fmt.Errorf("decoding value blob for %s: %w", entry.FullPath, protowire.ParseError(tagLen))
|
||||
}
|
||||
var valLen int
|
||||
if num == entryChunksField && typ == protowire.BytesType {
|
||||
chunk, n := protowire.ConsumeBytes(rest[tagLen:])
|
||||
if n < 0 {
|
||||
return fmt.Errorf("decoding value blob for %s: %w", entry.FullPath, protowire.ParseError(n))
|
||||
}
|
||||
valLen = n
|
||||
end, err := chunkExtentEnd(chunk)
|
||||
if err != nil {
|
||||
return fmt.Errorf("decoding value blob for %s: %w", entry.FullPath, err)
|
||||
}
|
||||
if end > chunkExtent {
|
||||
chunkExtent = end
|
||||
}
|
||||
if scratchPtr == nil {
|
||||
scratchPtr = attributesScratchPool.Get().(*[]byte)
|
||||
scratch = append((*scratchPtr)[:0], blob[:pos]...)
|
||||
}
|
||||
} else {
|
||||
valLen = protowire.ConsumeFieldValue(num, typ, rest[tagLen:])
|
||||
if valLen < 0 {
|
||||
return fmt.Errorf("decoding value blob for %s: %w", entry.FullPath, protowire.ParseError(valLen))
|
||||
}
|
||||
if scratchPtr != nil {
|
||||
scratch = append(scratch, rest[:tagLen+valLen]...)
|
||||
}
|
||||
}
|
||||
pos += tagLen + valLen
|
||||
}
|
||||
if scratchPtr != nil {
|
||||
attributes = scratch
|
||||
}
|
||||
|
||||
message := pbEntryPool.Get().(*filer_pb.Entry)
|
||||
defer func() {
|
||||
resetPbEntry(message)
|
||||
pbEntryPool.Put(message)
|
||||
}()
|
||||
|
||||
if err := proto.Unmarshal(attributes, message); err != nil {
|
||||
return fmt.Errorf("decoding value blob for %s: %v", entry.FullPath, err)
|
||||
}
|
||||
|
||||
FromPbEntryToExistingEntry(message, entry)
|
||||
|
||||
// FromPbEntryToExistingEntry took the size over a chunk list that is not
|
||||
// there, so fold in what the chunks actually reached. This is TotalSize.
|
||||
if chunkExtent > entry.FileSize {
|
||||
entry.FileSize = chunkExtent
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
|
||||
// chunkExtentEnd reports where one encoded FileChunk ends, the offset plus size
|
||||
// that TotalSize maximises over, without building the chunk. A chunk this
|
||||
// rejects is one the full decoder rejects too, so a listing never reports a
|
||||
// size for an entry that cannot be opened.
|
||||
//
|
||||
// A manifest chunk needs no special handling: it carries the offset and size of
|
||||
// the whole range it stands for, and TotalSize does not resolve it either.
|
||||
func chunkExtentEnd(chunk []byte) (uint64, error) {
|
||||
var offset int64
|
||||
var size uint64
|
||||
for len(chunk) > 0 {
|
||||
num, typ, tagLen := protowire.ConsumeTag(chunk)
|
||||
if tagLen < 0 {
|
||||
return 0, protowire.ParseError(tagLen)
|
||||
}
|
||||
chunk = chunk[tagLen:]
|
||||
if typ == protowire.VarintType && (num == fileChunkOffsetField || num == fileChunkSizeField) {
|
||||
v, n := protowire.ConsumeVarint(chunk)
|
||||
if n < 0 {
|
||||
return 0, protowire.ParseError(n)
|
||||
}
|
||||
if num == fileChunkOffsetField {
|
||||
offset = int64(v)
|
||||
} else {
|
||||
size = v
|
||||
}
|
||||
chunk = chunk[n:]
|
||||
continue
|
||||
}
|
||||
if typ == protowire.BytesType {
|
||||
v, n := protowire.ConsumeBytes(chunk)
|
||||
if n < 0 {
|
||||
return 0, protowire.ParseError(n)
|
||||
}
|
||||
if fileChunkMessageFields[num] {
|
||||
if err := validateMessage(v); err != nil {
|
||||
return 0, err
|
||||
}
|
||||
} else if fileChunkStringFields[num] && !utf8.Valid(v) {
|
||||
return 0, errors.New("invalid UTF-8 in string field")
|
||||
}
|
||||
chunk = chunk[n:]
|
||||
continue
|
||||
}
|
||||
n := protowire.ConsumeFieldValue(num, typ, chunk)
|
||||
if n < 0 {
|
||||
return 0, protowire.ParseError(n)
|
||||
}
|
||||
chunk = chunk[n:]
|
||||
}
|
||||
return uint64(offset + int64(size)), nil
|
||||
}
|
||||
|
||||
// validateMessage walks an encoded message to check it parses, which is all the
|
||||
// generated unmarshaller would do for one whose fields are scalars.
|
||||
func validateMessage(b []byte) error {
|
||||
for len(b) > 0 {
|
||||
num, typ, tagLen := protowire.ConsumeTag(b)
|
||||
if tagLen < 0 {
|
||||
return protowire.ParseError(tagLen)
|
||||
}
|
||||
valLen := protowire.ConsumeFieldValue(num, typ, b[tagLen:])
|
||||
if valLen < 0 {
|
||||
return protowire.ParseError(valLen)
|
||||
}
|
||||
b = b[tagLen+valLen:]
|
||||
}
|
||||
return nil
|
||||
}
|
||||
@@ -0,0 +1,444 @@
|
||||
package filer
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"math/rand"
|
||||
"os"
|
||||
"reflect"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
"google.golang.org/protobuf/encoding/protowire"
|
||||
"google.golang.org/protobuf/proto"
|
||||
"google.golang.org/protobuf/reflect/protoreflect"
|
||||
)
|
||||
|
||||
// decodeBothWays round-trips entry and returns what each decoder made of it.
|
||||
func decodeBothWays(t *testing.T, entry *Entry) (full, attrsOnly Entry) {
|
||||
t.Helper()
|
||||
blob, err := entry.EncodeAttributesAndChunks()
|
||||
if err != nil {
|
||||
t.Fatalf("encode: %v", err)
|
||||
}
|
||||
full.FullPath = entry.FullPath
|
||||
if err := full.DecodeAttributesAndChunks(blob); err != nil {
|
||||
t.Fatalf("full decode: %v", err)
|
||||
}
|
||||
attrsOnly.FullPath = entry.FullPath
|
||||
if err := attrsOnly.DecodeAttributesOnly(blob); err != nil {
|
||||
t.Fatalf("attributes-only decode: %v", err)
|
||||
}
|
||||
return full, attrsOnly
|
||||
}
|
||||
|
||||
// assertSameButChunks checks the two decoders agree on everything a listing
|
||||
// reads. Chunks are the deliberate exception; size is not, because an entry can
|
||||
// carry a zero FileSize and take its size from the chunks.
|
||||
func assertSameButChunks(t *testing.T, full, attrsOnly Entry) {
|
||||
t.Helper()
|
||||
if attrsOnly.Chunks != nil {
|
||||
t.Errorf("attributes-only decode built %d chunks, want none", len(attrsOnly.Chunks))
|
||||
}
|
||||
if full.FileSize != attrsOnly.FileSize {
|
||||
t.Errorf("FileSize = %d, want %d", attrsOnly.FileSize, full.FileSize)
|
||||
}
|
||||
if full.Size() != attrsOnly.Size() {
|
||||
t.Errorf("Size() = %d, want %d", attrsOnly.Size(), full.Size())
|
||||
}
|
||||
if !reflect.DeepEqual(full.Attr, attrsOnly.Attr) {
|
||||
t.Errorf("Attr = %+v, want %+v", attrsOnly.Attr, full.Attr)
|
||||
}
|
||||
if string(full.HardLinkId) != string(attrsOnly.HardLinkId) {
|
||||
t.Errorf("HardLinkId = %x, want %x", attrsOnly.HardLinkId, full.HardLinkId)
|
||||
}
|
||||
if full.HardLinkCounter != attrsOnly.HardLinkCounter {
|
||||
t.Errorf("HardLinkCounter = %d, want %d", attrsOnly.HardLinkCounter, full.HardLinkCounter)
|
||||
}
|
||||
if string(full.Content) != string(attrsOnly.Content) {
|
||||
t.Errorf("Content = %q, want %q", attrsOnly.Content, full.Content)
|
||||
}
|
||||
if full.Quota != attrsOnly.Quota {
|
||||
t.Errorf("Quota = %d, want %d", attrsOnly.Quota, full.Quota)
|
||||
}
|
||||
if full.WORMEnforcedAtTsNs != attrsOnly.WORMEnforcedAtTsNs {
|
||||
t.Errorf("WORMEnforcedAtTsNs = %d, want %d", attrsOnly.WORMEnforcedAtTsNs, full.WORMEnforcedAtTsNs)
|
||||
}
|
||||
if len(full.Extended) != len(attrsOnly.Extended) {
|
||||
t.Errorf("Extended has %d keys, want %d", len(attrsOnly.Extended), len(full.Extended))
|
||||
}
|
||||
for k, v := range full.Extended {
|
||||
if string(attrsOnly.Extended[k]) != string(v) {
|
||||
t.Errorf("Extended[%q] = %q, want %q", k, attrsOnly.Extended[k], v)
|
||||
}
|
||||
}
|
||||
if (full.Remote == nil) != (attrsOnly.Remote == nil) {
|
||||
t.Errorf("Remote presence differs: %v vs %v", attrsOnly.Remote != nil, full.Remote != nil)
|
||||
} else if full.Remote != nil && full.Remote.RemoteSize != attrsOnly.Remote.RemoteSize {
|
||||
t.Errorf("Remote.RemoteSize = %d, want %d", attrsOnly.Remote.RemoteSize, full.Remote.RemoteSize)
|
||||
}
|
||||
}
|
||||
|
||||
func chunkAt(offset int64, size uint64, i int) *filer_pb.FileChunk {
|
||||
return &filer_pb.FileChunk{
|
||||
FileId: fmt.Sprintf("3,01637037d6%04d", i),
|
||||
Offset: offset,
|
||||
Size: size,
|
||||
ModifiedTsNs: int64(1700000000+i) * 1e9,
|
||||
ETag: "1a2b3c4d5e6f7890",
|
||||
Fid: &filer_pb.FileId{VolumeId: uint32(3 + i), FileKey: uint64(i), Cookie: 0x1637037d},
|
||||
CipherKey: []byte{1, 2, 3, 4},
|
||||
}
|
||||
}
|
||||
|
||||
func TestDecodeAttributesOnlyMatchesFullDecode(t *testing.T) {
|
||||
now := time.Unix(1700000000, 123456789)
|
||||
|
||||
cases := []struct {
|
||||
name string
|
||||
entry *Entry
|
||||
}{
|
||||
{"no chunks", &Entry{
|
||||
FullPath: util.FullPath("/d/plain"),
|
||||
Attr: Attr{Mode: 0o644, Mtime: now, Crtime: now, Ctime: now, Uid: 99, Gid: 100, FileSize: 12},
|
||||
}},
|
||||
{"directory", &Entry{
|
||||
FullPath: util.FullPath("/d/sub"),
|
||||
Attr: Attr{Mode: os.ModeDir | 0o755, Mtime: now, Crtime: now, Uid: 99, Gid: 100},
|
||||
}},
|
||||
{"one chunk", &Entry{
|
||||
FullPath: util.FullPath("/d/one"),
|
||||
Attr: Attr{Mode: 0o644, Mtime: now, Crtime: now, Uid: 99, Gid: 100, FileSize: 4 << 20},
|
||||
Chunks: []*filer_pb.FileChunk{chunkAt(0, 4<<20, 0)},
|
||||
}},
|
||||
// The S3 copy and multipart paths deliberately store a zero FileSize and
|
||||
// let the chunks define it, so this is the case that forces the extent
|
||||
// walk rather than just skipping the field.
|
||||
{"zero FileSize, size comes from chunks", &Entry{
|
||||
FullPath: util.FullPath("/d/zerosize"),
|
||||
Attr: Attr{Mode: 0o644, Mtime: now, Crtime: now, Uid: 99, Gid: 100, FileSize: 0},
|
||||
Chunks: []*filer_pb.FileChunk{
|
||||
chunkAt(0, 4<<20, 0), chunkAt(4<<20, 4<<20, 1), chunkAt(8<<20, 1234, 2),
|
||||
},
|
||||
}},
|
||||
{"chunks out of order", &Entry{
|
||||
FullPath: util.FullPath("/d/unordered"),
|
||||
Attr: Attr{Mode: 0o644, Mtime: now, Crtime: now, Uid: 99, Gid: 100},
|
||||
Chunks: []*filer_pb.FileChunk{
|
||||
chunkAt(8<<20, 99, 0), chunkAt(0, 4<<20, 1), chunkAt(4<<20, 4<<20, 2),
|
||||
},
|
||||
}},
|
||||
{"stored FileSize larger than chunks", &Entry{
|
||||
FullPath: util.FullPath("/d/sparse"),
|
||||
Attr: Attr{Mode: 0o644, Mtime: now, Crtime: now, Uid: 99, Gid: 100, FileSize: 1 << 30},
|
||||
Chunks: []*filer_pb.FileChunk{chunkAt(0, 16, 0)},
|
||||
}},
|
||||
{"symlink", &Entry{
|
||||
FullPath: util.FullPath("/d/link"),
|
||||
Attr: Attr{Mode: os.ModeSymlink | 0o777, Mtime: now, Crtime: now, SymlinkTarget: "../target"},
|
||||
}},
|
||||
{"hard link", &Entry{
|
||||
FullPath: util.FullPath("/d/hard"),
|
||||
Attr: Attr{Mode: 0o644, Mtime: now, Crtime: now},
|
||||
HardLinkId: HardLinkId([]byte{9, 8, 7, 6}),
|
||||
HardLinkCounter: 3,
|
||||
}},
|
||||
{"inline content", &Entry{
|
||||
FullPath: util.FullPath("/d/inline"),
|
||||
Attr: Attr{Mode: 0o644, Mtime: now, Crtime: now},
|
||||
Content: []byte("hello world"),
|
||||
}},
|
||||
{"extended attributes", &Entry{
|
||||
FullPath: util.FullPath("/d/xattr"),
|
||||
Attr: Attr{Mode: 0o644, Mtime: now, Crtime: now},
|
||||
Extended: map[string][]byte{"a": []byte("1"), "b": []byte("2"), "Seaweed-X": []byte("y")},
|
||||
}},
|
||||
{"remote entry", &Entry{
|
||||
FullPath: util.FullPath("/d/remote"),
|
||||
Attr: Attr{Mode: 0o644, Mtime: now, Crtime: now, FileSize: 5},
|
||||
Remote: &filer_pb.RemoteEntry{RemoteSize: 4096, RemoteMtime: now.Unix() + 60, StorageName: "s3"},
|
||||
}},
|
||||
{"quota and worm", &Entry{
|
||||
FullPath: util.FullPath("/d/bucket"),
|
||||
Attr: Attr{Mode: os.ModeDir | 0o755, Mtime: now, Crtime: now},
|
||||
Quota: 1 << 40,
|
||||
WORMEnforcedAtTsNs: now.UnixNano(),
|
||||
}},
|
||||
}
|
||||
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
full, attrsOnly := decodeBothWays(t, tc.entry)
|
||||
assertSameButChunks(t, full, attrsOnly)
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestDecodeAttributesOnlyRandomEntries fuzzes the field combinations, since the
|
||||
// decoder walks the wire format by hand and has to stay in step with the
|
||||
// generated one as filer.proto grows.
|
||||
func TestDecodeAttributesOnlyRandomEntries(t *testing.T) {
|
||||
rnd := rand.New(rand.NewSource(1))
|
||||
for i := 0; i < 2000; i++ {
|
||||
now := time.Unix(1600000000+rnd.Int63n(1e8), rnd.Int63n(1e9))
|
||||
entry := &Entry{
|
||||
FullPath: util.FullPath(fmt.Sprintf("/d/f%d", i)),
|
||||
Attr: Attr{
|
||||
Mode: os.FileMode(rnd.Intn(0o777)),
|
||||
Mtime: now, Crtime: now, Ctime: now,
|
||||
Uid: uint32(rnd.Intn(70000)), Gid: uint32(rnd.Intn(70000)),
|
||||
FileSize: uint64(rnd.Int63n(1 << 34)),
|
||||
Inode: rnd.Uint64(),
|
||||
Rdev: uint32(rnd.Intn(1 << 20)),
|
||||
TtlSec: int32(rnd.Intn(1000)),
|
||||
},
|
||||
}
|
||||
if rnd.Intn(2) == 0 {
|
||||
entry.Attr.FileSize = 0
|
||||
}
|
||||
if rnd.Intn(4) == 0 {
|
||||
entry.Content = make([]byte, rnd.Intn(64))
|
||||
rnd.Read(entry.Content)
|
||||
}
|
||||
if rnd.Intn(4) == 0 {
|
||||
entry.Extended = map[string][]byte{}
|
||||
for k := 0; k < rnd.Intn(4); k++ {
|
||||
entry.Extended[fmt.Sprintf("k%d", k)] = []byte(fmt.Sprintf("v%d", rnd.Intn(1000)))
|
||||
}
|
||||
}
|
||||
if rnd.Intn(8) == 0 {
|
||||
entry.Remote = &filer_pb.RemoteEntry{RemoteSize: rnd.Int63n(1 << 30), RemoteMtime: now.Unix() + int64(rnd.Intn(120)) - 60}
|
||||
}
|
||||
for c := 0; c < rnd.Intn(20); c++ {
|
||||
entry.Chunks = append(entry.Chunks, chunkAt(rnd.Int63n(1<<30), uint64(rnd.Int63n(1<<22)), c))
|
||||
}
|
||||
full, attrsOnly := decodeBothWays(t, entry)
|
||||
assertSameButChunks(t, full, attrsOnly)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDecodeAttributesOnlyRejectsGarbage(t *testing.T) {
|
||||
var entry Entry
|
||||
entry.FullPath = util.FullPath("/d/bad")
|
||||
if err := entry.DecodeAttributesOnly([]byte{0xff, 0xff, 0xff, 0xff}); err == nil {
|
||||
t.Fatal("expected an error for a malformed blob")
|
||||
}
|
||||
}
|
||||
|
||||
func BenchmarkDecode(b *testing.B) {
|
||||
now := time.Unix(1700000000, 0)
|
||||
for _, n := range []int{0, 1, 4, 16, 64} {
|
||||
entry := &Entry{
|
||||
FullPath: util.FullPath("/images/image-00000001.jpg"),
|
||||
Attr: Attr{Mode: 0o644, Mtime: now, Crtime: now, Ctime: now, Uid: 99, Gid: 100, FileSize: uint64(n) * 4 << 20},
|
||||
}
|
||||
for i := 0; i < n; i++ {
|
||||
entry.Chunks = append(entry.Chunks, chunkAt(int64(i)*4<<20, 4<<20, i))
|
||||
}
|
||||
blob, err := entry.EncodeAttributesAndChunks()
|
||||
if err != nil {
|
||||
b.Fatal(err)
|
||||
}
|
||||
b.Run(fmt.Sprintf("chunks=%d/decoder=full", n), func(b *testing.B) {
|
||||
b.ReportAllocs()
|
||||
for i := 0; i < b.N; i++ {
|
||||
var out Entry
|
||||
if err := out.DecodeAttributesAndChunks(blob); err != nil {
|
||||
b.Fatal(err)
|
||||
}
|
||||
}
|
||||
})
|
||||
b.Run(fmt.Sprintf("chunks=%d/decoder=attrsonly", n), func(b *testing.B) {
|
||||
b.ReportAllocs()
|
||||
for i := 0; i < b.N; i++ {
|
||||
var out Entry
|
||||
if err := out.DecodeAttributesOnly(blob); err != nil {
|
||||
b.Fatal(err)
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestProtoEntryWithoutChunksKeepsSize covers the wire contract the filer's
|
||||
// omit_chunks listing relies on: the size a client needs survives in the
|
||||
// attributes once the chunk list is dropped, including for the entries that
|
||||
// store a zero FileSize and take their size from the chunks.
|
||||
func TestProtoEntryWithoutChunksKeepsSize(t *testing.T) {
|
||||
now := time.Unix(1700000000, 0)
|
||||
for _, storedSize := range []uint64{0, 7, 1 << 30} {
|
||||
stored := &Entry{
|
||||
FullPath: util.FullPath("/d/obj"),
|
||||
Attr: Attr{Mode: 0o644, Mtime: now, Crtime: now, FileSize: storedSize},
|
||||
Chunks: []*filer_pb.FileChunk{chunkAt(0, 4<<20, 0), chunkAt(4<<20, 1234, 1)},
|
||||
}
|
||||
blob, err := stored.EncodeAttributesAndChunks()
|
||||
if err != nil {
|
||||
t.Fatalf("encode: %v", err)
|
||||
}
|
||||
// What the filer holds after reading the entry out of its store.
|
||||
var loaded Entry
|
||||
loaded.FullPath = stored.FullPath
|
||||
if err := loaded.DecodeAttributesAndChunks(blob); err != nil {
|
||||
t.Fatalf("decode: %v", err)
|
||||
}
|
||||
want := FileSize(loaded.ToProtoEntry())
|
||||
|
||||
pbEntry := loaded.ToProtoEntry()
|
||||
pbEntry.Chunks = nil
|
||||
if got := FileSize(pbEntry); got != want {
|
||||
t.Errorf("stored FileSize %d: size over the wire = %d, want %d", storedSize, got, want)
|
||||
}
|
||||
if got := FromPbEntry("/d", pbEntry).Size(); got != want {
|
||||
t.Errorf("stored FileSize %d: size at the client = %d, want %d", storedSize, got, want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestChunkValidationCoversEveryField keeps the hand-rolled chunk walk honest as
|
||||
// filer.proto grows. The generated unmarshaller never sees the chunk bytes, so
|
||||
// every FileChunk field whose contents it would have checked -- a submessage, or
|
||||
// a proto3 string's UTF-8 -- has to be listed for the walk to check instead.
|
||||
func TestChunkValidationCoversEveryField(t *testing.T) {
|
||||
fields := (&filer_pb.FileChunk{}).ProtoReflect().Descriptor().Fields()
|
||||
for i := 0; i < fields.Len(); i++ {
|
||||
f := fields.Get(i)
|
||||
switch f.Kind() {
|
||||
case protoreflect.MessageKind, protoreflect.GroupKind:
|
||||
if !fileChunkMessageFields[protowire.Number(f.Number())] {
|
||||
t.Errorf("FileChunk.%s (field %d) is a message but is not in fileChunkMessageFields, so a corrupt one would pass the listing decoder and fail the full one", f.Name(), f.Number())
|
||||
}
|
||||
case protoreflect.StringKind:
|
||||
if !fileChunkStringFields[protowire.Number(f.Number())] {
|
||||
t.Errorf("FileChunk.%s (field %d) is a string but is not in fileChunkStringFields, so invalid UTF-8 would pass the listing decoder and fail the full one", f.Name(), f.Number())
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// corruptChunkBlobs builds entry blobs whose chunk bytes are damaged in ways the
|
||||
// generated unmarshaller rejects.
|
||||
func corruptChunkBlobs(t *testing.T) map[string][]byte {
|
||||
t.Helper()
|
||||
now := time.Unix(1700000000, 0)
|
||||
base := func() *filer_pb.FileChunk {
|
||||
return &filer_pb.FileChunk{Offset: 0, Size: 1024, Fid: &filer_pb.FileId{VolumeId: 3, FileKey: 7, Cookie: 9}}
|
||||
}
|
||||
blobFor := func(mangle func(raw []byte) []byte) []byte {
|
||||
chunk := base()
|
||||
chunkBytes, err := proto.Marshal(chunk)
|
||||
if err != nil {
|
||||
t.Fatalf("marshal chunk: %v", err)
|
||||
}
|
||||
chunkBytes = mangle(chunkBytes)
|
||||
var blob []byte
|
||||
blob = protowire.AppendTag(blob, entryChunksField, protowire.BytesType)
|
||||
blob = protowire.AppendBytes(blob, chunkBytes)
|
||||
attrs, err := proto.Marshal(&filer_pb.FuseAttributes{FileSize: 0, Mtime: now.Unix(), FileMode: 0o644})
|
||||
if err != nil {
|
||||
t.Fatalf("marshal attrs: %v", err)
|
||||
}
|
||||
blob = protowire.AppendTag(blob, 4, protowire.BytesType)
|
||||
return protowire.AppendBytes(blob, attrs)
|
||||
}
|
||||
|
||||
out := map[string][]byte{}
|
||||
// The exact probe from review: a nested fid whose payload is not a message.
|
||||
out["corrupt nested fid"] = blobFor(func(raw []byte) []byte {
|
||||
var b []byte
|
||||
b = protowire.AppendTag(b, 2, protowire.VarintType)
|
||||
b = protowire.AppendVarint(b, 0)
|
||||
b = protowire.AppendTag(b, 3, protowire.VarintType)
|
||||
b = protowire.AppendVarint(b, 1024)
|
||||
b = protowire.AppendTag(b, 7, protowire.BytesType)
|
||||
return protowire.AppendBytes(b, []byte{0xff, 0xff, 0xff, 0xff})
|
||||
})
|
||||
out["invalid utf8 in file_id"] = blobFor(func(raw []byte) []byte {
|
||||
var b []byte
|
||||
b = protowire.AppendTag(b, 1, protowire.BytesType)
|
||||
b = protowire.AppendBytes(b, []byte{0xff, 0xfe, 0xfd})
|
||||
b = protowire.AppendTag(b, 3, protowire.VarintType)
|
||||
return protowire.AppendVarint(b, 1024)
|
||||
})
|
||||
out["truncated chunk"] = blobFor(func(raw []byte) []byte { return raw[:len(raw)-1] })
|
||||
return out
|
||||
}
|
||||
|
||||
// TestDecodeAttributesOnlyRejectsWhatFullDecodeRejects is the invariant that
|
||||
// keeps a listing from showing a file that cannot then be opened: the fast path
|
||||
// must never accept a blob the full decoder turns away.
|
||||
func TestDecodeAttributesOnlyRejectsWhatFullDecodeRejects(t *testing.T) {
|
||||
for name, blob := range corruptChunkBlobs(t) {
|
||||
t.Run(name, func(t *testing.T) {
|
||||
var full, attrsOnly Entry
|
||||
full.FullPath = util.FullPath("/d/corrupt")
|
||||
attrsOnly.FullPath = full.FullPath
|
||||
fullErr := full.DecodeAttributesAndChunks(blob)
|
||||
attrsErr := attrsOnly.DecodeAttributesOnly(blob)
|
||||
if fullErr == nil {
|
||||
t.Skip("full decoder accepts this blob, nothing to match")
|
||||
}
|
||||
if attrsErr == nil {
|
||||
t.Errorf("full decode rejected the blob (%v) but attributes-only accepted it with FileSize=%d", fullErr, attrsOnly.FileSize)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestDecodeAttributesOnlyManifestChunk pins that a manifest chunk needs no
|
||||
// resolving: it carries the offset and size of the range it stands for, and
|
||||
// TotalSize does not resolve it either, so both decoders see one number.
|
||||
func TestDecodeAttributesOnlyManifestChunk(t *testing.T) {
|
||||
now := time.Unix(1700000000, 0)
|
||||
manifest := chunkAt(0, 64<<20, 0)
|
||||
manifest.IsChunkManifest = true
|
||||
entry := &Entry{
|
||||
FullPath: util.FullPath("/d/big"),
|
||||
// Zero stored size, so the manifest's extent is the only source.
|
||||
Attr: Attr{Mode: 0o644, Mtime: now, Crtime: now, FileSize: 0},
|
||||
Chunks: []*filer_pb.FileChunk{manifest},
|
||||
}
|
||||
full, attrsOnly := decodeBothWays(t, entry)
|
||||
assertSameButChunks(t, full, attrsOnly)
|
||||
if attrsOnly.FileSize != 64<<20 {
|
||||
t.Errorf("FileSize = %d, want %d from the manifest extent", attrsOnly.FileSize, 64<<20)
|
||||
}
|
||||
}
|
||||
|
||||
// TestDecodeAttributesOnlyFieldBeforeChunks exercises the prefix copy, which no
|
||||
// other case reaches: EncodeAttributesAndChunks emits chunks before every field
|
||||
// the other tests set, so `dropping` always turns on at pos 0 there.
|
||||
func TestDecodeAttributesOnlyFieldBeforeChunks(t *testing.T) {
|
||||
now := time.Unix(1700000000, 0)
|
||||
attrs, err := proto.Marshal(&filer_pb.FuseAttributes{FileSize: 0, Mtime: now.Unix(), FileMode: 0o755})
|
||||
if err != nil {
|
||||
t.Fatalf("marshal attrs: %v", err)
|
||||
}
|
||||
chunkBytes, err := proto.Marshal(chunkAt(0, 4<<20, 0))
|
||||
if err != nil {
|
||||
t.Fatalf("marshal chunk: %v", err)
|
||||
}
|
||||
// is_directory (2) ahead of chunks (3), so the walk has a prefix to copy.
|
||||
var blob []byte
|
||||
blob = protowire.AppendTag(blob, 2, protowire.VarintType)
|
||||
blob = protowire.AppendVarint(blob, 1)
|
||||
blob = protowire.AppendTag(blob, entryChunksField, protowire.BytesType)
|
||||
blob = protowire.AppendBytes(blob, chunkBytes)
|
||||
blob = protowire.AppendTag(blob, 4, protowire.BytesType)
|
||||
blob = protowire.AppendBytes(blob, attrs)
|
||||
|
||||
var full, attrsOnly Entry
|
||||
full.FullPath = util.FullPath("/d/dirwithchunks")
|
||||
attrsOnly.FullPath = full.FullPath
|
||||
if err := full.DecodeAttributesAndChunks(blob); err != nil {
|
||||
t.Fatalf("full decode: %v", err)
|
||||
}
|
||||
if err := attrsOnly.DecodeAttributesOnly(blob); err != nil {
|
||||
t.Fatalf("attributes-only decode: %v", err)
|
||||
}
|
||||
assertSameButChunks(t, full, attrsOnly)
|
||||
if attrsOnly.FileSize != 4<<20 {
|
||||
t.Errorf("FileSize = %d, want %d", attrsOnly.FileSize, 4<<20)
|
||||
}
|
||||
}
|
||||
@@ -236,7 +236,10 @@ func doMaybeManifestize(saveFunc SaveDataAsChunkFunctionType, inputChunks []*fil
|
||||
for i := 0; i+mergeFactor <= len(dataChunks); i += mergeFactor {
|
||||
chunk, err := mergefn(saveFunc, dataChunks[i:i+mergeFactor])
|
||||
if err != nil {
|
||||
return dataChunks, err
|
||||
// Return the manifests already written plus the chunks not yet
|
||||
// wrapped: a complete, deletable representation of every byte, so
|
||||
// callers can clean up or keep a usable chunk list.
|
||||
return append(chunks, dataChunks[i:]...), err
|
||||
}
|
||||
chunks = append(chunks, chunk)
|
||||
remaining -= mergeFactor
|
||||
|
||||
@@ -77,6 +77,38 @@ func TestDoMaybeManifestize(t *testing.T) {
|
||||
actual, _ := doMaybeManifestize(nil, mtest.inputs, 2, mockMerge)
|
||||
assertEqualChunks(t, mtest.expected, actual)
|
||||
}
|
||||
}
|
||||
|
||||
// A mid-run merge failure must still return every chunk that exists: the
|
||||
// manifests already written plus the chunks not yet wrapped, so callers can
|
||||
// delete or keep a complete set.
|
||||
func TestDoMaybeManifestizePartialFailure(t *testing.T) {
|
||||
inputs := []*filer_pb.FileChunk{
|
||||
{FileId: "0", IsChunkManifest: true},
|
||||
{FileId: "1", IsChunkManifest: false},
|
||||
{FileId: "2", IsChunkManifest: false},
|
||||
{FileId: "3", IsChunkManifest: false},
|
||||
{FileId: "4", IsChunkManifest: false},
|
||||
}
|
||||
calls := 0
|
||||
failingMerge := func(saveFunc SaveDataAsChunkFunctionType, dataChunks []*filer_pb.FileChunk) (*filer_pb.FileChunk, error) {
|
||||
calls++
|
||||
if calls > 1 {
|
||||
return nil, fmt.Errorf("merge failed")
|
||||
}
|
||||
return mockMerge(saveFunc, dataChunks)
|
||||
}
|
||||
actual, err := doMaybeManifestize(nil, inputs, 2, failingMerge)
|
||||
if err == nil {
|
||||
t.Fatalf("doMaybeManifestize() expected an error")
|
||||
}
|
||||
expected := []*filer_pb.FileChunk{
|
||||
{FileId: "0", IsChunkManifest: true},
|
||||
{FileId: "12", IsChunkManifest: true},
|
||||
{FileId: "3", IsChunkManifest: false},
|
||||
{FileId: "4", IsChunkManifest: false},
|
||||
}
|
||||
assertEqualChunks(t, expected, actual)
|
||||
|
||||
}
|
||||
|
||||
|
||||
@@ -71,9 +71,7 @@ func (f *Filer) notifyUpdateEvent(ctx context.Context, oldEntry, newEntry *Entry
|
||||
sink.Record(event)
|
||||
}
|
||||
|
||||
// Trigger empty folder cleanup for local events
|
||||
// Remote events are handled via MetaAggregator.onMetadataChangeEvent
|
||||
f.triggerLocalEmptyFolderCleanup(oldEntry, newEntry)
|
||||
f.onMetadataChangeEvent(event)
|
||||
|
||||
return event
|
||||
}
|
||||
@@ -121,41 +119,6 @@ func (f *Filer) logMetaEvent(ctx context.Context, event *filer_pb.SubscribeMetad
|
||||
|
||||
}
|
||||
|
||||
// triggerLocalEmptyFolderCleanup triggers empty folder cleanup for local events
|
||||
// This is needed because onMetadataChangeEvent is only called for remote peer events
|
||||
func (f *Filer) triggerLocalEmptyFolderCleanup(oldEntry, newEntry *Entry) {
|
||||
if f.EmptyFolderCleaner == nil || !f.EmptyFolderCleaner.IsEnabled() {
|
||||
return
|
||||
}
|
||||
|
||||
eventTime := time.Now()
|
||||
|
||||
// Handle delete events (oldEntry exists, newEntry is nil)
|
||||
if oldEntry != nil && newEntry == nil {
|
||||
dir, name := oldEntry.FullPath.DirAndName()
|
||||
f.EmptyFolderCleaner.OnDeleteEvent(dir, name, oldEntry.IsDirectory(), eventTime)
|
||||
}
|
||||
|
||||
// Handle create events (oldEntry is nil, newEntry exists)
|
||||
if oldEntry == nil && newEntry != nil {
|
||||
dir, name := newEntry.FullPath.DirAndName()
|
||||
f.EmptyFolderCleaner.OnCreateEvent(dir, name, newEntry.IsDirectory())
|
||||
}
|
||||
|
||||
// Handle rename/move events (both exist but paths differ)
|
||||
if oldEntry != nil && newEntry != nil {
|
||||
oldDir, oldName := oldEntry.FullPath.DirAndName()
|
||||
newDir, newName := newEntry.FullPath.DirAndName()
|
||||
|
||||
if oldDir != newDir || oldName != newName {
|
||||
// Treat old location as delete
|
||||
f.EmptyFolderCleaner.OnDeleteEvent(oldDir, oldName, oldEntry.IsDirectory(), eventTime)
|
||||
// Treat new location as create
|
||||
f.EmptyFolderCleaner.OnCreateEvent(newDir, newName, newEntry.IsDirectory())
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// metadataLogUploadLimit is the piece size a metadata log flush starts with. A
|
||||
// volume server refuses anything over its -fileSizeLimitMB (256 MB by default),
|
||||
// and a single oversized event — a CreateEntry carrying a large inline Content,
|
||||
|
||||
@@ -237,7 +237,7 @@ func (store *LevelDBStore) ListDirectoryPrefixedEntries(ctx context.Context, dir
|
||||
entry := &filer.Entry{
|
||||
FullPath: weed_util.NewFullPath(string(dirPath), fileName),
|
||||
}
|
||||
if decodeErr := entry.DecodeAttributesAndChunks(weed_util.MaybeDecompressData(iter.Value())); decodeErr != nil {
|
||||
if decodeErr := filer.DecodeListedEntry(ctx, entry, weed_util.MaybeDecompressData(iter.Value())); decodeErr != nil {
|
||||
err = decodeErr
|
||||
glog.V(0).InfofCtx(ctx, "list %s : %v", entry.FullPath, err)
|
||||
break
|
||||
|
||||
@@ -53,9 +53,10 @@ func (ma *MetaAggregator) OnPeerUpdate(update *master_pb.ClusterNodeUpdate, star
|
||||
|
||||
address := pb.ServerAddress(update.Address)
|
||||
if update.IsAdd {
|
||||
// cancel previous subscription if any
|
||||
if prevChan, found := ma.peerChans[address]; found {
|
||||
close(prevChan)
|
||||
// the peer is already followed, restarting would only lose the events
|
||||
// in between
|
||||
if _, found := ma.peerChans[address]; found {
|
||||
return
|
||||
}
|
||||
stopChan := make(chan struct{})
|
||||
ma.peerChans[address] = stopChan
|
||||
|
||||
@@ -0,0 +1,36 @@
|
||||
package filer
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
|
||||
)
|
||||
|
||||
func TestOnPeerUpdateRepeatedAdd(t *testing.T) {
|
||||
peer := pb.ServerAddress("127.0.0.1:1")
|
||||
ma := NewMetaAggregator(nil, pb.ServerAddress("127.0.0.1:2"), nil)
|
||||
|
||||
add := &master_pb.ClusterNodeUpdate{Address: string(peer), IsAdd: true}
|
||||
ma.OnPeerUpdate(add, time.Now())
|
||||
first, found := ma.peerChans[peer]
|
||||
if !found {
|
||||
t.Fatal("expecting a subscription after the first add")
|
||||
}
|
||||
|
||||
ma.OnPeerUpdate(add, time.Now())
|
||||
if ma.peerChans[peer] != first {
|
||||
t.Fatal("expecting the same subscription after a repeated add")
|
||||
}
|
||||
select {
|
||||
case <-first:
|
||||
t.Fatal("expecting the subscription to stay alive after a repeated add")
|
||||
default:
|
||||
}
|
||||
|
||||
ma.OnPeerUpdate(&master_pb.ClusterNodeUpdate{Address: string(peer)}, time.Now())
|
||||
if _, found := ma.peerChans[peer]; found {
|
||||
t.Fatal("expecting the subscription to be removed")
|
||||
}
|
||||
}
|
||||
@@ -1,10 +1,12 @@
|
||||
package filer
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util/log_buffer"
|
||||
)
|
||||
@@ -41,3 +43,45 @@ func TestNotifyUpdateEventRecordsRequestMetadataEvent(t *testing.T) {
|
||||
t.Fatal("expected event timestamp to be set")
|
||||
}
|
||||
}
|
||||
|
||||
func TestNotifyUpdateEventReloadsLocalFilerConfiguration(t *testing.T) {
|
||||
f := &Filer{
|
||||
Signature: 42,
|
||||
FilerConf: NewFilerConf(),
|
||||
LocalMetaLogBuffer: log_buffer.NewLogBuffer(
|
||||
"test",
|
||||
time.Hour,
|
||||
func(*log_buffer.LogBuffer, time.Time, time.Time, []byte, int64, int64) {},
|
||||
nil,
|
||||
nil,
|
||||
),
|
||||
}
|
||||
|
||||
updatedConf := NewFilerConf()
|
||||
if err := updatedConf.SetLocationConf(&filer_pb.FilerConf_PathConf{
|
||||
LocationPrefix: "/data/",
|
||||
Collection: "hot",
|
||||
}); err != nil {
|
||||
t.Fatalf("set location conf: %v", err)
|
||||
}
|
||||
var content bytes.Buffer
|
||||
if err := updatedConf.ToText(&content); err != nil {
|
||||
t.Fatalf("serialize filer conf: %v", err)
|
||||
}
|
||||
|
||||
f.NotifyUpdateEvent(context.Background(), nil, &Entry{
|
||||
FullPath: util.NewFullPath(DirectoryEtcSeaweedFS, FilerConfName),
|
||||
Attr: Attr{
|
||||
FileSize: uint64(content.Len()),
|
||||
},
|
||||
Content: content.Bytes(),
|
||||
}, false, false, nil)
|
||||
|
||||
locConf, found := f.FilerConf.GetLocationConf("/data/")
|
||||
if !found {
|
||||
t.Fatal("expected local filer configuration to reload /data/ rule")
|
||||
}
|
||||
if locConf.Collection != "hot" {
|
||||
t.Fatalf("collection = %q, want hot", locConf.Collection)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,93 @@
|
||||
// Package format maps the internal structure of container file formats onto
|
||||
// storage chunk boundaries.
|
||||
//
|
||||
// An adapter translates one format into three things the core understands: a
|
||||
// list of extent sizes, an alignment quantum, and an opaque payload. Adapters
|
||||
// never see chunks, file ids, or authorization; the core never learns what an
|
||||
// MPEG-TS packet or a parquet row group is.
|
||||
package format
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"io"
|
||||
"net/url"
|
||||
)
|
||||
|
||||
// LayoutKey is the filer Extended key holding the encoded Layout. The
|
||||
// x-seaweedfs- prefix keeps it out of HTTP response headers.
|
||||
const LayoutKey = "x-seaweedfs-format-layout"
|
||||
|
||||
// ViewParam is the query parameter selecting a format view on GET requests.
|
||||
// Adapters rendering self-referential URLs must use it.
|
||||
const ViewParam = "format.view"
|
||||
|
||||
// ErrNoSuchView reports that a view request addresses nothing servable; the
|
||||
// server answers 404.
|
||||
var ErrNoSuchView = errors.New("no such view")
|
||||
|
||||
// Layout describes how a file's structure maps to byte extents.
|
||||
type Layout struct {
|
||||
Format string // adapter name
|
||||
ExtentSizes []int64 // extent lengths in file order; they sum to the file size
|
||||
Align int64 // quantum for cutting inside an oversized extent; 1 cuts anywhere
|
||||
Payload []byte // adapter-owned metadata, opaque to the core
|
||||
}
|
||||
|
||||
// Hint carries the cheap identification signals available to Sniff.
|
||||
type Hint struct {
|
||||
Name string
|
||||
ContentType string
|
||||
Size int64
|
||||
Head []byte
|
||||
Tail []byte
|
||||
}
|
||||
|
||||
// Format is the mandatory adapter identity. Capabilities beyond it are
|
||||
// discovered by type assertion.
|
||||
type Format interface {
|
||||
Name() string
|
||||
}
|
||||
|
||||
// Sniffer cheaply recognizes the format from identification signals. Repack
|
||||
// gates on it before parsing, and policy-driven detection will rely on it;
|
||||
// ingest-only adapters, whose files carry their layout from birth, skip it.
|
||||
type Sniffer interface {
|
||||
Sniff(h Hint) bool
|
||||
}
|
||||
|
||||
// Indexer derives a Layout from the complete stored bytes.
|
||||
type Indexer interface {
|
||||
Index(ctx context.Context, r io.ReaderAt, size int64) (*Layout, error)
|
||||
}
|
||||
|
||||
// SidecarIndexer derives a Layout from an external index document supplied at
|
||||
// ingest, before the media bytes arrive.
|
||||
type SidecarIndexer interface {
|
||||
IndexSidecar(sidecar []byte) (*Layout, error)
|
||||
}
|
||||
|
||||
// Object is everything a Viewer may know about the file it serves.
|
||||
type Object struct {
|
||||
Name string
|
||||
Size int64
|
||||
Layout *Layout
|
||||
}
|
||||
|
||||
// ViewRequest carries the request parameters of a ?view= request.
|
||||
type ViewRequest struct {
|
||||
Query url.Values
|
||||
}
|
||||
|
||||
// ViewPlan tells the server what to serve. The server executes it on the
|
||||
// normal streaming path; adapters stay pure functions of request and layout.
|
||||
type ViewPlan struct {
|
||||
ContentType string
|
||||
Body []byte // rendered document; when nil, stream Extent instead
|
||||
Extent int
|
||||
}
|
||||
|
||||
// Viewer answers ?view= requests.
|
||||
type Viewer interface {
|
||||
View(req ViewRequest, obj Object) (*ViewPlan, error)
|
||||
}
|
||||
@@ -0,0 +1,64 @@
|
||||
// Package formattest is the conformance kit format adapters must pass:
|
||||
// indexers parse attacker-controlled bytes inside a storage daemon, so they
|
||||
// must never panic and every layout they accept must validate.
|
||||
package formattest
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"testing"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/format"
|
||||
)
|
||||
|
||||
// IndexTruncations feeds progressively truncated copies of a valid file to the
|
||||
// indexer. Any outcome is acceptable except a panic or an invalid layout.
|
||||
func IndexTruncations(t *testing.T, indexer format.Indexer, data []byte) {
|
||||
t.Helper()
|
||||
for i := 0; i <= 16; i++ {
|
||||
size := int64(len(data) * i / 16)
|
||||
layout, err := indexer.Index(context.Background(), bytes.NewReader(data[:size]), size)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
if validateErr := layout.Validate(size); validateErr != nil {
|
||||
t.Fatalf("Index() at %d/%d bytes returned an invalid layout: %v", size, len(data), validateErr)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// SidecarTruncations does the same for sidecar index documents.
|
||||
func SidecarTruncations(t *testing.T, indexer format.SidecarIndexer, sidecar []byte) {
|
||||
t.Helper()
|
||||
for i := 0; i <= 16; i++ {
|
||||
layout, err := indexer.IndexSidecar(sidecar[:len(sidecar)*i/16])
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
if validateErr := layout.Validate(-1); validateErr != nil {
|
||||
t.Fatalf("IndexSidecar() at %d/16 returned an invalid layout: %v", i, validateErr)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// EncodeRoundTrip checks that a layout survives the persistence codec.
|
||||
func EncodeRoundTrip(t *testing.T, layout *format.Layout) {
|
||||
t.Helper()
|
||||
encoded, err := layout.Encode()
|
||||
if err != nil {
|
||||
t.Fatalf("Encode() error = %v", err)
|
||||
}
|
||||
decoded, err := format.DecodeLayout(encoded)
|
||||
if err != nil {
|
||||
t.Fatalf("DecodeLayout() error = %v", err)
|
||||
}
|
||||
if decoded.Format != layout.Format || decoded.Align != layout.Align ||
|
||||
len(decoded.ExtentSizes) != len(layout.ExtentSizes) || !bytes.Equal(decoded.Payload, layout.Payload) {
|
||||
t.Fatalf("decoded layout %+v differs from %+v", decoded, layout)
|
||||
}
|
||||
for i := range layout.ExtentSizes {
|
||||
if decoded.ExtentSizes[i] != layout.ExtentSizes[i] {
|
||||
t.Fatalf("extent %d = %d, want %d", i, decoded.ExtentSizes[i], layout.ExtentSizes[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,284 @@
|
||||
// Package hlsts adapts single-file HLS MPEG-TS VOD assets: the ingest sidecar
|
||||
// is an FFmpeg-style EXT-X-BYTERANGE media playlist, extents are its segments,
|
||||
// and the view serves a rewritten playlist with plain numbered segment URLs.
|
||||
package hlsts
|
||||
|
||||
import (
|
||||
"bufio"
|
||||
"bytes"
|
||||
"encoding/binary"
|
||||
"errors"
|
||||
"fmt"
|
||||
"math"
|
||||
"net/url"
|
||||
"strconv"
|
||||
"strings"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/format"
|
||||
)
|
||||
|
||||
const (
|
||||
FormatName = "hls-ts"
|
||||
// TSPacketSize is the fixed MPEG-TS packet size; segment boundaries and
|
||||
// interior chunk cuts land on packet boundaries.
|
||||
TSPacketSize = 188
|
||||
|
||||
tsSyncByte = 0x47
|
||||
|
||||
PlaylistContentType = "application/vnd.apple.mpegurl"
|
||||
MediaContentType = "video/MP2T"
|
||||
)
|
||||
|
||||
func init() {
|
||||
format.Register(Adapter{})
|
||||
}
|
||||
|
||||
type Adapter struct{}
|
||||
|
||||
var (
|
||||
_ format.Sniffer = Adapter{}
|
||||
_ format.SidecarIndexer = Adapter{}
|
||||
_ format.Viewer = Adapter{}
|
||||
)
|
||||
|
||||
func (Adapter) Name() string { return FormatName }
|
||||
|
||||
func (Adapter) Sniff(h format.Hint) bool {
|
||||
if len(h.Head) > TSPacketSize {
|
||||
return h.Head[0] == tsSyncByte && h.Head[TSPacketSize] == tsSyncByte
|
||||
}
|
||||
return len(h.Head) > 0 && h.Head[0] == tsSyncByte
|
||||
}
|
||||
|
||||
// playlistInfo is the adapter payload: what the generated playback playlist
|
||||
// needs beyond the extent sizes.
|
||||
type playlistInfo struct {
|
||||
TargetDuration int64
|
||||
MediaSequence int64
|
||||
DurationsMs []int64
|
||||
}
|
||||
|
||||
func (p *playlistInfo) encode() []byte {
|
||||
out := binary.AppendUvarint(nil, uint64(p.TargetDuration))
|
||||
out = binary.AppendUvarint(out, uint64(p.MediaSequence))
|
||||
for _, durationMs := range p.DurationsMs {
|
||||
out = binary.AppendUvarint(out, uint64(durationMs))
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
func decodePlaylistInfo(payload []byte, extentCount int) (*playlistInfo, error) {
|
||||
reader := bytes.NewReader(payload)
|
||||
target, err := binary.ReadUvarint(reader)
|
||||
if err != nil || target == 0 || target > math.MaxInt32 {
|
||||
return nil, fmt.Errorf("invalid hls-ts target duration")
|
||||
}
|
||||
// mirror the ingest bound: the last segment number is sequence+count-1
|
||||
sequence, err := binary.ReadUvarint(reader)
|
||||
if err != nil || sequence > math.MaxInt64-uint64(extentCount-1) {
|
||||
return nil, fmt.Errorf("invalid hls-ts media sequence")
|
||||
}
|
||||
info := &playlistInfo{TargetDuration: int64(target), MediaSequence: int64(sequence), DurationsMs: make([]int64, extentCount)}
|
||||
for i := range info.DurationsMs {
|
||||
durationMs, err := binary.ReadUvarint(reader)
|
||||
if err != nil || durationMs == 0 || durationMs > math.MaxInt32 {
|
||||
return nil, fmt.Errorf("invalid hls-ts segment %d duration", i)
|
||||
}
|
||||
info.DurationsMs[i] = int64(durationMs)
|
||||
}
|
||||
if reader.Len() != 0 {
|
||||
return nil, fmt.Errorf("hls-ts payload has trailing bytes")
|
||||
}
|
||||
return info, nil
|
||||
}
|
||||
|
||||
// IndexSidecar parses a VOD media playlist whose segments reference one shared
|
||||
// media URI through EXT-X-BYTERANGE. Playlist state the generated playback
|
||||
// playlist cannot reproduce is rejected.
|
||||
func (Adapter) IndexSidecar(sidecar []byte) (*format.Layout, error) {
|
||||
scanner := bufio.NewScanner(bytes.NewReader(sidecar))
|
||||
scanner.Buffer(make([]byte, 64*1024), 1<<20)
|
||||
|
||||
info := &playlistInfo{}
|
||||
var sizes []int64
|
||||
var pendingDuration int64 // ms; 0 = no EXTINF pending
|
||||
var pendingSize, pendingOffset int64
|
||||
var havePendingRange bool
|
||||
var expectedOffset int64
|
||||
var mediaURI string
|
||||
var sawHeader, sawEndList bool
|
||||
var maxDurationMs int64
|
||||
|
||||
for scanner.Scan() {
|
||||
line := strings.TrimSpace(scanner.Text())
|
||||
switch {
|
||||
case line == "":
|
||||
case line == "#EXTM3U":
|
||||
sawHeader = true
|
||||
case line == "#EXT-X-ENDLIST":
|
||||
sawEndList = true
|
||||
case strings.HasPrefix(line, "#EXT-X-KEY:"),
|
||||
line == "#EXT-X-DISCONTINUITY",
|
||||
strings.HasPrefix(line, "#EXT-X-DISCONTINUITY-SEQUENCE:"),
|
||||
strings.HasPrefix(line, "#EXT-X-MAP:"),
|
||||
line == "#EXT-X-GAP",
|
||||
line == "#EXT-X-I-FRAMES-ONLY":
|
||||
return nil, fmt.Errorf("%s is not supported by hls-ts ingest", strings.SplitN(line, ":", 2)[0])
|
||||
case strings.HasPrefix(line, "#EXT-X-TARGETDURATION:"):
|
||||
value := strings.TrimSpace(strings.TrimPrefix(line, "#EXT-X-TARGETDURATION:"))
|
||||
target, err := strconv.ParseInt(value, 10, 32)
|
||||
if err != nil || target <= 0 {
|
||||
return nil, fmt.Errorf("invalid EXT-X-TARGETDURATION %q", value)
|
||||
}
|
||||
info.TargetDuration = target
|
||||
case strings.HasPrefix(line, "#EXT-X-MEDIA-SEQUENCE:"):
|
||||
value := strings.TrimSpace(strings.TrimPrefix(line, "#EXT-X-MEDIA-SEQUENCE:"))
|
||||
sequence, err := strconv.ParseInt(value, 10, 64)
|
||||
if err != nil || sequence < 0 {
|
||||
return nil, fmt.Errorf("invalid EXT-X-MEDIA-SEQUENCE %q", value)
|
||||
}
|
||||
info.MediaSequence = sequence
|
||||
case strings.HasPrefix(line, "#EXTINF:"):
|
||||
if pendingDuration != 0 {
|
||||
return nil, errors.New("EXTINF without a media URI for the previous segment")
|
||||
}
|
||||
value := strings.TrimSpace(strings.TrimPrefix(line, "#EXTINF:"))
|
||||
if comma := strings.IndexByte(value, ','); comma >= 0 {
|
||||
value = value[:comma]
|
||||
}
|
||||
seconds, err := strconv.ParseFloat(value, 64)
|
||||
if err != nil || seconds <= 0 || math.IsNaN(seconds) || math.IsInf(seconds, 0) || seconds > math.MaxInt32/1000 {
|
||||
return nil, fmt.Errorf("invalid EXTINF duration %q", value)
|
||||
}
|
||||
pendingDuration = int64(math.Round(seconds * 1000))
|
||||
if pendingDuration == 0 {
|
||||
pendingDuration = 1
|
||||
}
|
||||
case strings.HasPrefix(line, "#EXT-X-BYTERANGE:"):
|
||||
if pendingDuration == 0 {
|
||||
return nil, errors.New("EXT-X-BYTERANGE without a preceding EXTINF")
|
||||
}
|
||||
value := strings.TrimSpace(strings.TrimPrefix(line, "#EXT-X-BYTERANGE:"))
|
||||
lengthText, offsetText, hasOffset := strings.Cut(value, "@")
|
||||
length, err := strconv.ParseInt(lengthText, 10, 64)
|
||||
if err != nil || length <= 0 {
|
||||
return nil, fmt.Errorf("invalid EXT-X-BYTERANGE length %q", value)
|
||||
}
|
||||
offset := expectedOffset
|
||||
if hasOffset {
|
||||
if offset, err = strconv.ParseInt(offsetText, 10, 64); err != nil || offset < 0 {
|
||||
return nil, fmt.Errorf("invalid EXT-X-BYTERANGE offset %q", value)
|
||||
}
|
||||
}
|
||||
pendingSize, pendingOffset, havePendingRange = length, offset, true
|
||||
case strings.HasPrefix(line, "#"):
|
||||
// other tags carry no state the generated playlist must keep
|
||||
default:
|
||||
if pendingDuration == 0 || !havePendingRange {
|
||||
return nil, fmt.Errorf("media URI %q without EXTINF and EXT-X-BYTERANGE", line)
|
||||
}
|
||||
if mediaURI == "" {
|
||||
mediaURI = line
|
||||
} else if mediaURI != line {
|
||||
return nil, errors.New("hls-ts ingest requires one shared media URI")
|
||||
}
|
||||
if pendingOffset != expectedOffset {
|
||||
return nil, fmt.Errorf("non-contiguous byte range at segment %d: offset %d, expected %d", len(sizes), pendingOffset, expectedOffset)
|
||||
}
|
||||
if pendingSize%TSPacketSize != 0 {
|
||||
return nil, fmt.Errorf("segment %d size %d is not a multiple of the %d-byte TS packet", len(sizes), pendingSize, TSPacketSize)
|
||||
}
|
||||
if pendingSize > math.MaxInt64-expectedOffset {
|
||||
return nil, fmt.Errorf("byte range at segment %d overflows", len(sizes))
|
||||
}
|
||||
if len(sizes) >= format.MaxExtentCount {
|
||||
return nil, fmt.Errorf("playlist has more than %d segments", format.MaxExtentCount)
|
||||
}
|
||||
sizes = append(sizes, pendingSize)
|
||||
info.DurationsMs = append(info.DurationsMs, pendingDuration)
|
||||
if pendingDuration > maxDurationMs {
|
||||
maxDurationMs = pendingDuration
|
||||
}
|
||||
expectedOffset += pendingSize
|
||||
pendingDuration, havePendingRange = 0, false
|
||||
}
|
||||
}
|
||||
if err := scanner.Err(); err != nil {
|
||||
return nil, fmt.Errorf("read playlist: %w", err)
|
||||
}
|
||||
if !sawHeader {
|
||||
return nil, errors.New("playlist is missing EXTM3U")
|
||||
}
|
||||
if pendingDuration != 0 || havePendingRange {
|
||||
return nil, errors.New("playlist ended with an incomplete media segment")
|
||||
}
|
||||
if len(sizes) == 0 {
|
||||
return nil, errors.New("playlist has no media segments")
|
||||
}
|
||||
if !sawEndList {
|
||||
return nil, errors.New("only VOD playlists with EXT-X-ENDLIST are supported")
|
||||
}
|
||||
if info.MediaSequence > math.MaxInt64-int64(len(sizes)-1) {
|
||||
return nil, errors.New("EXT-X-MEDIA-SEQUENCE overflows segment numbering")
|
||||
}
|
||||
|
||||
// RFC 8216: EXT-X-TARGETDURATION must be at least each segment duration
|
||||
// rounded to the nearest integer.
|
||||
minimumTarget := (maxDurationMs + 500) / 1000
|
||||
if minimumTarget < 1 {
|
||||
minimumTarget = 1
|
||||
}
|
||||
if info.TargetDuration == 0 {
|
||||
info.TargetDuration = minimumTarget
|
||||
} else if info.TargetDuration < minimumTarget {
|
||||
return nil, fmt.Errorf("EXT-X-TARGETDURATION %d is smaller than the longest segment duration %d", info.TargetDuration, minimumTarget)
|
||||
}
|
||||
|
||||
layout := &format.Layout{
|
||||
Format: FormatName,
|
||||
ExtentSizes: sizes,
|
||||
Align: TSPacketSize,
|
||||
Payload: info.encode(),
|
||||
}
|
||||
// valid by construction, but enforce the formattest invariant explicitly
|
||||
if err := layout.Validate(-1); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return layout, nil
|
||||
}
|
||||
|
||||
// View serves the generated playlist, or maps ?seq=N to its extent.
|
||||
func (Adapter) View(req format.ViewRequest, obj format.Object) (*format.ViewPlan, error) {
|
||||
info, err := decodePlaylistInfo(obj.Layout.Payload, len(obj.Layout.ExtentSizes))
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
sequenceText := req.Query.Get("seq")
|
||||
if sequenceText == "" {
|
||||
return &format.ViewPlan{ContentType: PlaylistContentType, Body: renderPlaylist(obj.Name, info)}, nil
|
||||
}
|
||||
sequence, err := strconv.ParseInt(sequenceText, 10, 64)
|
||||
if err != nil || sequence < info.MediaSequence {
|
||||
return nil, format.ErrNoSuchView
|
||||
}
|
||||
index := sequence - info.MediaSequence
|
||||
if index >= int64(len(obj.Layout.ExtentSizes)) {
|
||||
return nil, format.ErrNoSuchView
|
||||
}
|
||||
return &format.ViewPlan{ContentType: MediaContentType, Extent: int(index)}, nil
|
||||
}
|
||||
|
||||
func renderPlaylist(name string, info *playlistInfo) []byte {
|
||||
var out strings.Builder
|
||||
out.WriteString("#EXTM3U\n#EXT-X-VERSION:3\n")
|
||||
fmt.Fprintf(&out, "#EXT-X-TARGETDURATION:%d\n", info.TargetDuration)
|
||||
fmt.Fprintf(&out, "#EXT-X-MEDIA-SEQUENCE:%d\n", info.MediaSequence)
|
||||
out.WriteString("#EXT-X-PLAYLIST-TYPE:VOD\n")
|
||||
escapedName := url.PathEscape(name)
|
||||
for i, durationMs := range info.DurationsMs {
|
||||
fmt.Fprintf(&out, "#EXTINF:%.3f,\n", float64(durationMs)/1000)
|
||||
fmt.Fprintf(&out, "%s?%s=%s&seq=%d\n", escapedName, format.ViewParam, FormatName, int64(i)+info.MediaSequence)
|
||||
}
|
||||
out.WriteString("#EXT-X-ENDLIST\n")
|
||||
return []byte(out.String())
|
||||
}
|
||||
@@ -0,0 +1,204 @@
|
||||
package hlsts
|
||||
|
||||
import (
|
||||
"net/url"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/format"
|
||||
"github.com/seaweedfs/seaweedfs/weed/format/formattest"
|
||||
)
|
||||
|
||||
const ffmpegPlaylist = `#EXTM3U
|
||||
#EXT-X-VERSION:4
|
||||
#EXT-X-TARGETDURATION:6
|
||||
#EXT-X-MEDIA-SEQUENCE:0
|
||||
#EXT-X-PLAYLIST-TYPE:VOD
|
||||
#EXTINF:6.000000,
|
||||
#EXT-X-BYTERANGE:1128@0
|
||||
video.ts
|
||||
#EXTINF:6.000000,
|
||||
#EXT-X-BYTERANGE:940
|
||||
video.ts
|
||||
#EXTINF:2.500000,
|
||||
#EXT-X-BYTERANGE:376@2068
|
||||
video.ts
|
||||
#EXT-X-ENDLIST
|
||||
`
|
||||
|
||||
func TestIndexSidecar(t *testing.T) {
|
||||
layout, err := Adapter{}.IndexSidecar([]byte(ffmpegPlaylist))
|
||||
if err != nil {
|
||||
t.Fatalf("IndexSidecar() error = %v", err)
|
||||
}
|
||||
wantSizes := []int64{1128, 940, 376}
|
||||
if len(layout.ExtentSizes) != len(wantSizes) {
|
||||
t.Fatalf("extents = %v, want %v", layout.ExtentSizes, wantSizes)
|
||||
}
|
||||
for i := range wantSizes {
|
||||
if layout.ExtentSizes[i] != wantSizes[i] {
|
||||
t.Fatalf("extent %d = %d, want %d", i, layout.ExtentSizes[i], wantSizes[i])
|
||||
}
|
||||
}
|
||||
if layout.Align != TSPacketSize || layout.Format != FormatName {
|
||||
t.Fatalf("layout = %+v", layout)
|
||||
}
|
||||
if err := layout.Validate(1128 + 940 + 376); err != nil {
|
||||
t.Fatalf("Validate() error = %v", err)
|
||||
}
|
||||
formattest.EncodeRoundTrip(t, layout)
|
||||
}
|
||||
|
||||
func TestIndexSidecarDefaultsTargetDuration(t *testing.T) {
|
||||
playlist := "#EXTM3U\n#EXTINF:5.6,\n#EXT-X-BYTERANGE:188@0\nv.ts\n#EXT-X-ENDLIST\n"
|
||||
layout, err := Adapter{}.IndexSidecar([]byte(playlist))
|
||||
if err != nil {
|
||||
t.Fatalf("IndexSidecar() error = %v", err)
|
||||
}
|
||||
info, err := decodePlaylistInfo(layout.Payload, len(layout.ExtentSizes))
|
||||
if err != nil {
|
||||
t.Fatalf("decodePlaylistInfo() error = %v", err)
|
||||
}
|
||||
if info.TargetDuration != 6 {
|
||||
t.Fatalf("TargetDuration = %d, want 6", info.TargetDuration)
|
||||
}
|
||||
}
|
||||
|
||||
func TestIndexSidecarRejections(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
playlist string
|
||||
wantErr string
|
||||
}{
|
||||
{"missing header", "#EXTINF:6,\n#EXT-X-BYTERANGE:188@0\nv.ts\n#EXT-X-ENDLIST\n", "EXTM3U"},
|
||||
{"missing endlist", "#EXTM3U\n#EXTINF:6,\n#EXT-X-BYTERANGE:188@0\nv.ts\n", "EXT-X-ENDLIST"},
|
||||
{"no segments", "#EXTM3U\n#EXT-X-ENDLIST\n", "no media segments"},
|
||||
{"encryption", "#EXTM3U\n#EXT-X-KEY:METHOD=AES-128\n#EXTINF:6,\n#EXT-X-BYTERANGE:188@0\nv.ts\n#EXT-X-ENDLIST\n", "EXT-X-KEY"},
|
||||
{"discontinuity", "#EXTM3U\n#EXTINF:6,\n#EXT-X-BYTERANGE:188@0\nv.ts\n#EXT-X-DISCONTINUITY\n#EXT-X-ENDLIST\n", "EXT-X-DISCONTINUITY"},
|
||||
{"map", "#EXTM3U\n#EXT-X-MAP:URI=\"init.mp4\"\n#EXTINF:6,\n#EXT-X-BYTERANGE:188@0\nv.ts\n#EXT-X-ENDLIST\n", "EXT-X-MAP"},
|
||||
{"gap", "#EXTM3U\n#EXT-X-GAP\n#EXTINF:6,\n#EXT-X-BYTERANGE:188@0\nv.ts\n#EXT-X-ENDLIST\n", "EXT-X-GAP"},
|
||||
{"iframes only", "#EXTM3U\n#EXT-X-I-FRAMES-ONLY\n#EXTINF:6,\n#EXT-X-BYTERANGE:188@0\nv.ts\n#EXT-X-ENDLIST\n", "EXT-X-I-FRAMES-ONLY"},
|
||||
{"no byterange", "#EXTM3U\n#EXTINF:6,\nv.ts\n#EXT-X-ENDLIST\n", "EXT-X-BYTERANGE"},
|
||||
{"gap in ranges", "#EXTM3U\n#EXTINF:6,\n#EXT-X-BYTERANGE:188@0\nv.ts\n#EXTINF:6,\n#EXT-X-BYTERANGE:188@376\nv.ts\n#EXT-X-ENDLIST\n", "non-contiguous"},
|
||||
{"unaligned size", "#EXTM3U\n#EXTINF:6,\n#EXT-X-BYTERANGE:100@0\nv.ts\n#EXT-X-ENDLIST\n", "TS packet"},
|
||||
{"two media files", "#EXTM3U\n#EXTINF:6,\n#EXT-X-BYTERANGE:188@0\na.ts\n#EXTINF:6,\n#EXT-X-BYTERANGE:188@188\nb.ts\n#EXT-X-ENDLIST\n", "one shared media URI"},
|
||||
{"target too small", "#EXTM3U\n#EXT-X-TARGETDURATION:2\n#EXTINF:6,\n#EXT-X-BYTERANGE:188@0\nv.ts\n#EXT-X-ENDLIST\n", "smaller than"},
|
||||
{"zero duration", "#EXTM3U\n#EXTINF:0,\n#EXT-X-BYTERANGE:188@0\nv.ts\n#EXT-X-ENDLIST\n", "EXTINF"},
|
||||
{"dangling extinf", "#EXTM3U\n#EXTINF:6,\n#EXT-X-BYTERANGE:188@0\nv.ts\n#EXTINF:6,\n#EXT-X-ENDLIST\n", "incomplete"},
|
||||
}
|
||||
for _, test := range tests {
|
||||
_, err := Adapter{}.IndexSidecar([]byte(test.playlist))
|
||||
if err == nil || !strings.Contains(err.Error(), test.wantErr) {
|
||||
t.Fatalf("%s: error = %v, want %q", test.name, err, test.wantErr)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestSidecarTruncations(t *testing.T) {
|
||||
formattest.SidecarTruncations(t, Adapter{}, []byte(ffmpegPlaylist))
|
||||
}
|
||||
|
||||
func viewObject(t *testing.T) format.Object {
|
||||
t.Helper()
|
||||
layout, err := Adapter{}.IndexSidecar([]byte(ffmpegPlaylist))
|
||||
if err != nil {
|
||||
t.Fatalf("IndexSidecar() error = %v", err)
|
||||
}
|
||||
return format.Object{Name: "movie.ts", Size: layout.TotalSize(), Layout: layout}
|
||||
}
|
||||
|
||||
func TestViewPlaylist(t *testing.T) {
|
||||
plan, err := Adapter{}.View(format.ViewRequest{Query: url.Values{}}, viewObject(t))
|
||||
if err != nil {
|
||||
t.Fatalf("View() error = %v", err)
|
||||
}
|
||||
if plan.ContentType != PlaylistContentType {
|
||||
t.Fatalf("ContentType = %q", plan.ContentType)
|
||||
}
|
||||
want := `#EXTM3U
|
||||
#EXT-X-VERSION:3
|
||||
#EXT-X-TARGETDURATION:6
|
||||
#EXT-X-MEDIA-SEQUENCE:0
|
||||
#EXT-X-PLAYLIST-TYPE:VOD
|
||||
#EXTINF:6.000,
|
||||
movie.ts?format.view=hls-ts&seq=0
|
||||
#EXTINF:6.000,
|
||||
movie.ts?format.view=hls-ts&seq=1
|
||||
#EXTINF:2.500,
|
||||
movie.ts?format.view=hls-ts&seq=2
|
||||
#EXT-X-ENDLIST
|
||||
`
|
||||
if string(plan.Body) != want {
|
||||
t.Fatalf("playlist = %q, want %q", plan.Body, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestViewSegment(t *testing.T) {
|
||||
obj := viewObject(t)
|
||||
plan, err := Adapter{}.View(format.ViewRequest{Query: url.Values{"seq": {"1"}}}, obj)
|
||||
if err != nil {
|
||||
t.Fatalf("View() error = %v", err)
|
||||
}
|
||||
if plan.Body != nil || plan.Extent != 1 || plan.ContentType != MediaContentType {
|
||||
t.Fatalf("plan = %+v", plan)
|
||||
}
|
||||
for _, bad := range []string{"3", "-1", "x", "9999999999999999999"} {
|
||||
if _, err := (Adapter{}).View(format.ViewRequest{Query: url.Values{"seq": {bad}}}, obj); err != format.ErrNoSuchView {
|
||||
t.Fatalf("seq %q: error = %v, want ErrNoSuchView", bad, err)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestViewSegmentHonorsMediaSequence(t *testing.T) {
|
||||
playlist := "#EXTM3U\n#EXT-X-MEDIA-SEQUENCE:10\n#EXTINF:6,\n#EXT-X-BYTERANGE:188@0\nv.ts\n#EXTINF:6,\n#EXT-X-BYTERANGE:376\nv.ts\n#EXT-X-ENDLIST\n"
|
||||
layout, err := Adapter{}.IndexSidecar([]byte(playlist))
|
||||
if err != nil {
|
||||
t.Fatalf("IndexSidecar() error = %v", err)
|
||||
}
|
||||
obj := format.Object{Name: "v.ts", Size: layout.TotalSize(), Layout: layout}
|
||||
plan, err := Adapter{}.View(format.ViewRequest{Query: url.Values{"seq": {"11"}}}, obj)
|
||||
if err != nil {
|
||||
t.Fatalf("View() error = %v", err)
|
||||
}
|
||||
if plan.Extent != 1 {
|
||||
t.Fatalf("Extent = %d, want 1", plan.Extent)
|
||||
}
|
||||
if _, err := (Adapter{}).View(format.ViewRequest{Query: url.Values{"seq": {"9"}}}, obj); err != format.ErrNoSuchView {
|
||||
t.Fatalf("seq below media sequence: error = %v, want ErrNoSuchView", err)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSniff(t *testing.T) {
|
||||
head := make([]byte, 400)
|
||||
head[0], head[TSPacketSize] = tsSyncByte, tsSyncByte
|
||||
if !(Adapter{}).Sniff(format.Hint{Head: head}) {
|
||||
t.Fatalf("Sniff() rejected TS head")
|
||||
}
|
||||
head[TSPacketSize] = 0
|
||||
if (Adapter{}).Sniff(format.Hint{Head: head}) {
|
||||
t.Fatalf("Sniff() accepted non-TS head")
|
||||
}
|
||||
}
|
||||
|
||||
// The ingest bound admits mediaSequence = MaxInt64-(count-1); the payload
|
||||
// decoder must accept the same boundary or every view of such an asset fails.
|
||||
func TestMediaSequenceBoundaryRoundTrip(t *testing.T) {
|
||||
playlist := "#EXTM3U\n#EXT-X-MEDIA-SEQUENCE:9223372036854775806\n" +
|
||||
"#EXTINF:6,\n#EXT-X-BYTERANGE:188@0\nv.ts\n#EXTINF:6,\n#EXT-X-BYTERANGE:188\nv.ts\n#EXT-X-ENDLIST\n"
|
||||
layout, err := Adapter{}.IndexSidecar([]byte(playlist))
|
||||
if err != nil {
|
||||
t.Fatalf("IndexSidecar() error = %v", err)
|
||||
}
|
||||
obj := format.Object{Name: "v.ts", Size: layout.TotalSize(), Layout: layout}
|
||||
if _, err := (Adapter{}).View(format.ViewRequest{Query: url.Values{}}, obj); err != nil {
|
||||
t.Fatalf("playlist view error = %v", err)
|
||||
}
|
||||
plan, err := Adapter{}.View(format.ViewRequest{Query: url.Values{"seq": {"9223372036854775807"}}}, obj)
|
||||
if err != nil {
|
||||
t.Fatalf("last segment view error = %v", err)
|
||||
}
|
||||
if plan.Extent != 1 {
|
||||
t.Fatalf("Extent = %d, want 1", plan.Extent)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,204 @@
|
||||
package format
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/binary"
|
||||
"fmt"
|
||||
"math"
|
||||
"sort"
|
||||
)
|
||||
|
||||
const (
|
||||
layoutVersion = 1
|
||||
|
||||
// MaxExtentCount bounds decoded layouts; it also bounds the chunk count a
|
||||
// layout can force on an entry.
|
||||
MaxExtentCount = 1 << 20
|
||||
// MaxPayloadBytes bounds the adapter payload carried in entry metadata.
|
||||
MaxPayloadBytes = 16 << 20
|
||||
// MaxFormatNameBytes bounds the encoded adapter name.
|
||||
MaxFormatNameBytes = 256
|
||||
)
|
||||
|
||||
// TotalSize returns the file size the layout describes.
|
||||
func (l *Layout) TotalSize() int64 {
|
||||
var total int64
|
||||
for _, size := range l.ExtentSizes {
|
||||
total += size
|
||||
}
|
||||
return total
|
||||
}
|
||||
|
||||
// ExtentRange returns the byte range of extent i.
|
||||
func (l *Layout) ExtentRange(i int) (offset, size int64, ok bool) {
|
||||
if i < 0 || i >= len(l.ExtentSizes) {
|
||||
return 0, 0, false
|
||||
}
|
||||
for _, extentSize := range l.ExtentSizes[:i] {
|
||||
offset += extentSize
|
||||
}
|
||||
return offset, l.ExtentSizes[i], true
|
||||
}
|
||||
|
||||
// Validate checks layout consistency. A negative fileSize skips the total size
|
||||
// check.
|
||||
func (l *Layout) Validate(fileSize int64) error {
|
||||
if l.Format == "" {
|
||||
return fmt.Errorf("layout has no format name")
|
||||
}
|
||||
if len(l.Format) > MaxFormatNameBytes {
|
||||
return fmt.Errorf("layout format name is too long: %d bytes", len(l.Format))
|
||||
}
|
||||
if l.Align < 1 {
|
||||
return fmt.Errorf("layout align %d is invalid", l.Align)
|
||||
}
|
||||
if len(l.ExtentSizes) == 0 {
|
||||
return fmt.Errorf("layout has no extents")
|
||||
}
|
||||
if len(l.ExtentSizes) > MaxExtentCount {
|
||||
return fmt.Errorf("layout has too many extents: %d", len(l.ExtentSizes))
|
||||
}
|
||||
if len(l.Payload) > MaxPayloadBytes {
|
||||
return fmt.Errorf("layout payload is too large: %d bytes", len(l.Payload))
|
||||
}
|
||||
var total int64
|
||||
for i, size := range l.ExtentSizes {
|
||||
if size <= 0 {
|
||||
return fmt.Errorf("extent %d has invalid size %d", i, size)
|
||||
}
|
||||
if size > math.MaxInt64-total {
|
||||
return fmt.Errorf("extent %d overflows the file size", i)
|
||||
}
|
||||
total += size
|
||||
}
|
||||
if fileSize >= 0 && total != fileSize {
|
||||
return fmt.Errorf("layout describes %d bytes but the file has %d", total, fileSize)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// Encode serializes the layout for the LayoutKey extended attribute.
|
||||
func (l *Layout) Encode() ([]byte, error) {
|
||||
if err := l.Validate(-1); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
out := []byte{layoutVersion}
|
||||
out = binary.AppendUvarint(out, uint64(len(l.Format)))
|
||||
out = append(out, l.Format...)
|
||||
out = binary.AppendUvarint(out, uint64(l.Align))
|
||||
out = binary.AppendUvarint(out, uint64(len(l.ExtentSizes)))
|
||||
for _, size := range l.ExtentSizes {
|
||||
out = binary.AppendUvarint(out, uint64(size))
|
||||
}
|
||||
out = binary.AppendUvarint(out, uint64(len(l.Payload)))
|
||||
out = append(out, l.Payload...)
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// DecodeLayout parses an encoded layout and validates it.
|
||||
func DecodeLayout(data []byte) (*Layout, error) {
|
||||
reader := bytes.NewReader(data)
|
||||
version, err := reader.ReadByte()
|
||||
if err != nil || version != layoutVersion {
|
||||
return nil, fmt.Errorf("unsupported layout version")
|
||||
}
|
||||
name, err := readUvarintBytes(reader, MaxFormatNameBytes)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("read layout format: %w", err)
|
||||
}
|
||||
align, err := binary.ReadUvarint(reader)
|
||||
if err != nil || align > math.MaxInt64 {
|
||||
return nil, fmt.Errorf("read layout align: invalid")
|
||||
}
|
||||
count, err := binary.ReadUvarint(reader)
|
||||
if err != nil || count > MaxExtentCount {
|
||||
return nil, fmt.Errorf("read layout extent count: invalid")
|
||||
}
|
||||
sizes := make([]int64, count)
|
||||
for i := range sizes {
|
||||
size, err := binary.ReadUvarint(reader)
|
||||
if err != nil || size > math.MaxInt64 {
|
||||
return nil, fmt.Errorf("read extent %d size: invalid", i)
|
||||
}
|
||||
sizes[i] = int64(size)
|
||||
}
|
||||
payload, err := readUvarintBytes(reader, MaxPayloadBytes)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("read layout payload: %w", err)
|
||||
}
|
||||
if reader.Len() != 0 {
|
||||
return nil, fmt.Errorf("layout has %d trailing bytes", reader.Len())
|
||||
}
|
||||
layout := &Layout{Format: string(name), ExtentSizes: sizes, Align: int64(align), Payload: payload}
|
||||
if err := layout.Validate(-1); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return layout, nil
|
||||
}
|
||||
|
||||
func readUvarintBytes(reader *bytes.Reader, limit uint64) ([]byte, error) {
|
||||
length, err := binary.ReadUvarint(reader)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if length > limit || length > uint64(reader.Len()) {
|
||||
return nil, fmt.Errorf("length %d is out of bounds", length)
|
||||
}
|
||||
if length == 0 {
|
||||
return nil, nil
|
||||
}
|
||||
data := make([]byte, length)
|
||||
if _, err := reader.Read(data); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return data, nil
|
||||
}
|
||||
|
||||
// Cutter yields upload chunk boundaries: every extent boundary, plus
|
||||
// align-quantized cuts inside extents larger than maxChunkSize. A non-positive
|
||||
// maxChunkSize keeps each extent in one chunk. Interior cuts are computed
|
||||
// lazily, so memory stays bounded by the extent count no matter how large an
|
||||
// untrusted layout declares its extents to be.
|
||||
type Cutter struct {
|
||||
starts []int64 // extent start offsets, ascending
|
||||
total int64
|
||||
quantum int64 // interior cut spacing; 0 = one chunk per extent
|
||||
}
|
||||
|
||||
func (l *Layout) Cutter(maxChunkSize int64) *Cutter {
|
||||
quantum := maxChunkSize
|
||||
if quantum > 0 && l.Align > 1 {
|
||||
quantum -= quantum % l.Align
|
||||
if quantum <= 0 {
|
||||
// An align larger than the chunk limit still cuts on whole atoms.
|
||||
quantum = l.Align
|
||||
}
|
||||
}
|
||||
starts := make([]int64, len(l.ExtentSizes))
|
||||
var offset int64
|
||||
for i, size := range l.ExtentSizes {
|
||||
starts[i] = offset
|
||||
offset += size
|
||||
}
|
||||
return &Cutter{starts: starts, total: offset, quantum: quantum}
|
||||
}
|
||||
|
||||
// NextChunkSize returns the size of the chunk starting at offset, or 0 past
|
||||
// the end. It satisfies the filer upload loop's ChunkBoundaries interface.
|
||||
func (c *Cutter) NextChunkSize(offset int64) int64 {
|
||||
if offset < 0 || offset >= c.total {
|
||||
return 0
|
||||
}
|
||||
i := sort.Search(len(c.starts), func(i int) bool { return c.starts[i] > offset }) - 1
|
||||
end := c.total
|
||||
if i+1 < len(c.starts) {
|
||||
end = c.starts[i+1]
|
||||
}
|
||||
remaining := end - offset
|
||||
if c.quantum > 0 {
|
||||
if step := c.quantum - (offset-c.starts[i])%c.quantum; step < remaining {
|
||||
return step
|
||||
}
|
||||
}
|
||||
return remaining
|
||||
}
|
||||
@@ -0,0 +1,192 @@
|
||||
package format
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestLayoutEncodeDecodeRoundTrip(t *testing.T) {
|
||||
layout := &Layout{
|
||||
Format: "hls-ts",
|
||||
ExtentSizes: []int64{188 * 3, 188 * 2, 188 * 7},
|
||||
Align: 188,
|
||||
Payload: []byte{1, 2, 3},
|
||||
}
|
||||
encoded, err := layout.Encode()
|
||||
if err != nil {
|
||||
t.Fatalf("Encode() error = %v", err)
|
||||
}
|
||||
decoded, err := DecodeLayout(encoded)
|
||||
if err != nil {
|
||||
t.Fatalf("DecodeLayout() error = %v", err)
|
||||
}
|
||||
if decoded.Format != layout.Format || decoded.Align != layout.Align {
|
||||
t.Fatalf("decoded = %+v, want %+v", decoded, layout)
|
||||
}
|
||||
if len(decoded.ExtentSizes) != len(layout.ExtentSizes) {
|
||||
t.Fatalf("extent count = %d, want %d", len(decoded.ExtentSizes), len(layout.ExtentSizes))
|
||||
}
|
||||
for i := range layout.ExtentSizes {
|
||||
if decoded.ExtentSizes[i] != layout.ExtentSizes[i] {
|
||||
t.Fatalf("extent %d = %d, want %d", i, decoded.ExtentSizes[i], layout.ExtentSizes[i])
|
||||
}
|
||||
}
|
||||
if string(decoded.Payload) != string(layout.Payload) {
|
||||
t.Fatalf("payload = %v, want %v", decoded.Payload, layout.Payload)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDecodeLayoutRejectsCorruptInput(t *testing.T) {
|
||||
layout := &Layout{Format: "parquet", ExtentSizes: []int64{10, 20}, Align: 1}
|
||||
encoded, err := layout.Encode()
|
||||
if err != nil {
|
||||
t.Fatalf("Encode() error = %v", err)
|
||||
}
|
||||
for cut := 0; cut < len(encoded); cut++ {
|
||||
if _, err := DecodeLayout(encoded[:cut]); err == nil {
|
||||
t.Fatalf("DecodeLayout() accepted truncation at %d", cut)
|
||||
}
|
||||
}
|
||||
if _, err := DecodeLayout(append(append([]byte{}, encoded...), 0)); err == nil {
|
||||
t.Fatalf("DecodeLayout() accepted trailing bytes")
|
||||
}
|
||||
}
|
||||
|
||||
func TestLayoutValidate(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
layout Layout
|
||||
fileSize int64
|
||||
wantErr string
|
||||
}{
|
||||
{"valid", Layout{Format: "x", ExtentSizes: []int64{5, 5}, Align: 1}, 10, ""},
|
||||
{"skip size check", Layout{Format: "x", ExtentSizes: []int64{5}, Align: 1}, -1, ""},
|
||||
{"wrong total", Layout{Format: "x", ExtentSizes: []int64{5, 5}, Align: 1}, 11, "but the file has"},
|
||||
{"zero extent", Layout{Format: "x", ExtentSizes: []int64{5, 0}, Align: 1}, -1, "invalid size"},
|
||||
{"no extents", Layout{Format: "x", Align: 1}, -1, "no extents"},
|
||||
{"bad align", Layout{Format: "x", ExtentSizes: []int64{5}, Align: 0}, -1, "align"},
|
||||
{"no name", Layout{ExtentSizes: []int64{5}, Align: 1}, -1, "format name"},
|
||||
{"name too long", Layout{Format: strings.Repeat("x", MaxFormatNameBytes+1), ExtentSizes: []int64{5}, Align: 1}, -1, "too long"},
|
||||
}
|
||||
for _, test := range tests {
|
||||
err := test.layout.Validate(test.fileSize)
|
||||
if test.wantErr == "" {
|
||||
if err != nil {
|
||||
t.Fatalf("%s: Validate() error = %v", test.name, err)
|
||||
}
|
||||
continue
|
||||
}
|
||||
if err == nil || !strings.Contains(err.Error(), test.wantErr) {
|
||||
t.Fatalf("%s: Validate() error = %v, want %q", test.name, err, test.wantErr)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestExtentRange(t *testing.T) {
|
||||
layout := &Layout{Format: "x", ExtentSizes: []int64{10, 20, 30}, Align: 1}
|
||||
offset, size, ok := layout.ExtentRange(1)
|
||||
if !ok || offset != 10 || size != 20 {
|
||||
t.Fatalf("ExtentRange(1) = (%d, %d, %v), want (10, 20, true)", offset, size, ok)
|
||||
}
|
||||
if _, _, ok := layout.ExtentRange(3); ok {
|
||||
t.Fatalf("ExtentRange(3) accepted out-of-range index")
|
||||
}
|
||||
if _, _, ok := layout.ExtentRange(-1); ok {
|
||||
t.Fatalf("ExtentRange(-1) accepted negative index")
|
||||
}
|
||||
}
|
||||
|
||||
// collectChunks walks the cutter the way the upload loop does.
|
||||
func collectChunks(t *testing.T, cutter *Cutter) [][2]int64 {
|
||||
t.Helper()
|
||||
var chunks [][2]int64
|
||||
var offset int64
|
||||
for {
|
||||
size := cutter.NextChunkSize(offset)
|
||||
if size <= 0 {
|
||||
return chunks
|
||||
}
|
||||
chunks = append(chunks, [2]int64{offset, size})
|
||||
offset += size
|
||||
}
|
||||
}
|
||||
|
||||
func TestCutterKeepsExtentBoundaries(t *testing.T) {
|
||||
layout := &Layout{Format: "x", ExtentSizes: []int64{5, 4}, Align: 1}
|
||||
chunks := collectChunks(t, layout.Cutter(16))
|
||||
want := [][2]int64{{0, 5}, {5, 4}}
|
||||
if len(chunks) != len(want) {
|
||||
t.Fatalf("chunks = %v, want %v", chunks, want)
|
||||
}
|
||||
for i := range want {
|
||||
if chunks[i] != want[i] {
|
||||
t.Fatalf("chunk %d = %v, want %v", i, chunks[i], want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestCutterSplitsOversizedExtentsOnAlign(t *testing.T) {
|
||||
// maxChunkSize 5 with align 2 quantizes down to 4-byte interior cuts.
|
||||
layout := &Layout{Format: "x", ExtentSizes: []int64{10, 3}, Align: 2}
|
||||
chunks := collectChunks(t, layout.Cutter(5))
|
||||
want := [][2]int64{{0, 4}, {4, 4}, {8, 2}, {10, 3}}
|
||||
if len(chunks) != len(want) {
|
||||
t.Fatalf("chunks = %v, want %v", chunks, want)
|
||||
}
|
||||
for i := range want {
|
||||
if chunks[i] != want[i] {
|
||||
t.Fatalf("chunk %d = %v, want %v", i, chunks[i], want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestCutterAlignLargerThanChunkLimit(t *testing.T) {
|
||||
// Align above maxChunkSize still cuts on whole atoms.
|
||||
layout := &Layout{Format: "x", ExtentSizes: []int64{20}, Align: 8}
|
||||
chunks := collectChunks(t, layout.Cutter(5))
|
||||
want := [][2]int64{{0, 8}, {8, 8}, {16, 4}}
|
||||
if len(chunks) != len(want) {
|
||||
t.Fatalf("chunks = %v, want %v", chunks, want)
|
||||
}
|
||||
for i := range want {
|
||||
if chunks[i] != want[i] {
|
||||
t.Fatalf("chunk %d = %v, want %v", i, chunks[i], want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestCutterUnlimitedKeepsOneChunkPerExtent(t *testing.T) {
|
||||
layout := &Layout{Format: "x", ExtentSizes: []int64{10, 3}, Align: 188}
|
||||
chunks := collectChunks(t, layout.Cutter(0))
|
||||
want := [][2]int64{{0, 10}, {10, 3}}
|
||||
if len(chunks) != len(want) {
|
||||
t.Fatalf("chunks = %v, want %v", chunks, want)
|
||||
}
|
||||
}
|
||||
|
||||
// A hostile layout may declare an enormous extent; the cutter must stay O(1)
|
||||
// per query instead of materializing every interior cut.
|
||||
func TestCutterHugeExtentStaysLazy(t *testing.T) {
|
||||
const quantum = 4 << 20 // 4MiB, already a multiple of align 1
|
||||
layout := &Layout{Format: "x", ExtentSizes: []int64{1 << 50, 188}, Align: 188}
|
||||
cutter := layout.Cutter(quantum)
|
||||
alignedQuantum := int64(quantum - quantum%188)
|
||||
if got := cutter.NextChunkSize(0); got != alignedQuantum {
|
||||
t.Fatalf("NextChunkSize(0) = %d, want %d", got, alignedQuantum)
|
||||
}
|
||||
if got := cutter.NextChunkSize(alignedQuantum * 1000); got != alignedQuantum {
|
||||
t.Fatalf("mid-extent chunk = %d, want %d", got, alignedQuantum)
|
||||
}
|
||||
// the final interior chunk stops at the extent boundary
|
||||
last := (int64(1<<50) / alignedQuantum) * alignedQuantum
|
||||
if got := cutter.NextChunkSize(last); got != int64(1<<50)-last {
|
||||
t.Fatalf("tail chunk = %d, want %d", got, int64(1<<50)-last)
|
||||
}
|
||||
// the next extent still cuts independently
|
||||
if got := cutter.NextChunkSize(1 << 50); got != 188 {
|
||||
t.Fatalf("second extent chunk = %d, want 188", got)
|
||||
}
|
||||
if got := cutter.NextChunkSize(1<<50 + 188); got != 0 {
|
||||
t.Fatalf("past end = %d, want 0", got)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,98 @@
|
||||
// Package parquet adapts parquet files: extents are cut at row-group starts,
|
||||
// with one trailing extent for the page indexes and footer. Readers that fetch
|
||||
// row groups by offset then hit exactly the covering chunks. No view is
|
||||
// needed; the whole benefit is delivered by alignment.
|
||||
package parquet
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"fmt"
|
||||
"io"
|
||||
"math"
|
||||
|
||||
parquetgo "github.com/parquet-go/parquet-go"
|
||||
"github.com/seaweedfs/seaweedfs/weed/format"
|
||||
)
|
||||
|
||||
const FormatName = "parquet"
|
||||
|
||||
var magic = []byte("PAR1")
|
||||
|
||||
func init() {
|
||||
format.Register(Adapter{})
|
||||
}
|
||||
|
||||
type Adapter struct{}
|
||||
|
||||
var (
|
||||
_ format.Sniffer = Adapter{}
|
||||
_ format.Indexer = Adapter{}
|
||||
)
|
||||
|
||||
func (Adapter) Name() string { return FormatName }
|
||||
|
||||
func (Adapter) Sniff(h format.Hint) bool {
|
||||
return bytes.HasPrefix(h.Head, magic) && bytes.HasSuffix(h.Tail, magic)
|
||||
}
|
||||
|
||||
// Index reads the footer and cuts one extent per row group. The leading magic
|
||||
// rides with the first row group; everything after the last row group (page
|
||||
// indexes, footer) forms the final extent.
|
||||
func (Adapter) Index(ctx context.Context, r io.ReaderAt, size int64) (*format.Layout, error) {
|
||||
file, err := parquetgo.OpenFile(r, size, parquetgo.SkipPageIndex(true), parquetgo.SkipBloomFilters(true))
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("open parquet: %w", err)
|
||||
}
|
||||
rowGroups := file.Metadata().RowGroups
|
||||
if len(rowGroups) == 0 {
|
||||
return nil, fmt.Errorf("parquet file has no row groups")
|
||||
}
|
||||
if len(rowGroups) >= format.MaxExtentCount {
|
||||
return nil, fmt.Errorf("parquet file has too many row groups: %d", len(rowGroups))
|
||||
}
|
||||
|
||||
starts := make([]int64, 0, len(rowGroups))
|
||||
var lastEnd int64
|
||||
for i, rowGroup := range rowGroups {
|
||||
start, end := int64(math.MaxInt64), int64(0)
|
||||
for _, column := range rowGroup.Columns {
|
||||
columnStart := column.MetaData.DataPageOffset
|
||||
if dictionary := column.MetaData.DictionaryPageOffset; dictionary > 0 && dictionary < columnStart {
|
||||
columnStart = dictionary
|
||||
}
|
||||
if columnStart < start {
|
||||
start = columnStart
|
||||
}
|
||||
if columnEnd := columnStart + column.MetaData.TotalCompressedSize; columnEnd > end {
|
||||
end = columnEnd
|
||||
}
|
||||
}
|
||||
if len(rowGroup.Columns) == 0 || start >= end || start < int64(len(magic)) || end > size {
|
||||
return nil, fmt.Errorf("row group %d has an invalid byte range [%d, %d)", i, start, end)
|
||||
}
|
||||
if i > 0 && start < lastEnd {
|
||||
return nil, fmt.Errorf("row group %d overlaps its predecessor", i)
|
||||
}
|
||||
starts = append(starts, start)
|
||||
lastEnd = end
|
||||
}
|
||||
if lastEnd >= size {
|
||||
return nil, fmt.Errorf("row groups leave no room for the footer")
|
||||
}
|
||||
|
||||
var sizes []int64
|
||||
var previous int64
|
||||
for _, start := range starts[1:] {
|
||||
sizes = append(sizes, start-previous)
|
||||
previous = start
|
||||
}
|
||||
sizes = append(sizes, lastEnd-previous)
|
||||
sizes = append(sizes, size-lastEnd)
|
||||
|
||||
layout := &format.Layout{Format: FormatName, ExtentSizes: sizes, Align: 1}
|
||||
if err := layout.Validate(size); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return layout, nil
|
||||
}
|
||||
@@ -0,0 +1,108 @@
|
||||
package parquet
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"testing"
|
||||
|
||||
parquetgo "github.com/parquet-go/parquet-go"
|
||||
"github.com/seaweedfs/seaweedfs/weed/format"
|
||||
"github.com/seaweedfs/seaweedfs/weed/format/formattest"
|
||||
)
|
||||
|
||||
type row struct {
|
||||
ID int64 `parquet:"id"`
|
||||
Name string `parquet:"name"`
|
||||
}
|
||||
|
||||
// buildParquet writes rowGroups row groups of rowsPerGroup rows each.
|
||||
func buildParquet(t *testing.T, rowGroups, rowsPerGroup int) []byte {
|
||||
t.Helper()
|
||||
var buf bytes.Buffer
|
||||
writer := parquetgo.NewGenericWriter[row](&buf)
|
||||
for g := 0; g < rowGroups; g++ {
|
||||
rows := make([]row, rowsPerGroup)
|
||||
for i := range rows {
|
||||
rows[i] = row{ID: int64(g*rowsPerGroup + i), Name: "some filler content to give row groups a little size"}
|
||||
}
|
||||
if _, err := writer.Write(rows); err != nil {
|
||||
t.Fatalf("Write() error = %v", err)
|
||||
}
|
||||
if err := writer.Flush(); err != nil {
|
||||
t.Fatalf("Flush() error = %v", err)
|
||||
}
|
||||
}
|
||||
if err := writer.Close(); err != nil {
|
||||
t.Fatalf("Close() error = %v", err)
|
||||
}
|
||||
return buf.Bytes()
|
||||
}
|
||||
|
||||
func TestIndex(t *testing.T) {
|
||||
data := buildParquet(t, 3, 100)
|
||||
size := int64(len(data))
|
||||
layout, err := Adapter{}.Index(context.Background(), bytes.NewReader(data), size)
|
||||
if err != nil {
|
||||
t.Fatalf("Index() error = %v", err)
|
||||
}
|
||||
// one extent per row group plus the trailing footer extent
|
||||
if len(layout.ExtentSizes) != 4 {
|
||||
t.Fatalf("extents = %v, want 4", layout.ExtentSizes)
|
||||
}
|
||||
if err := layout.Validate(size); err != nil {
|
||||
t.Fatalf("Validate() error = %v", err)
|
||||
}
|
||||
formattest.EncodeRoundTrip(t, layout)
|
||||
|
||||
// extent cuts must land on the row-group starts the footer declares
|
||||
file, err := parquetgo.OpenFile(bytes.NewReader(data), size)
|
||||
if err != nil {
|
||||
t.Fatalf("OpenFile() error = %v", err)
|
||||
}
|
||||
var offset int64
|
||||
for i, extentSize := range layout.ExtentSizes[:3] {
|
||||
if i > 0 {
|
||||
want := file.Metadata().RowGroups[i].Columns[0].MetaData.DataPageOffset
|
||||
if dictionary := file.Metadata().RowGroups[i].Columns[0].MetaData.DictionaryPageOffset; dictionary > 0 && dictionary < want {
|
||||
want = dictionary
|
||||
}
|
||||
if offset != want {
|
||||
t.Fatalf("extent %d starts at %d, row group starts at %d", i, offset, want)
|
||||
}
|
||||
}
|
||||
offset += extentSize
|
||||
}
|
||||
}
|
||||
|
||||
func TestIndexSingleRowGroup(t *testing.T) {
|
||||
data := buildParquet(t, 1, 10)
|
||||
layout, err := Adapter{}.Index(context.Background(), bytes.NewReader(data), int64(len(data)))
|
||||
if err != nil {
|
||||
t.Fatalf("Index() error = %v", err)
|
||||
}
|
||||
if len(layout.ExtentSizes) != 2 {
|
||||
t.Fatalf("extents = %v, want 2", layout.ExtentSizes)
|
||||
}
|
||||
}
|
||||
|
||||
func TestIndexRejectsNonParquet(t *testing.T) {
|
||||
data := []byte("this is not a parquet file, not even close, but long enough")
|
||||
if _, err := (Adapter{}).Index(context.Background(), bytes.NewReader(data), int64(len(data))); err == nil {
|
||||
t.Fatalf("Index() accepted junk")
|
||||
}
|
||||
}
|
||||
|
||||
func TestIndexTruncations(t *testing.T) {
|
||||
formattest.IndexTruncations(t, Adapter{}, buildParquet(t, 2, 50))
|
||||
}
|
||||
|
||||
func TestSniff(t *testing.T) {
|
||||
data := buildParquet(t, 1, 10)
|
||||
head, tail := data[:4], data[len(data)-4:]
|
||||
if !(Adapter{}).Sniff(format.Hint{Head: head, Tail: tail}) {
|
||||
t.Fatalf("Sniff() rejected parquet magic")
|
||||
}
|
||||
if (Adapter{}).Sniff(format.Hint{Head: []byte("PAR1"), Tail: []byte("nope")}) {
|
||||
t.Fatalf("Sniff() accepted missing tail magic")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,16 @@
|
||||
package format
|
||||
|
||||
// Adapters register from init(), so the map needs no locking.
|
||||
var formats = map[string]Format{}
|
||||
|
||||
func Register(f Format) {
|
||||
if _, ok := formats[f.Name()]; ok {
|
||||
panic("format: duplicate adapter " + f.Name())
|
||||
}
|
||||
formats[f.Name()] = f
|
||||
}
|
||||
|
||||
// ByName returns the registered adapter, or nil.
|
||||
func ByName(name string) Format {
|
||||
return formats[name]
|
||||
}
|
||||
@@ -15,6 +15,10 @@ type DirEntrySink interface {
|
||||
// AddEntryPlus is AddEntry for readdirplus, returning the attribute block
|
||||
// to fill in, or nil once the sink is full.
|
||||
AddEntryPlus(entry fuse.DirEntry) *fuse.EntryOut
|
||||
|
||||
// TakesLookupRef reports whether AddEntryPlus hands the sink a reference the
|
||||
// mount must hold until a FORGET returns it.
|
||||
TakesLookupRef() bool
|
||||
}
|
||||
|
||||
// fuseDirEntryList adapts the kernel reply buffer to DirEntrySink.
|
||||
@@ -30,6 +34,8 @@ func (l fuseDirEntryList) AddEntryPlus(entry fuse.DirEntry) *fuse.EntryOut {
|
||||
return l.AddDirLookupEntry(entry)
|
||||
}
|
||||
|
||||
func (l fuseDirEntryList) TakesLookupRef() bool { return true }
|
||||
|
||||
// ReadDirectoryInto runs a readdir against sink. ReadDir and ReadDirPlus are
|
||||
// this with the kernel reply buffer as the sink.
|
||||
func (wfs *WFS) ReadDirectoryInto(input *fuse.ReadIn, sink DirEntrySink, isPlusMode bool) fuse.Status {
|
||||
|
||||
@@ -137,6 +137,35 @@ func (i *InodeToPath) Lookup(path util.FullPath, unixTime int64, isDirectory boo
|
||||
return inode
|
||||
}
|
||||
|
||||
// IncrementNlookup takes one more reference on an inode already in the table,
|
||||
// reporting false if it is not there.
|
||||
func (i *InodeToPath) IncrementNlookup(inode uint64) bool {
|
||||
i.Lock()
|
||||
defer i.Unlock()
|
||||
entry, found := i.inode2path[inode]
|
||||
if !found {
|
||||
return false
|
||||
}
|
||||
entry.nlookup++
|
||||
return true
|
||||
}
|
||||
|
||||
// InodeForListing returns the inode number a readdir should report for path
|
||||
// without entering it in the table. Nothing is reserved, so the collision probe
|
||||
// Lookup does is skipped: the worst case is a repeated st_ino in one listing.
|
||||
func (i *InodeToPath) InodeForListing(path util.FullPath, unixTime int64, possibleInode uint64) uint64 {
|
||||
i.RLock()
|
||||
inode, found := i.path2inode[path]
|
||||
i.RUnlock()
|
||||
if found {
|
||||
return inode
|
||||
}
|
||||
if possibleInode != 0 {
|
||||
return possibleInode
|
||||
}
|
||||
return path.AsInode(unixTime)
|
||||
}
|
||||
|
||||
func (i *InodeToPath) AllocateInode(path util.FullPath, unixTime int64) uint64 {
|
||||
if path == "/" {
|
||||
return 1
|
||||
|
||||
@@ -43,6 +43,11 @@ type MetaCache struct {
|
||||
dedupRing dedupRingBuffer
|
||||
includeSystemEntries bool
|
||||
|
||||
// oversizedDirs are directories the mount refused to cache for their size.
|
||||
// Their listings read through to the filer, and a later visit fails fast
|
||||
// instead of streaming to the limit again to rediscover them.
|
||||
oversizedDirs map[util.FullPath]struct{}
|
||||
|
||||
// dirVersionFloors is each cached directory's listing snapshot: the
|
||||
// version of every child the listing covered, present or absent, unless
|
||||
// a later event gave that child its own record. One map write per build
|
||||
@@ -119,6 +124,7 @@ func NewMetaCache(dbFolder string, uidGidMapper *UidGidMapper, root util.FullPat
|
||||
buildingDirs: make(map[util.FullPath]*directoryBuildState),
|
||||
dedupRing: newDedupRingBuffer(),
|
||||
dirVersionFloors: make(map[util.FullPath]int64),
|
||||
oversizedDirs: make(map[util.FullPath]struct{}),
|
||||
}
|
||||
mc.invalidateWorker = util.NewAsyncBatchWorker(func(batch []EntryInvalidation) {
|
||||
for _, invalidation := range batch {
|
||||
@@ -626,7 +632,12 @@ func (mc *MetaCache) deleteFolderChildrenForRebuild(ctx context.Context, dirPath
|
||||
return nil
|
||||
}
|
||||
|
||||
func (mc *MetaCache) ListDirectoryEntries(ctx context.Context, dirPath util.FullPath, startFileName string, includeStartFile bool, limit int64, eachEntryFunc filer.ListEachEntryFunc) error {
|
||||
// ListDirectoryEntries reports the last name the store reached, which is not
|
||||
// the last name handed to eachEntryFunc: an expired child is dropped here after
|
||||
// the store has already spent it against limit. A caller paginating by count
|
||||
// would read a short batch as the end of the directory, and one resuming from
|
||||
// the last name it saw would re-read the dropped ones forever.
|
||||
func (mc *MetaCache) ListDirectoryEntries(ctx context.Context, dirPath util.FullPath, startFileName string, includeStartFile bool, limit int64, eachEntryFunc filer.ListEachEntryFunc) (lastFileName string, err error) {
|
||||
mc.RLock()
|
||||
defer mc.RUnlock()
|
||||
|
||||
@@ -635,17 +646,26 @@ func (mc *MetaCache) ListDirectoryEntries(ctx context.Context, dirPath util.Full
|
||||
glog.Warningf("unsynchronized dir: %v", dirPath)
|
||||
}
|
||||
|
||||
_, err := mc.localStore.ListDirectoryEntries(ctx, dirPath, startFileName, includeStartFile, limit, func(entry *filer.Entry) (bool, error) {
|
||||
return mc.localStore.ListDirectoryEntries(ctx, dirPath, startFileName, includeStartFile, limit, func(entry *filer.Entry) (bool, error) {
|
||||
if entry.TtlSec > 0 && entry.Crtime.Add(time.Duration(entry.TtlSec)*time.Second).Before(time.Now()) {
|
||||
return true, nil
|
||||
}
|
||||
mc.mapIdFromFilerToLocal(entry)
|
||||
return eachEntryFunc(entry)
|
||||
})
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
return err
|
||||
}
|
||||
|
||||
func (mc *MetaCache) markOversized(dirPath util.FullPath) {
|
||||
mc.Lock()
|
||||
defer mc.Unlock()
|
||||
mc.oversizedDirs[dirPath] = struct{}{}
|
||||
}
|
||||
|
||||
func (mc *MetaCache) isOversized(dirPath util.FullPath) bool {
|
||||
mc.RLock()
|
||||
defer mc.RUnlock()
|
||||
_, found := mc.oversizedDirs[dirPath]
|
||||
return found
|
||||
}
|
||||
|
||||
func (mc *MetaCache) Shutdown() {
|
||||
|
||||
@@ -110,7 +110,7 @@ func TestEnsureVisitedReplaysBufferedEventsAfterSnapshot(t *testing.T) {
|
||||
},
|
||||
}
|
||||
|
||||
if err := EnsureVisited(mc, accessor, util.FullPath("/dir")); err != nil {
|
||||
if err := EnsureVisited(mc, accessor, util.FullPath("/dir"), 0); err != nil {
|
||||
t.Fatalf("ensure visited: %v", err)
|
||||
}
|
||||
if applyErr != nil {
|
||||
@@ -504,7 +504,7 @@ func TestEnsureVisitedPreservesLocalOnlyEntry(t *testing.T) {
|
||||
}},
|
||||
}}
|
||||
|
||||
if err := EnsureVisited(mc, accessor, util.FullPath("/dir")); err != nil {
|
||||
if err := EnsureVisited(mc, accessor, util.FullPath("/dir"), 0); err != nil {
|
||||
t.Fatalf("ensure visited: %v", err)
|
||||
}
|
||||
if !mc.IsDirectoryCached(util.FullPath("/dir")) {
|
||||
@@ -550,7 +550,7 @@ func TestEnsureVisitedDropsUnpinnedStaleEntry(t *testing.T) {
|
||||
}},
|
||||
}}
|
||||
|
||||
if err := EnsureVisited(mc, accessor, util.FullPath("/dir")); err != nil {
|
||||
if err := EnsureVisited(mc, accessor, util.FullPath("/dir"), 0); err != nil {
|
||||
t.Fatalf("ensure visited: %v", err)
|
||||
}
|
||||
if entry, _, err := mc.FindEntry(context.Background(), util.FullPath("/dir/stale.txt")); err != filer_pb.ErrNotFound || entry != nil {
|
||||
@@ -597,7 +597,7 @@ func TestEnsureVisitedConfirmsTransientEmptyListing(t *testing.T) {
|
||||
},
|
||||
}}
|
||||
|
||||
if err := EnsureVisited(mc, accessor, util.FullPath("/dir")); err != nil {
|
||||
if err := EnsureVisited(mc, accessor, util.FullPath("/dir"), 0); err != nil {
|
||||
t.Fatalf("ensure visited: %v", err)
|
||||
}
|
||||
if !mc.IsDirectoryCached(util.FullPath("/dir")) {
|
||||
@@ -619,7 +619,7 @@ func TestEnsureVisitedCachesGenuinelyEmptyDirectory(t *testing.T) {
|
||||
}
|
||||
accessor := &buildFilerAccessor{client: client}
|
||||
|
||||
if err := EnsureVisited(mc, accessor, util.FullPath("/empty")); err != nil {
|
||||
if err := EnsureVisited(mc, accessor, util.FullPath("/empty"), 0); err != nil {
|
||||
t.Fatalf("ensure visited: %v", err)
|
||||
}
|
||||
if !mc.IsDirectoryCached(util.FullPath("/empty")) {
|
||||
|
||||
@@ -2,6 +2,7 @@ package meta_cache
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"time"
|
||||
|
||||
@@ -13,7 +14,19 @@ import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
)
|
||||
|
||||
func EnsureVisited(mc *MetaCache, client filer_pb.FilerClient, dirPath util.FullPath) error {
|
||||
// DirectoryTooLargeError reports a directory the mount refuses to cache
|
||||
// locally. Its listings read through to the filer instead.
|
||||
type DirectoryTooLargeError struct {
|
||||
Path util.FullPath
|
||||
}
|
||||
|
||||
func (e *DirectoryTooLargeError) Error() string {
|
||||
return fmt.Sprintf("directory %s is too large to cache locally", e.Path)
|
||||
}
|
||||
|
||||
// maxCacheableEntries is the directory size above which a build gives up, or 0
|
||||
// to cache everything.
|
||||
func EnsureVisited(mc *MetaCache, client filer_pb.FilerClient, dirPath util.FullPath, maxCacheableEntries int) error {
|
||||
// Collect all uncached paths from target directory up to root
|
||||
var uncachedPaths []util.FullPath
|
||||
currentPath := dirPath
|
||||
@@ -23,7 +36,15 @@ func EnsureVisited(mc *MetaCache, client filer_pb.FilerClient, dirPath util.Full
|
||||
if mc.isCachedFn(currentPath) {
|
||||
break
|
||||
}
|
||||
uncachedPaths = append(uncachedPaths, currentPath)
|
||||
if mc.isOversized(currentPath) {
|
||||
// The directory itself reads through; an ancestor is stepped over,
|
||||
// or it would wedge every listing beneath it forever.
|
||||
if currentPath == dirPath {
|
||||
return &DirectoryTooLargeError{Path: currentPath}
|
||||
}
|
||||
} else {
|
||||
uncachedPaths = append(uncachedPaths, currentPath)
|
||||
}
|
||||
|
||||
// Continue to parent directory
|
||||
if currentPath != mc.root {
|
||||
@@ -44,7 +65,16 @@ func EnsureVisited(mc *MetaCache, client filer_pb.FilerClient, dirPath util.Full
|
||||
for _, p := range uncachedPaths {
|
||||
path := p // capture for closure
|
||||
g.Go(func() error {
|
||||
return doEnsureVisited(ctx, mc, client, path)
|
||||
err := doEnsureVisited(ctx, mc, client, path, maxCacheableEntries)
|
||||
var tooLarge *DirectoryTooLargeError
|
||||
if errors.As(err, &tooLarge) && path != dirPath {
|
||||
// An ancestor found oversized just reads through; failing the
|
||||
// group here would cancel the builds of its cacheable
|
||||
// descendants, and the caller would treat the refusal as the
|
||||
// listed directory's own.
|
||||
return nil
|
||||
}
|
||||
return err
|
||||
})
|
||||
}
|
||||
return g.Wait()
|
||||
@@ -60,7 +90,7 @@ const (
|
||||
emptyRebuildConfirmDelay = 50 * time.Millisecond
|
||||
)
|
||||
|
||||
func doEnsureVisited(ctx context.Context, mc *MetaCache, client filer_pb.FilerClient, path util.FullPath) error {
|
||||
func doEnsureVisited(ctx context.Context, mc *MetaCache, client filer_pb.FilerClient, path util.FullPath, maxCacheableEntries int) error {
|
||||
// Use singleflight to deduplicate concurrent requests for the same path
|
||||
_, err, _ := mc.visitGroup.Do(string(path), func() (interface{}, error) {
|
||||
// Check for cancellation before starting
|
||||
@@ -116,6 +146,9 @@ func doEnsureVisited(ctx context.Context, mc *MetaCache, client filer_pb.FilerCl
|
||||
return nil
|
||||
}
|
||||
|
||||
if maxCacheableEntries > 0 && entryCount >= maxCacheableEntries {
|
||||
return &DirectoryTooLargeError{Path: path}
|
||||
}
|
||||
batch = append(batch, entry)
|
||||
entryCount++
|
||||
|
||||
@@ -143,6 +176,15 @@ func doEnsureVisited(ctx context.Context, mc *MetaCache, client filer_pb.FilerCl
|
||||
|
||||
entryCount, snapshotTsNs, fetchErr := reloadFromFiler()
|
||||
if fetchErr != nil {
|
||||
var tooLarge *DirectoryTooLargeError
|
||||
if errors.As(fetchErr, &tooLarge) {
|
||||
// Remember the refusal so the next visit fails fast instead of
|
||||
// streaming up to the limit again to rediscover it.
|
||||
mc.markOversized(path)
|
||||
glog.V(0).Infof("directory %s exceeds %d entries, reading it through instead of caching", path, maxCacheableEntries)
|
||||
cleanupBuild("oversized")
|
||||
return nil, fetchErr
|
||||
}
|
||||
cleanupBuild("failed")
|
||||
return nil, fmt.Errorf("list %s: %w", path, fetchErr)
|
||||
}
|
||||
|
||||
@@ -0,0 +1,138 @@
|
||||
package meta_cache
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"fmt"
|
||||
"testing"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/filer"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
"google.golang.org/grpc"
|
||||
)
|
||||
|
||||
func listResponses(n int) []*filer_pb.ListEntriesResponse {
|
||||
responses := make([]*filer_pb.ListEntriesResponse, 0, n)
|
||||
for i := 0; i < n; i++ {
|
||||
responses = append(responses, &filer_pb.ListEntriesResponse{
|
||||
Entry: &filer_pb.Entry{
|
||||
Name: fmt.Sprintf("f%05d", i),
|
||||
Attributes: &filer_pb.FuseAttributes{
|
||||
Crtime: 1, Mtime: 1, FileMode: 0o644, FileSize: 3,
|
||||
},
|
||||
},
|
||||
})
|
||||
}
|
||||
return responses
|
||||
}
|
||||
|
||||
// TestEnsureVisitedRefusesOversizedDirectory checks that a directory past the
|
||||
// limit is not cached, that the refusal is remembered, and that the partial
|
||||
// build leaves nothing behind in the local store.
|
||||
func TestEnsureVisitedRefusesOversizedDirectory(t *testing.T) {
|
||||
mc, _, _, _ := newTestMetaCache(t, map[util.FullPath]bool{"/": true})
|
||||
defer mc.Shutdown()
|
||||
|
||||
accessor := &buildFilerAccessor{client: &buildListClient{responses: listResponses(10)}}
|
||||
|
||||
err := EnsureVisited(mc, accessor, util.FullPath("/dir"), 5)
|
||||
var tooLarge *DirectoryTooLargeError
|
||||
if !errors.As(err, &tooLarge) {
|
||||
t.Fatalf("EnsureVisited = %v, want DirectoryTooLargeError", err)
|
||||
}
|
||||
if tooLarge.Path != util.FullPath("/dir") {
|
||||
t.Fatalf("oversized path = %s, want /dir", tooLarge.Path)
|
||||
}
|
||||
if mc.IsDirectoryCached(util.FullPath("/dir")) {
|
||||
t.Error("oversized directory reported cached")
|
||||
}
|
||||
// The aborted build must leave no partial children to be served later.
|
||||
count := 0
|
||||
if _, err := mc.ListDirectoryEntries(context.Background(), util.FullPath("/dir"), "", false, 100, func(e *filer.Entry) (bool, error) {
|
||||
count++
|
||||
return true, nil
|
||||
}); err != nil {
|
||||
t.Fatalf("list: %v", err)
|
||||
}
|
||||
if count != 0 {
|
||||
t.Errorf("local store still holds %d children of the aborted build", count)
|
||||
}
|
||||
|
||||
// A second visit fails fast without streaming to the limit again.
|
||||
if err := EnsureVisited(mc, accessor, util.FullPath("/dir"), 5); !errors.As(err, &tooLarge) {
|
||||
t.Fatalf("second EnsureVisited = %v, want DirectoryTooLargeError", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestEnsureVisitedStepsOverOversizedAncestor checks a huge ancestor does not
|
||||
// wedge the caching of its subdirectories.
|
||||
func TestEnsureVisitedStepsOverOversizedAncestor(t *testing.T) {
|
||||
mc, _, _, _ := newTestMetaCache(t, map[util.FullPath]bool{"/": true})
|
||||
defer mc.Shutdown()
|
||||
mc.markOversized(util.FullPath("/huge"))
|
||||
|
||||
accessor := &buildFilerAccessor{client: &buildListClient{responses: listResponses(3)}}
|
||||
if err := EnsureVisited(mc, accessor, util.FullPath("/huge/sub"), 5); err != nil {
|
||||
t.Fatalf("EnsureVisited under an oversized ancestor: %v", err)
|
||||
}
|
||||
if !mc.IsDirectoryCached(util.FullPath("/huge/sub")) {
|
||||
t.Error("subdirectory of an oversized ancestor was not cached")
|
||||
}
|
||||
if mc.IsDirectoryCached(util.FullPath("/huge")) {
|
||||
t.Error("oversized ancestor became cached")
|
||||
}
|
||||
}
|
||||
|
||||
// TestEnsureVisitedUnderTheLimitStillCaches pins that the gate does not change
|
||||
// behaviour for ordinary directories.
|
||||
func TestEnsureVisitedUnderTheLimitStillCaches(t *testing.T) {
|
||||
mc, _, _, _ := newTestMetaCache(t, map[util.FullPath]bool{"/": true})
|
||||
defer mc.Shutdown()
|
||||
|
||||
accessor := &buildFilerAccessor{client: &buildListClient{responses: listResponses(5)}}
|
||||
if err := EnsureVisited(mc, accessor, util.FullPath("/dir"), 5); err != nil {
|
||||
t.Fatalf("EnsureVisited: %v", err)
|
||||
}
|
||||
if !mc.IsDirectoryCached(util.FullPath("/dir")) {
|
||||
t.Error("directory at the limit was not cached")
|
||||
}
|
||||
}
|
||||
|
||||
// pathListClient serves canned listings per directory, so one visit can see
|
||||
// directories of different sizes.
|
||||
type pathListClient struct {
|
||||
filer_pb.SeaweedFilerClient
|
||||
perDir map[string][]*filer_pb.ListEntriesResponse
|
||||
}
|
||||
|
||||
func (c *pathListClient) ListEntries(ctx context.Context, in *filer_pb.ListEntriesRequest, opts ...grpc.CallOption) (grpc.ServerStreamingClient[filer_pb.ListEntriesResponse], error) {
|
||||
return &buildListStream{responses: c.perDir[in.Directory]}, nil
|
||||
}
|
||||
|
||||
// TestEnsureVisitedAncestorFoundOversizedMidVisit covers the first discovery:
|
||||
// the ancestor's refusal must neither cancel the descendant's build nor be
|
||||
// reported as the descendant's own.
|
||||
func TestEnsureVisitedAncestorFoundOversizedMidVisit(t *testing.T) {
|
||||
mc, _, _, _ := newTestMetaCache(t, map[util.FullPath]bool{"/": true})
|
||||
defer mc.Shutdown()
|
||||
|
||||
accessor := &buildFilerAccessor{client: &pathListClient{perDir: map[string][]*filer_pb.ListEntriesResponse{
|
||||
"/huge": listResponses(10),
|
||||
"/huge/sub": listResponses(3),
|
||||
}}}
|
||||
|
||||
if err := EnsureVisited(mc, accessor, util.FullPath("/huge/sub"), 5); err != nil {
|
||||
t.Fatalf("EnsureVisited: %v", err)
|
||||
}
|
||||
if !mc.IsDirectoryCached(util.FullPath("/huge/sub")) {
|
||||
t.Error("descendant of a just-discovered oversized ancestor was not cached")
|
||||
}
|
||||
if mc.IsDirectoryCached(util.FullPath("/huge")) {
|
||||
t.Error("oversized ancestor became cached")
|
||||
}
|
||||
if !mc.isOversized(util.FullPath("/huge")) {
|
||||
t.Error("oversized ancestor was not remembered")
|
||||
}
|
||||
}
|
||||
@@ -53,6 +53,7 @@ type Option struct {
|
||||
CacheDirForWrite string
|
||||
WriteBufferSizeMB int64
|
||||
CacheMetaTTlSec int
|
||||
CacheDirMaxEntries int
|
||||
DataCenter string
|
||||
Umask os.FileMode
|
||||
Quota int64
|
||||
@@ -801,9 +802,33 @@ func (wfs *WFS) onEntryInvalidation(invalidation meta_cache.EntryInvalidation) {
|
||||
if listener != nil {
|
||||
listener(invalidation)
|
||||
}
|
||||
wfs.invalidateKernelDirListing(invalidation.Path)
|
||||
wfs.invalidateOpenFileHandle(invalidation)
|
||||
}
|
||||
|
||||
// invalidateKernelDirListing drops the kernel's cached listing of the directory
|
||||
// holding path. Safe here because invalidations run on their own worker, never
|
||||
// on a thread serving a kernel request; notifying from a handler can deadlock
|
||||
// against the page it holds, which is why the file paths avoid InodeNotify.
|
||||
// A directory the kernel has not looked up has no inode here and nothing
|
||||
// cached, so it is skipped.
|
||||
func (wfs *WFS) invalidateKernelDirListing(path util.FullPath) {
|
||||
server := wfs.fuseServer
|
||||
if server == nil {
|
||||
return
|
||||
}
|
||||
dir, _ := path.DirAndName()
|
||||
dirInode, found := wfs.inodeToPath.GetInode(util.FullPath(dir))
|
||||
if !found {
|
||||
return
|
||||
}
|
||||
// ENOENT is the kernel not holding the inode, ENOSYS a kernel without the
|
||||
// notify; neither is worth a line.
|
||||
if status := server.InodeNotify(dirInode, 0, -1); status != fuse.OK && status != fuse.ENOENT && status != fuse.ENOSYS {
|
||||
glog.V(4).Infof("invalidate kernel listing of %s: %v", dir, status)
|
||||
}
|
||||
}
|
||||
|
||||
// MountRoot is the filer path this mount is rooted at. Event paths are absolute
|
||||
// on the filer; a front end that addresses files relative to the mount needs it
|
||||
// to translate them.
|
||||
|
||||
+113
-17
@@ -2,6 +2,7 @@ package mount
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
@@ -20,6 +21,11 @@ const (
|
||||
batchSize = 1000
|
||||
)
|
||||
|
||||
// readdirContext marks the meta cache listing as reading attributes only. A
|
||||
// readdir never looks at a chunk list, and building one per child is most of
|
||||
// the cost of decoding a wide directory.
|
||||
var readdirContext = filer_pb.WithChunksOmitted(context.Background())
|
||||
|
||||
// DirectoryHandle represents an open directory handle.
|
||||
// It maintains state for directory listing pagination and is protected by a mutex
|
||||
// to handle concurrent readdir operations from NFS-Ganesha and other multi-threaded clients.
|
||||
@@ -28,11 +34,15 @@ type DirectoryHandle struct {
|
||||
isFinished bool
|
||||
entryStream []*filer.Entry
|
||||
entryStreamOffset uint64
|
||||
snapshotTsNs int64 // snapshot timestamp for consistent readdir in direct mode
|
||||
// lastListedName is how far the store itself reached, which runs ahead of
|
||||
// the last visible entry whenever children are dropped as expired.
|
||||
lastListedName string
|
||||
snapshotTsNs int64 // snapshot timestamp for consistent readdir in direct mode
|
||||
}
|
||||
|
||||
func (dh *DirectoryHandle) reset() {
|
||||
dh.isFinished = false
|
||||
dh.lastListedName = ""
|
||||
dh.snapshotTsNs = 0
|
||||
// Nil out pointers to allow garbage collection of old entries,
|
||||
// then reuse the slice's capacity to avoid re-allocations.
|
||||
@@ -99,6 +109,12 @@ func (wfs *WFS) OpenDir(cancel <-chan struct{}, input *fuse.OpenIn, out *fuse.Op
|
||||
}
|
||||
dhid, _ := wfs.AcquireDirectoryHandle()
|
||||
out.Fh = uint64(dhid)
|
||||
// Let the kernel keep the listing in the directory's page cache, so
|
||||
// reopening the directory does not reach the mount at all. Local mutations
|
||||
// drop that cache in the kernel; remote ones arrive through the metadata
|
||||
// subscription, which notifies the kernel per changed directory. A kernel
|
||||
// too old for the flag ignores it.
|
||||
out.OpenFlags |= fuse.FOPEN_CACHE_DIR | fuse.FOPEN_KEEP_CACHE
|
||||
return fuse.OK
|
||||
}
|
||||
|
||||
@@ -156,6 +172,11 @@ func (wfs *WFS) doReadDirectory(input *fuse.ReadIn, out DirEntrySink, isPlusMode
|
||||
|
||||
if input.Offset == 0 {
|
||||
dh.reset()
|
||||
} else if input.Offset < dh.entryStreamOffset {
|
||||
// Seeking back before what the handle still holds. Start the directory
|
||||
// again rather than reporting nothing; the preload below refills up to
|
||||
// the requested offset.
|
||||
dh.reset()
|
||||
} else if dh.isFinished && input.Offset >= dh.entryStreamOffset {
|
||||
entryCurrentIndex := input.Offset - dh.entryStreamOffset
|
||||
if uint64(len(dh.entryStream)) <= entryCurrentIndex {
|
||||
@@ -170,12 +191,21 @@ func (wfs *WFS) doReadDirectory(input *fuse.ReadIn, out DirEntrySink, isPlusMode
|
||||
wfs.inodeToPath.TouchDirectory(dirPath)
|
||||
|
||||
var dirEntry fuse.DirEntry
|
||||
// Only a reference makes a child worth entering in the inode table: without
|
||||
// one nothing ever arrives to take the entry back out again.
|
||||
takesLookupRef := isPlusMode && out.TakesLookupRef()
|
||||
|
||||
// index is the position in entryStream, used to calculate the offset for next readdir
|
||||
processEachEntryFn := func(entry *filer.Entry, index int64) bool {
|
||||
dirEntry.Name = entry.Name()
|
||||
dirEntry.Mode = toSyscallMode(entry.Mode)
|
||||
inode := wfs.inodeToPath.Lookup(dirPath.Child(dirEntry.Name), entry.Crtime.Unix(), entry.IsDirectory(), len(entry.HardLinkId) > 0, entry.Inode, false)
|
||||
childPath := dirPath.Child(dirEntry.Name)
|
||||
var inode uint64
|
||||
if takesLookupRef {
|
||||
inode = wfs.inodeToPath.Lookup(childPath, entry.Crtime.Unix(), entry.IsDirectory(), len(entry.HardLinkId) > 0, entry.Inode, false)
|
||||
} else {
|
||||
inode = wfs.inodeToPath.InodeForListing(childPath, entry.Crtime.Unix(), entry.Inode)
|
||||
}
|
||||
dirEntry.Ino = inode
|
||||
|
||||
// Set Off to the next offset so client can resume from correct position
|
||||
@@ -191,11 +221,15 @@ func (wfs *WFS) doReadDirectory(input *fuse.ReadIn, out DirEntrySink, isPlusMode
|
||||
return false
|
||||
}
|
||||
if fh, found := wfs.fhMap.FindFileHandle(inode); found {
|
||||
glog.V(4).Infof("readdir opened file %s", dirPath.Child(dirEntry.Name))
|
||||
glog.V(4).Infof("readdir opened file %s", childPath)
|
||||
entry = filer.FromPbEntry(string(dirPath), fh.GetEntry().GetEntry())
|
||||
}
|
||||
wfs.outputFilerEntry(entryOut, inode, entry)
|
||||
wfs.inodeToPath.Lookup(dirPath.Child(dirEntry.Name), entry.Crtime.Unix(), entry.IsDirectory(), len(entry.HardLinkId) > 0, entry.Inode, true)
|
||||
// Taken only once the entry is really in the sink, so one that did not
|
||||
// fit leaves no reference behind. The fallback covers a racing Forget.
|
||||
if takesLookupRef && !wfs.inodeToPath.IncrementNlookup(inode) {
|
||||
wfs.inodeToPath.Lookup(childPath, entry.Crtime.Unix(), entry.IsDirectory(), len(entry.HardLinkId) > 0, entry.Inode, true)
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
@@ -223,21 +257,39 @@ func (wfs *WFS) doReadDirectory(input *fuse.ReadIn, out DirEntrySink, isPlusMode
|
||||
|
||||
// Read from cache first, then load next batch if needed
|
||||
if input.Offset >= dh.entryStreamOffset {
|
||||
// Drop what the client has walked past. Offsets are indexes into the
|
||||
// stream from entryStreamOffset, so advancing the two together keeps
|
||||
// them lined up; one entry is kept back because the next batch resumes
|
||||
// from the name immediately before the offset.
|
||||
if trim := int(input.Offset-dh.entryStreamOffset) - 1; trim > 0 && trim <= len(dh.entryStream) {
|
||||
copy(dh.entryStream, dh.entryStream[trim:])
|
||||
for i := len(dh.entryStream) - trim; i < len(dh.entryStream); i++ {
|
||||
dh.entryStream[i] = nil
|
||||
}
|
||||
dh.entryStream = dh.entryStream[:len(dh.entryStream)-trim]
|
||||
dh.entryStreamOffset += uint64(trim)
|
||||
}
|
||||
|
||||
// Handle case: new handle with non-zero offset but empty cache
|
||||
// This happens when NFS-Ganesha opens multiple directory handles
|
||||
if len(dh.entryStream) == 0 && input.Offset > dh.entryStreamOffset {
|
||||
skipCount := int64(input.Offset - dh.entryStreamOffset)
|
||||
|
||||
if err := meta_cache.EnsureVisited(wfs.metaCache, wfs, dirPath); err != nil {
|
||||
if err := wfs.ensureDirectoryVisited(dirPath); err != nil {
|
||||
var tooLarge *meta_cache.DirectoryTooLargeError
|
||||
if errors.As(err, &tooLarge) {
|
||||
return wfs.readDirectoryDirect(input, out, dh, dirPath, processEachEntryFn)
|
||||
}
|
||||
glog.Errorf("dir ReadDirAll %s: %v", dirPath, err)
|
||||
return fuse.EIO
|
||||
}
|
||||
|
||||
// Load entries from beginning to fill cache up to the requested offset
|
||||
loadErr := wfs.metaCache.ListDirectoryEntries(context.Background(), dirPath, "", false, skipCount+int64(batchSize), func(entry *filer.Entry) (bool, error) {
|
||||
storeLastName, loadErr := wfs.metaCache.ListDirectoryEntries(readdirContext, dirPath, "", false, skipCount+int64(batchSize), func(entry *filer.Entry) (bool, error) {
|
||||
dh.entryStream = append(dh.entryStream, entry)
|
||||
return true, nil
|
||||
})
|
||||
dh.lastListedName = storeLastName
|
||||
if loadErr != nil {
|
||||
glog.Errorf("list meta cache: %v", loadErr)
|
||||
return fuse.EIO
|
||||
@@ -248,6 +300,13 @@ func (wfs *WFS) doReadDirectory(input *fuse.ReadIn, out DirEntrySink, isPlusMode
|
||||
entryPreviousIndex := (input.Offset - dh.entryStreamOffset) - 1
|
||||
if uint64(len(dh.entryStream)) > entryPreviousIndex {
|
||||
lastEntryName = dh.entryStream[entryPreviousIndex].Name()
|
||||
} else {
|
||||
// The stream runs from the directory's first child, so failing to
|
||||
// reach the entry before this offset means the directory has since
|
||||
// shrunk past it. Listing on from an empty name would replay the
|
||||
// directory from the start and hand the client every name twice.
|
||||
dh.isFinished = true
|
||||
return fuse.OK
|
||||
}
|
||||
}
|
||||
|
||||
@@ -263,18 +322,29 @@ func (wfs *WFS) doReadDirectory(input *fuse.ReadIn, out DirEntrySink, isPlusMode
|
||||
}
|
||||
|
||||
// Cache exhausted, load next batch
|
||||
if err := meta_cache.EnsureVisited(wfs.metaCache, wfs, dirPath); err != nil {
|
||||
if err := wfs.ensureDirectoryVisited(dirPath); err != nil {
|
||||
var tooLarge *meta_cache.DirectoryTooLargeError
|
||||
if errors.As(err, &tooLarge) {
|
||||
// The direct path keeps the same pagination state on dh, so it
|
||||
// carries on from wherever the cached walk reached.
|
||||
return wfs.readDirectoryDirect(input, out, dh, dirPath, processEachEntryFn)
|
||||
}
|
||||
glog.Errorf("dir ReadDirAll %s: %v", dirPath, err)
|
||||
return fuse.EIO
|
||||
}
|
||||
|
||||
// Batch loading: fetch batchSize entries starting from lastEntryName
|
||||
loadedCount := 0
|
||||
// Page from where the store itself reached, not from the last entry the
|
||||
// sink saw. An expired child is counted against the batch and then
|
||||
// dropped, so resuming from the last visible name would re-read it every
|
||||
// round and never get past a batch that was entirely expired.
|
||||
if dh.lastListedName > lastEntryName {
|
||||
lastEntryName = dh.lastListedName
|
||||
}
|
||||
|
||||
bufferFull := false
|
||||
loadErr := wfs.metaCache.ListDirectoryEntries(context.Background(), dirPath, lastEntryName, false, int64(batchSize), func(entry *filer.Entry) (bool, error) {
|
||||
storeLastName, loadErr := wfs.metaCache.ListDirectoryEntries(readdirContext, dirPath, lastEntryName, false, int64(batchSize), func(entry *filer.Entry) (bool, error) {
|
||||
currentIndex := int64(len(dh.entryStream))
|
||||
dh.entryStream = append(dh.entryStream, entry)
|
||||
loadedCount++
|
||||
if !processEachEntryFn(entry, currentIndex) {
|
||||
bufferFull = true
|
||||
return false, nil
|
||||
@@ -285,10 +355,12 @@ func (wfs *WFS) doReadDirectory(input *fuse.ReadIn, out DirEntrySink, isPlusMode
|
||||
glog.Errorf("list meta cache: %v", loadErr)
|
||||
return fuse.EIO
|
||||
}
|
||||
dh.lastListedName = storeLastName
|
||||
|
||||
// Mark finished only when loading completed normally (not buffer full)
|
||||
// and we got fewer entries than requested
|
||||
if !bufferFull && loadedCount < batchSize {
|
||||
// The store reaching nothing is the only sound end-of-directory signal:
|
||||
// a batch can come back short because entries expired, not because the
|
||||
// directory ran out.
|
||||
if !bufferFull && storeLastName == "" {
|
||||
dh.isFinished = true
|
||||
}
|
||||
}
|
||||
@@ -296,13 +368,25 @@ func (wfs *WFS) doReadDirectory(input *fuse.ReadIn, out DirEntrySink, isPlusMode
|
||||
return fuse.OK
|
||||
}
|
||||
|
||||
// ensureDirectoryVisited pulls the directory into the local cache, unless it is
|
||||
// too large to cache: then the directory is marked read-through, so later
|
||||
// listings go straight to the filer without re-asking.
|
||||
func (wfs *WFS) ensureDirectoryVisited(dirPath util.FullPath) error {
|
||||
err := meta_cache.EnsureVisited(wfs.metaCache, wfs, dirPath, wfs.option.CacheDirMaxEntries)
|
||||
var tooLarge *meta_cache.DirectoryTooLargeError
|
||||
if errors.As(err, &tooLarge) {
|
||||
wfs.inodeToPath.MarkDirectoryReadThrough(dirPath, time.Now())
|
||||
}
|
||||
return err
|
||||
}
|
||||
|
||||
func (wfs *WFS) readDirectoryDirect(input *fuse.ReadIn, out DirEntrySink, dh *DirectoryHandle, dirPath util.FullPath, processEachEntryFn func(entry *filer.Entry, index int64) bool) fuse.Status {
|
||||
var lastEntryName string
|
||||
|
||||
if input.Offset >= dh.entryStreamOffset {
|
||||
if len(dh.entryStream) == 0 && input.Offset > dh.entryStreamOffset {
|
||||
skipCount := uint32(input.Offset-dh.entryStreamOffset) + batchSize
|
||||
entries, snapshotTs, err := loadDirectoryEntriesDirect(context.Background(), wfs, wfs.option.UidGidMapper, dirPath, "", false, skipCount, dh.snapshotTsNs, wfs.option.IncludeSystemEntries)
|
||||
entries, snapshotTs, err := loadDirectoryEntriesDirect(readdirContext, wfs, wfs.option.UidGidMapper, dirPath, "", false, skipCount, dh.snapshotTsNs, wfs.option.IncludeSystemEntries)
|
||||
if err != nil {
|
||||
glog.Errorf("list filer directory: %v", err)
|
||||
return fuse.EIO
|
||||
@@ -317,6 +401,11 @@ func (wfs *WFS) readDirectoryDirect(input *fuse.ReadIn, out DirEntrySink, dh *Di
|
||||
entryPreviousIndex := (input.Offset - dh.entryStreamOffset) - 1
|
||||
if uint64(len(dh.entryStream)) > entryPreviousIndex {
|
||||
lastEntryName = dh.entryStream[entryPreviousIndex].Name()
|
||||
} else {
|
||||
// See the cached path: the directory shrank past this offset, and
|
||||
// resuming from an empty name would replay it from the start.
|
||||
dh.isFinished = true
|
||||
return fuse.OK
|
||||
}
|
||||
}
|
||||
|
||||
@@ -331,7 +420,7 @@ func (wfs *WFS) readDirectoryDirect(input *fuse.ReadIn, out DirEntrySink, dh *Di
|
||||
}
|
||||
}
|
||||
|
||||
entries, snapshotTs, err := loadDirectoryEntriesDirect(context.Background(), wfs, wfs.option.UidGidMapper, dirPath, lastEntryName, false, batchSize, dh.snapshotTsNs, wfs.option.IncludeSystemEntries)
|
||||
entries, snapshotTs, err := loadDirectoryEntriesDirect(readdirContext, wfs, wfs.option.UidGidMapper, dirPath, lastEntryName, false, batchSize, dh.snapshotTsNs, wfs.option.IncludeSystemEntries)
|
||||
if err != nil {
|
||||
glog.Errorf("list filer directory: %v", err)
|
||||
return fuse.EIO
|
||||
@@ -361,7 +450,14 @@ func (wfs *WFS) readDirectoryDirect(input *fuse.ReadIn, out DirEntrySink, dh *Di
|
||||
}
|
||||
|
||||
func loadDirectoryEntriesDirect(ctx context.Context, client filer_pb.FilerClient, uidGidMapper *meta_cache.UidGidMapper, dirPath util.FullPath, startFileName string, includeStart bool, limit uint32, snapshotTsNs int64, includeSystemEntries bool) ([]*filer.Entry, int64, error) {
|
||||
entries := make([]*filer.Entry, 0, limit)
|
||||
// limit can be a client-supplied resume offset rather than a batch size, so
|
||||
// preallocating for it would size the slice from where the caller happened to
|
||||
// seek. Reserve a batch and let append find the rest.
|
||||
prealloc := limit
|
||||
if prealloc > batchSize {
|
||||
prealloc = batchSize
|
||||
}
|
||||
entries := make([]*filer.Entry, 0, prealloc)
|
||||
var actualSnapshotTsNs int64
|
||||
err := client.WithFilerClient(false, func(sc filer_pb.SeaweedFilerClient) error {
|
||||
var innerErr error
|
||||
|
||||
@@ -0,0 +1,233 @@
|
||||
package mount
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/go-fuse/v2/fuse"
|
||||
"github.com/seaweedfs/seaweedfs/weed/filer"
|
||||
"github.com/seaweedfs/seaweedfs/weed/mount/meta_cache"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
"google.golang.org/grpc"
|
||||
"google.golang.org/grpc/credentials/insecure"
|
||||
)
|
||||
|
||||
const benchDirEntryCount = 200000
|
||||
|
||||
// benchSink drains a readdir the way a front end does. sinkLimit is how many
|
||||
// entries one round accepts, standing in for the kernel reply buffer or the
|
||||
// WinFsp adapter's batch.
|
||||
type benchSink struct {
|
||||
plus bool
|
||||
takesRef bool
|
||||
sinkLimit int
|
||||
count int
|
||||
inodes []uint64
|
||||
lastOff uint64
|
||||
attrs []fuse.EntryOut
|
||||
}
|
||||
|
||||
func (s *benchSink) reset() {
|
||||
s.count = 0
|
||||
s.inodes = s.inodes[:0]
|
||||
s.attrs = s.attrs[:0]
|
||||
}
|
||||
|
||||
func (s *benchSink) AddEntry(entry fuse.DirEntry) bool {
|
||||
if s.count >= s.sinkLimit {
|
||||
return false
|
||||
}
|
||||
s.count++
|
||||
s.lastOff = entry.Off
|
||||
s.inodes = append(s.inodes, entry.Ino)
|
||||
return true
|
||||
}
|
||||
|
||||
func (s *benchSink) AddEntryPlus(entry fuse.DirEntry) *fuse.EntryOut {
|
||||
if s.count >= s.sinkLimit {
|
||||
return nil
|
||||
}
|
||||
s.count++
|
||||
s.lastOff = entry.Off
|
||||
s.inodes = append(s.inodes, entry.Ino)
|
||||
s.attrs = append(s.attrs, fuse.EntryOut{})
|
||||
return &s.attrs[len(s.attrs)-1]
|
||||
}
|
||||
|
||||
func (s *benchSink) TakesLookupRef() bool { return s.takesRef }
|
||||
|
||||
func inodeTableSize(i *InodeToPath) int {
|
||||
i.RLock()
|
||||
defer i.RUnlock()
|
||||
return len(i.inode2path)
|
||||
}
|
||||
|
||||
func newBenchWFS(tb testing.TB, dir util.FullPath, n int) *WFS {
|
||||
tb.Helper()
|
||||
|
||||
uidGidMapper, err := meta_cache.NewUidGidMapper("", "")
|
||||
if err != nil {
|
||||
tb.Fatalf("uid/gid mapper: %v", err)
|
||||
}
|
||||
|
||||
root := util.FullPath("/")
|
||||
option := &Option{
|
||||
ChunkSizeLimit: 1024,
|
||||
ConcurrentReaders: 1,
|
||||
VolumeServerAccess: "filerProxy",
|
||||
FilerAddresses: []pb.ServerAddress{pb.NewServerAddressWithGrpcPort("127.0.0.1:1", 1)},
|
||||
GrpcDialOption: grpc.WithTransportCredentials(insecure.NewCredentials()),
|
||||
FilerMountRootPath: "/",
|
||||
MountUid: 99,
|
||||
MountGid: 100,
|
||||
MountMode: 0o777,
|
||||
MountMtime: time.Now(),
|
||||
MountCtime: time.Now(),
|
||||
UidGidMapper: uidGidMapper,
|
||||
}
|
||||
|
||||
wfs := &WFS{
|
||||
option: option,
|
||||
signature: 1,
|
||||
inodeToPath: NewInodeToPath(root, 0),
|
||||
fhMap: NewFileHandleToInode(),
|
||||
dhMap: NewDirectoryHandleToInode(),
|
||||
fhLockTable: util.NewLockTable[FileHandleId](),
|
||||
hardLinkLockTable: util.NewLockTable[string](),
|
||||
}
|
||||
wfs.metaCache = meta_cache.NewMetaCache(
|
||||
filepath.Join(tb.TempDir(), "meta"),
|
||||
uidGidMapper,
|
||||
root,
|
||||
false,
|
||||
func(path util.FullPath) { wfs.inodeToPath.MarkChildrenCached(path) },
|
||||
func(path util.FullPath) bool { return wfs.inodeToPath.IsChildrenCached(path) },
|
||||
func(meta_cache.EntryInvalidation) {},
|
||||
nil,
|
||||
)
|
||||
tb.Cleanup(wfs.metaCache.Shutdown)
|
||||
|
||||
now := time.Now()
|
||||
ctx := context.Background()
|
||||
if err := wfs.metaCache.InsertEntry(ctx, &filer.Entry{
|
||||
FullPath: dir,
|
||||
Attr: filer.Attr{Mode: os.ModeDir | 0o755, Mtime: now, Crtime: now, Uid: 99, Gid: 100},
|
||||
}, 0); err != nil {
|
||||
tb.Fatalf("insert dir: %v", err)
|
||||
}
|
||||
for i := 0; i < n; i++ {
|
||||
child := dir.Child(fmt.Sprintf("image-%08d.jpg", i))
|
||||
if err := wfs.metaCache.InsertEntry(ctx, &filer.Entry{
|
||||
FullPath: child,
|
||||
// The filer stamps an inode on every entry it stores, so a listing
|
||||
// arrives with one and never has to derive its own.
|
||||
Attr: filer.Attr{Mode: 0o644, Mtime: now, Crtime: now, Uid: 99, Gid: 100, FileSize: 4 << 20, Inode: child.AsInode(now.Unix())},
|
||||
// A real file has chunks, and building them is most of what decoding
|
||||
// an entry costs.
|
||||
// No FileId: BeforeEntrySerialization reparses that legacy string
|
||||
// over Fid on the way in, which would make every entry's chunk
|
||||
// byte-identical instead of varying per file.
|
||||
Chunks: []*filer_pb.FileChunk{{
|
||||
Size: 4 << 20, ModifiedTsNs: now.UnixNano(),
|
||||
ETag: "1a2b3c4d5e6f7890",
|
||||
Fid: &filer_pb.FileId{VolumeId: 3, FileKey: uint64(i), Cookie: 0x1637037d},
|
||||
}},
|
||||
}, 0); err != nil {
|
||||
tb.Fatalf("insert entry %d: %v", i, err)
|
||||
}
|
||||
}
|
||||
|
||||
// Mark the tree cached so the listing is served from the meta cache and no
|
||||
// filer client is dialled.
|
||||
wfs.inodeToPath.MarkChildrenCached(root)
|
||||
wfs.inodeToPath.Lookup(dir, now.Unix(), true, false, 0, true)
|
||||
wfs.inodeToPath.MarkChildrenCached(dir)
|
||||
|
||||
return wfs
|
||||
}
|
||||
|
||||
// walkOnce enumerates the whole directory the way a front end does: repeated
|
||||
// rounds against one handle until the listing runs dry, returning whatever
|
||||
// references the round took.
|
||||
func walkOnce(tb testing.TB, wfs *WFS, dirInode uint64, sink *benchSink, forgets bool) int {
|
||||
dhid, _ := wfs.AcquireDirectoryHandle()
|
||||
defer wfs.ReleaseDirectoryHandle(dhid)
|
||||
|
||||
total := 0
|
||||
offset := uint64(0)
|
||||
for {
|
||||
sink.reset()
|
||||
status := wfs.doReadDirectory(&fuse.ReadIn{
|
||||
InHeader: fuse.InHeader{NodeId: dirInode},
|
||||
Fh: uint64(dhid),
|
||||
Offset: offset,
|
||||
Size: 1 << 20,
|
||||
}, sink, sink.plus)
|
||||
if status != fuse.OK {
|
||||
tb.Fatalf("readdir: %v", status)
|
||||
}
|
||||
if sink.count == 0 {
|
||||
return total
|
||||
}
|
||||
total += sink.count
|
||||
if forgets {
|
||||
// HasInode keeps this to the entries that really hold a reference,
|
||||
// so a listing that took none is not charged for a bogus Forget.
|
||||
for _, ino := range sink.inodes {
|
||||
if wfs.inodeToPath.HasInode(ino) {
|
||||
wfs.Forget(ino, 1)
|
||||
}
|
||||
}
|
||||
}
|
||||
if sink.lastOff <= offset {
|
||||
return total
|
||||
}
|
||||
offset = sink.lastOff
|
||||
}
|
||||
}
|
||||
|
||||
var benchCases = []struct {
|
||||
name string
|
||||
plus bool
|
||||
// ref mirrors the sink's TakesLookupRef. The kernel's reports true whatever
|
||||
// the mode, exactly as fuseDirEntryList does; it is the mode that decides
|
||||
// whether a reference is actually granted.
|
||||
ref bool
|
||||
// forgets is what the front end does after a round: the kernel returns one
|
||||
// FORGET per readdirplus entry and none for a plain readdir, and the WinFsp
|
||||
// adapter hands back everything a round gave it.
|
||||
forgets bool
|
||||
}{
|
||||
{"kernel_readdir", false, true, false},
|
||||
{"kernel_readdirplus", true, true, true},
|
||||
{"winfsp_readdirplus", true, false, true},
|
||||
}
|
||||
|
||||
func BenchmarkReadDirectory(b *testing.B) {
|
||||
dir := util.FullPath("/images")
|
||||
for _, tc := range benchCases {
|
||||
b.Run(tc.name, func(b *testing.B) {
|
||||
wfs := newBenchWFS(b, dir, benchDirEntryCount)
|
||||
dirInode, _ := wfs.inodeToPath.GetInode(dir)
|
||||
sink := &benchSink{plus: tc.plus, takesRef: tc.ref, sinkLimit: 4096}
|
||||
|
||||
b.ResetTimer()
|
||||
b.ReportAllocs()
|
||||
for i := 0; i < b.N; i++ {
|
||||
if got := walkOnce(b, wfs, dirInode, sink, tc.forgets); got != benchDirEntryCount+2 {
|
||||
b.Fatalf("listed %d entries, want %d", got, benchDirEntryCount+2)
|
||||
}
|
||||
}
|
||||
b.StopTimer()
|
||||
// What the listing left in the inode table, over the root and the
|
||||
// directory itself.
|
||||
b.ReportMetric(float64(inodeTableSize(wfs.inodeToPath)), "inodes_left")
|
||||
})
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,32 @@
|
||||
package mount
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"github.com/seaweedfs/go-fuse/v2/fuse"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
)
|
||||
|
||||
// TestOpenDirEnablesKernelListingCache pins the reply flags: without them the
|
||||
// kernel calls back for every enumeration, and losing them would silently
|
||||
// revert repeat listings to full walks of the mount.
|
||||
func TestOpenDirEnablesKernelListingCache(t *testing.T) {
|
||||
dir := util.FullPath("/images")
|
||||
wfs := newBenchWFS(t, dir, 4)
|
||||
dirInode, _ := wfs.inodeToPath.GetInode(dir)
|
||||
|
||||
var out fuse.OpenOut
|
||||
if status := wfs.OpenDir(nil, &fuse.OpenIn{InHeader: fuse.InHeader{NodeId: dirInode}}, &out); status != fuse.OK {
|
||||
t.Fatalf("OpenDir: %v", status)
|
||||
}
|
||||
defer wfs.ReleaseDir(&fuse.ReleaseIn{Fh: out.Fh})
|
||||
|
||||
for _, want := range []struct {
|
||||
name string
|
||||
flag uint32
|
||||
}{{"FOPEN_CACHE_DIR", fuse.FOPEN_CACHE_DIR}, {"FOPEN_KEEP_CACHE", fuse.FOPEN_KEEP_CACHE}} {
|
||||
if out.OpenFlags&want.flag == 0 {
|
||||
t.Errorf("OpenDir reply lacks %s", want.name)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,55 @@
|
||||
package mount
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/filer"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
)
|
||||
|
||||
// TestListDirectoryEntriesOmitsChunks covers the wiring the readdir speedup
|
||||
// rests on: that the context marker actually reaches the store's decode. Every
|
||||
// other test exercises the decoder directly, so a refactor that stopped
|
||||
// threading the context would revert the optimisation silently.
|
||||
func TestListDirectoryEntriesOmitsChunks(t *testing.T) {
|
||||
dir := util.FullPath("/images")
|
||||
wfs := newBenchWFS(t, dir, 4)
|
||||
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
ctx context.Context
|
||||
wantChunks bool
|
||||
}{
|
||||
{"plain listing keeps chunks", context.Background(), true},
|
||||
{"marked listing drops chunks", filer_pb.WithChunksOmitted(context.Background()), false},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
var seen int
|
||||
_, err := wfs.metaCache.ListDirectoryEntries(tc.ctx, dir, "", false, 100, func(entry *filer.Entry) (bool, error) {
|
||||
seen++
|
||||
if got := len(entry.Chunks) > 0; got != tc.wantChunks {
|
||||
t.Errorf("%s: has chunks = %v, want %v", entry.Name(), got, tc.wantChunks)
|
||||
}
|
||||
// The size has to survive either way, since that is what the
|
||||
// readdir reports.
|
||||
if entry.FileSize != 4<<20 {
|
||||
t.Errorf("%s: FileSize = %d, want %d", entry.Name(), entry.FileSize, 4<<20)
|
||||
}
|
||||
return true, nil
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("list: %v", err)
|
||||
}
|
||||
if seen != 4 {
|
||||
t.Fatalf("listed %d entries, want 4", seen)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
// readdirContext is what weedfs_dir_read.go actually passes.
|
||||
if !filer_pb.ChunksOmitted(readdirContext) {
|
||||
t.Error("readdirContext does not carry the chunks-omitted marker")
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,246 @@
|
||||
package mount
|
||||
|
||||
import (
|
||||
"context"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/seaweedfs/go-fuse/v2/fuse"
|
||||
"github.com/seaweedfs/seaweedfs/weed/filer"
|
||||
"github.com/seaweedfs/seaweedfs/weed/mount/meta_cache"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/util"
|
||||
"google.golang.org/grpc"
|
||||
"google.golang.org/grpc/credentials/insecure"
|
||||
)
|
||||
|
||||
// pagingSink collects names, accepting at most limit per round the way a kernel
|
||||
// reply buffer does.
|
||||
type pagingSink struct {
|
||||
limit int
|
||||
names []string
|
||||
lastOff uint64
|
||||
round int
|
||||
}
|
||||
|
||||
func (s *pagingSink) AddEntry(entry fuse.DirEntry) bool {
|
||||
if s.round >= s.limit {
|
||||
return false
|
||||
}
|
||||
s.round++
|
||||
s.names = append(s.names, entry.Name)
|
||||
s.lastOff = entry.Off
|
||||
return true
|
||||
}
|
||||
|
||||
func (s *pagingSink) AddEntryPlus(entry fuse.DirEntry) *fuse.EntryOut { return nil }
|
||||
func (s *pagingSink) TakesLookupRef() bool { return true }
|
||||
|
||||
func newPagingWFS(tb testing.TB, dir util.FullPath, names []string, ttlSec int32) *WFS {
|
||||
tb.Helper()
|
||||
uidGidMapper, err := meta_cache.NewUidGidMapper("", "")
|
||||
if err != nil {
|
||||
tb.Fatalf("uid/gid mapper: %v", err)
|
||||
}
|
||||
root := util.FullPath("/")
|
||||
wfs := &WFS{
|
||||
option: &Option{
|
||||
ChunkSizeLimit: 1024, ConcurrentReaders: 1, VolumeServerAccess: "filerProxy",
|
||||
FilerAddresses: []pb.ServerAddress{pb.NewServerAddressWithGrpcPort("127.0.0.1:1", 1)},
|
||||
GrpcDialOption: grpc.WithTransportCredentials(insecure.NewCredentials()),
|
||||
FilerMountRootPath: "/", MountUid: 99, MountGid: 100, MountMode: 0o777,
|
||||
MountMtime: time.Now(), MountCtime: time.Now(), UidGidMapper: uidGidMapper,
|
||||
},
|
||||
signature: 1, inodeToPath: NewInodeToPath(root, 0),
|
||||
fhMap: NewFileHandleToInode(), dhMap: NewDirectoryHandleToInode(),
|
||||
fhLockTable: util.NewLockTable[FileHandleId](), hardLinkLockTable: util.NewLockTable[string](),
|
||||
}
|
||||
wfs.metaCache = meta_cache.NewMetaCache(
|
||||
filepath.Join(tb.TempDir(), "meta"), uidGidMapper, root, false,
|
||||
func(p util.FullPath) { wfs.inodeToPath.MarkChildrenCached(p) },
|
||||
func(p util.FullPath) bool { return wfs.inodeToPath.IsChildrenCached(p) },
|
||||
func(meta_cache.EntryInvalidation) {}, nil,
|
||||
)
|
||||
tb.Cleanup(wfs.metaCache.Shutdown)
|
||||
|
||||
now := time.Now()
|
||||
ctx := context.Background()
|
||||
if err := wfs.metaCache.InsertEntry(ctx, &filer.Entry{
|
||||
FullPath: dir,
|
||||
Attr: filer.Attr{Mode: os.ModeDir | 0o755, Mtime: now, Crtime: now},
|
||||
}, 0); err != nil {
|
||||
tb.Fatalf("insert dir: %v", err)
|
||||
}
|
||||
for _, name := range names {
|
||||
child := dir.Child(name)
|
||||
attr := filer.Attr{Mode: 0o644, Mtime: now, Crtime: now, FileSize: 1, Inode: child.AsInode(now.Unix())}
|
||||
if ttlSec > 0 {
|
||||
// Expired well in the past, so the meta cache drops it after the
|
||||
// store has already counted it against the limit.
|
||||
attr.TtlSec = ttlSec
|
||||
attr.Crtime = now.Add(-time.Duration(ttlSec+60) * time.Second)
|
||||
}
|
||||
if err := wfs.metaCache.InsertEntry(ctx, &filer.Entry{FullPath: child, Attr: attr}, 0); err != nil {
|
||||
tb.Fatalf("insert %s: %v", name, err)
|
||||
}
|
||||
}
|
||||
wfs.inodeToPath.MarkChildrenCached(root)
|
||||
wfs.inodeToPath.Lookup(dir, now.Unix(), true, false, 0, true)
|
||||
wfs.inodeToPath.MarkChildrenCached(dir)
|
||||
return wfs
|
||||
}
|
||||
|
||||
// TestReadDirResumePastShrunkDirectory covers a client resuming a fresh handle
|
||||
// at a cookie from an earlier, larger listing. The directory has since shrunk
|
||||
// past that offset, and the readdir used to fall back to listing from the first
|
||||
// child again, handing back names the client had already consumed.
|
||||
func TestReadDirResumePastShrunkDirectory(t *testing.T) {
|
||||
dir := util.FullPath("/d")
|
||||
var names []string
|
||||
for i := 0; i < 100; i++ {
|
||||
names = append(names, fmt.Sprintf("f%03d", i))
|
||||
}
|
||||
wfs := newPagingWFS(t, dir, names, 0)
|
||||
dirInode, _ := wfs.inodeToPath.GetInode(dir)
|
||||
|
||||
dhid, _ := wfs.AcquireDirectoryHandle()
|
||||
defer wfs.ReleaseDirectoryHandle(dhid)
|
||||
|
||||
sink := &pagingSink{limit: 4096}
|
||||
status := wfs.doReadDirectory(&fuse.ReadIn{
|
||||
InHeader: fuse.InHeader{NodeId: dirInode},
|
||||
Fh: uint64(dhid),
|
||||
Offset: 152, // past the end of a directory that now holds 100
|
||||
Size: 1 << 20,
|
||||
}, sink, false)
|
||||
if status != fuse.OK {
|
||||
t.Fatalf("readdir: %v", status)
|
||||
}
|
||||
if len(sink.names) != 0 {
|
||||
t.Errorf("resuming past the end returned %d entries (first %q), want none", len(sink.names), sink.names[0])
|
||||
}
|
||||
}
|
||||
|
||||
// TestReadDirWithExpiredEntries covers a batch the store fills to the limit but
|
||||
// whose entries the meta cache then drops as expired. The post-filter count used
|
||||
// to be read as end-of-directory, silently hiding every later child.
|
||||
func TestReadDirWithExpiredEntries(t *testing.T) {
|
||||
dir := util.FullPath("/d")
|
||||
// One full store batch of entries that all expire, then live ones behind
|
||||
// them. Nothing survives the first batch, which is the case that latched.
|
||||
var names []string
|
||||
for i := 0; i < batchSize; i++ {
|
||||
names = append(names, fmt.Sprintf("a%05d", i))
|
||||
}
|
||||
wfs := newPagingWFS(t, dir, names, 30)
|
||||
|
||||
live := []string{"z001", "z002", "z003"}
|
||||
now := time.Now()
|
||||
for _, name := range live {
|
||||
child := dir.Child(name)
|
||||
if err := wfs.metaCache.InsertEntry(context.Background(), &filer.Entry{
|
||||
FullPath: child,
|
||||
Attr: filer.Attr{Mode: 0o644, Mtime: now, Crtime: now, FileSize: 1, Inode: child.AsInode(now.Unix())},
|
||||
}, 0); err != nil {
|
||||
t.Fatalf("insert %s: %v", name, err)
|
||||
}
|
||||
}
|
||||
|
||||
dirInode, _ := wfs.inodeToPath.GetInode(dir)
|
||||
dhid, _ := wfs.AcquireDirectoryHandle()
|
||||
defer wfs.ReleaseDirectoryHandle(dhid)
|
||||
|
||||
sink := &pagingSink{limit: 4096}
|
||||
var offset uint64
|
||||
for round := 0; round < 20; round++ {
|
||||
sink.round = 0
|
||||
status := wfs.doReadDirectory(&fuse.ReadIn{
|
||||
InHeader: fuse.InHeader{NodeId: dirInode},
|
||||
Fh: uint64(dhid),
|
||||
Offset: offset,
|
||||
Size: 1 << 20,
|
||||
}, sink, false)
|
||||
if status != fuse.OK {
|
||||
t.Fatalf("readdir: %v", status)
|
||||
}
|
||||
if sink.lastOff <= offset {
|
||||
break
|
||||
}
|
||||
offset = sink.lastOff
|
||||
}
|
||||
|
||||
var seen []string
|
||||
for _, n := range sink.names {
|
||||
if n != "." && n != ".." {
|
||||
seen = append(seen, n)
|
||||
}
|
||||
}
|
||||
if len(seen) != len(live) {
|
||||
t.Fatalf("listed %d entries %v, want the %d live ones %v", len(seen), seen, len(live), live)
|
||||
}
|
||||
}
|
||||
|
||||
// TestReadDirTrimsConsumedEntries checks that a walk does not accumulate the
|
||||
// whole directory in the handle. Offsets index into the stream from
|
||||
// entryStreamOffset, so the two have to advance together or the listing
|
||||
// silently misaligns.
|
||||
func TestReadDirTrimsConsumedEntries(t *testing.T) {
|
||||
dir := util.FullPath("/d")
|
||||
const total = 5000
|
||||
var names []string
|
||||
for i := 0; i < total; i++ {
|
||||
names = append(names, fmt.Sprintf("f%05d", i))
|
||||
}
|
||||
wfs := newPagingWFS(t, dir, names, 0)
|
||||
dirInode, _ := wfs.inodeToPath.GetInode(dir)
|
||||
|
||||
dhid, dh := wfs.AcquireDirectoryHandle()
|
||||
defer wfs.ReleaseDirectoryHandle(dhid)
|
||||
|
||||
// A small sink forces many rounds, which is when the stream would grow.
|
||||
sink := &pagingSink{limit: 64}
|
||||
var offset uint64
|
||||
var seen []string
|
||||
peak := 0
|
||||
for round := 0; round < 500; round++ {
|
||||
sink.round = 0
|
||||
before := len(sink.names)
|
||||
status := wfs.doReadDirectory(&fuse.ReadIn{
|
||||
InHeader: fuse.InHeader{NodeId: dirInode},
|
||||
Fh: uint64(dhid),
|
||||
Offset: offset,
|
||||
Size: 1 << 20,
|
||||
}, sink, false)
|
||||
if status != fuse.OK {
|
||||
t.Fatalf("readdir: %v", status)
|
||||
}
|
||||
if n := len(dh.entryStream); n > peak {
|
||||
peak = n
|
||||
}
|
||||
if len(sink.names) == before || sink.lastOff <= offset {
|
||||
break
|
||||
}
|
||||
offset = sink.lastOff
|
||||
}
|
||||
for _, n := range sink.names {
|
||||
if n != "." && n != ".." {
|
||||
seen = append(seen, n)
|
||||
}
|
||||
}
|
||||
|
||||
if len(seen) != total {
|
||||
t.Fatalf("listed %d entries, want %d", len(seen), total)
|
||||
}
|
||||
for i, name := range seen {
|
||||
if want := fmt.Sprintf("f%05d", i); name != want {
|
||||
t.Fatalf("entry %d is %q, want %q -- trimming misaligned the offsets", i, name, want)
|
||||
}
|
||||
}
|
||||
// Without trimming the handle ends up holding every entry.
|
||||
if peak >= total {
|
||||
t.Errorf("handle held %d entries at peak, want well under the %d in the directory", peak, total)
|
||||
}
|
||||
}
|
||||
@@ -1405,7 +1405,7 @@ func TestEmptyListingTrailerSnapshotSetsAbsenceFloor(t *testing.T) {
|
||||
startFakeFiler(t, wfs, &fakeFilerServer{listSnapshotTrailerTsNs: 4000})
|
||||
|
||||
wfs.inodeToPath.Lookup(util.FullPath("/dir"), time.Now().Unix(), true, false, 0, false)
|
||||
if err := meta_cache.EnsureVisited(wfs.metaCache, wfs, util.FullPath("/dir")); err != nil {
|
||||
if err := meta_cache.EnsureVisited(wfs.metaCache, wfs, util.FullPath("/dir"), 0); err != nil {
|
||||
t.Fatalf("EnsureVisited: %v", err)
|
||||
}
|
||||
|
||||
|
||||
@@ -651,6 +651,9 @@ func (s *readdirSink) AddEntryPlus(entry fuse.DirEntry) *fuse.EntryOut {
|
||||
return out
|
||||
}
|
||||
|
||||
// WinFsp has no FORGET, and every EntryOut is converted before the round ends.
|
||||
func (s *readdirSink) TakesLookupRef() bool { return false }
|
||||
|
||||
func (w *WinFS) Readdir(path string, fill func(name string, stat *cgofuse.Stat_t, ofst int64) bool, ofst int64, fh uint64) int {
|
||||
inode := w.inodeForHandle(w.dirInodes, fh)
|
||||
if inode == 0 {
|
||||
@@ -699,12 +702,6 @@ func (w *WinFS) Readdir(path string, fill func(name string, stat *cgofuse.Stat_t
|
||||
filled = false
|
||||
}
|
||||
}
|
||||
// A readdirplus entry carries a reference of its own. Give every one
|
||||
// of them back, including any the fill above stopped short of, or a
|
||||
// single walk of a wide directory strands one reference per child.
|
||||
for _, child := range sink.inodes {
|
||||
w.forget(child)
|
||||
}
|
||||
if !filled {
|
||||
return 0
|
||||
}
|
||||
|
||||
@@ -136,6 +136,10 @@ message ListEntriesRequest {
|
||||
bool inclusiveStartFrom = 4;
|
||||
uint32 limit = 5;
|
||||
int64 snapshot_ts_ns = 6;
|
||||
// Leave the chunk list out of every entry in the response. The attributes
|
||||
// still carry the file size, so a listing that only reads attributes can
|
||||
// ask for this and skip the largest part of the payload.
|
||||
bool omit_chunks = 7;
|
||||
}
|
||||
|
||||
message ListEntriesResponse {
|
||||
|
||||
@@ -440,8 +440,12 @@ type ListEntriesRequest struct {
|
||||
InclusiveStartFrom bool `protobuf:"varint,4,opt,name=inclusiveStartFrom,proto3" json:"inclusiveStartFrom,omitempty"`
|
||||
Limit uint32 `protobuf:"varint,5,opt,name=limit,proto3" json:"limit,omitempty"`
|
||||
SnapshotTsNs int64 `protobuf:"varint,6,opt,name=snapshot_ts_ns,json=snapshotTsNs,proto3" json:"snapshot_ts_ns,omitempty"`
|
||||
unknownFields protoimpl.UnknownFields
|
||||
sizeCache protoimpl.SizeCache
|
||||
// Leave the chunk list out of every entry in the response. The attributes
|
||||
// still carry the file size, so a listing that only reads attributes can
|
||||
// ask for this and skip the largest part of the payload.
|
||||
OmitChunks bool `protobuf:"varint,7,opt,name=omit_chunks,json=omitChunks,proto3" json:"omit_chunks,omitempty"`
|
||||
unknownFields protoimpl.UnknownFields
|
||||
sizeCache protoimpl.SizeCache
|
||||
}
|
||||
|
||||
func (x *ListEntriesRequest) Reset() {
|
||||
@@ -516,6 +520,13 @@ func (x *ListEntriesRequest) GetSnapshotTsNs() int64 {
|
||||
return 0
|
||||
}
|
||||
|
||||
func (x *ListEntriesRequest) GetOmitChunks() bool {
|
||||
if x != nil {
|
||||
return x.OmitChunks
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
type ListEntriesResponse struct {
|
||||
state protoimpl.MessageState `protogen:"open.v1"`
|
||||
Entry *Entry `protobuf:"bytes,1,opt,name=entry,proto3" json:"entry,omitempty"`
|
||||
@@ -6939,14 +6950,16 @@ const file_filer_proto_rawDesc = "" +
|
||||
"\x1cLookupDirectoryEntryResponse\x12%\n" +
|
||||
"\x05entry\x18\x01 \x01(\v2\x0f.filer_pb.EntryR\x05entry\x12\x1a\n" +
|
||||
"\tlog_ts_ns\x18\x02 \x01(\x03R\alogTsNs\x12#\n" +
|
||||
"\rlog_signature\x18\x03 \x01(\x05R\flogSignature\"\xe4\x01\n" +
|
||||
"\rlog_signature\x18\x03 \x01(\x05R\flogSignature\"\x85\x02\n" +
|
||||
"\x12ListEntriesRequest\x12\x1c\n" +
|
||||
"\tdirectory\x18\x01 \x01(\tR\tdirectory\x12\x16\n" +
|
||||
"\x06prefix\x18\x02 \x01(\tR\x06prefix\x12,\n" +
|
||||
"\x11startFromFileName\x18\x03 \x01(\tR\x11startFromFileName\x12.\n" +
|
||||
"\x12inclusiveStartFrom\x18\x04 \x01(\bR\x12inclusiveStartFrom\x12\x14\n" +
|
||||
"\x05limit\x18\x05 \x01(\rR\x05limit\x12$\n" +
|
||||
"\x0esnapshot_ts_ns\x18\x06 \x01(\x03R\fsnapshotTsNs\"b\n" +
|
||||
"\x0esnapshot_ts_ns\x18\x06 \x01(\x03R\fsnapshotTsNs\x12\x1f\n" +
|
||||
"\vomit_chunks\x18\a \x01(\bR\n" +
|
||||
"omitChunks\"b\n" +
|
||||
"\x13ListEntriesResponse\x12%\n" +
|
||||
"\x05entry\x18\x01 \x01(\v2\x0f.filer_pb.EntryR\x05entry\x12$\n" +
|
||||
"\x0esnapshot_ts_ns\x18\x02 \x01(\x03R\fsnapshotTsNs\"\xa1\x02\n" +
|
||||
|
||||
@@ -139,6 +139,7 @@ func DoSeaweedListWithSnapshot(ctx context.Context, client SeaweedFilerClient, f
|
||||
Limit: redLimit,
|
||||
InclusiveStartFrom: inclusive,
|
||||
SnapshotTsNs: snapshotTsNs,
|
||||
OmitChunks: ChunksOmitted(ctx),
|
||||
}
|
||||
|
||||
// Preserve the caller-requested snapshot so pagination uses the same
|
||||
|
||||
@@ -149,6 +149,16 @@ func (m *ListEntriesRequest) MarshalToSizedBufferVT(dAtA []byte) (int, error) {
|
||||
i -= len(m.unknownFields)
|
||||
copy(dAtA[i:], m.unknownFields)
|
||||
}
|
||||
if m.OmitChunks {
|
||||
i--
|
||||
if m.OmitChunks {
|
||||
dAtA[i] = 1
|
||||
} else {
|
||||
dAtA[i] = 0
|
||||
}
|
||||
i--
|
||||
dAtA[i] = 0x38
|
||||
}
|
||||
if m.SnapshotTsNs != 0 {
|
||||
i = protohelpers.EncodeVarint(dAtA, i, uint64(m.SnapshotTsNs))
|
||||
i--
|
||||
@@ -6285,6 +6295,9 @@ func (m *ListEntriesRequest) SizeVT() (n int) {
|
||||
if m.SnapshotTsNs != 0 {
|
||||
n += 1 + protohelpers.SizeOfVarint(uint64(m.SnapshotTsNs))
|
||||
}
|
||||
if m.OmitChunks {
|
||||
n += 2
|
||||
}
|
||||
n += len(m.unknownFields)
|
||||
return n
|
||||
}
|
||||
@@ -9140,6 +9153,26 @@ func (m *ListEntriesRequest) UnmarshalVT(dAtA []byte) error {
|
||||
break
|
||||
}
|
||||
}
|
||||
case 7:
|
||||
if wireType != 0 {
|
||||
return fmt.Errorf("proto: wrong wireType = %d for field OmitChunks", wireType)
|
||||
}
|
||||
var v int
|
||||
for shift := uint(0); ; shift += 7 {
|
||||
if shift >= 64 {
|
||||
return protohelpers.ErrIntOverflow
|
||||
}
|
||||
if iNdEx >= l {
|
||||
return io.ErrUnexpectedEOF
|
||||
}
|
||||
b := dAtA[iNdEx]
|
||||
iNdEx++
|
||||
v |= int(b&0x7F) << shift
|
||||
if b < 0x80 {
|
||||
break
|
||||
}
|
||||
}
|
||||
m.OmitChunks = bool(v != 0)
|
||||
default:
|
||||
iNdEx = preIndex
|
||||
skippy, err := protohelpers.Skip(dAtA[iNdEx:])
|
||||
|
||||
@@ -0,0 +1,22 @@
|
||||
package filer_pb
|
||||
|
||||
import "context"
|
||||
|
||||
type omitChunksKey struct{}
|
||||
|
||||
// WithChunksOmitted marks a listing as wanting attributes only. A listing over
|
||||
// gRPC then asks the filer to leave the chunk lists out, and a listing served
|
||||
// from a local store skips building them. The file size is carried in the
|
||||
// attributes either way.
|
||||
//
|
||||
// Only a caller that reads attributes and nothing else may use this. In
|
||||
// particular a listing that populates a cache, or one whose entries can be
|
||||
// written back or deleted, needs the chunks.
|
||||
func WithChunksOmitted(ctx context.Context) context.Context {
|
||||
return context.WithValue(ctx, omitChunksKey{}, true)
|
||||
}
|
||||
|
||||
// ChunksOmitted reports whether the listing wants attributes only.
|
||||
func ChunksOmitted(ctx context.Context) bool {
|
||||
return ctx.Value(omitChunksKey{}) != nil
|
||||
}
|
||||
+31
-17
@@ -135,10 +135,38 @@ func makeSubscribeMetadataFunc(option *MetadataFollowOption, processEventFn Proc
|
||||
|
||||
var pendingRefs []*filer_pb.LogFileChunkRef
|
||||
|
||||
// drainPendingRefs reads whatever chunk refs have accumulated. The server
|
||||
// sends refs in their own responses, so they can only be read once
|
||||
// something tells us the run of refs has ended — either a normal event, or
|
||||
// the end of the stream. A bounded subscription (StopTsNs set, range
|
||||
// already in the past) may get nothing but refs and then EOF, so draining
|
||||
// only on the former silently returns no events at all.
|
||||
drainPendingRefs := func() error {
|
||||
if len(pendingRefs) == 0 || option.LogFileReaderFn == nil {
|
||||
return nil
|
||||
}
|
||||
lastTs, readErr := ReadLogFileRefs(pendingRefs, option.LogFileReaderFn,
|
||||
option.StartTsNs, option.StopTsNs,
|
||||
PathFilter{
|
||||
PathPrefix: option.PathPrefix,
|
||||
AdditionalPathPrefixes: option.AdditionalPathPrefixes,
|
||||
DirectoriesToWatch: option.DirectoriesToWatch,
|
||||
},
|
||||
processEventFn)
|
||||
if readErr != nil {
|
||||
return fmt.Errorf("read log file refs: %w", readErr)
|
||||
}
|
||||
if lastTs > 0 {
|
||||
option.StartTsNs = lastTs
|
||||
}
|
||||
pendingRefs = nil
|
||||
return nil
|
||||
}
|
||||
|
||||
for {
|
||||
resp, listenErr := stream.Recv()
|
||||
if listenErr == io.EOF {
|
||||
return nil
|
||||
return drainPendingRefs()
|
||||
}
|
||||
if listenErr != nil {
|
||||
return listenErr
|
||||
@@ -151,22 +179,8 @@ func makeSubscribeMetadataFunc(option *MetadataFollowOption, processEventFn Proc
|
||||
}
|
||||
|
||||
// Process accumulated refs before handling normal events (transition point)
|
||||
if len(pendingRefs) > 0 && option.LogFileReaderFn != nil {
|
||||
lastTs, readErr := ReadLogFileRefs(pendingRefs, option.LogFileReaderFn,
|
||||
option.StartTsNs, option.StopTsNs,
|
||||
PathFilter{
|
||||
PathPrefix: option.PathPrefix,
|
||||
AdditionalPathPrefixes: option.AdditionalPathPrefixes,
|
||||
DirectoriesToWatch: option.DirectoriesToWatch,
|
||||
},
|
||||
processEventFn)
|
||||
if readErr != nil {
|
||||
return fmt.Errorf("read log file refs: %w", readErr)
|
||||
}
|
||||
if lastTs > 0 {
|
||||
option.StartTsNs = lastTs
|
||||
}
|
||||
pendingRefs = nil
|
||||
if err := drainPendingRefs(); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
// Process the envelope event (top-level fields) and any batched tail.
|
||||
|
||||
@@ -59,6 +59,8 @@ service Seaweed {
|
||||
}
|
||||
rpc VolumeGrow (VolumeGrowRequest) returns (VolumeGrowResponse) {
|
||||
}
|
||||
rpc CollectionStatistics (CollectionStatisticsRequest) returns (CollectionStatisticsResponse) {
|
||||
}
|
||||
}
|
||||
|
||||
//////////////////////////////////////////////////
|
||||
@@ -105,6 +107,16 @@ message Heartbeat {
|
||||
// physical disk capacity per disk type, in bytes, from the underlying filesystem
|
||||
map<string, uint64> disk_total_bytes = 25;
|
||||
map<string, uint64> disk_free_bytes = 26;
|
||||
|
||||
// Digest of every volume in this heartbeat's view of the server, letting the
|
||||
// master check its copy is current without being sent the whole list. Absent
|
||||
// from servers that do not compute it, and distinct from a digest of 0, which
|
||||
// is what a server holding no volumes reports.
|
||||
optional uint64 volume_digest = 27;
|
||||
// Volumes whose reported state changed since the last heartbeat, sent in
|
||||
// place of `volumes`. A master that does not understand this never sets
|
||||
// volume_digest_supported, so it keeps being sent the whole list.
|
||||
repeated VolumeInformationMessage changed_volumes = 28;
|
||||
}
|
||||
|
||||
message HeartbeatResponse {
|
||||
@@ -115,6 +127,12 @@ message HeartbeatResponse {
|
||||
repeated StorageBackend storage_backends = 5;
|
||||
repeated string duplicated_uuids = 6;
|
||||
bool preallocate = 7;
|
||||
// The master's view of this server's volumes disagrees with the reported
|
||||
// digest, so it needs the full volume list rather than changes alone.
|
||||
bool resend_full_volume_list = 8;
|
||||
// The master compares volume digests, so a server that reports one may send
|
||||
// changed_volumes in place of its whole list.
|
||||
bool volume_digest_supported = 9;
|
||||
}
|
||||
|
||||
message VolumeInformationMessage {
|
||||
@@ -315,6 +333,26 @@ message CollectionListResponse {
|
||||
repeated Collection collections = 1;
|
||||
}
|
||||
|
||||
// Summarises what each collection holds, so a caller tracking usage does not
|
||||
// have to be sent every volume in the cluster to add it up itself.
|
||||
message CollectionStatisticsRequest {
|
||||
}
|
||||
message CollectionStatisticsResponse {
|
||||
repeated CollectionStatistics collections = 1;
|
||||
}
|
||||
message CollectionStatistics {
|
||||
string collection = 1;
|
||||
uint64 file_count = 2;
|
||||
uint64 delete_count = 3;
|
||||
uint64 deleted_byte_count = 4;
|
||||
// one copy of the data: a single replica of a regular volume, the data
|
||||
// shards of an ec volume
|
||||
uint64 size = 5;
|
||||
// what is on disk: every replica, and parity shards
|
||||
uint64 physical_size = 6;
|
||||
uint64 volume_count = 7;
|
||||
}
|
||||
|
||||
message CollectionDeleteRequest {
|
||||
string name = 1;
|
||||
}
|
||||
|
||||
+537
-290
File diff suppressed because it is too large
Load Diff
@@ -44,6 +44,7 @@ const (
|
||||
Seaweed_RaftRemoveServer_FullMethodName = "/master_pb.Seaweed/RaftRemoveServer"
|
||||
Seaweed_RaftLeadershipTransfer_FullMethodName = "/master_pb.Seaweed/RaftLeadershipTransfer"
|
||||
Seaweed_VolumeGrow_FullMethodName = "/master_pb.Seaweed/VolumeGrow"
|
||||
Seaweed_CollectionStatistics_FullMethodName = "/master_pb.Seaweed/CollectionStatistics"
|
||||
)
|
||||
|
||||
// SeaweedClient is the client API for Seaweed service.
|
||||
@@ -75,6 +76,7 @@ type SeaweedClient interface {
|
||||
RaftRemoveServer(ctx context.Context, in *RaftRemoveServerRequest, opts ...grpc.CallOption) (*RaftRemoveServerResponse, error)
|
||||
RaftLeadershipTransfer(ctx context.Context, in *RaftLeadershipTransferRequest, opts ...grpc.CallOption) (*RaftLeadershipTransferResponse, error)
|
||||
VolumeGrow(ctx context.Context, in *VolumeGrowRequest, opts ...grpc.CallOption) (*VolumeGrowResponse, error)
|
||||
CollectionStatistics(ctx context.Context, in *CollectionStatisticsRequest, opts ...grpc.CallOption) (*CollectionStatisticsResponse, error)
|
||||
}
|
||||
|
||||
type seaweedClient struct {
|
||||
@@ -344,6 +346,16 @@ func (c *seaweedClient) VolumeGrow(ctx context.Context, in *VolumeGrowRequest, o
|
||||
return out, nil
|
||||
}
|
||||
|
||||
func (c *seaweedClient) CollectionStatistics(ctx context.Context, in *CollectionStatisticsRequest, opts ...grpc.CallOption) (*CollectionStatisticsResponse, error) {
|
||||
cOpts := append([]grpc.CallOption{grpc.StaticMethod()}, opts...)
|
||||
out := new(CollectionStatisticsResponse)
|
||||
err := c.cc.Invoke(ctx, Seaweed_CollectionStatistics_FullMethodName, in, out, cOpts...)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return out, nil
|
||||
}
|
||||
|
||||
// SeaweedServer is the server API for Seaweed service.
|
||||
// All implementations must embed UnimplementedSeaweedServer
|
||||
// for forward compatibility.
|
||||
@@ -373,6 +385,7 @@ type SeaweedServer interface {
|
||||
RaftRemoveServer(context.Context, *RaftRemoveServerRequest) (*RaftRemoveServerResponse, error)
|
||||
RaftLeadershipTransfer(context.Context, *RaftLeadershipTransferRequest) (*RaftLeadershipTransferResponse, error)
|
||||
VolumeGrow(context.Context, *VolumeGrowRequest) (*VolumeGrowResponse, error)
|
||||
CollectionStatistics(context.Context, *CollectionStatisticsRequest) (*CollectionStatisticsResponse, error)
|
||||
mustEmbedUnimplementedSeaweedServer()
|
||||
}
|
||||
|
||||
@@ -458,6 +471,9 @@ func (UnimplementedSeaweedServer) RaftLeadershipTransfer(context.Context, *RaftL
|
||||
func (UnimplementedSeaweedServer) VolumeGrow(context.Context, *VolumeGrowRequest) (*VolumeGrowResponse, error) {
|
||||
return nil, status.Error(codes.Unimplemented, "method VolumeGrow not implemented")
|
||||
}
|
||||
func (UnimplementedSeaweedServer) CollectionStatistics(context.Context, *CollectionStatisticsRequest) (*CollectionStatisticsResponse, error) {
|
||||
return nil, status.Error(codes.Unimplemented, "method CollectionStatistics not implemented")
|
||||
}
|
||||
func (UnimplementedSeaweedServer) mustEmbedUnimplementedSeaweedServer() {}
|
||||
func (UnimplementedSeaweedServer) testEmbeddedByValue() {}
|
||||
|
||||
@@ -896,6 +912,24 @@ func _Seaweed_VolumeGrow_Handler(srv interface{}, ctx context.Context, dec func(
|
||||
return interceptor(ctx, in, info, handler)
|
||||
}
|
||||
|
||||
func _Seaweed_CollectionStatistics_Handler(srv interface{}, ctx context.Context, dec func(interface{}) error, interceptor grpc.UnaryServerInterceptor) (interface{}, error) {
|
||||
in := new(CollectionStatisticsRequest)
|
||||
if err := dec(in); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
if interceptor == nil {
|
||||
return srv.(SeaweedServer).CollectionStatistics(ctx, in)
|
||||
}
|
||||
info := &grpc.UnaryServerInfo{
|
||||
Server: srv,
|
||||
FullMethod: Seaweed_CollectionStatistics_FullMethodName,
|
||||
}
|
||||
handler := func(ctx context.Context, req interface{}) (interface{}, error) {
|
||||
return srv.(SeaweedServer).CollectionStatistics(ctx, req.(*CollectionStatisticsRequest))
|
||||
}
|
||||
return interceptor(ctx, in, info, handler)
|
||||
}
|
||||
|
||||
// Seaweed_ServiceDesc is the grpc.ServiceDesc for Seaweed service.
|
||||
// It's only intended for direct use with grpc.RegisterService,
|
||||
// and not to be introspected or modified (even as a copy)
|
||||
@@ -991,6 +1025,10 @@ var Seaweed_ServiceDesc = grpc.ServiceDesc{
|
||||
MethodName: "VolumeGrow",
|
||||
Handler: _Seaweed_VolumeGrow_Handler,
|
||||
},
|
||||
{
|
||||
MethodName: "CollectionStatistics",
|
||||
Handler: _Seaweed_CollectionStatistics_Handler,
|
||||
},
|
||||
},
|
||||
Streams: []grpc.StreamDesc{
|
||||
{
|
||||
|
||||
@@ -1671,7 +1671,7 @@ func (iam *IdentityAccessManagement) authRequestWithAuthType(r *http.Request, ac
|
||||
|
||||
// Batch DeleteObjects keys arrive in the body, not the URL: a bucket-level check
|
||||
// here can't match object-scoped policies. DeleteMultipleObjectsHandler authorizes
|
||||
// each key via AuthorizeBatchDeleteKey.
|
||||
// each key via AuthorizeObjectDelete.
|
||||
if action == s3_constants.ACTION_WRITE && r.Method == http.MethodPost &&
|
||||
object == "" && r.URL.Query().Has("delete") {
|
||||
r.Header.Set(s3_constants.AmzAccountId, identity.Account.Id)
|
||||
@@ -2739,11 +2739,11 @@ func (iam *IdentityAccessManagement) AuthorizeCopySource(r *http.Request, identi
|
||||
return iam.VerifyActionPermission(srcReq, identity, Action(action), srcBucket, srcObject)
|
||||
}
|
||||
|
||||
// AuthorizeBatchDeleteKey authorizes one key from a DeleteObjects body. The route
|
||||
// Auth middleware only authenticated the caller (keys arrive in the body, not the
|
||||
// URL), so each key is checked here against a synthetic DELETE /<bucket>/<key> that
|
||||
// makes ResolveS3Action and buildResourceARN target the object. Mirrors AuthorizeCopySource.
|
||||
func (iam *IdentityAccessManagement) AuthorizeBatchDeleteKey(r *http.Request, identity *Identity, bucket, objectKey, versionId string) s3err.ErrorCode {
|
||||
// AuthorizeObjectDelete authorizes removing one key the request URL does not
|
||||
// name: a key from a DeleteObjects body, or the source of a RenameObject. It is
|
||||
// checked against a synthetic DELETE /<bucket>/<key> so that ResolveS3Action and
|
||||
// buildResourceARN target the object. Mirrors AuthorizeCopySource.
|
||||
func (iam *IdentityAccessManagement) AuthorizeObjectDelete(r *http.Request, identity *Identity, bucket, objectKey, versionId string) s3err.ErrorCode {
|
||||
if !iam.isEnabled() {
|
||||
return s3err.ErrNone
|
||||
}
|
||||
|
||||
@@ -15,7 +15,6 @@ import (
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/filer_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
|
||||
"github.com/seaweedfs/seaweedfs/weed/stats"
|
||||
"github.com/seaweedfs/seaweedfs/weed/storage/erasure_coding"
|
||||
)
|
||||
|
||||
const (
|
||||
@@ -46,12 +45,6 @@ func (c *CollectionInfo) LogicalSize() float64 {
|
||||
return c.Size - c.DeletedByteCount
|
||||
}
|
||||
|
||||
// volumeKey uniquely identifies a volume for deduplication
|
||||
type volumeKey struct {
|
||||
collection string
|
||||
volumeId uint32
|
||||
}
|
||||
|
||||
// startBucketSizeMetricsLoop periodically collects bucket size metrics and updates Prometheus gauges.
|
||||
// Uses a distributed lock to ensure only one S3 instance collects metrics at a time.
|
||||
// Should be called as a goroutine; stops when the provided context is cancelled.
|
||||
@@ -185,18 +178,28 @@ func (s3a *S3ApiServer) collectCollectionInfoFromMaster(ctx context.Context) (ma
|
||||
masterMap[string(master)] = master
|
||||
}
|
||||
|
||||
// Connect to any available master and get volume list with topology
|
||||
// Ask the master to summarise. Adding this up here instead would mean
|
||||
// being sent every volume in the cluster once a minute.
|
||||
collectionInfos := make(map[string]*CollectionInfo)
|
||||
|
||||
err := pb.WithOneOfGrpcMasterClients(false, masterMap, s3a.option.GrpcDialOption, func(client master_pb.SeaweedClient) error {
|
||||
resp, err := client.VolumeList(ctx, &master_pb.VolumeListRequest{})
|
||||
resp, err := client.CollectionStatistics(ctx, &master_pb.CollectionStatisticsRequest{})
|
||||
if err != nil {
|
||||
return fmt.Errorf("failed to get volume list: %w", err)
|
||||
return fmt.Errorf("failed to get collection statistics: %w", err)
|
||||
}
|
||||
if resp == nil || resp.TopologyInfo == nil {
|
||||
return fmt.Errorf("empty topology info from master")
|
||||
if resp == nil {
|
||||
return fmt.Errorf("empty collection statistics from master")
|
||||
}
|
||||
for _, c := range resp.Collections {
|
||||
collectionInfos[c.Collection] = &CollectionInfo{
|
||||
FileCount: float64(c.FileCount),
|
||||
DeleteCount: float64(c.DeleteCount),
|
||||
DeletedByteCount: float64(c.DeletedByteCount),
|
||||
Size: float64(c.Size),
|
||||
PhysicalSize: float64(c.PhysicalSize),
|
||||
VolumeCount: int(c.VolumeCount),
|
||||
}
|
||||
}
|
||||
collectCollectionInfoFromTopology(resp.TopologyInfo, collectionInfos)
|
||||
return nil
|
||||
})
|
||||
if err != nil {
|
||||
@@ -256,105 +259,3 @@ func (s3a *S3ApiServer) listBuckets(ctx context.Context) ([]*filer_pb.Entry, err
|
||||
|
||||
return buckets, err
|
||||
}
|
||||
|
||||
// ecVolumeAgg accumulates per-volume EC counts across the shard holders.
|
||||
// fileCount is volume-wide (every holder sees the same .ecx) so we take the
|
||||
// max across reporters to avoid a slow node with a not-yet-loaded .ecx
|
||||
// pinning the aggregate at 0. deleteCount is node-local to each .ecj
|
||||
// deletion journal, so it's summed across reporters.
|
||||
type ecVolumeAgg struct {
|
||||
collection string
|
||||
fileCount uint64
|
||||
deleteCount uint64
|
||||
}
|
||||
|
||||
// collectCollectionInfoFromTopology extracts collection info from topology.
|
||||
// Deduplicates by volume ID to correctly handle missing replicas.
|
||||
// Unlike dividing by copyCount (which would give wrong results if replicas are missing),
|
||||
// we track seen volume IDs and only count each volume once for logical size/count.
|
||||
// EC-encoded volumes are folded in via per-shard aggregation: every shard is
|
||||
// node-local (not a replica), so shard sizes are summed across nodes; the
|
||||
// per-volume file/delete counts carried on each shard message are deduped
|
||||
// via max/sum so the aggregate doesn't double-count or drop after a volume
|
||||
// is converted from regular to erasure coding.
|
||||
func collectCollectionInfoFromTopology(t *master_pb.TopologyInfo, collectionInfos map[string]*CollectionInfo) {
|
||||
// Track which volumes we've already seen to deduplicate by volume ID
|
||||
seenVolumes := make(map[volumeKey]bool)
|
||||
ecVolumes := make(map[volumeKey]*ecVolumeAgg)
|
||||
|
||||
for _, dc := range t.DataCenterInfos {
|
||||
for _, r := range dc.RackInfos {
|
||||
for _, dn := range r.DataNodeInfos {
|
||||
for _, diskInfo := range dn.DiskInfos {
|
||||
for _, vi := range diskInfo.VolumeInfos {
|
||||
c := vi.Collection
|
||||
cif, found := collectionInfos[c]
|
||||
if !found {
|
||||
cif = &CollectionInfo{}
|
||||
collectionInfos[c] = cif
|
||||
}
|
||||
|
||||
// Always add to physical size (all replicas)
|
||||
cif.PhysicalSize += float64(vi.Size)
|
||||
|
||||
// Check if we've already counted this volume for logical stats
|
||||
key := volumeKey{collection: c, volumeId: vi.Id}
|
||||
if seenVolumes[key] {
|
||||
// Already counted this volume, skip logical stats
|
||||
continue
|
||||
}
|
||||
seenVolumes[key] = true
|
||||
|
||||
// First time seeing this volume - add to logical stats
|
||||
cif.Size += float64(vi.Size)
|
||||
cif.FileCount += float64(vi.FileCount)
|
||||
cif.DeleteCount += float64(vi.DeleteCount)
|
||||
cif.DeletedByteCount += float64(vi.DeletedByteCount)
|
||||
cif.VolumeCount++
|
||||
}
|
||||
|
||||
for _, esi := range diskInfo.EcShardInfos {
|
||||
c := esi.Collection
|
||||
cif, found := collectionInfos[c]
|
||||
if !found {
|
||||
cif = &CollectionInfo{}
|
||||
collectionInfos[c] = cif
|
||||
}
|
||||
|
||||
// EC shards are node-local (no replication), so both
|
||||
// physical and logical shard sizes sum across nodes
|
||||
// without any dedupe. Logical size excludes parity
|
||||
// shards; physical size includes them. Upstream OSS
|
||||
// uses the fixed 10+4 ratio (dataShards=0 → default);
|
||||
// forks with per-volume ratio metadata can pass the
|
||||
// configured value here.
|
||||
cif.PhysicalSize += float64(erasure_coding.EcShardsTotalSize(esi))
|
||||
cif.Size += float64(erasure_coding.EcShardsDataSize(esi, 0))
|
||||
|
||||
key := volumeKey{collection: c, volumeId: esi.Id}
|
||||
agg, ok := ecVolumes[key]
|
||||
if !ok {
|
||||
agg = &ecVolumeAgg{collection: c}
|
||||
ecVolumes[key] = agg
|
||||
cif.VolumeCount++
|
||||
}
|
||||
if esi.FileCount > agg.fileCount {
|
||||
agg.fileCount = esi.FileCount
|
||||
}
|
||||
agg.deleteCount += esi.DeleteCount
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Fold deduped EC file/delete counts into each collection's totals.
|
||||
for _, agg := range ecVolumes {
|
||||
cif := collectionInfos[agg.collection]
|
||||
if cif == nil {
|
||||
continue
|
||||
}
|
||||
cif.FileCount += float64(agg.fileCount)
|
||||
cif.DeleteCount += float64(agg.deleteCount)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2,219 +2,8 @@ package s3api
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
|
||||
)
|
||||
|
||||
// TestCollectCollectionInfoFromTopologyEC verifies that EC-encoded volumes
|
||||
// contribute to per-collection logical/physical size, file count, and volume
|
||||
// count. Before this fix, encoding a volume to EC caused the bucket size
|
||||
// metrics exported to Prometheus to drop to zero for that volume.
|
||||
//
|
||||
// Layout: one 10+4 EC volume in collection "crm-docs-storage", 14 shards of
|
||||
// 1000 bytes each split across two nodes.
|
||||
// - nodeA holds data shards 0..6 (7 * 1000 = 7000)
|
||||
// - nodeB holds data shards 7..9 (3 * 1000 = 3000) and parity 10..13 (4 * 1000 = 4000)
|
||||
//
|
||||
// Expected:
|
||||
// - PhysicalSize = 14 * 1000 = 14000
|
||||
// - Size (logical, data shards) = 10 * 1000 = 10000
|
||||
// - FileCount = 100 total - (2 + 3) local deletes = 95 is NOT what we check; the
|
||||
// collector reports raw file_count and delete_count as separate gauges, so
|
||||
// we assert FileCount = 100 (max across reporters) and DeleteCount = 5 (sum).
|
||||
// - VolumeCount = 1 (one unique EC volume)
|
||||
func TestCollectCollectionInfoFromTopologyEC(t *testing.T) {
|
||||
nodeA := &master_pb.DataNodeInfo{
|
||||
DiskInfos: map[string]*master_pb.DiskInfo{
|
||||
"disk1": {
|
||||
EcShardInfos: []*master_pb.VolumeEcShardInformationMessage{
|
||||
{
|
||||
Id: 42,
|
||||
Collection: "crm-docs-storage",
|
||||
EcIndexBits: (1 << 0) | (1 << 1) | (1 << 2) | (1 << 3) | (1 << 4) | (1 << 5) | (1 << 6),
|
||||
ShardSizes: []int64{1000, 1000, 1000, 1000, 1000, 1000, 1000},
|
||||
FileCount: 100,
|
||||
DeleteCount: 2,
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
}
|
||||
nodeB := &master_pb.DataNodeInfo{
|
||||
DiskInfos: map[string]*master_pb.DiskInfo{
|
||||
"disk1": {
|
||||
EcShardInfos: []*master_pb.VolumeEcShardInformationMessage{
|
||||
{
|
||||
Id: 42,
|
||||
Collection: "crm-docs-storage",
|
||||
EcIndexBits: (1 << 7) | (1 << 8) | (1 << 9) | (1 << 10) | (1 << 11) | (1 << 12) | (1 << 13),
|
||||
ShardSizes: []int64{1000, 1000, 1000, 1000, 1000, 1000, 1000},
|
||||
FileCount: 100,
|
||||
DeleteCount: 3,
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
topo := &master_pb.TopologyInfo{
|
||||
DataCenterInfos: []*master_pb.DataCenterInfo{
|
||||
{
|
||||
RackInfos: []*master_pb.RackInfo{
|
||||
{
|
||||
DataNodeInfos: []*master_pb.DataNodeInfo{nodeA, nodeB},
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
got := make(map[string]*CollectionInfo)
|
||||
collectCollectionInfoFromTopology(topo, got)
|
||||
|
||||
info, ok := got["crm-docs-storage"]
|
||||
if !ok {
|
||||
t.Fatalf("expected collection crm-docs-storage, got: %v", got)
|
||||
}
|
||||
if info.PhysicalSize != 14000 {
|
||||
t.Errorf("PhysicalSize: got %.0f, want 14000", info.PhysicalSize)
|
||||
}
|
||||
if info.Size != 10000 {
|
||||
t.Errorf("Size (logical): got %.0f, want 10000", info.Size)
|
||||
}
|
||||
if info.FileCount != 100 {
|
||||
t.Errorf("FileCount: got %.0f, want 100 (max across reporters)", info.FileCount)
|
||||
}
|
||||
if info.DeleteCount != 5 {
|
||||
t.Errorf("DeleteCount: got %.0f, want 5 (sum across reporters)", info.DeleteCount)
|
||||
}
|
||||
if info.VolumeCount != 1 {
|
||||
t.Errorf("VolumeCount: got %d, want 1", info.VolumeCount)
|
||||
}
|
||||
}
|
||||
|
||||
// TestCollectCollectionInfoFromTopologyMixed verifies that regular and EC
|
||||
// volumes accumulate under the same collection without one clobbering the
|
||||
// other, which is the state during an in-progress EC conversion.
|
||||
func TestCollectCollectionInfoFromTopologyMixed(t *testing.T) {
|
||||
node := &master_pb.DataNodeInfo{
|
||||
DiskInfos: map[string]*master_pb.DiskInfo{
|
||||
"disk1": {
|
||||
VolumeInfos: []*master_pb.VolumeInformationMessage{
|
||||
{
|
||||
Id: 1,
|
||||
Collection: "bucket-mix",
|
||||
Size: 5000,
|
||||
FileCount: 50,
|
||||
DeleteCount: 1,
|
||||
DeletedByteCount: 100,
|
||||
},
|
||||
},
|
||||
EcShardInfos: []*master_pb.VolumeEcShardInformationMessage{
|
||||
{
|
||||
Id: 2,
|
||||
Collection: "bucket-mix",
|
||||
EcIndexBits: (1 << 0) | (1 << 1) | (1 << 10), // 2 data + 1 parity
|
||||
ShardSizes: []int64{3000, 3000, 3000},
|
||||
FileCount: 80,
|
||||
DeleteCount: 4,
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
topo := &master_pb.TopologyInfo{
|
||||
DataCenterInfos: []*master_pb.DataCenterInfo{
|
||||
{
|
||||
RackInfos: []*master_pb.RackInfo{
|
||||
{
|
||||
DataNodeInfos: []*master_pb.DataNodeInfo{node},
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
got := make(map[string]*CollectionInfo)
|
||||
collectCollectionInfoFromTopology(topo, got)
|
||||
|
||||
info, ok := got["bucket-mix"]
|
||||
if !ok {
|
||||
t.Fatalf("expected collection bucket-mix, got: %v", got)
|
||||
}
|
||||
// Regular volume: 5000 physical + logical. EC shards: 9000 physical,
|
||||
// 6000 logical (data shards 0 and 1).
|
||||
if info.PhysicalSize != 5000+9000 {
|
||||
t.Errorf("PhysicalSize: got %.0f, want 14000", info.PhysicalSize)
|
||||
}
|
||||
if info.Size != 5000+6000 {
|
||||
t.Errorf("Size: got %.0f, want 11000", info.Size)
|
||||
}
|
||||
// LogicalSize drops the 100 bytes of un-vacuumed garbage on the regular
|
||||
// volume; EC shards carry no DeletedByteCount here.
|
||||
if info.LogicalSize() != 11000-100 {
|
||||
t.Errorf("LogicalSize: got %.0f, want 10900", info.LogicalSize())
|
||||
}
|
||||
if info.FileCount != 50+80 {
|
||||
t.Errorf("FileCount: got %.0f, want 130", info.FileCount)
|
||||
}
|
||||
if info.DeleteCount != 1+4 {
|
||||
t.Errorf("DeleteCount: got %.0f, want 5", info.DeleteCount)
|
||||
}
|
||||
if info.VolumeCount != 2 {
|
||||
t.Errorf("VolumeCount: got %d, want 2", info.VolumeCount)
|
||||
}
|
||||
}
|
||||
|
||||
// TestCollectCollectionInfoFromTopologyECFileCountMaxDedupe verifies that a
|
||||
// slow shard holder reporting file_count=0 (because it has not yet finished
|
||||
// loading .ecx) does not pin the per-volume FileCount at 0.
|
||||
func TestCollectCollectionInfoFromTopologyECFileCountMaxDedupe(t *testing.T) {
|
||||
makeNode := func(bits uint32, sizes []int64, fileCount uint64) *master_pb.DataNodeInfo {
|
||||
return &master_pb.DataNodeInfo{
|
||||
DiskInfos: map[string]*master_pb.DiskInfo{
|
||||
"disk1": {
|
||||
EcShardInfos: []*master_pb.VolumeEcShardInformationMessage{
|
||||
{
|
||||
Id: 11,
|
||||
Collection: "bucket-b",
|
||||
EcIndexBits: bits,
|
||||
ShardSizes: sizes,
|
||||
FileCount: fileCount,
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
topo := &master_pb.TopologyInfo{
|
||||
DataCenterInfos: []*master_pb.DataCenterInfo{
|
||||
{
|
||||
RackInfos: []*master_pb.RackInfo{
|
||||
{
|
||||
DataNodeInfos: []*master_pb.DataNodeInfo{
|
||||
makeNode((1<<0)|(1<<1)|(1<<2)|(1<<3)|(1<<4)|(1<<5)|(1<<6), []int64{1, 1, 1, 1, 1, 1, 1}, 0),
|
||||
makeNode((1<<7)|(1<<8)|(1<<9)|(1<<10)|(1<<11)|(1<<12)|(1<<13), []int64{1, 1, 1, 1, 1, 1, 1}, 6),
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
got := make(map[string]*CollectionInfo)
|
||||
collectCollectionInfoFromTopology(topo, got)
|
||||
info, ok := got["bucket-b"]
|
||||
if !ok {
|
||||
t.Fatalf("expected collection bucket-b, got: %v", got)
|
||||
}
|
||||
if info.FileCount != 6 {
|
||||
t.Errorf("FileCount: got %.0f, want 6 (max across reporters)", info.FileCount)
|
||||
}
|
||||
}
|
||||
|
||||
// TestCollectionInfoLogicalSize verifies logical size excludes un-vacuumed
|
||||
// garbage and never goes negative. Quota enforcement runs on this value so a
|
||||
// bucket full of tombstones is not flipped read-only while its live data is
|
||||
|
||||
@@ -9,10 +9,10 @@ import (
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
// TestAuthorizeBatchDeleteKey_AwsCanonicalPolicy: a policy granting s3:DeleteObject
|
||||
// TestAuthorizeObjectDelete_AwsCanonicalPolicy: a policy granting s3:DeleteObject
|
||||
// on <bucket>/* must allow per-key batch deletes. Pre-fix the bucket-level check
|
||||
// built arn:aws:s3:::<bucket> and never matched the object-scoped policy.
|
||||
func TestAuthorizeBatchDeleteKey_AwsCanonicalPolicy(t *testing.T) {
|
||||
func TestAuthorizeObjectDelete_AwsCanonicalPolicy(t *testing.T) {
|
||||
const bucket = "test-bucket"
|
||||
const policyName = "delete-test-bucket-objects"
|
||||
|
||||
@@ -43,17 +43,17 @@ func TestAuthorizeBatchDeleteKey_AwsCanonicalPolicy(t *testing.T) {
|
||||
r := httptest.NewRequest("POST", "/"+bucket+"?delete", nil)
|
||||
|
||||
require.Equal(t, s3err.ErrNone,
|
||||
iam.AuthorizeBatchDeleteKey(r, identity, bucket, "objects/a.txt", ""),
|
||||
iam.AuthorizeObjectDelete(r, identity, bucket, "objects/a.txt", ""),
|
||||
"s3:DeleteObject on arn:aws:s3:::%s/* must allow deleting %s/objects/a.txt", bucket, bucket)
|
||||
|
||||
require.Equal(t, s3err.ErrAccessDenied,
|
||||
iam.AuthorizeBatchDeleteKey(r, identity, "other-bucket", "objects/a.txt", ""),
|
||||
iam.AuthorizeObjectDelete(r, identity, "other-bucket", "objects/a.txt", ""),
|
||||
"keys outside the granted bucket must be denied")
|
||||
}
|
||||
|
||||
// TestAuthorizeBatchDeleteKey_PrefixScopedPolicy: a prefix-scoped policy must allow
|
||||
// TestAuthorizeObjectDelete_PrefixScopedPolicy: a prefix-scoped policy must allow
|
||||
// batch deletes under the prefix and deny keys outside it, per-key.
|
||||
func TestAuthorizeBatchDeleteKey_PrefixScopedPolicy(t *testing.T) {
|
||||
func TestAuthorizeObjectDelete_PrefixScopedPolicy(t *testing.T) {
|
||||
const bucket = "test-bucket"
|
||||
const policyName = "delete-prefix-only"
|
||||
|
||||
@@ -84,10 +84,10 @@ func TestAuthorizeBatchDeleteKey_PrefixScopedPolicy(t *testing.T) {
|
||||
r := httptest.NewRequest("POST", "/"+bucket+"?delete", nil)
|
||||
|
||||
require.Equal(t, s3err.ErrNone,
|
||||
iam.AuthorizeBatchDeleteKey(r, identity, bucket, "safe/inside.txt", ""),
|
||||
iam.AuthorizeObjectDelete(r, identity, bucket, "safe/inside.txt", ""),
|
||||
"key under granted prefix must be allowed")
|
||||
|
||||
require.Equal(t, s3err.ErrAccessDenied,
|
||||
iam.AuthorizeBatchDeleteKey(r, identity, bucket, "danger/outside.txt", ""),
|
||||
iam.AuthorizeObjectDelete(r, identity, bucket, "danger/outside.txt", ""),
|
||||
"key outside the granted prefix must be denied per-key, not at the batch level")
|
||||
}
|
||||
|
||||
@@ -133,6 +133,7 @@ func (s *Server) finalizeCreateOnCommit(ctx context.Context, input createOnCommi
|
||||
message: "Failed to apply statistics updates: " + err.Error(),
|
||||
}
|
||||
}
|
||||
metadataBytes = refreshDefaultNameMapping(metadataBytes, newMetadata)
|
||||
// Same spec-compliance fixup we apply on create-table; ensures
|
||||
// v{N}.metadata.json files written through this create-on-commit path are
|
||||
// also readable by strict Iceberg clients reading directly from S3.
|
||||
|
||||
@@ -66,6 +66,30 @@ func (s *Server) handleUpdateTable(w http.ResponseWriter, r *http.Request) {
|
||||
return
|
||||
}
|
||||
}
|
||||
// Manifest repair runs once, as soon as the table location is known; on
|
||||
// commit retries the updates already reference the repaired files. Repair
|
||||
// is best effort end to end: the originals parsed already, so a repair
|
||||
// that fails to re-parse is discarded rather than failing the commit.
|
||||
manifestsRepaired := false
|
||||
repairManifests := func(location string) {
|
||||
if manifestsRepaired {
|
||||
return
|
||||
}
|
||||
manifestsRepaired = true
|
||||
repaired, changed := s.repairAddSnapshotManifests(r.Context(), location, raw.Updates)
|
||||
if !changed {
|
||||
return
|
||||
}
|
||||
repairedUpdates, repairedStatistics, err := parseCommitUpdates(repaired)
|
||||
if err != nil {
|
||||
glog.Warningf("Iceberg: repaired updates failed to parse, keeping originals: %v", err)
|
||||
return
|
||||
}
|
||||
raw.Updates = repaired
|
||||
req.Updates = repairedUpdates
|
||||
statisticsUpdates = repairedStatistics
|
||||
}
|
||||
|
||||
maxCommitAttempts := 3
|
||||
generatedLegacyUUID := uuid.New()
|
||||
stageCreateEnabled := isStageCreateEnabled()
|
||||
@@ -168,6 +192,8 @@ func (s *Server) handleUpdateTable(w http.ResponseWriter, r *http.Request) {
|
||||
}
|
||||
}
|
||||
|
||||
repairManifests(location)
|
||||
|
||||
result, reqErr := s.finalizeCreateOnCommit(r.Context(), createOnCommitInput{
|
||||
bucketARN: bucketARN,
|
||||
markerBucket: bucketName,
|
||||
@@ -232,6 +258,8 @@ func (s *Server) handleUpdateTable(w http.ResponseWriter, r *http.Request) {
|
||||
}
|
||||
}
|
||||
|
||||
repairManifests(location)
|
||||
|
||||
builder, err := table.MetadataBuilderFromBase(currentMetadata, getResp.MetadataLocation)
|
||||
if err != nil {
|
||||
writeError(w, http.StatusInternalServerError, "InternalServerError", "Failed to create metadata builder: "+err.Error())
|
||||
@@ -266,6 +294,7 @@ func (s *Server) handleUpdateTable(w http.ResponseWriter, r *http.Request) {
|
||||
writeError(w, http.StatusBadRequest, "BadRequestException", "Failed to apply statistics updates: "+err.Error())
|
||||
return
|
||||
}
|
||||
metadataBytes = refreshDefaultNameMapping(metadataBytes, newMetadata)
|
||||
// Same spec-compliance fixup we apply on create-table; ensures
|
||||
// v{N}.metadata.json files written during commit are also readable by
|
||||
// strict Iceberg clients reading directly from S3, and that the
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user