Files
seaweedfs/weed/storage/volume_info/volume_info.go
T
Chris LuandGitHub 75ec5ec193 admin: allow setting volume read-only and read/write modes (#11217)
* admin: support setting volume read-only and read/write modes

* admin: address PR review on volume access-mode persistence

Reject trailing JSON values in the SetVolumeReadOnly handler so
requests like {"read_only":true}{} no longer pass validation, and add
a trailing-value case to the invalid-request test.

Propagate .vif persistence failures through the access-mode chain.
PersistReadOnly now returns the SaveVolumeInfo error and rolls back
the in-memory volumeInfo on failure; Store.MarkVolumeReadonly and
Store.MarkVolumeWritable propagate that error and roll back their
noWrite flags, so the API reports failure instead of success while
restart would revert the mode.

* admin: make .vif persistence atomic and preserve error chain

SaveVolumeInfo now writes to a .vif.tmp file, syncs it, renames it
over the target, and fsyncs the directory. A write/sync/close failure
leaves the existing .vif intact, so the PersistReadOnly in-memory
rollback matches the durable state instead of diverging from a
partially written file that restart would apply.

Switch the error wrappers in PersistReadOnly, MarkVolumeReadonly, and
MarkVolumeWritable from %v to %w so callers can use errors.Is and
errors.As to classify persistence failures.

* admin: treat post-rename dir fsync failure as a warning

After os.Rename commits the new .vif, the on-disk file already holds
the requested mode. A directory fsync failure only risks losing the
rename across a crash; returning an error here would make
PersistReadOnly roll back in-memory state while the durable file keeps
the new mode, splitting the replica. Log the failure as a warning
instead, matching the best-effort nature of FsyncDir (already skipped
on Windows).

* admin: distinguish post-rename durability failures and use unique temp files

SaveVolumeInfo now uses os.CreateTemp for the staging file, preventing
concurrent saves for the same volume from colliding on a shared .tmp
path.

A directory fsync failure after os.Rename returns a
NotCrashDurableError instead of being silently swallowed. The rename
already committed the new metadata to disk, so PersistReadOnly,
MarkVolumeReadonly, and MarkVolumeWritable skip the in-memory rollback
for this error type (keeping state aligned with the durable file) while
still propagating the failure to the API. Pre-commit failures continue
to roll back as before.

* admin: continue post-commit work after NotCrashDurableError

MarkVolumeWritable now clears the EIO quarantine and the gRPC handlers
(makeVolumeReadonly step 3, makeVolumeWritable master notification)
proceed with their post-commit work when SaveVolumeInfo returns a
NotCrashDurableError, instead of aborting and leaving the volume
unavailable or the master unaware of the mode change. The durability
warning is still propagated to the API caller. Pre-commit failures
continue to abort early as before.

* admin: handle NotCrashDurableError in tier and EC callers

VolumeTierMoveDatFromRemote and VolumeEcShardsGenerate now check for
NotCrashDurableError from SaveVolumeInfo. When the rename has already
committed the new .vif, they continue with their post-commit work
(backend switch, remote deletion, keeping generated EC shards) instead
of aborting and leaving the on-disk metadata inconsistent with the
file layout. The durability warning is logged for the operator.
2026-09-07 18:40:37 -07:00

168 lines
5.5 KiB
Go

package volume_info
import (
"fmt"
"os"
"path/filepath"
"github.com/seaweedfs/seaweedfs/weed/glog"
"github.com/seaweedfs/seaweedfs/weed/pb/volume_server_pb"
"github.com/seaweedfs/seaweedfs/weed/util"
jsonpb "google.golang.org/protobuf/encoding/protojson"
)
// MaybeLoadVolumeInfo load the file data as *volume_server_pb.VolumeInfo, the returned volumeInfo will not be nil
func MaybeLoadVolumeInfo(fileName string) (volumeInfo *volume_server_pb.VolumeInfo, hasRemoteFile bool, hasVolumeInfoFile bool, err error) {
volumeInfo = &volume_server_pb.VolumeInfo{}
glog.V(1).Infof("maybeLoadVolumeInfo checks %s", fileName)
if exists, canRead, _, _, _ := util.CheckFile(fileName); !exists || !canRead {
if !exists {
return
}
hasVolumeInfoFile = true
if !canRead {
glog.Warningf("can not read %s", fileName)
err = fmt.Errorf("can not read %s", fileName)
return
}
return
}
hasVolumeInfoFile = true
glog.V(1).Infof("maybeLoadVolumeInfo reads %s", fileName)
fileData, readErr := os.ReadFile(fileName)
if readErr != nil {
glog.Warningf("fail to read %s : %v", fileName, readErr)
err = fmt.Errorf("fail to read %s : %v", fileName, readErr)
return
}
// Handle empty .vif files gracefully - treat as if file doesn't exist
// This can happen when ec.decode copies from a source that doesn't have a .vif file
if len(fileData) == 0 {
glog.Warningf("empty volume info file %s, treating as non-existent", fileName)
hasVolumeInfoFile = false
return
}
glog.V(1).Infof("maybeLoadVolumeInfo Unmarshal volume info %v", fileName)
if err = jsonpb.Unmarshal(fileData, volumeInfo); err != nil {
if oldVersionErr := tryOldVersionVolumeInfo(fileData, volumeInfo); oldVersionErr != nil {
glog.Warningf("unmarshal error: %v oldFormat: %v", err, oldVersionErr)
err = fmt.Errorf("unmarshal error: %w oldFormat: %v", err, oldVersionErr)
return
} else {
err = nil
}
}
if len(volumeInfo.GetFiles()) == 0 {
return
}
hasRemoteFile = true
return
}
// NotCrashDurableError indicates that the .vif file was renamed
// successfully but the directory entry may not survive a crash. The
// on-disk file already holds the new metadata, so callers should keep
// in-memory state aligned with the file rather than rolling back, while
// still propagating the durability failure to the user.
type NotCrashDurableError struct {
FileName string
Err error
}
func (e *NotCrashDurableError) Error() string {
return fmt.Sprintf("volume info %s saved but not crash-durable: %v", e.FileName, e.Err)
}
func (e *NotCrashDurableError) Unwrap() error { return e.Err }
func SaveVolumeInfo(fileName string, volumeInfo *volume_server_pb.VolumeInfo) error {
if exists, _, canWrite, _, _ := util.CheckFile(fileName); exists && !canWrite {
return fmt.Errorf("failed to check %s not writable", fileName)
}
m := jsonpb.MarshalOptions{
AllowPartial: true,
EmitUnpopulated: true,
Indent: " ",
}
text, marshalErr := m.Marshal(volumeInfo)
if marshalErr != nil {
return fmt.Errorf("failed to marshal %s: %v", fileName, marshalErr)
}
// Write atomically so a write/sync/close failure leaves the existing
// .vif file intact. PersistReadOnly rolls back in-memory state on
// error; the atomic rename guarantees the durable file still matches
// that rolled-back state rather than the requested mode. Use a
// unique temp file so concurrent saves for the same volume do not
// collide on a shared .tmp path.
f, err := os.CreateTemp(filepath.Dir(fileName), filepath.Base(fileName)+".tmp.*")
if err != nil {
return fmt.Errorf("failed to create temp file for %s: %w", fileName, err)
}
tmpName := f.Name()
if _, err := f.Write(text); err != nil {
f.Close()
os.Remove(tmpName)
return fmt.Errorf("failed to write %s: %w", fileName, err)
}
if err := f.Chmod(0644); err != nil {
f.Close()
os.Remove(tmpName)
return fmt.Errorf("failed to chmod %s: %w", fileName, err)
}
if err := f.Sync(); err != nil {
f.Close()
os.Remove(tmpName)
return fmt.Errorf("failed to sync %s: %w", fileName, err)
}
if err := f.Close(); err != nil {
os.Remove(tmpName)
return fmt.Errorf("failed to close %s: %w", fileName, err)
}
if err := os.Rename(tmpName, fileName); err != nil {
os.Remove(tmpName)
return fmt.Errorf("failed to rename %s: %w", fileName, err)
}
// The rename has committed the new metadata to the on-disk file.
// A directory fsync failure only risks losing the rename across a
// crash; the file content is already correct, so callers must not
// roll back in-memory state. Return NotCrashDurableError so they
// can distinguish this from a pre-commit failure and keep state
// aligned with the renamed file while still reporting the issue.
if err := util.FsyncDir(filepath.Dir(fileName)); err != nil {
glog.Warningf("fsync dir for %s: %v", fileName, err)
return &NotCrashDurableError{FileName: fileName, Err: err}
}
return nil
}
func tryOldVersionVolumeInfo(data []byte, volumeInfo *volume_server_pb.VolumeInfo) error {
oldVersionVolumeInfo := &volume_server_pb.OldVersionVolumeInfo{}
if err := jsonpb.Unmarshal(data, oldVersionVolumeInfo); err != nil {
return fmt.Errorf("failed to unmarshal old version volume info: %w", err)
}
volumeInfo.Files = oldVersionVolumeInfo.Files
volumeInfo.Version = oldVersionVolumeInfo.Version
volumeInfo.Replication = oldVersionVolumeInfo.Replication
volumeInfo.BytesOffset = oldVersionVolumeInfo.BytesOffset
volumeInfo.DatFileSize = oldVersionVolumeInfo.DatFileSize
volumeInfo.ExpireAtSec = oldVersionVolumeInfo.DestroyTime
volumeInfo.ReadOnly = oldVersionVolumeInfo.ReadOnly
return nil
}