Files
seaweedfs/weed/server/volume_grpc_copy.go
T
Chris LuandGitHub 37f3dff677 volume: validate the file extension in CopyFile and ReceiveFile (#10644)
* volume: validate the file extension in CopyFile and ReceiveFile

CopyFile and ReceiveFile build an on-disk path from the client-supplied
Ext. Both are intentionally ungated for cluster-internal peers, so a
value like "/../../x" is joined onto the volume directory and, once
path-cleaned, resolves outside it -- an EC-shard receive can then write,
and CopyFile read, anywhere the process can reach.

Constrain Ext to a real suffix (a leading dot followed by alphanumerics)
before it is used to build any path, so it can no longer carry a
separator or a parent reference.

* test: use an alphanumeric missing-file extension in the copy variants

The not-found and stop-offset-zero cases used ".definitely-missing" as a
deliberately absent source. The extension is now validated, and the hyphen
makes it invalid, so switch to ".missing" -- still a nonexistent file, but a
real extension shape.

* volume: validate the collection in CopyFile and ReceiveFile

The client-supplied Collection is folded into the on-disk path as
"<collection>_<vid>" by VolumeFileName and EcShardBaseFileName, both joined
with path.Join / util.Join. A Collection carrying a separator, e.g.
"../../x", therefore path-cleans to a target outside the volume directory,
the same escape the extension check just closed. Reject a collection that is
a bare parent reference or holds a separator; ordinary names ('.', '-' and
all) still pass.
2026-08-08 09:25:57 -07:00

835 lines
31 KiB
Go

package weed_server
import (
"context"
"fmt"
"io"
"math"
"os"
"strings"
"time"
"github.com/seaweedfs/seaweedfs/weed/pb/master_pb"
"github.com/seaweedfs/seaweedfs/weed/storage/backend"
"github.com/seaweedfs/seaweedfs/weed/glog"
"github.com/seaweedfs/seaweedfs/weed/operation"
"github.com/seaweedfs/seaweedfs/weed/pb"
"github.com/seaweedfs/seaweedfs/weed/pb/volume_server_pb"
"github.com/seaweedfs/seaweedfs/weed/storage"
"github.com/seaweedfs/seaweedfs/weed/storage/erasure_coding"
"github.com/seaweedfs/seaweedfs/weed/storage/idx"
"github.com/seaweedfs/seaweedfs/weed/storage/needle"
"github.com/seaweedfs/seaweedfs/weed/storage/types"
"github.com/seaweedfs/seaweedfs/weed/util"
)
const BufferSizeLimit = 1024 * 1024 * 2
// VolumeCopy copy the .idx .dat .vif files, and mount the volume
func (vs *VolumeServer) VolumeCopy(req *volume_server_pb.VolumeCopyRequest, stream volume_server_pb.VolumeServer_VolumeCopyServer) error {
if err := vs.checkGrpcAdminAuth(stream.Context()); err != nil {
return err
}
if err := vs.CheckMaintenanceMode(); err != nil {
return err
}
// A pre-existing local replica is NOT deleted up front. Deleting before the
// source is confirmed reachable destroys a healthy copy on a transient
// source outage (and, on retry, can lose the volume entirely). The delete is
// deferred until ReadVolumeFileStatus below proves the source holds the
// volume; readability alone is the gate (size/count comparisons invert after
// divergent vacuum/compaction and would block valid re-replication).
hasExistingVolume := vs.store.GetVolume(needle.VolumeId(req.VolumeId)) != nil
// the master will not start compaction for read-only volumes, so it is safe to just copy files directly
// copy .dat and .idx files
// read .idx .dat file size and timestamp
// send .idx file
// send .dat file
// confirm size and timestamp
var volFileInfoResp *volume_server_pb.ReadVolumeFileStatusResponse
var dataBaseFileName, indexBaseFileName, idxFileName, datFileName string
var hasRemoteDatFile bool
err := operation.WithVolumeServerClient(true, pb.ServerAddress(req.SourceDataNode), vs.grpcDialOption, func(client volume_server_pb.VolumeServerClient) error {
var err error
volFileInfoResp, err = client.ReadVolumeFileStatus(context.Background(),
&volume_server_pb.ReadVolumeFileStatusRequest{
VolumeId: req.VolumeId,
})
if nil != err {
return fmt.Errorf("read volume file status failed, %w", err)
}
// Source is reachable and holds the volume: only now is it safe to drop
// an existing local replica before overwriting its files.
if hasExistingVolume {
glog.V(0).Infof("volume %d already exists. deleting before copying from %s...", req.VolumeId, req.SourceDataNode)
// keep remote data: the inbound copy carries a .vif that may point at
// the same cloud-tier object the existing volume references.
if delErr := vs.store.DeleteVolume(needle.VolumeId(req.VolumeId), false, true); delErr != nil {
return fmt.Errorf("failed to delete existing volume %d: %v", req.VolumeId, delErr)
}
glog.V(0).Infof("deleted existing volume %d before copying.", req.VolumeId)
}
diskType := volFileInfoResp.DiskType
if req.DiskType != "" {
diskType = req.DiskType
}
hasRemoteDatFile = volFileInfoResp.VolumeInfo != nil && len(volFileInfoResp.VolumeInfo.Files) > 0
// a remote-backed volume only lands its .idx/.vif locally; the .dat stays in the tier
neededSpace := volFileInfoResp.DatFileSize
if hasRemoteDatFile {
neededSpace = volFileInfoResp.IdxFileSize
}
location := vs.store.FindFreeLocation(func(location *storage.DiskLocation) bool {
return location.DiskType == types.ToDiskType(diskType) &&
location.AvailableSpace.Load() > neededSpace
})
if location == nil {
return fmt.Errorf("%s %s", util.ErrVolumeNoSpaceLeft, types.ToDiskType(diskType).ReadableString())
}
dataBaseFileName = storage.VolumeFileName(location.Directory, volFileInfoResp.Collection, int(req.VolumeId))
indexBaseFileName = storage.VolumeFileName(location.IdxDirectory, volFileInfoResp.Collection, int(req.VolumeId))
// The .note marks the copy as in-progress; a leftover note fails the
// volume load on restart, so a write failure must abort the copy.
if noteErr := util.WriteFile(dataBaseFileName+".note", []byte(fmt.Sprintf("copying from %s", req.SourceDataNode)), 0755); noteErr != nil {
err = noteErr
return fmt.Errorf("write .note for volume %d: %w", req.VolumeId, noteErr)
}
defer func() {
if err != nil {
os.Remove(dataBaseFileName + ".dat")
os.Remove(indexBaseFileName + ".idx")
os.Remove(dataBaseFileName + ".vif")
os.Remove(dataBaseFileName + ".note")
}
}()
var preallocateSize int64
if grpcErr := pb.WithMasterClient(context.Background(), false, vs.GetMaster(context.Background()), vs.grpcDialOption, false, func(client master_pb.SeaweedClient) error {
resp, err := client.GetMasterConfiguration(context.Background(), &master_pb.GetMasterConfigurationRequest{})
if err != nil {
return fmt.Errorf("get master %s configuration: %v", vs.GetMaster(context.Background()), err)
}
if resp.VolumePreallocate {
preallocateSize = int64(resp.VolumeSizeLimitMB) * (1 << 20)
}
return nil
}); grpcErr != nil {
glog.V(0).Infof("connect to %s: %v", vs.GetMaster(context.Background()), grpcErr)
}
if preallocateSize > 0 && !hasRemoteDatFile {
volumeFile := dataBaseFileName + ".dat"
_, err := backend.CreateVolumeFile(volumeFile, preallocateSize, 0)
if err != nil {
return fmt.Errorf("create volume file %s: %v", volumeFile, err)
}
}
// println("source:", volFileInfoResp.String())
copyResponse := &volume_server_pb.VolumeCopyResponse{}
reportInterval := int64(1024 * 1024 * 128)
nextReportTarget := reportInterval
var modifiedTsNs int64
var sendErr error
var ioBytePerSecond int64
if req.IoBytePerSecond <= 0 {
ioBytePerSecond = vs.maintenanceBytePerSecond
} else {
ioBytePerSecond = req.IoBytePerSecond
}
throttler := util.NewWriteThrottler(ioBytePerSecond)
if !hasRemoteDatFile {
if modifiedTsNs, err = vs.doCopyFileWithThrottler(client, false, req.Collection, req.VolumeId, volFileInfoResp.CompactionRevision, volFileInfoResp.DatFileSize, dataBaseFileName, ".dat", false, true, func(processed int64) bool {
if processed > nextReportTarget {
copyResponse.ProcessedBytes = processed
if sendErr = stream.Send(copyResponse); sendErr != nil {
return false
}
nextReportTarget = processed + reportInterval
}
return true
}, throttler); err != nil {
return err
}
if sendErr != nil {
return sendErr
}
if modifiedTsNs > 0 {
os.Chtimes(dataBaseFileName+".dat", time.Unix(0, modifiedTsNs), time.Unix(0, modifiedTsNs))
}
}
if modifiedTsNs, err = vs.doCopyFileWithThrottler(client, false, req.Collection, req.VolumeId, volFileInfoResp.CompactionRevision, volFileInfoResp.IdxFileSize, indexBaseFileName, ".idx", false, false, nil, throttler); err != nil {
return err
}
if modifiedTsNs > 0 {
os.Chtimes(indexBaseFileName+".idx", time.Unix(0, modifiedTsNs), time.Unix(0, modifiedTsNs))
}
if modifiedTsNs, err = vs.doCopyFileWithThrottler(client, false, req.Collection, req.VolumeId, volFileInfoResp.CompactionRevision, 1024*1024, dataBaseFileName, ".vif", false, true, nil, throttler); err != nil {
return err
}
if modifiedTsNs > 0 {
os.Chtimes(dataBaseFileName+".vif", time.Unix(0, modifiedTsNs), time.Unix(0, modifiedTsNs))
}
// A leftover .note fails the load on the next restart, so a removal
// failure must fail the copy rather than be silently swallowed.
if noteErr := os.Remove(dataBaseFileName + ".note"); noteErr != nil && !os.IsNotExist(noteErr) {
err = noteErr
return fmt.Errorf("remove .note for volume %d: %w", req.VolumeId, noteErr)
}
return nil
})
if err != nil {
return err
}
if dataBaseFileName == "" {
return fmt.Errorf("not found volume %d file", req.VolumeId)
}
idxFileName = indexBaseFileName + ".idx"
datFileName = dataBaseFileName + ".dat"
defer func() {
if err != nil && dataBaseFileName != "" {
os.Remove(idxFileName)
os.Remove(datFileName)
os.Remove(dataBaseFileName + ".vif")
}
}()
if err = checkCopyFiles(volFileInfoResp, hasRemoteDatFile, idxFileName, datFileName); err != nil { // added by panyc16
return err
}
var lastAppendAtNs = volFileInfoResp.DatFileTimestampSeconds * uint64(time.Second)
if !hasRemoteDatFile {
if appendAtNs, appendErr := findLastAppendAtNsFromCopiedFiles(idxFileName, datFileName, needle.Version(volFileInfoResp.Version)); appendErr == nil && appendAtNs > 0 {
lastAppendAtNs = appendAtNs
} else if appendErr != nil {
glog.V(1).Infof("failed to find last append timestamp for volume %d: %v", req.VolumeId, appendErr)
}
}
// mount the volume
err = vs.store.MountVolume(needle.VolumeId(req.VolumeId))
if err != nil {
return fmt.Errorf("failed to mount volume %d: %v", req.VolumeId, err)
}
if err = stream.Send(&volume_server_pb.VolumeCopyResponse{
LastAppendAtNs: lastAppendAtNs,
}); err != nil {
glog.Errorf("send response: %v", err)
}
return err
}
func (vs *VolumeServer) doCopyFile(client volume_server_pb.VolumeServerClient, isEcVolume bool, collection string, vid, compactRevision uint32, stopOffset uint64, baseFileName, ext string, isAppend, ignoreSourceFileNotFound bool, progressFn storage.ProgressFunc) (modifiedTsNs int64, err error) {
return vs.doCopyFileWithThrottler(client, isEcVolume, collection, vid, compactRevision, stopOffset, baseFileName, ext, isAppend, ignoreSourceFileNotFound, progressFn, util.NewWriteThrottler(vs.maintenanceBytePerSecond))
}
func (vs *VolumeServer) doCopyFileWithThrottler(client volume_server_pb.VolumeServerClient, isEcVolume bool, collection string, vid, compactRevision uint32, stopOffset uint64, baseFileName, ext string, isAppend, ignoreSourceFileNotFound bool, progressFn storage.ProgressFunc, throttler *util.WriteThrottler) (modifiedTsNs int64, err error) {
copyFileClient, err := client.CopyFile(context.Background(), &volume_server_pb.CopyFileRequest{
VolumeId: vid,
Ext: ext,
CompactionRevision: compactRevision,
StopOffset: stopOffset,
Collection: collection,
IsEcVolume: isEcVolume,
IgnoreSourceFileNotFound: ignoreSourceFileNotFound,
})
if err != nil {
return modifiedTsNs, fmt.Errorf("failed to start copying volume %d %s file: %v", vid, ext, err)
}
modifiedTsNs, err = writeToFile(copyFileClient, baseFileName+ext, throttler, isAppend, ignoreSourceFileNotFound, progressFn)
if err != nil {
return modifiedTsNs, fmt.Errorf("failed to copy %s file: %v", baseFileName+ext, err)
}
return modifiedTsNs, nil
}
/*
*
only check the differ of the file size
todo: maybe should check the received count and deleted count of the volume
*/
func checkCopyFiles(originFileInf *volume_server_pb.ReadVolumeFileStatusResponse, hasRemoteDatFile bool, idxFileName, datFileName string) error {
stat, err := os.Stat(idxFileName)
if err != nil {
// If the idx file doesn't exist but the expected size is 0, that's OK (empty volume)
if os.IsNotExist(err) && originFileInf.IdxFileSize == 0 {
// empty volume, idx file not needed
} else {
return fmt.Errorf("stat idx file %s failed: %v", idxFileName, err)
}
} else if originFileInf.IdxFileSize != uint64(stat.Size()) {
return fmt.Errorf("idx file %s size [%v] is not same as origin file size [%v]",
idxFileName, stat.Size(), originFileInf.IdxFileSize)
}
if hasRemoteDatFile {
return nil
}
stat, err = os.Stat(datFileName)
if err != nil {
return fmt.Errorf("get dat file info failed, %w", err)
}
if originFileInf.DatFileSize != uint64(stat.Size()) {
return fmt.Errorf("the dat file size [%v] is not same as origin file size [%v]",
stat.Size(), originFileInf.DatFileSize)
}
return nil
}
func findLastAppendAtNsFromCopiedFiles(idxFileName, datFileName string, version needle.Version) (uint64, error) {
if version < needle.Version3 {
return 0, nil
}
idxFile, err := os.Open(idxFileName)
if err != nil {
return 0, fmt.Errorf("open idx file %s: %w", idxFileName, err)
}
defer idxFile.Close()
fi, err := idxFile.Stat()
if err != nil {
return 0, fmt.Errorf("stat idx file %s: %w", idxFileName, err)
}
if fi.Size() == 0 {
return 0, nil
}
if fi.Size()%int64(types.NeedleMapEntrySize) != 0 {
return 0, fmt.Errorf("unexpected idx file %s size: %d", idxFileName, fi.Size())
}
buf := make([]byte, types.NeedleMapEntrySize)
if _, err := idxFile.ReadAt(buf, fi.Size()-int64(types.NeedleMapEntrySize)); err != nil {
return 0, fmt.Errorf("read idx file %s: %w", idxFileName, err)
}
_, offset, _ := idx.IdxFileEntry(buf)
if offset.IsZero() {
return 0, nil
}
datFile, err := os.Open(datFileName)
if err != nil {
return 0, fmt.Errorf("open dat file %s: %w", datFileName, err)
}
defer datFile.Close()
datBackend := backend.NewDiskFile(datFile)
n, _, _, err := needle.ReadNeedleHeader(datBackend, version, offset.ToActualOffset())
if err != nil {
return 0, fmt.Errorf("read needle header %s offset %d: %w", datFileName, offset.ToActualOffset(), err)
}
tailOffset := offset.ToActualOffset() + int64(types.NeedleHeaderSize) + int64(n.Size)
tail := make([]byte, needle.NeedleChecksumSize+types.TimestampSize)
readCount, readErr := datBackend.ReadAt(tail, tailOffset)
if readErr == io.EOF && readCount == len(tail) {
readErr = nil
}
if readErr != nil {
return 0, fmt.Errorf("read needle tail %s offset %d: %w", datFileName, tailOffset, readErr)
}
return util.BytesToUint64(tail[needle.NeedleChecksumSize : needle.NeedleChecksumSize+types.TimestampSize]), nil
}
func writeToFile(client volume_server_pb.VolumeServer_CopyFileClient, fileName string, wt *util.WriteThrottler, isAppend, ignoreSourceFileNotFound bool, progressFn storage.ProgressFunc) (modifiedTsNs int64, err error) {
glog.V(4).Infof("writing to %s", fileName)
// For an optional copy (ignoreSourceFileNotFound), stage into a temp sibling
// and atomically rename on success, so a source that lacks the file cannot
// truncate a valid pre-existing destination. Mandatory copies write in place.
writePath := fileName
stageThenCommit := ignoreSourceFileNotFound && !isAppend
if stageThenCommit {
writePath = fileName + ".copying"
}
flags := os.O_WRONLY | os.O_CREATE | os.O_TRUNC
if isAppend {
flags = os.O_WRONLY | os.O_CREATE
}
dst, err := os.OpenFile(writePath, flags, 0644)
if err != nil {
return modifiedTsNs, fmt.Errorf("open file %s: %w", writePath, err)
}
// Track the destination handle through a closer that runs at most once.
// On Windows os.Remove fails while the file is still open, so any path
// that wants to delete the file we just created must close the handle
// first. The deferred call here is the safety net for normal returns.
dstClosed := false
closeDst := func() {
if dstClosed {
return
}
dstClosed = true
_ = dst.Close()
}
defer closeDst()
// removeIncomplete deletes the partially-written file we just opened
// with O_TRUNC. Used on stream / write / cancellation errors so a
// caller (notably VolumeEcShardsCopy distributing .ecx) doesn't end
// up with a 0-byte stub that downstream code mistakes for a valid
// empty file. Skip in isAppend mode — the existing content is not
// ours to remove, and resumable appends rely on partial state.
removeIncomplete := func(reason string) {
if isAppend {
return
}
closeDst()
if removeErr := os.Remove(writePath); removeErr != nil && !os.IsNotExist(removeErr) {
glog.Warningf("failed to remove incomplete file %s after %s: %v", writePath, reason, removeErr)
} else if removeErr == nil {
glog.V(1).Infof("removed incomplete file %s after %s", writePath, reason)
}
}
var progressedBytes int64
for {
resp, receiveErr := client.Recv()
if receiveErr == io.EOF {
break
}
if resp != nil && resp.ModifiedTsNs != 0 {
modifiedTsNs = resp.ModifiedTsNs
}
if receiveErr != nil {
removeIncomplete("receive error")
return modifiedTsNs, fmt.Errorf("receiving %s: %w", fileName, receiveErr)
}
if _, writeErr := dst.Write(resp.FileContent); writeErr != nil {
removeIncomplete("write error")
return modifiedTsNs, fmt.Errorf("write file %s: %w", fileName, writeErr)
}
progressedBytes += int64(len(resp.FileContent))
if progressFn != nil {
if !progressFn(progressedBytes) {
removeIncomplete("progress cancelled")
return modifiedTsNs, fmt.Errorf("interrupted copy operation")
}
}
wt.MaybeSlowdown(int64(len(resp.FileContent)))
}
// If we never received a modifiedTsNs, it means the source file did not exist.
// Remove the empty file we created to avoid leaving corrupted empty files.
// Note: We check modifiedTsNs (not progressedBytes) because an empty source file
// is valid and should result in an empty destination file.
if modifiedTsNs == 0 && !isAppend {
closeDst()
if removeErr := os.Remove(writePath); removeErr != nil && !os.IsNotExist(removeErr) {
glog.V(1).Infof("failed to remove empty file %s: %v", writePath, removeErr)
} else if removeErr == nil {
glog.V(1).Infof("removed empty file %s (source file not found)", writePath)
}
return modifiedTsNs, nil
}
// Commit the staged temp into place.
if stageThenCommit {
closeDst()
if renameErr := os.Rename(writePath, fileName); renameErr != nil {
os.Remove(writePath)
return modifiedTsNs, fmt.Errorf("commit copied file %s: %w", fileName, renameErr)
}
}
return modifiedTsNs, nil
}
func (vs *VolumeServer) ReadVolumeFileStatus(ctx context.Context, req *volume_server_pb.ReadVolumeFileStatusRequest) (*volume_server_pb.ReadVolumeFileStatusResponse, error) {
resp := &volume_server_pb.ReadVolumeFileStatusResponse{}
v := vs.store.GetVolume(needle.VolumeId(req.VolumeId))
if v == nil {
return nil, fmt.Errorf("not found volume id %d", req.VolumeId)
}
resp.VolumeId = req.VolumeId
datSize, idxSize, modTime := v.FileStat()
resp.DatFileSize = datSize
resp.IdxFileSize = idxSize
resp.DatFileTimestampSeconds = uint64(modTime.Unix())
resp.IdxFileTimestampSeconds = uint64(modTime.Unix())
resp.FileCount = v.FileCount()
resp.CompactionRevision = uint32(v.CompactionRevision)
resp.Collection = v.Collection
resp.DiskType = string(v.DiskType())
resp.VolumeInfo = v.GetVolumeInfo()
resp.Version = uint32(v.Version())
return resp, nil
}
// checkVolumeFileExtension guards the client-supplied Ext that CopyFile and
// ReceiveFile turn into an on-disk path. Both RPCs are intentionally ungated
// for cluster-internal peers (see volume_grpc_admin_auth_coverage_test.go), so
// this is the only check standing between a peer request and the os.Open /
// os.Create target: without it an Ext like "/../../x" is joined onto the volume
// directory and, once path-cleaned, resolves outside it. A genuine extension is
// a leading dot followed by alphanumerics -- ".dat", ".idx", ".vif", ".ecx",
// ".ecj", ".ecsum", ".ec00".. -- and never contains a separator or "..".
func checkVolumeFileExtension(ext string) error {
if len(ext) < 2 || ext[0] != '.' {
return fmt.Errorf("invalid file extension %q", ext)
}
for _, r := range ext[1:] {
if r < '0' || (r > '9' && r < 'A') || (r > 'Z' && r < 'a') || r > 'z' {
return fmt.Errorf("invalid file extension %q", ext)
}
}
return nil
}
// checkVolumeCollection guards the client-supplied Collection, which CopyFile
// and ReceiveFile fold into a path component ("<collection>_<vid>"). An empty
// collection is the default; any other value must be a single path element so a
// collection like "../../x" cannot climb out of the volume directory once
// path-cleaned. Collection names are user-facing and may hold '.' or '-', so
// this rejects only separators and bare parent references rather than the
// stricter alphanumeric rule used for extensions.
func checkVolumeCollection(collection string) error {
if collection == "." || collection == ".." || strings.ContainsAny(collection, `/\`) {
return fmt.Errorf("invalid collection %q", collection)
}
return nil
}
// CopyFile client pulls the volume related file from the source server.
// if req.CompactionRevision != math.MaxUint32, it ensures the compact revision is as expected
// The copying still stop at req.StopOffset, but you can set it to math.MaxUint64 in order to read all data.
func (vs *VolumeServer) CopyFile(req *volume_server_pb.CopyFileRequest, stream volume_server_pb.VolumeServer_CopyFileServer) error {
if err := checkVolumeFileExtension(req.Ext); err != nil {
return err
}
if err := checkVolumeCollection(req.Collection); err != nil {
return err
}
var fileName string
if !req.IsEcVolume {
v := vs.store.GetVolume(needle.VolumeId(req.VolumeId))
if v == nil {
return fmt.Errorf("not found volume id %d", req.VolumeId)
}
if uint32(v.CompactionRevision) != req.CompactionRevision && req.CompactionRevision != math.MaxUint32 {
return fmt.Errorf("volume %d is compacted", req.VolumeId)
}
v.SyncToDisk()
fileName = v.FileName(req.Ext)
} else {
// Sync EC volume files to disk before copying to ensure deletions are visible
// This fixes issue #7751 where deleted files in encoded volumes were not
// properly marked as deleted when decoded.
if ecVolume, found := vs.store.FindEcVolume(needle.VolumeId(req.VolumeId)); found {
ecVolume.Sync()
}
baseFileName := erasure_coding.EcShardBaseFileName(req.Collection, int(req.VolumeId)) + req.Ext
for _, location := range vs.store.Locations {
tName := util.Join(location.Directory, baseFileName)
if util.FileExists(tName) {
fileName = tName
}
tName = util.Join(location.IdxDirectory, baseFileName)
if util.FileExists(tName) {
fileName = tName
}
}
if fileName == "" {
if req.IgnoreSourceFileNotFound {
return nil
}
return fmt.Errorf("CopyFile not found ec volume id %d", req.VolumeId)
}
}
bytesToRead := int64(req.StopOffset)
file, err := os.Open(fileName)
if err != nil {
if os.IsNotExist(err) {
// If file doesn't exist and we're asked to copy 0 bytes (empty file),
// or if IgnoreSourceFileNotFound is set, treat as success
if req.IgnoreSourceFileNotFound || req.StopOffset == 0 {
return nil
}
}
return err
}
defer file.Close()
fileInfo, err := file.Stat()
if err != nil {
return err
}
fileModTsNs := fileInfo.ModTime().UnixNano()
buffer := make([]byte, BufferSizeLimit)
for bytesToRead > 0 {
bytesread, err := file.Read(buffer)
// println(fileName, "read", bytesread, "bytes, with target", bytesToRead)
if err != nil {
if err != io.EOF {
return err
}
// println(fileName, "read", bytesread, "bytes, with target", bytesToRead, "err", err.Error())
break
}
if int64(bytesread) > bytesToRead {
bytesread = int(bytesToRead)
}
err = stream.Send(&volume_server_pb.CopyFileResponse{
FileContent: buffer[:bytesread],
ModifiedTsNs: fileModTsNs,
})
if err != nil {
// println("sending", bytesread, "bytes err", err.Error())
return err
}
fileModTsNs = 0 // only send once
bytesToRead -= int64(bytesread)
}
// If no data has been sent in the loop (e.g. for an empty file, or when stopOffset is 0),
// we still need to send the ModifiedTsNs so the client knows the source file exists.
// fileModTsNs is set to 0 after the first send, so if it's still non-zero,
// we haven't sent anything yet.
if fileModTsNs != 0 {
err = stream.Send(&volume_server_pb.CopyFileResponse{
ModifiedTsNs: fileModTsNs,
})
if err != nil {
return err
}
}
return nil
}
// diskHoldsEcShardFile reports whether dir contains any <vid>.ecNN shard file,
// so a decoded <vid>.dat is never staged beside a shard. Unlike FindEcVolume,
// it also catches shards present on disk but not mounted.
func diskHoldsEcShardFile(dir, collection string, vid needle.VolumeId) bool {
base := erasure_coding.EcShardFileName(collection, dir, int(vid))
for i := 0; i < erasure_coding.MaxShardCount; i++ {
if fi, err := os.Stat(base + erasure_coding.ToExt(i)); err == nil && !fi.IsDir() && fi.Size() > 0 {
return true
}
}
return false
}
// ReceiveFile receives a file stream from client and writes it to storage
func (vs *VolumeServer) ReceiveFile(stream volume_server_pb.VolumeServer_ReceiveFileServer) error {
if err := vs.CheckMaintenanceMode(); err != nil {
return err
}
var fileInfo *volume_server_pb.ReceiveFileInfo
var targetFile *os.File
var filePath string
var bytesWritten uint64
defer func() {
if targetFile != nil {
targetFile.Close()
}
}()
for {
req, err := stream.Recv()
if err == io.EOF {
// Stream completed successfully
if targetFile != nil {
targetFile.Sync()
glog.V(1).Infof("Successfully received file %s (%d bytes)", filePath, bytesWritten)
}
return stream.SendAndClose(&volume_server_pb.ReceiveFileResponse{
BytesWritten: bytesWritten,
})
}
if err != nil {
// Clean up on error
if targetFile != nil {
targetFile.Close()
os.Remove(filePath)
}
glog.Errorf("Failed to receive stream: %v", err)
return fmt.Errorf("failed to receive stream: %v", err)
}
switch data := req.Data.(type) {
case *volume_server_pb.ReceiveFileRequest_Info:
// First message contains file info
fileInfo = data.Info
glog.V(1).Infof("ReceiveFile: volume %d, ext %s, collection %s, shard %d, size %d",
fileInfo.VolumeId, fileInfo.Ext, fileInfo.Collection, fileInfo.ShardId, fileInfo.FileSize)
if err := checkVolumeFileExtension(fileInfo.Ext); err != nil {
glog.Errorf("ReceiveFile: %v", err)
return stream.SendAndClose(&volume_server_pb.ReceiveFileResponse{
Error: err.Error(),
})
}
if err := checkVolumeCollection(fileInfo.Collection); err != nil {
glog.Errorf("ReceiveFile: %v", err)
return stream.SendAndClose(&volume_server_pb.ReceiveFileResponse{
Error: err.Error(),
})
}
if fileInfo.IsEcVolume {
// os.Create below truncates in place; a mounted EcVolume
// holds fds on the same inodes, so overwriting corrupts
// live readers.
if _, mounted := vs.store.FindEcVolume(needle.VolumeId(fileInfo.VolumeId)); mounted {
mountedDisks := vs.store.FindEcVolumeDiskIds(needle.VolumeId(fileInfo.VolumeId))
glog.Errorf("ReceiveFile: ec volume %d is mounted on disk_ids:%v; refusing overwrite for %s", fileInfo.VolumeId, mountedDisks, fileInfo.Ext)
return stream.SendAndClose(&volume_server_pb.ReceiveFileResponse{
Error: fmt.Sprintf("ec volume %d is mounted on disk_ids:%v; unmount before ReceiveFile", fileInfo.VolumeId, mountedDisks),
})
}
// disk_id=0 means "unset" (protobuf default), so auto-select
// using the same primitive as VolumeEcShardsCopy: prefer a
// disk that has the EC volume mounted, then a disk that owns
// the .ecx on disk (the volume hasn't been mounted yet —
// relevant when shards stream in mid-rebuild before any
// mount has happened; see #9212), then any HDD, then any
// disk.
var targetLocation *storage.DiskLocation
if fileInfo.DiskId > 0 {
if fileInfo.DiskId >= uint32(len(vs.store.Locations)) {
glog.Errorf("ReceiveFile: invalid disk_id %d: only have %d disks", fileInfo.DiskId, len(vs.store.Locations))
return stream.SendAndClose(&volume_server_pb.ReceiveFileResponse{
Error: fmt.Sprintf("invalid disk_id %d: only have %d disks", fileInfo.DiskId, len(vs.store.Locations)),
})
}
targetLocation = vs.store.Locations[fileInfo.DiskId]
} else {
// Pass the build's default data-shard count for the helper's
// free-slot maths; it's a parameter so custom-ratio builds
// (e.g. enterprise) can swap it without touching this file.
targetLocation = vs.store.FindEcShardTargetLocation(fileInfo.Collection, needle.VolumeId(fileInfo.VolumeId), erasure_coding.DataShardsCount)
}
if targetLocation == nil {
glog.Errorf("ReceiveFile: no storage location available")
return stream.SendAndClose(&volume_server_pb.ReceiveFileResponse{
Error: "no storage location available",
})
}
// Create EC shard file path
baseFileName := erasure_coding.EcShardBaseFileName(fileInfo.Collection, int(fileInfo.VolumeId))
filePath = util.Join(targetLocation.Directory, baseFileName+fileInfo.Ext)
} else {
// Regular volume file
v := vs.store.GetVolume(needle.VolumeId(fileInfo.VolumeId))
if v == nil {
if fileInfo.DiskType == "" {
glog.Errorf("ReceiveFile: volume %d not found", fileInfo.VolumeId)
return stream.SendAndClose(&volume_server_pb.ReceiveFileResponse{
Error: fmt.Sprintf("volume %d not found", fileInfo.VolumeId),
})
}
// Staged-new-volume mode (EC decode onto a clean peer): the
// volume does not exist here yet. Pick a free-slot disk location
// of the requested medium and stage the file as
// <base><ext>.copying, to be renamed into place and mounted by
// VolumeEcShardsToVolume(from_staged). .idx.copying/.vif.copying
// are not valid volume names, so the scanner never half-loads.
want := types.ToDiskType(fileInfo.DiskType)
stagedVid := needle.VolumeId(fileInfo.VolumeId)
loc := vs.store.FindFreeLocation(func(l *storage.DiskLocation) bool {
if l.DiskType != want {
return false
}
// Don't stage the decoded .dat onto a disk that holds a shard
// of this vid. Check the mounted map and the on-disk files, so
// an unmounted or orphan shard (on disk, absent from the map)
// is caught too.
if _, holds := l.FindEcVolume(stagedVid); holds {
return false
}
return !diskHoldsEcShardFile(l.Directory, fileInfo.Collection, stagedVid)
})
if loc == nil {
return stream.SendAndClose(&volume_server_pb.ReceiveFileResponse{
Error: fmt.Sprintf("no %s disk location with a free slot for volume %d", fileInfo.DiskType, fileInfo.VolumeId),
})
}
filePath = storage.VolumeFileName(loc.Directory, fileInfo.Collection, int(fileInfo.VolumeId)) + fileInfo.Ext + ".copying"
} else {
filePath = v.FileName(fileInfo.Ext)
}
}
// Create target file
targetFile, err = os.Create(filePath)
if err != nil {
glog.Errorf("ReceiveFile: failed to create file %s: %v", filePath, err)
return stream.SendAndClose(&volume_server_pb.ReceiveFileResponse{
Error: fmt.Sprintf("failed to create file: %v", err),
})
}
glog.V(1).Infof("ReceiveFile: created target file %s", filePath)
case *volume_server_pb.ReceiveFileRequest_FileContent:
// Subsequent messages contain file content
if targetFile == nil {
glog.Errorf("ReceiveFile: file info must be sent first")
return stream.SendAndClose(&volume_server_pb.ReceiveFileResponse{
Error: "file info must be sent first",
})
}
n, err := targetFile.Write(data.FileContent)
if err != nil {
targetFile.Close()
os.Remove(filePath)
glog.Errorf("ReceiveFile: failed to write to file %s: %v", filePath, err)
return stream.SendAndClose(&volume_server_pb.ReceiveFileResponse{
Error: fmt.Sprintf("failed to write file: %v", err),
})
}
bytesWritten += uint64(n)
glog.V(2).Infof("ReceiveFile: wrote %d bytes to %s (total: %d)", n, filePath, bytesWritten)
default:
glog.Errorf("ReceiveFile: unknown message type")
return stream.SendAndClose(&volume_server_pb.ReceiveFileResponse{
Error: "unknown message type",
})
}
}
}