Files
versitygw/backend/posix/with_otmpfile.go
T
Kyd CaoandBen McClelland 45a532e6a6 feat: add configurable file permissions for new objects
Replace the hardcoded 0644 defaultFilePerm with a NewFilePerm option on
the posix and scoutfs backends, exposed as the --file-perms flag and
VGW_FILE_PERMS env var alongside the existing dir-perms option.

The mode passed to open() is masked by the process umask, so the
O_TMPFILE path now chmods explicitly to match the CreateTemp fallback
path and give new objects the configured mode regardless of umask.
2026-08-28 08:51:03 -07:00

400 lines
11 KiB
Go

// Copyright 2023 Versity Software
// This file is licensed under the Apache License, Version 2.0
// (the "License"); you may not use this file except in compliance
// with the License. You may obtain a copy of the License at
//
// http://www.apache.org/licenses/LICENSE-2.0
//
// Unless required by applicable law or agreed to in writing,
// software distributed under the License is distributed on an
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
// KIND, either express or implied. See the License for the
// specific language governing permissions and limitations
// under the License.
//go:build linux
// +build linux
package posix
import (
"crypto/sha256"
"errors"
"fmt"
"io/fs"
"os"
"path/filepath"
"strconv"
"strings"
"syscall"
"time"
"github.com/versity/versitygw/auth"
"github.com/versity/versitygw/backend"
"github.com/versity/versitygw/s3err"
"golang.org/x/sys/unix"
)
const procfddir = "/proc/self/fd"
type tmpfile struct {
f *os.File
bucket string
objname string
isOTmp bool
procFDName string
useODirect bool
size int64
doChown bool
uid int
gid int
newDirPerm fs.FileMode
newFilePerm fs.FileMode
}
func (p *Posix) openTmpFile(dir, bucket, obj string, size int64, acct auth.Account, dofalloc bool, forceNoTmpFile bool, allowODirect odirectPolicy) (*tmpfile, error) {
uid, gid, doChown := p.getChownIDs(acct)
if forceNoTmpFile {
return p.openMkTemp(dir, bucket, obj, size, dofalloc, uid, gid, doChown, allowODirect)
}
// O_TMPFILE allows for a file handle to an unnamed file in the filesystem.
// This can help reduce contention within the namespace (parent directories),
// etc. And will auto cleanup the inode on close if we never link this
// file descriptor into the namespace.
// Not all filesystems support this, so fallback to CreateTemp for when
// this is not supported.
openFlags := unix.O_RDWR | unix.O_TMPFILE | unix.O_CLOEXEC
useODirect := false
if p.enableODirect && bool(allowODirect) {
openFlags |= unix.O_DIRECT
useODirect = true
}
filePerm := uint32(p.newFilePerm.Perm())
fd, err := unix.Open(dir, openFlags, filePerm)
if err != nil {
if errors.Is(err, syscall.EROFS) {
return nil, s3err.GetAPIError(s3err.ErrMethodNotAllowed)
}
if p.enableODirect && bool(allowODirect) && isODirectUnsupportedOpenErr(err) {
warnODirectUnsupportedOnce("openTmpFile", err)
fd, err = unix.Open(dir, unix.O_RDWR|unix.O_TMPFILE|unix.O_CLOEXEC, filePerm)
if err == nil {
useODirect = false
} else if errors.Is(err, syscall.EROFS) {
return nil, s3err.GetAPIError(s3err.ErrMethodNotAllowed)
}
}
if err != nil {
// O_TMPFILE not supported, try fallback
return p.openMkTemp(dir, bucket, obj, size, dofalloc, uid, gid, doChown, allowODirect)
}
}
// for O_TMPFILE, filename is /proc/self/fd/<fd> to be used
// later to link file into namespace
f := os.NewFile(uintptr(fd), filepath.Join(procfddir, strconv.Itoa(fd)))
// The mode passed to open() is masked by the process umask. Set the
// configured mode explicitly so new objects get the same permissions
// regardless of umask, and regardless of whether this or the CreateTemp
// fallback path (which also chmods) created the file.
err = f.Chmod(p.newFilePerm)
if err != nil {
f.Close()
return nil, fmt.Errorf("set temp file mode: %w", err)
}
tmp := &tmpfile{
f: f,
bucket: bucket,
objname: obj,
isOTmp: true,
procFDName: strconv.Itoa(fd),
useODirect: useODirect,
size: size,
doChown: doChown,
uid: uid,
gid: gid,
newDirPerm: p.newDirPerm,
newFilePerm: p.newFilePerm,
}
// falloc is best effort, its fine if this fails
if size > 0 && dofalloc {
tmp.falloc()
}
if doChown {
err := f.Chown(uid, gid)
if err != nil {
f.Close()
return nil, fmt.Errorf("set temp file ownership: %w", err)
}
}
return tmp, nil
}
func (p *Posix) openMkTemp(dir, bucket, obj string, size int64, dofalloc bool, uid, gid int, doChown bool, allowODirect odirectPolicy) (*tmpfile, error) {
err := backend.MkdirAll(dir, uid, gid, doChown, p.newDirPerm)
if err != nil {
if errors.Is(err, syscall.EROFS) {
return nil, s3err.GetAPIError(s3err.ErrMethodNotAllowed)
}
return nil, fmt.Errorf("make temp dir: %w", err)
}
f, err := os.CreateTemp(dir,
fmt.Sprintf("%x.", sha256.Sum256([]byte(obj))))
if err != nil {
if errors.Is(err, syscall.EROFS) {
return nil, s3err.GetAPIError(s3err.ErrMethodNotAllowed)
}
return nil, err
}
useODirect := false
if p.enableODirect && bool(allowODirect) {
name := f.Name()
if err := f.Close(); err != nil {
os.Remove(name)
return nil, fmt.Errorf("close temp file before O_DIRECT reopen: %w", err)
}
fd, err := unix.Open(name, unix.O_RDWR|unix.O_CLOEXEC|unix.O_DIRECT, uint32(p.newFilePerm.Perm()))
if err == nil {
f = os.NewFile(uintptr(fd), name)
useODirect = true
} else if isODirectUnsupportedOpenErr(err) {
warnODirectUnsupportedOnce("openMkTemp", err)
f, err = os.OpenFile(name, os.O_RDWR, 0)
if err != nil {
os.Remove(name)
return nil, fmt.Errorf("reopen temp file after O_DIRECT fallback: %w", err)
}
} else {
os.Remove(name)
return nil, fmt.Errorf("open temp file with O_DIRECT: %w", err)
}
}
tmp := &tmpfile{
f: f,
bucket: bucket,
objname: obj,
useODirect: useODirect,
size: size,
doChown: doChown,
uid: uid,
gid: gid,
newDirPerm: p.newDirPerm,
newFilePerm: p.newFilePerm,
}
// falloc is best effort, its fine if this fails
if size > 0 && dofalloc {
tmp.falloc()
}
if doChown {
err := f.Chown(uid, gid)
if err != nil {
f.Close()
os.Remove(f.Name())
return nil, fmt.Errorf("set temp file ownership: %w", err)
}
}
return tmp, nil
}
func (tmp *tmpfile) falloc() error {
err := syscall.Fallocate(int(tmp.f.Fd()), 0, 0, tmp.size)
if err != nil {
return fmt.Errorf("fallocate: %w", err)
}
return nil
}
const (
maxTmpFileNameRetries = 3
maxDirRecreateRetries = 3
initialBackoffMs = 1
maxBackoffMs = 1024 // ~1 second
)
// linkatOTmpfile links the O_TMPFILE identified by procdir/fdName into dir/basename.
// Handles EEXIST by linking to a temporary name and atomically renaming it into place.
func linkatOTmpfile(procdirFd, dirFd int, fdName, basename string) error {
err := unix.Linkat(procdirFd, fdName, dirFd, basename, unix.AT_SYMLINK_FOLLOW)
if !errors.Is(err, syscall.EEXIST) {
return err
}
// Linkat cannot overwrite an existing file; link to a temp name then rename atomically.
for retries := 1; ; retries++ {
tmpName := fmt.Sprintf(".%s.sgwtmp.%d", basename, time.Now().UnixNano())
err := unix.Linkat(procdirFd, fdName, dirFd, tmpName, unix.AT_SYMLINK_FOLLOW)
if errors.Is(err, syscall.EEXIST) && retries < maxTmpFileNameRetries {
continue
}
if err != nil {
return fmt.Errorf("cannot find free temporary file: %w", err)
}
err = unix.Renameat(dirFd, tmpName, dirFd, basename)
if err != nil {
// cleanup temp name previously linked into namespace
_ = unix.Unlinkat(dirFd, tmpName, 0)
return fmt.Errorf("overwriting renameat failed: %w", err)
}
return nil
}
}
func (tmp *tmpfile) link() error {
// make sure this is cleaned up in all error cases
defer tmp.f.Close()
// We use Linkat/Rename as the atomic operation for object puts. The
// upload is written to a temp (or unnamed/O_TMPFILE) file to not conflict
// with any other simultaneous uploads. The final operation is to move the
// temp file into place for the object. This ensures the object semantics
// of last upload completed wins and is not some combination of writes
// from simultaneous uploads.
objPath := filepath.Join(tmp.bucket, tmp.objname)
dir := filepath.Dir(objPath)
err := backend.MkdirAll(dir, tmp.uid, tmp.gid, tmp.doChown, tmp.newDirPerm)
if err != nil {
return fmt.Errorf("make parent dir: %w", err)
}
if !tmp.isOTmp {
// O_TMPFILE not supported, use fallback
return tmp.fallbackLink()
}
procdir, err := os.Open(procfddir)
if err != nil {
return fmt.Errorf("open proc dir: %w", err)
}
defer procdir.Close()
backoffMs := initialBackoffMs
var dirf *os.File
for {
dirf, err = os.Open(dir)
if errors.Is(err, fs.ErrNotExist) {
err := backend.MkdirAll(dir, tmp.uid, tmp.gid, tmp.doChown, tmp.newDirPerm)
if err != nil {
return fmt.Errorf("make parent dir: %w", err)
}
continue
}
if err != nil {
return fmt.Errorf("open parent dir: %w", err)
}
srcFDName := tmp.procFDName
if srcFDName == "" {
srcFDName = filepath.Base(tmp.f.Name())
}
err = linkatOTmpfile(int(procdir.Fd()), int(dirf.Fd()),
srcFDName, filepath.Base(objPath))
dirf.Close()
if errors.Is(err, syscall.ENOENT) {
// The directory was removed between open and linkat; backoff and retry.
// Add jitter to avoid synchronized retry waves.
sleepWithJitter(backoffMs)
backoffMs = min((backoffMs * 2), maxBackoffMs)
mkErr := backend.MkdirAll(dir, tmp.uid, tmp.gid, tmp.doChown, tmp.newDirPerm)
if mkErr != nil {
return fmt.Errorf("make parent dir: %w", mkErr)
}
continue
}
if err != nil {
return fmt.Errorf("link tmpfile (fd %q as %q): %w",
srcFDName, objPath, err)
}
break
}
err = tmp.f.Close()
if err != nil {
return fmt.Errorf("close tmpfile: %w", err)
}
return nil
}
func (tmp *tmpfile) fallbackLink() error {
tempname := tmp.f.Name()
// reset default file mode because CreateTemp uses 0600
tmp.f.Chmod(tmp.newFilePerm)
err := tmp.f.Close()
if err != nil {
return fmt.Errorf("close tmpfile: %w", err)
}
objPath := filepath.Join(tmp.bucket, tmp.objname)
dir := filepath.Dir(objPath)
err = os.Rename(tempname, objPath)
if errors.Is(err, syscall.ENOENT) {
// The parent directory was concurrently removed; backoff and retry.
backoffMs := initialBackoffMs
for range maxDirRecreateRetries {
// Add jitter to avoid synchronized retry waves.
sleepWithJitter(backoffMs)
backoffMs = min((backoffMs * 2), maxBackoffMs)
err = backend.MkdirAll(dir, tmp.uid, tmp.gid, tmp.doChown, tmp.newDirPerm)
if err != nil {
return fmt.Errorf("recreate parent dir: %w", err)
}
err = os.Rename(tempname, objPath)
if !errors.Is(err, syscall.ENOENT) {
break
}
}
}
if err != nil {
// rename only works for files within the same filesystem
// if this fails fallback to copy
backoffMs := initialBackoffMs
for range maxDirRecreateRetries {
err = backend.MoveFile(tempname, objPath, tmp.newFilePerm)
if !errors.Is(err, syscall.ENOENT) {
break
}
// Add jitter to avoid synchronized retry waves.
sleepWithJitter(backoffMs)
backoffMs = min((backoffMs * 2), maxBackoffMs)
// The parent directory was concurrently removed; recreate and retry.
mkErr := backend.MkdirAll(dir, tmp.uid, tmp.gid, tmp.doChown, tmp.newDirPerm)
if mkErr != nil {
return fmt.Errorf("recreate parent dir: %w", mkErr)
}
}
return err
}
return nil
}
func (tmp *tmpfile) cleanup() {
tmp.f.Close()
if !strings.HasPrefix(tmp.f.Name(), procfddir) {
os.Remove(tmp.f.Name())
}
}