mirror of
https://github.com/versity/versitygw.git
synced 2026-09-24 00:44:23 +00:00
This change introduces an opt-in O_DIRECT mode for POSIX object data paths while keeping behavior safe and predictable across different filesystems and kernel constraints. We now use direct I/O when available and beneficial, but preserve correctness by falling back to buffered I/O when runtime read behavior indicates alignment or capability mismatches. The implementation keeps fast paths available for full-object reads and descriptor-to-descriptor copy operations so kernel-level optimizations can still be used where possible. At the same time, it avoids global assumptions from single runtime failures and performs fallback at the stream level so requests can continue successfully without broad feature disablement.
389 lines
10 KiB
Go
389 lines
10 KiB
Go
// Copyright 2023 Versity Software
|
|
// This file is licensed under the Apache License, Version 2.0
|
|
// (the "License"); you may not use this file except in compliance
|
|
// with the License. You may obtain a copy of the License at
|
|
//
|
|
// http://www.apache.org/licenses/LICENSE-2.0
|
|
//
|
|
// Unless required by applicable law or agreed to in writing,
|
|
// software distributed under the License is distributed on an
|
|
// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
|
|
// KIND, either express or implied. See the License for the
|
|
// specific language governing permissions and limitations
|
|
// under the License.
|
|
|
|
//go:build linux
|
|
// +build linux
|
|
|
|
package posix
|
|
|
|
import (
|
|
"crypto/sha256"
|
|
"errors"
|
|
"fmt"
|
|
"io/fs"
|
|
"os"
|
|
"path/filepath"
|
|
"strconv"
|
|
"strings"
|
|
"syscall"
|
|
"time"
|
|
|
|
"github.com/versity/versitygw/auth"
|
|
"github.com/versity/versitygw/backend"
|
|
"github.com/versity/versitygw/s3err"
|
|
"golang.org/x/sys/unix"
|
|
)
|
|
|
|
const procfddir = "/proc/self/fd"
|
|
|
|
type tmpfile struct {
|
|
f *os.File
|
|
bucket string
|
|
objname string
|
|
isOTmp bool
|
|
procFDName string
|
|
useODirect bool
|
|
size int64
|
|
doChown bool
|
|
uid int
|
|
gid int
|
|
newDirPerm fs.FileMode
|
|
}
|
|
|
|
var (
|
|
// TODO: make this configurable
|
|
defaultFilePerm uint32 = 0644
|
|
)
|
|
|
|
func (p *Posix) openTmpFile(dir, bucket, obj string, size int64, acct auth.Account, dofalloc bool, forceNoTmpFile bool, allowODirect odirectPolicy) (*tmpfile, error) {
|
|
uid, gid, doChown := p.getChownIDs(acct)
|
|
|
|
if forceNoTmpFile {
|
|
return p.openMkTemp(dir, bucket, obj, size, dofalloc, uid, gid, doChown, allowODirect)
|
|
}
|
|
|
|
// O_TMPFILE allows for a file handle to an unnamed file in the filesystem.
|
|
// This can help reduce contention within the namespace (parent directories),
|
|
// etc. And will auto cleanup the inode on close if we never link this
|
|
// file descriptor into the namespace.
|
|
// Not all filesystems support this, so fallback to CreateTemp for when
|
|
// this is not supported.
|
|
openFlags := unix.O_RDWR | unix.O_TMPFILE | unix.O_CLOEXEC
|
|
useODirect := false
|
|
if p.enableODirect && bool(allowODirect) {
|
|
openFlags |= unix.O_DIRECT
|
|
useODirect = true
|
|
}
|
|
|
|
fd, err := unix.Open(dir, openFlags, defaultFilePerm)
|
|
if err != nil {
|
|
if errors.Is(err, syscall.EROFS) {
|
|
return nil, s3err.GetAPIError(s3err.ErrMethodNotAllowed)
|
|
}
|
|
|
|
if p.enableODirect && bool(allowODirect) && isODirectUnsupportedOpenErr(err) {
|
|
warnODirectUnsupportedOnce("openTmpFile", err)
|
|
|
|
fd, err = unix.Open(dir, unix.O_RDWR|unix.O_TMPFILE|unix.O_CLOEXEC, defaultFilePerm)
|
|
if err == nil {
|
|
useODirect = false
|
|
} else if errors.Is(err, syscall.EROFS) {
|
|
return nil, s3err.GetAPIError(s3err.ErrMethodNotAllowed)
|
|
}
|
|
}
|
|
|
|
if err != nil {
|
|
// O_TMPFILE not supported, try fallback
|
|
return p.openMkTemp(dir, bucket, obj, size, dofalloc, uid, gid, doChown, allowODirect)
|
|
}
|
|
}
|
|
|
|
// for O_TMPFILE, filename is /proc/self/fd/<fd> to be used
|
|
// later to link file into namespace
|
|
f := os.NewFile(uintptr(fd), filepath.Join(procfddir, strconv.Itoa(fd)))
|
|
|
|
tmp := &tmpfile{
|
|
f: f,
|
|
bucket: bucket,
|
|
objname: obj,
|
|
isOTmp: true,
|
|
procFDName: strconv.Itoa(fd),
|
|
useODirect: useODirect,
|
|
size: size,
|
|
doChown: doChown,
|
|
uid: uid,
|
|
gid: gid,
|
|
newDirPerm: p.newDirPerm,
|
|
}
|
|
|
|
// falloc is best effort, its fine if this fails
|
|
if size > 0 && dofalloc {
|
|
tmp.falloc()
|
|
}
|
|
|
|
if doChown {
|
|
err := f.Chown(uid, gid)
|
|
if err != nil {
|
|
f.Close()
|
|
return nil, fmt.Errorf("set temp file ownership: %w", err)
|
|
}
|
|
}
|
|
|
|
return tmp, nil
|
|
}
|
|
|
|
func (p *Posix) openMkTemp(dir, bucket, obj string, size int64, dofalloc bool, uid, gid int, doChown bool, allowODirect odirectPolicy) (*tmpfile, error) {
|
|
err := backend.MkdirAll(dir, uid, gid, doChown, p.newDirPerm)
|
|
if err != nil {
|
|
if errors.Is(err, syscall.EROFS) {
|
|
return nil, s3err.GetAPIError(s3err.ErrMethodNotAllowed)
|
|
}
|
|
return nil, fmt.Errorf("make temp dir: %w", err)
|
|
}
|
|
f, err := os.CreateTemp(dir,
|
|
fmt.Sprintf("%x.", sha256.Sum256([]byte(obj))))
|
|
if err != nil {
|
|
if errors.Is(err, syscall.EROFS) {
|
|
return nil, s3err.GetAPIError(s3err.ErrMethodNotAllowed)
|
|
}
|
|
return nil, err
|
|
}
|
|
|
|
useODirect := false
|
|
if p.enableODirect && bool(allowODirect) {
|
|
name := f.Name()
|
|
if err := f.Close(); err != nil {
|
|
os.Remove(name)
|
|
return nil, fmt.Errorf("close temp file before O_DIRECT reopen: %w", err)
|
|
}
|
|
|
|
fd, err := unix.Open(name, unix.O_RDWR|unix.O_CLOEXEC|unix.O_DIRECT, defaultFilePerm)
|
|
if err == nil {
|
|
f = os.NewFile(uintptr(fd), name)
|
|
useODirect = true
|
|
} else if isODirectUnsupportedOpenErr(err) {
|
|
warnODirectUnsupportedOnce("openMkTemp", err)
|
|
f, err = os.OpenFile(name, os.O_RDWR, 0)
|
|
if err != nil {
|
|
os.Remove(name)
|
|
return nil, fmt.Errorf("reopen temp file after O_DIRECT fallback: %w", err)
|
|
}
|
|
} else {
|
|
os.Remove(name)
|
|
return nil, fmt.Errorf("open temp file with O_DIRECT: %w", err)
|
|
}
|
|
}
|
|
|
|
tmp := &tmpfile{
|
|
f: f,
|
|
bucket: bucket,
|
|
objname: obj,
|
|
useODirect: useODirect,
|
|
size: size,
|
|
doChown: doChown,
|
|
uid: uid,
|
|
gid: gid,
|
|
}
|
|
// falloc is best effort, its fine if this fails
|
|
if size > 0 && dofalloc {
|
|
tmp.falloc()
|
|
}
|
|
|
|
if doChown {
|
|
err := f.Chown(uid, gid)
|
|
if err != nil {
|
|
f.Close()
|
|
os.Remove(f.Name())
|
|
return nil, fmt.Errorf("set temp file ownership: %w", err)
|
|
}
|
|
}
|
|
|
|
return tmp, nil
|
|
}
|
|
|
|
func (tmp *tmpfile) falloc() error {
|
|
err := syscall.Fallocate(int(tmp.f.Fd()), 0, 0, tmp.size)
|
|
if err != nil {
|
|
return fmt.Errorf("fallocate: %w", err)
|
|
}
|
|
return nil
|
|
}
|
|
|
|
const (
|
|
maxTmpFileNameRetries = 3
|
|
maxDirRecreateRetries = 3
|
|
initialBackoffMs = 1
|
|
maxBackoffMs = 1024 // ~1 second
|
|
)
|
|
|
|
// linkatOTmpfile links the O_TMPFILE identified by procdir/fdName into dir/basename.
|
|
// Handles EEXIST by linking to a temporary name and atomically renaming it into place.
|
|
func linkatOTmpfile(procdirFd, dirFd int, fdName, basename string) error {
|
|
err := unix.Linkat(procdirFd, fdName, dirFd, basename, unix.AT_SYMLINK_FOLLOW)
|
|
if !errors.Is(err, syscall.EEXIST) {
|
|
return err
|
|
}
|
|
// Linkat cannot overwrite an existing file; link to a temp name then rename atomically.
|
|
for retries := 1; ; retries++ {
|
|
tmpName := fmt.Sprintf(".%s.sgwtmp.%d", basename, time.Now().UnixNano())
|
|
err := unix.Linkat(procdirFd, fdName, dirFd, tmpName, unix.AT_SYMLINK_FOLLOW)
|
|
if errors.Is(err, syscall.EEXIST) && retries < maxTmpFileNameRetries {
|
|
continue
|
|
}
|
|
if err != nil {
|
|
return fmt.Errorf("cannot find free temporary file: %w", err)
|
|
}
|
|
err = unix.Renameat(dirFd, tmpName, dirFd, basename)
|
|
if err != nil {
|
|
// cleanup temp name previously linked into namespace
|
|
_ = unix.Unlinkat(dirFd, tmpName, 0)
|
|
return fmt.Errorf("overwriting renameat failed: %w", err)
|
|
}
|
|
return nil
|
|
}
|
|
}
|
|
|
|
func (tmp *tmpfile) link() error {
|
|
// make sure this is cleaned up in all error cases
|
|
defer tmp.f.Close()
|
|
|
|
// We use Linkat/Rename as the atomic operation for object puts. The
|
|
// upload is written to a temp (or unnamed/O_TMPFILE) file to not conflict
|
|
// with any other simultaneous uploads. The final operation is to move the
|
|
// temp file into place for the object. This ensures the object semantics
|
|
// of last upload completed wins and is not some combination of writes
|
|
// from simultaneous uploads.
|
|
objPath := filepath.Join(tmp.bucket, tmp.objname)
|
|
|
|
dir := filepath.Dir(objPath)
|
|
|
|
err := backend.MkdirAll(dir, tmp.uid, tmp.gid, tmp.doChown, tmp.newDirPerm)
|
|
if err != nil {
|
|
return fmt.Errorf("make parent dir: %w", err)
|
|
}
|
|
|
|
if !tmp.isOTmp {
|
|
// O_TMPFILE not supported, use fallback
|
|
return tmp.fallbackLink()
|
|
}
|
|
|
|
procdir, err := os.Open(procfddir)
|
|
if err != nil {
|
|
return fmt.Errorf("open proc dir: %w", err)
|
|
}
|
|
defer procdir.Close()
|
|
|
|
backoffMs := initialBackoffMs
|
|
var dirf *os.File
|
|
for {
|
|
dirf, err = os.Open(dir)
|
|
if errors.Is(err, fs.ErrNotExist) {
|
|
err := backend.MkdirAll(dir, tmp.uid, tmp.gid, tmp.doChown, tmp.newDirPerm)
|
|
if err != nil {
|
|
return fmt.Errorf("make parent dir: %w", err)
|
|
}
|
|
continue
|
|
}
|
|
if err != nil {
|
|
return fmt.Errorf("open parent dir: %w", err)
|
|
}
|
|
srcFDName := tmp.procFDName
|
|
if srcFDName == "" {
|
|
srcFDName = filepath.Base(tmp.f.Name())
|
|
}
|
|
err = linkatOTmpfile(int(procdir.Fd()), int(dirf.Fd()),
|
|
srcFDName, filepath.Base(objPath))
|
|
dirf.Close()
|
|
if errors.Is(err, syscall.ENOENT) {
|
|
// The directory was removed between open and linkat; backoff and retry.
|
|
// Add jitter to avoid synchronized retry waves.
|
|
sleepWithJitter(backoffMs)
|
|
backoffMs = min((backoffMs * 2), maxBackoffMs)
|
|
|
|
mkErr := backend.MkdirAll(dir, tmp.uid, tmp.gid, tmp.doChown, tmp.newDirPerm)
|
|
if mkErr != nil {
|
|
return fmt.Errorf("make parent dir: %w", mkErr)
|
|
}
|
|
continue
|
|
}
|
|
if err != nil {
|
|
return fmt.Errorf("link tmpfile (fd %q as %q): %w",
|
|
srcFDName, objPath, err)
|
|
}
|
|
break
|
|
}
|
|
|
|
err = tmp.f.Close()
|
|
if err != nil {
|
|
return fmt.Errorf("close tmpfile: %w", err)
|
|
}
|
|
|
|
return nil
|
|
}
|
|
|
|
func (tmp *tmpfile) fallbackLink() error {
|
|
tempname := tmp.f.Name()
|
|
|
|
// reset default file mode because CreateTemp uses 0600
|
|
tmp.f.Chmod(fs.FileMode(defaultFilePerm))
|
|
|
|
err := tmp.f.Close()
|
|
if err != nil {
|
|
return fmt.Errorf("close tmpfile: %w", err)
|
|
}
|
|
|
|
objPath := filepath.Join(tmp.bucket, tmp.objname)
|
|
dir := filepath.Dir(objPath)
|
|
err = os.Rename(tempname, objPath)
|
|
if errors.Is(err, syscall.ENOENT) {
|
|
// The parent directory was concurrently removed; backoff and retry.
|
|
backoffMs := initialBackoffMs
|
|
for range maxDirRecreateRetries {
|
|
// Add jitter to avoid synchronized retry waves.
|
|
sleepWithJitter(backoffMs)
|
|
backoffMs = min((backoffMs * 2), maxBackoffMs)
|
|
|
|
err = backend.MkdirAll(dir, tmp.uid, tmp.gid, tmp.doChown, tmp.newDirPerm)
|
|
if err != nil {
|
|
return fmt.Errorf("recreate parent dir: %w", err)
|
|
}
|
|
err = os.Rename(tempname, objPath)
|
|
if !errors.Is(err, syscall.ENOENT) {
|
|
break
|
|
}
|
|
}
|
|
}
|
|
if err != nil {
|
|
// rename only works for files within the same filesystem
|
|
// if this fails fallback to copy
|
|
backoffMs := initialBackoffMs
|
|
for range maxDirRecreateRetries {
|
|
err = backend.MoveFile(tempname, objPath, fs.FileMode(defaultFilePerm))
|
|
if !errors.Is(err, syscall.ENOENT) {
|
|
break
|
|
}
|
|
|
|
// Add jitter to avoid synchronized retry waves.
|
|
sleepWithJitter(backoffMs)
|
|
backoffMs = min((backoffMs * 2), maxBackoffMs)
|
|
|
|
// The parent directory was concurrently removed; recreate and retry.
|
|
mkErr := backend.MkdirAll(dir, tmp.uid, tmp.gid, tmp.doChown, tmp.newDirPerm)
|
|
if mkErr != nil {
|
|
return fmt.Errorf("recreate parent dir: %w", mkErr)
|
|
}
|
|
}
|
|
return err
|
|
}
|
|
|
|
return nil
|
|
}
|
|
|
|
func (tmp *tmpfile) cleanup() {
|
|
tmp.f.Close()
|
|
if !strings.HasPrefix(tmp.f.Name(), procfddir) {
|
|
os.Remove(tmp.f.Name())
|
|
}
|
|
}
|