Files
seaweedfs/weed/storage/needle_map_metric.go
T
b9ad62fc16 [Volume] Keep DAT and index state consistent after async batch Sync failure (#11425)
* fix 11400

* persist failed-recovery quarantine and harden rollback

- record the unavailable state in a .unavailable marker, fsync it, and
  re-arm it on load so a restart cannot serve an unverified pair
- quarantine the volume so heartbeats stop advertising it
- block MarkVolumeWritable while unavailable, rechecked under noWriteLock
- fail every request of a failed batch, not only the succeeded ones
- restore the needle map and truncate .dat on inline fsync rollback failure
- add truncateIndex for the sorted-file needle map
- mirror the fail-closed semantics in the Rust volume server

* volume: erase rolled-back mappings instead of leaving tombstones

A rolled-back batch or failed inline write used Delete() to undo a
needle that did not exist beforehand, leaving a tombstoned map entry
whose stale offset makes the next write to that needle fail reading a
header that no longer exists. Add removeMapping/restoreMapping to the
mappers so recovery erases entries that were absent before the batch
and reinstates the exact prior offset/size for ones that were,
including tombstones. The index row still goes through Delete so a
replay forgets the needle.

* volume: gate bulk readers on unavailable and fsync the marker's dir

- fsync_dir(&self.dir) synced the volume dir's parent, not the dir
  holding .unavailable; pass the marker path so the create survives
  a host crash
- export UnavailableError and check it in ReadAllNeedles,
  VolumeTailSender, VolumeIncrementalCopy, and IncrementalBackup so
  replica-sync paths cannot stream or append data from an unverified
  .dat/.idx pair; mirror on the Rust side via read_dat_slice,
  read_all_needles, dat_scan_plan, and the incremental-copy handler

* volume: drop issue references from comments near touched code

* volume: stop active scans when the volume becomes unavailable

The stream entry-point checks ran once per RPC, so a volume quarantined
by a failed recovery mid-scan kept serving data. Recheck availability
per needle/chunk on the detached read paths: tail scan and heartbeat,
read-all, incremental copy, incremental backup writes, and the Rust
StreamingBody chunk reads. Rust incremental copy also rejects a
quarantined volume before sync_to_disk touches the backend.

---------

Co-authored-by: Chris Lu <chris.lu@gmail.com>
2026-09-24 06:57:44 +08:00

232 lines
6.5 KiB
Go

package storage
import (
"fmt"
"io"
"os"
"sync/atomic"
"github.com/seaweedfs/seaweedfs/weed/storage/idx"
"github.com/seaweedfs/seaweedfs/weed/storage/needle"
. "github.com/seaweedfs/seaweedfs/weed/storage/types"
boom "github.com/tylertreat/BoomFilters"
)
type mapMetric struct {
DeletionCounter uint32 `json:"DeletionCounter"`
FileCounter uint32 `json:"FileCounter"`
DeletionByteCounter uint64 `json:"DeletionByteCounter"`
FileByteCounter uint64 `json:"FileByteCounter"`
MaximumFileKey uint64 `json:"MaxFileKey"`
// MaximumNeedleEnd is the largest (offset.ToActualOffset() +
// GetActualSize(size, version)) seen during the index walk. It is used
// at volume load to verify that no .idx entry references bytes past
// the end of the .dat — the deeper-than-tail corruption shape from
// issue #8928 — without paying for a second linear scan of the index.
MaximumNeedleEnd int64 `json:"MaxNeedleEnd"`
}
type batchMapMetricSnapshot struct {
deletionCounter uint32
fileCounter uint32
deletionByteCounter uint64
fileByteCounter uint64
maximumFileKey uint64
maximumNeedleEnd int64
}
func (mm *mapMetric) snapshotBatchMetrics() batchMapMetricSnapshot {
return batchMapMetricSnapshot{
deletionCounter: atomic.LoadUint32(&mm.DeletionCounter),
fileCounter: atomic.LoadUint32(&mm.FileCounter),
deletionByteCounter: atomic.LoadUint64(&mm.DeletionByteCounter),
fileByteCounter: atomic.LoadUint64(&mm.FileByteCounter),
maximumFileKey: atomic.LoadUint64(&mm.MaximumFileKey),
maximumNeedleEnd: atomic.LoadInt64(&mm.MaximumNeedleEnd),
}
}
func (mm *mapMetric) restoreBatchMetrics(snapshot batchMapMetricSnapshot) {
atomic.StoreUint32(&mm.DeletionCounter, snapshot.deletionCounter)
atomic.StoreUint32(&mm.FileCounter, snapshot.fileCounter)
atomic.StoreUint64(&mm.DeletionByteCounter, snapshot.deletionByteCounter)
atomic.StoreUint64(&mm.FileByteCounter, snapshot.fileByteCounter)
atomic.StoreUint64(&mm.MaximumFileKey, snapshot.maximumFileKey)
atomic.StoreInt64(&mm.MaximumNeedleEnd, snapshot.maximumNeedleEnd)
}
func (mm *mapMetric) logDelete(deletedByteCount Size) {
if mm == nil {
return
}
mm.LogDeletionCounter(deletedByteCount)
}
func (mm *mapMetric) logPut(key NeedleId, oldSize Size, newSize Size) {
if mm == nil {
return
}
mm.MaybeSetMaxFileKey(key)
mm.LogFileCounter(newSize)
if oldSize > 0 && oldSize.IsValid() {
mm.LogDeletionCounter(oldSize)
}
}
func (mm *mapMetric) LogFileCounter(newSize Size) {
if mm == nil {
return
}
atomic.AddUint32(&mm.FileCounter, 1)
atomic.AddUint64(&mm.FileByteCounter, uint64(newSize))
}
func (mm *mapMetric) LogDeletionCounter(oldSize Size) {
if mm == nil {
return
}
if oldSize > 0 {
atomic.AddUint32(&mm.DeletionCounter, 1)
atomic.AddUint64(&mm.DeletionByteCounter, uint64(oldSize))
}
}
func (mm *mapMetric) ContentSize() uint64 {
if mm == nil {
return 0
}
return atomic.LoadUint64(&mm.FileByteCounter)
}
func (mm *mapMetric) DeletedSize() uint64 {
if mm == nil {
return 0
}
return atomic.LoadUint64(&mm.DeletionByteCounter)
}
func (mm *mapMetric) FileCount() int {
if mm == nil {
return 0
}
return int(atomic.LoadUint32(&mm.FileCounter))
}
func (mm *mapMetric) DeletedCount() int {
if mm == nil {
return 0
}
return int(atomic.LoadUint32(&mm.DeletionCounter))
}
func (mm *mapMetric) MaxFileKey() NeedleId {
if mm == nil {
return 0
}
t := uint64(mm.MaximumFileKey)
return Uint64ToNeedleId(t)
}
func (mm *mapMetric) MaybeSetMaxFileKey(key NeedleId) {
if mm == nil {
return
}
if key > mm.MaxFileKey() {
atomic.StoreUint64(&mm.MaximumFileKey, uint64(key))
}
}
// MaybeSetMaxNeedleEnd updates MaximumNeedleEnd if the supplied entry's
// (offset + actual size) is larger than what we have seen so far. Skips
// deleted/zero-offset entries because they don't reserve space in .dat.
func (mm *mapMetric) MaybeSetMaxNeedleEnd(offset Offset, size Size, version needle.Version) {
if mm == nil || offset.IsZero() || !size.IsValid() {
return
}
end := offset.ToActualOffset() + needle.GetActualSize(size, version)
if end > atomic.LoadInt64(&mm.MaximumNeedleEnd) {
atomic.StoreInt64(&mm.MaximumNeedleEnd, end)
}
}
func (mm *mapMetric) MaxNeedleEnd() int64 {
if mm == nil {
return 0
}
return atomic.LoadInt64(&mm.MaximumNeedleEnd)
}
func needleMapMetricFromIndexFile(r *os.File, mm *mapMetric, version needle.Version) error {
var bf *boom.BloomFilter
buf := make([]byte, NeedleIdSize)
err := reverseWalkIndexFile(r, func(entryCount int64) {
bf = boom.NewBloomFilter(uint(entryCount), 0.001)
}, func(key NeedleId, offset Offset, size Size) error {
mm.MaybeSetMaxFileKey(key)
mm.MaybeSetMaxNeedleEnd(offset, size, version)
NeedleIdToBytes(buf, key)
if size.IsValid() {
mm.FileByteCounter += uint64(size)
}
mm.FileCounter++
if !bf.TestAndAdd(buf) {
// if !size.IsValid(), then this file is deleted already
if !size.IsValid() {
mm.DeletionCounter++
}
} else {
// deleted file
mm.DeletionCounter++
if size.IsValid() {
// previously already deleted file
mm.DeletionByteCounter += uint64(size)
}
}
return nil
})
return err
}
func newNeedleMapMetricFromIndexFile(r *os.File, version needle.Version) (mm *mapMetric, err error) {
mm = &mapMetric{}
err = needleMapMetricFromIndexFile(r, mm, version)
return
}
func reverseWalkIndexFile(r *os.File, initFn func(entryCount int64), fn func(key NeedleId, offset Offset, size Size) error) error {
fi, err := r.Stat()
if err != nil {
return fmt.Errorf("file %s stat error: %v", r.Name(), err)
}
fileSize := fi.Size()
if fileSize%NeedleMapEntrySize != 0 {
return fmt.Errorf("unexpected file %s size: %d", r.Name(), fileSize)
}
entryCount := fileSize / NeedleMapEntrySize
initFn(entryCount)
batchSize := int64(1024 * 4)
bytes := make([]byte, NeedleMapEntrySize*batchSize)
nextBatchSize := entryCount % batchSize
if nextBatchSize == 0 {
nextBatchSize = batchSize
}
remainingCount := entryCount - nextBatchSize
for remainingCount >= 0 {
n, e := r.ReadAt(bytes[:NeedleMapEntrySize*nextBatchSize], NeedleMapEntrySize*remainingCount)
// glog.V(0).Infoln("file", r.Name(), "readerOffset", NeedleMapEntrySize*remainingCount, "count", count, "e", e)
if e == io.EOF && n == int(NeedleMapEntrySize*nextBatchSize) {
e = nil
}
if e != nil {
return e
}
for i := int(nextBatchSize) - 1; i >= 0; i-- {
key, offset, size := idx.IdxFileEntry(bytes[i*NeedleMapEntrySize : i*NeedleMapEntrySize+NeedleMapEntrySize])
if e = fn(key, offset, size); e != nil {
return e
}
}
nextBatchSize = batchSize
remainingCount -= nextBatchSize
}
return nil
}