Files
seaweedfs/weed/storage/blockvol/testrunner/infra/fault.go
T
Ping QiuandClaude Opus 4.6 3557ae283f feat: Phase 10 CP10-3 -- NVMe/TCP Tier 1 optimizations, WAL admission control, benchmark platform
CP10-3 Tier 1 optimizations (T1-T4):
- TCP_NODELAY + 256KB socket buffers on NVMe/TCP connections
- Response batching: all C2H data chunks + CapsuleResp in single flush
- Tiered buffer pool (4KB/64KB/256KB sync.Pool) for write payloads
- Configurable MaxH2CDataLength wiring through controller/IC/chunking

BUG-CP103-1: NVMe write retry with jittered backoff for transient WAL pressure
- writeWithRetry() with bounded backoff [50/200/800ms]
- throttleOnWALPressure() pre-write delay above 90% WAL usage
- WALPressureProvider interface + NVMeAdapter.WALPressure()

BUG-CP103-2: Volume-level WAL admission control
- WALAdmission with counting semaphore (max concurrent writers)
- Soft watermark (0.7): small delay to desynchronize herd
- Hard watermark (0.9): block until flusher drains
- Single-deadline budget shared across watermark wait + semaphore
- Close-aware during both watermark and semaphore waits
- Wired into BlockVol.WriteLBA() and Trim()

Benchmark platform enhancements:
- NVMe benchmark actions and scenarios (A/B, CW sweep, IOQ sweep)
- Database benchmark actions (SQLite, pgbench)
- K8s operator QA reconciler tests
- New testrunner scenarios for HA, fault injection, CSI lifecycle

Test counts: 213 NVMe + 625 engine + operator + testrunner tests, all passing.

Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
2026-03-09 17:44:01 -07:00

139 lines
5.1 KiB
Go

package infra
import (
"context"
"fmt"
"strings"
"time"
)
// InjectNetem adds a netem delay on the node's outbound traffic to targetIP.
// Returns the cleanup command string (for storage in vars).
func InjectNetem(ctx context.Context, node *Node, targetIP string, delayMs int) (cleanupCmd string, err error) {
iface, _, code, err := node.RunRoot(ctx, fmt.Sprintf(
"ip route get %s | head -1 | awk '{for(i=1;i<=NF;i++) if($i==\"dev\") print $(i+1)}'", targetIP))
iface = strings.TrimSpace(iface)
if err != nil || code != 0 || iface == "" {
return "", fmt.Errorf("find interface for %s: iface=%q code=%d err=%v", targetIP, iface, code, err)
}
_, stderr, code, err := node.RunRoot(ctx, fmt.Sprintf(
"tc qdisc add dev %s root netem delay %dms", iface, delayMs))
if err != nil || code != 0 {
return "", fmt.Errorf("tc qdisc add: code=%d stderr=%s err=%v", code, stderr, err)
}
cleanupCmd = fmt.Sprintf("tc qdisc del dev %s root 2>/dev/null || true", iface)
return cleanupCmd, nil
}
// InjectIptablesDrop blocks outbound TCP traffic from node to targetIP on the given ports.
// Returns the cleanup command string.
func InjectIptablesDrop(ctx context.Context, node *Node, targetIP string, ports []int) (cleanupCmd string, err error) {
for i, port := range ports {
_, stderr, code, err := node.RunRoot(ctx, fmt.Sprintf(
"iptables -A OUTPUT -d %s -p tcp --dport %d -j DROP", targetIP, port))
if err != nil || code != 0 {
// Rollback already-added rules
for j := 0; j < i; j++ {
node.RunRoot(ctx, fmt.Sprintf(
"iptables -D OUTPUT -d %s -p tcp --dport %d -j DROP 2>/dev/null", targetIP, ports[j]))
}
return "", fmt.Errorf("iptables add port %d: code=%d stderr=%s err=%v", port, code, stderr, err)
}
}
// Build cleanup command that removes all rules.
// Use "|| true" so each removal succeeds even if the rule is already gone,
// and ";" so all ports are attempted even if one fails.
var cmds []string
for _, port := range ports {
cmds = append(cmds, fmt.Sprintf(
"iptables -D OUTPUT -d %s -p tcp --dport %d -j DROP 2>/dev/null || true", targetIP, port))
}
cleanupCmd = strings.Join(cmds, " ; ")
return cleanupCmd, nil
}
// FillDisk fills the filesystem at dir, leaving ~4MB free.
// Returns the cleanup command string.
func FillDisk(ctx context.Context, node *Node, dir string) (cleanupCmd string, err error) {
stdout, _, code, err := node.RunRoot(ctx, fmt.Sprintf(
"df -BM --output=avail %s | tail -1 | tr -d ' M'", dir))
if err != nil || code != 0 {
return "", fmt.Errorf("df: code=%d err=%v", code, err)
}
availMB := 0
fmt.Sscanf(strings.TrimSpace(stdout), "%d", &availMB)
if availMB < 8 {
return "", fmt.Errorf("not enough space to fill: %dMB available", availMB)
}
fillMB := availMB - 4
// Use fallocate (instant) instead of dd (linear time).
_, stderr, code, err := node.RunRoot(ctx, fmt.Sprintf(
"fallocate -l %dM %s/fillfile", fillMB, dir))
if err != nil || code != 0 {
// Fallback to dd if fallocate not available.
_, stderr, code, err = node.RunRoot(ctx, fmt.Sprintf(
"dd if=/dev/zero of=%s/fillfile bs=1M count=%d 2>/dev/null", dir, fillMB))
if err != nil || code != 0 {
stdout2, _, _, _ := node.RunRoot(ctx, fmt.Sprintf("test -f %s/fillfile && echo ok", dir))
if !strings.Contains(stdout2, "ok") {
return "", fmt.Errorf("fillDisk: code=%d stderr=%s err=%v", code, stderr, err)
}
}
}
cleanupCmd = fmt.Sprintf("rm -f %s/fillfile", dir)
return cleanupCmd, nil
}
// CorruptWALRegion overwrites nBytes within the WAL section of the volume file.
func CorruptWALRegion(ctx context.Context, node *Node, volPath string, nBytes int) error {
const walOffset = 4096 // SuperblockSize
stdout, _, code, err := node.RunRoot(ctx, fmt.Sprintf("stat -c %%s %s", volPath))
if err != nil || code != 0 {
return fmt.Errorf("stat %s: code=%d err=%v", volPath, code, err)
}
fileSize := 0
fmt.Sscanf(strings.TrimSpace(stdout), "%d", &fileSize)
walEnd := walOffset + 64*1024*1024
if walEnd > fileSize {
walEnd = fileSize
}
walUsable := walEnd - walOffset
if walUsable < nBytes*2 {
return fmt.Errorf("WAL region too small: %d", walUsable)
}
seekPos := walOffset + walUsable/3
_, stderr, code, err := node.RunRoot(ctx, fmt.Sprintf(
"python3 -c \"import sys; sys.stdout.buffer.write(b'\\xff'*%d)\" | dd of=%s bs=1 seek=%d conv=notrunc 2>/dev/null",
nBytes, volPath, seekPos))
if err != nil || code != 0 {
return fmt.Errorf("corrupt WAL region: code=%d stderr=%s err=%v", code, stderr, err)
}
return nil
}
// ClearFault executes a cleanup command stored in vars.
// Tolerates non-zero exit codes since cleanup commands are often
// idempotent (e.g. removing an already-removed iptables rule).
func ClearFault(ctx context.Context, node *Node, cleanupCmd string) error {
if cleanupCmd == "" {
return nil
}
cctx, cancel := context.WithTimeout(ctx, 10*time.Second)
defer cancel()
_, stderr, code, err := node.RunRoot(cctx, cleanupCmd)
if err != nil {
return fmt.Errorf("clear fault: code=%d stderr=%s err=%v", code, stderr, err)
}
// Non-zero exit is tolerated — cleanup commands use "|| true" but
// legacy cleanup strings might not, and double-cleanup is harmless.
return nil
}