mirror of
https://tangled.org/evan.jarrett.net/at-container-registry
synced 2026-09-30 05:55:34 +00:00
Each in-flight blob upload buffers up to 16MB, Docker pushes five layers at once per client, and nothing bounded the total. Writers also lived in the package-level map forever: a client that died mid-push left its writer, its buffer, and any hold-side S3 multipart session behind with no expiry. A process-wide budget (golang.org/x/sync semaphore, default 512MB, server.upload_buffer_budget_mb) now caps memory held in upload buffers. A writer charges its buffer's projected backing capacity before growing, so a config blob costs kilobytes and a full writer costs exactly one buffer, and releases once, on Commit, Cancel, or reap. A write that needs budget waits on the request's context with a five minute cap, outside the writer's lock so Cancel and the sweeper cannot queue behind it; that wait is backpressure on the client. The budget is clamped to at least one buffer so a single upload can never deadlock. A sweeper started with the other appview workers reaps writers idle past server.upload_idle_timeout (default 1h), aborting the hold-side multipart on a detached context and releasing the budget. It measures inactivity, not age, so a slow push is never reaped, and it skips a writer whose lock is held so it cannot race a live part upload. Write also gains a fix the budget made visible. It appended a whole chunk and checked afterwards, so the last chunk before a flush could land a few bytes past 16MB, which did not fit the backing array; bytes.Buffer doubled it to 32MB and Reset kept that for the rest of the upload. Only chunk sizes that tile 16MB exactly avoided it, and the network read loop promises no such thing. Every large layer could hold 32MB while the budget charged 16. Write now fills to exactly the threshold, flushes, and continues with the remainder, so capacity is pinned at 16MB for any chunk size, every part is exactly one buffer, and a single oversized Write streams through as parts instead of buffering whole. The test streams 24KB chunks across the boundary and fails against the old code with cap 33554432. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_018Yf1ZVA7sXYhQNb9tCo1m5
200 lines
7.1 KiB
Go
200 lines
7.1 KiB
Go
package storage
|
|
|
|
import (
|
|
"context"
|
|
"log/slog"
|
|
"sync/atomic"
|
|
"time"
|
|
|
|
"golang.org/x/sync/semaphore"
|
|
)
|
|
|
|
const (
|
|
// defaultUploadBufferBudget is the process-wide ceiling on bytes held in
|
|
// blob upload buffers. Every in-flight push holds a buffer of up to
|
|
// maxBufferSize, Docker pushes up to five layers at once per client, and
|
|
// nothing used to stop N clients from multiplying that out until the
|
|
// AppView was killed by the OOM reaper. 512MB is 32 full buffers, which is
|
|
// far more concurrency than a single AppView instance sees in practice
|
|
// while still fitting comfortably in the smallest deployment.
|
|
defaultUploadBufferBudget = 512 * 1024 * 1024 // 512MB
|
|
|
|
// defaultUploadIdleTimeout is how long a writer may go without a Write or
|
|
// a Commit before the sweeper treats it as abandoned. Generous on purpose:
|
|
// the signal is inactivity, not age, so a slow push that is still making
|
|
// progress is never reaped no matter how long it runs.
|
|
defaultUploadIdleTimeout = time.Hour
|
|
|
|
// uploadSweepInterval is how often abandoned writers are looked for. Not
|
|
// configurable: the meaningful knob is the timeout, and sweeping a map of
|
|
// a few dozen entries every five minutes costs nothing.
|
|
uploadSweepInterval = 5 * time.Minute
|
|
|
|
// uploadAbortTimeout bounds the hold-side abort issued for a reaped
|
|
// writer. It runs on a detached context, so it needs its own deadline.
|
|
uploadAbortTimeout = 30 * time.Second
|
|
|
|
// uploadBudgetWait bounds how long a single Write waits for budget when
|
|
// the request it belongs to has no deadline of its own. Blocking is the
|
|
// point (it is backpressure on the Docker client), but blocking forever is
|
|
// not: a writer that cannot get budget in this long fails its write and
|
|
// the client retries.
|
|
uploadBudgetWait = 5 * time.Minute
|
|
)
|
|
|
|
// bufferBudget is the process-wide allowance for bytes held in upload buffers.
|
|
//
|
|
// semaphore.Weighted does the waiting, and the held counter exists only so the
|
|
// sweeper and the tests can report what is outstanding: the semaphore itself
|
|
// does not expose its current weight.
|
|
type bufferBudget struct {
|
|
sem *semaphore.Weighted
|
|
limit int64
|
|
held atomic.Int64
|
|
}
|
|
|
|
func newBufferBudget(limit int64) *bufferBudget {
|
|
return &bufferBudget{sem: semaphore.NewWeighted(limit), limit: limit}
|
|
}
|
|
|
|
// acquire blocks until n bytes of budget are available, the context is done, or
|
|
// the wait cap expires. n is always at most maxBufferSize per step, and the
|
|
// budget is never smaller than maxBufferSize, so a single writer can always
|
|
// eventually be satisfied.
|
|
func (b *bufferBudget) acquire(ctx context.Context, n int64) error {
|
|
if n <= 0 {
|
|
return nil
|
|
}
|
|
waitCtx, cancel := context.WithTimeout(ctx, uploadBudgetWait)
|
|
defer cancel()
|
|
|
|
if err := b.sem.Acquire(waitCtx, n); err != nil {
|
|
return err
|
|
}
|
|
b.held.Add(n)
|
|
return nil
|
|
}
|
|
|
|
// release returns n bytes to the budget. Callers must release exactly what they
|
|
// acquired and no more; ProxyBlobWriter does that by tracking a single charged
|
|
// figure and zeroing it as it releases.
|
|
func (b *bufferBudget) release(n int64) {
|
|
if n <= 0 {
|
|
return
|
|
}
|
|
b.sem.Release(n)
|
|
b.held.Add(-n)
|
|
}
|
|
|
|
// uploadBudget is package level for the same reason globalUploads is: a
|
|
// ProxyBlobStore is built fresh on every registry request, so anything that has
|
|
// to be shared across requests (and across the uploads that outlive a single
|
|
// request) cannot hang off the instance. Threading it through RegistryContext
|
|
// would have given each request its own view of a limit that is only meaningful
|
|
// process-wide, and would have made every writer's release depend on which
|
|
// request happened to construct it.
|
|
//
|
|
// ConfigureUploads replaces it once at startup, before the listener is up.
|
|
var (
|
|
uploadBudget = newBufferBudget(defaultUploadBufferBudget)
|
|
uploadIdleTimeout = defaultUploadIdleTimeout
|
|
)
|
|
|
|
// ConfigureUploads applies the configured upload buffer budget and idle
|
|
// timeout. Call it once during startup, before serving: it replaces the
|
|
// semaphore outright rather than resizing it, which is only safe while nothing
|
|
// holds budget.
|
|
//
|
|
// A budget below maxBufferSize is raised to it. Anything less could never be
|
|
// acquired by even a single writer, so it would not throttle uploads, it would
|
|
// stall every one of them until the wait cap expired.
|
|
func ConfigureUploads(budgetBytes int64, idleTimeout time.Duration) {
|
|
if budgetBytes < maxBufferSize {
|
|
slog.Warn("Upload buffer budget below one buffer, raising to the minimum",
|
|
"component", "proxy_blob_store", "configured", budgetBytes, "minimum", maxBufferSize)
|
|
budgetBytes = maxBufferSize
|
|
}
|
|
if idleTimeout <= 0 {
|
|
idleTimeout = defaultUploadIdleTimeout
|
|
}
|
|
|
|
uploadBudget = newBufferBudget(budgetBytes)
|
|
uploadIdleTimeout = idleTimeout
|
|
|
|
slog.Info("Upload buffer budget configured",
|
|
"component", "proxy_blob_store", "budget_bytes", budgetBytes, "idle_timeout", idleTimeout)
|
|
}
|
|
|
|
// UploadStats reports what the in-flight uploads are currently holding: how
|
|
// many writers are tracked, and how much buffer budget they hold between them.
|
|
func UploadStats() (inFlight int, bytesHeld int64) {
|
|
globalUploadsMu.RLock()
|
|
inFlight = len(globalUploads)
|
|
globalUploadsMu.RUnlock()
|
|
return inFlight, uploadBudget.held.Load()
|
|
}
|
|
|
|
// StartUploadSweeper runs the abandoned-upload sweep until ctx is cancelled.
|
|
//
|
|
// Not a leased worker. globalUploads is per-process memory, so every instance
|
|
// must sweep its own map; electing one instance to do it would leave the others
|
|
// leaking exactly as they do today.
|
|
func StartUploadSweeper(ctx context.Context) {
|
|
go func() {
|
|
ticker := time.NewTicker(uploadSweepInterval)
|
|
defer ticker.Stop()
|
|
|
|
for {
|
|
select {
|
|
case <-ctx.Done():
|
|
return
|
|
case <-ticker.C:
|
|
sweepAbandonedUploads(time.Now())
|
|
}
|
|
}
|
|
}()
|
|
}
|
|
|
|
// sweepAbandonedUploads cancels every writer that has been idle longer than the
|
|
// configured timeout, and returns how many it reaped.
|
|
//
|
|
// A Docker client that dies mid-push leaves its writer in globalUploads with
|
|
// nothing to remove it: only Commit and Cancel do that, and neither is ever
|
|
// called. The writer's buffer, its budget, and the hold-side S3 multipart
|
|
// session it may have opened all stay put for the life of the process.
|
|
func sweepAbandonedUploads(now time.Time) int {
|
|
// Snapshot under the map lock and reap outside it. Reaping takes the
|
|
// writer's own lock (and issues a network call), and Commit takes the map
|
|
// lock while holding the writer's lock, so holding both here in the other
|
|
// order would deadlock.
|
|
globalUploadsMu.RLock()
|
|
candidates := make([]*ProxyBlobWriter, 0, len(globalUploads))
|
|
for _, w := range globalUploads {
|
|
candidates = append(candidates, w)
|
|
}
|
|
globalUploadsMu.RUnlock()
|
|
|
|
reaped := 0
|
|
for _, w := range candidates {
|
|
idle, ok := w.reapIfIdle(now, uploadIdleTimeout)
|
|
if !ok {
|
|
continue
|
|
}
|
|
reaped++
|
|
|
|
globalUploadsMu.Lock()
|
|
delete(globalUploads, w.id)
|
|
globalUploadsMu.Unlock()
|
|
|
|
slog.Info("Reaped abandoned upload",
|
|
"component", "proxy_blob_store/sweep", "id", w.id, "idle", idle, "size", w.Size())
|
|
}
|
|
|
|
if reaped > 0 {
|
|
inFlight, held := UploadStats()
|
|
slog.Info("Abandoned upload sweep finished",
|
|
"component", "proxy_blob_store/sweep", "reaped", reaped, "in_flight", inFlight, "bytes_held", held)
|
|
}
|
|
return reaped
|
|
}
|