remotes: add least-connections mirror strategy (round-robin remains default) (#122)
ci/woodpecker/tag/docker Pipeline was successful
ci/woodpecker/tag/docker Pipeline was successful
## Why The mirrorlist (PR #121) always load-balances round-robin. Round-robin is oblivious to how busy each mirror is, so a slow or saturated mirror keeps getting its fair share of new requests. This adds an opt-in **least-connections** strategy that favors the mirror currently handling the fewest in-flight requests, steering new work toward idle mirrors. Round-robin stays the default, so existing remotes are unchanged. ## How - **Model**: `models.Remote` gains `mirror_strategy` (`round_robin` default/empty, or `least_conn`). `ValidateMirrorStrategy` checks the enum and requires a non-empty mirrorlist for `least_conn`. Empty behaves as `round_robin` for back-compat. - **DB**: additive `mirror_strategy TEXT NOT NULL DEFAULT 'round_robin'` column (CREATE TABLE + `ADD COLUMN IF NOT EXISTS`), wired through remoteCols/scanRemote/CreateRemote/UpdateRemote; empty normalized to `round_robin` on write. - **Engine**: a per-remote, per-pool-URL atomic in-flight gauge is incremented around each upstream call (head/fetch/checkUpstream). For a `least_conn` remote the attempt order starts with the least-loaded pool URL (ties broken by the existing round-robin rotation). Round-robin path and failover order are unchanged; single-URL pools are a no-op. - **Tests**: unit tests for least-loaded selection, round-robin default, gauge inc/dec, single-URL no-op, and validation; DB round-trip covers the new column; docker e2e adds a `least_conn` distribution test plus a real `dnf` install through a `least_conn` remote. Back-compat: unset/empty `mirror_strategy` is `round_robin`, so all existing remotes keep their current behavior. Reviewed-on: #122 Co-authored-by: unkin-agent <unkin-agent@unkin.net> Co-committed-by: unkin-agent <unkin-agent@unkin.net>
This commit was merged in pull request #122.
This commit is contained in:
@@ -10,6 +10,7 @@ import (
|
||||
"io"
|
||||
"log/slog"
|
||||
"net/http"
|
||||
"sort"
|
||||
"strings"
|
||||
"sync"
|
||||
"sync/atomic"
|
||||
@@ -41,6 +42,11 @@ type Engine struct {
|
||||
// *atomic.Uint64) used to rotate the starting mirror across upstream base
|
||||
// URLs. Distribution is per-replica and approximate, which is fine.
|
||||
rrCounters sync.Map
|
||||
// inflight holds a per-remote, per-upstream-URL in-flight request gauge
|
||||
// (key "remoteName\x00baseURL" -> *atomic.Int64) used by the least_conn
|
||||
// mirror strategy to prefer the mirror currently handling the fewest
|
||||
// requests. Per-replica and approximate, which is fine.
|
||||
inflight sync.Map
|
||||
}
|
||||
|
||||
func NewEngine(db *database.DB, c *cache.Redis, s *storage.S3) *Engine {
|
||||
@@ -237,7 +243,9 @@ func (e *Engine) headUpstream(ctx context.Context, remote models.Remote, path st
|
||||
}
|
||||
var lastErr error
|
||||
for i, url := range order {
|
||||
ctr := e.beginAttempt(remote, url)
|
||||
result, err := e.headUpstreamOnce(ctx, withBaseURL(remote, url), path, prov)
|
||||
endAttempt(ctr)
|
||||
if err == nil {
|
||||
return result, nil
|
||||
}
|
||||
@@ -316,7 +324,9 @@ func (e *Engine) fetchFromUpstream(ctx context.Context, remote models.Remote, pa
|
||||
}
|
||||
var lastErr error
|
||||
for i, url := range order {
|
||||
ctr := e.beginAttempt(remote, url)
|
||||
result, err := e.fetchFromUpstreamOnce(ctx, withBaseURL(remote, url), path, prov, class, ttl, clientHeaders)
|
||||
endAttempt(ctr)
|
||||
if err == nil {
|
||||
return result, nil
|
||||
}
|
||||
@@ -517,7 +527,9 @@ func (e *Engine) checkUpstream(ctx context.Context, remote models.Remote, path,
|
||||
}
|
||||
var lastErr error
|
||||
for i, url := range order {
|
||||
ctr := e.beginAttempt(remote, url)
|
||||
notModified, err := e.checkUpstreamOnce(ctx, withBaseURL(remote, url), path, etag, prov)
|
||||
endAttempt(ctr)
|
||||
if err == nil {
|
||||
return notModified, nil
|
||||
}
|
||||
@@ -729,23 +741,61 @@ func isNetworkError(err error) bool {
|
||||
|
||||
// baseURLAttemptOrder returns the ordered upstream base URLs to try for a single
|
||||
// request, drawn from the remote's pool ([base_url] + mirrorlist). A multi-mirror
|
||||
// remote starts at the next round-robin position and advances linearly for
|
||||
// remote starts at a strategy-chosen position and advances linearly for
|
||||
// failover; a remote with no mirrorlist yields exactly [base_url], preserving the
|
||||
// original single-attempt behavior.
|
||||
// original single-attempt behavior. The default (round_robin) rotates the
|
||||
// starting mirror; least_conn starts with the mirror handling the fewest
|
||||
// in-flight requests. Failover order after the first pick is unchanged.
|
||||
func (e *Engine) baseURLAttemptOrder(remote models.Remote) []string {
|
||||
urls := remote.UpstreamPool()
|
||||
if len(urls) <= 1 {
|
||||
return urls
|
||||
}
|
||||
// Rotate by the round-robin cursor first so equal-load mirrors still spread
|
||||
// evenly; least_conn then stable-sorts this rotation by in-flight count.
|
||||
v, _ := e.rrCounters.LoadOrStore(remote.Name, new(atomic.Uint64))
|
||||
start := int(v.(*atomic.Uint64).Add(1) - 1)
|
||||
ordered := make([]string, len(urls))
|
||||
for i := range urls {
|
||||
ordered[i] = urls[(start+i)%len(urls)]
|
||||
}
|
||||
if remote.MirrorStrategy == models.MirrorStrategyLeastConn {
|
||||
sort.SliceStable(ordered, func(a, b int) bool {
|
||||
return e.inflightCounter(remote.Name, ordered[a]).Load() < e.inflightCounter(remote.Name, ordered[b]).Load()
|
||||
})
|
||||
}
|
||||
return ordered
|
||||
}
|
||||
|
||||
// inflightCounter returns the shared in-flight request gauge for a given
|
||||
// (remote, upstream URL), creating it on first use.
|
||||
func (e *Engine) inflightCounter(remoteName, url string) *atomic.Int64 {
|
||||
v, _ := e.inflight.LoadOrStore(remoteName+"\x00"+url, new(atomic.Int64))
|
||||
return v.(*atomic.Int64)
|
||||
}
|
||||
|
||||
// beginAttempt increments the in-flight gauge for a least_conn multi-mirror
|
||||
// remote before an upstream call and returns the counter to release; it is a
|
||||
// no-op (returns nil) for round-robin remotes and single-URL pools.
|
||||
func (e *Engine) beginAttempt(remote models.Remote, url string) *atomic.Int64 {
|
||||
if remote.MirrorStrategy != models.MirrorStrategyLeastConn {
|
||||
return nil
|
||||
}
|
||||
if len(remote.UpstreamPool()) <= 1 {
|
||||
return nil
|
||||
}
|
||||
ctr := e.inflightCounter(remote.Name, url)
|
||||
ctr.Add(1)
|
||||
return ctr
|
||||
}
|
||||
|
||||
// endAttempt decrements a gauge returned by beginAttempt, tolerating nil.
|
||||
func endAttempt(ctr *atomic.Int64) {
|
||||
if ctr != nil {
|
||||
ctr.Add(-1)
|
||||
}
|
||||
}
|
||||
|
||||
// withBaseURL narrows a remote's active BaseURL to a single selected mirror so
|
||||
// providers (UpstreamURL/AuthHeaders/RewriteResponse) operate on exactly that
|
||||
// upstream for this attempt.
|
||||
|
||||
Reference in New Issue
Block a user