remotes: support multiple base_urls with round-robin + failover
ci/woodpecker/pr/build Pipeline was successful
ci/woodpecker/pr/test Pipeline was successful
ci/woodpecker/pr/pre-commit Pipeline was successful

A remote's base_url may now be a single string OR a list of upstream
mirrors. When it is a list the shared proxy engine load-balances across
them round-robin and, on an upstream error/timeout/5xx, fails over to the
next mirror before returning an error. Because the selection happens in
the engine (not per provider), it applies to every remote package type.
Backward compatible: a bare-string base_url behaves exactly as before.

- add models.StringOrSlice (string-or-array JSON) and custom Remote
  (Un)MarshalJSON: base_url populates BaseURLs (full list) + BaseURL
  (active/first); marshals a single mirror back to a bare string
- add Remote.BaseURLList / ValidateBaseURLs; validate list is non-empty
  and every entry is an http/https URL in the v2 create/update handlers
- persist the full list in a new base_urls TEXT[] column (additive
  migration), keeping base_url in sync for old readers; only write
  base_urls for genuinely multi-mirror remotes
- engine: per-remote round-robin cursor + attempt ordering; wrap the
  fetch/head/revalidate upstream calls in a failover loop that narrows
  the remote to one selected mirror per attempt; only network errors and
  5xx fail over (404/403/... are returned as-is); circuit breaker stays
  keyed per remote and trips only after all mirrors fail
- add unit tests (JSON round-trip, engine round-robin/failover/single-URL,
  DB multi-URL round-trip) and a docker acceptance suite: round-robin
  distribution across two mock upstreams, failover past a dead primary,
  single-base_url regression, and a real dnf makecache+install through a
  two-mirror rpm remote whose primary is dead

Least-connections and a per-remote strategy selector are a follow-up PR.
This commit is contained in:
2026-08-13 08:26:17 +10:00
parent 73c0bfc670
commit 68a1f14e17
22 changed files with 921 additions and 19 deletions
+118
View File
@@ -11,6 +11,8 @@ import (
"log/slog"
"net/http"
"strings"
"sync"
"sync/atomic"
"time"
"git.unkin.net/unkin/artifactapi/internal/cache"
@@ -35,6 +37,10 @@ type Engine struct {
cas *storage.CAS
circuit *CircuitBreaker
accessLog chan database.AccessLogEntry
// rrCounters holds a per-remote round-robin cursor (remoteName ->
// *atomic.Uint64) used to rotate the starting mirror across upstream base
// URLs. Distribution is per-replica and approximate, which is fine.
rrCounters sync.Map
}
func NewEngine(db *database.DB, c *cache.Redis, s *storage.S3) *Engine {
@@ -222,7 +228,30 @@ func (e *Engine) Head(ctx context.Context, remote models.Remote, path string, pr
return e.headUpstream(ctx, remote, path, prov)
}
// headUpstream issues an upstream HEAD, load-balancing across the remote's base
// URLs and failing over to the next mirror on a network error or 5xx.
func (e *Engine) headUpstream(ctx context.Context, remote models.Remote, path string, prov provider.Provider) (*HeadResult, error) {
order := e.baseURLAttemptOrder(remote)
if len(order) == 0 {
return nil, &ProxyError{Status: http.StatusBadGateway, Message: "no upstream base_url configured"}
}
var lastErr error
for i, url := range order {
result, err := e.headUpstreamOnce(ctx, withBaseURL(remote, url), path, prov)
if err == nil {
return result, nil
}
lastErr = err
if i < len(order)-1 && shouldFailover(err) {
slog.Warn("upstream HEAD failed, failing over", "remote", remote.Name, "base_url", url, "error", err)
continue
}
return nil, err
}
return nil, lastErr
}
func (e *Engine) headUpstreamOnce(ctx context.Context, remote models.Remote, path string, prov provider.Provider) (*HeadResult, error) {
url := prov.UpstreamURL(remote, path)
authHeaders, err := prov.AuthHeaders(ctx, remote)
@@ -277,7 +306,31 @@ func (e *Engine) headUpstream(ctx context.Context, remote models.Remote, path st
return &HeadResult{ContentType: contentType, Size: resp.ContentLength, Source: "remote"}, nil
}
// fetchFromUpstream fetches an artifact from upstream, load-balancing across the
// remote's base URLs and failing over to the next mirror on a network error or
// 5xx before returning an error.
func (e *Engine) fetchFromUpstream(ctx context.Context, remote models.Remote, path string, prov provider.Provider, class Classification, ttl time.Duration, clientHeaders http.Header) (*FetchResult, error) {
order := e.baseURLAttemptOrder(remote)
if len(order) == 0 {
return nil, &ProxyError{Status: http.StatusBadGateway, Message: "no upstream base_url configured"}
}
var lastErr error
for i, url := range order {
result, err := e.fetchFromUpstreamOnce(ctx, withBaseURL(remote, url), path, prov, class, ttl, clientHeaders)
if err == nil {
return result, nil
}
lastErr = err
if i < len(order)-1 && shouldFailover(err) {
slog.Warn("upstream fetch failed, failing over", "remote", remote.Name, "base_url", url, "error", err)
continue
}
return nil, err
}
return nil, lastErr
}
func (e *Engine) fetchFromUpstreamOnce(ctx context.Context, remote models.Remote, path string, prov provider.Provider, class Classification, ttl time.Duration, clientHeaders http.Header) (*FetchResult, error) {
url := prov.UpstreamURL(remote, path)
authHeaders, err := prov.AuthHeaders(ctx, remote)
@@ -454,7 +507,31 @@ func (e *Engine) serveFromStore(ctx context.Context, remote models.Remote, path
}, nil
}
// checkUpstream issues a conditional upstream HEAD (If-None-Match), load
// balancing across the remote's base URLs and failing over to the next mirror on
// a network error or 5xx.
func (e *Engine) checkUpstream(ctx context.Context, remote models.Remote, path, etag string, prov provider.Provider) (bool, error) {
order := e.baseURLAttemptOrder(remote)
if len(order) == 0 {
return false, &ProxyError{Status: http.StatusBadGateway, Message: "no upstream base_url configured"}
}
var lastErr error
for i, url := range order {
notModified, err := e.checkUpstreamOnce(ctx, withBaseURL(remote, url), path, etag, prov)
if err == nil {
return notModified, nil
}
lastErr = err
if i < len(order)-1 && shouldFailover(err) {
slog.Warn("upstream revalidation failed, failing over", "remote", remote.Name, "base_url", url, "error", err)
continue
}
return false, err
}
return false, lastErr
}
func (e *Engine) checkUpstreamOnce(ctx context.Context, remote models.Remote, path, etag string, prov provider.Provider) (bool, error) {
url := prov.UpstreamURL(remote, path)
req, err := http.NewRequestWithContext(ctx, http.MethodHead, url, nil)
@@ -649,3 +726,44 @@ func isNetworkError(err error) bool {
var ue *UpstreamError
return errors.As(err, &ue)
}
// baseURLAttemptOrder returns the ordered upstream base URLs to try for a single
// request. A multi-mirror remote starts at the next round-robin position and
// advances linearly for failover; a single-URL remote yields exactly that one
// URL, preserving the original one-attempt behavior.
func (e *Engine) baseURLAttemptOrder(remote models.Remote) []string {
urls := remote.BaseURLList()
if len(urls) <= 1 {
return urls
}
v, _ := e.rrCounters.LoadOrStore(remote.Name, new(atomic.Uint64))
start := int(v.(*atomic.Uint64).Add(1) - 1)
ordered := make([]string, len(urls))
for i := range urls {
ordered[i] = urls[(start+i)%len(urls)]
}
return ordered
}
// withBaseURL narrows a remote's active BaseURL to a single selected mirror so
// providers (UpstreamURL/AuthHeaders/RewriteResponse) operate on exactly that
// upstream for this attempt.
func withBaseURL(remote models.Remote, url string) models.Remote {
remote.BaseURL = url
remote.BaseURLs = []string{url}
return remote
}
// shouldFailover reports whether an upstream attempt error is worth retrying
// against the next mirror: network errors/timeouts and upstream 5xx responses.
// Definitive statuses (404/403/401/...) are returned to the caller unchanged.
func shouldFailover(err error) bool {
if isNetworkError(err) {
return true
}
var pe *ProxyError
if errors.As(err, &pe) {
return pe.Status >= 500
}
return false
}