Retry failed GitHub release scans with backoff (#133)
ci/woodpecker/tag/docker Pipeline was successful

Releasing a GitHub sync lease always advanced `last_synced_at`, even after a failed scan (e.g. a rate-limit 403). A remote that failed once waited a full `mutable_ttl` before retrying, and a cold remote kept returning 503 until then.

- record scan outcomes in one shared lease helper for github_rpm/deb/alpine
- keep `last_synced_at` and the ETag on failure; retry from 60s with exponential backoff, capped at min(10m, ttl/4)
- honour `Retry-After` / `X-RateLimit-Reset`, clamped to `mutable_ttl`
- add `sync_failures` / `next_retry_at` columns (migration 0002)

Reviewed-on: #133
Co-authored-by: unkin-agent <unkin-agent@unkin.net>
Co-committed-by: unkin-agent <unkin-agent@unkin.net>
This commit was merged in pull request #133.
This commit is contained in:
2026-10-09 23:15:55 +11:00
committed by BenVincent
parent 1780b3d77c
commit 492607a164
18 changed files with 477 additions and 134 deletions
+45 -17
View File
@@ -7,6 +7,8 @@ import (
"github.com/jackc/pgx/v5"
"git.unkin.net/unkin/artifactapi/internal/provider"
"git.unkin.net/unkin/artifactapi/pkg/models"
)
@@ -30,22 +32,35 @@ func (db *DB) ListGitHubRPMRemotes(ctx context.Context) ([]models.Remote, error)
return remotes, rows.Err()
}
// ClaimGitHubSyncLease atomically claims the per-remote sync lease. It succeeds
// (claimed=true) only when the remote is due — never synced, or synced longer
// than freshness ago — and no live lease is held by another replica. This bounds
// total GitHub load to roughly one scan per freshness window regardless of how
// many replicas poll. The returned etag is the stored releases-list ETag, shared
// across replicas so a conditional request can short-circuit an unchanged repo.
// A zero freshness (used for prime scans) ignores the recency gate and claims
// whenever no live lease is held.
// ClaimGitHubSyncLease atomically claims the per-remote github_rpm sync lease.
// See claimSyncLease.
func (db *DB) ClaimGitHubSyncLease(ctx context.Context, remoteName, owner string, freshness, lease time.Duration) (bool, string, error) {
return db.claimSyncLease(ctx, "github_rpm_sync_state", remoteName, owner, freshness, lease)
}
// ReleaseGitHubSyncLease records a github_rpm scan outcome and frees the lease.
// See releaseSyncLease.
func (db *DB) ReleaseGitHubSyncLease(ctx context.Context, remoteName, owner string, res provider.SyncResult) error {
return db.releaseSyncLease(ctx, "github_rpm_sync_state", remoteName, owner, res)
}
// claimSyncLease atomically claims a per-remote sync lease in table. It succeeds
// (claimed=true) only when the remote is due and no live lease is held by
// another replica. Due means: a pending retry after a failed scan has reached
// next_retry_at, or, with no retry pending, the remote was never synced or was
// synced longer than freshness ago. A zero freshness (prime scans) ignores the
// recency gate but still honours a pending retry's backoff. The returned etag is
// the stored releases-list ETag, shared across replicas so a conditional request
// can short-circuit an unchanged repo.
func (db *DB) claimSyncLease(ctx context.Context, table, remoteName, owner string, freshness, lease time.Duration) (bool, string, error) {
row := db.Pool.QueryRow(ctx, `
INSERT INTO github_rpm_sync_state AS s (remote_name, sync_lease_owner, sync_lease_expires)
INSERT INTO `+table+` AS s (remote_name, sync_lease_owner, sync_lease_expires)
VALUES ($1, $2, now() + make_interval(secs => $4))
ON CONFLICT (remote_name) DO UPDATE
SET sync_lease_owner = $2,
sync_lease_expires = now() + make_interval(secs => $4)
WHERE (s.last_synced_at IS NULL OR s.last_synced_at < now() - make_interval(secs => $3))
WHERE (CASE WHEN s.next_retry_at IS NOT NULL THEN s.next_retry_at <= now()
ELSE s.last_synced_at IS NULL OR s.last_synced_at < now() - make_interval(secs => $3) END)
AND (s.sync_lease_expires IS NULL OR s.sync_lease_expires < now())
RETURNING s.etag
`, remoteName, owner, freshness.Seconds(), lease.Seconds())
@@ -60,14 +75,27 @@ func (db *DB) ClaimGitHubSyncLease(ctx context.Context, remoteName, owner string
return true, etag, nil
}
// ReleaseGitHubSyncLease records the completed scan and frees the lease. Only the
// owning replica may release; last_synced_at advances so the next poll waits a
// full freshness window, and etag is persisted for the next conditional request.
func (db *DB) ReleaseGitHubSyncLease(ctx context.Context, remoteName, owner, etag string, syncedAt time.Time) error {
// releaseSyncLease records a scan outcome and frees the lease; only the owning
// replica may release. Success advances last_synced_at, persists the etag and
// clears any retry. Failure leaves last_synced_at and etag untouched (the last
// good metadata keeps serving) and schedules next_retry_at with exponential
// backoff from the consecutive-failure count, never before res.RetryAt.
func (db *DB) releaseSyncLease(ctx context.Context, table, remoteName, owner string, res provider.SyncResult) error {
var retryAt *time.Time
if !res.RetryAt.IsZero() {
retryAt = &res.RetryAt
}
_, err := db.Pool.Exec(ctx, `
UPDATE github_rpm_sync_state
SET last_synced_at = $3, etag = $4, sync_lease_owner = '', sync_lease_expires = NULL
UPDATE `+table+`
SET last_synced_at = CASE WHEN $3 THEN last_synced_at ELSE now() END,
etag = CASE WHEN $3 THEN etag ELSE $4 END,
sync_failures = CASE WHEN $3 THEN sync_failures + 1 ELSE 0 END,
next_retry_at = CASE WHEN $3 THEN GREATEST(
now() + make_interval(secs => LEAST($5 * power(2, LEAST(sync_failures, 20)), $6)),
$7::timestamptz)
END,
sync_lease_owner = '', sync_lease_expires = NULL
WHERE remote_name = $1 AND sync_lease_owner = $2
`, remoteName, owner, syncedAt, etag)
`, remoteName, owner, res.Failed, res.Etag, res.Backoff.Seconds(), res.MaxBackoff.Seconds(), retryAt)
return err
}