Support adopting existing radosgw buckets and users
ci/woodpecker/pr/build Pipeline was successful
ci/woodpecker/pr/pre-commit Pipeline was successful
ci/woodpecker/pr/test Pipeline was successful

The operator previously assumed it created every user and bucket it managed:
reconciling an existing resource could overwrite its user attributes or wipe its
bucket policy, and deleting a CRD always deleted the underlying RGW object (only
Bucket had retainOnDelete). That made taking over pre-existing radosgw state
unsafe. Make adoption first-class.

- add retainOnDelete to ObjectStoreUser and BucketAccess (dedicated users), so
  deleting the CRD orphans the RGW user instead of deleting it (symmetric with
  Bucket)
- merge bucket policy instead of replacing it: the operator marks its own
  statements with a cephrgwop* Sid and preserves any statement it does not own,
  so adopting a bucket with a hand-written policy keeps it; add Bucket
  managePolicy (default true) to opt out of policy management entirely
- only reconcile user attributes the spec sets: DisplayName when non-empty and
  Suspended is now an optional *bool, so adopting a user does not reset them
- record adoption: ObjectStoreUser/Bucket status.adopted (+ printcolumn) is true
  when the RGW object already existed on first reconcile
- add GetBucketPolicy + MergeBucketPolicy; keyed adoption detection off the
  status identity field so a Pending owner wait does not mislabel it
- regenerate CRDs/deepcopy; add docs/adoption.md and
  config/samples/05-adoption.yaml; cover the merge in policy_test.go

Claude-Session: https://claude.ai/code/session_016CEncETbf8cvy1PhsHfFHM
This commit is contained in:
2026-07-25 00:15:10 +10:00
parent 619aa6751d
commit 54d3e38223
18 changed files with 578 additions and 50 deletions
+39 -12
View File
@@ -77,7 +77,11 @@ func (r *BucketReconciler) Reconcile(ctx context.Context, req ctrl.Request) (ctr
}
ownerUID := owner.Status.UID
// Ensure the bucket exists.
// Ensure the bucket exists. Record adoption once: whether the RGW bucket
// already existed the first time we reconciled this resource. status.BucketID
// is only set on a successful reconcile, so a Pending wait on the owner (or a
// transient failure) does not pollute the signal.
firstObserve := b.Status.BucketID == ""
info, err := r.Ceph.GetBucket(ctx, bucketName)
if ceph.IsNotFound(err) {
createSpec := ceph.CreateBucketSpec{
@@ -97,8 +101,14 @@ func (r *BucketReconciler) Reconcile(ctx context.Context, req ctrl.Request) (ctr
return r.fail(ctx, &b, "CreateFailed", err)
}
logger.Info("created bucket", "bucket", bucketName, "owner", ownerUID)
if firstObserve {
b.Status.Adopted = false
}
} else if err != nil {
return r.fail(ctx, &b, "LookupFailed", err)
} else if firstObserve {
b.Status.Adopted = true
logger.Info("adopted existing bucket", "bucket", bucketName, "owner", ownerUID)
}
bucketID := info.InstanceID()
@@ -129,17 +139,28 @@ func (r *BucketReconciler) Reconcile(ctx context.Context, req ctrl.Request) (ctr
}
}
// Render and apply the aggregate S3 policy from all BucketAccess grants.
grants, principals, err := r.collectGrants(ctx, b.Namespace, b.Name)
if err != nil {
return r.fail(ctx, &b, "GrantsFailed", err)
}
policy, err := ceph.BuildBucketPolicy(bucketName, grants)
if err != nil {
return r.fail(ctx, &b, "PolicyBuildFailed", err)
}
if err := r.Ceph.SetBucketPolicy(ctx, bucketName, bucketID, ownerUID, policy); err != nil {
return r.fail(ctx, &b, "PolicyFailed", err)
// Render and apply the aggregate S3 policy from all BucketAccess grants,
// unless the bucket opts out of policy management. The merge preserves any
// statements the operator does not own, so an adopted bucket keeps its
// existing policy.
principals := 0
if managePolicy(&b) {
grants, p, err := r.collectGrants(ctx, b.Namespace, b.Name)
if err != nil {
return r.fail(ctx, &b, "GrantsFailed", err)
}
existing, err := r.Ceph.GetBucketPolicy(ctx, bucketName, ownerUID)
if err != nil {
return r.fail(ctx, &b, "PolicyReadFailed", err)
}
policy, err := ceph.MergeBucketPolicy(existing, bucketName, grants)
if err != nil {
return r.fail(ctx, &b, "PolicyBuildFailed", err)
}
if err := r.Ceph.SetBucketPolicy(ctx, bucketName, bucketID, ownerUID, policy); err != nil {
return r.fail(ctx, &b, "PolicyFailed", err)
}
principals = p
}
b.Status.Phase = "Ready"
@@ -223,6 +244,12 @@ func grantKey(g ceph.Grant) string {
return string(b)
}
// managePolicy reports whether the operator should reconcile this bucket's S3
// policy. A nil ManagePolicy (the CRD default) is treated as true.
func managePolicy(b *v1alpha1.Bucket) bool {
return b.Spec.ManagePolicy == nil || *b.Spec.ManagePolicy
}
func (r *BucketReconciler) pending(ctx context.Context, b *v1alpha1.Bucket, reason, msg string) (ctrl.Result, error) {
b.Status.Phase = "Pending"
b.Status.ObservedGeneration = b.Generation
@@ -49,8 +49,9 @@ func (r *BucketAccessReconciler) Reconcile(ctx context.Context, req ctrl.Request
if !ba.DeletionTimestamp.IsZero() {
if controllerutil.ContainsFinalizer(&ba, finalizer) {
// Only delete a user the operator created for this grant.
if managed && uid != "" {
// Only delete a user the operator created for this grant, and only
// when the grant does not ask to retain it.
if managed && uid != "" && !ba.Spec.RetainOnDelete {
if err := r.Ceph.DeleteUser(ctx, uid); err != nil {
return r.fail(ctx, &ba, "DeleteFailed", err)
}
@@ -40,7 +40,9 @@ func (r *ObjectStoreUserReconciler) Reconcile(ctx context.Context, req ctrl.Requ
if !osu.DeletionTimestamp.IsZero() {
if controllerutil.ContainsFinalizer(&osu, finalizer) {
if err := r.Ceph.DeleteUser(ctx, uid); err != nil {
if osu.Spec.RetainOnDelete {
logger.Info("retaining RGW user on delete", "uid", uid)
} else if err := r.Ceph.DeleteUser(ctx, uid); err != nil {
return r.fail(ctx, &osu, "DeleteFailed", err)
}
controllerutil.RemoveFinalizer(&osu, finalizer)
@@ -65,17 +67,31 @@ func (r *ObjectStoreUserReconciler) Reconcile(ctx context.Context, req ctrl.Requ
Suspended: osu.Spec.Suspended,
}
if _, err := r.Ceph.GetUser(ctx, uid); ceph.IsNotFound(err) {
// Record adoption once: whether the RGW user already existed the first time
// we reconciled this resource (taken over rather than created). status.UID is
// only set on a successful reconcile, so it is a clean "never provisioned"
// signal that transient failures do not pollute.
firstObserve := osu.Status.UID == ""
_, getErr := r.Ceph.GetUser(ctx, uid)
switch {
case ceph.IsNotFound(getErr):
if _, err := r.Ceph.CreateUser(ctx, spec); err != nil {
return r.fail(ctx, &osu, "CreateFailed", err)
}
logger.Info("created RGW user", "uid", uid)
} else if err != nil {
return r.fail(ctx, &osu, "LookupFailed", err)
} else {
if firstObserve {
osu.Status.Adopted = false
}
case getErr != nil:
return r.fail(ctx, &osu, "LookupFailed", getErr)
default:
if _, err := r.Ceph.UpdateUser(ctx, spec); err != nil {
return r.fail(ctx, &osu, "UpdateFailed", err)
}
if firstObserve {
osu.Status.Adopted = true
logger.Info("adopted existing RGW user", "uid", uid)
}
}
if q := osu.Spec.Quota; q != nil {