Make intra-cluster NOTIFY loop-free (TSIG-keyed, no pod IPs in restart config)

v0.2.5 (PR #14) added an options-scope allow-notify enumerating the primary
pod IP on secondaries. Options-scope config feeds the config-hash annotation
that rolls the StatefulSet, so any config change rolled the pods, the primary
came back on a new pod IP, the operator re-rendered with the new IP, the hash
changed, the pods rolled again — an infinite roll loop across every
BindCluster. The prod deployment was reverted to v0.2.4.

Replace the pod-IP allow-notify with TSIG-authenticated NOTIFY:

- Secondaries render `allow-notify { key "<name>"; };` — a static key element
  with NO IPs. It depends only on the key name, so pod-IP churn can never
  change the render, the config-hash, or trigger a restart.
- The primary signs its outgoing NOTIFYs: the zone-scope also-notify entries
  (already enumerating replica pod IPs, applied via rndc addzone/modzone with
  NO restart) now carry `key "<name>"`.
- Key choice: reuse the cluster's catalog transfer TSIG key (TransferKeyRef).
  Secondaries already present it for AXFR and it is in keys.conf on every pod,
  so no new key plumbing is needed.

Add a permanent regression guard for the loop class:
- controller: reconcile the ConfigMap with the primary pod on two different
  IPs and assert the config-hash is byte-identical.
- render: render restart-scoped input and assert no pod IP appears in
  allow-notify; RenderInput no longer has any pod-IP field.

Zone-scope also-notify (rndc, no restart) legitimately still lists pod IPs;
only restart-scoped config must be pod-IP-independent.
This commit is contained in:
2026-07-25 23:22:45 +10:00
parent 671c43b05b
commit 61324ae89a
7 changed files with 295 additions and 70 deletions
+10 -8
View File
@@ -176,14 +176,6 @@ func (r *BindClusterReconciler) reconcileConfigMap(ctx context.Context, c *bindv
// across primary pod restarts (falls back to the pod IP when no primary
// Service exists; the Pod/Service watches re-render when it changes).
in := bind.RenderInput{Cluster: c, PrimaryAddress: primaryTransferAddress(ctx, r.Client, c)}
// Secondaries transfer from the stable primary Service ClusterIP, but the
// primary pod's NOTIFYs are sourced from its pod IP, which BIND refuses as
// "non-primary" unless it appears in allow-notify. Render the primary pod IP
// so intra-cluster NOTIFYs are accepted immediately (the Pod watch re-renders
// the ConfigMap when the pod IP changes across restarts).
if ip := primaryPodIP(ctx, r.Client, c); ip != "" {
in.PrimaryPodAddresses = []string{ip}
}
var acls bindv1alpha1.BindACLList
if err := r.List(ctx, &acls, client.InNamespace(c.Namespace)); err == nil {
@@ -239,6 +231,16 @@ func (r *BindClusterReconciler) reconcileConfigMap(ctx context.Context, c *bindv
}
}
// Secondaries accept intra-cluster NOTIFYs signed with the cluster's catalog
// transfer TSIG key (the primary signs its also-notify NOTIFYs with it — see
// bindzone_controller). Render `allow-notify { key "<name>"; }` — a static key
// element, NO pod IPs — so it never changes on pod churn and cannot re-render
// the restart-scoped config (the v0.2.5 roll loop). Resolve the catalog's
// TransferKeyRef to the BIND key name (KeyName override, else the ref).
if in.Catalog != nil && in.Catalog.Spec.TransferKeyRef != "" {
in.NotifyKeyName = tsigKeyName(ctx, r.Client, c.Namespace, in.Catalog.Spec.TransferKeyRef)
}
primaryConf, secondaryConf := bind.RenderNamedConf(in)
data := map[string]string{
"named.conf.primary": primaryConf,
+17 -5
View File
@@ -88,7 +88,13 @@ func (r *BindZoneReconciler) Reconcile(ctx context.Context, req ctrl.Request) (c
notifyTargets = secondaryPodIPs(ctx, r.Client, cluster)
}
zoneConfig, err := r.buildZoneConfig(ctx, &zone, r.zoneTransferKeyRef(ctx, &zone, cluster), notifyTargets)
// The catalog transfer TSIG key doubles as the intra-cluster NOTIFY key: the
// primary signs its also-notify NOTIFYs with it and secondaries accept them
// via `allow-notify { key "<key>"; }`. Keying the NOTIFYs is what lets the
// secondary's allow-notify be a static key element (no pod IPs), so pod-IP
// churn never re-renders restart-scoped config (the v0.2.5 roll loop).
transferKey := r.zoneTransferKeyRef(ctx, &zone, cluster)
zoneConfig, err := r.buildZoneConfig(ctx, &zone, transferKey, notifyTargets, transferKey)
if err != nil {
return r.setPhase(ctx, &zone, "Error", "ConfigError", err.Error())
}
@@ -142,8 +148,10 @@ func (r *BindZoneReconciler) Reconcile(ctx context.Context, req ctrl.Request) (c
// buildZoneConfig renders the inner clause passed to rndc addzone/modzone.
// transferKey, when set, is the catalog transfer TSIG key name; catalog member
// primary zones must allow AXFR with it so secondaries can pull them.
func (r *BindZoneReconciler) buildZoneConfig(ctx context.Context, zone *bindv1alpha1.BindZone, transferKey string, notifyTargets []string) (string, error) {
// primary zones must allow AXFR with it so secondaries can pull them. notifyKey,
// when set, is the TSIG key each also-notify entry is signed with, so
// secondaries can accept the NOTIFYs by key rather than by (churning) pod IP.
func (r *BindZoneReconciler) buildZoneConfig(ctx context.Context, zone *bindv1alpha1.BindZone, transferKey string, notifyTargets []string, notifyKey string) (string, error) {
zType := zone.Spec.Type
if zType == "" {
zType = bindv1alpha1.ZonePrimary
@@ -164,9 +172,13 @@ func (r *BindZoneReconciler) buildZoneConfig(ctx context.Context, zone *bindv1al
}
// NOTIFY only the secondaries we know about (their apex NS is the primary
// itself, so default `notify yes` would reach no one). `notify explicit`
// keeps NOTIFY off the query-serving VIP and scoped to the pod IPs.
// keeps NOTIFY off the query-serving VIP and scoped to the pod IPs. Each
// entry is signed with notifyKey so secondaries can admit the NOTIFY by key
// (`allow-notify { key ... }`) instead of by pod IP. This zone config is
// applied via rndc addzone/modzone — no pod restart — so listing pod IPs
// here is safe; only *restart-scoped* config must never depend on pod IPs.
if len(notifyTargets) > 0 {
parts = append(parts, "notify explicit", fmt.Sprintf("also-notify { %s }", terminateInline(notifyTargets)))
parts = append(parts, "notify explicit", fmt.Sprintf("also-notify { %s }", alsoNotifyList(notifyTargets, notifyKey)))
}
if zone.Spec.DNSSECPolicyRef != "" {
parts = append(parts, fmt.Sprintf("dnssec-policy \"%s\"", zone.Spec.DNSSECPolicyRef), "inline-signing yes")
+21 -4
View File
@@ -19,15 +19,32 @@ func TestBuildZoneConfigPrimaryAlsoNotify(t *testing.T) {
},
}
cfg, err := r.buildZoneConfig(context.Background(), zone, "transfer-key", []string{"10.42.2.6", "10.42.1.5"})
cfg, err := r.buildZoneConfig(context.Background(), zone, "transfer-key", []string{"10.42.2.6", "10.42.1.5"}, "externaldns-key")
if err != nil {
t.Fatalf("buildZoneConfig: %v", err)
}
if !strings.Contains(cfg, "notify explicit") {
t.Errorf("expected notify explicit in %q", cfg)
}
if !strings.Contains(cfg, "also-notify { 10.42.2.6; 10.42.1.5; }") {
t.Errorf("expected also-notify with the secondary IPs in %q", cfg)
// also-notify entries are keyed so secondaries can admit the NOTIFY by TSIG
// key (allow-notify { key ... }) instead of by (churning) pod IP.
if !strings.Contains(cfg, `also-notify { 10.42.2.6 key "externaldns-key"; 10.42.1.5 key "externaldns-key"; }`) {
t.Errorf("expected keyed also-notify with the secondary IPs in %q", cfg)
}
}
// also-notify entries carry no key when none is configured (unkeyed NOTIFY).
func TestBuildZoneConfigPrimaryAlsoNotifyUnkeyed(t *testing.T) {
r := &BindZoneReconciler{}
zone := &bindv1alpha1.BindZone{
Spec: bindv1alpha1.BindZoneSpec{ZoneName: "main.unkin.net", Type: bindv1alpha1.ZonePrimary},
}
cfg, err := r.buildZoneConfig(context.Background(), zone, "transfer-key", []string{"10.42.2.6"}, "")
if err != nil {
t.Fatalf("buildZoneConfig: %v", err)
}
if !strings.Contains(cfg, "also-notify { 10.42.2.6; }") {
t.Errorf("expected unkeyed also-notify in %q", cfg)
}
}
@@ -38,7 +55,7 @@ func TestBuildZoneConfigPrimaryNoNotifyTargets(t *testing.T) {
Spec: bindv1alpha1.BindZoneSpec{ZoneName: "main.unkin.net", Type: bindv1alpha1.ZonePrimary},
}
cfg, err := r.buildZoneConfig(context.Background(), zone, "transfer-key", nil)
cfg, err := r.buildZoneConfig(context.Background(), zone, "transfer-key", nil, "externaldns-key")
if err != nil {
t.Fatalf("buildZoneConfig: %v", err)
}
@@ -0,0 +1,108 @@
package controller
import (
"context"
"strings"
"testing"
corev1 "k8s.io/api/core/v1"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"k8s.io/apimachinery/pkg/runtime"
clientgoscheme "k8s.io/client-go/kubernetes/scheme"
"sigs.k8s.io/controller-runtime/pkg/client"
"sigs.k8s.io/controller-runtime/pkg/client/fake"
bindv1alpha1 "git.unkin.net/unkin/bind-operator/api/v1alpha1"
)
// TestConfigHashIndependentOfPrimaryPodIP is the permanent regression guard for
// the v0.2.5 rolling-restart loop.
//
// v0.2.5 rendered an options-scope `allow-notify { <primaryPodIP>; ... }` into
// the (restart-scoped) named.conf. The config-hash annotation that rolls the
// StatefulSet is computed over that render (BindClusterReconciler.configHash), so
// a config change rolled the pods, the primary pod came back on a NEW IP, the
// operator re-rendered with the new IP, the hash changed, the pods rolled again,
// and so on — an infinite roll loop across every BindCluster.
//
// The fix removed all pod IPs from restart-scoped config (secondaries now admit
// intra-cluster NOTIFYs by TSIG key: `allow-notify { key "X"; }`). This test
// reconciles the ConfigMap with the primary pod on one IP, computes the hash,
// then does it again with the primary pod on a DIFFERENT IP, and asserts the
// hash is byte-identical. If any pod-IP dependency ever creeps back into
// restart-scoped config, this fails and the loop-class bug is caught.
func TestConfigHashIndependentOfPrimaryPodIP(t *testing.T) {
scheme := runtime.NewScheme()
if err := clientgoscheme.AddToScheme(scheme); err != nil {
t.Fatal(err)
}
if err := bindv1alpha1.AddToScheme(scheme); err != nil {
t.Fatal(err)
}
const ns = "dns"
svcIP := "10.43.5.5" // stable primary Service ClusterIP (does NOT churn)
cluster := &bindv1alpha1.BindCluster{
ObjectMeta: metav1.ObjectMeta{Name: "auth", Namespace: ns},
Spec: bindv1alpha1.BindClusterSpec{
Mode: bindv1alpha1.ModeAuthoritative,
Replicas: 3,
PrimaryService: &bindv1alpha1.ClusterServiceSpec{},
},
}
primarySvc := &corev1.Service{
ObjectMeta: metav1.ObjectMeta{Name: primaryServiceName(cluster.Name), Namespace: ns},
Spec: corev1.ServiceSpec{ClusterIP: svcIP},
}
catalog := &bindv1alpha1.BindCatalogZone{
ObjectMeta: metav1.ObjectMeta{Name: "cat", Namespace: ns},
Spec: bindv1alpha1.BindCatalogZoneSpec{
ClusterRef: cluster.Name,
ZoneName: "catalog.internal",
TransferKeyRef: "externaldns-key",
},
}
tsig := &bindv1alpha1.BindTSIGKey{
ObjectMeta: metav1.ObjectMeta{Name: "externaldns-key", Namespace: ns},
Spec: bindv1alpha1.BindTSIGKeySpec{ClusterRef: cluster.Name},
}
// hashWithPrimaryIP reconciles the ConfigMap with the primary pod carrying the
// given IP, then returns the resulting config-hash.
hashWithPrimaryIP := func(ip string) string {
primaryPod := &corev1.Pod{
ObjectMeta: metav1.ObjectMeta{Name: primaryPodName(cluster.Name), Namespace: ns, Labels: commonLabels(cluster.Name)},
Status: corev1.PodStatus{PodIP: ip},
}
c := fake.NewClientBuilder().
WithScheme(scheme).
WithObjects(cluster, primarySvc, catalog, tsig, primaryPod).
Build()
r := &BindClusterReconciler{Client: c, Scheme: scheme}
ctx := context.Background()
if err := r.reconcileConfigMap(ctx, cluster); err != nil {
t.Fatalf("reconcileConfigMap: %v", err)
}
// Sanity: the pod IP must not have leaked into the rendered config.
var cm corev1.ConfigMap
if err := c.Get(ctx, client.ObjectKey{Namespace: ns, Name: configMapName(cluster.Name)}, &cm); err != nil {
t.Fatalf("get configmap: %v", err)
}
for k, v := range cm.Data {
if ip != "" && strings.Contains(v, ip) {
t.Fatalf("primary pod IP %s leaked into restart-scoped config %s:\n%s", ip, k, v)
}
}
return r.configHash(ctx, cluster)
}
h1 := hashWithPrimaryIP("10.42.3.197")
h2 := hashWithPrimaryIP("10.42.9.42") // primary pod restarted onto a new IP
if h1 == "" {
t.Fatal("config hash should not be empty")
}
if h1 != h2 {
t.Fatalf("config hash changed when only the primary pod IP changed — the v0.2.5 roll loop:\n%s\n%s", h1, h2)
}
}
+30 -2
View File
@@ -2,6 +2,7 @@ package controller
import (
"context"
"fmt"
"strings"
"sigs.k8s.io/controller-runtime/pkg/client"
@@ -58,12 +59,19 @@ func recordsToUpdates(zone string, records []bindv1alpha1.Record, defaultTTL int
// updateKeyName returns the TSIG key name (as used in named.conf) for a zone's
// update key, falling back to the object name.
func updateKeyName(ctx context.Context, c client.Client, zone *bindv1alpha1.BindZone) string {
ref := zone.Spec.UpdateKeyRef
return tsigKeyName(ctx, c, zone.Namespace, zone.Spec.UpdateKeyRef)
}
// tsigKeyName resolves a BindTSIGKey object reference to the TSIG key name used
// in named.conf (the KeyName override when set, otherwise the object name).
// Returns "" for an empty ref, and falls back to the ref if the object cannot be
// read.
func tsigKeyName(ctx context.Context, c client.Client, namespace, ref string) string {
if ref == "" {
return ""
}
var key bindv1alpha1.BindTSIGKey
if err := c.Get(ctx, client.ObjectKey{Namespace: zone.Namespace, Name: ref}, &key); err != nil {
if err := c.Get(ctx, client.ObjectKey{Namespace: namespace, Name: ref}, &key); err != nil {
return ref
}
if key.Spec.KeyName != "" {
@@ -86,3 +94,23 @@ func terminateInline(entries []string) string {
}
return strings.Join(parts, " ")
}
// alsoNotifyList renders also-notify entries, each optionally annotated with a
// TSIG key so the primary signs its NOTIFYs and secondaries can accept them by
// key (`allow-notify { key ... }`) rather than by pod IP. An entry that already
// carries a `key` clause is left untouched.
func alsoNotifyList(addrs []string, key string) string {
key = strings.TrimSpace(key)
var parts []string
for _, a := range addrs {
a = strings.TrimSpace(strings.TrimRight(a, ";"))
if a == "" {
continue
}
if key != "" && !strings.Contains(a, " key ") {
a = fmt.Sprintf("%s key \"%s\"", a, key)
}
parts = append(parts, a+";")
}
return strings.Join(parts, " ")
}