From 6a51351c666a5f21d6676ef8090630c80ca332af Mon Sep 17 00:00:00 2001 From: Brent Graveland Date: Sat, 19 Sep 2026 19:03:07 -0600 Subject: [PATCH 1/7] fix(shard): stop the status hot loop on a healthy shard A healthy Shard rewrote its status about ten times a second, forever. Two server-side-apply field managers wrote the same status fields: setStorageClassCondition's guard apply and updateStatus each behaved correctly alone, but together they flipped status.orchReady and status.poolsReady false/true and alternated the StorageClassValid condition's message between two strings, with no terminal state. The guard built its apply payload from a typed ShardStatus literal that only set Conditions, but an SSA payload is a complete statement of ownership: the zero values of every other field on that literal (orchReady:false, poolsReady:false) got serialised too, and with ForceOwnership the guard seized both from updateStatus on every reconcile, which made updateStatus reclaim them right back next time. The guard now applies an unstructured payload that carries only status.conditions, so StorageClassValid has exactly one writer. Other visible changes along the way: - the StorageClassValid condition's messages changed: one verdict now covers both the backup and pool StorageClass checks instead of each reporting its own; - pool StorageClass lookups now happen before the Multiorch reconcile block (where the backup lookup already ran), instead of after it; - pools are checked in sorted name order instead of Go's randomised map order, which was a second, independent source of the condition message flapping; - the missing-StorageClass warning event is now recorded in Reconcile itself rather than inside the validation helpers, and the FailedApply event wording for a validation error changed accordingly. Coverage said nothing, because both halves were executed by existing tests and each was right in isolation: the defect only existed when both writers hit one API server. Added coverage now pins the guard's single-field payload and skip-if-unchanged behaviour, plus that Reconcile itself never makes more than one guard apply per pass. Signed-off-by: Brent Graveland --- .../controller/shard/shard_controller.go | 67 +- .../controller/shard/storage_class_guard.go | 220 +-- .../shard/storage_class_guard_test.go | 1254 ++++++++++++++--- 3 files changed, 1243 insertions(+), 298 deletions(-) diff --git a/pkg/resource-handler/controller/shard/shard_controller.go b/pkg/resource-handler/controller/shard/shard_controller.go index d0962cb7..d1229ef4 100644 --- a/pkg/resource-handler/controller/shard/shard_controller.go +++ b/pkg/resource-handler/controller/shard/shard_controller.go @@ -254,26 +254,46 @@ func (r *ShardReconciler) Reconcile( return ctrl.Result{}, err } - if err := r.validateBackupStorageClassDependency(ctx, shard); err != nil { - if isMissingStorageClassDependency(err) { - logger.Info( - "StorageClass dependency missing for shared backup PVC; requeueing", - "after", - storageClassDependencyRequeue, - ) - return ctrl.Result{RequeueAfter: storageClassDependencyRequeue}, nil - } + // Every StorageClass the shard references is resolved here, in one pass, and + // the StorageClassValid condition is published once. Two call sites used to + // write that condition with their own message and overwrite each other on + // every reconcile. The gates stay where they were: this one stops before + // the shared backup PVC, the pool one further down stops before the + // workloads that consume pool storage. + storageClasses, err := r.validateStorageClassDependencies(ctx, shard) + if err != nil { + monitoring.RecordSpanError(span, err) + logger.Error(err, "Failed to validate StorageClass dependencies") + r.Recorder.Eventf( + shard, + "Warning", + "FailedApply", + "Failed to validate StorageClass dependencies: %v", + err, + ) + return ctrl.Result{}, err + } + if err := r.setStorageClassCondition(ctx, shard, storageClasses); err != nil { monitoring.RecordSpanError(span, err) - logger.Error(err, "Failed to validate backup StorageClass") + logger.Error(err, "Failed to set StorageClass condition") r.Recorder.Eventf( shard, "Warning", "FailedApply", - "Failed to validate backup StorageClass: %v", + "Failed to set StorageClass condition: %v", err, ) return ctrl.Result{}, err } + if isMissingStorageClassDependency(storageClasses.backupDependency) { + r.Recorder.Event(shard, "Warning", storageClassNotFoundReason, storageClasses.message) + logger.Info( + "StorageClass dependency missing for shared backup PVC; requeueing", + "after", + storageClassDependencyRequeue, + ) + return ctrl.Result{RequeueAfter: storageClassDependencyRequeue}, nil + } // Reconcile Multiorch - one Deployment and Service per cell { @@ -335,25 +355,14 @@ func (r *ShardReconciler) Reconcile( childSpan.End() } - if err := r.validatePoolStorageClassDependencies(ctx, shard); err != nil { - if isMissingStorageClassDependency(err) { - logger.Info( - "StorageClass dependency missing for pool resources; requeueing", - "after", - storageClassDependencyRequeue, - ) - return ctrl.Result{RequeueAfter: storageClassDependencyRequeue}, nil - } - monitoring.RecordSpanError(span, err) - logger.Error(err, "Failed to validate pool StorageClass dependencies") - r.Recorder.Eventf( - shard, - "Warning", - "FailedApply", - "Failed to validate pool StorageClass dependencies: %v", - err, + if isMissingStorageClassDependency(storageClasses.poolDependency) { + r.Recorder.Event(shard, "Warning", storageClassNotFoundReason, storageClasses.message) + logger.Info( + "StorageClass dependency missing for pool resources; requeueing", + "after", + storageClassDependencyRequeue, ) - return ctrl.Result{}, err + return ctrl.Result{RequeueAfter: storageClassDependencyRequeue}, nil } // Render the effective postgres config into the operator-owned ConfigMap and diff --git a/pkg/resource-handler/controller/shard/storage_class_guard.go b/pkg/resource-handler/controller/shard/storage_class_guard.go index 2004d645..548b2e1a 100644 --- a/pkg/resource-handler/controller/shard/storage_class_guard.go +++ b/pkg/resource-handler/controller/shard/storage_class_guard.go @@ -4,12 +4,16 @@ import ( "context" "errors" "fmt" + "maps" + "slices" "time" storagev1 "k8s.io/api/storage/v1" apierrors "k8s.io/apimachinery/pkg/api/errors" "k8s.io/apimachinery/pkg/api/meta" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/apis/meta/v1/unstructured" + "k8s.io/apimachinery/pkg/runtime" "sigs.k8s.io/controller-runtime/pkg/client" multigresv1alpha1 "github.com/multigres/multigres-operator/api/v1alpha1" @@ -37,6 +41,27 @@ func isMissingStorageClassDependency(err error) bool { return errors.As(err, &depErr) } +// storageClassCheck is the whole StorageClassValid verdict for one reconcile: +// the condition to publish, plus the dependency errors that gate the reconcile. +// +// It exists so the condition has exactly one writer per reconcile. The backup +// and pool checks used to write it independently with different messages, and +// since setStorageClassCondition's skip-if-unchanged test compares the +// persisted message, each call saw the other's message and rewrote it, forever. +// +// backupDependency and poolDependency are held apart because they gate +// different points in Reconcile: a missing backup class must stop before the +// shared backup PVC is created, a missing pool class before the pool +// workloads. Each is a *missingStorageClassDependencyError when set. +type storageClassCheck struct { + status metav1.ConditionStatus + reason string + message string + + backupDependency error + poolDependency error +} + func backupFilesystemStorageClassName(shard *multigresv1alpha1.Shard) string { if shard.Spec.Backup == nil || shard.Spec.Backup.Type != multigresv1alpha1.BackupTypeFilesystem { @@ -67,102 +92,102 @@ func (r *ShardReconciler) validateStorageClassExists( return true, nil } -func (r *ShardReconciler) validateBackupStorageClassDependency( +// validateStorageClassDependencies resolves every StorageClass the shard +// references and reduces them to a single verdict. It writes nothing: the +// caller publishes the condition once and then gates on the missing-class +// fields, which is what keeps the condition single-writer. +// +// A missing backup class short-circuits before the pools are looked at, so the +// reported message matches the order in which Reconcile gates on them. +func (r *ShardReconciler) validateStorageClassDependencies( ctx context.Context, shard *multigresv1alpha1.Shard, -) error { +) (storageClassCheck, error) { backupClass := backupFilesystemStorageClassName(shard) - if backupClass == "" { - return r.setStorageClassCondition( - ctx, - shard, - metav1.ConditionTrue, - storageClassNotSpecifiedReason, - "No explicit backup filesystem StorageClass configured; using cluster default", - ) - } - - exists, err := r.validateStorageClassExists(ctx, backupClass) - if err != nil { - return fmt.Errorf("failed to validate backup StorageClass %q: %w", backupClass, err) - } - if !exists { - msg := fmt.Sprintf("StorageClass %q not found for shared backup PVCs", backupClass) - if setErr := r.setStorageClassCondition( - ctx, - shard, - metav1.ConditionFalse, - storageClassNotFoundReason, - msg, - ); setErr != nil { - return setErr + if backupClass != "" { + exists, err := r.validateStorageClassExists(ctx, backupClass) + if err != nil { + return storageClassCheck{}, fmt.Errorf( + "failed to validate backup StorageClass %q: %w", + backupClass, + err, + ) + } + if !exists { + return storageClassCheck{ + status: metav1.ConditionFalse, + reason: storageClassNotFoundReason, + message: fmt.Sprintf( + "StorageClass %q not found for shared backup PVCs", + backupClass, + ), + backupDependency: &missingStorageClassDependencyError{className: backupClass}, + }, nil } - r.Recorder.Event(shard, "Warning", storageClassNotFoundReason, msg) - return &missingStorageClassDependencyError{className: backupClass} } - return r.setStorageClassCondition( - ctx, - shard, - metav1.ConditionTrue, - storageClassFoundReason, - fmt.Sprintf("StorageClass %q found for shared backup PVCs", backupClass), - ) -} - -func (r *ShardReconciler) validatePoolStorageClassDependencies( - ctx context.Context, - shard *multigresv1alpha1.Shard, -) error { - hasExplicitPoolStorageClass := false - for poolName, pool := range shard.Spec.Pools { - if pool.Storage.Class == "" { + // Pools are visited in name order because Spec.Pools is a map: reporting + // whichever missing class Go's randomised map iteration reached first would + // flap the condition message between reconciles. + hasExplicitPoolClass := false + for _, poolName := range slices.Sorted(maps.Keys(shard.Spec.Pools)) { + poolClass := shard.Spec.Pools[poolName].Storage.Class + if poolClass == "" { continue } - hasExplicitPoolStorageClass = true + hasExplicitPoolClass = true - exists, err := r.validateStorageClassExists(ctx, pool.Storage.Class) + exists, err := r.validateStorageClassExists(ctx, poolClass) if err != nil { - return fmt.Errorf( + return storageClassCheck{}, fmt.Errorf( "failed to validate StorageClass %q for pool %s: %w", - pool.Storage.Class, + poolClass, poolName, err, ) } if !exists { - msg := fmt.Sprintf( - "StorageClass %q not found for pool %s", - pool.Storage.Class, - poolName, - ) - if setErr := r.setStorageClassCondition( - ctx, - shard, - metav1.ConditionFalse, - storageClassNotFoundReason, - msg, - ); setErr != nil { - return setErr - } - r.Recorder.Event(shard, "Warning", storageClassNotFoundReason, msg) - return &missingStorageClassDependencyError{className: pool.Storage.Class} + return storageClassCheck{ + status: metav1.ConditionFalse, + reason: storageClassNotFoundReason, + message: fmt.Sprintf( + "StorageClass %q not found for pool %s", + poolClass, + poolName, + ), + poolDependency: &missingStorageClassDependencyError{className: poolClass}, + }, nil } } - reason := storageClassNotSpecifiedReason - message := "No explicit pool StorageClass configured; using cluster default" - if hasExplicitPoolStorageClass { - reason = storageClassFoundReason - message = "All explicit pool StorageClasses are present" + if backupClass == "" && !hasExplicitPoolClass { + return storageClassCheck{ + status: metav1.ConditionTrue, + reason: storageClassNotSpecifiedReason, + message: "No explicit backup filesystem or pool StorageClass configured; using cluster default", + }, nil } - return r.setStorageClassCondition(ctx, shard, metav1.ConditionTrue, reason, message) + + return storageClassCheck{ + status: metav1.ConditionTrue, + reason: storageClassFoundReason, + message: "All explicitly configured StorageClasses are present", + }, nil } // setStorageClassCondition patches the StorageClassValid condition using SSA. // Uses FieldOwner("multigres-resource-handler-guard") to avoid ownership conflicts // with updateStatus which uses FieldOwner("multigres-resource-handler"). // +// The payload is unstructured and carries status.conditions and nothing else. +// An SSA apply payload is a complete statement of what its field manager owns, +// so a payload built from a partially-populated typed struct silently asserts +// the zero value of every non-omitempty field in it: a ShardStatus literal that +// sets only Conditions still serialises orchReady:false and poolsReady:false, +// and with ForceOwnership it seizes both fields from updateStatus, which forces +// them straight back on its next write. That is a permanent ping-pong, one +// round trip per reconcile, each one scheduling the next reconcile. +// // Reads the latest condition from the API server (not the in-memory shard) to // avoid false skips when the in-memory object is stale. // TODO: This stale-safe condition skip logic is mirrored in the TopoServer @@ -170,9 +195,7 @@ func (r *ShardReconciler) validatePoolStorageClassDependencies( func (r *ShardReconciler) setStorageClassCondition( ctx context.Context, shard *multigresv1alpha1.Shard, - condStatus metav1.ConditionStatus, - reason string, - message string, + check storageClassCheck, ) error { // Read the latest from the API server so the skip-if-unchanged check // compares against the real persisted state, not a potentially stale @@ -184,9 +207,9 @@ func (r *ShardReconciler) setStorageClassCondition( existing := meta.FindStatusCondition(latest.Status.Conditions, conditionStorageClassValid) if existing != nil && - existing.Status == condStatus && - existing.Reason == reason && - existing.Message == message && + existing.Status == check.status && + existing.Reason == check.reason && + existing.Message == check.message && existing.ObservedGeneration == latest.Generation { return nil } @@ -194,32 +217,35 @@ func (r *ShardReconciler) setStorageClassCondition( // Preserve LastTransitionTime when the status hasn't transitioned, // matching the behaviour of meta.SetStatusCondition. now := metav1.Now() - if existing != nil && existing.Status == condStatus { + if existing != nil && existing.Status == check.status { now = existing.LastTransitionTime } - patchObj := &multigresv1alpha1.Shard{ - TypeMeta: metav1.TypeMeta{ - APIVersion: multigresv1alpha1.GroupVersion.String(), - Kind: "Shard", - }, - ObjectMeta: metav1.ObjectMeta{ - Name: shard.Name, - Namespace: shard.Namespace, + // Converted from the typed condition rather than hand-built so the wire + // encoding, notably the metav1.Time format, cannot drift from the API's. + cond, err := runtime.DefaultUnstructuredConverter.ToUnstructured(&metav1.Condition{ + Type: conditionStorageClassValid, + Status: check.status, + Reason: check.reason, + Message: check.message, + ObservedGeneration: latest.Generation, + LastTransitionTime: now, + }) + if err != nil { + return fmt.Errorf("failed to encode Shard StorageClass condition: %w", err) + } + + patchObj := &unstructured.Unstructured{Object: map[string]any{ + "apiVersion": multigresv1alpha1.GroupVersion.String(), + "kind": "Shard", + "metadata": map[string]any{ + "name": shard.Name, + "namespace": shard.Namespace, }, - Status: multigresv1alpha1.ShardStatus{ - Conditions: []metav1.Condition{ - { - Type: conditionStorageClassValid, - Status: condStatus, - Reason: reason, - Message: message, - ObservedGeneration: latest.Generation, - LastTransitionTime: now, - }, - }, + "status": map[string]any{ + "conditions": []any{cond}, }, - } + }} if err := r.Status().Patch( ctx, diff --git a/pkg/resource-handler/controller/shard/storage_class_guard_test.go b/pkg/resource-handler/controller/shard/storage_class_guard_test.go index 7df20f97..70137e36 100644 --- a/pkg/resource-handler/controller/shard/storage_class_guard_test.go +++ b/pkg/resource-handler/controller/shard/storage_class_guard_test.go @@ -1,19 +1,27 @@ package shard import ( + "context" + "encoding/json" "errors" + "maps" + "slices" + "strings" "testing" + "time" appsv1 "k8s.io/api/apps/v1" corev1 "k8s.io/api/core/v1" policyv1 "k8s.io/api/policy/v1" storagev1 "k8s.io/api/storage/v1" + apierrors "k8s.io/apimachinery/pkg/api/errors" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" "k8s.io/apimachinery/pkg/runtime" "k8s.io/utils/ptr" ctrl "sigs.k8s.io/controller-runtime" "sigs.k8s.io/controller-runtime/pkg/client" "sigs.k8s.io/controller-runtime/pkg/client/fake" + "sigs.k8s.io/controller-runtime/pkg/client/interceptor" "k8s.io/client-go/tools/record" @@ -22,191 +30,620 @@ import ( "github.com/multigres/multigres-operator/pkg/util/metadata" ) -func TestValidateBackupStorageClassDependency(t *testing.T) { +func TestValidateStorageClassDependencies(t *testing.T) { scheme := runtime.NewScheme() _ = multigresv1alpha1.AddToScheme(scheme) _ = storagev1.AddToScheme(scheme) - t.Run("no explicit backup class sets true not-specified condition", func(t *testing.T) { - shard := &multigresv1alpha1.Shard{ - ObjectMeta: metav1.ObjectMeta{Name: "test-shard", Namespace: "default"}, - } + newReconciler := func(objs ...client.Object) *ShardReconciler { c := fake.NewClientBuilder(). WithScheme(scheme). - WithObjects(shard). + WithObjects(objs...). WithStatusSubresource(&multigresv1alpha1.Shard{}). Build() - r := &ShardReconciler{Client: c, Scheme: scheme, Recorder: record.NewFakeRecorder(10)} + return &ShardReconciler{Client: c, Scheme: scheme, Recorder: record.NewFakeRecorder(10)} + } - if err := r.validateBackupStorageClassDependency(t.Context(), shard); err != nil { - t.Fatalf("unexpected error: %v", err) + filesystemBackup := func(class string) *multigresv1alpha1.BackupConfig { + return &multigresv1alpha1.BackupConfig{ + Type: multigresv1alpha1.BackupTypeFilesystem, + Filesystem: &multigresv1alpha1.FilesystemBackupConfig{ + Storage: multigresv1alpha1.StorageSpec{Class: class}, + }, + } + } + + t.Run("nothing explicit reports one not-specified verdict", func(t *testing.T) { + shard := &multigresv1alpha1.Shard{ + ObjectMeta: metav1.ObjectMeta{Name: "test-shard", Namespace: "default"}, + Spec: multigresv1alpha1.ShardSpec{ + Pools: map[multigresv1alpha1.PoolName]multigresv1alpha1.PoolSpec{ + "primary": {Storage: multigresv1alpha1.StorageSpec{Size: "10Gi"}}, + }, + }, } + r := newReconciler(shard) - var updated multigresv1alpha1.Shard - if err := c.Get(t.Context(), client.ObjectKeyFromObject(shard), &updated); err != nil { - t.Fatalf("failed to read shard: %v", err) + check, err := r.validateStorageClassDependencies(t.Context(), shard) + if err != nil { + t.Fatalf("unexpected error: %v", err) } - cond := findCondition(updated.Status.Conditions, conditionStorageClassValid) - if cond == nil || cond.Status != metav1.ConditionTrue || - cond.Reason != storageClassNotSpecifiedReason { - t.Fatalf("unexpected condition: %#v", cond) + if check.status != metav1.ConditionTrue || check.reason != storageClassNotSpecifiedReason { + t.Fatalf("unexpected verdict: %+v", check) + } + if check.backupDependency != nil || check.poolDependency != nil { + t.Fatalf("expected no dependency errors, got %+v", check) } }) - t.Run("missing backup class returns dependency error and false condition", func(t *testing.T) { + t.Run("explicit classes all present report found", func(t *testing.T) { shard := &multigresv1alpha1.Shard{ ObjectMeta: metav1.ObjectMeta{Name: "test-shard", Namespace: "default"}, Spec: multigresv1alpha1.ShardSpec{ - Backup: &multigresv1alpha1.BackupConfig{ - Type: multigresv1alpha1.BackupTypeFilesystem, - Filesystem: &multigresv1alpha1.FilesystemBackupConfig{ - Storage: multigresv1alpha1.StorageSpec{Class: "missing-sc"}, + Backup: filesystemBackup("backup-sc"), + Pools: map[multigresv1alpha1.PoolName]multigresv1alpha1.PoolSpec{ + "primary": { + Storage: multigresv1alpha1.StorageSpec{Size: "10Gi", Class: "fast"}, }, }, }, } - c := fake.NewClientBuilder(). - WithScheme(scheme). - WithObjects(shard). - WithStatusSubresource(&multigresv1alpha1.Shard{}). - Build() - r := &ShardReconciler{Client: c, Scheme: scheme, Recorder: record.NewFakeRecorder(10)} + r := newReconciler( + shard, + &storagev1.StorageClass{ObjectMeta: metav1.ObjectMeta{Name: "backup-sc"}}, + &storagev1.StorageClass{ObjectMeta: metav1.ObjectMeta{Name: "fast"}}, + ) - err := r.validateBackupStorageClassDependency(t.Context(), shard) - if err == nil || !isMissingStorageClassDependency(err) { - t.Fatalf("expected missing dependency error, got: %v", err) + check, err := r.validateStorageClassDependencies(t.Context(), shard) + if err != nil { + t.Fatalf("unexpected error: %v", err) } + if check.status != metav1.ConditionTrue || check.reason != storageClassFoundReason { + t.Fatalf("unexpected verdict: %+v", check) + } + }) - var updated multigresv1alpha1.Shard - if getErr := c.Get( - t.Context(), - client.ObjectKeyFromObject(shard), - &updated, - ); getErr != nil { - t.Fatalf("failed to read shard: %v", getErr) + // One writer means one verdict, and in the mixed cases the merged verdict + // reports Found where the old pair reported NotSpecified: the pool validator + // ran last and claimed the shard as unconfigured even when the backup class + // was explicit. The reason is published on the condition, so both directions + // of "some of it is explicit" are pinned here. + t.Run("explicit backup class with no pool class reports found", func(t *testing.T) { + shard := &multigresv1alpha1.Shard{ + ObjectMeta: metav1.ObjectMeta{Name: "test-shard", Namespace: "default"}, + Spec: multigresv1alpha1.ShardSpec{ + Backup: filesystemBackup("backup-sc"), + Pools: map[multigresv1alpha1.PoolName]multigresv1alpha1.PoolSpec{ + "primary": {Storage: multigresv1alpha1.StorageSpec{Size: "10Gi"}}, + }, + }, + } + r := newReconciler( + shard, + &storagev1.StorageClass{ObjectMeta: metav1.ObjectMeta{Name: "backup-sc"}}, + ) + + check, err := r.validateStorageClassDependencies(t.Context(), shard) + if err != nil { + t.Fatalf("unexpected error: %v", err) } - cond := findCondition(updated.Status.Conditions, conditionStorageClassValid) - if cond == nil || cond.Status != metav1.ConditionFalse || - cond.Reason != storageClassNotFoundReason { - t.Fatalf("unexpected condition: %#v", cond) + if check.status != metav1.ConditionTrue || check.reason != storageClassFoundReason { + t.Fatalf("unexpected verdict: %+v", check) } }) -} -func TestValidatePoolStorageClassDependencies(t *testing.T) { - scheme := runtime.NewScheme() - _ = multigresv1alpha1.AddToScheme(scheme) - _ = storagev1.AddToScheme(scheme) - - t.Run("no explicit pool class sets true not-specified condition", func(t *testing.T) { + t.Run("explicit pool class with no backup class reports found", func(t *testing.T) { shard := &multigresv1alpha1.Shard{ ObjectMeta: metav1.ObjectMeta{Name: "test-shard", Namespace: "default"}, Spec: multigresv1alpha1.ShardSpec{ Pools: map[multigresv1alpha1.PoolName]multigresv1alpha1.PoolSpec{ "primary": { - Storage: multigresv1alpha1.StorageSpec{Size: "10Gi"}, + Storage: multigresv1alpha1.StorageSpec{Size: "10Gi", Class: "fast"}, }, }, }, } - c := fake.NewClientBuilder(). - WithScheme(scheme). - WithObjects(shard). - WithStatusSubresource(&multigresv1alpha1.Shard{}). - Build() - r := &ShardReconciler{Client: c, Scheme: scheme, Recorder: record.NewFakeRecorder(10)} + r := newReconciler( + shard, + &storagev1.StorageClass{ObjectMeta: metav1.ObjectMeta{Name: "fast"}}, + ) - if err := r.validatePoolStorageClassDependencies(t.Context(), shard); err != nil { + check, err := r.validateStorageClassDependencies(t.Context(), shard) + if err != nil { t.Fatalf("unexpected error: %v", err) } + if check.status != metav1.ConditionTrue || check.reason != storageClassFoundReason { + t.Fatalf("unexpected verdict: %+v", check) + } + }) - var updated multigresv1alpha1.Shard - if err := c.Get(t.Context(), client.ObjectKeyFromObject(shard), &updated); err != nil { - t.Fatalf("failed to read shard: %v", err) + t.Run("missing backup class reports the backup dependency", func(t *testing.T) { + shard := &multigresv1alpha1.Shard{ + ObjectMeta: metav1.ObjectMeta{Name: "test-shard", Namespace: "default"}, + Spec: multigresv1alpha1.ShardSpec{Backup: filesystemBackup("missing-sc")}, + } + r := newReconciler(shard) + + check, err := r.validateStorageClassDependencies(t.Context(), shard) + if err != nil { + t.Fatalf("unexpected error: %v", err) + } + if check.status != metav1.ConditionFalse || check.reason != storageClassNotFoundReason { + t.Fatalf("unexpected verdict: %+v", check) + } + if !isMissingStorageClassDependency(check.backupDependency) { + t.Fatalf("expected backup dependency error, got %v", check.backupDependency) } - cond := findCondition(updated.Status.Conditions, conditionStorageClassValid) - if cond == nil || cond.Status != metav1.ConditionTrue || - cond.Reason != storageClassNotSpecifiedReason { - t.Fatalf("unexpected condition: %#v", cond) + if check.poolDependency != nil { + t.Fatalf("expected no pool dependency error, got %v", check.poolDependency) + } + if !strings.Contains(check.message, `"missing-sc"`) { + t.Fatalf("message must name the class, got %q", check.message) } }) - t.Run("all explicit pool classes present sets true found condition", func(t *testing.T) { + t.Run("missing pool class reports the pool dependency", func(t *testing.T) { shard := &multigresv1alpha1.Shard{ ObjectMeta: metav1.ObjectMeta{Name: "test-shard", Namespace: "default"}, Spec: multigresv1alpha1.ShardSpec{ Pools: map[multigresv1alpha1.PoolName]multigresv1alpha1.PoolSpec{ "primary": { - Storage: multigresv1alpha1.StorageSpec{Size: "10Gi", Class: "fast"}, + Storage: multigresv1alpha1.StorageSpec{Size: "10Gi", Class: "missing-sc"}, }, }, }, } - c := fake.NewClientBuilder(). - WithScheme(scheme). - WithObjects(shard, &storagev1.StorageClass{ObjectMeta: metav1.ObjectMeta{Name: "fast"}}). - WithStatusSubresource(&multigresv1alpha1.Shard{}). - Build() - r := &ShardReconciler{Client: c, Scheme: scheme, Recorder: record.NewFakeRecorder(10)} + r := newReconciler(shard) - if err := r.validatePoolStorageClassDependencies(t.Context(), shard); err != nil { + check, err := r.validateStorageClassDependencies(t.Context(), shard) + if err != nil { t.Fatalf("unexpected error: %v", err) } - - var updated multigresv1alpha1.Shard - if err := c.Get(t.Context(), client.ObjectKeyFromObject(shard), &updated); err != nil { - t.Fatalf("failed to read shard: %v", err) + if check.status != metav1.ConditionFalse || check.reason != storageClassNotFoundReason { + t.Fatalf("unexpected verdict: %+v", check) } - cond := findCondition(updated.Status.Conditions, conditionStorageClassValid) - if cond == nil || cond.Status != metav1.ConditionTrue || - cond.Reason != storageClassFoundReason { - t.Fatalf("unexpected condition: %#v", cond) + if !isMissingStorageClassDependency(check.poolDependency) { + t.Fatalf("expected pool dependency error, got %v", check.poolDependency) + } + if check.backupDependency != nil { + t.Fatalf("expected no backup dependency error, got %v", check.backupDependency) + } + if !strings.Contains(check.message, "primary") { + t.Fatalf("message must name the pool, got %q", check.message) } }) - t.Run( - "missing explicit pool class returns dependency error and false condition", - func(t *testing.T) { - shard := &multigresv1alpha1.Shard{ - ObjectMeta: metav1.ObjectMeta{Name: "test-shard", Namespace: "default"}, - Spec: multigresv1alpha1.ShardSpec{ - Pools: map[multigresv1alpha1.PoolName]multigresv1alpha1.PoolSpec{ - "primary": { - Storage: multigresv1alpha1.StorageSpec{ - Size: "10Gi", - Class: "missing-sc", - }, + t.Run("missing backup class wins over a missing pool class", func(t *testing.T) { + shard := &multigresv1alpha1.Shard{ + ObjectMeta: metav1.ObjectMeta{Name: "test-shard", Namespace: "default"}, + Spec: multigresv1alpha1.ShardSpec{ + Backup: filesystemBackup("missing-backup-sc"), + Pools: map[multigresv1alpha1.PoolName]multigresv1alpha1.PoolSpec{ + "primary": { + Storage: multigresv1alpha1.StorageSpec{ + Size: "10Gi", + Class: "missing-pool-sc", }, }, }, - } - c := fake.NewClientBuilder(). - WithScheme(scheme). - WithObjects(shard). - WithStatusSubresource(&multigresv1alpha1.Shard{}). - Build() - r := &ShardReconciler{Client: c, Scheme: scheme, Recorder: record.NewFakeRecorder(10)} + }, + } + r := newReconciler(shard) - err := r.validatePoolStorageClassDependencies(t.Context(), shard) - if err == nil || !isMissingStorageClassDependency(err) { - t.Fatalf("expected missing dependency error, got: %v", err) - } + check, err := r.validateStorageClassDependencies(t.Context(), shard) + if err != nil { + t.Fatalf("unexpected error: %v", err) + } + if !isMissingStorageClassDependency(check.backupDependency) || + check.poolDependency != nil { + t.Fatalf("expected only the backup dependency, got %+v", check) + } + }) + + // Spec.Pools is a map, so an unordered scan would report whichever missing + // pool Go's iteration reached first and flap the condition message. + t.Run("the reported pool is stable across calls", func(t *testing.T) { + shard := &multigresv1alpha1.Shard{ + ObjectMeta: metav1.ObjectMeta{Name: "test-shard", Namespace: "default"}, + Spec: multigresv1alpha1.ShardSpec{ + Pools: map[multigresv1alpha1.PoolName]multigresv1alpha1.PoolSpec{ + "aaa": { + Storage: multigresv1alpha1.StorageSpec{Size: "10Gi", Class: "missing-a"}, + }, + "bbb": { + Storage: multigresv1alpha1.StorageSpec{Size: "10Gi", Class: "missing-b"}, + }, + "ccc": { + Storage: multigresv1alpha1.StorageSpec{Size: "10Gi", Class: "missing-c"}, + }, + }, + }, + } + r := newReconciler(shard) - var updated multigresv1alpha1.Shard - if getErr := c.Get( - t.Context(), - client.ObjectKeyFromObject(shard), - &updated, - ); getErr != nil { - t.Fatalf("failed to read shard: %v", getErr) + want := "" + for i := range 30 { + check, err := r.validateStorageClassDependencies(t.Context(), shard) + if err != nil { + t.Fatalf("unexpected error: %v", err) + } + if i == 0 { + want = check.message + } + if check.message != want { + t.Fatalf("message flapped: %q then %q", want, check.message) } - cond := findCondition(updated.Status.Conditions, conditionStorageClassValid) - if cond == nil || cond.Status != metav1.ConditionFalse || - cond.Reason != storageClassNotFoundReason { - t.Fatalf("unexpected condition: %#v", cond) + } + if !strings.Contains(want, "aaa") { + t.Fatalf("expected the first pool by name, got %q", want) + } + }) +} + +// TestSetStorageClassCondition_AppliesOnlyConditions pins the defect that made +// a healthy shard rewrite its status forever. The guard's apply payload is a +// complete statement of what its field manager owns, so it must carry +// status.conditions and nothing else: a payload built from a typed ShardStatus +// literal also serialises orchReady:false and poolsReady:false, and with +// ForceOwnership it seizes both from updateStatus on every reconcile. +// +// The assertion is on the serialised payload rather than on the Go value, +// because the Go value looks correct in both the fixed and the broken version. +// The zero values only become an assertion once they are marshalled. +// +// It cannot be made against the object after a round trip here: the fake client +// does not scope an apply patch to the payload's fields, it replaces the whole +// status, so orchReady and poolsReady are lost either way. The round-trip half +// of this belongs to an apiserver, and lives in TestShardStatusQuiesces under +// test/suite. +func TestSetStorageClassCondition_AppliesOnlyConditions(t *testing.T) { + t.Parallel() + + scheme := runtime.NewScheme() + _ = multigresv1alpha1.AddToScheme(scheme) + _ = storagev1.AddToScheme(scheme) + + shard := &multigresv1alpha1.Shard{ + ObjectMeta: metav1.ObjectMeta{Name: "test-shard", Namespace: "default"}, + Status: multigresv1alpha1.ShardStatus{ + OrchReady: true, + PoolsReady: true, + }, + } + + baseClient := fake.NewClientBuilder(). + WithScheme(scheme). + WithObjects(shard). + WithStatusSubresource(&multigresv1alpha1.Shard{}). + Build() + + var captured []byte + fakeClient := testutil.NewFakeClientWithFailures(baseClient, &testutil.FailureConfig{ + OnStatusPatch: func(obj client.Object) error { + raw, err := json.Marshal(obj) + if err != nil { + t.Fatalf("marshal patch payload: %v", err) } + captured = raw + return nil }, - ) + }) + + r := &ShardReconciler{Client: fakeClient, Scheme: scheme, Recorder: record.NewFakeRecorder(10)} + + check := storageClassCheck{ + status: metav1.ConditionTrue, + reason: storageClassNotSpecifiedReason, + message: "No explicit backup filesystem or pool StorageClass configured; using cluster default", + } + if err := r.setStorageClassCondition(t.Context(), shard, check); err != nil { + t.Fatalf("setStorageClassCondition: %v", err) + } + if captured == nil { + t.Fatal("guard did not apply a status patch") + } + + var payload struct { + Status map[string]json.RawMessage `json:"status"` + } + if err := json.Unmarshal(captured, &payload); err != nil { + t.Fatalf("unmarshal patch payload: %v", err) + } + keys := slices.Sorted(maps.Keys(payload.Status)) + if !slices.Equal(keys, []string{"conditions"}) { + t.Fatalf("guard payload must own status.conditions only, got %v", keys) + } + + var conditions []metav1.Condition + if err := json.Unmarshal(payload.Status["conditions"], &conditions); err != nil { + t.Fatalf("unmarshal conditions: %v", err) + } + if len(conditions) != 1 || conditions[0].Type != conditionStorageClassValid { + t.Fatalf("guard payload must carry exactly the %s condition, got %+v", + conditionStorageClassValid, conditions) + } + if conditions[0].Status != check.status || conditions[0].Reason != check.reason || + conditions[0].Message != check.message { + t.Fatalf("condition does not match the verdict: %+v", conditions[0]) + } +} + +// TestStorageClassCondition_IsStableAcrossReconciles pins the second defect: +// two validations wrote the same condition type with different messages, so +// every reconcile rewrote the other's and the condition never settled. Once it +// has one writer, the second cycle must not write at all. +// +// This drives the guard cycle directly rather than Reconcile, because a full +// reconcile calls updateStatus again afterwards and the fake client's apply +// drops the guard's condition when it does (see +// TestSetStorageClassCondition_AppliesOnlyConditions). The full-reconcile form +// of this assertion is TestShardStatusQuiesces under test/suite. +func TestStorageClassCondition_IsStableAcrossReconciles(t *testing.T) { + t.Parallel() + + scheme := runtime.NewScheme() + _ = multigresv1alpha1.AddToScheme(scheme) + _ = storagev1.AddToScheme(scheme) + + // No explicit StorageClass anywhere: the case both validations used to + // claim with a message of their own. + shard := &multigresv1alpha1.Shard{ + ObjectMeta: metav1.ObjectMeta{Name: "test-shard", Namespace: "default"}, + Spec: multigresv1alpha1.ShardSpec{ + Pools: map[multigresv1alpha1.PoolName]multigresv1alpha1.PoolSpec{ + "primary": {Storage: multigresv1alpha1.StorageSpec{Size: "10Gi"}}, + }, + }, + } + + baseClient := fake.NewClientBuilder(). + WithScheme(scheme). + WithObjects(shard). + WithStatusSubresource(&multigresv1alpha1.Shard{}). + Build() + + patches := 0 + fakeClient := testutil.NewFakeClientWithFailures(baseClient, &testutil.FailureConfig{ + OnStatusPatch: func(client.Object) error { + patches++ + return nil + }, + }) + + r := &ShardReconciler{Client: fakeClient, Scheme: scheme, Recorder: record.NewFakeRecorder(50)} + + reconcileStorageClasses := func() []byte { + check, err := r.validateStorageClassDependencies(t.Context(), shard) + if err != nil { + t.Fatalf("validate: %v", err) + } + if err := r.setStorageClassCondition(t.Context(), shard, check); err != nil { + t.Fatalf("set condition: %v", err) + } + + var got multigresv1alpha1.Shard + if err := baseClient.Get(t.Context(), client.ObjectKeyFromObject(shard), &got); err != nil { + t.Fatalf("read shard: %v", err) + } + cond := findCondition(got.Status.Conditions, conditionStorageClassValid) + if cond == nil { + t.Fatalf("no %s condition", conditionStorageClassValid) + } + raw, err := json.Marshal(cond) + if err != nil { + t.Fatalf("marshal condition: %v", err) + } + return raw + } + + first := reconcileStorageClasses() + if patches != 1 { + t.Fatalf("first cycle must apply the condition once, applied %d times", patches) + } + + second := reconcileStorageClasses() + if string(first) != string(second) { + t.Fatalf( + "StorageClassValid condition moved between reconciles:\n %s\n %s", + first, + second, + ) + } + if patches != 1 { + t.Fatalf("second cycle rewrote a settled condition: %d applies total", patches) + } +} + +// guardStatusApply is one status apply captured under the guard's field +// owner: the top-level keys of the applied status object, and the single +// StorageClassValid condition it carried. +type guardStatusApply struct { + statusKeys []string + condition metav1.Condition +} + +// TestReconcile_StorageClassConditionSettlesAcrossReconciles drives the real +// ShardReconciler.Reconcile, not the guard cycle in isolation, because the hot +// loop only existed when the guard's apply and updateStatus's apply landed +// against the same object in the same reconcile: a reviewer showed that +// reintroducing a second writer for StorageClassValid (e.g. splitting the +// backup and pool checks back into two condition-setting calls) passes +// TestStorageClassCondition_IsStableAcrossReconciles above, because that test +// drives the merged validateStorageClassDependencies directly and never +// exercises whatever calls the write path from within one Reconcile pass. +// +// It captures every guard-owned status apply's payload directly, identified +// by the real client.SubResourcePatchOptions.FieldManager rather than by +// payload shape, so the assertions hold regardless of how the guard's payload +// happens to be built. +// +// It does not assert that the second reconcile makes zero guard applies, even +// though on a real API server it would: the fake client's SSA implementation +// (managedfields.DeducedTypeConverter, since the Shard CRD's structural schema +// isn't available to it) treats the +listType=map Conditions slice as atomic, +// so updateStatus's own apply - a different field owner, earlier in the same +// Reconcile - replaces status.conditions wholesale and erases whatever the +// guard wrote in the previous reconcile. That forces the guard to see no +// existing condition and reapply every single call, on both fixed and broken +// code, which is exactly why TestStorageClassCondition_IsStableAcrossReconciles +// avoids driving Reconcile for its own assertion. What is real and worth +// pinning here: each individual Reconcile call makes at most one guard apply +// (a second writer within one pass would make two), and the verdict it writes +// does not flap from one reconcile to the next. The true single-writer, +// settles-for-real proof against a schema-aware apply is +// TestShardStatusQuiesces under test/suite. +func TestReconcile_StorageClassConditionSettlesAcrossReconciles(t *testing.T) { + t.Parallel() + + scheme := runtime.NewScheme() + _ = multigresv1alpha1.AddToScheme(scheme) + _ = appsv1.AddToScheme(scheme) + _ = corev1.AddToScheme(scheme) + _ = policyv1.AddToScheme(scheme) + _ = storagev1.AddToScheme(scheme) + + // No explicit StorageClass anywhere: the case the guard and updateStatus + // used to fight over. + shard := &multigresv1alpha1.Shard{ + ObjectMeta: metav1.ObjectMeta{Name: "stable-shard", Namespace: "default"}, + Spec: multigresv1alpha1.ShardSpec{ + DatabaseName: "testdb", + TableGroupName: "default", + Multiorch: multigresv1alpha1.MultiorchSpec{ + Cells: []multigresv1alpha1.CellName{"zone1"}, + }, + Pools: map[multigresv1alpha1.PoolName]multigresv1alpha1.PoolSpec{ + "primary": { + Cells: []multigresv1alpha1.CellName{"zone1"}, + Type: "replica", + ReplicasPerCell: ptr.To(int32(1)), + Storage: multigresv1alpha1.StorageSpec{Size: "10Gi"}, + }, + }, + }, + } + setTestPostgresPasswordSecretRef(shard) + + var guardApplies []guardStatusApply + fakeClient := fake.NewClientBuilder(). + WithScheme(scheme). + WithObjects(shard, testPostgresPasswordSecretForShard(shard)). + WithStatusSubresource(&multigresv1alpha1.Shard{}). + WithInterceptorFuncs(interceptor.Funcs{ + SubResourcePatch: func( + ctx context.Context, + c client.Client, + subResourceName string, + obj client.Object, + patch client.Patch, + opts ...client.SubResourcePatchOption, + ) error { + popts := (&client.SubResourcePatchOptions{}).ApplyOptions(opts) + if popts.FieldManager == "multigres-resource-handler-guard" { + raw, err := json.Marshal(obj) + if err != nil { + t.Fatalf("marshal guard payload: %v", err) + } + var payload struct { + Status map[string]json.RawMessage `json:"status"` + } + if err := json.Unmarshal(raw, &payload); err != nil { + t.Fatalf("unmarshal guard payload: %v", err) + } + var conditions []metav1.Condition + if raw, ok := payload.Status["conditions"]; ok { + if err := json.Unmarshal(raw, &conditions); err != nil { + t.Fatalf("unmarshal guard conditions: %v", err) + } + } + if len(conditions) != 1 { + t.Fatalf( + "guard payload must carry exactly one condition, got %+v", + conditions, + ) + } + guardApplies = append(guardApplies, guardStatusApply{ + statusKeys: slices.Sorted(maps.Keys(payload.Status)), + condition: conditions[0], + }) + } + return c.SubResource(subResourceName).Patch(ctx, obj, patch, opts...) + }, + }). + Build() + + r := &ShardReconciler{ + Client: fakeClient, + Scheme: scheme, + Recorder: record.NewFakeRecorder(100), + CreateTopoStore: newMemoryTopoFactory(), + } + + req := ctrl.Request{NamespacedName: client.ObjectKeyFromObject(shard)} + + assertOnlyOwnsConditions := func(t *testing.T, apply guardStatusApply) { + t.Helper() + if !slices.Equal(apply.statusKeys, []string{"conditions"}) { + t.Fatalf("guard payload must own status.conditions only, got %v", apply.statusKeys) + } + if apply.condition.Type != conditionStorageClassValid { + t.Fatalf("guard payload must carry the %s condition, got %+v", + conditionStorageClassValid, apply.condition) + } + } + + if _, err := r.Reconcile(t.Context(), req); err != nil { + t.Fatalf("first reconcile: %v", err) + } + if len(guardApplies) != 1 { + t.Fatalf( + "first reconcile must make exactly one guard apply (a second writer is back), got %d", + len(guardApplies), + ) + } + first := guardApplies[0] + assertOnlyOwnsConditions(t, first) + if first.condition.Status != metav1.ConditionTrue || + first.condition.Reason != storageClassNotSpecifiedReason { + t.Fatalf("unexpected first verdict: %+v", first.condition) + } + + // The fake client's SSA status apply replaces the whole object rather than + // scoping to the applied fields, which drops Spec. Restore it exactly as the + // existing multi-reconcile loop in TestShardReconciler_Reconcile does, so the + // second reconcile sees the same spec as the first rather than erroring on a + // shard with no pools. + var stored multigresv1alpha1.Shard + if err := fakeClient.Get(t.Context(), req.NamespacedName, &stored); err != nil { + t.Fatalf("read shard before restoring spec: %v", err) + } + stored.Spec = shard.Spec + if err := fakeClient.Update(t.Context(), &stored); err != nil { + t.Fatalf("restore shard spec: %v", err) + } + + if _, err := r.Reconcile(t.Context(), req); err != nil { + t.Fatalf("second reconcile: %v", err) + } + if len(guardApplies) != 2 { + t.Fatalf( + "second reconcile must make exactly one guard apply of its own (a second writer "+ + "is back), got %d total", + len(guardApplies), + ) + } + second := guardApplies[1] + assertOnlyOwnsConditions(t, second) + + if second.condition.Status != first.condition.Status || + second.condition.Reason != first.condition.Reason || + second.condition.Message != first.condition.Message { + t.Fatalf( + "StorageClassValid verdict flapped between reconciles:\n %+v\n %+v", + first.condition, + second.condition, + ) + } } func TestReconcile_MissingStorageClassReturnsDependencyRequeueEvenWhenPVCExists(t *testing.T) { @@ -406,88 +843,561 @@ func TestShardReconciler_FieldOwnershipIsolation(t *testing.T) { t.Fatal("updateStatus patch must contain Available condition") } }) +} - t.Run( - "guard patch contains only StorageClassValid condition and no other status fields", - func(t *testing.T) { - t.Parallel() +// storageClassGateScheme registers everything a full Reconcile of the gate +// fixture below touches. +func storageClassGateScheme() *runtime.Scheme { + scheme := runtime.NewScheme() + _ = multigresv1alpha1.AddToScheme(scheme) + _ = appsv1.AddToScheme(scheme) + _ = corev1.AddToScheme(scheme) + _ = policyv1.AddToScheme(scheme) + _ = storagev1.AddToScheme(scheme) + return scheme +} - shard := &multigresv1alpha1.Shard{ - ObjectMeta: metav1.ObjectMeta{ - Name: "test-field-owner-2", - Namespace: "default", +// storageClassGateShard is a shard that reconciles far enough to pass both +// storage-class gates, so a test can make exactly one of them fire by naming a +// class that does not exist. +func storageClassGateShard(backupClass, poolClass string) *multigresv1alpha1.Shard { + return &multigresv1alpha1.Shard{ + ObjectMeta: metav1.ObjectMeta{ + Name: "gate-shard", + Namespace: "default", + Labels: map[string]string{ + metadata.LabelMultigresCluster: "test-cluster", + }, + }, + Spec: multigresv1alpha1.ShardSpec{ + DatabaseName: "db", + TableGroupName: "tg", + ShardName: "s1", + PostgresPasswordSecretRef: multigresv1alpha1.PostgresPasswordSecretRef{ + Name: testPostgresAuthRefName, + Key: PostgresPasswordSecretKey, + }, + Multiorch: multigresv1alpha1.MultiorchSpec{ + Cells: []multigresv1alpha1.CellName{"zone1"}, + }, + Backup: &multigresv1alpha1.BackupConfig{ + Type: multigresv1alpha1.BackupTypeFilesystem, + Filesystem: &multigresv1alpha1.FilesystemBackupConfig{ + Storage: multigresv1alpha1.StorageSpec{Size: "10Gi", Class: backupClass}, }, - Spec: multigresv1alpha1.ShardSpec{ - Pools: map[multigresv1alpha1.PoolName]multigresv1alpha1.PoolSpec{ - "primary": { - Storage: multigresv1alpha1.StorageSpec{ - Size: "10Gi", - Class: "fast-ssd", - }, - }, - }, + }, + Pools: map[multigresv1alpha1.PoolName]multigresv1alpha1.PoolSpec{ + "primary": { + ReplicasPerCell: ptr.To(int32(1)), + Cells: []multigresv1alpha1.CellName{"zone1"}, + Type: "readWrite", + Storage: multigresv1alpha1.StorageSpec{Size: "10Gi", Class: poolClass}, }, - } - sc := &storagev1.StorageClass{ObjectMeta: metav1.ObjectMeta{Name: "fast-ssd"}} + }, + }, + } +} +func childExists(t *testing.T, c client.Client, obj client.Object, namespace, name string) bool { + t.Helper() + err := c.Get(t.Context(), client.ObjectKey{Namespace: namespace, Name: name}, obj) + if err == nil { + return true + } + if !apierrors.IsNotFound(err) { + t.Fatalf("unexpected error reading %T %s: %v", obj, name, err) + } + return false +} + +// TestReconcile_MissingBackupStorageClassStopsBeforeTheSharedBackupPVC and its +// pool twin below pin where the two gates sit, not just that they fire. The +// returned RequeueAfter is the same wherever a gate is placed, so the only +// observable that moves when a gate moves is which children the pass created: +// a missing backup class must stop before the shared backup PVC is applied, and +// a missing pool class must stop after it and before anything that consumes +// pool storage. +func TestReconcile_MissingBackupStorageClassStopsBeforeTheSharedBackupPVC(t *testing.T) { + scheme := storageClassGateScheme() + shard := storageClassGateShard("missing-backup-sc", "") + + c := fake.NewClientBuilder(). + WithScheme(scheme). + WithObjects(shard, testPostgresPasswordSecretForShard(shard)). + WithStatusSubresource(&multigresv1alpha1.Shard{}). + Build() + + r := &ShardReconciler{ + Client: c, + Scheme: scheme, + Recorder: record.NewFakeRecorder(100), + APIReader: c, + CreateTopoStore: newMemoryTopoFactory(), + } + + result, err := r.Reconcile(t.Context(), ctrl.Request{ + NamespacedName: client.ObjectKeyFromObject(shard), + }) + if err != nil { + t.Fatalf("expected non-error dependency requeue, got error: %v", err) + } + if result.RequeueAfter != storageClassDependencyRequeue { + t.Fatalf("requeueAfter = %v, want %v", result.RequeueAfter, storageClassDependencyRequeue) + } + + ns := shard.Namespace + if !childExists(t, c, &corev1.ConfigMap{}, ns, PgHbaConfigMapName(shard.Name)) { + t.Error("pg_hba ConfigMap is missing: the backup gate moved above the shared ConfigMaps") + } + if childExists(t, c, &appsv1.Deployment{}, ns, buildHashedMultiorchName(shard, "zone1")) { + t.Error("Multiorch Deployment was created: the backup gate moved below the Multiorch block") + } + if childExists(t, c, &corev1.PersistentVolumeClaim{}, ns, BuildSharedBackupPVCName(shard)) { + t.Error( + "shared backup PVC was created against a StorageClass that does not exist: " + + "the backup gate moved below the backup PVC block", + ) + } + if childExists(t, c, &corev1.ConfigMap{}, ns, PostgresConfigMapName(shard.Name)) { + t.Error("postgres config ConfigMap was created: the reconcile ran past both gates") + } +} + +func TestReconcile_MissingPoolStorageClassStopsAfterTheSharedBackupPVC(t *testing.T) { + scheme := storageClassGateScheme() + shard := storageClassGateShard("backup-sc", "missing-pool-sc") + + c := fake.NewClientBuilder(). + WithScheme(scheme). + WithObjects( + shard, + testPostgresPasswordSecretForShard(shard), + &storagev1.StorageClass{ObjectMeta: metav1.ObjectMeta{Name: "backup-sc"}}, + ). + WithStatusSubresource(&multigresv1alpha1.Shard{}). + Build() + + r := &ShardReconciler{ + Client: c, + Scheme: scheme, + Recorder: record.NewFakeRecorder(100), + APIReader: c, + CreateTopoStore: newMemoryTopoFactory(), + } + + result, err := r.Reconcile(t.Context(), ctrl.Request{ + NamespacedName: client.ObjectKeyFromObject(shard), + }) + if err != nil { + t.Fatalf("expected non-error dependency requeue, got error: %v", err) + } + if result.RequeueAfter != storageClassDependencyRequeue { + t.Fatalf("requeueAfter = %v, want %v", result.RequeueAfter, storageClassDependencyRequeue) + } + + ns := shard.Namespace + if !childExists(t, c, &appsv1.Deployment{}, ns, buildHashedMultiorchName(shard, "zone1")) { + t.Error("Multiorch Deployment is missing: the pool gate moved above the Multiorch block") + } + if !childExists(t, c, &corev1.PersistentVolumeClaim{}, ns, BuildSharedBackupPVCName(shard)) { + t.Error( + "shared backup PVC is missing: a missing pool class must not stop the reconcile " + + "before the backup PVC, whose own StorageClass is present", + ) + } + if childExists(t, c, &corev1.ConfigMap{}, ns, PostgresConfigMapName(shard.Name)) { + t.Error("postgres config ConfigMap was created: the pool gate moved below it") + } + if childExists(t, c, &corev1.Pod{}, ns, BuildPoolPodName(shard, "primary", "zone1", 0)) { + t.Error( + "pool pod was created against a StorageClass that does not exist: " + + "the pool gate no longer precedes the workloads that consume pool storage", + ) + } +} + +// persistStorageClassCondition writes a prior StorageClassValid verdict so a +// test can drive setStorageClassCondition against known persisted state. The +// generation is read back rather than assumed, because the skip compares the +// condition's observedGeneration against the object's. +func persistStorageClassCondition( + t *testing.T, + c client.Client, + key client.ObjectKey, + build func(generation int64) metav1.Condition, +) { + t.Helper() + + var shard multigresv1alpha1.Shard + if err := c.Get(t.Context(), key, &shard); err != nil { + t.Fatalf("read shard: %v", err) + } + shard.Status.Conditions = []metav1.Condition{build(shard.Generation)} + if err := c.Status().Update(t.Context(), &shard); err != nil { + t.Fatalf("seed condition: %v", err) + } +} + +// TestSetStorageClassCondition_RepublishesWhenOneComparedFieldDiffers takes the +// skip-if-unchanged test apart conjunct by conjunct. Dropping any one of them +// widens the skip so it swallows a real change, and a case where several fields +// differ at once cannot see that, so each case here differs from the persisted +// condition in exactly one compared field. +// +// Some of these field combinations are not reachable verdicts: no production +// verdict pairs StorageClassFound with False. The unit under test is +// setStorageClassCondition, whose contract is per-field, and driving it with a +// storageClassCheck directly is what makes one field at a time possible. +func TestSetStorageClassCondition_RepublishesWhenOneComparedFieldDiffers(t *testing.T) { + t.Parallel() + + const foundMessage = "All explicitly configured StorageClasses are present" + settled := storageClassCheck{ + status: metav1.ConditionTrue, + reason: storageClassFoundReason, + message: foundMessage, + } + + cases := []struct { + name string + persisted func(generation int64) metav1.Condition + wantPatch bool + wantReason string + }{ + { + name: "status differs", + persisted: func(generation int64) metav1.Condition { + return metav1.Condition{ + Type: conditionStorageClassValid, + Status: metav1.ConditionFalse, + Reason: storageClassFoundReason, + Message: foundMessage, + ObservedGeneration: generation, + LastTransitionTime: metav1.Now(), + } + }, + wantPatch: true, + }, + { + name: "reason differs", + persisted: func(generation int64) metav1.Condition { + return metav1.Condition{ + Type: conditionStorageClassValid, + Status: metav1.ConditionTrue, + Reason: storageClassNotSpecifiedReason, + Message: foundMessage, + ObservedGeneration: generation, + LastTransitionTime: metav1.Now(), + } + }, + wantPatch: true, + }, + { + name: "message differs", + persisted: func(generation int64) metav1.Condition { + return metav1.Condition{ + Type: conditionStorageClassValid, + Status: metav1.ConditionTrue, + Reason: storageClassFoundReason, + Message: "a message from an older build", + ObservedGeneration: generation, + LastTransitionTime: metav1.Now(), + } + }, + wantPatch: true, + }, + { + name: "observedGeneration differs", + persisted: func(generation int64) metav1.Condition { + return metav1.Condition{ + Type: conditionStorageClassValid, + Status: metav1.ConditionTrue, + Reason: storageClassFoundReason, + Message: foundMessage, + ObservedGeneration: generation - 1, + LastTransitionTime: metav1.Now(), + } + }, + wantPatch: true, + }, + { + name: "nothing differs", + persisted: func(generation int64) metav1.Condition { + return metav1.Condition{ + Type: conditionStorageClassValid, + Status: metav1.ConditionTrue, + Reason: storageClassFoundReason, + Message: foundMessage, + ObservedGeneration: generation, + LastTransitionTime: metav1.Now(), + } + }, + wantPatch: false, + }, + } + + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + t.Parallel() + + scheme := runtime.NewScheme() + _ = multigresv1alpha1.AddToScheme(scheme) + _ = storagev1.AddToScheme(scheme) + + shard := &multigresv1alpha1.Shard{ + ObjectMeta: metav1.ObjectMeta{Name: "test-shard", Namespace: "default"}, + } baseClient := fake.NewClientBuilder(). WithScheme(scheme). - WithObjects(shard, sc). + WithObjects(shard). WithStatusSubresource(&multigresv1alpha1.Shard{}). Build() - var capturedPatchObj client.Object - fakeClient := testutil.NewFakeClientWithFailures(baseClient, &testutil.FailureConfig{ - OnStatusPatch: func(obj client.Object) error { - capturedPatchObj = obj - return nil - }, - }) + key := client.ObjectKeyFromObject(shard) + persistStorageClassCondition(t, baseClient, key, tc.persisted) + patches := 0 + fakeClient := testutil.NewFakeClientWithFailures( + baseClient, + &testutil.FailureConfig{ + OnStatusPatch: func(client.Object) error { + patches++ + return nil + }, + }, + ) r := &ShardReconciler{ Client: fakeClient, Scheme: scheme, Recorder: record.NewFakeRecorder(10), } - if err := r.validatePoolStorageClassDependencies(t.Context(), shard); err != nil { - t.Fatalf("guard: %v", err) + if err := r.setStorageClassCondition(t.Context(), shard, settled); err != nil { + t.Fatalf("setStorageClassCondition: %v", err) } - patchShard, ok := capturedPatchObj.(*multigresv1alpha1.Shard) - if !ok { - t.Fatalf("expected *Shard patch, got %T", capturedPatchObj) + want := 0 + if tc.wantPatch { + want = 1 } - - // Exactly one condition: StorageClassValid. - if len(patchShard.Status.Conditions) != 1 { - t.Fatalf("guard patch must contain exactly 1 condition, got %d: %v", - len(patchShard.Status.Conditions), patchShard.Status.Conditions) - } - scCond := &patchShard.Status.Conditions[0] - if scCond.Type != conditionStorageClassValid { - t.Fatalf("expected %s condition, got %s", conditionStorageClassValid, scCond.Type) + if patches != want { + t.Fatalf("applied %d patches, want %d", patches, want) } - if scCond.Status != metav1.ConditionTrue || scCond.Reason != storageClassFoundReason { - t.Fatalf("unexpected condition: status=%s reason=%s", scCond.Status, scCond.Reason) + if !tc.wantPatch { + return } - // No other status fields should be set in the guard patch. - if patchShard.Status.Phase != "" { - t.Fatalf("guard patch must not set Phase, got %q", patchShard.Status.Phase) - } - if patchShard.Status.Message != "" { - t.Fatalf("guard patch must not set Message, got %q", patchShard.Status.Message) + var got multigresv1alpha1.Shard + if err := baseClient.Get(t.Context(), key, &got); err != nil { + t.Fatalf("read shard: %v", err) } - if patchShard.Status.PodRoles != nil { - t.Fatal("guard patch must not set PodRoles") + cond := findCondition(got.Status.Conditions, conditionStorageClassValid) + if cond == nil { + t.Fatalf("no %s condition", conditionStorageClassValid) } - if patchShard.Status.ReadyReplicas != 0 { - t.Fatalf( - "guard patch must not set ReadyReplicas, got %d", - patchShard.Status.ReadyReplicas, - ) + if cond.Status != settled.status || cond.Reason != settled.reason || + cond.Message != settled.message || cond.ObservedGeneration != got.Generation { + t.Fatalf("published condition does not match the verdict: %+v", *cond) } + }) + } +} + +// TestStorageClassCondition_UpdatesWhenTheVerdictChanges is the other half of +// TestStorageClassCondition_IsStableAcrossReconciles: the skip has to hold a +// settled condition still, and it has to let a changed verdict through. The +// missing StorageClass appears between the two cycles, so status, reason, +// message and lastTransitionTime all have to move with it. +func TestStorageClassCondition_UpdatesWhenTheVerdictChanges(t *testing.T) { + t.Parallel() + + scheme := runtime.NewScheme() + _ = multigresv1alpha1.AddToScheme(scheme) + _ = storagev1.AddToScheme(scheme) + + shard := &multigresv1alpha1.Shard{ + ObjectMeta: metav1.ObjectMeta{Name: "test-shard", Namespace: "default"}, + Spec: multigresv1alpha1.ShardSpec{ + Pools: map[multigresv1alpha1.PoolName]multigresv1alpha1.PoolSpec{ + "primary": { + Storage: multigresv1alpha1.StorageSpec{Size: "10Gi", Class: "appears-later"}, + }, + }, + }, + } + + baseClient := fake.NewClientBuilder(). + WithScheme(scheme). + WithObjects(shard). + WithStatusSubresource(&multigresv1alpha1.Shard{}). + Build() + + patches := 0 + fakeClient := testutil.NewFakeClientWithFailures(baseClient, &testutil.FailureConfig{ + OnStatusPatch: func(client.Object) error { + patches++ + return nil + }, + }) + r := &ShardReconciler{Client: fakeClient, Scheme: scheme, Recorder: record.NewFakeRecorder(50)} + + key := client.ObjectKeyFromObject(shard) + cycle := func() metav1.Condition { + t.Helper() + + check, err := r.validateStorageClassDependencies(t.Context(), shard) + if err != nil { + t.Fatalf("validate: %v", err) + } + if err := r.setStorageClassCondition(t.Context(), shard, check); err != nil { + t.Fatalf("set condition: %v", err) + } + + var got multigresv1alpha1.Shard + if err := baseClient.Get(t.Context(), key, &got); err != nil { + t.Fatalf("read shard: %v", err) + } + cond := findCondition(got.Status.Conditions, conditionStorageClassValid) + if cond == nil { + t.Fatalf("no %s condition", conditionStorageClassValid) + } + return *cond + } + + before := cycle() + if before.Status != metav1.ConditionFalse || before.Reason != storageClassNotFoundReason { + t.Fatalf("first cycle must report the missing class: %+v", before) + } + if patches != 1 { + t.Fatalf("first cycle applied %d patches, want 1", patches) + } + + // Backdated because metav1.Time serialises at second precision and both + // cycles run inside the same second, which would make a rewritten + // lastTransitionTime indistinguishable from a preserved one. + backdated := metav1.NewTime(time.Now().Add(-time.Hour).Truncate(time.Second)) + persistStorageClassCondition(t, baseClient, key, func(generation int64) metav1.Condition { + return metav1.Condition{ + Type: before.Type, + Status: before.Status, + Reason: before.Reason, + Message: before.Message, + ObservedGeneration: generation, + LastTransitionTime: backdated, + } + }) + + if err := baseClient.Create(t.Context(), &storagev1.StorageClass{ + ObjectMeta: metav1.ObjectMeta{Name: "appears-later"}, + }); err != nil { + t.Fatalf("create StorageClass: %v", err) + } + + after := cycle() + if patches != 2 { + t.Fatalf("the changed verdict was skipped: %d patches total", patches) + } + if after.Status != metav1.ConditionTrue { + t.Errorf("status = %s, want %s", after.Status, metav1.ConditionTrue) + } + if after.Reason != storageClassFoundReason { + t.Errorf("reason = %s, want %s", after.Reason, storageClassFoundReason) + } + if after.Message == before.Message { + t.Errorf("message did not move off the missing-class text: %q", after.Message) + } + if want := "All explicitly configured StorageClasses are present"; after.Message != want { + t.Errorf("message = %q, want %q", after.Message, want) + } + if !after.LastTransitionTime.After(backdated.Time) { + t.Errorf( + "lastTransitionTime = %s, want it moved past %s: the condition transitioned", + after.LastTransitionTime, + backdated, + ) + } +} + +// TestStorageClassCondition_PreservesLastTransitionTimeWithoutATransition +// covers the republish an upgrade from the two-writer build performs: the +// persisted condition still carries one of the two old messages, the verdict is +// the same True it always was, so the message has to be rewritten while +// lastTransitionTime stays put, matching meta.SetStatusCondition. +func TestStorageClassCondition_PreservesLastTransitionTimeWithoutATransition(t *testing.T) { + t.Parallel() + + scheme := runtime.NewScheme() + _ = multigresv1alpha1.AddToScheme(scheme) + _ = storagev1.AddToScheme(scheme) + + shard := &multigresv1alpha1.Shard{ + ObjectMeta: metav1.ObjectMeta{Name: "test-shard", Namespace: "default"}, + Spec: multigresv1alpha1.ShardSpec{ + Pools: map[multigresv1alpha1.PoolName]multigresv1alpha1.PoolSpec{ + "primary": {Storage: multigresv1alpha1.StorageSpec{Size: "10Gi"}}, + }, }, - ) + } + + baseClient := fake.NewClientBuilder(). + WithScheme(scheme). + WithObjects(shard). + WithStatusSubresource(&multigresv1alpha1.Shard{}). + Build() + + key := client.ObjectKeyFromObject(shard) + staleMessage := "No explicit pool StorageClass configured; using cluster default" + backdated := metav1.NewTime(time.Now().Add(-time.Hour).Truncate(time.Second)) + persistStorageClassCondition(t, baseClient, key, func(generation int64) metav1.Condition { + return metav1.Condition{ + Type: conditionStorageClassValid, + Status: metav1.ConditionTrue, + Reason: storageClassNotSpecifiedReason, + Message: staleMessage, + ObservedGeneration: generation, + LastTransitionTime: backdated, + } + }) + + patches := 0 + fakeClient := testutil.NewFakeClientWithFailures(baseClient, &testutil.FailureConfig{ + OnStatusPatch: func(client.Object) error { + patches++ + return nil + }, + }) + r := &ShardReconciler{Client: fakeClient, Scheme: scheme, Recorder: record.NewFakeRecorder(50)} + + check, err := r.validateStorageClassDependencies(t.Context(), shard) + if err != nil { + t.Fatalf("validate: %v", err) + } + if err := r.setStorageClassCondition(t.Context(), shard, check); err != nil { + t.Fatalf("set condition: %v", err) + } + if patches != 1 { + t.Fatalf("the stale message was not republished: %d patches", patches) + } + + var got multigresv1alpha1.Shard + if err := baseClient.Get(t.Context(), key, &got); err != nil { + t.Fatalf("read shard: %v", err) + } + cond := findCondition(got.Status.Conditions, conditionStorageClassValid) + if cond == nil { + t.Fatalf("no %s condition", conditionStorageClassValid) + } + if cond.Message == staleMessage { + t.Fatalf("message was not rewritten: %q", cond.Message) + } + if cond.Status != metav1.ConditionTrue { + t.Fatalf("status = %s, want %s", cond.Status, metav1.ConditionTrue) + } + if !cond.LastTransitionTime.Time.Equal(backdated.Time) { + t.Errorf( + "lastTransitionTime = %s, want it preserved at %s: the status did not transition", + cond.LastTransitionTime, + backdated, + ) + } } From 325807d1e4962823c70ebc45b8123a847591cd18 Mon Sep 17 00:00:00 2001 From: Brent Graveland Date: Sat, 19 Sep 2026 19:03:55 -0600 Subject: [PATCH 2/7] test(suite): add the scenario tests The operator half of the multi-controller suite: the fakes, the cluster fixture, shard identity helpers, and the scenario tests themselves, written as a consumer of github.com/multigres/testkit/ctrltest. A scenario test exercises a joint between controllers, which is what the existing tier cannot reach. The deletion protocol runs shard to tablegroup to multigrescluster; the fan-out is asserted as a sequence rather than an end state; the race test pins two reconcilers writing one object. Others cover shard lifecycle, selector impostors, transitions and round-trip equality, and thrash. Seven live operator defects are pinned with KnownDefect, so each one reproduces its own evidence on every run rather than only in a document. A pin passes while its defect is present and fails the day it is fixed, so the fix has to replace the pin with a positive assertion in the same change. The pool scale-up pin constructs its race rather than sampling it. The defect needs the new pooler to register after the shard has converged, which happens naturally about a third of the time. poolerSim can hold new registrations for one namespace, so the test holds, scales up, waits for two pool pods and one entry in status.podRoles to stay stable, then releases. The pooler then appears in etcd with no Kubernetes event to announce it. That precondition is deliberately a stability window rather than RequireQuiescent: once the defect is fixed, the shard requeues while the pooler is held and the namespace never goes quiet. Tests open with newCase(t), which allocates the namespace, registers it at the reconcile gate and attaches the failure dump, and everything hangs off that receiver: the assertions, the harness, the client verbs, and this package's own vocabulary. Sub(t) is the subtest form, keeping the namespace while binding the subtest T. Bare(t) is for the pure-logic tests that never touch the cluster. Tests scope their work to the case namespace throughout. envtest never really deletes a namespace, so an unscoped List would see everything every earlier test in the run created. Signed-off-by: Brent Graveland --- Makefile | 22 +- go.mod | 11 +- go.sum | 10 +- test/suite/case.go | 68 ++ test/suite/fakes.go | 265 ++++++ test/suite/fixture.go | 113 +++ test/suite/golden_test.go | 67 ++ test/suite/identity.go | 98 ++ test/suite/identity_test.go | 239 +++++ test/suite/main_test.go | 66 ++ test/suite/scenario_deletion_test.go | 295 ++++++ test/suite/scenario_fanout_test.go | 159 ++++ test/suite/scenario_race_test.go | 335 +++++++ test/suite/scenario_selector_impostor_test.go | 643 +++++++++++++ test/suite/scenario_shard_lifecycle_test.go | 896 ++++++++++++++++++ test/suite/scenario_shard_quiescence_test.go | 26 + test/suite/scenario_thrash_test.go | 416 ++++++++ test/suite/scenario_transitions_test.go | 466 +++++++++ test/suite/shard_requeue_test.go | 183 ++++ test/suite/suite.go | 223 +++++ test/suite/suite_test.go | 123 +++ .../multigateway-deployment.golden.yaml | 89 ++ test/suite/types.go | 48 + 23 files changed, 4849 insertions(+), 12 deletions(-) create mode 100644 test/suite/case.go create mode 100644 test/suite/fakes.go create mode 100644 test/suite/fixture.go create mode 100644 test/suite/golden_test.go create mode 100644 test/suite/identity.go create mode 100644 test/suite/identity_test.go create mode 100644 test/suite/main_test.go create mode 100644 test/suite/scenario_deletion_test.go create mode 100644 test/suite/scenario_fanout_test.go create mode 100644 test/suite/scenario_race_test.go create mode 100644 test/suite/scenario_selector_impostor_test.go create mode 100644 test/suite/scenario_shard_lifecycle_test.go create mode 100644 test/suite/scenario_shard_quiescence_test.go create mode 100644 test/suite/scenario_thrash_test.go create mode 100644 test/suite/scenario_transitions_test.go create mode 100644 test/suite/shard_requeue_test.go create mode 100644 test/suite/suite.go create mode 100644 test/suite/suite_test.go create mode 100644 test/suite/testdata/multigateway-deployment.golden.yaml create mode 100644 test/suite/types.go diff --git a/Makefile b/Makefile index 235d2b31..9aef3644 100644 --- a/Makefile +++ b/Makefile @@ -278,22 +278,38 @@ build-installer: manifests generate kustomize ## Generate consolidated install Y ##@ Test +# test/suite is the multi-controller envtest suite. It carries no build tag, so +# every `go test ./...` call site has to exclude it by path or it lands in the +# required check before it is ready. That is one filter per call site, which is +# the deliberate trade against a tag that someone forgets on a new file. +# +# -v is load-bearing rather than cosmetic. Each KnownDefect pin logs the defect +# it is standing on while that defect is still present, and without -v go test +# discards the output of a passing test, so a green CI run shows none of them. +# The suite is meant to be readable as the operator's live defect list, and -v +# is what makes that list visible without waiting for a pin to expire. +.PHONY: test-suite +test-suite: manifests generate fmt vet setup-envtest ## Run the multi-controller test suite + KUBEBUILDER_ASSETS="$(shell $(ENVTEST) use $(ENVTEST_K8S_VERSION) --bin-dir $(LOCALBIN) -p path)" \ + go test -v -p 1 -timeout 20m ./test/suite/... + .PHONY: test test: manifests generate fmt vet ## Run tests (no integration testing) KUBEBUILDER_ASSETS="$(shell $(ENVTEST) use $(ENVTEST_K8S_VERSION) --bin-dir $(LOCALBIN) -p path)" \ - go test -p 1 $$(go list ./... | grep -v /e2e) -coverprofile=cover.out + go test -p 1 $$(go list ./... | grep -v /e2e | grep -v /test/suite) -coverprofile=cover.out .PHONY: test-integration test-integration: manifests generate fmt vet setup-envtest ## Run integration tests KUBEBUILDER_ASSETS="$(shell $(ENVTEST) use $(ENVTEST_K8S_VERSION) --bin-dir $(LOCALBIN) -p path)" \ - go test -p 1 -tags=integration,verbose $$(go list ./... | grep -v /e2e) -coverprofile=cover.out + go test -p 1 -tags=integration,verbose $$(go list ./... | grep -v /e2e | grep -v /test/suite) -coverprofile=cover.out .PHONY: test-coverage test-coverage: manifests generate fmt vet setup-envtest ## Generate coverage report with HTML @mkdir -p coverage @echo "==> Generating coverage..." KUBEBUILDER_ASSETS="$(shell $(ENVTEST) use $(ENVTEST_K8S_VERSION) --bin-dir $(LOCALBIN) -p path)" \ - go test -p 1 -tags=integration,verbose ./... -coverprofile=coverage/combined.out -covermode=atomic + go test -p 1 -tags=integration,verbose $$(go list ./... | grep -v /e2e | grep -v /test/suite) \ + -coverprofile=coverage/combined.out -covermode=atomic @echo "==> Generating HTML report..." @go tool cover -html=coverage/combined.out -o=coverage/combined.html @echo "Generated: coverage/combined.html" diff --git a/go.mod b/go.mod index afd81bb4..ea64cd6e 100644 --- a/go.mod +++ b/go.mod @@ -1,11 +1,12 @@ module github.com/multigres/multigres-operator -go 1.26.6 +go 1.27 require ( github.com/go-logr/logr v1.4.4 github.com/google/go-cmp v0.7.0 github.com/multigres/multigres v0.0.0-20260925193740-522b90425a83 + github.com/multigres/testkit v0.2.1 github.com/prometheus/client_golang v1.24.1 github.com/prometheus/client_model v0.6.3 github.com/stretchr/testify v1.12.1 @@ -15,14 +16,16 @@ require ( go.opentelemetry.io/otel v1.46.0 go.opentelemetry.io/otel/sdk v1.46.0 go.opentelemetry.io/otel/trace v1.46.0 + go.uber.org/goleak v1.3.0 google.golang.org/grpc v1.83.2 google.golang.org/protobuf v1.36.12 k8s.io/api v0.37.0 k8s.io/apimachinery v0.37.0 k8s.io/client-go v0.37.0 - k8s.io/utils v0.0.0-20260626114624-be93311217bd - sigs.k8s.io/controller-runtime v0.25.0 + k8s.io/utils v0.0.0-20260707023825-cf1189d6abe3 + sigs.k8s.io/controller-runtime v0.25.1 sigs.k8s.io/e2e-framework v0.7.0 + sigs.k8s.io/structured-merge-diff/v6 v6.4.2 ) require ( @@ -128,7 +131,6 @@ require ( go.opentelemetry.io/otel/sdk/log v0.22.0 // indirect go.opentelemetry.io/otel/sdk/metric v1.46.0 // indirect go.opentelemetry.io/proto/otlp v1.11.0 // indirect - go.uber.org/goleak v1.3.0 // indirect go.uber.org/multierr v1.11.0 // indirect go.uber.org/zap v1.27.1 // indirect go.yaml.in/yaml/v2 v2.4.4 // indirect @@ -157,7 +159,6 @@ require ( sigs.k8s.io/apiserver-network-proxy/konnectivity-client v0.36.0 // indirect sigs.k8s.io/json v0.0.0-20250730193827-2d320260d730 // indirect sigs.k8s.io/randfill v1.0.0 // indirect - sigs.k8s.io/structured-merge-diff/v6 v6.4.2 // indirect sigs.k8s.io/yaml v1.6.0 // indirect ) diff --git a/go.sum b/go.sum index f263bbdf..ad8e666b 100644 --- a/go.sum +++ b/go.sum @@ -199,6 +199,8 @@ github.com/modern-go/reflect2 v1.0.3-0.20250322232337-35a7c28c31ee h1:W5t00kpgFd github.com/modern-go/reflect2 v1.0.3-0.20250322232337-35a7c28c31ee/go.mod h1:yWuevngMOJpCy52FWWMvUC8ws7m/LJsjYzDa0/r8luk= github.com/multigres/multigres v0.0.0-20260925193740-522b90425a83 h1:IdyFGtwc9pEZEzqs6h5JfavAnErWfuKwj3d42qkZUx8= github.com/multigres/multigres v0.0.0-20260925193740-522b90425a83/go.mod h1:Ov2hrkOguWSkCS2QIhAdguFeG5GlZ3v4WGqIdqkQ7Tg= +github.com/multigres/testkit v0.2.1 h1:1SOV2jevblZBpobzaoFAy/lVqb2Donihjc+EovmbIAo= +github.com/multigres/testkit v0.2.1/go.mod h1:3ONhsV/PNOUke7PID5HPlnxTLyQCcfOQ/JfLCRkSLOY= github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822 h1:C3w9PqII01/Oq1c1nUAm88MOHcQC9l5mIlSMApZMrHA= github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822/go.mod h1:+n7T8mK8HuQTcFwEeznm/DIxMOiR9yIdICNftLE1DvQ= github.com/onsi/ginkgo/v2 v2.27.4 h1:fcEcQW/A++6aZAZQNUmNjvA9PSOzefMJBerHJ4t8v8Y= @@ -457,12 +459,12 @@ k8s.io/kube-openapi v0.0.0-20260721132016-d427ff9ee9ad h1:oXImqH8mQNk7PmvzKhmN3d k8s.io/kube-openapi v0.0.0-20260721132016-d427ff9ee9ad/go.mod h1:0/mqHCVhlumdJ3BhCfnjSZQE037nAhNodh1/hK0T8/I= k8s.io/streaming v0.37.0 h1:iPBUZLZiKt5bV+lxJurASMOV07VuBhNpiwJt2//AWrM= k8s.io/streaming v0.37.0/go.mod h1:APlJR26ZWRcVy5bIEj0QRrKUXROtBHPcxl2NT7EAzPU= -k8s.io/utils v0.0.0-20260626114624-be93311217bd h1:Ea7fgQ5we8Y9T0OX5o0dAHzQOBRI07D/dEYRaB9ZZEs= -k8s.io/utils v0.0.0-20260626114624-be93311217bd/go.mod h1:xDxuJ0whA3d0I4mf/C4ppKHxXynQ+fxnkmQH0vTHnuk= +k8s.io/utils v0.0.0-20260707023825-cf1189d6abe3 h1:jVkFFVfXdXP74B/zbO3hM3hpSFD0xvhQ5U686DPurkE= +k8s.io/utils v0.0.0-20260707023825-cf1189d6abe3/go.mod h1:M2s5JB1lIYP3jzZdorPLHXIPJzt9vv2muW5a6L9DtNM= sigs.k8s.io/apiserver-network-proxy/konnectivity-client v0.36.0 h1:/YpDJ4vReG7ZmzSpBGxduXgywWkJU9zHubgJG03MT+Y= sigs.k8s.io/apiserver-network-proxy/konnectivity-client v0.36.0/go.mod h1:tJo1aepTXyR+8Xs3sUsGBDk4Ub2AM5dPAPKJx0mpm5c= -sigs.k8s.io/controller-runtime v0.25.0 h1:44KgRUPew331KSJpNu8zJow3iTR5W0p/SfrHdw3lV40= -sigs.k8s.io/controller-runtime v0.25.0/go.mod h1:4QqLdT6z/L6Olj8JJCtvztid4/fnIiYsfaTFScegctc= +sigs.k8s.io/controller-runtime v0.25.1 h1:BKgU9OeE8xv8EbbM8cY0NVzTQs35rokkdq1jh12fMb4= +sigs.k8s.io/controller-runtime v0.25.1/go.mod h1:4QqLdT6z/L6Olj8JJCtvztid4/fnIiYsfaTFScegctc= sigs.k8s.io/e2e-framework v0.7.0 h1:AHkySTC6MvnnMbVSxaO4z1m2MhQKNFP+2Ihs5pRNLlM= sigs.k8s.io/e2e-framework v0.7.0/go.mod h1:1ZgXkUSjmnf18/JgHZNEATWjv48O5lJm9aI1QIsRdbw= sigs.k8s.io/json v0.0.0-20250730193827-2d320260d730 h1:IpInykpT6ceI+QxKBbEflcR5EXP7sU1kvOlxwZh5txg= diff --git a/test/suite/case.go b/test/suite/case.go new file mode 100644 index 00000000..f43e5091 --- /dev/null +++ b/test/suite/case.go @@ -0,0 +1,68 @@ +package suite + +import ( + "testing" + + "github.com/multigres/testkit/ctrltest" +) + +// C is this operator's test context: the generic harness handle from +// ctrltest, plus the multigres vocabulary. +// +// A local type because Go cannot add methods to another package's, which is +// the point rather than a workaround. pkg/ctrltest is meant to be copied into +// other operators, so it carries assertions and harness pointers and knows +// nothing about shards or poolers; each consumer wraps it and hangs its own +// domain on the same receiver. Everything ctrltest offers is promoted, so +// c.NoError and c.WaitForClusterHealthy read alike at the call site. +type C struct { + *ctrltest.C +} + +// newCase opens a test context on its own namespace. +// +// Every test in this package should start with one. It allocates the +// namespace, activates the reconcile gate for it, and registers the failure +// dump, so a failing test prints the interleaved op log and reconcile records +// rather than only the assertion message. +func newCase(t *testing.T) *C { + t.Helper() + return &C{C: Suite.Case(t)} +} + +// newBareCase opens a test context with no namespace, for the tests in this +// package that are pure logic and never touch the cluster. +// +// identity_test.go is all of them: MembersOf and ShardPVCOf take objects and +// return answers. newCase would allocate a real namespace against envtest and +// register it at the reconcile gate for each one, which buys nothing. +func newBareCase(t *testing.T) *C { + t.Helper() + return &C{C: ctrltest.Bare(t)} +} + +// Sub binds this case to a subtest's T while keeping its namespace. Use it +// for a t.Run that asserts about objects the parent test created; use newCase +// for a subtest that wants a namespace of its own. +// +// It shadows the embedded ctrltest.C.Sub so that one name always hands back +// this package's C, with the multigres vocabulary still on it. Without the +// shadow a subtest would silently drop to the generic type and lose every +// method below. +func (c *C) Sub(t *testing.T) *C { + t.Helper() + return &C{C: c.C.Sub(t)} +} + +// Check returns a C whose assertions report and continue rather than abort, +// shadowed for the same reason as Sub: without it c.Check() hands back a +// *ctrltest.C and a collecting assertion silently loses every method below. +func (c *C) Check() *C { + return &C{C: c.C.Check()} +} + +// Assert at compile time that both shadows hand back this package's type. The +// regression they guard against is silent: dropping to *ctrltest.C still +// compiles at every existing call site, and only stops compiling once someone +// chains a multigres method off one of them. +var _ = func(c *C) (*C, *C) { return c.Check(), c.Sub(nil) } diff --git a/test/suite/fakes.go b/test/suite/fakes.go new file mode 100644 index 00000000..6b6691ff --- /dev/null +++ b/test/suite/fakes.go @@ -0,0 +1,265 @@ +package suite + +import ( + "context" + "fmt" + "sort" + "strings" + "sync" + "time" + + "github.com/multigres/multigres/go/common/rpcclient" + "github.com/multigres/multigres/go/common/topoclient" + "github.com/multigres/multigres/go/common/topoclient/memorytopo" + cm "github.com/multigres/multigres/go/pb/clustermetadata" + md "github.com/multigres/multigres/go/pb/multipoolermanagerdata" + corev1 "k8s.io/api/core/v1" + "sigs.k8s.io/controller-runtime/pkg/client" + + multigresv1alpha1 "github.com/multigres/multigres-operator/api/v1alpha1" + shardcontroller "github.com/multigres/multigres-operator/pkg/resource-handler/controller/shard" + "github.com/multigres/multigres-operator/pkg/util/metadata" +) + +// defaultSimCell is the cell every fixture uses. The topo store needs its cells +// declared up front, so a test using a different cell name needs this widened. +const defaultSimCell = "zone-a" + +// topoRegistry hands out one in-memory topology store per namespace. +// +// Per namespace rather than one shared store, because namespace-per-test is the +// suite's isolation boundary and a single store would let one test's cluster +// see another's multipoolers. Both CreateTopoStore seams resolve through here: +// the shard's carries the object, the cluster's carries only a DNS address, so +// that one recovers the namespace by parsing it. +type topoRegistry struct { + ctx context.Context + + mu sync.Mutex + stores map[string]topoclient.Store + facts map[string]*memorytopo.Factory +} + +func newTopoRegistry(ctx context.Context) *topoRegistry { + return &topoRegistry{ + ctx: ctx, + stores: map[string]topoclient.Store{}, + facts: map[string]*memorytopo.Factory{}, + } +} + +// Store returns the namespace's store, creating it on first use. +func (r *topoRegistry) Store(ns string) topoclient.Store { + r.mu.Lock() + defer r.mu.Unlock() + if s, ok := r.stores[ns]; ok { + return s + } + store, factory := memorytopo.NewServerAndFactory(r.ctx, defaultSimCell) + r.stores[ns] = store + r.facts[ns] = factory + return store +} + +func (r *topoRegistry) client(ns string) (topoclient.Store, error) { + r.Store(ns) + r.mu.Lock() + factory := r.facts[ns] + r.mu.Unlock() + return topoclient.NewWithFactory( + factory, "", []string{""}, topoclient.NewDefaultTopoConfig(), + ), nil +} + +// ForShard is ShardReconciler.CreateTopoStore. +func (r *topoRegistry) ForShard(shard *Shard) (topoclient.Store, error) { + return r.client(shard.Namespace) +} + +// ForClusterRef is MultigresClusterReconciler.CreateTopoStore. The ref carries +// no namespace, only the Service address the cluster controller built as +// "-global-topo..svc:2379", so the namespace comes back out +// of the address. Left unstubbed, this seam dials a real etcd and the cluster +// never reaches TopologyReady. +func (r *topoRegistry) ForClusterRef( + ref multigresv1alpha1.GlobalTopoServerRef, +) (topoclient.Store, error) { + ns, err := namespaceFromTopoAddress(ref.Address) + if err != nil { + return nil, err + } + return r.client(ns) +} + +func namespaceFromTopoAddress(address string) (string, error) { + parts := strings.Split(address, ".") + if len(parts) < 2 || parts[1] == "" { + return "", fmt.Errorf("cannot derive namespace from topo address %q", address) + } + return parts[1], nil +} + +// poolerSim is the data plane: it registers a multipooler per pool pod in that +// namespace's topology store and answers a healthy Status RPC for each. +// +// Without it the shard controller stalls at PostureConsistent=Unknown +// (AwaitingPoolerRegistration) and never reaches a terminal state. +// +// Deliberately generous: every pod is healthy, always, and the lowest-numbered +// pod is primary. It models no ordering, no failure, and no latency, so a test +// that needs any of those needs a better fake than this one. +type poolerSim struct { + c client.Client + rpc *rpcclient.FakeClient + topo *topoRegistry + interval time.Duration + + mu sync.Mutex + registered map[string]bool + held map[string]bool +} + +// HoldRegistrations stops this fake registering any *new* pooler in ns until +// the returned function is called. Poolers already registered keep answering. +// +// It exists to make a race deterministic instead of sampled. The defect it was +// built for needs a shard to converge having seen fewer poolers than pods, +// which happens on its own only when registration loses a race against the +// last reconcile: about half the time, measured. A test that waits for that by +// chance detects a regression about half the time too, which is what the +// twelve-attempt statistical pin it replaced was paying for. +// +// Holding lets a test construct the precondition on purpose: hold, scale up, +// wait until the shard has demonstrably converged short, then release. One +// attempt, and the regression either survives the release or it does not. +func (p *poolerSim) HoldRegistrations(ns string) func() { + p.mu.Lock() + if p.held == nil { + p.held = map[string]bool{} + } + p.held[ns] = true + p.mu.Unlock() + + var once sync.Once + return func() { + once.Do(func() { + p.mu.Lock() + delete(p.held, ns) + p.mu.Unlock() + }) + } +} + +func (p *poolerSim) run(ctx context.Context) { + p.registered = map[string]bool{} + t := time.NewTicker(p.interval) + defer t.Stop() + for { + select { + case <-ctx.Done(): + return + case <-t.C: + p.tick(ctx) + } + } +} + +func (p *poolerSim) tick(ctx context.Context) { + pods := &corev1.PodList{} + if err := p.c.List(ctx, pods, + client.MatchingLabels{metadata.LabelAppComponent: shardcontroller.PoolComponentName}, + ); err != nil { + return + } + byNamespace := map[string][]*corev1.Pod{} + for i := range pods.Items { + pod := &pods.Items[i] + if !pod.DeletionTimestamp.IsZero() { + continue + } + byNamespace[pod.Namespace] = append(byNamespace[pod.Namespace], pod) + } + for ns, group := range byNamespace { + p.tickNamespace(ctx, ns, group) + } +} + +func (p *poolerSim) tickNamespace(ctx context.Context, ns string, pods []*corev1.Pod) { + sort.Slice(pods, func(i, j int) bool { return pods[i].Name < pods[j].Name }) + + ids := make([]*cm.ID, 0, len(pods)) + for _, pod := range pods { + ids = append(ids, &cm.ID{ + Cell: pod.Labels[metadata.LabelMultigresCell], + Name: pod.Name, + }) + } + if len(ids) == 0 { + return + } + leader := ids[0] + rule := &cm.ShardRule{ + RuleNumber: &cm.RuleNumber{CoordinatorTerm: 2}, + LeaderId: leader, + CohortMembers: ids, + DurabilityPolicy: topoclient.AtLeastN(1), + } + store := p.topo.Store(ns) + + for i, pod := range pods { + id := ids[i] + role := cm.RoutingRole_ROUTING_ROLE_REPLICA + resp := &md.StatusResponse{ + Status: &md.Status{ + IsInitialized: true, + PostgresReady: true, + PostgresStatus: md.PostgresStatus_POSTGRES_STATUS_STANDBY, + }, + AvailabilityStatus: &cm.AvailabilityStatus{ + CohortEligibilityStatus: &cm.CohortEligibilityStatus{ + Signal: cm.CohortEligibilitySignal_COHORT_ELIGIBILITY_SIGNAL_ELIGIBLE, + }, + }, + ConsensusStatus: &cm.ConsensusStatus{ + Id: id, + CurrentPosition: &cm.PoolerPosition{Position: &cm.RulePosition{Decision: rule}}, + }, + } + if id.Name == leader.Name { + role = cm.RoutingRole_ROUTING_ROLE_PRIMARY + resp.Status.PostgresStatus = md.PostgresStatus_POSTGRES_STATUS_PRIMARY + resp.Status.PrimaryStatus = &md.PrimaryStatus{ + Ready: true, + ConnectedFollowers: ids[1:], + } + } + p.rpc.SetStatusResponse(topoclient.ComponentIDString(id), resp) + + key := ns + "/" + pod.Name + p.mu.Lock() + already := p.registered[key] + heldBack := p.held[ns] + p.mu.Unlock() + // A held namespace still gets its Status RPC answered above, so pods + // already registered stay healthy and the shard keeps converging. Only + // the new registration waits, which is the whole point. + if already || heldBack { + continue + } + pooler := &cm.Multipooler{ + Id: id, + Hostname: pod.Name, + ShardKey: &cm.ShardKey{ + Database: pod.Labels[metadata.LabelMultigresDatabase], + TableGroup: pod.Labels[metadata.LabelMultigresTableGroup], + Shard: pod.Labels[metadata.LabelMultigresShard], + }, + RoutingState: &cm.RoutingState{Role: role}, + } + if err := store.RegisterMultipooler(ctx, pooler, true); err == nil { + p.mu.Lock() + p.registered[key] = true + p.mu.Unlock() + } + } +} diff --git a/test/suite/fixture.go b/test/suite/fixture.go new file mode 100644 index 00000000..440b017a --- /dev/null +++ b/test/suite/fixture.go @@ -0,0 +1,113 @@ +package suite + +import ( + "fmt" + "time" + + corev1 "k8s.io/api/core/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "sigs.k8s.io/controller-runtime/pkg/client" + + multigresv1alpha1 "github.com/multigres/multigres-operator/api/v1alpha1" +) + +// The name of a Secret object, not a credential. gosec matches on the string +// value rather than on how it is used, so the suppression has to be explicit. +// +//nolint:gosec // G101: this is a Secret's name; the password itself is set below +const adminSecretName = "multigres-admin-password" + +// MinimalCluster creates the equivalent of config/samples/minimal.yaml in ns, +// along with the password Secret it references, and returns it. +// +// Built in Go rather than read from the sample, because the e2e loader +// (framework.MustLoadCluster) is behind //go:build e2e and this package +// deliberately has no build tag. +// +// The Secret is intentionally unlabelled, matching what a user would create. +// Under the production cache config that makes it invisible to the cached +// client, so this fixture also exercises why the reconcilers hold an APIReader. +func (c *C) MinimalCluster(name string) *MultigresCluster { + c.Helper() + return c.newCluster(name) +} + +// newCluster creates the admin Secret every cluster in this package +// references, then the cluster itself, applying any spec adjustments in +// between. +// +// The four fixtures here were identical for twenty lines and diverged only +// at Spec.Databases, which is the argument for this existing: the cost of +// the copy grows with the test count rather than being a debt that stays +// fixed, and the next person writing a scenario test copies whichever +// fixture they happened to read. +// +// The adjustment is a callback rather than a returned unsaved object so +// that creating the cluster cannot be forgotten. A fixture that built an +// object and never persisted it would leave the test waiting on a +// convergence that had no reason to start. +// +// The Secret is intentionally unlabelled, matching what a user would create. +// Under the production cache config that makes it invisible to the cached +// client, so this fixture also exercises why the reconcilers hold an +// APIReader. +func (c *C) newCluster(name string, with ...func(*MultigresClusterSpec)) *MultigresCluster { + c.Helper() + + secret := &corev1.Secret{ + ObjectMeta: metav1.ObjectMeta{Name: adminSecretName, Namespace: c.NS}, + StringData: map[string]string{"password": "postgres"}, + } + c.NoError(c.Create(secret), "create password secret") + + cluster := &MultigresCluster{ + ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: c.NS}, + Spec: MultigresClusterSpec{ + PostgresPasswordSecretRef: PostgresPasswordSecretRef{ + Name: adminSecretName, + Key: "password", + }, + PVCDeletionPolicy: &PVCDeletionPolicy{ + WhenDeleted: multigresv1alpha1.DeletePVCRetentionPolicy, + WhenScaled: multigresv1alpha1.DeletePVCRetentionPolicy, + }, + Cells: []CellConfig{ + {Name: defaultSimCell, ZoneID: "us-central1-a"}, + }, + }, + } + for _, adjust := range with { + adjust(&cluster.Spec) + } + c.NoError(c.Create(cluster), "create MultigresCluster") + return cluster +} + +// WaitForClusterHealthy blocks until the cluster reports PhaseHealthy. +// +// One method rather than the seven copies pass 1 left behind: two named +// helpers (waitForClusterHealthy in the thrash file and waitForHealthy in the +// transitions file, byte-identical to each other) and five inlined +// Eventually blocks. They were hard to see as duplicates while each was +// wrapped in its own t-and-namespace threading. +// +// It is the convergence check nearly every scenario test starts from, which +// the old comment on one of the copies said out loud without anyone acting +// on it. +// +// 30 seconds because that is what all seven used. It is a convergence wait +// for a whole cluster under five controllers, not a single object read, so it +// is deliberately far longer than any assertion budget. +func (c *C) WaitForClusterHealthy(cluster *MultigresCluster) { + c.Helper() + c.Eventually(30*time.Second, "cluster to report Healthy", func() error { + got := &MultigresCluster{} + if err := c.Get(client.ObjectKeyFromObject(cluster), got); err != nil { + return err + } + if got.Status.Phase != multigresv1alpha1.PhaseHealthy { + return fmt.Errorf("phase is %q", got.Status.Phase) + } + return nil + }) +} diff --git a/test/suite/golden_test.go b/test/suite/golden_test.go new file mode 100644 index 00000000..f68746ed --- /dev/null +++ b/test/suite/golden_test.go @@ -0,0 +1,67 @@ +package suite + +import ( + "testing" + + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/runtime" + + multigresv1alpha1 "github.com/multigres/multigres-operator/api/v1alpha1" + cellcontroller "github.com/multigres/multigres-operator/pkg/resource-handler/controller/cell" + "github.com/multigres/testkit/golden" +) + +// TestGoldenMultigatewayDeployment pilots golden.AssertYAML against +// BuildMultigatewayDeployment, called directly rather than through the +// running cell controller: the builder is exported and reachable from this +// package, so the pilot exercises a pure function with no generated fields to +// strip, rather than an object fetched from envtest. +// +// It builds its own scheme rather than reaching for Suite.Scheme: the +// builder call needs nothing envtest boots, and coupling to Suite would make +// this test pay for the whole suite's startup for no benefit. +// +// The fixture pins both Spec.Observability (via OTEL_EXPORTER_OTLP_ENDPOINT) +// and Spec.Images.Multigateway to fixed values: a golden over a builder's +// output must not depend on values that routine maintenance changes. +func TestGoldenMultigatewayDeployment(t *testing.T) { + // BuildMultigatewayDeployment resolves OTEL settings from the process + // environment when Spec.Observability is nil (as it is below), so the + // golden file is only stable once that read is pinned. "disabled" is the + // builder's own sentinel for suppressing every OTEL var. + t.Setenv("OTEL_EXPORTER_OTLP_ENDPOINT", "disabled") + c := newBareCase(t) + + scheme := runtime.NewScheme() + c.NoError(multigresv1alpha1.AddToScheme(scheme), "add to scheme") + + cell := &multigresv1alpha1.Cell{ + ObjectMeta: metav1.ObjectMeta{ + Name: "golden-cell", + Namespace: "default", + UID: "golden-cell-uid", + Labels: map[string]string{"multigres.com/cluster": "golden-cluster"}, + }, + Spec: multigresv1alpha1.CellSpec{ + Name: "zone1", + GlobalTopoServer: multigresv1alpha1.GlobalTopoServerRef{ + Address: "global-topo:2379", + RootPath: "/multigres/global", + Implementation: "etcd", + }, + LogLevels: multigresv1alpha1.ComponentLogLevels{ + Multigateway: "info", + }, + Images: multigresv1alpha1.CellImages{ + Multigateway: multigresv1alpha1.ImageRef( + "ghcr.io/multigres/multigres:golden-fixture", + ), + }, + }, + } + + got, err := cellcontroller.BuildMultigatewayDeployment(cell, scheme) + c.NoError(err, "BuildMultigatewayDeployment") + + golden.AssertYAML(t, got, "testdata/multigateway-deployment.golden.yaml") +} diff --git a/test/suite/identity.go b/test/suite/identity.go new file mode 100644 index 00000000..5112c0c6 --- /dev/null +++ b/test/suite/identity.go @@ -0,0 +1,98 @@ +package suite + +import ( + "context" + "fmt" + "sort" + "strings" + + "sigs.k8s.io/controller-runtime/pkg/client" + + shardcontroller "github.com/multigres/multigres-operator/pkg/resource-handler/controller/shard" + "github.com/multigres/testkit/ctrltest" +) + +// Members is a snapshot of a Shard's pod roles, taken at the moment of a call. +type Members struct { + Primary string // pod name with role PRIMARY + Replicas []string // pod names with role REPLICA, sorted + Quarantined []string // pod names with role QUARANTINED, sorted +} + +// MembersOf reads the Shard's status.podRoles and classifies every pod. +// Error-returning because callers include KnownDefect bodies. +// +// The role set here is {PRIMARY, REPLICA, QUARANTINED}, not the +// {PRIMARY, REPLICA, DRAINED} the CRD doc comment on PodRoles still claims +// (api/v1alpha1/shard_types.go:329, stale since e3677f0). PodRoles has one +// writer, reconcile_data_plane.go, fed entirely by GetPoolerStatus in +// pkg/data-handler/topo/pooler.go, where roleName is one of exactly those +// three literals. DRAINED cannot be produced by the real operator today, so +// it falls to the default arm below like any other unrecognized value. +func MembersOf(ctx context.Context, c client.Client, key client.ObjectKey) (Members, error) { + shard := &Shard{} + if err := c.Get(ctx, key, shard); err != nil { + return Members{}, fmt.Errorf("get shard %s: %w", key, err) + } + + roles := shard.Status.PodRoles + if len(roles) == 0 { + return Members{}, fmt.Errorf("shard %s: status.podRoles is empty", key) + } + + var primaries, replicas, quarantined []string + for pod, role := range roles { + switch role { + case "PRIMARY": + primaries = append(primaries, pod) + case "REPLICA": + replicas = append(replicas, pod) + case "QUARANTINED": + quarantined = append(quarantined, pod) + default: + return Members{}, fmt.Errorf( + "shard %s: pod %s has unrecognized role %q", key, pod, role, + ) + } + } + + switch len(primaries) { + case 0: + return Members{}, fmt.Errorf("shard %s: no pod has role PRIMARY", key) + case 1: + default: + sort.Strings(primaries) + return Members{}, fmt.Errorf( + "shard %s: more than one pod has role PRIMARY: %s", + key, strings.Join(primaries, ", "), + ) + } + + sort.Strings(replicas) + sort.Strings(quarantined) + + return Members{ + Primary: primaries[0], + Replicas: replicas, + Quarantined: quarantined, + }, nil +} + +// ShardPVCOf returns the PVC bound by the named pool pod's data volume. +// +// Pool pods only. The toposerver controller declares its own +// DataVolumeName ("data", in its statefulset builder), so this returns a +// not-found error for a toposerver pod rather than that pod's data PVC. A +// loud error rather than a wrong answer, but the restriction is not +// visible in the signature. +// +// A pool pod can carry a second PVC-backed volume (the filesystem backup +// volume, when shard.Spec.Backup.Type is Filesystem; see +// buildSharedBackupVolume in the shard controller), which is why the +// underlying lookup selects by volume name rather than requiring the pod +// to have exactly one PVC volume. +func ShardPVCOf( + ctx context.Context, c client.Client, ns, pod string, +) (string, error) { + return ctrltest.PVCOf(ctx, c, ns, pod, shardcontroller.DataVolumeName) +} diff --git a/test/suite/identity_test.go b/test/suite/identity_test.go new file mode 100644 index 00000000..3be6874b --- /dev/null +++ b/test/suite/identity_test.go @@ -0,0 +1,239 @@ +package suite + +import ( + "testing" + + corev1 "k8s.io/api/core/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/runtime" + "sigs.k8s.io/controller-runtime/pkg/client" + "sigs.k8s.io/controller-runtime/pkg/client/fake" + + multigresv1alpha1 "github.com/multigres/multigres-operator/api/v1alpha1" + shardcontroller "github.com/multigres/multigres-operator/pkg/resource-handler/controller/shard" +) + +// identityFakeClient builds a fake client that knows about both the +// operator's CRDs and core types, since a Shard's status and a Pod's +// volumes both need to round-trip through it. +func identityFakeClient(objs ...client.Object) client.Client { + scheme := runtime.NewScheme() + if err := multigresv1alpha1.AddToScheme(scheme); err != nil { + panic(err) + } + if err := corev1.AddToScheme(scheme); err != nil { + panic(err) + } + return fake.NewClientBuilder(). + WithScheme(scheme). + WithObjects(objs...). + WithStatusSubresource(&Shard{}). + Build() +} + +func shardWithRoles(ns, name string, roles map[string]string) *Shard { + return &Shard{ + ObjectMeta: metav1.ObjectMeta{Namespace: ns, Name: name}, + Status: multigresv1alpha1.ShardStatus{ + PodRoles: roles, + }, + } +} + +func TestIdentityMembersOfClassifiesAndSortsReplicas(t *testing.T) { + c := newBareCase(t) + shard := shardWithRoles("ns1", "shard1", map[string]string{ + "pod-primary": "PRIMARY", + "pod-replica-b": "REPLICA", + "pod-replica-a": "REPLICA", + }) + fc := identityFakeClient(shard) + + members, err := MembersOf(c.Context(), fc, client.ObjectKeyFromObject(shard)) + c.NoError(err, "MembersOf") + + c.Check().Eq("pod-primary", members.Primary, "Primary") + want := []string{"pod-replica-a", "pod-replica-b"} + gotSorted := len(members.Replicas) == len(want) && + members.Replicas[0] == want[0] && members.Replicas[1] == want[1] + c.Check().True(gotSorted, "Replicas = %v, want %v (sorted)", members.Replicas, want) +} + +func TestIdentityMembersOfSeparatesQuarantinedFromReplicas(t *testing.T) { + c := newBareCase(t) + shard := shardWithRoles("ns1", "shard1", map[string]string{ + "pod-primary": "PRIMARY", + "pod-replica": "REPLICA", + "pod-quarantined": "QUARANTINED", + }) + fc := identityFakeClient(shard) + + members, err := MembersOf(c.Context(), fc, client.ObjectKeyFromObject(shard)) + c.NoError(err, "MembersOf") + + c.Check().EqDiff([]string{"pod-replica"}, members.Replicas, "Replicas") + c.Check().EqDiff([]string{"pod-quarantined"}, members.Quarantined, "Quarantined") + c.NotContains(members.Replicas, "pod-quarantined", "quarantined pod leaked into Replicas") +} + +// TestIdentityMembersOfUnrecognizedRoleErrors pins the guard that catches any +// role value outside {PRIMARY, REPLICA, QUARANTINED}, the set the operator's +// single writer (pkg/data-handler/topo/pooler.go) can actually produce. DRAINED +// is deliberately used as the unknown value here: the CRD doc comment on +// PodRoles still lists it, but commit e3677f0 removed it from the writer, so +// it is exactly the stale value a careless "known roles" list would still +// accept. +func TestIdentityMembersOfUnrecognizedRoleErrors(t *testing.T) { + c := newBareCase(t) + shard := shardWithRoles("ns1", "shard1", map[string]string{ + "pod-primary": "PRIMARY", + "pod-drained": "DRAINED", + }) + fc := identityFakeClient(shard) + + _, err := MembersOf(c.Context(), fc, client.ObjectKeyFromObject(shard)) + c.Error(err, "want an error for an unrecognized role") + + c.Check(). + ErrorContains(err, "unrecognized role", "want the error to call DRAINED an unrecognized role") + c.Check().ErrorContains(err, "DRAINED", "want the error to name DRAINED specifically") +} + +// TestIdentityMembersOfErrorCases pins the three distinct error cases the +// brief calls out. An absent primary and a duplicated primary are both bugs +// a test should be able to pin, but they are different bugs, so a test that +// only checked "err != nil" could not tell them apart. +func TestIdentityMembersOfErrorCases(t *testing.T) { + c := newBareCase(t) + cases := []struct { + name string + roles map[string]string + wantErrs []string + }{ + { + name: "empty PodRoles", + roles: map[string]string{}, + wantErrs: []string{"empty"}, + }, + { + name: "no primary", + roles: map[string]string{ + "pod-a": "REPLICA", + "pod-b": "QUARANTINED", + }, + wantErrs: []string{"no pod has role PRIMARY"}, + }, + { + name: "two primaries", + roles: map[string]string{ + "pod-a": "PRIMARY", + "pod-b": "PRIMARY", + }, + wantErrs: []string{"more than one pod has role PRIMARY", "pod-a", "pod-b"}, + }, + } + + // Each case's error must be distinguishable from the other two, so this + // collects every message actually produced and cross-checks that no + // case's message satisfies another case's expectation. + messages := make(map[string]string, len(cases)) + + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + c := newBareCase(t) + shard := shardWithRoles("ns1", "shard1", tc.roles) + fc := identityFakeClient(shard) + + _, err := MembersOf(c.Context(), fc, client.ObjectKeyFromObject(shard)) + c.Error(err, "want an error for %s", tc.name) + for _, want := range tc.wantErrs { + c.Check().ErrorContains(err, want) + } + messages[tc.name] = err.Error() + }) + } + + c.Check().NotEq(messages["empty PodRoles"], messages["no primary"], + "empty PodRoles and no-primary produced the same error message") + c.Check().NotEq(messages["no primary"], messages["two primaries"], + "no-primary and two-primaries produced the same error message") + c.Check().NotEq(messages["empty PodRoles"], messages["two primaries"], + "empty PodRoles and two-primaries produced the same error message") +} + +// identityPVCVolume is a thin copy of relations_test.go's pvcVolume. Kept +// separate rather than shared, since after Task 2.1 the original lives in +// another package. +func identityPVCVolume(name, claim string) corev1.Volume { + return corev1.Volume{ + Name: name, + VolumeSource: corev1.VolumeSource{ + PersistentVolumeClaim: &corev1.PersistentVolumeClaimVolumeSource{ + ClaimName: claim, + }, + }, + } +} + +// identityPodWithVolumes is a thin copy of relations_test.go's +// podWithVolumes. Kept separate rather than shared, since after Task 2.1 the +// original lives in another package. +func identityPodWithVolumes(ns, name string, volumes ...corev1.Volume) *corev1.Pod { + return &corev1.Pod{ + ObjectMeta: metav1.ObjectMeta{Namespace: ns, Name: name}, + Spec: corev1.PodSpec{ + Containers: []corev1.Container{{Name: "c", Image: "busybox"}}, + Volumes: volumes, + }, + } +} + +// TestIdentityShardPVCOfUsesTheShardDataVolumeName pins which volume name +// the wrapper supplies. Nothing else does: PVCOf's own tests pass a +// literal, so a wrapper handing over the wrong constant is invisible. +func TestIdentityShardPVCOfUsesTheShardDataVolumeName(t *testing.T) { + c := newBareCase(t) + p := identityPodWithVolumes( + "ns1", "pod-a", + identityPVCVolume(shardcontroller.DataVolumeName, "pod-a-data"), + identityPVCVolume("backup-data", "pod-a-backup"), + ) + fc := identityFakeClient(p) + + claim, err := ShardPVCOf(c.Context(), fc, "ns1", "pod-a") + c.NoError(err, "ShardPVCOf") + c.Check().Eq("pod-a-data", claim) +} + +// TestIdentityMembersOfSortsEnoughReplicasToCatchMapOrder pins the sorts in +// MembersOf. PodRoles is a Go map and Go randomises map iteration, so two +// replicas would agree with insertion order half the time and the assertion +// would be a coin flip. Five in reverse order leaves a 1-in-120 chance of +// passing against an unsorted implementation. +func TestIdentityMembersOfSortsEnoughReplicasToCatchMapOrder(t *testing.T) { + c := newBareCase(t) + shard := &Shard{ + ObjectMeta: metav1.ObjectMeta{Name: "shard-0", Namespace: "ns"}, + Status: multigresv1alpha1.ShardStatus{PodRoles: map[string]string{ + "pool-primary": "PRIMARY", + "pool-e": "REPLICA", + "pool-d": "REPLICA", + "pool-c": "REPLICA", + "pool-b": "REPLICA", + "pool-a": "REPLICA", + "quar-e": "QUARANTINED", + "quar-d": "QUARANTINED", + "quar-c": "QUARANTINED", + "quar-b": "QUARANTINED", + "quar-a": "QUARANTINED", + }}, + } + fc := identityFakeClient(shard) + + got, err := MembersOf(c.Context(), fc, client.ObjectKeyFromObject(shard)) + c.NoError(err, "MembersOf") + wantReplicas := []string{"pool-a", "pool-b", "pool-c", "pool-d", "pool-e"} + wantQuarantined := []string{"quar-a", "quar-b", "quar-c", "quar-d", "quar-e"} + c.Check().EqDiff(wantReplicas, got.Replicas, "Replicas") + c.Check().EqDiff(wantQuarantined, got.Quarantined, "Quarantined") +} diff --git a/test/suite/main_test.go b/test/suite/main_test.go new file mode 100644 index 00000000..214d750e --- /dev/null +++ b/test/suite/main_test.go @@ -0,0 +1,66 @@ +package suite + +import ( + "fmt" + "os" + "runtime/pprof" + "testing" + + "go.uber.org/goleak" +) + +// TestMain boots the suite once, runs the package, tears it down, and only then +// checks for leaked goroutines. +// +// Deliberately not goleak.VerifyTestMain: that calls m.Run() and checks +// immediately afterwards, with no hook in between. Since envtest and the +// manager live for the whole package rather than for one test, the check would +// run while both were still up and report the entire manager as leaked. +// +// The ignore list is empty on purpose. A clean start/stop leaks nothing +// measurable (verified across 16 cycles before this suite existed), so every +// future entry should be justified against a real stack trace rather than +// pre-loaded against suspects. A pre-loaded ignore is a permanent hole. +func TestMain(m *testing.M) { + s, teardown, err := Boot() + if err != nil { + fmt.Fprintf(os.Stderr, "suite boot failed: %v\n", err) + os.Exit(1) + } + Suite = s + + code := m.Run() + + if err := teardown(); err != nil { + fmt.Fprintf(os.Stderr, "suite teardown failed: %v\n", err) + if code == 0 { + code = 1 + } + } + + // Only leak-check a passing run: a failed test may have left its own + // goroutines behind, and reporting those on top of a real failure buries + // the real failure. + if code == 0 { + if err := goleak.Find(); err != nil { + fmt.Fprintf(os.Stderr, "goroutine leak after suite teardown: %v\n", err) + dumpGoroutineLeakProfile() + code = 1 + } + } + os.Exit(code) +} + +// dumpGoroutineLeakProfile writes the runtime's own leak profile when the +// toolchain has one. It returns nil before Go 1.27, so this is a no-op today +// and becomes a diagnostic on the next toolchain bump, with no build tag. +// +// Count() does not run the leak detector but WriteTo() does, so the profile has +// to be driven through WriteTo and the count read afterwards, never before. +func dumpGoroutineLeakProfile() { + p := pprof.Lookup("goroutineleak") + if p == nil { + return + } + _ = p.WriteTo(os.Stderr, 1) +} diff --git a/test/suite/scenario_deletion_test.go b/test/suite/scenario_deletion_test.go new file mode 100644 index 00000000..2d2a872d --- /dev/null +++ b/test/suite/scenario_deletion_test.go @@ -0,0 +1,295 @@ +package suite + +import ( + "strings" + "testing" + "time" + + apierrors "k8s.io/apimachinery/pkg/api/errors" + "k8s.io/utils/ptr" + "sigs.k8s.io/controller-runtime/pkg/client" + + multigresv1alpha1 "github.com/multigres/multigres-operator/api/v1alpha1" + multigresclustercontroller "github.com/multigres/multigres-operator/pkg/cluster-handler/controller/multigrescluster" + tablegroupcontroller "github.com/multigres/multigres-operator/pkg/cluster-handler/controller/tablegroup" + "github.com/multigres/testkit/ctrltest" +) + +// TestReadyForDeletionProtocol asserts the three-hop ReadyForDeletion protocol +// found in pkg/resource-handler/controller/shard/reconcile_deletion.go, +// pkg/cluster-handler/controller/tablegroup/tablegroup_controller.go and +// pkg/cluster-handler/controller/multigrescluster/reconcile_databases.go: a +// Shard sets multigresv1alpha1.ConditionReadyForDeletion on itself once every +// pool pod has drained, its parent TableGroup sets the same condition on +// itself once every child Shard has, and MultigresCluster deletes a TableGroup +// only once that TableGroup reports it. +// +// This is orphan pruning, not whole-cluster teardown. Deleting a +// MultigresCluster goes through MultigresClusterReconciler.handleDeletion, +// which lists and Deletes its Cells and TableGroups directly and never +// consults ConditionReadyForDeletion at all; that path was added in +// da7d639177b0 as a narrower fix scoped explicitly to "orphan pruning" (its +// own commit message), for the case where a TableGroup or Cell falls out of a +// still-live cluster's spec and must drain before it is safe to remove. A +// whole-cluster delete has nothing left to protect by draining, so it tears +// down directly instead. The three-hop protocol is therefore reachable only +// through that narrower path, which this test drives directly: it builds an +// orphan TableGroup (one MultigresCluster.Spec.Databases entry will never +// name) with multigresclustercontroller.BuildTableGroup, the same builder +// production code uses, so multigrescluster's reconcileDatabases treats it +// exactly as it would treat a TableGroup a user just removed from spec. +func TestReadyForDeletionProtocol(t *testing.T) { + c := newCase(t) + ns := c.NS + cluster := c.MinimalCluster("scenario-del") + + c.WaitForClusterHealthy(cluster) + + // Copy the real TableGroup's resolved GlobalTopoServer ref and component + // Images rather than re-deriving them: both are resolved in-memory once + // per MultigresCluster reconcile (globalTopoRef by the unexported + // globalTopoRef method, Images by resolveImages) and never written back to + // MultigresCluster.Spec, so the cluster object this test already holds + // still has them blank. Re-deriving either by hand risks building an + // orphan that fails validation or that the topology store does not + // recognize, for reasons unrelated to what this test asserts. + realTGs := &TableGroupList{} + c.NoError(c.List(realTGs, + client.MatchingLabels{"multigres.com/cluster": cluster.Name}), + "list tablegroups") + c.Len(realTGs.Items, 1, + "want exactly one TableGroup before introducing an orphan") + globalTopoRef := realTGs.Items[0].Spec.GlobalTopoServer + cluster.Spec.Images = multigresv1alpha1.ClusterImages{ + Multiorch: realTGs.Items[0].Spec.Images.Multiorch, + Multipooler: realTGs.Items[0].Spec.Images.Multipooler, + Postgres: realTGs.Items[0].Spec.Images.Postgres, + ImagePullPolicy: realTGs.Items[0].Spec.Images.ImagePullPolicy, + ImagePullSecrets: realTGs.Items[0].Spec.Images.ImagePullSecrets, + } + + // The orphan's one shard has no pools and zero Multiorch replicas, so it + // creates no Pods. That keeps the pod drain state machine, which is its + // own protocol, out of this test's way: with zero pods, + // ShardReconciler.handlePendingDeletion takes the "no pods" branch and + // sets ConditionReadyForDeletion on its very first pass. What this test + // asserts is the condition handoff between the three controllers, not + // how long draining a pod takes. + // + // Multiorch.Cells is set explicitly because getMultiorchCells + // (shard_controller.go) falls back to the union of pool cells when it is + // empty, and errors out when that is empty too; with no pools, leaving + // Cells unset turns every normal (non-deletion) reconcile of this Shard + // into a reconcile error, which is retried on controller-runtime's own + // backoff rather than this suite's compressed one and made the protocol's + // timing depend on that backoff instead of on the protocol. + // + // Opened before the orphan TableGroup exists, not after: five reconcilers + // run concurrently (scenario_fanout_test.go documents the same discipline for + // cursors, for the same reason), and multigrescluster can notice and + // annotate an orphan TableGroup within the reconcile pass that follows its + // creation. A stream opened even one line later could start listening + // after that annotation, and the condition flips it drives, have already + // landed, which is exactly the 4ms-window flake this replaces. + st := c.Watch(&TableGroupList{}, &ShardList{}) + + orphanTG, err := multigresclustercontroller.BuildTableGroup( + cluster, + DatabaseConfig{Name: "postgres"}, + &TableGroupConfig{Name: "orphan"}, + []multigresv1alpha1.ShardResolvedSpec{{ + Name: "0-inf", + Multiorch: multigresv1alpha1.MultiorchSpec{ + StatelessSpec: multigresv1alpha1.StatelessSpec{Replicas: ptr.To(int32(0))}, + Cells: []CellName{defaultSimCell}, + }, + Pools: map[PoolName]PoolSpec{}, + }}, + globalTopoRef, + Suite.Scheme, + ) + c.NoError(err, "build orphan tablegroup") + c.NoError(c.Create(orphanTG), "create orphan tablegroup") + orphanKey := client.ObjectKeyFromObject(orphanTG) + + // Pre-create the child Shard, from the now-persisted orphanTG (so its + // owner reference carries a real UID), immediately after the TableGroup + // rather than leaving TableGroupReconciler to create it on its first + // normal pass. Without this, a real race exists: if multigrescluster + // annotates the brand-new TableGroup with AnnotationPendingDeletion before + // TableGroupReconciler has ever run stepApplyDesiredShards for it, + // handlePendingDeletion lists zero child Shards and reports + // ReadyForDeletion vacuously, having consulted nothing. Creating the Shard + // here, keyed identically to what stepApplyDesiredShards would build, + // guarantees the child the protocol is supposed to drain exists before + // either controller's watch can fire. AlreadyExists is fine rather than + // fatal: it means TableGroupReconciler's own first pass won the race and + // applied this same Shard first, which is the other safe ordering. + shardCR, err := tablegroupcontroller.BuildShard( + orphanTG, + &orphanTG.Spec.Shards[0], + Suite.Scheme, + ) + c.NoError(err, "build orphan shard") + if err := c.Create(shardCR); err != nil && + !apierrors.IsAlreadyExists(err) { + c.NoError(err, "create orphan shard") + } + orphanShardKey := client.ObjectKeyFromObject(shardCR) + clusterKey := client.ObjectKeyFromObject(cluster) + + // Corroborating evidence, independent of how fast the three controllers + // converge: the interceptor records every reconcile pass, including ones + // that wrote nothing, so a pass that asked to be woken again in 5s while + // something was still pending is permanent history even if the object it + // was about is deleted moments later. Both waits are scoped by object key, + // so neither can be satisfied by the pre-existing healthy + // TableGroup/Shard's own unrelated reconciles. + // + // What this actually proves is narrower than it looks: both + // TableGroupReconciler.handlePendingDeletion and reconcileDatabases also + // take a 5s-requeue path the first time they see an orphan (setting the + // PendingDeletion annotation itself sets `allReady`/`pendingDeletion` and + // requeues), so mutating away only the later + // `meta.IsStatusConditionTrue(...ConditionReadyForDeletion)` guard still + // leaves that earlier requeue in place and these two waits keep passing - + // verified by making exactly that mutation in each function and watching + // these two lines stay green while the poll below caught it instead. What + // these two waits do rule out is a version that deletes an orphan in the + // very same pass that first notices it, with no intervening wait at all. + Suite.Reconciles.WaitForRequeue(t, "tablegroup", orphanKey, 5*time.Second, 10*time.Second) + Suite.Reconciles.WaitForRequeue( + t, + "multigrescluster", + clusterKey, + 5*time.Second, + 10*time.Second, + ) + + // Primary evidence for the specific guard on each hop, read off the event + // stream rather than sampled by polling: the write-up measured the + // TableGroup's parent deleting it 4.3ms after ReadyForDeletion is set, + // against a 20ms poll, and showed no poll interval fixes that because one + // sample already costs about as long as the state persists. The stream is + // push rather than sample, so it sees the transition however briefly it + // held. sawX latches record having observed each condition true at least + // once, so that reaching a later state (the TableGroup gone) without ever + // having latched an earlier one (its own condition, or its child Shard's) + // is still caught even if every step landed inside one 50ms reorder + // window, or before this loop's first read. + // + // Mutation verified: deleting the + // `if !meta.IsStatusConditionTrue(s.Status.Conditions, + // multigresv1alpha1.ConditionReadyForDeletion) { allReady = false }` guard + // in TableGroupReconciler.handlePendingDeletion, or the equivalent guard + // over item.Status.Conditions in reconcileDatabases, each independently + // makes this fail (tried one at a time): the TableGroup reports + // ReadyForDeletion, or is deleted, before its Shard's own condition is + // ever observed true. + sawShardReady := false + sawTableGroupReady := false + tgGone := false + deadline := time.Now().Add(30 * time.Second) + for !tgGone { + ev, err := st.Next(time.Until(deadline)) + c.NoError(err, + "waiting for the protocol to reach Shard ready, then TableGroup ready, "+ + "then TableGroup deleted, in that order") + switch { + case ev.Key == orphanShardKey && ev.Kind == "Shard": + if conditionSetTrue(ev, string(multigresv1alpha1.ConditionReadyForDeletion)) { + sawShardReady = true + } + case ev.Key == orphanKey && ev.Kind == "TableGroup": + if ev.Type == "deleted" { + tgGone = true + continue + } + if conditionSetTrue(ev, string(multigresv1alpha1.ConditionReadyForDeletion)) { + c.True(sawShardReady, + "TableGroup %s reported ReadyForDeletion before Shard %s ever did", + orphanKey.Name, orphanShardKey.Name) + sawTableGroupReady = true + } + } + } + // A relist can win the race against the last blocked read: Next selects + // over the event channel and the failure channel, and Go picks at random + // when both are ready. If it hands back the deletion, the loop exits and + // the entry guard never runs again, so the terminal error would go + // unobserved and the assertions below would blame the operator for history + // the harness lost. + c.NoError(st.Terminal(), + "the event stream failed, so every assertion over it is void") + + c.True(sawShardReady, + "TableGroup %s was deleted before Shard %s ever reported ReadyForDeletion", + orphanKey.Name, orphanShardKey.Name) + c.True(sawTableGroupReady, + "TableGroup %s was deleted before it ever reported ReadyForDeletion", + orphanKey.Name) + + // Attribution check: confirm the delete that made the TableGroup + // disappear was actually issued by multigrescluster, using a static scan + // of the completed op log rather than a live cursor wait. A live + // CursorFor(ns, "multigrescluster") wait was not usable for any step + // above: a single MultigresCluster reconcile pass unconditionally + // re-applies the healthy default Cell and TableGroup before ever reaching + // the orphan-pruning loop, so the next op in that scope is legitimately + // something else almost every time, and WaitForNext does not skip ahead + // to find a match (that is WaitForMatching, which this suite marks as an + // escape hatch not to be reached for). The op log has already stopped + // growing with respect to this object by the time we reach this check, so + // a static scan carries none of Cursor's live-ordering caveats. + deletedByCluster := false + for _, op := range Suite.Ops.OpsInNamespace(ns) { + if op.Controller == "multigrescluster" && op.Verb == "delete" && + ctrltest.KindSuffix(op.Kind) == "TableGroup" && op.Key.Name == orphanKey.Name { + deletedByCluster = true + break + } + } + c.True(deletedByCluster, + "TableGroup %s disappeared without a recorded delete from multigrescluster", + orphanKey.Name) +} + +// conditionSetTrue reports whether ev records a status.conditions entry whose +// type field arrived at conditionType, with that same entry's status field +// arrived at "True", in the same event. +// +// Changed and Transitions carry a diff, not the object's current state, so +// the type and the status of one array slot have to be read out of the same +// event to know which condition moved; a status flip alone does not say +// which condition it belongs to. Requiring the type to change in the same +// event rather than looking it up separately is sound here because every +// condition this test watches for (Shard and TableGroup's own +// ConditionReadyForDeletion) is only ever set once, straight to True: neither +// controller ever writes it False first, so its type always appears fresh +// alongside the status that makes it true. +// +// If that premise ever breaks, and a controller sets the condition False +// before True, this helper stops recognising the transition and the test fails +// with an ordering complaint against the operator rather than against itself. +// So a sudden "reported ReadyForDeletion before X ever did" failure is worth +// checking here second: the Cell controller already writes a condition False +// first, so the premise holds by habit rather than by rule. +func conditionSetTrue(ev ctrltest.Event, conditionType string) bool { + wantType := `"` + conditionType + `"` + for path, transition := range ev.Transitions { + // The status.conditions prefix is checked as well as the leaf, so a + // future array of objects carrying both a type and a status field + // cannot start feeding this helper silently. No such array exists on + // either status today; the guard is what keeps the doc comment above + // true rather than merely true for now. + if !strings.HasPrefix(path, "status.conditions[") || + !strings.HasSuffix(path, "].type") || transition.To != wantType { + continue + } + statusPath := strings.TrimSuffix(path, "type") + "status" + if status, ok := ev.Transitions[statusPath]; ok && status.To == `"True"` { + return true + } + } + return false +} diff --git a/test/suite/scenario_fanout_test.go b/test/suite/scenario_fanout_test.go new file mode 100644 index 00000000..d0578763 --- /dev/null +++ b/test/suite/scenario_fanout_test.go @@ -0,0 +1,159 @@ +package suite + +import ( + "testing" + "time" + + "k8s.io/utils/ptr" + + multigresv1alpha1 "github.com/multigres/multigres-operator/api/v1alpha1" + "github.com/multigres/multigres-operator/pkg/util/name" + "github.com/multigres/testkit/ctrltest" +) + +// fanoutCluster builds a MultigresCluster with one cell and one database, +// table group and shard: enough for the cluster controller to fan out to +// every child kind it owns (TopoServer, Cell, TableGroup) and for the +// TableGroup controller it creates to fan out to a Shard of its own. +func (c *C) fanoutCluster(clusterName string) *MultigresCluster { + c.Helper() + return c.newCluster(clusterName, func(s *MultigresClusterSpec) { + s.Databases = []DatabaseConfig{ + { + Name: "postgres", + Default: true, + TableGroups: []TableGroupConfig{ + { + Name: "default", + Default: true, + Shards: []ShardConfig{{ + Name: "0-inf", + Spec: &ShardInlineSpec{ + Multiorch: multigresv1alpha1.MultiorchSpec{ + StatelessSpec: multigresv1alpha1.StatelessSpec{ + Replicas: ptr.To(int32(1)), + }, + }, + Pools: map[PoolName]PoolSpec{ + "primary": { + ReplicasPerCell: ptr.To(int32(1)), + Type: "readWrite", + Cells: []CellName{ + defaultSimCell, + }, + }, + }, + }, + }}, + }, + }, + }, + } + }) +} + +// findOp returns the index of the first op in ops matching kind and name. +// batch has already been validated as an exact multiset by WaitForAll, so a +// miss here would mean this helper's own matching is wrong, not that the op +// is absent. +func (c *C) findOp(ops []ctrltest.Op, kind, objName string) int { + c.Helper() + for i, op := range ops { + if ctrltest.KindSuffix(op.Kind) == kind && op.Key.Name == objName { + return i + } + } + c.Fatalf("no %s %q op among %v", kind, objName, ops) + return -1 +} + +// TestClusterFanOutSequence asserts that creating a MultigresCluster fans out +// to TopoServer, Cell and TableGroup attributed to the multigrescluster +// controller, in that relative order, and that the second-order fan-out to +// Shard is attributed to the tablegroup controller rather than to +// multigrescluster. +// +// Ordering choice: multigrescluster_controller.go calls +// reconcileGlobalComponents, then reconcileCells, then (after +// reconcileTopology, which makes no Kubernetes writes here) reconcileDatabases, +// as sequential statements inside one Reconcile call. That is a genuine, +// code-level guarantee, so TopoServer < Cell < TableGroup is asserted below. +// Nothing is asserted about the relative order of the Multiadmin/MultiadminWeb +// writes reconcileGlobalComponents also makes: they are unconditional (there +// is no way to disable them from the spec) and sit between the TopoServer and +// Cell writes, but their order relative to each other, or to TopoServer and +// Cell, is not what this test is about and is left unspecified. +// +// Matcher choice: WaitForNext cannot express this, because those +// Multiadmin/MultiadminWeb writes are real, deterministic, and land between +// TopoServer and Cell, so a strict next-op chain from TopoServer would hit a +// Deployment patch instead of Cell. WaitForAll is the right tool instead: one +// call consumes the whole first pass as a multiset, which (a) still proves +// each of TopoServer/Cell/TableGroup was written by multigrescluster and +// nothing else was (in particular, that multigrescluster never itself writes +// a Shard), and (b) hands back the ops in recorded order, which is what the +// ordering check below reads. WaitForMatching was avoided entirely: it would +// let the assertion silently skip past a misordered write instead of failing +// on it, which is exactly what an ordering test must not do. +func TestClusterFanOutSequence(t *testing.T) { + c := newCase(t) + const clusterName = "fanout" + + // Cursors are opened before the cluster exists, not after: five + // reconcilers run concurrently (one goroutine each, per suite.go), so a + // cursor opened even one line late can start its scan after another + // controller's reaction to the same write has already landed, and then + // miss the very op it was meant to catch. + clusterCur := c.Cursor("multigrescluster") + tgCur := c.Cursor("tablegroup") + + c.fanoutCluster(clusterName) + + topoServerName := clusterName + "-global-topo" + cellName := name.JoinWithConstraints(name.DefaultConstraints, clusterName, defaultSimCell) + tableGroupName := name.JoinWithConstraints( + name.DefaultConstraints, clusterName, "postgres", "default", + ) + shardName := name.JoinWithConstraints( + name.DefaultConstraints, clusterName, "postgres", "default", "0-inf", + ) + + batch := clusterCur.WaitForAll(t, []ctrltest.Expect{ + // ensureClusterFinalizer, then resolveImages recording the default + // image set: both patches of the cluster object itself, before any + // child is touched. + ctrltest.ExpectPatch("MultigresCluster", clusterName), + ctrltest.ExpectPatch("MultigresCluster", clusterName), + ctrltest.ExpectPatch("TopoServer", topoServerName), + ctrltest.ExpectPatch("Deployment", clusterName+"-multiadmin"), + ctrltest.ExpectPatch("Service", clusterName+"-multiadmin"), + ctrltest.ExpectPatch("Deployment", clusterName+"-multiadmin-web"), + ctrltest.ExpectPatch("Service", clusterName+"-multiadmin-web"), + ctrltest.ExpectPatch("Service", clusterName+"-multigateway"), + ctrltest.ExpectPatch("Service", clusterName+"-multigateway-replica"), + ctrltest.ExpectPatch("Cell", cellName), + ctrltest.ExpectPatch("TableGroup", tableGroupName), + ctrltest.ExpectStatusPatch("MultigresCluster", clusterName), + }, 30*time.Second) + + topoIdx := c.findOp(batch, "TopoServer", topoServerName) + cellIdx := c.findOp(batch, "Cell", cellName) + tgIdx := c.findOp(batch, "TableGroup", tableGroupName) + c.True(topoIdx < cellIdx, + "expected TopoServer before Cell in the multigrescluster controller's "+ + "writes, got positions %d, %d in %v", topoIdx, cellIdx, batch) + c.True(cellIdx < tgIdx, + "expected Cell before TableGroup in the multigrescluster controller's "+ + "writes, got positions %d, %d in %v", cellIdx, tgIdx, batch) + + // Second-order fan-out: the TableGroup controller, not the cluster + // controller, creates the Shard. Applying the desired Shard is the first + // write TableGroupReconciler makes (stepListChildShards only reads), so + // WaitForNext is sound here without any of the batching above. The + // attribution claim itself comes from the cursor's scope, not from a + // separate check: tgCur can only ever return an op whose Controller is + // "tablegroup" (see Recorder.firstMatchFrom), and the WaitForAll batch + // above already proved multigrescluster wrote no Shard of its own, since + // one would have shown up there as an unexpected op. + tgCur.WaitForNext(t, "patch", "Shard", shardName, 30*time.Second) +} diff --git a/test/suite/scenario_race_test.go b/test/suite/scenario_race_test.go new file mode 100644 index 00000000..8ac8a766 --- /dev/null +++ b/test/suite/scenario_race_test.go @@ -0,0 +1,335 @@ +package suite + +import ( + "fmt" + "sort" + "strings" + "testing" + "time" + + "k8s.io/apimachinery/pkg/api/equality" + "sigs.k8s.io/controller-runtime/pkg/client" + "sigs.k8s.io/structured-merge-diff/v6/fieldpath" + + "github.com/multigres/testkit/ctrltest" +) + +// shardProbeAnnotation is a key neither controller's applied payload ever +// mentions: tablegroup's BuildShard only sets an annotation map at all when the +// TableGroup carries a project-ref annotation, which MinimalCluster's fixture +// never does. Writing it from the test is therefore a mutation neither +// manager's SSA apply will contend with or revert, and it is the only way this +// test can put a fresh, known-real change on the Shard once the cluster has +// converged, since a converged tablegroup no longer gets re-triggered by its own +// stale watch. +const shardProbeAnnotation = "scenario-race-test.multigres.com/probe" + +// TestTwoControllersWriteOneShard names the two-writer relationship between the +// shard and tablegroup controllers on one object, which every other test in +// this package runs past without seeing: each of them boots the suite with +// every reconciler live, but none asserts anything about a Shard being written +// by both. +// +// The relationship was measured, not assumed: before a fix, the shard +// controller wrote its own status about ten times a second forever, and while +// that ran, the tablegroup controller issued 3,641 patches against the Shard in +// three minutes, every one a no-op. Fixing the hot loop did not remove the +// two-writer relationship, only the churn it used to cause, so it is still +// worth a name. +func TestTwoControllersWriteOneShard(t *testing.T) { + c := newCase(t) + cluster := c.MinimalCluster("race") + + c.WaitForClusterHealthy(cluster) + key := c.shardKey() + + t.Run("shard and tablegroup both write the Shard", func(t *testing.T) { + c := c.Sub(t) + writers := map[string]bool{} + for _, op := range Suite.Ops.OpsInNamespace(c.NS) { + if ctrltest.KindSuffix(op.Kind) == "Shard" && op.Key.Name == key.Name { + writers[op.Controller] = true + } + } + for _, want := range []string{"shard", "tablegroup"} { + c.Check().True(writers[want], + "the %s controller has no recorded write to Shard %s; writers seen: %v", + want, key.Name, writers) + } + }) + + // Runs before the no-op check below, which mutates the Shard: a manager + // entry left over from that mutation would be a false-positive third writer + // and would blur what this is actually about, which is the two production + // managers. + t.Run("field ownership on the Shard is disjoint", func(t *testing.T) { + c := c.Sub(t) + // Several Shard status writes carry no field owner, so the API server + // attributes them to the manager that happens to be the process name, + // and that manager ends up co-owning fields the shard controller's own + // applier claims. Retire this pin with an explicit owner on every + // status write, and replace it with c.Empty on the conflicts. + conflicts := c.fieldOwnershipConflicts(key) + c.KnownDefect("MGO-SHARD-STATUS-WRITES-NO-FIELD-OWNER", func() error { + if len(conflicts) == 0 { + return nil + } + return fmt.Errorf( + "Shard %s has fields claimed by more than one field manager, "+ + "the same shape of defect as the status hot loop:\n %s", + key.Name, strings.Join(conflicts, "\n ")) + }) + }) + + t.Run("tablegroup's patches to the Shard never change it", func(t *testing.T) { + c.Sub(t).requireTableGroupPatchesAreNoOps(key) + }) +} + +// fieldOwnershipConflicts returns every field path on the Shard that more +// than one field manager claims, sorted. An empty result is the invariant, and +// it is the check that generalises: it would have caught the status hot-loop +// directly; the loop was two field managers each asserting a value for the +// same field, ObservedGeneration or a phase, and the API server obliging both. +// A dump of writes only shows that as churn; managedFields shows it as the +// overlap it is. +// +// Field ownership is expressed by the API server as a Set per manager, encoded +// in metadata.managedFields[].fieldsV1: this decodes each manager's Set with +// the same library the server itself uses and collects every field path that +// is a member of more than one. +func (c *C) fieldOwnershipConflicts(key client.ObjectKey) []string { + c.Helper() + + shard := &Shard{} + c.NoError(c.Get(key, shard), "get shard %s", key) + + claimants := map[string][]string{} + for _, mf := range shard.ManagedFields { + if mf.FieldsV1 == nil { + continue + } + set := &fieldpath.Set{} + c.NoError(set.FromJSON(mf.FieldsV1.GetRawReader()), + "decode managedFields for manager %s", mf.Manager) + set.Iterate(func(p fieldpath.Path) { + path := p.String() + claimants[path] = append(claimants[path], mf.Manager) + }) + } + + // A Shard with no decodable managedFields claim at all would make the + // disjointness check below pass having compared nothing, which is the one + // way this assertion can go green without the invariant holding. Nothing in + // this function's own reasoning rules that out, so it is asserted here + // rather than inherited from the writer check above. + c.True(len(claimants) > 0, + "Shard %s yielded no decodable managedFields claims, so field ownership "+ + "was not checked at all; %d managedFields entries were present", + key.Name, len(shard.ManagedFields)) + + var conflicts []string + for path, managers := range claimants { + distinct := map[string]bool{} + for _, m := range managers { + distinct[m] = true + } + if len(distinct) <= 1 { + continue + } + names := make([]string, 0, len(distinct)) + for m := range distinct { + names = append(names, m) + } + sort.Strings(names) + conflicts = append(conflicts, fmt.Sprintf("%s claimed by %v", path, names)) + } + + sort.Strings(conflicts) + return conflicts +} + +// requireTableGroupPatchesAreNoOps is assertion 2 from the brief, corrected to +// the harness as it exists now rather than as it was written: convergence is +// about 0.3s and requeues are compressed, so a fixed wall-clock window no +// longer separates from scheduling noise. It also has nothing to observe by the +// time a test could get around to sleeping: once the cluster is Healthy, +// tablegroup's own SSA patches to the Shard stop generating new watch events +// (a true no-op patch does not bump resourceVersion, so Owns(&Shard{}) never +// re-fires), so tablegroup simply stops reconciling and there is no ten-second +// window in which anything would happen anyway. +// +// So instead of waiting, this drives the relationship directly: it makes a +// small, known write of its own to the Shard (an annotation neither manager's +// apply payload mentions, see shardProbeAnnotation) to produce a fresh watch +// event, then waits for a COMPLETE tablegroup reconcile pass that both began +// after that write and wrote this Shard (see awaitTableGroupPassAfter), read +// from Suite.Reconciles rather than the raw op cursor. That matters: a +// reconcile record is only appended once the whole pass has returned, so +// waiting on it cannot observe half a pass the way a plain op-log cursor can, +// where a later step of the very pass that just satisfied the wait +// (tablegroup's own status-patch, a step after the Shard patch) can still land +// and be mistaken for the next pass's write. +// +// The no-op check itself compares content, not resourceVersion. The shard +// controller reconciles this same object roughly every clamp interval even at +// steady state (see ctrltest.RequeueClamp), so by the time tablegroup's pass +// has been observed, some other write to the Shard has almost always also +// landed in the same rough window; attributing a resourceVersion move to +// "whichever controller wrote most recently" is exactly as unsound as the +// wall-clock method this replaces; it was tried and produced a false pass +// under mutation (see the report). tablegroup's SSA apply payload is a +// complete statement of what it owns: spec, labels and ownerReferences, +// never status (see BuildShard), so comparing that payload's own fields +// before and after the pass answers the question directly and is immune to +// any concurrent, status-only write from the shard controller, no matter how +// often it fires. +// +// Repeated several times rather than once, since a single pass proves nothing +// about whether "never" holds. +func (c *C) requireTableGroupPatchesAreNoOps(key client.ObjectKey) { + c.Helper() + + const passesToObserve = 5 + for i := range passesToObserve { + before := &Shard{} + c.NoError(c.Get(key, before), "get shard %s", key) + + probedFrom := c.probeShard(key, i) + pass := awaitTableGroupPassAfter(c, c.NS, key, probedFrom) + + after := &Shard{} + c.NoError(c.Get(key, after), "get shard %s", key) + + c.True(equality.Semantic.DeepEqual(before.Spec, after.Spec), + "tablegroup's pass %s changed Shard %s's spec, the field surface its "+ + "SSA apply owns:\n before: %+v\n after: %+v", + pass, key.Name, before.Spec, after.Spec) + c.True( + equality.Semantic.DeepEqual(before.OwnerReferences, after.OwnerReferences), + "tablegroup's pass %s changed Shard %s's ownerReferences:\n before: %+v\n after: %+v", + pass, + key.Name, + before.OwnerReferences, + after.OwnerReferences, + ) + c.True(equality.Semantic.DeepEqual(before.Labels, after.Labels), + "tablegroup's pass %s changed Shard %s's labels:\n before: %+v\n after: %+v", + pass, key.Name, before.Labels, after.Labels) + } +} + +// awaitTableGroupPassAfter blocks until the tablegroup controller has completed +// a reconcile pass that started after notBefore and wrote the Shard at key, and +// returns that pass. +// +// The bar is Reconcile.Start measured against an instant captured BEFORE the +// probe write was issued, and both halves of that were arrived at by measuring +// a wrong version of it. +// +// Start, rather than a position in the log, because a record is appended when +// its pass RETURNS. A pass already in flight when the probe lands is therefore +// filed after the probe while having begun before it, so an index cursor +// admits it however freshly it was seeded. Five controllers converge this +// cluster before the first probe and leave dozens of finished passes behind, +// and a scan from index zero matched those exclusively: every iteration was +// satisfied by a pass that had started roughly half a second before the probe +// it was supposed to be reacting to. +// +// Before the write rather than after it, because a write becomes visible to +// watchers when the API server commits it, which is strictly before the +// client's own call returns. The gap is small but it is on the wrong side: the +// reacting tablegroup pass starts within a few tenths of a millisecond of the +// probe Patch returning, and measurably often starts just before it. A mark +// taken after the write returned then rejects the very pass it is waiting for, +// and since the probe is the only thing that writes this Shard once the cluster +// has converged, no later pass arrives to replace it. The wait cannot then +// succeed at any timeout, which is what made it fail about one run in three. A +// mark taken before the write has no such edge, because no pass can react to a +// write that has not been issued yet. +// +// The cost of moving the mark earlier is one API round trip of slack, in which a +// pass that did not see the probe would be accepted. That is bounded by a round +// trip instead of by the whole log, and at steady state nothing but this test +// writes the Shard, so the only candidate is a second pass caused by the +// previous iteration's probe. That pass has already been observed to completion +// before this iteration's mark is taken. +// +// Rescanning the namespace's whole log on each poll, rather than carrying a +// cursor across polls or across iterations, is deliberate for a related reason: +// the log is ordered by completion, not by start, so an index cursor and a +// start-time bar disagree about which entries are still candidates, and the +// cursor is the one that can step over the pass being waited for. The scan is +// bounded by one namespace's history and runs at most once per poll, so the +// cost of being right here is nothing worth optimising. +// +// No bookkeeping is needed to stop one iteration matching an earlier +// iteration's pass: that pass had already returned, and so had already started, +// before this iteration's mark was taken. +func awaitTableGroupPassAfter( + c *C, + ns string, + key client.ObjectKey, + notBefore time.Time, +) ctrltest.Reconcile { + c.Helper() + + var pass ctrltest.Reconcile + what := fmt.Sprintf("a tablegroup pass on Shard %s beginning after the probe", key.Name) + c.Eventually(10*time.Second, what, func() error { + for _, r := range Suite.Reconciles.InNamespace(ns) { + if r.Controller != "tablegroup" || !r.Start.After(notBefore) { + continue + } + for _, op := range Suite.Reconciles.Ops(r) { + if ctrltest.KindSuffix(op.Kind) == "Shard" && op.Key.Name == key.Name { + pass = r + return nil + } + } + } + return fmt.Errorf( + "no tablegroup pass touching Shard %s has begun since the probe write", + key.Name, + ) + }) + return pass +} + +// probeShard sets a test-owned annotation on the Shard to a fresh value and +// returns the instant just before that write was issued, which is the latest +// mark a pass reacting to it is guaranteed to start after. +// +// Returning the instant the write completed is the obvious choice and is the +// wrong one: the API server commits the write and dispatches the watch event +// before the client's Patch call returns, so the reacting pass is often already +// running by then. See awaitTableGroupPassAfter. +// +// Neither field manager's apply payload mentions the annotation (tablegroup's +// BuildShard only sets an annotation map at all when the TableGroup carries a +// project-ref annotation, which MinimalCluster's fixture never does), so it is +// a change neither manager's SSA apply will contend with or revert. It is a +// merge patch rather than a full Update, so it carries no resourceVersion +// precondition and cannot spuriously conflict with either controller's own +// concurrent write to the same object. Writing it is the only way this test +// can put a fresh, real change on the Shard once the cluster has converged, +// since a converged tablegroup no longer gets re-triggered by its own stale +// watch (see requireTableGroupPatchesAreNoOps). +func (c *C) probeShard(key client.ObjectKey, seq int) time.Time { + c.Helper() + shard := &Shard{} + c.NoError(c.Get(key, shard), "get shard %s", key) + base := shard.DeepCopy() + if shard.Annotations == nil { + shard.Annotations = map[string]string{} + } + shard.Annotations[shardProbeAnnotation] = fmt.Sprintf("%d", seq) + + issued := time.Now() + c.NoError( + c.Patch(shard, client.MergeFrom(base)), + "annotate shard %s", + key, + ) + return issued +} diff --git a/test/suite/scenario_selector_impostor_test.go b/test/suite/scenario_selector_impostor_test.go new file mode 100644 index 00000000..2ae0c28e --- /dev/null +++ b/test/suite/scenario_selector_impostor_test.go @@ -0,0 +1,643 @@ +// Selector inventory for the multigrescluster and shard controllers: every +// List call in pkg/cluster-handler/controller/multigrescluster/ and +// pkg/resource-handler/controller/shard/ that carries a label selector, the +// selector itself, and what this sweep found or judged about it. +// +// 35 List call sites in the two packages, 12 in multigrescluster and 23 in +// shard, of which 32 carry a label selector; the three that do not are +// recorded below as out of scope rather than omitted, so that the count can +// be rederived from this block. Line numbers point at the List( call, not at +// the MatchingLabels argument. Cross-checked for the other spellings a +// selector can take (MatchingLabelsSelector, HasLabels, a raw +// ListOptions{LabelSelector}): neither package uses any of them. +// +// multigrescluster (pkg/cluster-handler/controller/multigrescluster/): +// +// - reconcile_cells.go:23 CellList {cluster} +// Feeds the active/orphan diff for cells removed from spec. The eventual +// delete is gated behind the AnnotationPendingDeletion + ConditionReady- +// ForDeletion handshake (reconcile_cells.go:92-125), not a raw sweep. +// SAFE (condition-gated). Not tested here. +// +// - reconcile_databases.go:23 TableGroupList {cluster} +// Same shape as reconcile_cells.go:23, for TableGroups. SAFE (condition- +// gated); the protocol itself is already covered by +// TestReadyForDeletionProtocol in scenario_deletion_test.go. +// +// - reconcile_topology.go:245 CellList {cluster} +// Read-only: collects names of cells pending deletion, for topology +// pruning. Never mutates or deletes anything itself. SAFE. +// +// - reconcile_global.go:113 TopoServerList {cluster} +// CONFIRMED LIVE DEFECT (Defect 2 in +// tasks/multigres-operator-bugs-found-by-the-suite.md). No component +// filter, no owner-reference check: when global topology is external, +// every TopoServer carrying the cluster label is deleted, including a +// cell's own local TopoServer (component "local-topo"), which this +// selector cannot tell apart from the managed global one (component +// "global-topo"). TESTED below: +// TestSelectorImpostorGlobalTopoPruneDeletesCellOwnedLocalTopoServer. +// +// - multigrescluster_controller.go:401 CellList {cluster}, in +// handleDeletion (whole-cluster teardown). Raw delete, no owner-ref +// check. Reachable only while the cluster itself is being deleted, and +// an impostor sharing the label would first be picked up by +// reconcile_cells.go's own condition-gated path above (the cell +// controller reconciles any Cell object regardless of who owns it), +// which confounds an isolated impostor test for this exact line. +// JUDGED UNREACHABLE IN ISOLATION within this pass; not tested. See the +// task report for the reasoning in full. +// +// - multigrescluster_controller.go:419 TableGroupList {cluster}, in +// handleDeletion. Same shape and same confound as line 401, via +// reconcile_databases.go:23's condition-gated path. Not tested. +// +// - multigrescluster_controller.go:439 PersistentVolumeClaimList +// {cluster, component=toposerver}, in handleDeletion. Raw delete, no +// owner-ref check, but the function's own comment states the intent: +// these PVCs "may outlive their TopoServer when a cluster switches from +// managed to external topology," i.e. an unowned PVC with this label +// pair is the expected steady state this code exists to clean up, not +// an anomaly. SAFE BY DESIGN for this task's "should do nothing to an +// object it does not own" heuristic, because the whole point here is +// that ownership cannot be established for what it is meant to sweep. +// Not pinned as a defect; the design's blast radius (anything bearing +// these two labels, from any source, is eligible) is flagged in the +// report rather than pinned as a KnownDefect. +// +// This is SAFE while shard_controller.go:589 below is a DEFECT on what +// looks like the same evidence, an unowned PVC being acted on, and the +// two verdicts are worth reading together rather than one at a time. +// The difference is documented intent, which is the only thing that can +// separate them: :439 says an unowned PVC carrying these labels is +// precisely what it exists to sweep, whereas :589's own doc comment +// (shard_controller.go:571-574) and its call site +// (shard_controller.go:425-426) both say it exists to fix up ownerRefs +// on the shard's own PVCs across a mid-lifecycle policy change. Acting +// on a stranger is the job in one and an accident in the other. +// +// - multigrescluster_controller.go:603 MultigresClusterList, InNamespace +// only. Not a label selector (a map function for CoreTemplate/ +// CellTemplate/ShardTemplate change fanout). Out of scope. +// +// - certificate.go:394 (via pkg/util/certs.List) no label selector, +// InNamespace only. The eventual delete (certs.Prune) checks +// OwnedBy(cert, ownerUID) before deleting anything: the one place in +// this controller that already does what reconcile_global.go:113 does +// not. SAFE, and the contrast is the original bug write-up's own point. +// +// - status.go:125, :150, :220 CellList / TableGroupList / TopoServerList +// {cluster}. Purely read, to aggregate MultigresCluster.Status; nothing +// is ever mutated or deleted at these call sites. SAFE for this task's +// "does the controller act on it" question. A mislabelled object here +// would only ever skew the cluster's own reported status, a different +// risk this task's Quiet()-shaped assertion cannot express and which is +// not assessed here. +// +// shard (pkg/resource-handler/controller/shard/): +// +// - shard_controller.go:589 (reconcilePVCOwnerRefs) PersistentVolumeClaimList +// {cluster, database, tablegroup, shard} (no pool, no component). NEW +// DEFECT found by this sweep: the selector is the shard's four identity +// keys and nothing else, so it cannot tell the shard's own PVC from any +// other object carrying the same identity, and the only test applied to +// a match before adoption is whether it already carries a ref with this +// shard's UID (shard_controller.go:625-631). A PVC belonging to nobody +// fails that test, so it is adopted: SetControllerReference + Patch, +// whenever the shard's effective PVCDeletionPolicy resolves to Delete. +// +// There is a real ownership check here, and naming it correctly matters +// because it bounds the defect. ctrl.SetControllerReference returns +// AlreadyOwnedError when the object already carries a different +// controller ownerRef (controller-runtime v0.25.0, +// pkg/controller/controllerutil/controllerutil.go:97-99), so this code +// does not adopt a PVC that already belongs to someone else. What it +// adopts is a PVC with no controller owner at all. TESTED below: +// TestSelectorImpostorShardOwnerRefReconcileAdoptsUnrelatedPVC. +// +// - reconcile_deletion.go:57 DeploymentList {cluster, database, +// tablegroup, shard}, in the Shard's own handleDeletion. Raw delete, no +// owner-ref check. Same family as shard_controller.go:589. Not +// independently tested given this pass's budget. +// +// - reconcile_deletion.go:90 PodList same 4-key selector, same +// function. Raw delete, no owner-ref check. Lower interference risk +// than the multigrescluster Cell/TableGroup case above, since a Shard's +// own teardown is not cascaded through another controller's graceful +// orphan protocol. Not tested here. +// +// - reconcile_deletion.go:166 (cleanupShardPVCs) PersistentVolumeClaimList +// same 4-key selector. Every match is marked orphan or deleted with no +// owner-ref check, gated only by shardPVCShouldBeCleaned's policy read. +// Same family as shard_controller.go:589 at a different lifecycle +// point. Not independently tested. +// +// - reconcile_deletion.go:252 (handlePendingDeletion) PodList same +// 4-key selector. Runs the drain state machine (initiateDrain / +// clearDrainAnnotations / Delete) against any match once the Shard +// itself carries the PendingDeletion annotation. Same family; combining +// it with a graceful shard-level orphan flow adds the same entanglement +// seen in the multigrescluster Cell/TableGroup case. Not tested. +// +// - reconcile_data_plane.go:290, :367 PodList same 4-key selector. +// Read-only relative to the pods themselves (feeds +// shard.Status.PodRoles and posture.Evaluate). SAFE. +// +// - reconcile_data_plane.go:542 (reconcileDrainState) PodList same +// 4-key selector. MUTATES a matching pod (clears its drain annotations) +// when isDrainStale holds, which requires the pod's pool label to +// resolve to a real shard.Spec.Pools entry, a name that parses to an +// in-range replica ordinal, and a spec judged unchanged from desired. +// A real candidate, but reproducing that combination on a synthetic +// impostor is disproportionate for this pass; deferred. +// +// - reconcile_data_plane.go:673 (reconcilePoolerPrune) PodList same +// 4-key selector. The action it drives (topo.MarkDeadPoolers) writes to +// the fake topology store, not to the Kubernetes object, so this +// suite's k8s-event Stream cannot observe the outcome either way. +// UNTESTABLE WITH THIS HARNESS. +// +// - reconcile_quarantine.go:83 (reconcileQuarantineRemediation) PodList +// same 4-key selector. Deletes a pod and hard-deletes its PVC, but only +// for names the fake topology store reports as LIFECYCLE_QUARANTINED. +// Driving that requires reaching into the suite's internal topology +// registry; deferred. +// +// - disruption.go:30 PodList {cluster, database, tablegroup, shard, +// component=Pool} (adds the component key the multigrescluster prune +// lacks). Read-only (feeds canStartDisruption's decision). SAFE. +// +// - maintenance_surge.go:300, :338 PodList component+cell-scoped via +// shardPDBLabels/metadata.GetSelectorLabels. Read-only. SAFE. +// +// - postgres_config.go:266 ShardList, InNamespace only. Not a label +// selector (map function for ConfigMap-change fanout). Out of scope. +// +// - reconcile_shared_infra.go:373 PodList the PDB's own +// component+pool+cell selector. Read-only (sizes MinAvailable). SAFE. +// +// - reconcile_shared_infra.go:411 PodDisruptionBudgetList same PDB +// selector. Deletes an unmatched PDB, but only those that also pass +// metav1.IsControlledBy(pdb, shard): an explicit owner check right in +// the loop. SAFE, and the "done right" counterpart to +// shard_controller.go:589. +// +// - reconcile_pool_pods.go:50 PodList pool+cell-scoped +// (buildPoolLabelsWithCell). DEFECT of the same class as +// shard_controller.go:589, and the strongest untested candidate left in +// this inventory. The list populates existingPods +// (reconcile_pool_pods.go:69-73), which is then walked by name with no +// ownership check at all: +// +// Phase 0, syncDrainedLabels (:82, body :943-:974), iterates every map +// member and patches multigres.com/pod-role whenever +// resolvePodRole(shard, pod.Name) disagrees with the label the pod +// carries, so a pod this shard does not own has that label written or +// stripped. Phase 2, handleScaleDown (:118, body :499), classifies any +// member whose name does not parse as - (resolvePodIndex, +// :1240-:1250) or whose index is at or beyond effectiveReplicas as an +// extra pod (:537-:540), then drains (initiateDrain, :650) and deletes +// it (:567). A plausible impostor Pod carrying the pool and cell labels +// and any non-numeric name suffix is therefore drained and deleted. +// Secondary consequence at the same site: isPoolHealthy(existingPods, +// ...) (:585) counts an impostor as a pool member, so a non-ready one +// blocks legitimate scale-down of the real pool. +// +// NOT PINNED in this pass, and recorded here rather than left implied: +// a pin is a test, and this one needs a Pod-kind impostor against +// DataPlaneSim.tickPods, whose write set differs from tickPVCs's and +// has not been derived (see the TestSelectorImpostorShardOwnerRef... +// caveat below). Naming it SAFE, as an earlier revision of this block +// did, was the error worth correcting: an unexamined site is a gap, and +// a gap signed SAFE is worse than one left open. +// +// - reconcile_pool_pods.go:60 PersistentVolumeClaimList pool+cell- +// scoped. SAFE, but name-keyed rather than read-only, which is the +// accurate justification: pvcutil.ClearOrphan (:204) and +// expandPVCIfNeeded (:216) both write to list members, and what makes +// them safe is that every access is existingPVCs[pvcName] where +// pvcName comes from BuildPoolDataPVCName, a deterministic desired +// name. An impostor under any other name is never indexed, so it is +// genuinely untouched. Unlike :50 above, which walks the map itself. +// +// - reconcile_pool_pods.go:1116 PersistentVolumeClaimList pool+cell- +// scoped, counts non-orphan PVCs. Read-only. SAFE. +// +// - reload.go:69 PodList {cluster, database, tablegroup, shard, +// component=Pool}. Read-only (feeds the reload decision). SAFE. +// +// - status.go:221 PodList pool+cell-scoped (buildPoolLabelsWithCell). +// Read-only (status aggregation). SAFE. +// +// - status.go:365 PodList buildMultiorchLabelsWithCell selector. +// Read-only (crash-loop detection for status). SAFE. +// +// - reconcile_readiness.go:33 PodList {cluster, database, tablegroup, +// shard, component=Pool}. Read-only (readiness aggregation). SAFE. + +package suite + +import ( + "errors" + "fmt" + "testing" + "time" + + corev1 "k8s.io/api/core/v1" + "k8s.io/apimachinery/pkg/api/resource" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/utils/ptr" + "sigs.k8s.io/controller-runtime/pkg/client" + + multigresv1alpha1 "github.com/multigres/multigres-operator/api/v1alpha1" + topopkg "github.com/multigres/multigres-operator/pkg/data-handler/topo" + "github.com/multigres/multigres-operator/pkg/util/metadata" + "github.com/multigres/multigres-operator/pkg/util/name" + "github.com/multigres/testkit/ctrltest" +) + +// errLocalTopoDeleted is what the TopoServer script's invariant returns when +// it sees the deletion that test pins. +// +// A sentinel rather than a match on the harness's message text, because +// KnownDefect reads any non-nil error as "the pinned defect is still live". +// This pin's script can produce several other errors that are not that +// deletion, and every one of them would otherwise keep the pin green: a +// legitimate toposerver status write that the step's declaration failed to +// account for, a step that timed out because that declaration has drifted from +// what the controller now writes. Those are facts about this test, not about +// the operator, so the check body has to be able to tell them apart, and it +// cannot do that by reading a string the harness is free to reformat. +var errLocalTopoDeleted = errors.New( + "the cell's own local TopoServer was deleted", +) + +// selectorImpostorNudgeAnnotation is a key no controller's applied payload +// ever mentions, following the same reasoning as shardProbeAnnotation in +// scenario_race_test.go: tablegroup's BuildShard sets an annotation map on a Shard +// only when its TableGroup carries a project-ref annotation +// (pkg/cluster-handler/controller/tablegroup/builders.go:37-48), which +// MinimalCluster's fixture never does, so writing this key is a mutation +// neither manager's SSA apply contends with or reverts. +// +// It has to enqueue the Shard to be useful, and it does: the shard +// controller's For(&Shard{}) (shard_controller.go:693) +// carries no predicate, so a metadata-only patch is a reconcile trigger. That +// is the whole reason the key exists, since a converged Shard has nothing left +// to re-trigger it and reconcilePVCOwnerRefs only looks at an impostor on a +// pass that actually runs. +const selectorImpostorNudgeAnnotation = "scenario-selector-impostor-test.multigres.com/nudge" + +// externalGlobalTopoCluster creates a MultigresCluster whose global topology is +// external and whose one cell manages its own local TopoServer, the exact +// combination tasks/multigres-operator-external-topo-deletes-local.md +// reproduced on a live cluster: it is what makes reconcileGlobalTopoServer's +// desired-is-nil branch run on every reconcile, while still giving the cell +// controller a local TopoServer of its own to keep reapplying. +// +// It returns an error rather than calling t.Fatalf as MinimalCluster does, +// because its caller runs it inside a Script step's do: a Fatalf there would +// Goexit out of the middle of a script, whereas Script.TryStep routes a failing +// do to the test's own Fatalf and, crucially, never lets it reach KnownDefect +// as though it were evidence about the operator. +// +// The password Secret is created here rather than shared with MinimalCluster +// because the two fixtures differ in every other field; what is worth keeping +// in step is the deliberate choice to leave it unlabelled, which is what a user +// would create and what makes the reconcilers' APIReader necessary. +func (c *C) externalGlobalTopoCluster(clusterName string) error { + c.Helper() + + secret := &corev1.Secret{ + ObjectMeta: metav1.ObjectMeta{Name: adminSecretName, Namespace: c.NS}, + StringData: map[string]string{"password": "postgres"}, + } + if err := c.Create(secret); err != nil { + return fmt.Errorf("create password secret: %w", err) + } + + cluster := &MultigresCluster{ + ObjectMeta: metav1.ObjectMeta{Name: clusterName, Namespace: c.NS}, + Spec: MultigresClusterSpec{ + PostgresPasswordSecretRef: PostgresPasswordSecretRef{ + Name: adminSecretName, + Key: "password", + }, + PVCDeletionPolicy: &PVCDeletionPolicy{ + WhenDeleted: multigresv1alpha1.DeletePVCRetentionPolicy, + WhenScaled: multigresv1alpha1.DeletePVCRetentionPolicy, + }, + GlobalTopoServer: &multigresv1alpha1.GlobalTopoServerSpec{ + External: &multigresv1alpha1.ExternalTopoServerSpec{ + Endpoints: []multigresv1alpha1.EndpointUrl{ + "https://external-topo.invalid:2379", + }, + }, + }, + Cells: []CellConfig{ + { + Name: defaultSimCell, + ZoneID: "us-central1-a", + Spec: &multigresv1alpha1.CellInlineSpec{ + LocalTopoServer: &multigresv1alpha1.LocalTopoServerSpec{ + Etcd: &multigresv1alpha1.EtcdSpec{ + Replicas: ptr.To(int32(1)), + }, + }, + }, + }, + }, + }, + } + if err := c.Create(cluster); err != nil { + return fmt.Errorf("create MultigresCluster: %w", err) + } + return nil +} + +// TestSelectorImpostorGlobalTopoPruneDeletesCellOwnedLocalTopoServer pins +// Defect 2 from tasks/multigres-operator-bugs-found-by-the-suite.md: +// reconcile_global.go:113 lists TopoServers by cluster label alone (no +// component filter, no owner-reference check) and deletes every match +// whenever global topology is external. A cell's own local TopoServer +// carries that same cluster label, so it is not this controller's to +// manage, but the selector cannot tell the difference. +// +// The impostor here is not synthetic: it is the real local TopoServer the cell +// controller legitimately creates and keeps reapplying, which is exactly what +// makes the write-up call this "a permanent create/delete loop" rather than a +// one-off. That permanence is also what shapes the script below, because it +// rules out the move every other test in this suite makes first. An object +// caught in a create/delete loop never settles, so there is no converged +// namespace to open a watch onto: neither RequireQuiescent nor a poll for a +// stable TopoServer can be used here, and waiting a fixed margin for the +// toposerver controller to stop writing is a guess at how long another actor's +// work takes, which is the thing this suite exists to refuse. +// +// So the watch opens first, on an empty namespace, and the fixture is created +// inside the script's own step. Every legitimate write the toposerver +// controller then makes to the object is named as a permitted change, which +// leaves the deletion as the one event nothing accounts for, and leaves nothing +// to wait out. +func TestSelectorImpostorGlobalTopoPruneDeletesCellOwnedLocalTopoServer(t *testing.T) { + c := newCase(t) + const clusterName = "ext-global-topo" + cellResourceName := name.JoinWithConstraints( + name.DefaultConstraints, clusterName, string(defaultSimCell), + ) + localTopoName := topopkg.ManagedLocalTopoServerName(cellResourceName) + + script := c.NewScript(&TopoServerList{}) + + // The deletion is this test's entire claim, so it is asserted directly + // rather than inferred from being whatever event no step happened to + // permit. Script.TryStep and Script.TryFinish both run the invariants + // against an event before comparing it to the permitted set + // (script.go:224, :243, :294), so the delete is reported as this violation + // wherever it lands: while the step is still waiting on a status write, + // inside its settle window, or inside Finish's horizon. Combined with the + // sentinel above, that is what makes the pin's evidence the deletion on + // every run instead of whichever event happened to arrive first. + script.Invariant( + "the cell's own local TopoServer is never deleted", + func(ev ctrltest.Event) error { + if ev.Type == "deleted" && ev.Kind == "TopoServer" && + ev.Key.Name == localTopoName { + return errLocalTopoDeleted + } + return nil + }, + ) + + c.KnownDefect( + "pkg/cluster-handler/controller/multigrescluster/reconcile_global.go:113 "+ + "(external-global-topo prune selector has no component filter or "+ + "owner-reference check, so it also deletes a cell's own local TopoServer)", + func() error { + stepErr := script.TryStep( + "the cell controller creates its own local TopoServer and the "+ + "toposerver controller settles its status on it", + func() error { + return c.externalGlobalTopoCluster(clusterName) + }, + // The toposerver controller's whole settling sequence on a + // TopoServer it has just been handed, measured over four runs + // against a cluster whose global topology is managed so this + // prune never fires, which is the one way to observe what the + // object does when it is left alone: the first condition, then + // the client and peer endpoints once the etcd StatefulSet + // exists, then Ready once DataPlaneSim has ticked that + // StatefulSet ready. Three writes, in that order, then quiet + // indefinitely. + // + // The paths are the narrowest that pick out one write each. + // status.conditions[0] belongs only to the first and + // status.clientService only to the second; the third's paths + // are a subset of the second's, so it is matched by + // elimination, which is what assignEvents does a search rather + // than a greedy first match for. + // + // No ordering is declared between them even though one was + // observed, because ordering is opt-in for changes that follow + // from the code and nothing here needs it: the deletion is + // caught by the invariant above, not by an order violation. + ctrltest.Added("TopoServer", localTopoName), + ctrltest.Changed("TopoServer", localTopoName, "status.conditions[0].type"), + ctrltest.Changed("TopoServer", localTopoName, + "status.clientService", "status.peerService"), + ctrltest.Changed("TopoServer", localTopoName, "status.phase"), + ) + // TryFinish runs whatever the step returned, so that the script + // ends with Finish exactly once and the end-of-script backstop is + // satisfied on every path through this body. While the defect is + // live the step returns long before the object has finished + // settling, so the end of the script still has to be closed. + // + // The horizon is not load-bearing in either direction, which is the + // point of choosing it freely. While the defect is live nothing + // depends on it: the delete lands about 15ms after the create, + // inside the step. Once the defect is fixed this is the only window + // left in which a later prune pass could still be caught, and a + // longer horizon can only refuse more events, never permit one. + finishErr := script.TryFinish(10 * time.Second) + + switch { + case errors.Is(stepErr, errLocalTopoDeleted): + return stepErr + case errors.Is(finishErr, errLocalTopoDeleted): + return finishErr + case stepErr != nil: + // Anything else is this test's own declaration or pacing + // rather than evidence about the operator, and a pin that + // confirmed on it would survive the fix it is supposed to + // expire on. Fatalf is the right side of the line + // Script.fatalf already draws for the same reason. + c.Fatalf("the script's declaration of the toposerver "+ + "controller's settling sequence did not hold, which is a "+ + "fact about this test rather than about the prune it pins: %v", + stepErr) + case finishErr != nil: + c.Fatalf("the script's end was not quiet, and not because of "+ + "the deletion this test pins, which is a fact about this "+ + "test rather than about the prune: %v", finishErr) + } + return nil + }, + ) +} + +// TestSelectorImpostorShardOwnerRefReconcileAdoptsUnrelatedPVC pins a new +// defect found by this sweep: reconcilePVCOwnerRefs +// (shard_controller.go:589) lists PersistentVolumeClaims by the shard's four +// identity labels alone (cluster, database, tablegroup, shard; no pool, no +// component), so it cannot tell the shard's own PVC from any other object +// carrying the same identity, and the only test it applies to a match before +// adopting it is whether that match already carries a ref with this shard's +// UID. A PVC belonging to nobody fails that test and is adopted, whenever the +// shard's effective PVCDeletionPolicy resolves to Delete. +// +// The bound on the defect is worth stating precisely, because it is what a fix +// has to be aimed at. There is an ownership check in this path: +// ctrl.SetControllerReference returns AlreadyOwnedError when the object already +// carries a different controller ownerRef, so this code does not take a PVC +// that belongs to someone else. What it takes is a PVC with no controller owner +// at all. The missing check is not "does this belong to somebody else" but +// "does this belong to me", and the selector is what cannot answer it. +// +// The impostor is a bare PersistentVolumeClaim carrying just those four +// labels and no pool label, the same shape shard_controller.go's own +// "shared backup PVC" branch expects, and no owner reference at all: the +// shape a PVC left behind by some other process, or a since-recreated +// resource under the same identity, would plausibly have. +// +// RequireQuiescent runs before the script's watch opens so the baseline it +// replays is the shard's own already-settled PVCs (its pool data PVCs and +// its backup PVC), named explicitly rather than guessed: this suite's own +// discipline is that an already-populated namespace's replay is the first +// step's problem to permit, not something to dodge by racing the watch +// ahead of convergence. Unlike the TopoServer test above, that is available +// here, because the shard does converge. +// +// The impostor itself is created after the watch opens, and its own +// creation-then-bind is declared with Before: DataPlaneSim +// (pkg/ctrltest/datasim.go) binds every PersistentVolumeClaim in the cluster +// regardless of who it belongs to, as a stand-in for the volume provisioner +// envtest does not run, and that status patch (status.phase/accessModes/ +// capacity) has to be permitted explicitly or it is indistinguishable from +// the actual ownerRef adoption this test is pinning. Declaring the pair +// with Before, rather than waiting for Bound out-of-band first, is what +// keeps the watch open across the one window where the real defect could +// otherwise race in unobserved, immediately after creation and before this +// suite's own fake gets to it. +func TestSelectorImpostorShardOwnerRefReconcileAdoptsUnrelatedPVC(t *testing.T) { + c := newCase(t) + ns := c.NS + c.MinimalCluster("pvc-adopt") + + shard := c.awaitShard() + + c.RequireQuiescent(time.Second, 30*time.Second) + + existingPVCs := &corev1.PersistentVolumeClaimList{} + c.NoError(c.List(existingPVCs), "list existing PVCs") + + impostor := &corev1.PersistentVolumeClaim{ + ObjectMeta: metav1.ObjectMeta{ + Name: "impostor-shared-pvc", + Namespace: ns, + Labels: map[string]string{ + metadata.LabelMultigresCluster: shard.Labels[metadata.LabelMultigresCluster], + metadata.LabelMultigresDatabase: string(shard.Spec.DatabaseName), + metadata.LabelMultigresTableGroup: string(shard.Spec.TableGroupName), + metadata.LabelMultigresShard: string(shard.Spec.ShardName), + }, + }, + Spec: corev1.PersistentVolumeClaimSpec{ + AccessModes: []corev1.PersistentVolumeAccessMode{corev1.ReadWriteOnce}, + Resources: corev1.VolumeResourceRequirements{ + Requests: corev1.ResourceList{ + corev1.ResourceStorage: resource.MustParse("1Gi"), + }, + }, + }, + } + + allow := make([]ctrltest.Allow, 0, len(existingPVCs.Items)+1) + for _, pvc := range existingPVCs.Items { + allow = append(allow, ctrltest.Added("PersistentVolumeClaim", pvc.Name)) + } + allow = append(allow, ctrltest.Before( + ctrltest.Added("PersistentVolumeClaim", impostor.Name), + // Narrowed to status.phase rather than left open: DataPlaneSim's bind + // always touches it alongside accessModes/capacity, so naming it is + // enough to identify that write and that write only. Left open, this + // leaf would also happily absorb the ownerRef adoption this test + // exists to catch, since Changed with no paths matches any + // modification at all. + ctrltest.Changed("PersistentVolumeClaim", impostor.Name, "status.phase"), + )) + + script := c.NewScript(&corev1.PersistentVolumeClaimList{}) + + c.KnownDefect( + "pkg/resource-handler/controller/shard/shard_controller.go:589 "+ + "(reconcilePVCOwnerRefs selects on the shard's four identity labels "+ + "alone, so it adopts any PVC carrying them that has no controller "+ + "ownerRef, with nothing establishing the PVC is the shard's own)", + func() error { + stepErr := script.TryStep( + "baseline PVCs replay; the impostor is created, then bound like "+ + "any other PVC by this suite's data-plane fake", + func() error { + if err := c.Create(impostor); err != nil { + return err + } + // The impostor's own creation cannot trigger the shard's + // reconcile loop (it carries no owner reference, so + // Owns(&PersistentVolumeClaim{}) has nothing to map it back + // to), and RequireQuiescent above means nothing else is left + // to either: measured empirically, a fully quiesced shard + // does not reconcile again on its own. A metadata-only nudge + // on the Shard itself is what actually gets + // reconcilePVCOwnerRefs to run again and look at the + // impostor; this write is on ShardList, not the + // PersistentVolumeClaimList this script watches, so it needs + // no entry of its own in allow. + shardCopy := shard.DeepCopy() + patch := client.MergeFrom(shardCopy.DeepCopy()) + if shardCopy.Annotations == nil { + shardCopy.Annotations = map[string]string{} + } + shardCopy.Annotations[selectorImpostorNudgeAnnotation] = time.Now(). + UTC(). + Format(time.RFC3339Nano) + return c.Patch(shardCopy, patch) + }, + allow..., + ) + // Run unconditionally, and 10s, for the reasons given at the same + // call in the TopoServer test above. Unlike that test this one + // needs no sentinel to tell two permitted-set outcomes apart: the + // adoption is a modification of the impostor, and the only other + // modification anything makes to it is DataPlaneSim's bind, which + // the step permits by name and by path. So there is no second + // route to a non-nil error through a permitted-set mismatch. + // + // That is narrower than "no second route at all", and the + // difference matters for how much this pin can be trusted. A + // TryStep timeout is also a non-nil error, and KnownDefect reads + // any non-nil error as the defect still being live, so if the data + // plane fake never binds the impostor or a baseline PVC name + // drifts, this pin survives the operator fix that should have + // retired it. That is the general limitation of pinned steps + // stated on TryStep, and this call site is not exempt from it. The + // TopoServer test's sentinel-plus-Fatalf discrimination is the + // honest pattern if this ever needs to be tightened. + finishErr := script.TryFinish(10 * time.Second) + if stepErr != nil { + return stepErr + } + return finishErr + }, + ) +} diff --git a/test/suite/scenario_shard_lifecycle_test.go b/test/suite/scenario_shard_lifecycle_test.go new file mode 100644 index 00000000..d494a34d --- /dev/null +++ b/test/suite/scenario_shard_lifecycle_test.go @@ -0,0 +1,896 @@ +package suite + +import ( + "context" + "fmt" + "slices" + "strings" + "testing" + "time" + + corev1 "k8s.io/api/core/v1" + storagev1 "k8s.io/api/storage/v1" + apierrors "k8s.io/apimachinery/pkg/api/errors" + "k8s.io/apimachinery/pkg/api/meta" + "k8s.io/apimachinery/pkg/api/resource" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/utils/ptr" + "sigs.k8s.io/controller-runtime/pkg/client" + + multigresv1alpha1 "github.com/multigres/multigres-operator/api/v1alpha1" + "github.com/multigres/multigres-operator/pkg/resolver" + shardcontroller "github.com/multigres/multigres-operator/pkg/resource-handler/controller/shard" + "github.com/multigres/multigres-operator/pkg/util/metadata" + "github.com/multigres/multigres-operator/pkg/util/name" + "github.com/multigres/testkit/ctrltest" +) + +const lifecycleShardName multigresv1alpha1.ShardName = "0-inf" + +// blockedObservationWindow is how long steps 5 and 6 watch for a reaction +// that never comes. +// +// The shard controller requeues every disruptionRecoveryRequeue (5s, +// pkg/resource-handler/controller/shard/disruption.go:17) for as long as a +// disruption is refused, so a window several times that long is the +// difference between "the operator tried repeatedly and refused every time" +// and "the operator had not got round to trying yet". A Quiet() step on its +// own asserts silence only over the 250ms settle window, which for this +// question would be almost nothing. +const blockedObservationWindow = 20 * time.Second + +// lifecycleShardRef is a Shard carrying only the fields BuildPoolPodName, +// BuildPoolDataPVCName and BuildSharedBackupPVCName read: the cluster label +// and the three spec names. Those four values are known before the real +// Shard exists, since lifecycleCluster (below) chooses them, which lets the +// test predict a pod or PVC's name ahead of the create that produces it, +// using the operator's own name builders rather than a hand-rolled format +// string. It is never sent to the API server. +func lifecycleShardRef(clusterName string) *Shard { + return &Shard{ + ObjectMeta: metav1.ObjectMeta{ + Labels: map[string]string{metadata.LabelMultigresCluster: clusterName}, + }, + Spec: multigresv1alpha1.ShardSpec{ + DatabaseName: resolver.DefaultSystemDatabaseName, + TableGroupName: resolver.DefaultSystemTableGroupName, + ShardName: lifecycleShardName, + }, + } +} + +// lifecycleStorageClassName is the StorageClass lifecycleCluster's pool +// references, distinct per namespace so concurrently-running instances of +// this test never collide on the same cluster-scoped object. +func lifecycleStorageClassName(ns string) string { + return ns + "-lifecycle-expandable" +} + +// lifecycleCluster creates the same MultigresCluster MinimalCluster does +// (one cell, one database, one table group, one shard), plus a StorageClass +// with AllowVolumeExpansion set and the pool's Storage.Class pointed at it. +// +// Two facts force this rather than a plain call to MinimalCluster. First, +// PopulateClusterDefaults's own injection of Databases is commented +// "in-memory" for a reason: with no mutating webhook running in this suite, +// nothing ever writes it back to the MultigresCluster object itself, every +// reconcile recomputes it from scratch, and cluster.Spec.Databases stays +// permanently empty on the server unless a caller sets it explicitly. That +// alone would still allow starting from MinimalCluster's bare cluster and +// seeding Databases later, in updateLifecyclePool's first call. But second, +// storageClassName is immutable once a PersistentVolumeClaim exists, and step +// 3 needs the pool's existing data PVCs to already reference a StorageClass +// that allows expansion, or the API server's resize admission check refuses +// the request outright ("only dynamically provisioned pvc can be resized"). +// So the StorageClass has to be in place, and referenced, from this create. +// +// The shape mirrors what resolver.PopulateClusterDefaults would have +// injected for a MinimalCluster (one pool, one cell, the two-replica floor +// pkg/resolver/shard.go computes for a single-cell pool): every attribute +// this script mutates is still reached by changing a cluster that already +// exists, this just makes explicit at creation the one attribute (storage +// class) that cannot be introduced by a later mutation. +func (c *C) lifecycleCluster(clusterName string) *MultigresCluster { + c.Helper() + + scName := lifecycleStorageClassName(c.NS) + allowExpansion := true + sc := &storagev1.StorageClass{ + ObjectMeta: metav1.ObjectMeta{Name: scName}, + Provisioner: "multigres-test/no-op", + AllowVolumeExpansion: &allowExpansion, + } + c.NoError(c.Create(sc), "create StorageClass %s", scName) + // Cluster-scoped, so the per-test namespace delete does not reach it and + // every run would otherwise leave one behind in the shared envtest API + // server for the life of the package. context.Background rather than + // c.Context, which is already cancelled by the time cleanups run. + c.Cleanup(func() { + _ = c.Client().Delete(context.Background(), sc) + }) + + return c.newCluster(clusterName, func(s *MultigresClusterSpec) { + s.Databases = []DatabaseConfig{{ + Name: resolver.DefaultSystemDatabaseName, + Default: true, + TableGroups: []TableGroupConfig{{ + Name: resolver.DefaultSystemTableGroupName, + Default: true, + Shards: []ShardConfig{{ + Name: lifecycleShardName, + Spec: &ShardInlineSpec{ + Pools: map[PoolName]PoolSpec{ + resolver.DefaultPoolName: { + Type: "readWrite", + Cells: []CellName{defaultSimCell}, + ReplicasPerCell: ptr.To(int32(2)), + Storage: multigresv1alpha1.StorageSpec{Class: scName}, + }, + }, + }, + }}, + }}, + }} + }) +} + +// updateLifecyclePool re-reads the cluster and applies mutate to the +// "default" pool's spec, retrying on a conflict from a concurrent status +// write. The cluster's own status subresource is patched by the +// multigrescluster controller on a completely separate write path, but any +// spec Update still carries the resourceVersion it read, so a status patch +// landing between our Get and our Update aborts it. +func (c *C) updateLifecyclePool( + clusterName string, + mutate func(*PoolSpec), +) { + c.Helper() + key := client.ObjectKey{Namespace: c.NS, Name: clusterName} + for { + cluster := &MultigresCluster{} + c.NoError(c.Get(key, cluster), "get cluster %s", clusterName) + pools := cluster.Spec.Databases[0].TableGroups[0].Shards[0].Spec.Pools + pool := pools[resolver.DefaultPoolName] + mutate(&pool) + pools[resolver.DefaultPoolName] = pool + + err := c.Update(cluster) + if err == nil { + return + } + if !apierrors.IsConflict(err) { + c.Fatalf("update cluster %s: %v", clusterName, err) + } + } +} + +// waitFor is eventually without the t.Fatalf, returning the last error instead. +// +// A KnownDefect body has to be able to contain a wait: KnownDefect reads a +// returned error as the pinned defect still being live, and eventually ends the +// goroutine through t.Fatalf rather than returning, so a pin whose subject is +// "this never happens" cannot be written with eventually at all. +func waitFor(timeout time.Duration, cond func() error) error { + deadline := time.Now().Add(timeout) + for { + err := cond() + if err == nil { + return nil + } + if time.Now().After(deadline) { + return err + } + time.Sleep(250 * time.Millisecond) + } +} + +// drainStateSeen reports nil once any pod in ns carries the drain state +// machine's annotation, and an error while none does. +// +// This is the first write a drain makes: initiateDrain patches the annotation +// to Requested (pkg/data-handler/drain/drain_helpers.go) before anything else +// moves, and ExecuteDrainStateMachine advances it one state per reconcile from +// there. So the annotation's mere presence is the earliest observable evidence +// that a drain started, which is exactly what step 2's pin is about, and unlike +// a count of pod events it does not depend on how many reconciles the operator +// took to get anywhere. +func (c *C) drainStateSeen() error { + c.Helper() + pods := &corev1.PodList{} + if err := c.List(pods); err != nil { + return err + } + for i := range pods.Items { + if _, ok := pods.Items[i].Annotations[metadata.AnnotationDrainState]; ok { + return nil + } + } + return fmt.Errorf("no pod in %s carries %s", c.NS, metadata.AnnotationDrainState) +} + +// poolPodFingerprints maps every pod in ns to its UID and resourceVersion. +// +// Comparing two of these across a window is how this file asserts that the +// operator left the pods alone, and it replaces what a Quiet() step used to say +// about pods before pods came out of the script's watch entirely (see the +// script's own comment in TestShardLifecycle). It is not the weaker claim: +// resourceVersion is monotonic per object and moves on every write the API +// server accepts, so any patch at all, by the operator or by the data-plane +// fake, shows up as a changed fingerprint. A pod created or deleted in the +// window changes the key set, and a delete-and-recreate at the same name +// changes the UID. What it does not do is care when any of that happened, which +// is the whole reason to read pods rather than watch them: the claim is about +// the operator's writes and not about the fake's pacing. +func (c *C) poolPodFingerprints() map[string]string { + c.Helper() + pods := &corev1.PodList{} + c.NoError(c.List(pods), "list pods in %s", c.NS) + out := make(map[string]string, len(pods.Items)) + for i := range pods.Items { + p := &pods.Items[i] + out[p.Name] = string(p.UID) + "@" + p.ResourceVersion + } + return out +} + +// podFingerprintDiff describes how two poolPodFingerprints snapshots differ, or +// returns "" when they are identical. Sorted, so a failure message is the same +// text on every run rather than whatever order the map iterated in. +func podFingerprintDiff(before, after map[string]string) string { + var notes []string + for name, was := range before { + now, ok := after[name] + switch { + case !ok: + notes = append(notes, fmt.Sprintf("%s was deleted", name)) + case now != was: + notes = append(notes, fmt.Sprintf("%s was written (%s to %s)", name, was, now)) + } + } + for name := range after { + if _, ok := before[name]; !ok { + notes = append(notes, fmt.Sprintf("%s was created", name)) + } + } + slices.Sort(notes) + return strings.Join(notes, "; ") +} + +// firstCreateIndex returns the position in ops of controller's first accepted +// create of kind/name, and whether it made one. +// +// This is how step 4's ordering claim survives pods leaving the script's watch. +// The op log's order is arrival at the recorder's mutex, which recorder.go is +// explicit is program order within one controller and not a causal order across +// controllers, so two indexes are only comparable when both name the same +// controller. Both creates here are the shard controller's, in one pass of +// createMissingResources, which is what makes the comparison sound and is why +// the controller is a parameter rather than left implicit. +func firstCreateIndex(ops []ctrltest.Op, controller, kind, name string) (int, bool) { + for i, op := range ops { + if op.Controller == controller && op.Verb == "create" && + ctrltest.KindSuffix(op.Kind) == kind && op.Key.Name == name { + return i, true + } + } + return 0, false +} + +// podRoleViolation reports whether err is one of the errors MembersOf returns +// about what status.podRoles actually says, as opposed to a transient "not +// reconciled yet" state (shard not found, status.podRoles still empty) that +// the standing invariant below must not treat as a violation. +// +// The unrecognized-role error has to count, not just the two primary-count +// ones. MembersOf returns it from its classification loop, before it counts +// primaries at all, so a snapshot holding one pod in a role this suite does +// not know (a future DRAINED, say) and two pods reporting PRIMARY comes back +// as the unrecognized-role error alone. Reading that as "not a violation" +// would wave the two primaries through, which is the one thing the invariant +// exists to catch. +// +// Matching on message text is the only option MembersOf offers, since it +// exports no sentinel errors. That coupling is invisible from identity.go, so +// rewording a message there disables this check silently; closing it needs +// either sentinels in identity.go or a unit test pinning these strings, and +// both are outside this file. +func podRoleViolation(err error) bool { + if err == nil { + return false + } + msg := err.Error() + return strings.Contains(msg, "no pod has role PRIMARY") || + strings.Contains(msg, "more than one pod has role PRIMARY") || + strings.Contains(msg, "has unrecognized role ") +} + +// rollingUpdateDrift reports whether the Shard says wantPods of its pool pods +// have drifted from their desired spec, through the RollingUpdate condition +// handleRollingUpdates writes (reconcile_pool_pods.go:735-751). +// +// This is the only object state the operator changes in response to a spec +// change it then refuses to act on: the drain it would start next is blocked +// before it writes anything to a pod, and the DisruptionBlocked Event that +// refusal records goes through client-go's per-object spam filter (burst 25, +// one refill per 300s), which a chatty Shard inside a short envtest run has +// already spent. Steps 5 and 6 need this because a step asserting that +// nothing happened asserts nothing at all unless the stimulus provably +// arrived first. +// +// The expected message is built from the same format string the operator +// uses, so this is coupled to that wording. The coupling is deliberate and +// fails in the safe direction: a reworded message makes the step fail loudly +// rather than quietly stop asserting. +func rollingUpdateDrift(key client.ObjectKey, wantPods int) error { + shard := &Shard{} + if err := Suite.Client.Get(context.Background(), key, shard); err != nil { + return err + } + cond := meta.FindStatusCondition(shard.Status.Conditions, "RollingUpdate") + if cond == nil { + return fmt.Errorf("shard %s has no RollingUpdate condition yet", key) + } + want := fmt.Sprintf("%d pods need update in pool %s", wantPods, resolver.DefaultPoolName) + if cond.Status != metav1.ConditionTrue || cond.Reason != "PodsDrifted" || + cond.Message != want { + return fmt.Errorf( + "shard %s reports RollingUpdate=%s reason=%s %q, want True PodsDrifted %q", + key, cond.Status, cond.Reason, cond.Message, want, + ) + } + return nil +} + +// lifecycleResources builds a concrete, distinguishable resource request/limit +// pair so successive calls with different label values are guaranteed to +// differ from both the resolver's own defaults and from each other. +func lifecycleResources(cpuReq, memReq, cpuLim, memLim string) corev1.ResourceRequirements { + return corev1.ResourceRequirements{ + Requests: corev1.ResourceList{ + corev1.ResourceCPU: resource.MustParse(cpuReq), + corev1.ResourceMemory: resource.MustParse(memReq), + }, + Limits: corev1.ResourceList{ + corev1.ResourceCPU: resource.MustParse(cpuLim), + corev1.ResourceMemory: resource.MustParse(memLim), + }, + } +} + +// TestShardLifecycle starts from the smallest cluster this suite can +// converge (one pool, two pods, under the two-replica floor a single-cell +// pool always resolves to) and mutates it forward: resource change, storage +// change, scale up, a second resource change with three members present, +// and scale back down. Every attribute is reached by changing a cluster +// that already exists, never by constructing the end state, so this +// exercises the operator's transition paths rather than its defaulting +// paths. +func TestShardLifecycle(t *testing.T) { + c := newCase(t) + ns := c.NS + const clusterName = "lifecycle" + + shardKey := client.ObjectKey{ + Namespace: ns, + Name: name.JoinWithConstraints( + name.DefaultConstraints, + clusterName, + string(resolver.DefaultSystemDatabaseName), + string(resolver.DefaultSystemTableGroupName), + string(lifecycleShardName), + ), + } + shardRef := lifecycleShardRef(clusterName) + const cellName = defaultSimCell + + poolName := string(resolver.DefaultPoolName) + pod0 := shardcontroller.BuildPoolPodName(shardRef, poolName, cellName, 0) + pvc0 := shardcontroller.BuildPoolDataPVCName(shardRef, poolName, cellName, 0) + pod1 := shardcontroller.BuildPoolPodName(shardRef, poolName, cellName, 1) + pvc1 := shardcontroller.BuildPoolDataPVCName(shardRef, poolName, cellName, 1) + backupPVC := shardcontroller.BuildSharedBackupPVCName(shardRef) + + // Step 1: converge, out of the script's sight, and then open the script + // over what converging produced. + // + // lifecycleCluster mirrors what MinimalCluster plus the resolver's own + // defaults would produce (one cell, one pool at the two-replica floor + // pkg/resolver/shard.go computes for a single-cell pool under the default + // AT_LEAST_2 durability policy), so "the smallest possible cluster" + // already has a primary and a replica the moment it is healthy: pods 0 + // and 1, their data PVCs, and the shard's one shared backup PVC. + // + // The watch therefore opens after its own fixture, against this suite's + // standing rule that a stream is established before the objects it + // watches are created. That rule exists so a reaction to the create + // cannot land before anyone is listening, and the events it protects here + // are ones no assertion reads: the real claims of this script are steps 2 + // through 6, each about one mutation of a cluster that already exists, + // and none of them looks at how the cluster got there. + // + // What the closed step it replaces did read was the convergence sequence + // itself, and that count is not the operator's to keep. A step's + // allow-list is an exact multiset with no "N events of this kind" form, + // so a declared count is sound only where the code fixes it. Pod + // readiness here is paced by the data-plane sim (pkg/ctrltest/datasim.go) + // racing reconcilePoolerReadiness's own readiness-gate patch on the same + // Pod, and measured over runs of this test a pod settles in three status + // modifications most of the time and four sometimes, which failed step 1 + // on an unpermitted event about one run in five. A fourth speculative + // Changed leaf would only move that boundary, since nothing bounds the + // sequence at three or four either. Do not restore the closed step: it + // pinned the harness, not the operator. + // + // RequireQuiescent is what stands in its place, and over the convergence + // it is strictly the stronger claim: no projected state change on any of + // watchedKinds() and no attempted write for five seconds, rather than a + // count of Pod and PVC events. It is also what makes the baseline below + // a fixed set rather than a race, because it establishes that nothing is + // still moving when the watch opens. + c.lifecycleCluster(clusterName) + c.Eventually(60*time.Second, "shard to report Healthy", func() error { + shard := &Shard{} + if err := c.Get(shardKey, shard); err != nil { + return err + } + if shard.Status.Phase != multigresv1alpha1.PhaseHealthy { + return fmt.Errorf("shard phase is %q", shard.Status.Phase) + } + return nil + }) + c.RequireQuiescent(5*time.Second, 60*time.Second) + + // PersistentVolumeClaim and not Pod, which is the general form of what + // step 1 above ran into rather than a second workaround for it. + // + // A closed step counts events, so it is sound only over writes whose number + // follows from the operator's code. Every PVC event in this namespace is + // one of those: the operator creates each PVC once, patches it once per + // resize, and the data-plane fake's bind is a single terminal transition + // from Pending to Bound rather than a progression, so it too is one write + // by construction. Pod status is the opposite. The fake walks a pod's + // readiness conditions in however many passes it takes while + // reconcilePoolerReadiness writes its readiness gate on the same object, + // and the number of modifications that takes is a property of that race: + // three most of the time, four often enough to fail one run in five, + // measured here twice, at step 1 and again at step 4. + // + // Note what is not available as a middle road: keeping Pod in the watch + // while declining to enumerate status modifications. There is no "N events + // of this kind" form, so a watched kind's events must each be permitted by + // name or they fail the step in flight. Watching pods therefore forces the + // enumeration, which is why pods come out of the watch altogether and every + // pod-level claim below is a direct read instead. Those reads are not the + // weaker choice; see poolPodFingerprints for why a state comparison is at + // least as strong here as an event count, and firstCreateIndex for how the + // one genuine ordering claim is made from the operator's own write log. + s := c.NewScript(&corev1.PersistentVolumeClaimList{}) + + // Standing invariant: exactly one PRIMARY whenever status.podRoles is + // non-empty, no pod in a role this suite does not know, and no pod + // quarantined. MembersOf distinguishes those errors from every other read + // failure (shard not found, podRoles still empty), and podRoleViolation is + // what keeps one of those from being reported as a violation. The script + // opens after convergence, so podRoles is populated by the time this is + // registered, but steps 2 through 6 mutate the pool and the shard rewrites + // podRoles as those land, so a read taken mid-rewrite is still ordinary. + // + // Quarantined is checked explicitly because it is its own bucket in + // Members, disjoint from Replicas: a script that only ever asserted + // primary-count and replica-count could watch a pod sit quarantined for + // its entire length without ever naming it. This script never triggers + // quarantine, so any appearance here is itself something to catch. + // + // Written as a closure and registered, rather than only registered, + // because an invariant is sampled once per event and this script now + // watches one kind: the steps that matter most to it, 2 and 5 and 6, are + // steps where the operator is refused and no event arrives at all, so + // registration alone would leave those windows unsampled. requirePodRoles + // is called directly at the end of each of them. + checkPodRoles := func() error { + members, err := MembersOf(context.Background(), c.Client(), shardKey) + if err != nil { + if podRoleViolation(err) { + return err + } + return nil + } + if len(members.Quarantined) > 0 { + return fmt.Errorf( + "pod(s) unexpectedly quarantined: %s", strings.Join(members.Quarantined, ", "), + ) + } + return nil + } + requirePodRoles := func(where string) { + c.Helper() + c.Check().NoError(checkPodRoles(), "pod roles after %s", where) + } + s.Invariant( + "shard has exactly one primary and no quarantined pod", + func(_ ctrltest.Event) error { return checkPodRoles() }, + ) + + // The watch carries no starting resourceVersion, so the API server replays + // the namespace's PVCs as "added" and this step permits exactly that + // baseline. Naming the three from the operator's own name builders, rather + // than from a List taken a moment earlier, is what keeps it an assertion: + // it says the converged pool's storage is those three claims and nothing + // else, so a fourth PVC or a misnamed one fails here instead of being + // permitted by whatever happened to exist. + s.Step("the converged pool's PVCs replay as the baseline", nil, + ctrltest.Added("PersistentVolumeClaim", pvc0), + ctrltest.Added("PersistentVolumeClaim", pvc1), + ctrltest.Added("PersistentVolumeClaim", backupPVC), + ) + + s.Step("cluster settles before any change", nil, ctrltest.Quiet()) + + // The pod half of that same claim, made as a read because pods are not + // watched: the converged pool is pods 0 and 1 and nothing else. This is + // what the baseline's two Added("Pod", ...) leaves used to say. + converged := c.poolPodFingerprints() + for _, want := range []string{pod0, pod1} { + c.HasKey(converged, want, "converged pool has no pod %s", want) + } + c.Eq(2, len(converged), "want exactly %s and %s", pod0, pod1) + + // The precondition that makes step 2 meaningful: a two-member pool, one of + // them primary, which is the cohort size the defect below is about. + // + // Read under Eventually rather than once. The convergence check above is + // paced by the data plane sim at 250ms, but status.podRoles is written + // through the pooler sim at 500ms, so both pods can be ready while the + // second pod's role has not propagated yet. A single read lands in that + // window often enough to matter: observed failing one full run in six, + // reporting one primary and zero replicas. + var initialMembers Members + c.Eventually( + 30*time.Second, + "the pool to report one primary and one replica", + func() error { + members, err := MembersOf(c.Context(), c.Client(), shardKey) + if err != nil { + return err + } + if len(members.Replicas) != 1 || members.Primary == "" { + return fmt.Errorf("got %+v", members) + } + initialMembers = members + return nil + }, + ) + + // Step 2: change CPU and memory on the pool. This is written as the + // positive assertion the brief describes, and it does not pass: the + // mutation lands in the Shard spec and podNeedsUpdate correctly flags both + // pool pods as drifted (the Shard reports RollingUpdate=True + // reason=PodsDrifted, "2 pods need update in pool default"), but + // handleRollingUpdates never drains either of them. canStartDisruption + // (disruption.go) calls posture.CheckDisruption, which calls + // consensus.CheckSufficientRecruitment against the two-pooler rule + // test/suite/fakes.go's poolerSim registers; that function's own majority + // rule is len(cohort)/2+1, which for a two-member cohort is 2, so + // excluding either pod to disrupt it always leaves 1 short. This is not + // gated by DurabilityPolicy at all: fakes.go's rule sets AtLeastN(1), not + // the cluster's AT_LEAST_2 default, so the floor here comes from the + // majority check that runs before any policy-specific one, and it blocks + // a two-member pool categorically, not just under this suite's default. + // + // This step is pinned, where steps 5 and 6 below are not, because this one + // is a claim about the operator. A single-cell pool takes ReplicasPerCell 2 + // from the resolver's own default (pkg/resolver/shard.go:102-113), and the + // operator's purpose-built escape hatch for a member that cannot be + // disrupted, reconcileCellMaintenanceSurge, is gated on + // MULTI_CELL_AT_LEAST_2 with exactly two cells + // (maintenance_surge.go:349-351), so the shape the operator defaults to + // gets no surge and no other route. The pin may never expire, which is + // worth saying plainly: it expires if the majority rule changes for a + // 2-cohort, if the single-cell default moves off 2, or if the surge gate + // widens to the condition that actually triggers it, and not otherwise. + // + // The pin is that no drain ever starts, and it is written as exactly that: + // the drain state annotation never appears on any pod in the namespace. + // What it replaced was a permitted set enumerating the whole nine-event + // cycle a drain would produce on each of the two pods, which pinned the + // same defect less precisely and could not survive the readiness leaves + // three of those nine were (see step 1). Naming the annotation is the + // better pin on its own terms: initiateDrain patches it before the drain + // machine does anything else, so a drain that started and then stalled + // halfway confirms the pin today and would be indistinguishable from + // "never started" under a count of events that never arrived. + // + // The positive half comes first and is not part of the pin. A step that + // asserts nothing happened asserts nothing at all unless the stimulus + // provably arrived, so the drift condition is waited on with eventually, + // which fails the test outright rather than confirming the pin, exactly as + // KnownDefect's contract requires of a setup step. + // + // The step's own Quiet() carries the closed-world half over PVCs: a rolling + // update rewrites pods and leaves their claims alone, so the PVC silence + // here is the assertion that the operator did not take some other action + // instead of the one it refused. + s.Step("change CPU and memory on the pool", func() error { + c.updateLifecyclePool(clusterName, func(p *PoolSpec) { + p.Postgres.Resources = lifecycleResources("100m", "128Mi", "200m", "256Mi") + }) + c.Eventually( + 30*time.Second, + "the shard to report both pool pods drifted", + func() error { + return rollingUpdateDrift(shardKey, len(initialMembers.Replicas)+1) + }, + ) + podsBefore := c.poolPodFingerprints() + c.KnownDefect("shard-lifecycle-two-member-pool-never-rolls", + func() error { + return waitFor(blockedObservationWindow, func() error { + return c.drainStateSeen() + }) + }, + ) + // Stronger than the pin and independent of it: not only did no drain + // annotation appear, no pod was written to at all while we watched. + c.Check().Eq("", podFingerprintDiff(podsBefore, c.poolPodFingerprints()), + "the refused rolling update still touched pods") + requirePodRoles("the refused rolling update") + return nil + }, ctrltest.Quiet()) + + // Step 3: change storage. expandPVCIfNeeded patches the PVC's storage + // request directly; it is not drift the pod's spec hash notices (the pod + // references the PVC by name, not by size), so no pod event follows. + // This mutation is independent of step 2's never-applied one and is + // unaffected by it. + s.Step("change storage", func() error { + c.updateLifecyclePool(clusterName, func(p *PoolSpec) { + p.Storage.Size = "2Gi" + }) + return nil + }, ctrltest.Changed("PersistentVolumeClaim", pvc0), ctrltest.Changed("PersistentVolumeClaim", pvc1)) + + // Step 4: add a replica. The closed step covers the new PVC, created once + // by the operator and bound once by the data-plane fake. The new pod is + // waited on as a read, and the one genuine code-level ordering here, that + // the PVC exists before the pod that binds it, is asserted below from the + // operator's own write log rather than from the arrival order of two + // watches. + // + // This step is where the second instance of step 1's problem was measured: + // it declared three Changed("Pod", pod2) leaves for the readiness settle + // and failed on a fourth, one run in eight, with the same + // status.conditions paths. Those leaves are gone rather than widened. + pod2 := shardcontroller.BuildPoolPodName(shardRef, poolName, cellName, 2) + pvc2 := shardcontroller.BuildPoolDataPVCName(shardRef, poolName, cellName, 2) + + s.Step("add a replica", func() error { + c.updateLifecyclePool(clusterName, func(p *PoolSpec) { + p.ReplicasPerCell = ptr.To(int32(3)) + }) + c.Eventually(30*time.Second, "new pod ready", func() error { + pod := &corev1.Pod{} + key := client.ObjectKey{Namespace: ns, Name: pod2} + if err := c.Get(key, pod); err != nil { + return err + } + for _, cond := range pod.Status.Conditions { + if cond.Type == corev1.PodReady && cond.Status == corev1.ConditionTrue { + return nil + } + } + return fmt.Errorf("pod %s not Ready yet", pod2) + }) + return nil + }, + ctrltest.Added("PersistentVolumeClaim", pvc2), + ctrltest.Changed("PersistentVolumeClaim", pvc2), + ) + + // The ordering claim, from the write log: createMissingResources creates + // the data PVC and then the pod that mounts it, in that order, within one + // pass (reconcile_pool_pods.go:178-198 and the Create that follows it). + // Both are the shard controller's writes, which is what makes their + // relative position in the log program order rather than arrival noise; + // see firstCreateIndex. + ops := Suite.Ops.OpsInNamespace(ns) + pvcAt, pvcCreated := firstCreateIndex(ops, "shard", "PersistentVolumeClaim", pvc2) + podAt, podCreated := firstCreateIndex(ops, "shard", "Pod", pod2) + c.Check().True(pvcCreated, "the shard controller never created PVC %s", pvc2) + c.Check().True(podCreated, "the shard controller never created pod %s", pod2) + // Only meaningful once both exist. firstCreateIndex reports a miss as + // index 0, so an absent pod would otherwise read as one created before + // its PVC, and the run would carry a confident ordering complaint about + // an object that was never created. The switch this replaced got that + // right by being mutually exclusive; three independent checks have to say + // it explicitly. + if pvcCreated && podCreated { + c.Check().True(pvcAt <= podAt, + "pod %s was created before the PVC %s it binds (ops %d and %d)", + pod2, pvc2, podAt, pvcAt) + } + + requirePodRoles("the scale-up") + + // The new pod must bind the new PVC, not either existing one: resolved + // through ShardPVCOf against the live pod rather than assumed from the names + // above. + c.Eventually(30*time.Second, "new pod bound to a PVC", func() error { + _, err := ShardPVCOf(c.Context(), c.Client(), ns, pod2) + return err + }) + boundTo, err := ShardPVCOf(c.Context(), c.Client(), ns, pod2) + c.NoError(err, "ShardPVCOf(%s)", pod2) + // One comparison, not two: pvc0, pvc1 and pvc2 are distinct names, so + // "it is pvc2" already says "it is not either existing PVC", and a second + // check against those two could never fire. + c.Eq(pvc2, boundTo, "pod %s is bound to the wrong PVC, want the new PVC rather than %s or %s", + pod2, pvc0, pvc1) + + // Steps 5 and 6 do not assert the guarantee this script was written to + // assert, and say so rather than implying otherwise. + // + // That guarantee, documented at + // pkg/resource-handler/controller/shard/reconcile_pool_pods.go:711, is + // that handleRollingUpdates drains drifted pods one at a time, replicas + // before the primary, with the primary's switchover requested before it is + // touched, and that handleScaleDown removes an extra replica by draining + // it. Nothing in this harness can observe any of it, because no drain ever + // starts at any cohort size. canStartDisruption calls + // posture.CheckDisruption, whose revocation check + // (consensus.CheckSufficientRecruitment) requires that the pod being + // excluded cannot satisfy the durability policy on its own, and + // test/suite/fakes.go's poolerSim registers DurabilityPolicy: + // topoclient.AtLeastN(1) regardless of the shard's configured policy, so + // any single excluded pod satisfies it alone and the check refuses every + // exclusion. handleScaleDown calls the identical gate + // (reconcile_pool_pods.go:629), which is why step 6 is in the same + // position as step 5. + // + // Neither step is a KnownDefect, and that is the point. A pin records a + // live operator defect and expires on the day the operator is fixed; this + // is a harness fidelity gap, so a pin here could never expire, which makes + // it a suppression wearing a pin's clothes. Nor is threading the shard's + // real policy through fakes.go known to be enough to reach these steps: at + // AtLeastN(2) the failure moves earlier instead, the shard loses its + // PRIMARY from status.podRoles altogether, the standing invariant above + // fires during step 4, and the operator hot-loops "No primary in podRoles, + // requeueing to re-read topology". Whether that is further harness + // infidelity (the fake models no election and no promotion, and registers + // each pooler's routing role once, at first sight) or an operator defect at + // AT_LEAST_2 with three members is unresolved, and settling it is the + // prerequisite for writing these two steps for real. + // + // What is left is the pair of facts this harness can establish, asserted + // positively: the operator sees the change, and then does nothing about it + // for as long as we watch. Both halves carry weight. Without the first the + // silence would also be satisfied by a mutation that never reached the + // shard controller at all, which is the vacuous assertion this suite + // exists to refuse; without the second there is no tripwire. The day + // either half changes, for a harness reason or an operator one, these + // steps fail and force the question open again. + // + // One note for whoever picks that up, because it is not obvious from the + // guarantee's wording: the switchover half has no Pod or PVC footprint to + // assert even in principle. handleRollingUpdates "requests a switchover" + // by calling the same initiateDrain annotation patch it uses for a replica + // (drain_helpers.go:58) and recording a RollingUpdateStarted Event on the + // Shard. So the only Kubernetes-visible difference between draining the + // primary and draining a replica is an Event, on an object this script + // does not watch, delivered over the spam-filtered path rollingUpdateDrift + // describes. Asserting it needs a different observation, not a better + // permitted set. + + // Step 5: change CPU and memory again, now with three members. Resolved + // through MembersOf rather than assumed from index, both as the + // precondition that makes the step meaningful (silence about a rolling + // update is only interesting over a pool that really does hold three + // members, one of them primary) and because the drifted-pod count + // asserted below is derived from it. + members, err := MembersOf(c.Context(), c.Client(), shardKey) + c.NoError(err, "MembersOf before step 5") + c.Eq(2, len(members.Replicas), "want two replicas before step 5, got %+v", members) + c.NotEq("", members.Primary, "want a primary before step 5, got %+v", members) + poolPods := len(members.Replicas) + 1 + + s.Step("change CPU and memory again, now with three members", func() error { + podsBefore := c.poolPodFingerprints() + c.updateLifecyclePool(clusterName, func(p *PoolSpec) { + p.Postgres.Resources = lifecycleResources("150m", "192Mi", "300m", "384Mi") + }) + // The count is what makes this non-vacuous. Step 2's change was never + // applied either, so two pods have been drifted since then and a bare + // "RollingUpdate is True" would have been satisfied before this step + // ran; only the third pod, created at step 4 from the then-current + // spec, drifts because of this mutation. + c.Eventually( + 30*time.Second, + "the shard to report every pool pod drifted", + func() error { + return rollingUpdateDrift(shardKey, poolPods) + }, + ) + // Slept inside the step's own action rather than after it, so the + // closed world covers the whole window: any event the operator + // produces while we wait is buffered by the stream and fails the + // Quiet() below as an unpermitted change. + time.Sleep(blockedObservationWindow) + // The pod half of the same silence, which the Quiet() cannot carry now + // that pods are read rather than watched. Snapshotted before the + // mutation rather than after the drift wait, so the window compared + // here is the whole step and not just its tail, which is the span the + // Quiet() covers. + c.Check().Eq("", podFingerprintDiff(podsBefore, c.poolPodFingerprints()), + "the refused rolling update still touched pods") + requirePodRoles("the refused three-member rolling update") + return nil + }, ctrltest.Quiet()) + + // Step 6: scale the pool back down, which should drain and remove one + // replica. Same shape as step 5: prove the request reached the Shard the + // controller reconciles, then watch it be refused. + // + // When this becomes assertable, do not resolve the pod being removed as + // the highest-named replica. selectShardScaleDownPod (disruption.go:144) + // selects on the pod index being at or above the desired replica count, + // and the two only agree here because this fake always elects the + // lowest-indexed pod primary: with pod 2 primary, the highest-named + // replica is pod 1 and the operator would remove pod 2. Derive it from the + // index, confirm through MembersOf that it is not the primary, and take + // its PVC from ShardPVCOf. The permitted set then wants the PVC's change before + // the pod's deletion, not after: cleanupDrainedPod patches the data PVC's + // orphan mark (reconcile_pool_pods.go:560) before the Delete at :567, and + // it patches rather than deletes because orphanByRemainingCount holds + // while the pool has three data PVCs and the threshold is 3 + // (shard_controller.go:49). + membersBeforeScaleDown, err := MembersOf(c.Context(), c.Client(), shardKey) + c.NoError(err, "MembersOf before step 6") + c.Eq(2, len(membersBeforeScaleDown.Replicas), + "want two replicas before step 6, got %+v", membersBeforeScaleDown) + c.NotEq("", membersBeforeScaleDown.Primary, + "want a primary before step 6, got %+v", membersBeforeScaleDown) + + s.Step("scale the pool back down by one replica", func() error { + podsBefore := c.poolPodFingerprints() + c.updateLifecyclePool(clusterName, func(p *PoolSpec) { + p.ReplicasPerCell = ptr.To(int32(2)) + }) + // Scale-down writes no condition of its own, so what is proved here is + // narrower than step 5's: the desired count reached the Shard spec, + // which is the input handleScaleDown reads, while three pods are still + // live for it to act on. + c.Eventually( + 30*time.Second, + "the scale-down to reach the Shard spec", + func() error { + shard := &Shard{} + if err := c.Get(shardKey, shard); err != nil { + return err + } + pool, ok := shard.Spec.Pools[resolver.DefaultPoolName] + if !ok { + return fmt.Errorf("shard %s has no %q pool", shardKey, resolver.DefaultPoolName) + } + if got := ptr.Deref(pool.ReplicasPerCell, -1); got != 2 { + return fmt.Errorf("pool %q wants %d replicas per cell, not 2", + resolver.DefaultPoolName, got) + } + return nil + }, + ) + time.Sleep(blockedObservationWindow) + // The claim step 6 exists to make, in the only form this harness can + // state it while the gate refuses every exclusion: no pod was removed, + // and no pod was written to either. When the gate opens, the primary's + // pod and PVC must still be here and the replica's must not, which is + // what the note above is about; the fingerprint comparison is the half + // of that which is assertable today, and it is resolved from the live + // pods rather than from the names, so it does not assume which pod the + // operator would have picked. + c.Check().Eq("", podFingerprintDiff(podsBefore, c.poolPodFingerprints()), + "the refused scale-down still touched pods") + requirePodRoles("the refused scale-down") + return nil + }, ctrltest.Quiet()) + + s.Finish(time.Second) +} diff --git a/test/suite/scenario_shard_quiescence_test.go b/test/suite/scenario_shard_quiescence_test.go new file mode 100644 index 00000000..dedc79ef --- /dev/null +++ b/test/suite/scenario_shard_quiescence_test.go @@ -0,0 +1,26 @@ +package suite + +import ( + "testing" + "time" +) + +// TestShardStatusQuiesces pins the fix for the shard status hot loop. It was +// written red, against the defect described below, and went green when the two +// server-side-apply defects behind it were fixed. +// +// A healthy Shard should stop writing once its status reflects reality, and it +// did not: two server-side-apply defects fought each other forever, so +// status.orchReady and status.poolsReady flipped false/true and the +// StorageClassValid condition's message alternated between two strings, each +// several times a second, with no terminal state. Keeping the measurement here +// is the point of the test: a status that converges is the property, and these +// are the fields that used to prove it did not. +func TestShardStatusQuiesces(t *testing.T) { + c := newCase(t) + cluster := c.MinimalCluster("quiesce") + + c.WaitForClusterHealthy(cluster) + + c.RequireQuiescent(10*time.Second, 30*time.Second) +} diff --git a/test/suite/scenario_thrash_test.go b/test/suite/scenario_thrash_test.go new file mode 100644 index 00000000..71d14d63 --- /dev/null +++ b/test/suite/scenario_thrash_test.go @@ -0,0 +1,416 @@ +package suite + +import ( + "fmt" + "testing" + "time" + + corev1 "k8s.io/api/core/v1" + "k8s.io/utils/ptr" + "sigs.k8s.io/controller-runtime/pkg/client" + + shardcontroller "github.com/multigres/multigres-operator/pkg/resource-handler/controller/shard" + "github.com/multigres/multigres-operator/pkg/util/metadata" +) + +// TestThrash adds and removes things faster than the operator can converge, +// then asserts it lands in the right end state and stops. Both subtests +// assert only two things: what the namespace looks like once it settles, and +// that it does settle at all. +// +// Neither subtest uses a Script. A Script's steps are a closed-world +// assertion of one interleaving of events, and thrash does not produce one +// interleaving: which reconciler wins a given race, how many redundant SSA +// applies land, and how many requeues fire along the way are all legitimately +// unpredictable once changes are issued faster than the operator can react to +// them. A Script step here would be asserting an ordering nobody can +// guarantee, which is exactly the kind of flake this suite exists to stop +// writing. So the claim is only about the end state and about quiescence, +// made with ordinary reads and RequireQuiescent, not about the path taken to +// get there. Do not "improve" this into a Script. +func TestThrash(t *testing.T) { + t.Run("pool replica thrash", testPoolReplicaThrash) + t.Run("child CR thrash", testChildCRThrash) +} + +// thrashPoolName matches resolver.DefaultPoolName, the name the operator +// gives the pool it injects when a cluster specifies none. Importing +// pkg/resolver for one string constant did not seem worth the dependency, so +// this is a literal deliberately kept next to its cross-check. +const thrashPoolName = PoolName("default") + +// testPoolReplicaThrash scales one pool 1 -> 2 -> 1 -> 2 with no wait between +// changes, then requires the namespace to go quiet and every pool PVC to be +// bound to a live pod. +// +// It does not assert on this thrashed shard's status.podRoles: whether the +// last scale-up's role lands there is a race against the defect that +// pinPoolScaleUpRoleStale constructs deterministically on a namespace of its +// own, so asserting it here would only sample that race. +func testPoolReplicaThrash(t *testing.T) { + c := newCase(t) + cluster := c.poolThrashCluster("pool-thrash", 1) + c.WaitForClusterHealthy(cluster) + + // Fired back to back, not waited on between calls: each Patch is an + // unconditional merge patch computed against the object's state as of the + // previous call in this loop, so it lands regardless of what the operator + // has or hasn't done with the prior one yet. That is the thrash. + for _, n := range []int32{2, 1, 2} { + c.scalePoolTo(cluster, n) + } + + // MGO-POOL-SCALEUP-ROLE-STALE: once a shard reconciled to Healthy with N + // poolers, raising replicasPerCell to add an (N+1)th sometimes never gets + // that pooler's role into shard.Status.PodRoles, permanently. + // + // Whether it bites depends on whether the pooler registers before or + // after the reconcile that declares the shard converged, so sampled + // naturally it reproduces about a third of the time. The pin constructs + // that ordering instead, which makes it deterministic. + pinPoolScaleUpRoleStale(t) + + c.RequireQuiescent(10*time.Second, 90*time.Second) + + // The end state, asserted on this shard rather than inferred from the + // pin's namespace. Pod readiness is a Kubernetes-level fact that does + // not travel through shard.Status.PodRoles, so this is immune to the defect + // pinned above: without it a lost or reverted final scale-up settles at one + // pod, one bound PVC and a quiet namespace, and every other assertion here + // is satisfied by that. + live := c.liveReadyPoolPodNames() + c.Check().Len(live, 2, "ready pods") + + c.requireNoOrphanedPoolPVCs(live) +} + +// pinPoolScaleUpRoleStale pins that scaling a pool from one to two does not +// land the new pooler's role in status.podRoles when the pooler registers +// after the shard has already converged. +// +// On its own namespace and its own cluster, with no thrash, because the +// defect never needed one: a bare single scale-up reproduces it, and the +// thrash above only found it first. +// +// Nothing wakes the shard once it is Healthy: a registration is a write to +// the topology store, with no Kubernetes event behind it, and the shard does +// not requeue itself while a managed pod is still awaiting its pooler. The fix +// is that requeue; with it the role lands well inside a second here, because +// the suite compresses requeues, so the window below is generous. +func pinPoolScaleUpRoleStale(t *testing.T) { + t.Helper() + + c := newCase(t) + cluster := c.poolThrashCluster("scaleup", 1) + c.WaitForClusterHealthy(cluster) + key := c.shardKey() + + // Hold registration before scaling, so the second pooler cannot register + // until this test says so. Without the hold the defect's precondition, a + // shard that converged having seen fewer poolers than pods, arrives only + // when registration loses a race against the last reconcile. Measured + // 2026-09-19 by disabling the fix and running this six times: three runs + // caught the regression and three did not. Constructing the state instead + // of waiting for it takes that from roughly half to always. + release := poolers.HoldRegistrations(c.NS) + defer release() + + c.scalePoolTo(cluster, 2) + + // The precondition itself, waited on rather than assumed: two pool pods + // exist and the shard has settled on a PodRoles that knows about one. A + // release before this point would prove nothing, because the reconcile + // that notices the new pooler might be one the scale-up was going to + // trigger anyway. + c.Eventually( + 60*time.Second, + "the shard to converge having seen fewer poolers than pods", + func() error { + pods := &corev1.PodList{} + if err := c.List(pods, client.MatchingLabels{ + metadata.LabelMultigresPool: string(thrashPoolName), + }); err != nil { + return err + } + if len(pods.Items) != 2 { + return fmt.Errorf("want 2 pool pods, got %d", len(pods.Items)) + } + shard := &Shard{} + if err := c.Get(key, shard); err != nil { + return err + } + if len(shard.Status.PodRoles) != 1 { + return fmt.Errorf("want 1 pod role while held, got %d", len(shard.Status.PodRoles)) + } + return nil + }, + ) + // Deliberately not RequireQuiescent. A held namespace never goes quiet + // once the fix is in, because the requeue this pins is firing on its + // backoff the whole time, so quiescence holds before the fix and cannot + // after it. A stability window is true either way: it confirms the shard + // has settled on one role rather than being mid-pass, which is all the + // release needs. + stable := time.Now().Add(3 * time.Second) + for time.Now().Before(stable) { + shard := &Shard{} + c.NoError(c.Get(key, shard), "read the shard while registration is held") + c.Eq(1, len(shard.Status.PodRoles), + "a held pooler reached status.podRoles, so the hold is not holding") + time.Sleep(250 * time.Millisecond) + } + + // Now the pooler appears, with no Kubernetes event to announce it: a + // registration is a write to etcd. Only a requeue the operator asked for + // itself can notice, which is the thing this pins. + release() + + // Retire this pin by replacing it with c.Eventually on the same condition. + c.KnownDefect("MGO-POOL-SCALEUP-ROLE-STALE", func() error { + var members Members + deadline := time.Now().Add(20 * time.Second) + for { + var err error + members, err = MembersOf(c.Context(), c.Client(), key) + // A read failure is the check's own setup failing, not the + // defect, so it fails the test rather than keeping the pin green. + c.NoError(err, "read shard members") + if len(members.Replicas) == 1 && len(members.Quarantined) == 0 { + return nil + } + if time.Now().After(deadline) { + break + } + time.Sleep(250 * time.Millisecond) + } + return fmt.Errorf( + "the scaled-up pooler registered after the shard converged and its role "+ + "never reached status.podRoles within 20s: got %+v", members) + }) +} + +func (c *C) poolThrashCluster( + name string, + replicasPerCell int32, +) *MultigresCluster { + c.Helper() + return c.newCluster(name, func(s *MultigresClusterSpec) { + s.Databases = []DatabaseConfig{{ + Name: "postgres", + Default: true, + TableGroups: []TableGroupConfig{{ + Name: "default", + Default: true, + Shards: []ShardConfig{{ + Name: "0-inf", + Spec: &ShardInlineSpec{ + Pools: map[PoolName]PoolSpec{ + thrashPoolName: poolSpecWithReplicas(replicasPerCell), + }, + }, + }}, + }}, + }} + }) +} + +func poolSpecWithReplicas(n int32) PoolSpec { + return PoolSpec{ + Type: "readWrite", + Cells: []CellName{defaultSimCell}, + ReplicasPerCell: ptr.To(n), + } +} + +// scalePoolTo patches cluster's pool to n replicas per cell via a merge patch +// against cluster's own in-memory state, not a fresh read of the server. That +// makes each call in a back-to-back thrash loop independent of whatever the +// operator has done with the previous one: the patch always states the full +// desired pool spec, so it lands regardless of the server's current state. +func (c *C) scalePoolTo(cluster *MultigresCluster, n int32) { + c.Helper() + base := cluster.DeepCopy() + pools := cluster.Spec.Databases[0].TableGroups[0].Shards[0].Spec.Pools + pools[thrashPoolName] = poolSpecWithReplicas(n) + c.NoError( + c.Patch(cluster, client.MergeFrom(base)), + "scale pool %s to %d replicas per cell", + thrashPoolName, + n, + ) +} + +// liveReadyPoolPodNames lists Ready pods belonging to the thrashed pool. +// +// This deliberately does not go through MembersOf/shard.Status.PodRoles: that +// path is exactly what MGO-POOL-SCALEUP-ROLE-STALE (see +// testPoolReplicaThrash) breaks, and a pod being live is a Kubernetes-level +// fact independent of whether the operator's own role bookkeeping has caught +// up to it. +func (c *C) liveReadyPoolPodNames() []string { + c.Helper() + pods := &corev1.PodList{} + c.NoError( + c.List(pods, client.MatchingLabels{metadata.LabelMultigresPool: string(thrashPoolName)}), + "list pool pods", + ) + var names []string + for i := range pods.Items { + if podReady(&pods.Items[i]) { + names = append(names, pods.Items[i].Name) + } + } + return names +} + +func podReady(p *corev1.Pod) bool { + for _, c := range p.Status.Conditions { + if c.Type == corev1.PodReady { + return c.Status == corev1.ConditionTrue + } + } + return false +} + +// requireNoOrphanedPoolPVCs asserts every PVC belonging to the thrashed pool +// is bound to one of livePods. It resolves the binding with ShardPVCOf, reading it +// off each live pod's own volumes, rather than reconstructing a PVC name from +// the pool/cell/ordinal and asserting the two strings match: a pod bound to +// the wrong PVC would pass a name-arithmetic check and fail this one. +func (c *C) requireNoOrphanedPoolPVCs(livePods []string) { + c.Helper() + + bound := map[string]bool{} + for _, pod := range livePods { + pvcName, err := ShardPVCOf(c.Context(), c.Client(), c.NS, pod) + if err != nil { + c.Fatalf("resolve PVC bound to live pod %s: %v", pod, err) + } + bound[pvcName] = true + } + + pvcs := &corev1.PersistentVolumeClaimList{} + c.NoError( + c.List(pvcs, client.MatchingLabels{metadata.LabelMultigresPool: string(thrashPoolName)}), + "list pool PVCs", + ) + for _, pvc := range pvcs.Items { + if !bound[pvc.Name] { + c.Errorf( + "PVC %s belongs to pool %s but is not bound to any live pod; live pods: %v", + pvc.Name, thrashPoolName, livePods, + ) + } + } +} + +// testChildCRThrash deletes a Shard out from under its TableGroup and lets +// the parent recreate it, three times in a row, then requires the shard to +// reconverge and the namespace to go quiet. +// +// This is the likeliest spot in the wave to find a live defect: the +// ReadyForDeletion protocol between the shard and tablegroup controllers is +// already the subject of two filed defects (a vacuous ReadyForDeletion, and +// whole-cluster teardown skipping the drain), and it found a third here, pinned +// below. +func testChildCRThrash(t *testing.T) { + c := newCase(t) + ns := c.NS + cluster := c.MinimalCluster("cr-thrash") + c.WaitForClusterHealthy(cluster) + + key := c.shardKey() + shard := &Shard{} + + for i := 0; i < 3; i++ { + c.NoError(c.Get(key, shard), "cycle %d: get shard %s", i, key.Name) + oldUID := shard.UID + c.NoError(c.Delete(shard), "cycle %d: delete shard %s", i, key.Name) + + // The only wait in this loop: for the parent to have recreated a + // replacement (a new UID at the same name), which is a mechanical + // precondition for the next delete to hit a live object rather than a + // no-op against one already gone. It is not a wait for the replacement + // to converge, and the loop does not wait for that before deleting + // again: that is the thrash. + what := fmt.Sprintf("the tablegroup to recreate Shard %s after delete #%d", key.Name, i+1) + c.Eventually(30*time.Second, what, func() error { + got := &Shard{} + if err := c.Get(key, got); err != nil { + return err + } + if got.UID == oldUID { + return fmt.Errorf("shard %s not yet recreated", key.Name) + } + return nil + }) + } + + c.NoError(c.Get(key, shard), "get final shard incarnation") + + c.WaitForClusterHealthy(cluster) + + c.Eventually(60*time.Second, "the shard to report one primary and one replica", + func() error { + members, err := MembersOf(c.Context(), c.Client(), key) + if err != nil { + return err + } + if len(members.Replicas) != 1 || len(members.Quarantined) != 0 { + return fmt.Errorf( + "want 1 primary + 1 replica + 0 quarantined, got %+v", members, + ) + } + return nil + }, + ) + + c.RequireQuiescent(10*time.Second, 90*time.Second) + + // A second live defect, found by this test: the shared backup PVC never + // has its orphan label cleared when a torn-down Shard's replacement + // reclaims it. + // + // reconcileSharedBackupPVC (reconcile_shared_infra.go) reapplies the PVC by + // server-side apply from BuildSharedBackupPVC's payload, which never + // mentions multigres.com/orphan-since, so SSA leaves that label exactly as + // cleanupShardPVCs (reconcile_deletion.go) left it during the prior + // teardown: marked orphan. Contrast the per-pool data PVC path, which + // explicitly calls pvcutil.ClearOrphan on reuse + // (reconcile_pool_pods.go:204). The shared backup PVC has no equivalent + // call anywhere in the shard controller. + // + // Net effect: after any teardown-and-recreate of a Shard whose backup PVC + // survives (WhenDeleted=Delete, which MinimalCluster sets, still only + // orphans rather than deletes it in-line, because resolvePodIndex cannot + // parse an ordinal out of a backup PVC's name-hash suffix and the !hasIndex + // arm short-circuits before pvcOrphanReplicasThreshold is consulted at + // all), the backup PVC is left labeled orphan + // forever, even though it is immediately reclaimed and stays in active use + // by the reconverged, healthy shard. The multigres-gc CronJob acts on + // exactly that label, so in a real cluster this is a live backup volume + // scheduled for deletion out from under a running shard. + backupPVCKey := client.ObjectKey{ + Namespace: ns, + Name: shardcontroller.BuildSharedBackupPVCName(shard), + } + pvc := &corev1.PersistentVolumeClaim{} + c.NoError( + c.Get(backupPVCKey, pvc), + "get shared backup PVC %s", + backupPVCKey.Name, + ) + c.KnownDefect("MGO-BACKUP-PVC-ORPHAN-STALE", func() error { + since, stale := pvc.Labels[metadata.LabelOrphan] + if !stale { + return nil + } + return fmt.Errorf( + "shared backup PVC %s still carries %s=%s from an earlier teardown, though the "+ + "shard that owns it (uid %s) has reconverged healthy: reconcileSharedBackupPVC's "+ + "server-side apply never clears the label on reuse, unlike the per-pool data PVC "+ + "path (pvcutil.ClearOrphan in reconcile_pool_pods.go)", + backupPVCKey.Name, metadata.LabelOrphan, since, shard.UID, + ) + }) +} diff --git a/test/suite/scenario_transitions_test.go b/test/suite/scenario_transitions_test.go new file mode 100644 index 00000000..74a5986d --- /dev/null +++ b/test/suite/scenario_transitions_test.go @@ -0,0 +1,466 @@ +package suite + +import ( + "fmt" + "maps" + "strings" + "testing" + "time" + + corev1 "k8s.io/api/core/v1" + "sigs.k8s.io/controller-runtime/pkg/client" + + multigresv1alpha1 "github.com/multigres/multigres-operator/api/v1alpha1" + "github.com/multigres/testkit/ctrltest" +) + +// TestTransitions is the round-trip suite: for a handful of optional +// MultigresCluster fields, set the field, then unset it, and require that the +// cluster has nothing left to do. No subtest enumerates what cleanup it +// expects to see; the closing assertion fails on any activity at all, whatever +// it is, which is what finds a leak without anyone having to name it first. +// +// That closing assertion takes one of two forms below, and the choice is per +// field rather than stylistic. Where the field's consequence lands on a plain +// corev1 kind whose event count is deterministic, it is a Script step +// permitting nothing but Quiet(). Where it does not, it is RequireQuiescent, +// which makes the same closed-world claim over the twelve kinds of +// watchedKinds() plus the write recorder rather than over the one or two kinds +// a Script could usefully watch, and which opens and closes its own watch +// instead of needing one opened before the fixture exists. A Script opened +// late over two kinds for a second is strictly weaker than the primitive it +// would be standing in for, so a field that cannot use the first form uses the +// second rather than a token version of the first. +// +// Each subtest also compares state directly across the round trip, which is +// not redundant with either form. An object created during the "set" and then +// left alone emits no event on the way back, so residue that manifests as the +// absence of an expected deletion is invisible to an event stream by +// construction, whichever kinds it watches. +// +// How far that comparison reaches differs per subtest, and none of them reach +// every kind. Backup compares PVC names and storage requests, DurabilityPolicy +// compares the TableGroup and Shard mirrors, and PVCDeletionPolicy compares its +// own field plus the namespace's PVC requests. Residue of a kind a subtest does +// not read still passes it: an orphaned ConfigMap or Service would pass all +// three. +// +// The field list came from api/v1alpha1/multigrescluster_types.go rather than +// from this task's brief, as instructed. Three of the four named fields exist +// as optional fields on MultigresClusterSpec and are covered below: +// PVCDeletionPolicy, Backup and DurabilityPolicy. The fourth, "a pool's +// Replicas", does not exist under that name: PoolSpec (shard_types.go) has no +// Replicas field, only ReplicasPerCell *int32. Per this task's brief, a field +// that does not exist under the given name is reported rather than silently +// substituted, so there is no fourth subtest here. +func TestTransitions(t *testing.T) { + t.Run("PVCDeletionPolicy", testPVCDeletionPolicyRoundTrip) + t.Run("Backup", testBackupRoundTrip) + t.Run("DurabilityPolicy", testDurabilityPolicyRoundTrip) +} + +// updateCluster applies mutate to a fresh read of the cluster and writes it +// back, failing the test rather than returning an error: a write that does not +// land is this file's own setup breaking, never an observation about the +// operator. +func (c *C) updateCluster( + cluster *MultigresCluster, + mutate func(*MultigresCluster), +) { + c.Helper() + got := &MultigresCluster{} + c.NoError(c.Get(client.ObjectKeyFromObject(cluster), got), "get cluster") + mutate(got) + c.NoError(c.Update(got), "update cluster") +} + +// pvcRequests reads every PVC in ns with the storage request it carries, for +// the snapshot-and-compare half of a round trip. Comparing the whole map +// catches a PVC that appeared, a PVC that went away and was never recreated, +// and a request that moved and stayed moved, none of which the event stream +// can report once the object stops changing. +func (c *C) pvcRequests() map[string]string { + c.Helper() + pvcs := &corev1.PersistentVolumeClaimList{} + c.NoError(c.List(pvcs), "list PVCs") + out := make(map[string]string, len(pvcs.Items)) + for _, pvc := range pvcs.Items { + out[pvc.Name] = pvc.Spec.Resources.Requests.Storage().String() + } + return out +} + +// shardBackupSizes reads the backup storage size each Shard has resolved. That +// is the value BuildSharedBackupPVC applies the shared backup PVC from +// (pool_pvc.go), so it is where a Backup write has to arrive for the PVC to +// see it. +func (c *C) shardBackupSizes() map[string]string { + c.Helper() + shards := &ShardList{} + c.NoError(c.List(shards), "list Shards") + if len(shards.Items) == 0 { + c.Fatalf("no Shards in %s to read a resolved backup size from", c.NS) + } + out := make(map[string]string, len(shards.Items)) + for _, shard := range shards.Items { + size := "" + if shard.Spec.Backup != nil && shard.Spec.Backup.Filesystem != nil { + size = shard.Spec.Backup.Filesystem.Storage.Size + } + out[shard.Name] = size + } + return out +} + +// durabilityMirrors reads the DurabilityPolicy every TableGroup and Shard in +// ns currently carries. The field's only other consumer is the topology store, +// which this suite fakes in memory (fakes.go) and which writes a policy of its +// own regardless, so these mirrored spec fields are the whole of what this +// field does that anything here can observe. +func (c *C) durabilityMirrors() map[string]string { + c.Helper() + out := map[string]string{} + tgs := &TableGroupList{} + c.NoError(c.List(tgs), "list TableGroups") + for _, tg := range tgs.Items { + out["TableGroup/"+tg.Name] = tg.Spec.DurabilityPolicy + } + shards := &ShardList{} + c.NoError(c.List(shards), "list Shards") + for _, shard := range shards.Items { + out["Shard/"+shard.Name] = shard.Spec.DurabilityPolicy + } + if len(out) == 0 { + c.Fatalf("no TableGroups or Shards in %s to read a DurabilityPolicy from", c.NS) + } + return out +} + +// testPVCDeletionPolicyRoundTrip watches only PersistentVolumeClaim. +// +// PVC is a plain corev1 kind with no status.conditions of its own, so writing +// to it never sets off the generation-bump status-condition churn that +// TableGroup and Shard produce on every spec write (see +// testDurabilityPolicyRoundTrip for where that churn made a Step-based +// assertion unusable). PVCDeletionPolicy's real consequence, +// reconcilePVCOwnerRefs (pkg/resource-handler/controller/shard/shard_controller.go), +// lands on PVCs directly, so this narrower watch still sees it, and nothing +// wider is needed to catch a leak here. +func testPVCDeletionPolicyRoundTrip(t *testing.T) { + c := newCase(t) + sc := c.NewScript(&corev1.PersistentVolumeClaimList{}) + sc.StepTimeout = 10 * time.Second + + cluster := c.MinimalCluster("pvcdp") + c.WaitForClusterHealthy(cluster) + + // Snapshotted so the round trip is checked against the namespace's PVCs and + // not only against the field's own value. A PVC created during the set and + // then left alone emits nothing on the way back, so the closing Quiet step + // cannot see it. + pvcsAtStart := c.pvcRequests() + c.RequireQuiescent(5*time.Second, 30*time.Second) + + // The script opened before MinimalCluster, per NewScript's own contract, + // so its watch replays every PVC the fixture created as an Added event + // ("an already-populated namespace is replayed as a run of added + // events"). Permitting that baseline explicitly, by listing what actually + // exists now that convergence is independently confirmed, is the + // documented way to handle it, and it is a statement about the fixture, + // not a guess about this field's behaviour. + pvcs := &corev1.PersistentVolumeClaimList{} + c.NoError(c.List(pvcs), "list PVCs") + c.NotEmpty(pvcs.Items, "MinimalCluster created no PVCs to test PVCDeletionPolicy against") + baseline := make([]ctrltest.Allow, 0, len(pvcs.Items)*2) + pvcNames := make([]string, 0, len(pvcs.Items)) + for _, pvc := range pvcs.Items { + baseline = append(baseline, + ctrltest.Added("PersistentVolumeClaim", pvc.Name), + ctrltest.Changed("PersistentVolumeClaim", pvc.Name)) + pvcNames = append(pvcNames, pvc.Name) + } + sc.Step("PVCs created and bound during initial convergence", nil, baseline...) + + sc.Step("cluster settled", nil, ctrltest.Quiet()) + + original := cluster.Spec.PVCDeletionPolicy.DeepCopy() + + consequences := make([]ctrltest.Allow, 0, len(pvcNames)) + for _, name := range pvcNames { + consequences = append(consequences, ctrltest.Changed("PersistentVolumeClaim", name)) + } + + sc.Step("set PVCDeletionPolicy to Retain/Retain", func() error { + got := &MultigresCluster{} + if err := c.Get(client.ObjectKeyFromObject(cluster), got); err != nil { + return err + } + got.Spec.PVCDeletionPolicy = &PVCDeletionPolicy{ + WhenDeleted: multigresv1alpha1.RetainPVCRetentionPolicy, + WhenScaled: multigresv1alpha1.RetainPVCRetentionPolicy, + } + return c.Update(got) + }, consequences...) + + sc.Step("settled after setting the field", nil, ctrltest.Quiet()) + + sc.Step("unset PVCDeletionPolicy", func() error { + got := &MultigresCluster{} + if err := c.Get(client.ObjectKeyFromObject(cluster), got); err != nil { + return err + } + got.Spec.PVCDeletionPolicy = nil + return c.Update(got) + }, consequences...) + + sc.Step("nothing left to do after the round trip", nil, ctrltest.Quiet()) + + sc.Finish(time.Second) + + // PVCDeletionPolicy carries a CRD-level +kubebuilder:default, so a nil + // pointer never survives the API server: MinimalCluster's explicit + // {Delete, Delete} and an omitted field both resolve to the same stored + // value. The round trip is checked against what the field actually reads + // back as, not against a literal nil. + final := &MultigresCluster{} + c.NoError(c.Get(client.ObjectKeyFromObject(cluster), final), "get final cluster") + c.Check().EqDiff(original, final.Spec.PVCDeletionPolicy, + "PVCDeletionPolicy round trip not lossless") + c.Check().EqDiff(pvcsAtStart, c.pvcRequests(), + "PVCs differ across the round trip") +} + +// testBackupRoundTrip takes Backup out to an explicit backup storage size and +// back. It is the subtest that found a defect, and the pin below is the +// finding rather than an aside. +// +// Backup's only consequence any watch in this suite can see is the shared +// backup PVC's storage request (pool_pvc.go). The rest of what the field +// feeds is the topology store, faked in memory here (fakes.go), or needs a +// Secret the caller precreates (reconcile_shared_infra.go). So the round trip +// is driven through that size, and the size is where the operator breaks. +// +// Which size, measured against this harness rather than assumed, because once +// the PVC is bound and the data-plane fake has copied its request into +// status.capacity (datasim.go) every candidate is refused for a different +// reason: +// +// - smaller, 5Gi against the resolver's 10Gi default, is refused by PVC +// validation: "spec.resources.requests.storage: Forbidden: field can not +// be less than status.capacity". That rule is unconditional in Kubernetes, +// so any user of the operator can reach it. +// - larger, 20Gi, is refused by the PersistentVolumeClaimResize admission +// plugin: "only dynamically provisioned pvc can be resized and the +// storageclass that provisions the pvc must support resize", because this +// suite creates no StorageClass. A real cluster whose class sets +// allowVolumeExpansion would accept it, so that refusal is an artifact of +// the fixture and is deliberately not what gets pinned here. +// - equal but written in another unit, 10240Mi, is canonicalised back to +// 10Gi by the API server and bumps no resourceVersion, so it is invisible +// rather than illegal. +// +// The shrink is therefore the transition worth driving: legal on the +// MultigresCluster, propagated all the way to the PVC apply, and refused +// there. +func testBackupRoundTrip(t *testing.T) { + c := newCase(t) + ns := c.NS + cluster := c.MinimalCluster("backup") + c.WaitForClusterHealthy(cluster) + c.RequireQuiescent(5*time.Second, 30*time.Second) + + pvcsBefore := c.pvcRequests() + sizesBefore := c.shardBackupSizes() + original := cluster.Spec.Backup.DeepCopy() + + c.updateCluster(cluster, func(c *MultigresCluster) { + c.Spec.Backup = &multigresv1alpha1.BackupConfig{ + Type: multigresv1alpha1.BackupTypeFilesystem, + Filesystem: &multigresv1alpha1.FilesystemBackupConfig{ + Storage: multigresv1alpha1.StorageSpec{Size: "5Gi"}, + }, + } + }) + + // The defect: the refused PVC apply returns an error from Reconcile + // (shard_controller.go), so everything after that block is skipped for as + // long as the field stays lowered, the postgres ConfigMap render, the pool + // pods, PDB sizing and reconcilePVCOwnerRefs, and controller-runtime + // retries forever. The status update is NOT skipped: updateStatus runs + // early in Reconcile, deliberately, so a wedged shard keeps reporting a + // current observedGeneration and phase. That is what makes this defect + // silent, and it is why a triage that looks for a stale status will not + // find one. A user who lowers this field wedges + // the shard's whole reconcile loop, with no clamping, no rejection at + // admission, and no condition on the Shard saying why. + // + // Pinned as non-quiescence rather than as a missing PVC event, because a + // missing event is what the API server's refusal guarantees on every run + // forever: no operator change could ever retire that pin, and a pin that + // cannot expire is a suppression. This one expires the day the operator + // clamps the value, because then the namespace goes quiet and quiescent + // returns nil. + // + // Two neighbouring fixes do not expire it through quiescence, and the + // difference is worth knowing before trusting this pin as a tracker. An + // admission refusal fails the test earlier, at the update call, so the + // suite still goes red but by another route. A fix that gives up and + // records a terminal condition only after retrying past this horizon's + // activity budget leaves the pin green, so that one has to be noticed by a + // human reading this comment rather than by the pin flipping. + // + // What this pin rests on, measured 2026-09-18 rather than assumed, because + // it used to rest on something weaker than it looked. The wedged pass makes + // several accepted no-op writes (both pg_hba and exporter-queries + // ConfigMaps, the multiorch Deployment and Service, a Shard status patch) + // before it reaches the refused PVC apply, so before the recorder counted + // rejected writes, this pin observed non-quiescence only through those + // earlier writes. Reordering updateStatus, a refactor with no behavioural + // intent, would have flipped it to "appears fixed" with the defect fully + // present. The recorder now counts the refused apply itself, which is the + // one signal the defect cannot occur without, so that particular + // reordering can no longer fool it. + // + // It is not yet true that the rejection alone carries the pin. Counting + // only rejected writes, the same wedge measured non-quiescent on one run + // and quiet on the next, 12 refused applies being right at the edge of a + // 5s window inside an 11s horizon as the retry backoff spreads them out. + // So the margin still comes from the signals combined. Widening the + // horizon is not the fix (see below); if this pin ever needs to stand on + // the rejection by itself, count refused reconcile passes directly rather + // than inferring them from a quiet window. + // + // The horizon is short on purpose. controller-runtime retries a failing + // Reconcile with exponential backoff, so the gap between failing passes + // grows without bound and a long enough horizon would let the backoff + // itself supply the quiet window while the shard is still wedged. + // Calibrated both ways on this harness, three runs each: measured from + // immediately after a legal spec write this goes quiet in about 5.3s, + // measured from immediately after this write it never goes quiet inside + // 11s, over 12 failing reconcile passes. + c.KnownDefect("MGO-BACKUP-PVC-SHRINK-WEDGES-SHARD-RECONCILE", func() error { + err := c.TryQuiescent(5*time.Second, 11*time.Second) + // A lost watch voids the measurement instead of observing the + // defect, and a non-nil error here is read as the defect still + // being present, which would hold this pin green on an unrelated + // failure. + c.True(err == nil || !strings.Contains(err.Error(), "is void"), + "quiescence over %s is void, so it is no evidence either way: %v", ns, err) + return err + }) + + // Where the value actually got to, as an executable claim rather than a + // comment, because the mechanism is easy to misread: it does reach the + // Shard, so the PVC is applied from the size the user asked for and the + // refusal happens at the API server. A fix aimed at the resolver's backup + // defaulting would land on code that is behaving correctly. + c.Eventually(15*time.Second, "the lowered size to reach every Shard", func() error { + for name, size := range c.shardBackupSizes() { + if size != "5Gi" { + return fmt.Errorf("Shard %s resolved backup size is %q", name, size) + } + } + return nil + }) + + c.updateCluster(cluster, func(c *MultigresCluster) { + c.Spec.Backup = nil + }) + + // The gate that makes the closing assertion mean something. The namespace + // is not quiet when the unset lands, so silence afterwards would be + // ambiguous between the operator having processed it and the retry backoff + // having merely grown past the window. The resolved size returning to + // where it started is positive evidence that the unset propagated back + // down to where the PVC is applied from. It is a wait, not the assertion. + c.Eventually(30*time.Second, "the unset to reach every Shard", func() error { + if got := c.shardBackupSizes(); !maps.Equal(got, sizesBefore) { + return fmt.Errorf("resolved backup sizes are %v, want %v", got, sizesBefore) + } + return nil + }) + + // The round trip's assertion: once the cluster is back to the + // configuration it started in, a converged operator has nothing left to + // do, and this is that claim over every kind the operator writes plus + // every write it makes. + c.RequireQuiescent(5*time.Second, 30*time.Second) + + c.Check().EqDiff(pvcsBefore, c.pvcRequests(), "PVCs did not round trip") + final := &MultigresCluster{} + c.NoError(c.Get(client.ObjectKeyFromObject(cluster), final), "get final cluster") + c.Check().EqDiff(original, final.Spec.Backup, "Backup round trip not lossless") +} + +// testDurabilityPolicyRoundTrip has no Script, and the reason is worth +// recording because the shape it would need does not exist in the runner. +// +// DurabilityPolicy is mirrored into TableGroup.Spec and Shard.Spec +// (builders_tablegroup.go, tablegroup/builders.go), and any spec write to +// either bumps its generation, which sends every controller watching it back +// to re-stamp its own status conditions' observedGeneration. That catch-up +// took a different number of passes on every run measured while writing this +// test: watching only TableGroup and Shard, the same single field write +// settled after 19, then 13, then 9 total events across three otherwise +// identical runs. A Step's allow-list is an exact multiset, with no "N events +// of this kind" wildcard available, so declaring one against a count that +// moves between runs would flake on this suite's own harness rather than on +// the operator, which is a worse failure than not writing the assertion at +// all. +// +// What is left for a Script to assert over those kinds is that nothing further +// happened, and RequireQuiescent asserts that strictly better: five seconds +// over twelve kinds plus the write recorder, rather than about a second over +// two, and it needs no watch opened before the fixture, which over these kinds +// is not possible to combine with an exact Step anyway. So the closing +// RequireQuiescent is this subtest's closed-world assertion, standing where +// the other form's Quiet() step stands. +func testDurabilityPolicyRoundTrip(t *testing.T) { + c := newCase(t) + cluster := c.MinimalCluster("durability") + c.WaitForClusterHealthy(cluster) + c.RequireQuiescent(5*time.Second, 30*time.Second) + + original := cluster.Spec.DurabilityPolicy + mirrorsBefore := c.durabilityMirrors() + + c.updateCluster(cluster, func(c *MultigresCluster) { + c.Spec.DurabilityPolicy = "MULTI_CELL_AT_LEAST_2" + }) + + // Not an expectation about cleanup, which this test writes none of. It is + // the guard that keeps the round trip from being vacuous: if setting the + // field moved nothing anywhere, unsetting it could not leak anything and + // the closing assertion would be proving nothing about this field. + c.Eventually(30*time.Second, "the set policy to reach every mirror", func() error { + for name, policy := range c.durabilityMirrors() { + if policy != "MULTI_CELL_AT_LEAST_2" { + return fmt.Errorf("%s carries %q", name, policy) + } + } + return nil + }) + c.RequireQuiescent(5*time.Second, 30*time.Second) + + c.updateCluster(cluster, func(c *MultigresCluster) { + c.Spec.DurabilityPolicy = original + }) + + // The mirrors returning is the state half of the round trip, and no event + // assertion can make it: a mirror left holding the set value emits nothing + // once it stops changing, so silence and correctness would be the same + // observation. Also the gate that the operator processed the unset before + // the assertion below asks for silence. + c.Eventually(30*time.Second, "the unset policy to reach every mirror", func() error { + if got := c.durabilityMirrors(); !maps.Equal(got, mirrorsBefore) { + return fmt.Errorf("mirrors are %v, want %v", got, mirrorsBefore) + } + return nil + }) + + c.RequireQuiescent(5*time.Second, 30*time.Second) + + final := &MultigresCluster{} + c.NoError(c.Get(client.ObjectKeyFromObject(cluster), final), "get final cluster") + c.Check().Eq(original, final.Spec.DurabilityPolicy, "DurabilityPolicy round trip not lossless") +} diff --git a/test/suite/shard_requeue_test.go b/test/suite/shard_requeue_test.go new file mode 100644 index 00000000..6cd6311d --- /dev/null +++ b/test/suite/shard_requeue_test.go @@ -0,0 +1,183 @@ +package suite + +import ( + "fmt" + "testing" + "time" + + "sigs.k8s.io/controller-runtime/pkg/client" + + "github.com/multigres/testkit/ctrltest" +) + +// certBootstrapBudget is how long a test will wait for the shard controller to +// finish generating pgBackRest's CA and server certificates. +// +// It is the one wait in this package deliberately larger than 30s, and it is +// sized against the step it waits on rather than against convergence. +// reconcilePgBackRestCerts generates two RSA keys, which is the most expensive +// thing this suite does and by far the most variable under -race: measured at +// 14s in one run of the suite and 75s in another, on the same machine. A +// number sized for the 14s case turns the 75s case into a red suite that says +// nothing about the operator. +const certBootstrapBudget = 2 * time.Minute + +// waitForShardPastPKI blocks until the shard controller has written a pool Pod +// in ns, which is the evidence this suite has that the controller is past +// certificate generation. +// +// It exists to keep a slow precondition out of an assertion's budget. The +// shard reconciles fourteen steps in a fixed order: the pgBackRest certificate +// step is fifth, reconcilePool is twelfth, and reconcileDataPlane, the only +// step that returns the pooler-registration requeue, is last. A test that +// starts its requeue budget before the keys exist is timing RSA keygen, and it +// goes red when the crypto was slow rather than when the operator was wrong. +// Split in two, each wait is sized against what it actually waits for. +// +// A pool Pod is the signal rather than the certificate Secrets themselves for +// two reasons. It sits between the two steps that matter, so it proves the +// keygen is behind us without depending on it being the immediately preceding +// step. And a Shard creating Pods for its pools is a more stable fact than the +// name the operator builds its Secrets from, which a test has no business +// knowing. It is read from the recorder rather than from the apiserver so that +// what is observed is the shard controller having written, not an object that +// something else could have created. +func (c *C) waitForShardPastPKI() { + c.Helper() + c.Eventually( + certBootstrapBudget, + "the shard controller to get past certificate generation", + func() error { + for _, op := range Suite.Ops.OpsInNamespace(c.NS) { + if op.Controller == "shard" && ctrltest.KindSuffix(op.Kind) == "Pod" { + return nil + } + } + return fmt.Errorf("the shard controller has written no pool Pod yet") + }, + ) +} + +// holdPoolerRegistration keeps every pooler in this case's namespace out of +// the topology store for the rest of the test. +// +// The shard controller only asks for its one-minute requeue when it finds no +// poolers registered at all. Left to the data-plane fake, which registers on +// a 500ms tick as soon as a pool pod exists, whether the shard ever sees that +// empty topology is a race: when the first pooler registers before the +// shard's first data-plane pass, the shard goes straight to "some poolers", +// skips the requeue these tests wait for, and on today's operator returns no +// requeue at all (MGO-POOL-SCALEUP-ROLE-STALE). Measured at 1 run in 20 under +// full-suite load. Holding registration makes the empty topology the only +// state the shard can see. +func (c *C) holdPoolerRegistration() { + c.Helper() + c.Cleanup(poolers.HoldRegistrations(c.NS)) +} + +// awaitShard waits for the cluster controller to create this namespace's one +// Shard and returns it. +// +// Exactly one, not at least one. Both polls this replaces went on to read +// Items[0], and the weaker form picked element zero out of a set whose size +// it never checked. Every fixture in this package declares a single +// ShardConfig and suite_test.go asserts that, so the stronger claim is true +// today and costs nothing to state. +// +// It is what happens when that stops being true that decides it. Sharding is +// this project's whole point, so a multi-shard fixture is a matter of time, +// and at that moment "at least one" stays green while silently asserting +// about whichever Shard the API server happened to return first. "Exactly +// one" fails saying it got two, which points at the fixture that changed. A +// test that wants a particular shard out of several should name it rather +// than index into a list. +func (c *C) awaitShard() Shard { + c.Helper() + var shard Shard + c.Eventually(30*time.Second, "the cluster's one Shard to exist", func() error { + shards := &ShardList{} + if err := c.List(shards); err != nil { + return err + } + if len(shards.Items) != 1 { + return fmt.Errorf("want exactly one Shard, got %d", len(shards.Items)) + } + shard = shards.Items[0] + return nil + }) + return shard +} + +// shardKey is awaitShard's key, for the callers that only need to address it. +func (c *C) shardKey() client.ObjectKey { + c.Helper() + shard := c.awaitShard() + return client.ObjectKeyFromObject(&shard) +} + +// TestShardAsksForAMinuteAwaitingPoolerRegistration is the assertion that +// replaces waiting a minute for the same information, and the reason requeue +// compression does not hide the defect it compresses. +// +// While no multipooler has registered in the topology store the shard +// controller asks to be woken in a minute. Nothing in Kubernetes watches that +// store, so in production nothing can wake it sooner, and before compression +// every convergence test in this package paid that minute whenever the last pod +// event happened to land before the data plane fake registered its poolers. +// Which test paid was a coin flip and the suite's wall time swung by a minute +// between runs for no visible reason. +// +// Compressed, the poll comes back in 50ms and the minute survives here as a +// fact about the operator. Fixing it is not this suite's job, and this +// assertion is what should fail when somebody does fix it. +// +// This test asserts about the operator and nothing else. That the suite clamps +// what it saw here is a fact about the harness and belongs to +// TestSuiteCompressesRequeues, so that a clamp change reports itself as a +// clamp change rather than as the shard controller's polling having moved. +func TestShardAsksForAMinuteAwaitingPoolerRegistration(t *testing.T) { + c := newCase(t) + c.holdPoolerRegistration() + c.MinimalCluster("requeue") + + key := c.shardKey() + c.waitForShardPastPKI() + + got := Suite.Reconciles.WaitForRequeue(t, "shard", key, time.Minute, 30*time.Second) + + // Exactly a minute, not merely at least a minute. One minute is the shard + // controller's only requeue of that length (poolerRegistrationRetryDelay + // in reconcile_data_plane.go), so the duration identifies the code path + // that the reconcile boundary itself cannot: ctrl.Result carries no + // reason, and the AwaitingPoolerRegistration reason lives on the shard's + // PostureConsistent condition, which by the time a test can read it has + // usually already moved on. + c.Check().Eq(time.Minute, got.RequestedAfter, "the shard's requeue duration") +} + +// TestSuiteCompressesRequeues is the canary on the harness half of the +// bargain, and it is live rather than a unit test for a reason the unit tests +// cannot cover: TestInterceptorCompressesRequeue proves that an interceptor +// clamps, not that this suite wired one around the real controllers. If that +// wiring is ever broken, compression dies silently, every timeout in the +// package regains its old "maybe it is just waiting" ambiguity, and nothing +// fails except runtimes nobody reads. +// +// It deliberately does not name a duration the operator chose. Any requeue +// longer than the clamp will do, so that fixing the shard controller's minute +// changes what this test observes but not whether it passes. +func TestSuiteCompressesRequeues(t *testing.T) { + c := newCase(t) + c.holdPoolerRegistration() + c.MinimalCluster("clamp") + + key := c.shardKey() + c.waitForShardPastPKI() + + got := Suite.Reconciles.WaitForRequeue(t, "shard", key, 2*ctrltest.RequeueClamp, 30*time.Second) + + c.Check().Eq(ctrltest.RequeueClamp, got.Result.RequeueAfter, + "controller-runtime's requeue-after duration") + c.Check(). + True(got.Compressed(), "Compressed() = false on a pass that asked for %s", got.RequestedAfter) +} diff --git a/test/suite/suite.go b/test/suite/suite.go new file mode 100644 index 00000000..428d78a3 --- /dev/null +++ b/test/suite/suite.go @@ -0,0 +1,223 @@ +// Package suite is the multi-controller envtest harness for this operator: one +// manager running every reconciler the operator runs, so behaviour that is a +// protocol between controllers becomes testable. +// +// The generic half lives in pkg/ctrltest. What is left here is everything that +// is about this operator specifically: its scheme, its CRDs, its cache config, +// its five reconcilers, and the data plane doubles those reconcilers need. +// +// It has no build tag. Exclusion from the ordinary test targets is by path +// filter, and the entrypoint is `make test-suite`. +package suite + +import ( + "context" + "fmt" + "path/filepath" + "time" + + "github.com/multigres/multigres/go/common/rpcclient" + appsv1 "k8s.io/api/apps/v1" + corev1 "k8s.io/api/core/v1" + networkingv1 "k8s.io/api/networking/v1" + policyv1 "k8s.io/api/policy/v1" + storagev1 "k8s.io/api/storage/v1" + "k8s.io/apimachinery/pkg/runtime" + clientgoscheme "k8s.io/client-go/kubernetes/scheme" + "k8s.io/utils/ptr" + "sigs.k8s.io/controller-runtime/pkg/client" + "sigs.k8s.io/controller-runtime/pkg/controller" + "sigs.k8s.io/controller-runtime/pkg/manager" + + multigresv1alpha1 "github.com/multigres/multigres-operator/api/v1alpha1" + "github.com/multigres/multigres-operator/pkg/cacheopts" + multigresclustercontroller "github.com/multigres/multigres-operator/pkg/cluster-handler/controller/multigrescluster" + tablegroupcontroller "github.com/multigres/multigres-operator/pkg/cluster-handler/controller/tablegroup" + "github.com/multigres/multigres-operator/pkg/data-handler/poolerclient" + cellcontroller "github.com/multigres/multigres-operator/pkg/resource-handler/controller/cell" + shardcontroller "github.com/multigres/multigres-operator/pkg/resource-handler/controller/shard" + toposervercontroller "github.com/multigres/multigres-operator/pkg/resource-handler/controller/toposerver" + "github.com/multigres/testkit/ctrltest" +) + +// OperatorNamespace stands in for the namespace the operator deploys into. The +// cache treats it specially (unfiltered), so the suite has to have one for the +// production cache config to mean anything. +const OperatorNamespace = "multigres-operator-system" + +// Suite is the suite-wide harness, booted once by TestMain. One envtest and one +// manager serve the whole package; isolate with Suite.Namespace(t). +var Suite *ctrltest.Suite + +// The operator's data plane doubles. Deliberately not suite members: every +// operator's data plane is different, so ctrltest has no place to put these and +// a test reaching for them is reaching for something about this operator. +var ( + rpc *rpcclient.FakeClient + topo *topoRegistry + poolers *poolerSim +) + +// Boot brings up envtest and the manager. The returned function tears both down +// and must run before any goroutine leak check, since the manager owns +// goroutines that only exit once its context is cancelled. +func Boot() (*ctrltest.Suite, func() error, error) { + scheme := runtime.NewScheme() + for _, add := range []func(*runtime.Scheme) error{ + clientgoscheme.AddToScheme, + multigresv1alpha1.AddToScheme, + appsv1.AddToScheme, + corev1.AddToScheme, + policyv1.AddToScheme, + networkingv1.AddToScheme, + storagev1.AddToScheme, + } { + if err := add(scheme); err != nil { + return nil, nil, fmt.Errorf("add to scheme: %w", err) + } + } + + return ctrltest.Boot(ctrltest.Options{ + Scheme: scheme, + CRDPaths: []string{filepath.Join("..", "..", "config", "crd", "bases")}, + WatchedKinds: watchedKinds(), + SimInterval: 250 * time.Millisecond, + Managers: []ctrltest.ManagerOptions{{ + Name: "multigres-operator", + CacheOptions: cacheopts.New(OperatorNamespace), + OperatorNamespace: OperatorNamespace, + // Matches main.go: the operator raises these to avoid client-side + // throttling once several controllers are reconciling at once. + QPS: 50, + Burst: 100, + Register: register, + }}, + }) +} + +// watchedKinds is every kind the operator writes in a test namespace. A kind +// missing here is a kind whose churn RequireQuiescent cannot see. +func watchedKinds() []client.ObjectList { + return []client.ObjectList{ + &multigresv1alpha1.MultigresClusterList{}, + &TopoServerList{}, + &multigresv1alpha1.CellList{}, + &TableGroupList{}, + &ShardList{}, + &corev1.PodList{}, + &appsv1.DeploymentList{}, + &appsv1.StatefulSetList{}, + &corev1.PersistentVolumeClaimList{}, + &corev1.ConfigMapList{}, + &corev1.ServiceList{}, + &policyv1.PodDisruptionBudgetList{}, + } +} + +// register wires every reconciler exactly as cmd/multigres-operator/main.go +// does, differing only in the seams a test has to fake: the topology store and +// the multipooler RPC client, and in the interceptor each one's reconcile +// boundary is wrapped in. +// +// Each reconciler is a named variable rather than the anonymous composite +// literal this used to be, and that is load bearing rather than tidying. +// SetupWithManagerReconciler substitutes the reconcile boundary and nothing +// else: the receiver stays live on the enqueue path, because map functions and +// predicates bind to it when the builder runs and call its client, and +// ShardReconciler keeps mutable state on itself. So the same pointer has to be +// both the receiver and what the interceptor delegates to. +func register(ctx context.Context, mgr manager.Manager, s *ctrltest.Suite) error { + base := mgr.GetClient() + + rpc = rpcclient.NewFakeClient() + topo = newTopoRegistry(ctx) + + // The pooler fake runs for the life of the suite, across every namespace, + // because the manager it feeds is also suite-wide. + poolers = &poolerSim{c: s.Client, rpc: rpc, topo: topo, interval: 500 * time.Millisecond} + go poolers.run(ctx) + + // Deliberately a bare option struct rather than each controller's + // production options. Every controller sets MaxConcurrentReconciles to 20 + // in its own SetupWithManager and then lets a caller-supplied + // controller.Options replace that wholesale, so this suite's options + // decide the value; it is set to 1 here rather than inherited by omission + // from the controller-runtime default, which is what used to happen. + // + // That divergence from production is load-bearing in both directions. + // It is what makes the recorder's per-controller ordering assertions + // meaningful: one reconcile goroutine per controller means writes are + // issued and recorded in program order. It is also what this suite + // therefore cannot catch, namely a controller racing itself across + // concurrent reconciles of different objects. Raising this to match + // production would silently turn every ordering assertion in the + // scenario tests into a race that fails a few times a week and reads as operator + // flakiness, and it would also break Interceptor.Ops, which identifies a + // pass's writes by an op log range that only one in-flight reconcile per + // controller can make unambiguous. + opts := controller.Options{ + SkipNameValidation: ptr.To(true), + MaxConcurrentReconciles: 1, + } + + cluster := &multigresclustercontroller.MultigresClusterReconciler{ + Client: s.Ops.For("multigrescluster", base), + Scheme: mgr.GetScheme(), + Recorder: mgr.GetEventRecorderFor("multigrescluster-controller"), + APIReader: mgr.GetAPIReader(), + CreateTopoStore: topo.ForClusterRef, + } + if err := cluster.SetupWithManagerReconciler( + mgr, s.Reconciles.Wrap("multigrescluster", cluster), opts, + ); err != nil { + return fmt.Errorf("setup multigrescluster: %w", err) + } + + tableGroup := &tablegroupcontroller.TableGroupReconciler{ + Client: s.Ops.For("tablegroup", base), + Scheme: mgr.GetScheme(), + Recorder: mgr.GetEventRecorderFor("tablegroup-controller"), + } + if err := tableGroup.SetupWithManagerReconciler( + mgr, s.Reconciles.Wrap("tablegroup", tableGroup), opts, + ); err != nil { + return fmt.Errorf("setup tablegroup: %w", err) + } + + cell := &cellcontroller.CellReconciler{ + Client: s.Ops.For("cell", base), + Scheme: mgr.GetScheme(), + Recorder: mgr.GetEventRecorderFor("cell-controller"), + } + if err := cell.SetupWithManagerReconciler( + mgr, s.Reconciles.Wrap("cell", cell), opts, + ); err != nil { + return fmt.Errorf("setup cell: %w", err) + } + + topoServer := &toposervercontroller.TopoServerReconciler{ + Client: s.Ops.For("toposerver", base), + Scheme: mgr.GetScheme(), + Recorder: mgr.GetEventRecorderFor("toposerver-controller"), + } + if err := topoServer.SetupWithManagerReconciler( + mgr, s.Reconciles.Wrap("toposerver", topoServer), opts, + ); err != nil { + return fmt.Errorf("setup toposerver: %w", err) + } + + shard := &shardcontroller.ShardReconciler{ + Client: s.Ops.For("shard", base), + Scheme: mgr.GetScheme(), + Recorder: mgr.GetEventRecorderFor("shard-controller"), + APIReader: mgr.GetAPIReader(), + PoolerClients: poolerclient.Static(rpc), + CreateTopoStore: topo.ForShard, + } + if err := shard.SetupWithManagerReconciler( + mgr, s.Reconciles.Wrap("shard", shard), opts, + ); err != nil { + return fmt.Errorf("setup shard: %w", err) + } + return nil +} diff --git a/test/suite/suite_test.go b/test/suite/suite_test.go new file mode 100644 index 00000000..2b8c8667 --- /dev/null +++ b/test/suite/suite_test.go @@ -0,0 +1,123 @@ +package suite + +import ( + "fmt" + "strings" + "testing" + "time" + + corev1 "k8s.io/api/core/v1" + apierrors "k8s.io/apimachinery/pkg/api/errors" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "sigs.k8s.io/controller-runtime/pkg/client" + + multigresv1alpha1 "github.com/multigres/multigres-operator/api/v1alpha1" +) + +// TestClusterConvergesUnderAllControllers is the harness proof: one +// MultigresCluster, five live reconcilers, and a faked data plane, reaching a +// terminal healthy state. +// +// It asserts nothing about behaviour that single-controller tests already +// cover. Its job is to fail loudly if the harness itself stops working. +func TestClusterConvergesUnderAllControllers(t *testing.T) { + c := newCase(t) + ns := c.NS + cluster := c.MinimalCluster("minimal") + + c.Eventually(30*time.Second, "child CRs to be created", func() error { + topos := &TopoServerList{} + if err := c.List(topos); err != nil { + return err + } + cells := &multigresv1alpha1.CellList{} + if err := c.List(cells); err != nil { + return err + } + tgs := &TableGroupList{} + if err := c.List(tgs); err != nil { + return err + } + shards := &ShardList{} + if err := c.List(shards); err != nil { + return err + } + if len(topos.Items) == 0 || len(cells.Items) == 0 || + len(tgs.Items) == 0 || len(shards.Items) == 0 { + return fmt.Errorf("have %d TopoServer, %d Cell, %d TableGroup, %d Shard", + len(topos.Items), len(cells.Items), len(tgs.Items), len(shards.Items)) + } + return nil + }) + + c.WaitForClusterHealthy(cluster) + + // Attribution is the other half of the harness: a test that cannot say + // which controller wrote cannot assert a protocol between controllers. + wrote := map[string]bool{} + for _, op := range Suite.Ops.OpsInNamespace(ns) { + wrote[op.Controller] = true + } + for _, name := range []string{"multigrescluster", "cell", "toposerver", "tablegroup", "shard"} { + c.Check().True(wrote[name], "no recorded writes from the %s controller; "+ + "either it never ran or attribution is broken", name) + } +} + +// TestNamespacesAreIsolated runs two clusters at once to prove the isolation +// boundary holds, since every fake behind the suite (the topology store above +// all) is shared process-wide and keyed by namespace. +func TestNamespacesAreIsolated(t *testing.T) { + t.Parallel() + + for _, name := range []string{"iso-a", "iso-b"} { + t.Run(name, func(t *testing.T) { + t.Parallel() + c := newCase(t) + ns := c.NS + cluster := c.MinimalCluster(strings.TrimPrefix(name, "iso-")) + + c.WaitForClusterHealthy(cluster) + + shards := &ShardList{} + c.NoError(c.List(shards)) + c.Len(shards.Items, 1, "in %s", ns) + }) + } +} + +// TestProductionCacheConfigIsInEffect guards the fidelity trap: the manager +// must cache exactly what cmd/multigres-operator/main.go caches. +// +// Under the production config an unlabelled Secret outside the operator's own +// namespace is invisible to the cached client, which is why the reconcilers +// carry an APIReader at all. A manager built with default cache options makes +// every cached read behave differently from production and quietly voids the +// premise that these are the real controllers wired as in main.go. +// +// This assertion is only possible because the config lives in pkg/cacheopts +// rather than being copied out of package main, which cannot be imported. +func TestProductionCacheConfigIsInEffect(t *testing.T) { + c := newCase(t) + ns := c.NS + + secret := &corev1.Secret{ + ObjectMeta: metav1.ObjectMeta{Name: "unlabelled", Namespace: ns}, + StringData: map[string]string{"password": "postgres"}, + } + c.NoError(c.Create(secret), "create secret") + key := client.ObjectKeyFromObject(secret) + + c.Eventually(30*time.Second, "the APIReader to see the secret", func() error { + return Suite.Manager("multigres-operator"). + Mgr.GetAPIReader(). + Get(c.Context(), key, &corev1.Secret{}) + }) + + err := Suite.Manager("multigres-operator"). + Mgr.GetClient(). + Get(c.Context(), key, &corev1.Secret{}) + c.True(apierrors.IsNotFound(err), + "cached client should not see an unlabelled Secret outside %s, got err=%v", + OperatorNamespace, err) +} diff --git a/test/suite/testdata/multigateway-deployment.golden.yaml b/test/suite/testdata/multigateway-deployment.golden.yaml new file mode 100644 index 00000000..c0c3211a --- /dev/null +++ b/test/suite/testdata/multigateway-deployment.golden.yaml @@ -0,0 +1,89 @@ +metadata: + labels: + app.kubernetes.io/component: multigateway + app.kubernetes.io/instance: golden-cluster + app.kubernetes.io/managed-by: multigres-operator + app.kubernetes.io/name: multigres + app.kubernetes.io/part-of: multigres + multigres.com/cell: zone1 + name: golden-cluster-zone1-multigateway-90bac2c9 + namespace: default + ownerReferences: + - apiVersion: multigres.com/v1alpha1 + blockOwnerDeletion: true + controller: true + kind: Cell + name: golden-cell + uid: golden-cell-uid +spec: + replicas: 1 + selector: + matchLabels: + app.kubernetes.io/component: multigateway + app.kubernetes.io/instance: golden-cluster + multigres.com/cell: zone1 + strategy: {} + template: + metadata: + annotations: + multigres.com/project-ref: golden-cluster + labels: + app.kubernetes.io/component: multigateway + app.kubernetes.io/instance: golden-cluster + app.kubernetes.io/managed-by: multigres-operator + app.kubernetes.io/name: multigres + app.kubernetes.io/part-of: multigres + multigres.com/cell: zone1 + spec: + containers: + - args: + - multigateway + - --http-port + - "15100" + - --grpc-port + - "15170" + - --pg-port + - "5432" + - --pg-replica-port + - "5433" + - --topo-global-server-addresses + - global-topo:2379 + - --topo-global-root + - /multigres/global + - --cell + - zone1 + - --log-level + - info + image: ghcr.io/multigres/multigres:golden-fixture + livenessProbe: + httpGet: + path: /live + port: 15100 + periodSeconds: 10 + name: multigateway + ports: + - containerPort: 15100 + name: http + protocol: TCP + - containerPort: 15170 + name: grpc + protocol: TCP + - containerPort: 5432 + name: postgres + protocol: TCP + - containerPort: 5433 + name: pg-replica + protocol: TCP + readinessProbe: + httpGet: + path: /ready + port: 15100 + periodSeconds: 5 + resources: {} + startupProbe: + failureThreshold: 30 + httpGet: + path: /ready + port: 15100 + periodSeconds: 5 +status: {} diff --git a/test/suite/types.go b/test/suite/types.go new file mode 100644 index 00000000..622c0d01 --- /dev/null +++ b/test/suite/types.go @@ -0,0 +1,48 @@ +package suite + +import multigresv1alpha1 "github.com/multigres/multigres-operator/api/v1alpha1" + +// The API types this package builds most, without the package qualifier. +// +// multigresv1alpha1.MultigresCluster is thirty-three characters and says +// "multigres" twice, and a test body that constructs a dozen API objects +// spends more width on the qualifier than on what it is asserting. +// +// Two rules keep this from becoming a dot-import in disguise, which would +// trade that width for provenance nobody can recover: +// +// - Only types, never values or functions. multigresv1alpha1.PhaseHealthy +// and multigresv1alpha1.DeletePVCRetentionPolicy stay qualified, because +// a bare PhaseHealthy in an assertion genuinely does read as though it +// could be this package's own. A composite literal names its type on the +// line above, so &Shard{} does not have the same problem. +// - Only types used three or more times here. Aliasing a type used once +// saves eighteen characters and costs the next reader a lookup. +// +// Deliberately the same names as upstream, so there is nothing to learn and +// nothing to bikeshed: this drops the qualifier and changes nothing else. +// MultigresCluster in particular keeps its full name; a bare Cluster is far +// too overloaded in a workspace where that word also means a Kubernetes +// cluster and an EKS cluster. +// +// Orthogonal to any future rename of the multigresv1alpha1 alias itself, +// which is a repo-wide question across 217 files. These read the same either +// way. +type ( + MultigresCluster = multigresv1alpha1.MultigresCluster + Shard = multigresv1alpha1.Shard + PoolSpec = multigresv1alpha1.PoolSpec + ShardList = multigresv1alpha1.ShardList + PVCDeletionPolicy = multigresv1alpha1.PVCDeletionPolicy + CellConfig = multigresv1alpha1.CellConfig + MultigresClusterSpec = multigresv1alpha1.MultigresClusterSpec + PoolName = multigresv1alpha1.PoolName + PostgresPasswordSecretRef = multigresv1alpha1.PostgresPasswordSecretRef + TableGroupList = multigresv1alpha1.TableGroupList + CellName = multigresv1alpha1.CellName + DatabaseConfig = multigresv1alpha1.DatabaseConfig + TableGroupConfig = multigresv1alpha1.TableGroupConfig + ShardConfig = multigresv1alpha1.ShardConfig + ShardInlineSpec = multigresv1alpha1.ShardInlineSpec + TopoServerList = multigresv1alpha1.TopoServerList +) From a9abec7a687c2748a435e43e54289f719e502c30 Mon Sep 17 00:00:00 2001 From: Brent Graveland Date: Sat, 19 Sep 2026 19:03:55 -0600 Subject: [PATCH 3/7] ci: add an advisory job for the multi-controller suite make test-suite runs it, and the other test targets exclude test/suite by path filter: the tier has no build tag, deliberately, so nothing can be hidden behind one. The job is advisory rather than required, because the suite pins live operator defects and a required check would block every PR on defects nobody is fixing in that PR. It runs verbosely, which is what makes a passing run readable: a pin that is doing its job logs and passes, and Go discards that output without -v, so a suite with seven live pins would otherwise print the same thing as a suite with none. Signed-off-by: Brent Graveland --- .github/workflows/build-and-release.yaml | 4 ++- .github/workflows/test-suite.yaml | 40 ++++++++++++++++++++++++ 2 files changed, 43 insertions(+), 1 deletion(-) create mode 100644 .github/workflows/test-suite.yaml diff --git a/.github/workflows/build-and-release.yaml b/.github/workflows/build-and-release.yaml index aedf85a0..9ceb2873 100644 --- a/.github/workflows/build-and-release.yaml +++ b/.github/workflows/build-and-release.yaml @@ -50,7 +50,9 @@ jobs: cache: false - name: Run tests - run: go test ./... -coverprofile=./cover.out -covermode=atomic -coverpkg=./... + run: | + go test $(go list ./... | grep -v /test/suite) \ + -coverprofile=./cover.out -covermode=atomic -coverpkg=./... - name: Run observer tests against current operator API working-directory: tools/observer diff --git a/.github/workflows/test-suite.yaml b/.github/workflows/test-suite.yaml new file mode 100644 index 00000000..ce840c22 --- /dev/null +++ b/.github/workflows/test-suite.yaml @@ -0,0 +1,40 @@ +# This job is advisory. It is deliberately not a required check and is not +# referenced by any other workflow's needs. A red run here means the suite +# caught a real problem, not that CI itself is broken: treat a failure as a +# finding to investigate, not as noise to wave off or restart until it goes +# green. +name: Test suite + +on: + pull_request: {} + # main only: pull_request already covers any branch with a PR open, so a + # wider push trigger would run the suite twice for it. main is here to catch + # a bad merge. + push: + branches: + - main + workflow_dispatch: {} + +permissions: + contents: read + +jobs: + test-suite: + runs-on: ubuntu-latest + timeout-minutes: 30 + permissions: + contents: read + steps: + - name: Check out code + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + + - name: Install Go + uses: actions/setup-go@b7ad1dad31e06c5925ef5d2fc7ad053ef454303e # v7.0.0 + with: + go-version-file: go.mod + cache: false + + - name: Run test suite + run: make test-suite From a6b6626931012d6a6870a84e41d2ab5125ef4733 Mon Sep 17 00:00:00 2001 From: Brent Graveland Date: Sat, 19 Sep 2026 19:15:55 -0600 Subject: [PATCH 4/7] test: add a race-detector target for the multi-controller suite Nothing in this repo ran -race, which is an odd gap for the one suite where five controllers share a manager. Measured on the first run: 183s against a 166s baseline and zero data races. The 10% is cheaper than expected because this suite spends most of its wall clock waiting for controllers to converge, and the race detector does not slow down waiting. Separate from test-suite anyway. Certificate generation is the one CPU-bound step and has been measured swinging between 14 and 75 seconds under -race, which is enough to turn a wait sized against the normal run into a flake, so the timeout here is deliberately loose. The operator holds exactly one piece of state across reconcile goroutines, ShardReconciler.postureStrikes, and it is mutex-guarded with controller- runtime already serialising per object key. So this is a standing check that the answer has not changed rather than a hunt for a known race. Signed-off-by: Brent Graveland --- Makefile | 27 +++++++++++++++++++++++++++ 1 file changed, 27 insertions(+) diff --git a/Makefile b/Makefile index 9aef3644..a842b125 100644 --- a/Makefile +++ b/Makefile @@ -293,6 +293,33 @@ test-suite: manifests generate fmt vet setup-envtest ## Run the multi-controller KUBEBUILDER_ASSETS="$(shell $(ENVTEST) use $(ENVTEST_K8S_VERSION) --bin-dir $(LOCALBIN) -p path)" \ go test -v -p 1 -timeout 20m ./test/suite/... +# A separate target rather than a flag on the one above. Measured 2026-09-19: +# 183s against a 166s baseline, so about 10% rather than the roughly-double a +# CPU-bound suite would pay. This one spends most of its wall clock waiting for +# controllers to converge, and the race detector does not slow down waiting. +# +# Kept separate anyway, because the cost is not the same everywhere: certificate +# generation is the one CPU-bound step here and has been measured swinging +# between 14 and 75 seconds under -race, which is enough to turn a wait sized +# against the normal run into a flake. A budget that holds on both is looser +# than the default target should carry. +# +# Worth having at all because this suite is the only place five controllers +# share one manager, and the operator holds exactly one piece of state across +# reconcile goroutines: ShardReconciler.postureStrikes, a map guarded by a +# mutex. Nothing here exercises contention on it today, since the suite pins +# MaxConcurrentReconciles to 1 and controller-runtime already serialises +# reconciles per object key, so this is a standing check that the answer has +# not changed rather than a hunt for a known race. +# +# The timeout is generous rather than tight: the instrumented run is only +# slightly slower on average, but its slow tail is much fatter, and a timeout +# that fires on the tail reads as a hang rather than as the flake it is. +.PHONY: test-suite-race +test-suite-race: manifests generate fmt vet setup-envtest ## Run the multi-controller test suite under the race detector + KUBEBUILDER_ASSETS="$(shell $(ENVTEST) use $(ENVTEST_K8S_VERSION) --bin-dir $(LOCALBIN) -p path)" \ + go test -race -v -p 1 -timeout 40m ./test/suite/... + .PHONY: test test: manifests generate fmt vet ## Run tests (no integration testing) KUBEBUILDER_ASSETS="$(shell $(ENVTEST) use $(ENVTEST_K8S_VERSION) --bin-dir $(LOCALBIN) -p path)" \ From 6589421a0379e7ee6d21199c2dc0d80877d26e81 Mon Sep 17 00:00:00 2001 From: Brent Graveland Date: Fri, 25 Sep 2026 08:46:41 -0600 Subject: [PATCH 5/7] fix(shard): give every status write an explicit field owner A patch without client.FieldOwner does not opt out of field management. The API server derives one from the client's User-Agent, so the write gets an owner nobody named and nobody can see, and that owner then co-owns whatever fields the patch touched. Four Shard status writes did this, and the fields they claimed are also claimed by updateStatus on every reconcile, so two managers held them jointly. That is the same structure as the status hot loop: two managers, one object, overlapping fields. It was not looping, because these are merge patches that agree on values rather than applies that disagree, but it is one changed value away from the same outcome. The Pod status write in reconcile_readiness.go takes a distinct manager rather than the Shard's. It claims one condition on a Pod whose status otherwise belongs to kubelet, and a manager name is the only record of which concern took a field. The multi-controller suite pinned this with a KnownDefect on the field-ownership check in TestTwoControllersWriteOneShard. The pin becomes the positive assertion it was waiting to be: no field on the Shard is claimed by more than one manager. Signed-off-by: Brent Graveland --- .../controller/shard/reconcile_data_plane.go | 22 +++++++++++++++---- .../controller/shard/reconcile_deletion.go | 7 +++++- .../controller/shard/reconcile_readiness.go | 11 +++++++++- test/suite/scenario_race_test.go | 18 ++++----------- 4 files changed, 38 insertions(+), 20 deletions(-) diff --git a/pkg/resource-handler/controller/shard/reconcile_data_plane.go b/pkg/resource-handler/controller/shard/reconcile_data_plane.go index 57ac0301..0ce880e1 100644 --- a/pkg/resource-handler/controller/shard/reconcile_data_plane.go +++ b/pkg/resource-handler/controller/shard/reconcile_data_plane.go @@ -205,8 +205,12 @@ func (r *ShardReconciler) reconcileDataPlane( shard, fmt.Sprintf("Failed to check backup health: %v", err), ) - if patchErr := r.Status(). - Patch(ctx, shard, client.MergeFrom(backupBase)); patchErr != nil { + if patchErr := r.Status().Patch( + ctx, + shard, + client.MergeFrom(backupBase), + client.FieldOwner("multigres-resource-handler"), + ); patchErr != nil { return ctrl.Result{}, fmt.Errorf("update unavailable backup status: %w", patchErr) } } else if result != nil { @@ -222,7 +226,12 @@ func (r *ShardReconciler) reconcileDataPlane( r.Recorder.Event(shard, "Warning", "BackupStale", result.Message) } - if err := r.Status().Patch(ctx, shard, client.MergeFrom(backupBase)); err != nil { + if err := r.Status().Patch( + ctx, + shard, + client.MergeFrom(backupBase), + client.FieldOwner("multigres-resource-handler"), + ); err != nil { monitoring.RecordSpanError(childSpan, err) childSpan.End() logger.Error(err, "Failed to update shard backup status") @@ -317,7 +326,12 @@ func (r *ShardReconciler) reconcilePodRoles( } if rolesChanged { - if err := r.Status().Patch(ctx, shard, client.MergeFrom(statusBase)); err != nil { + if err := r.Status().Patch( + ctx, + shard, + client.MergeFrom(statusBase), + client.FieldOwner("multigres-resource-handler"), + ); err != nil { logger.Error(err, "Failed to update shard pod roles") } } diff --git a/pkg/resource-handler/controller/shard/reconcile_deletion.go b/pkg/resource-handler/controller/shard/reconcile_deletion.go index 4a949499..af17193c 100644 --- a/pkg/resource-handler/controller/shard/reconcile_deletion.go +++ b/pkg/resource-handler/controller/shard/reconcile_deletion.go @@ -342,7 +342,12 @@ func (r *ShardReconciler) handlePendingDeletion( ObservedGeneration: shard.Generation, LastTransitionTime: metav1.Now(), }) - if err := r.Status().Patch(ctx, shard, client.MergeFrom(statusBase)); err != nil { + if err := r.Status().Patch( + ctx, + shard, + client.MergeFrom(statusBase), + client.FieldOwner("multigres-resource-handler"), + ); err != nil { return ctrl.Result{}, fmt.Errorf("setting ReadyForDeletion condition: %w", err) } logger.Info("Set ReadyForDeletion condition") diff --git a/pkg/resource-handler/controller/shard/reconcile_readiness.go b/pkg/resource-handler/controller/shard/reconcile_readiness.go index fe6886f2..71234eae 100644 --- a/pkg/resource-handler/controller/shard/reconcile_readiness.go +++ b/pkg/resource-handler/controller/shard/reconcile_readiness.go @@ -71,7 +71,16 @@ func (r *ShardReconciler) reconcilePoolerReadiness( Reason: observation.Reason, Message: observation.Message, }) - if err := r.Status().Patch(ctx, pod, client.MergeFrom(base)); err != nil { + // Named apart from the Shard's own status manager: the claim here is over + // one condition on a Pod whose status otherwise belongs to kubelet, not + // over the Shard's status, and a manager name is the only record of which + // concern took a field. Same reasoning as the storage-class guard. + if err := r.Status().Patch( + ctx, + pod, + client.MergeFrom(base), + client.FieldOwner("multigres-resource-handler-readiness"), + ); err != nil { return fmt.Errorf("patch pooler readiness for pod %s: %w", pod.Name, err) } } diff --git a/test/suite/scenario_race_test.go b/test/suite/scenario_race_test.go index 8ac8a766..7b233ab8 100644 --- a/test/suite/scenario_race_test.go +++ b/test/suite/scenario_race_test.go @@ -64,21 +64,11 @@ func TestTwoControllersWriteOneShard(t *testing.T) { // managers. t.Run("field ownership on the Shard is disjoint", func(t *testing.T) { c := c.Sub(t) - // Several Shard status writes carry no field owner, so the API server - // attributes them to the manager that happens to be the process name, - // and that manager ends up co-owning fields the shard controller's own - // applier claims. Retire this pin with an explicit owner on every - // status write, and replace it with c.Empty on the conflicts. conflicts := c.fieldOwnershipConflicts(key) - c.KnownDefect("MGO-SHARD-STATUS-WRITES-NO-FIELD-OWNER", func() error { - if len(conflicts) == 0 { - return nil - } - return fmt.Errorf( - "Shard %s has fields claimed by more than one field manager, "+ - "the same shape of defect as the status hot loop:\n %s", - key.Name, strings.Join(conflicts, "\n ")) - }) + c.Empty(conflicts, + "Shard %s has fields claimed by more than one field manager, "+ + "the same shape of defect as the status hot loop:\n %s", + key.Name, strings.Join(conflicts, "\n ")) }) t.Run("tablegroup's patches to the Shard never change it", func(t *testing.T) { From 235f5ede8a4b93d9d03d8b0f0d86df1505033d72 Mon Sep 17 00:00:00 2001 From: Brent Graveland Date: Fri, 25 Sep 2026 09:48:42 -0600 Subject: [PATCH 6/7] chore: build with Go 1.27.1 testkit's assertions are generic methods, which need Go 1.27, so adding it as a test dependency raised this module's go directive from 1.26.6. Pin it to the exact release and move the builder image with it: official Go images set GOTOOLCHAIN=local, so a 1.26.6 builder refuses a module that asks for 1.27. golangci-lint goes to v2.13.2. v2.12.2's bundled staticcheck IR builder cannot parse Go 1.27 syntax and panics on the standard library (buildir: package "poll": unexpected expr: *ast.KeyValueExpr), which fails every lint run on Linux. Go 1.27 support landed in v2.13.0. The newer staticcheck names deprecated symbols by full package path, so four of the existing controller-runtime deprecation exclusions stopped matching. Their patterns now accept either form rather than adding new exclusions; the deprecations and the follow-up migration are unchanged. The golangci-lint binary's name now also carries the Go version it was built with. CI restores bin/ from older caches, and a name keyed only on the linter's version reused a binary built by go1.26 after the bump, which refused to load a go1.27 module. Any future Go bump would repeat that. tools/observer is a separate module that does not use testkit and stays on 1.26.6. Signed-off-by: Brent Graveland --- .golangci.toml | 13 +++++++++---- Dockerfile | 2 +- Makefile | 16 ++++++++++------ go.mod | 2 +- 4 files changed, 21 insertions(+), 12 deletions(-) diff --git a/.golangci.toml b/.golangci.toml index cf89d727..ce4a124a 100644 --- a/.golangci.toml +++ b/.golangci.toml @@ -10,11 +10,16 @@ enable = [ "gosec" ] # for a dedicated follow-up migration; excluded here so the dependency bump # that introduced the deprecations isn't blocked on an unrelated, # wide-reaching refactor. +# +# The optional package-path prefix is there because staticcheck names the +# symbol differently across versions: bare ("client.Apply") before +# golangci-lint v2.13, fully qualified ("sigs.k8s.io/.../pkg/client.Apply") +# from it. rules = [ - { linters = [ "staticcheck" ], text = "SA1019: client.Apply is deprecated" }, - { linters = [ "staticcheck" ], text = "SA1019: scheme.Builder is deprecated" }, - { linters = [ "staticcheck" ], text = "SA1019: webhook.CustomDefaulter is deprecated" }, - { linters = [ "staticcheck" ], text = "SA1019: webhook.CustomValidator is deprecated" }, + { linters = [ "staticcheck" ], text = 'SA1019: (\S+/)?client\.Apply is deprecated' }, + { linters = [ "staticcheck" ], text = 'SA1019: (\S+/)?scheme\.Builder is deprecated' }, + { linters = [ "staticcheck" ], text = 'SA1019: (\S+/)?webhook\.CustomDefaulter is deprecated' }, + { linters = [ "staticcheck" ], text = 'SA1019: (\S+/)?webhook\.CustomValidator is deprecated' }, { linters = [ "staticcheck" ], text = "WithCustomDefaulter is deprecated" }, { linters = [ "staticcheck" ], text = "WithCustomValidator is deprecated" }, { linters = [ "staticcheck" ], text = "GetEventRecorderFor is deprecated" }, diff --git a/Dockerfile b/Dockerfile index 3c215ac3..32a24b31 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,6 +1,6 @@ # Containerfile for multigres-operator -FROM --platform=$BUILDPLATFORM golang:1.26.6-alpine3.23 AS builder +FROM --platform=$BUILDPLATFORM golang:1.27.1-alpine3.23 AS builder ARG TARGETOS ARG TARGETARCH diff --git a/Makefile b/Makefile index a842b125..2c274622 100644 --- a/Makefile +++ b/Makefile @@ -108,7 +108,7 @@ KUSTOMIZE_VERSION ?= v5.6.0 # renovate: datasource=github-releases depName=kubernetes-sigs/controller-tools CONTROLLER_TOOLS_VERSION ?= v0.18.0 # renovate: datasource=github-releases depName=golangci/golangci-lint -GOLANGCI_LINT_VERSION ?= v2.12.2 +GOLANGCI_LINT_VERSION ?= v2.13.2 CERT_MANAGER_VERSION ?= v1.19.2 @@ -722,10 +722,13 @@ $(ENVTEST): $(LOCALBIN) golangci-lint: $(GOLANGCI_LINT) ## Download golangci-lint locally if necessary. # golangci-lint's own go.mod selects an older toolchain than this module # targets, and a linter built with a lower Go version refuses to run. Pin the -# build toolchain to the one resolved by this module's go.mod. +# build toolchain to the one resolved by this module's go.mod, and put that +# version in the binary's name: CI restores bin/ from older caches, and a +# name keyed only on the linter's version would reuse a binary built by the +# previous toolchain after a Go bump. $(GOLANGCI_LINT): export GOTOOLCHAIN = $(shell go env GOVERSION) $(GOLANGCI_LINT): $(LOCALBIN) - $(call go-install-tool,$(GOLANGCI_LINT),github.com/golangci/golangci-lint/v2/cmd/golangci-lint,$(GOLANGCI_LINT_VERSION)) + $(call go-install-tool,$(GOLANGCI_LINT),github.com/golangci/golangci-lint/v2/cmd/golangci-lint,$(GOLANGCI_LINT_VERSION),$(shell go env GOVERSION)) .PHONY: install-certmanager install-certmanager: ## Install Cert-Manager into the cluster @@ -738,16 +741,17 @@ install-certmanager: ## Install Cert-Manager into the cluster # $1 - target path with name of binary # $2 - package url which can be installed # $3 - specific version of package +# $4 - optional extra suffix for the binary's name, e.g. the Go version it was built with define go-install-tool -@[ -f "$(1)-$(3)" ] && [ "$$(readlink -- "$(1)" 2>/dev/null)" = "$(1)-$(3)" ] || { \ +@[ -f "$(1)-$(3)$(if $(4),-$(4))" ] && [ "$$(readlink -- "$(1)" 2>/dev/null)" = "$(1)-$(3)$(if $(4),-$(4))" ] || { \ set -e; \ package=$(2)@$(3) ;\ echo "Downloading $${package}" ;\ rm -f $(1) ;\ GOBIN=$(LOCALBIN) go install $${package} ;\ -mv $(1) $(1)-$(3) ;\ +mv $(1) $(1)-$(3)$(if $(4),-$(4)) ;\ } ;\ -ln -sf $$(realpath $(1)-$(3)) $(1) +ln -sf $$(realpath $(1)-$(3)$(if $(4),-$(4))) $(1) endef ##@ Backward Compatibility Aliases diff --git a/go.mod b/go.mod index ea64cd6e..9baf7fbd 100644 --- a/go.mod +++ b/go.mod @@ -1,6 +1,6 @@ module github.com/multigres/multigres-operator -go 1.27 +go 1.27.1 require ( github.com/go-logr/logr v1.4.4 From 7d062a19c2a92faa562024bd376a8462ef03a32e Mon Sep 17 00:00:00 2001 From: Brent Graveland Date: Sat, 26 Sep 2026 12:34:50 -0600 Subject: [PATCH 7/7] fix(shard): requeue while a shard has not converged The commit that stopped the shard controller's status hot loop ("stop the status hot loop on a healthy shard") removed a driver that used to rerun every shard several times a second regardless of what reconcile asked for. That accidentally covered for a gap: once a posture observation settles (nothing inconsistent, nothing incomplete, or an unsettled observation accepted after its debounce), the shard requests no further reconcile at all, even when a managed pod has never reached posture readiness, or a shard mid-bootstrap has every pooler registered but no primary elected yet. Nothing in Kubernetes watches the topology store, so nothing else wakes the shard either: an accepted RPC blip or role mismatch leaves a pod NotReady or a shard Degraded until controller-runtime's 10h resync, and a fresh cluster's pool pods never go Ready at all. reconcilePosture now requests a requeue for any not-converged state, unsettled or merely not-ready, once the existing debounce for unsettled observations ends. The delay is clamped elapsed time since the shard was first observed not converged, five seconds to one minute, with up to 20% upward jitter (so five seconds to about seventy-two seconds including jitter), and resets the moment the shard converges. Elapsed time rather than a per-reconcile count, because pod status transitions, drain requeues and the operator's own status patches are each their own reconcile with no predicate filtering them, and a burst of those must not by itself run the backoff up to its ceiling. Scopes the posture pod list to pool pods: a shard's multiorch pod carries the same four identity labels and would otherwise count as a pod that never becomes ready. The pod-roles, drain, and pooler-prune lists are scoped the same way for consistency; pooler-prune's filter is the one that matters operationally, since it decides which topology poolers this reconcile marks LIFECYCLE_SHUTDOWN, and every pool pod has carried the component label since the pool controller was introduced, so this is not a behaviour change for existing clusters. test/suite's pool-scale-up pin is now a positive assertion: the role landing in status.podRoles after a scale-up whose pooler registers late used to be a coin flip, and is deterministic with this fix in place. Signed-off-by: Brent Graveland --- pkg/data-handler/posture/posture.go | 11 +- .../controller/shard/reconcile_data_plane.go | 197 +++- ...oncile_data_plane_posture_internal_test.go | 22 +- .../controller/shard/reconcile_deletion.go | 4 + .../shard/registration_requeue_test.go | 858 ++++++++++++++++++ .../controller/shard/shard_controller.go | 14 + test/suite/scenario_thrash_test.go | 87 +- 7 files changed, 1135 insertions(+), 58 deletions(-) create mode 100644 pkg/resource-handler/controller/shard/registration_requeue_test.go diff --git a/pkg/data-handler/posture/posture.go b/pkg/data-handler/posture/posture.go index 049b8ceb..8d0a83b6 100644 --- a/pkg/data-handler/posture/posture.go +++ b/pkg/data-handler/posture/posture.go @@ -47,6 +47,15 @@ type Readiness struct { Message string } +// reasonAwaitingRegistration is the readiness reason carried by a managed pod +// that has no corresponding pooler in the shard topology. +// +// Every managed pod is seeded with it and only overwritten once a topology +// entry matches. Note this is NOT what Result.Incomplete reports: that covers +// an unreachable cell, a topology entry with no matching pod, or an UNKNOWN +// posture, all of which are the opposite direction. +const reasonAwaitingRegistration = "AwaitingRegistration" + // Evaluate compares each managed pooler's observed postgres state with its // topology role. It returns nil when topology contains no active poolers, as // during bootstrap. @@ -61,7 +70,7 @@ func Evaluate( readiness := make(map[string]Readiness, len(managedPodNames)) for _, podName := range managedPodNames { readiness[podName] = Readiness{ - Reason: "AwaitingRegistration", + Reason: reasonAwaitingRegistration, Message: "pooler has not registered in the shard topology", } } diff --git a/pkg/resource-handler/controller/shard/reconcile_data_plane.go b/pkg/resource-handler/controller/shard/reconcile_data_plane.go index 0ce880e1..7bb1bd49 100644 --- a/pkg/resource-handler/controller/shard/reconcile_data_plane.go +++ b/pkg/resource-handler/controller/shard/reconcile_data_plane.go @@ -10,6 +10,7 @@ import ( corev1 "k8s.io/api/core/v1" meta "k8s.io/apimachinery/pkg/api/meta" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/util/wait" ctrl "sigs.k8s.io/controller-runtime" "sigs.k8s.io/controller-runtime/pkg/client" "sigs.k8s.io/controller-runtime/pkg/log" @@ -73,7 +74,12 @@ func (r *ShardReconciler) reconcileDataPlane( // A resolver error breaks the sequence of posture observations. It must // not let a strike from before the transport outage combine with the - // first unsettled observation after recovery. + // first unsettled observation after recovery. notConvergedSince needs no + // matching reset here: this path (like the nil-topology one below) + // returns its own fixed delay independent of that map, so a stale entry + // left over from before the outage only shortens the path to the + // backoff's ceiling once posture resumes, it never suppresses a requeue + // that is actually owed. That is intentional, not a gap. r.recordPostureObservation(shard, false) // A partial observation cannot clear a confirmed posture failure. Keep a @@ -279,12 +285,16 @@ func (r *ShardReconciler) reconcilePodRoles( ) { logger := log.FromContext(ctx) - // List managed pods for this shard (same pattern as reconcilePoolerPrune). + // List managed pool pods for this shard (same pattern as + // reconcilePoolerPrune). Scoped to pool pods: only they can ever match a + // topology pooler entry, and a shard's multiorch pod carries the same four + // identity labels but is never one. lbls := map[string]string{ metadata.LabelMultigresCluster: shard.Labels[metadata.LabelMultigresCluster], metadata.LabelMultigresDatabase: string(shard.Spec.DatabaseName), metadata.LabelMultigresTableGroup: string(shard.Spec.TableGroupName), metadata.LabelMultigresShard: string(shard.Spec.ShardName), + metadata.LabelAppComponent: PoolComponentName, } podList := &corev1.PodList{} if err := r.List(ctx, podList, @@ -345,6 +355,22 @@ const ( // client cannot be built yet (e.g. operator client cert not issued). poolerClientRetryDelay = 10 * time.Second + // A shard that is not converged, whatever the reason, is requeued on a + // backoff rather than left to the 10h resync. The delay is clamped elapsed + // time since the shard was first observed not-converged, not a per-pass + // count: a burst of unrelated pod/status events must not itself advance + // the backoff. The floor is the multipooler container's own readiness + // probe period, so the operator never polls topology more aggressively + // than the kubelet polls the pod, and the ceiling matches the + // empty-topology path. + readinessBackoffMinDelay = 5 * time.Second + readinessBackoffMaxDelay = time.Minute + // readinessBackoffJitter adds upward jitter. controller-runtime does not + // jitter RequeueAfter, and a fleet that lost convergence together + // (topology restart, mass scale-up) would otherwise retry in lockstep + // against the topology server that just came back. + readinessBackoffJitter = 0.2 + reasonPoolerClientUnavailable = "PoolerClientUnavailable" reasonAwaitingPoolerRegistration = "AwaitingPoolerRegistration" reasonObservationPending = "ObservationPending" @@ -357,11 +383,17 @@ func (r *ShardReconciler) reconcilePosture( shard *multigresv1alpha1.Shard, rpcClient rpcclient.MultipoolerClient, ) (time.Duration, error) { + // Pool pods only: a shard's multiorch pod carries the same four identity + // labels (buildMultiorchLabelsWithCell merges them in) but is never a + // pooler, so an unfiltered list seeds it AwaitingRegistration forever and + // anyPodNotReady never clears. reconcilePoolerReadiness already scopes + // this way; posture.Evaluate must match it. lbls := map[string]string{ metadata.LabelMultigresCluster: shard.Labels[metadata.LabelMultigresCluster], metadata.LabelMultigresDatabase: string(shard.Spec.DatabaseName), metadata.LabelMultigresTableGroup: string(shard.Spec.TableGroupName), metadata.LabelMultigresShard: string(shard.Spec.ShardName), + metadata.LabelAppComponent: PoolComponentName, } podList := &corev1.PodList{} if err := r.List(ctx, podList, @@ -388,6 +420,8 @@ func (r *ShardReconciler) reconcilePosture( // An empty topology is expected during bootstrap, but it is not a settled // posture observation. Keep polling until poolers register rather than // leaving a previous transport condition stuck until the periodic resync. + // This path always returns a fixed 1-minute delay of its own regardless of + // notConvergedSince, so it needs no matching reset there either. r.recordPostureObservation(shard, false) setPostureUnknownUnlessFalse( shard, @@ -450,12 +484,78 @@ func (r *ShardReconciler) reconcilePosture( r.Recorder.Event(shard, "Warning", reason, result.Message) } + // A shard is converged only once posture is settled AND every managed pod + // has reached posture readiness. Neither half implies the other: an + // accepted mismatch or accepted Incomplete observation (strikes past + // threshold) is settled but still has a pod sitting NotReady, and a shard + // mid-bootstrap with every pooler registered but no primary elected yet is + // neither inconsistent nor incomplete but has nothing actually ready. + // Registration itself, and any change multiorch makes in topology, are + // writes to the topology store with no Kubernetes event behind them, so + // nothing but this controller's own requeue will ever look again. Without + // it, a settled-but-not-ready shard sits until controller-runtime's 10h + // resync; a shard with a genuinely stuck pod (Pending, crash-looping, + // quarantined) sits forever, since a lost watch never recovers it. + // + // The debounce above stays authoritative for whether an unsettled + // observation is accepted into status: this never fires while + // `unsettled && strikes < postureStrikeThreshold`, so it cannot preempt or + // shorten that window, only pick up once it ends. + notConverged := unsettled || anyPodNotReady(result.Readiness) + elapsed := r.recordNotConverged(shard, notConverged) + if unsettled && strikes < postureStrikeThreshold { return postureDebounceRequeueDelay, nil } + if notConverged { + return readinessBackoffDelay(elapsed), nil + } return 0, nil } +// anyPodNotReady reports whether any managed pod has not yet reached posture +// readiness. +// +// Deliberately broader than a check for the seeded "AwaitingRegistration" +// reason alone. A pod whose pooler has never registered carries that reason, +// but every other reason poolerReadiness can return (NotInitialized, +// PostgresNotReady, CohortIneligible, NotCohortMember) means Ready is false +// too, and all of them are reachable by a shard that Evaluate reports as +// neither inconsistent nor incomplete, since that result only compares +// postures against topology roles and does not require a primary to exist. A +// shard mid-bootstrap with every pooler registered but no primary elected +// settles there, looking "consistent" while nothing is actually ready. +func anyPodNotReady(readiness map[string]posture.Readiness) bool { + for _, r := range readiness { + if !r.Ready { + return true + } + } + return false +} + +// readinessBackoffDelay clamps elapsed (time since the shard was first +// observed not-converged) to [readinessBackoffMinDelay, +// readinessBackoffMaxDelay], with upward jitter. +// +// Elapsed-time-based rather than a per-pass count: Owns(&corev1.Pod{}) and +// For(&Shard{}) carry no predicates, so a pod's Pending -> ContainerCreating -> +// per-container-Ready transitions, an operator gate patch, and a drain's 2s +// requeues are each their own reconcile. A scale-up alone is at least five of +// those before there is anything to register, and a wall-clock burst of +// events must not by itself run the delay up to the ceiling: two reconciles a +// second apart are five seconds not-converged either way, whether they were +// one pass or five. +func readinessBackoffDelay(elapsed time.Duration) time.Duration { + d := elapsed + if d < readinessBackoffMinDelay { + d = readinessBackoffMinDelay + } else if d > readinessBackoffMaxDelay { + d = readinessBackoffMaxDelay + } + return wait.Jitter(d, readinessBackoffJitter) +} + func setPostureUnknownUnlessFalse( shard *multigresv1alpha1.Shard, reason string, @@ -504,6 +604,18 @@ func withDataPlaneRequeue( return result } +// recordPostureObservation counts consecutive unsettled observations for one +// shard and returns the running total. +// +// A settled observation deletes the entry rather than writing zero. The two +// mean the same thing to every reader, since a missing key reads as zero, but +// they differ in what the map holds: writing zero keeps an entry for every +// shard this process has ever reconciled, while deleting keeps only the +// shards currently accumulating strikes, which in a healthy cluster is none. +// A Go map's table is sized by its peak simultaneous entries and does not +// shrink on delete, so this is what keeps that peak bounded by currently +// unsettled shards rather than by cumulative shards over the operator's +// lifetime. func (r *ShardReconciler) recordPostureObservation( shard *multigresv1alpha1.Shard, unsettled bool, @@ -512,17 +624,76 @@ func (r *ShardReconciler) recordPostureObservation( r.postureStrikesMu.Lock() defer r.postureStrikesMu.Unlock() + if !unsettled { + delete(r.postureStrikes, key) + return 0 + } if r.postureStrikes == nil { r.postureStrikes = make(map[string]int) } - if unsettled { - r.postureStrikes[key]++ - } else { - r.postureStrikes[key] = 0 - } + r.postureStrikes[key]++ return r.postureStrikes[key] } +// recordNotConverged tracks, per shard, the time a shard was first observed +// not converged (unsettled, or some managed pod not posture-ready), and +// returns how long that has been true. Same delete-on-settle pattern as +// recordPostureObservation, and a separate map for the same reason that one +// is not reused here: this one picks the backoff, and posture strikes must +// keep gating only posture.Apply. +// +// Storing the first-seen time rather than a per-pass count is what makes +// readinessBackoffDelay a function of elapsed wall-clock time instead of +// event count; see readinessBackoffDelay's own doc for why that matters. +func (r *ShardReconciler) recordNotConverged( + shard *multigresv1alpha1.Shard, + notConverged bool, +) time.Duration { + key := fmt.Sprintf("%s/%s", shard.Namespace, shard.Name) + now := r.now() + + r.notConvergedMu.Lock() + defer r.notConvergedMu.Unlock() + if !notConverged { + delete(r.notConvergedSince, key) + return 0 + } + if r.notConvergedSince == nil { + r.notConvergedSince = make(map[string]time.Time) + } + first, ok := r.notConvergedSince[key] + if !ok { + first = now + r.notConvergedSince[key] = first + } + return now.Sub(first) +} + +// now returns the reconciler's clock, defaulting to time.Now so production +// code never has to set Clock. Tests inject Clock to drive +// recordNotConverged's elapsed-time math deterministically. +func (r *ShardReconciler) now() time.Time { + if r.Clock != nil { + return r.Clock() + } + return time.Now() +} + +// forgetStrikes drops namespace/name's entry from both strike maps. Called on +// a Shard's deletion and not-found paths so a shard deleted mid-backoff, in +// either counter, does not leak its entry for the life of the process. +func (r *ShardReconciler) forgetStrikes(namespace, name string) { + key := fmt.Sprintf("%s/%s", namespace, name) + + r.postureStrikesMu.Lock() + delete(r.postureStrikes, key) + r.postureStrikesMu.Unlock() + + r.notConvergedMu.Lock() + delete(r.notConvergedSince, key) + r.notConvergedMu.Unlock() +} + // reconcileDrainState iterates pods with drain annotations and runs the // drain state machine for each one. func (r *ShardReconciler) reconcileDrainState( @@ -532,11 +703,17 @@ func (r *ShardReconciler) reconcileDrainState( ) (bool, error) { logger := log.FromContext(ctx) + // Pool pods only: the drain-requested annotation this loop acts on is only + // ever set by the pool scale-down/rolling-update path (reconcile_pool_pods.go), + // never on a shard's multiorch pod, so this is currently a no-op filter. + // Scoped anyway for the same reason reconcilePosture now is: relying on an + // annotation nothing else sets is a coincidence, not a guarantee. lbls := map[string]string{ metadata.LabelMultigresCluster: shard.Labels[metadata.LabelMultigresCluster], metadata.LabelMultigresDatabase: string(shard.Spec.DatabaseName), metadata.LabelMultigresTableGroup: string(shard.Spec.TableGroupName), metadata.LabelMultigresShard: string(shard.Spec.ShardName), + metadata.LabelAppComponent: PoolComponentName, } podList := &corev1.PodList{} if err := r.List( @@ -655,11 +832,17 @@ func (r *ShardReconciler) reconcilePoolerPrune( return } + // Pool pods only: topo.MarkDeadPoolers matches this set's names against + // topology pooler entries, and a multiorch pod's name never matches one + // (it is not a pooler), so including it here is currently a no-op. Scoped + // anyway to keep this list's meaning ("pods that can be poolers") aligned + // with what it is actually used for. lbls := map[string]string{ metadata.LabelMultigresCluster: shard.Labels[metadata.LabelMultigresCluster], metadata.LabelMultigresDatabase: string(shard.Spec.DatabaseName), metadata.LabelMultigresTableGroup: string(shard.Spec.TableGroupName), metadata.LabelMultigresShard: string(shard.Spec.ShardName), + metadata.LabelAppComponent: PoolComponentName, } podList := &corev1.PodList{} if err := r.List( diff --git a/pkg/resource-handler/controller/shard/reconcile_data_plane_posture_internal_test.go b/pkg/resource-handler/controller/shard/reconcile_data_plane_posture_internal_test.go index 28e2a918..cd83c8e0 100644 --- a/pkg/resource-handler/controller/shard/reconcile_data_plane_posture_internal_test.go +++ b/pkg/resource-handler/controller/shard/reconcile_data_plane_posture_internal_test.go @@ -114,6 +114,10 @@ func postureTestReconciler( }, c } +// postureTestPod is a pool pod: reconcilePosture's own pod list is scoped to +// PoolComponentName (a shard's multiorch pod carries the same four identity +// labels but is never a pooler), so a pod fixture without this label is +// invisible to it regardless of what the test otherwise sets up. func postureTestPod() *corev1.Pod { return &corev1.Pod{ObjectMeta: metav1.ObjectMeta{ Name: "pooler-0", @@ -123,6 +127,7 @@ func postureTestPod() *corev1.Pod { metadata.LabelMultigresDatabase: "database", metadata.LabelMultigresTableGroup: "table-group", metadata.LabelMultigresShard: "0", + metadata.LabelAppComponent: PoolComponentName, }, }} } @@ -244,8 +249,14 @@ func TestReconcilePostureDebouncesFirstInconsistency(t *testing.T) { if err != nil { t.Fatalf("second reconcilePosture() error = %v", err) } - if retryAfter != 0 { - t.Error("second inconsistent posture observation requested another debounce requeue") + // The mismatch is now accepted into status (PostureConsistent=False), but + // this mock pooler never reports IsInitialized/PostgresReady, so it has + // also never reached posture readiness. An accepted-but-not-ready shard + // must keep requesting a requeue: nothing but this controller's own + // backoff will ever look again, since neither a status recovery in + // topology nor a role fix changes a Kubernetes object. + if retryAfter <= 0 { + t.Error("second inconsistent posture observation (still not ready) requested no requeue") } if !conditionIsFalse(shard.Status.Conditions, posture.ConditionConsistent) { t.Errorf( @@ -296,8 +307,11 @@ func TestReconcilePostureDebouncesFirstIncompleteObservation(t *testing.T) { if err != nil { t.Fatalf("second reconcilePosture() error = %v", err) } - if retryAfter != 0 { - t.Error("second incomplete posture observation requested another debounce requeue") + // Accepted into status (Unknown/ObservationIncomplete), but a pooler whose + // Status RPC errors has also never reached posture readiness, so this must + // still requeue rather than strand the pod until the 10h resync. + if retryAfter <= 0 { + t.Error("second incomplete posture observation (still not ready) requested no requeue") } for _, condition := range shard.Status.Conditions { if condition.Type != posture.ConditionConsistent { diff --git a/pkg/resource-handler/controller/shard/reconcile_deletion.go b/pkg/resource-handler/controller/shard/reconcile_deletion.go index af17193c..13dbaa6b 100644 --- a/pkg/resource-handler/controller/shard/reconcile_deletion.go +++ b/pkg/resource-handler/controller/shard/reconcile_deletion.go @@ -130,6 +130,10 @@ func (r *ShardReconciler) handleDeletion( return ctrl.Result{}, err } + // The shard is done reconciling; drop its strike entries so a shard + // deleted mid-backoff does not leak its counters. + r.forgetStrikes(shard.Namespace, shard.Name) + // Remove the finalizer last so Kubernetes can finish deletion now that the // PVC cleanup has run. if slices.Contains(shard.Finalizers, shardFinalizer) { diff --git a/pkg/resource-handler/controller/shard/registration_requeue_test.go b/pkg/resource-handler/controller/shard/registration_requeue_test.go new file mode 100644 index 00000000..b9a5ff99 --- /dev/null +++ b/pkg/resource-handler/controller/shard/registration_requeue_test.go @@ -0,0 +1,858 @@ +package shard + +import ( + "errors" + "fmt" + "testing" + "time" + + "github.com/multigres/multigres/go/common/rpcclient" + "github.com/multigres/multigres/go/common/topoclient" + "github.com/multigres/multigres/go/common/topoclient/memorytopo" + "github.com/multigres/multigres/go/pb/clustermetadata" + multipoolermanagerdatapb "github.com/multigres/multigres/go/pb/multipoolermanagerdata" + corev1 "k8s.io/api/core/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/types" + "k8s.io/client-go/tools/record" + ctrl "sigs.k8s.io/controller-runtime" + "sigs.k8s.io/controller-runtime/pkg/client" + "sigs.k8s.io/controller-runtime/pkg/client/fake" + + multigresv1alpha1 "github.com/multigres/multigres-operator/api/v1alpha1" + "github.com/multigres/multigres-operator/pkg/data-handler/poolerclient" + "github.com/multigres/multigres-operator/pkg/util/metadata" +) + +// fakeClock lets a test drive recordNotConverged's elapsed-time math without +// sleeping. Set r.Clock = clk.now. +type fakeClock struct { + t time.Time +} + +func (c *fakeClock) now() time.Time { return c.t } + +func (c *fakeClock) advance(d time.Duration) { c.t = c.t.Add(d) } + +// wantDelayRange mirrors readinessBackoffDelay's own clamp so a change to +// that clamp is caught here rather than only against itself. +func wantDelayRange(elapsed time.Duration) (min, max time.Duration) { + min = elapsed + if min < readinessBackoffMinDelay { + min = readinessBackoffMinDelay + } else if min > readinessBackoffMaxDelay { + min = readinessBackoffMaxDelay + } + max = time.Duration(float64(min) * 1.2) + return min, max +} + +// TestReadinessBackoffDelayClampsElapsedTime pins the clamp. The lower bound +// matters as much as the upper one: a delay that could return zero would +// reinstate the defect this exists to fix, since a zero RequeueAfter means no +// requeue at all rather than an immediate one. +func TestReadinessBackoffDelayClampsElapsedTime(t *testing.T) { + t.Parallel() + + for _, tc := range []struct { + elapsed time.Duration + min, max time.Duration + }{ + {elapsed: 0, min: 5 * time.Second, max: 6 * time.Second}, + {elapsed: 3 * time.Second, min: 5 * time.Second, max: 6 * time.Second}, + {elapsed: 5 * time.Second, min: 5 * time.Second, max: 6 * time.Second}, + {elapsed: 30 * time.Second, min: 30 * time.Second, max: 36 * time.Second}, + {elapsed: 59 * time.Second, min: 59 * time.Second, max: 70800 * time.Millisecond}, + {elapsed: time.Minute, min: time.Minute, max: 72 * time.Second}, + {elapsed: 70 * time.Second, min: time.Minute, max: 72 * time.Second}, + {elapsed: time.Hour, min: time.Minute, max: 72 * time.Second}, + } { + // Jitter is random, so the bound has to hold across repeats rather + // than on one draw. + for range 200 { + got := readinessBackoffDelay(tc.elapsed) + if got < tc.min || got > tc.max { + t.Fatalf("elapsed=%v: delay %v outside [%v, %v]", + tc.elapsed, got, tc.min, tc.max) + } + } + } +} + +// TestReadinessBackoffDelayIsNeverZero is the one that dies if the clamp is +// removed. A zero duration is not "retry immediately", it is "do not +// requeue", which is exactly how a shard ends up stranded not-converged. +func TestReadinessBackoffDelayIsNeverZero(t *testing.T) { + t.Parallel() + + for _, elapsed := range []time.Duration{ + -5 * time.Second, -1, 0, time.Second, 5 * time.Second, + 30 * time.Second, time.Minute, time.Hour, + } { + for range 50 { + if got := readinessBackoffDelay(elapsed); got <= 0 { + t.Fatalf("elapsed=%v produced a non-positive delay %v, "+ + "which controller-runtime reads as no requeue", elapsed, got) + } + } + } +} + +// TestReadinessBackoffDelayJitters guards the fleet-lockstep property: a +// constant delay would have every shard that lost convergence together retry +// in lockstep against the topology server that just came back. +func TestReadinessBackoffDelayJitters(t *testing.T) { + t.Parallel() + + seen := map[time.Duration]bool{} + for range 200 { + seen[readinessBackoffDelay(30*time.Second)] = true + } + if len(seen) < 10 { + t.Fatalf("only %d distinct delays across 200 draws; jitter is not applied", len(seen)) + } +} + +// shardNamed is the minimum a strike counter reads. +func shardNamed(ns, name string) *multigresv1alpha1.Shard { + return &multigresv1alpha1.Shard{ + ObjectMeta: metav1.ObjectMeta{Namespace: ns, Name: name}, + } +} + +// TestPostureStrikesLeaveNoEntryOnceSettled is what deleting on settle +// actually buys: a map's table is sized by its peak simultaneous entries and +// does not shrink on delete, so writing zero instead kept an entry for every +// shard the process had ever reconciled. +func TestPostureStrikesLeaveNoEntryOnceSettled(t *testing.T) { + t.Parallel() + + r := &ShardReconciler{} + for i := range 1000 { + s := shardNamed("ns", fmt.Sprintf("shard-%d", i)) + r.recordPostureObservation(s, true) + r.recordPostureObservation(s, false) + } + + if got := len(r.postureStrikes); got != 0 { + t.Fatalf("a thousand shards seen and settled left %d entries, want 0", got) + } +} + +// TestPostureStrikesDoNotSurviveRecreation pins the intended semantic: a +// recreated Shard (a different object at the same namespace/name) opens at +// strike 1, not at whatever count its predecessor left. Posture strikes gate +// posture.Apply (when an unsettled observation is accepted into status); they +// do not pick a requeue delay, that is notConvergedSince's job. +func TestPostureStrikesDoNotSurviveRecreation(t *testing.T) { + t.Parallel() + + r := &ShardReconciler{} + s := shardNamed("ns", "shard-0") + + for range 5 { + r.recordPostureObservation(s, true) + } + if got := r.recordPostureObservation(s, false); got != 0 { + t.Fatalf("a settled observation reported %d strikes, want 0", got) + } + + // The replacement is a different object at the same key, which is what + // the tablegroup controller creates after a Shard is deleted. + if got := r.recordPostureObservation(shardNamed("ns", "shard-0"), true); got != 1 { + t.Fatalf("a recreated shard opened at %d strikes, want 1", got) + } +} + +// TestPostureStrikesCountConsecutiveUnsettled pins what the counter is for, +// so settling on delete cannot be "fixed" into never counting at all. +func TestPostureStrikesCountConsecutiveUnsettled(t *testing.T) { + t.Parallel() + + r := &ShardReconciler{} + s := shardNamed("ns", "shard-0") + for want := 1; want <= 3; want++ { + if got := r.recordPostureObservation(s, true); got != want { + t.Fatalf("consecutive unsettled observation %d reported %d strikes", want, got) + } + } + // Shards are counted independently, which is the only reason the map has + // keys at all. + if got := r.recordPostureObservation(shardNamed("ns", "other"), true); got != 1 { + t.Fatalf("a second shard opened at %d strikes, want 1", got) + } + if got := r.recordPostureObservation(s, true); got != 4 { + t.Fatalf("the first shard reported %d strikes after a second shard, want 4", got) + } +} + +// TestNotConvergedSinceLeavesNoEntryOnceSettled mirrors +// TestPostureStrikesLeaveNoEntryOnceSettled for the not-converged-since map: a +// shard deleted mid-backoff must not leak its entry, and neither must one +// that simply converges. +func TestNotConvergedSinceLeavesNoEntryOnceSettled(t *testing.T) { + t.Parallel() + + r := &ShardReconciler{} + for i := range 1000 { + s := shardNamed("ns", fmt.Sprintf("shard-%d", i)) + r.recordNotConverged(s, true) + r.recordNotConverged(s, false) + } + + if got := len(r.notConvergedSince); got != 0 { + t.Fatalf("a thousand shards seen and settled left %d entries, want 0", got) + } +} + +// TestNotConvergedSinceTracksElapsedTime pins the elapsed-time semantics: the +// first not-converged observation opens the clock, later ones read the time +// since then rather than a per-call count, and a settled observation clears +// it so a later not-converged spell starts over rather than resuming. +func TestNotConvergedSinceTracksElapsedTime(t *testing.T) { + t.Parallel() + + clk := &fakeClock{t: time.Unix(1_700_000_000, 0)} + r := &ShardReconciler{Clock: clk.now} + s := shardNamed("ns", "shard-0") + + if got := r.recordNotConverged(s, true); got != 0 { + t.Fatalf("first not-converged observation reported elapsed %v, want 0", got) + } + clk.advance(37 * time.Second) + if got := r.recordNotConverged(s, true); got != 37*time.Second { + t.Fatalf("second not-converged observation reported elapsed %v, want 37s", got) + } + // A burst of same-instant calls (a wave of unrelated pod events) must not + // itself advance the elapsed time. + if got := r.recordNotConverged(s, true); got != 37*time.Second { + t.Fatalf( + "third not-converged observation (no time passed) reported elapsed %v, want 37s", got, + ) + } + + if got := r.recordNotConverged(s, false); got != 0 { + t.Fatalf("a settled observation reported elapsed %v, want 0", got) + } + if got := r.recordNotConverged(s, true); got != 0 { + t.Fatalf("a fresh not-converged spell reported elapsed %v, want 0 (not resumed)", got) + } +} + +// TestForgetStrikesDropsBothCounters pins that a Shard's strike entries do +// not survive forgetStrikes, in either counter. +func TestForgetStrikesDropsBothCounters(t *testing.T) { + t.Parallel() + + r := &ShardReconciler{} + s := shardNamed("ns", "shard-0") + r.recordPostureObservation(s, true) + r.recordNotConverged(s, true) + + r.forgetStrikes(s.Namespace, s.Name) + + if got := len(r.postureStrikes); got != 0 { + t.Fatalf("posture strikes: %d entries survived forgetStrikes, want 0", got) + } + if got := len(r.notConvergedSince); got != 0 { + t.Fatalf("not-converged-since: %d entries survived forgetStrikes, want 0", got) + } +} + +// TestHandleDeletionForgetsStrikes drives the deletion cleanup through the +// real controller path (handleDeletion) rather than calling forgetStrikes +// directly, so a regression that stops handleDeletion from reaching it is +// caught here rather than only in the helper's own unit test. +func TestHandleDeletionForgetsStrikes(t *testing.T) { + shard := postureTestShard() + shard.Finalizers = []string{shardFinalizer} + now := metav1.Now() + shard.DeletionTimestamp = &now + + scheme := postureTestScheme(t) + c := fake.NewClientBuilder(). + WithScheme(scheme). + WithObjects(shard). + WithStatusSubresource(&multigresv1alpha1.Shard{}). + Build() + r := &ShardReconciler{ + Client: c, + Scheme: scheme, + Recorder: record.NewFakeRecorder(20), + } + + key := fmt.Sprintf("%s/%s", shard.Namespace, shard.Name) + r.postureStrikes = map[string]int{key: 2} + r.notConvergedSince = map[string]time.Time{key: time.Now()} + + if _, err := r.handleDeletion(t.Context(), shard); err != nil { + t.Fatalf("handleDeletion() error = %v", err) + } + + if _, ok := r.postureStrikes[key]; ok { + t.Errorf("posture strikes entry for %s survived handleDeletion", key) + } + if _, ok := r.notConvergedSince[key]; ok { + t.Errorf("not-converged-since entry for %s survived handleDeletion", key) + } +} + +// TestReconcileForgetsStrikesOnNotFound drives the not-found cleanup through +// Reconcile itself: a Shard already gone from the API server (the common case +// once handleDeletion above has already run and removed the finalizer) must +// still have its strike entries dropped, as a backstop. +func TestReconcileForgetsStrikesOnNotFound(t *testing.T) { + scheme := postureTestScheme(t) + c := fake.NewClientBuilder().WithScheme(scheme).Build() + r := &ShardReconciler{ + Client: c, + Scheme: scheme, + Recorder: record.NewFakeRecorder(20), + } + + key := "default/gone-shard" + r.postureStrikes = map[string]int{key: 3} + r.notConvergedSince = map[string]time.Time{key: time.Now()} + + req := ctrl.Request{ + NamespacedName: types.NamespacedName{Namespace: "default", Name: "gone-shard"}, + } + if _, err := r.Reconcile(t.Context(), req); err != nil { + t.Fatalf("Reconcile() error = %v", err) + } + + if _, ok := r.postureStrikes[key]; ok { + t.Errorf("posture strikes entry for %s survived Reconcile on a missing Shard", key) + } + if _, ok := r.notConvergedSince[key]; ok { + t.Errorf("not-converged-since entry for %s survived Reconcile on a missing Shard", key) + } +} + +// gateTestReconciler is postureTestReconciler plus a Pod status subresource, +// for tests that read back the PoolerDataReady gate condition a real +// apiserver would only apply through Status().Patch. +func gateTestReconciler( + t *testing.T, + shard *multigresv1alpha1.Shard, + rpc rpcclient.MultipoolerClient, + objects ...client.Object, +) (*ShardReconciler, client.Client) { + t.Helper() + scheme := postureTestScheme(t) + allObjects := append([]client.Object{shard}, objects...) + c := fake.NewClientBuilder(). + WithScheme(scheme). + WithObjects(allObjects...). + WithStatusSubresource(&multigresv1alpha1.Shard{}, &corev1.Pod{}). + Build() + return &ShardReconciler{ + Client: c, + Scheme: scheme, + Recorder: record.NewFakeRecorder(20), + PoolerClients: poolerclient.Static(rpc), + CreateTopoStore: newMemoryTopoFactory(), + }, c +} + +// registeredReplica registers id in store as a non-primary member of shard, +// with an RPC status that reads Ready via poolerReadiness: initialized, +// accepting connections, cohort-eligible, and a committed member of its own +// single-member rule. Standing in for "this pooler has finished bootstrapping +// and has somewhere to belong," independent of whether anyone in the shard has +// been elected primary. +func registeredReplica( + t *testing.T, + store topoclient.Store, + rpc *rpcclient.FakeClient, + shard *multigresv1alpha1.Shard, + cell, name string, +) topoclient.ComponentID { + t.Helper() + id := &clustermetadata.ID{Cell: cell, Name: name} + if err := store.RegisterMultipooler(t.Context(), &clustermetadata.Multipooler{ + Id: id, + Hostname: name, + ShardKey: &clustermetadata.ShardKey{ + Database: string(shard.Spec.DatabaseName), + TableGroup: string(shard.Spec.TableGroupName), + Shard: string(shard.Spec.ShardName), + }, + RoutingState: &clustermetadata.RoutingState{ + Role: clustermetadata.RoutingRole_ROUTING_ROLE_REPLICA, + }, + }, false); err != nil { + t.Fatalf("register pooler %s: %v", name, err) + } + + componentID := topoclient.ComponentIDString(id) + rpc.SetStatusResponse(componentID, readyStatusResponse(id)) + return componentID +} + +// readyStatusResponse is a fully posture-ready StatusResponse for id: a +// replica, initialized, accepting connections, cohort-eligible, and a +// committed member of its own single-member rule. +func readyStatusResponse(id *clustermetadata.ID) *multipoolermanagerdatapb.StatusResponse { + return &multipoolermanagerdatapb.StatusResponse{ + Status: &multipoolermanagerdatapb.Status{ + IsInitialized: true, + PostgresReady: true, + PostgresStatus: multipoolermanagerdatapb.PostgresStatus_POSTGRES_STATUS_STANDBY, + }, + AvailabilityStatus: &clustermetadata.AvailabilityStatus{ + CohortEligibilityStatus: &clustermetadata.CohortEligibilityStatus{ + Signal: clustermetadata.CohortEligibilitySignal_COHORT_ELIGIBILITY_SIGNAL_ELIGIBLE, + }, + }, + ConsensusStatus: &clustermetadata.ConsensusStatus{ + Id: id, + CurrentPosition: &clustermetadata.PoolerPosition{ + Position: &clustermetadata.RulePosition{ + Decision: &clustermetadata.ShardRule{ + RuleNumber: &clustermetadata.RuleNumber{CoordinatorTerm: 1}, + LeaderId: id, + CohortMembers: []*clustermetadata.ID{id}, + DurabilityPolicy: topoclient.AtLeastN(1), + }, + }, + }, + }, + } +} + +// notYetSettledReplica registers id in store as a non-primary member of +// shard, with an RPC status that is initialized, accepting connections, and +// cohort-eligible, but has not yet committed a rule naming any cohort +// members. Standing in for a shard mid-bootstrap where every pooler has +// registered but multiorch has not yet elected a primary or committed a +// durability rule: nobody is a "postgres primary" (so nothing looks +// inconsistent) and nobody has an unmatched topology entry or an unreadable +// RPC (so nothing looks incomplete), but nobody is ready either. +func notYetSettledReplica( + t *testing.T, + store topoclient.Store, + rpc *rpcclient.FakeClient, + shard *multigresv1alpha1.Shard, + cell, name string, +) *clustermetadata.ID { + t.Helper() + id := &clustermetadata.ID{Cell: cell, Name: name} + if err := store.RegisterMultipooler(t.Context(), &clustermetadata.Multipooler{ + Id: id, + Hostname: name, + ShardKey: &clustermetadata.ShardKey{ + Database: string(shard.Spec.DatabaseName), + TableGroup: string(shard.Spec.TableGroupName), + Shard: string(shard.Spec.ShardName), + }, + RoutingState: &clustermetadata.RoutingState{ + Role: clustermetadata.RoutingRole_ROUTING_ROLE_REPLICA, + }, + }, false); err != nil { + t.Fatalf("register pooler %s: %v", name, err) + } + + componentID := topoclient.ComponentIDString(id) + rpc.SetStatusResponse(componentID, &multipoolermanagerdatapb.StatusResponse{ + Status: &multipoolermanagerdatapb.Status{ + IsInitialized: true, + PostgresReady: true, + PostgresStatus: multipoolermanagerdatapb.PostgresStatus_POSTGRES_STATUS_STANDBY, + }, + AvailabilityStatus: &clustermetadata.AvailabilityStatus{ + CohortEligibilityStatus: &clustermetadata.CohortEligibilityStatus{ + Signal: clustermetadata.CohortEligibilitySignal_COHORT_ELIGIBILITY_SIGNAL_ELIGIBLE, + }, + }, + // No ConsensusStatus: nothing has been committed yet, so + // committedCohortContains is false for everyone. + }) + return id +} + +// TestReconcilePostureConvergedShardReturnsZero pins the hot-loop fix's own +// invariant: a converged shard returns 0 and leaves no map entry. It includes +// a multiorch pod built from BuildMultiorchDeployment, since a shard's +// multiorch pod carries the same four identity labels as its pool pods, so a +// selector that forgets to scope to pool pods seeds it AwaitingRegistration +// forever and this shard never returns 0. +func TestReconcilePostureConvergedShardReturnsZero(t *testing.T) { + shard := postureTestShard() + shard.Labels[metadata.LabelMultigresDatabase] = "database" + shard.Labels[metadata.LabelMultigresTableGroup] = "table-group" + shard.Labels[metadata.LabelMultigresShard] = "0" + + dep, err := BuildMultiorchDeployment(shard, "cell1", postureTestScheme(t)) + if err != nil { + t.Fatalf("BuildMultiorchDeployment() error = %v", err) + } + orch := &corev1.Pod{ObjectMeta: metav1.ObjectMeta{ + Name: "multiorch-abc", Namespace: shard.Namespace, Labels: dep.Spec.Template.Labels, + }} + + _, factory := memorytopo.NewServerAndFactory(t.Context(), "cell1") + store := topoclient.NewWithFactory(factory, "", []string{""}, topoclient.NewDefaultTopoConfig()) + defer func() { _ = store.Close() }() + rpc := rpcclient.NewFakeClient() + registeredReplica(t, store, rpc, shard, "cell1", "pooler-0") + + pool := postureTestPod() + r, _ := postureTestReconciler(t, shard, rpc, pool, orch) + + delay, err := r.reconcilePosture(t.Context(), store, shard, rpc) + if err != nil { + t.Fatalf("reconcilePosture() error = %v", err) + } + if delay != 0 { + t.Errorf("delay = %v, want 0 for a converged shard", delay) + } + key := fmt.Sprintf("%s/%s", shard.Namespace, shard.Name) + if _, ok := r.postureStrikes[key]; ok { + t.Errorf("posture strikes entry left for a converged shard") + } + if _, ok := r.notConvergedSince[key]; ok { + t.Errorf("not-converged-since entry left for a converged shard") + } + if conditionIsFalse(shard.Status.Conditions, "PostureConsistent") { + t.Errorf("conditions = %#v, want no failure for a converged shard", shard.Status.Conditions) + } +} + +// TestReconcilePostureAcceptedIncompleteObservationStillRequeues covers an +// accepted-but-not-ready observation: a Status RPC failure makes a pod +// UNKNOWN, which makes it Incomplete, which makes the shard unsettled. After +// the strike threshold the observation is accepted into status, but the +// failing pod has also never reached posture readiness, so this must keep +// requesting a requeue rather than stranding it until the 10h resync. +func TestReconcilePostureAcceptedIncompleteObservationStillRequeues(t *testing.T) { + shard := postureTestShard() + _, factory := memorytopo.NewServerAndFactory(t.Context(), "cell1") + store := topoclient.NewWithFactory(factory, "", []string{""}, topoclient.NewDefaultTopoConfig()) + defer func() { _ = store.Close() }() + rpc := rpcclient.NewFakeClient() + registeredReplica(t, store, rpc, shard, "cell1", "pooler-0") + bad := registeredReplica(t, store, rpc, shard, "cell1", "pooler-1") + rpc.Errors[bad] = errors.New("dial: connection refused") + + p0 := postureTestPod() + p1 := postureTestPod() + p1.Name = "pooler-1" + r, _ := postureTestReconciler(t, shard, rpc, p0, p1) + + first, err := r.reconcilePosture(t.Context(), store, shard, rpc) + if err != nil { + t.Fatalf("first reconcilePosture() error = %v", err) + } + if first != postureDebounceRequeueDelay { + t.Errorf("first delay = %v, want the %v debounce", first, postureDebounceRequeueDelay) + } + + for pass := 2; pass <= 4; pass++ { + delay, err := r.reconcilePosture(t.Context(), store, shard, rpc) + if err != nil { + t.Fatalf("pass %d reconcilePosture() error = %v", pass, err) + } + if delay <= 0 { + t.Errorf( + "pass %d: delay = %v, want non-zero (RPC failure still unresolved)", pass, delay, + ) + } + } +} + +// TestReconcilePostureAcceptedMismatchAndNotReadyStillRequeues covers a +// mismatch compounded with a pod that has never reached posture readiness at +// all (this mock pooler never reports IsInitialized/PostgresReady): a +// replica reporting postgres PRIMARY is a role mismatch. After the strike +// threshold it is accepted as PostureConsistent=False, and this must keep +// requesting a requeue. +func TestReconcilePostureAcceptedMismatchAndNotReadyStillRequeues(t *testing.T) { + shard := postureTestShard() + _, factory := memorytopo.NewServerAndFactory(t.Context(), "cell1") + store := topoclient.NewWithFactory(factory, "", []string{""}, topoclient.NewDefaultTopoConfig()) + defer func() { _ = store.Close() }() + rpc := rpcclient.NewFakeClient() + id := &clustermetadata.ID{Cell: "cell1", Name: "pooler-0"} + if err := store.RegisterMultipooler(t.Context(), &clustermetadata.Multipooler{ + Id: id, + Hostname: "pooler-0", + ShardKey: &clustermetadata.ShardKey{ + Database: string(shard.Spec.DatabaseName), + TableGroup: string(shard.Spec.TableGroupName), + Shard: string(shard.Spec.ShardName), + }, + RoutingState: &clustermetadata.RoutingState{ + Role: clustermetadata.RoutingRole_ROUTING_ROLE_REPLICA, + }, + }, false); err != nil { + t.Fatalf("register pooler: %v", err) + } + componentID := topoclient.ComponentIDString(id) + rpc.SetStatusResponse(componentID, &multipoolermanagerdatapb.StatusResponse{ + Status: &multipoolermanagerdatapb.Status{ + PostgresStatus: multipoolermanagerdatapb.PostgresStatus_POSTGRES_STATUS_PRIMARY, + }, + }) + r, _ := postureTestReconciler(t, shard, rpc, postureTestPod()) + + first, err := r.reconcilePosture(t.Context(), store, shard, rpc) + if err != nil { + t.Fatalf("first reconcilePosture() error = %v", err) + } + if first != postureDebounceRequeueDelay { + t.Errorf("first delay = %v, want the %v debounce", first, postureDebounceRequeueDelay) + } + + for pass := 2; pass <= 4; pass++ { + delay, err := r.reconcilePosture(t.Context(), store, shard, rpc) + if err != nil { + t.Fatalf("pass %d reconcilePosture() error = %v", pass, err) + } + if delay <= 0 { + t.Errorf("pass %d: delay = %v, want non-zero (mismatch still unresolved)", pass, delay) + } + if !conditionIsFalse(shard.Status.Conditions, "PostureConsistent") { + t.Errorf("pass %d: conditions = %#v, want PostureConsistent=False once accepted", + pass, shard.Status.Conditions) + } + } +} + +// TestReconcilePostureAcceptedMismatchWithReadyPodsStillRequeues is the other +// half of the accepted-mismatch case: the pod itself is fully ready +// (registeredReplica: initialized, accepting connections, cohort-eligible, +// a committed cohort member), and the only thing wrong is that it reports +// postgres PRIMARY while topology still has it as REPLICA. anyPodNotReady is +// false throughout, so unsettled is the only thing driving this requeue: a +// shard where every pod is ready but multiorch and postgres disagree about +// who is primary must not be left Degraded until the 10h resync once that +// disagreement is accepted into status. +func TestReconcilePostureAcceptedMismatchWithReadyPodsStillRequeues(t *testing.T) { + shard := postureTestShard() + _, factory := memorytopo.NewServerAndFactory(t.Context(), "cell1") + store := topoclient.NewWithFactory(factory, "", []string{""}, topoclient.NewDefaultTopoConfig()) + defer func() { _ = store.Close() }() + rpc := rpcclient.NewFakeClient() + id := registeredReplica(t, store, rpc, shard, "cell1", "pooler-0") + rpc.StatusResponses[id].Response.Status.PostgresStatus = multipoolermanagerdatapb.PostgresStatus_POSTGRES_STATUS_PRIMARY + r, _ := postureTestReconciler(t, shard, rpc, postureTestPod()) + + first, err := r.reconcilePosture(t.Context(), store, shard, rpc) + if err != nil { + t.Fatalf("first reconcilePosture() error = %v", err) + } + if first != postureDebounceRequeueDelay { + t.Errorf("first delay = %v, want the %v debounce", first, postureDebounceRequeueDelay) + } + + for pass := 2; pass <= 4; pass++ { + delay, err := r.reconcilePosture(t.Context(), store, shard, rpc) + if err != nil { + t.Fatalf("pass %d reconcilePosture() error = %v", pass, err) + } + if delay <= 0 { + t.Errorf( + "pass %d: delay = %v, want non-zero (mismatch still unresolved, though the pod is ready)", + pass, + delay, + ) + } + if !conditionIsFalse(shard.Status.Conditions, "PostureConsistent") { + t.Errorf("pass %d: conditions = %#v, want PostureConsistent=False once accepted", + pass, shard.Status.Conditions) + } + } +} + +// TestReconcilePostureRequeuesWhileAPodAwaitsItsPooler covers a shard with one +// settled, registered pooler and one managed pod that never registers: it +// must keep requesting a requeue, and the requested delay must grow with +// elapsed wall-clock time rather than sit at a fixed floor forever. +// +// Drives a fake clock directly so the growth assertion is exact rather than +// "second draw happened to be bigger": elapsed time is measured directly +// regardless of how many reconcile passes it took to get there, so a mutation +// that turns the backoff back into a per-pass count cannot pass by chance. +func TestReconcilePostureRequeuesWhileAPodAwaitsItsPooler(t *testing.T) { + shard := postureTestShard() + _, factory := memorytopo.NewServerAndFactory(t.Context(), "cell1") + store := topoclient.NewWithFactory(factory, "", []string{""}, topoclient.NewDefaultTopoConfig()) + defer func() { _ = store.Close() }() + + rpc := rpcclient.NewFakeClient() + registeredReplica(t, store, rpc, shard, "cell1", "pooler-0") + + settledPod := postureTestPod() + awaitingPod := postureTestPod() + awaitingPod.Name = "pooler-1" + + r, _ := postureTestReconciler(t, shard, rpc, settledPod, awaitingPod) + clk := &fakeClock{t: time.Unix(1_700_000_000, 0)} + r.Clock = clk.now + + for _, tc := range []struct { + advance time.Duration + elapsed time.Duration + }{ + {advance: 0, elapsed: 0}, + {advance: 20 * time.Second, elapsed: 20 * time.Second}, + {advance: 50 * time.Second, elapsed: 70 * time.Second}, + } { + clk.advance(tc.advance) + delay, err := r.reconcilePosture(t.Context(), store, shard, rpc) + if err != nil { + t.Fatalf("reconcilePosture() error = %v", err) + } + min, max := wantDelayRange(tc.elapsed) + if delay < min || delay > max { + t.Errorf("at elapsed=%v: delay = %v, want in [%v, %v]", tc.elapsed, delay, min, max) + } + } +} + +// TestReconcilePostureBacksOffThenClearsOnceAPrimaryIsElected covers the +// bring-up state a minimal cluster passes through before a primary is +// elected: every managed pod has a topology entry (nothing is "awaiting +// registration" in the topology-match sense), and nothing is inconsistent or +// incomplete (Evaluate only compares observed postgres primaries against +// topology roles and finds none of either), but nobody has committed a cohort +// membership because no primary has been elected yet. +// +// Drives a fake clock through several passes to pin the actual delay range, +// checks the PoolerDataReady gate stays False while waiting, then flips the +// fixture to a committed primary and checks the requeue stops, the gate goes +// True, and the not-converged-since entry is gone. +func TestReconcilePostureBacksOffThenClearsOnceAPrimaryIsElected(t *testing.T) { + shard := postureTestShard() + _, factory := memorytopo.NewServerAndFactory(t.Context(), "cell1") + store := topoclient.NewWithFactory(factory, "", []string{""}, topoclient.NewDefaultTopoConfig()) + defer func() { _ = store.Close() }() + + rpc := rpcclient.NewFakeClient() + id0 := notYetSettledReplica(t, store, rpc, shard, "cell1", "pooler-0") + id1 := notYetSettledReplica(t, store, rpc, shard, "cell1", "pooler-1") + + pod0 := postureTestPod() + pod1 := postureTestPod() + pod1.Name = "pooler-1" + + r, c := gateTestReconciler(t, shard, rpc, pod0, pod1) + clk := &fakeClock{t: time.Unix(1_700_000_000, 0)} + r.Clock = clk.now + + for _, tc := range []struct { + advance time.Duration + elapsed time.Duration + }{ + {advance: 0, elapsed: 0}, + {advance: 20 * time.Second, elapsed: 20 * time.Second}, + {advance: 50 * time.Second, elapsed: 70 * time.Second}, + } { + clk.advance(tc.advance) + delay, err := r.reconcilePosture(t.Context(), store, shard, rpc) + if err != nil { + t.Fatalf("reconcilePosture() error = %v", err) + } + min, max := wantDelayRange(tc.elapsed) + if delay < min || delay > max { + t.Errorf("at elapsed=%v: delay = %v, want in [%v, %v]", tc.elapsed, delay, min, max) + } + } + + for _, pod := range []*corev1.Pod{pod0, pod1} { + got := &corev1.Pod{} + if err := c.Get(t.Context(), client.ObjectKeyFromObject(pod), got); err != nil { + t.Fatalf("get pod %s: %v", pod.Name, err) + } + condition := readinessCondition(got.Status.Conditions) + if condition == nil || condition.Status != corev1.ConditionFalse { + t.Errorf("pod %s readiness condition = %#v, want False while waiting for a primary", + pod.Name, condition) + } + } + + // The fixture reaches a settled state: both poolers commit a rule naming + // pooler-0 as leader, so both are cohort-eligible members of the same + // durability rule and pooler-0's postgres reports PRIMARY, matching the + // topology role a leader-designate needs. + if err := store.RegisterMultipooler(t.Context(), &clustermetadata.Multipooler{ + Id: id0, + Hostname: "pooler-0", + ShardKey: &clustermetadata.ShardKey{ + Database: string(shard.Spec.DatabaseName), + TableGroup: string(shard.Spec.TableGroupName), + Shard: string(shard.Spec.ShardName), + }, + RoutingState: &clustermetadata.RoutingState{ + Role: clustermetadata.RoutingRole_ROUTING_ROLE_PRIMARY, + }, + }, true); err != nil { + t.Fatalf("promote pooler-0 in topology: %v", err) + } + rule := &clustermetadata.ShardRule{ + RuleNumber: &clustermetadata.RuleNumber{CoordinatorTerm: 1}, + LeaderId: id0, + CohortMembers: []*clustermetadata.ID{id0, id1}, + DurabilityPolicy: topoclient.AtLeastN(1), + } + for _, elected := range []struct { + id *clustermetadata.ID + primary bool + }{{id0, true}, {id1, false}} { + status := multipoolermanagerdatapb.PostgresStatus_POSTGRES_STATUS_STANDBY + if elected.primary { + status = multipoolermanagerdatapb.PostgresStatus_POSTGRES_STATUS_PRIMARY + } + componentID := topoclient.ComponentIDString(elected.id) + rpc.SetStatusResponse(componentID, &multipoolermanagerdatapb.StatusResponse{ + Status: &multipoolermanagerdatapb.Status{ + IsInitialized: true, + PostgresReady: true, + PostgresStatus: status, + }, + AvailabilityStatus: &clustermetadata.AvailabilityStatus{ + CohortEligibilityStatus: &clustermetadata.CohortEligibilityStatus{ + Signal: clustermetadata.CohortEligibilitySignal_COHORT_ELIGIBILITY_SIGNAL_ELIGIBLE, + }, + }, + ConsensusStatus: &clustermetadata.ConsensusStatus{ + Id: elected.id, + CurrentPosition: &clustermetadata.PoolerPosition{ + Position: &clustermetadata.RulePosition{Decision: rule}, + }, + }, + }) + } + + clk.advance(time.Second) + delay, err := r.reconcilePosture(t.Context(), store, shard, rpc) + if err != nil { + t.Fatalf("reconcilePosture() after election error = %v", err) + } + if delay != 0 { + t.Errorf("delay after a primary is elected = %v, want 0", delay) + } + key := fmt.Sprintf("%s/%s", shard.Namespace, shard.Name) + if _, ok := r.notConvergedSince[key]; ok { + t.Errorf("not-converged-since entry left after a primary is elected") + } + if conditionIsFalse(shard.Status.Conditions, "PostureConsistent") { + t.Errorf( + "conditions = %#v, want no failure once a primary is elected", + shard.Status.Conditions, + ) + } + + for _, pod := range []*corev1.Pod{pod0, pod1} { + got := &corev1.Pod{} + if err := c.Get(t.Context(), client.ObjectKeyFromObject(pod), got); err != nil { + t.Fatalf("get pod %s: %v", pod.Name, err) + } + condition := readinessCondition(got.Status.Conditions) + if condition == nil || condition.Status != corev1.ConditionTrue { + t.Errorf("pod %s readiness condition = %#v, want True once a primary is elected", + pod.Name, condition) + } + } +} diff --git a/pkg/resource-handler/controller/shard/shard_controller.go b/pkg/resource-handler/controller/shard/shard_controller.go index d1229ef4..a36c2a7e 100644 --- a/pkg/resource-handler/controller/shard/shard_controller.go +++ b/pkg/resource-handler/controller/shard/shard_controller.go @@ -79,9 +79,22 @@ type ShardReconciler struct { APIReader client.Reader PoolerClients poolerclient.Resolver CreateTopoStore func(*multigresv1alpha1.Shard) (topoclient.Store, error) + // Clock overrides time.Now for recordNotConverged's elapsed-time math. + // Nil in production; tests inject it to drive the readiness backoff + // deterministically. + Clock func() time.Time postureStrikesMu sync.Mutex postureStrikes map[string]int + + // notConvergedMu and notConvergedSince record, per shard, the time it was + // first observed not converged: unsettled (posture debounce past + // threshold, i.e. an accepted mismatch or Incomplete observation), or some + // managed pod not yet posture-ready. Kept separate from postureStrikes, + // which gates posture.Apply: folding this into that counter would change + // when Apply fires. + notConvergedMu sync.Mutex + notConvergedSince map[string]time.Time } // Reconcile manages pool pods, PVCs, services, and data-plane topology for a Shard. @@ -109,6 +122,7 @@ func (r *ShardReconciler) Reconcile( if err := r.Get(ctx, req.NamespacedName, shard); err != nil { if errors.IsNotFound(err) { logger.Info("Shard resource not found, ignoring") + r.forgetStrikes(req.Namespace, req.Name) return ctrl.Result{}, nil } monitoring.RecordSpanError(span, err) diff --git a/test/suite/scenario_thrash_test.go b/test/suite/scenario_thrash_test.go index 71d14d63..f4718deb 100644 --- a/test/suite/scenario_thrash_test.go +++ b/test/suite/scenario_thrash_test.go @@ -43,10 +43,10 @@ const thrashPoolName = PoolName("default") // changes, then requires the namespace to go quiet and every pool PVC to be // bound to a live pod. // -// It does not assert on this thrashed shard's status.podRoles: whether the -// last scale-up's role lands there is a race against the defect that -// pinPoolScaleUpRoleStale constructs deterministically on a namespace of its -// own, so asserting it here would only sample that race. +// It does not assert on this thrashed shard's status.podRoles itself: +// requirePoolScaleUpLandsRole asserts that claim deterministically on a +// namespace of its own, and asserting it here too would only be a second, +// weaker sample of the same thing. func testPoolReplicaThrash(t *testing.T) { c := newCase(t) cluster := c.poolThrashCluster("pool-thrash", 1) @@ -60,15 +60,19 @@ func testPoolReplicaThrash(t *testing.T) { c.scalePoolTo(cluster, n) } - // MGO-POOL-SCALEUP-ROLE-STALE: once a shard reconciled to Healthy with N - // poolers, raising replicasPerCell to add an (N+1)th sometimes never gets - // that pooler's role into shard.Status.PodRoles, permanently. + // Fixed: scaling a pool from N to N+1 lands the new pooler's role in + // shard.Status.PodRoles even when the pooler registers in topology after + // the shard has already reconciled to Healthy. The shard controller's + // readiness requeue is what closes the gap, since nothing else notices a + // registration: it is a write to the topology store, with no Kubernetes + // event behind it. // - // Whether it bites depends on whether the pooler registers before or - // after the reconcile that declares the shard converged, so sampled - // naturally it reproduces about a third of the time. The pin constructs - // that ordering instead, which makes it deterministic. - pinPoolScaleUpRoleStale(t) + // The pin this replaces was statistical because the defect was: whether it + // bit depended on whether the pooler registered before or after the + // reconcile that declared the shard converged, so it reproduced about a + // third of the time. Fixed, it is deterministic, so a positive assertion + // replaces what used to be a KnownDefect pin. + requirePoolScaleUpLandsRole(t) c.RequireQuiescent(10*time.Second, 90*time.Second) @@ -84,20 +88,19 @@ func testPoolReplicaThrash(t *testing.T) { c.requireNoOrphanedPoolPVCs(live) } -// pinPoolScaleUpRoleStale pins that scaling a pool from one to two does not -// land the new pooler's role in status.podRoles when the pooler registers -// after the shard has already converged. +// requirePoolScaleUpLandsRole asserts that scaling a pool from one to two +// lands the new pooler's role in status.podRoles. // // On its own namespace and its own cluster, with no thrash, because the -// defect never needed one: a bare single scale-up reproduces it, and the -// thrash above only found it first. +// defect this replaces never needed one: a bare single scale-up reproduced +// it, and the thrash above only found it first. // // Nothing wakes the shard once it is Healthy: a registration is a write to -// the topology store, with no Kubernetes event behind it, and the shard does -// not requeue itself while a managed pod is still awaiting its pooler. The fix -// is that requeue; with it the role lands well inside a second here, because -// the suite compresses requeues, so the window below is generous. -func pinPoolScaleUpRoleStale(t *testing.T) { +// the topology store, with no Kubernetes event behind it. The shard +// controller now requeues on a backoff while any managed pod has not reached +// posture readiness, so the role lands well inside a second here, because the +// suite compresses requeues, so the window below is generous. +func requirePoolScaleUpLandsRole(t *testing.T) { t.Helper() c := newCase(t) @@ -162,31 +165,24 @@ func pinPoolScaleUpRoleStale(t *testing.T) { // Now the pooler appears, with no Kubernetes event to announce it: a // registration is a write to etcd. Only a requeue the operator asked for - // itself can notice, which is the thing this pins. + // itself can notice, which is the thing this asserts. release() - // Retire this pin by replacing it with c.Eventually on the same condition. - c.KnownDefect("MGO-POOL-SCALEUP-ROLE-STALE", func() error { - var members Members - deadline := time.Now().Add(20 * time.Second) - for { - var err error - members, err = MembersOf(c.Context(), c.Client(), key) + c.Eventually( + 60*time.Second, + "the scaled-up pooler's role to reach status.podRoles", + func() error { + members, err := MembersOf(c.Context(), c.Client(), key) // A read failure is the check's own setup failing, not the - // defect, so it fails the test rather than keeping the pin green. + // convergence it is waiting for, so it fails the test immediately + // rather than retrying it silently until the timeout. c.NoError(err, "read shard members") - if len(members.Replicas) == 1 && len(members.Quarantined) == 0 { - return nil - } - if time.Now().After(deadline) { - break + if len(members.Replicas) != 1 || len(members.Quarantined) != 0 { + return fmt.Errorf("got %+v", members) } - time.Sleep(250 * time.Millisecond) - } - return fmt.Errorf( - "the scaled-up pooler registered after the shard converged and its role "+ - "never reached status.podRoles within 20s: got %+v", members) - }) + return nil + }, + ) } func (c *C) poolThrashCluster( @@ -243,10 +239,9 @@ func (c *C) scalePoolTo(cluster *MultigresCluster, n int32) { // liveReadyPoolPodNames lists Ready pods belonging to the thrashed pool. // // This deliberately does not go through MembersOf/shard.Status.PodRoles: that -// path is exactly what MGO-POOL-SCALEUP-ROLE-STALE (see -// testPoolReplicaThrash) breaks, and a pod being live is a Kubernetes-level -// fact independent of whether the operator's own role bookkeeping has caught -// up to it. +// path is the one testPoolReplicaThrash's pool-scale-up assertion covers +// directly, and a pod being live is a Kubernetes-level fact independent of +// whether the operator's own role bookkeeping has caught up to it. func (c *C) liveReadyPoolPodNames() []string { c.Helper() pods := &corev1.PodList{}