diff --git a/.github/workflows/build-and-release.yaml b/.github/workflows/build-and-release.yaml index aedf85a0..9ceb2873 100644 --- a/.github/workflows/build-and-release.yaml +++ b/.github/workflows/build-and-release.yaml @@ -50,7 +50,9 @@ jobs: cache: false - name: Run tests - run: go test ./... -coverprofile=./cover.out -covermode=atomic -coverpkg=./... + run: | + go test $(go list ./... | grep -v /test/suite) \ + -coverprofile=./cover.out -covermode=atomic -coverpkg=./... - name: Run observer tests against current operator API working-directory: tools/observer diff --git a/.github/workflows/test-suite.yaml b/.github/workflows/test-suite.yaml new file mode 100644 index 00000000..ce840c22 --- /dev/null +++ b/.github/workflows/test-suite.yaml @@ -0,0 +1,40 @@ +# This job is advisory. It is deliberately not a required check and is not +# referenced by any other workflow's needs. A red run here means the suite +# caught a real problem, not that CI itself is broken: treat a failure as a +# finding to investigate, not as noise to wave off or restart until it goes +# green. +name: Test suite + +on: + pull_request: {} + # main only: pull_request already covers any branch with a PR open, so a + # wider push trigger would run the suite twice for it. main is here to catch + # a bad merge. + push: + branches: + - main + workflow_dispatch: {} + +permissions: + contents: read + +jobs: + test-suite: + runs-on: ubuntu-latest + timeout-minutes: 30 + permissions: + contents: read + steps: + - name: Check out code + uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + + - name: Install Go + uses: actions/setup-go@b7ad1dad31e06c5925ef5d2fc7ad053ef454303e # v7.0.0 + with: + go-version-file: go.mod + cache: false + + - name: Run test suite + run: make test-suite diff --git a/.golangci.toml b/.golangci.toml index cf89d727..ce4a124a 100644 --- a/.golangci.toml +++ b/.golangci.toml @@ -10,11 +10,16 @@ enable = [ "gosec" ] # for a dedicated follow-up migration; excluded here so the dependency bump # that introduced the deprecations isn't blocked on an unrelated, # wide-reaching refactor. +# +# The optional package-path prefix is there because staticcheck names the +# symbol differently across versions: bare ("client.Apply") before +# golangci-lint v2.13, fully qualified ("sigs.k8s.io/.../pkg/client.Apply") +# from it. rules = [ - { linters = [ "staticcheck" ], text = "SA1019: client.Apply is deprecated" }, - { linters = [ "staticcheck" ], text = "SA1019: scheme.Builder is deprecated" }, - { linters = [ "staticcheck" ], text = "SA1019: webhook.CustomDefaulter is deprecated" }, - { linters = [ "staticcheck" ], text = "SA1019: webhook.CustomValidator is deprecated" }, + { linters = [ "staticcheck" ], text = 'SA1019: (\S+/)?client\.Apply is deprecated' }, + { linters = [ "staticcheck" ], text = 'SA1019: (\S+/)?scheme\.Builder is deprecated' }, + { linters = [ "staticcheck" ], text = 'SA1019: (\S+/)?webhook\.CustomDefaulter is deprecated' }, + { linters = [ "staticcheck" ], text = 'SA1019: (\S+/)?webhook\.CustomValidator is deprecated' }, { linters = [ "staticcheck" ], text = "WithCustomDefaulter is deprecated" }, { linters = [ "staticcheck" ], text = "WithCustomValidator is deprecated" }, { linters = [ "staticcheck" ], text = "GetEventRecorderFor is deprecated" }, diff --git a/Dockerfile b/Dockerfile index 3c215ac3..32a24b31 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,6 +1,6 @@ # Containerfile for multigres-operator -FROM --platform=$BUILDPLATFORM golang:1.26.6-alpine3.23 AS builder +FROM --platform=$BUILDPLATFORM golang:1.27.1-alpine3.23 AS builder ARG TARGETOS ARG TARGETARCH diff --git a/Makefile b/Makefile index 235d2b31..2c274622 100644 --- a/Makefile +++ b/Makefile @@ -108,7 +108,7 @@ KUSTOMIZE_VERSION ?= v5.6.0 # renovate: datasource=github-releases depName=kubernetes-sigs/controller-tools CONTROLLER_TOOLS_VERSION ?= v0.18.0 # renovate: datasource=github-releases depName=golangci/golangci-lint -GOLANGCI_LINT_VERSION ?= v2.12.2 +GOLANGCI_LINT_VERSION ?= v2.13.2 CERT_MANAGER_VERSION ?= v1.19.2 @@ -278,22 +278,65 @@ build-installer: manifests generate kustomize ## Generate consolidated install Y ##@ Test +# test/suite is the multi-controller envtest suite. It carries no build tag, so +# every `go test ./...` call site has to exclude it by path or it lands in the +# required check before it is ready. That is one filter per call site, which is +# the deliberate trade against a tag that someone forgets on a new file. +# +# -v is load-bearing rather than cosmetic. Each KnownDefect pin logs the defect +# it is standing on while that defect is still present, and without -v go test +# discards the output of a passing test, so a green CI run shows none of them. +# The suite is meant to be readable as the operator's live defect list, and -v +# is what makes that list visible without waiting for a pin to expire. +.PHONY: test-suite +test-suite: manifests generate fmt vet setup-envtest ## Run the multi-controller test suite + KUBEBUILDER_ASSETS="$(shell $(ENVTEST) use $(ENVTEST_K8S_VERSION) --bin-dir $(LOCALBIN) -p path)" \ + go test -v -p 1 -timeout 20m ./test/suite/... + +# A separate target rather than a flag on the one above. Measured 2026-09-19: +# 183s against a 166s baseline, so about 10% rather than the roughly-double a +# CPU-bound suite would pay. This one spends most of its wall clock waiting for +# controllers to converge, and the race detector does not slow down waiting. +# +# Kept separate anyway, because the cost is not the same everywhere: certificate +# generation is the one CPU-bound step here and has been measured swinging +# between 14 and 75 seconds under -race, which is enough to turn a wait sized +# against the normal run into a flake. A budget that holds on both is looser +# than the default target should carry. +# +# Worth having at all because this suite is the only place five controllers +# share one manager, and the operator holds exactly one piece of state across +# reconcile goroutines: ShardReconciler.postureStrikes, a map guarded by a +# mutex. Nothing here exercises contention on it today, since the suite pins +# MaxConcurrentReconciles to 1 and controller-runtime already serialises +# reconciles per object key, so this is a standing check that the answer has +# not changed rather than a hunt for a known race. +# +# The timeout is generous rather than tight: the instrumented run is only +# slightly slower on average, but its slow tail is much fatter, and a timeout +# that fires on the tail reads as a hang rather than as the flake it is. +.PHONY: test-suite-race +test-suite-race: manifests generate fmt vet setup-envtest ## Run the multi-controller test suite under the race detector + KUBEBUILDER_ASSETS="$(shell $(ENVTEST) use $(ENVTEST_K8S_VERSION) --bin-dir $(LOCALBIN) -p path)" \ + go test -race -v -p 1 -timeout 40m ./test/suite/... + .PHONY: test test: manifests generate fmt vet ## Run tests (no integration testing) KUBEBUILDER_ASSETS="$(shell $(ENVTEST) use $(ENVTEST_K8S_VERSION) --bin-dir $(LOCALBIN) -p path)" \ - go test -p 1 $$(go list ./... | grep -v /e2e) -coverprofile=cover.out + go test -p 1 $$(go list ./... | grep -v /e2e | grep -v /test/suite) -coverprofile=cover.out .PHONY: test-integration test-integration: manifests generate fmt vet setup-envtest ## Run integration tests KUBEBUILDER_ASSETS="$(shell $(ENVTEST) use $(ENVTEST_K8S_VERSION) --bin-dir $(LOCALBIN) -p path)" \ - go test -p 1 -tags=integration,verbose $$(go list ./... | grep -v /e2e) -coverprofile=cover.out + go test -p 1 -tags=integration,verbose $$(go list ./... | grep -v /e2e | grep -v /test/suite) -coverprofile=cover.out .PHONY: test-coverage test-coverage: manifests generate fmt vet setup-envtest ## Generate coverage report with HTML @mkdir -p coverage @echo "==> Generating coverage..." KUBEBUILDER_ASSETS="$(shell $(ENVTEST) use $(ENVTEST_K8S_VERSION) --bin-dir $(LOCALBIN) -p path)" \ - go test -p 1 -tags=integration,verbose ./... -coverprofile=coverage/combined.out -covermode=atomic + go test -p 1 -tags=integration,verbose $$(go list ./... | grep -v /e2e | grep -v /test/suite) \ + -coverprofile=coverage/combined.out -covermode=atomic @echo "==> Generating HTML report..." @go tool cover -html=coverage/combined.out -o=coverage/combined.html @echo "Generated: coverage/combined.html" @@ -679,10 +722,13 @@ $(ENVTEST): $(LOCALBIN) golangci-lint: $(GOLANGCI_LINT) ## Download golangci-lint locally if necessary. # golangci-lint's own go.mod selects an older toolchain than this module # targets, and a linter built with a lower Go version refuses to run. Pin the -# build toolchain to the one resolved by this module's go.mod. +# build toolchain to the one resolved by this module's go.mod, and put that +# version in the binary's name: CI restores bin/ from older caches, and a +# name keyed only on the linter's version would reuse a binary built by the +# previous toolchain after a Go bump. $(GOLANGCI_LINT): export GOTOOLCHAIN = $(shell go env GOVERSION) $(GOLANGCI_LINT): $(LOCALBIN) - $(call go-install-tool,$(GOLANGCI_LINT),github.com/golangci/golangci-lint/v2/cmd/golangci-lint,$(GOLANGCI_LINT_VERSION)) + $(call go-install-tool,$(GOLANGCI_LINT),github.com/golangci/golangci-lint/v2/cmd/golangci-lint,$(GOLANGCI_LINT_VERSION),$(shell go env GOVERSION)) .PHONY: install-certmanager install-certmanager: ## Install Cert-Manager into the cluster @@ -695,16 +741,17 @@ install-certmanager: ## Install Cert-Manager into the cluster # $1 - target path with name of binary # $2 - package url which can be installed # $3 - specific version of package +# $4 - optional extra suffix for the binary's name, e.g. the Go version it was built with define go-install-tool -@[ -f "$(1)-$(3)" ] && [ "$$(readlink -- "$(1)" 2>/dev/null)" = "$(1)-$(3)" ] || { \ +@[ -f "$(1)-$(3)$(if $(4),-$(4))" ] && [ "$$(readlink -- "$(1)" 2>/dev/null)" = "$(1)-$(3)$(if $(4),-$(4))" ] || { \ set -e; \ package=$(2)@$(3) ;\ echo "Downloading $${package}" ;\ rm -f $(1) ;\ GOBIN=$(LOCALBIN) go install $${package} ;\ -mv $(1) $(1)-$(3) ;\ +mv $(1) $(1)-$(3)$(if $(4),-$(4)) ;\ } ;\ -ln -sf $$(realpath $(1)-$(3)) $(1) +ln -sf $$(realpath $(1)-$(3)$(if $(4),-$(4))) $(1) endef ##@ Backward Compatibility Aliases diff --git a/go.mod b/go.mod index afd81bb4..9baf7fbd 100644 --- a/go.mod +++ b/go.mod @@ -1,11 +1,12 @@ module github.com/multigres/multigres-operator -go 1.26.6 +go 1.27.1 require ( github.com/go-logr/logr v1.4.4 github.com/google/go-cmp v0.7.0 github.com/multigres/multigres v0.0.0-20260925193740-522b90425a83 + github.com/multigres/testkit v0.2.1 github.com/prometheus/client_golang v1.24.1 github.com/prometheus/client_model v0.6.3 github.com/stretchr/testify v1.12.1 @@ -15,14 +16,16 @@ require ( go.opentelemetry.io/otel v1.46.0 go.opentelemetry.io/otel/sdk v1.46.0 go.opentelemetry.io/otel/trace v1.46.0 + go.uber.org/goleak v1.3.0 google.golang.org/grpc v1.83.2 google.golang.org/protobuf v1.36.12 k8s.io/api v0.37.0 k8s.io/apimachinery v0.37.0 k8s.io/client-go v0.37.0 - k8s.io/utils v0.0.0-20260626114624-be93311217bd - sigs.k8s.io/controller-runtime v0.25.0 + k8s.io/utils v0.0.0-20260707023825-cf1189d6abe3 + sigs.k8s.io/controller-runtime v0.25.1 sigs.k8s.io/e2e-framework v0.7.0 + sigs.k8s.io/structured-merge-diff/v6 v6.4.2 ) require ( @@ -128,7 +131,6 @@ require ( go.opentelemetry.io/otel/sdk/log v0.22.0 // indirect go.opentelemetry.io/otel/sdk/metric v1.46.0 // indirect go.opentelemetry.io/proto/otlp v1.11.0 // indirect - go.uber.org/goleak v1.3.0 // indirect go.uber.org/multierr v1.11.0 // indirect go.uber.org/zap v1.27.1 // indirect go.yaml.in/yaml/v2 v2.4.4 // indirect @@ -157,7 +159,6 @@ require ( sigs.k8s.io/apiserver-network-proxy/konnectivity-client v0.36.0 // indirect sigs.k8s.io/json v0.0.0-20250730193827-2d320260d730 // indirect sigs.k8s.io/randfill v1.0.0 // indirect - sigs.k8s.io/structured-merge-diff/v6 v6.4.2 // indirect sigs.k8s.io/yaml v1.6.0 // indirect ) diff --git a/go.sum b/go.sum index f263bbdf..ad8e666b 100644 --- a/go.sum +++ b/go.sum @@ -199,6 +199,8 @@ github.com/modern-go/reflect2 v1.0.3-0.20250322232337-35a7c28c31ee h1:W5t00kpgFd github.com/modern-go/reflect2 v1.0.3-0.20250322232337-35a7c28c31ee/go.mod h1:yWuevngMOJpCy52FWWMvUC8ws7m/LJsjYzDa0/r8luk= github.com/multigres/multigres v0.0.0-20260925193740-522b90425a83 h1:IdyFGtwc9pEZEzqs6h5JfavAnErWfuKwj3d42qkZUx8= github.com/multigres/multigres v0.0.0-20260925193740-522b90425a83/go.mod h1:Ov2hrkOguWSkCS2QIhAdguFeG5GlZ3v4WGqIdqkQ7Tg= +github.com/multigres/testkit v0.2.1 h1:1SOV2jevblZBpobzaoFAy/lVqb2Donihjc+EovmbIAo= +github.com/multigres/testkit v0.2.1/go.mod h1:3ONhsV/PNOUke7PID5HPlnxTLyQCcfOQ/JfLCRkSLOY= github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822 h1:C3w9PqII01/Oq1c1nUAm88MOHcQC9l5mIlSMApZMrHA= github.com/munnerz/goautoneg v0.0.0-20191010083416-a7dc8b61c822/go.mod h1:+n7T8mK8HuQTcFwEeznm/DIxMOiR9yIdICNftLE1DvQ= github.com/onsi/ginkgo/v2 v2.27.4 h1:fcEcQW/A++6aZAZQNUmNjvA9PSOzefMJBerHJ4t8v8Y= @@ -457,12 +459,12 @@ k8s.io/kube-openapi v0.0.0-20260721132016-d427ff9ee9ad h1:oXImqH8mQNk7PmvzKhmN3d k8s.io/kube-openapi v0.0.0-20260721132016-d427ff9ee9ad/go.mod h1:0/mqHCVhlumdJ3BhCfnjSZQE037nAhNodh1/hK0T8/I= k8s.io/streaming v0.37.0 h1:iPBUZLZiKt5bV+lxJurASMOV07VuBhNpiwJt2//AWrM= k8s.io/streaming v0.37.0/go.mod h1:APlJR26ZWRcVy5bIEj0QRrKUXROtBHPcxl2NT7EAzPU= -k8s.io/utils v0.0.0-20260626114624-be93311217bd h1:Ea7fgQ5we8Y9T0OX5o0dAHzQOBRI07D/dEYRaB9ZZEs= -k8s.io/utils v0.0.0-20260626114624-be93311217bd/go.mod h1:xDxuJ0whA3d0I4mf/C4ppKHxXynQ+fxnkmQH0vTHnuk= +k8s.io/utils v0.0.0-20260707023825-cf1189d6abe3 h1:jVkFFVfXdXP74B/zbO3hM3hpSFD0xvhQ5U686DPurkE= +k8s.io/utils v0.0.0-20260707023825-cf1189d6abe3/go.mod h1:M2s5JB1lIYP3jzZdorPLHXIPJzt9vv2muW5a6L9DtNM= sigs.k8s.io/apiserver-network-proxy/konnectivity-client v0.36.0 h1:/YpDJ4vReG7ZmzSpBGxduXgywWkJU9zHubgJG03MT+Y= sigs.k8s.io/apiserver-network-proxy/konnectivity-client v0.36.0/go.mod h1:tJo1aepTXyR+8Xs3sUsGBDk4Ub2AM5dPAPKJx0mpm5c= -sigs.k8s.io/controller-runtime v0.25.0 h1:44KgRUPew331KSJpNu8zJow3iTR5W0p/SfrHdw3lV40= -sigs.k8s.io/controller-runtime v0.25.0/go.mod h1:4QqLdT6z/L6Olj8JJCtvztid4/fnIiYsfaTFScegctc= +sigs.k8s.io/controller-runtime v0.25.1 h1:BKgU9OeE8xv8EbbM8cY0NVzTQs35rokkdq1jh12fMb4= +sigs.k8s.io/controller-runtime v0.25.1/go.mod h1:4QqLdT6z/L6Olj8JJCtvztid4/fnIiYsfaTFScegctc= sigs.k8s.io/e2e-framework v0.7.0 h1:AHkySTC6MvnnMbVSxaO4z1m2MhQKNFP+2Ihs5pRNLlM= sigs.k8s.io/e2e-framework v0.7.0/go.mod h1:1ZgXkUSjmnf18/JgHZNEATWjv48O5lJm9aI1QIsRdbw= sigs.k8s.io/json v0.0.0-20250730193827-2d320260d730 h1:IpInykpT6ceI+QxKBbEflcR5EXP7sU1kvOlxwZh5txg= diff --git a/pkg/data-handler/posture/posture.go b/pkg/data-handler/posture/posture.go index 049b8ceb..8d0a83b6 100644 --- a/pkg/data-handler/posture/posture.go +++ b/pkg/data-handler/posture/posture.go @@ -47,6 +47,15 @@ type Readiness struct { Message string } +// reasonAwaitingRegistration is the readiness reason carried by a managed pod +// that has no corresponding pooler in the shard topology. +// +// Every managed pod is seeded with it and only overwritten once a topology +// entry matches. Note this is NOT what Result.Incomplete reports: that covers +// an unreachable cell, a topology entry with no matching pod, or an UNKNOWN +// posture, all of which are the opposite direction. +const reasonAwaitingRegistration = "AwaitingRegistration" + // Evaluate compares each managed pooler's observed postgres state with its // topology role. It returns nil when topology contains no active poolers, as // during bootstrap. @@ -61,7 +70,7 @@ func Evaluate( readiness := make(map[string]Readiness, len(managedPodNames)) for _, podName := range managedPodNames { readiness[podName] = Readiness{ - Reason: "AwaitingRegistration", + Reason: reasonAwaitingRegistration, Message: "pooler has not registered in the shard topology", } } diff --git a/pkg/resource-handler/controller/shard/reconcile_data_plane.go b/pkg/resource-handler/controller/shard/reconcile_data_plane.go index 57ac0301..7bb1bd49 100644 --- a/pkg/resource-handler/controller/shard/reconcile_data_plane.go +++ b/pkg/resource-handler/controller/shard/reconcile_data_plane.go @@ -10,6 +10,7 @@ import ( corev1 "k8s.io/api/core/v1" meta "k8s.io/apimachinery/pkg/api/meta" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/util/wait" ctrl "sigs.k8s.io/controller-runtime" "sigs.k8s.io/controller-runtime/pkg/client" "sigs.k8s.io/controller-runtime/pkg/log" @@ -73,7 +74,12 @@ func (r *ShardReconciler) reconcileDataPlane( // A resolver error breaks the sequence of posture observations. It must // not let a strike from before the transport outage combine with the - // first unsettled observation after recovery. + // first unsettled observation after recovery. notConvergedSince needs no + // matching reset here: this path (like the nil-topology one below) + // returns its own fixed delay independent of that map, so a stale entry + // left over from before the outage only shortens the path to the + // backoff's ceiling once posture resumes, it never suppresses a requeue + // that is actually owed. That is intentional, not a gap. r.recordPostureObservation(shard, false) // A partial observation cannot clear a confirmed posture failure. Keep a @@ -205,8 +211,12 @@ func (r *ShardReconciler) reconcileDataPlane( shard, fmt.Sprintf("Failed to check backup health: %v", err), ) - if patchErr := r.Status(). - Patch(ctx, shard, client.MergeFrom(backupBase)); patchErr != nil { + if patchErr := r.Status().Patch( + ctx, + shard, + client.MergeFrom(backupBase), + client.FieldOwner("multigres-resource-handler"), + ); patchErr != nil { return ctrl.Result{}, fmt.Errorf("update unavailable backup status: %w", patchErr) } } else if result != nil { @@ -222,7 +232,12 @@ func (r *ShardReconciler) reconcileDataPlane( r.Recorder.Event(shard, "Warning", "BackupStale", result.Message) } - if err := r.Status().Patch(ctx, shard, client.MergeFrom(backupBase)); err != nil { + if err := r.Status().Patch( + ctx, + shard, + client.MergeFrom(backupBase), + client.FieldOwner("multigres-resource-handler"), + ); err != nil { monitoring.RecordSpanError(childSpan, err) childSpan.End() logger.Error(err, "Failed to update shard backup status") @@ -270,12 +285,16 @@ func (r *ShardReconciler) reconcilePodRoles( ) { logger := log.FromContext(ctx) - // List managed pods for this shard (same pattern as reconcilePoolerPrune). + // List managed pool pods for this shard (same pattern as + // reconcilePoolerPrune). Scoped to pool pods: only they can ever match a + // topology pooler entry, and a shard's multiorch pod carries the same four + // identity labels but is never one. lbls := map[string]string{ metadata.LabelMultigresCluster: shard.Labels[metadata.LabelMultigresCluster], metadata.LabelMultigresDatabase: string(shard.Spec.DatabaseName), metadata.LabelMultigresTableGroup: string(shard.Spec.TableGroupName), metadata.LabelMultigresShard: string(shard.Spec.ShardName), + metadata.LabelAppComponent: PoolComponentName, } podList := &corev1.PodList{} if err := r.List(ctx, podList, @@ -317,7 +336,12 @@ func (r *ShardReconciler) reconcilePodRoles( } if rolesChanged { - if err := r.Status().Patch(ctx, shard, client.MergeFrom(statusBase)); err != nil { + if err := r.Status().Patch( + ctx, + shard, + client.MergeFrom(statusBase), + client.FieldOwner("multigres-resource-handler"), + ); err != nil { logger.Error(err, "Failed to update shard pod roles") } } @@ -331,6 +355,22 @@ const ( // client cannot be built yet (e.g. operator client cert not issued). poolerClientRetryDelay = 10 * time.Second + // A shard that is not converged, whatever the reason, is requeued on a + // backoff rather than left to the 10h resync. The delay is clamped elapsed + // time since the shard was first observed not-converged, not a per-pass + // count: a burst of unrelated pod/status events must not itself advance + // the backoff. The floor is the multipooler container's own readiness + // probe period, so the operator never polls topology more aggressively + // than the kubelet polls the pod, and the ceiling matches the + // empty-topology path. + readinessBackoffMinDelay = 5 * time.Second + readinessBackoffMaxDelay = time.Minute + // readinessBackoffJitter adds upward jitter. controller-runtime does not + // jitter RequeueAfter, and a fleet that lost convergence together + // (topology restart, mass scale-up) would otherwise retry in lockstep + // against the topology server that just came back. + readinessBackoffJitter = 0.2 + reasonPoolerClientUnavailable = "PoolerClientUnavailable" reasonAwaitingPoolerRegistration = "AwaitingPoolerRegistration" reasonObservationPending = "ObservationPending" @@ -343,11 +383,17 @@ func (r *ShardReconciler) reconcilePosture( shard *multigresv1alpha1.Shard, rpcClient rpcclient.MultipoolerClient, ) (time.Duration, error) { + // Pool pods only: a shard's multiorch pod carries the same four identity + // labels (buildMultiorchLabelsWithCell merges them in) but is never a + // pooler, so an unfiltered list seeds it AwaitingRegistration forever and + // anyPodNotReady never clears. reconcilePoolerReadiness already scopes + // this way; posture.Evaluate must match it. lbls := map[string]string{ metadata.LabelMultigresCluster: shard.Labels[metadata.LabelMultigresCluster], metadata.LabelMultigresDatabase: string(shard.Spec.DatabaseName), metadata.LabelMultigresTableGroup: string(shard.Spec.TableGroupName), metadata.LabelMultigresShard: string(shard.Spec.ShardName), + metadata.LabelAppComponent: PoolComponentName, } podList := &corev1.PodList{} if err := r.List(ctx, podList, @@ -374,6 +420,8 @@ func (r *ShardReconciler) reconcilePosture( // An empty topology is expected during bootstrap, but it is not a settled // posture observation. Keep polling until poolers register rather than // leaving a previous transport condition stuck until the periodic resync. + // This path always returns a fixed 1-minute delay of its own regardless of + // notConvergedSince, so it needs no matching reset there either. r.recordPostureObservation(shard, false) setPostureUnknownUnlessFalse( shard, @@ -436,12 +484,78 @@ func (r *ShardReconciler) reconcilePosture( r.Recorder.Event(shard, "Warning", reason, result.Message) } + // A shard is converged only once posture is settled AND every managed pod + // has reached posture readiness. Neither half implies the other: an + // accepted mismatch or accepted Incomplete observation (strikes past + // threshold) is settled but still has a pod sitting NotReady, and a shard + // mid-bootstrap with every pooler registered but no primary elected yet is + // neither inconsistent nor incomplete but has nothing actually ready. + // Registration itself, and any change multiorch makes in topology, are + // writes to the topology store with no Kubernetes event behind them, so + // nothing but this controller's own requeue will ever look again. Without + // it, a settled-but-not-ready shard sits until controller-runtime's 10h + // resync; a shard with a genuinely stuck pod (Pending, crash-looping, + // quarantined) sits forever, since a lost watch never recovers it. + // + // The debounce above stays authoritative for whether an unsettled + // observation is accepted into status: this never fires while + // `unsettled && strikes < postureStrikeThreshold`, so it cannot preempt or + // shorten that window, only pick up once it ends. + notConverged := unsettled || anyPodNotReady(result.Readiness) + elapsed := r.recordNotConverged(shard, notConverged) + if unsettled && strikes < postureStrikeThreshold { return postureDebounceRequeueDelay, nil } + if notConverged { + return readinessBackoffDelay(elapsed), nil + } return 0, nil } +// anyPodNotReady reports whether any managed pod has not yet reached posture +// readiness. +// +// Deliberately broader than a check for the seeded "AwaitingRegistration" +// reason alone. A pod whose pooler has never registered carries that reason, +// but every other reason poolerReadiness can return (NotInitialized, +// PostgresNotReady, CohortIneligible, NotCohortMember) means Ready is false +// too, and all of them are reachable by a shard that Evaluate reports as +// neither inconsistent nor incomplete, since that result only compares +// postures against topology roles and does not require a primary to exist. A +// shard mid-bootstrap with every pooler registered but no primary elected +// settles there, looking "consistent" while nothing is actually ready. +func anyPodNotReady(readiness map[string]posture.Readiness) bool { + for _, r := range readiness { + if !r.Ready { + return true + } + } + return false +} + +// readinessBackoffDelay clamps elapsed (time since the shard was first +// observed not-converged) to [readinessBackoffMinDelay, +// readinessBackoffMaxDelay], with upward jitter. +// +// Elapsed-time-based rather than a per-pass count: Owns(&corev1.Pod{}) and +// For(&Shard{}) carry no predicates, so a pod's Pending -> ContainerCreating -> +// per-container-Ready transitions, an operator gate patch, and a drain's 2s +// requeues are each their own reconcile. A scale-up alone is at least five of +// those before there is anything to register, and a wall-clock burst of +// events must not by itself run the delay up to the ceiling: two reconciles a +// second apart are five seconds not-converged either way, whether they were +// one pass or five. +func readinessBackoffDelay(elapsed time.Duration) time.Duration { + d := elapsed + if d < readinessBackoffMinDelay { + d = readinessBackoffMinDelay + } else if d > readinessBackoffMaxDelay { + d = readinessBackoffMaxDelay + } + return wait.Jitter(d, readinessBackoffJitter) +} + func setPostureUnknownUnlessFalse( shard *multigresv1alpha1.Shard, reason string, @@ -490,6 +604,18 @@ func withDataPlaneRequeue( return result } +// recordPostureObservation counts consecutive unsettled observations for one +// shard and returns the running total. +// +// A settled observation deletes the entry rather than writing zero. The two +// mean the same thing to every reader, since a missing key reads as zero, but +// they differ in what the map holds: writing zero keeps an entry for every +// shard this process has ever reconciled, while deleting keeps only the +// shards currently accumulating strikes, which in a healthy cluster is none. +// A Go map's table is sized by its peak simultaneous entries and does not +// shrink on delete, so this is what keeps that peak bounded by currently +// unsettled shards rather than by cumulative shards over the operator's +// lifetime. func (r *ShardReconciler) recordPostureObservation( shard *multigresv1alpha1.Shard, unsettled bool, @@ -498,17 +624,76 @@ func (r *ShardReconciler) recordPostureObservation( r.postureStrikesMu.Lock() defer r.postureStrikesMu.Unlock() + if !unsettled { + delete(r.postureStrikes, key) + return 0 + } if r.postureStrikes == nil { r.postureStrikes = make(map[string]int) } - if unsettled { - r.postureStrikes[key]++ - } else { - r.postureStrikes[key] = 0 - } + r.postureStrikes[key]++ return r.postureStrikes[key] } +// recordNotConverged tracks, per shard, the time a shard was first observed +// not converged (unsettled, or some managed pod not posture-ready), and +// returns how long that has been true. Same delete-on-settle pattern as +// recordPostureObservation, and a separate map for the same reason that one +// is not reused here: this one picks the backoff, and posture strikes must +// keep gating only posture.Apply. +// +// Storing the first-seen time rather than a per-pass count is what makes +// readinessBackoffDelay a function of elapsed wall-clock time instead of +// event count; see readinessBackoffDelay's own doc for why that matters. +func (r *ShardReconciler) recordNotConverged( + shard *multigresv1alpha1.Shard, + notConverged bool, +) time.Duration { + key := fmt.Sprintf("%s/%s", shard.Namespace, shard.Name) + now := r.now() + + r.notConvergedMu.Lock() + defer r.notConvergedMu.Unlock() + if !notConverged { + delete(r.notConvergedSince, key) + return 0 + } + if r.notConvergedSince == nil { + r.notConvergedSince = make(map[string]time.Time) + } + first, ok := r.notConvergedSince[key] + if !ok { + first = now + r.notConvergedSince[key] = first + } + return now.Sub(first) +} + +// now returns the reconciler's clock, defaulting to time.Now so production +// code never has to set Clock. Tests inject Clock to drive +// recordNotConverged's elapsed-time math deterministically. +func (r *ShardReconciler) now() time.Time { + if r.Clock != nil { + return r.Clock() + } + return time.Now() +} + +// forgetStrikes drops namespace/name's entry from both strike maps. Called on +// a Shard's deletion and not-found paths so a shard deleted mid-backoff, in +// either counter, does not leak its entry for the life of the process. +func (r *ShardReconciler) forgetStrikes(namespace, name string) { + key := fmt.Sprintf("%s/%s", namespace, name) + + r.postureStrikesMu.Lock() + delete(r.postureStrikes, key) + r.postureStrikesMu.Unlock() + + r.notConvergedMu.Lock() + delete(r.notConvergedSince, key) + r.notConvergedMu.Unlock() +} + // reconcileDrainState iterates pods with drain annotations and runs the // drain state machine for each one. func (r *ShardReconciler) reconcileDrainState( @@ -518,11 +703,17 @@ func (r *ShardReconciler) reconcileDrainState( ) (bool, error) { logger := log.FromContext(ctx) + // Pool pods only: the drain-requested annotation this loop acts on is only + // ever set by the pool scale-down/rolling-update path (reconcile_pool_pods.go), + // never on a shard's multiorch pod, so this is currently a no-op filter. + // Scoped anyway for the same reason reconcilePosture now is: relying on an + // annotation nothing else sets is a coincidence, not a guarantee. lbls := map[string]string{ metadata.LabelMultigresCluster: shard.Labels[metadata.LabelMultigresCluster], metadata.LabelMultigresDatabase: string(shard.Spec.DatabaseName), metadata.LabelMultigresTableGroup: string(shard.Spec.TableGroupName), metadata.LabelMultigresShard: string(shard.Spec.ShardName), + metadata.LabelAppComponent: PoolComponentName, } podList := &corev1.PodList{} if err := r.List( @@ -641,11 +832,17 @@ func (r *ShardReconciler) reconcilePoolerPrune( return } + // Pool pods only: topo.MarkDeadPoolers matches this set's names against + // topology pooler entries, and a multiorch pod's name never matches one + // (it is not a pooler), so including it here is currently a no-op. Scoped + // anyway to keep this list's meaning ("pods that can be poolers") aligned + // with what it is actually used for. lbls := map[string]string{ metadata.LabelMultigresCluster: shard.Labels[metadata.LabelMultigresCluster], metadata.LabelMultigresDatabase: string(shard.Spec.DatabaseName), metadata.LabelMultigresTableGroup: string(shard.Spec.TableGroupName), metadata.LabelMultigresShard: string(shard.Spec.ShardName), + metadata.LabelAppComponent: PoolComponentName, } podList := &corev1.PodList{} if err := r.List( diff --git a/pkg/resource-handler/controller/shard/reconcile_data_plane_posture_internal_test.go b/pkg/resource-handler/controller/shard/reconcile_data_plane_posture_internal_test.go index 28e2a918..cd83c8e0 100644 --- a/pkg/resource-handler/controller/shard/reconcile_data_plane_posture_internal_test.go +++ b/pkg/resource-handler/controller/shard/reconcile_data_plane_posture_internal_test.go @@ -114,6 +114,10 @@ func postureTestReconciler( }, c } +// postureTestPod is a pool pod: reconcilePosture's own pod list is scoped to +// PoolComponentName (a shard's multiorch pod carries the same four identity +// labels but is never a pooler), so a pod fixture without this label is +// invisible to it regardless of what the test otherwise sets up. func postureTestPod() *corev1.Pod { return &corev1.Pod{ObjectMeta: metav1.ObjectMeta{ Name: "pooler-0", @@ -123,6 +127,7 @@ func postureTestPod() *corev1.Pod { metadata.LabelMultigresDatabase: "database", metadata.LabelMultigresTableGroup: "table-group", metadata.LabelMultigresShard: "0", + metadata.LabelAppComponent: PoolComponentName, }, }} } @@ -244,8 +249,14 @@ func TestReconcilePostureDebouncesFirstInconsistency(t *testing.T) { if err != nil { t.Fatalf("second reconcilePosture() error = %v", err) } - if retryAfter != 0 { - t.Error("second inconsistent posture observation requested another debounce requeue") + // The mismatch is now accepted into status (PostureConsistent=False), but + // this mock pooler never reports IsInitialized/PostgresReady, so it has + // also never reached posture readiness. An accepted-but-not-ready shard + // must keep requesting a requeue: nothing but this controller's own + // backoff will ever look again, since neither a status recovery in + // topology nor a role fix changes a Kubernetes object. + if retryAfter <= 0 { + t.Error("second inconsistent posture observation (still not ready) requested no requeue") } if !conditionIsFalse(shard.Status.Conditions, posture.ConditionConsistent) { t.Errorf( @@ -296,8 +307,11 @@ func TestReconcilePostureDebouncesFirstIncompleteObservation(t *testing.T) { if err != nil { t.Fatalf("second reconcilePosture() error = %v", err) } - if retryAfter != 0 { - t.Error("second incomplete posture observation requested another debounce requeue") + // Accepted into status (Unknown/ObservationIncomplete), but a pooler whose + // Status RPC errors has also never reached posture readiness, so this must + // still requeue rather than strand the pod until the 10h resync. + if retryAfter <= 0 { + t.Error("second incomplete posture observation (still not ready) requested no requeue") } for _, condition := range shard.Status.Conditions { if condition.Type != posture.ConditionConsistent { diff --git a/pkg/resource-handler/controller/shard/reconcile_deletion.go b/pkg/resource-handler/controller/shard/reconcile_deletion.go index 4a949499..13dbaa6b 100644 --- a/pkg/resource-handler/controller/shard/reconcile_deletion.go +++ b/pkg/resource-handler/controller/shard/reconcile_deletion.go @@ -130,6 +130,10 @@ func (r *ShardReconciler) handleDeletion( return ctrl.Result{}, err } + // The shard is done reconciling; drop its strike entries so a shard + // deleted mid-backoff does not leak its counters. + r.forgetStrikes(shard.Namespace, shard.Name) + // Remove the finalizer last so Kubernetes can finish deletion now that the // PVC cleanup has run. if slices.Contains(shard.Finalizers, shardFinalizer) { @@ -342,7 +346,12 @@ func (r *ShardReconciler) handlePendingDeletion( ObservedGeneration: shard.Generation, LastTransitionTime: metav1.Now(), }) - if err := r.Status().Patch(ctx, shard, client.MergeFrom(statusBase)); err != nil { + if err := r.Status().Patch( + ctx, + shard, + client.MergeFrom(statusBase), + client.FieldOwner("multigres-resource-handler"), + ); err != nil { return ctrl.Result{}, fmt.Errorf("setting ReadyForDeletion condition: %w", err) } logger.Info("Set ReadyForDeletion condition") diff --git a/pkg/resource-handler/controller/shard/reconcile_readiness.go b/pkg/resource-handler/controller/shard/reconcile_readiness.go index fe6886f2..71234eae 100644 --- a/pkg/resource-handler/controller/shard/reconcile_readiness.go +++ b/pkg/resource-handler/controller/shard/reconcile_readiness.go @@ -71,7 +71,16 @@ func (r *ShardReconciler) reconcilePoolerReadiness( Reason: observation.Reason, Message: observation.Message, }) - if err := r.Status().Patch(ctx, pod, client.MergeFrom(base)); err != nil { + // Named apart from the Shard's own status manager: the claim here is over + // one condition on a Pod whose status otherwise belongs to kubelet, not + // over the Shard's status, and a manager name is the only record of which + // concern took a field. Same reasoning as the storage-class guard. + if err := r.Status().Patch( + ctx, + pod, + client.MergeFrom(base), + client.FieldOwner("multigres-resource-handler-readiness"), + ); err != nil { return fmt.Errorf("patch pooler readiness for pod %s: %w", pod.Name, err) } } diff --git a/pkg/resource-handler/controller/shard/registration_requeue_test.go b/pkg/resource-handler/controller/shard/registration_requeue_test.go new file mode 100644 index 00000000..b9a5ff99 --- /dev/null +++ b/pkg/resource-handler/controller/shard/registration_requeue_test.go @@ -0,0 +1,858 @@ +package shard + +import ( + "errors" + "fmt" + "testing" + "time" + + "github.com/multigres/multigres/go/common/rpcclient" + "github.com/multigres/multigres/go/common/topoclient" + "github.com/multigres/multigres/go/common/topoclient/memorytopo" + "github.com/multigres/multigres/go/pb/clustermetadata" + multipoolermanagerdatapb "github.com/multigres/multigres/go/pb/multipoolermanagerdata" + corev1 "k8s.io/api/core/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/types" + "k8s.io/client-go/tools/record" + ctrl "sigs.k8s.io/controller-runtime" + "sigs.k8s.io/controller-runtime/pkg/client" + "sigs.k8s.io/controller-runtime/pkg/client/fake" + + multigresv1alpha1 "github.com/multigres/multigres-operator/api/v1alpha1" + "github.com/multigres/multigres-operator/pkg/data-handler/poolerclient" + "github.com/multigres/multigres-operator/pkg/util/metadata" +) + +// fakeClock lets a test drive recordNotConverged's elapsed-time math without +// sleeping. Set r.Clock = clk.now. +type fakeClock struct { + t time.Time +} + +func (c *fakeClock) now() time.Time { return c.t } + +func (c *fakeClock) advance(d time.Duration) { c.t = c.t.Add(d) } + +// wantDelayRange mirrors readinessBackoffDelay's own clamp so a change to +// that clamp is caught here rather than only against itself. +func wantDelayRange(elapsed time.Duration) (min, max time.Duration) { + min = elapsed + if min < readinessBackoffMinDelay { + min = readinessBackoffMinDelay + } else if min > readinessBackoffMaxDelay { + min = readinessBackoffMaxDelay + } + max = time.Duration(float64(min) * 1.2) + return min, max +} + +// TestReadinessBackoffDelayClampsElapsedTime pins the clamp. The lower bound +// matters as much as the upper one: a delay that could return zero would +// reinstate the defect this exists to fix, since a zero RequeueAfter means no +// requeue at all rather than an immediate one. +func TestReadinessBackoffDelayClampsElapsedTime(t *testing.T) { + t.Parallel() + + for _, tc := range []struct { + elapsed time.Duration + min, max time.Duration + }{ + {elapsed: 0, min: 5 * time.Second, max: 6 * time.Second}, + {elapsed: 3 * time.Second, min: 5 * time.Second, max: 6 * time.Second}, + {elapsed: 5 * time.Second, min: 5 * time.Second, max: 6 * time.Second}, + {elapsed: 30 * time.Second, min: 30 * time.Second, max: 36 * time.Second}, + {elapsed: 59 * time.Second, min: 59 * time.Second, max: 70800 * time.Millisecond}, + {elapsed: time.Minute, min: time.Minute, max: 72 * time.Second}, + {elapsed: 70 * time.Second, min: time.Minute, max: 72 * time.Second}, + {elapsed: time.Hour, min: time.Minute, max: 72 * time.Second}, + } { + // Jitter is random, so the bound has to hold across repeats rather + // than on one draw. + for range 200 { + got := readinessBackoffDelay(tc.elapsed) + if got < tc.min || got > tc.max { + t.Fatalf("elapsed=%v: delay %v outside [%v, %v]", + tc.elapsed, got, tc.min, tc.max) + } + } + } +} + +// TestReadinessBackoffDelayIsNeverZero is the one that dies if the clamp is +// removed. A zero duration is not "retry immediately", it is "do not +// requeue", which is exactly how a shard ends up stranded not-converged. +func TestReadinessBackoffDelayIsNeverZero(t *testing.T) { + t.Parallel() + + for _, elapsed := range []time.Duration{ + -5 * time.Second, -1, 0, time.Second, 5 * time.Second, + 30 * time.Second, time.Minute, time.Hour, + } { + for range 50 { + if got := readinessBackoffDelay(elapsed); got <= 0 { + t.Fatalf("elapsed=%v produced a non-positive delay %v, "+ + "which controller-runtime reads as no requeue", elapsed, got) + } + } + } +} + +// TestReadinessBackoffDelayJitters guards the fleet-lockstep property: a +// constant delay would have every shard that lost convergence together retry +// in lockstep against the topology server that just came back. +func TestReadinessBackoffDelayJitters(t *testing.T) { + t.Parallel() + + seen := map[time.Duration]bool{} + for range 200 { + seen[readinessBackoffDelay(30*time.Second)] = true + } + if len(seen) < 10 { + t.Fatalf("only %d distinct delays across 200 draws; jitter is not applied", len(seen)) + } +} + +// shardNamed is the minimum a strike counter reads. +func shardNamed(ns, name string) *multigresv1alpha1.Shard { + return &multigresv1alpha1.Shard{ + ObjectMeta: metav1.ObjectMeta{Namespace: ns, Name: name}, + } +} + +// TestPostureStrikesLeaveNoEntryOnceSettled is what deleting on settle +// actually buys: a map's table is sized by its peak simultaneous entries and +// does not shrink on delete, so writing zero instead kept an entry for every +// shard the process had ever reconciled. +func TestPostureStrikesLeaveNoEntryOnceSettled(t *testing.T) { + t.Parallel() + + r := &ShardReconciler{} + for i := range 1000 { + s := shardNamed("ns", fmt.Sprintf("shard-%d", i)) + r.recordPostureObservation(s, true) + r.recordPostureObservation(s, false) + } + + if got := len(r.postureStrikes); got != 0 { + t.Fatalf("a thousand shards seen and settled left %d entries, want 0", got) + } +} + +// TestPostureStrikesDoNotSurviveRecreation pins the intended semantic: a +// recreated Shard (a different object at the same namespace/name) opens at +// strike 1, not at whatever count its predecessor left. Posture strikes gate +// posture.Apply (when an unsettled observation is accepted into status); they +// do not pick a requeue delay, that is notConvergedSince's job. +func TestPostureStrikesDoNotSurviveRecreation(t *testing.T) { + t.Parallel() + + r := &ShardReconciler{} + s := shardNamed("ns", "shard-0") + + for range 5 { + r.recordPostureObservation(s, true) + } + if got := r.recordPostureObservation(s, false); got != 0 { + t.Fatalf("a settled observation reported %d strikes, want 0", got) + } + + // The replacement is a different object at the same key, which is what + // the tablegroup controller creates after a Shard is deleted. + if got := r.recordPostureObservation(shardNamed("ns", "shard-0"), true); got != 1 { + t.Fatalf("a recreated shard opened at %d strikes, want 1", got) + } +} + +// TestPostureStrikesCountConsecutiveUnsettled pins what the counter is for, +// so settling on delete cannot be "fixed" into never counting at all. +func TestPostureStrikesCountConsecutiveUnsettled(t *testing.T) { + t.Parallel() + + r := &ShardReconciler{} + s := shardNamed("ns", "shard-0") + for want := 1; want <= 3; want++ { + if got := r.recordPostureObservation(s, true); got != want { + t.Fatalf("consecutive unsettled observation %d reported %d strikes", want, got) + } + } + // Shards are counted independently, which is the only reason the map has + // keys at all. + if got := r.recordPostureObservation(shardNamed("ns", "other"), true); got != 1 { + t.Fatalf("a second shard opened at %d strikes, want 1", got) + } + if got := r.recordPostureObservation(s, true); got != 4 { + t.Fatalf("the first shard reported %d strikes after a second shard, want 4", got) + } +} + +// TestNotConvergedSinceLeavesNoEntryOnceSettled mirrors +// TestPostureStrikesLeaveNoEntryOnceSettled for the not-converged-since map: a +// shard deleted mid-backoff must not leak its entry, and neither must one +// that simply converges. +func TestNotConvergedSinceLeavesNoEntryOnceSettled(t *testing.T) { + t.Parallel() + + r := &ShardReconciler{} + for i := range 1000 { + s := shardNamed("ns", fmt.Sprintf("shard-%d", i)) + r.recordNotConverged(s, true) + r.recordNotConverged(s, false) + } + + if got := len(r.notConvergedSince); got != 0 { + t.Fatalf("a thousand shards seen and settled left %d entries, want 0", got) + } +} + +// TestNotConvergedSinceTracksElapsedTime pins the elapsed-time semantics: the +// first not-converged observation opens the clock, later ones read the time +// since then rather than a per-call count, and a settled observation clears +// it so a later not-converged spell starts over rather than resuming. +func TestNotConvergedSinceTracksElapsedTime(t *testing.T) { + t.Parallel() + + clk := &fakeClock{t: time.Unix(1_700_000_000, 0)} + r := &ShardReconciler{Clock: clk.now} + s := shardNamed("ns", "shard-0") + + if got := r.recordNotConverged(s, true); got != 0 { + t.Fatalf("first not-converged observation reported elapsed %v, want 0", got) + } + clk.advance(37 * time.Second) + if got := r.recordNotConverged(s, true); got != 37*time.Second { + t.Fatalf("second not-converged observation reported elapsed %v, want 37s", got) + } + // A burst of same-instant calls (a wave of unrelated pod events) must not + // itself advance the elapsed time. + if got := r.recordNotConverged(s, true); got != 37*time.Second { + t.Fatalf( + "third not-converged observation (no time passed) reported elapsed %v, want 37s", got, + ) + } + + if got := r.recordNotConverged(s, false); got != 0 { + t.Fatalf("a settled observation reported elapsed %v, want 0", got) + } + if got := r.recordNotConverged(s, true); got != 0 { + t.Fatalf("a fresh not-converged spell reported elapsed %v, want 0 (not resumed)", got) + } +} + +// TestForgetStrikesDropsBothCounters pins that a Shard's strike entries do +// not survive forgetStrikes, in either counter. +func TestForgetStrikesDropsBothCounters(t *testing.T) { + t.Parallel() + + r := &ShardReconciler{} + s := shardNamed("ns", "shard-0") + r.recordPostureObservation(s, true) + r.recordNotConverged(s, true) + + r.forgetStrikes(s.Namespace, s.Name) + + if got := len(r.postureStrikes); got != 0 { + t.Fatalf("posture strikes: %d entries survived forgetStrikes, want 0", got) + } + if got := len(r.notConvergedSince); got != 0 { + t.Fatalf("not-converged-since: %d entries survived forgetStrikes, want 0", got) + } +} + +// TestHandleDeletionForgetsStrikes drives the deletion cleanup through the +// real controller path (handleDeletion) rather than calling forgetStrikes +// directly, so a regression that stops handleDeletion from reaching it is +// caught here rather than only in the helper's own unit test. +func TestHandleDeletionForgetsStrikes(t *testing.T) { + shard := postureTestShard() + shard.Finalizers = []string{shardFinalizer} + now := metav1.Now() + shard.DeletionTimestamp = &now + + scheme := postureTestScheme(t) + c := fake.NewClientBuilder(). + WithScheme(scheme). + WithObjects(shard). + WithStatusSubresource(&multigresv1alpha1.Shard{}). + Build() + r := &ShardReconciler{ + Client: c, + Scheme: scheme, + Recorder: record.NewFakeRecorder(20), + } + + key := fmt.Sprintf("%s/%s", shard.Namespace, shard.Name) + r.postureStrikes = map[string]int{key: 2} + r.notConvergedSince = map[string]time.Time{key: time.Now()} + + if _, err := r.handleDeletion(t.Context(), shard); err != nil { + t.Fatalf("handleDeletion() error = %v", err) + } + + if _, ok := r.postureStrikes[key]; ok { + t.Errorf("posture strikes entry for %s survived handleDeletion", key) + } + if _, ok := r.notConvergedSince[key]; ok { + t.Errorf("not-converged-since entry for %s survived handleDeletion", key) + } +} + +// TestReconcileForgetsStrikesOnNotFound drives the not-found cleanup through +// Reconcile itself: a Shard already gone from the API server (the common case +// once handleDeletion above has already run and removed the finalizer) must +// still have its strike entries dropped, as a backstop. +func TestReconcileForgetsStrikesOnNotFound(t *testing.T) { + scheme := postureTestScheme(t) + c := fake.NewClientBuilder().WithScheme(scheme).Build() + r := &ShardReconciler{ + Client: c, + Scheme: scheme, + Recorder: record.NewFakeRecorder(20), + } + + key := "default/gone-shard" + r.postureStrikes = map[string]int{key: 3} + r.notConvergedSince = map[string]time.Time{key: time.Now()} + + req := ctrl.Request{ + NamespacedName: types.NamespacedName{Namespace: "default", Name: "gone-shard"}, + } + if _, err := r.Reconcile(t.Context(), req); err != nil { + t.Fatalf("Reconcile() error = %v", err) + } + + if _, ok := r.postureStrikes[key]; ok { + t.Errorf("posture strikes entry for %s survived Reconcile on a missing Shard", key) + } + if _, ok := r.notConvergedSince[key]; ok { + t.Errorf("not-converged-since entry for %s survived Reconcile on a missing Shard", key) + } +} + +// gateTestReconciler is postureTestReconciler plus a Pod status subresource, +// for tests that read back the PoolerDataReady gate condition a real +// apiserver would only apply through Status().Patch. +func gateTestReconciler( + t *testing.T, + shard *multigresv1alpha1.Shard, + rpc rpcclient.MultipoolerClient, + objects ...client.Object, +) (*ShardReconciler, client.Client) { + t.Helper() + scheme := postureTestScheme(t) + allObjects := append([]client.Object{shard}, objects...) + c := fake.NewClientBuilder(). + WithScheme(scheme). + WithObjects(allObjects...). + WithStatusSubresource(&multigresv1alpha1.Shard{}, &corev1.Pod{}). + Build() + return &ShardReconciler{ + Client: c, + Scheme: scheme, + Recorder: record.NewFakeRecorder(20), + PoolerClients: poolerclient.Static(rpc), + CreateTopoStore: newMemoryTopoFactory(), + }, c +} + +// registeredReplica registers id in store as a non-primary member of shard, +// with an RPC status that reads Ready via poolerReadiness: initialized, +// accepting connections, cohort-eligible, and a committed member of its own +// single-member rule. Standing in for "this pooler has finished bootstrapping +// and has somewhere to belong," independent of whether anyone in the shard has +// been elected primary. +func registeredReplica( + t *testing.T, + store topoclient.Store, + rpc *rpcclient.FakeClient, + shard *multigresv1alpha1.Shard, + cell, name string, +) topoclient.ComponentID { + t.Helper() + id := &clustermetadata.ID{Cell: cell, Name: name} + if err := store.RegisterMultipooler(t.Context(), &clustermetadata.Multipooler{ + Id: id, + Hostname: name, + ShardKey: &clustermetadata.ShardKey{ + Database: string(shard.Spec.DatabaseName), + TableGroup: string(shard.Spec.TableGroupName), + Shard: string(shard.Spec.ShardName), + }, + RoutingState: &clustermetadata.RoutingState{ + Role: clustermetadata.RoutingRole_ROUTING_ROLE_REPLICA, + }, + }, false); err != nil { + t.Fatalf("register pooler %s: %v", name, err) + } + + componentID := topoclient.ComponentIDString(id) + rpc.SetStatusResponse(componentID, readyStatusResponse(id)) + return componentID +} + +// readyStatusResponse is a fully posture-ready StatusResponse for id: a +// replica, initialized, accepting connections, cohort-eligible, and a +// committed member of its own single-member rule. +func readyStatusResponse(id *clustermetadata.ID) *multipoolermanagerdatapb.StatusResponse { + return &multipoolermanagerdatapb.StatusResponse{ + Status: &multipoolermanagerdatapb.Status{ + IsInitialized: true, + PostgresReady: true, + PostgresStatus: multipoolermanagerdatapb.PostgresStatus_POSTGRES_STATUS_STANDBY, + }, + AvailabilityStatus: &clustermetadata.AvailabilityStatus{ + CohortEligibilityStatus: &clustermetadata.CohortEligibilityStatus{ + Signal: clustermetadata.CohortEligibilitySignal_COHORT_ELIGIBILITY_SIGNAL_ELIGIBLE, + }, + }, + ConsensusStatus: &clustermetadata.ConsensusStatus{ + Id: id, + CurrentPosition: &clustermetadata.PoolerPosition{ + Position: &clustermetadata.RulePosition{ + Decision: &clustermetadata.ShardRule{ + RuleNumber: &clustermetadata.RuleNumber{CoordinatorTerm: 1}, + LeaderId: id, + CohortMembers: []*clustermetadata.ID{id}, + DurabilityPolicy: topoclient.AtLeastN(1), + }, + }, + }, + }, + } +} + +// notYetSettledReplica registers id in store as a non-primary member of +// shard, with an RPC status that is initialized, accepting connections, and +// cohort-eligible, but has not yet committed a rule naming any cohort +// members. Standing in for a shard mid-bootstrap where every pooler has +// registered but multiorch has not yet elected a primary or committed a +// durability rule: nobody is a "postgres primary" (so nothing looks +// inconsistent) and nobody has an unmatched topology entry or an unreadable +// RPC (so nothing looks incomplete), but nobody is ready either. +func notYetSettledReplica( + t *testing.T, + store topoclient.Store, + rpc *rpcclient.FakeClient, + shard *multigresv1alpha1.Shard, + cell, name string, +) *clustermetadata.ID { + t.Helper() + id := &clustermetadata.ID{Cell: cell, Name: name} + if err := store.RegisterMultipooler(t.Context(), &clustermetadata.Multipooler{ + Id: id, + Hostname: name, + ShardKey: &clustermetadata.ShardKey{ + Database: string(shard.Spec.DatabaseName), + TableGroup: string(shard.Spec.TableGroupName), + Shard: string(shard.Spec.ShardName), + }, + RoutingState: &clustermetadata.RoutingState{ + Role: clustermetadata.RoutingRole_ROUTING_ROLE_REPLICA, + }, + }, false); err != nil { + t.Fatalf("register pooler %s: %v", name, err) + } + + componentID := topoclient.ComponentIDString(id) + rpc.SetStatusResponse(componentID, &multipoolermanagerdatapb.StatusResponse{ + Status: &multipoolermanagerdatapb.Status{ + IsInitialized: true, + PostgresReady: true, + PostgresStatus: multipoolermanagerdatapb.PostgresStatus_POSTGRES_STATUS_STANDBY, + }, + AvailabilityStatus: &clustermetadata.AvailabilityStatus{ + CohortEligibilityStatus: &clustermetadata.CohortEligibilityStatus{ + Signal: clustermetadata.CohortEligibilitySignal_COHORT_ELIGIBILITY_SIGNAL_ELIGIBLE, + }, + }, + // No ConsensusStatus: nothing has been committed yet, so + // committedCohortContains is false for everyone. + }) + return id +} + +// TestReconcilePostureConvergedShardReturnsZero pins the hot-loop fix's own +// invariant: a converged shard returns 0 and leaves no map entry. It includes +// a multiorch pod built from BuildMultiorchDeployment, since a shard's +// multiorch pod carries the same four identity labels as its pool pods, so a +// selector that forgets to scope to pool pods seeds it AwaitingRegistration +// forever and this shard never returns 0. +func TestReconcilePostureConvergedShardReturnsZero(t *testing.T) { + shard := postureTestShard() + shard.Labels[metadata.LabelMultigresDatabase] = "database" + shard.Labels[metadata.LabelMultigresTableGroup] = "table-group" + shard.Labels[metadata.LabelMultigresShard] = "0" + + dep, err := BuildMultiorchDeployment(shard, "cell1", postureTestScheme(t)) + if err != nil { + t.Fatalf("BuildMultiorchDeployment() error = %v", err) + } + orch := &corev1.Pod{ObjectMeta: metav1.ObjectMeta{ + Name: "multiorch-abc", Namespace: shard.Namespace, Labels: dep.Spec.Template.Labels, + }} + + _, factory := memorytopo.NewServerAndFactory(t.Context(), "cell1") + store := topoclient.NewWithFactory(factory, "", []string{""}, topoclient.NewDefaultTopoConfig()) + defer func() { _ = store.Close() }() + rpc := rpcclient.NewFakeClient() + registeredReplica(t, store, rpc, shard, "cell1", "pooler-0") + + pool := postureTestPod() + r, _ := postureTestReconciler(t, shard, rpc, pool, orch) + + delay, err := r.reconcilePosture(t.Context(), store, shard, rpc) + if err != nil { + t.Fatalf("reconcilePosture() error = %v", err) + } + if delay != 0 { + t.Errorf("delay = %v, want 0 for a converged shard", delay) + } + key := fmt.Sprintf("%s/%s", shard.Namespace, shard.Name) + if _, ok := r.postureStrikes[key]; ok { + t.Errorf("posture strikes entry left for a converged shard") + } + if _, ok := r.notConvergedSince[key]; ok { + t.Errorf("not-converged-since entry left for a converged shard") + } + if conditionIsFalse(shard.Status.Conditions, "PostureConsistent") { + t.Errorf("conditions = %#v, want no failure for a converged shard", shard.Status.Conditions) + } +} + +// TestReconcilePostureAcceptedIncompleteObservationStillRequeues covers an +// accepted-but-not-ready observation: a Status RPC failure makes a pod +// UNKNOWN, which makes it Incomplete, which makes the shard unsettled. After +// the strike threshold the observation is accepted into status, but the +// failing pod has also never reached posture readiness, so this must keep +// requesting a requeue rather than stranding it until the 10h resync. +func TestReconcilePostureAcceptedIncompleteObservationStillRequeues(t *testing.T) { + shard := postureTestShard() + _, factory := memorytopo.NewServerAndFactory(t.Context(), "cell1") + store := topoclient.NewWithFactory(factory, "", []string{""}, topoclient.NewDefaultTopoConfig()) + defer func() { _ = store.Close() }() + rpc := rpcclient.NewFakeClient() + registeredReplica(t, store, rpc, shard, "cell1", "pooler-0") + bad := registeredReplica(t, store, rpc, shard, "cell1", "pooler-1") + rpc.Errors[bad] = errors.New("dial: connection refused") + + p0 := postureTestPod() + p1 := postureTestPod() + p1.Name = "pooler-1" + r, _ := postureTestReconciler(t, shard, rpc, p0, p1) + + first, err := r.reconcilePosture(t.Context(), store, shard, rpc) + if err != nil { + t.Fatalf("first reconcilePosture() error = %v", err) + } + if first != postureDebounceRequeueDelay { + t.Errorf("first delay = %v, want the %v debounce", first, postureDebounceRequeueDelay) + } + + for pass := 2; pass <= 4; pass++ { + delay, err := r.reconcilePosture(t.Context(), store, shard, rpc) + if err != nil { + t.Fatalf("pass %d reconcilePosture() error = %v", pass, err) + } + if delay <= 0 { + t.Errorf( + "pass %d: delay = %v, want non-zero (RPC failure still unresolved)", pass, delay, + ) + } + } +} + +// TestReconcilePostureAcceptedMismatchAndNotReadyStillRequeues covers a +// mismatch compounded with a pod that has never reached posture readiness at +// all (this mock pooler never reports IsInitialized/PostgresReady): a +// replica reporting postgres PRIMARY is a role mismatch. After the strike +// threshold it is accepted as PostureConsistent=False, and this must keep +// requesting a requeue. +func TestReconcilePostureAcceptedMismatchAndNotReadyStillRequeues(t *testing.T) { + shard := postureTestShard() + _, factory := memorytopo.NewServerAndFactory(t.Context(), "cell1") + store := topoclient.NewWithFactory(factory, "", []string{""}, topoclient.NewDefaultTopoConfig()) + defer func() { _ = store.Close() }() + rpc := rpcclient.NewFakeClient() + id := &clustermetadata.ID{Cell: "cell1", Name: "pooler-0"} + if err := store.RegisterMultipooler(t.Context(), &clustermetadata.Multipooler{ + Id: id, + Hostname: "pooler-0", + ShardKey: &clustermetadata.ShardKey{ + Database: string(shard.Spec.DatabaseName), + TableGroup: string(shard.Spec.TableGroupName), + Shard: string(shard.Spec.ShardName), + }, + RoutingState: &clustermetadata.RoutingState{ + Role: clustermetadata.RoutingRole_ROUTING_ROLE_REPLICA, + }, + }, false); err != nil { + t.Fatalf("register pooler: %v", err) + } + componentID := topoclient.ComponentIDString(id) + rpc.SetStatusResponse(componentID, &multipoolermanagerdatapb.StatusResponse{ + Status: &multipoolermanagerdatapb.Status{ + PostgresStatus: multipoolermanagerdatapb.PostgresStatus_POSTGRES_STATUS_PRIMARY, + }, + }) + r, _ := postureTestReconciler(t, shard, rpc, postureTestPod()) + + first, err := r.reconcilePosture(t.Context(), store, shard, rpc) + if err != nil { + t.Fatalf("first reconcilePosture() error = %v", err) + } + if first != postureDebounceRequeueDelay { + t.Errorf("first delay = %v, want the %v debounce", first, postureDebounceRequeueDelay) + } + + for pass := 2; pass <= 4; pass++ { + delay, err := r.reconcilePosture(t.Context(), store, shard, rpc) + if err != nil { + t.Fatalf("pass %d reconcilePosture() error = %v", pass, err) + } + if delay <= 0 { + t.Errorf("pass %d: delay = %v, want non-zero (mismatch still unresolved)", pass, delay) + } + if !conditionIsFalse(shard.Status.Conditions, "PostureConsistent") { + t.Errorf("pass %d: conditions = %#v, want PostureConsistent=False once accepted", + pass, shard.Status.Conditions) + } + } +} + +// TestReconcilePostureAcceptedMismatchWithReadyPodsStillRequeues is the other +// half of the accepted-mismatch case: the pod itself is fully ready +// (registeredReplica: initialized, accepting connections, cohort-eligible, +// a committed cohort member), and the only thing wrong is that it reports +// postgres PRIMARY while topology still has it as REPLICA. anyPodNotReady is +// false throughout, so unsettled is the only thing driving this requeue: a +// shard where every pod is ready but multiorch and postgres disagree about +// who is primary must not be left Degraded until the 10h resync once that +// disagreement is accepted into status. +func TestReconcilePostureAcceptedMismatchWithReadyPodsStillRequeues(t *testing.T) { + shard := postureTestShard() + _, factory := memorytopo.NewServerAndFactory(t.Context(), "cell1") + store := topoclient.NewWithFactory(factory, "", []string{""}, topoclient.NewDefaultTopoConfig()) + defer func() { _ = store.Close() }() + rpc := rpcclient.NewFakeClient() + id := registeredReplica(t, store, rpc, shard, "cell1", "pooler-0") + rpc.StatusResponses[id].Response.Status.PostgresStatus = multipoolermanagerdatapb.PostgresStatus_POSTGRES_STATUS_PRIMARY + r, _ := postureTestReconciler(t, shard, rpc, postureTestPod()) + + first, err := r.reconcilePosture(t.Context(), store, shard, rpc) + if err != nil { + t.Fatalf("first reconcilePosture() error = %v", err) + } + if first != postureDebounceRequeueDelay { + t.Errorf("first delay = %v, want the %v debounce", first, postureDebounceRequeueDelay) + } + + for pass := 2; pass <= 4; pass++ { + delay, err := r.reconcilePosture(t.Context(), store, shard, rpc) + if err != nil { + t.Fatalf("pass %d reconcilePosture() error = %v", pass, err) + } + if delay <= 0 { + t.Errorf( + "pass %d: delay = %v, want non-zero (mismatch still unresolved, though the pod is ready)", + pass, + delay, + ) + } + if !conditionIsFalse(shard.Status.Conditions, "PostureConsistent") { + t.Errorf("pass %d: conditions = %#v, want PostureConsistent=False once accepted", + pass, shard.Status.Conditions) + } + } +} + +// TestReconcilePostureRequeuesWhileAPodAwaitsItsPooler covers a shard with one +// settled, registered pooler and one managed pod that never registers: it +// must keep requesting a requeue, and the requested delay must grow with +// elapsed wall-clock time rather than sit at a fixed floor forever. +// +// Drives a fake clock directly so the growth assertion is exact rather than +// "second draw happened to be bigger": elapsed time is measured directly +// regardless of how many reconcile passes it took to get there, so a mutation +// that turns the backoff back into a per-pass count cannot pass by chance. +func TestReconcilePostureRequeuesWhileAPodAwaitsItsPooler(t *testing.T) { + shard := postureTestShard() + _, factory := memorytopo.NewServerAndFactory(t.Context(), "cell1") + store := topoclient.NewWithFactory(factory, "", []string{""}, topoclient.NewDefaultTopoConfig()) + defer func() { _ = store.Close() }() + + rpc := rpcclient.NewFakeClient() + registeredReplica(t, store, rpc, shard, "cell1", "pooler-0") + + settledPod := postureTestPod() + awaitingPod := postureTestPod() + awaitingPod.Name = "pooler-1" + + r, _ := postureTestReconciler(t, shard, rpc, settledPod, awaitingPod) + clk := &fakeClock{t: time.Unix(1_700_000_000, 0)} + r.Clock = clk.now + + for _, tc := range []struct { + advance time.Duration + elapsed time.Duration + }{ + {advance: 0, elapsed: 0}, + {advance: 20 * time.Second, elapsed: 20 * time.Second}, + {advance: 50 * time.Second, elapsed: 70 * time.Second}, + } { + clk.advance(tc.advance) + delay, err := r.reconcilePosture(t.Context(), store, shard, rpc) + if err != nil { + t.Fatalf("reconcilePosture() error = %v", err) + } + min, max := wantDelayRange(tc.elapsed) + if delay < min || delay > max { + t.Errorf("at elapsed=%v: delay = %v, want in [%v, %v]", tc.elapsed, delay, min, max) + } + } +} + +// TestReconcilePostureBacksOffThenClearsOnceAPrimaryIsElected covers the +// bring-up state a minimal cluster passes through before a primary is +// elected: every managed pod has a topology entry (nothing is "awaiting +// registration" in the topology-match sense), and nothing is inconsistent or +// incomplete (Evaluate only compares observed postgres primaries against +// topology roles and finds none of either), but nobody has committed a cohort +// membership because no primary has been elected yet. +// +// Drives a fake clock through several passes to pin the actual delay range, +// checks the PoolerDataReady gate stays False while waiting, then flips the +// fixture to a committed primary and checks the requeue stops, the gate goes +// True, and the not-converged-since entry is gone. +func TestReconcilePostureBacksOffThenClearsOnceAPrimaryIsElected(t *testing.T) { + shard := postureTestShard() + _, factory := memorytopo.NewServerAndFactory(t.Context(), "cell1") + store := topoclient.NewWithFactory(factory, "", []string{""}, topoclient.NewDefaultTopoConfig()) + defer func() { _ = store.Close() }() + + rpc := rpcclient.NewFakeClient() + id0 := notYetSettledReplica(t, store, rpc, shard, "cell1", "pooler-0") + id1 := notYetSettledReplica(t, store, rpc, shard, "cell1", "pooler-1") + + pod0 := postureTestPod() + pod1 := postureTestPod() + pod1.Name = "pooler-1" + + r, c := gateTestReconciler(t, shard, rpc, pod0, pod1) + clk := &fakeClock{t: time.Unix(1_700_000_000, 0)} + r.Clock = clk.now + + for _, tc := range []struct { + advance time.Duration + elapsed time.Duration + }{ + {advance: 0, elapsed: 0}, + {advance: 20 * time.Second, elapsed: 20 * time.Second}, + {advance: 50 * time.Second, elapsed: 70 * time.Second}, + } { + clk.advance(tc.advance) + delay, err := r.reconcilePosture(t.Context(), store, shard, rpc) + if err != nil { + t.Fatalf("reconcilePosture() error = %v", err) + } + min, max := wantDelayRange(tc.elapsed) + if delay < min || delay > max { + t.Errorf("at elapsed=%v: delay = %v, want in [%v, %v]", tc.elapsed, delay, min, max) + } + } + + for _, pod := range []*corev1.Pod{pod0, pod1} { + got := &corev1.Pod{} + if err := c.Get(t.Context(), client.ObjectKeyFromObject(pod), got); err != nil { + t.Fatalf("get pod %s: %v", pod.Name, err) + } + condition := readinessCondition(got.Status.Conditions) + if condition == nil || condition.Status != corev1.ConditionFalse { + t.Errorf("pod %s readiness condition = %#v, want False while waiting for a primary", + pod.Name, condition) + } + } + + // The fixture reaches a settled state: both poolers commit a rule naming + // pooler-0 as leader, so both are cohort-eligible members of the same + // durability rule and pooler-0's postgres reports PRIMARY, matching the + // topology role a leader-designate needs. + if err := store.RegisterMultipooler(t.Context(), &clustermetadata.Multipooler{ + Id: id0, + Hostname: "pooler-0", + ShardKey: &clustermetadata.ShardKey{ + Database: string(shard.Spec.DatabaseName), + TableGroup: string(shard.Spec.TableGroupName), + Shard: string(shard.Spec.ShardName), + }, + RoutingState: &clustermetadata.RoutingState{ + Role: clustermetadata.RoutingRole_ROUTING_ROLE_PRIMARY, + }, + }, true); err != nil { + t.Fatalf("promote pooler-0 in topology: %v", err) + } + rule := &clustermetadata.ShardRule{ + RuleNumber: &clustermetadata.RuleNumber{CoordinatorTerm: 1}, + LeaderId: id0, + CohortMembers: []*clustermetadata.ID{id0, id1}, + DurabilityPolicy: topoclient.AtLeastN(1), + } + for _, elected := range []struct { + id *clustermetadata.ID + primary bool + }{{id0, true}, {id1, false}} { + status := multipoolermanagerdatapb.PostgresStatus_POSTGRES_STATUS_STANDBY + if elected.primary { + status = multipoolermanagerdatapb.PostgresStatus_POSTGRES_STATUS_PRIMARY + } + componentID := topoclient.ComponentIDString(elected.id) + rpc.SetStatusResponse(componentID, &multipoolermanagerdatapb.StatusResponse{ + Status: &multipoolermanagerdatapb.Status{ + IsInitialized: true, + PostgresReady: true, + PostgresStatus: status, + }, + AvailabilityStatus: &clustermetadata.AvailabilityStatus{ + CohortEligibilityStatus: &clustermetadata.CohortEligibilityStatus{ + Signal: clustermetadata.CohortEligibilitySignal_COHORT_ELIGIBILITY_SIGNAL_ELIGIBLE, + }, + }, + ConsensusStatus: &clustermetadata.ConsensusStatus{ + Id: elected.id, + CurrentPosition: &clustermetadata.PoolerPosition{ + Position: &clustermetadata.RulePosition{Decision: rule}, + }, + }, + }) + } + + clk.advance(time.Second) + delay, err := r.reconcilePosture(t.Context(), store, shard, rpc) + if err != nil { + t.Fatalf("reconcilePosture() after election error = %v", err) + } + if delay != 0 { + t.Errorf("delay after a primary is elected = %v, want 0", delay) + } + key := fmt.Sprintf("%s/%s", shard.Namespace, shard.Name) + if _, ok := r.notConvergedSince[key]; ok { + t.Errorf("not-converged-since entry left after a primary is elected") + } + if conditionIsFalse(shard.Status.Conditions, "PostureConsistent") { + t.Errorf( + "conditions = %#v, want no failure once a primary is elected", + shard.Status.Conditions, + ) + } + + for _, pod := range []*corev1.Pod{pod0, pod1} { + got := &corev1.Pod{} + if err := c.Get(t.Context(), client.ObjectKeyFromObject(pod), got); err != nil { + t.Fatalf("get pod %s: %v", pod.Name, err) + } + condition := readinessCondition(got.Status.Conditions) + if condition == nil || condition.Status != corev1.ConditionTrue { + t.Errorf("pod %s readiness condition = %#v, want True once a primary is elected", + pod.Name, condition) + } + } +} diff --git a/pkg/resource-handler/controller/shard/shard_controller.go b/pkg/resource-handler/controller/shard/shard_controller.go index d0962cb7..a36c2a7e 100644 --- a/pkg/resource-handler/controller/shard/shard_controller.go +++ b/pkg/resource-handler/controller/shard/shard_controller.go @@ -79,9 +79,22 @@ type ShardReconciler struct { APIReader client.Reader PoolerClients poolerclient.Resolver CreateTopoStore func(*multigresv1alpha1.Shard) (topoclient.Store, error) + // Clock overrides time.Now for recordNotConverged's elapsed-time math. + // Nil in production; tests inject it to drive the readiness backoff + // deterministically. + Clock func() time.Time postureStrikesMu sync.Mutex postureStrikes map[string]int + + // notConvergedMu and notConvergedSince record, per shard, the time it was + // first observed not converged: unsettled (posture debounce past + // threshold, i.e. an accepted mismatch or Incomplete observation), or some + // managed pod not yet posture-ready. Kept separate from postureStrikes, + // which gates posture.Apply: folding this into that counter would change + // when Apply fires. + notConvergedMu sync.Mutex + notConvergedSince map[string]time.Time } // Reconcile manages pool pods, PVCs, services, and data-plane topology for a Shard. @@ -109,6 +122,7 @@ func (r *ShardReconciler) Reconcile( if err := r.Get(ctx, req.NamespacedName, shard); err != nil { if errors.IsNotFound(err) { logger.Info("Shard resource not found, ignoring") + r.forgetStrikes(req.Namespace, req.Name) return ctrl.Result{}, nil } monitoring.RecordSpanError(span, err) @@ -254,26 +268,46 @@ func (r *ShardReconciler) Reconcile( return ctrl.Result{}, err } - if err := r.validateBackupStorageClassDependency(ctx, shard); err != nil { - if isMissingStorageClassDependency(err) { - logger.Info( - "StorageClass dependency missing for shared backup PVC; requeueing", - "after", - storageClassDependencyRequeue, - ) - return ctrl.Result{RequeueAfter: storageClassDependencyRequeue}, nil - } + // Every StorageClass the shard references is resolved here, in one pass, and + // the StorageClassValid condition is published once. Two call sites used to + // write that condition with their own message and overwrite each other on + // every reconcile. The gates stay where they were: this one stops before + // the shared backup PVC, the pool one further down stops before the + // workloads that consume pool storage. + storageClasses, err := r.validateStorageClassDependencies(ctx, shard) + if err != nil { + monitoring.RecordSpanError(span, err) + logger.Error(err, "Failed to validate StorageClass dependencies") + r.Recorder.Eventf( + shard, + "Warning", + "FailedApply", + "Failed to validate StorageClass dependencies: %v", + err, + ) + return ctrl.Result{}, err + } + if err := r.setStorageClassCondition(ctx, shard, storageClasses); err != nil { monitoring.RecordSpanError(span, err) - logger.Error(err, "Failed to validate backup StorageClass") + logger.Error(err, "Failed to set StorageClass condition") r.Recorder.Eventf( shard, "Warning", "FailedApply", - "Failed to validate backup StorageClass: %v", + "Failed to set StorageClass condition: %v", err, ) return ctrl.Result{}, err } + if isMissingStorageClassDependency(storageClasses.backupDependency) { + r.Recorder.Event(shard, "Warning", storageClassNotFoundReason, storageClasses.message) + logger.Info( + "StorageClass dependency missing for shared backup PVC; requeueing", + "after", + storageClassDependencyRequeue, + ) + return ctrl.Result{RequeueAfter: storageClassDependencyRequeue}, nil + } // Reconcile Multiorch - one Deployment and Service per cell { @@ -335,25 +369,14 @@ func (r *ShardReconciler) Reconcile( childSpan.End() } - if err := r.validatePoolStorageClassDependencies(ctx, shard); err != nil { - if isMissingStorageClassDependency(err) { - logger.Info( - "StorageClass dependency missing for pool resources; requeueing", - "after", - storageClassDependencyRequeue, - ) - return ctrl.Result{RequeueAfter: storageClassDependencyRequeue}, nil - } - monitoring.RecordSpanError(span, err) - logger.Error(err, "Failed to validate pool StorageClass dependencies") - r.Recorder.Eventf( - shard, - "Warning", - "FailedApply", - "Failed to validate pool StorageClass dependencies: %v", - err, + if isMissingStorageClassDependency(storageClasses.poolDependency) { + r.Recorder.Event(shard, "Warning", storageClassNotFoundReason, storageClasses.message) + logger.Info( + "StorageClass dependency missing for pool resources; requeueing", + "after", + storageClassDependencyRequeue, ) - return ctrl.Result{}, err + return ctrl.Result{RequeueAfter: storageClassDependencyRequeue}, nil } // Render the effective postgres config into the operator-owned ConfigMap and diff --git a/pkg/resource-handler/controller/shard/storage_class_guard.go b/pkg/resource-handler/controller/shard/storage_class_guard.go index 2004d645..548b2e1a 100644 --- a/pkg/resource-handler/controller/shard/storage_class_guard.go +++ b/pkg/resource-handler/controller/shard/storage_class_guard.go @@ -4,12 +4,16 @@ import ( "context" "errors" "fmt" + "maps" + "slices" "time" storagev1 "k8s.io/api/storage/v1" apierrors "k8s.io/apimachinery/pkg/api/errors" "k8s.io/apimachinery/pkg/api/meta" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/apis/meta/v1/unstructured" + "k8s.io/apimachinery/pkg/runtime" "sigs.k8s.io/controller-runtime/pkg/client" multigresv1alpha1 "github.com/multigres/multigres-operator/api/v1alpha1" @@ -37,6 +41,27 @@ func isMissingStorageClassDependency(err error) bool { return errors.As(err, &depErr) } +// storageClassCheck is the whole StorageClassValid verdict for one reconcile: +// the condition to publish, plus the dependency errors that gate the reconcile. +// +// It exists so the condition has exactly one writer per reconcile. The backup +// and pool checks used to write it independently with different messages, and +// since setStorageClassCondition's skip-if-unchanged test compares the +// persisted message, each call saw the other's message and rewrote it, forever. +// +// backupDependency and poolDependency are held apart because they gate +// different points in Reconcile: a missing backup class must stop before the +// shared backup PVC is created, a missing pool class before the pool +// workloads. Each is a *missingStorageClassDependencyError when set. +type storageClassCheck struct { + status metav1.ConditionStatus + reason string + message string + + backupDependency error + poolDependency error +} + func backupFilesystemStorageClassName(shard *multigresv1alpha1.Shard) string { if shard.Spec.Backup == nil || shard.Spec.Backup.Type != multigresv1alpha1.BackupTypeFilesystem { @@ -67,102 +92,102 @@ func (r *ShardReconciler) validateStorageClassExists( return true, nil } -func (r *ShardReconciler) validateBackupStorageClassDependency( +// validateStorageClassDependencies resolves every StorageClass the shard +// references and reduces them to a single verdict. It writes nothing: the +// caller publishes the condition once and then gates on the missing-class +// fields, which is what keeps the condition single-writer. +// +// A missing backup class short-circuits before the pools are looked at, so the +// reported message matches the order in which Reconcile gates on them. +func (r *ShardReconciler) validateStorageClassDependencies( ctx context.Context, shard *multigresv1alpha1.Shard, -) error { +) (storageClassCheck, error) { backupClass := backupFilesystemStorageClassName(shard) - if backupClass == "" { - return r.setStorageClassCondition( - ctx, - shard, - metav1.ConditionTrue, - storageClassNotSpecifiedReason, - "No explicit backup filesystem StorageClass configured; using cluster default", - ) - } - - exists, err := r.validateStorageClassExists(ctx, backupClass) - if err != nil { - return fmt.Errorf("failed to validate backup StorageClass %q: %w", backupClass, err) - } - if !exists { - msg := fmt.Sprintf("StorageClass %q not found for shared backup PVCs", backupClass) - if setErr := r.setStorageClassCondition( - ctx, - shard, - metav1.ConditionFalse, - storageClassNotFoundReason, - msg, - ); setErr != nil { - return setErr + if backupClass != "" { + exists, err := r.validateStorageClassExists(ctx, backupClass) + if err != nil { + return storageClassCheck{}, fmt.Errorf( + "failed to validate backup StorageClass %q: %w", + backupClass, + err, + ) + } + if !exists { + return storageClassCheck{ + status: metav1.ConditionFalse, + reason: storageClassNotFoundReason, + message: fmt.Sprintf( + "StorageClass %q not found for shared backup PVCs", + backupClass, + ), + backupDependency: &missingStorageClassDependencyError{className: backupClass}, + }, nil } - r.Recorder.Event(shard, "Warning", storageClassNotFoundReason, msg) - return &missingStorageClassDependencyError{className: backupClass} } - return r.setStorageClassCondition( - ctx, - shard, - metav1.ConditionTrue, - storageClassFoundReason, - fmt.Sprintf("StorageClass %q found for shared backup PVCs", backupClass), - ) -} - -func (r *ShardReconciler) validatePoolStorageClassDependencies( - ctx context.Context, - shard *multigresv1alpha1.Shard, -) error { - hasExplicitPoolStorageClass := false - for poolName, pool := range shard.Spec.Pools { - if pool.Storage.Class == "" { + // Pools are visited in name order because Spec.Pools is a map: reporting + // whichever missing class Go's randomised map iteration reached first would + // flap the condition message between reconciles. + hasExplicitPoolClass := false + for _, poolName := range slices.Sorted(maps.Keys(shard.Spec.Pools)) { + poolClass := shard.Spec.Pools[poolName].Storage.Class + if poolClass == "" { continue } - hasExplicitPoolStorageClass = true + hasExplicitPoolClass = true - exists, err := r.validateStorageClassExists(ctx, pool.Storage.Class) + exists, err := r.validateStorageClassExists(ctx, poolClass) if err != nil { - return fmt.Errorf( + return storageClassCheck{}, fmt.Errorf( "failed to validate StorageClass %q for pool %s: %w", - pool.Storage.Class, + poolClass, poolName, err, ) } if !exists { - msg := fmt.Sprintf( - "StorageClass %q not found for pool %s", - pool.Storage.Class, - poolName, - ) - if setErr := r.setStorageClassCondition( - ctx, - shard, - metav1.ConditionFalse, - storageClassNotFoundReason, - msg, - ); setErr != nil { - return setErr - } - r.Recorder.Event(shard, "Warning", storageClassNotFoundReason, msg) - return &missingStorageClassDependencyError{className: pool.Storage.Class} + return storageClassCheck{ + status: metav1.ConditionFalse, + reason: storageClassNotFoundReason, + message: fmt.Sprintf( + "StorageClass %q not found for pool %s", + poolClass, + poolName, + ), + poolDependency: &missingStorageClassDependencyError{className: poolClass}, + }, nil } } - reason := storageClassNotSpecifiedReason - message := "No explicit pool StorageClass configured; using cluster default" - if hasExplicitPoolStorageClass { - reason = storageClassFoundReason - message = "All explicit pool StorageClasses are present" + if backupClass == "" && !hasExplicitPoolClass { + return storageClassCheck{ + status: metav1.ConditionTrue, + reason: storageClassNotSpecifiedReason, + message: "No explicit backup filesystem or pool StorageClass configured; using cluster default", + }, nil } - return r.setStorageClassCondition(ctx, shard, metav1.ConditionTrue, reason, message) + + return storageClassCheck{ + status: metav1.ConditionTrue, + reason: storageClassFoundReason, + message: "All explicitly configured StorageClasses are present", + }, nil } // setStorageClassCondition patches the StorageClassValid condition using SSA. // Uses FieldOwner("multigres-resource-handler-guard") to avoid ownership conflicts // with updateStatus which uses FieldOwner("multigres-resource-handler"). // +// The payload is unstructured and carries status.conditions and nothing else. +// An SSA apply payload is a complete statement of what its field manager owns, +// so a payload built from a partially-populated typed struct silently asserts +// the zero value of every non-omitempty field in it: a ShardStatus literal that +// sets only Conditions still serialises orchReady:false and poolsReady:false, +// and with ForceOwnership it seizes both fields from updateStatus, which forces +// them straight back on its next write. That is a permanent ping-pong, one +// round trip per reconcile, each one scheduling the next reconcile. +// // Reads the latest condition from the API server (not the in-memory shard) to // avoid false skips when the in-memory object is stale. // TODO: This stale-safe condition skip logic is mirrored in the TopoServer @@ -170,9 +195,7 @@ func (r *ShardReconciler) validatePoolStorageClassDependencies( func (r *ShardReconciler) setStorageClassCondition( ctx context.Context, shard *multigresv1alpha1.Shard, - condStatus metav1.ConditionStatus, - reason string, - message string, + check storageClassCheck, ) error { // Read the latest from the API server so the skip-if-unchanged check // compares against the real persisted state, not a potentially stale @@ -184,9 +207,9 @@ func (r *ShardReconciler) setStorageClassCondition( existing := meta.FindStatusCondition(latest.Status.Conditions, conditionStorageClassValid) if existing != nil && - existing.Status == condStatus && - existing.Reason == reason && - existing.Message == message && + existing.Status == check.status && + existing.Reason == check.reason && + existing.Message == check.message && existing.ObservedGeneration == latest.Generation { return nil } @@ -194,32 +217,35 @@ func (r *ShardReconciler) setStorageClassCondition( // Preserve LastTransitionTime when the status hasn't transitioned, // matching the behaviour of meta.SetStatusCondition. now := metav1.Now() - if existing != nil && existing.Status == condStatus { + if existing != nil && existing.Status == check.status { now = existing.LastTransitionTime } - patchObj := &multigresv1alpha1.Shard{ - TypeMeta: metav1.TypeMeta{ - APIVersion: multigresv1alpha1.GroupVersion.String(), - Kind: "Shard", - }, - ObjectMeta: metav1.ObjectMeta{ - Name: shard.Name, - Namespace: shard.Namespace, + // Converted from the typed condition rather than hand-built so the wire + // encoding, notably the metav1.Time format, cannot drift from the API's. + cond, err := runtime.DefaultUnstructuredConverter.ToUnstructured(&metav1.Condition{ + Type: conditionStorageClassValid, + Status: check.status, + Reason: check.reason, + Message: check.message, + ObservedGeneration: latest.Generation, + LastTransitionTime: now, + }) + if err != nil { + return fmt.Errorf("failed to encode Shard StorageClass condition: %w", err) + } + + patchObj := &unstructured.Unstructured{Object: map[string]any{ + "apiVersion": multigresv1alpha1.GroupVersion.String(), + "kind": "Shard", + "metadata": map[string]any{ + "name": shard.Name, + "namespace": shard.Namespace, }, - Status: multigresv1alpha1.ShardStatus{ - Conditions: []metav1.Condition{ - { - Type: conditionStorageClassValid, - Status: condStatus, - Reason: reason, - Message: message, - ObservedGeneration: latest.Generation, - LastTransitionTime: now, - }, - }, + "status": map[string]any{ + "conditions": []any{cond}, }, - } + }} if err := r.Status().Patch( ctx, diff --git a/pkg/resource-handler/controller/shard/storage_class_guard_test.go b/pkg/resource-handler/controller/shard/storage_class_guard_test.go index 7df20f97..70137e36 100644 --- a/pkg/resource-handler/controller/shard/storage_class_guard_test.go +++ b/pkg/resource-handler/controller/shard/storage_class_guard_test.go @@ -1,19 +1,27 @@ package shard import ( + "context" + "encoding/json" "errors" + "maps" + "slices" + "strings" "testing" + "time" appsv1 "k8s.io/api/apps/v1" corev1 "k8s.io/api/core/v1" policyv1 "k8s.io/api/policy/v1" storagev1 "k8s.io/api/storage/v1" + apierrors "k8s.io/apimachinery/pkg/api/errors" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" "k8s.io/apimachinery/pkg/runtime" "k8s.io/utils/ptr" ctrl "sigs.k8s.io/controller-runtime" "sigs.k8s.io/controller-runtime/pkg/client" "sigs.k8s.io/controller-runtime/pkg/client/fake" + "sigs.k8s.io/controller-runtime/pkg/client/interceptor" "k8s.io/client-go/tools/record" @@ -22,191 +30,620 @@ import ( "github.com/multigres/multigres-operator/pkg/util/metadata" ) -func TestValidateBackupStorageClassDependency(t *testing.T) { +func TestValidateStorageClassDependencies(t *testing.T) { scheme := runtime.NewScheme() _ = multigresv1alpha1.AddToScheme(scheme) _ = storagev1.AddToScheme(scheme) - t.Run("no explicit backup class sets true not-specified condition", func(t *testing.T) { - shard := &multigresv1alpha1.Shard{ - ObjectMeta: metav1.ObjectMeta{Name: "test-shard", Namespace: "default"}, - } + newReconciler := func(objs ...client.Object) *ShardReconciler { c := fake.NewClientBuilder(). WithScheme(scheme). - WithObjects(shard). + WithObjects(objs...). WithStatusSubresource(&multigresv1alpha1.Shard{}). Build() - r := &ShardReconciler{Client: c, Scheme: scheme, Recorder: record.NewFakeRecorder(10)} + return &ShardReconciler{Client: c, Scheme: scheme, Recorder: record.NewFakeRecorder(10)} + } - if err := r.validateBackupStorageClassDependency(t.Context(), shard); err != nil { - t.Fatalf("unexpected error: %v", err) + filesystemBackup := func(class string) *multigresv1alpha1.BackupConfig { + return &multigresv1alpha1.BackupConfig{ + Type: multigresv1alpha1.BackupTypeFilesystem, + Filesystem: &multigresv1alpha1.FilesystemBackupConfig{ + Storage: multigresv1alpha1.StorageSpec{Class: class}, + }, + } + } + + t.Run("nothing explicit reports one not-specified verdict", func(t *testing.T) { + shard := &multigresv1alpha1.Shard{ + ObjectMeta: metav1.ObjectMeta{Name: "test-shard", Namespace: "default"}, + Spec: multigresv1alpha1.ShardSpec{ + Pools: map[multigresv1alpha1.PoolName]multigresv1alpha1.PoolSpec{ + "primary": {Storage: multigresv1alpha1.StorageSpec{Size: "10Gi"}}, + }, + }, } + r := newReconciler(shard) - var updated multigresv1alpha1.Shard - if err := c.Get(t.Context(), client.ObjectKeyFromObject(shard), &updated); err != nil { - t.Fatalf("failed to read shard: %v", err) + check, err := r.validateStorageClassDependencies(t.Context(), shard) + if err != nil { + t.Fatalf("unexpected error: %v", err) } - cond := findCondition(updated.Status.Conditions, conditionStorageClassValid) - if cond == nil || cond.Status != metav1.ConditionTrue || - cond.Reason != storageClassNotSpecifiedReason { - t.Fatalf("unexpected condition: %#v", cond) + if check.status != metav1.ConditionTrue || check.reason != storageClassNotSpecifiedReason { + t.Fatalf("unexpected verdict: %+v", check) + } + if check.backupDependency != nil || check.poolDependency != nil { + t.Fatalf("expected no dependency errors, got %+v", check) } }) - t.Run("missing backup class returns dependency error and false condition", func(t *testing.T) { + t.Run("explicit classes all present report found", func(t *testing.T) { shard := &multigresv1alpha1.Shard{ ObjectMeta: metav1.ObjectMeta{Name: "test-shard", Namespace: "default"}, Spec: multigresv1alpha1.ShardSpec{ - Backup: &multigresv1alpha1.BackupConfig{ - Type: multigresv1alpha1.BackupTypeFilesystem, - Filesystem: &multigresv1alpha1.FilesystemBackupConfig{ - Storage: multigresv1alpha1.StorageSpec{Class: "missing-sc"}, + Backup: filesystemBackup("backup-sc"), + Pools: map[multigresv1alpha1.PoolName]multigresv1alpha1.PoolSpec{ + "primary": { + Storage: multigresv1alpha1.StorageSpec{Size: "10Gi", Class: "fast"}, }, }, }, } - c := fake.NewClientBuilder(). - WithScheme(scheme). - WithObjects(shard). - WithStatusSubresource(&multigresv1alpha1.Shard{}). - Build() - r := &ShardReconciler{Client: c, Scheme: scheme, Recorder: record.NewFakeRecorder(10)} + r := newReconciler( + shard, + &storagev1.StorageClass{ObjectMeta: metav1.ObjectMeta{Name: "backup-sc"}}, + &storagev1.StorageClass{ObjectMeta: metav1.ObjectMeta{Name: "fast"}}, + ) - err := r.validateBackupStorageClassDependency(t.Context(), shard) - if err == nil || !isMissingStorageClassDependency(err) { - t.Fatalf("expected missing dependency error, got: %v", err) + check, err := r.validateStorageClassDependencies(t.Context(), shard) + if err != nil { + t.Fatalf("unexpected error: %v", err) } + if check.status != metav1.ConditionTrue || check.reason != storageClassFoundReason { + t.Fatalf("unexpected verdict: %+v", check) + } + }) - var updated multigresv1alpha1.Shard - if getErr := c.Get( - t.Context(), - client.ObjectKeyFromObject(shard), - &updated, - ); getErr != nil { - t.Fatalf("failed to read shard: %v", getErr) + // One writer means one verdict, and in the mixed cases the merged verdict + // reports Found where the old pair reported NotSpecified: the pool validator + // ran last and claimed the shard as unconfigured even when the backup class + // was explicit. The reason is published on the condition, so both directions + // of "some of it is explicit" are pinned here. + t.Run("explicit backup class with no pool class reports found", func(t *testing.T) { + shard := &multigresv1alpha1.Shard{ + ObjectMeta: metav1.ObjectMeta{Name: "test-shard", Namespace: "default"}, + Spec: multigresv1alpha1.ShardSpec{ + Backup: filesystemBackup("backup-sc"), + Pools: map[multigresv1alpha1.PoolName]multigresv1alpha1.PoolSpec{ + "primary": {Storage: multigresv1alpha1.StorageSpec{Size: "10Gi"}}, + }, + }, + } + r := newReconciler( + shard, + &storagev1.StorageClass{ObjectMeta: metav1.ObjectMeta{Name: "backup-sc"}}, + ) + + check, err := r.validateStorageClassDependencies(t.Context(), shard) + if err != nil { + t.Fatalf("unexpected error: %v", err) } - cond := findCondition(updated.Status.Conditions, conditionStorageClassValid) - if cond == nil || cond.Status != metav1.ConditionFalse || - cond.Reason != storageClassNotFoundReason { - t.Fatalf("unexpected condition: %#v", cond) + if check.status != metav1.ConditionTrue || check.reason != storageClassFoundReason { + t.Fatalf("unexpected verdict: %+v", check) } }) -} -func TestValidatePoolStorageClassDependencies(t *testing.T) { - scheme := runtime.NewScheme() - _ = multigresv1alpha1.AddToScheme(scheme) - _ = storagev1.AddToScheme(scheme) - - t.Run("no explicit pool class sets true not-specified condition", func(t *testing.T) { + t.Run("explicit pool class with no backup class reports found", func(t *testing.T) { shard := &multigresv1alpha1.Shard{ ObjectMeta: metav1.ObjectMeta{Name: "test-shard", Namespace: "default"}, Spec: multigresv1alpha1.ShardSpec{ Pools: map[multigresv1alpha1.PoolName]multigresv1alpha1.PoolSpec{ "primary": { - Storage: multigresv1alpha1.StorageSpec{Size: "10Gi"}, + Storage: multigresv1alpha1.StorageSpec{Size: "10Gi", Class: "fast"}, }, }, }, } - c := fake.NewClientBuilder(). - WithScheme(scheme). - WithObjects(shard). - WithStatusSubresource(&multigresv1alpha1.Shard{}). - Build() - r := &ShardReconciler{Client: c, Scheme: scheme, Recorder: record.NewFakeRecorder(10)} + r := newReconciler( + shard, + &storagev1.StorageClass{ObjectMeta: metav1.ObjectMeta{Name: "fast"}}, + ) - if err := r.validatePoolStorageClassDependencies(t.Context(), shard); err != nil { + check, err := r.validateStorageClassDependencies(t.Context(), shard) + if err != nil { t.Fatalf("unexpected error: %v", err) } + if check.status != metav1.ConditionTrue || check.reason != storageClassFoundReason { + t.Fatalf("unexpected verdict: %+v", check) + } + }) - var updated multigresv1alpha1.Shard - if err := c.Get(t.Context(), client.ObjectKeyFromObject(shard), &updated); err != nil { - t.Fatalf("failed to read shard: %v", err) + t.Run("missing backup class reports the backup dependency", func(t *testing.T) { + shard := &multigresv1alpha1.Shard{ + ObjectMeta: metav1.ObjectMeta{Name: "test-shard", Namespace: "default"}, + Spec: multigresv1alpha1.ShardSpec{Backup: filesystemBackup("missing-sc")}, + } + r := newReconciler(shard) + + check, err := r.validateStorageClassDependencies(t.Context(), shard) + if err != nil { + t.Fatalf("unexpected error: %v", err) + } + if check.status != metav1.ConditionFalse || check.reason != storageClassNotFoundReason { + t.Fatalf("unexpected verdict: %+v", check) + } + if !isMissingStorageClassDependency(check.backupDependency) { + t.Fatalf("expected backup dependency error, got %v", check.backupDependency) } - cond := findCondition(updated.Status.Conditions, conditionStorageClassValid) - if cond == nil || cond.Status != metav1.ConditionTrue || - cond.Reason != storageClassNotSpecifiedReason { - t.Fatalf("unexpected condition: %#v", cond) + if check.poolDependency != nil { + t.Fatalf("expected no pool dependency error, got %v", check.poolDependency) + } + if !strings.Contains(check.message, `"missing-sc"`) { + t.Fatalf("message must name the class, got %q", check.message) } }) - t.Run("all explicit pool classes present sets true found condition", func(t *testing.T) { + t.Run("missing pool class reports the pool dependency", func(t *testing.T) { shard := &multigresv1alpha1.Shard{ ObjectMeta: metav1.ObjectMeta{Name: "test-shard", Namespace: "default"}, Spec: multigresv1alpha1.ShardSpec{ Pools: map[multigresv1alpha1.PoolName]multigresv1alpha1.PoolSpec{ "primary": { - Storage: multigresv1alpha1.StorageSpec{Size: "10Gi", Class: "fast"}, + Storage: multigresv1alpha1.StorageSpec{Size: "10Gi", Class: "missing-sc"}, }, }, }, } - c := fake.NewClientBuilder(). - WithScheme(scheme). - WithObjects(shard, &storagev1.StorageClass{ObjectMeta: metav1.ObjectMeta{Name: "fast"}}). - WithStatusSubresource(&multigresv1alpha1.Shard{}). - Build() - r := &ShardReconciler{Client: c, Scheme: scheme, Recorder: record.NewFakeRecorder(10)} + r := newReconciler(shard) - if err := r.validatePoolStorageClassDependencies(t.Context(), shard); err != nil { + check, err := r.validateStorageClassDependencies(t.Context(), shard) + if err != nil { t.Fatalf("unexpected error: %v", err) } - - var updated multigresv1alpha1.Shard - if err := c.Get(t.Context(), client.ObjectKeyFromObject(shard), &updated); err != nil { - t.Fatalf("failed to read shard: %v", err) + if check.status != metav1.ConditionFalse || check.reason != storageClassNotFoundReason { + t.Fatalf("unexpected verdict: %+v", check) } - cond := findCondition(updated.Status.Conditions, conditionStorageClassValid) - if cond == nil || cond.Status != metav1.ConditionTrue || - cond.Reason != storageClassFoundReason { - t.Fatalf("unexpected condition: %#v", cond) + if !isMissingStorageClassDependency(check.poolDependency) { + t.Fatalf("expected pool dependency error, got %v", check.poolDependency) + } + if check.backupDependency != nil { + t.Fatalf("expected no backup dependency error, got %v", check.backupDependency) + } + if !strings.Contains(check.message, "primary") { + t.Fatalf("message must name the pool, got %q", check.message) } }) - t.Run( - "missing explicit pool class returns dependency error and false condition", - func(t *testing.T) { - shard := &multigresv1alpha1.Shard{ - ObjectMeta: metav1.ObjectMeta{Name: "test-shard", Namespace: "default"}, - Spec: multigresv1alpha1.ShardSpec{ - Pools: map[multigresv1alpha1.PoolName]multigresv1alpha1.PoolSpec{ - "primary": { - Storage: multigresv1alpha1.StorageSpec{ - Size: "10Gi", - Class: "missing-sc", - }, + t.Run("missing backup class wins over a missing pool class", func(t *testing.T) { + shard := &multigresv1alpha1.Shard{ + ObjectMeta: metav1.ObjectMeta{Name: "test-shard", Namespace: "default"}, + Spec: multigresv1alpha1.ShardSpec{ + Backup: filesystemBackup("missing-backup-sc"), + Pools: map[multigresv1alpha1.PoolName]multigresv1alpha1.PoolSpec{ + "primary": { + Storage: multigresv1alpha1.StorageSpec{ + Size: "10Gi", + Class: "missing-pool-sc", }, }, }, - } - c := fake.NewClientBuilder(). - WithScheme(scheme). - WithObjects(shard). - WithStatusSubresource(&multigresv1alpha1.Shard{}). - Build() - r := &ShardReconciler{Client: c, Scheme: scheme, Recorder: record.NewFakeRecorder(10)} + }, + } + r := newReconciler(shard) - err := r.validatePoolStorageClassDependencies(t.Context(), shard) - if err == nil || !isMissingStorageClassDependency(err) { - t.Fatalf("expected missing dependency error, got: %v", err) - } + check, err := r.validateStorageClassDependencies(t.Context(), shard) + if err != nil { + t.Fatalf("unexpected error: %v", err) + } + if !isMissingStorageClassDependency(check.backupDependency) || + check.poolDependency != nil { + t.Fatalf("expected only the backup dependency, got %+v", check) + } + }) + + // Spec.Pools is a map, so an unordered scan would report whichever missing + // pool Go's iteration reached first and flap the condition message. + t.Run("the reported pool is stable across calls", func(t *testing.T) { + shard := &multigresv1alpha1.Shard{ + ObjectMeta: metav1.ObjectMeta{Name: "test-shard", Namespace: "default"}, + Spec: multigresv1alpha1.ShardSpec{ + Pools: map[multigresv1alpha1.PoolName]multigresv1alpha1.PoolSpec{ + "aaa": { + Storage: multigresv1alpha1.StorageSpec{Size: "10Gi", Class: "missing-a"}, + }, + "bbb": { + Storage: multigresv1alpha1.StorageSpec{Size: "10Gi", Class: "missing-b"}, + }, + "ccc": { + Storage: multigresv1alpha1.StorageSpec{Size: "10Gi", Class: "missing-c"}, + }, + }, + }, + } + r := newReconciler(shard) - var updated multigresv1alpha1.Shard - if getErr := c.Get( - t.Context(), - client.ObjectKeyFromObject(shard), - &updated, - ); getErr != nil { - t.Fatalf("failed to read shard: %v", getErr) + want := "" + for i := range 30 { + check, err := r.validateStorageClassDependencies(t.Context(), shard) + if err != nil { + t.Fatalf("unexpected error: %v", err) + } + if i == 0 { + want = check.message + } + if check.message != want { + t.Fatalf("message flapped: %q then %q", want, check.message) } - cond := findCondition(updated.Status.Conditions, conditionStorageClassValid) - if cond == nil || cond.Status != metav1.ConditionFalse || - cond.Reason != storageClassNotFoundReason { - t.Fatalf("unexpected condition: %#v", cond) + } + if !strings.Contains(want, "aaa") { + t.Fatalf("expected the first pool by name, got %q", want) + } + }) +} + +// TestSetStorageClassCondition_AppliesOnlyConditions pins the defect that made +// a healthy shard rewrite its status forever. The guard's apply payload is a +// complete statement of what its field manager owns, so it must carry +// status.conditions and nothing else: a payload built from a typed ShardStatus +// literal also serialises orchReady:false and poolsReady:false, and with +// ForceOwnership it seizes both from updateStatus on every reconcile. +// +// The assertion is on the serialised payload rather than on the Go value, +// because the Go value looks correct in both the fixed and the broken version. +// The zero values only become an assertion once they are marshalled. +// +// It cannot be made against the object after a round trip here: the fake client +// does not scope an apply patch to the payload's fields, it replaces the whole +// status, so orchReady and poolsReady are lost either way. The round-trip half +// of this belongs to an apiserver, and lives in TestShardStatusQuiesces under +// test/suite. +func TestSetStorageClassCondition_AppliesOnlyConditions(t *testing.T) { + t.Parallel() + + scheme := runtime.NewScheme() + _ = multigresv1alpha1.AddToScheme(scheme) + _ = storagev1.AddToScheme(scheme) + + shard := &multigresv1alpha1.Shard{ + ObjectMeta: metav1.ObjectMeta{Name: "test-shard", Namespace: "default"}, + Status: multigresv1alpha1.ShardStatus{ + OrchReady: true, + PoolsReady: true, + }, + } + + baseClient := fake.NewClientBuilder(). + WithScheme(scheme). + WithObjects(shard). + WithStatusSubresource(&multigresv1alpha1.Shard{}). + Build() + + var captured []byte + fakeClient := testutil.NewFakeClientWithFailures(baseClient, &testutil.FailureConfig{ + OnStatusPatch: func(obj client.Object) error { + raw, err := json.Marshal(obj) + if err != nil { + t.Fatalf("marshal patch payload: %v", err) } + captured = raw + return nil }, - ) + }) + + r := &ShardReconciler{Client: fakeClient, Scheme: scheme, Recorder: record.NewFakeRecorder(10)} + + check := storageClassCheck{ + status: metav1.ConditionTrue, + reason: storageClassNotSpecifiedReason, + message: "No explicit backup filesystem or pool StorageClass configured; using cluster default", + } + if err := r.setStorageClassCondition(t.Context(), shard, check); err != nil { + t.Fatalf("setStorageClassCondition: %v", err) + } + if captured == nil { + t.Fatal("guard did not apply a status patch") + } + + var payload struct { + Status map[string]json.RawMessage `json:"status"` + } + if err := json.Unmarshal(captured, &payload); err != nil { + t.Fatalf("unmarshal patch payload: %v", err) + } + keys := slices.Sorted(maps.Keys(payload.Status)) + if !slices.Equal(keys, []string{"conditions"}) { + t.Fatalf("guard payload must own status.conditions only, got %v", keys) + } + + var conditions []metav1.Condition + if err := json.Unmarshal(payload.Status["conditions"], &conditions); err != nil { + t.Fatalf("unmarshal conditions: %v", err) + } + if len(conditions) != 1 || conditions[0].Type != conditionStorageClassValid { + t.Fatalf("guard payload must carry exactly the %s condition, got %+v", + conditionStorageClassValid, conditions) + } + if conditions[0].Status != check.status || conditions[0].Reason != check.reason || + conditions[0].Message != check.message { + t.Fatalf("condition does not match the verdict: %+v", conditions[0]) + } +} + +// TestStorageClassCondition_IsStableAcrossReconciles pins the second defect: +// two validations wrote the same condition type with different messages, so +// every reconcile rewrote the other's and the condition never settled. Once it +// has one writer, the second cycle must not write at all. +// +// This drives the guard cycle directly rather than Reconcile, because a full +// reconcile calls updateStatus again afterwards and the fake client's apply +// drops the guard's condition when it does (see +// TestSetStorageClassCondition_AppliesOnlyConditions). The full-reconcile form +// of this assertion is TestShardStatusQuiesces under test/suite. +func TestStorageClassCondition_IsStableAcrossReconciles(t *testing.T) { + t.Parallel() + + scheme := runtime.NewScheme() + _ = multigresv1alpha1.AddToScheme(scheme) + _ = storagev1.AddToScheme(scheme) + + // No explicit StorageClass anywhere: the case both validations used to + // claim with a message of their own. + shard := &multigresv1alpha1.Shard{ + ObjectMeta: metav1.ObjectMeta{Name: "test-shard", Namespace: "default"}, + Spec: multigresv1alpha1.ShardSpec{ + Pools: map[multigresv1alpha1.PoolName]multigresv1alpha1.PoolSpec{ + "primary": {Storage: multigresv1alpha1.StorageSpec{Size: "10Gi"}}, + }, + }, + } + + baseClient := fake.NewClientBuilder(). + WithScheme(scheme). + WithObjects(shard). + WithStatusSubresource(&multigresv1alpha1.Shard{}). + Build() + + patches := 0 + fakeClient := testutil.NewFakeClientWithFailures(baseClient, &testutil.FailureConfig{ + OnStatusPatch: func(client.Object) error { + patches++ + return nil + }, + }) + + r := &ShardReconciler{Client: fakeClient, Scheme: scheme, Recorder: record.NewFakeRecorder(50)} + + reconcileStorageClasses := func() []byte { + check, err := r.validateStorageClassDependencies(t.Context(), shard) + if err != nil { + t.Fatalf("validate: %v", err) + } + if err := r.setStorageClassCondition(t.Context(), shard, check); err != nil { + t.Fatalf("set condition: %v", err) + } + + var got multigresv1alpha1.Shard + if err := baseClient.Get(t.Context(), client.ObjectKeyFromObject(shard), &got); err != nil { + t.Fatalf("read shard: %v", err) + } + cond := findCondition(got.Status.Conditions, conditionStorageClassValid) + if cond == nil { + t.Fatalf("no %s condition", conditionStorageClassValid) + } + raw, err := json.Marshal(cond) + if err != nil { + t.Fatalf("marshal condition: %v", err) + } + return raw + } + + first := reconcileStorageClasses() + if patches != 1 { + t.Fatalf("first cycle must apply the condition once, applied %d times", patches) + } + + second := reconcileStorageClasses() + if string(first) != string(second) { + t.Fatalf( + "StorageClassValid condition moved between reconciles:\n %s\n %s", + first, + second, + ) + } + if patches != 1 { + t.Fatalf("second cycle rewrote a settled condition: %d applies total", patches) + } +} + +// guardStatusApply is one status apply captured under the guard's field +// owner: the top-level keys of the applied status object, and the single +// StorageClassValid condition it carried. +type guardStatusApply struct { + statusKeys []string + condition metav1.Condition +} + +// TestReconcile_StorageClassConditionSettlesAcrossReconciles drives the real +// ShardReconciler.Reconcile, not the guard cycle in isolation, because the hot +// loop only existed when the guard's apply and updateStatus's apply landed +// against the same object in the same reconcile: a reviewer showed that +// reintroducing a second writer for StorageClassValid (e.g. splitting the +// backup and pool checks back into two condition-setting calls) passes +// TestStorageClassCondition_IsStableAcrossReconciles above, because that test +// drives the merged validateStorageClassDependencies directly and never +// exercises whatever calls the write path from within one Reconcile pass. +// +// It captures every guard-owned status apply's payload directly, identified +// by the real client.SubResourcePatchOptions.FieldManager rather than by +// payload shape, so the assertions hold regardless of how the guard's payload +// happens to be built. +// +// It does not assert that the second reconcile makes zero guard applies, even +// though on a real API server it would: the fake client's SSA implementation +// (managedfields.DeducedTypeConverter, since the Shard CRD's structural schema +// isn't available to it) treats the +listType=map Conditions slice as atomic, +// so updateStatus's own apply - a different field owner, earlier in the same +// Reconcile - replaces status.conditions wholesale and erases whatever the +// guard wrote in the previous reconcile. That forces the guard to see no +// existing condition and reapply every single call, on both fixed and broken +// code, which is exactly why TestStorageClassCondition_IsStableAcrossReconciles +// avoids driving Reconcile for its own assertion. What is real and worth +// pinning here: each individual Reconcile call makes at most one guard apply +// (a second writer within one pass would make two), and the verdict it writes +// does not flap from one reconcile to the next. The true single-writer, +// settles-for-real proof against a schema-aware apply is +// TestShardStatusQuiesces under test/suite. +func TestReconcile_StorageClassConditionSettlesAcrossReconciles(t *testing.T) { + t.Parallel() + + scheme := runtime.NewScheme() + _ = multigresv1alpha1.AddToScheme(scheme) + _ = appsv1.AddToScheme(scheme) + _ = corev1.AddToScheme(scheme) + _ = policyv1.AddToScheme(scheme) + _ = storagev1.AddToScheme(scheme) + + // No explicit StorageClass anywhere: the case the guard and updateStatus + // used to fight over. + shard := &multigresv1alpha1.Shard{ + ObjectMeta: metav1.ObjectMeta{Name: "stable-shard", Namespace: "default"}, + Spec: multigresv1alpha1.ShardSpec{ + DatabaseName: "testdb", + TableGroupName: "default", + Multiorch: multigresv1alpha1.MultiorchSpec{ + Cells: []multigresv1alpha1.CellName{"zone1"}, + }, + Pools: map[multigresv1alpha1.PoolName]multigresv1alpha1.PoolSpec{ + "primary": { + Cells: []multigresv1alpha1.CellName{"zone1"}, + Type: "replica", + ReplicasPerCell: ptr.To(int32(1)), + Storage: multigresv1alpha1.StorageSpec{Size: "10Gi"}, + }, + }, + }, + } + setTestPostgresPasswordSecretRef(shard) + + var guardApplies []guardStatusApply + fakeClient := fake.NewClientBuilder(). + WithScheme(scheme). + WithObjects(shard, testPostgresPasswordSecretForShard(shard)). + WithStatusSubresource(&multigresv1alpha1.Shard{}). + WithInterceptorFuncs(interceptor.Funcs{ + SubResourcePatch: func( + ctx context.Context, + c client.Client, + subResourceName string, + obj client.Object, + patch client.Patch, + opts ...client.SubResourcePatchOption, + ) error { + popts := (&client.SubResourcePatchOptions{}).ApplyOptions(opts) + if popts.FieldManager == "multigres-resource-handler-guard" { + raw, err := json.Marshal(obj) + if err != nil { + t.Fatalf("marshal guard payload: %v", err) + } + var payload struct { + Status map[string]json.RawMessage `json:"status"` + } + if err := json.Unmarshal(raw, &payload); err != nil { + t.Fatalf("unmarshal guard payload: %v", err) + } + var conditions []metav1.Condition + if raw, ok := payload.Status["conditions"]; ok { + if err := json.Unmarshal(raw, &conditions); err != nil { + t.Fatalf("unmarshal guard conditions: %v", err) + } + } + if len(conditions) != 1 { + t.Fatalf( + "guard payload must carry exactly one condition, got %+v", + conditions, + ) + } + guardApplies = append(guardApplies, guardStatusApply{ + statusKeys: slices.Sorted(maps.Keys(payload.Status)), + condition: conditions[0], + }) + } + return c.SubResource(subResourceName).Patch(ctx, obj, patch, opts...) + }, + }). + Build() + + r := &ShardReconciler{ + Client: fakeClient, + Scheme: scheme, + Recorder: record.NewFakeRecorder(100), + CreateTopoStore: newMemoryTopoFactory(), + } + + req := ctrl.Request{NamespacedName: client.ObjectKeyFromObject(shard)} + + assertOnlyOwnsConditions := func(t *testing.T, apply guardStatusApply) { + t.Helper() + if !slices.Equal(apply.statusKeys, []string{"conditions"}) { + t.Fatalf("guard payload must own status.conditions only, got %v", apply.statusKeys) + } + if apply.condition.Type != conditionStorageClassValid { + t.Fatalf("guard payload must carry the %s condition, got %+v", + conditionStorageClassValid, apply.condition) + } + } + + if _, err := r.Reconcile(t.Context(), req); err != nil { + t.Fatalf("first reconcile: %v", err) + } + if len(guardApplies) != 1 { + t.Fatalf( + "first reconcile must make exactly one guard apply (a second writer is back), got %d", + len(guardApplies), + ) + } + first := guardApplies[0] + assertOnlyOwnsConditions(t, first) + if first.condition.Status != metav1.ConditionTrue || + first.condition.Reason != storageClassNotSpecifiedReason { + t.Fatalf("unexpected first verdict: %+v", first.condition) + } + + // The fake client's SSA status apply replaces the whole object rather than + // scoping to the applied fields, which drops Spec. Restore it exactly as the + // existing multi-reconcile loop in TestShardReconciler_Reconcile does, so the + // second reconcile sees the same spec as the first rather than erroring on a + // shard with no pools. + var stored multigresv1alpha1.Shard + if err := fakeClient.Get(t.Context(), req.NamespacedName, &stored); err != nil { + t.Fatalf("read shard before restoring spec: %v", err) + } + stored.Spec = shard.Spec + if err := fakeClient.Update(t.Context(), &stored); err != nil { + t.Fatalf("restore shard spec: %v", err) + } + + if _, err := r.Reconcile(t.Context(), req); err != nil { + t.Fatalf("second reconcile: %v", err) + } + if len(guardApplies) != 2 { + t.Fatalf( + "second reconcile must make exactly one guard apply of its own (a second writer "+ + "is back), got %d total", + len(guardApplies), + ) + } + second := guardApplies[1] + assertOnlyOwnsConditions(t, second) + + if second.condition.Status != first.condition.Status || + second.condition.Reason != first.condition.Reason || + second.condition.Message != first.condition.Message { + t.Fatalf( + "StorageClassValid verdict flapped between reconciles:\n %+v\n %+v", + first.condition, + second.condition, + ) + } } func TestReconcile_MissingStorageClassReturnsDependencyRequeueEvenWhenPVCExists(t *testing.T) { @@ -406,88 +843,561 @@ func TestShardReconciler_FieldOwnershipIsolation(t *testing.T) { t.Fatal("updateStatus patch must contain Available condition") } }) +} - t.Run( - "guard patch contains only StorageClassValid condition and no other status fields", - func(t *testing.T) { - t.Parallel() +// storageClassGateScheme registers everything a full Reconcile of the gate +// fixture below touches. +func storageClassGateScheme() *runtime.Scheme { + scheme := runtime.NewScheme() + _ = multigresv1alpha1.AddToScheme(scheme) + _ = appsv1.AddToScheme(scheme) + _ = corev1.AddToScheme(scheme) + _ = policyv1.AddToScheme(scheme) + _ = storagev1.AddToScheme(scheme) + return scheme +} - shard := &multigresv1alpha1.Shard{ - ObjectMeta: metav1.ObjectMeta{ - Name: "test-field-owner-2", - Namespace: "default", +// storageClassGateShard is a shard that reconciles far enough to pass both +// storage-class gates, so a test can make exactly one of them fire by naming a +// class that does not exist. +func storageClassGateShard(backupClass, poolClass string) *multigresv1alpha1.Shard { + return &multigresv1alpha1.Shard{ + ObjectMeta: metav1.ObjectMeta{ + Name: "gate-shard", + Namespace: "default", + Labels: map[string]string{ + metadata.LabelMultigresCluster: "test-cluster", + }, + }, + Spec: multigresv1alpha1.ShardSpec{ + DatabaseName: "db", + TableGroupName: "tg", + ShardName: "s1", + PostgresPasswordSecretRef: multigresv1alpha1.PostgresPasswordSecretRef{ + Name: testPostgresAuthRefName, + Key: PostgresPasswordSecretKey, + }, + Multiorch: multigresv1alpha1.MultiorchSpec{ + Cells: []multigresv1alpha1.CellName{"zone1"}, + }, + Backup: &multigresv1alpha1.BackupConfig{ + Type: multigresv1alpha1.BackupTypeFilesystem, + Filesystem: &multigresv1alpha1.FilesystemBackupConfig{ + Storage: multigresv1alpha1.StorageSpec{Size: "10Gi", Class: backupClass}, }, - Spec: multigresv1alpha1.ShardSpec{ - Pools: map[multigresv1alpha1.PoolName]multigresv1alpha1.PoolSpec{ - "primary": { - Storage: multigresv1alpha1.StorageSpec{ - Size: "10Gi", - Class: "fast-ssd", - }, - }, - }, + }, + Pools: map[multigresv1alpha1.PoolName]multigresv1alpha1.PoolSpec{ + "primary": { + ReplicasPerCell: ptr.To(int32(1)), + Cells: []multigresv1alpha1.CellName{"zone1"}, + Type: "readWrite", + Storage: multigresv1alpha1.StorageSpec{Size: "10Gi", Class: poolClass}, }, - } - sc := &storagev1.StorageClass{ObjectMeta: metav1.ObjectMeta{Name: "fast-ssd"}} + }, + }, + } +} +func childExists(t *testing.T, c client.Client, obj client.Object, namespace, name string) bool { + t.Helper() + err := c.Get(t.Context(), client.ObjectKey{Namespace: namespace, Name: name}, obj) + if err == nil { + return true + } + if !apierrors.IsNotFound(err) { + t.Fatalf("unexpected error reading %T %s: %v", obj, name, err) + } + return false +} + +// TestReconcile_MissingBackupStorageClassStopsBeforeTheSharedBackupPVC and its +// pool twin below pin where the two gates sit, not just that they fire. The +// returned RequeueAfter is the same wherever a gate is placed, so the only +// observable that moves when a gate moves is which children the pass created: +// a missing backup class must stop before the shared backup PVC is applied, and +// a missing pool class must stop after it and before anything that consumes +// pool storage. +func TestReconcile_MissingBackupStorageClassStopsBeforeTheSharedBackupPVC(t *testing.T) { + scheme := storageClassGateScheme() + shard := storageClassGateShard("missing-backup-sc", "") + + c := fake.NewClientBuilder(). + WithScheme(scheme). + WithObjects(shard, testPostgresPasswordSecretForShard(shard)). + WithStatusSubresource(&multigresv1alpha1.Shard{}). + Build() + + r := &ShardReconciler{ + Client: c, + Scheme: scheme, + Recorder: record.NewFakeRecorder(100), + APIReader: c, + CreateTopoStore: newMemoryTopoFactory(), + } + + result, err := r.Reconcile(t.Context(), ctrl.Request{ + NamespacedName: client.ObjectKeyFromObject(shard), + }) + if err != nil { + t.Fatalf("expected non-error dependency requeue, got error: %v", err) + } + if result.RequeueAfter != storageClassDependencyRequeue { + t.Fatalf("requeueAfter = %v, want %v", result.RequeueAfter, storageClassDependencyRequeue) + } + + ns := shard.Namespace + if !childExists(t, c, &corev1.ConfigMap{}, ns, PgHbaConfigMapName(shard.Name)) { + t.Error("pg_hba ConfigMap is missing: the backup gate moved above the shared ConfigMaps") + } + if childExists(t, c, &appsv1.Deployment{}, ns, buildHashedMultiorchName(shard, "zone1")) { + t.Error("Multiorch Deployment was created: the backup gate moved below the Multiorch block") + } + if childExists(t, c, &corev1.PersistentVolumeClaim{}, ns, BuildSharedBackupPVCName(shard)) { + t.Error( + "shared backup PVC was created against a StorageClass that does not exist: " + + "the backup gate moved below the backup PVC block", + ) + } + if childExists(t, c, &corev1.ConfigMap{}, ns, PostgresConfigMapName(shard.Name)) { + t.Error("postgres config ConfigMap was created: the reconcile ran past both gates") + } +} + +func TestReconcile_MissingPoolStorageClassStopsAfterTheSharedBackupPVC(t *testing.T) { + scheme := storageClassGateScheme() + shard := storageClassGateShard("backup-sc", "missing-pool-sc") + + c := fake.NewClientBuilder(). + WithScheme(scheme). + WithObjects( + shard, + testPostgresPasswordSecretForShard(shard), + &storagev1.StorageClass{ObjectMeta: metav1.ObjectMeta{Name: "backup-sc"}}, + ). + WithStatusSubresource(&multigresv1alpha1.Shard{}). + Build() + + r := &ShardReconciler{ + Client: c, + Scheme: scheme, + Recorder: record.NewFakeRecorder(100), + APIReader: c, + CreateTopoStore: newMemoryTopoFactory(), + } + + result, err := r.Reconcile(t.Context(), ctrl.Request{ + NamespacedName: client.ObjectKeyFromObject(shard), + }) + if err != nil { + t.Fatalf("expected non-error dependency requeue, got error: %v", err) + } + if result.RequeueAfter != storageClassDependencyRequeue { + t.Fatalf("requeueAfter = %v, want %v", result.RequeueAfter, storageClassDependencyRequeue) + } + + ns := shard.Namespace + if !childExists(t, c, &appsv1.Deployment{}, ns, buildHashedMultiorchName(shard, "zone1")) { + t.Error("Multiorch Deployment is missing: the pool gate moved above the Multiorch block") + } + if !childExists(t, c, &corev1.PersistentVolumeClaim{}, ns, BuildSharedBackupPVCName(shard)) { + t.Error( + "shared backup PVC is missing: a missing pool class must not stop the reconcile " + + "before the backup PVC, whose own StorageClass is present", + ) + } + if childExists(t, c, &corev1.ConfigMap{}, ns, PostgresConfigMapName(shard.Name)) { + t.Error("postgres config ConfigMap was created: the pool gate moved below it") + } + if childExists(t, c, &corev1.Pod{}, ns, BuildPoolPodName(shard, "primary", "zone1", 0)) { + t.Error( + "pool pod was created against a StorageClass that does not exist: " + + "the pool gate no longer precedes the workloads that consume pool storage", + ) + } +} + +// persistStorageClassCondition writes a prior StorageClassValid verdict so a +// test can drive setStorageClassCondition against known persisted state. The +// generation is read back rather than assumed, because the skip compares the +// condition's observedGeneration against the object's. +func persistStorageClassCondition( + t *testing.T, + c client.Client, + key client.ObjectKey, + build func(generation int64) metav1.Condition, +) { + t.Helper() + + var shard multigresv1alpha1.Shard + if err := c.Get(t.Context(), key, &shard); err != nil { + t.Fatalf("read shard: %v", err) + } + shard.Status.Conditions = []metav1.Condition{build(shard.Generation)} + if err := c.Status().Update(t.Context(), &shard); err != nil { + t.Fatalf("seed condition: %v", err) + } +} + +// TestSetStorageClassCondition_RepublishesWhenOneComparedFieldDiffers takes the +// skip-if-unchanged test apart conjunct by conjunct. Dropping any one of them +// widens the skip so it swallows a real change, and a case where several fields +// differ at once cannot see that, so each case here differs from the persisted +// condition in exactly one compared field. +// +// Some of these field combinations are not reachable verdicts: no production +// verdict pairs StorageClassFound with False. The unit under test is +// setStorageClassCondition, whose contract is per-field, and driving it with a +// storageClassCheck directly is what makes one field at a time possible. +func TestSetStorageClassCondition_RepublishesWhenOneComparedFieldDiffers(t *testing.T) { + t.Parallel() + + const foundMessage = "All explicitly configured StorageClasses are present" + settled := storageClassCheck{ + status: metav1.ConditionTrue, + reason: storageClassFoundReason, + message: foundMessage, + } + + cases := []struct { + name string + persisted func(generation int64) metav1.Condition + wantPatch bool + wantReason string + }{ + { + name: "status differs", + persisted: func(generation int64) metav1.Condition { + return metav1.Condition{ + Type: conditionStorageClassValid, + Status: metav1.ConditionFalse, + Reason: storageClassFoundReason, + Message: foundMessage, + ObservedGeneration: generation, + LastTransitionTime: metav1.Now(), + } + }, + wantPatch: true, + }, + { + name: "reason differs", + persisted: func(generation int64) metav1.Condition { + return metav1.Condition{ + Type: conditionStorageClassValid, + Status: metav1.ConditionTrue, + Reason: storageClassNotSpecifiedReason, + Message: foundMessage, + ObservedGeneration: generation, + LastTransitionTime: metav1.Now(), + } + }, + wantPatch: true, + }, + { + name: "message differs", + persisted: func(generation int64) metav1.Condition { + return metav1.Condition{ + Type: conditionStorageClassValid, + Status: metav1.ConditionTrue, + Reason: storageClassFoundReason, + Message: "a message from an older build", + ObservedGeneration: generation, + LastTransitionTime: metav1.Now(), + } + }, + wantPatch: true, + }, + { + name: "observedGeneration differs", + persisted: func(generation int64) metav1.Condition { + return metav1.Condition{ + Type: conditionStorageClassValid, + Status: metav1.ConditionTrue, + Reason: storageClassFoundReason, + Message: foundMessage, + ObservedGeneration: generation - 1, + LastTransitionTime: metav1.Now(), + } + }, + wantPatch: true, + }, + { + name: "nothing differs", + persisted: func(generation int64) metav1.Condition { + return metav1.Condition{ + Type: conditionStorageClassValid, + Status: metav1.ConditionTrue, + Reason: storageClassFoundReason, + Message: foundMessage, + ObservedGeneration: generation, + LastTransitionTime: metav1.Now(), + } + }, + wantPatch: false, + }, + } + + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + t.Parallel() + + scheme := runtime.NewScheme() + _ = multigresv1alpha1.AddToScheme(scheme) + _ = storagev1.AddToScheme(scheme) + + shard := &multigresv1alpha1.Shard{ + ObjectMeta: metav1.ObjectMeta{Name: "test-shard", Namespace: "default"}, + } baseClient := fake.NewClientBuilder(). WithScheme(scheme). - WithObjects(shard, sc). + WithObjects(shard). WithStatusSubresource(&multigresv1alpha1.Shard{}). Build() - var capturedPatchObj client.Object - fakeClient := testutil.NewFakeClientWithFailures(baseClient, &testutil.FailureConfig{ - OnStatusPatch: func(obj client.Object) error { - capturedPatchObj = obj - return nil - }, - }) + key := client.ObjectKeyFromObject(shard) + persistStorageClassCondition(t, baseClient, key, tc.persisted) + patches := 0 + fakeClient := testutil.NewFakeClientWithFailures( + baseClient, + &testutil.FailureConfig{ + OnStatusPatch: func(client.Object) error { + patches++ + return nil + }, + }, + ) r := &ShardReconciler{ Client: fakeClient, Scheme: scheme, Recorder: record.NewFakeRecorder(10), } - if err := r.validatePoolStorageClassDependencies(t.Context(), shard); err != nil { - t.Fatalf("guard: %v", err) + if err := r.setStorageClassCondition(t.Context(), shard, settled); err != nil { + t.Fatalf("setStorageClassCondition: %v", err) } - patchShard, ok := capturedPatchObj.(*multigresv1alpha1.Shard) - if !ok { - t.Fatalf("expected *Shard patch, got %T", capturedPatchObj) + want := 0 + if tc.wantPatch { + want = 1 } - - // Exactly one condition: StorageClassValid. - if len(patchShard.Status.Conditions) != 1 { - t.Fatalf("guard patch must contain exactly 1 condition, got %d: %v", - len(patchShard.Status.Conditions), patchShard.Status.Conditions) - } - scCond := &patchShard.Status.Conditions[0] - if scCond.Type != conditionStorageClassValid { - t.Fatalf("expected %s condition, got %s", conditionStorageClassValid, scCond.Type) + if patches != want { + t.Fatalf("applied %d patches, want %d", patches, want) } - if scCond.Status != metav1.ConditionTrue || scCond.Reason != storageClassFoundReason { - t.Fatalf("unexpected condition: status=%s reason=%s", scCond.Status, scCond.Reason) + if !tc.wantPatch { + return } - // No other status fields should be set in the guard patch. - if patchShard.Status.Phase != "" { - t.Fatalf("guard patch must not set Phase, got %q", patchShard.Status.Phase) - } - if patchShard.Status.Message != "" { - t.Fatalf("guard patch must not set Message, got %q", patchShard.Status.Message) + var got multigresv1alpha1.Shard + if err := baseClient.Get(t.Context(), key, &got); err != nil { + t.Fatalf("read shard: %v", err) } - if patchShard.Status.PodRoles != nil { - t.Fatal("guard patch must not set PodRoles") + cond := findCondition(got.Status.Conditions, conditionStorageClassValid) + if cond == nil { + t.Fatalf("no %s condition", conditionStorageClassValid) } - if patchShard.Status.ReadyReplicas != 0 { - t.Fatalf( - "guard patch must not set ReadyReplicas, got %d", - patchShard.Status.ReadyReplicas, - ) + if cond.Status != settled.status || cond.Reason != settled.reason || + cond.Message != settled.message || cond.ObservedGeneration != got.Generation { + t.Fatalf("published condition does not match the verdict: %+v", *cond) } + }) + } +} + +// TestStorageClassCondition_UpdatesWhenTheVerdictChanges is the other half of +// TestStorageClassCondition_IsStableAcrossReconciles: the skip has to hold a +// settled condition still, and it has to let a changed verdict through. The +// missing StorageClass appears between the two cycles, so status, reason, +// message and lastTransitionTime all have to move with it. +func TestStorageClassCondition_UpdatesWhenTheVerdictChanges(t *testing.T) { + t.Parallel() + + scheme := runtime.NewScheme() + _ = multigresv1alpha1.AddToScheme(scheme) + _ = storagev1.AddToScheme(scheme) + + shard := &multigresv1alpha1.Shard{ + ObjectMeta: metav1.ObjectMeta{Name: "test-shard", Namespace: "default"}, + Spec: multigresv1alpha1.ShardSpec{ + Pools: map[multigresv1alpha1.PoolName]multigresv1alpha1.PoolSpec{ + "primary": { + Storage: multigresv1alpha1.StorageSpec{Size: "10Gi", Class: "appears-later"}, + }, + }, + }, + } + + baseClient := fake.NewClientBuilder(). + WithScheme(scheme). + WithObjects(shard). + WithStatusSubresource(&multigresv1alpha1.Shard{}). + Build() + + patches := 0 + fakeClient := testutil.NewFakeClientWithFailures(baseClient, &testutil.FailureConfig{ + OnStatusPatch: func(client.Object) error { + patches++ + return nil + }, + }) + r := &ShardReconciler{Client: fakeClient, Scheme: scheme, Recorder: record.NewFakeRecorder(50)} + + key := client.ObjectKeyFromObject(shard) + cycle := func() metav1.Condition { + t.Helper() + + check, err := r.validateStorageClassDependencies(t.Context(), shard) + if err != nil { + t.Fatalf("validate: %v", err) + } + if err := r.setStorageClassCondition(t.Context(), shard, check); err != nil { + t.Fatalf("set condition: %v", err) + } + + var got multigresv1alpha1.Shard + if err := baseClient.Get(t.Context(), key, &got); err != nil { + t.Fatalf("read shard: %v", err) + } + cond := findCondition(got.Status.Conditions, conditionStorageClassValid) + if cond == nil { + t.Fatalf("no %s condition", conditionStorageClassValid) + } + return *cond + } + + before := cycle() + if before.Status != metav1.ConditionFalse || before.Reason != storageClassNotFoundReason { + t.Fatalf("first cycle must report the missing class: %+v", before) + } + if patches != 1 { + t.Fatalf("first cycle applied %d patches, want 1", patches) + } + + // Backdated because metav1.Time serialises at second precision and both + // cycles run inside the same second, which would make a rewritten + // lastTransitionTime indistinguishable from a preserved one. + backdated := metav1.NewTime(time.Now().Add(-time.Hour).Truncate(time.Second)) + persistStorageClassCondition(t, baseClient, key, func(generation int64) metav1.Condition { + return metav1.Condition{ + Type: before.Type, + Status: before.Status, + Reason: before.Reason, + Message: before.Message, + ObservedGeneration: generation, + LastTransitionTime: backdated, + } + }) + + if err := baseClient.Create(t.Context(), &storagev1.StorageClass{ + ObjectMeta: metav1.ObjectMeta{Name: "appears-later"}, + }); err != nil { + t.Fatalf("create StorageClass: %v", err) + } + + after := cycle() + if patches != 2 { + t.Fatalf("the changed verdict was skipped: %d patches total", patches) + } + if after.Status != metav1.ConditionTrue { + t.Errorf("status = %s, want %s", after.Status, metav1.ConditionTrue) + } + if after.Reason != storageClassFoundReason { + t.Errorf("reason = %s, want %s", after.Reason, storageClassFoundReason) + } + if after.Message == before.Message { + t.Errorf("message did not move off the missing-class text: %q", after.Message) + } + if want := "All explicitly configured StorageClasses are present"; after.Message != want { + t.Errorf("message = %q, want %q", after.Message, want) + } + if !after.LastTransitionTime.After(backdated.Time) { + t.Errorf( + "lastTransitionTime = %s, want it moved past %s: the condition transitioned", + after.LastTransitionTime, + backdated, + ) + } +} + +// TestStorageClassCondition_PreservesLastTransitionTimeWithoutATransition +// covers the republish an upgrade from the two-writer build performs: the +// persisted condition still carries one of the two old messages, the verdict is +// the same True it always was, so the message has to be rewritten while +// lastTransitionTime stays put, matching meta.SetStatusCondition. +func TestStorageClassCondition_PreservesLastTransitionTimeWithoutATransition(t *testing.T) { + t.Parallel() + + scheme := runtime.NewScheme() + _ = multigresv1alpha1.AddToScheme(scheme) + _ = storagev1.AddToScheme(scheme) + + shard := &multigresv1alpha1.Shard{ + ObjectMeta: metav1.ObjectMeta{Name: "test-shard", Namespace: "default"}, + Spec: multigresv1alpha1.ShardSpec{ + Pools: map[multigresv1alpha1.PoolName]multigresv1alpha1.PoolSpec{ + "primary": {Storage: multigresv1alpha1.StorageSpec{Size: "10Gi"}}, + }, }, - ) + } + + baseClient := fake.NewClientBuilder(). + WithScheme(scheme). + WithObjects(shard). + WithStatusSubresource(&multigresv1alpha1.Shard{}). + Build() + + key := client.ObjectKeyFromObject(shard) + staleMessage := "No explicit pool StorageClass configured; using cluster default" + backdated := metav1.NewTime(time.Now().Add(-time.Hour).Truncate(time.Second)) + persistStorageClassCondition(t, baseClient, key, func(generation int64) metav1.Condition { + return metav1.Condition{ + Type: conditionStorageClassValid, + Status: metav1.ConditionTrue, + Reason: storageClassNotSpecifiedReason, + Message: staleMessage, + ObservedGeneration: generation, + LastTransitionTime: backdated, + } + }) + + patches := 0 + fakeClient := testutil.NewFakeClientWithFailures(baseClient, &testutil.FailureConfig{ + OnStatusPatch: func(client.Object) error { + patches++ + return nil + }, + }) + r := &ShardReconciler{Client: fakeClient, Scheme: scheme, Recorder: record.NewFakeRecorder(50)} + + check, err := r.validateStorageClassDependencies(t.Context(), shard) + if err != nil { + t.Fatalf("validate: %v", err) + } + if err := r.setStorageClassCondition(t.Context(), shard, check); err != nil { + t.Fatalf("set condition: %v", err) + } + if patches != 1 { + t.Fatalf("the stale message was not republished: %d patches", patches) + } + + var got multigresv1alpha1.Shard + if err := baseClient.Get(t.Context(), key, &got); err != nil { + t.Fatalf("read shard: %v", err) + } + cond := findCondition(got.Status.Conditions, conditionStorageClassValid) + if cond == nil { + t.Fatalf("no %s condition", conditionStorageClassValid) + } + if cond.Message == staleMessage { + t.Fatalf("message was not rewritten: %q", cond.Message) + } + if cond.Status != metav1.ConditionTrue { + t.Fatalf("status = %s, want %s", cond.Status, metav1.ConditionTrue) + } + if !cond.LastTransitionTime.Time.Equal(backdated.Time) { + t.Errorf( + "lastTransitionTime = %s, want it preserved at %s: the status did not transition", + cond.LastTransitionTime, + backdated, + ) + } } diff --git a/test/suite/case.go b/test/suite/case.go new file mode 100644 index 00000000..f43e5091 --- /dev/null +++ b/test/suite/case.go @@ -0,0 +1,68 @@ +package suite + +import ( + "testing" + + "github.com/multigres/testkit/ctrltest" +) + +// C is this operator's test context: the generic harness handle from +// ctrltest, plus the multigres vocabulary. +// +// A local type because Go cannot add methods to another package's, which is +// the point rather than a workaround. pkg/ctrltest is meant to be copied into +// other operators, so it carries assertions and harness pointers and knows +// nothing about shards or poolers; each consumer wraps it and hangs its own +// domain on the same receiver. Everything ctrltest offers is promoted, so +// c.NoError and c.WaitForClusterHealthy read alike at the call site. +type C struct { + *ctrltest.C +} + +// newCase opens a test context on its own namespace. +// +// Every test in this package should start with one. It allocates the +// namespace, activates the reconcile gate for it, and registers the failure +// dump, so a failing test prints the interleaved op log and reconcile records +// rather than only the assertion message. +func newCase(t *testing.T) *C { + t.Helper() + return &C{C: Suite.Case(t)} +} + +// newBareCase opens a test context with no namespace, for the tests in this +// package that are pure logic and never touch the cluster. +// +// identity_test.go is all of them: MembersOf and ShardPVCOf take objects and +// return answers. newCase would allocate a real namespace against envtest and +// register it at the reconcile gate for each one, which buys nothing. +func newBareCase(t *testing.T) *C { + t.Helper() + return &C{C: ctrltest.Bare(t)} +} + +// Sub binds this case to a subtest's T while keeping its namespace. Use it +// for a t.Run that asserts about objects the parent test created; use newCase +// for a subtest that wants a namespace of its own. +// +// It shadows the embedded ctrltest.C.Sub so that one name always hands back +// this package's C, with the multigres vocabulary still on it. Without the +// shadow a subtest would silently drop to the generic type and lose every +// method below. +func (c *C) Sub(t *testing.T) *C { + t.Helper() + return &C{C: c.C.Sub(t)} +} + +// Check returns a C whose assertions report and continue rather than abort, +// shadowed for the same reason as Sub: without it c.Check() hands back a +// *ctrltest.C and a collecting assertion silently loses every method below. +func (c *C) Check() *C { + return &C{C: c.C.Check()} +} + +// Assert at compile time that both shadows hand back this package's type. The +// regression they guard against is silent: dropping to *ctrltest.C still +// compiles at every existing call site, and only stops compiling once someone +// chains a multigres method off one of them. +var _ = func(c *C) (*C, *C) { return c.Check(), c.Sub(nil) } diff --git a/test/suite/fakes.go b/test/suite/fakes.go new file mode 100644 index 00000000..6b6691ff --- /dev/null +++ b/test/suite/fakes.go @@ -0,0 +1,265 @@ +package suite + +import ( + "context" + "fmt" + "sort" + "strings" + "sync" + "time" + + "github.com/multigres/multigres/go/common/rpcclient" + "github.com/multigres/multigres/go/common/topoclient" + "github.com/multigres/multigres/go/common/topoclient/memorytopo" + cm "github.com/multigres/multigres/go/pb/clustermetadata" + md "github.com/multigres/multigres/go/pb/multipoolermanagerdata" + corev1 "k8s.io/api/core/v1" + "sigs.k8s.io/controller-runtime/pkg/client" + + multigresv1alpha1 "github.com/multigres/multigres-operator/api/v1alpha1" + shardcontroller "github.com/multigres/multigres-operator/pkg/resource-handler/controller/shard" + "github.com/multigres/multigres-operator/pkg/util/metadata" +) + +// defaultSimCell is the cell every fixture uses. The topo store needs its cells +// declared up front, so a test using a different cell name needs this widened. +const defaultSimCell = "zone-a" + +// topoRegistry hands out one in-memory topology store per namespace. +// +// Per namespace rather than one shared store, because namespace-per-test is the +// suite's isolation boundary and a single store would let one test's cluster +// see another's multipoolers. Both CreateTopoStore seams resolve through here: +// the shard's carries the object, the cluster's carries only a DNS address, so +// that one recovers the namespace by parsing it. +type topoRegistry struct { + ctx context.Context + + mu sync.Mutex + stores map[string]topoclient.Store + facts map[string]*memorytopo.Factory +} + +func newTopoRegistry(ctx context.Context) *topoRegistry { + return &topoRegistry{ + ctx: ctx, + stores: map[string]topoclient.Store{}, + facts: map[string]*memorytopo.Factory{}, + } +} + +// Store returns the namespace's store, creating it on first use. +func (r *topoRegistry) Store(ns string) topoclient.Store { + r.mu.Lock() + defer r.mu.Unlock() + if s, ok := r.stores[ns]; ok { + return s + } + store, factory := memorytopo.NewServerAndFactory(r.ctx, defaultSimCell) + r.stores[ns] = store + r.facts[ns] = factory + return store +} + +func (r *topoRegistry) client(ns string) (topoclient.Store, error) { + r.Store(ns) + r.mu.Lock() + factory := r.facts[ns] + r.mu.Unlock() + return topoclient.NewWithFactory( + factory, "", []string{""}, topoclient.NewDefaultTopoConfig(), + ), nil +} + +// ForShard is ShardReconciler.CreateTopoStore. +func (r *topoRegistry) ForShard(shard *Shard) (topoclient.Store, error) { + return r.client(shard.Namespace) +} + +// ForClusterRef is MultigresClusterReconciler.CreateTopoStore. The ref carries +// no namespace, only the Service address the cluster controller built as +// "-global-topo..svc:2379", so the namespace comes back out +// of the address. Left unstubbed, this seam dials a real etcd and the cluster +// never reaches TopologyReady. +func (r *topoRegistry) ForClusterRef( + ref multigresv1alpha1.GlobalTopoServerRef, +) (topoclient.Store, error) { + ns, err := namespaceFromTopoAddress(ref.Address) + if err != nil { + return nil, err + } + return r.client(ns) +} + +func namespaceFromTopoAddress(address string) (string, error) { + parts := strings.Split(address, ".") + if len(parts) < 2 || parts[1] == "" { + return "", fmt.Errorf("cannot derive namespace from topo address %q", address) + } + return parts[1], nil +} + +// poolerSim is the data plane: it registers a multipooler per pool pod in that +// namespace's topology store and answers a healthy Status RPC for each. +// +// Without it the shard controller stalls at PostureConsistent=Unknown +// (AwaitingPoolerRegistration) and never reaches a terminal state. +// +// Deliberately generous: every pod is healthy, always, and the lowest-numbered +// pod is primary. It models no ordering, no failure, and no latency, so a test +// that needs any of those needs a better fake than this one. +type poolerSim struct { + c client.Client + rpc *rpcclient.FakeClient + topo *topoRegistry + interval time.Duration + + mu sync.Mutex + registered map[string]bool + held map[string]bool +} + +// HoldRegistrations stops this fake registering any *new* pooler in ns until +// the returned function is called. Poolers already registered keep answering. +// +// It exists to make a race deterministic instead of sampled. The defect it was +// built for needs a shard to converge having seen fewer poolers than pods, +// which happens on its own only when registration loses a race against the +// last reconcile: about half the time, measured. A test that waits for that by +// chance detects a regression about half the time too, which is what the +// twelve-attempt statistical pin it replaced was paying for. +// +// Holding lets a test construct the precondition on purpose: hold, scale up, +// wait until the shard has demonstrably converged short, then release. One +// attempt, and the regression either survives the release or it does not. +func (p *poolerSim) HoldRegistrations(ns string) func() { + p.mu.Lock() + if p.held == nil { + p.held = map[string]bool{} + } + p.held[ns] = true + p.mu.Unlock() + + var once sync.Once + return func() { + once.Do(func() { + p.mu.Lock() + delete(p.held, ns) + p.mu.Unlock() + }) + } +} + +func (p *poolerSim) run(ctx context.Context) { + p.registered = map[string]bool{} + t := time.NewTicker(p.interval) + defer t.Stop() + for { + select { + case <-ctx.Done(): + return + case <-t.C: + p.tick(ctx) + } + } +} + +func (p *poolerSim) tick(ctx context.Context) { + pods := &corev1.PodList{} + if err := p.c.List(ctx, pods, + client.MatchingLabels{metadata.LabelAppComponent: shardcontroller.PoolComponentName}, + ); err != nil { + return + } + byNamespace := map[string][]*corev1.Pod{} + for i := range pods.Items { + pod := &pods.Items[i] + if !pod.DeletionTimestamp.IsZero() { + continue + } + byNamespace[pod.Namespace] = append(byNamespace[pod.Namespace], pod) + } + for ns, group := range byNamespace { + p.tickNamespace(ctx, ns, group) + } +} + +func (p *poolerSim) tickNamespace(ctx context.Context, ns string, pods []*corev1.Pod) { + sort.Slice(pods, func(i, j int) bool { return pods[i].Name < pods[j].Name }) + + ids := make([]*cm.ID, 0, len(pods)) + for _, pod := range pods { + ids = append(ids, &cm.ID{ + Cell: pod.Labels[metadata.LabelMultigresCell], + Name: pod.Name, + }) + } + if len(ids) == 0 { + return + } + leader := ids[0] + rule := &cm.ShardRule{ + RuleNumber: &cm.RuleNumber{CoordinatorTerm: 2}, + LeaderId: leader, + CohortMembers: ids, + DurabilityPolicy: topoclient.AtLeastN(1), + } + store := p.topo.Store(ns) + + for i, pod := range pods { + id := ids[i] + role := cm.RoutingRole_ROUTING_ROLE_REPLICA + resp := &md.StatusResponse{ + Status: &md.Status{ + IsInitialized: true, + PostgresReady: true, + PostgresStatus: md.PostgresStatus_POSTGRES_STATUS_STANDBY, + }, + AvailabilityStatus: &cm.AvailabilityStatus{ + CohortEligibilityStatus: &cm.CohortEligibilityStatus{ + Signal: cm.CohortEligibilitySignal_COHORT_ELIGIBILITY_SIGNAL_ELIGIBLE, + }, + }, + ConsensusStatus: &cm.ConsensusStatus{ + Id: id, + CurrentPosition: &cm.PoolerPosition{Position: &cm.RulePosition{Decision: rule}}, + }, + } + if id.Name == leader.Name { + role = cm.RoutingRole_ROUTING_ROLE_PRIMARY + resp.Status.PostgresStatus = md.PostgresStatus_POSTGRES_STATUS_PRIMARY + resp.Status.PrimaryStatus = &md.PrimaryStatus{ + Ready: true, + ConnectedFollowers: ids[1:], + } + } + p.rpc.SetStatusResponse(topoclient.ComponentIDString(id), resp) + + key := ns + "/" + pod.Name + p.mu.Lock() + already := p.registered[key] + heldBack := p.held[ns] + p.mu.Unlock() + // A held namespace still gets its Status RPC answered above, so pods + // already registered stay healthy and the shard keeps converging. Only + // the new registration waits, which is the whole point. + if already || heldBack { + continue + } + pooler := &cm.Multipooler{ + Id: id, + Hostname: pod.Name, + ShardKey: &cm.ShardKey{ + Database: pod.Labels[metadata.LabelMultigresDatabase], + TableGroup: pod.Labels[metadata.LabelMultigresTableGroup], + Shard: pod.Labels[metadata.LabelMultigresShard], + }, + RoutingState: &cm.RoutingState{Role: role}, + } + if err := store.RegisterMultipooler(ctx, pooler, true); err == nil { + p.mu.Lock() + p.registered[key] = true + p.mu.Unlock() + } + } +} diff --git a/test/suite/fixture.go b/test/suite/fixture.go new file mode 100644 index 00000000..440b017a --- /dev/null +++ b/test/suite/fixture.go @@ -0,0 +1,113 @@ +package suite + +import ( + "fmt" + "time" + + corev1 "k8s.io/api/core/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "sigs.k8s.io/controller-runtime/pkg/client" + + multigresv1alpha1 "github.com/multigres/multigres-operator/api/v1alpha1" +) + +// The name of a Secret object, not a credential. gosec matches on the string +// value rather than on how it is used, so the suppression has to be explicit. +// +//nolint:gosec // G101: this is a Secret's name; the password itself is set below +const adminSecretName = "multigres-admin-password" + +// MinimalCluster creates the equivalent of config/samples/minimal.yaml in ns, +// along with the password Secret it references, and returns it. +// +// Built in Go rather than read from the sample, because the e2e loader +// (framework.MustLoadCluster) is behind //go:build e2e and this package +// deliberately has no build tag. +// +// The Secret is intentionally unlabelled, matching what a user would create. +// Under the production cache config that makes it invisible to the cached +// client, so this fixture also exercises why the reconcilers hold an APIReader. +func (c *C) MinimalCluster(name string) *MultigresCluster { + c.Helper() + return c.newCluster(name) +} + +// newCluster creates the admin Secret every cluster in this package +// references, then the cluster itself, applying any spec adjustments in +// between. +// +// The four fixtures here were identical for twenty lines and diverged only +// at Spec.Databases, which is the argument for this existing: the cost of +// the copy grows with the test count rather than being a debt that stays +// fixed, and the next person writing a scenario test copies whichever +// fixture they happened to read. +// +// The adjustment is a callback rather than a returned unsaved object so +// that creating the cluster cannot be forgotten. A fixture that built an +// object and never persisted it would leave the test waiting on a +// convergence that had no reason to start. +// +// The Secret is intentionally unlabelled, matching what a user would create. +// Under the production cache config that makes it invisible to the cached +// client, so this fixture also exercises why the reconcilers hold an +// APIReader. +func (c *C) newCluster(name string, with ...func(*MultigresClusterSpec)) *MultigresCluster { + c.Helper() + + secret := &corev1.Secret{ + ObjectMeta: metav1.ObjectMeta{Name: adminSecretName, Namespace: c.NS}, + StringData: map[string]string{"password": "postgres"}, + } + c.NoError(c.Create(secret), "create password secret") + + cluster := &MultigresCluster{ + ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: c.NS}, + Spec: MultigresClusterSpec{ + PostgresPasswordSecretRef: PostgresPasswordSecretRef{ + Name: adminSecretName, + Key: "password", + }, + PVCDeletionPolicy: &PVCDeletionPolicy{ + WhenDeleted: multigresv1alpha1.DeletePVCRetentionPolicy, + WhenScaled: multigresv1alpha1.DeletePVCRetentionPolicy, + }, + Cells: []CellConfig{ + {Name: defaultSimCell, ZoneID: "us-central1-a"}, + }, + }, + } + for _, adjust := range with { + adjust(&cluster.Spec) + } + c.NoError(c.Create(cluster), "create MultigresCluster") + return cluster +} + +// WaitForClusterHealthy blocks until the cluster reports PhaseHealthy. +// +// One method rather than the seven copies pass 1 left behind: two named +// helpers (waitForClusterHealthy in the thrash file and waitForHealthy in the +// transitions file, byte-identical to each other) and five inlined +// Eventually blocks. They were hard to see as duplicates while each was +// wrapped in its own t-and-namespace threading. +// +// It is the convergence check nearly every scenario test starts from, which +// the old comment on one of the copies said out loud without anyone acting +// on it. +// +// 30 seconds because that is what all seven used. It is a convergence wait +// for a whole cluster under five controllers, not a single object read, so it +// is deliberately far longer than any assertion budget. +func (c *C) WaitForClusterHealthy(cluster *MultigresCluster) { + c.Helper() + c.Eventually(30*time.Second, "cluster to report Healthy", func() error { + got := &MultigresCluster{} + if err := c.Get(client.ObjectKeyFromObject(cluster), got); err != nil { + return err + } + if got.Status.Phase != multigresv1alpha1.PhaseHealthy { + return fmt.Errorf("phase is %q", got.Status.Phase) + } + return nil + }) +} diff --git a/test/suite/golden_test.go b/test/suite/golden_test.go new file mode 100644 index 00000000..f68746ed --- /dev/null +++ b/test/suite/golden_test.go @@ -0,0 +1,67 @@ +package suite + +import ( + "testing" + + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/runtime" + + multigresv1alpha1 "github.com/multigres/multigres-operator/api/v1alpha1" + cellcontroller "github.com/multigres/multigres-operator/pkg/resource-handler/controller/cell" + "github.com/multigres/testkit/golden" +) + +// TestGoldenMultigatewayDeployment pilots golden.AssertYAML against +// BuildMultigatewayDeployment, called directly rather than through the +// running cell controller: the builder is exported and reachable from this +// package, so the pilot exercises a pure function with no generated fields to +// strip, rather than an object fetched from envtest. +// +// It builds its own scheme rather than reaching for Suite.Scheme: the +// builder call needs nothing envtest boots, and coupling to Suite would make +// this test pay for the whole suite's startup for no benefit. +// +// The fixture pins both Spec.Observability (via OTEL_EXPORTER_OTLP_ENDPOINT) +// and Spec.Images.Multigateway to fixed values: a golden over a builder's +// output must not depend on values that routine maintenance changes. +func TestGoldenMultigatewayDeployment(t *testing.T) { + // BuildMultigatewayDeployment resolves OTEL settings from the process + // environment when Spec.Observability is nil (as it is below), so the + // golden file is only stable once that read is pinned. "disabled" is the + // builder's own sentinel for suppressing every OTEL var. + t.Setenv("OTEL_EXPORTER_OTLP_ENDPOINT", "disabled") + c := newBareCase(t) + + scheme := runtime.NewScheme() + c.NoError(multigresv1alpha1.AddToScheme(scheme), "add to scheme") + + cell := &multigresv1alpha1.Cell{ + ObjectMeta: metav1.ObjectMeta{ + Name: "golden-cell", + Namespace: "default", + UID: "golden-cell-uid", + Labels: map[string]string{"multigres.com/cluster": "golden-cluster"}, + }, + Spec: multigresv1alpha1.CellSpec{ + Name: "zone1", + GlobalTopoServer: multigresv1alpha1.GlobalTopoServerRef{ + Address: "global-topo:2379", + RootPath: "/multigres/global", + Implementation: "etcd", + }, + LogLevels: multigresv1alpha1.ComponentLogLevels{ + Multigateway: "info", + }, + Images: multigresv1alpha1.CellImages{ + Multigateway: multigresv1alpha1.ImageRef( + "ghcr.io/multigres/multigres:golden-fixture", + ), + }, + }, + } + + got, err := cellcontroller.BuildMultigatewayDeployment(cell, scheme) + c.NoError(err, "BuildMultigatewayDeployment") + + golden.AssertYAML(t, got, "testdata/multigateway-deployment.golden.yaml") +} diff --git a/test/suite/identity.go b/test/suite/identity.go new file mode 100644 index 00000000..5112c0c6 --- /dev/null +++ b/test/suite/identity.go @@ -0,0 +1,98 @@ +package suite + +import ( + "context" + "fmt" + "sort" + "strings" + + "sigs.k8s.io/controller-runtime/pkg/client" + + shardcontroller "github.com/multigres/multigres-operator/pkg/resource-handler/controller/shard" + "github.com/multigres/testkit/ctrltest" +) + +// Members is a snapshot of a Shard's pod roles, taken at the moment of a call. +type Members struct { + Primary string // pod name with role PRIMARY + Replicas []string // pod names with role REPLICA, sorted + Quarantined []string // pod names with role QUARANTINED, sorted +} + +// MembersOf reads the Shard's status.podRoles and classifies every pod. +// Error-returning because callers include KnownDefect bodies. +// +// The role set here is {PRIMARY, REPLICA, QUARANTINED}, not the +// {PRIMARY, REPLICA, DRAINED} the CRD doc comment on PodRoles still claims +// (api/v1alpha1/shard_types.go:329, stale since e3677f0). PodRoles has one +// writer, reconcile_data_plane.go, fed entirely by GetPoolerStatus in +// pkg/data-handler/topo/pooler.go, where roleName is one of exactly those +// three literals. DRAINED cannot be produced by the real operator today, so +// it falls to the default arm below like any other unrecognized value. +func MembersOf(ctx context.Context, c client.Client, key client.ObjectKey) (Members, error) { + shard := &Shard{} + if err := c.Get(ctx, key, shard); err != nil { + return Members{}, fmt.Errorf("get shard %s: %w", key, err) + } + + roles := shard.Status.PodRoles + if len(roles) == 0 { + return Members{}, fmt.Errorf("shard %s: status.podRoles is empty", key) + } + + var primaries, replicas, quarantined []string + for pod, role := range roles { + switch role { + case "PRIMARY": + primaries = append(primaries, pod) + case "REPLICA": + replicas = append(replicas, pod) + case "QUARANTINED": + quarantined = append(quarantined, pod) + default: + return Members{}, fmt.Errorf( + "shard %s: pod %s has unrecognized role %q", key, pod, role, + ) + } + } + + switch len(primaries) { + case 0: + return Members{}, fmt.Errorf("shard %s: no pod has role PRIMARY", key) + case 1: + default: + sort.Strings(primaries) + return Members{}, fmt.Errorf( + "shard %s: more than one pod has role PRIMARY: %s", + key, strings.Join(primaries, ", "), + ) + } + + sort.Strings(replicas) + sort.Strings(quarantined) + + return Members{ + Primary: primaries[0], + Replicas: replicas, + Quarantined: quarantined, + }, nil +} + +// ShardPVCOf returns the PVC bound by the named pool pod's data volume. +// +// Pool pods only. The toposerver controller declares its own +// DataVolumeName ("data", in its statefulset builder), so this returns a +// not-found error for a toposerver pod rather than that pod's data PVC. A +// loud error rather than a wrong answer, but the restriction is not +// visible in the signature. +// +// A pool pod can carry a second PVC-backed volume (the filesystem backup +// volume, when shard.Spec.Backup.Type is Filesystem; see +// buildSharedBackupVolume in the shard controller), which is why the +// underlying lookup selects by volume name rather than requiring the pod +// to have exactly one PVC volume. +func ShardPVCOf( + ctx context.Context, c client.Client, ns, pod string, +) (string, error) { + return ctrltest.PVCOf(ctx, c, ns, pod, shardcontroller.DataVolumeName) +} diff --git a/test/suite/identity_test.go b/test/suite/identity_test.go new file mode 100644 index 00000000..3be6874b --- /dev/null +++ b/test/suite/identity_test.go @@ -0,0 +1,239 @@ +package suite + +import ( + "testing" + + corev1 "k8s.io/api/core/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/runtime" + "sigs.k8s.io/controller-runtime/pkg/client" + "sigs.k8s.io/controller-runtime/pkg/client/fake" + + multigresv1alpha1 "github.com/multigres/multigres-operator/api/v1alpha1" + shardcontroller "github.com/multigres/multigres-operator/pkg/resource-handler/controller/shard" +) + +// identityFakeClient builds a fake client that knows about both the +// operator's CRDs and core types, since a Shard's status and a Pod's +// volumes both need to round-trip through it. +func identityFakeClient(objs ...client.Object) client.Client { + scheme := runtime.NewScheme() + if err := multigresv1alpha1.AddToScheme(scheme); err != nil { + panic(err) + } + if err := corev1.AddToScheme(scheme); err != nil { + panic(err) + } + return fake.NewClientBuilder(). + WithScheme(scheme). + WithObjects(objs...). + WithStatusSubresource(&Shard{}). + Build() +} + +func shardWithRoles(ns, name string, roles map[string]string) *Shard { + return &Shard{ + ObjectMeta: metav1.ObjectMeta{Namespace: ns, Name: name}, + Status: multigresv1alpha1.ShardStatus{ + PodRoles: roles, + }, + } +} + +func TestIdentityMembersOfClassifiesAndSortsReplicas(t *testing.T) { + c := newBareCase(t) + shard := shardWithRoles("ns1", "shard1", map[string]string{ + "pod-primary": "PRIMARY", + "pod-replica-b": "REPLICA", + "pod-replica-a": "REPLICA", + }) + fc := identityFakeClient(shard) + + members, err := MembersOf(c.Context(), fc, client.ObjectKeyFromObject(shard)) + c.NoError(err, "MembersOf") + + c.Check().Eq("pod-primary", members.Primary, "Primary") + want := []string{"pod-replica-a", "pod-replica-b"} + gotSorted := len(members.Replicas) == len(want) && + members.Replicas[0] == want[0] && members.Replicas[1] == want[1] + c.Check().True(gotSorted, "Replicas = %v, want %v (sorted)", members.Replicas, want) +} + +func TestIdentityMembersOfSeparatesQuarantinedFromReplicas(t *testing.T) { + c := newBareCase(t) + shard := shardWithRoles("ns1", "shard1", map[string]string{ + "pod-primary": "PRIMARY", + "pod-replica": "REPLICA", + "pod-quarantined": "QUARANTINED", + }) + fc := identityFakeClient(shard) + + members, err := MembersOf(c.Context(), fc, client.ObjectKeyFromObject(shard)) + c.NoError(err, "MembersOf") + + c.Check().EqDiff([]string{"pod-replica"}, members.Replicas, "Replicas") + c.Check().EqDiff([]string{"pod-quarantined"}, members.Quarantined, "Quarantined") + c.NotContains(members.Replicas, "pod-quarantined", "quarantined pod leaked into Replicas") +} + +// TestIdentityMembersOfUnrecognizedRoleErrors pins the guard that catches any +// role value outside {PRIMARY, REPLICA, QUARANTINED}, the set the operator's +// single writer (pkg/data-handler/topo/pooler.go) can actually produce. DRAINED +// is deliberately used as the unknown value here: the CRD doc comment on +// PodRoles still lists it, but commit e3677f0 removed it from the writer, so +// it is exactly the stale value a careless "known roles" list would still +// accept. +func TestIdentityMembersOfUnrecognizedRoleErrors(t *testing.T) { + c := newBareCase(t) + shard := shardWithRoles("ns1", "shard1", map[string]string{ + "pod-primary": "PRIMARY", + "pod-drained": "DRAINED", + }) + fc := identityFakeClient(shard) + + _, err := MembersOf(c.Context(), fc, client.ObjectKeyFromObject(shard)) + c.Error(err, "want an error for an unrecognized role") + + c.Check(). + ErrorContains(err, "unrecognized role", "want the error to call DRAINED an unrecognized role") + c.Check().ErrorContains(err, "DRAINED", "want the error to name DRAINED specifically") +} + +// TestIdentityMembersOfErrorCases pins the three distinct error cases the +// brief calls out. An absent primary and a duplicated primary are both bugs +// a test should be able to pin, but they are different bugs, so a test that +// only checked "err != nil" could not tell them apart. +func TestIdentityMembersOfErrorCases(t *testing.T) { + c := newBareCase(t) + cases := []struct { + name string + roles map[string]string + wantErrs []string + }{ + { + name: "empty PodRoles", + roles: map[string]string{}, + wantErrs: []string{"empty"}, + }, + { + name: "no primary", + roles: map[string]string{ + "pod-a": "REPLICA", + "pod-b": "QUARANTINED", + }, + wantErrs: []string{"no pod has role PRIMARY"}, + }, + { + name: "two primaries", + roles: map[string]string{ + "pod-a": "PRIMARY", + "pod-b": "PRIMARY", + }, + wantErrs: []string{"more than one pod has role PRIMARY", "pod-a", "pod-b"}, + }, + } + + // Each case's error must be distinguishable from the other two, so this + // collects every message actually produced and cross-checks that no + // case's message satisfies another case's expectation. + messages := make(map[string]string, len(cases)) + + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + c := newBareCase(t) + shard := shardWithRoles("ns1", "shard1", tc.roles) + fc := identityFakeClient(shard) + + _, err := MembersOf(c.Context(), fc, client.ObjectKeyFromObject(shard)) + c.Error(err, "want an error for %s", tc.name) + for _, want := range tc.wantErrs { + c.Check().ErrorContains(err, want) + } + messages[tc.name] = err.Error() + }) + } + + c.Check().NotEq(messages["empty PodRoles"], messages["no primary"], + "empty PodRoles and no-primary produced the same error message") + c.Check().NotEq(messages["no primary"], messages["two primaries"], + "no-primary and two-primaries produced the same error message") + c.Check().NotEq(messages["empty PodRoles"], messages["two primaries"], + "empty PodRoles and two-primaries produced the same error message") +} + +// identityPVCVolume is a thin copy of relations_test.go's pvcVolume. Kept +// separate rather than shared, since after Task 2.1 the original lives in +// another package. +func identityPVCVolume(name, claim string) corev1.Volume { + return corev1.Volume{ + Name: name, + VolumeSource: corev1.VolumeSource{ + PersistentVolumeClaim: &corev1.PersistentVolumeClaimVolumeSource{ + ClaimName: claim, + }, + }, + } +} + +// identityPodWithVolumes is a thin copy of relations_test.go's +// podWithVolumes. Kept separate rather than shared, since after Task 2.1 the +// original lives in another package. +func identityPodWithVolumes(ns, name string, volumes ...corev1.Volume) *corev1.Pod { + return &corev1.Pod{ + ObjectMeta: metav1.ObjectMeta{Namespace: ns, Name: name}, + Spec: corev1.PodSpec{ + Containers: []corev1.Container{{Name: "c", Image: "busybox"}}, + Volumes: volumes, + }, + } +} + +// TestIdentityShardPVCOfUsesTheShardDataVolumeName pins which volume name +// the wrapper supplies. Nothing else does: PVCOf's own tests pass a +// literal, so a wrapper handing over the wrong constant is invisible. +func TestIdentityShardPVCOfUsesTheShardDataVolumeName(t *testing.T) { + c := newBareCase(t) + p := identityPodWithVolumes( + "ns1", "pod-a", + identityPVCVolume(shardcontroller.DataVolumeName, "pod-a-data"), + identityPVCVolume("backup-data", "pod-a-backup"), + ) + fc := identityFakeClient(p) + + claim, err := ShardPVCOf(c.Context(), fc, "ns1", "pod-a") + c.NoError(err, "ShardPVCOf") + c.Check().Eq("pod-a-data", claim) +} + +// TestIdentityMembersOfSortsEnoughReplicasToCatchMapOrder pins the sorts in +// MembersOf. PodRoles is a Go map and Go randomises map iteration, so two +// replicas would agree with insertion order half the time and the assertion +// would be a coin flip. Five in reverse order leaves a 1-in-120 chance of +// passing against an unsorted implementation. +func TestIdentityMembersOfSortsEnoughReplicasToCatchMapOrder(t *testing.T) { + c := newBareCase(t) + shard := &Shard{ + ObjectMeta: metav1.ObjectMeta{Name: "shard-0", Namespace: "ns"}, + Status: multigresv1alpha1.ShardStatus{PodRoles: map[string]string{ + "pool-primary": "PRIMARY", + "pool-e": "REPLICA", + "pool-d": "REPLICA", + "pool-c": "REPLICA", + "pool-b": "REPLICA", + "pool-a": "REPLICA", + "quar-e": "QUARANTINED", + "quar-d": "QUARANTINED", + "quar-c": "QUARANTINED", + "quar-b": "QUARANTINED", + "quar-a": "QUARANTINED", + }}, + } + fc := identityFakeClient(shard) + + got, err := MembersOf(c.Context(), fc, client.ObjectKeyFromObject(shard)) + c.NoError(err, "MembersOf") + wantReplicas := []string{"pool-a", "pool-b", "pool-c", "pool-d", "pool-e"} + wantQuarantined := []string{"quar-a", "quar-b", "quar-c", "quar-d", "quar-e"} + c.Check().EqDiff(wantReplicas, got.Replicas, "Replicas") + c.Check().EqDiff(wantQuarantined, got.Quarantined, "Quarantined") +} diff --git a/test/suite/main_test.go b/test/suite/main_test.go new file mode 100644 index 00000000..214d750e --- /dev/null +++ b/test/suite/main_test.go @@ -0,0 +1,66 @@ +package suite + +import ( + "fmt" + "os" + "runtime/pprof" + "testing" + + "go.uber.org/goleak" +) + +// TestMain boots the suite once, runs the package, tears it down, and only then +// checks for leaked goroutines. +// +// Deliberately not goleak.VerifyTestMain: that calls m.Run() and checks +// immediately afterwards, with no hook in between. Since envtest and the +// manager live for the whole package rather than for one test, the check would +// run while both were still up and report the entire manager as leaked. +// +// The ignore list is empty on purpose. A clean start/stop leaks nothing +// measurable (verified across 16 cycles before this suite existed), so every +// future entry should be justified against a real stack trace rather than +// pre-loaded against suspects. A pre-loaded ignore is a permanent hole. +func TestMain(m *testing.M) { + s, teardown, err := Boot() + if err != nil { + fmt.Fprintf(os.Stderr, "suite boot failed: %v\n", err) + os.Exit(1) + } + Suite = s + + code := m.Run() + + if err := teardown(); err != nil { + fmt.Fprintf(os.Stderr, "suite teardown failed: %v\n", err) + if code == 0 { + code = 1 + } + } + + // Only leak-check a passing run: a failed test may have left its own + // goroutines behind, and reporting those on top of a real failure buries + // the real failure. + if code == 0 { + if err := goleak.Find(); err != nil { + fmt.Fprintf(os.Stderr, "goroutine leak after suite teardown: %v\n", err) + dumpGoroutineLeakProfile() + code = 1 + } + } + os.Exit(code) +} + +// dumpGoroutineLeakProfile writes the runtime's own leak profile when the +// toolchain has one. It returns nil before Go 1.27, so this is a no-op today +// and becomes a diagnostic on the next toolchain bump, with no build tag. +// +// Count() does not run the leak detector but WriteTo() does, so the profile has +// to be driven through WriteTo and the count read afterwards, never before. +func dumpGoroutineLeakProfile() { + p := pprof.Lookup("goroutineleak") + if p == nil { + return + } + _ = p.WriteTo(os.Stderr, 1) +} diff --git a/test/suite/scenario_deletion_test.go b/test/suite/scenario_deletion_test.go new file mode 100644 index 00000000..2d2a872d --- /dev/null +++ b/test/suite/scenario_deletion_test.go @@ -0,0 +1,295 @@ +package suite + +import ( + "strings" + "testing" + "time" + + apierrors "k8s.io/apimachinery/pkg/api/errors" + "k8s.io/utils/ptr" + "sigs.k8s.io/controller-runtime/pkg/client" + + multigresv1alpha1 "github.com/multigres/multigres-operator/api/v1alpha1" + multigresclustercontroller "github.com/multigres/multigres-operator/pkg/cluster-handler/controller/multigrescluster" + tablegroupcontroller "github.com/multigres/multigres-operator/pkg/cluster-handler/controller/tablegroup" + "github.com/multigres/testkit/ctrltest" +) + +// TestReadyForDeletionProtocol asserts the three-hop ReadyForDeletion protocol +// found in pkg/resource-handler/controller/shard/reconcile_deletion.go, +// pkg/cluster-handler/controller/tablegroup/tablegroup_controller.go and +// pkg/cluster-handler/controller/multigrescluster/reconcile_databases.go: a +// Shard sets multigresv1alpha1.ConditionReadyForDeletion on itself once every +// pool pod has drained, its parent TableGroup sets the same condition on +// itself once every child Shard has, and MultigresCluster deletes a TableGroup +// only once that TableGroup reports it. +// +// This is orphan pruning, not whole-cluster teardown. Deleting a +// MultigresCluster goes through MultigresClusterReconciler.handleDeletion, +// which lists and Deletes its Cells and TableGroups directly and never +// consults ConditionReadyForDeletion at all; that path was added in +// da7d639177b0 as a narrower fix scoped explicitly to "orphan pruning" (its +// own commit message), for the case where a TableGroup or Cell falls out of a +// still-live cluster's spec and must drain before it is safe to remove. A +// whole-cluster delete has nothing left to protect by draining, so it tears +// down directly instead. The three-hop protocol is therefore reachable only +// through that narrower path, which this test drives directly: it builds an +// orphan TableGroup (one MultigresCluster.Spec.Databases entry will never +// name) with multigresclustercontroller.BuildTableGroup, the same builder +// production code uses, so multigrescluster's reconcileDatabases treats it +// exactly as it would treat a TableGroup a user just removed from spec. +func TestReadyForDeletionProtocol(t *testing.T) { + c := newCase(t) + ns := c.NS + cluster := c.MinimalCluster("scenario-del") + + c.WaitForClusterHealthy(cluster) + + // Copy the real TableGroup's resolved GlobalTopoServer ref and component + // Images rather than re-deriving them: both are resolved in-memory once + // per MultigresCluster reconcile (globalTopoRef by the unexported + // globalTopoRef method, Images by resolveImages) and never written back to + // MultigresCluster.Spec, so the cluster object this test already holds + // still has them blank. Re-deriving either by hand risks building an + // orphan that fails validation or that the topology store does not + // recognize, for reasons unrelated to what this test asserts. + realTGs := &TableGroupList{} + c.NoError(c.List(realTGs, + client.MatchingLabels{"multigres.com/cluster": cluster.Name}), + "list tablegroups") + c.Len(realTGs.Items, 1, + "want exactly one TableGroup before introducing an orphan") + globalTopoRef := realTGs.Items[0].Spec.GlobalTopoServer + cluster.Spec.Images = multigresv1alpha1.ClusterImages{ + Multiorch: realTGs.Items[0].Spec.Images.Multiorch, + Multipooler: realTGs.Items[0].Spec.Images.Multipooler, + Postgres: realTGs.Items[0].Spec.Images.Postgres, + ImagePullPolicy: realTGs.Items[0].Spec.Images.ImagePullPolicy, + ImagePullSecrets: realTGs.Items[0].Spec.Images.ImagePullSecrets, + } + + // The orphan's one shard has no pools and zero Multiorch replicas, so it + // creates no Pods. That keeps the pod drain state machine, which is its + // own protocol, out of this test's way: with zero pods, + // ShardReconciler.handlePendingDeletion takes the "no pods" branch and + // sets ConditionReadyForDeletion on its very first pass. What this test + // asserts is the condition handoff between the three controllers, not + // how long draining a pod takes. + // + // Multiorch.Cells is set explicitly because getMultiorchCells + // (shard_controller.go) falls back to the union of pool cells when it is + // empty, and errors out when that is empty too; with no pools, leaving + // Cells unset turns every normal (non-deletion) reconcile of this Shard + // into a reconcile error, which is retried on controller-runtime's own + // backoff rather than this suite's compressed one and made the protocol's + // timing depend on that backoff instead of on the protocol. + // + // Opened before the orphan TableGroup exists, not after: five reconcilers + // run concurrently (scenario_fanout_test.go documents the same discipline for + // cursors, for the same reason), and multigrescluster can notice and + // annotate an orphan TableGroup within the reconcile pass that follows its + // creation. A stream opened even one line later could start listening + // after that annotation, and the condition flips it drives, have already + // landed, which is exactly the 4ms-window flake this replaces. + st := c.Watch(&TableGroupList{}, &ShardList{}) + + orphanTG, err := multigresclustercontroller.BuildTableGroup( + cluster, + DatabaseConfig{Name: "postgres"}, + &TableGroupConfig{Name: "orphan"}, + []multigresv1alpha1.ShardResolvedSpec{{ + Name: "0-inf", + Multiorch: multigresv1alpha1.MultiorchSpec{ + StatelessSpec: multigresv1alpha1.StatelessSpec{Replicas: ptr.To(int32(0))}, + Cells: []CellName{defaultSimCell}, + }, + Pools: map[PoolName]PoolSpec{}, + }}, + globalTopoRef, + Suite.Scheme, + ) + c.NoError(err, "build orphan tablegroup") + c.NoError(c.Create(orphanTG), "create orphan tablegroup") + orphanKey := client.ObjectKeyFromObject(orphanTG) + + // Pre-create the child Shard, from the now-persisted orphanTG (so its + // owner reference carries a real UID), immediately after the TableGroup + // rather than leaving TableGroupReconciler to create it on its first + // normal pass. Without this, a real race exists: if multigrescluster + // annotates the brand-new TableGroup with AnnotationPendingDeletion before + // TableGroupReconciler has ever run stepApplyDesiredShards for it, + // handlePendingDeletion lists zero child Shards and reports + // ReadyForDeletion vacuously, having consulted nothing. Creating the Shard + // here, keyed identically to what stepApplyDesiredShards would build, + // guarantees the child the protocol is supposed to drain exists before + // either controller's watch can fire. AlreadyExists is fine rather than + // fatal: it means TableGroupReconciler's own first pass won the race and + // applied this same Shard first, which is the other safe ordering. + shardCR, err := tablegroupcontroller.BuildShard( + orphanTG, + &orphanTG.Spec.Shards[0], + Suite.Scheme, + ) + c.NoError(err, "build orphan shard") + if err := c.Create(shardCR); err != nil && + !apierrors.IsAlreadyExists(err) { + c.NoError(err, "create orphan shard") + } + orphanShardKey := client.ObjectKeyFromObject(shardCR) + clusterKey := client.ObjectKeyFromObject(cluster) + + // Corroborating evidence, independent of how fast the three controllers + // converge: the interceptor records every reconcile pass, including ones + // that wrote nothing, so a pass that asked to be woken again in 5s while + // something was still pending is permanent history even if the object it + // was about is deleted moments later. Both waits are scoped by object key, + // so neither can be satisfied by the pre-existing healthy + // TableGroup/Shard's own unrelated reconciles. + // + // What this actually proves is narrower than it looks: both + // TableGroupReconciler.handlePendingDeletion and reconcileDatabases also + // take a 5s-requeue path the first time they see an orphan (setting the + // PendingDeletion annotation itself sets `allReady`/`pendingDeletion` and + // requeues), so mutating away only the later + // `meta.IsStatusConditionTrue(...ConditionReadyForDeletion)` guard still + // leaves that earlier requeue in place and these two waits keep passing - + // verified by making exactly that mutation in each function and watching + // these two lines stay green while the poll below caught it instead. What + // these two waits do rule out is a version that deletes an orphan in the + // very same pass that first notices it, with no intervening wait at all. + Suite.Reconciles.WaitForRequeue(t, "tablegroup", orphanKey, 5*time.Second, 10*time.Second) + Suite.Reconciles.WaitForRequeue( + t, + "multigrescluster", + clusterKey, + 5*time.Second, + 10*time.Second, + ) + + // Primary evidence for the specific guard on each hop, read off the event + // stream rather than sampled by polling: the write-up measured the + // TableGroup's parent deleting it 4.3ms after ReadyForDeletion is set, + // against a 20ms poll, and showed no poll interval fixes that because one + // sample already costs about as long as the state persists. The stream is + // push rather than sample, so it sees the transition however briefly it + // held. sawX latches record having observed each condition true at least + // once, so that reaching a later state (the TableGroup gone) without ever + // having latched an earlier one (its own condition, or its child Shard's) + // is still caught even if every step landed inside one 50ms reorder + // window, or before this loop's first read. + // + // Mutation verified: deleting the + // `if !meta.IsStatusConditionTrue(s.Status.Conditions, + // multigresv1alpha1.ConditionReadyForDeletion) { allReady = false }` guard + // in TableGroupReconciler.handlePendingDeletion, or the equivalent guard + // over item.Status.Conditions in reconcileDatabases, each independently + // makes this fail (tried one at a time): the TableGroup reports + // ReadyForDeletion, or is deleted, before its Shard's own condition is + // ever observed true. + sawShardReady := false + sawTableGroupReady := false + tgGone := false + deadline := time.Now().Add(30 * time.Second) + for !tgGone { + ev, err := st.Next(time.Until(deadline)) + c.NoError(err, + "waiting for the protocol to reach Shard ready, then TableGroup ready, "+ + "then TableGroup deleted, in that order") + switch { + case ev.Key == orphanShardKey && ev.Kind == "Shard": + if conditionSetTrue(ev, string(multigresv1alpha1.ConditionReadyForDeletion)) { + sawShardReady = true + } + case ev.Key == orphanKey && ev.Kind == "TableGroup": + if ev.Type == "deleted" { + tgGone = true + continue + } + if conditionSetTrue(ev, string(multigresv1alpha1.ConditionReadyForDeletion)) { + c.True(sawShardReady, + "TableGroup %s reported ReadyForDeletion before Shard %s ever did", + orphanKey.Name, orphanShardKey.Name) + sawTableGroupReady = true + } + } + } + // A relist can win the race against the last blocked read: Next selects + // over the event channel and the failure channel, and Go picks at random + // when both are ready. If it hands back the deletion, the loop exits and + // the entry guard never runs again, so the terminal error would go + // unobserved and the assertions below would blame the operator for history + // the harness lost. + c.NoError(st.Terminal(), + "the event stream failed, so every assertion over it is void") + + c.True(sawShardReady, + "TableGroup %s was deleted before Shard %s ever reported ReadyForDeletion", + orphanKey.Name, orphanShardKey.Name) + c.True(sawTableGroupReady, + "TableGroup %s was deleted before it ever reported ReadyForDeletion", + orphanKey.Name) + + // Attribution check: confirm the delete that made the TableGroup + // disappear was actually issued by multigrescluster, using a static scan + // of the completed op log rather than a live cursor wait. A live + // CursorFor(ns, "multigrescluster") wait was not usable for any step + // above: a single MultigresCluster reconcile pass unconditionally + // re-applies the healthy default Cell and TableGroup before ever reaching + // the orphan-pruning loop, so the next op in that scope is legitimately + // something else almost every time, and WaitForNext does not skip ahead + // to find a match (that is WaitForMatching, which this suite marks as an + // escape hatch not to be reached for). The op log has already stopped + // growing with respect to this object by the time we reach this check, so + // a static scan carries none of Cursor's live-ordering caveats. + deletedByCluster := false + for _, op := range Suite.Ops.OpsInNamespace(ns) { + if op.Controller == "multigrescluster" && op.Verb == "delete" && + ctrltest.KindSuffix(op.Kind) == "TableGroup" && op.Key.Name == orphanKey.Name { + deletedByCluster = true + break + } + } + c.True(deletedByCluster, + "TableGroup %s disappeared without a recorded delete from multigrescluster", + orphanKey.Name) +} + +// conditionSetTrue reports whether ev records a status.conditions entry whose +// type field arrived at conditionType, with that same entry's status field +// arrived at "True", in the same event. +// +// Changed and Transitions carry a diff, not the object's current state, so +// the type and the status of one array slot have to be read out of the same +// event to know which condition moved; a status flip alone does not say +// which condition it belongs to. Requiring the type to change in the same +// event rather than looking it up separately is sound here because every +// condition this test watches for (Shard and TableGroup's own +// ConditionReadyForDeletion) is only ever set once, straight to True: neither +// controller ever writes it False first, so its type always appears fresh +// alongside the status that makes it true. +// +// If that premise ever breaks, and a controller sets the condition False +// before True, this helper stops recognising the transition and the test fails +// with an ordering complaint against the operator rather than against itself. +// So a sudden "reported ReadyForDeletion before X ever did" failure is worth +// checking here second: the Cell controller already writes a condition False +// first, so the premise holds by habit rather than by rule. +func conditionSetTrue(ev ctrltest.Event, conditionType string) bool { + wantType := `"` + conditionType + `"` + for path, transition := range ev.Transitions { + // The status.conditions prefix is checked as well as the leaf, so a + // future array of objects carrying both a type and a status field + // cannot start feeding this helper silently. No such array exists on + // either status today; the guard is what keeps the doc comment above + // true rather than merely true for now. + if !strings.HasPrefix(path, "status.conditions[") || + !strings.HasSuffix(path, "].type") || transition.To != wantType { + continue + } + statusPath := strings.TrimSuffix(path, "type") + "status" + if status, ok := ev.Transitions[statusPath]; ok && status.To == `"True"` { + return true + } + } + return false +} diff --git a/test/suite/scenario_fanout_test.go b/test/suite/scenario_fanout_test.go new file mode 100644 index 00000000..d0578763 --- /dev/null +++ b/test/suite/scenario_fanout_test.go @@ -0,0 +1,159 @@ +package suite + +import ( + "testing" + "time" + + "k8s.io/utils/ptr" + + multigresv1alpha1 "github.com/multigres/multigres-operator/api/v1alpha1" + "github.com/multigres/multigres-operator/pkg/util/name" + "github.com/multigres/testkit/ctrltest" +) + +// fanoutCluster builds a MultigresCluster with one cell and one database, +// table group and shard: enough for the cluster controller to fan out to +// every child kind it owns (TopoServer, Cell, TableGroup) and for the +// TableGroup controller it creates to fan out to a Shard of its own. +func (c *C) fanoutCluster(clusterName string) *MultigresCluster { + c.Helper() + return c.newCluster(clusterName, func(s *MultigresClusterSpec) { + s.Databases = []DatabaseConfig{ + { + Name: "postgres", + Default: true, + TableGroups: []TableGroupConfig{ + { + Name: "default", + Default: true, + Shards: []ShardConfig{{ + Name: "0-inf", + Spec: &ShardInlineSpec{ + Multiorch: multigresv1alpha1.MultiorchSpec{ + StatelessSpec: multigresv1alpha1.StatelessSpec{ + Replicas: ptr.To(int32(1)), + }, + }, + Pools: map[PoolName]PoolSpec{ + "primary": { + ReplicasPerCell: ptr.To(int32(1)), + Type: "readWrite", + Cells: []CellName{ + defaultSimCell, + }, + }, + }, + }, + }}, + }, + }, + }, + } + }) +} + +// findOp returns the index of the first op in ops matching kind and name. +// batch has already been validated as an exact multiset by WaitForAll, so a +// miss here would mean this helper's own matching is wrong, not that the op +// is absent. +func (c *C) findOp(ops []ctrltest.Op, kind, objName string) int { + c.Helper() + for i, op := range ops { + if ctrltest.KindSuffix(op.Kind) == kind && op.Key.Name == objName { + return i + } + } + c.Fatalf("no %s %q op among %v", kind, objName, ops) + return -1 +} + +// TestClusterFanOutSequence asserts that creating a MultigresCluster fans out +// to TopoServer, Cell and TableGroup attributed to the multigrescluster +// controller, in that relative order, and that the second-order fan-out to +// Shard is attributed to the tablegroup controller rather than to +// multigrescluster. +// +// Ordering choice: multigrescluster_controller.go calls +// reconcileGlobalComponents, then reconcileCells, then (after +// reconcileTopology, which makes no Kubernetes writes here) reconcileDatabases, +// as sequential statements inside one Reconcile call. That is a genuine, +// code-level guarantee, so TopoServer < Cell < TableGroup is asserted below. +// Nothing is asserted about the relative order of the Multiadmin/MultiadminWeb +// writes reconcileGlobalComponents also makes: they are unconditional (there +// is no way to disable them from the spec) and sit between the TopoServer and +// Cell writes, but their order relative to each other, or to TopoServer and +// Cell, is not what this test is about and is left unspecified. +// +// Matcher choice: WaitForNext cannot express this, because those +// Multiadmin/MultiadminWeb writes are real, deterministic, and land between +// TopoServer and Cell, so a strict next-op chain from TopoServer would hit a +// Deployment patch instead of Cell. WaitForAll is the right tool instead: one +// call consumes the whole first pass as a multiset, which (a) still proves +// each of TopoServer/Cell/TableGroup was written by multigrescluster and +// nothing else was (in particular, that multigrescluster never itself writes +// a Shard), and (b) hands back the ops in recorded order, which is what the +// ordering check below reads. WaitForMatching was avoided entirely: it would +// let the assertion silently skip past a misordered write instead of failing +// on it, which is exactly what an ordering test must not do. +func TestClusterFanOutSequence(t *testing.T) { + c := newCase(t) + const clusterName = "fanout" + + // Cursors are opened before the cluster exists, not after: five + // reconcilers run concurrently (one goroutine each, per suite.go), so a + // cursor opened even one line late can start its scan after another + // controller's reaction to the same write has already landed, and then + // miss the very op it was meant to catch. + clusterCur := c.Cursor("multigrescluster") + tgCur := c.Cursor("tablegroup") + + c.fanoutCluster(clusterName) + + topoServerName := clusterName + "-global-topo" + cellName := name.JoinWithConstraints(name.DefaultConstraints, clusterName, defaultSimCell) + tableGroupName := name.JoinWithConstraints( + name.DefaultConstraints, clusterName, "postgres", "default", + ) + shardName := name.JoinWithConstraints( + name.DefaultConstraints, clusterName, "postgres", "default", "0-inf", + ) + + batch := clusterCur.WaitForAll(t, []ctrltest.Expect{ + // ensureClusterFinalizer, then resolveImages recording the default + // image set: both patches of the cluster object itself, before any + // child is touched. + ctrltest.ExpectPatch("MultigresCluster", clusterName), + ctrltest.ExpectPatch("MultigresCluster", clusterName), + ctrltest.ExpectPatch("TopoServer", topoServerName), + ctrltest.ExpectPatch("Deployment", clusterName+"-multiadmin"), + ctrltest.ExpectPatch("Service", clusterName+"-multiadmin"), + ctrltest.ExpectPatch("Deployment", clusterName+"-multiadmin-web"), + ctrltest.ExpectPatch("Service", clusterName+"-multiadmin-web"), + ctrltest.ExpectPatch("Service", clusterName+"-multigateway"), + ctrltest.ExpectPatch("Service", clusterName+"-multigateway-replica"), + ctrltest.ExpectPatch("Cell", cellName), + ctrltest.ExpectPatch("TableGroup", tableGroupName), + ctrltest.ExpectStatusPatch("MultigresCluster", clusterName), + }, 30*time.Second) + + topoIdx := c.findOp(batch, "TopoServer", topoServerName) + cellIdx := c.findOp(batch, "Cell", cellName) + tgIdx := c.findOp(batch, "TableGroup", tableGroupName) + c.True(topoIdx < cellIdx, + "expected TopoServer before Cell in the multigrescluster controller's "+ + "writes, got positions %d, %d in %v", topoIdx, cellIdx, batch) + c.True(cellIdx < tgIdx, + "expected Cell before TableGroup in the multigrescluster controller's "+ + "writes, got positions %d, %d in %v", cellIdx, tgIdx, batch) + + // Second-order fan-out: the TableGroup controller, not the cluster + // controller, creates the Shard. Applying the desired Shard is the first + // write TableGroupReconciler makes (stepListChildShards only reads), so + // WaitForNext is sound here without any of the batching above. The + // attribution claim itself comes from the cursor's scope, not from a + // separate check: tgCur can only ever return an op whose Controller is + // "tablegroup" (see Recorder.firstMatchFrom), and the WaitForAll batch + // above already proved multigrescluster wrote no Shard of its own, since + // one would have shown up there as an unexpected op. + tgCur.WaitForNext(t, "patch", "Shard", shardName, 30*time.Second) +} diff --git a/test/suite/scenario_race_test.go b/test/suite/scenario_race_test.go new file mode 100644 index 00000000..7b233ab8 --- /dev/null +++ b/test/suite/scenario_race_test.go @@ -0,0 +1,325 @@ +package suite + +import ( + "fmt" + "sort" + "strings" + "testing" + "time" + + "k8s.io/apimachinery/pkg/api/equality" + "sigs.k8s.io/controller-runtime/pkg/client" + "sigs.k8s.io/structured-merge-diff/v6/fieldpath" + + "github.com/multigres/testkit/ctrltest" +) + +// shardProbeAnnotation is a key neither controller's applied payload ever +// mentions: tablegroup's BuildShard only sets an annotation map at all when the +// TableGroup carries a project-ref annotation, which MinimalCluster's fixture +// never does. Writing it from the test is therefore a mutation neither +// manager's SSA apply will contend with or revert, and it is the only way this +// test can put a fresh, known-real change on the Shard once the cluster has +// converged, since a converged tablegroup no longer gets re-triggered by its own +// stale watch. +const shardProbeAnnotation = "scenario-race-test.multigres.com/probe" + +// TestTwoControllersWriteOneShard names the two-writer relationship between the +// shard and tablegroup controllers on one object, which every other test in +// this package runs past without seeing: each of them boots the suite with +// every reconciler live, but none asserts anything about a Shard being written +// by both. +// +// The relationship was measured, not assumed: before a fix, the shard +// controller wrote its own status about ten times a second forever, and while +// that ran, the tablegroup controller issued 3,641 patches against the Shard in +// three minutes, every one a no-op. Fixing the hot loop did not remove the +// two-writer relationship, only the churn it used to cause, so it is still +// worth a name. +func TestTwoControllersWriteOneShard(t *testing.T) { + c := newCase(t) + cluster := c.MinimalCluster("race") + + c.WaitForClusterHealthy(cluster) + key := c.shardKey() + + t.Run("shard and tablegroup both write the Shard", func(t *testing.T) { + c := c.Sub(t) + writers := map[string]bool{} + for _, op := range Suite.Ops.OpsInNamespace(c.NS) { + if ctrltest.KindSuffix(op.Kind) == "Shard" && op.Key.Name == key.Name { + writers[op.Controller] = true + } + } + for _, want := range []string{"shard", "tablegroup"} { + c.Check().True(writers[want], + "the %s controller has no recorded write to Shard %s; writers seen: %v", + want, key.Name, writers) + } + }) + + // Runs before the no-op check below, which mutates the Shard: a manager + // entry left over from that mutation would be a false-positive third writer + // and would blur what this is actually about, which is the two production + // managers. + t.Run("field ownership on the Shard is disjoint", func(t *testing.T) { + c := c.Sub(t) + conflicts := c.fieldOwnershipConflicts(key) + c.Empty(conflicts, + "Shard %s has fields claimed by more than one field manager, "+ + "the same shape of defect as the status hot loop:\n %s", + key.Name, strings.Join(conflicts, "\n ")) + }) + + t.Run("tablegroup's patches to the Shard never change it", func(t *testing.T) { + c.Sub(t).requireTableGroupPatchesAreNoOps(key) + }) +} + +// fieldOwnershipConflicts returns every field path on the Shard that more +// than one field manager claims, sorted. An empty result is the invariant, and +// it is the check that generalises: it would have caught the status hot-loop +// directly; the loop was two field managers each asserting a value for the +// same field, ObservedGeneration or a phase, and the API server obliging both. +// A dump of writes only shows that as churn; managedFields shows it as the +// overlap it is. +// +// Field ownership is expressed by the API server as a Set per manager, encoded +// in metadata.managedFields[].fieldsV1: this decodes each manager's Set with +// the same library the server itself uses and collects every field path that +// is a member of more than one. +func (c *C) fieldOwnershipConflicts(key client.ObjectKey) []string { + c.Helper() + + shard := &Shard{} + c.NoError(c.Get(key, shard), "get shard %s", key) + + claimants := map[string][]string{} + for _, mf := range shard.ManagedFields { + if mf.FieldsV1 == nil { + continue + } + set := &fieldpath.Set{} + c.NoError(set.FromJSON(mf.FieldsV1.GetRawReader()), + "decode managedFields for manager %s", mf.Manager) + set.Iterate(func(p fieldpath.Path) { + path := p.String() + claimants[path] = append(claimants[path], mf.Manager) + }) + } + + // A Shard with no decodable managedFields claim at all would make the + // disjointness check below pass having compared nothing, which is the one + // way this assertion can go green without the invariant holding. Nothing in + // this function's own reasoning rules that out, so it is asserted here + // rather than inherited from the writer check above. + c.True(len(claimants) > 0, + "Shard %s yielded no decodable managedFields claims, so field ownership "+ + "was not checked at all; %d managedFields entries were present", + key.Name, len(shard.ManagedFields)) + + var conflicts []string + for path, managers := range claimants { + distinct := map[string]bool{} + for _, m := range managers { + distinct[m] = true + } + if len(distinct) <= 1 { + continue + } + names := make([]string, 0, len(distinct)) + for m := range distinct { + names = append(names, m) + } + sort.Strings(names) + conflicts = append(conflicts, fmt.Sprintf("%s claimed by %v", path, names)) + } + + sort.Strings(conflicts) + return conflicts +} + +// requireTableGroupPatchesAreNoOps is assertion 2 from the brief, corrected to +// the harness as it exists now rather than as it was written: convergence is +// about 0.3s and requeues are compressed, so a fixed wall-clock window no +// longer separates from scheduling noise. It also has nothing to observe by the +// time a test could get around to sleeping: once the cluster is Healthy, +// tablegroup's own SSA patches to the Shard stop generating new watch events +// (a true no-op patch does not bump resourceVersion, so Owns(&Shard{}) never +// re-fires), so tablegroup simply stops reconciling and there is no ten-second +// window in which anything would happen anyway. +// +// So instead of waiting, this drives the relationship directly: it makes a +// small, known write of its own to the Shard (an annotation neither manager's +// apply payload mentions, see shardProbeAnnotation) to produce a fresh watch +// event, then waits for a COMPLETE tablegroup reconcile pass that both began +// after that write and wrote this Shard (see awaitTableGroupPassAfter), read +// from Suite.Reconciles rather than the raw op cursor. That matters: a +// reconcile record is only appended once the whole pass has returned, so +// waiting on it cannot observe half a pass the way a plain op-log cursor can, +// where a later step of the very pass that just satisfied the wait +// (tablegroup's own status-patch, a step after the Shard patch) can still land +// and be mistaken for the next pass's write. +// +// The no-op check itself compares content, not resourceVersion. The shard +// controller reconciles this same object roughly every clamp interval even at +// steady state (see ctrltest.RequeueClamp), so by the time tablegroup's pass +// has been observed, some other write to the Shard has almost always also +// landed in the same rough window; attributing a resourceVersion move to +// "whichever controller wrote most recently" is exactly as unsound as the +// wall-clock method this replaces; it was tried and produced a false pass +// under mutation (see the report). tablegroup's SSA apply payload is a +// complete statement of what it owns: spec, labels and ownerReferences, +// never status (see BuildShard), so comparing that payload's own fields +// before and after the pass answers the question directly and is immune to +// any concurrent, status-only write from the shard controller, no matter how +// often it fires. +// +// Repeated several times rather than once, since a single pass proves nothing +// about whether "never" holds. +func (c *C) requireTableGroupPatchesAreNoOps(key client.ObjectKey) { + c.Helper() + + const passesToObserve = 5 + for i := range passesToObserve { + before := &Shard{} + c.NoError(c.Get(key, before), "get shard %s", key) + + probedFrom := c.probeShard(key, i) + pass := awaitTableGroupPassAfter(c, c.NS, key, probedFrom) + + after := &Shard{} + c.NoError(c.Get(key, after), "get shard %s", key) + + c.True(equality.Semantic.DeepEqual(before.Spec, after.Spec), + "tablegroup's pass %s changed Shard %s's spec, the field surface its "+ + "SSA apply owns:\n before: %+v\n after: %+v", + pass, key.Name, before.Spec, after.Spec) + c.True( + equality.Semantic.DeepEqual(before.OwnerReferences, after.OwnerReferences), + "tablegroup's pass %s changed Shard %s's ownerReferences:\n before: %+v\n after: %+v", + pass, + key.Name, + before.OwnerReferences, + after.OwnerReferences, + ) + c.True(equality.Semantic.DeepEqual(before.Labels, after.Labels), + "tablegroup's pass %s changed Shard %s's labels:\n before: %+v\n after: %+v", + pass, key.Name, before.Labels, after.Labels) + } +} + +// awaitTableGroupPassAfter blocks until the tablegroup controller has completed +// a reconcile pass that started after notBefore and wrote the Shard at key, and +// returns that pass. +// +// The bar is Reconcile.Start measured against an instant captured BEFORE the +// probe write was issued, and both halves of that were arrived at by measuring +// a wrong version of it. +// +// Start, rather than a position in the log, because a record is appended when +// its pass RETURNS. A pass already in flight when the probe lands is therefore +// filed after the probe while having begun before it, so an index cursor +// admits it however freshly it was seeded. Five controllers converge this +// cluster before the first probe and leave dozens of finished passes behind, +// and a scan from index zero matched those exclusively: every iteration was +// satisfied by a pass that had started roughly half a second before the probe +// it was supposed to be reacting to. +// +// Before the write rather than after it, because a write becomes visible to +// watchers when the API server commits it, which is strictly before the +// client's own call returns. The gap is small but it is on the wrong side: the +// reacting tablegroup pass starts within a few tenths of a millisecond of the +// probe Patch returning, and measurably often starts just before it. A mark +// taken after the write returned then rejects the very pass it is waiting for, +// and since the probe is the only thing that writes this Shard once the cluster +// has converged, no later pass arrives to replace it. The wait cannot then +// succeed at any timeout, which is what made it fail about one run in three. A +// mark taken before the write has no such edge, because no pass can react to a +// write that has not been issued yet. +// +// The cost of moving the mark earlier is one API round trip of slack, in which a +// pass that did not see the probe would be accepted. That is bounded by a round +// trip instead of by the whole log, and at steady state nothing but this test +// writes the Shard, so the only candidate is a second pass caused by the +// previous iteration's probe. That pass has already been observed to completion +// before this iteration's mark is taken. +// +// Rescanning the namespace's whole log on each poll, rather than carrying a +// cursor across polls or across iterations, is deliberate for a related reason: +// the log is ordered by completion, not by start, so an index cursor and a +// start-time bar disagree about which entries are still candidates, and the +// cursor is the one that can step over the pass being waited for. The scan is +// bounded by one namespace's history and runs at most once per poll, so the +// cost of being right here is nothing worth optimising. +// +// No bookkeeping is needed to stop one iteration matching an earlier +// iteration's pass: that pass had already returned, and so had already started, +// before this iteration's mark was taken. +func awaitTableGroupPassAfter( + c *C, + ns string, + key client.ObjectKey, + notBefore time.Time, +) ctrltest.Reconcile { + c.Helper() + + var pass ctrltest.Reconcile + what := fmt.Sprintf("a tablegroup pass on Shard %s beginning after the probe", key.Name) + c.Eventually(10*time.Second, what, func() error { + for _, r := range Suite.Reconciles.InNamespace(ns) { + if r.Controller != "tablegroup" || !r.Start.After(notBefore) { + continue + } + for _, op := range Suite.Reconciles.Ops(r) { + if ctrltest.KindSuffix(op.Kind) == "Shard" && op.Key.Name == key.Name { + pass = r + return nil + } + } + } + return fmt.Errorf( + "no tablegroup pass touching Shard %s has begun since the probe write", + key.Name, + ) + }) + return pass +} + +// probeShard sets a test-owned annotation on the Shard to a fresh value and +// returns the instant just before that write was issued, which is the latest +// mark a pass reacting to it is guaranteed to start after. +// +// Returning the instant the write completed is the obvious choice and is the +// wrong one: the API server commits the write and dispatches the watch event +// before the client's Patch call returns, so the reacting pass is often already +// running by then. See awaitTableGroupPassAfter. +// +// Neither field manager's apply payload mentions the annotation (tablegroup's +// BuildShard only sets an annotation map at all when the TableGroup carries a +// project-ref annotation, which MinimalCluster's fixture never does), so it is +// a change neither manager's SSA apply will contend with or revert. It is a +// merge patch rather than a full Update, so it carries no resourceVersion +// precondition and cannot spuriously conflict with either controller's own +// concurrent write to the same object. Writing it is the only way this test +// can put a fresh, real change on the Shard once the cluster has converged, +// since a converged tablegroup no longer gets re-triggered by its own stale +// watch (see requireTableGroupPatchesAreNoOps). +func (c *C) probeShard(key client.ObjectKey, seq int) time.Time { + c.Helper() + shard := &Shard{} + c.NoError(c.Get(key, shard), "get shard %s", key) + base := shard.DeepCopy() + if shard.Annotations == nil { + shard.Annotations = map[string]string{} + } + shard.Annotations[shardProbeAnnotation] = fmt.Sprintf("%d", seq) + + issued := time.Now() + c.NoError( + c.Patch(shard, client.MergeFrom(base)), + "annotate shard %s", + key, + ) + return issued +} diff --git a/test/suite/scenario_selector_impostor_test.go b/test/suite/scenario_selector_impostor_test.go new file mode 100644 index 00000000..2ae0c28e --- /dev/null +++ b/test/suite/scenario_selector_impostor_test.go @@ -0,0 +1,643 @@ +// Selector inventory for the multigrescluster and shard controllers: every +// List call in pkg/cluster-handler/controller/multigrescluster/ and +// pkg/resource-handler/controller/shard/ that carries a label selector, the +// selector itself, and what this sweep found or judged about it. +// +// 35 List call sites in the two packages, 12 in multigrescluster and 23 in +// shard, of which 32 carry a label selector; the three that do not are +// recorded below as out of scope rather than omitted, so that the count can +// be rederived from this block. Line numbers point at the List( call, not at +// the MatchingLabels argument. Cross-checked for the other spellings a +// selector can take (MatchingLabelsSelector, HasLabels, a raw +// ListOptions{LabelSelector}): neither package uses any of them. +// +// multigrescluster (pkg/cluster-handler/controller/multigrescluster/): +// +// - reconcile_cells.go:23 CellList {cluster} +// Feeds the active/orphan diff for cells removed from spec. The eventual +// delete is gated behind the AnnotationPendingDeletion + ConditionReady- +// ForDeletion handshake (reconcile_cells.go:92-125), not a raw sweep. +// SAFE (condition-gated). Not tested here. +// +// - reconcile_databases.go:23 TableGroupList {cluster} +// Same shape as reconcile_cells.go:23, for TableGroups. SAFE (condition- +// gated); the protocol itself is already covered by +// TestReadyForDeletionProtocol in scenario_deletion_test.go. +// +// - reconcile_topology.go:245 CellList {cluster} +// Read-only: collects names of cells pending deletion, for topology +// pruning. Never mutates or deletes anything itself. SAFE. +// +// - reconcile_global.go:113 TopoServerList {cluster} +// CONFIRMED LIVE DEFECT (Defect 2 in +// tasks/multigres-operator-bugs-found-by-the-suite.md). No component +// filter, no owner-reference check: when global topology is external, +// every TopoServer carrying the cluster label is deleted, including a +// cell's own local TopoServer (component "local-topo"), which this +// selector cannot tell apart from the managed global one (component +// "global-topo"). TESTED below: +// TestSelectorImpostorGlobalTopoPruneDeletesCellOwnedLocalTopoServer. +// +// - multigrescluster_controller.go:401 CellList {cluster}, in +// handleDeletion (whole-cluster teardown). Raw delete, no owner-ref +// check. Reachable only while the cluster itself is being deleted, and +// an impostor sharing the label would first be picked up by +// reconcile_cells.go's own condition-gated path above (the cell +// controller reconciles any Cell object regardless of who owns it), +// which confounds an isolated impostor test for this exact line. +// JUDGED UNREACHABLE IN ISOLATION within this pass; not tested. See the +// task report for the reasoning in full. +// +// - multigrescluster_controller.go:419 TableGroupList {cluster}, in +// handleDeletion. Same shape and same confound as line 401, via +// reconcile_databases.go:23's condition-gated path. Not tested. +// +// - multigrescluster_controller.go:439 PersistentVolumeClaimList +// {cluster, component=toposerver}, in handleDeletion. Raw delete, no +// owner-ref check, but the function's own comment states the intent: +// these PVCs "may outlive their TopoServer when a cluster switches from +// managed to external topology," i.e. an unowned PVC with this label +// pair is the expected steady state this code exists to clean up, not +// an anomaly. SAFE BY DESIGN for this task's "should do nothing to an +// object it does not own" heuristic, because the whole point here is +// that ownership cannot be established for what it is meant to sweep. +// Not pinned as a defect; the design's blast radius (anything bearing +// these two labels, from any source, is eligible) is flagged in the +// report rather than pinned as a KnownDefect. +// +// This is SAFE while shard_controller.go:589 below is a DEFECT on what +// looks like the same evidence, an unowned PVC being acted on, and the +// two verdicts are worth reading together rather than one at a time. +// The difference is documented intent, which is the only thing that can +// separate them: :439 says an unowned PVC carrying these labels is +// precisely what it exists to sweep, whereas :589's own doc comment +// (shard_controller.go:571-574) and its call site +// (shard_controller.go:425-426) both say it exists to fix up ownerRefs +// on the shard's own PVCs across a mid-lifecycle policy change. Acting +// on a stranger is the job in one and an accident in the other. +// +// - multigrescluster_controller.go:603 MultigresClusterList, InNamespace +// only. Not a label selector (a map function for CoreTemplate/ +// CellTemplate/ShardTemplate change fanout). Out of scope. +// +// - certificate.go:394 (via pkg/util/certs.List) no label selector, +// InNamespace only. The eventual delete (certs.Prune) checks +// OwnedBy(cert, ownerUID) before deleting anything: the one place in +// this controller that already does what reconcile_global.go:113 does +// not. SAFE, and the contrast is the original bug write-up's own point. +// +// - status.go:125, :150, :220 CellList / TableGroupList / TopoServerList +// {cluster}. Purely read, to aggregate MultigresCluster.Status; nothing +// is ever mutated or deleted at these call sites. SAFE for this task's +// "does the controller act on it" question. A mislabelled object here +// would only ever skew the cluster's own reported status, a different +// risk this task's Quiet()-shaped assertion cannot express and which is +// not assessed here. +// +// shard (pkg/resource-handler/controller/shard/): +// +// - shard_controller.go:589 (reconcilePVCOwnerRefs) PersistentVolumeClaimList +// {cluster, database, tablegroup, shard} (no pool, no component). NEW +// DEFECT found by this sweep: the selector is the shard's four identity +// keys and nothing else, so it cannot tell the shard's own PVC from any +// other object carrying the same identity, and the only test applied to +// a match before adoption is whether it already carries a ref with this +// shard's UID (shard_controller.go:625-631). A PVC belonging to nobody +// fails that test, so it is adopted: SetControllerReference + Patch, +// whenever the shard's effective PVCDeletionPolicy resolves to Delete. +// +// There is a real ownership check here, and naming it correctly matters +// because it bounds the defect. ctrl.SetControllerReference returns +// AlreadyOwnedError when the object already carries a different +// controller ownerRef (controller-runtime v0.25.0, +// pkg/controller/controllerutil/controllerutil.go:97-99), so this code +// does not adopt a PVC that already belongs to someone else. What it +// adopts is a PVC with no controller owner at all. TESTED below: +// TestSelectorImpostorShardOwnerRefReconcileAdoptsUnrelatedPVC. +// +// - reconcile_deletion.go:57 DeploymentList {cluster, database, +// tablegroup, shard}, in the Shard's own handleDeletion. Raw delete, no +// owner-ref check. Same family as shard_controller.go:589. Not +// independently tested given this pass's budget. +// +// - reconcile_deletion.go:90 PodList same 4-key selector, same +// function. Raw delete, no owner-ref check. Lower interference risk +// than the multigrescluster Cell/TableGroup case above, since a Shard's +// own teardown is not cascaded through another controller's graceful +// orphan protocol. Not tested here. +// +// - reconcile_deletion.go:166 (cleanupShardPVCs) PersistentVolumeClaimList +// same 4-key selector. Every match is marked orphan or deleted with no +// owner-ref check, gated only by shardPVCShouldBeCleaned's policy read. +// Same family as shard_controller.go:589 at a different lifecycle +// point. Not independently tested. +// +// - reconcile_deletion.go:252 (handlePendingDeletion) PodList same +// 4-key selector. Runs the drain state machine (initiateDrain / +// clearDrainAnnotations / Delete) against any match once the Shard +// itself carries the PendingDeletion annotation. Same family; combining +// it with a graceful shard-level orphan flow adds the same entanglement +// seen in the multigrescluster Cell/TableGroup case. Not tested. +// +// - reconcile_data_plane.go:290, :367 PodList same 4-key selector. +// Read-only relative to the pods themselves (feeds +// shard.Status.PodRoles and posture.Evaluate). SAFE. +// +// - reconcile_data_plane.go:542 (reconcileDrainState) PodList same +// 4-key selector. MUTATES a matching pod (clears its drain annotations) +// when isDrainStale holds, which requires the pod's pool label to +// resolve to a real shard.Spec.Pools entry, a name that parses to an +// in-range replica ordinal, and a spec judged unchanged from desired. +// A real candidate, but reproducing that combination on a synthetic +// impostor is disproportionate for this pass; deferred. +// +// - reconcile_data_plane.go:673 (reconcilePoolerPrune) PodList same +// 4-key selector. The action it drives (topo.MarkDeadPoolers) writes to +// the fake topology store, not to the Kubernetes object, so this +// suite's k8s-event Stream cannot observe the outcome either way. +// UNTESTABLE WITH THIS HARNESS. +// +// - reconcile_quarantine.go:83 (reconcileQuarantineRemediation) PodList +// same 4-key selector. Deletes a pod and hard-deletes its PVC, but only +// for names the fake topology store reports as LIFECYCLE_QUARANTINED. +// Driving that requires reaching into the suite's internal topology +// registry; deferred. +// +// - disruption.go:30 PodList {cluster, database, tablegroup, shard, +// component=Pool} (adds the component key the multigrescluster prune +// lacks). Read-only (feeds canStartDisruption's decision). SAFE. +// +// - maintenance_surge.go:300, :338 PodList component+cell-scoped via +// shardPDBLabels/metadata.GetSelectorLabels. Read-only. SAFE. +// +// - postgres_config.go:266 ShardList, InNamespace only. Not a label +// selector (map function for ConfigMap-change fanout). Out of scope. +// +// - reconcile_shared_infra.go:373 PodList the PDB's own +// component+pool+cell selector. Read-only (sizes MinAvailable). SAFE. +// +// - reconcile_shared_infra.go:411 PodDisruptionBudgetList same PDB +// selector. Deletes an unmatched PDB, but only those that also pass +// metav1.IsControlledBy(pdb, shard): an explicit owner check right in +// the loop. SAFE, and the "done right" counterpart to +// shard_controller.go:589. +// +// - reconcile_pool_pods.go:50 PodList pool+cell-scoped +// (buildPoolLabelsWithCell). DEFECT of the same class as +// shard_controller.go:589, and the strongest untested candidate left in +// this inventory. The list populates existingPods +// (reconcile_pool_pods.go:69-73), which is then walked by name with no +// ownership check at all: +// +// Phase 0, syncDrainedLabels (:82, body :943-:974), iterates every map +// member and patches multigres.com/pod-role whenever +// resolvePodRole(shard, pod.Name) disagrees with the label the pod +// carries, so a pod this shard does not own has that label written or +// stripped. Phase 2, handleScaleDown (:118, body :499), classifies any +// member whose name does not parse as - (resolvePodIndex, +// :1240-:1250) or whose index is at or beyond effectiveReplicas as an +// extra pod (:537-:540), then drains (initiateDrain, :650) and deletes +// it (:567). A plausible impostor Pod carrying the pool and cell labels +// and any non-numeric name suffix is therefore drained and deleted. +// Secondary consequence at the same site: isPoolHealthy(existingPods, +// ...) (:585) counts an impostor as a pool member, so a non-ready one +// blocks legitimate scale-down of the real pool. +// +// NOT PINNED in this pass, and recorded here rather than left implied: +// a pin is a test, and this one needs a Pod-kind impostor against +// DataPlaneSim.tickPods, whose write set differs from tickPVCs's and +// has not been derived (see the TestSelectorImpostorShardOwnerRef... +// caveat below). Naming it SAFE, as an earlier revision of this block +// did, was the error worth correcting: an unexamined site is a gap, and +// a gap signed SAFE is worse than one left open. +// +// - reconcile_pool_pods.go:60 PersistentVolumeClaimList pool+cell- +// scoped. SAFE, but name-keyed rather than read-only, which is the +// accurate justification: pvcutil.ClearOrphan (:204) and +// expandPVCIfNeeded (:216) both write to list members, and what makes +// them safe is that every access is existingPVCs[pvcName] where +// pvcName comes from BuildPoolDataPVCName, a deterministic desired +// name. An impostor under any other name is never indexed, so it is +// genuinely untouched. Unlike :50 above, which walks the map itself. +// +// - reconcile_pool_pods.go:1116 PersistentVolumeClaimList pool+cell- +// scoped, counts non-orphan PVCs. Read-only. SAFE. +// +// - reload.go:69 PodList {cluster, database, tablegroup, shard, +// component=Pool}. Read-only (feeds the reload decision). SAFE. +// +// - status.go:221 PodList pool+cell-scoped (buildPoolLabelsWithCell). +// Read-only (status aggregation). SAFE. +// +// - status.go:365 PodList buildMultiorchLabelsWithCell selector. +// Read-only (crash-loop detection for status). SAFE. +// +// - reconcile_readiness.go:33 PodList {cluster, database, tablegroup, +// shard, component=Pool}. Read-only (readiness aggregation). SAFE. + +package suite + +import ( + "errors" + "fmt" + "testing" + "time" + + corev1 "k8s.io/api/core/v1" + "k8s.io/apimachinery/pkg/api/resource" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/utils/ptr" + "sigs.k8s.io/controller-runtime/pkg/client" + + multigresv1alpha1 "github.com/multigres/multigres-operator/api/v1alpha1" + topopkg "github.com/multigres/multigres-operator/pkg/data-handler/topo" + "github.com/multigres/multigres-operator/pkg/util/metadata" + "github.com/multigres/multigres-operator/pkg/util/name" + "github.com/multigres/testkit/ctrltest" +) + +// errLocalTopoDeleted is what the TopoServer script's invariant returns when +// it sees the deletion that test pins. +// +// A sentinel rather than a match on the harness's message text, because +// KnownDefect reads any non-nil error as "the pinned defect is still live". +// This pin's script can produce several other errors that are not that +// deletion, and every one of them would otherwise keep the pin green: a +// legitimate toposerver status write that the step's declaration failed to +// account for, a step that timed out because that declaration has drifted from +// what the controller now writes. Those are facts about this test, not about +// the operator, so the check body has to be able to tell them apart, and it +// cannot do that by reading a string the harness is free to reformat. +var errLocalTopoDeleted = errors.New( + "the cell's own local TopoServer was deleted", +) + +// selectorImpostorNudgeAnnotation is a key no controller's applied payload +// ever mentions, following the same reasoning as shardProbeAnnotation in +// scenario_race_test.go: tablegroup's BuildShard sets an annotation map on a Shard +// only when its TableGroup carries a project-ref annotation +// (pkg/cluster-handler/controller/tablegroup/builders.go:37-48), which +// MinimalCluster's fixture never does, so writing this key is a mutation +// neither manager's SSA apply contends with or reverts. +// +// It has to enqueue the Shard to be useful, and it does: the shard +// controller's For(&Shard{}) (shard_controller.go:693) +// carries no predicate, so a metadata-only patch is a reconcile trigger. That +// is the whole reason the key exists, since a converged Shard has nothing left +// to re-trigger it and reconcilePVCOwnerRefs only looks at an impostor on a +// pass that actually runs. +const selectorImpostorNudgeAnnotation = "scenario-selector-impostor-test.multigres.com/nudge" + +// externalGlobalTopoCluster creates a MultigresCluster whose global topology is +// external and whose one cell manages its own local TopoServer, the exact +// combination tasks/multigres-operator-external-topo-deletes-local.md +// reproduced on a live cluster: it is what makes reconcileGlobalTopoServer's +// desired-is-nil branch run on every reconcile, while still giving the cell +// controller a local TopoServer of its own to keep reapplying. +// +// It returns an error rather than calling t.Fatalf as MinimalCluster does, +// because its caller runs it inside a Script step's do: a Fatalf there would +// Goexit out of the middle of a script, whereas Script.TryStep routes a failing +// do to the test's own Fatalf and, crucially, never lets it reach KnownDefect +// as though it were evidence about the operator. +// +// The password Secret is created here rather than shared with MinimalCluster +// because the two fixtures differ in every other field; what is worth keeping +// in step is the deliberate choice to leave it unlabelled, which is what a user +// would create and what makes the reconcilers' APIReader necessary. +func (c *C) externalGlobalTopoCluster(clusterName string) error { + c.Helper() + + secret := &corev1.Secret{ + ObjectMeta: metav1.ObjectMeta{Name: adminSecretName, Namespace: c.NS}, + StringData: map[string]string{"password": "postgres"}, + } + if err := c.Create(secret); err != nil { + return fmt.Errorf("create password secret: %w", err) + } + + cluster := &MultigresCluster{ + ObjectMeta: metav1.ObjectMeta{Name: clusterName, Namespace: c.NS}, + Spec: MultigresClusterSpec{ + PostgresPasswordSecretRef: PostgresPasswordSecretRef{ + Name: adminSecretName, + Key: "password", + }, + PVCDeletionPolicy: &PVCDeletionPolicy{ + WhenDeleted: multigresv1alpha1.DeletePVCRetentionPolicy, + WhenScaled: multigresv1alpha1.DeletePVCRetentionPolicy, + }, + GlobalTopoServer: &multigresv1alpha1.GlobalTopoServerSpec{ + External: &multigresv1alpha1.ExternalTopoServerSpec{ + Endpoints: []multigresv1alpha1.EndpointUrl{ + "https://external-topo.invalid:2379", + }, + }, + }, + Cells: []CellConfig{ + { + Name: defaultSimCell, + ZoneID: "us-central1-a", + Spec: &multigresv1alpha1.CellInlineSpec{ + LocalTopoServer: &multigresv1alpha1.LocalTopoServerSpec{ + Etcd: &multigresv1alpha1.EtcdSpec{ + Replicas: ptr.To(int32(1)), + }, + }, + }, + }, + }, + }, + } + if err := c.Create(cluster); err != nil { + return fmt.Errorf("create MultigresCluster: %w", err) + } + return nil +} + +// TestSelectorImpostorGlobalTopoPruneDeletesCellOwnedLocalTopoServer pins +// Defect 2 from tasks/multigres-operator-bugs-found-by-the-suite.md: +// reconcile_global.go:113 lists TopoServers by cluster label alone (no +// component filter, no owner-reference check) and deletes every match +// whenever global topology is external. A cell's own local TopoServer +// carries that same cluster label, so it is not this controller's to +// manage, but the selector cannot tell the difference. +// +// The impostor here is not synthetic: it is the real local TopoServer the cell +// controller legitimately creates and keeps reapplying, which is exactly what +// makes the write-up call this "a permanent create/delete loop" rather than a +// one-off. That permanence is also what shapes the script below, because it +// rules out the move every other test in this suite makes first. An object +// caught in a create/delete loop never settles, so there is no converged +// namespace to open a watch onto: neither RequireQuiescent nor a poll for a +// stable TopoServer can be used here, and waiting a fixed margin for the +// toposerver controller to stop writing is a guess at how long another actor's +// work takes, which is the thing this suite exists to refuse. +// +// So the watch opens first, on an empty namespace, and the fixture is created +// inside the script's own step. Every legitimate write the toposerver +// controller then makes to the object is named as a permitted change, which +// leaves the deletion as the one event nothing accounts for, and leaves nothing +// to wait out. +func TestSelectorImpostorGlobalTopoPruneDeletesCellOwnedLocalTopoServer(t *testing.T) { + c := newCase(t) + const clusterName = "ext-global-topo" + cellResourceName := name.JoinWithConstraints( + name.DefaultConstraints, clusterName, string(defaultSimCell), + ) + localTopoName := topopkg.ManagedLocalTopoServerName(cellResourceName) + + script := c.NewScript(&TopoServerList{}) + + // The deletion is this test's entire claim, so it is asserted directly + // rather than inferred from being whatever event no step happened to + // permit. Script.TryStep and Script.TryFinish both run the invariants + // against an event before comparing it to the permitted set + // (script.go:224, :243, :294), so the delete is reported as this violation + // wherever it lands: while the step is still waiting on a status write, + // inside its settle window, or inside Finish's horizon. Combined with the + // sentinel above, that is what makes the pin's evidence the deletion on + // every run instead of whichever event happened to arrive first. + script.Invariant( + "the cell's own local TopoServer is never deleted", + func(ev ctrltest.Event) error { + if ev.Type == "deleted" && ev.Kind == "TopoServer" && + ev.Key.Name == localTopoName { + return errLocalTopoDeleted + } + return nil + }, + ) + + c.KnownDefect( + "pkg/cluster-handler/controller/multigrescluster/reconcile_global.go:113 "+ + "(external-global-topo prune selector has no component filter or "+ + "owner-reference check, so it also deletes a cell's own local TopoServer)", + func() error { + stepErr := script.TryStep( + "the cell controller creates its own local TopoServer and the "+ + "toposerver controller settles its status on it", + func() error { + return c.externalGlobalTopoCluster(clusterName) + }, + // The toposerver controller's whole settling sequence on a + // TopoServer it has just been handed, measured over four runs + // against a cluster whose global topology is managed so this + // prune never fires, which is the one way to observe what the + // object does when it is left alone: the first condition, then + // the client and peer endpoints once the etcd StatefulSet + // exists, then Ready once DataPlaneSim has ticked that + // StatefulSet ready. Three writes, in that order, then quiet + // indefinitely. + // + // The paths are the narrowest that pick out one write each. + // status.conditions[0] belongs only to the first and + // status.clientService only to the second; the third's paths + // are a subset of the second's, so it is matched by + // elimination, which is what assignEvents does a search rather + // than a greedy first match for. + // + // No ordering is declared between them even though one was + // observed, because ordering is opt-in for changes that follow + // from the code and nothing here needs it: the deletion is + // caught by the invariant above, not by an order violation. + ctrltest.Added("TopoServer", localTopoName), + ctrltest.Changed("TopoServer", localTopoName, "status.conditions[0].type"), + ctrltest.Changed("TopoServer", localTopoName, + "status.clientService", "status.peerService"), + ctrltest.Changed("TopoServer", localTopoName, "status.phase"), + ) + // TryFinish runs whatever the step returned, so that the script + // ends with Finish exactly once and the end-of-script backstop is + // satisfied on every path through this body. While the defect is + // live the step returns long before the object has finished + // settling, so the end of the script still has to be closed. + // + // The horizon is not load-bearing in either direction, which is the + // point of choosing it freely. While the defect is live nothing + // depends on it: the delete lands about 15ms after the create, + // inside the step. Once the defect is fixed this is the only window + // left in which a later prune pass could still be caught, and a + // longer horizon can only refuse more events, never permit one. + finishErr := script.TryFinish(10 * time.Second) + + switch { + case errors.Is(stepErr, errLocalTopoDeleted): + return stepErr + case errors.Is(finishErr, errLocalTopoDeleted): + return finishErr + case stepErr != nil: + // Anything else is this test's own declaration or pacing + // rather than evidence about the operator, and a pin that + // confirmed on it would survive the fix it is supposed to + // expire on. Fatalf is the right side of the line + // Script.fatalf already draws for the same reason. + c.Fatalf("the script's declaration of the toposerver "+ + "controller's settling sequence did not hold, which is a "+ + "fact about this test rather than about the prune it pins: %v", + stepErr) + case finishErr != nil: + c.Fatalf("the script's end was not quiet, and not because of "+ + "the deletion this test pins, which is a fact about this "+ + "test rather than about the prune: %v", finishErr) + } + return nil + }, + ) +} + +// TestSelectorImpostorShardOwnerRefReconcileAdoptsUnrelatedPVC pins a new +// defect found by this sweep: reconcilePVCOwnerRefs +// (shard_controller.go:589) lists PersistentVolumeClaims by the shard's four +// identity labels alone (cluster, database, tablegroup, shard; no pool, no +// component), so it cannot tell the shard's own PVC from any other object +// carrying the same identity, and the only test it applies to a match before +// adopting it is whether that match already carries a ref with this shard's +// UID. A PVC belonging to nobody fails that test and is adopted, whenever the +// shard's effective PVCDeletionPolicy resolves to Delete. +// +// The bound on the defect is worth stating precisely, because it is what a fix +// has to be aimed at. There is an ownership check in this path: +// ctrl.SetControllerReference returns AlreadyOwnedError when the object already +// carries a different controller ownerRef, so this code does not take a PVC +// that belongs to someone else. What it takes is a PVC with no controller owner +// at all. The missing check is not "does this belong to somebody else" but +// "does this belong to me", and the selector is what cannot answer it. +// +// The impostor is a bare PersistentVolumeClaim carrying just those four +// labels and no pool label, the same shape shard_controller.go's own +// "shared backup PVC" branch expects, and no owner reference at all: the +// shape a PVC left behind by some other process, or a since-recreated +// resource under the same identity, would plausibly have. +// +// RequireQuiescent runs before the script's watch opens so the baseline it +// replays is the shard's own already-settled PVCs (its pool data PVCs and +// its backup PVC), named explicitly rather than guessed: this suite's own +// discipline is that an already-populated namespace's replay is the first +// step's problem to permit, not something to dodge by racing the watch +// ahead of convergence. Unlike the TopoServer test above, that is available +// here, because the shard does converge. +// +// The impostor itself is created after the watch opens, and its own +// creation-then-bind is declared with Before: DataPlaneSim +// (pkg/ctrltest/datasim.go) binds every PersistentVolumeClaim in the cluster +// regardless of who it belongs to, as a stand-in for the volume provisioner +// envtest does not run, and that status patch (status.phase/accessModes/ +// capacity) has to be permitted explicitly or it is indistinguishable from +// the actual ownerRef adoption this test is pinning. Declaring the pair +// with Before, rather than waiting for Bound out-of-band first, is what +// keeps the watch open across the one window where the real defect could +// otherwise race in unobserved, immediately after creation and before this +// suite's own fake gets to it. +func TestSelectorImpostorShardOwnerRefReconcileAdoptsUnrelatedPVC(t *testing.T) { + c := newCase(t) + ns := c.NS + c.MinimalCluster("pvc-adopt") + + shard := c.awaitShard() + + c.RequireQuiescent(time.Second, 30*time.Second) + + existingPVCs := &corev1.PersistentVolumeClaimList{} + c.NoError(c.List(existingPVCs), "list existing PVCs") + + impostor := &corev1.PersistentVolumeClaim{ + ObjectMeta: metav1.ObjectMeta{ + Name: "impostor-shared-pvc", + Namespace: ns, + Labels: map[string]string{ + metadata.LabelMultigresCluster: shard.Labels[metadata.LabelMultigresCluster], + metadata.LabelMultigresDatabase: string(shard.Spec.DatabaseName), + metadata.LabelMultigresTableGroup: string(shard.Spec.TableGroupName), + metadata.LabelMultigresShard: string(shard.Spec.ShardName), + }, + }, + Spec: corev1.PersistentVolumeClaimSpec{ + AccessModes: []corev1.PersistentVolumeAccessMode{corev1.ReadWriteOnce}, + Resources: corev1.VolumeResourceRequirements{ + Requests: corev1.ResourceList{ + corev1.ResourceStorage: resource.MustParse("1Gi"), + }, + }, + }, + } + + allow := make([]ctrltest.Allow, 0, len(existingPVCs.Items)+1) + for _, pvc := range existingPVCs.Items { + allow = append(allow, ctrltest.Added("PersistentVolumeClaim", pvc.Name)) + } + allow = append(allow, ctrltest.Before( + ctrltest.Added("PersistentVolumeClaim", impostor.Name), + // Narrowed to status.phase rather than left open: DataPlaneSim's bind + // always touches it alongside accessModes/capacity, so naming it is + // enough to identify that write and that write only. Left open, this + // leaf would also happily absorb the ownerRef adoption this test + // exists to catch, since Changed with no paths matches any + // modification at all. + ctrltest.Changed("PersistentVolumeClaim", impostor.Name, "status.phase"), + )) + + script := c.NewScript(&corev1.PersistentVolumeClaimList{}) + + c.KnownDefect( + "pkg/resource-handler/controller/shard/shard_controller.go:589 "+ + "(reconcilePVCOwnerRefs selects on the shard's four identity labels "+ + "alone, so it adopts any PVC carrying them that has no controller "+ + "ownerRef, with nothing establishing the PVC is the shard's own)", + func() error { + stepErr := script.TryStep( + "baseline PVCs replay; the impostor is created, then bound like "+ + "any other PVC by this suite's data-plane fake", + func() error { + if err := c.Create(impostor); err != nil { + return err + } + // The impostor's own creation cannot trigger the shard's + // reconcile loop (it carries no owner reference, so + // Owns(&PersistentVolumeClaim{}) has nothing to map it back + // to), and RequireQuiescent above means nothing else is left + // to either: measured empirically, a fully quiesced shard + // does not reconcile again on its own. A metadata-only nudge + // on the Shard itself is what actually gets + // reconcilePVCOwnerRefs to run again and look at the + // impostor; this write is on ShardList, not the + // PersistentVolumeClaimList this script watches, so it needs + // no entry of its own in allow. + shardCopy := shard.DeepCopy() + patch := client.MergeFrom(shardCopy.DeepCopy()) + if shardCopy.Annotations == nil { + shardCopy.Annotations = map[string]string{} + } + shardCopy.Annotations[selectorImpostorNudgeAnnotation] = time.Now(). + UTC(). + Format(time.RFC3339Nano) + return c.Patch(shardCopy, patch) + }, + allow..., + ) + // Run unconditionally, and 10s, for the reasons given at the same + // call in the TopoServer test above. Unlike that test this one + // needs no sentinel to tell two permitted-set outcomes apart: the + // adoption is a modification of the impostor, and the only other + // modification anything makes to it is DataPlaneSim's bind, which + // the step permits by name and by path. So there is no second + // route to a non-nil error through a permitted-set mismatch. + // + // That is narrower than "no second route at all", and the + // difference matters for how much this pin can be trusted. A + // TryStep timeout is also a non-nil error, and KnownDefect reads + // any non-nil error as the defect still being live, so if the data + // plane fake never binds the impostor or a baseline PVC name + // drifts, this pin survives the operator fix that should have + // retired it. That is the general limitation of pinned steps + // stated on TryStep, and this call site is not exempt from it. The + // TopoServer test's sentinel-plus-Fatalf discrimination is the + // honest pattern if this ever needs to be tightened. + finishErr := script.TryFinish(10 * time.Second) + if stepErr != nil { + return stepErr + } + return finishErr + }, + ) +} diff --git a/test/suite/scenario_shard_lifecycle_test.go b/test/suite/scenario_shard_lifecycle_test.go new file mode 100644 index 00000000..d494a34d --- /dev/null +++ b/test/suite/scenario_shard_lifecycle_test.go @@ -0,0 +1,896 @@ +package suite + +import ( + "context" + "fmt" + "slices" + "strings" + "testing" + "time" + + corev1 "k8s.io/api/core/v1" + storagev1 "k8s.io/api/storage/v1" + apierrors "k8s.io/apimachinery/pkg/api/errors" + "k8s.io/apimachinery/pkg/api/meta" + "k8s.io/apimachinery/pkg/api/resource" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/utils/ptr" + "sigs.k8s.io/controller-runtime/pkg/client" + + multigresv1alpha1 "github.com/multigres/multigres-operator/api/v1alpha1" + "github.com/multigres/multigres-operator/pkg/resolver" + shardcontroller "github.com/multigres/multigres-operator/pkg/resource-handler/controller/shard" + "github.com/multigres/multigres-operator/pkg/util/metadata" + "github.com/multigres/multigres-operator/pkg/util/name" + "github.com/multigres/testkit/ctrltest" +) + +const lifecycleShardName multigresv1alpha1.ShardName = "0-inf" + +// blockedObservationWindow is how long steps 5 and 6 watch for a reaction +// that never comes. +// +// The shard controller requeues every disruptionRecoveryRequeue (5s, +// pkg/resource-handler/controller/shard/disruption.go:17) for as long as a +// disruption is refused, so a window several times that long is the +// difference between "the operator tried repeatedly and refused every time" +// and "the operator had not got round to trying yet". A Quiet() step on its +// own asserts silence only over the 250ms settle window, which for this +// question would be almost nothing. +const blockedObservationWindow = 20 * time.Second + +// lifecycleShardRef is a Shard carrying only the fields BuildPoolPodName, +// BuildPoolDataPVCName and BuildSharedBackupPVCName read: the cluster label +// and the three spec names. Those four values are known before the real +// Shard exists, since lifecycleCluster (below) chooses them, which lets the +// test predict a pod or PVC's name ahead of the create that produces it, +// using the operator's own name builders rather than a hand-rolled format +// string. It is never sent to the API server. +func lifecycleShardRef(clusterName string) *Shard { + return &Shard{ + ObjectMeta: metav1.ObjectMeta{ + Labels: map[string]string{metadata.LabelMultigresCluster: clusterName}, + }, + Spec: multigresv1alpha1.ShardSpec{ + DatabaseName: resolver.DefaultSystemDatabaseName, + TableGroupName: resolver.DefaultSystemTableGroupName, + ShardName: lifecycleShardName, + }, + } +} + +// lifecycleStorageClassName is the StorageClass lifecycleCluster's pool +// references, distinct per namespace so concurrently-running instances of +// this test never collide on the same cluster-scoped object. +func lifecycleStorageClassName(ns string) string { + return ns + "-lifecycle-expandable" +} + +// lifecycleCluster creates the same MultigresCluster MinimalCluster does +// (one cell, one database, one table group, one shard), plus a StorageClass +// with AllowVolumeExpansion set and the pool's Storage.Class pointed at it. +// +// Two facts force this rather than a plain call to MinimalCluster. First, +// PopulateClusterDefaults's own injection of Databases is commented +// "in-memory" for a reason: with no mutating webhook running in this suite, +// nothing ever writes it back to the MultigresCluster object itself, every +// reconcile recomputes it from scratch, and cluster.Spec.Databases stays +// permanently empty on the server unless a caller sets it explicitly. That +// alone would still allow starting from MinimalCluster's bare cluster and +// seeding Databases later, in updateLifecyclePool's first call. But second, +// storageClassName is immutable once a PersistentVolumeClaim exists, and step +// 3 needs the pool's existing data PVCs to already reference a StorageClass +// that allows expansion, or the API server's resize admission check refuses +// the request outright ("only dynamically provisioned pvc can be resized"). +// So the StorageClass has to be in place, and referenced, from this create. +// +// The shape mirrors what resolver.PopulateClusterDefaults would have +// injected for a MinimalCluster (one pool, one cell, the two-replica floor +// pkg/resolver/shard.go computes for a single-cell pool): every attribute +// this script mutates is still reached by changing a cluster that already +// exists, this just makes explicit at creation the one attribute (storage +// class) that cannot be introduced by a later mutation. +func (c *C) lifecycleCluster(clusterName string) *MultigresCluster { + c.Helper() + + scName := lifecycleStorageClassName(c.NS) + allowExpansion := true + sc := &storagev1.StorageClass{ + ObjectMeta: metav1.ObjectMeta{Name: scName}, + Provisioner: "multigres-test/no-op", + AllowVolumeExpansion: &allowExpansion, + } + c.NoError(c.Create(sc), "create StorageClass %s", scName) + // Cluster-scoped, so the per-test namespace delete does not reach it and + // every run would otherwise leave one behind in the shared envtest API + // server for the life of the package. context.Background rather than + // c.Context, which is already cancelled by the time cleanups run. + c.Cleanup(func() { + _ = c.Client().Delete(context.Background(), sc) + }) + + return c.newCluster(clusterName, func(s *MultigresClusterSpec) { + s.Databases = []DatabaseConfig{{ + Name: resolver.DefaultSystemDatabaseName, + Default: true, + TableGroups: []TableGroupConfig{{ + Name: resolver.DefaultSystemTableGroupName, + Default: true, + Shards: []ShardConfig{{ + Name: lifecycleShardName, + Spec: &ShardInlineSpec{ + Pools: map[PoolName]PoolSpec{ + resolver.DefaultPoolName: { + Type: "readWrite", + Cells: []CellName{defaultSimCell}, + ReplicasPerCell: ptr.To(int32(2)), + Storage: multigresv1alpha1.StorageSpec{Class: scName}, + }, + }, + }, + }}, + }}, + }} + }) +} + +// updateLifecyclePool re-reads the cluster and applies mutate to the +// "default" pool's spec, retrying on a conflict from a concurrent status +// write. The cluster's own status subresource is patched by the +// multigrescluster controller on a completely separate write path, but any +// spec Update still carries the resourceVersion it read, so a status patch +// landing between our Get and our Update aborts it. +func (c *C) updateLifecyclePool( + clusterName string, + mutate func(*PoolSpec), +) { + c.Helper() + key := client.ObjectKey{Namespace: c.NS, Name: clusterName} + for { + cluster := &MultigresCluster{} + c.NoError(c.Get(key, cluster), "get cluster %s", clusterName) + pools := cluster.Spec.Databases[0].TableGroups[0].Shards[0].Spec.Pools + pool := pools[resolver.DefaultPoolName] + mutate(&pool) + pools[resolver.DefaultPoolName] = pool + + err := c.Update(cluster) + if err == nil { + return + } + if !apierrors.IsConflict(err) { + c.Fatalf("update cluster %s: %v", clusterName, err) + } + } +} + +// waitFor is eventually without the t.Fatalf, returning the last error instead. +// +// A KnownDefect body has to be able to contain a wait: KnownDefect reads a +// returned error as the pinned defect still being live, and eventually ends the +// goroutine through t.Fatalf rather than returning, so a pin whose subject is +// "this never happens" cannot be written with eventually at all. +func waitFor(timeout time.Duration, cond func() error) error { + deadline := time.Now().Add(timeout) + for { + err := cond() + if err == nil { + return nil + } + if time.Now().After(deadline) { + return err + } + time.Sleep(250 * time.Millisecond) + } +} + +// drainStateSeen reports nil once any pod in ns carries the drain state +// machine's annotation, and an error while none does. +// +// This is the first write a drain makes: initiateDrain patches the annotation +// to Requested (pkg/data-handler/drain/drain_helpers.go) before anything else +// moves, and ExecuteDrainStateMachine advances it one state per reconcile from +// there. So the annotation's mere presence is the earliest observable evidence +// that a drain started, which is exactly what step 2's pin is about, and unlike +// a count of pod events it does not depend on how many reconciles the operator +// took to get anywhere. +func (c *C) drainStateSeen() error { + c.Helper() + pods := &corev1.PodList{} + if err := c.List(pods); err != nil { + return err + } + for i := range pods.Items { + if _, ok := pods.Items[i].Annotations[metadata.AnnotationDrainState]; ok { + return nil + } + } + return fmt.Errorf("no pod in %s carries %s", c.NS, metadata.AnnotationDrainState) +} + +// poolPodFingerprints maps every pod in ns to its UID and resourceVersion. +// +// Comparing two of these across a window is how this file asserts that the +// operator left the pods alone, and it replaces what a Quiet() step used to say +// about pods before pods came out of the script's watch entirely (see the +// script's own comment in TestShardLifecycle). It is not the weaker claim: +// resourceVersion is monotonic per object and moves on every write the API +// server accepts, so any patch at all, by the operator or by the data-plane +// fake, shows up as a changed fingerprint. A pod created or deleted in the +// window changes the key set, and a delete-and-recreate at the same name +// changes the UID. What it does not do is care when any of that happened, which +// is the whole reason to read pods rather than watch them: the claim is about +// the operator's writes and not about the fake's pacing. +func (c *C) poolPodFingerprints() map[string]string { + c.Helper() + pods := &corev1.PodList{} + c.NoError(c.List(pods), "list pods in %s", c.NS) + out := make(map[string]string, len(pods.Items)) + for i := range pods.Items { + p := &pods.Items[i] + out[p.Name] = string(p.UID) + "@" + p.ResourceVersion + } + return out +} + +// podFingerprintDiff describes how two poolPodFingerprints snapshots differ, or +// returns "" when they are identical. Sorted, so a failure message is the same +// text on every run rather than whatever order the map iterated in. +func podFingerprintDiff(before, after map[string]string) string { + var notes []string + for name, was := range before { + now, ok := after[name] + switch { + case !ok: + notes = append(notes, fmt.Sprintf("%s was deleted", name)) + case now != was: + notes = append(notes, fmt.Sprintf("%s was written (%s to %s)", name, was, now)) + } + } + for name := range after { + if _, ok := before[name]; !ok { + notes = append(notes, fmt.Sprintf("%s was created", name)) + } + } + slices.Sort(notes) + return strings.Join(notes, "; ") +} + +// firstCreateIndex returns the position in ops of controller's first accepted +// create of kind/name, and whether it made one. +// +// This is how step 4's ordering claim survives pods leaving the script's watch. +// The op log's order is arrival at the recorder's mutex, which recorder.go is +// explicit is program order within one controller and not a causal order across +// controllers, so two indexes are only comparable when both name the same +// controller. Both creates here are the shard controller's, in one pass of +// createMissingResources, which is what makes the comparison sound and is why +// the controller is a parameter rather than left implicit. +func firstCreateIndex(ops []ctrltest.Op, controller, kind, name string) (int, bool) { + for i, op := range ops { + if op.Controller == controller && op.Verb == "create" && + ctrltest.KindSuffix(op.Kind) == kind && op.Key.Name == name { + return i, true + } + } + return 0, false +} + +// podRoleViolation reports whether err is one of the errors MembersOf returns +// about what status.podRoles actually says, as opposed to a transient "not +// reconciled yet" state (shard not found, status.podRoles still empty) that +// the standing invariant below must not treat as a violation. +// +// The unrecognized-role error has to count, not just the two primary-count +// ones. MembersOf returns it from its classification loop, before it counts +// primaries at all, so a snapshot holding one pod in a role this suite does +// not know (a future DRAINED, say) and two pods reporting PRIMARY comes back +// as the unrecognized-role error alone. Reading that as "not a violation" +// would wave the two primaries through, which is the one thing the invariant +// exists to catch. +// +// Matching on message text is the only option MembersOf offers, since it +// exports no sentinel errors. That coupling is invisible from identity.go, so +// rewording a message there disables this check silently; closing it needs +// either sentinels in identity.go or a unit test pinning these strings, and +// both are outside this file. +func podRoleViolation(err error) bool { + if err == nil { + return false + } + msg := err.Error() + return strings.Contains(msg, "no pod has role PRIMARY") || + strings.Contains(msg, "more than one pod has role PRIMARY") || + strings.Contains(msg, "has unrecognized role ") +} + +// rollingUpdateDrift reports whether the Shard says wantPods of its pool pods +// have drifted from their desired spec, through the RollingUpdate condition +// handleRollingUpdates writes (reconcile_pool_pods.go:735-751). +// +// This is the only object state the operator changes in response to a spec +// change it then refuses to act on: the drain it would start next is blocked +// before it writes anything to a pod, and the DisruptionBlocked Event that +// refusal records goes through client-go's per-object spam filter (burst 25, +// one refill per 300s), which a chatty Shard inside a short envtest run has +// already spent. Steps 5 and 6 need this because a step asserting that +// nothing happened asserts nothing at all unless the stimulus provably +// arrived first. +// +// The expected message is built from the same format string the operator +// uses, so this is coupled to that wording. The coupling is deliberate and +// fails in the safe direction: a reworded message makes the step fail loudly +// rather than quietly stop asserting. +func rollingUpdateDrift(key client.ObjectKey, wantPods int) error { + shard := &Shard{} + if err := Suite.Client.Get(context.Background(), key, shard); err != nil { + return err + } + cond := meta.FindStatusCondition(shard.Status.Conditions, "RollingUpdate") + if cond == nil { + return fmt.Errorf("shard %s has no RollingUpdate condition yet", key) + } + want := fmt.Sprintf("%d pods need update in pool %s", wantPods, resolver.DefaultPoolName) + if cond.Status != metav1.ConditionTrue || cond.Reason != "PodsDrifted" || + cond.Message != want { + return fmt.Errorf( + "shard %s reports RollingUpdate=%s reason=%s %q, want True PodsDrifted %q", + key, cond.Status, cond.Reason, cond.Message, want, + ) + } + return nil +} + +// lifecycleResources builds a concrete, distinguishable resource request/limit +// pair so successive calls with different label values are guaranteed to +// differ from both the resolver's own defaults and from each other. +func lifecycleResources(cpuReq, memReq, cpuLim, memLim string) corev1.ResourceRequirements { + return corev1.ResourceRequirements{ + Requests: corev1.ResourceList{ + corev1.ResourceCPU: resource.MustParse(cpuReq), + corev1.ResourceMemory: resource.MustParse(memReq), + }, + Limits: corev1.ResourceList{ + corev1.ResourceCPU: resource.MustParse(cpuLim), + corev1.ResourceMemory: resource.MustParse(memLim), + }, + } +} + +// TestShardLifecycle starts from the smallest cluster this suite can +// converge (one pool, two pods, under the two-replica floor a single-cell +// pool always resolves to) and mutates it forward: resource change, storage +// change, scale up, a second resource change with three members present, +// and scale back down. Every attribute is reached by changing a cluster +// that already exists, never by constructing the end state, so this +// exercises the operator's transition paths rather than its defaulting +// paths. +func TestShardLifecycle(t *testing.T) { + c := newCase(t) + ns := c.NS + const clusterName = "lifecycle" + + shardKey := client.ObjectKey{ + Namespace: ns, + Name: name.JoinWithConstraints( + name.DefaultConstraints, + clusterName, + string(resolver.DefaultSystemDatabaseName), + string(resolver.DefaultSystemTableGroupName), + string(lifecycleShardName), + ), + } + shardRef := lifecycleShardRef(clusterName) + const cellName = defaultSimCell + + poolName := string(resolver.DefaultPoolName) + pod0 := shardcontroller.BuildPoolPodName(shardRef, poolName, cellName, 0) + pvc0 := shardcontroller.BuildPoolDataPVCName(shardRef, poolName, cellName, 0) + pod1 := shardcontroller.BuildPoolPodName(shardRef, poolName, cellName, 1) + pvc1 := shardcontroller.BuildPoolDataPVCName(shardRef, poolName, cellName, 1) + backupPVC := shardcontroller.BuildSharedBackupPVCName(shardRef) + + // Step 1: converge, out of the script's sight, and then open the script + // over what converging produced. + // + // lifecycleCluster mirrors what MinimalCluster plus the resolver's own + // defaults would produce (one cell, one pool at the two-replica floor + // pkg/resolver/shard.go computes for a single-cell pool under the default + // AT_LEAST_2 durability policy), so "the smallest possible cluster" + // already has a primary and a replica the moment it is healthy: pods 0 + // and 1, their data PVCs, and the shard's one shared backup PVC. + // + // The watch therefore opens after its own fixture, against this suite's + // standing rule that a stream is established before the objects it + // watches are created. That rule exists so a reaction to the create + // cannot land before anyone is listening, and the events it protects here + // are ones no assertion reads: the real claims of this script are steps 2 + // through 6, each about one mutation of a cluster that already exists, + // and none of them looks at how the cluster got there. + // + // What the closed step it replaces did read was the convergence sequence + // itself, and that count is not the operator's to keep. A step's + // allow-list is an exact multiset with no "N events of this kind" form, + // so a declared count is sound only where the code fixes it. Pod + // readiness here is paced by the data-plane sim (pkg/ctrltest/datasim.go) + // racing reconcilePoolerReadiness's own readiness-gate patch on the same + // Pod, and measured over runs of this test a pod settles in three status + // modifications most of the time and four sometimes, which failed step 1 + // on an unpermitted event about one run in five. A fourth speculative + // Changed leaf would only move that boundary, since nothing bounds the + // sequence at three or four either. Do not restore the closed step: it + // pinned the harness, not the operator. + // + // RequireQuiescent is what stands in its place, and over the convergence + // it is strictly the stronger claim: no projected state change on any of + // watchedKinds() and no attempted write for five seconds, rather than a + // count of Pod and PVC events. It is also what makes the baseline below + // a fixed set rather than a race, because it establishes that nothing is + // still moving when the watch opens. + c.lifecycleCluster(clusterName) + c.Eventually(60*time.Second, "shard to report Healthy", func() error { + shard := &Shard{} + if err := c.Get(shardKey, shard); err != nil { + return err + } + if shard.Status.Phase != multigresv1alpha1.PhaseHealthy { + return fmt.Errorf("shard phase is %q", shard.Status.Phase) + } + return nil + }) + c.RequireQuiescent(5*time.Second, 60*time.Second) + + // PersistentVolumeClaim and not Pod, which is the general form of what + // step 1 above ran into rather than a second workaround for it. + // + // A closed step counts events, so it is sound only over writes whose number + // follows from the operator's code. Every PVC event in this namespace is + // one of those: the operator creates each PVC once, patches it once per + // resize, and the data-plane fake's bind is a single terminal transition + // from Pending to Bound rather than a progression, so it too is one write + // by construction. Pod status is the opposite. The fake walks a pod's + // readiness conditions in however many passes it takes while + // reconcilePoolerReadiness writes its readiness gate on the same object, + // and the number of modifications that takes is a property of that race: + // three most of the time, four often enough to fail one run in five, + // measured here twice, at step 1 and again at step 4. + // + // Note what is not available as a middle road: keeping Pod in the watch + // while declining to enumerate status modifications. There is no "N events + // of this kind" form, so a watched kind's events must each be permitted by + // name or they fail the step in flight. Watching pods therefore forces the + // enumeration, which is why pods come out of the watch altogether and every + // pod-level claim below is a direct read instead. Those reads are not the + // weaker choice; see poolPodFingerprints for why a state comparison is at + // least as strong here as an event count, and firstCreateIndex for how the + // one genuine ordering claim is made from the operator's own write log. + s := c.NewScript(&corev1.PersistentVolumeClaimList{}) + + // Standing invariant: exactly one PRIMARY whenever status.podRoles is + // non-empty, no pod in a role this suite does not know, and no pod + // quarantined. MembersOf distinguishes those errors from every other read + // failure (shard not found, podRoles still empty), and podRoleViolation is + // what keeps one of those from being reported as a violation. The script + // opens after convergence, so podRoles is populated by the time this is + // registered, but steps 2 through 6 mutate the pool and the shard rewrites + // podRoles as those land, so a read taken mid-rewrite is still ordinary. + // + // Quarantined is checked explicitly because it is its own bucket in + // Members, disjoint from Replicas: a script that only ever asserted + // primary-count and replica-count could watch a pod sit quarantined for + // its entire length without ever naming it. This script never triggers + // quarantine, so any appearance here is itself something to catch. + // + // Written as a closure and registered, rather than only registered, + // because an invariant is sampled once per event and this script now + // watches one kind: the steps that matter most to it, 2 and 5 and 6, are + // steps where the operator is refused and no event arrives at all, so + // registration alone would leave those windows unsampled. requirePodRoles + // is called directly at the end of each of them. + checkPodRoles := func() error { + members, err := MembersOf(context.Background(), c.Client(), shardKey) + if err != nil { + if podRoleViolation(err) { + return err + } + return nil + } + if len(members.Quarantined) > 0 { + return fmt.Errorf( + "pod(s) unexpectedly quarantined: %s", strings.Join(members.Quarantined, ", "), + ) + } + return nil + } + requirePodRoles := func(where string) { + c.Helper() + c.Check().NoError(checkPodRoles(), "pod roles after %s", where) + } + s.Invariant( + "shard has exactly one primary and no quarantined pod", + func(_ ctrltest.Event) error { return checkPodRoles() }, + ) + + // The watch carries no starting resourceVersion, so the API server replays + // the namespace's PVCs as "added" and this step permits exactly that + // baseline. Naming the three from the operator's own name builders, rather + // than from a List taken a moment earlier, is what keeps it an assertion: + // it says the converged pool's storage is those three claims and nothing + // else, so a fourth PVC or a misnamed one fails here instead of being + // permitted by whatever happened to exist. + s.Step("the converged pool's PVCs replay as the baseline", nil, + ctrltest.Added("PersistentVolumeClaim", pvc0), + ctrltest.Added("PersistentVolumeClaim", pvc1), + ctrltest.Added("PersistentVolumeClaim", backupPVC), + ) + + s.Step("cluster settles before any change", nil, ctrltest.Quiet()) + + // The pod half of that same claim, made as a read because pods are not + // watched: the converged pool is pods 0 and 1 and nothing else. This is + // what the baseline's two Added("Pod", ...) leaves used to say. + converged := c.poolPodFingerprints() + for _, want := range []string{pod0, pod1} { + c.HasKey(converged, want, "converged pool has no pod %s", want) + } + c.Eq(2, len(converged), "want exactly %s and %s", pod0, pod1) + + // The precondition that makes step 2 meaningful: a two-member pool, one of + // them primary, which is the cohort size the defect below is about. + // + // Read under Eventually rather than once. The convergence check above is + // paced by the data plane sim at 250ms, but status.podRoles is written + // through the pooler sim at 500ms, so both pods can be ready while the + // second pod's role has not propagated yet. A single read lands in that + // window often enough to matter: observed failing one full run in six, + // reporting one primary and zero replicas. + var initialMembers Members + c.Eventually( + 30*time.Second, + "the pool to report one primary and one replica", + func() error { + members, err := MembersOf(c.Context(), c.Client(), shardKey) + if err != nil { + return err + } + if len(members.Replicas) != 1 || members.Primary == "" { + return fmt.Errorf("got %+v", members) + } + initialMembers = members + return nil + }, + ) + + // Step 2: change CPU and memory on the pool. This is written as the + // positive assertion the brief describes, and it does not pass: the + // mutation lands in the Shard spec and podNeedsUpdate correctly flags both + // pool pods as drifted (the Shard reports RollingUpdate=True + // reason=PodsDrifted, "2 pods need update in pool default"), but + // handleRollingUpdates never drains either of them. canStartDisruption + // (disruption.go) calls posture.CheckDisruption, which calls + // consensus.CheckSufficientRecruitment against the two-pooler rule + // test/suite/fakes.go's poolerSim registers; that function's own majority + // rule is len(cohort)/2+1, which for a two-member cohort is 2, so + // excluding either pod to disrupt it always leaves 1 short. This is not + // gated by DurabilityPolicy at all: fakes.go's rule sets AtLeastN(1), not + // the cluster's AT_LEAST_2 default, so the floor here comes from the + // majority check that runs before any policy-specific one, and it blocks + // a two-member pool categorically, not just under this suite's default. + // + // This step is pinned, where steps 5 and 6 below are not, because this one + // is a claim about the operator. A single-cell pool takes ReplicasPerCell 2 + // from the resolver's own default (pkg/resolver/shard.go:102-113), and the + // operator's purpose-built escape hatch for a member that cannot be + // disrupted, reconcileCellMaintenanceSurge, is gated on + // MULTI_CELL_AT_LEAST_2 with exactly two cells + // (maintenance_surge.go:349-351), so the shape the operator defaults to + // gets no surge and no other route. The pin may never expire, which is + // worth saying plainly: it expires if the majority rule changes for a + // 2-cohort, if the single-cell default moves off 2, or if the surge gate + // widens to the condition that actually triggers it, and not otherwise. + // + // The pin is that no drain ever starts, and it is written as exactly that: + // the drain state annotation never appears on any pod in the namespace. + // What it replaced was a permitted set enumerating the whole nine-event + // cycle a drain would produce on each of the two pods, which pinned the + // same defect less precisely and could not survive the readiness leaves + // three of those nine were (see step 1). Naming the annotation is the + // better pin on its own terms: initiateDrain patches it before the drain + // machine does anything else, so a drain that started and then stalled + // halfway confirms the pin today and would be indistinguishable from + // "never started" under a count of events that never arrived. + // + // The positive half comes first and is not part of the pin. A step that + // asserts nothing happened asserts nothing at all unless the stimulus + // provably arrived, so the drift condition is waited on with eventually, + // which fails the test outright rather than confirming the pin, exactly as + // KnownDefect's contract requires of a setup step. + // + // The step's own Quiet() carries the closed-world half over PVCs: a rolling + // update rewrites pods and leaves their claims alone, so the PVC silence + // here is the assertion that the operator did not take some other action + // instead of the one it refused. + s.Step("change CPU and memory on the pool", func() error { + c.updateLifecyclePool(clusterName, func(p *PoolSpec) { + p.Postgres.Resources = lifecycleResources("100m", "128Mi", "200m", "256Mi") + }) + c.Eventually( + 30*time.Second, + "the shard to report both pool pods drifted", + func() error { + return rollingUpdateDrift(shardKey, len(initialMembers.Replicas)+1) + }, + ) + podsBefore := c.poolPodFingerprints() + c.KnownDefect("shard-lifecycle-two-member-pool-never-rolls", + func() error { + return waitFor(blockedObservationWindow, func() error { + return c.drainStateSeen() + }) + }, + ) + // Stronger than the pin and independent of it: not only did no drain + // annotation appear, no pod was written to at all while we watched. + c.Check().Eq("", podFingerprintDiff(podsBefore, c.poolPodFingerprints()), + "the refused rolling update still touched pods") + requirePodRoles("the refused rolling update") + return nil + }, ctrltest.Quiet()) + + // Step 3: change storage. expandPVCIfNeeded patches the PVC's storage + // request directly; it is not drift the pod's spec hash notices (the pod + // references the PVC by name, not by size), so no pod event follows. + // This mutation is independent of step 2's never-applied one and is + // unaffected by it. + s.Step("change storage", func() error { + c.updateLifecyclePool(clusterName, func(p *PoolSpec) { + p.Storage.Size = "2Gi" + }) + return nil + }, ctrltest.Changed("PersistentVolumeClaim", pvc0), ctrltest.Changed("PersistentVolumeClaim", pvc1)) + + // Step 4: add a replica. The closed step covers the new PVC, created once + // by the operator and bound once by the data-plane fake. The new pod is + // waited on as a read, and the one genuine code-level ordering here, that + // the PVC exists before the pod that binds it, is asserted below from the + // operator's own write log rather than from the arrival order of two + // watches. + // + // This step is where the second instance of step 1's problem was measured: + // it declared three Changed("Pod", pod2) leaves for the readiness settle + // and failed on a fourth, one run in eight, with the same + // status.conditions paths. Those leaves are gone rather than widened. + pod2 := shardcontroller.BuildPoolPodName(shardRef, poolName, cellName, 2) + pvc2 := shardcontroller.BuildPoolDataPVCName(shardRef, poolName, cellName, 2) + + s.Step("add a replica", func() error { + c.updateLifecyclePool(clusterName, func(p *PoolSpec) { + p.ReplicasPerCell = ptr.To(int32(3)) + }) + c.Eventually(30*time.Second, "new pod ready", func() error { + pod := &corev1.Pod{} + key := client.ObjectKey{Namespace: ns, Name: pod2} + if err := c.Get(key, pod); err != nil { + return err + } + for _, cond := range pod.Status.Conditions { + if cond.Type == corev1.PodReady && cond.Status == corev1.ConditionTrue { + return nil + } + } + return fmt.Errorf("pod %s not Ready yet", pod2) + }) + return nil + }, + ctrltest.Added("PersistentVolumeClaim", pvc2), + ctrltest.Changed("PersistentVolumeClaim", pvc2), + ) + + // The ordering claim, from the write log: createMissingResources creates + // the data PVC and then the pod that mounts it, in that order, within one + // pass (reconcile_pool_pods.go:178-198 and the Create that follows it). + // Both are the shard controller's writes, which is what makes their + // relative position in the log program order rather than arrival noise; + // see firstCreateIndex. + ops := Suite.Ops.OpsInNamespace(ns) + pvcAt, pvcCreated := firstCreateIndex(ops, "shard", "PersistentVolumeClaim", pvc2) + podAt, podCreated := firstCreateIndex(ops, "shard", "Pod", pod2) + c.Check().True(pvcCreated, "the shard controller never created PVC %s", pvc2) + c.Check().True(podCreated, "the shard controller never created pod %s", pod2) + // Only meaningful once both exist. firstCreateIndex reports a miss as + // index 0, so an absent pod would otherwise read as one created before + // its PVC, and the run would carry a confident ordering complaint about + // an object that was never created. The switch this replaced got that + // right by being mutually exclusive; three independent checks have to say + // it explicitly. + if pvcCreated && podCreated { + c.Check().True(pvcAt <= podAt, + "pod %s was created before the PVC %s it binds (ops %d and %d)", + pod2, pvc2, podAt, pvcAt) + } + + requirePodRoles("the scale-up") + + // The new pod must bind the new PVC, not either existing one: resolved + // through ShardPVCOf against the live pod rather than assumed from the names + // above. + c.Eventually(30*time.Second, "new pod bound to a PVC", func() error { + _, err := ShardPVCOf(c.Context(), c.Client(), ns, pod2) + return err + }) + boundTo, err := ShardPVCOf(c.Context(), c.Client(), ns, pod2) + c.NoError(err, "ShardPVCOf(%s)", pod2) + // One comparison, not two: pvc0, pvc1 and pvc2 are distinct names, so + // "it is pvc2" already says "it is not either existing PVC", and a second + // check against those two could never fire. + c.Eq(pvc2, boundTo, "pod %s is bound to the wrong PVC, want the new PVC rather than %s or %s", + pod2, pvc0, pvc1) + + // Steps 5 and 6 do not assert the guarantee this script was written to + // assert, and say so rather than implying otherwise. + // + // That guarantee, documented at + // pkg/resource-handler/controller/shard/reconcile_pool_pods.go:711, is + // that handleRollingUpdates drains drifted pods one at a time, replicas + // before the primary, with the primary's switchover requested before it is + // touched, and that handleScaleDown removes an extra replica by draining + // it. Nothing in this harness can observe any of it, because no drain ever + // starts at any cohort size. canStartDisruption calls + // posture.CheckDisruption, whose revocation check + // (consensus.CheckSufficientRecruitment) requires that the pod being + // excluded cannot satisfy the durability policy on its own, and + // test/suite/fakes.go's poolerSim registers DurabilityPolicy: + // topoclient.AtLeastN(1) regardless of the shard's configured policy, so + // any single excluded pod satisfies it alone and the check refuses every + // exclusion. handleScaleDown calls the identical gate + // (reconcile_pool_pods.go:629), which is why step 6 is in the same + // position as step 5. + // + // Neither step is a KnownDefect, and that is the point. A pin records a + // live operator defect and expires on the day the operator is fixed; this + // is a harness fidelity gap, so a pin here could never expire, which makes + // it a suppression wearing a pin's clothes. Nor is threading the shard's + // real policy through fakes.go known to be enough to reach these steps: at + // AtLeastN(2) the failure moves earlier instead, the shard loses its + // PRIMARY from status.podRoles altogether, the standing invariant above + // fires during step 4, and the operator hot-loops "No primary in podRoles, + // requeueing to re-read topology". Whether that is further harness + // infidelity (the fake models no election and no promotion, and registers + // each pooler's routing role once, at first sight) or an operator defect at + // AT_LEAST_2 with three members is unresolved, and settling it is the + // prerequisite for writing these two steps for real. + // + // What is left is the pair of facts this harness can establish, asserted + // positively: the operator sees the change, and then does nothing about it + // for as long as we watch. Both halves carry weight. Without the first the + // silence would also be satisfied by a mutation that never reached the + // shard controller at all, which is the vacuous assertion this suite + // exists to refuse; without the second there is no tripwire. The day + // either half changes, for a harness reason or an operator one, these + // steps fail and force the question open again. + // + // One note for whoever picks that up, because it is not obvious from the + // guarantee's wording: the switchover half has no Pod or PVC footprint to + // assert even in principle. handleRollingUpdates "requests a switchover" + // by calling the same initiateDrain annotation patch it uses for a replica + // (drain_helpers.go:58) and recording a RollingUpdateStarted Event on the + // Shard. So the only Kubernetes-visible difference between draining the + // primary and draining a replica is an Event, on an object this script + // does not watch, delivered over the spam-filtered path rollingUpdateDrift + // describes. Asserting it needs a different observation, not a better + // permitted set. + + // Step 5: change CPU and memory again, now with three members. Resolved + // through MembersOf rather than assumed from index, both as the + // precondition that makes the step meaningful (silence about a rolling + // update is only interesting over a pool that really does hold three + // members, one of them primary) and because the drifted-pod count + // asserted below is derived from it. + members, err := MembersOf(c.Context(), c.Client(), shardKey) + c.NoError(err, "MembersOf before step 5") + c.Eq(2, len(members.Replicas), "want two replicas before step 5, got %+v", members) + c.NotEq("", members.Primary, "want a primary before step 5, got %+v", members) + poolPods := len(members.Replicas) + 1 + + s.Step("change CPU and memory again, now with three members", func() error { + podsBefore := c.poolPodFingerprints() + c.updateLifecyclePool(clusterName, func(p *PoolSpec) { + p.Postgres.Resources = lifecycleResources("150m", "192Mi", "300m", "384Mi") + }) + // The count is what makes this non-vacuous. Step 2's change was never + // applied either, so two pods have been drifted since then and a bare + // "RollingUpdate is True" would have been satisfied before this step + // ran; only the third pod, created at step 4 from the then-current + // spec, drifts because of this mutation. + c.Eventually( + 30*time.Second, + "the shard to report every pool pod drifted", + func() error { + return rollingUpdateDrift(shardKey, poolPods) + }, + ) + // Slept inside the step's own action rather than after it, so the + // closed world covers the whole window: any event the operator + // produces while we wait is buffered by the stream and fails the + // Quiet() below as an unpermitted change. + time.Sleep(blockedObservationWindow) + // The pod half of the same silence, which the Quiet() cannot carry now + // that pods are read rather than watched. Snapshotted before the + // mutation rather than after the drift wait, so the window compared + // here is the whole step and not just its tail, which is the span the + // Quiet() covers. + c.Check().Eq("", podFingerprintDiff(podsBefore, c.poolPodFingerprints()), + "the refused rolling update still touched pods") + requirePodRoles("the refused three-member rolling update") + return nil + }, ctrltest.Quiet()) + + // Step 6: scale the pool back down, which should drain and remove one + // replica. Same shape as step 5: prove the request reached the Shard the + // controller reconciles, then watch it be refused. + // + // When this becomes assertable, do not resolve the pod being removed as + // the highest-named replica. selectShardScaleDownPod (disruption.go:144) + // selects on the pod index being at or above the desired replica count, + // and the two only agree here because this fake always elects the + // lowest-indexed pod primary: with pod 2 primary, the highest-named + // replica is pod 1 and the operator would remove pod 2. Derive it from the + // index, confirm through MembersOf that it is not the primary, and take + // its PVC from ShardPVCOf. The permitted set then wants the PVC's change before + // the pod's deletion, not after: cleanupDrainedPod patches the data PVC's + // orphan mark (reconcile_pool_pods.go:560) before the Delete at :567, and + // it patches rather than deletes because orphanByRemainingCount holds + // while the pool has three data PVCs and the threshold is 3 + // (shard_controller.go:49). + membersBeforeScaleDown, err := MembersOf(c.Context(), c.Client(), shardKey) + c.NoError(err, "MembersOf before step 6") + c.Eq(2, len(membersBeforeScaleDown.Replicas), + "want two replicas before step 6, got %+v", membersBeforeScaleDown) + c.NotEq("", membersBeforeScaleDown.Primary, + "want a primary before step 6, got %+v", membersBeforeScaleDown) + + s.Step("scale the pool back down by one replica", func() error { + podsBefore := c.poolPodFingerprints() + c.updateLifecyclePool(clusterName, func(p *PoolSpec) { + p.ReplicasPerCell = ptr.To(int32(2)) + }) + // Scale-down writes no condition of its own, so what is proved here is + // narrower than step 5's: the desired count reached the Shard spec, + // which is the input handleScaleDown reads, while three pods are still + // live for it to act on. + c.Eventually( + 30*time.Second, + "the scale-down to reach the Shard spec", + func() error { + shard := &Shard{} + if err := c.Get(shardKey, shard); err != nil { + return err + } + pool, ok := shard.Spec.Pools[resolver.DefaultPoolName] + if !ok { + return fmt.Errorf("shard %s has no %q pool", shardKey, resolver.DefaultPoolName) + } + if got := ptr.Deref(pool.ReplicasPerCell, -1); got != 2 { + return fmt.Errorf("pool %q wants %d replicas per cell, not 2", + resolver.DefaultPoolName, got) + } + return nil + }, + ) + time.Sleep(blockedObservationWindow) + // The claim step 6 exists to make, in the only form this harness can + // state it while the gate refuses every exclusion: no pod was removed, + // and no pod was written to either. When the gate opens, the primary's + // pod and PVC must still be here and the replica's must not, which is + // what the note above is about; the fingerprint comparison is the half + // of that which is assertable today, and it is resolved from the live + // pods rather than from the names, so it does not assume which pod the + // operator would have picked. + c.Check().Eq("", podFingerprintDiff(podsBefore, c.poolPodFingerprints()), + "the refused scale-down still touched pods") + requirePodRoles("the refused scale-down") + return nil + }, ctrltest.Quiet()) + + s.Finish(time.Second) +} diff --git a/test/suite/scenario_shard_quiescence_test.go b/test/suite/scenario_shard_quiescence_test.go new file mode 100644 index 00000000..dedc79ef --- /dev/null +++ b/test/suite/scenario_shard_quiescence_test.go @@ -0,0 +1,26 @@ +package suite + +import ( + "testing" + "time" +) + +// TestShardStatusQuiesces pins the fix for the shard status hot loop. It was +// written red, against the defect described below, and went green when the two +// server-side-apply defects behind it were fixed. +// +// A healthy Shard should stop writing once its status reflects reality, and it +// did not: two server-side-apply defects fought each other forever, so +// status.orchReady and status.poolsReady flipped false/true and the +// StorageClassValid condition's message alternated between two strings, each +// several times a second, with no terminal state. Keeping the measurement here +// is the point of the test: a status that converges is the property, and these +// are the fields that used to prove it did not. +func TestShardStatusQuiesces(t *testing.T) { + c := newCase(t) + cluster := c.MinimalCluster("quiesce") + + c.WaitForClusterHealthy(cluster) + + c.RequireQuiescent(10*time.Second, 30*time.Second) +} diff --git a/test/suite/scenario_thrash_test.go b/test/suite/scenario_thrash_test.go new file mode 100644 index 00000000..f4718deb --- /dev/null +++ b/test/suite/scenario_thrash_test.go @@ -0,0 +1,411 @@ +package suite + +import ( + "fmt" + "testing" + "time" + + corev1 "k8s.io/api/core/v1" + "k8s.io/utils/ptr" + "sigs.k8s.io/controller-runtime/pkg/client" + + shardcontroller "github.com/multigres/multigres-operator/pkg/resource-handler/controller/shard" + "github.com/multigres/multigres-operator/pkg/util/metadata" +) + +// TestThrash adds and removes things faster than the operator can converge, +// then asserts it lands in the right end state and stops. Both subtests +// assert only two things: what the namespace looks like once it settles, and +// that it does settle at all. +// +// Neither subtest uses a Script. A Script's steps are a closed-world +// assertion of one interleaving of events, and thrash does not produce one +// interleaving: which reconciler wins a given race, how many redundant SSA +// applies land, and how many requeues fire along the way are all legitimately +// unpredictable once changes are issued faster than the operator can react to +// them. A Script step here would be asserting an ordering nobody can +// guarantee, which is exactly the kind of flake this suite exists to stop +// writing. So the claim is only about the end state and about quiescence, +// made with ordinary reads and RequireQuiescent, not about the path taken to +// get there. Do not "improve" this into a Script. +func TestThrash(t *testing.T) { + t.Run("pool replica thrash", testPoolReplicaThrash) + t.Run("child CR thrash", testChildCRThrash) +} + +// thrashPoolName matches resolver.DefaultPoolName, the name the operator +// gives the pool it injects when a cluster specifies none. Importing +// pkg/resolver for one string constant did not seem worth the dependency, so +// this is a literal deliberately kept next to its cross-check. +const thrashPoolName = PoolName("default") + +// testPoolReplicaThrash scales one pool 1 -> 2 -> 1 -> 2 with no wait between +// changes, then requires the namespace to go quiet and every pool PVC to be +// bound to a live pod. +// +// It does not assert on this thrashed shard's status.podRoles itself: +// requirePoolScaleUpLandsRole asserts that claim deterministically on a +// namespace of its own, and asserting it here too would only be a second, +// weaker sample of the same thing. +func testPoolReplicaThrash(t *testing.T) { + c := newCase(t) + cluster := c.poolThrashCluster("pool-thrash", 1) + c.WaitForClusterHealthy(cluster) + + // Fired back to back, not waited on between calls: each Patch is an + // unconditional merge patch computed against the object's state as of the + // previous call in this loop, so it lands regardless of what the operator + // has or hasn't done with the prior one yet. That is the thrash. + for _, n := range []int32{2, 1, 2} { + c.scalePoolTo(cluster, n) + } + + // Fixed: scaling a pool from N to N+1 lands the new pooler's role in + // shard.Status.PodRoles even when the pooler registers in topology after + // the shard has already reconciled to Healthy. The shard controller's + // readiness requeue is what closes the gap, since nothing else notices a + // registration: it is a write to the topology store, with no Kubernetes + // event behind it. + // + // The pin this replaces was statistical because the defect was: whether it + // bit depended on whether the pooler registered before or after the + // reconcile that declared the shard converged, so it reproduced about a + // third of the time. Fixed, it is deterministic, so a positive assertion + // replaces what used to be a KnownDefect pin. + requirePoolScaleUpLandsRole(t) + + c.RequireQuiescent(10*time.Second, 90*time.Second) + + // The end state, asserted on this shard rather than inferred from the + // pin's namespace. Pod readiness is a Kubernetes-level fact that does + // not travel through shard.Status.PodRoles, so this is immune to the defect + // pinned above: without it a lost or reverted final scale-up settles at one + // pod, one bound PVC and a quiet namespace, and every other assertion here + // is satisfied by that. + live := c.liveReadyPoolPodNames() + c.Check().Len(live, 2, "ready pods") + + c.requireNoOrphanedPoolPVCs(live) +} + +// requirePoolScaleUpLandsRole asserts that scaling a pool from one to two +// lands the new pooler's role in status.podRoles. +// +// On its own namespace and its own cluster, with no thrash, because the +// defect this replaces never needed one: a bare single scale-up reproduced +// it, and the thrash above only found it first. +// +// Nothing wakes the shard once it is Healthy: a registration is a write to +// the topology store, with no Kubernetes event behind it. The shard +// controller now requeues on a backoff while any managed pod has not reached +// posture readiness, so the role lands well inside a second here, because the +// suite compresses requeues, so the window below is generous. +func requirePoolScaleUpLandsRole(t *testing.T) { + t.Helper() + + c := newCase(t) + cluster := c.poolThrashCluster("scaleup", 1) + c.WaitForClusterHealthy(cluster) + key := c.shardKey() + + // Hold registration before scaling, so the second pooler cannot register + // until this test says so. Without the hold the defect's precondition, a + // shard that converged having seen fewer poolers than pods, arrives only + // when registration loses a race against the last reconcile. Measured + // 2026-09-19 by disabling the fix and running this six times: three runs + // caught the regression and three did not. Constructing the state instead + // of waiting for it takes that from roughly half to always. + release := poolers.HoldRegistrations(c.NS) + defer release() + + c.scalePoolTo(cluster, 2) + + // The precondition itself, waited on rather than assumed: two pool pods + // exist and the shard has settled on a PodRoles that knows about one. A + // release before this point would prove nothing, because the reconcile + // that notices the new pooler might be one the scale-up was going to + // trigger anyway. + c.Eventually( + 60*time.Second, + "the shard to converge having seen fewer poolers than pods", + func() error { + pods := &corev1.PodList{} + if err := c.List(pods, client.MatchingLabels{ + metadata.LabelMultigresPool: string(thrashPoolName), + }); err != nil { + return err + } + if len(pods.Items) != 2 { + return fmt.Errorf("want 2 pool pods, got %d", len(pods.Items)) + } + shard := &Shard{} + if err := c.Get(key, shard); err != nil { + return err + } + if len(shard.Status.PodRoles) != 1 { + return fmt.Errorf("want 1 pod role while held, got %d", len(shard.Status.PodRoles)) + } + return nil + }, + ) + // Deliberately not RequireQuiescent. A held namespace never goes quiet + // once the fix is in, because the requeue this pins is firing on its + // backoff the whole time, so quiescence holds before the fix and cannot + // after it. A stability window is true either way: it confirms the shard + // has settled on one role rather than being mid-pass, which is all the + // release needs. + stable := time.Now().Add(3 * time.Second) + for time.Now().Before(stable) { + shard := &Shard{} + c.NoError(c.Get(key, shard), "read the shard while registration is held") + c.Eq(1, len(shard.Status.PodRoles), + "a held pooler reached status.podRoles, so the hold is not holding") + time.Sleep(250 * time.Millisecond) + } + + // Now the pooler appears, with no Kubernetes event to announce it: a + // registration is a write to etcd. Only a requeue the operator asked for + // itself can notice, which is the thing this asserts. + release() + + c.Eventually( + 60*time.Second, + "the scaled-up pooler's role to reach status.podRoles", + func() error { + members, err := MembersOf(c.Context(), c.Client(), key) + // A read failure is the check's own setup failing, not the + // convergence it is waiting for, so it fails the test immediately + // rather than retrying it silently until the timeout. + c.NoError(err, "read shard members") + if len(members.Replicas) != 1 || len(members.Quarantined) != 0 { + return fmt.Errorf("got %+v", members) + } + return nil + }, + ) +} + +func (c *C) poolThrashCluster( + name string, + replicasPerCell int32, +) *MultigresCluster { + c.Helper() + return c.newCluster(name, func(s *MultigresClusterSpec) { + s.Databases = []DatabaseConfig{{ + Name: "postgres", + Default: true, + TableGroups: []TableGroupConfig{{ + Name: "default", + Default: true, + Shards: []ShardConfig{{ + Name: "0-inf", + Spec: &ShardInlineSpec{ + Pools: map[PoolName]PoolSpec{ + thrashPoolName: poolSpecWithReplicas(replicasPerCell), + }, + }, + }}, + }}, + }} + }) +} + +func poolSpecWithReplicas(n int32) PoolSpec { + return PoolSpec{ + Type: "readWrite", + Cells: []CellName{defaultSimCell}, + ReplicasPerCell: ptr.To(n), + } +} + +// scalePoolTo patches cluster's pool to n replicas per cell via a merge patch +// against cluster's own in-memory state, not a fresh read of the server. That +// makes each call in a back-to-back thrash loop independent of whatever the +// operator has done with the previous one: the patch always states the full +// desired pool spec, so it lands regardless of the server's current state. +func (c *C) scalePoolTo(cluster *MultigresCluster, n int32) { + c.Helper() + base := cluster.DeepCopy() + pools := cluster.Spec.Databases[0].TableGroups[0].Shards[0].Spec.Pools + pools[thrashPoolName] = poolSpecWithReplicas(n) + c.NoError( + c.Patch(cluster, client.MergeFrom(base)), + "scale pool %s to %d replicas per cell", + thrashPoolName, + n, + ) +} + +// liveReadyPoolPodNames lists Ready pods belonging to the thrashed pool. +// +// This deliberately does not go through MembersOf/shard.Status.PodRoles: that +// path is the one testPoolReplicaThrash's pool-scale-up assertion covers +// directly, and a pod being live is a Kubernetes-level fact independent of +// whether the operator's own role bookkeeping has caught up to it. +func (c *C) liveReadyPoolPodNames() []string { + c.Helper() + pods := &corev1.PodList{} + c.NoError( + c.List(pods, client.MatchingLabels{metadata.LabelMultigresPool: string(thrashPoolName)}), + "list pool pods", + ) + var names []string + for i := range pods.Items { + if podReady(&pods.Items[i]) { + names = append(names, pods.Items[i].Name) + } + } + return names +} + +func podReady(p *corev1.Pod) bool { + for _, c := range p.Status.Conditions { + if c.Type == corev1.PodReady { + return c.Status == corev1.ConditionTrue + } + } + return false +} + +// requireNoOrphanedPoolPVCs asserts every PVC belonging to the thrashed pool +// is bound to one of livePods. It resolves the binding with ShardPVCOf, reading it +// off each live pod's own volumes, rather than reconstructing a PVC name from +// the pool/cell/ordinal and asserting the two strings match: a pod bound to +// the wrong PVC would pass a name-arithmetic check and fail this one. +func (c *C) requireNoOrphanedPoolPVCs(livePods []string) { + c.Helper() + + bound := map[string]bool{} + for _, pod := range livePods { + pvcName, err := ShardPVCOf(c.Context(), c.Client(), c.NS, pod) + if err != nil { + c.Fatalf("resolve PVC bound to live pod %s: %v", pod, err) + } + bound[pvcName] = true + } + + pvcs := &corev1.PersistentVolumeClaimList{} + c.NoError( + c.List(pvcs, client.MatchingLabels{metadata.LabelMultigresPool: string(thrashPoolName)}), + "list pool PVCs", + ) + for _, pvc := range pvcs.Items { + if !bound[pvc.Name] { + c.Errorf( + "PVC %s belongs to pool %s but is not bound to any live pod; live pods: %v", + pvc.Name, thrashPoolName, livePods, + ) + } + } +} + +// testChildCRThrash deletes a Shard out from under its TableGroup and lets +// the parent recreate it, three times in a row, then requires the shard to +// reconverge and the namespace to go quiet. +// +// This is the likeliest spot in the wave to find a live defect: the +// ReadyForDeletion protocol between the shard and tablegroup controllers is +// already the subject of two filed defects (a vacuous ReadyForDeletion, and +// whole-cluster teardown skipping the drain), and it found a third here, pinned +// below. +func testChildCRThrash(t *testing.T) { + c := newCase(t) + ns := c.NS + cluster := c.MinimalCluster("cr-thrash") + c.WaitForClusterHealthy(cluster) + + key := c.shardKey() + shard := &Shard{} + + for i := 0; i < 3; i++ { + c.NoError(c.Get(key, shard), "cycle %d: get shard %s", i, key.Name) + oldUID := shard.UID + c.NoError(c.Delete(shard), "cycle %d: delete shard %s", i, key.Name) + + // The only wait in this loop: for the parent to have recreated a + // replacement (a new UID at the same name), which is a mechanical + // precondition for the next delete to hit a live object rather than a + // no-op against one already gone. It is not a wait for the replacement + // to converge, and the loop does not wait for that before deleting + // again: that is the thrash. + what := fmt.Sprintf("the tablegroup to recreate Shard %s after delete #%d", key.Name, i+1) + c.Eventually(30*time.Second, what, func() error { + got := &Shard{} + if err := c.Get(key, got); err != nil { + return err + } + if got.UID == oldUID { + return fmt.Errorf("shard %s not yet recreated", key.Name) + } + return nil + }) + } + + c.NoError(c.Get(key, shard), "get final shard incarnation") + + c.WaitForClusterHealthy(cluster) + + c.Eventually(60*time.Second, "the shard to report one primary and one replica", + func() error { + members, err := MembersOf(c.Context(), c.Client(), key) + if err != nil { + return err + } + if len(members.Replicas) != 1 || len(members.Quarantined) != 0 { + return fmt.Errorf( + "want 1 primary + 1 replica + 0 quarantined, got %+v", members, + ) + } + return nil + }, + ) + + c.RequireQuiescent(10*time.Second, 90*time.Second) + + // A second live defect, found by this test: the shared backup PVC never + // has its orphan label cleared when a torn-down Shard's replacement + // reclaims it. + // + // reconcileSharedBackupPVC (reconcile_shared_infra.go) reapplies the PVC by + // server-side apply from BuildSharedBackupPVC's payload, which never + // mentions multigres.com/orphan-since, so SSA leaves that label exactly as + // cleanupShardPVCs (reconcile_deletion.go) left it during the prior + // teardown: marked orphan. Contrast the per-pool data PVC path, which + // explicitly calls pvcutil.ClearOrphan on reuse + // (reconcile_pool_pods.go:204). The shared backup PVC has no equivalent + // call anywhere in the shard controller. + // + // Net effect: after any teardown-and-recreate of a Shard whose backup PVC + // survives (WhenDeleted=Delete, which MinimalCluster sets, still only + // orphans rather than deletes it in-line, because resolvePodIndex cannot + // parse an ordinal out of a backup PVC's name-hash suffix and the !hasIndex + // arm short-circuits before pvcOrphanReplicasThreshold is consulted at + // all), the backup PVC is left labeled orphan + // forever, even though it is immediately reclaimed and stays in active use + // by the reconverged, healthy shard. The multigres-gc CronJob acts on + // exactly that label, so in a real cluster this is a live backup volume + // scheduled for deletion out from under a running shard. + backupPVCKey := client.ObjectKey{ + Namespace: ns, + Name: shardcontroller.BuildSharedBackupPVCName(shard), + } + pvc := &corev1.PersistentVolumeClaim{} + c.NoError( + c.Get(backupPVCKey, pvc), + "get shared backup PVC %s", + backupPVCKey.Name, + ) + c.KnownDefect("MGO-BACKUP-PVC-ORPHAN-STALE", func() error { + since, stale := pvc.Labels[metadata.LabelOrphan] + if !stale { + return nil + } + return fmt.Errorf( + "shared backup PVC %s still carries %s=%s from an earlier teardown, though the "+ + "shard that owns it (uid %s) has reconverged healthy: reconcileSharedBackupPVC's "+ + "server-side apply never clears the label on reuse, unlike the per-pool data PVC "+ + "path (pvcutil.ClearOrphan in reconcile_pool_pods.go)", + backupPVCKey.Name, metadata.LabelOrphan, since, shard.UID, + ) + }) +} diff --git a/test/suite/scenario_transitions_test.go b/test/suite/scenario_transitions_test.go new file mode 100644 index 00000000..74a5986d --- /dev/null +++ b/test/suite/scenario_transitions_test.go @@ -0,0 +1,466 @@ +package suite + +import ( + "fmt" + "maps" + "strings" + "testing" + "time" + + corev1 "k8s.io/api/core/v1" + "sigs.k8s.io/controller-runtime/pkg/client" + + multigresv1alpha1 "github.com/multigres/multigres-operator/api/v1alpha1" + "github.com/multigres/testkit/ctrltest" +) + +// TestTransitions is the round-trip suite: for a handful of optional +// MultigresCluster fields, set the field, then unset it, and require that the +// cluster has nothing left to do. No subtest enumerates what cleanup it +// expects to see; the closing assertion fails on any activity at all, whatever +// it is, which is what finds a leak without anyone having to name it first. +// +// That closing assertion takes one of two forms below, and the choice is per +// field rather than stylistic. Where the field's consequence lands on a plain +// corev1 kind whose event count is deterministic, it is a Script step +// permitting nothing but Quiet(). Where it does not, it is RequireQuiescent, +// which makes the same closed-world claim over the twelve kinds of +// watchedKinds() plus the write recorder rather than over the one or two kinds +// a Script could usefully watch, and which opens and closes its own watch +// instead of needing one opened before the fixture exists. A Script opened +// late over two kinds for a second is strictly weaker than the primitive it +// would be standing in for, so a field that cannot use the first form uses the +// second rather than a token version of the first. +// +// Each subtest also compares state directly across the round trip, which is +// not redundant with either form. An object created during the "set" and then +// left alone emits no event on the way back, so residue that manifests as the +// absence of an expected deletion is invisible to an event stream by +// construction, whichever kinds it watches. +// +// How far that comparison reaches differs per subtest, and none of them reach +// every kind. Backup compares PVC names and storage requests, DurabilityPolicy +// compares the TableGroup and Shard mirrors, and PVCDeletionPolicy compares its +// own field plus the namespace's PVC requests. Residue of a kind a subtest does +// not read still passes it: an orphaned ConfigMap or Service would pass all +// three. +// +// The field list came from api/v1alpha1/multigrescluster_types.go rather than +// from this task's brief, as instructed. Three of the four named fields exist +// as optional fields on MultigresClusterSpec and are covered below: +// PVCDeletionPolicy, Backup and DurabilityPolicy. The fourth, "a pool's +// Replicas", does not exist under that name: PoolSpec (shard_types.go) has no +// Replicas field, only ReplicasPerCell *int32. Per this task's brief, a field +// that does not exist under the given name is reported rather than silently +// substituted, so there is no fourth subtest here. +func TestTransitions(t *testing.T) { + t.Run("PVCDeletionPolicy", testPVCDeletionPolicyRoundTrip) + t.Run("Backup", testBackupRoundTrip) + t.Run("DurabilityPolicy", testDurabilityPolicyRoundTrip) +} + +// updateCluster applies mutate to a fresh read of the cluster and writes it +// back, failing the test rather than returning an error: a write that does not +// land is this file's own setup breaking, never an observation about the +// operator. +func (c *C) updateCluster( + cluster *MultigresCluster, + mutate func(*MultigresCluster), +) { + c.Helper() + got := &MultigresCluster{} + c.NoError(c.Get(client.ObjectKeyFromObject(cluster), got), "get cluster") + mutate(got) + c.NoError(c.Update(got), "update cluster") +} + +// pvcRequests reads every PVC in ns with the storage request it carries, for +// the snapshot-and-compare half of a round trip. Comparing the whole map +// catches a PVC that appeared, a PVC that went away and was never recreated, +// and a request that moved and stayed moved, none of which the event stream +// can report once the object stops changing. +func (c *C) pvcRequests() map[string]string { + c.Helper() + pvcs := &corev1.PersistentVolumeClaimList{} + c.NoError(c.List(pvcs), "list PVCs") + out := make(map[string]string, len(pvcs.Items)) + for _, pvc := range pvcs.Items { + out[pvc.Name] = pvc.Spec.Resources.Requests.Storage().String() + } + return out +} + +// shardBackupSizes reads the backup storage size each Shard has resolved. That +// is the value BuildSharedBackupPVC applies the shared backup PVC from +// (pool_pvc.go), so it is where a Backup write has to arrive for the PVC to +// see it. +func (c *C) shardBackupSizes() map[string]string { + c.Helper() + shards := &ShardList{} + c.NoError(c.List(shards), "list Shards") + if len(shards.Items) == 0 { + c.Fatalf("no Shards in %s to read a resolved backup size from", c.NS) + } + out := make(map[string]string, len(shards.Items)) + for _, shard := range shards.Items { + size := "" + if shard.Spec.Backup != nil && shard.Spec.Backup.Filesystem != nil { + size = shard.Spec.Backup.Filesystem.Storage.Size + } + out[shard.Name] = size + } + return out +} + +// durabilityMirrors reads the DurabilityPolicy every TableGroup and Shard in +// ns currently carries. The field's only other consumer is the topology store, +// which this suite fakes in memory (fakes.go) and which writes a policy of its +// own regardless, so these mirrored spec fields are the whole of what this +// field does that anything here can observe. +func (c *C) durabilityMirrors() map[string]string { + c.Helper() + out := map[string]string{} + tgs := &TableGroupList{} + c.NoError(c.List(tgs), "list TableGroups") + for _, tg := range tgs.Items { + out["TableGroup/"+tg.Name] = tg.Spec.DurabilityPolicy + } + shards := &ShardList{} + c.NoError(c.List(shards), "list Shards") + for _, shard := range shards.Items { + out["Shard/"+shard.Name] = shard.Spec.DurabilityPolicy + } + if len(out) == 0 { + c.Fatalf("no TableGroups or Shards in %s to read a DurabilityPolicy from", c.NS) + } + return out +} + +// testPVCDeletionPolicyRoundTrip watches only PersistentVolumeClaim. +// +// PVC is a plain corev1 kind with no status.conditions of its own, so writing +// to it never sets off the generation-bump status-condition churn that +// TableGroup and Shard produce on every spec write (see +// testDurabilityPolicyRoundTrip for where that churn made a Step-based +// assertion unusable). PVCDeletionPolicy's real consequence, +// reconcilePVCOwnerRefs (pkg/resource-handler/controller/shard/shard_controller.go), +// lands on PVCs directly, so this narrower watch still sees it, and nothing +// wider is needed to catch a leak here. +func testPVCDeletionPolicyRoundTrip(t *testing.T) { + c := newCase(t) + sc := c.NewScript(&corev1.PersistentVolumeClaimList{}) + sc.StepTimeout = 10 * time.Second + + cluster := c.MinimalCluster("pvcdp") + c.WaitForClusterHealthy(cluster) + + // Snapshotted so the round trip is checked against the namespace's PVCs and + // not only against the field's own value. A PVC created during the set and + // then left alone emits nothing on the way back, so the closing Quiet step + // cannot see it. + pvcsAtStart := c.pvcRequests() + c.RequireQuiescent(5*time.Second, 30*time.Second) + + // The script opened before MinimalCluster, per NewScript's own contract, + // so its watch replays every PVC the fixture created as an Added event + // ("an already-populated namespace is replayed as a run of added + // events"). Permitting that baseline explicitly, by listing what actually + // exists now that convergence is independently confirmed, is the + // documented way to handle it, and it is a statement about the fixture, + // not a guess about this field's behaviour. + pvcs := &corev1.PersistentVolumeClaimList{} + c.NoError(c.List(pvcs), "list PVCs") + c.NotEmpty(pvcs.Items, "MinimalCluster created no PVCs to test PVCDeletionPolicy against") + baseline := make([]ctrltest.Allow, 0, len(pvcs.Items)*2) + pvcNames := make([]string, 0, len(pvcs.Items)) + for _, pvc := range pvcs.Items { + baseline = append(baseline, + ctrltest.Added("PersistentVolumeClaim", pvc.Name), + ctrltest.Changed("PersistentVolumeClaim", pvc.Name)) + pvcNames = append(pvcNames, pvc.Name) + } + sc.Step("PVCs created and bound during initial convergence", nil, baseline...) + + sc.Step("cluster settled", nil, ctrltest.Quiet()) + + original := cluster.Spec.PVCDeletionPolicy.DeepCopy() + + consequences := make([]ctrltest.Allow, 0, len(pvcNames)) + for _, name := range pvcNames { + consequences = append(consequences, ctrltest.Changed("PersistentVolumeClaim", name)) + } + + sc.Step("set PVCDeletionPolicy to Retain/Retain", func() error { + got := &MultigresCluster{} + if err := c.Get(client.ObjectKeyFromObject(cluster), got); err != nil { + return err + } + got.Spec.PVCDeletionPolicy = &PVCDeletionPolicy{ + WhenDeleted: multigresv1alpha1.RetainPVCRetentionPolicy, + WhenScaled: multigresv1alpha1.RetainPVCRetentionPolicy, + } + return c.Update(got) + }, consequences...) + + sc.Step("settled after setting the field", nil, ctrltest.Quiet()) + + sc.Step("unset PVCDeletionPolicy", func() error { + got := &MultigresCluster{} + if err := c.Get(client.ObjectKeyFromObject(cluster), got); err != nil { + return err + } + got.Spec.PVCDeletionPolicy = nil + return c.Update(got) + }, consequences...) + + sc.Step("nothing left to do after the round trip", nil, ctrltest.Quiet()) + + sc.Finish(time.Second) + + // PVCDeletionPolicy carries a CRD-level +kubebuilder:default, so a nil + // pointer never survives the API server: MinimalCluster's explicit + // {Delete, Delete} and an omitted field both resolve to the same stored + // value. The round trip is checked against what the field actually reads + // back as, not against a literal nil. + final := &MultigresCluster{} + c.NoError(c.Get(client.ObjectKeyFromObject(cluster), final), "get final cluster") + c.Check().EqDiff(original, final.Spec.PVCDeletionPolicy, + "PVCDeletionPolicy round trip not lossless") + c.Check().EqDiff(pvcsAtStart, c.pvcRequests(), + "PVCs differ across the round trip") +} + +// testBackupRoundTrip takes Backup out to an explicit backup storage size and +// back. It is the subtest that found a defect, and the pin below is the +// finding rather than an aside. +// +// Backup's only consequence any watch in this suite can see is the shared +// backup PVC's storage request (pool_pvc.go). The rest of what the field +// feeds is the topology store, faked in memory here (fakes.go), or needs a +// Secret the caller precreates (reconcile_shared_infra.go). So the round trip +// is driven through that size, and the size is where the operator breaks. +// +// Which size, measured against this harness rather than assumed, because once +// the PVC is bound and the data-plane fake has copied its request into +// status.capacity (datasim.go) every candidate is refused for a different +// reason: +// +// - smaller, 5Gi against the resolver's 10Gi default, is refused by PVC +// validation: "spec.resources.requests.storage: Forbidden: field can not +// be less than status.capacity". That rule is unconditional in Kubernetes, +// so any user of the operator can reach it. +// - larger, 20Gi, is refused by the PersistentVolumeClaimResize admission +// plugin: "only dynamically provisioned pvc can be resized and the +// storageclass that provisions the pvc must support resize", because this +// suite creates no StorageClass. A real cluster whose class sets +// allowVolumeExpansion would accept it, so that refusal is an artifact of +// the fixture and is deliberately not what gets pinned here. +// - equal but written in another unit, 10240Mi, is canonicalised back to +// 10Gi by the API server and bumps no resourceVersion, so it is invisible +// rather than illegal. +// +// The shrink is therefore the transition worth driving: legal on the +// MultigresCluster, propagated all the way to the PVC apply, and refused +// there. +func testBackupRoundTrip(t *testing.T) { + c := newCase(t) + ns := c.NS + cluster := c.MinimalCluster("backup") + c.WaitForClusterHealthy(cluster) + c.RequireQuiescent(5*time.Second, 30*time.Second) + + pvcsBefore := c.pvcRequests() + sizesBefore := c.shardBackupSizes() + original := cluster.Spec.Backup.DeepCopy() + + c.updateCluster(cluster, func(c *MultigresCluster) { + c.Spec.Backup = &multigresv1alpha1.BackupConfig{ + Type: multigresv1alpha1.BackupTypeFilesystem, + Filesystem: &multigresv1alpha1.FilesystemBackupConfig{ + Storage: multigresv1alpha1.StorageSpec{Size: "5Gi"}, + }, + } + }) + + // The defect: the refused PVC apply returns an error from Reconcile + // (shard_controller.go), so everything after that block is skipped for as + // long as the field stays lowered, the postgres ConfigMap render, the pool + // pods, PDB sizing and reconcilePVCOwnerRefs, and controller-runtime + // retries forever. The status update is NOT skipped: updateStatus runs + // early in Reconcile, deliberately, so a wedged shard keeps reporting a + // current observedGeneration and phase. That is what makes this defect + // silent, and it is why a triage that looks for a stale status will not + // find one. A user who lowers this field wedges + // the shard's whole reconcile loop, with no clamping, no rejection at + // admission, and no condition on the Shard saying why. + // + // Pinned as non-quiescence rather than as a missing PVC event, because a + // missing event is what the API server's refusal guarantees on every run + // forever: no operator change could ever retire that pin, and a pin that + // cannot expire is a suppression. This one expires the day the operator + // clamps the value, because then the namespace goes quiet and quiescent + // returns nil. + // + // Two neighbouring fixes do not expire it through quiescence, and the + // difference is worth knowing before trusting this pin as a tracker. An + // admission refusal fails the test earlier, at the update call, so the + // suite still goes red but by another route. A fix that gives up and + // records a terminal condition only after retrying past this horizon's + // activity budget leaves the pin green, so that one has to be noticed by a + // human reading this comment rather than by the pin flipping. + // + // What this pin rests on, measured 2026-09-18 rather than assumed, because + // it used to rest on something weaker than it looked. The wedged pass makes + // several accepted no-op writes (both pg_hba and exporter-queries + // ConfigMaps, the multiorch Deployment and Service, a Shard status patch) + // before it reaches the refused PVC apply, so before the recorder counted + // rejected writes, this pin observed non-quiescence only through those + // earlier writes. Reordering updateStatus, a refactor with no behavioural + // intent, would have flipped it to "appears fixed" with the defect fully + // present. The recorder now counts the refused apply itself, which is the + // one signal the defect cannot occur without, so that particular + // reordering can no longer fool it. + // + // It is not yet true that the rejection alone carries the pin. Counting + // only rejected writes, the same wedge measured non-quiescent on one run + // and quiet on the next, 12 refused applies being right at the edge of a + // 5s window inside an 11s horizon as the retry backoff spreads them out. + // So the margin still comes from the signals combined. Widening the + // horizon is not the fix (see below); if this pin ever needs to stand on + // the rejection by itself, count refused reconcile passes directly rather + // than inferring them from a quiet window. + // + // The horizon is short on purpose. controller-runtime retries a failing + // Reconcile with exponential backoff, so the gap between failing passes + // grows without bound and a long enough horizon would let the backoff + // itself supply the quiet window while the shard is still wedged. + // Calibrated both ways on this harness, three runs each: measured from + // immediately after a legal spec write this goes quiet in about 5.3s, + // measured from immediately after this write it never goes quiet inside + // 11s, over 12 failing reconcile passes. + c.KnownDefect("MGO-BACKUP-PVC-SHRINK-WEDGES-SHARD-RECONCILE", func() error { + err := c.TryQuiescent(5*time.Second, 11*time.Second) + // A lost watch voids the measurement instead of observing the + // defect, and a non-nil error here is read as the defect still + // being present, which would hold this pin green on an unrelated + // failure. + c.True(err == nil || !strings.Contains(err.Error(), "is void"), + "quiescence over %s is void, so it is no evidence either way: %v", ns, err) + return err + }) + + // Where the value actually got to, as an executable claim rather than a + // comment, because the mechanism is easy to misread: it does reach the + // Shard, so the PVC is applied from the size the user asked for and the + // refusal happens at the API server. A fix aimed at the resolver's backup + // defaulting would land on code that is behaving correctly. + c.Eventually(15*time.Second, "the lowered size to reach every Shard", func() error { + for name, size := range c.shardBackupSizes() { + if size != "5Gi" { + return fmt.Errorf("Shard %s resolved backup size is %q", name, size) + } + } + return nil + }) + + c.updateCluster(cluster, func(c *MultigresCluster) { + c.Spec.Backup = nil + }) + + // The gate that makes the closing assertion mean something. The namespace + // is not quiet when the unset lands, so silence afterwards would be + // ambiguous between the operator having processed it and the retry backoff + // having merely grown past the window. The resolved size returning to + // where it started is positive evidence that the unset propagated back + // down to where the PVC is applied from. It is a wait, not the assertion. + c.Eventually(30*time.Second, "the unset to reach every Shard", func() error { + if got := c.shardBackupSizes(); !maps.Equal(got, sizesBefore) { + return fmt.Errorf("resolved backup sizes are %v, want %v", got, sizesBefore) + } + return nil + }) + + // The round trip's assertion: once the cluster is back to the + // configuration it started in, a converged operator has nothing left to + // do, and this is that claim over every kind the operator writes plus + // every write it makes. + c.RequireQuiescent(5*time.Second, 30*time.Second) + + c.Check().EqDiff(pvcsBefore, c.pvcRequests(), "PVCs did not round trip") + final := &MultigresCluster{} + c.NoError(c.Get(client.ObjectKeyFromObject(cluster), final), "get final cluster") + c.Check().EqDiff(original, final.Spec.Backup, "Backup round trip not lossless") +} + +// testDurabilityPolicyRoundTrip has no Script, and the reason is worth +// recording because the shape it would need does not exist in the runner. +// +// DurabilityPolicy is mirrored into TableGroup.Spec and Shard.Spec +// (builders_tablegroup.go, tablegroup/builders.go), and any spec write to +// either bumps its generation, which sends every controller watching it back +// to re-stamp its own status conditions' observedGeneration. That catch-up +// took a different number of passes on every run measured while writing this +// test: watching only TableGroup and Shard, the same single field write +// settled after 19, then 13, then 9 total events across three otherwise +// identical runs. A Step's allow-list is an exact multiset, with no "N events +// of this kind" wildcard available, so declaring one against a count that +// moves between runs would flake on this suite's own harness rather than on +// the operator, which is a worse failure than not writing the assertion at +// all. +// +// What is left for a Script to assert over those kinds is that nothing further +// happened, and RequireQuiescent asserts that strictly better: five seconds +// over twelve kinds plus the write recorder, rather than about a second over +// two, and it needs no watch opened before the fixture, which over these kinds +// is not possible to combine with an exact Step anyway. So the closing +// RequireQuiescent is this subtest's closed-world assertion, standing where +// the other form's Quiet() step stands. +func testDurabilityPolicyRoundTrip(t *testing.T) { + c := newCase(t) + cluster := c.MinimalCluster("durability") + c.WaitForClusterHealthy(cluster) + c.RequireQuiescent(5*time.Second, 30*time.Second) + + original := cluster.Spec.DurabilityPolicy + mirrorsBefore := c.durabilityMirrors() + + c.updateCluster(cluster, func(c *MultigresCluster) { + c.Spec.DurabilityPolicy = "MULTI_CELL_AT_LEAST_2" + }) + + // Not an expectation about cleanup, which this test writes none of. It is + // the guard that keeps the round trip from being vacuous: if setting the + // field moved nothing anywhere, unsetting it could not leak anything and + // the closing assertion would be proving nothing about this field. + c.Eventually(30*time.Second, "the set policy to reach every mirror", func() error { + for name, policy := range c.durabilityMirrors() { + if policy != "MULTI_CELL_AT_LEAST_2" { + return fmt.Errorf("%s carries %q", name, policy) + } + } + return nil + }) + c.RequireQuiescent(5*time.Second, 30*time.Second) + + c.updateCluster(cluster, func(c *MultigresCluster) { + c.Spec.DurabilityPolicy = original + }) + + // The mirrors returning is the state half of the round trip, and no event + // assertion can make it: a mirror left holding the set value emits nothing + // once it stops changing, so silence and correctness would be the same + // observation. Also the gate that the operator processed the unset before + // the assertion below asks for silence. + c.Eventually(30*time.Second, "the unset policy to reach every mirror", func() error { + if got := c.durabilityMirrors(); !maps.Equal(got, mirrorsBefore) { + return fmt.Errorf("mirrors are %v, want %v", got, mirrorsBefore) + } + return nil + }) + + c.RequireQuiescent(5*time.Second, 30*time.Second) + + final := &MultigresCluster{} + c.NoError(c.Get(client.ObjectKeyFromObject(cluster), final), "get final cluster") + c.Check().Eq(original, final.Spec.DurabilityPolicy, "DurabilityPolicy round trip not lossless") +} diff --git a/test/suite/shard_requeue_test.go b/test/suite/shard_requeue_test.go new file mode 100644 index 00000000..6cd6311d --- /dev/null +++ b/test/suite/shard_requeue_test.go @@ -0,0 +1,183 @@ +package suite + +import ( + "fmt" + "testing" + "time" + + "sigs.k8s.io/controller-runtime/pkg/client" + + "github.com/multigres/testkit/ctrltest" +) + +// certBootstrapBudget is how long a test will wait for the shard controller to +// finish generating pgBackRest's CA and server certificates. +// +// It is the one wait in this package deliberately larger than 30s, and it is +// sized against the step it waits on rather than against convergence. +// reconcilePgBackRestCerts generates two RSA keys, which is the most expensive +// thing this suite does and by far the most variable under -race: measured at +// 14s in one run of the suite and 75s in another, on the same machine. A +// number sized for the 14s case turns the 75s case into a red suite that says +// nothing about the operator. +const certBootstrapBudget = 2 * time.Minute + +// waitForShardPastPKI blocks until the shard controller has written a pool Pod +// in ns, which is the evidence this suite has that the controller is past +// certificate generation. +// +// It exists to keep a slow precondition out of an assertion's budget. The +// shard reconciles fourteen steps in a fixed order: the pgBackRest certificate +// step is fifth, reconcilePool is twelfth, and reconcileDataPlane, the only +// step that returns the pooler-registration requeue, is last. A test that +// starts its requeue budget before the keys exist is timing RSA keygen, and it +// goes red when the crypto was slow rather than when the operator was wrong. +// Split in two, each wait is sized against what it actually waits for. +// +// A pool Pod is the signal rather than the certificate Secrets themselves for +// two reasons. It sits between the two steps that matter, so it proves the +// keygen is behind us without depending on it being the immediately preceding +// step. And a Shard creating Pods for its pools is a more stable fact than the +// name the operator builds its Secrets from, which a test has no business +// knowing. It is read from the recorder rather than from the apiserver so that +// what is observed is the shard controller having written, not an object that +// something else could have created. +func (c *C) waitForShardPastPKI() { + c.Helper() + c.Eventually( + certBootstrapBudget, + "the shard controller to get past certificate generation", + func() error { + for _, op := range Suite.Ops.OpsInNamespace(c.NS) { + if op.Controller == "shard" && ctrltest.KindSuffix(op.Kind) == "Pod" { + return nil + } + } + return fmt.Errorf("the shard controller has written no pool Pod yet") + }, + ) +} + +// holdPoolerRegistration keeps every pooler in this case's namespace out of +// the topology store for the rest of the test. +// +// The shard controller only asks for its one-minute requeue when it finds no +// poolers registered at all. Left to the data-plane fake, which registers on +// a 500ms tick as soon as a pool pod exists, whether the shard ever sees that +// empty topology is a race: when the first pooler registers before the +// shard's first data-plane pass, the shard goes straight to "some poolers", +// skips the requeue these tests wait for, and on today's operator returns no +// requeue at all (MGO-POOL-SCALEUP-ROLE-STALE). Measured at 1 run in 20 under +// full-suite load. Holding registration makes the empty topology the only +// state the shard can see. +func (c *C) holdPoolerRegistration() { + c.Helper() + c.Cleanup(poolers.HoldRegistrations(c.NS)) +} + +// awaitShard waits for the cluster controller to create this namespace's one +// Shard and returns it. +// +// Exactly one, not at least one. Both polls this replaces went on to read +// Items[0], and the weaker form picked element zero out of a set whose size +// it never checked. Every fixture in this package declares a single +// ShardConfig and suite_test.go asserts that, so the stronger claim is true +// today and costs nothing to state. +// +// It is what happens when that stops being true that decides it. Sharding is +// this project's whole point, so a multi-shard fixture is a matter of time, +// and at that moment "at least one" stays green while silently asserting +// about whichever Shard the API server happened to return first. "Exactly +// one" fails saying it got two, which points at the fixture that changed. A +// test that wants a particular shard out of several should name it rather +// than index into a list. +func (c *C) awaitShard() Shard { + c.Helper() + var shard Shard + c.Eventually(30*time.Second, "the cluster's one Shard to exist", func() error { + shards := &ShardList{} + if err := c.List(shards); err != nil { + return err + } + if len(shards.Items) != 1 { + return fmt.Errorf("want exactly one Shard, got %d", len(shards.Items)) + } + shard = shards.Items[0] + return nil + }) + return shard +} + +// shardKey is awaitShard's key, for the callers that only need to address it. +func (c *C) shardKey() client.ObjectKey { + c.Helper() + shard := c.awaitShard() + return client.ObjectKeyFromObject(&shard) +} + +// TestShardAsksForAMinuteAwaitingPoolerRegistration is the assertion that +// replaces waiting a minute for the same information, and the reason requeue +// compression does not hide the defect it compresses. +// +// While no multipooler has registered in the topology store the shard +// controller asks to be woken in a minute. Nothing in Kubernetes watches that +// store, so in production nothing can wake it sooner, and before compression +// every convergence test in this package paid that minute whenever the last pod +// event happened to land before the data plane fake registered its poolers. +// Which test paid was a coin flip and the suite's wall time swung by a minute +// between runs for no visible reason. +// +// Compressed, the poll comes back in 50ms and the minute survives here as a +// fact about the operator. Fixing it is not this suite's job, and this +// assertion is what should fail when somebody does fix it. +// +// This test asserts about the operator and nothing else. That the suite clamps +// what it saw here is a fact about the harness and belongs to +// TestSuiteCompressesRequeues, so that a clamp change reports itself as a +// clamp change rather than as the shard controller's polling having moved. +func TestShardAsksForAMinuteAwaitingPoolerRegistration(t *testing.T) { + c := newCase(t) + c.holdPoolerRegistration() + c.MinimalCluster("requeue") + + key := c.shardKey() + c.waitForShardPastPKI() + + got := Suite.Reconciles.WaitForRequeue(t, "shard", key, time.Minute, 30*time.Second) + + // Exactly a minute, not merely at least a minute. One minute is the shard + // controller's only requeue of that length (poolerRegistrationRetryDelay + // in reconcile_data_plane.go), so the duration identifies the code path + // that the reconcile boundary itself cannot: ctrl.Result carries no + // reason, and the AwaitingPoolerRegistration reason lives on the shard's + // PostureConsistent condition, which by the time a test can read it has + // usually already moved on. + c.Check().Eq(time.Minute, got.RequestedAfter, "the shard's requeue duration") +} + +// TestSuiteCompressesRequeues is the canary on the harness half of the +// bargain, and it is live rather than a unit test for a reason the unit tests +// cannot cover: TestInterceptorCompressesRequeue proves that an interceptor +// clamps, not that this suite wired one around the real controllers. If that +// wiring is ever broken, compression dies silently, every timeout in the +// package regains its old "maybe it is just waiting" ambiguity, and nothing +// fails except runtimes nobody reads. +// +// It deliberately does not name a duration the operator chose. Any requeue +// longer than the clamp will do, so that fixing the shard controller's minute +// changes what this test observes but not whether it passes. +func TestSuiteCompressesRequeues(t *testing.T) { + c := newCase(t) + c.holdPoolerRegistration() + c.MinimalCluster("clamp") + + key := c.shardKey() + c.waitForShardPastPKI() + + got := Suite.Reconciles.WaitForRequeue(t, "shard", key, 2*ctrltest.RequeueClamp, 30*time.Second) + + c.Check().Eq(ctrltest.RequeueClamp, got.Result.RequeueAfter, + "controller-runtime's requeue-after duration") + c.Check(). + True(got.Compressed(), "Compressed() = false on a pass that asked for %s", got.RequestedAfter) +} diff --git a/test/suite/suite.go b/test/suite/suite.go new file mode 100644 index 00000000..428d78a3 --- /dev/null +++ b/test/suite/suite.go @@ -0,0 +1,223 @@ +// Package suite is the multi-controller envtest harness for this operator: one +// manager running every reconciler the operator runs, so behaviour that is a +// protocol between controllers becomes testable. +// +// The generic half lives in pkg/ctrltest. What is left here is everything that +// is about this operator specifically: its scheme, its CRDs, its cache config, +// its five reconcilers, and the data plane doubles those reconcilers need. +// +// It has no build tag. Exclusion from the ordinary test targets is by path +// filter, and the entrypoint is `make test-suite`. +package suite + +import ( + "context" + "fmt" + "path/filepath" + "time" + + "github.com/multigres/multigres/go/common/rpcclient" + appsv1 "k8s.io/api/apps/v1" + corev1 "k8s.io/api/core/v1" + networkingv1 "k8s.io/api/networking/v1" + policyv1 "k8s.io/api/policy/v1" + storagev1 "k8s.io/api/storage/v1" + "k8s.io/apimachinery/pkg/runtime" + clientgoscheme "k8s.io/client-go/kubernetes/scheme" + "k8s.io/utils/ptr" + "sigs.k8s.io/controller-runtime/pkg/client" + "sigs.k8s.io/controller-runtime/pkg/controller" + "sigs.k8s.io/controller-runtime/pkg/manager" + + multigresv1alpha1 "github.com/multigres/multigres-operator/api/v1alpha1" + "github.com/multigres/multigres-operator/pkg/cacheopts" + multigresclustercontroller "github.com/multigres/multigres-operator/pkg/cluster-handler/controller/multigrescluster" + tablegroupcontroller "github.com/multigres/multigres-operator/pkg/cluster-handler/controller/tablegroup" + "github.com/multigres/multigres-operator/pkg/data-handler/poolerclient" + cellcontroller "github.com/multigres/multigres-operator/pkg/resource-handler/controller/cell" + shardcontroller "github.com/multigres/multigres-operator/pkg/resource-handler/controller/shard" + toposervercontroller "github.com/multigres/multigres-operator/pkg/resource-handler/controller/toposerver" + "github.com/multigres/testkit/ctrltest" +) + +// OperatorNamespace stands in for the namespace the operator deploys into. The +// cache treats it specially (unfiltered), so the suite has to have one for the +// production cache config to mean anything. +const OperatorNamespace = "multigres-operator-system" + +// Suite is the suite-wide harness, booted once by TestMain. One envtest and one +// manager serve the whole package; isolate with Suite.Namespace(t). +var Suite *ctrltest.Suite + +// The operator's data plane doubles. Deliberately not suite members: every +// operator's data plane is different, so ctrltest has no place to put these and +// a test reaching for them is reaching for something about this operator. +var ( + rpc *rpcclient.FakeClient + topo *topoRegistry + poolers *poolerSim +) + +// Boot brings up envtest and the manager. The returned function tears both down +// and must run before any goroutine leak check, since the manager owns +// goroutines that only exit once its context is cancelled. +func Boot() (*ctrltest.Suite, func() error, error) { + scheme := runtime.NewScheme() + for _, add := range []func(*runtime.Scheme) error{ + clientgoscheme.AddToScheme, + multigresv1alpha1.AddToScheme, + appsv1.AddToScheme, + corev1.AddToScheme, + policyv1.AddToScheme, + networkingv1.AddToScheme, + storagev1.AddToScheme, + } { + if err := add(scheme); err != nil { + return nil, nil, fmt.Errorf("add to scheme: %w", err) + } + } + + return ctrltest.Boot(ctrltest.Options{ + Scheme: scheme, + CRDPaths: []string{filepath.Join("..", "..", "config", "crd", "bases")}, + WatchedKinds: watchedKinds(), + SimInterval: 250 * time.Millisecond, + Managers: []ctrltest.ManagerOptions{{ + Name: "multigres-operator", + CacheOptions: cacheopts.New(OperatorNamespace), + OperatorNamespace: OperatorNamespace, + // Matches main.go: the operator raises these to avoid client-side + // throttling once several controllers are reconciling at once. + QPS: 50, + Burst: 100, + Register: register, + }}, + }) +} + +// watchedKinds is every kind the operator writes in a test namespace. A kind +// missing here is a kind whose churn RequireQuiescent cannot see. +func watchedKinds() []client.ObjectList { + return []client.ObjectList{ + &multigresv1alpha1.MultigresClusterList{}, + &TopoServerList{}, + &multigresv1alpha1.CellList{}, + &TableGroupList{}, + &ShardList{}, + &corev1.PodList{}, + &appsv1.DeploymentList{}, + &appsv1.StatefulSetList{}, + &corev1.PersistentVolumeClaimList{}, + &corev1.ConfigMapList{}, + &corev1.ServiceList{}, + &policyv1.PodDisruptionBudgetList{}, + } +} + +// register wires every reconciler exactly as cmd/multigres-operator/main.go +// does, differing only in the seams a test has to fake: the topology store and +// the multipooler RPC client, and in the interceptor each one's reconcile +// boundary is wrapped in. +// +// Each reconciler is a named variable rather than the anonymous composite +// literal this used to be, and that is load bearing rather than tidying. +// SetupWithManagerReconciler substitutes the reconcile boundary and nothing +// else: the receiver stays live on the enqueue path, because map functions and +// predicates bind to it when the builder runs and call its client, and +// ShardReconciler keeps mutable state on itself. So the same pointer has to be +// both the receiver and what the interceptor delegates to. +func register(ctx context.Context, mgr manager.Manager, s *ctrltest.Suite) error { + base := mgr.GetClient() + + rpc = rpcclient.NewFakeClient() + topo = newTopoRegistry(ctx) + + // The pooler fake runs for the life of the suite, across every namespace, + // because the manager it feeds is also suite-wide. + poolers = &poolerSim{c: s.Client, rpc: rpc, topo: topo, interval: 500 * time.Millisecond} + go poolers.run(ctx) + + // Deliberately a bare option struct rather than each controller's + // production options. Every controller sets MaxConcurrentReconciles to 20 + // in its own SetupWithManager and then lets a caller-supplied + // controller.Options replace that wholesale, so this suite's options + // decide the value; it is set to 1 here rather than inherited by omission + // from the controller-runtime default, which is what used to happen. + // + // That divergence from production is load-bearing in both directions. + // It is what makes the recorder's per-controller ordering assertions + // meaningful: one reconcile goroutine per controller means writes are + // issued and recorded in program order. It is also what this suite + // therefore cannot catch, namely a controller racing itself across + // concurrent reconciles of different objects. Raising this to match + // production would silently turn every ordering assertion in the + // scenario tests into a race that fails a few times a week and reads as operator + // flakiness, and it would also break Interceptor.Ops, which identifies a + // pass's writes by an op log range that only one in-flight reconcile per + // controller can make unambiguous. + opts := controller.Options{ + SkipNameValidation: ptr.To(true), + MaxConcurrentReconciles: 1, + } + + cluster := &multigresclustercontroller.MultigresClusterReconciler{ + Client: s.Ops.For("multigrescluster", base), + Scheme: mgr.GetScheme(), + Recorder: mgr.GetEventRecorderFor("multigrescluster-controller"), + APIReader: mgr.GetAPIReader(), + CreateTopoStore: topo.ForClusterRef, + } + if err := cluster.SetupWithManagerReconciler( + mgr, s.Reconciles.Wrap("multigrescluster", cluster), opts, + ); err != nil { + return fmt.Errorf("setup multigrescluster: %w", err) + } + + tableGroup := &tablegroupcontroller.TableGroupReconciler{ + Client: s.Ops.For("tablegroup", base), + Scheme: mgr.GetScheme(), + Recorder: mgr.GetEventRecorderFor("tablegroup-controller"), + } + if err := tableGroup.SetupWithManagerReconciler( + mgr, s.Reconciles.Wrap("tablegroup", tableGroup), opts, + ); err != nil { + return fmt.Errorf("setup tablegroup: %w", err) + } + + cell := &cellcontroller.CellReconciler{ + Client: s.Ops.For("cell", base), + Scheme: mgr.GetScheme(), + Recorder: mgr.GetEventRecorderFor("cell-controller"), + } + if err := cell.SetupWithManagerReconciler( + mgr, s.Reconciles.Wrap("cell", cell), opts, + ); err != nil { + return fmt.Errorf("setup cell: %w", err) + } + + topoServer := &toposervercontroller.TopoServerReconciler{ + Client: s.Ops.For("toposerver", base), + Scheme: mgr.GetScheme(), + Recorder: mgr.GetEventRecorderFor("toposerver-controller"), + } + if err := topoServer.SetupWithManagerReconciler( + mgr, s.Reconciles.Wrap("toposerver", topoServer), opts, + ); err != nil { + return fmt.Errorf("setup toposerver: %w", err) + } + + shard := &shardcontroller.ShardReconciler{ + Client: s.Ops.For("shard", base), + Scheme: mgr.GetScheme(), + Recorder: mgr.GetEventRecorderFor("shard-controller"), + APIReader: mgr.GetAPIReader(), + PoolerClients: poolerclient.Static(rpc), + CreateTopoStore: topo.ForShard, + } + if err := shard.SetupWithManagerReconciler( + mgr, s.Reconciles.Wrap("shard", shard), opts, + ); err != nil { + return fmt.Errorf("setup shard: %w", err) + } + return nil +} diff --git a/test/suite/suite_test.go b/test/suite/suite_test.go new file mode 100644 index 00000000..2b8c8667 --- /dev/null +++ b/test/suite/suite_test.go @@ -0,0 +1,123 @@ +package suite + +import ( + "fmt" + "strings" + "testing" + "time" + + corev1 "k8s.io/api/core/v1" + apierrors "k8s.io/apimachinery/pkg/api/errors" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "sigs.k8s.io/controller-runtime/pkg/client" + + multigresv1alpha1 "github.com/multigres/multigres-operator/api/v1alpha1" +) + +// TestClusterConvergesUnderAllControllers is the harness proof: one +// MultigresCluster, five live reconcilers, and a faked data plane, reaching a +// terminal healthy state. +// +// It asserts nothing about behaviour that single-controller tests already +// cover. Its job is to fail loudly if the harness itself stops working. +func TestClusterConvergesUnderAllControllers(t *testing.T) { + c := newCase(t) + ns := c.NS + cluster := c.MinimalCluster("minimal") + + c.Eventually(30*time.Second, "child CRs to be created", func() error { + topos := &TopoServerList{} + if err := c.List(topos); err != nil { + return err + } + cells := &multigresv1alpha1.CellList{} + if err := c.List(cells); err != nil { + return err + } + tgs := &TableGroupList{} + if err := c.List(tgs); err != nil { + return err + } + shards := &ShardList{} + if err := c.List(shards); err != nil { + return err + } + if len(topos.Items) == 0 || len(cells.Items) == 0 || + len(tgs.Items) == 0 || len(shards.Items) == 0 { + return fmt.Errorf("have %d TopoServer, %d Cell, %d TableGroup, %d Shard", + len(topos.Items), len(cells.Items), len(tgs.Items), len(shards.Items)) + } + return nil + }) + + c.WaitForClusterHealthy(cluster) + + // Attribution is the other half of the harness: a test that cannot say + // which controller wrote cannot assert a protocol between controllers. + wrote := map[string]bool{} + for _, op := range Suite.Ops.OpsInNamespace(ns) { + wrote[op.Controller] = true + } + for _, name := range []string{"multigrescluster", "cell", "toposerver", "tablegroup", "shard"} { + c.Check().True(wrote[name], "no recorded writes from the %s controller; "+ + "either it never ran or attribution is broken", name) + } +} + +// TestNamespacesAreIsolated runs two clusters at once to prove the isolation +// boundary holds, since every fake behind the suite (the topology store above +// all) is shared process-wide and keyed by namespace. +func TestNamespacesAreIsolated(t *testing.T) { + t.Parallel() + + for _, name := range []string{"iso-a", "iso-b"} { + t.Run(name, func(t *testing.T) { + t.Parallel() + c := newCase(t) + ns := c.NS + cluster := c.MinimalCluster(strings.TrimPrefix(name, "iso-")) + + c.WaitForClusterHealthy(cluster) + + shards := &ShardList{} + c.NoError(c.List(shards)) + c.Len(shards.Items, 1, "in %s", ns) + }) + } +} + +// TestProductionCacheConfigIsInEffect guards the fidelity trap: the manager +// must cache exactly what cmd/multigres-operator/main.go caches. +// +// Under the production config an unlabelled Secret outside the operator's own +// namespace is invisible to the cached client, which is why the reconcilers +// carry an APIReader at all. A manager built with default cache options makes +// every cached read behave differently from production and quietly voids the +// premise that these are the real controllers wired as in main.go. +// +// This assertion is only possible because the config lives in pkg/cacheopts +// rather than being copied out of package main, which cannot be imported. +func TestProductionCacheConfigIsInEffect(t *testing.T) { + c := newCase(t) + ns := c.NS + + secret := &corev1.Secret{ + ObjectMeta: metav1.ObjectMeta{Name: "unlabelled", Namespace: ns}, + StringData: map[string]string{"password": "postgres"}, + } + c.NoError(c.Create(secret), "create secret") + key := client.ObjectKeyFromObject(secret) + + c.Eventually(30*time.Second, "the APIReader to see the secret", func() error { + return Suite.Manager("multigres-operator"). + Mgr.GetAPIReader(). + Get(c.Context(), key, &corev1.Secret{}) + }) + + err := Suite.Manager("multigres-operator"). + Mgr.GetClient(). + Get(c.Context(), key, &corev1.Secret{}) + c.True(apierrors.IsNotFound(err), + "cached client should not see an unlabelled Secret outside %s, got err=%v", + OperatorNamespace, err) +} diff --git a/test/suite/testdata/multigateway-deployment.golden.yaml b/test/suite/testdata/multigateway-deployment.golden.yaml new file mode 100644 index 00000000..c0c3211a --- /dev/null +++ b/test/suite/testdata/multigateway-deployment.golden.yaml @@ -0,0 +1,89 @@ +metadata: + labels: + app.kubernetes.io/component: multigateway + app.kubernetes.io/instance: golden-cluster + app.kubernetes.io/managed-by: multigres-operator + app.kubernetes.io/name: multigres + app.kubernetes.io/part-of: multigres + multigres.com/cell: zone1 + name: golden-cluster-zone1-multigateway-90bac2c9 + namespace: default + ownerReferences: + - apiVersion: multigres.com/v1alpha1 + blockOwnerDeletion: true + controller: true + kind: Cell + name: golden-cell + uid: golden-cell-uid +spec: + replicas: 1 + selector: + matchLabels: + app.kubernetes.io/component: multigateway + app.kubernetes.io/instance: golden-cluster + multigres.com/cell: zone1 + strategy: {} + template: + metadata: + annotations: + multigres.com/project-ref: golden-cluster + labels: + app.kubernetes.io/component: multigateway + app.kubernetes.io/instance: golden-cluster + app.kubernetes.io/managed-by: multigres-operator + app.kubernetes.io/name: multigres + app.kubernetes.io/part-of: multigres + multigres.com/cell: zone1 + spec: + containers: + - args: + - multigateway + - --http-port + - "15100" + - --grpc-port + - "15170" + - --pg-port + - "5432" + - --pg-replica-port + - "5433" + - --topo-global-server-addresses + - global-topo:2379 + - --topo-global-root + - /multigres/global + - --cell + - zone1 + - --log-level + - info + image: ghcr.io/multigres/multigres:golden-fixture + livenessProbe: + httpGet: + path: /live + port: 15100 + periodSeconds: 10 + name: multigateway + ports: + - containerPort: 15100 + name: http + protocol: TCP + - containerPort: 15170 + name: grpc + protocol: TCP + - containerPort: 5432 + name: postgres + protocol: TCP + - containerPort: 5433 + name: pg-replica + protocol: TCP + readinessProbe: + httpGet: + path: /ready + port: 15100 + periodSeconds: 5 + resources: {} + startupProbe: + failureThreshold: 30 + httpGet: + path: /ready + port: 15100 + periodSeconds: 5 +status: {} diff --git a/test/suite/types.go b/test/suite/types.go new file mode 100644 index 00000000..622c0d01 --- /dev/null +++ b/test/suite/types.go @@ -0,0 +1,48 @@ +package suite + +import multigresv1alpha1 "github.com/multigres/multigres-operator/api/v1alpha1" + +// The API types this package builds most, without the package qualifier. +// +// multigresv1alpha1.MultigresCluster is thirty-three characters and says +// "multigres" twice, and a test body that constructs a dozen API objects +// spends more width on the qualifier than on what it is asserting. +// +// Two rules keep this from becoming a dot-import in disguise, which would +// trade that width for provenance nobody can recover: +// +// - Only types, never values or functions. multigresv1alpha1.PhaseHealthy +// and multigresv1alpha1.DeletePVCRetentionPolicy stay qualified, because +// a bare PhaseHealthy in an assertion genuinely does read as though it +// could be this package's own. A composite literal names its type on the +// line above, so &Shard{} does not have the same problem. +// - Only types used three or more times here. Aliasing a type used once +// saves eighteen characters and costs the next reader a lookup. +// +// Deliberately the same names as upstream, so there is nothing to learn and +// nothing to bikeshed: this drops the qualifier and changes nothing else. +// MultigresCluster in particular keeps its full name; a bare Cluster is far +// too overloaded in a workspace where that word also means a Kubernetes +// cluster and an EKS cluster. +// +// Orthogonal to any future rename of the multigresv1alpha1 alias itself, +// which is a repo-wide question across 217 files. These read the same either +// way. +type ( + MultigresCluster = multigresv1alpha1.MultigresCluster + Shard = multigresv1alpha1.Shard + PoolSpec = multigresv1alpha1.PoolSpec + ShardList = multigresv1alpha1.ShardList + PVCDeletionPolicy = multigresv1alpha1.PVCDeletionPolicy + CellConfig = multigresv1alpha1.CellConfig + MultigresClusterSpec = multigresv1alpha1.MultigresClusterSpec + PoolName = multigresv1alpha1.PoolName + PostgresPasswordSecretRef = multigresv1alpha1.PostgresPasswordSecretRef + TableGroupList = multigresv1alpha1.TableGroupList + CellName = multigresv1alpha1.CellName + DatabaseConfig = multigresv1alpha1.DatabaseConfig + TableGroupConfig = multigresv1alpha1.TableGroupConfig + ShardConfig = multigresv1alpha1.ShardConfig + ShardInlineSpec = multigresv1alpha1.ShardInlineSpec + TopoServerList = multigresv1alpha1.TopoServerList +)