From 4e7071e1eda9ee95b939b174dde4a71244c52879 Mon Sep 17 00:00:00 2001 From: Jonathan Siegel <248302+usiegj00@users.noreply.github.com> Date: Mon, 21 Sep 2026 07:04:49 +0900 Subject: [PATCH 01/24] ci: add an end-to-end test for operator-managed mode (#176) Brings up minikube, installs the chart from the checkout and exercises a real RedisFailover with sentinel.enabled: false - master election, replication, an operator-driven failover, and the master Service endpoints afterwards. Logs are collected on failure. Operator-managed mode is the default since v4.0.0 and the recent bugs in it (#161, #167) were both found from a live cluster's logs rather than from CI. #165 covers the rollout at the integration level; this covers a running cluster. Triggers on every pull request rather than a single base branch, so a change to the workflow can be tested on its own pull request. Takes about 9 minutes. --- .github/workflows/e2e.yml | 271 ++++++++++++++++++++++++++++++++++++++ 1 file changed, 271 insertions(+) create mode 100644 .github/workflows/e2e.yml diff --git a/.github/workflows/e2e.yml b/.github/workflows/e2e.yml new file mode 100644 index 000000000..62d9643f2 --- /dev/null +++ b/.github/workflows/e2e.yml @@ -0,0 +1,271 @@ +name: E2E Tests + +# End-to-end coverage for operator-managed (Sentinel-less) mode, the default +# since v4.0.0. Brings up minikube, installs the chart from this checkout and +# exercises a real RedisFailover: master election, replication, an +# operator-driven failover, and the master Service endpoints afterwards. +# +# Runs on every pull request. A workflow that only triggered for one base +# branch could not be tested on the pull request that changed it. +on: + pull_request: + workflow_dispatch: + +jobs: + e2e-sentinel-free: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v6 + + - name: Set up minikube + uses: medyagh/setup-minikube@v0.0.21 + with: + kubernetes-version: v1.32.0 + + - name: Build operator image + run: | + docker build -t redis-operator:e2e -f docker/app/Dockerfile . + minikube image load redis-operator:e2e + + - name: Install CRD + run: kubectl apply --server-side -f manifests/databases.spotahome.com_redisfailovers.yaml + + - name: Deploy operator + run: | + helm upgrade --install redis-operator ./charts/redisoperator \ + --set image.repository=redis-operator \ + --set image.tag=e2e \ + --set image.pullPolicy=Never \ + --wait --timeout=120s + + - name: Wait for operator ready + run: | + kubectl rollout status deployment/redis-operator --timeout=60s + + - name: Create RedisFailover with sentinel disabled + run: | + kubectl apply -f - </dev/null; then + kubectl rollout status statefulset/rfr-test-no-sentinel --timeout=180s && break + fi + echo "Waiting for StatefulSet to be created... ($i/60)" + sleep 5 + done + + - name: Verify NO Sentinel resources exist + run: | + echo "Verifying Sentinel resources do NOT exist..." + + # Check Sentinel Deployment does not exist + if kubectl get deployment rfs-test-no-sentinel 2>/dev/null; then + echo "✗ Sentinel Deployment exists but should not" + exit 1 + fi + echo "✓ No Sentinel Deployment" + + # Check Sentinel Service does not exist + if kubectl get service rfs-test-no-sentinel 2>/dev/null; then + echo "✗ Sentinel Service exists but should not" + exit 1 + fi + echo "✓ No Sentinel Service" + + # Check Sentinel ConfigMap does not exist + if kubectl get configmap rfs-test-no-sentinel 2>/dev/null; then + echo "✗ Sentinel ConfigMap exists but should not" + exit 1 + fi + echo "✓ No Sentinel ConfigMap" + + - name: Verify Redis resources exist + run: | + echo "Verifying Redis resources exist..." + + kubectl get statefulset rfr-test-no-sentinel + echo "✓ Redis StatefulSet exists" + + kubectl get service rfrm-test-no-sentinel + echo "✓ Master Service exists" + + kubectl get service rfrs-test-no-sentinel + echo "✓ Slave Service exists" + + - name: Verify master is elected + run: | + echo "Checking that exactly one master exists..." + MASTER_COUNT=0 + for pod in rfr-test-no-sentinel-0 rfr-test-no-sentinel-1; do + ROLE=$(kubectl exec $pod -- redis-cli INFO replication 2>/dev/null | grep "role:" | tr -d '\r') + echo "$pod: $ROLE" + if [[ "$ROLE" == "role:master" ]]; then + MASTER_COUNT=$((MASTER_COUNT + 1)) + MASTER_POD=$pod + fi + done + + if [[ "$MASTER_COUNT" -ne 1 ]]; then + echo "✗ Expected exactly 1 master, found $MASTER_COUNT" + exit 1 + fi + echo "✓ Exactly one master: $MASTER_POD" + + - name: Write test data + run: | + echo "Writing test data to master..." + MASTER_POD="" + for pod in rfr-test-no-sentinel-0 rfr-test-no-sentinel-1; do + ROLE=$(kubectl exec $pod -- redis-cli INFO replication 2>/dev/null | grep "role:" | tr -d '\r') + if [[ "$ROLE" == "role:master" ]]; then + MASTER_POD=$pod + break + fi + done + + for i in {1..100}; do + kubectl exec $MASTER_POD -- redis-cli SET "key:$i" "value:$i" > /dev/null + done + echo "✓ Wrote 100 keys to $MASTER_POD" + + - name: Verify replication + run: | + echo "Verifying data replicated to replica..." + sleep 3 + + for pod in rfr-test-no-sentinel-0 rfr-test-no-sentinel-1; do + ROLE=$(kubectl exec $pod -- redis-cli INFO replication 2>/dev/null | grep "role:" | tr -d '\r') + KEY_COUNT=$(kubectl exec $pod -- redis-cli DBSIZE | grep -oE '[0-9]+') + echo "$pod ($ROLE): $KEY_COUNT keys" + done + + - name: Test operator-managed failover + run: | + echo "Testing operator-managed failover..." + + # Find the master + MASTER_POD="" + REPLICA_POD="" + for pod in rfr-test-no-sentinel-0 rfr-test-no-sentinel-1; do + ROLE=$(kubectl exec $pod -- redis-cli INFO replication 2>/dev/null | grep "role:" | tr -d '\r') + if [[ "$ROLE" == "role:master" ]]; then + MASTER_POD=$pod + else + REPLICA_POD=$pod + fi + done + + echo "Current master: $MASTER_POD" + echo "Current replica: $REPLICA_POD" + + # Delete the master pod to trigger failover + echo "Deleting master pod $MASTER_POD..." + kubectl delete pod $MASTER_POD + + # Wait for failover + echo "Waiting for failover (up to 60s)..." + for i in {1..20}; do + sleep 3 + # Check if the replica became master + ROLE=$(kubectl exec $REPLICA_POD -- redis-cli INFO replication 2>/dev/null | grep "role:" | tr -d '\r' || echo "") + echo " Check $i: $REPLICA_POD is $ROLE" + if [[ "$ROLE" == "role:master" ]]; then + echo "✓ Failover complete: $REPLICA_POD is now master" + break + fi + done + + # Verify final state + ROLE=$(kubectl exec $REPLICA_POD -- redis-cli INFO replication 2>/dev/null | grep "role:" | tr -d '\r') + if [[ "$ROLE" != "role:master" ]]; then + echo "✗ Failover failed: $REPLICA_POD is still $ROLE" + kubectl logs -l app.kubernetes.io/name=redisoperator --tail=50 || true + exit 1 + fi + + - name: Wait for pod recovery + run: | + echo "Waiting for deleted pod to recover..." + kubectl wait --for=condition=Ready pod/rfr-test-no-sentinel-0 --timeout=120s || true + kubectl wait --for=condition=Ready pod/rfr-test-no-sentinel-1 --timeout=120s || true + + - name: Verify data survived failover + run: | + echo "Verifying data survived failover..." + + # Find current master + for pod in rfr-test-no-sentinel-0 rfr-test-no-sentinel-1; do + if kubectl get pod $pod 2>/dev/null | grep -q Running; then + VALUE=$(kubectl exec $pod -- redis-cli GET "key:50" 2>/dev/null || echo "") + if [[ "$VALUE" == "value:50" ]]; then + echo "✓ $pod has test data (key:50 = $VALUE)" + else + echo "⚠ $pod: key:50 = '$VALUE'" + fi + fi + done + + - name: Verify master service endpoints + run: | + echo "Checking master service endpoints..." + + # Get the current master IP + MASTER_IP="" + for pod in rfr-test-no-sentinel-0 rfr-test-no-sentinel-1; do + if kubectl get pod $pod 2>/dev/null | grep -q Running; then + ROLE=$(kubectl exec $pod -- redis-cli INFO replication 2>/dev/null | grep "role:" | tr -d '\r' || echo "") + if [[ "$ROLE" == "role:master" ]]; then + MASTER_IP=$(kubectl get pod $pod -o jsonpath='{.status.podIP}') + echo "Current master: $pod ($MASTER_IP)" + break + fi + fi + done + + # Check master service endpoints + ENDPOINTS=$(kubectl get endpoints rfrm-test-no-sentinel -o jsonpath='{.subsets[0].addresses[0].ip}' 2>/dev/null || echo "") + echo "Master service endpoint: $ENDPOINTS" + + if [[ "$ENDPOINTS" == "$MASTER_IP" ]]; then + echo "✓ Master service points to correct master" + else + echo "⚠ Master service endpoint mismatch (may need more time to update)" + fi + + - name: Collect logs on failure + if: failure() + run: | + echo "=== Operator logs ===" + kubectl logs -l app.kubernetes.io/name=redisoperator --tail=100 || true + echo "=== Redis pod 0 logs ===" + kubectl logs rfr-test-no-sentinel-0 --tail=50 || true + echo "=== Redis pod 1 logs ===" + kubectl logs rfr-test-no-sentinel-1 --tail=50 || true + echo "=== Pod descriptions ===" + kubectl describe pod -l redisfailovers.databases.spotahome.com/name=test-no-sentinel || true + echo "=== Services ===" + kubectl get svc | grep test-no-sentinel || true + echo "=== Events ===" + kubectl get events --sort-by='.lastTimestamp' | tail -30 From d29e685dd8b7940ddb7bd40e3a3d597b26f90fa4 Mon Sep 17 00:00:00 2001 From: Michael Strassberger Date: Mon, 21 Sep 2026 00:20:30 +0200 Subject: [PATCH 02/24] Fix cluster_ok metric leak by handling RedisFailover deletion via a finalizer (#174) * Fix cluster_ok metric leak by handling RedisFailover deletion via a finalizer kooper v2's generic controller never calls Handle() on resource deletion: by the time its DeleteFunc fires, the object is already gone from the informer's local indexer, and the processor that resolves a dequeued key back to an object just no-ops when the key no longer resolves (see newIndexerProcessor in kooper/v2/controller/processor.go). Because of this, metrics.DeleteCluster - which correctly removes a RedisFailover's cluster_ok Prometheus gauge series - had zero callers, so cluster_ok leaked a stale series forever after every RedisFailover deletion. Add a finalizer (redisfailovers.databases.spotahome.com/finalizer) so a delete becomes an ordinary object update that Handle() does receive: the API server holds the object (DeletionTimestamp set, finalizer still present) until we remove it. Handle() now: - registers the finalizer on any RedisFailover that doesn't have it yet, before validation, so even one that never becomes valid still gets it - when DeletionTimestamp is set and the finalizer is present, calls mClient.DeleteCluster and removes just the finalizer from the list - when DeletionTimestamp is set but the finalizer is already gone, no-ops (cleanup already ran on a previous reconcile) Adds RedisFailover.PatchRedisFailoverFinalizers (service/k8s) doing a JSON merge patch against metadata.finalizers, following the same "always send the full desired value" pattern UpdateRedisFailoverStatus already uses for the status subresource. Mocks for both k8s.RedisFailover and k8s.Services are hand-patched to add the new method: the pinned mockery v2.20.0 panics on modern Go's export data, and regenerating with a newer mockery risked unrelated style-wide diff noise, so the new method was added by hand matching the existing generated style exactly. This is purely additive (new interface method, new object field usage) so it should not conflict with downstream forks that share this code's ancestry (e.g. buildio/redis-operator). Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01G4mmyo3sqLwf23m5FE2nBE * Cover PatchRedisFailoverFinalizers and Handle's finalizer error paths Codecov flagged this PR's patch coverage at 40%: PatchRedisFailoverFinalizers (service/k8s/redisfailover.go) had zero test coverage, and Handle's two new error-propagation branches (finalizer-add failure, deletion-cleanup finalizer- removal failure) were untested. - TestRedisFailoverServicePatchRedisFailoverFinalizers (service/k8s): success, nil-finalizers-becomes-empty-array, and not-found-returns-error, mirroring the existing UpdateRedisFailoverStatus test's fake-clientset style. - TestHandleFinalizerRegistrationErrorPropagates: a failed finalizer-add patch stops Handle before Validate/Ensure/CheckAndHeal. - TestHandleDeletionFinalizerRemovalErrorPropagates: a failed finalizer-removal patch during deletion cleanup is still propagated, after DeleteCluster has already fired. Handle and PatchRedisFailoverFinalizers are now both at 100% coverage (verified with go tool cover -func). go build/vet/test and gofmt all pass. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01G4mmyo3sqLwf23m5FE2nBE --------- Co-authored-by: Claude --- mocks/operator/redisfailover/RedisFailover.go | 14 ++ mocks/service/k8s/Services.go | 14 ++ operator/redisfailover/handler.go | 38 ++++ operator/redisfailover/handler_test.go | 201 +++++++++++++++++- service/k8s/redisfailover.go | 29 +++ service/k8s/redisfailover_test.go | 45 ++++ 6 files changed, 339 insertions(+), 2 deletions(-) diff --git a/mocks/operator/redisfailover/RedisFailover.go b/mocks/operator/redisfailover/RedisFailover.go index 05ecd735f..4bdad7543 100644 --- a/mocks/operator/redisfailover/RedisFailover.go +++ b/mocks/operator/redisfailover/RedisFailover.go @@ -76,6 +76,20 @@ func (_m *RedisFailover) UpdateRedisFailoverStatus(ctx context.Context, namespac _m.Called(ctx, namespace, redisFailover, opts) } +// PatchRedisFailoverFinalizers provides a mock function with given fields: ctx, namespace, name, finalizers, opts +func (_m *RedisFailover) PatchRedisFailoverFinalizers(ctx context.Context, namespace string, name string, finalizers []string, opts v1.PatchOptions) error { + ret := _m.Called(ctx, namespace, name, finalizers, opts) + + var r0 error + if rf, ok := ret.Get(0).(func(context.Context, string, string, []string, v1.PatchOptions) error); ok { + r0 = rf(ctx, namespace, name, finalizers, opts) + } else { + r0 = ret.Error(0) + } + + return r0 +} + type mockConstructorTestingTNewRedisFailover interface { mock.TestingT Cleanup(func()) diff --git a/mocks/service/k8s/Services.go b/mocks/service/k8s/Services.go index 355722d67..edb46af89 100644 --- a/mocks/service/k8s/Services.go +++ b/mocks/service/k8s/Services.go @@ -733,6 +733,20 @@ func (_m *Services) ListStatefulSets(namespace string) (*appsv1.StatefulSetList, return r0, r1 } +// PatchRedisFailoverFinalizers provides a mock function with given fields: ctx, namespace, name, finalizers, opts +func (_m *Services) PatchRedisFailoverFinalizers(ctx context.Context, namespace string, name string, finalizers []string, opts metav1.PatchOptions) error { + ret := _m.Called(ctx, namespace, name, finalizers, opts) + + var r0 error + if rf, ok := ret.Get(0).(func(context.Context, string, string, []string, metav1.PatchOptions) error); ok { + r0 = rf(ctx, namespace, name, finalizers, opts) + } else { + r0 = ret.Error(0) + } + + return r0 +} + // UpdateConfigMap provides a mock function with given fields: namespace, configMap func (_m *Services) UpdateConfigMap(namespace string, configMap *v1.ConfigMap) error { ret := _m.Called(namespace, configMap) diff --git a/operator/redisfailover/handler.go b/operator/redisfailover/handler.go index 65aedd01f..fa40d7b77 100644 --- a/operator/redisfailover/handler.go +++ b/operator/redisfailover/handler.go @@ -4,6 +4,7 @@ import ( "context" "fmt" "regexp" + "slices" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" "k8s.io/apimachinery/pkg/runtime" @@ -20,6 +21,20 @@ const ( rfLabelManagedByKey = "app.kubernetes.io/managed-by" rfLabelNameKey = "redisfailovers.databases.spotahome.com/name" skipReconcileAnnotation = "redisfailovers.databases.spotahome.com/skip-reconcile" + // redisFailoverFinalizer is what makes RedisFailover deletion visible to + // Handle at all. Without a finalizer, kooper's generic controller never + // calls Handle for a delete: by the time its DeleteFunc fires, the + // object is already gone from the informer's local indexer, and the + // processor that resolves a queued key back to an object just no-ops + // when the key no longer resolves (see newIndexerProcessor in + // kooper/v2/controller/processor.go) - Handle is never invoked with a + // nil/absent object standing in for "this was deleted". A finalizer + // makes the API server hold the object (with DeletionTimestamp set) + // until we remove it, which turns "delete" into an ordinary object we + // still see via Handle, and gives us a hook to clean up state that + // only exists outside the object itself, i.e. the cluster_ok metrics + // series (see the DeletionTimestamp branch in Handle). + redisFailoverFinalizer = "redisfailovers.databases.spotahome.com/finalizer" ) var ( @@ -60,6 +75,29 @@ func (r *RedisFailoverHandler) Handle(_ context.Context, obj runtime.Object) err return fmt.Errorf("can't handle the received object: not a redisfailover") } + // Deletion cleanup and finalizer registration run before anything else, + // including Validate(): an object that never passes validation must + // still get a finalizer (so its eventual deletion is observable here) + // and must still have its metrics cleaned up on the way out. + if rf.DeletionTimestamp != nil { + if !slices.Contains(rf.Finalizers, redisFailoverFinalizer) { + // Finalizer already removed (or never added) - nothing left to do. + return nil + } + r.mClient.DeleteCluster(rf.Namespace, rf.Name) + remaining := slices.DeleteFunc(slices.Clone(rf.Finalizers), func(f string) bool { + return f == redisFailoverFinalizer + }) + return r.k8sservice.PatchRedisFailoverFinalizers(context.Background(), rf.Namespace, rf.Name, remaining, metav1.PatchOptions{}) + } + + if !slices.Contains(rf.Finalizers, redisFailoverFinalizer) { + finalizers := append(slices.Clone(rf.Finalizers), redisFailoverFinalizer) + if err := r.k8sservice.PatchRedisFailoverFinalizers(context.Background(), rf.Namespace, rf.Name, finalizers, metav1.PatchOptions{}); err != nil { + return err + } + } + if rf.Annotations[skipReconcileAnnotation] == "true" { r.logger.WithField("redisfailover", rf.Name).WithField("namespace", rf.Namespace). Infof("skip-reconcile annotation set to true, skipping reconciliation") diff --git a/operator/redisfailover/handler_test.go b/operator/redisfailover/handler_test.go index 1a3445a26..0677667ec 100644 --- a/operator/redisfailover/handler_test.go +++ b/operator/redisfailover/handler_test.go @@ -4,9 +4,11 @@ import ( "context" "errors" "testing" + "time" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/mock" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" "k8s.io/utils/ptr" redisfailoverv1 "github.com/saremox/redis-operator/api/redisfailover/v1" @@ -18,6 +20,22 @@ import ( rfOperator "github.com/saremox/redis-operator/operator/redisfailover" ) +// redisFailoverFinalizerMirror mirrors the unexported constant of the same +// name defined in operator/redisfailover/handler.go. +const redisFailoverFinalizerMirror = "redisfailovers.databases.spotahome.com/finalizer" + +// fakeRecorder wraps metrics.Dummy (whose concrete type is unexported, so it +// can't be embedded directly) and records DeleteCluster calls, which is the +// one signal these tests need to observe. +type fakeRecorder struct { + metrics.Recorder + deleteClusterCalls []string // "namespace/name" per call +} + +func (f *fakeRecorder) DeleteCluster(namespace, name string) { + f.deleteClusterCalls = append(f.deleteClusterCalls, namespace+"/"+name) +} + // skipReconcileAnnotationKey mirrors the unexported constant of the same name // defined in operator/redisfailover/handler.go. const skipReconcileAnnotationKey = "redisfailovers.databases.spotahome.com/skip-reconcile" @@ -74,12 +92,17 @@ func TestHandleSkipReconcileAnnotation(t *testing.T) { config := generateConfig() mk := &mK8SService.Services{} // CheckAndHeal always defers updateStatus, on every return path; - // only reached when reconciliation isn't skipped. - mk.On("UpdateRedisFailoverStatus", mock.Anything, mock.Anything, mock.Anything, mock.Anything).Return() + // only reached when reconciliation isn't skipped, hence Maybe(). + mk.On("UpdateRedisFailoverStatus", mock.Anything, mock.Anything, mock.Anything, mock.Anything).Maybe().Return() mrfc := &mRFService.RedisFailoverCheck{} mrfh := &mRFService.RedisFailoverHeal{} mrfs := &mRFService.RedisFailoverClient{} + // Finalizer registration runs before the skip-reconcile check, so + // every case (skipped or not) triggers it on this fixture, which + // starts with no finalizers. + mk.On("PatchRedisFailoverFinalizers", mock.Anything, rf.Namespace, rf.Name, mock.Anything, mock.Anything).Once().Return(nil) + if !test.expectSkip { // Minimal Ensure() expectations for bootstrapping without exporter // or sentinels (see TestEnsure in ensurer_test.go). @@ -113,6 +136,7 @@ func TestHandleSkipReconcileAnnotation(t *testing.T) { mrfs.AssertExpectations(t) mrfc.AssertExpectations(t) } + mk.AssertExpectations(t) }) } } @@ -149,12 +173,17 @@ func TestHandleValidateError(t *testing.T) { mrfh := &mRFService.RedisFailoverHeal{} mrfs := &mRFService.RedisFailoverClient{} + // Finalizer registration runs before Validate(), so it's still expected + // even though this RF fails validation. + mk.On("PatchRedisFailoverFinalizers", mock.Anything, rf.Namespace, rf.Name, mock.Anything, mock.Anything).Once().Return(nil) + handler := rfOperator.NewRedisFailoverHandler(config, mrfs, mrfc, mrfh, mk, metrics.Dummy, log.Dummy) err := handler.Handle(context.Background(), rf) assert.Error(err) mrfs.AssertNotCalled(t, "EnsureNotPresentRedisService", mock.Anything) mrfc.AssertNotCalled(t, "IsRedisRunning", mock.Anything) + mk.AssertExpectations(t) } // TestHandleEnsureError verifies that Handle propagates an error from Ensure @@ -174,6 +203,7 @@ func TestHandleEnsureError(t *testing.T) { // Only the very first Ensure() call is mocked, and it fails - nothing // after it (in Ensure or CheckAndHeal) should ever be invoked. mrfs.On("EnsureNotPresentRedisService", rf).Once().Return(ensureErr) + mk.On("PatchRedisFailoverFinalizers", mock.Anything, rf.Namespace, rf.Name, mock.Anything, mock.Anything).Once().Return(nil) handler := rfOperator.NewRedisFailoverHandler(config, mrfs, mrfc, mrfh, mk, metrics.Dummy, log.Dummy) err := handler.Handle(context.Background(), rf) @@ -181,6 +211,7 @@ func TestHandleEnsureError(t *testing.T) { assert.Equal(ensureErr, err) mrfc.AssertNotCalled(t, "IsRedisRunning", mock.Anything) mrfs.AssertExpectations(t) + mk.AssertExpectations(t) } // TestHandleCheckAndHealError verifies that Handle propagates an error from @@ -212,6 +243,7 @@ func TestHandleCheckAndHealError(t *testing.T) { mrfs.On("EnsureRedisReadinessConfigMap", rf, mock.Anything, mock.Anything).Once().Return(nil) mrfs.On("EnsureRedisConfigMap", rf, mock.Anything, mock.Anything).Once().Return(nil) mrfs.On("EnsureRedisStatefulset", rf, mock.Anything, mock.Anything).Once().Return(nil) + mk.On("PatchRedisFailoverFinalizers", mock.Anything, rf.Namespace, rf.Name, mock.Anything, mock.Anything).Once().Return(nil) // CheckAndHeal routes to checkAndHealOperatorManagedMode and fails at // GetNumberMasters. @@ -224,6 +256,7 @@ func TestHandleCheckAndHealError(t *testing.T) { assert.Equal(checkErr, err) mrfs.AssertExpectations(t) mrfc.AssertExpectations(t) + mk.AssertExpectations(t) } // rfLabelManagedByKeyMirror, rfLabelNameKeyMirror and operatorNameMirror @@ -311,6 +344,7 @@ func TestHandleGetLabelsWhitelistFiltering(t *testing.T) { mrfs.On("EnsureRedisShutdownConfigMap", rf, mock.Anything, mock.Anything).Once().Return(nil) mrfs.On("EnsureRedisReadinessConfigMap", rf, mock.Anything, mock.Anything).Once().Return(nil) mrfs.On("EnsureRedisStatefulset", rf, mock.Anything, mock.Anything).Once().Return(nil) + mk.On("PatchRedisFailoverFinalizers", mock.Anything, rf.Namespace, rf.Name, mock.Anything, mock.Anything).Once().Return(nil) mrfc.On("IsRedisRunning", rf).Once().Return(false) @@ -332,10 +366,168 @@ func TestHandleGetLabelsWhitelistFiltering(t *testing.T) { mrfs.AssertExpectations(t) mrfc.AssertExpectations(t) + mk.AssertExpectations(t) }) } } +// TestHandleAddsFinalizerOnFreshRF verifies that Handle registers +// redisFailoverFinalizer on a RedisFailover that doesn't yet have it, adding +// it to whatever finalizers were already present rather than replacing them. +func TestHandleAddsFinalizerOnFreshRF(t *testing.T) { + assert := assert.New(t) + + rf := generateRF(false, true) + rf.Finalizers = []string{"some.other/finalizer"} + + config := generateConfig() + mk := &mK8SService.Services{} + mrfc := &mRFService.RedisFailoverCheck{} + mrfh := &mRFService.RedisFailoverHeal{} + mrfs := &mRFService.RedisFailoverClient{} + + mk.On("PatchRedisFailoverFinalizers", mock.Anything, rf.Namespace, rf.Name, + []string{"some.other/finalizer", redisFailoverFinalizerMirror}, mock.Anything).Once().Return(nil) + // CheckAndHeal always defers updateStatus, on every return path. + mk.On("UpdateRedisFailoverStatus", mock.Anything, mock.Anything, mock.Anything, mock.Anything).Return() + + mrfs.On("EnsureNotPresentRedisService", rf).Once().Return(nil) + mrfs.On("EnsureNotPresentSentinelResources", rf).Once().Return(nil) + mrfs.On("EnsureRedisMasterService", rf, mock.Anything, mock.Anything).Once().Return(nil) + mrfs.On("EnsureRedisSlaveService", rf, mock.Anything, mock.Anything).Once().Return(nil) + mrfs.On("EnsureRedisConfigMap", rf, mock.Anything, mock.Anything).Once().Return(nil) + mrfs.On("EnsureRedisShutdownConfigMap", rf, mock.Anything, mock.Anything).Once().Return(nil) + mrfs.On("EnsureRedisReadinessConfigMap", rf, mock.Anything, mock.Anything).Once().Return(nil) + mrfs.On("EnsureRedisStatefulset", rf, mock.Anything, mock.Anything).Once().Return(nil) + mrfc.On("IsRedisRunning", rf).Once().Return(false) + + handler := rfOperator.NewRedisFailoverHandler(config, mrfs, mrfc, mrfh, mk, metrics.Dummy, log.Dummy) + err := handler.Handle(context.Background(), rf) + + assert.NoError(err) + mk.AssertExpectations(t) + mrfs.AssertExpectations(t) + mrfc.AssertExpectations(t) +} + +// TestHandleDeletionCleansUpMetricsAndRemovesFinalizer verifies that Handle, +// when given a RedisFailover with a DeletionTimestamp and the finalizer still +// present, cleans up its cluster_ok metrics series and removes the finalizer +// (leaving any other finalizers intact) without ever calling Ensure or +// CheckAndHeal. +func TestHandleDeletionCleansUpMetricsAndRemovesFinalizer(t *testing.T) { + assert := assert.New(t) + + rf := generateRF(false, true) + now := metav1.NewTime(time.Now()) + rf.DeletionTimestamp = &now + rf.Finalizers = []string{"some.other/finalizer", redisFailoverFinalizerMirror} + + config := generateConfig() + mk := &mK8SService.Services{} + mrfc := &mRFService.RedisFailoverCheck{} + mrfh := &mRFService.RedisFailoverHeal{} + mrfs := &mRFService.RedisFailoverClient{} + mClient := &fakeRecorder{Recorder: metrics.Dummy} + + mk.On("PatchRedisFailoverFinalizers", mock.Anything, rf.Namespace, rf.Name, + []string{"some.other/finalizer"}, mock.Anything).Once().Return(nil) + + handler := rfOperator.NewRedisFailoverHandler(config, mrfs, mrfc, mrfh, mk, mClient, log.Dummy) + err := handler.Handle(context.Background(), rf) + + assert.NoError(err) + assert.Equal([]string{rf.Namespace + "/" + rf.Name}, mClient.deleteClusterCalls) + mk.AssertExpectations(t) + mrfs.AssertNotCalled(t, "EnsureNotPresentRedisService", mock.Anything) + mrfc.AssertNotCalled(t, "IsRedisRunning", mock.Anything) +} + +// TestHandleDeletionWithoutFinalizerIsNoop verifies that Handle does nothing +// (no metrics cleanup, no finalizer patch) for a RedisFailover that already +// has its DeletionTimestamp set but no longer carries redisFailoverFinalizer - +// i.e. cleanup already ran on a previous reconcile. +func TestHandleDeletionWithoutFinalizerIsNoop(t *testing.T) { + assert := assert.New(t) + + rf := generateRF(false, true) + now := metav1.NewTime(time.Now()) + rf.DeletionTimestamp = &now + rf.Finalizers = []string{"some.other/finalizer"} + + config := generateConfig() + mk := &mK8SService.Services{} + mrfc := &mRFService.RedisFailoverCheck{} + mrfh := &mRFService.RedisFailoverHeal{} + mrfs := &mRFService.RedisFailoverClient{} + mClient := &fakeRecorder{Recorder: metrics.Dummy} + + handler := rfOperator.NewRedisFailoverHandler(config, mrfs, mrfc, mrfh, mk, mClient, log.Dummy) + err := handler.Handle(context.Background(), rf) + + assert.NoError(err) + assert.Empty(mClient.deleteClusterCalls) + mk.AssertNotCalled(t, "PatchRedisFailoverFinalizers", mock.Anything, mock.Anything, mock.Anything, mock.Anything, mock.Anything) + mrfs.AssertNotCalled(t, "EnsureNotPresentRedisService", mock.Anything) + mrfc.AssertNotCalled(t, "IsRedisRunning", mock.Anything) +} + +// TestHandleFinalizerRegistrationErrorPropagates verifies that Handle stops +// and returns the error when adding the finalizer to a fresh RedisFailover +// fails, without going on to Validate/Ensure/CheckAndHeal. +func TestHandleFinalizerRegistrationErrorPropagates(t *testing.T) { + assert := assert.New(t) + + rf := generateRF(false, true) + patchErr := errors.New("patch boom") + + config := generateConfig() + mk := &mK8SService.Services{} + mrfc := &mRFService.RedisFailoverCheck{} + mrfh := &mRFService.RedisFailoverHeal{} + mrfs := &mRFService.RedisFailoverClient{} + + mk.On("PatchRedisFailoverFinalizers", mock.Anything, rf.Namespace, rf.Name, + []string{redisFailoverFinalizerMirror}, mock.Anything).Once().Return(patchErr) + + handler := rfOperator.NewRedisFailoverHandler(config, mrfs, mrfc, mrfh, mk, metrics.Dummy, log.Dummy) + err := handler.Handle(context.Background(), rf) + + assert.Equal(patchErr, err) + mk.AssertExpectations(t) + mrfs.AssertNotCalled(t, "EnsureNotPresentRedisService", mock.Anything) + mrfc.AssertNotCalled(t, "IsRedisRunning", mock.Anything) +} + +// TestHandleDeletionFinalizerRemovalErrorPropagates verifies that Handle +// propagates an error from removing the finalizer during deletion cleanup, +// after DeleteCluster has already been called. +func TestHandleDeletionFinalizerRemovalErrorPropagates(t *testing.T) { + assert := assert.New(t) + + rf := generateRF(false, true) + now := metav1.NewTime(time.Now()) + rf.DeletionTimestamp = &now + rf.Finalizers = []string{redisFailoverFinalizerMirror} + patchErr := errors.New("patch boom") + + config := generateConfig() + mk := &mK8SService.Services{} + mrfc := &mRFService.RedisFailoverCheck{} + mrfh := &mRFService.RedisFailoverHeal{} + mrfs := &mRFService.RedisFailoverClient{} + mClient := &fakeRecorder{Recorder: metrics.Dummy} + + mk.On("PatchRedisFailoverFinalizers", mock.Anything, rf.Namespace, rf.Name, []string{}, mock.Anything).Once().Return(patchErr) + + handler := rfOperator.NewRedisFailoverHandler(config, mrfs, mrfc, mrfh, mk, mClient, log.Dummy) + err := handler.Handle(context.Background(), rf) + + assert.Equal(patchErr, err) + assert.Equal([]string{rf.Namespace + "/" + rf.Name}, mClient.deleteClusterCalls) + mk.AssertExpectations(t) +} + // TestHandleRecordsClusterMetrics verifies Handle reports the RedisFailover's // health via mClient.SetClusterOK/SetClusterError - the signal actually used // to know a failover succeeded or failed - rather than just exercising these @@ -396,6 +588,11 @@ func TestHandleRecordsClusterMetrics(t *testing.T) { mrfs := &mRFService.RedisFailoverClient{} test.setup(rf, mrfs, mrfc) + // Finalizer registration runs before validation/Ensure/CheckAndHeal, + // so every case here triggers it on this fixture, which starts with + // no finalizers. + mk.On("PatchRedisFailoverFinalizers", mock.Anything, rf.Namespace, rf.Name, mock.Anything, mock.Anything).Once().Return(nil) + mrec := &mMetrics.Recorder{} mrec.On(test.wantMethod, rf.Namespace, rf.Name).Once() if test.extraMetricsStub != nil { diff --git a/service/k8s/redisfailover.go b/service/k8s/redisfailover.go index 4d6c66ded..15278ce64 100644 --- a/service/k8s/redisfailover.go +++ b/service/k8s/redisfailover.go @@ -22,6 +22,10 @@ type RedisFailover interface { // WatchRedisFailovers watches the redisfailovers on a cluster. WatchRedisFailovers(ctx context.Context, namespace string, opts metav1.ListOptions) (watch.Interface, error) UpdateRedisFailoverStatus(ctx context.Context, namespace string, redisFailover *redisfailoverv1.RedisFailover, opts metav1.PatchOptions) + // PatchRedisFailoverFinalizers replaces a RedisFailover's finalizers list + // with the given one. Finalizers live under metadata, not the status + // subresource, so this can't go through UpdateRedisFailoverStatus. + PatchRedisFailoverFinalizers(ctx context.Context, namespace string, name string, finalizers []string, opts metav1.PatchOptions) error } // RedisFailoverService is the RedisFailover service implementation using API calls to kubernetes. @@ -82,3 +86,28 @@ func (r *RedisFailoverService) UpdateRedisFailoverStatus(ctx context.Context, na r.logger.Errorf("Error while patching RedisFailover status %s/%s : %s", rf.Namespace, rf.Name, err.Error()) } } + +// PatchRedisFailoverFinalizers satisfies redisfailover.Service interface. +// A JSON merge patch replaces the whole finalizers array, so the caller must +// pass the complete list it wants the object to end up with (add/remove +// against the finalizers it read, not just the one entry it cares about) - +// same reasoning as UpdateRedisFailoverStatus always sending all three +// status fields. +func (r *RedisFailoverService) PatchRedisFailoverFinalizers(ctx context.Context, namespace string, name string, finalizers []string, opts metav1.PatchOptions) error { + if finalizers == nil { + finalizers = []string{} + } + patch := map[string]interface{}{ + "metadata": map[string]interface{}{ + "finalizers": finalizers, + }, + } + patchBytes, _ := json.Marshal(patch) + + _, err := r.k8sCli.DatabasesV1().RedisFailovers(namespace).Patch(ctx, name, types.MergePatchType, patchBytes, opts) + recordMetrics(namespace, "RedisFailover", metrics.NOT_APPLICABLE, "PATCH", err, r.metricsRecorder) + if err != nil { + r.logger.Errorf("Error while patching RedisFailover finalizers %s/%s : %s", namespace, name, err.Error()) + } + return err +} diff --git a/service/k8s/redisfailover_test.go b/service/k8s/redisfailover_test.go index 890b7cc3c..77aff01ce 100644 --- a/service/k8s/redisfailover_test.go +++ b/service/k8s/redisfailover_test.go @@ -134,3 +134,48 @@ func TestRedisFailoverServiceUpdateRedisFailoverStatus(t *testing.T) { assert.Equal(t, trickyRF.Status.Message, got.Status.Message) }) } + +func TestRedisFailoverServicePatchRedisFailoverFinalizers(t *testing.T) { + testns := "testns" + + rf := &redisfailoverv1.RedisFailover{ + ObjectMeta: metav1.ObjectMeta{ + Name: "rf1", + Namespace: testns, + Finalizers: []string{"some.other/finalizer"}, + }, + } + + t.Run("patches the finalizers of an existing RedisFailover", func(t *testing.T) { + crdcli := redisfailoverfake.NewSimpleClientset(rf) + service := k8s.NewRedisFailoverService(crdcli, log.Dummy, metrics.Dummy) + + err := service.PatchRedisFailoverFinalizers(context.TODO(), testns, rf.Name, + []string{"some.other/finalizer", "redisfailovers.databases.spotahome.com/finalizer"}, metav1.PatchOptions{}) + assert.NoError(t, err) + + got, err := crdcli.DatabasesV1().RedisFailovers(testns).Get(context.TODO(), rf.Name, metav1.GetOptions{}) + assert.NoError(t, err) + assert.Equal(t, []string{"some.other/finalizer", "redisfailovers.databases.spotahome.com/finalizer"}, got.Finalizers) + }) + + t.Run("a nil finalizers list patches to an empty array rather than leaving finalizers untouched", func(t *testing.T) { + crdcli := redisfailoverfake.NewSimpleClientset(rf) + service := k8s.NewRedisFailoverService(crdcli, log.Dummy, metrics.Dummy) + + err := service.PatchRedisFailoverFinalizers(context.TODO(), testns, rf.Name, nil, metav1.PatchOptions{}) + assert.NoError(t, err) + + got, err := crdcli.DatabasesV1().RedisFailovers(testns).Get(context.TODO(), rf.Name, metav1.GetOptions{}) + assert.NoError(t, err) + assert.Empty(t, got.Finalizers) + }) + + t.Run("returns an error when patching a non-existent RedisFailover", func(t *testing.T) { + crdcli := redisfailoverfake.NewSimpleClientset() + service := k8s.NewRedisFailoverService(crdcli, log.Dummy, metrics.Dummy) + + err := service.PatchRedisFailoverFinalizers(context.TODO(), testns, "does-not-exist", []string{"some/finalizer"}, metav1.PatchOptions{}) + assert.Error(t, err) + }) +} From d3509402d1c6ba2aa9c59e8f97210dfd5c1830fa Mon Sep 17 00:00:00 2001 From: Michael Strassberger Date: Mon, 21 Sep 2026 00:58:01 +0200 Subject: [PATCH 03/24] Cut CI wall-clock time and fix the broken e2e pod-readiness gate (#177) * Run the two integration tests in parallel to cut CI wall-clock time TestRedisFailover (creation_test.go, sentinel-managed) and TestRedisFailoverOperatorManagedModeRollout (operator-managed) ran back to back in the same go test process - 206.66s + 205.84s = ~412s combined, per a real CI run's -v output (Saremox/redis-operator#174, job 35540389923/106157089205). They don't share any state that would make that unsafe: - separate namespaces (rf-integration-tests vs rf-integration-tests-operator-managed) - separate in-process operator instances, each with its own leader-election lease scoped to its own namespace (confirmed in the same CI run's logs - no lease contention between them) - separate Secrets, separate k8s clientset instances (client-go clientsets are safe for concurrent use) - both use metrics.Dummy/log.Dummy, so no shared Prometheus registry Adding t.Parallel() as the first statement in both lets Go's test runner execute them concurrently instead of sequentially, which should cut roughly half of that ~412s off the Integration test job's wall time (the long pole of the whole CI run - every other job finishes in under 3.5 minutes). The same redis:7.2.12-alpine image backs both tests' pods, so running them concurrently doesn't double the image-pull cost either: kubelet dedupes concurrent pulls of the same image on a node. This can't be verified against a real cluster in this environment (no k8s available), so validation here is build/vet only: `go build/vet ./...` and `go build/vet -tags integration ./test/...` all pass. The real before/after timing comparison happens in this PR's own CI run. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01G4mmyo3sqLwf23m5FE2nBE * ci: fix e2e pod-readiness gate and shave wall time off the CI run - e2e.yml: kubectl rollout status only supports RollingUpdate, but the Redis StatefulSet uses OnDelete, so the "pods ready" gate errored out on every run and silently burned all 60 retries under bash's &&-list errexit exemption. Switch to kubectl wait on .status.readyReplicas, which the StatefulSet controller populates regardless of strategy. - ci.yaml: drop integration-test's needs: [check, unit-test] gate - the matrix doesn't depend on either job's output, so serializing them was pure wall time. - ci.yaml: pre-pull the redis image in the background right after checkout, only waiting on it just before the Go tests start, so the pull is hidden behind conntrack/minikube/CRD setup instead of paying for it inline while the tests poll for pods to become ready. - operator_managed_rollout_test.go: reduce ommRedisSize to 2, matching creation_test.go's already-lighter footprint and cutting one pod's worth of startup/rollout time from the heaviest subtest. - creation_test.go: fix a leaked operator goroutine in TestRedisFailover - Run() was called with context.Background() and the stopC channel used for "cleanup" was never wired to it, so closing stopC did nothing and the controller (plus its informers/leader-election) kept running for the rest of the test binary's life. Use a cancelable context instead. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01G4mmyo3sqLwf23m5FE2nBE --------- Co-authored-by: Claude --- .github/workflows/ci.yaml | 22 ++++++++++++++++- .github/workflows/e2e.yml | 13 ++++++++-- .../redisfailover/creation_test.go | 24 ++++++++++++++----- .../operator_managed_rollout_test.go | 9 ++++++- 4 files changed, 58 insertions(+), 10 deletions(-) diff --git a/.github/workflows/ci.yaml b/.github/workflows/ci.yaml index bb424d465..b4b22cf22 100644 --- a/.github/workflows/ci.yaml +++ b/.github/workflows/ci.yaml @@ -69,7 +69,10 @@ jobs: integration-test: name: Integration test runs-on: ubuntu-24.04 - needs: [check, unit-test] + # No `needs` gate: this matrix doesn't depend on check/unit-test's + # output, so gating it behind them only serializes two independent + # wall-clock costs. Letting it start immediately in parallel is strictly + # faster; a failure here fails the run exactly the same either way. strategy: matrix: kubernetes: [1.35.8, 1.36.4, 1.37.0] @@ -78,6 +81,16 @@ jobs: - uses: actions/setup-go@v7 with: go-version-file: go.mod + - name: Pre-pull redis image in the background + # driver=none means minikube schedules pods straight onto this + # runner's own docker daemon, so this image is already the exact one + # the in-process operator will ask the cluster to run. Kicking the + # pull off now and only waiting on it right before the Go tests start + # hides its wall time behind conntrack/minikube/CRD setup below, + # instead of paying for it inline while waitForPodsReady polls. + run: | + docker pull redis:7.2.12-alpine & + echo "REDIS_PULL_PID=$!" >> "$GITHUB_ENV" - name: Install conntrack run: sudo apt-get install -y conntrack - name: Prepare CNI config directory @@ -97,6 +110,13 @@ jobs: container-runtime: docker - name: Add redisfailover CRD run: kubectl create -f manifests/databases.spotahome.com_redisfailovers.yaml + - name: Wait for redis image pre-pull to finish + # $REDIS_PULL_PID isn't a child of this step's shell (each step is + # its own process), so `wait` can't be used on it - poll instead. + run: | + while kill -0 "$REDIS_PULL_PID" 2>/dev/null; do + sleep 1 + done - run: make ci-integration-test chart-test: diff --git a/.github/workflows/e2e.yml b/.github/workflows/e2e.yml index 62d9643f2..59c9bf181 100644 --- a/.github/workflows/e2e.yml +++ b/.github/workflows/e2e.yml @@ -67,15 +67,24 @@ jobs: - name: Wait for Redis pods ready run: | - echo "Waiting for Redis StatefulSet..." + echo "Waiting for Redis StatefulSet to be created..." for i in {1..60}; do if kubectl get statefulset rfr-test-no-sentinel 2>/dev/null; then - kubectl rollout status statefulset/rfr-test-no-sentinel --timeout=180s && break + break fi echo "Waiting for StatefulSet to be created... ($i/60)" sleep 5 done + # The StatefulSet uses OnDelete (the operator drives pod replacement + # itself), which `kubectl rollout status` refuses to wait on - it + # only supports RollingUpdate and errors out immediately, so the + # loop above used to burn all 60 retries every run without ever + # actually waiting on pod readiness. readyReplicas is populated by + # the StatefulSet controller regardless of update strategy. + echo "Waiting for Redis pods to become Ready..." + kubectl wait --for=jsonpath='{.status.readyReplicas}'=2 statefulset/rfr-test-no-sentinel --timeout=180s + - name: Verify NO Sentinel resources exist run: | echo "Verifying Sentinel resources do NOT exist..." diff --git a/test/integration/redisfailover/creation_test.go b/test/integration/redisfailover/creation_test.go index fb5917daf..fcd81adf4 100644 --- a/test/integration/redisfailover/creation_test.go +++ b/test/integration/redisfailover/creation_test.go @@ -129,16 +129,21 @@ func (c *clients) prepareNS() error { return err } -func (c *clients) cleanup(stopC chan struct{}) { +func (c *clients) cleanup(cancel context.CancelFunc) { c.k8sClient.CoreV1().Namespaces().Delete(context.Background(), namespace, metav1.DeleteOptions{}) - close(stopC) + cancel() } func TestRedisFailover(t *testing.T) { + // Runs alongside TestRedisFailoverOperatorManagedModeRollout: separate + // namespaces, separate in-process operator instances (each with its own + // leader-election lease scoped to its own namespace), separate Secrets - + // nothing here is shared state, so there's no reason to pay for the two + // tests' pod-startup waits back to back instead of concurrently. + t.Parallel() + require := require.New(t) - // Create signal channels. - stopC := make(chan struct{}) errC := make(chan error) kubeconfig := os.Getenv("KUBECONFIG") @@ -178,12 +183,19 @@ func TestRedisFailover(t *testing.T) { redisfailoverOperator, err := redisfailover.New(redisfailover.Config{}, k8sservice, k8sClient, namespace, redisClient, metrics.Dummy, log.Dummy) require.NoError(err) + // Its own cancelable context, not context.Background(): without this, + // nothing ever stopped the operator goroutine below - closing the old + // stopC channel here was a no-op since Run() was never wired to observe + // it, so the controller (and its informers/leader-election) kept running + // for the rest of the test binary's life after this test finished. + runCtx, cancelRun := context.WithCancel(context.Background()) + go func() { - errC <- redisfailoverOperator.Run(context.Background()) + errC <- redisfailoverOperator.Run(runCtx) }() // Prepare cleanup for when the test ends - defer clients.cleanup(stopC) + defer clients.cleanup(cancelRun) // There's no external readiness signal for "the operator started"; this // just fails fast if it crashed immediately instead of silently waiting diff --git a/test/integration/redisfailover/operator_managed_rollout_test.go b/test/integration/redisfailover/operator_managed_rollout_test.go index 2f0dd9941..736b56158 100644 --- a/test/integration/redisfailover/operator_managed_rollout_test.go +++ b/test/integration/redisfailover/operator_managed_rollout_test.go @@ -40,7 +40,7 @@ import ( const ( ommNamespace = "rf-integration-tests-operator-managed" ommName = "testing-omm" - ommRedisSize = int32(3) + ommRedisSize = int32(2) ommAuthSecretPath = "redis-auth-omm" ommTestPass = "test-pass-omm" ) @@ -178,6 +178,13 @@ func (c *ommClients) onlyMaster(labelSelector string) (string, error) { // at all, and the only one that exercises a rollout (a StatefulSet template // change) rather than just initial creation. func TestRedisFailoverOperatorManagedModeRollout(t *testing.T) { + // Runs alongside TestRedisFailover (creation_test.go): separate + // namespaces, separate in-process operator instances (each with its own + // leader-election lease scoped to its own namespace), separate Secrets - + // nothing here is shared state, so there's no reason to pay for the two + // tests' pod-startup waits back to back instead of concurrently. + t.Parallel() + require := require.New(t) stopC := make(chan struct{}) From c64fdea990fcbd879742f89d399fdc17bb52de65 Mon Sep 17 00:00:00 2001 From: Michael Strassberger Date: Mon, 21 Sep 2026 01:32:11 +0200 Subject: [PATCH 04/24] test: give integration-test operators an explicit short SyncInterval (#178) The two integration tests construct their in-process operator with a zero-value redisfailover.Config{}, so Config.SyncInterval is 0. That flows into kooper's ResyncInterval, which falls back to *kooper's own* default of 3 minutes when unset - unlike production, which always sets one explicitly via -sync-interval (default 30, cmd/utils/flags.go). That 3-minute fallback is reachable in practice: CheckAndHeal's deferred status patch is sometimes byte-identical to what's already stored (a steady-state reconcile with no health-state transition), and the apiserver/etcd treat a truly no-op write as a no-op - no new resourceVersion, no watch event. Since the operator only watches the RedisFailover CR itself (not Pods/StatefulSets), that patch was the only thing re-triggering the next Handle() call. TestRedisFailoverOperatorManagedModeRollout needs two separate Handle() calls to replace both pods (slave, then master), so a missed self-trigger between them stalls for the full resync interval - matching the ~300-313s outliers seen in CI (~180s stall + ~130s real work), always in that one subtest and nowhere else, since it's the only test that depends on a second, chained reconcile. Giving both operators a short explicit SyncInterval keeps that same stall - if it happens at all - under a couple of seconds instead of 3 minutes, which should make the rollout subtest's timing consistent across runs. This masks the symptom in tests, matching what production already does via its own default; it doesn't fix the underlying gap (the controller's self-trigger depends on incidental status content changes rather than being guaranteed). That's tracked separately. Claude-Session: https://claude.ai/code/session_01G4mmyo3sqLwf23m5FE2nBE Co-authored-by: Claude --- test/integration/redisfailover/creation_test.go | 14 ++++++++++++-- .../redisfailover/operator_managed_rollout_test.go | 13 ++++++++++++- 2 files changed, 24 insertions(+), 3 deletions(-) diff --git a/test/integration/redisfailover/creation_test.go b/test/integration/redisfailover/creation_test.go index fcd81adf4..1eb2b5888 100644 --- a/test/integration/redisfailover/creation_test.go +++ b/test/integration/redisfailover/creation_test.go @@ -179,8 +179,18 @@ func TestRedisFailover(t *testing.T) { // Wait for the namespace to be ready, rather than guessing how long that takes. require.NoError(waitForNamespaceActive(k8sClient, namespace, 15*time.Second)) - // Create operator and run. - redisfailoverOperator, err := redisfailover.New(redisfailover.Config{}, k8sservice, k8sClient, namespace, redisClient, metrics.Dummy, log.Dummy) + // Create operator and run. SyncInterval is set explicitly (production + // always sets one via -sync-interval, default 30) rather than left at + // the zero value: a zero Config.SyncInterval means kooper's own + // ResyncInterval falls back to *its* default of 3 minutes, and the + // controller's status patch after a no-op reconcile can be byte-identical + // to what's already stored - which the apiserver treats as a no-op write + // that never reaches watchers, so nothing re-triggers Handle() until the + // next full resync. A multi-step change (like the rollout below, which + // needs two separate Handle() calls to replace two pods) can then stall + // for the full resync interval. A short one here keeps that stall short + // instead of letting it hit the 3-minute fallback. + redisfailoverOperator, err := redisfailover.New(redisfailover.Config{SyncInterval: 2}, k8sservice, k8sClient, namespace, redisClient, metrics.Dummy, log.Dummy) require.NoError(err) // Its own cancelable context, not context.Background(): without this, diff --git a/test/integration/redisfailover/operator_managed_rollout_test.go b/test/integration/redisfailover/operator_managed_rollout_test.go index 736b56158..249787498 100644 --- a/test/integration/redisfailover/operator_managed_rollout_test.go +++ b/test/integration/redisfailover/operator_managed_rollout_test.go @@ -216,7 +216,18 @@ func TestRedisFailoverOperatorManagedModeRollout(t *testing.T) { require.NoError(waitForNamespaceActive(k8sClient, ommNamespace, 15*time.Second)) k8sservice := k8s.New(k8sClient, customClient, log.Dummy, metrics.Dummy) - redisfailoverOperator, err := redisfailover.New(redisfailover.Config{}, k8sservice, k8sClient, ommNamespace, redisClient, metrics.Dummy, log.Dummy) + // SyncInterval is set explicitly (production always sets one via + // -sync-interval, default 30) rather than left at the zero value: a zero + // Config.SyncInterval means kooper's own ResyncInterval falls back to + // *its* default of 3 minutes, and the controller's status patch after a + // no-op reconcile can be byte-identical to what's already stored - which + // the apiserver treats as a no-op write that never reaches watchers, so + // nothing re-triggers Handle() until the next full resync. The rollout + // below needs two separate Handle() calls to replace both pods (the + // slave, then the master), so a missed self-trigger between them can + // stall for the full resync interval - a short one here keeps that + // stall short instead of letting it hit the 3-minute fallback. + redisfailoverOperator, err := redisfailover.New(redisfailover.Config{SyncInterval: 2}, k8sservice, k8sClient, ommNamespace, redisClient, metrics.Dummy, log.Dummy) require.NoError(err) go func() { From fee5935a8b616166aefb72bea7776a814c3bff32 Mon Sep 17 00:00:00 2001 From: "dependabot[bot]" <49699333+dependabot[bot]@users.noreply.github.com> Date: Mon, 21 Sep 2026 23:08:48 +0200 Subject: [PATCH 05/24] chore(deps): bump actions/checkout from 6 to 7 (#183) Bumps [actions/checkout](https://github.com/actions/checkout) from 6 to 7. - [Release notes](https://github.com/actions/checkout/releases) - [Changelog](https://github.com/actions/checkout/blob/main/CHANGELOG.md) - [Commits](https://github.com/actions/checkout/compare/v6...v7) --- updated-dependencies: - dependency-name: actions/checkout dependency-version: '7' dependency-type: direct:production update-type: version-update:semver-major ... Signed-off-by: dependabot[bot] Co-authored-by: dependabot[bot] <49699333+dependabot[bot]@users.noreply.github.com> --- .github/workflows/e2e.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.github/workflows/e2e.yml b/.github/workflows/e2e.yml index 59c9bf181..32e30ea4f 100644 --- a/.github/workflows/e2e.yml +++ b/.github/workflows/e2e.yml @@ -15,7 +15,7 @@ jobs: e2e-sentinel-free: runs-on: ubuntu-latest steps: - - uses: actions/checkout@v6 + - uses: actions/checkout@v7 - name: Set up minikube uses: medyagh/setup-minikube@v0.0.21 From 674cf3aec1964db205a11dc124d2de4c9dfa56d5 Mon Sep 17 00:00:00 2001 From: Michael Strassberger Date: Thu, 24 Sep 2026 23:46:32 +0200 Subject: [PATCH 06/24] Potential fix for code scanning alert no. 28: Workflow does not contain permissions (#184) Co-authored-by: Copilot Autofix powered by AI <62310815+github-advanced-security[bot]@users.noreply.github.com> --- .github/workflows/e2e.yml | 3 +++ 1 file changed, 3 insertions(+) diff --git a/.github/workflows/e2e.yml b/.github/workflows/e2e.yml index 32e30ea4f..6bfb6daad 100644 --- a/.github/workflows/e2e.yml +++ b/.github/workflows/e2e.yml @@ -11,6 +11,9 @@ on: pull_request: workflow_dispatch: +permissions: + contents: read + jobs: e2e-sentinel-free: runs-on: ubuntu-latest From b3b24e44721023a0ae01a63160e2ecef6d7bc19d Mon Sep 17 00:00:00 2001 From: Michael Strassberger Date: Fri, 25 Sep 2026 15:39:04 +0200 Subject: [PATCH 07/24] Disconnect a demoted master's clients once it leaves the master Service (#182) A Service only routes new connections, so clients connected through the master Service stay on a pod after it is demoted to a replica and keep getting READONLY errors. When a pod's role label changes from master to slave, the operator now waits (in the background, up to 10s) for the pod's IP to leave the master Service's EndpointSlices, then 2s for kube-proxy, and closes the pod's normal and pub/sub connections (CLIENT KILL TYPE normal|pubsub) so clients reconnect to the new master. Replication links are left alone. This covers Sentinel-driven demotions too, which the operator only relabels. - On by default; --disconnect-clients-on-demotion=false turns it off. - Needs get/list on discovery.k8s.io endpointslices (chart, kustomize and examples updated). Without it, or on timeout, it disconnects anyway. - e2e: a client writing through the master Service must end up on the new master after master and replica are swapped, in both directions. Claude-Session: https://claude.ai/code/session_01G4mmyo3sqLwf23m5FE2nBE Co-authored-by: Claude --- .github/workflows/e2e.yml | 39 +++ .../templates/service-account.yaml | 7 + cmd/utils/flags.go | 23 +- cmd/utils/flags_test.go | 42 +++ .../all-redis-operator-resources.yaml | 7 + example/operator/roles.yaml | 7 + .../components/rbac/clusterrole.yaml | 7 + metrics/metrics.go | 1 + mocks/service/redis/Client.go | 14 + operator/redisfailover/config.go | 2 + operator/redisfailover/factory.go | 12 +- operator/redisfailover/service/check.go | 15 +- operator/redisfailover/service/demotion.go | 133 ++++++++ .../redisfailover/service/demotion_test.go | 304 ++++++++++++++++++ operator/redisfailover/service/heal.go | 19 +- service/redis/client.go | 35 ++ service/redis/client_test.go | 123 +++++++ 17 files changed, 758 insertions(+), 32 deletions(-) create mode 100644 cmd/utils/flags_test.go create mode 100644 operator/redisfailover/service/demotion.go create mode 100644 operator/redisfailover/service/demotion_test.go diff --git a/.github/workflows/e2e.yml b/.github/workflows/e2e.yml index 6bfb6daad..93a617345 100644 --- a/.github/workflows/e2e.yml +++ b/.github/workflows/e2e.yml @@ -266,6 +266,45 @@ jobs: echo "⚠ Master service endpoint mismatch (may need more time to update)" fi + - name: Test clients of a demoted master move to the new master + run: | + # redis-cli -r keeps one connection open and reconnects as soon as it is closed. + kubectl run demotion-client --image=redis:7.2.12-alpine --image-pull-policy=IfNotPresent --restart=Never --command -- \ + sh -c 'while true; do redis-cli -h rfrm-test-no-sentinel -r -1 -i 0.01 INCR e2e:demotion; sleep 0.01; done' + kubectl wait --for=condition=Ready pod/demotion-client --timeout=60s + + writing() { local out; out=$(kubectl logs demotion-client --tail=5); [[ -n "$out" ]] && ! grep -qvE '^[0-9]+$' <<<"$out"; } + readonly_errors() { kubectl logs demotion-client | grep -c READONLY || true; } + for i in {1..30}; do writing && break; sleep 2; done + writing || { echo "✗ Client never wrote to the master"; kubectl logs demotion-client --tail=20; exit 1; } + + # Both directions, since the operator relabels pods in list order. + for round in 1 2; do + for pod in rfr-test-no-sentinel-0 rfr-test-no-sentinel-1; do + if kubectl exec $pod -- redis-cli INFO replication | grep -q "role:master"; then OLD_MASTER=$pod; else NEW_MASTER=$pod; fi + done + echo "Round $round: demoting $OLD_MASTER in favour of $NEW_MASTER" + ERRORS_BEFORE=$(readonly_errors) + kubectl exec $NEW_MASTER -- redis-cli REPLICAOF NO ONE + kubectl exec $OLD_MASTER -- redis-cli REPLICAOF $(kubectl get pod $NEW_MASTER -o jsonpath='{.status.podIP}') 6379 + + for i in {1..45}; do + [[ "$(kubectl get pod $OLD_MASTER -o jsonpath='{.metadata.labels.redisfailovers-role}')" == slave ]] && break + sleep 2 + done + MOVED=false + for i in {1..10}; do + sleep 3 + if writing && (( $(readonly_errors) > ERRORS_BEFORE )); then MOVED=true; break; fi + done + if [[ "$MOVED" != true ]]; then + echo "✗ Client still stuck on the demoted master" + kubectl logs demotion-client --tail=20 + exit 1 + fi + echo "✓ Client reconnected to the new master" + done + - name: Collect logs on failure if: failure() run: | diff --git a/charts/redisoperator/templates/service-account.yaml b/charts/redisoperator/templates/service-account.yaml index 1e8b79e86..41340aaaf 100644 --- a/charts/redisoperator/templates/service-account.yaml +++ b/charts/redisoperator/templates/service-account.yaml @@ -42,6 +42,13 @@ rules: - get - list - update + - apiGroups: + - discovery.k8s.io + resources: + - endpointslices + verbs: + - get + - list - apiGroups: - "" resources: diff --git a/cmd/utils/flags.go b/cmd/utils/flags.go index be83c4fb9..029d3c1f8 100644 --- a/cmd/utils/flags.go +++ b/cmd/utils/flags.go @@ -13,16 +13,17 @@ import ( // CMDFlags are the flags used by the cmd // TODO: improve flags. type CMDFlags struct { - KubeConfig string - SupportedNamespacesRegex string - Development bool - ListenAddr string - MetricsPath string - K8sQueriesPerSecond int - K8sQueriesBurstable int - Concurrency int - SyncInterval int - LogLevel string + KubeConfig string + SupportedNamespacesRegex string + Development bool + ListenAddr string + MetricsPath string + K8sQueriesPerSecond int + K8sQueriesBurstable int + Concurrency int + SyncInterval int + LogLevel string + DisconnectClientsOnDemotion bool } // Init initializes and parse the flags @@ -41,6 +42,7 @@ func (c *CMDFlags) Init() { flag.IntVar(&c.Concurrency, "concurrency", 3, "Number of conccurent workers meant to process events") flag.IntVar(&c.SyncInterval, "sync-interval", 30, "Number of seconds between checks") flag.StringVar(&c.LogLevel, "log-level", "info", "set log level") + flag.BoolVar(&c.DisconnectClientsOnDemotion, "disconnect-clients-on-demotion", true, "Close a redis pod's normal and pub/sub client connections when it stops being the master, so clients reconnect to the new master instead of staying on a replica") // Parse flags flag.Parse() @@ -57,5 +59,6 @@ func (c *CMDFlags) ToRedisOperatorConfig() redisfailover.Config { Concurrency: c.Concurrency, SyncInterval: c.SyncInterval, SupportedNamespacesRegex: c.SupportedNamespacesRegex, + KeepClientsOnDemotion: !c.DisconnectClientsOnDemotion, } } diff --git a/cmd/utils/flags_test.go b/cmd/utils/flags_test.go new file mode 100644 index 000000000..a2e5d52e2 --- /dev/null +++ b/cmd/utils/flags_test.go @@ -0,0 +1,42 @@ +package utils + +import ( + "flag" + "os" + "testing" + + "github.com/stretchr/testify/assert" +) + +func TestToRedisOperatorConfigDisconnectClientsOnDemotion(t *testing.T) { + enabled := (&CMDFlags{DisconnectClientsOnDemotion: true}).ToRedisOperatorConfig() + assert.False(t, enabled.KeepClientsOnDemotion) + + disabled := (&CMDFlags{DisconnectClientsOnDemotion: false}).ToRedisOperatorConfig() + assert.True(t, disabled.KeepClientsOnDemotion) +} + +func TestInitDisconnectClientsOnDemotion(t *testing.T) { + tests := map[string]struct { + args []string + want bool + }{ + "on by default": {want: true}, + "turned off": {args: []string{"--disconnect-clients-on-demotion=false"}, want: false}, + } + + for name, test := range tests { + t.Run(name, func(t *testing.T) { + defer func(args []string, commandLine *flag.FlagSet) { + os.Args, flag.CommandLine = args, commandLine + }(os.Args, flag.CommandLine) + os.Args = append([]string{"redis-operator"}, test.args...) + flag.CommandLine = flag.NewFlagSet("redis-operator", flag.ContinueOnError) + + var flags CMDFlags + flags.Init() + + assert.Equal(t, test.want, flags.DisconnectClientsOnDemotion) + }) + } +} diff --git a/example/operator/all-redis-operator-resources.yaml b/example/operator/all-redis-operator-resources.yaml index dad4bd536..92d2b193d 100644 --- a/example/operator/all-redis-operator-resources.yaml +++ b/example/operator/all-redis-operator-resources.yaml @@ -104,6 +104,13 @@ rules: - leases verbs: - "*" + - apiGroups: + - discovery.k8s.io + resources: + - endpointslices + verbs: + - get + - list --- apiVersion: v1 diff --git a/example/operator/roles.yaml b/example/operator/roles.yaml index da969654e..799cbc9aa 100644 --- a/example/operator/roles.yaml +++ b/example/operator/roles.yaml @@ -25,6 +25,13 @@ rules: - get - list - update + - apiGroups: + - discovery.k8s.io + resources: + - endpointslices + verbs: + - get + - list - apiGroups: - "" resources: diff --git a/manifests/kustomize/components/rbac/clusterrole.yaml b/manifests/kustomize/components/rbac/clusterrole.yaml index 4e4025d9b..6dfc5f2a3 100644 --- a/manifests/kustomize/components/rbac/clusterrole.yaml +++ b/manifests/kustomize/components/rbac/clusterrole.yaml @@ -25,6 +25,13 @@ rules: - get - list - update + - apiGroups: + - discovery.k8s.io + resources: + - endpointslices + verbs: + - get + - list - apiGroups: - "" resources: diff --git a/metrics/metrics.go b/metrics/metrics.go index 190663ac1..c0d2466ae 100644 --- a/metrics/metrics.go +++ b/metrics/metrics.go @@ -71,6 +71,7 @@ const ( CHECK_SENTINEL_QUORUM = "SENTINEL_CKQUORUM" SLAVE_IS_READY = "CHECK_IF_SLAVE_IS_READY" GET_REPLICATION_INFO = "GET_REPLICATION_INFO" + DISCONNECT_CLIENTS = "DISCONNECT_CLIENTS_ON_DEMOTED_INSTANCE" ) var ( // used for grabage collection of metrics diff --git a/mocks/service/redis/Client.go b/mocks/service/redis/Client.go index 332361a03..8498b61f6 100644 --- a/mocks/service/redis/Client.go +++ b/mocks/service/redis/Client.go @@ -13,6 +13,20 @@ type Client struct { mock.Mock } +// DisconnectClients provides a mock function with given fields: ip, port, password +func (_m *Client) DisconnectClients(ip string, port string, password string) error { + ret := _m.Called(ip, port, password) + + var r0 error + if rf, ok := ret.Get(0).(func(string, string, string) error); ok { + r0 = rf(ip, port, password) + } else { + r0 = ret.Error(0) + } + + return r0 +} + // GetNumberSentinelSlavesInMemory provides a mock function with given fields: ip func (_m *Client) GetNumberSentinelSlavesInMemory(ip string) (int32, error) { ret := _m.Called(ip) diff --git a/operator/redisfailover/config.go b/operator/redisfailover/config.go index d384f81ac..1f5c32e64 100644 --- a/operator/redisfailover/config.go +++ b/operator/redisfailover/config.go @@ -7,4 +7,6 @@ type Config struct { Concurrency int SyncInterval int SupportedNamespacesRegex string + // KeepClientsOnDemotion is negative so the zero Config disconnects. + KeepClientsOnDemotion bool } diff --git a/operator/redisfailover/factory.go b/operator/redisfailover/factory.go index 11c213a92..8057bfb35 100644 --- a/operator/redisfailover/factory.go +++ b/operator/redisfailover/factory.go @@ -25,6 +25,9 @@ import ( const ( operatorName = "redis-operator" lockKey = "redis-failover-lease" + + endpointRemovalTimeout = 10 * time.Second + kubeProxySyncGrace = 2 * time.Second ) // New will create an operator that is responsible for managing all the required stuff @@ -32,8 +35,13 @@ const ( func New(cfg Config, k8sService k8s.Services, k8sClient kubernetes.Interface, lockNamespace string, redisClient redis.Client, kooperMetricsRecorder metrics.Recorder, logger log.Logger) (controller.Controller, error) { // Create internal services. rfService := rfservice.NewRedisFailoverKubeClient(k8sService, logger, kooperMetricsRecorder) - rfChecker := rfservice.NewRedisFailoverChecker(k8sService, redisClient, logger, kooperMetricsRecorder) - rfHealer := rfservice.NewRedisFailoverHealer(k8sService, redisClient, logger) + var opts []rfservice.Option + if !cfg.KeepClientsOnDemotion { + disconnector := rfservice.NewClientDisconnector(k8sClient, redisClient, logger, endpointRemovalTimeout, kubeProxySyncGrace) + opts = append(opts, rfservice.WithClientDisconnector(disconnector)) + } + rfChecker := rfservice.NewRedisFailoverChecker(k8sService, redisClient, logger, kooperMetricsRecorder, opts...) + rfHealer := rfservice.NewRedisFailoverHealer(k8sService, redisClient, logger, opts...) // Create the handlers. rfHandler := NewRedisFailoverHandler(cfg, rfService, rfChecker, rfHealer, k8sService, kooperMetricsRecorder, logger) diff --git a/operator/redisfailover/service/check.go b/operator/redisfailover/service/check.go index 5923b5f3a..99e395751 100644 --- a/operator/redisfailover/service/check.go +++ b/operator/redisfailover/service/check.go @@ -71,15 +71,17 @@ type RedisFailoverChecker struct { redisClient redis.Client logger log.Logger metricsClient metrics.Recorder + opts options } // NewRedisFailoverChecker creates an object of the RedisFailoverChecker struct -func NewRedisFailoverChecker(k8sService k8s.Services, redisClient redis.Client, logger log.Logger, metricsClient metrics.Recorder) *RedisFailoverChecker { +func NewRedisFailoverChecker(k8sService k8s.Services, redisClient redis.Client, logger log.Logger, metricsClient metrics.Recorder, opts ...Option) *RedisFailoverChecker { return &RedisFailoverChecker{ k8sService: k8sService, redisClient: redisClient, logger: logger, metricsClient: metricsClient, + opts: applyOptions(opts), } } @@ -116,13 +118,8 @@ func (r *RedisFailoverChecker) setMasterLabelIfNecessary(namespace string, pod c return r.k8sService.UpdatePodLabels(namespace, pod.Name, generateRedisMasterRoleLabel()) } -func (r *RedisFailoverChecker) setSlaveLabelIfNecessary(namespace string, pod corev1.Pod) error { - for labelKey, labelValue := range pod.Labels { - if labelKey == redisRoleLabelKey && labelValue == redisRoleLabelSlave { - return nil - } - } - return r.k8sService.UpdatePodLabels(namespace, pod.Name, generateRedisSlaveRoleLabel()) +func (r *RedisFailoverChecker) setSlaveLabelIfNecessary(rf *redisfailoverv1.RedisFailover, pod corev1.Pod, port, password string) error { + return setSlaveLabel(r.k8sService, r.opts, rf, pod, port, password) } // CheckAllSlavesFromMaster controlls that all slaves have the same master (the real one) @@ -151,7 +148,7 @@ func (r *RedisFailoverChecker) CheckAllSlavesFromMaster(master string, rf *redis return err } } else { - err = r.setSlaveLabelIfNecessary(rf.Namespace, rp) + err = r.setSlaveLabelIfNecessary(rf, rp, rport, password) if err != nil { return err } diff --git a/operator/redisfailover/service/demotion.go b/operator/redisfailover/service/demotion.go new file mode 100644 index 000000000..37f2fdf9e --- /dev/null +++ b/operator/redisfailover/service/demotion.go @@ -0,0 +1,133 @@ +package service + +import ( + "context" + "slices" + "sync" + "time" + + corev1 "k8s.io/api/core/v1" + discoveryv1 "k8s.io/api/discovery/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/util/wait" + "k8s.io/client-go/kubernetes" + + redisfailoverv1 "github.com/saremox/redis-operator/api/redisfailover/v1" + "github.com/saremox/redis-operator/log" + "github.com/saremox/redis-operator/service/k8s" + "github.com/saremox/redis-operator/service/redis" +) + +const endpointPollInterval = 200 * time.Millisecond + +// Option configures optional behaviour shared by RedisFailoverChecker and +// RedisFailoverHealer. +type Option func(*options) + +type options struct { + disconnector ClientDisconnector +} + +func applyOptions(opts []Option) options { + var o options + for _, opt := range opts { + opt(&o) + } + return o +} + +// WithClientDisconnector sets what closes a pod's client connections when +// its role label moves from master to slave. Without it they are left open. +func WithClientDisconnector(d ClientDisconnector) Option { + return func(o *options) { + o.disconnector = d + } +} + +// ClientDisconnector closes the client connections of a pod that has stopped +// being the master, so clients reconnect through the master Service. +type ClientDisconnector interface { + DisconnectDemoted(rf *redisfailoverv1.RedisFailover, pod corev1.Pod, port, password string) +} + +// setSlaveLabel gives pod the slave role label if it doesn't already have it, +// and disconnects its clients if the label it replaces was master. +func setSlaveLabel(k8sService k8s.Services, o options, rf *redisfailoverv1.RedisFailover, pod corev1.Pod, port, password string) error { + previousRole := pod.Labels[redisRoleLabelKey] + if previousRole == redisRoleLabelSlave { + return nil + } + if err := k8sService.UpdatePodLabels(rf.Namespace, pod.Name, generateRedisSlaveRoleLabel()); err != nil { + return err + } + if previousRole == redisRoleLabelMaster && o.disconnector != nil { + o.disconnector.DisconnectDemoted(rf, pod, port, password) + } + return nil +} + +type endpointAwareDisconnector struct { + kubeClient kubernetes.Interface + redisClient redis.Client + logger log.Logger + timeout time.Duration + grace time.Duration + pending sync.Map +} + +// NewClientDisconnector returns a ClientDisconnector that works in the +// background: it waits (up to timeout) for the pod to leave the master +// Service's EndpointSlices, then for grace so kube-proxy can catch up, and +// only then disconnects. Disconnecting earlier sends clients straight back. +func NewClientDisconnector(kubeClient kubernetes.Interface, redisClient redis.Client, logger log.Logger, timeout, grace time.Duration) ClientDisconnector { + return &endpointAwareDisconnector{ + kubeClient: kubeClient, + redisClient: redisClient, + logger: logger, + timeout: timeout, + grace: grace, + } +} + +func (d *endpointAwareDisconnector) DisconnectDemoted(rf *redisfailoverv1.RedisFailover, pod corev1.Pod, port, password string) { + key := rf.Namespace + "/" + pod.Name + if _, busy := d.pending.LoadOrStore(key, struct{}{}); busy { + return + } + go func() { + defer d.pending.Delete(key) + d.disconnect(rf, pod, port, password) + }() +} + +func (d *endpointAwareDisconnector) disconnect(rf *redisfailoverv1.RedisFailover, pod corev1.Pod, port, password string) { + logger := d.logger.WithField("redisfailover", rf.Name).WithField("namespace", rf.Namespace) + service := GetRedisMasterName(rf) + if err := d.waitForEndpointRemoval(rf.Namespace, service, pod.Status.PodIP); err != nil { + logger.Warningf("Pod %s may still be behind Service %s: %v", pod.Name, service, err) + } + time.Sleep(d.grace) + + logger.Infof("Pod %s is no longer the master, disconnecting its clients", pod.Name) + if err := d.redisClient.DisconnectClients(pod.Status.PodIP, port, password); err != nil { + logger.Warningf("Could not disconnect clients of demoted pod %s: %v", pod.Name, err) + } +} + +func (d *endpointAwareDisconnector) waitForEndpointRemoval(namespace, service, ip string) error { + selector := metav1.ListOptions{LabelSelector: discoveryv1.LabelServiceName + "=" + service} + return wait.PollUntilContextTimeout(context.Background(), endpointPollInterval, d.timeout, true, func(ctx context.Context) (bool, error) { + endpointSlices, err := d.kubeClient.DiscoveryV1().EndpointSlices(namespace).List(ctx, selector) + if err != nil { + return false, err + } + for _, endpointSlice := range endpointSlices.Items { + for _, endpoint := range endpointSlice.Endpoints { + if slices.Contains(endpoint.Addresses, ip) { + return false, nil + } + } + } + return true, nil + }) +} diff --git a/operator/redisfailover/service/demotion_test.go b/operator/redisfailover/service/demotion_test.go new file mode 100644 index 000000000..e448db270 --- /dev/null +++ b/operator/redisfailover/service/demotion_test.go @@ -0,0 +1,304 @@ +package service_test + +import ( + "context" + "errors" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + "github.com/stretchr/testify/require" + corev1 "k8s.io/api/core/v1" + discoveryv1 "k8s.io/api/discovery/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/runtime" + "k8s.io/client-go/kubernetes/fake" + k8stesting "k8s.io/client-go/testing" + + redisfailoverv1 "github.com/saremox/redis-operator/api/redisfailover/v1" + "github.com/saremox/redis-operator/log" + "github.com/saremox/redis-operator/metrics" + mK8SService "github.com/saremox/redis-operator/mocks/service/k8s" + mRedisService "github.com/saremox/redis-operator/mocks/service/redis" + rfservice "github.com/saremox/redis-operator/operator/redisfailover/service" +) + +var ( + masterRoleLabel = map[string]string{"redisfailovers-role": "master"} + slaveRoleLabel = map[string]string{"redisfailovers-role": "slave"} +) + +func podWithRole(name, ip string, labels map[string]string) corev1.Pod { + return corev1.Pod{ + ObjectMeta: metav1.ObjectMeta{Name: name, Labels: labels}, + Status: corev1.PodStatus{PodIP: ip, Phase: corev1.PodRunning}, + } +} + +type fakeDisconnector struct { + calls *[]string +} + +func (f fakeDisconnector) DisconnectDemoted(_ *redisfailoverv1.RedisFailover, pod corev1.Pod, _, _ string) { + *f.calls = append(*f.calls, "disconnect "+pod.Name) +} + +func TestCheckAllSlavesFromMasterDemotedMasterLabel(t *testing.T) { + tests := []struct { + name string + oldLabels map[string]string + noDisconnector bool + expectRelabel bool + expectedActions []string + }{ + { + name: "master label replaced: clients disconnected after relabel", + oldLabels: masterRoleLabel, + expectRelabel: true, + expectedActions: []string{"relabel", "disconnect old-master"}, + }, + { + name: "master label replaced without a disconnector: label only", + oldLabels: masterRoleLabel, + noDisconnector: true, + expectRelabel: true, + expectedActions: []string{"relabel"}, + }, + { + name: "unlabelled pod gets its first label: nothing to disconnect", + oldLabels: nil, + expectRelabel: true, + expectedActions: []string{"relabel"}, + }, + { + name: "already labelled slave: untouched", + oldLabels: slaveRoleLabel, + }, + } + + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + rf := generateRF() + pods := &corev1.PodList{Items: []corev1.Pod{ + podWithRole("new-master", "0.0.0.0", masterRoleLabel), + podWithRole("old-master", "1.1.1.1", test.oldLabels), + }} + + var actions []string + ms := &mK8SService.Services{} + ms.On("GetStatefulSetPods", namespace, rfservice.GetRedisName(rf)).Once().Return(pods, nil) + if test.expectRelabel { + ms.On("UpdatePodLabels", namespace, "old-master", slaveRoleLabel).Once().Return(nil). + Run(func(mock.Arguments) { actions = append(actions, "relabel") }) + } + mr := &mRedisService.Client{} + mr.On("GetSlaveOf", "0.0.0.0", "0", "").Once().Return("", nil) + mr.On("GetSlaveOf", "1.1.1.1", "0", "").Once().Return("0.0.0.0", nil) + + var opts []rfservice.Option + if !test.noDisconnector { + opts = append(opts, rfservice.WithClientDisconnector(fakeDisconnector{calls: &actions})) + } + checker := rfservice.NewRedisFailoverChecker(ms, mr, log.DummyLogger{}, metrics.Dummy, opts...) + err := checker.CheckAllSlavesFromMaster("0.0.0.0", rf) + + assert.NoError(t, err) + ms.AssertExpectations(t) + mr.AssertExpectations(t) + assert.Equal(t, test.expectedActions, actions) + }) + } +} + +// A pod still behind the master Service would get its clients straight back. +func TestCheckAllSlavesFromMasterRelabelFailureSkipsDisconnect(t *testing.T) { + assert := assert.New(t) + rf := generateRF() + pods := &corev1.PodList{Items: []corev1.Pod{ + podWithRole("old-master", "1.1.1.1", masterRoleLabel), + }} + + ms := &mK8SService.Services{} + ms.On("GetStatefulSetPods", namespace, rfservice.GetRedisName(rf)).Once().Return(pods, nil) + ms.On("UpdatePodLabels", namespace, "old-master", slaveRoleLabel).Once().Return(errors.New("conflict")) + var disconnects []string + + checker := rfservice.NewRedisFailoverChecker(ms, &mRedisService.Client{}, log.DummyLogger{}, metrics.Dummy, + rfservice.WithClientDisconnector(fakeDisconnector{calls: &disconnects})) + err := checker.CheckAllSlavesFromMaster("0.0.0.0", rf) + + assert.Error(err) + assert.Empty(disconnects) +} + +// Replicas that merely get SLAVEOF re-issued keep their (read) clients. +func TestSetMasterOnAllDisconnectsOnlyTheDemotedMaster(t *testing.T) { + assert := assert.New(t) + rf := generateRF() + pods := &corev1.PodList{Items: []corev1.Pod{ + podWithRole("new-master", "0.0.0.0", masterRoleLabel), + podWithRole("old-master", "1.1.1.1", masterRoleLabel), + podWithRole("replica", "2.2.2.2", slaveRoleLabel), + }} + + ms := &mK8SService.Services{} + ms.On("GetStatefulSetPods", namespace, rfservice.GetRedisName(rf)).Once().Return(pods, nil) + ms.On("UpdatePodLabels", namespace, "old-master", slaveRoleLabel).Once().Return(nil) + mr := &mRedisService.Client{} + mr.On("IsMaster", "0.0.0.0", "0", "").Return(true, nil) + mr.On("MakeSlaveOfWithPort", "1.1.1.1", "0.0.0.0", "0", "").Once().Return(nil) + mr.On("MakeSlaveOfWithPort", "2.2.2.2", "0.0.0.0", "0", "").Once().Return(nil) + var disconnects []string + + healer := rfservice.NewRedisFailoverHealer(ms, mr, log.DummyLogger{}, + rfservice.WithClientDisconnector(fakeDisconnector{calls: &disconnects})) + err := healer.SetMasterOnAll("0.0.0.0", rf) + + assert.NoError(err) + ms.AssertExpectations(t) + mr.AssertExpectations(t) + assert.Equal([]string{"disconnect old-master"}, disconnects) +} + +// Bootstrap mode re-issues SLAVEOF on every reconcile; that must not disconnect anyone. +func TestSetExternalMasterOnAllNeverDisconnectsClients(t *testing.T) { + assert := assert.New(t) + rf := generateRF() + pods := &corev1.PodList{Items: []corev1.Pod{ + podWithRole("pod-0", "0.0.0.0", masterRoleLabel), + podWithRole("pod-1", "1.1.1.1", slaveRoleLabel), + }} + + ms := &mK8SService.Services{} + ms.On("GetStatefulSetPods", namespace, rfservice.GetRedisName(rf)).Return(pods, nil) + mr := &mRedisService.Client{} + mr.On("MakeSlaveOfWithPort", mock.Anything, "5.5.5.5", "6379", "").Return(nil) + var disconnects []string + + healer := rfservice.NewRedisFailoverHealer(ms, mr, log.DummyLogger{}, + rfservice.WithClientDisconnector(fakeDisconnector{calls: &disconnects})) + for reconcile := 0; reconcile < 3; reconcile++ { + assert.NoError(healer.SetExternalMasterOnAll("5.5.5.5", "6379", rf)) + } + + mr.AssertNumberOfCalls(t, "MakeSlaveOfWithPort", 6) + assert.Empty(disconnects) + ms.AssertNotCalled(t, "UpdatePodLabels", mock.Anything, mock.Anything, mock.Anything) +} + +func masterEndpointSlice(rf *redisfailoverv1.RedisFailover, ips ...string) *discoveryv1.EndpointSlice { + return endpointSlice(rfservice.GetRedisMasterName(rf), ips...) +} + +func endpointSlice(service string, ips ...string) *discoveryv1.EndpointSlice { + slice := &discoveryv1.EndpointSlice{ObjectMeta: metav1.ObjectMeta{ + Name: service + "-abcde", + Namespace: namespace, + Labels: map[string]string{discoveryv1.LabelServiceName: service}, + }} + for _, ip := range ips { + slice.Endpoints = append(slice.Endpoints, discoveryv1.Endpoint{Addresses: []string{ip}}) + } + return slice +} + +func disconnectClientsMock(err error) (*mRedisService.Client, chan time.Time) { + called := make(chan time.Time, 1) + mr := &mRedisService.Client{} + mr.On("DisconnectClients", "1.1.1.1", "0", "").Once().Return(err). + Run(func(mock.Arguments) { called <- time.Now() }) + return mr, called +} + +func TestClientDisconnectorWaitsForPodToLeaveMasterService(t *testing.T) { + rf := generateRF() + kubeClient := fake.NewClientset( + masterEndpointSlice(rf, "0.0.0.0", "1.1.1.1"), + endpointSlice("unrelated", "1.1.1.1"), + ) + mr, called := disconnectClientsMock(nil) + disconnector := rfservice.NewClientDisconnector(kubeClient, mr, log.DummyLogger{}, time.Minute, 0) + + demoted := podWithRole("old-master", "1.1.1.1", slaveRoleLabel) + disconnector.DisconnectDemoted(rf, demoted, "0", "") + disconnector.DisconnectDemoted(rf, demoted, "0", "") + + select { + case <-called: + t.Fatal("clients disconnected while the pod was still behind the master Service") + case <-time.After(time.Second): + } + + _, err := kubeClient.DiscoveryV1().EndpointSlices(namespace). + Update(context.Background(), masterEndpointSlice(rf, "0.0.0.0"), metav1.UpdateOptions{}) + require.NoError(t, err) + + select { + case <-called: + case <-time.After(5 * time.Second): + t.Fatal("clients not disconnected after the pod left the master Service") + } + time.Sleep(500 * time.Millisecond) + mr.AssertNumberOfCalls(t, "DisconnectClients", 1) +} + +func TestClientDisconnectorWaitsForGraceAfterRemoval(t *testing.T) { + rf := generateRF() + kubeClient := fake.NewClientset(masterEndpointSlice(rf, "0.0.0.0")) + mr, called := disconnectClientsMock(nil) + grace := 500 * time.Millisecond + disconnector := rfservice.NewClientDisconnector(kubeClient, mr, log.DummyLogger{}, time.Minute, grace) + + start := time.Now() + disconnector.DisconnectDemoted(rf, podWithRole("old-master", "1.1.1.1", slaveRoleLabel), "0", "") + + select { + case at := <-called: + assert.GreaterOrEqual(t, at.Sub(start), grace) + case <-time.After(5 * time.Second): + t.Fatal("clients not disconnected") + } +} + +func TestClientDisconnectorFallsBackToDisconnecting(t *testing.T) { + tests := []struct { + name string + timeout time.Duration + setup func(*fake.Clientset) + }{ + { + name: "pod never leaves the master Service", + timeout: 500 * time.Millisecond, + }, + { + name: "EndpointSlices can't be listed", + timeout: time.Minute, + setup: func(c *fake.Clientset) { + c.PrependReactor("list", "endpointslices", func(k8stesting.Action) (bool, runtime.Object, error) { + return true, nil, errors.New("forbidden") + }) + }, + }, + } + + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + rf := generateRF() + kubeClient := fake.NewClientset(masterEndpointSlice(rf, "1.1.1.1")) + if test.setup != nil { + test.setup(kubeClient) + } + mr, called := disconnectClientsMock(errors.New("NOPERM")) + disconnector := rfservice.NewClientDisconnector(kubeClient, mr, log.DummyLogger{}, test.timeout, 0) + + disconnector.DisconnectDemoted(rf, podWithRole("old-master", "1.1.1.1", slaveRoleLabel), "0", "") + + select { + case <-called: + case <-time.After(5 * time.Second): + t.Fatal("clients not disconnected") + } + }) + } +} diff --git a/operator/redisfailover/service/heal.go b/operator/redisfailover/service/heal.go index c7130c075..de8214ebf 100644 --- a/operator/redisfailover/service/heal.go +++ b/operator/redisfailover/service/heal.go @@ -39,15 +39,17 @@ type RedisFailoverHealer struct { k8sService k8s.Services redisClient redis.Client logger log.Logger + opts options } // NewRedisFailoverHealer creates an object of the RedisFailoverChecker struct -func NewRedisFailoverHealer(k8sService k8s.Services, redisClient redis.Client, logger log.Logger) *RedisFailoverHealer { +func NewRedisFailoverHealer(k8sService k8s.Services, redisClient redis.Client, logger log.Logger, opts ...Option) *RedisFailoverHealer { logger = logger.With("service", "redis.healer") return &RedisFailoverHealer{ k8sService: k8sService, redisClient: redisClient, logger: logger, + opts: applyOptions(opts), } } @@ -60,13 +62,8 @@ func (r *RedisFailoverHealer) setMasterLabelIfNecessary(namespace string, pod v1 return r.k8sService.UpdatePodLabels(namespace, pod.Name, generateRedisMasterRoleLabel()) } -func (r *RedisFailoverHealer) setSlaveLabelIfNecessary(namespace string, pod v1.Pod) error { - for labelKey, labelValue := range pod.Labels { - if labelKey == redisRoleLabelKey && labelValue == redisRoleLabelSlave { - return nil - } - } - return r.k8sService.UpdatePodLabels(namespace, pod.Name, generateRedisSlaveRoleLabel()) +func (r *RedisFailoverHealer) setSlaveLabelIfNecessary(rf *redisfailoverv1.RedisFailover, pod v1.Pod, port, password string) error { + return setSlaveLabel(r.k8sService, r.opts, rf, pod, port, password) } func (r *RedisFailoverHealer) MakeMaster(ip string, rf *redisfailoverv1.RedisFailover) error { @@ -137,7 +134,7 @@ func (r *RedisFailoverHealer) SetOldestAsMaster(rf *redisfailoverv1.RedisFailove r.logger.WithField("redisfailover", rf.Name).WithField("namespace", rf.Namespace).Errorf("Make slave failed, slave pod ip: %s, master ip: %s, error: %v", pod.Status.PodIP, newMasterIP, err) } - err = r.setSlaveLabelIfNecessary(rf.Namespace, pod) + err = r.setSlaveLabelIfNecessary(rf, pod, port, password) if err != nil { return err } @@ -210,7 +207,7 @@ func (r *RedisFailoverHealer) SetMasterOnAll(masterIP string, rf *redisfailoverv continue } - err = r.setSlaveLabelIfNecessary(rf.Namespace, pod) + err = r.setSlaveLabelIfNecessary(rf, pod, port, password) if err != nil { return err } @@ -368,7 +365,7 @@ func (r *RedisFailoverHealer) PromoteBestReplica(newMasterIP string, rf *redisfa continue } - if err := r.setSlaveLabelIfNecessary(rf.Namespace, rp); err != nil { + if err := r.setSlaveLabelIfNecessary(rf, rp, port, password); err != nil { r.logger.WithField("redisfailover", rf.Name).WithField("namespace", rf.Namespace). Errorf("Failed to set slave label on pod %s: %v", rp.Name, err) reconcileErrs = append(reconcileErrs, err) diff --git a/service/redis/client.go b/service/redis/client.go index 89cdcbb13..c5dcdf0e5 100644 --- a/service/redis/client.go +++ b/service/redis/client.go @@ -38,6 +38,7 @@ type Client interface { MakeMaster(ip, port, password string) error MakeSlaveOf(ip, masterIP, password string) error MakeSlaveOfWithPort(ip, masterIP, masterPort, password string) error + DisconnectClients(ip, port, password string) error GetSentinelMonitor(ip string) (string, string, error) SetCustomSentinelConfig(ip string, configs []string) error SetCustomRedisConfig(ip string, port string, configs []string, password string) error @@ -344,6 +345,40 @@ func (c *client) MakeSlaveOfWithPort(ip, masterIP, masterPort, password string) return nil } +func closeClient(rClient *rediscli.Client) { + if err := rClient.Close(); err != nil { + log.Error(err.Error()) + } +} + +// DisconnectClients closes every normal and pub/sub client connection on the +// given instance. Replication links are left alone. +func (c *client) DisconnectClients(ip, port, password string) error { + options := &rediscli.Options{ + Addr: net.JoinHostPort(ip, port), + Password: password, + DB: 0, + } + rClient := rediscli.NewClient(options) + defer closeClient(rClient) + + var errs []error + for _, clientType := range []string{"normal", "pubsub"} { + if err := rClient.ClientKillByFilter(context.TODO(), "TYPE", clientType).Err(); err != nil { + errs = append(errs, fmt.Errorf("CLIENT KILL TYPE %s: %w", clientType, err)) + if IsUnreachableError(err) { + break + } + } + } + if err := errors.Join(errs...); err != nil { + c.metricsRecorder.RecordRedisOperation(metrics.KIND_REDIS, ip, metrics.DISCONNECT_CLIENTS, metrics.FAIL, getRedisError(errs[0])) + return err + } + c.metricsRecorder.RecordRedisOperation(metrics.KIND_REDIS, ip, metrics.DISCONNECT_CLIENTS, metrics.SUCCESS, metrics.NOT_APPLICABLE) + return nil +} + func (c *client) GetSentinelMonitor(ip string) (string, string, error) { options := &rediscli.Options{ Addr: net.JoinHostPort(ip, sentinelPort), diff --git a/service/redis/client_test.go b/service/redis/client_test.go index 031d5836a..6880eed6b 100644 --- a/service/redis/client_test.go +++ b/service/redis/client_test.go @@ -16,6 +16,7 @@ package redis import ( "context" "errors" + "io" "net" "os" "strconv" @@ -327,6 +328,128 @@ func TestMakeSlaveOfWithPort_SamePortDifferentIP(t *testing.T) { assert.True(t, linkUp) } +// dialRaw uses a raw connection: a pooled client would silently redial. +func dialRaw(t *testing.T, addr, cmd, wantReply string) net.Conn { + t.Helper() + conn, err := net.Dial("tcp", addr) + require.NoError(t, err) + t.Cleanup(func() { _ = conn.Close() }) + require.NoError(t, conn.SetDeadline(time.Now().Add(2*time.Second))) + _, err = conn.Write([]byte(cmd)) + require.NoError(t, err) + reply := make([]byte, len(wantReply)) + _, err = io.ReadFull(conn, reply) + require.NoError(t, err) + require.Equal(t, wantReply, string(reply)) + return conn +} + +// requireClosedByServer requires io.EOF; a read timeout means still open. +func requireClosedByServer(t *testing.T, conn net.Conn) { + t.Helper() + require.NoError(t, conn.SetDeadline(time.Now().Add(5*time.Second))) + _, err := conn.Read(make([]byte, 1)) + if netErr, ok := err.(net.Error); ok && netErr.Timeout() { + t.Fatalf("read timed out: the server never closed the connection: %v", err) + } + require.ErrorIs(t, err, io.EOF) +} + +// SLAVEOF is re-issued against healthy replicas, so it must not disconnect. +// Same port, different IPs: see TestMakeSlaveOfWithPort_MismatchedTargetPort. +func TestMakeSlaveOfWithPort_LeavesClientConnectionsAlone(t *testing.T) { + requireRedisServer(t) + otherIP, ok := nonLoopbackIPv4() + if !ok { + t.Skip("no non-loopback IPv4 address available on this machine") + } + port, err := findFreePort() + require.NoError(t, err) + + a := startRedisProcessOnAddr(t, testLoopbackIP, port) + b := startRedisProcessOnAddr(t, otherIP, port) + c := newTestClient() + + appConn := dialRaw(t, a.Addr(), "PING\r\n", "+PONG\r\n") + + require.NoError(t, c.MakeSlaveOfWithPort(a.IP, b.IP, strconv.Itoa(b.Port), "")) + + require.NoError(t, appConn.SetDeadline(time.Now().Add(2*time.Second))) + _, err = appConn.Write([]byte("PING\r\n")) + require.NoError(t, err) + pong := make([]byte, 7) + _, err = io.ReadFull(appConn, pong) + require.NoError(t, err, "SLAVEOF must not close existing client connections") + assert.Equal(t, "+PONG\r\n", string(pong)) +} + +func replicaLinkID(t *testing.T, master *redisProc) string { + t.Helper() + rc := rediscli.NewClient(&rediscli.Options{Addr: master.Addr()}) + defer func() { _ = rc.Close() }() + list, err := rc.Do(bgCtx(), "CLIENT", "LIST", "TYPE", "replica").Text() + require.NoError(t, err) + fields := strings.Fields(list) + require.NotEmpty(t, fields, "no replica connected to master") + require.True(t, strings.HasPrefix(fields[0], "id="), "unexpected CLIENT LIST output: %q", list) + return fields[0] +} + +func TestDisconnectClients_ClosesNormalAndPubSubClientsOnly(t *testing.T) { + requireRedisServer(t) + master := startRedisProcess(t, "--repl-diskless-sync-delay", "0") + replica := startReplicaOf(t, master) + c := newTestClient() + + require.True(t, waitForCondition(t, 10*time.Second, func() bool { + info, err := c.GetReplicationInfo(replica.IP, strconv.Itoa(replica.Port), "") + return err == nil && info.MasterLinkStatus == "up" + }), "replica never finished syncing") + linkBefore := replicaLinkID(t, master) + + normalConn := dialRaw(t, master.Addr(), "PING\r\n", "+PONG\r\n") + pubsubConn := dialRaw(t, master.Addr(), "SUBSCRIBE ch\r\n", + "*3\r\n$9\r\nsubscribe\r\n$2\r\nch\r\n:1\r\n") + + require.NoError(t, c.DisconnectClients(master.IP, strconv.Itoa(master.Port), "")) + + requireClosedByServer(t, normalConn) + requireClosedByServer(t, pubsubConn) + assert.Equal(t, linkBefore, replicaLinkID(t, master), + "the replication link must survive: a new client ID means it was killed and redialled") +} + +func TestDisconnectClients_ReturnsErrorWhenDenied(t *testing.T) { + requireRedisServer(t) + a := startRedisProcess(t, "--user", "default", "on", "nopass", "~*", "&*", "+@all", "-client|kill") + c := newTestClient() + + err := c.DisconnectClients(a.IP, strconv.Itoa(a.Port), "") + require.Error(t, err) + assert.ErrorContains(t, err, "TYPE normal") + assert.ErrorContains(t, err, "TYPE pubsub") + assert.ErrorContains(t, err, "NOPERM") +} + +// Unreachable: don't pay the dial timeout twice. +func TestDisconnectClients_ConnectionError(t *testing.T) { + port, err := findFreePort() + require.NoError(t, err) + c := newTestClient() + + err = c.DisconnectClients(testLoopbackIP, strconv.Itoa(port), "") + require.Error(t, err) + assert.ErrorContains(t, err, "TYPE normal") + assert.NotContains(t, err.Error(), "TYPE pubsub", "the second kill should be skipped once the node is known to be unreachable") +} + +func TestCloseClient_LogsCloseError(t *testing.T) { + rClient := rediscli.NewClient(&rediscli.Options{Addr: net.JoinHostPort(testLoopbackIP, "0")}) + closeClient(rClient) + require.Error(t, rClient.Close(), "a second Close should fail") + assert.NotPanics(t, func() { closeClient(rClient) }) +} + // TestMakeSlaveOfWithPort_MismatchedTargetPort documents a real bug found // while building this test suite (not fixed here, per instructions - see // the task summary for the full report): From 08a63be9f695f63b793a2a5b9a87679b0c712728 Mon Sep 17 00:00:00 2001 From: Michael Strassberger Date: Fri, 25 Sep 2026 23:02:43 +0200 Subject: [PATCH 08/24] Add a skill for running kind in the Claude Code cloud sandbox (#187) A plain `kind create cluster` fails in the sandbox, and a working cluster can't pull images. The skill's scripts: - patch the kind config for the sandbox's cgroup v1 kernel (failCgroupV1: false, containerd restrict_oom_score_adj); - mirror docker.io and quay.io through a local registry, since the node can't reach any registry and the operator defaults to pullPolicy Always; - route the pod subnet to the host for the integration tests; - build the operator image without docker/app/Dockerfile, whose apk step has no network here. .gitignore now lets .claude/skills be committed. Claude-Session: https://claude.ai/code/session_01G4mmyo3sqLwf23m5FE2nBE Co-authored-by: Claude --- .claude/skills/kind-cluster/SKILL.md | 88 ++++++++++++++++++++++ .claude/skills/kind-cluster/build-image.sh | 27 +++++++ .claude/skills/kind-cluster/kind-up.sh | 72 ++++++++++++++++++ .claude/skills/kind-cluster/registry.sh | 63 ++++++++++++++++ .gitignore | 3 +- 5 files changed, 252 insertions(+), 1 deletion(-) create mode 100644 .claude/skills/kind-cluster/SKILL.md create mode 100755 .claude/skills/kind-cluster/build-image.sh create mode 100755 .claude/skills/kind-cluster/kind-up.sh create mode 100755 .claude/skills/kind-cluster/registry.sh diff --git a/.claude/skills/kind-cluster/SKILL.md b/.claude/skills/kind-cluster/SKILL.md new file mode 100644 index 000000000..dc4927960 --- /dev/null +++ b/.claude/skills/kind-cluster/SKILL.md @@ -0,0 +1,88 @@ +--- +name: kind-cluster +description: Create a kind Kubernetes cluster inside the Claude Code cloud sandbox and run this repo's integration tests or a Helm-installed operator against it. Use when a change needs e2e testing on a real API server. +--- + +# kind cluster in the cloud sandbox + +A plain `kind create cluster` fails in the sandbox, and a working cluster +still can't pull images. The scripts here handle all of this. + +## Create + +```sh +.claude/skills/kind-cluster/kind-up.sh e2e v1.35.0 244 +export KUBECONFIG=/tmp/kind-e2e/kubeconfig +``` + +The script: + +1. Starts `dockerd` if it isn't running. +2. Writes a kind config with two sandbox fixes: + - `failCgroupV1: false`: the sandbox is cgroup v1, and kubelet >= 1.35 + refuses to start on it. + - containerd `restrict_oom_score_adj = true`: the kernel rejects negative + `oom_score_adj`, so every pod sandbox would fail with `can't get final + child's PID from pipe: EOF`. This affects every node version, 1.34 too. +3. Makes a local registry, `kind-registry`, the node's mirror for docker.io + and quay.io (`registry.sh`). + - The node can't reach any registry, because its `HTTPS_PROXY` points at + the sandbox's 127.0.0.1 proxy. + - `kind load` isn't enough: the operator defaults to `imagePullPolicy: + Always`, so preloaded images still fail with `ImagePullBackOff`. +4. Copies the images from `api/redisfailover/v1/defaults.go`, plus any in + `EXTRA_IMAGES`, into the registry. + - quay.io is blocked from the host too, so `registry.sh` falls back to the + Docker Hub and `mirror.gcr.io` copies. + - Add another image any time with `registry.sh push IMAGE`. +5. Adds a host route to the pod subnet: the integration tests connect to + Redis pod IPs. +6. Applies the RedisFailover CRD. + +## Several clusters at once + +Give each cluster its own name and subnet octet, e.g. `a … 181` and +`b … 185`. The pod subnets must differ, or the host routes collide. The +registry is shared. The machine has 4 CPUs, so two single-node clusters is +the practical limit. + +## Integration tests + +```sh +go test ./test/integration/... -tags integration -v -timeout 30m +``` + +They run the operator in-process against `$KUBECONFIG`. + +## Operator image and Helm + +```sh +.claude/skills/kind-cluster/build-image.sh pr # redis-operator:pr +helm upgrade --install redis-operator ./charts/redisoperator \ + --set image.repository=redis-operator --set image.tag=pr --wait +``` + +- `docker/app/Dockerfile` fails here, because `apk add` can't reach the + package mirrors. `build-image.sh` builds the binary on the host instead. +- Install helm with `GOBIN=/usr/local/bin go install helm.sh/helm/v3/cmd/helm@v3.19.0` + if it's missing. +- `.github/workflows/e2e.yml` has a RedisFailover manifest and checks you + can reuse. +- Don't run the integration tests while a Helm-installed operator is + running: both would reconcile the test's RedisFailovers. + +## Debugging a failed create + +Add `--retain` to keep the node. Then read +`docker exec NAME-control-plane journalctl -u kubelet --no-pager` and +`journalctl -u containerd`. + +## Clean up + +```sh +kind delete cluster --name e2e +docker rm -f kind-registry # once no cluster needs it +``` + +This leaves the host route behind. It is harmless, and a new cluster +replaces it. diff --git a/.claude/skills/kind-cluster/build-image.sh b/.claude/skills/kind-cluster/build-image.sh new file mode 100755 index 000000000..0b196677d --- /dev/null +++ b/.claude/skills/kind-cluster/build-image.sh @@ -0,0 +1,27 @@ +#!/usr/bin/env bash +# Builds the operator from the current checkout as redis-operator:TAG and +# serves it to the kind nodes through the local registry. +# +# Usage: build-image.sh [TAG] (default TAG: dev) +# +# docker/app/Dockerfile can't be used in the sandbox: its `apk add` steps have +# no route to the package mirrors. Build the binary on the host instead and +# copy it into the same alpine base with the same non-root user. +set -euo pipefail + +tag=${1:-dev} +repo=$(git rev-parse --show-toplevel) +here=$(cd "$(dirname "$0")" && pwd) +ctx=$(mktemp -d) +trap 'rm -rf "$ctx"' EXIT + +(cd "$repo" && CGO_ENABLED=0 go build -o "$ctx/redis-operator" -ldflags "-w" ./cmd/redisoperator) +cat >"$ctx/Dockerfile" <<'EOF' +FROM alpine:latest +COPY redis-operator /usr/local/bin/redis-operator +RUN addgroup -g 1000 rf && adduser -D -u 1000 -G rf rf +USER rf +ENTRYPOINT ["/usr/local/bin/redis-operator"] +EOF +docker build -q -t "redis-operator:$tag" "$ctx" >/dev/null +"$here/registry.sh" push "redis-operator:$tag" diff --git a/.claude/skills/kind-cluster/kind-up.sh b/.claude/skills/kind-cluster/kind-up.sh new file mode 100755 index 000000000..ecca2727d --- /dev/null +++ b/.claude/skills/kind-cluster/kind-up.sh @@ -0,0 +1,72 @@ +#!/usr/bin/env bash +# Creates a kind cluster that works inside the Claude Code cloud sandbox and +# prepares it for this repo's integration tests. +# +# Usage: kind-up.sh NAME [NODE_VERSION] [SUBNET] +# NAME cluster name; state goes to /tmp/kind-NAME +# NODE_VERSION kindest/node tag (default v1.35.0) +# SUBNET second octet of the pod subnet 10.SUBNET.0.0/16 (default 244). +# Use a different one per cluster when running several. +set -euo pipefail + +name=${1:?usage: kind-up.sh NAME [NODE_VERSION] [SUBNET]} +version=${2:-v1.35.0} +subnet=${3:-244} +dir=/tmp/kind-$name +repo=$(git rev-parse --show-toplevel) +here=$(cd "$(dirname "$0")" && pwd) +pod_cidr=10.$subnet.0.0/16 +node=$name-control-plane +mkdir -p "$dir" + +if ! docker info >/dev/null 2>&1; then + (dockerd >/tmp/dockerd.log 2>&1 &) + for _ in $(seq 1 30); do docker info >/dev/null 2>&1 && break; sleep 1; done +fi + +# The sandbox runs cgroup v1, which kubelet >= 1.35 refuses by default, and +# its kernel rejects negative oom_score_adj values, which containerd sets +# on every pod sandbox unless restricted. +cat >"$dir/kind.yaml" < library/redis:7, quay.io/a/b:1 -> a/b:1. +repo_path() { + local first=${1%%/*} + if [[ $1 != */* ]]; then + echo "library/$1" + elif [[ $first == *.* || $first == *:* ]]; then + echo "${1#*/}" + else + echo "$1" + fi +} + +case $cmd in +connect) + if ! docker inspect "$reg" >/dev/null 2>&1; then + docker pull -q registry:2 >/dev/null 2>&1 || + { docker pull -q mirror.gcr.io/library/registry:2 >/dev/null && docker tag mirror.gcr.io/library/registry:2 registry:2; } + docker run -d --restart=always --name "$reg" -p 127.0.0.1:5001:5000 registry:2 >/dev/null + fi + docker network connect kind "$reg" 2>/dev/null || true + # Plain HTTP, so containerd doesn't send it through the unreachable + # HTTPS proxy. + for host in docker.io quay.io; do + docker exec "$arg-control-plane" mkdir -p "/etc/containerd/certs.d/$host" + printf '[host."http://%s:5000"]\n capabilities = ["pull", "resolve"]\n' "$reg" | + docker exec -i "$arg-control-plane" cp /dev/stdin "/etc/containerd/certs.d/$host/hosts.toml" + done + ;; +push) + img=$arg + path=$(repo_path "$img") + if ! docker image inspect "$img" >/dev/null 2>&1; then + # quay.io is blocked and Docker Hub may rate limit; try Docker Hub's copy + # and mirror.gcr.io too. + for src in "$img" "$path" "mirror.gcr.io/$path"; do + if docker pull -q "$src" >/dev/null 2>&1; then + [[ $src == "$img" ]] || docker tag "$src" "$img" + break + fi + done + fi + docker tag "$img" "localhost:5001/$path" + docker push -q --platform linux/amd64 "localhost:5001/$path" >/dev/null + echo "$img -> $reg/$path" + ;; +*) + echo "usage: registry.sh connect CLUSTER | push IMAGE" >&2 + exit 1 + ;; +esac diff --git a/.gitignore b/.gitignore index e12fa332c..e2cf74da8 100644 --- a/.gitignore +++ b/.gitignore @@ -4,5 +4,6 @@ .idea/ /tmp vendor -.claude/ +.claude/* +!.claude/skills/ coverage.out \ No newline at end of file From 8f84d29d7d544efddce744bc6061be9d185b509e Mon Sep 17 00:00:00 2001 From: Michael Strassberger Date: Sat, 26 Sep 2026 00:10:46 +0200 Subject: [PATCH 09/24] Fall back to a mirror for every image the kind skill pulls (#188) build-image.sh let docker build pull alpine straight from Docker Hub, which fails when Docker Hub rate limits (429). registry.sh already fell back to Docker Hub's copy and mirror.gcr.io for the images it pushes. Move that fallback into a `registry.sh pull` subcommand and use it for alpine, the registry:2 image and the kind node image. Claude-Session: https://claude.ai/code/session_01G4mmyo3sqLwf23m5FE2nBE Co-authored-by: Claude --- .claude/skills/kind-cluster/build-image.sh | 2 ++ .claude/skills/kind-cluster/kind-up.sh | 1 + .claude/skills/kind-cluster/registry.sh | 38 ++++++++++++++-------- 3 files changed, 27 insertions(+), 14 deletions(-) diff --git a/.claude/skills/kind-cluster/build-image.sh b/.claude/skills/kind-cluster/build-image.sh index 0b196677d..5296466c2 100755 --- a/.claude/skills/kind-cluster/build-image.sh +++ b/.claude/skills/kind-cluster/build-image.sh @@ -23,5 +23,7 @@ RUN addgroup -g 1000 rf && adduser -D -u 1000 -G rf rf USER rf ENTRYPOINT ["/usr/local/bin/redis-operator"] EOF +# docker build pulls a missing base image straight from Docker Hub. +"$here/registry.sh" pull alpine:latest docker build -q -t "redis-operator:$tag" "$ctx" >/dev/null "$here/registry.sh" push "redis-operator:$tag" diff --git a/.claude/skills/kind-cluster/kind-up.sh b/.claude/skills/kind-cluster/kind-up.sh index ecca2727d..4c4e8f02a 100755 --- a/.claude/skills/kind-cluster/kind-up.sh +++ b/.claude/skills/kind-cluster/kind-up.sh @@ -47,6 +47,7 @@ nodes: failCgroupV1: false EOF +"$here/registry.sh" pull "kindest/node:$version" kind create cluster --name "$name" --image "kindest/node:$version" \ --config "$dir/kind.yaml" --kubeconfig "$dir/kubeconfig" --wait 180s export KUBECONFIG=$dir/kubeconfig diff --git a/.claude/skills/kind-cluster/registry.sh b/.claude/skills/kind-cluster/registry.sh index 8c3e3879c..66dc0e32e 100755 --- a/.claude/skills/kind-cluster/registry.sh +++ b/.claude/skills/kind-cluster/registry.sh @@ -4,10 +4,11 @@ # # Usage: registry.sh connect CLUSTER start the registry, make it CLUSTER's mirror # registry.sh push IMAGE copy IMAGE from the host into it +# registry.sh pull IMAGE pull IMAGE to the host, via a mirror if needed set -euo pipefail cmd=${1:-} -arg=${2:?usage: registry.sh connect CLUSTER | push IMAGE} +arg=${2:?usage: registry.sh connect CLUSTER | push IMAGE | pull IMAGE} reg=kind-registry # Repository path without the registry host, as containerd asks the mirror @@ -23,11 +24,26 @@ repo_path() { fi } +# Pulls IMAGE unless the host has it. quay.io is blocked and Docker Hub may +# rate limit, so fall back to Docker Hub's copy and mirror.gcr.io. +ensure_local() { + local img=$1 path src + docker image inspect "$img" >/dev/null 2>&1 && return + path=$(repo_path "$img") + for src in "$img" "$path" "mirror.gcr.io/$path"; do + if docker pull -q "$src" >/dev/null 2>&1; then + [[ $src == "$img" ]] || docker tag "$src" "$img" + return + fi + done + echo "failed to pull $img" >&2 + return 1 +} + case $cmd in connect) if ! docker inspect "$reg" >/dev/null 2>&1; then - docker pull -q registry:2 >/dev/null 2>&1 || - { docker pull -q mirror.gcr.io/library/registry:2 >/dev/null && docker tag mirror.gcr.io/library/registry:2 registry:2; } + ensure_local registry:2 docker run -d --restart=always --name "$reg" -p 127.0.0.1:5001:5000 registry:2 >/dev/null fi docker network connect kind "$reg" 2>/dev/null || true @@ -39,25 +55,19 @@ connect) docker exec -i "$arg-control-plane" cp /dev/stdin "/etc/containerd/certs.d/$host/hosts.toml" done ;; +pull) + ensure_local "$arg" + ;; push) img=$arg path=$(repo_path "$img") - if ! docker image inspect "$img" >/dev/null 2>&1; then - # quay.io is blocked and Docker Hub may rate limit; try Docker Hub's copy - # and mirror.gcr.io too. - for src in "$img" "$path" "mirror.gcr.io/$path"; do - if docker pull -q "$src" >/dev/null 2>&1; then - [[ $src == "$img" ]] || docker tag "$src" "$img" - break - fi - done - fi + ensure_local "$img" docker tag "$img" "localhost:5001/$path" docker push -q --platform linux/amd64 "localhost:5001/$path" >/dev/null echo "$img -> $reg/$path" ;; *) - echo "usage: registry.sh connect CLUSTER | push IMAGE" >&2 + echo "usage: registry.sh connect CLUSTER | push IMAGE | pull IMAGE" >&2 exit 1 ;; esac From 8314bf5ce30df39398de0daaf02bebe2901cca50 Mon Sep 17 00:00:00 2001 From: Michael Strassberger Date: Sat, 26 Sep 2026 00:11:20 +0200 Subject: [PATCH 10/24] Pass watch bookmarks and errors through the namespace filter (#186) With client-go 1.35+ the informer streams its initial list and waits for the initial-events-end bookmark. The RedisFailover watch filter dropped it, since it has no namespace, so with a restrictive --supported-namespaces-regex the operator never finished syncing and never reconciled. It also dropped watch errors such as 410 Gone. Let bookmark and error events through, and run the integration tests with a regex that only matches their namespace. Claude-Session: https://claude.ai/code/session_01G4mmyo3sqLwf23m5FE2nBE Co-authored-by: Claude --- operator/redisfailover/factory.go | 6 +++ operator/redisfailover/factory_test.go | 54 +++++++++++++++++++ .../redisfailover/creation_test.go | 2 +- .../operator_managed_rollout_test.go | 2 +- 4 files changed, 62 insertions(+), 2 deletions(-) diff --git a/operator/redisfailover/factory.go b/operator/redisfailover/factory.go index 8057bfb35..f14ae2317 100644 --- a/operator/redisfailover/factory.go +++ b/operator/redisfailover/factory.go @@ -97,6 +97,12 @@ func NewRedisFailoverRetriever(cfg Config, cli k8s.Services) controller.Retrieve return watcher, err } watcher = watch.Filter(watcher, func(event watch.Event) (watch.Event, bool) { + // Bookmarks and errors belong to no namespace. The informer + // needs them to finish its initial sync and to relist after + // an expired watch. + if event.Type == watch.Bookmark || event.Type == watch.Error { + return event, true + } rf, ok := event.Object.(*redisfailoverv1.RedisFailover) if !ok { return event, false diff --git a/operator/redisfailover/factory_test.go b/operator/redisfailover/factory_test.go index fbe324257..577b19d72 100644 --- a/operator/redisfailover/factory_test.go +++ b/operator/redisfailover/factory_test.go @@ -3,16 +3,22 @@ package redisfailover import ( "context" "errors" + "sync/atomic" "testing" + "time" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/mock" "github.com/stretchr/testify/require" corev1 "k8s.io/api/core/v1" + apierrors "k8s.io/apimachinery/pkg/api/errors" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" "k8s.io/apimachinery/pkg/watch" + clientfeatures "k8s.io/client-go/features" + clientfeaturestesting "k8s.io/client-go/features/testing" fakekubernetes "k8s.io/client-go/kubernetes/fake" + "k8s.io/client-go/tools/cache" kooperlog "github.com/spotahome/kooper/v2/log" @@ -150,6 +156,54 @@ func TestNewRedisFailoverRetrieverWatchNilWatcherNoError(t *testing.T) { mk.AssertExpectations(t) } +// rfInformer builds an informer on NewRedisFailoverRetriever the way the +// controller does. +func rfInformer(t *testing.T, mk *mK8SService.Services) cache.SharedIndexInformer { + t.Helper() + retriever := NewRedisFailoverRetriever(Config{SupportedNamespacesRegex: "^allowed$"}, mk) + lw := &cache.ListWatch{ + ListWithContextFunc: retriever.List, + WatchFuncWithContext: retriever.Watch, + } + informer := cache.NewSharedIndexInformer(lw, nil, 0, cache.Indexers{}) + ctx, cancel := context.WithCancel(context.Background()) + t.Cleanup(cancel) + go informer.RunWithContext(ctx) + return informer +} + +func TestRedisFailoverInformerSyncsFromAWatchListStream(t *testing.T) { + clientfeaturestesting.SetFeatureDuringTest(t, clientfeatures.WatchListClient, true) + mk := &mK8SService.Services{} + stream := watch.NewFake() + mk.On("WatchRedisFailovers", mock.Anything, "", mock.Anything).Return(stream, nil) + + informer := rfInformer(t, mk) + stream.Add(&redisfailoverv1.RedisFailover{ObjectMeta: metav1.ObjectMeta{Name: "rf", Namespace: "allowed", ResourceVersion: "5"}}) + stream.Action(watch.Bookmark, &redisfailoverv1.RedisFailover{ObjectMeta: metav1.ObjectMeta{ + ResourceVersion: "6", + Annotations: map[string]string{metav1.InitialEventsAnnotationKey: "true"}, + }}) + + assert.Eventually(t, informer.HasSynced, 3*time.Second, 10*time.Millisecond, "the end-of-initial-events bookmark must reach the informer") +} + +func TestRedisFailoverInformerRelistsOnAnExpiredWatch(t *testing.T) { + clientfeaturestesting.SetFeatureDuringTest(t, clientfeatures.WatchListClient, false) + mk := &mK8SService.Services{} + var lists atomic.Int32 + mk.On("ListRedisFailovers", mock.Anything, "", mock.Anything).Run(func(mock.Arguments) { lists.Add(1) }). + Return(&redisfailoverv1.RedisFailoverList{ListMeta: metav1.ListMeta{ResourceVersion: "1"}}, nil) + stream := watch.NewFake() + mk.On("WatchRedisFailovers", mock.Anything, "", mock.Anything).Return(stream, nil) + + informer := rfInformer(t, mk) + assert.Eventually(t, informer.HasSynced, 3*time.Second, 10*time.Millisecond) + stream.Error(&apierrors.NewResourceExpired("too old resource version").ErrStatus) + + assert.Eventually(t, func() bool { return lists.Load() > 1 }, 5*time.Second, 10*time.Millisecond, "an expired watch must make the informer relist") +} + // ----------------------------------------------------------------------- // kooperlogger.WithKV // ----------------------------------------------------------------------- diff --git a/test/integration/redisfailover/creation_test.go b/test/integration/redisfailover/creation_test.go index 1eb2b5888..2c939ab0b 100644 --- a/test/integration/redisfailover/creation_test.go +++ b/test/integration/redisfailover/creation_test.go @@ -190,7 +190,7 @@ func TestRedisFailover(t *testing.T) { // needs two separate Handle() calls to replace two pods) can then stall // for the full resync interval. A short one here keeps that stall short // instead of letting it hit the 3-minute fallback. - redisfailoverOperator, err := redisfailover.New(redisfailover.Config{SyncInterval: 2}, k8sservice, k8sClient, namespace, redisClient, metrics.Dummy, log.Dummy) + redisfailoverOperator, err := redisfailover.New(redisfailover.Config{SyncInterval: 2, SupportedNamespacesRegex: "^" + namespace + "$"}, k8sservice, k8sClient, namespace, redisClient, metrics.Dummy, log.Dummy) require.NoError(err) // Its own cancelable context, not context.Background(): without this, diff --git a/test/integration/redisfailover/operator_managed_rollout_test.go b/test/integration/redisfailover/operator_managed_rollout_test.go index 249787498..44cacf5b8 100644 --- a/test/integration/redisfailover/operator_managed_rollout_test.go +++ b/test/integration/redisfailover/operator_managed_rollout_test.go @@ -227,7 +227,7 @@ func TestRedisFailoverOperatorManagedModeRollout(t *testing.T) { // slave, then the master), so a missed self-trigger between them can // stall for the full resync interval - a short one here keeps that // stall short instead of letting it hit the 3-minute fallback. - redisfailoverOperator, err := redisfailover.New(redisfailover.Config{SyncInterval: 2}, k8sservice, k8sClient, ommNamespace, redisClient, metrics.Dummy, log.Dummy) + redisfailoverOperator, err := redisfailover.New(redisfailover.Config{SyncInterval: 2, SupportedNamespacesRegex: "^" + ommNamespace + "$"}, k8sservice, k8sClient, ommNamespace, redisClient, metrics.Dummy, log.Dummy) require.NoError(err) go func() { From d075fc5b660d1639530933b3a81e74e8fa3884b2 Mon Sep 17 00:00:00 2001 From: Michael Strassberger Date: Sat, 26 Sep 2026 00:34:58 +0200 Subject: [PATCH 11/24] Reconcile a RedisFailover when its pods change (#181) A rollout replaces one pod per reconcile, and each next step waited for the resync because nothing queued the RedisFailover again. Replace kooper's controller with one that has a single queue keyed by RedisFailover, fed by RedisFailover events and by events on the pods of handled RedisFailovers, so pod changes drive the next step. The pod cache keeps metadata only, and a pod watch that can't sync doesn't block reconciling. One queued event is counted per update unless a pod's owner changed, as with kooper. Because the next reconcile now follows a delete immediately: - replace another pod only once the last replacement has settled (all pods present, none terminating, updated pods ready), and log at Info why a rollout waits; - in operator-managed mode, wait for a terminating master that is still ready to exit before electing a new one. A stopping replica doesn't block the election; - don't dial terminating pods in CheckAllSlavesFromMaster; their IP is often gone and every rollout step would stall for the dial timeout. Claude-Session: https://claude.ai/code/session_01G4mmyo3sqLwf23m5FE2nBE Co-authored-by: Claude --- operator/redisfailover/checker.go | 62 +++ operator/redisfailover/checker_test.go | 170 ++++++- operator/redisfailover/controller.go | 240 ++++++++++ operator/redisfailover/controller_test.go | 427 ++++++++++++++++++ operator/redisfailover/factory.go | 16 +- operator/redisfailover/factory_test.go | 7 + operator/redisfailover/handler.go | 11 +- operator/redisfailover/service/check.go | 10 + operator/redisfailover/service/check_test.go | 34 ++ operator/redisfailover/util/pod.go | 9 + operator/redisfailover/util/pod_test.go | 13 + .../redisfailover/creation_test.go | 14 +- .../operator_managed_rollout_test.go | 15 +- 13 files changed, 979 insertions(+), 49 deletions(-) create mode 100644 operator/redisfailover/controller.go create mode 100644 operator/redisfailover/controller_test.go diff --git a/operator/redisfailover/checker.go b/operator/redisfailover/checker.go index 2ef4c4fca..559cf3370 100644 --- a/operator/redisfailover/checker.go +++ b/operator/redisfailover/checker.go @@ -3,15 +3,18 @@ package redisfailover import ( "context" "errors" + "fmt" "strconv" "time" "github.com/saremox/redis-operator/service/k8s" + appsv1 "k8s.io/api/apps/v1" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" redisfailoverv1 "github.com/saremox/redis-operator/api/redisfailover/v1" "github.com/saremox/redis-operator/metrics" rfservice "github.com/saremox/redis-operator/operator/redisfailover/service" + "github.com/saremox/redis-operator/operator/redisfailover/util" "github.com/saremox/redis-operator/service/redis" ) @@ -59,6 +62,9 @@ func (r *RedisFailoverHandler) UpdateRedisesPods(rf *redisfailoverv1.RedisFailov return err } if revision != ssUR { + if settled, err := r.redisPodsSettled(rf, ssUR); err != nil || !settled { + return err + } //Delete pod and wait next round to check if the new one is synced err = r.rfHealer.DeletePod(pod, rf) if err != nil { @@ -113,6 +119,9 @@ func (r *RedisFailoverHandler) UpdateRedisesPods(rf *redisfailoverv1.RedisFailov } } + if settled, err := r.redisPodsSettled(rf, ssUR); err != nil || !settled { + return err + } err = r.rfHealer.DeletePod(master, rf) if err != nil { return err @@ -125,6 +134,53 @@ func (r *RedisFailoverHandler) UpdateRedisesPods(rf *redisfailoverv1.RedisFailov return nil } +// redisPodsSettled reports whether the last redis pod replacement has +// finished: the StatefulSet has all its pods, none is being deleted, and every +// pod already on the update revision is ready. Pod events start the next +// reconcile right after a delete, so without this check a rollout would delete +// several pods at once. +func (r *RedisFailoverHandler) redisPodsSettled(rf *redisfailoverv1.RedisFailover, updateRevision string) (bool, error) { + pods, err := r.k8sservice.GetStatefulSetPods(rf.Namespace, rfservice.GetRedisName(rf)) + if err != nil { + return false, err + } + wait := func(reason string) (bool, error) { + r.logger.WithField("namespace", rf.Namespace).WithField("name", rf.Name).Infof("redis rollout waits: %s", reason) + return false, nil + } + if len(pods.Items) < int(rf.Spec.Redis.Replicas) { + return wait(fmt.Sprintf("%d of %d pods exist", len(pods.Items), rf.Spec.Redis.Replicas)) + } + for i := range pods.Items { + pod := &pods.Items[i] + if pod.DeletionTimestamp != nil { + return wait("pod " + pod.Name + " is terminating") + } + if pod.Labels[appsv1.ControllerRevisionHashLabelKey] == updateRevision && !util.PodIsReady(pod) { + return wait("pod " + pod.Name + " is not ready") + } + } + return true, nil +} + +// masterPodStopping reports whether the master's pod is being deleted but +// still ready, i.e. still taking writes. A pod on a lost node is not ready, +// so it doesn't block anything. +func (r *RedisFailoverHandler) masterPodStopping(rf *redisfailoverv1.RedisFailover) (bool, error) { + pods, err := r.k8sservice.GetStatefulSetPods(rf.Namespace, rfservice.GetRedisName(rf)) + if err != nil { + return false, err + } + for i := range pods.Items { + pod := &pods.Items[i] + if pod.DeletionTimestamp != nil && rfservice.IsMasterPod(pod) && util.PodIsReady(pod) { + r.logger.WithField("namespace", rf.Namespace).WithField("name", rf.Name).WithField("pod", pod.Name).Info("waiting for the stopping master pod to exit before electing a master") + return true, nil + } + } + return false, nil +} + // CheckAndHeal runs verifcation checks to ensure the RedisFailover is in an expected and healthy state. // If the checks do not match up to expectations, an attempt will be made to "heal" the RedisFailover into a healthy state. func (r *RedisFailoverHandler) CheckAndHeal(rf *redisfailoverv1.RedisFailover) error { @@ -411,6 +467,12 @@ func (r *RedisFailoverHandler) checkAndHealOperatorManagedMode(rf *redisfailover switch nMasters { case 0: + // A master whose pod is being deleted is no longer counted but may + // still take writes. Wait for it to stop so a promoted replica + // doesn't lose them. + if stopping, err := r.masterPodStopping(rf); err != nil || stopping { + return err + } // No master available - elect one setRedisCheckerMetrics(r.mClient, "redis", rf.Namespace, rf.Name, metrics.NO_MASTER, metrics.NOT_APPLICABLE, errors.New("no masters detected")) r.logger.WithField("redisfailover", rf.ObjectMeta.Name).WithField("namespace", rf.ObjectMeta.Namespace).Warningf("No master available, operator will elect one") diff --git a/operator/redisfailover/checker_test.go b/operator/redisfailover/checker_test.go index c71fa0aca..011277719 100644 --- a/operator/redisfailover/checker_test.go +++ b/operator/redisfailover/checker_test.go @@ -310,7 +310,7 @@ func TestCheckAndHeal(t *testing.T) { sentinel := "1.1.1.1" config := generateConfig() - mk := &mK8SService.Services{} + mk := settledK8sServices() // CheckAndHeal always defers updateStatus, on every return path. mk.On("UpdateRedisFailoverStatus", mock.Anything, mock.Anything, mock.Anything, mock.Anything).Return() mrfs := &mRFService.RedisFailoverClient{} @@ -753,7 +753,7 @@ func TestCheckAndHealOperatorManagedMode(t *testing.T) { rf := operatorManagedRF() config := generateConfig() - mk := &mK8SService.Services{} + mk := settledK8sServices() // CheckAndHeal always defers updateStatus, on every return path. mk.On("UpdateRedisFailoverStatus", mock.Anything, mock.Anything, mock.Anything, mock.Anything).Return() mrfs := &mRFService.RedisFailoverClient{} @@ -820,7 +820,7 @@ func TestUpdateStatusLastChanged(t *testing.T) { rf.Status.LastChanged = "2020-01-01T00:00:00Z" config := generateConfig() - mk := &mK8SService.Services{} + mk := settledK8sServices() mk.On("UpdateRedisFailoverStatus", mock.Anything, mock.Anything, mock.Anything, mock.Anything).Return() mrfs := &mRFService.RedisFailoverClient{} mrfc := &mRFService.RedisFailoverCheck{} @@ -848,7 +848,7 @@ func TestUpdateStatusLastChanged(t *testing.T) { rf.Status.LastChanged = previousLastChanged config := generateConfig() - mk := &mK8SService.Services{} + mk := settledK8sServices() mk.On("UpdateRedisFailoverStatus", mock.Anything, mock.Anything, mock.Anything, mock.Anything).Return() mrfs := &mRFService.RedisFailoverClient{} mrfc := &mRFService.RedisFailoverCheck{} @@ -1230,7 +1230,7 @@ func TestCheckAndHealPlainModeErrorBranches(t *testing.T) { } config := generateConfig() - mk := &mK8SService.Services{} + mk := settledK8sServices() // CheckAndHeal always defers updateStatus, on every return path. mk.On("UpdateRedisFailoverStatus", mock.Anything, mock.Anything, mock.Anything, mock.Anything).Return() mrfs := &mRFService.RedisFailoverClient{} @@ -1379,7 +1379,7 @@ func TestCheckAndHealBootstrapModeErrorBranches(t *testing.T) { rf.Spec.BootstrapNode.AllowSentinels = test.allowSentinels config := generateConfig() - mk := &mK8SService.Services{} + mk := settledK8sServices() // CheckAndHeal always defers updateStatus, on every return path. mk.On("UpdateRedisFailoverStatus", mock.Anything, mock.Anything, mock.Anything, mock.Anything).Return() mrfs := &mRFService.RedisFailoverClient{} @@ -1974,7 +1974,7 @@ func TestUpdate(t *testing.T) { } } - mk := &mK8SService.Services{} + mk := settledK8sServices() handler := rfOperator.NewRedisFailoverHandler(config, mrfs, mrfc, mrfh, mk, metrics.Dummy, log.Dummy) err := handler.UpdateRedisesPods(rf) @@ -2020,7 +2020,7 @@ func TestUpdateRedisesPodsOperatorManagedModeSkipsSentinelGate(t *testing.T) { mrfc.On("GetRedisRevisionHash", "master", rf).Once().Return("9", nil) // stale mrfh.On("DeletePod", "master", rf).Once().Return(nil) - mk := &mK8SService.Services{} + mk := settledK8sServices() handler := rfOperator.NewRedisFailoverHandler(config, mrfs, mrfc, mrfh, mk, metrics.Dummy, log.Dummy) err := handler.UpdateRedisesPods(rf) @@ -2136,7 +2136,7 @@ func TestUpdateRedisesPodsErrorBranches(t *testing.T) { rf := generateRF(false, false) config := generateConfig() - mk := &mK8SService.Services{} + mk := settledK8sServices() mrfs := &mRFService.RedisFailoverClient{} mrfc := &mRFService.RedisFailoverCheck{} mrfh := &mRFService.RedisFailoverHeal{} @@ -2153,3 +2153,155 @@ func TestUpdateRedisesPodsErrorBranches(t *testing.T) { }) } } + +// settledK8sServices returns a k8s service mock whose redis pods are all up, +// so the pod-replacement and master-election waits don't hold anything back. +func settledK8sServices() *mK8SService.Services { + mk := &mK8SService.Services{} + mk.On("GetStatefulSetPods", mock.Anything, mock.Anything).Maybe().Return(&corev1.PodList{Items: make([]corev1.Pod, 5)}, nil) + return mk +} + +func redisPod(revision string, ready, deleting bool) corev1.Pod { + pod := corev1.Pod{ObjectMeta: metav1.ObjectMeta{Labels: map[string]string{appsv1.ControllerRevisionHashLabelKey: revision}}} + if ready { + pod.Status.Conditions = []corev1.PodCondition{{Type: corev1.PodReady, Status: corev1.ConditionTrue}} + } + if deleting { + pod.DeletionTimestamp = &metav1.Time{Time: time.Now()} + } + return pod +} + +func masterPod(pod corev1.Pod) corev1.Pod { + pod.Labels["redisfailovers-role"] = "master" + return pod +} + +func TestUpdateRedisesPodsWaitsForTheLastReplacement(t *testing.T) { + tests := []struct { + name string + pods []corev1.Pod + podsErr error + wantDelete bool + }{ + { + name: "all pods up", + pods: []corev1.Pod{redisPod("old", true, false), redisPod("old", true, false), redisPod("new", true, false)}, + wantDelete: true, + }, + { + name: "a pod is being deleted", + pods: []corev1.Pod{redisPod("old", true, false), redisPod("old", true, false), redisPod("old", true, true)}, + }, + { + name: "the deleted pod is not recreated yet", + pods: []corev1.Pod{redisPod("old", true, false), redisPod("old", true, false)}, + }, + { + name: "the recreated pod is not ready yet", + pods: []corev1.Pod{redisPod("old", true, false), redisPod("old", true, false), redisPod("new", false, false)}, + }, + { + name: "an old pod that is not ready doesn't block", + pods: []corev1.Pod{redisPod("old", true, false), redisPod("old", false, false), redisPod("new", true, false)}, + wantDelete: true, + }, + { + name: "listing pods fails", + podsErr: errors.New("list err"), + }, + } + + for _, test := range tests { + for _, stale := range []string{"slave", "master"} { + t.Run(test.name+"/"+stale, func(t *testing.T) { + rf := operatorManagedRF() + rf.Spec.Redis.Replicas = 3 + mrfc := &mRFService.RedisFailoverCheck{} + mrfh := &mRFService.RedisFailoverHeal{} + mk := &mK8SService.Services{} + mrfc.On("GetRedisesIPs", rf).Once().Return([]string{"10.0.0.1"}, nil) + mrfc.On("GetMasterIP", rf).Once().Return("10.0.0.1", nil) + mrfc.On("GetStatefulSetUpdateRevision", rf).Once().Return("new", nil) + if stale == "slave" { + mrfc.On("GetRedisesSlavesPods", rf).Once().Return([]string{"slave"}, nil) + } else { + mrfc.On("GetRedisesSlavesPods", rf).Once().Return([]string{}, nil) + mrfc.On("GetRedisesMasterPod", rf).Once().Return("master", nil) + } + mrfc.On("GetRedisRevisionHash", stale, rf).Once().Return("old", nil) + mk.On("GetStatefulSetPods", rf.Namespace, rfservice.GetRedisName(rf)).Once().Return(&corev1.PodList{Items: test.pods}, test.podsErr) + if test.wantDelete { + mrfh.On("DeletePod", stale, rf).Once().Return(nil) + } + + handler := rfOperator.NewRedisFailoverHandler(generateConfig(), &mRFService.RedisFailoverClient{}, mrfc, mrfh, mk, metrics.Dummy, log.Dummy) + err := handler.UpdateRedisesPods(rf) + + assert.Equal(t, test.podsErr, err) + mrfc.AssertExpectations(t) + mrfh.AssertExpectations(t) + mk.AssertExpectations(t) + }) + } + } +} + +func TestOperatorManagedModeWaitsForAStoppingMasterBeforeElecting(t *testing.T) { + tests := []struct { + name string + pods []corev1.Pod + podsErr error + wantElect bool + }{ + { + name: "no pod is stopping", + pods: []corev1.Pod{redisPod("1", true, false), redisPod("1", true, false)}, + wantElect: true, + }, + { + name: "the old master is stopping and still ready", + pods: []corev1.Pod{redisPod("1", true, false), masterPod(redisPod("1", true, true))}, + }, + { + name: "a stopping master that is not ready (lost node) doesn't block", + pods: []corev1.Pod{redisPod("1", true, false), masterPod(redisPod("1", false, true))}, + wantElect: true, + }, + { + name: "a stopping replica doesn't block", + pods: []corev1.Pod{redisPod("1", true, false), redisPod("1", true, true)}, + wantElect: true, + }, + { + name: "listing pods fails", + podsErr: errors.New("list err"), + }, + } + + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + rf := operatorManagedRF() + mrfc := &mRFService.RedisFailoverCheck{} + mrfh := &mRFService.RedisFailoverHeal{} + mk := &mK8SService.Services{} + mk.On("UpdateRedisFailoverStatus", mock.Anything, mock.Anything, mock.Anything, mock.Anything).Return() + mrfc.On("IsRedisRunningQuorum", rf).Once().Return(true) + mrfc.On("GetNumberMasters", rf).Once().Return(0, nil) + mk.On("GetStatefulSetPods", rf.Namespace, rfservice.GetRedisName(rf)).Once().Return(&corev1.PodList{Items: test.pods}, test.podsErr) + if test.wantElect { + mrfc.On("GetBestReplicaForPromotion", rf).Once().Return(&rfservice.ReplicaInfo{IP: "10.0.0.2"}, nil) + mrfh.On("PromoteBestReplica", "10.0.0.2", rf).Once().Return(nil) + } + + handler := rfOperator.NewRedisFailoverHandler(generateConfig(), &mRFService.RedisFailoverClient{}, mrfc, mrfh, mk, metrics.Dummy, log.Dummy) + err := handler.CheckAndHeal(rf) + + assert.Equal(t, test.podsErr, err) + mrfc.AssertExpectations(t) + mrfh.AssertExpectations(t) + mk.AssertExpectations(t) + }) + } +} diff --git a/operator/redisfailover/controller.go b/operator/redisfailover/controller.go new file mode 100644 index 000000000..fd119bed3 --- /dev/null +++ b/operator/redisfailover/controller.go @@ -0,0 +1,240 @@ +package redisfailover + +import ( + "context" + "errors" + "fmt" + "sync" + "time" + + "github.com/spotahome/kooper/v2/controller" + "github.com/spotahome/kooper/v2/controller/leaderelection" + corev1 "k8s.io/api/core/v1" + "k8s.io/apimachinery/pkg/api/meta" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/runtime" + "k8s.io/apimachinery/pkg/util/wait" + "k8s.io/apimachinery/pkg/watch" + "k8s.io/client-go/kubernetes" + "k8s.io/client-go/tools/cache" + "k8s.io/client-go/util/workqueue" + + "github.com/saremox/redis-operator/log" +) + +const controllerName = "redisfailover" + +// rfController reconciles RedisFailovers from one queue keyed by RedisFailover. +// Events on a RedisFailover's pods queue that RedisFailover, so a pod change +// (deleted, recreated, ready) drives the next reconcile without waiting for +// the resync. +type rfController struct { + handler controller.Handler + rfInformer cache.SharedIndexInformer + podInformer cache.SharedIndexInformer + queue workqueue.TypedInterface[string] + workers int + leRunner leaderelection.Runner + metrics controller.MetricsRecorder + logger log.Logger + + // queuedAt feeds the in-queue duration metric. + mu sync.Mutex + queuedAt map[string]time.Time +} + +func newRFController(handler controller.Handler, rfRetriever controller.Retriever, podLW cache.ListerWatcher, resync time.Duration, workers int, leRunner leaderelection.Runner, metrics controller.MetricsRecorder, logger log.Logger) (*rfController, error) { + if resync <= 0 { + resync = 3 * time.Minute + } + if workers <= 0 { + workers = 3 + } + c := &rfController{ + handler: handler, + rfInformer: cache.NewSharedIndexInformer(listerWatcher(rfRetriever), nil, resync, cache.Indexers{}), + podInformer: cache.NewSharedIndexInformer(podLW, &corev1.Pod{}, 0, cache.Indexers{}), + queue: workqueue.NewTyped[string](), + workers: workers, + leRunner: leRunner, + metrics: metrics, + logger: logger, + queuedAt: map[string]time.Time{}, + } + + // Only the metrics registration can fail here; the informers are new. + _, rfErr := c.rfInformer.AddEventHandlerWithResyncPeriod(c.eventHandler(rfKey), resync) + _, podErr := c.podInformer.AddEventHandler(c.eventHandler(c.podOwner)) + queueLen := func(context.Context) int { return c.queue.Len() } + if err := errors.Join( + c.podInformer.SetTransform(podMetadataOnly), + rfErr, + podErr, + metrics.RegisterResourceQueueLengthFunc(controllerName, queueLen), + ); err != nil { + return nil, err + } + return c, nil +} + +func listerWatcher(r controller.Retriever) cache.ListerWatcher { + return &cache.ListWatch{ + ListWithContextFunc: r.List, + WatchFuncWithContext: r.Watch, + } +} + +// podMetadataOnly keeps only what podOwnerKey needs in the pod cache. +func podMetadataOnly(obj any) (any, error) { + pod, ok := obj.(*corev1.Pod) + if !ok { + return obj, nil + } + return &corev1.Pod{ObjectMeta: metav1.ObjectMeta{ + Name: pod.Name, + Namespace: pod.Namespace, + UID: pod.UID, + ResourceVersion: pod.ResourceVersion, + Labels: pod.Labels, + }}, nil +} + +func rfKey(obj any) (string, bool) { + key, err := cache.DeletionHandlingMetaNamespaceKeyFunc(obj) + return key, err == nil +} + +func podOwnerKey(obj any) (string, bool) { + if tombstone, ok := obj.(cache.DeletedFinalStateUnknown); ok { + obj = tombstone.Obj + } + m, err := meta.Accessor(obj) + if err != nil { + return "", false + } + name := m.GetLabels()[rfLabelNameKey] + if name == "" { + return "", false + } + return m.GetNamespace() + "/" + name, true +} + +// podOwner returns the key of the pod's RedisFailover, if this operator +// handles that RedisFailover. +func (c *rfController) podOwner(obj any) (string, bool) { + key, ok := podOwnerKey(obj) + if !ok { + return "", false + } + _, exists, _ := c.rfInformer.GetIndexer().GetByKey(key) + return key, exists +} + +func (c *rfController) eventHandler(keyOf func(any) (string, bool)) cache.ResourceEventHandler { + enqueue := func(obj any) { + if key, ok := keyOf(obj); ok { + c.enqueue(key) + } + } + return cache.ResourceEventHandlerFuncs{ + AddFunc: enqueue, + UpdateFunc: func(old, obj any) { + // A pod whose owner label changed concerns both owners. + oldKey, oldOK := keyOf(old) + key, ok := keyOf(obj) + if oldOK && (!ok || oldKey != key) { + c.enqueue(oldKey) + } + if ok { + c.enqueue(key) + } + }, + DeleteFunc: enqueue, + } +} + +func (c *rfController) enqueue(key string) { + c.mu.Lock() + if _, ok := c.queuedAt[key]; !ok { + c.queuedAt[key] = time.Now() + } + c.mu.Unlock() + c.metrics.IncResourceEventQueued(context.Background(), controllerName, false) + c.queue.Add(key) +} + +// Run satisfies controller.Controller. +func (c *rfController) Run(ctx context.Context) error { + if c.leRunner == nil { + return c.run(ctx) + } + return c.leRunner.Run(func() error { return c.run(ctx) }) +} + +func (c *rfController) run(ctx context.Context) error { + defer c.queue.ShutDown() + + c.logger.Infof("starting controller") + go c.rfInformer.RunWithContext(ctx) + go c.podInformer.RunWithContext(ctx) + // Pod events only speed reconciles up, so a pod watch that can't sync + // (e.g. RBAC) must not block reconciling. + if !cache.WaitForNamedCacheSyncWithContext(ctx, c.rfInformer.HasSynced) { + return fmt.Errorf("timed out waiting for caches to sync") + } + + for range c.workers { + go wait.UntilWithContext(ctx, func(ctx context.Context) { + for c.processNext(ctx) { + } + }, time.Second) + } + <-ctx.Done() + c.logger.Infof("stopping controller") + return nil +} + +func (c *rfController) processNext(ctx context.Context) bool { + key, shutdown := c.queue.Get() + if shutdown { + return false + } + defer c.queue.Done(key) + + c.mu.Lock() + if queuedAt, ok := c.queuedAt[key]; ok { + c.metrics.ObserveResourceInQueueDuration(ctx, controllerName, queuedAt) + delete(c.queuedAt, key) + } + c.mu.Unlock() + + start := time.Now() + err := c.process(ctx, key) + c.metrics.ObserveResourceProcessingDuration(ctx, controllerName, err == nil, start) + if err != nil { + c.logger.WithField("object-key", key).Errorf("error on object processing: %v", err) + } + return true +} + +func (c *rfController) process(ctx context.Context, key string) error { + obj, exists, err := c.rfInformer.GetIndexer().GetByKey(key) + if err != nil || !exists { + return err + } + return c.handler.Handle(ctx, obj.(runtime.Object)) +} + +// newPodListWatch lists and watches the pods of every RedisFailover. +func newPodListWatch(k8sClient kubernetes.Interface) *cache.ListWatch { + return &cache.ListWatch{ + ListWithContextFunc: func(ctx context.Context, options metav1.ListOptions) (runtime.Object, error) { + options.LabelSelector = rfLabelNameKey + return k8sClient.CoreV1().Pods("").List(ctx, options) + }, + WatchFuncWithContext: func(ctx context.Context, options metav1.ListOptions) (watch.Interface, error) { + options.LabelSelector = rfLabelNameKey + return k8sClient.CoreV1().Pods("").Watch(ctx, options) + }, + } +} diff --git a/operator/redisfailover/controller_test.go b/operator/redisfailover/controller_test.go new file mode 100644 index 000000000..ec2b4b13e --- /dev/null +++ b/operator/redisfailover/controller_test.go @@ -0,0 +1,427 @@ +package redisfailover + +import ( + "context" + "errors" + "sync" + "sync/atomic" + "testing" + "time" + + "github.com/spotahome/kooper/v2/controller" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + corev1 "k8s.io/api/core/v1" + apierrors "k8s.io/apimachinery/pkg/api/errors" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/runtime" + "k8s.io/apimachinery/pkg/watch" + clientfeatures "k8s.io/client-go/features" + clientfeaturestesting "k8s.io/client-go/features/testing" + fakekubernetes "k8s.io/client-go/kubernetes/fake" + k8stesting "k8s.io/client-go/testing" + "k8s.io/client-go/tools/cache" + + redisfailoverv1 "github.com/saremox/redis-operator/api/redisfailover/v1" + "github.com/saremox/redis-operator/log" + "github.com/saremox/redis-operator/metrics" +) + +type staticRFRetriever struct { + items []redisfailoverv1.RedisFailover +} + +func (r staticRFRetriever) List(context.Context, metav1.ListOptions) (runtime.Object, error) { + return &redisfailoverv1.RedisFailoverList{Items: r.items}, nil +} + +func (r staticRFRetriever) Watch(context.Context, metav1.ListOptions) (watch.Interface, error) { + return watch.NewFake(), nil +} + +type recordingHandler struct { + mu sync.Mutex + calls []string + running atomic.Int32 + overlap atomic.Bool + delay time.Duration +} + +func (h *recordingHandler) Handle(_ context.Context, obj runtime.Object) error { + if h.running.Add(1) > 1 { + h.overlap.Store(true) + } + defer h.running.Add(-1) + time.Sleep(h.delay) + rf := obj.(*redisfailoverv1.RedisFailover) + h.mu.Lock() + h.calls = append(h.calls, rf.Namespace+"/"+rf.Name) + h.mu.Unlock() + return nil +} + +func (h *recordingHandler) count() int { + h.mu.Lock() + defer h.mu.Unlock() + return len(h.calls) +} + +func rfPod(name, rf string) *corev1.Pod { + return &corev1.Pod{ObjectMeta: metav1.ObjectMeta{ + Name: name, + Namespace: "ns", + Labels: map[string]string{rfLabelNameKey: rf}, + }} +} + +func startController(t *testing.T, h controller.Handler, kube *fakekubernetes.Clientset, rfs ...redisfailoverv1.RedisFailover) *rfController { + t.Helper() + return startControllerWith(t, h, kube, time.Hour, metrics.Dummy, rfs...) +} + +// startControllerWith returns once the pod watch is registered, so pods +// created afterwards produce events. +func startControllerWith(t *testing.T, h controller.Handler, kube *fakekubernetes.Clientset, resync time.Duration, mrec metrics.Recorder, rfs ...redisfailoverv1.RedisFailover) *rfController { + t.Helper() + clientfeaturestesting.SetFeatureDuringTest(t, clientfeatures.WatchListClient, false) + watching := make(chan struct{}) + var once sync.Once + kube.PrependWatchReactor("pods", func(action k8stesting.Action) (bool, watch.Interface, error) { + w, err := kube.Tracker().Watch(action.GetResource(), action.GetNamespace()) + once.Do(func() { close(watching) }) + return true, w, err + }) + c, err := newRFController(h, staticRFRetriever{items: rfs}, newPodListWatch(kube), resync, 3, nil, mrec, log.Dummy) + require.NoError(t, err) + ctx, cancel := context.WithCancel(context.Background()) + done := make(chan error) + go func() { done <- c.Run(ctx) }() + <-watching + t.Cleanup(func() { + cancel() + require.NoError(t, <-done) + }) + return c +} + +func TestRFControllerPodEventsReconcileTheirRedisFailover(t *testing.T) { + rf := redisfailoverv1.RedisFailover{ObjectMeta: metav1.ObjectMeta{Name: "rf", Namespace: "ns"}} + kube := fakekubernetes.NewClientset() + h := &recordingHandler{} + startController(t, h, kube, rf) + + require.Eventually(t, func() bool { return h.count() == 1 }, 5*time.Second, 10*time.Millisecond) + + _, err := kube.CoreV1().Pods("ns").Create(context.Background(), rfPod("rfr-rf-0", "rf"), metav1.CreateOptions{}) + require.NoError(t, err) + require.Eventually(t, func() bool { return h.count() == 2 }, 5*time.Second, 10*time.Millisecond) + + ready := rfPod("rfr-rf-0", "rf") + ready.Status.Conditions = []corev1.PodCondition{{Type: corev1.PodReady, Status: corev1.ConditionTrue}} + _, err = kube.CoreV1().Pods("ns").UpdateStatus(context.Background(), ready, metav1.UpdateOptions{}) + require.NoError(t, err) + require.Eventually(t, func() bool { return h.count() == 3 }, 5*time.Second, 10*time.Millisecond) + + _, err = kube.CoreV1().Pods("ns").Create(context.Background(), rfPod("rfr-other-0", "other"), metav1.CreateOptions{}) + require.NoError(t, err) + require.NoError(t, kube.CoreV1().Pods("ns").Delete(context.Background(), "rfr-rf-0", metav1.DeleteOptions{})) + require.Eventually(t, func() bool { return h.count() == 4 }, 5*time.Second, 10*time.Millisecond) + time.Sleep(200 * time.Millisecond) + + h.mu.Lock() + defer h.mu.Unlock() + assert.Equal(t, []string{"ns/rf", "ns/rf", "ns/rf", "ns/rf"}, h.calls) +} + +func TestRFControllerNeverReconcilesARedisFailoverConcurrently(t *testing.T) { + rf := redisfailoverv1.RedisFailover{ObjectMeta: metav1.ObjectMeta{Name: "rf", Namespace: "ns"}} + kube := fakekubernetes.NewClientset() + h := &recordingHandler{delay: 50 * time.Millisecond} + startController(t, h, kube, rf) + require.Eventually(t, func() bool { return h.count() == 1 }, 5*time.Second, 10*time.Millisecond) + + for _, name := range []string{"rfr-rf-0", "rfr-rf-1", "rfr-rf-2", "rfs-rf-a", "rfs-rf-b"} { + _, err := kube.CoreV1().Pods("ns").Create(context.Background(), rfPod(name, "rf"), metav1.CreateOptions{}) + require.NoError(t, err) + } + require.Eventually(t, func() bool { return h.count() >= 2 && h.running.Load() == 0 }, 5*time.Second, 10*time.Millisecond) + time.Sleep(200 * time.Millisecond) + + assert.False(t, h.overlap.Load()) + assert.Less(t, h.count(), 7, "queued events for the same RedisFailover coalesce") +} + +func TestPodOwnerKey(t *testing.T) { + key, ok := podOwnerKey(rfPod("rfr-rf-0", "rf")) + assert.True(t, ok) + assert.Equal(t, "ns/rf", key) + + key, ok = podOwnerKey(cache.DeletedFinalStateUnknown{Key: "ns/rfr-rf-0", Obj: rfPod("rfr-rf-0", "rf")}) + assert.True(t, ok) + assert.Equal(t, "ns/rf", key) + + _, ok = podOwnerKey(&corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: "unlabelled", Namespace: "ns"}}) + assert.False(t, ok) + + _, ok = podOwnerKey("not an object") + assert.False(t, ok) +} + +func TestPodMetadataOnly(t *testing.T) { + pod := rfPod("rfr-rf-0", "rf") + pod.Spec.Containers = []corev1.Container{{Name: "redis"}} + pod.Status.PodIP = "10.0.0.1" + + obj, err := podMetadataOnly(pod) + require.NoError(t, err) + assert.Equal(t, &corev1.Pod{ObjectMeta: pod.ObjectMeta}, obj) + + obj, err = podMetadataOnly("passthrough") + require.NoError(t, err) + assert.Equal(t, "passthrough", obj) +} + +type failingQueueMetrics struct{ metrics.Recorder } + +func (failingQueueMetrics) RegisterResourceQueueLengthFunc(string, func(context.Context) int) error { + return errors.New("already registered") +} + +func TestNewRFControllerReturnsMetricsRegistrationError(t *testing.T) { + _, err := newRFController(&recordingHandler{}, staticRFRetriever{}, newPodListWatch(fakekubernetes.NewClientset()), 0, 0, nil, failingQueueMetrics{metrics.Dummy}, log.Dummy) + assert.EqualError(t, err, "already registered") +} + +type queueLengthMetrics struct { + metrics.Recorder + queueLen func(context.Context) int +} + +func (m *queueLengthMetrics) RegisterResourceQueueLengthFunc(_ string, f func(context.Context) int) error { + m.queueLen = f + return nil +} + +type recordingRunner struct{ ran bool } + +func (r *recordingRunner) Run(f func() error) error { + r.ran = true + return f() +} + +func TestRFControllerRunsUnderLeaderElection(t *testing.T) { + runner := &recordingRunner{} + mrec := &queueLengthMetrics{Recorder: metrics.Dummy} + c, err := newRFController(&recordingHandler{}, staticRFRetriever{}, newPodListWatch(fakekubernetes.NewClientset()), time.Hour, 1, runner, mrec, log.Dummy) + require.NoError(t, err) + c.enqueue("ns/rf") + assert.Equal(t, 1, mrec.queueLen(context.Background())) + ctx, cancel := context.WithCancel(context.Background()) + cancel() + + assert.EqualError(t, c.Run(ctx), "timed out waiting for caches to sync") + assert.True(t, runner.ran) +} + +type failingHandler struct{ calls atomic.Int32 } + +func (h *failingHandler) Handle(context.Context, runtime.Object) error { + h.calls.Add(1) + return errors.New("boom") +} + +func TestRFControllerKeepsWorkingAfterHandleErrors(t *testing.T) { + rf := redisfailoverv1.RedisFailover{ObjectMeta: metav1.ObjectMeta{Name: "rf", Namespace: "ns"}} + kube := fakekubernetes.NewClientset() + h := &failingHandler{} + startController(t, h, kube, rf) + require.Eventually(t, func() bool { return h.calls.Load() == 1 }, 5*time.Second, 10*time.Millisecond) + + _, err := kube.CoreV1().Pods("ns").Create(context.Background(), rfPod("rfr-rf-0", "rf"), metav1.CreateOptions{}) + require.NoError(t, err) + require.Eventually(t, func() bool { return h.calls.Load() == 2 }, 5*time.Second, 10*time.Millisecond) +} + +func (h *recordingHandler) countFor(key string) int { + h.mu.Lock() + defer h.mu.Unlock() + n := 0 + for _, c := range h.calls { + if c == key { + n++ + } + } + return n +} + +func TestRFControllerReconcilesBothOwnersWhenAPodChangesOwner(t *testing.T) { + a := redisfailoverv1.RedisFailover{ObjectMeta: metav1.ObjectMeta{Name: "a", Namespace: "ns"}} + b := redisfailoverv1.RedisFailover{ObjectMeta: metav1.ObjectMeta{Name: "b", Namespace: "ns"}} + kube := fakekubernetes.NewClientset() + h := &recordingHandler{} + startController(t, h, kube, a, b) + require.Eventually(t, func() bool { return h.count() == 2 }, 5*time.Second, 10*time.Millisecond) + + pod, err := kube.CoreV1().Pods("ns").Create(context.Background(), rfPod("rfr-a-0", "a"), metav1.CreateOptions{}) + require.NoError(t, err) + require.Eventually(t, func() bool { return h.countFor("ns/a") == 2 }, 5*time.Second, 10*time.Millisecond) + + pod.Labels[rfLabelNameKey] = "b" + _, err = kube.CoreV1().Pods("ns").Update(context.Background(), pod, metav1.UpdateOptions{}) + require.NoError(t, err) + require.Eventually(t, func() bool { return h.countFor("ns/a") == 3 && h.countFor("ns/b") == 2 }, 5*time.Second, 10*time.Millisecond) +} + +func TestRFControllerReconcilesWithoutPodAccess(t *testing.T) { + clientfeaturestesting.SetFeatureDuringTest(t, clientfeatures.WatchListClient, false) + rf := redisfailoverv1.RedisFailover{ObjectMeta: metav1.ObjectMeta{Name: "rf", Namespace: "ns"}} + kube := fakekubernetes.NewClientset() + kube.PrependReactor("list", "pods", func(k8stesting.Action) (bool, runtime.Object, error) { + return true, nil, apierrors.NewForbidden(corev1.Resource("pods"), "", errors.New("denied")) + }) + h := &recordingHandler{} + c, err := newRFController(h, staticRFRetriever{items: []redisfailoverv1.RedisFailover{rf}}, newPodListWatch(kube), time.Hour, 1, nil, metrics.Dummy, log.Dummy) + require.NoError(t, err) + ctx, cancel := context.WithCancel(context.Background()) + done := make(chan error) + go func() { done <- c.Run(ctx) }() + + require.Eventually(t, func() bool { return h.count() == 1 }, 5*time.Second, 10*time.Millisecond) + cancel() + require.NoError(t, <-done) +} + +func TestRFControllerPodOwnerIgnoresUnhandledRedisFailovers(t *testing.T) { + c, err := newRFController(&recordingHandler{}, staticRFRetriever{}, newPodListWatch(fakekubernetes.NewClientset()), time.Hour, 1, nil, metrics.Dummy, log.Dummy) + require.NoError(t, err) + require.NoError(t, c.rfInformer.GetIndexer().Add(&redisfailoverv1.RedisFailover{ObjectMeta: metav1.ObjectMeta{Name: "rf", Namespace: "ns"}})) + + key, ok := c.podOwner(rfPod("rfr-rf-0", "rf")) + assert.True(t, ok) + assert.Equal(t, "ns/rf", key) + + _, ok = c.podOwner(rfPod("rfr-other-0", "other")) + assert.False(t, ok) + + _, ok = c.podOwner(&corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: "unlabelled", Namespace: "ns"}}) + assert.False(t, ok) +} + +func TestRFControllerSkipsDeletedRedisFailovers(t *testing.T) { + h := &recordingHandler{} + c, err := newRFController(h, staticRFRetriever{}, newPodListWatch(fakekubernetes.NewClientset()), time.Hour, 1, nil, metrics.Dummy, log.Dummy) + require.NoError(t, err) + + assert.NoError(t, c.process(context.Background(), "ns/gone")) + assert.Zero(t, h.count()) +} + +func TestRFControllerResyncsRedisFailovers(t *testing.T) { + rf := redisfailoverv1.RedisFailover{ObjectMeta: metav1.ObjectMeta{Name: "rf", Namespace: "ns"}} + h := &recordingHandler{} + startControllerWith(t, h, fakekubernetes.NewClientset(), 50*time.Millisecond, metrics.Dummy, rf) + + require.Eventually(t, func() bool { return h.count() >= 2 }, 5*time.Second, 10*time.Millisecond) +} + +func TestNewRFControllerDefaultsWorkers(t *testing.T) { + c, err := newRFController(&recordingHandler{}, staticRFRetriever{}, newPodListWatch(fakekubernetes.NewClientset()), 0, 0, nil, metrics.Dummy, log.Dummy) + require.NoError(t, err) + assert.Equal(t, 3, c.workers) +} + +func TestPodListWatchSelectsRedisFailoverPods(t *testing.T) { + kube := fakekubernetes.NewClientset() + lw := newPodListWatch(kube) + _, err := lw.ListWithContext(context.Background(), metav1.ListOptions{}) + require.NoError(t, err) + w, err := lw.WatchWithContext(context.Background(), metav1.ListOptions{}) + require.NoError(t, err) + w.Stop() + + actions := kube.Actions() + require.Len(t, actions, 2) + assert.Equal(t, rfLabelNameKey, actions[0].(k8stesting.ListActionImpl).GetListRestrictions().Labels.String()) + assert.Equal(t, rfLabelNameKey, actions[1].(k8stesting.WatchActionImpl).GetWatchRestrictions().Labels.String()) +} + +func TestRFControllerCachesPodMetadataOnly(t *testing.T) { + rf := redisfailoverv1.RedisFailover{ObjectMeta: metav1.ObjectMeta{Name: "rf", Namespace: "ns"}} + kube := fakekubernetes.NewClientset() + h := &recordingHandler{} + c := startController(t, h, kube, rf) + require.Eventually(t, func() bool { return h.count() == 1 }, 5*time.Second, 10*time.Millisecond) + + pod := rfPod("rfr-rf-0", "rf") + pod.Spec.Containers = []corev1.Container{{Name: "redis"}} + _, err := kube.CoreV1().Pods("ns").Create(context.Background(), pod, metav1.CreateOptions{}) + require.NoError(t, err) + require.Eventually(t, func() bool { return len(c.podInformer.GetStore().List()) == 1 }, 5*time.Second, 10*time.Millisecond) + + assert.Empty(t, c.podInformer.GetStore().List()[0].(*corev1.Pod).Spec.Containers) +} + +type recordingMetrics struct { + metrics.Recorder + mu sync.Mutex + queued int + inQueue int + processed []bool +} + +func (m *recordingMetrics) IncResourceEventQueued(context.Context, string, bool) { + m.mu.Lock() + defer m.mu.Unlock() + m.queued++ +} + +func (m *recordingMetrics) ObserveResourceInQueueDuration(context.Context, string, time.Time) { + m.mu.Lock() + defer m.mu.Unlock() + m.inQueue++ +} + +func (m *recordingMetrics) ObserveResourceProcessingDuration(_ context.Context, _ string, success bool, _ time.Time) { + m.mu.Lock() + defer m.mu.Unlock() + m.processed = append(m.processed, success) +} + +func (m *recordingMetrics) snapshot() (int, int, []bool) { + m.mu.Lock() + defer m.mu.Unlock() + return m.queued, m.inQueue, append([]bool(nil), m.processed...) +} + +func TestRFControllerRecordsMetrics(t *testing.T) { + rf := redisfailoverv1.RedisFailover{ObjectMeta: metav1.ObjectMeta{Name: "rf", Namespace: "ns"}} + kube := fakekubernetes.NewClientset() + h := &failingHandler{} + mrec := &recordingMetrics{Recorder: metrics.Dummy} + startControllerWith(t, h, kube, time.Hour, mrec, rf) + require.Eventually(t, func() bool { _, _, p := mrec.snapshot(); return len(p) == 1 }, 5*time.Second, 10*time.Millisecond) + + queued, inQueue, processed := mrec.snapshot() + assert.Equal(t, 1, queued) + assert.Equal(t, 1, inQueue) + assert.Equal(t, []bool{false}, processed) +} + +func TestRFControllerCountsOneEventPerUpdate(t *testing.T) { + mrec := &recordingMetrics{Recorder: metrics.Dummy} + c, err := newRFController(&recordingHandler{}, staticRFRetriever{}, newPodListWatch(fakekubernetes.NewClientset()), time.Hour, 1, nil, mrec, log.Dummy) + require.NoError(t, err) + a := &redisfailoverv1.RedisFailover{ObjectMeta: metav1.ObjectMeta{Name: "a", Namespace: "ns"}} + b := &redisfailoverv1.RedisFailover{ObjectMeta: metav1.ObjectMeta{Name: "b", Namespace: "ns"}} + require.NoError(t, c.rfInformer.GetIndexer().Add(a)) + require.NoError(t, c.rfInformer.GetIndexer().Add(b)) + + c.eventHandler(rfKey).OnUpdate(a, a) + queued, _, _ := mrec.snapshot() + assert.Equal(t, 1, queued) + + c.eventHandler(c.podOwner).OnUpdate(rfPod("rfr-a-0", "a"), rfPod("rfr-a-0", "b")) + queued, _, _ = mrec.snapshot() + assert.Equal(t, 3, queued) + assert.Equal(t, 2, c.queue.Len()) +} diff --git a/operator/redisfailover/factory.go b/operator/redisfailover/factory.go index f14ae2317..8fa053b8f 100644 --- a/operator/redisfailover/factory.go +++ b/operator/redisfailover/factory.go @@ -54,17 +54,11 @@ func New(cfg Config, k8sService k8s.Services, k8sClient kubernetes.Interface, lo return nil, err } - // Create our controller. - return controller.New(&controller.Config{ - Handler: rfHandler, - Retriever: rfRetriever, - LeaderElector: leSVC, - MetricsRecorder: kooperMetricsRecorder, - Logger: kooperLogger, - Name: "redisfailover", - ResyncInterval: time.Duration(cfg.SyncInterval) * time.Second, - ConcurrentWorkers: cfg.Concurrency, - }) + c, err := newRFController(rfHandler, rfRetriever, newPodListWatch(k8sClient), time.Duration(cfg.SyncInterval)*time.Second, cfg.Concurrency, leSVC, kooperMetricsRecorder, logger.WithField("operator", "redisfailover")) + if err != nil { + return nil, err + } + return c, nil } func NewRedisFailoverRetriever(cfg Config, cli k8s.Services) controller.Retriever { diff --git a/operator/redisfailover/factory_test.go b/operator/redisfailover/factory_test.go index 577b19d72..8d8b6f1dc 100644 --- a/operator/redisfailover/factory_test.go +++ b/operator/redisfailover/factory_test.go @@ -285,3 +285,10 @@ func TestNewPropagatesLeaderElectionError(t *testing.T) { mk.AssertExpectations(t) mr.AssertExpectations(t) } + +func TestNewPropagatesControllerError(t *testing.T) { + ctrl, err := New(Config{SupportedNamespacesRegex: ".*"}, &mK8SService.Services{}, fakekubernetes.NewClientset(), "test-namespace", &mRedisService.Client{}, failingQueueMetrics{metrics.Dummy}, log.Dummy) + + assert.Error(t, err) + assert.Nil(t, ctrl) +} diff --git a/operator/redisfailover/handler.go b/operator/redisfailover/handler.go index fa40d7b77..f8d65f7e5 100644 --- a/operator/redisfailover/handler.go +++ b/operator/redisfailover/handler.go @@ -22,12 +22,11 @@ const ( rfLabelNameKey = "redisfailovers.databases.spotahome.com/name" skipReconcileAnnotation = "redisfailovers.databases.spotahome.com/skip-reconcile" // redisFailoverFinalizer is what makes RedisFailover deletion visible to - // Handle at all. Without a finalizer, kooper's generic controller never - // calls Handle for a delete: by the time its DeleteFunc fires, the - // object is already gone from the informer's local indexer, and the - // processor that resolves a queued key back to an object just no-ops - // when the key no longer resolves (see newIndexerProcessor in - // kooper/v2/controller/processor.go) - Handle is never invoked with a + // Handle at all. Without a finalizer, the controller never calls Handle + // for a delete: by the time its DeleteFunc fires, the object is already + // gone from the informer's local indexer, and resolving a queued key + // back to an object just no-ops when the key no longer resolves (see + // rfController.process) - Handle is never invoked with a // nil/absent object standing in for "this was deleted". A finalizer // makes the API server hold the object (with DeletionTimestamp set) // until we remove it, which turns "delete" into an ordinary object we diff --git a/operator/redisfailover/service/check.go b/operator/redisfailover/service/check.go index 99e395751..f1b8d7af5 100644 --- a/operator/redisfailover/service/check.go +++ b/operator/redisfailover/service/check.go @@ -109,6 +109,11 @@ func (r *RedisFailoverChecker) CheckSentinelNumber(rf *redisfailoverv1.RedisFail return nil } +// IsMasterPod reports whether the pod is labelled as the redis master. +func IsMasterPod(pod *corev1.Pod) bool { + return pod.Labels[redisRoleLabelKey] == redisRoleLabelMaster +} + func (r *RedisFailoverChecker) setMasterLabelIfNecessary(namespace string, pod corev1.Pod) error { for labelKey, labelValue := range pod.Labels { if labelKey == redisRoleLabelKey && labelValue == redisRoleLabelMaster { @@ -154,6 +159,11 @@ func (r *RedisFailoverChecker) CheckAllSlavesFromMaster(master string, rf *redis } } + if rp.DeletionTimestamp != nil { + // A terminating pod is going away, and dialing its IP can block + // for the whole connection timeout. + continue + } slave, err := r.redisClient.GetSlaveOf(rp.Status.PodIP, rport, password) if err != nil { // The pod is unreachable - typically the old master on a downed node. diff --git a/operator/redisfailover/service/check_test.go b/operator/redisfailover/service/check_test.go index 270b00cfe..0753fb6e0 100644 --- a/operator/redisfailover/service/check_test.go +++ b/operator/redisfailover/service/check_test.go @@ -235,6 +235,40 @@ func TestCheckAllSlavesFromMasterLabelsMasterDespiteUnreachablePod(t *testing.T) ms.AssertExpectations(t) // proves the master-role label was applied } +func TestCheckAllSlavesFromMasterDoesNotDialTerminatingPods(t *testing.T) { + rf := generateRF() + pods := &corev1.PodList{ + Items: []corev1.Pod{ + { + ObjectMeta: metav1.ObjectMeta{Name: "rfr-test-0", Labels: map[string]string{"redisfailovers-role": "master"}}, + Status: corev1.PodStatus{PodIP: "10.0.0.1", Phase: corev1.PodRunning}, + }, + { + ObjectMeta: metav1.ObjectMeta{Name: "rfr-test-1", DeletionTimestamp: &metav1.Time{Time: time.Now()}}, + Status: corev1.PodStatus{PodIP: "10.0.0.2", Phase: corev1.PodRunning}, + }, + }, + } + + ms := &mK8SService.Services{} + ms.On("GetStatefulSetPods", namespace, rfservice.GetRedisName(rf)).Once().Return(pods, nil) + ms.On("UpdatePodLabels", namespace, "rfr-test-1", map[string]string{"redisfailovers-role": "slave"}).Once().Return(nil) + mr := &mRedisService.Client{} + mr.On("GetSlaveOf", "10.0.0.1", "0", "").Once().Return("", nil) + + checker := rfservice.NewRedisFailoverChecker(ms, mr, log.DummyLogger{}, metrics.Dummy) + + assert.NoError(t, checker.CheckAllSlavesFromMaster("10.0.0.1", rf)) + ms.AssertExpectations(t) + mr.AssertExpectations(t) +} + +func TestIsMasterPod(t *testing.T) { + assert.True(t, rfservice.IsMasterPod(&corev1.Pod{ObjectMeta: metav1.ObjectMeta{Labels: map[string]string{"redisfailovers-role": "master"}}})) + assert.False(t, rfservice.IsMasterPod(&corev1.Pod{ObjectMeta: metav1.ObjectMeta{Labels: map[string]string{"redisfailovers-role": "slave"}}})) + assert.False(t, rfservice.IsMasterPod(&corev1.Pod{})) +} + func TestCheckAllSlavesFromMasterDifferentMaster(t *testing.T) { assert := assert.New(t) diff --git a/operator/redisfailover/util/pod.go b/operator/redisfailover/util/pod.go index d0e4b0137..d11fb1e13 100644 --- a/operator/redisfailover/util/pod.go +++ b/operator/redisfailover/util/pod.go @@ -9,3 +9,12 @@ func PodIsTerminal(pod *v1.Pod) bool { func PodIsScheduling(pod *v1.Pod) bool { return pod.DeletionTimestamp != nil || pod.Status.Phase == v1.PodPending } + +func PodIsReady(pod *v1.Pod) bool { + for _, c := range pod.Status.Conditions { + if c.Type == v1.PodReady { + return c.Status == v1.ConditionTrue + } + } + return false +} diff --git a/operator/redisfailover/util/pod_test.go b/operator/redisfailover/util/pod_test.go index 7b8a82579..8de45ea10 100644 --- a/operator/redisfailover/util/pod_test.go +++ b/operator/redisfailover/util/pod_test.go @@ -220,3 +220,16 @@ func TestPodIsTerminal(t *testing.T) { }) } } + +func TestPodIsReady(t *testing.T) { + withReady := func(status corev1.ConditionStatus) *corev1.Pod { + return &corev1.Pod{Status: corev1.PodStatus{Conditions: []corev1.PodCondition{ + {Type: corev1.PodScheduled, Status: corev1.ConditionTrue}, + {Type: corev1.PodReady, Status: status}, + }}} + } + assert.True(t, util.PodIsReady(withReady(corev1.ConditionTrue))) + assert.False(t, util.PodIsReady(withReady(corev1.ConditionFalse))) + assert.False(t, util.PodIsReady(withReady(corev1.ConditionUnknown))) + assert.False(t, util.PodIsReady(&corev1.Pod{})) +} diff --git a/test/integration/redisfailover/creation_test.go b/test/integration/redisfailover/creation_test.go index 2c939ab0b..bd8876566 100644 --- a/test/integration/redisfailover/creation_test.go +++ b/test/integration/redisfailover/creation_test.go @@ -179,17 +179,9 @@ func TestRedisFailover(t *testing.T) { // Wait for the namespace to be ready, rather than guessing how long that takes. require.NoError(waitForNamespaceActive(k8sClient, namespace, 15*time.Second)) - // Create operator and run. SyncInterval is set explicitly (production - // always sets one via -sync-interval, default 30) rather than left at - // the zero value: a zero Config.SyncInterval means kooper's own - // ResyncInterval falls back to *its* default of 3 minutes, and the - // controller's status patch after a no-op reconcile can be byte-identical - // to what's already stored - which the apiserver treats as a no-op write - // that never reaches watchers, so nothing re-triggers Handle() until the - // next full resync. A multi-step change (like the rollout below, which - // needs two separate Handle() calls to replace two pods) can then stall - // for the full resync interval. A short one here keeps that stall short - // instead of letting it hit the 3-minute fallback. + // Create operator and run. A short resync: waiting for the sentinels to + // see the new slaves before replacing the master isn't driven by any + // Kubernetes event. redisfailoverOperator, err := redisfailover.New(redisfailover.Config{SyncInterval: 2, SupportedNamespacesRegex: "^" + namespace + "$"}, k8sservice, k8sClient, namespace, redisClient, metrics.Dummy, log.Dummy) require.NoError(err) diff --git a/test/integration/redisfailover/operator_managed_rollout_test.go b/test/integration/redisfailover/operator_managed_rollout_test.go index 44cacf5b8..99b69cc0a 100644 --- a/test/integration/redisfailover/operator_managed_rollout_test.go +++ b/test/integration/redisfailover/operator_managed_rollout_test.go @@ -216,18 +216,9 @@ func TestRedisFailoverOperatorManagedModeRollout(t *testing.T) { require.NoError(waitForNamespaceActive(k8sClient, ommNamespace, 15*time.Second)) k8sservice := k8s.New(k8sClient, customClient, log.Dummy, metrics.Dummy) - // SyncInterval is set explicitly (production always sets one via - // -sync-interval, default 30) rather than left at the zero value: a zero - // Config.SyncInterval means kooper's own ResyncInterval falls back to - // *its* default of 3 minutes, and the controller's status patch after a - // no-op reconcile can be byte-identical to what's already stored - which - // the apiserver treats as a no-op write that never reaches watchers, so - // nothing re-triggers Handle() until the next full resync. The rollout - // below needs two separate Handle() calls to replace both pods (the - // slave, then the master), so a missed self-trigger between them can - // stall for the full resync interval - a short one here keeps that - // stall short instead of letting it hit the 3-minute fallback. - redisfailoverOperator, err := redisfailover.New(redisfailover.Config{SyncInterval: 2, SupportedNamespacesRegex: "^" + ommNamespace + "$"}, k8sservice, k8sClient, ommNamespace, redisClient, metrics.Dummy, log.Dummy) + // The resync is far longer than the rollout's timeout, so only pod events + // can drive the rollout's steps. + redisfailoverOperator, err := redisfailover.New(redisfailover.Config{SyncInterval: 600, SupportedNamespacesRegex: "^" + ommNamespace + "$"}, k8sservice, k8sClient, ommNamespace, redisClient, metrics.Dummy, log.Dummy) require.NoError(err) go func() { From b519090c2daeb2f737734922f9da2f6c6f70a417 Mon Sep 17 00:00:00 2001 From: Michael Strassberger Date: Sat, 26 Sep 2026 00:36:05 +0200 Subject: [PATCH 12/24] Drop the kooper dependency (#185) Port what kooper still provided onto client-go and prometheus: - the controller metrics, with the same kooper_controller_* names, labels, help texts and buckets; - leader election, with the same lease, timings and identity, and no Events written. Leader election and shutdown: - On SIGTERM the controller stops taking queued reconciles and main waits up to 5s, kooper's grace period, for running ones. The lease is then released so another replica takes over at once; if they don't finish in time, the process exits without releasing it, as before. Stopping before the cache has synced is a clean stop. - When the lease is lost, Run returns errLeadershipLost at once, without waiting for a running reconcile, and the process exits. Kooper exited 5s after the loss. - A lost lease is never released, a lease another replica holds is left alone, and a lease acquired while shutting down is released. Claude-Session: https://claude.ai/code/session_01G4mmyo3sqLwf23m5FE2nBE Co-authored-by: Claude --- .github/copilot-instructions.md | 2 +- cmd/redisoperator/main.go | 36 ++- cmd/utils/flags.go | 3 +- go.mod | 1 - go.sum | 2 - metrics/controller.go | 93 ++++++ metrics/controller_test.go | 74 +++++ metrics/dummy.go | 8 +- metrics/metrics.go | 10 +- operator/redisfailover/controller.go | 65 +++-- operator/redisfailover/controller_test.go | 101 +++++-- operator/redisfailover/factory.go | 32 +-- operator/redisfailover/factory_test.go | 53 +--- operator/redisfailover/leaderelection.go | 161 +++++++++++ operator/redisfailover/leaderelection_test.go | 266 ++++++++++++++++++ 15 files changed, 749 insertions(+), 158 deletions(-) create mode 100644 metrics/controller.go create mode 100644 metrics/controller_test.go create mode 100644 operator/redisfailover/leaderelection.go create mode 100644 operator/redisfailover/leaderelection_test.go diff --git a/.github/copilot-instructions.md b/.github/copilot-instructions.md index 4cdbaa41d..469eab5b4 100644 --- a/.github/copilot-instructions.md +++ b/.github/copilot-instructions.md @@ -63,7 +63,7 @@ make helm-test ## Kubernetes Operator Patterns -- The operator uses the `kooper` framework (`github.com/spotahome/kooper/v2`) for controller/reconciler wiring +- The controller is built directly on client-go informers and a workqueue (`operator/redisfailover/controller.go`) - The reconciliation loop is in `operator/redisfailover/` - All Kubernetes resources created by the operator carry owner references pointing to the `RedisFailover` CR - Redis Statefulsets use the prefix `rfr-`; Sentinel Deployments use `rfs-` diff --git a/cmd/redisoperator/main.go b/cmd/redisoperator/main.go index 107bc7e0b..3b9d8d0e4 100644 --- a/cmd/redisoperator/main.go +++ b/cmd/redisoperator/main.go @@ -24,7 +24,9 @@ import ( ) const ( - gracePeriod = 5 * time.Second + // shutdownTimeout bounds how long SIGTERM waits for running reconciles + // and the lease release. + shutdownTimeout = 5 * time.Second metricsNamespace = "redis_operator" ) @@ -32,7 +34,6 @@ const ( type Main struct { flags *utils.CMDFlags logger log.Logger - stopC chan struct{} } // New returns a Main object. @@ -49,9 +50,7 @@ func New(logger log.Logger) Main { // Run execs the program. func (m *Main) Run() error { - // Create signal channels. - m.stopC = make(chan struct{}) - errC := make(chan error) + errC := make(chan error, 1) // Set correct logging. err := m.logger.Set(log.Level(strings.ToLower(m.flags.LogLevel))) @@ -93,23 +92,30 @@ func (m *Main) Run() error { return err } + ctx, cancel := context.WithCancel(context.Background()) + defer cancel() go func() { - errC <- redisfailoverOperator.Run(context.Background()) + errC <- redisfailoverOperator.Run(ctx) }() // Await signals. sigC := m.createSignalCapturer() - var finalErr error select { case <-sigC: m.logger.Infof("Signal captured, exiting...") + // Let the operator finish its reconciles and release the lease. + cancel() + select { + case err := <-errC: + return err + case <-time.After(shutdownTimeout): + m.logger.Warningf("Operator did not stop within %s, exiting without releasing the leader lease", shutdownTimeout) + return nil + } case err := <-errC: m.logger.Errorf("Error received: %s, exiting...", err) - finalErr = err + return err } - - m.stop(m.stopC) - return finalErr } func (m *Main) createSignalCapturer() <-chan os.Signal { @@ -118,14 +124,6 @@ func (m *Main) createSignalCapturer() <-chan os.Signal { return sigC } -func (m *Main) stop(stopC chan struct{}) { - m.logger.Infof("Stopping everything, waiting %s...", gracePeriod) - - // stop everything and let them time to stop - close(stopC) - time.Sleep(gracePeriod) -} - func getNamespace() string { // This way assumes you've set the POD_NAMESPACE environment // variable using the downward API. This check has to be done first diff --git a/cmd/utils/flags.go b/cmd/utils/flags.go index 029d3c1f8..48c58f1f6 100644 --- a/cmd/utils/flags.go +++ b/cmd/utils/flags.go @@ -37,8 +37,7 @@ func (c *CMDFlags) Init() { flag.StringVar(&c.MetricsPath, "metrics-path", "/metrics", "Path to serve the metrics.") flag.IntVar(&c.K8sQueriesPerSecond, "k8s-cli-qps-limit", 100, "Number of allowed queries per second by kubernetes client without client side throttling") flag.IntVar(&c.K8sQueriesBurstable, "k8s-cli-burstable-limit", 100, "Number of allowed burst requests by kubernetes client without client side throttling") - // default is 3 for conccurency because kooper also defines 3 as default - // reference: https://github.com/spotahome/kooper/blob/master/controller/controller.go#L89 + // 3 is also the controller's fallback for a concurrency of 0 or less. flag.IntVar(&c.Concurrency, "concurrency", 3, "Number of conccurent workers meant to process events") flag.IntVar(&c.SyncInterval, "sync-interval", 30, "Number of seconds between checks") flag.StringVar(&c.LogLevel, "log-level", "info", "set log level") diff --git a/go.mod b/go.mod index d0002a1bf..e16d61333 100644 --- a/go.mod +++ b/go.mod @@ -6,7 +6,6 @@ require ( github.com/go-redis/redis/v8 v8.11.5 github.com/prometheus/client_golang v1.24.1 github.com/sirupsen/logrus v1.10.2 - github.com/spotahome/kooper/v2 v2.10.0 github.com/stretchr/testify v1.12.1 k8s.io/api v0.37.0 k8s.io/apiextensions-apiserver v0.37.0 diff --git a/go.sum b/go.sum index a9677a2bb..02810d0c0 100644 --- a/go.sum +++ b/go.sum @@ -94,8 +94,6 @@ github.com/sirupsen/logrus v1.10.2 h1:G2SED73/qrAu6YwbdxOD6peLkCBI3z7L+ykJFTXJBB github.com/sirupsen/logrus v1.10.2/go.mod h1:SLEg8TqYulVKKfIGHldVp2K2aYz2DKSVBq4g/H5bR7Q= github.com/spf13/pflag v1.0.10 h1:4EBh2KAYBwaONj6b2Ye1GiHfwjqyROoF4RwYO+vPwFk= github.com/spf13/pflag v1.0.10/go.mod h1:McXfInJRrz4CZXVZOBLb0bTZqETkiAhM9Iw0y3An2Bg= -github.com/spotahome/kooper/v2 v2.10.0 h1:A8Mt2Ul3gn+X0Rt3bfi+wW6kMr3DImOCIahAjJKOb2A= -github.com/spotahome/kooper/v2 v2.10.0/go.mod h1:oEhovyoOMmB13TTPdtdFqVCPcdaxGZUqPo4GdX7Mc44= github.com/stretchr/objx v0.1.0/go.mod h1:HFkY916IF+rwdDfMAkV7OtwuqBVzrE8GR6GFx+wExME= github.com/stretchr/objx v0.5.3 h1:jmXUvGomnU1o3W/V5h2VEradbpJDwGrzugQQvL0POH4= github.com/stretchr/objx v0.5.3/go.mod h1:rDQraq+vQZU7Fde9LOZLr8Tax6zZvy4kuNKF+QYS+U0= diff --git a/metrics/controller.go b/metrics/controller.go new file mode 100644 index 000000000..6b77b8580 --- /dev/null +++ b/metrics/controller.go @@ -0,0 +1,93 @@ +package metrics + +import ( + "context" + "fmt" + "strconv" + "time" + + "github.com/prometheus/client_golang/prometheus" +) + +// The controller metrics keep the names they had when the operator was built +// on kooper, so existing dashboards and alerts keep working. +const controllerMetricsNamespace = "kooper" + +// ControllerRecorder records the metrics of the operator's reconcile queue. +type ControllerRecorder interface { + IncResourceEventQueued(ctx context.Context, controller string, isRequeue bool) + ObserveResourceInQueueDuration(ctx context.Context, controller string, queuedAt time.Time) + ObserveResourceProcessingDuration(ctx context.Context, controller string, success bool, startProcessingAt time.Time) + RegisterResourceQueueLengthFunc(controller string, f func(context.Context) int) error +} + +type controllerRecorder struct { + reg prometheus.Registerer + queuedEventsTotal *prometheus.CounterVec + inQueueEventDuration *prometheus.HistogramVec + processedEventDuration *prometheus.HistogramVec +} + +func newControllerRecorder(reg prometheus.Registerer) *controllerRecorder { + r := &controllerRecorder{ + reg: reg, + queuedEventsTotal: prometheus.NewCounterVec(prometheus.CounterOpts{ + Namespace: controllerMetricsNamespace, + Subsystem: promControllerSubsystem, + Name: "queued_events_total", + Help: "Total number of events queued.", + }, []string{"controller", "requeue"}), + inQueueEventDuration: prometheus.NewHistogramVec(prometheus.HistogramOpts{ + Namespace: controllerMetricsNamespace, + Subsystem: promControllerSubsystem, + Name: "event_in_queue_duration_seconds", + Help: "The duration of an event in the queue.", + Buckets: []float64{.01, .05, .1, .25, .5, 1, 3, 10, 20, 60, 150, 300}, + }, []string{"controller"}), + processedEventDuration: prometheus.NewHistogramVec(prometheus.HistogramOpts{ + Namespace: controllerMetricsNamespace, + Subsystem: promControllerSubsystem, + Name: "processed_event_duration_seconds", + Help: "The duration for an event to be processed.", + Buckets: prometheus.DefBuckets, + }, []string{"controller", "success"}), + } + reg.MustRegister(r.queuedEventsTotal, r.inQueueEventDuration, r.processedEventDuration) + return r +} + +func (r *controllerRecorder) IncResourceEventQueued(_ context.Context, controller string, isRequeue bool) { + r.queuedEventsTotal.WithLabelValues(controller, strconv.FormatBool(isRequeue)).Inc() +} + +func (r *controllerRecorder) ObserveResourceInQueueDuration(_ context.Context, controller string, queuedAt time.Time) { + r.inQueueEventDuration.WithLabelValues(controller).Observe(time.Since(queuedAt).Seconds()) +} + +func (r *controllerRecorder) ObserveResourceProcessingDuration(_ context.Context, controller string, success bool, startProcessingAt time.Time) { + r.processedEventDuration.WithLabelValues(controller, strconv.FormatBool(success)).Observe(time.Since(startProcessingAt).Seconds()) +} + +func (r *controllerRecorder) RegisterResourceQueueLengthFunc(controller string, f func(context.Context) int) error { + err := r.reg.Register(prometheus.NewGaugeFunc(prometheus.GaugeOpts{ + Namespace: controllerMetricsNamespace, + Subsystem: promControllerSubsystem, + Name: "event_queue_length", + Help: "Length of the controller resource queue.", + ConstLabels: prometheus.Labels{"controller": controller}, + }, func() float64 { return float64(f(context.Background())) })) + if err != nil { + return fmt.Errorf("could not register ResourceQueueLengthFunc metrics: %w", err) + } + return nil +} + +type dummyControllerRecorder struct{} + +func (dummyControllerRecorder) IncResourceEventQueued(context.Context, string, bool) {} +func (dummyControllerRecorder) ObserveResourceInQueueDuration(context.Context, string, time.Time) {} +func (dummyControllerRecorder) ObserveResourceProcessingDuration(context.Context, string, bool, time.Time) { +} +func (dummyControllerRecorder) RegisterResourceQueueLengthFunc(string, func(context.Context) int) error { + return nil +} diff --git a/metrics/controller_test.go b/metrics/controller_test.go new file mode 100644 index 000000000..3bebbc0d7 --- /dev/null +++ b/metrics/controller_test.go @@ -0,0 +1,74 @@ +package metrics + +import ( + "context" + "strings" + "testing" + "time" + + "github.com/prometheus/client_golang/prometheus" + "github.com/prometheus/client_golang/prometheus/testutil" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +func TestControllerMetricsKeepTheirNames(t *testing.T) { + reg := prometheus.NewRegistry() + rec := NewRecorder("my_metrics", reg) + ctx := context.Background() + + rec.IncResourceEventQueued(ctx, "redisfailover", false) + rec.IncResourceEventQueued(ctx, "redisfailover", true) + rec.ObserveResourceInQueueDuration(ctx, "redisfailover", time.Now()) + rec.ObserveResourceProcessingDuration(ctx, "redisfailover", true, time.Now()) + require.NoError(t, rec.RegisterResourceQueueLengthFunc("redisfailover", func(context.Context) int { return 4 })) + + expected := ` +# HELP kooper_controller_event_queue_length Length of the controller resource queue. +# TYPE kooper_controller_event_queue_length gauge +kooper_controller_event_queue_length{controller="redisfailover"} 4 +# HELP kooper_controller_queued_events_total Total number of events queued. +# TYPE kooper_controller_queued_events_total counter +kooper_controller_queued_events_total{controller="redisfailover",requeue="false"} 1 +kooper_controller_queued_events_total{controller="redisfailover",requeue="true"} 1 +` + assert.NoError(t, testutil.GatherAndCompare(reg, strings.NewReader(expected), + "kooper_controller_event_queue_length", "kooper_controller_queued_events_total")) + + families, err := reg.Gather() + require.NoError(t, err) + buckets := map[string][]float64{} + labels := map[string][]string{} + for _, mf := range families { + if mf.GetType().String() != "HISTOGRAM" { + continue + } + m := mf.GetMetric()[0] + for _, b := range m.GetHistogram().GetBucket() { + buckets[mf.GetName()] = append(buckets[mf.GetName()], b.GetUpperBound()) + } + for _, l := range m.GetLabel() { + labels[mf.GetName()] = append(labels[mf.GetName()], l.GetName()+"="+l.GetValue()) + } + } + assert.Equal(t, map[string][]float64{ + "kooper_controller_event_in_queue_duration_seconds": {.01, .05, .1, .25, .5, 1, 3, 10, 20, 60, 150, 300}, + "kooper_controller_processed_event_duration_seconds": prometheus.DefBuckets, + }, buckets) + assert.Equal(t, map[string][]string{ + "kooper_controller_event_in_queue_duration_seconds": {"controller=redisfailover"}, + "kooper_controller_processed_event_duration_seconds": {"controller=redisfailover", "success=true"}, + }, labels) + + assert.EqualError(t, rec.RegisterResourceQueueLengthFunc("redisfailover", func(context.Context) int { return 0 }), + `could not register ResourceQueueLengthFunc metrics: duplicate metrics collector registration attempted`) +} + +func TestDummyControllerRecorder(t *testing.T) { + assert.NotPanics(t, func() { + Dummy.IncResourceEventQueued(context.Background(), "c", false) + Dummy.ObserveResourceInQueueDuration(context.Background(), "c", time.Now()) + Dummy.ObserveResourceProcessingDuration(context.Background(), "c", true, time.Now()) + }) + assert.NoError(t, Dummy.RegisterResourceQueueLengthFunc("c", func(context.Context) int { return 0 })) +} diff --git a/metrics/dummy.go b/metrics/dummy.go index d50050cc1..3f5087677 100644 --- a/metrics/dummy.go +++ b/metrics/dummy.go @@ -1,17 +1,13 @@ package metrics -import ( - koopercontroller "github.com/spotahome/kooper/v2/controller" -) - // Dummy is a handy instnce of a dummy instrumenter, most of the times it will be used on tests. var Dummy = &dummy{ - MetricsRecorder: koopercontroller.DummyMetricsRecorder, + ControllerRecorder: dummyControllerRecorder{}, } // dummy is a dummy implementation of Instrumenter. type dummy struct { - koopercontroller.MetricsRecorder + ControllerRecorder } func (d *dummy) SetClusterOK(namespace string, name string) {} diff --git a/metrics/metrics.go b/metrics/metrics.go index c0d2466ae..140917926 100644 --- a/metrics/metrics.go +++ b/metrics/metrics.go @@ -8,8 +8,6 @@ import ( "github.com/prometheus/client_golang/prometheus" "github.com/saremox/redis-operator/log" - koopercontroller "github.com/spotahome/kooper/v2/controller" - kooperprometheus "github.com/spotahome/kooper/v2/metrics/prometheus" ) const ( @@ -97,7 +95,7 @@ type checkMetricInfo struct { // Instrumenter is the interface that will collect the metrics and has ability to send/expose those metrics. type Recorder interface { - koopercontroller.MetricsRecorder + ControllerRecorder // ClusterOK metrics SetClusterOK(namespace string, name string) @@ -123,7 +121,7 @@ type recorder struct { sentinelCheck *prometheus.CounterVec // indicates any error encountered in managed sentinel instance(s) k8sServiceOperations *prometheus.CounterVec // number of operations performed on k8s redisOperations *prometheus.CounterVec // number of operations performed on redis/sentinel instances - koopercontroller.MetricsRecorder + ControllerRecorder } // NewPrometheusMetrics returns a new PromMetrics object. @@ -181,9 +179,7 @@ func NewRecorder(namespace string, reg prometheus.Registerer) Recorder { sentinelCheck: sentinelCheck, k8sServiceOperations: k8sServiceOperations, redisOperations: redisOperations, - MetricsRecorder: kooperprometheus.New(kooperprometheus.Config{ - Registerer: reg, - }), + ControllerRecorder: newControllerRecorder(reg), } // Register metrics. diff --git a/operator/redisfailover/controller.go b/operator/redisfailover/controller.go index fd119bed3..9329eccc4 100644 --- a/operator/redisfailover/controller.go +++ b/operator/redisfailover/controller.go @@ -3,12 +3,9 @@ package redisfailover import ( "context" "errors" - "fmt" "sync" "time" - "github.com/spotahome/kooper/v2/controller" - "github.com/spotahome/kooper/v2/controller/leaderelection" corev1 "k8s.io/api/core/v1" "k8s.io/apimachinery/pkg/api/meta" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" @@ -20,22 +17,33 @@ import ( "k8s.io/client-go/util/workqueue" "github.com/saremox/redis-operator/log" + "github.com/saremox/redis-operator/metrics" ) const controllerName = "redisfailover" +// Controller runs the operator until its context is done. +type Controller interface { + Run(ctx context.Context) error +} + +// Handler reconciles one object. +type Handler interface { + Handle(ctx context.Context, obj runtime.Object) error +} + // rfController reconciles RedisFailovers from one queue keyed by RedisFailover. // Events on a RedisFailover's pods queue that RedisFailover, so a pod change // (deleted, recreated, ready) drives the next reconcile without waiting for // the resync. type rfController struct { - handler controller.Handler + handler Handler rfInformer cache.SharedIndexInformer podInformer cache.SharedIndexInformer queue workqueue.TypedInterface[string] workers int - leRunner leaderelection.Runner - metrics controller.MetricsRecorder + leRunner leaderRunner + metrics metrics.ControllerRecorder logger log.Logger // queuedAt feeds the in-queue duration metric. @@ -43,7 +51,7 @@ type rfController struct { queuedAt map[string]time.Time } -func newRFController(handler controller.Handler, rfRetriever controller.Retriever, podLW cache.ListerWatcher, resync time.Duration, workers int, leRunner leaderelection.Runner, metrics controller.MetricsRecorder, logger log.Logger) (*rfController, error) { +func newRFController(handler Handler, rfLW, podLW cache.ListerWatcher, resync time.Duration, workers int, leRunner leaderRunner, mrec metrics.ControllerRecorder, logger log.Logger) (*rfController, error) { if resync <= 0 { resync = 3 * time.Minute } @@ -52,12 +60,12 @@ func newRFController(handler controller.Handler, rfRetriever controller.Retrieve } c := &rfController{ handler: handler, - rfInformer: cache.NewSharedIndexInformer(listerWatcher(rfRetriever), nil, resync, cache.Indexers{}), + rfInformer: cache.NewSharedIndexInformer(rfLW, nil, resync, cache.Indexers{}), podInformer: cache.NewSharedIndexInformer(podLW, &corev1.Pod{}, 0, cache.Indexers{}), queue: workqueue.NewTyped[string](), workers: workers, leRunner: leRunner, - metrics: metrics, + metrics: mrec, logger: logger, queuedAt: map[string]time.Time{}, } @@ -70,20 +78,13 @@ func newRFController(handler controller.Handler, rfRetriever controller.Retrieve c.podInformer.SetTransform(podMetadataOnly), rfErr, podErr, - metrics.RegisterResourceQueueLengthFunc(controllerName, queueLen), + mrec.RegisterResourceQueueLengthFunc(controllerName, queueLen), ); err != nil { return nil, err } return c, nil } -func listerWatcher(r controller.Retriever) cache.ListerWatcher { - return &cache.ListWatch{ - ListWithContextFunc: r.List, - WatchFuncWithContext: r.Watch, - } -} - // podMetadataOnly keeps only what podOwnerKey needs in the pod cache. func podMetadataOnly(obj any) (any, error) { pod, ok := obj.(*corev1.Pod) @@ -163,34 +164,40 @@ func (c *rfController) enqueue(key string) { c.queue.Add(key) } -// Run satisfies controller.Controller. +// Run satisfies Controller. func (c *rfController) Run(ctx context.Context) error { if c.leRunner == nil { return c.run(ctx) } - return c.leRunner.Run(func() error { return c.run(ctx) }) + return c.leRunner.Run(ctx, c.run) } func (c *rfController) run(ctx context.Context) error { - defer c.queue.ShutDown() - c.logger.Infof("starting controller") go c.rfInformer.RunWithContext(ctx) go c.podInformer.RunWithContext(ctx) // Pod events only speed reconciles up, so a pod watch that can't sync // (e.g. RBAC) must not block reconciling. + // The wait only fails once ctx is done, i.e. on shutdown. if !cache.WaitForNamedCacheSyncWithContext(ctx, c.rfInformer.HasSynced) { - return fmt.Errorf("timed out waiting for caches to sync") + c.logger.Infof("controller stopped before its cache synced") + return nil } + var workers sync.WaitGroup for range c.workers { - go wait.UntilWithContext(ctx, func(ctx context.Context) { - for c.processNext(ctx) { - } - }, time.Second) + workers.Go(func() { + wait.UntilWithContext(ctx, func(ctx context.Context) { + for c.processNext(ctx) { + } + }, time.Second) + }) } <-ctx.Done() - c.logger.Infof("stopping controller") + c.logger.Infof("stopping controller, waiting for running reconciles") + c.queue.ShutDown() + workers.Wait() + c.logger.Infof("controller stopped") return nil } @@ -200,6 +207,10 @@ func (c *rfController) processNext(ctx context.Context) bool { return false } defer c.queue.Done(key) + // A shut down queue still hands out what it holds; drop it once stopping. + if ctx.Err() != nil { + return false + } c.mu.Lock() if queuedAt, ok := c.queuedAt[key]; ok { diff --git a/operator/redisfailover/controller_test.go b/operator/redisfailover/controller_test.go index ec2b4b13e..e9a52f907 100644 --- a/operator/redisfailover/controller_test.go +++ b/operator/redisfailover/controller_test.go @@ -8,7 +8,6 @@ import ( "testing" "time" - "github.com/spotahome/kooper/v2/controller" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" corev1 "k8s.io/api/core/v1" @@ -27,16 +26,15 @@ import ( "github.com/saremox/redis-operator/metrics" ) -type staticRFRetriever struct { - items []redisfailoverv1.RedisFailover -} - -func (r staticRFRetriever) List(context.Context, metav1.ListOptions) (runtime.Object, error) { - return &redisfailoverv1.RedisFailoverList{Items: r.items}, nil -} - -func (r staticRFRetriever) Watch(context.Context, metav1.ListOptions) (watch.Interface, error) { - return watch.NewFake(), nil +func staticRFs(items ...redisfailoverv1.RedisFailover) *cache.ListWatch { + return &cache.ListWatch{ + ListWithContextFunc: func(context.Context, metav1.ListOptions) (runtime.Object, error) { + return &redisfailoverv1.RedisFailoverList{Items: items}, nil + }, + WatchFuncWithContext: func(context.Context, metav1.ListOptions) (watch.Interface, error) { + return watch.NewFake(), nil + }, + } } type recordingHandler struct { @@ -74,14 +72,14 @@ func rfPod(name, rf string) *corev1.Pod { }} } -func startController(t *testing.T, h controller.Handler, kube *fakekubernetes.Clientset, rfs ...redisfailoverv1.RedisFailover) *rfController { +func startController(t *testing.T, h Handler, kube *fakekubernetes.Clientset, rfs ...redisfailoverv1.RedisFailover) *rfController { t.Helper() return startControllerWith(t, h, kube, time.Hour, metrics.Dummy, rfs...) } // startControllerWith returns once the pod watch is registered, so pods // created afterwards produce events. -func startControllerWith(t *testing.T, h controller.Handler, kube *fakekubernetes.Clientset, resync time.Duration, mrec metrics.Recorder, rfs ...redisfailoverv1.RedisFailover) *rfController { +func startControllerWith(t *testing.T, h Handler, kube *fakekubernetes.Clientset, resync time.Duration, mrec metrics.ControllerRecorder, rfs ...redisfailoverv1.RedisFailover) *rfController { t.Helper() clientfeaturestesting.SetFeatureDuringTest(t, clientfeatures.WatchListClient, false) watching := make(chan struct{}) @@ -91,7 +89,7 @@ func startControllerWith(t *testing.T, h controller.Handler, kube *fakekubernete once.Do(func() { close(watching) }) return true, w, err }) - c, err := newRFController(h, staticRFRetriever{items: rfs}, newPodListWatch(kube), resync, 3, nil, mrec, log.Dummy) + c, err := newRFController(h, staticRFs(rfs...), newPodListWatch(kube), resync, 3, nil, mrec, log.Dummy) require.NoError(t, err) ctx, cancel := context.WithCancel(context.Background()) done := make(chan error) @@ -188,7 +186,7 @@ func (failingQueueMetrics) RegisterResourceQueueLengthFunc(string, func(context. } func TestNewRFControllerReturnsMetricsRegistrationError(t *testing.T) { - _, err := newRFController(&recordingHandler{}, staticRFRetriever{}, newPodListWatch(fakekubernetes.NewClientset()), 0, 0, nil, failingQueueMetrics{metrics.Dummy}, log.Dummy) + _, err := newRFController(&recordingHandler{}, staticRFs(), newPodListWatch(fakekubernetes.NewClientset()), 0, 0, nil, failingQueueMetrics{metrics.Dummy}, log.Dummy) assert.EqualError(t, err, "already registered") } @@ -204,25 +202,80 @@ func (m *queueLengthMetrics) RegisterResourceQueueLengthFunc(_ string, f func(co type recordingRunner struct{ ran bool } -func (r *recordingRunner) Run(f func() error) error { +func (r *recordingRunner) Run(ctx context.Context, f func(context.Context) error) error { r.ran = true - return f() + return f(ctx) } func TestRFControllerRunsUnderLeaderElection(t *testing.T) { runner := &recordingRunner{} mrec := &queueLengthMetrics{Recorder: metrics.Dummy} - c, err := newRFController(&recordingHandler{}, staticRFRetriever{}, newPodListWatch(fakekubernetes.NewClientset()), time.Hour, 1, runner, mrec, log.Dummy) + c, err := newRFController(&recordingHandler{}, staticRFs(), newPodListWatch(fakekubernetes.NewClientset()), time.Hour, 1, runner, mrec, log.Dummy) require.NoError(t, err) c.enqueue("ns/rf") assert.Equal(t, 1, mrec.queueLen(context.Background())) ctx, cancel := context.WithCancel(context.Background()) cancel() - assert.EqualError(t, c.Run(ctx), "timed out waiting for caches to sync") + assert.NoError(t, c.Run(ctx), "shutting down before the cache synced is not an error") assert.True(t, runner.ran) } +type blockingHandler struct { + once sync.Once + calls atomic.Int32 + started chan struct{} + release chan struct{} +} + +func (h *blockingHandler) Handle(context.Context, runtime.Object) error { + h.calls.Add(1) + h.once.Do(func() { close(h.started) }) + <-h.release + return nil +} + +func TestRFControllerRunWaitsForInFlightReconciles(t *testing.T) { + rf := redisfailoverv1.RedisFailover{ObjectMeta: metav1.ObjectMeta{Name: "rf", Namespace: "ns"}} + clientfeaturestesting.SetFeatureDuringTest(t, clientfeatures.WatchListClient, false) + h := &blockingHandler{started: make(chan struct{}), release: make(chan struct{})} + c, err := newRFController(h, staticRFs(rf), newPodListWatch(fakekubernetes.NewClientset()), time.Hour, 1, nil, metrics.Dummy, log.Dummy) + require.NoError(t, err) + ctx, cancel := context.WithCancel(context.Background()) + done := make(chan error) + go func() { done <- c.Run(ctx) }() + + <-h.started + cancel() + select { + case <-done: + t.Fatal("Run returned while a reconcile was still running") + case <-time.After(200 * time.Millisecond): + } + close(h.release) + assert.NoError(t, <-done) +} + +func TestRFControllerStopsTakingQueuedReconcilesOnceCancelled(t *testing.T) { + clientfeaturestesting.SetFeatureDuringTest(t, clientfeatures.WatchListClient, false) + var rfs []redisfailoverv1.RedisFailover + for _, name := range []string{"a", "b", "c", "d"} { + rfs = append(rfs, redisfailoverv1.RedisFailover{ObjectMeta: metav1.ObjectMeta{Name: name, Namespace: "ns"}}) + } + h := &blockingHandler{started: make(chan struct{}), release: make(chan struct{})} + c, err := newRFController(h, staticRFs(rfs...), newPodListWatch(fakekubernetes.NewClientset()), time.Hour, 1, nil, metrics.Dummy, log.Dummy) + require.NoError(t, err) + ctx, cancel := context.WithCancel(context.Background()) + done := make(chan error) + go func() { done <- c.Run(ctx) }() + + <-h.started + cancel() + close(h.release) + assert.NoError(t, <-done) + assert.Equal(t, int32(1), h.calls.Load(), "the queued RedisFailovers are not reconciled after cancel") +} + type failingHandler struct{ calls atomic.Int32 } func (h *failingHandler) Handle(context.Context, runtime.Object) error { @@ -280,7 +333,7 @@ func TestRFControllerReconcilesWithoutPodAccess(t *testing.T) { return true, nil, apierrors.NewForbidden(corev1.Resource("pods"), "", errors.New("denied")) }) h := &recordingHandler{} - c, err := newRFController(h, staticRFRetriever{items: []redisfailoverv1.RedisFailover{rf}}, newPodListWatch(kube), time.Hour, 1, nil, metrics.Dummy, log.Dummy) + c, err := newRFController(h, staticRFs(rf), newPodListWatch(kube), time.Hour, 1, nil, metrics.Dummy, log.Dummy) require.NoError(t, err) ctx, cancel := context.WithCancel(context.Background()) done := make(chan error) @@ -292,7 +345,7 @@ func TestRFControllerReconcilesWithoutPodAccess(t *testing.T) { } func TestRFControllerPodOwnerIgnoresUnhandledRedisFailovers(t *testing.T) { - c, err := newRFController(&recordingHandler{}, staticRFRetriever{}, newPodListWatch(fakekubernetes.NewClientset()), time.Hour, 1, nil, metrics.Dummy, log.Dummy) + c, err := newRFController(&recordingHandler{}, staticRFs(), newPodListWatch(fakekubernetes.NewClientset()), time.Hour, 1, nil, metrics.Dummy, log.Dummy) require.NoError(t, err) require.NoError(t, c.rfInformer.GetIndexer().Add(&redisfailoverv1.RedisFailover{ObjectMeta: metav1.ObjectMeta{Name: "rf", Namespace: "ns"}})) @@ -309,7 +362,7 @@ func TestRFControllerPodOwnerIgnoresUnhandledRedisFailovers(t *testing.T) { func TestRFControllerSkipsDeletedRedisFailovers(t *testing.T) { h := &recordingHandler{} - c, err := newRFController(h, staticRFRetriever{}, newPodListWatch(fakekubernetes.NewClientset()), time.Hour, 1, nil, metrics.Dummy, log.Dummy) + c, err := newRFController(h, staticRFs(), newPodListWatch(fakekubernetes.NewClientset()), time.Hour, 1, nil, metrics.Dummy, log.Dummy) require.NoError(t, err) assert.NoError(t, c.process(context.Background(), "ns/gone")) @@ -325,7 +378,7 @@ func TestRFControllerResyncsRedisFailovers(t *testing.T) { } func TestNewRFControllerDefaultsWorkers(t *testing.T) { - c, err := newRFController(&recordingHandler{}, staticRFRetriever{}, newPodListWatch(fakekubernetes.NewClientset()), 0, 0, nil, metrics.Dummy, log.Dummy) + c, err := newRFController(&recordingHandler{}, staticRFs(), newPodListWatch(fakekubernetes.NewClientset()), 0, 0, nil, metrics.Dummy, log.Dummy) require.NoError(t, err) assert.Equal(t, 3, c.workers) } @@ -409,7 +462,7 @@ func TestRFControllerRecordsMetrics(t *testing.T) { func TestRFControllerCountsOneEventPerUpdate(t *testing.T) { mrec := &recordingMetrics{Recorder: metrics.Dummy} - c, err := newRFController(&recordingHandler{}, staticRFRetriever{}, newPodListWatch(fakekubernetes.NewClientset()), time.Hour, 1, nil, mrec, log.Dummy) + c, err := newRFController(&recordingHandler{}, staticRFs(), newPodListWatch(fakekubernetes.NewClientset()), time.Hour, 1, nil, mrec, log.Dummy) require.NoError(t, err) a := &redisfailoverv1.RedisFailover{ObjectMeta: metav1.ObjectMeta{Name: "a", Namespace: "ns"}} b := &redisfailoverv1.RedisFailover{ObjectMeta: metav1.ObjectMeta{Name: "b", Namespace: "ns"}} diff --git a/operator/redisfailover/factory.go b/operator/redisfailover/factory.go index 8fa053b8f..6b45ed752 100644 --- a/operator/redisfailover/factory.go +++ b/operator/redisfailover/factory.go @@ -5,9 +5,6 @@ import ( "regexp" "time" - "github.com/spotahome/kooper/v2/controller" - "github.com/spotahome/kooper/v2/controller/leaderelection" - kooperlog "github.com/spotahome/kooper/v2/log" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" "k8s.io/apimachinery/pkg/runtime" "k8s.io/apimachinery/pkg/watch" @@ -32,43 +29,42 @@ const ( // New will create an operator that is responsible for managing all the required stuff // to create redis failovers. -func New(cfg Config, k8sService k8s.Services, k8sClient kubernetes.Interface, lockNamespace string, redisClient redis.Client, kooperMetricsRecorder metrics.Recorder, logger log.Logger) (controller.Controller, error) { +func New(cfg Config, k8sService k8s.Services, k8sClient kubernetes.Interface, lockNamespace string, redisClient redis.Client, metricsRecorder metrics.Recorder, logger log.Logger) (Controller, error) { // Create internal services. - rfService := rfservice.NewRedisFailoverKubeClient(k8sService, logger, kooperMetricsRecorder) + rfService := rfservice.NewRedisFailoverKubeClient(k8sService, logger, metricsRecorder) var opts []rfservice.Option if !cfg.KeepClientsOnDemotion { disconnector := rfservice.NewClientDisconnector(k8sClient, redisClient, logger, endpointRemovalTimeout, kubeProxySyncGrace) opts = append(opts, rfservice.WithClientDisconnector(disconnector)) } - rfChecker := rfservice.NewRedisFailoverChecker(k8sService, redisClient, logger, kooperMetricsRecorder, opts...) + rfChecker := rfservice.NewRedisFailoverChecker(k8sService, redisClient, logger, metricsRecorder, opts...) rfHealer := rfservice.NewRedisFailoverHealer(k8sService, redisClient, logger, opts...) // Create the handlers. - rfHandler := NewRedisFailoverHandler(cfg, rfService, rfChecker, rfHealer, k8sService, kooperMetricsRecorder, logger) + rfHandler := NewRedisFailoverHandler(cfg, rfService, rfChecker, rfHealer, k8sService, metricsRecorder, logger) rfRetriever := NewRedisFailoverRetriever(cfg, k8sService) - kooperLogger := kooperlogger{Logger: logger.WithField("operator", "redisfailover")} - // Leader election service. - leSVC, err := leaderelection.NewDefault(lockKey, lockNamespace, k8sClient, kooperLogger) + logger = logger.WithField("operator", "redisfailover") + leRunner, err := newLeaseRunner(lockKey, lockNamespace, k8sClient, logger) if err != nil { return nil, err } - c, err := newRFController(rfHandler, rfRetriever, newPodListWatch(k8sClient), time.Duration(cfg.SyncInterval)*time.Second, cfg.Concurrency, leSVC, kooperMetricsRecorder, logger.WithField("operator", "redisfailover")) + c, err := newRFController(rfHandler, rfRetriever, newPodListWatch(k8sClient), time.Duration(cfg.SyncInterval)*time.Second, cfg.Concurrency, leRunner, metricsRecorder, logger) if err != nil { return nil, err } return c, nil } -func NewRedisFailoverRetriever(cfg Config, cli k8s.Services) controller.Retriever { +func NewRedisFailoverRetriever(cfg Config, cli k8s.Services) *cache.ListWatch { isNamespaceSupported := func(rf redisfailoverv1.RedisFailover) bool { match, _ := regexp.Match(cfg.SupportedNamespacesRegex, []byte(rf.Namespace)) return match } // check in the startup whether the regex compiles - return controller.MustRetrieverFromListerWatcher(&cache.ListWatch{ + return &cache.ListWatch{ ListWithContextFunc: func(ctx context.Context, options metav1.ListOptions) (runtime.Object, error) { rfList, err := cli.ListRedisFailovers(ctx, "", options) if err != nil { @@ -105,13 +101,5 @@ func NewRedisFailoverRetriever(cfg Config, cli k8s.Services) controller.Retrieve }) return watcher, err }, - }) -} - -type kooperlogger struct { - log.Logger -} - -func (k kooperlogger) WithKV(kv kooperlog.KV) kooperlog.Logger { - return kooperlogger{Logger: k.WithFields(kv)} + } } diff --git a/operator/redisfailover/factory_test.go b/operator/redisfailover/factory_test.go index 8d8b6f1dc..8c56c69ae 100644 --- a/operator/redisfailover/factory_test.go +++ b/operator/redisfailover/factory_test.go @@ -20,8 +20,6 @@ import ( fakekubernetes "k8s.io/client-go/kubernetes/fake" "k8s.io/client-go/tools/cache" - kooperlog "github.com/spotahome/kooper/v2/log" - redisfailoverv1 "github.com/saremox/redis-operator/api/redisfailover/v1" "github.com/saremox/redis-operator/log" "github.com/saremox/redis-operator/metrics" @@ -29,10 +27,6 @@ import ( mRedisService "github.com/saremox/redis-operator/mocks/service/redis" ) -// This file is deliberately in package redisfailover (white-box) rather than -// redisfailover_test, because kooperlogger and its WithKV method are -// unexported and can't be reached from outside the package. - // ----------------------------------------------------------------------- // NewRedisFailoverRetriever - List // ----------------------------------------------------------------------- @@ -51,7 +45,7 @@ func TestNewRedisFailoverRetrieverListFiltersByNamespace(t *testing.T) { mk.On("ListRedisFailovers", mock.Anything, "", mock.Anything).Once().Return(rfList, nil) retriever := NewRedisFailoverRetriever(cfg, mk) - obj, err := retriever.List(context.Background(), metav1.ListOptions{}) + obj, err := retriever.ListWithContext(context.Background(), metav1.ListOptions{}) require.NoError(t, err) got, ok := obj.(*redisfailoverv1.RedisFailoverList) @@ -73,7 +67,7 @@ func TestNewRedisFailoverRetrieverListPropagatesError(t *testing.T) { mk.On("ListRedisFailovers", mock.Anything, "", mock.Anything).Once().Return(nil, wantErr) retriever := NewRedisFailoverRetriever(cfg, mk) - obj, err := retriever.List(context.Background(), metav1.ListOptions{}) + obj, err := retriever.ListWithContext(context.Background(), metav1.ListOptions{}) assert.Equal(t, wantErr, err) assert.Nil(t, obj) @@ -92,7 +86,7 @@ func TestNewRedisFailoverRetrieverWatchFiltersByNamespace(t *testing.T) { mk.On("WatchRedisFailovers", mock.Anything, "", mock.Anything).Once().Return(fakeWatcher, nil) retriever := NewRedisFailoverRetriever(cfg, mk) - w, err := retriever.Watch(context.Background(), metav1.ListOptions{}) + w, err := retriever.WatchWithContext(context.Background(), metav1.ListOptions{}) require.NoError(t, err) require.NotNil(t, w) @@ -127,7 +121,7 @@ func TestNewRedisFailoverRetrieverWatchPropagatesError(t *testing.T) { mk.On("WatchRedisFailovers", mock.Anything, "", mock.Anything).Once().Return(nil, wantErr) retriever := NewRedisFailoverRetriever(cfg, mk) - w, err := retriever.Watch(context.Background(), metav1.ListOptions{}) + w, err := retriever.WatchWithContext(context.Background(), metav1.ListOptions{}) assert.Equal(t, wantErr, err) assert.Nil(t, w) @@ -149,7 +143,7 @@ func TestNewRedisFailoverRetrieverWatchNilWatcherNoError(t *testing.T) { var w watch.Interface var err error assert.NotPanics(t, func() { - w, err = retriever.Watch(context.Background(), metav1.ListOptions{}) + w, err = retriever.WatchWithContext(context.Background(), metav1.ListOptions{}) }) assert.NoError(t, err) assert.Nil(t, w) @@ -160,11 +154,7 @@ func TestNewRedisFailoverRetrieverWatchNilWatcherNoError(t *testing.T) { // controller does. func rfInformer(t *testing.T, mk *mK8SService.Services) cache.SharedIndexInformer { t.Helper() - retriever := NewRedisFailoverRetriever(Config{SupportedNamespacesRegex: "^allowed$"}, mk) - lw := &cache.ListWatch{ - ListWithContextFunc: retriever.List, - WatchFuncWithContext: retriever.Watch, - } + lw := NewRedisFailoverRetriever(Config{SupportedNamespacesRegex: "^allowed$"}, mk) informer := cache.NewSharedIndexInformer(lw, nil, 0, cache.Indexers{}) ctx, cancel := context.WithCancel(context.Background()) t.Cleanup(cancel) @@ -204,37 +194,6 @@ func TestRedisFailoverInformerRelistsOnAnExpiredWatch(t *testing.T) { assert.Eventually(t, func() bool { return lists.Load() > 1 }, 5*time.Second, 10*time.Millisecond, "an expired watch must make the informer relist") } -// ----------------------------------------------------------------------- -// kooperlogger.WithKV -// ----------------------------------------------------------------------- - -// kvCapturingLogger is a minimal log.Logger test double that records the -// values passed to WithFields so we can assert kooperlogger.WithKV forwards -// its kooperlog.KV argument correctly. -type kvCapturingLogger struct { - log.DummyLogger - lastKV map[string]interface{} -} - -func (l *kvCapturingLogger) WithFields(values map[string]interface{}) log.Logger { - l.lastKV = values - return l -} - -func TestKooperLoggerWithKV(t *testing.T) { - fake := &kvCapturingLogger{} - kl := kooperlogger{Logger: fake} - - kv := kooperlog.KV{"foo": "bar", "n": 1} - result := kl.WithKV(kv) - - wrapped, ok := result.(kooperlogger) - require.True(t, ok) - - assert.Equal(t, map[string]interface{}{"foo": "bar", "n": 1}, fake.lastKV) - assert.Same(t, fake, wrapped.Logger) -} - // ----------------------------------------------------------------------- // New // ----------------------------------------------------------------------- diff --git a/operator/redisfailover/leaderelection.go b/operator/redisfailover/leaderelection.go new file mode 100644 index 000000000..5759710ca --- /dev/null +++ b/operator/redisfailover/leaderelection.go @@ -0,0 +1,161 @@ +package redisfailover + +import ( + "context" + "errors" + "fmt" + "os" + "time" + + apierrors "k8s.io/apimachinery/pkg/api/errors" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/util/uuid" + "k8s.io/client-go/kubernetes" + "k8s.io/client-go/tools/leaderelection" + "k8s.io/client-go/tools/leaderelection/resourcelock" + + "github.com/saremox/redis-operator/log" +) + +const ( + leaseDuration = 15 * time.Second + renewDeadline = 10 * time.Second + retryPeriod = 2 * time.Second +) + +var ( + hostname = os.Hostname + errLeadershipLost = errors.New("leadership lost") +) + +// leaderRunner runs a function only while this instance holds the lease. +type leaderRunner interface { + Run(ctx context.Context, f func(context.Context) error) error +} + +type leaseRunner struct { + lock resourcelock.Interface + leaseDuration, renewDeadline, retryPeriod time.Duration + logger log.Logger +} + +func newLeaseRunner(key, namespace string, k8scli kubernetes.Interface, logger log.Logger) (*leaseRunner, error) { + if namespace == "" { + return nil, fmt.Errorf("running in leader election mode requires the namespace running") + } + if key == "" { + return nil, fmt.Errorf("running in leader election mode requires a key for identification the different instances") + } + host, err := hostname() + if err != nil { + return nil, err + } + + return &leaseRunner{ + lock: &resourcelock.LeaseLock{ + LeaseMeta: metav1.ObjectMeta{Namespace: namespace, Name: key}, + Client: k8scli.CoordinationV1(), + LockConfig: resourcelock.ResourceLockConfig{Identity: host + "_" + string(uuid.NewUUID())}, + }, + leaseDuration: leaseDuration, + renewDeadline: renewDeadline, + retryPeriod: retryPeriod, + logger: logger.WithField("source-service", "leader-election").WithField("leader-election-id", namespace+"/"+key), + }, nil +} + +// Run runs f once this instance holds the lease, until ctx is done or the +// lease is lost. When ctx is done, Run waits for f and then releases the +// lease. When the lease is lost, Run returns errLeadershipLost at once so the +// process can exit before another replica takes over, even if f is still +// running. +func (r *leaseRunner) Run(ctx context.Context, f func(context.Context) error) error { + // The election has its own context: once this instance leads, it keeps + // renewing the lease until f has stopped, even if ctx is already done. + electionCtx, stopElection := context.WithCancel(context.WithoutCancel(ctx)) + defer stopElection() + + leading := make(chan context.Context, 1) + electionDone := make(chan struct{}) + go func() { + defer close(electionDone) + // No ReleaseOnCancel: client-go would also release a lost lease. + leaderelection.RunOrDie(electionCtx, leaderelection.LeaderElectionConfig{ + Lock: r.lock, + LeaseDuration: r.leaseDuration, + RenewDeadline: r.renewDeadline, + RetryPeriod: r.retryPeriod, + Callbacks: leaderelection.LeaderCallbacks{ + OnStartedLeading: func(leaderCtx context.Context) { leading <- leaderCtx }, + OnStoppedLeading: func() {}, + }, + }) + }() + r.logger.Infof("running in leader election mode, waiting to acquire leadership...") + + // Without ctx being done, the election only ends after leading, and then + // leading always receives the leader context, done if the lease is lost. + var err error + select { + case leaderCtx := <-leading: + if ctx.Err() == nil { + err = r.lead(ctx, leaderCtx, f) + } + case <-ctx.Done(): + } + if errors.Is(err, errLeadershipLost) { + r.logger.Warningf("leader lease lost") + return err + } + stopElection() + <-electionDone + r.release() + return err +} + +// lead runs f until ctx is done or the lease is lost. +func (r *leaseRunner) lead(ctx, leaderCtx context.Context, f func(context.Context) error) error { + r.logger.Infof("lead acquire, starting...") + runCtx, cancelRun := context.WithCancel(leaderCtx) + defer cancelRun() + stopRun := context.AfterFunc(ctx, cancelRun) + defer stopRun() + + done := make(chan error, 1) + go func() { done <- f(runCtx) }() + var err error + select { + case err = <-done: + r.logger.Infof("lead execution stopped") + case <-leaderCtx.Done(): + } + if leaderCtx.Err() != nil { + return errLeadershipLost + } + return err +} + +// release gives up the lease if this instance holds it, so another replica +// takes over at once instead of waiting for it to expire. +func (r *leaseRunner) release() { + ctx, cancel := context.WithTimeout(context.Background(), r.renewDeadline) + defer cancel() + current, _, err := r.lock.Get(ctx) + if apierrors.IsNotFound(err) || (err == nil && current.HolderIdentity != r.lock.Identity()) { + return + } + if err == nil { + now := metav1.NewTime(time.Now()) + err = r.lock.Update(ctx, resourcelock.LeaderElectionRecord{ + LeaderTransitions: current.LeaderTransitions, + LeaseDurationSeconds: 1, + RenewTime: now, + AcquireTime: now, + }) + } + if err != nil { + r.logger.Warningf("could not release the leader lease, it expires in %s: %v", r.leaseDuration, err) + return + } + r.logger.Infof("released the leader lease") +} diff --git a/operator/redisfailover/leaderelection_test.go b/operator/redisfailover/leaderelection_test.go new file mode 100644 index 000000000..ecff4fec3 --- /dev/null +++ b/operator/redisfailover/leaderelection_test.go @@ -0,0 +1,266 @@ +package redisfailover + +import ( + "context" + "errors" + "sync/atomic" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + coordinationv1 "k8s.io/api/coordination/v1" + apierrors "k8s.io/apimachinery/pkg/api/errors" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/runtime" + fakekubernetes "k8s.io/client-go/kubernetes/fake" + k8stesting "k8s.io/client-go/testing" + "k8s.io/utils/ptr" + + "github.com/saremox/redis-operator/log" +) + +func TestNewLeaseRunnerValidates(t *testing.T) { + _, err := newLeaseRunner(lockKey, "", fakekubernetes.NewClientset(), log.Dummy) + assert.EqualError(t, err, "running in leader election mode requires the namespace running") + + _, err = newLeaseRunner("", "ns", fakekubernetes.NewClientset(), log.Dummy) + assert.EqualError(t, err, "running in leader election mode requires a key for identification the different instances") + + defer func(h func() (string, error)) { hostname = h }(hostname) + hostname = func() (string, error) { return "", errors.New("no hostname") } + _, err = newLeaseRunner(lockKey, "ns", fakekubernetes.NewClientset(), log.Dummy) + assert.EqualError(t, err, "no hostname") +} + +func TestNewLeaseRunnerKeepsKooperTimings(t *testing.T) { + r, err := newLeaseRunner(lockKey, "ns", fakekubernetes.NewClientset(), log.Dummy) + require.NoError(t, err) + assert.Equal(t, 15*time.Second, r.leaseDuration) + assert.Equal(t, 10*time.Second, r.renewDeadline) + assert.Equal(t, 2*time.Second, r.retryPeriod) +} + +func fastLeaseRunner(t *testing.T, kube *fakekubernetes.Clientset) *leaseRunner { + t.Helper() + r, err := newLeaseRunner(lockKey, "ns", kube, log.Dummy) + require.NoError(t, err) + r.leaseDuration, r.renewDeadline, r.retryPeriod = 300*time.Millisecond, 200*time.Millisecond, 50*time.Millisecond + return r +} + +func leaseHolder(t *testing.T, kube *fakekubernetes.Clientset) string { + t.Helper() + lease, err := kube.CoordinationV1().Leases("ns").Get(context.Background(), lockKey, metav1.GetOptions{}) + require.NoError(t, err) + if lease.Spec.HolderIdentity == nil { + return "" + } + return *lease.Spec.HolderIdentity +} + +func TestLeaseRunnerReleasesTheLeaseWhenFReturns(t *testing.T) { + kube := fakekubernetes.NewClientset() + r := fastLeaseRunner(t, kube) + + var holder string + err := r.Run(context.Background(), func(context.Context) error { + holder = leaseHolder(t, kube) + return errors.New("done") + }) + + assert.EqualError(t, err, "done") + assert.Equal(t, r.lock.Identity(), holder) + assert.Empty(t, leaseHolder(t, kube)) +} + +func TestLeaseRunnerStopsFThenReleasesTheLeaseOnShutdown(t *testing.T) { + kube := fakekubernetes.NewClientset() + r := fastLeaseRunner(t, kube) + ctx, cancel := context.WithCancel(context.Background()) + + var holderWhenStopped string + err := r.Run(ctx, func(runCtx context.Context) error { + cancel() + <-runCtx.Done() + holderWhenStopped = leaseHolder(t, kube) + return nil + }) + + assert.NoError(t, err) + assert.Equal(t, r.lock.Identity(), holderWhenStopped, "the lease is held until f has returned") + assert.Empty(t, leaseHolder(t, kube)) + events, err := kube.CoreV1().Events("").List(context.Background(), metav1.ListOptions{}) + require.NoError(t, err) + assert.Empty(t, events.Items, "like kooper, no leader election events are written") +} + +func TestLeaseRunnerReturnsAtOnceWhenLeadershipIsLost(t *testing.T) { + kube := fakekubernetes.NewClientset() + r := fastLeaseRunner(t, kube) + var leading atomic.Bool + kube.PrependReactor("update", "leases", func(k8stesting.Action) (bool, runtime.Object, error) { + if leading.Load() { + return true, nil, apierrors.NewServiceUnavailable("down") + } + return false, nil, nil + }) + fStopping := make(chan struct{}) + finishF := make(chan struct{}) + defer close(finishF) + + err := r.Run(context.Background(), func(runCtx context.Context) error { + leading.Store(true) + <-runCtx.Done() + close(fStopping) + <-finishF // a reconcile that is still running + return nil + }) + + assert.ErrorIs(t, err, errLeadershipLost) + select { + case <-fStopping: + case <-time.After(5 * time.Second): + t.Fatal("f's context was not cancelled") + } +} + +func TestLeaseRunnerReturnsWhenCancelledBeforeLeading(t *testing.T) { + ctx, cancel := context.WithCancel(context.Background()) + cancel() + r := fastLeaseRunner(t, fakekubernetes.NewClientset()) + + err := r.Run(ctx, func(context.Context) error { + t.Error("f must not run") + return nil + }) + + assert.NoError(t, err) +} + +func TestLeaseRunnerStopsWaitingWhenAnotherReplicaLeads(t *testing.T) { + kube := fakekubernetes.NewClientset() + other := fastLeaseRunner(t, kube) + // Leases store whole seconds, so the holder needs at least one. + other.leaseDuration, other.renewDeadline = 2*time.Second, time.Second + otherCtx, stopOther := context.WithCancel(context.Background()) + otherDone := make(chan error) + otherLeading := make(chan struct{}) + go func() { + otherDone <- other.Run(otherCtx, func(ctx context.Context) error { + close(otherLeading) + <-ctx.Done() + return nil + }) + }() + <-otherLeading + defer func() { + stopOther() + require.NoError(t, <-otherDone) + }() + + r := fastLeaseRunner(t, kube) + ctx, cancel := context.WithTimeout(context.Background(), 500*time.Millisecond) + defer cancel() + start := time.Now() + err := r.Run(ctx, func(context.Context) error { + t.Error("f must not run while another replica leads") + return nil + }) + + assert.NoError(t, err) + assert.Less(t, time.Since(start), 2*time.Second) + assert.Equal(t, other.lock.Identity(), leaseHolder(t, kube)) +} + +func TestLeaseRunnerReleasesALeaseAcquiredWhileStopping(t *testing.T) { + kube := fakekubernetes.NewClientset() + r := fastLeaseRunner(t, kube) + ctx, cancel := context.WithCancel(context.Background()) + kube.PrependReactor("create", "leases", func(k8stesting.Action) (bool, runtime.Object, error) { + cancel() + return false, nil, nil + }) + + err := r.Run(ctx, func(runCtx context.Context) error { + <-runCtx.Done() + return nil + }) + + assert.NoError(t, err) + assert.Empty(t, leaseHolder(t, kube)) +} + +func TestLeaseRunnerDoesNotReleaseALostLease(t *testing.T) { + kube := fakekubernetes.NewClientset() + r := fastLeaseRunner(t, kube) + var leading atomic.Bool + var releases atomic.Int32 + kube.PrependReactor("update", "leases", func(action k8stesting.Action) (bool, runtime.Object, error) { + lease := action.(k8stesting.UpdateAction).GetObject().(*coordinationv1.Lease) + if lease.Spec.HolderIdentity == nil || *lease.Spec.HolderIdentity == "" { + releases.Add(1) + } + if leading.Load() { + return true, nil, apierrors.NewServiceUnavailable("down") + } + return false, nil, nil + }) + + err := r.Run(context.Background(), func(runCtx context.Context) error { + leading.Store(true) + <-runCtx.Done() + return nil + }) + + assert.ErrorIs(t, err, errLeadershipLost) + assert.Zero(t, releases.Load()) +} + +func TestLeaseRunnerKeepsGoingWhenReleaseFails(t *testing.T) { + kube := fakekubernetes.NewClientset() + r := fastLeaseRunner(t, kube) + var stopped atomic.Bool + kube.PrependReactor("update", "leases", func(k8stesting.Action) (bool, runtime.Object, error) { + if stopped.Load() { + return true, nil, apierrors.NewServiceUnavailable("down") + } + return false, nil, nil + }) + + err := r.Run(context.Background(), func(context.Context) error { + stopped.Store(true) + return nil + }) + + assert.NoError(t, err) + assert.Equal(t, r.lock.Identity(), leaseHolder(t, kube), "the lease is left to expire") +} + +func TestLeaseRunnerLeavesALeaseTakenOverByAnotherReplica(t *testing.T) { + kube := fakekubernetes.NewClientset() + r := fastLeaseRunner(t, kube) + // The fake client has no optimistic locking; reject this runner's renewals + // once the other replica holds the lease, as the API server would. + var takenOver atomic.Bool + kube.PrependReactor("update", "leases", func(action k8stesting.Action) (bool, runtime.Object, error) { + lease := action.(k8stesting.UpdateAction).GetObject().(*coordinationv1.Lease) + if takenOver.Load() && ptr.Deref(lease.Spec.HolderIdentity, "") == r.lock.Identity() { + return true, nil, apierrors.NewConflict(coordinationv1.Resource("leases"), lockKey, errors.New("taken over")) + } + return false, nil, nil + }) + + err := r.Run(context.Background(), func(context.Context) error { + takenOver.Store(true) + lease, err := kube.CoordinationV1().Leases("ns").Get(context.Background(), lockKey, metav1.GetOptions{}) + require.NoError(t, err) + lease.Spec.HolderIdentity = ptr.To("other") + _, err = kube.CoordinationV1().Leases("ns").Update(context.Background(), lease, metav1.UpdateOptions{}) + require.NoError(t, err) + return nil + }) + + assert.NoError(t, err) + assert.Equal(t, "other", leaseHolder(t, kube)) +} From 6163d62f4c0a62c8d541990712cbcf0e7e09c1ca Mon Sep 17 00:00:00 2001 From: Michael Strassberger Date: Sat, 26 Sep 2026 01:21:03 +0200 Subject: [PATCH 13/24] fix: SentinelCheckQuorum NOQUORUM dead code + MakeSlaveOfWithPort wrong port (#128) SentinelCheckQuorum: real Sentinel's CKQUORUM NOQUORUM outcome comes back as a RESP error, not a successful string reply - so the intended NOQUORUM handling (parsing "(error)"/"NOQUORUM" out of a successful result) could never execute; callers only ever saw the raw driver error. NOQUORUM is now classified from err.Error() directly, restoring the intended "quorum Not available" message and NOQUORUM metrics tag. MakeSlaveOfWithPort(ip, masterIP, masterPort, password) connected to the *target* redis instance at (ip, masterPort) - using the master's port to reach the target, with no parameter for the target's own port. This only worked because every caller except SetExternalMasterOnAll happens to pass the same port for target and master (both come from this RedisFailover's own spec.redis.port). SetExternalMasterOnAll's externally-supplied bootstrap master port can genuinely differ, in which case the client silently connected to the wrong instance (or, when target and master share an IP, issued SLAVEOF to the master itself) instead of failing loudly. MakeSlaveOfWithPort now takes an explicit port for the target, separate from masterPort; all call sites, the mock, and MakeSlaveOf updated accordingly. Both found and confirmed against real Redis/Sentinel 7.0.15 while writing coverage tests for this file. Co-authored-by: Claude --- mocks/service/redis/Client.go | 10 +- .../redisfailover/service/demotion_test.go | 6 +- operator/redisfailover/service/heal.go | 12 +- operator/redisfailover/service/heal_test.go | 26 ++-- service/redis/client.go | 36 +++-- service/redis/client_test.go | 131 ++++++++---------- 6 files changed, 108 insertions(+), 113 deletions(-) diff --git a/mocks/service/redis/Client.go b/mocks/service/redis/Client.go index 8498b61f6..388d04a80 100644 --- a/mocks/service/redis/Client.go +++ b/mocks/service/redis/Client.go @@ -182,13 +182,13 @@ func (_m *Client) MakeSlaveOf(ip string, masterIP string, password string) error return r0 } -// MakeSlaveOfWithPort provides a mock function with given fields: ip, masterIP, masterPort, password -func (_m *Client) MakeSlaveOfWithPort(ip string, masterIP string, masterPort string, password string) error { - ret := _m.Called(ip, masterIP, masterPort, password) +// MakeSlaveOfWithPort provides a mock function with given fields: ip, port, masterIP, masterPort, password +func (_m *Client) MakeSlaveOfWithPort(ip string, port string, masterIP string, masterPort string, password string) error { + ret := _m.Called(ip, port, masterIP, masterPort, password) var r0 error - if rf, ok := ret.Get(0).(func(string, string, string, string) error); ok { - r0 = rf(ip, masterIP, masterPort, password) + if rf, ok := ret.Get(0).(func(string, string, string, string, string) error); ok { + r0 = rf(ip, port, masterIP, masterPort, password) } else { r0 = ret.Error(0) } diff --git a/operator/redisfailover/service/demotion_test.go b/operator/redisfailover/service/demotion_test.go index e448db270..571dee843 100644 --- a/operator/redisfailover/service/demotion_test.go +++ b/operator/redisfailover/service/demotion_test.go @@ -147,8 +147,8 @@ func TestSetMasterOnAllDisconnectsOnlyTheDemotedMaster(t *testing.T) { ms.On("UpdatePodLabels", namespace, "old-master", slaveRoleLabel).Once().Return(nil) mr := &mRedisService.Client{} mr.On("IsMaster", "0.0.0.0", "0", "").Return(true, nil) - mr.On("MakeSlaveOfWithPort", "1.1.1.1", "0.0.0.0", "0", "").Once().Return(nil) - mr.On("MakeSlaveOfWithPort", "2.2.2.2", "0.0.0.0", "0", "").Once().Return(nil) + mr.On("MakeSlaveOfWithPort", "1.1.1.1", "0", "0.0.0.0", "0", "").Once().Return(nil) + mr.On("MakeSlaveOfWithPort", "2.2.2.2", "0", "0.0.0.0", "0", "").Once().Return(nil) var disconnects []string healer := rfservice.NewRedisFailoverHealer(ms, mr, log.DummyLogger{}, @@ -173,7 +173,7 @@ func TestSetExternalMasterOnAllNeverDisconnectsClients(t *testing.T) { ms := &mK8SService.Services{} ms.On("GetStatefulSetPods", namespace, rfservice.GetRedisName(rf)).Return(pods, nil) mr := &mRedisService.Client{} - mr.On("MakeSlaveOfWithPort", mock.Anything, "5.5.5.5", "6379", "").Return(nil) + mr.On("MakeSlaveOfWithPort", mock.Anything, "0", "5.5.5.5", "6379", "").Return(nil) var disconnects []string healer := rfservice.NewRedisFailoverHealer(ms, mr, log.DummyLogger{}, diff --git a/operator/redisfailover/service/heal.go b/operator/redisfailover/service/heal.go index de8214ebf..644416b51 100644 --- a/operator/redisfailover/service/heal.go +++ b/operator/redisfailover/service/heal.go @@ -130,7 +130,7 @@ func (r *RedisFailoverHealer) SetOldestAsMaster(rf *redisfailoverv1.RedisFailove newMasterIP = pod.Status.PodIP } else { r.logger.Infof("Making pod %s slave of %s", pod.Name, newMasterIP) - if err := r.redisClient.MakeSlaveOfWithPort(pod.Status.PodIP, newMasterIP, port, password); err != nil { + if err := r.redisClient.MakeSlaveOfWithPort(pod.Status.PodIP, port, newMasterIP, port, password); err != nil { r.logger.WithField("redisfailover", rf.Name).WithField("namespace", rf.Namespace).Errorf("Make slave failed, slave pod ip: %s, master ip: %s, error: %v", pod.Status.PodIP, newMasterIP, err) } @@ -199,7 +199,7 @@ func (r *RedisFailoverHealer) SetMasterOnAll(masterIP string, rf *redisfailoverv continue } r.logger.WithField("redisfailover", rf.Name).WithField("namespace", rf.Namespace).Infof("Making pod %s slave of %s", pod.Name, masterIP) - if err := r.redisClient.MakeSlaveOfWithPort(pod.Status.PodIP, masterIP, port, password); err != nil { + if err := r.redisClient.MakeSlaveOfWithPort(pod.Status.PodIP, port, masterIP, port, password); err != nil { // The pod is unreachable - typically the old master on a downed // node. Skip it and keep repointing the reachable slaves instead // of aborting; it will re-sync via sentinel once its node is back. @@ -229,9 +229,13 @@ func (r *RedisFailoverHealer) SetExternalMasterOnAll(masterIP, masterPort string return err } + // The target pods' own port, which is not necessarily masterPort - the + // external bootstrap master can be configured on a different port than + // this RedisFailover's own Redis pods. + port := getRedisPort(rf.Spec.Redis.Port) for _, pod := range ssp.Items { r.logger.WithField("redisfailover", rf.Name).WithField("namespace", rf.Namespace).Infof("Making pod %s slave of %s:%s", pod.Name, masterIP, masterPort) - if err := r.redisClient.MakeSlaveOfWithPort(pod.Status.PodIP, masterIP, masterPort, password); err != nil { + if err := r.redisClient.MakeSlaveOfWithPort(pod.Status.PodIP, port, masterIP, masterPort, password); err != nil { return err } @@ -358,7 +362,7 @@ func (r *RedisFailoverHealer) PromoteBestReplica(newMasterIP string, rf *redisfa r.logger.WithField("redisfailover", rf.Name).WithField("namespace", rf.Namespace). Infof("Making pod %s slave of %s", rp.Name, newMasterIP) - if err := r.redisClient.MakeSlaveOfWithPort(rp.Status.PodIP, newMasterIP, port, password); err != nil { + if err := r.redisClient.MakeSlaveOfWithPort(rp.Status.PodIP, port, newMasterIP, port, password); err != nil { r.logger.WithField("redisfailover", rf.Name).WithField("namespace", rf.Namespace). Errorf("Failed to make %s slave of %s: %v", rp.Status.PodIP, newMasterIP, err) reconcileErrs = append(reconcileErrs, err) diff --git a/operator/redisfailover/service/heal_test.go b/operator/redisfailover/service/heal_test.go index 8884a4c3c..c55f5c990 100644 --- a/operator/redisfailover/service/heal_test.go +++ b/operator/redisfailover/service/heal_test.go @@ -96,7 +96,7 @@ func TestSetOldestAsMasterMultiplePodsMakeSlaveOfError(t *testing.T) { ms.On("UpdatePodLabels", namespace, mock.AnythingOfType("string"), mock.Anything).Return(nil) mr := &mRedisService.Client{} mr.On("MakeMaster", "0.0.0.0", "0", "").Once().Return(nil) - mr.On("MakeSlaveOfWithPort", "1.1.1.1", "0.0.0.0", "0", "").Once().Return(errors.New("")) + mr.On("MakeSlaveOfWithPort", "1.1.1.1", "0", "0.0.0.0", "0", "").Once().Return(errors.New("")) healer := rfservice.NewRedisFailoverHealer(ms, mr, log.DummyLogger{}) @@ -129,7 +129,7 @@ func TestSetOldestAsMasterMultiplePods(t *testing.T) { ms.On("UpdatePodLabels", namespace, mock.AnythingOfType("string"), mock.Anything).Return(nil) mr := &mRedisService.Client{} mr.On("MakeMaster", "0.0.0.0", "0", "").Once().Return(nil) - mr.On("MakeSlaveOfWithPort", "1.1.1.1", "0.0.0.0", "0", "").Once().Return(nil) + mr.On("MakeSlaveOfWithPort", "1.1.1.1", "0", "0.0.0.0", "0", "").Once().Return(nil) healer := rfservice.NewRedisFailoverHealer(ms, mr, log.DummyLogger{}) @@ -172,7 +172,7 @@ func TestSetOldestAsMasterOrdering(t *testing.T) { ms.On("UpdatePodLabels", namespace, mock.AnythingOfType("string"), mock.Anything).Return(nil) mr := &mRedisService.Client{} mr.On("MakeMaster", "1.1.1.1", "0", "").Once().Return(nil) - mr.On("MakeSlaveOfWithPort", "0.0.0.0", "1.1.1.1", "0", "").Once().Return(nil) + mr.On("MakeSlaveOfWithPort", "0.0.0.0", "0", "1.1.1.1", "0", "").Once().Return(nil) healer := rfservice.NewRedisFailoverHealer(ms, mr, log.DummyLogger{}) @@ -277,9 +277,9 @@ func TestSetMasterOnAllMakeSlaveOfErrorIsSkipped(t *testing.T) { ms.On("UpdatePodLabels", namespace, mock.AnythingOfType("string"), mock.Anything).Return(nil) mr := &mRedisService.Client{} mr.On("IsMaster", "0.0.0.0", "0", "").Return(true, nil) - mr.On("MakeSlaveOfWithPort", "1.1.1.1", "0.0.0.0", "0", "").Once().Return(errors.New("i/o timeout")) + mr.On("MakeSlaveOfWithPort", "1.1.1.1", "0", "0.0.0.0", "0", "").Once().Return(errors.New("i/o timeout")) // The reachable slave must still be repointed even though the previous pod failed. - mr.On("MakeSlaveOfWithPort", "2.2.2.2", "0.0.0.0", "0", "").Once().Return(nil) + mr.On("MakeSlaveOfWithPort", "2.2.2.2", "0", "0.0.0.0", "0", "").Once().Return(nil) healer := rfservice.NewRedisFailoverHealer(ms, mr, log.DummyLogger{}) @@ -313,7 +313,7 @@ func TestSetMasterOnAll(t *testing.T) { ms.On("UpdatePodLabels", namespace, mock.AnythingOfType("string"), mock.Anything).Return(nil) mr := &mRedisService.Client{} mr.On("IsMaster", "0.0.0.0", "0", "").Return(true, nil) - mr.On("MakeSlaveOfWithPort", "1.1.1.1", "0.0.0.0", "0", "").Once().Return(nil) + mr.On("MakeSlaveOfWithPort", "1.1.1.1", "0", "0.0.0.0", "0", "").Once().Return(nil) healer := rfservice.NewRedisFailoverHealer(ms, mr, log.DummyLogger{}) @@ -376,7 +376,7 @@ func TestSetMasterOnAllSlaveAlreadyLabeled(t *testing.T) { ms.On("GetStatefulSetPods", namespace, rfservice.GetRedisName(rf)).Once().Return(pods, nil) mr := &mRedisService.Client{} mr.On("IsMaster", "0.0.0.0", "0", "").Return(true, nil) - mr.On("MakeSlaveOfWithPort", "1.1.1.1", "0.0.0.0", "0", "").Once().Return(nil) + mr.On("MakeSlaveOfWithPort", "1.1.1.1", "0", "0.0.0.0", "0", "").Once().Return(nil) healer := rfservice.NewRedisFailoverHealer(ms, mr, log.DummyLogger{}) @@ -435,12 +435,12 @@ func TestSetExternalMasterOnAll(t *testing.T) { mr := &mRedisService.Client{} if !expectError { - mr.On("MakeSlaveOfWithPort", "0.0.0.0", "5.5.5.5", "6379", "").Once().Return(nil) + mr.On("MakeSlaveOfWithPort", "0.0.0.0", "0", "5.5.5.5", "6379", "").Once().Return(nil) if test.errorOnMakeSlaveOf { expectError = true - mr.On("MakeSlaveOfWithPort", "1.1.1.1", "5.5.5.5", "6379", "").Once().Return(errors.New("")) + mr.On("MakeSlaveOfWithPort", "1.1.1.1", "0", "5.5.5.5", "6379", "").Once().Return(errors.New("")) } else { - mr.On("MakeSlaveOfWithPort", "1.1.1.1", "5.5.5.5", "6379", "").Once().Return(nil) + mr.On("MakeSlaveOfWithPort", "1.1.1.1", "0", "5.5.5.5", "6379", "").Once().Return(nil) } } @@ -491,7 +491,7 @@ func TestPromoteBestReplicaSuccess(t *testing.T) { mr := &mRedisService.Client{} mr.On("MakeMaster", newMasterIP, "0", "").Once().Return(nil) - mr.On("MakeSlaveOfWithPort", replicaIP, newMasterIP, "0", "").Once().Return(nil) + mr.On("MakeSlaveOfWithPort", replicaIP, "0", newMasterIP, "0", "").Once().Return(nil) healer := rfservice.NewRedisFailoverHealer(ms, mr, log.DummyLogger{}) @@ -533,7 +533,7 @@ func TestPromoteBestReplicaReplicaRepointerFails(t *testing.T) { mr := &mRedisService.Client{} mr.On("MakeMaster", newMasterIP, "0", "").Once().Return(nil) - mr.On("MakeSlaveOfWithPort", replicaIP, newMasterIP, "0", "").Once().Return(errors.New("replica repoint failed")) + mr.On("MakeSlaveOfWithPort", replicaIP, "0", newMasterIP, "0", "").Once().Return(errors.New("replica repoint failed")) healer := rfservice.NewRedisFailoverHealer(ms, mr, log.DummyLogger{}) @@ -577,7 +577,7 @@ func TestPromoteBestReplicaLabelUpdateFails(t *testing.T) { mr := &mRedisService.Client{} mr.On("MakeMaster", newMasterIP, "0", "").Once().Return(nil) - mr.On("MakeSlaveOfWithPort", replicaIP, newMasterIP, "0", "").Once().Return(nil) + mr.On("MakeSlaveOfWithPort", replicaIP, "0", newMasterIP, "0", "").Once().Return(nil) healer := rfservice.NewRedisFailoverHealer(ms, mr, log.DummyLogger{}) diff --git a/service/redis/client.go b/service/redis/client.go index c5dcdf0e5..3feab37ed 100644 --- a/service/redis/client.go +++ b/service/redis/client.go @@ -37,7 +37,7 @@ type Client interface { MonitorRedisWithPort(ip, monitor, port, quorum, password string) error MakeMaster(ip, port, password string) error MakeSlaveOf(ip, masterIP, password string) error - MakeSlaveOfWithPort(ip, masterIP, masterPort, password string) error + MakeSlaveOfWithPort(ip, port, masterIP, masterPort, password string) error DisconnectClients(ip, port, password string) error GetSentinelMonitor(ip string) (string, string, error) SetCustomSentinelConfig(ip string, configs []string) error @@ -321,12 +321,17 @@ func (c *client) MakeMaster(ip string, port string, password string) error { } func (c *client) MakeSlaveOf(ip, masterIP, password string) error { - return c.MakeSlaveOfWithPort(ip, masterIP, redisPort, password) + return c.MakeSlaveOfWithPort(ip, redisPort, masterIP, redisPort, password) } -func (c *client) MakeSlaveOfWithPort(ip, masterIP, masterPort, password string) error { +// MakeSlaveOfWithPort reconfigures the Redis instance at ip:port to become a +// replica of masterIP:masterPort. port is the target's own listening port - +// it must not be assumed to equal masterPort, since a target and its master +// can be configured with different ports (e.g. Bootstrapping mode's +// externally supplied master port). +func (c *client) MakeSlaveOfWithPort(ip, port, masterIP, masterPort, password string) error { options := &rediscli.Options{ - Addr: net.JoinHostPort(ip, masterPort), // this is IP and Port for the RedisFailover redis + Addr: net.JoinHostPort(ip, port), Password: password, DB: 0, } @@ -514,6 +519,17 @@ func (c *client) SentinelCheckQuorum(ip string) error { res, err := cmd.Result() if err != nil { + // SENTINEL CKQUORUM's NOQUORUM outcome comes back over the wire as a + // genuine RESP error whose text starts with "NOQUORUM", not as a + // successful string reply - so it has to be classified here, before + // the success-path string parsing below (which can only ever see + // the "OK ..." success message, since res is empty whenever err is + // non-nil). + if strings.Contains(err.Error(), "NOQUORUM") { + log.Debugf("SentinelCheckQuorum: quorum not available: %s", err.Error()) + c.metricsRecorder.RecordRedisOperation(metrics.KIND_SENTINEL, ip, metrics.CHECK_SENTINEL_QUORUM, metrics.SUCCESS, "NOQUORUM") + return fmt.Errorf("quorum Not available") + } log.Warnf("Unable to get result for CKQUORUM comand") c.metricsRecorder.RecordRedisOperation(metrics.KIND_SENTINEL, ip, metrics.CHECK_SENTINEL_QUORUM, metrics.FAIL, getRedisError(err)) return err @@ -521,18 +537,8 @@ func (c *client) SentinelCheckQuorum(ip string) error { log.Debugf("SentinelCheckQuorum cmd result: %s", res) s := strings.Split(res, " ") status := s[0] - quorum := s[1] - - if status == "" { - log.Errorf("quorum command result unexpected output") - c.metricsRecorder.RecordRedisOperation(metrics.KIND_SENTINEL, ip, metrics.CHECK_SENTINEL_QUORUM, metrics.FAIL, "quorum command result unexpected output") - return fmt.Errorf("quorum command result unexpected output") - } - if status == "(error)" && quorum == "NOQUORUM" { - c.metricsRecorder.RecordRedisOperation(metrics.KIND_SENTINEL, ip, metrics.CHECK_SENTINEL_QUORUM, metrics.SUCCESS, "NOQUORUM") - return fmt.Errorf("quorum Not available") - } else if status == "OK" { + if status == "OK" { c.metricsRecorder.RecordRedisOperation(metrics.KIND_SENTINEL, ip, metrics.CHECK_SENTINEL_QUORUM, metrics.SUCCESS, "QUORUM") return nil } else { diff --git a/service/redis/client_test.go b/service/redis/client_test.go index 6880eed6b..74817a1c8 100644 --- a/service/redis/client_test.go +++ b/service/redis/client_test.go @@ -6,12 +6,14 @@ package redis // though it is the entire wire-protocol layer the operator uses to talk to // Redis and Sentinel. // -// Two behaviours here were verified empirically against real Redis 7.0.15 -// rather than assumed, per the investigation that prompted this test suite: -// - SENTINEL CKQUORUM's NOQUORUM outcome (see TestSentinelCheckQuorum_NoQuorum). +// Several behaviours here were verified empirically against real Redis +// 7.0.15 rather than assumed, per the investigation that prompted this test +// suite - each surfaced a real bug in client.go, fixed in the same change +// as its test: // - CONFIG SET aclfile's immutability (see TestSetCustomRedisConfig_ACLFile). -// The aclfile finding surfaced a real bug in client.go, fixed in the same -// change as this test (see the comment on TestSetCustomRedisConfig_ACLFile). +// - SENTINEL CKQUORUM's NOQUORUM outcome (see TestSentinelCheckQuorum_NoQuorum). +// - MakeSlaveOfWithPort connecting to the master's port instead of the +// target's own port (see TestMakeSlaveOfWithPort_MismatchedTargetPort). import ( "context" @@ -244,18 +246,15 @@ func TestMakeMaster_ConnectionError(t *testing.T) { assert.Error(t, err) } -// TestMakeSlaveOfWithPort_ConnectionError points at a port nothing is -// listening on to exercise MakeSlaveOfWithPort's own connection-error -// branch. It intentionally does not attempt to also reach a real master - -// see TestMakeSlaveOfWithPort_MismatchedTargetPort for why the address -// MakeSlaveOfWithPort actually connects to is (ip, masterPort), not -// (ip, targetPort). +// TestMakeSlaveOfWithPort_ConnectionError points the target's own port at a +// port nothing is listening on to exercise MakeSlaveOfWithPort's own +// connection-error branch. func TestMakeSlaveOfWithPort_ConnectionError(t *testing.T) { port, err := findFreePort() require.NoError(t, err) c := newTestClient() - err = c.MakeSlaveOfWithPort(testLoopbackIP, "10.0.0.1", strconv.Itoa(port), "") + err = c.MakeSlaveOfWithPort(testLoopbackIP, strconv.Itoa(port), "10.0.0.1", "6379", "") assert.Error(t, err) } @@ -310,7 +309,7 @@ func TestMakeSlaveOfWithPort_SamePortDifferentIP(t *testing.T) { require.NoError(t, err) require.True(t, isMaster) - err = c.MakeSlaveOfWithPort(a.IP, b.IP, strconv.Itoa(b.Port), "") + err = c.MakeSlaveOfWithPort(a.IP, strconv.Itoa(a.Port), b.IP, strconv.Itoa(b.Port), "") require.NoError(t, err) waitForCondition(t, 5*time.Second, func() bool { @@ -372,7 +371,7 @@ func TestMakeSlaveOfWithPort_LeavesClientConnectionsAlone(t *testing.T) { appConn := dialRaw(t, a.Addr(), "PING\r\n", "+PONG\r\n") - require.NoError(t, c.MakeSlaveOfWithPort(a.IP, b.IP, strconv.Itoa(b.Port), "")) + require.NoError(t, c.MakeSlaveOfWithPort(a.IP, strconv.Itoa(a.Port), b.IP, strconv.Itoa(b.Port), "")) require.NoError(t, appConn.SetDeadline(time.Now().Add(2*time.Second))) _, err = appConn.Write([]byte("PING\r\n")) @@ -450,55 +449,47 @@ func TestCloseClient_LogsCloseError(t *testing.T) { assert.NotPanics(t, func() { closeClient(rClient) }) } -// TestMakeSlaveOfWithPort_MismatchedTargetPort documents a real bug found -// while building this test suite (not fixed here, per instructions - see -// the task summary for the full report): -// -// MakeSlaveOfWithPort(ip, masterIP, masterPort, password) connects to the -// *target* redis instance (the one it's about to issue SLAVEOF against) at -// `net.JoinHostPort(ip, masterPort)` - i.e. it uses the *master's* port to -// reach the target, with no separate parameter for the target's own port. -// That's only correct when the target and the master happen to listen on -// the same port number (true in this operator's normal deployment model, -// where every managed Redis instance uses one shared configured port across -// different pod IPs - see the SamePortDifferentIP test above for that -// case). When the target's actual port differs from the master's port, this -// silently does the wrong thing rather than failing loudly: +// TestMakeSlaveOfWithPort_MismatchedTargetPort covers a real bug found while +// building this test suite: MakeSlaveOfWithPort used to have no separate +// parameter for the target's own port, and connected to the *target* at +// `net.JoinHostPort(ip, masterPort)` - i.e. it used the *master's* port to +// reach the target. That was only correct when the target and the master +// happened to share a port number (true in this operator's normal +// deployment model - see TestMakeSlaveOfWithPort_SamePortDifferentIP for +// that case). When the target's actual port differed from the master's +// port (reachable via Bootstrapping's externally supplied master port), +// this silently connected to the wrong instance rather than failing loudly. // -// Here `a` (the intended target) and `b` (the intended master) share the -// same IP (both loopback) but listen on *different* ports. Calling -// MakeSlaveOfWithPort(a.IP, b.IP, b.Port, "") computes the connection -// address as (a.IP, b.Port) - since a.IP == b.IP, that address actually -// belongs to `b`'s own server, not `a`'s. The client ends up connected to -// `b` and issues `SLAVEOF b.IP b.Port` *to b itself*. Real Redis accepts -// `SLAVEOF ` without error (confirmed manually against 7.0.15) and -// becomes a slave of itself with master_link_status permanently "down" - -// so the call returns a misleading nil error, `b` is left in a broken -// self-replicating state, and `a` (the actual intended target) is never -// contacted at all and remains an untouched master. +// MakeSlaveOfWithPort now takes an explicit port for the target, separate +// from masterPort. Here `a` (the target) and `b` (the master) share the +// same IP (both loopback) but listen on *different* ports - exactly the +// case that used to misroute the connection to `b` itself. This asserts +// the fix: `a` is correctly reconfigured as a replica of `b`, and `b` is +// left untouched as a master. func TestMakeSlaveOfWithPort_MismatchedTargetPort(t *testing.T) { requireRedisServer(t) a := startRedisProcess(t) b := startRedisProcess(t) c := newTestClient() - require.NotEqual(t, a.Port, b.Port, "test setup requires distinct ports to reproduce the bug") + require.NotEqual(t, a.Port, b.Port, "test setup requires distinct ports to exercise the target/master port distinction") - err := c.MakeSlaveOfWithPort(a.IP, b.IP, strconv.Itoa(b.Port), "") - require.NoError(t, err, "the call itself does not surface an error - that's the point of this bug") - - // `a`, the intended target, was never actually reached. - masterOf, err := c.GetSlaveOf(a.IP, strconv.Itoa(a.Port), "") + err := c.MakeSlaveOfWithPort(a.IP, strconv.Itoa(a.Port), b.IP, strconv.Itoa(b.Port), "") require.NoError(t, err) - assert.Empty(t, masterOf, "the intended target should be untouched, still a master") - // `b` was told to replicate from itself instead. - waitForCondition(t, 2*time.Second, func() bool { - masterOf, err := c.GetSlaveOf(b.IP, strconv.Itoa(b.Port), "") + // `a`, the actual target, should be reconfigured as a replica of `b`. + waitForCondition(t, 5*time.Second, func() bool { + masterOf, err := c.GetSlaveOf(a.IP, strconv.Itoa(a.Port), "") return err == nil && masterOf == b.IP }) + masterOf, err := c.GetSlaveOf(a.IP, strconv.Itoa(a.Port), "") + require.NoError(t, err) + assert.Equal(t, b.IP, masterOf, "the target should be reconfigured as a replica of the intended master") + + // `b`, the master, should be untouched - still its own master, not + // pointed at itself or at `a`. masterOf, err = c.GetSlaveOf(b.IP, strconv.Itoa(b.Port), "") require.NoError(t, err) - assert.Equal(t, b.IP, masterOf, "the master ends up pointed at itself instead of the target being reconfigured") + assert.Empty(t, masterOf, "the master should be untouched, still a master") } // TestMakeSlaveOf covers the MakeSlaveOf convenience wrapper, which hardcodes @@ -1003,30 +994,23 @@ func TestSentinelCheckQuorum_OK(t *testing.T) { assert.NoError(t, err) } -// TestSentinelCheckQuorum_NoQuorum documents a real bug found while building -// this test suite (not fixed here, per instructions - see the task summary -// for the full report): -// -// Real Sentinel's CKQUORUM reply for the NOQUORUM case comes back as a RESP -// *error* reply (confirmed here against real Redis 7.0.15: `redis-cli -// --no-raw` shows `(error) NOQUORUM ...`), not as a successful string reply -// whose text happens to start with the literal characters "(error)". Since -// go-redis's SentinelClient.CkQuorum uses a StringCmd, that RESP error -// becomes cmd.Err()/the err returned by cmd.Result() - so -// client.go's SentinelCheckQuorum takes its `if err != nil { ... return err -// }` branch immediately. +// TestSentinelCheckQuorum_NoQuorum covers a real bug found while building +// this test suite: real Sentinel's CKQUORUM reply for the NOQUORUM case +// comes back as a RESP *error* reply (confirmed here against real Redis +// 7.0.15: `redis-cli --no-raw` shows `(error) NOQUORUM ...`), not as a +// successful string reply whose text happens to start with the literal +// characters "(error)". Since go-redis's SentinelClient.CkQuorum uses a +// StringCmd, that RESP error becomes cmd.Err()/the err returned by +// cmd.Result() - so client.go's SentinelCheckQuorum used to take its `if +// err != nil { ... return err }` branch immediately, and its own intended +// NOQUORUM handling (checking `s := strings.Split(res, " ")` for a literal +// "(error)"/"NOQUORUM" pair) could never execute, since res is only +// non-empty when err is nil. // -// The subsequent code - `s := strings.Split(res, " ")` followed by `if -// status == "(error)" && quorum == "NOQUORUM"` - can therefore never -// execute on a real NOQUORUM response: `res` is only non-empty when err is -// nil, i.e. only for the OK case, so status will always be "OK" whenever -// that branch is reached. In other words, the code's own intended NOQUORUM -// handling (returning `errors.New("quorum Not available")`) is dead code; -// what actually gets returned to callers today is the raw driver error -// (whose message happens to also mention NOQUORUM, so callers checking -// err != nil for failure still work correctly - this is a -// dead-code/messaging bug, not a functional regression for existing -// callers as far as we can tell). +// SentinelCheckQuorum now classifies NOQUORUM directly from err.Error() +// (which real Sentinel prefixes with "NOQUORUM ...") before the +// success-path string parsing, restoring the intended "quorum Not +// available" message and NOQUORUM metrics tag. // --------------------------------------------------------------------- // getRedisError // @@ -1066,6 +1050,7 @@ func TestSentinelCheckQuorum_NoQuorum(t *testing.T) { err = c.SentinelCheckQuorum(env.sentinel.IP) require.Error(t, err, "CKQUORUM should fail when quorum exceeds the number of known sentinels") + assert.Equal(t, "quorum Not available", err.Error(), "the intended NOQUORUM message should be reachable, not just the raw driver error") } // TestSentinelFunctions_SentinelUnreachable exercises the connection-error From de80035f55cb72c70faea3040c242b7ce01bf03d Mon Sep 17 00:00:00 2001 From: Michael Strassberger Date: Sat, 26 Sep 2026 01:32:53 +0200 Subject: [PATCH 14/24] feat(exporter): make the exporter metrics port configurable (dnse #30) (#144) * feat(exporter): make the exporter metrics port configurable (#30) The sentinel exporter port was hardcoded to 9355 (and the redis exporter to 9121), so the listen port, container port and metrics service port could not be changed. Add an optional Exporter.port field that overrides the listen/container/service port for both the redis and sentinel exporters, defaulting to the previous values when left at 0. Regenerated CRD manifests for the new field. Refs upstream spotahome/redis-operator#531. Co-authored-by: Binh Nguyen (cherry picked from commit 00c37f260682bf9ee070c97eab629f816ba2bfe7) * Make the redis exporter listen on a custom port and bound exporter ports A custom spec.redis.exporter.port only changed the declared container and Service port; the exporter itself kept listening on 9121. Set REDIS_EXPORTER_WEB_LISTEN_ADDRESS when a custom port is set, like the sentinel exporter does. Reject exporter ports outside 0-65535 in the CRD schema and in Validate(). Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01G4mmyo3sqLwf23m5FE2nBE --------- Co-authored-by: Binh Nguyen <37066217+binhnguyenduc@users.noreply.github.com> Co-authored-by: Binh Nguyen Co-authored-by: Claude --- api/redisfailover/v1/types.go | 6 ++ api/redisfailover/v1/validate.go | 6 ++ api/redisfailover/v1/validate_test.go | 17 ++++++ ...atabases.spotahome.com_redisfailovers.yaml | 18 ++++++ ...atabases.spotahome.com_redisfailovers.yaml | 18 ++++++ ...atabases.spotahome.com_redisfailovers.yaml | 18 ++++++ operator/redisfailover/service/generator.go | 39 +++++++++++-- .../redisfailover/service/generator_test.go | 57 +++++++++++++++++++ 8 files changed, 173 insertions(+), 6 deletions(-) diff --git a/api/redisfailover/v1/types.go b/api/redisfailover/v1/types.go index 740a1c08e..e48cb6458 100644 --- a/api/redisfailover/v1/types.go +++ b/api/redisfailover/v1/types.go @@ -136,6 +136,12 @@ type Exporter struct { Args []string `json:"args,omitempty"` Env []corev1.EnvVar `json:"env,omitempty"` Resources *corev1.ResourceRequirements `json:"resources,omitempty"` + // Port the exporter sidecar listens on and the metrics service exposes. + // Defaults to 9121 for the redis exporter and 9355 for the sentinel exporter + // when left as 0. + // +kubebuilder:validation:Minimum=0 + // +kubebuilder:validation:Maximum=65535 + Port int32 `json:"port,omitempty"` } // SentinelConfigCopy defines the specification for the sentinel exporter diff --git a/api/redisfailover/v1/validate.go b/api/redisfailover/v1/validate.go index 2d54b7792..1dcd214d7 100644 --- a/api/redisfailover/v1/validate.go +++ b/api/redisfailover/v1/validate.go @@ -35,6 +35,12 @@ func (r *RedisFailover) Validate() error { } } + for name, port := range map[string]int32{"redis": r.Spec.Redis.Exporter.Port, "sentinel": r.Spec.Sentinel.Exporter.Port} { + if port < 0 || port > 65535 { + return fmt.Errorf("%s.exporter.port %d must be between 1 and 65535, or 0 for the default", name, port) + } + } + if r.Bootstrapping() { if r.Spec.BootstrapNode.Host == "" { return errors.New("BootstrapNode must include a host when provided") diff --git a/api/redisfailover/v1/validate_test.go b/api/redisfailover/v1/validate_test.go index 56bc79d9e..c66484616 100644 --- a/api/redisfailover/v1/validate_test.go +++ b/api/redisfailover/v1/validate_test.go @@ -187,3 +187,20 @@ func TestValidatePreservesExistingStatus(t *testing.T) { LastChanged: "2026-01-01T00:00:00Z", }, rf.Status) } + +func TestValidateExporterPort(t *testing.T) { + for _, port := range []int32{0, 1, 65535} { + rf := generateRedisFailover("test", nil) + rf.Spec.Redis.Exporter.Port = port + rf.Spec.Sentinel.Exporter.Port = port + assert.NoError(t, rf.Validate(), "port %d", port) + } + + rf := generateRedisFailover("test", nil) + rf.Spec.Redis.Exporter.Port = -1 + assert.EqualError(t, rf.Validate(), "redis.exporter.port -1 must be between 1 and 65535, or 0 for the default") + + rf = generateRedisFailover("test", nil) + rf.Spec.Sentinel.Exporter.Port = 65536 + assert.EqualError(t, rf.Validate(), "sentinel.exporter.port 65536 must be between 1 and 65535, or 0 for the default") +} diff --git a/charts/redisoperator/crds/databases.spotahome.com_redisfailovers.yaml b/charts/redisoperator/crds/databases.spotahome.com_redisfailovers.yaml index 578035e28..7d63967d5 100644 --- a/charts/redisoperator/crds/databases.spotahome.com_redisfailovers.yaml +++ b/charts/redisoperator/crds/databases.spotahome.com_redisfailovers.yaml @@ -2044,6 +2044,15 @@ spec: description: PullPolicy describes a policy for if/when to pull a container image type: string + port: + description: |- + Port the exporter sidecar listens on and the metrics service exposes. + Defaults to 9121 for the redis exporter and 9355 for the sentinel exporter + when left as 0. + format: int32 + maximum: 65535 + minimum: 0 + type: integer resources: description: ResourceRequirements describes the compute resource requirements. @@ -10343,6 +10352,15 @@ spec: description: PullPolicy describes a policy for if/when to pull a container image type: string + port: + description: |- + Port the exporter sidecar listens on and the metrics service exposes. + Defaults to 9121 for the redis exporter and 9355 for the sentinel exporter + when left as 0. + format: int32 + maximum: 65535 + minimum: 0 + type: integer resources: description: ResourceRequirements describes the compute resource requirements. diff --git a/manifests/databases.spotahome.com_redisfailovers.yaml b/manifests/databases.spotahome.com_redisfailovers.yaml index 578035e28..7d63967d5 100644 --- a/manifests/databases.spotahome.com_redisfailovers.yaml +++ b/manifests/databases.spotahome.com_redisfailovers.yaml @@ -2044,6 +2044,15 @@ spec: description: PullPolicy describes a policy for if/when to pull a container image type: string + port: + description: |- + Port the exporter sidecar listens on and the metrics service exposes. + Defaults to 9121 for the redis exporter and 9355 for the sentinel exporter + when left as 0. + format: int32 + maximum: 65535 + minimum: 0 + type: integer resources: description: ResourceRequirements describes the compute resource requirements. @@ -10343,6 +10352,15 @@ spec: description: PullPolicy describes a policy for if/when to pull a container image type: string + port: + description: |- + Port the exporter sidecar listens on and the metrics service exposes. + Defaults to 9121 for the redis exporter and 9355 for the sentinel exporter + when left as 0. + format: int32 + maximum: 65535 + minimum: 0 + type: integer resources: description: ResourceRequirements describes the compute resource requirements. diff --git a/manifests/kustomize/base/databases.spotahome.com_redisfailovers.yaml b/manifests/kustomize/base/databases.spotahome.com_redisfailovers.yaml index 578035e28..7d63967d5 100644 --- a/manifests/kustomize/base/databases.spotahome.com_redisfailovers.yaml +++ b/manifests/kustomize/base/databases.spotahome.com_redisfailovers.yaml @@ -2044,6 +2044,15 @@ spec: description: PullPolicy describes a policy for if/when to pull a container image type: string + port: + description: |- + Port the exporter sidecar listens on and the metrics service exposes. + Defaults to 9121 for the redis exporter and 9355 for the sentinel exporter + when left as 0. + format: int32 + maximum: 65535 + minimum: 0 + type: integer resources: description: ResourceRequirements describes the compute resource requirements. @@ -10343,6 +10352,15 @@ spec: description: PullPolicy describes a policy for if/when to pull a container image type: string + port: + description: |- + Port the exporter sidecar listens on and the metrics service exposes. + Defaults to 9121 for the redis exporter and 9355 for the sentinel exporter + when left as 0. + format: int32 + maximum: 65535 + minimum: 0 + type: integer resources: description: ResourceRequirements describes the compute resource requirements. diff --git a/operator/redisfailover/service/generator.go b/operator/redisfailover/service/generator.go index c803186a8..dd6765986 100644 --- a/operator/redisfailover/service/generator.go +++ b/operator/redisfailover/service/generator.go @@ -80,10 +80,11 @@ func generateSentinelService(rf *redisfailoverv1.RedisFailover, labels map[strin // The sentinel exporter sidecar listens on sentinelExporterPort, but without // a matching service port there is no way to scrape it through the service. if rf.Spec.Sentinel.Exporter.Enabled { + port := sentinelExporterListenPort(rf) svc.Spec.Ports = append(svc.Spec.Ports, corev1.ServicePort{ Name: "metrics", - Port: sentinelExporterPort, - TargetPort: intstr.FromInt(sentinelExporterPort), + Port: port, + TargetPort: intstr.FromInt(int(port)), Protocol: corev1.ProtocolTCP, }) } @@ -117,7 +118,7 @@ func generateRedisService(rf *redisfailoverv1.RedisFailover, labels map[string]s ClusterIP: corev1.ClusterIPNone, Ports: []corev1.ServicePort{ { - Port: exporterPort, + Port: redisExporterListenPort(rf), Protocol: corev1.ProtocolTCP, Name: exporterPortName, }, @@ -789,6 +790,24 @@ var exporterDefaultResourceRequirements = corev1.ResourceRequirements{ }, } +// redisExporterListenPort returns the port the redis exporter listens on, +// falling back to the built-in default when the spec leaves it unset. +func redisExporterListenPort(rf *redisfailoverv1.RedisFailover) int32 { + if p := rf.Spec.Redis.Exporter.Port; p != 0 { + return p + } + return exporterPort +} + +// sentinelExporterListenPort returns the port the sentinel exporter listens on, +// falling back to the built-in default when the spec leaves it unset. +func sentinelExporterListenPort(rf *redisfailoverv1.RedisFailover) int32 { + if p := rf.Spec.Sentinel.Exporter.Port; p != 0 { + return p + } + return sentinelExporterPort +} + func createRedisExporterContainer(rf *redisfailoverv1.RedisFailover) corev1.Container { resources := exporterDefaultResourceRequirements if rf.Spec.Redis.Exporter.Resources != nil { @@ -812,7 +831,7 @@ func createRedisExporterContainer(rf *redisfailoverv1.RedisFailover) corev1.Cont Ports: []corev1.ContainerPort{ { Name: "metrics", - ContainerPort: exporterPort, + ContainerPort: redisExporterListenPort(rf), Protocol: corev1.ProtocolTCP, }, }, @@ -821,6 +840,13 @@ func createRedisExporterContainer(rf *redisfailoverv1.RedisFailover) corev1.Cont redisEnv := getRedisExporterEnv(rf) container.Env = append(container.Env, redisEnv...) + // Only for a custom port, so default pod templates stay unchanged. + if rf.Spec.Redis.Exporter.Port != 0 { + container.Env = append(container.Env, corev1.EnvVar{ + Name: "REDIS_EXPORTER_WEB_LISTEN_ADDRESS", + Value: fmt.Sprintf("0.0.0.0:%d", rf.Spec.Redis.Exporter.Port), + }) + } return container } @@ -830,6 +856,7 @@ func createSentinelExporterContainer(rf *redisfailoverv1.RedisFailover) corev1.C if rf.Spec.Sentinel.Exporter.Resources != nil { resources = *rf.Spec.Sentinel.Exporter.Resources } + listenPort := sentinelExporterListenPort(rf) container := corev1.Container{ Name: sentinelExporterContainerName, Image: rf.Spec.Sentinel.Exporter.Image, @@ -845,7 +872,7 @@ func createSentinelExporterContainer(rf *redisfailoverv1.RedisFailover) corev1.C }, }, corev1.EnvVar{ Name: "REDIS_EXPORTER_WEB_LISTEN_ADDRESS", - Value: fmt.Sprintf("0.0.0.0:%[1]v", sentinelExporterPort), + Value: fmt.Sprintf("0.0.0.0:%[1]v", listenPort), }, corev1.EnvVar{ Name: "REDIS_ADDR", Value: "redis://127.0.0.1:26379", @@ -854,7 +881,7 @@ func createSentinelExporterContainer(rf *redisfailoverv1.RedisFailover) corev1.C Ports: []corev1.ContainerPort{ { Name: "metrics", - ContainerPort: sentinelExporterPort, + ContainerPort: listenPort, Protocol: corev1.ProtocolTCP, }, }, diff --git a/operator/redisfailover/service/generator_test.go b/operator/redisfailover/service/generator_test.go index 7a3f823c6..a502ce736 100644 --- a/operator/redisfailover/service/generator_test.go +++ b/operator/redisfailover/service/generator_test.go @@ -1162,10 +1162,17 @@ func TestSentinelServiceExporterPort(t *testing.T) { TargetPort: intstr.FromInt(9355), Protocol: corev1.ProtocolTCP, } + customMetricsPort := corev1.ServicePort{ + Name: "metrics", + Port: 19355, + TargetPort: intstr.FromInt(19355), + Protocol: corev1.ProtocolTCP, + } tests := []struct { name string exporterEnabled bool + exporterPort int32 expectedPorts []corev1.ServicePort }{ { @@ -1178,6 +1185,12 @@ func TestSentinelServiceExporterPort(t *testing.T) { exporterEnabled: true, expectedPorts: []corev1.ServicePort{sentinelPort, metricsPort}, }, + { + name: "exporter port override is honoured", + exporterEnabled: true, + exporterPort: 19355, + expectedPorts: []corev1.ServicePort{sentinelPort, customMetricsPort}, + }, } for _, test := range tests { @@ -1186,6 +1199,7 @@ func TestSentinelServiceExporterPort(t *testing.T) { rf := generateRF() rf.Spec.Sentinel.Exporter.Enabled = test.exporterEnabled + rf.Spec.Sentinel.Exporter.Port = test.exporterPort generatedService := corev1.Service{} @@ -3769,6 +3783,49 @@ func TestRedisExporterCustomResources(t *testing.T) { } } +func TestRedisExporterListensOnCustomPort(t *testing.T) { + tests := []struct { + name string + port int32 + wantPort int32 + wantListen string + }{ + {name: "default port leaves the listen address alone", wantPort: 9121}, + {name: "custom port sets the listen address", port: 19121, wantPort: 19121, wantListen: "0.0.0.0:19121"}, + } + + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + assert := assert.New(t) + rf := generateRF() + rf.Spec.Redis.Exporter.Enabled = true + rf.Spec.Redis.Exporter.Port = test.port + + var gotSS *appsv1.StatefulSet + ms := &mK8SService.Services{} + ms.On("CreateOrUpdatePodDisruptionBudget", namespace, mock.Anything).Once().Return(nil, nil) + ms.On("CreateOrUpdateStatefulSet", namespace, mock.Anything).Once().Run(func(args mock.Arguments) { + gotSS = args.Get(1).(*appsv1.StatefulSet) + }).Return(nil) + + client := rfservice.NewRedisFailoverKubeClient(ms, log.Dummy, metrics.Dummy) + assert.NoError(client.EnsureRedisStatefulset(rf, nil, []metav1.OwnerReference{})) + + if assert.NotNil(gotSS) && assert.Len(gotSS.Spec.Template.Spec.Containers, 2) { + exporter := gotSS.Spec.Template.Spec.Containers[1] + assert.Equal(test.wantPort, exporter.Ports[0].ContainerPort) + var listen string + for _, env := range exporter.Env { + if env.Name == "REDIS_EXPORTER_WEB_LISTEN_ADDRESS" { + listen = env.Value + } + } + assert.Equal(test.wantListen, listen) + } + }) + } +} + // --------------------------------------------------------------------------- // getRedisExporterEnv / envExists // --------------------------------------------------------------------------- From b234930bddb455123a50f5686838791574c8e328 Mon Sep 17 00:00:00 2001 From: Michael Strassberger Date: Sat, 26 Sep 2026 09:51:54 +0200 Subject: [PATCH 15/24] feat(heal): optionally protect the redis master from autoscaler eviction (dnse #35) (#147) * feat(heal): optionally protect the redis master from autoscaler eviction (#35) Add an opt-in redis.preventMasterEviction flag. When enabled, the operator annotates the current master pod with cluster-autoscaler.kubernetes.io/safe-to-evict=false and marks slaves true, so the cluster autoscaler will not drain the node running the master and force an avoidable failover. The annotation follows the master as the role moves, in both the checker and healer relabel paths. Adds a Pod.UpdatePodAnnotations service method (JSON merge patch) and the matching mock. Regenerated the CRD for the new field. Refs upstream spotahome/redis-operator#689. Co-authored-by: Binh Nguyen (cherry picked from commit 85bc2f31ce062167199aad586e0bdf8763d2f8d6) * Mark a demoted master evictable only after it is relabelled setSlaveLabel set safe-to-evict=true before changing the role label, so a failed relabel left a pod still labelled master, and still behind the master Service, that the autoscaler could evict. Relabel first and annotate afterwards. Also test UpdatePodAnnotations against the fake clientset. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01G4mmyo3sqLwf23m5FE2nBE * Test that a master is not relabelled when pinning it fails Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01G4mmyo3sqLwf23m5FE2nBE --------- Co-authored-by: Binh Nguyen <37066217+binhnguyenduc@users.noreply.github.com> Co-authored-by: Binh Nguyen Co-authored-by: Claude --- README.md | 7 + api/redisfailover/v1/types.go | 5 + ...atabases.spotahome.com_redisfailovers.yaml | 7 + ...atabases.spotahome.com_redisfailovers.yaml | 7 + ...atabases.spotahome.com_redisfailovers.yaml | 7 + mocks/service/k8s/Services.go | 14 ++ operator/redisfailover/service/check.go | 29 ++++- operator/redisfailover/service/constants.go | 4 + operator/redisfailover/service/demotion.go | 20 +-- operator/redisfailover/service/heal.go | 13 +- .../master_eviction_annotation_test.go | 123 ++++++++++++++++++ service/k8s/pod.go | 22 ++++ service/k8s/pod_test.go | 39 ++++++ 13 files changed, 279 insertions(+), 18 deletions(-) create mode 100644 operator/redisfailover/service/master_eviction_annotation_test.go diff --git a/README.md b/README.md index 07fe625e6..4170228d0 100644 --- a/README.md +++ b/README.md @@ -133,6 +133,13 @@ This redis-failover will be managed by the operator, resulting in the following **NOTE**: `NAME` is the named provided when creating the RedisFailover. **IMPORTANT**: the name of the redis-failover to be created cannot be longer than 48 characters, due to prepend of redis/sentinel identification and statefulset limitation. +### Protect the master from cluster-autoscaler eviction + +Setting `redis.preventMasterEviction: true` makes the operator annotate the current master pod with +`cluster-autoscaler.kubernetes.io/safe-to-evict: "false"` (and mark slaves `"true"`), so the +cluster-autoscaler will not drain the node running the master and trigger an avoidable failover. The +annotation follows the master as it moves. Defaults to `false` (no annotation is managed). + ### Persistence The operator can add persistence to Redis data. By default, an `emptyDir` will be used, so the data is not saved. diff --git a/api/redisfailover/v1/types.go b/api/redisfailover/v1/types.go index e48cb6458..97b737fa1 100644 --- a/api/redisfailover/v1/types.go +++ b/api/redisfailover/v1/types.go @@ -72,6 +72,11 @@ type RedisSettings struct { CustomReadinessProbe *corev1.Probe `json:"customReadinessProbe,omitempty"` CustomStartupProbe *corev1.Probe `json:"customStartupProbe,omitempty"` DisablePodDisruptionBudget bool `json:"disablePodDisruptionBudget,omitempty"` + // PreventMasterEviction, when true, annotates the current master pod with + // cluster-autoscaler.kubernetes.io/safe-to-evict=false so the cluster + // autoscaler will not drain the node running the master. Slaves are marked + // evictable. Defaults to false. + PreventMasterEviction bool `json:"preventMasterEviction,omitempty"` } // SentinelSettings defines the specification of the sentinel cluster diff --git a/charts/redisoperator/crds/databases.spotahome.com_redisfailovers.yaml b/charts/redisoperator/crds/databases.spotahome.com_redisfailovers.yaml index 7d63967d5..d7a360146 100644 --- a/charts/redisoperator/crds/databases.spotahome.com_redisfailovers.yaml +++ b/charts/redisoperator/crds/databases.spotahome.com_redisfailovers.yaml @@ -7184,6 +7184,13 @@ spec: port: format: int32 type: integer + preventMasterEviction: + description: |- + PreventMasterEviction, when true, annotates the current master pod with + cluster-autoscaler.kubernetes.io/safe-to-evict=false so the cluster + autoscaler will not drain the node running the master. Slaves are marked + evictable. Defaults to false. + type: boolean priorityClassName: type: string replicas: diff --git a/manifests/databases.spotahome.com_redisfailovers.yaml b/manifests/databases.spotahome.com_redisfailovers.yaml index 7d63967d5..d7a360146 100644 --- a/manifests/databases.spotahome.com_redisfailovers.yaml +++ b/manifests/databases.spotahome.com_redisfailovers.yaml @@ -7184,6 +7184,13 @@ spec: port: format: int32 type: integer + preventMasterEviction: + description: |- + PreventMasterEviction, when true, annotates the current master pod with + cluster-autoscaler.kubernetes.io/safe-to-evict=false so the cluster + autoscaler will not drain the node running the master. Slaves are marked + evictable. Defaults to false. + type: boolean priorityClassName: type: string replicas: diff --git a/manifests/kustomize/base/databases.spotahome.com_redisfailovers.yaml b/manifests/kustomize/base/databases.spotahome.com_redisfailovers.yaml index 7d63967d5..d7a360146 100644 --- a/manifests/kustomize/base/databases.spotahome.com_redisfailovers.yaml +++ b/manifests/kustomize/base/databases.spotahome.com_redisfailovers.yaml @@ -7184,6 +7184,13 @@ spec: port: format: int32 type: integer + preventMasterEviction: + description: |- + PreventMasterEviction, when true, annotates the current master pod with + cluster-autoscaler.kubernetes.io/safe-to-evict=false so the cluster + autoscaler will not drain the node running the master. Slaves are marked + evictable. Defaults to false. + type: boolean priorityClassName: type: string replicas: diff --git a/mocks/service/k8s/Services.go b/mocks/service/k8s/Services.go index edb46af89..59903d582 100644 --- a/mocks/service/k8s/Services.go +++ b/mocks/service/k8s/Services.go @@ -789,6 +789,20 @@ func (_m *Services) UpdatePodDisruptionBudget(namespace string, podDisruptionBud return r0 } +// UpdatePodAnnotations provides a mock function with given fields: namespace, podName, annotations +func (_m *Services) UpdatePodAnnotations(namespace string, podName string, annotations map[string]string) error { + ret := _m.Called(namespace, podName, annotations) + + var r0 error + if rf, ok := ret.Get(0).(func(string, string, map[string]string) error); ok { + r0 = rf(namespace, podName, annotations) + } else { + r0 = ret.Error(0) + } + + return r0 +} + // UpdatePodLabels provides a mock function with given fields: namespace, podName, labels func (_m *Services) UpdatePodLabels(namespace string, podName string, labels map[string]string) error { ret := _m.Called(namespace, podName, labels) diff --git a/operator/redisfailover/service/check.go b/operator/redisfailover/service/check.go index f1b8d7af5..96a0785c9 100644 --- a/operator/redisfailover/service/check.go +++ b/operator/redisfailover/service/check.go @@ -114,13 +114,36 @@ func IsMasterPod(pod *corev1.Pod) bool { return pod.Labels[redisRoleLabelKey] == redisRoleLabelMaster } -func (r *RedisFailoverChecker) setMasterLabelIfNecessary(namespace string, pod corev1.Pod) error { +// applyMasterEvictionAnnotation keeps the cluster-autoscaler safe-to-evict +// annotation in sync with a pod's role, but only when the RedisFailover opts in +// via spec.redis.preventMasterEviction. The master is pinned (false) and slaves +// are marked evictable (true). It reads the desired state off the already-fetched +// pod object and skips the patch when the annotation is already correct, so it is +// safe to call on every reconcile without extra API writes. +func applyMasterEvictionAnnotation(k8sService k8s.Services, rf *redisfailoverv1.RedisFailover, pod corev1.Pod, isMaster bool) error { + if !rf.Spec.Redis.PreventMasterEviction { + return nil + } + desired := "true" + if isMaster { + desired = "false" + } + if pod.Annotations[masterSafeToEvictAnnotation] == desired { + return nil + } + return k8sService.UpdatePodAnnotations(rf.Namespace, pod.Name, map[string]string{masterSafeToEvictAnnotation: desired}) +} + +func (r *RedisFailoverChecker) setMasterLabelIfNecessary(rf *redisfailoverv1.RedisFailover, pod corev1.Pod) error { + if err := applyMasterEvictionAnnotation(r.k8sService, rf, pod, true); err != nil { + return err + } for labelKey, labelValue := range pod.Labels { if labelKey == redisRoleLabelKey && labelValue == redisRoleLabelMaster { return nil } } - return r.k8sService.UpdatePodLabels(namespace, pod.Name, generateRedisMasterRoleLabel()) + return r.k8sService.UpdatePodLabels(rf.Namespace, pod.Name, generateRedisMasterRoleLabel()) } func (r *RedisFailoverChecker) setSlaveLabelIfNecessary(rf *redisfailoverv1.RedisFailover, pod corev1.Pod, port, password string) error { @@ -148,7 +171,7 @@ func (r *RedisFailoverChecker) CheckAllSlavesFromMaster(master string, rf *redis var wrongMasterErr error for _, rp := range rps.Items { if rp.Status.PodIP == master { - err = r.setMasterLabelIfNecessary(rf.Namespace, rp) + err = r.setMasterLabelIfNecessary(rf, rp) if err != nil { return err } diff --git a/operator/redisfailover/service/constants.go b/operator/redisfailover/service/constants.go index 1fdc356d9..04da4ab80 100644 --- a/operator/redisfailover/service/constants.go +++ b/operator/redisfailover/service/constants.go @@ -46,3 +46,7 @@ const ( // template hash, which the existing revision-based staleness check in // UpdateRedisesPods already uses to roll pods one at a time. const redisAuthSecretChecksumAnnotation = "redisfailovers.databases.spotahome.com/secret-checksum" + +// masterSafeToEvictAnnotation is the cluster-autoscaler annotation used to keep +// the node running the redis master from being drained during scale-down. +const masterSafeToEvictAnnotation = "cluster-autoscaler.kubernetes.io/safe-to-evict" diff --git a/operator/redisfailover/service/demotion.go b/operator/redisfailover/service/demotion.go index 37f2fdf9e..439473085 100644 --- a/operator/redisfailover/service/demotion.go +++ b/operator/redisfailover/service/demotion.go @@ -51,19 +51,19 @@ type ClientDisconnector interface { } // setSlaveLabel gives pod the slave role label if it doesn't already have it, -// and disconnects its clients if the label it replaces was master. +// and disconnects its clients if the label it replaces was master. Only then +// does it mark the pod evictable. func setSlaveLabel(k8sService k8s.Services, o options, rf *redisfailoverv1.RedisFailover, pod corev1.Pod, port, password string) error { previousRole := pod.Labels[redisRoleLabelKey] - if previousRole == redisRoleLabelSlave { - return nil - } - if err := k8sService.UpdatePodLabels(rf.Namespace, pod.Name, generateRedisSlaveRoleLabel()); err != nil { - return err - } - if previousRole == redisRoleLabelMaster && o.disconnector != nil { - o.disconnector.DisconnectDemoted(rf, pod, port, password) + if previousRole != redisRoleLabelSlave { + if err := k8sService.UpdatePodLabels(rf.Namespace, pod.Name, generateRedisSlaveRoleLabel()); err != nil { + return err + } + if previousRole == redisRoleLabelMaster && o.disconnector != nil { + o.disconnector.DisconnectDemoted(rf, pod, port, password) + } } - return nil + return applyMasterEvictionAnnotation(k8sService, rf, pod, false) } type endpointAwareDisconnector struct { diff --git a/operator/redisfailover/service/heal.go b/operator/redisfailover/service/heal.go index 644416b51..17a130212 100644 --- a/operator/redisfailover/service/heal.go +++ b/operator/redisfailover/service/heal.go @@ -53,13 +53,16 @@ func NewRedisFailoverHealer(k8sService k8s.Services, redisClient redis.Client, l } } -func (r *RedisFailoverHealer) setMasterLabelIfNecessary(namespace string, pod v1.Pod) error { +func (r *RedisFailoverHealer) setMasterLabelIfNecessary(rf *redisfailoverv1.RedisFailover, pod v1.Pod) error { + if err := applyMasterEvictionAnnotation(r.k8sService, rf, pod, true); err != nil { + return err + } for labelKey, labelValue := range pod.Labels { if labelKey == redisRoleLabelKey && labelValue == redisRoleLabelMaster { return nil } } - return r.k8sService.UpdatePodLabels(namespace, pod.Name, generateRedisMasterRoleLabel()) + return r.k8sService.UpdatePodLabels(rf.Namespace, pod.Name, generateRedisMasterRoleLabel()) } func (r *RedisFailoverHealer) setSlaveLabelIfNecessary(rf *redisfailoverv1.RedisFailover, pod v1.Pod, port, password string) error { @@ -84,7 +87,7 @@ func (r *RedisFailoverHealer) MakeMaster(ip string, rf *redisfailoverv1.RedisFai } for _, rp := range rps.Items { if rp.Status.PodIP == ip { - return r.setMasterLabelIfNecessary(rf.Namespace, rp) + return r.setMasterLabelIfNecessary(rf, rp) } } return nil @@ -122,7 +125,7 @@ func (r *RedisFailoverHealer) SetOldestAsMaster(rf *redisfailoverv1.RedisFailove continue } - err = r.setMasterLabelIfNecessary(rf.Namespace, pod) + err = r.setMasterLabelIfNecessary(rf, pod) if err != nil { return err } @@ -337,7 +340,7 @@ func (r *RedisFailoverHealer) PromoteBestReplica(newMasterIP string, rf *redisfa // Step 2: Update pod labels for the new master for _, rp := range rps.Items { if rp.Status.PodIP == newMasterIP { - if err := r.setMasterLabelIfNecessary(rf.Namespace, rp); err != nil { + if err := r.setMasterLabelIfNecessary(rf, rp); err != nil { r.logger.WithField("redisfailover", rf.Name).WithField("namespace", rf.Namespace). Errorf("Failed to set master label on pod %s: %v", rp.Name, err) return err diff --git a/operator/redisfailover/service/master_eviction_annotation_test.go b/operator/redisfailover/service/master_eviction_annotation_test.go new file mode 100644 index 000000000..d5702c86f --- /dev/null +++ b/operator/redisfailover/service/master_eviction_annotation_test.go @@ -0,0 +1,123 @@ +package service + +import ( + "errors" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + corev1 "k8s.io/api/core/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + + redisfailoverv1 "github.com/saremox/redis-operator/api/redisfailover/v1" + mK8SService "github.com/saremox/redis-operator/mocks/service/k8s" + "github.com/saremox/redis-operator/service/k8s" +) + +func rfWithEvictionProtection(enabled bool) *redisfailoverv1.RedisFailover { + return &redisfailoverv1.RedisFailover{ + ObjectMeta: metav1.ObjectMeta{Name: "test", Namespace: "testns"}, + Spec: redisfailoverv1.RedisFailoverSpec{ + Redis: redisfailoverv1.RedisSettings{PreventMasterEviction: enabled}, + }, + } +} + +func podNamed(name string, annotations map[string]string) corev1.Pod { + return corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: name, Annotations: annotations}} +} + +func TestApplyMasterEvictionAnnotation(t *testing.T) { + t.Run("flag disabled makes no annotation call", func(t *testing.T) { + ms := &mK8SService.Services{} + err := applyMasterEvictionAnnotation(ms, rfWithEvictionProtection(false), podNamed("p0", nil), true) + assert.NoError(t, err) + ms.AssertNotCalled(t, "UpdatePodAnnotations", mock.Anything, mock.Anything, mock.Anything) + }) + + t.Run("master is pinned with safe-to-evict false", func(t *testing.T) { + ms := &mK8SService.Services{} + ms.On("UpdatePodAnnotations", "testns", "p0", map[string]string{ + masterSafeToEvictAnnotation: "false", + }).Once().Return(nil) + + err := applyMasterEvictionAnnotation(ms, rfWithEvictionProtection(true), podNamed("p0", nil), true) + assert.NoError(t, err) + ms.AssertExpectations(t) + }) + + t.Run("slave is marked evictable with safe-to-evict true", func(t *testing.T) { + ms := &mK8SService.Services{} + ms.On("UpdatePodAnnotations", "testns", "p1", map[string]string{ + masterSafeToEvictAnnotation: "true", + }).Once().Return(nil) + + err := applyMasterEvictionAnnotation(ms, rfWithEvictionProtection(true), podNamed("p1", nil), false) + assert.NoError(t, err) + ms.AssertExpectations(t) + }) + + t.Run("no call when annotation already at desired value", func(t *testing.T) { + ms := &mK8SService.Services{} + pod := podNamed("p0", map[string]string{masterSafeToEvictAnnotation: "false"}) + err := applyMasterEvictionAnnotation(ms, rfWithEvictionProtection(true), pod, true) + assert.NoError(t, err) + ms.AssertNotCalled(t, "UpdatePodAnnotations", mock.Anything, mock.Anything, mock.Anything) + }) +} + +func TestSetSlaveLabelMarksSlavesEvictable(t *testing.T) { + ms := &mK8SService.Services{} + ms.On("UpdatePodAnnotations", "testns", "p1", map[string]string{ + masterSafeToEvictAnnotation: "true", + }).Once().Return(nil) + pod := podNamed("p1", nil) + pod.Labels = generateRedisSlaveRoleLabel() + + err := setSlaveLabel(ms, options{}, rfWithEvictionProtection(true), pod, "0", "") + assert.NoError(t, err) + ms.AssertExpectations(t) +} + +func TestSetSlaveLabelKeepsADemotedMasterPinnedUntilRelabelled(t *testing.T) { + pod := podNamed("p0", map[string]string{masterSafeToEvictAnnotation: "false"}) + pod.Labels = generateRedisMasterRoleLabel() + + ms := &mK8SService.Services{} + ms.On("UpdatePodLabels", "testns", "p0", generateRedisSlaveRoleLabel()).Once().Return(errors.New("boom")) + err := setSlaveLabel(ms, options{}, rfWithEvictionProtection(true), pod, "0", "") + assert.EqualError(t, err, "boom") + ms.AssertNotCalled(t, "UpdatePodAnnotations", mock.Anything, mock.Anything, mock.Anything) + + var calls []string + ms = &mK8SService.Services{} + ms.On("UpdatePodLabels", "testns", "p0", generateRedisSlaveRoleLabel()).Once(). + Run(func(mock.Arguments) { calls = append(calls, "label") }).Return(nil) + ms.On("UpdatePodAnnotations", "testns", "p0", map[string]string{masterSafeToEvictAnnotation: "true"}).Once(). + Run(func(mock.Arguments) { calls = append(calls, "annotate") }).Return(nil) + assert.NoError(t, setSlaveLabel(ms, options{}, rfWithEvictionProtection(true), pod, "0", "")) + assert.Equal(t, []string{"label", "annotate"}, calls) + ms.AssertExpectations(t) +} + +func TestSetMasterLabelStopsWhenPinningFails(t *testing.T) { + setters := map[string]func(k8s.Services, *redisfailoverv1.RedisFailover, corev1.Pod) error{ + "checker": func(ms k8s.Services, rf *redisfailoverv1.RedisFailover, pod corev1.Pod) error { + return (&RedisFailoverChecker{k8sService: ms}).setMasterLabelIfNecessary(rf, pod) + }, + "healer": func(ms k8s.Services, rf *redisfailoverv1.RedisFailover, pod corev1.Pod) error { + return (&RedisFailoverHealer{k8sService: ms}).setMasterLabelIfNecessary(rf, pod) + }, + } + for name, setMaster := range setters { + t.Run(name, func(t *testing.T) { + ms := &mK8SService.Services{} + ms.On("UpdatePodAnnotations", "testns", "p0", map[string]string{masterSafeToEvictAnnotation: "false"}).Once().Return(errors.New("boom")) + + err := setMaster(ms, rfWithEvictionProtection(true), podNamed("p0", nil)) + + assert.EqualError(t, err, "boom") + ms.AssertNotCalled(t, "UpdatePodLabels", mock.Anything, mock.Anything, mock.Anything) + }) + } +} diff --git a/service/k8s/pod.go b/service/k8s/pod.go index 390471dc0..66305bc5a 100644 --- a/service/k8s/pod.go +++ b/service/k8s/pod.go @@ -20,6 +20,7 @@ type Pod interface { DeletePod(namespace string, name string) error ListPods(namespace string) (*corev1.PodList, error) UpdatePodLabels(namespace, podName string, labels map[string]string) error + UpdatePodAnnotations(namespace, podName string, annotations map[string]string) error } // PodService is the pod service implementation using API calls to kubernetes. @@ -88,3 +89,24 @@ func (p *PodService) UpdatePodLabels(namespace, podName string, labels map[strin } return err } + +// UpdatePodAnnotations sets the given annotations on a pod. It uses a JSON merge +// patch so the annotations map is created when absent and existing annotations +// are left untouched, unlike the JSON-patch "replace" used for labels. +func (p *PodService) UpdatePodAnnotations(namespace, podName string, annotations map[string]string) error { + p.logger.Infof("Update pod annotations, namespace: %s, pod name: %s, annotations: %v", namespace, podName, annotations) + + patch := map[string]interface{}{ + "metadata": map[string]interface{}{ + "annotations": annotations, + }, + } + payloadBytes, _ := json.Marshal(patch) + + _, err := p.kubeClient.CoreV1().Pods(namespace).Patch(context.TODO(), podName, types.MergePatchType, payloadBytes, metav1.PatchOptions{}) + recordMetrics(namespace, "Pod", podName, "PATCH", err, p.metricsRecorder) + if err != nil { + p.logger.Errorf("Update pod annotations failed, namespace: %s, pod name: %s, error: %v", namespace, podName, err) + } + return err +} diff --git a/service/k8s/pod_test.go b/service/k8s/pod_test.go index 12cee1f49..ac4021336 100644 --- a/service/k8s/pod_test.go +++ b/service/k8s/pod_test.go @@ -231,3 +231,42 @@ func TestPodServiceList(t *testing.T) { assertTest.Equal([]kubetesting.Action{newPodListAction(testns)}, mcli.Actions()) }) } + +func TestPodServiceUpdatePodAnnotations(t *testing.T) { + testns := "testns" + tests := []struct { + name string + existing map[string]string + expected map[string]string + }{ + { + name: "creates the annotations on a pod without any", + expected: map[string]string{"safe-to-evict": "false"}, + }, + { + name: "keeps unrelated annotations and replaces the given key", + existing: map[string]string{"other": "kept", "safe-to-evict": "true"}, + expected: map[string]string{"other": "kept", "safe-to-evict": "false"}, + }, + } + + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + pod := &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: "testpod", Namespace: testns, Annotations: test.existing}} + mcli := kubernetes.NewClientset(pod) + service := k8s.NewPodService(mcli, log.Dummy, metrics.Dummy) + + assert.NoError(t, service.UpdatePodAnnotations(testns, "testpod", map[string]string{"safe-to-evict": "false"})) + + got, err := mcli.CoreV1().Pods(testns).Get(context.TODO(), "testpod", metav1.GetOptions{}) + assert.NoError(t, err) + assert.Equal(t, test.expected, got.Annotations) + }) + } + + t.Run("returns not found for a missing pod", func(t *testing.T) { + service := k8s.NewPodService(kubernetes.NewClientset(), log.Dummy, metrics.Dummy) + err := service.UpdatePodAnnotations(testns, "missing", map[string]string{"safe-to-evict": "false"}) + assert.True(t, kubeerrors.IsNotFound(err)) + }) +} From 480be21a0d0754e33d8412336f6e129df01c01eb Mon Sep 17 00:00:00 2001 From: Michael Strassberger Date: Sat, 26 Sep 2026 09:53:54 +0200 Subject: [PATCH 16/24] feat(sentinel): configurable deployment strategy and PDB minAvailable (dnse #33) (#146) * feat(sentinel): configurable deployment strategy and PDB minAvailable (#33) Two rollout/availability knobs that were previously hardcoded: - sentinel.strategy overrides the sentinel Deployment update strategy (e.g. rollingUpdate maxSurge/maxUnavailable), avoiding a deadlocked rolling update when required anti-affinity plus replicas==nodes leaves no room to surge. - redis/sentinel.podDisruptionBudgetMinAvailable overrides the PDB minAvailable (previously fixed at 2, or 1 when replicas<=2). Regenerated CRD + deepcopy for the new fields. Refs upstream spotahome/redis-operator#662, #516, #598. Co-authored-by: Binh Nguyen (cherry picked from commit a53177dd9a6ac4e5c351f1e430c3ace774081308) * Compare a configured sentinel strategy instead of discarding it deploymentUpToDate blanked the stored strategy before comparing, so a configured sentinel.strategy never matched and every reconcile updated the Deployment, while removing one was never applied. Compare both strategies with the API server's defaults filled in instead. Also fix the sentinel PDB default docs, which derive from the sentinel replicas rather than the redis ones, and cover the sentinel override. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01G4mmyo3sqLwf23m5FE2nBE --------- Co-authored-by: Binh Nguyen <37066217+binhnguyenduc@users.noreply.github.com> Co-authored-by: Binh Nguyen Co-authored-by: Claude --- README.md | 20 +++ api/redisfailover/v1/types.go | 13 ++ api/redisfailover/v1/zz_generated.deepcopy.go | 12 ++ ...atabases.spotahome.com_redisfailovers.yaml | 67 +++++++++ ...atabases.spotahome.com_redisfailovers.yaml | 67 +++++++++ ...atabases.spotahome.com_redisfailovers.yaml | 67 +++++++++ operator/redisfailover/service/client.go | 10 +- .../service/deployment_strategy_pdb_test.go | 139 ++++++++++++++++++ operator/redisfailover/service/generator.go | 1 + service/k8s/deployment_test.go | 41 ++++++ service/k8s/resourcediff.go | 28 +++- 11 files changed, 461 insertions(+), 4 deletions(-) create mode 100644 operator/redisfailover/service/deployment_strategy_pdb_test.go diff --git a/README.md b/README.md index 4170228d0..2056c22a5 100644 --- a/README.md +++ b/README.md @@ -140,6 +140,26 @@ Setting `redis.preventMasterEviction: true` makes the operator annotate the curr cluster-autoscaler will not drain the node running the master and trigger an avoidable failover. The annotation follows the master as it moves. Defaults to `false` (no annotation is managed). +### Sentinel update strategy and PodDisruptionBudget + +The sentinel `Deployment` update strategy can be overridden via `sentinel.strategy` (e.g. to set +`rollingUpdate.maxSurge`/`maxUnavailable`). This helps when required anti-affinity plus +`replicas == nodes` would otherwise deadlock the default rolling update: + +```yaml +spec: + sentinel: + strategy: + type: RollingUpdate + rollingUpdate: + maxSurge: 1 + maxUnavailable: 0 +``` + +The `PodDisruptionBudget` `minAvailable` for each component defaults to `2` (or `1` when that +component's `replicas <= 2`). Override it per component with `redis.podDisruptionBudgetMinAvailable` / +`sentinel.podDisruptionBudgetMinAvailable` (an integer or percentage string such as `"60%"`). + ### Persistence The operator can add persistence to Redis data. By default, an `emptyDir` will be used, so the data is not saved. diff --git a/api/redisfailover/v1/types.go b/api/redisfailover/v1/types.go index 97b737fa1..68a929550 100644 --- a/api/redisfailover/v1/types.go +++ b/api/redisfailover/v1/types.go @@ -1,8 +1,10 @@ package v1 import ( + appsv1 "k8s.io/api/apps/v1" corev1 "k8s.io/api/core/v1" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/util/intstr" ) // +genclient @@ -77,6 +79,9 @@ type RedisSettings struct { // autoscaler will not drain the node running the master. Slaves are marked // evictable. Defaults to false. PreventMasterEviction bool `json:"preventMasterEviction,omitempty"` + // PodDisruptionBudgetMinAvailable overrides the PodDisruptionBudget + // minAvailable for the redis pods. Defaults to 2 (or 1 when replicas <= 2). + PodDisruptionBudgetMinAvailable *intstr.IntOrString `json:"podDisruptionBudgetMinAvailable,omitempty"` } // SentinelSettings defines the specification of the sentinel cluster @@ -118,6 +123,14 @@ type SentinelSettings struct { CustomReadinessProbe *corev1.Probe `json:"customReadinessProbe,omitempty"` CustomStartupProbe *corev1.Probe `json:"customStartupProbe,omitempty"` DisablePodDisruptionBudget bool `json:"disablePodDisruptionBudget,omitempty"` + // Strategy overrides the sentinel Deployment update strategy (e.g. to set + // rollingUpdate maxSurge/maxUnavailable). Defaults to the Kubernetes default + // RollingUpdate strategy when unset. + Strategy appsv1.DeploymentStrategy `json:"strategy,omitempty"` + // PodDisruptionBudgetMinAvailable overrides the PodDisruptionBudget + // minAvailable for the sentinel pods. Defaults to 2 (or 1 when sentinel + // replicas <= 2). + PodDisruptionBudgetMinAvailable *intstr.IntOrString `json:"podDisruptionBudgetMinAvailable,omitempty"` } // AuthSettings contains settings about auth diff --git a/api/redisfailover/v1/zz_generated.deepcopy.go b/api/redisfailover/v1/zz_generated.deepcopy.go index bf538242e..cbe2e21b6 100644 --- a/api/redisfailover/v1/zz_generated.deepcopy.go +++ b/api/redisfailover/v1/zz_generated.deepcopy.go @@ -8,6 +8,7 @@ import ( corev1 "k8s.io/api/core/v1" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" "k8s.io/apimachinery/pkg/runtime" + "k8s.io/apimachinery/pkg/util/intstr" ) // DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. @@ -361,6 +362,11 @@ func (in *RedisSettings) DeepCopyInto(out *RedisSettings) { *out = new(corev1.Probe) (*in).DeepCopyInto(*out) } + if in.PodDisruptionBudgetMinAvailable != nil { + in, out := &in.PodDisruptionBudgetMinAvailable, &out.PodDisruptionBudgetMinAvailable + *out = new(intstr.IntOrString) + **out = **in + } } // DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new RedisSettings. @@ -542,6 +548,12 @@ func (in *SentinelSettings) DeepCopyInto(out *SentinelSettings) { *out = new(corev1.Probe) (*in).DeepCopyInto(*out) } + in.Strategy.DeepCopyInto(&out.Strategy) + if in.PodDisruptionBudgetMinAvailable != nil { + in, out := &in.PodDisruptionBudgetMinAvailable, &out.PodDisruptionBudgetMinAvailable + *out = new(intstr.IntOrString) + **out = **in + } } // DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new SentinelSettings. diff --git a/charts/redisoperator/crds/databases.spotahome.com_redisfailovers.yaml b/charts/redisoperator/crds/databases.spotahome.com_redisfailovers.yaml index d7a360146..08e2edbb3 100644 --- a/charts/redisoperator/crds/databases.spotahome.com_redisfailovers.yaml +++ b/charts/redisoperator/crds/databases.spotahome.com_redisfailovers.yaml @@ -7181,6 +7181,14 @@ spec: additionalProperties: type: string type: object + podDisruptionBudgetMinAvailable: + anyOf: + - type: integer + - type: string + description: |- + PodDisruptionBudgetMinAvailable overrides the PodDisruptionBudget + minAvailable for the redis pods. Defaults to 2 (or 1 when replicas <= 2). + x-kubernetes-int-or-string: true port: format: int32 type: integer @@ -15501,6 +15509,15 @@ spec: additionalProperties: type: string type: object + podDisruptionBudgetMinAvailable: + anyOf: + - type: integer + - type: string + description: |- + PodDisruptionBudgetMinAvailable overrides the PodDisruptionBudget + minAvailable for the sentinel pods. Defaults to 2 (or 1 when sentinel + replicas <= 2). + x-kubernetes-int-or-string: true priorityClassName: type: string replicas: @@ -15809,6 +15826,56 @@ spec: type: object startupConfigMap: type: string + strategy: + description: |- + Strategy overrides the sentinel Deployment update strategy (e.g. to set + rollingUpdate maxSurge/maxUnavailable). Defaults to the Kubernetes default + RollingUpdate strategy when unset. + properties: + rollingUpdate: + description: |- + Rolling update config params. Present only if DeploymentStrategyType = + RollingUpdate. + properties: + maxSurge: + anyOf: + - type: integer + - type: string + description: |- + The maximum number of pods that can be scheduled above the desired number of + pods. + Value can be an absolute number (ex: 5) or a percentage of desired pods (ex: 10%). + This can not be 0 if MaxUnavailable is 0. + Absolute number is calculated from percentage by rounding up. + Defaults to 25%. + Example: when this is set to 30%, the new ReplicaSet can be scaled up immediately when + the rolling update starts, such that the total number of old and new pods do not exceed + 130% of desired pods. Once old pods have been killed, + new ReplicaSet can be scaled up further, ensuring that total number of pods running + at any time during the update is at most 130% of desired pods. + x-kubernetes-int-or-string: true + maxUnavailable: + anyOf: + - type: integer + - type: string + description: |- + The maximum number of pods that can be unavailable during the update. + Value can be an absolute number (ex: 5) or a percentage of desired pods (ex: 10%). + Absolute number is calculated from percentage by rounding down. + This can not be 0 if MaxSurge is 0. + Defaults to 25%. + Example: when this is set to 30%, the old ReplicaSet can be scaled down to 70% of desired pods + immediately when the rolling update starts. Once new pods are ready, old ReplicaSet + can be scaled down further, followed by scaling up the new ReplicaSet, ensuring + that the total number of pods available at all times during the update is at + least 70% of desired pods. + x-kubernetes-int-or-string: true + type: object + type: + description: Type of deployment. Can be "Recreate" or "RollingUpdate". + Default is RollingUpdate. + type: string + type: object tolerations: items: description: |- diff --git a/manifests/databases.spotahome.com_redisfailovers.yaml b/manifests/databases.spotahome.com_redisfailovers.yaml index d7a360146..08e2edbb3 100644 --- a/manifests/databases.spotahome.com_redisfailovers.yaml +++ b/manifests/databases.spotahome.com_redisfailovers.yaml @@ -7181,6 +7181,14 @@ spec: additionalProperties: type: string type: object + podDisruptionBudgetMinAvailable: + anyOf: + - type: integer + - type: string + description: |- + PodDisruptionBudgetMinAvailable overrides the PodDisruptionBudget + minAvailable for the redis pods. Defaults to 2 (or 1 when replicas <= 2). + x-kubernetes-int-or-string: true port: format: int32 type: integer @@ -15501,6 +15509,15 @@ spec: additionalProperties: type: string type: object + podDisruptionBudgetMinAvailable: + anyOf: + - type: integer + - type: string + description: |- + PodDisruptionBudgetMinAvailable overrides the PodDisruptionBudget + minAvailable for the sentinel pods. Defaults to 2 (or 1 when sentinel + replicas <= 2). + x-kubernetes-int-or-string: true priorityClassName: type: string replicas: @@ -15809,6 +15826,56 @@ spec: type: object startupConfigMap: type: string + strategy: + description: |- + Strategy overrides the sentinel Deployment update strategy (e.g. to set + rollingUpdate maxSurge/maxUnavailable). Defaults to the Kubernetes default + RollingUpdate strategy when unset. + properties: + rollingUpdate: + description: |- + Rolling update config params. Present only if DeploymentStrategyType = + RollingUpdate. + properties: + maxSurge: + anyOf: + - type: integer + - type: string + description: |- + The maximum number of pods that can be scheduled above the desired number of + pods. + Value can be an absolute number (ex: 5) or a percentage of desired pods (ex: 10%). + This can not be 0 if MaxUnavailable is 0. + Absolute number is calculated from percentage by rounding up. + Defaults to 25%. + Example: when this is set to 30%, the new ReplicaSet can be scaled up immediately when + the rolling update starts, such that the total number of old and new pods do not exceed + 130% of desired pods. Once old pods have been killed, + new ReplicaSet can be scaled up further, ensuring that total number of pods running + at any time during the update is at most 130% of desired pods. + x-kubernetes-int-or-string: true + maxUnavailable: + anyOf: + - type: integer + - type: string + description: |- + The maximum number of pods that can be unavailable during the update. + Value can be an absolute number (ex: 5) or a percentage of desired pods (ex: 10%). + Absolute number is calculated from percentage by rounding down. + This can not be 0 if MaxSurge is 0. + Defaults to 25%. + Example: when this is set to 30%, the old ReplicaSet can be scaled down to 70% of desired pods + immediately when the rolling update starts. Once new pods are ready, old ReplicaSet + can be scaled down further, followed by scaling up the new ReplicaSet, ensuring + that the total number of pods available at all times during the update is at + least 70% of desired pods. + x-kubernetes-int-or-string: true + type: object + type: + description: Type of deployment. Can be "Recreate" or "RollingUpdate". + Default is RollingUpdate. + type: string + type: object tolerations: items: description: |- diff --git a/manifests/kustomize/base/databases.spotahome.com_redisfailovers.yaml b/manifests/kustomize/base/databases.spotahome.com_redisfailovers.yaml index d7a360146..08e2edbb3 100644 --- a/manifests/kustomize/base/databases.spotahome.com_redisfailovers.yaml +++ b/manifests/kustomize/base/databases.spotahome.com_redisfailovers.yaml @@ -7181,6 +7181,14 @@ spec: additionalProperties: type: string type: object + podDisruptionBudgetMinAvailable: + anyOf: + - type: integer + - type: string + description: |- + PodDisruptionBudgetMinAvailable overrides the PodDisruptionBudget + minAvailable for the redis pods. Defaults to 2 (or 1 when replicas <= 2). + x-kubernetes-int-or-string: true port: format: int32 type: integer @@ -15501,6 +15509,15 @@ spec: additionalProperties: type: string type: object + podDisruptionBudgetMinAvailable: + anyOf: + - type: integer + - type: string + description: |- + PodDisruptionBudgetMinAvailable overrides the PodDisruptionBudget + minAvailable for the sentinel pods. Defaults to 2 (or 1 when sentinel + replicas <= 2). + x-kubernetes-int-or-string: true priorityClassName: type: string replicas: @@ -15809,6 +15826,56 @@ spec: type: object startupConfigMap: type: string + strategy: + description: |- + Strategy overrides the sentinel Deployment update strategy (e.g. to set + rollingUpdate maxSurge/maxUnavailable). Defaults to the Kubernetes default + RollingUpdate strategy when unset. + properties: + rollingUpdate: + description: |- + Rolling update config params. Present only if DeploymentStrategyType = + RollingUpdate. + properties: + maxSurge: + anyOf: + - type: integer + - type: string + description: |- + The maximum number of pods that can be scheduled above the desired number of + pods. + Value can be an absolute number (ex: 5) or a percentage of desired pods (ex: 10%). + This can not be 0 if MaxUnavailable is 0. + Absolute number is calculated from percentage by rounding up. + Defaults to 25%. + Example: when this is set to 30%, the new ReplicaSet can be scaled up immediately when + the rolling update starts, such that the total number of old and new pods do not exceed + 130% of desired pods. Once old pods have been killed, + new ReplicaSet can be scaled up further, ensuring that total number of pods running + at any time during the update is at most 130% of desired pods. + x-kubernetes-int-or-string: true + maxUnavailable: + anyOf: + - type: integer + - type: string + description: |- + The maximum number of pods that can be unavailable during the update. + Value can be an absolute number (ex: 5) or a percentage of desired pods (ex: 10%). + Absolute number is calculated from percentage by rounding down. + This can not be 0 if MaxSurge is 0. + Defaults to 25%. + Example: when this is set to 30%, the old ReplicaSet can be scaled down to 70% of desired pods + immediately when the rolling update starts. Once new pods are ready, old ReplicaSet + can be scaled down further, followed by scaling up the new ReplicaSet, ensuring + that the total number of pods available at all times during the update is at + least 70% of desired pods. + x-kubernetes-int-or-string: true + type: object + type: + description: Type of deployment. Can be "Recreate" or "RollingUpdate". + Default is RollingUpdate. + type: string + type: object tolerations: items: description: |- diff --git a/operator/redisfailover/service/client.go b/operator/redisfailover/service/client.go index c47458d35..9a85adadc 100644 --- a/operator/redisfailover/service/client.go +++ b/operator/redisfailover/service/client.go @@ -87,7 +87,7 @@ func (r *RedisFailoverKubeClient) EnsureSentinelConfigMap(rf *redisfailoverv1.Re // EnsureSentinelDeployment makes sure the sentinel deployment exists in the desired state func (r *RedisFailoverKubeClient) EnsureSentinelDeployment(rf *redisfailoverv1.RedisFailover, labels map[string]string, ownerRefs []metav1.OwnerReference) error { if !rf.Spec.Sentinel.DisablePodDisruptionBudget { - if err := r.ensurePodDisruptionBudget(rf, sentinelName, sentinelRoleName, labels, ownerRefs, rf.Spec.Sentinel.Replicas); err != nil { + if err := r.ensurePodDisruptionBudget(rf, sentinelName, sentinelRoleName, rf.Spec.Sentinel.PodDisruptionBudgetMinAvailable, labels, ownerRefs, rf.Spec.Sentinel.Replicas); err != nil { return err } } @@ -119,7 +119,7 @@ func (r *RedisFailoverKubeClient) ensureSentinelServiceAccount(rf *redisfailover // EnsureRedisStatefulset makes sure the redis statefulset exists in the desired state func (r *RedisFailoverKubeClient) EnsureRedisStatefulset(rf *redisfailoverv1.RedisFailover, labels map[string]string, ownerRefs []metav1.OwnerReference) error { if !rf.Spec.Redis.DisablePodDisruptionBudget { - if err := r.ensurePodDisruptionBudget(rf, redisName, redisRoleName, labels, ownerRefs, rf.Spec.Redis.Replicas); err != nil { + if err := r.ensurePodDisruptionBudget(rf, redisName, redisRoleName, rf.Spec.Redis.PodDisruptionBudgetMinAvailable, labels, ownerRefs, rf.Spec.Redis.Replicas); err != nil { return err } } @@ -264,7 +264,8 @@ func (r *RedisFailoverKubeClient) EnsureRedisSlaveService(rf *redisfailoverv1.Re // ensurePodDisruptionBudget creates or updates a PDB for the given component. // replicas must be the replica count of the component being protected (not a different component). -func (r *RedisFailoverKubeClient) ensurePodDisruptionBudget(rf *redisfailoverv1.RedisFailover, name string, component string, labels map[string]string, ownerRefs []metav1.OwnerReference, replicas int32) error { +// minAvailableOverride, when non-nil, replaces the replicas-derived default. +func (r *RedisFailoverKubeClient) ensurePodDisruptionBudget(rf *redisfailoverv1.RedisFailover, name string, component string, minAvailableOverride *intstr.IntOrString, labels map[string]string, ownerRefs []metav1.OwnerReference, replicas int32) error { name = generateName(name, rf.Name) namespace := rf.Namespace @@ -272,6 +273,9 @@ func (r *RedisFailoverKubeClient) ensurePodDisruptionBudget(rf *redisfailoverv1. if replicas <= 2 { minAvailable = intstr.FromInt(1) } + if minAvailableOverride != nil { + minAvailable = *minAvailableOverride + } selectorLabels := generateSelectorLabels(component, rf.Name) metaLabels := util.MergeLabels(labels, selectorLabels) diff --git a/operator/redisfailover/service/deployment_strategy_pdb_test.go b/operator/redisfailover/service/deployment_strategy_pdb_test.go new file mode 100644 index 000000000..3972806fa --- /dev/null +++ b/operator/redisfailover/service/deployment_strategy_pdb_test.go @@ -0,0 +1,139 @@ +package service_test + +import ( + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + appsv1 "k8s.io/api/apps/v1" + policyv1 "k8s.io/api/policy/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/util/intstr" + + "github.com/saremox/redis-operator/log" + "github.com/saremox/redis-operator/metrics" + mK8SService "github.com/saremox/redis-operator/mocks/service/k8s" + rfservice "github.com/saremox/redis-operator/operator/redisfailover/service" +) + +func TestSentinelDeploymentStrategy(t *testing.T) { + assert := assert.New(t) + + maxSurge := intstr.FromInt(1) + maxUnavailable := intstr.FromInt(0) + strategy := appsv1.DeploymentStrategy{ + Type: appsv1.RollingUpdateDeploymentStrategyType, + RollingUpdate: &appsv1.RollingUpdateDeployment{ + MaxSurge: &maxSurge, + MaxUnavailable: &maxUnavailable, + }, + } + + rf := generateRF() + rf.Spec.Sentinel.Strategy = strategy + + var gotStrategy appsv1.DeploymentStrategy + ms := &mK8SService.Services{} + ms.On("CreateOrUpdatePodDisruptionBudget", namespace, mock.Anything).Once().Return(nil, nil) + ms.On("CreateOrUpdateServiceAccount", namespace, mock.Anything).Once().Return(nil) + ms.On("CreateOrUpdateDeployment", namespace, mock.Anything).Once().Run(func(args mock.Arguments) { + gotStrategy = args.Get(1).(*appsv1.Deployment).Spec.Strategy + }).Return(nil) + + client := rfservice.NewRedisFailoverKubeClient(ms, log.Dummy, metrics.Dummy) + err := client.EnsureSentinelDeployment(rf, nil, []metav1.OwnerReference{}) + + assert.NoError(err) + assert.Equal(strategy, gotStrategy) +} + +func TestPodDisruptionBudgetMinAvailableOverride(t *testing.T) { + tests := []struct { + name string + override *intstr.IntOrString + redisReplic int32 + expected intstr.IntOrString + }{ + { + name: "default with >2 replicas is 2", + override: nil, + redisReplic: 3, + expected: intstr.FromInt(2), + }, + { + name: "default with <=2 replicas is 1", + override: nil, + redisReplic: 2, + expected: intstr.FromInt(1), + }, + { + name: "explicit override wins", + override: ptrIOS(intstr.FromString("60%")), + redisReplic: 3, + expected: intstr.FromString("60%"), + }, + } + + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + assert := assert.New(t) + + rf := generateRF() + rf.Spec.Redis.Replicas = test.redisReplic + rf.Spec.Redis.PodDisruptionBudgetMinAvailable = test.override + + var gotMinAvailable *intstr.IntOrString + ms := &mK8SService.Services{} + ms.On("CreateOrUpdatePodDisruptionBudget", namespace, mock.Anything).Once().Run(func(args mock.Arguments) { + gotMinAvailable = args.Get(1).(*policyv1.PodDisruptionBudget).Spec.MinAvailable + }).Return(nil) + ms.On("CreateOrUpdateStatefulSet", namespace, mock.Anything).Once().Return(nil) + + client := rfservice.NewRedisFailoverKubeClient(ms, log.Dummy, metrics.Dummy) + err := client.EnsureRedisStatefulset(rf, nil, []metav1.OwnerReference{}) + + assert.NoError(err) + assert.NotNil(gotMinAvailable) + assert.Equal(test.expected, *gotMinAvailable) + }) + } +} + +func ptrIOS(v intstr.IntOrString) *intstr.IntOrString { return &v } + +func TestSentinelPodDisruptionBudgetMinAvailable(t *testing.T) { + tests := []struct { + name string + override *intstr.IntOrString + expected intstr.IntOrString + }{ + {name: "defaults from the sentinel replicas, not the redis ones", expected: intstr.FromInt(1)}, + {name: "explicit override wins", override: ptrIOS(intstr.FromString("60%")), expected: intstr.FromString("60%")}, + } + + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + assert := assert.New(t) + + rf := generateRF() + rf.Spec.Redis.Replicas = 5 + rf.Spec.Sentinel.Replicas = 2 + rf.Spec.Sentinel.PodDisruptionBudgetMinAvailable = test.override + + var gotMinAvailable *intstr.IntOrString + ms := &mK8SService.Services{} + ms.On("CreateOrUpdatePodDisruptionBudget", namespace, mock.Anything).Once().Run(func(args mock.Arguments) { + gotMinAvailable = args.Get(1).(*policyv1.PodDisruptionBudget).Spec.MinAvailable + }).Return(nil) + ms.On("CreateOrUpdateServiceAccount", namespace, mock.Anything).Once().Return(nil) + ms.On("CreateOrUpdateDeployment", namespace, mock.Anything).Once().Return(nil) + + client := rfservice.NewRedisFailoverKubeClient(ms, log.Dummy, metrics.Dummy) + assert.NoError(client.EnsureSentinelDeployment(rf, nil, []metav1.OwnerReference{})) + + if assert.NotNil(gotMinAvailable) { + assert.Equal(test.expected, *gotMinAvailable) + } + }) + } +} diff --git a/operator/redisfailover/service/generator.go b/operator/redisfailover/service/generator.go index dd6765986..848343091 100644 --- a/operator/redisfailover/service/generator.go +++ b/operator/redisfailover/service/generator.go @@ -590,6 +590,7 @@ func generateSentinelDeployment(rf *redisfailoverv1.RedisFailover, labels map[st }, Spec: appsv1.DeploymentSpec{ Replicas: &rf.Spec.Sentinel.Replicas, + Strategy: rf.Spec.Sentinel.Strategy, Selector: &metav1.LabelSelector{ MatchLabels: selectorLabels, }, diff --git a/service/k8s/deployment_test.go b/service/k8s/deployment_test.go index e63cd7368..2baa9f913 100644 --- a/service/k8s/deployment_test.go +++ b/service/k8s/deployment_test.go @@ -12,6 +12,7 @@ import ( metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" "k8s.io/apimachinery/pkg/runtime" "k8s.io/apimachinery/pkg/runtime/schema" + "k8s.io/apimachinery/pkg/util/intstr" kubernetes "k8s.io/client-go/kubernetes/fake" kubetesting "k8s.io/client-go/testing" "k8s.io/utils/ptr" @@ -192,6 +193,22 @@ func serverDefaulted(d *appsv1.Deployment) *appsv1.Deployment { return d } +func withStrategy(d *appsv1.Deployment, s appsv1.DeploymentStrategy) *appsv1.Deployment { + d.Spec.Strategy = s + return d +} + +func recreate() appsv1.DeploymentStrategy { + return appsv1.DeploymentStrategy{Type: appsv1.RecreateDeploymentStrategyType} +} + +func rollingUpdate(maxSurge, maxUnavailable *intstr.IntOrString) appsv1.DeploymentStrategy { + return appsv1.DeploymentStrategy{ + Type: appsv1.RollingUpdateDeploymentStrategyType, + RollingUpdate: &appsv1.RollingUpdateDeployment{MaxSurge: maxSurge, MaxUnavailable: maxUnavailable}, + } +} + func TestDeploymentServiceObjectUpToDate(t *testing.T) { testns := "testns" @@ -242,6 +259,30 @@ func TestDeploymentServiceObjectUpToDate(t *testing.T) { desired: realisticDeployment(3), expectUpdates: 1, }, + { + name: "a configured strategy the server stored as-is is a no-op", + stored: withStrategy(realisticDeployment(3), recreate()), + desired: withStrategy(realisticDeployment(3), recreate()), + expectUpdates: 0, + }, + { + name: "a partly configured strategy the server filled in is a no-op", + stored: withStrategy(realisticDeployment(3), rollingUpdate(ptr.To(intstr.FromInt32(1)), ptr.To(intstr.FromString("25%")))), + desired: withStrategy(realisticDeployment(3), rollingUpdate(ptr.To(intstr.FromInt32(1)), nil)), + expectUpdates: 0, + }, + { + name: "a changed strategy triggers an update", + stored: withStrategy(realisticDeployment(3), rollingUpdate(ptr.To(intstr.FromInt32(1)), ptr.To(intstr.FromString("25%")))), + desired: withStrategy(realisticDeployment(3), recreate()), + expectUpdates: 1, + }, + { + name: "removing a configured strategy triggers an update", + stored: withStrategy(realisticDeployment(3), recreate()), + desired: realisticDeployment(3), + expectUpdates: 1, + }, { name: "a deployment-controller-owned annotation on stored does not trigger an update", // generateSentinelDeployment never sets Deployment-level diff --git a/service/k8s/resourcediff.go b/service/k8s/resourcediff.go index 654b9a6ed..94010ec92 100644 --- a/service/k8s/resourcediff.go +++ b/service/k8s/resourcediff.go @@ -5,6 +5,7 @@ import ( corev1 "k8s.io/api/core/v1" policyv1 "k8s.io/api/policy/v1" "k8s.io/apimachinery/pkg/api/equality" + "k8s.io/apimachinery/pkg/util/intstr" ) // The *UpToDate functions below all follow the same pattern: compare a live @@ -70,12 +71,37 @@ func deploymentUpToDate(stored, desired *appsv1.Deployment) bool { normalized := stored.Spec.DeepCopy() normalized.RevisionHistoryLimit = nil normalized.ProgressDeadlineSeconds = nil - normalized.Strategy = appsv1.DeploymentStrategy{} + if equality.Semantic.DeepEqual(defaultedStrategy(stored.Spec.Strategy), defaultedStrategy(desired.Spec.Strategy)) { + normalized.Strategy = desired.Spec.Strategy + } normalizePodSpecForComparison(&normalized.Template.Spec) return equality.Semantic.DeepEqual(normalized, &desired.Spec) } +// defaultedStrategy fills in what the API server defaults in a Deployment +// strategy, so a stored strategy compares equal to the desired one it came from. +func defaultedStrategy(s appsv1.DeploymentStrategy) appsv1.DeploymentStrategy { + s = *s.DeepCopy() + if s.Type == "" { + s.Type = appsv1.RollingUpdateDeploymentStrategyType + } + if s.Type != appsv1.RollingUpdateDeploymentStrategyType { + return s + } + if s.RollingUpdate == nil { + s.RollingUpdate = &appsv1.RollingUpdateDeployment{} + } + quarter := intstr.FromString("25%") + if s.RollingUpdate.MaxUnavailable == nil { + s.RollingUpdate.MaxUnavailable = &quarter + } + if s.RollingUpdate.MaxSurge == nil { + s.RollingUpdate.MaxSurge = &quarter + } + return s +} + // serviceUpToDate is statefulSetUpToDate's counterpart for Service. See its // doc comment for the general comparison strategy. // From 6f877ece3085339e6fd1f30cb6088fc392e02d81 Mon Sep 17 00:00:00 2001 From: Michael Strassberger Date: Sat, 26 Sep 2026 09:55:14 +0200 Subject: [PATCH 17/24] feat(env): allow custom env vars on redis and sentinel main containers (dnse #34) (#145) * feat(env): allow custom env vars on redis and sentinel main containers (#34) Only the exporter sidecar accepted custom environment variables. Add an optional env field to the redis and sentinel specs, injected into their main containers. On the redis container the user's env is placed before the operator-injected vars (REDIS_ADDR/PORT/USER/PASSWORD) so those keep precedence under Kubernetes last-wins semantics. Regenerated CRD + deepcopy for the new fields. Refs upstream spotahome/redis-operator#290. Co-authored-by: Binh Nguyen (cherry picked from commit 4d51d46592498e13861c8a1c6aaaadbbb5508b28) * Test that DeepCopy clones the redis and sentinel env vars Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01G4mmyo3sqLwf23m5FE2nBE --------- Co-authored-by: Binh Nguyen <37066217+binhnguyenduc@users.noreply.github.com> Co-authored-by: Binh Nguyen Co-authored-by: Claude --- README.md | 7 + api/redisfailover/v1/deepcopy_test.go | 21 ++ api/redisfailover/v1/types.go | 2 + api/redisfailover/v1/zz_generated.deepcopy.go | 14 + ...atabases.spotahome.com_redisfailovers.yaml | 314 ++++++++++++++++++ ...atabases.spotahome.com_redisfailovers.yaml | 314 ++++++++++++++++++ ...atabases.spotahome.com_redisfailovers.yaml | 314 ++++++++++++++++++ operator/redisfailover/service/generator.go | 7 +- .../redisfailover/service/pod_env_test.go | 89 +++++ 9 files changed, 1081 insertions(+), 1 deletion(-) create mode 100644 operator/redisfailover/service/pod_env_test.go diff --git a/README.md b/README.md index 2056c22a5..a2c14461a 100644 --- a/README.md +++ b/README.md @@ -222,6 +222,13 @@ By default, redis and sentinel will be called with the basic command, giving the If necessary, this command can be changed with the `command` option inside redis/sentinel spec. An example can be found in the [custom command example file](example/redisfailover/custom-command.yaml). +### Custom environment variables + +Extra environment variables can be injected into the redis and sentinel **main** containers via +`redis.env` / `sentinel.env` (standard Kubernetes `EnvVar` entries). The operator's own variables +(`REDIS_ADDR`, `REDIS_PORT`, `REDIS_USER`, `REDIS_PASSWORD`) always take precedence, so a +user-supplied variable that reuses one of those names cannot override it. + ### Custom Priority Class To use a custom Kubernetes [Priority Class](https://kubernetes.io/docs/concepts/configuration/pod-priority-preemption/#priorityclass) for Redis and/or Sentinel pods, you can set the `priorityClassName` in the redis/sentinel spec, this attribute has no default and depends on the specific cluster configuration. **Note:** the operator doesn't create the referenced `Priority Class` resource. diff --git a/api/redisfailover/v1/deepcopy_test.go b/api/redisfailover/v1/deepcopy_test.go index 1ef27badb..ed0da849e 100644 --- a/api/redisfailover/v1/deepcopy_test.go +++ b/api/redisfailover/v1/deepcopy_test.go @@ -4,6 +4,7 @@ import ( "testing" "github.com/stretchr/testify/assert" + corev1 "k8s.io/api/core/v1" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" "k8s.io/utils/ptr" ) @@ -36,3 +37,23 @@ func TestSentinelSettingsDeepCopyClonesPointerFields(t *testing.T) { assert.Equal(metav1.Duration{Duration: 10}, *original.FailoverTimeout, "mutating the clone must not affect the original") } } + +func TestSettingsDeepCopyClonesEnv(t *testing.T) { + assert := assert.New(t) + env := func() []corev1.EnvVar { + return []corev1.EnvVar{{ + Name: "FROM_SECRET", + ValueFrom: &corev1.EnvVarSource{SecretKeyRef: &corev1.SecretKeySelector{Key: "a"}}, + }} + } + redis := &RedisSettings{Env: env()} + sentinel := &SentinelSettings{Env: env()} + + redisClone := redis.DeepCopy() + sentinelClone := sentinel.DeepCopy() + redisClone.Env[0].ValueFrom.SecretKeyRef.Key = "b" + sentinelClone.Env[0].ValueFrom.SecretKeyRef.Key = "b" + + assert.Equal(env(), redis.Env, "mutating the clone must not affect the original") + assert.Equal(env(), sentinel.Env, "mutating the clone must not affect the original") +} diff --git a/api/redisfailover/v1/types.go b/api/redisfailover/v1/types.go index 68a929550..586ff964e 100644 --- a/api/redisfailover/v1/types.go +++ b/api/redisfailover/v1/types.go @@ -45,6 +45,7 @@ type RedisSettings struct { Replicas int32 `json:"replicas,omitempty"` Port int32 `json:"port,omitempty"` Resources corev1.ResourceRequirements `json:"resources,omitempty"` + Env []corev1.EnvVar `json:"env,omitempty"` CustomConfig []string `json:"customConfig,omitempty"` CustomCommandRenames []RedisCommandRename `json:"customCommandRenames,omitempty"` Command []string `json:"command,omitempty"` @@ -97,6 +98,7 @@ type SentinelSettings struct { ImagePullPolicy corev1.PullPolicy `json:"imagePullPolicy,omitempty"` Replicas int32 `json:"replicas,omitempty"` Resources corev1.ResourceRequirements `json:"resources,omitempty"` + Env []corev1.EnvVar `json:"env,omitempty"` CustomConfig []string `json:"customConfig,omitempty"` Command []string `json:"command,omitempty"` StartupConfigMap string `json:"startupConfigMap,omitempty"` diff --git a/api/redisfailover/v1/zz_generated.deepcopy.go b/api/redisfailover/v1/zz_generated.deepcopy.go index cbe2e21b6..cc8b70f06 100644 --- a/api/redisfailover/v1/zz_generated.deepcopy.go +++ b/api/redisfailover/v1/zz_generated.deepcopy.go @@ -247,6 +247,13 @@ func (in *RedisFailoverStatus) DeepCopy() *RedisFailoverStatus { func (in *RedisSettings) DeepCopyInto(out *RedisSettings) { *out = *in in.Resources.DeepCopyInto(&out.Resources) + if in.Env != nil { + in, out := &in.Env, &out.Env + *out = make([]corev1.EnvVar, len(*in)) + for i := range *in { + (*in)[i].DeepCopyInto(&(*out)[i]) + } + } if in.CustomConfig != nil { in, out := &in.CustomConfig, &out.CustomConfig *out = make([]string, len(*in)) @@ -438,6 +445,13 @@ func (in *SentinelSettings) DeepCopyInto(out *SentinelSettings) { **out = **in } in.Resources.DeepCopyInto(&out.Resources) + if in.Env != nil { + in, out := &in.Env, &out.Env + *out = make([]corev1.EnvVar, len(*in)) + for i := range *in { + (*in)[i].DeepCopyInto(&(*out)[i]) + } + } if in.CustomConfig != nil { in, out := &in.CustomConfig, &out.CustomConfig *out = make([]string, len(*in)) diff --git a/charts/redisoperator/crds/databases.spotahome.com_redisfailovers.yaml b/charts/redisoperator/crds/databases.spotahome.com_redisfailovers.yaml index 08e2edbb3..fe2e28bda 100644 --- a/charts/redisoperator/crds/databases.spotahome.com_redisfailovers.yaml +++ b/charts/redisoperator/crds/databases.spotahome.com_redisfailovers.yaml @@ -1676,6 +1676,163 @@ spec: dnsPolicy: description: DNSPolicy defines how a pod's DNS will be configured. type: string + env: + items: + description: EnvVar represents an environment variable present + in a Container. + properties: + name: + description: |- + Name of the environment variable. + May consist of any printable ASCII characters except '='. + type: string + value: + description: |- + Variable references $(VAR_NAME) are expanded + using the previously defined environment variables in the container and + any service environment variables. If a variable cannot be resolved, + the reference in the input string will be unchanged. Double $$ are reduced + to a single $, which allows for escaping the $(VAR_NAME) syntax: i.e. + "$$(VAR_NAME)" will produce the string literal "$(VAR_NAME)". + Escaped references will never be expanded, regardless of whether the variable + exists or not. + Defaults to "". + type: string + valueFrom: + description: Source for the environment variable's value. + Cannot be used if value is not empty. + properties: + configMapKeyRef: + description: Selects a key of a ConfigMap. + properties: + key: + description: The key to select. + type: string + name: + default: "" + description: |- + Name of the referent. + This field is effectively required, but due to backwards compatibility is + allowed to be empty. Instances of this type with an empty value here are + almost certainly wrong. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#names + type: string + optional: + description: Specify whether the ConfigMap or its + key must be defined + type: boolean + required: + - key + type: object + x-kubernetes-map-type: atomic + fieldRef: + description: |- + Selects a field of the pod: supports metadata.name, metadata.namespace, `metadata.labels['']`, `metadata.annotations['']`, + spec.nodeName, spec.serviceAccountName, status.hostIP, status.podIP, status.podIPs. + properties: + apiVersion: + description: Version of the schema the FieldPath + is written in terms of, defaults to "v1". + type: string + fieldPath: + description: Path of the field to select in the + specified API version. + type: string + required: + - fieldPath + type: object + x-kubernetes-map-type: atomic + fileKeyRef: + description: |- + FileKeyRef selects a key of the env file. + Requires the EnvFiles feature gate to be enabled. + properties: + key: + description: |- + The key within the env file. An invalid key will prevent the pod from starting. + The keys defined within a source may consist of any printable ASCII characters except '='. + During Alpha stage of the EnvFiles feature gate, the key size is limited to 128 characters. + type: string + optional: + default: false + description: |- + Specify whether the file or its key must be defined. If the file or key + does not exist, then the env var is not published. + If optional is set to true and the specified key does not exist, + the environment variable will not be set in the Pod's containers. + + If optional is set to false and the specified key does not exist, + an error will be returned during Pod creation. + type: boolean + path: + description: |- + The path within the volume from which to select the file. + Must be relative and may not contain the '..' path or start with '..'. + type: string + volumeName: + description: The name of the volume mount containing + the env file. + type: string + required: + - key + - path + - volumeName + type: object + x-kubernetes-map-type: atomic + resourceFieldRef: + description: |- + Selects a resource of the container: only resources limits and requests + (limits.cpu, limits.memory, limits.ephemeral-storage, requests.cpu, requests.memory and requests.ephemeral-storage) are currently supported. + properties: + containerName: + description: 'Container name: required for volumes, + optional for env vars' + type: string + divisor: + anyOf: + - type: integer + - type: string + description: Specifies the output format of the + exposed resources, defaults to "1" + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + resource: + description: 'Required: resource to select' + type: string + required: + - resource + type: object + x-kubernetes-map-type: atomic + secretKeyRef: + description: Selects a key of a secret in the pod's + namespace + properties: + key: + description: The key of the secret to select from. Must + be a valid secret key. + type: string + name: + default: "" + description: |- + Name of the referent. + This field is effectively required, but due to backwards compatibility is + allowed to be empty. Instances of this type with an empty value here are + almost certainly wrong. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#names + type: string + optional: + description: Specify whether the Secret or its key + must be defined + type: boolean + required: + - key + type: object + x-kubernetes-map-type: atomic + type: object + required: + - name + type: object + type: array exporter: description: Exporter defines the specification for the redis/sentinel exporter @@ -9999,6 +10156,163 @@ spec: Enabled controls whether Sentinel is deployed. When false, the operator manages failover instead of Sentinel. Defaults to true. type: boolean + env: + items: + description: EnvVar represents an environment variable present + in a Container. + properties: + name: + description: |- + Name of the environment variable. + May consist of any printable ASCII characters except '='. + type: string + value: + description: |- + Variable references $(VAR_NAME) are expanded + using the previously defined environment variables in the container and + any service environment variables. If a variable cannot be resolved, + the reference in the input string will be unchanged. Double $$ are reduced + to a single $, which allows for escaping the $(VAR_NAME) syntax: i.e. + "$$(VAR_NAME)" will produce the string literal "$(VAR_NAME)". + Escaped references will never be expanded, regardless of whether the variable + exists or not. + Defaults to "". + type: string + valueFrom: + description: Source for the environment variable's value. + Cannot be used if value is not empty. + properties: + configMapKeyRef: + description: Selects a key of a ConfigMap. + properties: + key: + description: The key to select. + type: string + name: + default: "" + description: |- + Name of the referent. + This field is effectively required, but due to backwards compatibility is + allowed to be empty. Instances of this type with an empty value here are + almost certainly wrong. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#names + type: string + optional: + description: Specify whether the ConfigMap or its + key must be defined + type: boolean + required: + - key + type: object + x-kubernetes-map-type: atomic + fieldRef: + description: |- + Selects a field of the pod: supports metadata.name, metadata.namespace, `metadata.labels['']`, `metadata.annotations['']`, + spec.nodeName, spec.serviceAccountName, status.hostIP, status.podIP, status.podIPs. + properties: + apiVersion: + description: Version of the schema the FieldPath + is written in terms of, defaults to "v1". + type: string + fieldPath: + description: Path of the field to select in the + specified API version. + type: string + required: + - fieldPath + type: object + x-kubernetes-map-type: atomic + fileKeyRef: + description: |- + FileKeyRef selects a key of the env file. + Requires the EnvFiles feature gate to be enabled. + properties: + key: + description: |- + The key within the env file. An invalid key will prevent the pod from starting. + The keys defined within a source may consist of any printable ASCII characters except '='. + During Alpha stage of the EnvFiles feature gate, the key size is limited to 128 characters. + type: string + optional: + default: false + description: |- + Specify whether the file or its key must be defined. If the file or key + does not exist, then the env var is not published. + If optional is set to true and the specified key does not exist, + the environment variable will not be set in the Pod's containers. + + If optional is set to false and the specified key does not exist, + an error will be returned during Pod creation. + type: boolean + path: + description: |- + The path within the volume from which to select the file. + Must be relative and may not contain the '..' path or start with '..'. + type: string + volumeName: + description: The name of the volume mount containing + the env file. + type: string + required: + - key + - path + - volumeName + type: object + x-kubernetes-map-type: atomic + resourceFieldRef: + description: |- + Selects a resource of the container: only resources limits and requests + (limits.cpu, limits.memory, limits.ephemeral-storage, requests.cpu, requests.memory and requests.ephemeral-storage) are currently supported. + properties: + containerName: + description: 'Container name: required for volumes, + optional for env vars' + type: string + divisor: + anyOf: + - type: integer + - type: string + description: Specifies the output format of the + exposed resources, defaults to "1" + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + resource: + description: 'Required: resource to select' + type: string + required: + - resource + type: object + x-kubernetes-map-type: atomic + secretKeyRef: + description: Selects a key of a secret in the pod's + namespace + properties: + key: + description: The key of the secret to select from. Must + be a valid secret key. + type: string + name: + default: "" + description: |- + Name of the referent. + This field is effectively required, but due to backwards compatibility is + allowed to be empty. Instances of this type with an empty value here are + almost certainly wrong. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#names + type: string + optional: + description: Specify whether the Secret or its key + must be defined + type: boolean + required: + - key + type: object + x-kubernetes-map-type: atomic + type: object + required: + - name + type: object + type: array exporter: description: Exporter defines the specification for the redis/sentinel exporter diff --git a/manifests/databases.spotahome.com_redisfailovers.yaml b/manifests/databases.spotahome.com_redisfailovers.yaml index 08e2edbb3..fe2e28bda 100644 --- a/manifests/databases.spotahome.com_redisfailovers.yaml +++ b/manifests/databases.spotahome.com_redisfailovers.yaml @@ -1676,6 +1676,163 @@ spec: dnsPolicy: description: DNSPolicy defines how a pod's DNS will be configured. type: string + env: + items: + description: EnvVar represents an environment variable present + in a Container. + properties: + name: + description: |- + Name of the environment variable. + May consist of any printable ASCII characters except '='. + type: string + value: + description: |- + Variable references $(VAR_NAME) are expanded + using the previously defined environment variables in the container and + any service environment variables. If a variable cannot be resolved, + the reference in the input string will be unchanged. Double $$ are reduced + to a single $, which allows for escaping the $(VAR_NAME) syntax: i.e. + "$$(VAR_NAME)" will produce the string literal "$(VAR_NAME)". + Escaped references will never be expanded, regardless of whether the variable + exists or not. + Defaults to "". + type: string + valueFrom: + description: Source for the environment variable's value. + Cannot be used if value is not empty. + properties: + configMapKeyRef: + description: Selects a key of a ConfigMap. + properties: + key: + description: The key to select. + type: string + name: + default: "" + description: |- + Name of the referent. + This field is effectively required, but due to backwards compatibility is + allowed to be empty. Instances of this type with an empty value here are + almost certainly wrong. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#names + type: string + optional: + description: Specify whether the ConfigMap or its + key must be defined + type: boolean + required: + - key + type: object + x-kubernetes-map-type: atomic + fieldRef: + description: |- + Selects a field of the pod: supports metadata.name, metadata.namespace, `metadata.labels['']`, `metadata.annotations['']`, + spec.nodeName, spec.serviceAccountName, status.hostIP, status.podIP, status.podIPs. + properties: + apiVersion: + description: Version of the schema the FieldPath + is written in terms of, defaults to "v1". + type: string + fieldPath: + description: Path of the field to select in the + specified API version. + type: string + required: + - fieldPath + type: object + x-kubernetes-map-type: atomic + fileKeyRef: + description: |- + FileKeyRef selects a key of the env file. + Requires the EnvFiles feature gate to be enabled. + properties: + key: + description: |- + The key within the env file. An invalid key will prevent the pod from starting. + The keys defined within a source may consist of any printable ASCII characters except '='. + During Alpha stage of the EnvFiles feature gate, the key size is limited to 128 characters. + type: string + optional: + default: false + description: |- + Specify whether the file or its key must be defined. If the file or key + does not exist, then the env var is not published. + If optional is set to true and the specified key does not exist, + the environment variable will not be set in the Pod's containers. + + If optional is set to false and the specified key does not exist, + an error will be returned during Pod creation. + type: boolean + path: + description: |- + The path within the volume from which to select the file. + Must be relative and may not contain the '..' path or start with '..'. + type: string + volumeName: + description: The name of the volume mount containing + the env file. + type: string + required: + - key + - path + - volumeName + type: object + x-kubernetes-map-type: atomic + resourceFieldRef: + description: |- + Selects a resource of the container: only resources limits and requests + (limits.cpu, limits.memory, limits.ephemeral-storage, requests.cpu, requests.memory and requests.ephemeral-storage) are currently supported. + properties: + containerName: + description: 'Container name: required for volumes, + optional for env vars' + type: string + divisor: + anyOf: + - type: integer + - type: string + description: Specifies the output format of the + exposed resources, defaults to "1" + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + resource: + description: 'Required: resource to select' + type: string + required: + - resource + type: object + x-kubernetes-map-type: atomic + secretKeyRef: + description: Selects a key of a secret in the pod's + namespace + properties: + key: + description: The key of the secret to select from. Must + be a valid secret key. + type: string + name: + default: "" + description: |- + Name of the referent. + This field is effectively required, but due to backwards compatibility is + allowed to be empty. Instances of this type with an empty value here are + almost certainly wrong. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#names + type: string + optional: + description: Specify whether the Secret or its key + must be defined + type: boolean + required: + - key + type: object + x-kubernetes-map-type: atomic + type: object + required: + - name + type: object + type: array exporter: description: Exporter defines the specification for the redis/sentinel exporter @@ -9999,6 +10156,163 @@ spec: Enabled controls whether Sentinel is deployed. When false, the operator manages failover instead of Sentinel. Defaults to true. type: boolean + env: + items: + description: EnvVar represents an environment variable present + in a Container. + properties: + name: + description: |- + Name of the environment variable. + May consist of any printable ASCII characters except '='. + type: string + value: + description: |- + Variable references $(VAR_NAME) are expanded + using the previously defined environment variables in the container and + any service environment variables. If a variable cannot be resolved, + the reference in the input string will be unchanged. Double $$ are reduced + to a single $, which allows for escaping the $(VAR_NAME) syntax: i.e. + "$$(VAR_NAME)" will produce the string literal "$(VAR_NAME)". + Escaped references will never be expanded, regardless of whether the variable + exists or not. + Defaults to "". + type: string + valueFrom: + description: Source for the environment variable's value. + Cannot be used if value is not empty. + properties: + configMapKeyRef: + description: Selects a key of a ConfigMap. + properties: + key: + description: The key to select. + type: string + name: + default: "" + description: |- + Name of the referent. + This field is effectively required, but due to backwards compatibility is + allowed to be empty. Instances of this type with an empty value here are + almost certainly wrong. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#names + type: string + optional: + description: Specify whether the ConfigMap or its + key must be defined + type: boolean + required: + - key + type: object + x-kubernetes-map-type: atomic + fieldRef: + description: |- + Selects a field of the pod: supports metadata.name, metadata.namespace, `metadata.labels['']`, `metadata.annotations['']`, + spec.nodeName, spec.serviceAccountName, status.hostIP, status.podIP, status.podIPs. + properties: + apiVersion: + description: Version of the schema the FieldPath + is written in terms of, defaults to "v1". + type: string + fieldPath: + description: Path of the field to select in the + specified API version. + type: string + required: + - fieldPath + type: object + x-kubernetes-map-type: atomic + fileKeyRef: + description: |- + FileKeyRef selects a key of the env file. + Requires the EnvFiles feature gate to be enabled. + properties: + key: + description: |- + The key within the env file. An invalid key will prevent the pod from starting. + The keys defined within a source may consist of any printable ASCII characters except '='. + During Alpha stage of the EnvFiles feature gate, the key size is limited to 128 characters. + type: string + optional: + default: false + description: |- + Specify whether the file or its key must be defined. If the file or key + does not exist, then the env var is not published. + If optional is set to true and the specified key does not exist, + the environment variable will not be set in the Pod's containers. + + If optional is set to false and the specified key does not exist, + an error will be returned during Pod creation. + type: boolean + path: + description: |- + The path within the volume from which to select the file. + Must be relative and may not contain the '..' path or start with '..'. + type: string + volumeName: + description: The name of the volume mount containing + the env file. + type: string + required: + - key + - path + - volumeName + type: object + x-kubernetes-map-type: atomic + resourceFieldRef: + description: |- + Selects a resource of the container: only resources limits and requests + (limits.cpu, limits.memory, limits.ephemeral-storage, requests.cpu, requests.memory and requests.ephemeral-storage) are currently supported. + properties: + containerName: + description: 'Container name: required for volumes, + optional for env vars' + type: string + divisor: + anyOf: + - type: integer + - type: string + description: Specifies the output format of the + exposed resources, defaults to "1" + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + resource: + description: 'Required: resource to select' + type: string + required: + - resource + type: object + x-kubernetes-map-type: atomic + secretKeyRef: + description: Selects a key of a secret in the pod's + namespace + properties: + key: + description: The key of the secret to select from. Must + be a valid secret key. + type: string + name: + default: "" + description: |- + Name of the referent. + This field is effectively required, but due to backwards compatibility is + allowed to be empty. Instances of this type with an empty value here are + almost certainly wrong. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#names + type: string + optional: + description: Specify whether the Secret or its key + must be defined + type: boolean + required: + - key + type: object + x-kubernetes-map-type: atomic + type: object + required: + - name + type: object + type: array exporter: description: Exporter defines the specification for the redis/sentinel exporter diff --git a/manifests/kustomize/base/databases.spotahome.com_redisfailovers.yaml b/manifests/kustomize/base/databases.spotahome.com_redisfailovers.yaml index 08e2edbb3..fe2e28bda 100644 --- a/manifests/kustomize/base/databases.spotahome.com_redisfailovers.yaml +++ b/manifests/kustomize/base/databases.spotahome.com_redisfailovers.yaml @@ -1676,6 +1676,163 @@ spec: dnsPolicy: description: DNSPolicy defines how a pod's DNS will be configured. type: string + env: + items: + description: EnvVar represents an environment variable present + in a Container. + properties: + name: + description: |- + Name of the environment variable. + May consist of any printable ASCII characters except '='. + type: string + value: + description: |- + Variable references $(VAR_NAME) are expanded + using the previously defined environment variables in the container and + any service environment variables. If a variable cannot be resolved, + the reference in the input string will be unchanged. Double $$ are reduced + to a single $, which allows for escaping the $(VAR_NAME) syntax: i.e. + "$$(VAR_NAME)" will produce the string literal "$(VAR_NAME)". + Escaped references will never be expanded, regardless of whether the variable + exists or not. + Defaults to "". + type: string + valueFrom: + description: Source for the environment variable's value. + Cannot be used if value is not empty. + properties: + configMapKeyRef: + description: Selects a key of a ConfigMap. + properties: + key: + description: The key to select. + type: string + name: + default: "" + description: |- + Name of the referent. + This field is effectively required, but due to backwards compatibility is + allowed to be empty. Instances of this type with an empty value here are + almost certainly wrong. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#names + type: string + optional: + description: Specify whether the ConfigMap or its + key must be defined + type: boolean + required: + - key + type: object + x-kubernetes-map-type: atomic + fieldRef: + description: |- + Selects a field of the pod: supports metadata.name, metadata.namespace, `metadata.labels['']`, `metadata.annotations['']`, + spec.nodeName, spec.serviceAccountName, status.hostIP, status.podIP, status.podIPs. + properties: + apiVersion: + description: Version of the schema the FieldPath + is written in terms of, defaults to "v1". + type: string + fieldPath: + description: Path of the field to select in the + specified API version. + type: string + required: + - fieldPath + type: object + x-kubernetes-map-type: atomic + fileKeyRef: + description: |- + FileKeyRef selects a key of the env file. + Requires the EnvFiles feature gate to be enabled. + properties: + key: + description: |- + The key within the env file. An invalid key will prevent the pod from starting. + The keys defined within a source may consist of any printable ASCII characters except '='. + During Alpha stage of the EnvFiles feature gate, the key size is limited to 128 characters. + type: string + optional: + default: false + description: |- + Specify whether the file or its key must be defined. If the file or key + does not exist, then the env var is not published. + If optional is set to true and the specified key does not exist, + the environment variable will not be set in the Pod's containers. + + If optional is set to false and the specified key does not exist, + an error will be returned during Pod creation. + type: boolean + path: + description: |- + The path within the volume from which to select the file. + Must be relative and may not contain the '..' path or start with '..'. + type: string + volumeName: + description: The name of the volume mount containing + the env file. + type: string + required: + - key + - path + - volumeName + type: object + x-kubernetes-map-type: atomic + resourceFieldRef: + description: |- + Selects a resource of the container: only resources limits and requests + (limits.cpu, limits.memory, limits.ephemeral-storage, requests.cpu, requests.memory and requests.ephemeral-storage) are currently supported. + properties: + containerName: + description: 'Container name: required for volumes, + optional for env vars' + type: string + divisor: + anyOf: + - type: integer + - type: string + description: Specifies the output format of the + exposed resources, defaults to "1" + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + resource: + description: 'Required: resource to select' + type: string + required: + - resource + type: object + x-kubernetes-map-type: atomic + secretKeyRef: + description: Selects a key of a secret in the pod's + namespace + properties: + key: + description: The key of the secret to select from. Must + be a valid secret key. + type: string + name: + default: "" + description: |- + Name of the referent. + This field is effectively required, but due to backwards compatibility is + allowed to be empty. Instances of this type with an empty value here are + almost certainly wrong. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#names + type: string + optional: + description: Specify whether the Secret or its key + must be defined + type: boolean + required: + - key + type: object + x-kubernetes-map-type: atomic + type: object + required: + - name + type: object + type: array exporter: description: Exporter defines the specification for the redis/sentinel exporter @@ -9999,6 +10156,163 @@ spec: Enabled controls whether Sentinel is deployed. When false, the operator manages failover instead of Sentinel. Defaults to true. type: boolean + env: + items: + description: EnvVar represents an environment variable present + in a Container. + properties: + name: + description: |- + Name of the environment variable. + May consist of any printable ASCII characters except '='. + type: string + value: + description: |- + Variable references $(VAR_NAME) are expanded + using the previously defined environment variables in the container and + any service environment variables. If a variable cannot be resolved, + the reference in the input string will be unchanged. Double $$ are reduced + to a single $, which allows for escaping the $(VAR_NAME) syntax: i.e. + "$$(VAR_NAME)" will produce the string literal "$(VAR_NAME)". + Escaped references will never be expanded, regardless of whether the variable + exists or not. + Defaults to "". + type: string + valueFrom: + description: Source for the environment variable's value. + Cannot be used if value is not empty. + properties: + configMapKeyRef: + description: Selects a key of a ConfigMap. + properties: + key: + description: The key to select. + type: string + name: + default: "" + description: |- + Name of the referent. + This field is effectively required, but due to backwards compatibility is + allowed to be empty. Instances of this type with an empty value here are + almost certainly wrong. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#names + type: string + optional: + description: Specify whether the ConfigMap or its + key must be defined + type: boolean + required: + - key + type: object + x-kubernetes-map-type: atomic + fieldRef: + description: |- + Selects a field of the pod: supports metadata.name, metadata.namespace, `metadata.labels['']`, `metadata.annotations['']`, + spec.nodeName, spec.serviceAccountName, status.hostIP, status.podIP, status.podIPs. + properties: + apiVersion: + description: Version of the schema the FieldPath + is written in terms of, defaults to "v1". + type: string + fieldPath: + description: Path of the field to select in the + specified API version. + type: string + required: + - fieldPath + type: object + x-kubernetes-map-type: atomic + fileKeyRef: + description: |- + FileKeyRef selects a key of the env file. + Requires the EnvFiles feature gate to be enabled. + properties: + key: + description: |- + The key within the env file. An invalid key will prevent the pod from starting. + The keys defined within a source may consist of any printable ASCII characters except '='. + During Alpha stage of the EnvFiles feature gate, the key size is limited to 128 characters. + type: string + optional: + default: false + description: |- + Specify whether the file or its key must be defined. If the file or key + does not exist, then the env var is not published. + If optional is set to true and the specified key does not exist, + the environment variable will not be set in the Pod's containers. + + If optional is set to false and the specified key does not exist, + an error will be returned during Pod creation. + type: boolean + path: + description: |- + The path within the volume from which to select the file. + Must be relative and may not contain the '..' path or start with '..'. + type: string + volumeName: + description: The name of the volume mount containing + the env file. + type: string + required: + - key + - path + - volumeName + type: object + x-kubernetes-map-type: atomic + resourceFieldRef: + description: |- + Selects a resource of the container: only resources limits and requests + (limits.cpu, limits.memory, limits.ephemeral-storage, requests.cpu, requests.memory and requests.ephemeral-storage) are currently supported. + properties: + containerName: + description: 'Container name: required for volumes, + optional for env vars' + type: string + divisor: + anyOf: + - type: integer + - type: string + description: Specifies the output format of the + exposed resources, defaults to "1" + pattern: ^(\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))(([KMGTPE]i)|[numkMGTPE]|([eE](\+|-)?(([0-9]+(\.[0-9]*)?)|(\.[0-9]+))))?$ + x-kubernetes-int-or-string: true + resource: + description: 'Required: resource to select' + type: string + required: + - resource + type: object + x-kubernetes-map-type: atomic + secretKeyRef: + description: Selects a key of a secret in the pod's + namespace + properties: + key: + description: The key of the secret to select from. Must + be a valid secret key. + type: string + name: + default: "" + description: |- + Name of the referent. + This field is effectively required, but due to backwards compatibility is + allowed to be empty. Instances of this type with an empty value here are + almost certainly wrong. + More info: https://kubernetes.io/docs/concepts/overview/working-with-objects/names/#names + type: string + optional: + description: Specify whether the Secret or its key + must be defined + type: boolean + required: + - key + type: object + x-kubernetes-map-type: atomic + type: object + required: + - name + type: object + type: array exporter: description: Exporter defines the specification for the redis/sentinel exporter diff --git a/operator/redisfailover/service/generator.go b/operator/redisfailover/service/generator.go index 848343091..82d4f10cd 100644 --- a/operator/redisfailover/service/generator.go +++ b/operator/redisfailover/service/generator.go @@ -558,8 +558,12 @@ func generateRedisStatefulSet(rf *redisfailoverv1.RedisFailover, labels map[stri ss.Spec.Template.Spec.Containers = append(ss.Spec.Template.Spec.Containers, extraContainers...) } + // User-supplied env is placed before the operator-injected vars so that, on + // duplicate names (Kubernetes last-wins), the operator's REDIS_ADDR/PORT/USER/ + // PASSWORD keep precedence and can't be silently overridden. redisEnv := getRedisEnv(rf) - ss.Spec.Template.Spec.Containers[0].Env = append(ss.Spec.Template.Spec.Containers[0].Env, redisEnv...) + mainEnv := append(ss.Spec.Template.Spec.Containers[0].Env, rf.Spec.Redis.Env...) + ss.Spec.Template.Spec.Containers[0].Env = append(mainEnv, redisEnv...) return ss } @@ -650,6 +654,7 @@ func generateSentinelDeployment(rf *redisfailoverv1.RedisFailover, labels map[st Image: rf.Spec.Sentinel.Image, ImagePullPolicy: pullPolicy(rf.Spec.Sentinel.ImagePullPolicy), SecurityContext: getContainerSecurityContext(rf.Spec.Sentinel.ContainerSecurityContext), + Env: rf.Spec.Sentinel.Env, Ports: []corev1.ContainerPort{ { Name: "sentinel", diff --git a/operator/redisfailover/service/pod_env_test.go b/operator/redisfailover/service/pod_env_test.go new file mode 100644 index 000000000..1ba454962 --- /dev/null +++ b/operator/redisfailover/service/pod_env_test.go @@ -0,0 +1,89 @@ +package service_test + +import ( + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + appsv1 "k8s.io/api/apps/v1" + corev1 "k8s.io/api/core/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + + "github.com/saremox/redis-operator/log" + "github.com/saremox/redis-operator/metrics" + mK8SService "github.com/saremox/redis-operator/mocks/service/k8s" + rfservice "github.com/saremox/redis-operator/operator/redisfailover/service" +) + +// lastValueOf returns the value of the last env var with the given name, which +// is the one that takes effect under Kubernetes' last-wins semantics. +func lastValueOf(env []corev1.EnvVar, name string) (string, bool) { + value, found := "", false + for _, e := range env { + if e.Name == name { + value, found = e.Value, true + } + } + return value, found +} + +func hasEnv(env []corev1.EnvVar, name, value string) bool { + for _, e := range env { + if e.Name == name && e.Value == value { + return true + } + } + return false +} + +func TestRedisMainContainerEnv(t *testing.T) { + assert := assert.New(t) + + rf := generateRF() + rf.Spec.Redis.Port = 6379 + rf.Spec.Redis.Env = []corev1.EnvVar{ + {Name: "MY_CUSTOM", Value: "hello"}, + // Duplicates an operator-injected var; the operator value must still win. + {Name: "REDIS_PORT", Value: "1"}, + } + + var got []corev1.EnvVar + ms := &mK8SService.Services{} + ms.On("CreateOrUpdatePodDisruptionBudget", namespace, mock.Anything).Once().Return(nil, nil) + ms.On("CreateOrUpdateStatefulSet", namespace, mock.Anything).Once().Run(func(args mock.Arguments) { + got = args.Get(1).(*appsv1.StatefulSet).Spec.Template.Spec.Containers[0].Env + }).Return(nil) + + client := rfservice.NewRedisFailoverKubeClient(ms, log.Dummy, metrics.Dummy) + err := client.EnsureRedisStatefulset(rf, nil, []metav1.OwnerReference{}) + + assert.NoError(err) + assert.True(hasEnv(got, "MY_CUSTOM", "hello"), "custom user env must be injected") + // Operator REDIS_PORT (6379) is appended after the user's, so it wins. + port, found := lastValueOf(got, "REDIS_PORT") + assert.True(found) + assert.Equal("6379", port, "operator-injected env must take precedence over a user duplicate") +} + +func TestSentinelMainContainerEnv(t *testing.T) { + assert := assert.New(t) + + rf := generateRF() + rf.Spec.Sentinel.Env = []corev1.EnvVar{ + {Name: "MY_CUSTOM", Value: "world"}, + } + + var got []corev1.EnvVar + ms := &mK8SService.Services{} + ms.On("CreateOrUpdatePodDisruptionBudget", namespace, mock.Anything).Once().Return(nil, nil) + ms.On("CreateOrUpdateServiceAccount", namespace, mock.Anything).Once().Return(nil) + ms.On("CreateOrUpdateDeployment", namespace, mock.Anything).Once().Run(func(args mock.Arguments) { + got = args.Get(1).(*appsv1.Deployment).Spec.Template.Spec.Containers[0].Env + }).Return(nil) + + client := rfservice.NewRedisFailoverKubeClient(ms, log.Dummy, metrics.Dummy) + err := client.EnsureSentinelDeployment(rf, nil, []metav1.OwnerReference{}) + + assert.NoError(err) + assert.True(hasEnv(got, "MY_CUSTOM", "world"), "custom user env must be injected into sentinel") +} From f43873c919d0d0231397ba7ffc91c9b2151c7946 Mon Sep 17 00:00:00 2001 From: Michael Strassberger Date: Sat, 26 Sep 2026 10:37:35 +0200 Subject: [PATCH 18/24] Update dependabot.yml to remove Kubernetes ignore rule Removed ignore rule for Kubernetes dependencies in Dependabot configuration. --- .github/dependabot.yml | 3 --- 1 file changed, 3 deletions(-) diff --git a/.github/dependabot.yml b/.github/dependabot.yml index c34da64db..d30f881ef 100644 --- a/.github/dependabot.yml +++ b/.github/dependabot.yml @@ -4,9 +4,6 @@ updates: directory: "/" schedule: interval: "daily" - ignore: - # Ignore Kubernetes dependencies to have full control on them. - - dependency-name: "k8s.io/*" - package-ecosystem: "github-actions" directory: "/" schedule: From 616fcfd0b876fff5827aa41d4e725441b593a2df Mon Sep 17 00:00:00 2001 From: Michael Strassberger Date: Sat, 26 Sep 2026 23:30:30 +0200 Subject: [PATCH 19/24] feat(redis): operator-managed maxmemory and maxmemory-policy (#190) Add opt-in redis.maxMemory {percent, policy}: maxmemory is min(limit * percent / 100, limit - 32Mi) of the redis container's memory limit, percent defaults to 75 and policy to noeviction. Applied at runtime with CONFIG SET, so enabling it rolls no pods. - Upgrade safety: keys set in customConfig take precedence, so instances are unaffected until they opt in. Without a memory limit >= 64Mi, maxmemory is not managed and the reason is set in the status message instead of stopping the reconcile. replica-ignore-maxmemory no is rejected, and running pods are set to yes as removing it from customConfig does not reset it. - The target follows the smallest pod's limit, capped by the configured one (applied limit from the pod status, else the pod spec): replicas hold the whole dataset and any of them can be promoted. With OnDelete a raised limit applies once every pod runs with it, a lowered one before smaller pods are rolled out. While a pod runs below 64Mi, maxmemory is left alone. - maxmemory is only lowered below the memory in use under allkeys-* (volatile-* could evict every key with a TTL and still not fit), where it stays while Redis evicts down to it. Each lowering is verified on the master and rolled back if it stopped being master or, without eviction, the usage does not fit. A master without maxmemory, e.g. after a restart, whose data does not fit the target is capped at the smallest running container less 32Mi, or at the memory in use. - While a stale pod is about to be replaced with a smaller limit and the master's maxmemory or memory in use does not fit it yet, the rollout is held and the reason is set in the status message. Pods count as stale until the StatefulSet controller has observed the latest spec. - When raising, maxmemory is set before the policy; when lowering, after. Bootstrap mode applies managed maxmemory before replacing pods. Claude-Session: https://claude.ai/code/session_01G4mmyo3sqLwf23m5FE2nBE Co-authored-by: Claude --- README.md | 19 + api/redisfailover/v1/defaults.go | 2 + api/redisfailover/v1/maxmemory.go | 93 +++ api/redisfailover/v1/maxmemory_test.go | 105 +++ api/redisfailover/v1/types.go | 15 + api/redisfailover/v1/validate.go | 4 + api/redisfailover/v1/zz_generated.deepcopy.go | 20 + ...atabases.spotahome.com_redisfailovers.yaml | 26 + example/redisfailover/maxmemory.yaml | 15 + ...atabases.spotahome.com_redisfailovers.yaml | 26 + ...atabases.spotahome.com_redisfailovers.yaml | 26 + metrics/metrics.go | 1 + .../service/RedisFailoverHeal.go | 25 + mocks/service/redis/Client.go | 26 + .../apply_custom_config_internal_test.go | 98 +++ operator/redisfailover/checker.go | 88 +- operator/redisfailover/checker_test.go | 66 ++ operator/redisfailover/service/constants.go | 1 + operator/redisfailover/service/generator.go | 2 +- operator/redisfailover/service/heal.go | 1 + operator/redisfailover/service/maxmemory.go | 334 ++++++++ .../redisfailover/service/maxmemory_test.go | 782 ++++++++++++++++++ service/redis/client.go | 59 ++ service/redis/memory_test.go | 91 ++ 24 files changed, 1910 insertions(+), 15 deletions(-) create mode 100644 api/redisfailover/v1/maxmemory.go create mode 100644 api/redisfailover/v1/maxmemory_test.go create mode 100644 example/redisfailover/maxmemory.yaml create mode 100644 operator/redisfailover/service/maxmemory.go create mode 100644 operator/redisfailover/service/maxmemory_test.go create mode 100644 service/redis/memory_test.go diff --git a/README.md b/README.md index a2c14461a..80d153bf6 100644 --- a/README.md +++ b/README.md @@ -189,6 +189,25 @@ To have the ability of this configuration to be changed "on the fly," without th **Important 2**: do **NOT** change the options used for control the redis/sentinel such as `port`, `bind`, `dir`, etc. +### Managed maxmemory + +With `redis.maxMemory` the operator sets `maxmemory` and `maxmemory-policy` from the redis container's memory limit, see the [maxmemory example file](example/redisfailover/maxmemory.yaml). It requires a memory limit of at least 64Mi; otherwise `maxmemory` is not managed and the reason is in the status message. + +`maxmemory` is `percent` (default `75`) of the limit, keeping at least 32Mi free. `policy` defaults to `noeviction`. + +| Limit | maxmemory | +|---|---| +| 64Mi | 32Mi | +| 96Mi | 64Mi | +| 128Mi | 96Mi | +| 1Gi | 768Mi | + +Keys set in `customConfig` take precedence; `replica-ignore-maxmemory no` is rejected, as replicas would evict on their own, and running pods are set to `yes`. To migrate, update the CRD and the operator, add `maxMemory`, then remove `maxmemory` and `maxmemory-policy` from `customConfig`. Removing `maxMemory` leaves the running pods at their current values until they are replaced; set them in `customConfig` to keep them. + +`maxmemory` follows the smallest redis pod, as replicas hold the whole dataset and any of them can be promoted: a raised limit applies once every pod runs with it, a lowered one before the pods are replaced. `maxmemory` is only lowered below the memory in use under an `allkeys-*` policy, as `volatile-*` could evict every key with a TTL and still not fit; otherwise it is kept and the reason is in the status message. Until the data fits, the operator does not replace pods with the smaller limit. Pods recreated for other reasons, e.g. a node drain, get the smaller limit anyway. + +For small instances, the default `client-output-buffer-limit` for `pubsub` (32mb) and `replica` (256mb) can exceed the free part of the limit; lower them with `customConfig`. Replicas buffer a whole `MULTI`/`EXEC` or `EVAL` before applying it, so one large batch can get a replica OOM-killed. + ### Custom shutdown script By default, a custom shutdown file is given. This file makes redis to `SAVE` it's data, and when Sentinel is enabled and redis is master, it'll call sentinel to ask for failover. diff --git a/api/redisfailover/v1/defaults.go b/api/redisfailover/v1/defaults.go index 18fd9699e..fd930ea81 100644 --- a/api/redisfailover/v1/defaults.go +++ b/api/redisfailover/v1/defaults.go @@ -13,6 +13,8 @@ const ( defaultExporterImage = "quay.io/oliver006/redis_exporter:v1.80.0-alpine" defaultImage = "redis:7.2.12-alpine" defaultRedisPort = 6379 + defaultMaxMemoryPercent = 75 + defaultMaxMemoryPolicy = "noeviction" HealthyState = "Healthy" NotHealthyState = "NotHealthy" ) diff --git a/api/redisfailover/v1/maxmemory.go b/api/redisfailover/v1/maxmemory.go new file mode 100644 index 000000000..0bcf96c13 --- /dev/null +++ b/api/redisfailover/v1/maxmemory.go @@ -0,0 +1,93 @@ +package v1 + +import ( + "fmt" + "strings" + + corev1 "k8s.io/api/core/v1" +) + +const ( + // MaxMemoryReserve is the minimum headroom below the limit, which matters + // more than Percent for small limits. + MaxMemoryReserve = 32 << 20 + // MinManagedMemoryLimit is the smallest memory limit managed mode accepts. + MinManagedMemoryLimit = 64 << 20 +) + +var maxMemoryPolicies = []string{ + "noeviction", + "allkeys-lru", "allkeys-lfu", "allkeys-random", + "volatile-lru", "volatile-lfu", "volatile-random", "volatile-ttl", +} + +// MaxMemoryFor returns the managed maxmemory for a memory limit, or 0 when +// maxmemory is not managed. +func (r *RedisFailover) MaxMemoryFor(limit int64) int64 { + mm := r.Spec.Redis.MaxMemory + if mm == nil { + return 0 + } + target := limit * int64(mm.Percent) / 100 + if capped := limit - MaxMemoryReserve; capped < target { + target = capped + } + return target +} + +// CustomConfigSets reports whether redis customConfig sets the given key. It +// parses entries like SetCustomRedisConfig, which skips e.g. a leading space. +func (r *RedisFailover) CustomConfigSets(key string) bool { + for _, c := range r.Spec.Redis.CustomConfig { + if param, _, _ := strings.Cut(c, " "); strings.EqualFold(param, key) { + return true + } + } + return false +} + +func (r *RedisFailover) validateMaxMemory() error { + mm := r.Spec.Redis.MaxMemory + if mm == nil { + return nil + } + if mm.Percent == 0 { + mm.Percent = defaultMaxMemoryPercent + } + if mm.Percent < 10 || mm.Percent > 95 { + return fmt.Errorf("redis.maxMemory.percent %d must be between 10 and 95", mm.Percent) + } + if mm.Policy == "" { + mm.Policy = defaultMaxMemoryPolicy + } + valid := false + for _, p := range maxMemoryPolicies { + valid = valid || mm.Policy == p + } + if !valid { + return fmt.Errorf("redis.maxMemory.policy %q must be one of %s", mm.Policy, strings.Join(maxMemoryPolicies, ", ")) + } + // Replicas enforcing maxmemory would evict on their own and diverge from + // the master. Fatal, as skipping management would leave the managed + // values on the replicas while customConfig enables enforcing them. + for _, c := range r.Spec.Redis.CustomConfig { + if param, value, _ := strings.Cut(c, " "); (strings.EqualFold(param, "replica-ignore-maxmemory") || strings.EqualFold(param, "slave-ignore-maxmemory")) && strings.EqualFold(strings.TrimSpace(value), "no") { + return fmt.Errorf("redis.maxMemory does not support %q in customConfig", c) + } + } + return nil +} + +// ManagedMaxMemoryError reports why maxmemory cannot be managed. Unlike +// Validate errors, which stop the whole reconcile, it only disables managed +// maxmemory: the API server accepts these specs. +func (r *RedisFailover) ManagedMaxMemoryError() error { + limit, ok := r.Spec.Redis.Resources.Limits[corev1.ResourceMemory] + if !ok { + return fmt.Errorf("redis.maxMemory requires redis.resources.limits.memory") + } + if limit.Value() < MinManagedMemoryLimit { + return fmt.Errorf("redis.maxMemory requires redis.resources.limits.memory of at least 64Mi, got %s", limit.String()) + } + return nil +} diff --git a/api/redisfailover/v1/maxmemory_test.go b/api/redisfailover/v1/maxmemory_test.go new file mode 100644 index 000000000..a5eafab5d --- /dev/null +++ b/api/redisfailover/v1/maxmemory_test.go @@ -0,0 +1,105 @@ +package v1 + +import ( + "testing" + + "github.com/stretchr/testify/assert" + corev1 "k8s.io/api/core/v1" + "k8s.io/apimachinery/pkg/api/resource" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" +) + +func rfWithMemoryLimit(limit string, mm *MaxMemorySettings) *RedisFailover { + rf := &RedisFailover{ObjectMeta: metav1.ObjectMeta{Name: "test"}} + if limit != "" { + rf.Spec.Redis.Resources.Limits = corev1.ResourceList{corev1.ResourceMemory: resource.MustParse(limit)} + } + rf.Spec.Redis.MaxMemory = mm + return rf +} + +func TestMaxMemoryFor(t *testing.T) { + tests := []struct { + limit string + percent int32 + expected string + }{ + {"64Mi", 75, "32Mi"}, + {"96Mi", 75, "64Mi"}, + {"128Mi", 75, "96Mi"}, + {"1Gi", 75, "768Mi"}, + {"128Mi", 90, "96Mi"}, + {"1Gi", 50, "512Mi"}, + } + for _, test := range tests { + t.Run(test.limit, func(t *testing.T) { + rf := rfWithMemoryLimit(test.limit, &MaxMemorySettings{Percent: test.percent}) + expected := resource.MustParse(test.expected) + limit := resource.MustParse(test.limit) + assert.Equal(t, expected.Value(), rf.MaxMemoryFor(limit.Value())) + }) + } + assert.Zero(t, rfWithMemoryLimit("1Gi", nil).MaxMemoryFor(1<<30)) +} + +func TestCustomConfigSets(t *testing.T) { + rf := &RedisFailover{} + rf.Spec.Redis.CustomConfig = []string{"maxmemory-samples 10", "MAXMEMORY 100mb", " maxmemory-policy allkeys-lru"} + assert.True(t, rf.CustomConfigSets("maxmemory")) + // Skipped by SetCustomRedisConfig, so not an override. + assert.False(t, rf.CustomConfigSets("maxmemory-policy")) +} + +func TestValidateMaxMemory(t *testing.T) { + tests := []struct { + name string + settings MaxMemorySettings + customConfig []string + expectedError string + }{ + {name: "defaults"}, + {name: "replicas enforcing maxmemory", customConfig: []string{"slave-ignore-maxmemory NO"}, expectedError: `redis.maxMemory does not support "slave-ignore-maxmemory NO" in customConfig`}, + {name: "replicas ignoring maxmemory", customConfig: []string{"replica-ignore-maxmemory yes"}}, + {name: "percent too low", settings: MaxMemorySettings{Percent: 5}, expectedError: "redis.maxMemory.percent 5 must be between 10 and 95"}, + {name: "percent too high", settings: MaxMemorySettings{Percent: 96}, expectedError: "redis.maxMemory.percent 96 must be between 10 and 95"}, + {name: "unknown policy", settings: MaxMemorySettings{Policy: "lru"}, expectedError: `redis.maxMemory.policy "lru" must be one of`}, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + settings := test.settings + // Requirements for managing maxmemory are not validated here. + rf := rfWithMemoryLimit("", &settings) + rf.Spec.Redis.CustomConfig = test.customConfig + err := rf.Validate() + if test.expectedError != "" { + assert.ErrorContains(t, err, test.expectedError) + return + } + assert.NoError(t, err) + assert.Equal(t, MaxMemorySettings{Percent: 75, Policy: "noeviction"}, *rf.Spec.Redis.MaxMemory) + }) + } +} + +func TestManagedMaxMemoryError(t *testing.T) { + tests := []struct { + name string + limit string + expectedError string + }{ + {name: "minimum limit", limit: "64Mi"}, + {name: "no limit", expectedError: "redis.maxMemory requires redis.resources.limits.memory"}, + {name: "limit too small", limit: "63Mi", expectedError: "redis.maxMemory requires redis.resources.limits.memory of at least 64Mi, got 63Mi"}, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + rf := rfWithMemoryLimit(test.limit, &MaxMemorySettings{}) + err := rf.ManagedMaxMemoryError() + if test.expectedError != "" { + assert.EqualError(t, err, test.expectedError) + return + } + assert.NoError(t, err) + }) + } +} diff --git a/api/redisfailover/v1/types.go b/api/redisfailover/v1/types.go index 586ff964e..a7b9b9dc6 100644 --- a/api/redisfailover/v1/types.go +++ b/api/redisfailover/v1/types.go @@ -83,6 +83,21 @@ type RedisSettings struct { // PodDisruptionBudgetMinAvailable overrides the PodDisruptionBudget // minAvailable for the redis pods. Defaults to 2 (or 1 when replicas <= 2). PodDisruptionBudgetMinAvailable *intstr.IntOrString `json:"podDisruptionBudgetMinAvailable,omitempty"` + // MaxMemory lets the operator set maxmemory and maxmemory-policy from the + // redis container's memory limit. Values set in customConfig take precedence. + MaxMemory *MaxMemorySettings `json:"maxMemory,omitempty"` +} + +// MaxMemorySettings configures the operator-managed maxmemory. +type MaxMemorySettings struct { + // Percent of the memory limit used as maxmemory. At least 32Mi of the + // limit is always left free. Defaults to 75. + // +kubebuilder:validation:Minimum=10 + // +kubebuilder:validation:Maximum=95 + Percent int32 `json:"percent,omitempty"` + // Policy is the maxmemory-policy. Defaults to noeviction. + // +kubebuilder:validation:Enum=noeviction;allkeys-lru;allkeys-lfu;allkeys-random;volatile-lru;volatile-lfu;volatile-random;volatile-ttl + Policy string `json:"policy,omitempty"` } // SentinelSettings defines the specification of the sentinel cluster diff --git a/api/redisfailover/v1/validate.go b/api/redisfailover/v1/validate.go index 1dcd214d7..0dbaa8d58 100644 --- a/api/redisfailover/v1/validate.go +++ b/api/redisfailover/v1/validate.go @@ -41,6 +41,10 @@ func (r *RedisFailover) Validate() error { } } + if err := r.validateMaxMemory(); err != nil { + return err + } + if r.Bootstrapping() { if r.Spec.BootstrapNode.Host == "" { return errors.New("BootstrapNode must include a host when provided") diff --git a/api/redisfailover/v1/zz_generated.deepcopy.go b/api/redisfailover/v1/zz_generated.deepcopy.go index cc8b70f06..78e1446aa 100644 --- a/api/redisfailover/v1/zz_generated.deepcopy.go +++ b/api/redisfailover/v1/zz_generated.deepcopy.go @@ -126,6 +126,21 @@ func (in *Exporter) DeepCopy() *Exporter { return out } +// DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. +func (in *MaxMemorySettings) DeepCopyInto(out *MaxMemorySettings) { + *out = *in +} + +// DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new MaxMemorySettings. +func (in *MaxMemorySettings) DeepCopy() *MaxMemorySettings { + if in == nil { + return nil + } + out := new(MaxMemorySettings) + in.DeepCopyInto(out) + return out +} + // DeepCopyInto is an autogenerated deepcopy function, copying the receiver, writing into out. in must be non-nil. func (in *RedisCommandRename) DeepCopyInto(out *RedisCommandRename) { *out = *in @@ -374,6 +389,11 @@ func (in *RedisSettings) DeepCopyInto(out *RedisSettings) { *out = new(intstr.IntOrString) **out = **in } + if in.MaxMemory != nil { + in, out := &in.MaxMemory, &out.MaxMemory + *out = new(MaxMemorySettings) + **out = **in + } } // DeepCopy is an autogenerated deepcopy function, copying the receiver, creating a new RedisSettings. diff --git a/charts/redisoperator/crds/databases.spotahome.com_redisfailovers.yaml b/charts/redisoperator/crds/databases.spotahome.com_redisfailovers.yaml index fe2e28bda..64c2e04b2 100644 --- a/charts/redisoperator/crds/databases.spotahome.com_redisfailovers.yaml +++ b/charts/redisoperator/crds/databases.spotahome.com_redisfailovers.yaml @@ -7330,6 +7330,32 @@ spec: - name type: object type: array + maxMemory: + description: |- + MaxMemory lets the operator set maxmemory and maxmemory-policy from the + redis container's memory limit. Values set in customConfig take precedence. + properties: + percent: + description: |- + Percent of the memory limit used as maxmemory. At least 32Mi of the + limit is always left free. Defaults to 75. + format: int32 + maximum: 95 + minimum: 10 + type: integer + policy: + description: Policy is the maxmemory-policy. Defaults to noeviction. + enum: + - noeviction + - allkeys-lru + - allkeys-lfu + - allkeys-random + - volatile-lru + - volatile-lfu + - volatile-random + - volatile-ttl + type: string + type: object nodeSelector: additionalProperties: type: string diff --git a/example/redisfailover/maxmemory.yaml b/example/redisfailover/maxmemory.yaml new file mode 100644 index 000000000..d197685f5 --- /dev/null +++ b/example/redisfailover/maxmemory.yaml @@ -0,0 +1,15 @@ +apiVersion: databases.spotahome.com/v1 +kind: RedisFailover +metadata: + name: redisfailover +spec: + redis: + replicas: 3 + resources: + requests: + memory: 128Mi + limits: + memory: 128Mi + maxMemory: + percent: 75 + policy: noeviction diff --git a/manifests/databases.spotahome.com_redisfailovers.yaml b/manifests/databases.spotahome.com_redisfailovers.yaml index fe2e28bda..64c2e04b2 100644 --- a/manifests/databases.spotahome.com_redisfailovers.yaml +++ b/manifests/databases.spotahome.com_redisfailovers.yaml @@ -7330,6 +7330,32 @@ spec: - name type: object type: array + maxMemory: + description: |- + MaxMemory lets the operator set maxmemory and maxmemory-policy from the + redis container's memory limit. Values set in customConfig take precedence. + properties: + percent: + description: |- + Percent of the memory limit used as maxmemory. At least 32Mi of the + limit is always left free. Defaults to 75. + format: int32 + maximum: 95 + minimum: 10 + type: integer + policy: + description: Policy is the maxmemory-policy. Defaults to noeviction. + enum: + - noeviction + - allkeys-lru + - allkeys-lfu + - allkeys-random + - volatile-lru + - volatile-lfu + - volatile-random + - volatile-ttl + type: string + type: object nodeSelector: additionalProperties: type: string diff --git a/manifests/kustomize/base/databases.spotahome.com_redisfailovers.yaml b/manifests/kustomize/base/databases.spotahome.com_redisfailovers.yaml index fe2e28bda..64c2e04b2 100644 --- a/manifests/kustomize/base/databases.spotahome.com_redisfailovers.yaml +++ b/manifests/kustomize/base/databases.spotahome.com_redisfailovers.yaml @@ -7330,6 +7330,32 @@ spec: - name type: object type: array + maxMemory: + description: |- + MaxMemory lets the operator set maxmemory and maxmemory-policy from the + redis container's memory limit. Values set in customConfig take precedence. + properties: + percent: + description: |- + Percent of the memory limit used as maxmemory. At least 32Mi of the + limit is always left free. Defaults to 75. + format: int32 + maximum: 95 + minimum: 10 + type: integer + policy: + description: Policy is the maxmemory-policy. Defaults to noeviction. + enum: + - noeviction + - allkeys-lru + - allkeys-lfu + - allkeys-random + - volatile-lru + - volatile-lfu + - volatile-random + - volatile-ttl + type: string + type: object nodeSelector: additionalProperties: type: string diff --git a/metrics/metrics.go b/metrics/metrics.go index 140917926..5fc8c58d2 100644 --- a/metrics/metrics.go +++ b/metrics/metrics.go @@ -69,6 +69,7 @@ const ( CHECK_SENTINEL_QUORUM = "SENTINEL_CKQUORUM" SLAVE_IS_READY = "CHECK_IF_SLAVE_IS_READY" GET_REPLICATION_INFO = "GET_REPLICATION_INFO" + GET_MEMORY_INFO = "GET_MEMORY_INFO" DISCONNECT_CLIENTS = "DISCONNECT_CLIENTS_ON_DEMOTED_INSTANCE" ) diff --git a/mocks/operator/redisfailover/service/RedisFailoverHeal.go b/mocks/operator/redisfailover/service/RedisFailoverHeal.go index 958122eb2..7bf6acf34 100644 --- a/mocks/operator/redisfailover/service/RedisFailoverHeal.go +++ b/mocks/operator/redisfailover/service/RedisFailoverHeal.go @@ -6,6 +6,7 @@ import ( mock "github.com/stretchr/testify/mock" v1 "github.com/saremox/redis-operator/api/redisfailover/v1" + service "github.com/saremox/redis-operator/operator/redisfailover/service" ) // RedisFailoverHeal is an autogenerated mock type for the RedisFailoverHeal type @@ -27,6 +28,30 @@ func (_m *RedisFailoverHeal) DeletePod(podName string, rFailover *v1.RedisFailov return r0 } +// EnsureRedisMaxMemory provides a mock function with given fields: rFailover, master, redises +func (_m *RedisFailoverHeal) EnsureRedisMaxMemory(rFailover *v1.RedisFailover, master string, redises []string) (service.MaxMemoryResult, error) { + ret := _m.Called(rFailover, master, redises) + + var r0 service.MaxMemoryResult + var r1 error + if rf, ok := ret.Get(0).(func(*v1.RedisFailover, string, []string) (service.MaxMemoryResult, error)); ok { + return rf(rFailover, master, redises) + } + if rf, ok := ret.Get(0).(func(*v1.RedisFailover, string, []string) service.MaxMemoryResult); ok { + r0 = rf(rFailover, master, redises) + } else { + r0 = ret.Get(0).(service.MaxMemoryResult) + } + + if rf, ok := ret.Get(1).(func(*v1.RedisFailover, string, []string) error); ok { + r1 = rf(rFailover, master, redises) + } else { + r1 = ret.Error(1) + } + + return r0, r1 +} + // MakeMaster provides a mock function with given fields: ip, rFailover func (_m *RedisFailoverHeal) MakeMaster(ip string, rFailover *v1.RedisFailover) error { ret := _m.Called(ip, rFailover) diff --git a/mocks/service/redis/Client.go b/mocks/service/redis/Client.go index 388d04a80..7fab24243 100644 --- a/mocks/service/redis/Client.go +++ b/mocks/service/redis/Client.go @@ -27,6 +27,32 @@ func (_m *Client) DisconnectClients(ip string, port string, password string) err return r0 } +// GetMemoryInfo provides a mock function with given fields: ip, port, password +func (_m *Client) GetMemoryInfo(ip string, port string, password string) (*redis.MemoryInfo, error) { + ret := _m.Called(ip, port, password) + + var r0 *redis.MemoryInfo + var r1 error + if rf, ok := ret.Get(0).(func(string, string, string) (*redis.MemoryInfo, error)); ok { + return rf(ip, port, password) + } + if rf, ok := ret.Get(0).(func(string, string, string) *redis.MemoryInfo); ok { + r0 = rf(ip, port, password) + } else { + if ret.Get(0) != nil { + r0 = ret.Get(0).(*redis.MemoryInfo) + } + } + + if rf, ok := ret.Get(1).(func(string, string, string) error); ok { + r1 = rf(ip, port, password) + } else { + r1 = ret.Error(1) + } + + return r0, r1 +} + // GetNumberSentinelSlavesInMemory provides a mock function with given fields: ip func (_m *Client) GetNumberSentinelSlavesInMemory(ip string) (int32, error) { ret := _m.Called(ip) diff --git a/operator/redisfailover/apply_custom_config_internal_test.go b/operator/redisfailover/apply_custom_config_internal_test.go index 545e3a049..0e9344510 100644 --- a/operator/redisfailover/apply_custom_config_internal_test.go +++ b/operator/redisfailover/apply_custom_config_internal_test.go @@ -15,6 +15,7 @@ import ( "github.com/saremox/redis-operator/metrics" mRFService "github.com/saremox/redis-operator/mocks/operator/redisfailover/service" mK8SService "github.com/saremox/redis-operator/mocks/service/k8s" + rfservice "github.com/saremox/redis-operator/operator/redisfailover/service" ) func newCustomConfigTestRF() *redisfailoverv1.RedisFailover { @@ -73,3 +74,100 @@ func TestApplyRedisCustomConfigAbortsOnConfigError(t *testing.T) { assert.Contains(err.Error(), "unknown parameter") mrfh.AssertExpectations(t) } + +// A blocked maxmemory change is reported in the status message without failing +// the reconcile. +func TestEnsureRedisMaxMemoryReportsBlockedMaxMemory(t *testing.T) { + assert := assert.New(t) + + rf := newCustomConfigTestRF() + rf.Spec.Redis.MaxMemory = &redisfailoverv1.MaxMemorySettings{Percent: 75, Policy: "noeviction"} + rf.Status.State = redisfailoverv1.HealthyState + ips := []string{"10.0.0.1", "10.0.0.2"} + + mrfc := &mRFService.RedisFailoverCheck{} + mrfc.On("GetRedisesIPs", rf).Once().Return(ips, nil) + mrfc.On("GetMasterIP", rf).Once().Return("10.0.0.1", nil) + + mrfh := &mRFService.RedisFailoverHeal{} + mrfh.On("EnsureRedisMaxMemory", rf, "10.0.0.1", ips).Once().Return(rfservice.MaxMemoryResult{Message: "maxmemory kept", HoldRollout: true}, nil) + + handler := NewRedisFailoverHandler(Config{}, &mRFService.RedisFailoverClient{}, mrfc, mrfh, &mK8SService.Services{}, metrics.Dummy, log.Dummy) + + hold, err := handler.ensureRedisMaxMemory(rf, "10.0.0.1") + + assert.NoError(err) + assert.True(hold) + assert.Equal(redisfailoverv1.HealthyState, rf.Status.State) + assert.Equal("maxmemory kept", rf.Status.Message) + mrfh.AssertExpectations(t) +} + +// The master is re-resolved before maxmemory is applied, so a failover earlier +// in the reconcile does not make a replica the memory reference. +func TestEnsureRedisMaxMemoryUsesCurrentMaster(t *testing.T) { + rf := newCustomConfigTestRF() + rf.Spec.Redis.MaxMemory = &redisfailoverv1.MaxMemorySettings{Percent: 75, Policy: "noeviction"} + ips := []string{"10.0.0.1", "10.0.0.2"} + + mrfc := &mRFService.RedisFailoverCheck{} + mrfc.On("GetRedisesIPs", rf).Once().Return(ips, nil) + mrfc.On("GetMasterIP", rf).Once().Return("10.0.0.2", nil) + + mrfh := &mRFService.RedisFailoverHeal{} + mrfh.On("EnsureRedisMaxMemory", rf, "10.0.0.2", ips).Once().Return(rfservice.MaxMemoryResult{}, nil) + + handler := NewRedisFailoverHandler(Config{}, &mRFService.RedisFailoverClient{}, mrfc, mrfh, &mK8SService.Services{}, metrics.Dummy, log.Dummy) + + hold, err := handler.ensureRedisMaxMemory(rf, "10.0.0.1") + assert.NoError(t, err) + assert.False(t, hold) + mrfh.AssertExpectations(t) +} + +// Without a single master, maxmemory and the pod rollout wait for the next reconcile. +func TestEnsureRedisMaxMemorySkippedWithoutMaster(t *testing.T) { + rf := newCustomConfigTestRF() + rf.Spec.Redis.MaxMemory = &redisfailoverv1.MaxMemorySettings{Percent: 75, Policy: "noeviction"} + ips := []string{"10.0.0.1", "10.0.0.2"} + + mrfc := &mRFService.RedisFailoverCheck{} + mrfc.On("GetRedisesIPs", rf).Once().Return(ips, nil) + mrfc.On("GetMasterIP", rf).Once().Return("", errors.New("ambiguous master count")) + + // No EnsureRedisMaxMemory expectation - calling it would panic the mock. + mrfh := &mRFService.RedisFailoverHeal{} + + handler := NewRedisFailoverHandler(Config{}, &mRFService.RedisFailoverClient{}, mrfc, mrfh, &mK8SService.Services{}, metrics.Dummy, log.Dummy) + + hold, err := handler.ensureRedisMaxMemory(rf, "10.0.0.1") + assert.NoError(t, err) + assert.True(t, hold) +} + +func TestEnsureRedisMaxMemoryErrors(t *testing.T) { + rf := newCustomConfigTestRF() + rf.Spec.Redis.MaxMemory = &redisfailoverv1.MaxMemorySettings{Percent: 75, Policy: "noeviction"} + ips := []string{"10.0.0.1"} + + t.Run("listing the redises", func(t *testing.T) { + mrfc := &mRFService.RedisFailoverCheck{} + mrfc.On("GetRedisesIPs", rf).Once().Return(nil, errors.New("boom")) + handler := NewRedisFailoverHandler(Config{}, &mRFService.RedisFailoverClient{}, mrfc, &mRFService.RedisFailoverHeal{}, &mK8SService.Services{}, metrics.Dummy, log.Dummy) + + _, err := handler.ensureRedisMaxMemory(rf, "") + assert.EqualError(t, err, "boom") + }) + + t.Run("applying maxmemory", func(t *testing.T) { + mrfc := &mRFService.RedisFailoverCheck{} + mrfc.On("GetRedisesIPs", rf).Once().Return(ips, nil) + mrfh := &mRFService.RedisFailoverHeal{} + mrfh.On("EnsureRedisMaxMemory", rf, "", ips).Once().Return(rfservice.MaxMemoryResult{}, errors.New("ERR boom")) + handler := NewRedisFailoverHandler(Config{}, &mRFService.RedisFailoverClient{}, mrfc, mrfh, &mK8SService.Services{}, metrics.Dummy, log.Dummy) + + _, err := handler.ensureRedisMaxMemory(rf, "") + assert.EqualError(t, err, "ERR boom") + mrfh.AssertExpectations(t) + }) +} diff --git a/operator/redisfailover/checker.go b/operator/redisfailover/checker.go index 559cf3370..e982c90ae 100644 --- a/operator/redisfailover/checker.go +++ b/operator/redisfailover/checker.go @@ -377,6 +377,10 @@ func (r *RedisFailoverHandler) CheckAndHeal(rf *redisfailoverv1.RedisFailover) e } err = r.applyRedisCustomConfig(rf) + var holdRollout bool + if err == nil { + holdRollout, err = r.ensureRedisMaxMemory(rf, master) + } setRedisCheckerMetrics(r.mClient, "redis", rf.Namespace, rf.Name, metrics.APPLY_REDIS_CONFIG, metrics.NOT_APPLICABLE, err) if err != nil { rf.Status = redisfailoverv1.RedisFailoverStatus{ @@ -386,13 +390,15 @@ func (r *RedisFailoverHandler) CheckAndHeal(rf *redisfailoverv1.RedisFailover) e return err } - err = r.UpdateRedisesPods(rf) - if err != nil { - rf.Status = redisfailoverv1.RedisFailoverStatus{ - State: redisfailoverv1.NotHealthyState, - Message: "unable to update redis PODs", + if !holdRollout { + err = r.UpdateRedisesPods(rf) + if err != nil { + rf.Status = redisfailoverv1.RedisFailoverStatus{ + State: redisfailoverv1.NotHealthyState, + Message: "unable to update redis PODs", + } + return err } - return err } sentinels, err := r.rfChecker.GetSentinelsIPs(rf) @@ -465,6 +471,7 @@ func (r *RedisFailoverHandler) checkAndHealOperatorManagedMode(rf *redisfailover return err } + var master string switch nMasters { case 0: // A master whose pod is being deleted is no longer counted but may @@ -552,6 +559,8 @@ func (r *RedisFailoverHandler) checkAndHealOperatorManagedMode(rf *redisfailover return nil } + master = masterIP + // Master is healthy - ensure all slaves are connected to it err = r.rfChecker.CheckAllSlavesFromMaster(masterIP, rf) setRedisCheckerMetrics(r.mClient, "redis", rf.Namespace, rf.Name, metrics.SLAVE_WRONG_MASTER, metrics.NOT_APPLICABLE, err) @@ -580,6 +589,10 @@ func (r *RedisFailoverHandler) checkAndHealOperatorManagedMode(rf *redisfailover // Apply custom Redis configuration err = r.applyRedisCustomConfig(rf) + var holdRollout bool + if err == nil { + holdRollout, err = r.ensureRedisMaxMemory(rf, master) + } setRedisCheckerMetrics(r.mClient, "redis", rf.Namespace, rf.Name, metrics.APPLY_REDIS_CONFIG, metrics.NOT_APPLICABLE, err) if err != nil { rf.Status = redisfailoverv1.RedisFailoverStatus{ @@ -590,13 +603,15 @@ func (r *RedisFailoverHandler) checkAndHealOperatorManagedMode(rf *redisfailover } // Update stale pods - err = r.UpdateRedisesPods(rf) - if err != nil { - rf.Status = redisfailoverv1.RedisFailoverStatus{ - State: redisfailoverv1.NotHealthyState, - Message: "unable to update redis pods", + if !holdRollout { + err = r.UpdateRedisesPods(rf) + if err != nil { + rf.Status = redisfailoverv1.RedisFailoverStatus{ + State: redisfailoverv1.NotHealthyState, + Message: "unable to update redis pods", + } + return err } - return err } return nil @@ -616,14 +631,26 @@ func (r *RedisFailoverHandler) checkAndHealBootstrapMode(rf *redisfailoverv1.Red return nil } - err := r.UpdateRedisesPods(rf) + // Before UpdateRedisesPods, so a lowered memory limit can hold the rollout. + holdRollout, err := r.ensureRedisMaxMemory(rf, "") if err != nil { + setRedisCheckerMetrics(r.mClient, "redis", rf.Namespace, rf.Name, metrics.APPLY_REDIS_CONFIG, metrics.NOT_APPLICABLE, err) rf.Status = redisfailoverv1.RedisFailoverStatus{ State: redisfailoverv1.NotHealthyState, - Message: "unable to update Redis PODs", + Message: "unable to set Redis maxmemory", } return err } + if !holdRollout { + err = r.UpdateRedisesPods(rf) + if err != nil { + rf.Status = redisfailoverv1.RedisFailoverStatus{ + State: redisfailoverv1.NotHealthyState, + Message: "unable to update Redis PODs", + } + return err + } + } err = r.applyRedisCustomConfig(rf) setRedisCheckerMetrics(r.mClient, "redis", rf.Namespace, rf.Name, metrics.APPLY_REDIS_CONFIG, metrics.NOT_APPLICABLE, err) if err != nil { @@ -708,6 +735,39 @@ func (r *RedisFailoverHandler) applyRedisCustomConfig(rf *redisfailoverv1.RedisF return nil } +// ensureRedisMaxMemory applies the managed maxmemory. master is "" when +// bootstrapping. It reports whether the pod rollout must be held. +func (r *RedisFailoverHandler) ensureRedisMaxMemory(rf *redisfailoverv1.RedisFailover, master string) (bool, error) { + if rf.Spec.Redis.MaxMemory == nil { + return false, nil + } + redises, err := r.rfChecker.GetRedisesIPs(rf) + if err != nil { + return false, err + } + logger := r.logger.WithField("redisfailover", rf.Name).WithField("namespace", rf.Namespace) + if master != "" { + // A failover earlier in this reconcile would make the check run against + // a replica. Without a master nothing is checked, so hold the rollout. + if master, err = r.rfChecker.GetMasterIP(rf); err != nil { + logger.Warningf("Skipping maxmemory and the pod rollout, unable to resolve the master: %s", err.Error()) + return true, nil + } + } + result, err := r.rfHealer.EnsureRedisMaxMemory(rf, master, redises) + if err != nil { + return false, err + } + if result.Message != "" { + logger.Warningf("%s", result.Message) + rf.Status.Message = result.Message + } + if result.HoldRollout { + logger.Warningf("Holding the pod rollout until maxmemory fits the lowered memory limit") + } + return result.HoldRollout, nil +} + func (r *RedisFailoverHandler) checkAndHealSentinels(rf *redisfailoverv1.RedisFailover, sentinels []string) error { for _, sip := range sentinels { err := r.rfChecker.CheckSentinelNumberInMemory(sip, rf) diff --git a/operator/redisfailover/checker_test.go b/operator/redisfailover/checker_test.go index 011277719..10c3cacb5 100644 --- a/operator/redisfailover/checker_test.go +++ b/operator/redisfailover/checker_test.go @@ -678,6 +678,24 @@ func TestCheckAndHealOperatorManagedMode(t *testing.T) { wantErr: false, wantState: v1.HealthyState, }, + { + // No UpdateRedisesPods expectations: replacing pods would panic the mock. + name: "single master - maxmemory holds the pod rollout", + setup: func(mrfc *mRFService.RedisFailoverCheck, mrfh *mRFService.RedisFailoverHeal, rf *v1.RedisFailover) { + rf.Spec.Redis.MaxMemory = &v1.MaxMemorySettings{Percent: 75, Policy: "noeviction"} + mrfc.On("IsRedisRunningQuorum", rf).Once().Return(true) + mrfc.On("GetNumberMasters", rf).Once().Return(1, nil) + mrfc.On("CheckMasterHealth", rf).Once().Return(true, master, nil) + mrfc.On("CheckAllSlavesFromMaster", master, rf).Once().Return(nil) + mrfc.On("GetRedisesIPs", rf).Twice().Return([]string{master}, nil) + mrfh.On("SetRedisCustomConfig", master, rf).Once().Return(nil) + mrfc.On("GetMasterIP", rf).Once().Return(master, nil) + mrfh.On("EnsureRedisMaxMemory", rf, master, []string{master}).Once().Return(rfservice.MaxMemoryResult{Message: "maxmemory kept", HoldRollout: true}, nil) + }, + wantErr: false, + wantState: v1.HealthyState, + wantMessage: "maxmemory kept", + }, { name: "single master - healthy, slaves fixed, config and pods updated", setup: func(mrfc *mRFService.RedisFailoverCheck, mrfh *mRFService.RedisFailoverHeal, rf *v1.RedisFailover) { @@ -1062,6 +1080,27 @@ func TestCheckAndHealPlainModeErrorBranches(t *testing.T) { wantState: v1.NotHealthyState, wantMessage: "unable to update redis PODs", }, + { + // No UpdateRedisesPods expectations: replacing pods would panic the mock. + name: "maxmemory holds the pod rollout", + rfMod: func(rf *v1.RedisFailover) { + rf.Spec.Redis.MaxMemory = &v1.MaxMemorySettings{Percent: 75, Policy: "noeviction"} + }, + setup: func(mrfc *mRFService.RedisFailoverCheck, mrfh *mRFService.RedisFailoverHeal, rf *v1.RedisFailover) { + mrfc.On("IsRedisRunningQuorum", rf).Once().Return(true) + mrfc.On("IsSentinelRunningQuorum", rf).Once().Return(true) + mrfc.On("GetNumberMasters", rf).Once().Return(1, nil) + mrfc.On("GetMasterIP", rf).Twice().Return(master, nil) + mrfc.On("CheckAllSlavesFromMaster", master, rf).Once().Return(nil) + mrfc.On("GetRedisesIPs", rf).Twice().Return([]string{master}, nil) + mrfh.On("SetRedisCustomConfig", master, rf).Once().Return(nil) + mrfh.On("EnsureRedisMaxMemory", rf, master, []string{master}).Once().Return(rfservice.MaxMemoryResult{Message: "maxmemory kept", HoldRollout: true}, nil) + mrfc.On("GetSentinelsIPs", rf).Once().Return(nil, errors.New("sentinels ips err")) + }, + wantErr: true, + wantState: v1.NotHealthyState, + wantMessage: "unable to get sentinels IPs", + }, { name: "GetSentinelsIPs fails", setup: func(mrfc *mRFService.RedisFailoverCheck, mrfh *mRFService.RedisFailoverHeal, rf *v1.RedisFailover) { @@ -1326,6 +1365,33 @@ func TestCheckAndHealBootstrapModeErrorBranches(t *testing.T) { wantState: v1.NotHealthyState, wantMessage: "unable to set Redis custom config", }, + { + name: "EnsureRedisMaxMemory fails", + setup: func(mrfc *mRFService.RedisFailoverCheck, mrfh *mRFService.RedisFailoverHeal, rf *v1.RedisFailover) { + rf.Spec.Redis.MaxMemory = &v1.MaxMemorySettings{Percent: 75, Policy: "noeviction"} + mrfc.On("IsRedisRunning", rf).Once().Return(true) + mrfc.On("GetRedisesIPs", rf).Once().Return([]string{bootstrapMaster}, nil) + mrfh.On("EnsureRedisMaxMemory", rf, "", []string{bootstrapMaster}).Once().Return(rfservice.MaxMemoryResult{}, errors.New("maxmemory err")) + }, + wantErr: true, + wantState: v1.NotHealthyState, + wantMessage: "unable to set Redis maxmemory", + }, + { + // No UpdateRedisesPods expectations: replacing pods would panic the mock. + name: "maxmemory holds the pod rollout", + setup: func(mrfc *mRFService.RedisFailoverCheck, mrfh *mRFService.RedisFailoverHeal, rf *v1.RedisFailover) { + rf.Spec.Redis.MaxMemory = &v1.MaxMemorySettings{Percent: 75, Policy: "noeviction"} + mrfc.On("IsRedisRunning", rf).Once().Return(true) + mrfc.On("GetRedisesIPs", rf).Twice().Return([]string{bootstrapMaster}, nil) + mrfh.On("EnsureRedisMaxMemory", rf, "", []string{bootstrapMaster}).Once().Return(rfservice.MaxMemoryResult{Message: "maxmemory kept", HoldRollout: true}, nil) + mrfh.On("SetRedisCustomConfig", bootstrapMaster, rf).Once().Return(nil) + mrfh.On("SetExternalMasterOnAll", bootstrapMaster, bootstrapMasterPort, rf).Once().Return(nil) + }, + wantErr: false, + wantState: v1.HealthyState, + wantMessage: "maxmemory kept", + }, { name: "sentinels allowed but not running", allowSentinels: true, diff --git a/operator/redisfailover/service/constants.go b/operator/redisfailover/service/constants.go index 04da4ab80..4eaa2641c 100644 --- a/operator/redisfailover/service/constants.go +++ b/operator/redisfailover/service/constants.go @@ -25,6 +25,7 @@ const ( redisShutdownName = "r-s" redisReadinessName = "r-readiness" redisRoleName = "redis" + redisContainerName = "redis" sentinelServiceAccountName = "s-sa" appLabel = "redis-failover" hostnameTopologyKey = "kubernetes.io/hostname" diff --git a/operator/redisfailover/service/generator.go b/operator/redisfailover/service/generator.go index 82d4f10cd..835e79be4 100644 --- a/operator/redisfailover/service/generator.go +++ b/operator/redisfailover/service/generator.go @@ -433,7 +433,7 @@ func generateRedisStatefulSet(rf *redisfailoverv1.RedisFailover, labels map[stri TerminationGracePeriodSeconds: &terminationGracePeriodSeconds, Containers: []corev1.Container{ { - Name: "redis", + Name: redisContainerName, Image: rf.Spec.Redis.Image, ImagePullPolicy: pullPolicy(rf.Spec.Redis.ImagePullPolicy), SecurityContext: getContainerSecurityContext(rf.Spec.Redis.ContainerSecurityContext), diff --git a/operator/redisfailover/service/heal.go b/operator/redisfailover/service/heal.go index 17a130212..ba42a426e 100644 --- a/operator/redisfailover/service/heal.go +++ b/operator/redisfailover/service/heal.go @@ -32,6 +32,7 @@ type RedisFailoverHeal interface { SetRedisCustomConfig(ip string, rFailover *redisfailoverv1.RedisFailover) error DeletePod(podName string, rFailover *redisfailoverv1.RedisFailover) error PromoteBestReplica(newMasterIP string, rFailover *redisfailoverv1.RedisFailover) error + EnsureRedisMaxMemory(rFailover *redisfailoverv1.RedisFailover, master string, redises []string) (MaxMemoryResult, error) } // RedisFailoverHealer is our implementation of RedisFailoverCheck interface diff --git a/operator/redisfailover/service/maxmemory.go b/operator/redisfailover/service/maxmemory.go new file mode 100644 index 000000000..64253400b --- /dev/null +++ b/operator/redisfailover/service/maxmemory.go @@ -0,0 +1,334 @@ +package service + +import ( + "errors" + "fmt" + "strings" + "time" + + appsv1 "k8s.io/api/apps/v1" + corev1 "k8s.io/api/core/v1" + + redisfailoverv1 "github.com/saremox/redis-operator/api/redisfailover/v1" + "github.com/saremox/redis-operator/service/k8s" + "github.com/saremox/redis-operator/service/redis" +) + +var ( + evictionWait = 5 * time.Second + evictionPollInterval = 250 * time.Millisecond +) + +// MaxMemoryResult is the outcome of EnsureRedisMaxMemory. +type MaxMemoryResult struct { + // Message explains why maxmemory, or the memory in use, is above its + // target. + Message string + // HoldRollout is set when a lowered memory limit is pending but the + // master's maxmemory or memory in use does not fit it yet. Replicas are + // replaced first and a full sync ignores maxmemory, so they would be + // OOM-killed. + HoldRollout bool +} + +// EnsureRedisMaxMemory applies the managed maxmemory-policy and maxmemory, +// except for keys set in customConfig. master is "" when bootstrapping. +// +// The target derives from the smallest pod's limit, capped by the configured +// one: replicas ignore maxmemory and hold the whole dataset, and a failover can +// promote any of them. With OnDelete, a raised limit therefore applies once +// every pod runs with it, a lowered one before smaller pods are rolled out. +// Lowering below the memory in use needs an allkeys-* policy, which evicts down +// to it. Otherwise it is verified on the master and rolled back if it does not +// fit, because a lowered value left in place would pass as applied on the next +// reconcile. +func (r *RedisFailoverHealer) EnsureRedisMaxMemory(rf *redisfailoverv1.RedisFailover, master string, redises []string) (MaxMemoryResult, error) { + mm := rf.Spec.Redis.MaxMemory + if mm == nil { + return MaxMemoryResult{}, nil + } + if err := rf.ManagedMaxMemoryError(); err != nil { + return MaxMemoryResult{Message: "maxmemory not managed: " + err.Error()}, nil + } + password, err := k8s.GetRedisPassword(r.k8sService, rf) + if err != nil { + return MaxMemoryResult{}, err + } + port := getRedisPort(rf.Spec.Redis.Port) + + policyManaged := !rf.CustomConfigSets("maxmemory-policy") + setPolicy := func() error { + if !policyManaged { + return nil + } + return r.setRedisConfigOn(rf, redises, port, password, "maxmemory-policy "+mm.Policy) + } + + // Removing "replica-ignore-maxmemory no" from customConfig does not + // reset it on running replicas, and bootstrap mode applies customConfig + // afterwards. Validation only lets customConfig set it to yes. + if err := r.setRedisConfigOn(rf, redises, port, password, "replica-ignore-maxmemory yes"); err != nil { + return MaxMemoryResult{}, err + } + if rf.CustomConfigSets("maxmemory") { + return MaxMemoryResult{}, setPolicy() + } + rp, err := r.redisMemoryLimits(rf) + if err != nil { + return MaxMemoryResult{}, err + } + downsizing := rp.downsizing + kept := func(msg string) MaxMemoryResult { + return MaxMemoryResult{Message: msg, HoldRollout: downsizing} + } + if master == "" { + if err := setPolicy(); err != nil { + return MaxMemoryResult{}, err + } + msg, err := r.ensureMaxMemoryWithoutMaster(rf, redises, rp.limits, port, password) + if err != nil || msg == "" { + return MaxMemoryResult{}, err + } + return kept(msg), nil + } + + if _, ok := rp.limits[master]; !ok { + // A master below the managed minimum holds less than any pod it is + // replaced with. + return MaxMemoryResult{}, setPolicy() + } + if rp.smallest < redisfailoverv1.MinManagedMemoryLimit { + // A replica below the managed minimum is replaced before the master. + return MaxMemoryResult{}, setPolicy() + } + target := rf.MaxMemoryFor(rp.smallest) + setMaxMemory := func(ips []string, bytes int64) error { + return r.setRedisConfigOn(rf, ips, port, password, fmt.Sprintf("maxmemory %d", bytes)) + } + current, err := r.redisClient.GetMemoryInfo(master, port, password) + if err != nil { + return MaxMemoryResult{}, err + } + if current.Loading { + // E.g. a restarted master, which would seem to fit a lower value. + return MaxMemoryResult{HoldRollout: downsizing}, setPolicy() + } + policy := current.MaxMemoryPolicy + if policyManaged { + policy = mm.Policy + } + // Only allkeys-* surely evicts enough: volatile-* could evict every key + // with a TTL and still not fit, which restoring maxmemory cannot undo. + evicts := strings.HasPrefix(policy, "allkeys-") + // Without the memory in use, which would change the status on every + // reconcile and trigger the next one. + evicting := fmt.Sprintf("maxmemory lowered to %s, evicting down to it", formatBytes(target)) + + if !lowers(current.MaxMemory, target) { + // Before the policy, so a new evicting policy does not evict what + // fits under the raised value. + if err := setMaxMemory(redises, target); err != nil { + return MaxMemoryResult{}, err + } + if evicts && downsizing && current.UsedMemory > target { + return kept(evicting), setPolicy() + } + return MaxMemoryResult{}, setPolicy() + } + if err := setPolicy(); err != nil { + return MaxMemoryResult{}, err + } + + others := make([]string, 0, len(redises)) + for _, ip := range redises { + if ip != master { + others = append(others, ip) + } + } + notFit := func() string { + return fmt.Sprintf("maxmemory kept at %s: lowering it to %s would not fit the memory in use under policy %s", + formatBytes(current.MaxMemory), formatBytes(target), policy) + } + + if current.UsedMemory > target && !evicts { + if current.MaxMemory == 0 { + // E.g. a restarted master: bounded by the smallest running + // container's reserve, or the memory in use. + current.MaxMemory = max(current.UsedMemory, rp.smallestRunning-redisfailoverv1.MaxMemoryReserve) + if err := r.setRedisConfigOn(rf, []string{master}, port, password, fmt.Sprintf("maxmemory %d", current.MaxMemory)); err != nil { + return MaxMemoryResult{}, err + } + } + // Replicas follow the kept value, so a promoted one behaves the same. + return kept(notFit()), setMaxMemory(others, current.MaxMemory) + } + + // Usage can grow or a failover happen meanwhile. + wait := time.Duration(0) + if evicts { + wait = evictionWait + } + after, err := r.lowerAndWait(master, port, password, target, wait) + if err == nil && after.Role == "master" && (evicts || after.UsedMemory <= target) { + if err := setMaxMemory(others, target); err != nil || after.UsedMemory <= target { + return MaxMemoryResult{}, err + } + // Redis keeps evicting in the background, which can take longer for + // a large dataset. + return kept(evicting), nil + } + if rerr := r.redisClient.SetCustomRedisConfig(master, port, []string{fmt.Sprintf("maxmemory %d", current.MaxMemory)}, password); rerr != nil || err != nil { + return MaxMemoryResult{}, errors.Join(err, rerr) + } + if after.Role != "master" { + // The new master is not known here, and capping it by its own + // container would lower it without checking its usage. + return kept(fmt.Sprintf("maxmemory kept at %s: %s stopped being the master while lowering it", formatBytes(current.MaxMemory), master)), nil + } + return kept(notFit()), setMaxMemory(others, current.MaxMemory) +} + +// ensureMaxMemoryWithoutMaster handles bootstrapping, where every pod is a +// replica. Replicas do not evict, so each one is only lowered when it fits. +func (r *RedisFailoverHealer) ensureMaxMemoryWithoutMaster(rf *redisfailoverv1.RedisFailover, redises []string, limits map[string]int64, port, password string) (string, error) { + msg := "" + for _, ip := range redises { + limit, ok := limits[ip] + if !ok { + continue + } + target := rf.MaxMemoryFor(limit) + mi, err := r.redisClient.GetMemoryInfo(ip, port, password) + if err != nil { + if redis.IsUnreachableError(err) { + continue + } + return "", err + } + if mi.Loading { + continue + } + if lowers(mi.MaxMemory, target) && mi.UsedMemory > target { + msg = fmt.Sprintf("maxmemory kept at %s: lowering it to %s would not fit the memory in use on %s", + formatBytes(mi.MaxMemory), formatBytes(target), ip) + continue + } + if err := r.setRedisConfigOn(rf, []string{ip}, port, password, fmt.Sprintf("maxmemory %d", target)); err != nil { + return "", err + } + } + return msg, nil +} + +// redisPods describes the redis pods by IP. +type redisPods struct { + // limits holds the smaller of the configured and the running redis + // container's limit, preferring the limit the kubelet reports as applied + // over the pod spec. A pod without a limit counts as having the configured + // one, which it gets on replacement; pods below MinManagedMemoryLimit are + // left out until replaced. + limits map[string]int64 + // smallest is the smallest limit, including pods left out of limits; + // smallestRunning the same without capping by the configured limit. + smallest, smallestRunning int64 + // downsizing reports whether a stale pod is about to be replaced with a + // smaller limit. + downsizing bool +} + +func (r *RedisFailoverHealer) redisMemoryLimits(rf *redisfailoverv1.RedisFailover) (redisPods, error) { + pods, err := r.k8sService.GetStatefulSetPods(rf.Namespace, GetRedisName(rf)) + if err != nil { + return redisPods{}, err + } + ss, err := r.k8sService.GetStatefulSet(rf.Namespace, GetRedisName(rf)) + if err != nil { + return redisPods{}, err + } + spec := rf.Spec.Redis.Resources.Limits.Memory().Value() + rp := redisPods{limits: map[string]int64{}} + for _, pod := range pods.Items { + if pod.Status.PodIP == "" { + continue + } + var podLimits corev1.ResourceList + for _, c := range pod.Spec.Containers { + if c.Name == redisContainerName { + podLimits = c.Resources.Limits + } + } + for _, cs := range pod.Status.ContainerStatuses { + if cs.Name == redisContainerName && cs.Resources != nil { + podLimits = cs.Resources.Limits + } + } + limit, running := spec, spec + if memory, ok := podLimits[corev1.ResourceMemory]; ok { + running = memory.Value() + } + // A larger container that is up to date, e.g. resized by a VPA, is + // not replaced. The update revision may still be the previous one + // until the controller observes the latest spec. + stale := ss.Status.ObservedGeneration < ss.Generation || pod.Labels[appsv1.ControllerRevisionHashLabelKey] != ss.Status.UpdateRevision + if memory, ok := podLimits[corev1.ResourceMemory]; ok && memory.Value() < spec { + limit = memory.Value() + } else if stale && (!ok || memory.Value() > spec) { + rp.downsizing = true + } + if limit >= redisfailoverv1.MinManagedMemoryLimit { + rp.limits[pod.Status.PodIP] = limit + } + if rp.smallest == 0 || limit < rp.smallest { + rp.smallest = limit + } + if rp.smallestRunning == 0 || running < rp.smallestRunning { + rp.smallestRunning = running + } + } + return rp, nil +} + +// setRedisConfigOn applies one config to the given redises, skipping unreachable ones. +func (r *RedisFailoverHealer) setRedisConfigOn(rf *redisfailoverv1.RedisFailover, ips []string, port, password, config string) error { + for _, ip := range ips { + if err := r.redisClient.SetCustomRedisConfig(ip, port, []string{config}, password); err != nil { + if redis.IsUnreachableError(err) { + r.logger.WithField("redisfailover", rf.Name).WithField("namespace", rf.Namespace).Warningf("Skipping %s on unreachable redis %s: %s", config, ip, err.Error()) + continue + } + return err + } + } + return nil +} + +// lowerAndWait sets maxmemory on the master and returns its memory info once +// the memory in use fits, or after wait has passed. +func (r *RedisFailoverHealer) lowerAndWait(master, port, password string, target int64, wait time.Duration) (*redis.MemoryInfo, error) { + if err := r.redisClient.SetCustomRedisConfig(master, port, []string{fmt.Sprintf("maxmemory %d", target)}, password); err != nil { + return nil, err + } + deadline := time.Now().Add(wait) + for { + mi, err := r.redisClient.GetMemoryInfo(master, port, password) + if err != nil { + return nil, err + } + if mi.UsedMemory <= target || !time.Now().Before(deadline) { + return mi, nil + } + time.Sleep(evictionPollInterval) + } +} + +// lowers reports whether setting maxmemory to target lowers it; 0 is unlimited. +func lowers(current, target int64) bool { + return current == 0 || target < current +} + +func formatBytes(b int64) string { + if b == 0 { + return "unlimited" + } + return fmt.Sprintf("%.1fMi", float64(b)/(1<<20)) +} diff --git a/operator/redisfailover/service/maxmemory_test.go b/operator/redisfailover/service/maxmemory_test.go new file mode 100644 index 000000000..b1a526bb7 --- /dev/null +++ b/operator/redisfailover/service/maxmemory_test.go @@ -0,0 +1,782 @@ +package service + +import ( + "errors" + "fmt" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + appsv1 "k8s.io/api/apps/v1" + corev1 "k8s.io/api/core/v1" + "k8s.io/apimachinery/pkg/api/resource" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + + redisfailoverv1 "github.com/saremox/redis-operator/api/redisfailover/v1" + "github.com/saremox/redis-operator/log" + mK8SService "github.com/saremox/redis-operator/mocks/service/k8s" + mRedisService "github.com/saremox/redis-operator/mocks/service/redis" + "github.com/saremox/redis-operator/service/redis" +) + +const mi = int64(1 << 20) + +var ( + mmMaster = "10.0.0.1" + mmReplica = "10.0.0.2" + mmRedises = []string{mmMaster, mmReplica} + // 1Gi limit at 75% -> 768Mi + mmTarget = 768 * mi +) + +func rfWithMaxMemory(policy string, customConfig ...string) *redisfailoverv1.RedisFailover { + return &redisfailoverv1.RedisFailover{ + ObjectMeta: metav1.ObjectMeta{Name: "test", Namespace: "testns"}, + Spec: redisfailoverv1.RedisFailoverSpec{ + Redis: redisfailoverv1.RedisSettings{ + Resources: corev1.ResourceRequirements{ + Limits: corev1.ResourceList{corev1.ResourceMemory: resource.MustParse("1Gi")}, + }, + CustomConfig: customConfig, + MaxMemory: &redisfailoverv1.MaxMemorySettings{Percent: 75, Policy: policy}, + }, + }, + } +} + +// redisMock accepts replicas being set to ignore maxmemory, which managed +// maxmemory does on every reconcile. +func redisMock() *mRedisService.Client { + mr := &mRedisService.Client{} + mr.On("SetCustomRedisConfig", mock.Anything, "0", []string{"replica-ignore-maxmemory yes"}, "").Maybe().Return(nil) + return mr +} + +func expectSet(mr *mRedisService.Client, ip, config string) { + mr.On("SetCustomRedisConfig", ip, "0", []string{config}, "").Once().Return(nil) +} + +func maxMemoryConfig(b int64) string { return fmt.Sprintf("maxmemory %d", b) } + +const updateRevision = "new" + +// redisPod returns a stale redis pod with the given memory limit in its spec +// and, when not empty, the limit the kubelet reports as applied. +func redisPod(ip, specLimit, appliedLimit string) corev1.Pod { + pod := corev1.Pod{ + ObjectMeta: metav1.ObjectMeta{Name: "rfr-" + ip, Labels: map[string]string{appsv1.ControllerRevisionHashLabelKey: "old"}}, + Status: corev1.PodStatus{PodIP: ip}, + } + container := corev1.Container{Name: redisContainerName} + if specLimit != "" { + container.Resources.Limits = corev1.ResourceList{corev1.ResourceMemory: resource.MustParse(specLimit)} + } + pod.Spec.Containers = []corev1.Container{{Name: "exporter"}, container} + if appliedLimit != "" { + pod.Status.ContainerStatuses = []corev1.ContainerStatus{{ + Name: redisContainerName, + Resources: &corev1.ResourceRequirements{Limits: corev1.ResourceList{corev1.ResourceMemory: resource.MustParse(appliedLimit)}}, + }} + } + return pod +} + +func upToDate(pod corev1.Pod) corev1.Pod { + pod.Labels = map[string]string{appsv1.ControllerRevisionHashLabelKey: updateRevision} + return pod +} + +func k8sWithPods(pods ...corev1.Pod) *mK8SService.Services { + return k8sWithStatefulSet(&appsv1.StatefulSet{Status: appsv1.StatefulSetStatus{UpdateRevision: updateRevision}}, pods...) +} + +func k8sWithStatefulSet(ss *appsv1.StatefulSet, pods ...corev1.Pod) *mK8SService.Services { + if len(pods) == 0 { + pods = []corev1.Pod{redisPod(mmMaster, "1Gi", ""), redisPod(mmReplica, "1Gi", "")} + } + ms := &mK8SService.Services{} + ms.On("GetStatefulSetPods", "testns", mock.Anything).Return(&corev1.PodList{Items: pods}, nil) + ms.On("GetStatefulSet", "testns", mock.Anything).Return(ss, nil) + return ms +} + +func runEnsureMaxMemory(t *testing.T, rf *redisfailoverv1.RedisFailover, master string, mr *mRedisService.Client, pods ...corev1.Pod) MaxMemoryResult { + t.Helper() + healer := NewRedisFailoverHealer(k8sWithPods(pods...), mr, log.Dummy) + result, err := healer.EnsureRedisMaxMemory(rf, master, mmRedises) + assert.NoError(t, err) + mr.AssertExpectations(t) + return result +} + +// afterLowering is the master's memory info read back after lowering maxmemory. +func afterLowering(used int64) *redis.MemoryInfo { + return &redis.MemoryInfo{UsedMemory: used, Role: "master"} +} + +func TestEnsureRedisMaxMemoryUnmanagedDoesNothing(t *testing.T) { + rf := rfWithMaxMemory("noeviction") + rf.Spec.Redis.MaxMemory = nil + assert.Empty(t, runEnsureMaxMemory(t, rf, mmMaster, &mRedisService.Client{})) +} + +// A spec the API server accepts but that cannot be managed is reported, not applied. +func TestEnsureRedisMaxMemoryReportsUnmanageableSpec(t *testing.T) { + rf := rfWithMaxMemory("noeviction") + rf.Spec.Redis.Resources.Limits = nil + result := runEnsureMaxMemory(t, rf, mmMaster, &mRedisService.Client{}) + assert.Equal(t, MaxMemoryResult{Message: "maxmemory not managed: redis.maxMemory requires redis.resources.limits.memory"}, result) +} + +func expectPolicy(mr *mRedisService.Client, policy string) { + expectSet(mr, mmMaster, "maxmemory-policy "+policy) + expectSet(mr, mmReplica, "maxmemory-policy "+policy) +} + +func expectMemoryInfo(mr *mRedisService.Client, info *redis.MemoryInfo) { + mr.On("GetMemoryInfo", mmMaster, "0", "").Once().Return(info, nil) +} + +func withShortEvictionWait(t *testing.T) { + w, p := evictionWait, evictionPollInterval + evictionWait, evictionPollInterval = 20*time.Millisecond, time.Millisecond + t.Cleanup(func() { evictionWait, evictionPollInterval = w, p }) +} + +func TestEnsureRedisMaxMemoryRaises(t *testing.T) { + tests := map[string]*redis.MemoryInfo{ + "above the current value": {MaxMemory: 512 * mi, UsedMemory: 600 * mi, MaxMemoryPolicy: "noeviction"}, + "already at target": {MaxMemory: mmTarget, UsedMemory: 800 * mi, MaxMemoryPolicy: "noeviction"}, + } + for name, info := range tests { + t.Run(name, func(t *testing.T) { + mr := redisMock() + expectPolicy(mr, "noeviction") + expectMemoryInfo(mr, info) + expectSet(mr, mmMaster, maxMemoryConfig(mmTarget)) + expectSet(mr, mmReplica, maxMemoryConfig(mmTarget)) + + assert.Empty(t, runEnsureMaxMemory(t, rfWithMaxMemory("noeviction"), mmMaster, mr)) + }) + } +} + +// Lowering is verified on the master before the replicas follow. +func TestEnsureRedisMaxMemoryLowers(t *testing.T) { + withShortEvictionWait(t) + tests := map[string]struct { + policy string + current *redis.MemoryInfo + after []int64 + }{ + "from unlimited": {"noeviction", &redis.MemoryInfo{UsedMemory: 100 * mi}, []int64{100 * mi}}, + "when it fits": {"noeviction", &redis.MemoryInfo{MaxMemory: 900 * mi, UsedMemory: 700 * mi}, []int64{700 * mi}}, + "allkeys evicts": {"allkeys-lru", &redis.MemoryInfo{UsedMemory: 800 * mi}, []int64{790 * mi, 700 * mi}}, + } + for name, test := range tests { + t.Run(name, func(t *testing.T) { + test.current.MaxMemoryPolicy = test.policy + mr := redisMock() + expectPolicy(mr, test.policy) + expectMemoryInfo(mr, test.current) + expectSet(mr, mmMaster, maxMemoryConfig(mmTarget)) + for _, used := range test.after { + expectMemoryInfo(mr, afterLowering(used)) + } + expectSet(mr, mmReplica, maxMemoryConfig(mmTarget)) + + assert.Empty(t, runEnsureMaxMemory(t, rfWithMaxMemory(test.policy), mmMaster, mr)) + }) + } +} + +// Without an eviction policy that can free memory, nothing is lowered and the +// replicas follow the master's current value. +func TestEnsureRedisMaxMemoryKeepsValueThatDoesNotFit(t *testing.T) { + tests := map[string]string{ + "noeviction": "noeviction", + // Evicting every key with a TTL might still not fit. + "volatile-lru": "volatile-lru", + } + for name, policy := range tests { + t.Run(name, func(t *testing.T) { + mr := redisMock() + expectPolicy(mr, policy) + expectMemoryInfo(mr, &redis.MemoryInfo{MaxMemory: 900 * mi, UsedMemory: 800 * mi, MaxMemoryPolicy: policy}) + expectSet(mr, mmReplica, maxMemoryConfig(900*mi)) + + result := runEnsureMaxMemory(t, rfWithMaxMemory(policy), mmMaster, mr) + assert.Equal(t, "maxmemory kept at 900.0Mi: lowering it to 768.0Mi would not fit the memory in use under policy "+policy, result.Message) + }) + } +} + +// A lowered value that turns out not to fit is restored on the master. +func TestEnsureRedisMaxMemoryRestoresValueThatDoesNotFit(t *testing.T) { + mr := redisMock() + expectPolicy(mr, "noeviction") + expectMemoryInfo(mr, &redis.MemoryInfo{MaxMemory: 900 * mi, UsedMemory: 700 * mi, MaxMemoryPolicy: "noeviction"}) + expectSet(mr, mmMaster, maxMemoryConfig(mmTarget)) + expectMemoryInfo(mr, afterLowering(790*mi)) + expectSet(mr, mmMaster, maxMemoryConfig(900*mi)) + expectSet(mr, mmReplica, maxMemoryConfig(900*mi)) + + result := runEnsureMaxMemory(t, rfWithMaxMemory("noeviction"), mmMaster, mr) + assert.Equal(t, "maxmemory kept at 900.0Mi: lowering it to 768.0Mi would not fit the memory in use under policy noeviction", result.Message) +} + +// Evicting a large dataset can take longer than the wait; the lowered value +// stays and the replicas follow it. +func TestEnsureRedisMaxMemoryKeepsEvictingDownToTheTarget(t *testing.T) { + withShortEvictionWait(t) + const evicting = "maxmemory lowered to 768.0Mi, evicting down to it" + t.Run("while lowering", func(t *testing.T) { + mr := redisMock() + expectPolicy(mr, "allkeys-lru") + expectMemoryInfo(mr, &redis.MemoryInfo{MaxMemory: 900 * mi, UsedMemory: 800 * mi, MaxMemoryPolicy: "allkeys-lru"}) + expectSet(mr, mmMaster, maxMemoryConfig(mmTarget)) + mr.On("GetMemoryInfo", mmMaster, "0", "").Return(afterLowering(790*mi), nil) + expectSet(mr, mmReplica, maxMemoryConfig(mmTarget)) + + assert.Equal(t, MaxMemoryResult{Message: evicting}, runEnsureMaxMemory(t, rfWithMaxMemory("allkeys-lru"), mmMaster, mr)) + }) + + // A full cache under heavy writes briefly goes above maxmemory. + t.Run("not at the target without a downsize", func(t *testing.T) { + mr := redisMock() + expectMemoryInfo(mr, &redis.MemoryInfo{MaxMemory: mmTarget, UsedMemory: 790 * mi, MaxMemoryPolicy: "allkeys-lru"}) + expectSet(mr, mmMaster, maxMemoryConfig(mmTarget)) + expectSet(mr, mmReplica, maxMemoryConfig(mmTarget)) + expectPolicy(mr, "allkeys-lru") + + assert.Empty(t, runEnsureMaxMemory(t, rfWithMaxMemory("allkeys-lru"), mmMaster, mr)) + }) +} + +// A master loading its data, e.g. after a restart, uses less memory than it +// will once loaded. +func TestEnsureRedisMaxMemoryWaitsForALoadingMaster(t *testing.T) { + t.Run("with a master", func(t *testing.T) { + rf := rfWithMaxMemory("noeviction") + rf.Spec.Redis.Resources.Limits[corev1.ResourceMemory] = resource.MustParse("512Mi") + mr := redisMock() + expectPolicy(mr, "noeviction") + expectMemoryInfo(mr, &redis.MemoryInfo{UsedMemory: 100 * mi, Loading: true}) + + assert.Equal(t, MaxMemoryResult{HoldRollout: true}, runEnsureMaxMemory(t, rf, mmMaster, mr)) + }) + + t.Run("without a master", func(t *testing.T) { + mr := redisMock() + expectPolicy(mr, "noeviction") + expectMemoryInfo(mr, &redis.MemoryInfo{UsedMemory: 100 * mi, Loading: true}) + mr.On("GetMemoryInfo", mmReplica, "0", "").Once().Return(&redis.MemoryInfo{UsedMemory: 100 * mi}, nil) + expectSet(mr, mmReplica, maxMemoryConfig(mmTarget)) + + assert.Empty(t, runEnsureMaxMemory(t, rfWithMaxMemory("noeviction"), "", mr)) + }) +} + +func TestEnsureRedisMaxMemoryCustomConfigTakesPrecedence(t *testing.T) { + t.Run("maxmemory", func(t *testing.T) { + mr := redisMock() + expectPolicy(mr, "noeviction") + + assert.Empty(t, runEnsureMaxMemory(t, rfWithMaxMemory("noeviction", "MaxMemory 100mb"), mmMaster, mr)) + }) + + t.Run("maxmemory-policy", func(t *testing.T) { + mr := redisMock() + expectMemoryInfo(mr, &redis.MemoryInfo{MaxMemory: 512 * mi, MaxMemoryPolicy: "allkeys-lru"}) + expectSet(mr, mmMaster, maxMemoryConfig(mmTarget)) + expectSet(mr, mmReplica, maxMemoryConfig(mmTarget)) + + assert.Empty(t, runEnsureMaxMemory(t, rfWithMaxMemory("noeviction", "maxmemory-policy allkeys-lru", "maxmemory-samples 10"), mmMaster, mr)) + }) +} + +func TestEnsureRedisMaxMemoryWithoutMaster(t *testing.T) { + unreachable := errors.New("dial tcp 10.0.0.2:6379: connect: connection refused") + + t.Run("replicas never evict", func(t *testing.T) { + mr := redisMock() + expectPolicy(mr, "allkeys-lru") + expectMemoryInfo(mr, &redis.MemoryInfo{UsedMemory: 100 * mi}) + expectSet(mr, mmMaster, maxMemoryConfig(mmTarget)) + mr.On("GetMemoryInfo", mmReplica, "0", "").Once().Return(&redis.MemoryInfo{UsedMemory: 800 * mi}, nil) + + result := runEnsureMaxMemory(t, rfWithMaxMemory("allkeys-lru"), "", mr) + assert.Equal(t, "maxmemory kept at unlimited: lowering it to 768.0Mi would not fit the memory in use on 10.0.0.2", result.Message) + }) + + t.Run("unreachable redises are skipped", func(t *testing.T) { + mr := redisMock() + expectSet(mr, mmMaster, "maxmemory-policy noeviction") + mr.On("SetCustomRedisConfig", mmReplica, "0", []string{"maxmemory-policy noeviction"}, "").Once().Return(unreachable) + expectMemoryInfo(mr, &redis.MemoryInfo{UsedMemory: 100 * mi}) + mr.On("GetMemoryInfo", mmReplica, "0", "").Once().Return(nil, unreachable) + expectSet(mr, mmMaster, maxMemoryConfig(mmTarget)) + + assert.Empty(t, runEnsureMaxMemory(t, rfWithMaxMemory("noeviction"), "", mr)) + }) + + t.Run("pods without a known limit are skipped", func(t *testing.T) { + mr := redisMock() + expectPolicy(mr, "noeviction") + expectMemoryInfo(mr, &redis.MemoryInfo{UsedMemory: 100 * mi}) + expectSet(mr, mmMaster, maxMemoryConfig(mmTarget)) + + pending := redisPod("", "1Gi", "") + assert.Empty(t, runEnsureMaxMemory(t, rfWithMaxMemory("noeviction"), "", mr, redisPod(mmMaster, "1Gi", ""), redisPod(mmReplica, "32Mi", ""), pending)) + }) + + t.Run("errors", func(t *testing.T) { + fail := errors.New("ERR boom") + tests := map[string]func(mr *mRedisService.Client){ + "setting the policy": func(mr *mRedisService.Client) { + mr.On("SetCustomRedisConfig", mmMaster, "0", []string{"maxmemory-policy noeviction"}, "").Once().Return(fail) + }, + "reading the memory": func(mr *mRedisService.Client) { + expectPolicy(mr, "noeviction") + mr.On("GetMemoryInfo", mmMaster, "0", "").Once().Return(nil, fail) + }, + "setting maxmemory": func(mr *mRedisService.Client) { + expectPolicy(mr, "noeviction") + expectMemoryInfo(mr, &redis.MemoryInfo{UsedMemory: 100 * mi}) + mr.On("SetCustomRedisConfig", mmMaster, "0", []string{maxMemoryConfig(mmTarget)}, "").Once().Return(fail) + }, + } + for name, setup := range tests { + t.Run(name, func(t *testing.T) { + mr := redisMock() + setup(mr) + healer := NewRedisFailoverHealer(k8sWithPods(), mr, log.Dummy) + _, err := healer.EnsureRedisMaxMemory(rfWithMaxMemory("noeviction"), "", mmRedises) + assert.ErrorIs(t, err, fail) + mr.AssertExpectations(t) + }) + } + }) +} + +func TestEnsureRedisMaxMemoryErrors(t *testing.T) { + withShortEvictionWait(t) + fail := errors.New("ERR boom") + evicting := &redis.MemoryInfo{MaxMemory: 900 * mi, UsedMemory: 800 * mi, MaxMemoryPolicy: "allkeys-lru"} + tests := map[string]func(mr *mRedisService.Client){ + "setting the policy": func(mr *mRedisService.Client) { + expectMemoryInfo(mr, evicting) + mr.On("SetCustomRedisConfig", mmMaster, "0", []string{"maxmemory-policy allkeys-lru"}, "").Once().Return(fail) + }, + "reading the master's memory": func(mr *mRedisService.Client) { + mr.On("GetMemoryInfo", mmMaster, "0", "").Once().Return(nil, fail) + }, + "raising": func(mr *mRedisService.Client) { + expectMemoryInfo(mr, &redis.MemoryInfo{MaxMemory: 512 * mi}) + mr.On("SetCustomRedisConfig", mmMaster, "0", []string{maxMemoryConfig(mmTarget)}, "").Once().Return(fail) + }, + "setting the policy after raising": func(mr *mRedisService.Client) { + expectMemoryInfo(mr, &redis.MemoryInfo{MaxMemory: 512 * mi}) + expectSet(mr, mmMaster, maxMemoryConfig(mmTarget)) + expectSet(mr, mmReplica, maxMemoryConfig(mmTarget)) + mr.On("SetCustomRedisConfig", mmMaster, "0", []string{"maxmemory-policy allkeys-lru"}, "").Once().Return(fail) + }, + // The previous value is restored whenever the attempt fails. + "lowering the master": func(mr *mRedisService.Client) { + expectPolicy(mr, "allkeys-lru") + expectMemoryInfo(mr, evicting) + mr.On("SetCustomRedisConfig", mmMaster, "0", []string{maxMemoryConfig(mmTarget)}, "").Once().Return(fail) + expectSet(mr, mmMaster, maxMemoryConfig(900*mi)) + }, + "verifying the lowered value": func(mr *mRedisService.Client) { + expectPolicy(mr, "allkeys-lru") + expectMemoryInfo(mr, evicting) + expectSet(mr, mmMaster, maxMemoryConfig(mmTarget)) + mr.On("GetMemoryInfo", mmMaster, "0", "").Once().Return(nil, fail) + expectSet(mr, mmMaster, maxMemoryConfig(900*mi)) + }, + "lowering the replicas": func(mr *mRedisService.Client) { + expectPolicy(mr, "allkeys-lru") + expectMemoryInfo(mr, evicting) + expectSet(mr, mmMaster, maxMemoryConfig(mmTarget)) + expectMemoryInfo(mr, afterLowering(700*mi)) + mr.On("SetCustomRedisConfig", mmReplica, "0", []string{maxMemoryConfig(mmTarget)}, "").Once().Return(fail) + }, + "restoring the master": func(mr *mRedisService.Client) { + expectPolicy(mr, "allkeys-lru") + expectMemoryInfo(mr, evicting) + expectSet(mr, mmMaster, maxMemoryConfig(mmTarget)) + expectMemoryInfo(mr, &redis.MemoryInfo{UsedMemory: 700 * mi, Role: "slave"}) + mr.On("SetCustomRedisConfig", mmMaster, "0", []string{maxMemoryConfig(900 * mi)}, "").Once().Return(fail) + }, + } + for name, setup := range tests { + t.Run(name, func(t *testing.T) { + mr := redisMock() + setup(mr) + healer := NewRedisFailoverHealer(k8sWithPods(), mr, log.Dummy) + _, err := healer.EnsureRedisMaxMemory(rfWithMaxMemory("allkeys-lru"), mmMaster, mmRedises) + assert.ErrorIs(t, err, fail) + mr.AssertExpectations(t) + }) + } + + t.Run("listing the pods", func(t *testing.T) { + ms := &mK8SService.Services{} + ms.On("GetStatefulSetPods", "testns", mock.Anything).Once().Return(nil, fail) + healer := NewRedisFailoverHealer(ms, redisMock(), log.Dummy) + _, err := healer.EnsureRedisMaxMemory(rfWithMaxMemory("noeviction"), mmMaster, mmRedises) + assert.ErrorIs(t, err, fail) + }) + + t.Run("reading the statefulset", func(t *testing.T) { + ms := &mK8SService.Services{} + ms.On("GetStatefulSetPods", "testns", mock.Anything).Once().Return(&corev1.PodList{}, nil) + ms.On("GetStatefulSet", "testns", mock.Anything).Once().Return(nil, fail) + healer := NewRedisFailoverHealer(ms, redisMock(), log.Dummy) + _, err := healer.EnsureRedisMaxMemory(rfWithMaxMemory("noeviction"), mmMaster, mmRedises) + assert.ErrorIs(t, err, fail) + }) + + t.Run("reading the password", func(t *testing.T) { + rf := rfWithMaxMemory("noeviction") + rf.Spec.Auth.SecretPath = "redis-auth" + ms := &mK8SService.Services{} + ms.On("GetSecret", "testns", "redis-auth").Once().Return(nil, fail) + healer := NewRedisFailoverHealer(ms, redisMock(), log.Dummy) + _, err := healer.EnsureRedisMaxMemory(rf, mmMaster, mmRedises) + assert.ErrorIs(t, err, fail) + }) +} + +// maxmemory follows the limit of the running master, not the spec: with +// OnDelete the master keeps its old container until it is replaced. +func TestEnsureRedisMaxMemoryFollowsTheRunningMasterLimit(t *testing.T) { + tests := map[string]corev1.Pod{ + "old limit in the pod spec": redisPod(mmMaster, "512Mi", ""), + "applied limit preferred over the spec": redisPod(mmMaster, "1Gi", "512Mi"), + } + for name, master := range tests { + t.Run(name, func(t *testing.T) { + mr := redisMock() + expectPolicy(mr, "noeviction") + expectMemoryInfo(mr, &redis.MemoryInfo{MaxMemory: 256 * mi, UsedMemory: 100 * mi}) + // 512Mi at 75%, not the 768Mi the 1Gi spec would give. + expectSet(mr, mmMaster, maxMemoryConfig(384*mi)) + expectSet(mr, mmReplica, maxMemoryConfig(384*mi)) + + assert.Empty(t, runEnsureMaxMemory(t, rfWithMaxMemory("noeviction"), mmMaster, mr, master, redisPod(mmReplica, "1Gi", ""))) + }) + } + + // A lowered limit applies before the smaller replicas are rolled out. + t.Run("lowered spec limit", func(t *testing.T) { + withShortEvictionWait(t) + rf := rfWithMaxMemory("noeviction") + rf.Spec.Redis.Resources.Limits[corev1.ResourceMemory] = resource.MustParse("512Mi") + mr := redisMock() + expectPolicy(mr, "noeviction") + expectMemoryInfo(mr, &redis.MemoryInfo{MaxMemory: mmTarget, UsedMemory: 100 * mi}) + expectSet(mr, mmMaster, maxMemoryConfig(384*mi)) + expectMemoryInfo(mr, afterLowering(100*mi)) + expectSet(mr, mmReplica, maxMemoryConfig(384*mi)) + + assert.Empty(t, runEnsureMaxMemory(t, rf, mmMaster, mr)) + }) + + t.Run("master still below the managed minimum", func(t *testing.T) { + mr := redisMock() + expectPolicy(mr, "noeviction") + + assert.Empty(t, runEnsureMaxMemory(t, rfWithMaxMemory("noeviction"), mmMaster, mr, redisPod(mmMaster, "32Mi", ""), redisPod(mmReplica, "1Gi", ""))) + }) + + // Without a limit the pod is about to get the configured one. + t.Run("master without a limit", func(t *testing.T) { + mr := redisMock() + expectMemoryInfo(mr, &redis.MemoryInfo{MaxMemory: 512 * mi, UsedMemory: 100 * mi}) + expectSet(mr, mmMaster, maxMemoryConfig(mmTarget)) + expectSet(mr, mmReplica, maxMemoryConfig(mmTarget)) + expectPolicy(mr, "noeviction") + + assert.Empty(t, runEnsureMaxMemory(t, rfWithMaxMemory("noeviction"), mmMaster, mr, redisPod(mmMaster, "", ""), redisPod(mmReplica, "1Gi", ""))) + }) +} + +// When raising, maxmemory goes up before an evicting policy is enabled. +func TestEnsureRedisMaxMemoryRaisesBeforeChangingThePolicy(t *testing.T) { + mr := redisMock() + expectMemoryInfo(mr, &redis.MemoryInfo{MaxMemory: 512 * mi, UsedMemory: 500 * mi, MaxMemoryPolicy: "noeviction"}) + expectSet(mr, mmMaster, maxMemoryConfig(mmTarget)) + expectSet(mr, mmReplica, maxMemoryConfig(mmTarget)) + expectPolicy(mr, "allkeys-lru") + + assert.Empty(t, runEnsureMaxMemory(t, rfWithMaxMemory("allkeys-lru"), mmMaster, mr)) + + var order []string + for _, call := range mr.Calls { + if call.Method == "SetCustomRedisConfig" && call.Arguments.Get(2).([]string)[0] != "replica-ignore-maxmemory yes" { + order = append(order, call.Arguments.Get(2).([]string)[0]) + } + } + assert.Equal(t, []string{ + maxMemoryConfig(mmTarget), maxMemoryConfig(mmTarget), + "maxmemory-policy allkeys-lru", "maxmemory-policy allkeys-lru", + }, order) +} + +// When lowering, the new policy is in place before it is relied on to evict. +func TestEnsureRedisMaxMemoryChangesThePolicyBeforeLowering(t *testing.T) { + withShortEvictionWait(t) + mr := redisMock() + expectMemoryInfo(mr, &redis.MemoryInfo{MaxMemory: 900 * mi, UsedMemory: 800 * mi, MaxMemoryPolicy: "noeviction"}) + expectPolicy(mr, "allkeys-lru") + expectSet(mr, mmMaster, maxMemoryConfig(mmTarget)) + expectMemoryInfo(mr, afterLowering(700*mi)) + expectSet(mr, mmReplica, maxMemoryConfig(mmTarget)) + + assert.Empty(t, runEnsureMaxMemory(t, rfWithMaxMemory("allkeys-lru"), mmMaster, mr)) + var sets []string + for _, call := range mr.Calls { + if call.Method == "SetCustomRedisConfig" { + sets = append(sets, call.Arguments.Get(2).([]string)[0]) + } + } + assert.Equal(t, []string{ + "replica-ignore-maxmemory yes", "replica-ignore-maxmemory yes", + "maxmemory-policy allkeys-lru", "maxmemory-policy allkeys-lru", + maxMemoryConfig(mmTarget), maxMemoryConfig(mmTarget), + }, sets) +} + +// While pods are about to be replaced with a smaller limit, a master that +// cannot be brought down to fit it holds the rollout. +func TestEnsureRedisMaxMemoryHoldsTheRolloutOfALoweredLimit(t *testing.T) { + withShortEvictionWait(t) + lowered := func() *redisfailoverv1.RedisFailover { + rf := rfWithMaxMemory("noeviction") + rf.Spec.Redis.Resources.Limits[corev1.ResourceMemory] = resource.MustParse("512Mi") + return rf + } + + t.Run("dataset does not fit", func(t *testing.T) { + mr := redisMock() + expectPolicy(mr, "noeviction") + expectMemoryInfo(mr, &redis.MemoryInfo{MaxMemory: mmTarget, UsedMemory: 500 * mi}) + expectSet(mr, mmReplica, maxMemoryConfig(mmTarget)) + + result := runEnsureMaxMemory(t, lowered(), mmMaster, mr) + assert.True(t, result.HoldRollout) + assert.Contains(t, result.Message, "would not fit") + }) + + t.Run("master changed while lowering", func(t *testing.T) { + mr := redisMock() + expectPolicy(mr, "noeviction") + expectMemoryInfo(mr, &redis.MemoryInfo{MaxMemory: mmTarget, UsedMemory: 100 * mi}) + expectSet(mr, mmMaster, maxMemoryConfig(384*mi)) + expectMemoryInfo(mr, &redis.MemoryInfo{UsedMemory: 100 * mi, Role: "slave"}) + expectSet(mr, mmMaster, maxMemoryConfig(mmTarget)) + + result := runEnsureMaxMemory(t, lowered(), mmMaster, mr) + assert.True(t, result.HoldRollout) + assert.Equal(t, "maxmemory kept at 768.0Mi: 10.0.0.1 stopped being the master while lowering it", result.Message) + }) + + t.Run("master below the managed minimum", func(t *testing.T) { + mr := redisMock() + expectPolicy(mr, "noeviction") + + // Holding would keep the master from ever being replaced. + result := runEnsureMaxMemory(t, lowered(), mmMaster, mr, redisPod(mmMaster, "32Mi", ""), redisPod(mmReplica, "1Gi", "")) + assert.False(t, result.HoldRollout) + }) + + t.Run("still evicting at the target", func(t *testing.T) { + mr := redisMock() + expectMemoryInfo(mr, &redis.MemoryInfo{MaxMemory: 384 * mi, UsedMemory: 400 * mi}) + expectSet(mr, mmMaster, maxMemoryConfig(384*mi)) + expectSet(mr, mmReplica, maxMemoryConfig(384*mi)) + expectPolicy(mr, "allkeys-lru") + + rf := lowered() + rf.Spec.Redis.MaxMemory.Policy = "allkeys-lru" + assert.True(t, runEnsureMaxMemory(t, rf, mmMaster, mr).HoldRollout) + }) + + t.Run("without a master", func(t *testing.T) { + mr := redisMock() + expectPolicy(mr, "noeviction") + expectMemoryInfo(mr, &redis.MemoryInfo{MaxMemory: mmTarget, UsedMemory: 500 * mi}) + mr.On("GetMemoryInfo", mmReplica, "0", "").Once().Return(&redis.MemoryInfo{MaxMemory: 384 * mi, UsedMemory: 100 * mi}, nil) + expectSet(mr, mmReplica, maxMemoryConfig(384*mi)) + + assert.True(t, runEnsureMaxMemory(t, lowered(), "", mr).HoldRollout) + }) + + t.Run("not while the limit stays", func(t *testing.T) { + mr := redisMock() + expectPolicy(mr, "noeviction") + expectMemoryInfo(mr, &redis.MemoryInfo{MaxMemory: 900 * mi, UsedMemory: 800 * mi}) + expectSet(mr, mmReplica, maxMemoryConfig(900*mi)) + + result := runEnsureMaxMemory(t, rfWithMaxMemory("noeviction"), mmMaster, mr) + assert.False(t, result.HoldRollout) + assert.NotEmpty(t, result.Message) + }) + + // Under noeviction, a full master overshoots maxmemory by the last write. + t.Run("not for a full master at the target", func(t *testing.T) { + mr := redisMock() + expectMemoryInfo(mr, &redis.MemoryInfo{MaxMemory: 384 * mi, UsedMemory: 385 * mi}) + expectSet(mr, mmMaster, maxMemoryConfig(384*mi)) + expectSet(mr, mmReplica, maxMemoryConfig(384*mi)) + expectPolicy(mr, "noeviction") + + assert.Empty(t, runEnsureMaxMemory(t, lowered(), mmMaster, mr)) + }) +} + +// Replicas ignore maxmemory and hold the master's whole dataset, and any of +// them can be promoted, so the smallest container sets maxmemory. +func TestEnsureRedisMaxMemoryFollowsTheSmallestPod(t *testing.T) { + t.Run("replica still on the old limit", func(t *testing.T) { + mr := redisMock() + expectPolicy(mr, "noeviction") + expectMemoryInfo(mr, &redis.MemoryInfo{MaxMemory: 384 * mi, UsedMemory: 100 * mi}) + expectSet(mr, mmMaster, maxMemoryConfig(384*mi)) + expectSet(mr, mmReplica, maxMemoryConfig(384*mi)) + + assert.Empty(t, runEnsureMaxMemory(t, rfWithMaxMemory("noeviction"), mmMaster, mr, upToDate(redisPod(mmMaster, "1Gi", "")), redisPod(mmReplica, "1Gi", "512Mi"))) + }) + + t.Run("replica below the managed minimum", func(t *testing.T) { + mr := redisMock() + expectPolicy(mr, "noeviction") + + assert.Empty(t, runEnsureMaxMemory(t, rfWithMaxMemory("noeviction"), mmMaster, mr, upToDate(redisPod(mmMaster, "1Gi", "")), redisPod(mmReplica, "32Mi", ""))) + }) +} + +// A restarted redis runs without maxmemory; one that does not fit the target +// is capped instead of being left unlimited. +func TestEnsureRedisMaxMemoryCapsAnUnlimitedMaster(t *testing.T) { + tests := map[string]struct{ used, capped int64 }{ + // 1Gi less the 32Mi reserve. + "at the reserve": {800 * mi, 992 * mi}, + "at the memory in use": {1000 * mi, 1000 * mi}, + } + for name, test := range tests { + t.Run(name, func(t *testing.T) { + mr := redisMock() + expectPolicy(mr, "noeviction") + expectMemoryInfo(mr, &redis.MemoryInfo{UsedMemory: test.used}) + expectSet(mr, mmMaster, maxMemoryConfig(test.capped)) + expectSet(mr, mmReplica, maxMemoryConfig(test.capped)) + + result := runEnsureMaxMemory(t, rfWithMaxMemory("noeviction"), mmMaster, mr) + assert.Contains(t, result.Message, "maxmemory kept at "+formatBytes(test.capped)) + }) + } + + t.Run("by the running container during a downsize", func(t *testing.T) { + rf := rfWithMaxMemory("noeviction") + rf.Spec.Redis.Resources.Limits[corev1.ResourceMemory] = resource.MustParse("512Mi") + mr := redisMock() + expectPolicy(mr, "noeviction") + expectMemoryInfo(mr, &redis.MemoryInfo{UsedMemory: 500 * mi}) + expectSet(mr, mmMaster, maxMemoryConfig(992*mi)) + expectSet(mr, mmReplica, maxMemoryConfig(992*mi)) + + assert.True(t, runEnsureMaxMemory(t, rf, mmMaster, mr).HoldRollout) + }) + + t.Run("error", func(t *testing.T) { + fail := errors.New("ERR boom") + mr := redisMock() + expectPolicy(mr, "noeviction") + expectMemoryInfo(mr, &redis.MemoryInfo{UsedMemory: 800 * mi}) + mr.On("SetCustomRedisConfig", mmMaster, "0", []string{maxMemoryConfig(992 * mi)}, "").Once().Return(fail) + _, err := NewRedisFailoverHealer(k8sWithPods(), mr, log.Dummy).EnsureRedisMaxMemory(rfWithMaxMemory("noeviction"), mmMaster, mmRedises) + assert.ErrorIs(t, err, fail) + }) +} + +// Only stale pods are replaced: an up-to-date larger container, e.g. resized +// by a VPA, does not hold the rollout. +func TestEnsureRedisMaxMemoryDoesNotHoldForUpToDatePods(t *testing.T) { + rf := rfWithMaxMemory("noeviction") + rf.Spec.Redis.Resources.Limits[corev1.ResourceMemory] = resource.MustParse("512Mi") + mr := redisMock() + expectPolicy(mr, "noeviction") + expectMemoryInfo(mr, &redis.MemoryInfo{MaxMemory: mmTarget, UsedMemory: 500 * mi}) + expectSet(mr, mmReplica, maxMemoryConfig(mmTarget)) + + result := runEnsureMaxMemory(t, rf, mmMaster, mr, upToDate(redisPod(mmMaster, "1Gi", "")), upToDate(redisPod(mmReplica, "", ""))) + assert.False(t, result.HoldRollout) + assert.Contains(t, result.Message, "would not fit") +} + +// Until the controller observes a lowered limit, its update revision is the +// previous one, which the pods may still be on. +func TestEnsureRedisMaxMemoryHoldsBeforeTheRevisionIsUpdated(t *testing.T) { + rf := rfWithMaxMemory("noeviction") + rf.Spec.Redis.Resources.Limits[corev1.ResourceMemory] = resource.MustParse("512Mi") + mr := redisMock() + expectPolicy(mr, "noeviction") + expectMemoryInfo(mr, &redis.MemoryInfo{MaxMemory: mmTarget, UsedMemory: 500 * mi}) + expectSet(mr, mmReplica, maxMemoryConfig(mmTarget)) + + ss := &appsv1.StatefulSet{ObjectMeta: metav1.ObjectMeta{Generation: 2}, Status: appsv1.StatefulSetStatus{ObservedGeneration: 1, UpdateRevision: updateRevision}} + healer := NewRedisFailoverHealer(k8sWithStatefulSet(ss, upToDate(redisPod(mmMaster, "1Gi", "")), upToDate(redisPod(mmReplica, "1Gi", ""))), mr, log.Dummy) + result, err := healer.EnsureRedisMaxMemory(rf, mmMaster, mmRedises) + assert.NoError(t, err) + assert.True(t, result.HoldRollout) + mr.AssertExpectations(t) +} + +// Removing "replica-ignore-maxmemory no" from customConfig does not reset it +// on running replicas, so managed maxmemory resets it on every reconcile. +func TestEnsureRedisMaxMemoryMakesReplicasIgnoreMaxMemory(t *testing.T) { + t.Run("set", func(t *testing.T) { + mr := &mRedisService.Client{} + expectSet(mr, mmMaster, "replica-ignore-maxmemory yes") + expectSet(mr, mmReplica, "replica-ignore-maxmemory yes") + expectPolicy(mr, "noeviction") + expectMemoryInfo(mr, &redis.MemoryInfo{MaxMemory: mmTarget, UsedMemory: 100 * mi}) + expectSet(mr, mmMaster, maxMemoryConfig(mmTarget)) + expectSet(mr, mmReplica, maxMemoryConfig(mmTarget)) + + assert.Empty(t, runEnsureMaxMemory(t, rfWithMaxMemory("noeviction"), mmMaster, mr)) + }) + + t.Run("with maxmemory in customConfig", func(t *testing.T) { + mr := &mRedisService.Client{} + expectSet(mr, mmMaster, "replica-ignore-maxmemory yes") + expectSet(mr, mmReplica, "replica-ignore-maxmemory yes") + expectPolicy(mr, "noeviction") + + assert.Empty(t, runEnsureMaxMemory(t, rfWithMaxMemory("noeviction", "maxmemory 100mb"), mmMaster, mr)) + }) + + t.Run("also when customConfig sets it", func(t *testing.T) { + mr := &mRedisService.Client{} + expectSet(mr, mmMaster, "replica-ignore-maxmemory yes") + expectSet(mr, mmReplica, "replica-ignore-maxmemory yes") + expectPolicy(mr, "noeviction") + expectMemoryInfo(mr, &redis.MemoryInfo{MaxMemory: mmTarget, UsedMemory: 100 * mi}) + expectSet(mr, mmMaster, maxMemoryConfig(mmTarget)) + expectSet(mr, mmReplica, maxMemoryConfig(mmTarget)) + + assert.Empty(t, runEnsureMaxMemory(t, rfWithMaxMemory("noeviction", "slave-ignore-maxmemory yes"), mmMaster, mr)) + }) + + t.Run("error", func(t *testing.T) { + fail := errors.New("ERR boom") + mr := &mRedisService.Client{} + mr.On("SetCustomRedisConfig", mmMaster, "0", []string{"replica-ignore-maxmemory yes"}, "").Once().Return(fail) + _, err := NewRedisFailoverHealer(k8sWithPods(), mr, log.Dummy).EnsureRedisMaxMemory(rfWithMaxMemory("noeviction"), mmMaster, mmRedises) + assert.ErrorIs(t, err, fail) + }) +} diff --git a/service/redis/client.go b/service/redis/client.go index 3feab37ed..0c33bf045 100644 --- a/service/redis/client.go +++ b/service/redis/client.go @@ -26,6 +26,15 @@ type ReplicationInfo struct { SyncInProgress bool // true if slave is syncing } +// MemoryInfo contains the memory figures of a Redis instance relevant to maxmemory +type MemoryInfo struct { + MaxMemory int64 + MaxMemoryPolicy string + UsedMemory int64 // used_memory minus mem_not_counted_for_evict, as compared against maxmemory + Role string + Loading bool // used_memory does not show the whole dataset yet +} + // Client defines the functions neccesary to connect to redis and sentinel to get or set what we nned type Client interface { GetNumberSentinelsInMemory(ip string) (int32, error) @@ -45,6 +54,7 @@ type Client interface { SlaveIsReady(ip, port, password string) (bool, error) SentinelCheckQuorum(ip string) error GetReplicationInfo(ip, port, password string) (*ReplicationInfo, error) + GetMemoryInfo(ip, port, password string) (*MemoryInfo, error) } type client struct { @@ -737,6 +747,55 @@ func (c *client) GetReplicationInfo(ip, port, password string) (*ReplicationInfo return replInfo, nil } +// GetMemoryInfo returns the maxmemory settings, memory usage and role of a Redis instance. +func (c *client) GetMemoryInfo(ip, port, password string) (*MemoryInfo, error) { + rClient := rediscli.NewClient(&rediscli.Options{ + Addr: net.JoinHostPort(ip, port), + Password: password, + DB: 0, + }) + defer func(rClient *rediscli.Client) { + if err := rClient.Close(); err != nil { + log.Error(err.Error()) + } + }(rClient) + + mi := &MemoryInfo{} + var notCounted int64 + for _, section := range []string{"memory", "replication", "persistence"} { + info, err := rClient.Info(context.TODO(), section).Result() + if err != nil { + c.metricsRecorder.RecordRedisOperation(metrics.KIND_REDIS, ip, metrics.GET_MEMORY_INFO, metrics.FAIL, getRedisError(err)) + return nil, err + } + for _, line := range strings.Split(info, "\n") { + key, value, ok := strings.Cut(strings.TrimSpace(line), ":") + if !ok { + continue + } + n, _ := strconv.ParseInt(value, 10, 64) + switch key { + case "maxmemory": + mi.MaxMemory = n + case "maxmemory_policy": + mi.MaxMemoryPolicy = value + case "role": + mi.Role = value + case "used_memory": + mi.UsedMemory = n + case "mem_not_counted_for_evict": + notCounted = n + case "loading": + mi.Loading = n == 1 + } + } + } + mi.UsedMemory -= notCounted + + c.metricsRecorder.RecordRedisOperation(metrics.KIND_REDIS, ip, metrics.GET_MEMORY_INFO, metrics.SUCCESS, metrics.NOT_APPLICABLE) + return mi, nil +} + func getRedisError(err error) string { if strings.Contains(err.Error(), "NOAUTH") { return metrics.NOAUTH diff --git a/service/redis/memory_test.go b/service/redis/memory_test.go new file mode 100644 index 000000000..3e2a5fa85 --- /dev/null +++ b/service/redis/memory_test.go @@ -0,0 +1,91 @@ +package redis + +import ( + "fmt" + "strconv" + "strings" + "testing" + "time" + + rediscli "github.com/go-redis/redis/v8" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" +) + +// fillRedis writes n 1KiB keys. +func fillRedis(t *testing.T, rc *rediscli.Client, n int) { + t.Helper() + value := strings.Repeat("x", 1024) + pipe := rc.Pipeline() + for i := 0; i < n; i++ { + pipe.Set(bgCtx(), fmt.Sprintf("key:%d", i), value, 0) + } + _, err := pipe.Exec(bgCtx()) + require.NoError(t, err) +} + +func TestGetMemoryInfo(t *testing.T) { + requireRedisServer(t) + r := startRedisProcess(t) + rc := rediscli.NewClient(&rediscli.Options{Addr: r.Addr()}) + defer func() { _ = rc.Close() }() + fillRedis(t, rc, 100) + c := newTestClient() + require.NoError(t, c.SetCustomRedisConfig(r.IP, strconv.Itoa(r.Port), []string{"maxmemory 104857600", "maxmemory-policy allkeys-lru"}, "")) + + mi, err := c.GetMemoryInfo(r.IP, strconv.Itoa(r.Port), "") + require.NoError(t, err) + assert.Equal(t, int64(104857600), mi.MaxMemory) + assert.Equal(t, "allkeys-lru", mi.MaxMemoryPolicy) + assert.Equal(t, "master", mi.Role) + assert.Greater(t, mi.UsedMemory, int64(200*1024)) +} + +func TestGetMemoryInfo_ConnectionError(t *testing.T) { + port, err := findFreePort() + require.NoError(t, err) + _, err = newTestClient().GetMemoryInfo(testLoopbackIP, strconv.Itoa(port), "") + assert.Error(t, err) +} + +// Lowering maxmemory below the memory in use makes Redis evict right away +// under allkeys-*; EnsureRedisMaxMemory relies on that. +func TestLoweringMaxMemoryEvicts(t *testing.T) { + requireRedisServer(t) + r := startRedisProcess(t) + rc := rediscli.NewClient(&rediscli.Options{Addr: r.Addr()}) + defer func() { _ = rc.Close() }() + fillRedis(t, rc, 4000) + c := newTestClient() + port := strconv.Itoa(r.Port) + before, err := c.GetMemoryInfo(r.IP, port, "") + require.NoError(t, err) + target := before.UsedMemory - 1<<20 + require.NoError(t, c.SetCustomRedisConfig(r.IP, port, []string{"maxmemory-policy allkeys-lru", fmt.Sprintf("maxmemory %d", target)}, "")) + + assert.Eventually(t, func() bool { + after, err := c.GetMemoryInfo(r.IP, port, "") + return err == nil && after.UsedMemory <= target + }, time.Second, 50*time.Millisecond) +} + +func TestGetMemoryInfoWhileLoading(t *testing.T) { + requireRedisServer(t) + r := startRedisProcess(t, "--enable-debug-command", "yes", "--key-load-delay", "500", "--loading-process-events-interval-bytes", "1024") + rc := rediscli.NewClient(&rediscli.Options{Addr: r.Addr()}) + defer func() { _ = rc.Close() }() + fillRedis(t, rc, 2000) + c := newTestClient() + port := strconv.Itoa(r.Port) + mi, err := c.GetMemoryInfo(r.IP, port, "") + require.NoError(t, err) + assert.False(t, mi.Loading) + + reloaded := make(chan error, 1) + go func() { reloaded <- rc.Do(bgCtx(), "debug", "reload").Err() }() + assert.Eventually(t, func() bool { + mi, err := c.GetMemoryInfo(r.IP, port, "") + return err == nil && mi.Loading + }, 5*time.Second, 10*time.Millisecond) + require.NoError(t, <-reloaded) +} From a0112131efc41e67f4f27a7d2d9f36d1a5fa0476 Mon Sep 17 00:00:00 2001 From: Michael Strassberger Date: Sat, 26 Sep 2026 23:31:04 +0200 Subject: [PATCH 20/24] feat(redis): resize redis pods in place (#191) When a stale redis pod's revision differs from the update revision only in container cpu and memory, resize it through the pods/resize subresource instead of recreating it, then move its revision label to the update revision once the kubelet reports the resources as applied (safe with OnDelete). No restart, no data reload, no master failover; still one pod at a time, replicas first. On by default from Kubernetes 1.33; spec.redis.inPlaceResize: Disabled opts out. - The resize starts from the pod's own resources and sets the update revision's cpu and memory values, so admission defaults (e.g. a LimitRange) are kept and a resize superseded by a newer revision is corrected. Memory decreases are detected against the applied limits. - The pod is recreated as before when the update changes anything else (resource claims included), adds or removes requests or limits, lowers a memory limit before Kubernetes 1.35, or when the API server rejects the resize, e.g. for a QoS class change or without the new RBAC. It is also recreated when the kubelet does not report the container's resources, as one without in-place resize support would leave the pod spec, which maxmemory derives from, ahead of the container. - Kubelet state: infeasible, deferred for more than 5 minutes, failing for more than 5 minutes (e.g. a memory decrease below the usage it sees, which includes the page cache), or not applied within 5 minutes without any condition falls back to recreating the pod. Timeouts count from the latest resize request, kept in a pod annotation. Conditions from an earlier pod generation are ignored. An error detecting the cluster's support is retried rather than recreating the pod. The operator needs patch on pods/resize and get on controllerrevisions, granted in the chart, the kustomize and the example manifests. Claude-Session: https://claude.ai/code/session_01G4mmyo3sqLwf23m5FE2nBE Co-authored-by: Claude --- README.md | 8 +- api/redisfailover/v1/types.go | 5 + ...atabases.spotahome.com_redisfailovers.yaml | 9 + .../templates/service-account.yaml | 12 + .../all-redis-operator-resources.yaml | 2 + example/operator/roles.yaml | 2 + ...atabases.spotahome.com_redisfailovers.yaml | 9 + ...atabases.spotahome.com_redisfailovers.yaml | 9 + .../components/rbac/clusterrole.yaml | 2 + .../service/RedisFailoverHeal.go | 24 + mocks/service/k8s/Services.go | 66 +++ operator/redisfailover/checker.go | 27 +- operator/redisfailover/checker_test.go | 61 ++- operator/redisfailover/service/check.go | 6 + operator/redisfailover/service/check_test.go | 15 + operator/redisfailover/service/constants.go | 4 + operator/redisfailover/service/heal.go | 1 + operator/redisfailover/service/inplace.go | 329 ++++++++++++ .../redisfailover/service/inplace_test.go | 478 ++++++++++++++++++ service/k8s/pod.go | 64 +++ service/k8s/resize_test.go | 103 ++++ service/k8s/statefulset.go | 8 + 22 files changed, 1238 insertions(+), 6 deletions(-) create mode 100644 operator/redisfailover/service/inplace.go create mode 100644 operator/redisfailover/service/inplace_test.go create mode 100644 service/k8s/resize_test.go diff --git a/README.md b/README.md index 80d153bf6..7e38dc001 100644 --- a/README.md +++ b/README.md @@ -202,12 +202,18 @@ With `redis.maxMemory` the operator sets `maxmemory` and `maxmemory-policy` from | 128Mi | 96Mi | | 1Gi | 768Mi | -Keys set in `customConfig` take precedence; `replica-ignore-maxmemory no` is rejected, as replicas would evict on their own, and running pods are set to `yes`. To migrate, update the CRD and the operator, add `maxMemory`, then remove `maxmemory` and `maxmemory-policy` from `customConfig`. Removing `maxMemory` leaves the running pods at their current values until they are replaced; set them in `customConfig` to keep them. +Keys set in `customConfig` take precedence; `replica-ignore-maxmemory no` is rejected, as replicas would evict on their own, and running pods are set to `yes`. To migrate, update the CRD and the operator, add `maxMemory`, then remove `maxmemory` and `maxmemory-policy` from `customConfig`. Removing `maxMemory` leaves the running pods at their current values until they are recreated, which an in-place resize does not do; set them in `customConfig` to keep them. `maxmemory` follows the smallest redis pod, as replicas hold the whole dataset and any of them can be promoted: a raised limit applies once every pod runs with it, a lowered one before the pods are replaced. `maxmemory` is only lowered below the memory in use under an `allkeys-*` policy, as `volatile-*` could evict every key with a TTL and still not fit; otherwise it is kept and the reason is in the status message. Until the data fits, the operator does not replace pods with the smaller limit. Pods recreated for other reasons, e.g. a node drain, get the smaller limit anyway. For small instances, the default `client-output-buffer-limit` for `pubsub` (32mb) and `replica` (256mb) can exceed the free part of the limit; lower them with `customConfig`. Replicas buffer a whole `MULTI`/`EXEC` or `EVAL` before applying it, so one large batch can get a replica OOM-killed. +### In-place resize + +On Kubernetes 1.33 or later, an update that only changes container cpu or memory resizes the redis pods in place instead of recreating them, so no data is reloaded and the master does not fail over. Pods are resized one at a time, replicas first. Lowering a memory limit in place needs Kubernetes 1.35. Set `redis.inPlaceResize: Disabled` to always recreate the pods. + +A pod is still recreated when the update changes anything else, adds or removes requests or limits, or changes the pod's QoS class, when the node's kubelet does not support in-place resize, and when the kubelet reports the resize as infeasible, defers it or fails it for more than 5 minutes, or does not apply it within 5 minutes without reporting why. The operator needs `patch` on `pods/resize` and `get` on `controllerrevisions`, which the chart, the kustomize and the example manifests grant; without them the pods are recreated. + ### Custom shutdown script By default, a custom shutdown file is given. This file makes redis to `SAVE` it's data, and when Sentinel is enabled and redis is master, it'll call sentinel to ask for failover. diff --git a/api/redisfailover/v1/types.go b/api/redisfailover/v1/types.go index a7b9b9dc6..a25dfcee4 100644 --- a/api/redisfailover/v1/types.go +++ b/api/redisfailover/v1/types.go @@ -86,6 +86,11 @@ type RedisSettings struct { // MaxMemory lets the operator set maxmemory and maxmemory-policy from the // redis container's memory limit. Values set in customConfig take precedence. MaxMemory *MaxMemorySettings `json:"maxMemory,omitempty"` + // InPlaceResize controls whether redis pods whose update only changes + // container resources are resized in place instead of being recreated. + // Defaults to Enabled. + // +kubebuilder:validation:Enum=Enabled;Disabled + InPlaceResize string `json:"inPlaceResize,omitempty"` } // MaxMemorySettings configures the operator-managed maxmemory. diff --git a/charts/redisoperator/crds/databases.spotahome.com_redisfailovers.yaml b/charts/redisoperator/crds/databases.spotahome.com_redisfailovers.yaml index 64c2e04b2..ff3995af7 100644 --- a/charts/redisoperator/crds/databases.spotahome.com_redisfailovers.yaml +++ b/charts/redisoperator/crds/databases.spotahome.com_redisfailovers.yaml @@ -5799,6 +5799,15 @@ spec: type: object x-kubernetes-map-type: atomic type: array + inPlaceResize: + description: |- + InPlaceResize controls whether redis pods whose update only changes + container resources are resized in place instead of being recreated. + Defaults to Enabled. + enum: + - Enabled + - Disabled + type: string initContainers: items: description: A single application container that you want to diff --git a/charts/redisoperator/templates/service-account.yaml b/charts/redisoperator/templates/service-account.yaml index 41340aaaf..ef6bd8007 100644 --- a/charts/redisoperator/templates/service-account.yaml +++ b/charts/redisoperator/templates/service-account.yaml @@ -74,6 +74,18 @@ rules: - secrets verbs: - "get" + - apiGroups: + - "" + resources: + - pods/resize + verbs: + - patch + - apiGroups: + - apps + resources: + - controllerrevisions + verbs: + - get - apiGroups: - apps resources: diff --git a/example/operator/all-redis-operator-resources.yaml b/example/operator/all-redis-operator-resources.yaml index 92d2b193d..5904646a6 100644 --- a/example/operator/all-redis-operator-resources.yaml +++ b/example/operator/all-redis-operator-resources.yaml @@ -77,6 +77,7 @@ rules: - serviceaccounts - persistentvolumeclaims - persistentvolumeclaims/finalizers + - pods/resize verbs: - "*" - apiGroups: @@ -90,6 +91,7 @@ rules: resources: - deployments - statefulsets + - controllerrevisions verbs: - "*" - apiGroups: diff --git a/example/operator/roles.yaml b/example/operator/roles.yaml index 799cbc9aa..189f94a65 100644 --- a/example/operator/roles.yaml +++ b/example/operator/roles.yaml @@ -44,6 +44,7 @@ rules: - serviceaccounts - persistentvolumeclaims - persistentvolumeclaims/finalizers + - pods/resize verbs: - "*" - apiGroups: @@ -51,6 +52,7 @@ rules: resources: - deployments - statefulsets + - controllerrevisions verbs: - "*" - apiGroups: diff --git a/manifests/databases.spotahome.com_redisfailovers.yaml b/manifests/databases.spotahome.com_redisfailovers.yaml index 64c2e04b2..ff3995af7 100644 --- a/manifests/databases.spotahome.com_redisfailovers.yaml +++ b/manifests/databases.spotahome.com_redisfailovers.yaml @@ -5799,6 +5799,15 @@ spec: type: object x-kubernetes-map-type: atomic type: array + inPlaceResize: + description: |- + InPlaceResize controls whether redis pods whose update only changes + container resources are resized in place instead of being recreated. + Defaults to Enabled. + enum: + - Enabled + - Disabled + type: string initContainers: items: description: A single application container that you want to diff --git a/manifests/kustomize/base/databases.spotahome.com_redisfailovers.yaml b/manifests/kustomize/base/databases.spotahome.com_redisfailovers.yaml index 64c2e04b2..ff3995af7 100644 --- a/manifests/kustomize/base/databases.spotahome.com_redisfailovers.yaml +++ b/manifests/kustomize/base/databases.spotahome.com_redisfailovers.yaml @@ -5799,6 +5799,15 @@ spec: type: object x-kubernetes-map-type: atomic type: array + inPlaceResize: + description: |- + InPlaceResize controls whether redis pods whose update only changes + container resources are resized in place instead of being recreated. + Defaults to Enabled. + enum: + - Enabled + - Disabled + type: string initContainers: items: description: A single application container that you want to diff --git a/manifests/kustomize/components/rbac/clusterrole.yaml b/manifests/kustomize/components/rbac/clusterrole.yaml index 6dfc5f2a3..729b21bfb 100644 --- a/manifests/kustomize/components/rbac/clusterrole.yaml +++ b/manifests/kustomize/components/rbac/clusterrole.yaml @@ -44,6 +44,7 @@ rules: - serviceaccounts - persistentvolumeclaims - persistentvolumeclaims/finalizers + - pods/resize verbs: - "*" - apiGroups: @@ -51,6 +52,7 @@ rules: resources: - deployments - statefulsets + - controllerrevisions verbs: - "*" - apiGroups: diff --git a/mocks/operator/redisfailover/service/RedisFailoverHeal.go b/mocks/operator/redisfailover/service/RedisFailoverHeal.go index 7bf6acf34..fec48e058 100644 --- a/mocks/operator/redisfailover/service/RedisFailoverHeal.go +++ b/mocks/operator/redisfailover/service/RedisFailoverHeal.go @@ -94,6 +94,30 @@ func (_m *RedisFailoverHeal) NewSentinelMonitorWithPort(ip string, monitor strin return r0 } +// ResizePodInPlace provides a mock function with given fields: rFailover, podName, updateRevision +func (_m *RedisFailoverHeal) ResizePodInPlace(rFailover *v1.RedisFailover, podName string, updateRevision string) (service.ResizeResult, error) { + ret := _m.Called(rFailover, podName, updateRevision) + + var r0 service.ResizeResult + var r1 error + if rf, ok := ret.Get(0).(func(*v1.RedisFailover, string, string) (service.ResizeResult, error)); ok { + return rf(rFailover, podName, updateRevision) + } + if rf, ok := ret.Get(0).(func(*v1.RedisFailover, string, string) service.ResizeResult); ok { + r0 = rf(rFailover, podName, updateRevision) + } else { + r0 = ret.Get(0).(service.ResizeResult) + } + + if rf, ok := ret.Get(1).(func(*v1.RedisFailover, string, string) error); ok { + r1 = rf(rFailover, podName, updateRevision) + } else { + r1 = ret.Error(1) + } + + return r0, r1 +} + // RestoreSentinel provides a mock function with given fields: ip func (_m *RedisFailoverHeal) RestoreSentinel(ip string) error { ret := _m.Called(ip) diff --git a/mocks/service/k8s/Services.go b/mocks/service/k8s/Services.go index 59903d582..49755c0b7 100644 --- a/mocks/service/k8s/Services.go +++ b/mocks/service/k8s/Services.go @@ -13,6 +13,8 @@ import ( policyv1 "k8s.io/api/policy/v1" + k8s "github.com/saremox/redis-operator/service/k8s" + redisfailoverv1 "github.com/saremox/redis-operator/api/redisfailover/v1" v1 "k8s.io/api/core/v1" @@ -317,6 +319,32 @@ func (_m *Services) GetConfigMap(namespace string, name string) (*v1.ConfigMap, return r0, r1 } +// GetControllerRevision provides a mock function with given fields: namespace, name +func (_m *Services) GetControllerRevision(namespace string, name string) (*appsv1.ControllerRevision, error) { + ret := _m.Called(namespace, name) + + var r0 *appsv1.ControllerRevision + var r1 error + if rf, ok := ret.Get(0).(func(string, string) (*appsv1.ControllerRevision, error)); ok { + return rf(namespace, name) + } + if rf, ok := ret.Get(0).(func(string, string) *appsv1.ControllerRevision); ok { + r0 = rf(namespace, name) + } else { + if ret.Get(0) != nil { + r0 = ret.Get(0).(*appsv1.ControllerRevision) + } + } + + if rf, ok := ret.Get(1).(func(string, string) error); ok { + r1 = rf(namespace, name) + } else { + r1 = ret.Error(1) + } + + return r0, r1 +} + // GetDeployment provides a mock function with given fields: namespace, name func (_m *Services) GetDeployment(namespace string, name string) (*appsv1.Deployment, error) { ret := _m.Called(namespace, name) @@ -747,6 +775,44 @@ func (_m *Services) PatchRedisFailoverFinalizers(ctx context.Context, namespace return r0 } +// PodResizeSupport provides a mock function with given fields: +func (_m *Services) PodResizeSupport() (k8s.PodResizeSupport, error) { + ret := _m.Called() + + var r0 k8s.PodResizeSupport + var r1 error + if rf, ok := ret.Get(0).(func() (k8s.PodResizeSupport, error)); ok { + return rf() + } + if rf, ok := ret.Get(0).(func() k8s.PodResizeSupport); ok { + r0 = rf() + } else { + r0 = ret.Get(0).(k8s.PodResizeSupport) + } + + if rf, ok := ret.Get(1).(func() error); ok { + r1 = rf() + } else { + r1 = ret.Error(1) + } + + return r0, r1 +} + +// ResizePod provides a mock function with given fields: namespace, podName, resources +func (_m *Services) ResizePod(namespace string, podName string, resources map[string]v1.ResourceRequirements) error { + ret := _m.Called(namespace, podName, resources) + + var r0 error + if rf, ok := ret.Get(0).(func(string, string, map[string]v1.ResourceRequirements) error); ok { + r0 = rf(namespace, podName, resources) + } else { + r0 = ret.Error(0) + } + + return r0 +} + // UpdateConfigMap provides a mock function with given fields: namespace, configMap func (_m *Services) UpdateConfigMap(namespace string, configMap *v1.ConfigMap) error { ret := _m.Called(namespace, configMap) diff --git a/operator/redisfailover/checker.go b/operator/redisfailover/checker.go index e982c90ae..60bac831c 100644 --- a/operator/redisfailover/checker.go +++ b/operator/redisfailover/checker.go @@ -65,6 +65,9 @@ func (r *RedisFailoverHandler) UpdateRedisesPods(rf *redisfailoverv1.RedisFailov if settled, err := r.redisPodsSettled(rf, ssUR); err != nil || !settled { return err } + if recreate, err := r.resizeInPlace(rf, pod, ssUR); err != nil || !recreate { + return err + } //Delete pod and wait next round to check if the new one is synced err = r.rfHealer.DeletePod(pod, rf) if err != nil { @@ -87,6 +90,14 @@ func (r *RedisFailoverHandler) UpdateRedisesPods(rf *redisfailoverv1.RedisFailov return err } if masterRevision != ssUR { + // Resizing in place needs no failover, so it skips the gate below. + if settled, err := r.redisPodsSettled(rf, ssUR); err != nil || !settled { + return err + } + if recreate, err := r.resizeInPlace(rf, master, ssUR); err != nil || !recreate { + return err + } + // Deleting the master makes sentinel run a failover. Only do that once // every sentinel has a quorum (majority) of the freshly (re)started // slaves in memory - the redis-side readiness checked above is not @@ -119,9 +130,6 @@ func (r *RedisFailoverHandler) UpdateRedisesPods(rf *redisfailoverv1.RedisFailov } } - if settled, err := r.redisPodsSettled(rf, ssUR); err != nil || !settled { - return err - } err = r.rfHealer.DeletePod(master, rf) if err != nil { return err @@ -134,6 +142,19 @@ func (r *RedisFailoverHandler) UpdateRedisesPods(rf *redisfailoverv1.RedisFailov return nil } +// resizeInPlace tries to move a stale pod to the update revision without +// recreating it. It reports whether the pod has to be recreated instead. +func (r *RedisFailoverHandler) resizeInPlace(rf *redisfailoverv1.RedisFailover, pod, updateRevision string) (bool, error) { + result, err := r.rfHealer.ResizePodInPlace(rf, pod, updateRevision) + if err != nil { + return false, err + } + if result.Action == rfservice.ResizeWaiting && result.Message != "" { + rf.Status.Message = result.Message + } + return result.Action == rfservice.ResizeRecreate, nil +} + // redisPodsSettled reports whether the last redis pod replacement has // finished: the StatefulSet has all its pods, none is being deleted, and every // pod already on the update revision is ready. Pod events start the next diff --git a/operator/redisfailover/checker_test.go b/operator/redisfailover/checker_test.go index 10c3cacb5..e8e5a317f 100644 --- a/operator/redisfailover/checker_test.go +++ b/operator/redisfailover/checker_test.go @@ -2007,6 +2007,7 @@ func TestUpdate(t *testing.T) { // master is deleted later and only after the sentinel gate, so // its DeletePod expectation is set in the master block below. if pod.pod.Labels[appsv1.ControllerRevisionHashLabelKey] != test.ssVersion && !pod.master { + mrfh.On("ResizePodInPlace", rf, pod.pod.Name, mock.Anything).Once().Return(rfservice.ResizeResult{}, nil) mrfh.On("DeletePod", pod.pod.Name, rf).Once().Return(nil) next = false break @@ -2026,8 +2027,10 @@ func TestUpdate(t *testing.T) { } } if masterStale { - // Before replacing the master the operator checks every - // sentinel has a quorum of the slaves in memory. + // An in-place resize is tried first; before recreating the + // master the operator checks every sentinel has a quorum of + // the slaves in memory. + mrfh.On("ResizePodInPlace", rf, "master", mock.Anything).Once().Return(rfservice.ResizeResult{}, nil) mrfc.On("GetSentinelsIPs", rf).Once().Return([]string{"sentinel0"}, nil) if test.sentinelSlavesShort { mrfc.On("CheckSentinelSlavesNumberQuorumInMemory", "sentinel0", rf).Once().Return(errors.New("redis slaves in sentinel memory below quorum")) @@ -2084,6 +2087,7 @@ func TestUpdateRedisesPodsOperatorManagedModeSkipsSentinelGate(t *testing.T) { mrfc.On("GetRedisesSlavesPods", rf).Once().Return([]string{}, nil) mrfc.On("GetRedisesMasterPod", rf).Once().Return("master", nil) mrfc.On("GetRedisRevisionHash", "master", rf).Once().Return("9", nil) // stale + mrfh.On("ResizePodInPlace", rf, "master", mock.Anything).Once().Return(rfservice.ResizeResult{}, nil) mrfh.On("DeletePod", "master", rf).Once().Return(nil) mk := settledK8sServices() @@ -2153,6 +2157,7 @@ func TestUpdateRedisesPodsErrorBranches(t *testing.T) { mrfc.On("GetStatefulSetUpdateRevision", rf).Once().Return("1", nil) mrfc.On("GetRedisesSlavesPods", rf).Once().Return([]string{"slave1"}, nil) mrfc.On("GetRedisRevisionHash", "slave1", rf).Once().Return("stale", nil) + mrfh.On("ResizePodInPlace", rf, "slave1", mock.Anything).Once().Return(rfservice.ResizeResult{}, nil) mrfh.On("DeletePod", "slave1", rf).Once().Return(errors.New("delete err")) }, }, @@ -2176,6 +2181,7 @@ func TestUpdateRedisesPodsErrorBranches(t *testing.T) { mrfc.On("GetRedisesSlavesPods", rf).Once().Return([]string{}, nil) mrfc.On("GetRedisesMasterPod", rf).Once().Return(master, nil) mrfc.On("GetRedisRevisionHash", master, rf).Once().Return("stale", nil) + mrfh.On("ResizePodInPlace", rf, master, mock.Anything).Once().Return(rfservice.ResizeResult{}, nil) mrfc.On("GetSentinelsIPs", rf).Once().Return(nil, errors.New("sentinels ips err")) }, }, @@ -2190,6 +2196,7 @@ func TestUpdateRedisesPodsErrorBranches(t *testing.T) { mrfc.On("GetRedisRevisionHash", master, rf).Once().Return("stale", nil) mrfc.On("GetSentinelsIPs", rf).Once().Return([]string{"sentinel0"}, nil) mrfc.On("CheckSentinelSlavesNumberQuorumInMemory", "sentinel0", rf).Once().Return(nil) + mrfh.On("ResizePodInPlace", rf, master, mock.Anything).Once().Return(rfservice.ResizeResult{}, nil) mrfh.On("DeletePod", master, rf).Once().Return(errors.New("delete master err")) }, }, @@ -2299,6 +2306,7 @@ func TestUpdateRedisesPodsWaitsForTheLastReplacement(t *testing.T) { mrfc.On("GetRedisRevisionHash", stale, rf).Once().Return("old", nil) mk.On("GetStatefulSetPods", rf.Namespace, rfservice.GetRedisName(rf)).Once().Return(&corev1.PodList{Items: test.pods}, test.podsErr) if test.wantDelete { + mrfh.On("ResizePodInPlace", rf, stale, mock.Anything).Once().Return(rfservice.ResizeResult{}, nil) mrfh.On("DeletePod", stale, rf).Once().Return(nil) } @@ -2371,3 +2379,52 @@ func TestOperatorManagedModeWaitsForAStoppingMasterBeforeElecting(t *testing.T) }) } } + +// A pod resized in place is neither deleted nor, as master, failed over. +func TestUpdateRedisesPodsResizesInPlace(t *testing.T) { + for _, stale := range []string{"slave", "master"} { + for _, action := range []rfservice.ResizeAction{rfservice.ResizeWaiting, rfservice.ResizeDone} { + t.Run(fmt.Sprintf("%s/%d", stale, action), func(t *testing.T) { + rf := generateRF(false, false) + mrfc := &mRFService.RedisFailoverCheck{} + mrfh := &mRFService.RedisFailoverHeal{} + mrfc.On("GetRedisesIPs", rf).Once().Return([]string{"10.0.0.1"}, nil) + mrfc.On("GetMasterIP", rf).Once().Return("10.0.0.1", nil) + mrfc.On("GetStatefulSetUpdateRevision", rf).Once().Return("new", nil) + if stale == "slave" { + mrfc.On("GetRedisesSlavesPods", rf).Once().Return([]string{"slave"}, nil) + } else { + mrfc.On("GetRedisesSlavesPods", rf).Once().Return([]string{}, nil) + mrfc.On("GetRedisesMasterPod", rf).Once().Return("master", nil) + } + mrfc.On("GetRedisRevisionHash", stale, rf).Once().Return("old", nil) + // No DeletePod or sentinel expectations: calling them would panic the mock. + mrfh.On("ResizePodInPlace", rf, stale, "new").Once().Return(rfservice.ResizeResult{Action: action, Message: "resizing"}, nil) + + handler := rfOperator.NewRedisFailoverHandler(generateConfig(), &mRFService.RedisFailoverClient{}, mrfc, mrfh, settledK8sServices(), metrics.Dummy, log.Dummy) + assert.NoError(t, handler.UpdateRedisesPods(rf)) + if action == rfservice.ResizeWaiting { + assert.Equal(t, "resizing", rf.Status.Message) + } + mrfc.AssertExpectations(t) + mrfh.AssertExpectations(t) + }) + } + } +} + +func TestUpdateRedisesPodsResizeError(t *testing.T) { + rf := generateRF(false, false) + mrfc := &mRFService.RedisFailoverCheck{} + mrfh := &mRFService.RedisFailoverHeal{} + mrfc.On("GetRedisesIPs", rf).Once().Return([]string{"10.0.0.1"}, nil) + mrfc.On("GetMasterIP", rf).Once().Return("10.0.0.1", nil) + mrfc.On("GetStatefulSetUpdateRevision", rf).Once().Return("new", nil) + mrfc.On("GetRedisesSlavesPods", rf).Once().Return([]string{"slave"}, nil) + mrfc.On("GetRedisRevisionHash", "slave", rf).Once().Return("old", nil) + mrfh.On("ResizePodInPlace", rf, "slave", "new").Once().Return(rfservice.ResizeResult{}, errors.New("resize err")) + + handler := rfOperator.NewRedisFailoverHandler(generateConfig(), &mRFService.RedisFailoverClient{}, mrfc, mrfh, settledK8sServices(), metrics.Dummy, log.Dummy) + assert.EqualError(t, handler.UpdateRedisesPods(rf), "resize err") + mrfh.AssertExpectations(t) +} diff --git a/operator/redisfailover/service/check.go b/operator/redisfailover/service/check.go index 96a0785c9..d977ddc87 100644 --- a/operator/redisfailover/service/check.go +++ b/operator/redisfailover/service/check.go @@ -548,6 +548,12 @@ func (r *RedisFailoverChecker) GetRedisRevisionHash(podName string, rFailover *r return "", errors.New("labels not found") } + // A pod being resized in place is on no revision until the resize is + // applied, even when its label matches again, e.g. after a revert. + if pod.Annotations[resizeRequestedAnnotation] != "" { + return "", nil + } + val := pod.Labels[appsv1.ControllerRevisionHashLabelKey] return val, nil diff --git a/operator/redisfailover/service/check_test.go b/operator/redisfailover/service/check_test.go index 0753fb6e0..8ed2d51a5 100644 --- a/operator/redisfailover/service/check_test.go +++ b/operator/redisfailover/service/check_test.go @@ -1244,6 +1244,21 @@ func TestGetRedisRevisionHash(t *testing.T) { expectedHash: "10", expectedError: nil, }, + { + name: "being resized in place", + pod: &corev1.Pod{ + ObjectMeta: metav1.ObjectMeta{ + Labels: map[string]string{ + appsv1.ControllerRevisionHashLabelKey: "10", + }, + Annotations: map[string]string{ + "redisfailovers.databases.spotahome.com/resize-requested-at": "2026-01-01T00:00:00Z", + }, + }, + }, + expectedHash: "", + expectedError: nil, + }, { name: "no pod", pod: nil, diff --git a/operator/redisfailover/service/constants.go b/operator/redisfailover/service/constants.go index 4eaa2641c..473b2ed5b 100644 --- a/operator/redisfailover/service/constants.go +++ b/operator/redisfailover/service/constants.go @@ -48,6 +48,10 @@ const ( // UpdateRedisesPods already uses to roll pods one at a time. const redisAuthSecretChecksumAnnotation = "redisfailovers.databases.spotahome.com/secret-checksum" +// resizeRequestedAnnotation holds when the operator last requested an +// in-place resize of the pod, and is cleared once the resize is applied. +const resizeRequestedAnnotation = "redisfailovers.databases.spotahome.com/resize-requested-at" + // masterSafeToEvictAnnotation is the cluster-autoscaler annotation used to keep // the node running the redis master from being drained during scale-down. const masterSafeToEvictAnnotation = "cluster-autoscaler.kubernetes.io/safe-to-evict" diff --git a/operator/redisfailover/service/heal.go b/operator/redisfailover/service/heal.go index ba42a426e..ad4a51239 100644 --- a/operator/redisfailover/service/heal.go +++ b/operator/redisfailover/service/heal.go @@ -33,6 +33,7 @@ type RedisFailoverHeal interface { DeletePod(podName string, rFailover *redisfailoverv1.RedisFailover) error PromoteBestReplica(newMasterIP string, rFailover *redisfailoverv1.RedisFailover) error EnsureRedisMaxMemory(rFailover *redisfailoverv1.RedisFailover, master string, redises []string) (MaxMemoryResult, error) + ResizePodInPlace(rFailover *redisfailoverv1.RedisFailover, podName, updateRevision string) (ResizeResult, error) } // RedisFailoverHealer is our implementation of RedisFailoverCheck interface diff --git a/operator/redisfailover/service/inplace.go b/operator/redisfailover/service/inplace.go new file mode 100644 index 000000000..5da7eab44 --- /dev/null +++ b/operator/redisfailover/service/inplace.go @@ -0,0 +1,329 @@ +package service + +import ( + "encoding/json" + "fmt" + "time" + + appsv1 "k8s.io/api/apps/v1" + corev1 "k8s.io/api/core/v1" + "k8s.io/apimachinery/pkg/api/equality" + apierrors "k8s.io/apimachinery/pkg/api/errors" + + redisfailoverv1 "github.com/saremox/redis-operator/api/redisfailover/v1" + "github.com/saremox/redis-operator/service/k8s" +) + +// ResizeAction is what the rollout does with a pod that is not on the update revision. +type ResizeAction int + +const ( + // ResizeRecreate means the pod has to be deleted and recreated. + ResizeRecreate ResizeAction = iota + // ResizeWaiting means an in-place resize of the pod is in progress. + ResizeWaiting + // ResizeDone means the pod was resized in place and moved to the update revision. + ResizeDone +) + +// ResizeResult is the outcome of ResizePodInPlace. +type ResizeResult struct { + Action ResizeAction + // Message explains a resize that is stuck or fell back to recreating the pod. + Message string +} + +// inPlaceResizeTimeout bounds how long a deferred or failing resize is waited +// for before the pod is recreated instead. +var inPlaceResizeTimeout = 5 * time.Minute + +var resizableResources = []corev1.ResourceName{corev1.ResourceCPU, corev1.ResourceMemory} + +// ResizePodInPlace moves a redis pod to the update revision without +// recreating it when the revisions differ only in container cpu and memory. +// The kubelet applies the resize without restarting the containers; once the +// applied resources match, the pod's revision label is updated, which +// UpdateRedisesPods then treats as up to date. The StatefulSet uses OnDelete, +// so its controller never replaces the relabelled pod. +func (r *RedisFailoverHealer) ResizePodInPlace(rf *redisfailoverv1.RedisFailover, podName, updateRevision string) (ResizeResult, error) { + logger := r.logger.WithField("redisfailover", rf.Name).WithField("namespace", rf.Namespace).WithField("pod", podName) + recreate := func(reason string) (ResizeResult, error) { + logger.Infof("Recreating the pod instead of resizing it in place: %s", reason) + return ResizeResult{Action: ResizeRecreate, Message: reason}, nil + } + waiting := func(format string, args ...interface{}) (ResizeResult, error) { + return ResizeResult{Action: ResizeWaiting, Message: fmt.Sprintf(format, args...)}, nil + } + + if rf.Spec.Redis.InPlaceResize == "Disabled" { + return ResizeResult{Action: ResizeRecreate}, nil + } + support, err := r.k8sService.PodResizeSupport() + if err != nil { + return ResizeResult{}, err + } + if !support.Supported { + return ResizeResult{Action: ResizeRecreate}, nil + } + pod, err := r.k8sService.GetPod(rf.Namespace, podName) + if err != nil { + return ResizeResult{}, err + } + desired, reason, err := r.inPlaceResources(rf, pod, updateRevision, support) + if err != nil { + return ResizeResult{}, err + } + if reason != "" { + return recreate(reason) + } + + if !podRequests(pod, desired) { + if err := r.markResizeRequested(rf, podName); err != nil { + return ResizeResult{}, err + } + if err := r.k8sService.ResizePod(rf.Namespace, podName, desired); err != nil { + // Rejected by the API server, e.g. missing RBAC or a disabled feature gate. + if apierrors.IsForbidden(err) || apierrors.IsInvalid(err) || apierrors.IsNotFound(err) || apierrors.IsMethodNotSupported(err) { + return recreate("resize rejected: " + err.Error()) + } + return ResizeResult{}, err + } + logger.Infof("Resizing the pod in place") + return waiting("resizing pod %s in place", podName) + } + + // A condition keeps its transition time when a newer resize supersedes + // the one it reported on. + sinceLatest := func(t time.Time) time.Duration { + if requested, err := time.Parse(time.RFC3339, pod.Annotations[resizeRequestedAnnotation]); err == nil && requested.After(t) { + t = requested + } + return time.Since(t) + } + if c := podCondition(pod, corev1.PodResizePending); c != nil { + if c.Reason == corev1.PodReasonInfeasible { + return recreate("resize infeasible: " + c.Message) + } + if sinceLatest(c.LastTransitionTime.Time) > inPlaceResizeTimeout { + return recreate("resize deferred for too long: " + c.Message) + } + return waiting("resize of pod %s deferred: %s", podName, c.Message) + } + if c := podCondition(pod, corev1.PodResizeInProgress); c != nil { + // E.g. a memory limit below the usage the kubelet sees, which counts + // the page cache. + if c.Reason == corev1.PodReasonError && sinceLatest(c.LastTransitionTime.Time) > inPlaceResizeTimeout { + return recreate("resize failed: " + c.Message) + } + return waiting("resize of pod %s in progress", podName) + } + if !podApplied(pod, desired) { + // The kubelet may not have picked the resize up yet. + requested, err := time.Parse(time.RFC3339, pod.Annotations[resizeRequestedAnnotation]) + if err != nil { + if err := r.markResizeRequested(rf, podName); err != nil { + return ResizeResult{}, err + } + } else if time.Since(requested) > inPlaceResizeTimeout { + return recreate("resize not applied by the kubelet") + } + return waiting("resize of pod %s in progress", podName) + } + + if err := r.k8sService.UpdatePodLabels(rf.Namespace, podName, map[string]string{appsv1.ControllerRevisionHashLabelKey: updateRevision}); err != nil { + return ResizeResult{}, err + } + if err := r.k8sService.UpdatePodAnnotations(rf.Namespace, podName, map[string]string{resizeRequestedAnnotation: ""}); err != nil { + return ResizeResult{}, err + } + logger.Infof("Resized the pod in place") + return ResizeResult{Action: ResizeDone}, nil +} + +func (r *RedisFailoverHealer) markResizeRequested(rf *redisfailoverv1.RedisFailover, podName string) error { + return r.k8sService.UpdatePodAnnotations(rf.Namespace, podName, map[string]string{resizeRequestedAnnotation: time.Now().UTC().Format(time.RFC3339)}) +} + +// inPlaceResources returns the container resources of the update revision +// when the pod can be moved to it by an in-place resize, or the reason it +// cannot. +func (r *RedisFailoverHealer) inPlaceResources(rf *redisfailoverv1.RedisFailover, pod *corev1.Pod, updateRevision string, support k8s.PodResizeSupport) (map[string]corev1.ResourceRequirements, string, error) { + current, err := r.revisionTemplate(rf, pod.Labels[appsv1.ControllerRevisionHashLabelKey]) + if current == nil || err != nil { + return nil, "the pod's revision is not available", err + } + update, err := r.revisionTemplate(rf, updateRevision) + if update == nil || err != nil { + return nil, "the update revision is not available", err + } + if !equality.Semantic.DeepEqual(withoutResources(current), withoutResources(update)) { + return nil, "the update changes more than container resources", nil + } + + old := map[string]corev1.ResourceRequirements{} + for _, c := range current.Spec.Containers { + old[c.Name] = withDefaultRequests(c.Resources) + } + // The pod's own resources are the base: admission (e.g. a LimitRange) may + // have added values that neither revision has. + actual, running := map[string]corev1.ResourceRequirements{}, map[string]corev1.ResourceRequirements{} + for _, c := range pod.Spec.Containers { + actual[c.Name], running[c.Name] = c.Resources, c.Resources + } + reported := false + for _, cs := range pod.Status.ContainerStatuses { + if cs.Resources != nil { + running[cs.Name] = *cs.Resources + reported = reported || cs.Name == redisContainerName + } + } + if !reported { + // E.g. a kubelet without in-place resize support, which would leave + // the pod spec that maxmemory derives from ahead of the container. + return nil, "the kubelet does not report the container's resources", nil + } + desired := map[string]corev1.ResourceRequirements{} + lowersMemoryLimit := false + for _, c := range update.Spec.Containers { + d, o := withDefaultRequests(c.Resources), old[c.Name] + for _, list := range [][2]corev1.ResourceList{{o.Requests, d.Requests}, {o.Limits, d.Limits}} { + if len(list[0]) != len(list[1]) { + return nil, "requests or limits are added or removed", nil + } + for name, q := range list[0] { + n, ok := list[1][name] + if !ok { + return nil, "requests or limits are added or removed", nil + } + if name != corev1.ResourceCPU && name != corev1.ResourceMemory && !q.Equal(n) { + return nil, "only cpu and memory can be resized in place", nil + } + } + } + // The pod may already run a superseded revision's values. + base := actual[c.Name] + target := *base.DeepCopy() + target.Requests = overlay(target.Requests, d.Requests) + target.Limits = overlay(target.Limits, d.Limits) + desired[c.Name] = target + if q, ok := running[c.Name].Limits[corev1.ResourceMemory]; ok && target.Limits.Memory().Cmp(q) < 0 { + lowersMemoryLimit = true + } + } + if lowersMemoryLimit && !support.MemoryLimitDecrease { + return nil, "lowering a memory limit in place needs Kubernetes 1.35", nil + } + return desired, "", nil +} + +// overlay sets the cpu and memory values of new on list. +func overlay(list, new corev1.ResourceList) corev1.ResourceList { + for _, name := range resizableResources { + if n, ok := new[name]; ok { + if list == nil { + list = corev1.ResourceList{} + } + list[name] = n + } + } + return list +} + +func (r *RedisFailoverHealer) revisionTemplate(rf *redisfailoverv1.RedisFailover, revision string) (*corev1.PodTemplateSpec, error) { + // The revision hash label and the update revision are the ControllerRevision's name. + cr, err := r.k8sService.GetControllerRevision(rf.Namespace, revision) + // Forbidden without the RBAC added for in-place resize: recreate instead. + if apierrors.IsNotFound(err) || apierrors.IsForbidden(err) { + return nil, nil + } + if err != nil { + return nil, err + } + // The StatefulSet controller stores the template as a patch of the + // StatefulSet: {"spec":{"template":{...}}}. + var data struct { + Spec struct { + Template corev1.PodTemplateSpec `json:"template"` + } `json:"spec"` + } + if err := json.Unmarshal(cr.Data.Raw, &data); err != nil { + return nil, err + } + return &data.Spec.Template, nil +} + +func withoutResources(t *corev1.PodTemplateSpec) *corev1.PodTemplateSpec { + t = t.DeepCopy() + for i := range t.Spec.Containers { + t.Spec.Containers[i].Resources.Requests = nil + t.Spec.Containers[i].Resources.Limits = nil + } + return t +} + +// withDefaultRequests sets unset requests to their limits, as the API server +// does for pods but not for the templates they are created from. +func withDefaultRequests(r corev1.ResourceRequirements) corev1.ResourceRequirements { + r = *r.DeepCopy() + for name, q := range r.Limits { + if _, ok := r.Requests[name]; !ok { + if r.Requests == nil { + r.Requests = corev1.ResourceList{} + } + r.Requests[name] = q + } + } + return r +} + +// podRequests reports whether the pod spec already requests the desired resources. +func podRequests(pod *corev1.Pod, desired map[string]corev1.ResourceRequirements) bool { + for _, c := range pod.Spec.Containers { + if d, ok := desired[c.Name]; ok && !sameResizable(c.Resources, d) { + return false + } + } + return true +} + +// podApplied reports whether the kubelet reports the desired resources as applied. +func podApplied(pod *corev1.Pod, desired map[string]corev1.ResourceRequirements) bool { + applied := map[string]*corev1.ResourceRequirements{} + for _, cs := range pod.Status.ContainerStatuses { + applied[cs.Name] = cs.Resources + } + for name, d := range desired { + if a := applied[name]; a == nil || !sameResizable(*a, d) { + return false + } + } + return true +} + +func sameResizable(a, b corev1.ResourceRequirements) bool { + for _, name := range resizableResources { + for _, list := range [][2]corev1.ResourceList{{a.Requests, b.Requests}, {a.Limits, b.Limits}} { + qa, oka := list[0][name] + qb, okb := list[1][name] + if oka != okb || !qa.Equal(qb) { + return false + } + } + } + return true +} + +// podCondition returns the pod's true condition of type t, ignoring one left +// from an earlier resize request. +func podCondition(pod *corev1.Pod, t corev1.PodConditionType) *corev1.PodCondition { + for i := range pod.Status.Conditions { + c := &pod.Status.Conditions[i] + if c.ObservedGeneration != 0 && c.ObservedGeneration < pod.Generation { + continue + } + if c.Type == t && c.Status == corev1.ConditionTrue { + return &pod.Status.Conditions[i] + } + } + return nil +} diff --git a/operator/redisfailover/service/inplace_test.go b/operator/redisfailover/service/inplace_test.go new file mode 100644 index 000000000..fddd962f1 --- /dev/null +++ b/operator/redisfailover/service/inplace_test.go @@ -0,0 +1,478 @@ +package service + +import ( + "encoding/json" + "errors" + "testing" + "time" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + appsv1 "k8s.io/api/apps/v1" + corev1 "k8s.io/api/core/v1" + apierrors "k8s.io/apimachinery/pkg/api/errors" + "k8s.io/apimachinery/pkg/api/resource" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/runtime" + "k8s.io/apimachinery/pkg/runtime/schema" + + redisfailoverv1 "github.com/saremox/redis-operator/api/redisfailover/v1" + "github.com/saremox/redis-operator/log" + mK8SService "github.com/saremox/redis-operator/mocks/service/k8s" + "github.com/saremox/redis-operator/service/k8s" +) + +const resizePod = "rfr-test-0" + +func resources(cpu, memory string) corev1.ResourceRequirements { + return corev1.ResourceRequirements{Limits: corev1.ResourceList{ + corev1.ResourceCPU: resource.MustParse(cpu), + corev1.ResourceMemory: resource.MustParse(memory), + }} +} + +func podTemplate(redis corev1.ResourceRequirements) corev1.PodTemplateSpec { + return corev1.PodTemplateSpec{Spec: corev1.PodSpec{Containers: []corev1.Container{ + {Name: redisContainerName, Image: "redis:7", Resources: redis}, + {Name: "exporter", Image: "exporter"}, + }}} +} + +func revision(name string, t corev1.PodTemplateSpec) *appsv1.ControllerRevision { + raw, _ := json.Marshal(map[string]interface{}{"spec": map[string]interface{}{"template": t}}) + return &appsv1.ControllerRevision{ObjectMeta: metav1.ObjectMeta{Name: name}, Data: runtime.RawExtension{Raw: raw}} +} + +// stalePod is a pod created from the "old" revision, with spec resources +// requested and status resources applied as given. +func stalePod(requested, applied corev1.ResourceRequirements, conditions ...corev1.PodCondition) *corev1.Pod { + pod := &corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: resizePod, Labels: map[string]string{appsv1.ControllerRevisionHashLabelKey: "old"}}} + requested, applied = withDefaultRequests(requested), withDefaultRequests(applied) + pod.Spec.Containers = []corev1.Container{{Name: redisContainerName, Resources: requested}, {Name: "exporter"}} + pod.Status.ContainerStatuses = []corev1.ContainerStatus{ + {Name: redisContainerName, Resources: &applied}, + {Name: "exporter", Resources: &corev1.ResourceRequirements{}}, + } + pod.Status.Conditions = conditions + return pod +} + +func condition(t corev1.PodConditionType, reason string, age time.Duration) corev1.PodCondition { + return corev1.PodCondition{Type: t, Status: corev1.ConditionTrue, Reason: reason, Message: "msg", LastTransitionTime: metav1.NewTime(time.Now().Add(-age))} +} + +// podFrom is a pod created from the "old" revision template, with its +// resources applied. +func podFrom(t corev1.PodTemplateSpec) *corev1.Pod { + pod := stalePod(corev1.ResourceRequirements{}, corev1.ResourceRequirements{}) + pod.Spec.Containers, pod.Status.ContainerStatuses = nil, nil + for _, c := range t.Spec.Containers { + r := withDefaultRequests(c.Resources) + pod.Spec.Containers = append(pod.Spec.Containers, corev1.Container{Name: c.Name, Resources: r}) + pod.Status.ContainerStatuses = append(pod.Status.ContainerStatuses, corev1.ContainerStatus{Name: c.Name, Resources: &r}) + } + return pod +} + +type resizeCase struct { + support k8s.PodResizeSupport + old, new corev1.PodTemplateSpec + pod *corev1.Pod +} + +func runResize(t *testing.T, c resizeCase, setup func(ms *mK8SService.Services)) (ResizeResult, *mK8SService.Services, error) { + t.Helper() + rf := &redisfailoverv1.RedisFailover{ObjectMeta: metav1.ObjectMeta{Name: "test", Namespace: "testns"}} + ms := &mK8SService.Services{} + ms.On("PodResizeSupport").Return(c.support, nil) + ms.On("GetPod", "testns", resizePod).Return(c.pod, nil) + ms.On("GetControllerRevision", "testns", "old").Return(revision("old", c.old), nil) + ms.On("GetControllerRevision", "testns", "new").Return(revision("new", c.new), nil) + ms.On("UpdatePodAnnotations", "testns", resizePod, mock.Anything).Maybe().Return(nil) + if setup != nil { + setup(ms) + } + result, err := NewRedisFailoverHealer(ms, nil, log.Dummy).ResizePodInPlace(rf, resizePod, "new") + return result, ms, err +} + +var fullSupport = k8s.PodResizeSupport{Supported: true, MemoryLimitDecrease: true} + +func TestResizePodInPlaceRequestsTheResize(t *testing.T) { + old, new := resources("500m", "1Gi"), resources("1", "2Gi") + var got map[string]corev1.ResourceRequirements + result, ms, err := runResize(t, resizeCase{fullSupport, podTemplate(old), podTemplate(new), stalePod(old, old)}, func(ms *mK8SService.Services) { + ms.On("ResizePod", "testns", resizePod, mock.Anything).Once().Run(func(args mock.Arguments) { + got = args.Get(2).(map[string]corev1.ResourceRequirements) + }).Return(nil) + }) + assert.NoError(t, err) + assert.Equal(t, ResizeWaiting, result.Action) + ms.AssertExpectations(t) + // Requests are defaulted to the limits, as the API server does for pods. + assert.Equal(t, withDefaultRequests(new), got[redisContainerName]) + assert.Contains(t, got, "exporter") +} + +func TestResizePodInPlaceFollowsTheKubelet(t *testing.T) { + old, new := resources("500m", "1Gi"), resources("1", "2Gi") + tests := map[string]struct { + pod *corev1.Pod + action ResizeAction + }{ + "not applied yet": {stalePod(new, old), ResizeWaiting}, + "infeasible": {stalePod(new, old, condition(corev1.PodResizePending, corev1.PodReasonInfeasible, 0)), ResizeRecreate}, + "deferred": {stalePod(new, old, condition(corev1.PodResizePending, corev1.PodReasonDeferred, time.Minute)), ResizeWaiting}, + "deferred for too long": {stalePod(new, old, condition(corev1.PodResizePending, corev1.PodReasonDeferred, time.Hour)), ResizeRecreate}, + "in progress": {stalePod(new, old, condition(corev1.PodResizeInProgress, "", time.Hour)), ResizeWaiting}, + "failing": {stalePod(new, old, condition(corev1.PodResizeInProgress, corev1.PodReasonError, time.Minute)), ResizeWaiting}, + "failing for too long": {stalePod(new, old, condition(corev1.PodResizeInProgress, corev1.PodReasonError, time.Hour)), ResizeRecreate}, + "applied": {stalePod(new, new), ResizeDone}, + "pending after applying": {stalePod(new, new, condition(corev1.PodResizePending, corev1.PodReasonDeferred, 0)), ResizeWaiting}, + } + for name, test := range tests { + t.Run(name, func(t *testing.T) { + result, ms, err := runResize(t, resizeCase{fullSupport, podTemplate(old), podTemplate(new), test.pod}, func(ms *mK8SService.Services) { + if test.action == ResizeDone { + ms.On("UpdatePodLabels", "testns", resizePod, map[string]string{appsv1.ControllerRevisionHashLabelKey: "new"}).Once().Return(nil) + } + }) + assert.NoError(t, err) + assert.Equal(t, test.action, result.Action) + ms.AssertExpectations(t) + if test.action == ResizeDone { + ms.AssertCalled(t, "UpdatePodAnnotations", "testns", resizePod, map[string]string{resizeRequestedAnnotation: ""}) + } + }) + } +} + +// The kubelet's memory usage includes the page cache, so it can refuse a +// memory decrease that the recreated pod fits. +func TestResizePodInPlaceRecreatesForARefusedMemoryDecrease(t *testing.T) { + old, new := resources("1", "2Gi"), resources("1", "1Gi") + failing := func(age time.Duration) *corev1.Pod { + c := condition(corev1.PodResizeInProgress, corev1.PodReasonError, age) + c.Message = "cannot decrease memory limits: attempting to set container 'redis' memory limit (1073741824) below current usage (1181116006)" + return stalePod(new, old, c) + } + result, _, err := runResize(t, resizeCase{fullSupport, podTemplate(old), podTemplate(new), failing(time.Minute)}, nil) + assert.NoError(t, err) + assert.Equal(t, ResizeWaiting, result.Action) + + result, _, err = runResize(t, resizeCase{fullSupport, podTemplate(old), podTemplate(new), failing(time.Hour)}, nil) + assert.NoError(t, err) + assert.Equal(t, ResizeRecreate, result.Action) +} + +// A resize that the kubelet neither applies nor reports on in time falls back +// to recreating the pod. +func TestResizePodInPlaceTimesOutWithoutCondition(t *testing.T) { + old, new := resources("500m", "1Gi"), resources("1", "2Gi") + requested := func(age time.Duration) *corev1.Pod { + pod := stalePod(new, old) + pod.Annotations = map[string]string{resizeRequestedAnnotation: time.Now().Add(-age).UTC().Format(time.RFC3339)} + return pod + } + noAnnotation := func(ms *mK8SService.Services) { + ms.ExpectedCalls = ms.ExpectedCalls[:len(ms.ExpectedCalls)-1] + } + + result, ms, err := runResize(t, resizeCase{fullSupport, podTemplate(old), podTemplate(new), requested(time.Minute)}, noAnnotation) + assert.NoError(t, err) + assert.Equal(t, ResizeWaiting, result.Action) + ms.AssertExpectations(t) + + result, _, err = runResize(t, resizeCase{fullSupport, podTemplate(old), podTemplate(new), requested(time.Hour)}, noAnnotation) + assert.NoError(t, err) + assert.Equal(t, ResizeRecreate, result.Action) + assert.Equal(t, "resize not applied by the kubelet", result.Message) + + // Without a recorded request, the timeout starts now. + result, ms, err = runResize(t, resizeCase{fullSupport, podTemplate(old), podTemplate(new), stalePod(new, old)}, nil) + assert.NoError(t, err) + assert.Equal(t, ResizeWaiting, result.Action) + ms.AssertCalled(t, "UpdatePodAnnotations", "testns", resizePod, mock.Anything) +} + +// A condition from an earlier resize request is not acted on. +func TestResizePodInPlaceIgnoresStaleConditions(t *testing.T) { + old, new := resources("500m", "1Gi"), resources("1", "2Gi") + stale := condition(corev1.PodResizePending, corev1.PodReasonInfeasible, time.Hour) + stale.ObservedGeneration = 1 + pod := stalePod(new, old, stale) + pod.Generation = 2 + result, _, err := runResize(t, resizeCase{fullSupport, podTemplate(old), podTemplate(new), pod}, nil) + assert.NoError(t, err) + assert.Equal(t, ResizeWaiting, result.Action) +} + +// Values added by admission, e.g. a LimitRange default, are kept. +func TestResizePodInPlaceKeepsAdmittedResources(t *testing.T) { + memory := func(size string) corev1.ResourceRequirements { + return corev1.ResourceRequirements{Limits: corev1.ResourceList{corev1.ResourceMemory: resource.MustParse(size)}} + } + admitted := func(r corev1.ResourceRequirements) corev1.ResourceRequirements { + r = withDefaultRequests(r) + r.Requests[corev1.ResourceCPU] = resource.MustParse("100m") + return r + } + var got map[string]corev1.ResourceRequirements + _, ms, err := runResize(t, resizeCase{fullSupport, podTemplate(memory("1Gi")), podTemplate(memory("2Gi")), stalePod(admitted(memory("1Gi")), admitted(memory("1Gi")))}, func(ms *mK8SService.Services) { + ms.On("ResizePod", "testns", resizePod, mock.Anything).Once().Run(func(args mock.Arguments) { + got = args.Get(2).(map[string]corev1.ResourceRequirements) + }).Return(nil) + }) + assert.NoError(t, err) + ms.AssertExpectations(t) + assert.Equal(t, admitted(memory("2Gi")), got[redisContainerName]) + + // Once applied, the pod matches and moves to the update revision. + result, _, err := runResize(t, resizeCase{fullSupport, podTemplate(memory("1Gi")), podTemplate(memory("2Gi")), stalePod(admitted(memory("2Gi")), admitted(memory("2Gi")))}, func(ms *mK8SService.Services) { + ms.On("UpdatePodLabels", "testns", resizePod, mock.Anything).Once().Return(nil) + }) + assert.NoError(t, err) + assert.Equal(t, ResizeDone, result.Action) +} + +func TestResizePodInPlaceRecreates(t *testing.T) { + cpu := func(r corev1.ResourceRequirements) corev1.ResourceRequirements { + r.Requests = corev1.ResourceList{corev1.ResourceCPU: resource.MustParse("100m")} + delete(r.Limits, corev1.ResourceCPU) + return r + } + withStorage := func(r corev1.ResourceRequirements, size string) corev1.ResourceRequirements { + r.Limits[corev1.ResourceEphemeralStorage] = resource.MustParse(size) + return r + } + other := podTemplate(resources("1", "2Gi")) + other.Spec.Containers[0].Image = "redis:8" + claimed := podTemplate(resources("1", "1Gi")) + claimed.Spec.Containers[0].Resources.Claims = []corev1.ResourceClaim{{Name: "gpu"}} + tests := map[string]struct { + support k8s.PodResizeSupport + old, new corev1.PodTemplateSpec + reason string + }{ + "unsupported cluster": {k8s.PodResizeSupport{}, podTemplate(resources("1", "1Gi")), podTemplate(resources("2", "1Gi")), ""}, + "more than resources change": {fullSupport, podTemplate(resources("1", "1Gi")), other, "the update changes more than container resources"}, + "resource claims change": {fullSupport, podTemplate(resources("1", "1Gi")), claimed, "the update changes more than container resources"}, + "limit replaced": {fullSupport, podTemplate(resources("1", "1Gi")), podTemplate(corev1.ResourceRequirements{Limits: corev1.ResourceList{corev1.ResourceCPU: resource.MustParse("1"), corev1.ResourceEphemeralStorage: resource.MustParse("1Gi")}}), "requests or limits are added or removed"}, + "limit removed": {fullSupport, podTemplate(resources("1", "1Gi")), podTemplate(cpu(resources("1", "1Gi"))), "requests or limits are added or removed"}, + "other resources change": {fullSupport, podTemplate(withStorage(resources("1", "1Gi"), "1Gi")), podTemplate(withStorage(resources("1", "1Gi"), "2Gi")), "only cpu and memory can be resized in place"}, + "memory decrease before 1.35": {k8s.PodResizeSupport{Supported: true}, podTemplate(resources("1", "2Gi")), podTemplate(resources("1", "1Gi")), "lowering a memory limit in place needs Kubernetes 1.35"}, + } + for name, test := range tests { + t.Run(name, func(t *testing.T) { + result, _, err := runResize(t, resizeCase{test.support, test.old, test.new, podFrom(test.old)}, nil) + assert.NoError(t, err) + assert.Equal(t, ResizeRecreate, result.Action) + assert.Equal(t, test.reason, result.Message) + }) + } +} + +// A kubelet without in-place resize support reports no container resources. +func TestResizePodInPlaceRecreatesOnAKubeletWithoutSupport(t *testing.T) { + old := podTemplate(resources("1", "1Gi")) + pod := podFrom(old) + pod.Status.ContainerStatuses = nil + result, _, err := runResize(t, resizeCase{fullSupport, old, podTemplate(resources("1", "2Gi")), pod}, nil) + assert.NoError(t, err) + assert.Equal(t, ResizeRecreate, result.Action) + assert.Equal(t, "the kubelet does not report the container's resources", result.Message) +} + +func TestResizePodInPlaceRecreatesWithoutRevisionsOrSupport(t *testing.T) { + rf := &redisfailoverv1.RedisFailover{ObjectMeta: metav1.ObjectMeta{Name: "test", Namespace: "testns"}} + old := resources("1", "1Gi") + notFound := apierrors.NewNotFound(schema.GroupResource{Group: "apps", Resource: "controllerrevisions"}, "old") + + t.Run("disabled", func(t *testing.T) { + rf := rf.DeepCopy() + rf.Spec.Redis.InPlaceResize = "Disabled" + result, err := NewRedisFailoverHealer(&mK8SService.Services{}, nil, log.Dummy).ResizePodInPlace(rf, resizePod, "new") + assert.NoError(t, err) + assert.Equal(t, ResizeRecreate, result.Action) + }) + + t.Run("revisions forbidden", func(t *testing.T) { + ms := &mK8SService.Services{} + ms.On("PodResizeSupport").Return(fullSupport, nil) + ms.On("GetPod", "testns", resizePod).Return(stalePod(old, old), nil) + ms.On("GetControllerRevision", "testns", "old").Return(nil, apierrors.NewForbidden(schema.GroupResource{Group: "apps", Resource: "controllerrevisions"}, "old", errors.New("rbac"))) + result, err := NewRedisFailoverHealer(ms, nil, log.Dummy).ResizePodInPlace(rf, resizePod, "new") + assert.NoError(t, err) + assert.Equal(t, ResizeRecreate, result.Action) + }) + + for _, missing := range []string{"old", "new"} { + t.Run(missing+" revision gone", func(t *testing.T) { + ms := &mK8SService.Services{} + ms.On("PodResizeSupport").Return(fullSupport, nil) + ms.On("GetPod", "testns", resizePod).Return(stalePod(old, old), nil) + for _, rev := range []string{"old", "new"} { + if rev == missing { + ms.On("GetControllerRevision", "testns", rev).Return(nil, notFound) + } else { + ms.On("GetControllerRevision", "testns", rev).Return(revision(rev, podTemplate(old)), nil) + } + } + result, err := NewRedisFailoverHealer(ms, nil, log.Dummy).ResizePodInPlace(rf, resizePod, "new") + assert.NoError(t, err) + assert.Equal(t, ResizeRecreate, result.Action) + }) + } +} + +func TestResizePodInPlaceErrors(t *testing.T) { + old, new := resources("500m", "1Gi"), resources("1", "2Gi") + fail := errors.New("boom") + + t.Run("resize rejected", func(t *testing.T) { + forbidden := apierrors.NewForbidden(schema.GroupResource{Resource: "pods/resize"}, resizePod, fail) + result, _, err := runResize(t, resizeCase{fullSupport, podTemplate(old), podTemplate(new), stalePod(old, old)}, func(ms *mK8SService.Services) { + ms.On("ResizePod", "testns", resizePod, mock.Anything).Return(forbidden) + }) + assert.NoError(t, err) + assert.Equal(t, ResizeRecreate, result.Action) + }) + + tests := map[string]func(ms *mK8SService.Services){ + // Retried, as recreating the master would fail over. + "detecting support": func(ms *mK8SService.Services) { + ms.ExpectedCalls = nil + ms.On("PodResizeSupport").Return(k8s.PodResizeSupport{}, fail) + }, + "resizing": func(ms *mK8SService.Services) { + ms.On("ResizePod", "testns", resizePod, mock.Anything).Return(fail) + }, + "reading the pod": func(ms *mK8SService.Services) { + ms.ExpectedCalls = nil + ms.On("PodResizeSupport").Return(fullSupport, nil) + ms.On("GetPod", "testns", resizePod).Return(nil, fail) + }, + "reading a revision": func(ms *mK8SService.Services) { + ms.ExpectedCalls = nil + ms.On("PodResizeSupport").Return(fullSupport, nil) + ms.On("GetPod", "testns", resizePod).Return(stalePod(old, old), nil) + ms.On("GetControllerRevision", "testns", "old").Return(nil, fail) + }, + } + for name, setup := range tests { + t.Run(name, func(t *testing.T) { + _, _, err := runResize(t, resizeCase{fullSupport, podTemplate(old), podTemplate(new), stalePod(old, old)}, setup) + assert.ErrorIs(t, err, fail) + }) + } + + t.Run("corrupt revision", func(t *testing.T) { + _, _, err := runResize(t, resizeCase{fullSupport, podTemplate(old), podTemplate(new), stalePod(old, old)}, func(ms *mK8SService.Services) { + ms.ExpectedCalls = nil + ms.On("PodResizeSupport").Return(fullSupport, nil) + ms.On("GetPod", "testns", resizePod).Return(stalePod(old, old), nil) + ms.On("GetControllerRevision", "testns", "old").Return(&appsv1.ControllerRevision{Data: runtime.RawExtension{Raw: []byte("{")}}, nil) + }) + assert.Error(t, err) + }) + + for name, pod := range map[string]*corev1.Pod{"requesting the resize": stalePod(old, old), "recording the request": stalePod(new, old)} { + t.Run(name, func(t *testing.T) { + _, _, err := runResize(t, resizeCase{fullSupport, podTemplate(old), podTemplate(new), pod}, func(ms *mK8SService.Services) { + ms.ExpectedCalls = ms.ExpectedCalls[:len(ms.ExpectedCalls)-1] + ms.On("UpdatePodAnnotations", "testns", resizePod, mock.Anything).Return(fail) + }) + assert.ErrorIs(t, err, fail) + }) + } + + t.Run("relabelling", func(t *testing.T) { + _, _, err := runResize(t, resizeCase{fullSupport, podTemplate(old), podTemplate(new), stalePod(new, new)}, func(ms *mK8SService.Services) { + ms.On("UpdatePodLabels", "testns", resizePod, mock.Anything).Return(fail) + }) + assert.ErrorIs(t, err, fail) + }) + + t.Run("clearing the request", func(t *testing.T) { + _, _, err := runResize(t, resizeCase{fullSupport, podTemplate(old), podTemplate(new), stalePod(new, new)}, func(ms *mK8SService.Services) { + ms.ExpectedCalls = ms.ExpectedCalls[:len(ms.ExpectedCalls)-1] + ms.On("UpdatePodLabels", "testns", resizePod, mock.Anything).Return(nil) + ms.On("UpdatePodAnnotations", "testns", resizePod, mock.Anything).Return(fail) + }) + assert.ErrorIs(t, err, fail) + }) +} + +func TestOverlay(t *testing.T) { + list := func(kv ...string) corev1.ResourceList { + l := corev1.ResourceList{} + for i := 0; i < len(kv); i += 2 { + l[corev1.ResourceName(kv[i])] = resource.MustParse(kv[i+1]) + } + return l + } + assert.Equal(t, list("memory", "2Gi", "cpu", "1"), overlay(list("memory", "1Gi", "cpu", "1"), list("memory", "2Gi"))) + assert.Equal(t, list("memory", "2Gi", "ephemeral-storage", "1Gi"), overlay(list("ephemeral-storage", "1Gi"), list("memory", "2Gi", "ephemeral-storage", "2Gi"))) + assert.Equal(t, list("cpu", "1"), overlay(nil, list("cpu", "1"))) + assert.Nil(t, overlay(nil, nil)) +} + +// A pod still running an earlier resize that a newer revision superseded is +// resized to the newer revision's values, not only the ones it changes. +func TestResizePodInPlaceSupersededResize(t *testing.T) { + a, b, c := resources("1", "1Gi"), resources("2", "1Gi"), resources("1", "2Gi") + var got map[string]corev1.ResourceRequirements + _, ms, err := runResize(t, resizeCase{fullSupport, podTemplate(a), podTemplate(c), stalePod(b, b)}, func(ms *mK8SService.Services) { + ms.On("ResizePod", "testns", resizePod, mock.Anything).Once().Run(func(args mock.Arguments) { + got = args.Get(2).(map[string]corev1.ResourceRequirements) + }).Return(nil) + }) + assert.NoError(t, err) + ms.AssertExpectations(t) + assert.Equal(t, withDefaultRequests(c), got[redisContainerName]) + + // Reverting to the pod's own revision resizes the pod back. + pod := stalePod(b, a) + pod.Labels[appsv1.ControllerRevisionHashLabelKey] = "new" + got = nil + _, _, err = runResize(t, resizeCase{fullSupport, podTemplate(a), podTemplate(a), pod}, func(ms *mK8SService.Services) { + ms.On("ResizePod", "testns", resizePod, mock.Anything).Once().Run(func(args mock.Arguments) { + got = args.Get(2).(map[string]corev1.ResourceRequirements) + }).Return(nil) + }) + assert.NoError(t, err) + assert.Equal(t, withDefaultRequests(a), got[redisContainerName]) + + // Reverting a memory increase lowers the pod's limit, although the revisions have the same. + result, _, err := runResize(t, resizeCase{k8s.PodResizeSupport{Supported: true}, podTemplate(a), podTemplate(a), stalePod(c, c)}, nil) + assert.NoError(t, err) + assert.Equal(t, ResizeRecreate, result.Action) + assert.Equal(t, "lowering a memory limit in place needs Kubernetes 1.35", result.Message) + + // Before 1.35 the API server refuses to revert an increase the kubelet has not applied yet. + invalid := apierrors.NewInvalid(schema.GroupKind{Kind: "Pod"}, resizePod, nil) + result, _, err = runResize(t, resizeCase{k8s.PodResizeSupport{Supported: true}, podTemplate(a), podTemplate(a), stalePod(c, a)}, func(ms *mK8SService.Services) { + ms.On("ResizePod", "testns", resizePod, mock.Anything).Once().Return(invalid) + }) + assert.NoError(t, err) + assert.Equal(t, ResizeRecreate, result.Action) + assert.Equal(t, "resize rejected: "+invalid.Error(), result.Message) +} + +// A condition keeps its transition time when a newer resize supersedes the +// one it reported on, so the timeout also counts from the latest request. +func TestResizePodInPlaceTimesOutFromTheLatestRequest(t *testing.T) { + old, new := resources("500m", "1Gi"), resources("1", "2Gi") + for name, c := range map[string]corev1.PodCondition{ + "deferred": condition(corev1.PodResizePending, corev1.PodReasonDeferred, time.Hour), + "failing": condition(corev1.PodResizeInProgress, corev1.PodReasonError, time.Hour), + } { + t.Run(name, func(t *testing.T) { + pod := stalePod(new, old, c) + pod.Annotations = map[string]string{resizeRequestedAnnotation: time.Now().Add(-time.Minute).UTC().Format(time.RFC3339)} + result, _, err := runResize(t, resizeCase{fullSupport, podTemplate(old), podTemplate(new), pod}, nil) + assert.NoError(t, err) + assert.Equal(t, ResizeWaiting, result.Action) + }) + } +} diff --git a/service/k8s/pod.go b/service/k8s/pod.go index 66305bc5a..7015f0d57 100644 --- a/service/k8s/pod.go +++ b/service/k8s/pod.go @@ -3,6 +3,12 @@ package k8s import ( "context" "encoding/json" + "fmt" + "sort" + "strconv" + "strings" + "sync" + "time" "k8s.io/apimachinery/pkg/types" @@ -21,6 +27,17 @@ type Pod interface { ListPods(namespace string) (*corev1.PodList, error) UpdatePodLabels(namespace, podName string, labels map[string]string) error UpdatePodAnnotations(namespace, podName string, annotations map[string]string) error + ResizePod(namespace, podName string, resources map[string]corev1.ResourceRequirements) error + PodResizeSupport() (PodResizeSupport, error) +} + +// PodResizeSupport describes the cluster's support for in-place pod resize. +type PodResizeSupport struct { + // Supported is set from Kubernetes 1.33, where in-place resize is on by default. + Supported bool + // MemoryLimitDecrease is set from Kubernetes 1.35, which allows lowering a + // memory limit without restarting the container. + MemoryLimitDecrease bool } // PodService is the pod service implementation using API calls to kubernetes. @@ -28,8 +45,16 @@ type PodService struct { kubeClient kubernetes.Interface logger log.Logger metricsRecorder metrics.Recorder + + resizeSupportMu sync.Mutex + resizeSupport PodResizeSupport + resizeSupportChecked time.Time } +// podResizeSupportTTL bounds how long the detected support is cached, so a +// cluster upgrade is picked up without restarting the operator. +const podResizeSupportTTL = 10 * time.Minute + // NewPodService returns a new Pod KubeService. func NewPodService(kubeClient kubernetes.Interface, logger log.Logger, metricsRecorder metrics.Recorder) *PodService { logger = logger.With("service", "k8s.pod") @@ -110,3 +135,42 @@ func (p *PodService) UpdatePodAnnotations(namespace, podName string, annotations } return err } + +// ResizePod sets the resources of the named containers through the pod's +// resize subresource. +func (p *PodService) ResizePod(namespace, podName string, resources map[string]corev1.ResourceRequirements) error { + names := make([]string, 0, len(resources)) + for name := range resources { + names = append(names, name) + } + sort.Strings(names) + containers := make([]map[string]interface{}, 0, len(resources)) + for _, name := range names { + containers = append(containers, map[string]interface{}{"name": name, "resources": resources[name]}) + } + payloadBytes, _ := json.Marshal(map[string]interface{}{"spec": map[string]interface{}{"containers": containers}}) + _, err := p.kubeClient.CoreV1().Pods(namespace).Patch(context.TODO(), podName, types.StrategicMergePatchType, payloadBytes, metav1.PatchOptions{}, "resize") + recordMetrics(namespace, "Pod", podName, "RESIZE", err, p.metricsRecorder) + return err +} + +// PodResizeSupport reports the cluster's support for in-place pod resize. +func (p *PodService) PodResizeSupport() (PodResizeSupport, error) { + p.resizeSupportMu.Lock() + defer p.resizeSupportMu.Unlock() + if !p.resizeSupportChecked.IsZero() && time.Since(p.resizeSupportChecked) < podResizeSupportTTL { + return p.resizeSupport, nil + } + version, err := p.kubeClient.Discovery().ServerVersion() + if err != nil { + return PodResizeSupport{}, err + } + // Minor can carry a suffix, e.g. "35+". + minor, err := strconv.Atoi(strings.TrimRight(version.Minor, "+")) + if err != nil || version.Major != "1" { + return PodResizeSupport{}, fmt.Errorf("unexpected server version %s.%s", version.Major, version.Minor) + } + p.resizeSupport = PodResizeSupport{Supported: minor >= 33, MemoryLimitDecrease: minor >= 35} + p.resizeSupportChecked = time.Now() + return p.resizeSupport, nil +} diff --git a/service/k8s/resize_test.go b/service/k8s/resize_test.go new file mode 100644 index 000000000..95d101bdf --- /dev/null +++ b/service/k8s/resize_test.go @@ -0,0 +1,103 @@ +package k8s_test + +import ( + "encoding/json" + "errors" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + appsv1 "k8s.io/api/apps/v1" + corev1 "k8s.io/api/core/v1" + "k8s.io/apimachinery/pkg/api/resource" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/apimachinery/pkg/runtime" + "k8s.io/apimachinery/pkg/version" + fakediscovery "k8s.io/client-go/discovery/fake" + kubernetes "k8s.io/client-go/kubernetes/fake" + kubetesting "k8s.io/client-go/testing" + + "github.com/saremox/redis-operator/log" + "github.com/saremox/redis-operator/metrics" + "github.com/saremox/redis-operator/service/k8s" +) + +func TestPodServiceResizePod(t *testing.T) { + client := &kubernetes.Clientset{} + var patch kubetesting.PatchActionImpl + client.AddReactor("patch", "pods", func(action kubetesting.Action) (bool, runtime.Object, error) { + patch = action.(kubetesting.PatchActionImpl) + return true, &corev1.Pod{}, nil + }) + limits := corev1.ResourceRequirements{Limits: corev1.ResourceList{corev1.ResourceMemory: resource.MustParse("1Gi")}} + + err := k8s.NewPodService(client, log.Dummy, metrics.Dummy).ResizePod("ns", "pod", map[string]corev1.ResourceRequirements{"redis": limits, "exporter": {}}) + require.NoError(t, err) + + assert.Equal(t, "resize", patch.GetSubresource()) + var body struct { + Spec struct { + Containers []corev1.Container `json:"containers"` + } `json:"spec"` + } + require.NoError(t, json.Unmarshal(patch.GetPatch(), &body)) + require.Len(t, body.Spec.Containers, 2) + assert.Equal(t, "exporter", body.Spec.Containers[0].Name) + assert.Equal(t, "redis", body.Spec.Containers[1].Name) + assert.Equal(t, "1Gi", body.Spec.Containers[1].Resources.Limits.Memory().String()) +} + +func TestPodServicePodResizeSupport(t *testing.T) { + tests := map[string]struct { + minor string + want k8s.PodResizeSupport + wantErr bool + }{ + "1.32": {minor: "32", want: k8s.PodResizeSupport{}}, + "1.33": {minor: "33", want: k8s.PodResizeSupport{Supported: true}}, + "1.35+": {minor: "35+", want: k8s.PodResizeSupport{Supported: true, MemoryLimitDecrease: true}}, + "unparsed": {minor: "x", wantErr: true}, + } + for name, test := range tests { + t.Run(name, func(t *testing.T) { + client := kubernetes.NewClientset() + client.Discovery().(*fakediscovery.FakeDiscovery).FakedServerVersion = &version.Info{Major: "1", Minor: test.minor} + service := k8s.NewPodService(client, log.Dummy, metrics.Dummy) + + got, err := service.PodResizeSupport() + if test.wantErr { + assert.Error(t, err) + return + } + require.NoError(t, err) + assert.Equal(t, test.want, got) + + // Cached: a changed server version is not seen right away. + client.Discovery().(*fakediscovery.FakeDiscovery).FakedServerVersion = &version.Info{Major: "1", Minor: "40"} + got, err = service.PodResizeSupport() + require.NoError(t, err) + assert.Equal(t, test.want, got) + }) + } +} + +func TestPodServicePodResizeSupportDiscoveryError(t *testing.T) { + client := kubernetes.NewClientset() + client.PrependReactor("get", "version", func(kubetesting.Action) (bool, runtime.Object, error) { + return true, nil, errors.New("discovery down") + }) + _, err := k8s.NewPodService(client, log.Dummy, metrics.Dummy).PodResizeSupport() + assert.EqualError(t, err, "discovery down") +} + +func TestStatefulSetServiceGetControllerRevision(t *testing.T) { + revision := &appsv1.ControllerRevision{ObjectMeta: metav1.ObjectMeta{Name: "rfr-test-abc", Namespace: "ns"}, Revision: 2} + client := kubernetes.NewClientset(revision) + + got, err := k8s.NewStatefulSetService(client, log.Dummy, metrics.Dummy).GetControllerRevision("ns", "rfr-test-abc") + require.NoError(t, err) + assert.Equal(t, int64(2), got.Revision) + + _, err = k8s.NewStatefulSetService(client, log.Dummy, metrics.Dummy).GetControllerRevision("ns", "missing") + assert.Error(t, err) +} diff --git a/service/k8s/statefulset.go b/service/k8s/statefulset.go index 2048927b3..386ac4114 100644 --- a/service/k8s/statefulset.go +++ b/service/k8s/statefulset.go @@ -29,6 +29,7 @@ type StatefulSet interface { CreateOrUpdateStatefulSet(namespace string, statefulSet *appsv1.StatefulSet) error DeleteStatefulSet(namespace string, name string) error ListStatefulSets(namespace string) (*appsv1.StatefulSetList, error) + GetControllerRevision(namespace, name string) (*appsv1.ControllerRevision, error) } // StatefulSetService is the service account service implementation using API calls to kubernetes. @@ -190,3 +191,10 @@ func (s *StatefulSetService) ListStatefulSets(namespace string) (*appsv1.Statefu recordMetrics(namespace, "StatefulSet", metrics.NOT_APPLICABLE, "LIST", err, s.metricsRecorder) return stsList, err } + +// GetControllerRevision returns the named ControllerRevision. +func (s *StatefulSetService) GetControllerRevision(namespace, name string) (*appsv1.ControllerRevision, error) { + revision, err := s.kubeClient.AppsV1().ControllerRevisions(namespace).Get(context.TODO(), name, metav1.GetOptions{}) + recordMetrics(namespace, "ControllerRevision", name, "GET", err, s.metricsRecorder) + return revision, err +} From 3494105ae650234ed247daca96c154874d0657c7 Mon Sep 17 00:00:00 2001 From: Michael Strassberger Date: Sun, 27 Sep 2026 02:18:24 +0200 Subject: [PATCH 21/24] fix(failover): don't promote while a ready Redis pod doesn't answer (#193) * fix(failover): don't promote while a ready Redis pod doesn't answer GetNumberMasters skipped a pod it could not query, so a master that was briefly unreachable counted as absent. Zero masters drives promotion, so the operator replaced a running master: in operator-managed mode a 20s CLIENT PAUSE on the master was enough. A ready pod that doesn't answer now makes the count unknown when no master answered, and the reconcile stops. A pod Kubernetes has marked not ready is still skipped, so failover after a node loss or a crashed Redis goes ahead once the pod leaves the master Service. Ported from powerhome/redis-operator#106. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01G4mmyo3sqLwf23m5FE2nBE * fix(redis): shorten the Redis client timeouts Every client used the go-redis defaults: a 3s read timeout, 5s dial timeout and 3 retries. A pod that accepts connections but never answers, like one on a frozen node, held each call for 12s, and a failover after a node loss waited on several of them. Use 2s timeouts and one retry, so such a call gives up after 4s. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01G4mmyo3sqLwf23m5FE2nBE --------- Co-authored-by: Claude --- operator/redisfailover/service/check.go | 24 +++- operator/redisfailover/service/check_test.go | 46 ++++++++ service/redis/client.go | 111 ++++++------------- service/redis/client_test.go | 23 ++++ 4 files changed, 119 insertions(+), 85 deletions(-) diff --git a/operator/redisfailover/service/check.go b/operator/redisfailover/service/check.go index d977ddc87..03952fd74 100644 --- a/operator/redisfailover/service/check.go +++ b/operator/redisfailover/service/check.go @@ -383,10 +383,13 @@ func (r *RedisFailoverChecker) GetMasterIP(rf *redisfailoverv1.RedisFailover) (s return masters[0], nil } -// GetNumberMasters returns the number of redis nodes that are working as a master +// GetNumberMasters returns the number of redis nodes that are working as a master. +// A ready pod that does not answer may still be the master, so if no pod +// answers as master it returns an error rather than zero, and callers don't +// promote over it. A pod Kubernetes has marked not ready is skipped. func (r *RedisFailoverChecker) GetNumberMasters(rf *redisfailoverv1.RedisFailover) (int, error) { nMasters := 0 - rips, err := r.GetRedisesIPs(rf) + rps, err := r.k8sService.GetStatefulSetPods(rf.Namespace, GetRedisName(rf)) if err != nil { r.logger.Error(err.Error()) return nMasters, err @@ -398,17 +401,28 @@ func (r *RedisFailoverChecker) GetNumberMasters(rf *redisfailoverv1.RedisFailove return nMasters, err } + var unanswered error rport := getRedisPort(rf.Spec.Redis.Port) - for _, rip := range rips { - master, err := r.redisClient.IsMaster(rip, rport, password) + for i := range rps.Items { + rp := &rps.Items[i] + if rp.Status.Phase != corev1.PodRunning || rp.DeletionTimestamp != nil { + continue + } + master, err := r.redisClient.IsMaster(rp.Status.PodIP, rport, password) if err != nil { - r.logger.Errorf("Get redis info failed, maybe this node is not ready, pod ip: %s", rip) + r.logger.Errorf("Get redis info failed, maybe this node is not ready, pod ip: %s", rp.Status.PodIP) + if unanswered == nil && util.PodIsReady(rp) { + unanswered = fmt.Errorf("ready redis pod %s did not answer: %w", rp.Name, err) + } continue } if master { nMasters++ } } + if nMasters == 0 && unanswered != nil { + return nMasters, unanswered + } return nMasters, nil } diff --git a/operator/redisfailover/service/check_test.go b/operator/redisfailover/service/check_test.go index 8ed2d51a5..822c2162b 100644 --- a/operator/redisfailover/service/check_test.go +++ b/operator/redisfailover/service/check_test.go @@ -821,6 +821,52 @@ func TestGetNumberMastersIsMasterError(t *testing.T) { assert.NoError(err) } +func TestGetNumberMastersReadyPodUnanswered(t *testing.T) { + ready := []corev1.PodCondition{{Type: corev1.PodReady, Status: corev1.ConditionTrue}} + pods := &corev1.PodList{ + Items: []corev1.Pod{ + { + ObjectMeta: metav1.ObjectMeta{Name: "redis-0"}, + Status: corev1.PodStatus{PodIP: "0.0.0.0", Phase: corev1.PodRunning, Conditions: ready}, + }, + { + ObjectMeta: metav1.ObjectMeta{Name: "redis-1"}, + Status: corev1.PodStatus{PodIP: "1.1.1.1", Phase: corev1.PodRunning, Conditions: ready}, + }, + }, + } + + tests := []struct { + name string + otherIsMaster bool + expN int + expErr bool + }{ + {name: "no master answered", otherIsMaster: false, expN: 0, expErr: true}, + {name: "another master answered", otherIsMaster: true, expN: 1, expErr: false}, + } + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + rf := generateRF() + ms := &mK8SService.Services{} + ms.On("GetStatefulSetPods", namespace, rfservice.GetRedisName(rf)).Once().Return(pods, nil) + mr := &mRedisService.Client{} + mr.On("IsMaster", "0.0.0.0", "0", "").Once().Return(false, errors.New("i/o timeout")) + mr.On("IsMaster", "1.1.1.1", "0", "").Once().Return(test.otherIsMaster, nil) + + checker := rfservice.NewRedisFailoverChecker(ms, mr, log.DummyLogger{}, metrics.Dummy) + + n, err := checker.GetNumberMasters(rf) + assert.Equal(t, test.expN, n) + if test.expErr { + assert.ErrorContains(t, err, "redis-0") + } else { + assert.NoError(t, err) + } + }) + } +} + func TestGetNumberMasters(t *testing.T) { assert := assert.New(t) diff --git a/service/redis/client.go b/service/redis/client.go index 0c33bf045..764a44ba7 100644 --- a/service/redis/client.go +++ b/service/redis/client.go @@ -8,6 +8,7 @@ import ( "regexp" "strconv" "strings" + "time" rediscli "github.com/go-redis/redis/v8" "github.com/saremox/redis-operator/log" @@ -89,13 +90,23 @@ var ( redisMasterHostRE = regexp.MustCompile(redisMasterHostREString) ) +// Redis answers the operator in milliseconds. The go-redis defaults (3s read +// timeout, 5s dial timeout, 3 retries) let one unreachable pod hold a +// reconcile for 12-20s per call. +func redisOptions(addr, password string) *rediscli.Options { + return &rediscli.Options{ + Addr: addr, + Password: password, + DialTimeout: 2 * time.Second, + ReadTimeout: 2 * time.Second, + WriteTimeout: 2 * time.Second, + MaxRetries: 1, + } +} + // GetNumberSentinelsInMemory return the number of sentinels that the requested sentinel has func (c *client) GetNumberSentinelsInMemory(ip string) (int32, error) { - options := &rediscli.Options{ - Addr: net.JoinHostPort(ip, sentinelPort), - Password: "", - DB: 0, - } + options := redisOptions(net.JoinHostPort(ip, sentinelPort), "") rClient := rediscli.NewClient(options) defer func(rClient *rediscli.Client) { err := rClient.Close() @@ -132,11 +143,7 @@ func (c *client) GetNumberSentinelsInMemory(ip string) (int32, error) { // GetNumberSentinelSlavesInMemory return the number of sentinels that the requested sentinel has func (c *client) GetNumberSentinelSlavesInMemory(ip string) (int32, error) { - options := &rediscli.Options{ - Addr: net.JoinHostPort(ip, sentinelPort), - Password: "", - DB: 0, - } + options := redisOptions(net.JoinHostPort(ip, sentinelPort), "") rClient := rediscli.NewClient(options) defer func(rClient *rediscli.Client) { err := rClient.Close() @@ -181,11 +188,7 @@ func isSentinelReady(info string) error { // ResetSentinel sends a sentinel reset * for the given sentinel func (c *client) ResetSentinel(ip string) error { - options := &rediscli.Options{ - Addr: net.JoinHostPort(ip, sentinelPort), - Password: "", - DB: 0, - } + options := redisOptions(net.JoinHostPort(ip, sentinelPort), "") rClient := rediscli.NewClient(options) defer func(rClient *rediscli.Client) { err := rClient.Close() @@ -211,11 +214,7 @@ func (c *client) ResetSentinel(ip string) error { // GetSlaveOf returns the master of the given redis, or nil if it's master func (c *client) GetSlaveOf(ip, port, password string) (string, error) { - options := &rediscli.Options{ - Addr: net.JoinHostPort(ip, port), - Password: password, - DB: 0, - } + options := redisOptions(net.JoinHostPort(ip, port), password) rClient := rediscli.NewClient(options) defer func(rClient *rediscli.Client) { err := rClient.Close() @@ -239,11 +238,7 @@ func (c *client) GetSlaveOf(ip, port, password string) (string, error) { } func (c *client) IsMaster(ip, port, password string) (bool, error) { - options := &rediscli.Options{ - Addr: net.JoinHostPort(ip, port), - Password: password, - DB: 0, - } + options := redisOptions(net.JoinHostPort(ip, port), password) rClient := rediscli.NewClient(options) defer func(rClient *rediscli.Client) { err := rClient.Close() @@ -265,11 +260,7 @@ func (c *client) MonitorRedis(ip, monitor, quorum, password string) error { } func (c *client) MonitorRedisWithPort(ip, monitor, port, quorum, password string) error { - options := &rediscli.Options{ - Addr: net.JoinHostPort(ip, sentinelPort), - Password: "", - DB: 0, - } + options := redisOptions(net.JoinHostPort(ip, sentinelPort), "") rClient := rediscli.NewClient(options) defer func(rClient *rediscli.Client) { err := rClient.Close() @@ -310,11 +301,7 @@ func (c *client) MonitorRedisWithPort(ip, monitor, port, quorum, password string } func (c *client) MakeMaster(ip string, port string, password string) error { - options := &rediscli.Options{ - Addr: net.JoinHostPort(ip, port), - Password: password, - DB: 0, - } + options := redisOptions(net.JoinHostPort(ip, port), password) rClient := rediscli.NewClient(options) defer func(rClient *rediscli.Client) { err := rClient.Close() @@ -340,11 +327,7 @@ func (c *client) MakeSlaveOf(ip, masterIP, password string) error { // can be configured with different ports (e.g. Bootstrapping mode's // externally supplied master port). func (c *client) MakeSlaveOfWithPort(ip, port, masterIP, masterPort, password string) error { - options := &rediscli.Options{ - Addr: net.JoinHostPort(ip, port), - Password: password, - DB: 0, - } + options := redisOptions(net.JoinHostPort(ip, port), password) rClient := rediscli.NewClient(options) defer func(rClient *rediscli.Client) { err := rClient.Close() @@ -369,11 +352,7 @@ func closeClient(rClient *rediscli.Client) { // DisconnectClients closes every normal and pub/sub client connection on the // given instance. Replication links are left alone. func (c *client) DisconnectClients(ip, port, password string) error { - options := &rediscli.Options{ - Addr: net.JoinHostPort(ip, port), - Password: password, - DB: 0, - } + options := redisOptions(net.JoinHostPort(ip, port), password) rClient := rediscli.NewClient(options) defer closeClient(rClient) @@ -395,11 +374,7 @@ func (c *client) DisconnectClients(ip, port, password string) error { } func (c *client) GetSentinelMonitor(ip string) (string, string, error) { - options := &rediscli.Options{ - Addr: net.JoinHostPort(ip, sentinelPort), - Password: "", - DB: 0, - } + options := redisOptions(net.JoinHostPort(ip, sentinelPort), "") rClient := rediscli.NewClient(options) defer func(rClient *rediscli.Client) { err := rClient.Close() @@ -425,11 +400,7 @@ func (c *client) GetSentinelMonitor(ip string) (string, string, error) { } func (c *client) SetCustomSentinelConfig(ip string, configs []string) error { - options := &rediscli.Options{ - Addr: net.JoinHostPort(ip, sentinelPort), - Password: "", - DB: 0, - } + options := redisOptions(net.JoinHostPort(ip, sentinelPort), "") rClient := rediscli.NewClient(options) defer func(rClient *rediscli.Client) { err := rClient.Close() @@ -513,11 +484,7 @@ func (c *client) getSentinelMasterInfo(rClient *rediscli.Client) (map[string]str func (c *client) SentinelCheckQuorum(ip string) error { - options := &rediscli.Options{ - Addr: net.JoinHostPort(ip, sentinelPort), - Password: "", - DB: 0, - } + options := redisOptions(net.JoinHostPort(ip, sentinelPort), "") rClient := rediscli.NewSentinelClient(options) defer func(rClient *rediscli.SentinelClient) { err := rClient.Close() @@ -559,11 +526,7 @@ func (c *client) SentinelCheckQuorum(ip string) error { } func (c *client) SetCustomRedisConfig(ip string, port string, configs []string, password string) error { - options := &rediscli.Options{ - Addr: net.JoinHostPort(ip, port), - Password: password, - DB: 0, - } + options := redisOptions(net.JoinHostPort(ip, port), password) rClient := rediscli.NewClient(options) defer func(rClient *rediscli.Client) { err := rClient.Close() @@ -653,11 +616,7 @@ func (c *client) getConfigParameters(config string) (parameter string, value str } func (c *client) SlaveIsReady(ip, port, password string) (bool, error) { - options := &rediscli.Options{ - Addr: net.JoinHostPort(ip, port), - Password: password, - DB: 0, - } + options := redisOptions(net.JoinHostPort(ip, port), password) rClient := rediscli.NewClient(options) defer func(rClient *rediscli.Client) { err := rClient.Close() @@ -681,11 +640,7 @@ func (c *client) SlaveIsReady(ip, port, password string) (bool, error) { // GetReplicationInfo returns detailed replication information for a Redis instance. // This is used for operator-managed failover to select the best replica for promotion. func (c *client) GetReplicationInfo(ip, port, password string) (*ReplicationInfo, error) { - options := &rediscli.Options{ - Addr: net.JoinHostPort(ip, port), - Password: password, - DB: 0, - } + options := redisOptions(net.JoinHostPort(ip, port), password) rClient := rediscli.NewClient(options) defer func(rClient *rediscli.Client) { err := rClient.Close() @@ -749,11 +704,7 @@ func (c *client) GetReplicationInfo(ip, port, password string) (*ReplicationInfo // GetMemoryInfo returns the maxmemory settings, memory usage and role of a Redis instance. func (c *client) GetMemoryInfo(ip, port, password string) (*MemoryInfo, error) { - rClient := rediscli.NewClient(&rediscli.Options{ - Addr: net.JoinHostPort(ip, port), - Password: password, - DB: 0, - }) + rClient := rediscli.NewClient(redisOptions(net.JoinHostPort(ip, port), password)) defer func(rClient *rediscli.Client) { if err := rClient.Close(); err != nil { log.Error(err.Error()) diff --git a/service/redis/client_test.go b/service/redis/client_test.go index 74817a1c8..3e5d2ef95 100644 --- a/service/redis/client_test.go +++ b/service/redis/client_test.go @@ -50,6 +50,29 @@ func newTestClientStruct() *client { // (shared read-only master+replica+sentinel environment) // --------------------------------------------------------------------- +// A frozen node accepts connections and never replies. +func TestIsMasterUnresponsiveTimesOut(t *testing.T) { + l, err := net.Listen("tcp", "127.0.0.1:0") + require.NoError(t, err) + defer func() { _ = l.Close() }() + go func() { + for { + conn, err := l.Accept() + if err != nil { + return + } + defer func() { _ = conn.Close() }() + } + }() + host, port, err := net.SplitHostPort(l.Addr().String()) + require.NoError(t, err) + + start := time.Now() + _, err = newTestClient().IsMaster(host, port, "") + assert.Error(t, err) + assert.Less(t, time.Since(start), 6*time.Second) +} + func TestIsMaster(t *testing.T) { env := getSharedEnv(t) c := newTestClient() From 905cfd813dc922e5f013971331c773acfd781adf Mon Sep 17 00:00:00 2001 From: Michael Strassberger Date: Sun, 27 Sep 2026 14:29:14 +0200 Subject: [PATCH 22/24] fix(chart): fit the CRD in the upgrade hook's ConfigMap (#195) The CRD YAML is 1.09MB, over the 1MiB a ConfigMap holds, so with crds.upgradeHook.enabled every install and upgrade failed. Store it as compact JSON (587KB, descriptions kept) and apply it server-side, as it is also too large for a client-side apply's last-applied annotation. Claude-Session: https://claude.ai/code/session_01G4mmyo3sqLwf23m5FE2nBE Co-authored-by: Claude --- .../redisoperator/templates/crd-upgrade-hook.yaml | 13 +++++++++++-- 1 file changed, 11 insertions(+), 2 deletions(-) diff --git a/charts/redisoperator/templates/crd-upgrade-hook.yaml b/charts/redisoperator/templates/crd-upgrade-hook.yaml index c33a53d2d..ba5bb2e57 100644 --- a/charts/redisoperator/templates/crd-upgrade-hook.yaml +++ b/charts/redisoperator/templates/crd-upgrade-hook.yaml @@ -13,8 +13,11 @@ metadata: helm.sh/hook: pre-install,pre-upgrade helm.sh/hook-weight: "-6" helm.sh/hook-delete-policy: before-hook-creation,hook-succeeded +# The CRD YAML is over the 1MiB a ConfigMap holds; compact JSON fits. data: - {{- (.Files.Glob "crds/*.yaml").AsConfig | nindent 2 }} + {{- range $path, $_ := .Files.Glob "crds/*.yaml" }} + {{ base $path | trimSuffix ".yaml" }}.json: {{ $.Files.Get $path | fromYaml | toJson | quote }} + {{- end }} --- # This hook Job gets its own minimal ServiceAccount/ClusterRole rather than # reusing the operator's own: the operator's ServiceAccount needs broad @@ -123,11 +126,17 @@ spec: - name: crds-upgrade image: "{{ .Values.crds.upgradeHook.image.repository }}:{{ .Values.crds.upgradeHook.image.tag | default "v1.36.2" }}" imagePullPolicy: {{ .Values.crds.upgradeHook.image.pullPolicy }} + # Server-side, as the CRD is too large for the last-applied + # annotation of a client-side apply. command: - kubectl - apply + - --server-side + - --force-conflicts + {{- range $path, $_ := .Files.Glob "crds/*.yaml" }} - -f - - /crds + - /crds/{{ base $path | trimSuffix ".yaml" }}.json + {{- end }} volumeMounts: - name: crds mountPath: /crds From 759b67b17800644aeb03acbb19df7bcb1838c538 Mon Sep 17 00:00:00 2001 From: Michael Strassberger Date: Sun, 27 Sep 2026 22:06:50 +0200 Subject: [PATCH 23/24] fix(chart): set fsGroup on the operator pod, not its container (#196) fsGroup is a PodSecurityContext field. In the container securityContext the API server dropped it with an unknown-field warning, and strict validation rejects the Deployment. Move it to a new podSecurityContext value rendered on the pod spec. Claude-Session: https://claude.ai/code/session_01G4mmyo3sqLwf23m5FE2nBE Co-authored-by: Claude --- charts/redisoperator/templates/deployment.yaml | 4 ++++ charts/redisoperator/values.yaml | 6 +++++- 2 files changed, 9 insertions(+), 1 deletion(-) diff --git a/charts/redisoperator/templates/deployment.yaml b/charts/redisoperator/templates/deployment.yaml index 781415084..826ce7afa 100644 --- a/charts/redisoperator/templates/deployment.yaml +++ b/charts/redisoperator/templates/deployment.yaml @@ -30,6 +30,10 @@ spec: spec: serviceAccountName: {{ template "chart.serviceAccountName" . }} enableServiceLinks: false + {{- with .Values.podSecurityContext }} + securityContext: + {{- toYaml . | nindent 8 }} + {{- end }} {{- if (and .Values.imageCredentials.create (not .Values.imageCredentials.existsSecrets)) }} imagePullSecrets: - name: {{ $fullName }}-{{ $name }} diff --git a/charts/redisoperator/values.yaml b/charts/redisoperator/values.yaml index afa64e3fd..05a1ce643 100644 --- a/charts/redisoperator/values.yaml +++ b/charts/redisoperator/values.yaml @@ -55,7 +55,6 @@ container: securityContext: allowPrivilegeEscalation: false readOnlyRootFilesystem: true - fsGroup: 1000 runAsGroup: 1000 runAsNonRoot: true runAsUser: 1000 @@ -65,6 +64,11 @@ securityContext: drop: - ALL +# Pod [security context](https://kubernetes.io/docs/tasks/configure-pod-container/security-context/#set-the-security-context-for-a-pod). +# See the [API reference](https://kubernetes.io/docs/reference/kubernetes-api/workload-resources/pod-v1/#security-context) for details. +podSecurityContext: + fsGroup: 1000 + # Container resource [requests and limits](https://kubernetes.io/docs/concepts/configuration/manage-resources-containers/). # See the [API reference](https://kubernetes.io/docs/reference/kubernetes-api/workload-resources/pod-v1/#resources) for details. # @default -- No requests or limits. From 861a011a0f99ee1f1524ae1ec41029bd5097f2cd Mon Sep 17 00:00:00 2001 From: Michael Strassberger Date: Sun, 27 Sep 2026 22:10:02 +0200 Subject: [PATCH 24/24] fix(auth): apply a changed Redis password in place (#194) * fix(auth): apply a changed Redis password in place Changing the password in the auth secret, or adding or removing auth.secretPath, left the RedisFailover wedged: Redis reads requirepass only at startup, so every pod refused the operator's new password, CheckAndHeal failed on its first check, and the rolling update that would have restarted the pods onto the secret was never reached. Each pass also tried to promote a master and failed only on the password. Restarting the pods one at a time can't apply it either: a restarted replica can't authenticate to a master still on the old password. The operator now remembers, per RedisFailover, the password every running Redis last accepted. When the secret differs, it logs in with that one and runs CONFIG SET masterauth and requirepass on each Redis, then gives the Sentinels the new auth-pass. Replication and existing connections stay up, and the rolling update then restarts the pods onto the secret. If the operator restarted in between it has no old password; the RedisFailover reports it and the pods have to be deleted by hand. The readiness probe authenticates with the password its container started with, so after the change it failed on every pod not yet restarted, the master included. A refused password now counts as ready. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01G4mmyo3sqLwf23m5FE2nBE * fix(auth): keep the password out of errors and converge Sentinel The Sentinel auth-pass went through SetCustomSentinelConfig as the config string "auth-pass ", and that function's parse error quotes the string, so the password could reach the reconcile error log (CodeQL go/clear-text-logging). Set it with a dedicated SetSentinelAuthPass instead. Give the Sentinels the password whenever the Redis pods all accept it, not only when the previous password is known. Otherwise an operator restart between changing the Redis pods and the Sentinels left the Sentinels on the old password for good. The step only runs when the password isn't already cached, so this is one SENTINEL SET per operator start. Count a Redis pod that isn't running yet as incomplete, so the password isn't cached while a pod may still start on the old pod template. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01G4mmyo3sqLwf23m5FE2nBE * fix(auth): use the shared client timeouts for the password calls Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01G4mmyo3sqLwf23m5FE2nBE * test(auth): cover the password change's error paths Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01G4mmyo3sqLwf23m5FE2nBE * fix(auth): don't let one pod or Sentinel hold up a password change - Give the Sentinels the new password as soon as every running Redis accepts it, instead of waiting for pods yet to start, and remember it separately so a Pending pod doesn't rewrite their config each sync. - Skip an unreachable Sentinel and retry it, rather than failing the whole reconcile. - With no password remembered after an operator restart, remember the one every running Redis accepts. A Redis without a password is changed without needing the old one. - Point to putting the previous password back, not deleting the pods, when the old password is unknown, and document that the exporter and pre-stop save keep the old password until their pod restarts. - Read the secret once per reconcile. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01G4mmyo3sqLwf23m5FE2nBE --------- Co-authored-by: Claude --- README.md | 6 +- metrics/metrics.go | 1 + .../service/RedisFailoverHeal.go | 48 ++++ mocks/service/redis/Client.go | 28 +++ operator/redisfailover/checker.go | 65 ++++++ operator/redisfailover/checker_test.go | 26 +++ operator/redisfailover/handler.go | 5 + operator/redisfailover/handler_test.go | 24 ++ .../redisfailover/password_internal_test.go | 123 ++++++++++ operator/redisfailover/service/generator.go | 7 + operator/redisfailover/service/heal.go | 2 + operator/redisfailover/service/password.go | 91 ++++++++ .../redisfailover/service/password_test.go | 214 ++++++++++++++++++ service/redis/client.go | 57 +++++ service/redis/client_test.go | 73 ++++++ 15 files changed, 769 insertions(+), 1 deletion(-) create mode 100644 operator/redisfailover/password_internal_test.go create mode 100644 operator/redisfailover/service/password.go create mode 100644 operator/redisfailover/service/password_test.go diff --git a/README.md b/README.md index 7e38dc001..827f1c48d 100644 --- a/README.md +++ b/README.md @@ -385,7 +385,11 @@ spec: ``` You need to set secretPath as the secret name which is created before. -Rotating the password (updating the `password` key of that same Secret in place) is safe: the operator watches a checksum of the current password and rolls the Redis pods, one at a time, whenever it changes. +Rotating the password (updating the `password` key of that same Secret in place), adding `auth.secretPath` or removing it is safe. The operator first switches every running Redis, and the Sentinels, to the new password in place with `CONFIG SET`, which keeps replication up, and then restarts the Redis pods one at a time onto the Secret. New connections need the new password right away. + +Until a pod restarts, whatever reads the password from its environment keeps the old one: the exporter sidecar can't authenticate, the pre-stop `SAVE` fails, and so do custom probes using `$REDIS_PASSWORD`. + +The operator knows the old password only from memory. If it restarted between the change and its next check, it can't switch the pods. The RedisFailover then reports `unable to apply the configured password`. Put the previous password back in the Secret, wait for the RedisFailover to become healthy, then change it again. The password is read from that Secret and passed to `redis-server` via `--requirepass`/`--masterauth` sourced from an environment variable; it is **not** written into the redis ConfigMap, so it never diff --git a/metrics/metrics.go b/metrics/metrics.go index 5fc8c58d2..906d01d13 100644 --- a/metrics/metrics.go +++ b/metrics/metrics.go @@ -71,6 +71,7 @@ const ( GET_REPLICATION_INFO = "GET_REPLICATION_INFO" GET_MEMORY_INFO = "GET_MEMORY_INFO" DISCONNECT_CLIENTS = "DISCONNECT_CLIENTS_ON_DEMOTED_INSTANCE" + SET_PASSWORD = "SET_PASSWORD" ) var ( // used for grabage collection of metrics diff --git a/mocks/operator/redisfailover/service/RedisFailoverHeal.go b/mocks/operator/redisfailover/service/RedisFailoverHeal.go index fec48e058..e34a806f4 100644 --- a/mocks/operator/redisfailover/service/RedisFailoverHeal.go +++ b/mocks/operator/redisfailover/service/RedisFailoverHeal.go @@ -14,6 +14,54 @@ type RedisFailoverHeal struct { mock.Mock } +// ApplyPassword provides a mock function with given fields: rFailover, password, previous +func (_m *RedisFailoverHeal) ApplyPassword(rFailover *v1.RedisFailover, password string, previous string) (bool, error) { + ret := _m.Called(rFailover, password, previous) + + var r0 bool + var r1 error + if rf, ok := ret.Get(0).(func(*v1.RedisFailover, string, string) (bool, error)); ok { + return rf(rFailover, password, previous) + } + if rf, ok := ret.Get(0).(func(*v1.RedisFailover, string, string) bool); ok { + r0 = rf(rFailover, password, previous) + } else { + r0 = ret.Get(0).(bool) + } + + if rf, ok := ret.Get(1).(func(*v1.RedisFailover, string, string) error); ok { + r1 = rf(rFailover, password, previous) + } else { + r1 = ret.Error(1) + } + + return r0, r1 +} + +// ApplySentinelPassword provides a mock function with given fields: rFailover, password +func (_m *RedisFailoverHeal) ApplySentinelPassword(rFailover *v1.RedisFailover, password string) (bool, error) { + ret := _m.Called(rFailover, password) + + var r0 bool + var r1 error + if rf, ok := ret.Get(0).(func(*v1.RedisFailover, string) (bool, error)); ok { + return rf(rFailover, password) + } + if rf, ok := ret.Get(0).(func(*v1.RedisFailover, string) bool); ok { + r0 = rf(rFailover, password) + } else { + r0 = ret.Get(0).(bool) + } + + if rf, ok := ret.Get(1).(func(*v1.RedisFailover, string) error); ok { + r1 = rf(rFailover, password) + } else { + r1 = ret.Error(1) + } + + return r0, r1 +} + // DeletePod provides a mock function with given fields: podName, rFailover func (_m *RedisFailoverHeal) DeletePod(podName string, rFailover *v1.RedisFailover) error { ret := _m.Called(podName, rFailover) diff --git a/mocks/service/redis/Client.go b/mocks/service/redis/Client.go index 7fab24243..3672813fc 100644 --- a/mocks/service/redis/Client.go +++ b/mocks/service/redis/Client.go @@ -306,6 +306,34 @@ func (_m *Client) SetCustomSentinelConfig(ip string, configs []string) error { return r0 } +// SetPassword provides a mock function with given fields: ip, port, password, newPassword +func (_m *Client) SetPassword(ip string, port string, password string, newPassword string) error { + ret := _m.Called(ip, port, password, newPassword) + + var r0 error + if rf, ok := ret.Get(0).(func(string, string, string, string) error); ok { + r0 = rf(ip, port, password, newPassword) + } else { + r0 = ret.Error(0) + } + + return r0 +} + +// SetSentinelAuthPass provides a mock function with given fields: ip, password +func (_m *Client) SetSentinelAuthPass(ip string, password string) error { + ret := _m.Called(ip, password) + + var r0 error + if rf, ok := ret.Get(0).(func(string, string) error); ok { + r0 = rf(ip, password) + } else { + r0 = ret.Error(0) + } + + return r0 +} + // SlaveIsReady provides a mock function with given fields: ip, port, password func (_m *Client) SlaveIsReady(ip string, port string, password string) (bool, error) { ret := _m.Called(ip, port, password) diff --git a/operator/redisfailover/checker.go b/operator/redisfailover/checker.go index 60bac831c..74cdba5e3 100644 --- a/operator/redisfailover/checker.go +++ b/operator/redisfailover/checker.go @@ -202,6 +202,62 @@ func (r *RedisFailoverHandler) masterPodStopping(rf *redisfailoverv1.RedisFailov return false, nil } +// passwordState is the password the Redis pods and the Sentinels were last +// brought onto. +type passwordState struct { + redis string + sentinel string +} + +func passwordKey(rf *redisfailoverv1.RedisFailover) string { + return rf.Namespace + "/" + rf.Name +} + +// applyPassword applies a changed auth secret to the running Redis and the +// Sentinels. Nothing is checked while both are on the secret. +func (r *RedisFailoverHandler) applyPassword(rf *redisfailoverv1.RedisFailover) error { + password, err := k8s.GetRedisPassword(r.k8sservice, rf) + if err != nil { + return err + } + key := passwordKey(rf) + v, known := r.passwords.Load(key) + state, _ := v.(passwordState) + if known && state.redis == password && state.sentinel == password { + return nil + } + + if !known || state.redis != password { + previous := password + if known { + previous = state.redis + } + complete, err := r.rfHealer.ApplyPassword(rf, password, previous) + if err != nil { + return err + } + // A pod yet to start keeps the old password in play. With none known, + // the one every running pod accepts is the best there is. + if complete || !known { + state.redis = password + } + } + + // Every running Redis now accepts the password, so the Sentinels need it + // to reach them. + if state.sentinel != password { + complete, err := r.rfHealer.ApplySentinelPassword(rf, password) + if err != nil { + return err + } + if complete { + state.sentinel = password + } + } + r.passwords.Store(key, state) + return nil +} + // CheckAndHeal runs verifcation checks to ensure the RedisFailover is in an expected and healthy state. // If the checks do not match up to expectations, an attempt will be made to "heal" the RedisFailover into a healthy state. func (r *RedisFailoverHandler) CheckAndHeal(rf *redisfailoverv1.RedisFailover) error { @@ -215,6 +271,15 @@ func (r *RedisFailoverHandler) CheckAndHeal(rf *redisfailoverv1.RedisFailover) e defer updateStatus(r.k8sservice, rf, oldState, oldLastChanged) + // Every check below authenticates, so a changed password goes first. + if err := r.applyPassword(rf); err != nil { + rf.Status = redisfailoverv1.RedisFailoverStatus{ + State: redisfailoverv1.NotHealthyState, + Message: "unable to apply the configured password", + } + return err + } + if rf.Bootstrapping() { return r.checkAndHealBootstrapMode(rf) } diff --git a/operator/redisfailover/checker_test.go b/operator/redisfailover/checker_test.go index e8e5a317f..63f03ffcc 100644 --- a/operator/redisfailover/checker_test.go +++ b/operator/redisfailover/checker_test.go @@ -316,6 +316,8 @@ func TestCheckAndHeal(t *testing.T) { mrfs := &mRFService.RedisFailoverClient{} mrfc := &mRFService.RedisFailoverCheck{} mrfh := &mRFService.RedisFailoverHeal{} + mrfh.On("ApplyPassword", mock.Anything, mock.Anything, mock.Anything).Maybe().Return(true, nil) + mrfh.On("ApplySentinelPassword", mock.Anything, mock.Anything).Maybe().Return(true, nil) // Normal CheckAndHeal gates on a quorum; bootstrap mode still gates on // the full set, so route the mock to whichever the code under test calls. @@ -777,6 +779,8 @@ func TestCheckAndHealOperatorManagedMode(t *testing.T) { mrfs := &mRFService.RedisFailoverClient{} mrfc := &mRFService.RedisFailoverCheck{} mrfh := &mRFService.RedisFailoverHeal{} + mrfh.On("ApplyPassword", mock.Anything, mock.Anything, mock.Anything).Maybe().Return(true, nil) + mrfh.On("ApplySentinelPassword", mock.Anything, mock.Anything).Maybe().Return(true, nil) test.setup(mrfc, mrfh, rf) @@ -843,6 +847,8 @@ func TestUpdateStatusLastChanged(t *testing.T) { mrfs := &mRFService.RedisFailoverClient{} mrfc := &mRFService.RedisFailoverCheck{} mrfh := &mRFService.RedisFailoverHeal{} + mrfh.On("ApplyPassword", mock.Anything, mock.Anything, mock.Anything).Maybe().Return(true, nil) + mrfh.On("ApplySentinelPassword", mock.Anything, mock.Anything).Maybe().Return(true, nil) setupHealthyPass(mrfc, mrfh, rf) @@ -871,6 +877,8 @@ func TestUpdateStatusLastChanged(t *testing.T) { mrfs := &mRFService.RedisFailoverClient{} mrfc := &mRFService.RedisFailoverCheck{} mrfh := &mRFService.RedisFailoverHeal{} + mrfh.On("ApplyPassword", mock.Anything, mock.Anything, mock.Anything).Maybe().Return(true, nil) + mrfh.On("ApplySentinelPassword", mock.Anything, mock.Anything).Maybe().Return(true, nil) setupHealthyPass(mrfc, mrfh, rf) @@ -1275,6 +1283,8 @@ func TestCheckAndHealPlainModeErrorBranches(t *testing.T) { mrfs := &mRFService.RedisFailoverClient{} mrfc := &mRFService.RedisFailoverCheck{} mrfh := &mRFService.RedisFailoverHeal{} + mrfh.On("ApplyPassword", mock.Anything, mock.Anything, mock.Anything).Maybe().Return(true, nil) + mrfh.On("ApplySentinelPassword", mock.Anything, mock.Anything).Maybe().Return(true, nil) test.setup(mrfc, mrfh, rf) @@ -1451,6 +1461,8 @@ func TestCheckAndHealBootstrapModeErrorBranches(t *testing.T) { mrfs := &mRFService.RedisFailoverClient{} mrfc := &mRFService.RedisFailoverCheck{} mrfh := &mRFService.RedisFailoverHeal{} + mrfh.On("ApplyPassword", mock.Anything, mock.Anything, mock.Anything).Maybe().Return(true, nil) + mrfh.On("ApplySentinelPassword", mock.Anything, mock.Anything).Maybe().Return(true, nil) test.setup(mrfc, mrfh, rf) @@ -1992,6 +2004,8 @@ func TestUpdate(t *testing.T) { } } mrfh := &mRFService.RedisFailoverHeal{} + mrfh.On("ApplyPassword", mock.Anything, mock.Anything, mock.Anything).Maybe().Return(true, nil) + mrfh.On("ApplySentinelPassword", mock.Anything, mock.Anything).Maybe().Return(true, nil) if next { replicas := []string{"slave1", "slave2"} @@ -2080,6 +2094,8 @@ func TestUpdateRedisesPodsOperatorManagedModeSkipsSentinelGate(t *testing.T) { mrfs := &mRFService.RedisFailoverClient{} mrfc := &mRFService.RedisFailoverCheck{} mrfh := &mRFService.RedisFailoverHeal{} + mrfh.On("ApplyPassword", mock.Anything, mock.Anything, mock.Anything).Maybe().Return(true, nil) + mrfh.On("ApplySentinelPassword", mock.Anything, mock.Anything).Maybe().Return(true, nil) mrfc.On("GetRedisesIPs", rf).Once().Return([]string{"1.1.1.1"}, nil) mrfc.On("GetMasterIP", rf).Once().Return("1.1.1.1", nil) @@ -2213,6 +2229,8 @@ func TestUpdateRedisesPodsErrorBranches(t *testing.T) { mrfs := &mRFService.RedisFailoverClient{} mrfc := &mRFService.RedisFailoverCheck{} mrfh := &mRFService.RedisFailoverHeal{} + mrfh.On("ApplyPassword", mock.Anything, mock.Anything, mock.Anything).Maybe().Return(true, nil) + mrfh.On("ApplySentinelPassword", mock.Anything, mock.Anything).Maybe().Return(true, nil) test.setup(mrfc, mrfh, rf) @@ -2293,6 +2311,8 @@ func TestUpdateRedisesPodsWaitsForTheLastReplacement(t *testing.T) { rf.Spec.Redis.Replicas = 3 mrfc := &mRFService.RedisFailoverCheck{} mrfh := &mRFService.RedisFailoverHeal{} + mrfh.On("ApplyPassword", mock.Anything, mock.Anything, mock.Anything).Maybe().Return(true, nil) + mrfh.On("ApplySentinelPassword", mock.Anything, mock.Anything).Maybe().Return(true, nil) mk := &mK8SService.Services{} mrfc.On("GetRedisesIPs", rf).Once().Return([]string{"10.0.0.1"}, nil) mrfc.On("GetMasterIP", rf).Once().Return("10.0.0.1", nil) @@ -2359,6 +2379,8 @@ func TestOperatorManagedModeWaitsForAStoppingMasterBeforeElecting(t *testing.T) rf := operatorManagedRF() mrfc := &mRFService.RedisFailoverCheck{} mrfh := &mRFService.RedisFailoverHeal{} + mrfh.On("ApplyPassword", mock.Anything, mock.Anything, mock.Anything).Maybe().Return(true, nil) + mrfh.On("ApplySentinelPassword", mock.Anything, mock.Anything).Maybe().Return(true, nil) mk := &mK8SService.Services{} mk.On("UpdateRedisFailoverStatus", mock.Anything, mock.Anything, mock.Anything, mock.Anything).Return() mrfc.On("IsRedisRunningQuorum", rf).Once().Return(true) @@ -2388,6 +2410,8 @@ func TestUpdateRedisesPodsResizesInPlace(t *testing.T) { rf := generateRF(false, false) mrfc := &mRFService.RedisFailoverCheck{} mrfh := &mRFService.RedisFailoverHeal{} + mrfh.On("ApplyPassword", mock.Anything, mock.Anything, mock.Anything).Maybe().Return(true, nil) + mrfh.On("ApplySentinelPassword", mock.Anything, mock.Anything).Maybe().Return(true, nil) mrfc.On("GetRedisesIPs", rf).Once().Return([]string{"10.0.0.1"}, nil) mrfc.On("GetMasterIP", rf).Once().Return("10.0.0.1", nil) mrfc.On("GetStatefulSetUpdateRevision", rf).Once().Return("new", nil) @@ -2417,6 +2441,8 @@ func TestUpdateRedisesPodsResizeError(t *testing.T) { rf := generateRF(false, false) mrfc := &mRFService.RedisFailoverCheck{} mrfh := &mRFService.RedisFailoverHeal{} + mrfh.On("ApplyPassword", mock.Anything, mock.Anything, mock.Anything).Maybe().Return(true, nil) + mrfh.On("ApplySentinelPassword", mock.Anything, mock.Anything).Maybe().Return(true, nil) mrfc.On("GetRedisesIPs", rf).Once().Return([]string{"10.0.0.1"}, nil) mrfc.On("GetMasterIP", rf).Once().Return("10.0.0.1", nil) mrfc.On("GetStatefulSetUpdateRevision", rf).Once().Return("new", nil) diff --git a/operator/redisfailover/handler.go b/operator/redisfailover/handler.go index f8d65f7e5..6d8591f64 100644 --- a/operator/redisfailover/handler.go +++ b/operator/redisfailover/handler.go @@ -5,6 +5,7 @@ import ( "fmt" "regexp" "slices" + "sync" metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" "k8s.io/apimachinery/pkg/runtime" @@ -52,6 +53,9 @@ type RedisFailoverHandler struct { rfHealer rfservice.RedisFailoverHeal mClient metrics.Recorder logger log.Logger + // passwords holds a passwordState per namespace/name, so a changed secret + // can be applied with the old password. + passwords sync.Map } // NewRedisFailoverHandler returns a new RF handler @@ -84,6 +88,7 @@ func (r *RedisFailoverHandler) Handle(_ context.Context, obj runtime.Object) err return nil } r.mClient.DeleteCluster(rf.Namespace, rf.Name) + r.passwords.Delete(passwordKey(rf)) remaining := slices.DeleteFunc(slices.Clone(rf.Finalizers), func(f string) bool { return f == redisFailoverFinalizer }) diff --git a/operator/redisfailover/handler_test.go b/operator/redisfailover/handler_test.go index 0677667ec..663231002 100644 --- a/operator/redisfailover/handler_test.go +++ b/operator/redisfailover/handler_test.go @@ -96,6 +96,8 @@ func TestHandleSkipReconcileAnnotation(t *testing.T) { mk.On("UpdateRedisFailoverStatus", mock.Anything, mock.Anything, mock.Anything, mock.Anything).Maybe().Return() mrfc := &mRFService.RedisFailoverCheck{} mrfh := &mRFService.RedisFailoverHeal{} + mrfh.On("ApplyPassword", mock.Anything, mock.Anything, mock.Anything).Maybe().Return(true, nil) + mrfh.On("ApplySentinelPassword", mock.Anything, mock.Anything).Maybe().Return(true, nil) mrfs := &mRFService.RedisFailoverClient{} // Finalizer registration runs before the skip-reconcile check, so @@ -150,6 +152,8 @@ func TestHandleNotARedisFailover(t *testing.T) { mk := &mK8SService.Services{} mrfc := &mRFService.RedisFailoverCheck{} mrfh := &mRFService.RedisFailoverHeal{} + mrfh.On("ApplyPassword", mock.Anything, mock.Anything, mock.Anything).Maybe().Return(true, nil) + mrfh.On("ApplySentinelPassword", mock.Anything, mock.Anything).Maybe().Return(true, nil) mrfs := &mRFService.RedisFailoverClient{} handler := rfOperator.NewRedisFailoverHandler(config, mrfs, mrfc, mrfh, mk, metrics.Dummy, log.Dummy) @@ -171,6 +175,8 @@ func TestHandleValidateError(t *testing.T) { mk := &mK8SService.Services{} mrfc := &mRFService.RedisFailoverCheck{} mrfh := &mRFService.RedisFailoverHeal{} + mrfh.On("ApplyPassword", mock.Anything, mock.Anything, mock.Anything).Maybe().Return(true, nil) + mrfh.On("ApplySentinelPassword", mock.Anything, mock.Anything).Maybe().Return(true, nil) mrfs := &mRFService.RedisFailoverClient{} // Finalizer registration runs before Validate(), so it's still expected @@ -198,6 +204,8 @@ func TestHandleEnsureError(t *testing.T) { mk := &mK8SService.Services{} mrfc := &mRFService.RedisFailoverCheck{} mrfh := &mRFService.RedisFailoverHeal{} + mrfh.On("ApplyPassword", mock.Anything, mock.Anything, mock.Anything).Maybe().Return(true, nil) + mrfh.On("ApplySentinelPassword", mock.Anything, mock.Anything).Maybe().Return(true, nil) mrfs := &mRFService.RedisFailoverClient{} // Only the very first Ensure() call is mocked, and it fails - nothing @@ -233,6 +241,8 @@ func TestHandleCheckAndHealError(t *testing.T) { mk.On("UpdateRedisFailoverStatus", mock.Anything, mock.Anything, mock.Anything, mock.Anything).Return() mrfc := &mRFService.RedisFailoverCheck{} mrfh := &mRFService.RedisFailoverHeal{} + mrfh.On("ApplyPassword", mock.Anything, mock.Anything, mock.Anything).Maybe().Return(true, nil) + mrfh.On("ApplySentinelPassword", mock.Anything, mock.Anything).Maybe().Return(true, nil) mrfs := &mRFService.RedisFailoverClient{} mrfs.On("EnsureNotPresentRedisService", rf).Once().Return(nil) @@ -328,6 +338,8 @@ func TestHandleGetLabelsWhitelistFiltering(t *testing.T) { mk.On("UpdateRedisFailoverStatus", mock.Anything, mock.Anything, mock.Anything, mock.Anything).Return() mrfc := &mRFService.RedisFailoverCheck{} mrfh := &mRFService.RedisFailoverHeal{} + mrfh.On("ApplyPassword", mock.Anything, mock.Anything, mock.Anything).Maybe().Return(true, nil) + mrfh.On("ApplySentinelPassword", mock.Anything, mock.Anything).Maybe().Return(true, nil) mrfs := &mRFService.RedisFailoverClient{} var gotLabels map[string]string @@ -384,6 +396,8 @@ func TestHandleAddsFinalizerOnFreshRF(t *testing.T) { mk := &mK8SService.Services{} mrfc := &mRFService.RedisFailoverCheck{} mrfh := &mRFService.RedisFailoverHeal{} + mrfh.On("ApplyPassword", mock.Anything, mock.Anything, mock.Anything).Maybe().Return(true, nil) + mrfh.On("ApplySentinelPassword", mock.Anything, mock.Anything).Maybe().Return(true, nil) mrfs := &mRFService.RedisFailoverClient{} mk.On("PatchRedisFailoverFinalizers", mock.Anything, rf.Namespace, rf.Name, @@ -427,6 +441,8 @@ func TestHandleDeletionCleansUpMetricsAndRemovesFinalizer(t *testing.T) { mk := &mK8SService.Services{} mrfc := &mRFService.RedisFailoverCheck{} mrfh := &mRFService.RedisFailoverHeal{} + mrfh.On("ApplyPassword", mock.Anything, mock.Anything, mock.Anything).Maybe().Return(true, nil) + mrfh.On("ApplySentinelPassword", mock.Anything, mock.Anything).Maybe().Return(true, nil) mrfs := &mRFService.RedisFailoverClient{} mClient := &fakeRecorder{Recorder: metrics.Dummy} @@ -459,6 +475,8 @@ func TestHandleDeletionWithoutFinalizerIsNoop(t *testing.T) { mk := &mK8SService.Services{} mrfc := &mRFService.RedisFailoverCheck{} mrfh := &mRFService.RedisFailoverHeal{} + mrfh.On("ApplyPassword", mock.Anything, mock.Anything, mock.Anything).Maybe().Return(true, nil) + mrfh.On("ApplySentinelPassword", mock.Anything, mock.Anything).Maybe().Return(true, nil) mrfs := &mRFService.RedisFailoverClient{} mClient := &fakeRecorder{Recorder: metrics.Dummy} @@ -485,6 +503,8 @@ func TestHandleFinalizerRegistrationErrorPropagates(t *testing.T) { mk := &mK8SService.Services{} mrfc := &mRFService.RedisFailoverCheck{} mrfh := &mRFService.RedisFailoverHeal{} + mrfh.On("ApplyPassword", mock.Anything, mock.Anything, mock.Anything).Maybe().Return(true, nil) + mrfh.On("ApplySentinelPassword", mock.Anything, mock.Anything).Maybe().Return(true, nil) mrfs := &mRFService.RedisFailoverClient{} mk.On("PatchRedisFailoverFinalizers", mock.Anything, rf.Namespace, rf.Name, @@ -515,6 +535,8 @@ func TestHandleDeletionFinalizerRemovalErrorPropagates(t *testing.T) { mk := &mK8SService.Services{} mrfc := &mRFService.RedisFailoverCheck{} mrfh := &mRFService.RedisFailoverHeal{} + mrfh.On("ApplyPassword", mock.Anything, mock.Anything, mock.Anything).Maybe().Return(true, nil) + mrfh.On("ApplySentinelPassword", mock.Anything, mock.Anything).Maybe().Return(true, nil) mrfs := &mRFService.RedisFailoverClient{} mClient := &fakeRecorder{Recorder: metrics.Dummy} @@ -585,6 +607,8 @@ func TestHandleRecordsClusterMetrics(t *testing.T) { mk.On("UpdateRedisFailoverStatus", mock.Anything, mock.Anything, mock.Anything, mock.Anything).Return() mrfc := &mRFService.RedisFailoverCheck{} mrfh := &mRFService.RedisFailoverHeal{} + mrfh.On("ApplyPassword", mock.Anything, mock.Anything, mock.Anything).Maybe().Return(true, nil) + mrfh.On("ApplySentinelPassword", mock.Anything, mock.Anything).Maybe().Return(true, nil) mrfs := &mRFService.RedisFailoverClient{} test.setup(rf, mrfs, mrfc) diff --git a/operator/redisfailover/password_internal_test.go b/operator/redisfailover/password_internal_test.go new file mode 100644 index 000000000..ac949be43 --- /dev/null +++ b/operator/redisfailover/password_internal_test.go @@ -0,0 +1,123 @@ +package redisfailover + +import ( + "context" + "errors" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/mock" + corev1 "k8s.io/api/core/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + + redisfailoverv1 "github.com/saremox/redis-operator/api/redisfailover/v1" + "github.com/saremox/redis-operator/log" + "github.com/saremox/redis-operator/metrics" + mRFService "github.com/saremox/redis-operator/mocks/operator/redisfailover/service" + mK8SService "github.com/saremox/redis-operator/mocks/service/k8s" +) + +func newPasswordTestHandler(password *string) (*RedisFailoverHandler, *redisfailoverv1.RedisFailover, *mRFService.RedisFailoverHeal) { + rf := &redisfailoverv1.RedisFailover{ + ObjectMeta: metav1.ObjectMeta{Name: "test", Namespace: "testns"}, + Spec: redisfailoverv1.RedisFailoverSpec{Auth: redisfailoverv1.AuthSettings{SecretPath: "redis-auth"}}, + } + ms := &mK8SService.Services{} + ms.On("GetSecret", "testns", "redis-auth").Return(func(string, string) (*corev1.Secret, error) { + return &corev1.Secret{Data: map[string][]byte{"password": []byte(*password)}}, nil + }) + ms.On("UpdateRedisFailoverStatus", mock.Anything, mock.Anything, mock.Anything, mock.Anything).Return() + mrfh := &mRFService.RedisFailoverHeal{} + handler := NewRedisFailoverHandler(Config{}, &mRFService.RedisFailoverClient{}, &mRFService.RedisFailoverCheck{}, mrfh, ms, metrics.Dummy, log.Dummy) + return handler, rf, mrfh +} + +func TestApplyPasswordRemembersAcceptedPassword(t *testing.T) { + password := "v1" + handler, rf, mrfh := newPasswordTestHandler(&password) + + // With none known, the password every running pod accepts is remembered, + // and the Sentinels are retried until every one has it. + mrfh.On("ApplyPassword", rf, "v1", "v1").Once().Return(false, nil) + mrfh.On("ApplySentinelPassword", rf, "v1").Once().Return(false, nil) + assert.NoError(t, handler.applyPassword(rf)) + mrfh.On("ApplySentinelPassword", rf, "v1").Once().Return(true, nil) + assert.NoError(t, handler.applyPassword(rf)) + + // Nothing to check while the secret is unchanged. + assert.NoError(t, handler.applyPassword(rf)) + + // A changed secret is applied with the password the pods accepted. The + // Sentinels get it as soon as the running pods do, while a pod yet to + // start keeps the old password in use for the pods. + password = "v2" + mrfh.On("ApplyPassword", rf, "v2", "v1").Twice().Return(false, nil) + mrfh.On("ApplySentinelPassword", rf, "v2").Once().Return(true, nil) + assert.NoError(t, handler.applyPassword(rf)) + assert.NoError(t, handler.applyPassword(rf)) + mrfh.On("ApplyPassword", rf, "v2", "v1").Once().Return(true, nil) + assert.NoError(t, handler.applyPassword(rf)) + assert.NoError(t, handler.applyPassword(rf)) + + mrfh.AssertExpectations(t) + mrfh.AssertNumberOfCalls(t, "ApplyPassword", 4) + mrfh.AssertNumberOfCalls(t, "ApplySentinelPassword", 3) +} + +func TestApplyPasswordForgetsDeletedRedisFailover(t *testing.T) { + password := "v1" + handler, rf, _ := newPasswordTestHandler(&password) + handler.passwords.Store(passwordKey(rf), passwordState{redis: "v1", sentinel: "v1"}) + + now := metav1.Now() + rf.DeletionTimestamp = &now + rf.Finalizers = []string{redisFailoverFinalizer} + ms := handler.k8sservice.(*mK8SService.Services) + ms.On("PatchRedisFailoverFinalizers", mock.Anything, "testns", "test", mock.Anything, mock.Anything).Return(nil) + assert.NoError(t, handler.Handle(context.Background(), rf)) + + _, cached := handler.passwords.Load(passwordKey(rf)) + assert.False(t, cached) +} + +func TestCheckAndHealReportsAnUnappliedPassword(t *testing.T) { + boom := errors.New("boom") + tests := []struct { + name string + applyErr error + sentinelErr error + }{ + {name: "failed apply", applyErr: boom}, + {name: "failed sentinel apply", sentinelErr: boom}, + } + + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + password := "v1" + handler, rf, mrfh := newPasswordTestHandler(&password) + mrfh.On("ApplyPassword", rf, "v1", "v1").Return(true, test.applyErr) + mrfh.On("ApplySentinelPassword", rf, "v1").Return(false, test.sentinelErr) + + assert.ErrorIs(t, handler.CheckAndHeal(rf), boom) + assert.Equal(t, redisfailoverv1.NotHealthyState, rf.Status.State) + assert.Equal(t, "unable to apply the configured password", rf.Status.Message) + _, cached := handler.passwords.Load(passwordKey(rf)) + assert.False(t, cached) + }) + } +} + +func TestCheckAndHealReportsAnUnreadableSecret(t *testing.T) { + rf := &redisfailoverv1.RedisFailover{ + ObjectMeta: metav1.ObjectMeta{Name: "test", Namespace: "testns"}, + Spec: redisfailoverv1.RedisFailoverSpec{Auth: redisfailoverv1.AuthSettings{SecretPath: "redis-auth"}}, + } + boom := errors.New("boom") + ms := &mK8SService.Services{} + ms.On("GetSecret", "testns", "redis-auth").Return(nil, boom) + ms.On("UpdateRedisFailoverStatus", mock.Anything, mock.Anything, mock.Anything, mock.Anything).Return() + handler := NewRedisFailoverHandler(Config{}, &mRFService.RedisFailoverClient{}, &mRFService.RedisFailoverCheck{}, &mRFService.RedisFailoverHeal{}, ms, metrics.Dummy, log.Dummy) + + assert.ErrorIs(t, handler.CheckAndHeal(rf), boom) + assert.Equal(t, "unable to apply the configured password", rf.Status.Message) +} diff --git a/operator/redisfailover/service/generator.go b/operator/redisfailover/service/generator.go index 835e79be4..66fdb4025 100644 --- a/operator/redisfailover/service/generator.go +++ b/operator/redisfailover/service/generator.go @@ -334,6 +334,13 @@ fi cmd="${cmd} info replication" +# The operator changes the password of a running Redis in place, and the pod +# restarts onto it later. Until then a refused password says nothing about +# replication. +if echo "${cmd}" | xargs -0 sh -c 2>&1 | grep -qE "NOAUTH|WRONGPASS"; then + exit 0 +fi + check_master(){ exit 0 } diff --git a/operator/redisfailover/service/heal.go b/operator/redisfailover/service/heal.go index ad4a51239..6aaab59f2 100644 --- a/operator/redisfailover/service/heal.go +++ b/operator/redisfailover/service/heal.go @@ -34,6 +34,8 @@ type RedisFailoverHeal interface { PromoteBestReplica(newMasterIP string, rFailover *redisfailoverv1.RedisFailover) error EnsureRedisMaxMemory(rFailover *redisfailoverv1.RedisFailover, master string, redises []string) (MaxMemoryResult, error) ResizePodInPlace(rFailover *redisfailoverv1.RedisFailover, podName, updateRevision string) (ResizeResult, error) + ApplyPassword(rFailover *redisfailoverv1.RedisFailover, password, previous string) (bool, error) + ApplySentinelPassword(rFailover *redisfailoverv1.RedisFailover, password string) (bool, error) } // RedisFailoverHealer is our implementation of RedisFailoverCheck interface diff --git a/operator/redisfailover/service/password.go b/operator/redisfailover/service/password.go new file mode 100644 index 000000000..266ad1409 --- /dev/null +++ b/operator/redisfailover/service/password.go @@ -0,0 +1,91 @@ +package service + +import ( + "errors" + "fmt" + + redisfailoverv1 "github.com/saremox/redis-operator/api/redisfailover/v1" + "github.com/saremox/redis-operator/service/redis" + v1 "k8s.io/api/core/v1" +) + +// ApplyPassword brings every Redis onto password. Redis reads requirepass only +// at startup, and restarting the pods one at a time can't apply a new one: a +// restarted replica can't authenticate to a master still on the old password. +// So a Redis still on previous is changed in place, and the rolling update then +// restarts the pods onto the secret. +// +// It returns an error if a running Redis refuses password and can't be +// changed, and true once every Redis pod runs and accepts it. +func (r *RedisFailoverHealer) ApplyPassword(rf *redisfailoverv1.RedisFailover, password, previous string) (bool, error) { + rps, err := r.k8sService.GetStatefulSetPods(rf.Namespace, GetRedisName(rf)) + if err != nil { + return false, err + } + + port := getRedisPort(rf.Spec.Redis.Port) + complete := true + var errs []error + for _, rp := range rps.Items { + if rp.DeletionTimestamp != nil { + continue + } + // A pod yet to start may still come up on the old pod template. + if rp.Status.Phase != v1.PodRunning { + complete = false + continue + } + _, err := r.redisClient.IsMaster(rp.Status.PodIP, port, password) + if err == nil { + continue + } + if !redis.IsAuthError(err) { + complete = false + continue + } + current := previous + if redis.IsNoPasswordError(err) { + current = "" + } + if current == password { + errs = append(errs, fmt.Errorf("redis pod %s refuses the configured password and the operator doesn't know the one it runs with; put the previous password back in the secret until the RedisFailover is healthy, then change it again", rp.Name)) + continue + } + if err := r.redisClient.SetPassword(rp.Status.PodIP, port, current, password); err != nil { + errs = append(errs, fmt.Errorf("changing the password of redis pod %s: %w", rp.Name, err)) + continue + } + r.logger.WithField("redisfailover", rf.Name).WithField("namespace", rf.Namespace).Infof("Changed the password of redis pod %s", rp.Name) + } + if err := errors.Join(errs...); err != nil { + return false, err + } + return complete, nil +} + +// ApplySentinelPassword gives the Sentinels the password to authenticate to +// Redis with. It returns true once every running Sentinel has it. +func (r *RedisFailoverHealer) ApplySentinelPassword(rf *redisfailoverv1.RedisFailover, password string) (bool, error) { + if !rf.SentinelsAllowed() { + return true, nil + } + sps, err := r.k8sService.GetDeploymentPods(rf.Namespace, GetSentinelName(rf)) + if err != nil { + return false, err + } + complete := true + for _, sp := range sps.Items { + if sp.Status.Phase != v1.PodRunning || sp.DeletionTimestamp != nil { + continue + } + if err := r.redisClient.SetSentinelAuthPass(sp.Status.PodIP, password); err != nil { + if redis.IsUnreachableError(err) { + r.logger.WithField("redisfailover", rf.Name).WithField("namespace", rf.Namespace).Warningf("Sentinel pod %s is unreachable, its password is changed later: %v", sp.Name, err) + complete = false + continue + } + return false, fmt.Errorf("changing the password of sentinel pod %s: %w", sp.Name, err) + } + } + return complete, nil +} diff --git a/operator/redisfailover/service/password_test.go b/operator/redisfailover/service/password_test.go new file mode 100644 index 000000000..133568233 --- /dev/null +++ b/operator/redisfailover/service/password_test.go @@ -0,0 +1,214 @@ +package service_test + +import ( + "errors" + "testing" + + "github.com/stretchr/testify/assert" + corev1 "k8s.io/api/core/v1" + metav1 "k8s.io/apimachinery/pkg/apis/meta/v1" + "k8s.io/utils/ptr" + + "github.com/saremox/redis-operator/log" + mK8SService "github.com/saremox/redis-operator/mocks/service/k8s" + mRedisService "github.com/saremox/redis-operator/mocks/service/redis" + rfservice "github.com/saremox/redis-operator/operator/redisfailover/service" +) + +func runningPod(name, ip string) corev1.Pod { + return corev1.Pod{ + ObjectMeta: metav1.ObjectMeta{Name: name}, + Status: corev1.PodStatus{PodIP: ip, Phase: corev1.PodRunning}, + } +} + +func TestApplyPassword(t *testing.T) { + wrongpass := errors.New("WRONGPASS invalid username-password pair or user is disabled.") + nopass := errors.New("ERR AUTH called without any password configured for the default user. Are you sure your configuration is correct?") + redises := &corev1.PodList{Items: []corev1.Pod{runningPod("rfr-0", "10.0.0.1"), runningPod("rfr-1", "10.0.0.2")}} + + tests := []struct { + name string + previous string + pending bool + deleting bool + // errors IsMaster returns with the new password, per pod IP + refuse map[string]error + setErr error + // the password SetPassword logs in with, per pod IP + expSet map[string]string + expComplete bool + expErr string + }{ + { + name: "every pod accepts the password", + previous: "new", + expComplete: true, + }, + { + name: "a pod on the previous password is changed in place", + previous: "old", + refuse: map[string]error{"10.0.0.2": wrongpass}, + expSet: map[string]string{"10.0.0.2": "old"}, + expComplete: true, + }, + { + name: "a pod without a password is changed without knowing the previous one", + previous: "new", + refuse: map[string]error{"10.0.0.1": nopass}, + expSet: map[string]string{"10.0.0.1": ""}, + expComplete: true, + }, + { + name: "a pod yet to start leaves it incomplete", + previous: "new", + pending: true, + expComplete: false, + }, + { + name: "a pod being deleted is skipped", + previous: "new", + deleting: true, + expComplete: true, + }, + { + name: "a refused password with no previous one to use", + previous: "new", + refuse: map[string]error{"10.0.0.1": wrongpass}, + expErr: "put the previous password back in the secret", + }, + { + name: "a failed change is returned", + previous: "old", + refuse: map[string]error{"10.0.0.2": wrongpass}, + setErr: errors.New("i/o timeout"), + expSet: map[string]string{"10.0.0.2": "old"}, + expErr: "changing the password of redis pod rfr-1", + }, + { + name: "an unreachable pod leaves it incomplete", + previous: "old", + refuse: map[string]error{"10.0.0.1": errors.New("i/o timeout")}, + expComplete: false, + }, + } + + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + rf := generateRF() + + ms := &mK8SService.Services{} + pods := redises.DeepCopy() + if test.pending { + pods.Items = append(pods.Items, corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: "rfr-2"}, Status: corev1.PodStatus{Phase: corev1.PodPending}}) + } + if test.deleting { + deleting := runningPod("rfr-2", "10.0.0.3") + deleting.DeletionTimestamp = &metav1.Time{} + pods.Items = append(pods.Items, deleting) + } + ms.On("GetStatefulSetPods", namespace, rfservice.GetRedisName(rf)).Return(pods, nil) + mr := &mRedisService.Client{} + for _, p := range redises.Items { + mr.On("IsMaster", p.Status.PodIP, "0", "new").Return(false, test.refuse[p.Status.PodIP]) + } + for ip, current := range test.expSet { + mr.On("SetPassword", ip, "0", current, "new").Once().Return(test.setErr) + } + + healer := rfservice.NewRedisFailoverHealer(ms, mr, log.DummyLogger{}) + complete, err := healer.ApplyPassword(rf, "new", test.previous) + if test.expErr != "" { + assert.ErrorContains(t, err, test.expErr) + } else { + assert.NoError(t, err) + } + assert.Equal(t, test.expComplete, complete) + mr.AssertExpectations(t) + mr.AssertNumberOfCalls(t, "SetPassword", len(test.expSet)) + }) + } +} + +func TestApplyPasswordListError(t *testing.T) { + boom := errors.New("boom") + rf := generateRF() + ms := &mK8SService.Services{} + ms.On("GetStatefulSetPods", namespace, rfservice.GetRedisName(rf)).Return(nil, boom) + + healer := rfservice.NewRedisFailoverHealer(ms, &mRedisService.Client{}, log.DummyLogger{}) + complete, err := healer.ApplyPassword(rf, "new", "old") + assert.ErrorIs(t, err, boom) + assert.False(t, complete) +} + +func TestApplySentinelPassword(t *testing.T) { + stopped := corev1.Pod{ObjectMeta: metav1.ObjectMeta{Name: "rfs-2"}, Status: corev1.PodStatus{PodIP: "10.0.1.3", Phase: corev1.PodFailed}} + deleting := runningPod("rfs-3", "10.0.1.4") + deleting.DeletionTimestamp = &metav1.Time{} + sentinels := &corev1.PodList{Items: []corev1.Pod{runningPod("rfs-0", "10.0.1.1"), runningPod("rfs-1", "10.0.1.2"), stopped, deleting}} + + tests := []struct { + name string + sentinel bool + listErr error + setErr error + expCalls int + expComplete bool + expErr string + }{ + { + name: "without sentinels there is nothing to do", + expComplete: true, + }, + { + name: "every running sentinel gets the password", + sentinel: true, + expCalls: 2, + expComplete: true, + }, + { + name: "an unreachable sentinel leaves it incomplete", + sentinel: true, + setErr: errors.New("dial tcp 10.0.1.1:26379: i/o timeout"), + expCalls: 2, + expComplete: false, + }, + { + name: "a sentinel refusing the change is returned", + sentinel: true, + setErr: errors.New("ERR No such master with that name"), + expCalls: 1, + expErr: "changing the password of sentinel pod rfs-0", + }, + { + name: "a failed pod list is returned", + sentinel: true, + listErr: errors.New("boom"), + expErr: "boom", + }, + } + + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + rf := generateRF() + rf.Spec.Sentinel.Enabled = ptr.To(test.sentinel) + + ms := &mK8SService.Services{} + ms.On("GetDeploymentPods", namespace, rfservice.GetSentinelName(rf)).Return(sentinels, test.listErr) + mr := &mRedisService.Client{} + mr.On("SetSentinelAuthPass", "10.0.1.1", "new").Return(test.setErr) + mr.On("SetSentinelAuthPass", "10.0.1.2", "new").Return(nil) + + healer := rfservice.NewRedisFailoverHealer(ms, mr, log.DummyLogger{}) + complete, err := healer.ApplySentinelPassword(rf, "new") + if test.expErr != "" { + assert.ErrorContains(t, err, test.expErr) + } else { + assert.NoError(t, err) + } + assert.Equal(t, test.expComplete, complete) + mr.AssertNumberOfCalls(t, "SetSentinelAuthPass", test.expCalls) + }) + } +} diff --git a/service/redis/client.go b/service/redis/client.go index 764a44ba7..16ac7a174 100644 --- a/service/redis/client.go +++ b/service/redis/client.go @@ -56,6 +56,8 @@ type Client interface { SentinelCheckQuorum(ip string) error GetReplicationInfo(ip, port, password string) (*ReplicationInfo, error) GetMemoryInfo(ip, port, password string) (*MemoryInfo, error) + SetPassword(ip, port, password, newPassword string) error + SetSentinelAuthPass(ip, password string) error } type client struct { @@ -702,6 +704,43 @@ func (c *client) GetReplicationInfo(ip, port, password string) (*ReplicationInfo return replInfo, nil } +// SetPassword changes the password a running Redis requires and the one it +// uses to authenticate to its master. Connections already authenticated, +// including replication links, stay up. +func (c *client) SetPassword(ip, port, password, newPassword string) error { + rClient := rediscli.NewClient(redisOptions(net.JoinHostPort(ip, port), password)) + defer func(rClient *rediscli.Client) { + if err := rClient.Close(); err != nil { + log.Error(err.Error()) + } + }(rClient) + for _, param := range []string{"masterauth", "requirepass"} { + if err := rClient.ConfigSet(context.TODO(), param, newPassword).Err(); err != nil { + c.metricsRecorder.RecordRedisOperation(metrics.KIND_REDIS, ip, metrics.SET_PASSWORD, metrics.FAIL, getRedisError(err)) + return err + } + } + c.metricsRecorder.RecordRedisOperation(metrics.KIND_REDIS, ip, metrics.SET_PASSWORD, metrics.SUCCESS, metrics.NOT_APPLICABLE) + return nil +} + +// SetSentinelAuthPass sets the password a Sentinel uses to authenticate to the +// Redis it monitors. +func (c *client) SetSentinelAuthPass(ip, password string) error { + rClient := rediscli.NewClient(redisOptions(net.JoinHostPort(ip, sentinelPort), "")) + defer func(rClient *rediscli.Client) { + if err := rClient.Close(); err != nil { + log.Error(err.Error()) + } + }(rClient) + if err := rClient.Do(context.TODO(), "SENTINEL", "SET", masterName, "auth-pass", password).Err(); err != nil { + c.metricsRecorder.RecordRedisOperation(metrics.KIND_SENTINEL, ip, metrics.SET_PASSWORD, metrics.FAIL, getRedisError(err)) + return err + } + c.metricsRecorder.RecordRedisOperation(metrics.KIND_SENTINEL, ip, metrics.SET_PASSWORD, metrics.SUCCESS, metrics.NOT_APPLICABLE) + return nil +} + // GetMemoryInfo returns the maxmemory settings, memory usage and role of a Redis instance. func (c *client) GetMemoryInfo(ip, port, password string) (*MemoryInfo, error) { rClient := rediscli.NewClient(redisOptions(net.JoinHostPort(ip, port), password)) @@ -763,6 +802,24 @@ func getRedisError(err error) string { } } +// IsAuthError reports whether Redis refused the password it was given, or +// was given one while it has none configured. +func IsAuthError(err error) bool { + if err == nil { + return false + } + msg := err.Error() + return strings.Contains(msg, "WRONGPASS") || + strings.Contains(msg, "NOAUTH") || + IsNoPasswordError(err) +} + +// IsNoPasswordError reports whether Redis was given a password while it has +// none configured. +func IsNoPasswordError(err error) bool { + return err != nil && strings.Contains(err.Error(), "without any password configured") +} + // IsUnreachableError reports whether err means the redis node could not be // reached (dial/timeout/reset), as opposed to the node being reached and // rejecting the command. Callers use it to skip a down node instead of aborting diff --git a/service/redis/client_test.go b/service/redis/client_test.go index 3e5d2ef95..5f5cd0bba 100644 --- a/service/redis/client_test.go +++ b/service/redis/client_test.go @@ -245,6 +245,14 @@ func TestSlaveIsReady_ConnectionError(t *testing.T) { assert.Error(t, err) } +func TestSetPassword_ConnectionError(t *testing.T) { + port, err := findFreePort() + require.NoError(t, err) + c := newTestClient() + + assert.Error(t, c.SetPassword(testLoopbackIP, strconv.Itoa(port), "", "p1")) +} + func TestGetReplicationInfo_ConnectionError(t *testing.T) { port, err := findFreePort() require.NoError(t, err) @@ -1076,6 +1084,13 @@ func TestSentinelCheckQuorum_NoQuorum(t *testing.T) { assert.Equal(t, "quorum Not available", err.Error(), "the intended NOQUORUM message should be reachable, not just the raw driver error") } +func TestSetSentinelAuthPass(t *testing.T) { + env := getSharedEnv(t) + c := newTestClient() + require.NoError(t, c.SetSentinelAuthPass(env.sentinel.IP, "s3cr3t")) + require.NoError(t, c.SetSentinelAuthPass(env.sentinel.IP, "")) +} + // TestSentinelFunctions_SentinelUnreachable exercises the connection-error // branch of the various Sentinel-facing Client methods (the `if err != nil` // branch immediately following the Info()/Process() call to the sentinel, @@ -1126,6 +1141,9 @@ func TestSentinelFunctions_SentinelUnreachable(t *testing.T) { err = c.MonitorRedisWithPort(env.sentinel.IP, testLoopbackIP, redisPort, "1", "") assert.Error(t, err, "MonitorRedisWithPort should fail once nothing is listening on the sentinel port") + + err = c.SetSentinelAuthPass(env.sentinel.IP, "s3cr3t") + assert.Error(t, err, "SetSentinelAuthPass should fail once nothing is listening on the sentinel port") } // --------------------------------------------------------------------- @@ -1198,3 +1216,58 @@ func TestIsUnreachableError(t *testing.T) { }) } } + +func TestIsAuthError(t *testing.T) { + tests := []struct { + name string + err error + expected bool + }{ + {name: "nil", err: nil, expected: false}, + {name: "wrong password", err: errors.New("WRONGPASS invalid username-password pair or user is disabled."), expected: true}, + {name: "no password given", err: errors.New("NOAUTH Authentication required."), expected: true}, + {name: "password given to a redis without one", err: errors.New("ERR AUTH called without any password configured for the default user. Are you sure your configuration is correct?"), expected: true}, + {name: "unreachable", err: errors.New("dial tcp 10.0.0.1:6379: i/o timeout"), expected: false}, + } + + for _, test := range tests { + t.Run(test.name, func(t *testing.T) { + assert.Equal(t, test.expected, IsAuthError(test.err)) + }) + } +} + +func TestIsNoPasswordError(t *testing.T) { + assert.False(t, IsNoPasswordError(nil)) + assert.False(t, IsNoPasswordError(errors.New("WRONGPASS invalid username-password pair or user is disabled."))) + assert.True(t, IsNoPasswordError(errors.New("ERR AUTH called without any password configured for the default user. Are you sure your configuration is correct?"))) +} + +// Adding, changing and removing a password in place keeps replication up. +func TestSetPassword(t *testing.T) { + requireRedisServer(t) + master := startRedisProcess(t) + replica := startReplicaOf(t, master) + c := newTestClient() + mport, rport := strconv.Itoa(master.Port), strconv.Itoa(replica.Port) + + linkUp := func(password string) bool { + return waitForCondition(t, 15*time.Second, func() bool { + info, err := c.GetReplicationInfo(replica.IP, rport, password) + return err == nil && info.MasterLinkStatus == "up" + }) + } + require.True(t, linkUp("")) + + for _, step := range []struct{ from, to string }{{"", "p1"}, {"p1", "p2"}, {"p2", ""}} { + require.NoError(t, c.SetPassword(replica.IP, rport, step.from, step.to)) + require.NoError(t, c.SetPassword(master.IP, mport, step.from, step.to)) + + _, err := c.IsMaster(master.IP, mport, step.from) + assert.True(t, IsAuthError(err), "old password %q: %v", step.from, err) + isMaster, err := c.IsMaster(master.IP, mport, step.to) + require.NoError(t, err) + assert.True(t, isMaster) + assert.True(t, linkUp(step.to), "replication after %q -> %q", step.from, step.to) + } +}