Files
my-rook-config/pkg/operator/ceph/cluster/predicate.go
T
Anas Khan 47b4167b70 core: reconcile when node OSD topology labels change
shouldReconcileChangedNode only diffed the node spec, so changes to node
labels, including rook's own topology.rook.io/* labels, were never detected.
Relabeling a node therefore did not trigger a CephCluster reconcile until some
other condition eventually did.

Trigger a reconcile from the node update predicate when one of the OSD topology
labels rook cares about changes. The whitelist reuses the canonical label set
from topology.GetDefaultTopologyLabels() (the hostname, the Kubernetes
region/zone labels, and the topology.rook.io/* labels), so a reconcile is
scoped to topology changes instead of firing on any label or annotation change.
The UpdateFunc returns true directly for these changes because onK8sNode()
returns false for a node that is already an OSD host, which would otherwise
swallow the relabel.

Resolves #17652

Signed-off-by: Anas Khan <anas@anaskhan.me>
2026-07-07 07:38:15 +05:30

319 lines
12 KiB
Go

/*
Copyright 2020 The Rook Authors. All rights reserved.
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
*/
// Package cluster to manage a Ceph cluster.
package cluster
import (
"context"
"strings"
"github.com/google/go-cmp/cmp"
"github.com/google/go-cmp/cmp/cmpopts"
cephv1 "github.com/rook/rook/pkg/apis/ceph.rook.io/v1"
"github.com/rook/rook/pkg/clusterd"
discoverDaemon "github.com/rook/rook/pkg/daemon/discover"
"github.com/rook/rook/pkg/operator/ceph/cluster/osd/topology"
opcontroller "github.com/rook/rook/pkg/operator/ceph/controller"
"github.com/rook/rook/pkg/operator/k8sutil"
"github.com/rook/rook/pkg/util/log"
corev1 "k8s.io/api/core/v1"
"k8s.io/apimachinery/pkg/api/resource"
v1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"sigs.k8s.io/controller-runtime/pkg/client"
"sigs.k8s.io/controller-runtime/pkg/event"
"sigs.k8s.io/controller-runtime/pkg/predicate"
)
// Always trigger a reconcile when a secret is deleted. This will cause a
// reconciliation failure to happen immediately in hopes of alerting the end
// user to the configuration problem.
var changedOrDeleted = predicate.Or(
predicate.TypedResourceVersionChangedPredicate[*corev1.Secret]{},
predicate.TypedFuncs[*corev1.Secret]{
DeleteFunc: func(e event.TypedDeleteEvent[*corev1.Secret]) bool {
return true
},
},
)
func shouldReconcileChangedNode(objOld, objNew *corev1.Node) bool {
// do not watch node if only resourceversion got changed
resourceQtyComparer := cmpopts.IgnoreFields(v1.ObjectMeta{}, "ResourceVersion")
diff := cmp.Diff(objOld.Spec, objNew.Spec, resourceQtyComparer)
// do not watch node if only LastHeartbeatTime got changed
resourceQtyComparer = cmpopts.IgnoreFields(corev1.NodeCondition{}, "LastHeartbeatTime")
diff1 := cmp.Diff(objOld.Spec, objNew.Spec, resourceQtyComparer)
if diff == "" && diff1 == "" {
return false
}
return true
}
// topologyLabels is the set of node labels Rook uses to build the OSD CRUSH map
// topology (the hostname, the Kubernetes region/zone labels, and the
// topology.rook.io/* labels). Only changes to one of these labels are relevant
// to Rook; other label or annotation changes on a node must not trigger a
// reconcile.
var topologyLabels = strings.Split(topology.GetDefaultTopologyLabels(), ",")
// nodeTopologyLabelsChanged returns true if the value of any of the OSD topology
// labels Rook cares about differs between the old and new node. This detects a
// topology label being added, removed, or changed.
func nodeTopologyLabelsChanged(objOld, objNew *corev1.Node) bool {
oldLabels := objOld.GetLabels()
newLabels := objNew.GetLabels()
for _, label := range topologyLabels {
if oldLabels[label] != newLabels[label] {
return true
}
}
return false
}
// predicateForNodeWatcher is the predicate function to trigger reconcile on Node events
func predicateForNodeWatcher[T *corev1.Node](ctx context.Context, client client.Client, context *clusterd.Context, opNamespace string) predicate.TypedFuncs[T] {
return predicate.TypedFuncs[T]{
CreateFunc: func(e event.TypedCreateEvent[T]) bool {
obj := (*corev1.Node)(e.Object)
if e.IsInInitialList {
// Skip node handling while the controller runtime is initializing the cache.
return false
}
clientCluster := newClientCluster(client, obj.GetNamespace(), context)
return clientCluster.onK8sNode(ctx, obj, opNamespace)
},
UpdateFunc: func(e event.TypedUpdateEvent[T]) bool {
objOld := (*corev1.Node)(e.ObjectOld)
objNew := (*corev1.Node)(e.ObjectNew)
// When one of the OSD topology labels Rook cares about (for example
// the topology.rook.io/* labels) changes, trigger a reconcile
// directly. Otherwise onK8sNode() would return false for a node that
// is already an OSD host and the relabel would never be propagated to
// the CRUSH map.
if nodeTopologyLabelsChanged(objOld, objNew) {
return true
}
if !shouldReconcileChangedNode(objOld, objNew) {
return false
}
clientCluster := newClientCluster(client, objNew.GetNamespace(), context)
return clientCluster.onK8sNode(ctx, objNew, opNamespace)
},
DeleteFunc: func(e event.TypedDeleteEvent[T]) bool {
return false
},
GenericFunc: func(e event.TypedGenericEvent[T]) bool {
return false
},
}
}
// predicateForHotPlugCMWatcher is the predicate function to trigger reconcile on ConfigMap events (hot-plug)
func predicateForHotPlugCMWatcher[T *corev1.ConfigMap](client client.Client) predicate.TypedFuncs[T] {
return predicate.TypedFuncs[T]{
UpdateFunc: func(e event.TypedUpdateEvent[T]) bool {
objOld := (*corev1.ConfigMap)(e.ObjectOld)
objNew := (*corev1.ConfigMap)(e.ObjectNew)
if !isHotPlugCM(objNew) {
return false
}
clientCluster := newClientCluster(client, objNew.GetNamespace(), &clusterd.Context{})
return clientCluster.onDeviceCMUpdate(objOld, objNew)
},
DeleteFunc: func(e event.TypedDeleteEvent[T]) bool {
// TODO: if the configmap goes away we could retrigger rook-discover DS
// However at this point the returned bool can only trigger a reconcile of the CephCluster object
// Definitely non-trivial but nice to have in the future
return false
},
CreateFunc: func(e event.TypedCreateEvent[T]) bool {
return false
},
GenericFunc: func(e event.TypedGenericEvent[T]) bool {
return false
},
}
}
// isHotPlugCM informs whether the object is the cm for hot-plug disk
func isHotPlugCM(cm *corev1.ConfigMap) bool {
// Get the labels
labels := cm.GetLabels()
labelVal, labelKeyExist := labels[k8sutil.AppAttr]
if labelKeyExist && labelVal == discoverDaemon.AppName {
return true
}
return false
}
// predicateForClusterConfigMapWatcher is the predicate function to trigger reconcile on operator config ConfigMap events
func predicateForClusterConfigMapWatcher[T *corev1.ConfigMap](ctx context.Context, client client.Client) predicate.TypedFuncs[T] {
return predicate.TypedFuncs[T]{
UpdateFunc: func(e event.TypedUpdateEvent[T]) bool {
objNew := (*corev1.ConfigMap)(e.ObjectNew)
objOld := (*corev1.ConfigMap)(e.ObjectOld)
if objNew.Name == opcontroller.OperatorSettingConfigMapName {
if objOld.Data["ROOK_USE_CSI_OPERATOR"] != objNew.Data["ROOK_USE_CSI_OPERATOR"] {
log.NamespacedInfo(objNew.Namespace, logger, "ROOK_USE_CSI_OPERATOR changed from %q to %q", objOld.Data["ROOK_USE_CSI_OPERATOR"], objNew.Data["ROOK_USE_CSI_OPERATOR"])
return true
}
}
if isFloatingMonConfigMap(ctx, client, objNew) {
log.NamespacedInfo(objNew.Namespace, logger, "floating mon ConfigMap %q changed", objNew.Name)
return true
}
return false
},
CreateFunc: func(e event.TypedCreateEvent[T]) bool {
objNew := (*corev1.ConfigMap)(e.Object)
return objNew.Name == opcontroller.OperatorSettingConfigMapName
},
DeleteFunc: func(e event.TypedDeleteEvent[T]) bool {
return false
},
GenericFunc: func(e event.TypedGenericEvent[T]) bool {
return false
},
}
}
// isFloatingMonConfigMap checks if any floating mon configuration refers to the updated ConfigMap
func isFloatingMonConfigMap(ctx context.Context, client client.Client, cm *corev1.ConfigMap) bool {
clusterList := cephv1.CephClusterList{}
err := client.List(ctx, &clusterList)
if err != nil {
return false
}
for _, cluster := range clusterList.Items {
if cluster.Spec.Mon.FloatingMon.ConfigMapName != "" {
if cm.GetName() == cluster.Spec.Mon.FloatingMon.ConfigMapName {
return true
}
}
}
return false
}
func watchControllerPredicate[T *cephv1.CephCluster](ctx context.Context, c client.Client) predicate.TypedFuncs[T] {
return predicate.TypedFuncs[T]{
CreateFunc: func(e event.TypedCreateEvent[T]) bool {
if opcontroller.DuplicateCephClusters(ctx, c, e.Object, true) {
return false
}
logger.Debug("create event from a CR")
return true
},
DeleteFunc: func(e event.TypedDeleteEvent[T]) bool {
logger.Debug("delete event from a CR")
return true
},
UpdateFunc: func(e event.TypedUpdateEvent[T]) bool {
// We still need to check on update event since the user must delete the additional CR
// Until this is done, the user can still update the CR and the operator will reconcile
// This should not happen
if opcontroller.DuplicateCephClusters(ctx, c, e.ObjectOld, true) {
return false
}
// resource.Quantity has non-exportable fields, so we use its comparator method
resourceQtyComparer := cmp.Comparer(func(x, y resource.Quantity) bool { return x.Cmp(y) == 0 })
objOld := (*cephv1.CephCluster)(e.ObjectOld)
objNew := (*cephv1.CephCluster)(e.ObjectNew)
log.NamespacedDebug(objNew.Namespace, logger, "update event on CephCluster CR")
// If the user sets the skip-reconcile label, avoid doing anything (including manager reloads)
// for this CephCluster.
//
// However, we still want to trigger a reconcile when the label is added/removed so the
// controller can early-exit and record a "ReconcileSkipped" event (when label is added),
// and also resume reconciliation immediately when the label is removed.
_, oldHasSkipLabel := objOld.GetLabels()[cephv1.SkipReconcileLabelKey]
_, newHasSkipLabel := objNew.GetLabels()[cephv1.SkipReconcileLabelKey]
if oldHasSkipLabel != newHasSkipLabel {
log.NamespacedDebug(objNew.Namespace, logger, "CephCluster %q skip-reconcile label changed (%v -> %v); triggering reconcile", objNew.Name, oldHasSkipLabel, newHasSkipLabel)
return true
}
if newHasSkipLabel {
log.NamespacedInfo(objNew.Namespace, logger, "object %q matched on update but %q label is set, doing nothing", objNew.Name, cephv1.SkipReconcileLabelKey)
return false
}
// Backwards/alternative label used by some controllers to avoid reconciling an object.
// If this label is set on the object, let's not reconcile that request.
if opcontroller.IsDoNotReconcile(objNew.GetLabels()) {
log.NamespacedDebug(objNew.Namespace, logger, "object %q matched on update but %q label is set, doing nothing", opcontroller.DoNotReconcileLabelName, objNew.Name)
return false
}
diff := cmp.Diff(objOld.Spec, objNew.Spec, resourceQtyComparer)
if diff != "" {
log.NamespacedInfo(objNew.Namespace, logger, "CR has changed for %q. diff=%s", objNew.Name, diff)
if objNew.Spec.CleanupPolicy.HasDataDirCleanPolicy() {
log.NamespacedInfo(objNew.Namespace, logger, "skipping orchestration for cluster object %q in namespace %q because its cleanup policy is set. not reloading the manager", objNew.GetName(), objNew.GetNamespace())
return false
}
// Stop any ongoing orchestration
opcontroller.ReloadManager()
return false
} else if !objOld.GetDeletionTimestamp().Equal(objNew.GetDeletionTimestamp()) {
log.NamespacedInfo(objNew.Namespace, logger, "CR %q is going be deleted, cancelling any ongoing orchestration", objNew.Name)
// Stop any ongoing orchestration
opcontroller.ReloadManager()
return false
} else if objOld.GetGeneration() != objNew.GetGeneration() {
log.NamespacedDebug(objNew.Namespace, logger, "reconciling CephCluster %q with changed generation", objNew.Name)
return true
}
return false
},
GenericFunc: func(e event.TypedGenericEvent[T]) bool {
return false
},
}
}