forked from rook/rook
This commit is a large refactor on how the operator starts, stops and how it starts various sub-components such as the ceph-csi driver. It also refines the way we cancel orchestrations. We don't use breakpoints anymore but send our self a SIGUP to reload our controller runtime manager. The reload will happen under different circonstances like: * a new adminission controller secret is created/deleted/changed * a CephCluster CR is edited As mentioned earlier, the csi driver now has its own controller, just like flex. It reacts to change in the operator config map for particular ROOK_CSI_ fields. A second new controller for the operator's general config has been created, it manages: * the logging level * the ceph CLI command timeout * the discovery daemon The operator reacts much more rapidly to cancellation events by stopping the manager's context and reloading it. Signed-off-by: Sébastien Han <seb@redhat.com>
99 lines
3.4 KiB
Go
99 lines
3.4 KiB
Go
/*
|
|
Copyright 2020 The Rook Authors. All rights reserved.
|
|
|
|
Licensed under the Apache License, Version 2.0 (the "License");
|
|
you may not use this file except in compliance with the License.
|
|
You may obtain a copy of the License at
|
|
|
|
http://www.apache.org/licenses/LICENSE-2.0
|
|
|
|
Unless required by applicable law or agreed to in writing, software
|
|
distributed under the License is distributed on an "AS IS" BASIS,
|
|
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
See the License for the specific language governing permissions and
|
|
limitations under the License.
|
|
*/
|
|
|
|
// Package cluster to manage a Ceph cluster.
|
|
package cluster
|
|
|
|
import (
|
|
"context"
|
|
|
|
cephv1 "github.com/rook/rook/pkg/apis/ceph.rook.io/v1"
|
|
cephclient "github.com/rook/rook/pkg/daemon/ceph/client"
|
|
"github.com/rook/rook/pkg/operator/ceph/cluster/mon"
|
|
"github.com/rook/rook/pkg/operator/ceph/cluster/osd"
|
|
)
|
|
|
|
var (
|
|
monitorDaemonList = []string{"mon", "osd", "status"}
|
|
)
|
|
|
|
func (c *ClusterController) configureCephMonitoring(cluster *cluster, clusterInfo *cephclient.ClusterInfo) {
|
|
var isEnabled bool
|
|
for _, daemon := range monitorDaemonList {
|
|
// Is the monitoring enabled for that daemon?
|
|
isEnabled = isMonitoringEnabled(daemon, cluster.Spec)
|
|
if health, ok := cluster.monitoringRoutines[daemon]; ok {
|
|
// If the context Err() is nil this means it hasn't been cancelled yet
|
|
if health.internalCtx.Err() == nil {
|
|
logger.Debugf("monitoring routine for %q is already running", daemon)
|
|
if !isEnabled {
|
|
cluster.monitoringRoutines[daemon].internalCancel()
|
|
}
|
|
}
|
|
} else {
|
|
if isEnabled {
|
|
// Instantiate the monitoring goroutine context from the parent context
|
|
// They can individually be cancelled and will be cancelled when the parent context is cancelled
|
|
internalCtx, internalCancel := context.WithCancel(c.OpManagerCtx)
|
|
|
|
cluster.monitoringRoutines[daemon] = &clusterHealth{
|
|
internalCtx: internalCtx,
|
|
internalCancel: internalCancel,
|
|
}
|
|
|
|
// Run the go routine
|
|
c.startMonitoringCheck(cluster, clusterInfo, daemon)
|
|
}
|
|
}
|
|
}
|
|
}
|
|
|
|
func isMonitoringEnabled(daemon string, clusterSpec *cephv1.ClusterSpec) bool {
|
|
switch daemon {
|
|
case "mon":
|
|
return !clusterSpec.HealthCheck.DaemonHealth.Monitor.Disabled
|
|
|
|
case "osd":
|
|
return !clusterSpec.HealthCheck.DaemonHealth.ObjectStorageDaemon.Disabled
|
|
|
|
case "status":
|
|
return !clusterSpec.HealthCheck.DaemonHealth.Status.Disabled
|
|
}
|
|
|
|
return false
|
|
}
|
|
|
|
func (c *ClusterController) startMonitoringCheck(cluster *cluster, clusterInfo *cephclient.ClusterInfo, daemon string) {
|
|
switch daemon {
|
|
case "mon":
|
|
healthChecker := mon.NewHealthChecker(cluster.mons)
|
|
logger.Infof("enabling ceph %s monitoring goroutine for cluster %q", daemon, cluster.Namespace)
|
|
go healthChecker.Check(cluster.monitoringRoutines[daemon].internalCtx)
|
|
|
|
case "osd":
|
|
if !cluster.Spec.External.Enable {
|
|
c.osdChecker = osd.NewOSDHealthMonitor(c.context, clusterInfo, cluster.Spec.RemoveOSDsIfOutAndSafeToRemove, cluster.Spec.HealthCheck)
|
|
logger.Infof("enabling ceph %s monitoring goroutine for cluster %q", daemon, cluster.Namespace)
|
|
go c.osdChecker.Start(cluster.monitoringRoutines[daemon].internalCtx)
|
|
}
|
|
|
|
case "status":
|
|
cephChecker := newCephStatusChecker(c.context, clusterInfo, cluster.Spec)
|
|
logger.Infof("enabling ceph %s monitoring goroutine for cluster %q", daemon, cluster.Namespace)
|
|
go cephChecker.checkCephStatus(cluster.monitoringRoutines[daemon].internalCtx)
|
|
}
|
|
}
|