ceph: convert the CephCluster controller to the controller-runtime

This is the final conversion to controller-runtime conversion. This time the
CephCluster CRD has been converted to use the controller-runtime
library.
The controller incorporates all the previous watchers too, so the Node
and hot-plug configmap are been watched too.
Only the operator setting configmap is not being watcher since it's not
related to the CephCluster CRD.
Not only the patch converts to controller-runtime but also tries to
re-organize the tree of the repo to actually make the code more
readable and have better functions/methods/tests separations.

Closes: https://github.com/rook/rook/issues/4939
Signed-off-by: Sébastien Han <seb@redhat.com>
This commit is contained in:
Sébastien Han
2020-04-28 09:40:35 +02:00
parent 008670bc79
commit f27fd207ce
55 changed files with 2635 additions and 1848 deletions
+1
View File
@@ -8,6 +8,7 @@
- Added a [toolbox job](Documentation/ceph-toolbox.md#toolbox-job) for running a script with Ceph commands, similar to running commands in the Rook toolbox.
- Ceph RBD Mirror daemon has been extracted to its own CRD, it has been removed from the `CephCluster` CRD, see the [rbd-mirror crd](Documentation/ceph-rbd-mirror-crd.html).
- CephCluster CRD has been converted to use the controller-runtime framework.
### EdgeFS
+2 -3
View File
@@ -176,9 +176,8 @@ spec:
pattern: ^$|^yes-really-destroy-data$
placement: {}
resources: {}
# somehow this is breaking the status, but let's keep this here so we don't forget it once we move to controller-runtime
# subresources:
# status: {}
subresources:
status: {}
additionalPrinterColumns:
- name: DataDirHostPath
type: string
@@ -23,3 +23,188 @@ spec:
type: integer
subresources:
status: {}
---
apiVersion: apiextensions.k8s.io/v1beta1
kind: CustomResourceDefinition
metadata:
name: cephclusters.ceph.rook.io
spec:
group: ceph.rook.io
names:
kind: CephCluster
listKind: CephClusterList
plural: cephclusters
singular: cephcluster
scope: Namespaced
version: v1
validation:
openAPIV3Schema:
properties:
spec:
properties:
annotations: {}
cephVersion:
properties:
allowUnsupported:
type: boolean
image:
type: string
dashboard:
properties:
enabled:
type: boolean
urlPrefix:
type: string
port:
type: integer
minimum: 0
maximum: 65535
ssl:
type: boolean
dataDirHostPath:
pattern: ^/(\S+)
type: string
disruptionManagement:
properties:
machineDisruptionBudgetNamespace:
type: string
managePodBudgets:
type: boolean
osdMaintenanceTimeout:
type: integer
manageMachineDisruptionBudgets:
type: boolean
skipUpgradeChecks:
type: boolean
continueUpgradeAfterChecksEvenIfNotHealthy:
type: boolean
mon:
properties:
allowMultiplePerNode:
type: boolean
count:
maximum: 9
minimum: 0
type: integer
volumeClaimTemplate: {}
mgr:
properties:
modules:
items:
properties:
name:
type: string
enabled:
type: boolean
network:
properties:
hostNetwork:
type: boolean
provider:
type: string
selectors: {}
storage:
properties:
disruptionManagement:
properties:
machineDisruptionBudgetNamespace:
type: string
managePodBudgets:
type: boolean
osdMaintenanceTimeout:
type: integer
manageMachineDisruptionBudgets:
type: boolean
useAllNodes:
type: boolean
nodes:
items:
properties:
name:
type: string
config:
properties:
metadataDevice:
type: string
storeType:
type: string
pattern: ^(filestore|bluestore)$
databaseSizeMB:
type: string
walSizeMB:
type: string
journalSizeMB:
type: string
osdsPerDevice:
type: string
encryptedDevice:
type: string
pattern: ^(true|false)$
useAllDevices:
type: boolean
deviceFilter:
type: string
devicePathFilter:
type: string
devices:
type: array
items:
properties:
name:
type: string
config: {}
resources: {}
type: array
useAllDevices:
type: boolean
deviceFilter:
type: string
devicePathFilter:
type: string
config: {}
storageClassDeviceSets: {}
monitoring:
properties:
enabled:
type: boolean
rulesNamespace:
type: string
removeOSDsIfOutAndSafeToRemove:
type: boolean
external:
properties:
enable:
type: boolean
cleanupPolicy:
properties:
deleteDataDirOnHosts:
type: string
pattern: ^$|^yes-really-destroy-data$
placement: {}
resources: {}
subresources:
status: {}
additionalPrinterColumns:
- name: DataDirHostPath
type: string
description: Directory used on the K8s nodes
JSONPath: .spec.dataDirHostPath
- name: MonCount
type: string
description: Number of MONs
JSONPath: .spec.mon.count
- name: Age
type: date
JSONPath: .metadata.creationTimestamp
- name: Phase
type: string
description: Phase
JSONPath: .status.phase
- name: Message
type: string
description: Message
JSONPath: .status.message
- name: Health
type: string
description: Ceph Health
JSONPath: .status.ceph.health
+5 -4
View File
@@ -22,9 +22,10 @@ import (
"github.com/rook/rook/pkg/clusterd"
"github.com/rook/rook/pkg/daemon/ceph/agent/flexvolume/attachment"
operator "github.com/rook/rook/pkg/operator/ceph"
cluster "github.com/rook/rook/pkg/operator/ceph/cluster"
"github.com/rook/rook/pkg/operator/ceph/cluster/mon"
"github.com/rook/rook/pkg/operator/ceph/csi"
"github.com/rook/rook/pkg/operator/ceph/disruption"
"github.com/rook/rook/pkg/operator/k8sutil"
"github.com/rook/rook/pkg/util/flags"
"github.com/spf13/cobra"
@@ -55,7 +56,7 @@ func init() {
operatorCmd.Flags().StringVar(&csi.CephFSProvisionerSTSTemplatePath, "csi-cephfs-provisioner-sts-template-path", csi.DefaultCephFSProvisionerSTSTemplatePath, "path to ceph-csi cephfs provisioner statefulset template")
operatorCmd.Flags().StringVar(&csi.CephFSProvisionerDepTemplatePath, "csi-cephfs-provisioner-dep-template-path", csi.DefaultCephFSProvisionerDepTemplatePath, "path to ceph-csi cephfs provisioner deployment template")
operatorCmd.Flags().BoolVar(&disruption.EnableMachineDisruptionBudget, "enable-machine-disruption-budget", false, "enable fencing controllers")
operatorCmd.Flags().BoolVar(&cluster.EnableMachineDisruptionBudget, "enable-machine-disruption-budget", false, "enable fencing controllers")
flags.SetFlagsFromEnv(operatorCmd.Flags(), rook.RookEnvVarPrefix)
flags.SetLoggingFlags(operatorCmd.Flags())
@@ -68,7 +69,7 @@ func startOperator(cmd *cobra.Command, args []string) error {
rook.LogStartupInfo(operatorCmd.Flags())
logger.Infof("starting operator")
logger.Info("starting Rook-Ceph operator")
context := createContext()
context.NetworkInfo = clusterd.NetworkInfo{}
context.ConfigDir = k8sutil.DataDir
@@ -82,7 +83,7 @@ func startOperator(cmd *cobra.Command, args []string) error {
op := operator.New(context, volumeAttachment, rookImage, serviceAccountName)
err = op.Run()
if err != nil {
rook.TerminateFatal(errors.Wrapf(err, "failed to run operator\n"))
rook.TerminateFatal(errors.Wrap(err, "failed to run operator\n"))
}
return nil
+2 -2
View File
@@ -26,10 +26,10 @@ import (
"github.com/pkg/errors"
"github.com/rook/rook/cmd/rook/rook"
osddaemon "github.com/rook/rook/pkg/daemon/ceph/osd"
"github.com/rook/rook/pkg/operator/ceph/cluster"
"github.com/rook/rook/pkg/operator/ceph/cluster/mon"
oposd "github.com/rook/rook/pkg/operator/ceph/cluster/osd"
osdcfg "github.com/rook/rook/pkg/operator/ceph/cluster/osd/config"
opcontroller "github.com/rook/rook/pkg/operator/ceph/controller"
"github.com/rook/rook/pkg/operator/k8sutil"
"github.com/rook/rook/pkg/util/flags"
"github.com/spf13/cobra"
@@ -209,7 +209,7 @@ func prepareOSD(cmd *cobra.Command, args []string) error {
logger.Infof("crush location of osd: %s", crushLocation)
forceFormat := false
ownerRef := cluster.ClusterOwnerRef(clusterInfo.Name, ownerRefID)
ownerRef := opcontroller.ClusterOwnerRef(clusterInfo.Name, ownerRefID)
kv := k8sutil.NewConfigMapKVStore(clusterInfo.Name, context.Clientset, ownerRef)
agent := osddaemon.NewAgent(context, dataDevices, cfg.metadataDevice, forceFormat,
cfg.storeConfig, &clusterInfo, cfg.nodeName, kv, cfg.pvcBacked)
+5
View File
@@ -13,6 +13,7 @@ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
*/
package clusterd
import (
@@ -24,6 +25,7 @@ import (
apiextensionsclient "k8s.io/apiextensions-apiserver/pkg/client/clientset/clientset"
"k8s.io/client-go/kubernetes"
"k8s.io/client-go/rest"
"sigs.k8s.io/controller-runtime/pkg/client"
)
// Context for loading or applying the configuration state of a service.
@@ -35,6 +37,9 @@ type Context struct {
// Clientset is a connection to the core kubernetes API
Clientset kubernetes.Interface
// Represents the Client provided by the controller-runtime package to interact with Kubernetes objects
Client client.Client
// APIExtensionClientset is a connection to the API Extension kubernetes API
APIExtensionClientset apiextensionsclient.Interface
+3 -2
View File
@@ -26,7 +26,8 @@ import (
"github.com/rook/rook/pkg/clusterd"
"github.com/rook/rook/pkg/daemon/ceph/agent/flexvolume"
"github.com/rook/rook/pkg/daemon/ceph/agent/flexvolume/attachment"
opcluster "github.com/rook/rook/pkg/operator/ceph/cluster"
opcontroller "github.com/rook/rook/pkg/operator/ceph/controller"
"github.com/rook/rook/pkg/operator/k8sutil"
"k8s.io/apimachinery/pkg/runtime"
"k8s.io/client-go/tools/cache"
@@ -67,7 +68,7 @@ func (c *ClusterController) StartWatch(namespace string, stopCh chan struct{}) {
}
logger.Infof("start watching cluster resources")
go k8sutil.WatchCR(opcluster.ClusterResource, namespace, resourceHandlerFuncs, c.context.RookClientset.CephV1().RESTClient(), &cephv1.CephCluster{}, stopCh)
go k8sutil.WatchCR(opcontroller.ClusterResource, namespace, resourceHandlerFuncs, c.context.RookClientset.CephV1().RESTClient(), &cephv1.CephCluster{}, stopCh)
}
func (c *ClusterController) onDelete(obj interface{}) {
+11
View File
@@ -173,3 +173,14 @@ func isCrushFieldSet(fieldName string, pairs []string) bool {
func formatProperty(name, value string) string {
return fmt.Sprintf("%s=%s", name, value)
}
// GetOSDOnHost returns the list of osds running on a given host
func GetOSDOnHost(context *clusterd.Context, clusterName, node string) (string, error) {
args := []string{"osd", "crush", "ls", node}
buf, err := NewCephCommand(context, clusterName, args).Run()
if err != nil {
return "", errors.Wrapf(err, "failed to get osd list on host. %s", string(buf))
}
return string(buf), nil
}
+14
View File
@@ -245,6 +245,20 @@ func TestGetCrushMap(t *testing.T) {
assert.Equal(t, 2, len(crush.Rules))
}
func TestGetOSDOnHost(t *testing.T) {
executor := &exectest.MockExecutor{}
executor.MockExecuteCommandWithOutputFile = func(command, outputFile string, args ...string) (string, error) {
logger.Infof("Command: %s %v", command, args)
if args[1] == "crush" && args[2] == "ls" {
return "[\"osd.2\",\"osd.0\",\"osd.1\"]", nil
}
return "", errors.Errorf("unexpected ceph command '%v'", args)
}
_, err := GetOSDOnHost(&clusterd.Context{Executor: executor}, "rook-ceph", "my-host")
assert.Nil(t, err)
}
func TestCrushName(t *testing.T) {
// each is slightly different than the last
crushNames := []string{
+3 -3
View File
@@ -60,7 +60,7 @@ func getAllCephDaemonVersionsString(context *clusterd.Context, clusterName strin
args := []string{"versions"}
buf, err := NewCephCommand(context, clusterName, args).Run()
if err != nil {
return "", errors.Wrapf(err, "failed to run 'ceph versions")
return "", errors.Wrapf(err, "failed to run 'ceph versions. %s", string(buf))
}
output := string(buf)
logger.Debug(output)
@@ -78,7 +78,7 @@ func GetCephMonVersion(context *clusterd.Context, clusterName string) (*cephver.
v, err := cephver.ExtractCephVersion(output)
if err != nil {
return nil, errors.Wrapf(err, "failed to extract ceph version")
return nil, errors.Wrap(err, "failed to extract ceph version")
}
return v, nil
@@ -95,7 +95,7 @@ func GetAllCephDaemonVersions(context *clusterd.Context, clusterName string) (*C
var cephVersionsResult CephDaemonsVersions
err = json.Unmarshal([]byte(output), &cephVersionsResult)
if err != nil {
return nil, errors.Wrapf(err, "failed to retrieve ceph versions results")
return nil, errors.Wrap(err, "failed to retrieve ceph versions results")
}
return &cephVersionsResult, nil
+1 -1
View File
@@ -99,7 +99,7 @@ func TestGenerateConfigFile(t *testing.T) {
CephVersion: cephver.Nautilus,
}
isInitialized := clusterInfo.IsInitialized()
isInitialized := clusterInfo.IsInitialized(true)
assert.True(t, isInitialized)
// generate the config file to disk now
+13 -5
View File
@@ -53,17 +53,25 @@ type ExternalCred struct {
// in. This method exists less out of necessity than the desire to be explicit about the lifecycle
// of the ClusterInfo struct during startup, specifically that it is expected to exist after the
// Rook operator has started up or connected to the first components of the Ceph cluster.
func (c *ClusterInfo) IsInitialized() bool {
func (c *ClusterInfo) IsInitialized(logError bool) bool {
var isInitialized bool
if c == nil {
logger.Error("clusterInfo is nil")
if logError {
logger.Error("clusterInfo is nil")
}
} else if c.FSID == "" {
logger.Error("cluster fsid is empty")
if logError {
logger.Error("cluster fsid is empty")
}
} else if c.MonitorSecret == "" {
logger.Error("monitor secret is empty")
if logError {
logger.Error("monitor secret is empty")
}
} else if c.AdminSecret == "" {
logger.Error("admin secret is empty")
if logError {
logger.Error("admin secret is empty")
}
} else {
isInitialized = true
}
+66 -29
View File
@@ -18,15 +18,20 @@ limitations under the License.
package cluster
import (
"context"
"os"
"time"
"github.com/pkg/errors"
cephv1 "github.com/rook/rook/pkg/apis/ceph.rook.io/v1"
"github.com/rook/rook/pkg/clusterd"
"github.com/rook/rook/pkg/daemon/ceph/client"
cephclient "github.com/rook/rook/pkg/daemon/ceph/client"
"github.com/rook/rook/pkg/daemon/ceph/config"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
opcontroller "github.com/rook/rook/pkg/operator/ceph/controller"
cephver "github.com/rook/rook/pkg/operator/ceph/version"
kerrors "k8s.io/apimachinery/pkg/api/errors"
"k8s.io/apimachinery/pkg/types"
"sigs.k8s.io/controller-runtime/pkg/client"
)
const (
@@ -36,21 +41,23 @@ const (
// cephStatusChecker aggregates the mon/cluster info needed to check the health of the monitors
type cephStatusChecker struct {
context *clusterd.Context
namespace string
resourceName string
interval time.Duration
externalCred config.ExternalCred
context *clusterd.Context
resourceName string
interval time.Duration
externalCred config.ExternalCred
client client.Client
namespacedName types.NamespacedName
}
// newCephStatusChecker creates a new HealthChecker object
func newCephStatusChecker(context *clusterd.Context, namespace, resourceName string, externalCred config.ExternalCred) *cephStatusChecker {
func newCephStatusChecker(context *clusterd.Context, resourceName string, externalCred config.ExternalCred, namespacedName types.NamespacedName) *cephStatusChecker {
c := &cephStatusChecker{
context: context,
namespace: namespace,
resourceName: resourceName,
interval: defaultStatusCheckInterval,
externalCred: externalCred,
context: context,
resourceName: resourceName,
interval: defaultStatusCheckInterval,
externalCred: externalCred,
client: context.Client,
namespacedName: namespacedName,
}
// allow overriding the check interval with an env var on the operator
@@ -83,24 +90,24 @@ func (c *cephStatusChecker) checkCephStatus(stopCh chan struct{}) {
// checkStatus queries the status of ceph health then updates the CR status
func (c *cephStatusChecker) checkStatus() {
var status client.CephStatus
var status cephclient.CephStatus
var err error
logger.Debugf("checking health of cluster")
// Set the user health check to the admin user
healthCheckUser := client.AdminUsername
healthCheckUser := cephclient.AdminUsername
// This is an external cluster OR if the admin keyring is not present
// As of 1.3 an external cluster is deployed it uses a different user to check ceph's status
if c.externalCred.Username != "" && c.externalCred.Secret != "" {
if c.externalCred.Username != client.AdminUsername {
if c.externalCred.Username != cephclient.AdminUsername {
healthCheckUser = c.externalCred.Username
}
}
// Check ceph's status
status, err = client.StatusWithUser(c.context, c.namespace, healthCheckUser)
status, err = cephclient.StatusWithUser(c.context, c.namespacedName.Namespace, healthCheckUser)
if err != nil {
logger.Errorf("failed to get ceph status. %v", err)
return
@@ -108,30 +115,33 @@ func (c *cephStatusChecker) checkStatus() {
logger.Debugf("Cluster status: %+v", status)
if err := c.updateCephStatus(&status); err != nil {
logger.Errorf("failed to query cluster status in namespace %q. %v", c.namespace, err)
logger.Errorf("failed to query cluster status in namespace %q. %v", c.namespacedName.Namespace, err)
}
}
// updateCephStatus detects the latest health status from ceph and updates the CR status
func (c *cephStatusChecker) updateCephStatus(status *client.CephStatus) error {
// get the most recent cluster CRD object
cluster, err := c.context.RookClientset.CephV1().CephClusters(c.namespace).Get(c.resourceName, metav1.GetOptions{})
// updateStatus updates an object with a given status
func (c *cephStatusChecker) updateCephStatus(status *cephclient.CephStatus) error {
cephCluster := &cephv1.CephCluster{}
err := c.client.Get(context.TODO(), c.namespacedName, cephCluster)
if err != nil {
return errors.Wrapf(err, "failed to get cluster from namespace %s prior to updating its status", c.namespace)
if kerrors.IsNotFound(err) {
logger.Debug("CephCluster resource not found. Ignoring since object must be deleted.")
return nil
}
return errors.Wrapf(err, "failed to retrieve ceph cluster %q to update status to %+v", c.namespacedName.Name, status)
}
// translate the ceph status struct to the crd status
cluster.Status.CephStatus = toCustomResourceStatus(cluster.Status, status)
if _, err := c.context.RookClientset.CephV1().CephClusters(c.namespace).Update(cluster); err != nil {
return errors.Wrapf(err, "failed to update cluster %s status", c.namespace)
cephCluster.Status.CephStatus = toCustomResourceStatus(cephCluster.Status, status)
if err := opcontroller.UpdateStatus(c.client, cephCluster); err != nil {
return errors.Wrapf(err, "failed to update cluster %q status", c.namespacedName.Namespace)
}
logger.Debugf("ceph cluster %q status updated to %+v", c.namespacedName.Name, status)
return nil
}
// toCustomResourceStatus converts the ceph status to the struct expected for the CephCluster CR status
func toCustomResourceStatus(currentStatus cephv1.ClusterStatus, newStatus *client.CephStatus) *cephv1.CephStatus {
func toCustomResourceStatus(currentStatus cephv1.ClusterStatus, newStatus *cephclient.CephStatus) *cephv1.CephStatus {
s := &cephv1.CephStatus{
Health: newStatus.Health.Status,
LastChecked: formatTime(time.Now().UTC()),
@@ -157,3 +167,30 @@ func toCustomResourceStatus(currentStatus cephv1.ClusterStatus, newStatus *clien
func formatTime(t time.Time) string {
return t.Format(time.RFC3339)
}
func (c *ClusterController) updateClusterCephVersion(image string, cephVersion cephver.CephVersion) {
logger.Infof("cluster %q: version %q detected for image %q", c.namespacedName.Namespace, cephVersion.String(), image)
cephCluster := &cephv1.CephCluster{}
err := c.client.Get(context.TODO(), c.namespacedName, cephCluster)
if err != nil {
if kerrors.IsNotFound(err) {
logger.Debug("CephCluster resource not found. Ignoring since object must be deleted.")
return
}
logger.Errorf("failed to retrieve ceph cluster %q to update ceph version to %+v. %v", c.namespacedName.Name, cephVersion, err)
return
}
cephClusterVersion := &cephv1.ClusterVersion{
Image: image,
Version: opcontroller.GetCephVersionLabel(cephVersion),
}
// update the Ceph version on the retrieved cluster object
// do not overwrite the ceph status that is updated in a separate goroutine
cephCluster.Status.CephVersion = cephClusterVersion
if err := opcontroller.UpdateStatus(c.client, cephCluster); err != nil {
logger.Errorf("failed to update cluster %q version. %v", c.namespacedName.Name, err)
return
}
}
+5
View File
@@ -193,3 +193,8 @@ func (c *ClusterController) getMonSecret(namespace string) (string, error) {
return clusterInfo.MonitorSecret, nil
}
func hasCleanupPolicy(cephCluster *cephv1.CephCluster) bool {
policy := cephCluster.Spec.CleanupPolicy
return policy.DeleteDataDirOnHosts != ""
}
+215 -310
View File
@@ -18,30 +18,27 @@ limitations under the License.
package cluster
import (
"reflect"
"sort"
"path"
"strings"
"sync"
"time"
"github.com/pkg/errors"
"github.com/rook/rook/pkg/operator/k8sutil/cmdreporter"
"github.com/google/go-cmp/cmp"
cephv1 "github.com/rook/rook/pkg/apis/ceph.rook.io/v1"
rookv1 "github.com/rook/rook/pkg/apis/rook.io/v1"
"github.com/rook/rook/pkg/clusterd"
"github.com/rook/rook/pkg/daemon/ceph/client"
cephconfig "github.com/rook/rook/pkg/daemon/ceph/config"
cephclient "github.com/rook/rook/pkg/operator/ceph/client"
"github.com/rook/rook/pkg/operator/ceph/cluster/crash"
"github.com/rook/rook/pkg/operator/ceph/cluster/mgr"
"github.com/rook/rook/pkg/operator/ceph/cluster/mon"
"github.com/rook/rook/pkg/operator/ceph/cluster/osd"
"github.com/rook/rook/pkg/operator/ceph/config"
"github.com/rook/rook/pkg/operator/ceph/controller"
"github.com/rook/rook/pkg/operator/ceph/csi"
"github.com/rook/rook/pkg/operator/ceph/object/bucket"
cephver "github.com/rook/rook/pkg/operator/ceph/version"
"k8s.io/apimachinery/pkg/api/resource"
v1 "k8s.io/api/core/v1"
kerrors "k8s.io/apimachinery/pkg/api/errors"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
)
@@ -66,8 +63,7 @@ type cluster struct {
isUpgrade bool
}
func newCluster(c *cephv1.CephCluster, context *clusterd.Context, csiMutex *sync.Mutex) *cluster {
ownerRef := ClusterOwnerRef(c.Name, string(c.UID))
func newCluster(c *cephv1.CephCluster, context *clusterd.Context, csiMutex *sync.Mutex, ownerRef *metav1.OwnerReference) *cluster {
return &cluster{
// at this phase of the cluster creation process, the identity components of the cluster are
// not yet established. we reserve this struct which is filled in as soon as the cluster's
@@ -78,138 +74,15 @@ func newCluster(c *cephv1.CephCluster, context *clusterd.Context, csiMutex *sync
context: context,
crdName: c.Name,
stopCh: make(chan struct{}),
ownerRef: ownerRef,
mons: mon.New(context, c.Namespace, c.Spec.DataDirHostPath, c.Spec.Network, ownerRef, csiMutex),
ownerRef: *ownerRef,
mons: mon.New(context, c.Namespace, c.Spec.DataDirHostPath, c.Spec.Network, *ownerRef, csiMutex),
}
}
// detectCephVersion loads the ceph version from the image and checks that it meets the version requirements to
// run in the cluster
func (c *cluster) detectCephVersion(rookImage, cephImage string, timeout time.Duration) (*cephver.CephVersion, error) {
logger.Infof("detecting the ceph image version for image %s...", cephImage)
versionReporter, err := cmdreporter.New(
c.context.Clientset, &c.ownerRef,
detectVersionName, detectVersionName, c.Namespace,
[]string{"ceph"}, []string{"--version"},
rookImage, cephImage)
if err != nil {
return nil, errors.Wrapf(err, "failed to set up ceph version job")
}
job := versionReporter.Job()
job.Spec.Template.Spec.ServiceAccountName = "rook-ceph-cmd-reporter"
// Apply the same node selector and tolerations for the ceph version detection as the mon daemons
cephv1.GetMonPlacement(c.Spec.Placement).ApplyToPodSpec(&job.Spec.Template.Spec)
stdout, stderr, retcode, err := versionReporter.Run(timeout)
if err != nil {
return nil, errors.Wrapf(err, "failed to complete ceph version job")
}
if retcode != 0 {
return nil, errors.Errorf(`ceph version job returned failure with retcode %d.
stdout: %s
stderr: %s`, retcode, stdout, stderr)
}
version, err := cephver.ExtractCephVersion(stdout)
if err != nil {
return nil, errors.Wrapf(err, "failed to extract ceph version")
}
logger.Infof("Detected ceph image version: %q", version)
return version, nil
}
func (c *cluster) validateCephVersion(version *cephver.CephVersion) error {
if !c.Spec.External.Enable {
if !version.IsAtLeast(cephver.Minimum) {
return errors.Errorf("the version does not meet the minimum version %q", cephver.Minimum.String())
}
if !version.Supported() {
if !c.Spec.CephVersion.AllowUnsupported {
return errors.Errorf("allowUnsupported must be set to true to run with this version %q", version.String())
}
logger.Warningf("unsupported ceph version detected: %q, pursuing", version)
}
}
// The following tries to determine if the operator can proceed with an upgrade because we come from an OnAdd() call
// If the cluster was unhealthy and someone injected a new image version, an upgrade was triggered but failed because the cluster is not healthy
// Then after this, if the operator gets restarted we are not able to fail if the cluster is not healthy, the following tries to determine the
// state we are in and if we should upgrade or not
// Try to load clusterInfo so we can compare the running version with the one from the spec image
clusterInfo, _, _, err := mon.LoadClusterInfo(c.context, c.Namespace)
if err == nil {
// Write connection info (ceph config file and keyring) for ceph commands
err = mon.WriteConnectionConfig(c.context, clusterInfo)
if err != nil {
logger.Errorf("failed to write config. attempting to continue. %v", err)
}
}
if !clusterInfo.IsInitialized() {
// If not initialized, this is likely a new cluster so there is nothing to do
logger.Debug("cluster not initialized, nothing to validate")
return nil
}
if c.Spec.External.Enable && c.Spec.CephVersion.Image != "" {
c.Info.CephVersion, err = controller.ValidateCephVersionsBetweenLocalAndExternalClusters(c.context, c.Namespace, *version)
if err != nil {
return errors.Wrapf(err, "failed to validate ceph version between external and local")
}
}
// On external cluster setup, if we don't bootstrap any resources in the Kubernetes cluster then
// there is no need to validate the Ceph image further
if c.Spec.External.Enable && c.Spec.CephVersion.Image == "" {
logger.Debug("no spec image specified on external cluster, not validating Ceph version.")
return nil
}
// Get cluster running versions
versions, err := client.GetAllCephDaemonVersions(c.context, c.Namespace)
if err != nil {
logger.Errorf("failed to get ceph daemons versions. %v", err)
return nil
}
runningVersions := *versions
differentImages, err := diffImageSpecAndClusterRunningVersion(*version, runningVersions)
if err != nil {
logger.Errorf("failed to determine if we should upgrade or not. %v", err)
// we shouldn't block the orchestration if we can't determine the version of the image spec, we proceed anyway in best effort
// we won't be able to check if there is an update or not and what to do, so we don't check the cluster status either
// This will happen if someone uses ceph/daemon:latest-master for instance
return nil
}
if differentImages {
// If the image version changed let's make sure we can safely upgrade
// check ceph's status, if not healthy we fail
cephHealthy := client.IsCephHealthy(c.context, c.Namespace)
if !cephHealthy {
if c.Spec.SkipUpgradeChecks {
logger.Warning("ceph is not healthy but SkipUpgradeChecks is set, forcing upgrade.")
} else {
return errors.Errorf("ceph status in namespace %s is not healthy, refusing to upgrade. fix the cluster and re-edit the cluster CR to trigger a new orchestation update", c.Namespace)
}
}
c.isUpgrade = true
}
return nil
}
// initialized checks if the cluster has ever completed a successful orchestration since the operator has started
func (c *cluster) initialized() bool {
return c.initCompleted
}
func (c *cluster) createInstance(rookImage string, cephVersion cephver.CephVersion) error {
var err error
// Set orchestration lock, implying the orchestation is in progress
c.setOrchestrationNeeded()
// execute an orchestration until
@@ -217,13 +90,15 @@ func (c *cluster) createInstance(rookImage string, cephVersion cephver.CephVersi
// while no other goroutine is already running a cluster update
for c.checkSetOrchestrationStatus() == true {
if err != nil {
logger.Errorf("There was an orchestration error, but there is another orchestration pending; proceeding with next orchestration run (which may succeed). %v", err)
logger.Errorf("there was an orchestration error, but there is another orchestration pending; proceeding with next orchestration run (which may succeed). %v", err)
}
// Use a DeepCopy of the spec to avoid using an inconsistent data-set
spec := c.Spec.DeepCopy()
// Run ceph orchestration
err = c.doOrchestration(rookImage, cephVersion, spec)
// Orchestration is done, remove the lock
c.unsetOrchestrationStatus()
}
@@ -235,199 +110,163 @@ func (c *cluster) doOrchestration(rookImage string, cephVersion cephver.CephVers
// These settings should only be modified by a user after they are initialized
err := populateConfigOverrideConfigMap(c.context, c.Namespace, c.ownerRef)
if err != nil {
return errors.Wrapf(err, "failed to populate config override config map")
return errors.Wrap(err, "failed to populate config override config map")
}
if c.Spec.External.Enable {
// Apply CRD ConfigOverrideName to the external cluster
err = config.SetDefaultConfigs(c.context, c.Namespace, c.Info, cephv1.NetworkSpec{})
if err != nil {
// Mons are up, so something else is wrong
return errors.Wrapf(err, "failed to set Rook and/or user-defined Ceph config options on the external cluster monitors")
}
// The cluster Identity must be established at this point
if !c.Info.IsInitialized() {
return errors.Errorf("the cluster identity was not established: %+v", c.Info)
}
} else {
// This gets triggered on CR update so let's not run that (mon/mgr/osd daemons)
// Start the mon pods
clusterInfo, err := c.mons.Start(c.Info, rookImage, cephVersion, *c.Spec)
if err != nil {
return errors.Wrapf(err, "failed to start the mons")
}
c.Info = clusterInfo // mons return the cluster's info
// The cluster Identity must be established at this point
if !c.Info.IsInitialized() {
return errors.Errorf("the cluster identity was not established: %+v", c.Info)
}
// Execute actions after the monitors are up and running
logger.Debug("monitors are up and running, executing post actions")
err = c.postMonStartupActions()
if err != nil {
return errors.Wrapf(err, "failed to execute post actions after all the monitors started")
}
// If this is an upgrade, notify all the child controllers
if c.isUpgrade {
logger.Info("upgrade in progress, notifying child CRs")
err := c.notifyChildControllerOfUpgrade()
if err != nil {
return errors.Wrap(err, "failed to notify child CRs of upgrade")
}
}
mgrs := mgr.New(c.Info, c.context, c.Namespace, rookImage,
spec.CephVersion, cephv1.GetMgrPlacement(spec.Placement), cephv1.GetMgrAnnotations(c.Spec.Annotations),
spec.Network, spec.Dashboard, spec.Monitoring, spec.Mgr, cephv1.GetMgrResources(spec.Resources),
cephv1.GetMgrPriorityClassName(spec.PriorityClassNames), c.ownerRef, c.Spec.DataDirHostPath, c.Spec.SkipUpgradeChecks)
err = mgrs.Start()
if err != nil {
return errors.Wrapf(err, "failed to start the ceph mgr")
}
// Start the OSDs
osds := osd.New(c.Info, c.context, c.Namespace, rookImage, spec.CephVersion, spec.Storage, spec.DataDirHostPath,
cephv1.GetOSDPlacement(spec.Placement), cephv1.GetOSDAnnotations(spec.Annotations), spec.Network,
cephv1.GetOSDResources(spec.Resources), cephv1.GetPrepareOSDResources(spec.Resources), cephv1.GetOSDPriorityClassName(spec.PriorityClassNames), c.ownerRef, c.Spec.SkipUpgradeChecks, c.Spec.ContinueUpgradeAfterChecksEvenIfNotHealthy)
err = osds.Start()
if err != nil {
return errors.Wrapf(err, "failed to start the osds")
}
logger.Infof("Done creating rook instance in namespace %s", c.Namespace)
c.initCompleted = true
// Start the mon pods
clusterInfo, err := c.mons.Start(c.Info, rookImage, cephVersion, *c.Spec)
if err != nil {
return errors.Wrap(err, "failed to start ceph monitors")
}
c.Info = clusterInfo
// The cluster Identity must be established at this point
if !c.Info.IsInitialized(true) {
return errors.New("the cluster identity was not established")
}
// Execute actions after the monitors are up and running
logger.Debug("monitors are up and running, executing post actions")
err = c.postMonStartupActions()
if err != nil {
return errors.Wrap(err, "failed to execute post actions after all the ceph monitors started")
}
// If this is an upgrade, notify all the child controllers
if c.isUpgrade {
logger.Info("upgrade in progress, notifying child CRs")
err := c.notifyChildControllerOfUpgrade()
if err != nil {
return errors.Wrap(err, "failed to notify child CRs of upgrade")
}
}
// Start Ceph manager
mgrs := mgr.New(c.Info, c.context, c.Namespace, rookImage,
spec.CephVersion, cephv1.GetMgrPlacement(spec.Placement), cephv1.GetMgrAnnotations(c.Spec.Annotations),
spec.Network, spec.Dashboard, spec.Monitoring, spec.Mgr, cephv1.GetMgrResources(spec.Resources),
cephv1.GetMgrPriorityClassName(spec.PriorityClassNames), c.ownerRef, c.Spec.DataDirHostPath, c.Spec.SkipUpgradeChecks)
err = mgrs.Start()
if err != nil {
return errors.Wrap(err, "failed to start ceph mgr")
}
// Start the OSDs
osds := osd.New(c.Info, c.context, c.Namespace, rookImage, spec.CephVersion, spec.Storage, spec.DataDirHostPath,
cephv1.GetOSDPlacement(spec.Placement), cephv1.GetOSDAnnotations(spec.Annotations), spec.Network,
cephv1.GetOSDResources(spec.Resources), cephv1.GetPrepareOSDResources(spec.Resources), cephv1.GetOSDPriorityClassName(spec.PriorityClassNames), c.ownerRef, c.Spec.SkipUpgradeChecks, c.Spec.ContinueUpgradeAfterChecksEvenIfNotHealthy)
err = osds.Start()
if err != nil {
return errors.Wrap(err, "failed to start ceph osds")
}
logger.Infof("done reconciling ceph cluster in namespace %q", c.Namespace)
// We should be done updating by now
if c.isUpgrade {
c.printOverallCephVersion()
// reset the isUpgrade flag
c.isUpgrade = false
}
// Orchestration is done
c.initCompleted = true
return nil
}
func clusterChanged(oldCluster, newCluster cephv1.ClusterSpec, clusterRef *cluster) (bool, string) {
func (c *ClusterController) initializeCluster(cluster *cluster, clusterObj *cephv1.CephCluster) {
cluster.Spec = &clusterObj.Spec
// sort the nodes by name then compare to see if there are changes
sort.Sort(rookv1.NodesByName(oldCluster.Storage.Nodes))
sort.Sort(rookv1.NodesByName(newCluster.Storage.Nodes))
// any change in the crd will trigger an orchestration
if !reflect.DeepEqual(oldCluster, newCluster) {
diff := ""
func() {
defer func() {
if err := recover(); err != nil {
logger.Warningf("Encountered an issue getting cluster change differences: %v", err)
}
}()
// resource.Quantity has non-exportable fields, so we use its comparator method
resourceQtyComparer := cmp.Comparer(func(x, y resource.Quantity) bool { return x.Cmp(y) == 0 })
diff = cmp.Diff(oldCluster, newCluster, resourceQtyComparer)
}()
if diff != "" {
logger.Infof("The Cluster CR has changed. diff=%s", diff)
return true, diff
}
}
return false, ""
}
func (c *cluster) setOrchestrationNeeded() {
c.orchMux.Lock()
c.orchestrationNeeded = true
c.orchMux.Unlock()
}
// unsetOrchestrationStatus resets the orchestrationRunning-flag
func (c *cluster) unsetOrchestrationStatus() {
c.orchMux.Lock()
defer c.orchMux.Unlock()
c.orchestrationRunning = false
}
// checkSetOrchestrationStatus is responsible to do orchestration as long as there is a request needed
func (c *cluster) checkSetOrchestrationStatus() bool {
c.orchMux.Lock()
defer c.orchMux.Unlock()
// check if there is an orchestration needed currently
if c.orchestrationNeeded == true && c.orchestrationRunning == false {
// there is an orchestration needed
// allow to enter the orchestration-loop
c.orchestrationNeeded = false
c.orchestrationRunning = true
return true
}
return false
}
// This function compare the Ceph spec image and the cluster running version
// It returns true if the image is different and false if identical
func diffImageSpecAndClusterRunningVersion(imageSpecVersion cephver.CephVersion, runningVersions client.CephDaemonsVersions) (bool, error) {
numberOfCephVersions := len(runningVersions.Overall)
if numberOfCephVersions == 0 {
// let's return immediately
return false, errors.Errorf("no 'overall' section in the ceph versions. %+v", runningVersions.Overall)
}
if numberOfCephVersions > 1 {
// let's return immediately
logger.Warningf("it looks like we have more than one ceph version running. triggering upgrade. %+v:", runningVersions.Overall)
return true, nil
}
if numberOfCephVersions == 1 {
for v := range runningVersions.Overall {
version, err := cephver.ExtractCephVersion(v)
if err != nil {
logger.Errorf("failed to extract ceph version. %v", err)
return false, err
}
clusterRunningVersion := *version
// If this is the same version
if cephver.IsIdentical(clusterRunningVersion, imageSpecVersion) {
logger.Debugf("both cluster and image spec versions are identical, doing nothing %s", imageSpecVersion.String())
return false, nil
}
if cephver.IsSuperior(imageSpecVersion, clusterRunningVersion) {
logger.Infof("image spec version %s is higher than the running cluster version %s, upgrading", imageSpecVersion.String(), clusterRunningVersion.String())
return true, nil
}
if cephver.IsInferior(imageSpecVersion, clusterRunningVersion) {
return true, errors.Errorf("image spec version %s is lower than the running cluster version %s, downgrading is not supported", imageSpecVersion.String(), clusterRunningVersion.String())
}
// Check if the dataDirHostPath is located in the disallowed paths list
cleanDataDirHostPath := path.Clean(cluster.Spec.DataDirHostPath)
for _, b := range disallowedHostDirectories {
if cleanDataDirHostPath == b {
logger.Errorf("dataDirHostPath (given: %q) must not be used, conflicts with %q internal path", cluster.Spec.DataDirHostPath, b)
return
}
}
return false, nil
// Depending on the cluster type choose the correct orchestation
if cluster.Spec.External.Enable {
err := c.configureExternalCephCluster(cluster)
if err != nil {
config.ConditionExport(c.context, c.namespacedName, cephv1.ConditionFailure, v1.ConditionTrue, "ClusterFailure", "Failed to configure external ceph cluster")
logger.Errorf("failed to configure external ceph cluster. %v", err)
return
}
} else {
err := c.configureLocalCephCluster(cluster, clusterObj)
if err != nil {
logger.Errorf("failed to configure local ceph cluster. %v", err)
return
}
}
// Start client CRD watcher
clientController := cephclient.NewClientController(c.context, cluster.Namespace)
clientController.StartWatch(cluster.stopCh)
// Start the object bucket provisioner
bucketProvisioner := bucket.NewProvisioner(c.context, cluster.Namespace)
// note: the error return below is ignored and is expected to be removed from the
// bucket library's `NewProvisioner` function
bucketController, _ := bucket.NewBucketController(c.context.KubeConfig, bucketProvisioner)
go bucketController.Run(cluster.stopCh)
// Populate ClusterInfo with the last value
cluster.mons.ClusterInfo = cluster.Info
// Start mon health checker
healthChecker := mon.NewHealthChecker(cluster.mons, cluster.Spec)
go healthChecker.Check(cluster.stopCh)
if !cluster.Spec.External.Enable {
// Start the osd health checker only if running OSDs in the local ceph cluster
c.osdChecker = osd.NewMonitor(c.context, cluster.Namespace, cluster.Spec.RemoveOSDsIfOutAndSafeToRemove, cluster.Info.CephVersion)
go c.osdChecker.Start(cluster.stopCh)
}
// Start the ceph status checker
cephChecker := newCephStatusChecker(c.context, cluster.Namespace, cluster.mons.ClusterInfo.ExternalCred, c.namespacedName)
go cephChecker.checkCephStatus(cluster.stopCh)
}
// postMonStartupActions is a collection of actions to run once the monitors are up and running
// It gets executed right after the main mon Start() method
// Basically, it is executed between the monitors and the manager sequence
func (c *cluster) postMonStartupActions() error {
// Create CSI Kubernetes Secrets
err := csi.CreateCSISecrets(c.context, c.Namespace, &c.ownerRef)
func (c *ClusterController) configureLocalCephCluster(cluster *cluster, clusterObj *cephv1.CephCluster) error {
// Cluster Spec validation
err := c.preClusterStartValidation(cluster, clusterObj)
if err != nil {
return errors.Wrapf(err, "failed to create csi kubernetes secrets")
return errors.Wrap(err, "failed to perform validation before cluster creation")
}
// Create crash collector Kubernetes Secret
err = crash.CreateCrashCollectorSecret(c.context, c.Namespace, &c.ownerRef)
// Pass down the client to interact with Kubernetes objects
// This will be used later down by spec code to create objects like deployment, services etc
cluster.context.Client = c.client
// Run image validation job
cephVersion, isUpgrade, err := c.detectAndValidateCephVersion(cluster)
if err != nil {
return errors.Wrapf(err, "failed to create crash collector kubernetes secret")
return errors.Wrap(err, "failed the ceph version check")
}
// Enable Ceph messenger 2 protocol on Nautilus
if err := client.EnableMessenger2(c.context, c.Namespace); err != nil {
return errors.Wrapf(err, "failed to enable Ceph messenger version 2.")
// Set the value of isUpgrade based on the image discovery done by detectAndValidateCephVersion()
cluster.isUpgrade = isUpgrade
// Set the condition to the cluster object
message := config.CheckConditionReady(c.context, c.namespacedName)
config.ConditionExport(c.context, c.namespacedName, cephv1.ConditionProgressing, v1.ConditionTrue, "ClusterProgressing", message)
// Run the orchestration
err = cluster.createInstance(c.rookImage, *cephVersion)
if err != nil {
config.ConditionExport(c.context, c.namespacedName, cephv1.ConditionFailure, v1.ConditionTrue, "ClusterFailure", "Failed to create cluster")
return errors.Wrap(err, "failed to create cluster")
}
// Set the condition to the cluster object
config.ConditionExport(c.context, c.namespacedName, cephv1.ConditionReady, v1.ConditionTrue, "ClusterCreated", "Cluster created successfully")
return nil
}
@@ -482,3 +321,69 @@ func (c *cluster) notifyChildControllerOfUpgrade() error {
return nil
}
// Validate the cluster Specs
func (c *ClusterController) preClusterStartValidation(cluster *cluster, clusterObj *cephv1.CephCluster) error {
if cluster.Spec.Mon.Count == 0 {
logger.Warningf("mon count should be at least 1, will use default value of %d", mon.DefaultMonCount)
cluster.Spec.Mon.Count = mon.DefaultMonCount
}
if cluster.Spec.Mon.Count%2 == 0 {
logger.Warningf("mon count is even (given: %d), should be uneven, continuing", cluster.Spec.Mon.Count)
}
if len(cluster.Spec.Storage.Directories) != 0 {
logger.Warning("running osds on directory is not supported anymore, use devices instead.")
}
if cluster.Spec.Network.IsMultus() {
_, isPublic := cluster.Spec.Network.Selectors[config.PublicNetworkSelectorKeyName]
_, isCluster := cluster.Spec.Network.Selectors[config.ClusterNetworkSelectorKeyName]
if !isPublic && !isCluster {
return errors.New("both network selector values for public and cluster selector cannot be empty for multus provider")
}
for _, selector := range config.NetworkSelectors {
// If one selector is empty, we continue
// This means a single interface is used both public and cluster network
if _, ok := cluster.Spec.Network.Selectors[selector]; !ok {
continue
}
// Get network attachment definition
_, err := c.context.NetworkClient.NetworkAttachmentDefinitions(cluster.Namespace).Get(cluster.Spec.Network.Selectors[selector], metav1.GetOptions{})
if err != nil {
if kerrors.IsNotFound(err) {
return errors.Wrapf(err, "specified network attachment definition for selector %q does not exist", selector)
}
return errors.Wrapf(err, "failed to fetch network attachment definition for selector %q", selector)
}
}
}
logger.Debug("cluster spec successfully validated")
return nil
}
// postMonStartupActions is a collection of actions to run once the monitors are up and running
// It gets executed right after the main mon Start() method
// Basically, it is executed between the monitors and the manager sequence
func (c *cluster) postMonStartupActions() error {
// Create CSI Kubernetes Secrets
err := csi.CreateCSISecrets(c.context, c.Namespace, &c.ownerRef)
if err != nil {
return errors.Wrapf(err, "failed to create csi kubernetes secrets")
}
// Create crash collector Kubernetes Secret
err = crash.CreateCrashCollectorSecret(c.context, c.Namespace, &c.ownerRef)
if err != nil {
return errors.Wrapf(err, "failed to create crash collector kubernetes secret")
}
// Enable Ceph messenger 2 protocol on Nautilus
if err := client.EnableMessenger2(c.context, c.Namespace); err != nil {
return errors.Wrapf(err, "failed to enable Ceph messenger version 2.")
}
return nil
}
@@ -0,0 +1,221 @@
/*
Copyright 2020 The Rook Authors. All rights reserved.
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
*/
// Package cluster to manage a Ceph cluster.
package cluster
import (
"time"
"github.com/pkg/errors"
cephv1 "github.com/rook/rook/pkg/apis/ceph.rook.io/v1"
"github.com/rook/rook/pkg/clusterd"
"github.com/rook/rook/pkg/daemon/ceph/client"
cephconfig "github.com/rook/rook/pkg/daemon/ceph/config"
"github.com/rook/rook/pkg/operator/ceph/cluster/crash"
"github.com/rook/rook/pkg/operator/ceph/cluster/mon"
"github.com/rook/rook/pkg/operator/ceph/config"
"github.com/rook/rook/pkg/operator/ceph/csi"
"github.com/rook/rook/pkg/operator/k8sutil"
v1 "k8s.io/api/core/v1"
kerrors "k8s.io/apimachinery/pkg/api/errors"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"k8s.io/client-go/kubernetes"
)
func (c *ClusterController) configureExternalCephCluster(cluster *cluster) error {
// Make sure the spec contains all the information we need
err := validateExternalClusterSpec(cluster)
if err != nil {
return errors.Wrap(err, "failed to validate external cluster specs")
}
config.ConditionExport(c.context, c.namespacedName, cephv1.ConditionConnecting, v1.ConditionTrue, "ClusterConnecting", "Cluster is connecting")
// loop until we find the secret necessary to connect to the external cluster
// then populate clusterInfo
cluster.Info = populateExternalClusterInfo(c.context, c.namespacedName.Namespace)
// If the user to check the ceph health and status is not the admin,
// we validate that ExternalCred has been populated correctly,
// then we check if the key (whether admin or not) is encoded in base64
if !isExternalHealthCheckUserAdmin(cluster.Info.AdminSecret) {
if !cluster.Info.IsInitializedExternalCred(true) {
return errors.New("invalid user health checker credentials")
}
if !cephconfig.IsKeyringBase64Encoded(cluster.Info.ExternalCred.Secret) {
return errors.Errorf("invalid user health checker key %q", cluster.Info.ExternalCred.Username)
}
} else {
// If the client.admin is used
if !cephconfig.IsKeyringBase64Encoded(cluster.Info.AdminSecret) {
return errors.Errorf("invalid user health checker key %q", client.AdminUsername)
}
}
// Write connection info (ceph config file and keyring) for ceph commands
if cluster.Spec.CephVersion.Image == "" {
err = mon.WriteConnectionConfig(c.context, cluster.Info)
if err != nil {
logger.Errorf("failed to write config. attempting to continue. %v", err)
}
}
// Validate versions (local and external)
// If no image is specified we don't perform any checks
if cluster.Spec.CephVersion.Image != "" {
_, _, err = c.detectAndValidateCephVersion(cluster)
if err != nil {
return errors.Wrap(err, "failed to detect and validate ceph version")
}
// Write the rook-config-override configmap (used by various daemons to apply config overrides)
// If we don't do this, daemons will never start, waiting forever for this configmap to be present
//
// Only do this when doing a bit of management...
logger.Info("creating 'rook-ceph-config' configmap.")
err = populateConfigOverrideConfigMap(cluster.context, c.namespacedName.Namespace, cluster.ownerRef)
if err != nil {
return errors.Wrapf(err, "failed to populate config override config map")
}
}
// The cluster Identity must be established at this point
if !cluster.Info.IsInitialized(true) {
return errors.New("the cluster identity was not established")
}
logger.Info("external cluster identity established")
// Create CSI Secrets only if the user has provided the admin key
if cluster.Info.AdminSecret != mon.AdminSecretName {
err = csi.CreateCSISecrets(c.context, c.namespacedName.Namespace, &cluster.ownerRef)
if err != nil {
return errors.Wrap(err, "failed to create csi kubernetes secrets")
}
}
// Create CSI config map
err = csi.CreateCsiConfigMap(c.namespacedName.Namespace, c.context.Clientset, &cluster.ownerRef)
if err != nil {
return errors.Wrap(err, "failed to create csi config map")
}
// Save CSI configmap
err = csi.SaveClusterConfig(c.context.Clientset, c.namespacedName.Namespace, cluster.Info, c.csiConfigMutex)
if err != nil {
return errors.Wrap(err, "failed to update csi cluster config")
}
logger.Info("successfully updated csi config map")
// Create Crash Collector Secret
// In 14.2.5 the crash daemon will read the client.crash key instead of the admin key
if !cluster.Spec.CrashCollector.Disable {
err = crash.CreateCrashCollectorSecret(c.context, c.namespacedName.Namespace, &cluster.ownerRef)
if err != nil {
return errors.Wrap(err, "failed to create crash collector kubernetes secret")
}
}
// Everything went well so let's update the CR's status to "connected"
config.ConditionExport(c.context, c.namespacedName, cephv1.ConditionConnected, v1.ConditionTrue, "ClusterConnected", "Cluster connected successfully")
// Mark initialization has done
cluster.initCompleted = true
return nil
}
func purgeExternalCluster(clientset kubernetes.Interface, namespace string) error {
// Purge the config maps
cmsToDelete := []string{
mon.EndpointConfigMapName,
config.StoreName,
k8sutil.ConfigOverrideName,
}
for _, cm := range cmsToDelete {
err := clientset.CoreV1().ConfigMaps(namespace).Delete(cm, &metav1.DeleteOptions{})
if err != nil && !kerrors.IsNotFound(err) {
logger.Errorf("failed to delete config map %+v. %v", cm, err)
}
}
// Purge the secrets
secretsToDelete := []string{
mon.AppName,
mon.OperatorCreds,
csi.CsiRBDNodeSecret,
csi.CsiRBDProvisionerSecret,
csi.CsiCephFSNodeSecret,
csi.CsiCephFSProvisionerSecret,
}
for _, secret := range secretsToDelete {
err := clientset.CoreV1().Secrets(namespace).Delete(secret, &metav1.DeleteOptions{})
if err != nil && !kerrors.IsNotFound(err) {
logger.Errorf("failed to delete config map %+v. %v", secret, err)
}
}
return nil
}
func validateExternalClusterSpec(cluster *cluster) error {
if cluster.Spec.CephVersion.Image != "" {
if cluster.Spec.DataDirHostPath == "" {
return errors.New("dataDirHostPath must be specified")
}
}
return nil
}
// Add validation in the code to fail if the external cluster has no OSDs keep waiting
func populateExternalClusterInfo(context *clusterd.Context, namespace string) *cephconfig.ClusterInfo {
var clusterInfo *cephconfig.ClusterInfo
for {
var err error
clusterInfo, _, _, err = mon.LoadClusterInfo(context, namespace)
if err != nil {
logger.Warningf("waiting for the connection info of the external cluster. retrying in %s.", externalConnectionRetry.String())
time.Sleep(externalConnectionRetry)
continue
} else {
// If an admin key was provided we don't need to load the other resources
// Some people might want to give the admin key
// The necessary users/keys/secrets will be created by Rook
// This is also done to allow backward compatibility
if isExternalHealthCheckUserAdmin(clusterInfo.AdminSecret) {
break
}
externalCred, err := mon.ValidateAndLoadExternalClusterSecrets(context, namespace)
if err != nil {
logger.Warningf("waiting for the connection info of the external cluster. retrying in %s.", externalConnectionRetry.String())
logger.Debugf("%v", err)
time.Sleep(externalConnectionRetry)
continue
} else {
clusterInfo.ExternalCred = externalCred
logger.Infof("found the cluster info to connect to the external cluster. will use %q to check health and monitor status. mons=%+v", clusterInfo.ExternalCred.Username, clusterInfo.Monitors)
break
}
}
}
return clusterInfo
}
func isExternalHealthCheckUserAdmin(adminSecret string) bool {
return adminSecret != mon.AdminSecretName
}
@@ -1,5 +1,5 @@
/*
Copyright 2018 The Rook Authors. All rights reserved.
Copyright 2020 The Rook Authors. All rights reserved.
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
@@ -14,22 +14,26 @@ See the License for the specific language governing permissions and
limitations under the License.
*/
// Package cluster to manage a Ceph cluster.
package cluster
import (
"github.com/pkg/errors"
"testing"
cephv1 "github.com/rook/rook/pkg/apis/ceph.rook.io/v1"
"github.com/rook/rook/pkg/operator/ceph/cluster/mon"
"github.com/stretchr/testify/assert"
)
func getClusterObject(obj interface{}) (cluster *cephv1.CephCluster, err error) {
var ok bool
cluster, ok = obj.(*cephv1.CephCluster)
if ok {
// the cluster object is of the latest type, simply return it
cluster = cluster.DeepCopy()
return cluster, nil
}
func TestValidateExternalClusterSpec(t *testing.T) {
c := &cluster{Spec: &cephv1.ClusterSpec{}, mons: &mon.Cluster{}}
err := validateExternalClusterSpec(c)
assert.NoError(t, err)
return nil, errors.Errorf("not a known cluster object: %+v", obj)
c.Spec.CephVersion.Image = "ceph/ceph:v14.2.9"
err = validateExternalClusterSpec(c)
assert.Error(t, err)
c.Spec.DataDirHostPath = "path"
err = validateExternalClusterSpec(c)
assert.NoError(t, err, err)
}
File diff suppressed because it is too large Load Diff
@@ -21,14 +21,10 @@ import (
"testing"
"time"
"github.com/pkg/errors"
cephv1 "github.com/rook/rook/pkg/apis/ceph.rook.io/v1"
rookv1 "github.com/rook/rook/pkg/apis/rook.io/v1"
rookalpha "github.com/rook/rook/pkg/apis/rook.io/v1alpha2"
rookfake "github.com/rook/rook/pkg/client/clientset/versioned/fake"
"github.com/rook/rook/pkg/clusterd"
"github.com/rook/rook/pkg/daemon/ceph/agent/flexvolume/attachment"
"github.com/rook/rook/pkg/operator/ceph/cluster/mon"
"github.com/rook/rook/pkg/operator/k8sutil"
testop "github.com/rook/rook/pkg/operator/test"
"github.com/stretchr/testify/assert"
@@ -146,121 +142,3 @@ func TestClusterDeleteFlexDisabled(t *testing.T) {
// Ensure that the listing of volume attachments was never called.
assert.Equal(t, 0, listCount)
}
func TestClusterChanged(t *testing.T) {
// a new node added, should be a change
old := cephv1.ClusterSpec{
Storage: rookv1.StorageScopeSpec{
Nodes: []rookv1.Node{
{Name: "node1", Selection: rookv1.Selection{Devices: []rookv1.Device{{Name: "sda"}}}},
},
},
}
new := cephv1.ClusterSpec{
Storage: rookv1.StorageScopeSpec{
Nodes: []rookv1.Node{
{Name: "node1", Selection: rookv1.Selection{Devices: []rookv1.Device{{Name: "sda"}}}},
{Name: "node2", Selection: rookv1.Selection{Devices: []rookv1.Device{{Name: "sda"}}}},
},
},
}
c := &cluster{Spec: &cephv1.ClusterSpec{}, mons: &mon.Cluster{}}
changed, diff := clusterChanged(old, new, c)
assert.True(t, changed)
assert.NotEqual(t, diff, "")
assert.Equal(t, 0, c.Spec.Mon.Count)
// a node was removed, should be a change
old.Storage.Nodes = []rookv1.Node{
{Name: "node1", Selection: rookv1.Selection{Devices: []rookv1.Device{{Name: "sda"}}}},
{Name: "node2", Selection: rookv1.Selection{Devices: []rookv1.Device{{Name: "sda"}}}},
}
new.Storage.Nodes = []rookv1.Node{
{Name: "node1", Selection: rookv1.Selection{Devices: []rookv1.Device{{Name: "sda"}}}},
}
changed, diff = clusterChanged(old, new, c)
assert.True(t, changed)
assert.NotEqual(t, diff, "")
// the nodes being in a different order should not be a change
old.Storage.Nodes = []rookv1.Node{
{Name: "node1", Selection: rookv1.Selection{Devices: []rookv1.Device{{Name: "sda"}}}},
{Name: "node2", Selection: rookv1.Selection{Devices: []rookv1.Device{{Name: "sda"}}}},
}
new.Storage.Nodes = []rookv1.Node{
{Name: "node2", Selection: rookv1.Selection{Devices: []rookv1.Device{{Name: "sda"}}}},
{Name: "node1", Selection: rookv1.Selection{Devices: []rookv1.Device{{Name: "sda"}}}},
}
changed, diff = clusterChanged(old, new, c)
assert.False(t, changed)
assert.Equal(t, 0, c.Spec.Mon.Count)
assert.Equal(t, "", diff)
// If the number of mons changes, the cluster would be updated
new.Mon.Count = 3
new.Mon.AllowMultiplePerNode = true
changed, diff = clusterChanged(old, new, c)
assert.True(t, changed)
assert.NotEqual(t, diff, "")
}
func TestRemoveFinalizer(t *testing.T) {
clientset := testop.New(t, 3)
context := &clusterd.Context{
Clientset: clientset,
RookClientset: rookfake.NewSimpleClientset(),
}
operatorConfigCallbacks := []func() error{
func() error {
logger.Infof("test success callback")
return nil
},
}
addCallbacks := []func() error{
func() error {
logger.Infof("test success callback")
return errors.New("test failed callback")
},
}
controller := NewClusterController(context, "", &attachment.MockAttachment{}, operatorConfigCallbacks, addCallbacks)
// *****************************************
// start with a current version ceph cluster
// *****************************************
cluster := &cephv1.CephCluster{
ObjectMeta: metav1.ObjectMeta{
Name: "cluster-1893",
Namespace: "namespace-6551",
Finalizers: []string{finalizerName},
},
}
// create the cluster initially so it exists in the k8s api
cluster, err := context.RookClientset.CephV1().CephClusters(cluster.Namespace).Create(cluster)
assert.NoError(t, err)
assert.Len(t, cluster.Finalizers, 1)
// remove the finalizer from the cluster object
controller.removeFinalizer(cluster)
// verify the finalizer was removed
cluster, err = context.RookClientset.CephV1().CephClusters(cluster.Namespace).Get(cluster.Name, metav1.GetOptions{})
assert.NoError(t, err)
assert.NotNil(t, cluster)
assert.Len(t, cluster.Finalizers, 0)
}
func TestValidateExternalClusterSpec(t *testing.T) {
c := &cluster{Spec: &cephv1.ClusterSpec{}, mons: &mon.Cluster{}}
err := validateExternalClusterSpec(c)
assert.NoError(t, err)
c.Spec.CephVersion.Image = "ceph/ceph:v14.2.9"
err = validateExternalClusterSpec(c)
assert.Error(t, err)
c.Spec.DataDirHostPath = "path"
err = validateExternalClusterSpec(c)
assert.NoError(t, err, err)
}
+12 -5
View File
@@ -46,17 +46,24 @@ const (
// Add adds a new Controller based on nodedrain.ReconcileNode and registers the relevant watches and handlers
func Add(mgr manager.Manager, context *clusterd.Context) error {
reconcileNode := &ReconcileNode{
return add(mgr, newReconciler(mgr, context))
}
// newReconciler returns a new reconcile.Reconciler
func newReconciler(mgr manager.Manager, context *clusterd.Context) reconcile.Reconciler {
return &ReconcileNode{
client: mgr.GetClient(),
scheme: mgr.GetScheme(),
}
reconciler := reconcile.Reconciler(reconcileNode)
}
func add(mgr manager.Manager, r reconcile.Reconciler) error {
// Create a new controller
c, err := controller.New(controllerName, mgr, controller.Options{Reconciler: reconciler})
c, err := controller.New(controllerName, mgr, controller.Options{Reconciler: r})
if err != nil {
return errors.Wrapf(err, "failed to create a new %q", controllerName)
}
logger.Info("successfully started")
// Watch for changes to the nodes
specChangePredicate := predicate.Funcs{
@@ -75,7 +82,7 @@ func Add(mgr manager.Manager, context *clusterd.Context) error {
logger.Debugf("watch for changes to the nodes")
err = c.Watch(&source.Kind{Type: &corev1.Node{}}, &handler.EnqueueRequestForObject{}, specChangePredicate)
if err != nil {
return errors.Wrapf(err, "failed to watch for node changes")
return errors.Wrap(err, "failed to watch for node changes")
}
// Watch for changes to the ceph-crash deployments
@@ -103,7 +110,7 @@ func Add(mgr manager.Manager, context *clusterd.Context) error {
},
)
if err != nil {
return errors.Wrapf(err, "failed to watch for changes on the ceph-crash deployment")
return errors.Wrap(err, "failed to watch for changes on the ceph-crash deployment")
}
// Watch for changes to the ceph pod nodename and enqueue their nodes
+6 -15
View File
@@ -18,7 +18,6 @@ package crash
import (
"context"
"time"
"github.com/pkg/errors"
"github.com/rook/rook/pkg/operator/ceph/cluster/mgr"
@@ -47,11 +46,6 @@ import (
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
)
const (
getVersionRetryInterval = 5
getVersionMaxRetries = 60
)
var (
logger = capnslog.NewPackageLogger("github.com/rook/rook", controllerName)
// Implement reconcile.Reconciler so the controller can reconcile objects
@@ -79,7 +73,6 @@ func (r *ReconcileNode) Reconcile(request reconcile.Request) (reconcile.Result,
}
func (r *ReconcileNode) reconcile(request reconcile.Request) (reconcile.Result, error) {
logger.Debugf("reconciling node: %q", request.Name)
// get the node object
@@ -221,15 +214,13 @@ func (r *ReconcileNode) cephPodList() ([]corev1.Pod, error) {
// getImageVersion returns the CephVersion registered for a specified image (if any) and whether any image was found.
func getImageVersion(cephCluster cephv1.CephCluster) (*version.CephVersion, error) {
for i := 0; i < getVersionMaxRetries; i++ {
// If the Ceph cluster has not yet recorded the image and version for the current image in its spec, then the Crash
// controller should wait for the version to be detected.
if cephCluster.Status.CephVersion != nil && cephCluster.Spec.CephVersion.Image == cephCluster.Status.CephVersion.Image {
logger.Debugf("ceph version found %q", cephCluster.Status.CephVersion.Version)
return controller.ExtractCephVersionFromLabel(cephCluster.Status.CephVersion.Version)
}
<-time.After(time.Second * getVersionRetryInterval)
// If the Ceph cluster has not yet recorded the image and version for the current image in its spec, then the Crash
// controller should wait for the version to be detected.
if cephCluster.Status.CephVersion != nil && cephCluster.Spec.CephVersion.Image == cephCluster.Status.CephVersion.Image {
logger.Debugf("ceph version found %q", cephCluster.Status.CephVersion.Version)
return controller.ExtractCephVersionFromLabel(cephCluster.Status.CephVersion.Version)
}
return nil, errors.New("attempt to determine ceph version for the current cluster image timed out")
}
-6
View File
@@ -21,7 +21,6 @@ import (
"github.com/rook/rook/pkg/operator/ceph/config"
"github.com/rook/rook/pkg/operator/ceph/config/keyring"
apps "k8s.io/api/apps/v1"
"k8s.io/apimachinery/pkg/api/errors"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
)
@@ -79,8 +78,3 @@ func (c *Cluster) generateKeyring(m *mgrConfig) (string, error) {
keyring := fmt.Sprintf(keyringTemplate, m.DaemonID, key)
return keyring, s.CreateOrUpdate(m.ResourceName, keyring)
}
func (c *Cluster) associateKeyring(existingKeyring string, d *apps.Deployment) error {
s := keyring.GetSecretStoreForDeployment(c.context, d)
return s.CreateOrUpdate(d.GetName(), existingKeyring)
}
+3 -9
View File
@@ -159,8 +159,9 @@ func (c *Cluster) Start() error {
DataPathMap: config.NewStatelessDaemonDataPathMap(config.MgrType, daemonID, c.Namespace, c.dataDirHostPath),
}
// generate keyring specific to this mgr daemon saved to k8s secret
keyring, err := c.generateKeyring(mgrConfig)
// We set the owner reference of the Secret to the Object controller instead of the replicaset
// because we watch for that resource and reconcile if anything happens to it
_, err := c.generateKeyring(mgrConfig)
if err != nil {
return errors.Wrapf(err, "failed to generate keyring for %q", resourceName)
}
@@ -185,13 +186,6 @@ func (c *Cluster) Start() error {
logger.Errorf("failed to update mgr deployment %q. %v", resourceName, err)
}
}
if existingDeployment, err := c.context.Clientset.AppsV1().Deployments(c.Namespace).Get(d.GetName(), metav1.GetOptions{}); err != nil {
logger.Warningf("failed to find mgr deployment %q for keyring association. %v", resourceName, err)
} else {
if err = c.associateKeyring(keyring, existingDeployment); err != nil {
logger.Warningf("failed to associate keyring with mgr deployment %q. %v", resourceName, err)
}
}
}
if err := c.configureDashboardService(); err != nil {
+1 -1
View File
@@ -127,7 +127,7 @@ func CreateOrLoadClusterInfo(context *clusterd.Context, namespace string, ownerR
func ValidateAndLoadExternalClusterSecrets(context *clusterd.Context, namespace string) (cephconfig.ExternalCred, error) {
var externalCred cephconfig.ExternalCred
secret, err := context.Clientset.CoreV1().Secrets(namespace).Get(operatorCreds, metav1.GetOptions{})
secret, err := context.Clientset.CoreV1().Secrets(namespace).Get(OperatorCreds, metav1.GetOptions{})
if err != nil {
if !kerrors.IsNotFound(err) {
return externalCred, errors.Wrap(err, "failed to get external user secret")
+3 -2
View File
@@ -59,8 +59,9 @@ const (
MappingKey = "mapping"
// AppName is the name of the secret storing cluster mon.admin key, fsid and name
AppName = "rook-ceph-mon"
operatorCreds = "rook-ceph-operator-creds"
AppName = "rook-ceph-mon"
// OperatorCreds is the name of the secret
OperatorCreds = "rook-ceph-operator-creds"
monNodeAttr = "mon_node"
monClusterAttr = "mon_cluster"
tprName = "mon.rook.io"
+4 -4
View File
@@ -174,7 +174,7 @@ func TestOperatorRestart(t *testing.T) {
// start a basic cluster
info, err := c.Start(c.ClusterInfo, c.rookVersion, cephver.Nautilus, c.spec)
assert.Nil(t, err)
assert.True(t, info.IsInitialized())
assert.True(t, info.IsInitialized(true))
validateStart(t, c)
@@ -183,7 +183,7 @@ func TestOperatorRestart(t *testing.T) {
// starting again should be a no-op, but will not result in an error
info, err = c.Start(c.ClusterInfo, c.rookVersion, cephver.Nautilus, c.spec)
assert.Nil(t, err)
assert.True(t, info.IsInitialized())
assert.True(t, info.IsInitialized(true))
validateStart(t, c)
}
@@ -202,7 +202,7 @@ func TestOperatorRestartHostNetwork(t *testing.T) {
// start a basic cluster
info, err := c.Start(c.ClusterInfo, c.rookVersion, cephver.Nautilus, c.spec)
assert.Nil(t, err)
assert.True(t, info.IsInitialized())
assert.True(t, info.IsInitialized(true))
validateStart(t, c)
@@ -212,7 +212,7 @@ func TestOperatorRestartHostNetwork(t *testing.T) {
// starting again should be a no-op, but still results in an error
info, err = c.Start(c.ClusterInfo, c.rookVersion, cephver.Nautilus, c.spec)
assert.Nil(t, err)
assert.True(t, info.IsInitialized(), info)
assert.True(t, info.IsInitialized(true), info)
validateStart(t, c)
}
@@ -0,0 +1,66 @@
/*
Copyright 2020 The Rook Authors. All rights reserved.
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
*/
// Package cluster to manage a Ceph cluster.
package cluster
import (
"os"
"reflect"
opcontroller "github.com/rook/rook/pkg/operator/ceph/controller"
"github.com/rook/rook/pkg/operator/k8sutil"
v1 "k8s.io/api/core/v1"
)
// StartOperatorSettingsWatch starts the operator settings watcher
func (c *ClusterController) StartOperatorSettingsWatch(namespace string, stopCh chan struct{}) {
operatorNamespace := os.Getenv(k8sutil.PodNamespaceEnvVar)
// watch for "rook-ceph-operator-config" ConfigMap
k8sutil.StartOperatorSettingsWatch(c.context, operatorNamespace, opcontroller.OperatorSettingConfigMapName,
c.operatorConfigChange,
func(oldObj, newObj interface{}) {
if reflect.DeepEqual(oldObj, newObj) {
return
}
c.operatorConfigChange(newObj)
return
}, nil, stopCh)
}
// StopWatch stop watchers
func (c *ClusterController) StopWatch() {
for _, cluster := range c.clusterMap {
close(cluster.stopCh)
}
c.clusterMap = make(map[string]*cluster)
}
func (c *ClusterController) operatorConfigChange(obj interface{}) {
cm, ok := obj.(*v1.ConfigMap)
if !ok {
logger.Warningf("Expected ConfigMap but handler received %T. %#v", obj, obj)
return
}
logger.Infof("ConfigMap %q changes detected. Updating configurations", cm.Name)
for _, callback := range c.operatorConfigCallbacks {
if err := callback(); err != nil {
logger.Errorf("%v", err)
}
}
return
}
-6
View File
@@ -21,7 +21,6 @@ import (
"strconv"
"github.com/rook/rook/pkg/operator/ceph/config/keyring"
apps "k8s.io/api/apps/v1"
)
const (
@@ -48,8 +47,3 @@ func (c *Cluster) generateKeyring(osdID int) (string, error) {
keyring := fmt.Sprintf(keyringTemplate, osdIDStr, key)
return keyring, s.CreateOrUpdate(deploymentName, keyring)
}
func (c *Cluster) associateKeyring(existingKeyring string, d *apps.Deployment) error {
s := keyring.GetSecretStoreForDeployment(c.context, d)
return s.CreateOrUpdate(d.GetName(), existingKeyring)
}
+17 -35
View File
@@ -449,7 +449,7 @@ func (c *Cluster) startOSDDaemonsOnPVC(pvcName string, config *provisionConfig,
// keyring must be generated before deployment creation in order to avoid a race condition resulting
// in intermittent failure of first-attempt OSD pods.
keyring, err := c.generateKeyring(osd.ID)
_, err := c.generateKeyring(osd.ID)
if err != nil {
errMsg := fmt.Sprintf("failed to create keyring for pvc %q, osd %v. %v", osdProps.crushHostname, osd, err)
config.addError(errMsg)
@@ -471,24 +471,18 @@ func (c *Cluster) startOSDDaemonsOnPVC(pvcName string, config *provisionConfig,
continue
}
createdDeployment, createErr := c.context.Clientset.AppsV1().Deployments(c.Namespace).Create(dp)
_, createErr := c.context.Clientset.AppsV1().Deployments(c.Namespace).Create(dp)
if createErr != nil {
if !kerrors.IsAlreadyExists(createErr) {
if kerrors.IsAlreadyExists(createErr) {
logger.Infof("deployment for osd %d already exists. updating if needed", osd.ID)
if err = updateDeploymentAndWait(c.context, dp, c.Namespace, opconfig.OsdType, strconv.Itoa(osd.ID), c.skipUpgradeChecks, c.continueUpgradeAfterChecksEvenIfNotHealthy); err != nil {
logger.Errorf("failed to update osd deployment %d. %v", osd.ID, err)
}
} else {
// we failed to create job, update the orchestration status for this pvc
logger.Warningf("failed to create osd deployment for pvc %q, osd %v. %v", osdProps.pvc.ClaimName, osd, createErr)
continue
}
logger.Infof("deployment for osd %d already exists. updating if needed", osd.ID)
createdDeployment, err = c.context.Clientset.AppsV1().Deployments(c.Namespace).Get(dp.Name, metav1.GetOptions{})
if err != nil {
logger.Warningf("failed to get existing OSD deployment %q for update. %v", dp.Name, err)
continue
}
}
err = c.associateKeyring(keyring, createdDeployment)
if err != nil {
logger.Errorf("failed to associate keyring for pvc %q, osd %v. %v", osdProps.pvc.ClaimName, osd, err)
}
if createErr != nil && kerrors.IsAlreadyExists(createErr) {
@@ -529,7 +523,7 @@ func (c *Cluster) startOSDDaemonsOnNode(nodeName string, config *provisionConfig
// keyring must be generated before deployment creation in order to avoid a race condition resulting
// in intermittent failure of first-attempt OSD pods.
keyring, err := c.generateKeyring(osd.ID)
_, err := c.generateKeyring(osd.ID)
if err != nil {
errMsg := fmt.Sprintf("failed to create keyring for node %q, osd %v. %v", n.Name, osd, err)
config.addError(errMsg)
@@ -551,30 +545,18 @@ func (c *Cluster) startOSDDaemonsOnNode(nodeName string, config *provisionConfig
continue
}
createdDeployment, createErr := c.context.Clientset.AppsV1().Deployments(c.Namespace).Create(dp)
_, createErr := c.context.Clientset.AppsV1().Deployments(c.Namespace).Create(dp)
if createErr != nil {
if !kerrors.IsAlreadyExists(createErr) {
// we failed to create job, update the orchestration status for this node
if kerrors.IsAlreadyExists(createErr) {
logger.Infof("deployment for osd %d already exists. updating if needed", osd.ID)
if err = updateDeploymentAndWait(c.context, dp, c.Namespace, opconfig.OsdType, strconv.Itoa(osd.ID), c.skipUpgradeChecks, c.continueUpgradeAfterChecksEvenIfNotHealthy); err != nil {
logger.Errorf("failed to update osd deployment %d. %v", osd.ID, err)
}
} else {
// we failed to create job, update the orchestration status for this pvc
logger.Warningf("failed to create osd deployment for node %q, osd %+v. %v", n.Name, osd, createErr)
continue
}
logger.Infof("deployment for osd %d already exists. updating if needed", osd.ID)
createdDeployment, err = c.context.Clientset.AppsV1().Deployments(c.Namespace).Get(dp.Name, metav1.GetOptions{})
if err != nil {
logger.Warningf("failed to get existing OSD deployment %q for update. %v", dp.Name, err)
continue
}
}
err = c.associateKeyring(keyring, createdDeployment)
if err != nil {
logger.Errorf("failed to associate keyring for node %q, osd %v. %v", n.Name, osd, err)
}
if createErr != nil && kerrors.IsAlreadyExists(createErr) {
if err = updateDeploymentAndWait(c.context, dp, c.Namespace, opconfig.OsdType, strconv.Itoa(osd.ID), c.skipUpgradeChecks, c.continueUpgradeAfterChecksEvenIfNotHealthy); err != nil {
logger.Errorf("failed to update osd deployment %d. %v", osd.ID, err)
}
}
logger.Infof("started deployment for osd %d", osd.ID)
}
+102
View File
@@ -0,0 +1,102 @@
/*
Copyright 2020 The Rook Authors. All rights reserved.
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
*/
// Package cluster to manage a Ceph cluster.
package cluster
import (
"github.com/rook/rook/pkg/clusterd"
discoverDaemon "github.com/rook/rook/pkg/daemon/discover"
"github.com/rook/rook/pkg/operator/k8sutil"
corev1 "k8s.io/api/core/v1"
"k8s.io/apimachinery/pkg/runtime"
"sigs.k8s.io/controller-runtime/pkg/client"
"sigs.k8s.io/controller-runtime/pkg/event"
"sigs.k8s.io/controller-runtime/pkg/predicate"
)
// predicateForNodeWatcher is the predicate function to trigger reconcile on Node events
func predicateForNodeWatcher(client client.Client, context *clusterd.Context) predicate.Funcs {
return predicate.Funcs{
CreateFunc: func(e event.CreateEvent) bool {
clientCluster := newClientCluster(client, e.Meta.GetNamespace(), context)
return clientCluster.onK8sNode(e.Object)
},
UpdateFunc: func(e event.UpdateEvent) bool {
clientCluster := newClientCluster(client, e.MetaNew.GetNamespace(), context)
return clientCluster.onK8sNode(e.ObjectNew)
},
DeleteFunc: func(e event.DeleteEvent) bool {
return false
},
GenericFunc: func(e event.GenericEvent) bool {
return false
},
}
}
// predicateForHotPlugCMWatcher is the predicate function to trigger reconcile on ConfigMap events (hot-plug)
func predicateForHotPlugCMWatcher(client client.Client) predicate.Funcs {
return predicate.Funcs{
UpdateFunc: func(e event.UpdateEvent) bool {
isHotPlugCM := isHotPlugCM(e.ObjectNew)
if !isHotPlugCM {
logger.Debugf("hot-plug cm watcher: only reconcile on hot plug cm changes, this %q cm is handled by another watcher", e.MetaNew.GetName())
return false
}
clientCluster := newClientCluster(client, e.MetaNew.GetNamespace(), &clusterd.Context{})
return clientCluster.onDeviceCMUpdate(e.ObjectOld, e.ObjectNew)
},
DeleteFunc: func(e event.DeleteEvent) bool {
// TODO: if the configmap goes away we could retrigger rook-discover DS
// However at this point the returned bool can only trigger a reconcile of the CephCluster object
// Definitely non-trivial but nice to have in the future
return false
},
CreateFunc: func(e event.CreateEvent) bool {
return false
},
GenericFunc: func(e event.GenericEvent) bool {
return false
},
}
}
// isHotPlugCM informs whether the object is the cm for hot-plug disk
func isHotPlugCM(obj runtime.Object) bool {
// If not a ConfigMap, let's not reconcile
cm, ok := obj.(*corev1.ConfigMap)
if !ok {
return false
}
// Get the labels
labels := cm.GetLabels()
labelVal, labelKeyExist := labels[k8sutil.AppAttr]
if labelKeyExist && labelVal == discoverDaemon.AppName {
return true
}
return false
}
@@ -1,5 +1,5 @@
/*
Copyright 2018 The Rook Authors. All rights reserved.
Copyright 2020 The Rook Authors. All rights reserved.
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
@@ -21,16 +21,26 @@ import (
cephv1 "github.com/rook/rook/pkg/apis/ceph.rook.io/v1"
"github.com/stretchr/testify/assert"
corev1 "k8s.io/api/core/v1"
)
func TestGetClusterObject(t *testing.T) {
// get a current version cluster object, should return with no error and no migration needed
cluster, err := getClusterObject(&cephv1.CephCluster{})
assert.NotNil(t, cluster)
assert.Nil(t, err)
func TestIsHotPlugCM(t *testing.T) {
dum := &cephv1.CephBlockPool{}
// try to get an object that isn't a cluster, should return with an error
cluster, err = getClusterObject(&map[string]string{})
assert.Nil(t, cluster)
assert.NotNil(t, err)
b := isHotPlugCM(dum)
assert.False(t, b)
cm := &corev1.ConfigMap{}
b = isHotPlugCM(cm)
assert.False(t, b)
cm.Labels = map[string]string{
"foo": "bar",
}
b = isHotPlugCM(cm)
assert.False(t, b)
cm.Labels["app"] = "rook-discover"
b = isHotPlugCM(cm)
assert.True(t, b)
}
@@ -21,6 +21,11 @@ import (
"github.com/rook/rook/pkg/clusterd"
"github.com/rook/rook/pkg/operator/ceph/cluster/crash"
"github.com/rook/rook/pkg/operator/ceph/cluster/rbd"
"github.com/rook/rook/pkg/operator/ceph/disruption/clusterdisruption"
"github.com/rook/rook/pkg/operator/ceph/disruption/controllerconfig"
"github.com/rook/rook/pkg/operator/ceph/disruption/machinedisruption"
"github.com/rook/rook/pkg/operator/ceph/disruption/machinelabel"
"github.com/rook/rook/pkg/operator/ceph/disruption/nodedrain"
"github.com/rook/rook/pkg/operator/ceph/file"
"github.com/rook/rook/pkg/operator/ceph/nfs"
"github.com/rook/rook/pkg/operator/ceph/object"
@@ -30,6 +35,23 @@ import (
"sigs.k8s.io/controller-runtime/pkg/manager"
)
var (
// EnableMachineDisruptionBudget checks whether machine disruption budget is enabled
EnableMachineDisruptionBudget bool
)
// AddToManagerFuncsMaintenance is a list of functions to add all Controllers to the Manager (entrypoint for controller)
var AddToManagerFuncsMaintenance = []func(manager.Manager, *controllerconfig.Context) error{
nodedrain.Add,
clusterdisruption.Add,
}
// MachineDisruptionBudgetAddToManagerFuncs is a list of fencing related functions to add all Controllers to the Manager (entrypoint for controller)
var MachineDisruptionBudgetAddToManagerFuncs = []func(manager.Manager, *controllerconfig.Context) error{
machinelabel.Add,
machinedisruption.Add,
}
// AddToManagerFuncs is a list of functions to add all Controllers to the Manager (entrypoint for controller)
var AddToManagerFuncs = []func(manager.Manager, *clusterd.Context) error{
crash.Add,
@@ -44,16 +66,38 @@ var AddToManagerFuncs = []func(manager.Manager, *clusterd.Context) error{
// AddToManager adds all the registered controllers to the passed manager.
// each controller package will have an Add method listed in AddToManagerFuncs
// which will setup all the necessary watch
func AddToManager(m manager.Manager, c *clusterd.Context) error {
func AddToManager(m manager.Manager, c *controllerconfig.Context, clusterController *ClusterController) error {
if c == nil {
return errors.New("nil context passed")
}
// Run CephCluster CR
if err := Add(m, c.ClusterdContext, clusterController); err != nil {
return err
}
// Add Ceph child CR controllers
for _, f := range AddToManagerFuncs {
if err := f(m, c.ClusterdContext); err != nil {
return err
}
}
// Add maintenance controllers
for _, f := range AddToManagerFuncsMaintenance {
if err := f(m, c); err != nil {
return err
}
}
// If machine disruption budget is enabled let's add the controllers
if EnableMachineDisruptionBudget {
for _, f := range MachineDisruptionBudgetAddToManagerFuncs {
if err := f(m, c); err != nil {
return err
}
}
}
return nil
}
+80
View File
@@ -0,0 +1,80 @@
/*
Copyright 2020 The Rook Authors. All rights reserved.
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
*/
// Package cluster to manage a Ceph cluster.
package cluster
import (
"github.com/pkg/errors"
"github.com/rook/rook/pkg/clusterd"
"github.com/rook/rook/pkg/operator/k8sutil"
v1 "k8s.io/api/core/v1"
kerrors "k8s.io/apimachinery/pkg/api/errors"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
)
func (c *cluster) setOrchestrationNeeded() {
c.orchMux.Lock()
c.orchestrationNeeded = true
c.orchMux.Unlock()
}
// unsetOrchestrationStatus resets the orchestrationRunning-flag
func (c *cluster) unsetOrchestrationStatus() {
c.orchMux.Lock()
defer c.orchMux.Unlock()
c.orchestrationRunning = false
}
// checkSetOrchestrationStatus is responsible to do orchestration as long as there is a request needed
func (c *cluster) checkSetOrchestrationStatus() bool {
c.orchMux.Lock()
defer c.orchMux.Unlock()
// check if there is an orchestration needed currently
if c.orchestrationNeeded == true && c.orchestrationRunning == false {
// there is an orchestration needed
// allow to enter the orchestration-loop
c.orchestrationNeeded = false
c.orchestrationRunning = true
return true
}
return false
}
// populateConfigOverrideConfigMap creates the "rook-config-override" config map
// Its content allows modifying Ceph configuration flags
func populateConfigOverrideConfigMap(context *clusterd.Context, namespace string, ownerRef metav1.OwnerReference) error {
placeholderConfig := map[string]string{
k8sutil.ConfigOverrideVal: "",
}
cm := &v1.ConfigMap{
ObjectMeta: metav1.ObjectMeta{
Name: k8sutil.ConfigOverrideName,
},
Data: placeholderConfig,
}
k8sutil.SetOwnerRef(&cm.ObjectMeta, &ownerRef)
_, err := context.Clientset.CoreV1().ConfigMaps(namespace).Create(cm)
if err != nil && !kerrors.IsAlreadyExists(err) {
return errors.Wrapf(err, "failed to create override configmap %s", namespace)
}
return nil
}
+237
View File
@@ -0,0 +1,237 @@
/*
Copyright 2020 The Rook Authors. All rights reserved.
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
*/
// Package cluster to manage a Ceph cluster.
package cluster
import (
"time"
"github.com/pkg/errors"
cephv1 "github.com/rook/rook/pkg/apis/ceph.rook.io/v1"
"github.com/rook/rook/pkg/daemon/ceph/client"
daemonclient "github.com/rook/rook/pkg/daemon/ceph/client"
"github.com/rook/rook/pkg/operator/ceph/cluster/mon"
"github.com/rook/rook/pkg/operator/ceph/controller"
cephver "github.com/rook/rook/pkg/operator/ceph/version"
"github.com/rook/rook/pkg/operator/k8sutil/cmdreporter"
)
func (c *ClusterController) detectAndValidateCephVersion(cluster *cluster) (*cephver.CephVersion, bool, error) {
version, err := cluster.detectCephVersion(c.rookImage, cluster.Spec.CephVersion.Image, detectCephVersionTimeout)
if err != nil {
return nil, false, err
}
logger.Info("validating ceph version from provided image")
if err := cluster.validateCephVersion(version); err != nil {
return nil, cluster.isUpgrade, err
}
// Update ceph version field in cluster object status
c.updateClusterCephVersion(cluster.Spec.CephVersion.Image, *version)
return version, cluster.isUpgrade, nil
}
func (c *cluster) printOverallCephVersion() {
versions, err := daemonclient.GetAllCephDaemonVersions(c.context, c.Namespace)
if err != nil {
logger.Errorf("failed to get ceph daemons versions. %v", err)
return
}
if len(versions.Overall) == 1 {
for v := range versions.Overall {
version, err := cephver.ExtractCephVersion(v)
if err != nil {
logger.Errorf("failed to extract ceph version. %v", err)
return
}
vv := *version
logger.Infof("successfully upgraded cluster to version: %q", vv.String())
}
} else {
// This shouldn't happen, but let's log just in case
logger.Warningf("upgrade orchestration completed but somehow we still have more than one Ceph version running. %v:", versions.Overall)
}
}
// This function compare the Ceph spec image and the cluster running version
// It returns true if the image is different and false if identical
func diffImageSpecAndClusterRunningVersion(imageSpecVersion cephver.CephVersion, runningVersions client.CephDaemonsVersions) (bool, error) {
numberOfCephVersions := len(runningVersions.Overall)
if numberOfCephVersions == 0 {
// let's return immediately
return false, errors.Errorf("no 'overall' section in the ceph versions. %+v", runningVersions.Overall)
}
if numberOfCephVersions > 1 {
// let's return immediately
logger.Warningf("it looks like we have more than one ceph version running. triggering upgrade. %+v:", runningVersions.Overall)
return true, nil
}
if numberOfCephVersions == 1 {
for v := range runningVersions.Overall {
version, err := cephver.ExtractCephVersion(v)
if err != nil {
logger.Errorf("failed to extract ceph version. %v", err)
return false, err
}
clusterRunningVersion := *version
// If this is the same version
if cephver.IsIdentical(clusterRunningVersion, imageSpecVersion) {
logger.Debugf("both cluster and image spec versions are identical, doing nothing %s", imageSpecVersion.String())
return false, nil
}
if cephver.IsSuperior(imageSpecVersion, clusterRunningVersion) {
logger.Infof("image spec version %s is higher than the running cluster version %s, upgrading", imageSpecVersion.String(), clusterRunningVersion.String())
return true, nil
}
if cephver.IsInferior(imageSpecVersion, clusterRunningVersion) {
return true, errors.Errorf("image spec version %s is lower than the running cluster version %s, downgrading is not supported", imageSpecVersion.String(), clusterRunningVersion.String())
}
}
}
return false, nil
}
// detectCephVersion loads the ceph version from the image and checks that it meets the version requirements to
// run in the cluster
func (c *cluster) detectCephVersion(rookImage, cephImage string, timeout time.Duration) (*cephver.CephVersion, error) {
logger.Infof("detecting the ceph image version for image %s...", cephImage)
versionReporter, err := cmdreporter.New(
c.context.Clientset, &c.ownerRef,
detectVersionName, detectVersionName, c.Namespace,
[]string{"ceph"}, []string{"--version"},
rookImage, cephImage)
if err != nil {
return nil, errors.Wrapf(err, "failed to set up ceph version job")
}
job := versionReporter.Job()
job.Spec.Template.Spec.ServiceAccountName = "rook-ceph-cmd-reporter"
// Apply the same node selector and tolerations for the ceph version detection as the mon daemons
cephv1.GetMonPlacement(c.Spec.Placement).ApplyToPodSpec(&job.Spec.Template.Spec)
stdout, stderr, retcode, err := versionReporter.Run(timeout)
if err != nil {
return nil, errors.Wrapf(err, "failed to complete ceph version job")
}
if retcode != 0 {
return nil, errors.Errorf(`ceph version job returned failure with retcode %d.
stdout: %s
stderr: %s`, retcode, stdout, stderr)
}
version, err := cephver.ExtractCephVersion(stdout)
if err != nil {
return nil, errors.Wrapf(err, "failed to extract ceph version")
}
logger.Infof("detected ceph image version: %q", version)
return version, nil
}
func (c *cluster) validateCephVersion(version *cephver.CephVersion) error {
if !c.Spec.External.Enable {
if !version.IsAtLeast(cephver.Minimum) {
return errors.Errorf("the version does not meet the minimum version %q", cephver.Minimum.String())
}
if !version.Supported() {
if !c.Spec.CephVersion.AllowUnsupported {
return errors.Errorf("allowUnsupported must be set to true to run with this version %q", version.String())
}
logger.Warningf("unsupported ceph version detected: %q, pursuing", version)
}
}
// The following tries to determine if the operator can proceed with an upgrade because we come from an OnAdd() call
// If the cluster was unhealthy and someone injected a new image version, an upgrade was triggered but failed because the cluster is not healthy
// Then after this, if the operator gets restarted we are not able to fail if the cluster is not healthy, the following tries to determine the
// state we are in and if we should upgrade or not
// Try to load clusterInfo so we can compare the running version with the one from the spec image
clusterInfo, _, _, err := mon.LoadClusterInfo(c.context, c.Namespace)
if err == nil {
// Write connection info (ceph config file and keyring) for ceph commands
err = mon.WriteConnectionConfig(c.context, clusterInfo)
if err != nil {
logger.Errorf("failed to write config. attempting to continue. %v", err)
}
}
if !clusterInfo.IsInitialized(false) {
// If not initialized, this is likely a new cluster so there is nothing to do
logger.Debug("cluster not initialized, nothing to validate")
return nil
}
if c.Spec.External.Enable && c.Spec.CephVersion.Image != "" {
c.Info.CephVersion, err = controller.ValidateCephVersionsBetweenLocalAndExternalClusters(c.context, c.Namespace, *version)
if err != nil {
return errors.Wrapf(err, "failed to validate ceph version between external and local")
}
}
// On external cluster setup, if we don't bootstrap any resources in the Kubernetes cluster then
// there is no need to validate the Ceph image further
if c.Spec.External.Enable && c.Spec.CephVersion.Image == "" {
logger.Debug("no spec image specified on external cluster, not validating Ceph version.")
return nil
}
// Get cluster running versions
versions, err := client.GetAllCephDaemonVersions(c.context, c.Namespace)
if err != nil {
logger.Errorf("failed to get ceph daemons versions, this typically happens during the first cluster initialization. %v", err)
return nil
}
runningVersions := *versions
differentImages, err := diffImageSpecAndClusterRunningVersion(*version, runningVersions)
if err != nil {
logger.Errorf("failed to determine if we should upgrade or not. %v", err)
// we shouldn't block the orchestration if we can't determine the version of the image spec, we proceed anyway in best effort
// we won't be able to check if there is an update or not and what to do, so we don't check the cluster status either
// This will happen if someone uses ceph/daemon:latest-master for instance
return nil
}
if differentImages {
// If the image version changed let's make sure we can safely upgrade
// check ceph's status, if not healthy we fail
cephHealthy := client.IsCephHealthy(c.context, c.Namespace)
if !cephHealthy {
if c.Spec.SkipUpgradeChecks {
logger.Warning("ceph is not healthy but SkipUpgradeChecks is set, forcing upgrade.")
} else {
return errors.Errorf("ceph status in namespace %s is not healthy, refusing to upgrade. fix the cluster and re-edit the cluster CR to trigger a new orchestation update", c.Namespace)
}
}
// This is an upgrade
logger.Info("upgrading ceph cluster to %q", version.String())
c.isUpgrade = true
}
return nil
}
+180
View File
@@ -0,0 +1,180 @@
/*
Copyright 2020 The Rook Authors. All rights reserved.
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
*/
// Package cluster to manage a Ceph cluster.
package cluster
import (
"context"
cephv1 "github.com/rook/rook/pkg/apis/ceph.rook.io/v1"
"github.com/rook/rook/pkg/clusterd"
cephclient "github.com/rook/rook/pkg/daemon/ceph/client"
discoverDaemon "github.com/rook/rook/pkg/daemon/discover"
"github.com/rook/rook/pkg/operator/k8sutil"
v1 "k8s.io/api/core/v1"
"k8s.io/apimachinery/pkg/runtime"
"sigs.k8s.io/controller-runtime/pkg/client"
)
// clientCluster struct contains a client to interact with Kubernetes object
// as well as the NamespacedName (used in requests)
type clientCluster struct {
client client.Client
namespace string
context *clusterd.Context
}
func newClientCluster(client client.Client, namespace string, context *clusterd.Context) *clientCluster {
return &clientCluster{
client: client,
namespace: namespace,
context: context,
}
}
// onK8sNodeAdd is trigger when a node is added in the Kubernetes cluster
// func (c *clientCluster) onK8sNodeAdd(object runtime.Object) bool {
func (c *clientCluster) onK8sNode(object runtime.Object) bool {
node, ok := object.(*v1.Node)
if !ok {
return false
}
// Get CephCluster
cluster := c.getCephCluster()
if k8sutil.GetNodeSchedulable(*node) == false {
logger.Debugf("node watcher: skipping cluster update. added node %q is unschedulable", node.Labels[v1.LabelHostname])
return false
}
if k8sutil.NodeIsTolerable(*node, cephv1.GetOSDPlacement(cluster.Spec.Placement).Tolerations, false) == false {
logger.Debugf("node watcher: node since it is not tolerable for cluster %q, skipping", cluster.Namespace)
return false
}
if cluster.Spec.Storage.UseAllNodes == false {
logger.Debugf("node watcher: do not use all Nodes in cluster %q, skipping", cluster.Namespace)
return false
}
// Too strict? this replaces clusterInfo == nil
if cluster.Status.Phase != cephv1.ConditionReady {
logger.Debugf("node watcher: cluster %q is not ready. skipping orchestration", cluster.Namespace)
return false
}
valid, _ := k8sutil.ValidNode(*node, cluster.Spec.Placement.All())
if valid {
nodeName := node.Name
hostname, ok := node.Labels[v1.LabelHostname]
if ok && hostname != "" {
nodeName = hostname
}
// Make sure we can call Ceph properly
// Is the node in the CRUSH map already?
// If so we don't need to reconcile, this is done to avoid double reconcile on operator restart
osds, err := cephclient.GetOSDOnHost(c.context, cluster.Namespace, nodeName)
if err != nil {
// If it fails, this might be due to the the operator just starting and catching an add event for that node
logger.Debugf("failed to get osds on node %q", nodeName)
return false
}
// If they are OSDs in the CRUSH map and if the host exists in the CRUSH map, don't reconcile
if osds != "" {
// This is Debug level because the node receives frequent updates and this will polute the logs
logger.Debugf("node watcher: node %q is already an OSD node with %q", nodeName, osds)
} else {
logger.Infof("node watcher: adding node %q to cluster %q", node.Labels[v1.LabelHostname], cluster.Namespace)
return true
}
}
return false
}
// onDeviceCMUpdate is trigger when the hot plug config map is updated
func (c *clientCluster) onDeviceCMUpdate(oldObj, newObj runtime.Object) bool {
oldCm, ok := oldObj.(*v1.ConfigMap)
if !ok {
return false
}
logger.Debugf("hot-plug cm watcher: onDeviceCMUpdate old device cm: %+v", oldCm)
newCm, ok := newObj.(*v1.ConfigMap)
if !ok {
return false
}
logger.Debugf("hot-plug cm watcher: onDeviceCMUpdate new device cm: %+v", newCm)
oldDevStr, ok := oldCm.Data[discoverDaemon.LocalDiskCMData]
if !ok {
logger.Warning("hot-plug cm watcher: unexpected old configmap data")
return false
}
newDevStr, ok := newCm.Data[discoverDaemon.LocalDiskCMData]
if !ok {
logger.Warning("hot-plug cm watcher: unexpected new configmap data")
return false
}
devicesEqual, err := discoverDaemon.DeviceListsEqual(oldDevStr, newDevStr)
if err != nil {
logger.Warningf("hot-plug cm watcher: failed to compare device lists. %v", err)
return false
}
if devicesEqual {
logger.Debug("hot-plug cm watcher: device lists are equal. skipping orchestration")
return false
}
// Get CephCluster
cluster := c.getCephCluster()
if cluster.Status.Phase != cephv1.ConditionReady {
logger.Debugf("hot-plug cm watcher: cluster %q is not ready. skipping orchestration.", cluster.Namespace)
return false
}
if len(cluster.Spec.Storage.StorageClassDeviceSets) > 0 {
logger.Info("hot-plug cm watcher: skip orchestration on device config map update for OSDs on PVC")
return false
}
logger.Infof("hot-plug cm watcher: running orchestration for namespace %q after device change", cluster.Namespace)
return true
}
func (c *clientCluster) getCephCluster() *cephv1.CephCluster {
clusterList := &cephv1.CephClusterList{}
err := c.client.List(context.TODO(), clusterList, client.InNamespace(c.namespace))
if err != nil {
logger.Debugf("%q: failed to fetch CephCluster %v", controllerName, err)
return &cephv1.CephCluster{}
}
if len(clusterList.Items) == 0 {
logger.Debugf("%q: no CephCluster resource found in namespace %q", controllerName, c.namespace)
return &cephv1.CephCluster{}
}
return &clusterList.Items[0]
}
+125
View File
@@ -0,0 +1,125 @@
/*
Copyright 2020 The Rook Authors. All rights reserved.
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
*/
package cluster
import (
"os"
"testing"
"github.com/coreos/pkg/capnslog"
cephv1 "github.com/rook/rook/pkg/apis/ceph.rook.io/v1"
"github.com/rook/rook/pkg/client/clientset/versioned/scheme"
"github.com/rook/rook/pkg/clusterd"
"github.com/rook/rook/pkg/operator/k8sutil"
"github.com/stretchr/testify/assert"
corev1 "k8s.io/api/core/v1"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"k8s.io/apimachinery/pkg/runtime"
"sigs.k8s.io/controller-runtime/pkg/client/fake"
)
func TestOnDeviceCMUpdate(t *testing.T) {
// Set DEBUG logging
capnslog.SetGlobalLogLevel(capnslog.DEBUG)
os.Setenv("ROOK_LOG_LEVEL", "DEBUG")
dum := &corev1.Service{}
ns := "rook-ceph"
// Register operator types with the runtime scheme.
s := scheme.Scheme
s.AddKnownTypes(cephv1.SchemeGroupVersion, &cephv1.CephCluster{})
cephCluster := &cephv1.CephCluster{
ObjectMeta: metav1.ObjectMeta{
Name: ns,
Namespace: ns,
},
Status: cephv1.ClusterStatus{
Phase: "",
},
}
object := []runtime.Object{
cephCluster,
}
// Create a fake client to mock API calls.
client := fake.NewFakeClientWithScheme(s, object...)
clientCluster := newClientCluster(client, ns, &clusterd.Context{})
// Dummy object
b := clientCluster.onDeviceCMUpdate(dum, dum)
assert.False(t, b)
// No Data in the cm
oldCM := &corev1.ConfigMap{}
newCM := &corev1.ConfigMap{}
b = clientCluster.onDeviceCMUpdate(oldCM, newCM)
assert.False(t, b)
devices := []byte(`
[
{
"name": "dm-0",
"parent": ".",
"hasChildren": false,
"devLinks": "/dev/disk/by-id/dm-name-ceph--bee31cdd--e899--4f9a--9e77--df71cfad66f9-osd--data--b5df7900--0cf0--4b1a--a337--7b57c9f0111b/dev/disk/by-id/dm-uuid-LVM-B10SBHeAy5yF6l2OM3p3EqTQbUAYc6JI63n8ZZPTmxRHXTJHmQ4YTAIBCJqY931Z",
"size": 31138512896,
"uuid": "aafee853-1b8d-4a15-83a9-17825728befc",
"serial": "",
"type": "lvm",
"rotational": true,
"readOnly": false,
"Partitions": [
{
"Name": "ceph--bee31cdd--e899--4f9a--9e77--df71cfad66f9-osd--data--b5df7900--0cf0--4b1a--a337--7b57c9f0111b",
"Size": 0,
"Label": "",
"Filesystem": ""
}
],
"filesystem": "ceph_bluestore",
"vendor": "",
"model": "",
"wwn": "",
"wwnVendorExtension": "",
"empty": false,
"real-path": "/dev/mapper/ceph--bee31cdd--e899--4f9a--9e77--df71cfad66f9-osd--data--b5df7900--0cf0--4b1a--a337--7b57c9f0111b"
}
]`)
oldData := make(map[string]string, 1)
oldData["devices"] = "[{}]"
oldCM.Data = oldData
newData := make(map[string]string, 1)
newData["devices"] = string(devices)
newCM.Data = newData
// now there is a diff but cluster is not ready
b = clientCluster.onDeviceCMUpdate(oldCM, newCM)
assert.False(t, b)
// finally the cluster is ready and we can reconcile
// Add ready status to the CephCluster
cephCluster.Status.Phase = k8sutil.ReadyStatus
client = fake.NewFakeClientWithScheme(s, object...)
clientCluster.client = client
b = clientCluster.onDeviceCMUpdate(oldCM, newCM)
assert.True(t, b)
}
+29 -17
View File
@@ -18,13 +18,16 @@ limitations under the License.
package config
import (
"context"
"time"
"github.com/pkg/errors"
cephv1 "github.com/rook/rook/pkg/apis/ceph.rook.io/v1"
"github.com/rook/rook/pkg/clusterd"
v1 "k8s.io/api/core/v1"
kerrors "k8s.io/apimachinery/pkg/api/errors"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"k8s.io/apimachinery/pkg/types"
)
var (
@@ -33,8 +36,8 @@ var (
)
// ConditionExport function will export each condition into the cluster custom resource
func ConditionExport(context *clusterd.Context, namespace, name string, conditionType cephv1.ConditionType, status v1.ConditionStatus, reason, message string) {
setCondition(context, namespace, name, cephv1.Condition{
func ConditionExport(context *clusterd.Context, namespaceName types.NamespacedName, conditionType cephv1.ConditionType, status v1.ConditionStatus, reason, message string) {
setCondition(context, namespaceName, cephv1.Condition{
Type: conditionType,
Status: status,
Reason: reason,
@@ -43,11 +46,18 @@ func ConditionExport(context *clusterd.Context, namespace, name string, conditio
}
// setCondition updates the conditions of the cluster custom resource
func setCondition(context *clusterd.Context, namespace, name string, newCondition cephv1.Condition) {
cluster, err := context.RookClientset.CephV1().CephClusters(namespace).Get(name, metav1.GetOptions{})
func setCondition(c *clusterd.Context, namespaceName types.NamespacedName, newCondition cephv1.Condition) {
cluster := &cephv1.CephCluster{}
err := c.Client.Get(context.TODO(), namespaceName, cluster)
if err != nil {
logger.Errorf("failed to get cluster %v", err)
if kerrors.IsNotFound(err) {
logger.Errorf("no CephCluster could not be found. %+v", err)
return
}
logger.Errorf("failed to get CephCluster object. %+v", err)
return
}
if conditions == nil {
conditions = &cluster.Status.Conditions
if cluster.Status.Conditions != nil {
@@ -76,14 +86,15 @@ func setCondition(context *clusterd.Context, namespace, name string, newConditio
cluster.Status.State = state
}
cluster.Status.Message = newCondition.Message
logger.Infof("CephCluster %q status: %q. %q", namespace, cluster.Status.Phase, cluster.Status.Message)
logger.Infof("CephCluster %q status: %q. %q", namespaceName.Namespace, cluster.Status.Phase, cluster.Status.Message)
}
if _, err := context.RookClientset.CephV1().CephClusters(namespace).Update(cluster); err != nil {
logger.Errorf("failed to update cluster condition %v", err)
err = c.Client.Update(context.TODO(), cluster)
if err != nil {
logger.Errorf("failed to update cluster condition to %+v. %v", newCondition, err)
}
if newCondition.Type == cephv1.ConditionReady {
checkConditionFalse(context, namespace, name)
checkConditionFalse(c, namespaceName)
}
}
@@ -118,7 +129,7 @@ func translatePhasetoState(phase cephv1.ConditionType) cephv1.ClusterState {
}
// Updating the status of Progressing, Updating or Upgrading to False once cluster is Ready
func checkConditionFalse(context *clusterd.Context, namespace, name string) {
func checkConditionFalse(context *clusterd.Context, namespaceName types.NamespacedName) {
tempConditionList := []cephv1.ConditionType{cephv1.ConditionUpdating, cephv1.ConditionUpgrading, cephv1.ConditionProgressing}
var tempCondition cephv1.ConditionType
for _, conditionType := range tempConditionList {
@@ -138,24 +149,24 @@ func checkConditionFalse(context *clusterd.Context, namespace, name string) {
reason = "ProgressingCompleted"
message = "Cluster progression is completed"
}
ConditionExport(context, namespace, name, tempCondition, v1.ConditionFalse, reason, message)
ConditionExport(context, namespaceName, tempCondition, v1.ConditionFalse, reason, message)
}
// ConditionInitialize initializes some of the conditions at the beginning of cluster creation
func ConditionInitialize(context *clusterd.Context, namespace, name string) {
setCondition(context, namespace, name, cephv1.Condition{
func ConditionInitialize(context *clusterd.Context, namespaceName types.NamespacedName) {
setCondition(context, namespaceName, cephv1.Condition{
Type: cephv1.ConditionFailure,
Status: v1.ConditionFalse,
Reason: "",
Message: "",
})
setCondition(context, namespace, name, cephv1.Condition{
setCondition(context, namespaceName, cephv1.Condition{
Type: cephv1.ConditionIgnored,
Status: v1.ConditionFalse,
Reason: "",
Message: "",
})
setCondition(context, namespace, name, cephv1.Condition{
setCondition(context, namespaceName, cephv1.Condition{
Type: cephv1.ConditionUpgrading,
Status: v1.ConditionFalse,
Reason: "",
@@ -172,8 +183,9 @@ func conditionMapping(conditions []cephv1.Condition) {
}
// CheckConditionReady checks whether the cluster is Ready and returns the message for the Progressing ConditionType
func CheckConditionReady(context *clusterd.Context, namespace, name string) string {
cluster, err := context.RookClientset.CephV1().CephClusters(namespace).Get(name, metav1.GetOptions{})
func CheckConditionReady(c *clusterd.Context, namespaceName types.NamespacedName) string {
cluster := &cephv1.CephCluster{}
err := c.Client.Get(context.TODO(), namespaceName, cluster)
if err != nil {
logger.Errorf("failed to get cluster %v", err)
}
@@ -18,12 +18,16 @@ package controller
import (
"context"
"fmt"
"reflect"
"strings"
"time"
cephv1 "github.com/rook/rook/pkg/apis/ceph.rook.io/v1"
"github.com/rook/rook/pkg/clusterd"
cephclient "github.com/rook/rook/pkg/daemon/ceph/client"
"github.com/rook/rook/pkg/operator/k8sutil"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"k8s.io/apimachinery/pkg/types"
"sigs.k8s.io/controller-runtime/pkg/client"
"sigs.k8s.io/controller-runtime/pkg/reconcile"
@@ -86,3 +90,25 @@ func IsReadyToReconcile(c client.Client, clustercontext *clusterd.Context, names
logger.Infof("%s: CephCluster %q found but skipping reconcile since Ceph health is %q", controllerName, cephCluster.Name, status.Health.Status)
return cephCluster.Spec, false, cephClusterExists, WaitForRequeueIfCephClusterNotReady
}
// ClusterOwnerRef represents the owner reference of the CephCluster CR
func ClusterOwnerRef(clusterName, clusterID string) metav1.OwnerReference {
blockOwner := true
return metav1.OwnerReference{
APIVersion: fmt.Sprintf("%s/%s", ClusterResource.Group, ClusterResource.Version),
Kind: ClusterResource.Kind,
Name: clusterName,
UID: types.UID(clusterID),
BlockOwnerDeletion: &blockOwner,
}
}
// ClusterResource operator-kit Custom Resource Definition
var ClusterResource = k8sutil.CustomResource{
Name: "cephcluster",
Plural: "cephclusters",
Group: cephv1.CustomResourceGroup,
Version: cephv1.Version,
Kind: reflect.TypeOf(cephv1.CephCluster{}).Name(),
APIVersion: fmt.Sprintf("%s/%s", cephv1.CustomResourceGroup, cephv1.Version),
}
+66
View File
@@ -0,0 +1,66 @@
/*
Copyright 2020 The Rook Authors. All rights reserved.
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
*/
package controller
import (
"context"
"github.com/pkg/errors"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"k8s.io/apimachinery/pkg/apis/meta/v1/unstructured"
"k8s.io/apimachinery/pkg/runtime"
ctrl "sigs.k8s.io/controller-runtime"
"sigs.k8s.io/controller-runtime/pkg/client"
"sigs.k8s.io/controller-runtime/pkg/client/apiutil"
"sigs.k8s.io/controller-runtime/pkg/handler"
)
// ObjectToCRMapper returns the list of a given object type metadata
// It is used to trigger a reconcile object Kind A when watching object Kind B
// So we reconcile Kind A instead of Kind B
// For instance, we watch for CephCluster CR changes but want to reconcile CephFilesystem based on a Spec change
func ObjectToCRMapper(c client.Client, ro runtime.Object, scheme *runtime.Scheme) (handler.Mapper, error) {
if _, ok := ro.(metav1.ListInterface); !ok {
return nil, errors.Errorf("expected a metav1.ListInterface, got %T instead", ro)
}
gvk, err := apiutil.GVKForObject(ro, scheme)
if err != nil {
return nil, err
}
return handler.ToRequestsFunc(func(o handler.MapObject) []ctrl.Request {
list := &unstructured.UnstructuredList{}
list.SetGroupVersionKind(gvk)
err := c.List(context.TODO(), list)
if err != nil {
return nil
}
results := []ctrl.Request{}
for _, obj := range list.Items {
results = append(results, ctrl.Request{
NamespacedName: client.ObjectKey{
Namespace: obj.GetNamespace(),
Name: obj.GetName(),
},
})
}
return results
}), nil
}
@@ -0,0 +1,68 @@
/*
Copyright 2020 The Rook Authors. All rights reserved.
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
*/
package controller
import (
"reflect"
"testing"
cephv1 "github.com/rook/rook/pkg/apis/ceph.rook.io/v1"
"github.com/rook/rook/pkg/client/clientset/versioned/scheme"
"github.com/stretchr/testify/assert"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
"k8s.io/apimachinery/pkg/runtime"
ctrl "sigs.k8s.io/controller-runtime"
"sigs.k8s.io/controller-runtime/pkg/client"
"sigs.k8s.io/controller-runtime/pkg/client/fake"
"sigs.k8s.io/controller-runtime/pkg/handler"
)
func TestObjectToCRMapper(t *testing.T) {
fs := &cephv1.CephFilesystem{
ObjectMeta: metav1.ObjectMeta{
Name: name,
Namespace: namespace,
},
TypeMeta: metav1.TypeMeta{
Kind: reflect.TypeOf(cephv1.CephFilesystem{}).Name(),
},
}
// Objects to track in the fake client.
objects := []runtime.Object{
&cephv1.CephFilesystemList{},
fs,
}
// Register operator types with the runtime scheme.
s := scheme.Scheme
s.AddKnownTypes(cephv1.SchemeGroupVersion, &cephv1.CephFilesystemList{})
s.AddKnownTypes(cephv1.SchemeGroupVersion, &cephv1.CephFilesystem{})
s.AddKnownTypes(cephv1.SchemeGroupVersion, &cephv1.CephCluster{})
// Create a fake client to mock API calls.
cl := fake.NewFakeClientWithScheme(s, objects...)
// Fake reconcile request
fakeRequest := []ctrl.Request{
{NamespacedName: client.ObjectKey{Name: "my-pool", Namespace: "rook-ceph"}},
}
handlerFunc, err := ObjectToCRMapper(cl, objects[0], s)
assert.NoError(t, err)
assert.ElementsMatch(t, fakeRequest, handlerFunc.Map(handler.MapObject{Object: fs}))
}
+256 -61
View File
@@ -24,6 +24,10 @@ import (
"github.com/google/go-cmp/cmp"
"github.com/pkg/errors"
cephv1 "github.com/rook/rook/pkg/apis/ceph.rook.io/v1"
"github.com/rook/rook/pkg/operator/ceph/config"
"github.com/rook/rook/pkg/operator/k8sutil"
appsv1 "k8s.io/api/apps/v1"
corev1 "k8s.io/api/core/v1"
"k8s.io/apimachinery/pkg/api/meta"
"k8s.io/apimachinery/pkg/api/resource"
"k8s.io/apimachinery/pkg/runtime"
@@ -33,6 +37,9 @@ import (
const (
cephVersionLabelKey = "ceph_version"
// Unfortunately this is a duplicate of the const EndpointConfigMapName in the mon package, but done to avoid import cycle
endpointConfigMapName = "rook-ceph-mon-endpoints"
doNotReconcileLabelName = "do_not_reconcile"
)
// WatchControllerPredicate is a special update filter for update events
@@ -43,28 +50,33 @@ const (
func WatchControllerPredicate() predicate.Funcs {
return predicate.Funcs{
CreateFunc: func(e event.CreateEvent) bool {
logger.Debug("create event from the parent object")
logger.Debug("create event from a CR")
return true
},
DeleteFunc: func(e event.DeleteEvent) bool {
logger.Debug("delete event from the parent object")
logger.Debug("delete event from a CR")
return true
},
UpdateFunc: func(e event.UpdateEvent) bool {
logger.Debug("update event from the parent object")
logger.Debug("update event from a CR")
// resource.Quantity has non-exportable fields, so we use its comparator method
resourceQtyComparer := cmp.Comparer(func(x, y resource.Quantity) bool { return x.Cmp(y) == 0 })
switch objOld := e.ObjectOld.(type) {
case *cephv1.CephObjectStore:
objNew := e.ObjectNew.(*cephv1.CephObjectStore)
logger.Debug("update event from the parent object CephObjectStore")
logger.Debug("update event on CephObjectStore CR")
// If the labels "do_not_reconcile" is set on the object, let's not reconcile that request
isDoNotReconcile := isDoNotReconcile(objNew.GetLabels())
if isDoNotReconcile {
logger.Debugf("object %q matched on update but %q label is set, doing nothing", doNotReconcileLabelName, objNew.Name)
return false
}
diff := cmp.Diff(objOld.Spec, objNew.Spec, resourceQtyComparer)
if diff != "" || objOld.GetDeletionTimestamp() != objNew.GetDeletionTimestamp() {
// Checking if diff is not empty so we don't print it when the CR gets deleted
if diff != "" {
logger.Infof("CR has changed for %q. diff=%s", objNew.Name, diff)
}
if diff != "" {
logger.Infof("CR has changed for %q. diff=%s", objNew.Name, diff)
} else if objOld.GetDeletionTimestamp() != objNew.GetDeletionTimestamp() {
logger.Debugf("CR %q is going be deleted", objNew.Name)
return true
} else if objOld.GetGeneration() != objNew.GetGeneration() {
logger.Debugf("skipping resource %q update with unchanged spec", objNew.Name)
@@ -77,13 +89,18 @@ func WatchControllerPredicate() predicate.Funcs {
case *cephv1.CephObjectStoreUser:
objNew := e.ObjectNew.(*cephv1.CephObjectStoreUser)
logger.Debug("update event from the parent object CephObjectStoreUser")
logger.Debug("update event on CephObjectStoreUser CR")
// If the labels "do_not_reconcile" is set on the object, let's not reconcile that request
isDoNotReconcile := isDoNotReconcile(objNew.GetLabels())
if isDoNotReconcile {
logger.Debugf("object %q matched on update but %q label is set, doing nothing", doNotReconcileLabelName, objNew.Name)
return false
}
diff := cmp.Diff(objOld.Spec, objNew.Spec, resourceQtyComparer)
if diff != "" || objOld.GetDeletionTimestamp() != objNew.GetDeletionTimestamp() {
// Checking if diff is not empty so we don't print it when the CR gets deleted
if diff != "" {
logger.Infof("CR has changed for %q. diff=%s", objNew.Name, diff)
}
if diff != "" {
logger.Infof("CR has changed for %q. diff=%s", objNew.Name, diff)
} else if objOld.GetDeletionTimestamp() != objNew.GetDeletionTimestamp() {
logger.Debugf("CR %q is going be deleted", objNew.Name)
return true
} else if objOld.GetGeneration() != objNew.GetGeneration() {
logger.Debugf("skipping resource %q update with unchanged spec", objNew.Name)
@@ -91,10 +108,15 @@ func WatchControllerPredicate() predicate.Funcs {
case *cephv1.CephBlockPool:
objNew := e.ObjectNew.(*cephv1.CephBlockPool)
logger.Debug("update event from the parent object CephBlockPool")
logger.Debug("update event on CephBlockPool CR")
// If the labels "do_not_reconcile" is set on the object, let's not reconcile that request
isDoNotReconcile := isDoNotReconcile(objNew.GetLabels())
if isDoNotReconcile {
logger.Debugf("object %q matched on update but %q label is set, doing nothing", doNotReconcileLabelName, objNew.Name)
return false
}
diff := cmp.Diff(objOld.Spec, objNew.Spec, resourceQtyComparer)
if diff != "" || objOld.GetDeletionTimestamp() != objNew.GetDeletionTimestamp() {
// Checking if diff is not empty so we don't print it when the CR gets deleted
if diff != "" {
logger.Infof("CR has changed for %q. diff=%s", objNew.Name, diff)
}
@@ -105,13 +127,18 @@ func WatchControllerPredicate() predicate.Funcs {
case *cephv1.CephFilesystem:
objNew := e.ObjectNew.(*cephv1.CephFilesystem)
logger.Debug("update event from the parent object CephFilesystem")
logger.Debug("update event on CephFilesystem CR")
// If the labels "do_not_reconcile" is set on the object, let's not reconcile that request
isDoNotReconcile := isDoNotReconcile(objNew.GetLabels())
if isDoNotReconcile {
logger.Debugf("object %q matched on update but %q label is set, doing nothing", doNotReconcileLabelName, objNew.Name)
return false
}
diff := cmp.Diff(objOld.Spec, objNew.Spec, resourceQtyComparer)
if diff != "" || objOld.GetDeletionTimestamp() != objNew.GetDeletionTimestamp() {
// Checking if diff is not empty so we don't print it when the CR gets deleted
if diff != "" {
logger.Infof("CR has changed for %q. diff=%s", objNew.Name, diff)
}
if diff != "" {
logger.Infof("CR has changed for %q. diff=%s", objNew.Name, diff)
} else if objOld.GetDeletionTimestamp() != objNew.GetDeletionTimestamp() {
logger.Debugf("CR %q is going be deleted", objNew.Name)
return true
} else if objOld.GetGeneration() != objNew.GetGeneration() {
logger.Debugf("skipping resource %q update with unchanged spec", objNew.Name)
@@ -124,33 +151,73 @@ func WatchControllerPredicate() predicate.Funcs {
case *cephv1.CephNFS:
objNew := e.ObjectNew.(*cephv1.CephNFS)
logger.Debug("update event from the parent object CephNFS")
logger.Debug("update event on CephNFS CR")
// If the labels "do_not_reconcile" is set on the object, let's not reconcile that request
isDoNotReconcile := isDoNotReconcile(objNew.GetLabels())
if isDoNotReconcile {
logger.Debugf("object %q matched on update but %q label is set, doing nothing", doNotReconcileLabelName, objNew.Name)
return false
}
diff := cmp.Diff(objOld.Spec, objNew.Spec, resourceQtyComparer)
if diff != "" || objOld.GetDeletionTimestamp() != objNew.GetDeletionTimestamp() {
// Checking if diff is not empty so we don't print it when the CR gets deleted
if diff != "" {
logger.Infof("CR has changed for %q. diff=%s", objNew.Name, diff)
}
if diff != "" {
logger.Infof("CR has changed for %q. diff=%s", objNew.Name, diff)
} else if objOld.GetDeletionTimestamp() != objNew.GetDeletionTimestamp() {
logger.Debugf("CR %q is going be deleted", objNew.Name)
return true
} else if objOld.GetGeneration() != objNew.GetGeneration() {
logger.Debugf("skipping resource %q update with unchanged spec", objNew.Name)
}
case *cephv1.CephRBDMirror:
objNew := e.ObjectNew.(*cephv1.CephRBDMirror)
logger.Debug("update event from the parent object CephRBDMirror")
diff := cmp.Diff(objOld.Spec, objNew.Spec, resourceQtyComparer)
if diff != "" || objOld.GetDeletionTimestamp() != objNew.GetDeletionTimestamp() {
// Checking if diff is not empty so we don't print it when the CR gets deleted
if diff != "" {
logger.Infof("CR has changed for %q. diff=%s", objNew.Name, diff)
}
// Handling upgrades
isUpgrade := isUpgrade(objOld.GetLabels(), objNew.GetLabels())
if isUpgrade {
return true
}
case *cephv1.CephRBDMirror:
objNew := e.ObjectNew.(*cephv1.CephRBDMirror)
logger.Debug("update event on CephRBDMirror CR")
// If the labels "do_not_reconcile" is set on the object, let's not reconcile that request
isDoNotReconcile := isDoNotReconcile(objNew.GetLabels())
if isDoNotReconcile {
logger.Debugf("object %q matched on update but %q label is set, doing nothing", doNotReconcileLabelName, objNew.Name)
return false
}
diff := cmp.Diff(objOld.Spec, objNew.Spec, resourceQtyComparer)
if diff != "" {
logger.Infof("CR has changed for %q. diff=%s", objNew.Name, diff)
} else if objOld.GetDeletionTimestamp() != objNew.GetDeletionTimestamp() {
logger.Debugf("CR %q is going be deleted", objNew.Name)
return true
} else if objOld.GetGeneration() != objNew.GetGeneration() {
logger.Debugf("skipping resource %q update with unchanged spec", objNew.Name)
}
// Handling upgrades
isUpgrade := isUpgrade(objOld.GetLabels(), objNew.GetLabels())
if isUpgrade {
return true
}
case *cephv1.CephCluster:
objNew := e.ObjectNew.(*cephv1.CephCluster)
logger.Debug("update event on CephCluster CR")
// If the labels "do_not_reconcile" is set on the object, let's not reconcile that request
isDoNotReconcile := isDoNotReconcile(objNew.GetLabels())
if isDoNotReconcile {
logger.Debugf("object %q matched on update but %q label is set, doing nothing", doNotReconcileLabelName, objNew.Name)
return false
}
diff := cmp.Diff(objOld.Spec, objNew.Spec, resourceQtyComparer)
if diff != "" {
logger.Infof("CR has changed for %q. diff=%s", objNew.Name, diff)
return true
} else if objOld.GetDeletionTimestamp() != objNew.GetDeletionTimestamp() {
logger.Debugf("CR %q is going be deleted", objNew.Name)
return true
} else if objOld.GetGeneration() != objNew.GetGeneration() {
logger.Debugf("skipping resource %q update with unchanged spec", objNew.Name)
}
}
logger.Debug("wont update unknown object")
return false
},
GenericFunc: func(e event.GenericEvent) bool {
@@ -160,7 +227,7 @@ func WatchControllerPredicate() predicate.Funcs {
}
// objectChanged checks whether the object has been updated
func objectChanged(oldObj, newObj runtime.Object) (bool, error) {
func objectChanged(oldObj, newObj runtime.Object, objectName string) (bool, error) {
var doReconcile bool
old := oldObj.DeepCopyObject()
new := newObj.DeepCopyObject()
@@ -181,7 +248,7 @@ func objectChanged(oldObj, newObj runtime.Object) (bool, error) {
return doReconcile, nil
}
return isValidEvent(diff.Patch), nil
return isValidEvent(diff.Patch, objectName), nil
}
// WatchPredicateForNonCRDObject is a special filter for create events
@@ -199,35 +266,82 @@ func WatchPredicateForNonCRDObject(owner runtime.Object, scheme *runtime.Scheme)
CreateFunc: func(e event.CreateEvent) bool {
return false
},
DeleteFunc: func(e event.DeleteEvent) bool {
match, object, err := ownerMatcher.Match(e.Object)
if err != nil {
logger.Errorf("failed to check if object kind %q matched. %v", e.Object.GetObjectKind(), err)
}
objectName := object.GetName()
if match {
logger.Debugf("object %q matched on delete", object.GetName())
// If the resource is a CM, we might want to ignore it since some of them are ephemeral
isCMToIgnoreOnDelete := isCMToIgnoreOnDelete(e.Object)
if isCMToIgnoreOnDelete {
return false
}
// If the resource is a canary deployment we don't reconcile because it's ephemeral
isCanary := isCanary(e.Object)
if isCanary {
return false
}
logger.Infof("object %q matched on delete, reconciling", objectName)
return true
}
logger.Debugf("object %q did not match on delete", object.GetName())
logger.Debugf("object %q did not match on delete", objectName)
return false
},
UpdateFunc: func(e event.UpdateEvent) bool {
match, object, err := ownerMatcher.Match(e.ObjectNew)
if err != nil {
logger.Errorf("failed to check if object matched. %v", err)
}
objectName := object.GetName()
if match {
logger.Debugf("object %q matched on update", object.GetName())
objectChanged, err := objectChanged(e.ObjectOld, e.ObjectNew)
// If the labels "do_not_reconcile" is set on the object, let's not reconcile that request
isDoNotReconcile := isDoNotReconcile(object.GetLabels())
if isDoNotReconcile {
logger.Debugf("object %q matched on update but %q label is set, doing nothing", doNotReconcileLabelName, objectName)
return false
}
logger.Debugf("object %q matched on update", objectName)
// CONFIGMAP WHITELIST
// Only reconcile on rook-config-override CM changes
isCMTConfigOverride := isCMTConfigOverride(e.ObjectNew)
if !isCMTConfigOverride {
return false
}
// SECRETS BLACKLIST
// If the resource is a Secret, we might want to ignore it
// We want to reconcile Secrets in case their content gets altered
isSecretToIgnoreOnUpdate := isSecretToIgnoreOnUpdate(e.ObjectNew)
if isSecretToIgnoreOnUpdate {
return false
}
// If the resource is a deployment we don't reconcile
_, ok := e.ObjectNew.(*appsv1.Deployment)
if ok {
return false
}
// did the object change?
objectChanged, err := objectChanged(e.ObjectOld, e.ObjectNew, objectName)
if err != nil {
logger.Errorf("failed to check if object %q changed. %v", object.GetName(), err)
logger.Errorf("failed to check if object %q changed. %v", objectName, err)
}
return objectChanged
}
return false
},
GenericFunc: func(e event.GenericEvent) bool {
return false
},
@@ -237,29 +351,31 @@ func WatchPredicateForNonCRDObject(owner runtime.Object, scheme *runtime.Scheme)
// isValidEvent analyses the diff between two objects events and determines
// if we should reconcile that event or not
// The goal is to avoid double-reconcile as much as possible
func isValidEvent(patch []byte) bool {
func isValidEvent(patch []byte, objectName string) bool {
patchString := string(patch)
// Seem a bit 'weak' but since we can't get a real struct of the object
// (unless we use the unstructured package, but that over complicates things)
// That's probably the most straightforward approach for now...
//
// The downscale only shows a "deletionTimestamp" which is not appropriate to catch
if strings.Contains(patchString, "Created new replica set") {
logger.Debug("don't reconcile on replicaset addition")
return false
}
// It looks like there is a diff
// if the status changed, we do nothing
var p map[string]interface{}
json.Unmarshal(patch, &p)
// don't reconcile on status update on an object (e.g. status "creating")
delete(p, "status")
// Do not reconcile on metadata change since managedFields are often updated by the server
delete(p, "metadata")
// If the patch is now empty, we don't reconcile, nothing changed
if len(p) == 0 {
return false
}
logger.Infof("will reconcile based on patch %s", patchString)
// Re-marshal to get the last diff
patch, err := json.Marshal(p)
if err != nil {
logger.Infof("controller will reconcile resource %q based on patch: %s", objectName, patchString)
}
// If after all the filtering there is still something in the patch, we reconcile
logger.Infof("controller will reconcile resource %q based on patch: %s", objectName, string(patch))
return true
}
@@ -284,3 +400,82 @@ func isUpgrade(oldLabels, newLabels map[string]string) bool {
return false
}
func isCanary(obj runtime.Object) bool {
// If not a deployment, let's not reconcile
d, ok := obj.(*appsv1.Deployment)
if !ok {
return false
}
// Get the labels
labels := d.GetLabels()
labelVal, labelKeyExist := labels["mon_canary"]
if labelKeyExist && labelVal == "true" {
logger.Debugf("do not reconcile %q on monitor canary deployments", d.Name)
return true
}
return false
}
func isCMTConfigOverride(obj runtime.Object) bool {
// If not a ConfigMap, let's not reconcile
cm, ok := obj.(*corev1.ConfigMap)
if !ok {
return false
}
objectName := cm.GetName()
if objectName == k8sutil.ConfigOverrideName {
return true
}
return false
}
func isCMToIgnoreOnDelete(obj runtime.Object) bool {
// If not a ConfigMap, let's not reconcile
cm, ok := obj.(*corev1.ConfigMap)
if !ok {
return false
}
objectName := cm.GetName()
// is it the object the temporarily osd config map?
if strings.HasPrefix(objectName, "rook-ceph-osd-") && strings.HasSuffix(objectName, "-status") {
logger.Debugf("do not reconcile on %q config map changes", objectName)
return true
}
return false
}
func isSecretToIgnoreOnUpdate(obj runtime.Object) bool {
// If not a Secret, let's not reconcile
s, ok := obj.(*corev1.Secret)
if !ok {
return false
}
objectName := s.GetName()
switch objectName {
case config.StoreName:
logger.Debugf("do not reconcile on %q secret changes", objectName)
return true
}
return false
}
func isDoNotReconcile(labels map[string]string) bool {
value, ok := labels[doNotReconcileLabelName]
// Nothing exists
if ok && value == "true" {
return true
}
return false
}
+134 -2
View File
@@ -21,7 +21,10 @@ import (
"testing"
cephv1 "github.com/rook/rook/pkg/apis/ceph.rook.io/v1"
"github.com/rook/rook/pkg/operator/ceph/config"
"github.com/stretchr/testify/assert"
appsv1 "k8s.io/api/apps/v1"
corev1 "k8s.io/api/core/v1"
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
)
@@ -65,13 +68,13 @@ func TestObjectChanged(t *testing.T) {
}
// Identical
changed, err := objectChanged(oldObject, newObject)
changed, err := objectChanged(oldObject, newObject, "foo")
assert.NoError(t, err)
assert.False(t, changed)
// Replica size changed
oldObject.Spec.Replicated.Size = newReplicas
changed, err = objectChanged(oldObject, newObject)
changed, err = objectChanged(oldObject, newObject, "foo")
assert.NoError(t, err)
assert.True(t, changed)
}
@@ -102,3 +105,132 @@ func TestIsUpgrade(t *testing.T) {
b = isUpgrade(oldLabel, newLabel)
assert.True(t, b, fmt.Sprintf("%v,%v", oldLabel, newLabel))
}
func TestIsValidEvent(t *testing.T) {
obj := "rook-ceph-mon-a"
valid := []byte(`{
"metadata": {},
"spec": {},
"status": {
"conditions": [
{
"message": "ReplicaSet \"rook-ceph-mon-b-784fc58bf8\" is progressing.",
"reason": "ReplicaSetUpdated",
"type": "Progressing"
}
]
}
}`)
b := isValidEvent(valid, obj)
assert.True(t, b)
valid = []byte(`{"foo": "bar"}`)
b = isValidEvent(valid, obj)
assert.True(t, b)
invalid := []byte(`{
"metadata": {},
"status": {},
}`)
b = isValidEvent(invalid, obj)
assert.False(t, b)
}
func TestIsCanary(t *testing.T) {
dum := &cephv1.CephBlockPool{}
b := isCanary(dum)
assert.False(t, b)
d := &appsv1.Deployment{}
b = isCanary(d)
assert.False(t, b)
d.Labels = map[string]string{
"foo": "bar",
}
b = isCanary(d)
assert.False(t, b)
d.Labels["mon_canary"] = "true"
b = isCanary(d)
assert.True(t, b)
}
func TestIsCMToIgnoreOnUpdate(t *testing.T) {
dum := &cephv1.CephBlockPool{}
b := isCMTConfigOverride(dum)
assert.False(t, b)
cm := &corev1.ConfigMap{}
b = isCMTConfigOverride(cm)
assert.False(t, b)
cm.Name = "rook-ceph-mon-endpoints"
b = isCMTConfigOverride(cm)
assert.False(t, b)
cm.Name = "rook-config-override"
b = isCMTConfigOverride(cm)
assert.True(t, b)
}
func TestIsCMToIgnoreOnDelete(t *testing.T) {
dum := &cephv1.CephBlockPool{}
b := isCMToIgnoreOnDelete(dum)
assert.False(t, b)
cm := &corev1.ConfigMap{}
b = isCMToIgnoreOnDelete(cm)
assert.False(t, b)
cm.Name = "rook-ceph-mon-endpoints"
b = isCMToIgnoreOnDelete(cm)
assert.False(t, b)
cm.Name = "rook-ceph-osd-minikube-status"
b = isCMToIgnoreOnDelete(cm)
assert.True(t, b)
}
func TestIsSecretToIgnoreOnUpdate(t *testing.T) {
dum := &cephv1.CephBlockPool{}
b := isSecretToIgnoreOnUpdate(dum)
assert.False(t, b)
s := &corev1.Secret{}
b = isSecretToIgnoreOnUpdate(s)
assert.False(t, b)
s.Name = "foo"
b = isSecretToIgnoreOnUpdate(s)
assert.False(t, b)
s.Name = config.StoreName
b = isSecretToIgnoreOnUpdate(s)
assert.True(t, b)
}
func TestIsDoNotReconcile(t *testing.T) {
l := map[string]string{
"foo": "bar",
}
// value not present
b := isDoNotReconcile(l)
assert.False(t, b)
// good value wrong content
l["do_not_reconcile"] = "false"
b = isDoNotReconcile(l)
assert.False(t, b)
// good value and good content
l["do_not_reconcile"] = "true"
b = isDoNotReconcile(l)
assert.True(t, b)
}
+2 -8
View File
@@ -18,7 +18,6 @@ package operator
import (
"github.com/rook/rook/pkg/operator/ceph/cluster"
controllers "github.com/rook/rook/pkg/operator/ceph/disruption"
"github.com/rook/rook/pkg/operator/ceph/disruption/controllerconfig"
"sigs.k8s.io/controller-runtime/pkg/client/config"
@@ -26,7 +25,6 @@ import (
)
func (o *Operator) startManager(namespaceToWatch string, stopCh <-chan struct{}) {
// Set up a manager
mgrOpts := manager.Options{
LeaderElection: false,
@@ -39,17 +37,13 @@ func (o *Operator) startManager(namespaceToWatch string, stopCh <-chan struct{})
logger.Errorf("failed to get client config for controller-runtime manager. %v", err)
return
}
mgr, err := manager.New(kubeConfig, mgrOpts)
if err != nil {
logger.Errorf("failed to set up overall controller-runtime manager. %v", err)
return
}
// Add the registered controllers to the manager (entrypoint for controllers)
err = cluster.AddToManager(mgr, o.context)
if err != nil {
logger.Errorf("failed to add controllers to controller-runtime manager. %v", err)
}
// options to pass to the controllers
controllerOpts := &controllerconfig.Context{
RookImage: o.rookImage,
@@ -58,7 +52,7 @@ func (o *Operator) startManager(namespaceToWatch string, stopCh <-chan struct{})
ReconcileCanaries: &controllerconfig.LockingBool{},
}
// Add the registered controllers to the manager (entrypoint for controllers)
err = controllers.AddToManager(mgr, controllerOpts)
err = cluster.AddToManager(mgr, controllerOpts, o.clusterController)
if err != nil {
logger.Errorf("failed to add controllers to controller-runtime manager. %v", err)
}
@@ -29,7 +29,8 @@ import (
"github.com/pkg/errors"
cephv1 "github.com/rook/rook/pkg/apis/ceph.rook.io/v1"
cephClient "github.com/rook/rook/pkg/daemon/ceph/client"
cephCluster "github.com/rook/rook/pkg/operator/ceph/cluster"
opcontroller "github.com/rook/rook/pkg/operator/ceph/controller"
"github.com/rook/rook/pkg/operator/ceph/disruption/controllerconfig"
"github.com/rook/rook/pkg/operator/ceph/disruption/machinelabel"
kerrors "k8s.io/apimachinery/pkg/api/errors"
@@ -105,7 +106,7 @@ func (r *MachineDisruptionReconciler) reconcile(request reconcile.Request) (reco
MDBCephClusterNamespaceLabelKey: request.Namespace,
MDBCephClusterNameLabelKey: request.Name,
},
OwnerReferences: []metav1.OwnerReference{cephCluster.ClusterOwnerRef(cephClusterInstance.GetName(), string(cephClusterInstance.GetUID()))},
OwnerReferences: []metav1.OwnerReference{opcontroller.ClusterOwnerRef(cephClusterInstance.GetName(), string(cephClusterInstance.GetUID()))},
},
Spec: healthchecking.MachineDisruptionBudgetSpec{
MaxUnavailable: &maxUnavailable,
@@ -1,71 +0,0 @@
/*
Copyright 2019 The Rook Authors. All rights reserved.
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
*/
// Package controllers contains all the controller-runtime controllers and
// exports
package disruption
import (
"github.com/pkg/errors"
"github.com/rook/rook/pkg/operator/ceph/disruption/clusterdisruption"
"github.com/rook/rook/pkg/operator/ceph/disruption/controllerconfig"
"github.com/rook/rook/pkg/operator/ceph/disruption/machinedisruption"
"github.com/rook/rook/pkg/operator/ceph/disruption/machinelabel"
"github.com/rook/rook/pkg/operator/ceph/disruption/nodedrain"
"sigs.k8s.io/controller-runtime/pkg/manager"
)
var (
EnableMachineDisruptionBudget bool
)
// AddToManagerFuncs is a list of functions to add all Controllers to the Manager (entrypoint for controller)
var AddToManagerFuncs = []func(manager.Manager, *controllerconfig.Context) error{
nodedrain.Add,
clusterdisruption.Add,
}
// MachineDisruptionBudgetAddToManagerFuncs is a list of fencing related functions to add all Controllers to the Manager (entrypoint for controller)
var MachineDisruptionBudgetAddToManagerFuncs = []func(manager.Manager, *controllerconfig.Context) error{
machinelabel.Add,
machinedisruption.Add,
}
// AddToManager adds all the registered controllers to the passed manager.
// each controller package will have an Add method listed in AddToManagerFuncs
// which will setup all the necessary watch
func AddToManager(m manager.Manager, c *controllerconfig.Context) error {
if c == nil {
return errors.New("nil controllercontext passed")
}
for _, f := range AddToManagerFuncs {
if err := f(m, c); err != nil {
return err
}
}
if EnableMachineDisruptionBudget {
for _, f := range MachineDisruptionBudgetAddToManagerFuncs {
if err := f(m, c); err != nil {
return err
}
}
}
return nil
}
+1
View File
@@ -101,6 +101,7 @@ func add(mgr manager.Manager, r reconcile.Reconciler) error {
if err != nil {
return err
}
logger.Info("successfully started")
// Watch for changes on the CephFilesystem CRD object
err = c.Watch(&source.Kind{Type: &cephv1.CephFilesystem{TypeMeta: controllerTypeMeta}}, &handler.EnqueueRequestForObject{}, opcontroller.WatchControllerPredicate())
+1
View File
@@ -100,6 +100,7 @@ func add(mgr manager.Manager, r reconcile.Reconciler) error {
if err != nil {
return err
}
logger.Info("successfully started")
// Watch for changes on the cephNFS CRD object
err = c.Watch(&source.Kind{Type: &cephv1.CephNFS{TypeMeta: controllerTypeMeta}}, &handler.EnqueueRequestForObject{}, opcontroller.WatchControllerPredicate())
+1
View File
@@ -101,6 +101,7 @@ func add(mgr manager.Manager, r reconcile.Reconciler) error {
if err != nil {
return err
}
logger.Info("successfully started")
// Watch for changes on the cephObjectStore CRD object
err = c.Watch(&source.Kind{Type: &cephv1.CephObjectStore{TypeMeta: controllerTypeMeta}}, &handler.EnqueueRequestForObject{}, opcontroller.WatchControllerPredicate())
+1 -8
View File
@@ -156,19 +156,12 @@ func (c *clusterConfig) startRGWPods() error {
return errors.Wrapf(createErr, "failed to create rgw deployment")
}
logger.Infof("object store %q deployment %q already exists. updating if needed", c.store.Name, deployment.Name)
_, err = c.context.Clientset.AppsV1().Deployments(c.store.Namespace).Get(deployment.Name, metav1.GetOptions{})
if err != nil {
return errors.Wrapf(err, "failed to get existing rgw deployment %q for update", deployment.Name)
}
}
// Generate the mime.types file after the rep. controller as well for the same reason as keyring
if createErr != nil && kerrors.IsAlreadyExists(createErr) {
if err := updateDeploymentAndWait(c.context, deployment, c.store.Namespace, config.RgwType, daemonLetterID, c.skipUpgradeChecks, c.clusterSpec.ContinueUpgradeAfterChecksEvenIfNotHealthy); err != nil {
return errors.Wrapf(err, "failed to update object store %q deployment %q", c.store.Name, deployment.Name)
}
}
// Generate the mime.types file after the rep. controller as well for the same reason as keyring
if err := c.generateMimeTypes(); err != nil {
return errors.Wrap(err, "failed to generate the rgw mime.types config")
}
@@ -95,6 +95,7 @@ func add(mgr manager.Manager, r reconcile.Reconciler) error {
if err != nil {
return err
}
logger.Info("successfully started")
// Watch for changes on the CephObjectStoreUser CRD object
err = c.Watch(&source.Kind{Type: &cephv1.CephObjectStoreUser{TypeMeta: controllerTypeMeta}}, &handler.EnqueueRequestForObject{}, opcontroller.WatchControllerPredicate())
+40 -33
View File
@@ -31,7 +31,7 @@ import (
"github.com/rook/rook/pkg/daemon/ceph/agent/flexvolume/attachment"
"github.com/rook/rook/pkg/operator/ceph/agent"
"github.com/rook/rook/pkg/operator/ceph/cluster"
cephController "github.com/rook/rook/pkg/operator/ceph/controller"
opcontroller "github.com/rook/rook/pkg/operator/ceph/controller"
"github.com/rook/rook/pkg/operator/ceph/csi"
"github.com/rook/rook/pkg/operator/ceph/provisioner"
"github.com/rook/rook/pkg/operator/discover"
@@ -56,15 +56,18 @@ var provisionerConfigs = map[string]string{
}
var (
// Whether to enable the flex driver. If true, the rook-ceph-agent daemonset will be started.
// EnableFlexDriver Whether to enable the flex driver. If true, the rook-ceph-agent daemonset will be started.
EnableFlexDriver = true
// Whether to enable the daemon for device discovery. If true, the rook-ceph-discover daemonset will be started.
// EnableDiscoveryDaemon Whether to enable the daemon for device discovery. If true, the rook-ceph-discover daemonset will be started.
EnableDiscoveryDaemon = true
// ImmediateRetryResult Return this for a immediate retry of the reconciliation loop with the same request object.
ImmediateRetryResult = reconcile.Result{Requeue: true}
// WaitForRequeueIfCephClusterNotReadyAfter requeue after 10sec if the operator is not ready
WaitForRequeueIfCephClusterNotReadyAfter = 10 * time.Second
// WaitForRequeueIfCephClusterNotReady waits for the CephCluster to be ready
WaitForRequeueIfCephClusterNotReady = reconcile.Result{Requeue: true, RequeueAfter: WaitForRequeueIfCephClusterNotReadyAfter}
)
@@ -84,7 +87,7 @@ type Operator struct {
// New creates an operator instance
func New(context *clusterd.Context, volumeAttachmentWrapper attachment.Attachment, rookImage, securityAccount string) *Operator {
schemes := []k8sutil.CustomResource{cluster.ClusterResource, attachment.VolumeResource}
schemes := []k8sutil.CustomResource{opcontroller.ClusterResource, attachment.VolumeResource}
operatorNamespace := os.Getenv(k8sutil.PodNamespaceEnvVar)
o := &Operator{
@@ -108,57 +111,61 @@ func New(context *clusterd.Context, volumeAttachmentWrapper attachment.Attachmen
func (o *Operator) Run() error {
if o.operatorNamespace == "" {
return errors.Errorf("rook operator namespace is not provided. expose it via downward API in the rook operator manifest file using environment variable %s", k8sutil.PodNamespaceEnvVar)
return errors.Errorf("rook operator namespace is not provided. expose it via downward API in the rook operator manifest file using environment variable %q", k8sutil.PodNamespaceEnvVar)
}
if EnableDiscoveryDaemon {
rookDiscover := discover.New(o.context.Clientset)
if err := rookDiscover.Start(o.operatorNamespace, o.rookImage, o.securityAccount, true); err != nil {
return errors.Wrapf(err, "error starting device discovery daemonset")
return errors.Wrap(err, "failed to start device discovery daemonset")
}
}
serverVersion, err := o.context.Clientset.Discovery().ServerVersion()
if err != nil {
return errors.Wrapf(err, "error getting server version")
return errors.Wrap(err, "failed to get server version")
}
// Initialize signal handler
signalChan := make(chan os.Signal, 1)
stopChan := make(chan struct{})
signal.Notify(signalChan, syscall.SIGINT, syscall.SIGTERM)
// Run volume provisioner for each of the supported configurations
for name, vendor := range provisionerConfigs {
volumeProvisioner := provisioner.New(o.context, vendor)
pc := controller.NewProvisionController(
o.context.Clientset,
name,
volumeProvisioner,
serverVersion.GitVersion,
)
go pc.Run(stopChan)
logger.Infof("rook-provisioner %s started using %s flex vendor dir", name, vendor)
// For Flex Driver, run volume provisioner for each of the supported configurations
if EnableFlexDriver {
for name, vendor := range provisionerConfigs {
volumeProvisioner := provisioner.New(o.context, vendor)
pc := controller.NewProvisionController(
o.context.Clientset,
name,
volumeProvisioner,
serverVersion.GitVersion,
)
go pc.Run(stopChan)
logger.Infof("rook-provisioner %q started using %q flex vendor dir", name, vendor)
}
}
var namespaceToWatch string
if os.Getenv("ROOK_CURRENT_NAMESPACE_ONLY") == "true" {
logger.Infof("Watching the current namespace for a cluster CRD")
logger.Infof("watching the current namespace for a ceph cluster CR")
namespaceToWatch = o.operatorNamespace
} else {
logger.Infof("Watching all namespaces for cluster CRDs")
logger.Infof("watching all namespaces for ceph cluster CRs")
namespaceToWatch = v1.NamespaceAll
}
// Start the controller-runtime Manager.
go o.startManager(namespaceToWatch, stopChan)
// watch for changes to the rook clusters
o.clusterController.StartWatch(namespaceToWatch, stopChan)
// Start the operator setting watcher
go o.clusterController.StartOperatorSettingsWatch(namespaceToWatch, stopChan)
// Signal handler to stop the operator
for {
select {
case <-signalChan:
logger.Infof("shutdown signal received, exiting...")
logger.Info("shutdown signal received, exiting...")
close(stopChan)
o.clusterController.StopWatch()
return nil
@@ -239,7 +246,7 @@ func (o *Operator) updateDrivers() error {
func (o *Operator) setCSIParams() error {
var err error
csiEnableRBD, err := k8sutil.GetOperatorSetting(o.context.Clientset, cephController.OperatorSettingConfigMapName, "ROOK_CSI_ENABLE_RBD", "true")
csiEnableRBD, err := k8sutil.GetOperatorSetting(o.context.Clientset, opcontroller.OperatorSettingConfigMapName, "ROOK_CSI_ENABLE_RBD", "true")
if err != nil {
return errors.Wrap(err, "unable to determine if CSI driver for RBD is enabled")
}
@@ -247,7 +254,7 @@ func (o *Operator) setCSIParams() error {
return errors.Wrap(err, "unable to parse value for 'ROOK_CSI_ENABLE_RBD'")
}
csiEnableCephFS, err := k8sutil.GetOperatorSetting(o.context.Clientset, cephController.OperatorSettingConfigMapName, "ROOK_CSI_ENABLE_CEPHFS", "true")
csiEnableCephFS, err := k8sutil.GetOperatorSetting(o.context.Clientset, opcontroller.OperatorSettingConfigMapName, "ROOK_CSI_ENABLE_CEPHFS", "true")
if err != nil {
return errors.Wrap(err, "unable to determine if CSI driver for CephFS is enabled")
}
@@ -255,7 +262,7 @@ func (o *Operator) setCSIParams() error {
return errors.Wrap(err, "unable to parse value for 'ROOK_CSI_ENABLE_CEPHFS'")
}
csiAllowUnsupported, err := k8sutil.GetOperatorSetting(o.context.Clientset, cephController.OperatorSettingConfigMapName, "ROOK_CSI_ALLOW_UNSUPPORTED_VERSION", "false")
csiAllowUnsupported, err := k8sutil.GetOperatorSetting(o.context.Clientset, opcontroller.OperatorSettingConfigMapName, "ROOK_CSI_ALLOW_UNSUPPORTED_VERSION", "false")
if err != nil {
return errors.Wrap(err, "unable to determine if unsupported version is allowed")
}
@@ -263,7 +270,7 @@ func (o *Operator) setCSIParams() error {
return errors.Wrap(err, "unable to parse value for 'ROOK_CSI_ALLOW_UNSUPPORTED_VERSION'")
}
csiEnableCSIGRPCMetrics, err := k8sutil.GetOperatorSetting(o.context.Clientset, cephController.OperatorSettingConfigMapName, "ROOK_CSI_ENABLE_GRPC_METRICS", "true")
csiEnableCSIGRPCMetrics, err := k8sutil.GetOperatorSetting(o.context.Clientset, opcontroller.OperatorSettingConfigMapName, "ROOK_CSI_ENABLE_GRPC_METRICS", "true")
if err != nil {
return errors.Wrap(err, "unable to determine if CSI GRPC metrics is enabled")
}
@@ -271,27 +278,27 @@ func (o *Operator) setCSIParams() error {
return errors.Wrap(err, "unable to parse value for 'ROOK_CSI_ENABLE_GRPC_METRICS'")
}
csi.CSIParam.CSIPluginImage, err = k8sutil.GetOperatorSetting(o.context.Clientset, cephController.OperatorSettingConfigMapName, "ROOK_CSI_CEPH_IMAGE", csi.DefaultCSIPluginImage)
csi.CSIParam.CSIPluginImage, err = k8sutil.GetOperatorSetting(o.context.Clientset, opcontroller.OperatorSettingConfigMapName, "ROOK_CSI_CEPH_IMAGE", csi.DefaultCSIPluginImage)
if err != nil {
return errors.Wrap(err, "unable to configure CSI plugin image")
}
csi.CSIParam.RegistrarImage, err = k8sutil.GetOperatorSetting(o.context.Clientset, cephController.OperatorSettingConfigMapName, "ROOK_CSI_REGISTRAR_IMAGE", csi.DefaultRegistrarImage)
csi.CSIParam.RegistrarImage, err = k8sutil.GetOperatorSetting(o.context.Clientset, opcontroller.OperatorSettingConfigMapName, "ROOK_CSI_REGISTRAR_IMAGE", csi.DefaultRegistrarImage)
if err != nil {
return errors.Wrap(err, "unable to configure CSI registrar image")
}
csi.CSIParam.ProvisionerImage, err = k8sutil.GetOperatorSetting(o.context.Clientset, cephController.OperatorSettingConfigMapName, "ROOK_CSI_PROVISIONER_IMAGE", csi.DefaultProvisionerImage)
csi.CSIParam.ProvisionerImage, err = k8sutil.GetOperatorSetting(o.context.Clientset, opcontroller.OperatorSettingConfigMapName, "ROOK_CSI_PROVISIONER_IMAGE", csi.DefaultProvisionerImage)
if err != nil {
return errors.Wrap(err, "unable to configure CSI provisioner image")
}
csi.CSIParam.AttacherImage, err = k8sutil.GetOperatorSetting(o.context.Clientset, cephController.OperatorSettingConfigMapName, "ROOK_CSI_ATTACHER_IMAGE", csi.DefaultAttacherImage)
csi.CSIParam.AttacherImage, err = k8sutil.GetOperatorSetting(o.context.Clientset, opcontroller.OperatorSettingConfigMapName, "ROOK_CSI_ATTACHER_IMAGE", csi.DefaultAttacherImage)
if err != nil {
return errors.Wrap(err, "unable to configure CSI attacher image")
}
csi.CSIParam.SnapshotterImage, err = k8sutil.GetOperatorSetting(o.context.Clientset, cephController.OperatorSettingConfigMapName, "ROOK_CSI_SNAPSHOTTER_IMAGE", csi.DefaultSnapshotterImage)
csi.CSIParam.SnapshotterImage, err = k8sutil.GetOperatorSetting(o.context.Clientset, opcontroller.OperatorSettingConfigMapName, "ROOK_CSI_SNAPSHOTTER_IMAGE", csi.DefaultSnapshotterImage)
if err != nil {
return errors.Wrap(err, "unable to configure CSI snapshotter image")
}
csi.CSIParam.KubeletDirPath, err = k8sutil.GetOperatorSetting(o.context.Clientset, cephController.OperatorSettingConfigMapName, "ROOK_CSI_KUBELET_DIR_PATH", csi.DefaultKubeletDirPath)
csi.CSIParam.KubeletDirPath, err = k8sutil.GetOperatorSetting(o.context.Clientset, opcontroller.OperatorSettingConfigMapName, "ROOK_CSI_KUBELET_DIR_PATH", csi.DefaultKubeletDirPath)
if err != nil {
return errors.Wrap(err, "unable to configure CSI kubelet directory path")
}
+2 -2
View File
@@ -22,7 +22,7 @@ import (
"github.com/rook/rook/pkg/clusterd"
"github.com/rook/rook/pkg/daemon/ceph/agent/flexvolume/attachment"
"github.com/rook/rook/pkg/operator/ceph/cluster"
opcontroller "github.com/rook/rook/pkg/operator/ceph/controller"
"github.com/rook/rook/pkg/operator/test"
"github.com/stretchr/testify/assert"
)
@@ -38,7 +38,7 @@ func TestOperator(t *testing.T) {
assert.Equal(t, context, o.context)
assert.Equal(t, len(o.resources), 2)
for _, r := range o.resources {
if r.Name != cluster.ClusterResource.Name && r.Name != attachment.VolumeResource.Name {
if r.Name != opcontroller.ClusterResource.Name && r.Name != attachment.VolumeResource.Name {
assert.Fail(t, fmt.Sprintf("Resource %s is not valid", r.Name))
}
}
+1
View File
@@ -93,6 +93,7 @@ func add(mgr manager.Manager, r reconcile.Reconciler) error {
if err != nil {
return err
}
logger.Info("successfully started")
// Watch for changes on the CephBlockPool CRD object
err = c.Watch(&source.Kind{Type: &cephv1.CephBlockPool{TypeMeta: controllerTypeMeta}}, &handler.EnqueueRequestForObject{}, opcontroller.WatchControllerPredicate())
+3
View File
@@ -40,6 +40,9 @@ type CustomResource struct {
// Kind is the serialized interface of the resource.
Kind string
// APIVersion is the full API version name (combine Group and Version)
APIVersion string
}
// WatchCR begins watching the custom resource (CRD). The call will block until a Done signal is raised during in the context.