forked from rook/rook
We now create and delete pools in parallel. Before this patch, the deletion: ``` 2021-06-08 13:48:26.557328 I | cephclient: no images/snapshosts present in pool "my-store.rgw.control" 2021-06-08 13:48:26.557383 I | cephclient: purging pool "my-store.rgw.control" (id=1) 2021-06-08 13:48:27.030021 I | ceph-object-controller: done disabling the dashboard api secret key 2021-06-08 13:48:28.800661 I | cephclient: purge completed for pool "my-store.rgw.control" 2021-06-08 13:48:29.085744 I | cephclient: no images/snapshosts present in pool "my-store.rgw.meta" 2021-06-08 13:48:29.085762 I | cephclient: purging pool "my-store.rgw.meta" (id=3) 2021-06-08 13:48:30.835487 I | cephclient: purge completed for pool "my-store.rgw.meta" 2021-06-08 13:48:31.128349 I | cephclient: no images/snapshosts present in pool "my-store.rgw.log" 2021-06-08 13:48:31.128368 I | cephclient: purging pool "my-store.rgw.log" (id=4) 2021-06-08 13:48:32.864376 I | cephclient: purge completed for pool "my-store.rgw.log" 2021-06-08 13:48:33.180923 I | cephclient: no images/snapshosts present in pool "my-store.rgw.buckets.index" 2021-06-08 13:48:33.181049 I | cephclient: purging pool "my-store.rgw.buckets.index" (id=5) 2021-06-08 13:48:34.892163 I | cephclient: purge completed for pool "my-store.rgw.buckets.index" 2021-06-08 13:48:35.179952 I | cephclient: no images/snapshosts present in pool "my-store.rgw.buckets.non-ec" 2021-06-08 13:48:35.179994 I | cephclient: purging pool "my-store.rgw.buckets.non-ec" (id=6) 2021-06-08 13:48:37.376067 I | cephclient: purge completed for pool "my-store.rgw.buckets.non-ec" 2021-06-08 13:48:37.665201 I | cephclient: no images/snapshosts present in pool "my-store.rgw.buckets.data" 2021-06-08 13:48:37.665218 I | cephclient: purging pool "my-store.rgw.buckets.data" (id=8) 2021-06-08 13:48:39.394717 I | cephclient: purge completed for pool "my-store.rgw.buckets.data" 2021-06-08 13:48:39.725528 I | cephclient: no images/snapshosts present in pool ".rgw.root" 2021-06-08 13:48:39.725567 I | cephclient: purging pool ".rgw.root" (id=7) 2021-06-08 13:48:41.424628 I | cephclient: purge completed for pool ".rgw.root" ``` It took 15sec to cleanup... Now with this patch: ``` 2021-06-08 14:15:25.621472 I | cephclient: no images/snapshosts present in pool "my-store.rgw.buckets.index" 2021-06-08 14:15:25.621570 I | cephclient: purging pool "my-store.rgw.buckets.index" (id=12) 2021-06-08 14:15:25.668708 I | cephclient: no images/snapshosts present in pool "my-store.rgw.log" 2021-06-08 14:15:25.668729 I | cephclient: purging pool "my-store.rgw.log" (id=11) 2021-06-08 14:15:25.693002 I | cephclient: no images/snapshosts present in pool "my-store.rgw.meta" 2021-06-08 14:15:25.693050 I | cephclient: purging pool "my-store.rgw.meta" (id=10) 2021-06-08 14:15:25.698830 I | cephclient: no images/snapshosts present in pool "my-store.rgw.control" 2021-06-08 14:15:25.698854 I | cephclient: purging pool "my-store.rgw.control" (id=9) 2021-06-08 14:15:25.701732 I | cephclient: no images/snapshosts present in pool "my-store.rgw.buckets.non-ec" 2021-06-08 14:15:25.701758 I | cephclient: purging pool "my-store.rgw.buckets.non-ec" (id=13) 2021-06-08 14:15:25.702816 I | cephclient: no images/snapshosts present in pool ".rgw.root" 2021-06-08 14:15:25.702836 I | cephclient: purging pool ".rgw.root" (id=14) 2021-06-08 14:15:25.717144 I | cephclient: no images/snapshosts present in pool "my-store.rgw.buckets.data" 2021-06-08 14:15:25.717212 I | cephclient: purging pool "my-store.rgw.buckets.data" (id=15) 2021-06-08 14:15:26.393246 I | ceph-object-controller: done disabling the dashboard api secret key ``` It tool around 1sec. The creation before this patch: ``` 2021-06-08 16:47:10.669484 I | ceph-spec: adding finalizer "cephobjectstore.ceph.rook.io" on "my-store" 2021-06-08 16:47:10.677253 E | ceph-object-controller: failed to set object store "rook-ceph/my-store" status to "Progressing". failed to update object "my-store" status: Operation cannot be fulfilled on cephobjectstores.ceph.rook.io "my-store": the object has been modified; please apply your changes to the latest version and try again 2021-06-08 16:47:10.682231 I | op-mon: parsing mon endpoints: b=10.111.63.108:6789,c=10.108.123.222:6789,a=10.100.223.211:6789 2021-06-08 16:47:10.985002 I | ceph-object-controller: reconciling object store deployments 2021-06-08 16:47:11.002364 I | ceph-object-controller: ceph object store gateway service running at 10.99.31.5 2021-06-08 16:47:11.002384 I | ceph-object-controller: reconciling object store pools 2021-06-08 16:47:14.336086 I | cephclient: setting pool property "compression_mode" to "none" on pool "my-store.rgw.control" 2021-06-08 16:47:16.115072 E | ceph-crashcollector-controller: node reconcile failed on op "unchanged": Operation cannot be fulfilled on deployments.apps "rook-ceph-crashcollector-minikube": the object has been modified; please apply your changes to the latest version and try again 2021-06-08 16:47:16.138224 E | ceph-crashcollector-controller: node reconcile failed on op "unchanged": Operation cannot be fulfilled on deployments.apps "rook-ceph-crashcollector-minikube": the object has been modified; please apply your changes to the latest version and try again 2021-06-08 16:47:16.391914 I | cephclient: creating replicated pool my-store.rgw.control succeeded 2021-06-08 16:47:16.391943 I | cephclient: setting pool property "pg_num_min" to "8" on pool "my-store.rgw.control" 2021-06-08 16:47:20.390562 I | cephclient: setting pool property "compression_mode" to "none" on pool "my-store.rgw.meta" 2021-06-08 16:47:22.423269 I | cephclient: creating replicated pool my-store.rgw.meta succeeded 2021-06-08 16:47:22.423294 I | cephclient: setting pool property "pg_num_min" to "8" on pool "my-store.rgw.meta" 2021-06-08 16:47:26.502874 I | cephclient: setting pool property "compression_mode" to "none" on pool "my-store.rgw.log" 2021-06-08 16:47:28.555252 I | cephclient: creating replicated pool my-store.rgw.log succeeded 2021-06-08 16:47:28.555308 I | cephclient: setting pool property "pg_num_min" to "8" on pool "my-store.rgw.log" 2021-06-08 16:47:32.619546 I | cephclient: setting pool property "compression_mode" to "none" on pool "my-store.rgw.buckets.index" 2021-06-08 16:47:34.663020 I | cephclient: creating replicated pool my-store.rgw.buckets.index succeeded 2021-06-08 16:47:34.663050 I | cephclient: setting pool property "pg_num_min" to "8" on pool "my-store.rgw.buckets.index" 2021-06-08 16:47:38.712886 I | cephclient: setting pool property "compression_mode" to "none" on pool "my-store.rgw.buckets.non-ec" 2021-06-08 16:47:40.748430 I | cephclient: creating replicated pool my-store.rgw.buckets.non-ec succeeded 2021-06-08 16:47:40.748467 I | cephclient: setting pool property "pg_num_min" to "8" on pool "my-store.rgw.buckets.non-ec" 2021-06-08 16:47:44.830659 I | cephclient: setting pool property "compression_mode" to "none" on pool ".rgw.root" 2021-06-08 16:47:46.875834 I | cephclient: creating replicated pool .rgw.root succeeded 2021-06-08 16:47:46.875863 I | cephclient: setting pool property "pg_num_min" to "8" on pool ".rgw.root" 2021-06-08 16:47:50.933717 I | cephclient: setting pool property "compression_mode" to "none" on pool "my-store.rgw.buckets.data" 2021-06-08 16:47:52.969461 I | cephclient: creating replicated pool my-store.rgw.buckets.data succeeded 2021-06-08 16:47:52.969549 I | ceph-object-controller: setting multisite settings for object store "my-store" 2021-06-08 16:47:53.585737 I | ceph-object-controller: Multisite for object-store: realm=my-store, zonegroup=my-store, zone=my-store 2021-06-08 16:47:53.585777 I | ceph-object-controller: multisite configuration for object-store my-store is complete 2021-06-08 16:47:53.585787 I | ceph-object-controller: creating object store "my-store" in namespace "rook-ceph" 2021-06-08 16:47:53.585802 I | cephclient: getting or creating ceph auth key "client.rgw.my.store.a" 2021-06-08 16:47:53.930416 I | ceph-object-controller: setting rgw config flags 2021-06-08 16:47:53.931363 I | op-config: setting "client.rgw.my.store.a"="rgw_enable_usage_log"="true" option to the mon configuration database 2021-06-08 16:47:54.201200 I | op-config: successfully set "client.rgw.my.store.a"="rgw_enable_usage_log"="true" option to the mon configuration database 2021-06-08 16:47:54.201225 I | op-config: setting "client.rgw.my.store.a"="rgw_zone"="my-store" option to the mon configuration database 2021-06-08 16:47:54.454589 I | op-config: successfully set "client.rgw.my.store.a"="rgw_zone"="my-store" option to the mon configuration database 2021-06-08 16:47:54.454614 I | op-config: setting "client.rgw.my.store.a"="rgw_zonegroup"="my-store" option to the mon configuration database 2021-06-08 16:47:54.710218 I | op-config: successfully set "client.rgw.my.store.a"="rgw_zonegroup"="my-store" option to the mon configuration database 2021-06-08 16:47:54.710237 I | op-config: setting "client.rgw.my.store.a"="rgw_log_nonexistent_bucket"="true" option to the mon configuration database 2021-06-08 16:47:54.969049 I | op-config: successfully set "client.rgw.my.store.a"="rgw_log_nonexistent_bucket"="true" option to the mon configuration database 2021-06-08 16:47:54.969069 I | op-config: setting "client.rgw.my.store.a"="rgw_log_object_name_utc"="true" option to the mon configuration database 2021-06-08 16:47:55.255491 I | op-config: successfully set "client.rgw.my.store.a"="rgw_log_object_name_utc"="true" option to the mon configuration database 2021-06-08 16:47:55.255631 I | ceph-object-controller: object store "my-store" deployment "rook-ceph-rgw-my-store-a" started 2021-06-08 16:47:55.279316 I | ceph-object-controller: enabling rgw dashboard 2021-06-08 16:47:55.361208 E | ceph-crashcollector-controller: node reconcile failed on op "unchanged": Operation cannot be fulfilled on deployments.apps "rook-ceph-crashcollector-minikube": the object has been modified; please apply your changes to the latest version and try again 2021-06-08 16:47:56.225904 I | ceph-object-controller: setting the dashboard api secret key 2021-06-08 16:47:56.225984 I | ceph-object-controller: created object store "my-store" in namespace "rook-ceph" 2021-06-08 16:47:56.589452 I | ceph-object-controller: starting rgw healthcheck 2021-06-08 16:47:56.650045 I | ceph-object-controller: done setting the dashboard api secret key ``` It took 46sec, and I've seen it taking almost a 1min sometimes. Now with this patch: ``` 2021-06-08 16:51:35.259558 I | ceph-spec: adding finalizer "cephobjectstore.ceph.rook.io" on "my-store" 2021-06-08 16:51:35.270524 E | ceph-object-controller: failed to set object store "rook-ceph/my-store" status to "Progressing". failed to update object "my-store" status: Operation cannot be fulfilled on cephobjectstores.ceph.rook.io "my-store": the object has been modified; please apply your changes to the latest version and try again 2021-06-08 16:51:35.274387 I | op-mon: parsing mon endpoints: b=10.111.63.108:6789,c=10.108.123.222:6789,a=10.100.223.211:6789 2021-06-08 16:51:35.599023 I | ceph-object-controller: reconciling object store deployments 2021-06-08 16:51:35.607256 I | ceph-object-controller: ceph object store gateway service running at 10.104.254.110 2021-06-08 16:51:35.607315 I | ceph-object-controller: reconciling object store pools 2021-06-08 16:51:39.337735 I | cephclient: setting pool property "compression_mode" to "none" on pool "my-store.rgw.buckets.non-ec" 2021-06-08 16:51:40.343290 I | cephclient: setting pool property "compression_mode" to "none" on pool ".rgw.root" 2021-06-08 16:51:40.346335 I | cephclient: setting pool property "compression_mode" to "none" on pool "my-store.rgw.buckets.index" 2021-06-08 16:51:40.362670 I | cephclient: setting pool property "compression_mode" to "none" on pool "my-store.rgw.meta" 2021-06-08 16:51:40.363445 I | cephclient: setting pool property "compression_mode" to "none" on pool "my-store.rgw.log" 2021-06-08 16:51:40.364321 I | cephclient: setting pool property "compression_mode" to "none" on pool "my-store.rgw.control" 2021-06-08 16:51:41.412870 I | cephclient: creating replicated pool my-store.rgw.buckets.non-ec succeeded 2021-06-08 16:51:41.412904 I | cephclient: setting pool property "pg_num_min" to "8" on pool "my-store.rgw.buckets.non-ec" 2021-06-08 16:51:42.460439 I | cephclient: creating replicated pool my-store.rgw.control succeeded 2021-06-08 16:51:42.460503 I | cephclient: setting pool property "pg_num_min" to "8" on pool "my-store.rgw.control" 2021-06-08 16:51:42.470666 I | cephclient: creating replicated pool my-store.rgw.meta succeeded 2021-06-08 16:51:42.470727 I | cephclient: setting pool property "pg_num_min" to "8" on pool "my-store.rgw.meta" 2021-06-08 16:51:42.473029 I | cephclient: creating replicated pool .rgw.root succeeded 2021-06-08 16:51:42.473064 I | cephclient: setting pool property "pg_num_min" to "8" on pool ".rgw.root" 2021-06-08 16:51:42.473895 I | cephclient: creating replicated pool my-store.rgw.buckets.index succeeded 2021-06-08 16:51:42.473923 I | cephclient: setting pool property "pg_num_min" to "8" on pool "my-store.rgw.buckets.index" 2021-06-08 16:51:42.478625 I | cephclient: creating replicated pool my-store.rgw.log succeeded 2021-06-08 16:51:42.478671 I | cephclient: setting pool property "pg_num_min" to "8" on pool "my-store.rgw.log" 2021-06-08 16:51:46.606905 I | cephclient: setting pool property "compression_mode" to "none" on pool "my-store.rgw.buckets.data" 2021-06-08 16:51:48.627320 I | cephclient: creating replicated pool my-store.rgw.buckets.data succeeded 2021-06-08 16:51:48.627368 I | ceph-object-controller: setting multisite settings for object store "my-store" 2021-06-08 16:51:49.325086 I | ceph-object-controller: Multisite for object-store: realm=my-store, zonegroup=my-store, zone=my-store 2021-06-08 16:51:49.325108 I | ceph-object-controller: multisite configuration for object-store my-store is complete 2021-06-08 16:51:49.325121 I | ceph-object-controller: creating object store "my-store" in namespace "rook-ceph" 2021-06-08 16:51:49.325134 I | cephclient: getting or creating ceph auth key "client.rgw.my.store.a" 2021-06-08 16:51:49.657898 I | ceph-object-controller: setting rgw config flags 2021-06-08 16:51:49.657920 I | op-config: setting "client.rgw.my.store.a"="rgw_log_object_name_utc"="true" option to the mon configuration database 2021-06-08 16:51:49.917898 I | op-config: successfully set "client.rgw.my.store.a"="rgw_log_object_name_utc"="true" option to the mon configuration database 2021-06-08 16:51:49.917917 I | op-config: setting "client.rgw.my.store.a"="rgw_enable_usage_log"="true" option to the mon configuration database 2021-06-08 16:51:50.184939 I | op-config: successfully set "client.rgw.my.store.a"="rgw_enable_usage_log"="true" option to the mon configuration database 2021-06-08 16:51:50.184970 I | op-config: setting "client.rgw.my.store.a"="rgw_zone"="my-store" option to the mon configuration database 2021-06-08 16:51:50.463799 I | op-config: successfully set "client.rgw.my.store.a"="rgw_zone"="my-store" option to the mon configuration database 2021-06-08 16:51:50.463821 I | op-config: setting "client.rgw.my.store.a"="rgw_zonegroup"="my-store" option to the mon configuration database 2021-06-08 16:51:50.720953 I | op-config: successfully set "client.rgw.my.store.a"="rgw_zonegroup"="my-store" option to the mon configuration database 2021-06-08 16:51:50.720999 I | op-config: setting "client.rgw.my.store.a"="rgw_log_nonexistent_bucket"="true" option to the mon configuration database 2021-06-08 16:51:50.974875 I | op-config: successfully set "client.rgw.my.store.a"="rgw_log_nonexistent_bucket"="true" option to the mon configuration database 2021-06-08 16:51:50.975024 I | ceph-object-controller: object store "my-store" deployment "rook-ceph-rgw-my-store-a" started 2021-06-08 16:51:51.019820 I | ceph-object-controller: enabling rgw dashboard 2021-06-08 16:51:51.980117 I | ceph-object-controller: created object store "my-store" in namespace "rook-ceph" 2021-06-08 16:51:51.980199 I | ceph-object-controller: setting the dashboard api secret key 2021-06-08 16:51:52.399997 I | ceph-object-controller: done setting the dashboard api secret key 2021-06-08 16:51:52.435936 I | ceph-object-controller: starting rgw healthcheck ``` It took 17sec. Signed-off-by: Sébastien Han <seb@redhat.com>
566 lines
22 KiB
Go
566 lines
22 KiB
Go
/*
|
|
Copyright 2016 The Rook Authors. All rights reserved.
|
|
|
|
Licensed under the Apache License, Version 2.0 (the "License");
|
|
you may not use this file except in compliance with the License.
|
|
You may obtain a copy of the License at
|
|
|
|
http://www.apache.org/licenses/LICENSE-2.0
|
|
|
|
Unless required by applicable law or agreed to in writing, software
|
|
distributed under the License is distributed on an "AS IS" BASIS,
|
|
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
See the License for the specific language governing permissions and
|
|
limitations under the License.
|
|
*/
|
|
|
|
package object
|
|
|
|
import (
|
|
"context"
|
|
"fmt"
|
|
"reflect"
|
|
"strings"
|
|
"syscall"
|
|
"time"
|
|
|
|
"github.com/coreos/pkg/capnslog"
|
|
bktclient "github.com/kube-object-storage/lib-bucket-provisioner/pkg/client/clientset/versioned"
|
|
"github.com/pkg/errors"
|
|
cephv1 "github.com/rook/rook/pkg/apis/ceph.rook.io/v1"
|
|
"github.com/rook/rook/pkg/clusterd"
|
|
cephclient "github.com/rook/rook/pkg/daemon/ceph/client"
|
|
"github.com/rook/rook/pkg/operator/ceph/cluster/mgr"
|
|
"github.com/rook/rook/pkg/operator/ceph/cluster/mon"
|
|
opconfig "github.com/rook/rook/pkg/operator/ceph/config"
|
|
opcontroller "github.com/rook/rook/pkg/operator/ceph/controller"
|
|
"github.com/rook/rook/pkg/operator/k8sutil"
|
|
"github.com/rook/rook/pkg/util/exec"
|
|
appsv1 "k8s.io/api/apps/v1"
|
|
corev1 "k8s.io/api/core/v1"
|
|
kerrors "k8s.io/apimachinery/pkg/api/errors"
|
|
metav1 "k8s.io/apimachinery/pkg/apis/meta/v1"
|
|
"k8s.io/apimachinery/pkg/runtime"
|
|
"k8s.io/apimachinery/pkg/types"
|
|
"sigs.k8s.io/controller-runtime/pkg/client"
|
|
"sigs.k8s.io/controller-runtime/pkg/controller"
|
|
"sigs.k8s.io/controller-runtime/pkg/handler"
|
|
"sigs.k8s.io/controller-runtime/pkg/manager"
|
|
"sigs.k8s.io/controller-runtime/pkg/reconcile"
|
|
"sigs.k8s.io/controller-runtime/pkg/source"
|
|
)
|
|
|
|
const (
|
|
controllerName = "ceph-object-controller"
|
|
)
|
|
|
|
var waitForRequeueIfObjectStoreNotReady = reconcile.Result{Requeue: true, RequeueAfter: 10 * time.Second}
|
|
|
|
var logger = capnslog.NewPackageLogger("github.com/rook/rook", controllerName)
|
|
|
|
// List of object resources to watch by the controller
|
|
var objectsToWatch = []client.Object{
|
|
&corev1.Secret{TypeMeta: metav1.TypeMeta{Kind: "Secret", APIVersion: corev1.SchemeGroupVersion.String()}},
|
|
&corev1.Service{TypeMeta: metav1.TypeMeta{Kind: "Service", APIVersion: corev1.SchemeGroupVersion.String()}},
|
|
&appsv1.Deployment{TypeMeta: metav1.TypeMeta{Kind: "Deployment", APIVersion: appsv1.SchemeGroupVersion.String()}},
|
|
}
|
|
|
|
var cephObjectStoreKind = reflect.TypeOf(cephv1.CephObjectStore{}).Name()
|
|
|
|
// Sets the type meta for the controller main object
|
|
var controllerTypeMeta = metav1.TypeMeta{
|
|
Kind: cephObjectStoreKind,
|
|
APIVersion: fmt.Sprintf("%s/%s", cephv1.CustomResourceGroup, cephv1.Version),
|
|
}
|
|
|
|
// ReconcileCephObjectStore reconciles a cephObjectStore object
|
|
type ReconcileCephObjectStore struct {
|
|
client client.Client
|
|
bktclient bktclient.Interface
|
|
scheme *runtime.Scheme
|
|
context *clusterd.Context
|
|
clusterSpec *cephv1.ClusterSpec
|
|
clusterInfo *cephclient.ClusterInfo
|
|
objectStoreChannels map[string]*objectStoreHealth
|
|
}
|
|
|
|
type objectStoreHealth struct {
|
|
stopChan chan struct{}
|
|
monitoringRunning bool
|
|
}
|
|
|
|
// Add creates a new cephObjectStore Controller and adds it to the Manager. The Manager will set fields on the Controller
|
|
// and Start it when the Manager is Started.
|
|
func Add(mgr manager.Manager, context *clusterd.Context) error {
|
|
return add(mgr, newReconciler(mgr, context))
|
|
}
|
|
|
|
// newReconciler returns a new reconcile.Reconciler
|
|
func newReconciler(mgr manager.Manager, context *clusterd.Context) reconcile.Reconciler {
|
|
// Add the cephv1 scheme to the manager scheme so that the controller knows about it
|
|
mgrScheme := mgr.GetScheme()
|
|
if err := cephv1.AddToScheme(mgr.GetScheme()); err != nil {
|
|
panic(err)
|
|
}
|
|
context.Client = mgr.GetClient()
|
|
return &ReconcileCephObjectStore{
|
|
client: mgr.GetClient(),
|
|
scheme: mgrScheme,
|
|
context: context,
|
|
bktclient: bktclient.NewForConfigOrDie(context.KubeConfig),
|
|
objectStoreChannels: make(map[string]*objectStoreHealth),
|
|
}
|
|
}
|
|
|
|
func add(mgr manager.Manager, r reconcile.Reconciler) error {
|
|
// Create a new controller
|
|
c, err := controller.New(controllerName, mgr, controller.Options{Reconciler: r})
|
|
if err != nil {
|
|
return err
|
|
}
|
|
logger.Info("successfully started")
|
|
|
|
// Watch for changes on the cephObjectStore CRD object
|
|
err = c.Watch(&source.Kind{Type: &cephv1.CephObjectStore{TypeMeta: controllerTypeMeta}}, &handler.EnqueueRequestForObject{}, opcontroller.WatchControllerPredicate())
|
|
if err != nil {
|
|
return err
|
|
}
|
|
|
|
// Watch all other resources
|
|
for _, t := range objectsToWatch {
|
|
err = c.Watch(&source.Kind{Type: t}, &handler.EnqueueRequestForOwner{
|
|
IsController: true,
|
|
OwnerType: &cephv1.CephObjectStore{},
|
|
}, opcontroller.WatchPredicateForNonCRDObject(&cephv1.CephObjectStore{TypeMeta: controllerTypeMeta}, mgr.GetScheme()))
|
|
if err != nil {
|
|
return err
|
|
}
|
|
}
|
|
|
|
// Build Handler function to return the list of ceph object
|
|
// This is used by the watchers below
|
|
handlerFunc, err := opcontroller.ObjectToCRMapper(mgr.GetClient(), &cephv1.CephObjectStoreList{}, mgr.GetScheme())
|
|
if err != nil {
|
|
return err
|
|
}
|
|
|
|
// Watch for CephCluster Spec changes that we want to propagate to us
|
|
err = c.Watch(&source.Kind{Type: &cephv1.CephCluster{
|
|
TypeMeta: metav1.TypeMeta{
|
|
Kind: opcontroller.ClusterResource.Kind,
|
|
APIVersion: opcontroller.ClusterResource.APIVersion,
|
|
},
|
|
},
|
|
}, handler.EnqueueRequestsFromMapFunc(handlerFunc), opcontroller.WatchCephClusterPredicate())
|
|
if err != nil {
|
|
return err
|
|
}
|
|
|
|
return nil
|
|
}
|
|
|
|
// Reconcile reads that state of the cluster for a cephObjectStore object and makes changes based on the state read
|
|
// and what is in the cephObjectStore.Spec
|
|
// The Controller will requeue the Request to be processed again if the returned error is non-nil or
|
|
// Result.Requeue is true, otherwise upon completion it will remove the work from the queue.
|
|
func (r *ReconcileCephObjectStore) Reconcile(context context.Context, request reconcile.Request) (reconcile.Result, error) {
|
|
// workaround because the rook logging mechanism is not compatible with the controller-runtime logging interface
|
|
reconcileResponse, err := r.reconcile(request)
|
|
if err != nil {
|
|
logger.Errorf("failed to reconcile %v", err)
|
|
}
|
|
|
|
return reconcileResponse, err
|
|
}
|
|
|
|
func (r *ReconcileCephObjectStore) reconcile(request reconcile.Request) (reconcile.Result, error) {
|
|
// Fetch the cephObjectStore instance
|
|
cephObjectStore := &cephv1.CephObjectStore{}
|
|
err := r.client.Get(context.TODO(), request.NamespacedName, cephObjectStore)
|
|
if err != nil {
|
|
if kerrors.IsNotFound(err) {
|
|
logger.Debug("cephObjectStore resource not found. Ignoring since object must be deleted.")
|
|
return reconcile.Result{}, nil
|
|
}
|
|
// Error reading the object - requeue the request.
|
|
return reconcile.Result{}, errors.Wrap(err, "failed to get cephObjectStore")
|
|
}
|
|
|
|
// Set a finalizer so we can do cleanup before the object goes away
|
|
err = opcontroller.AddFinalizerIfNotPresent(r.client, cephObjectStore)
|
|
if err != nil {
|
|
return reconcile.Result{}, errors.Wrap(err, "failed to add finalizer")
|
|
}
|
|
|
|
// The CR was just created, initializing status fields
|
|
if cephObjectStore.Status == nil {
|
|
// The store is not available so let's not build the status Info yet
|
|
updateStatus(r.client, request.NamespacedName, cephv1.ConditionProgressing, map[string]string{})
|
|
}
|
|
|
|
// Make sure a CephCluster is present otherwise do nothing
|
|
cephCluster, isReadyToReconcile, cephClusterExists, reconcileResponse := opcontroller.IsReadyToReconcile(r.client, r.context, request.NamespacedName, controllerName)
|
|
if !isReadyToReconcile {
|
|
// This handles the case where the Ceph Cluster is gone and we want to delete that CR
|
|
// We skip the deleteStore() function since everything is gone already
|
|
//
|
|
// Also, only remove the finalizer if the CephCluster is gone
|
|
// If not, we should wait for it to be ready
|
|
// This handles the case where the operator is not ready to accept Ceph command but the cluster exists
|
|
if !cephObjectStore.GetDeletionTimestamp().IsZero() && !cephClusterExists {
|
|
// Remove finalizer
|
|
err := opcontroller.RemoveFinalizer(r.client, cephObjectStore)
|
|
if err != nil {
|
|
return reconcile.Result{}, errors.Wrap(err, "failed to remove finalizer")
|
|
}
|
|
|
|
// Return and do not requeue. Successful deletion.
|
|
return reconcile.Result{}, nil
|
|
}
|
|
|
|
return reconcileResponse, nil
|
|
}
|
|
r.clusterSpec = &cephCluster.Spec
|
|
|
|
// Initialize the channel for this object store
|
|
// This allows us to track multiple ObjectStores in the same namespace
|
|
_, ok := r.objectStoreChannels[cephObjectStore.Name]
|
|
if !ok {
|
|
r.objectStoreChannels[cephObjectStore.Name] = &objectStoreHealth{
|
|
stopChan: make(chan struct{}),
|
|
monitoringRunning: false,
|
|
}
|
|
}
|
|
|
|
// Populate clusterInfo during each reconcile
|
|
r.clusterInfo, _, _, err = mon.LoadClusterInfo(r.context, request.NamespacedName.Namespace)
|
|
if err != nil {
|
|
return reconcile.Result{}, errors.Wrap(err, "failed to populate cluster info")
|
|
}
|
|
|
|
// Populate CephVersion
|
|
currentCephVersion, err := cephclient.LeastUptodateDaemonVersion(r.context, r.clusterInfo, opconfig.MonType)
|
|
if err != nil {
|
|
if strings.Contains(err.Error(), opcontroller.UninitializedCephConfigError) {
|
|
logger.Info(opcontroller.OperatorNotInitializedMessage)
|
|
return opcontroller.WaitForRequeueIfOperatorNotInitialized, nil
|
|
}
|
|
return reconcile.Result{}, errors.Wrapf(err, "failed to retrieve current ceph %q version", opconfig.MonType)
|
|
}
|
|
r.clusterInfo.CephVersion = currentCephVersion
|
|
|
|
// DELETE: the CR was deleted
|
|
if !cephObjectStore.GetDeletionTimestamp().IsZero() {
|
|
logger.Debugf("deleting store %q", cephObjectStore.Name)
|
|
|
|
if ok {
|
|
select {
|
|
case <-r.objectStoreChannels[cephObjectStore.Name].stopChan:
|
|
// channel was closed
|
|
break
|
|
default:
|
|
// Close the channel to stop the healthcheck of the endpoint
|
|
close(r.objectStoreChannels[cephObjectStore.Name].stopChan)
|
|
}
|
|
|
|
response, okToDelete := r.verifyObjectBucketCleanup(cephObjectStore)
|
|
if !okToDelete {
|
|
// If the object store cannot be deleted, requeue the request for deletion to see if the conditions
|
|
// will eventually be satisfied such as the object buckets being removed
|
|
return response, nil
|
|
}
|
|
|
|
response, okToDelete = r.verifyObjectUserCleanup(cephObjectStore)
|
|
if !okToDelete {
|
|
// If the object store cannot be deleted, requeue the request for deletion to see if the conditions
|
|
// will eventually be satisfied such as the object users being removed
|
|
return response, nil
|
|
}
|
|
|
|
cfg := clusterConfig{
|
|
context: r.context,
|
|
store: cephObjectStore,
|
|
clusterSpec: r.clusterSpec,
|
|
clusterInfo: r.clusterInfo,
|
|
}
|
|
cfg.deleteStore()
|
|
|
|
// Remove object store from the map
|
|
delete(r.objectStoreChannels, cephObjectStore.Name)
|
|
}
|
|
|
|
// Remove finalizer
|
|
err = opcontroller.RemoveFinalizer(r.client, cephObjectStore)
|
|
if err != nil {
|
|
return reconcile.Result{}, errors.Wrap(err, "failed to remove finalizer")
|
|
}
|
|
|
|
// Return and do not requeue. Successful deletion.
|
|
return reconcile.Result{}, nil
|
|
}
|
|
|
|
// validate the store settings
|
|
if err := r.validateStore(cephObjectStore); err != nil {
|
|
return reconcile.Result{}, errors.Wrapf(err, "invalid object store %q arguments", cephObjectStore.Name)
|
|
}
|
|
|
|
// If the CephCluster has enabled the "pg_autoscaler" module and is running Nautilus
|
|
// we force the pg_autoscale_mode to "on"
|
|
_, propertyExists := cephObjectStore.Spec.DataPool.Parameters[cephclient.PgAutoscaleModeProperty]
|
|
if mgr.IsModuleInSpec(cephCluster.Spec.Mgr.Modules, mgr.PgautoscalerModuleName) &&
|
|
!currentCephVersion.IsAtLeastOctopus() &&
|
|
!propertyExists {
|
|
if len(cephObjectStore.Spec.DataPool.Parameters) == 0 {
|
|
cephObjectStore.Spec.DataPool.Parameters = make(map[string]string)
|
|
}
|
|
cephObjectStore.Spec.DataPool.Parameters[cephclient.PgAutoscaleModeProperty] = cephclient.PgAutoscaleModeOn
|
|
}
|
|
|
|
// CREATE/UPDATE
|
|
_, err = r.reconcileCreateObjectStore(cephObjectStore, request.NamespacedName, cephCluster.Spec)
|
|
if err != nil {
|
|
return r.setFailedStatus(request.NamespacedName, "failed to create object store deployments", err)
|
|
}
|
|
|
|
// Set Progressing status, we are done reconciling, the health check go routine will update the status
|
|
updateStatus(r.client, request.NamespacedName, cephv1.ConditionProgressing, buildStatusInfo(cephObjectStore))
|
|
|
|
// Return and do not requeue
|
|
logger.Debug("done reconciling")
|
|
return reconcile.Result{}, nil
|
|
}
|
|
|
|
func (r *ReconcileCephObjectStore) reconcileCreateObjectStore(cephObjectStore *cephv1.CephObjectStore, namespacedName types.NamespacedName, cluster cephv1.ClusterSpec) (reconcile.Result, error) {
|
|
ownerInfo := k8sutil.NewOwnerInfo(cephObjectStore, r.scheme)
|
|
cfg := clusterConfig{
|
|
context: r.context,
|
|
clusterInfo: r.clusterInfo,
|
|
store: cephObjectStore,
|
|
rookVersion: r.clusterSpec.CephVersion.Image,
|
|
clusterSpec: r.clusterSpec,
|
|
DataPathMap: opconfig.NewStatelessDaemonDataPathMap(opconfig.RgwType, cephObjectStore.Name, cephObjectStore.Namespace, r.clusterSpec.DataDirHostPath),
|
|
client: r.client,
|
|
ownerInfo: ownerInfo,
|
|
}
|
|
objContext := NewContext(r.context, r.clusterInfo, cephObjectStore.Name)
|
|
objContext.UID = string(cephObjectStore.UID)
|
|
|
|
var err error
|
|
|
|
if cephObjectStore.Spec.IsExternal() {
|
|
logger.Info("reconciling external object store")
|
|
|
|
// RECONCILE SERVICE
|
|
logger.Info("reconciling object store service")
|
|
_, err = cfg.reconcileService(cephObjectStore)
|
|
if err != nil {
|
|
return r.setFailedStatus(namespacedName, "failed to reconcile service", err)
|
|
}
|
|
|
|
// RECONCILE ENDPOINTS
|
|
// Always add the endpoint AFTER the service otherwise it will get overridden
|
|
logger.Info("reconciling external object store endpoint")
|
|
err = cfg.reconcileExternalEndpoint(cephObjectStore)
|
|
if err != nil {
|
|
return r.setFailedStatus(namespacedName, "failed to reconcile external endpoint", err)
|
|
}
|
|
} else {
|
|
logger.Info("reconciling object store deployments")
|
|
|
|
// Reconcile realm/zonegroup/zone CRs & update their names
|
|
realmName, zoneGroupName, zoneName, reconcileResponse, err := r.reconcileMultisiteCRs(cephObjectStore)
|
|
if err != nil {
|
|
return reconcileResponse, err
|
|
}
|
|
|
|
// Reconcile Ceph Zone if Multisite
|
|
if cephObjectStore.Spec.IsMultisite() {
|
|
reconcileResponse, err := r.reconcileCephZone(cephObjectStore, zoneGroupName, realmName)
|
|
if err != nil {
|
|
return reconcileResponse, err
|
|
}
|
|
}
|
|
|
|
objContext.Realm = realmName
|
|
objContext.ZoneGroup = zoneGroupName
|
|
objContext.Zone = zoneName
|
|
logger.Debugf("realm for object-store is %q, zone group for object-store is %q, zone for object-store is %q", objContext.Realm, objContext.ZoneGroup, objContext.Zone)
|
|
|
|
// RECONCILE SERVICE
|
|
logger.Debug("reconciling object store service")
|
|
serviceIP, err := cfg.reconcileService(cephObjectStore)
|
|
if err != nil {
|
|
return r.setFailedStatus(namespacedName, "failed to reconcile service", err)
|
|
}
|
|
|
|
// Reconcile Pool Creation
|
|
if !cephObjectStore.Spec.IsMultisite() {
|
|
logger.Info("reconciling object store pools")
|
|
err = CreatePools(objContext, r.clusterSpec, cephObjectStore.Spec.MetadataPool, cephObjectStore.Spec.DataPool)
|
|
if err != nil {
|
|
return r.setFailedStatus(namespacedName, "failed to create object pools", err)
|
|
}
|
|
}
|
|
|
|
// Reconcile Multisite Creation
|
|
logger.Infof("setting multisite settings for object store %q", cephObjectStore.Name)
|
|
err = setMultisite(objContext, cephObjectStore, serviceIP)
|
|
if err != nil {
|
|
return r.setFailedStatus(namespacedName, "failed to configure multisite for object store", err)
|
|
}
|
|
|
|
// Create or Update Store
|
|
err = cfg.createOrUpdateStore(realmName, zoneGroupName, zoneName)
|
|
if err != nil {
|
|
return reconcile.Result{}, errors.Wrapf(err, "failed to create object store %q", cephObjectStore.Name)
|
|
}
|
|
}
|
|
|
|
// Start monitoring
|
|
if !cephObjectStore.Spec.HealthCheck.Bucket.Disabled {
|
|
r.startMonitoring(cephObjectStore, objContext, namespacedName)
|
|
}
|
|
|
|
return reconcile.Result{}, nil
|
|
}
|
|
|
|
func (r *ReconcileCephObjectStore) reconcileCephZone(store *cephv1.CephObjectStore, zoneGroupName string, realmName string) (reconcile.Result, error) {
|
|
realmArg := fmt.Sprintf("--rgw-realm=%s", realmName)
|
|
zoneGroupArg := fmt.Sprintf("--rgw-zonegroup=%s", zoneGroupName)
|
|
zoneArg := fmt.Sprintf("--rgw-zone=%s", store.Spec.Zone.Name)
|
|
objContext := NewContext(r.context, r.clusterInfo, store.Name)
|
|
|
|
_, err := RunAdminCommandNoMultisite(objContext, true, "zone", "get", realmArg, zoneGroupArg, zoneArg)
|
|
if err != nil {
|
|
if code, ok := exec.ExitStatus(err); ok && code == int(syscall.ENOENT) {
|
|
return waitForRequeueIfObjectStoreNotReady, errors.Wrapf(err, "ceph zone %q not found", store.Spec.Zone.Name)
|
|
} else {
|
|
return waitForRequeueIfObjectStoreNotReady, errors.Wrapf(err, "radosgw-admin zone get failed with code %d", code)
|
|
}
|
|
}
|
|
|
|
logger.Infof("Zone %q found in Ceph cluster will include object store %q", store.Spec.Zone.Name, store.Name)
|
|
return reconcile.Result{}, nil
|
|
}
|
|
|
|
func (r *ReconcileCephObjectStore) reconcileMultisiteCRs(cephObjectStore *cephv1.CephObjectStore) (string, string, string, reconcile.Result, error) {
|
|
if cephObjectStore.Spec.IsMultisite() {
|
|
zoneName := cephObjectStore.Spec.Zone.Name
|
|
zone := &cephv1.CephObjectZone{}
|
|
err := r.client.Get(context.TODO(), types.NamespacedName{Name: zoneName, Namespace: cephObjectStore.Namespace}, zone)
|
|
if err != nil {
|
|
if kerrors.IsNotFound(err) {
|
|
return "", "", "", waitForRequeueIfObjectStoreNotReady, err
|
|
}
|
|
return "", "", "", waitForRequeueIfObjectStoreNotReady, errors.Wrapf(err, "error getting CephObjectZone %q", cephObjectStore.Spec.Zone.Name)
|
|
}
|
|
logger.Debugf("CephObjectZone resource %s found", zone.Name)
|
|
|
|
zonegroup := &cephv1.CephObjectZoneGroup{}
|
|
err = r.client.Get(context.TODO(), types.NamespacedName{Name: zone.Spec.ZoneGroup, Namespace: cephObjectStore.Namespace}, zonegroup)
|
|
if err != nil {
|
|
if kerrors.IsNotFound(err) {
|
|
return "", "", "", waitForRequeueIfObjectStoreNotReady, err
|
|
}
|
|
return "", "", "", waitForRequeueIfObjectStoreNotReady, errors.Wrapf(err, "error getting CephObjectZoneGroup %q", zone.Spec.ZoneGroup)
|
|
}
|
|
logger.Debugf("CephObjectZoneGroup resource %s found", zonegroup.Name)
|
|
|
|
realm := &cephv1.CephObjectRealm{}
|
|
err = r.client.Get(context.TODO(), types.NamespacedName{Name: zonegroup.Spec.Realm, Namespace: cephObjectStore.Namespace}, realm)
|
|
if err != nil {
|
|
if kerrors.IsNotFound(err) {
|
|
return "", "", "", waitForRequeueIfObjectStoreNotReady, err
|
|
}
|
|
return "", "", "", waitForRequeueIfObjectStoreNotReady, errors.Wrapf(err, "error getting CephObjectRealm %q", zonegroup.Spec.Realm)
|
|
}
|
|
logger.Debugf("CephObjectRealm resource %s found", realm.Name)
|
|
|
|
return realm.Name, zonegroup.Name, zone.Name, reconcile.Result{}, nil
|
|
}
|
|
|
|
return cephObjectStore.Name, cephObjectStore.Name, cephObjectStore.Name, reconcile.Result{}, nil
|
|
}
|
|
|
|
func (r *ReconcileCephObjectStore) verifyObjectBucketCleanup(objectstore *cephv1.CephObjectStore) (reconcile.Result, bool) {
|
|
bktProvisioner := GetObjectBucketProvisioner(r.context, objectstore.Namespace)
|
|
bktProvisioner = strings.Replace(bktProvisioner, "/", "-", -1)
|
|
selector := fmt.Sprintf("bucket-provisioner=%s", bktProvisioner)
|
|
objectBuckets, err := r.bktclient.ObjectbucketV1alpha1().ObjectBuckets().List(context.TODO(), metav1.ListOptions{LabelSelector: selector})
|
|
if err != nil {
|
|
logger.Errorf("failed to delete object store. failed to list buckets for objectstore %q in namespace %q", objectstore.Name, objectstore.Namespace)
|
|
return opcontroller.WaitForRequeueIfFinalizerBlocked, false
|
|
}
|
|
|
|
if len(objectBuckets.Items) == 0 {
|
|
logger.Infof("no buckets found for objectstore %q in namespace %q", objectstore.Name, objectstore.Namespace)
|
|
return reconcile.Result{}, true
|
|
}
|
|
|
|
bucketNames := make([]string, 0)
|
|
for _, bucket := range objectBuckets.Items {
|
|
bucketNames = append(bucketNames, bucket.Name)
|
|
}
|
|
|
|
logger.Errorf("failed to delete object store. buckets for objectstore %q in namespace %q are not cleaned up. remaining buckets: %+v", objectstore.Name, objectstore.Namespace, bucketNames)
|
|
return opcontroller.WaitForRequeueIfFinalizerBlocked, false
|
|
}
|
|
|
|
func (r *ReconcileCephObjectStore) startMonitoring(objectstore *cephv1.CephObjectStore, objContext *Context, namespacedName types.NamespacedName) {
|
|
// Start monitoring object store
|
|
if r.objectStoreChannels[objectstore.Name].monitoringRunning {
|
|
logger.Debug("external rgw endpoint monitoring go routine already running!")
|
|
return
|
|
}
|
|
|
|
// Set the monitoring flag so we don't start more than one go routine
|
|
r.objectStoreChannels[objectstore.Name].monitoringRunning = true
|
|
|
|
var port int32
|
|
|
|
if objectstore.Spec.IsTLSEnabled() {
|
|
port = objectstore.Spec.Gateway.SecurePort
|
|
} else if objectstore.Spec.Gateway.Port != 0 {
|
|
port = objectstore.Spec.Gateway.Port
|
|
} else {
|
|
logger.Error("At least one of Port or SecurePort should be non-zero")
|
|
return
|
|
}
|
|
|
|
rgwChecker := newBucketChecker(r.context, objContext, port, r.client, namespacedName, &objectstore.Spec)
|
|
|
|
// Fetch the admin ops user
|
|
accessKey, secretKey, err := GetAdminOPSUserCredentials(r.context, r.clusterInfo, objContext, objectstore)
|
|
if err != nil {
|
|
logger.Errorf("failed to create or retrieve rgw admin ops user. %v", err)
|
|
return
|
|
}
|
|
rgwChecker.objContext.adminOpsUserAccessKey = accessKey
|
|
rgwChecker.objContext.adminOpsUserSecretKey = secretKey
|
|
|
|
logger.Info("starting rgw healthcheck")
|
|
go rgwChecker.checkObjectStore(r.objectStoreChannels[objectstore.Name].stopChan)
|
|
}
|
|
|
|
func (r *ReconcileCephObjectStore) verifyObjectUserCleanup(objectstore *cephv1.CephObjectStore) (reconcile.Result, bool) {
|
|
ctx := context.TODO()
|
|
cephObjectUsers, err := r.context.RookClientset.CephV1().CephObjectStoreUsers(objectstore.Namespace).List(ctx, metav1.ListOptions{})
|
|
if err != nil {
|
|
logger.Errorf("failed to delete object store. failed to list user for objectstore %q in namespace %q", objectstore.Name, objectstore.Namespace)
|
|
return opcontroller.WaitForRequeueIfFinalizerBlocked, false
|
|
}
|
|
|
|
if len(cephObjectUsers.Items) == 0 {
|
|
logger.Infof("no users found for objectstore %q in namespace %q", objectstore.Name, objectstore.Namespace)
|
|
return reconcile.Result{}, true
|
|
}
|
|
|
|
userNames := make([]string, 0)
|
|
for _, user := range cephObjectUsers.Items {
|
|
userNames = append(userNames, user.Name)
|
|
}
|
|
|
|
logger.Errorf("failed to delete object store. users for objectstore %q in namespace %q are not cleaned up. remaining users: %+v", objectstore.Name, objectstore.Namespace, userNames)
|
|
return opcontroller.WaitForRequeueIfFinalizerBlocked, false
|
|
}
|