forked from rook/rook
validate_cluster.sh built its ceph exec command around a toolbox pod
name captured once at script start, without waiting for the toolbox to
be ready. When the toolbox container was still being created, or the
pod had been replaced, every check failed with 'unable to upgrade
connection: container not found' for the entire wait window and the
mon quorum check timed out without ever observing the cluster. This
intermittently failed the 'wait for ceph cluster N to be ready' steps
of canary jobs (observed on master pushes of multi-cluster-mirroring
and the canary job between 2026-03-10 and 2026-04-24, always with the
container-not-found signature filling the whole window).
Wait for the toolbox deployment rollout up front, and exec through
deploy/rook-ceph-tools so each call resolves a currently-ready pod,
matching how the other CI scripts invoke the toolbox.
Also call wait_for_daemon directly instead of 'return $(...)'. The
command substitution ran wait_for_daemon in a subshell and captured
its output, so the 'current status' diagnostics printed on timeout
were never displayed; worse, the captured text became the argument of
'return', which fails with 'numeric argument required' and aborts the
script with a misleading exit code 2. The osd variant's captured
'Return value' debug echo had the same problem and is removed.
Signed-off-by: Joshua Hoblitt <josh@hoblitt.com>
(cherry picked from commit 98f088b54f)
178 lines
4.7 KiB
Bash
Executable File
178 lines
4.7 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
|
|
# Copyright 2021 The Rook Authors. All rights reserved.
|
|
#
|
|
# Licensed under the Apache License, Version 2.0 (the "License");
|
|
# you may not use this file except in compliance with the License.
|
|
# You may obtain a copy of the License at
|
|
#
|
|
# http://www.apache.org/licenses/LICENSE-2.0
|
|
#
|
|
# Unless required by applicable law or agreed to in writing, software
|
|
# distributed under the License is distributed on an "AS IS" BASIS,
|
|
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
# See the License for the specific language governing permissions and
|
|
# limitations under the License.
|
|
|
|
set -xEe
|
|
|
|
: "${DAEMON_TO_VALIDATE:=${1}}"
|
|
if [ -z "$DAEMON_TO_VALIDATE" ]; then
|
|
DAEMON_TO_VALIDATE=all
|
|
fi
|
|
# The second script arg is optional and depends on the daemon
|
|
if [ "$DAEMON_TO_VALIDATE" == "rgw" ]; then
|
|
export OBJECT_STORE_NAME=$2
|
|
else
|
|
export OSD_COUNT=$2
|
|
# default to the name of the object store from object-a.yaml
|
|
export OBJECT_STORE_NAME=store-a
|
|
fi
|
|
|
|
#############
|
|
# FUNCTIONS #
|
|
#############
|
|
# Wait for the toolbox to be ready before the per-daemon wait budgets start, and exec through
|
|
# the deployment so every call resolves a currently-ready pod instead of a name captured once
|
|
# (a pod whose container is still creating, or that has been replaced, fails every exec with
|
|
# "container not found").
|
|
kubectl -n rook-ceph rollout status deploy/rook-ceph-tools --timeout=120s
|
|
EXEC_COMMAND="kubectl -n rook-ceph exec deploy/rook-ceph-tools -- ceph --connect-timeout 10"
|
|
|
|
function wait_for_daemon() {
|
|
timeout=90
|
|
daemon_to_test=$1
|
|
while [ $timeout -ne 0 ]; do
|
|
if eval $daemon_to_test; then
|
|
return 0
|
|
fi
|
|
sleep 1
|
|
let timeout=timeout-1
|
|
done
|
|
echo "current status:"
|
|
$EXEC_COMMAND -s
|
|
|
|
return 1
|
|
}
|
|
|
|
function test_demo_mon {
|
|
wait_for_daemon "$EXEC_COMMAND -s | grep -sq quorum"
|
|
}
|
|
|
|
function test_demo_mgr {
|
|
wait_for_daemon "$EXEC_COMMAND -s | grep -sq 'mgr:'"
|
|
}
|
|
|
|
function test_demo_osd {
|
|
wait_for_daemon "$EXEC_COMMAND -s | grep -sq \"$OSD_COUNT osds: $OSD_COUNT up.*, $OSD_COUNT in.*\""
|
|
}
|
|
|
|
function test_demo_rgw {
|
|
timeout 360 bash -x <<-'EOF'
|
|
until [[ "$(kubectl -n rook-ceph get pods -l rgw=$OBJECT_STORE_NAME -o 'jsonpath={..status.conditions[?(@.type=="Ready")].status}')" == "True" ]]; do
|
|
echo "waiting for rgw pods to be ready"
|
|
sleep 5
|
|
done
|
|
EOF
|
|
}
|
|
|
|
function test_demo_mds {
|
|
echo "Waiting for the MDS to be ready"
|
|
# NOTE: metadata server always takes up to 5 sec to run
|
|
# so we first check if the pools exit, from that we assume that
|
|
# the process will start. We stop waiting after 10 seconds.
|
|
wait_for_daemon "$EXEC_COMMAND osd dump | grep -sq cephfs && $EXEC_COMMAND -s | grep -sq up"
|
|
}
|
|
|
|
function test_demo_rbd_mirror {
|
|
wait_for_daemon "$EXEC_COMMAND -s | grep -sq 'rbd-mirror:'"
|
|
}
|
|
|
|
function test_demo_fs_mirror {
|
|
wait_for_daemon "$EXEC_COMMAND -s | grep -sq 'cephfs-mirror:'"
|
|
}
|
|
|
|
function test_demo_pool {
|
|
wait_for_daemon "$EXEC_COMMAND -s | grep -sq '11 pools'"
|
|
}
|
|
|
|
function test_csi {
|
|
csi_pod_min_count="${CSI_POD_MIN_COUNT:-7}"
|
|
timeout 360 bash -x <<EOF
|
|
until [[ "\$(kubectl -n rook-ceph get pods --field-selector=status.phase=Running | grep -c csi.)" -ge "${csi_pod_min_count}" ]]; do
|
|
echo "waiting for csi pods to be ready (need ${csi_pod_min_count})"
|
|
kubectl -n rook-ceph get pods | grep csi
|
|
sleep 5
|
|
done
|
|
EOF
|
|
}
|
|
|
|
function test_nfs {
|
|
timeout 360 bash <<-'EOF'
|
|
until [[ "$(kubectl -n rook-ceph get pods --field-selector=status.phase=Running|grep -c ^rook-ceph-nfs-)" -eq 1 ]]; do
|
|
echo "waiting for nfs pods to be ready"
|
|
sleep 5
|
|
done
|
|
EOF
|
|
}
|
|
|
|
########
|
|
# MAIN #
|
|
########
|
|
test_csi
|
|
test_demo_mon
|
|
test_demo_mgr
|
|
|
|
if [[ "$DAEMON_TO_VALIDATE" == "all" ]]; then
|
|
daemons_list="osd mds rgw rbd_mirror fs_mirror nfs"
|
|
else
|
|
# change commas to space
|
|
comma_to_space=${DAEMON_TO_VALIDATE//,/ }
|
|
|
|
# transform to an array
|
|
IFS=" " read -r -a array <<<"$comma_to_space"
|
|
|
|
# sort and remove potential duplicate
|
|
daemons_list=$(echo "${array[@]}" | tr ' ' '\n' | sort -u | tr '\n' ' ')
|
|
fi
|
|
|
|
for daemon in $daemons_list; do
|
|
case "$daemon" in
|
|
mon)
|
|
continue
|
|
;;
|
|
mgr)
|
|
continue
|
|
;;
|
|
osd)
|
|
test_demo_osd
|
|
;;
|
|
mds)
|
|
test_demo_mds
|
|
;;
|
|
rgw)
|
|
test_demo_rgw
|
|
;;
|
|
rbd_mirror)
|
|
test_demo_rbd_mirror
|
|
;;
|
|
fs_mirror)
|
|
test_demo_fs_mirror
|
|
;;
|
|
nfs)
|
|
test_nfs
|
|
;;
|
|
*)
|
|
log "ERROR: unknown daemon to validate!"
|
|
log "Available daemon are: mon mgr osd mds rgw rbd_mirror fs_mirror"
|
|
exit 1
|
|
;;
|
|
esac
|
|
done
|
|
|
|
echo "Ceph is up and running, have a look!"
|
|
$EXEC_COMMAND -s
|
|
kubectl -n rook-ceph get pods
|
|
kubectl -n rook-ceph logs "$(kubectl -n rook-ceph -l app=rook-ceph-operator get pods -o jsonpath='{.items[*].metadata.name}')"
|
|
kubectl -n rook-ceph get cephcluster -o yaml
|