Files
my-rook-config/tests/scripts/validate_cluster.sh
T
Joshua Hoblitt 0a48779416 ci: fix toolbox exec race in validate_cluster.sh
validate_cluster.sh built its ceph exec command around a toolbox pod
name captured once at script start, without waiting for the toolbox to
be ready. When the toolbox container was still being created, or the
pod had been replaced, every check failed with 'unable to upgrade
connection: container not found' for the entire wait window and the
mon quorum check timed out without ever observing the cluster. This
intermittently failed the 'wait for ceph cluster N to be ready' steps
of canary jobs (observed on master pushes of multi-cluster-mirroring
and the canary job between 2026-03-10 and 2026-04-24, always with the
container-not-found signature filling the whole window).

Wait for the toolbox deployment rollout up front, and exec through
deploy/rook-ceph-tools so each call resolves a currently-ready pod,
matching how the other CI scripts invoke the toolbox.

Also call wait_for_daemon directly instead of 'return $(...)'. The
command substitution ran wait_for_daemon in a subshell and captured
its output, so the 'current status' diagnostics printed on timeout
were never displayed; worse, the captured text became the argument of
'return', which fails with 'numeric argument required' and aborts the
script with a misleading exit code 2. The osd variant's captured
'Return value' debug echo had the same problem and is removed.

Signed-off-by: Joshua Hoblitt <josh@hoblitt.com>
(cherry picked from commit 98f088b54f)
2026-06-16 21:13:28 +00:00

178 lines
4.7 KiB
Bash
Executable File

#!/usr/bin/env bash
# Copyright 2021 The Rook Authors. All rights reserved.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
set -xEe
: "${DAEMON_TO_VALIDATE:=${1}}"
if [ -z "$DAEMON_TO_VALIDATE" ]; then
DAEMON_TO_VALIDATE=all
fi
# The second script arg is optional and depends on the daemon
if [ "$DAEMON_TO_VALIDATE" == "rgw" ]; then
export OBJECT_STORE_NAME=$2
else
export OSD_COUNT=$2
# default to the name of the object store from object-a.yaml
export OBJECT_STORE_NAME=store-a
fi
#############
# FUNCTIONS #
#############
# Wait for the toolbox to be ready before the per-daemon wait budgets start, and exec through
# the deployment so every call resolves a currently-ready pod instead of a name captured once
# (a pod whose container is still creating, or that has been replaced, fails every exec with
# "container not found").
kubectl -n rook-ceph rollout status deploy/rook-ceph-tools --timeout=120s
EXEC_COMMAND="kubectl -n rook-ceph exec deploy/rook-ceph-tools -- ceph --connect-timeout 10"
function wait_for_daemon() {
timeout=90
daemon_to_test=$1
while [ $timeout -ne 0 ]; do
if eval $daemon_to_test; then
return 0
fi
sleep 1
let timeout=timeout-1
done
echo "current status:"
$EXEC_COMMAND -s
return 1
}
function test_demo_mon {
wait_for_daemon "$EXEC_COMMAND -s | grep -sq quorum"
}
function test_demo_mgr {
wait_for_daemon "$EXEC_COMMAND -s | grep -sq 'mgr:'"
}
function test_demo_osd {
wait_for_daemon "$EXEC_COMMAND -s | grep -sq \"$OSD_COUNT osds: $OSD_COUNT up.*, $OSD_COUNT in.*\""
}
function test_demo_rgw {
timeout 360 bash -x <<-'EOF'
until [[ "$(kubectl -n rook-ceph get pods -l rgw=$OBJECT_STORE_NAME -o 'jsonpath={..status.conditions[?(@.type=="Ready")].status}')" == "True" ]]; do
echo "waiting for rgw pods to be ready"
sleep 5
done
EOF
}
function test_demo_mds {
echo "Waiting for the MDS to be ready"
# NOTE: metadata server always takes up to 5 sec to run
# so we first check if the pools exit, from that we assume that
# the process will start. We stop waiting after 10 seconds.
wait_for_daemon "$EXEC_COMMAND osd dump | grep -sq cephfs && $EXEC_COMMAND -s | grep -sq up"
}
function test_demo_rbd_mirror {
wait_for_daemon "$EXEC_COMMAND -s | grep -sq 'rbd-mirror:'"
}
function test_demo_fs_mirror {
wait_for_daemon "$EXEC_COMMAND -s | grep -sq 'cephfs-mirror:'"
}
function test_demo_pool {
wait_for_daemon "$EXEC_COMMAND -s | grep -sq '11 pools'"
}
function test_csi {
csi_pod_min_count="${CSI_POD_MIN_COUNT:-7}"
timeout 360 bash -x <<EOF
until [[ "\$(kubectl -n rook-ceph get pods --field-selector=status.phase=Running | grep -c csi.)" -ge "${csi_pod_min_count}" ]]; do
echo "waiting for csi pods to be ready (need ${csi_pod_min_count})"
kubectl -n rook-ceph get pods | grep csi
sleep 5
done
EOF
}
function test_nfs {
timeout 360 bash <<-'EOF'
until [[ "$(kubectl -n rook-ceph get pods --field-selector=status.phase=Running|grep -c ^rook-ceph-nfs-)" -eq 1 ]]; do
echo "waiting for nfs pods to be ready"
sleep 5
done
EOF
}
########
# MAIN #
########
test_csi
test_demo_mon
test_demo_mgr
if [[ "$DAEMON_TO_VALIDATE" == "all" ]]; then
daemons_list="osd mds rgw rbd_mirror fs_mirror nfs"
else
# change commas to space
comma_to_space=${DAEMON_TO_VALIDATE//,/ }
# transform to an array
IFS=" " read -r -a array <<<"$comma_to_space"
# sort and remove potential duplicate
daemons_list=$(echo "${array[@]}" | tr ' ' '\n' | sort -u | tr '\n' ' ')
fi
for daemon in $daemons_list; do
case "$daemon" in
mon)
continue
;;
mgr)
continue
;;
osd)
test_demo_osd
;;
mds)
test_demo_mds
;;
rgw)
test_demo_rgw
;;
rbd_mirror)
test_demo_rbd_mirror
;;
fs_mirror)
test_demo_fs_mirror
;;
nfs)
test_nfs
;;
*)
log "ERROR: unknown daemon to validate!"
log "Available daemon are: mon mgr osd mds rgw rbd_mirror fs_mirror"
exit 1
;;
esac
done
echo "Ceph is up and running, have a look!"
$EXEC_COMMAND -s
kubectl -n rook-ceph get pods
kubectl -n rook-ceph logs "$(kubectl -n rook-ceph -l app=rook-ceph-operator get pods -o jsonpath='{.items[*].metadata.name}')"
kubectl -n rook-ceph get cephcluster -o yaml