From 44101e39c7b4760b08d10f730aa8895bfcc743da Mon Sep 17 00:00:00 2001 From: "goodolclint-claude[bot]" <323206664+goodolclint-claude[bot]@users.noreply.github.com> Date: Tue, 1 Sep 2026 05:38:08 +0000 Subject: [PATCH] ci: dump corosync state from both nested nodes on cluster test failure The API reports a joined-but-offline node as online=0 with no further detail, and the cleanup job destroys the nodes minutes later, so the reason corosync membership never forms has never reached a log. Read corosync.conf, corosync-cfgtool, pvecm status and the corosync journal off both nodes while they are still alive. Best-effort: never fails the caller. Co-Authored-By: Claude Opus 5 (1M context) --- .../scripts/diagnose-cluster.sh | 81 +++++++++++++++++++ 1 file changed, 81 insertions(+) create mode 100644 tests/infrastructure/scripts/diagnose-cluster.sh diff --git a/tests/infrastructure/scripts/diagnose-cluster.sh b/tests/infrastructure/scripts/diagnose-cluster.sh new file mode 100644 index 0000000..e7312a0 --- /dev/null +++ b/tests/infrastructure/scripts/diagnose-cluster.sh @@ -0,0 +1,81 @@ +#!/usr/bin/env bash +# Dump corosync state from both nested PVE nodes after a cluster test failure. +# +# Usage: diagnose-cluster.sh [8|9] +# +# The PVE API reports a joined-but-offline node as online=0 with no further +# detail; corosync's own view lives only on the nodes, which the cleanup job +# destroys minutes later. Best-effort: never fails the caller. +# +# Required env vars: +# PVE_PASSWORD Root password for the nested PVE instances +# +# Optional env vars: +# CONFIG_FILE Test config JSON (default: $CACHE_DIR/work/config.json) +# CACHE_DIR Shared cache mount (default: /opt/pve-integration) + +VERSION="${1:-9}" +CACHE_DIR="${CACHE_DIR:-/opt/pve-integration}" +CONFIG_FILE="${CONFIG_FILE:-$CACHE_DIR/work/config.json}" + +if [[ ! -f "$CONFIG_FILE" ]]; then + echo "diagnose-cluster: no config at $CONFIG_FILE — nothing to inspect" + exit 0 +fi + +if [[ -z "${PVE_PASSWORD:-}" ]]; then + echo "diagnose-cluster: PVE_PASSWORD unset — cannot reach the nodes" + exit 0 +fi + +SSH_OPTS=(-o StrictHostKeyChecking=no -o UserKnownHostsFile=/dev/null -o LogLevel=ERROR -o ConnectTimeout=10) + +dump_node() { + local label="$1" ip="$2" + echo + echo "══════════ $label ($ip) ══════════" + if [[ -z "$ip" || "$ip" == "null" ]]; then + echo " no address in $CONFIG_FILE" + return + fi + + sshpass -p "$PVE_PASSWORD" ssh "${SSH_OPTS[@]}" "root@${ip}" bash -s <<'REMOTE' 2>&1 || echo " ssh to $ip failed (rc=$?)" +set +e +echo "--- hostname / resolution ---" +hostname -f +echo "hostname -i: $(hostname -i 2>&1)" +grep -vE '^\s*#' /etc/hosts | grep -vE '^\s*$' +echo +echo "--- addresses ---" +ip -4 -o addr show scope global +echo +echo "--- corosync.conf ---" +cat /etc/pve/corosync.conf 2>&1 || cat /etc/corosync/corosync.conf 2>&1 +echo +echo "--- corosync-cfgtool -s ---" +corosync-cfgtool -s 2>&1 +echo +echo "--- pvecm status ---" +pvecm status 2>&1 +echo +echo "--- corosync service ---" +systemctl is-active corosync pve-cluster 2>&1 +echo +echo "--- journalctl -u corosync (last 60) ---" +journalctl -u corosync -n 60 --no-pager 2>&1 +echo +echo "--- journalctl -u pve-cluster (last 30) ---" +journalctl -u pve-cluster -n 30 --no-pager 2>&1 +REMOTE +} + +echo "=== Cluster diagnostics for PVE $VERSION ===" +node_a="$(jq -r ".pve${VERSION}.nodes.a.host // empty" "$CONFIG_FILE")" +node_b="$(jq -r ".pve${VERSION}.nodes.b.host // empty" "$CONFIG_FILE")" + +dump_node "node A" "$node_a" +dump_node "node B" "$node_b" + +echo +echo "=== End cluster diagnostics ===" +exit 0