Files
TA-ProxMenu/inc/deploy-proxmox-keepalived.sh

795 lines
28 KiB
Bash

#!/usr/bin/env bash
set -Eeuo pipefail
# -----------------------------------------------------------------------------
# Proxmox cluster Keepalived deployment
# Version 1.5.2
#
# - Discovers real Proxmox cluster members from /etc/pve/.members (QDevice-safe)
# - Prompts for the floating VIP/CIDR unless supplied on the command line
# - Discovers local Linux bridges and prompts the user to select one unless
# supplied on the command line
# - Selects each node's unicast source address from the VIP's subnet
# - Derives VRRP priority from the hostname's trailing numeric suffix:
# suffix 1 = 200, suffix 2 = 190, suffix 3 = 180, etc.
# - Deploys a Proxmox-aware health check including cluster quorum
# - Uses nopreempt so the VIP does not fail back after a recovered node returns
# - Uses unicast VRRP and validates that exactly one node owns the VIP after the
# VRRP election has had time to settle
# - Starts the highest-priority node first so initial ownership is deterministic
#
# Run this from any healthy, quorate Proxmox cluster node as root.
# -----------------------------------------------------------------------------
SCRIPT_VERSION="1.5.2"
VIP_CIDR=""
INTERFACE=""
PRIORITY_BASE=200
PRIORITY_STEP=10
VRID=51
ADVERT_INT=1
ASSUME_YES=0
SSH_OPTS=(
-o BatchMode=yes
-o ConnectTimeout=8
-o StrictHostKeyChecking=accept-new
)
usage() {
cat <<USAGE
Usage: $0 [options]
Options:
--vip CIDR Floating VIP in CIDR form. If omitted, prompt.
--interface IFACE Linux bridge to use. If omitted, show a selection menu.
--vrid N VRRP virtual_router_id, 1-255 (default: ${VRID})
-y, --yes Do not prompt before deployment
--version Show script version
-h, --help Show this help
Examples:
$0
$0 --vip <customer-vip>/<prefix> --interface vmbr0
$0 --vip <customer-vip>/<prefix> --interface vmbr1 --vrid 61 -y
USAGE
}
log() { printf '\n[%s] %s\n' "$(date '+%F %T')" "$*"; }
warn() { printf '\nWARNING: %s\n' "$*" >&2; }
die() { printf '\nERROR: %s\n' "$*" >&2; exit 1; }
while [[ $# -gt 0 ]]; do
case "$1" in
--vip)
[[ $# -ge 2 ]] || die "--vip requires a value"
VIP_CIDR="$2"; shift 2 ;;
--interface)
[[ $# -ge 2 ]] || die "--interface requires a value"
INTERFACE="$2"; shift 2 ;;
--vrid)
[[ $# -ge 2 ]] || die "--vrid requires a value"
VRID="$2"; shift 2 ;;
-y|--yes)
ASSUME_YES=1; shift ;;
--version)
echo "deploy-proxmox-keepalived.sh ${SCRIPT_VERSION}"; exit 0 ;;
-h|--help)
usage; exit 0 ;;
*)
die "Unknown option: $1" ;;
esac
done
[[ $EUID -eq 0 ]] || die "Run this script as root."
command -v pvecm >/dev/null 2>&1 || die "pvecm not found. Run this on a Proxmox VE node."
command -v python3 >/dev/null 2>&1 || die "python3 is required."
command -v ip >/dev/null 2>&1 || die "iproute2 is required."
printf '\nProxmox Keepalived Cluster Deployment v%s\n' "$SCRIPT_VERSION"
printf '%s\n' '------------------------------------------------'
[[ "$PRIORITY_BASE" =~ ^[0-9]+$ ]] || die "Priority base must be numeric."
[[ "$PRIORITY_STEP" =~ ^[0-9]+$ ]] || die "Priority step must be numeric."
[[ "$VRID" =~ ^[0-9]+$ ]] || die "VRID must be numeric."
(( VRID >= 1 && VRID <= 255 )) || die "VRID must be between 1 and 255."
validate_vip_cidr() {
local cidr="$1"
python3 - "$cidr" <<'PY' >/dev/null 2>&1
import ipaddress
import sys
try:
iface = ipaddress.ip_interface(sys.argv[1])
except ValueError:
raise SystemExit(1)
if iface.version != 4:
raise SystemExit(1)
# A shared Ethernet VIP needs room for the VIP plus at least two real hosts.
if iface.network.prefixlen > 30:
raise SystemExit(1)
# Do not allow the subnet's network or broadcast address as the VIP.
if iface.ip in (iface.network.network_address, iface.network.broadcast_address):
raise SystemExit(1)
raise SystemExit(0)
PY
}
ip_is_in_vip_subnet() {
local ip_addr="$1"
python3 - "$VIP_CIDR" "$ip_addr" <<'PY' >/dev/null 2>&1
import ipaddress
import sys
network = ipaddress.ip_interface(sys.argv[1]).network
try:
address = ipaddress.ip_address(sys.argv[2])
except ValueError:
raise SystemExit(1)
raise SystemExit(0 if address in network else 1)
PY
}
# Select the Linux bridge first. The VIP is customer-specific, so when the VIP
# is not supplied on the command line we use the selected bridge's *actual*
# addressing to show the operator which network(s) are valid. Nothing here
# assumes a particular customer subnet.
if [[ -z "$INTERFACE" ]]; then
mapfile -t LOCAL_BRIDGES < <(
for bridge_dir in /sys/class/net/*/bridge; do
[[ -d "$bridge_dir" ]] || continue
basename "$(dirname "$bridge_dir")"
done | sort -V
)
(( ${#LOCAL_BRIDGES[@]} > 0 )) || die "No Linux bridge interfaces were discovered on this node."
printf '\nAvailable Linux bridges on %s:\n\n' "$(hostname -s)"
for i in "${!LOCAL_BRIDGES[@]}"; do
bridge="${LOCAL_BRIDGES[$i]}"
ipv4="$(ip -4 -o addr show dev "$bridge" scope global 2>/dev/null | awk '{print $4}' | paste -sd ',' -)"
[[ -n "$ipv4" ]] || ipv4="no IPv4 address"
printf ' %2d) %-16s %s\n' "$((i + 1))" "$bridge" "$ipv4"
done
while true; do
printf '\n'
read -r -p "Select bridge [1-${#LOCAL_BRIDGES[@]}]: " bridge_choice
if [[ "$bridge_choice" =~ ^[0-9]+$ ]] \
&& (( bridge_choice >= 1 && bridge_choice <= ${#LOCAL_BRIDGES[@]} )); then
INTERFACE="${LOCAL_BRIDGES[$((bridge_choice - 1))]}"
break
fi
printf 'Invalid selection.\n' >&2
done
fi
[[ "$INTERFACE" =~ ^[A-Za-z0-9_.:-]+$ ]] || die "Invalid interface name: $INTERFACE"
ip link show "$INTERFACE" >/dev/null 2>&1 || die "Interface ${INTERFACE} does not exist on the local node."
[[ -d "/sys/class/net/${INTERFACE}/bridge" ]] || die "Interface ${INTERFACE} is not a Linux bridge."
# Read the real IPv4 networks configured on the selected bridge. These are used
# only to guide/validate the user's VIP entry; they are never replaced by a
# hard-coded network.
mapfile -t LOCAL_BRIDGE_CIDRS < <(
ip -4 -o addr show dev "$INTERFACE" scope global 2>/dev/null \
| awk '{print $4}' \
| sort -u
)
(( ${#LOCAL_BRIDGE_CIDRS[@]} > 0 )) \
|| die "Selected bridge ${INTERFACE} has no global IPv4 address on the local node."
mapfile -t LOCAL_BRIDGE_NETWORKS < <(
python3 - "${LOCAL_BRIDGE_CIDRS[@]}" <<'PY'
import ipaddress
import sys
seen = set()
for raw in sys.argv[1:]:
try:
iface = ipaddress.ip_interface(raw)
except ValueError:
continue
network = str(iface.network)
if network not in seen:
seen.add(network)
print(network)
PY
)
printf '\nSelected bridge: %s\n' "$INTERFACE"
printf 'IPv4 addressing on this node:\n'
for cidr in "${LOCAL_BRIDGE_CIDRS[@]}"; do
network="$(python3 - "$cidr" <<'PY'
import ipaddress, sys
print(ipaddress.ip_interface(sys.argv[1]).network)
PY
)"
printf ' %-18s network %s\n' "$cidr" "$network"
done
vip_matches_local_bridge() {
local vip_cidr="$1"
shift
python3 - "$vip_cidr" "$@" <<'PY' >/dev/null 2>&1
import ipaddress
import sys
try:
vip = ipaddress.ip_interface(sys.argv[1])
except ValueError:
raise SystemExit(1)
for raw in sys.argv[2:]:
try:
bridge = ipaddress.ip_interface(raw)
except ValueError:
continue
# Do not allow an already-present VIP itself to satisfy the check on a
# redeploy; there must be a real node address in the same network.
if bridge.ip == vip.ip:
continue
if bridge.network == vip.network:
raise SystemExit(0)
raise SystemExit(1)
PY
}
# Prompt for the VIP after bridge selection so every customer sees their own
# network information before entering the floating address.
if [[ -z "$VIP_CIDR" ]]; then
while true; do
printf '\n'
read -r -p "Enter the unused floating VIP/CIDR for ${INTERFACE}: " VIP_CIDR
if ! validate_vip_cidr "$VIP_CIDR"; then
printf 'Invalid IPv4 VIP/CIDR. Enter an unused host address with a /1 through /30 prefix.\n' >&2
continue
fi
if ! vip_matches_local_bridge "$VIP_CIDR" "${LOCAL_BRIDGE_CIDRS[@]}"; then
printf 'The VIP must be in one of the selected bridge networks, using the same prefix:\n' >&2
for network in "${LOCAL_BRIDGE_NETWORKS[@]}"; do
printf ' %s\n' "$network" >&2
done
continue
fi
break
done
else
validate_vip_cidr "$VIP_CIDR" || die "Invalid VIP/CIDR: $VIP_CIDR"
vip_matches_local_bridge "$VIP_CIDR" "${LOCAL_BRIDGE_CIDRS[@]}" \
|| die "VIP ${VIP_CIDR} is not on a network configured on ${INTERFACE}. Valid local network(s): ${LOCAL_BRIDGE_NETWORKS[*]}"
fi
# Derive the bare VIP address only after the customer-specific VIP has been
# validated against the selected bridge.
VIP_IP="$(python3 - "$VIP_CIDR" <<'PY'
import ipaddress
import sys
print(ipaddress.ip_interface(sys.argv[1]).ip)
PY
)"
VIP_PREFIX="${VIP_CIDR#*/}"
# This check intentionally uses the Proxmox-reported quorum state.
if ! timeout 5 pvecm status 2>/dev/null | grep -Eq '^Quorate:[[:space:]]+Yes[[:space:]]*$'; then
die "The local node does not currently see a quorate cluster. Refusing deployment."
fi
# Use pmxcfs' JSON membership file for node discovery instead of parsing the
# human-formatted output of `pvecm nodes`. This is important on clusters with a
# QDevice: pvecm adds a Qdevice status column (for example A,V,NMW), which shifts
# the Name column and makes fixed-column awk parsing unsafe. /etc/pve/.members
# contains only real Proxmox cluster members in its nodelist.
MEMBERS_FILE="/etc/pve/.members"
[[ -r "$MEMBERS_FILE" ]] || die "Cannot read ${MEMBERS_FILE}; pmxcfs/cluster membership data is unavailable."
LOCAL_NODE="$(python3 - "$MEMBERS_FILE" <<'PYMEM'
import json, sys
try:
with open(sys.argv[1], 'r', encoding='utf-8') as f:
data = json.load(f)
except Exception:
raise SystemExit(1)
name = data.get('nodename')
if not isinstance(name, str) or not name:
raise SystemExit(1)
print(name)
PYMEM
)" || die "Unable to determine the local Proxmox node name from ${MEMBERS_FILE}."
mapfile -t NODES < <(python3 - "$MEMBERS_FILE" <<'PYMEM'
import json, sys
try:
with open(sys.argv[1], 'r', encoding='utf-8') as f:
data = json.load(f)
except Exception:
raise SystemExit(1)
nodelist = data.get('nodelist')
if not isinstance(nodelist, dict):
raise SystemExit(1)
items = []
for name, info in nodelist.items():
if not isinstance(name, str) or not name:
continue
node_id = 0
if isinstance(info, dict):
try:
node_id = int(info.get('id', 0))
except (TypeError, ValueError):
node_id = 0
items.append((node_id, name))
for _, name in sorted(items, key=lambda x: (x[0], x[1])):
print(name)
PYMEM
)
(( ${#NODES[@]} >= 2 )) || die "Fewer than two Proxmox cluster nodes were found in ${MEMBERS_FILE}."
printf '%s\n' "${NODES[@]}" | grep -Fxq "$LOCAL_NODE" \
|| die "Local node ${LOCAL_NODE} is not present in ${MEMBERS_FILE} nodelist."
DEPLOY_TAG="$(date +%Y%m%d-%H%M%S)-$$"
TMPDIR="$(mktemp -d)"
trap 'rm -rf "$TMPDIR"' EXIT
remote_exec() {
local node="$1"
local cmd="$2"
if [[ "$node" == "$LOCAL_NODE" ]]; then
bash -lc "$cmd"
else
ssh "${SSH_OPTS[@]}" "root@${node}" "$cmd"
fi
}
copy_to_node() {
local node="$1"
local src="$2"
local dst="$3"
if [[ "$node" == "$LOCAL_NODE" ]]; then
cp -f "$src" "$dst"
else
scp -q "${SSH_OPTS[@]}" "$src" "root@${node}:${dst}"
fi
}
# Protect against duplicate node records before using associative arrays.
mapfile -t UNIQUE_NODES < <(printf '%s\n' "${NODES[@]}" | awk 'NF && !seen[$0]++')
(( ${#UNIQUE_NODES[@]} == ${#NODES[@]} )) || die "Duplicate node names were returned by ${MEMBERS_FILE}."
NODES=("${UNIQUE_NODES[@]}")
declare -A NODE_IP
declare -A NODE_PRIORITY
declare -A PRIORITY_OWNER
log "Preflight: discovered ${#NODES[@]} Proxmox nodes from ${MEMBERS_FILE}."
# First pass: verify SSH/access, interface addresses, health, suffixes and priorities.
for node in "${NODES[@]}"; do
if [[ "$node" != "$LOCAL_NODE" ]]; then
ssh "${SSH_OPTS[@]}" "root@${node}" true \
|| die "Passwordless root SSH to ${node} failed. No changes have been made."
fi
remote_exec "$node" "command -v ip >/dev/null 2>&1 && command -v systemctl >/dev/null 2>&1 && command -v pvecm >/dev/null 2>&1 && command -v timeout >/dev/null 2>&1" \
|| die "${node} is missing one or more required base commands (ip, systemctl, pvecm, timeout)."
remote_exec "$node" "ip link show '$INTERFACE' >/dev/null 2>&1" \
|| die "Interface ${INTERFACE} does not exist on ${node}."
remote_exec "$node" "test -d '/sys/class/net/${INTERFACE}/bridge'" \
|| die "Interface ${INTERFACE} exists on ${node}, but it is not a Linux bridge."
# Gather all global IPv4 CIDRs on this bridge, excluding the requested VIP
# so rerunning the installer while Keepalived is active is safe.
mapfile -t candidate_cidrs < <(
remote_exec "$node" \
"ip -4 -o addr show dev '$INTERFACE' scope global | awk '{print \$4}' | grep -v '^${VIP_IP}/' || true"
)
subnet_matches=()
prefix_mismatches=()
for candidate_cidr in "${candidate_cidrs[@]}"; do
[[ "$candidate_cidr" =~ ^([0-9]{1,3}\.){3}[0-9]{1,3}/[0-9]{1,2}$ ]] || continue
candidate="${candidate_cidr%/*}"
candidate_prefix="${candidate_cidr#*/}"
if ip_is_in_vip_subnet "$candidate"; then
if [[ "$candidate_prefix" == "$VIP_PREFIX" ]]; then
subnet_matches+=("$candidate")
else
prefix_mismatches+=("$candidate_cidr")
fi
fi
done
if (( ${#subnet_matches[@]} == 0 && ${#prefix_mismatches[@]} > 0 )); then
die "${node}:${INTERFACE} has address(es) inside the VIP network but with a different prefix (${prefix_mismatches[*]}). VIP ${VIP_CIDR} must use the same prefix as the management bridge."
fi
(( ${#subnet_matches[@]} > 0 )) \
|| die "No non-VIP IPv4 address in ${VIP_CIDR} with matching /${VIP_PREFIX} prefix was found on ${node}:${INTERFACE}."
ip="${subnet_matches[0]}"
if (( ${#subnet_matches[@]} > 1 )); then
warn "${node}:${INTERFACE} has multiple addresses in the VIP subnet (${subnet_matches[*]}). Using ${ip} as the VRRP source."
fi
[[ "$node" =~ ([0-9]+)$ ]] \
|| die "Hostname '${node}' has no trailing numeric suffix."
# 10# prevents numbers with leading zeroes from being interpreted as octal.
# Host 1 starts at priority 200, then each suffix step drops by 10:
# *1 = 200, *2 = 190, *3 = 180, etc.
suffix=$((10#${BASH_REMATCH[1]}))
(( suffix >= 1 )) || die "Hostname suffix for ${node} must be 1 or greater."
priority=$((PRIORITY_BASE - ((suffix - 1) * PRIORITY_STEP)))
(( priority >= 1 && priority <= 254 )) \
|| die "Calculated priority ${priority} for ${node} is outside 1..254. Host suffix is too large for base ${PRIORITY_BASE} with step ${PRIORITY_STEP}."
if [[ -n "${PRIORITY_OWNER[$priority]:-}" ]]; then
die "Priority collision: ${node} and ${PRIORITY_OWNER[$priority]} both calculate to ${priority}."
fi
# Verify the remote node itself currently sees quorum before touching it.
remote_exec "$node" \
"timeout 5 pvecm status 2>/dev/null | grep -Eq '^Quorate:[[:space:]]+Yes[[:space:]]*$'" \
|| die "${node} does not currently report Quorate: Yes. No changes have been made."
NODE_IP["$node"]="$ip"
NODE_PRIORITY["$node"]="$priority"
PRIORITY_OWNER["$priority"]="$node"
done
# Prevent accidentally assigning a VIP equal to a real node address.
for node in "${NODES[@]}"; do
[[ "${NODE_IP[$node]}" != "$VIP_IP" ]] \
|| die "VIP ${VIP_IP} is already the selected management address of ${node}."
done
# Sort nodes by descending priority. This matters with nopreempt: start the
# highest-priority node first so it deterministically becomes the initial MASTER.
mapfile -t SORTED_NODES < <(
for node in "${NODES[@]}"; do
printf '%s %s\n' "${NODE_PRIORITY[$node]}" "$node"
done | sort -k1,1nr -k2,2V | awk '{print $2}'
)
printf '\nDeployment plan:\n'
printf ' VIP: %s\n' "$VIP_CIDR"
printf ' Interface: %s\n' "$INTERFACE"
printf ' VRID: %s\n' "$VRID"
printf ' Priority: suffix 1 = %s, then -%s per suffix\n' "$PRIORITY_BASE" "$PRIORITY_STEP"
printf ' Failback: disabled (nopreempt)\n'
printf ' VRRP mode: unicast, IP protocol 112\n\n'
printf ' %-28s %-16s %-8s\n' "NODE" "MANAGEMENT IP" "PRIORITY"
printf ' %-28s %-16s %-8s\n' "----------------------------" "----------------" "--------"
for node in "${SORTED_NODES[@]}"; do
printf ' %-28s %-16s %-8s\n' "$node" "${NODE_IP[$node]}" "${NODE_PRIORITY[$node]}"
done
printf '\nIMPORTANT: VRRP uses IP protocol 112, not TCP/UDP. It must be permitted\n'
printf 'between the listed node IPs on the selected bridge/network. The script will\n'
printf 'wait for the election to settle and detect the common failure case where\n'
printf 'blocked VRRP traffic causes multiple nodes to claim the VIP.\n'
if (( ASSUME_YES == 0 )); then
printf '\nThis will install/reconfigure Keepalived on every listed node.\n'
read -r -p 'Proceed? [y/N] ' answer
[[ "$answer" =~ ^[Yy]$ ]] || { echo "Aborted."; exit 0; }
fi
log "Installing Keepalived/curl and staging configuration on all nodes."
STAGE_DIR="/run/proxmox-keepalived-deploy-${DEPLOY_TAG}"
for node in "${NODES[@]}"; do
log "Preparing ${node}"
remote_exec "$node" \
"DEBIAN_FRONTEND=noninteractive apt-get update -qq && DEBIAN_FRONTEND=noninteractive apt-get install -y keepalived curl >/dev/null"
remote_exec "$node" \
"rm -rf '$STAGE_DIR' && install -d -o root -g root -m 0700 '$STAGE_DIR'"
# Generate a node-specific health check so the HTTPS probe tests the actual
# management address on the selected bridge instead of assuming pveproxy is
# bound to 127.0.0.1:8006.
health_file="$TMPDIR/check-proxmox-keepalived-${node}.sh"
cat > "$health_file" <<HEALTH
#!/usr/bin/env bash
set -u
# Managed by deploy-proxmox-keepalived.sh v${SCRIPT_VERSION}
# Node: ${node}
# Management address: ${NODE_IP[$node]}
# A node may own the management VIP only if the Proxmox management stack,
# Corosync and cluster quorum are all healthy.
systemctl is-active --quiet pveproxy || exit 1
systemctl is-active --quiet pvedaemon || exit 1
systemctl is-active --quiet pve-cluster || exit 1
systemctl is-active --quiet corosync || exit 1
# A surviving but non-quorate node is intentionally ineligible for the VIP.
timeout 3 pvecm status 2>/dev/null \
| grep -Eq '^Quorate:[[:space:]]+Yes[[:space:]]*\$' || exit 1
# Verify the actual externally-facing GUI/API endpoint is answering on the
# address Keepalived uses for this node. TLS verification is intentionally
# disabled because Proxmox commonly uses its cluster CA/node certificate here.
curl --silent --show-error --fail --insecure \
--connect-timeout 2 --max-time 3 \
'https://${NODE_IP[$node]}:8006/' \
>/dev/null 2>&1 || exit 1
exit 0
HEALTH
chmod 0750 "$health_file"
bash -n "$health_file" || die "Internal error: generated health script for ${node} is invalid."
remote_health="$STAGE_DIR/check-proxmox-keepalived.sh"
copy_to_node "$node" "$health_file" "$remote_health"
remote_exec "$node" \
"chown root:root '$remote_health' && chmod 0750 '$remote_health' && bash -n '$remote_health'" \
|| die "Generated health script failed syntax validation on ${node}."
# Test the staged script before touching the live health-check path. This is
# important on a redeploy: an already-running Keepalived instance may be
# executing the live script every few seconds.
# /run is commonly mounted noexec on hardened systems. Execute the staged
# script through Bash so preflight validation works even when the staging
# filesystem itself does not permit execve(2). The final installed script is
# still executed directly from /usr/local/sbin by Keepalived.
remote_exec "$node" "bash '$remote_health'" \
|| die "Staged health check failed on ${node}. No live Keepalived files have been replaced."
config_file="$TMPDIR/keepalived-${node}.conf"
{
cat <<CONF
# Managed by deploy-proxmox-keepalived.sh v${SCRIPT_VERSION}
# Deployment: ${DEPLOY_TAG}
global_defs {
router_id ${node//[^A-Za-z0-9_]/_}
script_user root
enable_script_security
}
vrrp_script chk_proxmox {
script "/usr/local/sbin/check-proxmox-keepalived.sh"
interval 2
timeout 4
fall 3
rise 3
init_fail
}
vrrp_instance PROXMOX_VIP {
state BACKUP
interface ${INTERFACE}
virtual_router_id ${VRID}
priority ${NODE_PRIORITY[$node]}
advert_int ${ADVERT_INT}
nopreempt
unicast_src_ip ${NODE_IP[$node]}
track_src_ip
check_unicast_src
unicast_fault_no_peer
unicast_peer {
CONF
for peer in "${NODES[@]}"; do
[[ "$peer" == "$node" ]] && continue
printf ' %s\n' "${NODE_IP[$peer]}"
done
cat <<CONF
}
virtual_ipaddress {
${VIP_CIDR} dev ${INTERFACE}
}
track_script {
chk_proxmox
}
}
CONF
} > "$config_file"
remote_conf="$STAGE_DIR/keepalived.conf"
remote_validate_conf="$STAGE_DIR/keepalived.validate.conf"
copy_to_node "$node" "$config_file" "$remote_conf"
remote_exec "$node" "chown root:root '$remote_conf' && chmod 0600 '$remote_conf'"
# Validate the staged config against the Keepalived version installed on the
# actual node without changing the live script/config. The validation copy
# points at the root-only staged health script so enable_script_security is
# exercised too.
remote_exec "$node" \
"sed 's#/usr/local/sbin/check-proxmox-keepalived.sh#$remote_health#' '$remote_conf' > '$remote_validate_conf' && chown root:root '$remote_validate_conf' && chmod 0600 '$remote_validate_conf' && keepalived -t -f '$remote_validate_conf'" \
|| die "Keepalived rejected the generated config for ${node}. No live Keepalived files have been replaced."
done
# From this point onward, failures should not leave multiple VRRP masters alive.
stop_keepalived_everywhere() {
local n
for n in "${NODES[@]}"; do
remote_exec "$n" "systemctl stop keepalived" >/dev/null 2>&1 || true
done
}
activation_die() {
local message="$1"
warn "$message"
warn "Stopping Keepalived on every cluster node to avoid leaving an ambiguous VIP state."
stop_keepalived_everywhere
die "$message"
}
log "All configurations validated. Resetting Keepalived election state."
stop_keepalived_everywhere
# If the VIP remains after Keepalived has been stopped everywhere, it is being
# assigned statically or by some other mechanism. Starting VRRP in that state
# would be unsafe.
for node in "${NODES[@]}"; do
if remote_exec "$node" \
"ip -4 -o addr show dev '$INTERFACE' | awk '{print \$4}' | cut -d/ -f1 | grep -Fxq '$VIP_IP'"; then
die "VIP ${VIP_IP} remains assigned on ${node}:${INTERFACE} after Keepalived was stopped. Remove the static/other VIP assignment before deploying."
fi
done
log "Installing validated health checks on all nodes."
for node in "${NODES[@]}"; do
remote_exec "$node" \
"if [ -f /usr/local/sbin/check-proxmox-keepalived.sh ]; then cp -a /usr/local/sbin/check-proxmox-keepalived.sh /usr/local/sbin/check-proxmox-keepalived.sh.pre-proxmox-vip.${DEPLOY_TAG}; fi; install -o root -g root -m 0750 '$STAGE_DIR/check-proxmox-keepalived.sh' /usr/local/sbin/check-proxmox-keepalived.sh" \
|| activation_die "Failed to install the health check on ${node}."
done
# Validate the exact final config now that its referenced live health script
# exists. Do this on every node before replacing any live Keepalived config.
log "Performing final Keepalived configuration validation on all nodes."
for node in "${NODES[@]}"; do
remote_exec "$node" "keepalived -t -f '$STAGE_DIR/keepalived.conf'" \
|| activation_die "Final Keepalived validation failed on ${node}."
done
log "Installing validated Keepalived configs on all nodes."
for node in "${NODES[@]}"; do
remote_exec "$node" \
"if [ -f /etc/keepalived/keepalived.conf ]; then cp -a /etc/keepalived/keepalived.conf /etc/keepalived/keepalived.conf.pre-proxmox-vip.${DEPLOY_TAG}; fi; install -o root -g root -m 0644 '$STAGE_DIR/keepalived.conf' /etc/keepalived/keepalived.conf; systemctl enable keepalived >/dev/null" \
|| activation_die "Failed to install the Keepalived configuration on ${node}."
done
# One last direct health execution catches anything that changed between
# staging and activation. Keepalived is still stopped cluster-wide here.
for node in "${NODES[@]}"; do
remote_exec "$node" "/usr/local/sbin/check-proxmox-keepalived.sh" \
|| activation_die "Final health check failed on ${node}; Keepalived remains stopped cluster-wide."
done
MASTER_NODE="${SORTED_NODES[0]}"
log "Starting highest-priority node first: ${MASTER_NODE} (${NODE_PRIORITY[$MASTER_NODE]})."
remote_exec "$MASTER_NODE" "systemctl start keepalived" \
|| activation_die "Keepalived failed to start on ${MASTER_NODE}."
# init_fail requires successful health checks before the VRRP instance becomes
# eligible. Allow enough time for rise=3, the VRRP master-down timer and normal
# scheduling jitter.
master_ready=0
for _ in {1..20}; do
if remote_exec "$MASTER_NODE" \
"ip -4 -o addr show dev '$INTERFACE' | awk '{print \$4}' | cut -d/ -f1 | grep -Fxq '$VIP_IP'"; then
master_ready=1
break
fi
sleep 1
done
(( master_ready == 1 )) \
|| activation_die "${MASTER_NODE} did not acquire VIP ${VIP_IP}. Check: journalctl -u keepalived"
for node in "${SORTED_NODES[@]:1}"; do
log "Starting backup node ${node}."
remote_exec "$node" "systemctl start keepalived" \
|| activation_die "Keepalived failed to start on backup node ${node}."
done
# Do not verify ownership immediately. If VRRP protocol 112 is blocked, every
# isolated BACKUP can take several seconds to decide the MASTER is absent and
# promote itself. Waiting past the normal master-down interval turns that common
# firewall/multicast-unicast failure into a detectable multiple-owner condition.
SETTLE_SECONDS=$((ADVERT_INT * 5 + 2))
log "Waiting ${SETTLE_SECONDS}s for the VRRP election to settle."
sleep "$SETTLE_SECONDS"
log "Verifying VIP ownership and service state."
owners=()
for node in "${NODES[@]}"; do
remote_exec "$node" "systemctl is-active --quiet keepalived" \
|| activation_die "Keepalived is not active on ${node}."
if remote_exec "$node" \
"ip -4 -o addr show dev '$INTERFACE' | awk '{print \$4}' | cut -d/ -f1 | grep -Fxq '$VIP_IP'"; then
owners+=("$node")
fi
done
(( ${#owners[@]} == 1 )) \
|| activation_die "Expected exactly one VIP owner after the election settled, found ${#owners[@]}: ${owners[*]:-none}. Verify IP protocol 112 is allowed between all node IPs and that VRID ${VRID} is not conflicting on this L2 network."
OWNER_NODE="${owners[0]}"
# Verify the VIP itself answers the Proxmox API from the deployment node.
if ! curl --silent --show-error --fail --insecure \
--connect-timeout 2 --max-time 5 \
"https://${VIP_IP}:8006/" >/dev/null 2>&1; then
warn "VIP ${VIP_IP} is assigned to ${OWNER_NODE}, but the HTTPS test through the VIP failed."
warn "Check routing/firewall rules and any custom pveproxy LISTEN_IP configuration."
else
log "VIP HTTPS test succeeded."
fi
# Staging lives under /run and is no longer needed after successful activation.
for node in "${NODES[@]}"; do
remote_exec "$node" "rm -rf '$STAGE_DIR'" >/dev/null 2>&1 || true
done
cat <<RESULT
Deployment complete.
VIP: ${VIP_IP}
Current owner: ${OWNER_NODE}
Priority: ${NODE_PRIORITY[$OWNER_NODE]}
URL: https://${VIP_IP}:8006/
Health requirements for VIP ownership:
- pveproxy active
- pvedaemon active
- pve-cluster active
- corosync active
- pvecm reports Quorate: Yes
- Proxmox HTTPS responds on the node's selected ${INTERFACE} address
Failback is disabled with nopreempt. After a failure, the surviving MASTER keeps
ownership when a higher-priority node returns.
Configured Proxmox hosts:
RESULT
printf ' %-28s %-16s %-8s %-8s\n' "HOSTNAME" "IP ADDRESS" "PRIORITY" "STATE"
printf ' %-28s %-16s %-8s %-8s\n' "----------------------------" "----------------" "--------" "--------"
for node in "${SORTED_NODES[@]}"; do
state="BACKUP"
[[ "$node" == "$OWNER_NODE" ]] && state="MASTER"
printf ' %-28s %-16s %-8s %-8s\n' "$node" "${NODE_IP[$node]}" "${NODE_PRIORITY[$node]}" "$state"
done
cat <<RESULT
Network requirement:
Allow VRRP (IP protocol 112) between all of the node IPs listed above.
Useful commands:
ip -br addr show ${INTERFACE}
systemctl status keepalived
journalctl -u keepalived -f
/usr/local/sbin/check-proxmox-keepalived.sh; echo \$?
RESULT