#!/usr/bin/env bash set -euo pipefail REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" KUBECONFIG_PATH="${KUBECONFIG:-${LAB_KUBECONFIG_PATH:-/home/jv/.kube/config}}" PIMOX_HOST="${LAB_PIMOX_HOST:-192.168.100.80}" PIMOX_USER="${LAB_PIMOX_USER:-jv}" PIMOX_KEY="${LAB_PIMOX_SSH_KEY_PATH:-${HOME}/.ssh/id_ed25519}" PIMOX_QM_BIN="${LAB_PIMOX_QM_BIN:-/usr/sbin/qm}" WORKER_COUNT="${LAB_PIMOX_WORKER_COUNT:-1}" WORKER_BASE_VMID="${LAB_PIMOX_WORKER_BASE_VMID:-9010}" WORKER_NODE_PREFIX="${LAB_PIMOX_WORKER_NODE_PREFIX:-pimox-worker}" WORKER_SKIP_INDEXES="${LAB_PIMOX_SKIP_WORKER_INDEXES:-}" usage() { cat < [args] Commands: list List desired Pimox workers, VM status, and node status tailnet Install/join Tailscale on generated Pimox workers start [index|all] Start Pimox worker VM(s) stop [index|all] Gracefully stop Pimox worker VM(s) restart [index|all] Gracefully restart Pimox worker VM(s) drain Drain a Kubernetes worker node uncordon Mark a Kubernetes worker node schedulable recreate-plan Print the safe recreate sequence for one Pimox worker rebalance Restart evicted/pending pods and show scheduling pressure EOF } have() { command -v "$1" >/dev/null 2>&1 } is_skipped() { local index="$1" local item local normalized="${WORKER_SKIP_INDEXES//,/ }" for item in $normalized; do if [ "$item" = "$index" ]; then return 0 fi done return 1 } validate_index() { local index="$1" if ! [[ "$index" =~ ^[0-9]+$ ]] || [ "$index" -lt 1 ]; then echo "worker index must be a positive integer, got '$index'" >&2 exit 1 fi } vmid_for_index() { local index="$1" echo $((WORKER_BASE_VMID + index - 1)) } node_for_index() { local index="$1" printf '%s-%02d\n' "$WORKER_NODE_PREFIX" "$index" } pimox_ssh() { ssh -i "$PIMOX_KEY" -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new "${PIMOX_USER}@${PIMOX_HOST}" "$@" } kubectl_cmd() { kubectl --kubeconfig "$KUBECONFIG_PATH" "$@" } target_indexes() { local target="${1:-all}" local index if [ "$target" = "all" ]; then for ((index = 1; index <= WORKER_COUNT; index++)); do if ! is_skipped "$index"; then echo "$index" fi done else validate_index "$target" echo "$target" fi } list_workers() { local index local vmid local node local status local node_status printf '%-8s %-8s %-24s %-16s %-16s\n' "INDEX" "VMID" "NODE" "VM" "K8S" for ((index = 1; index <= WORKER_COUNT; index++)); do vmid="$(vmid_for_index "$index")" node="$(node_for_index "$index")" if is_skipped "$index"; then printf '%-8s %-8s %-24s %-16s %-16s\n' "$index" "$vmid" "$node" "skipped" "skipped" continue fi status="$(pimox_ssh "sudo '$PIMOX_QM_BIN' status '$vmid'" 2>/dev/null | awk '{print $2}' || echo unknown)" if have kubectl && [ -s "$KUBECONFIG_PATH" ]; then node_status="$(kubectl_cmd get node "$node" --no-headers 2>/dev/null | awk '{print $2}' || echo missing)" else node_status="kubectl-unavailable" fi printf '%-8s %-8s %-24s %-16s %-16s\n' "$index" "$vmid" "$node" "$status" "$node_status" done } start_workers() { local index local vmid for index in $(target_indexes "${1:-all}"); do vmid="$(vmid_for_index "$index")" echo "Starting Pimox worker index $index VM $vmid..." pimox_ssh "sudo '$PIMOX_QM_BIN' start '$vmid'" done } stop_workers() { local index local vmid for index in $(target_indexes "${1:-all}"); do vmid="$(vmid_for_index "$index")" echo "Gracefully stopping Pimox worker index $index VM $vmid..." pimox_ssh "sudo '$PIMOX_QM_BIN' shutdown '$vmid' --timeout '${LAB_WORKER_SHUTDOWN_TIMEOUT:-90}'" || { echo "Graceful shutdown failed for VM $vmid. Use LAB_CLUSTER_STOP_FORCE=true ./jeannie stop-cluster if a force stop is intended." >&2 exit 1 } done } restart_workers() { local index local vmid local timeout="${LAB_WORKER_SHUTDOWN_TIMEOUT:-90}" for index in $(target_indexes "${1:-all}"); do vmid="$(vmid_for_index "$index")" echo "Restarting Pimox worker index $index VM $vmid..." pimox_ssh "set -eu if ! sudo '$PIMOX_QM_BIN' status '$vmid' >/dev/null 2>&1; then echo 'Pimox worker VM $vmid does not exist.' >&2 exit 1 fi if sudo '$PIMOX_QM_BIN' status '$vmid' | grep -q 'status: running'; then if ! sudo '$PIMOX_QM_BIN' shutdown '$vmid' --timeout '$timeout'; then echo 'Graceful shutdown timed out; forcing stop for Pimox worker VM $vmid.' sudo '$PIMOX_QM_BIN' stop '$vmid' fi fi sudo '$PIMOX_QM_BIN' start '$vmid'" done } drain_node() { local node="${1:-}" if [ -z "$node" ]; then echo "workers drain requires a node name." >&2 exit 1 fi kubectl_cmd drain "$node" --ignore-daemonsets --delete-emptydir-data --timeout="${LAB_WORKER_DRAIN_TIMEOUT:-10m}" } uncordon_node() { local node="${1:-}" if [ -z "$node" ]; then echo "workers uncordon requires a node name." >&2 exit 1 fi kubectl_cmd uncordon "$node" } recreate_plan() { local index="${1:-}" local vmid local node validate_index "$index" vmid="$(vmid_for_index "$index")" node="$(node_for_index "$index")" cat <&2 exit 1 fi echo "Current node pressure:" kubectl_cmd top nodes 2>/dev/null || kubectl_cmd get nodes -o wide echo echo "Problem pods:" kubectl_cmd get pods -A --field-selector=status.phase!=Running,status.phase!=Succeeded -o wide || true echo echo "Restarting pending/failed pods is intentionally manual. Use kubectl delete pod for specific pods after checking events." } tailnet_setup() { local var_file="${LAB_CLUSTER_VAR_FILE:-${REPO_ROOT}/.lab/cluster-workers.auto.tfvars.json}" local auth_key_file="${LAB_PIMOX_WORKER_TAILSCALE_AUTH_KEY_FILE:-${HOME}/.config/homelab/tailscale-pimox-worker.authkey}" local worker_key_prefix="${LAB_PIMOX_WORKER_KEY_PREFIX:-pimox}" local accept_routes="${LAB_PIMOX_WORKER_TAILSCALE_ACCEPT_ROUTES:-false}" local pod_egress_snat="${LAB_PIMOX_WORKER_TAILSCALE_POD_EGRESS_SNAT:-true}" local keys=() local key local cmd if [ ! -s "$var_file" ]; then echo "Missing generated worker var file: $var_file" >&2 echo "Run ./jeannie up once, or set LAB_CLUSTER_VAR_FILE." >&2 exit 1 fi if [ -z "${LAB_PIMOX_WORKER_TAILSCALE_AUTH_KEY:-}" ] && [ ! -s "$auth_key_file" ]; then echo "Missing Pimox worker Tailscale auth key." >&2 echo "Create $auth_key_file outside Git or set LAB_PIMOX_WORKER_TAILSCALE_AUTH_KEY." >&2 exit 1 fi while IFS= read -r key; do [ -n "$key" ] && keys+=("$key") done < <(python3 - "$var_file" "$worker_key_prefix" <<'PY' import json import sys var_file, prefix = sys.argv[1:3] with open(var_file, encoding="utf-8") as handle: worker_nodes = (json.load(handle).get("worker_nodes") or {}) for key in sorted(worker_nodes): if key.startswith(prefix): print(key) PY ) if ((${#keys[@]} == 0)); then echo "No Pimox worker keys found in $var_file with prefix '$worker_key_prefix'." >&2 exit 1 fi cmd=(tofu -chdir="${REPO_ROOT}/bootstrap/cluster" apply -var-file="$var_file") for key in "${keys[@]}"; do cmd+=("-replace=null_resource.kubeadm_worker[\"${key}\"]") done printf 'Re-running cluster worker bootstrap with Tailscale enabled for keys: %s\n' "${keys[*]}" if [ -n "${LAB_PIMOX_WORKER_TAILSCALE_AUTH_KEY:-}" ]; then printf 'Auth key source: LAB_PIMOX_WORKER_TAILSCALE_AUTH_KEY\n' else printf 'Auth key source: %s\n' "$auth_key_file" fi env \ TF_VAR_worker_tailscale_enabled=true \ TF_VAR_worker_tailscale_accept_routes="$accept_routes" \ TF_VAR_worker_tailscale_pod_egress_snat="$pod_egress_snat" \ "${cmd[@]}" } case "${1:-}" in list) list_workers ;; tailnet) tailnet_setup ;; start) start_workers "${2:-all}" ;; stop) stop_workers "${2:-all}" ;; restart) restart_workers "${2:-all}" ;; drain) drain_node "${2:-}" ;; uncordon) uncordon_node "${2:-}" ;; recreate-plan) recreate_plan "${2:-}" ;; rebalance) rebalance ;; -h|--help|help|"") usage ;; *) usage >&2 exit 1 ;; esac