diff --git a/lib/jeannie/commands.sh b/lib/jeannie/commands.sh new file mode 100644 index 0000000..7fe6353 --- /dev/null +++ b/lib/jeannie/commands.sh @@ -0,0 +1,2368 @@ +plan_homelab() { + local target="${1:-all}" + local stack + local -a stacks=() + + require_debian_server "plan" + + if [[ "${target}" == "all" ]]; then + if [[ -z "${LAB_CLUSTER_VAR_FILE:-}" ]]; then + prepare_cluster_worker_var_file false + fi + stacks=( + "bootstrap/provisioning" + "bootstrap/cluster" + "bootstrap/platform" + "bootstrap/apps" + "bootstrap/edge" + ) + else + stack="$(tofu_stack_from_plan_target "${target}")" + if [[ "${stack}" == "bootstrap/cluster" && -z "${LAB_CLUSTER_VAR_FILE:-}" ]]; then + prepare_cluster_worker_var_file false + fi + stacks=("${stack}") + fi + + for stack in "${stacks[@]}"; do + echo + echo "Planning ${stack}..." + run_tofu_plan_stack "${stack}" + done +} + +rebuild_cluster() { + require_debian_server "rebuild-cluster" + + export WORKER_SSH_TARGETS="${WORKER_SSH_TARGETS:-}" + export LAB_INCLUDE_RASPBERRY_WORKER="${LAB_INCLUDE_RASPBERRY_WORKER:-false}" + export LAB_PIMOX_TEMPLATE_REPLACE_EXISTING="${LAB_PIMOX_TEMPLATE_REPLACE_EXISTING:-true}" + export LAB_PIMOX_WORKER_COUNT="${LAB_PIMOX_WORKER_COUNT:-2}" + export LAB_PIMOX_WORKER_REPLACE_EXISTING="${LAB_PIMOX_WORKER_REPLACE_EXISTING:-true}" + export TF_VAR_force_worker_rejoin="${TF_VAR_force_worker_rejoin:-true}" + + echo "Rebuilding the Kubernetes cluster without touching external Gitea..." + jeannie_log_start "rebuild-cluster" + jeannie_step_plan 10 + + run_step "Preflight" homelab_preflight + run_step "Nuke existing cluster state" nuke_for_rebuild + run_step "Pimox provisioning and workers" run_pimox_pipeline + run_step "OpenWrt VM" run_openwrt_pipeline + run_step "Worker var file" ensure_cluster_worker_var_file + run_step "Cluster OpenTofu apply" run_tofu_stack "bootstrap/cluster" + run_step "Version report" doctor_versions_report + run_step "Platform OpenTofu apply" run_tofu_stack "bootstrap/platform" + run_step "Applications" apps + run_step "Edge OpenTofu apply" run_tofu_stack "bootstrap/edge" + + echo "Cluster rebuild successfully completed." + echo "Log: ${JEANNIE_LOG_FILE}" +} + +nuke_for_rebuild() { + LAB_NUKE_SKIP_CONFIRM=true nuke +} + +cluster_worker_targets() { + local worker_ssh_targets="${WORKER_SSH_TARGETS-}" + local var_file="${REPO_ROOT}/.lab/cluster-workers.auto.tfvars.json" + + CLUSTER_WORKER_TARGETS=() + read -r -a CLUSTER_WORKER_TARGETS <<< "${worker_ssh_targets}" + + if [[ -s "${var_file}" ]] && command -v python3 >/dev/null 2>&1; then + while IFS= read -r target; do + [[ -n "${target}" ]] || continue + if [[ ! " ${CLUSTER_WORKER_TARGETS[*]-} " =~ [[:space:]]${target}[[:space:]] ]]; then + CLUSTER_WORKER_TARGETS+=("${target}") + fi + done < <(python3 - "${var_file}" <<'PY' +import json +import sys + +var_file = sys.argv[1] +with open(var_file, encoding="utf-8") as handle: + document = json.load(handle) + +for key, node in sorted((document.get("worker_nodes") or {}).items()): + host = node.get("host") + user = node.get("user") + if host and user: + print(f"{user}@{host}") +PY +) + fi +} + +stop_local_kubernetes_services() { + echo "--> Stopping local Kubernetes services on the Debian control plane..." + sudo systemctl stop kubelet 2>/dev/null || true + if command -v crictl >/dev/null 2>&1; then + sudo crictl stop --all 2>/dev/null || true + elif command -v ctr >/dev/null 2>&1; then + sudo ctr -n k8s.io tasks ls -q 2>/dev/null | + while IFS= read -r task; do + [[ -n "${task}" ]] || continue + sudo ctr -n k8s.io tasks kill -s SIGTERM "${task}" 2>/dev/null || true + done + sleep 3 + sudo ctr -n k8s.io tasks ls -q 2>/dev/null | + while IFS= read -r task; do + [[ -n "${task}" ]] || continue + sudo ctr -n k8s.io tasks kill -s SIGKILL "${task}" 2>/dev/null || true + sudo ctr -n k8s.io tasks rm -f "${task}" 2>/dev/null || true + done + fi + if truthy "${LAB_CLUSTER_STOP_CONTAINERD:-false}"; then + sudo systemctl stop containerd 2>/dev/null || true + sudo systemctl reset-failed containerd 2>/dev/null || true + fi + sudo systemctl reset-failed kubelet 2>/dev/null || true +} + +start_local_kubernetes_services() { + echo "--> Starting local Kubernetes services on the Debian control plane..." + sudo systemctl start containerd + sudo systemctl start docker 2>/dev/null || true + sudo systemctl start kubelet +} + +stop_remote_kubernetes_services() { + local target="$1" + + echo "--> Stopping Kubernetes services on remote worker ${target}..." + if ! ssh -o ConnectTimeout=5 "${target}" \ + "sudo systemctl stop kubelet 2>/dev/null || true +if command -v crictl >/dev/null 2>&1; then + sudo crictl stop --all 2>/dev/null || true +elif command -v ctr >/dev/null 2>&1; then + sudo ctr -n k8s.io tasks ls -q 2>/dev/null | while IFS= read -r task; do + [ -n \"\$task\" ] || continue + sudo ctr -n k8s.io tasks kill -s SIGTERM \"\$task\" 2>/dev/null || true + done + sleep 3 + sudo ctr -n k8s.io tasks ls -q 2>/dev/null | while IFS= read -r task; do + [ -n \"\$task\" ] || continue + sudo ctr -n k8s.io tasks kill -s SIGKILL \"\$task\" 2>/dev/null || true + sudo ctr -n k8s.io tasks rm -f \"\$task\" 2>/dev/null || true + done +fi +if [ '${LAB_CLUSTER_STOP_CONTAINERD:-false}' = 'true' ]; then sudo systemctl stop containerd 2>/dev/null || true; sudo systemctl reset-failed containerd 2>/dev/null || true; fi +sudo systemctl reset-failed kubelet 2>/dev/null || true"; then + echo "Warning: could not SSH to ${target}; continuing because Pimox VM shutdown follows." >&2 + fi +} + +start_remote_kubernetes_services() { + local target="$1" + + echo "--> Starting Kubernetes services on remote worker ${target}..." + ssh -o ConnectTimeout=5 "${target}" \ + "sudo systemctl start containerd && sudo systemctl start kubelet" +} + +pimox_worker_count_default() { + local spec_file="${REPO_ROOT}/.lab/pimox-workers.tsv" + local var_file="${REPO_ROOT}/.lab/cluster-workers.auto.tfvars.json" + local worker_key_prefix="${LAB_PIMOX_WORKER_KEY_PREFIX:-pimox}" + local worker_node_prefix="${LAB_PIMOX_WORKER_NODE_PREFIX:-pimox-worker}" + local max_count=0 + local count + local state_count + + if [[ -s "${spec_file}" ]]; then + count="$(awk -F'\t' -v key_prefix="${worker_key_prefix}" -v node_prefix="${worker_node_prefix}-" ' + $1 ~ "^" key_prefix "[0-9]+$" { + worker_index = substr($1, length(key_prefix) + 1) + 0 + if (worker_index > max) max = worker_index + } + $4 ~ "^" node_prefix "[0-9]+$" { + worker_index = substr($4, length(node_prefix) + 1) + 0 + if (worker_index > max) max = worker_index + } + END { print max + 0 } + ' "${spec_file}")" + if [[ "${count}" =~ ^[0-9]+$ && "${count}" -gt "${max_count}" ]]; then + max_count="${count}" + fi + fi + + if [[ -s "${var_file}" ]] && command -v python3 >/dev/null 2>&1; then + count="$(python3 - "${var_file}" "${worker_key_prefix}" "${worker_node_prefix}" <<'PY' +import json +import re +import sys + +var_file, worker_key_prefix, worker_node_prefix = sys.argv[1:4] +with open(var_file, encoding="utf-8") as handle: + document = json.load(handle) +nodes = document.get("worker_nodes") or {} +highest = 0 +key_pattern = re.compile(rf"^{re.escape(worker_key_prefix)}(\d+)$") +node_pattern = re.compile(rf"^{re.escape(worker_node_prefix)}-(\d+)$") +for key, node in nodes.items(): + for candidate in (key, str(node.get("node_name", ""))): + match = key_pattern.match(candidate) or node_pattern.match(candidate) + if match: + highest = max(highest, int(match.group(1))) +print(highest) +PY +)" + if [[ "${count}" =~ ^[0-9]+$ && "${count}" -gt "${max_count}" ]]; then + max_count="${count}" + fi + fi + + state_count="$(tofu -chdir="${REPO_ROOT}/bootstrap/cluster" state show null_resource.worker_nodes_required 2>/dev/null | + awk -F'"' '/"worker_count"[[:space:]]*=/ { print $4; found = 1 } END { exit found ? 0 : 1 }' || true)" + if [[ "${state_count}" =~ ^[0-9]+$ && "${state_count}" -gt "${max_count}" ]]; then + max_count="${state_count}" + fi + + count="${LAB_PIMOX_DEFAULT_WORKER_COUNT:-1}" + if [[ "${count}" =~ ^[0-9]+$ && "${count}" -gt "${max_count}" ]]; then + max_count="${count}" + fi + + printf '%s\n' "${max_count}" +} + +pimox_worker_count_effective() { + local configured="${LAB_PIMOX_WORKER_COUNT:-}" + local detected + + detected="$(pimox_worker_count_default)" + if [[ "${configured}" =~ ^[0-9]+$ && "${configured}" -gt "${detected}" ]]; then + printf '%s\n' "${configured}" + return 0 + fi + printf '%s\n' "${detected}" +} + +stop_pimox_worker_vms() { + local pimox_host="${LAB_PIMOX_HOST:-${TF_VAR_pimox_host:-192.168.100.80}}" + local pimox_user="${LAB_PIMOX_USER:-${TF_VAR_pimox_user:-jv}}" + local pimox_key="${LAB_PIMOX_SSH_KEY_PATH:-${TF_VAR_pimox_ssh_key_path:-/home/jv/.ssh/id_ed25519}}" + local qm_bin="${LAB_PIMOX_QM_BIN:-${TF_VAR_pimox_qm_bin:-/usr/sbin/qm}}" + local worker_count + local worker_base_vmid="${LAB_PIMOX_WORKER_BASE_VMID:-9010}" + local worker_skip_indexes="${LAB_PIMOX_SKIP_WORKER_INDEXES:-}" + local shutdown_timeout="${LAB_CLUSTER_STOP_VM_TIMEOUT_SECONDS:-120}" + local force_stop="${LAB_CLUSTER_STOP_FORCE:-true}" + local index + local vmid + + worker_count="$(pimox_worker_count_effective)" + if ! [[ "${worker_count}" =~ ^[0-9]+$ ]]; then + echo "LAB_PIMOX_WORKER_COUNT must be an integer, got '${worker_count}'." >&2 + exit 1 + fi + if ! [[ "${shutdown_timeout}" =~ ^[0-9]+$ ]]; then + echo "LAB_CLUSTER_STOP_VM_TIMEOUT_SECONDS must be an integer, got '${shutdown_timeout}'." >&2 + exit 1 + fi + if ((worker_count == 0)); then + return 0 + fi + + echo "--> Stopping Pimox worker VMs on ${pimox_host}..." + for ((index = 1; index <= worker_count; index++)); do + if worker_index_is_skipped "${index}" "${worker_skip_indexes}"; then + echo "Skipping Pimox worker index ${index} because LAB_PIMOX_SKIP_WORKER_INDEXES=${worker_skip_indexes}." + continue + fi + vmid=$((worker_base_vmid + index - 1)) + pimox_ssh "${pimox_host}" "${pimox_user}" "${pimox_key}" "set -eu +if ! sudo '${qm_bin}' status '${vmid}' >/dev/null 2>&1; then + echo 'Pimox worker VM ${vmid} does not exist; skipping.' + exit 0 +fi +if sudo '${qm_bin}' status '${vmid}' | grep -q 'status: stopped'; then + echo 'Pimox worker VM ${vmid} is already stopped.' + exit 0 +fi +echo 'Gracefully shutting down Pimox worker VM ${vmid}...' +if sudo '${qm_bin}' shutdown '${vmid}' --timeout '${shutdown_timeout}'; then + exit 0 +fi +if [ '${force_stop}' = 'true' ]; then + echo 'Graceful shutdown timed out; forcing stop for Pimox worker VM ${vmid}.' + sudo '${qm_bin}' stop '${vmid}' +else + echo 'Graceful shutdown timed out for Pimox worker VM ${vmid}. Set LAB_CLUSTER_STOP_FORCE=true to force stop.' >&2 + exit 1 +fi" + done +} + +verify_pimox_worker_vms_stopped() { + local pimox_host="${LAB_PIMOX_HOST:-${TF_VAR_pimox_host:-192.168.100.80}}" + local pimox_user="${LAB_PIMOX_USER:-${TF_VAR_pimox_user:-jv}}" + local pimox_key="${LAB_PIMOX_SSH_KEY_PATH:-${TF_VAR_pimox_ssh_key_path:-/home/jv/.ssh/id_ed25519}}" + local qm_bin="${LAB_PIMOX_QM_BIN:-${TF_VAR_pimox_qm_bin:-/usr/sbin/qm}}" + local worker_count + local worker_base_vmid="${LAB_PIMOX_WORKER_BASE_VMID:-9010}" + local worker_skip_indexes="${LAB_PIMOX_SKIP_WORKER_INDEXES:-}" + local index + local vmid + local failures=0 + + worker_count="$(pimox_worker_count_effective)" + for ((index = 1; index <= worker_count; index++)); do + if worker_index_is_skipped "${index}" "${worker_skip_indexes}"; then + continue + fi + vmid=$((worker_base_vmid + index - 1)) + if ! pimox_ssh "${pimox_host}" "${pimox_user}" "${pimox_key}" "sudo '${qm_bin}' status '${vmid}' 2>/dev/null | grep -q 'status: stopped'"; then + echo "Pimox worker VM ${vmid} is not stopped." >&2 + failures=$((failures + 1)) + fi + done + + ((failures == 0)) +} + +destroy_pimox_worker_vms() { + local pimox_host="${LAB_PIMOX_HOST:-${TF_VAR_pimox_host:-192.168.100.80}}" + local pimox_user="${LAB_PIMOX_USER:-${TF_VAR_pimox_user:-jv}}" + local pimox_key="${LAB_PIMOX_SSH_KEY_PATH:-${TF_VAR_pimox_ssh_key_path:-/home/jv/.ssh/id_ed25519}}" + local qm_bin="${LAB_PIMOX_QM_BIN:-${TF_VAR_pimox_qm_bin:-/usr/sbin/qm}}" + local worker_count + local worker_base_vmid="${LAB_PIMOX_WORKER_BASE_VMID:-9010}" + local worker_skip_indexes="${LAB_PIMOX_SKIP_WORKER_INDEXES:-}" + local index + local vmid + + if ! truthy "${LAB_NUKE_DESTROY_PIMOX_WORKERS:-true}"; then + echo "--> Leaving Pimox worker VMs intact because LAB_NUKE_DESTROY_PIMOX_WORKERS=${LAB_NUKE_DESTROY_PIMOX_WORKERS}." + return 0 + fi + worker_count="$(pimox_worker_count_effective)" + if ! [[ "${worker_count}" =~ ^[0-9]+$ ]]; then + echo "LAB_PIMOX_WORKER_COUNT must be an integer, got '${worker_count}'." >&2 + exit 1 + fi + if ((worker_count == 0)); then + return 0 + fi + + echo "--> Destroying Pimox worker VMs on ${pimox_host}..." + for ((index = 1; index <= worker_count; index++)); do + if worker_index_is_skipped "${index}" "${worker_skip_indexes}"; then + echo "Skipping Pimox worker index ${index} because LAB_PIMOX_SKIP_WORKER_INDEXES=${worker_skip_indexes}." + continue + fi + vmid=$((worker_base_vmid + index - 1)) + pimox_ssh "${pimox_host}" "${pimox_user}" "${pimox_key}" "set -eu +if ! sudo '${qm_bin}' status '${vmid}' >/dev/null 2>&1; then + echo 'Pimox worker VM ${vmid} does not exist; skipping.' + exit 0 +fi +if sudo '${qm_bin}' config '${vmid}' | grep -q '^template: 1$'; then + echo 'Pimox VM ${vmid} is a template; refusing to destroy it as a worker.' >&2 + exit 1 +fi +sudo '${qm_bin}' stop '${vmid}' >/dev/null 2>&1 || true +sudo '${qm_bin}' destroy '${vmid}' --purge 1 >/dev/null 2>&1 || sudo '${qm_bin}' destroy '${vmid}'" + done +} + +start_pimox_worker_vms() { + local pimox_host="${LAB_PIMOX_HOST:-${TF_VAR_pimox_host:-192.168.100.80}}" + local pimox_user="${LAB_PIMOX_USER:-${TF_VAR_pimox_user:-jv}}" + local pimox_key="${LAB_PIMOX_SSH_KEY_PATH:-${TF_VAR_pimox_ssh_key_path:-/home/jv/.ssh/id_ed25519}}" + local qm_bin="${LAB_PIMOX_QM_BIN:-${TF_VAR_pimox_qm_bin:-/usr/sbin/qm}}" + local worker_count + local worker_base_vmid="${LAB_PIMOX_WORKER_BASE_VMID:-9010}" + local worker_skip_indexes="${LAB_PIMOX_SKIP_WORKER_INDEXES:-}" + local index + local vmid + + worker_count="$(pimox_worker_count_effective)" + if ! [[ "${worker_count}" =~ ^[0-9]+$ ]]; then + echo "LAB_PIMOX_WORKER_COUNT must be an integer, got '${worker_count}'." >&2 + exit 1 + fi + if ((worker_count == 0)); then + return 0 + fi + + echo "--> Starting Pimox worker VMs on ${pimox_host}..." + for ((index = 1; index <= worker_count; index++)); do + if worker_index_is_skipped "${index}" "${worker_skip_indexes}"; then + echo "Skipping Pimox worker index ${index} because LAB_PIMOX_SKIP_WORKER_INDEXES=${worker_skip_indexes}." + continue + fi + vmid=$((worker_base_vmid + index - 1)) + pimox_ssh "${pimox_host}" "${pimox_user}" "${pimox_key}" "set -eu +if ! sudo '${qm_bin}' status '${vmid}' >/dev/null 2>&1; then + echo 'Pimox worker VM ${vmid} does not exist; skipping.' + exit 0 +fi +if sudo '${qm_bin}' status '${vmid}' | grep -q 'status: running'; then + echo 'Pimox worker VM ${vmid} is already running.' + exit 0 +fi +echo 'Starting Pimox worker VM ${vmid}...' +sudo '${qm_bin}' start '${vmid}'" + done +} + +stop_cluster() { + local target + local failures=0 + declare -a CLUSTER_WORKER_TARGETS=() + + require_debian_server "stop-cluster" + cluster_worker_targets + + echo "Stopping Kubernetes cluster runtime without destroying state..." + for target in "${CLUSTER_WORKER_TARGETS[@]}"; do + stop_remote_kubernetes_services "${target}" + done + stop_pimox_worker_vms + stop_local_kubernetes_services + if kubernetes_api_reachable; then + echo "Kubernetes API is still reachable after stop; control-plane containers may still be running." >&2 + failures=$((failures + 1)) + fi + if ! verify_pimox_worker_vms_stopped; then + failures=$((failures + 1)) + fi + if ((failures > 0)); then + echo "Cluster stop finished with ${failures} verification failure(s)." >&2 + return 1 + fi + echo "Kubernetes runtime stopped. OpenTofu state, kubeadm files, PV data, and VM disks were left intact." +} + +start_cluster() { + local target + declare -a CLUSTER_WORKER_TARGETS=() + + require_debian_server "start-cluster" + cluster_worker_targets + + echo "Starting Kubernetes cluster runtime without rebuilding state..." + start_local_kubernetes_services + start_pimox_worker_vms + for target in "${CLUSTER_WORKER_TARGETS[@]}"; do + start_remote_kubernetes_services "${target}" + done + echo "Kubernetes runtime start requested. Use 'kubectl get nodes -o wide' to watch readiness." +} + +kubernetes_api_reachable() { + kubectl --kubeconfig "${KUBECONFIG_PATH}" get --raw=/readyz >/dev/null 2>&1 +} + +wait_for_kubernetes_api_stable() { + local timeout_seconds="${1:-300}" + local stable_seconds="${2:-30}" + local interval_seconds="${3:-5}" + local elapsed=0 + local stable_elapsed=0 + + echo "Waiting for Kubernetes API to stay ready for ${stable_seconds}s..." + until ((stable_elapsed >= stable_seconds)); do + if kubernetes_api_reachable; then + stable_elapsed=$((stable_elapsed + interval_seconds)) + else + stable_elapsed=0 + fi + + if ((stable_elapsed >= stable_seconds)); then + return 0 + fi + if ((elapsed >= timeout_seconds)); then + echo "Kubernetes API did not stay ready for ${stable_seconds}s within ${timeout_seconds}s." >&2 + return 1 + fi + + sleep "${interval_seconds}" + elapsed=$((elapsed + interval_seconds)) + done +} + +ensure_kubernetes_api_ready_for_tofu_stack() { + local stack="$1" + + case "${stack}" in + bootstrap/platform|bootstrap/apps) + wait_for_kubernetes_api_stable \ + "${LAB_KUBERNETES_API_WAIT_TIMEOUT_SECONDS:-300}" \ + "${LAB_KUBERNETES_API_STABLE_SECONDS:-30}" \ + "${LAB_KUBERNETES_API_STABLE_INTERVAL_SECONDS:-5}" + ;; + esac +} + +cluster_control_plane_tracked() { + tofu_state_has_resource "bootstrap/cluster" "null_resource.kubeadm_control_plane" +} + +cluster_admin_kubeconfig_present() { + [[ -s "${KUBECONFIG_PATH}" ]] || + sudo test -s /etc/kubernetes/admin.conf 2>/dev/null || + [[ -n "${KUBECONFIG:-}" && -s "${KUBECONFIG}" ]] +} + +wait_for_kubernetes_api() { + local timeout_seconds="${1:-180}" + local elapsed=0 + + until kubernetes_api_reachable; do + if ((elapsed >= timeout_seconds)); then + echo "Kubernetes API did not become reachable after ${timeout_seconds}s." >&2 + return 1 + fi + sleep 5 + elapsed=$((elapsed + 5)) + done +} + +ensure_existing_cluster_started_for_up() { + if ! cluster_control_plane_tracked; then + echo "No tracked kubeadm control plane found; bootstrap/cluster will create one." + return 0 + fi + + if ! cluster_admin_kubeconfig_present; then + cat >&2 <&2 + fi +} + +check_pimox_worker_tailnet_egress() { + local spec_file="${REPO_ROOT}/.lab/pimox-workers.tsv" + local probe_host="${LAB_PIMOX_WORKER_TAILNET_PROBE_HOST:-${LAB_DEBIAN_TAILSCALE_IP:-}}" + local probe_port="${LAB_PIMOX_WORKER_TAILNET_PROBE_PORT:-${LAB_GITEA_HTTP_PORT:-3000}}" + local failures=0 + local found=0 + local worker_key + local host + local user + local node_name + local key_path + local output + + if [[ ! -s "${spec_file}" ]]; then + echo "No Pimox worker spec found at ${spec_file}; run ./jeannie up or set LAB_CLUSTER_VAR_FILE." + return 1 + fi + if [[ -z "${probe_host}" ]]; then + echo "No tailnet probe host configured; set LAB_PIMOX_WORKER_TAILNET_PROBE_HOST or LAB_DEBIAN_TAILSCALE_IP." + return 1 + fi + + while IFS=$'\t' read -r worker_key host user node_name key_path; do + [[ -n "${worker_key}" && -n "${host}" && -n "${user}" && -n "${node_name}" && -n "${key_path}" ]] || continue + found=1 + printf '%-28s ' "${node_name}" + if output="$(ssh -n -i "${key_path}" -o BatchMode=yes -o ConnectTimeout=8 -o StrictHostKeyChecking=accept-new "${user}@${host}" " +set -eu +command -v tailscale >/dev/null || { echo 'tailscale command missing'; exit 1; } +sudo systemctl is-active --quiet tailscaled || { echo 'tailscaled is not active'; exit 1; } +tailscale_ip=\"\$(tailscale ip -4 2>/dev/null | head -n 1 || true)\" +[ -n \"\$tailscale_ip\" ] || { echo 'worker is not joined to Tailscale'; exit 1; } +timeout 6 bash -c 'exec 3<>/dev/tcp/${probe_host}/${probe_port}' || { echo 'tailnet TCP probe failed'; exit 1; } +echo \"tailscale_ip=\$tailscale_ip\" +" 2>&1)"; then + printf 'ok - %s:%s reachable over tailnet (%s)\n' "${probe_host}" "${probe_port}" "${output}" + else + printf 'fail - %s\n' "$(printf '%s' "${output}" | head -n 1)" + failures=$((failures + 1)) + fi + done <"${spec_file}" + + if ((found == 0)); then + echo "No Pimox workers were listed in ${spec_file}." + return 1 + fi + + ((failures == 0)) +} + +status_http() { + local name="$1" + local url="$2" + + if command -v curl >/dev/null 2>&1; then + printf '%-28s ' "${name}" + curl -k -sS -o /dev/null --connect-timeout 5 --max-time 10 -w '%{http_code} %{url_effective}\n' "${url}" || + printf 'unreachable %s\n' "${url}" + else + printf '%-28s curl not installed\n' "${name}" + fi +} + +status_http_ok() { + local url="$1" + local status + + if ! command -v curl >/dev/null 2>&1; then + echo "curl not installed" >&2 + return 1 + fi + + status="$(curl -k -sS -o /dev/null --connect-timeout 5 --max-time 10 -w '%{http_code}' "${url}")" || { + echo "unreachable ${url}" >&2 + return 1 + } + case "${status}" in + 2* | 3*) + return 0 + ;; + *) + echo "HTTP ${status} from ${url}" >&2 + return 1 + ;; + esac +} + +status_http_reachable() { + local url="$1" + local status + + if ! command -v curl >/dev/null 2>&1; then + echo "curl not installed" >&2 + return 1 + fi + + status="$(curl -k -sS -o /dev/null --connect-timeout 5 --max-time 10 -w '%{http_code}' "${url}")" || { + echo "unreachable ${url}" >&2 + return 1 + } + case "${status}" in + 2* | 3* | 4*) + return 0 + ;; + *) + echo "HTTP ${status} from ${url}" >&2 + return 1 + ;; + esac +} + +status_gitea_public_ok() { + local root_url="${LAB_GITEA_ROOT_URL:-${LAB_PUBLIC_URL:?LAB_PUBLIC_URL is required from homelab.yml}/git/}" + + status_http_reachable "${root_url}" +} + +status_cascade_check() { + local description="$1" + local output + shift + + if truthy "${JEANNIE_REPORT_MODE:-false}"; then + report_ui_check fail "${STATUS_CURRENT_SECTION:-Status}" "${description}" "./jeannie explain status" "" "" "$@" || return 1 + return 0 + fi + + printf '%-36s ' "${description}" + if output="$("$@" 2>&1)"; then + printf 'ok\n' + if truthy "${LAB_STATUS_SHOW_OK_OUTPUT:-false}" && [[ -n "${output}" ]]; then + printf '%s\n' "${output}" | sed -n '1,6p' | sed 's/^/ /' + fi + return 0 + fi + + printf 'fail\n' + if [[ -n "${output}" ]]; then + printf '%s\n' "${output}" | sed -n '1,8p' | sed 's/^/ /' + fi + return 1 +} + +status_cascade_warn() { + local description="$1" + local output + shift + + if truthy "${JEANNIE_REPORT_MODE:-false}"; then + report_ui_check warn "${STATUS_CURRENT_SECTION:-Status}" "${description}" "./jeannie explain status" "" "" "$@" + return 0 + fi + + printf '%-36s ' "${description}" + if output="$("$@" 2>&1)"; then + printf 'ok\n' + if truthy "${LAB_STATUS_SHOW_OK_OUTPUT:-false}" && [[ -n "${output}" ]]; then + printf '%s\n' "${output}" | sed -n '1,6p' | sed 's/^/ /' + fi + return 0 + fi + + printf 'warn\n' + if [[ -n "${output}" ]]; then + printf '%s\n' "${output}" | sed -n '1,6p' | sed 's/^/ /' + fi + return 0 +} + +status_container_running() { + local container="$1" + + if ! command -v docker >/dev/null 2>&1; then + echo "docker not installed" >&2 + return 1 + fi + if ! sudo docker inspect -f '{{.State.Running}}' "${container}" 2>/dev/null | grep -qx true; then + echo "container ${container} is not running" >&2 + return 1 + fi +} + +status_kubernetes_api_ok() { + if ! command -v kubectl >/dev/null 2>&1; then + echo "kubectl not installed" >&2 + return 1 + fi + if [[ ! -s "${KUBECONFIG_PATH}" ]]; then + echo "kubeconfig missing: ${KUBECONFIG_PATH}" >&2 + return 1 + fi + kubectl --kubeconfig "${KUBECONFIG_PATH}" get --raw=/readyz >/dev/null +} + +status_kubernetes_nodes_ready() { + local nodes + local not_ready + + nodes="$(kubectl --kubeconfig "${KUBECONFIG_PATH}" get nodes --no-headers 2>/dev/null)" || return 1 + if [[ -z "${nodes}" ]]; then + echo "no Kubernetes nodes found" >&2 + return 1 + fi + not_ready="$(awk '$2 !~ /^Ready/ { print $1 ":" $2 }' <<<"${nodes}")" + if [[ -n "${not_ready}" ]]; then + echo "not ready: ${not_ready}" >&2 + return 1 + fi +} + +status_kubernetes_deployment_ready() { + local namespace="$1" + local deployment="$2" + + kubectl --kubeconfig "${KUBECONFIG_PATH}" -n "${namespace}" rollout status "deployment/${deployment}" --timeout=5s >/dev/null +} + +status_no_problem_pods() { + local rows + + rows="$( + kubectl --kubeconfig "${KUBECONFIG_PATH}" get pods -A --no-headers 2>/dev/null | + awk '$4 != "Running" && $4 != "Completed" { print $1 "/" $2 ":" $4 }' + )" + if [[ -n "${rows}" ]]; then + printf '%s\n' "${rows}" >&2 + return 1 + fi +} + +status_recent_deployments() { + local rows + + rows="$( + kubectl --kubeconfig "${KUBECONFIG_PATH}" get deployments -A \ + -o custom-columns='NAMESPACE:.metadata.namespace,NAME:.metadata.name,READY:.status.readyReplicas,DESIRED:.spec.replicas,UPDATED:.status.updatedReplicas,AVAILABLE:.status.availableReplicas,AGE:.metadata.creationTimestamp' \ + --no-headers 2>/dev/null | + awk '$3 != $4 || $5 != $4 || $6 != $4 { print }' + )" + if [[ -n "${rows}" ]]; then + printf '%s\n' "${rows}" >&2 + return 1 + fi +} + +status_pod_restart_pressure() { + local threshold="${LAB_STATUS_RESTART_THRESHOLD:-3}" + local rows + + rows="$( + kubectl --kubeconfig "${KUBECONFIG_PATH}" get pods -A --no-headers 2>/dev/null | + awk -v threshold="${threshold}" '$5 + 0 >= threshold { print $1 "/" $2 ": restarts=" $5 " age=" $6 }' + )" + if [[ -n "${rows}" ]]; then + printf '%s\n' "${rows}" >&2 + return 1 + fi +} + +status_node_pressure() { + local rows + + rows="$( + kubectl --kubeconfig "${KUBECONFIG_PATH}" get nodes -o jsonpath='{range .items[*]}{.metadata.name}{" "}{range .status.conditions[*]}{.type}{"="}{.status}{" "}{end}{"\n"}{end}' 2>/dev/null | + awk '/MemoryPressure=True|DiskPressure=True|PIDPressure=True|NetworkUnavailable=True/ { print }' + )" + if [[ -n "${rows}" ]]; then + printf '%s\n' "${rows}" >&2 + return 1 + fi +} + +status_disk_pressure() { + local threshold="${LAB_STATUS_DISK_USE_THRESHOLD:-85}" + local rows + + rows="$(df -P / /data 2>/dev/null | awk -v threshold="${threshold}" 'NR > 1 { gsub(/%/, "", $5); if ($5 + 0 >= threshold) print $6 ": " $5 "% used" }')" + if [[ -n "${rows}" ]]; then + printf '%s\n' "${rows}" >&2 + return 1 + fi +} + +status_traefik_error_rates() { + local since="${LAB_STATUS_LOG_SINCE:-30m}" + local sample + local count_502 + local count_404 + + sample="$(kubectl --kubeconfig "${KUBECONFIG_PATH}" -n traefik logs -l app.kubernetes.io/name=traefik --since="${since}" --tail=2000 2>/dev/null || true)" + if [[ -z "${sample}" ]]; then + echo "no recent Traefik logs available" >&2 + return 1 + fi + + count_502="$(grep -Eac '(^|[^0-9])50[234]([^0-9]|$)|level=error' <<<"${sample}" || true)" + count_404="$(grep -Eac '(^|[^0-9])404([^0-9]|$)' <<<"${sample}" || true)" + if ((count_502 > 0)); then + printf '5xx/error lines in last %s: %s\n' "${since}" "${count_502}" >&2 + return 1 + fi + printf '404 lines in last %s: %s\n' "${since}" "${count_404}" >&2 +} + +status_gitea_recent_errors() { + local since="${LAB_STATUS_LOG_SINCE:-30m}" + local container="${GITEA_CONTAINER_NAME:-homelab-gitea}" + local sample + local matches + + if ! command -v docker >/dev/null 2>&1; then + echo "docker not installed" >&2 + return 1 + fi + if ! sudo docker inspect "${container}" >/dev/null 2>&1; then + echo "container ${container} not found" >&2 + return 1 + fi + + sample="$(sudo docker logs --since "${since}" "${container}" 2>&1 || true)" + matches="$(grep -Eai 'error|panic|fatal|authentication failed|denied| 50[0-9] ' <<<"${sample}" | tail -10 || true)" + if [[ -n "${matches}" ]]; then + printf '%s\n' "${matches}" >&2 + return 1 + fi +} + +status_registry_ok() { + status_http_ok "http://${LAB_REGISTRY_ENDPOINT:-192.168.100.73:30500}/v2/" +} + +status_rpi_uptime_kuma_ok() { + local rpi_host="${LAB_RPI_HOST:-${LAB_RASPBERRY_HOST:-192.168.100.89}}" + local rpi_user="${LAB_RPI_USER:-${LAB_RASPBERRY_USER:-jv}}" + local rpi_key="${LAB_RPI_SSH_KEY_PATH:-${LAB_RASPBERRY_SSH_KEY_PATH:-/home/jv/.ssh/id_ed25519}}" + local uptime_kuma_port="${UPTIME_KUMA_PORT:-3001}" + + ssh -i "${rpi_key}" -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new "${rpi_user}@${rpi_host}" \ + "curl -fsS --max-time 5 'http://127.0.0.1:${uptime_kuma_port}/' >/dev/null" +} + +status_pimox_workers_running() { + local pimox_host="${LAB_PIMOX_HOST:-${TF_VAR_pimox_host:-192.168.100.80}}" + local pimox_user="${LAB_PIMOX_USER:-${TF_VAR_pimox_user:-jv}}" + local pimox_key="${LAB_PIMOX_SSH_KEY_PATH:-${TF_VAR_pimox_ssh_key_path:-/home/jv/.ssh/id_ed25519}}" + local qm_bin="${LAB_PIMOX_QM_BIN:-${TF_VAR_pimox_qm_bin:-/usr/sbin/qm}}" + local worker_count + local worker_base_vmid="${LAB_PIMOX_WORKER_BASE_VMID:-9010}" + local worker_skip_indexes="${LAB_PIMOX_SKIP_WORKER_INDEXES:-}" + local index + local vmid + local failures=0 + + worker_count="$(pimox_worker_count_effective)" + if ! [[ "${worker_count}" =~ ^[0-9]+$ ]]; then + echo "invalid worker count ${worker_count}" >&2 + return 1 + fi + + for ((index = 1; index <= worker_count; index++)); do + if worker_index_is_skipped "${index}" "${worker_skip_indexes}"; then + continue + fi + vmid=$((worker_base_vmid + index - 1)) + if ! pimox_ssh "${pimox_host}" "${pimox_user}" "${pimox_key}" "sudo '${qm_bin}' status '${vmid}' | grep -q 'status: running'"; then + echo "VM ${vmid} is not running" >&2 + failures=$((failures + 1)) + fi + done + + ((failures == 0)) +} + +status_cascade_report() { + local failures=0 + + status_section "Status Cascade" + status_cascade_check "inventory YAML" inventory_check || failures=$((failures + 1)) + status_cascade_check "Debian Docker root" check_debian_docker_root || failures=$((failures + 1)) + status_cascade_check "Debian Docker service" check_systemd_active docker || failures=$((failures + 1)) + status_cascade_check "Debian containerd service" check_systemd_active containerd || failures=$((failures + 1)) + status_cascade_check "Debian Tailscale IP" check_debian_tailscale_ip || failures=$((failures + 1)) + + status_section "Git And Local Services" + status_cascade_check "Gitea container" status_container_running "${GITEA_CONTAINER_NAME:-homelab-gitea}" || failures=$((failures + 1)) + status_cascade_check "Gitea local HTTP" status_http_ok "http://127.0.0.1:${LAB_GITEA_HTTP_PORT:-3000}/" || failures=$((failures + 1)) + status_cascade_warn "Arr Radarr HTTP" status_http_ok "http://127.0.0.1:7878/" + status_cascade_warn "Arr Sonarr HTTP" status_http_ok "http://127.0.0.1:8989/" + status_cascade_warn "Arr Prowlarr HTTP" status_http_ok "http://127.0.0.1:9696/" + + status_section "RPi Bootstrap And DNS" + status_cascade_check "RPi SSH" check_rpi_ssh || failures=$((failures + 1)) + status_cascade_check "RPi Docker root state" check_rpi_docker_root_state || failures=$((failures + 1)) + status_cascade_check "Pi-hole DNS" check_pihole_dns_query || failures=$((failures + 1)) + status_cascade_warn "RPi Tailscale IP" check_rpi_tailscale_ip + status_cascade_warn "Uptime Kuma HTTP" status_rpi_uptime_kuma_ok + + status_section "Pimox And Cluster" + status_cascade_check "Pimox storage" check_pimox_storage || failures=$((failures + 1)) + status_cascade_check "Pimox workers running" status_pimox_workers_running || failures=$((failures + 1)) + status_cascade_check "Kubernetes API" status_kubernetes_api_ok || failures=$((failures + 1)) + status_cascade_check "Kubernetes nodes Ready" status_kubernetes_nodes_ready || failures=$((failures + 1)) + status_cascade_warn "No problem pods" status_no_problem_pods + + status_section "What Broke Signals" + status_cascade_warn "Backup freshness" backup_status + status_cascade_warn "Recent deployments healthy" status_recent_deployments + status_cascade_warn "Pod restart pressure" status_pod_restart_pressure + status_cascade_warn "Node pressure" status_node_pressure + status_cascade_warn "Disk usage" status_disk_pressure + status_cascade_warn "Traefik 5xx/404 signals" status_traefik_error_rates + status_cascade_warn "Gitea recent errors" status_gitea_recent_errors + + status_section "Platform And Apps" + status_cascade_check "Local registry" status_registry_ok || failures=$((failures + 1)) + status_cascade_check "Traefik deployment" status_kubernetes_deployment_ready traefik traefik || failures=$((failures + 1)) + status_cascade_check "Website deployment" status_kubernetes_deployment_ready website-production php-website-deployment || failures=$((failures + 1)) + status_cascade_warn "Argo CD server" status_kubernetes_deployment_ready argocd argocd-server + + status_section "Edge And Public URLs" + status_cascade_check "OCI edge SSH" check_edge_ssh || failures=$((failures + 1)) + status_cascade_check "Traefik LoadBalancer HTTP" status_http_reachable "http://${LAB_TRAEFIK_LB_IP:?LAB_TRAEFIK_LB_IP is required from homelab.yml}/" || failures=$((failures + 1)) + status_cascade_check "Website public URL" status_http_ok "${LAB_PUBLIC_URL:?LAB_PUBLIC_URL is required from homelab.yml}/" || failures=$((failures + 1)) + status_cascade_check "Gitea public route" status_gitea_public_ok || failures=$((failures + 1)) + + if truthy "${JEANNIE_REPORT_MODE:-false}"; then + return "${failures}" + fi + + if ((failures > 0)); then + printf '\nStatus cascade found %s blocking failure(s).\n' "${failures}" >&2 + return 1 + fi + + printf '\nStatus cascade passed.\n' +} + +backstage_brain_enabled() { + local provider="${LAB_AI_GATEWAY_PROVIDER:-ollama}" + local enabled="${LAB_BACKSTAGE_BRAIN_ENABLED:-false}" + + [[ "${provider,,}" == "ollama" ]] && truthy "${enabled}" +} + +backstage_brain_note() { + local topic="$1" + local endpoint="${LAB_AI_GATEWAY_URL:-${LAB_OLLAMA_URL:-http://127.0.0.1:11434}}" + local model="${LAB_AI_GATEWAY_MODEL:-qwen2.5:0.5b}" + local timeout="${LAB_AI_GATEWAY_TIMEOUT_SECONDS:-20}" + local index_dir="${LAB_AI_KNOWLEDGE_INDEX_DIR:-/data/homelab-ai/index}" + local context + local retrieved_context="" + + if ! backstage_brain_enabled; then + return 0 + fi + if ! command -v python3 >/dev/null 2>&1; then + return 0 + fi + + context="$(cat)" + if [[ -s "${index_dir}/index.json" ]]; then + retrieved_context="$("${REPO_ROOT}/scripts/query-homelab-ai-index" --index-dir "${index_dir}" --context-only --limit 3 "${topic} ${context}" 2>/dev/null || true)" + fi + if [[ -n "${retrieved_context}" ]]; then + context="${context} + +Retrieved homelab knowledge: +${retrieved_context}" + fi + BACKSTAGE_CONTEXT="${context}" python3 - "${endpoint}" "${model}" "${timeout}" "${topic}" <<'PY' || true +import json +import os +import sys +import urllib.error +import urllib.request + +endpoint, model, timeout, topic = sys.argv[1:5] +context = os.environ.get("BACKSTAGE_CONTEXT", "").strip() +try: + timeout_seconds = int(timeout) +except ValueError: + timeout_seconds = 20 + +prompt = f"""You are the backstage helper for a personal homelab script named jeannie. +Be concise and operational. Do not invent facts. Use only the supplied context. +Return at most 5 bullets. Focus on likely cause, next command, and risk. + +Topic: {topic} + +Context: +{context} +""" + +payload = { + "model": model, + "prompt": prompt, + "stream": False, + "options": { + "temperature": 0.1, + "num_predict": 220, + }, +} + +try: + tags_request = urllib.request.Request(f"{endpoint.rstrip('/')}/api/tags") + with urllib.request.urlopen(tags_request, timeout=3) as response: + tags = json.loads(response.read().decode("utf-8")) + available = {item.get("name", "") for item in tags.get("models", [])} + if model not in available and f"{model}:latest" not in available: + print(f"\n== Backstage Notes ==\nOllama is reachable, but model '{model}' is not pulled. Run: ollama pull {model}") + raise SystemExit(0) + + request = urllib.request.Request( + f"{endpoint.rstrip('/')}/api/generate", + data=json.dumps(payload).encode("utf-8"), + headers={"Content-Type": "application/json"}, + method="POST", + ) + with urllib.request.urlopen(request, timeout=timeout_seconds) as response: + result = json.loads(response.read().decode("utf-8")) +except (OSError, TimeoutError, urllib.error.URLError, json.JSONDecodeError): + raise SystemExit(0) + +note = (result.get("response") or "").strip() +if note: + print("\n== Backstage Notes ==") + print(note) +PY +} + +status_systemd_units() { + local unit + + for unit in "$@"; do + if systemctl list-unit-files --no-legend "${unit}.service" 2>/dev/null | grep -q .; then + printf '%-28s %s\n' "${unit}" "$(systemctl is-active "${unit}" 2>/dev/null || true)" + else + printf '%-28s not-installed\n' "${unit}" + fi + done +} + +status_compose_stack() { + local name="$1" + local directory="$2" + + printf '\n-- Docker Compose: %s (%s)\n' "${name}" "${directory}" + if [[ ! -d "${directory}" ]]; then + printf 'missing directory\n' + return 0 + fi + if ! command -v docker >/dev/null 2>&1; then + printf 'docker not installed\n' + return 0 + fi + if ! sudo docker compose version >/dev/null 2>&1; then + printf 'docker compose unavailable\n' + return 0 + fi + + (cd "${directory}" && sudo docker compose ps) || true +} + +status_kubernetes() { + status_section "Kubernetes" + + if ! command -v kubectl >/dev/null 2>&1; then + printf 'kubectl not installed\n' + return 0 + fi + if [[ ! -s "${KUBECONFIG_PATH}" ]]; then + printf 'kubeconfig missing: %s\n' "${KUBECONFIG_PATH}" + return 0 + fi + if ! kubectl --kubeconfig "${KUBECONFIG_PATH}" get --raw=/readyz >/dev/null 2>&1; then + printf 'API server not reachable through %s\n' "${KUBECONFIG_PATH}" + return 0 + fi + + status_run "nodes" kubectl --kubeconfig "${KUBECONFIG_PATH}" get nodes -o wide + status_kubernetes_problem_pods + status_run "traefik services" kubectl --kubeconfig "${KUBECONFIG_PATH}" -n traefik get svc,pods -o wide + status_run "website pods" kubectl --kubeconfig "${KUBECONFIG_PATH}" -n website-production get pods -o wide +} + +status_kubernetes_problem_pods() { + local rows + + printf '\n-- pods not Running/Completed\n' + rows="$( + kubectl --kubeconfig "${KUBECONFIG_PATH}" get pods -A --no-headers -o wide | + awk '$4 != "Running" && $4 != "Completed" { print }' + )" + if [[ -n "${rows}" ]]; then + printf '%s\n' "${rows}" + else + printf 'none\n' + fi +} + +status_pimox_workers() { + local pimox_host="${LAB_PIMOX_HOST:-${TF_VAR_pimox_host:-192.168.100.80}}" + local pimox_user="${LAB_PIMOX_USER:-${TF_VAR_pimox_user:-jv}}" + local pimox_key="${LAB_PIMOX_SSH_KEY_PATH:-${TF_VAR_pimox_ssh_key_path:-/home/jv/.ssh/id_ed25519}}" + local qm_bin="${LAB_PIMOX_QM_BIN:-${TF_VAR_pimox_qm_bin:-/usr/sbin/qm}}" + local worker_count + local worker_base_vmid="${LAB_PIMOX_WORKER_BASE_VMID:-9010}" + local worker_skip_indexes="${LAB_PIMOX_SKIP_WORKER_INDEXES:-}" + local index + local vmid + + status_section "Pimox Workers" + + worker_count="$(pimox_worker_count_effective)" + if ! [[ "${worker_count}" =~ ^[0-9]+$ ]]; then + printf 'invalid LAB_PIMOX_WORKER_COUNT: %s\n' "${worker_count}" + return 0 + fi + if ((worker_count == 0)); then + printf 'worker count is 0\n' + return 0 + fi + + for ((index = 1; index <= worker_count; index++)); do + if worker_index_is_skipped "${index}" "${worker_skip_indexes}"; then + printf 'worker index %s skipped\n' "${index}" + continue + fi + vmid=$((worker_base_vmid + index - 1)) + printf 'VM %-6s ' "${vmid}" + pimox_ssh "${pimox_host}" "${pimox_user}" "${pimox_key}" "sudo '${qm_bin}' status '${vmid}'" 2>/dev/null || + printf 'unreachable or missing\n' + done +} + +status_rpi_services() { + local rpi_host="${LAB_RPI_HOST:-${LAB_RASPBERRY_HOST:-192.168.100.89}}" + local rpi_user="${LAB_RPI_USER:-${LAB_RASPBERRY_USER:-jv}}" + local rpi_key="${LAB_RPI_SSH_KEY_PATH:-${LAB_RASPBERRY_SSH_KEY_PATH:-/home/jv/.ssh/id_ed25519}}" + local install_dir="${LAB_RPI_SERVICES_INSTALL_DIR:-/opt/homelab-rpi-services}" + local docker_nvme_root="${LAB_RPI_DOCKER_ROOT:-${LAB_RPI_DOCKER_NVME_ROOT:-/nvme-storage/docker}}" + local docker_fallback_root="${LAB_RPI_DOCKER_FALLBACK_ROOT:-/var/lib/docker}" + local pihole_container="${PIHOLE_CONTAINER_NAME:-homelab-pihole}" + local unbound_container="${UNBOUND_CONTAINER_NAME:-homelab-unbound}" + local uptime_kuma_port="${UPTIME_KUMA_PORT:-3001}" + + status_section "RPi Services" + ssh -i "${rpi_key}" -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new "${rpi_user}@${rpi_host}" " +set +e +dns_query() { + server=\"\$1\" + port=\"\$2\" + name=\"\$3\" + python3 - \"\$server\" \"\$port\" \"\$name\" <<'PY' +import random +import socket +import struct +import sys + +server, port, name = sys.argv[1], int(sys.argv[2]), sys.argv[3].rstrip('.') +query_id = random.randrange(0, 65536) +packet = struct.pack('!HHHHHH', query_id, 0x0100, 1, 0, 0, 0) +for label in name.split('.'): + packet += bytes([len(label)]) + label.encode('ascii') +packet += b'\x00' + struct.pack('!HH', 1, 1) +sock = socket.socket(socket.AF_INET, socket.SOCK_DGRAM) +sock.settimeout(5) +sock.sendto(packet, (server, port)) +data, _ = sock.recvfrom(512) +if len(data) < 12: + raise SystemExit('short DNS response') +response_id, flags, _qd, an, _ns, _ar = struct.unpack('!HHHHHH', data[:12]) +rcode = flags & 0x000f +if response_id != query_id: + raise SystemExit('mismatched DNS response id') +if rcode != 0: + raise SystemExit(f'DNS rcode {rcode}') +if an < 1: + raise SystemExit('DNS response had no answers') +print(f'ok answers={an}') +PY +} +check() { + label=\"\$1\" + shift + printf '%-34s ' \"\$label\" + output=\"\$("\$@" 2>&1)\" + status=\$? + if [ \"\$status\" -eq 0 ]; then + printf 'ok' + if [ -n \"\$output\" ]; then + printf ' - %s' \"\$(printf '%s' \"\$output\" | head -n 1)\" + fi + printf '\n' + else + printf 'fail - %s\n' \"\$(printf '%s' \"\$output\" | head -n 1)\" + fi + return 0 +} +docker_root=\"\$(sudo docker info --format '{{.DockerRootDir}}' 2>/dev/null || true)\" +printf '%-34s %s\n' 'Docker root' \"\${docker_root:-unknown}\" +if [ \"\$docker_root\" = '${docker_nvme_root}' ]; then + if mountpoint -q '${docker_nvme_root}'; then + printf '%-34s ok - nvme mount active\n' 'Docker root class' + else + printf '%-34s warn - configured for nvme but mountpoint is missing\n' 'Docker root class' + fi +elif [ \"\$docker_root\" = '${docker_fallback_root}' ]; then + printf '%-34s warn - fallback root active; expected ${docker_nvme_root}\n' 'Docker root class' +else + printf '%-34s warn - unexpected docker root\n' 'Docker root class' +fi +if [ -d '${install_dir}' ]; then + cd '${install_dir}' + sudo docker compose ps || true + unbound_ip=\"\$(sudo docker inspect -f '{{range.NetworkSettings.Networks}}{{.IPAddress}}{{end}}' '${unbound_container}' 2>/dev/null || true)\" + check 'Pi-hole DNS query' dns_query 127.0.0.1 53 cloudflare.com + if [ -n \"\$unbound_ip\" ]; then + check 'Unbound DNS query' dns_query \"\$unbound_ip\" 53 cloudflare.com + else + printf '%-34s fail - container IP unavailable\n' 'Unbound DNS query' + fi + upstreams=\"\$(sudo docker inspect -f '{{range .Config.Env}}{{println .}}{{end}}' '${pihole_container}' 2>/dev/null | grep -E '^PIHOLE_DNS_=|^FTLCONF_dns_upstreams=' || true)\" + if printf '%s\n' \"\$upstreams\" | grep -Eq '1\\.1\\.1\\.1|9\\.9\\.9\\.9|8\\.8\\.8\\.8'; then + printf '%-34s ok - public fallback configured\n' 'Pi-hole fallback config' + else + printf '%-34s warn - no public fallback found in visible config\n' 'Pi-hole fallback config' + fi + check 'Fallback resolver 1.1.1.1' dns_query 1.1.1.1 53 cloudflare.com + check 'Fallback resolver 9.9.9.9' dns_query 9.9.9.9 53 cloudflare.com + check 'Uptime Kuma HTTP' curl -fsS --max-time 5 \"http://127.0.0.1:${uptime_kuma_port}/\" +else + echo 'RPi services install directory is missing: ${install_dir}' +fi" || printf 'unable to reach %s@%s\n' "${rpi_user}" "${rpi_host}" +} + +status_report() { + local cascade_status=0 + local heal_mode="" + local status_json + local -a status_args=() + + require_debian_server "status" + + while (($# > 0)); do + case "$1" in + --heal-plan) + heal_mode="plan" + shift + ;; + --heal) + heal_mode="apply" + shift + ;; + *) + status_args+=("$1") + shift + ;; + esac + done + + if ! report_ui_parse_args "${status_args[@]}"; then + return 1 + fi + + if [[ -n "${heal_mode}" ]]; then + status_json="$(mktemp "${TMPDIR:-/tmp}/jeannie-status-json.XXXXXX")" + # shellcheck disable=SC2034 + REPORT_UI_JSON=true + # shellcheck disable=SC2034 + REPORT_UI_ONLY=problems + report_ui_begin + JEANNIE_REPORT_MODE=true status_cascade_report >/dev/null || cascade_status=$? + report_ui_render "Jeannie Status" >"${status_json}" + report_ui_cleanup + if [[ "${heal_mode}" == "plan" ]]; then + "${REPO_ROOT}/scripts/heal" plan --status-json "${status_json}" --ai + else + "${REPO_ROOT}/scripts/heal" apply --status-json "${status_json}" --ai + fi + rm -f "${status_json}" + return "${cascade_status}" + fi + + if [[ "${REPORT_UI_VERBOSE}" != "true" ]]; then + report_ui_begin + JEANNIE_REPORT_MODE=true status_cascade_report || cascade_status=$? + report_ui_render "Jeannie Status" + report_ui_cleanup + return "${cascade_status}" + fi + + status_cascade_report || cascade_status=$? + + if ! truthy "${LAB_STATUS_DETAILS:-false}" && [[ "${REPORT_UI_DETAILS}" != "true" ]]; then + return "${cascade_status}" + fi + + status_section "Host" + printf 'hostname: %s\n' "$(hostname -f 2>/dev/null || hostname)" + printf 'date: %s\n' "$(date -Is)" + printf 'uptime: %s\n' "$(uptime -p 2>/dev/null || uptime)" + status_run "memory" free -h + status_run "disk" df -h / /data /data/openebs/local + + status_section "Systemd" + status_systemd_units ssh docker containerd kubelet tailscaled homelab-gitea-runner + + status_section "Docker" + if command -v docker >/dev/null 2>&1; then + status_run "docker containers" sudo docker ps --format 'table {{.Names}}\t{{.Image}}\t{{.Status}}\t{{.Ports}}' + else + printf 'docker not installed\n' + fi + status_compose_stack "gitea" "/data/homelab-gitea" + status_compose_stack "heimdall" "${LAB_HEIMDALL_INSTALL_DIR:-/data/homelab-heimdall}" + status_compose_stack "arr-stack" "${REPO_ROOT}/infra/arr-stack" + + status_kubernetes + status_pimox_workers + status_rpi_services + + status_section "Tailscale" + if command -v tailscale >/dev/null 2>&1; then + status_run "tailscale ip" tailscale ip -4 + status_run "tailscale peers" tailscale status + else + printf 'tailscale not installed\n' + fi + + status_section "HTTP" + status_http "website public" "${LAB_PUBLIC_URL:?LAB_PUBLIC_URL is required from homelab.yml}/" + status_http "gitea public" "${LAB_GITEA_ROOT_URL:-${LAB_PUBLIC_URL:?LAB_PUBLIC_URL is required from homelab.yml}/git/}" + status_http "heimdall public" "${LAB_HEIMDALL_PUBLIC_URL:-https://heimdall.${LAB_DOMAIN:?LAB_DOMAIN is required from homelab.yml}/}" + status_http "gitea local" "http://127.0.0.1:${LAB_GITEA_HTTP_PORT:-3000}/" + status_http "heimdall local" "http://127.0.0.1:${LAB_HEIMDALL_HTTP_PORT:-8082}/" + status_http "traefik lb" "http://${LAB_TRAEFIK_LB_IP:?LAB_TRAEFIK_LB_IP is required from homelab.yml}/" + status_http "arr radarr" "http://127.0.0.1:7878/" + status_http "arr sonarr" "http://127.0.0.1:8989/" + status_http "arr prowlarr" "http://127.0.0.1:9696/" + + return "${cascade_status}" +} + +check_edge_traefik_backend() { + local edge_host="${LAB_EDGE_HOST:?LAB_EDGE_HOST is required from homelab.yml}" + local edge_user="${LAB_EDGE_USER:-ubuntu}" + local traefik_ip="${LAB_TRAEFIK_LB_IP:?LAB_TRAEFIK_LB_IP is required from homelab.yml}" + + ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new "${edge_user}@${edge_host}" \ + "curl -sS -o /dev/null --connect-timeout 5 --max-time 10 -w '%{http_code}' 'http://${traefik_ip}:80/' | grep -Eq '^[234]'" +} + +check_edge_gitea_backend() { + local edge_host="${LAB_EDGE_HOST:?LAB_EDGE_HOST is required from homelab.yml}" + local edge_user="${LAB_EDGE_USER:-ubuntu}" + local gitea_ts_ip="${LAB_GITEA_TAILSCALE_IP:?LAB_GITEA_TAILSCALE_IP is required from homelab.yml}" + local gitea_http_port="${LAB_GITEA_HTTP_PORT:-3000}" + + ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new "${edge_user}@${edge_host}" \ + "curl -sS -o /dev/null --connect-timeout 5 --max-time 10 -w '%{http_code}' 'http://${gitea_ts_ip}:${gitea_http_port}/' | grep -Eq '^[234]'" +} + +check_edge_heimdall_backend() { + local edge_host="${LAB_EDGE_HOST:?LAB_EDGE_HOST is required from homelab.yml}" + local edge_user="${LAB_EDGE_USER:-ubuntu}" + local heimdall_host="${LAB_HEIMDALL_BACKEND_HOST:-${LAB_GITEA_TAILSCALE_IP:?LAB_GITEA_TAILSCALE_IP is required from homelab.yml}}" + local heimdall_http_port="${LAB_HEIMDALL_HTTP_PORT:-8082}" + + ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new "${edge_user}@${edge_host}" \ + "curl -sS -o /dev/null --connect-timeout 5 --max-time 10 -w '%{http_code}' 'http://${heimdall_host}:${heimdall_http_port}/' | grep -Eq '^[234]'" +} + +check_gitea_git_remote() { + git ls-remote gitea main >/dev/null +} + +check_gitea_ssh() { + local gitea_host="${LAB_GITEA_HOST:-192.168.100.73}" + local gitea_ssh_port="${LAB_GITEA_SSH_PORT:-32222}" + + ssh -T -p "${gitea_ssh_port}" -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new "git@${gitea_host}" 2>&1 | + grep -Eiq 'successfully authenticated|Hi |shell access is disabled' +} + +doctor_edge() { + local edge_host="${LAB_EDGE_HOST:?LAB_EDGE_HOST is required from homelab.yml}" + local edge_user="${LAB_EDGE_USER:-ubuntu}" + local traefik_ip="${LAB_TRAEFIK_LB_IP:?LAB_TRAEFIK_LB_IP is required from homelab.yml}" + local gitea_ts_ip="${LAB_GITEA_TAILSCALE_IP:?LAB_GITEA_TAILSCALE_IP is required from homelab.yml}" + local gitea_http_port="${LAB_GITEA_HTTP_PORT:-3000}" + local heimdall_host="${LAB_HEIMDALL_BACKEND_HOST:-${gitea_ts_ip}}" + local heimdall_http_port="${LAB_HEIMDALL_HTTP_PORT:-8082}" + local failures=0 + + require_debian_server "doctor-edge" + if ! report_ui_parse_args "$@"; then + return 1 + fi + + if [[ "${REPORT_UI_VERBOSE}" != "true" ]]; then + report_ui_begin + report_ui_check fail "Public URLs" "Website public" "./jeannie doctor-edge --verbose" "" "" status_http_ok "${LAB_PUBLIC_URL:?LAB_PUBLIC_URL is required from homelab.yml}/" || failures=$((failures + 1)) + report_ui_check fail "Public URLs" "Gitea public" "./jeannie doctor-gitea" "" "" status_gitea_public_ok || failures=$((failures + 1)) + report_ui_check fail "Edge" "OCI edge SSH" "./jeannie preflight" "" "" check_edge_ssh || failures=$((failures + 1)) + report_ui_check fail "Backends" "Traefik from edge" "./jeannie doctor-cluster" "" "" check_edge_traefik_backend || failures=$((failures + 1)) + report_ui_check fail "Backends" "Gitea from edge" "./jeannie doctor-gitea" "" "" check_edge_gitea_backend || failures=$((failures + 1)) + report_ui_check fail "Backends" "Heimdall from edge" "./jeannie deploy-heimdall" "" "" check_edge_heimdall_backend || failures=$((failures + 1)) + report_ui_render "Doctor Edge" + report_ui_cleanup + return "${failures}" + fi + + status_section "Doctor Edge" + status_http "website public" "${LAB_PUBLIC_URL:?LAB_PUBLIC_URL is required from homelab.yml}/" + status_http "gitea public" "${LAB_GITEA_ROOT_URL:-${LAB_PUBLIC_URL:?LAB_PUBLIC_URL is required from homelab.yml}/git/}" + status_http "heimdall public" "${LAB_HEIMDALL_PUBLIC_URL:-https://heimdall.${LAB_DOMAIN:?LAB_DOMAIN is required from homelab.yml}/}" + + printf '\n-- edge backend reachability\n' + if ! ssh -o BatchMode=yes -o ConnectTimeout=10 -o StrictHostKeyChecking=accept-new "${edge_user}@${edge_host}" " +set +e +printf '%-28s ' 'traefik backend' +curl -sS -o /dev/null --connect-timeout 5 --max-time 10 -w '%{http_code}\n' 'http://${traefik_ip}:80/' || echo unreachable +printf '%-28s ' 'gitea backend' +curl -sS -o /dev/null --connect-timeout 5 --max-time 10 -w '%{http_code}\n' 'http://${gitea_ts_ip}:${gitea_http_port}/' || echo unreachable +printf '%-28s ' 'heimdall backend' +curl -sS -o /dev/null --connect-timeout 5 --max-time 10 -w '%{http_code}\n' 'http://${heimdall_host}:${heimdall_http_port}/' || echo unreachable +printf '\n-- tailscale\n' +tailscale status || true +"; then + printf 'unable to reach edge host %s@%s\n' "${edge_user}" "${edge_host}" + fi + + cat <<'EOF' + +Next steps: +- If only Traefik is unreachable, check cluster/MetalLB or run ./jeannie doctor-cluster. +- If only Gitea is unreachable, run ./jeannie doctor-gitea. +- If only Heimdall is unreachable, run ./jeannie deploy-heimdall. +- If both are unreachable from OCI, check Tailscale routes and bootstrap/edge. +- Runbook: docs/runbooks/edge-failures.md +EOF + + backstage_brain_note "doctor-edge" </dev/null 2>&1; then + sudo crictl ps -a | head -50 || true + else + printf 'crictl not installed\n' + fi + + cat <<'EOF' + +Next steps: +- If API server is unreachable after a stop, run ./jeannie start-cluster. +- If nodes are NotReady, inspect kube-system pods and containerd/kubelet on that node. +- If Pimox workers are stopped, start them with ./jeannie start-cluster. +- If kubelet versions drift, run ./jeannie doctor-versions. +- Runbook: docs/runbooks/cluster-stop-start-failures.md +EOF + + backstage_brain_note "doctor-cluster" <&2 <<'EOF' +nuke destroys Kubernetes state and Pimox worker VMs. +Rerun with LAB_CONFIRM_NUKE=homelab when that destructive action is intended. +EOF + exit 1 + fi + + echo "Brutally nuking the homelab infrastructure..." + cluster_worker_targets + + if ((${#CLUSTER_WORKER_TARGETS[@]} > 0)); then + mkdir -p "$(dirname "${worker_known_hosts_file}")" + touch "${worker_known_hosts_file}" + chmod 0600 "${worker_known_hosts_file}" + fi + + echo "--> Terminating local OpenTofu tasks..." + killall tofu terraform 2>/dev/null || true + + echo "--> Eviscerating local Kubernetes components..." + cleanup_node + sudo rm -f "${KUBECONFIG_PATH}" + + for target in "${CLUSTER_WORKER_TARGETS[@]}"; do + echo "--> Eviscerating remote Kubernetes components (${target})..." + target_host="${target#*@}" + target_host="${target_host%%:*}" + ssh-keygen -R "${target_host}" -f "${worker_known_hosts_file}" >/dev/null 2>&1 || true + if ! ssh -o ConnectTimeout=5 -o StrictHostKeyChecking=accept-new -o UserKnownHostsFile="${worker_known_hosts_file}" "${target}" "bash -s" <<'EOF' +set -euo pipefail + +cleanup_calico_links() { + ip link show | awk -F: '/^[0-9]+: cali/ {print $2}' | cut -d@ -f1 | xargs -r -n1 sudo ip link delete 2>/dev/null || true + sudo ip link delete vxlan.calico 2>/dev/null || true + sudo ip link delete tunl0 2>/dev/null || true + sudo ip link delete cni0 2>/dev/null || true + sudo ip link delete kube-ipvs0 2>/dev/null || true + ip netns list | awk '/^(cni-|calico)/ {print $1}' | xargs -r -n1 sudo ip netns delete 2>/dev/null || true +} + +cleanup_iptables() { + sudo iptables -F || true + sudo iptables -X || true + sudo iptables -t nat -F || true + sudo iptables -t nat -X || true + sudo iptables -t mangle -F || true + sudo iptables -t mangle -X || true + sudo iptables -t raw -F || true + sudo iptables -t raw -X || true + if command -v ipvsadm >/dev/null 2>&1; then + sudo ipvsadm --clear || true + fi +} + +cleanup_calico_runtime_files() { + local path + + for path in /run/calico /var/run/calico; do + if sudo test -e "${path}"; then + sudo find "${path}" -path '*/cgroup*' -prune -o -mindepth 1 -exec rm -rf -- {} + 2>/dev/null || true + sudo rmdir "${path}" 2>/dev/null || true + fi + done +} + +restore_node_dns() { + sudo rm -f /etc/systemd/resolved.conf.d/homelab-k8s.conf + if sudo test -e /etc/resolv.conf.homelab-k8s-backup; then + sudo rm -f /etc/resolv.conf + sudo mv /etc/resolv.conf.homelab-k8s-backup /etc/resolv.conf + fi + sudo systemctl restart systemd-resolved 2>/dev/null || true +} + +cleanup_mounts() { + if command -v findmnt >/dev/null 2>&1; then + local mount_root + while IFS= read -r mountpoint; do + sudo umount -f "${mountpoint}" 2>/dev/null || sudo umount -l "${mountpoint}" 2>/dev/null || true + done < <( + for mount_root in /var/lib/kubelet /var/lib/containerd /run/calico /run/calico/cgroup /var/run/calico /var/run/calico/cgroup; do + findmnt -Rno TARGET "${mount_root}" 2>/dev/null || true + done | sort -ru + ) + fi + while IFS= read -r mountpoint; do + sudo umount -f "${mountpoint}" 2>/dev/null || sudo umount -l "${mountpoint}" 2>/dev/null || true + done < <(find /var/lib/kubelet/pods -mindepth 2 -maxdepth 5 -type d 2>/dev/null || true) + sudo umount -f /var/lib/containerd/srun/* 2>/dev/null || sudo umount -l /var/lib/containerd/srun/* 2>/dev/null || true +} + +sudo kubeadm reset --force || true +sudo systemctl stop kubelet 2>/dev/null || true +sudo systemctl stop containerd 2>/dev/null || true +sudo killall containerd-shim-runc-v2 2>/dev/null || true + +cleanup_mounts + +sudo rm -rf \ + /etc/kubernetes/ \ + /var/lib/etcd/ \ + /var/lib/kubelet/ \ + /var/lib/cni/ \ + /etc/cni/net.d \ + /run/flannel \ + /var/lib/calico \ + /var/log/calico \ + /var/lib/containerd/* \ + /run/containerd/* \ + /etc/containerd/certs.d \ + /etc/containerd/config.toml +cleanup_calico_runtime_files +sudo rm -f /opt/cni/bin/calico /opt/cni/bin/calico-ipam + +cleanup_iptables +cleanup_calico_links +restore_node_dns + +sudo mkdir -p /etc/containerd/certs.d +sudo systemctl reset-failed kubelet containerd 2>/dev/null || true +sudo systemctl start containerd 2>/dev/null || true +EOF + then + echo "Remote cleanup failed for ${target}; not deleting OpenTofu state." >&2 + exit 1 + fi + done + + destroy_pimox_worker_vms + + docker buildx rm lab-builder 2>/dev/null || true + docker rm -f buildx_buildkit_lab-builder0 2>/dev/null || true + rm -f "${BUILDX_CONFIG}" || true + + echo "--> Backing up local OpenTofu tracking state files..." + backup_tofu_state + + echo "--> Deleting OpenTofu tracking state files..." + rm -rf "${REPO_ROOT}"/bootstrap/cluster/terraform.tfstate* + rm -f "${REPO_ROOT}"/bootstrap/cluster/.terraform.tfstate.lock.info + rm -rf "${REPO_ROOT}"/bootstrap/cluster/.terraform/ + rm -rf "${REPO_ROOT}"/bootstrap/platform/terraform.tfstate* + rm -f "${REPO_ROOT}"/bootstrap/platform/.terraform.tfstate.lock.info + rm -rf "${REPO_ROOT}"/bootstrap/platform/.terraform/ + rm -rf "${REPO_ROOT}"/bootstrap/apps/terraform.tfstate* + rm -f "${REPO_ROOT}"/bootstrap/apps/.terraform.tfstate.lock.info + rm -rf "${REPO_ROOT}"/bootstrap/apps/.terraform/ + rm -rf "${REPO_ROOT}"/bootstrap/edge/terraform.tfstate* + rm -f "${REPO_ROOT}"/bootstrap/edge/.terraform.tfstate.lock.info + rm -rf "${REPO_ROOT}"/bootstrap/edge/.terraform/ + rm -f "${REPO_ROOT}/.lab/pimox-workers.tsv" "${REPO_ROOT}/.lab/cluster-workers.auto.tfvars.json" + + echo "Destruction complete. Retained data under /data/openebs/local was left intact." +} + +ensure_sops_age_tools() { + local missing_packages=() + + if ! command -v age-keygen >/dev/null 2>&1; then + missing_packages+=(age) + fi + if ! command -v sops >/dev/null 2>&1; then + missing_packages+=(sops) + fi + + if ((${#missing_packages[@]} > 0)); then + echo "Installing missing secret-management tools: ${missing_packages[*]}" + sudo apt-get update + sudo apt-get install -y --no-install-recommends "${missing_packages[@]}" + fi + + if ! command -v age-keygen >/dev/null 2>&1; then + echo "age-keygen is still unavailable after package installation." >&2 + exit 1 + fi + if ! command -v sops >/dev/null 2>&1; then + echo "sops is still unavailable after package installation." >&2 + exit 1 + fi +} + +sops_age_key_file() { + printf '%s\n' "${SOPS_AGE_KEY_FILE:-${HOME}/.config/sops/age/keys.txt}" +} + +sops_age_recipient() { + local key_file="$1" + + awk -F': ' '/^# public key:/ { print $2; exit }' "${key_file}" +} + +secrets_init() { + local key_file + local key_dir + local recipient + local config_file="${SOPS_CONFIG:-${REPO_ROOT}/.sops.yaml}" + local example_file="${REPO_ROOT}/.sops.yaml.example" + + require_debian_server "secrets-init" + ensure_sops_age_tools + + key_file="$(sops_age_key_file)" + key_dir="$(dirname "${key_file}")" + mkdir -p "${key_dir}" + chmod 700 "${key_dir}" + + if [[ ! -s "${key_file}" ]]; then + echo "Generating age identity at ${key_file}..." + age-keygen -o "${key_file}" + chmod 600 "${key_file}" + else + echo "Using existing age identity at ${key_file}." + chmod 600 "${key_file}" + fi + + recipient="$(sops_age_recipient "${key_file}")" + if [[ -z "${recipient}" ]]; then + echo "Could not read the public recipient from ${key_file}." >&2 + exit 1 + fi + + if [[ -e "${config_file}" ]]; then + if grep -q 'age1replacewithyourpublicrecipient' "${config_file}"; then + echo "${config_file} still contains the placeholder age recipient." >&2 + exit 1 + fi + echo "SOPS config already exists: ${config_file}" + else + if [[ ! -s "${example_file}" ]]; then + echo "Missing ${example_file}" >&2 + exit 1 + fi + sed "s/age1replacewithyourpublicrecipient/${recipient}/g" "${example_file}" >"${config_file}" + echo "Wrote ${config_file} with Debian host age recipient ${recipient}." + fi + + echo "Private age key: ${key_file}" + echo "Public age recipient: ${recipient}" + echo "Review and commit ${config_file}; never commit ${key_file}." +} + +secrets_check_file() { + local file="$1" + local key_file="$2" + + case "${file}" in + *.json) + grep -q '"sops"' "${file}" || { + echo "${file} does not look SOPS-encrypted." >&2 + return 1 + } + ;; + *) + grep -q '^sops:' "${file}" || { + echo "${file} does not look SOPS-encrypted." >&2 + return 1 + } + ;; + esac + + if [[ -s "${key_file}" ]]; then + SOPS_AGE_KEY_FILE="${key_file}" sops -d "${file}" >/dev/null + fi +} + +secrets_check() { + local config_file="${SOPS_CONFIG:-${REPO_ROOT}/.sops.yaml}" + local key_file + local recipient="" + local failures=0 + local found_files=0 + local file + + key_file="$(sops_age_key_file)" + + if [[ ! -s "${config_file}" ]]; then + echo "Missing ${config_file}. Run ./jeannie secrets-init on the Debian host." >&2 + failures=$((failures + 1)) + elif grep -q 'age1replacewithyourpublicrecipient' "${config_file}"; then + echo "${config_file} still contains the placeholder age recipient." >&2 + failures=$((failures + 1)) + fi + + if [[ -s "${key_file}" ]]; then + recipient="$(sops_age_recipient "${key_file}")" + if [[ -n "${recipient}" && -s "${config_file}" ]] && ! grep -q "${recipient}" "${config_file}"; then + echo "${config_file} does not include this host's age recipient ${recipient}." >&2 + failures=$((failures + 1)) + fi + else + echo "No local age key found at ${key_file}; encrypted file structure will be checked without decrypting." + fi + + if command -v git >/dev/null 2>&1; then + while IFS= read -r file; do + [[ -n "${file}" ]] || continue + found_files=$((found_files + 1)) + secrets_check_file "${REPO_ROOT}/${file}" "${key_file}" || failures=$((failures + 1)) + done < <(git -C "${REPO_ROOT}" ls-files '*.secret.yaml' '*.secret.yml' '*.secret.json' '*.enc.yaml' '*.enc.yml' '*.enc.json') + fi + + if ((failures > 0)); then + echo "Secret checks failed with ${failures} issue(s)." >&2 + exit 1 + fi + + if ((found_files == 0)); then + echo "SOPS config checks passed. No encrypted secret files are committed yet." + else + echo "SOPS config checks passed for ${found_files} encrypted secret file(s)." + fi +} + +tailnet_policy_check() { + "${REPO_ROOT}/scripts/validate-tailnet-policy" +} + +ai_index() { + local index_dir="${LAB_AI_KNOWLEDGE_INDEX_DIR:-/data/homelab-ai/index}" + + require_debian_server "ai-index" + ensure_python3 + if [[ "${index_dir}" == /data/* ]]; then + sudo mkdir -p "${index_dir}" + sudo chown "${USER}:$(id -gn)" "${index_dir}" + fi + "${REPO_ROOT}/scripts/build-homelab-ai-index" --index-dir "${index_dir}" +} + +ai_check() { + local index_dir="${LAB_AI_KNOWLEDGE_INDEX_DIR:-/data/homelab-ai/index}" + local endpoint="${LAB_AI_GATEWAY_URL:-${LAB_OLLAMA_URL:-http://127.0.0.1:11434}}" + local model="${LAB_AI_GATEWAY_MODEL:-qwen2.5:0.5b}" + local failures=0 + + require_debian_server "ai-check" + + status_section "AI Knowledge" + if command -v python3 >/dev/null 2>&1; then + printf '%-28s ok\n' "python3" + else + printf '%-28s missing\n' "python3" + failures=$((failures + 1)) + fi + + if [[ -s "${index_dir}/index.json" ]]; then + printf '%-28s %s\n' "knowledge index" "${index_dir}/index.json" + else + printf '%-28s missing - run ./jeannie ai-index\n' "knowledge index" + failures=$((failures + 1)) + fi + + if "${REPO_ROOT}/scripts/query-homelab-ai-index" --index-dir "${index_dir}" --context-only --limit 2 "gitea edge traefik" >/dev/null 2>&1; then + printf '%-28s ok\n' "retrieval smoke test" + else + printf '%-28s failed\n' "retrieval smoke test" + failures=$((failures + 1)) + fi + + if backstage_brain_enabled; then + if curl -fsS --max-time 5 "${endpoint%/}/api/tags" >/dev/null 2>&1; then + printf '%-28s ok - %s\n' "ollama endpoint" "${endpoint}" + if python3 - "${endpoint}" "${model}" <<'PY' +import json +import sys +import urllib.request + +endpoint, model = sys.argv[1:3] +with urllib.request.urlopen(f"{endpoint.rstrip('/')}/api/tags", timeout=5) as response: + tags = json.loads(response.read().decode("utf-8")) +models = {item.get("name", "") for item in tags.get("models", [])} +raise SystemExit(0 if model in models or f"{model}:latest" in models else 1) +PY + then + printf '%-28s ok - %s\n' "ollama model" "${model}" + else + printf '%-28s missing - run: ollama pull %s\n' "ollama model" "${model}" + failures=$((failures + 1)) + fi + else + printf '%-28s unreachable - %s\n' "ollama endpoint" "${endpoint}" + failures=$((failures + 1)) + fi + else + printf '%-28s disabled\n' "backstage helper" + fi + + if ((failures > 0)); then + echo "AI check failed with ${failures} issue(s)." >&2 + exit 1 + fi + echo "AI check passed." +} + +ask_homelab() { + require_debian_server "ask" + "${REPO_ROOT}/scripts/ask" "${@:2}" +} + +ai_evals() { + "${REPO_ROOT}/scripts/ai-evals" "${@:2}" +} + +impact() { + "${REPO_ROOT}/scripts/impact" "${@:2}" +} + +review_last_change() { + "${REPO_ROOT}/scripts/review-last-change" "${@:2}" +} + +security_scan() { + require_debian_server "security-scan" + "${REPO_ROOT}/scripts/security-scan" all +} + +security_prepare() { + require_debian_server "security-prepare" + "${REPO_ROOT}/scripts/security-scan" prepare +} + +security_zap() { + require_debian_server "security-zap" + "${REPO_ROOT}/scripts/security-scan" zap +} + +security_k8s() { + require_debian_server "security-k8s" + "${REPO_ROOT}/scripts/security-scan" kube-bench +} + +security_host() { + require_debian_server "security-host" + "${REPO_ROOT}/scripts/security-scan" host +} + +security_trivy() { + require_debian_server "security-trivy" + "${REPO_ROOT}/scripts/security-scan" trivy +} + +security_secrets() { + require_debian_server "security-secrets" + "${REPO_ROOT}/scripts/security-scan" secrets +} + +security_nuclei() { + require_debian_server "security-nuclei" + "${REPO_ROOT}/scripts/security-scan" nuclei +} + +security_web() { + require_debian_server "security-web" + "${REPO_ROOT}/scripts/security-scan" web +} + +security_logs() { + require_debian_server "security-logs" + "${REPO_ROOT}/scripts/security-logs" +} + +security_runtime() { + require_debian_server "security-runtime" + if ! command -v kubectl >/dev/null 2>&1; then + echo "kubectl is required for security-runtime." >&2 + exit 1 + fi + + kubectl --kubeconfig "${KUBECONFIG}" -n kube-system rollout status daemonset/tetragon --timeout=30s + kubectl --kubeconfig "${KUBECONFIG}" -n kube-system get pods -l app.kubernetes.io/name=tetragon -o wide + cat <<'EOF' + +Watch recent Tetragon events: + kubectl -n kube-system logs -l app.kubernetes.io/name=tetragon -c export-stdout --tail=100 -f + +Useful practice trigger: + kubectl -n security-lab exec deploy/juice-shop -- sh -c 'id; uname -a' +EOF +} + +security_attack_path() { + require_debian_server "security-attack-path" + "${REPO_ROOT}/scripts/security-attack-path" "${@:2}" +} + +prompt_injection_lab() { + "${REPO_ROOT}/scripts/prompt-injection-lab" "${@:2}" +} + +incident_commander() { + "${REPO_ROOT}/scripts/incident-commander" "${@:2}" +} + +safety_case() { + "${REPO_ROOT}/scripts/safety-case" "${@:2}" +} + +agent_sandbox() { + "${REPO_ROOT}/scripts/agent-sandbox" "${@:2}" +} + +ai_scheduler() { + "${REPO_ROOT}/scripts/ai-scheduler" "${@:2}" +} + +ai_memory() { + "${REPO_ROOT}/scripts/ai-memory" "${@:2}" +} + +red_blue_loop() { + "${REPO_ROOT}/scripts/red-blue-loop" "${@:2}" +} + +promote() { + "${REPO_ROOT}/scripts/promote" "${@:2}" +} + +model_observe() { + "${REPO_ROOT}/scripts/model-observe" "${@:2}" +} + +heal() { + require_debian_server "heal" + "${REPO_ROOT}/scripts/heal" "${@:2}" +} + +validate_homelab() { + "${REPO_ROOT}/scripts/validate-homelab" +} + +access_audit() { + require_debian_server "access-audit" + "${REPO_ROOT}/scripts/access-audit" +} + +drill_restore() { + require_debian_server "drill-restore" + "${REPO_ROOT}/scripts/restore-drill" all +} + +drill_pihole_restore() { + "${REPO_ROOT}/scripts/restore-drill" pihole +} + +kubeconfig_readonly() { + require_debian_server "kubeconfig-readonly" + "${REPO_ROOT}/scripts/kubeconfig-readonly" +} + +capacity_report() { + require_debian_server "capacity" + "${REPO_ROOT}/scripts/capacity-report" "${@:2}" +} + +recover_plan() { + require_debian_server "recover-plan" + "${REPO_ROOT}/scripts/recover-plan" +} + +gitops_status() { + require_debian_server "gitops-status" + "${REPO_ROOT}/scripts/gitops-status" +} + +cert_check() { + "${REPO_ROOT}/scripts/cert-check" +} + +release_snapshot() { + require_debian_server "release-snapshot" + "${REPO_ROOT}/scripts/release-snapshot" +} + +backup_status() { + require_debian_server "backup-status" + "${REPO_ROOT}/scripts/backup-status" +} + +synthetic_checks() { + require_debian_server "synthetic-checks" + "${REPO_ROOT}/scripts/synthetic-checks" +} + +resource_budget() { + require_debian_server "resource-budget" + "${REPO_ROOT}/scripts/resource-budget" "${@:2}" +} + +capacity_advisor() { + require_debian_server "capacity-advisor" + "${REPO_ROOT}/scripts/capacity-advisor" "${@:2}" +} + +capacity_limits() { + require_debian_server "capacity-limits" + "${REPO_ROOT}/scripts/capacity-limits" "${@:2}" +} + +control_plane() { + require_debian_server "control-plane" + "${REPO_ROOT}/scripts/control-plane" "${@:2}" +} + +artifact_cache() { + require_debian_server "artifact-cache" + "${REPO_ROOT}/scripts/artifact-cache" "${@:2}" +} + +blockchain_devnet() { + require_debian_server "blockchain-devnet" + "${REPO_ROOT}/scripts/blockchain-devnet" "${@:2}" +} + +blockchain_test() { + "${REPO_ROOT}/scripts/blockchain-test" "${@:2}" +} + +blockchain_wallet() { + "${REPO_ROOT}/scripts/blockchain-wallet" "${@:2}" +} + +golden_ledger() { + "${REPO_ROOT}/scripts/golden-ledger" "${@:2}" +} + +route_inventory() { + require_debian_server "route-inventory" + "${REPO_ROOT}/scripts/route-inventory" +} + +explain() { + if [ "${2:-}" != "output" ]; then + require_debian_server "explain" + fi + "${REPO_ROOT}/scripts/explain" "${@:2}" +} + +recover_power() { + require_debian_server "recover-power" + record_change_journal "recover-power" "$@" + "${REPO_ROOT}/scripts/recover-power" "${@:2}" +} + +scorecard() { + require_debian_server "scorecard" + "${REPO_ROOT}/scripts/scorecard" +} + +grafana_dashboards() { + require_debian_server "grafana-dashboards" + "${REPO_ROOT}/scripts/grafana-dashboards" "${@:2}" +} + +workers_manage() { + require_debian_server "workers" + "${REPO_ROOT}/scripts/workers" "${@:2}" +} + +record_change_journal() { + local command_name="$1" + shift || true + + "${REPO_ROOT}/scripts/change-journal" append "${command_name}" "$*" || true +} + +change_journal() { + "${REPO_ROOT}/scripts/change-journal" "${@:2}" +} + +homelab_map() { + "${REPO_ROOT}/scripts/homelab-map" "${@:2}" +} +