#!/bin/bash
#
# Proxmox Node Recovery Script
# Attempts to recover a Proxmox node without rebooting the physical server
#
# Usage: ./proxmox-node-recovery.sh [--force] [--skip-cluster] [--skip-storage]
#
# Recovery Philosophy:
#   1. Diagnose first, understand the problem
#   2. Try simple fixes first (service restarts)
#   3. Only escalate to invasive recovery if simple fixes fail
#   4. NEVER do anything that risks data loss without explicit confirmation
#
# Author: DartNode Operations
#

set -o pipefail

# Colors for output
RED='\033[0;31m'
GREEN='\033[0;32m'
YELLOW='\033[1;33m'
BLUE='\033[0;34m'
CYAN='\033[0;36m'
MAGENTA='\033[0;35m'
NC='\033[0m' # No Color

# Configuration
LOG_FILE="/var/log/proxmox-recovery-$(date +%Y%m%d-%H%M%S).log"
FORCE_MODE=false
SKIP_CLUSTER=false
SKIP_STORAGE=false
TIMEOUT_SECONDS=30
DRY_RUN=false

# Track what we've found
declare -A ISSUES_FOUND
declare -a CRITICAL_ISSUES

# Parse arguments
while [[ $# -gt 0 ]]; do
    case $1 in
        --force)
            FORCE_MODE=true
            shift
            ;;
        --skip-cluster)
            SKIP_CLUSTER=true
            shift
            ;;
        --skip-storage)
            SKIP_STORAGE=true
            shift
            ;;
        --dry-run)
            DRY_RUN=true
            shift
            ;;
        --help|-h)
            echo "Proxmox Node Recovery Script"
            echo ""
            echo "Usage: $0 [OPTIONS]"
            echo ""
            echo "Options:"
            echo "  --force         Skip confirmation prompts (still confirms destructive actions)"
            echo "  --skip-cluster  Skip cluster recovery steps"
            echo "  --skip-storage  Skip storage recovery steps"
            echo "  --dry-run       Show what would be done without making changes"
            echo "  --help, -h      Show this help message"
            exit 0
            ;;
        *)
            echo "Unknown option: $1"
            exit 1
            ;;
    esac
done

#=============================================================================
# UTILITY FUNCTIONS
#=============================================================================

log() {
    local level=$1
    shift
    local message="$@"
    local timestamp=$(date '+%Y-%m-%d %H:%M:%S')
    echo -e "${timestamp} [${level}] ${message}" | tee -a "$LOG_FILE"
}

info() { log "INFO" "${BLUE}$@${NC}"; }
success() { log "SUCCESS" "${GREEN}$@${NC}"; }
warn() { log "WARN" "${YELLOW}$@${NC}"; }
error() { log "ERROR" "${RED}$@${NC}"; }
critical() { log "CRITICAL" "${RED}$@${NC}"; CRITICAL_ISSUES+=("$@"); }
section() {
    echo "" | tee -a "$LOG_FILE"
    echo -e "${CYAN}═══════════════════════════════════════════════════════════════${NC}" | tee -a "$LOG_FILE"
    log "SECTION" "${CYAN}$@${NC}"
    echo -e "${CYAN}═══════════════════════════════════════════════════════════════${NC}" | tee -a "$LOG_FILE"
    echo "" | tee -a "$LOG_FILE"
}

phase() {
    echo "" | tee -a "$LOG_FILE"
    echo -e "${MAGENTA}┏━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━┓${NC}" | tee -a "$LOG_FILE"
    echo -e "${MAGENTA}┃  $@${NC}" | tee -a "$LOG_FILE"
    echo -e "${MAGENTA}┗━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━┛${NC}" | tee -a "$LOG_FILE"
    echo "" | tee -a "$LOG_FILE"
}

check_root() {
    if [[ $EUID -ne 0 ]]; then
        error "This script must be run as root"
        exit 1
    fi
}

# Standard confirmation - skipped with --force
confirm() {
    if [[ "$FORCE_MODE" == "true" ]]; then
        return 0
    fi
    local message=$1
    read -p "$message [y/N]: " response
    [[ "$response" =~ ^[Yy]$ ]]
}

# Dangerous confirmation - NEVER skipped, requires typing 'yes'
confirm_dangerous() {
    local message=$1
    echo -e "${RED}╔═══════════════════════════════════════════════════════════════╗${NC}"
    echo -e "${RED}║  ⚠️  WARNING: POTENTIALLY DESTRUCTIVE ACTION                  ║${NC}"
    echo -e "${RED}╚═══════════════════════════════════════════════════════════════╝${NC}"
    echo -e "${YELLOW}$message${NC}"
    echo ""
    read -p "Type 'yes' to confirm: " response
    [[ "$response" == "yes" ]]
}

service_running() {
    systemctl is-active --quiet "$1"
}

# Execute or show what would be done
execute() {
    if [[ "$DRY_RUN" == "true" ]]; then
        info "[DRY-RUN] Would execute: $@"
        return 0
    else
        "$@"
    fi
}

# Safe service restart with timeout
safe_restart_service() {
    local service=$1
    local timeout=${2:-60}

    info "Attempting to restart $service (timeout: ${timeout}s)..."

    if [[ "$DRY_RUN" == "true" ]]; then
        info "[DRY-RUN] Would restart $service"
        return 0
    fi

    if timeout "$timeout" systemctl restart "$service" 2>&1; then
        success "$service restarted successfully"
        return 0
    else
        error "Failed to restart $service within ${timeout}s"
        return 1
    fi
}

#=============================================================================
# PHASE 1: DIAGNOSTICS
#=============================================================================

diagnose_system() {
    section "System Health Check"

    info "Hostname: $(hostname)"
    info "Uptime: $(uptime -p)"
    info "Load average: $(cat /proc/loadavg)"

    # Memory status
    local mem_total=$(free -m | awk '/Mem:/ {print $2}')
    local mem_used=$(free -m | awk '/Mem:/ {print $3}')
    local mem_percent=$((mem_used * 100 / mem_total))

    if [[ "$mem_percent" -gt 95 ]]; then
        critical "Memory usage critical: ${mem_percent}% (${mem_used}MB / ${mem_total}MB)"
        ISSUES_FOUND["memory_critical"]=1
    elif [[ "$mem_percent" -gt 85 ]]; then
        warn "Memory usage high: ${mem_percent}%"
    else
        info "Memory usage: ${mem_percent}% (${mem_used}MB / ${mem_total}MB)"
    fi

    # Check for OOM conditions
    local oom_count=$(dmesg 2>/dev/null | grep -c "Out of memory" || echo "0")
    if [[ "$oom_count" -gt 0 ]]; then
        warn "OOM killer has been invoked $oom_count time(s) since boot"
        ISSUES_FOUND["oom_detected"]=1
    fi

    # Disk space on critical partitions
    info "Checking disk space..."
    while IFS= read -r line; do
        local mount=$(echo "$line" | awk '{print $6}')
        local usage=$(echo "$line" | awk '{print $5}' | tr -d '%')
        local avail=$(echo "$line" | awk '{print $4}')

        if [[ "$usage" -gt 95 ]]; then
            critical "CRITICAL: $mount is ${usage}% full (${avail} available)"
            ISSUES_FOUND["disk_full_${mount}"]=1
        elif [[ "$usage" -gt 85 ]]; then
            warn "$mount is ${usage}% full (${avail} available)"
        else
            success "$mount: ${usage}% used (${avail} available)"
        fi
    done < <(df -h / /var /tmp /var/lib/vz 2>/dev/null | tail -n +2 | sort -u)
}

diagnose_disks() {
    section "Disk Health Diagnostics"

    # Check for failed disks via SMART
    info "Checking SMART status for all disks..."

    if ! command -v smartctl &>/dev/null; then
        warn "smartctl not available - install smartmontools for disk health checks"
    else
        for disk in /dev/sd[a-z] /dev/nvme[0-9]n[0-9]; do
            [[ -b "$disk" ]] || continue

            local disk_name=$(basename "$disk")

            # Check if SMART is supported and enabled
            if ! smartctl -i "$disk" 2>/dev/null | grep -q "SMART support is: Enabled"; then
                info "$disk_name: SMART not enabled or not supported"
                continue
            fi

            local smart_health=$(smartctl -H "$disk" 2>/dev/null | grep -iE "SMART overall-health|SMART Health Status")

            if echo "$smart_health" | grep -qi "PASSED\|OK"; then
                success "$disk_name: SMART health OK"
            elif echo "$smart_health" | grep -qi "FAILED"; then
                critical "$disk_name: SMART health FAILED - disk may be failing!"
                ISSUES_FOUND["smart_failed_${disk_name}"]=1

                # Show which attributes are failing
                smartctl -A "$disk" 2>/dev/null | grep -i "pre-fail" | grep -v "^$" | head -5
            else
                # Check for concerning SMART attributes
                local reallocated=$(smartctl -A "$disk" 2>/dev/null | grep -i "Reallocated_Sector" | awk '{print $NF}')
                local pending=$(smartctl -A "$disk" 2>/dev/null | grep -i "Current_Pending_Sector" | awk '{print $NF}')
                local uncorrectable=$(smartctl -A "$disk" 2>/dev/null | grep -i "Offline_Uncorrectable" | awk '{print $NF}')

                local has_issues=false
                if [[ -n "$reallocated" && "$reallocated" -gt 0 ]]; then
                    warn "$disk_name: Has $reallocated reallocated sectors (potential disk wear)"
                    has_issues=true
                fi
                if [[ -n "$pending" && "$pending" -gt 0 ]]; then
                    warn "$disk_name: Has $pending pending sectors (may indicate problems)"
                    has_issues=true
                fi
                if [[ -n "$uncorrectable" && "$uncorrectable" -gt 0 ]]; then
                    error "$disk_name: Has $uncorrectable uncorrectable sectors!"
                    ISSUES_FOUND["disk_errors_${disk_name}"]=1
                    has_issues=true
                fi

                if [[ "$has_issues" == "false" ]]; then
                    success "$disk_name: SMART attributes OK"
                fi
            fi
        done
    fi

    # Check for disk I/O errors in dmesg
    info "Checking for disk I/O errors in kernel log..."
    local io_errors=$(dmesg 2>/dev/null | grep -iE "I/O error|medium error|sense error|failed command|sector error" | tail -10)
    if [[ -n "$io_errors" ]]; then
        warn "Disk I/O errors detected in kernel messages:"
        echo "$io_errors" | while read line; do
            warn "  $line"
        done
        ISSUES_FOUND["io_errors"]=1
    else
        success "No disk I/O errors in kernel messages"
    fi

    # Check for offline or missing block devices
    info "Checking for missing block devices..."
    if [[ -f /etc/fstab ]]; then
        while IFS= read -r line; do
            [[ "$line" =~ ^[[:space:]]*# ]] && continue
            [[ -z "$line" ]] && continue

            local device=$(echo "$line" | awk '{print $1}')
            local mountpoint=$(echo "$line" | awk '{print $2}')
            local fstype=$(echo "$line" | awk '{print $3}')

            # Skip special entries
            [[ "$fstype" =~ ^(swap|tmpfs|devpts|sysfs|proc|none)$ ]] && continue
            [[ "$mountpoint" == "none" ]] && continue
            [[ "$device" =~ ^# ]] && continue

            # Resolve UUID/LABEL to actual device
            local actual_device="$device"
            if [[ "$device" =~ ^UUID= ]]; then
                local uuid="${device#UUID=}"
                actual_device=$(blkid -U "$uuid" 2>/dev/null)
            elif [[ "$device" =~ ^LABEL= ]]; then
                local label="${device#LABEL=}"
                actual_device=$(blkid -L "$label" 2>/dev/null)
            fi

            if [[ -z "$actual_device" ]]; then
                critical "MISSING DEVICE: $device (mountpoint: $mountpoint)"
                ISSUES_FOUND["missing_device_${mountpoint}"]=1
            elif [[ ! -b "$actual_device" ]]; then
                critical "DEVICE NOT A BLOCK DEVICE: $actual_device (mountpoint: $mountpoint)"
                ISSUES_FOUND["invalid_device_${mountpoint}"]=1
            fi
        done < /etc/fstab
    fi

    # Check for missing/failed RAID members (mdadm)
    if command -v mdadm &>/dev/null; then
        shopt -s nullglob
        local md_devices=(/dev/md*)
        shopt -u nullglob

        if [[ ${#md_devices[@]} -gt 0 ]]; then
            info "Checking software RAID status..."
            for md in "${md_devices[@]}"; do
                [[ -b "$md" ]] || continue
                local md_name=$(basename "$md")
                local md_status=$(mdadm --detail "$md" 2>/dev/null | grep "State :" | awk -F: '{print $2}' | xargs)

                if [[ "$md_status" =~ clean|active ]]; then
                    success "$md_name: $md_status"
                elif [[ "$md_status" =~ degraded ]]; then
                    critical "$md_name: DEGRADED - missing disk member!"
                    ISSUES_FOUND["raid_degraded_${md_name}"]=1

                    # Show which members are missing
                    mdadm --detail "$md" 2>/dev/null | grep -E "removed|faulty|spare" | while read line; do
                        error "  $line"
                    done
                elif [[ "$md_status" =~ inactive ]]; then
                    critical "$md_name: INACTIVE"
                    ISSUES_FOUND["raid_inactive_${md_name}"]=1
                else
                    warn "$md_name: $md_status"
                fi
            done
        fi
    fi

    # Check for unmounted filesystems that should be mounted
    info "Checking for unmounted filesystems from /etc/fstab..."
    check_unmounted_filesystems
}

check_unmounted_filesystems() {
    # Parse fstab for filesystems that should be mounted but aren't

    local temp_file="/tmp/unmounted_filesystems.$$"
    rm -f "$temp_file"

    while IFS= read -r line; do
        # Skip comments and empty lines
        [[ "$line" =~ ^[[:space:]]*# ]] && continue
        [[ -z "$line" ]] && continue

        local device=$(echo "$line" | awk '{print $1}')
        local mountpoint=$(echo "$line" | awk '{print $2}')
        local fstype=$(echo "$line" | awk '{print $3}')
        local options=$(echo "$line" | awk '{print $4}')

        # Skip non-filesystem entries
        [[ "$fstype" =~ ^(swap|tmpfs|devpts|sysfs|proc|none)$ ]] && continue
        [[ "$mountpoint" == "none" ]] && continue

        # Skip noauto mounts
        [[ "$options" =~ noauto ]] && continue

        # Skip network mounts (handle separately)
        [[ "$fstype" =~ ^(nfs|nfs4|cifs|glusterfs|ceph)$ ]] && continue

        # Check if mounted
        if ! mountpoint -q "$mountpoint" 2>/dev/null; then
            warn "UNMOUNTED: $mountpoint ($device, $fstype)"
            ISSUES_FOUND["unmounted_${mountpoint}"]=1

            # Store for potential remount
            echo "$device|$mountpoint|$fstype|$options" >> "$temp_file"
        fi
    done < /etc/fstab

    if [[ ! -f "$temp_file" ]]; then
        success "All local filesystems are mounted"
    fi
}

diagnose_proxmox_services() {
    section "Proxmox Services Status"

    local services=(
        "pve-cluster:critical"
        "pvedaemon:critical"
        "pveproxy:critical"
        "pvestatd:important"
        "pvescheduler:important"
        "pve-firewall:optional"
        "corosync:cluster"
        "pve-ha-lrm:cluster"
        "pve-ha-crm:cluster"
        "spiceproxy:optional"
    )

    for service_info in "${services[@]}"; do
        local service="${service_info%%:*}"
        local importance="${service_info##*:}"

        if ! systemctl list-unit-files 2>/dev/null | grep -q "^${service}"; then
            continue
        fi

        local status=$(systemctl is-active "$service" 2>/dev/null)
        local enabled=$(systemctl is-enabled "$service" 2>/dev/null)

        case "$status" in
            active)
                success "$service: running (enabled: $enabled)"
                ;;
            inactive)
                if [[ "$importance" == "critical" ]]; then
                    critical "$service: NOT RUNNING"
                    ISSUES_FOUND["service_down_${service}"]=1
                else
                    warn "$service: inactive (enabled: $enabled)"
                fi
                ;;
            failed)
                critical "$service: FAILED"
                ISSUES_FOUND["service_failed_${service}"]=1
                # Get failure reason
                local fail_reason=$(systemctl status "$service" 2>/dev/null | grep -A2 "Active:" | tail -1)
                error "  Reason: $fail_reason"
                ;;
            *)
                warn "$service: $status (enabled: $enabled)"
                ;;
        esac
    done

    # Check web interface
    if curl -s -k --connect-timeout 5 "https://localhost:8006" &>/dev/null; then
        success "Web interface (port 8006) is responding"
    else
        warn "Web interface (port 8006) is NOT responding"
        ISSUES_FOUND["web_ui_down"]=1
    fi
}

diagnose_storage() {
    section "Storage Subsystem Diagnostics"

    if [[ "$SKIP_STORAGE" == "true" ]]; then
        info "Skipping storage diagnostics (--skip-storage flag)"
        return 0
    fi

    # Check for hung mount points (especially NFS/CIFS)
    info "Checking for hung network mounts..."
    local hung_count=0

    while IFS= read -r line; do
        local mountpoint=$(echo "$line" | awk '{print $3}')
        local fstype=$(echo "$line" | awk '{print $5}')

        if ! timeout 5 stat "$mountpoint" &>/dev/null; then
            critical "HUNG MOUNT: $mountpoint ($fstype)"
            ISSUES_FOUND["hung_mount_${mountpoint}"]=1
            ((hung_count++))
        fi
    done < <(mount | grep -E 'nfs|cifs|gluster|ceph')

    if [[ "$hung_count" -eq 0 ]]; then
        success "No hung network mounts detected"
    fi

    # Check ZFS if available
    if command -v zpool &>/dev/null && zpool list &>/dev/null 2>&1; then
        info "Checking ZFS pool status..."
        while IFS= read -r pool; do
            local health=$(zpool status "$pool" 2>/dev/null | grep "state:" | awk '{print $2}')
            case "$health" in
                ONLINE)
                    success "ZFS pool '$pool': ONLINE"
                    ;;
                DEGRADED)
                    critical "ZFS pool '$pool': DEGRADED"
                    ISSUES_FOUND["zfs_degraded_${pool}"]=1
                    zpool status "$pool" 2>/dev/null | grep -E "DEGRADED|FAULTED|OFFLINE|UNAVAIL" | head -5
                    ;;
                FAULTED)
                    critical "ZFS pool '$pool': FAULTED"
                    ISSUES_FOUND["zfs_faulted_${pool}"]=1
                    ;;
                *)
                    warn "ZFS pool '$pool': $health"
                    ;;
            esac
        done < <(zpool list -H -o name 2>/dev/null)
    fi

    # Check LVM
    info "Checking LVM status..."
    if timeout 10 vgs --noheadings 2>/dev/null; then
        vgs --noheadings 2>/dev/null | while read line; do
            local vg_name=$(echo "$line" | awk '{print $1}')
            local vg_status=$(echo "$line" | awk '{print $5}')
            if [[ "$vg_status" == "wz--n-" ]]; then
                success "VG '$vg_name': OK"
            else
                warn "VG '$vg_name': status=$vg_status"
            fi
        done
    else
        error "LVM commands timed out - storage subsystem may be hung"
        ISSUES_FOUND["lvm_hung"]=1
    fi

    # Proxmox storage status
    info "Checking Proxmox storage backends..."
    if timeout 15 pvesm status &>/dev/null; then
        pvesm status 2>/dev/null | tail -n +2 | while read line; do
            local storage=$(echo "$line" | awk '{print $1}')
            local type=$(echo "$line" | awk '{print $2}')
            local status=$(echo "$line" | awk '{print $3}')

            if [[ "$status" == "active" ]]; then
                success "Storage '$storage' ($type): active"
            else
                warn "Storage '$storage' ($type): $status"
                ISSUES_FOUND["storage_inactive_${storage}"]=1
            fi
        done
    else
        error "pvesm status timed out"
        ISSUES_FOUND["pvesm_hung"]=1
    fi
}

diagnose_cluster() {
    section "Cluster Diagnostics"

    if [[ "$SKIP_CLUSTER" == "true" ]]; then
        info "Skipping cluster diagnostics (--skip-cluster flag)"
        return 0
    fi

    # Check if this is a cluster node
    if [[ ! -f /etc/corosync/corosync.conf ]]; then
        info "This node is not part of a cluster (standalone mode)"
        return 0
    fi

    # Check /etc/pve accessibility (pmxcfs)
    info "Checking /etc/pve (pmxcfs) accessibility..."
    if mountpoint -q /etc/pve 2>/dev/null; then
        if timeout 5 ls /etc/pve &>/dev/null; then
            success "/etc/pve is mounted and accessible"
        else
            critical "/etc/pve is mounted but NOT accessible (pmxcfs hung)"
            ISSUES_FOUND["pmxcfs_hung"]=1
        fi
    else
        critical "/etc/pve is NOT mounted"
        ISSUES_FOUND["pmxcfs_unmounted"]=1
    fi

    # Check corosync
    info "Checking corosync status..."
    if service_running corosync; then
        if timeout 5 corosync-cfgtool -s &>/dev/null; then
            success "Corosync is running and responsive"

            # Check ring status
            local ring_status=$(corosync-cfgtool -s 2>/dev/null)
            if echo "$ring_status" | grep -q "status.*=.*0"; then
                success "Corosync ring status: OK"
            else
                warn "Corosync ring may have issues"
                echo "$ring_status" | grep -E "id|status"
            fi
        else
            warn "Corosync is running but not responding to queries"
            ISSUES_FOUND["corosync_unresponsive"]=1
        fi
    else
        critical "Corosync is NOT running"
        ISSUES_FOUND["corosync_down"]=1
    fi

    # Check quorum
    if timeout 5 corosync-quorumtool -s &>/dev/null; then
        local quorate=$(corosync-quorumtool -s 2>/dev/null | grep -i "quorate" | head -1)
        if echo "$quorate" | grep -qi "yes"; then
            success "Cluster is quorate"
        else
            critical "Cluster is NOT quorate"
            ISSUES_FOUND["cluster_not_quorate"]=1
        fi
    fi
}

diagnose_vms() {
    section "VM/Container Status"

    # Check for stuck operations
    info "Checking for stuck VM locks..."
    local locks_found=false

    shopt -s nullglob
    for lock in /run/lock/qemu-server/*.lock; do
        locks_found=true
        local vmid=$(basename "$lock" .lock)

        # Check if VM process is actually running
        if pgrep -f "kvm.*-id $vmid" &>/dev/null; then
            info "VM $vmid has lock (process running - may be legitimate)"
        else
            warn "VM $vmid has STALE lock (no process found)"
            ISSUES_FOUND["stale_lock_vm_${vmid}"]=1
        fi
    done

    for lock in /run/lock/lxc/*.lock; do
        locks_found=true
        local ctid=$(basename "$lock" .lock)
        warn "Container $ctid has lock file"
        ISSUES_FOUND["lock_ct_${ctid}"]=1
    done
    shopt -u nullglob

    if [[ "$locks_found" == "false" ]]; then
        success "No VM/container locks found"
    fi

    # Count running VMs/CTs
    if timeout 10 qm list &>/dev/null; then
        local running_vms=$(qm list 2>/dev/null | grep -c running || echo 0)
        info "Running VMs: $running_vms"
    else
        warn "Unable to list VMs (qm command timed out)"
    fi

    if timeout 10 pct list &>/dev/null; then
        local running_cts=$(pct list 2>/dev/null | grep -c running || echo 0)
        info "Running containers: $running_cts"
    else
        warn "Unable to list containers (pct command timed out)"
    fi
}

#=============================================================================
# PHASE 2: QUICK RECOVERY (Service Restarts)
#=============================================================================

quick_recovery_services() {
    section "Quick Recovery: Service Restarts"

    info "Attempting to fix issues by restarting Proxmox services..."
    info "This is the safest recovery method and should be tried first."
    echo ""

    local services_to_restart=()
    local needs_restart=false

    # Check which services need attention
    if [[ -n "${ISSUES_FOUND[service_down_pvedaemon]}" ]] || [[ -n "${ISSUES_FOUND[service_failed_pvedaemon]}" ]]; then
        services_to_restart+=("pvedaemon")
        needs_restart=true
    fi

    if [[ -n "${ISSUES_FOUND[service_down_pveproxy]}" ]] || [[ -n "${ISSUES_FOUND[service_failed_pveproxy]}" ]] || [[ -n "${ISSUES_FOUND[web_ui_down]}" ]]; then
        services_to_restart+=("pveproxy")
        needs_restart=true
    fi

    if [[ -n "${ISSUES_FOUND[service_down_pvestatd]}" ]] || [[ -n "${ISSUES_FOUND[service_failed_pvestatd]}" ]]; then
        services_to_restart+=("pvestatd")
        needs_restart=true
    fi

    if [[ -n "${ISSUES_FOUND[pmxcfs_hung]}" ]] || [[ -n "${ISSUES_FOUND[pmxcfs_unmounted]}" ]]; then
        services_to_restart+=("pve-cluster")
        needs_restart=true
    fi

    if [[ -n "${ISSUES_FOUND[corosync_down]}" ]] || [[ -n "${ISSUES_FOUND[corosync_unresponsive]}" ]]; then
        services_to_restart+=("corosync")
        needs_restart=true
    fi

    # If nothing specific found but storage issues exist, restart pvestatd
    if [[ -n "${ISSUES_FOUND[pvesm_hung]}" ]]; then
        services_to_restart+=("pvestatd")
        needs_restart=true
    fi

    if [[ "$needs_restart" == "false" ]]; then
        success "No services need restarting based on diagnostics"
        return 0
    fi

    # Remove duplicates
    services_to_restart=($(echo "${services_to_restart[@]}" | tr ' ' '\n' | sort -u | tr '\n' ' '))

    info "Services identified for restart: ${services_to_restart[*]}"

    if ! confirm "Restart these services?"; then
        info "Skipping service restarts"
        return 0
    fi

    # Restart in proper order
    local restart_order=("corosync" "pve-cluster" "pvedaemon" "pvestatd" "pveproxy" "pvescheduler")

    for service in "${restart_order[@]}"; do
        if [[ " ${services_to_restart[*]} " =~ " ${service} " ]]; then
            safe_restart_service "$service" 60
            sleep 2
        fi
    done

    # Verify fixes
    echo ""
    info "Verifying service recovery..."
    sleep 3

    local all_fixed=true

    if [[ " ${services_to_restart[*]} " =~ " pveproxy " ]]; then
        if curl -s -k --connect-timeout 5 "https://localhost:8006" &>/dev/null; then
            success "Web interface is now responding"
            unset ISSUES_FOUND[web_ui_down]
        else
            warn "Web interface still not responding"
            all_fixed=false
        fi
    fi

    if [[ " ${services_to_restart[*]} " =~ " pve-cluster " ]]; then
        if timeout 5 ls /etc/pve &>/dev/null; then
            success "/etc/pve is now accessible"
            unset ISSUES_FOUND[pmxcfs_hung]
            unset ISSUES_FOUND[pmxcfs_unmounted]
        else
            warn "/etc/pve still not accessible"
            all_fixed=false
        fi
    fi

    if [[ "$all_fixed" == "true" ]]; then
        success "Quick recovery successful!"
        return 0
    else
        warn "Some issues remain - may need deeper recovery"
        return 1
    fi
}

#=============================================================================
# PHASE 3: DISK RECOVERY
#=============================================================================

recover_unmounted_filesystems() {
    section "Filesystem Recovery"

    local unmounted_file="/tmp/unmounted_filesystems.$$"

    if [[ ! -f "$unmounted_file" ]]; then
        success "No unmounted filesystems to recover"
        return 0
    fi

    info "Found unmounted filesystems that should be mounted:"
    cat "$unmounted_file" | while IFS='|' read device mountpoint fstype options; do
        warn "  $mountpoint ($device, $fstype)"
    done
    echo ""

    echo -e "${YELLOW}Before attempting to remount, this script will verify:${NC}"
    echo -e "${YELLOW}  1. The underlying device exists${NC}"
    echo -e "${YELLOW}  2. The filesystem passes a read-only check${NC}"
    echo -e "${YELLOW}  3. No data will be modified without your explicit approval${NC}"
    echo ""

    if ! confirm "Attempt to remount unmounted filesystems?"; then
        info "Skipping filesystem remount"
        rm -f "$unmounted_file"
        return 0
    fi

    while IFS='|' read device mountpoint fstype options; do
        echo ""
        info "Processing: $mountpoint"

        # Safety check: Does the device exist?
        local actual_device="$device"

        # Handle UUID= and LABEL= references
        if [[ "$device" =~ ^UUID= ]]; then
            local uuid="${device#UUID=}"
            actual_device=$(blkid -U "$uuid" 2>/dev/null)
            if [[ -z "$actual_device" ]]; then
                error "Cannot find device for UUID=$uuid"
                error "The disk may be missing, failed, or disconnected!"
                error "DO NOT attempt to remount - investigate the hardware first"
                continue
            fi
            info "Resolved UUID to device: $actual_device"
        elif [[ "$device" =~ ^LABEL= ]]; then
            local label="${device#LABEL=}"
            actual_device=$(blkid -L "$label" 2>/dev/null)
            if [[ -z "$actual_device" ]]; then
                error "Cannot find device for LABEL=$label"
                error "The disk may be missing, failed, or disconnected!"
                continue
            fi
            info "Resolved LABEL to device: $actual_device"
        fi

        # Check if device exists and is a block device
        if [[ ! -b "$actual_device" ]]; then
            error "Device $actual_device does not exist or is not a block device"
            error "This may indicate a failed or disconnected disk!"
            error "Check: dmesg | grep -i error"
            continue
        fi

        success "Device $actual_device exists and is valid"

        # For local filesystems, run a quick fsck check (read-only, non-destructive)
        if [[ "$fstype" =~ ^(ext[234])$ ]]; then
            info "Running filesystem check (read-only, non-destructive)..."

            local fsck_output=$(e2fsck -n "$actual_device" 2>&1)
            local fsck_result=$?

            if [[ $fsck_result -eq 0 ]]; then
                success "Filesystem check passed - filesystem is clean"
            elif [[ $fsck_result -eq 4 ]]; then
                error "Filesystem has ERRORS that need repair!"
                echo "$fsck_output" | grep -i "error\|warning" | head -5
                echo ""
                warn "To repair, you would need to run: e2fsck -f $actual_device"
                warn "This should be done with the filesystem UNMOUNTED"

                if ! confirm_dangerous "Mount anyway WITHOUT repair? This could cause data corruption!"; then
                    error "Skipping $mountpoint - repair filesystem first"
                    continue
                fi
            else
                warn "Filesystem check returned status $fsck_result"
            fi
        elif [[ "$fstype" == "xfs" ]]; then
            info "XFS filesystem - checking mount viability..."
            # XFS doesn't have a read-only check like ext4, but we can try a dry-run repair
            if ! xfs_repair -n "$actual_device" &>/dev/null; then
                warn "XFS filesystem may have issues"
                warn "Consider running: xfs_repair $actual_device (requires unmount)"
            else
                success "XFS filesystem appears healthy"
            fi
        fi

        # Create mountpoint if it doesn't exist
        if [[ ! -d "$mountpoint" ]]; then
            info "Creating mountpoint directory: $mountpoint"
            execute mkdir -p "$mountpoint"
        fi

        # Final confirmation before mount
        if ! confirm "Mount $device on $mountpoint?"; then
            info "Skipping $mountpoint"
            continue
        fi

        # Attempt mount
        info "Attempting to mount..."

        if execute mount "$mountpoint" 2>&1; then
            success "Successfully mounted $mountpoint"

            # Verify it's accessible
            if timeout 5 ls "$mountpoint" &>/dev/null; then
                success "Filesystem is accessible"

                # Show some stats
                local usage=$(df -h "$mountpoint" 2>/dev/null | tail -1 | awk '{print $5}')
                info "Usage: $usage"
            else
                warn "Mounted but not fully accessible - may have issues"
            fi
        else
            error "Failed to mount $mountpoint"
            error "Check: dmesg | tail -20"
            error "Check: journalctl -xe | tail -20"
        fi

    done < "$unmounted_file"

    rm -f "$unmounted_file"
}

recover_hung_nfs() {
    section "NFS Mount Recovery"

    # Find hung NFS mounts
    local hung_nfs=()

    while IFS= read -r line; do
        local mountpoint=$(echo "$line" | awk '{print $3}')
        if ! timeout 5 stat "$mountpoint" &>/dev/null; then
            hung_nfs+=("$line")
        fi
    done < <(mount | grep -E 'nfs|nfs4')

    if [[ ${#hung_nfs[@]} -eq 0 ]]; then
        success "No hung NFS mounts detected"
        return 0
    fi

    warn "Found ${#hung_nfs[@]} hung NFS mount(s):"
    for mount_info in "${hung_nfs[@]}"; do
        local device=$(echo "$mount_info" | awk '{print $1}')
        local mountpoint=$(echo "$mount_info" | awk '{print $3}')
        warn "  $mountpoint ($device)"
    done
    echo ""

    echo -e "${YELLOW}NFS Recovery Options:${NC}"
    echo -e "${YELLOW}  l = Lazy unmount: Detaches mount immediately, cleanup happens when no longer in use${NC}"
    echo -e "${YELLOW}      SAFE - No data loss, but processes may get stale file handle errors${NC}"
    echo -e "${YELLOW}  f = Force unmount: Forcefully unmounts (may leave stale handles)${NC}"
    echo -e "${YELLOW}      USE WITH CAUTION - May cause issues for running processes${NC}"
    echo -e "${YELLOW}  s = Skip: Leave mount alone${NC}"
    echo ""

    for mount_info in "${hung_nfs[@]}"; do
        local device=$(echo "$mount_info" | awk '{print $1}')
        local mountpoint=$(echo "$mount_info" | awk '{print $3}')

        echo ""
        info "Processing hung mount: $mountpoint"
        info "  Server: $device"

        # Check what's using the mount
        if command -v lsof &>/dev/null; then
            local procs=$(timeout 5 lsof "$mountpoint" 2>/dev/null | tail -n +2 | wc -l || echo "unknown")
            if [[ "$procs" != "unknown" && "$procs" -gt 0 ]]; then
                warn "  $procs process(es) may be using this mount"
            fi
        fi

        read -p "Action for $mountpoint [l=lazy, f=force, s=skip]: " action

        case "$action" in
            l|L)
                info "Performing lazy unmount..."
                info "  (Mount will be detached immediately, cleanup when no longer in use)"
                if execute umount -l "$mountpoint"; then
                    success "Lazy unmount initiated for $mountpoint"
                else
                    error "Lazy unmount failed"
                fi
                ;;
            f|F)
                warn "Force unmount may cause errors for processes using this mount"
                warn "Processes may receive 'Stale file handle' errors"

                if confirm "Proceed with force unmount of $mountpoint?"; then
                    # Kill processes first if fuser available
                    if command -v fuser &>/dev/null; then
                        info "Terminating processes using mount..."
                        timeout 10 fuser -km "$mountpoint" 2>/dev/null
                        sleep 2
                    fi

                    if execute umount -f "$mountpoint"; then
                        success "Force unmount successful"
                    else
                        warn "Force unmount failed, trying lazy unmount as fallback..."
                        execute umount -l "$mountpoint"
                    fi
                fi
                ;;
            s|S|*)
                info "Skipping $mountpoint"
                ;;
        esac
    done

    # Offer to restart NFS client
    echo ""
    if confirm "Restart NFS client services (recommended after unmounting)?"; then
        execute systemctl restart nfs-common 2>/dev/null
        execute systemctl restart rpc-statd 2>/dev/null
        success "NFS client services restarted"
    fi
}

#=============================================================================
# PHASE 4: DEEP RECOVERY (More Invasive)
#=============================================================================

deep_recovery_cluster() {
    section "Deep Cluster Recovery"

    if [[ "$SKIP_CLUSTER" == "true" ]]; then
        info "Skipping cluster recovery (--skip-cluster flag)"
        return 0
    fi

    # Only proceed if we still have cluster issues
    if [[ -z "${ISSUES_FOUND[pmxcfs_hung]}" ]] && [[ -z "${ISSUES_FOUND[pmxcfs_unmounted]}" ]] && \
       [[ -z "${ISSUES_FOUND[corosync_down]}" ]] && [[ -z "${ISSUES_FOUND[corosync_unresponsive]}" ]]; then
        success "No cluster issues requiring deep recovery"
        return 0
    fi

    warn "Deep cluster recovery will restart cluster services"
    warn "This may temporarily affect:"
    warn "  - HA failover capabilities"
    warn "  - Cluster-wide configuration access"
    warn "  - VM migrations in progress"
    echo ""

    if ! confirm "Proceed with deep cluster recovery?"; then
        return 0
    fi

    # Stop cluster services in reverse dependency order
    info "Stopping cluster services..."
    execute systemctl stop pve-ha-lrm 2>/dev/null
    execute systemctl stop pve-ha-crm 2>/dev/null
    execute systemctl stop pvestatd 2>/dev/null
    execute systemctl stop pveproxy 2>/dev/null
    execute systemctl stop pvedaemon 2>/dev/null
    sleep 2

    execute systemctl stop pve-cluster 2>/dev/null
    sleep 2

    # Kill any hung pmxcfs processes
    if pgrep pmxcfs &>/dev/null; then
        warn "Killing hung pmxcfs processes..."
        execute pkill -9 pmxcfs
        sleep 2
    fi

    # Unmount /etc/pve if still mounted
    if mountpoint -q /etc/pve 2>/dev/null; then
        info "Unmounting /etc/pve..."
        execute umount -l /etc/pve 2>/dev/null
        sleep 1
    fi

    # Restart cluster services in order
    info "Restarting cluster services..."
    execute systemctl start pve-cluster
    sleep 5

    # Verify /etc/pve
    if timeout 10 ls /etc/pve &>/dev/null; then
        success "/etc/pve is now accessible"

        # Restart dependent services
        execute systemctl start pvedaemon
        sleep 2
        execute systemctl start pveproxy
        execute systemctl start pvestatd
        execute systemctl start pve-ha-lrm 2>/dev/null
        execute systemctl start pve-ha-crm 2>/dev/null

        success "Cluster services restarted"
    else
        error "/etc/pve still not accessible after recovery attempt"
        error "Manual intervention required:"
        error "  1. Check corosync status: corosync-cfgtool -s"
        error "  2. Check cluster logs: journalctl -u corosync -u pve-cluster"
        error "  3. Verify network connectivity to other cluster nodes"
    fi
}

recover_stale_locks() {
    section "Stale Lock Recovery"

    local has_stale_locks=false

    for key in "${!ISSUES_FOUND[@]}"; do
        if [[ "$key" =~ ^stale_lock_ ]]; then
            has_stale_locks=true
            break
        fi
    done

    if [[ "$has_stale_locks" == "false" ]]; then
        success "No stale locks to recover"
        return 0
    fi

    warn "Stale locks can prevent VM/container operations"
    warn "Only remove locks if you're CERTAIN the VM/CT is not running"
    echo ""

    shopt -s nullglob
    for lock in /run/lock/qemu-server/*.lock; do
        local vmid=$(basename "$lock" .lock)

        # Triple-check process isn't running
        if pgrep -f "kvm.*-id $vmid" &>/dev/null; then
            info "VM $vmid: Process IS running - keeping lock (this is normal)"
            continue
        fi

        # Check task log for recent activity
        local recent_task=$(grep -l "VMID: $vmid" /var/log/pve/tasks/active 2>/dev/null | head -1)
        if [[ -n "$recent_task" ]]; then
            warn "VM $vmid: Has active task - lock may be legitimate"
            if ! confirm "Remove lock anyway for VM $vmid?"; then
                continue
            fi
        else
            warn "VM $vmid: Stale lock found (no process, no active task)"
            if ! confirm "Remove stale lock for VM $vmid?"; then
                continue
            fi
        fi

        execute rm -f "$lock"
        success "Removed lock for VM $vmid"
    done

    for lock in /run/lock/lxc/*.lock; do
        local ctid=$(basename "$lock" .lock)

        # Check if container is running
        if pct status "$ctid" 2>/dev/null | grep -q "running"; then
            info "Container $ctid: Is running - keeping lock"
            continue
        fi

        warn "Container $ctid: Lock file found"
        if confirm "Remove lock for container $ctid?"; then
            execute rm -f "$lock"
            success "Removed lock for container $ctid"
        fi
    done
    shopt -u nullglob
}

#=============================================================================
# PHASE 5: FINAL VERIFICATION
#=============================================================================

final_verification() {
    section "Final System Verification"

    local all_ok=true

    # Check services
    info "Checking critical services..."
    for service in pve-cluster pvedaemon pveproxy pvestatd; do
        if service_running "$service"; then
            success "$service: running"
        else
            error "$service: NOT running"
            all_ok=false
        fi
    done

    # Check web UI
    if curl -s -k --connect-timeout 5 "https://localhost:8006" &>/dev/null; then
        success "Web interface: responding"
    else
        error "Web interface: NOT responding"
        all_ok=false
    fi

    # Check /etc/pve if cluster
    if [[ -f /etc/corosync/corosync.conf ]]; then
        if timeout 5 ls /etc/pve &>/dev/null; then
            success "/etc/pve: accessible"
        else
            error "/etc/pve: NOT accessible"
            all_ok=false
        fi
    fi

    # Check storage
    if timeout 10 pvesm status &>/dev/null; then
        local inactive=$(pvesm status 2>/dev/null | grep -v "^Name" | grep -cv "active" || echo 0)
        if [[ "$inactive" -gt 0 ]]; then
            warn "Storage: $inactive storage(s) inactive"
        else
            success "Storage: all active"
        fi
    fi

    # Check for any remaining critical issues
    local remaining_critical=0
    for key in "${!ISSUES_FOUND[@]}"; do
        if [[ -n "${ISSUES_FOUND[$key]}" ]]; then
            ((remaining_critical++))
        fi
    done

    echo ""
    if [[ "$all_ok" == "true" && "$remaining_critical" -eq 0 ]]; then
        echo -e "${GREEN}═══════════════════════════════════════════════════════════════${NC}"
        echo -e "${GREEN}  ✓ NODE RECOVERY COMPLETE - System appears healthy${NC}"
        echo -e "${GREEN}═══════════════════════════════════════════════════════════════${NC}"
    elif [[ "$all_ok" == "true" ]]; then
        echo -e "${YELLOW}═══════════════════════════════════════════════════════════════${NC}"
        echo -e "${YELLOW}  ⚠ PARTIAL RECOVERY - Core services OK, some issues remain${NC}"
        echo -e "${YELLOW}═══════════════════════════════════════════════════════════════${NC}"
    else
        echo -e "${RED}═══════════════════════════════════════════════════════════════${NC}"
        echo -e "${RED}  ✗ RECOVERY INCOMPLETE - Critical issues remain${NC}"
        echo -e "${RED}═══════════════════════════════════════════════════════════════${NC}"
    fi

    echo ""
    info "Log file saved to: $LOG_FILE"
    echo ""
    info "Useful commands for further investigation:"
    info "  dmesg | tail -50              # Kernel messages"
    info "  journalctl -xe                # System logs"
    info "  tail -100 /var/log/syslog     # Syslog"
    info "  pvecm status                  # Cluster status"
    info "  zpool status                  # ZFS status (if applicable)"
    info "  cat /proc/mdstat              # RAID status (if applicable)"
}

#=============================================================================
# MAIN EXECUTION
#=============================================================================

main() {
    check_root

    echo -e "${CYAN}"
    cat << 'EOF'
╔═══════════════════════════════════════════════════════════════════════════╗
║                     PROXMOX NODE RECOVERY SCRIPT                          ║
║                                                                           ║
║  Recovery Philosophy:                                                     ║
║    1. Diagnose first - understand the problem                             ║
║    2. Try simple fixes (service restarts) BEFORE invasive recovery        ║
║    3. Check disk health before attempting storage operations              ║
║    4. NEVER risk data loss without explicit confirmation                  ║
║                                                                           ║
╚═══════════════════════════════════════════════════════════════════════════╝
EOF
    echo -e "${NC}"

    info "Log file: $LOG_FILE"
    [[ "$FORCE_MODE" == "true" ]] && warn "Force mode: enabled (destructive actions still require confirmation)"
    [[ "$DRY_RUN" == "true" ]] && info "Dry-run mode: enabled (no changes will be made)"
    [[ "$SKIP_CLUSTER" == "true" ]] && info "Skip cluster: enabled"
    [[ "$SKIP_STORAGE" == "true" ]] && info "Skip storage: enabled"
    echo ""

    if ! confirm "Start diagnostics and recovery?"; then
        info "Aborted by user"
        exit 0
    fi

    # Clean up any previous temp files
    rm -f /tmp/unmounted_filesystems.$$ 2>/dev/null

    #=========================================================================
    # PHASE 1: DIAGNOSTICS
    #=========================================================================
    phase "PHASE 1: DIAGNOSTICS"

    diagnose_system
    diagnose_disks
    diagnose_proxmox_services
    diagnose_storage
    diagnose_cluster
    diagnose_vms

    # Summary of issues found
    section "Diagnostic Summary"

    if [[ ${#CRITICAL_ISSUES[@]} -gt 0 ]]; then
        error "CRITICAL ISSUES FOUND:"
        for issue in "${CRITICAL_ISSUES[@]}"; do
            error "  • $issue"
        done
        echo ""
    fi

    local issue_count=${#ISSUES_FOUND[@]}
    if [[ "$issue_count" -eq 0 ]]; then
        success "No issues detected - system appears healthy"
        if ! confirm "Run recovery steps anyway?"; then
            final_verification
            exit 0
        fi
    else
        warn "Found $issue_count potential issue(s)"
        echo ""
    fi

    #=========================================================================
    # PHASE 2: QUICK RECOVERY (Service Restarts - Try This First!)
    #=========================================================================
    phase "PHASE 2: QUICK RECOVERY (Service Restarts)"

    info "Trying simple service restarts first..."
    info "This often resolves issues without needing deeper recovery."
    echo ""

    quick_recovery_services

    # Re-check if critical issues are resolved
    local still_broken=false

    # Quick re-test of critical items
    if ! curl -s -k --connect-timeout 5 "https://localhost:8006" &>/dev/null; then
        still_broken=true
    fi

    if [[ -f /etc/corosync/corosync.conf ]] && ! timeout 5 ls /etc/pve &>/dev/null; then
        still_broken=true
    fi

    if [[ "$still_broken" == "false" ]]; then
        success "Quick recovery resolved the main issues!"

        if ! confirm "Continue to disk/storage recovery anyway?"; then
            final_verification
            exit 0
        fi
    else
        warn "Some issues remain after quick recovery"
    fi

    if ! confirm "Proceed to disk and storage recovery?"; then
        final_verification
        exit 0
    fi

    #=========================================================================
    # PHASE 3: DISK/STORAGE RECOVERY
    #=========================================================================
    phase "PHASE 3: DISK & STORAGE RECOVERY"

    recover_unmounted_filesystems
    recover_hung_nfs

    #=========================================================================
    # PHASE 4: DEEP RECOVERY
    #=========================================================================
    phase "PHASE 4: DEEP RECOVERY"

    if ! confirm "Proceed to deep recovery steps?"; then
        final_verification
        exit 0
    fi

    deep_recovery_cluster
    recover_stale_locks

    #=========================================================================
    # PHASE 5: FINAL VERIFICATION
    #=========================================================================
    phase "PHASE 5: FINAL VERIFICATION"

    final_verification
}

main "$@"