#!/bin/bash
#
# LVM Recovery Script for Proxmox
# Fixes hung LVM that's blocking node operations
#
# Common causes of LVM hangs:
#   - Hung NFS/CIFS mount that LVM is scanning
#   - Missing/failed disk in a volume group
#   - Stuck device-mapper entries
#   - Blocked I/O on a disk
#
# Author: DartNode Operations
#

set -o pipefail

RED='\033[0;31m'
GREEN='\033[0;32m'
YELLOW='\033[1;33m'
BLUE='\033[0;34m'
CYAN='\033[0;36m'
NC='\033[0m'

LOG_FILE="/var/log/lvm-recovery-$(date +%Y%m%d-%H%M%S).log"

log() { echo -e "$(date '+%H:%M:%S') $@" | tee -a "$LOG_FILE"; }
info() { log "${BLUE}[INFO]${NC} $@"; }
success() { log "${GREEN}[OK]${NC} $@"; }
warn() { log "${YELLOW}[WARN]${NC} $@"; }
error() { log "${RED}[ERROR]${NC} $@"; }

section() {
    echo "" | tee -a "$LOG_FILE"
    echo -e "${CYAN}═══════════════════════════════════════════════════════════════${NC}" | tee -a "$LOG_FILE"
    echo -e "${CYAN}  $@${NC}" | tee -a "$LOG_FILE"
    echo -e "${CYAN}═══════════════════════════════════════════════════════════════${NC}" | tee -a "$LOG_FILE"
    echo "" | tee -a "$LOG_FILE"
}

# Check root
if [[ $EUID -ne 0 ]]; then
    echo "This script must be run as root"
    exit 1
fi

echo -e "${CYAN}"
cat << 'EOF'
╔═══════════════════════════════════════════════════════════════════════════╗
║                       LVM RECOVERY SCRIPT                                 ║
║                                                                           ║
║  This script diagnoses and fixes hung LVM issues                          ║
║  Common on Proxmox when disks fail or NFS mounts hang                     ║
╚═══════════════════════════════════════════════════════════════════════════╝
EOF
echo -e "${NC}"

info "Log file: $LOG_FILE"
echo ""

#=============================================================================
# STEP 1: Quick diagnosis - what's blocking?
#=============================================================================
section "STEP 1: Diagnosing LVM Hang"

# Check for processes stuck in D state (uninterruptible sleep - usually I/O wait)
info "Checking for processes stuck in D state (I/O wait)..."
d_state_procs=$(ps aux | awk '$8 ~ /D/ {print $0}')

if [[ -n "$d_state_procs" ]]; then
    warn "Found processes in D state (blocked on I/O):"
    echo "$d_state_procs" | while read line; do
        warn "  $line"
    done
    echo ""

    # Check what they're waiting on
    info "Checking what's blocking these processes..."
    ps aux | awk '$8 ~ /D/ {print $2}' | while read pid; do
        if [[ -f /proc/$pid/stack ]]; then
            local cmd=$(cat /proc/$pid/comm 2>/dev/null)
            echo -e "${YELLOW}PID $pid ($cmd) stack:${NC}"
            cat /proc/$pid/stack 2>/dev/null | head -10
            echo ""
        fi
    done
else
    success "No processes stuck in D state"
fi

# Check for hung NFS mounts - this is the #1 cause
info "Checking for hung NFS mounts..."
hung_nfs=()

while IFS= read -r line; do
    mountpoint=$(echo "$line" | awk '{print $3}')
    if ! timeout 3 stat "$mountpoint" &>/dev/null; then
        hung_nfs+=("$mountpoint")
        error "HUNG NFS: $mountpoint"
    fi
done < <(mount | grep -E 'nfs|nfs4' 2>/dev/null)

if [[ ${#hung_nfs[@]} -eq 0 ]]; then
    success "No hung NFS mounts detected"
else
    error "Found ${#hung_nfs[@]} hung NFS mount(s) - THIS IS LIKELY YOUR PROBLEM"
fi

# Check for missing/failed block devices
info "Checking for missing block devices referenced by LVM..."

# Get list of PVs without hanging
if timeout 5 pvs --noheadings -o pv_name 2>/dev/null; then
    pvs --noheadings -o pv_name 2>/dev/null | while read pv; do
        pv=$(echo "$pv" | xargs)  # trim whitespace
        if [[ ! -b "$pv" ]]; then
            error "MISSING PV: $pv does not exist!"
        fi
    done
else
    warn "pvs command is hanging - LVM is definitely stuck"
fi

# Check dmesg for disk errors
info "Recent disk errors from kernel:"
dmesg_errors=$(dmesg 2>/dev/null | grep -iE "I/O error|medium error|offline|failed|hung_task" | tail -10)
if [[ -n "$dmesg_errors" ]]; then
    echo "$dmesg_errors" | while read line; do
        warn "  $line"
    done
else
    success "No recent disk errors in dmesg"
fi

# Check for blocked device-mapper operations
info "Checking device-mapper status..."
if timeout 5 dmsetup status &>/dev/null; then
    success "dmsetup is responsive"
else
    error "dmsetup is HANGING - device-mapper is stuck"
fi

#=============================================================================
# STEP 2: Fix hung NFS first (if present)
#=============================================================================
if [[ ${#hung_nfs[@]} -gt 0 ]]; then
    section "STEP 2: Fixing Hung NFS Mounts"

    warn "Hung NFS mounts cause LVM to hang because LVM scans all mountpoints"
    warn "We need to unmount these first before LVM will respond"
    echo ""

    for mount in "${hung_nfs[@]}"; do
        info "Unmounting hung NFS: $mount"

        # Lazy unmount - detaches immediately
        if umount -l "$mount" 2>&1; then
            success "Lazy unmount sent for $mount"
        else
            warn "Lazy unmount failed, trying force..."
            umount -f "$mount" 2>&1
        fi
    done

    sleep 2

    # Kill any NFS-related processes that might be stuck
    info "Killing any stuck NFS processes..."
    pkill -9 -f "nfs" 2>/dev/null

    # Restart NFS client
    info "Restarting NFS client services..."
    systemctl restart rpc-statd 2>/dev/null
    systemctl restart nfs-common 2>/dev/null

    success "NFS cleanup complete"
    echo ""
fi

#=============================================================================
# STEP 3: LVM Configuration Fixes
#=============================================================================
section "STEP 3: LVM Configuration for Recovery"

# Backup current lvm.conf
if [[ -f /etc/lvm/lvm.conf ]]; then
    cp /etc/lvm/lvm.conf /etc/lvm/lvm.conf.backup.$(date +%Y%m%d-%H%M%S)
    success "Backed up /etc/lvm/lvm.conf"
fi

# Check current LVM filter settings
info "Current LVM device filter:"
grep -E "^\s*filter\s*=" /etc/lvm/lvm.conf 2>/dev/null || echo "  (no filter set - scanning all devices)"

# Check if md_component_detection is causing issues
info "Checking LVM md_component_detection setting..."
md_detect=$(grep -E "^\s*md_component_detection" /etc/lvm/lvm.conf 2>/dev/null)
if [[ -n "$md_detect" ]]; then
    info "  $md_detect"
fi

# Suggest filter if we found hung NFS
if [[ ${#hung_nfs[@]} -gt 0 ]]; then
    echo ""
    warn "RECOMMENDATION: Add LVM filter to exclude NFS paths"
    warn "Edit /etc/lvm/lvm.conf and set:"
    warn '  filter = [ "a|/dev/sd.*|", "a|/dev/nvme.*|", "a|/dev/dm-.*|", "r|.*|" ]'
    echo ""
fi

#=============================================================================
# STEP 4: Clear stuck LVM/DM state
#=============================================================================
section "STEP 4: Clearing Stuck LVM State"

# Stop services that use LVM
info "Stopping services that depend on LVM..."
systemctl stop pvestatd 2>/dev/null
systemctl stop lvm2-monitor 2>/dev/null

# Try to drop LVM cache
info "Dropping LVM cache..."
if timeout 10 vgchange --refresh 2>/dev/null; then
    success "VG refresh completed"
else
    warn "VG refresh timed out"
fi

# Try pvscan with specific options
info "Attempting pvscan with cache update..."
if timeout 30 pvscan --cache 2>&1; then
    success "pvscan cache updated"
else
    warn "pvscan still hanging - trying more aggressive fix"

    # Kill any stuck LVM processes
    info "Killing stuck LVM processes..."
    pkill -9 -f "pvscan" 2>/dev/null
    pkill -9 -f "vgscan" 2>/dev/null
    pkill -9 -f "lvscan" 2>/dev/null
    pkill -9 -f "lvs" 2>/dev/null
    pkill -9 -f "vgs" 2>/dev/null
    pkill -9 -f "pvs" 2>/dev/null

    sleep 2
fi

# Check if lvmetad is running and possibly causing issues
if pgrep lvmetad &>/dev/null; then
    info "lvmetad is running - restarting it..."
    systemctl restart lvm2-lvmetad 2>/dev/null || killall -9 lvmetad 2>/dev/null
    sleep 2
fi

#=============================================================================
# STEP 5: Identify problematic devices
#=============================================================================
section "STEP 5: Identifying Problematic Devices"

info "Scanning block devices (this may take a moment)..."

# List all block devices
echo ""
info "Block devices present:"
lsblk -d -o NAME,SIZE,TYPE,MOUNTPOINT,STATE 2>/dev/null | while read line; do
    info "  $line"
done

echo ""
info "Checking each block device for I/O responsiveness..."

for dev in /dev/sd[a-z] /dev/nvme[0-9]n[0-9]; do
    [[ -b "$dev" ]] || continue

    devname=$(basename "$dev")

    # Quick I/O test - try to read first sector
    if timeout 5 dd if="$dev" of=/dev/null bs=512 count=1 2>/dev/null; then
        success "$devname: responsive"
    else
        error "$devname: NOT RESPONDING - this disk may be failed/hung!"

        # Check if this device is part of any VG
        vg_info=$(pvs --noheadings -o vg_name "$dev" 2>/dev/null | xargs)
        if [[ -n "$vg_info" ]]; then
            error "  └─ Part of VG: $vg_info"
            error "  └─ This is likely causing your LVM hang!"
        fi
    fi
done

#=============================================================================
# STEP 6: Attempt LVM recovery
#=============================================================================
section "STEP 6: Attempting LVM Recovery"

# Now try LVM commands with timeout
info "Testing LVM responsiveness..."

echo -n "  pvs: "
if timeout 15 pvs --noheadings 2>/dev/null; then
    echo -e "${GREEN}OK${NC}"
    PVS_WORKS=true
else
    echo -e "${RED}TIMEOUT${NC}"
    PVS_WORKS=false
fi

echo -n "  vgs: "
if timeout 15 vgs --noheadings 2>/dev/null; then
    echo -e "${GREEN}OK${NC}"
    VGS_WORKS=true
else
    echo -e "${RED}TIMEOUT${NC}"
    VGS_WORKS=false
fi

echo -n "  lvs: "
if timeout 15 lvs --noheadings 2>/dev/null; then
    echo -e "${GREEN}OK${NC}"
    LVS_WORKS=true
else
    echo -e "${RED}TIMEOUT${NC}"
    LVS_WORKS=false
fi

echo ""

if [[ "$PVS_WORKS" == "true" && "$VGS_WORKS" == "true" && "$LVS_WORKS" == "true" ]]; then
    success "LVM is now responding!"

    # Show current state
    echo ""
    info "Current LVM state:"
    echo ""
    echo "Physical Volumes:"
    pvs 2>/dev/null
    echo ""
    echo "Volume Groups:"
    vgs 2>/dev/null
    echo ""
    echo "Logical Volumes:"
    lvs 2>/dev/null

else
    error "LVM is still hanging"
    echo ""
    warn "Additional steps to try:"
    echo ""
    echo "1. If a disk is failed/unresponsive, you may need to:"
    echo "   - Physically check/replace the disk"
    echo "   - Or remove it from the VG: vgreduce --removemissing <vgname>"
    echo ""
    echo "2. If device-mapper is stuck, try:"
    echo "   dmsetup remove_all --force"
    echo "   (WARNING: This will deactivate all LVs!)"
    echo ""
    echo "3. Check for hung kernel threads:"
    echo "   echo w > /proc/sysrq-trigger  # Show blocked tasks"
    echo "   dmesg | tail -50"
    echo ""
    echo "4. As last resort, you may need to reboot the node"
    echo "   (Schedule maintenance window first)"
fi

#=============================================================================
# STEP 7: Restart Proxmox services
#=============================================================================
section "STEP 7: Restarting Proxmox Services"

if [[ "$PVS_WORKS" == "true" ]]; then
    info "Restarting Proxmox services..."

    systemctl restart lvm2-monitor 2>/dev/null
    sleep 1

    systemctl restart pvestatd
    sleep 2

    systemctl restart pvedaemon
    sleep 2

    systemctl restart pveproxy
    sleep 2

    # Check web UI
    if curl -s -k --connect-timeout 5 "https://localhost:8006" &>/dev/null; then
        success "Web interface is responding"
    else
        warn "Web interface not responding yet - may need more time"
    fi

    # Check pvesm
    info "Checking storage status..."
    if timeout 15 pvesm status &>/dev/null; then
        pvesm status 2>/dev/null
        success "Storage subsystem is responding"
    else
        warn "Storage status still timing out"
    fi
else
    warn "Skipping Proxmox service restart - LVM still not working"
fi

#=============================================================================
# Summary
#=============================================================================
section "RECOVERY SUMMARY"

echo "Log file: $LOG_FILE"
echo ""

if [[ "$PVS_WORKS" == "true" && "$VGS_WORKS" == "true" && "$LVS_WORKS" == "true" ]]; then
    echo -e "${GREEN}════════════════════════════════════════════════════════════════${NC}"
    echo -e "${GREEN}  ✓ LVM RECOVERY SUCCESSFUL${NC}"
    echo -e "${GREEN}════════════════════════════════════════════════════════════════${NC}"
    echo ""
    echo "LVM is now responding. If you still have issues:"
    echo "  - Check: pvesm status"
    echo "  - Check: qm list / pct list"
    echo "  - Access web UI: https://$(hostname):8006"
else
    echo -e "${RED}════════════════════════════════════════════════════════════════${NC}"
    echo -e "${RED}  ✗ LVM STILL HUNG - MANUAL INTERVENTION REQUIRED${NC}"
    echo -e "${RED}════════════════════════════════════════════════════════════════${NC}"
    echo ""
    echo "Likely causes still present:"

    if [[ ${#hung_nfs[@]} -gt 0 ]]; then
        echo "  • Hung NFS mounts (may need hard reboot of NFS server)"
    fi

    echo "  • Failed/unresponsive disk"
    echo "  • Stuck device-mapper state"
    echo ""
    echo "Nuclear options (use with caution):"
    echo ""
    echo "  # Show what's blocking in kernel:"
    echo "  echo w > /proc/sysrq-trigger && dmesg | tail -30"
    echo ""
    echo "  # Force remove all device-mapper entries (DANGEROUS):"
    echo "  dmsetup remove_all --force"
    echo ""
    echo "  # Remove missing PVs from a VG:"
    echo "  vgreduce --removemissing --force <vgname>"
    echo ""
    echo "  # If all else fails, reboot may be required"
fi