#!/bin/bash

# DartNode Daemon Installer & Manager
# One-liner: curl -sSL https://pkg.dev.snaju.com/dn/dartnode.sh | bash -s install
# With token: curl -sSL https://pkg.dev.snaju.com/dn/dartnode.sh | bash -s install --token <token>

PKG_BASE="https://pkg.dev.snaju.com/dn"
API_BASE="https://api.dartnode.net"

# Registration token (can be set via --token flag or DARTNODE_TOKEN env var)
REGISTRATION_TOKEN="${DARTNODE_TOKEN:-}"

# Colors for interactive UI
RED='\033[0;31m'
GREEN='\033[0;32m'
YELLOW='\033[1;33m'
BLUE='\033[0;34m'
CYAN='\033[0;36m'
BOLD='\033[1m'
DIM='\033[2m'
NC='\033[0m' # No Color

# Detect OS
OS_TYPE="linux"
case "$(uname -s)" in
  Darwin) OS_TYPE="macos" ;;
  Linux)  OS_TYPE="linux" ;;
esac

# Set paths based on OS
if [ "$OS_TYPE" = "macos" ]; then
  DN_DAEMON_BIN="/opt/dn-daemon/dn-daemon"
  DN_DAEMON_CONFIG_DIR="/etc/dn-daemon"
  DN_DAEMON_SERVICE="com.dartnode.dn-daemon"
  DN_LAUNCHD_PLIST="/Library/LaunchDaemons/com.dartnode.dn-daemon.plist"
else
  DN_DAEMON_BIN="/usr/local/bin/dn-daemon"
  DN_DAEMON_CONFIG_DIR="/etc/dn-daemon"
  DN_DAEMON_SERVICE="dn-daemon"
fi

# Legacy paths (for migration)
LEGACY_BIN="/usr/local/bin/pve-stats-daemon"
LEGACY_CONFIG_DIR="/etc/pve-stats"
LEGACY_SERVICE="pve-stats"

# Interactive mode detection
INTERACTIVE=true
if [ ! -t 0 ]; then
  INTERACTIVE=false
fi

function header() {
  echo "    ___           _       __          _      "
  echo "   /   \\__ _ _ __| |_  /\\ \\ \\___   __| | ___ "
  echo "  / /\\ / _\` | '__| __|/  \\/ / _ \\ / _\` |/ _ \\"
  echo " / /_// (_| | |  | |_/ /\\  / (_) | (_| |  __/"
  echo "/___,' \\__,_|_|   \\__\\_\\ \\/ \\___/ \\__,_|\\___|"
  echo "                                             "
  echo "DartNode Daemon"
  echo "Copyright (c) 2024-2026, Snaju Inc. All Rights Reserved."
  echo ""
}

function help() {
  echo -e "Usage: ${BOLD}dartnode <command> [options]${NC}"
  echo ""
  echo -e "${BOLD}Commands:${NC}"
  if [ "$OS_TYPE" = "macos" ]; then
    echo "  install    - Install DN Daemon on this macOS host"
  else
    echo "  install    - Install DN Daemon on this Proxmox node"
  fi
  echo "  setup      - Run interactive setup wizard"
  echo "  config     - Configure daemon settings (interactive menu)"
  echo "  start      - Start the daemon"
  echo "  stop       - Stop the daemon"
  echo "  restart    - Restart the daemon"
  echo "  status     - Show daemon status"
  echo "  log        - Follow daemon logs"
  echo "  update     - Manually update daemon binary to latest version"
  echo "  version    - Show current daemon version"
  echo "  uninstall  - Remove daemon completely"
  if [ "$OS_TYPE" = "linux" ]; then
    echo "  migrate    - Migrate from legacy pve-stats to dn-daemon"
  fi
  echo "  help       - Show this help"
  echo ""
  echo -e "${BOLD}Diagnostics & Tools:${NC}"
  echo "  net                                - Network tools (interactive menu)"
  echo "  disk                               - Drive tools & LED identification"
  echo "  vms                                - List all VMs on this node"
  echo "  diag                               - Full node diagnostics"
  echo "  redis                              - Redis connection diagnostics"
  echo "  storage                            - Storage overview (LVM/RAID/disk)"
  echo "  fw                                 - Firewall rules overview"
  if [ "$OS_TYPE" = "linux" ]; then
    echo "  qmp <vmid>                         - Test QMP socket (PVE only)"
  fi
  echo ""
  echo -e "${BOLD}Repair & Jobs:${NC}"
  echo "  repair                                        - Diagnose & fix common node issues"
  echo "  jobs [list|clear|flush|inspect|retry|pve|locks] - Manage job queue & PVE tasks"
  echo ""
  if [ "$OS_TYPE" = "linux" ]; then
    echo -e "${BOLD}SMTP Gate:${NC}"
    echo "  smtp-gate status                              - Quick status check (delegates to smtp-gate-ctl)"
    echo "  Install: curl -sSL ${PKG_BASE}/smtp-gate.sh | bash -s install"
    echo "  Manage:  smtp-gate-ctl start|stop|restart|status|log|update|config"
  fi
  echo ""
  echo -e "${BOLD}Options:${NC}"
  echo "  --token <token>  - Registration token from DartNode admin panel"
  echo ""
  echo -e "Detected OS: ${CYAN}${OS_TYPE}${NC}"
  echo ""
  echo -e "${BOLD}Daemon Modes:${NC}"
  echo "  pve        - Proxmox VE (Linux KVM virtualization)"
  echo "  macos      - Apple Silicon (Tart-based macOS VMs)"
  echo ""
  echo -e "${BOLD}Auto-Update:${NC}"
  echo "  The daemon automatically checks for updates every 5 minutes."
  echo "  Disable with: \"auto_update\": false in config.json"
  echo ""
  echo -e "${BOLD}Quick Install (Interactive):${NC}"
  echo "  curl -sSL ${PKG_BASE}/dartnode.sh | bash -s install"
  echo ""
  echo -e "${BOLD}Quick Install (With Token):${NC}"
  echo "  curl -sSL ${PKG_BASE}/dartnode.sh | bash -s install --token YOUR_TOKEN"
  echo ""
}

function check_root() {
  if [ "$EUID" -ne 0 ]; then
    echo "Error: Please run as root"
    exit 1
  fi
}

function check_proxmox() {
  if [ ! -d "/var/run/qemu-server" ] && [ ! -d "/etc/pve" ]; then
    echo "Warning: This doesn't appear to be a Proxmox node."
    read -p "Continue anyway? (y/n) " -n 1 -r
    echo
    if [[ ! $REPLY =~ ^[Yy]$ ]]; then
      exit 1
    fi
  fi
}

function check_macos() {
  # Check macOS version (requires 13+)
  MACOS_VERSION=$(sw_vers -productVersion 2>/dev/null)
  MAJOR_VERSION=$(echo "$MACOS_VERSION" | cut -d. -f1)
  if [ -z "$MAJOR_VERSION" ] || [ "$MAJOR_VERSION" -lt 13 ]; then
    echo "Error: macOS 13 (Ventura) or later required."
    echo "Current version: ${MACOS_VERSION:-unknown}"
    exit 1
  fi

  # Check for Apple Silicon
  ARCH=$(uname -m)
  if [ "$ARCH" != "arm64" ]; then
    echo "Error: Apple Silicon (arm64) required."
    echo "Current architecture: $ARCH"
    exit 1
  fi

  echo "macOS ${MACOS_VERSION} on ${ARCH} detected."
}

function check_homebrew() {
  if ! command -v brew &> /dev/null; then
    echo "Homebrew not found. Installing..."
    /bin/bash -c "$(curl -fsSL https://raw.githubusercontent.com/Homebrew/install/HEAD/install.sh)"

    # Add to PATH for this session
    if [ -f "/opt/homebrew/bin/brew" ]; then
      eval "$(/opt/homebrew/bin/brew shellenv)"
    fi
  fi
}

function check_tart() {
  if ! command -v tart &> /dev/null; then
    echo "Tart not found. Installing via Homebrew..."
    brew install cirruslabs/cli/tart
  fi

  # Verify installation
  if ! tart --version &> /dev/null; then
    echo "Error: Tart installation failed."
    exit 1
  fi

  echo "Tart $(tart --version 2>&1 | head -1) installed."
}

function detect_legacy() {
  # Check if legacy installation exists
  if [ -f "${LEGACY_BIN}" ] || [ -d "${LEGACY_CONFIG_DIR}" ] || systemctl is-active --quiet ${LEGACY_SERVICE} 2>/dev/null; then
    return 0
  fi
  return 1
}

# ============================================
# Interactive Setup Functions
# ============================================

function print_step() {
  echo -e "${CYAN}==>${NC} ${BOLD}$1${NC}"
}

function print_success() {
  echo -e "${GREEN}[OK]${NC} $1"
}

function print_error() {
  echo -e "${RED}[ERROR]${NC} $1"
}

function print_warning() {
  echo -e "${YELLOW}[WARN]${NC} $1"
}

function print_info() {
  echo -e "${DIM}$1${NC}"
}

function select_option() {
  # Usage: select_option "prompt" option1 option2 option3 ...
  # Returns the selected option index (1-based)
  local prompt="$1"
  shift
  local options=("$@")
  local selected=1
  local total=${#options[@]}

  # Hide cursor
  tput civis 2>/dev/null

  while true; do
    # Clear previous output
    for ((i=0; i<total+2; i++)); do
      tput cuu1 2>/dev/null
      tput el 2>/dev/null
    done 2>/dev/null

    echo -e "${BOLD}${prompt}${NC}"
    echo ""

    for i in "${!options[@]}"; do
      local idx=$((i+1))
      if [ $idx -eq $selected ]; then
        echo -e "  ${GREEN}>${NC} ${BOLD}${options[$i]}${NC}"
      else
        echo -e "    ${DIM}${options[$i]}${NC}"
      fi
    done

    # Read single keypress
    read -rsn1 key
    case "$key" in
      A|k) # Up arrow or k
        ((selected--))
        [ $selected -lt 1 ] && selected=$total
        ;;
      B|j) # Down arrow or j
        ((selected++))
        [ $selected -gt $total ] && selected=1
        ;;
      "") # Enter
        break
        ;;
      [1-9]) # Number keys
        if [ "$key" -le "$total" ]; then
          selected=$key
          break
        fi
        ;;
    esac
  done

  # Show cursor
  tput cnorm 2>/dev/null

  echo "$selected"
}

function generate_secret() {
  # Generate a random 32-character hex string
  if command -v openssl &> /dev/null; then
    openssl rand -hex 16
  elif [ -f /dev/urandom ]; then
    head -c 16 /dev/urandom | xxd -p 2>/dev/null || cat /dev/urandom | tr -dc 'a-f0-9' | head -c 32
  else
    date +%s%N | sha256sum | head -c 32
  fi
}

# ============================================
# Token-Based Registration Functions
# ============================================

# JSON parsing helper - uses python3 for reliable parsing
function json_get() {
  local json="$1"
  local key="$2"

  if command -v python3 &> /dev/null; then
    echo "$json" | python3 -c "import sys,json; d=json.load(sys.stdin); print(d.get('data',d).get('$key',''))" 2>/dev/null
  else
    # Fallback to grep for simple cases
    echo "$json" | grep -o "\"$key\":\"[^\"]*\"" | cut -d'"' -f4
  fi
}

function json_get_nested() {
  local json="$1"
  local parent="$2"
  local key="$3"

  if command -v python3 &> /dev/null; then
    echo "$json" | python3 -c "import sys,json; d=json.load(sys.stdin); print(d.get('data',d).get('$parent',{}).get('$key',''))" 2>/dev/null
  else
    # Fallback to grep - less reliable for nested
    echo "$json" | grep -o "\"$key\":[^,}]*" | head -1 | sed 's/.*://' | tr -d '"'
  fi
}

function json_get_bool() {
  local json="$1"
  local key="$2"

  if command -v python3 &> /dev/null; then
    echo "$json" | python3 -c "import sys,json; d=json.load(sys.stdin); print('true' if d.get('data',d).get('$key') else 'false')" 2>/dev/null
  else
    echo "$json" | grep -q "\"$key\":true" && echo "true" || echo "false"
  fi
}

function validate_token() {
  local token="$1"

  print_step "Validating registration token..."

  local response
  response=$(curl -sS "${API_BASE}/node/status?token=${token}" 2>/dev/null)

  if [ -z "$response" ]; then
    print_error "Unable to connect to DartNode API"
    return 1
  fi

  # Check if response contains success
  if echo "$response" | grep -q '"success":true'; then
    local is_valid is_used
    is_valid=$(json_get_bool "$response" "isValid")
    is_used=$(json_get_bool "$response" "isUsed")

    if [ "$is_valid" = "true" ]; then
      print_success "Token is valid"
      return 0
    elif [ "$is_used" = "true" ]; then
      print_warning "Token has already been used"
      return 2
    else
      print_error "Token is expired"
      return 1
    fi
  else
    local error_msg
    error_msg=$(json_get "$response" "error")
    print_error "Invalid token: ${error_msg:-Unknown error}"
    return 1
  fi
}

function fetch_token_config() {
  local token="$1"

  print_step "Fetching configuration from DartNode..."

  local response
  response=$(curl -sS "${API_BASE}/node/config?token=${token}" 2>/dev/null)

  if [ -z "$response" ]; then
    print_error "Unable to fetch configuration"
    return 1
  fi

  if echo "$response" | grep -q '"success":true'; then
    echo "$response"
    return 0
  else
    local error_msg
    error_msg=$(json_get "$response" "error")
    print_error "Failed to fetch configuration: ${error_msg:-Unknown error}"
    return 1
  fi
}

function register_node_with_token() {
  local token="$1"
  local node_ip="$2"
  local api_secret="$3"
  local storage_name="${4:-VM_DATA}"
  local storage_type="${5:-raw}"
  local mode="${6:-pve}"

  print_step "Registering node with DartNode..."

  # Build base JSON payload
  local payload="{\"token\":\"${token}\",\"nodeIp\":\"${node_ip}\",\"apiSecret\":\"${api_secret}\",\"storageName\":\"${storage_name}\",\"storageType\":\"${storage_type}\"}"

  # For macOS nodes, include hardware info
  if [ "$mode" = "macos" ] && [ "$OS_TYPE" = "macos" ]; then
    local hw_info
    hw_info=$(get_macos_hardware_info)
    # Merge hardware info into payload
    local cpu cores ram_gb disk_gb os_version arch
    cpu=$(echo "$hw_info" | sed -n 's/.*"cpu":"\([^"]*\)".*/\1/p')
    cores=$(echo "$hw_info" | sed -n 's/.*"cores":\([0-9]*\).*/\1/p')
    ram_gb=$(echo "$hw_info" | sed -n 's/.*"ram_gb":\([0-9]*\).*/\1/p')
    disk_gb=$(echo "$hw_info" | sed -n 's/.*"disk_gb":\([0-9]*\).*/\1/p')
    os_version=$(echo "$hw_info" | sed -n 's/.*"os_version":"\([^"]*\)".*/\1/p')
    arch=$(echo "$hw_info" | sed -n 's/.*"arch":"\([^"]*\)".*/\1/p')

    payload="{\"token\":\"${token}\",\"nodeIp\":\"${node_ip}\",\"apiSecret\":\"${api_secret}\",\"storageName\":\"${storage_name}\",\"storageType\":\"${storage_type}\",\"hardwareInfo\":{\"cpu\":\"${cpu}\",\"cores\":${cores},\"ram_gb\":${ram_gb},\"disk_gb\":${disk_gb},\"os_version\":\"${os_version}\",\"arch\":\"${arch}\"}}"
  fi

  local response
  response=$(curl -sS -X POST "${API_BASE}/node/register" \
    -H "Content-Type: application/json" \
    -d "$payload" 2>/dev/null)

  if [ -z "$response" ]; then
    print_error "Unable to connect to DartNode API"
    return 1
  fi

  if echo "$response" | grep -q '"success":true'; then
    local node_name
    node_name=$(json_get "$response" "nodeName")
    print_success "Node registered successfully as: ${node_name}"
    return 0
  else
    local error_msg
    error_msg=$(json_get "$response" "error")
    print_error "Registration failed: ${error_msg:-Unknown error}"
    return 1
  fi
}

function get_primary_ip() {
  # Try to detect the primary IP address
  local ip=""

  if [ "$OS_TYPE" = "macos" ]; then
    ip=$(ipconfig getifaddr en0 2>/dev/null || ipconfig getifaddr en1 2>/dev/null)
  else
    # Try various methods on Linux
    ip=$(ip route get 1 2>/dev/null | awk '{print $7;exit}')
    if [ -z "$ip" ]; then
      ip=$(hostname -I 2>/dev/null | awk '{print $1}')
    fi
  fi

  echo "$ip"
}

function do_token_setup() {
  local token="$1"

  echo ""
  echo -e "${BOLD}============================================${NC}"
  echo -e "${BOLD}   DartNode Token-Based Node Registration   ${NC}"
  echo -e "${BOLD}============================================${NC}"
  echo ""

  # Validate the token first
  if ! validate_token "$token"; then
    exit 1
  fi

  # Fetch configuration
  local config_response
  config_response=$(fetch_token_config "$token")
  if [ $? -ne 0 ]; then
    exit 1
  fi

  # Parse config values using helper functions
  local node_name mode redis_host redis_port redis_password redis_db auth_secret
  node_name=$(json_get "$config_response" "nodeName")
  mode=$(json_get "$config_response" "mode")
  redis_host=$(json_get_nested "$config_response" "redis" "host")
  redis_port=$(json_get_nested "$config_response" "redis" "port")
  redis_password=$(json_get_nested "$config_response" "redis" "password")
  redis_db=$(json_get_nested "$config_response" "redis" "db")
  auth_secret=$(json_get "$config_response" "authSecret")

  # Set defaults
  redis_host=${redis_host:-127.0.0.1}
  redis_port=${redis_port:-6379}
  redis_db=${redis_db:-0}
  mode=${mode:-pve}

  echo -e "Node Name: ${CYAN}${node_name}${NC}"
  echo -e "Mode: ${CYAN}${mode}${NC}"
  echo ""

  # Detect if running non-interactively (piped install)
  local is_interactive=true
  if [ ! -t 0 ]; then
    is_interactive=false
  fi

  # Get node IP
  local detected_ip node_ip
  detected_ip=$(get_primary_ip)

  if [ "$is_interactive" = true ]; then
    echo -e "${BOLD}Network Configuration${NC}"
    read -p "Node IP Address [${detected_ip}]: " node_ip
    node_ip=${node_ip:-$detected_ip}
  else
    # Non-interactive: use detected IP automatically
    node_ip="$detected_ip"
    print_info "Auto-detected node IP: ${node_ip}"
  fi

  if [ -z "$node_ip" ]; then
    print_error "Node IP is required (could not auto-detect)"
    exit 1
  fi

  # Storage configuration
  local storage_name="VM_DATA"
  local storage_type="raw"

  if [ "$mode" = "pve" ] && [ "$is_interactive" = true ]; then
    echo ""
    echo -e "${BOLD}Storage Configuration${NC}"
    read -p "Storage Name [VM_DATA]: " storage_name
    storage_name=${storage_name:-VM_DATA}

    echo "Storage Type:"
    echo "  1) raw (recommended for SSD/NVMe)"
    echo "  2) qcow2"
    read -p "Select [1]: " storage_choice
    case $storage_choice in
      2) storage_type="qcow2" ;;
      *) storage_type="raw" ;;
    esac
  elif [ "$mode" = "macos" ]; then
    # macOS doesn't use traditional storage config
    storage_name="tart"
    storage_type="oci"
  fi

  echo ""

  # Test Redis connection
  print_step "Testing Redis connection..."
  if test_redis_connection "$redis_host" "$redis_port" "$redis_password"; then
    print_success "Redis connection successful"
  else
    print_warning "Could not verify Redis connection"
    if [ "$is_interactive" = true ]; then
      read -p "Continue anyway? (y/n) [y]: " continue_choice
      continue_choice=${continue_choice:-y}
      if [[ ! $continue_choice =~ ^[Yy]$ ]]; then
        exit 1
      fi
    else
      print_info "Continuing in non-interactive mode..."
    fi
  fi

  # Create config directory
  mkdir -p ${DN_DAEMON_CONFIG_DIR}

  # Write main config
  cat > "${DN_DAEMON_CONFIG_DIR}/config.json" <<-EOF
{
    "mode": "${mode}",
    "redis_host": "${redis_host}",
    "redis_port": ${redis_port},
    "redis_password": "${redis_password}",
    "redis_db": ${redis_db},
    "node_id": "${node_name}",
    "auto_update": true,
    "update_interval": 300,
    "health_reporting": true,
    "debug": false
}
EOF

  # Write module-specific config
  if [ "$mode" = "pve" ]; then
    cat > "${DN_DAEMON_CONFIG_DIR}/pve.json" <<-EOF
{
    "poll_interval": 5,
    "stats_ttl": 120,
    "storage_ttl": 300,
    "qmp_socket_dir": "/var/run/qemu-server",
    "backup_path": "/mnt/pve/dn-backups",
    "backup_interval": 300,
    "worker_count": 4,
    "vm_timeout": 8,
    "collect_timeout": 60,
    "vnc_enabled": true,
    "vnc_port": 5700,
    "vnc_socket_dir": "/var/run/qemu-server",
    "vnc_auth_secret": "${auth_secret}",
    "ssh_enabled": true,
    "ssh_auth_secret": "${auth_secret}"
}
EOF
  else
    # macOS config
    local tart_path="/opt/homebrew/bin/tart"
    if command -v tart &> /dev/null; then
      tart_path=$(which tart)
    fi

    cat > "${DN_DAEMON_CONFIG_DIR}/macos.json" <<-EOF
{
    "poll_interval": 5,
    "stats_ttl": 120,
    "storage_ttl": 300,
    "tart_path": "${tart_path}",
    "vm_storage_path": "${HOME}/.tart/vms",
    "worker_count": 4,
    "vm_timeout": 8,
    "collect_timeout": 60,
    "vnc_enabled": true,
    "vnc_port": 5700,
    "vnc_auth_secret": "${auth_secret}",
    "ssh_enabled": true,
    "ssh_auth_secret": "${auth_secret}",
    "job_poll_interval": 1,
    "guest_agent_port": 7777,
    "enable_nat": true,
    "nat_interface": "en0"
}
EOF
  fi

  print_success "Configuration saved!"

  # Register with DartNode API
  echo ""
  if ! register_node_with_token "$token" "$node_ip" "$auth_secret" "$storage_name" "$storage_type" "$mode"; then
    print_warning "Node registration failed. You may need to register manually."
  fi

  # Store auth secret for reference
  echo ""
  echo -e "${YELLOW}============================================${NC}"
  echo -e "${YELLOW}IMPORTANT: Authentication Secret${NC}"
  echo -e "${YELLOW}============================================${NC}"
  echo ""
  echo -e "VNC/SSH Auth Secret: ${BOLD}${auth_secret}${NC}"
  echo ""
  echo -e "${DIM}This has been automatically configured.${NC}"
  echo ""
}

function do_interactive_setup() {
  local mode="$1"

  echo ""
  echo -e "${BOLD}============================================${NC}"
  echo -e "${BOLD}       DartNode Daemon Configuration        ${NC}"
  echo -e "${BOLD}============================================${NC}"
  echo ""

  # If mode not provided, ask for it
  if [ -z "$mode" ]; then
    if [ "$OS_TYPE" = "macos" ]; then
      # On macOS, default to macos mode but allow pve for testing
      echo -e "${BOLD}Select daemon mode:${NC}"
      echo ""
      echo -e "  ${GREEN}1)${NC} ${BOLD}macOS${NC} - Apple Silicon VM hosting via Tart"
      echo -e "  ${DIM}2) PVE   - Proxmox VE (for testing only on macOS)${NC}"
      echo ""
      read -p "Enter selection [1]: " mode_choice
      mode_choice=${mode_choice:-1}
      case $mode_choice in
        2) mode="pve" ;;
        *) mode="macos" ;;
      esac
    else
      # On Linux, offer both options
      echo -e "${BOLD}Select daemon mode:${NC}"
      echo ""
      echo -e "  ${GREEN}1)${NC} ${BOLD}PVE${NC}   - Proxmox VE KVM virtualization"
      echo -e "  ${DIM}2) macOS - Apple Silicon (requires macOS host)${NC}"
      echo ""
      read -p "Enter selection [1]: " mode_choice
      mode_choice=${mode_choice:-1}
      case $mode_choice in
        2) mode="macos" ;;
        *) mode="pve" ;;
      esac
    fi
  fi

  echo ""
  print_step "Configuring for mode: ${mode}"
  echo ""

  # Get hostname as default node_id
  local default_node=$(hostname | tr '[:upper:]' '[:lower:]' | tr -cd 'a-z0-9-')

  # Redis configuration
  echo -e "${BOLD}Redis Connection${NC}"
  echo -e "${DIM}The daemon requires Redis for caching and job queues.${NC}"
  echo ""

  local redis_host redis_port redis_pass redis_db node_id

  read -p "Redis Host [127.0.0.1]: " redis_host
  redis_host=${redis_host:-127.0.0.1}

  read -p "Redis Port [6379]: " redis_port
  redis_port=${redis_port:-6379}

  read -p "Redis Password (leave empty for none): " redis_pass

  read -p "Redis Database [0]: " redis_db
  redis_db=${redis_db:-0}

  echo ""
  echo -e "${BOLD}Node Configuration${NC}"
  echo ""

  read -p "Node ID [${default_node}]: " node_id
  node_id=${node_id:-$default_node}

  # Test Redis connection
  echo ""
  print_step "Testing Redis connection..."
  if test_redis_connection "$redis_host" "$redis_port" "$redis_pass"; then
    print_success "Redis connection successful"
  else
    print_warning "Could not verify Redis connection. Please ensure Redis is running."
    read -p "Continue anyway? (y/n) [y]: " continue_choice
    continue_choice=${continue_choice:-y}
    if [[ ! $continue_choice =~ ^[Yy]$ ]]; then
      exit 1
    fi
  fi

  # Module-specific configuration
  echo ""
  if [ "$mode" = "macos" ]; then
    configure_macos_module "$redis_host" "$redis_port" "$redis_pass" "$redis_db" "$node_id"
  else
    configure_pve_module "$redis_host" "$redis_port" "$redis_pass" "$redis_db" "$node_id"
  fi

  echo ""
  print_success "Configuration complete!"
  echo ""
}

function test_redis_connection() {
  local host="$1"
  local port="$2"
  local pass="$3"

  if command -v redis-cli &> /dev/null; then
    local result
    if [ -n "$pass" ]; then
      result=$(redis-cli -h "$host" -p "$port" -a "$pass" ping 2>/dev/null)
    else
      result=$(redis-cli -h "$host" -p "$port" ping 2>/dev/null)
    fi
    [ "$result" == "PONG" ]
  else
    # Can't test without redis-cli, assume it's fine
    return 0
  fi
}

function configure_pve_module() {
  local redis_host="$1"
  local redis_port="$2"
  local redis_pass="$3"
  local redis_db="$4"
  local node_id="$5"

  echo -e "${BOLD}PVE Module Configuration${NC}"
  echo -e "${DIM}Configure Proxmox-specific settings.${NC}"
  echo ""

  local poll_interval qmp_socket_dir vnc_enabled vnc_port ssh_enabled

  read -p "Stats poll interval (seconds) [5]: " poll_interval
  poll_interval=${poll_interval:-5}

  read -p "QMP socket directory [/var/run/qemu-server]: " qmp_socket_dir
  qmp_socket_dir=${qmp_socket_dir:-/var/run/qemu-server}

  echo ""
  echo -e "${BOLD}Console Access${NC}"
  echo ""

  read -p "Enable VNC proxy? (y/n) [n]: " vnc_choice
  vnc_enabled=false
  vnc_port=5700
  if [[ $vnc_choice =~ ^[Yy]$ ]]; then
    vnc_enabled=true
    read -p "VNC proxy port [5700]: " vnc_port
    vnc_port=${vnc_port:-5700}
  fi

  read -p "Enable SSH proxy? (y/n) [n]: " ssh_choice
  ssh_enabled=false
  if [[ $ssh_choice =~ ^[Yy]$ ]]; then
    ssh_enabled=true
  fi

  # Generate shared secret if proxies are enabled
  local shared_secret=""
  if [ "$vnc_enabled" = true ] || [ "$ssh_enabled" = true ]; then
    shared_secret=$(generate_secret)
    echo ""
    print_info "Generated shared secret for proxy authentication."
  fi

  # Write main config
  cat > "${DN_DAEMON_CONFIG_DIR}/config.json" <<-EOF
{
    "mode": "pve",
    "redis_host": "${redis_host}",
    "redis_port": ${redis_port},
    "redis_password": "${redis_pass}",
    "redis_db": ${redis_db},
    "node_id": "${node_id}",
    "auto_update": true,
    "update_interval": 300,
    "health_reporting": true,
    "debug": false
}
EOF

  # Write PVE module config
  cat > "${DN_DAEMON_CONFIG_DIR}/pve.json" <<-EOF
{
    "poll_interval": ${poll_interval},
    "stats_ttl": 120,
    "storage_ttl": 300,
    "qmp_socket_dir": "${qmp_socket_dir}",
    "backup_path": "/mnt/pve/dn-backups",
    "backup_interval": 300,
    "worker_count": 4,
    "vm_timeout": 8,
    "collect_timeout": 60,
    "vnc_enabled": ${vnc_enabled},
    "vnc_port": ${vnc_port},
    "vnc_socket_dir": "${qmp_socket_dir}",
    "vnc_auth_secret": "${shared_secret}",
    "ssh_enabled": ${ssh_enabled},
    "ssh_auth_secret": "${shared_secret}"
}
EOF

  if [ -n "$shared_secret" ]; then
    echo ""
    echo -e "${YELLOW}IMPORTANT:${NC} Save this shared secret for your web application:"
    echo -e "${BOLD}${shared_secret}${NC}"
    echo ""
  fi
}

function configure_macos_module() {
  local redis_host="$1"
  local redis_port="$2"
  local redis_pass="$3"
  local redis_db="$4"
  local node_id="$5"

  echo -e "${BOLD}macOS Module Configuration${NC}"
  echo -e "${DIM}Configure Apple Silicon VM hosting settings.${NC}"
  echo ""

  local poll_interval tart_path vm_storage vnc_enabled vnc_port ssh_enabled guest_agent_port

  read -p "Stats poll interval (seconds) [5]: " poll_interval
  poll_interval=${poll_interval:-5}

  # Auto-detect Tart path
  local default_tart="/opt/homebrew/bin/tart"
  if command -v tart &> /dev/null; then
    default_tart=$(which tart)
  fi

  read -p "Tart binary path [${default_tart}]: " tart_path
  tart_path=${tart_path:-$default_tart}

  # VM storage path
  local default_storage="${HOME}/.tart/vms"
  read -p "VM storage path [${default_storage}]: " vm_storage
  vm_storage=${vm_storage:-$default_storage}

  echo ""
  echo -e "${BOLD}Guest Agent${NC}"
  echo ""

  read -p "Guest agent port [7777]: " guest_agent_port
  guest_agent_port=${guest_agent_port:-7777}

  echo ""
  echo -e "${BOLD}Console Access${NC}"
  echo ""

  read -p "Enable VNC proxy? (y/n) [y]: " vnc_choice
  vnc_choice=${vnc_choice:-y}
  vnc_enabled=false
  vnc_port=5700
  if [[ $vnc_choice =~ ^[Yy]$ ]]; then
    vnc_enabled=true
    read -p "VNC proxy port [5700]: " vnc_port
    vnc_port=${vnc_port:-5700}
  fi

  read -p "Enable SSH proxy? (y/n) [y]: " ssh_choice
  ssh_choice=${ssh_choice:-y}
  ssh_enabled=false
  if [[ $ssh_choice =~ ^[Yy]$ ]]; then
    ssh_enabled=true
  fi

  # Generate shared secret if proxies are enabled
  local shared_secret=""
  if [ "$vnc_enabled" = true ] || [ "$ssh_enabled" = true ]; then
    shared_secret=$(generate_secret)
    echo ""
    print_info "Generated shared secret for proxy authentication."
  fi

  echo ""
  echo -e "${BOLD}Network Configuration${NC}"
  echo ""

  local enable_nat default_gateway
  read -p "Enable NAT for VMs? (y/n) [y]: " nat_choice
  nat_choice=${nat_choice:-y}
  enable_nat=false
  if [[ $nat_choice =~ ^[Yy]$ ]]; then
    enable_nat=true
    read -p "Default gateway interface [en0]: " default_gateway
    default_gateway=${default_gateway:-en0}
  fi

  # Write main config
  cat > "${DN_DAEMON_CONFIG_DIR}/config.json" <<-EOF
{
    "mode": "macos",
    "redis_host": "${redis_host}",
    "redis_port": ${redis_port},
    "redis_password": "${redis_pass}",
    "redis_db": ${redis_db},
    "node_id": "${node_id}",
    "auto_update": true,
    "update_interval": 300,
    "health_reporting": true,
    "debug": false
}
EOF

  # Write macOS module config
  cat > "${DN_DAEMON_CONFIG_DIR}/macos.json" <<-EOF
{
    "poll_interval": ${poll_interval},
    "stats_ttl": 120,
    "storage_ttl": 300,
    "tart_path": "${tart_path}",
    "vm_storage_path": "${vm_storage}",
    "worker_count": 4,
    "vm_timeout": 8,
    "collect_timeout": 60,
    "vnc_enabled": ${vnc_enabled},
    "vnc_port": ${vnc_port},
    "vnc_auth_secret": "${shared_secret}",
    "ssh_enabled": ${ssh_enabled},
    "ssh_auth_secret": "${shared_secret}",
    "job_poll_interval": 1,
    "guest_agent_port": ${guest_agent_port},
    "enable_nat": ${enable_nat},
    "nat_interface": "${default_gateway:-en0}"
}
EOF

  if [ -n "$shared_secret" ]; then
    echo ""
    echo -e "${YELLOW}============================================${NC}"
    echo -e "${YELLOW}IMPORTANT: Save this shared secret!${NC}"
    echo -e "${YELLOW}============================================${NC}"
    echo ""
    echo -e "Add this to your DartNode web application config:"
    echo -e "${BOLD}DAEMON_SECRET=${shared_secret}${NC}"
    echo ""
    echo -e "${DIM}This secret is used for VNC/SSH proxy authentication.${NC}"
    echo ""
  fi
}

function do_migrate() {
  check_root

  if ! detect_legacy; then
    echo "No legacy pve-stats installation found."
    exit 0
  fi

  echo "Migrating from pve-stats to dn-daemon..."
  echo ""

  # Stop legacy service
  if systemctl is-active --quiet ${LEGACY_SERVICE} 2>/dev/null; then
    echo "Stopping legacy service..."
    systemctl stop ${LEGACY_SERVICE}
    systemctl disable ${LEGACY_SERVICE}
  fi

  # Create new config directory
  mkdir -p ${DN_DAEMON_CONFIG_DIR}

  # Migrate config if exists
  if [ -f "${LEGACY_CONFIG_DIR}/config.json" ]; then
    echo "Migrating configuration..."
    # Add mode field and copy config
    if command -v python3 &> /dev/null; then
      python3 -c "
import json
with open('${LEGACY_CONFIG_DIR}/config.json', 'r') as f:
    config = json.load(f)
config['mode'] = 'pve'
with open('${DN_DAEMON_CONFIG_DIR}/config.json', 'w') as f:
    json.dump(config, f, indent=4)
print('Configuration migrated successfully')
" 2>/dev/null
    else
      # Fallback: copy and manually add mode
      cp "${LEGACY_CONFIG_DIR}/config.json" "${DN_DAEMON_CONFIG_DIR}/config.json"
      # This is a simple approach - user may need to add mode manually
      echo "  Note: Please add '\"mode\": \"pve\"' to the config file"
    fi
  fi

  # Remove legacy files
  echo "Removing legacy files..."
  rm -f ${LEGACY_BIN}
  rm -f /etc/systemd/system/${LEGACY_SERVICE}.service
  systemctl daemon-reload

  echo ""
  echo "Migration complete! Run 'dartnode install' to install the new daemon."
  echo ""
}

function _install_bash_completion() {
  local comp_script='_dartnode() {
    local cur prev commands
    COMPREPLY=()
    cur="${COMP_WORDS[COMP_CWORD]}"
    prev="${COMP_WORDS[COMP_CWORD-1]}"

    commands="install setup config start stop restart status log update version uninstall migrate net disk drive led vms diag redis storage fw qmp repair jobs help"

    case "$prev" in
        dartnode)
            COMPREPLY=( $(compgen -W "$commands" -- "$cur") )
            return 0
            ;;
        net|network)
            COMPREPLY=( $(compgen -W "show scan identify watch up down" -- "$cur") )
            return 0
            ;;
        disk|drive|led)
            COMPREPLY=( $(compgen -W "list smart led-on led-off locate locate-off enclosure slots enclosure-led slot-led blink-array" -- "$cur") )
            return 0
            ;;
        jobs|job)
            COMPREPLY=( $(compgen -W "list clear flush inspect retry pve locks" -- "$cur") )
            return 0
            ;;
        identify|up|down)
            local ifaces=$(ls /sys/class/net/ 2>/dev/null | grep -v lo)
            COMPREPLY=( $(compgen -W "$ifaces" -- "$cur") )
            return 0
            ;;
        led-on|led-off|locate|locate-off)
            local devs=$(lsblk -dn -o NAME 2>/dev/null | sed "s|^|/dev/|")
            COMPREPLY=( $(compgen -W "$devs" -- "$cur") )
            return 0
            ;;
        qmp)
            local vmids=$(qm list 2>/dev/null | tail -n +2 | awk "{print \$1}")
            COMPREPLY=( $(compgen -W "$vmids" -- "$cur") )
            return 0
            ;;
    esac
}
complete -F _dartnode dartnode'

  # Install for bash
  local bash_comp_dir="/etc/bash_completion.d"
  if [ -d "$bash_comp_dir" ]; then
    echo "$comp_script" > "${bash_comp_dir}/dartnode"
    chmod 644 "${bash_comp_dir}/dartnode"
  elif [ -d "/usr/local/etc/bash_completion.d" ]; then
    echo "$comp_script" > "/usr/local/etc/bash_completion.d/dartnode"
    chmod 644 "/usr/local/etc/bash_completion.d/dartnode"
  fi

  # Install for zsh (if using bashcompinit or zsh-completions)
  if [ "$OS_TYPE" = "macos" ]; then
    local zsh_comp_dir="/usr/local/share/zsh/site-functions"
    [ -d "/opt/homebrew/share/zsh/site-functions" ] && zsh_comp_dir="/opt/homebrew/share/zsh/site-functions"
    if [ -d "$zsh_comp_dir" ]; then
      cat > "${zsh_comp_dir}/_dartnode" <<'ZSHEOF'
#compdef dartnode

_dartnode() {
  local -a commands net_subs disk_subs jobs_subs

  commands=(
    'install:Install DN Daemon'
    'setup:Run interactive setup wizard'
    'config:Configure daemon settings'
    'start:Start the daemon'
    'stop:Stop the daemon'
    'restart:Restart the daemon'
    'status:Show daemon status'
    'log:Follow daemon logs'
    'update:Update daemon binary'
    'version:Show daemon version'
    'uninstall:Remove daemon'
    'net:Network tools'
    'disk:Drive tools & LED identification'
    'vms:List VMs on this node'
    'diag:Full node diagnostics'
    'redis:Redis diagnostics'
    'storage:Storage overview'
    'fw:Firewall rules'
    'qmp:Test QMP socket'
    'repair:Diagnose & fix node issues'
    'jobs:Manage job queue'
    'help:Show help'
  )

  net_subs=('show' 'scan' 'identify' 'watch' 'up' 'down')
  disk_subs=('list' 'smart' 'led-on' 'led-off' 'locate' 'locate-off' 'enclosure' 'slots' 'enclosure-led' 'slot-led' 'blink-array')
  jobs_subs=('list' 'clear' 'flush' 'inspect' 'retry' 'pve' 'locks')

  if (( CURRENT == 2 )); then
    _describe 'command' commands
  elif (( CURRENT == 3 )); then
    case "${words[2]}" in
      net|network)
        _describe 'subcommand' net_subs ;;
      disk|drive|led)
        _describe 'subcommand' disk_subs ;;
      jobs|job)
        _describe 'subcommand' jobs_subs ;;
      qmp)
        local vmids=(${(f)"$(qm list 2>/dev/null | tail -n +2 | awk '{print $1}')"})
        _describe 'vmid' vmids ;;
      *)
        ;;
    esac
  elif (( CURRENT == 4 )); then
    case "${words[3]}" in
      identify|up|down)
        local ifaces=(${(f)"$(ls /sys/class/net/ 2>/dev/null | grep -v lo)"})
        _describe 'interface' ifaces ;;
      led-on|led-off|locate|locate-off)
        local devs=(${(f)"$(lsblk -dn -o NAME 2>/dev/null | sed 's|^|/dev/|')"})
        _describe 'device' devs ;;
    esac
  fi
}

_dartnode "$@"
ZSHEOF
      chmod 644 "${zsh_comp_dir}/_dartnode"
    fi
  fi
}

function do_install() {
  check_root

  # SMTP Gate has its own standalone installer
  if [ "${ARGS[0]}" = "smtp-gate" ]; then
    echo -e "${YELLOW}SMTP Gate now has a dedicated installer.${NC}"
    echo ""
    echo "Install with:"
    echo "  curl -sSL ${PKG_BASE}/smtp-gate.sh | bash -s install"
    echo ""
    echo "Or if already installed, manage with:"
    echo "  smtp-gate-ctl start|stop|restart|status|log|update|config|uninstall"
    return
  fi

  if [ "$OS_TYPE" = "macos" ]; then
    do_install_macos
  else
    do_install_linux
  fi
}

# ============================================
# SMTP Gate — Redirects to standalone installer
# ============================================

function do_smtp_gate_action() {
  if command -v smtp-gate-ctl &>/dev/null; then
    smtp-gate-ctl "${1:-${ARGS[0]}}"
  else
    echo -e "${YELLOW}SMTP Gate now has a dedicated management tool.${NC}"
    echo ""
    echo "Install with:"
    echo "  curl -sSL ${PKG_BASE}/smtp-gate.sh | bash -s install"
    echo ""
    echo "After installation, manage with: smtp-gate-ctl <command>"
  fi
}

# Flag to track if we're using token-based setup
USE_TOKEN_SETUP=false

function do_install_linux() {
  check_proxmox

  # Check for legacy installation
  if detect_legacy; then
    echo "Legacy pve-stats installation detected."
    read -p "Migrate to new dn-daemon? (y/n) " -n 1 -r
    echo
    if [[ $REPLY =~ ^[Yy]$ ]]; then
      do_migrate
    fi
  fi

  print_step "Installing DN Daemon (Linux)..."
  echo ""

  # Create directories
  mkdir -p /usr/local/dartnode
  mkdir -p ${DN_DAEMON_CONFIG_DIR}

  # Stop existing service if running
  if systemctl is-active --quiet ${DN_DAEMON_SERVICE} 2>/dev/null; then
    print_info "Stopping existing service..."
    systemctl stop ${DN_DAEMON_SERVICE}
  fi

  # Download this script
  print_step "Downloading dartnode CLI..."
  wget -q --show-progress -O /usr/local/dartnode/dartnode.sh ${PKG_BASE}/dartnode.sh
  chmod +x /usr/local/dartnode/dartnode.sh

  # Create symlinks
  ln -sf /usr/local/dartnode/dartnode.sh /usr/local/bin/dartnode
  ln -sf /usr/local/dartnode/dartnode.sh /bin/dartnode

  # Install bash completion
  _install_bash_completion

  # Download binary
  print_step "Downloading dn-daemon..."
  wget -q --show-progress -O ${DN_DAEMON_BIN} ${PKG_BASE}/dn-daemon
  chmod +x ${DN_DAEMON_BIN}

  # Create legacy symlink for backwards compatibility
  ln -sf ${DN_DAEMON_BIN} ${LEGACY_BIN}

  # Install systemd service
  print_step "Installing systemd service..."
  cat > "/etc/systemd/system/${DN_DAEMON_SERVICE}.service" <<-EOF
[Unit]
Description=DartNode Daemon
Documentation=https://dartnode.com
After=network.target
Wants=network-online.target

[Service]
Type=simple
ExecStart=${DN_DAEMON_BIN} -config ${DN_DAEMON_CONFIG_DIR}/config.json
Restart=always
RestartSec=5
StandardOutput=journal
StandardError=journal
SyslogIdentifier=dn-daemon

[Install]
WantedBy=multi-user.target
EOF

  systemctl daemon-reload
  systemctl enable ${DN_DAEMON_SERVICE}

  # Run setup based on whether we have a registration token
  if [ -n "$REGISTRATION_TOKEN" ]; then
    # Token-based setup
    do_token_setup "$REGISTRATION_TOKEN"
  elif [ ! -f "${DN_DAEMON_CONFIG_DIR}/config.json" ] || [ "$INTERACTIVE" = true ]; then
    # Interactive setup
    if [ -f "${DN_DAEMON_CONFIG_DIR}/config.json" ]; then
      echo ""
      read -p "Configuration exists. Run setup wizard? (y/n) [y]: " run_setup
      run_setup=${run_setup:-y}
      if [[ $run_setup =~ ^[Yy]$ ]]; then
        do_interactive_setup "pve"
      fi
    else
      do_interactive_setup "pve"
    fi
  fi

  echo ""
  echo -e "${GREEN}============================================${NC}"
  echo -e "${GREEN}        Installation Complete!              ${NC}"
  echo -e "${GREEN}============================================${NC}"
  echo ""
  echo "Commands:"
  echo "  dartnode start     - Start the daemon"
  echo "  dartnode status    - Check daemon status"
  echo "  dartnode config    - Reconfigure settings"
  echo "  dartnode log       - View daemon logs"
  echo ""

  # Ask to start
  read -p "Start daemon now? (y/n) [y]: " start_now
  start_now=${start_now:-y}
  if [[ $start_now =~ ^[Yy]$ ]]; then
    do_start
  fi
}

function do_install_macos() {
  check_macos
  check_homebrew
  check_tart

  echo ""
  print_step "Installing DN Daemon (macOS)..."
  echo ""

  # Create directories
  mkdir -p /opt/dn-daemon
  mkdir -p /usr/local/dartnode
  mkdir -p ${DN_DAEMON_CONFIG_DIR}
  mkdir -p /var/log/dn-daemon

  # Stop existing service if running
  if launchctl list | grep -q "${DN_DAEMON_SERVICE}"; then
    print_info "Stopping existing service..."
    launchctl unload ${DN_LAUNCHD_PLIST} 2>/dev/null
  fi

  # Download this script
  print_step "Downloading dartnode CLI..."
  curl -sSL -o /usr/local/dartnode/dartnode.sh ${PKG_BASE}/dartnode.sh
  chmod +x /usr/local/dartnode/dartnode.sh

  # Create symlinks
  ln -sf /usr/local/dartnode/dartnode.sh /usr/local/bin/dartnode
  ln -sf /usr/local/dartnode/dartnode.sh /usr/bin/dartnode 2>/dev/null || true

  # Install bash/zsh completion
  _install_bash_completion

  # Download macOS binary
  print_step "Downloading dn-daemon (darwin-arm64)..."
  curl -sSL -o ${DN_DAEMON_BIN} ${PKG_BASE}/dn-daemon-darwin-arm64
  chmod +x ${DN_DAEMON_BIN}

  # Download guest agent
  print_step "Downloading dn-guest-agent..."
  curl -sSL -o /opt/dn-daemon/dn-guest-agent ${PKG_BASE}/dn-guest-agent
  chmod +x /opt/dn-daemon/dn-guest-agent

  # Get current user for LaunchDaemon
  CURRENT_USER=$(whoami)

  # Install LaunchDaemon plist
  print_step "Installing LaunchDaemon..."
  cat > "${DN_LAUNCHD_PLIST}" <<-EOF
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE plist PUBLIC "-//Apple//DTD PLIST 1.0//EN" "http://www.apple.com/DTDs/PropertyList-1.0.dtd">
<plist version="1.0">
<dict>
    <key>Label</key>
    <string>${DN_DAEMON_SERVICE}</string>
    <key>ProgramArguments</key>
    <array>
        <string>${DN_DAEMON_BIN}</string>
        <string>-config</string>
        <string>${DN_DAEMON_CONFIG_DIR}/config.json</string>
    </array>
    <key>RunAtLoad</key>
    <true/>
    <key>KeepAlive</key>
    <true/>
    <key>StandardOutPath</key>
    <string>/var/log/dn-daemon/daemon.log</string>
    <key>StandardErrorPath</key>
    <string>/var/log/dn-daemon/daemon.err</string>
    <key>UserName</key>
    <string>${CURRENT_USER}</string>
    <key>EnvironmentVariables</key>
    <dict>
        <key>PATH</key>
        <string>/opt/homebrew/bin:/usr/local/bin:/usr/bin:/bin</string>
    </dict>
</dict>
</plist>
EOF

  # Run setup based on whether we have a registration token
  if [ -n "$REGISTRATION_TOKEN" ]; then
    # Token-based setup — fully automated
    do_token_setup "$REGISTRATION_TOKEN"
    setup_macos_host
  elif [ ! -f "${DN_DAEMON_CONFIG_DIR}/config.json" ] || [ "$INTERACTIVE" = true ]; then
    # Interactive setup
    if [ -f "${DN_DAEMON_CONFIG_DIR}/config.json" ]; then
      echo ""
      read -p "Configuration exists. Run setup wizard? (y/n) [y]: " run_setup
      run_setup=${run_setup:-y}
      if [[ $run_setup =~ ^[Yy]$ ]]; then
        do_interactive_setup "macos"
      fi
    else
      do_interactive_setup "macos"
    fi
    setup_macos_host
  fi

  echo ""
  echo -e "${GREEN}============================================${NC}"
  echo -e "${GREEN}  macOS Host Node Installation Complete!    ${NC}"
  echo -e "${GREEN}============================================${NC}"
  echo ""
  echo -e "${BOLD}System Summary:${NC}"
  echo -e "  macOS:   $(sw_vers -productVersion) on $(uname -m)"
  echo -e "  Tart:    $(tart --version 2>&1 | head -1)"
  echo -e "  Daemon:  ${DN_DAEMON_BIN}"
  echo -e "  Config:  ${DN_DAEMON_CONFIG_DIR}/"
  echo -e "  Logs:    /var/log/dn-daemon/"
  echo ""
  echo -e "${BOLD}Hardware:${NC}"
  echo -e "  CPU:     $(sysctl -n machdep.cpu.brand_string 2>/dev/null || echo 'Apple Silicon')"
  echo -e "  Cores:   $(sysctl -n hw.ncpu 2>/dev/null)"
  echo -e "  Memory:  $(( $(sysctl -n hw.memsize 2>/dev/null) / 1073741824 )) GB"
  echo -e "  Disk:    $(df -h / | awk 'NR==2 {print $2}') total"
  echo ""
  echo "Commands:"
  echo "  sudo dartnode start     - Start the daemon"
  echo "  sudo dartnode status    - Check daemon status"
  echo "  sudo dartnode config    - Reconfigure settings"
  echo "  sudo dartnode log       - View daemon logs"
  echo ""
  echo -e "${CYAN}Guest agent location:${NC}"
  echo "  /opt/dn-daemon/dn-guest-agent"
  echo ""
  echo -e "${DIM}Copy the guest agent to your macOS VM images for${NC}"
  echo -e "${DIM}in-VM stats collection and network configuration.${NC}"
  echo ""

  # Auto-start when using token (non-interactive), ask otherwise
  if [ -n "$REGISTRATION_TOKEN" ]; then
    do_start
  else
    read -p "Start daemon now? (y/n) [y]: " start_now
    start_now=${start_now:-y}
    if [[ $start_now =~ ^[Yy]$ ]]; then
      do_start
    fi
  fi
}

# ============================================
# macOS Host Provisioning
# ============================================

function setup_macos_host() {
  echo ""
  echo -e "${BOLD}============================================${NC}"
  echo -e "${BOLD}   macOS Host Node Provisioning             ${NC}"
  echo -e "${BOLD}============================================${NC}"
  echo ""

  setup_macos_pf_firewall
  setup_macos_sudoers
  setup_macos_system_tuning
  setup_macos_log_rotation
  setup_macos_base_images
}

function setup_macos_pf_firewall() {
  print_step "Configuring PF firewall for VM NAT..."

  # Create PF anchor directory
  mkdir -p /etc/pf.anchors

  # Create the DartNode PF anchor file
  cat > /etc/pf.anchors/com.dartnode <<-'EOF'
# DartNode VM NAT Rules
# Individual VM anchors are loaded as sub-anchors: dartnode-vm-{serviceId}
# This file is the parent anchor loaded by pf.conf
EOF

  # Create the main DartNode PF config that loads per-VM anchors
  cat > /etc/pf.anchors/com.dartnode.rules <<-'EOF'
# DartNode NAT — enable IP forwarding and load VM-specific anchors
nat-anchor "dartnode-vm-*"
rdr-anchor "dartnode-vm-*"
EOF

  # Check if pf.conf already has our anchor
  if ! grep -q "com.dartnode" /etc/pf.conf 2>/dev/null; then
    # Back up original
    cp /etc/pf.conf /etc/pf.conf.dartnode-backup 2>/dev/null

    # Append our anchor references
    cat >> /etc/pf.conf <<-'EOF'

# DartNode VM NAT anchors
nat-anchor "dartnode-vm-*"
rdr-anchor "dartnode-vm-*"
load anchor "com.dartnode" from "/etc/pf.anchors/com.dartnode"
EOF

    print_success "PF anchor added to /etc/pf.conf"
  else
    print_info "PF anchor already configured"
  fi

  # Enable IP forwarding
  sysctl -w net.inet.ip.forwarding=1 >/dev/null 2>&1
  sysctl -w net.inet6.ip6.forwarding=1 >/dev/null 2>&1

  # Make IP forwarding persistent across reboots
  if ! grep -q "net.inet.ip.forwarding" /etc/sysctl.conf 2>/dev/null; then
    echo "net.inet.ip.forwarding=1" >> /etc/sysctl.conf
    echo "net.inet6.ip6.forwarding=1" >> /etc/sysctl.conf
  fi

  # Enable PF if not already running
  pfctl -e 2>/dev/null || true
  pfctl -f /etc/pf.conf 2>/dev/null || true

  print_success "PF firewall configured for VM NAT"
}

function setup_macos_sudoers() {
  print_step "Configuring sudoers for daemon PF management..."

  local DAEMON_USER="${CURRENT_USER:-$(whoami)}"
  local SUDOERS_FILE="/etc/sudoers.d/dartnode-daemon"

  # Create sudoers entry so daemon can manage PF without password
  cat > "${SUDOERS_FILE}" <<-EOF
# DartNode Daemon — allow PF firewall management without password
${DAEMON_USER} ALL=(ALL) NOPASSWD: /sbin/pfctl
${DAEMON_USER} ALL=(ALL) NOPASSWD: /usr/bin/tee /etc/pf.anchors/dartnode-vm-*
${DAEMON_USER} ALL=(ALL) NOPASSWD: /bin/rm -f /etc/pf.anchors/dartnode-vm-*
EOF

  chmod 0440 "${SUDOERS_FILE}"

  # Validate sudoers syntax
  if visudo -cf "${SUDOERS_FILE}" >/dev/null 2>&1; then
    print_success "Sudoers configured for user: ${DAEMON_USER}"
  else
    print_error "Sudoers validation failed, removing file"
    rm -f "${SUDOERS_FILE}"
  fi
}

function setup_macos_system_tuning() {
  print_step "Applying system tuning for VM hosting..."

  # Increase file descriptor limits for many concurrent VM connections
  if [ ! -f /Library/LaunchDaemons/com.dartnode.sysctl.plist ]; then
    cat > /Library/LaunchDaemons/com.dartnode.sysctl.plist <<-'EOF'
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE plist PUBLIC "-//Apple//DTD PLIST 1.0//EN" "http://www.apple.com/DTDs/PropertyList-1.0.dtd">
<plist version="1.0">
<dict>
    <key>Label</key>
    <string>com.dartnode.sysctl</string>
    <key>ProgramArguments</key>
    <array>
        <string>/bin/sh</string>
        <string>-c</string>
        <string>sysctl -w kern.maxfiles=65536 kern.maxfilesperproc=32768 net.inet.ip.forwarding=1 net.inet6.ip6.forwarding=1</string>
    </array>
    <key>RunAtLoad</key>
    <true/>
</dict>
</plist>
EOF
    launchctl load /Library/LaunchDaemons/com.dartnode.sysctl.plist 2>/dev/null
  fi

  # Apply immediately
  sysctl -w kern.maxfiles=65536 >/dev/null 2>&1
  sysctl -w kern.maxfilesperproc=32768 >/dev/null 2>&1

  # Increase launchd resource limits for the daemon
  if [ ! -f /Library/LaunchDaemons/com.dartnode.limit.plist ]; then
    cat > /Library/LaunchDaemons/com.dartnode.limit.plist <<-'EOF'
<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE plist PUBLIC "-//Apple//DTD PLIST 1.0//EN" "http://www.apple.com/DTDs/PropertyList-1.0.dtd">
<plist version="1.0">
<dict>
    <key>Label</key>
    <string>com.dartnode.limit</string>
    <key>ProgramArguments</key>
    <array>
        <string>/bin/launchctl</string>
        <string>limit</string>
        <string>maxfiles</string>
        <string>65536</string>
        <string>65536</string>
    </array>
    <key>RunAtLoad</key>
    <true/>
</dict>
</plist>
EOF
    launchctl load /Library/LaunchDaemons/com.dartnode.limit.plist 2>/dev/null
  fi

  print_success "System tuning applied"
}

function setup_macos_log_rotation() {
  print_step "Configuring log rotation..."

  # Create newsyslog entry for daemon logs
  if [ ! -f /etc/newsyslog.d/com.dartnode.conf ]; then
    cat > /etc/newsyslog.d/com.dartnode.conf <<-EOF
# logfilename                          [owner:group]  mode count size(KB) when  flags [/pid_file] [sig_num]
/var/log/dn-daemon/daemon.log                         644  7     10240    *     GJ
/var/log/dn-daemon/daemon.err                         644  7     10240    *     GJ
EOF
    print_success "Log rotation configured (7 files, 10MB each)"
  else
    print_info "Log rotation already configured"
  fi
}

function setup_macos_base_images() {
  print_step "Checking base VM images..."

  # Check if we already have a base image
  local has_images=false
  if tart list --format json 2>/dev/null | grep -q '"Source" : "OCI"'; then
    has_images=true
    print_info "Base images already cached"
    tart list --format json 2>/dev/null | python3 -c "
import sys, json
vms = json.load(sys.stdin)
for vm in vms:
    if vm.get('Source') == 'OCI':
        print(f\"  {vm['Name']} ({vm.get('Size', '?')} GB)\")
" 2>/dev/null
    return
  fi

  if [ -n "$REGISTRATION_TOKEN" ] || [ "$INTERACTIVE" = false ]; then
    # Non-interactive: pull default base image
    echo "  Pulling macOS Sequoia base image (this may take 15-30 minutes)..."
    if tart pull ghcr.io/cirruslabs/macos-sequoia-base:latest 2>/dev/null; then
      print_success "Base image pulled: ghcr.io/cirruslabs/macos-sequoia-base:latest"
    else
      print_warning "Failed to pull base image. You can pull it later with:"
      echo "  tart pull ghcr.io/cirruslabs/macos-sequoia-base:latest"
    fi
  else
    # Interactive: ask which images to pull
    echo ""
    echo "  Available base images:"
    echo "    1) macOS Sequoia (recommended)"
    echo "    2) macOS Sonoma"
    echo "    3) Skip for now"
    echo ""
    read -p "  Pull base image? [1]: " image_choice
    image_choice=${image_choice:-1}

    case $image_choice in
      1)
        echo "  Pulling macOS Sequoia base image..."
        tart pull ghcr.io/cirruslabs/macos-sequoia-base:latest 2>/dev/null && \
          print_success "Base image pulled" || \
          print_warning "Pull failed. Try again later: tart pull ghcr.io/cirruslabs/macos-sequoia-base:latest"
        ;;
      2)
        echo "  Pulling macOS Sonoma base image..."
        tart pull ghcr.io/cirruslabs/macos-sonoma-base:latest 2>/dev/null && \
          print_success "Base image pulled" || \
          print_warning "Pull failed. Try again later: tart pull ghcr.io/cirruslabs/macos-sonoma-base:latest"
        ;;
      *)
        print_info "Skipping base image pull"
        ;;
    esac
  fi
}

function get_macos_hardware_info() {
  # Collect hardware info for node registration
  local cpu_brand cpu_cores total_ram_gb disk_total_gb macos_version

  cpu_brand=$(sysctl -n machdep.cpu.brand_string 2>/dev/null || echo "Apple Silicon")
  cpu_cores=$(sysctl -n hw.ncpu 2>/dev/null || echo "0")
  total_ram_gb=$(( $(sysctl -n hw.memsize 2>/dev/null || echo 0) / 1073741824 ))
  disk_total_gb=$(df -g / 2>/dev/null | awk 'NR==2 {print $2}' || echo "0")
  macos_version=$(sw_vers -productVersion 2>/dev/null || echo "unknown")

  # Return as JSON
  echo "{\"cpu\":\"${cpu_brand}\",\"cores\":${cpu_cores},\"ram_gb\":${total_ram_gb},\"disk_gb\":${disk_total_gb},\"os_version\":\"${macos_version}\",\"arch\":\"$(uname -m)\"}"
}

function do_config() {
  check_root

  if [ ! -f "${DN_DAEMON_CONFIG_DIR}/config.json" ]; then
    print_error "Config file not found. Run 'dartnode install' first."
    exit 1
  fi

  echo ""
  echo -e "${BOLD}DN Daemon Configuration${NC}"
  echo -e "${BOLD}=======================${NC}"
  echo ""

  # Read current mode
  local current_mode="pve"
  if command -v python3 &> /dev/null; then
    current_mode=$(python3 -c "import json; print(json.load(open('${DN_DAEMON_CONFIG_DIR}/config.json')).get('mode', 'pve'))" 2>/dev/null || echo "pve")
  else
    current_mode=$(grep -o '"mode"[^,]*' ${DN_DAEMON_CONFIG_DIR}/config.json | cut -d'"' -f4)
    current_mode=${current_mode:-pve}
  fi

  echo -e "Current mode: ${CYAN}${current_mode}${NC}"
  echo ""

  # Ask what to configure
  echo -e "${BOLD}What would you like to configure?${NC}"
  echo ""
  echo "  1) Full setup wizard (recommended)"
  echo "  2) Quick Redis config only"
  echo "  3) Module-specific settings only"
  echo "  4) Change daemon mode"
  echo ""
  read -p "Enter selection [1]: " config_choice
  config_choice=${config_choice:-1}

  case $config_choice in
    1)
      # Full interactive setup
      do_interactive_setup "$current_mode"
      ;;
    2)
      # Quick Redis config
      do_quick_redis_config "$current_mode"
      ;;
    3)
      # Module-specific settings
      if [ "$current_mode" = "macos" ]; then
        do_macos_module_config
      else
        do_pve_module_config
      fi
      ;;
    4)
      # Change mode
      do_change_mode
      ;;
    *)
      do_interactive_setup "$current_mode"
      ;;
  esac

  echo ""

  # Ask to start/restart
  if is_service_running; then
    read -p "Restart daemon to apply changes? (y/n) [y]: " restart_choice
    restart_choice=${restart_choice:-y}
    if [[ $restart_choice =~ ^[Yy]$ ]]; then
      do_restart
    fi
  else
    read -p "Start daemon now? (y/n) [y]: " start_choice
    start_choice=${start_choice:-y}
    if [[ $start_choice =~ ^[Yy]$ ]]; then
      do_start
    fi
  fi
}

function do_quick_redis_config() {
  local mode="$1"

  # Read current values
  local current_host current_port current_pass current_db current_node
  if command -v python3 &> /dev/null; then
    current_host=$(python3 -c "import json; print(json.load(open('${DN_DAEMON_CONFIG_DIR}/config.json'))['redis_host'])" 2>/dev/null || echo "127.0.0.1")
    current_port=$(python3 -c "import json; print(json.load(open('${DN_DAEMON_CONFIG_DIR}/config.json'))['redis_port'])" 2>/dev/null || echo "6379")
    current_pass=$(python3 -c "import json; print(json.load(open('${DN_DAEMON_CONFIG_DIR}/config.json'))['redis_password'])" 2>/dev/null || echo "")
    current_db=$(python3 -c "import json; print(json.load(open('${DN_DAEMON_CONFIG_DIR}/config.json')).get('redis_db', 0))" 2>/dev/null || echo "0")
    current_node=$(python3 -c "import json; print(json.load(open('${DN_DAEMON_CONFIG_DIR}/config.json'))['node_id'])" 2>/dev/null || echo "")
  else
    current_host=$(grep -o '"redis_host"[^,]*' ${DN_DAEMON_CONFIG_DIR}/config.json | cut -d'"' -f4)
    current_port=$(grep -o '"redis_port"[^,]*' ${DN_DAEMON_CONFIG_DIR}/config.json | grep -o '[0-9]*')
    current_pass=$(grep -o '"redis_password"[^,]*' ${DN_DAEMON_CONFIG_DIR}/config.json | cut -d'"' -f4)
    current_db="0"
    current_node=$(grep -o '"node_id"[^,]*' ${DN_DAEMON_CONFIG_DIR}/config.json | cut -d'"' -f4)
  fi

  local default_node=$(hostname | tr '[:upper:]' '[:lower:]' | tr -cd 'a-z0-9-')

  echo ""
  echo -e "${BOLD}Redis Configuration${NC}"
  echo ""

  read -p "Redis Host [${current_host}]: " redis_host
  redis_host=${redis_host:-$current_host}

  read -p "Redis Port [${current_port}]: " redis_port
  redis_port=${redis_port:-$current_port}

  read -p "Redis Password [${current_pass:-(none)}]: " redis_pass
  redis_pass=${redis_pass:-$current_pass}

  read -p "Redis Database [${current_db}]: " redis_db
  redis_db=${redis_db:-$current_db}

  read -p "Node ID [${current_node:-$default_node}]: " node_id
  node_id=${node_id:-${current_node:-$default_node}}

  # Test connection
  echo ""
  print_step "Testing Redis connection..."
  if test_redis_connection "$redis_host" "$redis_port" "$redis_pass"; then
    print_success "Redis connection successful"
  else
    print_warning "Could not verify Redis connection"
  fi

  # Write config
  cat > "${DN_DAEMON_CONFIG_DIR}/config.json" <<-EOF
{
    "mode": "${mode}",
    "redis_host": "${redis_host}",
    "redis_port": ${redis_port},
    "redis_password": "${redis_pass}",
    "redis_db": ${redis_db},
    "node_id": "${node_id}",
    "auto_update": true,
    "update_interval": 300,
    "health_reporting": true,
    "debug": false
}
EOF

  print_success "Configuration saved!"
}

function do_change_mode() {
  echo ""
  echo -e "${BOLD}Change Daemon Mode${NC}"
  echo ""
  echo -e "${YELLOW}Warning: Changing mode will reconfigure the daemon.${NC}"
  echo ""

  if [ "$OS_TYPE" = "macos" ]; then
    echo "  1) macOS - Apple Silicon VM hosting via Tart"
    echo "  2) PVE   - Proxmox VE (for testing only on macOS)"
  else
    echo "  1) PVE   - Proxmox VE KVM virtualization"
    echo "  2) macOS - Apple Silicon (requires macOS host)"
  fi
  echo ""
  read -p "Enter selection [1]: " mode_choice
  mode_choice=${mode_choice:-1}

  local new_mode
  if [ "$OS_TYPE" = "macos" ]; then
    case $mode_choice in
      2) new_mode="pve" ;;
      *) new_mode="macos" ;;
    esac
  else
    case $mode_choice in
      2) new_mode="macos" ;;
      *) new_mode="pve" ;;
    esac
  fi

  do_interactive_setup "$new_mode"
}

function do_pve_module_config() {
  echo ""
  echo -e "${BOLD}PVE Module Settings${NC}"
  echo ""

  local poll_interval vnc_enabled vnc_port ssh_enabled

  read -p "Stats poll interval (seconds) [5]: " poll_interval
  poll_interval=${poll_interval:-5}

  read -p "Enable VNC proxy? (y/n) [n]: " vnc_choice
  vnc_enabled=false
  vnc_port=5700
  if [[ $vnc_choice =~ ^[Yy]$ ]]; then
    vnc_enabled=true
    read -p "VNC proxy port [5700]: " vnc_port
    vnc_port=${vnc_port:-5700}
  fi

  read -p "Enable SSH proxy? (y/n) [n]: " ssh_choice
  ssh_enabled=false
  if [[ $ssh_choice =~ ^[Yy]$ ]]; then
    ssh_enabled=true
  fi

  local shared_secret=""
  if [ "$vnc_enabled" = true ] || [ "$ssh_enabled" = true ]; then
    shared_secret=$(generate_secret)
    echo ""
    echo -e "${YELLOW}Generated new shared secret: ${BOLD}${shared_secret}${NC}"
  fi

  # Update PVE config
  if [ -f "${DN_DAEMON_CONFIG_DIR}/pve.json" ]; then
    # Read existing values we want to preserve
    local qmp_dir="/var/run/qemu-server"
    if command -v python3 &> /dev/null; then
      qmp_dir=$(python3 -c "import json; print(json.load(open('${DN_DAEMON_CONFIG_DIR}/pve.json')).get('qmp_socket_dir', '/var/run/qemu-server'))" 2>/dev/null || echo "/var/run/qemu-server")
    fi

    cat > "${DN_DAEMON_CONFIG_DIR}/pve.json" <<-EOF
{
    "poll_interval": ${poll_interval},
    "stats_ttl": 120,
    "storage_ttl": 300,
    "qmp_socket_dir": "${qmp_dir}",
    "backup_path": "/mnt/pve/dn-backups",
    "backup_interval": 300,
    "worker_count": 4,
    "vm_timeout": 8,
    "collect_timeout": 60,
    "vnc_enabled": ${vnc_enabled},
    "vnc_port": ${vnc_port},
    "vnc_socket_dir": "${qmp_dir}",
    "vnc_auth_secret": "${shared_secret}",
    "ssh_enabled": ${ssh_enabled},
    "ssh_auth_secret": "${shared_secret}"
}
EOF
  fi

  print_success "PVE module configuration saved!"
}

function do_macos_module_config() {
  echo ""
  echo -e "${BOLD}macOS Module Settings${NC}"
  echo ""

  local poll_interval vnc_enabled vnc_port ssh_enabled guest_agent_port

  read -p "Stats poll interval (seconds) [5]: " poll_interval
  poll_interval=${poll_interval:-5}

  read -p "Guest agent port [7777]: " guest_agent_port
  guest_agent_port=${guest_agent_port:-7777}

  read -p "Enable VNC proxy? (y/n) [y]: " vnc_choice
  vnc_choice=${vnc_choice:-y}
  vnc_enabled=false
  vnc_port=5700
  if [[ $vnc_choice =~ ^[Yy]$ ]]; then
    vnc_enabled=true
    read -p "VNC proxy port [5700]: " vnc_port
    vnc_port=${vnc_port:-5700}
  fi

  read -p "Enable SSH proxy? (y/n) [y]: " ssh_choice
  ssh_choice=${ssh_choice:-y}
  ssh_enabled=false
  if [[ $ssh_choice =~ ^[Yy]$ ]]; then
    ssh_enabled=true
  fi

  local shared_secret=""
  if [ "$vnc_enabled" = true ] || [ "$ssh_enabled" = true ]; then
    shared_secret=$(generate_secret)
    echo ""
    echo -e "${YELLOW}Generated new shared secret: ${BOLD}${shared_secret}${NC}"
  fi

  # Update macOS config
  if [ -f "${DN_DAEMON_CONFIG_DIR}/macos.json" ]; then
    local tart_path="/opt/homebrew/bin/tart"
    local vm_storage="${HOME}/.tart/vms"
    if command -v python3 &> /dev/null; then
      tart_path=$(python3 -c "import json; print(json.load(open('${DN_DAEMON_CONFIG_DIR}/macos.json')).get('tart_path', '/opt/homebrew/bin/tart'))" 2>/dev/null || echo "/opt/homebrew/bin/tart")
      vm_storage=$(python3 -c "import json; print(json.load(open('${DN_DAEMON_CONFIG_DIR}/macos.json')).get('vm_storage_path', '~/.tart/vms'))" 2>/dev/null || echo "${HOME}/.tart/vms")
    fi

    cat > "${DN_DAEMON_CONFIG_DIR}/macos.json" <<-EOF
{
    "poll_interval": ${poll_interval},
    "stats_ttl": 120,
    "storage_ttl": 300,
    "tart_path": "${tart_path}",
    "vm_storage_path": "${vm_storage}",
    "worker_count": 4,
    "vm_timeout": 8,
    "collect_timeout": 60,
    "vnc_enabled": ${vnc_enabled},
    "vnc_port": ${vnc_port},
    "vnc_auth_secret": "${shared_secret}",
    "ssh_enabled": ${ssh_enabled},
    "ssh_auth_secret": "${shared_secret}",
    "job_poll_interval": 1,
    "guest_agent_port": ${guest_agent_port},
    "enable_nat": true,
    "nat_interface": "en0"
}
EOF
  fi

  print_success "macOS module configuration saved!"
}

function is_service_running() {
  if [ "$OS_TYPE" = "macos" ]; then
    launchctl list | grep -q "${DN_DAEMON_SERVICE}"
  else
    systemctl is-active --quiet ${DN_DAEMON_SERVICE}
  fi
}

function do_start() {
  check_root
  echo "Starting DN Daemon..."
  if [ "$OS_TYPE" = "macos" ]; then
    launchctl load ${DN_LAUNCHD_PLIST}
    sleep 1
    do_status
  else
    systemctl start ${DN_DAEMON_SERVICE}
    sleep 1
    systemctl status ${DN_DAEMON_SERVICE} --no-pager
  fi
}

function do_stop() {
  check_root
  echo "Stopping DN Daemon..."
  if [ "$OS_TYPE" = "macos" ]; then
    launchctl unload ${DN_LAUNCHD_PLIST}
  else
    systemctl stop ${DN_DAEMON_SERVICE}
  fi
  echo "Stopped."
}

function do_restart() {
  check_root
  echo "Restarting DN Daemon..."
  if [ "$OS_TYPE" = "macos" ]; then
    launchctl unload ${DN_LAUNCHD_PLIST} 2>/dev/null
    sleep 1
    launchctl load ${DN_LAUNCHD_PLIST}
    sleep 1
    do_status
  else
    systemctl restart ${DN_DAEMON_SERVICE}
    sleep 1
    systemctl status ${DN_DAEMON_SERVICE} --no-pager
  fi
}

function do_status() {
  if [ "$OS_TYPE" = "macos" ]; then
    if launchctl list | grep -q "${DN_DAEMON_SERVICE}"; then
      echo "DN Daemon is running"
      launchctl list | grep "${DN_DAEMON_SERVICE}"
    else
      echo "DN Daemon is not running"
    fi
  else
    systemctl status ${DN_DAEMON_SERVICE} --no-pager
  fi
}

function do_log() {
  if [ "$OS_TYPE" = "macos" ]; then
    tail -f /var/log/dn-daemon/daemon.log
  else
    journalctl -u ${DN_DAEMON_SERVICE} -f
  fi
}

function do_update() {
  check_root
  echo "Updating DN Daemon..."

  # Download new script
  echo "Updating dartnode CLI..."
  if [ "$OS_TYPE" = "macos" ]; then
    curl -sSL -o /usr/local/dartnode/dartnode.sh ${PKG_BASE}/dartnode.sh
  else
    wget -q --show-progress -O /usr/local/dartnode/dartnode.sh ${PKG_BASE}/dartnode.sh
  fi
  chmod +x /usr/local/dartnode/dartnode.sh

  # Stop service if running
  WAS_RUNNING=false
  if is_service_running; then
    WAS_RUNNING=true
    do_stop
  fi

  # Download new binary
  echo "Updating dn-daemon binary..."
  if [ "$OS_TYPE" = "macos" ]; then
    curl -sSL -o ${DN_DAEMON_BIN} ${PKG_BASE}/dn-daemon-darwin-arm64
    # Also update guest agent
    curl -sSL -o /opt/dn-daemon/dn-guest-agent ${PKG_BASE}/dn-guest-agent
    chmod +x /opt/dn-daemon/dn-guest-agent
  else
    wget -q --show-progress -O ${DN_DAEMON_BIN} ${PKG_BASE}/dn-daemon
  fi
  chmod +x ${DN_DAEMON_BIN}

  # Restart if was running
  if [ "$WAS_RUNNING" = true ]; then
    do_start
  fi

  echo ""
  echo "Update complete!"
}

function do_version() {
  if [ -x "${DN_DAEMON_BIN}" ]; then
    ${DN_DAEMON_BIN} -version

    # Also show remote version
    echo ""
    echo "Checking for latest version..."
    REMOTE_VERSION=$(curl -sS ${PKG_BASE}/version.txt 2>/dev/null)
    if [ -n "$REMOTE_VERSION" ]; then
      echo "Latest version: ${REMOTE_VERSION}"
    else
      echo "Unable to check remote version"
    fi
  else
    echo "DN Daemon is not installed."
    exit 1
  fi
}

function do_uninstall() {
  check_root
  echo "Uninstalling DN Daemon..."

  read -p "Are you sure? This will remove the daemon and config. (y/n) " -n 1 -r
  echo
  if [[ ! $REPLY =~ ^[Yy]$ ]]; then
    exit 0
  fi

  if [ "$OS_TYPE" = "macos" ]; then
    # Stop and unload LaunchDaemon
    launchctl unload ${DN_LAUNCHD_PLIST} 2>/dev/null

    # Remove files
    rm -f ${DN_DAEMON_BIN}
    rm -f /opt/dn-daemon/dn-guest-agent
    rm -f ${DN_LAUNCHD_PLIST}
    rm -rf ${DN_DAEMON_CONFIG_DIR}
    rm -f /usr/local/bin/dartnode
    rm -rf /usr/local/dartnode
    rm -rf /opt/dn-daemon
    rm -rf /var/log/dn-daemon
    # Remove completions
    rm -f /usr/local/etc/bash_completion.d/dartnode
    local zsh_comp="/opt/homebrew/share/zsh/site-functions/_dartnode"
    [ ! -f "$zsh_comp" ] && zsh_comp="/usr/local/share/zsh/site-functions/_dartnode"
    rm -f "$zsh_comp"
  else
    # Stop and disable service
    systemctl stop ${DN_DAEMON_SERVICE} 2>/dev/null
    systemctl disable ${DN_DAEMON_SERVICE} 2>/dev/null

    # Also stop legacy service if exists
    systemctl stop ${LEGACY_SERVICE} 2>/dev/null
    systemctl disable ${LEGACY_SERVICE} 2>/dev/null

    # Remove files
    rm -f ${DN_DAEMON_BIN}
    rm -f ${LEGACY_BIN}
    rm -f /etc/systemd/system/${DN_DAEMON_SERVICE}.service
    rm -f /etc/systemd/system/${LEGACY_SERVICE}.service
    rm -rf ${DN_DAEMON_CONFIG_DIR}
    rm -rf ${LEGACY_CONFIG_DIR}
    rm -f /usr/local/bin/dartnode
    rm -f /bin/dartnode
    rm -rf /usr/local/dartnode
    rm -f /etc/bash_completion.d/dartnode

    systemctl daemon-reload
  fi

  echo "DN Daemon uninstalled."
}

# ============================================
# Shared Infrastructure Helpers
# ============================================

function read_config_value() {
  local key="$1"
  local file="${2:-${DN_DAEMON_CONFIG_DIR}/config.json}"
  [ ! -f "$file" ] && return 1
  if command -v python3 &> /dev/null; then
    python3 -c "
import json, sys
v = json.load(open('${file}')).get('${key}', '')
print('' if v is None else v)
" 2>/dev/null
  else
    grep -o "\"${key}\"[[:space:]]*:[[:space:]]*[^,}]*" "$file" | head -1 | sed 's/.*:[[:space:]]*//' | tr -d '"' | tr -d ' '
  fi
}

function redis_cmd() {
  if ! command -v redis-cli &> /dev/null; then
    echo "redis-cli not found" >&2
    return 1
  fi
  local rh rp ra rd
  rh=$(read_config_value "redis_host"); rh=${rh:-127.0.0.1}
  rp=$(read_config_value "redis_port"); rp=${rp:-6379}
  ra=$(read_config_value "redis_password")
  rd=$(read_config_value "redis_db"); rd=${rd:-0}
  local auth=()
  [ -n "$ra" ] && auth=(-a "$ra")
  redis-cli -h "$rh" -p "$rp" -n "$rd" "${auth[@]}" --no-auth-warning "$@" 2>/dev/null
}

function get_daemon_mode() {
  local m; m=$(read_config_value "mode"); echo "${m:-pve}"
}

function get_node_id() {
  local n; n=$(read_config_value "node_id"); echo "${n:-$(hostname)}"
}

function format_speed() {
  local speed="$1"
  if [ -z "$speed" ] || ! [[ "$speed" =~ ^[0-9]+$ ]] || [ "$speed" -le 0 ] 2>/dev/null; then
    echo "-"; return
  fi
  if [ "$speed" -ge 1000 ]; then
    local whole=$((speed / 1000))
    local frac=$(( (speed % 1000) / 100 ))
    if [ "$frac" -eq 0 ]; then
      echo "${whole}G"
    else
      echo "${whole}.${frac}G"
    fi
  else
    echo "${speed}M"
  fi
}

# ============================================
# Interactive Pickers
# ============================================

# Pick a network interface from a numbered list
# Usage: local iface=$(_pick_iface "Select interface")
function _pick_iface() {
  local prompt="${1:-Select interface}"
  local ifaces=()

  for ifdir in /sys/class/net/*; do
    local name=$(basename "$ifdir")
    [ "$name" = "lo" ] && continue
    ifaces+=("$name")
  done

  if [ ${#ifaces[@]} -eq 0 ]; then
    print_error "No network interfaces found"
    return 1
  fi

  echo ""
  echo -e "  ${BOLD}${prompt}:${NC}"
  for i in "${!ifaces[@]}"; do
    local name="${ifaces[$i]}"
    local state=$(cat "/sys/class/net/${name}/operstate" 2>/dev/null || echo "unknown")
    local speed=$(cat "/sys/class/net/${name}/speed" 2>/dev/null || echo "")
    local mac=$(cat "/sys/class/net/${name}/address" 2>/dev/null || echo "")
    local info="${state}"
    [ -n "$speed" ] && [ "$speed" != "-1" ] && info+=" $(format_speed "$speed")"
    [ -n "$mac" ] && info+=" ${mac}"
    printf "   %2d) %-16s  %s\n" "$((i+1))" "$name" "$info"
  done
  echo ""
  read -p "  Enter number [1]: " choice
  choice=${choice:-1}

  if [ "$choice" -ge 1 ] && [ "$choice" -le ${#ifaces[@]} ] 2>/dev/null; then
    echo "${ifaces[$((choice-1))]}"
  else
    print_error "Invalid selection"
    return 1
  fi
}

# Pick a block device from a numbered list
# Usage: local dev=$(_pick_blockdev "Select drive")
function _pick_blockdev() {
  local prompt="${1:-Select drive}"
  local devs=()
  local labels=()

  for bd in /sys/block/sd* /sys/block/nvme*; do
    [ -e "$bd" ] || continue
    local name=$(basename "$bd")
    devs+=("/dev/$name")
    local model=$(cat "$bd/device/model" 2>/dev/null | xargs)
    local size_sectors=$(cat "$bd/size" 2>/dev/null)
    local size_gb=""
    if [ -n "$size_sectors" ] && [ "$size_sectors" -gt 0 ] 2>/dev/null; then
      size_gb="$(( size_sectors * 512 / 1073741824 ))G"
    fi
    local serial=$(cat "$bd/device/serial" 2>/dev/null | xargs)
    labels+=("${size_gb:-?}  ${model:-unknown}  ${serial}")
  done

  if [ ${#devs[@]} -eq 0 ]; then
    print_error "No block devices found"
    return 1
  fi

  echo ""
  echo -e "  ${BOLD}${prompt}:${NC}"
  for i in "${!devs[@]}"; do
    printf "   %2d) %-12s  %s\n" "$((i+1))" "${devs[$i]}" "${labels[$i]}"
  done
  echo ""
  read -p "  Enter number [1]: " choice
  choice=${choice:-1}

  if [ "$choice" -ge 1 ] && [ "$choice" -le ${#devs[@]} ] 2>/dev/null; then
    echo "${devs[$((choice-1))]}"
  else
    print_error "Invalid selection"
    return 1
  fi
}

# ============================================
# Network Commands (dartnode net)
# ============================================

function net_show_linux() {
  echo ""
  echo -e "  ${BOLD}Network Interfaces${NC}"
  echo ""

  # Collect IPv4 addresses
  declare -A iface_ips
  while IFS= read -r line; do
    local if_name if_ip
    if_name=$(echo "$line" | awk '{print $2}')
    if_ip=$(echo "$line" | awk '{print $4}' | cut -d/ -f1)
    if [ -n "${iface_ips[$if_name]+x}" ]; then
      iface_ips[$if_name]="${iface_ips[$if_name]}, $if_ip"
    else
      iface_ips[$if_name]="$if_ip"
    fi
  done < <(ip -o -4 addr show 2>/dev/null)

  # Collect bridge membership
  declare -A iface_bridge
  for brpath in /sys/class/net/*/bridge/; do
    [ -d "$brpath" ] || continue
    local br=$(basename "$(dirname "$brpath")")
    for portdir in /sys/class/net/"$br"/brif/*/; do
      [ -d "$portdir" ] || continue
      local port=$(basename "$portdir")
      iface_bridge[$port]="$br"
    done
  done

  # Header
  printf "  ${DIM}%-14s %-12s %-7s %-6s %-19s %s${NC}\n" "Interface" "Status" "Speed" "MTU" "MAC" "IP Address"
  printf "  ${DIM}%-14s %-12s %-7s %-6s %-19s %s${NC}\n" "──────────────" "────────────" "───────" "──────" "───────────────────" "───────────────"

  local printed_bridges=()

  for iface_path in /sys/class/net/*/; do
    local iface=$(basename "$iface_path")
    [ "$iface" = "lo" ] && continue
    # Skip bridge member ports (shown under their bridge)
    [ -n "${iface_bridge[$iface]+x}" ] && continue
    # Skip bridges (shown under their physical ports)
    [ -d "$iface_path/bridge" ] && continue

    local operstate carrier speed_raw mtu mac ip
    operstate=$(cat "$iface_path/operstate" 2>/dev/null || echo "unknown")
    carrier=$(cat "$iface_path/carrier" 2>/dev/null || echo "0")
    speed_raw=$(cat "$iface_path/speed" 2>/dev/null || echo "")
    mtu=$(cat "$iface_path/mtu" 2>/dev/null || echo "-")
    mac=$(cat "$iface_path/address" 2>/dev/null || echo "-")
    ip="${iface_ips[$iface]:-"-"}"
    local speed_str=$(format_speed "$speed_raw")

    local status_vis status_col
    if [ "$operstate" = "up" ]; then
      if [ "$carrier" = "1" ]; then
        status_vis="● UP"; status_col="${GREEN}"
      else
        status_vis="● NO-LINK"; status_col="${YELLOW}"
      fi
    else
      status_vis="○ DOWN"; status_col="${RED}"
    fi

    local ps=$(printf "%-12s" "$status_vis")
    printf "  %-14s ${status_col}%s${NC} %-7s %-6s %-19s %s\n" \
      "$iface" "$ps" "$speed_str" "$mtu" "$mac" "$ip"

    # Show bridges that use this physical interface
    if [ -d "$iface_path/device" ]; then
      for brdir in /sys/class/net/*/bridge/; do
        [ -d "$brdir" ] || continue
        local br=$(basename "$(dirname "$brdir")")
        [ -d "/sys/class/net/$br/brif/$iface" ] || continue
        local bo bm bma bi bsv bsc
        bo=$(cat "/sys/class/net/$br/operstate" 2>/dev/null || echo "unknown")
        bm=$(cat "/sys/class/net/$br/mtu" 2>/dev/null || echo "-")
        bma=$(cat "/sys/class/net/$br/address" 2>/dev/null || echo "-")
        bi="${iface_ips[$br]:-"-"}"
        if [ "$bo" = "up" ]; then
          bsv="● UP"; bsc="${GREEN}"
        else
          bsv="○ DOWN"; bsc="${RED}"
        fi
        local bps=$(printf "%-12s" "$bsv")
        printf "  ${DIM}├─${NC} %-11s ${bsc}%s${NC} %-7s %-6s %-19s %s\n" \
          "$br" "$bps" "-" "$bm" "$bma" "$bi"
        printed_bridges+=("$br")
      done
    fi
  done

  # Standalone bridges (not tied to a physical port we iterated)
  for brdir in /sys/class/net/*/bridge/; do
    [ -d "$brdir" ] || continue
    local br=$(basename "$(dirname "$brdir")")
    local skip=false
    for pb in "${printed_bridges[@]}"; do
      [ "$pb" = "$br" ] && skip=true && break
    done
    [ "$skip" = true ] && continue

    local bo bm bma bi bsv bsc
    bo=$(cat "/sys/class/net/$br/operstate" 2>/dev/null || echo "unknown")
    bm=$(cat "/sys/class/net/$br/mtu" 2>/dev/null || echo "-")
    bma=$(cat "/sys/class/net/$br/address" 2>/dev/null || echo "-")
    bi="${iface_ips[$br]:-"-"}"
    if [ "$bo" = "up" ]; then
      bsv="● UP"; bsc="${GREEN}"
    else
      bsv="○ DOWN"; bsc="${RED}"
    fi
    local bps=$(printf "%-12s" "$bsv")
    printf "  %-14s ${bsc}%s${NC} %-7s %-6s %-19s %s\n" \
      "$br" "$bps" "-" "$bm" "$bma" "$bi"
  done

  echo ""
}

function net_show_macos() {
  echo ""
  echo -e "  ${BOLD}Network Interfaces${NC}"
  echo ""

  printf "  ${DIM}%-14s %-12s %-19s %s${NC}\n" "Interface" "Status" "MAC" "IP Address"
  printf "  ${DIM}%-14s %-12s %-19s %s${NC}\n" "──────────────" "────────────" "───────────────────" "───────────────"

  local current_dev=""
  while IFS= read -r line; do
    if echo "$line" | grep -q "^Device:"; then
      current_dev=$(echo "$line" | sed 's/Device: //')
      local status ip mac sc
      if ifconfig "$current_dev" 2>/dev/null | grep -q "status: active"; then
        status="● UP"; sc="${GREEN}"
      else
        status="○ DOWN"; sc="${RED}"
      fi
      mac=$(ifconfig "$current_dev" 2>/dev/null | grep ether | awk '{print $2}')
      ip=$(ifconfig "$current_dev" 2>/dev/null | grep "inet " | awk '{print $2}' | head -1)
      mac=${mac:-"-"}
      ip=${ip:-"-"}
      local ps=$(printf "%-12s" "$status")
      printf "  %-14s ${sc}%s${NC} %-19s %s\n" "$current_dev" "$ps" "$mac" "$ip"
    fi
  done < <(networksetup -listallhardwareports 2>/dev/null)

  echo ""
}

function net_show() {
  if [ "$OS_TYPE" = "macos" ]; then
    net_show_macos
  else
    net_show_linux
  fi
}

function net_scan() {
  if [ "$OS_TYPE" = "macos" ]; then
    echo "Port discovery is only available on Linux."
    return 1
  fi

  if ! command -v lspci &> /dev/null; then
    print_error "lspci not found. Install pciutils: apt install pciutils"
    return 1
  fi

  echo ""
  echo -e "  ${BOLD}Physical Port Discovery${NC}"
  echo ""

  # Find all ethernet controllers on PCI bus
  local -a pci_addrs pci_descs iface_names iface_macs iface_links iface_speeds
  local count=0

  while IFS= read -r line; do
    [ -z "$line" ] && continue
    local pci_addr=$(echo "$line" | awk '{print $1}')
    local pci_desc=$(echo "$line" | sed "s/^${pci_addr} //; s/Ethernet controller: //")

    local iface="-"
    local mac="-"
    local link="NO"
    local speed="-"

    # Find interface for this PCI device
    if [ -d "/sys/bus/pci/devices/${pci_addr}/net" ]; then
      for netdir in /sys/bus/pci/devices/${pci_addr}/net/*/; do
        [ -d "$netdir" ] || continue
        iface=$(basename "$netdir")
        break
      done
    fi

    if [ "$iface" != "-" ]; then
      mac=$(cat "/sys/class/net/${iface}/address" 2>/dev/null || echo "-")

      # Check link - temporarily bring up if needed
      local was_down=false
      local operstate=$(cat "/sys/class/net/${iface}/operstate" 2>/dev/null || echo "down")
      if [ "$operstate" = "down" ]; then
        was_down=true
        ip link set "$iface" up 2>/dev/null
        sleep 1
      fi

      local carrier=$(cat "/sys/class/net/${iface}/carrier" 2>/dev/null || echo "0")
      if [ "$carrier" = "1" ]; then
        link="YES"
        local spd=$(cat "/sys/class/net/${iface}/speed" 2>/dev/null || echo "")
        speed=$(format_speed "$spd")
      fi

      # Restore original state
      if [ "$was_down" = true ]; then
        ip link set "$iface" down 2>/dev/null
      fi
    fi

    pci_addrs+=("$pci_addr")
    pci_descs+=("$pci_desc")
    iface_names+=("$iface")
    iface_macs+=("$mac")
    iface_links+=("$link")
    iface_speeds+=("$speed")
    ((count++))
  done < <(lspci -D 2>/dev/null | grep -i ethernet)

  if [ $count -eq 0 ]; then
    print_warning "No ethernet controllers found on PCI bus."
    return
  fi

  echo -e "  Found ${BOLD}${count}${NC} ethernet controller(s) on PCI bus:"
  echo ""

  # Print table
  printf "  ${DIM}%-14s %-12s %-32s %-19s %-8s %s${NC}\n" \
    "PCI Address" "Interface" "Device" "MAC" "Link" "Speed"
  printf "  ${DIM}%-14s %-12s %-32s %-19s %-8s %s${NC}\n" \
    "──────────────" "────────────" "────────────────────────────────" "───────────────────" "────────" "─────"

  for ((i=0; i<count; i++)); do
    local link_col link_vis
    if [ "${iface_links[$i]}" = "YES" ]; then
      link_vis="● YES"; link_col="${GREEN}"
    else
      link_vis="○ NO"; link_col="${RED}"
    fi
    local lps=$(printf "%-8s" "$link_vis")

    # Truncate device description to 30 chars
    local desc="${pci_descs[$i]}"
    [ ${#desc} -gt 30 ] && desc="${desc:0:29}…"

    printf "  %-14s %-12s %-32s %-19s ${link_col}%s${NC} %s\n" \
      "${pci_addrs[$i]}" "${iface_names[$i]}" "$desc" "${iface_macs[$i]}" "$lps" "${iface_speeds[$i]}"
  done

  echo ""

  # Interactive setup wizard
  net_scan_setup "$count"
}

function net_scan_setup() {
  local count="$1"

  # Access pci_addrs, iface_names, pci_descs, etc. from caller's scope (dynamic scoping)
  while true; do
    read -p "  Configure a port? Enter number (1-${count}) or 'q' to quit: " choice
    [ "$choice" = "q" ] || [ "$choice" = "Q" ] || [ -z "$choice" ] && break

    if ! [[ "$choice" =~ ^[0-9]+$ ]] || [ "$choice" -lt 1 ] || [ "$choice" -gt "$count" ]; then
      echo "  Invalid selection."
      continue
    fi

    local idx=$((choice - 1))
    local iface="${iface_names[$idx]}"

    if [ "$iface" = "-" ]; then
      print_error "No interface mapped to this PCI device. It may need a driver."
      continue
    fi

    echo ""
    echo -e "  ${DIM}───${NC} ${BOLD}${iface}${NC} (${pci_descs[$idx]}) ${DIM}───${NC}"
    echo ""
    echo "  1) Identify (blink LED)      5) Add to bridge"
    echo "  2) Bring UP                  6) Set MTU"
    echo "  3) Bring DOWN                7) Write to /etc/network/interfaces"
    echo "  4) Assign IP"
    echo ""

    while true; do
      read -p "  Action (or 'b' to go back): " action
      [ "$action" = "b" ] || [ "$action" = "B" ] && break

      case "$action" in
        1)
          if command -v ethtool &> /dev/null; then
            echo -n "  Blinking LED on ${iface} for 5 seconds..."
            ethtool --identify "$iface" 5 2>/dev/null
            echo " done"
            print_success "LED blink complete"
          else
            print_error "ethtool not found. Install: apt install ethtool"
          fi
          ;;
        2)
          ip link set "$iface" up 2>/dev/null
          sleep 1
          local carrier=$(cat "/sys/class/net/${iface}/carrier" 2>/dev/null || echo "0")
          print_success "${iface} is now UP"
          if [ "$carrier" = "1" ]; then
            local spd=$(cat "/sys/class/net/${iface}/speed" 2>/dev/null || echo "")
            local spd_fmt=$(format_speed "$spd")
            print_success "Link detected: yes (${spd_fmt})"
          else
            print_warning "Link detected: no"
          fi
          ;;
        3)
          ip link set "$iface" down 2>/dev/null
          print_success "${iface} is now DOWN"
          ;;
        4)
          read -p "  IP Address (CIDR, e.g. 10.0.0.5/24): " new_ip
          if [ -n "$new_ip" ]; then
            if ip addr add "$new_ip" dev "$iface" 2>/dev/null; then
              print_success "IP ${new_ip} assigned to ${iface}"
            else
              print_error "Failed to assign IP"
            fi
          fi
          ;;
        5)
          # Bridge management
          echo ""
          echo -e "  ${BOLD}Existing bridges:${NC}"
          local has_bridges=false
          for brdir in /sys/class/net/*/bridge/; do
            [ -d "$brdir" ] || continue
            has_bridges=true
            local br=$(basename "$(dirname "$brdir")")
            local br_ip
            br_ip=$(ip -o -4 addr show "$br" 2>/dev/null | awk '{print $4}' | cut -d/ -f1 | head -1)
            echo "    ${br} (${br_ip:-no IP})"
          done
          [ "$has_bridges" = false ] && echo "    (none)"
          echo ""

          read -p "  Add to existing bridge or create new? [existing/new]: " br_choice
          if [ "$br_choice" = "new" ]; then
            read -p "  Bridge name [vmbr2]: " br_name
            br_name=${br_name:-vmbr2}

            ip link add name "$br_name" type bridge 2>/dev/null
            ip link set "$br_name" up 2>/dev/null
            print_success "Bridge ${br_name} created"

            ip link set "$iface" master "$br_name" 2>/dev/null
            print_success "${iface} added to ${br_name}"

            local br_ip=""
            read -p "  Assign IP to bridge? (y/n) [n]: " assign_ip
            if [[ "$assign_ip" =~ ^[Yy]$ ]]; then
              read -p "  IP Address (CIDR): " br_ip
              if [ -n "$br_ip" ]; then
                ip addr add "$br_ip" dev "$br_name" 2>/dev/null
                print_success "IP ${br_ip} assigned to ${br_name}"
              fi
            fi

            _net_write_bridge_config "$iface" "$br_name" "$br_ip"
          else
            read -p "  Bridge name: " br_name
            if [ -n "$br_name" ] && [ -d "/sys/class/net/${br_name}/bridge" ]; then
              ip link set "$iface" master "$br_name" 2>/dev/null
              print_success "${iface} added to ${br_name}"
              _net_write_iface_config "$iface"
            else
              print_error "Bridge ${br_name} not found"
            fi
          fi
          ;;
        6)
          read -p "  MTU value [9000]: " new_mtu
          new_mtu=${new_mtu:-9000}
          if ip link set "$iface" mtu "$new_mtu" 2>/dev/null; then
            print_success "MTU set to ${new_mtu} on ${iface}"
          else
            print_error "Failed to set MTU"
          fi
          ;;
        7)
          _net_write_iface_config "$iface"
          ;;
        *)
          echo "  Invalid action."
          ;;
      esac
      echo ""
    done
    echo ""
  done
}

function _net_write_bridge_config() {
  local iface="$1"
  local bridge="$2"
  local bridge_ip="$3"

  read -p "  Write to /etc/network/interfaces? (y/n) [n]: " do_write
  [[ ! "$do_write" =~ ^[Yy]$ ]] && return

  local datestamp=$(date +%Y-%m-%d)
  local config=""
  config+="# Added by dartnode net scan - ${datestamp}\n"
  config+="auto ${iface}\n"
  config+="iface ${iface} inet manual\n"
  config+="\n"
  config+="auto ${bridge}\n"
  if [ -n "$bridge_ip" ]; then
    config+="iface ${bridge} inet static\n"
    config+="    address ${bridge_ip}\n"
  else
    config+="iface ${bridge} inet manual\n"
  fi
  config+="    bridge-ports ${iface}\n"
  config+="    bridge-stp off\n"
  config+="    bridge-fd 0\n"

  echo ""
  echo -e "  ${DIM}Preview:${NC}"
  printf "  %b" "$config" | sed 's/^/  /'
  echo ""

  read -p "  Confirm write? (y/n) [n]: " confirm
  if [[ "$confirm" =~ ^[Yy]$ ]]; then
    printf "\n%b" "$config" >> /etc/network/interfaces
    print_success "Configuration written to /etc/network/interfaces"
  fi
}

function _net_write_iface_config() {
  local iface="$1"

  read -p "  Write ${iface} to /etc/network/interfaces? (y/n) [n]: " do_write
  [[ ! "$do_write" =~ ^[Yy]$ ]] && return

  local datestamp=$(date +%Y-%m-%d)
  local config=""
  config+="# Added by dartnode net scan - ${datestamp}\n"
  config+="auto ${iface}\n"
  config+="iface ${iface} inet manual\n"

  echo ""
  echo -e "  ${DIM}Preview:${NC}"
  printf "  %b" "$config" | sed 's/^/  /'
  echo ""

  read -p "  Confirm write? (y/n) [n]: " confirm
  if [[ "$confirm" =~ ^[Yy]$ ]]; then
    printf "\n%b" "$config" >> /etc/network/interfaces
    print_success "Configuration written to /etc/network/interfaces"
  fi
}

function net_identify() {
  local iface="$1"
  if [ -z "$iface" ]; then
    iface=$(_pick_iface "Select interface to blink") || return 1
  fi
  if ! command -v ethtool &> /dev/null; then
    print_error "ethtool not found. Install: apt install ethtool"
    return 1
  fi
  if [ ! -d "/sys/class/net/${iface}" ]; then
    print_error "Interface ${iface} not found"
    return 1
  fi

  echo ""
  echo -e "  Blinking LED on ${BOLD}${iface}${NC}..."
  echo -e "  ${YELLOW}Press Ctrl+C to stop.${NC}"
  echo ""

  trap "echo ''; echo '  Stopping LED blink on ${iface}...'; print_success 'Stopped'; trap - SIGINT SIGTERM; return" SIGINT SIGTERM

  while true; do
    ethtool --identify "$iface" 5 2>/dev/null
  done
}

function net_watch() {
  echo -e "${BOLD}Network Monitor${NC} (Ctrl+C to exit)"
  while true; do
    clear
    echo -e "${BOLD}Network Monitor${NC} — $(date '+%H:%M:%S')"
    net_show
    sleep 1
  done
}

function net_up() {
  local iface="$1"
  if [ -z "$iface" ]; then
    iface=$(_pick_iface "Select interface to bring UP") || return 1
  fi
  check_root
  ip link set "$iface" up 2>/dev/null
  print_success "${iface} is now UP"
}

function net_down() {
  local iface="$1"
  if [ -z "$iface" ]; then
    iface=$(_pick_iface "Select interface to bring DOWN") || return 1
  fi
  check_root
  ip link set "$iface" down 2>/dev/null
  print_success "${iface} is now DOWN"
}

function do_net() {
  local subcmd="${ARGS[0]:-}"
  local arg="${ARGS[1]:-}"

  # Direct subcommand access for scripting
  case "$subcmd" in
    show)
      net_show; return ;;
    scan)
      check_root; net_scan; return ;;
    identify)
      check_root; net_identify "$arg"; return ;;
    watch)
      net_watch; return ;;
    up)
      net_up "$arg"; return ;;
    down)
      net_down "$arg"; return ;;
  esac

  # Interactive menu
  local node_id=$(get_node_id)
  local mode=$(get_daemon_mode)

  echo ""
  echo -e "  ${BOLD}Network Tools${NC} — ${node_id} (${mode})"
  echo ""
  echo "   1) Show interfaces           — status, speed, MAC, IPs"
  echo "   2) Physical port discovery    — scan PCI bus for all NICs"
  echo "   3) Identify port (blink LED)  — ethtool blink on a NIC"
  echo "   4) Live monitor               — auto-refreshing display"
  echo "   5) Bring interface UP"
  echo "   6) Bring interface DOWN"
  echo ""
  read -p "  Enter selection [1]: " net_choice
  net_choice=${net_choice:-1}

  case $net_choice in
    1)
      net_show
      ;;
    2)
      check_root
      net_scan
      ;;
    3)
      check_root
      net_identify
      ;;
    4)
      net_watch
      ;;
    5)
      net_up
      ;;
    6)
      net_down
      ;;
    *)
      echo "  Invalid selection."
      ;;
  esac
}

# ============================================
# Drive Tools (dartnode disk)
# ============================================

function disk_list() {
  echo ""
  echo -e "  ${BOLD}Block Devices${NC}"
  echo ""

  if command -v lsblk &>/dev/null; then
    lsblk -d -o NAME,SIZE,MODEL,SERIAL,ROTA,TRAN,STATE 2>/dev/null | while IFS= read -r line; do
      echo "  $line"
    done
  else
    # Fallback to sysfs
    for dev in /sys/block/sd* /sys/block/nvme*; do
      [ -e "$dev" ] || continue
      local name=$(basename "$dev")
      local model=$(cat "$dev/device/model" 2>/dev/null | xargs)
      local size_sectors=$(cat "$dev/size" 2>/dev/null)
      local size_gb=""
      if [ -n "$size_sectors" ] && [ "$size_sectors" -gt 0 ] 2>/dev/null; then
        size_gb="$(( size_sectors * 512 / 1073741824 ))G"
      fi
      local serial=$(cat "$dev/device/serial" 2>/dev/null | xargs)
      printf "  %-10s %8s  %-30s %s\n" "$name" "$size_gb" "$model" "$serial"
    done
  fi
  echo ""
}

function disk_smart_quick() {
  if ! command -v smartctl &>/dev/null; then
    print_error "smartctl not found. Install: apt install smartmontools"
    return 1
  fi

  check_root
  echo ""
  echo -e "  ${BOLD}SMART Health Summary${NC}"
  echo ""

  for dev in /dev/sd? /dev/nvme?n1; do
    [ -b "$dev" ] || continue
    local name=$(basename "$dev")
    local result=$(smartctl -H "$dev" 2>/dev/null | grep -i "overall\|result" | head -1)

    if echo "$result" | grep -qi "PASSED\|OK"; then
      echo -e "  ${GREEN}PASS${NC}  $name  — $result"
    elif [ -n "$result" ]; then
      echo -e "  ${RED}FAIL${NC}  $name  — $result"
    else
      echo -e "  ${YELLOW} ?? ${NC}  $name  — unable to read SMART data"
    fi
  done
  echo ""
}

function _led_locate_on() {
  local dev="$1"
  if command -v ledctl &>/dev/null; then
    ledctl locate="$dev" 2>/dev/null
    return $?
  fi
  # Fallback: sysfs enclosure locate
  local sysdev=$(basename "$dev")
  for slot in /sys/class/enclosure/*/*/device/block/"$sysdev"; do
    if [ -e "$slot" ]; then
      local slot_dir=$(dirname $(dirname $(dirname "$slot")))
      if [ -f "$slot_dir/locate" ]; then
        echo 1 > "$slot_dir/locate" 2>/dev/null
        return 0
      fi
    fi
  done
  return 1
}

function _led_locate_off() {
  local dev="$1"
  if command -v ledctl &>/dev/null; then
    ledctl locate_off="$dev" 2>/dev/null
    return $?
  fi
  local sysdev=$(basename "$dev")
  for slot in /sys/class/enclosure/*/*/device/block/"$sysdev"; do
    if [ -e "$slot" ]; then
      local slot_dir=$(dirname $(dirname $(dirname "$slot")))
      if [ -f "$slot_dir/locate" ]; then
        echo 0 > "$slot_dir/locate" 2>/dev/null
        return 0
      fi
    fi
  done
  return 1
}

function _check_led_support() {
  if command -v ledctl &>/dev/null; then
    return 0
  fi
  # Check for sysfs enclosure support
  for enc in /sys/class/enclosure/*/; do
    [ -d "$enc" ] && return 0
  done
  print_error "No LED control available (ledctl not found, no enclosure sysfs)"
  echo "  Install ledctl: apt install ledmon"
  return 1
}

function disk_led_on() {
  local dev="$1"
  if [ -z "$dev" ]; then
    dev=$(_pick_blockdev "Select drive to blink") || return 1
  fi
  if [ -z "$dev" ]; then
    print_error "No device specified"
    return 1
  fi

  check_root
  _check_led_support || return 1

  echo ""
  echo -e "  Enabling locate LED on ${BOLD}${dev}${NC}..."

  if _led_locate_on "$dev"; then
    print_success "Locate LED active on ${dev}"
  else
    print_error "Failed to activate LED on ${dev}"
    return 1
  fi

  echo ""
  echo -e "  ${YELLOW}Press Ctrl+C to turn off LED and exit.${NC}"
  echo ""

  trap "echo ''; echo '  Turning off locate LED on ${dev}...'; _led_locate_off '${dev}'; print_success 'LED off'; trap - SIGINT SIGTERM; return" SIGINT SIGTERM

  while true; do sleep 1; done
}

function disk_led_off() {
  local dev="$1"
  if [ -z "$dev" ]; then
    dev=$(_pick_blockdev "Select drive to stop blinking") || return 1
  fi
  if [ -z "$dev" ]; then
    print_error "No device specified"
    return 1
  fi

  check_root
  _check_led_support || return 1

  echo "  Turning off locate LED on ${dev}..."
  if _led_locate_off "$dev"; then
    print_success "Locate LED off on ${dev}"
  else
    print_error "Failed to turn off LED on ${dev}"
  fi
}

function disk_enclosure_list() {
  echo ""
  echo -e "  ${BOLD}Enclosure Slots${NC}"
  echo ""

  local found=false
  for enc in /sys/class/enclosure/*/; do
    [ -d "$enc" ] || continue
    found=true
    local enc_name=$(basename "$enc")
    echo -e "  ${BOLD}${enc_name}${NC}"

    for slot in "$enc"/Slot\ * "$enc"/ArrayDevice* "$enc"/Disk\ *; do
      [ -d "$slot" ] || continue
      local slot_name=$(basename "$slot")
      local status=$(cat "$slot/status" 2>/dev/null || echo "unknown")
      local locate=$(cat "$slot/locate" 2>/dev/null || echo "?")
      local fault=$(cat "$slot/fault" 2>/dev/null || echo "?")

      # Find which block device is in this slot
      local blkdev=""
      for bd in "$slot"/device/block/*; do
        [ -e "$bd" ] && blkdev=$(basename "$bd")
      done

      printf "    %-20s  status=%-10s  locate=%s  fault=%s" "$slot_name" "$status" "$locate" "$fault"
      [ -n "$blkdev" ] && printf "  dev=/dev/%s" "$blkdev"
      echo ""
    done
    echo ""
  done

  if [ "$found" = false ]; then
    print_warning "No SES enclosures found in /sys/class/enclosure/"
    echo "  This server may not have a SAS/SATA backplane with SES support."
    echo "  Try 'ledctl' (option 3) for direct AHCI/NVMe LED control instead."
  fi
}

function disk_enclosure_led() {
  check_root

  # Show available enclosures
  local enc_list=()
  for enc in /sys/class/enclosure/*/; do
    [ -d "$enc" ] && enc_list+=("$enc")
  done

  if [ ${#enc_list[@]} -eq 0 ]; then
    # Try sg_ses as alternative
    if ! command -v sg_ses &>/dev/null; then
      print_error "No enclosures found and sg_ses not available"
      echo "  Install sg3-utils: apt install sg3-utils"
      return 1
    fi

    echo "  No sysfs enclosures, trying sg_ses..."
    local sg_devs=()
    for sg in /dev/sg*; do
      if sg_ses -p 0x01 "$sg" &>/dev/null 2>&1; then
        sg_devs+=("$sg")
      fi
    done

    if [ ${#sg_devs[@]} -eq 0 ]; then
      print_error "No SES-capable devices found"
      return 1
    fi

    echo "  SES devices found: ${sg_devs[*]}"
    local sg_dev=""
    read -p "  SES device [${sg_devs[0]}]: " sg_dev
    sg_dev=${sg_dev:-${sg_devs[0]}}

    local slot_num=""
    read -p "  Slot number to identify: " slot_num
    if [ -z "$slot_num" ]; then
      print_error "No slot number specified"
      return 1
    fi

    echo "  Activating identify LED on slot ${slot_num}..."
    if sg_ses --index="$slot_num" --set=ident "$sg_dev" 2>/dev/null; then
      print_success "Identify LED active on slot ${slot_num} (${sg_dev})"
    else
      print_error "sg_ses failed to set identify LED"
      return 1
    fi

    echo ""
    echo -e "  ${YELLOW}Press Ctrl+C to turn off LED and exit.${NC}"
    echo ""

    trap "echo ''; echo '  Turning off identify LED on slot ${slot_num}...'; sg_ses --index='${slot_num}' --clear=ident '${sg_dev}' 2>/dev/null; print_success 'LED off'; trap - SIGINT SIGTERM; return" SIGINT SIGTERM

    while true; do sleep 1; done
  fi

  # Use sysfs locate
  echo ""
  disk_enclosure_list

  local slot_path=""
  read -p "  Slot name (e.g. 'Slot 00'): " slot_input
  if [ -z "$slot_input" ]; then
    print_error "No slot specified"
    return 1
  fi

  for enc in "${enc_list[@]}"; do
    local candidate="${enc}${slot_input}"
    if [ -d "$candidate" ] && [ -f "$candidate/locate" ]; then
      slot_path="$candidate/locate"
      break
    fi
  done

  if [ -z "$slot_path" ]; then
    print_error "Slot '${slot_input}' not found in any enclosure"
    return 1
  fi

  echo 1 > "$slot_path" 2>/dev/null
  print_success "Locate LED active on ${slot_input}"
  echo ""
  echo -e "  ${YELLOW}Press Ctrl+C to turn off LED and exit.${NC}"
  echo ""

  trap "echo ''; echo '  Turning off locate LED on ${slot_input}...'; echo 0 > '${slot_path}' 2>/dev/null; print_success 'LED off'; trap - SIGINT SIGTERM; return" SIGINT SIGTERM

  while true; do sleep 1; done
}

function disk_blink_array() {
  check_root

  if ! command -v mdadm &>/dev/null; then
    print_error "mdadm not found. Install: apt install mdadm"
    return 1
  fi

  _check_led_support || return 1

  # Discover md arrays
  local arrays=()
  for md in /dev/md*; do
    [ -b "$md" ] || continue
    arrays+=("$md")
  done

  if [ ${#arrays[@]} -eq 0 ]; then
    print_error "No MD RAID arrays found on this system"
    return 1
  fi

  local md_dev=""
  if [ ${#arrays[@]} -eq 1 ]; then
    md_dev="${arrays[0]}"
    echo "  Found array: ${md_dev}"
  else
    echo ""
    echo -e "  ${BOLD}Available arrays:${NC}"
    for i in "${!arrays[@]}"; do
      local detail=$(mdadm --detail "${arrays[$i]}" 2>/dev/null | grep "Raid Level" | awk '{print $NF}')
      local count=$(mdadm --detail "${arrays[$i]}" 2>/dev/null | grep "Active Devices" | awk '{print $NF}')
      echo "   $((i+1))) ${arrays[$i]}  (${detail:-unknown}, ${count:-?} active)"
    done
    echo ""
    read -p "  Select array [1]: " arr_choice
    arr_choice=${arr_choice:-1}
    md_dev="${arrays[$((arr_choice-1))]}"
  fi

  if [ ! -b "$md_dev" ]; then
    print_error "Invalid array: ${md_dev}"
    return 1
  fi

  # Get active drives
  local drives=$(mdadm --detail "$md_dev" 2>/dev/null | grep "active sync" | awk '{print $NF}')

  if [ -z "$drives" ]; then
    print_error "No active drives found in ${md_dev}"
    return 1
  fi

  local drive_count=$(echo "$drives" | wc -l | xargs)

  echo ""
  echo -e "  ${BOLD}Blink RAID Array${NC} — ${md_dev} (${drive_count} active drives)"
  echo ""
  echo "  Cycling locate LED across each drive one at a time."
  echo "  Watch for the bay that never lights up — that is your failed drive."
  echo ""
  echo -e "  ${YELLOW}Press Ctrl+C to turn off all locate LEDs.${NC}"
  echo ""

  _blink_array_cleanup() {
    echo ""
    echo "  Turning off locate LEDs..."
    for drive in $drives; do
      _led_locate_off "$drive"
    done
    print_success "All LEDs off"
    trap - SIGINT SIGTERM
    return 0
  }

  trap _blink_array_cleanup SIGINT SIGTERM

  while true; do
    for drive in $drives; do
      _led_locate_off "$drive"
      _led_locate_on "$drive"
      echo -e "  Locating: ${BOLD}${drive}${NC}"
      sleep 0.5
    done
  done
}

function do_disk() {
  local subcmd="${ARGS[0]:-}"
  local arg="${ARGS[1]:-}"

  # Direct subcommand access for scripting
  case "$subcmd" in
    list)
      disk_list; return ;;
    smart)
      disk_smart_quick; return ;;
    led-on|locate)
      disk_led_on "$arg"; return ;;
    led-off|locate-off)
      disk_led_off "$arg"; return ;;
    enclosure|slots)
      disk_enclosure_list; return ;;
    enclosure-led|slot-led)
      disk_enclosure_led; return ;;
    blink-array)
      disk_blink_array; return ;;
  esac

  # Interactive menu
  local node_id=$(get_node_id)

  echo ""
  echo -e "  ${BOLD}Drive Tools${NC} — ${node_id}"
  echo ""
  echo "   1) List all drives            — model, size, serial, transport"
  echo "   2) SMART health check         — quick pass/fail for all drives"
  echo "   3) Blink drive LED (ledctl)   — locate a drive on the backplane"
  echo "   4) Stop blinking LED          — turn off locate LED"
  echo "   5) Enclosure slots            — show SES enclosure slot mapping"
  echo "   6) Blink slot LED (sg_ses)    — identify by enclosure slot number"
  echo "   7) Blink RAID array           — I/O hammer to find dead drive"
  echo ""
  read -p "  Enter selection [1]: " disk_choice
  disk_choice=${disk_choice:-1}

  case $disk_choice in
    1)
      disk_list
      ;;
    2)
      disk_smart_quick
      ;;
    3)
      disk_led_on
      ;;
    4)
      disk_led_off
      ;;
    5)
      disk_enclosure_list
      ;;
    6)
      disk_enclosure_led
      ;;
    7)
      disk_blink_array
      ;;
    *)
      echo "  Invalid selection."
      ;;
  esac
}

# ============================================
# VM Commands (dartnode vms)
# ============================================

function do_vms() {
  local mode=$(get_daemon_mode)

  echo ""
  echo -e "  ${BOLD}Virtual Machines${NC}"
  echo ""

  if [ "$mode" = "pve" ]; then
    do_vms_pve
  elif [ "$mode" = "macos" ]; then
    do_vms_macos
  else
    if command -v qm &> /dev/null || [ -d "/etc/pve" ]; then
      do_vms_pve
    elif command -v tart &> /dev/null; then
      do_vms_macos
    else
      echo "  No VM platform detected."
    fi
  fi
}

function do_vms_pve() {
  if command -v qm &> /dev/null; then
    qm list 2>/dev/null
    if [ $? -ne 0 ]; then
      print_warning "qm list failed, scanning config files..."
      _vms_pve_fallback
    fi
  else
    _vms_pve_fallback
  fi
}

function _vms_pve_fallback() {
  local conf_dir="/etc/pve/qemu-server"
  if [ ! -d "$conf_dir" ]; then
    echo "  No PVE configuration directory found."
    return
  fi

  printf "  ${DIM}%-8s %-30s %-10s %-8s %-8s${NC}\n" "VMID" "Name" "Status" "Cores" "Memory"
  printf "  ${DIM}%-8s %-30s %-10s %-8s %-8s${NC}\n" "────────" "──────────────────────────────" "──────────" "────────" "────────"

  local found=false
  for conf in "${conf_dir}"/*.conf; do
    [ -f "$conf" ] || continue
    found=true
    local vmid=$(basename "$conf" .conf)
    local name cores memory
    name=$(grep "^name:" "$conf" 2>/dev/null | head -1 | sed 's/^name:[[:space:]]*//')
    cores=$(grep "^cores:" "$conf" 2>/dev/null | head -1 | sed 's/^cores:[[:space:]]*//')
    memory=$(grep "^memory:" "$conf" 2>/dev/null | head -1 | sed 's/^memory:[[:space:]]*//')

    local status_vis status_col
    if [ -S "/var/run/qemu-server/${vmid}.qmp" ]; then
      status_vis="running"; status_col="${GREEN}"
    else
      status_vis="stopped"; status_col="${DIM}"
    fi

    local sv=$(printf "%-10s" "$status_vis")
    printf "  %-8s %-30s ${status_col}%s${NC} %-8s %-8s\n" \
      "$vmid" "${name:-unnamed}" "$sv" "${cores:-?}" "${memory:-?}MB"
  done

  [ "$found" = false ] && echo "  No VMs found."
  echo ""
}

function do_vms_macos() {
  if command -v tart &> /dev/null; then
    tart list 2>/dev/null
  else
    local vm_storage
    vm_storage=$(read_config_value "vm_storage_path" "${DN_DAEMON_CONFIG_DIR}/macos.json")
    vm_storage=${vm_storage:-"$HOME/.tart/vms"}
    if [ -d "$vm_storage" ]; then
      echo "  VMs in ${vm_storage}:"
      for vm_dir in "${vm_storage}"/*/; do
        [ -d "$vm_dir" ] || continue
        echo "    $(basename "$vm_dir")"
      done
    else
      echo "  No VMs found."
    fi
  fi
  echo ""
}

# ============================================
# Diagnostics (dartnode diag)
# ============================================

function do_diag() {
  local mode=$(get_daemon_mode)
  local node_id=$(get_node_id)
  local pass=0 warn=0 fail=0 total=0

  echo ""
  echo -e "  ${BOLD}Node Diagnostics${NC} — ${node_id} (${mode})"
  echo -e "  ${DIM}$(date)${NC}"
  echo ""

  # 1. Daemon binary
  ((total++))
  if [ -x "${DN_DAEMON_BIN}" ]; then
    local ver=$(${DN_DAEMON_BIN} -version 2>/dev/null || echo "unknown")
    print_success "Daemon binary: ${DN_DAEMON_BIN} (${ver})"
    ((pass++))
  else
    print_error "Daemon binary not found at ${DN_DAEMON_BIN}"
    ((fail++))
  fi

  # 2. Daemon service
  ((total++))
  if is_service_running; then
    if [ "$OS_TYPE" = "linux" ]; then
      local pid=$(systemctl show -p MainPID --value ${DN_DAEMON_SERVICE} 2>/dev/null)
      local uptime=""
      if [ -n "$pid" ] && [ "$pid" != "0" ]; then
        uptime=$(ps -p "$pid" -o etime= 2>/dev/null | tr -d ' ')
      fi
      print_success "Service: running (PID ${pid}${uptime:+, uptime ${uptime}})"
    else
      print_success "Service: running"
    fi
    ((pass++))
  else
    print_error "Service: not running"
    ((fail++))
  fi

  # 3. Redis
  ((total++))
  if command -v redis-cli &> /dev/null; then
    local pong=$(redis_cmd PING 2>/dev/null)
    if [ "$pong" = "PONG" ]; then
      local key_count=$(redis_cmd DBSIZE 2>/dev/null | grep -o '[0-9]*')
      print_success "Redis: connected (${key_count:-?} keys)"
      ((pass++))

      # Check heartbeat
      ((total++))
      local hb_ttl=$(redis_cmd TTL "node:heartbeat:${node_id}" 2>/dev/null)
      if [ -n "$hb_ttl" ] && [ "$hb_ttl" != "-2" ]; then
        print_success "Heartbeat: alive (TTL ${hb_ttl}s)"
        ((pass++))
      else
        print_warning "Heartbeat: not found (daemon may not be reporting)"
        ((warn++))
      fi
    else
      print_error "Redis: connection failed"
      ((fail++))
    fi
  else
    print_warning "redis-cli not installed"
    ((warn++))
  fi

  # 4. Disk space
  echo ""
  local disk_paths=("/" "/var")
  if [ "$mode" = "pve" ] && [ -d "/mnt/pve" ]; then
    disk_paths+=("/mnt/pve")
  fi

  for dp in "${disk_paths[@]}"; do
    ((total++))
    if [ -d "$dp" ]; then
      local usage=$(df "$dp" 2>/dev/null | tail -1 | awk '{print $5}' | tr -d '%')
      local avail=$(df -h "$dp" 2>/dev/null | tail -1 | awk '{print $4}')
      if [ -n "$usage" ] && [[ "$usage" =~ ^[0-9]+$ ]]; then
        if [ "$usage" -ge 95 ]; then
          print_error "Disk ${dp}: ${usage}% used (${avail} free) — CRITICAL"
          ((fail++))
        elif [ "$usage" -ge 90 ]; then
          print_warning "Disk ${dp}: ${usage}% used (${avail} free)"
          ((warn++))
        else
          print_success "Disk ${dp}: ${usage}% used (${avail} free)"
          ((pass++))
        fi
      fi
    fi
  done

  # 5. Load & Memory
  echo ""
  ((total++))
  local ncpu load1
  if [ "$OS_TYPE" = "macos" ]; then
    ncpu=$(sysctl -n hw.ncpu 2>/dev/null || echo "1")
    load1=$(sysctl -n vm.loadavg 2>/dev/null | awk '{print $2}')
  else
    ncpu=$(nproc 2>/dev/null || echo "1")
    load1=$(cat /proc/loadavg 2>/dev/null | awk '{print $1}')
  fi
  if [ -n "$load1" ]; then
    local load_int=${load1%.*}
    if [ "${load_int:-0}" -ge "$((ncpu * 2))" ]; then
      print_warning "Load: ${load1} (${ncpu} CPUs) — high"
      ((warn++))
    else
      print_success "Load: ${load1} (${ncpu} CPUs)"
      ((pass++))
    fi
  fi

  ((total++))
  if [ "$OS_TYPE" = "linux" ]; then
    local mem_total mem_avail mem_pct
    mem_total=$(grep MemTotal /proc/meminfo 2>/dev/null | awk '{print $2}')
    mem_avail=$(grep MemAvailable /proc/meminfo 2>/dev/null | awk '{print $2}')
    if [ -n "$mem_total" ] && [ -n "$mem_avail" ] && [ "$mem_total" -gt 0 ]; then
      mem_pct=$(( (mem_total - mem_avail) * 100 / mem_total ))
      local mem_total_gb=$(awk "BEGIN{printf \"%.1f\", ${mem_total}/1048576}")
      local mem_avail_gb=$(awk "BEGIN{printf \"%.1f\", ${mem_avail}/1048576}")
      if [ "$mem_pct" -ge 95 ]; then
        print_error "Memory: ${mem_pct}% used (${mem_avail_gb}G free of ${mem_total_gb}G)"
        ((fail++))
      elif [ "$mem_pct" -ge 85 ]; then
        print_warning "Memory: ${mem_pct}% used (${mem_avail_gb}G free of ${mem_total_gb}G)"
        ((warn++))
      else
        print_success "Memory: ${mem_pct}% used (${mem_avail_gb}G free of ${mem_total_gb}G)"
        ((pass++))
      fi
    fi
  else
    print_success "Memory: see 'vm_stat' for details"
    ((pass++))
  fi

  # 6. PVE-specific checks
  if [ "$mode" = "pve" ]; then
    echo ""

    # Network bridges
    ((total++))
    local bridge_count=0 bridge_issues=0
    for brdir in /sys/class/net/*/bridge/; do
      [ -d "$brdir" ] || continue
      local br=$(basename "$(dirname "$brdir")")
      ((bridge_count++))
      local br_state=$(cat "/sys/class/net/$br/operstate" 2>/dev/null)
      local br_ports=$(ls "/sys/class/net/$br/brif/" 2>/dev/null | wc -l)
      if [ "$br_state" != "up" ] || [ "$br_ports" -eq 0 ]; then
        ((bridge_issues++))
      fi
    done
    if [ "$bridge_count" -gt 0 ] && [ "$bridge_issues" -eq 0 ]; then
      print_success "Bridges: ${bridge_count} configured, all UP with ports"
      ((pass++))
    elif [ "$bridge_count" -gt 0 ]; then
      print_warning "Bridges: ${bridge_count} configured, ${bridge_issues} with issues"
      ((warn++))
    else
      print_warning "Bridges: none found"
      ((warn++))
    fi

    # QMP sockets
    local qmp_dir
    qmp_dir=$(read_config_value "qmp_socket_dir" "${DN_DAEMON_CONFIG_DIR}/pve.json")
    qmp_dir=${qmp_dir:-/var/run/qemu-server}
    if [ -d "$qmp_dir" ]; then
      ((total++))
      local qmp_count=$(ls "$qmp_dir"/*.qmp 2>/dev/null | wc -l)
      print_success "QMP sockets: ${qmp_count} accessible in ${qmp_dir}"
      ((pass++))
    fi
  fi

  # Score
  echo ""
  echo -e "  ${DIM}──────────────────────────────────────${NC}"
  local score_color="${GREEN}"
  [ "$fail" -gt 0 ] && score_color="${RED}"
  [ "$warn" -gt 0 ] && [ "$fail" -eq 0 ] && score_color="${YELLOW}"
  echo -e "  ${BOLD}Score:${NC} ${score_color}${pass}/${total} passed${NC} (${GREEN}${pass} pass${NC}, ${YELLOW}${warn} warn${NC}, ${RED}${fail} fail${NC})"
  echo ""
}

# ============================================
# Redis Diagnostics (dartnode redis)
# ============================================

function do_redis_diag() {
  local mode=$(get_daemon_mode)
  local node_id=$(get_node_id)

  echo ""
  echo -e "  ${BOLD}Redis Diagnostics${NC}"
  echo ""

  # Config display
  local rh rp rd
  rh=$(read_config_value "redis_host"); rh=${rh:-127.0.0.1}
  rp=$(read_config_value "redis_port"); rp=${rp:-6379}
  rd=$(read_config_value "redis_db"); rd=${rd:-0}
  echo -e "  Connection: ${CYAN}${rh}:${rp}/${rd}${NC}"
  echo ""

  if ! command -v redis-cli &> /dev/null; then
    print_error "redis-cli not installed"
    echo "  Install: apt install redis-tools"
    return 1
  fi

  # Connection test
  local pong=$(redis_cmd PING)
  if [ "$pong" != "PONG" ]; then
    print_error "Connection failed"
    return 1
  fi
  print_success "Connected"

  # Server info
  local redis_ver=$(redis_cmd INFO server 2>/dev/null | grep "redis_version:" | cut -d: -f2 | tr -d '\r')
  local mem_used=$(redis_cmd INFO memory 2>/dev/null | grep "used_memory_human:" | cut -d: -f2 | tr -d '\r')
  echo -e "  Version: ${redis_ver:-unknown}"
  echo -e "  Memory:  ${mem_used:-unknown}"
  echo ""

  # VM stats keys
  local vm_keys=$(redis_cmd KEYS "vm:stats:*" 2>/dev/null | wc -l)
  echo -e "  ${BOLD}VM Stats:${NC} ${vm_keys} keys"

  if [ "$vm_keys" -gt 0 ]; then
    local min_ttl=999999 max_ttl=0
    while IFS= read -r key; do
      [ -z "$key" ] && continue
      local ttl=$(redis_cmd TTL "$key" 2>/dev/null)
      [ -z "$ttl" ] || [ "$ttl" = "-1" ] || [ "$ttl" = "-2" ] && continue
      [[ "$ttl" =~ ^[0-9]+$ ]] || continue
      [ "$ttl" -lt "$min_ttl" ] && min_ttl=$ttl
      [ "$ttl" -gt "$max_ttl" ] && max_ttl=$ttl
    done < <(redis_cmd KEYS "vm:stats:*" 2>/dev/null)
    [ "$min_ttl" -lt 999999 ] && echo "  TTL range: ${min_ttl}s - ${max_ttl}s"
  fi

  # Node heartbeat
  echo ""
  echo -e "  ${BOLD}Node Heartbeat:${NC}"
  local hb_data=$(redis_cmd GET "node:heartbeat:${node_id}" 2>/dev/null)
  if [ -n "$hb_data" ] && [ "$hb_data" != "(nil)" ]; then
    local hb_ttl=$(redis_cmd TTL "node:heartbeat:${node_id}" 2>/dev/null)
    print_success "Present (TTL ${hb_ttl}s)"
  else
    print_warning "Not found"
  fi

  # Health report
  local health_data=$(redis_cmd GET "node:health:${node_id}" 2>/dev/null)
  if [ -n "$health_data" ] && [ "$health_data" != "(nil)" ]; then
    local health_ttl=$(redis_cmd TTL "node:health:${node_id}" 2>/dev/null)
    echo -e "  ${BOLD}Health Report:${NC} present (TTL ${health_ttl}s)"
  fi

  # macOS job queue
  if [ "$mode" = "macos" ]; then
    echo ""
    echo -e "  ${BOLD}macOS Job Queue:${NC}"
    local job_len=$(redis_cmd LLEN "macos:jobs:${node_id}" 2>/dev/null)
    echo "  Pending jobs: ${job_len:-0}"
    if [ "${job_len:-0}" -gt 0 ]; then
      echo "  Next job:"
      local next=$(redis_cmd LINDEX "macos:jobs:${node_id}" 0 2>/dev/null)
      echo "    ${next:0:200}"
    fi
  fi

  echo ""
}

# ============================================
# Storage Overview (dartnode storage)
# ============================================

function do_storage() {
  echo ""
  echo -e "  ${BOLD}Storage Overview${NC}"
  echo ""

  if [ "$OS_TYPE" = "macos" ]; then
    do_storage_macos
  else
    do_storage_linux
  fi
}

function do_storage_linux() {
  # Disk usage
  echo -e "  ${BOLD}Disk Usage:${NC}"
  echo ""
  df -h / /var /tmp 2>/dev/null | awk 'NR==1{printf "  %-20s %-8s %-8s %-8s %-6s %s\n",$1,$2,$3,$4,$5,$6} NR>1{printf "  %-20s %-8s %-8s %-8s %-6s %s\n",$1,$2,$3,$4,$5,$6}'
  echo ""

  # LVM
  if command -v vgs &> /dev/null; then
    local vgs_out=$(vgs 2>/dev/null)
    if [ -n "$vgs_out" ]; then
      echo -e "  ${BOLD}LVM Volume Groups:${NC}"
      echo ""
      echo "$vgs_out" | sed 's/^/  /'
      echo ""
      echo -e "  ${BOLD}LVM Logical Volumes:${NC}"
      echo ""
      lvs 2>/dev/null | sed 's/^/  /'
      echo ""
    fi
  fi

  # RAID
  if [ -f /proc/mdstat ]; then
    local mdstat=$(cat /proc/mdstat 2>/dev/null)
    if echo "$mdstat" | grep -q "^md"; then
      echo -e "  ${BOLD}RAID Arrays:${NC}"
      echo ""
      echo "$mdstat" | sed 's/^/  /'
      echo ""
    fi
  fi

  # ZFS
  if command -v zpool &> /dev/null; then
    local zpool_out=$(zpool list 2>/dev/null)
    if [ -n "$zpool_out" ]; then
      echo -e "  ${BOLD}ZFS Pools:${NC}"
      echo ""
      echo "$zpool_out" | sed 's/^/  /'
      echo ""
    fi
  fi
}

function do_storage_macos() {
  echo -e "  ${BOLD}Disk Usage:${NC}"
  echo ""
  df -h / 2>/dev/null | awk 'NR==1{printf "  %-20s %-8s %-8s %-8s %-6s\n","Filesystem","Size","Used","Avail","Use%"} NR>1{printf "  %-20s %-8s %-8s %-8s %-6s\n",$1,$2,$3,$4,$5}'
  echo ""

  # VM storage
  local vm_storage
  vm_storage=$(read_config_value "vm_storage_path" "${DN_DAEMON_CONFIG_DIR}/macos.json")
  vm_storage=${vm_storage:-"$HOME/.tart/vms"}

  if [ -d "$vm_storage" ]; then
    echo -e "  ${BOLD}VM Storage:${NC} ${vm_storage}"
    echo ""
    for vm_dir in "${vm_storage}"/*/; do
      [ -d "$vm_dir" ] || continue
      local vm_name=$(basename "$vm_dir")
      local vm_size=$(du -sh "$vm_dir" 2>/dev/null | awk '{print $1}')
      printf "  %-30s %s\n" "$vm_name" "${vm_size:-?}"
    done
    echo ""
  fi
}

# ============================================
# Firewall Overview (dartnode fw)
# ============================================

function do_fw() {
  echo ""
  echo -e "  ${BOLD}Firewall Overview${NC}"
  echo ""

  if [ "$OS_TYPE" = "macos" ]; then
    do_fw_macos
  else
    do_fw_linux
  fi
}

function do_fw_linux() {
  if ! command -v iptables &> /dev/null; then
    print_warning "iptables not found"
    return
  fi

  # Filter table
  echo -e "  ${BOLD}Filter Table:${NC}"
  echo ""
  local filter_out=$(iptables -L -n --line-numbers 2>/dev/null)
  if [ -n "$filter_out" ]; then
    echo "$filter_out" | while IFS= read -r line; do
      if echo "$line" | grep -q "^Chain "; then
        echo -e "  ${CYAN}${line}${NC}"
      else
        echo "  $line"
      fi
    done
  else
    echo "  (empty or no permissions)"
  fi
  echo ""

  # NAT table
  echo -e "  ${BOLD}NAT Table:${NC}"
  echo ""
  local nat_out=$(iptables -t nat -L -n --line-numbers 2>/dev/null)
  if [ -n "$nat_out" ]; then
    echo "$nat_out" | while IFS= read -r line; do
      if echo "$line" | grep -q "^Chain "; then
        echo -e "  ${CYAN}${line}${NC}"
      else
        echo "  $line"
      fi
    done
  else
    echo "  (empty or no permissions)"
  fi
  echo ""
}

function do_fw_macos() {
  # PF status
  echo -e "  ${BOLD}PF Status:${NC}"
  echo ""
  local pf_status=$(pfctl -s info 2>/dev/null | head -5)
  if [ -n "$pf_status" ]; then
    echo "$pf_status" | sed 's/^/  /'
  else
    echo "  (unable to read PF status — run as root)"
  fi
  echo ""

  # PF rules
  echo -e "  ${BOLD}PF Rules:${NC}"
  echo ""
  local pf_rules=$(pfctl -s rules 2>/dev/null)
  if [ -n "$pf_rules" ]; then
    echo "$pf_rules" | sed 's/^/  /'
  else
    echo "  (no rules or no permissions)"
  fi
  echo ""

  # NAT rules
  echo -e "  ${BOLD}NAT Rules:${NC}"
  echo ""
  local nat_rules=$(pfctl -s nat 2>/dev/null)
  if [ -n "$nat_rules" ]; then
    echo "$nat_rules" | sed 's/^/  /'
  else
    echo "  (no NAT rules)"
  fi
  echo ""

  # DartNode anchor
  local dn_anchor=$(pfctl -a "com.dartnode" -s rules 2>/dev/null)
  if [ -n "$dn_anchor" ]; then
    echo -e "  ${BOLD}DartNode Anchor:${NC}"
    echo ""
    echo "$dn_anchor" | sed 's/^/  /'
    echo ""
  fi
}

# ============================================
# QMP Socket Test (dartnode qmp)
# ============================================

function do_qmp() {
  local vmid="${ARGS[0]:-}"
  if [ -z "$vmid" ]; then
    echo "Usage: dartnode qmp <vmid>"
    return 1
  fi

  if [ "$OS_TYPE" = "macos" ]; then
    echo "QMP socket test is only available on PVE/Linux."
    return 1
  fi

  local qmp_dir
  qmp_dir=$(read_config_value "qmp_socket_dir" "${DN_DAEMON_CONFIG_DIR}/pve.json")
  qmp_dir=${qmp_dir:-/var/run/qemu-server}

  local socket_path="${qmp_dir}/${vmid}.qmp"

  echo ""
  echo -e "  ${BOLD}QMP Socket Test${NC} — VM ${vmid}"
  echo ""

  # Check socket exists
  if [ ! -S "$socket_path" ]; then
    print_error "Socket not found: ${socket_path}"
    echo "  VM ${vmid} may not be running."
    return 1
  fi
  print_success "Socket exists: ${socket_path}"

  # Try to connect and query
  if command -v python3 &> /dev/null; then
    local result
    result=$(python3 -c "
import socket, json, sys
sock = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM)
try:
    sock.settimeout(5)
    sock.connect('${socket_path}')
    greeting = json.loads(sock.recv(4096))
    qemu_ver = greeting.get('QMP',{}).get('version',{}).get('qemu',{})
    print('Greeting: QMP ' + str(qemu_ver.get('major','?')) + '.' + str(qemu_ver.get('minor','?')))
    sock.send(json.dumps({'execute': 'qmp_capabilities'}).encode() + b'\n')
    resp = json.loads(sock.recv(4096))
    sock.send(json.dumps({'execute': 'query-status'}).encode() + b'\n')
    status = json.loads(sock.recv(4096))
    ret = status.get('return', {})
    print('Status: ' + ret.get('status', 'unknown'))
    print('Running: ' + str(ret.get('running', False)))
    sock.send(json.dumps({'execute': 'query-cpus-fast'}).encode() + b'\n')
    cpus = json.loads(sock.recv(4096))
    cpu_list = cpus.get('return', [])
    print('CPUs: ' + str(len(cpu_list)))
except Exception as e:
    print('Error: ' + str(e))
    sys.exit(1)
finally:
    sock.close()
" 2>/dev/null)

    if [ $? -eq 0 ]; then
      print_success "QMP handshake successful"
      echo ""
      echo "$result" | sed 's/^/  /'
    else
      print_error "QMP handshake failed"
      [ -n "$result" ] && echo "$result" | sed 's/^/  /'
    fi
  elif command -v socat &> /dev/null; then
    echo "  Using socat (limited output)..."
    local greeting
    greeting=$(echo '{"execute": "qmp_capabilities"}' | socat - UNIX-CONNECT:"${socket_path}" 2>/dev/null | head -2)
    if [ -n "$greeting" ]; then
      print_success "QMP connection successful"
      echo "  $greeting"
    else
      print_error "QMP connection failed"
    fi
  else
    print_error "Neither python3 nor socat available for QMP connection"
  fi

  echo ""
}

# ============================================
# Node Repair (dartnode repair)
# ============================================

function do_repair() {
  local mode=$(get_daemon_mode)
  local node_id=$(get_node_id)

  echo ""
  echo -e "  ${BOLD}Node Repair${NC} — ${node_id} (${mode})"
  echo -e "  ${DIM}$(date)${NC}"
  echo ""

  if [ "$OS_TYPE" = "macos" ]; then
    do_repair_macos
  else
    do_repair_pve
  fi
}

function do_repair_pve() {
  check_root
  local node_id=$(get_node_id)
  local fixed=0
  local issues=0

  echo -e "  ${BOLD}What would you like to repair?${NC}"
  echo ""
  echo -e "  ${BOLD}Quick Fixes:${NC}"
  echo "   1) Full auto-repair (scan & fix all)"
  echo "   2) Fix stuck/zombie QEMU processes"
  echo "   3) Clear stale PVE locks (config + lock files)"
  echo "   4) Fix broken daemon service"
  echo "   5) Repair LVM thin pool issues"
  echo "   6) Reset Redis state for this node"
  echo "   7) Fix stuck VM migration"
  echo "   8) Restart PVE cluster services"
  echo ""
  echo -e "  ${BOLD}Deep Recovery (from toolbox):${NC}"
  echo "   9) Fix hung LVM (NFS hang, missing disk, stuck device-mapper)"
  echo "  10) Deep node recovery (full Proxmox service + storage + cluster)"
  echo "  11) Fix hung NFS mounts"
  echo "  12) Disk health check (SMART, I/O errors, missing devices)"
  echo "  13) Recover unmounted filesystems"
  echo ""
  read -p "  Enter selection [1]: " repair_choice
  repair_choice=${repair_choice:-1}

  case $repair_choice in
    1) repair_pve_all ;;
    2) repair_stuck_qemu ;;
    3) repair_stale_locks ;;
    4) repair_daemon_service ;;
    5) repair_lvm_thin ;;
    6) repair_redis_state ;;
    7) repair_stuck_migration ;;
    8) repair_pve_cluster ;;
    9) repair_lvm_hung ;;
    10) repair_deep_node_recovery ;;
    11) repair_hung_nfs ;;
    12) repair_disk_health ;;
    13) repair_unmounted_filesystems ;;
    *)
      repair_pve_all
      ;;
  esac
}

function repair_pve_all() {
  local fixed=0
  local issues=0
  local node_id=$(get_node_id)

  print_step "Running full node repair scan..."
  echo ""

  # 0. Pre-check: D-state processes and hung NFS (most common cause of stuck nodes)
  echo -e "  ${BOLD}[pre] Checking for I/O hung processes...${NC}"
  local d_count=$(ps aux 2>/dev/null | awk '$8 ~ /D/' | wc -l)
  if [ "$d_count" -gt 0 ]; then
    ((issues++))
    print_warning "${d_count} process(es) stuck in D state (uninterruptible I/O)"

    # Quick NFS check
    local hung_nfs=0
    while IFS= read -r line; do
      local mountpoint=$(echo "$line" | awk '{print $3}')
      if ! timeout 3 stat "$mountpoint" &>/dev/null; then
        ((hung_nfs++))
        print_error "  Hung NFS mount: ${mountpoint}"
      fi
    done < <(mount 2>/dev/null | grep -E 'nfs|nfs4')

    if [ "$hung_nfs" -gt 0 ]; then
      print_error "  Hung NFS is likely blocking LVM and other operations!"
      read -p "  Run hung NFS/LVM recovery first? (y/n) [y]: " fix_nfs_lvm
      fix_nfs_lvm=${fix_nfs_lvm:-y}
      if [[ $fix_nfs_lvm =~ ^[Yy]$ ]]; then
        repair_lvm_hung
        ((fixed++))
      fi
    fi
  else
    print_success "No I/O hung processes"
  fi

  # 1. Stuck QEMU processes
  echo ""
  echo -e "  ${BOLD}[1/7] Checking for stuck QEMU processes...${NC}"
  local stuck_procs=$(repair_check_stuck_qemu)
  if [ "$stuck_procs" -gt 0 ]; then
    ((issues++))
    read -p "  Fix ${stuck_procs} stuck process(es)? (y/n) [y]: " fix_qemu
    fix_qemu=${fix_qemu:-y}
    if [[ $fix_qemu =~ ^[Yy]$ ]]; then
      repair_stuck_qemu
      ((fixed++))
    fi
  else
    print_success "No stuck QEMU processes"
  fi

  # 2. Stale PVE locks
  echo ""
  echo -e "  ${BOLD}[2/7] Checking for stale PVE locks...${NC}"
  local stale_locks=$(repair_check_stale_locks)
  if [ "$stale_locks" -gt 0 ]; then
    ((issues++))
    read -p "  Clear ${stale_locks} stale lock(s)? (y/n) [y]: " fix_locks
    fix_locks=${fix_locks:-y}
    if [[ $fix_locks =~ ^[Yy]$ ]]; then
      repair_stale_locks
      ((fixed++))
    fi
  else
    print_success "No stale locks"
  fi

  # 3. Daemon service
  echo ""
  echo -e "  ${BOLD}[3/7] Checking daemon service...${NC}"
  if ! is_service_running; then
    ((issues++))
    print_warning "Daemon is not running"
    read -p "  Repair and start daemon? (y/n) [y]: " fix_daemon
    fix_daemon=${fix_daemon:-y}
    if [[ $fix_daemon =~ ^[Yy]$ ]]; then
      repair_daemon_service
      ((fixed++))
    fi
  else
    print_success "Daemon is running"
  fi

  # 4. Redis connectivity
  echo ""
  echo -e "  ${BOLD}[4/7] Checking Redis...${NC}"
  if command -v redis-cli &> /dev/null; then
    local pong=$(redis_cmd PING 2>/dev/null)
    if [ "$pong" != "PONG" ]; then
      ((issues++))
      print_error "Redis connection failed"
      print_info "  Check redis configuration in ${DN_DAEMON_CONFIG_DIR}/config.json"
    else
      print_success "Redis connected"

      # Check stale heartbeat
      local hb_ttl=$(redis_cmd TTL "node:heartbeat:${node_id}" 2>/dev/null)
      if [ "$hb_ttl" = "-2" ]; then
        ((issues++))
        print_warning "Node heartbeat missing from Redis"
        if is_service_running; then
          print_info "  Daemon is running but not reporting — may need restart"
          read -p "  Restart daemon? (y/n) [y]: " fix_hb
          fix_hb=${fix_hb:-y}
          if [[ $fix_hb =~ ^[Yy]$ ]]; then
            do_restart
            ((fixed++))
          fi
        fi
      else
        print_success "Heartbeat present (TTL ${hb_ttl}s)"
      fi
    fi
  else
    print_warning "redis-cli not installed — skipping Redis checks"
  fi

  # 5. LVM thin pool
  echo ""
  echo -e "  ${BOLD}[5/7] Checking LVM thin pools...${NC}"
  if command -v lvs &> /dev/null; then
    local thin_issues=$(repair_check_lvm_thin)
    if [ "$thin_issues" -gt 0 ]; then
      ((issues++))
      read -p "  Attempt LVM thin pool repair? (y/n) [n]: " fix_lvm
      if [[ $fix_lvm =~ ^[Yy]$ ]]; then
        repair_lvm_thin
        ((fixed++))
      fi
    else
      print_success "LVM thin pools OK"
    fi
  else
    print_info "  LVM not available — skipping"
  fi

  # 6. Stuck tasks in PVE task log
  echo ""
  echo -e "  ${BOLD}[6/7] Checking for stuck PVE tasks...${NC}"
  local stuck_tasks=$(repair_check_stuck_tasks)
  if [ "$stuck_tasks" -gt 0 ]; then
    ((issues++))
    print_warning "${stuck_tasks} stuck task(s) found"
    read -p "  Clear stuck tasks? (y/n) [y]: " fix_tasks
    fix_tasks=${fix_tasks:-y}
    if [[ $fix_tasks =~ ^[Yy]$ ]]; then
      repair_stuck_tasks
      ((fixed++))
    fi
  else
    print_success "No stuck PVE tasks"
  fi

  # 7. Stale Redis job queue entries
  echo ""
  echo -e "  ${BOLD}[7/7] Checking Redis job queues...${NC}"
  if command -v redis-cli &> /dev/null; then
    local node_id=$(get_node_id)
    local stale_jobs=$(repair_check_stale_jobs)
    if [ "$stale_jobs" -gt 0 ]; then
      ((issues++))
      print_warning "${stale_jobs} stale job(s) in queue"
      read -p "  Clear stale jobs? (y/n) [y]: " fix_jobs
      fix_jobs=${fix_jobs:-y}
      if [[ $fix_jobs =~ ^[Yy]$ ]]; then
        do_jobs_clear_stale
        ((fixed++))
      fi
    else
      print_success "Job queues clean"
    fi
  fi

  # Summary
  echo ""
  echo -e "  ${DIM}──────────────────────────────────────${NC}"
  if [ "$issues" -eq 0 ]; then
    echo -e "  ${GREEN}${BOLD}Node is healthy — no issues found.${NC}"
  else
    echo -e "  ${BOLD}Found ${issues} issue(s), fixed ${fixed}.${NC}"
    if [ "$fixed" -lt "$issues" ]; then
      echo -e "  ${YELLOW}Some issues require manual intervention.${NC}"
    fi
  fi
  echo ""
}

function repair_check_stuck_qemu() {
  local count=0
  # Find QEMU processes whose VM config no longer exists or are orphaned
  while IFS= read -r line; do
    [ -z "$line" ] && continue
    local pid=$(echo "$line" | awk '{print $2}')
    local vmid=$(echo "$line" | grep -oP '(?<=-id )\d+' || echo "")
    if [ -n "$vmid" ]; then
      # Check if VM should be running (has a .qmp socket or PID file)
      if [ ! -S "/var/run/qemu-server/${vmid}.qmp" ] && [ ! -f "/var/run/qemu-server/${vmid}.pid" ]; then
        ((count++))
      fi
    fi
  done < <(ps aux 2>/dev/null | grep "[q]emu-system" || true)
  echo "$count"
}

function repair_stuck_qemu() {
  print_step "Scanning for stuck QEMU processes..."
  echo ""

  local found=false
  while IFS= read -r line; do
    [ -z "$line" ] && continue
    local pid=$(echo "$line" | awk '{print $2}')
    local vmid=$(echo "$line" | grep -oP '(?<=-id )\d+' || echo "unknown")
    local runtime=$(echo "$line" | awk '{print $10}')

    # Check for orphaned processes (no QMP socket)
    local orphan=false
    if [ "$vmid" != "unknown" ] && [ ! -S "/var/run/qemu-server/${vmid}.qmp" ]; then
      orphan=true
    fi

    if [ "$orphan" = true ]; then
      found=true
      print_warning "Orphaned QEMU process: PID ${pid}, VMID ${vmid}, runtime ${runtime}"
      read -p "    Kill PID ${pid}? (y/n) [n]: " kill_choice
      if [[ $kill_choice =~ ^[Yy]$ ]]; then
        kill "$pid" 2>/dev/null
        sleep 2
        if kill -0 "$pid" 2>/dev/null; then
          print_warning "  Process didn't terminate, sending SIGKILL..."
          kill -9 "$pid" 2>/dev/null
          sleep 1
        fi
        if ! kill -0 "$pid" 2>/dev/null; then
          print_success "  Process ${pid} terminated"
        else
          print_error "  Failed to kill process ${pid}"
        fi
      fi
    fi
  done < <(ps aux 2>/dev/null | grep "[q]emu-system" || true)

  if [ "$found" = false ]; then
    print_success "No stuck QEMU processes found"
  fi
  echo ""
}

function repair_check_stale_locks() {
  local count=0
  local conf_dir="/etc/pve/qemu-server"
  [ ! -d "$conf_dir" ] && echo "0" && return

  for conf in "${conf_dir}"/*.conf; do
    [ -f "$conf" ] || continue
    if grep -q "^lock:" "$conf" 2>/dev/null; then
      local vmid=$(basename "$conf" .conf)
      local lock_type=$(grep "^lock:" "$conf" | sed 's/^lock:[[:space:]]*//')
      # Check if the process that holds the lock is still alive
      local is_active=false
      if [ -S "/var/run/qemu-server/${vmid}.qmp" ]; then
        is_active=true
      fi
      # migration/backup locks on non-running VMs are stale
      if [ "$is_active" = false ]; then
        ((count++))
      fi
    fi
  done

  # Also count stale lock files in /var/lock and /run/lock
  local lock_dirs=("/run/lock/qemu-server" "/var/lock/qemu-server")
  for lock_dir in "${lock_dirs[@]}"; do
    [ -d "$lock_dir" ] || continue
    for lock in "${lock_dir}"/*.lock "${lock_dir}"/lock-*.conf; do
      [ -f "$lock" ] || continue
      local lock_name=$(basename "$lock")
      local vmid=""
      if [[ "$lock_name" =~ ^([0-9]+)\.lock$ ]]; then
        vmid="${BASH_REMATCH[1]}"
      elif [[ "$lock_name" =~ ^lock-([0-9]+)\.conf$ ]]; then
        vmid="${BASH_REMATCH[1]}"
      else
        continue
      fi
      if ! pgrep -f "kvm.*-id $vmid" &>/dev/null; then
        ((count++))
      fi
    done
  done

  echo "$count"
}

function repair_stale_locks() {
  print_step "Scanning for stale PVE locks..."
  echo ""

  local conf_dir="/etc/pve/qemu-server"
  if [ ! -d "$conf_dir" ]; then
    print_warning "PVE config directory not found"
    return
  fi

  local found=false
  for conf in "${conf_dir}"/*.conf; do
    [ -f "$conf" ] || continue
    if grep -q "^lock:" "$conf" 2>/dev/null; then
      local vmid=$(basename "$conf" .conf)
      local lock_type=$(grep "^lock:" "$conf" | sed 's/^lock:[[:space:]]*//')
      local vm_name=$(grep "^name:" "$conf" 2>/dev/null | head -1 | sed 's/^name:[[:space:]]*//')

      local is_active=false
      if [ -S "/var/run/qemu-server/${vmid}.qmp" ]; then
        is_active=true
      fi

      if [ "$is_active" = false ]; then
        found=true
        print_warning "Stale lock on VM ${vmid} (${vm_name:-unnamed}): ${lock_type}"
        read -p "    Remove lock? (y/n) [y]: " remove_lock
        remove_lock=${remove_lock:-y}
        if [[ $remove_lock =~ ^[Yy]$ ]]; then
          if command -v qm &> /dev/null; then
            qm unlock "$vmid" 2>/dev/null
            if [ $? -eq 0 ]; then
              print_success "  Lock removed from VM ${vmid}"
            else
              print_error "  Failed to remove lock via qm, trying manual removal..."
              sed -i '/^lock:/d' "$conf" 2>/dev/null
              print_success "  Lock line removed from config"
            fi
          else
            sed -i '/^lock:/d' "$conf" 2>/dev/null
            print_success "  Lock line removed from config"
          fi
        fi
      else
        print_info "  VM ${vmid} has lock '${lock_type}' but is running — skipping"
      fi
    fi
  done

  # Also check /run/lock/ and /var/lock/ files (from toolbox pattern)
  local lock_dirs=("/run/lock/qemu-server" "/var/lock/qemu-server")
  for lock_dir in "${lock_dirs[@]}"; do
    [ -d "$lock_dir" ] || continue

    # Check both *.lock and lock-*.conf patterns (PVE uses both)
    for lock in "${lock_dir}"/*.lock "${lock_dir}"/lock-*.conf; do
      [ -f "$lock" ] || continue
      local lock_name=$(basename "$lock")
      local vmid=""

      # Extract VMID from filename: either "12345.lock" or "lock-12345.conf"
      if [[ "$lock_name" =~ ^([0-9]+)\.lock$ ]]; then
        vmid="${BASH_REMATCH[1]}"
      elif [[ "$lock_name" =~ ^lock-([0-9]+)\.conf$ ]]; then
        vmid="${BASH_REMATCH[1]}"
      else
        continue
      fi

      # Triple-check process isn't running
      if pgrep -f "kvm.*-id $vmid" &>/dev/null; then
        print_info "  VM ${vmid}: process running — lock file is legitimate"
        continue
      fi

      found=true
      local age_secs=$(( $(date +%s) - $(stat -c %Y "$lock" 2>/dev/null || stat -f %m "$lock" 2>/dev/null || echo "0") ))
      local age_min=$(( age_secs / 60 ))
      print_warning "Stale lock file: ${lock} (no QEMU process for VM ${vmid}, age: ${age_min}m)"
      read -p "    Remove lock file? (y/n) [y]: " remove_lockfile
      remove_lockfile=${remove_lockfile:-y}
      if [[ $remove_lockfile =~ ^[Yy]$ ]]; then
        rm -f "$lock"
        print_success "  Lock file removed for VM ${vmid}"
      fi
    done
  done

  # Check container locks too
  local ct_lock_dir="/run/lock/lxc"
  if [ -d "$ct_lock_dir" ]; then
    for lock in "${ct_lock_dir}"/*.lock; do
      [ -f "$lock" ] || continue
      local ctid=$(basename "$lock" .lock)

      if command -v pct &> /dev/null && pct status "$ctid" 2>/dev/null | grep -q "running"; then
        print_info "  Container ${ctid}: running — lock is legitimate"
        continue
      fi

      found=true
      print_warning "Stale container lock: ${lock}"
      read -p "    Remove lock file? (y/n) [y]: " remove_ct_lock
      remove_ct_lock=${remove_ct_lock:-y}
      if [[ $remove_ct_lock =~ ^[Yy]$ ]]; then
        rm -f "$lock"
        print_success "  Lock file removed for container ${ctid}"
      fi
    done
  fi

  if [ "$found" = false ]; then
    print_success "No stale locks found"
  fi
  echo ""
}

function repair_daemon_service() {
  print_step "Repairing daemon service..."
  echo ""

  # Check binary
  if [ ! -x "${DN_DAEMON_BIN}" ]; then
    print_warning "Daemon binary missing or not executable"
    read -p "  Re-download daemon binary? (y/n) [y]: " dl_choice
    dl_choice=${dl_choice:-y}
    if [[ $dl_choice =~ ^[Yy]$ ]]; then
      print_step "Downloading dn-daemon..."
      if [ "$OS_TYPE" = "macos" ]; then
        curl -sSL -o ${DN_DAEMON_BIN} ${PKG_BASE}/dn-daemon-darwin-arm64
      else
        wget -q --show-progress -O ${DN_DAEMON_BIN} ${PKG_BASE}/dn-daemon
      fi
      chmod +x ${DN_DAEMON_BIN}
      print_success "Binary downloaded"
    fi
  else
    print_success "Daemon binary exists"
  fi

  # Check config
  if [ ! -f "${DN_DAEMON_CONFIG_DIR}/config.json" ]; then
    print_error "Config file missing at ${DN_DAEMON_CONFIG_DIR}/config.json"
    print_info "  Run 'dartnode setup' to create configuration"
    return
  fi
  print_success "Config file exists"

  # Check/repair systemd service (Linux)
  if [ "$OS_TYPE" = "linux" ]; then
    if [ ! -f "/etc/systemd/system/${DN_DAEMON_SERVICE}.service" ]; then
      print_warning "Systemd service file missing — recreating..."
      cat > "/etc/systemd/system/${DN_DAEMON_SERVICE}.service" <<-EOF
[Unit]
Description=DartNode Daemon
Documentation=https://dartnode.com
After=network.target
Wants=network-online.target

[Service]
Type=simple
ExecStart=${DN_DAEMON_BIN} -config ${DN_DAEMON_CONFIG_DIR}/config.json
Restart=always
RestartSec=5
StandardOutput=journal
StandardError=journal
SyslogIdentifier=dn-daemon

[Install]
WantedBy=multi-user.target
EOF
      systemctl daemon-reload
      systemctl enable ${DN_DAEMON_SERVICE}
      print_success "Service file recreated"
    fi

    # Check for failed state
    local svc_state=$(systemctl show -p ActiveState --value ${DN_DAEMON_SERVICE} 2>/dev/null)
    if [ "$svc_state" = "failed" ]; then
      print_warning "Service is in failed state — resetting..."
      systemctl reset-failed ${DN_DAEMON_SERVICE} 2>/dev/null
      print_success "Service state reset"
    fi
  fi

  # Try starting
  if ! is_service_running; then
    print_step "Starting daemon..."
    do_start
  else
    print_success "Daemon is already running"
  fi
  echo ""
}

function repair_check_lvm_thin() {
  local count=0
  if ! command -v lvs &> /dev/null; then
    echo "0"
    return
  fi

  # Check thin pools for high usage or metadata issues
  while IFS= read -r line; do
    [ -z "$line" ] && continue
    local data_pct=$(echo "$line" | awk '{print $6}' | tr -d '%')
    local meta_pct=$(echo "$line" | awk '{print $7}' | tr -d '%')
    if [ -n "$data_pct" ] && [[ "$data_pct" =~ ^[0-9]+(\.[0-9]+)?$ ]]; then
      local data_int=${data_pct%.*}
      if [ "${data_int:-0}" -ge 95 ]; then
        ((count++))
      fi
    fi
    if [ -n "$meta_pct" ] && [[ "$meta_pct" =~ ^[0-9]+(\.[0-9]+)?$ ]]; then
      local meta_int=${meta_pct%.*}
      if [ "${meta_int:-0}" -ge 90 ]; then
        ((count++))
      fi
    fi
  done < <(lvs --noheadings -o lv_name,vg_name,lv_attr,lv_size,pool_lv,data_percent,metadata_percent 2>/dev/null | grep "t")

  echo "$count"
}

function repair_lvm_thin() {
  print_step "Checking LVM thin pools..."
  echo ""

  if ! command -v lvs &> /dev/null; then
    print_warning "LVM tools not installed"
    return
  fi

  # Show thin pool status
  echo -e "  ${BOLD}Thin Pool Status:${NC}"
  echo ""
  lvs --noheadings -o lv_name,vg_name,lv_size,data_percent,metadata_percent,lv_attr 2>/dev/null | grep "t" | while IFS= read -r line; do
    [ -z "$line" ] && continue
    local lv_name=$(echo "$line" | awk '{print $1}')
    local vg_name=$(echo "$line" | awk '{print $2}')
    local lv_size=$(echo "$line" | awk '{print $3}')
    local data_pct=$(echo "$line" | awk '{print $4}')
    local meta_pct=$(echo "$line" | awk '{print $5}')

    local data_int=${data_pct%.*}
    local meta_int=${meta_pct%.*}

    local data_col="${GREEN}"
    [ "${data_int:-0}" -ge 80 ] && data_col="${YELLOW}"
    [ "${data_int:-0}" -ge 95 ] && data_col="${RED}"

    echo -e "    ${BOLD}${vg_name}/${lv_name}${NC} (${lv_size})"
    echo -e "      Data: ${data_col}${data_pct}%${NC}  Meta: ${meta_pct}%"

    if [ "${data_int:-0}" -ge 95 ]; then
      print_error "  Thin pool ${lv_name} is critically full!"
      echo ""
      echo "    Possible fixes:"
      echo "      1) Extend the thin pool:  lvextend -L +50G ${vg_name}/${lv_name}"
      echo "      2) Remove unused LVs:     lvremove ${vg_name}/<lv_name>"
      echo "      3) Run fstrim inside VMs to reclaim space"
      echo ""
    fi

    if [ "${meta_int:-0}" -ge 90 ]; then
      print_warning "  Thin pool metadata is almost full!"
      echo "    Fix: lvextend --poolmetadatasize +1G ${vg_name}/${lv_name}"
      echo ""
    fi
  done

  # Check for orphaned LVs (LVs with no matching VM config)
  echo ""
  echo -e "  ${BOLD}Checking for orphaned LVs...${NC}"
  local orphan_count=0
  while IFS= read -r line; do
    [ -z "$line" ] && continue
    local lv_name=$(echo "$line" | awk '{print $1}')
    local vg_name=$(echo "$line" | awk '{print $2}')

    # Pattern: vm-<vmid>-disk-<n>
    if [[ "$lv_name" =~ ^vm-([0-9]+)-disk- ]]; then
      local vmid="${BASH_REMATCH[1]}"
      if [ ! -f "/etc/pve/qemu-server/${vmid}.conf" ]; then
        ((orphan_count++))
        local lv_size=$(echo "$line" | awk '{print $4}')
        print_warning "  Orphaned LV: ${vg_name}/${lv_name} (${lv_size}) — VM ${vmid} config missing"
      fi
    fi
  done < <(lvs --noheadings -o lv_name,vg_name,lv_attr,lv_size 2>/dev/null | grep -v "t....-" || true)

  if [ "$orphan_count" -eq 0 ]; then
    print_success "No orphaned LVs found"
  else
    echo ""
    print_warning "${orphan_count} orphaned LV(s) found"
    print_info "  Use 'lvremove <vg>/<lv>' to remove after verifying they're not needed"
  fi
  echo ""
}

function repair_check_stuck_tasks() {
  local count=0
  if [ ! -d "/var/log/pve/tasks" ]; then
    echo "0"
    return
  fi

  # Check for active task files older than 2 hours
  local cutoff=$(date -d '2 hours ago' +%s 2>/dev/null || date -v-2H +%s 2>/dev/null || echo "0")
  for task_dir in /var/log/pve/tasks/active/*/; do
    [ -d "$task_dir" ] || continue
    for task_file in "${task_dir}"*; do
      [ -f "$task_file" ] || continue
      local task_mtime=$(stat -c %Y "$task_file" 2>/dev/null || stat -f %m "$task_file" 2>/dev/null || echo "0")
      if [ "$task_mtime" -lt "$cutoff" ] && [ "$cutoff" -gt 0 ]; then
        ((count++))
      fi
    done
  done
  echo "$count"
}

function repair_stuck_tasks() {
  print_step "Checking for stuck PVE tasks..."
  echo ""

  if [ ! -d "/var/log/pve/tasks" ]; then
    print_info "  PVE task directory not found"
    return
  fi

  local found=false
  local cutoff=$(date -d '2 hours ago' +%s 2>/dev/null || date -v-2H +%s 2>/dev/null || echo "0")

  for task_dir in /var/log/pve/tasks/active/*/; do
    [ -d "$task_dir" ] || continue
    for task_file in "${task_dir}"*; do
      [ -f "$task_file" ] || continue
      local task_name=$(basename "$task_file")
      local task_mtime=$(stat -c %Y "$task_file" 2>/dev/null || stat -f %m "$task_file" 2>/dev/null || echo "0")
      if [ "$task_mtime" -lt "$cutoff" ] && [ "$cutoff" -gt 0 ]; then
        found=true
        local age_hours=$(( ($(date +%s) - task_mtime) / 3600 ))
        print_warning "Stuck task: ${task_name} (${age_hours}h old)"
      fi
    done
  done

  if [ "$found" = true ]; then
    echo ""
    read -p "  Clear stuck active tasks? (y/n) [n]: " clear_tasks
    if [[ $clear_tasks =~ ^[Yy]$ ]]; then
      # Restart pvedaemon to release stuck tasks
      print_step "Restarting pvedaemon to clear stuck tasks..."
      force_restart_pve_service pvedaemon 30
    fi
  else
    print_success "No stuck tasks found"
  fi
  echo ""
}

function repair_stuck_migration() {
  check_root
  print_step "Checking for stuck migrations..."
  echo ""

  local conf_dir="/etc/pve/qemu-server"
  if [ ! -d "$conf_dir" ]; then
    print_warning "PVE config directory not found"
    return
  fi

  local found=false
  for conf in "${conf_dir}"/*.conf; do
    [ -f "$conf" ] || continue
    local lock_type=$(grep "^lock:" "$conf" 2>/dev/null | sed 's/^lock:[[:space:]]*//')
    if [ "$lock_type" = "migrate" ] || [ "$lock_type" = "moving" ]; then
      found=true
      local vmid=$(basename "$conf" .conf)
      local vm_name=$(grep "^name:" "$conf" 2>/dev/null | head -1 | sed 's/^name:[[:space:]]*//')
      print_warning "VM ${vmid} (${vm_name:-unnamed}) has migration lock"

      # Check if migration process exists
      local mig_pid=$(pgrep -f "migrate.*${vmid}" 2>/dev/null || true)
      if [ -n "$mig_pid" ]; then
        print_info "  Migration process PID: ${mig_pid}"
        read -p "    Kill migration process and clear lock? (y/n) [n]: " fix_mig
        if [[ $fix_mig =~ ^[Yy]$ ]]; then
          kill "$mig_pid" 2>/dev/null
          sleep 2
          kill -9 "$mig_pid" 2>/dev/null 2>&1
          if command -v qm &> /dev/null; then
            qm unlock "$vmid" 2>/dev/null
          else
            sed -i '/^lock:/d' "$conf" 2>/dev/null
          fi
          print_success "Migration aborted and lock cleared for VM ${vmid}"
        fi
      else
        print_info "  No migration process found — lock is stale"
        read -p "    Clear stale migration lock? (y/n) [y]: " clear_mig
        clear_mig=${clear_mig:-y}
        if [[ $clear_mig =~ ^[Yy]$ ]]; then
          if command -v qm &> /dev/null; then
            qm unlock "$vmid" 2>/dev/null
          else
            sed -i '/^lock:/d' "$conf" 2>/dev/null
          fi
          print_success "Migration lock cleared for VM ${vmid}"
        fi
      fi
    fi
  done

  # Also check Redis for stuck migration jobs
  if command -v redis-cli &> /dev/null; then
    local node_id=$(get_node_id)
    local mig_keys=$(redis_cmd KEYS "migration:*:${node_id}*" 2>/dev/null | wc -l)
    if [ "${mig_keys:-0}" -gt 0 ]; then
      found=true
      print_warning "${mig_keys} migration key(s) in Redis"
      read -p "  Clear migration Redis keys? (y/n) [n]: " clear_mig_redis
      if [[ $clear_mig_redis =~ ^[Yy]$ ]]; then
        redis_cmd KEYS "migration:*:${node_id}*" 2>/dev/null | while IFS= read -r key; do
          [ -z "$key" ] && continue
          redis_cmd DEL "$key" > /dev/null 2>&1
        done
        print_success "Migration Redis keys cleared"
      fi
    fi
  fi

  if [ "$found" = false ]; then
    print_success "No stuck migrations found"
  fi
  echo ""
}

function repair_redis_state() {
  check_root
  local node_id=$(get_node_id)

  echo ""
  print_step "Redis state cleanup for node: ${node_id}"
  echo ""

  if ! command -v redis-cli &> /dev/null; then
    print_error "redis-cli not installed"
    return 1
  fi

  local pong=$(redis_cmd PING 2>/dev/null)
  if [ "$pong" != "PONG" ]; then
    print_error "Redis connection failed"
    return 1
  fi

  echo -e "  ${BOLD}What to clean?${NC}"
  echo ""
  echo "  1) Clear stale VM stats (expired/orphaned)"
  echo "  2) Reset node heartbeat and health"
  echo "  3) Flush all job queues for this node"
  echo "  4) Clean everything for this node"
  echo ""
  read -p "  Enter selection [1]: " redis_choice
  redis_choice=${redis_choice:-1}

  case $redis_choice in
    1)
      print_step "Cleaning stale VM stats..."
      local cleaned=0
      while IFS= read -r key; do
        [ -z "$key" ] && continue
        local ttl=$(redis_cmd TTL "$key" 2>/dev/null)
        # Keys with no TTL or negative TTL are stale
        if [ "$ttl" = "-1" ]; then
          redis_cmd DEL "$key" > /dev/null 2>&1
          ((cleaned++))
          print_info "  Removed: ${key} (no TTL)"
        fi
      done < <(redis_cmd KEYS "vm:stats:*" 2>/dev/null)
      print_success "Cleaned ${cleaned} stale VM stats key(s)"
      ;;
    2)
      redis_cmd DEL "node:heartbeat:${node_id}" > /dev/null 2>&1
      redis_cmd DEL "node:health:${node_id}" > /dev/null 2>&1
      print_success "Heartbeat and health keys reset"
      print_info "  Daemon will recreate these on next cycle"
      ;;
    3)
      do_jobs_flush
      ;;
    4)
      print_warning "This will clear ALL data for node ${node_id}"
      read -p "  Are you sure? (yes/no): " confirm
      if [ "$confirm" = "yes" ]; then
        # VM stats
        while IFS= read -r key; do
          [ -z "$key" ] && continue
          redis_cmd DEL "$key" > /dev/null 2>&1
        done < <(redis_cmd KEYS "vm:stats:*" 2>/dev/null)

        # Node keys
        redis_cmd DEL "node:heartbeat:${node_id}" > /dev/null 2>&1
        redis_cmd DEL "node:health:${node_id}" > /dev/null 2>&1

        # Job queues
        local mode=$(get_daemon_mode)
        redis_cmd DEL "macos:jobs:${node_id}" > /dev/null 2>&1
        redis_cmd DEL "pve:jobs:${node_id}" > /dev/null 2>&1
        redis_cmd DEL "jobs:${node_id}" > /dev/null 2>&1

        # Migration keys
        redis_cmd KEYS "migration:*:${node_id}*" 2>/dev/null | while IFS= read -r key; do
          [ -z "$key" ] && continue
          redis_cmd DEL "$key" > /dev/null 2>&1
        done

        print_success "All Redis state cleared for node ${node_id}"
      else
        print_info "  Cancelled"
      fi
      ;;
  esac
  echo ""
}

function force_restart_pve_service() {
  # Restarts a PVE service with timeout and force-kill fallback
  local svc="$1"
  local svc_timeout="${2:-30}"

  echo -n "  Restarting ${svc}..."

  # Try graceful restart with timeout
  if timeout "$svc_timeout" systemctl restart "$svc" 2>/dev/null; then
    sleep 1
    if systemctl is-active --quiet "$svc" 2>/dev/null; then
      echo -e " ${GREEN}OK${NC}"
      return 0
    fi
  fi

  # Graceful restart timed out or failed — force kill
  echo -e " ${YELLOW}timed out${NC}"
  echo -n "  Force-stopping ${svc}..."
  systemctl kill -s SIGKILL "$svc" 2>/dev/null
  sleep 1

  # Clean up stale PID if needed
  local pid_file="/run/${svc}.pid"
  if [ -f "$pid_file" ]; then
    local stale_pid=$(cat "$pid_file" 2>/dev/null)
    if [ -n "$stale_pid" ] && kill -0 "$stale_pid" 2>/dev/null; then
      kill -9 "$stale_pid" 2>/dev/null
      sleep 1
    fi
    rm -f "$pid_file"
  fi

  # Also kill by name if still lingering
  pkill -9 -f "^${svc}" 2>/dev/null || true
  sleep 1

  # Now start fresh
  systemctl reset-failed "$svc" 2>/dev/null
  systemctl start "$svc" 2>/dev/null
  sleep 2
  if systemctl is-active --quiet "$svc" 2>/dev/null; then
    echo -e " ${GREEN}OK (force-restarted)${NC}"
    return 0
  else
    echo -e " ${RED}FAILED${NC}"
    return 1
  fi
}

function repair_pve_cluster() {
  check_root
  print_step "Restarting PVE cluster services..."
  echo ""

  # Clear stale lock files FIRST — these are what cause pvedaemon to hang
  print_step "Pre-clearing stale lock files to prevent hangs..."
  local lock_dirs=("/run/lock/qemu-server" "/var/lock/qemu-server")
  local cleared=0
  for lock_dir in "${lock_dirs[@]}"; do
    [ -d "$lock_dir" ] || continue
    for lock in "${lock_dir}"/*.lock "${lock_dir}"/lock-*.conf; do
      [ -f "$lock" ] || continue
      local lock_name=$(basename "$lock")
      local vmid=""
      if [[ "$lock_name" =~ ^([0-9]+)\.lock$ ]]; then
        vmid="${BASH_REMATCH[1]}"
      elif [[ "$lock_name" =~ ^lock-([0-9]+)\.conf$ ]]; then
        vmid="${BASH_REMATCH[1]}"
      else
        continue
      fi
      if ! pgrep -f "kvm.*-id $vmid" &>/dev/null; then
        rm -f "$lock"
        ((cleared++))
      fi
    done
  done
  if [ "$cleared" -gt 0 ]; then
    print_success "Cleared ${cleared} stale lock file(s)"
  else
    print_info "  No stale lock files to clear"
  fi
  echo ""

  local services=("pve-cluster" "pvedaemon" "pvestatd" "pveproxy")

  for svc in "${services[@]}"; do
    if systemctl is-active --quiet "$svc" 2>/dev/null || systemctl is-enabled --quiet "$svc" 2>/dev/null; then
      force_restart_pve_service "$svc" 30
    else
      echo -e "  ${DIM}${svc} — not installed/enabled${NC}"
    fi
  done

  echo ""

  # Also restart our daemon
  read -p "  Also restart dn-daemon? (y/n) [y]: " restart_dn
  restart_dn=${restart_dn:-y}
  if [[ $restart_dn =~ ^[Yy]$ ]]; then
    do_restart
  fi
  echo ""
}

# ============================================
# Deep Recovery (from toolbox scripts)
# ============================================

function repair_lvm_hung() {
  check_root
  echo ""
  echo -e "  ${BOLD}LVM Hang Recovery${NC}"
  echo -e "  ${DIM}Diagnoses and fixes hung LVM caused by failed disks, hung NFS, or stuck device-mapper${NC}"
  echo ""

  # Step 1: Check for D-state processes (uninterruptible I/O wait)
  print_step "Checking for processes stuck in D state (I/O wait)..."
  local d_procs=$(ps aux 2>/dev/null | awk '$8 ~ /D/ {print $0}')
  if [ -n "$d_procs" ]; then
    print_warning "Found processes blocked on I/O:"
    echo "$d_procs" | while IFS= read -r line; do
      echo -e "    ${DIM}${line}${NC}"
    done
    echo ""

    # Show kernel stacks for blocked processes
    print_info "Checking what's blocking these processes..."
    ps aux 2>/dev/null | awk '$8 ~ /D/ {print $2}' | while read pid; do
      if [ -f "/proc/$pid/stack" ]; then
        local cmd=$(cat "/proc/$pid/comm" 2>/dev/null)
        echo -e "    ${YELLOW}PID ${pid} (${cmd}):${NC}"
        head -10 "/proc/$pid/stack" 2>/dev/null | sed 's/^/      /'
        echo ""
      fi
    done
  else
    print_success "No processes stuck in D state"
  fi

  # Step 2: Check for hung NFS (top cause of LVM hangs)
  print_step "Checking for hung NFS mounts..."
  local hung_nfs=()
  while IFS= read -r line; do
    local mountpoint=$(echo "$line" | awk '{print $3}')
    if ! timeout 3 stat "$mountpoint" &>/dev/null; then
      hung_nfs+=("$mountpoint")
      print_error "HUNG NFS: ${mountpoint}"
    fi
  done < <(mount 2>/dev/null | grep -E 'nfs|nfs4')

  if [ ${#hung_nfs[@]} -eq 0 ]; then
    print_success "No hung NFS mounts"
  else
    print_error "Found ${#hung_nfs[@]} hung NFS mount(s) — THIS IS LIKELY YOUR PROBLEM"
    echo ""
    print_warning "Hung NFS causes LVM to hang because LVM scans all mountpoints"
    read -p "  Lazy unmount hung NFS mounts? (y/n) [y]: " fix_nfs
    fix_nfs=${fix_nfs:-y}
    if [[ $fix_nfs =~ ^[Yy]$ ]]; then
      for mnt in "${hung_nfs[@]}"; do
        echo -n "    Unmounting ${mnt}..."
        if umount -l "$mnt" 2>/dev/null; then
          echo -e " ${GREEN}OK${NC}"
        else
          umount -f "$mnt" 2>/dev/null
          echo -e " ${YELLOW}force${NC}"
        fi
      done
      sleep 2
      # Restart NFS client
      systemctl restart rpc-statd 2>/dev/null
      systemctl restart nfs-common 2>/dev/null
      print_success "NFS cleanup done"
    fi
  fi

  # Step 3: Check for missing block devices in LVM
  echo ""
  print_step "Checking for missing PVs..."
  if timeout 5 pvs --noheadings -o pv_name 2>/dev/null; then
    pvs --noheadings -o pv_name 2>/dev/null | while read pv; do
      pv=$(echo "$pv" | xargs)
      if [ -n "$pv" ] && [ ! -b "$pv" ]; then
        print_error "MISSING PV: ${pv} does not exist!"
      fi
    done
  else
    print_warning "pvs command is hanging — LVM is definitely stuck"
  fi

  # Step 4: Check device-mapper
  echo ""
  print_step "Checking device-mapper responsiveness..."
  if timeout 5 dmsetup status &>/dev/null; then
    print_success "dmsetup is responsive"
  else
    print_error "dmsetup is HANGING — device-mapper is stuck"
  fi

  # Step 5: Check disk I/O
  echo ""
  print_step "Checking disk I/O responsiveness..."
  for dev in /dev/sd[a-z] /dev/nvme[0-9]n[0-9]; do
    [ -b "$dev" ] || continue
    local devname=$(basename "$dev")
    if timeout 5 dd if="$dev" of=/dev/null bs=512 count=1 2>/dev/null; then
      print_success "${devname}: responsive"
    else
      print_error "${devname}: NOT RESPONDING — disk may be failed/hung!"
      local vg_info=$(timeout 5 pvs --noheadings -o vg_name "$dev" 2>/dev/null | xargs)
      if [ -n "$vg_info" ]; then
        print_error "  Part of VG: ${vg_info} — likely causing LVM hang"
      fi
    fi
  done

  # Step 6: Attempt LVM recovery
  echo ""
  print_step "Attempting LVM cache refresh..."
  systemctl stop pvestatd 2>/dev/null
  systemctl stop lvm2-monitor 2>/dev/null

  if timeout 10 vgchange --refresh 2>/dev/null; then
    print_success "VG refresh completed"
  else
    print_warning "VG refresh timed out"
  fi

  if timeout 30 pvscan --cache 2>&1 >/dev/null; then
    print_success "pvscan cache updated"
  else
    print_warning "pvscan still hanging — killing stuck LVM processes..."
    pkill -9 -f "pvscan" 2>/dev/null
    pkill -9 -f "vgscan" 2>/dev/null
    pkill -9 -f "lvscan" 2>/dev/null
    pkill -9 -f "lvs" 2>/dev/null
    pkill -9 -f "vgs" 2>/dev/null
    pkill -9 -f "pvs" 2>/dev/null
    sleep 2
  fi

  # Restart lvmetad if present
  if pgrep lvmetad &>/dev/null; then
    systemctl restart lvm2-lvmetad 2>/dev/null || killall -9 lvmetad 2>/dev/null
    sleep 2
  fi

  # Step 7: Verify
  echo ""
  print_step "Testing LVM responsiveness..."
  local pvs_ok=false vgs_ok=false lvs_ok=false

  echo -n "    pvs: "
  if timeout 15 pvs --noheadings 2>/dev/null >/dev/null; then
    echo -e "${GREEN}OK${NC}"; pvs_ok=true
  else
    echo -e "${RED}TIMEOUT${NC}"
  fi

  echo -n "    vgs: "
  if timeout 15 vgs --noheadings 2>/dev/null >/dev/null; then
    echo -e "${GREEN}OK${NC}"; vgs_ok=true
  else
    echo -e "${RED}TIMEOUT${NC}"
  fi

  echo -n "    lvs: "
  if timeout 15 lvs --noheadings 2>/dev/null >/dev/null; then
    echo -e "${GREEN}OK${NC}"; lvs_ok=true
  else
    echo -e "${RED}TIMEOUT${NC}"
  fi

  echo ""
  if [ "$pvs_ok" = true ] && [ "$vgs_ok" = true ] && [ "$lvs_ok" = true ]; then
    print_success "LVM is now responding!"
    echo ""
    # Restart Proxmox services
    read -p "  Restart Proxmox services? (y/n) [y]: " restart_pve
    restart_pve=${restart_pve:-y}
    if [[ $restart_pve =~ ^[Yy]$ ]]; then
      do_jobs_locks_auto
      timeout 30 systemctl restart lvm2-monitor 2>/dev/null; sleep 1
      force_restart_pve_service pvestatd 30
      force_restart_pve_service pvedaemon 30
      force_restart_pve_service pveproxy 30
      print_success "Services restarted"
    fi
  else
    print_error "LVM is still hanging — manual intervention required"
    echo ""
    echo "    Possible next steps:"
    echo "      1) Remove missing PVs:    vgreduce --removemissing --force <vgname>"
    echo "      2) Force remove DM:       dmsetup remove_all --force  (DANGEROUS)"
    echo "      3) Show blocked tasks:    echo w > /proc/sysrq-trigger && dmesg | tail -30"
    echo "      4) Add LVM filter to skip NFS paths in /etc/lvm/lvm.conf:"
    echo '         filter = [ "a|/dev/sd.*|", "a|/dev/nvme.*|", "a|/dev/dm-.*|", "r|.*|" ]'
    echo "      5) As last resort: schedule maintenance reboot"
  fi
  echo ""
}

function repair_deep_node_recovery() {
  check_root
  echo ""
  echo -e "  ${BOLD}Deep Proxmox Node Recovery${NC}"
  echo -e "  ${DIM}Full diagnostic and recovery — tries simple fixes before invasive ones${NC}"
  echo ""

  print_warning "This will diagnose and attempt to fix:"
  echo "    - Failed/hung Proxmox services"
  echo "    - Cluster/pmxcfs issues"
  echo "    - Storage backend problems"
  echo "    - Stale VM/container locks"
  echo "    - Memory/OOM conditions"
  echo ""
  read -p "  Proceed? (y/n) [y]: " proceed
  proceed=${proceed:-y}
  [[ ! $proceed =~ ^[Yy]$ ]] && return

  local pass=0 warn_count=0 fail_count=0

  # 1. System health
  echo ""
  echo -e "  ${BOLD}[1/6] System Health${NC}"
  echo ""

  # Memory
  if [ "$OS_TYPE" = "linux" ]; then
    local mem_total=$(grep MemTotal /proc/meminfo 2>/dev/null | awk '{print $2}')
    local mem_avail=$(grep MemAvailable /proc/meminfo 2>/dev/null | awk '{print $2}')
    if [ -n "$mem_total" ] && [ -n "$mem_avail" ] && [ "$mem_total" -gt 0 ]; then
      local mem_pct=$(( (mem_total - mem_avail) * 100 / mem_total ))
      if [ "$mem_pct" -ge 95 ]; then
        print_error "Memory CRITICAL: ${mem_pct}% used"; ((fail_count++))
      elif [ "$mem_pct" -ge 85 ]; then
        print_warning "Memory high: ${mem_pct}% used"; ((warn_count++))
      else
        print_success "Memory OK: ${mem_pct}% used"; ((pass++))
      fi
    fi

    # OOM check
    local oom_count=$(dmesg 2>/dev/null | grep -c "Out of memory" || echo "0")
    if [ "$oom_count" -gt 0 ]; then
      print_warning "OOM killer invoked ${oom_count} time(s) since boot"; ((warn_count++))
    fi
  fi

  # 2. Proxmox services
  echo ""
  echo -e "  ${BOLD}[2/6] Proxmox Services${NC}"
  echo ""

  local svc_issues=()
  for svc_info in "pve-cluster:critical" "pvedaemon:critical" "pveproxy:critical" "pvestatd:important" "pvescheduler:important" "corosync:cluster" "pve-ha-lrm:cluster" "pve-ha-crm:cluster"; do
    local svc="${svc_info%%:*}"
    local importance="${svc_info##*:}"

    if ! systemctl list-unit-files 2>/dev/null | grep -q "^${svc}"; then
      continue
    fi

    local status=$(systemctl is-active "$svc" 2>/dev/null)
    case "$status" in
      active)
        print_success "${svc}: running"; ((pass++))
        ;;
      inactive|failed)
        if [ "$importance" = "critical" ]; then
          print_error "${svc}: ${status}"; ((fail_count++))
          svc_issues+=("$svc")
        else
          print_warning "${svc}: ${status}"; ((warn_count++))
          svc_issues+=("$svc")
        fi
        ;;
    esac
  done

  # Web UI check
  if curl -s -k --connect-timeout 5 "https://localhost:8006" &>/dev/null; then
    print_success "Web interface (port 8006): responding"; ((pass++))
  else
    print_warning "Web interface: NOT responding"; ((warn_count++))
    svc_issues+=("pveproxy")
  fi

  # Fix services if needed
  if [ ${#svc_issues[@]} -gt 0 ]; then
    echo ""
    local unique_svcs=($(printf '%s\n' "${svc_issues[@]}" | sort -u))
    print_warning "Services needing attention: ${unique_svcs[*]}"
    read -p "  Restart failed services? (y/n) [y]: " fix_svcs
    fix_svcs=${fix_svcs:-y}
    if [[ $fix_svcs =~ ^[Yy]$ ]]; then
      # Restart in proper dependency order
      for svc in corosync pve-cluster pvedaemon pvestatd pveproxy pvescheduler pve-ha-lrm pve-ha-crm; do
        if printf '%s\n' "${unique_svcs[@]}" | grep -q "^${svc}$"; then
          echo -n "    Restarting ${svc}..."
          if timeout 60 systemctl restart "$svc" 2>/dev/null; then
            echo -e " ${GREEN}OK${NC}"
          else
            echo -e " ${RED}FAILED${NC}"
          fi
          sleep 2
        fi
      done
    fi
  fi

  # 3. Cluster / pmxcfs
  echo ""
  echo -e "  ${BOLD}[3/6] Cluster Status${NC}"
  echo ""

  if [ -f /etc/corosync/corosync.conf ]; then
    # Check /etc/pve
    if mountpoint -q /etc/pve 2>/dev/null; then
      if timeout 5 ls /etc/pve &>/dev/null; then
        print_success "/etc/pve is mounted and accessible"; ((pass++))
      else
        print_error "/etc/pve is mounted but NOT accessible (pmxcfs hung)"; ((fail_count++))
        echo ""
        read -p "  Attempt deep cluster recovery (stop/restart all cluster services)? (y/n) [n]: " deep_cluster
        if [[ $deep_cluster =~ ^[Yy]$ ]]; then
          print_step "Stopping cluster services..."
          for svc in pve-ha-lrm pve-ha-crm pvestatd pveproxy pvedaemon; do
            systemctl stop "$svc" 2>/dev/null
          done
          sleep 2
          systemctl stop pve-cluster 2>/dev/null
          sleep 2
          if pgrep pmxcfs &>/dev/null; then
            pkill -9 pmxcfs 2>/dev/null; sleep 2
          fi
          if mountpoint -q /etc/pve 2>/dev/null; then
            umount -l /etc/pve 2>/dev/null; sleep 1
          fi
          print_step "Restarting cluster services..."
          systemctl start pve-cluster 2>/dev/null; sleep 5
          if timeout 10 ls /etc/pve &>/dev/null; then
            print_success "/etc/pve recovered!"
            systemctl start pvedaemon 2>/dev/null; sleep 2
            systemctl start pveproxy 2>/dev/null
            systemctl start pvestatd 2>/dev/null
            systemctl start pve-ha-lrm 2>/dev/null
            systemctl start pve-ha-crm 2>/dev/null
          else
            print_error "/etc/pve still inaccessible after recovery"
          fi
        fi
      fi
    else
      print_error "/etc/pve is NOT mounted"; ((fail_count++))
    fi

    # Quorum check
    if timeout 5 corosync-quorumtool -s &>/dev/null; then
      local quorate=$(corosync-quorumtool -s 2>/dev/null | grep -i "quorate" | head -1)
      if echo "$quorate" | grep -qi "yes"; then
        print_success "Cluster is quorate"; ((pass++))
      else
        print_error "Cluster is NOT quorate"; ((fail_count++))
      fi
    fi
  else
    print_info "Standalone node (no cluster)"
  fi

  # 4. Storage
  echo ""
  echo -e "  ${BOLD}[4/6] Storage${NC}"
  echo ""

  # Check for hung mounts
  local hung_mounts=0
  while IFS= read -r line; do
    local mountpoint=$(echo "$line" | awk '{print $3}')
    if ! timeout 5 stat "$mountpoint" &>/dev/null; then
      print_error "HUNG MOUNT: ${mountpoint}"; ((fail_count++))
      ((hung_mounts++))
    fi
  done < <(mount 2>/dev/null | grep -E 'nfs|cifs|gluster|ceph')
  [ "$hung_mounts" -eq 0 ] && print_success "No hung network mounts"

  # LVM check
  if command -v vgs &> /dev/null; then
    if timeout 10 vgs --noheadings 2>/dev/null >/dev/null; then
      print_success "LVM is responsive"; ((pass++))
    else
      print_error "LVM commands timed out — storage may be hung"; ((fail_count++))
      read -p "  Run LVM hang recovery? (y/n) [n]: " fix_lvm
      if [[ $fix_lvm =~ ^[Yy]$ ]]; then
        repair_lvm_hung
      fi
    fi
  fi

  # PVE storage backends
  if timeout 15 pvesm status &>/dev/null; then
    local inactive_storage=$(pvesm status 2>/dev/null | tail -n +2 | grep -cv "active" || echo "0")
    if [ "$inactive_storage" -gt 0 ]; then
      print_warning "${inactive_storage} storage backend(s) inactive"; ((warn_count++))
    else
      print_success "All PVE storage backends active"; ((pass++))
    fi
  fi

  # 5. VM locks
  echo ""
  echo -e "  ${BOLD}[5/6] VM Locks & Tasks${NC}"
  echo ""

  local stale_lock_count=$(repair_check_stale_locks)
  if [ "$stale_lock_count" -gt 0 ]; then
    print_warning "${stale_lock_count} stale lock(s) detected"; ((warn_count++))
    read -p "  Clear stale locks? (y/n) [y]: " fix_locks
    fix_locks=${fix_locks:-y}
    if [[ $fix_locks =~ ^[Yy]$ ]]; then
      repair_stale_locks
    fi
  else
    print_success "No stale locks"; ((pass++))
  fi

  local stuck_tasks=$(repair_check_stuck_tasks)
  if [ "$stuck_tasks" -gt 0 ]; then
    print_warning "${stuck_tasks} stuck task(s)"; ((warn_count++))
    read -p "  Clear stuck tasks? (y/n) [y]: " fix_tasks
    fix_tasks=${fix_tasks:-y}
    if [[ $fix_tasks =~ ^[Yy]$ ]]; then
      repair_stuck_tasks
    fi
  else
    print_success "No stuck tasks"; ((pass++))
  fi

  # 6. DartNode daemon
  echo ""
  echo -e "  ${BOLD}[6/6] DartNode Daemon${NC}"
  echo ""

  if is_service_running; then
    print_success "Daemon is running"; ((pass++))
    local node_id=$(get_node_id)
    if command -v redis-cli &> /dev/null; then
      local hb_ttl=$(redis_cmd TTL "node:heartbeat:${node_id}" 2>/dev/null)
      if [ "$hb_ttl" != "-2" ]; then
        print_success "Heartbeat alive (TTL ${hb_ttl}s)"; ((pass++))
      else
        print_warning "Heartbeat missing — daemon may need restart"; ((warn_count++))
      fi
    fi
  else
    print_warning "Daemon not running"; ((warn_count++))
    read -p "  Start daemon? (y/n) [y]: " start_dn
    start_dn=${start_dn:-y}
    if [[ $start_dn =~ ^[Yy]$ ]]; then
      repair_daemon_service
    fi
  fi

  # Final summary
  echo ""
  echo -e "  ${DIM}──────────────────────────────────────${NC}"
  local total=$((pass + warn_count + fail_count))
  local score_color="${GREEN}"
  [ "$fail_count" -gt 0 ] && score_color="${RED}"
  [ "$warn_count" -gt 0 ] && [ "$fail_count" -eq 0 ] && score_color="${YELLOW}"

  if [ "$fail_count" -eq 0 ] && [ "$warn_count" -eq 0 ]; then
    echo -e "  ${GREEN}${BOLD}Node is healthy — all checks passed.${NC}"
  else
    echo -e "  ${BOLD}Score:${NC} ${score_color}${pass}/${total}${NC} (${GREEN}${pass} pass${NC}, ${YELLOW}${warn_count} warn${NC}, ${RED}${fail_count} fail${NC})"
    if [ "$fail_count" -gt 0 ]; then
      echo -e "  ${RED}Some critical issues remain.${NC}"
    fi
  fi
  echo ""
}

function repair_hung_nfs() {
  check_root
  echo ""
  echo -e "  ${BOLD}Hung NFS Mount Recovery${NC}"
  echo ""

  local hung_mounts=()
  while IFS= read -r line; do
    local device=$(echo "$line" | awk '{print $1}')
    local mountpoint=$(echo "$line" | awk '{print $3}')
    if ! timeout 5 stat "$mountpoint" &>/dev/null; then
      hung_mounts+=("${device}|${mountpoint}")
      print_error "HUNG: ${mountpoint} (${device})"
    else
      print_success "OK: ${mountpoint}"
    fi
  done < <(mount 2>/dev/null | grep -E 'nfs|nfs4')

  if [ ${#hung_mounts[@]} -eq 0 ]; then
    print_success "No hung NFS mounts detected"
    echo ""
    return
  fi

  echo ""
  echo -e "  ${BOLD}Recovery Options:${NC}"
  echo -e "    ${CYAN}l${NC} = Lazy unmount (safe — detaches immediately, cleanup later)"
  echo -e "    ${YELLOW}f${NC} = Force unmount (may cause stale file handle errors)"
  echo -e "    ${DIM}s${NC} = Skip"
  echo ""

  for entry in "${hung_mounts[@]}"; do
    local device="${entry%%|*}"
    local mountpoint="${entry##*|}"

    echo -e "  ${BOLD}${mountpoint}${NC} (${device})"

    # Check what's using the mount
    if command -v lsof &>/dev/null; then
      local procs=$(timeout 5 lsof "$mountpoint" 2>/dev/null | tail -n +2 | wc -l || echo "?")
      if [ "$procs" != "?" ] && [ "$procs" -gt 0 ]; then
        print_warning "  ${procs} process(es) may be using this mount"
      fi
    fi

    read -p "    Action [l=lazy, f=force, s=skip]: " nfs_action
    case "$nfs_action" in
      l|L)
        if umount -l "$mountpoint" 2>/dev/null; then
          print_success "  Lazy unmount initiated"
        else
          print_error "  Lazy unmount failed"
        fi
        ;;
      f|F)
        print_warning "  Force unmount may cause errors for processes using this mount"
        read -p "    Confirm force unmount? (y/n): " confirm_force
        if [[ $confirm_force =~ ^[Yy]$ ]]; then
          if command -v fuser &>/dev/null; then
            timeout 10 fuser -km "$mountpoint" 2>/dev/null
            sleep 2
          fi
          if umount -f "$mountpoint" 2>/dev/null; then
            print_success "  Force unmount successful"
          else
            print_warning "  Force failed, falling back to lazy..."
            umount -l "$mountpoint" 2>/dev/null
          fi
        fi
        ;;
      *)
        print_info "  Skipped"
        ;;
    esac
    echo ""
  done

  # Restart NFS client services
  read -p "  Restart NFS client services? (y/n) [y]: " restart_nfs
  restart_nfs=${restart_nfs:-y}
  if [[ $restart_nfs =~ ^[Yy]$ ]]; then
    systemctl restart nfs-common 2>/dev/null
    systemctl restart rpc-statd 2>/dev/null
    print_success "NFS client services restarted"
  fi
  echo ""
}

function repair_disk_health() {
  check_root
  echo ""
  echo -e "  ${BOLD}Disk Health Check${NC}"
  echo ""

  # SMART checks
  if command -v smartctl &>/dev/null; then
    print_step "Checking SMART status..."
    for disk in /dev/sd[a-z] /dev/nvme[0-9]n[0-9]; do
      [ -b "$disk" ] || continue
      local disk_name=$(basename "$disk")

      if ! smartctl -i "$disk" 2>/dev/null | grep -q "SMART support is: Enabled"; then
        print_info "${disk_name}: SMART not available"
        continue
      fi

      local smart_health=$(smartctl -H "$disk" 2>/dev/null | grep -iE "SMART overall-health|SMART Health Status")
      if echo "$smart_health" | grep -qi "PASSED\|OK"; then
        print_success "${disk_name}: SMART PASSED"
      elif echo "$smart_health" | grep -qi "FAILED"; then
        print_error "${disk_name}: SMART FAILED — disk may be failing!"
        smartctl -A "$disk" 2>/dev/null | grep -i "pre-fail" | grep -v "^$" | head -5 | sed 's/^/      /'
      else
        # Check individual attributes
        local reallocated=$(smartctl -A "$disk" 2>/dev/null | grep -i "Reallocated_Sector" | awk '{print $NF}')
        local pending=$(smartctl -A "$disk" 2>/dev/null | grep -i "Current_Pending_Sector" | awk '{print $NF}')
        local uncorrectable=$(smartctl -A "$disk" 2>/dev/null | grep -i "Offline_Uncorrectable" | awk '{print $NF}')

        local has_issues=false
        [ -n "$reallocated" ] && [ "$reallocated" -gt 0 ] 2>/dev/null && { print_warning "${disk_name}: ${reallocated} reallocated sectors"; has_issues=true; }
        [ -n "$pending" ] && [ "$pending" -gt 0 ] 2>/dev/null && { print_warning "${disk_name}: ${pending} pending sectors"; has_issues=true; }
        [ -n "$uncorrectable" ] && [ "$uncorrectable" -gt 0 ] 2>/dev/null && { print_error "${disk_name}: ${uncorrectable} uncorrectable sectors!"; has_issues=true; }
        [ "$has_issues" = false ] && print_success "${disk_name}: SMART attributes OK"
      fi
    done
  else
    print_warning "smartctl not available — install smartmontools for SMART checks"
  fi

  # Disk I/O errors in kernel log
  echo ""
  print_step "Checking kernel messages for disk errors..."
  local io_errors=$(dmesg 2>/dev/null | grep -iE "I/O error|medium error|sense error|failed command|sector error" | tail -10)
  if [ -n "$io_errors" ]; then
    print_warning "Disk I/O errors detected:"
    echo "$io_errors" | while IFS= read -r line; do
      echo -e "    ${DIM}${line}${NC}"
    done
  else
    print_success "No disk I/O errors in kernel messages"
  fi

  # Missing block devices from fstab
  echo ""
  print_step "Checking for missing devices from /etc/fstab..."
  if [ -f /etc/fstab ]; then
    local missing=0
    while IFS= read -r line; do
      [[ "$line" =~ ^[[:space:]]*# ]] && continue
      [ -z "$line" ] && continue
      local device=$(echo "$line" | awk '{print $1}')
      local mountpoint=$(echo "$line" | awk '{print $2}')
      local fstype=$(echo "$line" | awk '{print $3}')
      [[ "$fstype" =~ ^(swap|tmpfs|devpts|sysfs|proc|none)$ ]] && continue
      [ "$mountpoint" = "none" ] && continue

      local actual_device="$device"
      if [[ "$device" =~ ^UUID= ]]; then
        actual_device=$(blkid -U "${device#UUID=}" 2>/dev/null)
      elif [[ "$device" =~ ^LABEL= ]]; then
        actual_device=$(blkid -L "${device#LABEL=}" 2>/dev/null)
      fi

      if [ -z "$actual_device" ] || [ ! -b "$actual_device" ]; then
        print_error "MISSING: ${device} (mountpoint: ${mountpoint})"
        ((missing++))
      fi
    done < /etc/fstab
    [ "$missing" -eq 0 ] && print_success "All fstab devices present"
  fi

  # RAID check
  if command -v mdadm &>/dev/null; then
    local md_found=false
    for md in /dev/md*; do
      [ -b "$md" ] || continue
      md_found=true
      local md_name=$(basename "$md")
      local md_status=$(mdadm --detail "$md" 2>/dev/null | grep "State :" | awk -F: '{print $2}' | xargs)
      if [[ "$md_status" =~ clean|active ]]; then
        print_success "${md_name}: ${md_status}"
      elif [[ "$md_status" =~ degraded ]]; then
        print_error "${md_name}: DEGRADED — missing disk member!"
        mdadm --detail "$md" 2>/dev/null | grep -E "removed|faulty|spare" | sed 's/^/      /'
      else
        print_warning "${md_name}: ${md_status}"
      fi
    done
  fi

  echo ""
}

function repair_unmounted_filesystems() {
  check_root
  echo ""
  echo -e "  ${BOLD}Unmounted Filesystem Recovery${NC}"
  echo ""

  if [ ! -f /etc/fstab ]; then
    print_info "No /etc/fstab found"
    return
  fi

  local unmounted=()
  while IFS= read -r line; do
    [[ "$line" =~ ^[[:space:]]*# ]] && continue
    [ -z "$line" ] && continue
    local device=$(echo "$line" | awk '{print $1}')
    local mountpoint=$(echo "$line" | awk '{print $2}')
    local fstype=$(echo "$line" | awk '{print $3}')
    local options=$(echo "$line" | awk '{print $4}')

    [[ "$fstype" =~ ^(swap|tmpfs|devpts|sysfs|proc|none)$ ]] && continue
    [ "$mountpoint" = "none" ] && continue
    [[ "$options" =~ noauto ]] && continue
    [[ "$fstype" =~ ^(nfs|nfs4|cifs|glusterfs|ceph)$ ]] && continue

    if ! mountpoint -q "$mountpoint" 2>/dev/null; then
      unmounted+=("${device}|${mountpoint}|${fstype}")
      print_warning "UNMOUNTED: ${mountpoint} (${device}, ${fstype})"
    fi
  done < /etc/fstab

  if [ ${#unmounted[@]} -eq 0 ]; then
    print_success "All local filesystems are mounted"
    echo ""
    return
  fi

  echo ""
  read -p "  Attempt to remount unmounted filesystems? (y/n) [n]: " do_remount
  [[ ! $do_remount =~ ^[Yy]$ ]] && return

  for entry in "${unmounted[@]}"; do
    IFS='|' read -r device mountpoint fstype <<< "$entry"
    echo ""
    print_step "Processing: ${mountpoint}"

    # Resolve device
    local actual_device="$device"
    if [[ "$device" =~ ^UUID= ]]; then
      actual_device=$(blkid -U "${device#UUID=}" 2>/dev/null)
      if [ -z "$actual_device" ]; then
        print_error "  Cannot resolve ${device} — disk may be missing"
        continue
      fi
    elif [[ "$device" =~ ^LABEL= ]]; then
      actual_device=$(blkid -L "${device#LABEL=}" 2>/dev/null)
      if [ -z "$actual_device" ]; then
        print_error "  Cannot resolve ${device} — disk may be missing"
        continue
      fi
    fi

    if [ ! -b "$actual_device" ]; then
      print_error "  Device ${actual_device} not found"
      continue
    fi
    print_success "  Device ${actual_device} exists"

    # Filesystem check (read-only, non-destructive)
    if [[ "$fstype" =~ ^(ext[234])$ ]]; then
      print_info "  Running read-only filesystem check..."
      if e2fsck -n "$actual_device" &>/dev/null; then
        print_success "  Filesystem is clean"
      else
        print_warning "  Filesystem may have errors — consider: e2fsck -f ${actual_device}"
        read -p "    Mount anyway? (y/n) [n]: " mount_dirty
        [[ ! $mount_dirty =~ ^[Yy]$ ]] && continue
      fi
    elif [ "$fstype" = "xfs" ]; then
      if xfs_repair -n "$actual_device" &>/dev/null; then
        print_success "  XFS filesystem appears healthy"
      else
        print_warning "  XFS may have issues — consider: xfs_repair ${actual_device}"
      fi
    fi

    # Create mountpoint if needed
    [ ! -d "$mountpoint" ] && mkdir -p "$mountpoint"

    read -p "    Mount ${mountpoint}? (y/n) [y]: " do_mount
    do_mount=${do_mount:-y}
    if [[ $do_mount =~ ^[Yy]$ ]]; then
      if mount "$mountpoint" 2>/dev/null; then
        print_success "  Mounted successfully"
        if timeout 5 ls "$mountpoint" &>/dev/null; then
          local usage=$(df -h "$mountpoint" 2>/dev/null | tail -1 | awk '{print $5}')
          print_info "  Usage: ${usage}"
        fi
      else
        print_error "  Mount failed — check: dmesg | tail -20"
      fi
    fi
  done
  echo ""
}

function do_repair_macos() {
  check_root
  local node_id=$(get_node_id)
  local fixed=0
  local issues=0

  echo -e "  ${BOLD}macOS Node Repair${NC}"
  echo ""

  # 1. Daemon service
  echo -e "  ${BOLD}[1/4] Checking daemon service...${NC}"
  if ! is_service_running; then
    ((issues++))
    print_warning "Daemon is not running"
    read -p "  Repair and start? (y/n) [y]: " fix_daemon
    fix_daemon=${fix_daemon:-y}
    if [[ $fix_daemon =~ ^[Yy]$ ]]; then
      repair_daemon_service
      ((fixed++))
    fi
  else
    print_success "Daemon is running"
  fi

  # 2. Tart
  echo ""
  echo -e "  ${BOLD}[2/4] Checking Tart...${NC}"
  if command -v tart &> /dev/null; then
    print_success "Tart is installed: $(tart --version 2>&1 | head -1)"
  else
    ((issues++))
    print_error "Tart not found"
    read -p "  Install Tart via Homebrew? (y/n) [y]: " fix_tart
    fix_tart=${fix_tart:-y}
    if [[ $fix_tart =~ ^[Yy]$ ]]; then
      brew install cirruslabs/cli/tart
      ((fixed++))
    fi
  fi

  # 3. Redis
  echo ""
  echo -e "  ${BOLD}[3/4] Checking Redis...${NC}"
  if command -v redis-cli &> /dev/null; then
    local pong=$(redis_cmd PING 2>/dev/null)
    if [ "$pong" = "PONG" ]; then
      print_success "Redis connected"
    else
      ((issues++))
      print_error "Redis connection failed"
    fi
  else
    print_warning "redis-cli not installed"
  fi

  # 4. Stuck jobs
  echo ""
  echo -e "  ${BOLD}[4/4] Checking job queue...${NC}"
  if command -v redis-cli &> /dev/null; then
    local job_len=$(redis_cmd LLEN "macos:jobs:${node_id}" 2>/dev/null)
    if [ "${job_len:-0}" -gt 10 ]; then
      ((issues++))
      print_warning "Job queue has ${job_len} entries (may be backed up)"
      read -p "  Clear job queue? (y/n) [n]: " clear_jobs
      if [[ $clear_jobs =~ ^[Yy]$ ]]; then
        do_jobs_flush
        ((fixed++))
      fi
    else
      print_success "Job queue: ${job_len:-0} entries"
    fi
  fi

  # Summary
  echo ""
  echo -e "  ${DIM}──────────────────────────────────────${NC}"
  if [ "$issues" -eq 0 ]; then
    echo -e "  ${GREEN}${BOLD}Node is healthy — no issues found.${NC}"
  else
    echo -e "  ${BOLD}Found ${issues} issue(s), fixed ${fixed}.${NC}"
  fi
  echo ""
}

# ============================================
# Job Queue Management (dartnode jobs)
# ============================================

function do_jobs() {
  local subcmd="${ARGS[0]:-}"
  local arg="${ARGS[1]:-}"

  case "$subcmd" in
    ""|list)
      do_jobs_list
      ;;
    clear)
      do_jobs_clear_stale
      ;;
    flush)
      do_jobs_flush
      ;;
    inspect)
      do_jobs_inspect "$arg"
      ;;
    retry)
      do_jobs_retry "$arg"
      ;;
    pve|tasks)
      do_jobs_pve
      ;;
    locks)
      do_jobs_locks
      ;;
    *)
      echo "Usage: dartnode jobs [list|clear|flush|inspect|retry|pve|locks]"
      echo ""
      echo "  list               Show all queued jobs (Redis + PVE tasks)"
      echo "  clear              Remove stale/stuck jobs (older than 1h)"
      echo "  flush              Remove ALL jobs from queue"
      echo "  inspect <index>    Show full details of job at queue index"
      echo "  retry <index>      Re-queue a failed job"
      echo "  pve                Show and manage active Proxmox tasks"
      echo "  locks              Show and clear VM lock files"
      ;;
  esac
}

function do_jobs_list() {
  local mode=$(get_daemon_mode)
  local node_id=$(get_node_id)

  echo ""
  echo -e "  ${BOLD}Job Queue${NC} — ${node_id} (${mode})"
  echo ""

  if ! command -v redis-cli &> /dev/null; then
    print_error "redis-cli not installed"
    return 1
  fi

  local pong=$(redis_cmd PING 2>/dev/null)
  if [ "$pong" != "PONG" ]; then
    print_error "Redis connection failed"
    return 1
  fi

  # Check multiple possible queue key patterns
  local queue_keys=()
  if [ "$mode" = "macos" ]; then
    queue_keys+=("macos:jobs:${node_id}")
  elif [ "$mode" = "pve" ]; then
    queue_keys+=("pve:jobs:${node_id}")
  fi
  queue_keys+=("jobs:${node_id}")

  local total_jobs=0

  for qkey in "${queue_keys[@]}"; do
    local qlen=$(redis_cmd LLEN "$qkey" 2>/dev/null)
    [ -z "$qlen" ] || [ "$qlen" = "0" ] && continue

    total_jobs=$((total_jobs + qlen))

    echo -e "  ${CYAN}Queue: ${qkey}${NC} (${qlen} jobs)"
    echo ""

    printf "  ${DIM}%-5s %-12s %-20s %-10s %-22s %s${NC}\n" \
      "Idx" "Type" "VM/Target" "Status" "Created" "Details"
    printf "  ${DIM}%-5s %-12s %-20s %-10s %-22s %s${NC}\n" \
      "─────" "────────────" "────────────────────" "──────────" "──────────────────────" "──────────"

    local idx=0
    while [ "$idx" -lt "$qlen" ]; do
      local job=$(redis_cmd LINDEX "$qkey" "$idx" 2>/dev/null)
      if [ -n "$job" ] && [ "$job" != "(nil)" ]; then
        local job_type job_vm job_status job_created job_detail
        if command -v python3 &> /dev/null; then
          eval $(python3 -c "
import json, sys
try:
    j = json.loads('''${job}''')
    print('job_type=\"' + str(j.get('type', j.get('action', '?'))) + '\"')
    print('job_vm=\"' + str(j.get('vmName', j.get('vm_name', j.get('vmid', j.get('target', '?'))))) + '\"')
    print('job_status=\"' + str(j.get('status', 'queued')) + '\"')
    print('job_created=\"' + str(j.get('created', j.get('timestamp', j.get('queued_at', '?')))) + '\"')
    p = j.get('params', {})
    if isinstance(p, dict):
        detail_parts = []
        for k in ['image', 'cpu', 'memory', 'disk', 'reason']:
            if k in p:
                detail_parts.append(k + '=' + str(p[k]))
        print('job_detail=\"' + ', '.join(detail_parts[:3]) + '\"')
    else:
        print('job_detail=\"\"')
except:
    print('job_type=\"?\"')
    print('job_vm=\"?\"')
    print('job_status=\"?\"')
    print('job_created=\"?\"')
    print('job_detail=\"\"')
" 2>/dev/null)
        else
          job_type=$(echo "$job" | grep -o '"type":"[^"]*"' | cut -d'"' -f4)
          job_type=${job_type:-$(echo "$job" | grep -o '"action":"[^"]*"' | cut -d'"' -f4)}
          job_vm=$(echo "$job" | grep -o '"vmName":"[^"]*"' | cut -d'"' -f4)
          job_vm=${job_vm:-$(echo "$job" | grep -o '"vmid":"[^"]*"' | cut -d'"' -f4)}
          job_status="queued"
          job_created="?"
          job_detail=""
        fi

        # Truncate long values
        [ ${#job_vm} -gt 18 ] && job_vm="${job_vm:0:17}…"

        local status_col="${DIM}"
        case "$job_status" in
          queued)     status_col="${CYAN}" ;;
          running)    status_col="${GREEN}" ;;
          failed)     status_col="${RED}" ;;
          completed)  status_col="${DIM}" ;;
        esac

        printf "  %-5s %-12s %-20s ${status_col}%-10s${NC} %-22s %s\n" \
          "$idx" "${job_type:-?}" "${job_vm:-?}" "${job_status:-?}" "${job_created:-?}" "${job_detail:-}"
      fi
      ((idx++))
    done
    echo ""
  done

  if [ "$total_jobs" -eq 0 ]; then
    print_success "No jobs in queue"
    echo ""
  fi

  # Also check for active/processing keys
  local processing_keys=$(redis_cmd KEYS "job:active:${node_id}*" 2>/dev/null | wc -l)
  if [ "${processing_keys:-0}" -gt 0 ]; then
    echo -e "  ${YELLOW}Active/processing jobs: ${processing_keys}${NC}"
    echo ""
  fi

  # Check failed jobs list
  local failed_len=$(redis_cmd LLEN "jobs:failed:${node_id}" 2>/dev/null)
  if [ "${failed_len:-0}" -gt 0 ]; then
    echo -e "  ${RED}Failed jobs: ${failed_len}${NC}"
    echo -e "  ${DIM}Use 'dartnode jobs inspect <index>' to view details${NC}"
    echo ""
  fi

  # In PVE mode, also show active Proxmox tasks and lock files
  if [ "$mode" = "pve" ] && [ -d "/var/log/pve/tasks" ]; then
    local pve_task_count=0
    local cutoff=$(date -d '30 minutes ago' +%s 2>/dev/null || date -v-30M +%s 2>/dev/null || echo "0")

    for task_dir in /var/log/pve/tasks/active/*/; do
      [ -d "$task_dir" ] || continue
      for task_file in "${task_dir}"*; do
        [ -f "$task_file" ] || continue
        ((pve_task_count++))
      done
    done

    if [ "$pve_task_count" -gt 0 ]; then
      echo -e "  ${YELLOW}Active PVE tasks: ${pve_task_count}${NC}"
      echo -e "  ${DIM}Use 'dartnode jobs pve' to view and manage Proxmox tasks${NC}"
      echo ""
    fi

    # Count stale lock files
    local lock_count=0
    local lock_dirs=("/run/lock/qemu-server" "/var/lock/qemu-server")
    for lock_dir in "${lock_dirs[@]}"; do
      [ -d "$lock_dir" ] || continue
      for lock in "${lock_dir}"/*.lock "${lock_dir}"/lock-*.conf; do
        [ -f "$lock" ] || continue
        ((lock_count++))
      done
    done

    if [ "$lock_count" -gt 0 ]; then
      echo -e "  ${YELLOW}VM lock files: ${lock_count}${NC}"
      echo -e "  ${DIM}Use 'dartnode jobs locks' to view and clear lock files${NC}"
      echo ""
    fi
  fi
}

function do_jobs_pve() {
  local mode=$(get_daemon_mode)

  if [ "$mode" != "pve" ]; then
    print_info "PVE tasks are only available in PVE mode (current: ${mode})"
    return
  fi

  echo ""
  echo -e "  ${BOLD}Proxmox Active Tasks${NC}"
  echo ""

  if [ ! -d "/var/log/pve/tasks" ]; then
    print_info "PVE task directory not found"
    return
  fi

  local found=false
  local now=$(date +%s)

  printf "  ${DIM}%-42s %-8s %-14s %s${NC}\n" \
    "Task ID" "Age" "Status" "Details"
  printf "  ${DIM}%-42s %-8s %-14s %s${NC}\n" \
    "──────────────────────────────────────────" "────────" "──────────────" "──────────"

  for task_dir in /var/log/pve/tasks/active/*/; do
    [ -d "$task_dir" ] || continue
    for task_file in "${task_dir}"*; do
      [ -f "$task_file" ] || continue
      found=true
      local task_name=$(basename "$task_file")
      local task_mtime=$(stat -c %Y "$task_file" 2>/dev/null || stat -f %m "$task_file" 2>/dev/null || echo "0")
      local age_secs=$(( now - task_mtime ))
      local age_str=""

      if [ "$age_secs" -ge 3600 ]; then
        age_str="$(( age_secs / 3600 ))h $(( (age_secs % 3600) / 60 ))m"
      else
        age_str="$(( age_secs / 60 ))m"
      fi

      # Determine if stuck (>30min)
      local status_col="${GREEN}"
      local status="running"
      if [ "$age_secs" -ge 1800 ]; then
        status_col="${RED}"
        status="STUCK"
      elif [ "$age_secs" -ge 600 ]; then
        status_col="${YELLOW}"
        status="slow"
      fi

      # Try to read task type from the file content
      local task_detail=""
      task_detail=$(head -1 "$task_file" 2>/dev/null | tr -d '\n' | cut -c1-40)

      printf "  %-42s %-8s ${status_col}%-14s${NC} %s\n" \
        "$task_name" "$age_str" "$status" "${task_detail:-}"
    done
  done

  if [ "$found" = false ]; then
    print_success "No active PVE tasks"
    echo ""
    return
  fi

  echo ""

  # Also show lock files
  local lock_count=0
  local lock_dirs=("/run/lock/qemu-server" "/var/lock/qemu-server")
  for lock_dir in "${lock_dirs[@]}"; do
    [ -d "$lock_dir" ] || continue
    for lock in "${lock_dir}"/*.lock "${lock_dir}"/lock-*.conf; do
      [ -f "$lock" ] || continue
      ((lock_count++))
      local lock_name=$(basename "$lock")
      local lock_mtime=$(stat -c %Y "$lock" 2>/dev/null || stat -f %m "$lock" 2>/dev/null || echo "0")
      local lock_age=$(( (now - lock_mtime) / 60 ))
      print_info "  Lock file: ${lock} (${lock_age}m old)"
    done
  done

  echo ""
  echo -e "  ${BOLD}Actions:${NC}"
  echo "   1) Clear stuck tasks (restart pvedaemon)"
  echo "   2) Clear stale lock files"
  echo "   3) Clear both tasks and locks"
  echo "   4) Cancel"
  echo ""
  read -p "  Enter selection [4]: " pve_action
  pve_action=${pve_action:-4}

  case $pve_action in
    1)
      print_step "Clearing stale locks before restart..."
      do_jobs_locks_auto
      force_restart_pve_service pvedaemon 30
      ;;
    2)
      do_jobs_locks
      ;;
    3)
      do_jobs_locks
      force_restart_pve_service pvedaemon 30
      ;;
    *)
      print_info "Cancelled"
      ;;
  esac
  echo ""
}

function do_jobs_locks_auto() {
  # Non-interactive: clears all stale lock files without prompting
  # Used as a pre-step before service restarts to prevent hangs
  local cleared=0
  local lock_dirs=("/run/lock/qemu-server" "/var/lock/qemu-server")
  for lock_dir in "${lock_dirs[@]}"; do
    [ -d "$lock_dir" ] || continue
    for lock in "${lock_dir}"/*.lock "${lock_dir}"/lock-*.conf; do
      [ -f "$lock" ] || continue
      local lock_name=$(basename "$lock")
      local vmid=""
      if [[ "$lock_name" =~ ^([0-9]+)\.lock$ ]]; then
        vmid="${BASH_REMATCH[1]}"
      elif [[ "$lock_name" =~ ^lock-([0-9]+)\.conf$ ]]; then
        vmid="${BASH_REMATCH[1]}"
      else
        continue
      fi
      if ! pgrep -f "kvm.*-id $vmid" &>/dev/null; then
        rm -f "$lock"
        ((cleared++))
        # Also clear PVE config lock
        local pve_conf="/etc/pve/qemu-server/${vmid}.conf"
        if [ -f "$pve_conf" ] && grep -q "^lock:" "$pve_conf" 2>/dev/null; then
          if command -v qm &> /dev/null; then
            qm unlock "$vmid" 2>/dev/null
          else
            sed -i '/^lock:/d' "$pve_conf" 2>/dev/null
          fi
        fi
      fi
    done
  done
  if [ "$cleared" -gt 0 ]; then
    print_success "Auto-cleared ${cleared} stale lock file(s)"
  fi
}

function do_jobs_locks() {
  echo ""
  echo -e "  ${BOLD}VM Lock Files${NC}"
  echo ""

  local found=false
  local now=$(date +%s)
  local lock_dirs=("/run/lock/qemu-server" "/var/lock/qemu-server")

  for lock_dir in "${lock_dirs[@]}"; do
    [ -d "$lock_dir" ] || continue
    for lock in "${lock_dir}"/*.lock "${lock_dir}"/lock-*.conf; do
      [ -f "$lock" ] || continue
      local lock_name=$(basename "$lock")
      local vmid=""

      if [[ "$lock_name" =~ ^([0-9]+)\.lock$ ]]; then
        vmid="${BASH_REMATCH[1]}"
      elif [[ "$lock_name" =~ ^lock-([0-9]+)\.conf$ ]]; then
        vmid="${BASH_REMATCH[1]}"
      else
        continue
      fi

      local lock_mtime=$(stat -c %Y "$lock" 2>/dev/null || stat -f %m "$lock" 2>/dev/null || echo "0")
      local age_min=$(( (now - lock_mtime) / 60 ))
      local is_running=false

      if pgrep -f "kvm.*-id $vmid" &>/dev/null; then
        is_running=true
      fi

      found=true
      if [ "$is_running" = true ]; then
        print_info "  VM ${vmid}: ${lock} (${age_min}m old) — ${GREEN}process running${NC}"
      else
        print_warning "VM ${vmid}: ${lock} (${age_min}m old) — no QEMU process"
        read -p "    Remove lock file? (y/n) [y]: " remove_choice
        remove_choice=${remove_choice:-y}
        if [[ $remove_choice =~ ^[Yy]$ ]]; then
          rm -f "$lock"
          print_success "  Lock file removed for VM ${vmid}"

          # Also clear PVE config lock if present
          local pve_conf="/etc/pve/qemu-server/${vmid}.conf"
          if [ -f "$pve_conf" ] && grep -q "^lock:" "$pve_conf" 2>/dev/null; then
            if command -v qm &> /dev/null; then
              qm unlock "$vmid" 2>/dev/null
              print_success "  PVE config lock also cleared for VM ${vmid}"
            else
              sed -i '/^lock:/d' "$pve_conf" 2>/dev/null
              print_success "  PVE config lock line removed for VM ${vmid}"
            fi
          fi
        fi
      fi
    done
  done

  # Also check PVE config locks
  local conf_dir="/etc/pve/qemu-server"
  if [ -d "$conf_dir" ]; then
    for conf in "${conf_dir}"/*.conf; do
      [ -f "$conf" ] || continue
      if grep -q "^lock:" "$conf" 2>/dev/null; then
        local vmid=$(basename "$conf" .conf)
        local lock_type=$(grep "^lock:" "$conf" | sed 's/^lock:[[:space:]]*//')

        if [ -S "/var/run/qemu-server/${vmid}.qmp" ]; then
          continue
        fi

        found=true
        print_warning "VM ${vmid}: PVE config lock '${lock_type}' (no running process)"
        read -p "    Remove PVE config lock? (y/n) [y]: " remove_conf_lock
        remove_conf_lock=${remove_conf_lock:-y}
        if [[ $remove_conf_lock =~ ^[Yy]$ ]]; then
          if command -v qm &> /dev/null; then
            qm unlock "$vmid" 2>/dev/null
            print_success "  Lock removed from VM ${vmid}"
          else
            sed -i '/^lock:/d' "$conf" 2>/dev/null
            print_success "  Lock line removed from config"
          fi
        fi
      fi
    done
  fi

  if [ "$found" = false ]; then
    print_success "No lock files found"
  fi
  echo ""
}

function repair_check_stale_jobs() {
  local mode=$(get_daemon_mode)
  local node_id=$(get_node_id)
  local count=0

  if ! command -v redis-cli &> /dev/null; then
    echo "0"
    return
  fi

  local queue_keys=()
  if [ "$mode" = "macos" ]; then
    queue_keys+=("macos:jobs:${node_id}")
  elif [ "$mode" = "pve" ]; then
    queue_keys+=("pve:jobs:${node_id}")
  fi
  queue_keys+=("jobs:${node_id}")

  local cutoff=$(date -d '1 hour ago' +%s 2>/dev/null || date -v-1H +%s 2>/dev/null || echo "0")

  for qkey in "${queue_keys[@]}"; do
    local qlen=$(redis_cmd LLEN "$qkey" 2>/dev/null)
    [ -z "$qlen" ] || [ "$qlen" = "0" ] && continue

    local idx=0
    while [ "$idx" -lt "$qlen" ]; do
      local job=$(redis_cmd LINDEX "$qkey" "$idx" 2>/dev/null)
      if [ -n "$job" ] && command -v python3 &> /dev/null; then
        local ts=$(python3 -c "
import json
try:
    j = json.loads('''${job}''')
    ts = j.get('timestamp', j.get('created', j.get('queued_at', '')))
    if isinstance(ts, (int, float)):
        print(int(ts))
    else:
        print('0')
except:
    print('0')
" 2>/dev/null)
        if [ "${ts:-0}" -gt 0 ] && [ "$cutoff" -gt 0 ] && [ "$ts" -lt "$cutoff" ]; then
          ((count++))
        fi
      fi
      ((idx++))
    done
  done
  echo "$count"
}

function do_jobs_clear_stale() {
  local mode=$(get_daemon_mode)
  local node_id=$(get_node_id)

  echo ""
  print_step "Clearing stale jobs (older than 1 hour)..."
  echo ""

  if ! command -v redis-cli &> /dev/null; then
    print_error "redis-cli not installed"
    return 1
  fi

  local queue_keys=()
  if [ "$mode" = "macos" ]; then
    queue_keys+=("macos:jobs:${node_id}")
  elif [ "$mode" = "pve" ]; then
    queue_keys+=("pve:jobs:${node_id}")
  fi
  queue_keys+=("jobs:${node_id}")

  local cutoff=$(date -d '1 hour ago' +%s 2>/dev/null || date -v-1H +%s 2>/dev/null || echo "0")
  local removed=0

  for qkey in "${queue_keys[@]}"; do
    local qlen=$(redis_cmd LLEN "$qkey" 2>/dev/null)
    [ -z "$qlen" ] || [ "$qlen" = "0" ] && continue

    # Read all jobs, rebuild without stale ones
    local keep_jobs=()
    local idx=0
    while [ "$idx" -lt "$qlen" ]; do
      local job=$(redis_cmd LINDEX "$qkey" "$idx" 2>/dev/null)
      local is_stale=false

      if [ -n "$job" ] && command -v python3 &> /dev/null; then
        local ts=$(python3 -c "
import json
try:
    j = json.loads('''${job}''')
    ts = j.get('timestamp', j.get('created', j.get('queued_at', '')))
    if isinstance(ts, (int, float)):
        print(int(ts))
    else:
        print('0')
except:
    print('0')
" 2>/dev/null)
        if [ "${ts:-0}" -gt 0 ] && [ "$cutoff" -gt 0 ] && [ "$ts" -lt "$cutoff" ]; then
          is_stale=true
        fi
      fi

      if [ "$is_stale" = true ]; then
        ((removed++))
        local job_type="?"
        if command -v python3 &> /dev/null; then
          job_type=$(python3 -c "import json; print(json.loads('''${job}''').get('type','?'))" 2>/dev/null || echo "?")
        fi
        print_info "  Removed stale job: ${job_type}"
      else
        keep_jobs+=("$job")
      fi
      ((idx++))
    done

    # Rebuild queue with only fresh jobs
    if [ "$removed" -gt 0 ]; then
      redis_cmd DEL "$qkey" > /dev/null 2>&1
      for kept in "${keep_jobs[@]}"; do
        redis_cmd RPUSH "$qkey" "$kept" > /dev/null 2>&1
      done
    fi
  done

  # Also clean active job keys
  local active_cleaned=0
  while IFS= read -r key; do
    [ -z "$key" ] && continue
    redis_cmd DEL "$key" > /dev/null 2>&1
    ((active_cleaned++))
  done < <(redis_cmd KEYS "job:active:${node_id}*" 2>/dev/null)

  if [ "$active_cleaned" -gt 0 ]; then
    print_info "  Cleared ${active_cleaned} stale active job marker(s)"
  fi

  print_success "Removed ${removed} stale job(s)"
  echo ""
}

function do_jobs_flush() {
  local mode=$(get_daemon_mode)
  local node_id=$(get_node_id)

  echo ""
  print_step "Flushing ALL jobs for node: ${node_id}"
  echo ""

  if ! command -v redis-cli &> /dev/null; then
    print_error "redis-cli not installed"
    return 1
  fi

  # Count total before
  local queue_keys=()
  if [ "$mode" = "macos" ]; then
    queue_keys+=("macos:jobs:${node_id}")
  elif [ "$mode" = "pve" ]; then
    queue_keys+=("pve:jobs:${node_id}")
  fi
  queue_keys+=("jobs:${node_id}")

  local total=0
  for qkey in "${queue_keys[@]}"; do
    local qlen=$(redis_cmd LLEN "$qkey" 2>/dev/null)
    total=$((total + ${qlen:-0}))
  done

  if [ "$total" -eq 0 ]; then
    print_success "Job queues already empty"
    echo ""
    return
  fi

  print_warning "${total} job(s) will be permanently removed"
  read -p "  Are you sure? (yes/no): " confirm
  if [ "$confirm" = "yes" ]; then
    for qkey in "${queue_keys[@]}"; do
      redis_cmd DEL "$qkey" > /dev/null 2>&1
    done

    # Also clear active markers and failed list
    redis_cmd KEYS "job:active:${node_id}*" 2>/dev/null | while IFS= read -r key; do
      [ -z "$key" ] && continue
      redis_cmd DEL "$key" > /dev/null 2>&1
    done
    redis_cmd DEL "jobs:failed:${node_id}" > /dev/null 2>&1

    print_success "Flushed ${total} job(s)"
  else
    print_info "Cancelled"
  fi
  echo ""
}

function do_jobs_inspect() {
  local index="${1:-0}"
  local mode=$(get_daemon_mode)
  local node_id=$(get_node_id)

  echo ""
  echo -e "  ${BOLD}Job Details${NC} — index ${index}"
  echo ""

  if ! command -v redis-cli &> /dev/null; then
    print_error "redis-cli not installed"
    return 1
  fi

  # Find the right queue key
  local qkey=""
  if [ "$mode" = "macos" ]; then
    qkey="macos:jobs:${node_id}"
  elif [ "$mode" = "pve" ]; then
    qkey="pve:jobs:${node_id}"
  fi

  # Fallback
  local qlen=$(redis_cmd LLEN "$qkey" 2>/dev/null)
  if [ "${qlen:-0}" -eq 0 ]; then
    qkey="jobs:${node_id}"
    qlen=$(redis_cmd LLEN "$qkey" 2>/dev/null)
  fi

  if [ "${qlen:-0}" -eq 0 ]; then
    print_info "Queue is empty"
    echo ""
    return
  fi

  if [ "$index" -ge "${qlen:-0}" ]; then
    print_error "Index ${index} out of range (queue has ${qlen} entries)"
    echo ""
    return
  fi

  local job=$(redis_cmd LINDEX "$qkey" "$index" 2>/dev/null)
  if [ -z "$job" ] || [ "$job" = "(nil)" ]; then
    print_error "Job not found at index ${index}"
    echo ""
    return
  fi

  # Pretty-print JSON
  if command -v python3 &> /dev/null; then
    echo "$job" | python3 -m json.tool 2>/dev/null | sed 's/^/  /'
  else
    echo "  ${job}"
  fi

  echo ""
  echo -e "  ${DIM}Queue: ${qkey}, Index: ${index}${NC}"
  echo ""
}

function do_jobs_retry() {
  local index="${1:-}"
  local mode=$(get_daemon_mode)
  local node_id=$(get_node_id)

  if [ -z "$index" ]; then
    echo "Usage: dartnode jobs retry <index>"
    echo ""
    echo "  Re-queues a job from the failed list back to the active queue."
    echo "  Use 'dartnode jobs list' to see available indices."
    return 1
  fi

  echo ""

  if ! command -v redis-cli &> /dev/null; then
    print_error "redis-cli not installed"
    return 1
  fi

  local failed_key="jobs:failed:${node_id}"
  local failed_len=$(redis_cmd LLEN "$failed_key" 2>/dev/null)

  if [ "${failed_len:-0}" -eq 0 ]; then
    print_info "No failed jobs to retry"
    echo ""
    return
  fi

  if [ "$index" -ge "${failed_len:-0}" ]; then
    print_error "Index ${index} out of range (${failed_len} failed jobs)"
    echo ""
    return
  fi

  local job=$(redis_cmd LINDEX "$failed_key" "$index" 2>/dev/null)
  if [ -z "$job" ] || [ "$job" = "(nil)" ]; then
    print_error "Job not found"
    echo ""
    return
  fi

  # Determine target queue
  local target_key=""
  if [ "$mode" = "macos" ]; then
    target_key="macos:jobs:${node_id}"
  elif [ "$mode" = "pve" ]; then
    target_key="pve:jobs:${node_id}"
  else
    target_key="jobs:${node_id}"
  fi

  # Reset status to queued if possible
  if command -v python3 &> /dev/null; then
    job=$(python3 -c "
import json
j = json.loads('''${job}''')
j['status'] = 'queued'
j.pop('error', None)
j.pop('failed_at', None)
print(json.dumps(j))
" 2>/dev/null || echo "$job")
  fi

  redis_cmd RPUSH "$target_key" "$job" > /dev/null 2>&1
  print_success "Job re-queued to ${target_key}"

  # Show what was retried
  local job_type="?"
  if command -v python3 &> /dev/null; then
    job_type=$(python3 -c "import json; print(json.loads('''${job}''').get('type','?'))" 2>/dev/null || echo "?")
  fi
  print_info "  Type: ${job_type}"
  echo ""
}

# ============================================
# Parse Arguments
# ============================================

# Parse --token flag from any position
COMMAND=""
ARGS=()
while [[ $# -gt 0 ]]; do
  case $1 in
    --token)
      REGISTRATION_TOKEN="$2"
      shift 2
      ;;
    --token=*)
      REGISTRATION_TOKEN="${1#*=}"
      shift
      ;;
    --)
      shift
      ;;
    *)
      if [ -z "$COMMAND" ]; then
        COMMAND="$1"
      else
        ARGS+=("$1")
      fi
      shift
      ;;
  esac
done

# Default command
COMMAND=${COMMAND:-help}

# ============================================
# Main
# ============================================

header

case ${COMMAND} in
  install)
    do_install
    ;;
  setup)
    check_root
    if [ ! -d "${DN_DAEMON_CONFIG_DIR}" ]; then
      print_error "Daemon not installed. Run 'dartnode install' first."
      exit 1
    fi
    if [ -n "$REGISTRATION_TOKEN" ]; then
      do_token_setup "$REGISTRATION_TOKEN"
    else
      do_interactive_setup
    fi
    ;;
  config)
    do_config
    ;;
  start)
    do_start
    ;;
  stop)
    do_stop
    ;;
  restart)
    do_restart
    ;;
  status)
    do_status
    ;;
  log|logs)
    do_log
    ;;
  update)
    do_update
    ;;
  version|-v|--version)
    do_version
    ;;
  uninstall|remove)
    do_uninstall
    ;;
  migrate)
    do_migrate
    ;;
  net|network)
    do_net
    ;;
  disk|drive|led)
    do_disk
    ;;
  vms|vm)
    do_vms
    ;;
  diag|diagnostics|check)
    do_diag
    ;;
  redis)
    do_redis_diag
    ;;
  storage)
    do_storage
    ;;
  fw|firewall)
    do_fw
    ;;
  qmp)
    do_qmp
    ;;
  repair|fix)
    do_repair
    ;;
  jobs|job)
    do_jobs
    ;;
  smtp-gate)
    do_smtp_gate_action
    ;;
  help|--help|-h)
    help
    ;;
  *)
    echo "Unknown command: ${COMMAND}"
    echo ""
    help
    exit 1
    ;;
esac
