diff --git a/setup/so-functions b/setup/so-functions index bf95ea9d8..ca58dbbcb 100755 --- a/setup/so-functions +++ b/setup/so-functions @@ -1547,8 +1547,6 @@ reinstall_init() { local salt_services=( "salt-minion" ) fi - local service_retry_count=20 - { # remove all of root's cronjobs crontab -r -u root @@ -1563,31 +1561,48 @@ reinstall_init() { salt-call state.apply ca.remove -linfo --local --file-root=../salt - # Kill any salt processes (safely) + # Stop salt services and force-kill any lingering salt processes (including orphans + # from an earlier reinstall attempt where the unit file is gone but processes survive) + # so dnf remove salt can run cleanly for service in "${salt_services[@]}"; do - # Stop the service in the background so we can exit after a certain amount of time if check_service_status "$service"; then - systemctl stop "$service" & + info "Stopping $service via systemctl" + systemctl stop "$service" fi - local pid=$! - - local count=0 - while check_service_status "$service"; do - if [[ $count -gt $service_retry_count ]]; then - echo "Could not stop $service after 1 minute, exiting setup." - - # Stop the systemctl process trying to kill the service, show user a message, then exit setup - kill -9 $pid - fail_setup - fi - - sleep 5 - ((count++)) - done done + # Unconditionally force-kill any remaining salt binaries — these may be orphaned + # from a prior aborted reinstall (no unit file, so systemctl can't see them). + for salt_bin in salt-master salt-minion salt-call salt-cloud; do + if pgrep -f "/usr/bin/${salt_bin}" > /dev/null 2>&1; then + info "Force-killing lingering $salt_bin processes" + pkill -9 -ef "/usr/bin/${salt_bin}" 2>/dev/null + fi + done + # Catch stray `salt` CLI children from saltutil.kill_all_jobs / state.apply invocations + pkill -9 -ef "/usr/bin/python3 /bin/salt" 2>/dev/null + + # Give the kernel a moment to reap the killed processes before dnf removes the binaries + local kill_wait=0 + while pgrep -f "/usr/bin/salt-" > /dev/null 2>&1; do + if [[ $kill_wait -gt 10 ]]; then + info "Salt processes still present after SIGKILL + 10s wait; proceeding anyway" + pgrep -af "/usr/bin/salt-" | while read -r line; do info " lingering: $line"; done + break + fi + sleep 1 + ((kill_wait++)) + done + + # Clear the 'failed' state SIGKILL left on the units before removing the package + systemctl reset-failed salt-master.service salt-minion.service 2>/dev/null || true + # Remove all salt configs - rm -rf /etc/salt/engines/* /etc/salt/grains /etc/salt/master /etc/salt/master.d/* /etc/salt/minion /etc/salt/minion.d/* /etc/salt/pki/* /etc/salt/proxy /etc/salt/proxy.d/* /var/cache/salt/ + dnf -y remove salt + rm -rf /etc/salt/ /var/cache/salt/ + + # Drop systemd's in-memory references to the now-removed units + systemctl daemon-reload if command -v docker &> /dev/null; then # Stop and remove all so-* containers so files can be changed with more safety