Compare commits

..
Author SHA1 Message Date
Josh Patterson 84cd966736 FIX: prevent so-sensor-clean runaway loop and concurrent instances
The cleanup loop had no progress check or pass limit, so once /nsm was over
threshold with nothing left to reclaim it spun at full speed, writing 1.1 GB /
12.5M lines to sensor_clean.log in five hours. Stop when a pass removes
nothing, when a pass frees no space, or at MAX_PASSES, and drop the per-pass
"no old files" logging in favor of one actionable line.

Replace the pgrep guard with flock -n. A find|while read subshell inherits the
parent's argv, so pgrep -cf counted one instance as hundreds; it was also
check-then-act, which let cron stack up overlapping runs.

Paths now derive from SENSOR_DIR with env-overridable LOG/LOCK so the
over-threshold path can be tested against a scratch filesystem.
2026-09-14 12:14:13 -04:00
5 changed files with 78 additions and 74 deletions

No files matched your search

+78 -47
View File
@@ -8,21 +8,37 @@
# Elastic License 2.0.
SENSOR_DIR='/nsm'
SENSOR_DIR="${SENSOR_DIR:-/nsm}"
CRIT_DISK_USAGE=90
CUR_USAGE=$(df -P $SENSOR_DIR | tail -1 | awk '{print $5}' | tr -d %)
LOG="/opt/so/log/sensor_clean.log"
TODAY=$(date -u "+%Y-%m-%d")
LOG="${LOG:-/opt/so/log/sensor_clean.log}"
LOCK="${LOCK:-/var/tmp/so-sensor-clean.lock}"
MAX_PASSES=100
ZEEK_LOGS="$SENSOR_DIR/zeek/logs"
STRELKA_FILES="$SENSOR_DIR/strelka/processed"
SURICATA_LOGS="$SENSOR_DIR/suricata"
PCAPS="$SENSOR_DIR/pcapout"
log() {
echo "$(date) - $*" >>"$LOG"
}
disk_usage() {
df -P "$SENSOR_DIR" | tail -1 | awk '{print $5}' | tr -d %
}
disk_avail() {
df -P "$SENSOR_DIR" | tail -1 | awk '{print $4}'
}
# sets REMOVED=1 if anything was actually deleted
clean() {
## find the oldest Zeek logs directory
OLDEST_DIR=$(ls /nsm/zeek/logs/ | grep -v "current" | grep -v "stats" | grep -v "packetloss" | grep -v "zeek_clean" | sort | head -n 1)
if [ -z "$OLDEST_DIR" -o "$OLDEST_DIR" == ".." -o "$OLDEST_DIR" == "." ]; then
echo "$(date) - No old Zeek logs available to clean up in /nsm/zeek/logs/" >>$LOG
#exit 0
else
echo "$(date) - Removing directory: /nsm/zeek/logs/$OLDEST_DIR" >>$LOG
rm -rf /nsm/zeek/logs/"$OLDEST_DIR"
OLDEST_DIR=$(ls "$ZEEK_LOGS" 2>/dev/null | grep -v "current" | grep -v "stats" | grep -v "packetloss" | grep -v "zeek_clean" | sort | head -n 1)
if [ -n "$OLDEST_DIR" ]; then
log "Removing directory: $ZEEK_LOGS/$OLDEST_DIR"
rm -rf "$ZEEK_LOGS/$OLDEST_DIR"
REMOVED=1
fi
## Remarking for now, as we are moving extracted files to /nsm/strelka/processed
@@ -43,58 +59,73 @@ clean() {
#fi
## Clean up Zeek extracted files processed by Strelka
STRELKA_FILES='/nsm/strelka/processed'
OLDEST_STRELKA=$(find $STRELKA_FILES -type f -printf '%T+ %p\n' | sort -n | head -n 1)
if [ -z "$OLDEST_STRELKA" -o "$OLDEST_STRELKA" == ".." -o "$OLDEST_STRELKA" == "." ]; then
echo "$(date) - No old files available to clean up in $STRELKA_FILES" >>$LOG
else
OLDEST_STRELKA=$(find "$STRELKA_FILES" -type f -printf '%T+ %p\n' 2>/dev/null | sort -n | head -n 1)
if [ -n "$OLDEST_STRELKA" ]; then
OLDEST_STRELKA_DATE=$(echo $OLDEST_STRELKA | awk '{print $1}' | cut -d+ -f1)
OLDEST_STRELKA_FILE=$(echo $OLDEST_STRELKA | awk '{print $2}')
echo "$(date) - Removing extracted files for $OLDEST_STRELKA_DATE" >>$LOG
find $STRELKA_FILES -type f -printf '%T+ %p\n' | grep $OLDEST_STRELKA_DATE | awk '{print $2}' | while read FILE; do
echo "$(date) - Removing file: $FILE" >>$LOG
log "Removing extracted files for $OLDEST_STRELKA_DATE"
REMOVED=1
find "$STRELKA_FILES" -type f -printf '%T+ %p\n' 2>/dev/null | grep $OLDEST_STRELKA_DATE | awk '{print $2}' | while read FILE; do
log "Removing file: $FILE"
rm -f "$FILE"
done
fi
## Clean up Suricata log files
SURICATA_LOGS='/nsm/suricata'
OLDEST_SURICATA=$(find $SURICATA_LOGS -type f -printf '%T+ %p\n' | sort -n | head -n 1)
if [[ -z "$OLDEST_SURICATA" ]] || [[ "$OLDEST_SURICATA" == ".." ]] || [[ "$OLDEST_SURICATA" == "." ]]; then
echo "$(date) - No old files available to clean up in $SURICATA_LOGS" >>$LOG
else
OLDEST_SURICATA=$(find "$SURICATA_LOGS" -type f -printf '%T+ %p\n' 2>/dev/null | sort -n | head -n 1)
if [ -n "$OLDEST_SURICATA" ]; then
OLDEST_SURICATA_DATE=$(echo $OLDEST_SURICATA | awk '{print $1}' | cut -d+ -f1)
OLDEST_SURICATA_FILE=$(echo $OLDEST_SURICATA | awk '{print $2}')
echo "$(date) - Removing logs for $OLDEST_SURICATA_DATE" >>$LOG
find $SURICATA_LOGS -type f -printf '%T+ %p\n' | grep $OLDEST_SURICATA_DATE | awk '{print $2}' | while read FILE; do
echo "$(date) - Removing file: $FILE" >>$LOG
log "Removing logs for $OLDEST_SURICATA_DATE"
REMOVED=1
find "$SURICATA_LOGS" -type f -printf '%T+ %p\n' 2>/dev/null | grep $OLDEST_SURICATA_DATE | awk '{print $2}' | while read FILE; do
log "Removing file: $FILE"
rm -f "$FILE"
done
fi
## Clean up extracted pcaps
PCAPS='/nsm/pcapout'
OLDEST_PCAP=$(find $PCAPS -type f -printf '%T+ %p\n' | sort -n | head -n 1)
if [ -z "$OLDEST_PCAP" -o "$OLDEST_PCAP" == ".." -o "$OLDEST_PCAP" == "." ]; then
echo "$(date) - No old files available to clean up in $PCAPS" >>$LOG
else
OLDEST_PCAP=$(find "$PCAPS" -type f -printf '%T+ %p\n' 2>/dev/null | sort -n | head -n 1)
if [ -n "$OLDEST_PCAP" ]; then
OLDEST_PCAP_DATE=$(echo $OLDEST_PCAP | awk '{print $1}' | cut -d+ -f1)
OLDEST_PCAP_FILE=$(echo $OLDEST_PCAP | awk '{print $2}')
echo "$(date) - Removing extracted files for $OLDEST_PCAP_DATE" >>$LOG
find $PCAPS -type f -printf '%T+ %p\n' | grep $OLDEST_PCAP_DATE | awk '{print $2}' | while read FILE; do
echo "$(date) - Removing file: $FILE" >>$LOG
log "Removing extracted files for $OLDEST_PCAP_DATE"
REMOVED=1
find "$PCAPS" -type f -printf '%T+ %p\n' 2>/dev/null | grep $OLDEST_PCAP_DATE | awk '{print $2}' | while read FILE; do
log "Removing file: $FILE"
rm -f "$FILE"
done
fi
}
# Check to see if we are already running
NUM_RUNNING=$(pgrep -cf "/bin/bash /usr/sbin/so-sensor-clean")
[ "$NUM_RUNNING" -gt 1 ] && echo "$(date) - $NUM_RUNNING sensor clean script processes running...exiting." >>$LOG && exit 0
if [ "$CUR_USAGE" -gt "$CRIT_DISK_USAGE" ]; then
while [ "$CUR_USAGE" -gt "$CRIT_DISK_USAGE" ]; do
clean
CUR_USAGE=$(df -P $SENSOR_DIR | tail -1 | awk '{print $5}' | tr -d %)
done
# Only one instance at a time; the lock is the fd, so it releases on any exit
exec 9>"$LOCK" || exit 1
if ! flock -n 9; then
log "another so-sensor-clean is already running (lock $LOCK held); exiting"
exit 0
fi
CUR_USAGE=$(disk_usage)
[ "$CUR_USAGE" -gt "$CRIT_DISK_USAGE" ] || exit 0
log "$SENSOR_DIR at ${CUR_USAGE}% (threshold ${CRIT_DISK_USAGE}%); starting cleanup"
PASS=0
while [ "$CUR_USAGE" -gt "$CRIT_DISK_USAGE" ]; do
PASS=$((PASS + 1))
if [ "$PASS" -gt "$MAX_PASSES" ]; then
log "stopping after $MAX_PASSES passes; $SENSOR_DIR still at ${CUR_USAGE}%"
break
fi
REMOVED=0
BEFORE=$(disk_avail)
clean
CUR_USAGE=$(disk_usage)
if [ "$REMOVED" -eq 0 ]; then
log "nothing left to remove in $ZEEK_LOGS, $STRELKA_FILES, $SURICATA_LOGS, $PCAPS; $SENSOR_DIR still at ${CUR_USAGE}% - space is consumed outside of NSM cleanup scope"
break
fi
if [ "$(disk_avail)" -le "$BEFORE" ]; then
log "pass $PASS freed no space; $SENSOR_DIR still at ${CUR_USAGE}% - stopping until next run"
break
fi
done
-1
View File
@@ -18,7 +18,6 @@ zeek:
StatsLogEnable: 0
StatsLogExpireInterval: 0
StatusCmdShowAll: 0
StopWait: 1
CrashExpireInterval: 0
SitePolicyScripts: local.zeek
LogDir: /nsm/zeek/logs
-10
View File
@@ -9,19 +9,9 @@
include:
- zeek.sostatus
# Stop first so the entrypoint's SIGTERM trap can archive the final logs; docker_container.absent
# with force is a 'docker rm -f', which never delivers SIGTERM. force stays so the state still
# converges if the stop overruns.
so-zeek_stopped:
docker_container.stopped:
- name: so-zeek
- error_on_absent: False
so-zeek:
docker_container.absent:
- force: True
- require:
- docker_container: so-zeek_stopped
so-zeek_so-status.disabled:
file.comment:
-4
View File
@@ -19,10 +19,6 @@ so-zeek:
- restart_policy: unless-stopped
- start: True
- privileged: True
# Docker's default 10s grace is not enough for the entrypoint's SIGTERM trap to run
# 'zeekctl stop' and let StopWait archive the final logs. Overrunning it means SIGKILL,
# which strands those logs in spool/tmp and marks every node crashed on the next start.
- stop_timeout: 180
{% if DOCKERMERGED.containers['so-zeek'].ulimits %}
- ulimits:
{% for ULIMIT in DOCKERMERGED.containers['so-zeek'].ulimits %}
-12
View File
@@ -99,18 +99,6 @@ zeek:
regexFailureMessage: You must enter a whole number of days, or 0 to keep crash directories forever.
helpLink: zeek
advanced: True
StopWait:
description: >-
Set to 1 to make "zeekctl stop" wait for the final logs to be archived instead of
letting that finish in the background. Security Onion stops Zeek by stopping its
container, so anything still running in the background is killed when the container
exits - without this, the last logs of each run are stranded unarchived in
/nsm/zeek/spool/tmp and never reach Elasticsearch. It is read only for that reason.
regex: ^[01]$
regexFailureMessage: You must enter 0 or 1.
helpLink: zeek
advanced: True
readonly: True
MinDiskSpace:
description: >-
Percentage of free disk space below which ZeekControl reports a warning, or 0 to disable the check