Compare commits

..
Author SHA1 Message Date
Josh Patterson 56e3e44d04 Merge remote-tracking branch 'origin/3/dev' into zeekrestart 2026-09-11 15:06:28 -04:00
Josh Patterson 9652a2053b Stop the Zeek container gracefully
zeekctl's post-terminate archives the final logs in the background and returns
immediately unless StopWait is set, so the container exits and takes the archiving
with it, stranding unarchived logs in /nsm/zeek/spool/tmp on every restart. Docker's
default 10s grace is also too tight for the entrypoint's SIGTERM trap; overrunning it
means SIGKILL and crash directories on the next start.

Both are needed. StopWait alone gives the stop more work to do inside the same 10s
window, which was measured ending in SIGKILL with logs stranded in the spool.

disabled.sls used docker rm -f, which never delivers SIGTERM, so stop the container
before removing it.

Reported in discussion #16174.
2026-09-10 13:20:20 -04:00
5 changed files with 74 additions and 78 deletions

No files matched your search

+47 -78
View File
@@ -8,37 +8,21 @@
# Elastic License 2.0.
SENSOR_DIR="${SENSOR_DIR:-/nsm}"
SENSOR_DIR='/nsm'
CRIT_DISK_USAGE=90
LOG="${LOG:-/opt/so/log/sensor_clean.log}"
LOCK="${LOCK:-/var/tmp/so-sensor-clean.lock}"
MAX_PASSES=100
CUR_USAGE=$(df -P $SENSOR_DIR | tail -1 | awk '{print $5}' | tr -d %)
LOG="/opt/so/log/sensor_clean.log"
TODAY=$(date -u "+%Y-%m-%d")
ZEEK_LOGS="$SENSOR_DIR/zeek/logs"
STRELKA_FILES="$SENSOR_DIR/strelka/processed"
SURICATA_LOGS="$SENSOR_DIR/suricata"
PCAPS="$SENSOR_DIR/pcapout"
log() {
echo "$(date) - $*" >>"$LOG"
}
disk_usage() {
df -P "$SENSOR_DIR" | tail -1 | awk '{print $5}' | tr -d %
}
disk_avail() {
df -P "$SENSOR_DIR" | tail -1 | awk '{print $4}'
}
# sets REMOVED=1 if anything was actually deleted
clean() {
## find the oldest Zeek logs directory
OLDEST_DIR=$(ls "$ZEEK_LOGS" 2>/dev/null | grep -v "current" | grep -v "stats" | grep -v "packetloss" | grep -v "zeek_clean" | sort | head -n 1)
if [ -n "$OLDEST_DIR" ]; then
log "Removing directory: $ZEEK_LOGS/$OLDEST_DIR"
rm -rf "$ZEEK_LOGS/$OLDEST_DIR"
REMOVED=1
OLDEST_DIR=$(ls /nsm/zeek/logs/ | grep -v "current" | grep -v "stats" | grep -v "packetloss" | grep -v "zeek_clean" | sort | head -n 1)
if [ -z "$OLDEST_DIR" -o "$OLDEST_DIR" == ".." -o "$OLDEST_DIR" == "." ]; then
echo "$(date) - No old Zeek logs available to clean up in /nsm/zeek/logs/" >>$LOG
#exit 0
else
echo "$(date) - Removing directory: /nsm/zeek/logs/$OLDEST_DIR" >>$LOG
rm -rf /nsm/zeek/logs/"$OLDEST_DIR"
fi
## Remarking for now, as we are moving extracted files to /nsm/strelka/processed
@@ -59,73 +43,58 @@ clean() {
#fi
## Clean up Zeek extracted files processed by Strelka
OLDEST_STRELKA=$(find "$STRELKA_FILES" -type f -printf '%T+ %p\n' 2>/dev/null | sort -n | head -n 1)
if [ -n "$OLDEST_STRELKA" ]; then
STRELKA_FILES='/nsm/strelka/processed'
OLDEST_STRELKA=$(find $STRELKA_FILES -type f -printf '%T+ %p\n' | sort -n | head -n 1)
if [ -z "$OLDEST_STRELKA" -o "$OLDEST_STRELKA" == ".." -o "$OLDEST_STRELKA" == "." ]; then
echo "$(date) - No old files available to clean up in $STRELKA_FILES" >>$LOG
else
OLDEST_STRELKA_DATE=$(echo $OLDEST_STRELKA | awk '{print $1}' | cut -d+ -f1)
log "Removing extracted files for $OLDEST_STRELKA_DATE"
REMOVED=1
find "$STRELKA_FILES" -type f -printf '%T+ %p\n' 2>/dev/null | grep $OLDEST_STRELKA_DATE | awk '{print $2}' | while read FILE; do
log "Removing file: $FILE"
OLDEST_STRELKA_FILE=$(echo $OLDEST_STRELKA | awk '{print $2}')
echo "$(date) - Removing extracted files for $OLDEST_STRELKA_DATE" >>$LOG
find $STRELKA_FILES -type f -printf '%T+ %p\n' | grep $OLDEST_STRELKA_DATE | awk '{print $2}' | while read FILE; do
echo "$(date) - Removing file: $FILE" >>$LOG
rm -f "$FILE"
done
fi
## Clean up Suricata log files
OLDEST_SURICATA=$(find "$SURICATA_LOGS" -type f -printf '%T+ %p\n' 2>/dev/null | sort -n | head -n 1)
if [ -n "$OLDEST_SURICATA" ]; then
SURICATA_LOGS='/nsm/suricata'
OLDEST_SURICATA=$(find $SURICATA_LOGS -type f -printf '%T+ %p\n' | sort -n | head -n 1)
if [[ -z "$OLDEST_SURICATA" ]] || [[ "$OLDEST_SURICATA" == ".." ]] || [[ "$OLDEST_SURICATA" == "." ]]; then
echo "$(date) - No old files available to clean up in $SURICATA_LOGS" >>$LOG
else
OLDEST_SURICATA_DATE=$(echo $OLDEST_SURICATA | awk '{print $1}' | cut -d+ -f1)
log "Removing logs for $OLDEST_SURICATA_DATE"
REMOVED=1
find "$SURICATA_LOGS" -type f -printf '%T+ %p\n' 2>/dev/null | grep $OLDEST_SURICATA_DATE | awk '{print $2}' | while read FILE; do
log "Removing file: $FILE"
OLDEST_SURICATA_FILE=$(echo $OLDEST_SURICATA | awk '{print $2}')
echo "$(date) - Removing logs for $OLDEST_SURICATA_DATE" >>$LOG
find $SURICATA_LOGS -type f -printf '%T+ %p\n' | grep $OLDEST_SURICATA_DATE | awk '{print $2}' | while read FILE; do
echo "$(date) - Removing file: $FILE" >>$LOG
rm -f "$FILE"
done
fi
## Clean up extracted pcaps
OLDEST_PCAP=$(find "$PCAPS" -type f -printf '%T+ %p\n' 2>/dev/null | sort -n | head -n 1)
if [ -n "$OLDEST_PCAP" ]; then
PCAPS='/nsm/pcapout'
OLDEST_PCAP=$(find $PCAPS -type f -printf '%T+ %p\n' | sort -n | head -n 1)
if [ -z "$OLDEST_PCAP" -o "$OLDEST_PCAP" == ".." -o "$OLDEST_PCAP" == "." ]; then
echo "$(date) - No old files available to clean up in $PCAPS" >>$LOG
else
OLDEST_PCAP_DATE=$(echo $OLDEST_PCAP | awk '{print $1}' | cut -d+ -f1)
log "Removing extracted files for $OLDEST_PCAP_DATE"
REMOVED=1
find "$PCAPS" -type f -printf '%T+ %p\n' 2>/dev/null | grep $OLDEST_PCAP_DATE | awk '{print $2}' | while read FILE; do
log "Removing file: $FILE"
OLDEST_PCAP_FILE=$(echo $OLDEST_PCAP | awk '{print $2}')
echo "$(date) - Removing extracted files for $OLDEST_PCAP_DATE" >>$LOG
find $PCAPS -type f -printf '%T+ %p\n' | grep $OLDEST_PCAP_DATE | awk '{print $2}' | while read FILE; do
echo "$(date) - Removing file: $FILE" >>$LOG
rm -f "$FILE"
done
fi
}
# Only one instance at a time; the lock is the fd, so it releases on any exit
exec 9>"$LOCK" || exit 1
if ! flock -n 9; then
log "another so-sensor-clean is already running (lock $LOCK held); exiting"
exit 0
# Check to see if we are already running
NUM_RUNNING=$(pgrep -cf "/bin/bash /usr/sbin/so-sensor-clean")
[ "$NUM_RUNNING" -gt 1 ] && echo "$(date) - $NUM_RUNNING sensor clean script processes running...exiting." >>$LOG && exit 0
if [ "$CUR_USAGE" -gt "$CRIT_DISK_USAGE" ]; then
while [ "$CUR_USAGE" -gt "$CRIT_DISK_USAGE" ]; do
clean
CUR_USAGE=$(df -P $SENSOR_DIR | tail -1 | awk '{print $5}' | tr -d %)
done
fi
CUR_USAGE=$(disk_usage)
[ "$CUR_USAGE" -gt "$CRIT_DISK_USAGE" ] || exit 0
log "$SENSOR_DIR at ${CUR_USAGE}% (threshold ${CRIT_DISK_USAGE}%); starting cleanup"
PASS=0
while [ "$CUR_USAGE" -gt "$CRIT_DISK_USAGE" ]; do
PASS=$((PASS + 1))
if [ "$PASS" -gt "$MAX_PASSES" ]; then
log "stopping after $MAX_PASSES passes; $SENSOR_DIR still at ${CUR_USAGE}%"
break
fi
REMOVED=0
BEFORE=$(disk_avail)
clean
CUR_USAGE=$(disk_usage)
if [ "$REMOVED" -eq 0 ]; then
log "nothing left to remove in $ZEEK_LOGS, $STRELKA_FILES, $SURICATA_LOGS, $PCAPS; $SENSOR_DIR still at ${CUR_USAGE}% - space is consumed outside of NSM cleanup scope"
break
fi
if [ "$(disk_avail)" -le "$BEFORE" ]; then
log "pass $PASS freed no space; $SENSOR_DIR still at ${CUR_USAGE}% - stopping until next run"
break
fi
done
+1
View File
@@ -18,6 +18,7 @@ zeek:
StatsLogEnable: 0
StatsLogExpireInterval: 0
StatusCmdShowAll: 0
StopWait: 1
CrashExpireInterval: 0
SitePolicyScripts: local.zeek
LogDir: /nsm/zeek/logs
+10
View File
@@ -9,9 +9,19 @@
include:
- zeek.sostatus
# Stop first so the entrypoint's SIGTERM trap can archive the final logs; docker_container.absent
# with force is a 'docker rm -f', which never delivers SIGTERM. force stays so the state still
# converges if the stop overruns.
so-zeek_stopped:
docker_container.stopped:
- name: so-zeek
- error_on_absent: False
so-zeek:
docker_container.absent:
- force: True
- require:
- docker_container: so-zeek_stopped
so-zeek_so-status.disabled:
file.comment:
+4
View File
@@ -19,6 +19,10 @@ so-zeek:
- restart_policy: unless-stopped
- start: True
- privileged: True
# Docker's default 10s grace is not enough for the entrypoint's SIGTERM trap to run
# 'zeekctl stop' and let StopWait archive the final logs. Overrunning it means SIGKILL,
# which strands those logs in spool/tmp and marks every node crashed on the next start.
- stop_timeout: 180
{% if DOCKERMERGED.containers['so-zeek'].ulimits %}
- ulimits:
{% for ULIMIT in DOCKERMERGED.containers['so-zeek'].ulimits %}
+12
View File
@@ -99,6 +99,18 @@ zeek:
regexFailureMessage: You must enter a whole number of days, or 0 to keep crash directories forever.
helpLink: zeek
advanced: True
StopWait:
description: >-
Set to 1 to make "zeekctl stop" wait for the final logs to be archived instead of
letting that finish in the background. Security Onion stops Zeek by stopping its
container, so anything still running in the background is killed when the container
exits - without this, the last logs of each run are stranded unarchived in
/nsm/zeek/spool/tmp and never reach Elasticsearch. It is read only for that reason.
regex: ^[01]$
regexFailureMessage: You must enter 0 or 1.
helpLink: zeek
advanced: True
readonly: True
MinDiskSpace:
description: >-
Percentage of free disk space below which ZeekControl reports a warning, or 0 to disable the check