From 72f60fcaa96153ac2add88d41eaed87bb9f5d15f Mon Sep 17 00:00:00 2001 From: Josh Patterson Date: Thu, 24 Sep 2026 14:30:25 -0400 Subject: [PATCH 01/15] FIX: remove the docker socket from so-telegraf so-telegraf mounted /var/run/docker.sock and joined the host docker group on every node type. The :ro flag blocks write() to the inode, not connect() plus HTTP over the socket, so any code execution inside the container could reach POST /containers/create with Privileged:true and become root on the host. No telegraf script used the socket; the only consumer was the native [[inputs.docker]] plugin, and group_add 920 existed solely to feed it. Container metrics now come from so-container-stats, a collector that runs on the host from cron and writes influx line protocol to a file telegraf already had mounted. This is the pattern so-status, so-raid-status and so-elasticagent-status already use, so the privileged docker access stays on the host side where root cron already ran it. The collector runs as somon, a service account in the docker group with no login shell and a locked password, rather than root. Docker group membership is still root-equivalent on the host, so this is defense in depth rather than a privilege boundary. The telegraf scripts were root:939 mode 770, letting socore rewrite them for code execution inside the container; they are now 750, which still allows the read and execute telegraf needs. The container also ran with no group, giving it gid 0, and now runs as 939:939. That alone would have broken lasthighstate.sh, which reached /opt/so/log/salt only via the root group and could not tell an unreadable file from a missing one, so it silently reported a 56 year highstate age. Only the lasthighstate file is bind mounted now, and the script tests readability instead of existence. Everything inputs.docker collected beyond the five fields the shipped dashboards query is available per stat under telegraf:container_stats, annotated for SOC so an operator can enable it without editing files. Defaults reproduce the previous output exactly. With every stat enabled the emitted field set matches what inputs.docker wrote, verified by running the plugin against the live socket and diffing: 53 fields, no type mismatches, no field present on one side only. Two deliberate differences: max_usage carries the real cgroup peak where the daemon reports 0 on cgroup v2, and host-network containers emit no docker_container_net row, matching inputs.docker. Docker label tags are not restored, since nothing queries them and they cost significant cardinality. so-status, the influxdb size cron, so-elasticagent-status, so-raid-status and so-common-status-check truncated their output in place while telegraf read it, so telegraf periodically saw an empty file and logged a parse error or emitted empty values. They now write aside and rename. Measured on a live manager, the old so-status cron left status.log empty for 215 of 10997 reads. Tested on a fresh install, a converted grid and a 3.0 upgrade. --- .github/workflows/pythontest.yml | 25 + salt/common/init.sls | 4 +- salt/common/tools/sbin/so-common-status-check | 25 +- salt/common/tools/sbin_jinja/so-raid-status | 4 +- salt/influxdb/enabled.sls | 4 +- salt/manager/init.sls | 2 +- salt/telegraf/config.sls | 100 ++- salt/telegraf/defaults.yaml | 77 +++ salt/telegraf/disabled.sls | 5 + salt/telegraf/enabled.sls | 11 +- salt/telegraf/etc/telegraf.conf | 18 +- salt/telegraf/scripts/containers.sh | 29 + salt/telegraf/scripts/lasthighstate.sh | 10 +- salt/telegraf/soc_telegraf.yaml | 313 +++++++++ .../tools/sbin_jinja/so-container-stats | 398 +++++++++++ .../sbin_jinja/so-container-stats_test.py | 635 ++++++++++++++++++ setup/so-functions | 1 + 17 files changed, 1626 insertions(+), 35 deletions(-) create mode 100644 salt/telegraf/scripts/containers.sh create mode 100644 salt/telegraf/tools/sbin_jinja/so-container-stats create mode 100644 salt/telegraf/tools/sbin_jinja/so-container-stats_test.py diff --git a/.github/workflows/pythontest.yml b/.github/workflows/pythontest.yml index dc95b01c3..5d474f5df 100644 --- a/.github/workflows/pythontest.yml +++ b/.github/workflows/pythontest.yml @@ -6,6 +6,9 @@ on: - "salt/sensoroni/files/analyzers/**" - "salt/manager/tools/sbin/**" - "salt/_beacons/**" + - "salt/telegraf/tools/sbin_jinja/**" + - "salt/telegraf/defaults.yaml" + - "salt/telegraf/soc_telegraf.yaml" jobs: build: @@ -34,3 +37,25 @@ jobs: - name: Test with pytest run: | PYTHONPATH=${{ matrix.python-code-path }} pytest ${{ matrix.python-code-path }} --cov=${{ matrix.python-code-path }} --doctest-modules --cov-report=term --cov-fail-under=100 --cov-config=pytest.ini + + telegraf-collector: + # so-container-stats is a jinja template rather than an importable module, so it gets its + # own job: the test renders it the way salt does, then drives it with a faked docker engine + # and cgroup tree. No container runtime is needed. + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v3 + - name: Set up Python + uses: actions/setup-python@v3 + with: + python-version: "3.14" + - name: Install dependencies + run: | + python -m pip install --upgrade pip + python -m pip install flake8 pytest jinja2 pyyaml + - name: Lint with flake8 + run: | + flake8 salt/telegraf/tools/sbin_jinja/so-container-stats_test.py --config=pytest.ini + - name: Test with pytest + run: | + pytest salt/telegraf/tools/sbin_jinja/so-container-stats_test.py -v diff --git a/salt/common/init.sls b/salt/common/init.sls index 34b84f09d..ecb5fdccc 100644 --- a/salt/common/init.sls +++ b/salt/common/init.sls @@ -217,9 +217,11 @@ sostatus_log: - replace: False # Install sostatus check cron. This is used to populate Grid. +# telegraf reads status.log on the same minute boundary this runs, so write aside and rename +# rather than truncating the file it is reading so-status_check_cron: cron.present: - - name: '/usr/sbin/so-status -j > /opt/so/log/sostatus/status.log 2>&1' + - name: '/usr/sbin/so-status -j > /opt/so/log/sostatus/status.log.tmp 2>&1; mv -f /opt/so/log/sostatus/status.log.tmp /opt/so/log/sostatus/status.log' - identifier: so-status_check_cron - user: root - minute: '*/1' diff --git a/salt/common/tools/sbin/so-common-status-check b/salt/common/tools/sbin/so-common-status-check index cbef7309e..1f387570d 100644 --- a/salt/common/tools/sbin/so-common-status-check +++ b/salt/common/tools/sbin/so-common-status-check @@ -9,6 +9,7 @@ import sys import subprocess import os import json +import tempfile sys.path.append('/opt/saltstack/salt/lib/python3.10/site-packages/') import salt.config @@ -17,6 +18,21 @@ import salt.loader __opts__ = salt.config.minion_config('/etc/salt/minion') __grains__ = salt.loader.grains(__opts__) +def write_atomic(path, value): + # telegraf reads these files on its own schedule; replacing them by rename means it never + # reads a truncated file and reports an empty value as if it were real + directory = os.path.dirname(path) + handle, temp = tempfile.mkstemp(dir=directory) + try: + with os.fdopen(handle, 'w') as f: + f.write(str(value)) + os.chmod(temp, 0o644) + os.replace(temp, path) + except Exception: + os.path.exists(temp) and os.unlink(temp) + raise + + def check_needs_restarted(): osfam = __grains__['os_family'] val = '0' @@ -34,8 +50,7 @@ def check_needs_restarted(): else: fail("Unsupported OS") - with open(outfile, 'w') as f: - f.write(val) + write_atomic(outfile, val) def check_for_fps(): feat = 'fps' @@ -56,8 +71,7 @@ def check_for_fps(): # Unknown, so assume 0 fps = 0 - with open('/opt/so/log/sostatus/fps_enabled', 'w') as f: - f.write(str(fps)) + write_atomic('/opt/so/log/sostatus/fps_enabled', fps) def check_for_lks(): feat = 'Lks' @@ -80,8 +94,7 @@ def check_for_lks(): lks = 1 if lks: break - with open('/opt/so/log/sostatus/lks_enabled', 'w') as f: - f.write(str(lks)) + write_atomic('/opt/so/log/sostatus/lks_enabled', lks) def fail(msg): print(msg, file=sys.stderr) diff --git a/salt/common/tools/sbin_jinja/so-raid-status b/salt/common/tools/sbin_jinja/so-raid-status index ca3c34608..fefb1806f 100755 --- a/salt/common/tools/sbin_jinja/so-raid-status +++ b/salt/common/tools/sbin_jinja/so-raid-status @@ -125,4 +125,6 @@ else RAIDSTATUS=1 fi -echo "nsmraid=$RAIDSTATUS" > /opt/so/log/raid/status.log +# telegraf reads this file; write aside and rename so it never sees a half-written file +echo "nsmraid=$RAIDSTATUS" > /opt/so/log/raid/status.log.tmp +mv -f /opt/so/log/raid/status.log.tmp /opt/so/log/raid/status.log diff --git a/salt/influxdb/enabled.sls b/salt/influxdb/enabled.sls index ed3ac0de3..617a56e1e 100644 --- a/salt/influxdb/enabled.sls +++ b/salt/influxdb/enabled.sls @@ -94,9 +94,11 @@ metrics_link_file: - docker_container: so-influxdb # Install cron job to determine size of influxdb for telegraf +# telegraf reads this while the cron rewrites it, so write aside and rename rather than +# truncating in place. tgraflogdir recurses ownership, so the temp file is chowned to match get_influxdb_size: cron.present: - - name: 'du -s -k /nsm/influxdb | cut -f1 > /opt/so/log/telegraf/influxdb_size.log 2>&1' + - name: 'du -s -k /nsm/influxdb | cut -f1 > /opt/so/log/telegraf/influxdb_size.log.tmp 2>&1; chown 939:939 /opt/so/log/telegraf/influxdb_size.log.tmp; mv -f /opt/so/log/telegraf/influxdb_size.log.tmp /opt/so/log/telegraf/influxdb_size.log' - identifier: get_influxdb_size - user: root - minute: '*/1' diff --git a/salt/manager/init.sls b/salt/manager/init.sls index 87669fad4..cb0395d3c 100644 --- a/salt/manager/init.sls +++ b/salt/manager/init.sls @@ -166,7 +166,7 @@ so-repo-sync: so_fleetagent_status: cron.present: - - name: /usr/sbin/so-elasticagent-status > /opt/so/log/agents/agentstatus.log 2>&1 + - name: '/usr/sbin/so-elasticagent-status > /opt/so/log/agents/agentstatus.log.tmp 2>&1; mv -f /opt/so/log/agents/agentstatus.log.tmp /opt/so/log/agents/agentstatus.log' - identifier: so_fleetagent_status - user: root - minute: '*/5' diff --git a/salt/telegraf/config.sls b/salt/telegraf/config.sls index 18ac51ddd..64ca7070e 100644 --- a/salt/telegraf/config.sls +++ b/salt/telegraf/config.sls @@ -36,7 +36,7 @@ tgraf_sync_script_{{script}}: - name: /opt/so/conf/telegraf/scripts/{{script}} - user: root - group: 939 - - mode: 770 + - mode: 750 - template: jinja - source: salt://telegraf/scripts/{{script}} - defaults: @@ -49,7 +49,7 @@ tgraf_sync_script_esindexsize.sh: - name: /opt/so/conf/telegraf/scripts/esindexsize.sh - user: root - group: 939 - - mode: 770 + - mode: 750 - source: salt://telegraf/scripts/esindexsize.sh {# Copy conf/elasticsearch/curl.config for telegraf to use with esindexsize.sh #} tgraf_sync_escurl_conf: @@ -61,6 +61,80 @@ tgraf_sync_escurl_conf: - source: salt://elasticsearch/curl.config {% endif %} +# so-container-stats runs on the host as somon, a docker group member, so the container does +# not need the docker socket +somongroup: + group.present: + - name: somon + - gid: 961 + +# cron chdirs to $HOME before running a job, so home must exist +somon: + user.present: + - uid: 961 + - gid: 961 + - home: /opt/so/log/somon + - createhome: False + - shell: /sbin/nologin + - groups: + - docker + # renumbering an existing somon is a no-op on a fresh host and lets a host created + # before the id changed converge instead of failing the whole telegraf state + - allow_uid_change: True + - allow_gid_change: True + - require: + - group: somongroup + +somonlogdir: + file.directory: + - name: /opt/so/log/somon + - user: 961 + - group: 961 + - mode: 755 + # the lock file is not otherwise managed; recurse so a renumber rechowns it too + - recurse: + - user + - group + - require: + - user: somon + +containers_log: + file.managed: + - name: /opt/so/log/somon/containers.log + - user: 961 + - group: 961 + - mode: 644 + - replace: False + - require: + - file: somonlogdir + +# telegraf reads on the same minute boundary the collector runs, and docker stats takes +# seconds, so write aside and rename rather than truncating the file telegraf is reading. +# ; not && so a failed run replaces the file instead of leaving stale metrics behind. +# flock -n keeps a run that outlives its minute from racing the next one over the same tmp +# file; the skipped run leaves a stale containers.log, which containers.sh discards by age +so-container-stats_cron: + cron.present: + - name: "flock -n /opt/so/log/somon/containers.lock -c '/usr/sbin/so-container-stats > /opt/so/log/somon/containers.log.tmp 2>&1; mv -f /opt/so/log/somon/containers.log.tmp /opt/so/log/somon/containers.log'" + - identifier: so-container-stats_cron + - user: somon + - minute: '*/1' + - hour: '*' + - daymonth: '*' + - month: '*' + - dayweek: '*' + - require: + - user: somon + +# salt.lasthighstate touches this at order 9001, after the container starts; pre-create it so +# docker does not create a directory at the bind mount source +lasthighstate_placeholder: + file.managed: + - name: /opt/so/log/salt/lasthighstate + - mode: 644 + - replace: False + - makedirs: True + telegraf_sbin: file.recurse: - name: /usr/sbin @@ -69,14 +143,20 @@ telegraf_sbin: - group: root - file_mode: 755 -#telegraf_sbin_jinja: -# file.recurse: -# - name: /usr/sbin -# - source: salt://telegraf/tools/sbin_jinja -# - user: 939 -# - group: 939 -# - file_mode: 755 -# - template: jinja +# so-container-stats needs the per-stat toggles, so it renders instead of copying +tgraf_sbin_jinja: + file.recurse: + - name: /usr/sbin + - source: salt://telegraf/tools/sbin_jinja + - user: root + - group: root + - file_mode: 755 + # the unit test lives beside the script; it must not ship or be rendered as jinja + - exclude_pat: + - "*_test.py" + - template: jinja + - defaults: + CONTAINER_STATS: {{ TELEGRAFMERGED.container_stats }} tgrafconf: file.managed: diff --git a/salt/telegraf/defaults.yaml b/salt/telegraf/defaults.yaml index 24f58b157..0ff3539fd 100644 --- a/salt/telegraf/defaults.yaml +++ b/salt/telegraf/defaults.yaml @@ -10,6 +10,69 @@ telegraf: flush_jitter: '0s' debug: false quiet: false + container_stats: + tags: + identity: False + engine: + n_containers: False + n_containers_running: False + n_containers_stopped: False + n_containers_paused: False + n_images: False + n_cpus: False + n_goroutines: False + n_used_file_descriptors: False + n_listener_events: False + memory_total: False + cpu: + usage_percent: True + usage_total: False + usage_in_usermode: False + usage_in_kernelmode: False + usage_system: False + throttling_periods: False + throttling_throttled_periods: False + throttling_throttled_time: False + container_id: False + mem: + usage_percent: True + usage: False + limit: False + max_usage: False + active_anon: False + active_file: False + inactive_anon: False + inactive_file: False + unevictable: False + pgfault: False + pgmajfault: False + container_id: False + net: + rx_bytes: True + rx_packets: False + rx_errors: False + rx_dropped: False + tx_bytes: False + tx_packets: False + tx_errors: False + tx_dropped: False + container_id: False + blkio: + io_service_bytes_recursive_read: False + io_service_bytes_recursive_write: False + container_id: False + status: + uptime_ns: True + oomkilled: True + pid: False + exitcode: False + restart_count: False + started_at: False + finished_at: False + container_id: False + health: + health_status: False + failing_streak: False scripts: eval: - agentstatus.sh @@ -19,6 +82,7 @@ telegraf: - oldpcap.sh - os.sh - raid.sh + - containers.sh - sostatus.sh - suriloss.sh - surirules.sh @@ -34,6 +98,7 @@ telegraf: - os.sh - raid.sh - redis.sh + - containers.sh - sostatus.sh - suriloss.sh - surirules.sh @@ -47,6 +112,7 @@ telegraf: - os.sh - raid.sh - redis.sh + - containers.sh - sostatus.sh - features.sh managerhype: @@ -56,6 +122,7 @@ telegraf: - os.sh - raid.sh - redis.sh + - containers.sh - sostatus.sh - features.sh managersearch: @@ -66,12 +133,14 @@ telegraf: - os.sh - raid.sh - redis.sh + - containers.sh - sostatus.sh - features.sh import: - influxdbsize.sh - lasthighstate.sh - os.sh + - containers.sh - sostatus.sh sensor: - checkfiles.sh @@ -79,6 +148,7 @@ telegraf: - oldpcap.sh - os.sh - raid.sh + - containers.sh - sostatus.sh - suriloss.sh - surirules.sh @@ -93,6 +163,7 @@ telegraf: - os.sh - raid.sh - redis.sh + - containers.sh - sostatus.sh - suriloss.sh - surirules.sh @@ -101,12 +172,14 @@ telegraf: idh: - lasthighstate.sh - os.sh + - containers.sh - sostatus.sh searchnode: - eps.sh - lasthighstate.sh - os.sh - raid.sh + - containers.sh - sostatus.sh - features.sh receiver: @@ -115,16 +188,20 @@ telegraf: - os.sh - raid.sh - redis.sh + - containers.sh - sostatus.sh fleet: - lasthighstate.sh - os.sh + - containers.sh - sostatus.sh hypervisor: - lasthighstate.sh - os.sh + - containers.sh - sostatus.sh desktop: - lasthighstate.sh - os.sh + - containers.sh - sostatus.sh diff --git a/salt/telegraf/disabled.sls b/salt/telegraf/disabled.sls index 004d3d928..371708e04 100644 --- a/salt/telegraf/disabled.sls +++ b/salt/telegraf/disabled.sls @@ -13,6 +13,11 @@ so-telegraf: docker_container.absent: - force: True +so-container-stats_cron: + cron.absent: + - identifier: so-container-stats_cron + - user: somon + so-telegraf_so-status.disabled: file.comment: - name: /opt/so/conf/so-status/so-status.conf diff --git a/salt/telegraf/enabled.sls b/salt/telegraf/enabled.sls index 6a063e08b..addc55128 100644 --- a/salt/telegraf/enabled.sls +++ b/salt/telegraf/enabled.sls @@ -19,8 +19,7 @@ so-telegraf: docker_container.running: - image: {{ GLOBALS.registry_host }}:5000/{{ GLOBALS.image_repo }}/so-telegraf:{{ GLOBALS.so_version }} - restart_policy: unless-stopped - - user: 939 - - group_add: 939,920 + - user: 939:939 - environment: - HOST_ETC=/host/etc - HOST_SYS=/host/sys @@ -38,7 +37,6 @@ so-telegraf: - /opt/so/conf/telegraf/etc/telegraf.conf:/etc/telegraf/telegraf.conf:ro - /opt/so/conf/telegraf/node_config.json:/etc/telegraf/node_config.json:ro - /var/run/utmp:/var/run/utmp:ro - - /var/run/docker.sock:/var/run/docker.sock:ro - /:/host:ro - /sys:/host/sys:ro - /proc:/host/proc:ro @@ -51,7 +49,8 @@ so-telegraf: - /opt/so/log/suricata:/var/log/suricata:ro - /opt/so/log/raid:/var/log/raid:ro - /opt/so/log/sostatus:/var/log/sostatus:ro - - /opt/so/log/salt:/var/log/salt:ro + - /opt/so/log/somon:/var/log/somon:ro + - /opt/so/log/salt/lasthighstate:/var/log/salt/lasthighstate:ro - /opt/so/log/agents:/var/log/agents:ro {% if GLOBALS.is_manager or GLOBALS.role == 'so-heavynode' %} - /opt/so/conf/telegraf/etc/escurl.config:/etc/telegraf/elasticsearch.config:ro @@ -74,6 +73,8 @@ so-telegraf: {% endfor %} {% endif %} - watch: + - file: tgraf_sbin_jinja + - file: lasthighstate_placeholder - file: trusttheca - x509: telegraf_crt - x509: telegraf_key @@ -83,6 +84,8 @@ so-telegraf: - file: tgraf_sync_script_{{script}} {% endfor %} - require: + - file: lasthighstate_placeholder + - file: somonlogdir - file: trusttheca - x509: telegraf_crt - x509: telegraf_key diff --git a/salt/telegraf/etc/telegraf.conf b/salt/telegraf/etc/telegraf.conf index ae1b83084..ab51ac2a5 100644 --- a/salt/telegraf/etc/telegraf.conf +++ b/salt/telegraf/etc/telegraf.conf @@ -228,13 +228,6 @@ # ## bond interfaces. # # bond_interfaces = ["bond0"] -# # Read metrics about docker containers - [[inputs.docker]] -# ## Docker Endpoint -# ## To use TCP, set endpoint = "tcp://[ip]:[port]" -# ## To use environment variables (ie, docker-machine), set endpoint = "ENV" - endpoint = "unix:///var/run/docker.sock" -# # # Read stats from one or more Elasticsearch servers or clusters {%- if GLOBALS.is_manager or GLOBALS.role == 'so-heavynode' %} @@ -342,6 +335,17 @@ interval = "60s" {%- endif %} +{%- if 'containers.sh' in TELEGRAFMERGED.scripts[GLOBALS.role.split('-')[1]] %} +{%- do TELEGRAFMERGED.scripts[GLOBALS.role.split('-')[1]].remove('containers.sh') %} +[[inputs.exec]] + commands = [ + ["/scripts/containers.sh"] + ] + data_format = "influx" + timeout = "15s" + interval = "60s" +{%- endif %} + {%- if TELEGRAFMERGED.scripts[GLOBALS.role.split('-')[1]] | length > 0 %} [[inputs.exec]] commands = [ diff --git a/salt/telegraf/scripts/containers.sh b/salt/telegraf/scripts/containers.sh new file mode 100644 index 000000000..836d4a999 --- /dev/null +++ b/salt/telegraf/scripts/containers.sh @@ -0,0 +1,29 @@ +#!/bin/bash +# +# Copyright Security Onion Solutions LLC and/or licensed to Security Onion Solutions LLC under one +# or more contributor license agreements. Licensed under the Elastic License 2.0 as shown at +# https://securityonion.net/license; you may not use this file except in compliance with the +# Elastic License 2.0. + + + +# if this script isn't already running +if [[ ! "`pidof -x $(basename $0) -o %PPID`" ]]; then + + CONTAINERSLOG=/var/log/somon/containers.log + # the collector rewrites this every minute; report nothing rather than repeating a stale + # file as if it were current, in case a run was skipped or the collector is wedged + MAXAGE=150 + + if [ -r "$CONTAINERSLOG" ]; then + AGE=$(( $(date +%s) - $(stat -c %Y "$CONTAINERSLOG") )) + if [ "$AGE" -le "$MAXAGE" ]; then + cat $CONTAINERSLOG + fi + fi + + exit 0 + +fi + +exit 0 diff --git a/salt/telegraf/scripts/lasthighstate.sh b/salt/telegraf/scripts/lasthighstate.sh index 85f259bb8..a72c00ca9 100644 --- a/salt/telegraf/scripts/lasthighstate.sh +++ b/salt/telegraf/scripts/lasthighstate.sh @@ -8,10 +8,12 @@ # if this script isn't already running if [[ ! "`pidof -x $(basename $0) -o %PPID`" ]]; then - LAST_HIGHSTATE_END=$([ -e "/var/log/salt/lasthighstate" ] && date -r /var/log/salt/lasthighstate +%s || echo 0) - NOW=$(date +%s) - HIGHSTATE_AGE_SECONDS=$((NOW-LAST_HIGHSTATE_END)) - echo "salt highstate_age_seconds=$HIGHSTATE_AGE_SECONDS" + if [ -r "/var/log/salt/lasthighstate" ]; then + LAST_HIGHSTATE_END=$(date -r /var/log/salt/lasthighstate +%s) + NOW=$(date +%s) + HIGHSTATE_AGE_SECONDS=$((NOW-LAST_HIGHSTATE_END)) + echo "salt highstate_age_seconds=$HIGHSTATE_AGE_SECONDS" + fi fi diff --git a/salt/telegraf/soc_telegraf.yaml b/salt/telegraf/soc_telegraf.yaml index 0decdd06a..d92a89f2f 100644 --- a/salt/telegraf/soc_telegraf.yaml +++ b/salt/telegraf/soc_telegraf.yaml @@ -53,6 +53,319 @@ telegraf: forcedType: bool advanced: True helpLink: influxdb + container_stats: + tags: + identity: + description: Adds the container_image, container_version, engine_host and server_version tags to every container measurement, as the retired inputs.docker plugin did. Increases series cardinality. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + engine: + n_containers: + description: Total number of containers known to the Docker engine, running or not. Part of the engine-level docker measurement. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + n_containers_running: + description: Number of containers currently running. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + n_containers_stopped: + description: Number of containers currently stopped. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + n_containers_paused: + description: Number of containers currently paused. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + n_images: + description: Number of container images held by the Docker engine. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + n_cpus: + description: Number of CPUs the Docker engine reports for this host. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + n_goroutines: + description: Number of goroutines inside the Docker daemon. Diagnostic detail for the daemon itself. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + n_used_file_descriptors: + description: Number of file descriptors held open by the Docker daemon. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + n_listener_events: + description: Number of event listeners subscribed to the Docker daemon. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + memory_total: + description: Total physical memory the Docker engine reports for this host, in bytes. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + cpu: + usage_percent: + description: Percentage of host CPU consumed by the container. Required by the Container CPU Usage chart on the Security Onion Performance dashboard. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + usage_total: + description: Cumulative CPU time consumed by the container, in nanoseconds. Read from the cgroup. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + usage_in_usermode: + description: Cumulative CPU time consumed in user mode, in nanoseconds. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + usage_in_kernelmode: + description: Cumulative CPU time consumed in kernel mode, in nanoseconds. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + usage_system: + description: Cumulative host-wide CPU time, in nanoseconds. Used as the denominator when calculating container CPU percentage. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + throttling_periods: + description: Number of CPU enforcement periods the container has seen. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + throttling_throttled_periods: + description: Number of periods in which the container was throttled against its CPU limit. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + throttling_throttled_time: + description: Total time the container spent throttled, in nanoseconds. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + container_id: &containerid + description: Full 64 character container ID, emitted as a field on this measurement. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + mem: + usage_percent: + description: Percentage of its memory limit the container is using. Required by the Container Memory Usage chart on the Security Onion Performance dashboard. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + usage: + description: Container memory usage in bytes, excluding reclaimable page cache. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + limit: + description: Memory limit for the container in bytes. Reports total host memory when the container is unlimited. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + max_usage: + description: Peak memory usage for the container in bytes, read from the cgroup peak counter. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + active_anon: + description: Anonymous memory on the active LRU list, in bytes. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + active_file: + description: Page cache on the active LRU list, in bytes. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + inactive_anon: + description: Anonymous memory on the inactive LRU list, in bytes. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + inactive_file: + description: Page cache on the inactive LRU list, in bytes. This is the reclaimable cache subtracted from usage. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + unevictable: + description: Memory that cannot be reclaimed, in bytes. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + pgfault: + description: Cumulative number of page faults taken by the container. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + pgmajfault: + description: Cumulative number of major page faults, those requiring disk access. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + container_id: *containerid + net: + rx_bytes: + description: Bytes received by the container across all interfaces except loopback. Required by the Container Traffic - Inbound chart on the Security Onion Performance dashboard. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + rx_packets: + description: Packets received by the container. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + rx_errors: + description: Receive errors counted on the container interfaces. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + rx_dropped: + description: Received packets dropped by the container interfaces. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + tx_bytes: + description: Bytes transmitted by the container. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + tx_packets: + description: Packets transmitted by the container. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + tx_errors: + description: Transmit errors counted on the container interfaces. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + tx_dropped: + description: Transmitted packets dropped by the container interfaces. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + container_id: *containerid + blkio: + io_service_bytes_recursive_read: + description: Cumulative bytes read from block devices by the container. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + io_service_bytes_recursive_write: + description: Cumulative bytes written to block devices by the container. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + container_id: *containerid + status: + uptime_ns: + description: How long the container has been running, in nanoseconds. Required by the Container Uptime chart on the Security Onion Performance dashboard. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + oomkilled: + description: Whether the container was killed by the kernel out-of-memory handler. Required by the Most Recent Container Events table on the Security Onion Performance dashboard. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + pid: + description: Host process ID of the container main process. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + exitcode: + description: Exit code of the container main process. Meaningful once the container has stopped. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + restart_count: + description: Number of times the Docker engine has restarted this container. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + started_at: + description: Time the container last started, as a Unix timestamp in nanoseconds. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + finished_at: + description: Time the container last exited, as a Unix timestamp in nanoseconds. Absent for a container that has never exited. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + container_id: *containerid + health: + health_status: + description: Result of the container healthcheck, such as healthy, unhealthy or starting. Only emitted for containers that define a healthcheck. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + failing_streak: + description: Number of consecutive failed healthchecks. Only emitted for containers that define a healthcheck. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb scripts: eval: &telegrafscripts description: List of input.exec scripts to run for this node type. The script must be present in salt/telegraf/scripts. diff --git a/salt/telegraf/tools/sbin_jinja/so-container-stats b/salt/telegraf/tools/sbin_jinja/so-container-stats new file mode 100644 index 000000000..b11b0bcb4 --- /dev/null +++ b/salt/telegraf/tools/sbin_jinja/so-container-stats @@ -0,0 +1,398 @@ +#!/usr/bin/env python3 + +# Copyright Security Onion Solutions LLC and/or licensed to Security Onion Solutions LLC under one +# or more contributor license agreements. Licensed under the Elastic License 2.0 as shown at +# https://securityonion.net/license; you may not use this file except in compliance with the +# Elastic License 2.0. + +# Runs from cron as somon, a docker group member, and emits influx line protocol for telegraf to +# read. This exists so so-telegraf does not need the docker socket; the container reads the output +# file instead. +# Measurement/tag/field names match telegraf's inputs.docker plugin because the InfluxDB +# dashboards query them directly. Which fields are emitted is set per stat in SOC; see +# telegraf.container_stats in defaults.yaml. + +import json + +SETTINGS = json.loads('''{{ CONTAINER_STATS | tojson }}''') +{% raw %} +import os +import re +import subprocess +import sys +from datetime import datetime, timezone + +CGROUP_ROOT = '/sys/fs/cgroup' +NANOSEC = 10**9 +# cgroup and docker report cpu time in microseconds; inputs.docker published nanoseconds +USEC_TO_NSEC = 1000 + + +def want(group, field): + return bool(SETTINGS.get(group, {}).get(field)) + + +def wants_any(group, fields): + return any(want(group, field) for field in fields) + + +def docker(args): + proc = subprocess.run(['docker'] + args, stdout=subprocess.PIPE, stderr=subprocess.DEVNULL, encoding='utf-8') + if proc.returncode != 0: + print('Container system error; unable to query docker', file=sys.stderr) + sys.exit(1) + return proc.stdout + + +def escape_tag(value): + return value.replace(',', '\\,').replace(' ', '\\ ').replace('=', '\\=') + + +def quote(value): + return '"{}"'.format(str(value).replace('\\', '\\\\').replace('"', '\\"')) + + +def integer(value): + return '{}i'.format(int(value)) + + +def unsigned(value): + # inputs.docker published the cgroup and network counters as uint64; influx stores u and i + # as different field types, so matching it keeps historical series readable + return '{}u'.format(max(int(value), 0)) + + +def read_text(path): + try: + with open(path) as handle: + return handle.read() + except OSError: + return '' + + +def read_pairs(path): + # cgroup files such as memory.stat and cpu.stat are "key value" per line + values = {} + for line in read_text(path).splitlines(): + parts = line.split() + if len(parts) == 2: + try: + values[parts[0]] = int(parts[1]) + except ValueError: + pass + return values + + +def read_value(path): + raw = read_text(path).strip() + try: + return int(raw) + except ValueError: + return None + + +def cgroup_path(pid): + # "0::/system.slice/docker-.scope" on cgroup v2 + for line in read_text('/proc/{}/cgroup'.format(pid)).splitlines(): + parts = line.split(':', 2) + if len(parts) == 3 and parts[1] == '': + return CGROUP_ROOT + parts[2] + return None + + +def host_memory_total(): + match = re.search(r'MemTotal:\s+(\d+) kB', read_text('/proc/meminfo')) + return int(match.group(1)) * 1024 if match else 0 + + +def host_cpu_nanoseconds(): + # inputs.docker's usage_system is the host-wide cpu time the daemon reads from /proc/stat + for line in read_text('/proc/stat').splitlines(): + if line.startswith('cpu '): + ticks = sum(int(value) for value in line.split()[1:]) + return int(ticks * NANOSEC / os.sysconf('SC_CLK_TCK')) + return 0 + + +def parse_image(image): + # mirrors telegraf's internal/docker ParseImage so the tags match what inputs.docker emitted + domain = '' + remainder = image + if '/' in image: + head, _, tail = image.partition('/') + if '.' in head or ':' in head or head == 'localhost': + domain, remainder = head + '/', tail + if ':' in remainder: + name, _, version = remainder.rpartition(':') + return domain + name, version + return domain + remainder, 'unknown' + + +def to_nanoseconds(stamp): + # docker emits 9 fractional digits; fromisoformat takes at most 6 before python 3.11 + stamp = stamp.rstrip('Z') + if '.' in stamp: + whole, _, frac = stamp.partition('.') + stamp = whole + '.' + frac[:6] + try: + parsed = datetime.fromisoformat(stamp).replace(tzinfo=timezone.utc) + except ValueError: + return None + if parsed.year <= 1: + return None + return int(parsed.timestamp() * NANOSEC) + + +def percent(value): + try: + return float(value.strip().rstrip('%')) + except ValueError: + return 0.0 + + +def net_counters(pid): + # summed across interfaces except loopback, matching the daemon's per-container totals + rows = read_text('/proc/{}/net/dev'.format(pid)).splitlines()[2:] + if not rows: + return None + names = ['rx_bytes', 'rx_packets', 'rx_errors', 'rx_dropped'] + totals = dict.fromkeys(names + ['tx_bytes', 'tx_packets', 'tx_errors', 'tx_dropped'], 0) + found = False + for row in rows: + iface, _, rest = row.partition(':') + if iface.strip() == 'lo': + continue + columns = rest.split() + if len(columns) < 12: + continue + found = True + for index, name in enumerate(names): + totals[name] += int(columns[index]) + for index, name in enumerate(['tx_bytes', 'tx_packets', 'tx_errors', 'tx_dropped']): + totals[name] += int(columns[8 + index]) + return totals if found else None + + +def blkio_counters(path): + # inputs.docker emitted the device=total row unconditionally, so a container that has done + # no block io reports zeros rather than dropping out of the measurement entirely + totals = {'io_service_bytes_recursive_read': 0, 'io_service_bytes_recursive_write': 0} + rows = read_text(path + '/io.stat').splitlines() + for row in rows: + for token in row.split()[1:]: + key, _, value = token.partition('=') + try: + if key == 'rbytes': + totals['io_service_bytes_recursive_read'] += int(value) + elif key == 'wbytes': + totals['io_service_bytes_recursive_write'] += int(value) + except ValueError: + pass + return totals + + +def collect_stats(): + stats = {} + for line in docker(['stats', '--no-stream', '--format', '{{json .}}']).splitlines(): + line = line.strip() + if not line: + continue + try: + entry = json.loads(line) + except json.JSONDecodeError: + continue + stats[entry.get('Name', '')] = entry + return stats + + +def collect_inspect(): + ids = docker(['ps', '-aq']).split() + if not ids: + return [] + try: + return json.loads(docker(['inspect'] + ids)) + except json.JSONDecodeError: + return [] + + +def collect_info(): + try: + return json.loads(docker(['info', '--format', '{{json .}}'])) + except json.JSONDecodeError: + return {} + + +class Emitter: + def __init__(self): + self.lines = [] + + def add(self, measurement, tags, fields): + if not fields: + return + tagset = ','.join('{}={}'.format(key, escape_tag(value)) for key, value in tags) + body = ','.join('{}={}'.format(key, fields[key]) for key in sorted(fields)) + self.lines.append('{},{} {}'.format(measurement, tagset, body)) + + +def main(): + cpu_extra = ['usage_total', 'usage_in_usermode', 'usage_in_kernelmode', + 'throttling_periods', 'throttling_throttled_periods', 'throttling_throttled_time'] + mem_extra = ['usage', 'limit', 'max_usage', 'active_anon', 'active_file', 'inactive_anon', + 'inactive_file', 'unevictable', 'pgfault', 'pgmajfault'] + net_fields = ['rx_bytes', 'rx_packets', 'rx_errors', 'rx_dropped', + 'tx_bytes', 'tx_packets', 'tx_errors', 'tx_dropped'] + engine_fields = ['n_containers', 'n_containers_running', 'n_containers_stopped', + 'n_containers_paused', 'n_images', 'n_cpus', 'n_goroutines', + 'n_used_file_descriptors', 'n_listener_events'] + + identity = want('tags', 'identity') + need_stats = want('cpu', 'usage_percent') + need_cgroup_cpu = wants_any('cpu', cpu_extra) + need_cgroup_mem = wants_any('mem', mem_extra + ['usage_percent']) + need_blkio = wants_any('blkio', ['io_service_bytes_recursive_read', + 'io_service_bytes_recursive_write', 'container_id']) + need_net = wants_any('net', net_fields + ['container_id']) + need_info = identity or wants_any('engine', engine_fields + ['memory_total']) + + emitter = Emitter() + stats = collect_stats() if need_stats else {} + info = collect_info() if need_info else {} + engine_tags = [('engine_host', info.get('Name', '')), + ('server_version', info.get('ServerVersion', ''))] + system_ns = host_cpu_nanoseconds() if want('cpu', 'usage_system') else 0 + mem_total = host_memory_total() if wants_any('mem', ['limit', 'usage_percent']) else 0 + + if info: + counts = {'n_containers': 'Containers', 'n_containers_running': 'ContainersRunning', + 'n_containers_stopped': 'ContainersStopped', 'n_containers_paused': 'ContainersPaused', + 'n_images': 'Images', 'n_cpus': 'NCPU', 'n_goroutines': 'NGoroutines', + 'n_used_file_descriptors': 'NFd', 'n_listener_events': 'NEventsListener'} + fields = {name: integer(info[key]) for name, key in counts.items() + if want('engine', name) and key in info} + emitter.add('docker', engine_tags, fields) + # inputs.docker published memory_total as its own point + if want('engine', 'memory_total') and 'MemTotal' in info: + emitter.add('docker', engine_tags, {'memory_total': integer(info['MemTotal'])}) + + for container in collect_inspect(): + name = container.get('Name', '').lstrip('/') + if not name: + continue + state = container.get('State', {}) + status = state.get('Status', 'unknown') + pid = state.get('Pid') + container_id = container.get('Id', '') + tags = [('container_name', name), ('container_status', status)] + if identity: + image, version = parse_image(container.get('Config', {}).get('Image', '')) + tags += [('container_image', image), ('container_version', version)] + engine_tags + + status_fields = {} + if want('status', 'uptime_ns') or want('status', 'started_at') or want('status', 'finished_at'): + started = to_nanoseconds(state.get('StartedAt', '')) + finished = to_nanoseconds(state.get('FinishedAt', '')) + if started is not None: + if want('status', 'started_at'): + status_fields['started_at'] = integer(started) + if want('status', 'uptime_ns'): + end = finished if finished is not None and finished >= started else int( + datetime.now(timezone.utc).timestamp() * NANOSEC) + status_fields['uptime_ns'] = integer(end - started) + if finished is not None and want('status', 'finished_at'): + status_fields['finished_at'] = integer(finished) + if want('status', 'oomkilled'): + status_fields['oomkilled'] = 'true' if state.get('OOMKilled') else 'false' + if want('status', 'pid'): + status_fields['pid'] = integer(pid or 0) + if want('status', 'exitcode'): + status_fields['exitcode'] = integer(state.get('ExitCode', 0)) + if want('status', 'restart_count'): + status_fields['restart_count'] = integer(container.get('RestartCount', 0)) + if want('status', 'container_id'): + status_fields['container_id'] = quote(container_id) + emitter.add('docker_container_status', tags, status_fields) + + health = state.get('Health') + if health: + health_fields = {} + if want('health', 'health_status'): + health_fields['health_status'] = quote(health.get('Status', '')) + if want('health', 'failing_streak'): + health_fields['failing_streak'] = integer(health.get('FailingStreak', 0)) + emitter.add('docker_container_health', tags, health_fields) + + if status != 'running' or not pid: + continue + path = cgroup_path(pid) + entry = stats.get(name, {}) + + cpu_fields = {} + if want('cpu', 'usage_percent'): + cpu_fields['usage_percent'] = percent(entry.get('CPUPerc', '0%')) + if want('cpu', 'usage_system'): + cpu_fields['usage_system'] = unsigned(system_ns) + if want('cpu', 'container_id'): + cpu_fields['container_id'] = quote(container_id) + if need_cgroup_cpu and path: + cpu = read_pairs(path + '/cpu.stat') + mapping = {'usage_total': 'usage_usec', 'usage_in_usermode': 'user_usec', + 'usage_in_kernelmode': 'system_usec', + 'throttling_periods': 'nr_periods', + 'throttling_throttled_periods': 'nr_throttled', + 'throttling_throttled_time': 'throttled_usec'} + for field, key in mapping.items(): + if want('cpu', field) and key in cpu: + scale = 1 if field.startswith('throttling_') and field != 'throttling_throttled_time' else USEC_TO_NSEC + cpu_fields[field] = unsigned(cpu[key] * scale) + emitter.add('docker_container_cpu', tags + [('cpu', 'cpu-total')], cpu_fields) + + mem_fields = {} + if want('mem', 'container_id'): + mem_fields['container_id'] = quote(container_id) + if need_cgroup_mem and path: + memory = read_pairs(path + '/memory.stat') + for field in ['active_anon', 'active_file', 'inactive_anon', 'inactive_file', + 'unevictable', 'pgfault', 'pgmajfault']: + if want('mem', field) and field in memory: + mem_fields[field] = unsigned(memory[field]) + current = read_value(path + '/memory.current') + # inputs.docker reports usage net of reclaimable page cache + usage = max(current - memory.get('inactive_file', 0), 0) if current is not None else None + raw_limit = read_value(path + '/memory.max') + limit = raw_limit if raw_limit is not None else mem_total + if want('mem', 'usage') and usage is not None: + mem_fields['usage'] = unsigned(usage) + if want('mem', 'limit'): + mem_fields['limit'] = unsigned(limit) + if want('mem', 'usage_percent') and usage is not None: + # same ratio inputs.docker computes, from the same two values + mem_fields['usage_percent'] = usage / limit * 100.0 if limit else 0.0 + if want('mem', 'max_usage'): + peak = read_value(path + '/memory.peak') + if peak is not None: + mem_fields['max_usage'] = unsigned(peak) + emitter.add('docker_container_mem', tags, mem_fields) + + # inputs.docker emitted nothing for host-network containers; its Networks map was empty + if need_net and container.get('HostConfig', {}).get('NetworkMode', '') != 'host': + counters = net_counters(pid) + if counters is not None: + net_out = {field: unsigned(counters[field]) for field in net_fields if want('net', field)} + if want('net', 'container_id'): + net_out['container_id'] = quote(container_id) + emitter.add('docker_container_net', tags + [('network', 'total')], net_out) + + if need_blkio and path: + counters = blkio_counters(path) + if counters is not None: + blkio_out = {field: unsigned(value) for field, value in counters.items() if want('blkio', field)} + if want('blkio', 'container_id'): + blkio_out['container_id'] = quote(container_id) + emitter.add('docker_container_blkio', tags + [('device', 'total')], blkio_out) + + print('\n'.join(emitter.lines)) + + +if __name__ == '__main__': + main() +{% endraw %} diff --git a/salt/telegraf/tools/sbin_jinja/so-container-stats_test.py b/salt/telegraf/tools/sbin_jinja/so-container-stats_test.py new file mode 100644 index 000000000..fa6792860 --- /dev/null +++ b/salt/telegraf/tools/sbin_jinja/so-container-stats_test.py @@ -0,0 +1,635 @@ +# Copyright Security Onion Solutions LLC and/or licensed to Security Onion Solutions LLC under one +# or more contributor license agreements. Licensed under the Elastic License 2.0 as shown at +# https://securityonion.net/license; you may not use this file except in compliance with the +# Elastic License 2.0. + +# so-container-stats replaces telegraf's inputs.docker plugin, which was dropped along with the +# docker socket. The InfluxDB dashboards query its output by measurement, tag and field name, so +# these tests pin that contract: the shipped defaults, the per-stat toggles, the field types, and +# the value semantics copied from the plugin. +# +# The collector is a jinja template, so every test renders it the way salt does and imports the +# result. Docker and the cgroup filesystem are faked, so nothing here needs a container runtime. + +import contextlib +import importlib.util +import io +import json +import os +import tempfile +import unittest + +import jinja2 +import yaml + +HERE = os.path.dirname(os.path.abspath(__file__)) +TEMPLATE = os.path.join(HERE, 'so-container-stats') +DEFAULTS = os.path.join(HERE, '..', '..', 'defaults.yaml') + +# a container id is a 64 character hex string; the plugin published it in full +SOC_ID = 'a' * 64 +TELEGRAF_ID = 'b' * 64 +NGINX_ID = 'c' * 64 +IDSTOOLS_ID = 'd' * 64 + + +def shipped_defaults(): + with open(DEFAULTS) as handle: + return yaml.safe_load(handle)['telegraf']['container_stats'] + + +def all_enabled(): + return {group: {field: True for field in fields} for group, fields in shipped_defaults().items()} + + +def render(settings): + """Render the template as salt does, import it, and hand back the module.""" + with open(TEMPLATE) as handle: + source = handle.read() + rendered = jinja2.Template(source, keep_trailing_newline=True).render(CONTAINER_STATS=settings) + path = os.path.join(tempfile.mkdtemp(), 'so_container_stats.py') + with open(path, 'w') as handle: + handle.write(rendered) + spec = importlib.util.spec_from_file_location('so_container_stats', path) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def split_escaped(text, sep): + """Split on an unescaped, unquoted separator, the way influx line protocol is written.""" + parts, current, escaped, quoted = [], '', False, False + for char in text: + if escaped: + current += char + escaped = False + elif char == '\\': + current += char + escaped = True + elif char == '"': + quoted = not quoted + current += char + elif char == sep and not quoted: + parts.append(current) + current = '' + else: + current += char + parts.append(current) + return parts + + +def split_on_space(line): + escaped = quoted = False + for index, char in enumerate(line): + if escaped: + escaped = False + elif char == '\\': + escaped = True + elif char == '"': + quoted = not quoted + elif char == ' ' and not quoted: + return line[:index], line[index + 1:] + return line, '' + + +def parse(output): + """Parse line protocol into {(measurement, container): (tags, fields)}.""" + points = {} + for line in output.splitlines(): + line = line.strip() + if not line: + continue + head, fieldpart = split_on_space(line) + pieces = split_escaped(head, ',') + measurement, tags = pieces[0], {} + for piece in pieces[1:]: + key, _, value = piece.partition('=') + tags[key] = value + fields = {} + for piece in split_escaped(fieldpart, ','): + key, _, value = piece.partition('=') + fields[key] = value + key = (measurement, tags.get('container_name', '')) + if key in points: + # the engine measurement is published as two points, the way inputs.docker did + points[key][1].update(fields) + else: + points[key] = (tags, fields) + return points + + +def field_type(value): + if value.endswith('u'): + return 'unsigned' + if value.endswith('i'): + return 'integer' + if value.startswith('"'): + return 'string' + if value in ('true', 'false'): + return 'boolean' + return 'float' + + +def container(name, cid, pid, status='running', network='bridge', image='registry:5000/repo/img:3.4.0', + started='2026-09-24T10:00:00.123456789Z', finished='0001-01-01T00:00:00Z', health=None, + oomkilled=False, exitcode=0, restarts=0): + state = {'Status': status, 'Pid': pid, 'StartedAt': started, 'FinishedAt': finished, + 'OOMKilled': oomkilled, 'ExitCode': exitcode} + if health is not None: + state['Health'] = health + return {'Id': cid, 'Name': '/' + name, 'State': state, 'RestartCount': restarts, + 'Config': {'Image': image}, 'HostConfig': {'NetworkMode': network}} + + +class CollectorTestCase(unittest.TestCase): + """Builds a fake docker engine and cgroup tree, then runs the collector against it.""" + + def setUp(self): + self.inspect = [ + container('so-soc', SOC_ID, 1001), + container('so-telegraf', TELEGRAF_ID, 1002, network='host'), + container('so-nginx', NGINX_ID, 1003, health={'Status': 'healthy', 'FailingStreak': 0}), + ] + self.stats = { + 'so-soc': {'Name': 'so-soc', 'CPUPerc': '1.25%', 'MemPerc': '6.39%', 'NetIO': '1kB / 2kB'}, + 'so-telegraf': {'Name': 'so-telegraf', 'CPUPerc': '0.04%', 'MemPerc': '0.58%', 'NetIO': '0B / 0B'}, + 'so-nginx': {'Name': 'so-nginx', 'CPUPerc': '0.00%', 'MemPerc': '0.09%', 'NetIO': '3kB / 4kB'}, + } + self.info = {'Name': 'sohost', 'ServerVersion': '29.2.1', 'Containers': 3, 'ContainersRunning': 3, + 'ContainersStopped': 0, 'ContainersPaused': 0, 'Images': 9, 'NCPU': 8, + 'NGoroutines': 42, 'NFd': 77, 'NEventsListener': 1, 'MemTotal': 16000000000} + # one cgroup per container, addressed through /proc//cgroup exactly as the collector does + self.files = { + '/proc/meminfo': 'MemTotal: 15625000 kB\n', + '/proc/stat': 'cpu 100 200 300 400\ncpu0 1 2 3 4\n', + } + for pid in (1001, 1002, 1003): + self.files['/proc/%d/cgroup' % pid] = '0::/scope%d\n' % pid + self.cgroup(pid, 'cpu.stat', + 'usage_usec 1000\nuser_usec 600\nsystem_usec 400\nnr_periods 5\nnr_throttled 2\nthrottled_usec 700\n') + self.cgroup(pid, 'memory.stat', + 'active_anon 300\nactive_file 40\ninactive_anon 200\ninactive_file 50\nunevictable 0\npgfault 1234\npgmajfault 56\n') + self.cgroup(pid, 'memory.current', '1000\n') + self.cgroup(pid, 'memory.max', '4000\n') + self.cgroup(pid, 'memory.peak', '2500\n') + self.cgroup(pid, 'io.stat', '8:0 rbytes=100 wbytes=200 rios=1 wios=2\n252:0 rbytes=10 wbytes=20 rios=1 wios=1\n') + self.files['/proc/%d/net/dev' % pid] = ( + 'Inter-| Receive | Transmit\n' + ' face |bytes packets errs drop fifo frame compressed multicast|bytes packets errs drop fifo colls carrier compressed\n' + ' lo: 9999 99 9 9 0 0 0 0 9999 99 9 9 0 0 0 0\n' + ' eth0: 1000 10 1 2 0 0 0 0 2000 20 3 4 0 0 0 0\n' + ' eth1: 500 5 0 0 0 0 0 0 1000 10 0 0 0 0 0 0\n') + + def cgroup(self, pid, name, contents): + self.files['/sys/fs/cgroup/scope%d/%s' % (pid, name)] = contents + + def run_collector(self, settings=None, module=None): + module = module or render(settings if settings is not None else all_enabled()) + + def fake_docker(args): + if args[0] == 'stats': + return ''.join(json.dumps(entry) + '\n' for entry in self.stats.values()) + if args[0] == 'ps': + return ' '.join(entry['Id'] for entry in self.inspect) + if args[0] == 'inspect': + return json.dumps(self.inspect) + if args[0] == 'info': + return json.dumps(self.info) + raise AssertionError('unexpected docker call: %s' % args) + + module.docker = fake_docker + module.read_text = lambda path: self.files.get(path, '') + module.CGROUP_ROOT = '/sys/fs/cgroup' + buffer = io.StringIO() + with contextlib.redirect_stdout(buffer): + module.main() + self.output = buffer.getvalue() + return parse(self.output) + + +class TestShippedDefaults(CollectorTestCase): + + def test_defaults_emit_only_the_dashboard_fields(self): + # the Security Onion Performance dashboard queries exactly these five + points = self.run_collector(shipped_defaults()) + emitted = {(measurement, field) for (measurement, _), (_, fields) in points.items() for field in fields} + self.assertEqual(emitted, { + ('docker_container_cpu', 'usage_percent'), + ('docker_container_mem', 'usage_percent'), + ('docker_container_net', 'rx_bytes'), + ('docker_container_status', 'uptime_ns'), + ('docker_container_status', 'oomkilled'), + }) + + def test_defaults_do_not_emit_the_opt_in_measurements(self): + points = self.run_collector(shipped_defaults()) + measurements = {measurement for measurement, _ in points} + self.assertNotIn('docker', measurements) + self.assertNotIn('docker_container_blkio', measurements) + self.assertNotIn('docker_container_health', measurements) + + def test_defaults_carry_the_tags_the_dashboard_filters_on(self): + points = self.run_collector(shipped_defaults()) + tags, _ = points[('docker_container_cpu', 'so-soc')] + self.assertEqual(tags['container_status'], 'running') + self.assertEqual(tags['cpu'], 'cpu-total') + # identity tags are opt in, so they must be absent by default + self.assertNotIn('container_image', tags) + self.assertNotIn('engine_host', tags) + + +class TestToggles(CollectorTestCase): + + def test_enabling_one_stat_adds_only_that_field(self): + settings = shipped_defaults() + settings['cpu']['usage_total'] = True + _, fields = self.run_collector(settings)[('docker_container_cpu', 'so-soc')] + self.assertIn('usage_total', fields) + self.assertNotIn('usage_in_usermode', fields) + + def test_disabling_one_stat_leaves_its_neighbours(self): + settings = all_enabled() + settings['blkio']['io_service_bytes_recursive_read'] = False + _, fields = self.run_collector(settings)[('docker_container_blkio', 'so-soc')] + self.assertNotIn('io_service_bytes_recursive_read', fields) + self.assertIn('io_service_bytes_recursive_write', fields) + + def test_identity_tags_are_added_when_enabled(self): + tags, _ = self.run_collector(all_enabled())[('docker_container_cpu', 'so-soc')] + self.assertEqual(tags['container_image'], 'registry:5000/repo/img') + self.assertEqual(tags['container_version'], '3.4.0') + self.assertEqual(tags['engine_host'], 'sohost') + self.assertEqual(tags['server_version'], '29.2.1') + + def test_everything_off_emits_nothing(self): + settings = {group: {field: False for field in fields} for group, fields in shipped_defaults().items()} + self.assertEqual(self.run_collector(settings), {}) + + +class TestFieldTypes(CollectorTestCase): + """inputs.docker wrote the cgroup and network counters as unsigned; influx treats u and i as + different field types, so a mismatch breaks queries spanning the change.""" + + def test_counter_fields_are_unsigned(self): + points = self.run_collector(all_enabled()) + for measurement, field in (('docker_container_cpu', 'usage_total'), + ('docker_container_cpu', 'throttling_periods'), + ('docker_container_mem', 'usage'), + ('docker_container_mem', 'limit'), + ('docker_container_mem', 'pgfault'), + ('docker_container_net', 'rx_bytes'), + ('docker_container_blkio', 'io_service_bytes_recursive_read')): + _, fields = points[(measurement, 'so-soc')] + self.assertEqual(field_type(fields[field]), 'unsigned', '%s.%s' % (measurement, field)) + + def test_status_and_engine_fields_are_signed(self): + points = self.run_collector(all_enabled()) + _, status = points[('docker_container_status', 'so-soc')] + for field in ('uptime_ns', 'pid', 'exitcode', 'restart_count', 'started_at'): + self.assertEqual(field_type(status[field]), 'integer', field) + _, engine = points[('docker', '')] + self.assertEqual(field_type(engine['n_containers']), 'integer') + + def test_percentages_are_floats_and_ids_are_quoted_strings(self): + points = self.run_collector(all_enabled()) + _, cpu = points[('docker_container_cpu', 'so-soc')] + self.assertEqual(field_type(cpu['usage_percent']), 'float') + self.assertEqual(field_type(cpu['container_id']), 'string') + self.assertEqual(cpu['container_id'], '"%s"' % SOC_ID) + _, status = points[('docker_container_status', 'so-soc')] + self.assertEqual(field_type(status['oomkilled']), 'boolean') + _, health = points[('docker_container_health', 'so-nginx')] + self.assertEqual(health['health_status'], '"healthy"') + + +class TestMemorySemantics(CollectorTestCase): + + def test_usage_subtracts_reclaimable_page_cache(self): + # inputs.docker reports usage net of inactive_file: 1000 - 50 + _, fields = self.run_collector(all_enabled())[('docker_container_mem', 'so-soc')] + self.assertEqual(fields['usage'], '950u') + + def test_usage_percent_is_usage_over_limit(self): + _, fields = self.run_collector(all_enabled())[('docker_container_mem', 'so-soc')] + self.assertAlmostEqual(float(fields['usage_percent']), 950 / 4000 * 100.0) + + def test_limit_falls_back_to_host_memory_when_unlimited(self): + for pid in (1001, 1002, 1003): + self.cgroup(pid, 'memory.max', 'max\n') + _, fields = self.run_collector(all_enabled())[('docker_container_mem', 'so-soc')] + self.assertEqual(fields['limit'], '%du' % (15625000 * 1024)) + + def test_max_usage_reports_the_cgroup_peak(self): + # the daemon reports 0 on cgroup v2, so this deliberately carries the real peak + _, fields = self.run_collector(all_enabled())[('docker_container_mem', 'so-soc')] + self.assertEqual(fields['max_usage'], '2500u') + + +class TestCpuSemantics(CollectorTestCase): + + def test_microsecond_counters_are_published_as_nanoseconds(self): + _, fields = self.run_collector(all_enabled())[('docker_container_cpu', 'so-soc')] + self.assertEqual(fields['usage_total'], '1000000u') + self.assertEqual(fields['usage_in_usermode'], '600000u') + self.assertEqual(fields['usage_in_kernelmode'], '400000u') + self.assertEqual(fields['throttling_throttled_time'], '700000u') + + def test_throttling_counts_are_not_scaled(self): + _, fields = self.run_collector(all_enabled())[('docker_container_cpu', 'so-soc')] + self.assertEqual(fields['throttling_periods'], '5u') + self.assertEqual(fields['throttling_throttled_periods'], '2u') + + def test_usage_system_is_host_wide_cpu_time(self): + _, fields = self.run_collector(all_enabled())[('docker_container_cpu', 'so-soc')] + self.assertEqual(field_type(fields['usage_system']), 'unsigned') + self.assertNotEqual(fields['usage_system'], '0u') + + +class TestNetwork(CollectorTestCase): + + def test_counters_sum_interfaces_and_ignore_loopback(self): + _, fields = self.run_collector(all_enabled())[('docker_container_net', 'so-soc')] + self.assertEqual(fields['rx_bytes'], '1500u') + self.assertEqual(fields['rx_packets'], '15u') + self.assertEqual(fields['tx_bytes'], '3000u') + self.assertEqual(fields['rx_dropped'], '2u') + + def test_host_network_containers_emit_no_row(self): + # inputs.docker skipped these: its Networks map is empty for --net=host + points = self.run_collector(all_enabled()) + self.assertNotIn(('docker_container_net', 'so-telegraf'), points) + self.assertIn(('docker_container_net', 'so-soc'), points) + + def test_total_tag_is_present(self): + tags, _ = self.run_collector(all_enabled())[('docker_container_net', 'so-soc')] + self.assertEqual(tags['network'], 'total') + + +class TestBlkio(CollectorTestCase): + + def test_counters_sum_devices(self): + _, fields = self.run_collector(all_enabled())[('docker_container_blkio', 'so-soc')] + self.assertEqual(fields['io_service_bytes_recursive_read'], '110u') + self.assertEqual(fields['io_service_bytes_recursive_write'], '220u') + + def test_container_with_no_io_reports_zero_rather_than_disappearing(self): + for pid in (1001, 1002, 1003): + self.cgroup(pid, 'io.stat', '') + tags, fields = self.run_collector(all_enabled())[('docker_container_blkio', 'so-soc')] + self.assertEqual(fields['io_service_bytes_recursive_read'], '0u') + self.assertEqual(tags['device'], 'total') + + +class TestStatus(CollectorTestCase): + + def test_running_container_uptime_counts_from_start(self): + _, fields = self.run_collector(all_enabled())[('docker_container_status', 'so-soc')] + self.assertGreater(int(fields['uptime_ns'].rstrip('i')), 0) + self.assertNotIn('finished_at', fields) + + def test_exited_container_reports_its_lifetime_and_finished_at(self): + self.inspect.append(container('so-idstools', IDSTOOLS_ID, 0, status='exited', + started='2026-09-24T10:00:00.000000000Z', + finished='2026-09-24T10:00:02.000000000Z', exitcode=3, restarts=1)) + points = self.run_collector(all_enabled()) + tags, fields = points[('docker_container_status', 'so-idstools')] + self.assertEqual(tags['container_status'], 'exited') + self.assertEqual(fields['uptime_ns'], '2000000000i') + self.assertEqual(fields['finished_at'], '1790244002000000000i') + self.assertEqual(fields['exitcode'], '3i') + self.assertEqual(fields['restart_count'], '1i') + # a stopped container has no live stats, so only the status row is emitted + self.assertNotIn(('docker_container_cpu', 'so-idstools'), points) + + def test_health_is_emitted_only_for_containers_with_a_healthcheck(self): + points = self.run_collector(all_enabled()) + self.assertIn(('docker_container_health', 'so-nginx'), points) + self.assertNotIn(('docker_container_health', 'so-soc'), points) + + def test_oomkilled_is_reported(self): + self.inspect[0]['State']['OOMKilled'] = True + _, fields = self.run_collector(all_enabled())[('docker_container_status', 'so-soc')] + self.assertEqual(fields['oomkilled'], 'true') + + +class TestEngineMeasurement(CollectorTestCase): + + def test_engine_counts_come_from_docker_info(self): + _, fields = self.run_collector(all_enabled())[('docker', '')] + self.assertEqual(fields['n_containers'], '3i') + self.assertEqual(fields['n_cpus'], '8i') + self.assertEqual(fields['n_used_file_descriptors'], '77i') + + def test_engine_is_published_as_two_points(self): + # inputs.docker emitted memory_total on its own point, so keep that shape + self.run_collector(all_enabled()) + engine = [line for line in self.output.splitlines() if line.startswith('docker,')] + self.assertEqual(len(engine), 2) + self.assertTrue(any('memory_total=' in line for line in engine)) + self.assertTrue(any('n_containers=' in line for line in engine)) + + def test_engine_rows_are_tagged_with_host_and_version(self): + tags, _ = self.run_collector(all_enabled())[('docker', '')] + self.assertEqual(tags['engine_host'], 'sohost') + self.assertEqual(tags['server_version'], '29.2.1') + + +class TestLineProtocol(CollectorTestCase): + + def test_tag_values_are_escaped(self): + self.inspect[0]['Name'] = '/odd name,with=chars' + self.stats['odd name,with=chars'] = self.stats.pop('so-soc') + self.stats['odd name,with=chars']['Name'] = 'odd name,with=chars' + output = self.run_collector(all_enabled()) + self.assertIn(('docker_container_cpu', 'odd\\ name\\,with\\=chars'), output) + + def test_image_without_a_tag_reports_version_unknown(self): + self.inspect[0]['Config']['Image'] = 'busybox' + tags, _ = self.run_collector(all_enabled())[('docker_container_cpu', 'so-soc')] + self.assertEqual(tags['container_image'], 'busybox') + self.assertEqual(tags['container_version'], 'unknown') + + def test_every_line_has_a_measurement_tagset_and_fieldset(self): + module = render(all_enabled()) + + def fake_docker(args): + if args[0] == 'stats': + return ''.join(json.dumps(entry) + '\n' for entry in self.stats.values()) + if args[0] == 'ps': + return ' '.join(entry['Id'] for entry in self.inspect) + if args[0] == 'inspect': + return json.dumps(self.inspect) + return json.dumps(self.info) + + module.docker = fake_docker + module.read_text = lambda path: self.files.get(path, '') + buffer = io.StringIO() + with contextlib.redirect_stdout(buffer): + module.main() + lines = [line for line in buffer.getvalue().splitlines() if line.strip()] + self.assertTrue(lines) + for line in lines: + head, fieldpart = split_on_space(line) + self.assertIn(',', head, line) + self.assertIn('=', fieldpart, line) + self.assertFalse(fieldpart.endswith(','), line) + + +# group -> (measurement, the container whose row carries it) +GROUP_TARGET = { + 'engine': ('docker', ''), + 'cpu': ('docker_container_cpu', 'so-soc'), + 'mem': ('docker_container_mem', 'so-soc'), + 'net': ('docker_container_net', 'so-soc'), + 'blkio': ('docker_container_blkio', 'so-soc'), + 'status': ('docker_container_status', 'so-soc'), + 'health': ('docker_container_health', 'so-nginx'), +} + + +class TestEverySetting(CollectorTestCase): + """Whatever is offered in defaults.yaml has to actually be collectable. These tests are driven + off that file, so a new setting that is never wired up fails here rather than shipping.""" + + def setUp(self): + super().setUp() + # an exited container so status.finished_at has a value to report + self.inspect.append(container('so-idstools', IDSTOOLS_ID, 0, status='exited', + started='2026-09-24T10:00:00.000000000Z', + finished='2026-09-24T10:00:02.000000000Z')) + + def test_every_setting_emits_its_field_when_enabled(self): + points = self.run_collector(all_enabled()) + for group, fields in shipped_defaults().items(): + if group == 'tags': + continue + measurement, name = GROUP_TARGET[group] + if group == 'status': + # finished_at only exists for a container that has actually exited + _, exited = points[(measurement, 'so-idstools')] + self.assertIn('finished_at', exited) + _, emitted = points[(measurement, name)] + for field in fields: + if group == 'status' and field == 'finished_at': + continue + self.assertIn(field, emitted, '%s.%s is offered but never emitted' % (group, field)) + + def test_every_setting_is_individually_wired(self): + # enabling one stat on its own must produce exactly that field, proving each toggle is + # read rather than riding along with a neighbour + for group, fields in shipped_defaults().items(): + if group == 'tags': + continue + measurement, name = GROUP_TARGET[group] + for field in fields: + settings = {other: {key: False for key in values} for other, values in shipped_defaults().items()} + settings[group][field] = True + target = 'so-idstools' if (group == 'status' and field == 'finished_at') else name + points = self.run_collector(settings) + self.assertIn((measurement, target), points, '%s.%s emitted no row' % (group, field)) + _, emitted = points[(measurement, target)] + self.assertEqual(sorted(emitted), [field], '%s.%s did not emit itself alone' % (group, field)) + + def test_all_enabled_values_are_exact(self): + points = self.run_collector(all_enabled()) + clock = os.sysconf('SC_CLK_TCK') + expected = { + ('docker_container_cpu', 'so-soc'): { + 'usage_percent': '1.25', 'usage_total': '1000000u', 'usage_in_usermode': '600000u', + 'usage_in_kernelmode': '400000u', 'usage_system': '%du' % int(1000 * 10**9 / clock), + 'throttling_periods': '5u', 'throttling_throttled_periods': '2u', + 'throttling_throttled_time': '700000u', 'container_id': '"%s"' % SOC_ID, + }, + ('docker_container_mem', 'so-soc'): { + 'usage': '950u', 'limit': '4000u', 'max_usage': '2500u', 'active_anon': '300u', + 'active_file': '40u', 'inactive_anon': '200u', 'inactive_file': '50u', + 'unevictable': '0u', 'pgfault': '1234u', 'pgmajfault': '56u', + 'usage_percent': '23.75', 'container_id': '"%s"' % SOC_ID, + }, + ('docker_container_net', 'so-soc'): { + 'rx_bytes': '1500u', 'rx_packets': '15u', 'rx_errors': '1u', 'rx_dropped': '2u', + 'tx_bytes': '3000u', 'tx_packets': '30u', 'tx_errors': '3u', 'tx_dropped': '4u', + 'container_id': '"%s"' % SOC_ID, + }, + ('docker_container_blkio', 'so-soc'): { + 'io_service_bytes_recursive_read': '110u', + 'io_service_bytes_recursive_write': '220u', 'container_id': '"%s"' % SOC_ID, + }, + ('docker', ''): { + 'n_containers': '3i', 'n_containers_running': '3i', 'n_containers_stopped': '0i', + 'n_containers_paused': '0i', 'n_images': '9i', 'n_cpus': '8i', 'n_goroutines': '42i', + 'n_used_file_descriptors': '77i', 'n_listener_events': '1i', + 'memory_total': '16000000000i', + }, + ('docker_container_health', 'so-nginx'): { + 'health_status': '"healthy"', 'failing_streak': '0i', + }, + } + for key, fields in expected.items(): + _, emitted = points[key] + for field, value in fields.items(): + self.assertEqual(emitted[field], value, '%s %s' % (key[0], field)) + + def test_status_values_are_exact(self): + # uptime is relative to now, so it is checked separately from the fixed fields + points = self.run_collector(all_enabled()) + _, running = points[('docker_container_status', 'so-soc')] + self.assertEqual(running['pid'], '1001i') + self.assertEqual(running['exitcode'], '0i') + self.assertEqual(running['restart_count'], '0i') + self.assertEqual(running['oomkilled'], 'false') + self.assertEqual(running['container_id'], '"%s"' % SOC_ID) + self.assertEqual(int(running['started_at'].rstrip('i')) // 10**9, 1790244000) + self.assertGreater(int(running['uptime_ns'].rstrip('i')), 0) + _, exited = points[('docker_container_status', 'so-idstools')] + self.assertEqual(exited['uptime_ns'], '2000000000i') + self.assertEqual(exited['finished_at'], '1790244002000000000i') + + def test_tags_identity_toggle_controls_the_identity_tags(self): + settings = shipped_defaults() + settings['tags']['identity'] = False + tags, _ = self.run_collector(settings)[('docker_container_cpu', 'so-soc')] + self.assertEqual(sorted(tags), ['container_name', 'container_status', 'cpu']) + settings['tags']['identity'] = True + tags, _ = self.run_collector(settings)[('docker_container_cpu', 'so-soc')] + self.assertEqual(sorted(tags), ['container_image', 'container_name', 'container_status', + 'container_version', 'cpu', 'engine_host', 'server_version']) + + def test_identity_tags_cover_every_documented_tag(self): + tags, _ = self.run_collector(all_enabled())[('docker_container_cpu', 'so-soc')] + for tag in ('container_image', 'container_version', 'engine_host', 'server_version'): + self.assertIn(tag, tags) + + +class TestTemplate(unittest.TestCase): + + def test_template_renders_to_valid_python_for_the_shipped_defaults(self): + with open(TEMPLATE) as handle: + source = handle.read() + rendered = jinja2.Template(source, keep_trailing_newline=True).render(CONTAINER_STATS=shipped_defaults()) + compile(rendered, 'so-container-stats', 'exec') + self.assertNotIn('{%', rendered) + # the docker format strings must survive rendering untouched + self.assertIn('{{json .}}', rendered) + + def test_every_annotated_setting_exists_in_defaults(self): + # SOC reads both trees; an annotation without a default cannot be reverted in the UI + with open(os.path.join(HERE, '..', '..', 'soc_telegraf.yaml')) as handle: + annotated = yaml.safe_load(handle)['telegraf']['container_stats'] + defaults = shipped_defaults() + for group, fields in annotated.items(): + self.assertIn(group, defaults) + for field in fields: + self.assertIn(field, defaults[group], '%s.%s annotated but missing from defaults' % (group, field)) + + def test_every_default_setting_is_annotated_for_soc(self): + with open(os.path.join(HERE, '..', '..', 'soc_telegraf.yaml')) as handle: + annotated = yaml.safe_load(handle)['telegraf']['container_stats'] + for group, fields in shipped_defaults().items(): + self.assertIn(group, annotated) + for field in fields: + self.assertIn(field, annotated[group], '%s.%s missing a SOC annotation' % (group, field)) + + +if __name__ == '__main__': + unittest.main() diff --git a/setup/so-functions b/setup/so-functions index c5fc8fabf..41be9c7a7 100755 --- a/setup/so-functions +++ b/setup/so-functions @@ -1611,6 +1611,7 @@ reserve_group_ids() { logCmd "groupadd -g 949 elastic-agent" logCmd "groupadd -g 947 elastic-fleet" logCmd "groupadd -g 960 kafka" + logCmd "groupadd -g 961 somon" } reserve_ports() { From 26d895ccb7656ad5614deb292733f9b3ab656ffb Mon Sep 17 00:00:00 2001 From: Jason Ertel Date: Mon, 28 Sep 2026 13:47:16 -0400 Subject: [PATCH 02/15] alarms and ntf --- salt/soc/defaults.yaml | 9 +++++++- salt/soc/soc_soc.yaml | 48 ++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 56 insertions(+), 1 deletion(-) diff --git a/salt/soc/defaults.yaml b/salt/soc/defaults.yaml index 763768d72..64a734032 100644 --- a/salt/soc/defaults.yaml +++ b/salt/soc/defaults.yaml @@ -1496,7 +1496,7 @@ soc: verifyCert: false notification: dismissedPruneDays: 30 - enabled: false + enabled: true playbook: autoUpdateEnabled: true playbookImportFrequencySeconds: 86400 @@ -1634,6 +1634,13 @@ soc: database: securityonion user: "" password: "" + postgresmetrics: + host: "" + port: 5432 + sslMode: "require" + database: telegraf + user: "" + password: "" salt: queueDir: /opt/sensoroni/queue timeoutMs: 45000 diff --git a/salt/soc/soc_soc.yaml b/salt/soc/soc_soc.yaml index 3d1594437..641ac4d58 100644 --- a/salt/soc/soc_soc.yaml +++ b/salt/soc/soc_soc.yaml @@ -160,6 +160,7 @@ soc: description: Schedules that are shared across the Security Onion product. Modify via one of the SOC Schedules view. readonlyUi: True global: True + advanced: True forcedType: string syntax: json storage: db @@ -495,6 +496,7 @@ soc: description: JSON list of notifications. Modify via the SOC Notifications view. readonlyUi: True global: True + advanced: True forcedType: string syntax: json storage: db @@ -503,6 +505,10 @@ soc: description: The number of days to retain dismissed notifications. When a notification is dismissed, it will be pruned after this many days. Only one user need dismiss a notification for it to be pruned. forcedType: int global: True + maxListLimit: + description: Maximum number of notifications to display. + forcedType: int + global: True enabled: description: Enables or disables the SOC notification module. forcedType: bool @@ -533,6 +539,48 @@ soc: global: True sensitive: True advanced: True + postgresmetrics: + host: + description: Hostname or IP address of the PostgreSQL server used by Telegraf. Defaults to the manager hostname. + global: True + advanced: True + port: + description: Port of the PostgreSQL server used by Telegraf. + global: True + advanced: True + sslMode: + description: "Use encrypted connections to the PostgreSQL server used by Telegraf. Must be one of the following values: disable, allow, prefer, require, verify-ca, verify-full." + global: True + advanced: True + database: + description: Database to authenticate to on the PostgreSQL server. + global: True + advanced: True + user: + description: Username to authenticate to the PostgreSQL server used by Telegraf. + global: True + advanced: True + password: + description: Password used to authenticate to the PostgreSQL server used by Telegraf. + global: True + sensitive: True + advanced: True + cacheExpirationMs: + description: The interval (in milliseconds) to wait before querying the DB for updated metrics. + global: True + advanced: True + maxMetricAgeSeconds: + description: The maximum age (in seconds) of metrics to display in the SOC Grid Metrics view. Metrics older than this value will not be displayed. + global: True + advanced: True + alarms: + description: JSON list of metric alarms. Modify via the SOC Grid Alarms view. + readonlyUi: True + advanced: True + global: True + forcedType: string + syntax: json + storage: db salt: longRelayTimeoutMs: description: Duration (in milliseconds) to wait for a response from the Salt API when executing tasks known for being long running before giving up and showing an error on the SOC UI. From a06f08217ab68e4372d2cdffa83b86c32c3bd54f Mon Sep 17 00:00:00 2001 From: Mike Reeves Date: Tue, 29 Sep 2026 11:42:29 -0400 Subject: [PATCH 03/15] Raise SOAI Sonnet default small context limit to 1M Context is now flat-priced, so match contextLimitSmall to contextLimitLarge. With equal limits the SOC assistant hides the increase-context toggle. --- salt/soc/defaults.yaml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/salt/soc/defaults.yaml b/salt/soc/defaults.yaml index 64a734032..2783fb882 100644 --- a/salt/soc/defaults.yaml +++ b/salt/soc/defaults.yaml @@ -2806,7 +2806,7 @@ soc: - id: sonnet displayName: Claude Sonnet origin: USA - contextLimitSmall: 200000 + contextLimitSmall: 1000000 contextLimitLarge: 1000000 lowBalanceColorAlert: 500000 enabled: true From eb803dce0eb4777b7f3fdd08c38aee248ab89e8b Mon Sep 17 00:00:00 2001 From: Jason Ertel Date: Tue, 29 Sep 2026 12:02:48 -0400 Subject: [PATCH 04/15] resolve startup errors --- salt/common/tools/sbin/so-log-check | 1 + salt/soc/defaults.yaml | 7 ------- 2 files changed, 1 insertion(+), 7 deletions(-) diff --git a/salt/common/tools/sbin/so-log-check b/salt/common/tools/sbin/so-log-check index 39f36eacc..211a3f58d 100755 --- a/salt/common/tools/sbin/so-log-check +++ b/salt/common/tools/sbin/so-log-check @@ -177,6 +177,7 @@ if [[ $EXCLUDE_FALSE_POSITIVE_ERRORS == 'Y' ]]; then EXCLUDED_ERRORS="$EXCLUDED_ERRORS|Unexpected authorization header" # expected WARN log lines indicating invalid auth header EXCLUDED_ERRORS="$EXCLUDED_ERRORS|Missing ory_kratos_session cookie" # expected WARN log lines indicating invalid auth header EXCLUDED_ERRORS="$EXCLUDED_ERRORS|Static assets preprocessor only supports GET and HEAD requests" # expected WARN log lines indicating invalid auth header + EXCLUDED_ERRORS="$EXCLUDED_ERRORS|respondError" # respondError is a function name, output via http middleware as standard request logging fi if [[ $EXCLUDE_KNOWN_ERRORS == 'Y' ]]; then diff --git a/salt/soc/defaults.yaml b/salt/soc/defaults.yaml index 64a734032..3751f44b7 100644 --- a/salt/soc/defaults.yaml +++ b/salt/soc/defaults.yaml @@ -1634,13 +1634,6 @@ soc: database: securityonion user: "" password: "" - postgresmetrics: - host: "" - port: 5432 - sslMode: "require" - database: telegraf - user: "" - password: "" salt: queueDir: /opt/sensoroni/queue timeoutMs: 45000 From 855716846ad73317cd08673cecd3302f32f7b235 Mon Sep 17 00:00:00 2001 From: Corey Ogburn Date: Tue, 29 Sep 2026 16:50:04 -0600 Subject: [PATCH 05/15] New Automation Fields --- salt/soc/defaults.yaml | 8 +++++++ salt/soc/soc_soc.yaml | 50 ++++++++++++++++++++++++++++++++++++++++++ 2 files changed, 58 insertions(+) diff --git a/salt/soc/defaults.yaml b/salt/soc/defaults.yaml index a8d2b8496..4eeae5579 100644 --- a/salt/soc/defaults.yaml +++ b/salt/soc/defaults.yaml @@ -1561,6 +1561,14 @@ soc: reconcilePersona: "" toolUseTurnAttempts: 12 toolUseTurnDelayMs: 175 + agentSessionMaxTurns: 20 + agentStreamFlushIntervalMs: 1000 + agentStreamIdleTimeoutSeconds: 300 + automationSettings: + tickIntervalSeconds: 60 + maxConcurrentItems: 4 + maxQueuedItems: 0 + alertTriageEpoch: "2026-09-24T00:00:00Z" tools: filterEventFields: - "@timestamp" diff --git a/salt/soc/soc_soc.yaml b/salt/soc/soc_soc.yaml index adb19833e..23e8bb628 100644 --- a/salt/soc/soc_soc.yaml +++ b/salt/soc/soc_soc.yaml @@ -861,6 +861,16 @@ soc: description: Indicates if the Assistant Module should operate in agentic mode or not. If true, agents can work together to solve tasks. global: True forcedType: bool + automations: + template: + description: Scheduled automations for the Onion AI assistant, managed from the Agent Studio. Each automation is stored under its own generated ID so its history and rollback are independent of every other automation. + global: True + advanced: False + readonlyUi: True + duplicates: True + forcedType: string + syntax: json + helpLink: onion-ai agents: description: Agent definitions for the Onion AI assistant, managed from the Agent Studio. An entry naming a system agent overrides only the fields an admin may change; everything else comes from the built-in definition. global: True @@ -893,6 +903,9 @@ soc: - field: persona label: Persona multiline: True + - field: maxConcurrentInstances + label: Max Concurrent Instances + forcedType: int skills: description: Skill definitions for the Onion AI assistant, managed from the Agent Studio. An entry naming a system skill overrides only its enabled state and persona addendum; its tool set comes from the built-in definition. global: True @@ -996,6 +1009,43 @@ soc: description: The number of times to retry extracting memories from a session if errors occur. global: True advanced: True + agentSessionMaxTurns: + description: Maximum number of model turns a headless agent session, such as one started by an automation, may take before it is stopped. Turns taken by delegated sub-agents count toward this limit. A session that reaches the limit is recorded as failed. + global: True + advanced: True + forcedType: int + agentStreamFlushIntervalMs: + description: Milliseconds between writes of a streaming headless agent turn to the database. Lower values show progress sooner in the Agent Studio at the cost of more frequent Elasticsearch updates. + global: True + advanced: True + forcedType: int + agentStreamIdleTimeoutSeconds: + description: Seconds a streaming headless agent turn may go without receiving any output before it is abandoned and the session is recorded as failed. Set to 0 to disable the timeout. + global: True + advanced: True + forcedType: int + automationSettings: + tickIntervalSeconds: + description: How often, in seconds, the automation scheduler checks for automations that are due to run. Must be greater than 0. + global: True + advanced: True + forcedType: int + maxConcurrentItems: + description: Maximum number of automation work items that may run at the same time. Additional work items wait in the queue until a running item finishes. User chat sessions count toward this limit but are never held back by it. Set to 0 to disable the limit. + global: True + advanced: True + forcedType: int + maxQueuedItems: + description: Maximum number of automation work items that may wait to start. Once the queue is full, no new work items are created until the backlog drains. Set to 0 to disable the limit. + global: True + advanced: True + forcedType: int + alertTriageEpoch: + description: The earliest alert time the Alert Triage automation will consider. Alerts before this time are never triaged, which keeps a first run on an existing deployment from working through old history. Must be in UTC format (2026-09-24T00:00:00Z). + regex: '^(\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}(\.\d+)?Z)?$' + regexFailureMessage: Expecting date in RFC3339 format (2026-09-24T00:00:00Z) + global: True + advanced: True tools: filterEventFields: description: A whitelist of fields to return when OnionAI uses the query_events tool. All other fields are removed. One field per line. From d122ee7feac875e0cbaeebd080d3984557beee27 Mon Sep 17 00:00:00 2001 From: reyesj2 <94730068+reyesj2@users.noreply.github.com> Date: Wed, 30 Sep 2026 16:49:19 -0500 Subject: [PATCH 06/15] review integration-defaults weird_integrations mappings, removed unused, updated logstash integration naming --- salt/elasticfleet/integration-defaults.map.jinja | 7 ++----- salt/manager/tools/sbin/soup | 11 +++++++++++ 2 files changed, 13 insertions(+), 5 deletions(-) diff --git a/salt/elasticfleet/integration-defaults.map.jinja b/salt/elasticfleet/integration-defaults.map.jinja index 25e307331..559caad5a 100644 --- a/salt/elasticfleet/integration-defaults.map.jinja +++ b/salt/elasticfleet/integration-defaults.map.jinja @@ -30,17 +30,14 @@ 'azure_metrics.monitor': 'azure.monitor', 'azure_metrics.storage_account': 'azure.storage_account', 'azure_openai.metrics': 'azure.open_ai', - 'beat.state': 'beats.stack_monitoring.state', - 'beat.stats': 'beats.stack_monitoring.stats', - 'enterprisesearch.health': 'enterprisesearch.stack_monitoring.health', - 'enterprisesearch.stats': 'enterprisesearch.stack_monitoring.stats', 'kibana.cluster_actions': 'kibana.stack_monitoring.cluster_actions', 'kibana.cluster_rules': 'kibana.stack_monitoring.cluster_rules', 'kibana.node_actions': 'kibana.stack_monitoring.node_actions', 'kibana.node_rules': 'kibana.stack_monitoring.node_rules', 'kibana.stats': 'kibana.stack_monitoring.stats', 'kibana.status': 'kibana.stack_monitoring.status', - 'logstash.node_cel': 'logstash.stack_monitoring.node', + 'logstash.node': 'logstash.stack_monitoring.node', + 'logstash.node_cel': 'logstash.node', 'logstash.node_stats': 'logstash.stack_monitoring.node_stats', 'synthetics.browser': 'synthetics-browser', 'synthetics.browser_network': 'synthetics-browser.network', diff --git a/salt/manager/tools/sbin/soup b/salt/manager/tools/sbin/soup index 6a2af1c2d..20c2528dc 100755 --- a/salt/manager/tools/sbin/soup +++ b/salt/manager/tools/sbin/soup @@ -1183,6 +1183,13 @@ up_to_3.4.0() { echo "Removing so-kratos, so-hydra and so-soc so they are recreated on the soauth network." docker rm -f so-kratos so-hydra so-soc >> $SOUP_LOG 2>&1 + for template in so-metrics-logstash.node so-metrics-logstash.stack_monitoring.node; do + if ! remove_elasticsearch_index_template "$template" "logstash node and node_cel index patterns reversed"; then + FINAL_MESSAGE_QUEUE+=("WARNING: Unable to automatically remove the $template index template. Addon integration templates may fail to load until it is removed:") + FINAL_MESSAGE_QUEUE+=(" - sudo so-elasticsearch-query _index_template/$template -XDELETE && so-checkin") + fi + done + INSTALLEDVERSION=3.4.0 } @@ -1248,6 +1255,10 @@ valid_soauth_range() { } post_to_3.4.0() { + for idx in "metrics-logstash.node-default" "metrics-logstash.stack_monitoring.node-default"; do + rollover_index "$idx" + done + set_postversion 3.4.0 } ### 3.4.0 End ### From b4557e973c1ba5749630efbd80a872d8fc3eb878 Mon Sep 17 00:00:00 2001 From: Jason Ertel Date: Thu, 1 Oct 2026 10:03:39 -0400 Subject: [PATCH 07/15] support empty yaml files --- salt/manager/tools/sbin/so-yaml.py | 3 +- salt/manager/tools/sbin/so-yaml_test.py | 99 +++++++++++++++++++++++++ 2 files changed, 101 insertions(+), 1 deletion(-) diff --git a/salt/manager/tools/sbin/so-yaml.py b/salt/manager/tools/sbin/so-yaml.py index d0d5209f9..d6465e97b 100755 --- a/salt/manager/tools/sbin/so-yaml.py +++ b/salt/manager/tools/sbin/so-yaml.py @@ -42,7 +42,8 @@ def loadYaml(filename): try: with open(filename, "r") as file: content = file.read() - return yaml.safe_load(content) + loaded = yaml.safe_load(content) + return loaded if loaded is not None else {} except FileNotFoundError: print(f"File not found: {filename}", file=sys.stderr) sys.exit(1) diff --git a/salt/manager/tools/sbin/so-yaml_test.py b/salt/manager/tools/sbin/so-yaml_test.py index 56581f7e3..7eb30be1f 100644 --- a/salt/manager/tools/sbin/so-yaml_test.py +++ b/salt/manager/tools/sbin/so-yaml_test.py @@ -95,6 +95,21 @@ class TestRemove(unittest.TestCase): expected = "key1:\n child1: 123\n child2:\n deep2: ab\nkey2: false\n" self.assertEqual(actual, expected) + def test_remove_empty_file(self): + filename = "/tmp/so-yaml_test-remove-empty.yaml" + file = open(filename, "w") + file.close() + + code = soyaml.remove([filename, "key1"]) + self.assertEqual(code, 0) + + file = open(filename, "r") + actual = file.read() + file.close() + + self.assertEqual(actual, "{}\n") + + def test_remove_missing_args(self): with patch('sys.exit', new=MagicMock()) as sysmock: with patch('sys.stderr', new=StringIO()) as mock_stderr: @@ -294,6 +309,36 @@ class TestRemove(unittest.TestCase): expected = "key1:\n child1: 123\n child2:\n deep1: 45\n deep2: d\nkey2: false\nkey3:\n- e\n- f\n- g\n" self.assertEqual(actual, expected) + def test_add_empty_file(self): + filename = "/tmp/so-yaml_test-add-empty.yaml" + file = open(filename, "w") + file.close() + + code = soyaml.add([filename, "telegraf.output", "BOTH"]) + self.assertEqual(code, 0) + + file = open(filename, "r") + actual = file.read() + file.close() + + expected = "telegraf:\n output: BOTH\n" + self.assertEqual(actual, expected) + + def test_add_empty_file_simple(self): + filename = "/tmp/so-yaml_test-add-empty-simple.yaml" + file = open(filename, "w") + file.close() + + code = soyaml.add([filename, "telegraf", "BOTH"]) + self.assertEqual(code, 0) + + file = open(filename, "r") + actual = file.read() + file.close() + + expected = "telegraf: BOTH\n" + self.assertEqual(actual, expected) + def test_replace_missing_arg(self): with patch('sys.exit', new=MagicMock()) as sysmock: with patch('sys.stderr', new=StringIO()) as mock_stderr: @@ -346,6 +391,21 @@ class TestRemove(unittest.TestCase): expected = "key1:\n child1: 123\n child2:\n deep1: 46\nkey2: false\nkey3:\n- e\n- f\n- g\n" self.assertEqual(actual, expected) + def test_replace_empty_file(self): + filename = "/tmp/so-yaml_test-replace-empty.yaml" + file = open(filename, "w") + file.close() + + code = soyaml.replace([filename, "telegraf.output", "BOTH"]) + self.assertEqual(code, 0) + + file = open(filename, "r") + actual = file.read() + file.close() + + expected = "telegraf:\n output: BOTH\n" + self.assertEqual(actual, expected) + def test_convert(self): self.assertEqual(soyaml.convertType("foo"), "foo") self.assertEqual(soyaml.convertType("foo.bar"), "foo.bar") @@ -506,6 +566,18 @@ class TestRemove(unittest.TestCase): self.assertEqual(result, 2) self.assertEqual("", mock_stdout.getvalue()) + def test_get_empty_file(self): + with patch('sys.stdout', new=StringIO()) as mock_stdout: + with patch('sys.stderr', new=StringIO()) as mock_stderr: + filename = "/tmp/so-yaml_test-get-empty.yaml" + file = open(filename, "w") + file.close() + + result = soyaml.get([filename, "telegraf.output"]) + self.assertEqual(result, 2) + self.assertEqual("", mock_stdout.getvalue()) + self.assertIn("Key 'telegraf.output' not found by so-yaml.py", mock_stderr.getvalue()) + def test_get_usage(self): with patch('sys.exit', new=MagicMock()) as sysmock: with patch('sys.stderr', new=StringIO()) as mock_stderr: @@ -991,3 +1063,30 @@ class TestLoadYaml(unittest.TestCase): soyaml.loadYaml("/tmp/so-yaml_test-unreadable.yaml") sysmock.assert_called_with(1) self.assertIn("Error reading file", mock_stderr.getvalue()) + + def test_load_yaml_empty_file(self): + filename = "/tmp/so-yaml_test-load-empty.yaml" + file = open(filename, "w") + file.close() + + result = soyaml.loadYaml(filename) + self.assertEqual(result, {}) + + def test_load_yaml_whitespace_only(self): + filename = "/tmp/so-yaml_test-load-whitespace.yaml" + file = open(filename, "w") + file.write(" \n\n \n") + file.close() + + result = soyaml.loadYaml(filename) + self.assertEqual(result, {}) + + def test_load_yaml_comments_only(self): + filename = "/tmp/so-yaml_test-load-comments.yaml" + file = open(filename, "w") + file.write("# Just a comment\n# Another comment\n") + file.close() + + result = soyaml.loadYaml(filename) + self.assertEqual(result, {}) + From 523c39d4f21ad191de159eeb3cfaaa7dedcf2d65 Mon Sep 17 00:00:00 2001 From: Jason Ertel Date: Thu, 1 Oct 2026 10:09:20 -0400 Subject: [PATCH 08/15] fix flake --- salt/manager/tools/sbin/so-yaml_test.py | 2 -- 1 file changed, 2 deletions(-) diff --git a/salt/manager/tools/sbin/so-yaml_test.py b/salt/manager/tools/sbin/so-yaml_test.py index 7eb30be1f..7fa4cf8a7 100644 --- a/salt/manager/tools/sbin/so-yaml_test.py +++ b/salt/manager/tools/sbin/so-yaml_test.py @@ -109,7 +109,6 @@ class TestRemove(unittest.TestCase): self.assertEqual(actual, "{}\n") - def test_remove_missing_args(self): with patch('sys.exit', new=MagicMock()) as sysmock: with patch('sys.stderr', new=StringIO()) as mock_stderr: @@ -1089,4 +1088,3 @@ class TestLoadYaml(unittest.TestCase): result = soyaml.loadYaml(filename) self.assertEqual(result, {}) - From 89f8bcd19f291705569f3206595a8ec1adfc5e21 Mon Sep 17 00:00:00 2001 From: defensivedepth Date: Thu, 1 Oct 2026 10:28:10 -0400 Subject: [PATCH 09/15] evtx-import fixup --- .../grid-nodes_general/import-evtx-logs.json | 2 +- salt/elasticsearch/files/ingest/global@custom | 4 +-- salt/elasticsearch/files/ingest/import.evtx | 31 +++++++++++++++++++ 3 files changed, 34 insertions(+), 3 deletions(-) create mode 100644 salt/elasticsearch/files/ingest/import.evtx diff --git a/salt/elasticfleet/files/integrations/grid-nodes_general/import-evtx-logs.json b/salt/elasticfleet/files/integrations/grid-nodes_general/import-evtx-logs.json index 723370ba6..6f6bb8628 100644 --- a/salt/elasticfleet/files/integrations/grid-nodes_general/import-evtx-logs.json +++ b/salt/elasticfleet/files/integrations/grid-nodes_general/import-evtx-logs.json @@ -29,7 +29,7 @@ "\\.gz$" ], "include_files": [], - "processors": "- dissect:\n tokenizer: \"/nsm/import/%{import.id}/evtx/%{import.file}\"\n field: \"log.file.path\"\n target_prefix: \"\"\n- decode_json_fields:\n fields: [\"message\"]\n target: \"\"\n- drop_fields:\n fields: [\"host\"]\n ignore_missing: true\n- add_fields:\n target: data_stream\n fields:\n type: logs\n dataset: system.security\n- add_fields:\n target: event\n fields:\n dataset: system.security\n module: system\n imported: true\n- add_fields:\n target: \"@metadata\"\n fields:\n pipeline: logs-system.security-2.22.3\n- if:\n equals:\n winlog.channel: 'Microsoft-Windows-Sysmon/Operational'\n then: \n - add_fields:\n target: data_stream\n fields:\n dataset: windows.sysmon_operational\n - add_fields:\n target: event\n fields:\n dataset: windows.sysmon_operational\n module: windows\n imported: true\n - add_fields:\n target: \"@metadata\"\n fields:\n pipeline: logs-windows.sysmon_operational-3.9.0\n- if:\n equals:\n winlog.channel: 'Application'\n then: \n - add_fields:\n target: data_stream\n fields:\n dataset: system.application\n - add_fields:\n target: event\n fields:\n dataset: system.application\n - add_fields:\n target: \"@metadata\"\n fields:\n pipeline: logs-system.application-2.22.3\n- if:\n equals:\n winlog.channel: 'System'\n then: \n - add_fields:\n target: data_stream\n fields:\n dataset: system.system\n - add_fields:\n target: event\n fields:\n dataset: system.system\n - add_fields:\n target: \"@metadata\"\n fields:\n pipeline: logs-system.system-2.22.3\n \n- if:\n equals:\n winlog.channel: 'Microsoft-Windows-PowerShell/Operational'\n then: \n - add_fields:\n target: data_stream\n fields:\n dataset: windows.powershell_operational\n - add_fields:\n target: event\n fields:\n dataset: windows.powershell_operational\n module: windows\n - add_fields:\n target: \"@metadata\"\n fields:\n pipeline: logs-windows.powershell_operational-3.9.0\n- add_fields:\n target: data_stream\n fields:\n dataset: import", + "processors": "- dissect:\n tokenizer: \"/nsm/import/%{import.id}/evtx/%{import.file}\"\n field: \"log.file.path\"\n target_prefix: \"\"\n- decode_json_fields:\n fields: [\"message\"]\n target: \"\"\n- add_fields:\n target: event\n fields:\n dataset: windows.forwarded\n module: windows\n imported: true\n- add_fields:\n target: \"@metadata\"\n fields:\n pipeline: import.evtx\n- if:\n equals:\n winlog.channel: 'Security'\n then: \n - add_fields:\n target: event\n fields:\n dataset: system.security\n module: system\n- if:\n equals:\n winlog.channel: 'Windows PowerShell'\n then: \n - add_fields:\n target: event\n fields:\n dataset: windows.powershell\n module: windows\n- if:\n equals:\n winlog.channel: 'Microsoft-Windows-Sysmon/Operational'\n then: \n - add_fields:\n target: event\n fields:\n dataset: windows.sysmon_operational\n module: windows\n imported: true\n- if:\n equals:\n winlog.channel: 'Application'\n then: \n - add_fields:\n target: event\n fields:\n dataset: system.application\n- if:\n equals:\n winlog.channel: 'System'\n then: \n - add_fields:\n target: event\n fields:\n dataset: system.system\n \n- if:\n equals:\n winlog.channel: 'Microsoft-Windows-PowerShell/Operational'\n then: \n - add_fields:\n target: event\n fields:\n dataset: windows.powershell_operational\n module: windows\n- add_fields:\n target: data_stream\n fields:\n type: logs\n dataset: import", "tags": [ "import" ], diff --git a/salt/elasticsearch/files/ingest/global@custom b/salt/elasticsearch/files/ingest/global@custom index 979c5c1b8..ab1418cd2 100644 --- a/salt/elasticsearch/files/ingest/global@custom +++ b/salt/elasticsearch/files/ingest/global@custom @@ -99,7 +99,7 @@ }, { "set": { - "if": "ctx.tags != null && ctx.tags.contains('import')", + "if": "ctx.tags != null && ctx.tags.contains('import') && ctx._index != null && ctx._index.startsWith('logs-import-')", "override": true, "field": "data_stream.dataset", "value": "import" @@ -107,7 +107,7 @@ }, { "set": { - "if": "ctx.tags != null && ctx.tags.contains('import')", + "if": "ctx.tags != null && ctx.tags.contains('import') && ctx._index != null && ctx._index.startsWith('logs-import-')", "override": true, "field": "data_stream.namespace", "value": "so" diff --git a/salt/elasticsearch/files/ingest/import.evtx b/salt/elasticsearch/files/ingest/import.evtx new file mode 100644 index 000000000..99189130c --- /dev/null +++ b/salt/elasticsearch/files/ingest/import.evtx @@ -0,0 +1,31 @@ +{ + "description" : "import.evtx: normalize imported EVTX and reroute to logs--import", + "processors" : [ + { "script": { + "description": "Host from the event, not the importing node", + "lang": "painless", + "source": "Map host = ['os': ['type': 'windows', 'family': 'windows', 'platform': 'windows']]; def cn = ctx.winlog?.computer_name; if (cn != null && cn.toString().length() > 0) { String name = cn.toString(); int dot = name.indexOf('.'); if (dot > 0) { name = name.substring(0, dot); } host.put('hostname', name); host.put('name', name.toLowerCase()); } ctx.host = host;" + } }, + { "script": { + "description": "String event IDs, as Winlogbeat sends", + "lang": "painless", + "source": "if (ctx.winlog?.event_id != null) { ctx.winlog.event_id = ctx.winlog.event_id.toString(); } if (ctx.event?.code != null) { ctx.event.code = ctx.event.code.toString(); }" + } }, + { "script": { + "description": "Unnamed to param1..N, as Winlogbeat", + "lang": "painless", + "if": "ctx.winlog?.event_data?.Data instanceof Map && ctx.winlog.event_data.Data['#text'] != null", + "source": "def t = ctx.winlog.event_data.Data['#text']; List vals = t instanceof List ? t : [t]; for (int i = 0; i < vals.size(); i++) { ctx.winlog.event_data['param' + (i + 1)] = vals.get(i); } ctx.winlog.event_data.remove('Data');" + } }, + { "script": { + "description": "String values and LF line endings, as Winlogbeat", + "lang": "painless", + "if": "ctx.winlog?.event_data instanceof Map || ctx.winlog?.user_data instanceof Map", + "source": "String lf = String.valueOf((char) 10); String crlf = String.valueOf((char) 13) + lf; for (def key : ['event_data', 'user_data']) { def m = ctx.winlog[key]; if (!(m instanceof Map)) { continue; } for (def e : m.entrySet()) { def v = e.getValue(); if (v instanceof String) { e.setValue(v.replace(crlf, lf)); } else if (v instanceof Number || v instanceof Boolean) { e.setValue(v.toString()); } } }" + } }, + { "set": { "description": "event.kind, as Winlogbeat", "field": "event.kind", "value": "event", "override": false } }, + { "set": { "field": "data_stream.dataset", "copy_from": "event.dataset", "override": true, "ignore_empty_value": true } }, + { "set": { "field": "data_stream.namespace", "value": "import", "override": true } }, + { "reroute": { "dataset": "{{data_stream.dataset}}", "namespace": "{{data_stream.namespace}}" } } + ] +} From 2a4611df45e17b3ba98b376bddeaccf9f8a232b2 Mon Sep 17 00:00:00 2001 From: defensivedepth Date: Thu, 1 Oct 2026 10:48:55 -0400 Subject: [PATCH 10/15] Add caseless mappings --- salt/elasticsearch/defaults.yaml | 5 + .../so-fleet_process_caseless-1.json | 123 ++++++++++++++++++ .../so-fleet_system.security_caseless-1.json | 80 ++++++++++++ .../files/soc/sigma_playbook_pipeline.yaml | 1 - salt/soc/files/soc/sigma_so_pipeline.yaml | 10 ++ 5 files changed, 218 insertions(+), 1 deletion(-) create mode 100644 salt/elasticsearch/templates/component/elastic-agent/so-fleet_process_caseless-1.json create mode 100644 salt/elasticsearch/templates/component/elastic-agent/so-fleet_system.security_caseless-1.json diff --git a/salt/elasticsearch/defaults.yaml b/salt/elasticsearch/defaults.yaml index f67a01df9..c8ceab34d 100644 --- a/salt/elasticsearch/defaults.yaml +++ b/salt/elasticsearch/defaults.yaml @@ -3309,6 +3309,7 @@ elasticsearch: composed_of: - event-mappings - logs-system.security@package + - so-fleet_system.security_caseless-1 - logs-system.security@custom - so-fleet_integrations.ip_mappings-1 - so-fleet_globals-1 @@ -4175,6 +4176,7 @@ elasticsearch: index_template: composed_of: - logs-windows.forwarded@package + - so-fleet_process_caseless-1 - logs-windows.forwarded@custom - so-fleet_integrations.ip_mappings-1 - so-fleet_globals-1 @@ -4224,6 +4226,7 @@ elasticsearch: index_template: composed_of: - logs-windows.powershell@package + - so-fleet_process_caseless-1 - logs-windows.powershell@custom - so-fleet_integrations.ip_mappings-1 - so-fleet_globals-1 @@ -4273,6 +4276,7 @@ elasticsearch: index_template: composed_of: - logs-windows.powershell_operational@package + - so-fleet_process_caseless-1 - logs-windows.powershell_operational@custom - so-fleet_integrations.ip_mappings-1 - so-fleet_globals-1 @@ -4322,6 +4326,7 @@ elasticsearch: index_template: composed_of: - logs-windows.sysmon_operational@package + - so-fleet_process_caseless-1 - logs-windows.sysmon_operational@custom - so-fleet_integrations.ip_mappings-1 - so-fleet_globals-1 diff --git a/salt/elasticsearch/templates/component/elastic-agent/so-fleet_process_caseless-1.json b/salt/elasticsearch/templates/component/elastic-agent/so-fleet_process_caseless-1.json new file mode 100644 index 000000000..37b6413e9 --- /dev/null +++ b/salt/elasticsearch/templates/component/elastic-agent/so-fleet_process_caseless-1.json @@ -0,0 +1,123 @@ +{ + "_meta": { + "managed_by": "security_onion", + "managed": true, + "description": "Adds .caseless for Lucene queries. Restates each field's package type and .text." + }, + "template": { + "mappings": { + "properties": { + "process": { + "properties": { + "executable": { + "type": "keyword", + "ignore_above": 1024, + "fields": { + "caseless": { + "type": "keyword", + "ignore_above": 1024, + "normalizer": "lowercase" + }, + "text": { + "type": "match_only_text" + } + } + }, + "name": { + "type": "keyword", + "ignore_above": 1024, + "fields": { + "caseless": { + "type": "keyword", + "ignore_above": 1024, + "normalizer": "lowercase" + }, + "text": { + "type": "match_only_text" + } + } + }, + "command_line": { + "type": "wildcard", + "ignore_above": 1024, + "fields": { + "caseless": { + "type": "keyword", + "ignore_above": 1024, + "normalizer": "lowercase" + }, + "text": { + "type": "match_only_text" + } + } + }, + "parent": { + "properties": { + "executable": { + "type": "keyword", + "ignore_above": 1024, + "fields": { + "caseless": { + "type": "keyword", + "ignore_above": 1024, + "normalizer": "lowercase" + }, + "text": { + "type": "match_only_text" + } + } + }, + "name": { + "type": "keyword", + "ignore_above": 1024, + "fields": { + "caseless": { + "type": "keyword", + "ignore_above": 1024, + "normalizer": "lowercase" + }, + "text": { + "type": "match_only_text" + } + } + }, + "command_line": { + "type": "wildcard", + "ignore_above": 1024, + "fields": { + "caseless": { + "type": "keyword", + "ignore_above": 1024, + "normalizer": "lowercase" + }, + "text": { + "type": "match_only_text" + } + } + } + } + } + } + }, + "file": { + "properties": { + "path": { + "type": "keyword", + "ignore_above": 1024, + "fields": { + "caseless": { + "type": "keyword", + "ignore_above": 1024, + "normalizer": "lowercase" + }, + "text": { + "type": "match_only_text" + } + } + } + } + } + } + } + } +} diff --git a/salt/elasticsearch/templates/component/elastic-agent/so-fleet_system.security_caseless-1.json b/salt/elasticsearch/templates/component/elastic-agent/so-fleet_system.security_caseless-1.json new file mode 100644 index 000000000..d320364fb --- /dev/null +++ b/salt/elasticsearch/templates/component/elastic-agent/so-fleet_system.security_caseless-1.json @@ -0,0 +1,80 @@ +{ + "_meta": { + "managed_by": "security_onion", + "managed": true, + "description": "Adds .caseless for Lucene queries. Keeps each field's existing keyword type." + }, + "template": { + "mappings": { + "properties": { + "process": { + "properties": { + "command_line": { + "type": "keyword", + "ignore_above": 1024, + "fields": { + "caseless": { + "type": "keyword", + "ignore_above": 1024, + "normalizer": "lowercase" + } + } + }, + "parent": { + "properties": { + "executable": { + "type": "keyword", + "ignore_above": 1024, + "fields": { + "caseless": { + "type": "keyword", + "ignore_above": 1024, + "normalizer": "lowercase" + } + } + }, + "name": { + "type": "keyword", + "ignore_above": 1024, + "fields": { + "caseless": { + "type": "keyword", + "ignore_above": 1024, + "normalizer": "lowercase" + } + } + }, + "command_line": { + "type": "keyword", + "ignore_above": 1024, + "fields": { + "caseless": { + "type": "keyword", + "ignore_above": 1024, + "normalizer": "lowercase" + } + } + } + } + } + } + }, + "file": { + "properties": { + "path": { + "type": "keyword", + "ignore_above": 1024, + "fields": { + "caseless": { + "type": "keyword", + "ignore_above": 1024, + "normalizer": "lowercase" + } + } + } + } + } + } + } + } +} diff --git a/salt/soc/files/soc/sigma_playbook_pipeline.yaml b/salt/soc/files/soc/sigma_playbook_pipeline.yaml index a779c4dec..6d428d2a8 100644 --- a/salt/soc/files/soc/sigma_playbook_pipeline.yaml +++ b/salt/soc/files/soc/sigma_playbook_pipeline.yaml @@ -2,7 +2,6 @@ name: Security Onion - Playbook Pipeline priority: 97 transformations: # Route to lowercase-normalized .caseless subfields for case-insensitive matching. - # file.path.caseless exists on Defend only (Sysmon file events lack it); # registry.path / dll.path / file.name have no .caseless on any source. - id: case_insensitive_string_fields type: field_name_mapping diff --git a/salt/soc/files/soc/sigma_so_pipeline.yaml b/salt/soc/files/soc/sigma_so_pipeline.yaml index ae867be49..23348860d 100644 --- a/salt/soc/files/soc/sigma_so_pipeline.yaml +++ b/salt/soc/files/soc/sigma_so_pipeline.yaml @@ -26,6 +26,16 @@ transformations: type: set_state key: keep val: "_id, _index, _source" + # Not every source maps .caseless; EQL/ES|QL already match case-insensitively. + - id: caseless_to_parent_fields + type: field_name_mapping + mapping: + process.executable.caseless: process.executable + process.name.caseless: process.name + process.parent.executable.caseless: process.parent.executable + process.parent.name.caseless: process.parent.name + target.process.executable.caseless: target.process.executable + target.process.name.caseless: target.process.name - id: baseline_field_name_mapping type: field_name_mapping mapping: From 22bda638470f6948bad242e0d7b89e815620d041 Mon Sep 17 00:00:00 2001 From: Corey Ogburn Date: Wed, 30 Sep 2026 16:35:10 -0600 Subject: [PATCH 11/15] Unified Automations Remove the template and mark automations as advanced, readonly, and stored in the DB. --- salt/soc/soc_soc.yaml | 17 ++++++++--------- 1 file changed, 8 insertions(+), 9 deletions(-) diff --git a/salt/soc/soc_soc.yaml b/salt/soc/soc_soc.yaml index 23e8bb628..8aabbfa36 100644 --- a/salt/soc/soc_soc.yaml +++ b/salt/soc/soc_soc.yaml @@ -862,15 +862,14 @@ soc: global: True forcedType: bool automations: - template: - description: Scheduled automations for the Onion AI assistant, managed from the Agent Studio. Each automation is stored under its own generated ID so its history and rollback are independent of every other automation. - global: True - advanced: False - readonlyUi: True - duplicates: True - forcedType: string - syntax: json - helpLink: onion-ai + description: Scheduled automations for the Onion AI assistant, managed from the Agent Studio. + global: True + advanced: True + readonlyUi: True + storage: db + forcedType: string + syntax: json + helpLink: onion-ai agents: description: Agent definitions for the Onion AI assistant, managed from the Agent Studio. An entry naming a system agent overrides only the fields an admin may change; everything else comes from the built-in definition. global: True From 99322cf26a731d182aeef0aa41c860d2b9912085 Mon Sep 17 00:00:00 2001 From: defensivedepth Date: Thu, 1 Oct 2026 15:24:52 -0400 Subject: [PATCH 12/15] Add additional mapping --- salt/soc/files/soc/sigma_playbook_pipeline.yaml | 1 + 1 file changed, 1 insertion(+) diff --git a/salt/soc/files/soc/sigma_playbook_pipeline.yaml b/salt/soc/files/soc/sigma_playbook_pipeline.yaml index 6d428d2a8..b31f8ba5d 100644 --- a/salt/soc/files/soc/sigma_playbook_pipeline.yaml +++ b/salt/soc/files/soc/sigma_playbook_pipeline.yaml @@ -8,6 +8,7 @@ transformations: mapping: process.executable: process.executable.caseless process.parent.executable: process.parent.executable.caseless + process.parent.name: process.parent.name.caseless process.command_line: process.command_line.caseless process.parent.command_line: process.parent.command_line.caseless file.path: file.path.caseless From 43475452b31d2b48a3e38f395191458421052376 Mon Sep 17 00:00:00 2001 From: defensivedepth Date: Thu, 1 Oct 2026 19:17:52 -0400 Subject: [PATCH 13/15] set module --- .../files/integrations/grid-nodes_general/import-evtx-logs.json | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/salt/elasticfleet/files/integrations/grid-nodes_general/import-evtx-logs.json b/salt/elasticfleet/files/integrations/grid-nodes_general/import-evtx-logs.json index 6f6bb8628..2a3f80431 100644 --- a/salt/elasticfleet/files/integrations/grid-nodes_general/import-evtx-logs.json +++ b/salt/elasticfleet/files/integrations/grid-nodes_general/import-evtx-logs.json @@ -29,7 +29,7 @@ "\\.gz$" ], "include_files": [], - "processors": "- dissect:\n tokenizer: \"/nsm/import/%{import.id}/evtx/%{import.file}\"\n field: \"log.file.path\"\n target_prefix: \"\"\n- decode_json_fields:\n fields: [\"message\"]\n target: \"\"\n- add_fields:\n target: event\n fields:\n dataset: windows.forwarded\n module: windows\n imported: true\n- add_fields:\n target: \"@metadata\"\n fields:\n pipeline: import.evtx\n- if:\n equals:\n winlog.channel: 'Security'\n then: \n - add_fields:\n target: event\n fields:\n dataset: system.security\n module: system\n- if:\n equals:\n winlog.channel: 'Windows PowerShell'\n then: \n - add_fields:\n target: event\n fields:\n dataset: windows.powershell\n module: windows\n- if:\n equals:\n winlog.channel: 'Microsoft-Windows-Sysmon/Operational'\n then: \n - add_fields:\n target: event\n fields:\n dataset: windows.sysmon_operational\n module: windows\n imported: true\n- if:\n equals:\n winlog.channel: 'Application'\n then: \n - add_fields:\n target: event\n fields:\n dataset: system.application\n- if:\n equals:\n winlog.channel: 'System'\n then: \n - add_fields:\n target: event\n fields:\n dataset: system.system\n \n- if:\n equals:\n winlog.channel: 'Microsoft-Windows-PowerShell/Operational'\n then: \n - add_fields:\n target: event\n fields:\n dataset: windows.powershell_operational\n module: windows\n- add_fields:\n target: data_stream\n fields:\n type: logs\n dataset: import", + "processors": "- dissect:\n tokenizer: \"/nsm/import/%{import.id}/evtx/%{import.file}\"\n field: \"log.file.path\"\n target_prefix: \"\"\n- decode_json_fields:\n fields: [\"message\"]\n target: \"\"\n- add_fields:\n target: event\n fields:\n dataset: windows.forwarded\n module: windows\n imported: true\n- add_fields:\n target: \"@metadata\"\n fields:\n pipeline: import.evtx\n- if:\n equals:\n winlog.channel: 'Security'\n then: \n - add_fields:\n target: event\n fields:\n dataset: system.security\n module: system\n- if:\n equals:\n winlog.channel: 'Windows PowerShell'\n then: \n - add_fields:\n target: event\n fields:\n dataset: windows.powershell\n module: windows\n- if:\n equals:\n winlog.channel: 'Microsoft-Windows-Sysmon/Operational'\n then: \n - add_fields:\n target: event\n fields:\n dataset: windows.sysmon_operational\n module: windows\n imported: true\n- if:\n equals:\n winlog.channel: 'Application'\n then: \n - add_fields:\n target: event\n fields:\n dataset: system.application\n module: system\n- if:\n equals:\n winlog.channel: 'System'\n then: \n - add_fields:\n target: event\n fields:\n dataset: system.system\n module: system\n \n- if:\n equals:\n winlog.channel: 'Microsoft-Windows-PowerShell/Operational'\n then: \n - add_fields:\n target: event\n fields:\n dataset: windows.powershell_operational\n module: windows\n- add_fields:\n target: data_stream\n fields:\n type: logs\n dataset: import", "tags": [ "import" ], From 4ce7a06abee374b4f8aec2f70421f02887c273c2 Mon Sep 17 00:00:00 2001 From: Josh Patterson Date: Tue, 29 Sep 2026 17:39:05 -0400 Subject: [PATCH 14/15] upgrade docker to 29.8.1 and containerd.io to 2.3.6 Latest upstream stable for el9. All four NVRs are already carried by the SO prod repo, so no repo change is needed -- a so-repo-sync refresh is enough. Tested on a managersearch and a heavynode (OL 9.8, 3.4.0), upgrading from 29.2.1/2.2.1 both by hand and through the state itself: - The 29.8.1 RPM ships a byte-identical docker.service, so the full ExecStart override in files/iptables-disabled.conf still resolves correctly and the hand-written DOCKER/DOCKER-ISOLATION/DOCKER-USER chains came back byte-identical on both nodes across upgrade, restart and reboot. - update_holds re-pinned the versionlock from the old NVRs to the new ones without intervention, so soup's path needs no change. - docker-py 7.1.0 still creates sobridge and soauth (forced by removing both); bridges keep their configured kernel names rather than br-. - 29.6 changed how dynamic port allocation treats net.ipv4.ip_local_reserved_ports; Strelka's 57314 is both published and reserved, and docker-proxy still owns it with no bind errors. - docker ps --format json gained a HealthStatus key. Additive, so so-status, so-log-check and so-docker-prune all still parse it. - containerd 2.3.6 ships the same config.toml, and it is %config(noreplace) and unmodified on disk, so no .rpmnew and disabled_plugins=["cri"] survives. - Zeek/Suricata/Strelka pipeline verified end-to-end with so-test: 111k packets replayed, 0 capture loss, file extraction and ES ingest all landed. The manifest unknown exclusion in so-log-check still fires on 29.8.1 -- it comes from a tag lookup during the registry-to-registry image copy, not from the 29.2.1 upgrade the old comment blamed -- so only the comment changes. --- salt/common/tools/sbin/so-log-check | 2 +- salt/docker/init.sls | 8 ++++---- 2 files changed, 5 insertions(+), 5 deletions(-) diff --git a/salt/common/tools/sbin/so-log-check b/salt/common/tools/sbin/so-log-check index 211a3f58d..852592a10 100755 --- a/salt/common/tools/sbin/so-log-check +++ b/salt/common/tools/sbin/so-log-check @@ -241,7 +241,7 @@ if [[ $EXCLUDE_KNOWN_ERRORS == 'Y' ]]; then EXCLUDED_ERRORS="$EXCLUDED_ERRORS|marked for removal" # docker container getting recycled EXCLUDED_ERRORS="$EXCLUDED_ERRORS|tcp 127.0.0.1:6791: bind: address already in use" # so-elastic-fleet agent restarting. Seen starting w/ 8.18.8 https://github.com/elastic/kibana/issues/201459 EXCLUDED_ERRORS="$EXCLUDED_ERRORS|TransformTask\] \[logs-.*user so_kibana lacks the required permissions" # Known issue with integrations starting transform jobs that are explicitly not allowed to start as a system user - EXCLUDED_ERRORS="$EXCLUDED_ERRORS|manifest unknown" # appears in so-dockerregistry log for so-tcpreplay following docker upgrade to 29.2.1-1 + EXCLUDED_ERRORS="$EXCLUDED_ERRORS|manifest unknown" # so-dockerregistry logs a tag lookup miss during image copy; not tied to one docker version EXCLUDED_ERRORS="$EXCLUDED_ERRORS|Could not index event to Elasticsearch.*\"version\" => \"9.0.8\"" # Expected during Elastic upgrade temporarily, as policies referencing older pipelines are updated fi diff --git a/salt/docker/init.sls b/salt/docker/init.sls index 9945a8096..4917cc163 100644 --- a/salt/docker/init.sls +++ b/salt/docker/init.sls @@ -18,10 +18,10 @@ dockergroup: dockerheldpackages: pkg.installed: - pkgs: - - containerd.io: 2.2.1-1.el9 - - docker-ce: 3:29.2.1-1.el9 - - docker-ce-cli: 1:29.2.1-1.el9 - - docker-ce-rootless-extras: 29.2.1-1.el9 + - containerd.io: 2.3.6-1.el9 + - docker-ce: 3:29.8.1-1.el9 + - docker-ce-cli: 1:29.8.1-1.el9 + - docker-ce-rootless-extras: 29.8.1-1.el9 - hold: True - update_holds: True From 31c5190a1fab2a41229fd8e8709e75f47c837789 Mon Sep 17 00:00:00 2001 From: Josh Patterson Date: Wed, 30 Sep 2026 09:58:31 -0400 Subject: [PATCH 15/15] add missing comma --- salt/repo/client/map.jinja | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/salt/repo/client/map.jinja b/salt/repo/client/map.jinja index 21f52a5e7..3a44d8431 100644 --- a/salt/repo/client/map.jinja +++ b/salt/repo/client/map.jinja @@ -9,7 +9,7 @@ 'epel-testing.repo', 'saltstack.repo', 'salt-latest.repo', - 'wazuh.repo' + 'wazuh.repo', 'Rocky-Base.repo', 'Rocky-CR.repo', 'Rocky-Debuginfo.repo',