diff --git a/.github/workflows/pythontest.yml b/.github/workflows/pythontest.yml index dc95b01c3..5d474f5df 100644 --- a/.github/workflows/pythontest.yml +++ b/.github/workflows/pythontest.yml @@ -6,6 +6,9 @@ on: - "salt/sensoroni/files/analyzers/**" - "salt/manager/tools/sbin/**" - "salt/_beacons/**" + - "salt/telegraf/tools/sbin_jinja/**" + - "salt/telegraf/defaults.yaml" + - "salt/telegraf/soc_telegraf.yaml" jobs: build: @@ -34,3 +37,25 @@ jobs: - name: Test with pytest run: | PYTHONPATH=${{ matrix.python-code-path }} pytest ${{ matrix.python-code-path }} --cov=${{ matrix.python-code-path }} --doctest-modules --cov-report=term --cov-fail-under=100 --cov-config=pytest.ini + + telegraf-collector: + # so-container-stats is a jinja template rather than an importable module, so it gets its + # own job: the test renders it the way salt does, then drives it with a faked docker engine + # and cgroup tree. No container runtime is needed. + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v3 + - name: Set up Python + uses: actions/setup-python@v3 + with: + python-version: "3.14" + - name: Install dependencies + run: | + python -m pip install --upgrade pip + python -m pip install flake8 pytest jinja2 pyyaml + - name: Lint with flake8 + run: | + flake8 salt/telegraf/tools/sbin_jinja/so-container-stats_test.py --config=pytest.ini + - name: Test with pytest + run: | + pytest salt/telegraf/tools/sbin_jinja/so-container-stats_test.py -v diff --git a/salt/common/init.sls b/salt/common/init.sls index 34b84f09d..ecb5fdccc 100644 --- a/salt/common/init.sls +++ b/salt/common/init.sls @@ -217,9 +217,11 @@ sostatus_log: - replace: False # Install sostatus check cron. This is used to populate Grid. +# telegraf reads status.log on the same minute boundary this runs, so write aside and rename +# rather than truncating the file it is reading so-status_check_cron: cron.present: - - name: '/usr/sbin/so-status -j > /opt/so/log/sostatus/status.log 2>&1' + - name: '/usr/sbin/so-status -j > /opt/so/log/sostatus/status.log.tmp 2>&1; mv -f /opt/so/log/sostatus/status.log.tmp /opt/so/log/sostatus/status.log' - identifier: so-status_check_cron - user: root - minute: '*/1' diff --git a/salt/common/tools/sbin/so-common-status-check b/salt/common/tools/sbin/so-common-status-check index cbef7309e..1f387570d 100644 --- a/salt/common/tools/sbin/so-common-status-check +++ b/salt/common/tools/sbin/so-common-status-check @@ -9,6 +9,7 @@ import sys import subprocess import os import json +import tempfile sys.path.append('/opt/saltstack/salt/lib/python3.10/site-packages/') import salt.config @@ -17,6 +18,21 @@ import salt.loader __opts__ = salt.config.minion_config('/etc/salt/minion') __grains__ = salt.loader.grains(__opts__) +def write_atomic(path, value): + # telegraf reads these files on its own schedule; replacing them by rename means it never + # reads a truncated file and reports an empty value as if it were real + directory = os.path.dirname(path) + handle, temp = tempfile.mkstemp(dir=directory) + try: + with os.fdopen(handle, 'w') as f: + f.write(str(value)) + os.chmod(temp, 0o644) + os.replace(temp, path) + except Exception: + os.path.exists(temp) and os.unlink(temp) + raise + + def check_needs_restarted(): osfam = __grains__['os_family'] val = '0' @@ -34,8 +50,7 @@ def check_needs_restarted(): else: fail("Unsupported OS") - with open(outfile, 'w') as f: - f.write(val) + write_atomic(outfile, val) def check_for_fps(): feat = 'fps' @@ -56,8 +71,7 @@ def check_for_fps(): # Unknown, so assume 0 fps = 0 - with open('/opt/so/log/sostatus/fps_enabled', 'w') as f: - f.write(str(fps)) + write_atomic('/opt/so/log/sostatus/fps_enabled', fps) def check_for_lks(): feat = 'Lks' @@ -80,8 +94,7 @@ def check_for_lks(): lks = 1 if lks: break - with open('/opt/so/log/sostatus/lks_enabled', 'w') as f: - f.write(str(lks)) + write_atomic('/opt/so/log/sostatus/lks_enabled', lks) def fail(msg): print(msg, file=sys.stderr) diff --git a/salt/common/tools/sbin_jinja/so-raid-status b/salt/common/tools/sbin_jinja/so-raid-status index ca3c34608..fefb1806f 100755 --- a/salt/common/tools/sbin_jinja/so-raid-status +++ b/salt/common/tools/sbin_jinja/so-raid-status @@ -125,4 +125,6 @@ else RAIDSTATUS=1 fi -echo "nsmraid=$RAIDSTATUS" > /opt/so/log/raid/status.log +# telegraf reads this file; write aside and rename so it never sees a half-written file +echo "nsmraid=$RAIDSTATUS" > /opt/so/log/raid/status.log.tmp +mv -f /opt/so/log/raid/status.log.tmp /opt/so/log/raid/status.log diff --git a/salt/influxdb/enabled.sls b/salt/influxdb/enabled.sls index ed3ac0de3..617a56e1e 100644 --- a/salt/influxdb/enabled.sls +++ b/salt/influxdb/enabled.sls @@ -94,9 +94,11 @@ metrics_link_file: - docker_container: so-influxdb # Install cron job to determine size of influxdb for telegraf +# telegraf reads this while the cron rewrites it, so write aside and rename rather than +# truncating in place. tgraflogdir recurses ownership, so the temp file is chowned to match get_influxdb_size: cron.present: - - name: 'du -s -k /nsm/influxdb | cut -f1 > /opt/so/log/telegraf/influxdb_size.log 2>&1' + - name: 'du -s -k /nsm/influxdb | cut -f1 > /opt/so/log/telegraf/influxdb_size.log.tmp 2>&1; chown 939:939 /opt/so/log/telegraf/influxdb_size.log.tmp; mv -f /opt/so/log/telegraf/influxdb_size.log.tmp /opt/so/log/telegraf/influxdb_size.log' - identifier: get_influxdb_size - user: root - minute: '*/1' diff --git a/salt/manager/init.sls b/salt/manager/init.sls index 87669fad4..cb0395d3c 100644 --- a/salt/manager/init.sls +++ b/salt/manager/init.sls @@ -166,7 +166,7 @@ so-repo-sync: so_fleetagent_status: cron.present: - - name: /usr/sbin/so-elasticagent-status > /opt/so/log/agents/agentstatus.log 2>&1 + - name: '/usr/sbin/so-elasticagent-status > /opt/so/log/agents/agentstatus.log.tmp 2>&1; mv -f /opt/so/log/agents/agentstatus.log.tmp /opt/so/log/agents/agentstatus.log' - identifier: so_fleetagent_status - user: root - minute: '*/5' diff --git a/salt/telegraf/config.sls b/salt/telegraf/config.sls index 18ac51ddd..64ca7070e 100644 --- a/salt/telegraf/config.sls +++ b/salt/telegraf/config.sls @@ -36,7 +36,7 @@ tgraf_sync_script_{{script}}: - name: /opt/so/conf/telegraf/scripts/{{script}} - user: root - group: 939 - - mode: 770 + - mode: 750 - template: jinja - source: salt://telegraf/scripts/{{script}} - defaults: @@ -49,7 +49,7 @@ tgraf_sync_script_esindexsize.sh: - name: /opt/so/conf/telegraf/scripts/esindexsize.sh - user: root - group: 939 - - mode: 770 + - mode: 750 - source: salt://telegraf/scripts/esindexsize.sh {# Copy conf/elasticsearch/curl.config for telegraf to use with esindexsize.sh #} tgraf_sync_escurl_conf: @@ -61,6 +61,80 @@ tgraf_sync_escurl_conf: - source: salt://elasticsearch/curl.config {% endif %} +# so-container-stats runs on the host as somon, a docker group member, so the container does +# not need the docker socket +somongroup: + group.present: + - name: somon + - gid: 961 + +# cron chdirs to $HOME before running a job, so home must exist +somon: + user.present: + - uid: 961 + - gid: 961 + - home: /opt/so/log/somon + - createhome: False + - shell: /sbin/nologin + - groups: + - docker + # renumbering an existing somon is a no-op on a fresh host and lets a host created + # before the id changed converge instead of failing the whole telegraf state + - allow_uid_change: True + - allow_gid_change: True + - require: + - group: somongroup + +somonlogdir: + file.directory: + - name: /opt/so/log/somon + - user: 961 + - group: 961 + - mode: 755 + # the lock file is not otherwise managed; recurse so a renumber rechowns it too + - recurse: + - user + - group + - require: + - user: somon + +containers_log: + file.managed: + - name: /opt/so/log/somon/containers.log + - user: 961 + - group: 961 + - mode: 644 + - replace: False + - require: + - file: somonlogdir + +# telegraf reads on the same minute boundary the collector runs, and docker stats takes +# seconds, so write aside and rename rather than truncating the file telegraf is reading. +# ; not && so a failed run replaces the file instead of leaving stale metrics behind. +# flock -n keeps a run that outlives its minute from racing the next one over the same tmp +# file; the skipped run leaves a stale containers.log, which containers.sh discards by age +so-container-stats_cron: + cron.present: + - name: "flock -n /opt/so/log/somon/containers.lock -c '/usr/sbin/so-container-stats > /opt/so/log/somon/containers.log.tmp 2>&1; mv -f /opt/so/log/somon/containers.log.tmp /opt/so/log/somon/containers.log'" + - identifier: so-container-stats_cron + - user: somon + - minute: '*/1' + - hour: '*' + - daymonth: '*' + - month: '*' + - dayweek: '*' + - require: + - user: somon + +# salt.lasthighstate touches this at order 9001, after the container starts; pre-create it so +# docker does not create a directory at the bind mount source +lasthighstate_placeholder: + file.managed: + - name: /opt/so/log/salt/lasthighstate + - mode: 644 + - replace: False + - makedirs: True + telegraf_sbin: file.recurse: - name: /usr/sbin @@ -69,14 +143,20 @@ telegraf_sbin: - group: root - file_mode: 755 -#telegraf_sbin_jinja: -# file.recurse: -# - name: /usr/sbin -# - source: salt://telegraf/tools/sbin_jinja -# - user: 939 -# - group: 939 -# - file_mode: 755 -# - template: jinja +# so-container-stats needs the per-stat toggles, so it renders instead of copying +tgraf_sbin_jinja: + file.recurse: + - name: /usr/sbin + - source: salt://telegraf/tools/sbin_jinja + - user: root + - group: root + - file_mode: 755 + # the unit test lives beside the script; it must not ship or be rendered as jinja + - exclude_pat: + - "*_test.py" + - template: jinja + - defaults: + CONTAINER_STATS: {{ TELEGRAFMERGED.container_stats }} tgrafconf: file.managed: diff --git a/salt/telegraf/defaults.yaml b/salt/telegraf/defaults.yaml index 24f58b157..0ff3539fd 100644 --- a/salt/telegraf/defaults.yaml +++ b/salt/telegraf/defaults.yaml @@ -10,6 +10,69 @@ telegraf: flush_jitter: '0s' debug: false quiet: false + container_stats: + tags: + identity: False + engine: + n_containers: False + n_containers_running: False + n_containers_stopped: False + n_containers_paused: False + n_images: False + n_cpus: False + n_goroutines: False + n_used_file_descriptors: False + n_listener_events: False + memory_total: False + cpu: + usage_percent: True + usage_total: False + usage_in_usermode: False + usage_in_kernelmode: False + usage_system: False + throttling_periods: False + throttling_throttled_periods: False + throttling_throttled_time: False + container_id: False + mem: + usage_percent: True + usage: False + limit: False + max_usage: False + active_anon: False + active_file: False + inactive_anon: False + inactive_file: False + unevictable: False + pgfault: False + pgmajfault: False + container_id: False + net: + rx_bytes: True + rx_packets: False + rx_errors: False + rx_dropped: False + tx_bytes: False + tx_packets: False + tx_errors: False + tx_dropped: False + container_id: False + blkio: + io_service_bytes_recursive_read: False + io_service_bytes_recursive_write: False + container_id: False + status: + uptime_ns: True + oomkilled: True + pid: False + exitcode: False + restart_count: False + started_at: False + finished_at: False + container_id: False + health: + health_status: False + failing_streak: False scripts: eval: - agentstatus.sh @@ -19,6 +82,7 @@ telegraf: - oldpcap.sh - os.sh - raid.sh + - containers.sh - sostatus.sh - suriloss.sh - surirules.sh @@ -34,6 +98,7 @@ telegraf: - os.sh - raid.sh - redis.sh + - containers.sh - sostatus.sh - suriloss.sh - surirules.sh @@ -47,6 +112,7 @@ telegraf: - os.sh - raid.sh - redis.sh + - containers.sh - sostatus.sh - features.sh managerhype: @@ -56,6 +122,7 @@ telegraf: - os.sh - raid.sh - redis.sh + - containers.sh - sostatus.sh - features.sh managersearch: @@ -66,12 +133,14 @@ telegraf: - os.sh - raid.sh - redis.sh + - containers.sh - sostatus.sh - features.sh import: - influxdbsize.sh - lasthighstate.sh - os.sh + - containers.sh - sostatus.sh sensor: - checkfiles.sh @@ -79,6 +148,7 @@ telegraf: - oldpcap.sh - os.sh - raid.sh + - containers.sh - sostatus.sh - suriloss.sh - surirules.sh @@ -93,6 +163,7 @@ telegraf: - os.sh - raid.sh - redis.sh + - containers.sh - sostatus.sh - suriloss.sh - surirules.sh @@ -101,12 +172,14 @@ telegraf: idh: - lasthighstate.sh - os.sh + - containers.sh - sostatus.sh searchnode: - eps.sh - lasthighstate.sh - os.sh - raid.sh + - containers.sh - sostatus.sh - features.sh receiver: @@ -115,16 +188,20 @@ telegraf: - os.sh - raid.sh - redis.sh + - containers.sh - sostatus.sh fleet: - lasthighstate.sh - os.sh + - containers.sh - sostatus.sh hypervisor: - lasthighstate.sh - os.sh + - containers.sh - sostatus.sh desktop: - lasthighstate.sh - os.sh + - containers.sh - sostatus.sh diff --git a/salt/telegraf/disabled.sls b/salt/telegraf/disabled.sls index 004d3d928..371708e04 100644 --- a/salt/telegraf/disabled.sls +++ b/salt/telegraf/disabled.sls @@ -13,6 +13,11 @@ so-telegraf: docker_container.absent: - force: True +so-container-stats_cron: + cron.absent: + - identifier: so-container-stats_cron + - user: somon + so-telegraf_so-status.disabled: file.comment: - name: /opt/so/conf/so-status/so-status.conf diff --git a/salt/telegraf/enabled.sls b/salt/telegraf/enabled.sls index 6a063e08b..addc55128 100644 --- a/salt/telegraf/enabled.sls +++ b/salt/telegraf/enabled.sls @@ -19,8 +19,7 @@ so-telegraf: docker_container.running: - image: {{ GLOBALS.registry_host }}:5000/{{ GLOBALS.image_repo }}/so-telegraf:{{ GLOBALS.so_version }} - restart_policy: unless-stopped - - user: 939 - - group_add: 939,920 + - user: 939:939 - environment: - HOST_ETC=/host/etc - HOST_SYS=/host/sys @@ -38,7 +37,6 @@ so-telegraf: - /opt/so/conf/telegraf/etc/telegraf.conf:/etc/telegraf/telegraf.conf:ro - /opt/so/conf/telegraf/node_config.json:/etc/telegraf/node_config.json:ro - /var/run/utmp:/var/run/utmp:ro - - /var/run/docker.sock:/var/run/docker.sock:ro - /:/host:ro - /sys:/host/sys:ro - /proc:/host/proc:ro @@ -51,7 +49,8 @@ so-telegraf: - /opt/so/log/suricata:/var/log/suricata:ro - /opt/so/log/raid:/var/log/raid:ro - /opt/so/log/sostatus:/var/log/sostatus:ro - - /opt/so/log/salt:/var/log/salt:ro + - /opt/so/log/somon:/var/log/somon:ro + - /opt/so/log/salt/lasthighstate:/var/log/salt/lasthighstate:ro - /opt/so/log/agents:/var/log/agents:ro {% if GLOBALS.is_manager or GLOBALS.role == 'so-heavynode' %} - /opt/so/conf/telegraf/etc/escurl.config:/etc/telegraf/elasticsearch.config:ro @@ -74,6 +73,8 @@ so-telegraf: {% endfor %} {% endif %} - watch: + - file: tgraf_sbin_jinja + - file: lasthighstate_placeholder - file: trusttheca - x509: telegraf_crt - x509: telegraf_key @@ -83,6 +84,8 @@ so-telegraf: - file: tgraf_sync_script_{{script}} {% endfor %} - require: + - file: lasthighstate_placeholder + - file: somonlogdir - file: trusttheca - x509: telegraf_crt - x509: telegraf_key diff --git a/salt/telegraf/etc/telegraf.conf b/salt/telegraf/etc/telegraf.conf index ae1b83084..ab51ac2a5 100644 --- a/salt/telegraf/etc/telegraf.conf +++ b/salt/telegraf/etc/telegraf.conf @@ -228,13 +228,6 @@ # ## bond interfaces. # # bond_interfaces = ["bond0"] -# # Read metrics about docker containers - [[inputs.docker]] -# ## Docker Endpoint -# ## To use TCP, set endpoint = "tcp://[ip]:[port]" -# ## To use environment variables (ie, docker-machine), set endpoint = "ENV" - endpoint = "unix:///var/run/docker.sock" -# # # Read stats from one or more Elasticsearch servers or clusters {%- if GLOBALS.is_manager or GLOBALS.role == 'so-heavynode' %} @@ -342,6 +335,17 @@ interval = "60s" {%- endif %} +{%- if 'containers.sh' in TELEGRAFMERGED.scripts[GLOBALS.role.split('-')[1]] %} +{%- do TELEGRAFMERGED.scripts[GLOBALS.role.split('-')[1]].remove('containers.sh') %} +[[inputs.exec]] + commands = [ + ["/scripts/containers.sh"] + ] + data_format = "influx" + timeout = "15s" + interval = "60s" +{%- endif %} + {%- if TELEGRAFMERGED.scripts[GLOBALS.role.split('-')[1]] | length > 0 %} [[inputs.exec]] commands = [ diff --git a/salt/telegraf/scripts/containers.sh b/salt/telegraf/scripts/containers.sh new file mode 100644 index 000000000..836d4a999 --- /dev/null +++ b/salt/telegraf/scripts/containers.sh @@ -0,0 +1,29 @@ +#!/bin/bash +# +# Copyright Security Onion Solutions LLC and/or licensed to Security Onion Solutions LLC under one +# or more contributor license agreements. Licensed under the Elastic License 2.0 as shown at +# https://securityonion.net/license; you may not use this file except in compliance with the +# Elastic License 2.0. + + + +# if this script isn't already running +if [[ ! "`pidof -x $(basename $0) -o %PPID`" ]]; then + + CONTAINERSLOG=/var/log/somon/containers.log + # the collector rewrites this every minute; report nothing rather than repeating a stale + # file as if it were current, in case a run was skipped or the collector is wedged + MAXAGE=150 + + if [ -r "$CONTAINERSLOG" ]; then + AGE=$(( $(date +%s) - $(stat -c %Y "$CONTAINERSLOG") )) + if [ "$AGE" -le "$MAXAGE" ]; then + cat $CONTAINERSLOG + fi + fi + + exit 0 + +fi + +exit 0 diff --git a/salt/telegraf/scripts/lasthighstate.sh b/salt/telegraf/scripts/lasthighstate.sh index 85f259bb8..a72c00ca9 100644 --- a/salt/telegraf/scripts/lasthighstate.sh +++ b/salt/telegraf/scripts/lasthighstate.sh @@ -8,10 +8,12 @@ # if this script isn't already running if [[ ! "`pidof -x $(basename $0) -o %PPID`" ]]; then - LAST_HIGHSTATE_END=$([ -e "/var/log/salt/lasthighstate" ] && date -r /var/log/salt/lasthighstate +%s || echo 0) - NOW=$(date +%s) - HIGHSTATE_AGE_SECONDS=$((NOW-LAST_HIGHSTATE_END)) - echo "salt highstate_age_seconds=$HIGHSTATE_AGE_SECONDS" + if [ -r "/var/log/salt/lasthighstate" ]; then + LAST_HIGHSTATE_END=$(date -r /var/log/salt/lasthighstate +%s) + NOW=$(date +%s) + HIGHSTATE_AGE_SECONDS=$((NOW-LAST_HIGHSTATE_END)) + echo "salt highstate_age_seconds=$HIGHSTATE_AGE_SECONDS" + fi fi diff --git a/salt/telegraf/soc_telegraf.yaml b/salt/telegraf/soc_telegraf.yaml index 0decdd06a..d92a89f2f 100644 --- a/salt/telegraf/soc_telegraf.yaml +++ b/salt/telegraf/soc_telegraf.yaml @@ -53,6 +53,319 @@ telegraf: forcedType: bool advanced: True helpLink: influxdb + container_stats: + tags: + identity: + description: Adds the container_image, container_version, engine_host and server_version tags to every container measurement, as the retired inputs.docker plugin did. Increases series cardinality. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + engine: + n_containers: + description: Total number of containers known to the Docker engine, running or not. Part of the engine-level docker measurement. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + n_containers_running: + description: Number of containers currently running. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + n_containers_stopped: + description: Number of containers currently stopped. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + n_containers_paused: + description: Number of containers currently paused. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + n_images: + description: Number of container images held by the Docker engine. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + n_cpus: + description: Number of CPUs the Docker engine reports for this host. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + n_goroutines: + description: Number of goroutines inside the Docker daemon. Diagnostic detail for the daemon itself. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + n_used_file_descriptors: + description: Number of file descriptors held open by the Docker daemon. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + n_listener_events: + description: Number of event listeners subscribed to the Docker daemon. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + memory_total: + description: Total physical memory the Docker engine reports for this host, in bytes. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + cpu: + usage_percent: + description: Percentage of host CPU consumed by the container. Required by the Container CPU Usage chart on the Security Onion Performance dashboard. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + usage_total: + description: Cumulative CPU time consumed by the container, in nanoseconds. Read from the cgroup. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + usage_in_usermode: + description: Cumulative CPU time consumed in user mode, in nanoseconds. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + usage_in_kernelmode: + description: Cumulative CPU time consumed in kernel mode, in nanoseconds. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + usage_system: + description: Cumulative host-wide CPU time, in nanoseconds. Used as the denominator when calculating container CPU percentage. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + throttling_periods: + description: Number of CPU enforcement periods the container has seen. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + throttling_throttled_periods: + description: Number of periods in which the container was throttled against its CPU limit. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + throttling_throttled_time: + description: Total time the container spent throttled, in nanoseconds. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + container_id: &containerid + description: Full 64 character container ID, emitted as a field on this measurement. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + mem: + usage_percent: + description: Percentage of its memory limit the container is using. Required by the Container Memory Usage chart on the Security Onion Performance dashboard. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + usage: + description: Container memory usage in bytes, excluding reclaimable page cache. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + limit: + description: Memory limit for the container in bytes. Reports total host memory when the container is unlimited. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + max_usage: + description: Peak memory usage for the container in bytes, read from the cgroup peak counter. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + active_anon: + description: Anonymous memory on the active LRU list, in bytes. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + active_file: + description: Page cache on the active LRU list, in bytes. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + inactive_anon: + description: Anonymous memory on the inactive LRU list, in bytes. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + inactive_file: + description: Page cache on the inactive LRU list, in bytes. This is the reclaimable cache subtracted from usage. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + unevictable: + description: Memory that cannot be reclaimed, in bytes. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + pgfault: + description: Cumulative number of page faults taken by the container. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + pgmajfault: + description: Cumulative number of major page faults, those requiring disk access. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + container_id: *containerid + net: + rx_bytes: + description: Bytes received by the container across all interfaces except loopback. Required by the Container Traffic - Inbound chart on the Security Onion Performance dashboard. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + rx_packets: + description: Packets received by the container. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + rx_errors: + description: Receive errors counted on the container interfaces. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + rx_dropped: + description: Received packets dropped by the container interfaces. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + tx_bytes: + description: Bytes transmitted by the container. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + tx_packets: + description: Packets transmitted by the container. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + tx_errors: + description: Transmit errors counted on the container interfaces. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + tx_dropped: + description: Transmitted packets dropped by the container interfaces. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + container_id: *containerid + blkio: + io_service_bytes_recursive_read: + description: Cumulative bytes read from block devices by the container. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + io_service_bytes_recursive_write: + description: Cumulative bytes written to block devices by the container. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + container_id: *containerid + status: + uptime_ns: + description: How long the container has been running, in nanoseconds. Required by the Container Uptime chart on the Security Onion Performance dashboard. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + oomkilled: + description: Whether the container was killed by the kernel out-of-memory handler. Required by the Most Recent Container Events table on the Security Onion Performance dashboard. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + pid: + description: Host process ID of the container main process. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + exitcode: + description: Exit code of the container main process. Meaningful once the container has stopped. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + restart_count: + description: Number of times the Docker engine has restarted this container. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + started_at: + description: Time the container last started, as a Unix timestamp in nanoseconds. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + finished_at: + description: Time the container last exited, as a Unix timestamp in nanoseconds. Absent for a container that has never exited. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + container_id: *containerid + health: + health_status: + description: Result of the container healthcheck, such as healthy, unhealthy or starting. Only emitted for containers that define a healthcheck. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb + failing_streak: + description: Number of consecutive failed healthchecks. Only emitted for containers that define a healthcheck. Defaults to off. + forcedType: bool + global: True + advanced: True + helpLink: influxdb scripts: eval: &telegrafscripts description: List of input.exec scripts to run for this node type. The script must be present in salt/telegraf/scripts. diff --git a/salt/telegraf/tools/sbin_jinja/so-container-stats b/salt/telegraf/tools/sbin_jinja/so-container-stats new file mode 100644 index 000000000..b11b0bcb4 --- /dev/null +++ b/salt/telegraf/tools/sbin_jinja/so-container-stats @@ -0,0 +1,398 @@ +#!/usr/bin/env python3 + +# Copyright Security Onion Solutions LLC and/or licensed to Security Onion Solutions LLC under one +# or more contributor license agreements. Licensed under the Elastic License 2.0 as shown at +# https://securityonion.net/license; you may not use this file except in compliance with the +# Elastic License 2.0. + +# Runs from cron as somon, a docker group member, and emits influx line protocol for telegraf to +# read. This exists so so-telegraf does not need the docker socket; the container reads the output +# file instead. +# Measurement/tag/field names match telegraf's inputs.docker plugin because the InfluxDB +# dashboards query them directly. Which fields are emitted is set per stat in SOC; see +# telegraf.container_stats in defaults.yaml. + +import json + +SETTINGS = json.loads('''{{ CONTAINER_STATS | tojson }}''') +{% raw %} +import os +import re +import subprocess +import sys +from datetime import datetime, timezone + +CGROUP_ROOT = '/sys/fs/cgroup' +NANOSEC = 10**9 +# cgroup and docker report cpu time in microseconds; inputs.docker published nanoseconds +USEC_TO_NSEC = 1000 + + +def want(group, field): + return bool(SETTINGS.get(group, {}).get(field)) + + +def wants_any(group, fields): + return any(want(group, field) for field in fields) + + +def docker(args): + proc = subprocess.run(['docker'] + args, stdout=subprocess.PIPE, stderr=subprocess.DEVNULL, encoding='utf-8') + if proc.returncode != 0: + print('Container system error; unable to query docker', file=sys.stderr) + sys.exit(1) + return proc.stdout + + +def escape_tag(value): + return value.replace(',', '\\,').replace(' ', '\\ ').replace('=', '\\=') + + +def quote(value): + return '"{}"'.format(str(value).replace('\\', '\\\\').replace('"', '\\"')) + + +def integer(value): + return '{}i'.format(int(value)) + + +def unsigned(value): + # inputs.docker published the cgroup and network counters as uint64; influx stores u and i + # as different field types, so matching it keeps historical series readable + return '{}u'.format(max(int(value), 0)) + + +def read_text(path): + try: + with open(path) as handle: + return handle.read() + except OSError: + return '' + + +def read_pairs(path): + # cgroup files such as memory.stat and cpu.stat are "key value" per line + values = {} + for line in read_text(path).splitlines(): + parts = line.split() + if len(parts) == 2: + try: + values[parts[0]] = int(parts[1]) + except ValueError: + pass + return values + + +def read_value(path): + raw = read_text(path).strip() + try: + return int(raw) + except ValueError: + return None + + +def cgroup_path(pid): + # "0::/system.slice/docker-.scope" on cgroup v2 + for line in read_text('/proc/{}/cgroup'.format(pid)).splitlines(): + parts = line.split(':', 2) + if len(parts) == 3 and parts[1] == '': + return CGROUP_ROOT + parts[2] + return None + + +def host_memory_total(): + match = re.search(r'MemTotal:\s+(\d+) kB', read_text('/proc/meminfo')) + return int(match.group(1)) * 1024 if match else 0 + + +def host_cpu_nanoseconds(): + # inputs.docker's usage_system is the host-wide cpu time the daemon reads from /proc/stat + for line in read_text('/proc/stat').splitlines(): + if line.startswith('cpu '): + ticks = sum(int(value) for value in line.split()[1:]) + return int(ticks * NANOSEC / os.sysconf('SC_CLK_TCK')) + return 0 + + +def parse_image(image): + # mirrors telegraf's internal/docker ParseImage so the tags match what inputs.docker emitted + domain = '' + remainder = image + if '/' in image: + head, _, tail = image.partition('/') + if '.' in head or ':' in head or head == 'localhost': + domain, remainder = head + '/', tail + if ':' in remainder: + name, _, version = remainder.rpartition(':') + return domain + name, version + return domain + remainder, 'unknown' + + +def to_nanoseconds(stamp): + # docker emits 9 fractional digits; fromisoformat takes at most 6 before python 3.11 + stamp = stamp.rstrip('Z') + if '.' in stamp: + whole, _, frac = stamp.partition('.') + stamp = whole + '.' + frac[:6] + try: + parsed = datetime.fromisoformat(stamp).replace(tzinfo=timezone.utc) + except ValueError: + return None + if parsed.year <= 1: + return None + return int(parsed.timestamp() * NANOSEC) + + +def percent(value): + try: + return float(value.strip().rstrip('%')) + except ValueError: + return 0.0 + + +def net_counters(pid): + # summed across interfaces except loopback, matching the daemon's per-container totals + rows = read_text('/proc/{}/net/dev'.format(pid)).splitlines()[2:] + if not rows: + return None + names = ['rx_bytes', 'rx_packets', 'rx_errors', 'rx_dropped'] + totals = dict.fromkeys(names + ['tx_bytes', 'tx_packets', 'tx_errors', 'tx_dropped'], 0) + found = False + for row in rows: + iface, _, rest = row.partition(':') + if iface.strip() == 'lo': + continue + columns = rest.split() + if len(columns) < 12: + continue + found = True + for index, name in enumerate(names): + totals[name] += int(columns[index]) + for index, name in enumerate(['tx_bytes', 'tx_packets', 'tx_errors', 'tx_dropped']): + totals[name] += int(columns[8 + index]) + return totals if found else None + + +def blkio_counters(path): + # inputs.docker emitted the device=total row unconditionally, so a container that has done + # no block io reports zeros rather than dropping out of the measurement entirely + totals = {'io_service_bytes_recursive_read': 0, 'io_service_bytes_recursive_write': 0} + rows = read_text(path + '/io.stat').splitlines() + for row in rows: + for token in row.split()[1:]: + key, _, value = token.partition('=') + try: + if key == 'rbytes': + totals['io_service_bytes_recursive_read'] += int(value) + elif key == 'wbytes': + totals['io_service_bytes_recursive_write'] += int(value) + except ValueError: + pass + return totals + + +def collect_stats(): + stats = {} + for line in docker(['stats', '--no-stream', '--format', '{{json .}}']).splitlines(): + line = line.strip() + if not line: + continue + try: + entry = json.loads(line) + except json.JSONDecodeError: + continue + stats[entry.get('Name', '')] = entry + return stats + + +def collect_inspect(): + ids = docker(['ps', '-aq']).split() + if not ids: + return [] + try: + return json.loads(docker(['inspect'] + ids)) + except json.JSONDecodeError: + return [] + + +def collect_info(): + try: + return json.loads(docker(['info', '--format', '{{json .}}'])) + except json.JSONDecodeError: + return {} + + +class Emitter: + def __init__(self): + self.lines = [] + + def add(self, measurement, tags, fields): + if not fields: + return + tagset = ','.join('{}={}'.format(key, escape_tag(value)) for key, value in tags) + body = ','.join('{}={}'.format(key, fields[key]) for key in sorted(fields)) + self.lines.append('{},{} {}'.format(measurement, tagset, body)) + + +def main(): + cpu_extra = ['usage_total', 'usage_in_usermode', 'usage_in_kernelmode', + 'throttling_periods', 'throttling_throttled_periods', 'throttling_throttled_time'] + mem_extra = ['usage', 'limit', 'max_usage', 'active_anon', 'active_file', 'inactive_anon', + 'inactive_file', 'unevictable', 'pgfault', 'pgmajfault'] + net_fields = ['rx_bytes', 'rx_packets', 'rx_errors', 'rx_dropped', + 'tx_bytes', 'tx_packets', 'tx_errors', 'tx_dropped'] + engine_fields = ['n_containers', 'n_containers_running', 'n_containers_stopped', + 'n_containers_paused', 'n_images', 'n_cpus', 'n_goroutines', + 'n_used_file_descriptors', 'n_listener_events'] + + identity = want('tags', 'identity') + need_stats = want('cpu', 'usage_percent') + need_cgroup_cpu = wants_any('cpu', cpu_extra) + need_cgroup_mem = wants_any('mem', mem_extra + ['usage_percent']) + need_blkio = wants_any('blkio', ['io_service_bytes_recursive_read', + 'io_service_bytes_recursive_write', 'container_id']) + need_net = wants_any('net', net_fields + ['container_id']) + need_info = identity or wants_any('engine', engine_fields + ['memory_total']) + + emitter = Emitter() + stats = collect_stats() if need_stats else {} + info = collect_info() if need_info else {} + engine_tags = [('engine_host', info.get('Name', '')), + ('server_version', info.get('ServerVersion', ''))] + system_ns = host_cpu_nanoseconds() if want('cpu', 'usage_system') else 0 + mem_total = host_memory_total() if wants_any('mem', ['limit', 'usage_percent']) else 0 + + if info: + counts = {'n_containers': 'Containers', 'n_containers_running': 'ContainersRunning', + 'n_containers_stopped': 'ContainersStopped', 'n_containers_paused': 'ContainersPaused', + 'n_images': 'Images', 'n_cpus': 'NCPU', 'n_goroutines': 'NGoroutines', + 'n_used_file_descriptors': 'NFd', 'n_listener_events': 'NEventsListener'} + fields = {name: integer(info[key]) for name, key in counts.items() + if want('engine', name) and key in info} + emitter.add('docker', engine_tags, fields) + # inputs.docker published memory_total as its own point + if want('engine', 'memory_total') and 'MemTotal' in info: + emitter.add('docker', engine_tags, {'memory_total': integer(info['MemTotal'])}) + + for container in collect_inspect(): + name = container.get('Name', '').lstrip('/') + if not name: + continue + state = container.get('State', {}) + status = state.get('Status', 'unknown') + pid = state.get('Pid') + container_id = container.get('Id', '') + tags = [('container_name', name), ('container_status', status)] + if identity: + image, version = parse_image(container.get('Config', {}).get('Image', '')) + tags += [('container_image', image), ('container_version', version)] + engine_tags + + status_fields = {} + if want('status', 'uptime_ns') or want('status', 'started_at') or want('status', 'finished_at'): + started = to_nanoseconds(state.get('StartedAt', '')) + finished = to_nanoseconds(state.get('FinishedAt', '')) + if started is not None: + if want('status', 'started_at'): + status_fields['started_at'] = integer(started) + if want('status', 'uptime_ns'): + end = finished if finished is not None and finished >= started else int( + datetime.now(timezone.utc).timestamp() * NANOSEC) + status_fields['uptime_ns'] = integer(end - started) + if finished is not None and want('status', 'finished_at'): + status_fields['finished_at'] = integer(finished) + if want('status', 'oomkilled'): + status_fields['oomkilled'] = 'true' if state.get('OOMKilled') else 'false' + if want('status', 'pid'): + status_fields['pid'] = integer(pid or 0) + if want('status', 'exitcode'): + status_fields['exitcode'] = integer(state.get('ExitCode', 0)) + if want('status', 'restart_count'): + status_fields['restart_count'] = integer(container.get('RestartCount', 0)) + if want('status', 'container_id'): + status_fields['container_id'] = quote(container_id) + emitter.add('docker_container_status', tags, status_fields) + + health = state.get('Health') + if health: + health_fields = {} + if want('health', 'health_status'): + health_fields['health_status'] = quote(health.get('Status', '')) + if want('health', 'failing_streak'): + health_fields['failing_streak'] = integer(health.get('FailingStreak', 0)) + emitter.add('docker_container_health', tags, health_fields) + + if status != 'running' or not pid: + continue + path = cgroup_path(pid) + entry = stats.get(name, {}) + + cpu_fields = {} + if want('cpu', 'usage_percent'): + cpu_fields['usage_percent'] = percent(entry.get('CPUPerc', '0%')) + if want('cpu', 'usage_system'): + cpu_fields['usage_system'] = unsigned(system_ns) + if want('cpu', 'container_id'): + cpu_fields['container_id'] = quote(container_id) + if need_cgroup_cpu and path: + cpu = read_pairs(path + '/cpu.stat') + mapping = {'usage_total': 'usage_usec', 'usage_in_usermode': 'user_usec', + 'usage_in_kernelmode': 'system_usec', + 'throttling_periods': 'nr_periods', + 'throttling_throttled_periods': 'nr_throttled', + 'throttling_throttled_time': 'throttled_usec'} + for field, key in mapping.items(): + if want('cpu', field) and key in cpu: + scale = 1 if field.startswith('throttling_') and field != 'throttling_throttled_time' else USEC_TO_NSEC + cpu_fields[field] = unsigned(cpu[key] * scale) + emitter.add('docker_container_cpu', tags + [('cpu', 'cpu-total')], cpu_fields) + + mem_fields = {} + if want('mem', 'container_id'): + mem_fields['container_id'] = quote(container_id) + if need_cgroup_mem and path: + memory = read_pairs(path + '/memory.stat') + for field in ['active_anon', 'active_file', 'inactive_anon', 'inactive_file', + 'unevictable', 'pgfault', 'pgmajfault']: + if want('mem', field) and field in memory: + mem_fields[field] = unsigned(memory[field]) + current = read_value(path + '/memory.current') + # inputs.docker reports usage net of reclaimable page cache + usage = max(current - memory.get('inactive_file', 0), 0) if current is not None else None + raw_limit = read_value(path + '/memory.max') + limit = raw_limit if raw_limit is not None else mem_total + if want('mem', 'usage') and usage is not None: + mem_fields['usage'] = unsigned(usage) + if want('mem', 'limit'): + mem_fields['limit'] = unsigned(limit) + if want('mem', 'usage_percent') and usage is not None: + # same ratio inputs.docker computes, from the same two values + mem_fields['usage_percent'] = usage / limit * 100.0 if limit else 0.0 + if want('mem', 'max_usage'): + peak = read_value(path + '/memory.peak') + if peak is not None: + mem_fields['max_usage'] = unsigned(peak) + emitter.add('docker_container_mem', tags, mem_fields) + + # inputs.docker emitted nothing for host-network containers; its Networks map was empty + if need_net and container.get('HostConfig', {}).get('NetworkMode', '') != 'host': + counters = net_counters(pid) + if counters is not None: + net_out = {field: unsigned(counters[field]) for field in net_fields if want('net', field)} + if want('net', 'container_id'): + net_out['container_id'] = quote(container_id) + emitter.add('docker_container_net', tags + [('network', 'total')], net_out) + + if need_blkio and path: + counters = blkio_counters(path) + if counters is not None: + blkio_out = {field: unsigned(value) for field, value in counters.items() if want('blkio', field)} + if want('blkio', 'container_id'): + blkio_out['container_id'] = quote(container_id) + emitter.add('docker_container_blkio', tags + [('device', 'total')], blkio_out) + + print('\n'.join(emitter.lines)) + + +if __name__ == '__main__': + main() +{% endraw %} diff --git a/salt/telegraf/tools/sbin_jinja/so-container-stats_test.py b/salt/telegraf/tools/sbin_jinja/so-container-stats_test.py new file mode 100644 index 000000000..fa6792860 --- /dev/null +++ b/salt/telegraf/tools/sbin_jinja/so-container-stats_test.py @@ -0,0 +1,635 @@ +# Copyright Security Onion Solutions LLC and/or licensed to Security Onion Solutions LLC under one +# or more contributor license agreements. Licensed under the Elastic License 2.0 as shown at +# https://securityonion.net/license; you may not use this file except in compliance with the +# Elastic License 2.0. + +# so-container-stats replaces telegraf's inputs.docker plugin, which was dropped along with the +# docker socket. The InfluxDB dashboards query its output by measurement, tag and field name, so +# these tests pin that contract: the shipped defaults, the per-stat toggles, the field types, and +# the value semantics copied from the plugin. +# +# The collector is a jinja template, so every test renders it the way salt does and imports the +# result. Docker and the cgroup filesystem are faked, so nothing here needs a container runtime. + +import contextlib +import importlib.util +import io +import json +import os +import tempfile +import unittest + +import jinja2 +import yaml + +HERE = os.path.dirname(os.path.abspath(__file__)) +TEMPLATE = os.path.join(HERE, 'so-container-stats') +DEFAULTS = os.path.join(HERE, '..', '..', 'defaults.yaml') + +# a container id is a 64 character hex string; the plugin published it in full +SOC_ID = 'a' * 64 +TELEGRAF_ID = 'b' * 64 +NGINX_ID = 'c' * 64 +IDSTOOLS_ID = 'd' * 64 + + +def shipped_defaults(): + with open(DEFAULTS) as handle: + return yaml.safe_load(handle)['telegraf']['container_stats'] + + +def all_enabled(): + return {group: {field: True for field in fields} for group, fields in shipped_defaults().items()} + + +def render(settings): + """Render the template as salt does, import it, and hand back the module.""" + with open(TEMPLATE) as handle: + source = handle.read() + rendered = jinja2.Template(source, keep_trailing_newline=True).render(CONTAINER_STATS=settings) + path = os.path.join(tempfile.mkdtemp(), 'so_container_stats.py') + with open(path, 'w') as handle: + handle.write(rendered) + spec = importlib.util.spec_from_file_location('so_container_stats', path) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def split_escaped(text, sep): + """Split on an unescaped, unquoted separator, the way influx line protocol is written.""" + parts, current, escaped, quoted = [], '', False, False + for char in text: + if escaped: + current += char + escaped = False + elif char == '\\': + current += char + escaped = True + elif char == '"': + quoted = not quoted + current += char + elif char == sep and not quoted: + parts.append(current) + current = '' + else: + current += char + parts.append(current) + return parts + + +def split_on_space(line): + escaped = quoted = False + for index, char in enumerate(line): + if escaped: + escaped = False + elif char == '\\': + escaped = True + elif char == '"': + quoted = not quoted + elif char == ' ' and not quoted: + return line[:index], line[index + 1:] + return line, '' + + +def parse(output): + """Parse line protocol into {(measurement, container): (tags, fields)}.""" + points = {} + for line in output.splitlines(): + line = line.strip() + if not line: + continue + head, fieldpart = split_on_space(line) + pieces = split_escaped(head, ',') + measurement, tags = pieces[0], {} + for piece in pieces[1:]: + key, _, value = piece.partition('=') + tags[key] = value + fields = {} + for piece in split_escaped(fieldpart, ','): + key, _, value = piece.partition('=') + fields[key] = value + key = (measurement, tags.get('container_name', '')) + if key in points: + # the engine measurement is published as two points, the way inputs.docker did + points[key][1].update(fields) + else: + points[key] = (tags, fields) + return points + + +def field_type(value): + if value.endswith('u'): + return 'unsigned' + if value.endswith('i'): + return 'integer' + if value.startswith('"'): + return 'string' + if value in ('true', 'false'): + return 'boolean' + return 'float' + + +def container(name, cid, pid, status='running', network='bridge', image='registry:5000/repo/img:3.4.0', + started='2026-09-24T10:00:00.123456789Z', finished='0001-01-01T00:00:00Z', health=None, + oomkilled=False, exitcode=0, restarts=0): + state = {'Status': status, 'Pid': pid, 'StartedAt': started, 'FinishedAt': finished, + 'OOMKilled': oomkilled, 'ExitCode': exitcode} + if health is not None: + state['Health'] = health + return {'Id': cid, 'Name': '/' + name, 'State': state, 'RestartCount': restarts, + 'Config': {'Image': image}, 'HostConfig': {'NetworkMode': network}} + + +class CollectorTestCase(unittest.TestCase): + """Builds a fake docker engine and cgroup tree, then runs the collector against it.""" + + def setUp(self): + self.inspect = [ + container('so-soc', SOC_ID, 1001), + container('so-telegraf', TELEGRAF_ID, 1002, network='host'), + container('so-nginx', NGINX_ID, 1003, health={'Status': 'healthy', 'FailingStreak': 0}), + ] + self.stats = { + 'so-soc': {'Name': 'so-soc', 'CPUPerc': '1.25%', 'MemPerc': '6.39%', 'NetIO': '1kB / 2kB'}, + 'so-telegraf': {'Name': 'so-telegraf', 'CPUPerc': '0.04%', 'MemPerc': '0.58%', 'NetIO': '0B / 0B'}, + 'so-nginx': {'Name': 'so-nginx', 'CPUPerc': '0.00%', 'MemPerc': '0.09%', 'NetIO': '3kB / 4kB'}, + } + self.info = {'Name': 'sohost', 'ServerVersion': '29.2.1', 'Containers': 3, 'ContainersRunning': 3, + 'ContainersStopped': 0, 'ContainersPaused': 0, 'Images': 9, 'NCPU': 8, + 'NGoroutines': 42, 'NFd': 77, 'NEventsListener': 1, 'MemTotal': 16000000000} + # one cgroup per container, addressed through /proc//cgroup exactly as the collector does + self.files = { + '/proc/meminfo': 'MemTotal: 15625000 kB\n', + '/proc/stat': 'cpu 100 200 300 400\ncpu0 1 2 3 4\n', + } + for pid in (1001, 1002, 1003): + self.files['/proc/%d/cgroup' % pid] = '0::/scope%d\n' % pid + self.cgroup(pid, 'cpu.stat', + 'usage_usec 1000\nuser_usec 600\nsystem_usec 400\nnr_periods 5\nnr_throttled 2\nthrottled_usec 700\n') + self.cgroup(pid, 'memory.stat', + 'active_anon 300\nactive_file 40\ninactive_anon 200\ninactive_file 50\nunevictable 0\npgfault 1234\npgmajfault 56\n') + self.cgroup(pid, 'memory.current', '1000\n') + self.cgroup(pid, 'memory.max', '4000\n') + self.cgroup(pid, 'memory.peak', '2500\n') + self.cgroup(pid, 'io.stat', '8:0 rbytes=100 wbytes=200 rios=1 wios=2\n252:0 rbytes=10 wbytes=20 rios=1 wios=1\n') + self.files['/proc/%d/net/dev' % pid] = ( + 'Inter-| Receive | Transmit\n' + ' face |bytes packets errs drop fifo frame compressed multicast|bytes packets errs drop fifo colls carrier compressed\n' + ' lo: 9999 99 9 9 0 0 0 0 9999 99 9 9 0 0 0 0\n' + ' eth0: 1000 10 1 2 0 0 0 0 2000 20 3 4 0 0 0 0\n' + ' eth1: 500 5 0 0 0 0 0 0 1000 10 0 0 0 0 0 0\n') + + def cgroup(self, pid, name, contents): + self.files['/sys/fs/cgroup/scope%d/%s' % (pid, name)] = contents + + def run_collector(self, settings=None, module=None): + module = module or render(settings if settings is not None else all_enabled()) + + def fake_docker(args): + if args[0] == 'stats': + return ''.join(json.dumps(entry) + '\n' for entry in self.stats.values()) + if args[0] == 'ps': + return ' '.join(entry['Id'] for entry in self.inspect) + if args[0] == 'inspect': + return json.dumps(self.inspect) + if args[0] == 'info': + return json.dumps(self.info) + raise AssertionError('unexpected docker call: %s' % args) + + module.docker = fake_docker + module.read_text = lambda path: self.files.get(path, '') + module.CGROUP_ROOT = '/sys/fs/cgroup' + buffer = io.StringIO() + with contextlib.redirect_stdout(buffer): + module.main() + self.output = buffer.getvalue() + return parse(self.output) + + +class TestShippedDefaults(CollectorTestCase): + + def test_defaults_emit_only_the_dashboard_fields(self): + # the Security Onion Performance dashboard queries exactly these five + points = self.run_collector(shipped_defaults()) + emitted = {(measurement, field) for (measurement, _), (_, fields) in points.items() for field in fields} + self.assertEqual(emitted, { + ('docker_container_cpu', 'usage_percent'), + ('docker_container_mem', 'usage_percent'), + ('docker_container_net', 'rx_bytes'), + ('docker_container_status', 'uptime_ns'), + ('docker_container_status', 'oomkilled'), + }) + + def test_defaults_do_not_emit_the_opt_in_measurements(self): + points = self.run_collector(shipped_defaults()) + measurements = {measurement for measurement, _ in points} + self.assertNotIn('docker', measurements) + self.assertNotIn('docker_container_blkio', measurements) + self.assertNotIn('docker_container_health', measurements) + + def test_defaults_carry_the_tags_the_dashboard_filters_on(self): + points = self.run_collector(shipped_defaults()) + tags, _ = points[('docker_container_cpu', 'so-soc')] + self.assertEqual(tags['container_status'], 'running') + self.assertEqual(tags['cpu'], 'cpu-total') + # identity tags are opt in, so they must be absent by default + self.assertNotIn('container_image', tags) + self.assertNotIn('engine_host', tags) + + +class TestToggles(CollectorTestCase): + + def test_enabling_one_stat_adds_only_that_field(self): + settings = shipped_defaults() + settings['cpu']['usage_total'] = True + _, fields = self.run_collector(settings)[('docker_container_cpu', 'so-soc')] + self.assertIn('usage_total', fields) + self.assertNotIn('usage_in_usermode', fields) + + def test_disabling_one_stat_leaves_its_neighbours(self): + settings = all_enabled() + settings['blkio']['io_service_bytes_recursive_read'] = False + _, fields = self.run_collector(settings)[('docker_container_blkio', 'so-soc')] + self.assertNotIn('io_service_bytes_recursive_read', fields) + self.assertIn('io_service_bytes_recursive_write', fields) + + def test_identity_tags_are_added_when_enabled(self): + tags, _ = self.run_collector(all_enabled())[('docker_container_cpu', 'so-soc')] + self.assertEqual(tags['container_image'], 'registry:5000/repo/img') + self.assertEqual(tags['container_version'], '3.4.0') + self.assertEqual(tags['engine_host'], 'sohost') + self.assertEqual(tags['server_version'], '29.2.1') + + def test_everything_off_emits_nothing(self): + settings = {group: {field: False for field in fields} for group, fields in shipped_defaults().items()} + self.assertEqual(self.run_collector(settings), {}) + + +class TestFieldTypes(CollectorTestCase): + """inputs.docker wrote the cgroup and network counters as unsigned; influx treats u and i as + different field types, so a mismatch breaks queries spanning the change.""" + + def test_counter_fields_are_unsigned(self): + points = self.run_collector(all_enabled()) + for measurement, field in (('docker_container_cpu', 'usage_total'), + ('docker_container_cpu', 'throttling_periods'), + ('docker_container_mem', 'usage'), + ('docker_container_mem', 'limit'), + ('docker_container_mem', 'pgfault'), + ('docker_container_net', 'rx_bytes'), + ('docker_container_blkio', 'io_service_bytes_recursive_read')): + _, fields = points[(measurement, 'so-soc')] + self.assertEqual(field_type(fields[field]), 'unsigned', '%s.%s' % (measurement, field)) + + def test_status_and_engine_fields_are_signed(self): + points = self.run_collector(all_enabled()) + _, status = points[('docker_container_status', 'so-soc')] + for field in ('uptime_ns', 'pid', 'exitcode', 'restart_count', 'started_at'): + self.assertEqual(field_type(status[field]), 'integer', field) + _, engine = points[('docker', '')] + self.assertEqual(field_type(engine['n_containers']), 'integer') + + def test_percentages_are_floats_and_ids_are_quoted_strings(self): + points = self.run_collector(all_enabled()) + _, cpu = points[('docker_container_cpu', 'so-soc')] + self.assertEqual(field_type(cpu['usage_percent']), 'float') + self.assertEqual(field_type(cpu['container_id']), 'string') + self.assertEqual(cpu['container_id'], '"%s"' % SOC_ID) + _, status = points[('docker_container_status', 'so-soc')] + self.assertEqual(field_type(status['oomkilled']), 'boolean') + _, health = points[('docker_container_health', 'so-nginx')] + self.assertEqual(health['health_status'], '"healthy"') + + +class TestMemorySemantics(CollectorTestCase): + + def test_usage_subtracts_reclaimable_page_cache(self): + # inputs.docker reports usage net of inactive_file: 1000 - 50 + _, fields = self.run_collector(all_enabled())[('docker_container_mem', 'so-soc')] + self.assertEqual(fields['usage'], '950u') + + def test_usage_percent_is_usage_over_limit(self): + _, fields = self.run_collector(all_enabled())[('docker_container_mem', 'so-soc')] + self.assertAlmostEqual(float(fields['usage_percent']), 950 / 4000 * 100.0) + + def test_limit_falls_back_to_host_memory_when_unlimited(self): + for pid in (1001, 1002, 1003): + self.cgroup(pid, 'memory.max', 'max\n') + _, fields = self.run_collector(all_enabled())[('docker_container_mem', 'so-soc')] + self.assertEqual(fields['limit'], '%du' % (15625000 * 1024)) + + def test_max_usage_reports_the_cgroup_peak(self): + # the daemon reports 0 on cgroup v2, so this deliberately carries the real peak + _, fields = self.run_collector(all_enabled())[('docker_container_mem', 'so-soc')] + self.assertEqual(fields['max_usage'], '2500u') + + +class TestCpuSemantics(CollectorTestCase): + + def test_microsecond_counters_are_published_as_nanoseconds(self): + _, fields = self.run_collector(all_enabled())[('docker_container_cpu', 'so-soc')] + self.assertEqual(fields['usage_total'], '1000000u') + self.assertEqual(fields['usage_in_usermode'], '600000u') + self.assertEqual(fields['usage_in_kernelmode'], '400000u') + self.assertEqual(fields['throttling_throttled_time'], '700000u') + + def test_throttling_counts_are_not_scaled(self): + _, fields = self.run_collector(all_enabled())[('docker_container_cpu', 'so-soc')] + self.assertEqual(fields['throttling_periods'], '5u') + self.assertEqual(fields['throttling_throttled_periods'], '2u') + + def test_usage_system_is_host_wide_cpu_time(self): + _, fields = self.run_collector(all_enabled())[('docker_container_cpu', 'so-soc')] + self.assertEqual(field_type(fields['usage_system']), 'unsigned') + self.assertNotEqual(fields['usage_system'], '0u') + + +class TestNetwork(CollectorTestCase): + + def test_counters_sum_interfaces_and_ignore_loopback(self): + _, fields = self.run_collector(all_enabled())[('docker_container_net', 'so-soc')] + self.assertEqual(fields['rx_bytes'], '1500u') + self.assertEqual(fields['rx_packets'], '15u') + self.assertEqual(fields['tx_bytes'], '3000u') + self.assertEqual(fields['rx_dropped'], '2u') + + def test_host_network_containers_emit_no_row(self): + # inputs.docker skipped these: its Networks map is empty for --net=host + points = self.run_collector(all_enabled()) + self.assertNotIn(('docker_container_net', 'so-telegraf'), points) + self.assertIn(('docker_container_net', 'so-soc'), points) + + def test_total_tag_is_present(self): + tags, _ = self.run_collector(all_enabled())[('docker_container_net', 'so-soc')] + self.assertEqual(tags['network'], 'total') + + +class TestBlkio(CollectorTestCase): + + def test_counters_sum_devices(self): + _, fields = self.run_collector(all_enabled())[('docker_container_blkio', 'so-soc')] + self.assertEqual(fields['io_service_bytes_recursive_read'], '110u') + self.assertEqual(fields['io_service_bytes_recursive_write'], '220u') + + def test_container_with_no_io_reports_zero_rather_than_disappearing(self): + for pid in (1001, 1002, 1003): + self.cgroup(pid, 'io.stat', '') + tags, fields = self.run_collector(all_enabled())[('docker_container_blkio', 'so-soc')] + self.assertEqual(fields['io_service_bytes_recursive_read'], '0u') + self.assertEqual(tags['device'], 'total') + + +class TestStatus(CollectorTestCase): + + def test_running_container_uptime_counts_from_start(self): + _, fields = self.run_collector(all_enabled())[('docker_container_status', 'so-soc')] + self.assertGreater(int(fields['uptime_ns'].rstrip('i')), 0) + self.assertNotIn('finished_at', fields) + + def test_exited_container_reports_its_lifetime_and_finished_at(self): + self.inspect.append(container('so-idstools', IDSTOOLS_ID, 0, status='exited', + started='2026-09-24T10:00:00.000000000Z', + finished='2026-09-24T10:00:02.000000000Z', exitcode=3, restarts=1)) + points = self.run_collector(all_enabled()) + tags, fields = points[('docker_container_status', 'so-idstools')] + self.assertEqual(tags['container_status'], 'exited') + self.assertEqual(fields['uptime_ns'], '2000000000i') + self.assertEqual(fields['finished_at'], '1790244002000000000i') + self.assertEqual(fields['exitcode'], '3i') + self.assertEqual(fields['restart_count'], '1i') + # a stopped container has no live stats, so only the status row is emitted + self.assertNotIn(('docker_container_cpu', 'so-idstools'), points) + + def test_health_is_emitted_only_for_containers_with_a_healthcheck(self): + points = self.run_collector(all_enabled()) + self.assertIn(('docker_container_health', 'so-nginx'), points) + self.assertNotIn(('docker_container_health', 'so-soc'), points) + + def test_oomkilled_is_reported(self): + self.inspect[0]['State']['OOMKilled'] = True + _, fields = self.run_collector(all_enabled())[('docker_container_status', 'so-soc')] + self.assertEqual(fields['oomkilled'], 'true') + + +class TestEngineMeasurement(CollectorTestCase): + + def test_engine_counts_come_from_docker_info(self): + _, fields = self.run_collector(all_enabled())[('docker', '')] + self.assertEqual(fields['n_containers'], '3i') + self.assertEqual(fields['n_cpus'], '8i') + self.assertEqual(fields['n_used_file_descriptors'], '77i') + + def test_engine_is_published_as_two_points(self): + # inputs.docker emitted memory_total on its own point, so keep that shape + self.run_collector(all_enabled()) + engine = [line for line in self.output.splitlines() if line.startswith('docker,')] + self.assertEqual(len(engine), 2) + self.assertTrue(any('memory_total=' in line for line in engine)) + self.assertTrue(any('n_containers=' in line for line in engine)) + + def test_engine_rows_are_tagged_with_host_and_version(self): + tags, _ = self.run_collector(all_enabled())[('docker', '')] + self.assertEqual(tags['engine_host'], 'sohost') + self.assertEqual(tags['server_version'], '29.2.1') + + +class TestLineProtocol(CollectorTestCase): + + def test_tag_values_are_escaped(self): + self.inspect[0]['Name'] = '/odd name,with=chars' + self.stats['odd name,with=chars'] = self.stats.pop('so-soc') + self.stats['odd name,with=chars']['Name'] = 'odd name,with=chars' + output = self.run_collector(all_enabled()) + self.assertIn(('docker_container_cpu', 'odd\\ name\\,with\\=chars'), output) + + def test_image_without_a_tag_reports_version_unknown(self): + self.inspect[0]['Config']['Image'] = 'busybox' + tags, _ = self.run_collector(all_enabled())[('docker_container_cpu', 'so-soc')] + self.assertEqual(tags['container_image'], 'busybox') + self.assertEqual(tags['container_version'], 'unknown') + + def test_every_line_has_a_measurement_tagset_and_fieldset(self): + module = render(all_enabled()) + + def fake_docker(args): + if args[0] == 'stats': + return ''.join(json.dumps(entry) + '\n' for entry in self.stats.values()) + if args[0] == 'ps': + return ' '.join(entry['Id'] for entry in self.inspect) + if args[0] == 'inspect': + return json.dumps(self.inspect) + return json.dumps(self.info) + + module.docker = fake_docker + module.read_text = lambda path: self.files.get(path, '') + buffer = io.StringIO() + with contextlib.redirect_stdout(buffer): + module.main() + lines = [line for line in buffer.getvalue().splitlines() if line.strip()] + self.assertTrue(lines) + for line in lines: + head, fieldpart = split_on_space(line) + self.assertIn(',', head, line) + self.assertIn('=', fieldpart, line) + self.assertFalse(fieldpart.endswith(','), line) + + +# group -> (measurement, the container whose row carries it) +GROUP_TARGET = { + 'engine': ('docker', ''), + 'cpu': ('docker_container_cpu', 'so-soc'), + 'mem': ('docker_container_mem', 'so-soc'), + 'net': ('docker_container_net', 'so-soc'), + 'blkio': ('docker_container_blkio', 'so-soc'), + 'status': ('docker_container_status', 'so-soc'), + 'health': ('docker_container_health', 'so-nginx'), +} + + +class TestEverySetting(CollectorTestCase): + """Whatever is offered in defaults.yaml has to actually be collectable. These tests are driven + off that file, so a new setting that is never wired up fails here rather than shipping.""" + + def setUp(self): + super().setUp() + # an exited container so status.finished_at has a value to report + self.inspect.append(container('so-idstools', IDSTOOLS_ID, 0, status='exited', + started='2026-09-24T10:00:00.000000000Z', + finished='2026-09-24T10:00:02.000000000Z')) + + def test_every_setting_emits_its_field_when_enabled(self): + points = self.run_collector(all_enabled()) + for group, fields in shipped_defaults().items(): + if group == 'tags': + continue + measurement, name = GROUP_TARGET[group] + if group == 'status': + # finished_at only exists for a container that has actually exited + _, exited = points[(measurement, 'so-idstools')] + self.assertIn('finished_at', exited) + _, emitted = points[(measurement, name)] + for field in fields: + if group == 'status' and field == 'finished_at': + continue + self.assertIn(field, emitted, '%s.%s is offered but never emitted' % (group, field)) + + def test_every_setting_is_individually_wired(self): + # enabling one stat on its own must produce exactly that field, proving each toggle is + # read rather than riding along with a neighbour + for group, fields in shipped_defaults().items(): + if group == 'tags': + continue + measurement, name = GROUP_TARGET[group] + for field in fields: + settings = {other: {key: False for key in values} for other, values in shipped_defaults().items()} + settings[group][field] = True + target = 'so-idstools' if (group == 'status' and field == 'finished_at') else name + points = self.run_collector(settings) + self.assertIn((measurement, target), points, '%s.%s emitted no row' % (group, field)) + _, emitted = points[(measurement, target)] + self.assertEqual(sorted(emitted), [field], '%s.%s did not emit itself alone' % (group, field)) + + def test_all_enabled_values_are_exact(self): + points = self.run_collector(all_enabled()) + clock = os.sysconf('SC_CLK_TCK') + expected = { + ('docker_container_cpu', 'so-soc'): { + 'usage_percent': '1.25', 'usage_total': '1000000u', 'usage_in_usermode': '600000u', + 'usage_in_kernelmode': '400000u', 'usage_system': '%du' % int(1000 * 10**9 / clock), + 'throttling_periods': '5u', 'throttling_throttled_periods': '2u', + 'throttling_throttled_time': '700000u', 'container_id': '"%s"' % SOC_ID, + }, + ('docker_container_mem', 'so-soc'): { + 'usage': '950u', 'limit': '4000u', 'max_usage': '2500u', 'active_anon': '300u', + 'active_file': '40u', 'inactive_anon': '200u', 'inactive_file': '50u', + 'unevictable': '0u', 'pgfault': '1234u', 'pgmajfault': '56u', + 'usage_percent': '23.75', 'container_id': '"%s"' % SOC_ID, + }, + ('docker_container_net', 'so-soc'): { + 'rx_bytes': '1500u', 'rx_packets': '15u', 'rx_errors': '1u', 'rx_dropped': '2u', + 'tx_bytes': '3000u', 'tx_packets': '30u', 'tx_errors': '3u', 'tx_dropped': '4u', + 'container_id': '"%s"' % SOC_ID, + }, + ('docker_container_blkio', 'so-soc'): { + 'io_service_bytes_recursive_read': '110u', + 'io_service_bytes_recursive_write': '220u', 'container_id': '"%s"' % SOC_ID, + }, + ('docker', ''): { + 'n_containers': '3i', 'n_containers_running': '3i', 'n_containers_stopped': '0i', + 'n_containers_paused': '0i', 'n_images': '9i', 'n_cpus': '8i', 'n_goroutines': '42i', + 'n_used_file_descriptors': '77i', 'n_listener_events': '1i', + 'memory_total': '16000000000i', + }, + ('docker_container_health', 'so-nginx'): { + 'health_status': '"healthy"', 'failing_streak': '0i', + }, + } + for key, fields in expected.items(): + _, emitted = points[key] + for field, value in fields.items(): + self.assertEqual(emitted[field], value, '%s %s' % (key[0], field)) + + def test_status_values_are_exact(self): + # uptime is relative to now, so it is checked separately from the fixed fields + points = self.run_collector(all_enabled()) + _, running = points[('docker_container_status', 'so-soc')] + self.assertEqual(running['pid'], '1001i') + self.assertEqual(running['exitcode'], '0i') + self.assertEqual(running['restart_count'], '0i') + self.assertEqual(running['oomkilled'], 'false') + self.assertEqual(running['container_id'], '"%s"' % SOC_ID) + self.assertEqual(int(running['started_at'].rstrip('i')) // 10**9, 1790244000) + self.assertGreater(int(running['uptime_ns'].rstrip('i')), 0) + _, exited = points[('docker_container_status', 'so-idstools')] + self.assertEqual(exited['uptime_ns'], '2000000000i') + self.assertEqual(exited['finished_at'], '1790244002000000000i') + + def test_tags_identity_toggle_controls_the_identity_tags(self): + settings = shipped_defaults() + settings['tags']['identity'] = False + tags, _ = self.run_collector(settings)[('docker_container_cpu', 'so-soc')] + self.assertEqual(sorted(tags), ['container_name', 'container_status', 'cpu']) + settings['tags']['identity'] = True + tags, _ = self.run_collector(settings)[('docker_container_cpu', 'so-soc')] + self.assertEqual(sorted(tags), ['container_image', 'container_name', 'container_status', + 'container_version', 'cpu', 'engine_host', 'server_version']) + + def test_identity_tags_cover_every_documented_tag(self): + tags, _ = self.run_collector(all_enabled())[('docker_container_cpu', 'so-soc')] + for tag in ('container_image', 'container_version', 'engine_host', 'server_version'): + self.assertIn(tag, tags) + + +class TestTemplate(unittest.TestCase): + + def test_template_renders_to_valid_python_for_the_shipped_defaults(self): + with open(TEMPLATE) as handle: + source = handle.read() + rendered = jinja2.Template(source, keep_trailing_newline=True).render(CONTAINER_STATS=shipped_defaults()) + compile(rendered, 'so-container-stats', 'exec') + self.assertNotIn('{%', rendered) + # the docker format strings must survive rendering untouched + self.assertIn('{{json .}}', rendered) + + def test_every_annotated_setting_exists_in_defaults(self): + # SOC reads both trees; an annotation without a default cannot be reverted in the UI + with open(os.path.join(HERE, '..', '..', 'soc_telegraf.yaml')) as handle: + annotated = yaml.safe_load(handle)['telegraf']['container_stats'] + defaults = shipped_defaults() + for group, fields in annotated.items(): + self.assertIn(group, defaults) + for field in fields: + self.assertIn(field, defaults[group], '%s.%s annotated but missing from defaults' % (group, field)) + + def test_every_default_setting_is_annotated_for_soc(self): + with open(os.path.join(HERE, '..', '..', 'soc_telegraf.yaml')) as handle: + annotated = yaml.safe_load(handle)['telegraf']['container_stats'] + for group, fields in shipped_defaults().items(): + self.assertIn(group, annotated) + for field in fields: + self.assertIn(field, annotated[group], '%s.%s missing a SOC annotation' % (group, field)) + + +if __name__ == '__main__': + unittest.main() diff --git a/setup/so-functions b/setup/so-functions index c5fc8fabf..41be9c7a7 100755 --- a/setup/so-functions +++ b/setup/so-functions @@ -1611,6 +1611,7 @@ reserve_group_ids() { logCmd "groupadd -g 949 elastic-agent" logCmd "groupadd -g 947 elastic-fleet" logCmd "groupadd -g 960 kafka" + logCmd "groupadd -g 961 somon" } reserve_ports() {