mirror of
https://github.com/Security-Onion-Solutions/securityonion.git
synced 2026-09-30 11:37:16 +02:00
Compare commits
65
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
72f60fcaa9 | ||
|
|
2684a5ca95 | ||
|
|
bcee63bde5 | ||
|
|
6ce89eb323 | ||
|
|
ebab4b0d90 | ||
|
|
db60c27da2 | ||
|
|
f4defdfde0 | ||
|
|
36652e8f23 | ||
|
|
7bef194540 | ||
|
|
e2bf2837fe | ||
|
|
06704dad22 | ||
|
|
47fe0758d0 | ||
|
|
b71fd93f9d | ||
|
|
d9eff9aa9e | ||
|
|
d8884dbd99 | ||
|
|
6c0d4c15e8 | ||
|
|
2e2f62f265 | ||
|
|
b7a11a525c | ||
|
|
8eef95ea3e | ||
|
|
65e261475d | ||
|
|
bbc28c88b7 | ||
|
|
edaacf79a7 | ||
|
|
ecc643cd33 | ||
|
|
08aaf7948e | ||
|
|
bb57545d08 | ||
|
|
0f7adbbecc | ||
|
|
aeb4fe8f50 | ||
|
|
f3aa39c5a4 | ||
|
|
24077ba974 | ||
|
|
b3567405f9 | ||
|
|
1fc5bb7afa | ||
|
|
1f1d3ded41 | ||
|
|
1e86be11b2 | ||
|
|
f4518e2620 | ||
|
|
a9f7ffc3fe | ||
|
|
b018277d68 | ||
|
|
3be603e203 | ||
|
|
84cd966736 | ||
|
|
fee401a912 | ||
|
|
496b61966f | ||
|
|
52037314be | ||
|
|
9c12c10f96 | ||
|
|
9fc9be2cc9 | ||
|
|
7245843a3c | ||
|
|
a1d17417ea | ||
|
|
bee03d5bae | ||
|
|
56e3e44d04 | ||
|
|
32d1274b80 | ||
|
|
1624e8c094 | ||
|
|
3f3f091a7f | ||
|
|
cb48909578 | ||
|
|
3057775770 | ||
|
|
e4e8b90b9c | ||
|
|
223ace6ff3 | ||
|
|
66e7863336 | ||
|
|
8f253d17a6 | ||
|
|
9652a2053b | ||
|
|
191ee159ef | ||
|
|
a8bfe955a5 | ||
|
|
bd354abe83 | ||
|
|
37782fb45c | ||
|
|
cf3a4ebc27 | ||
|
|
c49008a413 | ||
|
|
fcbea1a1c4 | ||
|
|
6736f9c3a0 |
@@ -13,6 +13,7 @@ body:
|
||||
- 3.1.0
|
||||
- 3.2.0
|
||||
- 3.3.0
|
||||
- 3.4.0
|
||||
- Other (please provide detail below)
|
||||
validations:
|
||||
required: true
|
||||
|
||||
@@ -6,6 +6,9 @@ on:
|
||||
- "salt/sensoroni/files/analyzers/**"
|
||||
- "salt/manager/tools/sbin/**"
|
||||
- "salt/_beacons/**"
|
||||
- "salt/telegraf/tools/sbin_jinja/**"
|
||||
- "salt/telegraf/defaults.yaml"
|
||||
- "salt/telegraf/soc_telegraf.yaml"
|
||||
|
||||
jobs:
|
||||
build:
|
||||
@@ -34,3 +37,25 @@ jobs:
|
||||
- name: Test with pytest
|
||||
run: |
|
||||
PYTHONPATH=${{ matrix.python-code-path }} pytest ${{ matrix.python-code-path }} --cov=${{ matrix.python-code-path }} --doctest-modules --cov-report=term --cov-fail-under=100 --cov-config=pytest.ini
|
||||
|
||||
telegraf-collector:
|
||||
# so-container-stats is a jinja template rather than an importable module, so it gets its
|
||||
# own job: the test renders it the way salt does, then drives it with a faked docker engine
|
||||
# and cgroup tree. No container runtime is needed.
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v3
|
||||
with:
|
||||
python-version: "3.14"
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
python -m pip install --upgrade pip
|
||||
python -m pip install flake8 pytest jinja2 pyyaml
|
||||
- name: Lint with flake8
|
||||
run: |
|
||||
flake8 salt/telegraf/tools/sbin_jinja/so-container-stats_test.py --config=pytest.ini
|
||||
- name: Test with pytest
|
||||
run: |
|
||||
pytest salt/telegraf/tools/sbin_jinja/so-container-stats_test.py -v
|
||||
|
||||
+11
-11
@@ -1,17 +1,17 @@
|
||||
### 3.3.0-20260908 ISO image released on 2026/09/08
|
||||
### 3.3.0-20260911 ISO image released on 2026/09/11
|
||||
|
||||
|
||||
### Download and Verify
|
||||
|
||||
3.3.0-20260908 ISO image:
|
||||
https://download.securityonion.net/file/securityonion/securityonion-3.3.0-20260908.iso
|
||||
3.3.0-20260911 ISO image:
|
||||
https://download.securityonion.net/file/securityonion/securityonion-3.3.0-20260911.iso
|
||||
|
||||
MD5: 5A2C42D0083F2D7B4DC2178C30EBC05F
|
||||
SHA1: 505220A8A3315AFEE601C13772996018425CCD29
|
||||
SHA256: 6EB8401296A1D051FEC558C520D2D4AB1912A72D351A2426FBF8C87FEE2BA844
|
||||
MD5: 12B18433D3A2198A185892FF79CF638F
|
||||
SHA1: 2B3C2E1FA7A78ED1F956E7EDCC12E32593C14EEE
|
||||
SHA256: 0938C73B76CE30EC9E4394D312C79EA7CAC721B6818541697279A6221F7D870D
|
||||
|
||||
Signature for ISO image:
|
||||
https://github.com/Security-Onion-Solutions/securityonion/raw/3/main/sigs/securityonion-3.3.0-20260908.iso.sig
|
||||
https://github.com/Security-Onion-Solutions/securityonion/raw/3/main/sigs/securityonion-3.3.0-20260911.iso.sig
|
||||
|
||||
Signing key:
|
||||
https://raw.githubusercontent.com/Security-Onion-Solutions/securityonion/3/main/KEYS
|
||||
@@ -25,22 +25,22 @@ wget https://raw.githubusercontent.com/Security-Onion-Solutions/securityonion/3/
|
||||
|
||||
Download the signature file for the ISO:
|
||||
```
|
||||
wget https://github.com/Security-Onion-Solutions/securityonion/raw/3/main/sigs/securityonion-3.3.0-20260908.iso.sig
|
||||
wget https://github.com/Security-Onion-Solutions/securityonion/raw/3/main/sigs/securityonion-3.3.0-20260911.iso.sig
|
||||
```
|
||||
|
||||
Download the ISO image:
|
||||
```
|
||||
wget https://download.securityonion.net/file/securityonion/securityonion-3.3.0-20260908.iso
|
||||
wget https://download.securityonion.net/file/securityonion/securityonion-3.3.0-20260911.iso
|
||||
```
|
||||
|
||||
Verify the downloaded ISO image using the signature file:
|
||||
```
|
||||
gpg --verify securityonion-3.3.0-20260908.iso.sig securityonion-3.3.0-20260908.iso
|
||||
gpg --verify securityonion-3.3.0-20260911.iso.sig securityonion-3.3.0-20260911.iso
|
||||
```
|
||||
|
||||
The output should show "Good signature" and the Primary key fingerprint should match what's shown below:
|
||||
```
|
||||
gpg: Signature made Tue 08 Sep 2026 10:07:12 AM EDT using RSA key ID FE507013
|
||||
gpg: Signature made Fri 11 Sep 2026 11:23:56 AM EDT using RSA key ID FE507013
|
||||
gpg: Good signature from "Security Onion Solutions, LLC <info@securityonionsolutions.com>"
|
||||
gpg: WARNING: This key is not certified with a trusted signature!
|
||||
gpg: There is no indication that the signature belongs to the owner.
|
||||
|
||||
+20
-5
@@ -117,14 +117,25 @@ elastic_curl_config:
|
||||
{% endif %}
|
||||
|
||||
|
||||
# A non-root owner here can chmod the directory and replace any script in it, including
|
||||
# the root-owned ones. 555 is the mode the filesystem RPM ships; root ignores it anyway.
|
||||
usr_sbin_perms:
|
||||
file.directory:
|
||||
- name: /usr/sbin
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 555
|
||||
|
||||
common_sbin:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://common/tools/sbin
|
||||
- user: 939
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
- show_changes: False
|
||||
- require:
|
||||
- file: usr_sbin_perms
|
||||
{% if GLOBALS.role == 'so-heavynode' %}
|
||||
- exclude_pat:
|
||||
- so-pcap-import
|
||||
@@ -159,8 +170,8 @@ common_sbin_jinja:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://common/tools/sbin_jinja
|
||||
- user: 939
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
- template: jinja
|
||||
- show_changes: False
|
||||
@@ -173,6 +184,8 @@ so-status_script:
|
||||
file.managed:
|
||||
- name: /usr/sbin/so-status
|
||||
- source: salt://common/tools/sbin/so-status
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
|
||||
{% if GLOBALS.is_sensor %}
|
||||
@@ -204,9 +217,11 @@ sostatus_log:
|
||||
- replace: False
|
||||
|
||||
# Install sostatus check cron. This is used to populate Grid.
|
||||
# telegraf reads status.log on the same minute boundary this runs, so write aside and rename
|
||||
# rather than truncating the file it is reading
|
||||
so-status_check_cron:
|
||||
cron.present:
|
||||
- name: '/usr/sbin/so-status -j > /opt/so/log/sostatus/status.log 2>&1'
|
||||
- name: '/usr/sbin/so-status -j > /opt/so/log/sostatus/status.log.tmp 2>&1; mv -f /opt/so/log/sostatus/status.log.tmp /opt/so/log/sostatus/status.log'
|
||||
- identifier: so-status_check_cron
|
||||
- user: root
|
||||
- minute: '*/1'
|
||||
|
||||
@@ -18,47 +18,61 @@ copy_so-common_common_tools_sbin:
|
||||
- name: /opt/so/saltstack/default/salt/common/tools/sbin/so-common
|
||||
- source: {{UPDATE_DIR}}/salt/common/tools/sbin/so-common
|
||||
- force: True
|
||||
- preserve: True
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
|
||||
copy_so-image-common_common_tools_sbin:
|
||||
file.copy:
|
||||
- name: /opt/so/saltstack/default/salt/common/tools/sbin/so-image-common
|
||||
- source: {{UPDATE_DIR}}/salt/common/tools/sbin/so-image-common
|
||||
- force: True
|
||||
- preserve: True
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
|
||||
copy_soup_manager_tools_sbin:
|
||||
file.copy:
|
||||
- name: /opt/so/saltstack/default/salt/manager/tools/sbin/soup
|
||||
- source: {{UPDATE_DIR}}/salt/manager/tools/sbin/soup
|
||||
- force: True
|
||||
- preserve: True
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
|
||||
copy_so-firewall_manager_tools_sbin:
|
||||
file.copy:
|
||||
- name: /opt/so/saltstack/default/salt/manager/tools/sbin/so-firewall
|
||||
- source: {{UPDATE_DIR}}/salt/manager/tools/sbin/so-firewall
|
||||
- force: True
|
||||
- preserve: True
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
|
||||
copy_so-yaml_manager_tools_sbin:
|
||||
file.copy:
|
||||
- name: /opt/so/saltstack/default/salt/manager/tools/sbin/so-yaml.py
|
||||
- source: {{UPDATE_DIR}}/salt/manager/tools/sbin/so-yaml.py
|
||||
- force: True
|
||||
- preserve: True
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
|
||||
copy_so-repo-sync_manager_tools_sbin:
|
||||
file.copy:
|
||||
- name: /opt/so/saltstack/default/salt/manager/tools/sbin/so-repo-sync
|
||||
- source: {{UPDATE_DIR}}/salt/manager/tools/sbin/so-repo-sync
|
||||
- preserve: True
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
|
||||
copy_bootstrap-salt_manager_tools_sbin:
|
||||
file.copy:
|
||||
- name: /opt/so/saltstack/default/salt/salt/scripts/bootstrap-salt.sh
|
||||
- source: {{UPDATE_DIR}}/salt/salt/scripts/bootstrap-salt.sh
|
||||
- preserve: True
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 644
|
||||
|
||||
# This section is used to put the new script in place so that it can be called during soup.
|
||||
# It is faster than calling the states that normally manage them to put them in place.
|
||||
@@ -67,46 +81,60 @@ copy_so-common_sbin:
|
||||
- name: /usr/sbin/so-common
|
||||
- source: {{UPDATE_DIR}}/salt/common/tools/sbin/so-common
|
||||
- force: True
|
||||
- preserve: True
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
|
||||
copy_so-image-common_sbin:
|
||||
file.copy:
|
||||
- name: /usr/sbin/so-image-common
|
||||
- source: {{UPDATE_DIR}}/salt/common/tools/sbin/so-image-common
|
||||
- force: True
|
||||
- preserve: True
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
|
||||
copy_soup_sbin:
|
||||
file.copy:
|
||||
- name: /usr/sbin/soup
|
||||
- source: {{UPDATE_DIR}}/salt/manager/tools/sbin/soup
|
||||
- force: True
|
||||
- preserve: True
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
|
||||
copy_so-firewall_sbin:
|
||||
file.copy:
|
||||
- name: /usr/sbin/so-firewall
|
||||
- source: {{UPDATE_DIR}}/salt/manager/tools/sbin/so-firewall
|
||||
- force: True
|
||||
- preserve: True
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
|
||||
copy_so-yaml_sbin:
|
||||
file.copy:
|
||||
- name: /usr/sbin/so-yaml.py
|
||||
- source: {{UPDATE_DIR}}/salt/manager/tools/sbin/so-yaml.py
|
||||
- force: True
|
||||
- preserve: True
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
|
||||
copy_so-repo-sync_sbin:
|
||||
file.copy:
|
||||
- name: /usr/sbin/so-repo-sync
|
||||
- source: {{UPDATE_DIR}}/salt/manager/tools/sbin/so-repo-sync
|
||||
- force: True
|
||||
- preserve: True
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
|
||||
copy_bootstrap-salt_sbin:
|
||||
file.copy:
|
||||
- name: /usr/sbin/bootstrap-salt.sh
|
||||
- source: {{UPDATE_DIR}}/salt/salt/scripts/bootstrap-salt.sh
|
||||
- force: True
|
||||
- preserve: True
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
|
||||
@@ -240,7 +240,8 @@ copy_new_files() {
|
||||
cd $UPDATE_DIR
|
||||
rsync -a salt $DEFAULT_SALT_DIR/ --delete "${EXCLUDE_ARGS[@]}"
|
||||
rsync -a pillar $DEFAULT_SALT_DIR/ --delete "${EXCLUDE_ARGS[@]}"
|
||||
chown -R socore:socore $DEFAULT_SALT_DIR/
|
||||
# Root-executed code; SOC only needs to read it. Local dirs stay socore-owned.
|
||||
chown -R root:root $DEFAULT_SALT_DIR/
|
||||
cd /tmp
|
||||
}
|
||||
|
||||
|
||||
@@ -9,6 +9,7 @@ import sys
|
||||
import subprocess
|
||||
import os
|
||||
import json
|
||||
import tempfile
|
||||
|
||||
sys.path.append('/opt/saltstack/salt/lib/python3.10/site-packages/')
|
||||
import salt.config
|
||||
@@ -17,6 +18,21 @@ import salt.loader
|
||||
__opts__ = salt.config.minion_config('/etc/salt/minion')
|
||||
__grains__ = salt.loader.grains(__opts__)
|
||||
|
||||
def write_atomic(path, value):
|
||||
# telegraf reads these files on its own schedule; replacing them by rename means it never
|
||||
# reads a truncated file and reports an empty value as if it were real
|
||||
directory = os.path.dirname(path)
|
||||
handle, temp = tempfile.mkstemp(dir=directory)
|
||||
try:
|
||||
with os.fdopen(handle, 'w') as f:
|
||||
f.write(str(value))
|
||||
os.chmod(temp, 0o644)
|
||||
os.replace(temp, path)
|
||||
except Exception:
|
||||
os.path.exists(temp) and os.unlink(temp)
|
||||
raise
|
||||
|
||||
|
||||
def check_needs_restarted():
|
||||
osfam = __grains__['os_family']
|
||||
val = '0'
|
||||
@@ -34,8 +50,7 @@ def check_needs_restarted():
|
||||
else:
|
||||
fail("Unsupported OS")
|
||||
|
||||
with open(outfile, 'w') as f:
|
||||
f.write(val)
|
||||
write_atomic(outfile, val)
|
||||
|
||||
def check_for_fps():
|
||||
feat = 'fps'
|
||||
@@ -56,8 +71,7 @@ def check_for_fps():
|
||||
# Unknown, so assume 0
|
||||
fps = 0
|
||||
|
||||
with open('/opt/so/log/sostatus/fps_enabled', 'w') as f:
|
||||
f.write(str(fps))
|
||||
write_atomic('/opt/so/log/sostatus/fps_enabled', fps)
|
||||
|
||||
def check_for_lks():
|
||||
feat = 'Lks'
|
||||
@@ -80,8 +94,7 @@ def check_for_lks():
|
||||
lks = 1
|
||||
if lks:
|
||||
break
|
||||
with open('/opt/so/log/sostatus/lks_enabled', 'w') as f:
|
||||
f.write(str(lks))
|
||||
write_atomic('/opt/so/log/sostatus/lks_enabled', lks)
|
||||
|
||||
def fail(msg):
|
||||
print(msg, file=sys.stderr)
|
||||
|
||||
@@ -8,21 +8,37 @@
|
||||
# Elastic License 2.0.
|
||||
|
||||
|
||||
SENSOR_DIR='/nsm'
|
||||
SENSOR_DIR="${SENSOR_DIR:-/nsm}"
|
||||
CRIT_DISK_USAGE=90
|
||||
CUR_USAGE=$(df -P $SENSOR_DIR | tail -1 | awk '{print $5}' | tr -d %)
|
||||
LOG="/opt/so/log/sensor_clean.log"
|
||||
TODAY=$(date -u "+%Y-%m-%d")
|
||||
LOG="${LOG:-/opt/so/log/sensor_clean.log}"
|
||||
LOCK="${LOCK:-/var/tmp/so-sensor-clean.lock}"
|
||||
MAX_PASSES=100
|
||||
|
||||
ZEEK_LOGS="$SENSOR_DIR/zeek/logs"
|
||||
STRELKA_FILES="$SENSOR_DIR/strelka/processed"
|
||||
SURICATA_LOGS="$SENSOR_DIR/suricata"
|
||||
PCAPS="$SENSOR_DIR/pcapout"
|
||||
|
||||
log() {
|
||||
echo "$(date) - $*" >>"$LOG"
|
||||
}
|
||||
|
||||
disk_usage() {
|
||||
df -P "$SENSOR_DIR" | tail -1 | awk '{print $5}' | tr -d %
|
||||
}
|
||||
|
||||
disk_avail() {
|
||||
df -P "$SENSOR_DIR" | tail -1 | awk '{print $4}'
|
||||
}
|
||||
|
||||
# sets REMOVED=1 if anything was actually deleted
|
||||
clean() {
|
||||
## find the oldest Zeek logs directory
|
||||
OLDEST_DIR=$(ls /nsm/zeek/logs/ | grep -v "current" | grep -v "stats" | grep -v "packetloss" | grep -v "zeek_clean" | sort | head -n 1)
|
||||
if [ -z "$OLDEST_DIR" -o "$OLDEST_DIR" == ".." -o "$OLDEST_DIR" == "." ]; then
|
||||
echo "$(date) - No old Zeek logs available to clean up in /nsm/zeek/logs/" >>$LOG
|
||||
#exit 0
|
||||
else
|
||||
echo "$(date) - Removing directory: /nsm/zeek/logs/$OLDEST_DIR" >>$LOG
|
||||
rm -rf /nsm/zeek/logs/"$OLDEST_DIR"
|
||||
OLDEST_DIR=$(ls "$ZEEK_LOGS" 2>/dev/null | grep -v "current" | grep -v "stats" | grep -v "packetloss" | grep -v "zeek_clean" | sort | head -n 1)
|
||||
if [ -n "$OLDEST_DIR" ]; then
|
||||
log "Removing directory: $ZEEK_LOGS/$OLDEST_DIR"
|
||||
rm -rf "$ZEEK_LOGS/$OLDEST_DIR"
|
||||
REMOVED=1
|
||||
fi
|
||||
|
||||
## Remarking for now, as we are moving extracted files to /nsm/strelka/processed
|
||||
@@ -43,58 +59,73 @@ clean() {
|
||||
#fi
|
||||
|
||||
## Clean up Zeek extracted files processed by Strelka
|
||||
STRELKA_FILES='/nsm/strelka/processed'
|
||||
OLDEST_STRELKA=$(find $STRELKA_FILES -type f -printf '%T+ %p\n' | sort -n | head -n 1)
|
||||
if [ -z "$OLDEST_STRELKA" -o "$OLDEST_STRELKA" == ".." -o "$OLDEST_STRELKA" == "." ]; then
|
||||
echo "$(date) - No old files available to clean up in $STRELKA_FILES" >>$LOG
|
||||
else
|
||||
OLDEST_STRELKA=$(find "$STRELKA_FILES" -type f -printf '%T+ %p\n' 2>/dev/null | sort -n | head -n 1)
|
||||
if [ -n "$OLDEST_STRELKA" ]; then
|
||||
OLDEST_STRELKA_DATE=$(echo $OLDEST_STRELKA | awk '{print $1}' | cut -d+ -f1)
|
||||
OLDEST_STRELKA_FILE=$(echo $OLDEST_STRELKA | awk '{print $2}')
|
||||
echo "$(date) - Removing extracted files for $OLDEST_STRELKA_DATE" >>$LOG
|
||||
find $STRELKA_FILES -type f -printf '%T+ %p\n' | grep $OLDEST_STRELKA_DATE | awk '{print $2}' | while read FILE; do
|
||||
echo "$(date) - Removing file: $FILE" >>$LOG
|
||||
log "Removing extracted files for $OLDEST_STRELKA_DATE"
|
||||
REMOVED=1
|
||||
find "$STRELKA_FILES" -type f -printf '%T+ %p\n' 2>/dev/null | grep $OLDEST_STRELKA_DATE | awk '{print $2}' | while read FILE; do
|
||||
log "Removing file: $FILE"
|
||||
rm -f "$FILE"
|
||||
done
|
||||
fi
|
||||
|
||||
## Clean up Suricata log files
|
||||
SURICATA_LOGS='/nsm/suricata'
|
||||
OLDEST_SURICATA=$(find $SURICATA_LOGS -type f -printf '%T+ %p\n' | sort -n | head -n 1)
|
||||
if [[ -z "$OLDEST_SURICATA" ]] || [[ "$OLDEST_SURICATA" == ".." ]] || [[ "$OLDEST_SURICATA" == "." ]]; then
|
||||
echo "$(date) - No old files available to clean up in $SURICATA_LOGS" >>$LOG
|
||||
else
|
||||
OLDEST_SURICATA=$(find "$SURICATA_LOGS" -type f -printf '%T+ %p\n' 2>/dev/null | sort -n | head -n 1)
|
||||
if [ -n "$OLDEST_SURICATA" ]; then
|
||||
OLDEST_SURICATA_DATE=$(echo $OLDEST_SURICATA | awk '{print $1}' | cut -d+ -f1)
|
||||
OLDEST_SURICATA_FILE=$(echo $OLDEST_SURICATA | awk '{print $2}')
|
||||
echo "$(date) - Removing logs for $OLDEST_SURICATA_DATE" >>$LOG
|
||||
find $SURICATA_LOGS -type f -printf '%T+ %p\n' | grep $OLDEST_SURICATA_DATE | awk '{print $2}' | while read FILE; do
|
||||
echo "$(date) - Removing file: $FILE" >>$LOG
|
||||
log "Removing logs for $OLDEST_SURICATA_DATE"
|
||||
REMOVED=1
|
||||
find "$SURICATA_LOGS" -type f -printf '%T+ %p\n' 2>/dev/null | grep $OLDEST_SURICATA_DATE | awk '{print $2}' | while read FILE; do
|
||||
log "Removing file: $FILE"
|
||||
rm -f "$FILE"
|
||||
done
|
||||
fi
|
||||
|
||||
## Clean up extracted pcaps
|
||||
PCAPS='/nsm/pcapout'
|
||||
OLDEST_PCAP=$(find $PCAPS -type f -printf '%T+ %p\n' | sort -n | head -n 1)
|
||||
if [ -z "$OLDEST_PCAP" -o "$OLDEST_PCAP" == ".." -o "$OLDEST_PCAP" == "." ]; then
|
||||
echo "$(date) - No old files available to clean up in $PCAPS" >>$LOG
|
||||
else
|
||||
OLDEST_PCAP=$(find "$PCAPS" -type f -printf '%T+ %p\n' 2>/dev/null | sort -n | head -n 1)
|
||||
if [ -n "$OLDEST_PCAP" ]; then
|
||||
OLDEST_PCAP_DATE=$(echo $OLDEST_PCAP | awk '{print $1}' | cut -d+ -f1)
|
||||
OLDEST_PCAP_FILE=$(echo $OLDEST_PCAP | awk '{print $2}')
|
||||
echo "$(date) - Removing extracted files for $OLDEST_PCAP_DATE" >>$LOG
|
||||
find $PCAPS -type f -printf '%T+ %p\n' | grep $OLDEST_PCAP_DATE | awk '{print $2}' | while read FILE; do
|
||||
echo "$(date) - Removing file: $FILE" >>$LOG
|
||||
log "Removing extracted files for $OLDEST_PCAP_DATE"
|
||||
REMOVED=1
|
||||
find "$PCAPS" -type f -printf '%T+ %p\n' 2>/dev/null | grep $OLDEST_PCAP_DATE | awk '{print $2}' | while read FILE; do
|
||||
log "Removing file: $FILE"
|
||||
rm -f "$FILE"
|
||||
done
|
||||
fi
|
||||
}
|
||||
|
||||
# Check to see if we are already running
|
||||
NUM_RUNNING=$(pgrep -cf "/bin/bash /usr/sbin/so-sensor-clean")
|
||||
[ "$NUM_RUNNING" -gt 1 ] && echo "$(date) - $NUM_RUNNING sensor clean script processes running...exiting." >>$LOG && exit 0
|
||||
|
||||
if [ "$CUR_USAGE" -gt "$CRIT_DISK_USAGE" ]; then
|
||||
while [ "$CUR_USAGE" -gt "$CRIT_DISK_USAGE" ]; do
|
||||
clean
|
||||
CUR_USAGE=$(df -P $SENSOR_DIR | tail -1 | awk '{print $5}' | tr -d %)
|
||||
done
|
||||
# Only one instance at a time; the lock is the fd, so it releases on any exit
|
||||
exec 9>"$LOCK" || exit 1
|
||||
if ! flock -n 9; then
|
||||
log "another so-sensor-clean is already running (lock $LOCK held); exiting"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
CUR_USAGE=$(disk_usage)
|
||||
[ "$CUR_USAGE" -gt "$CRIT_DISK_USAGE" ] || exit 0
|
||||
|
||||
log "$SENSOR_DIR at ${CUR_USAGE}% (threshold ${CRIT_DISK_USAGE}%); starting cleanup"
|
||||
|
||||
PASS=0
|
||||
while [ "$CUR_USAGE" -gt "$CRIT_DISK_USAGE" ]; do
|
||||
PASS=$((PASS + 1))
|
||||
if [ "$PASS" -gt "$MAX_PASSES" ]; then
|
||||
log "stopping after $MAX_PASSES passes; $SENSOR_DIR still at ${CUR_USAGE}%"
|
||||
break
|
||||
fi
|
||||
|
||||
REMOVED=0
|
||||
BEFORE=$(disk_avail)
|
||||
clean
|
||||
CUR_USAGE=$(disk_usage)
|
||||
|
||||
if [ "$REMOVED" -eq 0 ]; then
|
||||
log "nothing left to remove in $ZEEK_LOGS, $STRELKA_FILES, $SURICATA_LOGS, $PCAPS; $SENSOR_DIR still at ${CUR_USAGE}% - space is consumed outside of NSM cleanup scope"
|
||||
break
|
||||
fi
|
||||
if [ "$(disk_avail)" -le "$BEFORE" ]; then
|
||||
log "pass $PASS freed no space; $SENSOR_DIR still at ${CUR_USAGE}% - stopping until next run"
|
||||
break
|
||||
fi
|
||||
done
|
||||
|
||||
@@ -125,4 +125,6 @@ else
|
||||
RAIDSTATUS=1
|
||||
fi
|
||||
|
||||
echo "nsmraid=$RAIDSTATUS" > /opt/so/log/raid/status.log
|
||||
# telegraf reads this file; write aside and rename so it never sees a half-written file
|
||||
echo "nsmraid=$RAIDSTATUS" > /opt/so/log/raid/status.log.tmp
|
||||
mv -f /opt/so/log/raid/status.log.tmp /opt/so/log/raid/status.log
|
||||
|
||||
@@ -1,6 +1,12 @@
|
||||
docker:
|
||||
range: '172.17.1.0/24'
|
||||
gateway: '172.17.1.1'
|
||||
networks:
|
||||
sobridge: {}
|
||||
soauth:
|
||||
range: '172.17.2.0/24'
|
||||
gateway: '172.17.2.1'
|
||||
manager_only: True
|
||||
ulimits:
|
||||
- name: nofile
|
||||
soft: 1048576
|
||||
@@ -58,18 +64,18 @@ docker:
|
||||
ulimits: []
|
||||
'so-kratos':
|
||||
final_octet: 28
|
||||
networks: ['soauth']
|
||||
port_bindings:
|
||||
- 0.0.0.0:4433:4433
|
||||
- 0.0.0.0:4434:4434
|
||||
custom_bind_mounts: []
|
||||
extra_hosts: []
|
||||
extra_env: []
|
||||
ulimits: []
|
||||
'so-hydra':
|
||||
final_octet: 30
|
||||
networks: ['soauth']
|
||||
port_bindings:
|
||||
- 0.0.0.0:4444:4444
|
||||
- 0.0.0.0:4445:4445
|
||||
custom_bind_mounts: []
|
||||
extra_hosts: []
|
||||
extra_env: []
|
||||
@@ -128,6 +134,7 @@ docker:
|
||||
ulimits: []
|
||||
'so-soc':
|
||||
final_octet: 34
|
||||
networks: ['sobridge', 'soauth']
|
||||
port_bindings:
|
||||
- 0.0.0.0:9822:9822
|
||||
custom_bind_mounts: []
|
||||
|
||||
@@ -1,8 +1,26 @@
|
||||
{% import_yaml 'docker/defaults.yaml' as DOCKERDEFAULTS %}
|
||||
{% set DOCKERMERGED = salt['pillar.get']('docker', DOCKERDEFAULTS.docker, merge=True) %}
|
||||
{% set RANGESPLIT = DOCKERMERGED.range.split('.') %}
|
||||
{% set FIRSTTHREE = RANGESPLIT[0] ~ '.' ~ RANGESPLIT[1] ~ '.' ~ RANGESPLIT[2] ~ '.' %}
|
||||
|
||||
{% if DOCKERMERGED.networks.sobridge is not mapping %}
|
||||
{% do DOCKERMERGED.networks.update({'sobridge': {}}) %}
|
||||
{% endif %}
|
||||
{% do DOCKERMERGED.networks['sobridge'].update({'range': DOCKERMERGED.range, 'gateway': DOCKERMERGED.gateway}) %}
|
||||
|
||||
{% for netname, net in DOCKERMERGED.networks.items() %}
|
||||
{% set RANGESPLIT = net.range.split('.') %}
|
||||
{% do net.update({'prefix': RANGESPLIT[0] ~ '.' ~ RANGESPLIT[1] ~ '.' ~ RANGESPLIT[2] ~ '.'}) %}
|
||||
{% endfor %}
|
||||
|
||||
{% for container, vals in DOCKERMERGED.containers.items() %}
|
||||
{% do DOCKERMERGED.containers[container].update({'ip': FIRSTTHREE ~ DOCKERMERGED.containers[container].final_octet}) %}
|
||||
{% set CONTAINER_NETS = vals.get('networks', ['sobridge']) %}
|
||||
{% set IPS = {} %}
|
||||
{% for netname in CONTAINER_NETS %}
|
||||
{% do IPS.update({netname: DOCKERMERGED.networks[netname].prefix ~ vals.final_octet}) %}
|
||||
{% endfor %}
|
||||
{% do DOCKERMERGED.containers[container].update({
|
||||
'networks': CONTAINER_NETS,
|
||||
'ips': IPS,
|
||||
'network': CONTAINER_NETS[0],
|
||||
'ip': IPS[CONTAINER_NETS[0]]
|
||||
}) %}
|
||||
{% endfor %}
|
||||
|
||||
+10
-6
@@ -71,15 +71,19 @@ dockerreserveports:
|
||||
- source: salt://common/files/99-reserved-ports.conf
|
||||
- name: /etc/sysctl.d/99-reserved-ports.conf
|
||||
|
||||
sos_docker_net:
|
||||
{% for NETNAME, NETWORK in DOCKERMERGED.networks.items() %}
|
||||
{% if not NETWORK.get('manager_only') or GLOBALS.get('is_manager', False) %}
|
||||
sos_docker_net_{{ NETNAME }}:
|
||||
docker_network.present:
|
||||
- name: sobridge
|
||||
- subnet: {{ DOCKERMERGED.range }}
|
||||
- gateway: {{ DOCKERMERGED.gateway }}
|
||||
- name: {{ NETNAME }}
|
||||
- subnet: {{ NETWORK.range }}
|
||||
- gateway: {{ NETWORK.gateway }}
|
||||
- options:
|
||||
com.docker.network.bridge.name: 'sobridge'
|
||||
com.docker.network.bridge.name: '{{ NETNAME }}'
|
||||
com.docker.network.driver.mtu: '1500'
|
||||
com.docker.network.bridge.enable_ip_masquerade: 'true'
|
||||
com.docker.network.bridge.enable_icc: 'true'
|
||||
com.docker.network.bridge.host_binding_ipv4: '0.0.0.0'
|
||||
- unless: ip l | grep sobridge
|
||||
- unless: ip l | grep {{ NETNAME }}
|
||||
{% endif %}
|
||||
{% endfor %}
|
||||
|
||||
@@ -7,6 +7,40 @@ docker:
|
||||
description: Default docker IP range for containers.
|
||||
helpLink: docker
|
||||
advanced: True
|
||||
networks:
|
||||
sobridge:
|
||||
description: |
|
||||
The default docker network, carrying most containers. Its range and gateway are taken
|
||||
from the docker.range and docker.gateway settings above rather than set here.
|
||||
helpLink: docker
|
||||
readonly: True
|
||||
advanced: True
|
||||
global: True
|
||||
soauth:
|
||||
range:
|
||||
description: |
|
||||
IP range for the soauth docker network, an isolated network for the authentication
|
||||
services, so that the Kratos and Hydra admin APIs are only reachable from the
|
||||
containers placed on it.
|
||||
helpLink: docker
|
||||
readonly: True
|
||||
advanced: True
|
||||
global: True
|
||||
gateway:
|
||||
description: Gateway for the soauth docker network.
|
||||
helpLink: docker
|
||||
readonly: True
|
||||
advanced: True
|
||||
global: True
|
||||
manager_only:
|
||||
description: |
|
||||
Limits the soauth network to grid members running the authentication containers,
|
||||
instead of creating it on every node.
|
||||
helpLink: docker
|
||||
readonly: True
|
||||
advanced: True
|
||||
global: True
|
||||
forcedType: bool
|
||||
ulimits:
|
||||
description: |
|
||||
Default ulimit settings applied to all containers via the Docker daemon. Each entry specifies a resource name (e.g. nofile, memlock, core, nproc) with soft and hard limits. Individual container ulimits override these defaults. Valid resource names include: cpu, fsize, data, stack, core, rss, nproc, nofile, memlock, as, locks, sigpending, msgqueue, nice, rtprio, rttime.
|
||||
@@ -34,6 +68,16 @@ docker:
|
||||
readonly: True
|
||||
advanced: True
|
||||
global: True
|
||||
networks:
|
||||
description: |
|
||||
Docker networks this container is attached to. The first entry is the container's
|
||||
primary network and determines the address its published ports are forwarded to.
|
||||
Defaults to sobridge when unset.
|
||||
helpLink: docker
|
||||
readonly: True
|
||||
advanced: True
|
||||
global: True
|
||||
forcedType: "[]string"
|
||||
port_bindings:
|
||||
description: List of port bindings for the container.
|
||||
helpLink: docker
|
||||
|
||||
@@ -33,8 +33,8 @@ elastalert_sbin:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://elastalert/tools/sbin
|
||||
- user: 933
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
|
||||
#elastalert_sbin_jinja:
|
||||
|
||||
@@ -39,8 +39,8 @@ elasticagent_sbin_jinja:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://elasticagent/tools/sbin_jinja
|
||||
- user: 949
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
- template: jinja
|
||||
|
||||
|
||||
@@ -31,8 +31,8 @@ elasticfleet_sbin:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://elasticfleet/tools/sbin
|
||||
- user: 947
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
- show_changes: False
|
||||
|
||||
@@ -40,8 +40,8 @@ elasticfleet_sbin_jinja:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://elasticfleet/tools/sbin_jinja
|
||||
- user: 947
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
- template: jinja
|
||||
- exclude_pat:
|
||||
@@ -81,8 +81,8 @@ eapackageupgrade:
|
||||
file.managed:
|
||||
- name: /usr/sbin/so-elastic-fleet-package-upgrade
|
||||
- source: salt://elasticfleet/tools/sbin_jinja/so-elastic-fleet-package-upgrade
|
||||
- user: 947
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
- template: jinja
|
||||
|
||||
|
||||
@@ -14,8 +14,8 @@ so-elastic-agent-install:
|
||||
file.managed:
|
||||
- name: /usr/sbin/so-elastic-agent-install
|
||||
- source: salt://elasticfleet/tools/sbin/so-elastic-agent-install
|
||||
- user: 947
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
- show_changes: False
|
||||
|
||||
|
||||
@@ -30,6 +30,56 @@ fleet_api() {
|
||||
curl -sK /opt/so/conf/elasticsearch/curl.config -L "localhost:5601/api/fleet/${QUERYPATH}" "$@" --retry 3 --retry-delay 10 --fail 2>/dev/null
|
||||
}
|
||||
|
||||
elastic_fleet_require_agent_policy() {
|
||||
local AGENT_POLICY=$1
|
||||
local POLICY_JSON
|
||||
|
||||
if ! POLICY_JSON=$(fleet_api "agent_policies/$AGENT_POLICY") || [ -z "$POLICY_JSON" ]; then
|
||||
echo "Error: Agent policy '$AGENT_POLICY' was not found or is not visible in the current Kibana space." >&2
|
||||
return 1
|
||||
fi
|
||||
|
||||
if ! jq -e '.item.package_policies | type == "array"' <<<"$POLICY_JSON" >/dev/null 2>&1; then
|
||||
echo "Error: Agent policy '$AGENT_POLICY' was not found or is not visible in the current Kibana space." >&2
|
||||
return 1
|
||||
fi
|
||||
|
||||
echo "$POLICY_JSON"
|
||||
}
|
||||
|
||||
# Print the single active enrollment token for POLICY_ID.
|
||||
# Exit 1: retryable (API failure, invalid response, no active token)
|
||||
# Exit 2: multiple active tokens - Shouldn't get into this state without manual intervention
|
||||
elastic_fleet_active_enrollment_token() {
|
||||
local POLICY_ID=$1
|
||||
local RESP TOKEN_COUNT API_KEY
|
||||
|
||||
if ! RESP=$(fleet_api "enrollment_api_keys?perPage=100" -H 'kbn-xsrf: true' -H 'Content-Type: application/json'); then
|
||||
echo "Error: Failed to retrieve enrollment tokens for agent policy '$POLICY_ID'." >&2
|
||||
return 1
|
||||
fi
|
||||
|
||||
if ! jq -e '.list' <<<"$RESP" >/dev/null 2>&1; then
|
||||
echo "Error: Invalid enrollment token response for agent policy '$POLICY_ID'." >&2
|
||||
return 1
|
||||
fi
|
||||
|
||||
TOKEN_COUNT=$(jq --arg pid "$POLICY_ID" '[.list[] | select(.policy_id == $pid and .active == true)] | length' <<<"$RESP")
|
||||
|
||||
if [ "${TOKEN_COUNT:-0}" -eq 0 ]; then
|
||||
echo "Error: No active enrollment token found for agent policy '$POLICY_ID'." >&2
|
||||
return 1
|
||||
fi
|
||||
|
||||
if [ "$TOKEN_COUNT" -gt 1 ]; then
|
||||
echo "Error: Found $TOKEN_COUNT active enrollment tokens for agent policy '$POLICY_ID'; expected exactly one." >&2
|
||||
return 2
|
||||
fi
|
||||
|
||||
API_KEY=$(jq -r --arg pid "$POLICY_ID" '.list[] | select(.policy_id == $pid and .active == true) | .api_key' <<<"$RESP")
|
||||
echo "$API_KEY"
|
||||
}
|
||||
|
||||
# Max number of concurrent Fleet write jobs (create/update). Override via env if needed.
|
||||
MAX_FLEET_JOBS=${MAX_FLEET_JOBS:-10}
|
||||
|
||||
@@ -62,15 +112,7 @@ elastic_fleet_load_integrations_dir() {
|
||||
i=0
|
||||
|
||||
# Fetch the agent policy a single time; we look up integration ids locally below.
|
||||
if ! POLICY_JSON=$(fleet_api "agent_policies/$AGENT_POLICY"); then
|
||||
echo "Error: Failed to retrieve agent policy '$AGENT_POLICY'."
|
||||
rm -f "$FAIL_FILE"
|
||||
rm -rf "$OUT_DIR"
|
||||
return 1
|
||||
fi
|
||||
|
||||
if ! jq -e '.item.package_policies' <<<"$POLICY_JSON" >/dev/null 2>&1; then
|
||||
echo "Error: Invalid agent policy response for '$AGENT_POLICY'."
|
||||
if ! POLICY_JSON=$(elastic_fleet_require_agent_policy "$AGENT_POLICY"); then
|
||||
rm -f "$FAIL_FILE"
|
||||
rm -rf "$OUT_DIR"
|
||||
return 1
|
||||
@@ -124,9 +166,15 @@ elastic_fleet_integration_check() {
|
||||
|
||||
JSON_STRING=$2
|
||||
|
||||
NAME=$(jq -r .name $JSON_STRING)
|
||||
NAME=$(jq -r .name "$JSON_STRING")
|
||||
INTEGRATION_ID=""
|
||||
|
||||
INTEGRATION_ID=$(/usr/sbin/so-elastic-fleet-agent-policy-view "$AGENT_POLICY" | jq -r '.item.package_policies[] | select(.name=="'"$NAME"'") | .id')
|
||||
local POLICY_JSON
|
||||
if ! POLICY_JSON=$(elastic_fleet_require_agent_policy "$AGENT_POLICY"); then
|
||||
return 1
|
||||
fi
|
||||
|
||||
INTEGRATION_ID=$(jq -r --arg n "$NAME" '.item.package_policies[]? | select(.name==$n) | .id' <<<"$POLICY_JSON")
|
||||
|
||||
}
|
||||
|
||||
@@ -148,7 +196,16 @@ elastic_fleet_integration_remove() {
|
||||
|
||||
NAME=$2
|
||||
|
||||
INTEGRATION_ID=$(/usr/sbin/so-elastic-fleet-agent-policy-view "$AGENT_POLICY" | jq -r '.item.package_policies[] | select(.name=="'"$NAME"'") | .id')
|
||||
local POLICY_JSON
|
||||
if ! POLICY_JSON=$(elastic_fleet_require_agent_policy "$AGENT_POLICY"); then
|
||||
return 1
|
||||
fi
|
||||
|
||||
INTEGRATION_ID=$(jq -r --arg n "$NAME" '.item.package_policies[]? | select(.name==$n) | .id' <<<"$POLICY_JSON")
|
||||
if [ -z "$INTEGRATION_ID" ]; then
|
||||
echo "Error: Integration '$NAME' was not found in agent policy '$AGENT_POLICY'." >&2
|
||||
return 1
|
||||
fi
|
||||
|
||||
JSON_STRING=$( jq -n \
|
||||
--arg INTEGRATIONID "$INTEGRATION_ID" \
|
||||
|
||||
@@ -13,7 +13,10 @@ ERROR=false
|
||||
for INTEGRATION in /opt/so/conf/elastic-fleet/integrations/elastic-defend/*.json
|
||||
do
|
||||
printf "\n\nInitial Endpoints Policy - Loading $INTEGRATION\n"
|
||||
elastic_fleet_integration_check "endpoints-initial" "$INTEGRATION"
|
||||
if ! elastic_fleet_integration_check "endpoints-initial" "$INTEGRATION"; then
|
||||
ERROR=true
|
||||
continue
|
||||
fi
|
||||
if [ -n "$INTEGRATION_ID" ]; then
|
||||
printf "\n\nIntegration $NAME exists - Upgrading integration policy\n"
|
||||
if ! elastic_fleet_integration_policy_upgrade "$INTEGRATION_ID"; then
|
||||
|
||||
+20
-5
@@ -7,20 +7,35 @@
|
||||
. /usr/sbin/so-elastic-fleet-common
|
||||
|
||||
# Get all the fleet policies
|
||||
json_output=$(curl -s -K /opt/so/conf/elasticsearch/curl.config -L -X GET "localhost:5601/api/fleet/agent_policies" -H 'kbn-xsrf: true')
|
||||
if ! json_output=$(fleet_api "agent_policies" -H 'kbn-xsrf: true'); then
|
||||
echo "Error: Failed to retrieve Fleet agent policies." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if ! jq -e '.items' <<<"$json_output" >/dev/null 2>&1; then
|
||||
echo "Error: Invalid Fleet agent policies response." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Extract the IDs that start with "FleetServer_"
|
||||
POLICY=$(echo "$json_output" | jq -r '.items[] | select(.id | startswith("FleetServer_")) | .id')
|
||||
POLICY=$(jq -r '.items[] | select(.id | startswith("FleetServer_")) | .id' <<<"$json_output")
|
||||
|
||||
# Iterate over each ID in the POLICY variable
|
||||
for POLICYNAME in $POLICY; do
|
||||
printf "\nUpdating Policy: $POLICYNAME\n"
|
||||
|
||||
# First get the Integration ID
|
||||
INTEGRATION_ID=$(/usr/sbin/so-elastic-fleet-agent-policy-view "$POLICYNAME" | jq -r '.item.package_policies[] | select(.package.name == "fleet_server") | .id')
|
||||
if ! POLICY_JSON=$(elastic_fleet_require_agent_policy "$POLICYNAME"); then
|
||||
exit 1
|
||||
fi
|
||||
|
||||
INTEGRATION_ID=$(jq -r '.item.package_policies[]? | select(.package.name == "fleet_server") | .id' <<<"$POLICY_JSON")
|
||||
if [ -z "$INTEGRATION_ID" ]; then
|
||||
echo "Error: fleet_server integration was not found in agent policy '$POLICYNAME'." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Modify the default integration policy to update the policy_id and an with the correct naming
|
||||
UPDATED_INTEGRATION_POLICY=$(jq --arg policy_id "$POLICYNAME" --arg name "fleet_server-$POLICYNAME" '
|
||||
UPDATED_INTEGRATION_POLICY=$(jq --arg policy_id "$POLICYNAME" --arg name "fleet_server-$POLICYNAME" '
|
||||
.policy_id = $policy_id |
|
||||
.name = $name' /opt/so/conf/elastic-fleet/integrations/fleet-server/fleet-server.json)
|
||||
|
||||
|
||||
@@ -22,12 +22,19 @@ NUM_RUNNING=$(pgrep -cf "/bin/bash /sbin/so-elastic-agent-gen-installers")
|
||||
|
||||
for i in {1..30}
|
||||
do
|
||||
ENROLLMENTOKEN=$(curl -K /opt/so/conf/elasticsearch/curl.config -L "localhost:5601/api/fleet/enrollment_api_keys?perPage=100" -H 'kbn-xsrf: true' -H 'Content-Type: application/json' | jq .list | jq -r -c '.[] | select(.policy_id | contains("endpoints-initial")) | .api_key')
|
||||
ENROLLMENTOKEN=$(elastic_fleet_active_enrollment_token "endpoints-initial")
|
||||
TOKEN_RC=$?
|
||||
if [ "$TOKEN_RC" -eq 2 ]; then
|
||||
exit 1
|
||||
fi
|
||||
FLEETHOST=$(curl -K /opt/so/conf/elasticsearch/curl.config 'http://localhost:5601/api/fleet/fleet_server_hosts/grid-default' | jq -r '.item.host_urls[]' | paste -sd ',')
|
||||
if [[ $FLEETHOST ]] && [[ $ENROLLMENTOKEN ]]; then break; else sleep 10; fi
|
||||
if [[ -n "$FLEETHOST" ]] && [[ -n "$ENROLLMENTOKEN" ]]; then
|
||||
break
|
||||
fi
|
||||
sleep 10
|
||||
done
|
||||
|
||||
if [[ -z $FLEETHOST ]] || [[ -z $ENROLLMENTOKEN ]]; then
|
||||
if [[ -z "$FLEETHOST" ]] || [[ -z "$ENROLLMENTOKEN" ]]; then
|
||||
printf "\nFleet Host URL, Enrollment Token or Elastic Version empty - exiting..."
|
||||
printf "\nFleet Host: $FLEETHOST, Enrollment Token: $ENROLLMENTOKEN\n"
|
||||
exit 1
|
||||
@@ -67,19 +74,25 @@ for GOOS in "${GOTARGETOS[@]}"; do
|
||||
GOARCH="amd64"
|
||||
if [[ $GOOS == 'darwin/arm64' ]]; then GOOS="darwin" && GOARCH="arm64"; fi
|
||||
printf "\n\n### Generating $GOOS/$GOARCH Installer...\n"
|
||||
docker run -e CGO_ENABLED=0 -e GOOS=$GOOS -e GOARCH=$GOARCH \
|
||||
if ! docker run -e CGO_ENABLED=0 -e GOOS=$GOOS -e GOARCH=$GOARCH \
|
||||
--mount type=bind,source=/etc/pki/tls/certs/,target=/workspace/files/cert/ \
|
||||
--mount type=bind,source=/nsm/elastic-agent-workspace/,target=/workspace/files/elastic-agent/ \
|
||||
--mount type=bind,source=/opt/so/saltstack/local/salt/elasticfleet/files/,target=/output/ \
|
||||
{{ GLOBALS.registry_host }}:5000/{{ GLOBALS.image_repo }}/so-elastic-agent-builder:{{ GLOBALS.so_version }} go build -ldflags "-X main.fleetHostURLsList=$FLEETHOST -X main.enrollmentToken=$ENROLLMENTOKEN" -o /output/so-elastic-agent_${GOOS}_${GOARCH}
|
||||
{{ GLOBALS.registry_host }}:5000/{{ GLOBALS.image_repo }}/so-elastic-agent-builder:{{ GLOBALS.so_version }} go build -ldflags "-X main.fleetHostURLsList=$FLEETHOST -X main.enrollmentToken=$ENROLLMENTOKEN" -o /output/so-elastic-agent_${GOOS}_${GOARCH}; then
|
||||
printf "\n### ERROR: Failed to generate $GOOS/$GOARCH installer. Exiting...\n"
|
||||
exit 1
|
||||
fi
|
||||
printf "\n### $GOOS/$GOARCH Installer Generated...\n"
|
||||
done
|
||||
|
||||
printf "\n\n### Generating MSI...\n"
|
||||
cp /opt/so/saltstack/local/salt/elasticfleet/files/so-elastic-agent_windows_amd64 /opt/so/saltstack/local/salt/elasticfleet/files/so-elastic-agent_windows_amd64.exe
|
||||
docker run \
|
||||
if ! docker run \
|
||||
--mount type=bind,source=/opt/so/saltstack/local/salt/elasticfleet/files/,target=/output/ -w /output \
|
||||
{{ GLOBALS.registry_host }}:5000/{{ GLOBALS.image_repo }}/so-elastic-agent-builder:{{ GLOBALS.so_version }} wixl -o so-elastic-agent_windows_amd64_msi --arch x64 /workspace/so-elastic-agent.wxs
|
||||
{{ GLOBALS.registry_host }}:5000/{{ GLOBALS.image_repo }}/so-elastic-agent-builder:{{ GLOBALS.so_version }} wixl -o so-elastic-agent_windows_amd64_msi --arch x64 /workspace/so-elastic-agent.wxs; then
|
||||
printf "\n### ERROR: Failed to generate MSI. Exiting...\n"
|
||||
exit 1
|
||||
fi
|
||||
printf "\n### MSI Generated...\n"
|
||||
|
||||
# Verify installers were created
|
||||
|
||||
@@ -202,26 +202,9 @@ fi
|
||||
### Finalization ###
|
||||
|
||||
# Query for Enrollment Tokens for default policies
|
||||
if ENDPOINTSENROLLMENTOKEN_RAW=$(fleet_api "enrollment_api_keys" -H 'kbn-xsrf: true' -H 'Content-Type: application/json'); then
|
||||
ENDPOINTSENROLLMENTOKEN=$(echo "$ENDPOINTSENROLLMENTOKEN_RAW" | jq .list | jq -r -c '.[] | select(.policy_id | contains("endpoints-initial")) | .api_key')
|
||||
else
|
||||
echo -e "\nFailed to query for Endpoints enrollment token"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if GRIDNODESENROLLMENTOKENGENERAL_RAW=$(fleet_api "enrollment_api_keys" -H 'kbn-xsrf: true' -H 'Content-Type: application/json'); then
|
||||
GRIDNODESENROLLMENTOKENGENERAL=$(echo "$GRIDNODESENROLLMENTOKENGENERAL_RAW" | jq .list | jq -r -c '.[] | select(.policy_id | contains("so-grid-nodes_general")) | .api_key')
|
||||
else
|
||||
echo -e "\nFailed to query for Grid nodes - General enrollment token"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if GRIDNODESENROLLMENTOKENHEAVY_RAW=$(fleet_api "enrollment_api_keys" -H 'kbn-xsrf: true' -H 'Content-Type: application/json'); then
|
||||
GRIDNODESENROLLMENTOKENHEAVY=$(echo "$GRIDNODESENROLLMENTOKENHEAVY_RAW" | jq .list | jq -r -c '.[] | select(.policy_id | contains("so-grid-nodes_heavy")) | .api_key')
|
||||
else
|
||||
echo -e "\nFailed to query for Grid nodes - Heavy enrollment token"
|
||||
exit 1
|
||||
fi
|
||||
ENDPOINTSENROLLMENTOKEN=$(elastic_fleet_active_enrollment_token "endpoints-initial") || exit 1
|
||||
GRIDNODESENROLLMENTOKENGENERAL=$(elastic_fleet_active_enrollment_token "so-grid-nodes_general") || exit 1
|
||||
GRIDNODESENROLLMENTOKENHEAVY=$(elastic_fleet_active_enrollment_token "so-grid-nodes_heavy") || exit 1
|
||||
|
||||
# Store needed data in minion pillar
|
||||
pillar_file=/opt/so/saltstack/local/pillar/minions/{{ GLOBALS.minion_id }}.sls
|
||||
|
||||
@@ -37,8 +37,8 @@ elasticsearch_sbin:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://elasticsearch/tools/sbin
|
||||
- user: 930
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
- exclude_pat:
|
||||
- so-elasticsearch-pipelines # exclude this because we need to watch it for changes, we sync it in another state
|
||||
@@ -49,8 +49,8 @@ so-elasticsearch-system-indices-patch-script:
|
||||
file.managed:
|
||||
- name: /usr/sbin/so-elasticsearch-system-indices-patch
|
||||
- source: salt://elasticsearch/tools/sbin/so-elasticsearch-system-indices-patch
|
||||
- user: 930
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
- show_changes: False
|
||||
|
||||
@@ -58,8 +58,8 @@ elasticsearch_sbin_jinja:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://elasticsearch/tools/sbin_jinja
|
||||
- user: 939
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
- template: jinja
|
||||
- exclude_pat:
|
||||
@@ -72,8 +72,8 @@ so-elasticsearch-ilm-policy-load-script:
|
||||
file.managed:
|
||||
- name: /usr/sbin/so-elasticsearch-ilm-policy-load
|
||||
- source: salt://elasticsearch/tools/sbin_jinja/so-elasticsearch-ilm-policy-load
|
||||
- user: 930
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 754
|
||||
- template: jinja
|
||||
- defaults:
|
||||
@@ -84,8 +84,8 @@ so-elasticsearch-pipelines-script:
|
||||
file.managed:
|
||||
- name: /usr/sbin/so-elasticsearch-pipelines
|
||||
- source: salt://elasticsearch/tools/sbin/so-elasticsearch-pipelines
|
||||
- user: 930
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 754
|
||||
- show_changes: False
|
||||
|
||||
|
||||
@@ -5,7 +5,8 @@
|
||||
{ "rename": { "field": "message2.proto", "target_field": "network.transport", "ignore_missing": true } },
|
||||
{ "rename": { "field": "message2.app_proto", "target_field": "network.protocol", "ignore_missing": true } },
|
||||
{ "rename": { "field": "message2.fileinfo.filename", "target_field": "file.name", "ignore_missing": true } },
|
||||
{ "rename": { "field": "message2.fileinfo.gaps", "target_field": "file.bytes.missing", "ignore_missing": true } },
|
||||
{ "rename": { "field": "message2.fileinfo.gaps", "target_field": "suricata.fileinfo.gaps", "ignore_missing": true } },
|
||||
{ "set": { "if": "ctx.suricata?.fileinfo?.gaps == false", "field": "file.bytes.missing", "value": 0 } },
|
||||
{ "rename": { "field": "message2.fileinfo.magic", "target_field": "file.mime_type", "ignore_missing": true } },
|
||||
{ "rename": { "field": "message2.fileinfo.md5", "target_field": "hash.md5", "ignore_missing": true } },
|
||||
{ "rename": { "field": "message2.fileinfo.sha1", "target_field": "hash.sha1", "ignore_missing": true } },
|
||||
|
||||
@@ -4,11 +4,19 @@
|
||||
{%- set role = GLOBALS.role.split('-')[1] %}
|
||||
{%- from 'firewall/containers.map.jinja' import NODE_CONTAINERS %}
|
||||
|
||||
{%- set NODE_NETWORKS = [] %}
|
||||
{%- for NETNAME, NETWORK in DOCKERMERGED.networks.items() %}
|
||||
{%- if not NETWORK.get('manager_only') or GLOBALS.get('is_manager', False) %}
|
||||
{%- do NODE_NETWORKS.append(NETNAME) %}
|
||||
{%- endif %}
|
||||
{%- endfor %}
|
||||
|
||||
{%- set PR = [] %}
|
||||
{%- set D1 = [] %}
|
||||
{%- set D2 = [] %}
|
||||
{%- for container in NODE_CONTAINERS %}
|
||||
{%- set IP = DOCKERMERGED.containers[container].ip %}
|
||||
{%- set BRIDGE = DOCKERMERGED.containers[container].network %}
|
||||
{%- if DOCKERMERGED.containers[container].port_bindings is defined %}
|
||||
{%- for binding in DOCKERMERGED.containers[container].port_bindings %}
|
||||
{#- cant split int so we convert to string #}
|
||||
@@ -35,11 +43,11 @@
|
||||
{%- endif %}
|
||||
{%- do PR.append("-A POSTROUTING -s " ~ DOCKERMERGED.containers[container].ip ~ "/32 -d " ~ DOCKERMERGED.containers[container].ip ~ "/32 -p " ~ proto ~ " -m " ~ proto ~ " --dport " ~ containerPort ~ " -j MASQUERADE") %}
|
||||
{%- if bindip | length and bindip != '0.0.0.0' %}
|
||||
{%- do D1.append("-A DOCKER -d " ~ bindip ~ "/32 ! -i sobridge -p " ~ proto ~ " -m " ~ proto ~ " --dport " ~ hostPort ~ " -j DNAT --to-destination " ~ DOCKERMERGED.containers[container].ip ~ ":" ~ containerPort) %}
|
||||
{%- do D1.append("-A DOCKER -d " ~ bindip ~ "/32 ! -i " ~ BRIDGE ~ " -p " ~ proto ~ " -m " ~ proto ~ " --dport " ~ hostPort ~ " -j DNAT --to-destination " ~ DOCKERMERGED.containers[container].ip ~ ":" ~ containerPort) %}
|
||||
{%- else %}
|
||||
{%- do D1.append("-A DOCKER ! -i sobridge -p " ~ proto ~ " -m " ~ proto ~ " --dport " ~ hostPort ~ " -j DNAT --to-destination " ~ DOCKERMERGED.containers[container].ip ~ ":" ~ containerPort) %}
|
||||
{%- do D1.append("-A DOCKER ! -i " ~ BRIDGE ~ " -p " ~ proto ~ " -m " ~ proto ~ " --dport " ~ hostPort ~ " -j DNAT --to-destination " ~ DOCKERMERGED.containers[container].ip ~ ":" ~ containerPort) %}
|
||||
{%- endif %}
|
||||
{%- do D2.append("-A DOCKER -d " ~ DOCKERMERGED.containers[container].ip ~ "/32 ! -i sobridge -o sobridge -p " ~ proto ~ " -m " ~ proto ~ " --dport " ~ containerPort ~ " -j ACCEPT") %}
|
||||
{%- do D2.append("-A DOCKER -d " ~ DOCKERMERGED.containers[container].ip ~ "/32 ! -i " ~ BRIDGE ~ " -o " ~ BRIDGE ~ " -p " ~ proto ~ " -m " ~ proto ~ " --dport " ~ containerPort ~ " -j ACCEPT") %}
|
||||
{%- endfor %}
|
||||
{%- endif %}
|
||||
{%- endfor %}
|
||||
@@ -52,11 +60,15 @@
|
||||
:DOCKER - [0:0]
|
||||
-A PREROUTING -m addrtype --dst-type LOCAL -j DOCKER
|
||||
-A OUTPUT ! -d 127.0.0.0/8 -m addrtype --dst-type LOCAL -j DOCKER
|
||||
-A POSTROUTING -s {{DOCKERMERGED.range}} ! -o sobridge -j MASQUERADE
|
||||
{%- for NETNAME in NODE_NETWORKS %}
|
||||
-A POSTROUTING -s {{ DOCKERMERGED.networks[NETNAME].range }} ! -o {{ NETNAME }} -j MASQUERADE
|
||||
{%- endfor %}
|
||||
{%- for rule in PR %}
|
||||
{{ rule }}
|
||||
{%- endfor %}
|
||||
-A DOCKER -i sobridge -j RETURN
|
||||
{%- for NETNAME in NODE_NETWORKS %}
|
||||
-A DOCKER -i {{ NETNAME }} -j RETURN
|
||||
{%- endfor %}
|
||||
{%- for rule in D1 %}
|
||||
{{ rule }}
|
||||
{%- endfor %}
|
||||
@@ -97,10 +109,12 @@ COMMIT
|
||||
{%- endif %}
|
||||
-A FORWARD -j DOCKER-USER
|
||||
-A FORWARD -j DOCKER-ISOLATION-STAGE-1
|
||||
-A FORWARD -o sobridge -m conntrack --ctstate RELATED,ESTABLISHED -j ACCEPT
|
||||
-A FORWARD -o sobridge -j DOCKER
|
||||
-A FORWARD -i sobridge ! -o sobridge -j ACCEPT
|
||||
-A FORWARD -i sobridge -o sobridge -j ACCEPT
|
||||
{%- for NETNAME in NODE_NETWORKS %}
|
||||
-A FORWARD -o {{ NETNAME }} -m conntrack --ctstate RELATED,ESTABLISHED -j ACCEPT
|
||||
-A FORWARD -o {{ NETNAME }} -j DOCKER
|
||||
-A FORWARD -i {{ NETNAME }} ! -o {{ NETNAME }} -j ACCEPT
|
||||
-A FORWARD -i {{ NETNAME }} -o {{ NETNAME }} -j ACCEPT
|
||||
{%- endfor %}
|
||||
-A FORWARD -m conntrack --ctstate RELATED,ESTABLISHED -j ACCEPT
|
||||
-A FORWARD -i lo -j ACCEPT
|
||||
-A FORWARD -m conntrack --ctstate INVALID -j DROP
|
||||
@@ -112,13 +126,18 @@ COMMIT
|
||||
{%- for rule in D2 %}
|
||||
{{ rule }}
|
||||
{%- endfor %}
|
||||
|
||||
-A DOCKER-ISOLATION-STAGE-1 -i sobridge ! -o sobridge -j DOCKER-ISOLATION-STAGE-2
|
||||
{% for NETNAME in NODE_NETWORKS %}
|
||||
-A DOCKER-ISOLATION-STAGE-1 -i {{ NETNAME }} ! -o {{ NETNAME }} -j DOCKER-ISOLATION-STAGE-2
|
||||
{%- endfor %}
|
||||
-A DOCKER-ISOLATION-STAGE-1 -j RETURN
|
||||
-A DOCKER-ISOLATION-STAGE-2 -o sobridge -j DROP
|
||||
{%- for NETNAME in NODE_NETWORKS %}
|
||||
-A DOCKER-ISOLATION-STAGE-2 -o {{ NETNAME }} -j DROP
|
||||
{%- endfor %}
|
||||
-A DOCKER-ISOLATION-STAGE-2 -j RETURN
|
||||
-A DOCKER-USER ! -i sobridge -o sobridge -m conntrack --ctstate RELATED,ESTABLISHED -j ACCEPT
|
||||
-A DOCKER-USER ! -i sobridge -o sobridge -j LOGGING
|
||||
{%- for NETNAME in NODE_NETWORKS %}
|
||||
-A DOCKER-USER ! -i {{ NETNAME }} -o {{ NETNAME }} -m conntrack --ctstate RELATED,ESTABLISHED -j ACCEPT
|
||||
-A DOCKER-USER ! -i {{ NETNAME }} -o {{ NETNAME }} -j LOGGING
|
||||
{%- endfor %}
|
||||
-A DOCKER-USER -j RETURN
|
||||
-A LOGGING -m limit --limit 2/min -j LOG --log-prefix "IPTables-dropped: "
|
||||
-A LOGGING -j DROP
|
||||
|
||||
@@ -4,8 +4,12 @@
|
||||
|
||||
{# add our ip to self #}
|
||||
{% do FIREWALL_DEFAULT.firewall.hostgroups.self.append(GLOBALS.node_ip) %}
|
||||
{# add dockernet range #}
|
||||
{% do FIREWALL_DEFAULT.firewall.hostgroups.dockernet.append(DOCKERMERGED.range) %}
|
||||
{# add dockernet ranges #}
|
||||
{% for NETNAME, NETWORK in DOCKERMERGED.networks.items() %}
|
||||
{% if not NETWORK.get('manager_only') or GLOBALS.get('is_manager', False) %}
|
||||
{% do FIREWALL_DEFAULT.firewall.hostgroups.dockernet.append(NETWORK.range) %}
|
||||
{% endif %}
|
||||
{% endfor %}
|
||||
|
||||
{% if GLOBALS.role == 'so-idh' %}
|
||||
{% from 'idh/opencanary_config.map.jinja' import IDH_PORTGROUPS %}
|
||||
|
||||
@@ -26,8 +26,8 @@ so-hydra:
|
||||
- hostname: hydra
|
||||
- name: so-hydra
|
||||
- networks:
|
||||
- sobridge:
|
||||
- ipv4_address: {{ DOCKERMERGED.containers['so-hydra'].ip }}
|
||||
- soauth:
|
||||
- ipv4_address: {{ DOCKERMERGED.containers['so-hydra'].ips['soauth'] }}
|
||||
- binds:
|
||||
- /opt/so/conf/hydra/:/hydra-conf:ro
|
||||
- /opt/so/log/hydra/:/hydra-log:rw
|
||||
@@ -73,7 +73,7 @@ delete_so-hydra_so-status.disabled:
|
||||
|
||||
wait_for_hydra:
|
||||
http.wait_for_successful_query:
|
||||
- name: 'http://{{ GLOBALS.manager }}:4444/health/alive'
|
||||
- name: 'http://{{ DOCKERMERGED.containers['so-hydra'].ips['soauth'] }}:4444/health/alive'
|
||||
- ssl: True
|
||||
- verify_ssl: False
|
||||
- status:
|
||||
|
||||
@@ -21,12 +21,16 @@ hypervisor_sbin:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://hypervisor/tools/sbin
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 744
|
||||
|
||||
hypervisor_sbin_jinja:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://hypervisor/tools/sbin_jinja
|
||||
- user: root
|
||||
- group: root
|
||||
- template: jinja
|
||||
- file_mode: 744
|
||||
|
||||
|
||||
+2
-2
@@ -86,8 +86,8 @@ idh_sbin:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://idh/tools/sbin
|
||||
- user: 939
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
|
||||
#idh_sbin_jinja:
|
||||
|
||||
@@ -41,8 +41,8 @@ influxdb_sbin:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://influxdb/tools/sbin
|
||||
- user: 939
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
|
||||
#influxdb_sbin_jinja:
|
||||
|
||||
@@ -94,9 +94,11 @@ metrics_link_file:
|
||||
- docker_container: so-influxdb
|
||||
|
||||
# Install cron job to determine size of influxdb for telegraf
|
||||
# telegraf reads this while the cron rewrites it, so write aside and rename rather than
|
||||
# truncating in place. tgraflogdir recurses ownership, so the temp file is chowned to match
|
||||
get_influxdb_size:
|
||||
cron.present:
|
||||
- name: 'du -s -k /nsm/influxdb | cut -f1 > /opt/so/log/telegraf/influxdb_size.log 2>&1'
|
||||
- name: 'du -s -k /nsm/influxdb | cut -f1 > /opt/so/log/telegraf/influxdb_size.log.tmp 2>&1; chown 939:939 /opt/so/log/telegraf/influxdb_size.log.tmp; mv -f /opt/so/log/telegraf/influxdb_size.log.tmp /opt/so/log/telegraf/influxdb_size.log'
|
||||
- identifier: get_influxdb_size
|
||||
- user: root
|
||||
- minute: '*/1'
|
||||
|
||||
@@ -30,16 +30,16 @@ kafka_sbin_tools:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://kafka/tools/sbin
|
||||
- user: 960
|
||||
- group: 960
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
|
||||
kafka_sbin_jinja_tools:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://kafka/tools/sbin_jinja
|
||||
- user: 960
|
||||
- group: 960
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
- template: jinja
|
||||
- defaults:
|
||||
|
||||
@@ -36,16 +36,16 @@ kibana_sbin:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://kibana/tools/sbin
|
||||
- user: 932
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
|
||||
kibana_sbin_jinja:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://kibana/tools/sbin_jinja
|
||||
- user: 932
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
- template: jinja
|
||||
- defaults:
|
||||
|
||||
@@ -19,8 +19,8 @@ so-kratos:
|
||||
- hostname: kratos
|
||||
- name: so-kratos
|
||||
- networks:
|
||||
- sobridge:
|
||||
- ipv4_address: {{ DOCKERMERGED.containers['so-kratos'].ip }}
|
||||
- soauth:
|
||||
- ipv4_address: {{ DOCKERMERGED.containers['so-kratos'].ips['soauth'] }}
|
||||
- binds:
|
||||
- /opt/so/conf/kratos/:/kratos-conf:ro
|
||||
- /opt/so/log/kratos/:/kratos-log:rw
|
||||
@@ -71,7 +71,7 @@ delete_so-kratos_so-status.disabled:
|
||||
|
||||
wait_for_kratos:
|
||||
http.wait_for_successful_query:
|
||||
- name: 'http://{{ GLOBALS.manager }}:4434/'
|
||||
- name: 'http://{{ DOCKERMERGED.containers['so-kratos'].ips['soauth'] }}:4434/'
|
||||
- ssl: True
|
||||
- verify_ssl: False
|
||||
- status:
|
||||
|
||||
@@ -6,6 +6,8 @@ so-fix-salt-ldap_script:
|
||||
file.managed:
|
||||
- name: /usr/sbin/so-fix-salt-ldap.py
|
||||
- source: salt://libvirt/64962/scripts/so-fix-salt-ldap.py
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 744
|
||||
|
||||
fix-salt-ldap:
|
||||
|
||||
@@ -40,8 +40,8 @@ logstash_sbin:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://logstash/tools/sbin
|
||||
- user: 931
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
|
||||
#logstash_sbin_jinja:
|
||||
|
||||
+12
-8
@@ -113,8 +113,8 @@ manager_sbin:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://manager/tools/sbin
|
||||
- user: 939
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
- exclude_pat:
|
||||
- "*_test.py"
|
||||
@@ -124,8 +124,8 @@ manager_sbin_jinja:
|
||||
file.recurse:
|
||||
- name: /usr/sbin/
|
||||
- source: salt://manager/tools/sbin_jinja/
|
||||
- user: socore
|
||||
- group: socore
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
- template: jinja
|
||||
- show_changes: False
|
||||
@@ -166,7 +166,7 @@ so-repo-sync:
|
||||
|
||||
so_fleetagent_status:
|
||||
cron.present:
|
||||
- name: /usr/sbin/so-elasticagent-status > /opt/so/log/agents/agentstatus.log 2>&1
|
||||
- name: '/usr/sbin/so-elasticagent-status > /opt/so/log/agents/agentstatus.log.tmp 2>&1; mv -f /opt/so/log/agents/agentstatus.log.tmp /opt/so/log/agents/agentstatus.log'
|
||||
- identifier: so_fleetagent_status
|
||||
- user: root
|
||||
- minute: '*/5'
|
||||
@@ -190,11 +190,15 @@ so_fleetagent_monitor:
|
||||
- month: '*'
|
||||
- dayweek: '*'
|
||||
|
||||
socore_own_saltstack_default:
|
||||
# This tree is the source of every root-executed script (/usr/sbin, reactors, _runners,
|
||||
# engines, salt-relay.sh). SOC mounts /opt/so/saltstack rw as uid 939 but only writes
|
||||
# under local/. Do not add dir_mode/file_mode here -- SOC reads default/ and 750/640
|
||||
# would break its config load.
|
||||
root_own_saltstack_default:
|
||||
file.directory:
|
||||
- name: /opt/so/saltstack/default
|
||||
- user: socore
|
||||
- group: socore
|
||||
- user: root
|
||||
- group: root
|
||||
- recurse:
|
||||
- user
|
||||
- group
|
||||
|
||||
@@ -106,7 +106,8 @@ while [[ $# -gt 0 ]]; do
|
||||
esac
|
||||
done
|
||||
|
||||
hydraUrl=${HYDRA_URL:-http://127.0.0.1:4445}
|
||||
hydraContainer=${HYDRA_CONTAINER:-so-hydra}
|
||||
hydraUrl=${HYDRA_URL:-http://localhost:4445}
|
||||
socRolesFile=${SOC_ROLES_FILE:-/opt/so/conf/soc/soc_clients_roles}
|
||||
soUID=${SOCORE_UID:-939}
|
||||
soGID=${SOCORE_GID:-939}
|
||||
@@ -124,6 +125,10 @@ function fail() {
|
||||
exit 1
|
||||
}
|
||||
|
||||
function hydraCurl() {
|
||||
docker exec "$hydraContainer" curl "$@"
|
||||
}
|
||||
|
||||
function require() {
|
||||
cmd=$1
|
||||
which "$1" 2>&1 > /dev/null
|
||||
@@ -133,8 +138,8 @@ function require() {
|
||||
# Verify this environment is capable of running this script
|
||||
function verifyEnvironment() {
|
||||
require "jq"
|
||||
require "curl"
|
||||
response=$(curl -Ss -L ${hydraUrl}/health/alive)
|
||||
require "docker"
|
||||
response=$(hydraCurl -Ss -L ${hydraUrl}/health/alive)
|
||||
[[ "$response" != '{"status":"ok"}' ]] && fail "Unable to communicate with Hydra; specify URL via HYDRA_URL environment variable"
|
||||
}
|
||||
|
||||
@@ -164,7 +169,7 @@ function ensureRoleFileExists() {
|
||||
}
|
||||
|
||||
function listClients() {
|
||||
response=$(curl -Ss -L -f ${hydraUrl}/admin/clients)
|
||||
response=$(hydraCurl -Ss -L -f ${hydraUrl}/admin/clients)
|
||||
[[ $? != 0 ]] && fail "Unable to communicate with Hydra"
|
||||
|
||||
clientIds=$(echo "${response}" | jq -r ".[] | .client_id" | sort)
|
||||
@@ -251,7 +256,7 @@ function createClient() {
|
||||
EOF
|
||||
)
|
||||
|
||||
response=$(curl -Ss -L --fail-with-body -X POST ${hydraUrl}/admin/clients -d "$body")
|
||||
response=$(hydraCurl -Ss -L --fail-with-body -X POST ${hydraUrl}/admin/clients -d "$body")
|
||||
if [[ $? != 0 ]]; then
|
||||
error=$(echo $response | jq .error)
|
||||
fail "Failed to submit request to Hydra: $error"
|
||||
@@ -283,7 +288,7 @@ function update() {
|
||||
EOF
|
||||
)
|
||||
|
||||
response=$(curl -Ss -L --fail-with-body -X PATCH ${hydraUrl}/admin/clients/$id -d "$body")
|
||||
response=$(hydraCurl -Ss -L --fail-with-body -X PATCH ${hydraUrl}/admin/clients/$id -d "$body")
|
||||
if [[ $? != 0 ]]; then
|
||||
error=$(echo $response | jq .error)
|
||||
fail "Failed to submit request to Hydra: $error"
|
||||
@@ -305,7 +310,7 @@ function generateSecret() {
|
||||
EOF
|
||||
)
|
||||
|
||||
response=$(curl -Ss -L --fail-with-body -X PATCH ${hydraUrl}/admin/clients/$id -d "$body")
|
||||
response=$(hydraCurl -Ss -L --fail-with-body -X PATCH ${hydraUrl}/admin/clients/$id -d "$body")
|
||||
if [[ $? != 0 ]]; then
|
||||
error=$(echo $response | jq .error)
|
||||
fail "Failed to submit request to Hydra: $error"
|
||||
@@ -317,7 +322,7 @@ function deleteClient() {
|
||||
|
||||
[[ ${identityId} == "" ]] && fail "Client not found"
|
||||
|
||||
response=$(curl -Ss -XDELETE -L --fail-with-body "${hydraUrl}/admin/clients/$identityId")
|
||||
response=$(hydraCurl -Ss -XDELETE -L --fail-with-body "${hydraUrl}/admin/clients/$identityId")
|
||||
if [[ $? != 0 ]]; then
|
||||
error=$(echo $response | jq .error)
|
||||
fail "Failed to submit request to Hydra: $error"
|
||||
|
||||
@@ -121,8 +121,14 @@ for i in "$@"; do
|
||||
esac
|
||||
done
|
||||
|
||||
PILLARFILE=/opt/so/saltstack/local/pillar/minions/$MINION_ID.sls
|
||||
ADVPILLARFILE=/opt/so/saltstack/local/pillar/minions/adv_$MINION_ID.sls
|
||||
if [[ -n "$MINION_ID" && ! "$MINION_ID" =~ ^[A-Za-z0-9._-]{1,253}$ ]]; then
|
||||
echo "Invalid minion id: $MINION_ID"
|
||||
log "ERROR" "Invalid minion id: $MINION_ID"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
readonly PILLARFILE=/opt/so/saltstack/local/pillar/minions/$MINION_ID.sls
|
||||
readonly ADVPILLARFILE=/opt/so/saltstack/local/pillar/minions/adv_$MINION_ID.sls
|
||||
|
||||
function getinstallinfo() {
|
||||
log "INFO" "Getting install info for minion $MINION_ID"
|
||||
@@ -133,10 +139,23 @@ function getinstallinfo() {
|
||||
return 1
|
||||
fi
|
||||
|
||||
while read -r var; do export "$var"; done <<< "$INSTALLVARS"
|
||||
if [ $? -ne 0 ]; then
|
||||
log "ERROR" "Failed to source install variables"
|
||||
return 1
|
||||
# install.txt is controlled by the minion; only accept known keys and never eval or export them
|
||||
local line key
|
||||
while IFS= read -r line; do
|
||||
[[ "$line" == *=* ]] || continue
|
||||
key=${line%%=*}
|
||||
case "$key" in
|
||||
MAINIP|MNIC|NODE_DESCRIPTION|ES_HEAP_SIZE|PATCHSCHEDULENAME|INTERFACE|NODETYPE|CORECOUNT|LSHOSTNAME|LSHEAP|CPUCORES|IDH_MGTRESTRICT|IDH_SERVICES)
|
||||
printf -v "$key" '%s' "${line#*=}"
|
||||
;;
|
||||
*)
|
||||
log "WARN" "Ignoring unexpected install var from $MINION_ID: ${key:0:64}"
|
||||
;;
|
||||
esac
|
||||
done <<< "$INSTALLVARS"
|
||||
|
||||
if [[ "$NODE_DESCRIPTION" == \'*\' ]]; then
|
||||
NODE_DESCRIPTION=${NODE_DESCRIPTION:1:-1}
|
||||
fi
|
||||
|
||||
log "INFO" "Fetched install info for $MINION_ID (node type: ${NODETYPE:-unset})"
|
||||
@@ -176,6 +195,12 @@ function pcapspace() {
|
||||
fi
|
||||
fi
|
||||
|
||||
# Must be checked before arithmetic expansion, which evaluates array subscripts
|
||||
if [[ ! "$SPACESIZE" =~ ^[0-9]+$ ]]; then
|
||||
log "ERROR" "Invalid disk size for $MINION_ID: ${SPACESIZE:0:64}"
|
||||
return 1
|
||||
fi
|
||||
|
||||
local s=$(( $SPACESIZE / 1000000 ))
|
||||
local s1=$(( $s / 4 * $PCAP_PERCENTAGE ))
|
||||
|
||||
@@ -1050,6 +1075,57 @@ function updateMineAndApplyStates() {
|
||||
fi
|
||||
}
|
||||
|
||||
# Values end up in a Jinja-rendered pillar and in bash, and may come from the minion
|
||||
function validate_minion_vars() {
|
||||
local error_msg=""
|
||||
# Inline rather than valid_ip4: so-common is not installed yet when setup runs -o=setup
|
||||
local octet='(25[0-5]|2[0-4][0-9]|1?[0-9]?[0-9])'
|
||||
local ip4_re="^($octet\.){3}$octet$"
|
||||
|
||||
case "$NODETYPE" in
|
||||
EVAL|STANDALONE|MANAGER|MANAGERSEARCH|MANAGERHYPE|IMPORT)
|
||||
# Manager pillars also rewrite the CA pillar, so never accept them from a remote node
|
||||
[[ "$OPERATION" == "setup" ]] || error_msg="Node type $NODETYPE can only be configured during setup"
|
||||
;;
|
||||
FLEET|IDH|HEAVYNODE|SENSOR|SEARCHNODE|RECEIVER|HYPERVISOR|DESKTOP)
|
||||
;;
|
||||
*)
|
||||
error_msg="Invalid node type: ${NODETYPE:0:64}"
|
||||
;;
|
||||
esac
|
||||
|
||||
if [[ -z "$error_msg" ]]; then
|
||||
if [[ ! "$MAINIP" =~ $ip4_re ]]; then
|
||||
error_msg="Invalid MAINIP: ${MAINIP:0:64}"
|
||||
elif [[ ! "$MNIC" =~ ^[A-Za-z0-9._-]*$ ]]; then
|
||||
error_msg="Invalid MNIC: ${MNIC:0:64}"
|
||||
elif [[ ! "$INTERFACE" =~ ^[A-Za-z0-9._-]*$ ]]; then
|
||||
error_msg="Invalid INTERFACE: ${INTERFACE:0:64}"
|
||||
elif [[ ! "$LSHOSTNAME" =~ ^[A-Za-z0-9._-]*$ ]]; then
|
||||
error_msg="Invalid LSHOSTNAME: ${LSHOSTNAME:0:64}"
|
||||
elif [[ ! "$ES_HEAP_SIZE" =~ ^([0-9]+[kKmMgG]?)?$ ]]; then
|
||||
error_msg="Invalid ES_HEAP_SIZE: ${ES_HEAP_SIZE:0:64}"
|
||||
elif [[ ! "$LSHEAP" =~ ^([0-9]+[kKmMgG]?)?$ ]]; then
|
||||
error_msg="Invalid LSHEAP: ${LSHEAP:0:64}"
|
||||
elif [[ ! "$CORECOUNT" =~ ^[0-9]*$ ]]; then
|
||||
error_msg="Invalid CORECOUNT: ${CORECOUNT:0:64}"
|
||||
elif [[ ! "$CPUCORES" =~ ^[0-9]*$ ]]; then
|
||||
error_msg="Invalid CPUCORES: ${CPUCORES:0:64}"
|
||||
elif [[ ! "$IDH_MGTRESTRICT" =~ ^(True|False)?$ ]]; then
|
||||
error_msg="Invalid IDH_MGTRESTRICT: ${IDH_MGTRESTRICT:0:64}"
|
||||
fi
|
||||
fi
|
||||
|
||||
if [[ -n "$error_msg" ]]; then
|
||||
log "ERROR" "$error_msg"
|
||||
echo "$error_msg"
|
||||
return 1
|
||||
fi
|
||||
|
||||
# Free text; removing braces is enough to prevent any Jinja delimiter
|
||||
NODE_DESCRIPTION=${NODE_DESCRIPTION//[\{\}[:cntrl:]]/}
|
||||
}
|
||||
|
||||
function setupMinionFiles() {
|
||||
log "INFO" "Setting up minion files for $MINION_ID (pillar: $PILLARFILE)"
|
||||
|
||||
@@ -1061,6 +1137,8 @@ function setupMinionFiles() {
|
||||
return 1
|
||||
fi
|
||||
|
||||
validate_minion_vars || return 1
|
||||
|
||||
# Create the base minion files
|
||||
create_minion_files || return 1
|
||||
|
||||
|
||||
@@ -124,8 +124,8 @@ copy_new_files() {
|
||||
|
||||
rsync -a salt $default_salt_dir/
|
||||
rsync -a pillar $default_salt_dir/
|
||||
chown -R socore:socore $default_salt_dir/salt
|
||||
chown -R socore:socore $default_salt_dir/pillar
|
||||
chown -R root:root $default_salt_dir/salt
|
||||
chown -R root:root $default_salt_dir/pillar
|
||||
chmod 755 $default_salt_dir/pillar/firewall/addfirewall.sh
|
||||
|
||||
rm -rf /tmp/sogh
|
||||
|
||||
@@ -129,7 +129,8 @@ while [[ $# -gt 0 ]]; do
|
||||
esac
|
||||
done
|
||||
|
||||
kratosUrl=${KRATOS_URL:-http://127.0.0.1:4434/admin}
|
||||
kratosContainer=${KRATOS_CONTAINER:-so-kratos}
|
||||
kratosUrl=${KRATOS_URL:-http://localhost:4434/admin}
|
||||
databasePath=${KRATOS_DB_PATH:-/nsm/kratos/db/db.sqlite}
|
||||
databaseTimeout=${KRATOS_DB_TIMEOUT:-5000}
|
||||
bcryptRounds=${BCRYPT_ROUNDS:-12}
|
||||
@@ -154,6 +155,10 @@ function fail() {
|
||||
exit 1
|
||||
}
|
||||
|
||||
function kratosCurl() {
|
||||
docker exec "$kratosContainer" curl "$@"
|
||||
}
|
||||
|
||||
function require() {
|
||||
cmd=$1
|
||||
which "$1" 2>&1 > /dev/null
|
||||
@@ -164,18 +169,18 @@ function require() {
|
||||
function verifyEnvironment() {
|
||||
require "htpasswd"
|
||||
require "jq"
|
||||
require "curl"
|
||||
require "docker"
|
||||
require "openssl"
|
||||
require "sqlite3"
|
||||
[[ ! -f $databasePath ]] && fail "Unable to find database file; specify path via KRATOS_DB_PATH environment variable"
|
||||
response=$(curl -Ss -L ${kratosUrl}/)
|
||||
response=$(kratosCurl -Ss -L ${kratosUrl}/)
|
||||
[[ "$response" != "404 page not found" ]] && fail "Unable to communicate with Kratos; specify URL via KRATOS_URL environment variable"
|
||||
}
|
||||
|
||||
function findIdByEmail() {
|
||||
email=${1,,}
|
||||
|
||||
response=$(curl -Ss -L ${kratosUrl}/identities)
|
||||
response=$(kratosCurl -Ss -L ${kratosUrl}/identities)
|
||||
identityId=$(echo "${response}" | jq -r ".[] | select(.verifiable_addresses[0].value == \"$email\") | .id")
|
||||
echo $identityId
|
||||
}
|
||||
@@ -416,7 +421,7 @@ function syncAll() {
|
||||
}
|
||||
|
||||
function listUsers() {
|
||||
response=$(curl -Ss -L ${kratosUrl}/identities)
|
||||
response=$(kratosCurl -Ss -L ${kratosUrl}/identities)
|
||||
[[ $? != 0 ]] && fail "Unable to communicate with Kratos"
|
||||
|
||||
users=$(echo "${response}" | jq -r ".[] | .verifiable_addresses[0].value" | sort)
|
||||
@@ -495,7 +500,7 @@ function createUser() {
|
||||
EOF
|
||||
)
|
||||
|
||||
response=$(curl -Ss -L ${kratosUrl}/identities -d "$addUserJson")
|
||||
response=$(kratosCurl -Ss -L ${kratosUrl}/identities -d "$addUserJson")
|
||||
[[ $? != 0 ]] && fail "Unable to communicate with Kratos"
|
||||
|
||||
identityId=$(echo "${response}" | jq -r ".id")
|
||||
@@ -518,7 +523,7 @@ function updateStatus() {
|
||||
identityId=$(findIdByEmail "$email")
|
||||
[[ ${identityId} == "" ]] && fail "User not found"
|
||||
|
||||
response=$(curl -Ss -L "${kratosUrl}/identities/$identityId")
|
||||
response=$(kratosCurl -Ss -L "${kratosUrl}/identities/$identityId")
|
||||
[[ $? != 0 ]] && fail "Unable to communicate with Kratos"
|
||||
|
||||
schemaId=$(echo "$response" | jq -r .schema_id)
|
||||
@@ -531,7 +536,7 @@ function updateStatus() {
|
||||
state="inactive"
|
||||
fi
|
||||
body="{ \"schema_id\": \"$schemaId\", \"state\": \"$state\", \"traits\": $traitBlock }"
|
||||
response=$(curl -fSsL -XPUT -H "Content-Type: application/json" "${kratosUrl}/identities/$identityId" -d "$body")
|
||||
response=$(kratosCurl -fSsL -XPUT -H "Content-Type: application/json" "${kratosUrl}/identities/$identityId" -d "$body")
|
||||
[[ $? != 0 ]] && fail "Unable to update user"
|
||||
}
|
||||
|
||||
@@ -550,7 +555,7 @@ function updateUserProfile() {
|
||||
identityId=$(findIdByEmail "$email")
|
||||
[[ ${identityId} == "" ]] && fail "User not found"
|
||||
|
||||
response=$(curl -Ss -L "${kratosUrl}/identities/$identityId")
|
||||
response=$(kratosCurl -Ss -L "${kratosUrl}/identities/$identityId")
|
||||
[[ $? != 0 ]] && fail "Unable to communicate with Kratos"
|
||||
|
||||
schemaId=$(echo "$response" | jq -r .schema_id)
|
||||
@@ -559,7 +564,7 @@ function updateUserProfile() {
|
||||
traitBlock="{\"email\":\"$email\",\"firstName\":\"$firstName\",\"lastName\":\"$lastName\",\"note\":\"$note\"}"
|
||||
|
||||
body="{ \"schema_id\": \"$schemaId\", \"state\": \"$state\", \"traits\": $traitBlock }"
|
||||
response=$(curl -fSsL -XPUT -H "Content-Type: application/json" "${kratosUrl}/identities/$identityId" -d "$body")
|
||||
response=$(kratosCurl -fSsL -XPUT -H "Content-Type: application/json" "${kratosUrl}/identities/$identityId" -d "$body")
|
||||
[[ $? != 0 ]] && fail "Unable to update user"
|
||||
}
|
||||
|
||||
@@ -569,7 +574,7 @@ function deleteUser() {
|
||||
identityId=$(findIdByEmail "$email")
|
||||
[[ ${identityId} == "" ]] && fail "User not found"
|
||||
|
||||
response=$(curl -Ss -XDELETE -L "${kratosUrl}/identities/$identityId")
|
||||
response=$(kratosCurl -Ss -XDELETE -L "${kratosUrl}/identities/$identityId")
|
||||
[[ $? != 0 ]] && fail "Unable to communicate with Kratos"
|
||||
|
||||
rolesTmpFile="${socRolesFile}.tmp"
|
||||
|
||||
@@ -28,6 +28,7 @@ INSTALLEDSALTVERSION=$(salt --versions-report | grep Salt: | awk '{print $2}')
|
||||
# percentage like "25%"). Empty means so-soup-grid-highstate uses the salt:auto_apply:batch
|
||||
# pillar default.
|
||||
BATCHSIZE=
|
||||
DEFAULT_DOCKER_RANGE='172.17.1.0/24'
|
||||
SOUP_LOG=/root/soup.log
|
||||
SOUP_DEBUG_LOG=/root/soup-debug.log
|
||||
WHATWOULDYOUSAYYAHDOHERE=soup
|
||||
@@ -120,6 +121,9 @@ check_err() {
|
||||
161)
|
||||
echo 'Required intermediate Elasticsearch upgrade not complete'
|
||||
;;
|
||||
162)
|
||||
echo 'One or more Elastic Agent nodes do not support the x86-64-v3 CPU instruction set'
|
||||
;;
|
||||
170)
|
||||
echo "Intermediate upgrade completed successfully to $next_step_so_version, but next soup to Security Onion $originally_requested_so_version could not be started automatically."
|
||||
echo "Start soup again manually to continue the upgrade to Security Onion $originally_requested_so_version."
|
||||
@@ -347,6 +351,83 @@ check_cluster_health() {
|
||||
exit 0
|
||||
}
|
||||
|
||||
no_soup_for_you() {
|
||||
echo ""
|
||||
echo "No soup for you!"
|
||||
exit 162
|
||||
}
|
||||
|
||||
check_cpu_compatibility() {
|
||||
# Roles running a container built from the so-elastic-agent image; mirrors the
|
||||
# elasticagent and elasticfleet entries in salt/reactor/pillar_push_map.yaml.
|
||||
local cpu_target='G@role:so-heavynode or G@role:so-eval or G@role:so-fleet or G@role:so-import or G@role:so-manager or G@role:so-managerhype or G@role:so-managersearch or G@role:so-standalone'
|
||||
local expected_nodes cpu_results node result confirm
|
||||
local -a unsupported=() offline=()
|
||||
|
||||
echo "Checking that Elastic Agent nodes support the x86-64-v3 CPU instruction set now required by Elastic."
|
||||
|
||||
if [[ "$SKIP_CPU_CHECK" == "true" ]]; then
|
||||
printf "\nSkipping the x86-64-v3 CPU check because --skip-cpu-check was specified.\n\n"
|
||||
return
|
||||
fi
|
||||
|
||||
# Nodes that never answer are absent from the results, so diff against who should have.
|
||||
expected_nodes=$(salt -C "$cpu_target" --preview-target --out=json 2>/dev/null | jq -r '.[]?') || true
|
||||
if [[ -z "$expected_nodes" ]]; then
|
||||
printf "\nCould not determine which nodes run the Elastic Agent, so the x86-64-v3 CPU check cannot run.\n"
|
||||
no_soup_for_you
|
||||
fi
|
||||
|
||||
cpu_results=$(salt -t 30 -C "$cpu_target" cmd.run "/lib64/ld-linux-x86-64.so.2 --help | grep x86-64-v3" --out=json 2>/dev/null) || true
|
||||
|
||||
while IFS= read -r node; do
|
||||
[[ -z "$node" ]] && continue
|
||||
result=$(jq -r --arg node "$node" '.[$node] // empty' <<< "$cpu_results" 2>/dev/null)
|
||||
if [[ -z "$result" || "$result" == *"did not return"* ]]; then
|
||||
offline+=("$node")
|
||||
elif [[ "$result" != *"x86-64-v3 (supported"* ]]; then
|
||||
# glibc appends "(supported, searched)" only when supported; the open paren keeps
|
||||
# this from matching a future "(unsupported".
|
||||
unsupported+=("$node")
|
||||
fi
|
||||
done <<< "$expected_nodes"
|
||||
|
||||
if [[ ${#unsupported[@]} -eq 0 && ${#offline[@]} -eq 0 ]]; then
|
||||
printf "\nAll Elastic Agent nodes support x86-64-v3. We can proceed with SOUP.\n\n"
|
||||
return
|
||||
fi
|
||||
|
||||
echo ""
|
||||
if [[ ${#unsupported[@]} -gt 0 ]]; then
|
||||
echo "The following node(s) do NOT support the x86-64-v3 CPU instruction set:"
|
||||
printf ' %s\n' "${unsupported[@]}"
|
||||
echo ""
|
||||
echo "Upstream Elastic now builds its binaries for x86-64-v3, so these nodes can no"
|
||||
echo "longer run Elastic. Upgrading them WILL BREAK them."
|
||||
echo ""
|
||||
fi
|
||||
if [[ ${#offline[@]} -gt 0 ]]; then
|
||||
echo "The following node(s) did not respond and could not be checked:"
|
||||
printf ' %s\n' "${offline[@]}"
|
||||
echo ""
|
||||
echo "These nodes are offline, so we cannot confirm they support x86-64-v3, which"
|
||||
echo "upstream Elastic now requires."
|
||||
echo ""
|
||||
fi
|
||||
|
||||
if [[ -n $UNATTENDED ]]; then
|
||||
echo "Unattended mode cannot prompt for an override. Re-run soup interactively, or pass --skip-cpu-check to bypass this check."
|
||||
no_soup_for_you
|
||||
fi
|
||||
|
||||
read -rp "Type 'override' to continue anyway, or press Enter to exit: " confirm
|
||||
if [[ "${confirm,,}" == "override" ]]; then
|
||||
printf "\nOverride accepted. Continuing at your own risk.\n\n"
|
||||
else
|
||||
no_soup_for_you
|
||||
fi
|
||||
}
|
||||
|
||||
check_fleet_server() {
|
||||
echo "Checking that Elastic Fleet Server is responding."
|
||||
# Modeled on the wait_for_so-elastic-fleet state check in elasticfleet/enabled.sls,
|
||||
@@ -525,6 +606,7 @@ preupgrade_changes() {
|
||||
[[ "$INSTALLEDVERSION" == "3.0.0" ]] && up_to_3.1.0
|
||||
[[ "$INSTALLEDVERSION" == "3.1.0" ]] && up_to_3.2.0
|
||||
[[ "$INSTALLEDVERSION" == "3.2.0" ]] && up_to_3.3.0
|
||||
[[ "$INSTALLEDVERSION" == "3.3.0" ]] && up_to_3.4.0
|
||||
true
|
||||
}
|
||||
|
||||
@@ -543,6 +625,7 @@ postupgrade_changes() {
|
||||
[[ "$POSTVERSION" == "3.0.0" ]] && post_to_3.1.0
|
||||
[[ "$POSTVERSION" == "3.1.0" ]] && post_to_3.2.0
|
||||
[[ "$POSTVERSION" == "3.2.0" ]] && post_to_3.3.0
|
||||
[[ "$POSTVERSION" == "3.3.0" ]] && post_to_3.4.0
|
||||
# All applicable post-upgrade steps completed; clear the resume marker.
|
||||
rm -f "$POSTVERSION_FILE"
|
||||
true
|
||||
@@ -1093,6 +1176,82 @@ post_to_3.3.0() {
|
||||
}
|
||||
### 3.3.0 End ###
|
||||
|
||||
### 3.4.0 Scripts ###
|
||||
up_to_3.4.0() {
|
||||
set_soauth_range
|
||||
|
||||
echo "Removing so-kratos, so-hydra and so-soc so they are recreated on the soauth network."
|
||||
docker rm -f so-kratos so-hydra so-soc >> $SOUP_LOG 2>&1
|
||||
|
||||
INSTALLEDVERSION=3.4.0
|
||||
}
|
||||
|
||||
set_soauth_range() {
|
||||
local pillar_file=/opt/so/saltstack/local/pillar/docker/soc_docker.sls
|
||||
local current_range suggested authnet authgw input
|
||||
|
||||
[[ -f "$pillar_file" ]] || return 0
|
||||
|
||||
current_range=$(so-yaml.py get -r "$pillar_file" docker.range 2>/dev/null) || return 0
|
||||
|
||||
# A default range gets the 172.17.2.0/24 from docker/defaults.yaml, same as a fresh
|
||||
# install, so there is nothing to ask about.
|
||||
[[ -n "$current_range" && "$current_range" != "$DEFAULT_DOCKER_RANGE" ]] || return 0
|
||||
|
||||
if so-yaml.py get -r "$pillar_file" docker.networks.soauth.range >/dev/null 2>&1; then
|
||||
return 0
|
||||
fi
|
||||
|
||||
suggested=$(echo "${current_range%%/*}" | awk -F'.' '{ printf "%s.%s.%s.%s", $1, $2, ($3 + 1) % 256, $4 }')
|
||||
|
||||
if [[ -z $UNATTENDED ]]; then
|
||||
echo ""
|
||||
echo "This grid uses a custom Docker range ($current_range). The authentication"
|
||||
echo "services are moving to their own isolated network, which needs a second /24"
|
||||
echo "that does not overlap it."
|
||||
echo ""
|
||||
while :; do
|
||||
read -rp "Enter the network without the /24 suffix, or press Enter for ${suggested}: " input
|
||||
[[ -z "$input" ]] && input="$suggested"
|
||||
if valid_soauth_range "$input" "$current_range"; then
|
||||
authnet="$input"
|
||||
break
|
||||
fi
|
||||
echo "That range must be a valid IPv4 network, must not be within 172.17.0.0/24, and must not overlap ${current_range}."
|
||||
done
|
||||
else
|
||||
if ! valid_soauth_range "$suggested" "$current_range"; then
|
||||
FINAL_MESSAGE_QUEUE+=("WARNING: Unable to pick a range for the authentication network alongside $current_range. Set it manually before the next highstate:")
|
||||
FINAL_MESSAGE_QUEUE+=(" - so-yaml.py add $pillar_file docker.networks.soauth.range <network>/24")
|
||||
FINAL_MESSAGE_QUEUE+=(" - so-yaml.py add $pillar_file docker.networks.soauth.gateway <gateway>")
|
||||
return 0
|
||||
fi
|
||||
authnet="$suggested"
|
||||
FINAL_MESSAGE_QUEUE+=("NOTE: The authentication services moved to an isolated Docker network and were assigned ${authnet}/24.")
|
||||
FINAL_MESSAGE_QUEUE+=(" - If that conflicts with your environment, update docker.networks.soauth in $pillar_file and run so-checkin.")
|
||||
fi
|
||||
|
||||
authgw=$(echo "$authnet" | awk -F'.' '{print $1,$2,$3,1}' OFS='.')
|
||||
|
||||
echo "Assigning the authentication network the range ${authnet}/24."
|
||||
so-yaml.py add "$pillar_file" docker.networks.soauth.range "${authnet}/24" >> $SOUP_LOG 2>&1
|
||||
so-yaml.py add "$pillar_file" docker.networks.soauth.gateway "$authgw" >> $SOUP_LOG 2>&1
|
||||
}
|
||||
|
||||
valid_soauth_range() {
|
||||
local candidate=$1 docker_range=$2
|
||||
|
||||
valid_ip4 "$candidate" || return 1
|
||||
[[ $candidate =~ ^172\.17\.0\. ]] && return 1
|
||||
[[ "${candidate}/24" == "$docker_range" ]] && return 1
|
||||
return 0
|
||||
}
|
||||
|
||||
post_to_3.4.0() {
|
||||
set_postversion 3.4.0
|
||||
}
|
||||
### 3.4.0 End ###
|
||||
|
||||
|
||||
repo_sync() {
|
||||
echo "Sync the local repo."
|
||||
@@ -1977,6 +2136,9 @@ main() {
|
||||
|
||||
echo "Let's see if we need to update Security Onion."
|
||||
upgrade_check
|
||||
|
||||
check_cpu_compatibility
|
||||
|
||||
upgrade_space
|
||||
|
||||
echo "Verifying Elasticsearch version compatibility across the grid before upgrading."
|
||||
@@ -2255,6 +2417,17 @@ fi
|
||||
echo "### soup has been served at $(date) ###"
|
||||
}
|
||||
|
||||
SKIP_CPU_CHECK=false
|
||||
declare -a SOUP_ARGS=()
|
||||
for arg in "$@"; do
|
||||
if [[ "$arg" == "--skip-cpu-check" ]]; then
|
||||
SKIP_CPU_CHECK=true
|
||||
else
|
||||
SOUP_ARGS+=("$arg")
|
||||
fi
|
||||
done
|
||||
set -- "${SOUP_ARGS[@]}"
|
||||
|
||||
while getopts ":b:f:y" opt; do
|
||||
case ${opt} in
|
||||
b )
|
||||
@@ -2278,7 +2451,7 @@ while getopts ":b:f:y" opt; do
|
||||
ISOLOC="$OPTARG"
|
||||
;;
|
||||
\? )
|
||||
echo "Usage: soup [-b] [-y] [-f <iso location>]"
|
||||
echo "Usage: soup [-b] [-y] [-f <iso location>] [--skip-cpu-check]"
|
||||
exit 1
|
||||
;;
|
||||
: )
|
||||
|
||||
@@ -57,8 +57,8 @@ nginx_sbin:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://nginx/tools/sbin
|
||||
- user: 939
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
|
||||
#nginx_sbin_jinja:
|
||||
|
||||
@@ -183,7 +183,7 @@ http {
|
||||
ssl_prefer_server_ciphers on;
|
||||
ssl_protocols TLSv1.2 TLSv1.3;
|
||||
|
||||
location ~* (^/login/.*|^/js/.*|^/css/.*|^/images/.*|^/pages/.*|^/docs/.*) {
|
||||
location ~* (^/login|^/login/.*|^/js/.*|^/css/.*|^/images/.*|^/pages/.*|^/docs/.*) {
|
||||
proxy_pass http://{{ GLOBALS.manager }}:9822;
|
||||
proxy_read_timeout 90;
|
||||
proxy_connect_timeout 90;
|
||||
|
||||
@@ -50,16 +50,16 @@ redis_sbin:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://redis/tools/sbin
|
||||
- user: 939
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
|
||||
redis_sbin_jinja:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://redis/tools/sbin_jinja
|
||||
- user: 939
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
- template: jinja
|
||||
|
||||
|
||||
+4
-2
@@ -3,6 +3,8 @@ salt_bootstrap:
|
||||
file.managed:
|
||||
- name: /usr/sbin/bootstrap-salt.sh
|
||||
- source: salt://salt/scripts/bootstrap-salt.sh
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
- show_changes: False
|
||||
|
||||
@@ -10,6 +12,6 @@ salt_sbin:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://salt/tools/sbin
|
||||
- user: 939
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
|
||||
@@ -35,6 +35,8 @@ combine_bond_script:
|
||||
file.managed:
|
||||
- name: /usr/sbin/so-combine-bond
|
||||
- source: salt://sensor/tools/sbin_jinja/so-combine-bond
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
- template: jinja
|
||||
- defaults:
|
||||
|
||||
@@ -64,8 +64,8 @@ sensoroni_sbin:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://sensoroni/tools/sbin
|
||||
- user: 939
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
|
||||
#sensoroni_sbin_jinja:
|
||||
|
||||
@@ -1,2 +1,3 @@
|
||||
requests>=2.31.0
|
||||
whoisit>=2.7.0
|
||||
requests>=2.34.0
|
||||
whoisit>=4.0.5
|
||||
anyio>=4.15.1
|
||||
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
Binary file not shown.
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
BIN
Binary file not shown.
+2
-2
@@ -171,8 +171,8 @@ soc_sbin:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://soc/tools/sbin
|
||||
- user: 939
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
|
||||
#soc_sbin_jinja:
|
||||
|
||||
@@ -14,6 +14,8 @@
|
||||
{% do SOCDEFAULTS.soc.config.server.modules[module].update({'hostUrl': application_url}) %}
|
||||
{% endfor %}
|
||||
|
||||
{% do SOCDEFAULTS.soc.config.server.modules.kratos.update({'publicHostUrl': 'http://' ~ DOCKERMERGED.containers['so-kratos'].ips['soauth'] ~ ':4433/'}) %}
|
||||
|
||||
{# add all grid heavy nodes to soc.server.modules.elastic.remoteHostUrls #}
|
||||
{% for node_type, minions in salt['pillar.get']('elasticsearch:nodes', {}).items() %}
|
||||
{% if node_type in ['heavynode'] %}
|
||||
|
||||
+69
-1
@@ -1380,6 +1380,7 @@ soc:
|
||||
retryFailureMaxAttempts: 5
|
||||
kratos:
|
||||
hostUrl:
|
||||
publicHostUrl:
|
||||
hydra:
|
||||
hostUrl:
|
||||
elastalertengine:
|
||||
@@ -1465,6 +1466,7 @@ soc:
|
||||
- core
|
||||
- emerging_threats_addon
|
||||
useEsql: false
|
||||
esqlCaseInsensitive: true
|
||||
elastic:
|
||||
hostUrl:
|
||||
remoteHostUrls: []
|
||||
@@ -1492,6 +1494,9 @@ soc:
|
||||
org: Security Onion
|
||||
bucket: telegraf/so_short_term
|
||||
verifyCert: false
|
||||
notification:
|
||||
dismissedPruneDays: 30
|
||||
enabled: false
|
||||
playbook:
|
||||
autoUpdateEnabled: true
|
||||
playbookImportFrequencySeconds: 86400
|
||||
@@ -1537,7 +1542,7 @@ soc:
|
||||
Orchestrator: sonnet@SOAI
|
||||
Investigator: gemma@SOAI
|
||||
DetectionEngineer: gemma@SOAI
|
||||
useMemory: true
|
||||
useMemory: false
|
||||
useMemoryScanner: false
|
||||
dontScanBefore: ""
|
||||
memoryScanIntervalSeconds: 300
|
||||
@@ -1556,6 +1561,69 @@ soc:
|
||||
reconcilePersona: ""
|
||||
toolUseTurnAttempts: 12
|
||||
toolUseTurnDelayMs: 175
|
||||
tools:
|
||||
filterEventFields:
|
||||
- "@timestamp"
|
||||
- "client.name"
|
||||
- "destination.ip"
|
||||
- "destination.port"
|
||||
- "destination.geo.country_name"
|
||||
- "dns.query.name"
|
||||
- "dns.query_name"
|
||||
- "event.action"
|
||||
- "event.category"
|
||||
- "event.module"
|
||||
- "event.dataset"
|
||||
- "event.outcome"
|
||||
- "event.severity"
|
||||
- "event.severity_label"
|
||||
- "event.type"
|
||||
- "event_data.agent.name"
|
||||
- "event_data.host.os.name"
|
||||
- "file.mime_type"
|
||||
- "file.name"
|
||||
- "hash.md5"
|
||||
- "hash.sha1"
|
||||
- "host.mac"
|
||||
- "host.name"
|
||||
- "host.os.name"
|
||||
- "http.method"
|
||||
- "http.useragent"
|
||||
- "http.virtual_host"
|
||||
- "log.id.uid"
|
||||
- "network.community_id"
|
||||
- "network.protocol"
|
||||
- "network.transport"
|
||||
- "notice.message"
|
||||
- "observer.name"
|
||||
- "process.name"
|
||||
- "process.executable"
|
||||
- "process.entity_id"
|
||||
- "process.command_line"
|
||||
- "process.Ext.ancestry"
|
||||
- "process.parent.entity_id"
|
||||
- "process.parent.command_line"
|
||||
- "rule.category"
|
||||
- "rule.name"
|
||||
- "rule.uuid"
|
||||
- "software.name"
|
||||
- "software.type"
|
||||
- "software.version.unparsed"
|
||||
- "source.ip"
|
||||
- "source.port"
|
||||
- "source.geo.country_name"
|
||||
- "ssh.cypher_algorithm"
|
||||
- "ssh.client"
|
||||
- "ssh.server"
|
||||
- "ssl.cipher"
|
||||
- "ssl.server_name"
|
||||
- "ssl.version"
|
||||
- "system.auth.sudo.command"
|
||||
- "user.name"
|
||||
- "user.domain"
|
||||
- "user.effective.name"
|
||||
- "weird.name"
|
||||
- "tags"
|
||||
onionconfig:
|
||||
saltstackDir: /opt/so/saltstack
|
||||
bypassEnabled: false
|
||||
|
||||
@@ -18,8 +18,8 @@ hypervisor_annotation:
|
||||
- name: /opt/so/saltstack/default/salt/hypervisor/soc_hypervisor.yaml
|
||||
- source: salt://soc/dyanno/hypervisor/soc_hypervisor.yaml.jinja
|
||||
- template: jinja
|
||||
- user: socore
|
||||
- group: socore
|
||||
- user: root
|
||||
- group: root
|
||||
- defaults:
|
||||
HYPERVISORS: {{ HYPERVISORS }}
|
||||
baseDomainStatus: {{ salt['pillar.get']('baseDomain:status', 'Initialized') }}
|
||||
|
||||
@@ -23,7 +23,9 @@ so-soc:
|
||||
- name: so-soc
|
||||
- networks:
|
||||
- sobridge:
|
||||
- ipv4_address: {{ DOCKERMERGED.containers['so-soc'].ip }}
|
||||
- ipv4_address: {{ DOCKERMERGED.containers['so-soc'].ips['sobridge'] }}
|
||||
- soauth:
|
||||
- ipv4_address: {{ DOCKERMERGED.containers['so-soc'].ips['soauth'] }}
|
||||
- binds:
|
||||
- /nsm/rules:/nsm/rules:rw
|
||||
- /opt/so/conf/strelka:/opt/sensoroni/yara:rw
|
||||
|
||||
@@ -1,6 +1,31 @@
|
||||
name: Security Onion Baseline Pipeline
|
||||
priority: 90
|
||||
transformations:
|
||||
# ES|QL scalar == returns null on multivalued fields; the
|
||||
# backend reads this key and emits MV_INTERSECTS instead.
|
||||
- id: declare_multivalue_fields
|
||||
type: set_state
|
||||
key: multivalue_fields
|
||||
val:
|
||||
- event.type
|
||||
- event.action
|
||||
- event.category
|
||||
- tags
|
||||
- process.args
|
||||
- related.ip
|
||||
- dns.resolved_ip
|
||||
- id: esql_default_index
|
||||
type: set_state
|
||||
key: index
|
||||
val: .ds-logs-*
|
||||
- id: esql_source_metadata
|
||||
type: set_state
|
||||
key: metadata
|
||||
val: "_id, _index, _source"
|
||||
- id: esql_source_keep
|
||||
type: set_state
|
||||
key: keep
|
||||
val: "_id, _index, _source"
|
||||
- id: baseline_field_name_mapping
|
||||
type: field_name_mapping
|
||||
mapping:
|
||||
|
||||
@@ -155,6 +155,14 @@ soc:
|
||||
description: Path to custom markdown templates for PDF report generation. All markdown files in this directory will be available as custom reports in the SOC Reports interface.
|
||||
global: True
|
||||
advanced: True
|
||||
schedules:
|
||||
title: Schedules
|
||||
description: Schedules that are shared across the Security Onion product. Modify via one of the SOC Schedules view.
|
||||
readonlyUi: True
|
||||
global: True
|
||||
forcedType: string
|
||||
syntax: json
|
||||
storage: db
|
||||
subgrids:
|
||||
title: Subordinate Grids
|
||||
description: |
|
||||
@@ -396,6 +404,11 @@ soc:
|
||||
global: True
|
||||
advanced: True
|
||||
forcedType: bool
|
||||
esqlCaseInsensitive:
|
||||
description: "Match string values case-insensitively when converting Sigma rules. Applies to ES|QL only"
|
||||
global: True
|
||||
advanced: True
|
||||
forcedType: bool
|
||||
elastic:
|
||||
index:
|
||||
description: Comma-separated list of indices or index patterns (wildcard "*" supported) that SOC will search for records.
|
||||
@@ -476,6 +489,24 @@ soc:
|
||||
global: True
|
||||
advanced: True
|
||||
forcedType: bool
|
||||
notification:
|
||||
destinations:
|
||||
title: Notification Destinations
|
||||
description: JSON list of notifications. Modify via the SOC Notifications view.
|
||||
readonlyUi: True
|
||||
global: True
|
||||
forcedType: string
|
||||
syntax: json
|
||||
storage: db
|
||||
dismissedPruneDays:
|
||||
title: Dismissed Retention Days
|
||||
description: The number of days to retain dismissed notifications. When a notification is dismissed, it will be pruned after this many days. Only one user need dismiss a notification for it to be pruned.
|
||||
forcedType: int
|
||||
global: True
|
||||
enabled:
|
||||
description: Enables or disables the SOC notification module.
|
||||
forcedType: bool
|
||||
global: True
|
||||
postgres:
|
||||
host:
|
||||
description: Hostname or IP address of the PostgreSQL server used by SOC. Defaults to the manager hostname.
|
||||
@@ -916,6 +947,11 @@ soc:
|
||||
description: The number of times to retry extracting memories from a session if errors occur.
|
||||
global: True
|
||||
advanced: True
|
||||
tools:
|
||||
filterEventFields:
|
||||
description: A whitelist of fields to return when OnionAI uses the query_events tool. All other fields are removed. One field per line.
|
||||
global: True
|
||||
multiline: True
|
||||
client:
|
||||
assistant:
|
||||
enabled:
|
||||
|
||||
@@ -51,8 +51,8 @@ strelka_sbin:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://strelka/tools/sbin
|
||||
- user: 939
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
|
||||
{% else %}
|
||||
|
||||
@@ -76,16 +76,16 @@ suricata_sbin:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://suricata/tools/sbin
|
||||
- user: 939
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
|
||||
suricata_sbin_jinja:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://suricata/tools/sbin_jinja
|
||||
- user: 939
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
- template: jinja
|
||||
|
||||
|
||||
+92
-12
@@ -36,7 +36,7 @@ tgraf_sync_script_{{script}}:
|
||||
- name: /opt/so/conf/telegraf/scripts/{{script}}
|
||||
- user: root
|
||||
- group: 939
|
||||
- mode: 770
|
||||
- mode: 750
|
||||
- template: jinja
|
||||
- source: salt://telegraf/scripts/{{script}}
|
||||
- defaults:
|
||||
@@ -49,7 +49,7 @@ tgraf_sync_script_esindexsize.sh:
|
||||
- name: /opt/so/conf/telegraf/scripts/esindexsize.sh
|
||||
- user: root
|
||||
- group: 939
|
||||
- mode: 770
|
||||
- mode: 750
|
||||
- source: salt://telegraf/scripts/esindexsize.sh
|
||||
{# Copy conf/elasticsearch/curl.config for telegraf to use with esindexsize.sh #}
|
||||
tgraf_sync_escurl_conf:
|
||||
@@ -61,22 +61,102 @@ tgraf_sync_escurl_conf:
|
||||
- source: salt://elasticsearch/curl.config
|
||||
{% endif %}
|
||||
|
||||
# so-container-stats runs on the host as somon, a docker group member, so the container does
|
||||
# not need the docker socket
|
||||
somongroup:
|
||||
group.present:
|
||||
- name: somon
|
||||
- gid: 961
|
||||
|
||||
# cron chdirs to $HOME before running a job, so home must exist
|
||||
somon:
|
||||
user.present:
|
||||
- uid: 961
|
||||
- gid: 961
|
||||
- home: /opt/so/log/somon
|
||||
- createhome: False
|
||||
- shell: /sbin/nologin
|
||||
- groups:
|
||||
- docker
|
||||
# renumbering an existing somon is a no-op on a fresh host and lets a host created
|
||||
# before the id changed converge instead of failing the whole telegraf state
|
||||
- allow_uid_change: True
|
||||
- allow_gid_change: True
|
||||
- require:
|
||||
- group: somongroup
|
||||
|
||||
somonlogdir:
|
||||
file.directory:
|
||||
- name: /opt/so/log/somon
|
||||
- user: 961
|
||||
- group: 961
|
||||
- mode: 755
|
||||
# the lock file is not otherwise managed; recurse so a renumber rechowns it too
|
||||
- recurse:
|
||||
- user
|
||||
- group
|
||||
- require:
|
||||
- user: somon
|
||||
|
||||
containers_log:
|
||||
file.managed:
|
||||
- name: /opt/so/log/somon/containers.log
|
||||
- user: 961
|
||||
- group: 961
|
||||
- mode: 644
|
||||
- replace: False
|
||||
- require:
|
||||
- file: somonlogdir
|
||||
|
||||
# telegraf reads on the same minute boundary the collector runs, and docker stats takes
|
||||
# seconds, so write aside and rename rather than truncating the file telegraf is reading.
|
||||
# ; not && so a failed run replaces the file instead of leaving stale metrics behind.
|
||||
# flock -n keeps a run that outlives its minute from racing the next one over the same tmp
|
||||
# file; the skipped run leaves a stale containers.log, which containers.sh discards by age
|
||||
so-container-stats_cron:
|
||||
cron.present:
|
||||
- name: "flock -n /opt/so/log/somon/containers.lock -c '/usr/sbin/so-container-stats > /opt/so/log/somon/containers.log.tmp 2>&1; mv -f /opt/so/log/somon/containers.log.tmp /opt/so/log/somon/containers.log'"
|
||||
- identifier: so-container-stats_cron
|
||||
- user: somon
|
||||
- minute: '*/1'
|
||||
- hour: '*'
|
||||
- daymonth: '*'
|
||||
- month: '*'
|
||||
- dayweek: '*'
|
||||
- require:
|
||||
- user: somon
|
||||
|
||||
# salt.lasthighstate touches this at order 9001, after the container starts; pre-create it so
|
||||
# docker does not create a directory at the bind mount source
|
||||
lasthighstate_placeholder:
|
||||
file.managed:
|
||||
- name: /opt/so/log/salt/lasthighstate
|
||||
- mode: 644
|
||||
- replace: False
|
||||
- makedirs: True
|
||||
|
||||
telegraf_sbin:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://telegraf/tools/sbin
|
||||
- user: 939
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
|
||||
#telegraf_sbin_jinja:
|
||||
# file.recurse:
|
||||
# - name: /usr/sbin
|
||||
# - source: salt://telegraf/tools/sbin_jinja
|
||||
# - user: 939
|
||||
# - group: 939
|
||||
# - file_mode: 755
|
||||
# - template: jinja
|
||||
# so-container-stats needs the per-stat toggles, so it renders instead of copying
|
||||
tgraf_sbin_jinja:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://telegraf/tools/sbin_jinja
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
# the unit test lives beside the script; it must not ship or be rendered as jinja
|
||||
- exclude_pat:
|
||||
- "*_test.py"
|
||||
- template: jinja
|
||||
- defaults:
|
||||
CONTAINER_STATS: {{ TELEGRAFMERGED.container_stats }}
|
||||
|
||||
tgrafconf:
|
||||
file.managed:
|
||||
|
||||
@@ -10,6 +10,69 @@ telegraf:
|
||||
flush_jitter: '0s'
|
||||
debug: false
|
||||
quiet: false
|
||||
container_stats:
|
||||
tags:
|
||||
identity: False
|
||||
engine:
|
||||
n_containers: False
|
||||
n_containers_running: False
|
||||
n_containers_stopped: False
|
||||
n_containers_paused: False
|
||||
n_images: False
|
||||
n_cpus: False
|
||||
n_goroutines: False
|
||||
n_used_file_descriptors: False
|
||||
n_listener_events: False
|
||||
memory_total: False
|
||||
cpu:
|
||||
usage_percent: True
|
||||
usage_total: False
|
||||
usage_in_usermode: False
|
||||
usage_in_kernelmode: False
|
||||
usage_system: False
|
||||
throttling_periods: False
|
||||
throttling_throttled_periods: False
|
||||
throttling_throttled_time: False
|
||||
container_id: False
|
||||
mem:
|
||||
usage_percent: True
|
||||
usage: False
|
||||
limit: False
|
||||
max_usage: False
|
||||
active_anon: False
|
||||
active_file: False
|
||||
inactive_anon: False
|
||||
inactive_file: False
|
||||
unevictable: False
|
||||
pgfault: False
|
||||
pgmajfault: False
|
||||
container_id: False
|
||||
net:
|
||||
rx_bytes: True
|
||||
rx_packets: False
|
||||
rx_errors: False
|
||||
rx_dropped: False
|
||||
tx_bytes: False
|
||||
tx_packets: False
|
||||
tx_errors: False
|
||||
tx_dropped: False
|
||||
container_id: False
|
||||
blkio:
|
||||
io_service_bytes_recursive_read: False
|
||||
io_service_bytes_recursive_write: False
|
||||
container_id: False
|
||||
status:
|
||||
uptime_ns: True
|
||||
oomkilled: True
|
||||
pid: False
|
||||
exitcode: False
|
||||
restart_count: False
|
||||
started_at: False
|
||||
finished_at: False
|
||||
container_id: False
|
||||
health:
|
||||
health_status: False
|
||||
failing_streak: False
|
||||
scripts:
|
||||
eval:
|
||||
- agentstatus.sh
|
||||
@@ -19,6 +82,7 @@ telegraf:
|
||||
- oldpcap.sh
|
||||
- os.sh
|
||||
- raid.sh
|
||||
- containers.sh
|
||||
- sostatus.sh
|
||||
- suriloss.sh
|
||||
- surirules.sh
|
||||
@@ -34,6 +98,7 @@ telegraf:
|
||||
- os.sh
|
||||
- raid.sh
|
||||
- redis.sh
|
||||
- containers.sh
|
||||
- sostatus.sh
|
||||
- suriloss.sh
|
||||
- surirules.sh
|
||||
@@ -47,6 +112,7 @@ telegraf:
|
||||
- os.sh
|
||||
- raid.sh
|
||||
- redis.sh
|
||||
- containers.sh
|
||||
- sostatus.sh
|
||||
- features.sh
|
||||
managerhype:
|
||||
@@ -56,6 +122,7 @@ telegraf:
|
||||
- os.sh
|
||||
- raid.sh
|
||||
- redis.sh
|
||||
- containers.sh
|
||||
- sostatus.sh
|
||||
- features.sh
|
||||
managersearch:
|
||||
@@ -66,12 +133,14 @@ telegraf:
|
||||
- os.sh
|
||||
- raid.sh
|
||||
- redis.sh
|
||||
- containers.sh
|
||||
- sostatus.sh
|
||||
- features.sh
|
||||
import:
|
||||
- influxdbsize.sh
|
||||
- lasthighstate.sh
|
||||
- os.sh
|
||||
- containers.sh
|
||||
- sostatus.sh
|
||||
sensor:
|
||||
- checkfiles.sh
|
||||
@@ -79,6 +148,7 @@ telegraf:
|
||||
- oldpcap.sh
|
||||
- os.sh
|
||||
- raid.sh
|
||||
- containers.sh
|
||||
- sostatus.sh
|
||||
- suriloss.sh
|
||||
- surirules.sh
|
||||
@@ -93,6 +163,7 @@ telegraf:
|
||||
- os.sh
|
||||
- raid.sh
|
||||
- redis.sh
|
||||
- containers.sh
|
||||
- sostatus.sh
|
||||
- suriloss.sh
|
||||
- surirules.sh
|
||||
@@ -101,12 +172,14 @@ telegraf:
|
||||
idh:
|
||||
- lasthighstate.sh
|
||||
- os.sh
|
||||
- containers.sh
|
||||
- sostatus.sh
|
||||
searchnode:
|
||||
- eps.sh
|
||||
- lasthighstate.sh
|
||||
- os.sh
|
||||
- raid.sh
|
||||
- containers.sh
|
||||
- sostatus.sh
|
||||
- features.sh
|
||||
receiver:
|
||||
@@ -115,16 +188,20 @@ telegraf:
|
||||
- os.sh
|
||||
- raid.sh
|
||||
- redis.sh
|
||||
- containers.sh
|
||||
- sostatus.sh
|
||||
fleet:
|
||||
- lasthighstate.sh
|
||||
- os.sh
|
||||
- containers.sh
|
||||
- sostatus.sh
|
||||
hypervisor:
|
||||
- lasthighstate.sh
|
||||
- os.sh
|
||||
- containers.sh
|
||||
- sostatus.sh
|
||||
desktop:
|
||||
- lasthighstate.sh
|
||||
- os.sh
|
||||
- containers.sh
|
||||
- sostatus.sh
|
||||
|
||||
@@ -13,6 +13,11 @@ so-telegraf:
|
||||
docker_container.absent:
|
||||
- force: True
|
||||
|
||||
so-container-stats_cron:
|
||||
cron.absent:
|
||||
- identifier: so-container-stats_cron
|
||||
- user: somon
|
||||
|
||||
so-telegraf_so-status.disabled:
|
||||
file.comment:
|
||||
- name: /opt/so/conf/so-status/so-status.conf
|
||||
|
||||
@@ -19,8 +19,7 @@ so-telegraf:
|
||||
docker_container.running:
|
||||
- image: {{ GLOBALS.registry_host }}:5000/{{ GLOBALS.image_repo }}/so-telegraf:{{ GLOBALS.so_version }}
|
||||
- restart_policy: unless-stopped
|
||||
- user: 939
|
||||
- group_add: 939,920
|
||||
- user: 939:939
|
||||
- environment:
|
||||
- HOST_ETC=/host/etc
|
||||
- HOST_SYS=/host/sys
|
||||
@@ -38,7 +37,6 @@ so-telegraf:
|
||||
- /opt/so/conf/telegraf/etc/telegraf.conf:/etc/telegraf/telegraf.conf:ro
|
||||
- /opt/so/conf/telegraf/node_config.json:/etc/telegraf/node_config.json:ro
|
||||
- /var/run/utmp:/var/run/utmp:ro
|
||||
- /var/run/docker.sock:/var/run/docker.sock:ro
|
||||
- /:/host:ro
|
||||
- /sys:/host/sys:ro
|
||||
- /proc:/host/proc:ro
|
||||
@@ -51,7 +49,8 @@ so-telegraf:
|
||||
- /opt/so/log/suricata:/var/log/suricata:ro
|
||||
- /opt/so/log/raid:/var/log/raid:ro
|
||||
- /opt/so/log/sostatus:/var/log/sostatus:ro
|
||||
- /opt/so/log/salt:/var/log/salt:ro
|
||||
- /opt/so/log/somon:/var/log/somon:ro
|
||||
- /opt/so/log/salt/lasthighstate:/var/log/salt/lasthighstate:ro
|
||||
- /opt/so/log/agents:/var/log/agents:ro
|
||||
{% if GLOBALS.is_manager or GLOBALS.role == 'so-heavynode' %}
|
||||
- /opt/so/conf/telegraf/etc/escurl.config:/etc/telegraf/elasticsearch.config:ro
|
||||
@@ -74,6 +73,8 @@ so-telegraf:
|
||||
{% endfor %}
|
||||
{% endif %}
|
||||
- watch:
|
||||
- file: tgraf_sbin_jinja
|
||||
- file: lasthighstate_placeholder
|
||||
- file: trusttheca
|
||||
- x509: telegraf_crt
|
||||
- x509: telegraf_key
|
||||
@@ -83,6 +84,8 @@ so-telegraf:
|
||||
- file: tgraf_sync_script_{{script}}
|
||||
{% endfor %}
|
||||
- require:
|
||||
- file: lasthighstate_placeholder
|
||||
- file: somonlogdir
|
||||
- file: trusttheca
|
||||
- x509: telegraf_crt
|
||||
- x509: telegraf_key
|
||||
|
||||
@@ -228,13 +228,6 @@
|
||||
# ## bond interfaces.
|
||||
# # bond_interfaces = ["bond0"]
|
||||
|
||||
# # Read metrics about docker containers
|
||||
[[inputs.docker]]
|
||||
# ## Docker Endpoint
|
||||
# ## To use TCP, set endpoint = "tcp://[ip]:[port]"
|
||||
# ## To use environment variables (ie, docker-machine), set endpoint = "ENV"
|
||||
endpoint = "unix:///var/run/docker.sock"
|
||||
#
|
||||
|
||||
# # Read stats from one or more Elasticsearch servers or clusters
|
||||
{%- if GLOBALS.is_manager or GLOBALS.role == 'so-heavynode' %}
|
||||
@@ -342,6 +335,17 @@
|
||||
interval = "60s"
|
||||
{%- endif %}
|
||||
|
||||
{%- if 'containers.sh' in TELEGRAFMERGED.scripts[GLOBALS.role.split('-')[1]] %}
|
||||
{%- do TELEGRAFMERGED.scripts[GLOBALS.role.split('-')[1]].remove('containers.sh') %}
|
||||
[[inputs.exec]]
|
||||
commands = [
|
||||
["/scripts/containers.sh"]
|
||||
]
|
||||
data_format = "influx"
|
||||
timeout = "15s"
|
||||
interval = "60s"
|
||||
{%- endif %}
|
||||
|
||||
{%- if TELEGRAFMERGED.scripts[GLOBALS.role.split('-')[1]] | length > 0 %}
|
||||
[[inputs.exec]]
|
||||
commands = [
|
||||
|
||||
@@ -0,0 +1,29 @@
|
||||
#!/bin/bash
|
||||
#
|
||||
# Copyright Security Onion Solutions LLC and/or licensed to Security Onion Solutions LLC under one
|
||||
# or more contributor license agreements. Licensed under the Elastic License 2.0 as shown at
|
||||
# https://securityonion.net/license; you may not use this file except in compliance with the
|
||||
# Elastic License 2.0.
|
||||
|
||||
|
||||
|
||||
# if this script isn't already running
|
||||
if [[ ! "`pidof -x $(basename $0) -o %PPID`" ]]; then
|
||||
|
||||
CONTAINERSLOG=/var/log/somon/containers.log
|
||||
# the collector rewrites this every minute; report nothing rather than repeating a stale
|
||||
# file as if it were current, in case a run was skipped or the collector is wedged
|
||||
MAXAGE=150
|
||||
|
||||
if [ -r "$CONTAINERSLOG" ]; then
|
||||
AGE=$(( $(date +%s) - $(stat -c %Y "$CONTAINERSLOG") ))
|
||||
if [ "$AGE" -le "$MAXAGE" ]; then
|
||||
cat $CONTAINERSLOG
|
||||
fi
|
||||
fi
|
||||
|
||||
exit 0
|
||||
|
||||
fi
|
||||
|
||||
exit 0
|
||||
@@ -8,10 +8,12 @@
|
||||
# if this script isn't already running
|
||||
if [[ ! "`pidof -x $(basename $0) -o %PPID`" ]]; then
|
||||
|
||||
LAST_HIGHSTATE_END=$([ -e "/var/log/salt/lasthighstate" ] && date -r /var/log/salt/lasthighstate +%s || echo 0)
|
||||
NOW=$(date +%s)
|
||||
HIGHSTATE_AGE_SECONDS=$((NOW-LAST_HIGHSTATE_END))
|
||||
echo "salt highstate_age_seconds=$HIGHSTATE_AGE_SECONDS"
|
||||
if [ -r "/var/log/salt/lasthighstate" ]; then
|
||||
LAST_HIGHSTATE_END=$(date -r /var/log/salt/lasthighstate +%s)
|
||||
NOW=$(date +%s)
|
||||
HIGHSTATE_AGE_SECONDS=$((NOW-LAST_HIGHSTATE_END))
|
||||
echo "salt highstate_age_seconds=$HIGHSTATE_AGE_SECONDS"
|
||||
fi
|
||||
|
||||
fi
|
||||
|
||||
|
||||
@@ -53,6 +53,319 @@ telegraf:
|
||||
forcedType: bool
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
container_stats:
|
||||
tags:
|
||||
identity:
|
||||
description: Adds the container_image, container_version, engine_host and server_version tags to every container measurement, as the retired inputs.docker plugin did. Increases series cardinality. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
engine:
|
||||
n_containers:
|
||||
description: Total number of containers known to the Docker engine, running or not. Part of the engine-level docker measurement. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
n_containers_running:
|
||||
description: Number of containers currently running. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
n_containers_stopped:
|
||||
description: Number of containers currently stopped. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
n_containers_paused:
|
||||
description: Number of containers currently paused. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
n_images:
|
||||
description: Number of container images held by the Docker engine. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
n_cpus:
|
||||
description: Number of CPUs the Docker engine reports for this host. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
n_goroutines:
|
||||
description: Number of goroutines inside the Docker daemon. Diagnostic detail for the daemon itself. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
n_used_file_descriptors:
|
||||
description: Number of file descriptors held open by the Docker daemon. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
n_listener_events:
|
||||
description: Number of event listeners subscribed to the Docker daemon. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
memory_total:
|
||||
description: Total physical memory the Docker engine reports for this host, in bytes. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
cpu:
|
||||
usage_percent:
|
||||
description: Percentage of host CPU consumed by the container. Required by the Container CPU Usage chart on the Security Onion Performance dashboard.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
usage_total:
|
||||
description: Cumulative CPU time consumed by the container, in nanoseconds. Read from the cgroup. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
usage_in_usermode:
|
||||
description: Cumulative CPU time consumed in user mode, in nanoseconds. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
usage_in_kernelmode:
|
||||
description: Cumulative CPU time consumed in kernel mode, in nanoseconds. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
usage_system:
|
||||
description: Cumulative host-wide CPU time, in nanoseconds. Used as the denominator when calculating container CPU percentage. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
throttling_periods:
|
||||
description: Number of CPU enforcement periods the container has seen. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
throttling_throttled_periods:
|
||||
description: Number of periods in which the container was throttled against its CPU limit. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
throttling_throttled_time:
|
||||
description: Total time the container spent throttled, in nanoseconds. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
container_id: &containerid
|
||||
description: Full 64 character container ID, emitted as a field on this measurement. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
mem:
|
||||
usage_percent:
|
||||
description: Percentage of its memory limit the container is using. Required by the Container Memory Usage chart on the Security Onion Performance dashboard.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
usage:
|
||||
description: Container memory usage in bytes, excluding reclaimable page cache. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
limit:
|
||||
description: Memory limit for the container in bytes. Reports total host memory when the container is unlimited. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
max_usage:
|
||||
description: Peak memory usage for the container in bytes, read from the cgroup peak counter. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
active_anon:
|
||||
description: Anonymous memory on the active LRU list, in bytes. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
active_file:
|
||||
description: Page cache on the active LRU list, in bytes. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
inactive_anon:
|
||||
description: Anonymous memory on the inactive LRU list, in bytes. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
inactive_file:
|
||||
description: Page cache on the inactive LRU list, in bytes. This is the reclaimable cache subtracted from usage. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
unevictable:
|
||||
description: Memory that cannot be reclaimed, in bytes. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
pgfault:
|
||||
description: Cumulative number of page faults taken by the container. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
pgmajfault:
|
||||
description: Cumulative number of major page faults, those requiring disk access. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
container_id: *containerid
|
||||
net:
|
||||
rx_bytes:
|
||||
description: Bytes received by the container across all interfaces except loopback. Required by the Container Traffic - Inbound chart on the Security Onion Performance dashboard.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
rx_packets:
|
||||
description: Packets received by the container. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
rx_errors:
|
||||
description: Receive errors counted on the container interfaces. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
rx_dropped:
|
||||
description: Received packets dropped by the container interfaces. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
tx_bytes:
|
||||
description: Bytes transmitted by the container. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
tx_packets:
|
||||
description: Packets transmitted by the container. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
tx_errors:
|
||||
description: Transmit errors counted on the container interfaces. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
tx_dropped:
|
||||
description: Transmitted packets dropped by the container interfaces. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
container_id: *containerid
|
||||
blkio:
|
||||
io_service_bytes_recursive_read:
|
||||
description: Cumulative bytes read from block devices by the container. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
io_service_bytes_recursive_write:
|
||||
description: Cumulative bytes written to block devices by the container. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
container_id: *containerid
|
||||
status:
|
||||
uptime_ns:
|
||||
description: How long the container has been running, in nanoseconds. Required by the Container Uptime chart on the Security Onion Performance dashboard.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
oomkilled:
|
||||
description: Whether the container was killed by the kernel out-of-memory handler. Required by the Most Recent Container Events table on the Security Onion Performance dashboard.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
pid:
|
||||
description: Host process ID of the container main process. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
exitcode:
|
||||
description: Exit code of the container main process. Meaningful once the container has stopped. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
restart_count:
|
||||
description: Number of times the Docker engine has restarted this container. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
started_at:
|
||||
description: Time the container last started, as a Unix timestamp in nanoseconds. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
finished_at:
|
||||
description: Time the container last exited, as a Unix timestamp in nanoseconds. Absent for a container that has never exited. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
container_id: *containerid
|
||||
health:
|
||||
health_status:
|
||||
description: Result of the container healthcheck, such as healthy, unhealthy or starting. Only emitted for containers that define a healthcheck. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
failing_streak:
|
||||
description: Number of consecutive failed healthchecks. Only emitted for containers that define a healthcheck. Defaults to off.
|
||||
forcedType: bool
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: influxdb
|
||||
scripts:
|
||||
eval: &telegrafscripts
|
||||
description: List of input.exec scripts to run for this node type. The script must be present in salt/telegraf/scripts.
|
||||
|
||||
@@ -0,0 +1,398 @@
|
||||
#!/usr/bin/env python3
|
||||
|
||||
# Copyright Security Onion Solutions LLC and/or licensed to Security Onion Solutions LLC under one
|
||||
# or more contributor license agreements. Licensed under the Elastic License 2.0 as shown at
|
||||
# https://securityonion.net/license; you may not use this file except in compliance with the
|
||||
# Elastic License 2.0.
|
||||
|
||||
# Runs from cron as somon, a docker group member, and emits influx line protocol for telegraf to
|
||||
# read. This exists so so-telegraf does not need the docker socket; the container reads the output
|
||||
# file instead.
|
||||
# Measurement/tag/field names match telegraf's inputs.docker plugin because the InfluxDB
|
||||
# dashboards query them directly. Which fields are emitted is set per stat in SOC; see
|
||||
# telegraf.container_stats in defaults.yaml.
|
||||
|
||||
import json
|
||||
|
||||
SETTINGS = json.loads('''{{ CONTAINER_STATS | tojson }}''')
|
||||
{% raw %}
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
from datetime import datetime, timezone
|
||||
|
||||
CGROUP_ROOT = '/sys/fs/cgroup'
|
||||
NANOSEC = 10**9
|
||||
# cgroup and docker report cpu time in microseconds; inputs.docker published nanoseconds
|
||||
USEC_TO_NSEC = 1000
|
||||
|
||||
|
||||
def want(group, field):
|
||||
return bool(SETTINGS.get(group, {}).get(field))
|
||||
|
||||
|
||||
def wants_any(group, fields):
|
||||
return any(want(group, field) for field in fields)
|
||||
|
||||
|
||||
def docker(args):
|
||||
proc = subprocess.run(['docker'] + args, stdout=subprocess.PIPE, stderr=subprocess.DEVNULL, encoding='utf-8')
|
||||
if proc.returncode != 0:
|
||||
print('Container system error; unable to query docker', file=sys.stderr)
|
||||
sys.exit(1)
|
||||
return proc.stdout
|
||||
|
||||
|
||||
def escape_tag(value):
|
||||
return value.replace(',', '\\,').replace(' ', '\\ ').replace('=', '\\=')
|
||||
|
||||
|
||||
def quote(value):
|
||||
return '"{}"'.format(str(value).replace('\\', '\\\\').replace('"', '\\"'))
|
||||
|
||||
|
||||
def integer(value):
|
||||
return '{}i'.format(int(value))
|
||||
|
||||
|
||||
def unsigned(value):
|
||||
# inputs.docker published the cgroup and network counters as uint64; influx stores u and i
|
||||
# as different field types, so matching it keeps historical series readable
|
||||
return '{}u'.format(max(int(value), 0))
|
||||
|
||||
|
||||
def read_text(path):
|
||||
try:
|
||||
with open(path) as handle:
|
||||
return handle.read()
|
||||
except OSError:
|
||||
return ''
|
||||
|
||||
|
||||
def read_pairs(path):
|
||||
# cgroup files such as memory.stat and cpu.stat are "key value" per line
|
||||
values = {}
|
||||
for line in read_text(path).splitlines():
|
||||
parts = line.split()
|
||||
if len(parts) == 2:
|
||||
try:
|
||||
values[parts[0]] = int(parts[1])
|
||||
except ValueError:
|
||||
pass
|
||||
return values
|
||||
|
||||
|
||||
def read_value(path):
|
||||
raw = read_text(path).strip()
|
||||
try:
|
||||
return int(raw)
|
||||
except ValueError:
|
||||
return None
|
||||
|
||||
|
||||
def cgroup_path(pid):
|
||||
# "0::/system.slice/docker-<id>.scope" on cgroup v2
|
||||
for line in read_text('/proc/{}/cgroup'.format(pid)).splitlines():
|
||||
parts = line.split(':', 2)
|
||||
if len(parts) == 3 and parts[1] == '':
|
||||
return CGROUP_ROOT + parts[2]
|
||||
return None
|
||||
|
||||
|
||||
def host_memory_total():
|
||||
match = re.search(r'MemTotal:\s+(\d+) kB', read_text('/proc/meminfo'))
|
||||
return int(match.group(1)) * 1024 if match else 0
|
||||
|
||||
|
||||
def host_cpu_nanoseconds():
|
||||
# inputs.docker's usage_system is the host-wide cpu time the daemon reads from /proc/stat
|
||||
for line in read_text('/proc/stat').splitlines():
|
||||
if line.startswith('cpu '):
|
||||
ticks = sum(int(value) for value in line.split()[1:])
|
||||
return int(ticks * NANOSEC / os.sysconf('SC_CLK_TCK'))
|
||||
return 0
|
||||
|
||||
|
||||
def parse_image(image):
|
||||
# mirrors telegraf's internal/docker ParseImage so the tags match what inputs.docker emitted
|
||||
domain = ''
|
||||
remainder = image
|
||||
if '/' in image:
|
||||
head, _, tail = image.partition('/')
|
||||
if '.' in head or ':' in head or head == 'localhost':
|
||||
domain, remainder = head + '/', tail
|
||||
if ':' in remainder:
|
||||
name, _, version = remainder.rpartition(':')
|
||||
return domain + name, version
|
||||
return domain + remainder, 'unknown'
|
||||
|
||||
|
||||
def to_nanoseconds(stamp):
|
||||
# docker emits 9 fractional digits; fromisoformat takes at most 6 before python 3.11
|
||||
stamp = stamp.rstrip('Z')
|
||||
if '.' in stamp:
|
||||
whole, _, frac = stamp.partition('.')
|
||||
stamp = whole + '.' + frac[:6]
|
||||
try:
|
||||
parsed = datetime.fromisoformat(stamp).replace(tzinfo=timezone.utc)
|
||||
except ValueError:
|
||||
return None
|
||||
if parsed.year <= 1:
|
||||
return None
|
||||
return int(parsed.timestamp() * NANOSEC)
|
||||
|
||||
|
||||
def percent(value):
|
||||
try:
|
||||
return float(value.strip().rstrip('%'))
|
||||
except ValueError:
|
||||
return 0.0
|
||||
|
||||
|
||||
def net_counters(pid):
|
||||
# summed across interfaces except loopback, matching the daemon's per-container totals
|
||||
rows = read_text('/proc/{}/net/dev'.format(pid)).splitlines()[2:]
|
||||
if not rows:
|
||||
return None
|
||||
names = ['rx_bytes', 'rx_packets', 'rx_errors', 'rx_dropped']
|
||||
totals = dict.fromkeys(names + ['tx_bytes', 'tx_packets', 'tx_errors', 'tx_dropped'], 0)
|
||||
found = False
|
||||
for row in rows:
|
||||
iface, _, rest = row.partition(':')
|
||||
if iface.strip() == 'lo':
|
||||
continue
|
||||
columns = rest.split()
|
||||
if len(columns) < 12:
|
||||
continue
|
||||
found = True
|
||||
for index, name in enumerate(names):
|
||||
totals[name] += int(columns[index])
|
||||
for index, name in enumerate(['tx_bytes', 'tx_packets', 'tx_errors', 'tx_dropped']):
|
||||
totals[name] += int(columns[8 + index])
|
||||
return totals if found else None
|
||||
|
||||
|
||||
def blkio_counters(path):
|
||||
# inputs.docker emitted the device=total row unconditionally, so a container that has done
|
||||
# no block io reports zeros rather than dropping out of the measurement entirely
|
||||
totals = {'io_service_bytes_recursive_read': 0, 'io_service_bytes_recursive_write': 0}
|
||||
rows = read_text(path + '/io.stat').splitlines()
|
||||
for row in rows:
|
||||
for token in row.split()[1:]:
|
||||
key, _, value = token.partition('=')
|
||||
try:
|
||||
if key == 'rbytes':
|
||||
totals['io_service_bytes_recursive_read'] += int(value)
|
||||
elif key == 'wbytes':
|
||||
totals['io_service_bytes_recursive_write'] += int(value)
|
||||
except ValueError:
|
||||
pass
|
||||
return totals
|
||||
|
||||
|
||||
def collect_stats():
|
||||
stats = {}
|
||||
for line in docker(['stats', '--no-stream', '--format', '{{json .}}']).splitlines():
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
try:
|
||||
entry = json.loads(line)
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
stats[entry.get('Name', '')] = entry
|
||||
return stats
|
||||
|
||||
|
||||
def collect_inspect():
|
||||
ids = docker(['ps', '-aq']).split()
|
||||
if not ids:
|
||||
return []
|
||||
try:
|
||||
return json.loads(docker(['inspect'] + ids))
|
||||
except json.JSONDecodeError:
|
||||
return []
|
||||
|
||||
|
||||
def collect_info():
|
||||
try:
|
||||
return json.loads(docker(['info', '--format', '{{json .}}']))
|
||||
except json.JSONDecodeError:
|
||||
return {}
|
||||
|
||||
|
||||
class Emitter:
|
||||
def __init__(self):
|
||||
self.lines = []
|
||||
|
||||
def add(self, measurement, tags, fields):
|
||||
if not fields:
|
||||
return
|
||||
tagset = ','.join('{}={}'.format(key, escape_tag(value)) for key, value in tags)
|
||||
body = ','.join('{}={}'.format(key, fields[key]) for key in sorted(fields))
|
||||
self.lines.append('{},{} {}'.format(measurement, tagset, body))
|
||||
|
||||
|
||||
def main():
|
||||
cpu_extra = ['usage_total', 'usage_in_usermode', 'usage_in_kernelmode',
|
||||
'throttling_periods', 'throttling_throttled_periods', 'throttling_throttled_time']
|
||||
mem_extra = ['usage', 'limit', 'max_usage', 'active_anon', 'active_file', 'inactive_anon',
|
||||
'inactive_file', 'unevictable', 'pgfault', 'pgmajfault']
|
||||
net_fields = ['rx_bytes', 'rx_packets', 'rx_errors', 'rx_dropped',
|
||||
'tx_bytes', 'tx_packets', 'tx_errors', 'tx_dropped']
|
||||
engine_fields = ['n_containers', 'n_containers_running', 'n_containers_stopped',
|
||||
'n_containers_paused', 'n_images', 'n_cpus', 'n_goroutines',
|
||||
'n_used_file_descriptors', 'n_listener_events']
|
||||
|
||||
identity = want('tags', 'identity')
|
||||
need_stats = want('cpu', 'usage_percent')
|
||||
need_cgroup_cpu = wants_any('cpu', cpu_extra)
|
||||
need_cgroup_mem = wants_any('mem', mem_extra + ['usage_percent'])
|
||||
need_blkio = wants_any('blkio', ['io_service_bytes_recursive_read',
|
||||
'io_service_bytes_recursive_write', 'container_id'])
|
||||
need_net = wants_any('net', net_fields + ['container_id'])
|
||||
need_info = identity or wants_any('engine', engine_fields + ['memory_total'])
|
||||
|
||||
emitter = Emitter()
|
||||
stats = collect_stats() if need_stats else {}
|
||||
info = collect_info() if need_info else {}
|
||||
engine_tags = [('engine_host', info.get('Name', '')),
|
||||
('server_version', info.get('ServerVersion', ''))]
|
||||
system_ns = host_cpu_nanoseconds() if want('cpu', 'usage_system') else 0
|
||||
mem_total = host_memory_total() if wants_any('mem', ['limit', 'usage_percent']) else 0
|
||||
|
||||
if info:
|
||||
counts = {'n_containers': 'Containers', 'n_containers_running': 'ContainersRunning',
|
||||
'n_containers_stopped': 'ContainersStopped', 'n_containers_paused': 'ContainersPaused',
|
||||
'n_images': 'Images', 'n_cpus': 'NCPU', 'n_goroutines': 'NGoroutines',
|
||||
'n_used_file_descriptors': 'NFd', 'n_listener_events': 'NEventsListener'}
|
||||
fields = {name: integer(info[key]) for name, key in counts.items()
|
||||
if want('engine', name) and key in info}
|
||||
emitter.add('docker', engine_tags, fields)
|
||||
# inputs.docker published memory_total as its own point
|
||||
if want('engine', 'memory_total') and 'MemTotal' in info:
|
||||
emitter.add('docker', engine_tags, {'memory_total': integer(info['MemTotal'])})
|
||||
|
||||
for container in collect_inspect():
|
||||
name = container.get('Name', '').lstrip('/')
|
||||
if not name:
|
||||
continue
|
||||
state = container.get('State', {})
|
||||
status = state.get('Status', 'unknown')
|
||||
pid = state.get('Pid')
|
||||
container_id = container.get('Id', '')
|
||||
tags = [('container_name', name), ('container_status', status)]
|
||||
if identity:
|
||||
image, version = parse_image(container.get('Config', {}).get('Image', ''))
|
||||
tags += [('container_image', image), ('container_version', version)] + engine_tags
|
||||
|
||||
status_fields = {}
|
||||
if want('status', 'uptime_ns') or want('status', 'started_at') or want('status', 'finished_at'):
|
||||
started = to_nanoseconds(state.get('StartedAt', ''))
|
||||
finished = to_nanoseconds(state.get('FinishedAt', ''))
|
||||
if started is not None:
|
||||
if want('status', 'started_at'):
|
||||
status_fields['started_at'] = integer(started)
|
||||
if want('status', 'uptime_ns'):
|
||||
end = finished if finished is not None and finished >= started else int(
|
||||
datetime.now(timezone.utc).timestamp() * NANOSEC)
|
||||
status_fields['uptime_ns'] = integer(end - started)
|
||||
if finished is not None and want('status', 'finished_at'):
|
||||
status_fields['finished_at'] = integer(finished)
|
||||
if want('status', 'oomkilled'):
|
||||
status_fields['oomkilled'] = 'true' if state.get('OOMKilled') else 'false'
|
||||
if want('status', 'pid'):
|
||||
status_fields['pid'] = integer(pid or 0)
|
||||
if want('status', 'exitcode'):
|
||||
status_fields['exitcode'] = integer(state.get('ExitCode', 0))
|
||||
if want('status', 'restart_count'):
|
||||
status_fields['restart_count'] = integer(container.get('RestartCount', 0))
|
||||
if want('status', 'container_id'):
|
||||
status_fields['container_id'] = quote(container_id)
|
||||
emitter.add('docker_container_status', tags, status_fields)
|
||||
|
||||
health = state.get('Health')
|
||||
if health:
|
||||
health_fields = {}
|
||||
if want('health', 'health_status'):
|
||||
health_fields['health_status'] = quote(health.get('Status', ''))
|
||||
if want('health', 'failing_streak'):
|
||||
health_fields['failing_streak'] = integer(health.get('FailingStreak', 0))
|
||||
emitter.add('docker_container_health', tags, health_fields)
|
||||
|
||||
if status != 'running' or not pid:
|
||||
continue
|
||||
path = cgroup_path(pid)
|
||||
entry = stats.get(name, {})
|
||||
|
||||
cpu_fields = {}
|
||||
if want('cpu', 'usage_percent'):
|
||||
cpu_fields['usage_percent'] = percent(entry.get('CPUPerc', '0%'))
|
||||
if want('cpu', 'usage_system'):
|
||||
cpu_fields['usage_system'] = unsigned(system_ns)
|
||||
if want('cpu', 'container_id'):
|
||||
cpu_fields['container_id'] = quote(container_id)
|
||||
if need_cgroup_cpu and path:
|
||||
cpu = read_pairs(path + '/cpu.stat')
|
||||
mapping = {'usage_total': 'usage_usec', 'usage_in_usermode': 'user_usec',
|
||||
'usage_in_kernelmode': 'system_usec',
|
||||
'throttling_periods': 'nr_periods',
|
||||
'throttling_throttled_periods': 'nr_throttled',
|
||||
'throttling_throttled_time': 'throttled_usec'}
|
||||
for field, key in mapping.items():
|
||||
if want('cpu', field) and key in cpu:
|
||||
scale = 1 if field.startswith('throttling_') and field != 'throttling_throttled_time' else USEC_TO_NSEC
|
||||
cpu_fields[field] = unsigned(cpu[key] * scale)
|
||||
emitter.add('docker_container_cpu', tags + [('cpu', 'cpu-total')], cpu_fields)
|
||||
|
||||
mem_fields = {}
|
||||
if want('mem', 'container_id'):
|
||||
mem_fields['container_id'] = quote(container_id)
|
||||
if need_cgroup_mem and path:
|
||||
memory = read_pairs(path + '/memory.stat')
|
||||
for field in ['active_anon', 'active_file', 'inactive_anon', 'inactive_file',
|
||||
'unevictable', 'pgfault', 'pgmajfault']:
|
||||
if want('mem', field) and field in memory:
|
||||
mem_fields[field] = unsigned(memory[field])
|
||||
current = read_value(path + '/memory.current')
|
||||
# inputs.docker reports usage net of reclaimable page cache
|
||||
usage = max(current - memory.get('inactive_file', 0), 0) if current is not None else None
|
||||
raw_limit = read_value(path + '/memory.max')
|
||||
limit = raw_limit if raw_limit is not None else mem_total
|
||||
if want('mem', 'usage') and usage is not None:
|
||||
mem_fields['usage'] = unsigned(usage)
|
||||
if want('mem', 'limit'):
|
||||
mem_fields['limit'] = unsigned(limit)
|
||||
if want('mem', 'usage_percent') and usage is not None:
|
||||
# same ratio inputs.docker computes, from the same two values
|
||||
mem_fields['usage_percent'] = usage / limit * 100.0 if limit else 0.0
|
||||
if want('mem', 'max_usage'):
|
||||
peak = read_value(path + '/memory.peak')
|
||||
if peak is not None:
|
||||
mem_fields['max_usage'] = unsigned(peak)
|
||||
emitter.add('docker_container_mem', tags, mem_fields)
|
||||
|
||||
# inputs.docker emitted nothing for host-network containers; its Networks map was empty
|
||||
if need_net and container.get('HostConfig', {}).get('NetworkMode', '') != 'host':
|
||||
counters = net_counters(pid)
|
||||
if counters is not None:
|
||||
net_out = {field: unsigned(counters[field]) for field in net_fields if want('net', field)}
|
||||
if want('net', 'container_id'):
|
||||
net_out['container_id'] = quote(container_id)
|
||||
emitter.add('docker_container_net', tags + [('network', 'total')], net_out)
|
||||
|
||||
if need_blkio and path:
|
||||
counters = blkio_counters(path)
|
||||
if counters is not None:
|
||||
blkio_out = {field: unsigned(value) for field, value in counters.items() if want('blkio', field)}
|
||||
if want('blkio', 'container_id'):
|
||||
blkio_out['container_id'] = quote(container_id)
|
||||
emitter.add('docker_container_blkio', tags + [('device', 'total')], blkio_out)
|
||||
|
||||
print('\n'.join(emitter.lines))
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
{% endraw %}
|
||||
@@ -0,0 +1,635 @@
|
||||
# Copyright Security Onion Solutions LLC and/or licensed to Security Onion Solutions LLC under one
|
||||
# or more contributor license agreements. Licensed under the Elastic License 2.0 as shown at
|
||||
# https://securityonion.net/license; you may not use this file except in compliance with the
|
||||
# Elastic License 2.0.
|
||||
|
||||
# so-container-stats replaces telegraf's inputs.docker plugin, which was dropped along with the
|
||||
# docker socket. The InfluxDB dashboards query its output by measurement, tag and field name, so
|
||||
# these tests pin that contract: the shipped defaults, the per-stat toggles, the field types, and
|
||||
# the value semantics copied from the plugin.
|
||||
#
|
||||
# The collector is a jinja template, so every test renders it the way salt does and imports the
|
||||
# result. Docker and the cgroup filesystem are faked, so nothing here needs a container runtime.
|
||||
|
||||
import contextlib
|
||||
import importlib.util
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
import tempfile
|
||||
import unittest
|
||||
|
||||
import jinja2
|
||||
import yaml
|
||||
|
||||
HERE = os.path.dirname(os.path.abspath(__file__))
|
||||
TEMPLATE = os.path.join(HERE, 'so-container-stats')
|
||||
DEFAULTS = os.path.join(HERE, '..', '..', 'defaults.yaml')
|
||||
|
||||
# a container id is a 64 character hex string; the plugin published it in full
|
||||
SOC_ID = 'a' * 64
|
||||
TELEGRAF_ID = 'b' * 64
|
||||
NGINX_ID = 'c' * 64
|
||||
IDSTOOLS_ID = 'd' * 64
|
||||
|
||||
|
||||
def shipped_defaults():
|
||||
with open(DEFAULTS) as handle:
|
||||
return yaml.safe_load(handle)['telegraf']['container_stats']
|
||||
|
||||
|
||||
def all_enabled():
|
||||
return {group: {field: True for field in fields} for group, fields in shipped_defaults().items()}
|
||||
|
||||
|
||||
def render(settings):
|
||||
"""Render the template as salt does, import it, and hand back the module."""
|
||||
with open(TEMPLATE) as handle:
|
||||
source = handle.read()
|
||||
rendered = jinja2.Template(source, keep_trailing_newline=True).render(CONTAINER_STATS=settings)
|
||||
path = os.path.join(tempfile.mkdtemp(), 'so_container_stats.py')
|
||||
with open(path, 'w') as handle:
|
||||
handle.write(rendered)
|
||||
spec = importlib.util.spec_from_file_location('so_container_stats', path)
|
||||
module = importlib.util.module_from_spec(spec)
|
||||
spec.loader.exec_module(module)
|
||||
return module
|
||||
|
||||
|
||||
def split_escaped(text, sep):
|
||||
"""Split on an unescaped, unquoted separator, the way influx line protocol is written."""
|
||||
parts, current, escaped, quoted = [], '', False, False
|
||||
for char in text:
|
||||
if escaped:
|
||||
current += char
|
||||
escaped = False
|
||||
elif char == '\\':
|
||||
current += char
|
||||
escaped = True
|
||||
elif char == '"':
|
||||
quoted = not quoted
|
||||
current += char
|
||||
elif char == sep and not quoted:
|
||||
parts.append(current)
|
||||
current = ''
|
||||
else:
|
||||
current += char
|
||||
parts.append(current)
|
||||
return parts
|
||||
|
||||
|
||||
def split_on_space(line):
|
||||
escaped = quoted = False
|
||||
for index, char in enumerate(line):
|
||||
if escaped:
|
||||
escaped = False
|
||||
elif char == '\\':
|
||||
escaped = True
|
||||
elif char == '"':
|
||||
quoted = not quoted
|
||||
elif char == ' ' and not quoted:
|
||||
return line[:index], line[index + 1:]
|
||||
return line, ''
|
||||
|
||||
|
||||
def parse(output):
|
||||
"""Parse line protocol into {(measurement, container): (tags, fields)}."""
|
||||
points = {}
|
||||
for line in output.splitlines():
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
head, fieldpart = split_on_space(line)
|
||||
pieces = split_escaped(head, ',')
|
||||
measurement, tags = pieces[0], {}
|
||||
for piece in pieces[1:]:
|
||||
key, _, value = piece.partition('=')
|
||||
tags[key] = value
|
||||
fields = {}
|
||||
for piece in split_escaped(fieldpart, ','):
|
||||
key, _, value = piece.partition('=')
|
||||
fields[key] = value
|
||||
key = (measurement, tags.get('container_name', ''))
|
||||
if key in points:
|
||||
# the engine measurement is published as two points, the way inputs.docker did
|
||||
points[key][1].update(fields)
|
||||
else:
|
||||
points[key] = (tags, fields)
|
||||
return points
|
||||
|
||||
|
||||
def field_type(value):
|
||||
if value.endswith('u'):
|
||||
return 'unsigned'
|
||||
if value.endswith('i'):
|
||||
return 'integer'
|
||||
if value.startswith('"'):
|
||||
return 'string'
|
||||
if value in ('true', 'false'):
|
||||
return 'boolean'
|
||||
return 'float'
|
||||
|
||||
|
||||
def container(name, cid, pid, status='running', network='bridge', image='registry:5000/repo/img:3.4.0',
|
||||
started='2026-09-24T10:00:00.123456789Z', finished='0001-01-01T00:00:00Z', health=None,
|
||||
oomkilled=False, exitcode=0, restarts=0):
|
||||
state = {'Status': status, 'Pid': pid, 'StartedAt': started, 'FinishedAt': finished,
|
||||
'OOMKilled': oomkilled, 'ExitCode': exitcode}
|
||||
if health is not None:
|
||||
state['Health'] = health
|
||||
return {'Id': cid, 'Name': '/' + name, 'State': state, 'RestartCount': restarts,
|
||||
'Config': {'Image': image}, 'HostConfig': {'NetworkMode': network}}
|
||||
|
||||
|
||||
class CollectorTestCase(unittest.TestCase):
|
||||
"""Builds a fake docker engine and cgroup tree, then runs the collector against it."""
|
||||
|
||||
def setUp(self):
|
||||
self.inspect = [
|
||||
container('so-soc', SOC_ID, 1001),
|
||||
container('so-telegraf', TELEGRAF_ID, 1002, network='host'),
|
||||
container('so-nginx', NGINX_ID, 1003, health={'Status': 'healthy', 'FailingStreak': 0}),
|
||||
]
|
||||
self.stats = {
|
||||
'so-soc': {'Name': 'so-soc', 'CPUPerc': '1.25%', 'MemPerc': '6.39%', 'NetIO': '1kB / 2kB'},
|
||||
'so-telegraf': {'Name': 'so-telegraf', 'CPUPerc': '0.04%', 'MemPerc': '0.58%', 'NetIO': '0B / 0B'},
|
||||
'so-nginx': {'Name': 'so-nginx', 'CPUPerc': '0.00%', 'MemPerc': '0.09%', 'NetIO': '3kB / 4kB'},
|
||||
}
|
||||
self.info = {'Name': 'sohost', 'ServerVersion': '29.2.1', 'Containers': 3, 'ContainersRunning': 3,
|
||||
'ContainersStopped': 0, 'ContainersPaused': 0, 'Images': 9, 'NCPU': 8,
|
||||
'NGoroutines': 42, 'NFd': 77, 'NEventsListener': 1, 'MemTotal': 16000000000}
|
||||
# one cgroup per container, addressed through /proc/<pid>/cgroup exactly as the collector does
|
||||
self.files = {
|
||||
'/proc/meminfo': 'MemTotal: 15625000 kB\n',
|
||||
'/proc/stat': 'cpu 100 200 300 400\ncpu0 1 2 3 4\n',
|
||||
}
|
||||
for pid in (1001, 1002, 1003):
|
||||
self.files['/proc/%d/cgroup' % pid] = '0::/scope%d\n' % pid
|
||||
self.cgroup(pid, 'cpu.stat',
|
||||
'usage_usec 1000\nuser_usec 600\nsystem_usec 400\nnr_periods 5\nnr_throttled 2\nthrottled_usec 700\n')
|
||||
self.cgroup(pid, 'memory.stat',
|
||||
'active_anon 300\nactive_file 40\ninactive_anon 200\ninactive_file 50\nunevictable 0\npgfault 1234\npgmajfault 56\n')
|
||||
self.cgroup(pid, 'memory.current', '1000\n')
|
||||
self.cgroup(pid, 'memory.max', '4000\n')
|
||||
self.cgroup(pid, 'memory.peak', '2500\n')
|
||||
self.cgroup(pid, 'io.stat', '8:0 rbytes=100 wbytes=200 rios=1 wios=2\n252:0 rbytes=10 wbytes=20 rios=1 wios=1\n')
|
||||
self.files['/proc/%d/net/dev' % pid] = (
|
||||
'Inter-| Receive | Transmit\n'
|
||||
' face |bytes packets errs drop fifo frame compressed multicast|bytes packets errs drop fifo colls carrier compressed\n'
|
||||
' lo: 9999 99 9 9 0 0 0 0 9999 99 9 9 0 0 0 0\n'
|
||||
' eth0: 1000 10 1 2 0 0 0 0 2000 20 3 4 0 0 0 0\n'
|
||||
' eth1: 500 5 0 0 0 0 0 0 1000 10 0 0 0 0 0 0\n')
|
||||
|
||||
def cgroup(self, pid, name, contents):
|
||||
self.files['/sys/fs/cgroup/scope%d/%s' % (pid, name)] = contents
|
||||
|
||||
def run_collector(self, settings=None, module=None):
|
||||
module = module or render(settings if settings is not None else all_enabled())
|
||||
|
||||
def fake_docker(args):
|
||||
if args[0] == 'stats':
|
||||
return ''.join(json.dumps(entry) + '\n' for entry in self.stats.values())
|
||||
if args[0] == 'ps':
|
||||
return ' '.join(entry['Id'] for entry in self.inspect)
|
||||
if args[0] == 'inspect':
|
||||
return json.dumps(self.inspect)
|
||||
if args[0] == 'info':
|
||||
return json.dumps(self.info)
|
||||
raise AssertionError('unexpected docker call: %s' % args)
|
||||
|
||||
module.docker = fake_docker
|
||||
module.read_text = lambda path: self.files.get(path, '')
|
||||
module.CGROUP_ROOT = '/sys/fs/cgroup'
|
||||
buffer = io.StringIO()
|
||||
with contextlib.redirect_stdout(buffer):
|
||||
module.main()
|
||||
self.output = buffer.getvalue()
|
||||
return parse(self.output)
|
||||
|
||||
|
||||
class TestShippedDefaults(CollectorTestCase):
|
||||
|
||||
def test_defaults_emit_only_the_dashboard_fields(self):
|
||||
# the Security Onion Performance dashboard queries exactly these five
|
||||
points = self.run_collector(shipped_defaults())
|
||||
emitted = {(measurement, field) for (measurement, _), (_, fields) in points.items() for field in fields}
|
||||
self.assertEqual(emitted, {
|
||||
('docker_container_cpu', 'usage_percent'),
|
||||
('docker_container_mem', 'usage_percent'),
|
||||
('docker_container_net', 'rx_bytes'),
|
||||
('docker_container_status', 'uptime_ns'),
|
||||
('docker_container_status', 'oomkilled'),
|
||||
})
|
||||
|
||||
def test_defaults_do_not_emit_the_opt_in_measurements(self):
|
||||
points = self.run_collector(shipped_defaults())
|
||||
measurements = {measurement for measurement, _ in points}
|
||||
self.assertNotIn('docker', measurements)
|
||||
self.assertNotIn('docker_container_blkio', measurements)
|
||||
self.assertNotIn('docker_container_health', measurements)
|
||||
|
||||
def test_defaults_carry_the_tags_the_dashboard_filters_on(self):
|
||||
points = self.run_collector(shipped_defaults())
|
||||
tags, _ = points[('docker_container_cpu', 'so-soc')]
|
||||
self.assertEqual(tags['container_status'], 'running')
|
||||
self.assertEqual(tags['cpu'], 'cpu-total')
|
||||
# identity tags are opt in, so they must be absent by default
|
||||
self.assertNotIn('container_image', tags)
|
||||
self.assertNotIn('engine_host', tags)
|
||||
|
||||
|
||||
class TestToggles(CollectorTestCase):
|
||||
|
||||
def test_enabling_one_stat_adds_only_that_field(self):
|
||||
settings = shipped_defaults()
|
||||
settings['cpu']['usage_total'] = True
|
||||
_, fields = self.run_collector(settings)[('docker_container_cpu', 'so-soc')]
|
||||
self.assertIn('usage_total', fields)
|
||||
self.assertNotIn('usage_in_usermode', fields)
|
||||
|
||||
def test_disabling_one_stat_leaves_its_neighbours(self):
|
||||
settings = all_enabled()
|
||||
settings['blkio']['io_service_bytes_recursive_read'] = False
|
||||
_, fields = self.run_collector(settings)[('docker_container_blkio', 'so-soc')]
|
||||
self.assertNotIn('io_service_bytes_recursive_read', fields)
|
||||
self.assertIn('io_service_bytes_recursive_write', fields)
|
||||
|
||||
def test_identity_tags_are_added_when_enabled(self):
|
||||
tags, _ = self.run_collector(all_enabled())[('docker_container_cpu', 'so-soc')]
|
||||
self.assertEqual(tags['container_image'], 'registry:5000/repo/img')
|
||||
self.assertEqual(tags['container_version'], '3.4.0')
|
||||
self.assertEqual(tags['engine_host'], 'sohost')
|
||||
self.assertEqual(tags['server_version'], '29.2.1')
|
||||
|
||||
def test_everything_off_emits_nothing(self):
|
||||
settings = {group: {field: False for field in fields} for group, fields in shipped_defaults().items()}
|
||||
self.assertEqual(self.run_collector(settings), {})
|
||||
|
||||
|
||||
class TestFieldTypes(CollectorTestCase):
|
||||
"""inputs.docker wrote the cgroup and network counters as unsigned; influx treats u and i as
|
||||
different field types, so a mismatch breaks queries spanning the change."""
|
||||
|
||||
def test_counter_fields_are_unsigned(self):
|
||||
points = self.run_collector(all_enabled())
|
||||
for measurement, field in (('docker_container_cpu', 'usage_total'),
|
||||
('docker_container_cpu', 'throttling_periods'),
|
||||
('docker_container_mem', 'usage'),
|
||||
('docker_container_mem', 'limit'),
|
||||
('docker_container_mem', 'pgfault'),
|
||||
('docker_container_net', 'rx_bytes'),
|
||||
('docker_container_blkio', 'io_service_bytes_recursive_read')):
|
||||
_, fields = points[(measurement, 'so-soc')]
|
||||
self.assertEqual(field_type(fields[field]), 'unsigned', '%s.%s' % (measurement, field))
|
||||
|
||||
def test_status_and_engine_fields_are_signed(self):
|
||||
points = self.run_collector(all_enabled())
|
||||
_, status = points[('docker_container_status', 'so-soc')]
|
||||
for field in ('uptime_ns', 'pid', 'exitcode', 'restart_count', 'started_at'):
|
||||
self.assertEqual(field_type(status[field]), 'integer', field)
|
||||
_, engine = points[('docker', '')]
|
||||
self.assertEqual(field_type(engine['n_containers']), 'integer')
|
||||
|
||||
def test_percentages_are_floats_and_ids_are_quoted_strings(self):
|
||||
points = self.run_collector(all_enabled())
|
||||
_, cpu = points[('docker_container_cpu', 'so-soc')]
|
||||
self.assertEqual(field_type(cpu['usage_percent']), 'float')
|
||||
self.assertEqual(field_type(cpu['container_id']), 'string')
|
||||
self.assertEqual(cpu['container_id'], '"%s"' % SOC_ID)
|
||||
_, status = points[('docker_container_status', 'so-soc')]
|
||||
self.assertEqual(field_type(status['oomkilled']), 'boolean')
|
||||
_, health = points[('docker_container_health', 'so-nginx')]
|
||||
self.assertEqual(health['health_status'], '"healthy"')
|
||||
|
||||
|
||||
class TestMemorySemantics(CollectorTestCase):
|
||||
|
||||
def test_usage_subtracts_reclaimable_page_cache(self):
|
||||
# inputs.docker reports usage net of inactive_file: 1000 - 50
|
||||
_, fields = self.run_collector(all_enabled())[('docker_container_mem', 'so-soc')]
|
||||
self.assertEqual(fields['usage'], '950u')
|
||||
|
||||
def test_usage_percent_is_usage_over_limit(self):
|
||||
_, fields = self.run_collector(all_enabled())[('docker_container_mem', 'so-soc')]
|
||||
self.assertAlmostEqual(float(fields['usage_percent']), 950 / 4000 * 100.0)
|
||||
|
||||
def test_limit_falls_back_to_host_memory_when_unlimited(self):
|
||||
for pid in (1001, 1002, 1003):
|
||||
self.cgroup(pid, 'memory.max', 'max\n')
|
||||
_, fields = self.run_collector(all_enabled())[('docker_container_mem', 'so-soc')]
|
||||
self.assertEqual(fields['limit'], '%du' % (15625000 * 1024))
|
||||
|
||||
def test_max_usage_reports_the_cgroup_peak(self):
|
||||
# the daemon reports 0 on cgroup v2, so this deliberately carries the real peak
|
||||
_, fields = self.run_collector(all_enabled())[('docker_container_mem', 'so-soc')]
|
||||
self.assertEqual(fields['max_usage'], '2500u')
|
||||
|
||||
|
||||
class TestCpuSemantics(CollectorTestCase):
|
||||
|
||||
def test_microsecond_counters_are_published_as_nanoseconds(self):
|
||||
_, fields = self.run_collector(all_enabled())[('docker_container_cpu', 'so-soc')]
|
||||
self.assertEqual(fields['usage_total'], '1000000u')
|
||||
self.assertEqual(fields['usage_in_usermode'], '600000u')
|
||||
self.assertEqual(fields['usage_in_kernelmode'], '400000u')
|
||||
self.assertEqual(fields['throttling_throttled_time'], '700000u')
|
||||
|
||||
def test_throttling_counts_are_not_scaled(self):
|
||||
_, fields = self.run_collector(all_enabled())[('docker_container_cpu', 'so-soc')]
|
||||
self.assertEqual(fields['throttling_periods'], '5u')
|
||||
self.assertEqual(fields['throttling_throttled_periods'], '2u')
|
||||
|
||||
def test_usage_system_is_host_wide_cpu_time(self):
|
||||
_, fields = self.run_collector(all_enabled())[('docker_container_cpu', 'so-soc')]
|
||||
self.assertEqual(field_type(fields['usage_system']), 'unsigned')
|
||||
self.assertNotEqual(fields['usage_system'], '0u')
|
||||
|
||||
|
||||
class TestNetwork(CollectorTestCase):
|
||||
|
||||
def test_counters_sum_interfaces_and_ignore_loopback(self):
|
||||
_, fields = self.run_collector(all_enabled())[('docker_container_net', 'so-soc')]
|
||||
self.assertEqual(fields['rx_bytes'], '1500u')
|
||||
self.assertEqual(fields['rx_packets'], '15u')
|
||||
self.assertEqual(fields['tx_bytes'], '3000u')
|
||||
self.assertEqual(fields['rx_dropped'], '2u')
|
||||
|
||||
def test_host_network_containers_emit_no_row(self):
|
||||
# inputs.docker skipped these: its Networks map is empty for --net=host
|
||||
points = self.run_collector(all_enabled())
|
||||
self.assertNotIn(('docker_container_net', 'so-telegraf'), points)
|
||||
self.assertIn(('docker_container_net', 'so-soc'), points)
|
||||
|
||||
def test_total_tag_is_present(self):
|
||||
tags, _ = self.run_collector(all_enabled())[('docker_container_net', 'so-soc')]
|
||||
self.assertEqual(tags['network'], 'total')
|
||||
|
||||
|
||||
class TestBlkio(CollectorTestCase):
|
||||
|
||||
def test_counters_sum_devices(self):
|
||||
_, fields = self.run_collector(all_enabled())[('docker_container_blkio', 'so-soc')]
|
||||
self.assertEqual(fields['io_service_bytes_recursive_read'], '110u')
|
||||
self.assertEqual(fields['io_service_bytes_recursive_write'], '220u')
|
||||
|
||||
def test_container_with_no_io_reports_zero_rather_than_disappearing(self):
|
||||
for pid in (1001, 1002, 1003):
|
||||
self.cgroup(pid, 'io.stat', '')
|
||||
tags, fields = self.run_collector(all_enabled())[('docker_container_blkio', 'so-soc')]
|
||||
self.assertEqual(fields['io_service_bytes_recursive_read'], '0u')
|
||||
self.assertEqual(tags['device'], 'total')
|
||||
|
||||
|
||||
class TestStatus(CollectorTestCase):
|
||||
|
||||
def test_running_container_uptime_counts_from_start(self):
|
||||
_, fields = self.run_collector(all_enabled())[('docker_container_status', 'so-soc')]
|
||||
self.assertGreater(int(fields['uptime_ns'].rstrip('i')), 0)
|
||||
self.assertNotIn('finished_at', fields)
|
||||
|
||||
def test_exited_container_reports_its_lifetime_and_finished_at(self):
|
||||
self.inspect.append(container('so-idstools', IDSTOOLS_ID, 0, status='exited',
|
||||
started='2026-09-24T10:00:00.000000000Z',
|
||||
finished='2026-09-24T10:00:02.000000000Z', exitcode=3, restarts=1))
|
||||
points = self.run_collector(all_enabled())
|
||||
tags, fields = points[('docker_container_status', 'so-idstools')]
|
||||
self.assertEqual(tags['container_status'], 'exited')
|
||||
self.assertEqual(fields['uptime_ns'], '2000000000i')
|
||||
self.assertEqual(fields['finished_at'], '1790244002000000000i')
|
||||
self.assertEqual(fields['exitcode'], '3i')
|
||||
self.assertEqual(fields['restart_count'], '1i')
|
||||
# a stopped container has no live stats, so only the status row is emitted
|
||||
self.assertNotIn(('docker_container_cpu', 'so-idstools'), points)
|
||||
|
||||
def test_health_is_emitted_only_for_containers_with_a_healthcheck(self):
|
||||
points = self.run_collector(all_enabled())
|
||||
self.assertIn(('docker_container_health', 'so-nginx'), points)
|
||||
self.assertNotIn(('docker_container_health', 'so-soc'), points)
|
||||
|
||||
def test_oomkilled_is_reported(self):
|
||||
self.inspect[0]['State']['OOMKilled'] = True
|
||||
_, fields = self.run_collector(all_enabled())[('docker_container_status', 'so-soc')]
|
||||
self.assertEqual(fields['oomkilled'], 'true')
|
||||
|
||||
|
||||
class TestEngineMeasurement(CollectorTestCase):
|
||||
|
||||
def test_engine_counts_come_from_docker_info(self):
|
||||
_, fields = self.run_collector(all_enabled())[('docker', '')]
|
||||
self.assertEqual(fields['n_containers'], '3i')
|
||||
self.assertEqual(fields['n_cpus'], '8i')
|
||||
self.assertEqual(fields['n_used_file_descriptors'], '77i')
|
||||
|
||||
def test_engine_is_published_as_two_points(self):
|
||||
# inputs.docker emitted memory_total on its own point, so keep that shape
|
||||
self.run_collector(all_enabled())
|
||||
engine = [line for line in self.output.splitlines() if line.startswith('docker,')]
|
||||
self.assertEqual(len(engine), 2)
|
||||
self.assertTrue(any('memory_total=' in line for line in engine))
|
||||
self.assertTrue(any('n_containers=' in line for line in engine))
|
||||
|
||||
def test_engine_rows_are_tagged_with_host_and_version(self):
|
||||
tags, _ = self.run_collector(all_enabled())[('docker', '')]
|
||||
self.assertEqual(tags['engine_host'], 'sohost')
|
||||
self.assertEqual(tags['server_version'], '29.2.1')
|
||||
|
||||
|
||||
class TestLineProtocol(CollectorTestCase):
|
||||
|
||||
def test_tag_values_are_escaped(self):
|
||||
self.inspect[0]['Name'] = '/odd name,with=chars'
|
||||
self.stats['odd name,with=chars'] = self.stats.pop('so-soc')
|
||||
self.stats['odd name,with=chars']['Name'] = 'odd name,with=chars'
|
||||
output = self.run_collector(all_enabled())
|
||||
self.assertIn(('docker_container_cpu', 'odd\\ name\\,with\\=chars'), output)
|
||||
|
||||
def test_image_without_a_tag_reports_version_unknown(self):
|
||||
self.inspect[0]['Config']['Image'] = 'busybox'
|
||||
tags, _ = self.run_collector(all_enabled())[('docker_container_cpu', 'so-soc')]
|
||||
self.assertEqual(tags['container_image'], 'busybox')
|
||||
self.assertEqual(tags['container_version'], 'unknown')
|
||||
|
||||
def test_every_line_has_a_measurement_tagset_and_fieldset(self):
|
||||
module = render(all_enabled())
|
||||
|
||||
def fake_docker(args):
|
||||
if args[0] == 'stats':
|
||||
return ''.join(json.dumps(entry) + '\n' for entry in self.stats.values())
|
||||
if args[0] == 'ps':
|
||||
return ' '.join(entry['Id'] for entry in self.inspect)
|
||||
if args[0] == 'inspect':
|
||||
return json.dumps(self.inspect)
|
||||
return json.dumps(self.info)
|
||||
|
||||
module.docker = fake_docker
|
||||
module.read_text = lambda path: self.files.get(path, '')
|
||||
buffer = io.StringIO()
|
||||
with contextlib.redirect_stdout(buffer):
|
||||
module.main()
|
||||
lines = [line for line in buffer.getvalue().splitlines() if line.strip()]
|
||||
self.assertTrue(lines)
|
||||
for line in lines:
|
||||
head, fieldpart = split_on_space(line)
|
||||
self.assertIn(',', head, line)
|
||||
self.assertIn('=', fieldpart, line)
|
||||
self.assertFalse(fieldpart.endswith(','), line)
|
||||
|
||||
|
||||
# group -> (measurement, the container whose row carries it)
|
||||
GROUP_TARGET = {
|
||||
'engine': ('docker', ''),
|
||||
'cpu': ('docker_container_cpu', 'so-soc'),
|
||||
'mem': ('docker_container_mem', 'so-soc'),
|
||||
'net': ('docker_container_net', 'so-soc'),
|
||||
'blkio': ('docker_container_blkio', 'so-soc'),
|
||||
'status': ('docker_container_status', 'so-soc'),
|
||||
'health': ('docker_container_health', 'so-nginx'),
|
||||
}
|
||||
|
||||
|
||||
class TestEverySetting(CollectorTestCase):
|
||||
"""Whatever is offered in defaults.yaml has to actually be collectable. These tests are driven
|
||||
off that file, so a new setting that is never wired up fails here rather than shipping."""
|
||||
|
||||
def setUp(self):
|
||||
super().setUp()
|
||||
# an exited container so status.finished_at has a value to report
|
||||
self.inspect.append(container('so-idstools', IDSTOOLS_ID, 0, status='exited',
|
||||
started='2026-09-24T10:00:00.000000000Z',
|
||||
finished='2026-09-24T10:00:02.000000000Z'))
|
||||
|
||||
def test_every_setting_emits_its_field_when_enabled(self):
|
||||
points = self.run_collector(all_enabled())
|
||||
for group, fields in shipped_defaults().items():
|
||||
if group == 'tags':
|
||||
continue
|
||||
measurement, name = GROUP_TARGET[group]
|
||||
if group == 'status':
|
||||
# finished_at only exists for a container that has actually exited
|
||||
_, exited = points[(measurement, 'so-idstools')]
|
||||
self.assertIn('finished_at', exited)
|
||||
_, emitted = points[(measurement, name)]
|
||||
for field in fields:
|
||||
if group == 'status' and field == 'finished_at':
|
||||
continue
|
||||
self.assertIn(field, emitted, '%s.%s is offered but never emitted' % (group, field))
|
||||
|
||||
def test_every_setting_is_individually_wired(self):
|
||||
# enabling one stat on its own must produce exactly that field, proving each toggle is
|
||||
# read rather than riding along with a neighbour
|
||||
for group, fields in shipped_defaults().items():
|
||||
if group == 'tags':
|
||||
continue
|
||||
measurement, name = GROUP_TARGET[group]
|
||||
for field in fields:
|
||||
settings = {other: {key: False for key in values} for other, values in shipped_defaults().items()}
|
||||
settings[group][field] = True
|
||||
target = 'so-idstools' if (group == 'status' and field == 'finished_at') else name
|
||||
points = self.run_collector(settings)
|
||||
self.assertIn((measurement, target), points, '%s.%s emitted no row' % (group, field))
|
||||
_, emitted = points[(measurement, target)]
|
||||
self.assertEqual(sorted(emitted), [field], '%s.%s did not emit itself alone' % (group, field))
|
||||
|
||||
def test_all_enabled_values_are_exact(self):
|
||||
points = self.run_collector(all_enabled())
|
||||
clock = os.sysconf('SC_CLK_TCK')
|
||||
expected = {
|
||||
('docker_container_cpu', 'so-soc'): {
|
||||
'usage_percent': '1.25', 'usage_total': '1000000u', 'usage_in_usermode': '600000u',
|
||||
'usage_in_kernelmode': '400000u', 'usage_system': '%du' % int(1000 * 10**9 / clock),
|
||||
'throttling_periods': '5u', 'throttling_throttled_periods': '2u',
|
||||
'throttling_throttled_time': '700000u', 'container_id': '"%s"' % SOC_ID,
|
||||
},
|
||||
('docker_container_mem', 'so-soc'): {
|
||||
'usage': '950u', 'limit': '4000u', 'max_usage': '2500u', 'active_anon': '300u',
|
||||
'active_file': '40u', 'inactive_anon': '200u', 'inactive_file': '50u',
|
||||
'unevictable': '0u', 'pgfault': '1234u', 'pgmajfault': '56u',
|
||||
'usage_percent': '23.75', 'container_id': '"%s"' % SOC_ID,
|
||||
},
|
||||
('docker_container_net', 'so-soc'): {
|
||||
'rx_bytes': '1500u', 'rx_packets': '15u', 'rx_errors': '1u', 'rx_dropped': '2u',
|
||||
'tx_bytes': '3000u', 'tx_packets': '30u', 'tx_errors': '3u', 'tx_dropped': '4u',
|
||||
'container_id': '"%s"' % SOC_ID,
|
||||
},
|
||||
('docker_container_blkio', 'so-soc'): {
|
||||
'io_service_bytes_recursive_read': '110u',
|
||||
'io_service_bytes_recursive_write': '220u', 'container_id': '"%s"' % SOC_ID,
|
||||
},
|
||||
('docker', ''): {
|
||||
'n_containers': '3i', 'n_containers_running': '3i', 'n_containers_stopped': '0i',
|
||||
'n_containers_paused': '0i', 'n_images': '9i', 'n_cpus': '8i', 'n_goroutines': '42i',
|
||||
'n_used_file_descriptors': '77i', 'n_listener_events': '1i',
|
||||
'memory_total': '16000000000i',
|
||||
},
|
||||
('docker_container_health', 'so-nginx'): {
|
||||
'health_status': '"healthy"', 'failing_streak': '0i',
|
||||
},
|
||||
}
|
||||
for key, fields in expected.items():
|
||||
_, emitted = points[key]
|
||||
for field, value in fields.items():
|
||||
self.assertEqual(emitted[field], value, '%s %s' % (key[0], field))
|
||||
|
||||
def test_status_values_are_exact(self):
|
||||
# uptime is relative to now, so it is checked separately from the fixed fields
|
||||
points = self.run_collector(all_enabled())
|
||||
_, running = points[('docker_container_status', 'so-soc')]
|
||||
self.assertEqual(running['pid'], '1001i')
|
||||
self.assertEqual(running['exitcode'], '0i')
|
||||
self.assertEqual(running['restart_count'], '0i')
|
||||
self.assertEqual(running['oomkilled'], 'false')
|
||||
self.assertEqual(running['container_id'], '"%s"' % SOC_ID)
|
||||
self.assertEqual(int(running['started_at'].rstrip('i')) // 10**9, 1790244000)
|
||||
self.assertGreater(int(running['uptime_ns'].rstrip('i')), 0)
|
||||
_, exited = points[('docker_container_status', 'so-idstools')]
|
||||
self.assertEqual(exited['uptime_ns'], '2000000000i')
|
||||
self.assertEqual(exited['finished_at'], '1790244002000000000i')
|
||||
|
||||
def test_tags_identity_toggle_controls_the_identity_tags(self):
|
||||
settings = shipped_defaults()
|
||||
settings['tags']['identity'] = False
|
||||
tags, _ = self.run_collector(settings)[('docker_container_cpu', 'so-soc')]
|
||||
self.assertEqual(sorted(tags), ['container_name', 'container_status', 'cpu'])
|
||||
settings['tags']['identity'] = True
|
||||
tags, _ = self.run_collector(settings)[('docker_container_cpu', 'so-soc')]
|
||||
self.assertEqual(sorted(tags), ['container_image', 'container_name', 'container_status',
|
||||
'container_version', 'cpu', 'engine_host', 'server_version'])
|
||||
|
||||
def test_identity_tags_cover_every_documented_tag(self):
|
||||
tags, _ = self.run_collector(all_enabled())[('docker_container_cpu', 'so-soc')]
|
||||
for tag in ('container_image', 'container_version', 'engine_host', 'server_version'):
|
||||
self.assertIn(tag, tags)
|
||||
|
||||
|
||||
class TestTemplate(unittest.TestCase):
|
||||
|
||||
def test_template_renders_to_valid_python_for_the_shipped_defaults(self):
|
||||
with open(TEMPLATE) as handle:
|
||||
source = handle.read()
|
||||
rendered = jinja2.Template(source, keep_trailing_newline=True).render(CONTAINER_STATS=shipped_defaults())
|
||||
compile(rendered, 'so-container-stats', 'exec')
|
||||
self.assertNotIn('{%', rendered)
|
||||
# the docker format strings must survive rendering untouched
|
||||
self.assertIn('{{json .}}', rendered)
|
||||
|
||||
def test_every_annotated_setting_exists_in_defaults(self):
|
||||
# SOC reads both trees; an annotation without a default cannot be reverted in the UI
|
||||
with open(os.path.join(HERE, '..', '..', 'soc_telegraf.yaml')) as handle:
|
||||
annotated = yaml.safe_load(handle)['telegraf']['container_stats']
|
||||
defaults = shipped_defaults()
|
||||
for group, fields in annotated.items():
|
||||
self.assertIn(group, defaults)
|
||||
for field in fields:
|
||||
self.assertIn(field, defaults[group], '%s.%s annotated but missing from defaults' % (group, field))
|
||||
|
||||
def test_every_default_setting_is_annotated_for_soc(self):
|
||||
with open(os.path.join(HERE, '..', '..', 'soc_telegraf.yaml')) as handle:
|
||||
annotated = yaml.safe_load(handle)['telegraf']['container_stats']
|
||||
for group, fields in shipped_defaults().items():
|
||||
self.assertIn(group, annotated)
|
||||
for field in fields:
|
||||
self.assertIn(field, annotated[group], '%s.%s missing a SOC annotation' % (group, field))
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
unittest.main()
|
||||
@@ -54,8 +54,8 @@
|
||||
{%
|
||||
do GLOBALS.update({
|
||||
'application_urls': {
|
||||
'hydra': 'http://' ~ GLOBALS.manager ~ ':4445/',
|
||||
'kratos': 'http://' ~ GLOBALS.manager ~ ':4434/',
|
||||
'hydra': 'http://' ~ DOCKERMERGED.containers['so-hydra'].ips['soauth'] ~ ':4445/',
|
||||
'kratos': 'http://' ~ DOCKERMERGED.containers['so-kratos'].ips['soauth'] ~ ':4434/',
|
||||
'elastic': 'https://' ~ GLOBALS.manager ~ ':9200/',
|
||||
'influxdb': 'https://' ~ GLOBALS.manager ~ ':8086/'
|
||||
}
|
||||
|
||||
@@ -101,8 +101,8 @@ zeek_sbin:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://zeek/tools/sbin
|
||||
- user: 939
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
|
||||
#zeek_sbin_jinja:
|
||||
|
||||
@@ -18,6 +18,7 @@ zeek:
|
||||
StatsLogEnable: 0
|
||||
StatsLogExpireInterval: 0
|
||||
StatusCmdShowAll: 0
|
||||
StopWait: 1
|
||||
CrashExpireInterval: 0
|
||||
SitePolicyScripts: local.zeek
|
||||
LogDir: /nsm/zeek/logs
|
||||
|
||||
@@ -9,9 +9,19 @@
|
||||
include:
|
||||
- zeek.sostatus
|
||||
|
||||
# Stop first so the entrypoint's SIGTERM trap can archive the final logs; docker_container.absent
|
||||
# with force is a 'docker rm -f', which never delivers SIGTERM. force stays so the state still
|
||||
# converges if the stop overruns.
|
||||
so-zeek_stopped:
|
||||
docker_container.stopped:
|
||||
- name: so-zeek
|
||||
- error_on_absent: False
|
||||
|
||||
so-zeek:
|
||||
docker_container.absent:
|
||||
- force: True
|
||||
- require:
|
||||
- docker_container: so-zeek_stopped
|
||||
|
||||
so-zeek_so-status.disabled:
|
||||
file.comment:
|
||||
|
||||
@@ -19,6 +19,10 @@ so-zeek:
|
||||
- restart_policy: unless-stopped
|
||||
- start: True
|
||||
- privileged: True
|
||||
# Docker's default 10s grace is not enough for the entrypoint's SIGTERM trap to run
|
||||
# 'zeekctl stop' and let StopWait archive the final logs. Overrunning it means SIGKILL,
|
||||
# which strands those logs in spool/tmp and marks every node crashed on the next start.
|
||||
- stop_timeout: 180
|
||||
{% if DOCKERMERGED.containers['so-zeek'].ulimits %}
|
||||
- ulimits:
|
||||
{% for ULIMIT in DOCKERMERGED.containers['so-zeek'].ulimits %}
|
||||
|
||||
@@ -99,6 +99,18 @@ zeek:
|
||||
regexFailureMessage: You must enter a whole number of days, or 0 to keep crash directories forever.
|
||||
helpLink: zeek
|
||||
advanced: True
|
||||
StopWait:
|
||||
description: >-
|
||||
Set to 1 to make "zeekctl stop" wait for the final logs to be archived instead of
|
||||
letting that finish in the background. Security Onion stops Zeek by stopping its
|
||||
container, so anything still running in the background is killed when the container
|
||||
exits - without this, the last logs of each run are stranded unarchived in
|
||||
/nsm/zeek/spool/tmp and never reach Elasticsearch. It is read only for that reason.
|
||||
regex: ^[01]$
|
||||
regexFailureMessage: You must enter 0 or 1.
|
||||
helpLink: zeek
|
||||
advanced: True
|
||||
readonly: True
|
||||
MinDiskSpace:
|
||||
description: >-
|
||||
Percentage of free disk space below which ZeekControl reports a warning, or 0 to disable the check
|
||||
|
||||
@@ -276,9 +276,20 @@ collect_dockernet() {
|
||||
whiptail_invalid_input
|
||||
whiptail_dockernet_sosnet "$DOCKERNET"
|
||||
done
|
||||
|
||||
whiptail_authnet_sosnet "$(adjacent_net "$DOCKERNET")"
|
||||
|
||||
while ! valid_ip4 "$AUTHNET" || [[ $AUTHNET =~ "172.17.0." ]] || [[ "$AUTHNET" == "$DOCKERNET" ]]; do
|
||||
whiptail_invalid_input
|
||||
whiptail_authnet_sosnet "$AUTHNET"
|
||||
done
|
||||
fi
|
||||
}
|
||||
|
||||
adjacent_net() {
|
||||
echo "$1" | awk -F'.' '{ printf "%s.%s.%s.%s", $1, $2, ($3 + 1) % 256, $4 }'
|
||||
}
|
||||
|
||||
collect_gateway() {
|
||||
whiptail_management_interface_gateway
|
||||
|
||||
@@ -1399,6 +1410,15 @@ docker_pillar() {
|
||||
"docker:"\
|
||||
" range: '$DOCKERNET/24'"\
|
||||
" gateway: '$DOCKERGATEWAY'" > $docker_pillar_file
|
||||
|
||||
if [ ! -z "$AUTHNET" ]; then
|
||||
AUTHGATEWAY=$(echo $AUTHNET | awk -F'.' '{print $1,$2,$3,1}' OFS='.')
|
||||
printf '%s\n'\
|
||||
" networks:"\
|
||||
" soauth:"\
|
||||
" range: '$AUTHNET/24'"\
|
||||
" gateway: '$AUTHGATEWAY'" >> $docker_pillar_file
|
||||
fi
|
||||
fi
|
||||
}
|
||||
|
||||
@@ -1591,6 +1611,7 @@ reserve_group_ids() {
|
||||
logCmd "groupadd -g 949 elastic-agent"
|
||||
logCmd "groupadd -g 947 elastic-fleet"
|
||||
logCmd "groupadd -g 960 kafka"
|
||||
logCmd "groupadd -g 961 somon"
|
||||
}
|
||||
|
||||
reserve_ports() {
|
||||
@@ -2105,6 +2126,8 @@ setup_salt_master_dirs() {
|
||||
|
||||
info "Chown the salt dirs on the manager for socore"
|
||||
logCmd "chown -R socore:socore /opt/so"
|
||||
# The default tree is root-executed code; SOC reads it but never writes it.
|
||||
logCmd "chown -R root:root $default_salt_dir"
|
||||
}
|
||||
|
||||
set_progress_str() {
|
||||
|
||||
@@ -365,6 +365,18 @@ whiptail_dockernet_sosnet() {
|
||||
|
||||
}
|
||||
|
||||
whiptail_authnet_sosnet() {
|
||||
|
||||
[ -n "$TESTING" ] && return
|
||||
|
||||
AUTHNET=$(whiptail --title "$whiptail_title" --inputbox \
|
||||
"\nEnter a second /24 size network range WITHOUT the /24 suffix. The authentication services are isolated on their own network so that the identity provider is not reachable from other containers. It must not overlap the range you just entered, and any range within 172.17.0.0/24 cannot be used." 13 65 "$1" 3>&1 1>&2 2>&3)
|
||||
|
||||
local exitstatus=$?
|
||||
whiptail_check_exitstatus $exitstatus
|
||||
|
||||
}
|
||||
|
||||
whiptail_end_settings() {
|
||||
[ -n "$TESTING" ] && return
|
||||
|
||||
@@ -427,6 +439,7 @@ whiptail_end_settings() {
|
||||
[[ -n $WEBUSER ]] && __append_end_msg "Web User: $WEBUSER"
|
||||
|
||||
[[ -n $DOCKERNET ]] && __append_end_msg "Docker network: $DOCKERNET/24"
|
||||
[[ -n $AUTHNET ]] && __append_end_msg "Authentication network: $AUTHNET/24"
|
||||
if [[ ${#ntp_servers[@]} -gt 0 ]]; then
|
||||
__append_end_msg "NTP Servers:"
|
||||
for server in "${ntp_servers[@]}"; do
|
||||
|
||||
Binary file not shown.
Reference in New Issue
Block a user