mirror of
https://github.com/Security-Onion-Solutions/securityonion.git
synced 2026-09-30 19:47:17 +02:00
Compare commits
354
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
72f60fcaa9 | ||
|
|
2684a5ca95 | ||
|
|
bcee63bde5 | ||
|
|
6ce89eb323 | ||
|
|
ebab4b0d90 | ||
|
|
db60c27da2 | ||
|
|
f4defdfde0 | ||
|
|
36652e8f23 | ||
|
|
7bef194540 | ||
|
|
e2bf2837fe | ||
|
|
06704dad22 | ||
|
|
47fe0758d0 | ||
|
|
b71fd93f9d | ||
|
|
d9eff9aa9e | ||
|
|
d8884dbd99 | ||
|
|
6c0d4c15e8 | ||
|
|
2e2f62f265 | ||
|
|
b7a11a525c | ||
|
|
8eef95ea3e | ||
|
|
65e261475d | ||
|
|
bbc28c88b7 | ||
|
|
edaacf79a7 | ||
|
|
ecc643cd33 | ||
|
|
08aaf7948e | ||
|
|
bb57545d08 | ||
|
|
0f7adbbecc | ||
|
|
aeb4fe8f50 | ||
|
|
f3aa39c5a4 | ||
|
|
24077ba974 | ||
|
|
b3567405f9 | ||
|
|
1fc5bb7afa | ||
|
|
1f1d3ded41 | ||
|
|
1e86be11b2 | ||
|
|
f4518e2620 | ||
|
|
a9f7ffc3fe | ||
|
|
b018277d68 | ||
|
|
3be603e203 | ||
|
|
84cd966736 | ||
|
|
fee401a912 | ||
|
|
496b61966f | ||
|
|
52037314be | ||
|
|
9c12c10f96 | ||
|
|
9fc9be2cc9 | ||
|
|
7245843a3c | ||
|
|
a1d17417ea | ||
|
|
bee03d5bae | ||
|
|
56e3e44d04 | ||
|
|
32d1274b80 | ||
|
|
1624e8c094 | ||
|
|
3f3f091a7f | ||
|
|
cb48909578 | ||
|
|
3057775770 | ||
|
|
e4e8b90b9c | ||
|
|
223ace6ff3 | ||
|
|
66e7863336 | ||
|
|
8f253d17a6 | ||
|
|
9652a2053b | ||
|
|
191ee159ef | ||
|
|
a8bfe955a5 | ||
|
|
bd354abe83 | ||
|
|
37782fb45c | ||
|
|
cf3a4ebc27 | ||
|
|
c49008a413 | ||
|
|
fcbea1a1c4 | ||
|
|
6736f9c3a0 | ||
|
|
8f14e96215 | ||
|
|
86ae51b2b5 | ||
|
|
2cde9abec2 | ||
|
|
57ce2cea0d | ||
|
|
ff7555d93b | ||
|
|
65b84026a8 | ||
|
|
721b1d6207 | ||
|
|
a74046fc24 | ||
|
|
ea539f8679 | ||
|
|
54bb0e2c12 | ||
|
|
67b4d82f62 | ||
|
|
30574fdbb9 | ||
|
|
64771e9e1d | ||
|
|
01ca33b90a | ||
|
|
599d19215b | ||
|
|
ee671e7ec9 | ||
|
|
42d429a11e | ||
|
|
332a5d11bc | ||
|
|
42a62c90a5 | ||
|
|
fb7e065590 | ||
|
|
bce6b0c1fe | ||
|
|
b9ba7df80c | ||
|
|
97fddc0719 | ||
|
|
a5deee1444 | ||
|
|
3585ccca79 | ||
|
|
dd035beec4 | ||
|
|
0dbb7803ef | ||
|
|
30deb00277 | ||
|
|
192363bc2f | ||
|
|
3d8f86883a | ||
|
|
96bef89ba9 | ||
|
|
a244640539 | ||
|
|
fdb975fdef | ||
|
|
f8401bef37 | ||
|
|
ca96a15091 | ||
|
|
1bac9a218e | ||
|
|
4786d359fb | ||
|
|
d771fbc444 | ||
|
|
85ab4c69e5 | ||
|
|
cb8e576d6b | ||
|
|
fae1754fec | ||
|
|
d33eb70af6 | ||
|
|
f45dcfdf73 | ||
|
|
62da505ea7 | ||
|
|
7e5b6f276f | ||
|
|
376d29e376 | ||
|
|
665772adb8 | ||
|
|
094b4d5e86 | ||
|
|
3a3667996c | ||
|
|
dfa6f0b454 | ||
|
|
9f6679c043 | ||
|
|
fb7d162de1 | ||
|
|
0ee8aa8079 | ||
|
|
a8785870af | ||
|
|
00f948e4d2 | ||
|
|
f6ab92fc24 | ||
|
|
5e9fd4a45b | ||
|
|
e7f54b49c4 | ||
|
|
a127ef5714 | ||
|
|
fcb889a30c | ||
|
|
99e1d83358 | ||
|
|
60052e0910 | ||
|
|
99c3c7f8aa | ||
|
|
cef1dcfcee | ||
|
|
cec3f7ed57 | ||
|
|
f566a8965d | ||
|
|
905cc1c0dd | ||
|
|
12744353fb | ||
|
|
52fc0cb828 | ||
|
|
247d9cdb34 | ||
|
|
088b761190 | ||
|
|
5c3a69d742 | ||
|
|
c1f256e630 | ||
|
|
2dcc81ea7d | ||
|
|
356da00395 | ||
|
|
d2ff29b7a8 | ||
|
|
7bdaf9338e | ||
|
|
35f545a858 | ||
|
|
dff3d76efd | ||
|
|
6f3f58bd70 | ||
|
|
d62c53fc92 | ||
|
|
2f2187f714 | ||
|
|
6c37bc1f9b | ||
|
|
c9a041ddb4 | ||
|
|
de3306e73c | ||
|
|
4b74e2c320 | ||
|
|
3744c0bd6c | ||
|
|
563b9d7c3b | ||
|
|
ec91f9b830 | ||
|
|
7f3f99880f | ||
|
|
3e7f508620 | ||
|
|
c4555a5514 | ||
|
|
d4d63fa60a | ||
|
|
dcb931b97c | ||
|
|
8e6b16bde0 | ||
|
|
63692aa1a0 | ||
|
|
2663ca87a2 | ||
|
|
a337a3e4f6 | ||
|
|
ea502e29d0 | ||
|
|
af222eed08 | ||
|
|
83e55ab0f3 | ||
|
|
ff82cc32a0 | ||
|
|
721d6483f0 | ||
|
|
a88562a348 | ||
|
|
64d7383233 | ||
|
|
7400e3dffa | ||
|
|
792b801086 | ||
|
|
3991e485c0 | ||
|
|
a87a910585 | ||
|
|
ba0dd38f4e | ||
|
|
b3467854a8 | ||
|
|
9ebf93cc26 | ||
|
|
d69234146e | ||
|
|
706d46b395 | ||
|
|
539389c78e | ||
|
|
65e81d3b3a | ||
|
|
546462c77f | ||
|
|
a7ddb7a975 | ||
|
|
fe4f7ad2f7 | ||
|
|
ee1d2167e8 | ||
|
|
2d0ea48c39 | ||
|
|
23d92316c1 | ||
|
|
668ab447a2 | ||
|
|
e5346af068 | ||
|
|
9762523849 | ||
|
|
e8ab6433ab | ||
|
|
9c20ef60f4 | ||
|
|
a2a4d9314d | ||
|
|
6abf382ea8 | ||
|
|
5d36d00dec | ||
|
|
f1f672892e | ||
|
|
a8053e2c9d | ||
|
|
36833fdad1 | ||
|
|
a92d10a1e3 | ||
|
|
d3da6b3939 | ||
|
|
e998a21b4d | ||
|
|
d77760c268 | ||
|
|
4f7ad76d5b | ||
|
|
41dba204c3 | ||
|
|
12bb16e89c | ||
|
|
45f1a1b8b1 | ||
|
|
23a9daf7a2 | ||
|
|
d3c6fdce7e | ||
|
|
c9642489d3 | ||
|
|
7b080797b4 | ||
|
|
d94c16eea1 | ||
|
|
3e3d409c42 | ||
|
|
a45ca12076 | ||
|
|
2e141a1ad7 | ||
|
|
4de8f0208f | ||
|
|
4aabf7d638 | ||
|
|
812310088e | ||
|
|
4e17390cfa | ||
|
|
57d629683d | ||
|
|
3666b5b0de | ||
|
|
7f64f143d7 | ||
|
|
162c66a705 | ||
|
|
14d11cc180 | ||
|
|
112fcf7804 | ||
|
|
120b426a79 | ||
|
|
72fd754a92 | ||
|
|
d5fddafa6a | ||
|
|
0c83d4e1fe | ||
|
|
baca444a7e | ||
|
|
4f5af93b38 | ||
|
|
b109ca4e9b | ||
|
|
e5969a12aa | ||
|
|
4d97b562eb | ||
|
|
f2d81cea3f | ||
|
|
6bd3c414bb | ||
|
|
16f958dac0 | ||
|
|
a57ff5f89e | ||
|
|
e2513daddc | ||
|
|
01a873b2d9 | ||
|
|
19957d9530 | ||
|
|
e4c14a9294 | ||
|
|
445ae58919 | ||
|
|
382dee1d06 | ||
|
|
f4aa9932ff | ||
|
|
4871098278 | ||
|
|
7b32c73da8 | ||
|
|
334978ad92 | ||
|
|
a21186ccce | ||
|
|
f77fad8087 | ||
|
|
963e475d1a | ||
|
|
ac46636196 | ||
|
|
f3d8bae13d | ||
|
|
894d323323 | ||
|
|
26eb8c3c18 | ||
|
|
4e1935f8a0 | ||
|
|
c4c1464b2a | ||
|
|
8d168e9661 | ||
|
|
f542d1e7ce | ||
|
|
b104f0955a | ||
|
|
89f4950521 | ||
|
|
387781c629 | ||
|
|
011749ad09 | ||
|
|
ad78e84ccd | ||
|
|
c4295b4e0a | ||
|
|
c2075ddafb | ||
|
|
4886034fef | ||
|
|
48a7d66964 | ||
|
|
6876b25280 | ||
|
|
30f3bddb8b | ||
|
|
811b799b0b | ||
|
|
aaea6dbd58 | ||
|
|
8095b82841 | ||
|
|
141116f550 | ||
|
|
f6d3cbe08d | ||
|
|
9e7e6edae0 | ||
|
|
6f61e7c901 | ||
|
|
cc2bfc26e2 | ||
|
|
073e32520b | ||
|
|
5867b50720 | ||
|
|
8a16ead33d | ||
|
|
f9b154ccef | ||
|
|
3503d0c33d | ||
|
|
517538a9a7 | ||
|
|
23c74f1727 | ||
|
|
b76f9d022e | ||
|
|
02318f065c | ||
|
|
f958212bea | ||
|
|
376607d292 | ||
|
|
186bf86e99 | ||
|
|
bd70dd53fb | ||
|
|
be7d8a2aa7 | ||
|
|
5178d5fd0e | ||
|
|
fee62ab976 | ||
|
|
618712469e | ||
|
|
8b488f9226 | ||
|
|
1657480d31 | ||
|
|
63d4061500 | ||
|
|
405dc52587 | ||
|
|
e42f7cd6fc | ||
|
|
8167ae3282 | ||
|
|
2cd889782d | ||
|
|
87a5639643 | ||
|
|
ed533efb7b | ||
|
|
5af6c56996 | ||
|
|
99e9fc1c3b | ||
|
|
89e6a746c8 | ||
|
|
52885e28c5 | ||
|
|
fbeac25ee9 | ||
|
|
8a3f5d0f81 | ||
|
|
9a71f64a35 | ||
|
|
40c02b3149 | ||
|
|
5fd5df54b4 | ||
|
|
66a1141b84 | ||
|
|
3310e19ee4 | ||
|
|
f441d98e71 | ||
|
|
a330bea25e | ||
|
|
33c24cd136 | ||
|
|
12f4447875 | ||
|
|
da94788255 | ||
|
|
fa2ae1b87f | ||
|
|
5bf9751adf | ||
|
|
3effdbc91e | ||
|
|
8836529496 | ||
|
|
b09c3776b7 | ||
|
|
dfdb1fbaeb | ||
|
|
61aa963a2d | ||
|
|
d71e80cf66 | ||
|
|
33a116357d | ||
|
|
8c17ae0f66 | ||
|
|
f54939b444 | ||
|
|
d48a22e37e | ||
|
|
6393d08e86 | ||
|
|
52791204e4 | ||
|
|
730c828bec | ||
|
|
b4e5171415 | ||
|
|
84decc1db6 | ||
|
|
7d4d6a0756 | ||
|
|
66c0a662fc | ||
|
|
778cc055ea | ||
|
|
932deab751 | ||
|
|
1281f0ee37 | ||
|
|
f774334b6c | ||
|
|
7fcace34c4 | ||
|
|
9541024eb7 | ||
|
|
0d166ef732 | ||
|
|
f7d2994f8b | ||
|
|
8f0757606d | ||
|
|
0a8f2e01a0 | ||
|
|
4546d7bc52 | ||
|
|
17849d8758 | ||
|
|
d3d30a587c | ||
|
|
034711d148 | ||
|
|
a0cf0489d6 | ||
|
|
613d31c8a6 |
@@ -12,6 +12,8 @@ body:
|
||||
- 3.0.0
|
||||
- 3.1.0
|
||||
- 3.2.0
|
||||
- 3.3.0
|
||||
- 3.4.0
|
||||
- Other (please provide detail below)
|
||||
validations:
|
||||
required: true
|
||||
|
||||
@@ -5,6 +5,10 @@ on:
|
||||
paths:
|
||||
- "salt/sensoroni/files/analyzers/**"
|
||||
- "salt/manager/tools/sbin/**"
|
||||
- "salt/_beacons/**"
|
||||
- "salt/telegraf/tools/sbin_jinja/**"
|
||||
- "salt/telegraf/defaults.yaml"
|
||||
- "salt/telegraf/soc_telegraf.yaml"
|
||||
|
||||
jobs:
|
||||
build:
|
||||
@@ -14,7 +18,7 @@ jobs:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
python-version: ["3.14"]
|
||||
python-code-path: ["salt/sensoroni/files/analyzers", "salt/manager/tools/sbin"]
|
||||
python-code-path: ["salt/sensoroni/files/analyzers", "salt/manager/tools/sbin", "salt/_beacons"]
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
@@ -33,3 +37,25 @@ jobs:
|
||||
- name: Test with pytest
|
||||
run: |
|
||||
PYTHONPATH=${{ matrix.python-code-path }} pytest ${{ matrix.python-code-path }} --cov=${{ matrix.python-code-path }} --doctest-modules --cov-report=term --cov-fail-under=100 --cov-config=pytest.ini
|
||||
|
||||
telegraf-collector:
|
||||
# so-container-stats is a jinja template rather than an importable module, so it gets its
|
||||
# own job: the test renders it the way salt does, then drives it with a faked docker engine
|
||||
# and cgroup tree. No container runtime is needed.
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v3
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@v3
|
||||
with:
|
||||
python-version: "3.14"
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
python -m pip install --upgrade pip
|
||||
python -m pip install flake8 pytest jinja2 pyyaml
|
||||
- name: Lint with flake8
|
||||
run: |
|
||||
flake8 salt/telegraf/tools/sbin_jinja/so-container-stats_test.py --config=pytest.ini
|
||||
- name: Test with pytest
|
||||
run: |
|
||||
pytest salt/telegraf/tools/sbin_jinja/so-container-stats_test.py -v
|
||||
|
||||
+11
-11
@@ -1,17 +1,17 @@
|
||||
### 3.1.0-20260528 ISO image released on 2026/05/28
|
||||
### 3.3.0-20260911 ISO image released on 2026/09/11
|
||||
|
||||
|
||||
### Download and Verify
|
||||
|
||||
3.1.0-20260528 ISO image:
|
||||
https://download.securityonion.net/file/securityonion/securityonion-3.1.0-20260528.iso
|
||||
3.3.0-20260911 ISO image:
|
||||
https://download.securityonion.net/file/securityonion/securityonion-3.3.0-20260911.iso
|
||||
|
||||
MD5: 9D6FF58DEEE24089D722C73169765B3E
|
||||
SHA1: 2B8B816B6CEC3B7F96B3C5E040EBF502DD2C412F
|
||||
SHA256: 62FAB57E247C843D6A04F0796D8162C732B65D82FC3E4A59D087135B9FD32912
|
||||
MD5: 12B18433D3A2198A185892FF79CF638F
|
||||
SHA1: 2B3C2E1FA7A78ED1F956E7EDCC12E32593C14EEE
|
||||
SHA256: 0938C73B76CE30EC9E4394D312C79EA7CAC721B6818541697279A6221F7D870D
|
||||
|
||||
Signature for ISO image:
|
||||
https://github.com/Security-Onion-Solutions/securityonion/raw/3/main/sigs/securityonion-3.1.0-20260528.iso.sig
|
||||
https://github.com/Security-Onion-Solutions/securityonion/raw/3/main/sigs/securityonion-3.3.0-20260911.iso.sig
|
||||
|
||||
Signing key:
|
||||
https://raw.githubusercontent.com/Security-Onion-Solutions/securityonion/3/main/KEYS
|
||||
@@ -25,22 +25,22 @@ wget https://raw.githubusercontent.com/Security-Onion-Solutions/securityonion/3/
|
||||
|
||||
Download the signature file for the ISO:
|
||||
```
|
||||
wget https://github.com/Security-Onion-Solutions/securityonion/raw/3/main/sigs/securityonion-3.1.0-20260528.iso.sig
|
||||
wget https://github.com/Security-Onion-Solutions/securityonion/raw/3/main/sigs/securityonion-3.3.0-20260911.iso.sig
|
||||
```
|
||||
|
||||
Download the ISO image:
|
||||
```
|
||||
wget https://download.securityonion.net/file/securityonion/securityonion-3.1.0-20260528.iso
|
||||
wget https://download.securityonion.net/file/securityonion/securityonion-3.3.0-20260911.iso
|
||||
```
|
||||
|
||||
Verify the downloaded ISO image using the signature file:
|
||||
```
|
||||
gpg --verify securityonion-3.1.0-20260528.iso.sig securityonion-3.1.0-20260528.iso
|
||||
gpg --verify securityonion-3.3.0-20260911.iso.sig securityonion-3.3.0-20260911.iso
|
||||
```
|
||||
|
||||
The output should show "Good signature" and the Primary key fingerprint should match what's shown below:
|
||||
```
|
||||
gpg: Signature made Wed 27 May 2026 03:03:59 PM EDT using RSA key ID FE507013
|
||||
gpg: Signature made Fri 11 Sep 2026 11:23:56 AM EDT using RSA key ID FE507013
|
||||
gpg: Good signature from "Security Onion Solutions, LLC <info@securityonionsolutions.com>"
|
||||
gpg: WARNING: This key is not certified with a trusted signature!
|
||||
gpg: There is no indication that the signature belongs to the owner.
|
||||
|
||||
@@ -3,6 +3,8 @@ base:
|
||||
- ca
|
||||
- global.soc_global
|
||||
- global.adv_global
|
||||
- salt.soc_salt
|
||||
- salt.adv_salt
|
||||
- docker.soc_docker
|
||||
- docker.adv_docker
|
||||
- influxdb.token
|
||||
|
||||
@@ -0,0 +1,145 @@
|
||||
# Copyright Security Onion Solutions LLC and/or licensed to Security Onion Solutions LLC under one
|
||||
# or more contributor license agreements. Licensed under the Elastic License 2.0 as shown at
|
||||
# https://securityonion.net/license; you may not use this file except in compliance with the
|
||||
# Elastic License 2.0.
|
||||
|
||||
# Custom salt beacon that watches hand-edited directories under
|
||||
# /opt/so/saltstack/local/salt/ for changes and emits a beacon event per changed
|
||||
# directory. This replaces the stock salt `inotify` beacon, which leaks a kernel
|
||||
# inotify instance every time the minion rebuilds the beacon loader's __context__
|
||||
# (orphaning the old pyinotify.Notifier without closing it) until
|
||||
# fs.inotify.max_user_instances is exhausted and the beacon dies with EMFILE.
|
||||
# Polling holds zero inotify instances, so the leak is impossible, and it keeps
|
||||
# firing during state runs (no blackout).
|
||||
#
|
||||
# Detection is poll-based with a per-directory fingerprint persisted to
|
||||
# WATERMARK_DIR: each pass walks the directory and hashes every file's
|
||||
# (relpath, st_mtime_ns, st_size), which catches content writes, additions,
|
||||
# moves, and deletions. A change in the digest emits one event; an unchanged
|
||||
# digest emits nothing. This makes it self-healing (a missed poll simply catches
|
||||
# up on the next one).
|
||||
#
|
||||
# Each emitted event carries the watched directory path under the configured tag
|
||||
# (e.g. salt/beacon/<minion>/local_files_beacon/zeek); the push_files reactor
|
||||
# looks the tag up in salt/reactor/pillar_push_map.yaml and writes a push intent,
|
||||
# after which the existing so-push-drainer / orch.push_batch pipeline takes over
|
||||
# unchanged.
|
||||
|
||||
import hashlib
|
||||
import logging
|
||||
import os
|
||||
import re
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
WATERMARK_DIR = '/opt/so/state'
|
||||
|
||||
# Temp/editor files that should not trigger a push. Mirrors the exclude regexes
|
||||
# the inotify beacon used. Matched against the full pathname.
|
||||
EXCLUDES = [
|
||||
re.compile(r'\.sw[a-z]$'),
|
||||
re.compile(r'~$'),
|
||||
re.compile(r'/4913$'),
|
||||
re.compile(r'/\.#'),
|
||||
]
|
||||
|
||||
|
||||
def __virtual__():
|
||||
return True
|
||||
|
||||
|
||||
def validate(config):
|
||||
return True, 'valid'
|
||||
|
||||
|
||||
def _paths_from_config(config):
|
||||
# The beacon config arrives as a list of single-key dicts (salt beacon style).
|
||||
# Merge it and return the {dir: tag} mapping under the 'paths' key.
|
||||
merged = {}
|
||||
if isinstance(config, list):
|
||||
for item in config:
|
||||
if isinstance(item, dict):
|
||||
merged.update(item)
|
||||
elif isinstance(config, dict):
|
||||
merged = config
|
||||
paths = merged.get('paths', {})
|
||||
return paths if isinstance(paths, dict) else {}
|
||||
|
||||
|
||||
def _excluded(pathname):
|
||||
for pattern in EXCLUDES:
|
||||
if pattern.search(pathname):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def _fingerprint(directory):
|
||||
# Stat-only walk; hash each file's (relpath, mtime_ns, size). Returns a hex
|
||||
# digest, or the digest of an empty tree if the directory does not exist.
|
||||
h = hashlib.sha1()
|
||||
if os.path.isdir(directory):
|
||||
entries = []
|
||||
for root, dirs, files in os.walk(directory):
|
||||
# zkg packages are git clones; .git churn would fire a state apply on its own.
|
||||
dirs[:] = [d for d in dirs if d != '.git']
|
||||
for name in files:
|
||||
full = os.path.join(root, name)
|
||||
if _excluded(full):
|
||||
continue
|
||||
try:
|
||||
st = os.stat(full)
|
||||
except OSError:
|
||||
continue
|
||||
rel = os.path.relpath(full, directory)
|
||||
entries.append('%s\0%d\0%d' % (rel, st.st_mtime_ns, st.st_size))
|
||||
for line in sorted(entries):
|
||||
h.update(line.encode('utf-8', 'surrogateescape'))
|
||||
h.update(b'\n')
|
||||
return h.hexdigest()
|
||||
|
||||
|
||||
def _watermark_file(tag, directory):
|
||||
# Keyed by directory: zeek/policy and zeek/zkg share the tag `zeek`.
|
||||
scope = hashlib.sha1(directory.encode('utf-8', 'surrogateescape')).hexdigest()[:12]
|
||||
return os.path.join(WATERMARK_DIR, 'local_files_beacon_%s_%s.hash' % (tag, scope))
|
||||
|
||||
|
||||
def _read_watermark(tag, directory):
|
||||
try:
|
||||
with open(_watermark_file(tag, directory), 'r') as f:
|
||||
return (f.read() or '').strip() or None
|
||||
except IOError:
|
||||
return None
|
||||
|
||||
|
||||
def _write_watermark(tag, directory, digest):
|
||||
path = _watermark_file(tag, directory)
|
||||
try:
|
||||
os.makedirs(WATERMARK_DIR, exist_ok=True)
|
||||
tmp = path + '.tmp'
|
||||
with open(tmp, 'w') as f:
|
||||
f.write(digest)
|
||||
os.rename(tmp, path)
|
||||
except OSError:
|
||||
log.exception('local_files_beacon: failed to persist watermark to %s', path)
|
||||
|
||||
|
||||
def beacon(config):
|
||||
retval = []
|
||||
|
||||
for directory, tag in _paths_from_config(config).items():
|
||||
digest = _fingerprint(directory)
|
||||
previous = _read_watermark(tag, directory)
|
||||
|
||||
# First run / missing watermark: seed the digest and emit nothing so a
|
||||
# fresh host does not fire a spurious fleetwide push.
|
||||
if previous is None:
|
||||
_write_watermark(tag, directory, digest)
|
||||
continue
|
||||
|
||||
if digest != previous:
|
||||
_write_watermark(tag, directory, digest)
|
||||
retval.append({'tag': tag, 'path': directory})
|
||||
log.info('local_files_beacon: change detected in %s, emitting %s', directory, tag)
|
||||
|
||||
return retval
|
||||
@@ -0,0 +1,221 @@
|
||||
# Copyright Security Onion Solutions LLC and/or licensed to Security Onion Solutions LLC under one
|
||||
# or more contributor license agreements. Licensed under the Elastic License 2.0 as shown at
|
||||
# https://securityonion.net/license; you may not use this file except in compliance with the
|
||||
# Elastic License 2.0.
|
||||
|
||||
import hashlib
|
||||
import os
|
||||
import shutil
|
||||
import tempfile
|
||||
import unittest
|
||||
from unittest.mock import patch
|
||||
|
||||
import local_files_beacon
|
||||
|
||||
|
||||
class TestRulesBeacon(unittest.TestCase):
|
||||
|
||||
def setUp(self):
|
||||
# Isolate all on-disk state (watermarks and the dirs we fingerprint) in a
|
||||
# throwaway tree, and point WATERMARK_DIR at it so the real read/write
|
||||
# helpers run against actual files.
|
||||
self.tmpdir = tempfile.mkdtemp()
|
||||
self.state = os.path.join(self.tmpdir, 'state')
|
||||
patcher = patch.object(local_files_beacon, 'WATERMARK_DIR', self.state)
|
||||
patcher.start()
|
||||
self.addCleanup(patcher.stop)
|
||||
|
||||
def tearDown(self):
|
||||
shutil.rmtree(self.tmpdir, ignore_errors=True)
|
||||
|
||||
def _make_dir(self, name, files=None):
|
||||
path = os.path.join(self.tmpdir, name)
|
||||
os.makedirs(path, exist_ok=True)
|
||||
for fname, content in (files or {}).items():
|
||||
with open(os.path.join(path, fname), 'w') as f:
|
||||
f.write(content)
|
||||
return path
|
||||
|
||||
# -- trivial contract -------------------------------------------------
|
||||
|
||||
def test_virtual_returns_true(self):
|
||||
self.assertTrue(local_files_beacon.__virtual__())
|
||||
|
||||
def test_validate_returns_valid(self):
|
||||
self.assertEqual(local_files_beacon.validate({}), (True, 'valid'))
|
||||
|
||||
# -- _paths_from_config -----------------------------------------------
|
||||
|
||||
def test_paths_from_config_list_of_dicts(self):
|
||||
config = [{'interval': 10}, {'paths': {'/a': 'suricata', '/b': 'strelka'}}]
|
||||
self.assertEqual(
|
||||
local_files_beacon._paths_from_config(config),
|
||||
{'/a': 'suricata', '/b': 'strelka'},
|
||||
)
|
||||
|
||||
def test_paths_from_config_plain_dict(self):
|
||||
self.assertEqual(
|
||||
local_files_beacon._paths_from_config({'paths': {'/a': 'suricata'}}),
|
||||
{'/a': 'suricata'},
|
||||
)
|
||||
|
||||
def test_paths_from_config_skips_non_dict_items(self):
|
||||
self.assertEqual(local_files_beacon._paths_from_config(['bogus', 42]), {})
|
||||
|
||||
def test_paths_from_config_paths_not_a_dict(self):
|
||||
self.assertEqual(local_files_beacon._paths_from_config({'paths': 'nope'}), {})
|
||||
|
||||
def test_paths_from_config_unexpected_type(self):
|
||||
self.assertEqual(local_files_beacon._paths_from_config('nonsense'), {})
|
||||
|
||||
# -- _excluded --------------------------------------------------------
|
||||
|
||||
def test_excluded_matches_temp_and_editor_files(self):
|
||||
for pathname in ('/rules/foo.swp', '/rules/foo~', '/rules/4913', '/rules/.#foo'):
|
||||
self.assertTrue(local_files_beacon._excluded(pathname), pathname)
|
||||
|
||||
def test_excluded_allows_real_rule_files(self):
|
||||
self.assertFalse(local_files_beacon._excluded('/rules/suricata.rules'))
|
||||
|
||||
# -- _fingerprint -----------------------------------------------------
|
||||
|
||||
def test_fingerprint_missing_dir_is_empty_tree_digest(self):
|
||||
missing = os.path.join(self.tmpdir, 'does-not-exist')
|
||||
self.assertEqual(local_files_beacon._fingerprint(missing), hashlib.sha1().hexdigest())
|
||||
|
||||
def test_fingerprint_changes_when_content_changes(self):
|
||||
d = self._make_dir('rules', {'a.rules': 'alert'})
|
||||
before = local_files_beacon._fingerprint(d)
|
||||
with open(os.path.join(d, 'a.rules'), 'w') as f:
|
||||
f.write('alert tcp any any -> any any') # different size
|
||||
self.assertNotEqual(local_files_beacon._fingerprint(d), before)
|
||||
|
||||
def test_fingerprint_ignores_excluded_files(self):
|
||||
d = self._make_dir('rules', {'a.rules': 'alert'})
|
||||
before = local_files_beacon._fingerprint(d)
|
||||
with open(os.path.join(d, 'a.rules.swp'), 'w') as f:
|
||||
f.write('editor swap')
|
||||
self.assertEqual(local_files_beacon._fingerprint(d), before)
|
||||
|
||||
def test_fingerprint_skips_unstatable_entries(self):
|
||||
# A dangling symlink appears in os.walk's file list but os.stat raises
|
||||
# OSError, exercising the except-continue path.
|
||||
d = self._make_dir('rules', {'a.rules': 'alert'})
|
||||
good = local_files_beacon._fingerprint(d)
|
||||
os.symlink(os.path.join(d, 'missing-target'), os.path.join(d, 'broken.link'))
|
||||
self.assertEqual(local_files_beacon._fingerprint(d), good)
|
||||
|
||||
def test_fingerprint_prunes_git_metadata(self):
|
||||
# zkg packages are git clones, so the watched tree carries .git.
|
||||
d = self._make_dir('zkg', {'pkg.zeek': 'print 1;'})
|
||||
before = local_files_beacon._fingerprint(d)
|
||||
git_dir = os.path.join(d, 'pkg', '.git', 'refs', 'heads')
|
||||
os.makedirs(git_dir)
|
||||
with open(os.path.join(git_dir, 'main'), 'w') as f:
|
||||
f.write('0' * 40)
|
||||
self.assertEqual(local_files_beacon._fingerprint(d), before)
|
||||
|
||||
def test_fingerprint_still_sees_worktree_next_to_git(self):
|
||||
d = self._make_dir('zkg', {'pkg.zeek': 'print 1;'})
|
||||
os.makedirs(os.path.join(d, 'pkg', '.git'))
|
||||
before = local_files_beacon._fingerprint(d)
|
||||
with open(os.path.join(d, 'pkg', 'scripts.zeek'), 'w') as f:
|
||||
f.write('print 2;')
|
||||
self.assertNotEqual(local_files_beacon._fingerprint(d), before)
|
||||
|
||||
# -- _read_watermark / _write_watermark -------------------------------
|
||||
|
||||
def test_watermark_round_trip(self):
|
||||
local_files_beacon._write_watermark('suricata', '/rules/suricata', 'deadbeef')
|
||||
self.assertEqual(
|
||||
local_files_beacon._read_watermark('suricata', '/rules/suricata'), 'deadbeef')
|
||||
|
||||
def test_read_watermark_missing_returns_none(self):
|
||||
self.assertIsNone(local_files_beacon._read_watermark('suricata', '/rules/suricata'))
|
||||
|
||||
def test_read_watermark_empty_file_returns_none(self):
|
||||
os.makedirs(self.state, exist_ok=True)
|
||||
with open(local_files_beacon._watermark_file('suricata', '/rules/suricata'), 'w') as f:
|
||||
f.write('')
|
||||
self.assertIsNone(local_files_beacon._read_watermark('suricata', '/rules/suricata'))
|
||||
|
||||
def test_write_watermark_swallows_oserror(self):
|
||||
with patch.object(local_files_beacon.os, 'makedirs', side_effect=OSError):
|
||||
local_files_beacon._write_watermark('suricata', '/rules/suricata', 'deadbeef')
|
||||
self.assertIsNone(local_files_beacon._read_watermark('suricata', '/rules/suricata'))
|
||||
|
||||
def test_watermark_file_differs_per_directory_within_one_tag(self):
|
||||
# zeek/policy and zeek/zkg share the tag 'zeek'.
|
||||
self.assertNotEqual(
|
||||
local_files_beacon._watermark_file('zeek', '/local/zeek/policy'),
|
||||
local_files_beacon._watermark_file('zeek', '/local/zeek/zkg'),
|
||||
)
|
||||
|
||||
def test_watermarks_are_independent_within_one_tag(self):
|
||||
local_files_beacon._write_watermark('zeek', '/local/zeek/policy', 'policyhash')
|
||||
local_files_beacon._write_watermark('zeek', '/local/zeek/zkg', 'zkghash')
|
||||
self.assertEqual(
|
||||
local_files_beacon._read_watermark('zeek', '/local/zeek/policy'), 'policyhash')
|
||||
self.assertEqual(
|
||||
local_files_beacon._read_watermark('zeek', '/local/zeek/zkg'), 'zkghash')
|
||||
|
||||
# -- beacon -----------------------------------------------------------
|
||||
|
||||
def _config(self, mapping):
|
||||
return [{'paths': mapping}]
|
||||
|
||||
def test_beacon_seeds_first_run_and_emits_nothing(self):
|
||||
with patch.object(local_files_beacon, '_fingerprint', return_value='hash1'), \
|
||||
patch.object(local_files_beacon, '_read_watermark', return_value=None), \
|
||||
patch.object(local_files_beacon, '_write_watermark') as mock_write:
|
||||
result = local_files_beacon.beacon(self._config({'/rules/suricata': 'suricata'}))
|
||||
self.assertEqual(result, [])
|
||||
mock_write.assert_called_once_with('suricata', '/rules/suricata', 'hash1')
|
||||
|
||||
def test_beacon_emits_on_change(self):
|
||||
with patch.object(local_files_beacon, '_fingerprint', return_value='newhash'), \
|
||||
patch.object(local_files_beacon, '_read_watermark', return_value='oldhash'), \
|
||||
patch.object(local_files_beacon, '_write_watermark') as mock_write:
|
||||
result = local_files_beacon.beacon(self._config({'/rules/suricata': 'suricata'}))
|
||||
self.assertEqual(result, [{'tag': 'suricata', 'path': '/rules/suricata'}])
|
||||
mock_write.assert_called_once_with('suricata', '/rules/suricata', 'newhash')
|
||||
|
||||
def test_beacon_no_change_emits_nothing(self):
|
||||
with patch.object(local_files_beacon, '_fingerprint', return_value='samehash'), \
|
||||
patch.object(local_files_beacon, '_read_watermark', return_value='samehash'), \
|
||||
patch.object(local_files_beacon, '_write_watermark') as mock_write:
|
||||
result = local_files_beacon.beacon(self._config({'/rules/suricata': 'suricata'}))
|
||||
self.assertEqual(result, [])
|
||||
mock_write.assert_not_called()
|
||||
|
||||
def test_beacon_end_to_end_with_real_files(self):
|
||||
# Exercise the full stack (real fingerprint + real watermark files) across
|
||||
# two poll passes: first seeds silently, second fires after a write.
|
||||
d = self._make_dir('rules', {'a.rules': 'alert'})
|
||||
config = self._config({d: 'suricata'})
|
||||
|
||||
self.assertEqual(local_files_beacon.beacon(config), []) # seed pass
|
||||
self.assertEqual(local_files_beacon.beacon(config), []) # unchanged pass
|
||||
|
||||
with open(os.path.join(d, 'b.rules'), 'w') as f:
|
||||
f.write('alert tcp any any -> any any')
|
||||
self.assertEqual(local_files_beacon.beacon(config), [{'tag': 'suricata', 'path': d}])
|
||||
|
||||
def test_beacon_two_dirs_one_tag_do_not_flap(self):
|
||||
# Tag-keyed watermarks would clobber each other and emit on every pass.
|
||||
policy = self._make_dir('zeek/policy', {'intel.dat': '#fields\tindicator'})
|
||||
zkg = self._make_dir('zeek/zkg', {'README': 'place packages here'})
|
||||
config = self._config({policy: 'zeek', zkg: 'zeek'})
|
||||
|
||||
self.assertEqual(local_files_beacon.beacon(config), []) # seed pass
|
||||
self.assertEqual(local_files_beacon.beacon(config), []) # idle
|
||||
self.assertEqual(local_files_beacon.beacon(config), []) # still idle
|
||||
|
||||
with open(os.path.join(policy, 'intel.dat'), 'a') as f:
|
||||
f.write('\nevil.com\tIntel::DOMAIN\tsource\n')
|
||||
self.assertEqual(local_files_beacon.beacon(config), [{'tag': 'zeek', 'path': policy}])
|
||||
self.assertEqual(local_files_beacon.beacon(config), []) # quiet again
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
unittest.main()
|
||||
@@ -0,0 +1,142 @@
|
||||
# Copyright Security Onion Solutions LLC and/or licensed to Security Onion Solutions LLC under one
|
||||
# or more contributor license agreements. Licensed under the Elastic License 2.0 as shown at
|
||||
# https://securityonion.net/license; you may not use this file except in compliance with the
|
||||
# Elastic License 2.0.
|
||||
|
||||
# Custom salt beacon that watches the SOC audit_settings table in postgres for
|
||||
# new settings changes and emits a beacon event per new row. This replaces the
|
||||
# inotify watch on /opt/so/saltstack/local/pillar -- instead of monitoring pillar
|
||||
# files on disk, we monitor the securityonion.audit_settings table that SOC writes to.
|
||||
#
|
||||
# Detection is poll-based with a monotonic `id` watermark persisted to
|
||||
# WATERMARK_FILE: each pass selects rows with id greater than the last id seen,
|
||||
# which makes it self-healing (a missed poll simply catches up on the next one).
|
||||
#
|
||||
# Each emitted event carries setting_id and node_id; the push_pillar reactor maps
|
||||
# setting_id -> app via pillar_push_map.yaml and writes a push intent, after which
|
||||
# the existing so-push-drainer / orch.push_batch pipeline takes over unchanged.
|
||||
|
||||
import logging
|
||||
import os
|
||||
import subprocess
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
WATERMARK_FILE = '/opt/so/state/postgres_pillar_beacon_watch.id'
|
||||
CONTAINER = 'so-postgres'
|
||||
DATABASE = 'securityonion'
|
||||
|
||||
# Unaligned, tuples-only psql output with a field separator that cannot appear in
|
||||
# an id/setting_id/node_id, so we can split each row reliably.
|
||||
FIELD_SEP = '\x1f'
|
||||
|
||||
|
||||
def __virtual__():
|
||||
return True
|
||||
|
||||
|
||||
def validate(config):
|
||||
return True, 'valid'
|
||||
|
||||
|
||||
def _read_watermark():
|
||||
# Returns the last processed id, or None if the watermark has not been seeded.
|
||||
try:
|
||||
with open(WATERMARK_FILE, 'r') as f:
|
||||
return int((f.read() or '').strip())
|
||||
except (IOError, ValueError):
|
||||
return None
|
||||
|
||||
|
||||
def _write_watermark(value):
|
||||
try:
|
||||
os.makedirs(os.path.dirname(WATERMARK_FILE), exist_ok=True)
|
||||
tmp = WATERMARK_FILE + '.tmp'
|
||||
with open(tmp, 'w') as f:
|
||||
f.write(str(int(value)))
|
||||
os.rename(tmp, WATERMARK_FILE)
|
||||
except OSError:
|
||||
log.exception('postgres_pillar_beacon: failed to persist watermark to %s', WATERMARK_FILE)
|
||||
|
||||
|
||||
def _query(sql):
|
||||
# Run a query against securityonion inside the so-postgres container over the unix
|
||||
# socket (trust auth, no password). Returns stdout on success, or None on any
|
||||
# failure so the caller can no-op and retry on the next interval.
|
||||
cmd = [
|
||||
'docker', 'exec', CONTAINER,
|
||||
'psql', '-U', 'postgres', '-d', DATABASE,
|
||||
'-tA', '-F', FIELD_SEP, '-c', sql,
|
||||
]
|
||||
try:
|
||||
result = subprocess.run(cmd, capture_output=True, text=True, timeout=30)
|
||||
except subprocess.TimeoutExpired:
|
||||
log.warning('postgres_pillar_beacon: psql timed out')
|
||||
return None
|
||||
except Exception:
|
||||
log.exception('postgres_pillar_beacon: failed to exec psql')
|
||||
return None
|
||||
if result.returncode != 0:
|
||||
log.warning('postgres_pillar_beacon: psql failed (rc=%s): %s',
|
||||
result.returncode, (result.stderr or '').strip())
|
||||
return None
|
||||
return result.stdout
|
||||
|
||||
|
||||
def beacon(config): # noqa: C901
|
||||
retval = []
|
||||
|
||||
watermark = _read_watermark()
|
||||
|
||||
# First run / missing watermark: seed to the current MAX(id) and emit nothing
|
||||
# so we never replay the entire settings history into a fleetwide push.
|
||||
if watermark is None:
|
||||
seed = _query('SELECT COALESCE(MAX(id), 0) FROM audit_settings;')
|
||||
if seed is None:
|
||||
return retval # postgres not ready yet; retry next interval
|
||||
try:
|
||||
_write_watermark(int((seed or '0').strip() or 0))
|
||||
except ValueError:
|
||||
log.warning('postgres_pillar_beacon: could not parse MAX(id) seed: %r', seed)
|
||||
return retval
|
||||
|
||||
rows = _query(
|
||||
"SELECT id, setting_id, COALESCE(node_id, '') FROM audit_settings "
|
||||
"WHERE id > %d ORDER BY id;" % watermark
|
||||
)
|
||||
if rows is None:
|
||||
return retval
|
||||
|
||||
max_id = watermark
|
||||
for line in rows.splitlines():
|
||||
# Do NOT str.strip() the whole line: Python treats the \x1f field
|
||||
# separator (and \x1c-\x1e) as whitespace, so stripping would eat an
|
||||
# empty trailing node_id field and make the row look malformed.
|
||||
if not line.strip():
|
||||
continue
|
||||
parts = line.split(FIELD_SEP)
|
||||
if len(parts) < 3:
|
||||
log.warning('postgres_pillar_beacon: skipping malformed row: %r', line)
|
||||
continue
|
||||
try:
|
||||
row_id = int(parts[0])
|
||||
except ValueError:
|
||||
log.warning('postgres_pillar_beacon: skipping row with non-int id: %r', line)
|
||||
continue
|
||||
setting_id = parts[1]
|
||||
node_id = parts[2]
|
||||
retval.append({
|
||||
'tag': 'audit_settings',
|
||||
'id': row_id,
|
||||
'setting_id': setting_id,
|
||||
'node_id': node_id,
|
||||
})
|
||||
if row_id > max_id:
|
||||
max_id = row_id
|
||||
|
||||
if max_id > watermark:
|
||||
_write_watermark(max_id)
|
||||
log.info('postgres_pillar_beacon: emitted %d change(s), watermark %d -> %d',
|
||||
len(retval), watermark, max_id)
|
||||
|
||||
return retval
|
||||
@@ -0,0 +1,165 @@
|
||||
# Copyright Security Onion Solutions LLC and/or licensed to Security Onion Solutions LLC under one
|
||||
# or more contributor license agreements. Licensed under the Elastic License 2.0 as shown at
|
||||
# https://securityonion.net/license; you may not use this file except in compliance with the
|
||||
# Elastic License 2.0.
|
||||
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import tempfile
|
||||
import unittest
|
||||
from unittest.mock import patch
|
||||
|
||||
import postgres_pillar_beacon
|
||||
|
||||
|
||||
class TestPostgresPillarBeacon(unittest.TestCase):
|
||||
|
||||
def setUp(self):
|
||||
# Point WATERMARK_FILE at a throwaway dir so the real read/write helpers
|
||||
# (and their os.makedirs/os.rename) run against actual files, then clean
|
||||
# it all up in tearDown.
|
||||
self.tmpdir = tempfile.mkdtemp()
|
||||
self.watermark = os.path.join(self.tmpdir, 'state', 'watch.id')
|
||||
patcher = patch.object(postgres_pillar_beacon, 'WATERMARK_FILE', self.watermark)
|
||||
patcher.start()
|
||||
self.addCleanup(patcher.stop)
|
||||
|
||||
def tearDown(self):
|
||||
shutil.rmtree(self.tmpdir, ignore_errors=True)
|
||||
|
||||
# -- trivial contract -------------------------------------------------
|
||||
|
||||
def test_virtual_returns_true(self):
|
||||
self.assertTrue(postgres_pillar_beacon.__virtual__())
|
||||
|
||||
def test_validate_returns_valid(self):
|
||||
self.assertEqual(postgres_pillar_beacon.validate({}), (True, 'valid'))
|
||||
|
||||
# -- _read_watermark --------------------------------------------------
|
||||
|
||||
def test_read_watermark_valid(self):
|
||||
postgres_pillar_beacon._write_watermark(42)
|
||||
self.assertEqual(postgres_pillar_beacon._read_watermark(), 42)
|
||||
|
||||
def test_read_watermark_missing_file_returns_none(self):
|
||||
# tmp watermark file was never created
|
||||
self.assertIsNone(postgres_pillar_beacon._read_watermark())
|
||||
|
||||
def test_read_watermark_garbage_returns_none(self):
|
||||
os.makedirs(os.path.dirname(self.watermark), exist_ok=True)
|
||||
with open(self.watermark, 'w') as f:
|
||||
f.write('nope')
|
||||
self.assertIsNone(postgres_pillar_beacon._read_watermark())
|
||||
|
||||
# -- _write_watermark -------------------------------------------------
|
||||
|
||||
def test_write_watermark_round_trip(self):
|
||||
postgres_pillar_beacon._write_watermark(7)
|
||||
with open(self.watermark) as f:
|
||||
self.assertEqual(f.read(), '7')
|
||||
|
||||
def test_write_watermark_swallows_oserror(self):
|
||||
with patch.object(postgres_pillar_beacon.os, 'makedirs', side_effect=OSError):
|
||||
# Must not raise; failure is logged and the beacon retries next pass.
|
||||
postgres_pillar_beacon._write_watermark(5)
|
||||
self.assertFalse(os.path.exists(self.watermark))
|
||||
|
||||
# -- _query -----------------------------------------------------------
|
||||
|
||||
def test_query_success_returns_stdout_and_builds_argv(self):
|
||||
completed = subprocess.CompletedProcess(args=[], returncode=0, stdout='rows', stderr='')
|
||||
with patch.object(postgres_pillar_beacon.subprocess, 'run', return_value=completed) as mock_run:
|
||||
result = postgres_pillar_beacon._query('SELECT 1;')
|
||||
self.assertEqual(result, 'rows')
|
||||
argv = mock_run.call_args[0][0]
|
||||
self.assertEqual(argv[:5], ['docker', 'exec', 'so-postgres', 'psql', '-U'])
|
||||
self.assertIn('SELECT 1;', argv)
|
||||
self.assertFalse(mock_run.call_args[1].get('shell', False))
|
||||
|
||||
def test_query_timeout_returns_none(self):
|
||||
with patch.object(postgres_pillar_beacon.subprocess, 'run',
|
||||
side_effect=subprocess.TimeoutExpired(cmd='psql', timeout=30)):
|
||||
self.assertIsNone(postgres_pillar_beacon._query('SELECT 1;'))
|
||||
|
||||
def test_query_generic_exception_returns_none(self):
|
||||
with patch.object(postgres_pillar_beacon.subprocess, 'run', side_effect=Exception('boom')):
|
||||
self.assertIsNone(postgres_pillar_beacon._query('SELECT 1;'))
|
||||
|
||||
def test_query_nonzero_returncode_returns_none(self):
|
||||
completed = subprocess.CompletedProcess(args=[], returncode=1, stdout='', stderr='bad')
|
||||
with patch.object(postgres_pillar_beacon.subprocess, 'run', return_value=completed):
|
||||
self.assertIsNone(postgres_pillar_beacon._query('SELECT 1;'))
|
||||
|
||||
# -- beacon: first run / seeding --------------------------------------
|
||||
|
||||
def test_beacon_seeds_when_postgres_not_ready(self):
|
||||
with patch.object(postgres_pillar_beacon, '_read_watermark', return_value=None), \
|
||||
patch.object(postgres_pillar_beacon, '_query', return_value=None), \
|
||||
patch.object(postgres_pillar_beacon, '_write_watermark') as mock_write:
|
||||
self.assertEqual(postgres_pillar_beacon.beacon({}), [])
|
||||
mock_write.assert_not_called()
|
||||
|
||||
def test_beacon_seeds_to_max_id_and_emits_nothing(self):
|
||||
with patch.object(postgres_pillar_beacon, '_read_watermark', return_value=None), \
|
||||
patch.object(postgres_pillar_beacon, '_query', return_value='7\n'), \
|
||||
patch.object(postgres_pillar_beacon, '_write_watermark') as mock_write:
|
||||
self.assertEqual(postgres_pillar_beacon.beacon({}), [])
|
||||
mock_write.assert_called_once_with(7)
|
||||
|
||||
def test_beacon_seed_unparseable_is_swallowed(self):
|
||||
with patch.object(postgres_pillar_beacon, '_read_watermark', return_value=None), \
|
||||
patch.object(postgres_pillar_beacon, '_query', return_value='abc'), \
|
||||
patch.object(postgres_pillar_beacon, '_write_watermark') as mock_write:
|
||||
self.assertEqual(postgres_pillar_beacon.beacon({}), [])
|
||||
mock_write.assert_not_called()
|
||||
|
||||
# -- beacon: steady state ---------------------------------------------
|
||||
|
||||
def test_beacon_query_failure_returns_empty(self):
|
||||
with patch.object(postgres_pillar_beacon, '_read_watermark', return_value=10), \
|
||||
patch.object(postgres_pillar_beacon, '_query', return_value=None), \
|
||||
patch.object(postgres_pillar_beacon, '_write_watermark') as mock_write:
|
||||
self.assertEqual(postgres_pillar_beacon.beacon({}), [])
|
||||
mock_write.assert_not_called()
|
||||
|
||||
def test_beacon_emits_events_and_advances_watermark(self):
|
||||
sep = postgres_pillar_beacon.FIELD_SEP
|
||||
rows = '11%s5%snode1\n12%s6%s\n' % (sep, sep, sep, sep)
|
||||
with patch.object(postgres_pillar_beacon, '_read_watermark', return_value=10), \
|
||||
patch.object(postgres_pillar_beacon, '_query', return_value=rows), \
|
||||
patch.object(postgres_pillar_beacon, '_write_watermark') as mock_write:
|
||||
result = postgres_pillar_beacon.beacon({})
|
||||
self.assertEqual(result, [
|
||||
{'tag': 'audit_settings', 'id': 11, 'setting_id': '5', 'node_id': 'node1'},
|
||||
{'tag': 'audit_settings', 'id': 12, 'setting_id': '6', 'node_id': ''},
|
||||
])
|
||||
mock_write.assert_called_once_with(12)
|
||||
|
||||
def test_beacon_skips_malformed_blank_and_noninteger_rows(self):
|
||||
sep = postgres_pillar_beacon.FIELD_SEP
|
||||
rows = (
|
||||
'\n' # blank line -> skipped
|
||||
'13%s7\n' # too few fields -> skipped
|
||||
'abc%s8%snodeX\n' # non-integer id -> skipped
|
||||
'14%s9%snodeY\n' # the one good row
|
||||
) % (sep, sep, sep, sep, sep)
|
||||
with patch.object(postgres_pillar_beacon, '_read_watermark', return_value=10), \
|
||||
patch.object(postgres_pillar_beacon, '_query', return_value=rows), \
|
||||
patch.object(postgres_pillar_beacon, '_write_watermark') as mock_write:
|
||||
result = postgres_pillar_beacon.beacon({})
|
||||
self.assertEqual(result, [
|
||||
{'tag': 'audit_settings', 'id': 14, 'setting_id': '9', 'node_id': 'nodeY'},
|
||||
])
|
||||
mock_write.assert_called_once_with(14)
|
||||
|
||||
def test_beacon_no_new_rows_does_not_advance_watermark(self):
|
||||
with patch.object(postgres_pillar_beacon, '_read_watermark', return_value=10), \
|
||||
patch.object(postgres_pillar_beacon, '_query', return_value=''), \
|
||||
patch.object(postgres_pillar_beacon, '_write_watermark') as mock_write:
|
||||
self.assertEqual(postgres_pillar_beacon.beacon({}), [])
|
||||
mock_write.assert_not_called()
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
unittest.main()
|
||||
+18
-19
@@ -3,31 +3,30 @@ import logging
|
||||
|
||||
def status():
|
||||
|
||||
cmd = "runuser -l zeek -c '/opt/zeek/bin/zeekctl status'"
|
||||
retval = __salt__['docker.run']('so-zeek', cmd)
|
||||
logging.info('zeekctl_module: zeekctl.status retval: %s' % retval)
|
||||
cmd = "runuser -l zeek -c '/opt/zeek/bin/zeekctl status'"
|
||||
retval = __salt__['docker.run']('so-zeek', cmd) # noqa: F821
|
||||
logging.info('zeekctl_module: zeekctl.status retval: %s' % retval)
|
||||
|
||||
return retval
|
||||
return retval
|
||||
|
||||
|
||||
def beacon(config):
|
||||
|
||||
retval = []
|
||||
retval = []
|
||||
|
||||
is_enabled = __salt__['healthcheck.is_enabled']()
|
||||
logging.info('zeek_beacon: healthcheck_is_enabled: %s' % is_enabled)
|
||||
is_enabled = __salt__['healthcheck.is_enabled']() # noqa: F821
|
||||
logging.info('zeek_beacon: healthcheck_is_enabled: %s' % is_enabled)
|
||||
|
||||
if is_enabled:
|
||||
zeekstatus = status().lower().split(' ')
|
||||
logging.info('zeek_beacon: zeekctl.status: %s' % str(zeekstatus))
|
||||
if 'stopped' in zeekstatus or 'crashed' in zeekstatus or 'error' in zeekstatus or 'error:' in zeekstatus:
|
||||
zeek_restart = True
|
||||
else:
|
||||
zeek_restart = False
|
||||
if is_enabled:
|
||||
zeekstatus = status().lower().split(' ')
|
||||
logging.info('zeek_beacon: zeekctl.status: %s' % str(zeekstatus))
|
||||
if 'stopped' in zeekstatus or 'crashed' in zeekstatus or 'error' in zeekstatus or 'error:' in zeekstatus:
|
||||
zeek_restart = True
|
||||
else:
|
||||
zeek_restart = False
|
||||
|
||||
__salt__['telegraf.send']('healthcheck zeek_restart=%s' % str(zeek_restart))
|
||||
retval.append({'zeek_restart': zeek_restart})
|
||||
logging.info('zeek_beacon: retval: %s' % str(retval))
|
||||
|
||||
return retval
|
||||
__salt__['telegraf.send']('healthcheck zeek_restart=%s' % str(zeek_restart)) # noqa: F821
|
||||
retval.append({'zeek_restart': zeek_restart})
|
||||
logging.info('zeek_beacon: retval: %s' % str(retval))
|
||||
|
||||
return retval
|
||||
|
||||
@@ -0,0 +1,59 @@
|
||||
# Copyright Security Onion Solutions LLC and/or licensed to Security Onion Solutions LLC under one
|
||||
# or more contributor license agreements. Licensed under the Elastic License 2.0 as shown at
|
||||
# https://securityonion.net/license; you may not use this file except in compliance with the
|
||||
# Elastic License 2.0.
|
||||
|
||||
import unittest
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
import zeek
|
||||
|
||||
ZEEKCTL_CMD = "runuser -l zeek -c '/opt/zeek/bin/zeekctl status'"
|
||||
|
||||
|
||||
class TestZeekBeacon(unittest.TestCase):
|
||||
|
||||
def setUp(self):
|
||||
# zeek.py relies on the __salt__ dunder that Salt injects at load time.
|
||||
# Nothing defines it under test, so we attach a dict of mock loader
|
||||
# functions to the module and remove it again afterwards.
|
||||
self.salt = {
|
||||
'docker.run': MagicMock(return_value='Zeek is running'),
|
||||
'healthcheck.is_enabled': MagicMock(return_value=True),
|
||||
'telegraf.send': MagicMock(),
|
||||
}
|
||||
zeek.__salt__ = self.salt
|
||||
self.addCleanup(lambda: delattr(zeek, '__salt__'))
|
||||
|
||||
# -- status -----------------------------------------------------------
|
||||
|
||||
def test_status_runs_zeekctl_and_returns_output(self):
|
||||
self.salt['docker.run'].return_value = 'Zeek is running'
|
||||
result = zeek.status()
|
||||
self.assertEqual(result, 'Zeek is running')
|
||||
self.salt['docker.run'].assert_called_once_with('so-zeek', ZEEKCTL_CMD)
|
||||
|
||||
# -- beacon -----------------------------------------------------------
|
||||
|
||||
def test_beacon_disabled_returns_empty_and_skips_telegraf(self):
|
||||
self.salt['healthcheck.is_enabled'].return_value = False
|
||||
self.assertEqual(zeek.beacon({}), [])
|
||||
self.salt['telegraf.send'].assert_not_called()
|
||||
|
||||
def test_beacon_running_reports_no_restart(self):
|
||||
self.salt['docker.run'].return_value = 'Zeek is running'
|
||||
self.assertEqual(zeek.beacon({}), [{'zeek_restart': False}])
|
||||
self.salt['telegraf.send'].assert_called_once_with('healthcheck zeek_restart=False')
|
||||
|
||||
def test_beacon_unhealthy_status_triggers_restart(self):
|
||||
# Each of these status tokens should flag a restart (the or-chain in beacon).
|
||||
for status_text in ('Zeek is stopped', 'Zeek crashed', 'Zeek error state', 'Zeek error:'):
|
||||
with self.subTest(status=status_text):
|
||||
self.salt['docker.run'].return_value = status_text
|
||||
self.salt['telegraf.send'].reset_mock()
|
||||
self.assertEqual(zeek.beacon({}), [{'zeek_restart': True}])
|
||||
self.salt['telegraf.send'].assert_called_once_with('healthcheck zeek_restart=True')
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
unittest.main()
|
||||
+34
-5
@@ -117,14 +117,25 @@ elastic_curl_config:
|
||||
{% endif %}
|
||||
|
||||
|
||||
# A non-root owner here can chmod the directory and replace any script in it, including
|
||||
# the root-owned ones. 555 is the mode the filesystem RPM ships; root ignores it anyway.
|
||||
usr_sbin_perms:
|
||||
file.directory:
|
||||
- name: /usr/sbin
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 555
|
||||
|
||||
common_sbin:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://common/tools/sbin
|
||||
- user: 939
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
- show_changes: False
|
||||
- require:
|
||||
- file: usr_sbin_perms
|
||||
{% if GLOBALS.role == 'so-heavynode' %}
|
||||
- exclude_pat:
|
||||
- so-pcap-import
|
||||
@@ -141,12 +152,26 @@ pin_nic_names:
|
||||
- file: common_sbin
|
||||
- file: statedir
|
||||
|
||||
# Once a node is actually running UEK8, the stock EL9 (RHCK) kernel packages are dead weight.
|
||||
# They can't be removed any earlier -- dnf protects the running kernel -- so the cleanup waits
|
||||
# for the reboot, which makes the highstate the natural place to catch it: fresh installs
|
||||
# reboot at the end of setup, and upgraded nodes reboot whenever the admin schedules it.
|
||||
# so-kernel-upgrade --cleanup checks rpm before touching dnf, so this costs an rpm query on
|
||||
# every highstate after the first pass. The package list lives in the script only, so there
|
||||
# is nothing here to drift out of sync with it.
|
||||
remove_stock_kernel:
|
||||
cmd.run:
|
||||
- name: /usr/sbin/so-kernel-upgrade --cleanup
|
||||
- onlyif: 'uname -r | grep -qE "^6\.[0-9]+.*uek"'
|
||||
- require:
|
||||
- file: common_sbin
|
||||
|
||||
common_sbin_jinja:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://common/tools/sbin_jinja
|
||||
- user: 939
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
- template: jinja
|
||||
- show_changes: False
|
||||
@@ -159,6 +184,8 @@ so-status_script:
|
||||
file.managed:
|
||||
- name: /usr/sbin/so-status
|
||||
- source: salt://common/tools/sbin/so-status
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
|
||||
{% if GLOBALS.is_sensor %}
|
||||
@@ -190,9 +217,11 @@ sostatus_log:
|
||||
- replace: False
|
||||
|
||||
# Install sostatus check cron. This is used to populate Grid.
|
||||
# telegraf reads status.log on the same minute boundary this runs, so write aside and rename
|
||||
# rather than truncating the file it is reading
|
||||
so-status_check_cron:
|
||||
cron.present:
|
||||
- name: '/usr/sbin/so-status -j > /opt/so/log/sostatus/status.log 2>&1'
|
||||
- name: '/usr/sbin/so-status -j > /opt/so/log/sostatus/status.log.tmp 2>&1; mv -f /opt/so/log/sostatus/status.log.tmp /opt/so/log/sostatus/status.log'
|
||||
- identifier: so-status_check_cron
|
||||
- user: root
|
||||
- minute: '*/1'
|
||||
|
||||
@@ -18,47 +18,61 @@ copy_so-common_common_tools_sbin:
|
||||
- name: /opt/so/saltstack/default/salt/common/tools/sbin/so-common
|
||||
- source: {{UPDATE_DIR}}/salt/common/tools/sbin/so-common
|
||||
- force: True
|
||||
- preserve: True
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
|
||||
copy_so-image-common_common_tools_sbin:
|
||||
file.copy:
|
||||
- name: /opt/so/saltstack/default/salt/common/tools/sbin/so-image-common
|
||||
- source: {{UPDATE_DIR}}/salt/common/tools/sbin/so-image-common
|
||||
- force: True
|
||||
- preserve: True
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
|
||||
copy_soup_manager_tools_sbin:
|
||||
file.copy:
|
||||
- name: /opt/so/saltstack/default/salt/manager/tools/sbin/soup
|
||||
- source: {{UPDATE_DIR}}/salt/manager/tools/sbin/soup
|
||||
- force: True
|
||||
- preserve: True
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
|
||||
copy_so-firewall_manager_tools_sbin:
|
||||
file.copy:
|
||||
- name: /opt/so/saltstack/default/salt/manager/tools/sbin/so-firewall
|
||||
- source: {{UPDATE_DIR}}/salt/manager/tools/sbin/so-firewall
|
||||
- force: True
|
||||
- preserve: True
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
|
||||
copy_so-yaml_manager_tools_sbin:
|
||||
file.copy:
|
||||
- name: /opt/so/saltstack/default/salt/manager/tools/sbin/so-yaml.py
|
||||
- source: {{UPDATE_DIR}}/salt/manager/tools/sbin/so-yaml.py
|
||||
- force: True
|
||||
- preserve: True
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
|
||||
copy_so-repo-sync_manager_tools_sbin:
|
||||
file.copy:
|
||||
- name: /opt/so/saltstack/default/salt/manager/tools/sbin/so-repo-sync
|
||||
- source: {{UPDATE_DIR}}/salt/manager/tools/sbin/so-repo-sync
|
||||
- preserve: True
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
|
||||
copy_bootstrap-salt_manager_tools_sbin:
|
||||
file.copy:
|
||||
- name: /opt/so/saltstack/default/salt/salt/scripts/bootstrap-salt.sh
|
||||
- source: {{UPDATE_DIR}}/salt/salt/scripts/bootstrap-salt.sh
|
||||
- preserve: True
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 644
|
||||
|
||||
# This section is used to put the new script in place so that it can be called during soup.
|
||||
# It is faster than calling the states that normally manage them to put them in place.
|
||||
@@ -67,46 +81,60 @@ copy_so-common_sbin:
|
||||
- name: /usr/sbin/so-common
|
||||
- source: {{UPDATE_DIR}}/salt/common/tools/sbin/so-common
|
||||
- force: True
|
||||
- preserve: True
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
|
||||
copy_so-image-common_sbin:
|
||||
file.copy:
|
||||
- name: /usr/sbin/so-image-common
|
||||
- source: {{UPDATE_DIR}}/salt/common/tools/sbin/so-image-common
|
||||
- force: True
|
||||
- preserve: True
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
|
||||
copy_soup_sbin:
|
||||
file.copy:
|
||||
- name: /usr/sbin/soup
|
||||
- source: {{UPDATE_DIR}}/salt/manager/tools/sbin/soup
|
||||
- force: True
|
||||
- preserve: True
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
|
||||
copy_so-firewall_sbin:
|
||||
file.copy:
|
||||
- name: /usr/sbin/so-firewall
|
||||
- source: {{UPDATE_DIR}}/salt/manager/tools/sbin/so-firewall
|
||||
- force: True
|
||||
- preserve: True
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
|
||||
copy_so-yaml_sbin:
|
||||
file.copy:
|
||||
- name: /usr/sbin/so-yaml.py
|
||||
- source: {{UPDATE_DIR}}/salt/manager/tools/sbin/so-yaml.py
|
||||
- force: True
|
||||
- preserve: True
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
|
||||
copy_so-repo-sync_sbin:
|
||||
file.copy:
|
||||
- name: /usr/sbin/so-repo-sync
|
||||
- source: {{UPDATE_DIR}}/salt/manager/tools/sbin/so-repo-sync
|
||||
- force: True
|
||||
- preserve: True
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
|
||||
copy_bootstrap-salt_sbin:
|
||||
file.copy:
|
||||
- name: /usr/sbin/bootstrap-salt.sh
|
||||
- source: {{UPDATE_DIR}}/salt/salt/scripts/bootstrap-salt.sh
|
||||
- force: True
|
||||
- preserve: True
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
|
||||
@@ -240,7 +240,8 @@ copy_new_files() {
|
||||
cd $UPDATE_DIR
|
||||
rsync -a salt $DEFAULT_SALT_DIR/ --delete "${EXCLUDE_ARGS[@]}"
|
||||
rsync -a pillar $DEFAULT_SALT_DIR/ --delete "${EXCLUDE_ARGS[@]}"
|
||||
chown -R socore:socore $DEFAULT_SALT_DIR/
|
||||
# Root-executed code; SOC only needs to read it. Local dirs stay socore-owned.
|
||||
chown -R root:root $DEFAULT_SALT_DIR/
|
||||
cd /tmp
|
||||
}
|
||||
|
||||
@@ -418,6 +419,19 @@ is_single_node_grid() {
|
||||
grep "role: so-" /etc/salt/grains | grep -E "eval|standalone|import" &> /dev/null
|
||||
}
|
||||
|
||||
remove_elasticsearch_index_template() {
|
||||
local template_name=$1
|
||||
local reason=${2:-"Removing index template for upgrade"}
|
||||
|
||||
# check if template exists
|
||||
if so-elasticsearch-query "_index_template/$template_name" --fail --retry 3 --retry-delay 5 >/dev/null 2>&1; then
|
||||
echo "Removing Elasticsearch index template: $template_name ($reason)"
|
||||
if ! so-elasticsearch-query "_index_template/$template_name" -XDELETE --fail --retry 3 --retry-delay 5; then
|
||||
return 1
|
||||
fi
|
||||
fi
|
||||
}
|
||||
|
||||
initialize_elasticsearch_indices() {
|
||||
local index_names=$1
|
||||
local default_entry=${2:-'{"@timestamp":"0"}'}
|
||||
@@ -602,42 +616,6 @@ run_check_net_err() {
|
||||
fi
|
||||
}
|
||||
|
||||
wait_for_salt_minion() {
|
||||
local minion="$1"
|
||||
local max_wait="${2:-30}"
|
||||
local interval="${3:-2}"
|
||||
local logfile="${4:-'/dev/stdout'}"
|
||||
local elapsed=0
|
||||
|
||||
echo "$(date '+%a %d %b %Y %H:%M:%S.%6N') - Waiting for salt-minion '$minion' to be ready..."
|
||||
|
||||
while [ $elapsed -lt $max_wait ]; do
|
||||
# Check if service is running
|
||||
echo "$(date '+%a %d %b %Y %H:%M:%S.%6N') - Check if salt-minion service is running"
|
||||
if ! systemctl is-active --quiet salt-minion; then
|
||||
echo "$(date '+%a %d %b %Y %H:%M:%S.%6N') - salt-minion service not running (elapsed: ${elapsed}s)"
|
||||
sleep $interval
|
||||
elapsed=$((elapsed + interval))
|
||||
continue
|
||||
fi
|
||||
echo "$(date '+%a %d %b %Y %H:%M:%S.%6N') - salt-minion service is running"
|
||||
|
||||
# Check if minion responds to ping
|
||||
echo "$(date '+%a %d %b %Y %H:%M:%S.%6N') - Check if $minion responds to ping"
|
||||
if salt "$minion" test.ping --timeout=3 --out=json 2>> "$logfile" | grep -q "true"; then
|
||||
echo "$(date '+%a %d %b %Y %H:%M:%S.%6N') - salt-minion '$minion' is connected and ready!"
|
||||
return 0
|
||||
fi
|
||||
|
||||
echo "$(date '+%a %d %b %Y %H:%M:%S.%6N') - Waiting... (${elapsed}s / ${max_wait}s)"
|
||||
sleep $interval
|
||||
elapsed=$((elapsed + interval))
|
||||
done
|
||||
|
||||
echo "$(date '+%a %d %b %Y %H:%M:%S.%6N') - ERROR: salt-minion '$minion' not ready after $max_wait seconds"
|
||||
return 1
|
||||
}
|
||||
|
||||
salt_minion_count() {
|
||||
local MINIONDIR="/opt/so/saltstack/local/pillar/minions"
|
||||
MINIONCOUNT=$(ls -la $MINIONDIR/*.sls | grep -v adv_ | wc -l)
|
||||
@@ -702,7 +680,7 @@ systemctl_func() {
|
||||
|
||||
echo ""
|
||||
echo "${echo_action^}ing $service_name service at $(date +"%T.%6N")"
|
||||
systemctl $action $service_name && echo "Successfully ${echo_action}ed $service_name." || echo "Failed to $action $service_name."
|
||||
systemctl $action $service_name && echo "Successfully ${echo_action}ed $service_name at $(date +"%T.%6N")." || echo "Failed to $action $service_name at $(date +"%T.%6N")."
|
||||
echo ""
|
||||
}
|
||||
|
||||
|
||||
@@ -9,6 +9,7 @@ import sys
|
||||
import subprocess
|
||||
import os
|
||||
import json
|
||||
import tempfile
|
||||
|
||||
sys.path.append('/opt/saltstack/salt/lib/python3.10/site-packages/')
|
||||
import salt.config
|
||||
@@ -17,6 +18,21 @@ import salt.loader
|
||||
__opts__ = salt.config.minion_config('/etc/salt/minion')
|
||||
__grains__ = salt.loader.grains(__opts__)
|
||||
|
||||
def write_atomic(path, value):
|
||||
# telegraf reads these files on its own schedule; replacing them by rename means it never
|
||||
# reads a truncated file and reports an empty value as if it were real
|
||||
directory = os.path.dirname(path)
|
||||
handle, temp = tempfile.mkstemp(dir=directory)
|
||||
try:
|
||||
with os.fdopen(handle, 'w') as f:
|
||||
f.write(str(value))
|
||||
os.chmod(temp, 0o644)
|
||||
os.replace(temp, path)
|
||||
except Exception:
|
||||
os.path.exists(temp) and os.unlink(temp)
|
||||
raise
|
||||
|
||||
|
||||
def check_needs_restarted():
|
||||
osfam = __grains__['os_family']
|
||||
val = '0'
|
||||
@@ -34,8 +50,7 @@ def check_needs_restarted():
|
||||
else:
|
||||
fail("Unsupported OS")
|
||||
|
||||
with open(outfile, 'w') as f:
|
||||
f.write(val)
|
||||
write_atomic(outfile, val)
|
||||
|
||||
def check_for_fps():
|
||||
feat = 'fps'
|
||||
@@ -56,8 +71,7 @@ def check_for_fps():
|
||||
# Unknown, so assume 0
|
||||
fps = 0
|
||||
|
||||
with open('/opt/so/log/sostatus/fps_enabled', 'w') as f:
|
||||
f.write(str(fps))
|
||||
write_atomic('/opt/so/log/sostatus/fps_enabled', fps)
|
||||
|
||||
def check_for_lks():
|
||||
feat = 'Lks'
|
||||
@@ -80,8 +94,7 @@ def check_for_lks():
|
||||
lks = 1
|
||||
if lks:
|
||||
break
|
||||
with open('/opt/so/log/sostatus/lks_enabled', 'w') as f:
|
||||
f.write(str(lks))
|
||||
write_atomic('/opt/so/log/sostatus/lks_enabled', lks)
|
||||
|
||||
def fail(msg):
|
||||
print(msg, file=sys.stderr)
|
||||
|
||||
@@ -5,53 +5,313 @@
|
||||
# https://securityonion.net/license; you may not use this file except in compliance with the
|
||||
# Elastic License 2.0.
|
||||
#
|
||||
# so-kernel-upgrade — switch the boot default to the installed UEK8 (6.x) kernel.
|
||||
# so-kernel-upgrade — install the UEK8 (6.x) kernel, make it the boot default, and once the
|
||||
# node is running it, remove the stock EL9 kernel.
|
||||
#
|
||||
# Security Onion is moving off the EL9 stock kernel / UEK7 (5.x) onto UEK8 (6.x).
|
||||
# Installing the kernel-uek-core package adds a UEK8 boot entry but does NOT make it the
|
||||
# default: kernel-install/grubby only auto-promote a new kernel within the running
|
||||
# kernel's flavor lineage, and we're crossing from a 5.x kernel to the new 6.x UEK flavor.
|
||||
# So even with UPDATEDEFAULT=yes and DEFAULTKERNEL=kernel-uek-core the box keeps booting
|
||||
# the old kernel. This tool finds the newest installed 6.x UEK kernel and makes it the
|
||||
# GRUB default via grubby so the next boot comes up on UEK8.
|
||||
# Security Onion is moving off the EL9 stock kernel (RHCK, 5.14) and UEK7 (5.15) onto UEK8
|
||||
# (6.x). Four things have to happen, and the tool has to drive each one:
|
||||
#
|
||||
# Idempotent: if the UEK8 kernel is already the default it does nothing. It only sets the
|
||||
# boot default; it does NOT reboot — the admin reboots the node on their own schedule.
|
||||
# 1. Populate. The manager mirrors the UEK8 packages into /nsm/kernelrepo via so-repo-sync,
|
||||
# and serves them to the grid over https://<manager>/kernelrepo. Until that sync runs the
|
||||
# repo is valid but EMPTY -- dnf resolves it happily and installs nothing, with no error.
|
||||
# 2. Install. A node on RHCK has no kernel-uek* package at all, so there is nothing for
|
||||
# 'dnf update' to upgrade. A node on UEK7 does have kernel-uek installed, so
|
||||
# 'dnf install kernel-uek' reports "Nothing to do" and exits 0 without installing 6.x.
|
||||
# Both cases need an explicit install of the UEK8 NEVRA.
|
||||
# 3. Boot it. Whether a newly installed UEK8 kernel becomes the boot default depends on the
|
||||
# RUNNING kernel's flavor. kernel-install/grubby (with UPDATEDEFAULT=yes) only auto-promote
|
||||
# within the running kernel's flavor lineage:
|
||||
# - From UEK7 (5.x, kernel-uek) the install stays in the kernel-uek lineage and IS
|
||||
# auto-promoted, so no grubby change is needed -- just make sure the repo is populated
|
||||
# and install UEK8.
|
||||
# - From the stock EL9 kernel (RHCK, 5.14, no UEK) it is a flavor CROSS that is NOT
|
||||
# auto-promoted, so the box keeps booting RHCK until grubby is told otherwise.
|
||||
# This tool inspects the running kernel and only runs 'grubby --set-default' for RHCK.
|
||||
# 4. Clean up. Once the node is actually RUNNING UEK8 the stock kernel packages are dead
|
||||
# weight -- disk in /boot and a stale GRUB entry. They cannot come off any earlier:
|
||||
# dnf's protect_running_kernel refuses to erase the booted kernel-core, so the removal
|
||||
# has to wait for the reboot. Waiting is also the safer sequencing on its own terms --
|
||||
# the node has proven it comes up on UEK8 before its fallback is deleted. That is why
|
||||
# the removal does not happen in the uek7 branch either, where dnf would allow it.
|
||||
#
|
||||
# Every one of those failure modes is silent by default. This tool handles each case and fails
|
||||
# loudly when it cannot, rather than reporting success while changing nothing.
|
||||
#
|
||||
# Invocation: with no arguments it drives the whole sequence for whatever kernel the node is
|
||||
# on. With --cleanup it does the step 4 removal ONLY, and no-ops on a node that isn't running
|
||||
# UEK8 yet -- that is the form the common highstate calls (remove_stock_kernel in
|
||||
# salt/common/init.sls) so the cleanup lands grid-wide after each node reboots.
|
||||
#
|
||||
# Manager vs minion: only the manager owns /nsm/kernelrepo, so only the manager can populate
|
||||
# it. If the repo is empty here, a manager runs so-repo-sync itself; a minion has no way to
|
||||
# fix it and exits non-zero telling the admin to sync the manager first.
|
||||
#
|
||||
# Idempotent: an already-installed, already-default UEK8 kernel is left alone. It only sets
|
||||
# the boot default; it does NOT reboot -- the admin reboots the node on their own schedule.
|
||||
|
||||
. /usr/sbin/so-common
|
||||
|
||||
# Client-side repo id (what dnf enables on this node, from repo/client/oracle.sls) vs the
|
||||
# reposync-side section in repodownload.conf that the manager mirrors from (mirrors the
|
||||
# securityonion/securityonionsync split for the main repo).
|
||||
KERNEL_REPO="securityonionkernel"
|
||||
KERNEL_REPO_SYNC="securityonionkernelsync"
|
||||
KERNEL_PKG="kernel-uek"
|
||||
KERNEL_REPO_DIR="/nsm/kernelrepo"
|
||||
REPOSYNC_CONF="/opt/so/conf/reposync/repodownload.conf"
|
||||
GLOBAL_PILLAR="/opt/so/saltstack/local/pillar/global/soc_global.sls"
|
||||
|
||||
# Stock EL9 (RHCK) kernel packages, removed only once the node is running UEK8 (see step 4
|
||||
# in the header). Left deliberately narrow: UEK7 kernel-uek builds age out on their own via
|
||||
# installonly_limit=3, and kernel-devel/kernel-headers are not touched.
|
||||
RHCK_PKGS="kernel kernel-core kernel-modules kernel-modules-core kernel-tools kernel-tools-libs"
|
||||
|
||||
log() { echo "[so-kernel-upgrade] $*"; }
|
||||
die() { echo "[so-kernel-upgrade] ERROR: $*" >&2; exit 1; }
|
||||
|
||||
[ "$(id -u)" -eq 0 ] || { log "must run as root"; exit 1; }
|
||||
command -v grubby >/dev/null 2>&1 || { log "grubby not found"; exit 1; }
|
||||
command -v grubby >/dev/null 2>&1 || die "grubby not found"
|
||||
command -v dnf >/dev/null 2>&1 || die "dnf not found"
|
||||
|
||||
ARCH="$(rpm -E '%{_arch}')"
|
||||
|
||||
is_airgap() {
|
||||
[ -f "$GLOBAL_PILLAR" ] && grep -q 'airgap: *[Tt]rue' "$GLOBAL_PILLAR"
|
||||
}
|
||||
|
||||
# Newest installed UEK8 (6.x) kernel known to the bootloader. UEK8 vmlinuz paths look like
|
||||
# /boot/vmlinuz-6.12.0-203.76.7.5.el9uek.x86_64; the 5.x UEK7 and 5.14 RHCK won't match.
|
||||
target="$(grubby --info=ALL 2>/dev/null \
|
||||
| sed -n 's/^kernel="\(.*\)"$/\1/p' \
|
||||
| grep -E '/vmlinuz-6\.[0-9]+.*uek' \
|
||||
| sort -V | tail -1)"
|
||||
# /boot/vmlinuz-6.12.0-204.92.4.2.el9uek.x86_64; UEK7 (5.15) and RHCK (5.14) won't match.
|
||||
find_uek8() {
|
||||
grubby --info=ALL 2>/dev/null \
|
||||
| sed -n 's/^kernel="\(.*\)"$/\1/p' \
|
||||
| grep -E '/vmlinuz-6\.[0-9]+.*uek' \
|
||||
| sort -V | tail -1
|
||||
}
|
||||
|
||||
if [ -z "$target" ]; then
|
||||
log "no installed 6.x UEK (UEK8) kernel found — confirm the kernel repo is assigned and"
|
||||
log "'dnf update' has installed kernel-uek-core. Nothing to do."
|
||||
# Classify the RUNNING kernel (uname -r) -- this, not what's installed, is what decides whether
|
||||
# a UEK8 install auto-promotes to the boot default:
|
||||
# uek8 6.x UEK already on the target line; nothing to do
|
||||
# uek7 5.x UEK a UEK8 install stays in the kernel-uek lineage and auto-promotes (no grubby)
|
||||
# rhck 5.14 EL9 crossing into the UEK flavor does NOT auto-promote (needs grubby --set-default)
|
||||
running_flavor() {
|
||||
case "$(uname -r)" in
|
||||
6.*uek*) echo uek8 ;;
|
||||
*uek*) echo uek7 ;;
|
||||
*) echo rhck ;;
|
||||
esac
|
||||
}
|
||||
|
||||
# Newest UEK8 kernel-uek NEVRA offered by the kernel repo, empty if the repo has none.
|
||||
# Restricted to the kernel repo so a UEK7 kernel-uek in the main repo can't be picked up,
|
||||
# and filtered to 6.x so we never "succeed" by reinstalling the 5.15 we already have.
|
||||
uek8_available() {
|
||||
dnf -q repoquery --disablerepo='*' --enablerepo="$KERNEL_REPO" \
|
||||
--arch="$ARCH" --latest-limit=1 \
|
||||
--qf '%{name}-%{evr}.%{arch}\n' "$KERNEL_PKG" 2>/dev/null \
|
||||
| grep -E "^${KERNEL_PKG}-6\." | tail -1
|
||||
}
|
||||
|
||||
kernelrepo_rpm_count() {
|
||||
find "$KERNEL_REPO_DIR" -maxdepth 1 -name '*.rpm' 2>/dev/null | wc -l
|
||||
}
|
||||
|
||||
# The kernel repo starts life as valid-but-empty (kernelrepo_init_empty in
|
||||
# salt/manager/init.sls) and is filled by so-repo-sync. During a soup, so-repo-sync runs
|
||||
# BEFORE the highstate deploys the [securityonionkernelsync] section into repodownload.conf, so
|
||||
# the first kernel-aware soup leaves the repo empty until the next nightly sync.
|
||||
sync_kernel_repo() {
|
||||
if is_airgap; then
|
||||
log "airgap install: $KERNEL_REPO_DIR is populated from the airgap ISO, not by so-repo-sync."
|
||||
return 1
|
||||
fi
|
||||
if ! grep -q "^\[${KERNEL_REPO_SYNC}\]" "$REPOSYNC_CONF" 2>/dev/null; then
|
||||
log "$REPOSYNC_CONF has no [${KERNEL_REPO_SYNC}] section -- run a highstate to deploy it."
|
||||
return 1
|
||||
fi
|
||||
|
||||
log "populating $KERNEL_REPO_DIR with so-repo-sync (mirrors upstream; can take several minutes)"
|
||||
su socore -c '/usr/sbin/so-repo-sync' || { log "so-repo-sync failed"; return 1; }
|
||||
|
||||
dnf -q clean expire-cache >/dev/null 2>&1
|
||||
return 0
|
||||
}
|
||||
|
||||
# Make the kernel repo actually able to serve a UEK8 package, or fail trying.
|
||||
ensure_kernel_repo() {
|
||||
# The repo is assigned by the repo.client highstate, and only once NICs are pinned by MAC
|
||||
# (/opt/so/state/nic_names_pinned) so the kernel swap can't renumber interfaces SO binds
|
||||
# by name. skip_if_unavailable=1 means a broken repo is silently ignored, so check first.
|
||||
if ! dnf -q repolist --enabled 2>/dev/null | awk '{print $1}' | grep -qx "$KERNEL_REPO"; then
|
||||
log "repo '$KERNEL_REPO' is not enabled on this node."
|
||||
log "Run a highstate first; the repo is skipped until /opt/so/state/nic_names_pinned"
|
||||
log "exists (run so-nic-pin) and this node's salt matches the version this release ships."
|
||||
die "kernel repo unavailable"
|
||||
fi
|
||||
|
||||
[ -n "$(uek8_available)" ] && return 0
|
||||
|
||||
log "repo '$KERNEL_REPO' is enabled but offers no UEK8 $KERNEL_PKG package"
|
||||
|
||||
if ! is_manager_node; then
|
||||
log "This is a minion; it consumes the kernel repo from the manager and cannot populate it."
|
||||
log "On the manager, run: su socore -c /usr/sbin/so-repo-sync"
|
||||
log "then re-run this script here."
|
||||
die "manager's kernel repo is empty"
|
||||
fi
|
||||
|
||||
log "this is a manager and $KERNEL_REPO_DIR holds $(kernelrepo_rpm_count) rpm(s)"
|
||||
sync_kernel_repo || die "could not populate $KERNEL_REPO_DIR"
|
||||
|
||||
[ -n "$(uek8_available)" ] \
|
||||
|| die "so-repo-sync completed but $KERNEL_REPO still offers no UEK8 $KERNEL_PKG"
|
||||
}
|
||||
|
||||
reboot_notice() {
|
||||
[ "$(uname -r)" = "$(basename "$1" | sed 's/^vmlinuz-//')" ] && return 0
|
||||
log "REBOOT REQUIRED to start using the UEK8 kernel (currently running $(uname -r))."
|
||||
# The stock kernel can't be removed until it stops being the running one, so say when
|
||||
# that will happen rather than leaving the admin to wonder if it was missed.
|
||||
[ -n "$(rhck_installed)" ] \
|
||||
&& log "The stock EL9 kernel is left in place until then; it is removed by the next highstate after the reboot."
|
||||
return 0
|
||||
}
|
||||
|
||||
# Keep future kernel updates on the UEK line rather than falling back to RHCK. Oracle ships
|
||||
# /etc/sysconfig/kernel; only rewrite it when it's actually pointing somewhere else.
|
||||
set_default_kernel_conf() {
|
||||
if [ -f /etc/sysconfig/kernel ] && ! grep -q '^DEFAULTKERNEL=kernel-uek-core$' /etc/sysconfig/kernel; then
|
||||
log "setting DEFAULTKERNEL=kernel-uek-core in /etc/sysconfig/kernel"
|
||||
sed -i 's/^DEFAULTKERNEL=.*/DEFAULTKERNEL=kernel-uek-core/' /etc/sysconfig/kernel
|
||||
fi
|
||||
}
|
||||
|
||||
# Which of RHCK_PKGS are actually installed, one per line. rpm -qa treats each argument as a
|
||||
# name glob and prints only what it finds, so a package that was never installed (or is
|
||||
# already gone) simply doesn't appear -- no "not installed" noise and no non-zero exit.
|
||||
rhck_installed() {
|
||||
rpm -qa $RHCK_PKGS 2>/dev/null
|
||||
}
|
||||
|
||||
# Remove the stock EL9 kernel. Only ever called once the running kernel is UEK8. The rpm
|
||||
# check above is the idempotency guard, so this is a cheap no-op on every highstate after
|
||||
# the first one -- it costs an rpm query, not a dnf transaction.
|
||||
remove_rhck() {
|
||||
local installed; installed="$(rhck_installed)"
|
||||
if [ -z "$installed" ]; then
|
||||
log "no stock EL9 (RHCK) kernel packages installed; nothing to remove."
|
||||
return 0
|
||||
fi
|
||||
|
||||
log "running UEK8; removing the stock EL9 (RHCK) kernel packages:"
|
||||
echo "$installed" | sed 's/^/[so-kernel-upgrade] /'
|
||||
dnf -y remove $RHCK_PKGS || die "failed to remove the stock EL9 kernel packages"
|
||||
|
||||
installed="$(rhck_installed)"
|
||||
[ -z "$installed" ] || die "dnf reported success but these remain: $(echo $installed)"
|
||||
log "stock EL9 kernel packages removed."
|
||||
}
|
||||
|
||||
# Make sure a UEK8 kernel is installed, leaving its boot entry in INSTALLED_UEK8. If one is
|
||||
# already present we leave the repo alone -- it may be disabled or empty and we don't need it
|
||||
# just to flip the boot default. Otherwise install the explicit NEVRA, not the bare package
|
||||
# name: on a UEK7 node 'dnf install kernel-uek' sees 5.15 already present, prints "Nothing to
|
||||
# do" and exits 0 without installing 6.x.
|
||||
ensure_uek8_installed() {
|
||||
INSTALLED_UEK8="$(find_uek8)"
|
||||
if [ -n "$INSTALLED_UEK8" ]; then
|
||||
log "UEK8 kernel already installed: $INSTALLED_UEK8"
|
||||
return 0
|
||||
fi
|
||||
|
||||
ensure_kernel_repo
|
||||
local nevra; nevra="$(uek8_available)"
|
||||
log "installing $nevra from $KERNEL_REPO"
|
||||
dnf -y install "$nevra" || die "failed to install $nevra"
|
||||
|
||||
INSTALLED_UEK8="$(find_uek8)"
|
||||
[ -n "$INSTALLED_UEK8" ] || die "$nevra installed but no 6.x UEK boot entry appeared -- check 'grubby --info=ALL'"
|
||||
log "installed UEK8 kernel: $INSTALLED_UEK8"
|
||||
}
|
||||
|
||||
# --cleanup does step 4 and nothing else. It exits 0 rather than failing on a node that
|
||||
# isn't on UEK8 yet: the highstate gates on 'uname -r' before calling this, and a state that
|
||||
# fails whenever that gate races would be worse than one that says what it's waiting for.
|
||||
case "$1" in
|
||||
"")
|
||||
;;
|
||||
--cleanup)
|
||||
if [ "$(running_flavor)" != uek8 ]; then
|
||||
log "not running a UEK8 kernel yet (currently $(uname -r)); leaving the stock EL9 kernel in place."
|
||||
log "Run so-kernel-upgrade with no arguments to install UEK8, then reboot."
|
||||
exit 0
|
||||
fi
|
||||
set_default_kernel_conf
|
||||
remove_rhck
|
||||
exit 0
|
||||
fi
|
||||
|
||||
current="$(grubby --default-kernel 2>/dev/null)"
|
||||
if [ "$current" = "$target" ]; then
|
||||
log "UEK8 kernel is already the boot default: $target"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
log "current default kernel: ${current:-unknown}"
|
||||
log "switching boot default to UEK8 kernel: $target"
|
||||
grubby --set-default="$target" || { log "ERROR: grubby --set-default failed for $target"; exit 1; }
|
||||
|
||||
# Verify the change actually took before claiming success.
|
||||
now="$(grubby --default-kernel 2>/dev/null)"
|
||||
if [ "$now" != "$target" ]; then
|
||||
log "ERROR: default kernel is still '${now:-unknown}' after set-default"
|
||||
;;
|
||||
*)
|
||||
echo "Usage: so-kernel-upgrade [--cleanup]" >&2
|
||||
echo " (no arguments) install UEK8, make it the boot default, clean up once it's running" >&2
|
||||
echo " --cleanup remove the stock EL9 kernel; no-op unless already running UEK8" >&2
|
||||
exit 1
|
||||
fi
|
||||
;;
|
||||
esac
|
||||
|
||||
log "boot default is now $target"
|
||||
log "REBOOT REQUIRED to start using the UEK8 kernel (currently running $(uname -r))."
|
||||
case "$(running_flavor)" in
|
||||
uek8)
|
||||
# Already on the 6.x UEK line. A plain 'dnf update' keeps this node current within the
|
||||
# lineage and auto-promotes newer builds, so there is no install or grubby work left --
|
||||
# only the step 4 cleanup, which this is the first point in the sequence that can run it.
|
||||
log "already running a UEK8 kernel ($(uname -r)); no kernel install needed."
|
||||
set_default_kernel_conf
|
||||
remove_rhck
|
||||
;;
|
||||
|
||||
uek7)
|
||||
# On a 5.x UEK kernel. Installing UEK8 stays inside the kernel-uek lineage, so dnf/grubby
|
||||
# (UPDATEDEFAULT=yes) auto-promote it and we do NOT touch grubby. A node still on UEK7
|
||||
# usually means the kernel repo was empty when it last updated, so populate it and install.
|
||||
log "running UEK7 kernel ($(uname -r)); the kernel repo was likely not yet populated when"
|
||||
log "this node last updated. Populating it and installing UEK8 -- the update stays on the"
|
||||
log "kernel-uek line, so it becomes the boot default automatically (no grubby change needed)."
|
||||
set_default_kernel_conf
|
||||
ensure_uek8_installed
|
||||
|
||||
now="$(grubby --default-kernel 2>/dev/null)"
|
||||
if [ "$now" = "$INSTALLED_UEK8" ]; then
|
||||
log "boot default auto-promoted to UEK8 kernel: $INSTALLED_UEK8"
|
||||
else
|
||||
log "WARNING: expected the UEK8 kernel to auto-promote but the default is still"
|
||||
log "'${now:-unknown}'. Run 'grubby --set-default=$INSTALLED_UEK8' to force it."
|
||||
fi
|
||||
reboot_notice "$INSTALLED_UEK8"
|
||||
;;
|
||||
|
||||
rhck)
|
||||
# On the stock EL9 kernel (5.14, no UEK installed). Crossing from RHCK into the UEK flavor
|
||||
# does NOT auto-promote -- kernel-install/grubby only auto-promote within the running
|
||||
# kernel's flavor lineage -- so after installing we must set the boot default explicitly.
|
||||
log "running stock EL9 (RHCK) kernel ($(uname -r)); installing UEK8 and setting it as the"
|
||||
log "boot default explicitly (a RHCK->UEK flavor change does not auto-promote)."
|
||||
set_default_kernel_conf
|
||||
ensure_uek8_installed
|
||||
target="$INSTALLED_UEK8"
|
||||
|
||||
current="$(grubby --default-kernel 2>/dev/null)"
|
||||
if [ "$current" = "$target" ]; then
|
||||
log "UEK8 kernel is already the boot default: $target"
|
||||
reboot_notice "$target"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
log "current default kernel: ${current:-unknown}"
|
||||
log "switching boot default to UEK8 kernel: $target"
|
||||
grubby --set-default="$target" || die "grubby --set-default failed for $target"
|
||||
|
||||
# Verify the change actually took before claiming success.
|
||||
now="$(grubby --default-kernel 2>/dev/null)"
|
||||
[ "$now" = "$target" ] || die "default kernel is still '${now:-unknown}' after set-default"
|
||||
|
||||
log "boot default is now $target"
|
||||
reboot_notice "$target"
|
||||
;;
|
||||
esac
|
||||
|
||||
@@ -132,6 +132,9 @@ if [[ $EXCLUDE_STARTUP_ERRORS == 'Y' ]]; then
|
||||
EXCLUDED_ERRORS="$EXCLUDED_ERRORS|HTTP 404: Not Found" # Salt loops until Kratos returns 200, during startup Kratos may not be ready
|
||||
EXCLUDED_ERRORS="$EXCLUDED_ERRORS|Cancelling deferred write event maybeFenceReplicas because the event queue is now closed" # Kafka controller log during shutdown/restart
|
||||
EXCLUDED_ERRORS="$EXCLUDED_ERRORS|Redis may have been restarted" # Redis likely restarted by salt
|
||||
EXCLUDED_ERRORS="$EXCLUDED_ERRORS|file already closed" # Go logging race condition during container restart
|
||||
EXCLUDED_ERRORS="$EXCLUDED_ERRORS|relation \"audit_settings\" does not exist" # salt checking for changes before SOC starts
|
||||
EXCLUDED_ERRORS="$EXCLUDED_ERRORS|Error in plugin: elasticsearch: Unable to retrieve master node information" # expected error while ES is upgrading/electing a master
|
||||
fi
|
||||
|
||||
if [[ $EXCLUDE_FALSE_POSITIVE_ERRORS == 'Y' ]]; then
|
||||
@@ -152,6 +155,8 @@ if [[ $EXCLUDE_FALSE_POSITIVE_ERRORS == 'Y' ]]; then
|
||||
EXCLUDED_ERRORS="$EXCLUDED_ERRORS|id.orig_h" # false positive (zeek test data)
|
||||
EXCLUDED_ERRORS="$EXCLUDED_ERRORS|emerging-all.rules" # false positive (error in rulename)
|
||||
EXCLUDED_ERRORS="$EXCLUDED_ERRORS|invalid query input" # false positive (Invalid user input in hunt query)
|
||||
EXCLUDED_ERRORS="$EXCLUDED_ERRORS|no data available for the requested dates" # false positive (pcap cypress test submits a job with an empty time frame)
|
||||
EXCLUDED_ERRORS="$EXCLUDED_ERRORS|no job processor" # false positive (same empty-time-frame job on import nodes, where no pcap processor runs)
|
||||
EXCLUDED_ERRORS="$EXCLUDED_ERRORS|example" # false positive (example test data)
|
||||
EXCLUDED_ERRORS="$EXCLUDED_ERRORS|status 200" # false positive (request successful, contained error string in content)
|
||||
EXCLUDED_ERRORS="$EXCLUDED_ERRORS|app_layer.error" # false positive (suricata 7) in stats.log e.g. app_layer.error.imap.parser | Total | 0
|
||||
@@ -167,6 +172,11 @@ if [[ $EXCLUDE_FALSE_POSITIVE_ERRORS == 'Y' ]]; then
|
||||
EXCLUDED_ERRORS="$EXCLUDED_ERRORS|Error while parsing document for index \[.ds-logs-kratos-so-.*object mapping for \[file\]" # false positive (mapping error occuring BEFORE kratos index has rolled over in 2.4.210)
|
||||
EXCLUDED_ERRORS="$EXCLUDED_ERRORS|No such container" # false positive (telegraf trying to run stats on an old container)
|
||||
EXCLUDED_ERRORS="$EXCLUDED_ERRORS|passwords do not match" # false positive (automated hydra test)
|
||||
EXCLUDED_ERRORS="$EXCLUDED_ERRORS|Request did not pass preprocessing" # expected WARN log lines indicating invalid auth header
|
||||
EXCLUDED_ERRORS="$EXCLUDED_ERRORS|Missing or invalid authorization header for bearer token" # expected WARN log lines indicating invalid auth header
|
||||
EXCLUDED_ERRORS="$EXCLUDED_ERRORS|Unexpected authorization header" # expected WARN log lines indicating invalid auth header
|
||||
EXCLUDED_ERRORS="$EXCLUDED_ERRORS|Missing ory_kratos_session cookie" # expected WARN log lines indicating invalid auth header
|
||||
EXCLUDED_ERRORS="$EXCLUDED_ERRORS|Static assets preprocessor only supports GET and HEAD requests" # expected WARN log lines indicating invalid auth header
|
||||
fi
|
||||
|
||||
if [[ $EXCLUDE_KNOWN_ERRORS == 'Y' ]]; then
|
||||
@@ -229,8 +239,9 @@ if [[ $EXCLUDE_KNOWN_ERRORS == 'Y' ]]; then
|
||||
EXCLUDED_ERRORS="$EXCLUDED_ERRORS|from NIC checksum offloading" # zeek reporter.log
|
||||
EXCLUDED_ERRORS="$EXCLUDED_ERRORS|marked for removal" # docker container getting recycled
|
||||
EXCLUDED_ERRORS="$EXCLUDED_ERRORS|tcp 127.0.0.1:6791: bind: address already in use" # so-elastic-fleet agent restarting. Seen starting w/ 8.18.8 https://github.com/elastic/kibana/issues/201459
|
||||
EXCLUDED_ERRORS="$EXCLUDED_ERRORS|TransformTask\] \[logs-(tychon|aws_billing|microsoft_defender_endpoint|armis|o365_metrics|microsoft_sentinel|snyk|cyera|island_browser).*user so_kibana lacks the required permissions \[(logs|metrics)-\1" # Known issue with integrations starting transform jobs that are explicitly not allowed to start as a system user. This error should not be seen on fresh ES 9.3.3 installs or after SO 3.1.0 with soups addition of check_transform_health_and_reauthorize()
|
||||
EXCLUDED_ERRORS="$EXCLUDED_ERRORS|TransformTask\] \[logs-.*user so_kibana lacks the required permissions" # Known issue with integrations starting transform jobs that are explicitly not allowed to start as a system user
|
||||
EXCLUDED_ERRORS="$EXCLUDED_ERRORS|manifest unknown" # appears in so-dockerregistry log for so-tcpreplay following docker upgrade to 29.2.1-1
|
||||
EXCLUDED_ERRORS="$EXCLUDED_ERRORS|Could not index event to Elasticsearch.*\"version\" => \"9.0.8\"" # Expected during Elastic upgrade temporarily, as policies referencing older pipelines are updated
|
||||
fi
|
||||
|
||||
RESULT=0
|
||||
@@ -295,4 +306,4 @@ else
|
||||
echo -e "\nResult: One or more errors found"
|
||||
fi
|
||||
|
||||
exit $RESULT
|
||||
exit $RESULT
|
||||
@@ -35,9 +35,6 @@ case $1 in
|
||||
"elastic-fleet"|"elasticfleet")
|
||||
docker_check_running "elastic-fleet" "--stop"
|
||||
docker rm "so-elastic-fleet" 2> /dev/null
|
||||
# Removing the elastic fleet state directory, so that the next startup re-enrolls with a fresh policy
|
||||
rm -rf /opt/so/conf/elastic-fleet/state
|
||||
|
||||
salt-call state.apply elasticfleet queue=True
|
||||
;;
|
||||
*)
|
||||
|
||||
@@ -8,21 +8,37 @@
|
||||
# Elastic License 2.0.
|
||||
|
||||
|
||||
SENSOR_DIR='/nsm'
|
||||
SENSOR_DIR="${SENSOR_DIR:-/nsm}"
|
||||
CRIT_DISK_USAGE=90
|
||||
CUR_USAGE=$(df -P $SENSOR_DIR | tail -1 | awk '{print $5}' | tr -d %)
|
||||
LOG="/opt/so/log/sensor_clean.log"
|
||||
TODAY=$(date -u "+%Y-%m-%d")
|
||||
LOG="${LOG:-/opt/so/log/sensor_clean.log}"
|
||||
LOCK="${LOCK:-/var/tmp/so-sensor-clean.lock}"
|
||||
MAX_PASSES=100
|
||||
|
||||
ZEEK_LOGS="$SENSOR_DIR/zeek/logs"
|
||||
STRELKA_FILES="$SENSOR_DIR/strelka/processed"
|
||||
SURICATA_LOGS="$SENSOR_DIR/suricata"
|
||||
PCAPS="$SENSOR_DIR/pcapout"
|
||||
|
||||
log() {
|
||||
echo "$(date) - $*" >>"$LOG"
|
||||
}
|
||||
|
||||
disk_usage() {
|
||||
df -P "$SENSOR_DIR" | tail -1 | awk '{print $5}' | tr -d %
|
||||
}
|
||||
|
||||
disk_avail() {
|
||||
df -P "$SENSOR_DIR" | tail -1 | awk '{print $4}'
|
||||
}
|
||||
|
||||
# sets REMOVED=1 if anything was actually deleted
|
||||
clean() {
|
||||
## find the oldest Zeek logs directory
|
||||
OLDEST_DIR=$(ls /nsm/zeek/logs/ | grep -v "current" | grep -v "stats" | grep -v "packetloss" | grep -v "zeek_clean" | sort | head -n 1)
|
||||
if [ -z "$OLDEST_DIR" -o "$OLDEST_DIR" == ".." -o "$OLDEST_DIR" == "." ]; then
|
||||
echo "$(date) - No old Zeek logs available to clean up in /nsm/zeek/logs/" >>$LOG
|
||||
#exit 0
|
||||
else
|
||||
echo "$(date) - Removing directory: /nsm/zeek/logs/$OLDEST_DIR" >>$LOG
|
||||
rm -rf /nsm/zeek/logs/"$OLDEST_DIR"
|
||||
OLDEST_DIR=$(ls "$ZEEK_LOGS" 2>/dev/null | grep -v "current" | grep -v "stats" | grep -v "packetloss" | grep -v "zeek_clean" | sort | head -n 1)
|
||||
if [ -n "$OLDEST_DIR" ]; then
|
||||
log "Removing directory: $ZEEK_LOGS/$OLDEST_DIR"
|
||||
rm -rf "$ZEEK_LOGS/$OLDEST_DIR"
|
||||
REMOVED=1
|
||||
fi
|
||||
|
||||
## Remarking for now, as we are moving extracted files to /nsm/strelka/processed
|
||||
@@ -43,58 +59,73 @@ clean() {
|
||||
#fi
|
||||
|
||||
## Clean up Zeek extracted files processed by Strelka
|
||||
STRELKA_FILES='/nsm/strelka/processed'
|
||||
OLDEST_STRELKA=$(find $STRELKA_FILES -type f -printf '%T+ %p\n' | sort -n | head -n 1)
|
||||
if [ -z "$OLDEST_STRELKA" -o "$OLDEST_STRELKA" == ".." -o "$OLDEST_STRELKA" == "." ]; then
|
||||
echo "$(date) - No old files available to clean up in $STRELKA_FILES" >>$LOG
|
||||
else
|
||||
OLDEST_STRELKA=$(find "$STRELKA_FILES" -type f -printf '%T+ %p\n' 2>/dev/null | sort -n | head -n 1)
|
||||
if [ -n "$OLDEST_STRELKA" ]; then
|
||||
OLDEST_STRELKA_DATE=$(echo $OLDEST_STRELKA | awk '{print $1}' | cut -d+ -f1)
|
||||
OLDEST_STRELKA_FILE=$(echo $OLDEST_STRELKA | awk '{print $2}')
|
||||
echo "$(date) - Removing extracted files for $OLDEST_STRELKA_DATE" >>$LOG
|
||||
find $STRELKA_FILES -type f -printf '%T+ %p\n' | grep $OLDEST_STRELKA_DATE | awk '{print $2}' | while read FILE; do
|
||||
echo "$(date) - Removing file: $FILE" >>$LOG
|
||||
log "Removing extracted files for $OLDEST_STRELKA_DATE"
|
||||
REMOVED=1
|
||||
find "$STRELKA_FILES" -type f -printf '%T+ %p\n' 2>/dev/null | grep $OLDEST_STRELKA_DATE | awk '{print $2}' | while read FILE; do
|
||||
log "Removing file: $FILE"
|
||||
rm -f "$FILE"
|
||||
done
|
||||
fi
|
||||
|
||||
## Clean up Suricata log files
|
||||
SURICATA_LOGS='/nsm/suricata'
|
||||
OLDEST_SURICATA=$(find $SURICATA_LOGS -type f -printf '%T+ %p\n' | sort -n | head -n 1)
|
||||
if [[ -z "$OLDEST_SURICATA" ]] || [[ "$OLDEST_SURICATA" == ".." ]] || [[ "$OLDEST_SURICATA" == "." ]]; then
|
||||
echo "$(date) - No old files available to clean up in $SURICATA_LOGS" >>$LOG
|
||||
else
|
||||
OLDEST_SURICATA=$(find "$SURICATA_LOGS" -type f -printf '%T+ %p\n' 2>/dev/null | sort -n | head -n 1)
|
||||
if [ -n "$OLDEST_SURICATA" ]; then
|
||||
OLDEST_SURICATA_DATE=$(echo $OLDEST_SURICATA | awk '{print $1}' | cut -d+ -f1)
|
||||
OLDEST_SURICATA_FILE=$(echo $OLDEST_SURICATA | awk '{print $2}')
|
||||
echo "$(date) - Removing logs for $OLDEST_SURICATA_DATE" >>$LOG
|
||||
find $SURICATA_LOGS -type f -printf '%T+ %p\n' | grep $OLDEST_SURICATA_DATE | awk '{print $2}' | while read FILE; do
|
||||
echo "$(date) - Removing file: $FILE" >>$LOG
|
||||
log "Removing logs for $OLDEST_SURICATA_DATE"
|
||||
REMOVED=1
|
||||
find "$SURICATA_LOGS" -type f -printf '%T+ %p\n' 2>/dev/null | grep $OLDEST_SURICATA_DATE | awk '{print $2}' | while read FILE; do
|
||||
log "Removing file: $FILE"
|
||||
rm -f "$FILE"
|
||||
done
|
||||
fi
|
||||
|
||||
## Clean up extracted pcaps
|
||||
PCAPS='/nsm/pcapout'
|
||||
OLDEST_PCAP=$(find $PCAPS -type f -printf '%T+ %p\n' | sort -n | head -n 1)
|
||||
if [ -z "$OLDEST_PCAP" -o "$OLDEST_PCAP" == ".." -o "$OLDEST_PCAP" == "." ]; then
|
||||
echo "$(date) - No old files available to clean up in $PCAPS" >>$LOG
|
||||
else
|
||||
OLDEST_PCAP=$(find "$PCAPS" -type f -printf '%T+ %p\n' 2>/dev/null | sort -n | head -n 1)
|
||||
if [ -n "$OLDEST_PCAP" ]; then
|
||||
OLDEST_PCAP_DATE=$(echo $OLDEST_PCAP | awk '{print $1}' | cut -d+ -f1)
|
||||
OLDEST_PCAP_FILE=$(echo $OLDEST_PCAP | awk '{print $2}')
|
||||
echo "$(date) - Removing extracted files for $OLDEST_PCAP_DATE" >>$LOG
|
||||
find $PCAPS -type f -printf '%T+ %p\n' | grep $OLDEST_PCAP_DATE | awk '{print $2}' | while read FILE; do
|
||||
echo "$(date) - Removing file: $FILE" >>$LOG
|
||||
log "Removing extracted files for $OLDEST_PCAP_DATE"
|
||||
REMOVED=1
|
||||
find "$PCAPS" -type f -printf '%T+ %p\n' 2>/dev/null | grep $OLDEST_PCAP_DATE | awk '{print $2}' | while read FILE; do
|
||||
log "Removing file: $FILE"
|
||||
rm -f "$FILE"
|
||||
done
|
||||
fi
|
||||
}
|
||||
|
||||
# Check to see if we are already running
|
||||
NUM_RUNNING=$(pgrep -cf "/bin/bash /usr/sbin/so-sensor-clean")
|
||||
[ "$NUM_RUNNING" -gt 1 ] && echo "$(date) - $NUM_RUNNING sensor clean script processes running...exiting." >>$LOG && exit 0
|
||||
|
||||
if [ "$CUR_USAGE" -gt "$CRIT_DISK_USAGE" ]; then
|
||||
while [ "$CUR_USAGE" -gt "$CRIT_DISK_USAGE" ]; do
|
||||
clean
|
||||
CUR_USAGE=$(df -P $SENSOR_DIR | tail -1 | awk '{print $5}' | tr -d %)
|
||||
done
|
||||
# Only one instance at a time; the lock is the fd, so it releases on any exit
|
||||
exec 9>"$LOCK" || exit 1
|
||||
if ! flock -n 9; then
|
||||
log "another so-sensor-clean is already running (lock $LOCK held); exiting"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
CUR_USAGE=$(disk_usage)
|
||||
[ "$CUR_USAGE" -gt "$CRIT_DISK_USAGE" ] || exit 0
|
||||
|
||||
log "$SENSOR_DIR at ${CUR_USAGE}% (threshold ${CRIT_DISK_USAGE}%); starting cleanup"
|
||||
|
||||
PASS=0
|
||||
while [ "$CUR_USAGE" -gt "$CRIT_DISK_USAGE" ]; do
|
||||
PASS=$((PASS + 1))
|
||||
if [ "$PASS" -gt "$MAX_PASSES" ]; then
|
||||
log "stopping after $MAX_PASSES passes; $SENSOR_DIR still at ${CUR_USAGE}%"
|
||||
break
|
||||
fi
|
||||
|
||||
REMOVED=0
|
||||
BEFORE=$(disk_avail)
|
||||
clean
|
||||
CUR_USAGE=$(disk_usage)
|
||||
|
||||
if [ "$REMOVED" -eq 0 ]; then
|
||||
log "nothing left to remove in $ZEEK_LOGS, $STRELKA_FILES, $SURICATA_LOGS, $PCAPS; $SENSOR_DIR still at ${CUR_USAGE}% - space is consumed outside of NSM cleanup scope"
|
||||
break
|
||||
fi
|
||||
if [ "$(disk_avail)" -le "$BEFORE" ]; then
|
||||
log "pass $PASS freed no space; $SENSOR_DIR still at ${CUR_USAGE}% - stopping until next run"
|
||||
break
|
||||
fi
|
||||
done
|
||||
|
||||
@@ -74,13 +74,13 @@ def output(options, console, code, data):
|
||||
summary = { "status_code": code, "containers": data }
|
||||
print(json.dumps(summary))
|
||||
elif "-q" not in options:
|
||||
if code == 2:
|
||||
console.print(" [bold yellow]:hourglass: [bold white]System appears to be starting. No highstate has completed since the system was restarted.")
|
||||
elif code == 99:
|
||||
if code == 99:
|
||||
console.print(" [bold red]:exclamation: [bold white]Installation does not appear to be complete. A highstate has not fully completed.")
|
||||
elif code == 100:
|
||||
console.print(" [bold red]:exclamation: [bold white]Installation encountered errors.")
|
||||
else:
|
||||
if code == 2:
|
||||
console.print(" [bold yellow]:hourglass: [bold white]System appears to be starting. No highstate has completed since the system was restarted. Container status is shown below.")
|
||||
table = Table(title = "Security Onion Status", show_edge = False, safe_box = True, box = box.MINIMAL)
|
||||
table.add_column("Container", justify="right", style="white", no_wrap=True)
|
||||
table.add_column("Status", justify="left", style="green", no_wrap=True)
|
||||
@@ -154,8 +154,14 @@ def check_status(options, console):
|
||||
code = check_installation_status(options, console)
|
||||
if code == 0:
|
||||
code = check_system_status(options, console)
|
||||
if code == 0:
|
||||
code, container_list = check_container_status(options, console)
|
||||
# Containers now start on boot without a highstate, so gather/display their
|
||||
# status even when the system is still "starting" (code 2). Keep the starting
|
||||
# code as the exit/status_code so SOC keeps showing the "restarting" message
|
||||
# on the Grid until a highstate completes.
|
||||
if code == 0 or code == 2:
|
||||
container_code, container_list = check_container_status(options, console)
|
||||
if code == 0:
|
||||
code = container_code
|
||||
output(options, console, code, container_list)
|
||||
return code
|
||||
|
||||
@@ -180,4 +186,3 @@ def main():
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
|
||||
|
||||
@@ -29,8 +29,6 @@ case $1 in
|
||||
"elasticfleet"|"elastic-fleet")
|
||||
docker_check_running "elastic-fleet" "--stop"
|
||||
docker rm "so-elastic-fleet" 2> /dev/null
|
||||
# Removing the elastic fleet state directory, so that the next startup re-enrolls with a fresh policy
|
||||
rm -rf /opt/so/conf/elastic-fleet/state
|
||||
;;
|
||||
*)
|
||||
docker_check_running "$1" "--stop"
|
||||
|
||||
@@ -125,4 +125,6 @@ else
|
||||
RAIDSTATUS=1
|
||||
fi
|
||||
|
||||
echo "nsmraid=$RAIDSTATUS" > /opt/so/log/raid/status.log
|
||||
# telegraf reads this file; write aside and rename so it never sees a half-written file
|
||||
echo "nsmraid=$RAIDSTATUS" > /opt/so/log/raid/status.log.tmp
|
||||
mv -f /opt/so/log/raid/status.log.tmp /opt/so/log/raid/status.log
|
||||
|
||||
@@ -1,5 +1,3 @@
|
||||
{% import_yaml 'salt/minion.defaults.yaml' as SALT_MINION_DEFAULTS -%}
|
||||
|
||||
#!/bin/bash
|
||||
#
|
||||
# Copyright Security Onion Solutions LLC and/or licensed to Security Onion Solutions LLC under one
|
||||
@@ -7,7 +5,7 @@
|
||||
# https://securityonion.net/license; you may not use this file except in compliance with the
|
||||
# Elastic License 2.0.
|
||||
|
||||
|
||||
{% from 'salt/schedule.map.jinja' import SCHEDULEMERGED %}
|
||||
|
||||
# this script checks the time the file /opt/so/log/salt/state-apply-test was last modified and restarts the salt-minion service if it is outside a threshold date/time
|
||||
# the file is modified via file.touch using a scheduled job healthcheck.salt-minion.state-apply-test that runs a state.apply.
|
||||
@@ -20,12 +18,14 @@
|
||||
|
||||
QUIET=false
|
||||
UPTIME_REQ=1800 #in seconds, how long the box has to be up before considering restarting salt-minion due to /opt/so/log/salt/state-apply-test not being touched
|
||||
HIGHSTATE_UPTIME_REQ=900 #in seconds; if the box has been up this long and no highstate has completed since boot, force one
|
||||
CURRENT_TIME=$(date +%s)
|
||||
SYSTEM_START_TIME=$(date -d "$(</proc/uptime awk '{print $1}') seconds ago" +%s)
|
||||
LAST_HIGHSTATE_END=$([ -e "/opt/so/log/salt/lasthighstate" ] && date -r /opt/so/log/salt/lasthighstate +%s || echo 0)
|
||||
LAST_HEALTHCHECK_STATE_APPLY=$([ -e "/opt/so/log/salt/state-apply-test" ] && date -r /opt/so/log/salt/state-apply-test +%s || echo 0)
|
||||
# SETTING THRESHOLD TO ANYTHING UNDER 600 seconds may cause a lot of salt-minion restarts since the job to touch the file occurs every 5-8 minutes by default
|
||||
THRESHOLD={{SALT_MINION_DEFAULTS.salt.minion.check_threshold}} #within how many seconds the file /opt/so/log/salt/state-apply-test must have been touched/modified before the salt minion is restarted
|
||||
# THRESHOLD is derived from the salt schedule highstate interval + 1 hour, so the minion-check grace period tracks the schedule automatically.
|
||||
THRESHOLD=$(( ({{ SCHEDULEMERGED.highstate_interval_minutes }} + 60) * 60 )) #within how many seconds the file /opt/so/log/salt/state-apply-test must have been touched/modified before the salt minion is restarted
|
||||
THRESHOLD_DATE=$((LAST_HEALTHCHECK_STATE_APPLY+THRESHOLD))
|
||||
|
||||
logCmd() {
|
||||
@@ -77,24 +77,50 @@ done
|
||||
|
||||
log "running so-salt-minion-check"
|
||||
|
||||
RESTARTED=false
|
||||
|
||||
# Check 1 (minion-restart-check): if the minion has stopped applying states (the
|
||||
# state-apply-test healthcheck file has gone stale), restart the salt-minion service.
|
||||
if [ $CURRENT_TIME -ge $((SYSTEM_START_TIME+$UPTIME_REQ)) ]; then
|
||||
if [ $THRESHOLD_DATE -le $CURRENT_TIME ]; then
|
||||
log "salt-minion is unable to apply states" E
|
||||
log "/opt/so/log/salt/healthcheck-state-apply not touched by required date: `date -d @$THRESHOLD_DATE`, last touched: `date -d @$LAST_HEALTHCHECK_STATE_APPLY`" I
|
||||
log "last highstate completed at `date -d @$LAST_HIGHSTATE_END`" I
|
||||
log "checking if any jobs are running" I
|
||||
log "[minion-restart-check] salt-minion is unable to apply states; restarting salt-minion" E
|
||||
log "[minion-restart-check] state-apply-test not touched by required date `date -d @$THRESHOLD_DATE`, last touched `date -d @$LAST_HEALTHCHECK_STATE_APPLY`" I
|
||||
log "[minion-restart-check] last highstate completed at `date -d @$LAST_HIGHSTATE_END`" I
|
||||
log "[minion-restart-check] checking if any jobs are running" I
|
||||
logCmd "salt-call --local saltutil.running" I
|
||||
log "ensure salt.minion-state-apply-test is enabled" I
|
||||
log "[minion-restart-check] ensure salt.minion-state-apply-test is enabled" I
|
||||
logCmd "salt-call state.enable salt.minion-state-apply-test" I
|
||||
log "ensure highstate is enabled" I
|
||||
log "[minion-restart-check] ensure highstate is enabled" I
|
||||
logCmd "salt-call state.enable highstate" I
|
||||
log "killing all salt-minion processes" I
|
||||
log "[minion-restart-check] killing all salt-minion processes" I
|
||||
logCmd "pkill -9 -ef /usr/bin/salt-minion" I
|
||||
log "starting salt-minion service" I
|
||||
log "[minion-restart-check] starting salt-minion service" I
|
||||
logCmd "systemctl start salt-minion" I
|
||||
log "[minion-restart-check] waiting for salt-minion to become ready, then applying highstate in the background (queued)" I
|
||||
nohup bash -c '/usr/sbin/so-salt-minion-wait; salt-call state.highstate queue=True' >> "/opt/so/log/salt/so-salt-minion-check" 2>&1 &
|
||||
RESTARTED=true
|
||||
else
|
||||
log "/opt/so/log/salt/healthcheck-state-apply last touched: `date -d @$LAST_HEALTHCHECK_STATE_APPLY` must be touched by `date -d @$THRESHOLD_DATE` to avoid salt-minion restart" I
|
||||
log "[minion-restart-check] healthy: state-apply-test last touched `date -d @$LAST_HEALTHCHECK_STATE_APPLY`, must go stale past `date -d @$THRESHOLD_DATE` to trigger a salt-minion restart" I
|
||||
fi
|
||||
else
|
||||
log "system uptime only $((CURRENT_TIME-SYSTEM_START_TIME)) seconds does not meet $UPTIME_REQ second requirement." I
|
||||
log "[minion-restart-check] skipped: system uptime $((CURRENT_TIME-SYSTEM_START_TIME))s is below the ${UPTIME_REQ}s minimum required before a salt-minion restart" I
|
||||
fi
|
||||
|
||||
# Check 2 (boot-highstate-check): if the host has been up long enough but no highstate
|
||||
# has completed since this boot, force one. This recovers a host whose boot highstate
|
||||
# (so-boot-highstate.service) failed or was skipped, even while the minion is otherwise
|
||||
# healthy (touching state-apply-test). We deliberately do NOT enable highstate here: if
|
||||
# soup has disabled it during an upgrade, Salt will refuse the highstate and we avoid
|
||||
# forcing one mid-upgrade.
|
||||
if $RESTARTED; then
|
||||
log "[boot-highstate-check] skipped: minion-restart-check already queued a highstate this run" I
|
||||
elif [ $CURRENT_TIME -lt $((SYSTEM_START_TIME+HIGHSTATE_UPTIME_REQ)) ]; then
|
||||
log "[boot-highstate-check] skipped: system uptime $((CURRENT_TIME-SYSTEM_START_TIME))s is below the ${HIGHSTATE_UPTIME_REQ}s minimum required before forcing a highstate" I
|
||||
elif [ $LAST_HIGHSTATE_END -ge $SYSTEM_START_TIME ]; then
|
||||
log "[boot-highstate-check] healthy: a highstate completed at `date -d @$LAST_HIGHSTATE_END`, after this boot at `date -d @$SYSTEM_START_TIME`" I
|
||||
elif salt-call --local saltutil.running 2>/dev/null | grep -q 'state.highstate'; then
|
||||
log "[boot-highstate-check] no highstate has completed since boot, but one is already running; skipping" I
|
||||
else
|
||||
log "[boot-highstate-check] no highstate has completed since boot after $((CURRENT_TIME-SYSTEM_START_TIME))s uptime; applying highstate" E
|
||||
nohup bash -c 'salt-call state.highstate -l info queue=True' >> "/opt/so/log/salt/so-salt-minion-check" 2>&1 &
|
||||
fi
|
||||
|
||||
@@ -1,6 +1,12 @@
|
||||
docker:
|
||||
range: '172.17.1.0/24'
|
||||
gateway: '172.17.1.1'
|
||||
networks:
|
||||
sobridge: {}
|
||||
soauth:
|
||||
range: '172.17.2.0/24'
|
||||
gateway: '172.17.2.1'
|
||||
manager_only: True
|
||||
ulimits:
|
||||
- name: nofile
|
||||
soft: 1048576
|
||||
@@ -58,18 +64,18 @@ docker:
|
||||
ulimits: []
|
||||
'so-kratos':
|
||||
final_octet: 28
|
||||
networks: ['soauth']
|
||||
port_bindings:
|
||||
- 0.0.0.0:4433:4433
|
||||
- 0.0.0.0:4434:4434
|
||||
custom_bind_mounts: []
|
||||
extra_hosts: []
|
||||
extra_env: []
|
||||
ulimits: []
|
||||
'so-hydra':
|
||||
final_octet: 30
|
||||
networks: ['soauth']
|
||||
port_bindings:
|
||||
- 0.0.0.0:4444:4444
|
||||
- 0.0.0.0:4445:4445
|
||||
custom_bind_mounts: []
|
||||
extra_hosts: []
|
||||
extra_env: []
|
||||
@@ -128,6 +134,7 @@ docker:
|
||||
ulimits: []
|
||||
'so-soc':
|
||||
final_octet: 34
|
||||
networks: ['sobridge', 'soauth']
|
||||
port_bindings:
|
||||
- 0.0.0.0:9822:9822
|
||||
custom_bind_mounts: []
|
||||
|
||||
@@ -1,8 +1,26 @@
|
||||
{% import_yaml 'docker/defaults.yaml' as DOCKERDEFAULTS %}
|
||||
{% set DOCKERMERGED = salt['pillar.get']('docker', DOCKERDEFAULTS.docker, merge=True) %}
|
||||
{% set RANGESPLIT = DOCKERMERGED.range.split('.') %}
|
||||
{% set FIRSTTHREE = RANGESPLIT[0] ~ '.' ~ RANGESPLIT[1] ~ '.' ~ RANGESPLIT[2] ~ '.' %}
|
||||
|
||||
{% if DOCKERMERGED.networks.sobridge is not mapping %}
|
||||
{% do DOCKERMERGED.networks.update({'sobridge': {}}) %}
|
||||
{% endif %}
|
||||
{% do DOCKERMERGED.networks['sobridge'].update({'range': DOCKERMERGED.range, 'gateway': DOCKERMERGED.gateway}) %}
|
||||
|
||||
{% for netname, net in DOCKERMERGED.networks.items() %}
|
||||
{% set RANGESPLIT = net.range.split('.') %}
|
||||
{% do net.update({'prefix': RANGESPLIT[0] ~ '.' ~ RANGESPLIT[1] ~ '.' ~ RANGESPLIT[2] ~ '.'}) %}
|
||||
{% endfor %}
|
||||
|
||||
{% for container, vals in DOCKERMERGED.containers.items() %}
|
||||
{% do DOCKERMERGED.containers[container].update({'ip': FIRSTTHREE ~ DOCKERMERGED.containers[container].final_octet}) %}
|
||||
{% set CONTAINER_NETS = vals.get('networks', ['sobridge']) %}
|
||||
{% set IPS = {} %}
|
||||
{% for netname in CONTAINER_NETS %}
|
||||
{% do IPS.update({netname: DOCKERMERGED.networks[netname].prefix ~ vals.final_octet}) %}
|
||||
{% endfor %}
|
||||
{% do DOCKERMERGED.containers[container].update({
|
||||
'networks': CONTAINER_NETS,
|
||||
'ips': IPS,
|
||||
'network': CONTAINER_NETS[0],
|
||||
'ip': IPS[CONTAINER_NETS[0]]
|
||||
}) %}
|
||||
{% endfor %}
|
||||
|
||||
+10
-6
@@ -71,15 +71,19 @@ dockerreserveports:
|
||||
- source: salt://common/files/99-reserved-ports.conf
|
||||
- name: /etc/sysctl.d/99-reserved-ports.conf
|
||||
|
||||
sos_docker_net:
|
||||
{% for NETNAME, NETWORK in DOCKERMERGED.networks.items() %}
|
||||
{% if not NETWORK.get('manager_only') or GLOBALS.get('is_manager', False) %}
|
||||
sos_docker_net_{{ NETNAME }}:
|
||||
docker_network.present:
|
||||
- name: sobridge
|
||||
- subnet: {{ DOCKERMERGED.range }}
|
||||
- gateway: {{ DOCKERMERGED.gateway }}
|
||||
- name: {{ NETNAME }}
|
||||
- subnet: {{ NETWORK.range }}
|
||||
- gateway: {{ NETWORK.gateway }}
|
||||
- options:
|
||||
com.docker.network.bridge.name: 'sobridge'
|
||||
com.docker.network.bridge.name: '{{ NETNAME }}'
|
||||
com.docker.network.driver.mtu: '1500'
|
||||
com.docker.network.bridge.enable_ip_masquerade: 'true'
|
||||
com.docker.network.bridge.enable_icc: 'true'
|
||||
com.docker.network.bridge.host_binding_ipv4: '0.0.0.0'
|
||||
- unless: ip l | grep sobridge
|
||||
- unless: ip l | grep {{ NETNAME }}
|
||||
{% endif %}
|
||||
{% endfor %}
|
||||
|
||||
@@ -1,12 +0,0 @@
|
||||
{% macro clear_stale_endpoint(container, network, ipv4, state_id=None) %}
|
||||
{{ container }}_{{ network }}_stale_endpoint:
|
||||
cmd.run:
|
||||
- name: docker network disconnect -f {{ network }} {{ container }} || true
|
||||
- onlyif: docker inspect {{ container }}
|
||||
- unless: >-
|
||||
docker inspect -f
|
||||
'{{ '{{' }} with index .NetworkSettings.Networks "{{ network }}" {{ '}}' }}{{ '{{' }} .IPAMConfig.IPv4Address {{ '}}' }}{{ '{{' }} end {{ '}}' }}'
|
||||
{{ container }} 2>/dev/null | grep -qx '{{ ipv4 }}'
|
||||
- require_in:
|
||||
- docker_container: {{ state_id or container }}
|
||||
{% endmacro %}
|
||||
@@ -7,6 +7,40 @@ docker:
|
||||
description: Default docker IP range for containers.
|
||||
helpLink: docker
|
||||
advanced: True
|
||||
networks:
|
||||
sobridge:
|
||||
description: |
|
||||
The default docker network, carrying most containers. Its range and gateway are taken
|
||||
from the docker.range and docker.gateway settings above rather than set here.
|
||||
helpLink: docker
|
||||
readonly: True
|
||||
advanced: True
|
||||
global: True
|
||||
soauth:
|
||||
range:
|
||||
description: |
|
||||
IP range for the soauth docker network, an isolated network for the authentication
|
||||
services, so that the Kratos and Hydra admin APIs are only reachable from the
|
||||
containers placed on it.
|
||||
helpLink: docker
|
||||
readonly: True
|
||||
advanced: True
|
||||
global: True
|
||||
gateway:
|
||||
description: Gateway for the soauth docker network.
|
||||
helpLink: docker
|
||||
readonly: True
|
||||
advanced: True
|
||||
global: True
|
||||
manager_only:
|
||||
description: |
|
||||
Limits the soauth network to grid members running the authentication containers,
|
||||
instead of creating it on every node.
|
||||
helpLink: docker
|
||||
readonly: True
|
||||
advanced: True
|
||||
global: True
|
||||
forcedType: bool
|
||||
ulimits:
|
||||
description: |
|
||||
Default ulimit settings applied to all containers via the Docker daemon. Each entry specifies a resource name (e.g. nofile, memlock, core, nproc) with soft and hard limits. Individual container ulimits override these defaults. Valid resource names include: cpu, fsize, data, stack, core, rss, nproc, nofile, memlock, as, locks, sigpending, msgqueue, nice, rtprio, rttime.
|
||||
@@ -34,6 +68,16 @@ docker:
|
||||
readonly: True
|
||||
advanced: True
|
||||
global: True
|
||||
networks:
|
||||
description: |
|
||||
Docker networks this container is attached to. The first entry is the container's
|
||||
primary network and determines the address its published ports are forwarded to.
|
||||
Defaults to sobridge when unset.
|
||||
helpLink: docker
|
||||
readonly: True
|
||||
advanced: True
|
||||
global: True
|
||||
forcedType: "[]string"
|
||||
port_bindings:
|
||||
description: List of port bindings for the container.
|
||||
helpLink: docker
|
||||
|
||||
@@ -9,7 +9,8 @@
|
||||
prune_images:
|
||||
cmd.run:
|
||||
- name: so-docker-prune
|
||||
- order: last
|
||||
- onlyif: command -v /usr/sbin/so-docker-prune >/dev/null 2>&1
|
||||
- order: 9000
|
||||
|
||||
{% else %}
|
||||
|
||||
|
||||
@@ -33,8 +33,8 @@ elastalert_sbin:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://elastalert/tools/sbin
|
||||
- user: 933
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
|
||||
#elastalert_sbin_jinja:
|
||||
|
||||
@@ -7,9 +7,6 @@
|
||||
{% if sls.split('.')[0] in allowed_states %}
|
||||
{% from 'vars/globals.map.jinja' import GLOBALS %}
|
||||
{% from 'docker/docker.map.jinja' import DOCKERMERGED %}
|
||||
{% from 'docker/macros/docker_endpoint.jinja' import clear_stale_endpoint %}
|
||||
|
||||
{{ clear_stale_endpoint('so-elastalert', 'sobridge', DOCKERMERGED.containers['so-elastalert'].ip) }}
|
||||
|
||||
include:
|
||||
- elastalert.config
|
||||
@@ -22,6 +19,7 @@ wait_for_elasticsearch:
|
||||
so-elastalert:
|
||||
docker_container.running:
|
||||
- image: {{ GLOBALS.registry_host }}:5000/{{ GLOBALS.image_repo }}/so-elastalert:{{ GLOBALS.so_version }}
|
||||
- restart_policy: unless-stopped
|
||||
- hostname: elastalert
|
||||
- name: so-elastalert
|
||||
- user: so-elastalert
|
||||
|
||||
@@ -7,9 +7,6 @@
|
||||
{% if sls.split('.')[0] in allowed_states %}
|
||||
{% from 'vars/globals.map.jinja' import GLOBALS %}
|
||||
{% from 'docker/docker.map.jinja' import DOCKERMERGED %}
|
||||
{% from 'docker/macros/docker_endpoint.jinja' import clear_stale_endpoint %}
|
||||
|
||||
{{ clear_stale_endpoint('so-elastic-fleet-package-registry', 'sobridge', DOCKERMERGED.containers['so-elastic-fleet-package-registry'].ip) }}
|
||||
|
||||
include:
|
||||
- elastic-fleet-package-registry.config
|
||||
@@ -18,6 +15,7 @@ include:
|
||||
so-elastic-fleet-package-registry:
|
||||
docker_container.running:
|
||||
- image: {{ GLOBALS.registry_host }}:5000/{{ GLOBALS.image_repo }}/so-elastic-fleet-package-registry:{{ GLOBALS.so_version }}
|
||||
- restart_policy: unless-stopped
|
||||
- name: so-elastic-fleet-package-registry
|
||||
- hostname: Fleet-package-reg-{{ GLOBALS.hostname }}
|
||||
- detach: True
|
||||
|
||||
@@ -39,8 +39,8 @@ elasticagent_sbin_jinja:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://elasticagent/tools/sbin_jinja
|
||||
- user: 949
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
- template: jinja
|
||||
|
||||
|
||||
@@ -7,9 +7,6 @@
|
||||
{% if sls.split('.')[0] in allowed_states %}
|
||||
{% from 'vars/globals.map.jinja' import GLOBALS %}
|
||||
{% from 'docker/docker.map.jinja' import DOCKERMERGED %}
|
||||
{% from 'docker/macros/docker_endpoint.jinja' import clear_stale_endpoint %}
|
||||
|
||||
{{ clear_stale_endpoint('so-elastic-agent', 'sobridge', DOCKERMERGED.containers['so-elastic-agent'].ip) }}
|
||||
|
||||
include:
|
||||
- ca
|
||||
@@ -19,6 +16,7 @@ include:
|
||||
so-elastic-agent:
|
||||
docker_container.running:
|
||||
- image: {{ GLOBALS.registry_host }}:5000/{{ GLOBALS.image_repo }}/so-elastic-agent:{{ GLOBALS.so_version }}
|
||||
- restart_policy: unless-stopped
|
||||
- name: so-elastic-agent
|
||||
- hostname: {{ GLOBALS.hostname }}
|
||||
- detach: True
|
||||
|
||||
@@ -31,8 +31,8 @@ elasticfleet_sbin:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://elasticfleet/tools/sbin
|
||||
- user: 947
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
- show_changes: False
|
||||
|
||||
@@ -40,8 +40,8 @@ elasticfleet_sbin_jinja:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://elasticfleet/tools/sbin_jinja
|
||||
- user: 947
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
- template: jinja
|
||||
- exclude_pat:
|
||||
@@ -81,8 +81,8 @@ eapackageupgrade:
|
||||
file.managed:
|
||||
- name: /usr/sbin/so-elastic-fleet-package-upgrade
|
||||
- source: salt://elasticfleet/tools/sbin_jinja/so-elastic-fleet-package-upgrade
|
||||
- user: 947
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
- template: jinja
|
||||
|
||||
|
||||
@@ -1,6 +1,5 @@
|
||||
elasticfleet:
|
||||
enabled: False
|
||||
patch_version: 9.3.3+build202604082258 # Elastic Agent specific patch release.
|
||||
enable_manager_output: True
|
||||
config:
|
||||
server:
|
||||
|
||||
@@ -8,14 +8,9 @@
|
||||
{% from 'vars/globals.map.jinja' import GLOBALS %}
|
||||
{% from 'docker/docker.map.jinja' import DOCKERMERGED %}
|
||||
{% from 'elasticfleet/map.jinja' import ELASTICFLEETMERGED %}
|
||||
{% from 'docker/macros/docker_endpoint.jinja' import clear_stale_endpoint %}
|
||||
|
||||
{# This value is generated during node install and stored in minion pillar #}
|
||||
{% set SERVICETOKEN = salt['pillar.get']('elasticfleet:config:server:es_token','') %}
|
||||
{# Prevent Elastic Agent from re-enrolling with a new agent.id everytime the container starts up.
|
||||
- if a fresh enrollment is needed use 'so-stop elasticfleet'
|
||||
#}
|
||||
{% set ENROLLED = salt['file.file_exists']('/opt/so/conf/elastic-fleet/state/fleet.enc') %}
|
||||
|
||||
include:
|
||||
- ca
|
||||
@@ -44,11 +39,10 @@ elasticagent_syncartifacts:
|
||||
{% endif %}
|
||||
|
||||
{% if SERVICETOKEN != '' %}
|
||||
{{ clear_stale_endpoint('so-elastic-fleet', 'sobridge', DOCKERMERGED.containers['so-elastic-fleet'].ip) }}
|
||||
|
||||
so-elastic-fleet:
|
||||
docker_container.running:
|
||||
- image: {{ GLOBALS.registry_host }}:5000/{{ GLOBALS.image_repo }}/so-elastic-agent:{{ GLOBALS.so_version }}
|
||||
- restart_policy: unless-stopped
|
||||
- name: so-elastic-fleet
|
||||
- hostname: FleetServer-{{ GLOBALS.hostname }}
|
||||
- detach: True
|
||||
@@ -72,7 +66,6 @@ so-elastic-fleet:
|
||||
- /etc/pki/elasticfleet-server.crt:/etc/pki/elasticfleet-server.crt:ro
|
||||
- /etc/pki/elasticfleet-server.key:/etc/pki/elasticfleet-server.key:ro
|
||||
- /etc/pki/tls/certs/intca.crt:/etc/pki/tls/certs/intca.crt:ro
|
||||
- /opt/so/conf/elastic-fleet/state:/usr/share/elastic-agent/state
|
||||
- /opt/so/log/elasticfleet:/usr/share/elastic-agent/logs
|
||||
{% if DOCKERMERGED.containers['so-elastic-fleet'].custom_bind_mounts %}
|
||||
{% for BIND in DOCKERMERGED.containers['so-elastic-fleet'].custom_bind_mounts %}
|
||||
@@ -80,7 +73,6 @@ so-elastic-fleet:
|
||||
{% endfor %}
|
||||
{% endif %}
|
||||
- environment:
|
||||
{% if not ENROLLED %}
|
||||
- FLEET_SERVER_ENABLE=true
|
||||
- FLEET_URL=https://{{ GLOBALS.hostname }}:8220
|
||||
- FLEET_SERVER_ELASTICSEARCH_HOST=https://{{ GLOBALS.manager }}:9200
|
||||
@@ -90,9 +82,6 @@ so-elastic-fleet:
|
||||
- FLEET_SERVER_CERT_KEY=/etc/pki/elasticfleet-server.key
|
||||
- FLEET_CA=/etc/pki/tls/certs/intca.crt
|
||||
- FLEET_SERVER_ELASTICSEARCH_CA=/etc/pki/tls/certs/intca.crt
|
||||
{% endif %}
|
||||
- STATE_PATH=/usr/share/elastic-agent/state
|
||||
- CONFIG_PATH=/usr/share/elastic-agent/state
|
||||
- LOGS_PATH=logs
|
||||
{% if DOCKERMERGED.containers['so-elastic-fleet'].extra_env %}
|
||||
{% for XTRAENV in DOCKERMERGED.containers['so-elastic-fleet'].extra_env %}
|
||||
@@ -111,7 +100,6 @@ so-elastic-fleet:
|
||||
- x509: etc_elasticfleet_crt
|
||||
- require:
|
||||
- file: trusttheca
|
||||
- file: eastatedir
|
||||
- x509: etc_elasticfleet_key
|
||||
- x509: etc_elasticfleet_crt
|
||||
|
||||
|
||||
@@ -5,7 +5,7 @@
|
||||
"package": {
|
||||
"name": "endpoint",
|
||||
"title": "Elastic Defend",
|
||||
"version": "9.3.0",
|
||||
"version": "9.4.1",
|
||||
"requires_root": true
|
||||
},
|
||||
"enabled": true,
|
||||
|
||||
@@ -29,7 +29,7 @@
|
||||
"\\.gz$"
|
||||
],
|
||||
"include_files": [],
|
||||
"processors": "- dissect:\n tokenizer: \"/nsm/import/%{import.id}/evtx/%{import.file}\"\n field: \"log.file.path\"\n target_prefix: \"\"\n- decode_json_fields:\n fields: [\"message\"]\n target: \"\"\n- drop_fields:\n fields: [\"host\"]\n ignore_missing: true\n- add_fields:\n target: data_stream\n fields:\n type: logs\n dataset: system.security\n- add_fields:\n target: event\n fields:\n dataset: system.security\n module: system\n imported: true\n- add_fields:\n target: \"@metadata\"\n fields:\n pipeline: logs-system.security-2.15.0\n- if:\n equals:\n winlog.channel: 'Microsoft-Windows-Sysmon/Operational'\n then: \n - add_fields:\n target: data_stream\n fields:\n dataset: windows.sysmon_operational\n - add_fields:\n target: event\n fields:\n dataset: windows.sysmon_operational\n module: windows\n imported: true\n - add_fields:\n target: \"@metadata\"\n fields:\n pipeline: logs-windows.sysmon_operational-3.8.0\n- if:\n equals:\n winlog.channel: 'Application'\n then: \n - add_fields:\n target: data_stream\n fields:\n dataset: system.application\n - add_fields:\n target: event\n fields:\n dataset: system.application\n - add_fields:\n target: \"@metadata\"\n fields:\n pipeline: logs-system.application-2.15.0\n- if:\n equals:\n winlog.channel: 'System'\n then: \n - add_fields:\n target: data_stream\n fields:\n dataset: system.system\n - add_fields:\n target: event\n fields:\n dataset: system.system\n - add_fields:\n target: \"@metadata\"\n fields:\n pipeline: logs-system.system-2.15.0\n \n- if:\n equals:\n winlog.channel: 'Microsoft-Windows-PowerShell/Operational'\n then: \n - add_fields:\n target: data_stream\n fields:\n dataset: windows.powershell_operational\n - add_fields:\n target: event\n fields:\n dataset: windows.powershell_operational\n module: windows\n - add_fields:\n target: \"@metadata\"\n fields:\n pipeline: logs-windows.powershell_operational-3.8.0\n- add_fields:\n target: data_stream\n fields:\n dataset: import",
|
||||
"processors": "- dissect:\n tokenizer: \"/nsm/import/%{import.id}/evtx/%{import.file}\"\n field: \"log.file.path\"\n target_prefix: \"\"\n- decode_json_fields:\n fields: [\"message\"]\n target: \"\"\n- drop_fields:\n fields: [\"host\"]\n ignore_missing: true\n- add_fields:\n target: data_stream\n fields:\n type: logs\n dataset: system.security\n- add_fields:\n target: event\n fields:\n dataset: system.security\n module: system\n imported: true\n- add_fields:\n target: \"@metadata\"\n fields:\n pipeline: logs-system.security-2.22.3\n- if:\n equals:\n winlog.channel: 'Microsoft-Windows-Sysmon/Operational'\n then: \n - add_fields:\n target: data_stream\n fields:\n dataset: windows.sysmon_operational\n - add_fields:\n target: event\n fields:\n dataset: windows.sysmon_operational\n module: windows\n imported: true\n - add_fields:\n target: \"@metadata\"\n fields:\n pipeline: logs-windows.sysmon_operational-3.9.0\n- if:\n equals:\n winlog.channel: 'Application'\n then: \n - add_fields:\n target: data_stream\n fields:\n dataset: system.application\n - add_fields:\n target: event\n fields:\n dataset: system.application\n - add_fields:\n target: \"@metadata\"\n fields:\n pipeline: logs-system.application-2.22.3\n- if:\n equals:\n winlog.channel: 'System'\n then: \n - add_fields:\n target: data_stream\n fields:\n dataset: system.system\n - add_fields:\n target: event\n fields:\n dataset: system.system\n - add_fields:\n target: \"@metadata\"\n fields:\n pipeline: logs-system.system-2.22.3\n \n- if:\n equals:\n winlog.channel: 'Microsoft-Windows-PowerShell/Operational'\n then: \n - add_fields:\n target: data_stream\n fields:\n dataset: windows.powershell_operational\n - add_fields:\n target: event\n fields:\n dataset: windows.powershell_operational\n module: windows\n - add_fields:\n target: \"@metadata\"\n fields:\n pipeline: logs-windows.powershell_operational-3.9.0\n- add_fields:\n target: data_stream\n fields:\n dataset: import",
|
||||
"tags": [
|
||||
"import"
|
||||
],
|
||||
|
||||
@@ -14,8 +14,8 @@ so-elastic-agent-install:
|
||||
file.managed:
|
||||
- name: /usr/sbin/so-elastic-agent-install
|
||||
- source: salt://elasticfleet/tools/sbin/so-elastic-agent-install
|
||||
- user: 947
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
- show_changes: False
|
||||
|
||||
|
||||
@@ -16,7 +16,6 @@
|
||||
'awsfirehose.metrics': 'aws.cloudwatch',
|
||||
'cribl.logs': 'cribl',
|
||||
'cribl.metrics': 'cribl',
|
||||
'sentinel_one_cloud_funnel.logins': 'sentinel_one_cloud_funnel.login',
|
||||
'azure_application_insights.app_insights': 'azure.app_insights',
|
||||
'azure_application_insights.app_state': 'azure.app_state',
|
||||
'azure_billing.billing': 'azure.billing',
|
||||
|
||||
@@ -68,6 +68,24 @@ so-elastic-fleet-package-upgrade:
|
||||
- require:
|
||||
- http: wait_for_so-kibana
|
||||
|
||||
# initial so-elasticsearch-templates run is earlier, but it can skip over templates that have component templates not yet installed to avoid elasticsearch rejecting the template.
|
||||
so-elasticsearch-templates-after-fleet-packages:
|
||||
cmd.run:
|
||||
- name: /usr/sbin/so-elasticsearch-templates-load
|
||||
- cwd: /opt/so
|
||||
- unless: test -f /opt/so/state/estemplates.txt
|
||||
- require:
|
||||
- cmd: so-elastic-fleet-package-upgrade
|
||||
|
||||
so-elastic-fleet-integration-upgrade:
|
||||
cmd.run:
|
||||
- name: /usr/sbin/so-elastic-fleet-integration-upgrade
|
||||
- retry:
|
||||
attempts: 3
|
||||
interval: 10
|
||||
- require:
|
||||
- cmd: so-elastic-fleet-package-upgrade
|
||||
|
||||
so-elastic-fleet-integrations:
|
||||
cmd.run:
|
||||
- name: /usr/sbin/so-elastic-fleet-integration-policy-load
|
||||
@@ -86,21 +104,13 @@ so-elastic-agent-grid-upgrade:
|
||||
- require:
|
||||
- http: wait_for_so-kibana
|
||||
|
||||
so-elastic-fleet-integration-upgrade:
|
||||
cmd.run:
|
||||
- name: /usr/sbin/so-elastic-fleet-integration-upgrade
|
||||
- retry:
|
||||
attempts: 3
|
||||
interval: 10
|
||||
- require:
|
||||
- http: wait_for_so-kibana
|
||||
|
||||
{# Optional integrations script doesn't need the retries like so-elastic-fleet-integration-upgrade which loads the default integrations #}
|
||||
so-elastic-fleet-addon-integrations:
|
||||
cmd.run:
|
||||
- name: /usr/sbin/so-elastic-fleet-optional-integrations-load
|
||||
- require:
|
||||
- http: wait_for_so-kibana
|
||||
- cmd: so-elasticsearch-templates-after-fleet-packages
|
||||
|
||||
{% if ELASTICFLEETMERGED.config.defend_filters.enable_auto_configuration %}
|
||||
so-elastic-defend-manage-filters-file-watch:
|
||||
|
||||
@@ -30,6 +30,56 @@ fleet_api() {
|
||||
curl -sK /opt/so/conf/elasticsearch/curl.config -L "localhost:5601/api/fleet/${QUERYPATH}" "$@" --retry 3 --retry-delay 10 --fail 2>/dev/null
|
||||
}
|
||||
|
||||
elastic_fleet_require_agent_policy() {
|
||||
local AGENT_POLICY=$1
|
||||
local POLICY_JSON
|
||||
|
||||
if ! POLICY_JSON=$(fleet_api "agent_policies/$AGENT_POLICY") || [ -z "$POLICY_JSON" ]; then
|
||||
echo "Error: Agent policy '$AGENT_POLICY' was not found or is not visible in the current Kibana space." >&2
|
||||
return 1
|
||||
fi
|
||||
|
||||
if ! jq -e '.item.package_policies | type == "array"' <<<"$POLICY_JSON" >/dev/null 2>&1; then
|
||||
echo "Error: Agent policy '$AGENT_POLICY' was not found or is not visible in the current Kibana space." >&2
|
||||
return 1
|
||||
fi
|
||||
|
||||
echo "$POLICY_JSON"
|
||||
}
|
||||
|
||||
# Print the single active enrollment token for POLICY_ID.
|
||||
# Exit 1: retryable (API failure, invalid response, no active token)
|
||||
# Exit 2: multiple active tokens - Shouldn't get into this state without manual intervention
|
||||
elastic_fleet_active_enrollment_token() {
|
||||
local POLICY_ID=$1
|
||||
local RESP TOKEN_COUNT API_KEY
|
||||
|
||||
if ! RESP=$(fleet_api "enrollment_api_keys?perPage=100" -H 'kbn-xsrf: true' -H 'Content-Type: application/json'); then
|
||||
echo "Error: Failed to retrieve enrollment tokens for agent policy '$POLICY_ID'." >&2
|
||||
return 1
|
||||
fi
|
||||
|
||||
if ! jq -e '.list' <<<"$RESP" >/dev/null 2>&1; then
|
||||
echo "Error: Invalid enrollment token response for agent policy '$POLICY_ID'." >&2
|
||||
return 1
|
||||
fi
|
||||
|
||||
TOKEN_COUNT=$(jq --arg pid "$POLICY_ID" '[.list[] | select(.policy_id == $pid and .active == true)] | length' <<<"$RESP")
|
||||
|
||||
if [ "${TOKEN_COUNT:-0}" -eq 0 ]; then
|
||||
echo "Error: No active enrollment token found for agent policy '$POLICY_ID'." >&2
|
||||
return 1
|
||||
fi
|
||||
|
||||
if [ "$TOKEN_COUNT" -gt 1 ]; then
|
||||
echo "Error: Found $TOKEN_COUNT active enrollment tokens for agent policy '$POLICY_ID'; expected exactly one." >&2
|
||||
return 2
|
||||
fi
|
||||
|
||||
API_KEY=$(jq -r --arg pid "$POLICY_ID" '.list[] | select(.policy_id == $pid and .active == true) | .api_key' <<<"$RESP")
|
||||
echo "$API_KEY"
|
||||
}
|
||||
|
||||
# Max number of concurrent Fleet write jobs (create/update). Override via env if needed.
|
||||
MAX_FLEET_JOBS=${MAX_FLEET_JOBS:-10}
|
||||
|
||||
@@ -62,15 +112,7 @@ elastic_fleet_load_integrations_dir() {
|
||||
i=0
|
||||
|
||||
# Fetch the agent policy a single time; we look up integration ids locally below.
|
||||
if ! POLICY_JSON=$(fleet_api "agent_policies/$AGENT_POLICY"); then
|
||||
echo "Error: Failed to retrieve agent policy '$AGENT_POLICY'."
|
||||
rm -f "$FAIL_FILE"
|
||||
rm -rf "$OUT_DIR"
|
||||
return 1
|
||||
fi
|
||||
|
||||
if ! jq -e '.item.package_policies' <<<"$POLICY_JSON" >/dev/null 2>&1; then
|
||||
echo "Error: Invalid agent policy response for '$AGENT_POLICY'."
|
||||
if ! POLICY_JSON=$(elastic_fleet_require_agent_policy "$AGENT_POLICY"); then
|
||||
rm -f "$FAIL_FILE"
|
||||
rm -rf "$OUT_DIR"
|
||||
return 1
|
||||
@@ -124,9 +166,15 @@ elastic_fleet_integration_check() {
|
||||
|
||||
JSON_STRING=$2
|
||||
|
||||
NAME=$(jq -r .name $JSON_STRING)
|
||||
NAME=$(jq -r .name "$JSON_STRING")
|
||||
INTEGRATION_ID=""
|
||||
|
||||
INTEGRATION_ID=$(/usr/sbin/so-elastic-fleet-agent-policy-view "$AGENT_POLICY" | jq -r '.item.package_policies[] | select(.name=="'"$NAME"'") | .id')
|
||||
local POLICY_JSON
|
||||
if ! POLICY_JSON=$(elastic_fleet_require_agent_policy "$AGENT_POLICY"); then
|
||||
return 1
|
||||
fi
|
||||
|
||||
INTEGRATION_ID=$(jq -r --arg n "$NAME" '.item.package_policies[]? | select(.name==$n) | .id' <<<"$POLICY_JSON")
|
||||
|
||||
}
|
||||
|
||||
@@ -148,7 +196,16 @@ elastic_fleet_integration_remove() {
|
||||
|
||||
NAME=$2
|
||||
|
||||
INTEGRATION_ID=$(/usr/sbin/so-elastic-fleet-agent-policy-view "$AGENT_POLICY" | jq -r '.item.package_policies[] | select(.name=="'"$NAME"'") | .id')
|
||||
local POLICY_JSON
|
||||
if ! POLICY_JSON=$(elastic_fleet_require_agent_policy "$AGENT_POLICY"); then
|
||||
return 1
|
||||
fi
|
||||
|
||||
INTEGRATION_ID=$(jq -r --arg n "$NAME" '.item.package_policies[]? | select(.name==$n) | .id' <<<"$POLICY_JSON")
|
||||
if [ -z "$INTEGRATION_ID" ]; then
|
||||
echo "Error: Integration '$NAME' was not found in agent policy '$AGENT_POLICY'." >&2
|
||||
return 1
|
||||
fi
|
||||
|
||||
JSON_STRING=$( jq -n \
|
||||
--arg INTEGRATIONID "$INTEGRATION_ID" \
|
||||
|
||||
@@ -13,7 +13,10 @@ ERROR=false
|
||||
for INTEGRATION in /opt/so/conf/elastic-fleet/integrations/elastic-defend/*.json
|
||||
do
|
||||
printf "\n\nInitial Endpoints Policy - Loading $INTEGRATION\n"
|
||||
elastic_fleet_integration_check "endpoints-initial" "$INTEGRATION"
|
||||
if ! elastic_fleet_integration_check "endpoints-initial" "$INTEGRATION"; then
|
||||
ERROR=true
|
||||
continue
|
||||
fi
|
||||
if [ -n "$INTEGRATION_ID" ]; then
|
||||
printf "\n\nIntegration $NAME exists - Upgrading integration policy\n"
|
||||
if ! elastic_fleet_integration_policy_upgrade "$INTEGRATION_ID"; then
|
||||
|
||||
+20
-5
@@ -7,20 +7,35 @@
|
||||
. /usr/sbin/so-elastic-fleet-common
|
||||
|
||||
# Get all the fleet policies
|
||||
json_output=$(curl -s -K /opt/so/conf/elasticsearch/curl.config -L -X GET "localhost:5601/api/fleet/agent_policies" -H 'kbn-xsrf: true')
|
||||
if ! json_output=$(fleet_api "agent_policies" -H 'kbn-xsrf: true'); then
|
||||
echo "Error: Failed to retrieve Fleet agent policies." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if ! jq -e '.items' <<<"$json_output" >/dev/null 2>&1; then
|
||||
echo "Error: Invalid Fleet agent policies response." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Extract the IDs that start with "FleetServer_"
|
||||
POLICY=$(echo "$json_output" | jq -r '.items[] | select(.id | startswith("FleetServer_")) | .id')
|
||||
POLICY=$(jq -r '.items[] | select(.id | startswith("FleetServer_")) | .id' <<<"$json_output")
|
||||
|
||||
# Iterate over each ID in the POLICY variable
|
||||
for POLICYNAME in $POLICY; do
|
||||
printf "\nUpdating Policy: $POLICYNAME\n"
|
||||
|
||||
# First get the Integration ID
|
||||
INTEGRATION_ID=$(/usr/sbin/so-elastic-fleet-agent-policy-view "$POLICYNAME" | jq -r '.item.package_policies[] | select(.package.name == "fleet_server") | .id')
|
||||
if ! POLICY_JSON=$(elastic_fleet_require_agent_policy "$POLICYNAME"); then
|
||||
exit 1
|
||||
fi
|
||||
|
||||
INTEGRATION_ID=$(jq -r '.item.package_policies[]? | select(.package.name == "fleet_server") | .id' <<<"$POLICY_JSON")
|
||||
if [ -z "$INTEGRATION_ID" ]; then
|
||||
echo "Error: fleet_server integration was not found in agent policy '$POLICYNAME'." >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Modify the default integration policy to update the policy_id and an with the correct naming
|
||||
UPDATED_INTEGRATION_POLICY=$(jq --arg policy_id "$POLICYNAME" --arg name "fleet_server-$POLICYNAME" '
|
||||
UPDATED_INTEGRATION_POLICY=$(jq --arg policy_id "$POLICYNAME" --arg name "fleet_server-$POLICYNAME" '
|
||||
.policy_id = $policy_id |
|
||||
.name = $name' /opt/so/conf/elastic-fleet/integrations/fleet-server/fleet-server.json)
|
||||
|
||||
|
||||
@@ -22,12 +22,19 @@ NUM_RUNNING=$(pgrep -cf "/bin/bash /sbin/so-elastic-agent-gen-installers")
|
||||
|
||||
for i in {1..30}
|
||||
do
|
||||
ENROLLMENTOKEN=$(curl -K /opt/so/conf/elasticsearch/curl.config -L "localhost:5601/api/fleet/enrollment_api_keys?perPage=100" -H 'kbn-xsrf: true' -H 'Content-Type: application/json' | jq .list | jq -r -c '.[] | select(.policy_id | contains("endpoints-initial")) | .api_key')
|
||||
ENROLLMENTOKEN=$(elastic_fleet_active_enrollment_token "endpoints-initial")
|
||||
TOKEN_RC=$?
|
||||
if [ "$TOKEN_RC" -eq 2 ]; then
|
||||
exit 1
|
||||
fi
|
||||
FLEETHOST=$(curl -K /opt/so/conf/elasticsearch/curl.config 'http://localhost:5601/api/fleet/fleet_server_hosts/grid-default' | jq -r '.item.host_urls[]' | paste -sd ',')
|
||||
if [[ $FLEETHOST ]] && [[ $ENROLLMENTOKEN ]]; then break; else sleep 10; fi
|
||||
if [[ -n "$FLEETHOST" ]] && [[ -n "$ENROLLMENTOKEN" ]]; then
|
||||
break
|
||||
fi
|
||||
sleep 10
|
||||
done
|
||||
|
||||
if [[ -z $FLEETHOST ]] || [[ -z $ENROLLMENTOKEN ]]; then
|
||||
if [[ -z "$FLEETHOST" ]] || [[ -z "$ENROLLMENTOKEN" ]]; then
|
||||
printf "\nFleet Host URL, Enrollment Token or Elastic Version empty - exiting..."
|
||||
printf "\nFleet Host: $FLEETHOST, Enrollment Token: $ENROLLMENTOKEN\n"
|
||||
exit 1
|
||||
@@ -67,19 +74,25 @@ for GOOS in "${GOTARGETOS[@]}"; do
|
||||
GOARCH="amd64"
|
||||
if [[ $GOOS == 'darwin/arm64' ]]; then GOOS="darwin" && GOARCH="arm64"; fi
|
||||
printf "\n\n### Generating $GOOS/$GOARCH Installer...\n"
|
||||
docker run -e CGO_ENABLED=0 -e GOOS=$GOOS -e GOARCH=$GOARCH \
|
||||
if ! docker run -e CGO_ENABLED=0 -e GOOS=$GOOS -e GOARCH=$GOARCH \
|
||||
--mount type=bind,source=/etc/pki/tls/certs/,target=/workspace/files/cert/ \
|
||||
--mount type=bind,source=/nsm/elastic-agent-workspace/,target=/workspace/files/elastic-agent/ \
|
||||
--mount type=bind,source=/opt/so/saltstack/local/salt/elasticfleet/files/,target=/output/ \
|
||||
{{ GLOBALS.registry_host }}:5000/{{ GLOBALS.image_repo }}/so-elastic-agent-builder:{{ GLOBALS.so_version }} go build -ldflags "-X main.fleetHostURLsList=$FLEETHOST -X main.enrollmentToken=$ENROLLMENTOKEN" -o /output/so-elastic-agent_${GOOS}_${GOARCH}
|
||||
{{ GLOBALS.registry_host }}:5000/{{ GLOBALS.image_repo }}/so-elastic-agent-builder:{{ GLOBALS.so_version }} go build -ldflags "-X main.fleetHostURLsList=$FLEETHOST -X main.enrollmentToken=$ENROLLMENTOKEN" -o /output/so-elastic-agent_${GOOS}_${GOARCH}; then
|
||||
printf "\n### ERROR: Failed to generate $GOOS/$GOARCH installer. Exiting...\n"
|
||||
exit 1
|
||||
fi
|
||||
printf "\n### $GOOS/$GOARCH Installer Generated...\n"
|
||||
done
|
||||
|
||||
printf "\n\n### Generating MSI...\n"
|
||||
cp /opt/so/saltstack/local/salt/elasticfleet/files/so-elastic-agent_windows_amd64 /opt/so/saltstack/local/salt/elasticfleet/files/so-elastic-agent_windows_amd64.exe
|
||||
docker run \
|
||||
if ! docker run \
|
||||
--mount type=bind,source=/opt/so/saltstack/local/salt/elasticfleet/files/,target=/output/ -w /output \
|
||||
{{ GLOBALS.registry_host }}:5000/{{ GLOBALS.image_repo }}/so-elastic-agent-builder:{{ GLOBALS.so_version }} wixl -o so-elastic-agent_windows_amd64_msi --arch x64 /workspace/so-elastic-agent.wxs
|
||||
{{ GLOBALS.registry_host }}:5000/{{ GLOBALS.image_repo }}/so-elastic-agent-builder:{{ GLOBALS.so_version }} wixl -o so-elastic-agent_windows_amd64_msi --arch x64 /workspace/so-elastic-agent.wxs; then
|
||||
printf "\n### ERROR: Failed to generate MSI. Exiting...\n"
|
||||
exit 1
|
||||
fi
|
||||
printf "\n### MSI Generated...\n"
|
||||
|
||||
# Verify installers were created
|
||||
|
||||
@@ -10,6 +10,25 @@
|
||||
|
||||
PKG_LOAD_FAILURES=0
|
||||
PKG_LOAD_FAILURES_NAMES=()
|
||||
PKG_UPGRADED=0
|
||||
|
||||
cleanup_elasticsearch_fleet_transforms() {
|
||||
local transforms transform_id attempt
|
||||
|
||||
if ! transforms=$(so-elasticsearch-query "_transform/logs-elasticsearch.index_pivot-default-*" --retry 1 --retry-delay 5); then
|
||||
return 0
|
||||
fi
|
||||
|
||||
while IFS= read -r transform_id; do
|
||||
[ -n "$transform_id" ] || continue
|
||||
for attempt in {1..3}; do
|
||||
if so-elasticsearch-query "_transform/$transform_id?force=true" -XDELETE --fail --retry 1 --retry-delay 5; then
|
||||
break
|
||||
fi
|
||||
sleep 5
|
||||
done
|
||||
done < <(jq -r '.transforms[]?.id' <<< "$transforms")
|
||||
}
|
||||
|
||||
{%- for PACKAGE in SUPPORTED_PACKAGES %}
|
||||
if INSTALLED_VERSION=$(elastic_fleet_package_version_check "{{ PACKAGE }}") && LATEST_VERSION=$(elastic_fleet_package_latest_version_check "{{ PACKAGE }}"); then
|
||||
@@ -17,10 +36,25 @@ if INSTALLED_VERSION=$(elastic_fleet_package_version_check "{{ PACKAGE }}") && L
|
||||
if [ "$INSTALLED_VERSION" == "$LATEST_VERSION" ]; then
|
||||
echo "{{ PACKAGE }} integration version $INSTALLED_VERSION is already at the reported latest version $LATEST_VERSION, skipping upgrade."
|
||||
else
|
||||
echo "Upgrading {{ PACKAGE }} package to version $LATEST_VERSION..."
|
||||
{%- if PACKAGE == 'elasticsearch' %}
|
||||
cleanup_elasticsearch_fleet_transforms
|
||||
{%- endif %}
|
||||
echo "Upgrading {{ PACKAGE }} package from $INSTALLED_VERSION to version $LATEST_VERSION..."
|
||||
if ! elastic_fleet_package_install "{{ PACKAGE }}" "$LATEST_VERSION"; then
|
||||
PKG_LOAD_FAILURES=$((PKG_LOAD_FAILURES + 1))
|
||||
PKG_LOAD_FAILURES_NAMES+=("{{ PACKAGE }}")
|
||||
# check that package has upgraded to the expected version after install command
|
||||
elif ! LATEST_VERSION=$(elastic_fleet_package_latest_version_check "{{ PACKAGE }}"); then
|
||||
echo "ERROR: Failed to get latest version information for integration {{ PACKAGE }} after upgrade attempt"
|
||||
PKG_LOAD_FAILURES=$((PKG_LOAD_FAILURES + 1))
|
||||
PKG_LOAD_FAILURES_NAMES+=("{{ PACKAGE }}")
|
||||
elif INSTALLED_VERSION=$(elastic_fleet_package_version_check "{{ PACKAGE }}") && [ "$INSTALLED_VERSION" == "$LATEST_VERSION" ]; then
|
||||
echo "{{ PACKAGE }} integration upgraded to version $LATEST_VERSION."
|
||||
PKG_UPGRADED=$((PKG_UPGRADED + 1))
|
||||
else
|
||||
echo "ERROR: {{ PACKAGE }} integration still at ${INSTALLED_VERSION:-unknown}; expected $LATEST_VERSION"
|
||||
PKG_LOAD_FAILURES=$((PKG_LOAD_FAILURES + 1))
|
||||
PKG_LOAD_FAILURES_NAMES+=("{{ PACKAGE }}")
|
||||
fi
|
||||
fi
|
||||
else
|
||||
@@ -30,6 +64,11 @@ else
|
||||
fi
|
||||
{%- endfor %}
|
||||
|
||||
if [ $PKG_UPGRADED -gt 0 ]; then
|
||||
echo "Elasticsearch template statefiles cleared after $PKG_UPGRADED package upgrade(s), so templates can reload."
|
||||
rm -f /opt/so/state/estemplates.txt /opt/so/state/addon_estemplates.txt
|
||||
fi
|
||||
|
||||
if [ $PKG_LOAD_FAILURES -gt 0 ]; then
|
||||
echo "ERROR: Failed to upgrade $PKG_LOAD_FAILURES package(s):"
|
||||
for PKG in "${PKG_LOAD_FAILURES_NAMES[@]}"; do
|
||||
|
||||
@@ -202,26 +202,9 @@ fi
|
||||
### Finalization ###
|
||||
|
||||
# Query for Enrollment Tokens for default policies
|
||||
if ENDPOINTSENROLLMENTOKEN_RAW=$(fleet_api "enrollment_api_keys" -H 'kbn-xsrf: true' -H 'Content-Type: application/json'); then
|
||||
ENDPOINTSENROLLMENTOKEN=$(echo "$ENDPOINTSENROLLMENTOKEN_RAW" | jq .list | jq -r -c '.[] | select(.policy_id | contains("endpoints-initial")) | .api_key')
|
||||
else
|
||||
echo -e "\nFailed to query for Endpoints enrollment token"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if GRIDNODESENROLLMENTOKENGENERAL_RAW=$(fleet_api "enrollment_api_keys" -H 'kbn-xsrf: true' -H 'Content-Type: application/json'); then
|
||||
GRIDNODESENROLLMENTOKENGENERAL=$(echo "$GRIDNODESENROLLMENTOKENGENERAL_RAW" | jq .list | jq -r -c '.[] | select(.policy_id | contains("so-grid-nodes_general")) | .api_key')
|
||||
else
|
||||
echo -e "\nFailed to query for Grid nodes - General enrollment token"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
if GRIDNODESENROLLMENTOKENHEAVY_RAW=$(fleet_api "enrollment_api_keys" -H 'kbn-xsrf: true' -H 'Content-Type: application/json'); then
|
||||
GRIDNODESENROLLMENTOKENHEAVY=$(echo "$GRIDNODESENROLLMENTOKENHEAVY_RAW" | jq .list | jq -r -c '.[] | select(.policy_id | contains("so-grid-nodes_heavy")) | .api_key')
|
||||
else
|
||||
echo -e "\nFailed to query for Grid nodes - Heavy enrollment token"
|
||||
exit 1
|
||||
fi
|
||||
ENDPOINTSENROLLMENTOKEN=$(elastic_fleet_active_enrollment_token "endpoints-initial") || exit 1
|
||||
GRIDNODESENROLLMENTOKENGENERAL=$(elastic_fleet_active_enrollment_token "so-grid-nodes_general") || exit 1
|
||||
GRIDNODESENROLLMENTOKENHEAVY=$(elastic_fleet_active_enrollment_token "so-grid-nodes_heavy") || exit 1
|
||||
|
||||
# Store needed data in minion pillar
|
||||
pillar_file=/opt/so/saltstack/local/pillar/minions/{{ GLOBALS.minion_id }}.sls
|
||||
|
||||
@@ -98,6 +98,13 @@ so-es-cluster-settings:
|
||||
- docker_container: so-elasticsearch
|
||||
- file: elasticsearch_sbin_jinja
|
||||
- http: wait_for_so-elasticsearch
|
||||
|
||||
so-elasticsearch-system-indices-patch:
|
||||
cmd.run:
|
||||
- name: /usr/sbin/so-elasticsearch-system-indices-patch
|
||||
- require:
|
||||
- http: wait_for_so-elasticsearch
|
||||
- file: so-elasticsearch-system-indices-patch-script
|
||||
{% endif %}
|
||||
|
||||
# heavynodes will only load ILM policies for SO managed indices. (Indicies defined in elasticsearch/defaults.yaml)
|
||||
|
||||
@@ -37,19 +37,29 @@ elasticsearch_sbin:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://elasticsearch/tools/sbin
|
||||
- user: 930
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
- exclude_pat:
|
||||
- so-elasticsearch-pipelines # exclude this because we need to watch it for changes, we sync it in another state
|
||||
- so-elasticsearch-system-indices-patch
|
||||
- show_changes: False
|
||||
|
||||
so-elasticsearch-system-indices-patch-script:
|
||||
file.managed:
|
||||
- name: /usr/sbin/so-elasticsearch-system-indices-patch
|
||||
- source: salt://elasticsearch/tools/sbin/so-elasticsearch-system-indices-patch
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
- show_changes: False
|
||||
|
||||
elasticsearch_sbin_jinja:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://elasticsearch/tools/sbin_jinja
|
||||
- user: 939
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
- template: jinja
|
||||
- exclude_pat:
|
||||
@@ -62,8 +72,8 @@ so-elasticsearch-ilm-policy-load-script:
|
||||
file.managed:
|
||||
- name: /usr/sbin/so-elasticsearch-ilm-policy-load
|
||||
- source: salt://elasticsearch/tools/sbin_jinja/so-elasticsearch-ilm-policy-load
|
||||
- user: 930
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 754
|
||||
- template: jinja
|
||||
- defaults:
|
||||
@@ -74,8 +84,8 @@ so-elasticsearch-pipelines-script:
|
||||
file.managed:
|
||||
- name: /usr/sbin/so-elasticsearch-pipelines
|
||||
- source: salt://elasticsearch/tools/sbin/so-elasticsearch-pipelines
|
||||
- user: 930
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 754
|
||||
- show_changes: False
|
||||
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
elasticsearch:
|
||||
enabled: false
|
||||
version: 9.3.3
|
||||
esheap: '600m'
|
||||
version: 9.4.5
|
||||
index_clean: true
|
||||
data_retention_method: DLM
|
||||
vm:
|
||||
@@ -3453,6 +3454,720 @@ elasticsearch:
|
||||
set_priority:
|
||||
priority: 50
|
||||
min_age: 30d
|
||||
so-metrics-system_x_core:
|
||||
data_stream_lifecycle:
|
||||
data_retention: 90d
|
||||
index_template:
|
||||
composed_of:
|
||||
- metrics@tsdb-settings
|
||||
- metrics-system.core@package
|
||||
- metrics@custom
|
||||
- system@custom
|
||||
- metrics-system.core@custom
|
||||
- ecs@mappings
|
||||
- so-fleet_integrations.ip_mappings-1
|
||||
- so-fleet_globals-1
|
||||
- so-fleet_agent_id_verification-1
|
||||
data_stream:
|
||||
allow_custom_routing: false
|
||||
hidden: false
|
||||
ignore_missing_component_templates:
|
||||
- metrics@custom
|
||||
- system@custom
|
||||
- metrics-system.core@custom
|
||||
index_patterns:
|
||||
- metrics-system.core-*
|
||||
priority: 501
|
||||
template:
|
||||
settings:
|
||||
index:
|
||||
lifecycle:
|
||||
name: so-metrics-system.core-logs
|
||||
mode: time_series
|
||||
number_of_replicas: 0
|
||||
policy:
|
||||
phases:
|
||||
cold:
|
||||
actions:
|
||||
set_priority:
|
||||
priority: 0
|
||||
min_age: 60d
|
||||
delete:
|
||||
actions:
|
||||
delete: {}
|
||||
min_age: 365d
|
||||
hot:
|
||||
actions:
|
||||
rollover:
|
||||
max_age: 30d
|
||||
max_primary_shard_size: 50gb
|
||||
set_priority:
|
||||
priority: 100
|
||||
min_age: 0ms
|
||||
warm:
|
||||
actions:
|
||||
set_priority:
|
||||
priority: 50
|
||||
min_age: 30d
|
||||
so-metrics-system_x_cpu:
|
||||
data_stream_lifecycle:
|
||||
data_retention: 90d
|
||||
index_template:
|
||||
composed_of:
|
||||
- metrics@tsdb-settings
|
||||
- metrics-system.cpu@package
|
||||
- metrics@custom
|
||||
- system@custom
|
||||
- metrics-system.cpu@custom
|
||||
- ecs@mappings
|
||||
- so-fleet_integrations.ip_mappings-1
|
||||
- so-fleet_globals-1
|
||||
- so-fleet_agent_id_verification-1
|
||||
data_stream:
|
||||
allow_custom_routing: false
|
||||
hidden: false
|
||||
ignore_missing_component_templates:
|
||||
- metrics@custom
|
||||
- system@custom
|
||||
- metrics-system.cpu@custom
|
||||
index_patterns:
|
||||
- metrics-system.cpu-*
|
||||
priority: 501
|
||||
template:
|
||||
settings:
|
||||
index:
|
||||
lifecycle:
|
||||
name: so-metrics-system.cpu-logs
|
||||
mode: time_series
|
||||
number_of_replicas: 0
|
||||
policy:
|
||||
phases:
|
||||
cold:
|
||||
actions:
|
||||
set_priority:
|
||||
priority: 0
|
||||
min_age: 60d
|
||||
delete:
|
||||
actions:
|
||||
delete: {}
|
||||
min_age: 365d
|
||||
hot:
|
||||
actions:
|
||||
rollover:
|
||||
max_age: 30d
|
||||
max_primary_shard_size: 50gb
|
||||
set_priority:
|
||||
priority: 100
|
||||
min_age: 0ms
|
||||
warm:
|
||||
actions:
|
||||
set_priority:
|
||||
priority: 50
|
||||
min_age: 30d
|
||||
so-metrics-system_x_diskio:
|
||||
data_stream_lifecycle:
|
||||
data_retention: 90d
|
||||
index_template:
|
||||
composed_of:
|
||||
- metrics@tsdb-settings
|
||||
- metrics-system.diskio@package
|
||||
- metrics@custom
|
||||
- system@custom
|
||||
- metrics-system.diskio@custom
|
||||
- ecs@mappings
|
||||
- so-fleet_integrations.ip_mappings-1
|
||||
- so-fleet_globals-1
|
||||
- so-fleet_agent_id_verification-1
|
||||
data_stream:
|
||||
allow_custom_routing: false
|
||||
hidden: false
|
||||
ignore_missing_component_templates:
|
||||
- metrics@custom
|
||||
- system@custom
|
||||
- metrics-system.diskio@custom
|
||||
index_patterns:
|
||||
- metrics-system.diskio-*
|
||||
priority: 501
|
||||
template:
|
||||
settings:
|
||||
index:
|
||||
lifecycle:
|
||||
name: so-metrics-system.diskio-logs
|
||||
mode: time_series
|
||||
number_of_replicas: 0
|
||||
policy:
|
||||
phases:
|
||||
cold:
|
||||
actions:
|
||||
set_priority:
|
||||
priority: 0
|
||||
min_age: 60d
|
||||
delete:
|
||||
actions:
|
||||
delete: {}
|
||||
min_age: 365d
|
||||
hot:
|
||||
actions:
|
||||
rollover:
|
||||
max_age: 30d
|
||||
max_primary_shard_size: 50gb
|
||||
set_priority:
|
||||
priority: 100
|
||||
min_age: 0ms
|
||||
warm:
|
||||
actions:
|
||||
set_priority:
|
||||
priority: 50
|
||||
min_age: 30d
|
||||
so-metrics-system_x_filesystem:
|
||||
data_stream_lifecycle:
|
||||
data_retention: 90d
|
||||
index_template:
|
||||
composed_of:
|
||||
- metrics@tsdb-settings
|
||||
- metrics-system.filesystem@package
|
||||
- metrics@custom
|
||||
- system@custom
|
||||
- metrics-system.filesystem@custom
|
||||
- ecs@mappings
|
||||
- so-fleet_integrations.ip_mappings-1
|
||||
- so-fleet_globals-1
|
||||
- so-fleet_agent_id_verification-1
|
||||
data_stream:
|
||||
allow_custom_routing: false
|
||||
hidden: false
|
||||
ignore_missing_component_templates:
|
||||
- metrics@custom
|
||||
- system@custom
|
||||
- metrics-system.filesystem@custom
|
||||
index_patterns:
|
||||
- metrics-system.filesystem-*
|
||||
priority: 501
|
||||
template:
|
||||
settings:
|
||||
index:
|
||||
lifecycle:
|
||||
name: so-metrics-system.filesystem-logs
|
||||
mode: time_series
|
||||
number_of_replicas: 0
|
||||
policy:
|
||||
phases:
|
||||
cold:
|
||||
actions:
|
||||
set_priority:
|
||||
priority: 0
|
||||
min_age: 60d
|
||||
delete:
|
||||
actions:
|
||||
delete: {}
|
||||
min_age: 365d
|
||||
hot:
|
||||
actions:
|
||||
rollover:
|
||||
max_age: 30d
|
||||
max_primary_shard_size: 50gb
|
||||
set_priority:
|
||||
priority: 100
|
||||
min_age: 0ms
|
||||
warm:
|
||||
actions:
|
||||
set_priority:
|
||||
priority: 50
|
||||
min_age: 30d
|
||||
so-metrics-system_x_fsstat:
|
||||
data_stream_lifecycle:
|
||||
data_retention: 90d
|
||||
index_template:
|
||||
composed_of:
|
||||
- metrics@tsdb-settings
|
||||
- metrics-system.fsstat@package
|
||||
- metrics@custom
|
||||
- system@custom
|
||||
- metrics-system.fsstat@custom
|
||||
- ecs@mappings
|
||||
- so-fleet_integrations.ip_mappings-1
|
||||
- so-fleet_globals-1
|
||||
- so-fleet_agent_id_verification-1
|
||||
data_stream:
|
||||
allow_custom_routing: false
|
||||
hidden: false
|
||||
ignore_missing_component_templates:
|
||||
- metrics@custom
|
||||
- system@custom
|
||||
- metrics-system.fsstat@custom
|
||||
index_patterns:
|
||||
- metrics-system.fsstat-*
|
||||
priority: 501
|
||||
template:
|
||||
settings:
|
||||
index:
|
||||
lifecycle:
|
||||
name: so-metrics-system.fsstat-logs
|
||||
mode: time_series
|
||||
number_of_replicas: 0
|
||||
policy:
|
||||
phases:
|
||||
cold:
|
||||
actions:
|
||||
set_priority:
|
||||
priority: 0
|
||||
min_age: 60d
|
||||
delete:
|
||||
actions:
|
||||
delete: {}
|
||||
min_age: 365d
|
||||
hot:
|
||||
actions:
|
||||
rollover:
|
||||
max_age: 30d
|
||||
max_primary_shard_size: 50gb
|
||||
set_priority:
|
||||
priority: 100
|
||||
min_age: 0ms
|
||||
warm:
|
||||
actions:
|
||||
set_priority:
|
||||
priority: 50
|
||||
min_age: 30d
|
||||
so-metrics-system_x_load:
|
||||
data_stream_lifecycle:
|
||||
data_retention: 90d
|
||||
index_template:
|
||||
composed_of:
|
||||
- metrics@tsdb-settings
|
||||
- metrics-system.load@package
|
||||
- metrics@custom
|
||||
- system@custom
|
||||
- metrics-system.load@custom
|
||||
- ecs@mappings
|
||||
- so-fleet_integrations.ip_mappings-1
|
||||
- so-fleet_globals-1
|
||||
- so-fleet_agent_id_verification-1
|
||||
data_stream:
|
||||
allow_custom_routing: false
|
||||
hidden: false
|
||||
ignore_missing_component_templates:
|
||||
- metrics@custom
|
||||
- system@custom
|
||||
- metrics-system.load@custom
|
||||
index_patterns:
|
||||
- metrics-system.load-*
|
||||
priority: 501
|
||||
template:
|
||||
settings:
|
||||
index:
|
||||
lifecycle:
|
||||
name: so-metrics-system.load-logs
|
||||
mode: time_series
|
||||
number_of_replicas: 0
|
||||
policy:
|
||||
phases:
|
||||
cold:
|
||||
actions:
|
||||
set_priority:
|
||||
priority: 0
|
||||
min_age: 60d
|
||||
delete:
|
||||
actions:
|
||||
delete: {}
|
||||
min_age: 365d
|
||||
hot:
|
||||
actions:
|
||||
rollover:
|
||||
max_age: 30d
|
||||
max_primary_shard_size: 50gb
|
||||
set_priority:
|
||||
priority: 100
|
||||
min_age: 0ms
|
||||
warm:
|
||||
actions:
|
||||
set_priority:
|
||||
priority: 50
|
||||
min_age: 30d
|
||||
so-metrics-system_x_memory:
|
||||
data_stream_lifecycle:
|
||||
data_retention: 90d
|
||||
index_template:
|
||||
composed_of:
|
||||
- metrics@tsdb-settings
|
||||
- metrics-system.memory@package
|
||||
- metrics@custom
|
||||
- system@custom
|
||||
- metrics-system.memory@custom
|
||||
- ecs@mappings
|
||||
- so-fleet_integrations.ip_mappings-1
|
||||
- so-fleet_globals-1
|
||||
- so-fleet_agent_id_verification-1
|
||||
data_stream:
|
||||
allow_custom_routing: false
|
||||
hidden: false
|
||||
ignore_missing_component_templates:
|
||||
- metrics@custom
|
||||
- system@custom
|
||||
- metrics-system.memory@custom
|
||||
index_patterns:
|
||||
- metrics-system.memory-*
|
||||
priority: 501
|
||||
template:
|
||||
settings:
|
||||
index:
|
||||
lifecycle:
|
||||
name: so-metrics-system.memory-logs
|
||||
mode: time_series
|
||||
number_of_replicas: 0
|
||||
policy:
|
||||
phases:
|
||||
cold:
|
||||
actions:
|
||||
set_priority:
|
||||
priority: 0
|
||||
min_age: 60d
|
||||
delete:
|
||||
actions:
|
||||
delete: {}
|
||||
min_age: 365d
|
||||
hot:
|
||||
actions:
|
||||
rollover:
|
||||
max_age: 30d
|
||||
max_primary_shard_size: 50gb
|
||||
set_priority:
|
||||
priority: 100
|
||||
min_age: 0ms
|
||||
warm:
|
||||
actions:
|
||||
set_priority:
|
||||
priority: 50
|
||||
min_age: 30d
|
||||
so-metrics-system_x_network:
|
||||
data_stream_lifecycle:
|
||||
data_retention: 90d
|
||||
index_template:
|
||||
composed_of:
|
||||
- metrics@tsdb-settings
|
||||
- metrics-system.network@package
|
||||
- metrics@custom
|
||||
- system@custom
|
||||
- metrics-system.network@custom
|
||||
- ecs@mappings
|
||||
- so-fleet_integrations.ip_mappings-1
|
||||
- so-fleet_globals-1
|
||||
- so-fleet_agent_id_verification-1
|
||||
data_stream:
|
||||
allow_custom_routing: false
|
||||
hidden: false
|
||||
ignore_missing_component_templates:
|
||||
- metrics@custom
|
||||
- system@custom
|
||||
- metrics-system.network@custom
|
||||
index_patterns:
|
||||
- metrics-system.network-*
|
||||
priority: 501
|
||||
template:
|
||||
settings:
|
||||
index:
|
||||
lifecycle:
|
||||
name: so-metrics-system.network-logs
|
||||
mode: time_series
|
||||
number_of_replicas: 0
|
||||
policy:
|
||||
phases:
|
||||
cold:
|
||||
actions:
|
||||
set_priority:
|
||||
priority: 0
|
||||
min_age: 60d
|
||||
delete:
|
||||
actions:
|
||||
delete: {}
|
||||
min_age: 365d
|
||||
hot:
|
||||
actions:
|
||||
rollover:
|
||||
max_age: 30d
|
||||
max_primary_shard_size: 50gb
|
||||
set_priority:
|
||||
priority: 100
|
||||
min_age: 0ms
|
||||
warm:
|
||||
actions:
|
||||
set_priority:
|
||||
priority: 50
|
||||
min_age: 30d
|
||||
so-metrics-system_x_ntp:
|
||||
data_stream_lifecycle:
|
||||
data_retention: 90d
|
||||
index_template:
|
||||
composed_of:
|
||||
- metrics@settings
|
||||
- metrics-system.ntp@package
|
||||
- metrics@custom
|
||||
- system@custom
|
||||
- metrics-system.ntp@custom
|
||||
- ecs@mappings
|
||||
- so-fleet_integrations.ip_mappings-1
|
||||
- so-fleet_globals-1
|
||||
- so-fleet_agent_id_verification-1
|
||||
data_stream:
|
||||
allow_custom_routing: false
|
||||
hidden: false
|
||||
ignore_missing_component_templates:
|
||||
- metrics@custom
|
||||
- system@custom
|
||||
- metrics-system.ntp@custom
|
||||
index_patterns:
|
||||
- metrics-system.ntp-*
|
||||
priority: 501
|
||||
template:
|
||||
settings:
|
||||
index:
|
||||
lifecycle:
|
||||
name: so-metrics-system.ntp-logs
|
||||
number_of_replicas: 0
|
||||
policy:
|
||||
phases:
|
||||
cold:
|
||||
actions:
|
||||
set_priority:
|
||||
priority: 0
|
||||
min_age: 60d
|
||||
delete:
|
||||
actions:
|
||||
delete: {}
|
||||
min_age: 365d
|
||||
hot:
|
||||
actions:
|
||||
rollover:
|
||||
max_age: 30d
|
||||
max_primary_shard_size: 50gb
|
||||
set_priority:
|
||||
priority: 100
|
||||
min_age: 0ms
|
||||
warm:
|
||||
actions:
|
||||
set_priority:
|
||||
priority: 50
|
||||
min_age: 30d
|
||||
so-metrics-system_x_process:
|
||||
data_stream_lifecycle:
|
||||
data_retention: 90d
|
||||
index_template:
|
||||
composed_of:
|
||||
- metrics@tsdb-settings
|
||||
- metrics-system.process@package
|
||||
- metrics@custom
|
||||
- system@custom
|
||||
- metrics-system.process@custom
|
||||
- ecs@mappings
|
||||
- so-fleet_integrations.ip_mappings-1
|
||||
- so-fleet_globals-1
|
||||
- so-fleet_agent_id_verification-1
|
||||
data_stream:
|
||||
allow_custom_routing: false
|
||||
hidden: false
|
||||
ignore_missing_component_templates:
|
||||
- metrics@custom
|
||||
- system@custom
|
||||
- metrics-system.process@custom
|
||||
index_patterns:
|
||||
- metrics-system.process-*
|
||||
priority: 501
|
||||
template:
|
||||
settings:
|
||||
index:
|
||||
lifecycle:
|
||||
name: so-metrics-system.process-logs
|
||||
mode: time_series
|
||||
number_of_replicas: 0
|
||||
policy:
|
||||
phases:
|
||||
cold:
|
||||
actions:
|
||||
set_priority:
|
||||
priority: 0
|
||||
min_age: 60d
|
||||
delete:
|
||||
actions:
|
||||
delete: {}
|
||||
min_age: 365d
|
||||
hot:
|
||||
actions:
|
||||
rollover:
|
||||
max_age: 30d
|
||||
max_primary_shard_size: 50gb
|
||||
set_priority:
|
||||
priority: 100
|
||||
min_age: 0ms
|
||||
warm:
|
||||
actions:
|
||||
set_priority:
|
||||
priority: 50
|
||||
min_age: 30d
|
||||
so-metrics-system_x_process_x_summary:
|
||||
data_stream_lifecycle:
|
||||
data_retention: 90d
|
||||
index_template:
|
||||
composed_of:
|
||||
- metrics@tsdb-settings
|
||||
- metrics-system.process.summary@package
|
||||
- metrics@custom
|
||||
- system@custom
|
||||
- metrics-system.process.summary@custom
|
||||
- ecs@mappings
|
||||
- so-fleet_integrations.ip_mappings-1
|
||||
- so-fleet_globals-1
|
||||
- so-fleet_agent_id_verification-1
|
||||
data_stream:
|
||||
allow_custom_routing: false
|
||||
hidden: false
|
||||
ignore_missing_component_templates:
|
||||
- metrics@custom
|
||||
- system@custom
|
||||
- metrics-system.process.summary@custom
|
||||
index_patterns:
|
||||
- metrics-system.process.summary-*
|
||||
priority: 501
|
||||
template:
|
||||
settings:
|
||||
index:
|
||||
lifecycle:
|
||||
name: so-metrics-system.process.summary-logs
|
||||
mode: time_series
|
||||
number_of_replicas: 0
|
||||
policy:
|
||||
phases:
|
||||
cold:
|
||||
actions:
|
||||
set_priority:
|
||||
priority: 0
|
||||
min_age: 60d
|
||||
delete:
|
||||
actions:
|
||||
delete: {}
|
||||
min_age: 365d
|
||||
hot:
|
||||
actions:
|
||||
rollover:
|
||||
max_age: 30d
|
||||
max_primary_shard_size: 50gb
|
||||
set_priority:
|
||||
priority: 100
|
||||
min_age: 0ms
|
||||
warm:
|
||||
actions:
|
||||
set_priority:
|
||||
priority: 50
|
||||
min_age: 30d
|
||||
so-metrics-system_x_socket_summary:
|
||||
data_stream_lifecycle:
|
||||
data_retention: 90d
|
||||
index_template:
|
||||
composed_of:
|
||||
- metrics@tsdb-settings
|
||||
- metrics-system.socket_summary@package
|
||||
- metrics@custom
|
||||
- system@custom
|
||||
- metrics-system.socket_summary@custom
|
||||
- ecs@mappings
|
||||
- so-fleet_integrations.ip_mappings-1
|
||||
- so-fleet_globals-1
|
||||
- so-fleet_agent_id_verification-1
|
||||
data_stream:
|
||||
allow_custom_routing: false
|
||||
hidden: false
|
||||
ignore_missing_component_templates:
|
||||
- metrics@custom
|
||||
- system@custom
|
||||
- metrics-system.socket_summary@custom
|
||||
index_patterns:
|
||||
- metrics-system.socket_summary-*
|
||||
priority: 501
|
||||
template:
|
||||
settings:
|
||||
index:
|
||||
lifecycle:
|
||||
name: so-metrics-system.socket_summary-logs
|
||||
mode: time_series
|
||||
number_of_replicas: 0
|
||||
policy:
|
||||
phases:
|
||||
cold:
|
||||
actions:
|
||||
set_priority:
|
||||
priority: 0
|
||||
min_age: 60d
|
||||
delete:
|
||||
actions:
|
||||
delete: {}
|
||||
min_age: 365d
|
||||
hot:
|
||||
actions:
|
||||
rollover:
|
||||
max_age: 30d
|
||||
max_primary_shard_size: 50gb
|
||||
set_priority:
|
||||
priority: 100
|
||||
min_age: 0ms
|
||||
warm:
|
||||
actions:
|
||||
set_priority:
|
||||
priority: 50
|
||||
min_age: 30d
|
||||
so-metrics-system_x_uptime:
|
||||
data_stream_lifecycle:
|
||||
data_retention: 90d
|
||||
index_template:
|
||||
composed_of:
|
||||
- metrics@tsdb-settings
|
||||
- metrics-system.uptime@package
|
||||
- metrics@custom
|
||||
- system@custom
|
||||
- metrics-system.uptime@custom
|
||||
- ecs@mappings
|
||||
- so-fleet_integrations.ip_mappings-1
|
||||
- so-fleet_globals-1
|
||||
- so-fleet_agent_id_verification-1
|
||||
data_stream:
|
||||
allow_custom_routing: false
|
||||
hidden: false
|
||||
ignore_missing_component_templates:
|
||||
- metrics@custom
|
||||
- system@custom
|
||||
- metrics-system.uptime@custom
|
||||
index_patterns:
|
||||
- metrics-system.uptime-*
|
||||
priority: 501
|
||||
template:
|
||||
settings:
|
||||
index:
|
||||
lifecycle:
|
||||
name: so-metrics-system.uptime-logs
|
||||
mode: time_series
|
||||
number_of_replicas: 0
|
||||
policy:
|
||||
phases:
|
||||
cold:
|
||||
actions:
|
||||
set_priority:
|
||||
priority: 0
|
||||
min_age: 60d
|
||||
delete:
|
||||
actions:
|
||||
delete: {}
|
||||
min_age: 365d
|
||||
hot:
|
||||
actions:
|
||||
rollover:
|
||||
max_age: 30d
|
||||
max_primary_shard_size: 50gb
|
||||
set_priority:
|
||||
priority: 100
|
||||
min_age: 0ms
|
||||
warm:
|
||||
actions:
|
||||
set_priority:
|
||||
priority: 50
|
||||
min_age: 30d
|
||||
so-logs-windows_x_forwarded:
|
||||
index_sorting: false
|
||||
data_stream_lifecycle:
|
||||
|
||||
@@ -10,9 +10,6 @@
|
||||
{% from 'elasticsearch/config.map.jinja' import ELASTICSEARCH_NODES %}
|
||||
{% from 'elasticsearch/config.map.jinja' import ELASTICSEARCH_SEED_HOSTS %}
|
||||
{% from 'elasticsearch/config.map.jinja' import ELASTICSEARCHMERGED %}
|
||||
{% from 'docker/macros/docker_endpoint.jinja' import clear_stale_endpoint %}
|
||||
|
||||
{{ clear_stale_endpoint('so-elasticsearch', 'sobridge', DOCKERMERGED.containers['so-elasticsearch'].ip) }}
|
||||
|
||||
include:
|
||||
- ca
|
||||
@@ -27,6 +24,7 @@ include:
|
||||
so-elasticsearch:
|
||||
docker_container.running:
|
||||
- image: {{ GLOBALS.registry_host }}:5000/{{ GLOBALS.image_repo }}/so-elasticsearch:{{ ELASTICSEARCHMERGED.version }}
|
||||
- restart_policy: unless-stopped
|
||||
- hostname: elasticsearch
|
||||
- name: so-elasticsearch
|
||||
- user: elasticsearch
|
||||
|
||||
+10
-10
@@ -118,70 +118,70 @@
|
||||
{
|
||||
"pipeline": {
|
||||
"tag": "pipeline_e16851a7",
|
||||
"name": "logs-pfsense.log-1.25.2-firewall",
|
||||
"name": "logs-pfsense.log-1.25.4-firewall",
|
||||
"if": "ctx.event.provider == 'filterlog'"
|
||||
}
|
||||
},
|
||||
{
|
||||
"pipeline": {
|
||||
"tag": "pipeline_828590b5",
|
||||
"name": "logs-pfsense.log-1.25.2-openvpn",
|
||||
"name": "logs-pfsense.log-1.25.4-openvpn",
|
||||
"if": "ctx.event.provider == 'openvpn'"
|
||||
}
|
||||
},
|
||||
{
|
||||
"pipeline": {
|
||||
"tag": "pipeline_9d37039c",
|
||||
"name": "logs-pfsense.log-1.25.2-ipsec",
|
||||
"name": "logs-pfsense.log-1.25.4-ipsec",
|
||||
"if": "ctx.event.provider == 'charon'"
|
||||
}
|
||||
},
|
||||
{
|
||||
"pipeline": {
|
||||
"tag": "pipeline_ad56bbca",
|
||||
"name": "logs-pfsense.log-1.25.2-dhcp",
|
||||
"name": "logs-pfsense.log-1.25.4-dhcp",
|
||||
"if": "[\"dhcpd\", \"dhclient\", \"dhcp6c\", \"dnsmasq-dhcp\"].contains(ctx.event.provider)"
|
||||
}
|
||||
},
|
||||
{
|
||||
"pipeline": {
|
||||
"tag": "pipeline_dd85553d",
|
||||
"name": "logs-pfsense.log-1.25.2-unbound",
|
||||
"name": "logs-pfsense.log-1.25.4-unbound",
|
||||
"if": "ctx.event.provider == 'unbound'"
|
||||
}
|
||||
},
|
||||
{
|
||||
"pipeline": {
|
||||
"tag": "pipeline_720ed255",
|
||||
"name": "logs-pfsense.log-1.25.2-haproxy",
|
||||
"name": "logs-pfsense.log-1.25.4-haproxy",
|
||||
"if": "ctx.event.provider == 'haproxy'"
|
||||
}
|
||||
},
|
||||
{
|
||||
"pipeline": {
|
||||
"tag": "pipeline_456beba5",
|
||||
"name": "logs-pfsense.log-1.25.2-php-fpm",
|
||||
"name": "logs-pfsense.log-1.25.4-php-fpm",
|
||||
"if": "ctx.event.provider == 'php-fpm'"
|
||||
}
|
||||
},
|
||||
{
|
||||
"pipeline": {
|
||||
"tag": "pipeline_a0d89375",
|
||||
"name": "logs-pfsense.log-1.25.2-squid",
|
||||
"name": "logs-pfsense.log-1.25.4-squid",
|
||||
"if": "ctx.event.provider == 'squid'"
|
||||
}
|
||||
},
|
||||
{
|
||||
"pipeline": {
|
||||
"tag": "pipeline_c2f1ed55",
|
||||
"name": "logs-pfsense.log-1.25.2-snort",
|
||||
"name": "logs-pfsense.log-1.25.4-snort",
|
||||
"if": "ctx.event.provider == 'snort'"
|
||||
}
|
||||
},
|
||||
{
|
||||
"pipeline": {
|
||||
"tag":"pipeline_33db1c9e",
|
||||
"name": "logs-pfsense.log-1.25.2-suricata",
|
||||
"name": "logs-pfsense.log-1.25.4-suricata",
|
||||
"if": "ctx.event.provider == 'suricata'"
|
||||
}
|
||||
},
|
||||
@@ -5,7 +5,8 @@
|
||||
{ "rename": { "field": "message2.proto", "target_field": "network.transport", "ignore_missing": true } },
|
||||
{ "rename": { "field": "message2.app_proto", "target_field": "network.protocol", "ignore_missing": true } },
|
||||
{ "rename": { "field": "message2.fileinfo.filename", "target_field": "file.name", "ignore_missing": true } },
|
||||
{ "rename": { "field": "message2.fileinfo.gaps", "target_field": "file.bytes.missing", "ignore_missing": true } },
|
||||
{ "rename": { "field": "message2.fileinfo.gaps", "target_field": "suricata.fileinfo.gaps", "ignore_missing": true } },
|
||||
{ "set": { "if": "ctx.suricata?.fileinfo?.gaps == false", "field": "file.bytes.missing", "value": 0 } },
|
||||
{ "rename": { "field": "message2.fileinfo.magic", "target_field": "file.mime_type", "ignore_missing": true } },
|
||||
{ "rename": { "field": "message2.fileinfo.md5", "target_field": "hash.md5", "ignore_missing": true } },
|
||||
{ "rename": { "field": "message2.fileinfo.sha1", "target_field": "hash.sha1", "ignore_missing": true } },
|
||||
|
||||
@@ -20,7 +20,8 @@ appender.rolling.strategy.type = DefaultRolloverStrategy
|
||||
appender.rolling.strategy.action.type = Delete
|
||||
appender.rolling.strategy.action.basepath = /var/log/elasticsearch
|
||||
appender.rolling.strategy.action.condition.type = IfFileName
|
||||
appender.rolling.strategy.action.condition.glob = *.log.gz
|
||||
# age delete regular securityonion.log.gz and gc.log.NN files
|
||||
appender.rolling.strategy.action.condition.regex = (?:.*[.]log[.]gz|gc[.]log[.][0-9]+)
|
||||
appender.rolling.strategy.action.condition.nested_condition.type = IfLastModified
|
||||
appender.rolling.strategy.action.condition.nested_condition.age = 7D
|
||||
|
||||
|
||||
@@ -64,6 +64,43 @@ elasticsearch:
|
||||
flood_stage:
|
||||
description: The max percentage of used disk space that will cause the node to take protective actions, such as blocking incoming events.
|
||||
helpLink: elasticsearch
|
||||
lifecycle:
|
||||
default:
|
||||
rollover:
|
||||
description: This property accepts a key value pair formatted string and configures the conditions that would trigger a data stream to rollover when it has lifecycle configured.
|
||||
forcedType: string
|
||||
regex: ^max_age=(auto|[1-9][0-9]*[hd]),max_primary_shard_size=[1-9][0-9]*gb,min_docs=(0|[1-9][0-9]*),max_primary_shard_docs=[1-9][0-9]*$
|
||||
regexFailureMessage: Must be in the format of "max_age=auto|<number><h|d>,max_primary_shard_size=<number>gb,min_docs=<number>,max_primary_shard_docs=<number>".
|
||||
advanced: True
|
||||
global: True
|
||||
data_streams:
|
||||
lifecycle:
|
||||
poll_interval:
|
||||
description: How often Elasticsearch checks what the next action is for all data streams with a built-in lifecycle.
|
||||
forcedType: string
|
||||
regex: "^[1-9][0-9]*[mhd]$"
|
||||
regexFailureMessage: Must be a number followed by m, h, or d.
|
||||
advanced: True
|
||||
global: true
|
||||
helpLink: elasticsearch
|
||||
target:
|
||||
merge:
|
||||
policy:
|
||||
merge_factor:
|
||||
description: Data stream lifecycle implements tail merging by updating the Lucene merge policy factor for the target backing index. The merge factor is both the number of segments that should be merged together, and the maximum number of segments that we expect to find.
|
||||
forcedType: int
|
||||
regex: "^[1-9][0-9]*$"
|
||||
advanced: True
|
||||
global: true
|
||||
helpLink: elasticsearch
|
||||
floor_segment:
|
||||
description: Data stream lifecycle implements tail merging by updating the Lucene merge policy floor segment for the target backing index. This floor segment size is a way to prevent indices from having a long tail of very small segments.
|
||||
forcedType: string
|
||||
regex: "^[1-9][0-9]*[MG]B$"
|
||||
regexFailureMessage: Must be a number followed by MB or GB, such as 100MB.
|
||||
advanced: True
|
||||
global: true
|
||||
helpLink: elasticsearch
|
||||
action:
|
||||
destructive_requires_name:
|
||||
description: Requires explicit index names when deleting indices. Prevents accidental deletion of indices via wildcard patterns.
|
||||
@@ -645,6 +682,9 @@ elasticsearch:
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: elasticsearch
|
||||
so-assistant-chat: *dataStreamSettings
|
||||
so-assistant-session: *dataStreamSettings
|
||||
so-elastic-agent-monitor: *dataStreamSettings
|
||||
so-logs-soc: *dataStreamSettings
|
||||
so-logs-system_x_auth: *dataStreamSettings
|
||||
so-logs-system_x_syslog: *dataStreamSettings
|
||||
@@ -667,7 +707,10 @@ elasticsearch:
|
||||
so-logs-elastic_agent_x_auditbeat: *dataStreamSettings
|
||||
so-logs-elastic_agent_x_cloudbeat: *dataStreamSettings
|
||||
so-logs-elastic_agent_x_endpoint_security: *dataStreamSettings
|
||||
so-logs-endpoint_x_actions: *dataStreamSettings
|
||||
so-logs-endpoint_x_action_x_responses: *dataStreamSettings
|
||||
so-logs-endpoint_x_alerts: *dataStreamSettings
|
||||
so-logs-endpoint_x_diagnostic_x_collection: *dataStreamSettings
|
||||
so-logs-endpoint_x_events_x_api: *dataStreamSettings
|
||||
so-logs-endpoint_x_events_x_file: *dataStreamSettings
|
||||
so-logs-endpoint_x_events_x_library: *dataStreamSettings
|
||||
@@ -675,6 +718,7 @@ elasticsearch:
|
||||
so-logs-endpoint_x_events_x_process: *dataStreamSettings
|
||||
so-logs-endpoint_x_events_x_registry: *dataStreamSettings
|
||||
so-logs-endpoint_x_events_x_security: *dataStreamSettings
|
||||
so-logs-endpoint_x_heartbeat: *dataStreamSettings
|
||||
so-logs-elastic_agent_x_filebeat: *dataStreamSettings
|
||||
so-logs-elastic_agent_x_fleet_server: *dataStreamSettings
|
||||
so-logs-elastic_agent_x_heartbeat: *dataStreamSettings
|
||||
@@ -690,6 +734,19 @@ elasticsearch:
|
||||
so-metrics-vsphere_x_datastore: *dataStreamSettings
|
||||
so-metrics-vsphere_x_host: *dataStreamSettings
|
||||
so-metrics-vsphere_x_virtualmachine: *dataStreamSettings
|
||||
so-metrics-system_x_core: *dataStreamSettings
|
||||
so-metrics-system_x_cpu: *dataStreamSettings
|
||||
so-metrics-system_x_diskio: *dataStreamSettings
|
||||
so-metrics-system_x_filesystem: *dataStreamSettings
|
||||
so-metrics-system_x_fsstat: *dataStreamSettings
|
||||
so-metrics-system_x_load: *dataStreamSettings
|
||||
so-metrics-system_x_memory: *dataStreamSettings
|
||||
so-metrics-system_x_network: *dataStreamSettings
|
||||
so-metrics-system_x_ntp: *dataStreamSettings
|
||||
so-metrics-system_x_process: *dataStreamSettings
|
||||
so-metrics-system_x_process_x_summary: *dataStreamSettings
|
||||
so-metrics-system_x_socket_summary: *dataStreamSettings
|
||||
so-metrics-system_x_uptime: *dataStreamSettings
|
||||
so-common: *dataStreamSettings
|
||||
so-endgame: *dataStreamSettings
|
||||
so-idh: *dataStreamSettings
|
||||
@@ -880,17 +937,6 @@ elasticsearch:
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: elasticsearch
|
||||
rollover:
|
||||
max_age:
|
||||
description: Maximum age of index. Once an index reaches this limit, it will be rolled over into a new index.
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: elasticsearch
|
||||
max_primary_shard_size:
|
||||
description: Maximum primary shard size. Once an index reaches this limit, it will be rolled over into a new index.
|
||||
global: True
|
||||
advanced: True
|
||||
helpLink: elasticsearch
|
||||
shrink:
|
||||
method:
|
||||
description: Shrink the index to a new index with fewer primary shards. Shrink operation is by count or size.
|
||||
@@ -987,8 +1033,6 @@ elasticsearch:
|
||||
helpLink: elasticsearch
|
||||
sos-backup: *indexSettings
|
||||
so-detection: *indexSettings
|
||||
so-assistant-chat: *indexSettings
|
||||
so-assistant-session: *indexSettings
|
||||
so-metrics-fleet_server_x_agent_status: &fleetMetricsSettings
|
||||
index_sorting:
|
||||
description: Sorts the index by event time, at the cost of additional processing resource consumption.
|
||||
|
||||
@@ -109,9 +109,15 @@
|
||||
{% if not settings.get('index_sorting', False) | to_bool and settings.index_template.template.settings.index.sort is defined %}
|
||||
{% do settings.index_template.template.settings.index.pop('sort') %}
|
||||
{% endif %}
|
||||
{% if DATA_RETENTION_METHOD == 'DLM' and settings.index_template.data_stream is defined and settings.data_stream_lifecycle is defined %}
|
||||
{% if settings.data_stream_lifecycle.data_retention is defined and settings.data_stream_lifecycle.data_retention %}
|
||||
{% do settings.index_template.template.update({'lifecycle': {'data_retention': settings.data_stream_lifecycle.data_retention}}) %}
|
||||
{% if DATA_RETENTION_METHOD == 'DLM' and settings.index_template.data_stream is defined %}
|
||||
{# Addon defaults are generated without data_stream_lifecycle, so fall back to global defaults. #}
|
||||
{% if settings.data_stream_lifecycle is defined %}
|
||||
{% set DATA_STREAM_LIFECYCLE = settings.data_stream_lifecycle %}
|
||||
{% else %}
|
||||
{% set DATA_STREAM_LIFECYCLE = DEFAULT_GLOBAL_OVERRIDES.data_stream_lifecycle %}
|
||||
{% endif %}
|
||||
{% if DATA_STREAM_LIFECYCLE.data_retention is defined and DATA_STREAM_LIFECYCLE.data_retention %}
|
||||
{% do settings.index_template.template.update({'lifecycle': {'data_retention': DATA_STREAM_LIFECYCLE.data_retention}}) %}
|
||||
{% else %}
|
||||
{% do settings.index_template.template.update({'lifecycle': {}}) %}
|
||||
{% endif %}
|
||||
|
||||
@@ -0,0 +1,203 @@
|
||||
#!/bin/bash
|
||||
# Copyright Security Onion Solutions LLC and/or licensed to Security Onion Solutions LLC under one
|
||||
# or more contributor license agreements. Licensed under the Elastic License 2.0 as shown at
|
||||
# https://securityonion.net/license; you may not use this file except in compliance with the
|
||||
# Elastic License 2.0.
|
||||
|
||||
set -eo pipefail
|
||||
|
||||
SETTINGS='{"index":{"auto_expand_replicas":"0-1"}}'
|
||||
KIBANA_PASSWORD=
|
||||
INDEX_PATTERNS=(
|
||||
'.entity_analytics.risk_score.lookup-*'
|
||||
'.entity_analytics.watchlists.*'
|
||||
'.entity_analytics.monitoring.users-*'
|
||||
'.entity_analytics.entity-leads-*'
|
||||
'.asset-criticality.asset-criticality-*'
|
||||
'.workflows-executions'
|
||||
'.workflows-step-executions'
|
||||
'.entities.v2.latest.security_*'
|
||||
'.entities.v2.history.security_*'
|
||||
'risk-score.risk-score-latest-*'
|
||||
'.metrics-endpoint.metadata_united_*'
|
||||
'.metrics-endpoint.metadata_current_*'
|
||||
)
|
||||
DATA_STREAM_PATTERNS=(
|
||||
'.entities.v2.updates.security_*'
|
||||
'risk-score.risk-score-*'
|
||||
'.rule-events'
|
||||
'.alert-actions'
|
||||
)
|
||||
TEMPLATE_PATTERNS=(
|
||||
'entities_v2_latest_security_default_index_template'
|
||||
'entities_v2_history_security_default_index_template'
|
||||
'.entities_v2_updates_security_default_index_template'
|
||||
'.risk-score.risk-score-default-index-template'
|
||||
'.rule-events'
|
||||
'.alert-actions'
|
||||
'.metrics-endpoint.metadata_united_default-template'
|
||||
'.metrics-endpoint.metadata_current_default-template'
|
||||
)
|
||||
|
||||
query_es() {
|
||||
if so-elasticsearch-query "$@" --fail --retry 3 --retry-delay 5; then
|
||||
return 0
|
||||
fi
|
||||
|
||||
# retry failed attempts with so_kibana user (system managed indices reject so_elastic user)
|
||||
local query_path="$1"
|
||||
shift
|
||||
|
||||
if [[ -z "$KIBANA_PASSWORD" ]]; then
|
||||
KIBANA_PASSWORD=$(salt-call pillar.get elasticsearch:auth:users:so_kibana_user:pass --out=newline_values_only)
|
||||
fi
|
||||
[[ -n "$KIBANA_PASSWORD" ]] || return 1
|
||||
|
||||
echo "Retrying ${query_path} as so_kibana." >&2
|
||||
curl -K /opt/so/conf/elasticsearch/curl.config --user "so_kibana:${KIBANA_PASSWORD}" \
|
||||
-s -k -L --fail --retry 3 --retry-delay 5 -H 'Content-Type: application/json' "https://localhost:9200/${query_path}" "$@"
|
||||
}
|
||||
|
||||
# add auto_expand_replicas=0-1 to given index
|
||||
set_auto_expand_replicas() {
|
||||
local index="$1"
|
||||
|
||||
echo "Setting auto_expand_replicas to 0-1 on ${index}."
|
||||
query_es "${index}/_settings" -XPUT -d "$SETTINGS" >/dev/null
|
||||
}
|
||||
|
||||
# resolve index patterns and find each index with an unassigned replica
|
||||
unassigned_replicas() {
|
||||
local pattern="$1"
|
||||
local resolved_indices response index
|
||||
|
||||
if ! resolved_indices=$(query_es "_resolve/index/${pattern}?expand_wildcards=all" 2>/dev/null); then
|
||||
return 0
|
||||
fi
|
||||
|
||||
while read -r index; do
|
||||
if ! response=$(query_es "_cat/shards/${index}?format=json&h=index,prirep,state" 2>/dev/null); then
|
||||
continue
|
||||
fi
|
||||
jq -r '.[]? | objects | select(.prirep == "r" and .state == "UNASSIGNED") | .index' <<<"$response"
|
||||
done < <(jq -r '.indices[]?.name' <<<"$resolved_indices")
|
||||
}
|
||||
|
||||
data_stream_indices() {
|
||||
local pattern="$1"
|
||||
local response
|
||||
|
||||
if ! response=$(query_es "_data_stream/${pattern}?expand_wildcards=all" 2>/dev/null); then
|
||||
return 0
|
||||
fi
|
||||
jq -r '.data_streams[]?.indices[]?.index_name' <<<"$response"
|
||||
}
|
||||
|
||||
update_system_indices() {
|
||||
local pattern="$1"
|
||||
local index
|
||||
|
||||
while read -r index; do
|
||||
[[ -n "$index" ]] && set_auto_expand_replicas "$index"
|
||||
done < <(unassigned_replicas "$pattern")
|
||||
}
|
||||
|
||||
# update data stream backing indices with unassigned replicas
|
||||
update_system_ds() {
|
||||
local pattern="$1"
|
||||
local index
|
||||
|
||||
while read -r index; do
|
||||
while read -r unassigned_index; do
|
||||
[[ -n "$unassigned_index" ]] && set_auto_expand_replicas "$unassigned_index"
|
||||
done < <(unassigned_replicas "$index")
|
||||
done < <(data_stream_indices "$pattern")
|
||||
}
|
||||
|
||||
has_unassigned_replicas() {
|
||||
local pattern="$1"
|
||||
local index
|
||||
|
||||
index=$(unassigned_replicas "$pattern" | sed -n '1p')
|
||||
[[ -n "$index" ]]
|
||||
}
|
||||
|
||||
data_stream_has_unassigned_replicas() {
|
||||
local pattern="$1"
|
||||
local index
|
||||
while read -r index; do
|
||||
has_unassigned_replicas "$index" && return 0
|
||||
done < <(data_stream_indices "$pattern")
|
||||
|
||||
return 1
|
||||
}
|
||||
|
||||
needs_patch() {
|
||||
local pattern
|
||||
for pattern in "${INDEX_PATTERNS[@]}"; do
|
||||
has_unassigned_replicas "$pattern" && return 0
|
||||
done
|
||||
for pattern in "${DATA_STREAM_PATTERNS[@]}"; do
|
||||
data_stream_has_unassigned_replicas "$pattern" && return 0
|
||||
done
|
||||
|
||||
return 1
|
||||
}
|
||||
|
||||
# get index templates, update with auto_expand_replicas=0-1, and PUT back. Keeping mappings/settings/aliases in-place
|
||||
update_system_templates() {
|
||||
local pattern="$1"
|
||||
local templates name response template auto_expand_replicas
|
||||
|
||||
if ! templates=$(query_es "_index_template/${pattern}" 2>/dev/null); then
|
||||
return 0
|
||||
fi
|
||||
while read -r name; do
|
||||
response=$(query_es "_index_template/${name}")
|
||||
template=$(jq -c '.index_templates[0].index_template' <<<"$response")
|
||||
auto_expand_replicas=$(jq -r '.template.settings["index.auto_expand_replicas"] // .template.settings.index.auto_expand_replicas // empty' <<<"$template")
|
||||
[[ "$auto_expand_replicas" == "0-1" ]] && continue
|
||||
|
||||
template=$(jq '
|
||||
if (.template.settings.index | type) == "object" then
|
||||
.template.settings.index.auto_expand_replicas = "0-1"
|
||||
else
|
||||
.template.settings["index.auto_expand_replicas"] = "0-1"
|
||||
end
|
||||
| del(.created_date_millis, .modified_date_millis)
|
||||
' <<<"$template")
|
||||
echo "Setting auto_expand_replicas to 0-1 on index template ${name}."
|
||||
query_es "_index_template/${name}" -XPUT -d "$template" >/dev/null
|
||||
done < <(jq -r '.index_templates[]?.name' <<<"$templates")
|
||||
}
|
||||
|
||||
if [[ "${1:-}" == "--check" ]]; then
|
||||
needs_patch
|
||||
exit $?
|
||||
fi
|
||||
|
||||
if [[ $# -ne 0 ]]; then
|
||||
echo "Usage: $0 [--check]" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
patched=false
|
||||
for pattern in "${INDEX_PATTERNS[@]}"; do
|
||||
if has_unassigned_replicas "$pattern"; then
|
||||
update_system_indices "$pattern"
|
||||
patched=true
|
||||
fi
|
||||
done
|
||||
|
||||
for pattern in "${DATA_STREAM_PATTERNS[@]}"; do
|
||||
if data_stream_has_unassigned_replicas "$pattern"; then
|
||||
update_system_ds "$pattern"
|
||||
patched=true
|
||||
fi
|
||||
done
|
||||
|
||||
if [[ "$patched" == true ]]; then
|
||||
for pattern in "${TEMPLATE_PATTERNS[@]}"; do
|
||||
update_system_templates "$pattern"
|
||||
done
|
||||
fi
|
||||
@@ -4,11 +4,19 @@
|
||||
{%- set role = GLOBALS.role.split('-')[1] %}
|
||||
{%- from 'firewall/containers.map.jinja' import NODE_CONTAINERS %}
|
||||
|
||||
{%- set NODE_NETWORKS = [] %}
|
||||
{%- for NETNAME, NETWORK in DOCKERMERGED.networks.items() %}
|
||||
{%- if not NETWORK.get('manager_only') or GLOBALS.get('is_manager', False) %}
|
||||
{%- do NODE_NETWORKS.append(NETNAME) %}
|
||||
{%- endif %}
|
||||
{%- endfor %}
|
||||
|
||||
{%- set PR = [] %}
|
||||
{%- set D1 = [] %}
|
||||
{%- set D2 = [] %}
|
||||
{%- for container in NODE_CONTAINERS %}
|
||||
{%- set IP = DOCKERMERGED.containers[container].ip %}
|
||||
{%- set BRIDGE = DOCKERMERGED.containers[container].network %}
|
||||
{%- if DOCKERMERGED.containers[container].port_bindings is defined %}
|
||||
{%- for binding in DOCKERMERGED.containers[container].port_bindings %}
|
||||
{#- cant split int so we convert to string #}
|
||||
@@ -35,11 +43,11 @@
|
||||
{%- endif %}
|
||||
{%- do PR.append("-A POSTROUTING -s " ~ DOCKERMERGED.containers[container].ip ~ "/32 -d " ~ DOCKERMERGED.containers[container].ip ~ "/32 -p " ~ proto ~ " -m " ~ proto ~ " --dport " ~ containerPort ~ " -j MASQUERADE") %}
|
||||
{%- if bindip | length and bindip != '0.0.0.0' %}
|
||||
{%- do D1.append("-A DOCKER -d " ~ bindip ~ "/32 ! -i sobridge -p " ~ proto ~ " -m " ~ proto ~ " --dport " ~ hostPort ~ " -j DNAT --to-destination " ~ DOCKERMERGED.containers[container].ip ~ ":" ~ containerPort) %}
|
||||
{%- do D1.append("-A DOCKER -d " ~ bindip ~ "/32 ! -i " ~ BRIDGE ~ " -p " ~ proto ~ " -m " ~ proto ~ " --dport " ~ hostPort ~ " -j DNAT --to-destination " ~ DOCKERMERGED.containers[container].ip ~ ":" ~ containerPort) %}
|
||||
{%- else %}
|
||||
{%- do D1.append("-A DOCKER ! -i sobridge -p " ~ proto ~ " -m " ~ proto ~ " --dport " ~ hostPort ~ " -j DNAT --to-destination " ~ DOCKERMERGED.containers[container].ip ~ ":" ~ containerPort) %}
|
||||
{%- do D1.append("-A DOCKER ! -i " ~ BRIDGE ~ " -p " ~ proto ~ " -m " ~ proto ~ " --dport " ~ hostPort ~ " -j DNAT --to-destination " ~ DOCKERMERGED.containers[container].ip ~ ":" ~ containerPort) %}
|
||||
{%- endif %}
|
||||
{%- do D2.append("-A DOCKER -d " ~ DOCKERMERGED.containers[container].ip ~ "/32 ! -i sobridge -o sobridge -p " ~ proto ~ " -m " ~ proto ~ " --dport " ~ containerPort ~ " -j ACCEPT") %}
|
||||
{%- do D2.append("-A DOCKER -d " ~ DOCKERMERGED.containers[container].ip ~ "/32 ! -i " ~ BRIDGE ~ " -o " ~ BRIDGE ~ " -p " ~ proto ~ " -m " ~ proto ~ " --dport " ~ containerPort ~ " -j ACCEPT") %}
|
||||
{%- endfor %}
|
||||
{%- endif %}
|
||||
{%- endfor %}
|
||||
@@ -52,11 +60,15 @@
|
||||
:DOCKER - [0:0]
|
||||
-A PREROUTING -m addrtype --dst-type LOCAL -j DOCKER
|
||||
-A OUTPUT ! -d 127.0.0.0/8 -m addrtype --dst-type LOCAL -j DOCKER
|
||||
-A POSTROUTING -s {{DOCKERMERGED.range}} ! -o sobridge -j MASQUERADE
|
||||
{%- for NETNAME in NODE_NETWORKS %}
|
||||
-A POSTROUTING -s {{ DOCKERMERGED.networks[NETNAME].range }} ! -o {{ NETNAME }} -j MASQUERADE
|
||||
{%- endfor %}
|
||||
{%- for rule in PR %}
|
||||
{{ rule }}
|
||||
{%- endfor %}
|
||||
-A DOCKER -i sobridge -j RETURN
|
||||
{%- for NETNAME in NODE_NETWORKS %}
|
||||
-A DOCKER -i {{ NETNAME }} -j RETURN
|
||||
{%- endfor %}
|
||||
{%- for rule in D1 %}
|
||||
{{ rule }}
|
||||
{%- endfor %}
|
||||
@@ -97,10 +109,12 @@ COMMIT
|
||||
{%- endif %}
|
||||
-A FORWARD -j DOCKER-USER
|
||||
-A FORWARD -j DOCKER-ISOLATION-STAGE-1
|
||||
-A FORWARD -o sobridge -m conntrack --ctstate RELATED,ESTABLISHED -j ACCEPT
|
||||
-A FORWARD -o sobridge -j DOCKER
|
||||
-A FORWARD -i sobridge ! -o sobridge -j ACCEPT
|
||||
-A FORWARD -i sobridge -o sobridge -j ACCEPT
|
||||
{%- for NETNAME in NODE_NETWORKS %}
|
||||
-A FORWARD -o {{ NETNAME }} -m conntrack --ctstate RELATED,ESTABLISHED -j ACCEPT
|
||||
-A FORWARD -o {{ NETNAME }} -j DOCKER
|
||||
-A FORWARD -i {{ NETNAME }} ! -o {{ NETNAME }} -j ACCEPT
|
||||
-A FORWARD -i {{ NETNAME }} -o {{ NETNAME }} -j ACCEPT
|
||||
{%- endfor %}
|
||||
-A FORWARD -m conntrack --ctstate RELATED,ESTABLISHED -j ACCEPT
|
||||
-A FORWARD -i lo -j ACCEPT
|
||||
-A FORWARD -m conntrack --ctstate INVALID -j DROP
|
||||
@@ -112,13 +126,18 @@ COMMIT
|
||||
{%- for rule in D2 %}
|
||||
{{ rule }}
|
||||
{%- endfor %}
|
||||
|
||||
-A DOCKER-ISOLATION-STAGE-1 -i sobridge ! -o sobridge -j DOCKER-ISOLATION-STAGE-2
|
||||
{% for NETNAME in NODE_NETWORKS %}
|
||||
-A DOCKER-ISOLATION-STAGE-1 -i {{ NETNAME }} ! -o {{ NETNAME }} -j DOCKER-ISOLATION-STAGE-2
|
||||
{%- endfor %}
|
||||
-A DOCKER-ISOLATION-STAGE-1 -j RETURN
|
||||
-A DOCKER-ISOLATION-STAGE-2 -o sobridge -j DROP
|
||||
{%- for NETNAME in NODE_NETWORKS %}
|
||||
-A DOCKER-ISOLATION-STAGE-2 -o {{ NETNAME }} -j DROP
|
||||
{%- endfor %}
|
||||
-A DOCKER-ISOLATION-STAGE-2 -j RETURN
|
||||
-A DOCKER-USER ! -i sobridge -o sobridge -m conntrack --ctstate RELATED,ESTABLISHED -j ACCEPT
|
||||
-A DOCKER-USER ! -i sobridge -o sobridge -j LOGGING
|
||||
{%- for NETNAME in NODE_NETWORKS %}
|
||||
-A DOCKER-USER ! -i {{ NETNAME }} -o {{ NETNAME }} -m conntrack --ctstate RELATED,ESTABLISHED -j ACCEPT
|
||||
-A DOCKER-USER ! -i {{ NETNAME }} -o {{ NETNAME }} -j LOGGING
|
||||
{%- endfor %}
|
||||
-A DOCKER-USER -j RETURN
|
||||
-A LOGGING -m limit --limit 2/min -j LOG --log-prefix "IPTables-dropped: "
|
||||
-A LOGGING -j DROP
|
||||
|
||||
@@ -4,8 +4,12 @@
|
||||
|
||||
{# add our ip to self #}
|
||||
{% do FIREWALL_DEFAULT.firewall.hostgroups.self.append(GLOBALS.node_ip) %}
|
||||
{# add dockernet range #}
|
||||
{% do FIREWALL_DEFAULT.firewall.hostgroups.dockernet.append(DOCKERMERGED.range) %}
|
||||
{# add dockernet ranges #}
|
||||
{% for NETNAME, NETWORK in DOCKERMERGED.networks.items() %}
|
||||
{% if not NETWORK.get('manager_only') or GLOBALS.get('is_manager', False) %}
|
||||
{% do FIREWALL_DEFAULT.firewall.hostgroups.dockernet.append(NETWORK.range) %}
|
||||
{% endif %}
|
||||
{% endfor %}
|
||||
|
||||
{% if GLOBALS.role == 'so-idh' %}
|
||||
{% from 'idh/opencanary_config.map.jinja' import IDH_PORTGROUPS %}
|
||||
|
||||
@@ -1,3 +1,3 @@
|
||||
global:
|
||||
pcapengine: SURICATA
|
||||
pipeline: REDIS
|
||||
pipeline: REDIS
|
||||
|
||||
@@ -14,9 +14,6 @@
|
||||
{% from 'docker/docker.map.jinja' import DOCKERMERGED %}
|
||||
{% from 'vars/globals.map.jinja' import GLOBALS %}
|
||||
{% if 'api' in salt['pillar.get']('features', []) %}
|
||||
{% from 'docker/macros/docker_endpoint.jinja' import clear_stale_endpoint %}
|
||||
|
||||
{{ clear_stale_endpoint('so-hydra', 'sobridge', DOCKERMERGED.containers['so-hydra'].ip) }}
|
||||
|
||||
include:
|
||||
- hydra.config
|
||||
@@ -25,11 +22,12 @@ include:
|
||||
so-hydra:
|
||||
docker_container.running:
|
||||
- image: {{ GLOBALS.registry_host }}:5000/{{ GLOBALS.image_repo }}/so-hydra:{{ GLOBALS.so_version }}
|
||||
- restart_policy: unless-stopped
|
||||
- hostname: hydra
|
||||
- name: so-hydra
|
||||
- networks:
|
||||
- sobridge:
|
||||
- ipv4_address: {{ DOCKERMERGED.containers['so-hydra'].ip }}
|
||||
- soauth:
|
||||
- ipv4_address: {{ DOCKERMERGED.containers['so-hydra'].ips['soauth'] }}
|
||||
- binds:
|
||||
- /opt/so/conf/hydra/:/hydra-conf:ro
|
||||
- /opt/so/log/hydra/:/hydra-log:rw
|
||||
@@ -61,7 +59,6 @@ so-hydra:
|
||||
- {{ ULIMIT.name }}={{ ULIMIT.soft }}:{{ ULIMIT.hard }}
|
||||
{% endfor %}
|
||||
{% endif %}
|
||||
- restart_policy: unless-stopped
|
||||
- watch:
|
||||
- file: hydraconfig
|
||||
- require:
|
||||
@@ -76,7 +73,7 @@ delete_so-hydra_so-status.disabled:
|
||||
|
||||
wait_for_hydra:
|
||||
http.wait_for_successful_query:
|
||||
- name: 'http://{{ GLOBALS.manager }}:4444/health/alive'
|
||||
- name: 'http://{{ DOCKERMERGED.containers['so-hydra'].ips['soauth'] }}:4444/health/alive'
|
||||
- ssl: True
|
||||
- verify_ssl: False
|
||||
- status:
|
||||
|
||||
@@ -21,12 +21,16 @@ hypervisor_sbin:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://hypervisor/tools/sbin
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 744
|
||||
|
||||
hypervisor_sbin_jinja:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://hypervisor/tools/sbin_jinja
|
||||
- user: root
|
||||
- group: root
|
||||
- template: jinja
|
||||
- file_mode: 744
|
||||
|
||||
|
||||
+2
-2
@@ -86,8 +86,8 @@ idh_sbin:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://idh/tools/sbin
|
||||
- user: 939
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
|
||||
#idh_sbin_jinja:
|
||||
|
||||
@@ -15,6 +15,7 @@ include:
|
||||
so-idh:
|
||||
docker_container.running:
|
||||
- image: {{ GLOBALS.registry_host }}:5000/{{ GLOBALS.image_repo }}/so-idh:{{ GLOBALS.so_version }}
|
||||
- restart_policy: unless-stopped
|
||||
- name: so-idh
|
||||
- detach: True
|
||||
- network_mode: host
|
||||
|
||||
@@ -41,8 +41,8 @@ influxdb_sbin:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://influxdb/tools/sbin
|
||||
- user: 939
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
|
||||
#influxdb_sbin_jinja:
|
||||
|
||||
@@ -9,9 +9,6 @@
|
||||
{% from 'docker/docker.map.jinja' import DOCKERMERGED %}
|
||||
{% set PASSWORD = salt['pillar.get']('secrets:influx_pass') %}
|
||||
{% set TOKEN = salt['pillar.get']('influxdb:token') %}
|
||||
{% from 'docker/macros/docker_endpoint.jinja' import clear_stale_endpoint %}
|
||||
|
||||
{{ clear_stale_endpoint('so-influxdb', 'sobridge', DOCKERMERGED.containers['so-influxdb'].ip) }}
|
||||
|
||||
include:
|
||||
- influxdb.ssl
|
||||
@@ -21,6 +18,7 @@ include:
|
||||
so-influxdb:
|
||||
docker_container.running:
|
||||
- image: {{ GLOBALS.registry_host }}:5000/{{ GLOBALS.image_repo }}/so-influxdb:{{ GLOBALS.so_version }}
|
||||
- restart_policy: unless-stopped
|
||||
- hostname: influxdb
|
||||
- networks:
|
||||
- sobridge:
|
||||
@@ -96,9 +94,11 @@ metrics_link_file:
|
||||
- docker_container: so-influxdb
|
||||
|
||||
# Install cron job to determine size of influxdb for telegraf
|
||||
# telegraf reads this while the cron rewrites it, so write aside and rename rather than
|
||||
# truncating in place. tgraflogdir recurses ownership, so the temp file is chowned to match
|
||||
get_influxdb_size:
|
||||
cron.present:
|
||||
- name: 'du -s -k /nsm/influxdb | cut -f1 > /opt/so/log/telegraf/influxdb_size.log 2>&1'
|
||||
- name: 'du -s -k /nsm/influxdb | cut -f1 > /opt/so/log/telegraf/influxdb_size.log.tmp 2>&1; chown 939:939 /opt/so/log/telegraf/influxdb_size.log.tmp; mv -f /opt/so/log/telegraf/influxdb_size.log.tmp /opt/so/log/telegraf/influxdb_size.log'
|
||||
- identifier: get_influxdb_size
|
||||
- user: root
|
||||
- minute: '*/1'
|
||||
|
||||
@@ -30,16 +30,16 @@ kafka_sbin_tools:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://kafka/tools/sbin
|
||||
- user: 960
|
||||
- group: 960
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
|
||||
kafka_sbin_jinja_tools:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://kafka/tools/sbin_jinja
|
||||
- user: 960
|
||||
- group: 960
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
- template: jinja
|
||||
- defaults:
|
||||
|
||||
@@ -16,9 +16,6 @@
|
||||
{% set KAFKANODES = salt['pillar.get']('kafka:nodes') %}
|
||||
{% set KAFKA_EXTERNAL_ACCESS = salt['pillar.get']('kafka:config:external_access:enabled', default=False) %}
|
||||
{% if 'gmd' in salt['pillar.get']('features', []) %}
|
||||
{% from 'docker/macros/docker_endpoint.jinja' import clear_stale_endpoint %}
|
||||
|
||||
{{ clear_stale_endpoint('so-kafka', 'sobridge', DOCKERMERGED.containers['so-kafka'].ip) }}
|
||||
|
||||
include:
|
||||
- kafka.ca
|
||||
@@ -30,6 +27,7 @@ include:
|
||||
so-kafka:
|
||||
docker_container.running:
|
||||
- image: {{ GLOBALS.registry_host }}:5000/{{ GLOBALS.image_repo }}/so-kafka:{{ GLOBALS.so_version }}
|
||||
- restart_policy: unless-stopped
|
||||
- hostname: so-kafka
|
||||
- name: so-kafka
|
||||
- networks:
|
||||
|
||||
@@ -36,16 +36,16 @@ kibana_sbin:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://kibana/tools/sbin
|
||||
- user: 932
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
|
||||
kibana_sbin_jinja:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://kibana/tools/sbin_jinja
|
||||
- user: 932
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
- template: jinja
|
||||
- defaults:
|
||||
|
||||
@@ -22,7 +22,7 @@ kibana:
|
||||
- default
|
||||
- file
|
||||
migrations:
|
||||
discardCorruptObjects: "9.3.3"
|
||||
discardCorruptObjects: "9.4.5"
|
||||
telemetry:
|
||||
enabled: False
|
||||
xpack:
|
||||
|
||||
@@ -8,9 +8,6 @@
|
||||
{% from 'docker/docker.map.jinja' import DOCKERMERGED %}
|
||||
{% from 'elasticsearch/config.map.jinja' import ELASTICSEARCHMERGED %}
|
||||
{% from 'vars/globals.map.jinja' import GLOBALS %}
|
||||
{% from 'docker/macros/docker_endpoint.jinja' import clear_stale_endpoint %}
|
||||
|
||||
{{ clear_stale_endpoint('so-kibana', 'sobridge', DOCKERMERGED.containers['so-kibana'].ip) }}
|
||||
|
||||
include:
|
||||
- kibana.config
|
||||
@@ -20,6 +17,7 @@ include:
|
||||
so-kibana:
|
||||
docker_container.running:
|
||||
- image: {{ GLOBALS.registry_host }}:5000/{{ GLOBALS.image_repo }}/so-kibana:{{ GLOBALS.so_version }}
|
||||
- restart_policy: unless-stopped
|
||||
- hostname: kibana
|
||||
- user: "932:0"
|
||||
- networks:
|
||||
|
||||
File diff suppressed because one or more lines are too long
@@ -9,5 +9,5 @@ SESSIONCOOKIE=$(curl -K /opt/so/conf/elasticsearch/curl.config -c - -X GET http:
|
||||
# Disable certain Features from showing up in the Kibana UI
|
||||
echo
|
||||
echo "Setting up default Kibana Space:"
|
||||
curl -K /opt/so/conf/elasticsearch/curl.config -b "sid=$SESSIONCOOKIE" -L -X PUT "localhost:5601/api/spaces/space/default" -H 'kbn-xsrf: true' -H 'Content-Type: application/json' -d' {"id":"default","name":"Default","disabledFeatures":["ml","enterpriseSearch","logs","infrastructure","apm","uptime","monitoring","stackAlerts","actions","securitySolutionCasesV3","inventory","dataQuality","searchSynonyms","searchQueryRules","enterpriseSearchApplications","enterpriseSearchAnalytics","securitySolutionTimeline","securitySolutionNotes","securitySolutionRulesV1","entityManager","streams","cloudConnect","slo"]} ' >> /opt/so/log/kibana/misc.log
|
||||
curl -K /opt/so/conf/elasticsearch/curl.config -b "sid=$SESSIONCOOKIE" -L -X PUT "localhost:5601/api/spaces/space/default" -H 'kbn-xsrf: true' -H 'Content-Type: application/json' -d' {"id":"default","name":"Default","disabledFeatures":["ml","enterpriseSearch","logs","infrastructure","apm","uptime","securitySolutionCasesV3","inventory","searchSynonyms","searchQueryRules","enterpriseSearchApplications","enterpriseSearchAnalytics","securitySolutionTimeline","securitySolutionNotes","securitySolutionRulesV4","securitySolutionAlertsV1","entityManager","slo","streams","anonymization","searchInferenceEndpoints","cloudConnect","queryActivity","automatic_import","stackAlerts","monitoring","dataQuality","actions"]} ' >> /opt/so/log/kibana/misc.log
|
||||
echo
|
||||
|
||||
@@ -7,9 +7,6 @@
|
||||
{% if sls.split('.')[0] in allowed_states %}
|
||||
{% from 'docker/docker.map.jinja' import DOCKERMERGED %}
|
||||
{% from 'vars/globals.map.jinja' import GLOBALS %}
|
||||
{% from 'docker/macros/docker_endpoint.jinja' import clear_stale_endpoint %}
|
||||
|
||||
{{ clear_stale_endpoint('so-kratos', 'sobridge', DOCKERMERGED.containers['so-kratos'].ip) }}
|
||||
|
||||
include:
|
||||
- kratos.config
|
||||
@@ -18,11 +15,12 @@ include:
|
||||
so-kratos:
|
||||
docker_container.running:
|
||||
- image: {{ GLOBALS.registry_host }}:5000/{{ GLOBALS.image_repo }}/so-kratos:{{ GLOBALS.so_version }}
|
||||
- restart_policy: unless-stopped
|
||||
- hostname: kratos
|
||||
- name: so-kratos
|
||||
- networks:
|
||||
- sobridge:
|
||||
- ipv4_address: {{ DOCKERMERGED.containers['so-kratos'].ip }}
|
||||
- soauth:
|
||||
- ipv4_address: {{ DOCKERMERGED.containers['so-kratos'].ips['soauth'] }}
|
||||
- binds:
|
||||
- /opt/so/conf/kratos/:/kratos-conf:ro
|
||||
- /opt/so/log/kratos/:/kratos-log:rw
|
||||
@@ -54,7 +52,6 @@ so-kratos:
|
||||
- {{ ULIMIT.name }}={{ ULIMIT.soft }}:{{ ULIMIT.hard }}
|
||||
{% endfor %}
|
||||
{% endif %}
|
||||
- restart_policy: unless-stopped
|
||||
- watch:
|
||||
- file: kratosschema
|
||||
- file: kratosconfig
|
||||
@@ -74,7 +71,7 @@ delete_so-kratos_so-status.disabled:
|
||||
|
||||
wait_for_kratos:
|
||||
http.wait_for_successful_query:
|
||||
- name: 'http://{{ GLOBALS.manager }}:4434/'
|
||||
- name: 'http://{{ DOCKERMERGED.containers['so-kratos'].ips['soauth'] }}:4434/'
|
||||
- ssl: True
|
||||
- verify_ssl: False
|
||||
- status:
|
||||
|
||||
@@ -6,6 +6,8 @@ so-fix-salt-ldap_script:
|
||||
file.managed:
|
||||
- name: /usr/sbin/so-fix-salt-ldap.py
|
||||
- source: salt://libvirt/64962/scripts/so-fix-salt-ldap.py
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 744
|
||||
|
||||
fix-salt-ldap:
|
||||
|
||||
@@ -0,0 +1,67 @@
|
||||
# This state is designed to run on a development manager running in a libvirt VM. It will map the default pillar and salt directories
|
||||
# from /opt/so/saltstack/default to your local development machine as the source path.
|
||||
# The VM requires a filesystem to be added. Only the source path should be changed to your development codebase
|
||||
# Driver: virtio-9p
|
||||
# Source path: ~/project/securityonion
|
||||
# Target path: saltDev
|
||||
|
||||
# If you want a directory to be RW, then kvm must have group privileges.
|
||||
# ll /home/user/projects/securityonion/salt/hypervisor
|
||||
# total 48
|
||||
# drwxrwxr-x 3 user kvm 4096 Feb 13 11:18 ./
|
||||
# drwxrwxr-x 64 user user 4096 Feb 13 10:32 ../
|
||||
# -rw-rw-r-- 1 user kvm 2238 Feb 12 15:06 defaults.yaml
|
||||
# -rw-rw-r-- 1 user kvm 1467 Feb 12 15:06 init.sls
|
||||
# -rw-rw-r-- 1 user kvm 70 Feb 13 09:37 soc_hypervisor.yaml
|
||||
# drwxrwxr-x 3 user kvm 4096 Feb 12 15:06 tools/
|
||||
|
||||
# Ensure required kernel modules are configured for loading
|
||||
/etc/modules-load.d/virtio-9p.conf:
|
||||
file.managed:
|
||||
- contents: |
|
||||
9pnet_virtio
|
||||
9pnet
|
||||
9p
|
||||
- mode: 644
|
||||
- user: root
|
||||
- group: root
|
||||
|
||||
# Load the kernel modules immediately (in the correct order)
|
||||
load_9p_modules:
|
||||
cmd.run:
|
||||
- names:
|
||||
- modprobe 9pnet_virtio
|
||||
- modprobe 9pnet
|
||||
- modprobe 9p
|
||||
- unless: lsmod | grep -E '9pnet_virtio|9pnet|9p'
|
||||
|
||||
# Ensure mount point exists
|
||||
/opt/so/saltstack/default:
|
||||
file.directory:
|
||||
- user: root
|
||||
- group: root
|
||||
- mode: 755
|
||||
- makedirs: True
|
||||
|
||||
# Configure fstab entry using mount.fstab_present
|
||||
# Configure fstab entry using mount.fstab_present
|
||||
saltdev_fstab:
|
||||
mount.fstab_present:
|
||||
- name: saltDev
|
||||
- fs_file: /opt/so/saltstack/default
|
||||
- fs_vfstype: 9p
|
||||
- fs_mntops: _netdev,trans=virtio,version=9p2000.L
|
||||
- fs_freq: 0
|
||||
- fs_passno: 0
|
||||
|
||||
# Mount the filesystem if not already mounted
|
||||
mount_saltdev:
|
||||
mount.mounted:
|
||||
- name: /opt/so/saltstack/default
|
||||
- device: saltDev
|
||||
- fstype: 9p
|
||||
- opts: _netdev,trans=virtio,version=9p2000.L
|
||||
- require:
|
||||
- file: /opt/so/saltstack/default
|
||||
- mount: saltdev_fstab
|
||||
- cmd: load_9p_modules
|
||||
@@ -150,6 +150,16 @@ logrotate:
|
||||
- extension .log
|
||||
- dateext
|
||||
- dateyesterday
|
||||
/opt/so/log/postgres/*_x_log:
|
||||
- daily
|
||||
- rotate 14
|
||||
- missingok
|
||||
- copytruncate
|
||||
- compress
|
||||
- create
|
||||
- extension .log
|
||||
- dateext
|
||||
- dateyesterday
|
||||
/opt/so/log/telegraf/*_x_log:
|
||||
- daily
|
||||
- rotate 14
|
||||
@@ -210,6 +220,36 @@ logrotate:
|
||||
- extension .log
|
||||
- dateext
|
||||
- dateyesterday
|
||||
/opt/so/log/salt/virtual_node_manager:
|
||||
- daily
|
||||
- rotate 14
|
||||
- missingok
|
||||
- copytruncate
|
||||
- compress
|
||||
- create
|
||||
- extension .log
|
||||
- dateext
|
||||
- dateyesterday
|
||||
/opt/so/log/salt/so-salt-cloud:
|
||||
- daily
|
||||
- rotate 14
|
||||
- missingok
|
||||
- copytruncate
|
||||
- compress
|
||||
- create
|
||||
- extension .log
|
||||
- dateext
|
||||
- dateyesterday
|
||||
/opt/so/log/salt/so-soup-grid-highstate:
|
||||
- daily
|
||||
- rotate 14
|
||||
- missingok
|
||||
- copytruncate
|
||||
- compress
|
||||
- create
|
||||
- extension .log
|
||||
- dateext
|
||||
- dateyesterday
|
||||
/nsm/idh/*_x_log:
|
||||
- daily
|
||||
- rotate 14
|
||||
|
||||
@@ -91,6 +91,13 @@ logrotate:
|
||||
multiline: True
|
||||
global: True
|
||||
forcedType: "[]string"
|
||||
"/opt/so/log/postgres/*_x_log":
|
||||
description: List of logrotate options for this file.
|
||||
title: /opt/so/log/postgres/*.log
|
||||
advanced: True
|
||||
multiline: True
|
||||
global: True
|
||||
forcedType: "[]string"
|
||||
"/opt/so/log/telegraf/*_x_log":
|
||||
description: List of logrotate options for this file.
|
||||
title: /opt/so/log/telegraf/*.log
|
||||
@@ -133,6 +140,27 @@ logrotate:
|
||||
multiline: True
|
||||
global: True
|
||||
forcedType: "[]string"
|
||||
"/opt/so/log/salt/virtual_node_manager":
|
||||
description: List of logrotate options for this file.
|
||||
title: /opt/so/log/salt/virtual_node_manager
|
||||
advanced: True
|
||||
multiline: True
|
||||
global: True
|
||||
forcedType: "[]string"
|
||||
"/opt/so/log/salt/so-salt-cloud":
|
||||
description: List of logrotate options for this file.
|
||||
title: /opt/so/log/salt/so-salt-cloud
|
||||
advanced: True
|
||||
multiline: True
|
||||
global: True
|
||||
forcedType: "[]string"
|
||||
"/opt/so/log/salt/so-soup-grid-highstate":
|
||||
description: List of logrotate options for this file.
|
||||
title: /opt/so/log/salt/so-soup-grid-highstate
|
||||
advanced: True
|
||||
multiline: True
|
||||
global: True
|
||||
forcedType: "[]string"
|
||||
"/nsm/idh/*_x_log":
|
||||
description: List of logrotate options for this file.
|
||||
title: /nsm/idh/*.log
|
||||
|
||||
@@ -40,8 +40,8 @@ logstash_sbin:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://logstash/tools/sbin
|
||||
- user: 931
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
|
||||
#logstash_sbin_jinja:
|
||||
@@ -81,6 +81,14 @@ ls_custom_pipeline_conf_{{assigned_pipeline}}_{{pipeline}}:
|
||||
|
||||
|
||||
{% for assigned_pipeline in ASSIGNED_PIPELINES %}
|
||||
{# a blank per-pipeline setting falls back to the global logstash.yml value #}
|
||||
{% set PARSED_OVERRIDES = LOGSTASH_MERGED.get('pipeline_settings', {}).get(assigned_pipeline, {}) %}
|
||||
{% if PARSED_OVERRIDES is not mapping %}
|
||||
{% do salt.log.warning('logstash: ignoring malformed pipeline_settings for pipeline ' ~ assigned_pipeline ~ '; expected a set of settings') %}
|
||||
{% endif %}
|
||||
{% set PIPELINE_OVERRIDES = PARSED_OVERRIDES if PARSED_OVERRIDES is mapping else {} %}
|
||||
{% set THREADS = PIPELINE_OVERRIDES.get('pipeline_x_workers') or LOGSTASH_MERGED.config.pipeline_x_workers %}
|
||||
{% set BATCH = PIPELINE_OVERRIDES.get('pipeline_x_batch_x_size') or LOGSTASH_MERGED.config.pipeline_x_batch_x_size %}
|
||||
{% for CONFIGFILE in LOGSTASH_MERGED.defined_pipelines[assigned_pipeline] %}
|
||||
ls_pipeline_{{assigned_pipeline}}_{{CONFIGFILE.split('.')[0] | replace("/","_") }}:
|
||||
file.managed:
|
||||
@@ -92,8 +100,8 @@ ls_pipeline_{{assigned_pipeline}}_{{CONFIGFILE.split('.')[0] | replace("/","_")
|
||||
GLOBALS: {{ GLOBALS }}
|
||||
ES_USER: "{{ salt['pillar.get']('elasticsearch:auth:users:so_elastic_user:user', '') }}"
|
||||
ES_PASS: "{{ salt['pillar.get']('elasticsearch:auth:users:so_elastic_user:pass', '') }}"
|
||||
THREADS: {{ LOGSTASH_MERGED.config.pipeline_x_workers }}
|
||||
BATCH: {{ LOGSTASH_MERGED.config.pipeline_x_batch_x_size }}
|
||||
THREADS: {{ THREADS }}
|
||||
BATCH: {{ BATCH }}
|
||||
{% else %}
|
||||
- name: /opt/so/conf/logstash/pipelines/{{assigned_pipeline}}/{{CONFIGFILE.split('/')[1]}}
|
||||
{% endif %}
|
||||
@@ -125,6 +133,14 @@ lspipelinesyml:
|
||||
- defaults:
|
||||
ASSIGNED_PIPELINES: {{ ASSIGNED_PIPELINES }}
|
||||
|
||||
lslog4j2:
|
||||
file.managed:
|
||||
- name: /opt/so/conf/logstash/etc/log4j2.properties
|
||||
- source: salt://logstash/etc/log4j2.properties.jinja
|
||||
- template: jinja
|
||||
- user: 931
|
||||
- group: 939
|
||||
|
||||
lsetcsync:
|
||||
file.recurse:
|
||||
- name: /opt/so/conf/logstash/etc
|
||||
@@ -133,7 +149,11 @@ lsetcsync:
|
||||
- group: 939
|
||||
- template: jinja
|
||||
- clean: True
|
||||
- exclude_pat: pipelines*
|
||||
{#- both names are matched: the .jinja source so the recurse does not copy it verbatim,
|
||||
and the rendered file so clean: True does not delete what lslog4j2 wrote #}
|
||||
- exclude_pat:
|
||||
- pipelines*
|
||||
- log4j2.properties*
|
||||
- defaults:
|
||||
LOGSTASH_MERGED: {{ LOGSTASH_MERGED }}
|
||||
|
||||
|
||||
@@ -42,6 +42,11 @@ logstash:
|
||||
custom2: []
|
||||
custom3: []
|
||||
custom4: []
|
||||
custom5: []
|
||||
custom6: []
|
||||
custom7: []
|
||||
custom8: []
|
||||
custom9: []
|
||||
pipeline_config:
|
||||
custom001: |-
|
||||
filter {
|
||||
@@ -60,10 +65,405 @@ logstash:
|
||||
custom008: PLACEHOLDER
|
||||
custom009: PLACEHOLDER
|
||||
custom010: PLACEHOLDER
|
||||
pipeline_settings:
|
||||
fleet:
|
||||
pipeline_x_workers: ''
|
||||
pipeline_x_batch_x_size: ''
|
||||
pipeline_x_batch_x_delay: ''
|
||||
pipeline_x_batch_x_metrics_x_sampling_mode: ''
|
||||
pipeline_x_ordered: ''
|
||||
pipeline_x_ecs_compatibility: ''
|
||||
pipeline_x_reloadable: ''
|
||||
queue_x_type: ''
|
||||
queue_x_max_bytes: ''
|
||||
queue_x_page_capacity: ''
|
||||
queue_x_max_events: ''
|
||||
queue_x_checkpoint_x_acks: ''
|
||||
queue_x_checkpoint_x_writes: ''
|
||||
queue_x_checkpoint_x_interval: ''
|
||||
queue_x_checkpoint_x_retry: ''
|
||||
queue_x_compression: ''
|
||||
queue_x_drain: ''
|
||||
dead_letter_queue_x_enable: ''
|
||||
dead_letter_queue_x_max_bytes: ''
|
||||
dead_letter_queue_x_flush_interval: ''
|
||||
dead_letter_queue_x_flush_check_interval: ''
|
||||
dead_letter_queue_x_storage_policy: ''
|
||||
dead_letter_queue_x_retain_x_age: ''
|
||||
path_x_queue: ''
|
||||
path_x_dead_letter_queue: ''
|
||||
config_x_debug: ''
|
||||
config_x_support_escapes: ''
|
||||
manager:
|
||||
pipeline_x_workers: ''
|
||||
pipeline_x_batch_x_size: ''
|
||||
pipeline_x_batch_x_delay: ''
|
||||
pipeline_x_batch_x_metrics_x_sampling_mode: ''
|
||||
pipeline_x_ordered: ''
|
||||
pipeline_x_ecs_compatibility: ''
|
||||
pipeline_x_reloadable: ''
|
||||
queue_x_type: ''
|
||||
queue_x_max_bytes: ''
|
||||
queue_x_page_capacity: ''
|
||||
queue_x_max_events: ''
|
||||
queue_x_checkpoint_x_acks: ''
|
||||
queue_x_checkpoint_x_writes: ''
|
||||
queue_x_checkpoint_x_interval: ''
|
||||
queue_x_checkpoint_x_retry: ''
|
||||
queue_x_compression: ''
|
||||
queue_x_drain: ''
|
||||
dead_letter_queue_x_enable: ''
|
||||
dead_letter_queue_x_max_bytes: ''
|
||||
dead_letter_queue_x_flush_interval: ''
|
||||
dead_letter_queue_x_flush_check_interval: ''
|
||||
dead_letter_queue_x_storage_policy: ''
|
||||
dead_letter_queue_x_retain_x_age: ''
|
||||
path_x_queue: ''
|
||||
path_x_dead_letter_queue: ''
|
||||
config_x_debug: ''
|
||||
config_x_support_escapes: ''
|
||||
receiver:
|
||||
pipeline_x_workers: ''
|
||||
pipeline_x_batch_x_size: ''
|
||||
pipeline_x_batch_x_delay: ''
|
||||
pipeline_x_batch_x_metrics_x_sampling_mode: ''
|
||||
pipeline_x_ordered: ''
|
||||
pipeline_x_ecs_compatibility: ''
|
||||
pipeline_x_reloadable: ''
|
||||
queue_x_type: ''
|
||||
queue_x_max_bytes: ''
|
||||
queue_x_page_capacity: ''
|
||||
queue_x_max_events: ''
|
||||
queue_x_checkpoint_x_acks: ''
|
||||
queue_x_checkpoint_x_writes: ''
|
||||
queue_x_checkpoint_x_interval: ''
|
||||
queue_x_checkpoint_x_retry: ''
|
||||
queue_x_compression: ''
|
||||
queue_x_drain: ''
|
||||
dead_letter_queue_x_enable: ''
|
||||
dead_letter_queue_x_max_bytes: ''
|
||||
dead_letter_queue_x_flush_interval: ''
|
||||
dead_letter_queue_x_flush_check_interval: ''
|
||||
dead_letter_queue_x_storage_policy: ''
|
||||
dead_letter_queue_x_retain_x_age: ''
|
||||
path_x_queue: ''
|
||||
path_x_dead_letter_queue: ''
|
||||
config_x_debug: ''
|
||||
config_x_support_escapes: ''
|
||||
search:
|
||||
pipeline_x_workers: ''
|
||||
pipeline_x_batch_x_size: ''
|
||||
pipeline_x_batch_x_delay: ''
|
||||
pipeline_x_batch_x_metrics_x_sampling_mode: ''
|
||||
pipeline_x_ordered: ''
|
||||
pipeline_x_ecs_compatibility: ''
|
||||
pipeline_x_reloadable: ''
|
||||
queue_x_type: ''
|
||||
queue_x_max_bytes: ''
|
||||
queue_x_page_capacity: ''
|
||||
queue_x_max_events: ''
|
||||
queue_x_checkpoint_x_acks: ''
|
||||
queue_x_checkpoint_x_writes: ''
|
||||
queue_x_checkpoint_x_interval: ''
|
||||
queue_x_checkpoint_x_retry: ''
|
||||
queue_x_compression: ''
|
||||
queue_x_drain: ''
|
||||
dead_letter_queue_x_enable: ''
|
||||
dead_letter_queue_x_max_bytes: ''
|
||||
dead_letter_queue_x_flush_interval: ''
|
||||
dead_letter_queue_x_flush_check_interval: ''
|
||||
dead_letter_queue_x_storage_policy: ''
|
||||
dead_letter_queue_x_retain_x_age: ''
|
||||
path_x_queue: ''
|
||||
path_x_dead_letter_queue: ''
|
||||
config_x_debug: ''
|
||||
config_x_support_escapes: ''
|
||||
custom0:
|
||||
pipeline_x_workers: ''
|
||||
pipeline_x_batch_x_size: ''
|
||||
pipeline_x_batch_x_delay: ''
|
||||
pipeline_x_batch_x_metrics_x_sampling_mode: ''
|
||||
pipeline_x_ordered: ''
|
||||
pipeline_x_ecs_compatibility: ''
|
||||
pipeline_x_reloadable: ''
|
||||
queue_x_type: ''
|
||||
queue_x_max_bytes: ''
|
||||
queue_x_page_capacity: ''
|
||||
queue_x_max_events: ''
|
||||
queue_x_checkpoint_x_acks: ''
|
||||
queue_x_checkpoint_x_writes: ''
|
||||
queue_x_checkpoint_x_interval: ''
|
||||
queue_x_checkpoint_x_retry: ''
|
||||
queue_x_compression: ''
|
||||
queue_x_drain: ''
|
||||
dead_letter_queue_x_enable: ''
|
||||
dead_letter_queue_x_max_bytes: ''
|
||||
dead_letter_queue_x_flush_interval: ''
|
||||
dead_letter_queue_x_flush_check_interval: ''
|
||||
dead_letter_queue_x_storage_policy: ''
|
||||
dead_letter_queue_x_retain_x_age: ''
|
||||
path_x_queue: ''
|
||||
path_x_dead_letter_queue: ''
|
||||
config_x_debug: ''
|
||||
config_x_support_escapes: ''
|
||||
custom1:
|
||||
pipeline_x_workers: ''
|
||||
pipeline_x_batch_x_size: ''
|
||||
pipeline_x_batch_x_delay: ''
|
||||
pipeline_x_batch_x_metrics_x_sampling_mode: ''
|
||||
pipeline_x_ordered: ''
|
||||
pipeline_x_ecs_compatibility: ''
|
||||
pipeline_x_reloadable: ''
|
||||
queue_x_type: ''
|
||||
queue_x_max_bytes: ''
|
||||
queue_x_page_capacity: ''
|
||||
queue_x_max_events: ''
|
||||
queue_x_checkpoint_x_acks: ''
|
||||
queue_x_checkpoint_x_writes: ''
|
||||
queue_x_checkpoint_x_interval: ''
|
||||
queue_x_checkpoint_x_retry: ''
|
||||
queue_x_compression: ''
|
||||
queue_x_drain: ''
|
||||
dead_letter_queue_x_enable: ''
|
||||
dead_letter_queue_x_max_bytes: ''
|
||||
dead_letter_queue_x_flush_interval: ''
|
||||
dead_letter_queue_x_flush_check_interval: ''
|
||||
dead_letter_queue_x_storage_policy: ''
|
||||
dead_letter_queue_x_retain_x_age: ''
|
||||
path_x_queue: ''
|
||||
path_x_dead_letter_queue: ''
|
||||
config_x_debug: ''
|
||||
config_x_support_escapes: ''
|
||||
custom2:
|
||||
pipeline_x_workers: ''
|
||||
pipeline_x_batch_x_size: ''
|
||||
pipeline_x_batch_x_delay: ''
|
||||
pipeline_x_batch_x_metrics_x_sampling_mode: ''
|
||||
pipeline_x_ordered: ''
|
||||
pipeline_x_ecs_compatibility: ''
|
||||
pipeline_x_reloadable: ''
|
||||
queue_x_type: ''
|
||||
queue_x_max_bytes: ''
|
||||
queue_x_page_capacity: ''
|
||||
queue_x_max_events: ''
|
||||
queue_x_checkpoint_x_acks: ''
|
||||
queue_x_checkpoint_x_writes: ''
|
||||
queue_x_checkpoint_x_interval: ''
|
||||
queue_x_checkpoint_x_retry: ''
|
||||
queue_x_compression: ''
|
||||
queue_x_drain: ''
|
||||
dead_letter_queue_x_enable: ''
|
||||
dead_letter_queue_x_max_bytes: ''
|
||||
dead_letter_queue_x_flush_interval: ''
|
||||
dead_letter_queue_x_flush_check_interval: ''
|
||||
dead_letter_queue_x_storage_policy: ''
|
||||
dead_letter_queue_x_retain_x_age: ''
|
||||
path_x_queue: ''
|
||||
path_x_dead_letter_queue: ''
|
||||
config_x_debug: ''
|
||||
config_x_support_escapes: ''
|
||||
custom3:
|
||||
pipeline_x_workers: ''
|
||||
pipeline_x_batch_x_size: ''
|
||||
pipeline_x_batch_x_delay: ''
|
||||
pipeline_x_batch_x_metrics_x_sampling_mode: ''
|
||||
pipeline_x_ordered: ''
|
||||
pipeline_x_ecs_compatibility: ''
|
||||
pipeline_x_reloadable: ''
|
||||
queue_x_type: ''
|
||||
queue_x_max_bytes: ''
|
||||
queue_x_page_capacity: ''
|
||||
queue_x_max_events: ''
|
||||
queue_x_checkpoint_x_acks: ''
|
||||
queue_x_checkpoint_x_writes: ''
|
||||
queue_x_checkpoint_x_interval: ''
|
||||
queue_x_checkpoint_x_retry: ''
|
||||
queue_x_compression: ''
|
||||
queue_x_drain: ''
|
||||
dead_letter_queue_x_enable: ''
|
||||
dead_letter_queue_x_max_bytes: ''
|
||||
dead_letter_queue_x_flush_interval: ''
|
||||
dead_letter_queue_x_flush_check_interval: ''
|
||||
dead_letter_queue_x_storage_policy: ''
|
||||
dead_letter_queue_x_retain_x_age: ''
|
||||
path_x_queue: ''
|
||||
path_x_dead_letter_queue: ''
|
||||
config_x_debug: ''
|
||||
config_x_support_escapes: ''
|
||||
custom4:
|
||||
pipeline_x_workers: ''
|
||||
pipeline_x_batch_x_size: ''
|
||||
pipeline_x_batch_x_delay: ''
|
||||
pipeline_x_batch_x_metrics_x_sampling_mode: ''
|
||||
pipeline_x_ordered: ''
|
||||
pipeline_x_ecs_compatibility: ''
|
||||
pipeline_x_reloadable: ''
|
||||
queue_x_type: ''
|
||||
queue_x_max_bytes: ''
|
||||
queue_x_page_capacity: ''
|
||||
queue_x_max_events: ''
|
||||
queue_x_checkpoint_x_acks: ''
|
||||
queue_x_checkpoint_x_writes: ''
|
||||
queue_x_checkpoint_x_interval: ''
|
||||
queue_x_checkpoint_x_retry: ''
|
||||
queue_x_compression: ''
|
||||
queue_x_drain: ''
|
||||
dead_letter_queue_x_enable: ''
|
||||
dead_letter_queue_x_max_bytes: ''
|
||||
dead_letter_queue_x_flush_interval: ''
|
||||
dead_letter_queue_x_flush_check_interval: ''
|
||||
dead_letter_queue_x_storage_policy: ''
|
||||
dead_letter_queue_x_retain_x_age: ''
|
||||
path_x_queue: ''
|
||||
path_x_dead_letter_queue: ''
|
||||
config_x_debug: ''
|
||||
config_x_support_escapes: ''
|
||||
custom5:
|
||||
pipeline_x_workers: ''
|
||||
pipeline_x_batch_x_size: ''
|
||||
pipeline_x_batch_x_delay: ''
|
||||
pipeline_x_batch_x_metrics_x_sampling_mode: ''
|
||||
pipeline_x_ordered: ''
|
||||
pipeline_x_ecs_compatibility: ''
|
||||
pipeline_x_reloadable: ''
|
||||
queue_x_type: ''
|
||||
queue_x_max_bytes: ''
|
||||
queue_x_page_capacity: ''
|
||||
queue_x_max_events: ''
|
||||
queue_x_checkpoint_x_acks: ''
|
||||
queue_x_checkpoint_x_writes: ''
|
||||
queue_x_checkpoint_x_interval: ''
|
||||
queue_x_checkpoint_x_retry: ''
|
||||
queue_x_compression: ''
|
||||
queue_x_drain: ''
|
||||
dead_letter_queue_x_enable: ''
|
||||
dead_letter_queue_x_max_bytes: ''
|
||||
dead_letter_queue_x_flush_interval: ''
|
||||
dead_letter_queue_x_flush_check_interval: ''
|
||||
dead_letter_queue_x_storage_policy: ''
|
||||
dead_letter_queue_x_retain_x_age: ''
|
||||
path_x_queue: ''
|
||||
path_x_dead_letter_queue: ''
|
||||
config_x_debug: ''
|
||||
config_x_support_escapes: ''
|
||||
custom6:
|
||||
pipeline_x_workers: ''
|
||||
pipeline_x_batch_x_size: ''
|
||||
pipeline_x_batch_x_delay: ''
|
||||
pipeline_x_batch_x_metrics_x_sampling_mode: ''
|
||||
pipeline_x_ordered: ''
|
||||
pipeline_x_ecs_compatibility: ''
|
||||
pipeline_x_reloadable: ''
|
||||
queue_x_type: ''
|
||||
queue_x_max_bytes: ''
|
||||
queue_x_page_capacity: ''
|
||||
queue_x_max_events: ''
|
||||
queue_x_checkpoint_x_acks: ''
|
||||
queue_x_checkpoint_x_writes: ''
|
||||
queue_x_checkpoint_x_interval: ''
|
||||
queue_x_checkpoint_x_retry: ''
|
||||
queue_x_compression: ''
|
||||
queue_x_drain: ''
|
||||
dead_letter_queue_x_enable: ''
|
||||
dead_letter_queue_x_max_bytes: ''
|
||||
dead_letter_queue_x_flush_interval: ''
|
||||
dead_letter_queue_x_flush_check_interval: ''
|
||||
dead_letter_queue_x_storage_policy: ''
|
||||
dead_letter_queue_x_retain_x_age: ''
|
||||
path_x_queue: ''
|
||||
path_x_dead_letter_queue: ''
|
||||
config_x_debug: ''
|
||||
config_x_support_escapes: ''
|
||||
custom7:
|
||||
pipeline_x_workers: ''
|
||||
pipeline_x_batch_x_size: ''
|
||||
pipeline_x_batch_x_delay: ''
|
||||
pipeline_x_batch_x_metrics_x_sampling_mode: ''
|
||||
pipeline_x_ordered: ''
|
||||
pipeline_x_ecs_compatibility: ''
|
||||
pipeline_x_reloadable: ''
|
||||
queue_x_type: ''
|
||||
queue_x_max_bytes: ''
|
||||
queue_x_page_capacity: ''
|
||||
queue_x_max_events: ''
|
||||
queue_x_checkpoint_x_acks: ''
|
||||
queue_x_checkpoint_x_writes: ''
|
||||
queue_x_checkpoint_x_interval: ''
|
||||
queue_x_checkpoint_x_retry: ''
|
||||
queue_x_compression: ''
|
||||
queue_x_drain: ''
|
||||
dead_letter_queue_x_enable: ''
|
||||
dead_letter_queue_x_max_bytes: ''
|
||||
dead_letter_queue_x_flush_interval: ''
|
||||
dead_letter_queue_x_flush_check_interval: ''
|
||||
dead_letter_queue_x_storage_policy: ''
|
||||
dead_letter_queue_x_retain_x_age: ''
|
||||
path_x_queue: ''
|
||||
path_x_dead_letter_queue: ''
|
||||
config_x_debug: ''
|
||||
config_x_support_escapes: ''
|
||||
custom8:
|
||||
pipeline_x_workers: ''
|
||||
pipeline_x_batch_x_size: ''
|
||||
pipeline_x_batch_x_delay: ''
|
||||
pipeline_x_batch_x_metrics_x_sampling_mode: ''
|
||||
pipeline_x_ordered: ''
|
||||
pipeline_x_ecs_compatibility: ''
|
||||
pipeline_x_reloadable: ''
|
||||
queue_x_type: ''
|
||||
queue_x_max_bytes: ''
|
||||
queue_x_page_capacity: ''
|
||||
queue_x_max_events: ''
|
||||
queue_x_checkpoint_x_acks: ''
|
||||
queue_x_checkpoint_x_writes: ''
|
||||
queue_x_checkpoint_x_interval: ''
|
||||
queue_x_checkpoint_x_retry: ''
|
||||
queue_x_compression: ''
|
||||
queue_x_drain: ''
|
||||
dead_letter_queue_x_enable: ''
|
||||
dead_letter_queue_x_max_bytes: ''
|
||||
dead_letter_queue_x_flush_interval: ''
|
||||
dead_letter_queue_x_flush_check_interval: ''
|
||||
dead_letter_queue_x_storage_policy: ''
|
||||
dead_letter_queue_x_retain_x_age: ''
|
||||
path_x_queue: ''
|
||||
path_x_dead_letter_queue: ''
|
||||
config_x_debug: ''
|
||||
config_x_support_escapes: ''
|
||||
custom9:
|
||||
pipeline_x_workers: ''
|
||||
pipeline_x_batch_x_size: ''
|
||||
pipeline_x_batch_x_delay: ''
|
||||
pipeline_x_batch_x_metrics_x_sampling_mode: ''
|
||||
pipeline_x_ordered: ''
|
||||
pipeline_x_ecs_compatibility: ''
|
||||
pipeline_x_reloadable: ''
|
||||
queue_x_type: ''
|
||||
queue_x_max_bytes: ''
|
||||
queue_x_page_capacity: ''
|
||||
queue_x_max_events: ''
|
||||
queue_x_checkpoint_x_acks: ''
|
||||
queue_x_checkpoint_x_writes: ''
|
||||
queue_x_checkpoint_x_interval: ''
|
||||
queue_x_checkpoint_x_retry: ''
|
||||
queue_x_compression: ''
|
||||
queue_x_drain: ''
|
||||
dead_letter_queue_x_enable: ''
|
||||
dead_letter_queue_x_max_bytes: ''
|
||||
dead_letter_queue_x_flush_interval: ''
|
||||
dead_letter_queue_x_flush_check_interval: ''
|
||||
dead_letter_queue_x_storage_policy: ''
|
||||
dead_letter_queue_x_retain_x_age: ''
|
||||
path_x_queue: ''
|
||||
path_x_dead_letter_queue: ''
|
||||
config_x_debug: ''
|
||||
config_x_support_escapes: ''
|
||||
settings:
|
||||
lsheap: 500m
|
||||
config:
|
||||
api_x_http_x_host: 0.0.0.0
|
||||
log_x_level: info
|
||||
log_x_format: plain
|
||||
path_x_logs: /var/log/logstash
|
||||
pipeline_x_workers: 1
|
||||
pipeline_x_batch_x_size: 125
|
||||
|
||||
@@ -10,9 +10,6 @@
|
||||
{% from 'logstash/map.jinja' import LOGSTASH_MERGED %}
|
||||
{% from 'logstash/map.jinja' import LOGSTASH_NODES %}
|
||||
{% set lsheap = LOGSTASH_MERGED.settings.lsheap %}
|
||||
{% from 'docker/macros/docker_endpoint.jinja' import clear_stale_endpoint %}
|
||||
|
||||
{{ clear_stale_endpoint('so-logstash', 'sobridge', DOCKERMERGED.containers['so-logstash'].ip) }}
|
||||
|
||||
include:
|
||||
- ca
|
||||
@@ -31,6 +28,7 @@ include:
|
||||
so-logstash:
|
||||
docker_container.running:
|
||||
- image: {{ GLOBALS.registry_host }}:5000/{{ GLOBALS.image_repo }}/so-logstash:{{ GLOBALS.so_version }}
|
||||
- restart_policy: unless-stopped
|
||||
- hostname: so-logstash
|
||||
- name: so-logstash
|
||||
- networks:
|
||||
@@ -107,6 +105,8 @@ so-logstash:
|
||||
{% endif %}
|
||||
- watch:
|
||||
- file: lsetcsync
|
||||
- file: lslog4j2
|
||||
- file: lspipelinesyml
|
||||
- file: trusttheca
|
||||
{% if GLOBALS.is_manager %}
|
||||
- file: elasticsearch_cacerts
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
{%- from 'logstash/map.jinja' import LOGSTASH_MERGED -%}
|
||||
status = error
|
||||
name = LogstashPropertiesConfig
|
||||
|
||||
@@ -16,8 +17,14 @@ name = LogstashPropertiesConfig
|
||||
appender.rolling.type = RollingFile
|
||||
appender.rolling.name = rolling
|
||||
appender.rolling.fileName = /var/log/logstash/logstash.log
|
||||
{%- if LOGSTASH_MERGED.config.get('log_x_format', 'plain') == 'json' %}
|
||||
appender.rolling.layout.type = JSONLayout
|
||||
appender.rolling.layout.compact = true
|
||||
appender.rolling.layout.eventEol = true
|
||||
{%- else %}
|
||||
appender.rolling.layout.type = PatternLayout
|
||||
appender.rolling.layout.pattern = [%d{ISO8601}][%-5p][%-25c] %.10000m%n
|
||||
{%- endif %}
|
||||
appender.rolling.filePattern = /var/log/logstash/logstash-%d{yyyy-MM-dd}.log.gz
|
||||
appender.rolling.policies.type = Policies
|
||||
appender.rolling.policies.time.type = TimeBasedTriggeringPolicy
|
||||
@@ -32,7 +39,5 @@ appender.rolling.strategy.action.condition.type = IfFileName
|
||||
appender.rolling.strategy.action.condition.glob = *.gz
|
||||
appender.rolling.strategy.action.condition.nested_condition.type = IfLastModified
|
||||
appender.rolling.strategy.action.condition.nested_condition.age = 7D
|
||||
rootLogger.level = info
|
||||
rootLogger.level = ${sys:ls.log.level}
|
||||
rootLogger.appenderRef.rolling.ref = rolling
|
||||
#rootLogger.level = ${sys:ls.log.level}
|
||||
#rootLogger.appenderRef.console.ref = ${sys:ls.log.format}_console
|
||||
@@ -1,4 +1,17 @@
|
||||
{%- from 'logstash/map.jinja' import LOGSTASH_MERGED %}
|
||||
{%- set PIPELINE_SETTINGS = LOGSTASH_MERGED.get('pipeline_settings', {}) %}
|
||||
{%- for assigned_pipeline in ASSIGNED_PIPELINES %}
|
||||
- pipeline.id: {{ assigned_pipeline }}
|
||||
path.config: "/usr/share/logstash/pipelines/{{ assigned_pipeline }}/"
|
||||
{%- set extra = PIPELINE_SETTINGS.get(assigned_pipeline, {}) %}
|
||||
{%- if extra is mapping %}
|
||||
{#- values are emitted unquoted so yaml re-infers the type logstash expects:
|
||||
4 as an integer, false as a boolean, 1024mb and auto as strings #}
|
||||
{%- for key, value in extra | dictsort %}
|
||||
{%- set rendered = key | replace('_x_', '.') %}
|
||||
{%- if value not in ['', None] and rendered not in ['pipeline.id', 'path.config'] %}
|
||||
{{ rendered }}: {{ value }}
|
||||
{%- endif %}
|
||||
{%- endfor %}
|
||||
{%- endif %}
|
||||
{% endfor -%}
|
||||
|
||||
@@ -16,6 +16,7 @@ logstash:
|
||||
heavynode: *assigned_pipelines
|
||||
searchnode: *assigned_pipelines
|
||||
manager: *assigned_pipelines
|
||||
managerhype: *assigned_pipelines
|
||||
managersearch: *assigned_pipelines
|
||||
fleet: *assigned_pipelines
|
||||
defined_pipelines:
|
||||
@@ -34,6 +35,11 @@ logstash:
|
||||
custom2: *defined_pipelines
|
||||
custom3: *defined_pipelines
|
||||
custom4: *defined_pipelines
|
||||
custom5: *defined_pipelines
|
||||
custom6: *defined_pipelines
|
||||
custom7: *defined_pipelines
|
||||
custom8: *defined_pipelines
|
||||
custom9: *defined_pipelines
|
||||
pipeline_config:
|
||||
custom001: &pipeline_config
|
||||
description: Pipeline configuration for Logstash
|
||||
@@ -51,6 +57,351 @@ logstash:
|
||||
custom008: *pipeline_config
|
||||
custom009: *pipeline_config
|
||||
custom010: *pipeline_config
|
||||
pipeline_settings:
|
||||
manager: &pipeline_settings
|
||||
pipeline_x_workers:
|
||||
description: >-
|
||||
Number of worker threads that run filters and outputs for this pipeline. May be set higher
|
||||
than the CPU core count when outputs spend time waiting on I/O. Leave blank to use the value
|
||||
from logstash.yml.
|
||||
title: pipeline.workers
|
||||
regex: '^$|^[1-9][0-9]*$'
|
||||
regexFailureMessage: Must be blank, or a positive whole number.
|
||||
advanced: True
|
||||
global: False
|
||||
helpLink: logstash
|
||||
pipeline_x_batch_x_size:
|
||||
description: >-
|
||||
Maximum number of events an individual worker thread collects before running filters and
|
||||
outputs. Larger batches are more efficient but increase heap use; total in-flight events is
|
||||
workers multiplied by batch size. Leave blank to use the value from logstash.yml.
|
||||
title: pipeline.batch.size
|
||||
regex: '^$|^[1-9][0-9]*$'
|
||||
regexFailureMessage: Must be blank, or a positive whole number.
|
||||
advanced: True
|
||||
global: False
|
||||
helpLink: logstash
|
||||
pipeline_x_batch_x_delay:
|
||||
description: >-
|
||||
Milliseconds a worker waits for the next event before running a batch that is not yet full.
|
||||
Leave blank to use the value from logstash.yml.
|
||||
title: pipeline.batch.delay
|
||||
regex: '^$|^[0-9]+$'
|
||||
regexFailureMessage: Must be blank, or a whole number.
|
||||
advanced: True
|
||||
global: False
|
||||
helpLink: logstash
|
||||
pipeline_x_batch_x_metrics_x_sampling_mode:
|
||||
description: >-
|
||||
Controls how often batch size metrics are collected for this pipeline, which helps tune
|
||||
pipeline.batch.size to the batch sizes actually being processed. Fuller sampling consumes
|
||||
additional heap. Elastic marks this setting as a technical preview that may change in a
|
||||
future release. Leave blank to use the value from logstash.yml.
|
||||
title: pipeline.batch.metrics.sampling_mode
|
||||
options:
|
||||
- ''
|
||||
- 'disabled'
|
||||
- 'minimal'
|
||||
- 'full'
|
||||
advanced: True
|
||||
global: False
|
||||
helpLink: logstash
|
||||
pipeline_x_ordered:
|
||||
description: >-
|
||||
Whether event order is preserved through this pipeline. auto enables ordering only when
|
||||
pipeline.workers is explicitly set to 1, and does nothing otherwise. Setting this to true
|
||||
requires pipeline.workers to be 1 as well; with more workers this pipeline fails to start.
|
||||
Leave blank to use the value from logstash.yml.
|
||||
title: pipeline.ordered
|
||||
options:
|
||||
- ''
|
||||
- 'auto'
|
||||
- 'true'
|
||||
- 'false'
|
||||
advanced: True
|
||||
global: False
|
||||
helpLink: logstash
|
||||
pipeline_x_ecs_compatibility:
|
||||
description: >-
|
||||
Elastic Common Schema compatibility mode for plugins in this pipeline. Security Onion sets
|
||||
this globally and it should rarely be changed per pipeline. Elastic considers values other
|
||||
than disabled to be BETA, and they may produce unintended consequences when upgrading
|
||||
Logstash. Leave blank to use the value from logstash.yml.
|
||||
title: pipeline.ecs_compatibility
|
||||
options:
|
||||
- ''
|
||||
- 'disabled'
|
||||
- 'v1'
|
||||
- 'v8'
|
||||
advanced: True
|
||||
global: False
|
||||
helpLink: logstash
|
||||
pipeline_x_reloadable:
|
||||
description: >-
|
||||
Whether this pipeline may be reloaded when its configuration changes. Leave blank to use the
|
||||
value from logstash.yml.
|
||||
title: pipeline.reloadable
|
||||
options:
|
||||
- ''
|
||||
- 'true'
|
||||
- 'false'
|
||||
advanced: True
|
||||
global: False
|
||||
helpLink: logstash
|
||||
queue_x_type:
|
||||
description: >-
|
||||
Queue backing this pipeline. persisted buffers events to disk under /nsm/logstash so they
|
||||
survive a restart, at some throughput cost; memory does not. Leave blank to use the value
|
||||
from logstash.yml.
|
||||
title: queue.type
|
||||
options:
|
||||
- ''
|
||||
- 'memory'
|
||||
- 'persisted'
|
||||
advanced: True
|
||||
global: False
|
||||
helpLink: logstash
|
||||
queue_x_max_bytes:
|
||||
description: >-
|
||||
Total capacity of the persistent queue for this pipeline, in bytes. Only applies when
|
||||
queue.type is persisted. The disk backing /nsm/logstash must have room for this much data or
|
||||
the pipeline fails to start, reporting that it was unable to allocate the space. If both
|
||||
queue.max_events and queue.max_bytes are set, whichever is reached first applies. Leave
|
||||
blank to use the value from logstash.yml.
|
||||
title: queue.max_bytes
|
||||
regex: '^$|^[0-9]+$|^[0-9]+(\.[0-9]+)?\s*(b|kb?|mb?|gb?|tb?|pb?)$'
|
||||
regexFailureMessage: Must be blank, or a size such as 512mb, 1gb, or 64k. Units are lowercase.
|
||||
advanced: True
|
||||
global: False
|
||||
helpLink: logstash
|
||||
queue_x_page_capacity:
|
||||
description: >-
|
||||
Size of the individual append-only page data files that make up the persistent queue for
|
||||
this pipeline. Only applies when queue.type is persisted. Leave blank to use the value from
|
||||
logstash.yml.
|
||||
title: queue.page_capacity
|
||||
regex: '^$|^[0-9]+$|^[0-9]+(\.[0-9]+)?\s*(b|kb?|mb?|gb?|tb?|pb?)$'
|
||||
regexFailureMessage: Must be blank, or a size such as 512mb, 1gb, or 64k. Units are lowercase.
|
||||
advanced: True
|
||||
global: False
|
||||
helpLink: logstash
|
||||
queue_x_max_events:
|
||||
description: >-
|
||||
Maximum number of unread events in the persistent queue for this pipeline. 0 means
|
||||
unlimited. Only applies when queue.type is persisted. Leave blank to use the value from
|
||||
logstash.yml.
|
||||
title: queue.max_events
|
||||
regex: '^$|^[0-9]+$'
|
||||
regexFailureMessage: Must be blank, or a whole number.
|
||||
advanced: True
|
||||
global: False
|
||||
helpLink: logstash
|
||||
queue_x_checkpoint_x_acks:
|
||||
description: >-
|
||||
Maximum number of acknowledged events before a checkpoint is forced. 0 means unlimited. Only
|
||||
applies when queue.type is persisted. Leave blank to use the value from logstash.yml.
|
||||
title: queue.checkpoint.acks
|
||||
regex: '^$|^[0-9]+$'
|
||||
regexFailureMessage: Must be blank, or a whole number.
|
||||
advanced: True
|
||||
global: False
|
||||
helpLink: logstash
|
||||
queue_x_checkpoint_x_writes:
|
||||
description: >-
|
||||
Maximum number of written events before a checkpoint is forced. Setting this to 1 gives
|
||||
maximum durability at a severe performance cost. 0 means unlimited. Only applies when
|
||||
queue.type is persisted. Leave blank to use the value from logstash.yml.
|
||||
title: queue.checkpoint.writes
|
||||
regex: '^$|^[0-9]+$'
|
||||
regexFailureMessage: Must be blank, or a whole number.
|
||||
advanced: True
|
||||
global: False
|
||||
helpLink: logstash
|
||||
queue_x_checkpoint_x_interval:
|
||||
description: >-
|
||||
Milliseconds between forced checkpoints on the persistent queue head page. 0 eliminates
|
||||
periodic checkpoints. Deprecated by Elastic as of Logstash 9.1. Only applies when queue.type
|
||||
is persisted. Leave blank to use the value from logstash.yml.
|
||||
title: queue.checkpoint.interval
|
||||
regex: '^$|^[0-9]+$'
|
||||
regexFailureMessage: Must be blank, or a whole number.
|
||||
advanced: True
|
||||
global: False
|
||||
helpLink: logstash
|
||||
queue_x_checkpoint_x_retry:
|
||||
description: >-
|
||||
When enabled, Logstash retries four times per attempted checkpoint write that fails; later
|
||||
errors are not retried. Elastic describes this as a workaround for failed checkpoint writes
|
||||
seen only on Windows and on filesystems with non-standard behaviour such as SANs, and does
|
||||
not recommend enabling it otherwise. Only applies when queue.type is persisted. Leave blank
|
||||
to use the value from logstash.yml.
|
||||
title: queue.checkpoint.retry
|
||||
options:
|
||||
- ''
|
||||
- 'true'
|
||||
- 'false'
|
||||
advanced: True
|
||||
global: False
|
||||
helpLink: logstash
|
||||
queue_x_compression:
|
||||
description: >-
|
||||
Compression applied to persistent queue pages for this pipeline, trading CPU for disk: speed
|
||||
favours the fastest operation, size the smallest files, and balanced sits between them. Once
|
||||
compressed events have been written, that queue cannot be read by Logstash releases earlier
|
||||
than 9.2. Only applies when queue.type is persisted. Leave blank to use the value from
|
||||
logstash.yml.
|
||||
title: queue.compression
|
||||
options:
|
||||
- ''
|
||||
- 'none'
|
||||
- 'speed'
|
||||
- 'balanced'
|
||||
- 'size'
|
||||
advanced: True
|
||||
global: False
|
||||
helpLink: logstash
|
||||
queue_x_drain:
|
||||
description: >-
|
||||
When enabled, Logstash waits for the persistent queue to drain before shutting down this
|
||||
pipeline. Draining a large queue makes shutdown take considerably longer. Only applies when
|
||||
queue.type is persisted. Leave blank to use the value from logstash.yml.
|
||||
title: queue.drain
|
||||
options:
|
||||
- ''
|
||||
- 'true'
|
||||
- 'false'
|
||||
advanced: True
|
||||
global: False
|
||||
helpLink: logstash
|
||||
dead_letter_queue_x_enable:
|
||||
description: >-
|
||||
Whether events this pipeline cannot process are written to a dead letter queue instead of
|
||||
being dropped. Leave blank to use the value from logstash.yml.
|
||||
title: dead_letter_queue.enable
|
||||
options:
|
||||
- ''
|
||||
- 'true'
|
||||
- 'false'
|
||||
advanced: True
|
||||
global: False
|
||||
helpLink: logstash
|
||||
dead_letter_queue_x_max_bytes:
|
||||
description: >-
|
||||
Total capacity of the dead letter queue for this pipeline, in bytes. Only applies when
|
||||
dead_letter_queue.enable is true. Leave blank to use the value from logstash.yml.
|
||||
title: dead_letter_queue.max_bytes
|
||||
regex: '^$|^[0-9]+$|^[0-9]+(\.[0-9]+)?\s*(b|kb?|mb?|gb?|tb?|pb?)$'
|
||||
regexFailureMessage: Must be blank, or a size such as 512mb, 1gb, or 64k. Units are lowercase.
|
||||
advanced: True
|
||||
global: False
|
||||
helpLink: logstash
|
||||
dead_letter_queue_x_flush_interval:
|
||||
description: >-
|
||||
Milliseconds before an incomplete dead letter queue segment is flushed and made available to
|
||||
the dead_letter_queue input. Lower values write more, smaller segment files; higher values
|
||||
add latency before events can be read. Only applies when dead_letter_queue.enable is true.
|
||||
Leave blank to use the value from logstash.yml.
|
||||
title: dead_letter_queue.flush_interval
|
||||
regex: '^$|^[0-9]+$'
|
||||
regexFailureMessage: Must be blank, or a whole number.
|
||||
advanced: True
|
||||
global: False
|
||||
helpLink: logstash
|
||||
dead_letter_queue_x_flush_check_interval:
|
||||
description: >-
|
||||
Milliseconds between checks for a stale dead letter queue segment needing a flush. Cannot be
|
||||
set lower than 1000. Smaller values rotate segments sooner at the cost of CPU. Only applies
|
||||
when dead_letter_queue.enable is true. Leave blank to use the value from logstash.yml.
|
||||
title: dead_letter_queue.flush_check_interval
|
||||
regex: '^$|^[0-9]+$'
|
||||
regexFailureMessage: Must be blank, or a whole number.
|
||||
advanced: True
|
||||
global: False
|
||||
helpLink: logstash
|
||||
dead_letter_queue_x_storage_policy:
|
||||
description: >-
|
||||
Action taken when dead_letter_queue.max_bytes is reached: drop_newer stops accepting new
|
||||
events, drop_older removes the oldest events to make room. Only applies when
|
||||
dead_letter_queue.enable is true. Leave blank to use the value from logstash.yml.
|
||||
title: dead_letter_queue.storage_policy
|
||||
options:
|
||||
- ''
|
||||
- 'drop_newer'
|
||||
- 'drop_older'
|
||||
advanced: True
|
||||
global: False
|
||||
helpLink: logstash
|
||||
dead_letter_queue_x_retain_x_age:
|
||||
description: >-
|
||||
How long an event is kept in the dead letter queue before Logstash removes it, such as 5d.
|
||||
Units are d, h, m and s; there is no default unit, so one must be given. Only applies when
|
||||
dead_letter_queue.enable is true. Leave blank to use the value from logstash.yml.
|
||||
title: dead_letter_queue.retain.age
|
||||
regex: '^$|^[0-9]+\s*[dhms]$'
|
||||
regexFailureMessage: Must be blank, or a number followed by d, h, m, or s, such as 5d.
|
||||
advanced: True
|
||||
global: False
|
||||
helpLink: logstash
|
||||
path_x_queue:
|
||||
description: >-
|
||||
Directory inside the Logstash container holding the persistent queue for this pipeline. The
|
||||
default lives under the /nsm/logstash bind mount; a path outside it will not survive a
|
||||
container restart. Logstash creates the directory if it is missing, requires it to be
|
||||
writable, and refuses to start if the path is a symlink. Only applies when queue.type is
|
||||
persisted. Leave blank to use the value from logstash.yml.
|
||||
title: path.queue
|
||||
advanced: True
|
||||
global: False
|
||||
helpLink: logstash
|
||||
path_x_dead_letter_queue:
|
||||
description: >-
|
||||
Directory inside the Logstash container holding the dead letter queue for this pipeline. The
|
||||
default lives under the /nsm/logstash bind mount; a path outside it will not survive a
|
||||
container restart. Logstash creates the directory if it is missing, requires it to be
|
||||
writable, and refuses to start if the path is a symlink. Only applies when
|
||||
dead_letter_queue.enable is true. Leave blank to use the value from logstash.yml.
|
||||
title: path.dead_letter_queue
|
||||
advanced: True
|
||||
global: False
|
||||
helpLink: logstash
|
||||
config_x_debug:
|
||||
description: >-
|
||||
Whether the fully compiled configuration for this pipeline is written to the log. The output
|
||||
may contain sensitive values from the pipeline configuration. Leave blank to use the value
|
||||
from logstash.yml.
|
||||
title: config.debug
|
||||
options:
|
||||
- ''
|
||||
- 'true'
|
||||
- 'false'
|
||||
advanced: True
|
||||
global: False
|
||||
helpLink: logstash
|
||||
config_x_support_escapes:
|
||||
description: >-
|
||||
Whether escape sequences such as \n and \t in this pipeline's quoted strings are
|
||||
interpreted. Leave blank to use the value from logstash.yml.
|
||||
title: config.support_escapes
|
||||
options:
|
||||
- ''
|
||||
- 'true'
|
||||
- 'false'
|
||||
advanced: True
|
||||
global: False
|
||||
helpLink: logstash
|
||||
fleet: *pipeline_settings
|
||||
receiver: *pipeline_settings
|
||||
search: *pipeline_settings
|
||||
custom0: *pipeline_settings
|
||||
custom1: *pipeline_settings
|
||||
custom2: *pipeline_settings
|
||||
custom3: *pipeline_settings
|
||||
custom4: *pipeline_settings
|
||||
custom5: *pipeline_settings
|
||||
custom6: *pipeline_settings
|
||||
custom7: *pipeline_settings
|
||||
custom8: *pipeline_settings
|
||||
custom9: *pipeline_settings
|
||||
settings:
|
||||
lsheap:
|
||||
description: Heap size to use for logstash
|
||||
@@ -62,6 +413,35 @@ logstash:
|
||||
helpLink: logstash
|
||||
readonly: True
|
||||
advanced: True
|
||||
log_x_level:
|
||||
description: >-
|
||||
Verbosity of the Logstash log at /opt/so/log/logstash/logstash.log. debug and trace produce
|
||||
a very large volume of log data on a busy node and should be used only while troubleshooting;
|
||||
the log rotates at 1GB and rotated files are deleted after 7 days. Setting this to debug is
|
||||
also what makes the per-pipeline config.debug setting emit anything.
|
||||
title: log.level
|
||||
options:
|
||||
- 'fatal'
|
||||
- 'error'
|
||||
- 'warn'
|
||||
- 'info'
|
||||
- 'debug'
|
||||
- 'trace'
|
||||
advanced: True
|
||||
global: False
|
||||
helpLink: logstash
|
||||
log_x_format:
|
||||
description: >-
|
||||
Layout of the Logstash log. plain writes human readable lines; json writes one JSON object
|
||||
per line, which is easier to parse but harder to read directly. The file name and location
|
||||
do not change.
|
||||
title: log.format
|
||||
options:
|
||||
- 'plain'
|
||||
- 'json'
|
||||
advanced: True
|
||||
global: False
|
||||
helpLink: logstash
|
||||
path_x_logs:
|
||||
description: Path inside the container to wrote logs.
|
||||
helpLink: logstash
|
||||
|
||||
@@ -0,0 +1,21 @@
|
||||
{% from 'vars/globals.map.jinja' import GLOBALS %}
|
||||
{% from 'salt/auto_apply.map.jinja' import AUTOAPPLY %}
|
||||
|
||||
include:
|
||||
- salt.minion
|
||||
|
||||
{% if GLOBALS.is_manager and AUTOAPPLY.enabled %}
|
||||
salt_beacons_pushstate:
|
||||
file.managed:
|
||||
- name: /etc/salt/minion.d/beacons_pushstate.conf
|
||||
- source: salt://manager/files/beacons_pushstate.conf.jinja
|
||||
- template: jinja
|
||||
- watch_in:
|
||||
- service: salt_minion_service
|
||||
{% else %}
|
||||
salt_beacons_pushstate:
|
||||
file.absent:
|
||||
- name: /etc/salt/minion.d/beacons_pushstate.conf
|
||||
- watch_in:
|
||||
- service: salt_minion_service
|
||||
{% endif %}
|
||||
@@ -0,0 +1,18 @@
|
||||
{% from 'salt/auto_apply.map.jinja' import AUTOAPPLY %}
|
||||
beacons:
|
||||
postgres_pillar_beacon:
|
||||
- interval: {{ AUTOAPPLY.drain_interval }}
|
||||
- disable_during_state_run: False
|
||||
local_files_beacon:
|
||||
- interval: {{ AUTOAPPLY.drain_interval }}
|
||||
- disable_during_state_run: False
|
||||
# Tags are app names in salt/reactor/pillar_push_map.yaml.
|
||||
# Allowlist on purpose: salt writes elsewhere under local/salt/ and would self-retrigger.
|
||||
- paths:
|
||||
/opt/so/saltstack/local/salt/suricata/rules: suricata
|
||||
/opt/so/saltstack/local/salt/strelka/rules/compiled: strelka
|
||||
/opt/so/saltstack/local/salt/zeek/policy: zeek
|
||||
/opt/so/saltstack/local/salt/zeek/zkg: zeek
|
||||
/opt/so/saltstack/local/salt/elasticsearch/files/ingest: elasticsearch
|
||||
/opt/so/saltstack/local/salt/elasticsearch/roles: elasticsearch
|
||||
/opt/so/saltstack/local/salt/logstash/pipelines/config/custom: logstash
|
||||
+14
-8
@@ -15,6 +15,7 @@ include:
|
||||
- manager.elasticsearch
|
||||
- manager.kibana
|
||||
- manager.managed_soc_annotations
|
||||
- manager.beacons
|
||||
|
||||
repo_log_dir:
|
||||
file.directory:
|
||||
@@ -112,8 +113,8 @@ manager_sbin:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://manager/tools/sbin
|
||||
- user: 939
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
- exclude_pat:
|
||||
- "*_test.py"
|
||||
@@ -123,8 +124,8 @@ manager_sbin_jinja:
|
||||
file.recurse:
|
||||
- name: /usr/sbin/
|
||||
- source: salt://manager/tools/sbin_jinja/
|
||||
- user: socore
|
||||
- group: socore
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
- template: jinja
|
||||
- show_changes: False
|
||||
@@ -165,7 +166,7 @@ so-repo-sync:
|
||||
|
||||
so_fleetagent_status:
|
||||
cron.present:
|
||||
- name: /usr/sbin/so-elasticagent-status > /opt/so/log/agents/agentstatus.log 2>&1
|
||||
- name: '/usr/sbin/so-elasticagent-status > /opt/so/log/agents/agentstatus.log.tmp 2>&1; mv -f /opt/so/log/agents/agentstatus.log.tmp /opt/so/log/agents/agentstatus.log'
|
||||
- identifier: so_fleetagent_status
|
||||
- user: root
|
||||
- minute: '*/5'
|
||||
@@ -189,11 +190,15 @@ so_fleetagent_monitor:
|
||||
- month: '*'
|
||||
- dayweek: '*'
|
||||
|
||||
socore_own_saltstack_default:
|
||||
# This tree is the source of every root-executed script (/usr/sbin, reactors, _runners,
|
||||
# engines, salt-relay.sh). SOC mounts /opt/so/saltstack rw as uid 939 but only writes
|
||||
# under local/. Do not add dir_mode/file_mode here -- SOC reads default/ and 750/640
|
||||
# would break its config load.
|
||||
root_own_saltstack_default:
|
||||
file.directory:
|
||||
- name: /opt/so/saltstack/default
|
||||
- user: socore
|
||||
- group: socore
|
||||
- user: root
|
||||
- group: root
|
||||
- recurse:
|
||||
- user
|
||||
- group
|
||||
@@ -260,6 +265,7 @@ surifiltersrules:
|
||||
- user: 939
|
||||
- group: 939
|
||||
|
||||
|
||||
{% else %}
|
||||
|
||||
{{sls}}_state_not_allowed:
|
||||
|
||||
@@ -106,7 +106,8 @@ while [[ $# -gt 0 ]]; do
|
||||
esac
|
||||
done
|
||||
|
||||
hydraUrl=${HYDRA_URL:-http://127.0.0.1:4445}
|
||||
hydraContainer=${HYDRA_CONTAINER:-so-hydra}
|
||||
hydraUrl=${HYDRA_URL:-http://localhost:4445}
|
||||
socRolesFile=${SOC_ROLES_FILE:-/opt/so/conf/soc/soc_clients_roles}
|
||||
soUID=${SOCORE_UID:-939}
|
||||
soGID=${SOCORE_GID:-939}
|
||||
@@ -124,6 +125,10 @@ function fail() {
|
||||
exit 1
|
||||
}
|
||||
|
||||
function hydraCurl() {
|
||||
docker exec "$hydraContainer" curl "$@"
|
||||
}
|
||||
|
||||
function require() {
|
||||
cmd=$1
|
||||
which "$1" 2>&1 > /dev/null
|
||||
@@ -133,8 +138,8 @@ function require() {
|
||||
# Verify this environment is capable of running this script
|
||||
function verifyEnvironment() {
|
||||
require "jq"
|
||||
require "curl"
|
||||
response=$(curl -Ss -L ${hydraUrl}/health/alive)
|
||||
require "docker"
|
||||
response=$(hydraCurl -Ss -L ${hydraUrl}/health/alive)
|
||||
[[ "$response" != '{"status":"ok"}' ]] && fail "Unable to communicate with Hydra; specify URL via HYDRA_URL environment variable"
|
||||
}
|
||||
|
||||
@@ -164,7 +169,7 @@ function ensureRoleFileExists() {
|
||||
}
|
||||
|
||||
function listClients() {
|
||||
response=$(curl -Ss -L -f ${hydraUrl}/admin/clients)
|
||||
response=$(hydraCurl -Ss -L -f ${hydraUrl}/admin/clients)
|
||||
[[ $? != 0 ]] && fail "Unable to communicate with Hydra"
|
||||
|
||||
clientIds=$(echo "${response}" | jq -r ".[] | .client_id" | sort)
|
||||
@@ -251,7 +256,7 @@ function createClient() {
|
||||
EOF
|
||||
)
|
||||
|
||||
response=$(curl -Ss -L --fail-with-body -X POST ${hydraUrl}/admin/clients -d "$body")
|
||||
response=$(hydraCurl -Ss -L --fail-with-body -X POST ${hydraUrl}/admin/clients -d "$body")
|
||||
if [[ $? != 0 ]]; then
|
||||
error=$(echo $response | jq .error)
|
||||
fail "Failed to submit request to Hydra: $error"
|
||||
@@ -283,7 +288,7 @@ function update() {
|
||||
EOF
|
||||
)
|
||||
|
||||
response=$(curl -Ss -L --fail-with-body -X PATCH ${hydraUrl}/admin/clients/$id -d "$body")
|
||||
response=$(hydraCurl -Ss -L --fail-with-body -X PATCH ${hydraUrl}/admin/clients/$id -d "$body")
|
||||
if [[ $? != 0 ]]; then
|
||||
error=$(echo $response | jq .error)
|
||||
fail "Failed to submit request to Hydra: $error"
|
||||
@@ -305,7 +310,7 @@ function generateSecret() {
|
||||
EOF
|
||||
)
|
||||
|
||||
response=$(curl -Ss -L --fail-with-body -X PATCH ${hydraUrl}/admin/clients/$id -d "$body")
|
||||
response=$(hydraCurl -Ss -L --fail-with-body -X PATCH ${hydraUrl}/admin/clients/$id -d "$body")
|
||||
if [[ $? != 0 ]]; then
|
||||
error=$(echo $response | jq .error)
|
||||
fail "Failed to submit request to Hydra: $error"
|
||||
@@ -317,7 +322,7 @@ function deleteClient() {
|
||||
|
||||
[[ ${identityId} == "" ]] && fail "Client not found"
|
||||
|
||||
response=$(curl -Ss -XDELETE -L --fail-with-body "${hydraUrl}/admin/clients/$identityId")
|
||||
response=$(hydraCurl -Ss -XDELETE -L --fail-with-body "${hydraUrl}/admin/clients/$identityId")
|
||||
if [[ $? != 0 ]]; then
|
||||
error=$(echo $response | jq .error)
|
||||
fail "Failed to submit request to Hydra: $error"
|
||||
|
||||
@@ -121,8 +121,14 @@ for i in "$@"; do
|
||||
esac
|
||||
done
|
||||
|
||||
PILLARFILE=/opt/so/saltstack/local/pillar/minions/$MINION_ID.sls
|
||||
ADVPILLARFILE=/opt/so/saltstack/local/pillar/minions/adv_$MINION_ID.sls
|
||||
if [[ -n "$MINION_ID" && ! "$MINION_ID" =~ ^[A-Za-z0-9._-]{1,253}$ ]]; then
|
||||
echo "Invalid minion id: $MINION_ID"
|
||||
log "ERROR" "Invalid minion id: $MINION_ID"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
readonly PILLARFILE=/opt/so/saltstack/local/pillar/minions/$MINION_ID.sls
|
||||
readonly ADVPILLARFILE=/opt/so/saltstack/local/pillar/minions/adv_$MINION_ID.sls
|
||||
|
||||
function getinstallinfo() {
|
||||
log "INFO" "Getting install info for minion $MINION_ID"
|
||||
@@ -133,11 +139,26 @@ function getinstallinfo() {
|
||||
return 1
|
||||
fi
|
||||
|
||||
while read -r var; do export "$var"; done <<< "$INSTALLVARS"
|
||||
if [ $? -ne 0 ]; then
|
||||
log "ERROR" "Failed to source install variables"
|
||||
return 1
|
||||
# install.txt is controlled by the minion; only accept known keys and never eval or export them
|
||||
local line key
|
||||
while IFS= read -r line; do
|
||||
[[ "$line" == *=* ]] || continue
|
||||
key=${line%%=*}
|
||||
case "$key" in
|
||||
MAINIP|MNIC|NODE_DESCRIPTION|ES_HEAP_SIZE|PATCHSCHEDULENAME|INTERFACE|NODETYPE|CORECOUNT|LSHOSTNAME|LSHEAP|CPUCORES|IDH_MGTRESTRICT|IDH_SERVICES)
|
||||
printf -v "$key" '%s' "${line#*=}"
|
||||
;;
|
||||
*)
|
||||
log "WARN" "Ignoring unexpected install var from $MINION_ID: ${key:0:64}"
|
||||
;;
|
||||
esac
|
||||
done <<< "$INSTALLVARS"
|
||||
|
||||
if [[ "$NODE_DESCRIPTION" == \'*\' ]]; then
|
||||
NODE_DESCRIPTION=${NODE_DESCRIPTION:1:-1}
|
||||
fi
|
||||
|
||||
log "INFO" "Fetched install info for $MINION_ID (node type: ${NODETYPE:-unset})"
|
||||
}
|
||||
|
||||
function pcapspace() {
|
||||
@@ -174,6 +195,12 @@ function pcapspace() {
|
||||
fi
|
||||
fi
|
||||
|
||||
# Must be checked before arithmetic expansion, which evaluates array subscripts
|
||||
if [[ ! "$SPACESIZE" =~ ^[0-9]+$ ]]; then
|
||||
log "ERROR" "Invalid disk size for $MINION_ID: ${SPACESIZE:0:64}"
|
||||
return 1
|
||||
fi
|
||||
|
||||
local s=$(( $SPACESIZE / 1000000 ))
|
||||
local s1=$(( $s / 4 * $PCAP_PERCENTAGE ))
|
||||
|
||||
@@ -483,6 +510,7 @@ function add_sensoroni_with_analyze_to_minion() {
|
||||
|
||||
# Sensor settings for the minion pillar
|
||||
function add_sensor_to_minion() {
|
||||
log "INFO" "Writing sensor configuration for $MINION_ID (interface: ${INTERFACE:-unset})"
|
||||
{
|
||||
echo "sensor:"
|
||||
echo " interface: '$INTERFACE'"
|
||||
@@ -509,6 +537,8 @@ function add_sensor_to_minion() {
|
||||
log "ERROR" "Failed to add sensor configuration to $PILLARFILE"
|
||||
return 1
|
||||
fi
|
||||
|
||||
log "INFO" "Wrote sensor configuration for $MINION_ID"
|
||||
}
|
||||
|
||||
function add_elastalert_to_minion() {
|
||||
@@ -581,11 +611,14 @@ function add_telegraf_to_minion() {
|
||||
# generates a password on first add and is a no-op on re-add so the cred
|
||||
# is stable across repeated so-minion runs. postgres.telegraf_users on the
|
||||
# manager creates/updates the DB role from the same pillar.
|
||||
so-telegraf-cred add "$MINION_ID"
|
||||
if [ $? -ne 0 ]; then
|
||||
log "ERROR" "Failed to provision postgres telegraf cred for $MINION_ID"
|
||||
return 1
|
||||
fi
|
||||
log "INFO" "Provisioning postgres telegraf credential for $MINION_ID"
|
||||
so-telegraf-cred add "$MINION_ID"
|
||||
local result=$?
|
||||
if [ $result -ne 0 ]; then
|
||||
log "ERROR" "Failed to provision postgres telegraf cred for $MINION_ID (exit code: $result)"
|
||||
return 1
|
||||
fi
|
||||
log "INFO" "Provisioned postgres telegraf credential for $MINION_ID"
|
||||
}
|
||||
|
||||
function add_influxdb_to_minion() {
|
||||
@@ -1042,8 +1075,59 @@ function updateMineAndApplyStates() {
|
||||
fi
|
||||
}
|
||||
|
||||
# Values end up in a Jinja-rendered pillar and in bash, and may come from the minion
|
||||
function validate_minion_vars() {
|
||||
local error_msg=""
|
||||
# Inline rather than valid_ip4: so-common is not installed yet when setup runs -o=setup
|
||||
local octet='(25[0-5]|2[0-4][0-9]|1?[0-9]?[0-9])'
|
||||
local ip4_re="^($octet\.){3}$octet$"
|
||||
|
||||
case "$NODETYPE" in
|
||||
EVAL|STANDALONE|MANAGER|MANAGERSEARCH|MANAGERHYPE|IMPORT)
|
||||
# Manager pillars also rewrite the CA pillar, so never accept them from a remote node
|
||||
[[ "$OPERATION" == "setup" ]] || error_msg="Node type $NODETYPE can only be configured during setup"
|
||||
;;
|
||||
FLEET|IDH|HEAVYNODE|SENSOR|SEARCHNODE|RECEIVER|HYPERVISOR|DESKTOP)
|
||||
;;
|
||||
*)
|
||||
error_msg="Invalid node type: ${NODETYPE:0:64}"
|
||||
;;
|
||||
esac
|
||||
|
||||
if [[ -z "$error_msg" ]]; then
|
||||
if [[ ! "$MAINIP" =~ $ip4_re ]]; then
|
||||
error_msg="Invalid MAINIP: ${MAINIP:0:64}"
|
||||
elif [[ ! "$MNIC" =~ ^[A-Za-z0-9._-]*$ ]]; then
|
||||
error_msg="Invalid MNIC: ${MNIC:0:64}"
|
||||
elif [[ ! "$INTERFACE" =~ ^[A-Za-z0-9._-]*$ ]]; then
|
||||
error_msg="Invalid INTERFACE: ${INTERFACE:0:64}"
|
||||
elif [[ ! "$LSHOSTNAME" =~ ^[A-Za-z0-9._-]*$ ]]; then
|
||||
error_msg="Invalid LSHOSTNAME: ${LSHOSTNAME:0:64}"
|
||||
elif [[ ! "$ES_HEAP_SIZE" =~ ^([0-9]+[kKmMgG]?)?$ ]]; then
|
||||
error_msg="Invalid ES_HEAP_SIZE: ${ES_HEAP_SIZE:0:64}"
|
||||
elif [[ ! "$LSHEAP" =~ ^([0-9]+[kKmMgG]?)?$ ]]; then
|
||||
error_msg="Invalid LSHEAP: ${LSHEAP:0:64}"
|
||||
elif [[ ! "$CORECOUNT" =~ ^[0-9]*$ ]]; then
|
||||
error_msg="Invalid CORECOUNT: ${CORECOUNT:0:64}"
|
||||
elif [[ ! "$CPUCORES" =~ ^[0-9]*$ ]]; then
|
||||
error_msg="Invalid CPUCORES: ${CPUCORES:0:64}"
|
||||
elif [[ ! "$IDH_MGTRESTRICT" =~ ^(True|False)?$ ]]; then
|
||||
error_msg="Invalid IDH_MGTRESTRICT: ${IDH_MGTRESTRICT:0:64}"
|
||||
fi
|
||||
fi
|
||||
|
||||
if [[ -n "$error_msg" ]]; then
|
||||
log "ERROR" "$error_msg"
|
||||
echo "$error_msg"
|
||||
return 1
|
||||
fi
|
||||
|
||||
# Free text; removing braces is enough to prevent any Jinja delimiter
|
||||
NODE_DESCRIPTION=${NODE_DESCRIPTION//[\{\}[:cntrl:]]/}
|
||||
}
|
||||
|
||||
function setupMinionFiles() {
|
||||
log "INFO" "Setting up minion files for $MINION_ID"
|
||||
log "INFO" "Setting up minion files for $MINION_ID (pillar: $PILLARFILE)"
|
||||
|
||||
# Check to see if nodetype is set
|
||||
if [ -z $NODETYPE ]; then
|
||||
@@ -1053,6 +1137,8 @@ function setupMinionFiles() {
|
||||
return 1
|
||||
fi
|
||||
|
||||
validate_minion_vars || return 1
|
||||
|
||||
# Create the base minion files
|
||||
create_minion_files || return 1
|
||||
|
||||
@@ -1069,7 +1155,10 @@ function setupMinionFiles() {
|
||||
fi
|
||||
|
||||
# Create node-specific configuration
|
||||
create$NODETYPE || return 1
|
||||
create$NODETYPE || {
|
||||
log "ERROR" "Failed to create $NODETYPE configuration for $MINION_ID"
|
||||
return 1
|
||||
}
|
||||
|
||||
# Ensure proper ownership after all content is written
|
||||
ensure_socore_ownership || return 1
|
||||
|
||||
@@ -0,0 +1,231 @@
|
||||
#!/opt/saltstack/salt/bin/python3
|
||||
|
||||
# Copyright Security Onion Solutions LLC and/or licensed to Security Onion Solutions LLC under one
|
||||
# or more contributor license agreements. Licensed under the Elastic License 2.0 as shown at
|
||||
# https://securityonion.net/license; you may not use this file except in compliance with the
|
||||
# Elastic License 2.0.
|
||||
|
||||
"""
|
||||
so-push-drainer
|
||||
===============
|
||||
|
||||
Scheduled drainer for the active-push feature. Runs on the manager every
|
||||
drain_interval seconds (default 15) via a salt schedule in salt/salt/push_drain_schedule.sls.
|
||||
|
||||
For each intent file under /opt/so/state/push_pending/*.json whose last_touch
|
||||
is older than debounce_seconds, this script:
|
||||
* concatenates the actions lists from every ready intent
|
||||
* dedupes by (state or __highstate__, tgt, tgt_type)
|
||||
* dispatches a single `salt-run state.orchestrate orch.push_batch --async`
|
||||
with the deduped actions list passed as pillar kwargs
|
||||
* deletes the contributed intent files on successful dispatch
|
||||
|
||||
Reactor sls files (push_files, push_pillar) write intents
|
||||
but never dispatch directly
|
||||
"""
|
||||
|
||||
import fcntl
|
||||
import glob
|
||||
import json
|
||||
import logging
|
||||
import logging.handlers
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
|
||||
import salt.client
|
||||
|
||||
PENDING_DIR = '/opt/so/state/push_pending'
|
||||
LOCK_FILE = os.path.join(PENDING_DIR, '.lock')
|
||||
LOG_FILE = '/opt/so/log/salt/so-push-drainer.log'
|
||||
|
||||
HIGHSTATE_SENTINEL = '__highstate__'
|
||||
|
||||
|
||||
def _make_logger():
|
||||
logger = logging.getLogger('so-push-drainer')
|
||||
logger.setLevel(logging.INFO)
|
||||
if not logger.handlers:
|
||||
os.makedirs(os.path.dirname(LOG_FILE), exist_ok=True)
|
||||
handler = logging.handlers.RotatingFileHandler(
|
||||
LOG_FILE, maxBytes=5 * 1024 * 1024, backupCount=3,
|
||||
)
|
||||
handler.setFormatter(logging.Formatter(
|
||||
'%(asctime)s | %(levelname)s | %(message)s',
|
||||
))
|
||||
logger.addHandler(handler)
|
||||
return logger
|
||||
|
||||
|
||||
def _load_push_cfg():
|
||||
"""Read the salt:auto_apply pillar subtree via salt-call. Returns a dict."""
|
||||
caller = salt.client.Caller()
|
||||
cfg = caller.cmd('pillar.get', 'salt:auto_apply', {})
|
||||
return cfg if isinstance(cfg, dict) else {}
|
||||
|
||||
|
||||
def _read_intent(path, log):
|
||||
try:
|
||||
with open(path, 'r') as f:
|
||||
return json.load(f)
|
||||
except (IOError, ValueError) as exc:
|
||||
log.warning('cannot read intent %s: %s', path, exc)
|
||||
return None
|
||||
except Exception:
|
||||
log.exception('unexpected error reading %s', path)
|
||||
return None
|
||||
|
||||
|
||||
def _dedupe_actions(actions):
|
||||
seen = set()
|
||||
deduped = []
|
||||
for action in actions:
|
||||
if not isinstance(action, dict):
|
||||
continue
|
||||
state_key = HIGHSTATE_SENTINEL if action.get('highstate') else action.get('state')
|
||||
tgt = action.get('tgt')
|
||||
tgt_type = action.get('tgt_type', 'compound')
|
||||
if not state_key or not tgt:
|
||||
continue
|
||||
key = (state_key, tgt, tgt_type)
|
||||
if key in seen:
|
||||
continue
|
||||
seen.add(key)
|
||||
deduped.append(action)
|
||||
return deduped
|
||||
|
||||
|
||||
def _dispatch(actions, log):
|
||||
pillar_arg = json.dumps({'actions': actions})
|
||||
cmd = [
|
||||
'salt-run',
|
||||
'state.orchestrate',
|
||||
'orch.push_batch',
|
||||
'pillar={}'.format(pillar_arg),
|
||||
'--async',
|
||||
]
|
||||
log.info('dispatching: %s', ' '.join(cmd[:3]) + ' pillar=<{} actions>'.format(len(actions)))
|
||||
try:
|
||||
result = subprocess.run(
|
||||
cmd, check=True, capture_output=True, text=True, timeout=60,
|
||||
)
|
||||
except subprocess.CalledProcessError as exc:
|
||||
log.error('dispatch failed (rc=%s): stdout=%s stderr=%s',
|
||||
exc.returncode, exc.stdout, exc.stderr)
|
||||
return False
|
||||
except subprocess.TimeoutExpired:
|
||||
log.error('dispatch timed out after 60s')
|
||||
return False
|
||||
except Exception:
|
||||
log.exception('dispatch raised')
|
||||
return False
|
||||
log.info('dispatch accepted: %s', (result.stdout or '').strip())
|
||||
return True
|
||||
|
||||
|
||||
def main():
|
||||
log = _make_logger()
|
||||
|
||||
if not os.path.isdir(PENDING_DIR):
|
||||
# Nothing to do; reactors create the dir on first use.
|
||||
return 0
|
||||
|
||||
try:
|
||||
push = _load_push_cfg()
|
||||
except Exception:
|
||||
log.exception('failed to read salt:auto_apply pillar; aborting drain pass')
|
||||
return 1
|
||||
|
||||
if not push.get('enabled', True):
|
||||
log.debug('push disabled; exiting')
|
||||
return 0
|
||||
|
||||
debounce_seconds = int(push.get('debounce_seconds', 30))
|
||||
|
||||
os.makedirs(PENDING_DIR, exist_ok=True)
|
||||
lock_fd = os.open(LOCK_FILE, os.O_CREAT | os.O_RDWR, 0o644)
|
||||
try:
|
||||
fcntl.flock(lock_fd, fcntl.LOCK_EX)
|
||||
|
||||
intent_files = [
|
||||
p for p in sorted(glob.glob(os.path.join(PENDING_DIR, '*.json')))
|
||||
if os.path.basename(p) != '.lock'
|
||||
]
|
||||
if not intent_files:
|
||||
return 0
|
||||
|
||||
now = time.time()
|
||||
ready = []
|
||||
skipped = 0
|
||||
broken = []
|
||||
for path in intent_files:
|
||||
intent = _read_intent(path, log)
|
||||
if not isinstance(intent, dict):
|
||||
broken.append(path)
|
||||
continue
|
||||
last_touch = intent.get('last_touch', 0)
|
||||
if now - last_touch < debounce_seconds:
|
||||
skipped += 1
|
||||
continue
|
||||
ready.append((path, intent))
|
||||
|
||||
for path in broken:
|
||||
try:
|
||||
os.unlink(path)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
if not ready:
|
||||
if skipped:
|
||||
log.debug('no ready intents (%d still in debounce window)', skipped)
|
||||
return 0
|
||||
|
||||
combined_actions = []
|
||||
oldest_first_touch = now
|
||||
all_paths = []
|
||||
for path, intent in ready:
|
||||
combined_actions.extend(intent.get('actions', []) or [])
|
||||
first = intent.get('first_touch', now)
|
||||
if first < oldest_first_touch:
|
||||
oldest_first_touch = first
|
||||
all_paths.extend(intent.get('paths', []) or [])
|
||||
|
||||
deduped = _dedupe_actions(combined_actions)
|
||||
if not deduped:
|
||||
log.warning('%d intent(s) had no usable actions; clearing', len(ready))
|
||||
for path, _ in ready:
|
||||
try:
|
||||
os.unlink(path)
|
||||
except OSError:
|
||||
pass
|
||||
return 0
|
||||
|
||||
debounce_duration = now - oldest_first_touch
|
||||
log.info(
|
||||
'draining %d intent(s): %d action(s) after dedupe (raw=%d), '
|
||||
'debounce_duration=%.1fs, paths=%s',
|
||||
len(ready), len(deduped), len(combined_actions),
|
||||
debounce_duration, all_paths[:20],
|
||||
)
|
||||
|
||||
if not _dispatch(deduped, log):
|
||||
log.warning('dispatch failed; leaving intent files in place for retry')
|
||||
return 1
|
||||
|
||||
for path, _ in ready:
|
||||
try:
|
||||
os.unlink(path)
|
||||
except OSError:
|
||||
log.exception('failed to remove drained intent %s', path)
|
||||
|
||||
return 0
|
||||
finally:
|
||||
try:
|
||||
fcntl.flock(lock_fd, fcntl.LOCK_UN)
|
||||
finally:
|
||||
os.close(lock_fd)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
sys.exit(main())
|
||||
@@ -124,8 +124,8 @@ copy_new_files() {
|
||||
|
||||
rsync -a salt $default_salt_dir/
|
||||
rsync -a pillar $default_salt_dir/
|
||||
chown -R socore:socore $default_salt_dir/salt
|
||||
chown -R socore:socore $default_salt_dir/pillar
|
||||
chown -R root:root $default_salt_dir/salt
|
||||
chown -R root:root $default_salt_dir/pillar
|
||||
chmod 755 $default_salt_dir/pillar/firewall/addfirewall.sh
|
||||
|
||||
rm -rf /tmp/sogh
|
||||
|
||||
@@ -0,0 +1,219 @@
|
||||
#!/bin/bash
|
||||
#
|
||||
# Copyright Security Onion Solutions LLC and/or licensed to Security Onion Solutions LLC under one
|
||||
# or more contributor license agreements. Licensed under the Elastic License 2.0 as shown at
|
||||
# https://securityonion.net/license; you may not use this file except in compliance with the
|
||||
# Elastic License 2.0.
|
||||
|
||||
# so-soup-grid-highstate
|
||||
# ======================
|
||||
# Drives a batched, role-tiered highstate across every non-manager minion in the
|
||||
# grid. soup fires this (detached) after it finishes upgrading the manager so the
|
||||
# rest of the grid converges immediately instead of waiting for its own scheduled
|
||||
# highstate -- which, since the schedule moved from 15 minutes to 120 minutes
|
||||
# (salt:schedule:highstate_interval_minutes), could otherwise leave nodes on the
|
||||
# old version for up to ~2.5 hours (interval + splay) while the manager runs new code.
|
||||
#
|
||||
# Work is done by the existing orch.push_batch orchestration (salt/orch/push_batch.sls),
|
||||
# the same runner the active-push drainer uses, so batching/queueing behavior matches.
|
||||
# Tiers are dispatched in declaration order: searchnodes/heavynodes (Elasticsearch data
|
||||
# nodes) first, then receivers, then everything else -- so the data tier converges before
|
||||
# the ingest tier before sensors/fleet/idh/etc.
|
||||
#
|
||||
# When soup also upgraded Salt itself, remote minions must first highstate onto the new
|
||||
# salt-minion package (top.sls gates every real state on G@saltversion, so a stale-version
|
||||
# minion only gets salt.minion until it upgrades and reconnects). --salt-upgraded runs that
|
||||
# preliminary pass and waits for the fleet to settle before the tiered pass.
|
||||
#
|
||||
# This is best-effort: soup has already completed by the time this runs, and the 120-minute
|
||||
# scheduled highstate remains the backstop for any node that is offline or missed a batch.
|
||||
|
||||
LOG_FILE=/opt/so/log/salt/so-soup-grid-highstate
|
||||
LOCK_FILE=/opt/so/state/so-soup-grid-highstate.lock
|
||||
SETTLE_MAX_WAIT=${GRID_HIGHSTATE_SETTLE_WAIT:-900} # backstop for the post-salt-upgrade settle loop
|
||||
SETTLE_INTERVAL=15
|
||||
SETTLE_STABLE_CHECKS=3
|
||||
# salt-minion on an upgraded node restarts ~30s after the upgrade state runs
|
||||
# (salt/salt/minion/init.sls start_minion_post_upgrade); wait past that before sampling
|
||||
# so the settle loop sees the drop-off instead of settling on the pre-restart set.
|
||||
SETTLE_INITIAL_WAIT=${GRID_HIGHSTATE_SETTLE_INITIAL_WAIT:-45}
|
||||
|
||||
BATCH=""
|
||||
BATCH_WAIT=""
|
||||
SALT_UPGRADED=false
|
||||
REASON="manual"
|
||||
|
||||
log() {
|
||||
echo "$(date '+%Y-%m-%d %H:%M:%S') | $*" | tee -a "$LOG_FILE"
|
||||
}
|
||||
|
||||
usage() {
|
||||
echo "Usage: so-soup-grid-highstate [--batch <spec>] [--batch-wait <sec>] [--salt-upgraded] [--reason <text>]"
|
||||
exit 1
|
||||
}
|
||||
|
||||
while [ $# -gt 0 ]; do
|
||||
case "$1" in
|
||||
--batch) BATCH="$2"; shift 2 ;;
|
||||
--batch-wait) BATCH_WAIT="$2"; shift 2 ;;
|
||||
--salt-upgraded) SALT_UPGRADED=true; shift ;;
|
||||
--reason) REASON="$2"; shift 2 ;;
|
||||
-h|--help) usage ;;
|
||||
*) echo "Unknown option: $1"; usage ;;
|
||||
esac
|
||||
done
|
||||
|
||||
mkdir -p "$(dirname "$LOG_FILE")" "$(dirname "$LOCK_FILE")"
|
||||
|
||||
# Serialize: a second invocation (e.g. two soups, or a manual run overlapping soup's)
|
||||
# should not dispatch a competing set of batches.
|
||||
exec 9>"$LOCK_FILE"
|
||||
if ! flock -n 9; then
|
||||
log "another so-soup-grid-highstate is already running (lock $LOCK_FILE held); exiting"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Resolve batch settings from the salt:auto_apply pillar when not overridden on the
|
||||
# command line, falling back to the same defaults orch.push_batch/salt.defaults use.
|
||||
if [ -z "$BATCH" ]; then
|
||||
BATCH=$(salt-call --out=newline_values_only pillar.get salt:auto_apply:batch 2>/dev/null)
|
||||
[ -z "$BATCH" ] && BATCH='10%'
|
||||
fi
|
||||
if [ -z "$BATCH_WAIT" ]; then
|
||||
BATCH_WAIT=$(salt-call --out=newline_values_only pillar.get salt:auto_apply:batch_wait 2>/dev/null)
|
||||
[ -z "$BATCH_WAIT" ] && BATCH_WAIT=15
|
||||
fi
|
||||
|
||||
MINIONID=$(salt-call --local --out=newline_values_only grains.get id 2>/dev/null)
|
||||
[ -z "$MINIONID" ] && MINIONID=$(cat /etc/salt/minion_id 2>/dev/null)
|
||||
if [ -z "$MINIONID" ]; then
|
||||
log "could not determine this minion's id; aborting"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Single-node grids (eval/standalone/import with no other accepted keys) have nothing
|
||||
# remote to push -- the manager already highstated during soup.
|
||||
NUM_ACCEPTED=$(salt-key --out=json --list=accepted 2>/dev/null | jq -r '.minions | length' 2>/dev/null)
|
||||
NUM_ACCEPTED=${NUM_ACCEPTED:-0}
|
||||
if [ "$NUM_ACCEPTED" -le 1 ]; then
|
||||
log "single node grid ($NUM_ACCEPTED accepted minion(s)); nothing to push (reason=$REASON)"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
log "starting grid highstate: reason=$REASON minion=$MINIONID accepted=$NUM_ACCEPTED batch=$BATCH batch_wait=$BATCH_WAIT salt_upgraded=$SALT_UPGRADED"
|
||||
|
||||
# Count minions currently responsive on the bus (includes this manager).
|
||||
count_up() {
|
||||
salt-run manage.up --out=json 2>/dev/null \
|
||||
| python3 -c 'import sys,json; print(len(json.load(sys.stdin)))' 2>/dev/null
|
||||
}
|
||||
|
||||
# Dispatch a single synchronous orch.push_batch run for the given actions JSON.
|
||||
# Synchronous is fine: soup launched us detached, so blocking here does not hold soup up.
|
||||
# expect_restart=true marks a dispatch (the salt-upgrade pass) where a non-zero rc is normal
|
||||
# because targets restart salt-minion mid-run -- so we don't log a misleading failure warning.
|
||||
dispatch() {
|
||||
local desc="$1"
|
||||
local actions="$2"
|
||||
local expect_restart="${3:-false}"
|
||||
local rc
|
||||
log "dispatching $desc"
|
||||
salt-run state.orchestrate orch.push_batch pillar="{\"actions\": $actions}" >>"$LOG_FILE" 2>&1
|
||||
rc=$?
|
||||
if [ "$rc" -eq 0 ]; then
|
||||
log "$desc dispatch completed (rc=0)"
|
||||
elif [ "$expect_restart" = "true" ]; then
|
||||
log "$desc returned rc=$rc; this is expected during a salt upgrade (targets restart salt-minion mid-run). Waiting for them to reconnect before the tiered pass."
|
||||
else
|
||||
log "WARNING: $desc dispatch returned rc=$rc; nodes it missed will converge on the scheduled highstate"
|
||||
fi
|
||||
}
|
||||
|
||||
# Wait for the reachable minion set to recover to its pre-upgrade size and hold steady.
|
||||
# Used after the salt-upgrade pass, where targets restart salt-minion (~30s delayed, see
|
||||
# salt/salt/minion/init.sls) and drop off the bus before reconnecting on the new version.
|
||||
# target = how many minions were reachable just before the pass; requiring up >= target keeps
|
||||
# us from releasing the tiered pass while nodes are still down for their restart (settling on
|
||||
# the not-yet-restarted subset). We deliberately compare against the pre-upgrade reachable
|
||||
# count, not accepted keys, so a node an operator intentionally powered off never stalls us.
|
||||
# Bounded by SETTLE_MAX_WAIT.
|
||||
wait_for_settle() {
|
||||
local target="$1"
|
||||
local elapsed=0 prev=-1 stable=0 up=0
|
||||
# Let the delayed salt-minion restart begin before we start counting stability, otherwise
|
||||
# we could see the pre-restart set as "stable" and settle before the drop-off even happens.
|
||||
sleep "$SETTLE_INITIAL_WAIT"
|
||||
elapsed=$SETTLE_INITIAL_WAIT
|
||||
while [ "$elapsed" -lt "$SETTLE_MAX_WAIT" ]; do
|
||||
up=$(count_up); up=${up:-0}
|
||||
if [ "$up" -ge "$target" ] && [ "$up" -eq "$prev" ]; then
|
||||
stable=$((stable + 1))
|
||||
[ "$stable" -ge "$SETTLE_STABLE_CHECKS" ] && break
|
||||
else
|
||||
stable=0
|
||||
fi
|
||||
prev=$up
|
||||
sleep "$SETTLE_INTERVAL"
|
||||
elapsed=$((elapsed + SETTLE_INTERVAL))
|
||||
done
|
||||
if [ "$up" -ge "$target" ]; then
|
||||
log "fleet recovered to ${up} minions up (>= pre-upgrade ${target}) after ${elapsed}s"
|
||||
else
|
||||
log "WARNING: ${SETTLE_MAX_WAIT}s settle backstop hit; only ${up}/${target} pre-upgrade minions back up; proceeding (stragglers converge on the scheduled highstate)"
|
||||
fi
|
||||
}
|
||||
|
||||
# Pass 0: when Salt itself was upgraded, remote minions still on the old version only match
|
||||
# top.sls's 'not G@saltversion' block (salt.minion, which performs the package upgrade). Push
|
||||
# an untiered highstate so they upgrade+reconnect, then wait for them to come back before the
|
||||
# real tiered pass applies the new version's states.
|
||||
if [ "$SALT_UPGRADED" = "true" ]; then
|
||||
PRE_UP=$(count_up); PRE_UP=${PRE_UP:-1}
|
||||
log "pre-upgrade reachable minions (incl. this manager): $PRE_UP"
|
||||
dispatch "salt-upgrade pass (all remote minions)" \
|
||||
"[{\"highstate\": true, \"tgt\": \"not $MINIONID\", \"tgt_type\": \"compound\", \"batch\": \"$BATCH\", \"batch_wait\": $BATCH_WAIT}]" \
|
||||
true
|
||||
log "waiting for minions to reconnect on the new salt version"
|
||||
wait_for_settle "$PRE_UP"
|
||||
fi
|
||||
|
||||
# Tiered pass: Elasticsearch data nodes first, then receivers, then the remainder. The last
|
||||
# tier is defined as the complement of the earlier tiers (and of this manager) so coverage is
|
||||
# exhaustive -- sensors, fleet, idh, desktop, hypervisor, and any future role are all included.
|
||||
TIER_TGTS=(
|
||||
"( *_searchnode or *_heavynode ) and not $MINIONID"
|
||||
"*_receiver and not $MINIONID"
|
||||
"not $MINIONID and not *_searchnode and not *_heavynode and not *_receiver"
|
||||
)
|
||||
|
||||
# Count minions a compound target matches, using the master's key/cache data (no execution).
|
||||
tier_count() {
|
||||
salt --out=json -C "$1" --preview-target 2>/dev/null | jq 'length' 2>/dev/null
|
||||
}
|
||||
|
||||
# Build the actions JSON, including only tiers that actually match minions. An empty target
|
||||
# would make orch.push_batch's salt.state step return "No minions returned" -- a failure --
|
||||
# even though nothing needed to run, and grids commonly lack a tier (no receiver, etc.).
|
||||
# Keep the JSON on a single line: salt parses `pillar=<value>` kwargs with a non-DOTALL
|
||||
# regex, so an embedded newline makes it treat the whole token as a positional saltenv
|
||||
# instead ("No matching salt environment for environment 'pillar=...'").
|
||||
actions=""
|
||||
for tgt in "${TIER_TGTS[@]}"; do
|
||||
n=$(tier_count "$tgt"); n=${n:-0}
|
||||
if [ "$n" -ge 1 ]; then
|
||||
[ -n "$actions" ] && actions="$actions, "
|
||||
actions="$actions{\"highstate\": true, \"tgt\": \"$tgt\", \"tgt_type\": \"compound\", \"batch\": \"$BATCH\", \"batch_wait\": $BATCH_WAIT}"
|
||||
log "tier matched $n minion(s): $tgt"
|
||||
else
|
||||
log "tier matched 0 minions, skipping: $tgt"
|
||||
fi
|
||||
done
|
||||
|
||||
if [ -z "$actions" ]; then
|
||||
log "no remote minions matched any tier; nothing to push (reason=$REASON)"
|
||||
exit 0
|
||||
fi
|
||||
dispatch "tiered pass (searchnodes/heavynodes -> receivers -> remainder)" "[$actions]"
|
||||
|
||||
log "grid highstate complete (reason=$REASON)"
|
||||
exit 0
|
||||
@@ -129,7 +129,8 @@ while [[ $# -gt 0 ]]; do
|
||||
esac
|
||||
done
|
||||
|
||||
kratosUrl=${KRATOS_URL:-http://127.0.0.1:4434/admin}
|
||||
kratosContainer=${KRATOS_CONTAINER:-so-kratos}
|
||||
kratosUrl=${KRATOS_URL:-http://localhost:4434/admin}
|
||||
databasePath=${KRATOS_DB_PATH:-/nsm/kratos/db/db.sqlite}
|
||||
databaseTimeout=${KRATOS_DB_TIMEOUT:-5000}
|
||||
bcryptRounds=${BCRYPT_ROUNDS:-12}
|
||||
@@ -154,6 +155,10 @@ function fail() {
|
||||
exit 1
|
||||
}
|
||||
|
||||
function kratosCurl() {
|
||||
docker exec "$kratosContainer" curl "$@"
|
||||
}
|
||||
|
||||
function require() {
|
||||
cmd=$1
|
||||
which "$1" 2>&1 > /dev/null
|
||||
@@ -164,18 +169,18 @@ function require() {
|
||||
function verifyEnvironment() {
|
||||
require "htpasswd"
|
||||
require "jq"
|
||||
require "curl"
|
||||
require "docker"
|
||||
require "openssl"
|
||||
require "sqlite3"
|
||||
[[ ! -f $databasePath ]] && fail "Unable to find database file; specify path via KRATOS_DB_PATH environment variable"
|
||||
response=$(curl -Ss -L ${kratosUrl}/)
|
||||
response=$(kratosCurl -Ss -L ${kratosUrl}/)
|
||||
[[ "$response" != "404 page not found" ]] && fail "Unable to communicate with Kratos; specify URL via KRATOS_URL environment variable"
|
||||
}
|
||||
|
||||
function findIdByEmail() {
|
||||
email=${1,,}
|
||||
|
||||
response=$(curl -Ss -L ${kratosUrl}/identities)
|
||||
response=$(kratosCurl -Ss -L ${kratosUrl}/identities)
|
||||
identityId=$(echo "${response}" | jq -r ".[] | select(.verifiable_addresses[0].value == \"$email\") | .id")
|
||||
echo $identityId
|
||||
}
|
||||
@@ -416,7 +421,7 @@ function syncAll() {
|
||||
}
|
||||
|
||||
function listUsers() {
|
||||
response=$(curl -Ss -L ${kratosUrl}/identities)
|
||||
response=$(kratosCurl -Ss -L ${kratosUrl}/identities)
|
||||
[[ $? != 0 ]] && fail "Unable to communicate with Kratos"
|
||||
|
||||
users=$(echo "${response}" | jq -r ".[] | .verifiable_addresses[0].value" | sort)
|
||||
@@ -495,7 +500,7 @@ function createUser() {
|
||||
EOF
|
||||
)
|
||||
|
||||
response=$(curl -Ss -L ${kratosUrl}/identities -d "$addUserJson")
|
||||
response=$(kratosCurl -Ss -L ${kratosUrl}/identities -d "$addUserJson")
|
||||
[[ $? != 0 ]] && fail "Unable to communicate with Kratos"
|
||||
|
||||
identityId=$(echo "${response}" | jq -r ".id")
|
||||
@@ -518,7 +523,7 @@ function updateStatus() {
|
||||
identityId=$(findIdByEmail "$email")
|
||||
[[ ${identityId} == "" ]] && fail "User not found"
|
||||
|
||||
response=$(curl -Ss -L "${kratosUrl}/identities/$identityId")
|
||||
response=$(kratosCurl -Ss -L "${kratosUrl}/identities/$identityId")
|
||||
[[ $? != 0 ]] && fail "Unable to communicate with Kratos"
|
||||
|
||||
schemaId=$(echo "$response" | jq -r .schema_id)
|
||||
@@ -531,7 +536,7 @@ function updateStatus() {
|
||||
state="inactive"
|
||||
fi
|
||||
body="{ \"schema_id\": \"$schemaId\", \"state\": \"$state\", \"traits\": $traitBlock }"
|
||||
response=$(curl -fSsL -XPUT -H "Content-Type: application/json" "${kratosUrl}/identities/$identityId" -d "$body")
|
||||
response=$(kratosCurl -fSsL -XPUT -H "Content-Type: application/json" "${kratosUrl}/identities/$identityId" -d "$body")
|
||||
[[ $? != 0 ]] && fail "Unable to update user"
|
||||
}
|
||||
|
||||
@@ -550,7 +555,7 @@ function updateUserProfile() {
|
||||
identityId=$(findIdByEmail "$email")
|
||||
[[ ${identityId} == "" ]] && fail "User not found"
|
||||
|
||||
response=$(curl -Ss -L "${kratosUrl}/identities/$identityId")
|
||||
response=$(kratosCurl -Ss -L "${kratosUrl}/identities/$identityId")
|
||||
[[ $? != 0 ]] && fail "Unable to communicate with Kratos"
|
||||
|
||||
schemaId=$(echo "$response" | jq -r .schema_id)
|
||||
@@ -559,7 +564,7 @@ function updateUserProfile() {
|
||||
traitBlock="{\"email\":\"$email\",\"firstName\":\"$firstName\",\"lastName\":\"$lastName\",\"note\":\"$note\"}"
|
||||
|
||||
body="{ \"schema_id\": \"$schemaId\", \"state\": \"$state\", \"traits\": $traitBlock }"
|
||||
response=$(curl -fSsL -XPUT -H "Content-Type: application/json" "${kratosUrl}/identities/$identityId" -d "$body")
|
||||
response=$(kratosCurl -fSsL -XPUT -H "Content-Type: application/json" "${kratosUrl}/identities/$identityId" -d "$body")
|
||||
[[ $? != 0 ]] && fail "Unable to update user"
|
||||
}
|
||||
|
||||
@@ -569,7 +574,7 @@ function deleteUser() {
|
||||
identityId=$(findIdByEmail "$email")
|
||||
[[ ${identityId} == "" ]] && fail "User not found"
|
||||
|
||||
response=$(curl -Ss -XDELETE -L "${kratosUrl}/identities/$identityId")
|
||||
response=$(kratosCurl -Ss -XDELETE -L "${kratosUrl}/identities/$identityId")
|
||||
[[ $? != 0 ]] && fail "Unable to communicate with Kratos"
|
||||
|
||||
rolesTmpFile="${socRolesFile}.tmp"
|
||||
|
||||
+479
-66
@@ -12,9 +12,23 @@
|
||||
UPDATE_DIR=/tmp/sogh/securityonion
|
||||
DEFAULT_SALT_DIR=/opt/so/saltstack/default
|
||||
INSTALLEDVERSION=$(cat /etc/soversion)
|
||||
POSTVERSION=$INSTALLEDVERSION
|
||||
# /etc/sopostversion is a soup-owned marker (no salt state manages it) tracking how
|
||||
# far the post-upgrade walk has progressed. Its presence means a prior upgrade did
|
||||
# not finish its post-upgrade steps; its contents are the resume point. It is read
|
||||
# here before preupgrade_changes mutates INSTALLEDVERSION and before any highstate
|
||||
# stamps /etc/soversion from the pillar.
|
||||
POSTVERSION_FILE=/etc/sopostversion
|
||||
if [ -f "$POSTVERSION_FILE" ]; then
|
||||
POSTVERSION=$(cat "$POSTVERSION_FILE")
|
||||
else
|
||||
POSTVERSION=$INSTALLEDVERSION
|
||||
fi
|
||||
INSTALLEDSALTVERSION=$(salt --versions-report | grep Salt: | awk '{print $2}')
|
||||
BATCHSIZE=5
|
||||
# Optional -b override for the grid highstate batch size (a count like "5" or a
|
||||
# percentage like "25%"). Empty means so-soup-grid-highstate uses the salt:auto_apply:batch
|
||||
# pillar default.
|
||||
BATCHSIZE=
|
||||
DEFAULT_DOCKER_RANGE='172.17.1.0/24'
|
||||
SOUP_LOG=/root/soup.log
|
||||
SOUP_DEBUG_LOG=/root/soup-debug.log
|
||||
WHATWOULDYOUSAYYAHDOHERE=soup
|
||||
@@ -23,6 +37,10 @@ NOTIFYCUSTOMELASTICCONFIG=false
|
||||
TOPFILE=/opt/so/saltstack/default/salt/top.sls
|
||||
BACKUPTOPFILE=/opt/so/saltstack/default/salt/top.sls.backup
|
||||
SALTUPGRADED=false
|
||||
# Set true once soup begins modifying the system (past the pre-flight checks), so the
|
||||
# EXIT trap can tell the user the update did not finish and must be re-run. Only the
|
||||
# pre-flight gates (ES compatibility, disk, network) fail before this is set.
|
||||
SOUP_UPGRADE_STARTED=false
|
||||
SALT_CLOUD_INSTALLED=false
|
||||
SALT_CLOUD_CONFIGURED=false
|
||||
# Check if salt-cloud is installed
|
||||
@@ -103,6 +121,9 @@ check_err() {
|
||||
161)
|
||||
echo 'Required intermediate Elasticsearch upgrade not complete'
|
||||
;;
|
||||
162)
|
||||
echo 'One or more Elastic Agent nodes do not support the x86-64-v3 CPU instruction set'
|
||||
;;
|
||||
170)
|
||||
echo "Intermediate upgrade completed successfully to $next_step_so_version, but next soup to Security Onion $originally_requested_so_version could not be started automatically."
|
||||
echo "Start soup again manually to continue the upgrade to Security Onion $originally_requested_so_version."
|
||||
@@ -123,6 +144,28 @@ check_err() {
|
||||
|
||||
echo "SOUP XTRACE debug log (if enabled) at $SOUP_DEBUG_LOG. Re-run soup with SOUP_DEBUG=1 to create $SOUP_DEBUG_LOG"
|
||||
|
||||
# If soup had already started modifying the system, make it unmistakable that the
|
||||
# update is incomplete and must be re-run. soup is resumable: a version upgrade
|
||||
# picks up from the /etc/sopostversion marker, and a hotfix re-applies because
|
||||
# /etc/sohotfix is only advanced after a successful highstate.
|
||||
if [[ "$SOUP_UPGRADE_STARTED" == "true" ]]; then
|
||||
echo ""
|
||||
echo "=============================================================================="
|
||||
echo " UPGRADE INCOMPLETE"
|
||||
echo "=============================================================================="
|
||||
echo " This soup run did NOT finish. Your Security Onion installation may be in a"
|
||||
echo " partially-updated state and is not yet fully upgraded."
|
||||
echo ""
|
||||
echo " Review the error above and $SOUP_LOG, resolve the underlying problem, then"
|
||||
echo " run soup again to resume and complete the update:"
|
||||
echo ""
|
||||
echo " sudo soup"
|
||||
echo ""
|
||||
echo " soup is resumable -- re-running it continues from where this run stopped."
|
||||
echo "=============================================================================="
|
||||
echo ""
|
||||
fi
|
||||
|
||||
exit $exit_code
|
||||
fi
|
||||
|
||||
@@ -291,6 +334,112 @@ check_pillar_items() {
|
||||
fi
|
||||
}
|
||||
|
||||
check_cluster_health() {
|
||||
# Require a 'green' cluster before upgrading
|
||||
echo "Checking Elasticsearch cluster health."
|
||||
if so-elasticsearch-query "_cluster/health?wait_for_status=green&timeout=120s" --fail > /dev/null 2>&1; then
|
||||
printf "\nThe Elasticsearch cluster is healthy (green). We can proceed with SOUP.\n\n"
|
||||
return
|
||||
fi
|
||||
|
||||
if command -v so-elasticsearch-troubleshoot > /dev/null 2>&1; then
|
||||
printf "\nRunning so-elasticsearch-troubleshoot for additional detail.\n"
|
||||
so-elasticsearch-troubleshoot || true
|
||||
fi
|
||||
|
||||
printf "\nThe Elasticsearch cluster is not green. Resolve the cluster health issue so the cluster is green before running SOUP again.\n\n"
|
||||
exit 0
|
||||
}
|
||||
|
||||
no_soup_for_you() {
|
||||
echo ""
|
||||
echo "No soup for you!"
|
||||
exit 162
|
||||
}
|
||||
|
||||
check_cpu_compatibility() {
|
||||
# Roles running a container built from the so-elastic-agent image; mirrors the
|
||||
# elasticagent and elasticfleet entries in salt/reactor/pillar_push_map.yaml.
|
||||
local cpu_target='G@role:so-heavynode or G@role:so-eval or G@role:so-fleet or G@role:so-import or G@role:so-manager or G@role:so-managerhype or G@role:so-managersearch or G@role:so-standalone'
|
||||
local expected_nodes cpu_results node result confirm
|
||||
local -a unsupported=() offline=()
|
||||
|
||||
echo "Checking that Elastic Agent nodes support the x86-64-v3 CPU instruction set now required by Elastic."
|
||||
|
||||
if [[ "$SKIP_CPU_CHECK" == "true" ]]; then
|
||||
printf "\nSkipping the x86-64-v3 CPU check because --skip-cpu-check was specified.\n\n"
|
||||
return
|
||||
fi
|
||||
|
||||
# Nodes that never answer are absent from the results, so diff against who should have.
|
||||
expected_nodes=$(salt -C "$cpu_target" --preview-target --out=json 2>/dev/null | jq -r '.[]?') || true
|
||||
if [[ -z "$expected_nodes" ]]; then
|
||||
printf "\nCould not determine which nodes run the Elastic Agent, so the x86-64-v3 CPU check cannot run.\n"
|
||||
no_soup_for_you
|
||||
fi
|
||||
|
||||
cpu_results=$(salt -t 30 -C "$cpu_target" cmd.run "/lib64/ld-linux-x86-64.so.2 --help | grep x86-64-v3" --out=json 2>/dev/null) || true
|
||||
|
||||
while IFS= read -r node; do
|
||||
[[ -z "$node" ]] && continue
|
||||
result=$(jq -r --arg node "$node" '.[$node] // empty' <<< "$cpu_results" 2>/dev/null)
|
||||
if [[ -z "$result" || "$result" == *"did not return"* ]]; then
|
||||
offline+=("$node")
|
||||
elif [[ "$result" != *"x86-64-v3 (supported"* ]]; then
|
||||
# glibc appends "(supported, searched)" only when supported; the open paren keeps
|
||||
# this from matching a future "(unsupported".
|
||||
unsupported+=("$node")
|
||||
fi
|
||||
done <<< "$expected_nodes"
|
||||
|
||||
if [[ ${#unsupported[@]} -eq 0 && ${#offline[@]} -eq 0 ]]; then
|
||||
printf "\nAll Elastic Agent nodes support x86-64-v3. We can proceed with SOUP.\n\n"
|
||||
return
|
||||
fi
|
||||
|
||||
echo ""
|
||||
if [[ ${#unsupported[@]} -gt 0 ]]; then
|
||||
echo "The following node(s) do NOT support the x86-64-v3 CPU instruction set:"
|
||||
printf ' %s\n' "${unsupported[@]}"
|
||||
echo ""
|
||||
echo "Upstream Elastic now builds its binaries for x86-64-v3, so these nodes can no"
|
||||
echo "longer run Elastic. Upgrading them WILL BREAK them."
|
||||
echo ""
|
||||
fi
|
||||
if [[ ${#offline[@]} -gt 0 ]]; then
|
||||
echo "The following node(s) did not respond and could not be checked:"
|
||||
printf ' %s\n' "${offline[@]}"
|
||||
echo ""
|
||||
echo "These nodes are offline, so we cannot confirm they support x86-64-v3, which"
|
||||
echo "upstream Elastic now requires."
|
||||
echo ""
|
||||
fi
|
||||
|
||||
if [[ -n $UNATTENDED ]]; then
|
||||
echo "Unattended mode cannot prompt for an override. Re-run soup interactively, or pass --skip-cpu-check to bypass this check."
|
||||
no_soup_for_you
|
||||
fi
|
||||
|
||||
read -rp "Type 'override' to continue anyway, or press Enter to exit: " confirm
|
||||
if [[ "${confirm,,}" == "override" ]]; then
|
||||
printf "\nOverride accepted. Continuing at your own risk.\n\n"
|
||||
else
|
||||
no_soup_for_you
|
||||
fi
|
||||
}
|
||||
|
||||
check_fleet_server() {
|
||||
echo "Checking that Elastic Fleet Server is responding."
|
||||
# Modeled on the wait_for_so-elastic-fleet state check in elasticfleet/enabled.sls,
|
||||
# which waits for HTTP 200 from the Fleet Server status API.
|
||||
if curl -sk --fail --retry 3 --retry-delay 10 --max-time 30 "https://localhost:8220/api/status" > /dev/null 2>&1; then
|
||||
printf "\nElastic Fleet Server is responding. We can proceed with SOUP.\n\n"
|
||||
else
|
||||
printf "\nElastic Fleet Server is not responding at https://localhost:8220/api/status. Please ensure Elastic Fleet is healthy before running SOUP again.\n\n"
|
||||
exit 0
|
||||
fi
|
||||
}
|
||||
|
||||
check_saltmaster_status() {
|
||||
set +e
|
||||
echo "Waiting on the Salt Master service to be ready."
|
||||
@@ -377,21 +526,66 @@ get_soup_script_hashes() {
|
||||
GITIMGCMN=$(md5sum $UPDATE_DIR/salt/common/tools/sbin/so-image-common | awk '{print $1}')
|
||||
CURRENTSOFIREWALL=$(md5sum /usr/sbin/so-firewall | awk '{print $1}')
|
||||
GITSOFIREWALL=$(md5sum $UPDATE_DIR/salt/manager/tools/sbin/so-firewall | awk '{print $1}')
|
||||
CURRENTSOYAML=$(md5sum /usr/sbin/so-yaml.py | awk '{print $1}')
|
||||
GITSOYAML=$(md5sum $UPDATE_DIR/salt/manager/tools/sbin/so-yaml.py | awk '{print $1}')
|
||||
}
|
||||
|
||||
highstate() {
|
||||
# Run a highstate.
|
||||
# Run a highstate with a retry attempt.
|
||||
if salt-call state.highstate -l info queue=True; then
|
||||
return 0
|
||||
fi
|
||||
|
||||
echo "Initial highstate attempt had a problem; retrying in 30 seconds."
|
||||
sleep 30
|
||||
salt-call state.highstate -l info queue=True
|
||||
}
|
||||
|
||||
upgrade_searchnode_elasticsearch() {
|
||||
# Run the elasticsearch state across the true elastic cluster (non-heavy) with a retry attempt
|
||||
# Excludes the manager, so that kibana & elasticfleet are not upgraded until searchnodes are upgraded.
|
||||
echo "Getting ready to upgrade Elasticsearch across the grid. This may take a while..."
|
||||
if salt -b 10% -C "I@elasticsearch:enabled and not G@role:so-heavynode and not G@id:${MINIONID}" state.apply elasticsearch queue=True; then
|
||||
return 0
|
||||
fi
|
||||
|
||||
echo "Initial elasticsearch state attempt had a problem; retrying in 30 seconds."
|
||||
sleep 30
|
||||
salt -b 10% -C "I@elasticsearch:enabled and not G@role:so-heavynode and not G@id:${MINIONID}" state.apply elasticsearch queue=True
|
||||
}
|
||||
|
||||
push_grid_highstate() {
|
||||
# Drive a batched, role-tiered highstate across the rest of the grid so remote minions
|
||||
# pick up this upgrade now instead of waiting up to ~2.5 hours for their own scheduled
|
||||
# highstate (the schedule moved from 15 to 120 minutes). so-soup-grid-highstate does the work
|
||||
# via orch.push_batch; it only exists once the manager highstate has deployed this
|
||||
# version's sbin files, so guard on it. Launch fully detached (setsid) so it survives an
|
||||
# SSH drop, and never let it affect soup's exit status -- it is best-effort with the
|
||||
# scheduled highstate as backstop.
|
||||
if [[ ! -x /usr/sbin/so-soup-grid-highstate ]]; then
|
||||
echo "so-soup-grid-highstate not present; remote nodes will converge on their scheduled highstate."
|
||||
return 0
|
||||
fi
|
||||
|
||||
local extra_args=()
|
||||
if [[ $SALTUPGRADED == true || $UPGRADESALT -eq 1 ]]; then
|
||||
extra_args+=(--salt-upgraded)
|
||||
fi
|
||||
if [[ -n "$BATCHSIZE" ]]; then
|
||||
extra_args+=(--batch "$BATCHSIZE")
|
||||
fi
|
||||
|
||||
echo "Dispatching a grid-wide highstate to remote nodes. Progress: /opt/so/log/salt/so-soup-grid-highstate"
|
||||
setsid nohup /usr/sbin/so-soup-grid-highstate --reason soup "${extra_args[@]}" >/dev/null 2>&1 &
|
||||
}
|
||||
|
||||
masterlock() {
|
||||
echo "Locking Salt Master"
|
||||
mv -v $TOPFILE $BACKUPTOPFILE
|
||||
# Render the real top file only for the host running soup; every other
|
||||
# minion gets an empty top (no states) while the master is upgrading.
|
||||
echo "{% if grains['id'] == '$MINIONID' %}" > $TOPFILE
|
||||
cat $BACKUPTOPFILE >> $TOPFILE
|
||||
echo "{% endif %}" >> $TOPFILE
|
||||
echo "base:" > $TOPFILE
|
||||
echo " $MINIONID:" >> $TOPFILE
|
||||
echo " - ca" >> $TOPFILE
|
||||
echo " - elasticsearch" >> $TOPFILE
|
||||
}
|
||||
|
||||
masterunlock() {
|
||||
@@ -411,9 +605,18 @@ preupgrade_changes() {
|
||||
[[ "$INSTALLEDVERSION" =~ ^2\.4\.21[0-9]+$ ]] && up_to_3.0.0
|
||||
[[ "$INSTALLEDVERSION" == "3.0.0" ]] && up_to_3.1.0
|
||||
[[ "$INSTALLEDVERSION" == "3.1.0" ]] && up_to_3.2.0
|
||||
[[ "$INSTALLEDVERSION" == "3.2.0" ]] && up_to_3.3.0
|
||||
[[ "$INSTALLEDVERSION" == "3.3.0" ]] && up_to_3.4.0
|
||||
true
|
||||
}
|
||||
|
||||
set_postversion() {
|
||||
# Persist post-upgrade walk progress so an interrupted upgrade can resume the
|
||||
# remaining steps on the next soup run (see /etc/sopostversion handling).
|
||||
POSTVERSION="$1"
|
||||
echo "$POSTVERSION" > "$POSTVERSION_FILE"
|
||||
}
|
||||
|
||||
postupgrade_changes() {
|
||||
# This function is to add any new pillar items if needed.
|
||||
echo "Running post upgrade processes."
|
||||
@@ -421,6 +624,10 @@ postupgrade_changes() {
|
||||
[[ "$POSTVERSION" =~ ^2\.4\.21[0-9]+$ ]] && post_to_3.0.0
|
||||
[[ "$POSTVERSION" == "3.0.0" ]] && post_to_3.1.0
|
||||
[[ "$POSTVERSION" == "3.1.0" ]] && post_to_3.2.0
|
||||
[[ "$POSTVERSION" == "3.2.0" ]] && post_to_3.3.0
|
||||
[[ "$POSTVERSION" == "3.3.0" ]] && post_to_3.4.0
|
||||
# All applicable post-upgrade steps completed; clear the resume marker.
|
||||
rm -f "$POSTVERSION_FILE"
|
||||
true
|
||||
}
|
||||
|
||||
@@ -513,7 +720,7 @@ post_to_3.0.0() {
|
||||
# convert yes/no in suricata pillars to true/false
|
||||
convert_suricata_yes_no
|
||||
|
||||
POSTVERSION=3.0.0
|
||||
set_postversion 3.0.0
|
||||
}
|
||||
|
||||
### 3.0.0 End ###
|
||||
@@ -691,6 +898,21 @@ ensure_postgres_local_pillar() {
|
||||
chown -R socore:socore "$dir"
|
||||
}
|
||||
|
||||
ensure_salt_local_pillar() {
|
||||
# The salt.auto_apply settings are a new SOC settings
|
||||
# module, so the new pillar/top.sls references salt.soc_salt / salt.adv_salt
|
||||
# unconditionally. Managers upgrading from before this change have no
|
||||
# /opt/so/saltstack/local/pillar/salt/ (make_some_dirs only runs at install
|
||||
# time), so the stubs must be created here before salt-master restarts against
|
||||
# the new top.sls.
|
||||
echo "Ensuring salt local pillar stubs exist."
|
||||
local dir=/opt/so/saltstack/local/pillar/salt
|
||||
mkdir -p "$dir"
|
||||
[[ -f "$dir/soc_salt.sls" ]] || touch "$dir/soc_salt.sls"
|
||||
[[ -f "$dir/adv_salt.sls" ]] || touch "$dir/adv_salt.sls"
|
||||
chown -R socore:socore "$dir"
|
||||
}
|
||||
|
||||
ensure_postgres_secret() {
|
||||
# On a fresh install, generate_passwords + secrets_pillar seed
|
||||
# secrets:postgres_pass in /opt/so/saltstack/local/pillar/secrets.sls. That
|
||||
@@ -740,7 +962,6 @@ fix_logstash_0013_lumberjack_pipeline_name() {
|
||||
up_to_3.1.0() {
|
||||
ensure_postgres_local_pillar
|
||||
ensure_postgres_secret
|
||||
determine_elastic_agent_upgrade
|
||||
elasticsearch_backup_index_templates
|
||||
# Clear existing component template state file.
|
||||
rm -f /opt/so/state/esfleet_component_templates.json
|
||||
@@ -777,7 +998,7 @@ post_to_3.1.0() {
|
||||
# Check for unhealthy / unauthorized integration transform jobs and attempt reauthorizations
|
||||
check_transform_health_and_reauthorize || true
|
||||
|
||||
POSTVERSION=3.1.0
|
||||
set_postversion 3.1.0
|
||||
}
|
||||
|
||||
### 3.1.0 End ###
|
||||
@@ -787,7 +1008,7 @@ post_to_3.1.0() {
|
||||
recollate_postgres() {
|
||||
echo ""
|
||||
echo "Recollating PostgreSQL databases. The following output may contain warnings about a version mismatch, followed by a note indicating that the collation version has been changed."
|
||||
for db in postgres securityonion so_telegraf; do
|
||||
for db in template1 postgres securityonion so_telegraf; do
|
||||
docker exec so-postgres psql -U postgres $db -c "reindex database $db"
|
||||
docker exec so-postgres psql -U postgres $db -c "alter database $db refresh collation version"
|
||||
done
|
||||
@@ -795,29 +1016,21 @@ recollate_postgres() {
|
||||
echo ""
|
||||
}
|
||||
|
||||
bootstrap_so_soc_database() {
|
||||
# init-db.sh is mounted into so-postgres at /docker-entrypoint-initdb.d/init-db.sh
|
||||
# and runs automatically only on a fresh data directory. Hosts upgrading from
|
||||
# 3.1.0 already have /nsm/postgres populated, so the so_soc bootstrap block
|
||||
# added in 3.2 never fires. Re-run the script explicitly; it's idempotent.
|
||||
echo "Bootstrapping database via init-db.sh."
|
||||
# The postgres image has no USER directive, so `docker exec` defaults to
|
||||
# root, and the container env intentionally omits POSTGRES_USER (the upstream
|
||||
# entrypoint defaults it transiently during first-init only). Recreate both
|
||||
# so psql inside init-db.sh resolves the connect user correctly.
|
||||
local exec_cmd="docker exec -u postgres -e POSTGRES_USER=postgres so-postgres bash /docker-entrypoint-initdb.d/init-db.sh"
|
||||
if ! /usr/sbin/so-postgres-wait; then
|
||||
FINAL_MESSAGE_QUEUE+=("WARNING: so-postgres was not ready during the 3.2.0 upgrade; the so_soc database may not have been bootstrapped. Re-run manually: $exec_cmd")
|
||||
scrub_postgres_log_passwords() {
|
||||
# Purge plaintext passwords a pre-3.2 postgres could log on DDL errors.
|
||||
local log=/opt/so/log/postgres/postgres.log
|
||||
[[ -f "$log" ]] || return 0
|
||||
if ! grep -qai "PASSWORD '" "$log" 2>/dev/null; then
|
||||
echo "No leaked passwords found in $log."
|
||||
return 0
|
||||
fi
|
||||
if ! $exec_cmd; then
|
||||
FINAL_MESSAGE_QUEUE+=("WARNING: init-db.sh failed inside so-postgres during the 3.2.0 upgrade; the database may not have been bootstrapped. Re-run manually: $exec_cmd")
|
||||
return 0
|
||||
fi
|
||||
echo "Database bootstrap complete."
|
||||
|
||||
echo "Restarting so-soc container to pick up database changes"
|
||||
docker restart so-soc
|
||||
echo "Removing leaked password statements from $log."
|
||||
local tmp
|
||||
tmp=$(mktemp)
|
||||
# Rewrite in place (cat >) to keep the inode postgres is writing to.
|
||||
grep -avi "PASSWORD '" "$log" > "$tmp" 2>/dev/null || true
|
||||
cat "$tmp" > "$log"
|
||||
rm -f "$tmp"
|
||||
}
|
||||
|
||||
# Existing grids should keep ILM unless an admin explicitly opts in to DLM.
|
||||
@@ -888,10 +1101,16 @@ update_kafka_metadata() {
|
||||
}
|
||||
|
||||
up_to_3.2.0() {
|
||||
ensure_salt_local_pillar
|
||||
|
||||
fix_logstash_0013_lumberjack_pipeline_name
|
||||
|
||||
pin_elasticsearch_data_retention_method
|
||||
|
||||
# Run so-elastic-fleet-es-url update with --force to ensure eval/import have
|
||||
# configured so-manager_elasicsearch as the default output for both monitoring and logs
|
||||
/usr/sbin/so-elastic-fleet-es-url-update --force
|
||||
|
||||
INSTALLEDVERSION=3.2.0
|
||||
}
|
||||
|
||||
@@ -899,20 +1118,139 @@ post_to_3.2.0() {
|
||||
# Recollate due to image OS rebase
|
||||
recollate_postgres
|
||||
|
||||
bootstrap_so_soc_database
|
||||
|
||||
# Including agent regen script here since it was missed in post_to_3.1.0
|
||||
echo "Regenerating Elastic Agent Installers"
|
||||
/sbin/so-elastic-agent-gen-installers
|
||||
# SOC database bootstrap is handled by the postgres.enabled highstate.
|
||||
scrub_postgres_log_passwords
|
||||
|
||||
kibana_backport_streams_index_template
|
||||
|
||||
update_kafka_metadata "4.3"
|
||||
|
||||
POSTVERSION=3.2.0
|
||||
set_postversion 3.2.0
|
||||
}
|
||||
### 3.2.0 End ###
|
||||
|
||||
### 3.3.0 Scripts ###
|
||||
up_to_3.3.0() {
|
||||
# download 9.4.5 elastic agent packages
|
||||
determine_elastic_agent_upgrade
|
||||
|
||||
# remove existing (patched) elasticsearch index template to match integration naming change
|
||||
if ! remove_elasticsearch_index_template "so-logs-sentinel_one_cloud_funnel.login" "sentinel_one_cloud_funnel.login changed to sentinel_one_cloud_funnel.logins"; then
|
||||
FINAL_MESSAGE_QUEUE+=("WARNING: Unable to automatically remove the so-logs-sentinel_one_cloud_funnel.login index template. This step can be performed manually using the following command:")
|
||||
FINAL_MESSAGE_QUEUE+=(" - sudo so-elasticsearch-query _index_template/so-logs-sentinel_one_cloud_funnel.login -XDELETE && so-checkin")
|
||||
fi
|
||||
|
||||
INSTALLEDVERSION=3.3.0
|
||||
}
|
||||
|
||||
### 3.2.0 End ###
|
||||
telegraf_repair() {
|
||||
# Only grids whose Telegraf partitions stalled need this; --check exits 1
|
||||
# when there is something to repair, so everyone else is left alone.
|
||||
local repair=/usr/sbin/so-telegraf-repair
|
||||
[[ -x "$repair" ]] || return 0
|
||||
docker ps --format '{{.Names}}' | grep -qx so-postgres || return 0
|
||||
|
||||
echo "Checking Telegraf metric partitions."
|
||||
local status=0
|
||||
"$repair" --check >> "$SOUP_LOG" 2>&1 || status=$?
|
||||
case "$status" in
|
||||
0) echo " Telegraf partitions are healthy; nothing to repair." ;;
|
||||
1) echo " Repairing stalled Telegraf partitions."
|
||||
"$repair" --yes \
|
||||
|| echo " warning: so-telegraf-repair failed; run it manually" >&2 ;;
|
||||
*) echo " Skipping; Telegraf is not storing metrics in Postgres on this host." ;;
|
||||
esac
|
||||
}
|
||||
|
||||
post_to_3.3.0() {
|
||||
# Recollate again since some internal DBs were excluded during 3.2.0 soup
|
||||
recollate_postgres
|
||||
|
||||
# Generate 9.4.5 elastic agent installers
|
||||
echo "Regenerating Elastic Agent Installers"
|
||||
/sbin/so-elastic-agent-gen-installers
|
||||
|
||||
telegraf_repair
|
||||
|
||||
set_postversion 3.3.0
|
||||
}
|
||||
### 3.3.0 End ###
|
||||
|
||||
### 3.4.0 Scripts ###
|
||||
up_to_3.4.0() {
|
||||
set_soauth_range
|
||||
|
||||
echo "Removing so-kratos, so-hydra and so-soc so they are recreated on the soauth network."
|
||||
docker rm -f so-kratos so-hydra so-soc >> $SOUP_LOG 2>&1
|
||||
|
||||
INSTALLEDVERSION=3.4.0
|
||||
}
|
||||
|
||||
set_soauth_range() {
|
||||
local pillar_file=/opt/so/saltstack/local/pillar/docker/soc_docker.sls
|
||||
local current_range suggested authnet authgw input
|
||||
|
||||
[[ -f "$pillar_file" ]] || return 0
|
||||
|
||||
current_range=$(so-yaml.py get -r "$pillar_file" docker.range 2>/dev/null) || return 0
|
||||
|
||||
# A default range gets the 172.17.2.0/24 from docker/defaults.yaml, same as a fresh
|
||||
# install, so there is nothing to ask about.
|
||||
[[ -n "$current_range" && "$current_range" != "$DEFAULT_DOCKER_RANGE" ]] || return 0
|
||||
|
||||
if so-yaml.py get -r "$pillar_file" docker.networks.soauth.range >/dev/null 2>&1; then
|
||||
return 0
|
||||
fi
|
||||
|
||||
suggested=$(echo "${current_range%%/*}" | awk -F'.' '{ printf "%s.%s.%s.%s", $1, $2, ($3 + 1) % 256, $4 }')
|
||||
|
||||
if [[ -z $UNATTENDED ]]; then
|
||||
echo ""
|
||||
echo "This grid uses a custom Docker range ($current_range). The authentication"
|
||||
echo "services are moving to their own isolated network, which needs a second /24"
|
||||
echo "that does not overlap it."
|
||||
echo ""
|
||||
while :; do
|
||||
read -rp "Enter the network without the /24 suffix, or press Enter for ${suggested}: " input
|
||||
[[ -z "$input" ]] && input="$suggested"
|
||||
if valid_soauth_range "$input" "$current_range"; then
|
||||
authnet="$input"
|
||||
break
|
||||
fi
|
||||
echo "That range must be a valid IPv4 network, must not be within 172.17.0.0/24, and must not overlap ${current_range}."
|
||||
done
|
||||
else
|
||||
if ! valid_soauth_range "$suggested" "$current_range"; then
|
||||
FINAL_MESSAGE_QUEUE+=("WARNING: Unable to pick a range for the authentication network alongside $current_range. Set it manually before the next highstate:")
|
||||
FINAL_MESSAGE_QUEUE+=(" - so-yaml.py add $pillar_file docker.networks.soauth.range <network>/24")
|
||||
FINAL_MESSAGE_QUEUE+=(" - so-yaml.py add $pillar_file docker.networks.soauth.gateway <gateway>")
|
||||
return 0
|
||||
fi
|
||||
authnet="$suggested"
|
||||
FINAL_MESSAGE_QUEUE+=("NOTE: The authentication services moved to an isolated Docker network and were assigned ${authnet}/24.")
|
||||
FINAL_MESSAGE_QUEUE+=(" - If that conflicts with your environment, update docker.networks.soauth in $pillar_file and run so-checkin.")
|
||||
fi
|
||||
|
||||
authgw=$(echo "$authnet" | awk -F'.' '{print $1,$2,$3,1}' OFS='.')
|
||||
|
||||
echo "Assigning the authentication network the range ${authnet}/24."
|
||||
so-yaml.py add "$pillar_file" docker.networks.soauth.range "${authnet}/24" >> $SOUP_LOG 2>&1
|
||||
so-yaml.py add "$pillar_file" docker.networks.soauth.gateway "$authgw" >> $SOUP_LOG 2>&1
|
||||
}
|
||||
|
||||
valid_soauth_range() {
|
||||
local candidate=$1 docker_range=$2
|
||||
|
||||
valid_ip4 "$candidate" || return 1
|
||||
[[ $candidate =~ ^172\.17\.0\. ]] && return 1
|
||||
[[ "${candidate}/24" == "$docker_range" ]] && return 1
|
||||
return 0
|
||||
}
|
||||
|
||||
post_to_3.4.0() {
|
||||
set_postversion 3.4.0
|
||||
}
|
||||
### 3.4.0 End ###
|
||||
|
||||
|
||||
repo_sync() {
|
||||
@@ -1061,8 +1399,20 @@ upgrade_check() {
|
||||
fi
|
||||
[[ -f /etc/sohotfix ]] && CURRENTHOTFIX=$(cat /etc/sohotfix)
|
||||
if [ "$INSTALLEDVERSION" == "$NEWVERSION" ]; then
|
||||
# A leftover post-version marker means a previous upgrade to this version
|
||||
# advanced /etc/soversion (the highstate stamps it from the pillar) but did not
|
||||
# finish its post-upgrade steps. Resume the upgrade instead of reporting "latest".
|
||||
if [ -f "$POSTVERSION_FILE" ] && [ "$(cat "$POSTVERSION_FILE")" != "$NEWVERSION" ]; then
|
||||
echo "A previous upgrade to $NEWVERSION did not complete its post-upgrade steps; resuming."
|
||||
is_hotfix=false
|
||||
return 0
|
||||
fi
|
||||
echo "Checking to see if there are hotfixes needed"
|
||||
if [ "$HOTFIXVERSION" == "$CURRENTHOTFIX" ]; then
|
||||
# Reaching here means we are at the target version and NOT resuming (the resume
|
||||
# check above returned otherwise). Clear any stale resume marker so a completed
|
||||
# upgrade is never mistaken for a partial one and re-run on a later invocation.
|
||||
rm -f "$POSTVERSION_FILE"
|
||||
echo "You are already running the latest version of Security Onion."
|
||||
exit 0
|
||||
else
|
||||
@@ -1141,7 +1491,7 @@ upgrade_salt() {
|
||||
|
||||
verify_latest_update_script() {
|
||||
get_soup_script_hashes
|
||||
if [[ "$CURRENTSOUP" == "$GITSOUP" && "$CURRENTCMN" == "$GITCMN" && "$CURRENTIMGCMN" == "$GITIMGCMN" && "$CURRENTSOFIREWALL" == "$GITSOFIREWALL" ]]; then
|
||||
if [[ "$CURRENTSOUP" == "$GITSOUP" && "$CURRENTCMN" == "$GITCMN" && "$CURRENTIMGCMN" == "$GITIMGCMN" && "$CURRENTSOFIREWALL" == "$GITSOFIREWALL" && "$CURRENTSOYAML" == "$GITSOYAML" ]]; then
|
||||
echo "This version of the soup script is up to date. Proceeding."
|
||||
else
|
||||
echo "You are not running the latest soup version. Updating soup and its components. This might take multiple runs to complete."
|
||||
@@ -1150,7 +1500,7 @@ verify_latest_update_script() {
|
||||
|
||||
# Verify that soup scripts updated as expected
|
||||
get_soup_script_hashes
|
||||
if [[ "$CURRENTSOUP" == "$GITSOUP" && "$CURRENTCMN" == "$GITCMN" && "$CURRENTIMGCMN" == "$GITIMGCMN" && "$CURRENTSOFIREWALL" == "$GITSOFIREWALL" ]]; then
|
||||
if [[ "$CURRENTSOUP" == "$GITSOUP" && "$CURRENTCMN" == "$GITCMN" && "$CURRENTIMGCMN" == "$GITIMGCMN" && "$CURRENTSOFIREWALL" == "$GITSOFIREWALL" && "$CURRENTSOYAML" == "$GITSOYAML" ]]; then
|
||||
echo "Succesfully updated soup scripts."
|
||||
else
|
||||
echo "There was a problem updating soup scripts. Trying to rerun script update."
|
||||
@@ -1171,10 +1521,12 @@ verify_es_version_compatibility() {
|
||||
local is_active_intermediate_upgrade=1
|
||||
# supported upgrade paths for SO-ES versions
|
||||
declare -A es_upgrade_map=(
|
||||
["8.18.4"]="8.18.6 8.18.8 9.0.8"
|
||||
["8.18.4"]="8.18.6 8.18.8 9.0.8"
|
||||
["8.18.6"]="8.18.8 9.0.8"
|
||||
["8.18.8"]="9.0.8"
|
||||
["9.0.8"]="9.3.3"
|
||||
["9.0.8"]="9.3.3 9.3.7 9.4.5"
|
||||
["9.3.3"]="9.3.7 9.4.5"
|
||||
["9.3.7"]="9.4.5"
|
||||
)
|
||||
|
||||
# Elasticsearch MUST upgrade through these versions
|
||||
@@ -1460,18 +1812,13 @@ verify_es_version_compatibility() {
|
||||
}
|
||||
|
||||
wait_for_salt_minion_with_restart() {
|
||||
local minion="$1"
|
||||
local max_wait="${2:-60}"
|
||||
local interval="${3:-3}"
|
||||
local logfile="$4"
|
||||
|
||||
wait_for_salt_minion "$minion" "$max_wait" "$interval" "$logfile"
|
||||
/usr/sbin/so-salt-minion-wait
|
||||
local result=$?
|
||||
|
||||
if [[ $result -ne 0 ]]; then
|
||||
echo "$(date '+%a %d %b %Y %H:%M:%S.%6N') - salt-minion not ready, attempting restart..."
|
||||
systemctl_func "restart" "salt-minion"
|
||||
wait_for_salt_minion "$minion" "$max_wait" "$interval" "$logfile"
|
||||
/usr/sbin/so-salt-minion-wait
|
||||
result=$?
|
||||
fi
|
||||
|
||||
@@ -1757,6 +2104,15 @@ main() {
|
||||
set_minionid
|
||||
MINION_ROLE=$(lookup_role)
|
||||
echo "Found that Security Onion $INSTALLEDVERSION is currently installed."
|
||||
# /etc/soversion is stamped to the target version before the upgrade fully
|
||||
# completes, so a lingering resume marker means this grid is only partially
|
||||
# upgraded even though the line above shows the target version. Make that explicit
|
||||
# so it is not mistaken for a finished upgrade.
|
||||
if [ -f "$POSTVERSION_FILE" ] && [ "$(cat "$POSTVERSION_FILE")" != "$INSTALLEDVERSION" ]; then
|
||||
echo ""
|
||||
echo "NOTE: A previous upgrade to $INSTALLEDVERSION did not finish. This grid is"
|
||||
echo " partially upgraded and this soup run will resume and complete it."
|
||||
fi
|
||||
echo ""
|
||||
check_minimum_version
|
||||
|
||||
@@ -1780,11 +2136,20 @@ main() {
|
||||
|
||||
echo "Let's see if we need to update Security Onion."
|
||||
upgrade_check
|
||||
|
||||
check_cpu_compatibility
|
||||
|
||||
upgrade_space
|
||||
|
||||
echo "Verifying Elasticsearch version compatibility across the grid before upgrading."
|
||||
verify_es_version_compatibility
|
||||
|
||||
# Pre-flight health checks: confirm the grid is in a good state before we change
|
||||
# anything. These run before any modifications, so a failure exits cleanly and the
|
||||
# operator can fix the issue and re-run soup.
|
||||
check_cluster_health
|
||||
check_fleet_server
|
||||
|
||||
echo "Checking for Salt Master and Minion updates."
|
||||
upgrade_check_salt
|
||||
set -e
|
||||
@@ -1801,6 +2166,7 @@ main() {
|
||||
fi
|
||||
|
||||
if [ "$is_hotfix" == "true" ]; then
|
||||
SOUP_UPGRADE_STARTED=true
|
||||
echo "Applying $HOTFIXVERSION hotfix"
|
||||
# since we don't run the backup.config_backup state on import we wont snapshot previous version states and pillars
|
||||
if [[ ! "$MINION_ROLE" == "import" ]]; then
|
||||
@@ -1811,10 +2177,19 @@ main() {
|
||||
create_local_directories "/opt/so/saltstack/default"
|
||||
apply_hotfix
|
||||
echo "Hotfix applied"
|
||||
update_version
|
||||
enable_highstate
|
||||
highstate
|
||||
# Record the hotfix only after the highstate succeeds. /etc/sohotfix is written
|
||||
# solely by soup (no salt state manages it), so deferring the write means a failed
|
||||
# hotfix highstate leaves the old hotfix value and re-running soup re-applies it,
|
||||
# rather than reporting "already latest". The soversion/pillar writes in
|
||||
# update_version are no-ops here since the version is unchanged for a hotfix.
|
||||
update_version
|
||||
# Push the hotfix out to the rest of the grid rather than waiting for the scheduled
|
||||
# highstate. Hotfixes never upgrade Salt, so no --salt-upgraded pass is needed.
|
||||
push_grid_highstate
|
||||
else
|
||||
SOUP_UPGRADE_STARTED=true
|
||||
echo ""
|
||||
echo "Performing upgrade from Security Onion $INSTALLEDVERSION to Security Onion $NEWVERSION."
|
||||
echo ""
|
||||
@@ -1870,6 +2245,10 @@ main() {
|
||||
copy_new_files
|
||||
echo ""
|
||||
create_local_directories "/opt/so/saltstack/default"
|
||||
# Seed the resume marker before the highstate stamps /etc/soversion to the new
|
||||
# version, so an interrupted upgrade is detectable as "not finished" on re-run.
|
||||
# POSTVERSION still holds the pre-upgrade (or prior resume) version here.
|
||||
[ -f "$POSTVERSION_FILE" ] || echo "$POSTVERSION" > "$POSTVERSION_FILE"
|
||||
update_version
|
||||
|
||||
echo ""
|
||||
@@ -1881,11 +2260,11 @@ main() {
|
||||
# Testing that salt-master is up by checking that is it connected to itself
|
||||
check_saltmaster_status
|
||||
|
||||
# update the salt-minion configs here and start the minion
|
||||
# update the salt-master and salt-minion configs here and start the minion
|
||||
# since highstate are disabled above, minion start should not trigger a highstate
|
||||
echo ""
|
||||
echo "Ensuring salt-minion configs are up-to-date."
|
||||
salt-call state.apply salt.minion -l info queue=True
|
||||
echo "Ensuring salt-master and salt-minion configs are up-to-date."
|
||||
salt-call state.apply salt.master -l info queue=True
|
||||
echo ""
|
||||
|
||||
# ensure the mine is updated and populated before highstates run, following the salt-master restart
|
||||
@@ -1899,9 +2278,9 @@ main() {
|
||||
enable_highstate
|
||||
|
||||
echo ""
|
||||
echo "Running a highstate. This could take several minutes."
|
||||
echo "Running a highstate at $(date +"%T.%6N"). This could take several minutes."
|
||||
set +e
|
||||
wait_for_salt_minion_with_restart "$MINIONID" "60" "3" "$SOUP_LOG" || fail "Salt minion was not running or ready."
|
||||
wait_for_salt_minion_with_restart || fail "Salt minion was not running or ready."
|
||||
highstate
|
||||
set -e
|
||||
|
||||
@@ -1913,8 +2292,8 @@ main() {
|
||||
|
||||
check_saltmaster_status
|
||||
|
||||
echo "Running a highstate to complete the Security Onion upgrade on this manager. This could take several minutes."
|
||||
wait_for_salt_minion_with_restart "$MINIONID" "60" "3" "$SOUP_LOG" || fail "Salt minion was not running or ready."
|
||||
echo "Running a highstate at $(date +"%T.%6N") to complete the Security Onion upgrade on this manager. This could take several minutes."
|
||||
wait_for_salt_minion_with_restart || fail "Salt minion was not running or ready."
|
||||
|
||||
# Stop long-running scripts to allow potentially updated scripts to load on the next execution.
|
||||
if pgrep salt-relay.sh > /dev/null 2>&1; then
|
||||
@@ -1926,12 +2305,28 @@ main() {
|
||||
|
||||
# ensure the mine is updated and populated before highstates run, following the salt-master restart
|
||||
update_salt_mine
|
||||
|
||||
|
||||
# kick off a searchnode elasticsearch upgrade
|
||||
set +e
|
||||
if [[ "$es_version" != "$target_es_version" ]]; then
|
||||
if salt-key -L accepted | grep -q "_searchnode$" 2>/dev/null; then
|
||||
# only run if there is atleast 1 searchnode
|
||||
upgrade_searchnode_elasticsearch
|
||||
fi
|
||||
fi
|
||||
set -e
|
||||
|
||||
highstate
|
||||
check_saltmaster_status
|
||||
postupgrade_changes
|
||||
[[ $is_airgap -eq 0 ]] && unmount_update
|
||||
|
||||
|
||||
if [[ "$es_version" != "$target_es_version" ]]; then
|
||||
# Run final elasticsearch / fleet state on manager to ensure addon index templates are created/regenerated and loaded
|
||||
echo "Running final Elastic states at $(date +"%T.%6N"), after upgrade to $NEWVERSION"
|
||||
salt-call state.apply elasticsearch,elasticfleet queue=True
|
||||
fi
|
||||
|
||||
echo ""
|
||||
echo "Upgrade to $NEWVERSION complete."
|
||||
|
||||
@@ -1975,13 +2370,18 @@ main() {
|
||||
|
||||
if [[ $NUM_MINIONS -gt 1 ]]; then
|
||||
|
||||
# Actively drive the rest of the grid to this version now. The scheduled highstate
|
||||
# runs only every 120 minutes (salt:schedule:highstate_interval_minutes), so without
|
||||
# this remote nodes could sit on the old version for a couple of hours after soup finishes.
|
||||
push_grid_highstate
|
||||
|
||||
cat << EOF
|
||||
|
||||
|
||||
|
||||
This appears to be a distributed deployment. Other nodes should update themselves at the next Salt highstate (typically within 15 minutes). Do not manually restart anything until you know that all the search/heavy nodes in your deployment are updated. This is especially important if you are using true clustering for Elasticsearch.
|
||||
This appears to be a distributed deployment. soup has dispatched a batched, grid-wide highstate to update the other nodes now: Elasticsearch data nodes (search/heavynodes) first, then receivers, then sensors and the remaining nodes. Progress is logged to /opt/so/log/salt/so-soup-grid-highstate, and you can watch nodes update from the Grid section of SOC. Do not manually restart anything until you know that all the search/heavynodes in your deployment are updated. This is especially important if you are using true clustering for Elasticsearch.
|
||||
|
||||
Each minion is on a random 15 minute check-in period and things like network bandwidth can be a factor in how long the actual upgrade takes. If you have a heavy node on a slow link, it is going to take a while to get the containers to it. Depending on what changes happened between the versions, Elasticsearch might not be able to talk to said heavy node until the update is complete.
|
||||
Nodes are updated in batches, and things like network bandwidth can be a factor in how long the actual upgrade takes. If you have a heavy node on a slow link, it is going to take a while to get the containers to it. Depending on what changes happened between the versions, Elasticsearch might not be able to talk to said heavy node until the update is complete. Any node that is offline or missed a batch will converge on its own scheduled highstate (every 120 minutes by default).
|
||||
|
||||
If it looks like you’re missing data after the upgrade, please avoid restarting services and instead make sure at least one search node has completed its upgrade. The best way to do this is to run 'sudo salt-call state.highstate' from a search node and make sure there are no errors. Typically if it works on one node it will work on the rest. Sensor nodes are less complex and will update as they check in so you can monitor those from the Grid section of SOC.
|
||||
|
||||
@@ -2017,12 +2417,25 @@ fi
|
||||
echo "### soup has been served at $(date) ###"
|
||||
}
|
||||
|
||||
SKIP_CPU_CHECK=false
|
||||
declare -a SOUP_ARGS=()
|
||||
for arg in "$@"; do
|
||||
if [[ "$arg" == "--skip-cpu-check" ]]; then
|
||||
SKIP_CPU_CHECK=true
|
||||
else
|
||||
SOUP_ARGS+=("$arg")
|
||||
fi
|
||||
done
|
||||
set -- "${SOUP_ARGS[@]}"
|
||||
|
||||
while getopts ":b:f:y" opt; do
|
||||
case ${opt} in
|
||||
b )
|
||||
BATCHSIZE="$OPTARG"
|
||||
if ! [[ "$BATCHSIZE" =~ ^[1-9][0-9]*$ ]]; then
|
||||
echo "Batch size must be a number greater than 0."
|
||||
# Accept either a plain count (e.g. 5) or a percentage (e.g. 25%); passed through
|
||||
# to so-soup-grid-highstate --batch, which salt's batch/batch_wait accepts in both forms.
|
||||
if ! [[ "$BATCHSIZE" =~ ^[1-9][0-9]*%?$ ]]; then
|
||||
echo "Batch size must be a number greater than 0, optionally with a trailing % (e.g. 5 or 25%)."
|
||||
exit 1
|
||||
fi
|
||||
;;
|
||||
@@ -2038,7 +2451,7 @@ while getopts ":b:f:y" opt; do
|
||||
ISOLOC="$OPTARG"
|
||||
;;
|
||||
\? )
|
||||
echo "Usage: soup [-b] [-y] [-f <iso location>]"
|
||||
echo "Usage: soup [-b] [-y] [-f <iso location>] [--skip-cpu-check]"
|
||||
exit 1
|
||||
;;
|
||||
: )
|
||||
|
||||
@@ -57,8 +57,8 @@ nginx_sbin:
|
||||
file.recurse:
|
||||
- name: /usr/sbin
|
||||
- source: salt://nginx/tools/sbin
|
||||
- user: 939
|
||||
- group: 939
|
||||
- user: root
|
||||
- group: root
|
||||
- file_mode: 755
|
||||
|
||||
#nginx_sbin_jinja:
|
||||
|
||||
@@ -8,7 +8,6 @@
|
||||
{% from 'vars/globals.map.jinja' import GLOBALS %}
|
||||
{% from 'docker/docker.map.jinja' import DOCKERMERGED %}
|
||||
{% from 'nginx/map.jinja' import NGINXMERGED %}
|
||||
{% from 'docker/macros/docker_endpoint.jinja' import clear_stale_endpoint %}
|
||||
|
||||
include:
|
||||
- nginx.ssl
|
||||
@@ -32,11 +31,10 @@ make-rule-dir-nginx:
|
||||
{% set container_config = 'so-nginx-fleet-node' %}
|
||||
{% endif %}
|
||||
|
||||
{{ clear_stale_endpoint('so-nginx', 'sobridge', DOCKERMERGED.containers[container_config].ip) }}
|
||||
|
||||
so-nginx:
|
||||
docker_container.running:
|
||||
- image: {{ GLOBALS.registry_host }}:5000/{{ GLOBALS.image_repo }}/so-nginx:{{ GLOBALS.so_version }}
|
||||
- restart_policy: unless-stopped
|
||||
- hostname: so-nginx
|
||||
- networks:
|
||||
- sobridge:
|
||||
|
||||
@@ -96,14 +96,14 @@ http {
|
||||
add_header X-XSS-Protection "1; mode=block";
|
||||
add_header X-Content-Type-Options nosniff;
|
||||
add_header Strict-Transport-Security "max-age=31536000; includeSubDomains";
|
||||
add_header referrer-Policy no-referrer;
|
||||
add_header Referrer-Policy no-referrer;
|
||||
|
||||
ssl_certificate "/etc/pki/nginx/server.crt";
|
||||
ssl_certificate_key "/etc/pki/nginx/server.key";
|
||||
ssl_session_cache shared:SSL:1m;
|
||||
ssl_session_timeout 10m;
|
||||
ssl_ciphers TLS_ECDHE_RSA_WITH_AES_256_GCM_SHA384:TLS_ECDHE_RSA_WITH_CHACHA20_POLY1305_SHA256:TLS_ECDHE_RSA_WITH_ARIA_256_GCM_SHA384:TLS_ECDHE_RSA_WITH_AES_128_GCM_SHA256:TLS_ECDHE_RSA_WITH_ARIA_128_GCM_SHA256:TLS_RSA_WITH_AES_256_GCM_SHA384:TLS_RSA_WITH_AES_256_CCM:TLS_RSA_WITH_ARIA_256_GCM_SHA384:TLS_RSA_WITH_AES_128_GCM_SHA256:TLS_RSA_WITH_AES_128_CCM:TLS_RSA_WITH_ARIA_128_GCM_SHA256;
|
||||
ssl_ecdh_curve secp521r1:secp384r1;
|
||||
ssl_ecdh_curve X25519:secp521r1:secp384r1;
|
||||
ssl_prefer_server_ciphers on;
|
||||
ssl_protocols TLSv1.2 TLSv1.3;
|
||||
}
|
||||
@@ -138,14 +138,14 @@ http {
|
||||
add_header X-XSS-Protection "1; mode=block";
|
||||
add_header X-Content-Type-Options nosniff;
|
||||
add_header Strict-Transport-Security "max-age=31536000; includeSubDomains";
|
||||
add_header referrer-Policy no-referrer;
|
||||
add_header Referrer-Policy no-referrer;
|
||||
|
||||
ssl_certificate "/etc/pki/nginx/server.crt";
|
||||
ssl_certificate_key "/etc/pki/nginx/server.key";
|
||||
ssl_session_cache shared:SSL:1m;
|
||||
ssl_session_timeout 10m;
|
||||
ssl_ciphers TLS_ECDHE_RSA_WITH_AES_256_GCM_SHA384:TLS_ECDHE_RSA_WITH_CHACHA20_POLY1305_SHA256:TLS_ECDHE_RSA_WITH_ARIA_256_GCM_SHA384:TLS_ECDHE_RSA_WITH_AES_128_GCM_SHA256:TLS_ECDHE_RSA_WITH_ARIA_128_GCM_SHA256:TLS_RSA_WITH_AES_256_GCM_SHA384:TLS_RSA_WITH_AES_256_CCM:TLS_RSA_WITH_ARIA_256_GCM_SHA384:TLS_RSA_WITH_AES_128_GCM_SHA256:TLS_RSA_WITH_AES_128_CCM:TLS_RSA_WITH_ARIA_128_GCM_SHA256;
|
||||
ssl_ecdh_curve secp521r1:secp384r1;
|
||||
ssl_ecdh_curve X25519:secp521r1:secp384r1;
|
||||
ssl_prefer_server_ciphers on;
|
||||
ssl_protocols TLSv1.2 TLSv1.3;
|
||||
location / {
|
||||
@@ -172,18 +172,18 @@ http {
|
||||
add_header X-XSS-Protection "1; mode=block";
|
||||
add_header X-Content-Type-Options nosniff;
|
||||
add_header Strict-Transport-Security "max-age=31536000; includeSubDomains";
|
||||
add_header referrer-Policy no-referrer;
|
||||
add_header Referrer-Policy no-referrer;
|
||||
|
||||
ssl_certificate "/etc/pki/nginx/server.crt";
|
||||
ssl_certificate_key "/etc/pki/nginx/server.key";
|
||||
ssl_session_cache shared:SSL:1m;
|
||||
ssl_session_timeout 10m;
|
||||
ssl_ciphers TLS_ECDHE_RSA_WITH_AES_256_GCM_SHA384:TLS_ECDHE_RSA_WITH_CHACHA20_POLY1305_SHA256:TLS_ECDHE_RSA_WITH_ARIA_256_GCM_SHA384:TLS_ECDHE_RSA_WITH_AES_128_GCM_SHA256:TLS_ECDHE_RSA_WITH_ARIA_128_GCM_SHA256:TLS_RSA_WITH_AES_256_GCM_SHA384:TLS_RSA_WITH_AES_256_CCM:TLS_RSA_WITH_ARIA_256_GCM_SHA384:TLS_RSA_WITH_AES_128_GCM_SHA256:TLS_RSA_WITH_AES_128_CCM:TLS_RSA_WITH_ARIA_128_GCM_SHA256;
|
||||
ssl_ecdh_curve secp521r1:secp384r1;
|
||||
ssl_ecdh_curve X25519:secp521r1:secp384r1;
|
||||
ssl_prefer_server_ciphers on;
|
||||
ssl_protocols TLSv1.2 TLSv1.3;
|
||||
|
||||
location ~* (^/login/.*|^/js/.*|^/css/.*|^/images/.*|^/pages/.*|^/docs/.*) {
|
||||
location ~* (^/login|^/login/.*|^/js/.*|^/css/.*|^/images/.*|^/pages/.*|^/docs/.*) {
|
||||
proxy_pass http://{{ GLOBALS.manager }}:9822;
|
||||
proxy_read_timeout 90;
|
||||
proxy_connect_timeout 90;
|
||||
@@ -198,6 +198,10 @@ http {
|
||||
}
|
||||
|
||||
location / {
|
||||
if ($http_authorization ~* "^Bearer .*$") {
|
||||
return 401;
|
||||
}
|
||||
|
||||
auth_request /auth/sessions/whoami;
|
||||
auth_request_set $userid $upstream_http_x_kratos_authenticated_identity_id;
|
||||
proxy_set_header x-user-id $userid;
|
||||
@@ -218,6 +222,13 @@ http {
|
||||
add_header Cache-Control "no-cache, no-store, must-revalidate";
|
||||
add_header Pragma "no-cache";
|
||||
add_header Expires "0";
|
||||
|
||||
add_header Content-Security-Policy "default-src 'self' 'unsafe-inline' 'unsafe-eval' https: data: blob: wss:; frame-ancestors 'self'";
|
||||
add_header X-Frame-Options SAMEORIGIN;
|
||||
add_header X-XSS-Protection "1; mode=block";
|
||||
add_header X-Content-Type-Options nosniff;
|
||||
add_header Strict-Transport-Security "max-age=31536000; includeSubDomains";
|
||||
add_header Referrer-Policy no-referrer;
|
||||
}
|
||||
|
||||
location ~ ^/auth/.*?(login|oidc/callback) {
|
||||
@@ -249,7 +260,7 @@ http {
|
||||
}
|
||||
|
||||
{% if 'api' in salt['pillar.get']('features', []) %}
|
||||
location ~* (^/oauth2/token.*|^.well-known/jwks.json|^.well-known/openid-configuration) {
|
||||
location ~* (^/oauth2/token.*|^/\.well-known/jwks.json|^/\.well-known/openid-configuration) {
|
||||
limit_req zone=auth_throttle burst={{ NGINXMERGED.config.throttle_login_burst }} nodelay;
|
||||
limit_req_status 429;
|
||||
proxy_pass http://{{ GLOBALS.manager }}:4444;
|
||||
@@ -383,6 +394,11 @@ http {
|
||||
if ($http_authorization = "") {
|
||||
return 403;
|
||||
}
|
||||
|
||||
if ($http_authorization ~* "^Bearer .*$") {
|
||||
return 401;
|
||||
}
|
||||
|
||||
proxy_pass http://{{ GLOBALS.manager }}:9822/;
|
||||
proxy_read_timeout 90;
|
||||
proxy_connect_timeout 90;
|
||||
@@ -399,7 +415,7 @@ http {
|
||||
error_page 429 = @error429;
|
||||
|
||||
location @error401 {
|
||||
if ($request_uri ~* (^/api/.*|^/connect/.*|^/oauth2/.*|^/.*\.map$)) {
|
||||
if ($request_uri ~* (^.*/api/.*|^.*/login.*|^.*/logout.*|^/connect/.*|^/oauth2/.*|^/.*\.map$)) {
|
||||
return 401;
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,37 @@
|
||||
{% from 'salt/auto_apply.map.jinja' import AUTOAPPLY %}
|
||||
{% set actions = salt['pillar.get']('actions', []) %}
|
||||
{% set BATCH = AUTOAPPLY.batch %}
|
||||
{% set BATCH_WAIT = AUTOAPPLY.batch_wait %}
|
||||
|
||||
{% for action in actions %}
|
||||
{% if action.get('highstate') %}
|
||||
apply_highstate_{{ loop.index }}:
|
||||
salt.state:
|
||||
- tgt: '{{ action.tgt }}'
|
||||
- tgt_type: {{ action.get('tgt_type', 'compound') }}
|
||||
- highstate: True
|
||||
- batch: {{ action.get('batch', BATCH) }}
|
||||
- batch_wait: {{ action.get('batch_wait', BATCH_WAIT) }}
|
||||
- kwarg:
|
||||
queue: 2
|
||||
{% else %}
|
||||
refresh_pillar_{{ loop.index }}:
|
||||
salt.function:
|
||||
- name: saltutil.refresh_pillar
|
||||
- tgt: '{{ action.tgt }}'
|
||||
- tgt_type: {{ action.get('tgt_type', 'compound') }}
|
||||
|
||||
apply_{{ action.state | replace('.', '_') }}_{{ loop.index }}:
|
||||
salt.state:
|
||||
- tgt: '{{ action.tgt }}'
|
||||
- tgt_type: {{ action.get('tgt_type', 'compound') }}
|
||||
- sls:
|
||||
- {{ action.state }}
|
||||
- batch: {{ action.get('batch', BATCH) }}
|
||||
- batch_wait: {{ action.get('batch_wait', BATCH_WAIT) }}
|
||||
- kwarg:
|
||||
queue: 2
|
||||
- require:
|
||||
- salt: refresh_pillar_{{ loop.index }}
|
||||
{% endif %}
|
||||
{% endfor %}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user