Files
securityonion/salt/postgres/telegraf_users.sls
T
Mike Reeves 5d36d00dec Fix Telegraf metrics falling into pg_partman default partitions
pg_cron's launcher connects to cron.database_name at postmaster start and is
registered BGW_NEVER_RESTART. On a host upgraded onto an existing /nsm/postgres
volume, init-db.sh never runs, so so_telegraf does not exist when PostgreSQL
starts -- the launcher dies and never retries. Salt then creates the database,
the extension, and the schedule, all of which succeed, but no worker is left to
fire the job. partman.run_maintenance_proc() therefore never runs: partitions
stop being premade after create_parent's initial window and every metric lands
in <parent>_default. Retention never fires either.

That state is self-perpetuating. Once the default partition holds rows for a day
with no child, PostgreSQL cannot create that child at all -- attaching it would
violate the default partition's constraint -- so maintenance aborts on the first
parent it reaches. Fixing the scheduler alone does not recover a stalled grid.

Point cron.database_name at the always-present postgres database and register the
job with cron.schedule_in_database targeting so_telegraf, so the launcher no
longer depends on database creation order. group_role drops any registration left
behind in so_telegraf, and both halves are guarded on the live GUC so applying
postgres.telegraf_users before the postgresql.conf change has restarted the
container skips instead of failing.

Maintenance now runs so_admin.telegraf_maintenance(), which drains stranded rows
out of any default partition before calling partman: expired rows are deleted,
the rest are repartitioned. It runs from the state on every highstate as well as
hourly from pg_cron, so a grid whose worker is dead still recovers on its own.
The routines live in a postgres-owned schema so_telegraf has no rights on, since
pg_cron executes them as postgres.

Existing grids are recovered by a marker-guarded repair state that truncates the
non-empty defaults once per host. The backlog is mostly past retention already
and moving tens of GB just to delete most of it is not worth the WAL.

Also raise premake from 3 to 7, reconciled onto existing parents in the retention
subcommand, so an outage has a week of headroom before anything reaches a
default partition, and add a check subcommand reporting partition age, default
occupancy and last job status.
2026-08-05 17:39:39 -04:00

101 lines
3.6 KiB
YAML+Jinja

# Copyright Security Onion Solutions LLC and/or licensed to Security Onion Solutions LLC under one
# or more contributor license agreements. Licensed under the Elastic License 2.0 as shown at
# https://securityonion.net/license; you may not use this file except in compliance with the
# Elastic License 2.0.
{% from 'allowed_states.map.jinja' import allowed_states %}
{% if sls.split('.')[0] in allowed_states %}
{% from 'vars/globals.map.jinja' import GLOBALS %}
{% from 'telegraf/map.jinja' import TELEGRAFMERGED %}
{# postgres.enabled declares the so-postgres container and postgres_wait_ready
that the requires below reference. Salt de-duplicates the circular include. #}
include:
- postgres.enabled
{% set TG_OUT = TELEGRAFMERGED.output | upper %}
{% if TG_OUT in ['POSTGRES', 'BOTH'] %}
# Ensure the shared Telegraf database exists. init-db.sh only runs on a
# fresh data dir, so hosts upgraded onto an existing /nsm/postgres volume
# would otherwise never get so_telegraf.
postgres_create_telegraf_db:
cmd.run:
- name: /usr/sbin/so-telegraf-postgres create_db
- require:
- cmd: postgres_wait_ready
- file: postgres_sbin
# Provision the shared group role and schema once. Every per-minion role is a
# member of so_telegraf, and each Telegraf connection does SET ROLE so_telegraf
# (via options='-c role=so_telegraf' in the connection string) so tables created
# on first write are owned by the group role and every member can INSERT/SELECT.
postgres_telegraf_group_role:
cmd.run:
- name: /usr/sbin/so-telegraf-postgres group_role
- require:
- cmd: postgres_create_telegraf_db
- file: postgres_sbin
{% set creds = salt['pillar.get']('telegraf:postgres_creds', {}) %}
{% for mid, entry in creds.items() %}
{% if entry.get('user') and entry.get('pass') %}
{% set u = entry.user %}
{% set p = entry.pass %}
postgres_telegraf_role_{{ u }}:
cmd.run:
- name: /usr/sbin/so-telegraf-postgres user
- env:
- ROLE_USER: {{ u | tojson }}
- ROLE_PASS: {{ p | tojson }}
- hide_output: True
- require:
- file: postgres_sbin
- cmd: postgres_telegraf_group_role
{% endif %}
{% endfor %}
# Reconcile partman retention from pillar. Runs after role/schema setup so
# any partitioned parents Telegraf has already created get their retention
# refreshed whenever postgres.telegraf.retention_days changes.
{% set retention = salt['pillar.get']('postgres:telegraf:retention_days', 14) | int %}
postgres_telegraf_retention_reconcile:
cmd.run:
- name: /usr/sbin/so-telegraf-postgres retention
- env:
- RETENTION_DAYS: {{ retention }}
- require:
- cmd: postgres_telegraf_group_role
- file: postgres_sbin
# One-time recovery for grids whose pg_cron job never ran. Truncating the
# defaults is destructive, so the watermark keeps it to one pass per host.
postgres_telegraf_partman_default_repair:
cmd.run:
- name: mkdir -p /opt/so/state && /usr/sbin/so-telegraf-postgres repair && touch /opt/so/state/telegraf_partman_default_repair
- creates: /opt/so/state/telegraf_partman_default_repair
- require:
- cmd: postgres_telegraf_retention_reconcile
- file: postgres_sbin
# Also run from the state, not just hourly from pg_cron, so a grid whose
# pg_cron worker is dead still recovers on its own.
postgres_telegraf_partman_maintenance:
cmd.run:
- name: /usr/sbin/so-telegraf-postgres maintenance
- require:
- cmd: postgres_telegraf_partman_default_repair
- file: postgres_sbin
{% endif %}
{% else %}
{{sls}}_state_not_allowed:
test.fail_without_changes:
- name: {{sls}}_state_not_allowed
{% endif %}