Add verified Prometheus backup pull to Atlas

This commit is contained in:
Fabio Scotto di Santolo
2026-09-30 21:21:48 +02:00
parent 3d2ef02c98
commit 0144600a4a
23 changed files with 851 additions and 22 deletions

View File

@@ -115,6 +115,15 @@ atlas_monitor_notifier: "{{ atlas_usb_reminder_notifier }}"
atlas_monitor_smart_devices: []
atlas_monitor_timers: []
atlas_monitor_failure_units: []
atlas_monitor_effective_timers: >-
{{ atlas_monitor_timers
+ ([{'name': 'atlas-prometheus-pull.timer', 'max_age_hours': 26}]
if atlas_manage_prometheus_backup_pull | bool and atlas_prometheus_pull_start_timer | bool
else []) }}
atlas_monitor_effective_failure_units: >-
{{ atlas_monitor_failure_units
+ (['atlas-prometheus-pull.service']
if atlas_manage_prometheus_backup_pull | bool else []) }}
atlas_monitor_remote_capacity: {}
atlas_monitor_pool_warning_percent: 80
atlas_monitor_pool_critical_percent: 90
@@ -141,6 +150,19 @@ atlas_music_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_music }}"
atlas_backup_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_backup }}"
atlas_host_backups_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_host_backups }}"
atlas_backup_prometheus_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_backup_prometheus }}"
atlas_manage_prometheus_backup_pull: false
atlas_prometheus_pull_ssh_dir: /etc/atlas-prometheus-pull
atlas_prometheus_pull_private_key_path: "{{ atlas_prometheus_pull_ssh_dir }}/id_ed25519"
atlas_prometheus_pull_known_hosts_path: "{{ atlas_prometheus_pull_ssh_dir }}/known_hosts"
atlas_prometheus_ssh_host_key: ""
atlas_prometheus_pull_source_user: prometheus-backup
atlas_prometheus_pull_source_port: 22
atlas_prometheus_pull_calendar: "*-*-* 03:00:00 Europe/Rome"
atlas_prometheus_pull_start_timer: false
atlas_prometheus_pull_keep_daily: 30
atlas_prometheus_pull_keep_weekly: 8
atlas_prometheus_pull_keep_monthly: 12
atlas_prometheus_pull_max_age_hours: 24
atlas_photobook_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_photobook }}"
atlas_45drives_repo_url: https://repo.45drives.com/repofiles/rocky/45drives-enterprise.repo

View File

@@ -0,0 +1,56 @@
#!/usr/bin/env python3
"""Prune only verified, named Prometheus backup versions after publication."""
import datetime as dt
import pathlib
import re
import shutil
import sys
def main() -> None:
if len(sys.argv) != 5:
raise SystemExit("Usage: atlas-prometheus-prune SNAPSHOTS DAILY WEEKLY MONTHLY")
root = pathlib.Path(sys.argv[1])
counts = [int(value) for value in sys.argv[2:]]
if not root.is_dir() or root.is_symlink() or min(counts) < 1:
raise SystemExit("Invalid backup directory or retention counts")
versions = []
for entry in root.iterdir():
if not entry.is_dir() or entry.is_symlink():
continue
if not re.fullmatch(r"[0-9]{8}T[0-9]{6}Z", entry.name):
continue
try:
when = dt.datetime.strptime(entry.name, "%Y%m%dT%H%M%SZ")
except ValueError:
continue
if not all((entry / name).is_file() for name in ("payload.tar", "payload.sha256", "metadata.json")):
continue
versions.append((when, entry))
versions.sort(reverse=True)
if not versions:
raise SystemExit("No published backup versions found; refusing to prune")
keep = {entry for _, entry in versions[: counts[0]]}
for count, key in (
(counts[1], lambda when: when.isocalendar()[:2]),
(counts[2], lambda when: (when.year, when.month)),
):
periods = set()
for when, entry in versions:
period = key(when)
if period in periods:
continue
periods.add(period)
keep.add(entry)
if len(periods) >= count:
break
for _, entry in versions:
if entry not in keep:
shutil.rmtree(entry)
if __name__ == "__main__":
main()

View File

@@ -23,6 +23,12 @@
- name: Import Atlas offline USB backup tasks
ansible.builtin.import_tasks: usb_backup.yml
- name: Import Atlas Prometheus backup pull identity tasks
ansible.builtin.import_tasks: prometheus_pull_identity.yml
- name: Import Atlas Prometheus backup pull job tasks
ansible.builtin.import_tasks: prometheus_pull_job.yml
- name: Import Atlas health monitoring tasks
ansible.builtin.import_tasks: monitoring.yml

View File

@@ -7,8 +7,8 @@
- atlas_zfs_pool != 'CHANGEME_ZFS_POOL'
- atlas_monitor_calendar | length > 0
- atlas_monitor_smart_devices | length > 0
- atlas_monitor_timers | length > 0
- atlas_monitor_failure_units | length > 0
- atlas_monitor_effective_timers | length > 0
- atlas_monitor_effective_failure_units | length > 0
- atlas_monitor_remote_capacity.user == atlas_borg_repository_user
- atlas_monitor_remote_capacity.host == atlas_borg_repository_host
- atlas_monitor_remote_capacity.run_as == atlas_borg_username
@@ -49,7 +49,7 @@
that:
- item.name is match('^[a-zA-Z0-9@_.-]+\\.timer$')
- item.max_age_hours | int >= 0
loop: "{{ atlas_monitor_timers }}"
loop: "{{ atlas_monitor_effective_timers }}"
loop_control:
label: "{{ item.name }}"
when: atlas_manage_monitoring | bool
@@ -59,7 +59,7 @@
ansible.builtin.assert:
that:
- item is match('^[a-zA-Z0-9@_.-]+\\.service$')
loop: "{{ atlas_monitor_failure_units }}"
loop: "{{ atlas_monitor_effective_failure_units }}"
when: atlas_manage_monitoring | bool
- name: Validate Atlas health monitor calendar
@@ -144,7 +144,7 @@
owner: root
group: root
mode: "0755"
loop: "{{ atlas_monitor_failure_units }}"
loop: "{{ atlas_monitor_effective_failure_units }}"
when: atlas_manage_monitoring | bool
- name: Notify 45Drives Alerts when an Atlas job fails
@@ -155,7 +155,7 @@
owner: root
group: root
mode: "0644"
loop: "{{ atlas_monitor_failure_units }}"
loop: "{{ atlas_monitor_effective_failure_units }}"
when: atlas_manage_monitoring | bool
- name: Reload systemd after installing Atlas monitoring

View File

@@ -0,0 +1,66 @@
---
- name: Validate Atlas Prometheus pull identity inputs
tags: [atlas, backup, prometheus_backup, prometheus_backup_key]
ansible.builtin.assert:
that:
- atlas_prometheus_pull_ssh_dir.startswith('/etc/')
- atlas_prometheus_pull_private_key_path.startswith(atlas_prometheus_pull_ssh_dir ~ '/')
- atlas_prometheus_pull_known_hosts_path.startswith(atlas_prometheus_pull_ssh_dir ~ '/')
- atlas_prometheus_ssh_host_key.startswith(
(hostvars['prometheus'].ansible_host | string) ~ ' ssh-ed25519 '
)
fail_msg: Pin the verified Prometheus ED25519 SSH host key before enabling the pull.
when: atlas_manage_prometheus_backup_pull | bool
- name: Create private Atlas Prometheus pull SSH directory
tags: [atlas, backup, prometheus_backup, prometheus_backup_key]
ansible.builtin.file:
path: "{{ atlas_prometheus_pull_ssh_dir }}"
state: directory
owner: root
group: root
mode: "0700"
when: atlas_manage_prometheus_backup_pull | bool
- name: Generate Atlas-only Prometheus pull SSH identity
tags: [atlas, backup, prometheus_backup, prometheus_backup_key]
ansible.builtin.command:
argv:
- ssh-keygen
- -q
- -t
- ed25519
- -N
- ""
- -C
- atlas-prometheus-pull@atlas
- -f
- "{{ atlas_prometheus_pull_private_key_path }}"
creates: "{{ atlas_prometheus_pull_private_key_path }}"
when: atlas_manage_prometheus_backup_pull | bool
- name: Protect Atlas-only Prometheus pull SSH identity
tags: [atlas, backup, prometheus_backup, prometheus_backup_key]
ansible.builtin.file:
path: "{{ item.path }}"
owner: root
group: root
mode: "{{ item.mode }}"
loop:
- { path: "{{ atlas_prometheus_pull_private_key_path }}", mode: "0600" }
- { path: "{{ atlas_prometheus_pull_private_key_path }}.pub", mode: "0644" }
loop_control:
label: "{{ item.path }}"
when:
- atlas_manage_prometheus_backup_pull | bool
- not ansible_check_mode
- name: Pin Prometheus SSH host key on Atlas
tags: [atlas, backup, prometheus_backup, prometheus_backup_key]
ansible.builtin.copy:
content: "{{ atlas_prometheus_ssh_host_key }}\n"
dest: "{{ atlas_prometheus_pull_known_hosts_path }}"
owner: root
group: root
mode: "0600"
when: atlas_manage_prometheus_backup_pull | bool

View File

@@ -0,0 +1,90 @@
---
- name: Validate Atlas Prometheus backup pull inputs
tags: [atlas, backup, prometheus_backup]
ansible.builtin.assert:
that:
- atlas_manage_storage | bool
- atlas_prometheus_pull_source_user is match('^[a-z_][a-z0-9_-]*$')
- atlas_prometheus_pull_source_port | int > 0
- atlas_prometheus_pull_source_port | int < 65536
- atlas_prometheus_pull_keep_daily | int > 0
- atlas_prometheus_pull_keep_weekly | int > 0
- atlas_prometheus_pull_keep_monthly | int > 0
- atlas_prometheus_pull_max_age_hours | int > 0
- atlas_backup_prometheus_mountpoint.startswith(atlas_mount_root ~ '/')
fail_msg: Define the Atlas backup destination, source account, and retention before enabling the pull.
when: atlas_manage_prometheus_backup_pull | bool
- name: Validate Atlas Prometheus backup pull calendar
tags: [atlas, backup, prometheus_backup]
ansible.builtin.command:
argv: [systemd-analyze, calendar, "{{ atlas_prometheus_pull_calendar }}"]
changed_when: false
check_mode: false
when: atlas_manage_prometheus_backup_pull | bool
- name: Create private Atlas Prometheus backup version directory
tags: [atlas, backup, prometheus_backup]
ansible.builtin.file:
path: "{{ atlas_backup_prometheus_mountpoint }}/snapshots"
state: directory
owner: root
group: root
mode: "0700"
when: atlas_manage_prometheus_backup_pull | bool
- name: Install Atlas Prometheus backup pull helper
tags: [atlas, backup, prometheus_backup]
ansible.builtin.template:
src: atlas-prometheus-pull.sh.j2
dest: /usr/local/sbin/atlas-prometheus-pull
owner: root
group: root
mode: "0750"
when: atlas_manage_prometheus_backup_pull | bool
- name: Install Atlas Prometheus backup retention helper
tags: [atlas, backup, prometheus_backup]
ansible.builtin.copy:
src: atlas-prometheus-prune.py
dest: /usr/local/libexec/atlas-prometheus-prune
owner: root
group: root
mode: "0750"
when: atlas_manage_prometheus_backup_pull | bool
- name: Install Atlas Prometheus backup pull systemd units
tags: [atlas, backup, prometheus_backup]
ansible.builtin.template:
src: "{{ item }}.j2"
dest: "/etc/systemd/system/{{ item }}"
owner: root
group: root
mode: "0644"
loop:
- atlas-prometheus-pull.service
- atlas-prometheus-pull.timer
loop_control:
label: "{{ item }}"
register: atlas_prometheus_pull_units
when: atlas_manage_prometheus_backup_pull | bool
- name: Reload systemd after Atlas Prometheus pull unit changes
tags: [atlas, backup, prometheus_backup]
ansible.builtin.systemd:
daemon_reload: true
when:
- atlas_manage_prometheus_backup_pull | bool
- atlas_prometheus_pull_units is changed
- not ansible_check_mode
- name: Enable Atlas Prometheus pull timer only after explicit activation
tags: [atlas, backup, prometheus_backup]
ansible.builtin.systemd:
name: atlas-prometheus-pull.timer
enabled: true
state: started
when:
- atlas_manage_prometheus_backup_pull | bool
- atlas_prometheus_pull_start_timer | bool
- not ansible_check_mode

View File

@@ -3,8 +3,8 @@
"backup_dataset": {{ (atlas_zfs_pool ~ '/' ~ atlas_zfs_dataset_backup) | to_json }},
"notifier": {{ atlas_monitor_notifier | to_json }},
"smart_devices": {{ atlas_monitor_smart_devices | to_json }},
"timers": {{ atlas_monitor_timers | to_json }},
"failure_units": {{ atlas_monitor_failure_units | to_json }},
"timers": {{ atlas_monitor_effective_timers | to_json }},
"failure_units": {{ atlas_monitor_effective_failure_units | to_json }},
"remote_capacity": {{ atlas_monitor_remote_capacity | to_json }},
"pool_warning_percent": {{ atlas_monitor_pool_warning_percent | int }},
"pool_critical_percent": {{ atlas_monitor_pool_critical_percent | int }},

View File

@@ -0,0 +1,19 @@
[Unit]
Description=Pull a prepared read-only Prometheus backup to Atlas
RequiresMountsFor={{ atlas_backup_prometheus_mountpoint }}
Wants=network-online.target
After=network-online.target zfs.target
ConditionFileIsExecutable=/usr/local/sbin/atlas-prometheus-pull
ConditionPathExists={{ atlas_prometheus_pull_private_key_path }}
ConditionPathExists={{ atlas_prometheus_pull_known_hosts_path }}
[Service]
Type=oneshot
ExecStart=/usr/local/sbin/atlas-prometheus-pull
User=root
Group=root
UMask=0077
TimeoutStartSec=infinity
Nice=15
IOSchedulingClass=best-effort
IOSchedulingPriority=7

View File

@@ -0,0 +1,75 @@
#!/usr/bin/env bash
set -Eeuo pipefail
umask 077
backup_root={{ atlas_backup_prometheus_mountpoint | quote }}
snapshots="$backup_root/snapshots"
stage=''
exec 9>/run/lock/atlas-prometheus-pull.lock
flock -n 9 || { echo 'A Prometheus pull is already running' >&2; exit 1; }
cleanup() {
local rc=$?
trap - EXIT
if (( rc != 0 )) && [[ -n "$stage" && -d "$stage" ]]; then
rm -rf -- "$stage"
fi
exit "$rc"
}
trap cleanup EXIT
zpool list -H -o name {{ atlas_zfs_pool | quote }} >/dev/null
findmnt -rn --mountpoint "$backup_root" >/dev/null
stage=$(mktemp -d "$backup_root/.staging.XXXXXXXX")
ssh_cmd='/usr/bin/ssh -F /dev/null -o BatchMode=yes -o StrictHostKeyChecking=yes -o UserKnownHostsFile={{ atlas_prometheus_pull_known_hosts_path }} -o IdentitiesOnly=yes -i {{ atlas_prometheus_pull_private_key_path }} -p {{ atlas_prometheus_pull_source_port }}'
rsync -a --partial --delay-updates -e "$ssh_cmd" \
{{ (atlas_prometheus_pull_source_user ~ '@' ~ hostvars['prometheus'].ansible_host ~ ':current/') | quote }} \
"$stage/"
test -s "$stage/payload.tar"
test -s "$stage/payload.sha256"
test -s "$stage/metadata.json"
(cd "$stage" && sha256sum -c payload.sha256)
tar -tf "$stage/payload.tar" >/dev/null
stamp=$(python3 - "$stage/metadata.json" <<'PY'
import json
import datetime as dt
import re
import sys
with open(sys.argv[1], encoding="utf-8") as stream:
metadata = json.load(stream)
stamp = metadata.get("created_utc", "")
if metadata.get("schema") != 1 or metadata.get("host") != "prometheus":
raise SystemExit("Unexpected Prometheus backup metadata")
if not re.fullmatch(r"[0-9]{8}T[0-9]{6}Z", stamp):
raise SystemExit("Invalid Prometheus backup timestamp")
created = dt.datetime.strptime(stamp, "%Y%m%dT%H%M%SZ").replace(tzinfo=dt.timezone.utc)
age = dt.datetime.now(dt.timezone.utc) - created
if age.total_seconds() < -300 or age > dt.timedelta(hours={{ atlas_prometheus_pull_max_age_hours }}):
raise SystemExit("Prometheus backup is outside the configured freshness window")
print(stamp)
PY
)
if [[ -e "$snapshots/$stamp" ]]; then
cmp "$stage/payload.sha256" "$snapshots/$stamp/payload.sha256"
cmp "$stage/metadata.json" "$snapshots/$stamp/metadata.json"
(cd "$snapshots/$stamp" && sha256sum -c payload.sha256)
rm -rf -- "${stage:?}"
stage=''
else
chown -R root:root "$stage"
chmod 0700 "$stage"
chmod 0600 "$stage/payload.tar" "$stage/payload.sha256" "$stage/metadata.json"
mv -- "$stage" "$snapshots/$stamp"
stage=''
fi
latest_link=$(readlink "$backup_root/latest" 2>/dev/null || true)
latest_stamp=${latest_link##*/}
if [[ -z "$latest_stamp" || "$stamp" > "$latest_stamp" ]]; then
ln -s "snapshots/$stamp" "$backup_root/.latest.new"
mv -Tf -- "$backup_root/.latest.new" "$backup_root/latest"
fi
python3 /usr/local/libexec/atlas-prometheus-prune "$snapshots" \
{{ atlas_prometheus_pull_keep_daily }} {{ atlas_prometheus_pull_keep_weekly }} {{ atlas_prometheus_pull_keep_monthly }}
echo "Verified and published Prometheus backup $stamp"

View File

@@ -0,0 +1,10 @@
[Unit]
Description=Schedule Atlas pull of prepared Prometheus backups
[Timer]
OnCalendar={{ atlas_prometheus_pull_calendar }}
Persistent=true
Unit=atlas-prometheus-pull.service
[Install]
WantedBy=timers.target

View File

@@ -0,0 +1,103 @@
---
- name: Validate Prometheus backup export identity inputs
tags: [services, backup, prometheus_backup]
ansible.builtin.assert:
that:
- inventory_hostname == 'prometheus'
- server_backup_username is match('^[a-z_][a-z0-9_-]*$')
- server_backup_username not in ['root', server_username]
- server_backup_export_root.startswith('/var/lib/')
- server_backup_public_key_name is match('^[a-z0-9_-]+$')
- hostvars['atlas'].atlas_manage_prometheus_backup_pull | default(false) | bool
fail_msg: Enable Atlas and Prometheus backup roles together with dedicated identity settings.
when: server_backup_export_enabled | bool
- name: Create dedicated Prometheus backup export group
tags: [services, backup, prometheus_backup]
ansible.builtin.group:
name: "{{ server_backup_username }}"
system: true
state: present
when: server_backup_export_enabled | bool
- name: Create locked Prometheus backup export account
tags: [services, backup, prometheus_backup]
ansible.builtin.user:
name: "{{ server_backup_username }}"
group: "{{ server_backup_username }}"
groups: []
append: false
comment: Read-only prepared backup export for Atlas
home: "{{ server_backup_export_root }}"
create_home: false
shell: /bin/bash
password_lock: true
system: true
state: present
when: server_backup_export_enabled | bool
- name: Require restricted rrsync helper on Prometheus
tags: [services, backup, prometheus_backup]
ansible.builtin.stat:
path: "{{ server_backup_rrsync_path }}"
register: server_backup_rrsync_file
when: server_backup_export_enabled | bool
- name: Validate restricted rrsync helper
tags: [services, backup, prometheus_backup]
ansible.builtin.assert:
that:
- server_backup_rrsync_file.stat.exists
- server_backup_rrsync_file.stat.isreg
- server_backup_rrsync_file.stat.pw_name == 'root'
fail_msg: Rocky rsync must provide the root-owned rrsync support script.
when: server_backup_export_enabled | bool
- name: Create prepared backup export root
tags: [services, backup, prometheus_backup]
ansible.builtin.file:
path: "{{ server_backup_export_root }}"
state: directory
owner: root
group: "{{ server_backup_username }}"
mode: "0750"
when: server_backup_export_enabled | bool
- name: Create restricted Prometheus backup SSH directories
tags: [services, backup, prometheus_backup]
ansible.builtin.file:
path: "{{ item }}"
state: directory
owner: root
group: "{{ server_backup_username }}"
mode: "0750"
loop:
- "{{ server_backup_export_root }}/.ssh"
- "{{ server_backup_export_root }}/.ssh/authorized_keys.d"
when: server_backup_export_enabled | bool
- name: Read Atlas public key for Prometheus backup pull
tags: [services, backup, prometheus_backup]
ansible.builtin.slurp:
src: "{{ hostvars['atlas'].atlas_prometheus_pull_private_key_path | default('/etc/atlas-prometheus-pull/id_ed25519') }}.pub"
delegate_to: atlas
become: true
register: server_backup_atlas_public_key
when:
- server_backup_export_enabled | bool
- not ansible_check_mode
- name: Authorize only restricted read-only backup access from Atlas
tags: [services, backup, prometheus_backup]
ansible.builtin.copy:
content: >-
{{ 'command="/usr/bin/python3 ' ~ server_backup_rrsync_path ~ ' -ro '
~ server_backup_export_root ~ '/versions",restrict '
~ (server_backup_atlas_public_key.content | b64decode | trim) ~ '\n' }}
dest: "{{ server_backup_export_root }}/.ssh/authorized_keys.d/{{ server_backup_public_key_name }}"
owner: root
group: "{{ server_backup_username }}"
mode: "0640"
when:
- server_backup_export_enabled | bool
- not ansible_check_mode

View File

@@ -0,0 +1,90 @@
---
- name: Validate Prometheus backup export job inputs
tags: [services, backup, prometheus_backup]
ansible.builtin.assert:
that:
- server_backup_export_source_keep | int >= 2
- server_backup_export_paths | length > 0
- server_backup_export_paths | unique | length == server_backup_export_paths | length
- >-
server_backup_export_paths
| select('match', '^[a-zA-Z0-9][a-zA-Z0-9._/-]*$') | list | length
== server_backup_export_paths | length
- >-
server_backup_export_paths
| reject('search', '(^|/)\.\.(/|$)') | list | length
== server_backup_export_paths | length
- >-
server_backup_export_excludes
| select('match', '^[a-zA-Z0-9][a-zA-Z0-9._/-]*$') | list | length
== server_backup_export_excludes | length
- >-
server_backup_export_excludes
| reject('search', '(^|/)\.\.(/|$)') | list | length
== server_backup_export_excludes | length
fail_msg: Define safe relative paths and at least two prepared export versions.
when: server_backup_export_enabled | bool
- name: Validate Prometheus backup export calendar
tags: [services, backup, prometheus_backup]
ansible.builtin.command:
argv: [systemd-analyze, calendar, "{{ server_backup_export_calendar }}"]
changed_when: false
check_mode: false
when: server_backup_export_enabled | bool
- name: Ensure prepared Prometheus backup versions directory exists
tags: [services, backup, prometheus_backup]
ansible.builtin.file:
path: "{{ server_backup_export_root }}/versions"
state: directory
owner: root
group: "{{ server_backup_username }}"
mode: "0750"
when: server_backup_export_enabled | bool
- name: Install Prometheus backup export helper
tags: [services, backup, prometheus_backup]
ansible.builtin.template:
src: prometheus-backup-export.sh.j2
dest: /usr/local/sbin/prometheus-backup-export
owner: root
group: root
mode: "0750"
when: server_backup_export_enabled | bool
- name: Install Prometheus backup export systemd units
tags: [services, backup, prometheus_backup]
ansible.builtin.template:
src: "{{ item }}.j2"
dest: "/etc/systemd/system/{{ item }}"
owner: root
group: root
mode: "0644"
loop:
- prometheus-backup-export.service
- prometheus-backup-export.timer
loop_control:
label: "{{ item }}"
register: server_backup_export_units
when: server_backup_export_enabled | bool
- name: Reload systemd after Prometheus backup export unit changes
tags: [services, backup, prometheus_backup]
ansible.builtin.systemd:
daemon_reload: true
when:
- server_backup_export_enabled | bool
- server_backup_export_units is changed
- not ansible_check_mode
- name: Enable Prometheus backup export timer only after explicit activation
tags: [services, backup, prometheus_backup]
ansible.builtin.systemd:
name: prometheus-backup-export.timer
enabled: true
state: started
when:
- server_backup_export_enabled | bool
- server_backup_export_start_timer | bool
- not ansible_check_mode

View File

@@ -53,6 +53,12 @@
tags: [services, podman]
ansible.builtin.include_tasks: podman-compose.yml
- name: Import Prometheus backup export identity tasks
ansible.builtin.import_tasks: backup_export_identity.yml
- name: Import Prometheus backup export job tasks
ansible.builtin.import_tasks: backup_export_job.yml
- name: Ensure server SSH authorized key fragments directory exists
tags: [services, ssh]
ansible.builtin.file:
@@ -77,13 +83,17 @@
when: server_ssh_authorized_keys | length > 0
- name: Configure server SSH authorized key fragments
tags: [services, ssh]
tags: [services, ssh, prometheus_backup]
ansible.builtin.lineinfile:
path: /etc/ssh/sshd_config
regexp: '^\s*AuthorizedKeysFile\s+'
line: >-
AuthorizedKeysFile {{ server_ssh_authorized_keys | map(attribute='name')
| map('regex_replace', '^', '%h/.ssh/authorized_keys.d/') | join(' ') }}
AuthorizedKeysFile {{
((server_ssh_authorized_keys | map(attribute='name')
| map('regex_replace', '^', '%h/.ssh/authorized_keys.d/') | list)
+ (['%h/.ssh/authorized_keys.d/' ~ server_backup_public_key_name]
if server_backup_export_enabled | bool else [])) | join(' ')
}}
state: present
validate: "sshd -t -f %s"
notify: Reload SSH service
@@ -100,11 +110,13 @@
notify: Reload SSH service
- name: Restrict SSH login to allowed users on server
tags: [services]
tags: [services, prometheus_backup]
ansible.builtin.lineinfile:
path: /etc/ssh/sshd_config
regexp: '^\s*AllowUsers\s+'
line: "AllowUsers {{ server_sshd_allow_users | join(' ') }}"
line: >-
AllowUsers {{ (server_sshd_allow_users
+ ([server_backup_username] if server_backup_export_enabled | bool else [])) | join(' ') }}
state: present
validate: "sshd -t -f %s"
notify: Reload SSH service

View File

@@ -0,0 +1,15 @@
[Unit]
Description=Prepare a read-only Prometheus application backup for Atlas
RequiresMountsFor=/opt/npm /opt/gitea {{ server_backup_export_root }}
ConditionFileIsExecutable=/usr/local/sbin/prometheus-backup-export
[Service]
Type=oneshot
ExecStart=/usr/local/sbin/prometheus-backup-export
User=root
Group=root
UMask=0077
TimeoutStartSec=infinity
Nice=10
IOSchedulingClass=best-effort
IOSchedulingPriority=7

View File

@@ -0,0 +1,92 @@
#!/usr/bin/env bash
set -Eeuo pipefail
umask 077
export_root={{ server_backup_export_root | quote }}
versions="$export_root/versions"
stack_unit=podman-compose-server.service
stamp=$(date -u +%Y%m%dT%H%M%SZ)
stage=''
stack_stopped=false
exec 9>/run/lock/prometheus-backup-export.lock
flock -n 9 || { echo 'A backup export is already running' >&2; exit 1; }
cleanup() {
local rc=$?
trap - EXIT
if "$stack_stopped"; then
if systemctl is-active --quiet "$stack_unit"; then
systemctl restart "$stack_unit" || rc=1
else
systemctl start "$stack_unit" || rc=1
fi
fi
if (( rc != 0 )) && [[ -n "$stage" && -d "$stage" ]]; then
rm -rf -- "$stage"
fi
exit "$rc"
}
trap cleanup EXIT
trap 'exit 129' HUP
trap 'exit 130' INT
trap 'exit 143' TERM
systemctl is-active --quiet "$stack_unit" || {
echo 'The managed Compose stack must be active before preparing a backup' >&2
exit 1
}
paths=(
{% for path in server_backup_export_paths %}
{{ path | quote }}
{% endfor %}
)
excludes=(
{% for path in server_backup_export_excludes %}
--exclude={{ path | quote }}
{% endfor %}
)
for path in "${paths[@]}"; do
[[ -e "/$path" ]] || { echo "Required backup path missing: /$path" >&2; exit 1; }
done
[[ ! -e "$versions/$stamp" ]] || { echo "Export version already exists: $stamp" >&2; exit 1; }
stage=$(mktemp -d "$export_root/.staging.XXXXXXXX")
# SQLite databases and their accompanying files are copied while both
# managed containers are stopped. The EXIT trap restarts the stack on error.
stack_stopped=true
systemctl stop "$stack_unit"
tar --acls --xattrs --selinux "${excludes[@]}" -C / -cf "$stage/payload.tar" "${paths[@]}"
systemctl start "$stack_unit"
for container in nginx-proxy-manager gitea; do
running=false
for _ in {1..30}; do
if [[ $(podman inspect --format '{{ '{{.State.Running}}' }}' "$container" 2>/dev/null) == true ]]; then
running=true
break
fi
sleep 2
done
"$running" || { echo "Container did not restart: $container" >&2; exit 1; }
done
stack_stopped=false
tar -tf "$stage/payload.tar" >/dev/null
(cd "$stage" && sha256sum payload.tar >payload.sha256)
printf '{"schema":1,"host":"prometheus","created_utc":"%s"}\n' "$stamp" >"$stage/metadata.json"
chown root:{{ server_backup_username }} "$stage" "$stage/payload.tar" "$stage/payload.sha256" "$stage/metadata.json"
chmod 0750 "$stage"
chmod 0640 "$stage/payload.tar" "$stage/payload.sha256" "$stage/metadata.json"
mv -- "$stage" "$versions/$stamp"
stage=''
ln -s "$stamp" "$versions/.current.new"
mv -Tf -- "$versions/.current.new" "$versions/current"
# Keep a small source-side safety window; Atlas owns long-term retention.
mapfile -t old_versions < <(find "$versions" -mindepth 1 -maxdepth 1 -type d \
-printf '%f\n' | grep -E '^[0-9]{8}T[0-9]{6}Z$' | sort -r | tail -n +{{ server_backup_export_source_keep + 1 }})
for old in "${old_versions[@]}"; do
rm -rf -- "${versions:?}/$old"
done
echo "Prepared Prometheus backup export $stamp"

View File

@@ -0,0 +1,10 @@
[Unit]
Description=Prepare daily Prometheus application backup for Atlas
[Timer]
OnCalendar={{ server_backup_export_calendar }}
Persistent=false
Unit=prometheus-backup-export.service
[Install]
WantedBy=timers.target