mirror of
https://github.com/fscotto/infra.git
synced 2026-10-06 06:49:49 +00:00
Add verified Prometheus backup pull to Atlas
This commit is contained in:
@@ -115,6 +115,15 @@ atlas_monitor_notifier: "{{ atlas_usb_reminder_notifier }}"
|
||||
atlas_monitor_smart_devices: []
|
||||
atlas_monitor_timers: []
|
||||
atlas_monitor_failure_units: []
|
||||
atlas_monitor_effective_timers: >-
|
||||
{{ atlas_monitor_timers
|
||||
+ ([{'name': 'atlas-prometheus-pull.timer', 'max_age_hours': 26}]
|
||||
if atlas_manage_prometheus_backup_pull | bool and atlas_prometheus_pull_start_timer | bool
|
||||
else []) }}
|
||||
atlas_monitor_effective_failure_units: >-
|
||||
{{ atlas_monitor_failure_units
|
||||
+ (['atlas-prometheus-pull.service']
|
||||
if atlas_manage_prometheus_backup_pull | bool else []) }}
|
||||
atlas_monitor_remote_capacity: {}
|
||||
atlas_monitor_pool_warning_percent: 80
|
||||
atlas_monitor_pool_critical_percent: 90
|
||||
@@ -141,6 +150,19 @@ atlas_music_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_music }}"
|
||||
atlas_backup_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_backup }}"
|
||||
atlas_host_backups_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_host_backups }}"
|
||||
atlas_backup_prometheus_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_backup_prometheus }}"
|
||||
atlas_manage_prometheus_backup_pull: false
|
||||
atlas_prometheus_pull_ssh_dir: /etc/atlas-prometheus-pull
|
||||
atlas_prometheus_pull_private_key_path: "{{ atlas_prometheus_pull_ssh_dir }}/id_ed25519"
|
||||
atlas_prometheus_pull_known_hosts_path: "{{ atlas_prometheus_pull_ssh_dir }}/known_hosts"
|
||||
atlas_prometheus_ssh_host_key: ""
|
||||
atlas_prometheus_pull_source_user: prometheus-backup
|
||||
atlas_prometheus_pull_source_port: 22
|
||||
atlas_prometheus_pull_calendar: "*-*-* 03:00:00 Europe/Rome"
|
||||
atlas_prometheus_pull_start_timer: false
|
||||
atlas_prometheus_pull_keep_daily: 30
|
||||
atlas_prometheus_pull_keep_weekly: 8
|
||||
atlas_prometheus_pull_keep_monthly: 12
|
||||
atlas_prometheus_pull_max_age_hours: 24
|
||||
atlas_photobook_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_photobook }}"
|
||||
|
||||
atlas_45drives_repo_url: https://repo.45drives.com/repofiles/rocky/45drives-enterprise.repo
|
||||
|
||||
56
ansible/roles/profile_atlas/files/atlas-prometheus-prune.py
Normal file
56
ansible/roles/profile_atlas/files/atlas-prometheus-prune.py
Normal file
@@ -0,0 +1,56 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Prune only verified, named Prometheus backup versions after publication."""
|
||||
|
||||
import datetime as dt
|
||||
import pathlib
|
||||
import re
|
||||
import shutil
|
||||
import sys
|
||||
|
||||
|
||||
def main() -> None:
|
||||
if len(sys.argv) != 5:
|
||||
raise SystemExit("Usage: atlas-prometheus-prune SNAPSHOTS DAILY WEEKLY MONTHLY")
|
||||
root = pathlib.Path(sys.argv[1])
|
||||
counts = [int(value) for value in sys.argv[2:]]
|
||||
if not root.is_dir() or root.is_symlink() or min(counts) < 1:
|
||||
raise SystemExit("Invalid backup directory or retention counts")
|
||||
versions = []
|
||||
for entry in root.iterdir():
|
||||
if not entry.is_dir() or entry.is_symlink():
|
||||
continue
|
||||
if not re.fullmatch(r"[0-9]{8}T[0-9]{6}Z", entry.name):
|
||||
continue
|
||||
try:
|
||||
when = dt.datetime.strptime(entry.name, "%Y%m%dT%H%M%SZ")
|
||||
except ValueError:
|
||||
continue
|
||||
if not all((entry / name).is_file() for name in ("payload.tar", "payload.sha256", "metadata.json")):
|
||||
continue
|
||||
versions.append((when, entry))
|
||||
versions.sort(reverse=True)
|
||||
if not versions:
|
||||
raise SystemExit("No published backup versions found; refusing to prune")
|
||||
|
||||
keep = {entry for _, entry in versions[: counts[0]]}
|
||||
for count, key in (
|
||||
(counts[1], lambda when: when.isocalendar()[:2]),
|
||||
(counts[2], lambda when: (when.year, when.month)),
|
||||
):
|
||||
periods = set()
|
||||
for when, entry in versions:
|
||||
period = key(when)
|
||||
if period in periods:
|
||||
continue
|
||||
periods.add(period)
|
||||
keep.add(entry)
|
||||
if len(periods) >= count:
|
||||
break
|
||||
|
||||
for _, entry in versions:
|
||||
if entry not in keep:
|
||||
shutil.rmtree(entry)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -23,6 +23,12 @@
|
||||
- name: Import Atlas offline USB backup tasks
|
||||
ansible.builtin.import_tasks: usb_backup.yml
|
||||
|
||||
- name: Import Atlas Prometheus backup pull identity tasks
|
||||
ansible.builtin.import_tasks: prometheus_pull_identity.yml
|
||||
|
||||
- name: Import Atlas Prometheus backup pull job tasks
|
||||
ansible.builtin.import_tasks: prometheus_pull_job.yml
|
||||
|
||||
- name: Import Atlas health monitoring tasks
|
||||
ansible.builtin.import_tasks: monitoring.yml
|
||||
|
||||
|
||||
@@ -7,8 +7,8 @@
|
||||
- atlas_zfs_pool != 'CHANGEME_ZFS_POOL'
|
||||
- atlas_monitor_calendar | length > 0
|
||||
- atlas_monitor_smart_devices | length > 0
|
||||
- atlas_monitor_timers | length > 0
|
||||
- atlas_monitor_failure_units | length > 0
|
||||
- atlas_monitor_effective_timers | length > 0
|
||||
- atlas_monitor_effective_failure_units | length > 0
|
||||
- atlas_monitor_remote_capacity.user == atlas_borg_repository_user
|
||||
- atlas_monitor_remote_capacity.host == atlas_borg_repository_host
|
||||
- atlas_monitor_remote_capacity.run_as == atlas_borg_username
|
||||
@@ -49,7 +49,7 @@
|
||||
that:
|
||||
- item.name is match('^[a-zA-Z0-9@_.-]+\\.timer$')
|
||||
- item.max_age_hours | int >= 0
|
||||
loop: "{{ atlas_monitor_timers }}"
|
||||
loop: "{{ atlas_monitor_effective_timers }}"
|
||||
loop_control:
|
||||
label: "{{ item.name }}"
|
||||
when: atlas_manage_monitoring | bool
|
||||
@@ -59,7 +59,7 @@
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- item is match('^[a-zA-Z0-9@_.-]+\\.service$')
|
||||
loop: "{{ atlas_monitor_failure_units }}"
|
||||
loop: "{{ atlas_monitor_effective_failure_units }}"
|
||||
when: atlas_manage_monitoring | bool
|
||||
|
||||
- name: Validate Atlas health monitor calendar
|
||||
@@ -144,7 +144,7 @@
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0755"
|
||||
loop: "{{ atlas_monitor_failure_units }}"
|
||||
loop: "{{ atlas_monitor_effective_failure_units }}"
|
||||
when: atlas_manage_monitoring | bool
|
||||
|
||||
- name: Notify 45Drives Alerts when an Atlas job fails
|
||||
@@ -155,7 +155,7 @@
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0644"
|
||||
loop: "{{ atlas_monitor_failure_units }}"
|
||||
loop: "{{ atlas_monitor_effective_failure_units }}"
|
||||
when: atlas_manage_monitoring | bool
|
||||
|
||||
- name: Reload systemd after installing Atlas monitoring
|
||||
|
||||
@@ -0,0 +1,66 @@
|
||||
---
|
||||
- name: Validate Atlas Prometheus pull identity inputs
|
||||
tags: [atlas, backup, prometheus_backup, prometheus_backup_key]
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- atlas_prometheus_pull_ssh_dir.startswith('/etc/')
|
||||
- atlas_prometheus_pull_private_key_path.startswith(atlas_prometheus_pull_ssh_dir ~ '/')
|
||||
- atlas_prometheus_pull_known_hosts_path.startswith(atlas_prometheus_pull_ssh_dir ~ '/')
|
||||
- atlas_prometheus_ssh_host_key.startswith(
|
||||
(hostvars['prometheus'].ansible_host | string) ~ ' ssh-ed25519 '
|
||||
)
|
||||
fail_msg: Pin the verified Prometheus ED25519 SSH host key before enabling the pull.
|
||||
when: atlas_manage_prometheus_backup_pull | bool
|
||||
|
||||
- name: Create private Atlas Prometheus pull SSH directory
|
||||
tags: [atlas, backup, prometheus_backup, prometheus_backup_key]
|
||||
ansible.builtin.file:
|
||||
path: "{{ atlas_prometheus_pull_ssh_dir }}"
|
||||
state: directory
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0700"
|
||||
when: atlas_manage_prometheus_backup_pull | bool
|
||||
|
||||
- name: Generate Atlas-only Prometheus pull SSH identity
|
||||
tags: [atlas, backup, prometheus_backup, prometheus_backup_key]
|
||||
ansible.builtin.command:
|
||||
argv:
|
||||
- ssh-keygen
|
||||
- -q
|
||||
- -t
|
||||
- ed25519
|
||||
- -N
|
||||
- ""
|
||||
- -C
|
||||
- atlas-prometheus-pull@atlas
|
||||
- -f
|
||||
- "{{ atlas_prometheus_pull_private_key_path }}"
|
||||
creates: "{{ atlas_prometheus_pull_private_key_path }}"
|
||||
when: atlas_manage_prometheus_backup_pull | bool
|
||||
|
||||
- name: Protect Atlas-only Prometheus pull SSH identity
|
||||
tags: [atlas, backup, prometheus_backup, prometheus_backup_key]
|
||||
ansible.builtin.file:
|
||||
path: "{{ item.path }}"
|
||||
owner: root
|
||||
group: root
|
||||
mode: "{{ item.mode }}"
|
||||
loop:
|
||||
- { path: "{{ atlas_prometheus_pull_private_key_path }}", mode: "0600" }
|
||||
- { path: "{{ atlas_prometheus_pull_private_key_path }}.pub", mode: "0644" }
|
||||
loop_control:
|
||||
label: "{{ item.path }}"
|
||||
when:
|
||||
- atlas_manage_prometheus_backup_pull | bool
|
||||
- not ansible_check_mode
|
||||
|
||||
- name: Pin Prometheus SSH host key on Atlas
|
||||
tags: [atlas, backup, prometheus_backup, prometheus_backup_key]
|
||||
ansible.builtin.copy:
|
||||
content: "{{ atlas_prometheus_ssh_host_key }}\n"
|
||||
dest: "{{ atlas_prometheus_pull_known_hosts_path }}"
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0600"
|
||||
when: atlas_manage_prometheus_backup_pull | bool
|
||||
90
ansible/roles/profile_atlas/tasks/prometheus_pull_job.yml
Normal file
90
ansible/roles/profile_atlas/tasks/prometheus_pull_job.yml
Normal file
@@ -0,0 +1,90 @@
|
||||
---
|
||||
- name: Validate Atlas Prometheus backup pull inputs
|
||||
tags: [atlas, backup, prometheus_backup]
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- atlas_manage_storage | bool
|
||||
- atlas_prometheus_pull_source_user is match('^[a-z_][a-z0-9_-]*$')
|
||||
- atlas_prometheus_pull_source_port | int > 0
|
||||
- atlas_prometheus_pull_source_port | int < 65536
|
||||
- atlas_prometheus_pull_keep_daily | int > 0
|
||||
- atlas_prometheus_pull_keep_weekly | int > 0
|
||||
- atlas_prometheus_pull_keep_monthly | int > 0
|
||||
- atlas_prometheus_pull_max_age_hours | int > 0
|
||||
- atlas_backup_prometheus_mountpoint.startswith(atlas_mount_root ~ '/')
|
||||
fail_msg: Define the Atlas backup destination, source account, and retention before enabling the pull.
|
||||
when: atlas_manage_prometheus_backup_pull | bool
|
||||
|
||||
- name: Validate Atlas Prometheus backup pull calendar
|
||||
tags: [atlas, backup, prometheus_backup]
|
||||
ansible.builtin.command:
|
||||
argv: [systemd-analyze, calendar, "{{ atlas_prometheus_pull_calendar }}"]
|
||||
changed_when: false
|
||||
check_mode: false
|
||||
when: atlas_manage_prometheus_backup_pull | bool
|
||||
|
||||
- name: Create private Atlas Prometheus backup version directory
|
||||
tags: [atlas, backup, prometheus_backup]
|
||||
ansible.builtin.file:
|
||||
path: "{{ atlas_backup_prometheus_mountpoint }}/snapshots"
|
||||
state: directory
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0700"
|
||||
when: atlas_manage_prometheus_backup_pull | bool
|
||||
|
||||
- name: Install Atlas Prometheus backup pull helper
|
||||
tags: [atlas, backup, prometheus_backup]
|
||||
ansible.builtin.template:
|
||||
src: atlas-prometheus-pull.sh.j2
|
||||
dest: /usr/local/sbin/atlas-prometheus-pull
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0750"
|
||||
when: atlas_manage_prometheus_backup_pull | bool
|
||||
|
||||
- name: Install Atlas Prometheus backup retention helper
|
||||
tags: [atlas, backup, prometheus_backup]
|
||||
ansible.builtin.copy:
|
||||
src: atlas-prometheus-prune.py
|
||||
dest: /usr/local/libexec/atlas-prometheus-prune
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0750"
|
||||
when: atlas_manage_prometheus_backup_pull | bool
|
||||
|
||||
- name: Install Atlas Prometheus backup pull systemd units
|
||||
tags: [atlas, backup, prometheus_backup]
|
||||
ansible.builtin.template:
|
||||
src: "{{ item }}.j2"
|
||||
dest: "/etc/systemd/system/{{ item }}"
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0644"
|
||||
loop:
|
||||
- atlas-prometheus-pull.service
|
||||
- atlas-prometheus-pull.timer
|
||||
loop_control:
|
||||
label: "{{ item }}"
|
||||
register: atlas_prometheus_pull_units
|
||||
when: atlas_manage_prometheus_backup_pull | bool
|
||||
|
||||
- name: Reload systemd after Atlas Prometheus pull unit changes
|
||||
tags: [atlas, backup, prometheus_backup]
|
||||
ansible.builtin.systemd:
|
||||
daemon_reload: true
|
||||
when:
|
||||
- atlas_manage_prometheus_backup_pull | bool
|
||||
- atlas_prometheus_pull_units is changed
|
||||
- not ansible_check_mode
|
||||
|
||||
- name: Enable Atlas Prometheus pull timer only after explicit activation
|
||||
tags: [atlas, backup, prometheus_backup]
|
||||
ansible.builtin.systemd:
|
||||
name: atlas-prometheus-pull.timer
|
||||
enabled: true
|
||||
state: started
|
||||
when:
|
||||
- atlas_manage_prometheus_backup_pull | bool
|
||||
- atlas_prometheus_pull_start_timer | bool
|
||||
- not ansible_check_mode
|
||||
@@ -3,8 +3,8 @@
|
||||
"backup_dataset": {{ (atlas_zfs_pool ~ '/' ~ atlas_zfs_dataset_backup) | to_json }},
|
||||
"notifier": {{ atlas_monitor_notifier | to_json }},
|
||||
"smart_devices": {{ atlas_monitor_smart_devices | to_json }},
|
||||
"timers": {{ atlas_monitor_timers | to_json }},
|
||||
"failure_units": {{ atlas_monitor_failure_units | to_json }},
|
||||
"timers": {{ atlas_monitor_effective_timers | to_json }},
|
||||
"failure_units": {{ atlas_monitor_effective_failure_units | to_json }},
|
||||
"remote_capacity": {{ atlas_monitor_remote_capacity | to_json }},
|
||||
"pool_warning_percent": {{ atlas_monitor_pool_warning_percent | int }},
|
||||
"pool_critical_percent": {{ atlas_monitor_pool_critical_percent | int }},
|
||||
|
||||
@@ -0,0 +1,19 @@
|
||||
[Unit]
|
||||
Description=Pull a prepared read-only Prometheus backup to Atlas
|
||||
RequiresMountsFor={{ atlas_backup_prometheus_mountpoint }}
|
||||
Wants=network-online.target
|
||||
After=network-online.target zfs.target
|
||||
ConditionFileIsExecutable=/usr/local/sbin/atlas-prometheus-pull
|
||||
ConditionPathExists={{ atlas_prometheus_pull_private_key_path }}
|
||||
ConditionPathExists={{ atlas_prometheus_pull_known_hosts_path }}
|
||||
|
||||
[Service]
|
||||
Type=oneshot
|
||||
ExecStart=/usr/local/sbin/atlas-prometheus-pull
|
||||
User=root
|
||||
Group=root
|
||||
UMask=0077
|
||||
TimeoutStartSec=infinity
|
||||
Nice=15
|
||||
IOSchedulingClass=best-effort
|
||||
IOSchedulingPriority=7
|
||||
@@ -0,0 +1,75 @@
|
||||
#!/usr/bin/env bash
|
||||
set -Eeuo pipefail
|
||||
umask 077
|
||||
|
||||
backup_root={{ atlas_backup_prometheus_mountpoint | quote }}
|
||||
snapshots="$backup_root/snapshots"
|
||||
stage=''
|
||||
exec 9>/run/lock/atlas-prometheus-pull.lock
|
||||
flock -n 9 || { echo 'A Prometheus pull is already running' >&2; exit 1; }
|
||||
|
||||
cleanup() {
|
||||
local rc=$?
|
||||
trap - EXIT
|
||||
if (( rc != 0 )) && [[ -n "$stage" && -d "$stage" ]]; then
|
||||
rm -rf -- "$stage"
|
||||
fi
|
||||
exit "$rc"
|
||||
}
|
||||
trap cleanup EXIT
|
||||
|
||||
zpool list -H -o name {{ atlas_zfs_pool | quote }} >/dev/null
|
||||
findmnt -rn --mountpoint "$backup_root" >/dev/null
|
||||
stage=$(mktemp -d "$backup_root/.staging.XXXXXXXX")
|
||||
ssh_cmd='/usr/bin/ssh -F /dev/null -o BatchMode=yes -o StrictHostKeyChecking=yes -o UserKnownHostsFile={{ atlas_prometheus_pull_known_hosts_path }} -o IdentitiesOnly=yes -i {{ atlas_prometheus_pull_private_key_path }} -p {{ atlas_prometheus_pull_source_port }}'
|
||||
rsync -a --partial --delay-updates -e "$ssh_cmd" \
|
||||
{{ (atlas_prometheus_pull_source_user ~ '@' ~ hostvars['prometheus'].ansible_host ~ ':current/') | quote }} \
|
||||
"$stage/"
|
||||
|
||||
test -s "$stage/payload.tar"
|
||||
test -s "$stage/payload.sha256"
|
||||
test -s "$stage/metadata.json"
|
||||
(cd "$stage" && sha256sum -c payload.sha256)
|
||||
tar -tf "$stage/payload.tar" >/dev/null
|
||||
stamp=$(python3 - "$stage/metadata.json" <<'PY'
|
||||
import json
|
||||
import datetime as dt
|
||||
import re
|
||||
import sys
|
||||
|
||||
with open(sys.argv[1], encoding="utf-8") as stream:
|
||||
metadata = json.load(stream)
|
||||
stamp = metadata.get("created_utc", "")
|
||||
if metadata.get("schema") != 1 or metadata.get("host") != "prometheus":
|
||||
raise SystemExit("Unexpected Prometheus backup metadata")
|
||||
if not re.fullmatch(r"[0-9]{8}T[0-9]{6}Z", stamp):
|
||||
raise SystemExit("Invalid Prometheus backup timestamp")
|
||||
created = dt.datetime.strptime(stamp, "%Y%m%dT%H%M%SZ").replace(tzinfo=dt.timezone.utc)
|
||||
age = dt.datetime.now(dt.timezone.utc) - created
|
||||
if age.total_seconds() < -300 or age > dt.timedelta(hours={{ atlas_prometheus_pull_max_age_hours }}):
|
||||
raise SystemExit("Prometheus backup is outside the configured freshness window")
|
||||
print(stamp)
|
||||
PY
|
||||
)
|
||||
if [[ -e "$snapshots/$stamp" ]]; then
|
||||
cmp "$stage/payload.sha256" "$snapshots/$stamp/payload.sha256"
|
||||
cmp "$stage/metadata.json" "$snapshots/$stamp/metadata.json"
|
||||
(cd "$snapshots/$stamp" && sha256sum -c payload.sha256)
|
||||
rm -rf -- "${stage:?}"
|
||||
stage=''
|
||||
else
|
||||
chown -R root:root "$stage"
|
||||
chmod 0700 "$stage"
|
||||
chmod 0600 "$stage/payload.tar" "$stage/payload.sha256" "$stage/metadata.json"
|
||||
mv -- "$stage" "$snapshots/$stamp"
|
||||
stage=''
|
||||
fi
|
||||
latest_link=$(readlink "$backup_root/latest" 2>/dev/null || true)
|
||||
latest_stamp=${latest_link##*/}
|
||||
if [[ -z "$latest_stamp" || "$stamp" > "$latest_stamp" ]]; then
|
||||
ln -s "snapshots/$stamp" "$backup_root/.latest.new"
|
||||
mv -Tf -- "$backup_root/.latest.new" "$backup_root/latest"
|
||||
fi
|
||||
python3 /usr/local/libexec/atlas-prometheus-prune "$snapshots" \
|
||||
{{ atlas_prometheus_pull_keep_daily }} {{ atlas_prometheus_pull_keep_weekly }} {{ atlas_prometheus_pull_keep_monthly }}
|
||||
echo "Verified and published Prometheus backup $stamp"
|
||||
@@ -0,0 +1,10 @@
|
||||
[Unit]
|
||||
Description=Schedule Atlas pull of prepared Prometheus backups
|
||||
|
||||
[Timer]
|
||||
OnCalendar={{ atlas_prometheus_pull_calendar }}
|
||||
Persistent=true
|
||||
Unit=atlas-prometheus-pull.service
|
||||
|
||||
[Install]
|
||||
WantedBy=timers.target
|
||||
Reference in New Issue
Block a user