mirror of
https://github.com/fscotto/infra.git
synced 2026-10-04 22:09:50 +00:00
Compare commits
52 Commits
de2c24d15c
...
main
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
9f95e68190 | ||
|
|
def3dbf313 | ||
|
|
a00602973c | ||
|
|
18eb2d2eb2 | ||
|
|
269fb13665 | ||
|
|
2dfe766b7b | ||
|
|
755f24bc72 | ||
|
|
7bc7f0e645 | ||
|
|
1577eec19d | ||
|
|
e30683c3d1 | ||
|
|
4bd6aafb53 | ||
|
|
9e76309833 | ||
|
|
ed3fee06e8 | ||
|
|
309d64b4ed | ||
|
|
dd33a4f55d | ||
|
|
0028fe8c4d | ||
|
|
12037fcc9a | ||
|
|
3f9a626759 | ||
|
|
31fedb8d44 | ||
|
|
9b5ee77905 | ||
|
|
54fb7d46d7 | ||
|
|
a609e68f42 | ||
|
|
06d3b175cb | ||
|
|
256d758b1a | ||
|
|
9d0013769c | ||
|
|
5b0f415163 | ||
|
|
3401b6137d | ||
|
|
bae6a9f554 | ||
|
|
c37483ba38 | ||
|
|
f491362365 | ||
|
|
5a1047adde | ||
|
|
802cb8c7ba | ||
|
|
0144600a4a | ||
|
|
3d2ef02c98 | ||
|
|
8844d00e24 | ||
|
|
9798fe3a12 | ||
|
|
702283b430 | ||
|
|
797087c66f | ||
|
|
fa1c8c0b82 | ||
|
|
0b6efc9ad8 | ||
|
|
361ee77d72 | ||
|
|
d4e40d423a | ||
|
|
0a5c2ac1a4 | ||
|
|
48a7f57f7e | ||
|
|
defa98c968 | ||
|
|
21e41f4fc1 | ||
|
|
e10c6694f8 | ||
|
|
4c10af3187 | ||
|
|
3ac732751c | ||
|
|
e837b0059b | ||
|
|
7e498514dd | ||
|
|
e7836ea25f |
412
AGENTS.md
412
AGENTS.md
@@ -25,6 +25,9 @@ Ansible-driven personal infrastructure repo for Fedora and Void desktops, Fedora
|
|||||||
- Preserve layering `all -> platform -> role -> desktop -> host`.
|
- Preserve layering `all -> platform -> role -> desktop -> host`.
|
||||||
- Keep `ansible/site.yml` small; orchestration belongs there, implementation belongs in roles.
|
- Keep `ansible/site.yml` small; orchestration belongs there, implementation belongs in roles.
|
||||||
- Prefer minimal, targeted edits. Preserve idempotency and existing ordering.
|
- Prefer minimal, targeted edits. Preserve idempotency and existing ordering.
|
||||||
|
- Keep completed one-time cleanup operations out of the playbook. Execute them directly
|
||||||
|
with explicit authorization; retain only the ongoing desired-state configuration and
|
||||||
|
historical documentation, not permanent cleanup flags or tasks.
|
||||||
- Use Git Flow branch prefixes: `feature/` for new functionality, `bugfix/` for non-urgent fixes,
|
- Use Git Flow branch prefixes: `feature/` for new functionality, `bugfix/` for non-urgent fixes,
|
||||||
`hotfix/` for urgent production fixes, `release/` for release preparation, and `support/` for
|
`hotfix/` for urgent production fixes, `release/` for release preparation, and `support/` for
|
||||||
maintained release lines. Do not use abbreviated prefixes such as `feat/`.
|
maintained release lines. Do not use abbreviated prefixes such as `feat/`.
|
||||||
@@ -54,14 +57,44 @@ Ansible-driven personal infrastructure repo for Fedora and Void desktops, Fedora
|
|||||||
- Emacs is disabled by default; temporary Emacs check: `ansible-playbook ansible/site.yml --limit <host> --tags emacs --check --diff -e emacs_enabled=true`
|
- Emacs is disabled by default; temporary Emacs check: `ansible-playbook ansible/site.yml --limit <host> --tags emacs --check --diff -e emacs_enabled=true`
|
||||||
- AI coding agents: `ansible-playbook ansible/site.yml --limit <host> --tags ai_agents --check --diff`
|
- AI coding agents: `ansible-playbook ansible/site.yml --limit <host> --tags ai_agents --check --diff`
|
||||||
- Mail bootstrap: `sh -n scripts/bootstrap_mail.sh` and `shellcheck scripts/bootstrap_mail.sh`
|
- Mail bootstrap: `sh -n scripts/bootstrap_mail.sh` and `shellcheck scripts/bootstrap_mail.sh`
|
||||||
- Server compose render: `podman-compose -f /opt/docker/server/docker-compose.yml config` and `systemctl status podman-compose-server`
|
- Server NPM Quadlet: `systemctl status prometheus-npm.service`; the Compose fallback is retired.
|
||||||
|
- Explicit Prometheus legacy cleanup (destructive only without check mode):
|
||||||
|
`ansible-playbook ansible/site.yml --limit prometheus --tags server_legacy_cleanup --check --diff -e server_legacy_cleanup=true`
|
||||||
- Atlas media stack:
|
- Atlas media stack:
|
||||||
`ansible-playbook ansible/site.yml --limit atlas --tags storage,sharing,containers --check --diff`
|
`ansible-playbook ansible/site.yml --limit atlas --tags storage,sharing,containers --check --diff`
|
||||||
|
- Atlas rootless Gitea staging (does not start Gitea):
|
||||||
|
`ansible-playbook ansible/site.yml --limit atlas --tags gitea --check --diff`
|
||||||
|
- Atlas canonical Gitea domain (restarts only Gitea on a real configuration change):
|
||||||
|
`ansible-playbook ansible/site.yml --limit atlas --tags gitea_public_domain --check --diff`
|
||||||
|
- Atlas Nextcloud/ONLYOFFICE steady state:
|
||||||
|
`ansible-playbook ansible/site.yml --limit atlas --tags nextcloud --check --diff`
|
||||||
|
- Atlas recurring consistent Nextcloud backup preparation:
|
||||||
|
`ansible-playbook ansible/site.yml --limit atlas --tags nextcloud_backup,monitoring --check --diff`
|
||||||
|
- Atlas iCloudPD storage and boot-started Quadlet:
|
||||||
|
`ansible-playbook ansible/site.yml --limit atlas --tags icloudpd --check --diff`
|
||||||
|
- Ongoing Gitea proxy configuration:
|
||||||
|
`ansible-playbook ansible/site.yml --limit prometheus --tags gitea_cutover,prometheus_backup --check --diff -e server_gitea_on_atlas=true`
|
||||||
|
and `ansible-playbook ansible/site.yml --limit atlas --tags gitea --check --diff`
|
||||||
|
- Atlas daily Navidrome music copy:
|
||||||
|
`ansible-playbook ansible/site.yml --limit atlas --tags music_sync --check --diff`
|
||||||
- Atlas network/share hardening:
|
- Atlas network/share hardening:
|
||||||
`ansible-playbook ansible/site.yml --limit atlas --tags hardening,sharing --check --diff`
|
`ansible-playbook ansible/site.yml --limit atlas --tags hardening,sharing --check --diff`
|
||||||
|
- Atlas ZFS snapshot retention and scrub timers:
|
||||||
|
`ansible-playbook ansible/site.yml --limit atlas --tags snapshots,scrub --check --diff`
|
||||||
|
- Atlas encrypted Borg backup:
|
||||||
|
`ansible-playbook ansible/site.yml --limit atlas --tags packages,borg --check --diff`
|
||||||
|
- Atlas Borg progress logging only:
|
||||||
|
`ansible-playbook ansible/site.yml --limit atlas --tags borg_logging --check --diff`
|
||||||
|
- Atlas manual offline USB backup and 45Drives Alerts reminder:
|
||||||
|
`ansible-playbook ansible/site.yml --limit atlas --tags usb_backup,usb_reminder --check --diff`
|
||||||
|
- Atlas pool, disk, capacity, temperature, and job monitoring:
|
||||||
|
`ansible-playbook ansible/site.yml --limit atlas --tags monitoring --check --diff`
|
||||||
|
- Atlas explicit post-restore SELinux relabeling:
|
||||||
|
`ansible-playbook ansible/site.yml --limit atlas --tags restorecon --check -e '{"atlas_restorecon_paths":["/zpool/archive"]}'`
|
||||||
- Prometheus/Aegis WireGuard gateway:
|
- Prometheus/Aegis WireGuard gateway:
|
||||||
`ansible-playbook ansible/site.yml --limit prometheus,aegis --tags wireguard --check --diff`
|
`ansible-playbook ansible/site.yml --limit prometheus,aegis --tags wireguard --check --diff`
|
||||||
- DuckDNS config only: `ansible-playbook ansible/site.yml --limit prometheus --tags duckdns --check --diff`
|
- Prometheus NPM Quadlet steady state (does not perform a cutover):
|
||||||
|
`ansible-playbook ansible/site.yml --limit prometheus --tags npm_quadlet --check --diff`
|
||||||
|
|
||||||
## Conventions
|
## Conventions
|
||||||
- Use FQCN Ansible modules.
|
- Use FQCN Ansible modules.
|
||||||
@@ -105,21 +138,29 @@ The dotfile vars follow the same split: `desktop_common_dotfiles` carries mode-i
|
|||||||
- Windows applications are installed manually and are not managed from the WSL profile.
|
- Windows applications are installed manually and are not managed from the WSL profile.
|
||||||
|
|
||||||
## Rocky Server Notes
|
## Rocky Server Notes
|
||||||
- DuckDNS is rendered by `profile_server` from host-local `server_duckdns_domain` and
|
- DuckDNS support is removed from the server profile, not feature-gated. No updater tasks,
|
||||||
`vault_duckdns_token`. Keep the rotated token in encrypted Vault or untracked local vars, never in
|
templates or enablement variables remain. Prometheus uses its static IP and `fscotto.co`;
|
||||||
dotfiles. The private `~/duckdns/duck.sh` keeps the existing entrypoint; rendering uses `no_log`
|
the local updater, log and cron job were already retired. External DuckDNS account/name
|
||||||
and disables diffs. Provisioning does not execute the updater or change its external schedule.
|
and existing encrypted token are outside this removal and remain untouched.
|
||||||
- `rocky_server` is a child of both `platform_rocky` and `server`; `prometheus` is its active target.
|
- `rocky_server` is a child of both `platform_rocky` and `server`; `prometheus` is its active target.
|
||||||
- The target must already provide `server_username` with local sudo access before the profile runs.
|
- The target must already provide `server_username` with local sudo access before the profile runs.
|
||||||
- The Rocky profile installs Podman and podman-compose, uses firewalld, preserves SELinux enforcement, and renders the
|
- The Rocky profile installs Podman and podman-compose. Prometheus explicitly retires the legacy
|
||||||
existing Nginx Proxy Manager/Gitea Compose stack with a `podman-compose-server` systemd unit. PostgreSQL and
|
Compose unit, files and final-export helper with `server_legacy_stack_retired: true`.
|
||||||
Navidrome are no longer part of the desired Prometheus configuration. The role does not stop or remove legacy
|
Its approved opt-in cleanup removed old application data on 2026-10-03; normal runs do not
|
||||||
containers, delete `/opt/postgres/data`, start the Compose stack, update DNS, or cut over traffic.
|
delete data or recreate the retired files. On Prometheus, Nginx Proxy Manager is now the rootful
|
||||||
|
`prometheus-npm.service` Quadlet with a pinned image digest and the existing `/opt/npm/data` and
|
||||||
|
`/opt/npm/letsencrypt` bind mounts. The rootful `server_web` bridge remains `10.89.0.0/24`.
|
||||||
|
Gitea runs on Atlas; PostgreSQL and Navidrome are absent from the desired Prometheus stack.
|
||||||
|
Normal runs do not delete legacy data, update DNS, or perform an implicit cutover;
|
||||||
|
destructive cleanup requires its explicit tag and opt-in extra-var.
|
||||||
- Firewalld enables SSH, Cockpit (`9090/tcp`), HTTP and HTTPS. Nginx Proxy Manager publishes `80/tcp` and
|
- Firewalld enables SSH, Cockpit (`9090/tcp`), HTTP and HTTPS. Nginx Proxy Manager publishes `80/tcp` and
|
||||||
`443/tcp`; bind its administration interface only to `127.0.0.1:81` and use `npm-tunnel` from Ikaros or Nymph.
|
`443/tcp`; bind its administration interface only to `127.0.0.1:81` and use `npm-tunnel` from Ikaros or Nymph.
|
||||||
Nextcloud remains disabled; do not provision `/srv/nextcloud` directories.
|
Nextcloud remains disabled; do not provision `/srv/nextcloud` directories.
|
||||||
- `scripts/migrate_prometheus_data.sh` is the separate, source-host-run NPM/Gitea migration path. It dry-runs by
|
- The completed Ubuntu-to-Rocky data migration script and its operational instructions
|
||||||
default and requires explicit source-stack quiescing before copying persistent Docker data with rsync.
|
have been removed; current provisioning does not provide that one-time migration path.
|
||||||
|
- Completed Gitea owner-migration, migration-restore and final-export tasks, helpers and flags
|
||||||
|
are removed. Current Gitea marker/ownership checks, recurring backups and proxy configuration
|
||||||
|
remain intact. `server_gitea_proxy_enabled` controls ongoing proxy management only.
|
||||||
- Atlas-only OpenZFS, NFS, Samba, and Syncthing stay selected through Atlas host variables and must not
|
- Atlas-only OpenZFS, NFS, Samba, and Syncthing stay selected through Atlas host variables and must not
|
||||||
leak into `rocky_server`. Cockpit plus its Navigator and Podman extensions are selected explicitly for
|
leak into `rocky_server`. Cockpit plus its Navigator and Podman extensions are selected explicitly for
|
||||||
Prometheus through its host variables.
|
Prometheus through its host variables.
|
||||||
@@ -152,7 +193,9 @@ The dotfile vars follow the same split: `desktop_common_dotfiles` carries mode-i
|
|||||||
- `profile_backend_phase1` temporarily runs rootless Navidrome and Syncthing on Atlas until Uranus replaces
|
- `profile_backend_phase1` temporarily runs rootless Navidrome and Syncthing on Atlas until Uranus replaces
|
||||||
them. It binds only to Atlas' LAN IP, never `wg0`; Navidrome and the Syncthing GUI admit only Aegis as
|
them. It binds only to Atlas' LAN IP, never `wg0`; Navidrome and the Syncthing GUI admit only Aegis as
|
||||||
the source-NAT gateway, while native Syncthing ports admit the configured LAN. It initializes fresh
|
the source-NAT gateway, while native Syncthing ports admit the configured LAN. It initializes fresh
|
||||||
state only and never migrates or deletes source application data.
|
state only and never migrates or deletes source application data. The enabled rootless
|
||||||
|
`atlas-music-sync.timer` copies `/zpool/archive/Music` to `/zpool/media/music` daily at 00:45
|
||||||
|
Europe/Rome without deleting destination files; it requires both datasets to be mounted.
|
||||||
- `wireguard_overlay` manages `wg0` between Prometheus (`10.0.0.1`) and Aegis (`10.0.0.2`). It persists private
|
- `wireguard_overlay` manages `wg0` between Prometheus (`10.0.0.1`) and Aegis (`10.0.0.2`). It persists private
|
||||||
keys only on their respective hosts, exchanges only derived public keys through Ansible, and verifies a real peer
|
keys only on their respective hosts, exchanges only derived public keys through Ansible, and verifies a real peer
|
||||||
handshake. Prometheus opens `51820/udp`; Aegis is the LAN gateway. Its persistent IPv4 forwarding, narrowly scoped
|
handshake. Prometheus opens `51820/udp`; Aegis is the LAN gateway. Its persistent IPv4 forwarding, narrowly scoped
|
||||||
@@ -163,37 +206,316 @@ The dotfile vars follow the same split: `desktop_common_dotfiles` carries mode-i
|
|||||||
|
|
||||||
## Atlas NAS TODO
|
## Atlas NAS TODO
|
||||||
Completed validation: the existing RAIDZ2 pool and datasets, SELinux, LAN firewall, SSH, Cockpit with
|
Completed validation: the existing RAIDZ2 pool and datasets, SELinux, LAN firewall, SSH, Cockpit with
|
||||||
the selected 45Drives plugins, encrypted SMB3 `Archive`, the Aegis-only NFSv4 `photobook` export, and
|
the selected 45Drives plugins, encrypted SMB3 `Archive`, the Aegis-only NFSv4 `photobook` export, and the
|
||||||
the former Prometheus--Atlas WireGuard path were operational. Aegis has validated NFSv4.2 read, write, delete,
|
Prometheus--Aegis WireGuard gateway are operational. The gateway handshake, forwarding, source masquerading,
|
||||||
and `all_squash` mapping to UID/GID `1100` end-to-end.
|
and TCP reachability to Atlas were verified. Temporary Navidrome and Syncthing are available through their
|
||||||
- Validate the Prometheus--Aegis WireGuard gateway after migration: peer handshake and counters, Aegis IPv4
|
manual NPM Proxy Hosts; Syncthing uses `/data/Org` backed by the SMB-shared Archive dataset. Aegis has also
|
||||||
forwarding and masquerading, and an NPM request from Prometheus to an Atlas LAN address. Add the Uranus VIP to
|
validated NFSv4.2 read, write, delete, and `all_squash` mapping to UID/GID `1100` end-to-end. The ZFS
|
||||||
Prometheus' Aegis peer when the cluster control plane is assigned.
|
snapshot timers are active; a recursive hourly snapshot and scheduled retention prune completed
|
||||||
- Validate temporary Atlas Navidrome and Syncthing through Aegis before creating their NPM Proxy Hosts.
|
successfully. The first monthly scrub completed successfully on 2026-10-04;
|
||||||
Keep NPM host configuration manual; plan their eventual Uranus migration with storage and routing declared
|
the actual service result and pool scan were independently verified.
|
||||||
separately from the NAS baseline.
|
|
||||||
- Decide whether a common SMB/NFS namespace is required. `Archive` (SMB) and `photobook` (NFS) are
|
### Priority 1 - Data protection
|
||||||
intentionally distinct today; only if a shared namespace is selected, finalize its UID/GID, group,
|
- [x] Deploy Ansible-managed recursive ZFS snapshots with 24 hourly, 30 daily, 8 weekly, and 12 monthly
|
||||||
and POSIX ACL model and test the same files through both protocols.
|
generations, plus a monthly scrub on the first Sunday at 03:00. The timers and first hourly snapshot were
|
||||||
- Keep `atlas_manage_media_stack` disabled until the future Immich deployment has validated `/dev/dri`,
|
verified on Atlas. Cockpit Scheduler is for visibility or manual operations only, and snapshot
|
||||||
container paths, and the required Vault database secret.
|
rollback is never automated.
|
||||||
- Add Ansible-managed ZFS snapshot retention and scrub timers. Use Cockpit Scheduler for visibility
|
- [x] Verify the first monthly ZFS scrub from its actual service result. On 2026-10-04
|
||||||
or manual operations, not as the only source of configuration, and never automate snapshot rollback.
|
it completed at 05:02 CEST after 2:02:04, repairing 0 B with zero errors;
|
||||||
- Add the Atlas-initiated least-privilege Prometheus backup pull: Prometheus exposes only prepared
|
the service exited successfully and the pool reported no known data errors.
|
||||||
|
- [x] Activate and validate the encrypted offsite Borg backup to the Hetzner Storage Box. Atlas uses the
|
||||||
|
dedicated SSH identity, pinned ED25519 host key, Vault-backed `repokey` encryption, and a locked
|
||||||
|
non-login `borg` account with no sudo or supplementary groups. The initial snapshot-consistent backup,
|
||||||
|
Borg repository check, and temporary-directory restore completed successfully; the restored `Archive`
|
||||||
|
tree matched the live data, and temporary snapshots and mounts were removed. The exported recovery key
|
||||||
|
was copied offline. Daily backup retries and logging, 30 daily, 8 weekly and 12 monthly archives,
|
||||||
|
compaction, and monthly repository checks are enabled. Runs report a ZFS-based estimated percentage.
|
||||||
|
On 2026-09-30 a successful incremental run also removed the stale 2026-09-29 snapshot and its own
|
||||||
|
temporary snapshot after exit; the earlier `RuntimeDirectory` cleanup failure is resolved.
|
||||||
|
- [x] Populate `/zpool/archive` with the currently available data so offsite and offline backup tests run
|
||||||
|
against a representative load.
|
||||||
|
- [x] Evaluate Borg against the populated pool. The 2026-09-29 archive took 1 h 32 min for 2.18 TB
|
||||||
|
original / 2.04 TB compressed data, with 13.49 GB deduplicated size; retention and compaction
|
||||||
|
succeeded. On 2026-09-30 a subsequent incremental archive completed in about 22 seconds with
|
||||||
|
successful cleanup. The monitor reported 37% Storage Box quota used. These are observed runs, not
|
||||||
|
a guarantee of future duration or compression ratio.
|
||||||
|
- [x] Add the UUID-bound offline USB backup with versioned rsync, locking, capacity checks, verification,
|
||||||
|
safe unmounting and a tested restore procedure; never trigger it for an arbitrary USB disk. The
|
||||||
|
LUKS/ext4 identities were read-only verified; the manual service and 45Drives Alerts reminder timer were
|
||||||
|
deployed on Atlas. Interactive LUKS unlock is part of the manual service; only the reminder is
|
||||||
|
scheduled for the first Saturday of each month at 10:00 Europe/Rome via the existing 45Drives
|
||||||
|
notifier. A manual test produced an Alerts notification, not an email. The first USB attempt failed
|
||||||
|
on a `security.selinux` xattr and was interrupted; the xattr filter is deployed and the temporary
|
||||||
|
recursive snapshot and open LUKS mapper were cleaned up. A later run reported checksum verification
|
||||||
|
and published the USB version, but failed while removing host-namespace ZFS snapshot mounts. Those
|
||||||
|
exact mounts and snapshots were cleaned up. An `ExecStopPost` helper now removes only the named
|
||||||
|
temporary snapshot after the backup process exits. A new full run checksum-verified and published a
|
||||||
|
USB version; the service ended successfully, the mapper closed, no temporary USB snapshot remained,
|
||||||
|
and the pool was healthy. On 2026-09-25 an independent, read-only USB restore test copied one file from
|
||||||
|
the published `atlas/latest` version into `/var/tmp` and matched contents, owner, mode, size, mtime and
|
||||||
|
POSIX ACL. The temporary copy and mount were removed, the mapper closed, and the pool remained healthy.
|
||||||
|
- [x] Test restores independently from a ZFS snapshot, Borg, and the offline USB backup before relying on
|
||||||
|
any backup path. The earlier Borg temporary-directory restore passed. On 2026-09-25 a separate,
|
||||||
|
read-only ZFS snapshot test restored one file to `/var/tmp`, confirmed matching contents, ownership,
|
||||||
|
mode, mtime and ACL, then removed its temporary copy and on-demand mount. This is a file-level smoke
|
||||||
|
test, not full dataset recovery. An independent USB file restore passed on 2026-09-25 with matching
|
||||||
|
content and metadata; the later scaled OS-rebuild rehearsal is documented under Priority 2.
|
||||||
|
- [x] Add monitoring and alerting for pool health, scrub/resilver, SMART data, temperatures, free space,
|
||||||
|
snapshot/local-backup growth, Hetzner Storage Box quota, and failed maintenance or backup timers.
|
||||||
|
The half-hourly Atlas health monitor and systemd final-failure hooks are deployed. A live probe
|
||||||
|
found no issues; the service and timer succeeded, and a labelled 45Drives Alerts test notification
|
||||||
|
was submitted. Alerts are deduplicated; email delivery is not claimed. The Storage Box quota probe
|
||||||
|
runs `df -m` over the dedicated pinned-key SSH identity and does not open the Borg repository.
|
||||||
|
The failed-job hook was corrected to pass the literal systemd unit name; its expansion was verified
|
||||||
|
on Atlas, but a new real failure notification has not been deliberately triggered.
|
||||||
|
|
||||||
|
### Priority 2 - NAS operability and recovery
|
||||||
|
- [x] Document and test disaster recovery in `docs/atlas-recovery.md`: the operator confirmed Vault
|
||||||
|
and Borg recovery material is available offline; provisional targets are RPO 24h/RTO 72h. On
|
||||||
|
2026-09-30 an isolated small Rocky VM was rebuilt with the Atlas Ansible roles, imported its
|
||||||
|
preserved RAIDZ2 pool without force/rewind, and restored a file from the preserved snapshot;
|
||||||
|
the second Ansible run was idempotent. Earlier independent production ZFS, USB, and Borg file
|
||||||
|
restore tests remain separate evidence. A production-size full restore, unclean import, and
|
||||||
|
measured 24h/72h compliance are not claimed.
|
||||||
|
- [x] Define a controlled Rocky kernel/OpenZFS update and reboot procedure in
|
||||||
|
`docs/atlas-updates.md`. The first real change-window execution is not yet
|
||||||
|
validated; the procedure never reboots automatically or upgrades pool features.
|
||||||
|
- [x] Add the Atlas-initiated least-privilege Prometheus backup pull: Prometheus exposes only prepared
|
||||||
read-only dumps through a dedicated account and Atlas retains the private SSH key, pinned host key,
|
read-only dumps through a dedicated account and Atlas retains the private SSH key, pinned host key,
|
||||||
atomic pull, verification, retention and systemd service/timer.
|
atomic pull, verification, retention and systemd service/timer. The dedicated key/account and unit
|
||||||
- Add the encrypted offsite backup with Borg to a Hetzner Storage Box: use a dedicated SSH identity,
|
files are deployed; live read-only SSH, shell denial, and write denial were verified. On 2026-09-30
|
||||||
pin the host key, keep Borg repository credentials and encryption material in Vault, use
|
a manual export, Atlas pull, checksum verification, and temporary restore passed; both SQLite
|
||||||
snapshot-consistent sources, and manage retries, logging, pruning, repository checks and restores.
|
databases passed integrity checks and a restored Git repository passed `git fsck`. Both daily
|
||||||
- Add the UUID-bound offline USB backup with versioned rsync, locking, capacity checks, verification,
|
timers are enabled for 02:00/03:00 Europe/Rome. On 2026-10-01 their first scheduled export and
|
||||||
safe unmounting and a tested restore procedure; never trigger it for an arbitrary USB disk.
|
pull succeeded: Atlas verified the payload checksum and published `20261001T000001Z` as `latest`.
|
||||||
- Add monitoring and alerting for pool health, scrub/resilver, SMART data, temperatures, free space and
|
- [x] Decide whether a common SMB/NFS namespace is required: no. `Archive` (SMB) and `photobook` (NFS)
|
||||||
failed backup timers, plus a controlled Rocky kernel/OpenZFS update and reboot procedure.
|
remain intentionally distinct; `docs/atlas-sharing-decision.md` records the decision. No ACL or export
|
||||||
- Document and test disaster recovery: rebuild Atlas with Ansible, import the existing pool, restore
|
change is authorized by this decision.
|
||||||
from snapshot/USB/Hetzner, preserve Vault and Borg recovery material offline, and define RPO/RTO.
|
|
||||||
- Optionally design iCloud photo ingestion and an Aegis persistent NFS mount as a separate workflow
|
### Priority 3 - Service expansion
|
||||||
after the storage and backup layers are validated; do not make either a dependency of the Atlas
|
- [x] Populate `/zpool/media/music` and validate Navidrome. On 2026-09-30, 21,158 files
|
||||||
baseline.
|
(93,937,810,350 regular-file bytes) were copied from `/zpool/archive/Music` using a temporary
|
||||||
|
ZFS snapshot; a checksum-based rsync dry run found no differences or extra files. Navidrome saw
|
||||||
|
all files through its read-only mount, completed a scan, indexed 18,168 tracks, and responded
|
||||||
|
over HTTP. Some imported playlists still reference obsolete Windows paths. The source was left
|
||||||
|
intact and the temporary snapshot was removed.
|
||||||
|
- [x] Schedule a daily, non-deleting copy from `Archive/Music` to the separate Navidrome music
|
||||||
|
dataset. The rootless `atlas-music-sync.timer` is enabled for 00:45 Europe/Rome; a manual
|
||||||
|
idempotent service run succeeded on 2026-10-01. The first scheduled run triggered at
|
||||||
|
00:45 CEST on 2026-10-02 and exited successfully (`Result=success`, status 0); the next
|
||||||
|
run is scheduled for 2026-10-03 00:45 CEST.
|
||||||
|
- [x] Design the staged Prometheus-to-Atlas Gitea migration in `docs/atlas-gitea-migration.md`.
|
||||||
|
The approved topology keeps NPM on Prometheus and moves HTTPS and public SSH (TCP/2222) together;
|
||||||
|
Gitea runs as an `admin`-owned rootless user Quadlet on Atlas with an internal `gitea` user.
|
||||||
|
The rootful-to-rootless data-layout
|
||||||
|
conversion passed an isolated restore rehearsal. The later partial cutover is tracked below.
|
||||||
|
- [x] Prepare the dedicated Atlas Gitea dataset, non-login UID/GID 1101 with a separate rootless Podman
|
||||||
|
sub-ID range, and disabled user Quadlet. On 2026-10-01 the targeted Ansible run and a second idempotent
|
||||||
|
run passed; the generated unit was inactive, with no staging HTTP/SSH listener. POSIX ACLs on only the
|
||||||
|
service-namespace parents grant this account traversal without access to sibling datasets.
|
||||||
|
- [x] Perform an isolated rootless restore rehearsal from the verified Prometheus backup. On 2026-10-01
|
||||||
|
the SHA-256-checked selective extraction and path/SSH conversion succeeded; SQLite `quick_check`
|
||||||
|
passed, all 33 repositories passed `git fsck`, and source/target public SSH host-key fingerprints
|
||||||
|
matched. The pinned rootless image answered HTTP and listened on internal SSH/2222 with
|
||||||
|
`--network none`; the temporary container was removed and the Quadlet stayed inactive. A second
|
||||||
|
restore run made no changes. This is a rehearsal copy, not the final consistent cutover copy.
|
||||||
|
- [x] Verify ZFS and Borg coverage of the staged Gitea dataset. On 2026-10-01 the managed recursive
|
||||||
|
hourly snapshot `atlas-auto-hourly-20261001T193401Z` included it, and the managed incremental
|
||||||
|
Borg archive `atlas-20261001T193420Z` included its database. A private one-file restore from
|
||||||
|
each independently matched the staged database and passed SQLite `quick_check`; temporary files
|
||||||
|
and snapshot mounts were removed, the Borg service ended successfully, and the pool was healthy.
|
||||||
|
- [x] Include the new Gitea dataset in a UUID-bound offline USB version and test a file restore
|
||||||
|
before accepting production writes. The operator's 2026-10-01 manual run published version
|
||||||
|
`20261001T201220Z-254397` successfully on 2026-10-02. Its Gitea database was restored to a
|
||||||
|
temporary directory from a read-only mount: contents, owner, group, mode, size, mtime and POSIX
|
||||||
|
ACL matched, and SQLite `quick_check` passed. Temporary files and mounts were removed, LUKS
|
||||||
|
was closed, and the pool remained healthy. A redundant run was stopped during verification;
|
||||||
|
its temporary snapshot was cleaned up and the service's resulting failed state was reset.
|
||||||
|
- [x] Install a separate opt-in final Gitea export helper on Prometheus. Its 2026-10-01 targeted
|
||||||
|
deployment and `bash -n` passed while Gitea and NPM stayed running. It refuses an active export
|
||||||
|
timer, stops only Gitea, verifies SQLite, publishes a checksum-verified Gitea-only version for
|
||||||
|
Atlas' existing pull, and leaves the source stopped on success. It was invoked on 2026-10-02
|
||||||
|
after the export timer was stopped; version `20261002T071525Z` was pulled and verified on Atlas.
|
||||||
|
- [x] Prepare the Atlas final-restore gate without replacing the rehearsal: it accepts only a
|
||||||
|
checksum-verified `gitea-cutover` export, refuses a running target, stages and validates the new
|
||||||
|
layout before replacing the marked rehearsal, and rolls back a failed swap. Synthetic success
|
||||||
|
and rollback tests passed on 2026-10-01. On 2026-10-02 the final gate replaced the rehearsal;
|
||||||
|
SQLite `quick_check`, all 33 repository `git fsck` checks, checksum and SSH host-key comparison passed.
|
||||||
|
- [x] Start the rootless Atlas Gitea Quadlet and move the primary HTTPS route. On 2026-10-02 Atlas
|
||||||
|
answered HTTP 200 through the Aegis gateway. NPM stayed on Prometheus; its variable upstream
|
||||||
|
required a managed Nginx `server_proxy.conf` override because runtime DNS ignores Compose
|
||||||
|
`extra_hosts`. The primary public HTTPS page and API returned 200, and `git ls-remote` succeeded
|
||||||
|
for a representative repository after NPM restart; the Navidrome and Syncthing Proxy Hosts also
|
||||||
|
responded. The source
|
||||||
|
Gitea container was removed from the desired Compose stack without deleting its data; the
|
||||||
|
Prometheus backup export timer resumed for NPM only. A post-cutover recursive ZFS snapshot and
|
||||||
|
encrypted Borg archive `atlas-20261002T073044Z` completed successfully.
|
||||||
|
- [x] Move the live Gitea Quadlet and dataset from the legacy host `gitea` account to `admin`
|
||||||
|
after a disposable snapshot-copy test of the pinned derived image. On 2026-10-02 the explicit
|
||||||
|
outage run stopped only legacy Gitea, made safety snapshot
|
||||||
|
`zpool/services/data/gitea@gitea-owner-migration-20261002T100104`, changed dataset ownership,
|
||||||
|
and validated loopback staging (HTTP 200, internal `gitea` UID/GID 1000, SQLite `quick_check`)
|
||||||
|
before promoting the `admin` Quadlet. Production LAN and public HTTPS returned 200; Navidrome
|
||||||
|
and Syncthing remained active, the pool was healthy, and the normal Gitea run changed nothing.
|
||||||
|
The old host account and data on Prometheus remain preserved; the old Atlas Quadlet and its
|
||||||
|
parent-dataset traverse ACL were removed. A subsequent normal run changed nothing.
|
||||||
|
- [x] Validate public Gitea SSH/2222 and an authenticated read from Ikaros. After the VPS
|
||||||
|
firewall was opened on 2026-10-02, TCP/2222 connected, the public ED25519 host-key
|
||||||
|
fingerprint matched Atlas, Gitea authenticated `fscotto` using the `ikaros` key, and
|
||||||
|
`git ls-remote` returned HEAD for `fscotto/infra.git` over public SSH.
|
||||||
|
- [x] Validate authenticated SSH pull and push. On 2026-10-02 the operator reported both
|
||||||
|
operations working through the public SSH endpoint; the earlier agent-run `git ls-remote`
|
||||||
|
remains the independent read-only check. The agent did not perform a test push.
|
||||||
|
- [x] Validate Gitea login and write via HTTPS. On 2026-10-03 the operator confirmed
|
||||||
|
authenticated web login and Git clone/pull/push through the public HTTPS endpoint. Do not
|
||||||
|
restart the stale source Gitea after Atlas has accepted writes.
|
||||||
|
- [x] Design and deploy the empty temporary Atlas Nextcloud/ONLYOFFICE stack on 2026-10-03.
|
||||||
|
The operator explicitly authorized empty internal service startup before the first scrub;
|
||||||
|
this does not close the scrub or protection checks. Four rootless Quadlets, separate
|
||||||
|
component datasets, pinned images/apps, Vault secrets, standard fabio/chiara users, a
|
||||||
|
separate application admin and the Famiglia folder are deployed. Cron and internal Office
|
||||||
|
connection checks succeeded; repeat deployment changed nothing. See `docs/atlas-nextcloud.md`.
|
||||||
|
- [x] Complete the authorized empty-stack public cutover on 2026-10-03 after operator
|
||||||
|
DNS/NPM configuration. Both hostnames passed TLS and HTTPS redirects; authenticated
|
||||||
|
web login, WebDAV, private-file isolation, Famiglia cross-user create/read/update/delete and
|
||||||
|
CalDAV/CardDAV discovery passed. The Office connector and public health/API asset passed.
|
||||||
|
Temporary test files were removed; no iCloud data was imported.
|
||||||
|
- [x] Test a manual consistent Nextcloud backup and isolated restore on 2026-10-04.
|
||||||
|
Paused application writers and cron, copied app/config/custom apps/themes and files,
|
||||||
|
dumped PostgreSQL, restored database roles and verified authenticated DAV contents,
|
||||||
|
account recovery and Famiglia permissions. Test containers had no external network,
|
||||||
|
published ports or live data mounts; they and the temporary restore copy were removed.
|
||||||
|
See `docs/atlas-nextcloud-recovery-test.md`; this is not recurring Borg/USB recovery evidence.
|
||||||
|
- [x] Integrate consistent Nextcloud bundles with recurring Borg and operator-started USB
|
||||||
|
backups, two-version local retention, interruption recovery and failure monitoring.
|
||||||
|
On 2026-10-04 Borg archive `atlas-20261004T095255Z` succeeded; its new bundle was
|
||||||
|
extracted from Hetzner and restored in isolation with checksums, accounts, Famiglia
|
||||||
|
permissions and authenticated DAV verified. No production database was replaced.
|
||||||
|
- [ ] Validate a new USB version and restore its consistent Nextcloud bundle after
|
||||||
|
operator connection/unlock. Dependency installation alone is not restore evidence.
|
||||||
|
- [ ] Complete Nextcloud desktop/mobile editing and synchronization acceptance before family import.
|
||||||
|
iCloud migration and future Uranus transfer remain separate operations, not playbook flags.
|
||||||
|
- [x] Move Gitea canonical HTTPS and SSH hostname to `git.fscotto.co` on
|
||||||
|
2026-10-03 through Ansible. Only Gitea restarted; second run changed nothing.
|
||||||
|
HTTPS and authenticated SSH reads returned the same repository HEAD.
|
||||||
|
The new NPM hostnames passed TLS/HTTP checks; old DuckDNS Proxy Hosts were
|
||||||
|
observed disabled. Details are in `docs/domain-fscotto-co.md`.
|
||||||
|
- [x] Confirm login on the new Gitea hostname and update remaining client remotes/integrations.
|
||||||
|
The operator confirmed completion on 2026-10-03; the agent did not perform a test push.
|
||||||
|
- [x] Remove obsolete DuckDNS NPM Proxy Hosts, unused certificates and the old upstream override.
|
||||||
|
The operator confirmed completion on 2026-10-03; no new agent runtime check was performed.
|
||||||
|
- [x] Review and remove completed one-time procedures from the playbook.
|
||||||
|
The operator confirmed completion on 2026-10-03.
|
||||||
|
- [x] Retire Prometheus' local DuckDNS updater on 2026-10-03 through Ansible:
|
||||||
|
the five-minute cron entry and private updater/log directory were removed.
|
||||||
|
Provisioning support was subsequently removed entirely; repeat cleanup changed nothing. HTTPS services, private NPM
|
||||||
|
administration and the export timer stayed healthy. The external name and Vault token
|
||||||
|
remain untouched for possible future use on a local host.
|
||||||
|
- [ ] Keep `atlas_manage_media_stack` disabled until the future Immich deployment has validated `/dev/dri`,
|
||||||
|
container paths, and the required Vault database secret.
|
||||||
|
|
||||||
|
### Priority 4 - Optional workflows
|
||||||
|
- [x] Deploy the declared Atlas iCloudPD state dataset and inactive rootless `admin` Quadlet.
|
||||||
|
Photos belong under `/zpool/archive/Pictures/iCloudPD`; private config/MFA state belongs in
|
||||||
|
`zpool/services/data/icloudpd`. Photobook remains reserved for Immich. Ansible now renders
|
||||||
|
`icloudpd.conf` with the Apple ID from the existing Vault key, but does not store the password
|
||||||
|
or manage MFA. Automatic startup was approved on 2026-10-03; the Quadlet now
|
||||||
|
uses `WantedBy=default.target` and Ansible keeps the service running.
|
||||||
|
The isolated no-network layout test is documented in
|
||||||
|
`docs/atlas-icloudpd-migration.md`. On 2026-10-02 Atlas deployment and a second idempotent run
|
||||||
|
passed; no app config existed at deployment. A manual first start on 2026-10-02 generated
|
||||||
|
`icloudpd.conf`; an Ansible run then replaced it with a private mode-0600 Vault-backed template
|
||||||
|
and an idempotent second run. The image later expanded the config, so Ansible now seeds it
|
||||||
|
only when absent and maintains the declared fields. Its launcher requires `traceroute`; the
|
||||||
|
rootless Quadlet grants only `NET_RAW`, tested in isolation and after restart. The service
|
||||||
|
was subsequently initialized interactively; initial ingestion is tracked below.
|
||||||
|
- [x] Retire Aegis iCloudPD completely. The operator authorized deleting its Quadlet,
|
||||||
|
`/var/lib/icloudpd` data, and MFA state despite an unaudited container overlay. After two
|
||||||
|
interactive-sudo runs on 2026-10-02, the unit is `not-found`/`inactive`, the Quadlet and state
|
||||||
|
directory are absent, and AdGuard remains active. The temporary retirement tasks have since
|
||||||
|
been removed from the Aegis role; it no longer manages iCloudPD.
|
||||||
|
- [x] Validate Atlas iCloudPD authentication and initial ingestion. On 2026-10-03 the active
|
||||||
|
rootless service logged `All photos and videos have been downloaded` at 02:16 and reported
|
||||||
|
completion for the user. The destination held 11,658 files (86,020,430,015 bytes); the preceding 24h
|
||||||
|
logs showed download activity without authentication failures or errors. A later read-only check
|
||||||
|
found the service still active. This confirms the initial download, not the next daily cycle.
|
||||||
|
- [x] Declare HEIC decoding for Fedora graphical desktops without converting the originals on Atlas.
|
||||||
|
The Fedora role installs RPM Fusion Free with a pinned signing-key fingerprint and
|
||||||
|
`libheif-freeworld` on Ikaros and Nymph. The package was confirmed installed on Ikaros on
|
||||||
|
2026-10-03; Nymph deployment and an actual image-opening test were not observed.
|
||||||
|
- [ ] Validate Atlas iCloudPD filesystem/SELinux/SMB access, the next daily sync, ZFS/Borg/USB
|
||||||
|
backup inclusion, and isolated restore of photos and private state. A recursive hourly snapshot
|
||||||
|
of `zpool/archive` exists after ingestion, but no iCloudPD-specific backup version or restore
|
||||||
|
has been verified. The first monthly scrub remains a separate open data-protection check.
|
||||||
|
|
||||||
|
## Prometheus NPM Quadlet cutover
|
||||||
|
- [x] Stage a rootful NPM Quadlet using the exact running image and the existing data/certificate
|
||||||
|
mounts, bridge subnet, public HTTP/HTTPS ports, and loopback-only administration port.
|
||||||
|
The generated service depends on `server-web-network.service` and is wanted by `multi-user.target`.
|
||||||
|
- [x] Take and verify the stopped-source export before switching owners. Version
|
||||||
|
`20261003T091009Z` was pulled to Atlas and its NPM SQLite database checked in isolation.
|
||||||
|
- [x] Cut over NPM to `prometheus-npm.service` on 2026-10-03. The legacy Compose unit is inactive
|
||||||
|
and disabled; the Quadlet is active with zero recorded restarts. Public Gitea and Syncthing
|
||||||
|
HTTPS returned 200 with valid TLS, while public TCP/81 remained unreachable.
|
||||||
|
- [x] Validate the post-cutover backup path. The export and Atlas pull published
|
||||||
|
`20261003T091633Z`; checksum, SQLite `quick_check`, ten proxy hosts, six certificate records,
|
||||||
|
both Quadlet files were present, and the complete Let's Encrypt tree (70 regular files plus
|
||||||
|
12 symlinks) matched the live data. A targeted normal Ansible run changed nothing. Details and rollback
|
||||||
|
boundaries are in `docs/prometheus-npm-quadlet.md`.
|
||||||
|
- [x] Remove only unused Gitea, Navidrome and PostgreSQL images with opt-in
|
||||||
|
Ansible tasks on 2026-10-03. Second run changed nothing; NPM stayed active
|
||||||
|
with zero restarts, HTTP/HTTPS passed, backup timer and SSH proxy stayed active.
|
||||||
|
This image-only step preserved data and fallback; the later approved deletion is tracked below. Validation:
|
||||||
|
`ansible-playbook ansible/site.yml --limit prometheus --tags server_image_cleanup --check --diff -e server_legacy_image_cleanup=true`
|
||||||
|
- [x] Complete explicitly approved old-data and Compose fallback removal on 2026-10-03.
|
||||||
|
Backup paths and mount dependencies were reconciled before deletion; repeat cleanup changed
|
||||||
|
nothing. The normal Compose/template/helper check did not recreate retired files.
|
||||||
|
A separately approved manual export/pull published `20261003T112906Z`; checksum and isolated
|
||||||
|
SQLite restore passed with ten proxy hosts and both Quadlet definitions. NPM, primary HTTPS,
|
||||||
|
WireGuard, SSH proxy and backup timer remained healthy; existing backup archives were preserved.
|
||||||
|
- [x] Retire the unused secondary Gitea hostname `git.ov-ad3410.infomaniak.ch`
|
||||||
|
on 2026-10-03. Its NPM Proxy Host was already soft-deleted and had no
|
||||||
|
associated certificate. Its Ansible domain and runtime override were removed;
|
||||||
|
nginx -t and reload passed without restarting NPM. Primary HTTPS returned 200
|
||||||
|
with valid TLS. At that step only `git.fscotto.duckdns.org` remained declared;
|
||||||
|
the subsequent domain transition and operator-confirmed cleanup are tracked above.
|
||||||
|
- [ ] Observe the first scheduled export and Atlas pull after the cutover; the manual end-to-end
|
||||||
|
cycle passed, but the next unattended cycle has not yet occurred.
|
||||||
|
|
||||||
|
## Cerberus Management Node (Deferred)
|
||||||
|
`cerberus` is postponed until the office in the new house is physically set up. It is not an inventory
|
||||||
|
host and this section is a design and implementation backlog, not authorization to provision it early.
|
||||||
|
|
||||||
|
The planned node is a Lenovo ThinkCentre M700 Tiny with an Intel Core i3-6100T, 8 GB RAM, a 256 GB SSD,
|
||||||
|
and native 1 Gbps Ethernet. It will connect to a multi-input KVM switch using a passive DisplayPort-to-HDMI
|
||||||
|
cable, sharing the monitor and peripherals with Ikaros. Fedora Sericea (immutable Fedora with the Sway
|
||||||
|
Wayland compositor) is the intended OS. Cerberus is an isolated management plane: a dedicated Toolbox
|
||||||
|
environment will run Ansible for future `uranus` cluster provisioning. Rootless Podman will host Grafana,
|
||||||
|
Prometheus, and Loki. The 256 GB local SSD is the hot tier retaining metrics and logs for 30 days; scheduled,
|
||||||
|
validated exports of older historical data will use a dedicated Atlas NFS dataset as cold storage.
|
||||||
|
|
||||||
|
### Implementation plan
|
||||||
|
- [ ] Confirm the office, KVM switch, passive DisplayPort-to-HDMI path, shared monitor/peripherals, and native
|
||||||
|
1 Gbps Ethernet are physically operational before adding Cerberus to inventory.
|
||||||
|
- [ ] Install and update Fedora Sericea with Sway; document the immutable-host lifecycle and keep host changes
|
||||||
|
declarative rather than treating the base OS as a mutable workstation.
|
||||||
|
- [ ] Model Cerberus as its own host with independent platform, role, desktop, network, and storage inputs;
|
||||||
|
do not repurpose Ikaros variables or make it a Uranus cluster member.
|
||||||
|
- [ ] Provision an isolated Toolbox-based Ansible controller with the required collections and a reproducible
|
||||||
|
project checkout; define its least-privilege SSH access, known-host handling, and Vault workflow without
|
||||||
|
storing secrets in the image or repository.
|
||||||
|
- [ ] Define the explicit Uranus provisioning workflow from Cerberus, including inventory boundaries,
|
||||||
|
validation-only runs, and separate approval for any destructive cluster operation.
|
||||||
|
- [ ] Design rootless Podman/Quadlet services for Grafana, Prometheus, and Loki, including persistent local
|
||||||
|
state, service ownership, LAN exposure/authentication, resource limits, updates, and backups.
|
||||||
|
- [ ] Size and enforce a 30-day local hot-retention policy for metrics and logs on the 256 GB SSD; validate
|
||||||
|
actual disk growth and alert before capacity exhaustion.
|
||||||
|
- [ ] Create and validate a dedicated Atlas NFS cold-storage dataset and least-privilege export for Cerberus;
|
||||||
|
do not use a broad existing share or couple it to unrelated Atlas application state.
|
||||||
|
- [ ] Implement scheduled, idempotent exports of data older than 30 days to the Atlas NFS cold tier, with
|
||||||
|
locking, capacity checks, integrity verification, retention rules, failure monitoring, and a tested restore.
|
||||||
|
- [ ] Validate management-plane recovery: rebuild Cerberus, restore observability history from Atlas, and
|
||||||
|
confirm that Uranus provisioning can resume without depending on unreproducible local state.
|
||||||
|
|
||||||
## Coding Agent Notes
|
## Coding Agent Notes
|
||||||
- Shared agent definitions and lifecycle flags live in `ai_agents` in `ansible/inventory/group_vars/all.yml`.
|
- Shared agent definitions and lifecycle flags live in `ai_agents` in `ansible/inventory/group_vars/all.yml`.
|
||||||
@@ -234,5 +556,5 @@ and `all_squash` mapping to UID/GID `1100` end-to-end.
|
|||||||
`/etc/resolv.conf` linked to `/run/systemd/resolve/resolv.conf`. LAN clients may use AdGuard, but
|
`/etc/resolv.conf` linked to `/run/systemd/resolve/resolv.conf`. LAN clients may use AdGuard, but
|
||||||
Aegis must use the independent upstream DNS declared by `aegis_host_dns_servers` so Greenboot does
|
Aegis must use the independent upstream DNS declared by `aegis_host_dns_servers` so Greenboot does
|
||||||
not depend on the AdGuard container during startup.
|
not depend on the AdGuard container during startup.
|
||||||
- iCloudPD requires post-deployment interactive MFA initialization; its cookie/configuration state is
|
- Aegis iCloudPD has been retired and is no longer managed by this role. Its service, Quadlet,
|
||||||
persisted in `/var/lib/icloudpd/config`.
|
data, and MFA state were removed with the operator's explicit authorization.
|
||||||
|
|||||||
411
README.it.md
411
README.it.md
@@ -95,6 +95,27 @@ Nota sullo stato attuale del playbook principale:
|
|||||||
- `ansible/site.yml` applica il profilo server Rocky a `prometheus` con DNF, systemd, dotfiles server e firewalld
|
- `ansible/site.yml` applica il profilo server Rocky a `prometheus` con DNF, systemd, dotfiles server e firewalld
|
||||||
- `ansible/site.yml` applica il profilo NAS Rocky su `atlas` tramite SSH remoto
|
- `ansible/site.yml` applica il profilo NAS Rocky su `atlas` tramite SSH remoto
|
||||||
|
|
||||||
|
## Nodo pianificato e posticipato: Cerberus
|
||||||
|
|
||||||
|
`cerberus` e un nodo di management **posticipato**, in attesa dell'allestimento
|
||||||
|
fisico dell'ufficio nella nuova casa. Non e ancora presente nell'inventory e non
|
||||||
|
esistono ruoli o playbook che lo prendano come target.
|
||||||
|
|
||||||
|
L'hardware previsto e un Lenovo ThinkCentre M700 Tiny (Intel Core i3-6100T,
|
||||||
|
8 GB di RAM e SSD da 256 GB) con Ethernet nativa a 1 Gbps. Condividera monitor
|
||||||
|
e periferiche di Ikaros tramite uno switch KVM a ingressi multipli, usando un
|
||||||
|
cavo passivo DisplayPort-HDMI per il collegamento video. Il sistema operativo
|
||||||
|
previsto e Fedora Sericea, la variante Fedora immutabile con compositor Wayland
|
||||||
|
Sway.
|
||||||
|
|
||||||
|
Cerberus sara un management plane isolato: Ansible verra eseguito in un ambiente
|
||||||
|
Toolbox dedicato per il provisioning del futuro cluster `uranus`, anziche da
|
||||||
|
Ikaros o da un host non gestito. Lo stack di osservabilita rootless Podman
|
||||||
|
eseguira Grafana, Prometheus e Loki. L'SSD locale sara l'hot storage, con
|
||||||
|
metriche e log conservati per 30 giorni; esportazioni programmate trasferiranno
|
||||||
|
i dati storici piu vecchi su un dataset Atlas montato via NFS come cold storage.
|
||||||
|
Il piano di implementazione, con prerequisiti espliciti, e in `AGENTS.md`.
|
||||||
|
|
||||||
## Desktop
|
## Desktop
|
||||||
|
|
||||||
Target operativi:
|
Target operativi:
|
||||||
@@ -161,6 +182,10 @@ Le applicazioni Windows sono installate e gestite manualmente; il profilo WSL no
|
|||||||
|
|
||||||
## Server
|
## Server
|
||||||
|
|
||||||
|
La migrazione dei servizi pubblici a `fscotto.co`, la gestione Ansible
|
||||||
|
degli URL Gitea e i passaggi ancora aperti per ritirare DuckDNS sono in
|
||||||
|
[`docs/domain-fscotto-co.md`](docs/domain-fscotto-co.md).
|
||||||
|
|
||||||
Sistema operativo:
|
Sistema operativo:
|
||||||
|
|
||||||
- Rocky Linux 9
|
- Rocky Linux 9
|
||||||
@@ -180,51 +205,36 @@ Lo stato attuale del profilo server include:
|
|||||||
- installazione pacchetti Rocky via DNF, EPEL e CRB
|
- installazione pacchetti Rocky via DNF, EPEL e CRB
|
||||||
- installazione di Podman e podman-compose
|
- installazione di Podman e podman-compose
|
||||||
- abilitazione dei servizi systemd dichiarati in inventory/group vars
|
- abilitazione dei servizi systemd dichiarati in inventory/group vars
|
||||||
- copia dei dotfiles server e rendering del `docker-compose.yml` per Nginx Proxy Manager e Gitea,
|
- copia dei dotfiles server e rendering del Quadlet rootful `prometheus-npm.service` per Nginx Proxy
|
||||||
piu l'unita `podman-compose-server` (attivazione manuale)
|
Manager; il vecchio fallback Compose è stato rimosso con autorizzazione esplicita
|
||||||
- attivazione di firewalld con SSH, Cockpit (`9090/tcp`), HTTP e HTTPS abilitati
|
- attivazione di firewalld con SSH, Cockpit (`9090/tcp`), HTTP e HTTPS abilitati
|
||||||
- Syncthing escluso dal profilo server Rocky
|
- Syncthing escluso dal profilo server Rocky
|
||||||
|
|
||||||
Il Compose desiderato su Prometheus non include piu Navidrome ne il database PostgreSQL obsoleto.
|
Il 2026-10-03 la pulizia opt-in autorizzata ha rimosso dati e immagini precedenti di Gitea,
|
||||||
Navidrome e Syncthing appartengono ad Atlas; Navidrome ufficiale usa invece SQLite. Il profilo non
|
Navidrome e PostgreSQL, directory obsolete vuote, helper finale Gitea e fallback Compose NPM.
|
||||||
arresta o rimuove automaticamente eventuali container legacy e non elimina `/opt/postgres/data`.
|
I servizi migrati restano su Atlas. `server_legacy_stack_retired: true` evita che i normali task
|
||||||
|
ricreino i residui; la cancellazione richiede `--tags server_legacy_cleanup` e
|
||||||
|
`-e server_legacy_cleanup=true`. NPM attivo e archivi di backup restano intatti.
|
||||||
|
Export, pull Atlas e restore SQLite isolato post-pulizia sono riusciti; il primo ciclo automatico
|
||||||
|
resta da osservare. Evidenze e confini del recovery:
|
||||||
|
[`docs/prometheus-npm-quadlet.md`](docs/prometheus-npm-quadlet.md).
|
||||||
|
|
||||||
Nginx Proxy Manager pubblica solo `80/tcp` e `443/tcp`; la sua interfaccia di amministrazione e
|
Nginx Proxy Manager pubblica solo `80/tcp` e `443/tcp`; la sua interfaccia di amministrazione e
|
||||||
associata a `127.0.0.1:81` ed e raggiungibile da Ikaros o Nymph con l'alias Bash `npm-tunnel`.
|
associata a `127.0.0.1:81` ed e raggiungibile da Ikaros o Nymph con l'alias Bash `npm-tunnel`.
|
||||||
Nextcloud resta disabilitato e il profilo non crea directory `/srv/nextcloud`.
|
Nextcloud resta disabilitato e il profilo non crea directory `/srv/nextcloud`.
|
||||||
|
|
||||||
La fase 1 su Atlas non modifica questo deployment NPM ne i suoi dati persistenti. Dopo aver attivato
|
La fase 1 su Atlas non modifica i dati persistenti NPM. I proxy host NPM usano gli upstream LAN
|
||||||
WireGuard e i servizi Atlas, configurare i proxy host NPM correnti con upstream Navidrome
|
`http://192.168.178.55:4533` per Navidrome e `http://192.168.178.55:8384` per la GUI Syncthing;
|
||||||
`http://10.0.0.2:4533` e upstream per la GUI Syncthing `http://10.0.0.2:8384`. Solo la GUI web di
|
Prometheus li raggiunge attraverso Aegis come gateway WireGuard. Solo la GUI web di Syncthing usa
|
||||||
Syncthing usa NPM; il traffico di sincronizzazione resta sulle porte native pubblicate esplicitamente solo
|
NPM; il traffico di sincronizzazione resta sulle porte native esposte sulla LAN dichiarata.
|
||||||
sull'indirizzo WireGuard di Atlas. Configurare l'autenticazione Syncthing e una policy di accesso NPM adeguata prima di pubblicare la GUI.
|
Mantenere l'autenticazione Syncthing e una policy di accesso NPM adeguata.
|
||||||
|
|
||||||
### DuckDNS
|
### Rimozione DuckDNS
|
||||||
|
|
||||||
`profile_server` genera `~/duckdns/duck.sh` con permessi `0700`, mantenendo il percorso dello
|
Il supporto DuckDNS è stato rimosso dal profilo server: non restano task, template,
|
||||||
script e `duck.log`. Definire `server_duckdns_domain` negli host vars del server e salvare il
|
variabili o flag di abilitazione. Prometheus usa IP statico e `fscotto.co`.
|
||||||
**nuovo token rigenerato** in `vault_duckdns_token`, nel Vault cifrato `secrets/vault.yml`
|
Updater locale, log e cron erano già stati rimossi. Nome/account DuckDNS esterni
|
||||||
(`ansible-vault edit secrets/vault.yml`) oppure negli override non versionati `secrets/vault.local.yml`.
|
ed eventuale token cifrato esistente restano invariati per un possibile uso futuro.
|
||||||
Non committare lo script generato e non passare il token sulla riga di comando. Il rendering
|
|
||||||
nasconde output e diff sensibili; lo script verifica TLS e passa il token a curl tramite stdin.
|
|
||||||
Il playbook non esegue lo script e non modifica la sua schedulazione esterna.
|
|
||||||
|
|
||||||
```bash
|
|
||||||
ansible-playbook ansible/site.yml --limit prometheus --tags duckdns --check --diff
|
|
||||||
ansible-playbook ansible/site.yml --limit prometheus --tags duckdns
|
|
||||||
```
|
|
||||||
|
|
||||||
La cancellazione dalla cronologia non revoca il token: rigenerarlo sul pannello DuckDNS.
|
|
||||||
Dopo la bonifica, riclonare gli altri checkout senza unire nuovamente la vecchia storia;
|
|
||||||
salvare separatamente eventuali modifiche non committate senza copiare segreti.
|
|
||||||
|
|
||||||
### Migrazione dati
|
|
||||||
|
|
||||||
Dopo il provisioning Rocky, eseguire `scripts/migrate_prometheus_data.sh` **sul server Ubuntu
|
|
||||||
sorgente**. Lo script usa rsync, e in dry-run di default; richiede `--quiesce-source --execute` per
|
|
||||||
fermare lo stack sorgente e copiare in modo consistente soltanto i dati di Nginx Proxy Manager e
|
|
||||||
Gitea. Non sposta Navidrome o Syncthing, non avvia container, non cancella dati e non esegue il
|
|
||||||
cutover.
|
|
||||||
|
|
||||||
Utente del profilo server:
|
Utente del profilo server:
|
||||||
|
|
||||||
@@ -246,90 +256,282 @@ ansible-playbook ansible/site.yml --limit prometheus -e server_username=myuser -
|
|||||||
|
|
||||||
## NAS
|
## NAS
|
||||||
|
|
||||||
`atlas` e un NAS Rocky Linux 9 raggiunto tramite SSH. Normalmente il pool ZFS esiste gia e il profilo
|
`atlas` è un NAS Rocky Linux 9 raggiunto via SSH. Normalmente il pool esiste già e il profilo gestisce
|
||||||
gestisce solo i dataset figli. Un bootstrap RAIDZ2 una tantum e disponibile solo con conferma esplicita
|
solo i dataset figli. La creazione iniziale del RAIDZ2 richiede esplicitamente `atlas_create_pool=true`
|
||||||
(`atlas_create_pool=true`) e quattro percorsi reali e verificati `/dev/disk/by-id/...` in
|
e quattro percorsi `/dev/disk/by-id/...` verificati in `atlas_zpool_disks`. Il ruolo non partiziona,
|
||||||
`atlas_zpool_disks`. Non partiziona, forza, distrugge, esegue rollback o modifica il layout vdev di un
|
forza, distrugge, ripristina né modifica il layout vdev di un pool esistente. I client Linux usano NFSv4,
|
||||||
pool esistente. I client Linux usano NFSv4, quelli Windows/WSL SMB; entrambi restano limitati alla LAN
|
quelli Windows/WSL SMB; l'accesso è limitato alla LAN configurata.
|
||||||
configurata.
|
|
||||||
|
|
||||||
Per il primo avvio fornire `vault_atlas_authorized_ssh_keys`, `vault_atlas_admin_password_hash`,
|
Per il primo avvio servono `vault_atlas_admin_password_hash`, `vault_atlas_samba_password` e
|
||||||
`vault_atlas_samba_password` e `vault_atlas_immich_db_password`. Eseguire il bootstrap tramite
|
`vault_atlas_immich_db_password`; il primo è un hash compatibile con `/etc/shadow`, non una password
|
||||||
l'amministratore esistente:
|
Cockpit in chiaro. Il bootstrap usa l'amministratore preesistente:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
ansible-playbook ansible/site.yml --limit atlas \
|
ansible-playbook ansible/site.yml --limit atlas \
|
||||||
-e atlas_connection_username=<existing-admin>
|
-e atlas_connection_username=<existing-admin>
|
||||||
```
|
```
|
||||||
|
|
||||||
`vault_atlas_admin_password_hash` deve essere un hash compatibile con `/etc/shadow`, non una
|
Le esecuzioni successive usano `atlas_admin_username`. Storage, condivisioni e firewall LAN sono
|
||||||
password Cockpit in chiaro. Le esecuzioni successive usano `atlas_admin_username`. Atlas dichiara
|
abilitati; prima dell'applicazione verificare pool, mountpoint, subnet e zona firewalld. La creazione
|
||||||
abilitati storage, condivisioni e regole firewall LAN. Prima della prima applicazione verificare pool e
|
del pool è protetta da un gate esplicito e avviene solo se è assente. Atlas non fa più parte della VPN
|
||||||
mountpoint esistenti, subnet LAN e zona firewalld attiva. `atlas_manage_media_stack` resta disabilitato
|
WireGuard: la vecchia interfaccia è stata ritirata manualmente dopo la verifica del collegamento tra
|
||||||
finche non saranno validati `/dev/dri`, i percorsi dei container e il segreto del database Immich.
|
Prometheus e Aegis. Le chiavi SSH autorizzate sono in file separati sotto
|
||||||
|
`~/.ssh/authorized_keys.d/`. `atlas_manage_media_stack` resta disabilitato finché `/dev/dri`, percorsi
|
||||||
|
dei container e segreto del database Immich non sono validati.
|
||||||
|
|
||||||
Con la gestione storage attiva, Atlas crea l'intera gerarchia sotto il pool `zpool` esistente o creato esplicitamente:
|
Sotto `zpool` Atlas crea `archive` (SMB), `services/data` con i dataset applicativi
|
||||||
`work`, `archive`, `archive/app_data`, i dataset applicativi separati
|
`services/data/navidrome` e `services/data/syncthing`, `media`, `media/music`, `media/photobook` e
|
||||||
`archive/app_data/navidrome` e `archive/app_data/syncthing`, `media`, `media/music`,
|
`backup/hosts/prometheus`. Archivio e applicazioni usano `zstd`; media, Syncthing e backup host usano
|
||||||
`media/photobook`, `backups`, `backups/services` e `backup_prometheus`. I dataset applicativi e
|
`lz4`. `backup` ha una riserva di `500G` che copre i discendenti. SELinux targeted è persistente;
|
||||||
di archivio usano `zstd`; media, Syncthing e backup dei servizi usano `lz4`;
|
l'eventuale riavvio necessario viene segnalato, non eseguito. Atlas assegna l'interfaccia primaria
|
||||||
`backups/services` mantiene inoltre una `refreservation` di `500G`.
|
alla zona firewalld gestita, rifiuta redirect e source route, registra i martian, mantiene il reverse-path
|
||||||
Atlas impone SELinux targeted in modo persistente e segnala, senza avviarlo, l’eventuale reboot necessario per attivarlo. Assegna esplicitamente l’interfaccia LAN primaria alla zona firewalld gestita e applica hardening persistente del kernel di rete: rifiuta redirect e source-route, registra i martian, usa reverse-path filtering loose per WireGuard e disabilita il forwarding IPv4. SSH consente solo l’amministratore dichiarato tramite chiave pubblica; root, password, agent e forwarding
|
filter loose e disabilita il forwarding IPv4. SSH consente soltanto l'amministratore dichiarato con
|
||||||
remoto sono disabilitati, mentre il forwarding locale resta disponibile per tunnel amministrativi privati. SMB3 pubblica `Archive` solo agli account Samba configurati con password in Vault e
|
chiave pubblica: root, password, agent forwarding e remote forwarding sono disabilitati, mentre il
|
||||||
ammette la LAN configurata su SMB3 cifrato e firmato, esclusivamente su TCP/445. NFSv4 esporta soltanto
|
forwarding locale resta disponibile per i tunnel amministrativi. SMB3 espone `Archive` agli account
|
||||||
`media/photobook` all'IP configurato di Aegis su TCP/2049, con `all_squash` verso UID/GID anonimi `1100`.
|
autorizzati da Vault sulla LAN, solo su TCP/445 con cifratura e firma obbligatorie. NFSv4 espone
|
||||||
|
soltanto `media/photobook` all'IP di Aegis su TCP/2049, con `all_squash` verso UID/GID `1100`.
|
||||||
|
|
||||||
L'account di sistema `immich` usa UID/GID `1100`, shell senza login, nessuna appartenenza a `wheel` e
|
L'account di sistema `immich` usa UID/GID `1100`, non ha shell di login né gruppo `wheel` e riceve i
|
||||||
i gruppi supplementari `video` e `render`. I Quadlet rootful di Immich Server, ML, cache compatibile
|
gruppi `video` e `render`. Lo stack Immich futuro prevede Quadlet rootful per Server, ML, cache,
|
||||||
Redis, PostgreSQL e NPM condividono una rete Podman. Immich viene eseguito come `1100:1100`; Server e
|
PostgreSQL e NPM su una rete Podman comune. Immich gira come `1100:1100`, Server e ML ricevono
|
||||||
ML ricevono `/dev/dri` e Photobook e montato in sola lettura su `/external/photobook`. NPM pubblica `80` e
|
`/dev/dri` e Photobook è montato in sola lettura su `/external/photobook`. NPM pubblica `80` e `443`;
|
||||||
`443`, mentre l'amministrazione resta vincolata a `127.0.0.1:81` per l'accesso tramite tunnel SSH.
|
l'interfaccia amministrativa resta su `127.0.0.1:81`, raggiungibile via tunnel SSH.
|
||||||
|
|
||||||
La fase 1 e limitata ai Quadlet utente rootless di Navidrome e Syncthing su Atlas. E abilitata nella
|
Atlas ospita temporaneamente Navidrome e Syncthing rootless fino alla sostituzione con Uranus. I
|
||||||
configurazione host di Atlas e puo essere impostata a `false` solo per una sospensione intenzionale. Navidrome ufficiale `0.63.2` usa il database SQLite sotto `/data` e
|
servizi sono inizializzati **ex novo**, senza migrare lo stato precedente, rispettivamente sotto
|
||||||
non supporta `ND_DATABASE_URL` ne un backend PostgreSQL esterno. Il servizio obsoleto `navidromedb`
|
`/zpool/services/data/navidrome` e `/zpool/services/data/syncthing`; la musica in
|
||||||
e quindi rimosso da Prometheus invece di essere replicato su Atlas. Il ruolo deriva i percorsi dal
|
`/zpool/media/music` è stata popolata separatamente da `/zpool/archive/Music` il 2026-09-30;
|
||||||
pool `zpool`, montato in `/zpool`: musica in sola lettura da `/zpool/media/music`, stato
|
Navidrome ha completato la scansione. Il timer rootless `atlas-music-sync.timer` copia i file nuovi
|
||||||
applicativo Navidrome e `navidrome.db` in `/zpool/archive/app_data/navidrome` e dati Syncthing in
|
o modificati ogni giorno alle 00:45 Europe/Rome, senza eliminare quelli presenti solo nella
|
||||||
`/zpool/archive/app_data/syncthing`. `profile_atlas` crea questi dataset quando
|
destinazione; entrambi i dataset ZFS devono essere montati. La prima esecuzione schedulata è
|
||||||
`atlas_manage_storage` e attivo; il ruolo backend verifica i mountpoint esatti prima di avviare i
|
riuscita il 2026-10-02. Alcune playlist originali contengono
|
||||||
container. Il ruolo backend non crea mai il pool. Il ruolo separato `wireguard_overlay`
|
ancora vecchi percorsi Windows. I servizi sono vincolati all'indirizzo LAN di Atlas
|
||||||
gestisce `wg0` tra Prometheus (`10.0.0.1`) e Atlas (`10.0.0.2`), genera una sola volta le chiavi
|
(`192.168.178.55`), mai a WireGuard. `wireguard_overlay` collega invece Prometheus (`10.0.0.1`)
|
||||||
private sui rispettivi host e scambia tramite Ansible soltanto quelle pubbliche. Solo Prometheus apre
|
e Aegis (`10.0.0.2`): le chiavi private restano sui rispettivi host e Ansible scambia solo le pubbliche.
|
||||||
pubblicamente `51820/udp`. Le porte backend sono ammesse esclusivamente nella zona firewalld WireGuard.
|
Prometheus apre `51820/udp`; Aegis inoltra soltanto il traffico overlay→LAN dichiarato e applica
|
||||||
|
source NAT, evitando interfacce VPN su Atlas/Uranus e route statiche sul router. Navidrome (`4533/tcp`)
|
||||||
|
e la GUI Syncthing (`8384/tcp`) ammettono solo Aegis, mentre le porte native Syncthing sono limitate
|
||||||
|
alla LAN. Dopo la verifica dei servizi, configurare manualmente i Proxy Host NPM verso
|
||||||
|
`http://192.168.178.55:4533` e `http://192.168.178.55:8384`. Il peer Prometheus include la LAN
|
||||||
|
negli `AllowedIPs`; aggiungere la VIP Uranus quando esisterà. Dopo il reload di firewalld, Ansible
|
||||||
|
ricarica le reti Podman rootful di Prometheus per conservare DNS e connettività del proxy.
|
||||||
|
|
||||||
`backend_phase1_start_services` resta falso durante il trasferimento dello stato applicativo, quindi
|
La migrazione Gitea da Prometheus ad Atlas è descritta in
|
||||||
la prima esecuzione reale del backend genera i Quadlet senza creare un database Atlas vuoto. Dopo aver
|
[`docs/atlas-gitea-migration.md`](docs/atlas-gitea-migration.md). Gitea usa un Quadlet rootless
|
||||||
arrestato Navidrome su Prometheus, copiare l'intera directory `/opt/navidrome/data/` in
|
di `admin` su un dataset dedicato; l'immagine derivata mantiene UID/GID 1000 ma chiama l'utente
|
||||||
`/zpool/archive/app_data/navidrome/`, preservando `navidrome.db` e gli eventuali file SQLite laterali.
|
interno `gitea`. NPM resta su Prometheus e l'HTTPS pubblico primario serve Atlas. L'SSH pubblico
|
||||||
Impostare quindi questa variabile a vero e rieseguire il ruolo per abilitare e avviare Navidrome e
|
su TCP/2222 autentica la chiave `ikaros` e un `git ls-remote` è riuscito; l'operatore ha
|
||||||
Syncthing. Il playbook non copia e non elimina mai i dati applicativi.
|
confermato pull e push SSH. Login e scrittura Git via HTTPS sono stati confermati il 2026-10-03. I dati sorgente restano
|
||||||
|
conservati su Prometheus senza avviarne il vecchio container.
|
||||||
|
|
||||||
Validare e generare i servizi Atlas con:
|
Validare il gateway con:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
ANSIBLE_LOCAL_TEMP=/tmp/ansible-local \
|
ANSIBLE_LOCAL_TEMP=/tmp/ansible-local \
|
||||||
ansible-playbook ansible/site.yml --limit atlas --tags storage
|
ansible-playbook ansible/site.yml --limit prometheus,aegis --tags wireguard --check --diff
|
||||||
|
|
||||||
ANSIBLE_LOCAL_TEMP=/tmp/ansible-local \
|
|
||||||
ansible-playbook ansible/site.yml --limit prometheus,atlas --tags wireguard
|
|
||||||
|
|
||||||
ANSIBLE_LOCAL_TEMP=/tmp/ansible-local \
|
|
||||||
ansible-playbook ansible/site.yml --limit atlas --tags backend_phase1 --check --diff
|
|
||||||
|
|
||||||
ANSIBLE_LOCAL_TEMP=/tmp/ansible-local \
|
|
||||||
ansible-playbook ansible/site.yml --limit atlas --tags backend_phase1
|
|
||||||
```
|
```
|
||||||
|
|
||||||
Per il cutover, arrestare il vecchio Navidrome prima di copiare la sua directory dati, verificare
|
La prima esecuzione reale WireGuard deve includere entrambi i peer. Se Aegis ha appena installato il
|
||||||
l'ownership dell'account `admin` su Atlas e confermare la presenza del database SQLite copiato prima
|
layer `wireguard-tools`, riavviarlo manualmente e rieseguire senza `--check`: il ruolo attende un
|
||||||
di impostare `backend_phase1_start_services: true` in `host_vars/atlas.yml`. Conservare i dati sorgente
|
handshake effettivo.
|
||||||
e il container legacy `navidromedb` fermo finche Navidrome su Atlas e una prova di restore non sono
|
|
||||||
stati validati.
|
|
||||||
|
|
||||||
Restano da completare retention delle snapshot, topologia Syncthing, validazione WireGuard/firewall,
|
Gli snapshot ZFS ricorsivi coprono l'intero pool: 24 orari al minuto 05, 30 giornalieri alle 00:15,
|
||||||
pull di backup da Prometheus, backup cifrati con Borg su una Hetzner Storage Box, backup USB,
|
8 settimanali la domenica alle 01:00 e 12 mensili il primo giorno alle 02:00. La retention elimina
|
||||||
monitoraggio e test di disaster recovery. Il backlog operativo dettagliato e in `AGENTS.md`.
|
solo gli snapshot con prefisso gestito `atlas-auto` e non esegue rollback. Lo scrub OpenZFS mensile è
|
||||||
|
previsto la prima domenica alle 03:00; il timer settimanale incompatibile è disabilitato. Il primo
|
||||||
|
snapshot orario ricorsivo è riuscito e la pulizia pianificata della retention è stata osservata il
|
||||||
|
2026-09-30. Il primo scrub mensile richiede ancora una verifica a runtime.
|
||||||
|
|
||||||
|
```bash
|
||||||
|
ANSIBLE_LOCAL_TEMP=/tmp/ansible-local \
|
||||||
|
ansible-playbook ansible/site.yml --limit atlas --tags snapshots,scrub --check --diff
|
||||||
|
```
|
||||||
|
|
||||||
|
Il backup Borg cifrato usa il sub-account Hetzner `u660064-sub1`, il repository relativo `./borg-data`
|
||||||
|
e Borg remoto 1.4 su SSH porta 23. La chiave ED25519 del server è fissata; una chiave client dedicata
|
||||||
|
appartiene all'account `borg`, bloccato e senza login, sudo o gruppi supplementari. La chiave privata
|
||||||
|
resta in `/etc/atlas-borg`; la passphrase proviene da `vault_atlas_borg_passphrase` ed è resa in un
|
||||||
|
file `0600`. Solo il wrapper root crea snapshot e mount; avvia il client come `borg` con il minimo
|
||||||
|
accesso temporaneo in lettura, senza concedergli gestione ZFS o sudo.
|
||||||
|
|
||||||
|
Il backup giornaliero parte alle 04:30 con un ritardo casuale fino a 30 minuti. Crea uno snapshot ZFS
|
||||||
|
ricorsivo temporaneo e ricostruisce tutti i dataset sotto `/zpool` in un albero di bind mount in sola
|
||||||
|
lettura, per inserirli in un unico archivio coerente. Il wrapper smonta ricorsivamente l'albero privato;
|
||||||
|
un helper `ExecStopPost` mirato rimuove eventuali mount dello snapshot nel namespace host e lo snapshot
|
||||||
|
temporaneo dopo l'uscita del processo. Borg conserva 30 archivi giornalieri, 8 settimanali e 12
|
||||||
|
mensili, poi compatta il repository. Il controllo completo di metadati e repository si svolge il 15
|
||||||
|
di ogni mese alle 06:00. Le operazioni usano un lock comune, journal e retry systemd limitati. Le
|
||||||
|
nuove esecuzioni riportano al massimo una riga di avanzamento al minuto: percentuale **stimata**,
|
||||||
|
dataset, file elaborati e byte originali/compressi/deduplicati. Il denominatore è la somma dei
|
||||||
|
`logicalreferenced` ZFS dello snapshot, non un totale Borg: può superare il 100% e non comprende
|
||||||
|
retention, compattazione o controlli. Le righe di progresso non riportano i nomi dei file; eventuali
|
||||||
|
warning possono farlo. Seguire il job con `sudo journalctl -fu atlas-borg-backup.service`; modifiche
|
||||||
|
all'helper non cambiano un'esecuzione già avviata.
|
||||||
|
|
||||||
|
Attivazione iniziale esplicita:
|
||||||
|
|
||||||
|
1. Inserire una passphrase unica in `secrets/vault.yml` con `ansible-vault edit`.
|
||||||
|
2. Generare e mostrare solo la chiave pubblica con
|
||||||
|
`ansible-playbook ansible/site.yml --limit atlas --tags borg_key`.
|
||||||
|
3. Installarla nel sub-account Hetzner, poi applicare con
|
||||||
|
`ansible-playbook ansible/site.yml --limit atlas --tags packages,borg`.
|
||||||
|
4. Copiare `secrets/recovery/atlas-borg-repokey.export` su un supporto davvero offline: la copia
|
||||||
|
locale ignorata da Git non è di per sé un backup offline.
|
||||||
|
|
||||||
|
Il ruolo inizializza solo un repository `repokey` assente, non accetta password SSH né host key non
|
||||||
|
fissate e non avvia manualmente il primo backup. Validazione:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
ANSIBLE_LOCAL_TEMP=/tmp/ansible-local \
|
||||||
|
ansible-playbook ansible/site.yml --limit atlas --tags packages,borg --check --diff
|
||||||
|
```
|
||||||
|
|
||||||
|
L'attivazione iniziale è riuscita: backup e controllo del repository, restore completo in una
|
||||||
|
directory temporanea confrontato con l'albero `Archive`, esportazione offline della chiave di recupero
|
||||||
|
e pulizia di snapshot/mount temporanei. Il 2026-09-25 un test separato da snapshot ZFS giornaliero ha
|
||||||
|
copiato un file di `/zpool/archive` in `/var/tmp`, verificando contenuto, proprietario, modalità,
|
||||||
|
mtime e ACL POSIX; copia e mount temporanei sono stati rimossi senza interrompere Borg. Non è un test
|
||||||
|
di ripristino dell'intero dataset.
|
||||||
|
|
||||||
|
L'archivio del pool popolato del 2026-09-29 ha richiesto 1 h 32 min per 2,18 TB originali / 2,04 TB
|
||||||
|
compressi, con 13,49 GB di dimensione deduplicata. Retention e compattazione sono riuscite, ma un
|
||||||
|
errore di permessi su `RuntimeDirectory` ha impedito la pulizia dello snapshot dopo il job. Dopo la
|
||||||
|
correzione, l'archivio incrementale del 2026-09-30 è terminato in circa 22 secondi, ha rimosso lo
|
||||||
|
snapshot residuo e quello corrente ed è terminato con stato 0. Il monitor ha rilevato il 37% della
|
||||||
|
quota Storage Box utilizzata. Questi risultati non predicono durata o compressione dei prossimi run.
|
||||||
|
|
||||||
|
Il backup USB offline è distribuito come **servizio solo manuale** (`atlas_manage_usb_backup: true`):
|
||||||
|
Ansible non formatta, sblocca, monta né avvia automaticamente il disco. Il disco esistente è stato
|
||||||
|
verificato in sola lettura il 2026-09-23: UUID LUKS `577b3c43-ea37-4611-81a9-39d555cdfbd4`,
|
||||||
|
UUID ext4 interno `758e2d2e-a427-4797-aad9-39c3a9f17c7e`, mapper `zpool-backup`. All'ispezione
|
||||||
|
era montato in `/mnt/zpool-backup`; il servizio richiede invece che il mapper **non sia montato** prima
|
||||||
|
dell'avvio. Se serve, `systemd-ask-password` chiede interattivamente la passphrase LUKS tramite
|
||||||
|
l'agente di `systemctl start` e la passa direttamente a `cryptsetup`, senza salvarla, esporla negli
|
||||||
|
argomenti o memorizzarla nella cache. Lo script monta il disco privatamente, crea uno snapshot ZFS
|
||||||
|
ricorsivo, copia tutti i dataset in `atlas/snapshots/<timestamp>/` con `rsync --link-dest`, verifica
|
||||||
|
con un dry-run basato sui checksum, aggiorna atomicamente `atlas/latest`, smonta e chiude LUKS. Un
|
||||||
|
errore non sostituisce `latest` né cancella versioni complete precedenti. Borg e USB possono operare
|
||||||
|
contemporaneamente su snapshot distinti, ma la lettura concorrente può ridurre il throughput.
|
||||||
|
|
||||||
|
La copia USB conserva le ACL ma non gli attributi estesi generici, compreso `security.selinux`: la
|
||||||
|
policy della destinazione deve ricreare le etichette dopo un restore. Per un percorso esplicito:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
ansible-playbook ansible/site.yml --limit atlas --tags restorecon \
|
||||||
|
-e '{"atlas_restorecon_paths":["/zpool/archive"]}'
|
||||||
|
```
|
||||||
|
|
||||||
|
Il task accetta solo percorsi sotto la radice del pool Atlas, esegue `restorecon -RFv` solo su quelli
|
||||||
|
indicati ed è altrimenti inattivo; non va lanciato sull'intero pool durante i run ordinari. Le vecchie
|
||||||
|
versioni USB non vengono eliminate automaticamente senza una retention deliberata. Il controllo di
|
||||||
|
capacità include il trasferimento stimato e una riserva libera di 10 GiB. Dopo un backup riuscito,
|
||||||
|
scollegare fisicamente il disco per renderlo davvero offline.
|
||||||
|
|
||||||
|
Validare la configurazione senza avviare il backup e, separatamente, un eventuale relabel pianificato:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
ANSIBLE_LOCAL_TEMP=/tmp/ansible-local \
|
||||||
|
ansible-playbook ansible/site.yml --limit atlas --tags usb_backup,usb_reminder --check --diff
|
||||||
|
ansible-playbook ansible/site.yml --limit atlas --tags restorecon --check \
|
||||||
|
-e '{"atlas_restorecon_paths":["/zpool/archive"]}'
|
||||||
|
```
|
||||||
|
|
||||||
|
Prima dell'avvio manuale smontare in sicurezza `/mnt/zpool-backup`, se ancora montato. Con il mapper
|
||||||
|
chiuso, `sudo systemctl start atlas-usb-backup.service` chiede la passphrase e avvia il backup; né la
|
||||||
|
password LUKS né un keyfile vanno in Ansible. Seguire con
|
||||||
|
`sudo journalctl -fu atlas-usb-backup.service`. **Non esiste un timer di backup USB.** Soltanto
|
||||||
|
`atlas-usb-reminder.timer` è schedulato il primo sabato del mese alle 10:00 `Europe/Rome`: invia un
|
||||||
|
promemoria al notifier 45Drives Houston, senza avviare il backup. Un test manuale ha prodotto una
|
||||||
|
notifica in 45Drives Alerts, **non un'email**; il log conferma l'invio della notifica, non la consegna
|
||||||
|
di posta. Il primo evento pianificato era il 2026-10-03 alle 10:00 CEST. Controllare timer e risultato
|
||||||
|
con `systemctl list-timers atlas-usb-reminder.timer` e in 45Drives Alerts.
|
||||||
|
|
||||||
|
Il primo tentativo USB del 2026-09-23 fallì su `security.selinux` e, dopo l'interruzione, lasciò
|
||||||
|
snapshot e mapper aperti. Applicato il filtro rsync, furono rimossi lo snapshot fallito, il mapper
|
||||||
|
smontato e lo stato failed; non rimase una copia valida di quel tentativo. Un run del 2026-09-24
|
||||||
|
pubblicò una versione verificata ma fallì nella distruzione dello snapshot a causa di mount
|
||||||
|
`.zfs/snapshot` aperti nel namespace host. Dopo la pulizia non forzata, è stato aggiunto un helper
|
||||||
|
`ExecStopPost` mirato e testato con uno snapshot usa-e-getta. Un run successivo del 2026-09-24 ha
|
||||||
|
verificato i checksum, pubblicato la versione ed è terminato con successo: mapper chiuso, nessuno
|
||||||
|
snapshot USB temporaneo e pool sano. Il 2026-09-25 un test di restore indipendente ha aperto il disco
|
||||||
|
in sola lettura, montato ext4 con `ro,noload`, copiato un file di 5.707.945 byte da `atlas/latest` in
|
||||||
|
una directory vuota sotto `/var/tmp` e confrontato contenuto, proprietario, modalità, dimensione,
|
||||||
|
mtime e ACL POSIX. Il test ha rimosso copia e mount temporanei, chiuso LUKS e lasciato il pool sano
|
||||||
|
mentre Borg continuava. È un test su file, non un esercizio completo di disaster recovery.
|
||||||
|
|
||||||
|
Il monitoraggio Atlas è eseguito ogni 30 minuti da `atlas-health-monitor.timer`. Sonde in sola
|
||||||
|
lettura controllano stato/errori del pool e dei vdev, scrub/resilver, SMART dei quattro dischi del
|
||||||
|
pool e dell'NVMe di sistema, temperature dei dischi e CPU, spazio di sistema/pool/snapshot, crescita
|
||||||
|
di `zpool/backup` e quota Hetzner tramite `df -m` via SSH con l'account `borg` e la chiave fissata.
|
||||||
|
La query remota non apre il repository Borg né il suo lock. Gli alert di crescita richiedono una
|
||||||
|
baseline di circa 24 ore. Sono controllati anche attivazione e freschezza dei timer; hook systemd
|
||||||
|
`OnFailure` segnalano errori di snapshot, scrub, Borg, USB, promemoria e monitoraggio. Il monitor non
|
||||||
|
riavvia Borg; avvisa solo se un run supera 14 giorni. Soglie e percorsi stabili dei dischi sono nelle
|
||||||
|
variabili host. Gli avvisi usano 45Drives Houston con deduplicazione; **la consegna email non è stata
|
||||||
|
verificata**. Il controllo live del 2026-09-25 non ha trovato problemi e ha inviato una notifica di
|
||||||
|
prova. Il 2026-09-30 il monitor ha rilevato zero problemi e una quota Storage Box occupata al 37%.
|
||||||
|
L'hook per i job falliti ora passa il nome letterale della unità systemd; l'espansione è stata
|
||||||
|
verificata senza inviare un falso allarme.
|
||||||
|
|
||||||
|
```bash
|
||||||
|
ansible-playbook ansible/site.yml --limit atlas --tags monitoring --check --diff
|
||||||
|
sudo /usr/local/libexec/atlas-health-monitor --dry-run
|
||||||
|
sudo journalctl -u atlas-health-monitor.service -n 100 --no-pager
|
||||||
|
systemctl list-timers atlas-health-monitor.timer
|
||||||
|
```
|
||||||
|
|
||||||
|
`--dry-run` non invia alert e non modifica lo stato del monitor. Un controllo reale si avvia con
|
||||||
|
`sudo systemctl start atlas-health-monitor.service`, senza avviare servizi di backup. Per una prova
|
||||||
|
etichettata di 45Drives Alerts usare
|
||||||
|
`sudo /usr/local/libexec/atlas-health-monitor --test-notification`.
|
||||||
|
|
||||||
|
### Timer systemd di Atlas
|
||||||
|
|
||||||
|
Tutti i dieci timer gestiti sono abilitati. Gli orari sono locali ad Atlas (`Europe/Rome`); Borg e
|
||||||
|
monitoraggio aggiungono il ritardo casuale indicato. Tutti hanno `Persistent=true`: un evento perso
|
||||||
|
viene recuperato quando il timer torna attivo.
|
||||||
|
|
||||||
|
| Timer | Pianificazione (`OnCalendar`) | Azione |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| `atlas-zfs-snapshot-hourly.timer` | `*-*-* *:05:00` — ogni ora al minuto 05 | Snapshot ricorsivo orario e retention |
|
||||||
|
| `atlas-zfs-snapshot-daily.timer` | `*-*-* 00:15:00` — ogni giorno alle 00:15 | Snapshot ricorsivo giornaliero e retention |
|
||||||
|
| `atlas-zfs-snapshot-weekly.timer` | `Sun *-*-* 01:00:00` — domenica alle 01:00 | Snapshot ricorsivo settimanale e retention |
|
||||||
|
| `atlas-zfs-snapshot-monthly.timer` | `*-*-01 02:00:00` — primo giorno del mese alle 02:00 | Snapshot ricorsivo mensile e retention |
|
||||||
|
| `zfs-scrub-monthly@zpool.timer` | `Sun *-*-01..07 03:00:00` — prima domenica alle 03:00 | Scrub ZFS |
|
||||||
|
| `atlas-borg-backup.timer` | `*-*-* 04:30:00` — ogni giorno alle 04:30, più 0–30 min casuali | Backup cifrato offsite |
|
||||||
|
| `atlas-borg-check.timer` | `*-*-15 06:00:00` — giorno 15 alle 06:00, più 0–30 min casuali | Controllo repository Borg |
|
||||||
|
| `atlas-usb-reminder.timer` | `Sat *-*-01..07 10:00:00 Europe/Rome` — primo sabato alle 10:00 | Solo promemoria 45Drives Alerts |
|
||||||
|
| `atlas-health-monitor.timer` | `*:0/30` — ogni mezz'ora, più 0–5 min casuali | Controlli di salute in sola lettura |
|
||||||
|
| `atlas-prometheus-pull.timer` | `*-*-* 03:00:00 Europe/Rome` — ogni giorno alle 03:00 | Pull e verifica del backup preparato su Prometheus |
|
||||||
|
|
||||||
|
`atlas-usb-backup.service` **non ha timer** e va avviato manualmente. Il timer del fornitore
|
||||||
|
`zfs-scrub-weekly@zpool.timer` è disabilitato a favore dello scrub mensile. Il timer di preparazione
|
||||||
|
su Prometheus è attivo alle 02:00 Europe/Rome; il primo ciclo pianificato è riuscito il 2026-10-01.
|
||||||
|
Un export, pull e ripristino temporaneo post-cutover NPM Quadlet sono riusciti il 2026-10-03;
|
||||||
|
il primo ciclo pianificato dopo quel cutover resta da osservare. Durante un backup Borg attivo,
|
||||||
|
`systemctl list-timers` può mostrare `-` per il prossimo evento senza che il timer sia disabilitato.
|
||||||
|
Per vedere la pianificazione corrente: `systemctl list-timers --all` su Atlas.
|
||||||
|
|
||||||
|
Nextcloud è previsto come servizio temporaneo su Atlas prima di Uranus, ma solo dopo la validazione
|
||||||
|
della protezione dei dati: richiede storage applicativo, database e cache separati, segreti Vault,
|
||||||
|
pubblicazione solo tramite NPM e Aegis, procedure di backup, aggiornamento e migrazione. Non
|
||||||
|
distribuirlo prima di completare la checklist di protezione dei dati.
|
||||||
|
|
||||||
|
Atlas è la destinazione dichiarata per iCloudPD. Ansible gestisce dataset, Quadlet rootless e
|
||||||
|
`icloudpd.conf` privato con Apple ID dal Vault: foto in `/zpool/archive/Pictures/iCloudPD`,
|
||||||
|
stato in `zpool/services/data/icloudpd`. Il primo avvio è stato manuale; password e MFA restano
|
||||||
|
gestiti interattivamente, senza avvio automatico al boot. L'inizializzazione è stata completata e
|
||||||
|
il download iniziale di foto e video è terminato il 2026-10-03. Su Aegis
|
||||||
|
il servizio, il Quadlet e `/var/lib/icloudpd` sono stati rimossi e verificati; il playbook Aegis
|
||||||
|
non contiene più task iCloudPD. L'accesso SMB e il ripristino dai backup dei nuovi dati restano
|
||||||
|
da verificare. L'export NFS Photobook resta
|
||||||
|
invariato. Dettagli in [`docs/atlas-icloudpd-migration.md`](docs/atlas-icloudpd-migration.md).
|
||||||
|
|
||||||
|
Il primo ciclo pianificato del backup di Prometheus e una prova di disaster recovery a dimensione reale
|
||||||
|
restano da verificare. Il 2026-09-30 una VM Rocky isolata ha superato ricostruzione OS con Ansible,
|
||||||
|
import del pool RAIDZ2 fittizio e ripristino da snapshot; RPO 24 ore/RTO 72 ore restano obiettivi
|
||||||
|
provvisori, non tempi misurati. Dettagli e limiti sono in `docs/atlas-recovery.md`. Il backlog
|
||||||
|
prioritizzato è in `AGENTS.md`.
|
||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
@@ -426,8 +628,8 @@ Questo significa che, allo stato attuale:
|
|||||||
- `deadalus` riceve il profilo Fedora WSL tramite play dev dedicati
|
- `deadalus` riceve il profilo Fedora WSL tramite play dev dedicati
|
||||||
- il server Rocky (`prometheus`) e gestito con pacchetti, servizi, dotfiles server e firewalld
|
- il server Rocky (`prometheus`) e gestito con pacchetti, servizi, dotfiles server e firewalld
|
||||||
- il NAS Rocky (`atlas`) usa un pool ZFS gia esistente, condivisioni NFSv4/SMB limitate alla LAN e Cockpit/45Drives
|
- il NAS Rocky (`atlas`) usa un pool ZFS gia esistente, condivisioni NFSv4/SMB limitate alla LAN e Cockpit/45Drives
|
||||||
- lo stack Compose server include soltanto `gitea` e `nginx-proxy-manager`; Navidrome e Syncthing
|
- NPM è un Quadlet rootful su Prometheus, mentre Gitea, Navidrome e Syncthing sono Quadlet
|
||||||
della fase 1 sono Quadlet rootless su Atlas
|
rootless su Atlas; il fallback Compose server è stato rimosso
|
||||||
|
|
||||||
# Dotfiles
|
# Dotfiles
|
||||||
|
|
||||||
@@ -533,9 +735,10 @@ ansible-playbook ansible/site.yml --limit <host> --tags <tag1>,<tag2> --check --
|
|||||||
ansible-playbook ansible/site.yml --limit <host> --start-at-task "<task name>" --check --diff
|
ansible-playbook ansible/site.yml --limit <host> --start-at-task "<task name>" --check --diff
|
||||||
ansible-lint ansible/roles/<role>
|
ansible-lint ansible/roles/<role>
|
||||||
yamllint ansible/path/to/file.yml
|
yamllint ansible/path/to/file.yml
|
||||||
podman-compose -f /opt/docker/server/docker-compose.yml config
|
ansible-playbook ansible/site.yml --limit prometheus --tags server_legacy_cleanup --check --diff -e server_legacy_cleanup=true
|
||||||
ansible-playbook ansible/site.yml --limit atlas --tags storage,sharing,containers --check --diff
|
ansible-playbook ansible/site.yml --limit atlas --tags storage,sharing,containers --check --diff
|
||||||
ansible-playbook ansible/site.yml --limit atlas --tags backend_phase1 --check --diff
|
ansible-playbook ansible/site.yml --limit atlas --tags backend_phase1 --check --diff
|
||||||
|
ansible-playbook ansible/site.yml --limit atlas --tags music_sync --check --diff
|
||||||
```
|
```
|
||||||
|
|
||||||
## Tag supportati dal playbook
|
## Tag supportati dal playbook
|
||||||
|
|||||||
352
README.md
352
README.md
@@ -67,6 +67,27 @@ The official ChatGPT desktop RPM is enabled only on `ikaros` and `nymph`. The
|
|||||||
playbook configures OpenAI's signed RPM repository and imports its pinned RPM
|
playbook configures OpenAI's signed RPM repository and imports its pinned RPM
|
||||||
signing key before installation; subsequent updates are handled by DNF.
|
signing key before installation; subsequent updates are handled by DNF.
|
||||||
|
|
||||||
|
## Deferred planned node: Cerberus
|
||||||
|
|
||||||
|
`cerberus` is a **postponed** management-plane node, pending the physical setup
|
||||||
|
of the office in the new house. It is not yet an inventory host and no role or
|
||||||
|
playbook targets it.
|
||||||
|
|
||||||
|
The planned hardware is a Lenovo ThinkCentre M700 Tiny (Intel Core i3-6100T,
|
||||||
|
8 GB RAM, and a 256 GB SSD) with native 1 Gbps Ethernet. It will share Ikaros'
|
||||||
|
monitor and peripherals through a multi-input KVM switch, using a passive
|
||||||
|
DisplayPort-to-HDMI cable for its video connection. Fedora Sericea, the
|
||||||
|
immutable Fedora variant with the Sway Wayland compositor, is the intended
|
||||||
|
operating system.
|
||||||
|
|
||||||
|
Cerberus will be an isolated management plane: Ansible will run from a
|
||||||
|
dedicated Toolbox environment to provision the future `uranus` cluster, rather
|
||||||
|
than from Ikaros or an unmanaged host. Its rootless Podman observability stack
|
||||||
|
will run Grafana, Prometheus, and Loki. The local SSD is the hot tier and
|
||||||
|
retains metrics and logs for 30 days; scheduled exports will place older
|
||||||
|
historical data on an NFS-mounted Atlas dataset as the cold tier. The detailed,
|
||||||
|
implementation-gated plan is maintained in `AGENTS.md`.
|
||||||
|
|
||||||
## Desktop profiles
|
## Desktop profiles
|
||||||
|
|
||||||
- `ikaros`: stable Fedora Workstation + GNOME desktop.
|
- `ikaros`: stable Fedora Workstation + GNOME desktop.
|
||||||
@@ -104,16 +125,26 @@ That gives it Fedora packages through DNF, Docker from the official repository,
|
|||||||
|
|
||||||
## Server
|
## Server
|
||||||
|
|
||||||
|
The public service domain transition to `fscotto.co`, Gitea canonical URL
|
||||||
|
management, and remaining DuckDNS retirement steps are documented in
|
||||||
|
[`docs/domain-fscotto-co.md`](docs/domain-fscotto-co.md).
|
||||||
|
|
||||||
`prometheus` is the Rocky Linux 9 server. It has no graphical environment and gets server-specific
|
`prometheus` is the Rocky Linux 9 server. It has no graphical environment and gets server-specific
|
||||||
dotfiles and templates. The profile provisions configuration only: it does not transfer data, start
|
dotfiles and templates. The profile does not transfer application data, update DNS, or perform an
|
||||||
the Compose stack, update DNS, or perform a cutover.
|
implicit service cutover.
|
||||||
|
|
||||||
The server profile installs platform-specific packages, Podman and podman-compose, declared systemd
|
The server profile installs platform-specific packages, Podman and podman-compose, declared systemd
|
||||||
services, and firewalld. The manually activated `podman-compose-server` unit contains the existing
|
services, and firewalld. Nginx Proxy Manager runs as the rootful `prometheus-npm.service` Quadlet.
|
||||||
Nginx Proxy Manager and Gitea services. The desired Compose file no longer includes Navidrome,
|
On 2026-10-03 the operator-approved opt-in cleanup removed old Gitea, Navidrome and PostgreSQL
|
||||||
Syncthing, or the obsolete Navidrome PostgreSQL database; their temporary Atlas deployment is managed
|
data/images, empty legacy directories, the Gitea final-export helper and the Compose rollback files.
|
||||||
by `profile_backend_phase1`. Applying the profile does not stop or remove legacy containers and does
|
The migrated services stay on Atlas. `server_legacy_stack_retired: true` prevents normal runs from
|
||||||
not delete `/opt/postgres/data`.
|
recreating retired files. Data deletion requires `--tags server_legacy_cleanup` and
|
||||||
|
`-e server_legacy_cleanup=true`; image-only cleanup has its own `server_image_cleanup` tag and flag.
|
||||||
|
Active NPM resources and existing backup archives remain preserved.
|
||||||
|
The post-cleanup export/pull and isolated SQLite restore passed; the first unattended cycle remains
|
||||||
|
pending. Evidence and recovery boundaries:
|
||||||
|
[`docs/prometheus-npm-quadlet.md`](docs/prometheus-npm-quadlet.md).
|
||||||
|
|
||||||
|
|
||||||
Firewalld enables SSH, Cockpit (`9090/tcp`), HTTP and HTTPS. Nginx Proxy Manager publishes only
|
Firewalld enables SSH, Cockpit (`9090/tcp`), HTTP and HTTPS. Nginx Proxy Manager publishes only
|
||||||
`80/tcp` and `443/tcp`; its administration interface is bound to `127.0.0.1:81` and can be reached
|
`80/tcp` and `443/tcp`; its administration interface is bound to `127.0.0.1:81` and can be reached
|
||||||
@@ -138,45 +169,12 @@ The target must already provide `server_username` with local sudo access.
|
|||||||
Prometheus authorizes its declared SSH public keys through separate files below
|
Prometheus authorizes its declared SSH public keys through separate files below
|
||||||
`~/.ssh/authorized_keys.d/`, while `sshd` is configured to read those files directly.
|
`~/.ssh/authorized_keys.d/`, while `sshd` is configured to read those files directly.
|
||||||
|
|
||||||
### DuckDNS
|
### DuckDNS retirement
|
||||||
|
|
||||||
`profile_server` renders `~/duckdns/duck.sh` with mode `0700`, keeping the existing updater path
|
DuckDNS support has been removed from the server profile: no tasks, templates,
|
||||||
and `duck.log`. Set `server_duckdns_domain` in the server's host vars and store the **rotated**
|
variables or enablement flags remain. Prometheus uses its static IP and `fscotto.co`.
|
||||||
`vault_duckdns_token` in encrypted `secrets/vault.yml` (using `ansible-vault edit secrets/vault.yml`)
|
The local updater, log and cron job were already removed. The external DuckDNS
|
||||||
or untracked `secrets/vault.local.yml`. Never commit the rendered script or put the token on a
|
name/account and existing encrypted token remain untouched for possible future use.
|
||||||
command line. Rendering hides secret output/diffs; the updater verifies TLS and passes the token
|
|
||||||
to curl through stdin. The playbook neither runs the updater nor changes its external schedule.
|
|
||||||
|
|
||||||
```bash
|
|
||||||
ansible-playbook ansible/site.yml --limit prometheus --tags duckdns --check --diff
|
|
||||||
ansible-playbook ansible/site.yml --limit prometheus --tags duckdns
|
|
||||||
```
|
|
||||||
|
|
||||||
An exposed token must be revoked/regenerated on DuckDNS: deleting it from Git history does not
|
|
||||||
revoke it. After a history cleanup, re-clone other checkouts rather than merging the old history
|
|
||||||
back in; preserve any uncommitted work separately without copying secrets.
|
|
||||||
|
|
||||||
### Data migration
|
|
||||||
|
|
||||||
Provision Rocky first, then run the migration script **on the retired Ubuntu source host**. It is
|
|
||||||
dry-run by default and requires an explicit source-stack stop before it can copy application data:
|
|
||||||
|
|
||||||
```bash
|
|
||||||
sudo ./scripts/migrate_prometheus_data.sh \
|
|
||||||
--destination rocky@179.237.102.172 \
|
|
||||||
--identity /root/.ssh/id_ed25519
|
|
||||||
|
|
||||||
sudo ./scripts/migrate_prometheus_data.sh \
|
|
||||||
--destination rocky@179.237.102.172 \
|
|
||||||
--identity /root/.ssh/id_ed25519 \
|
|
||||||
--quiesce-source --execute
|
|
||||||
```
|
|
||||||
|
|
||||||
The script copies only Nginx Proxy Manager and Gitea data. It does not delete data, move
|
|
||||||
Navidrome/Syncthing, copy `/home/git/.ssh`, start containers, update DNS, or perform a cutover. The
|
|
||||||
destination SSH host key must already be trusted and the destination account needs passwordless sudo
|
|
||||||
for `rsync`. It preserves ACLs but not extended attributes, so source SELinux labels are not
|
|
||||||
transferred; the Rocky Compose bind mounts apply their own `:Z` labels when containers start.
|
|
||||||
|
|
||||||
## DNS Filter
|
## DNS Filter
|
||||||
|
|
||||||
@@ -189,8 +187,8 @@ ansible/bootstrap/generate-aegis-ign.sh --write IMAGE DEVICE
|
|||||||
```
|
```
|
||||||
|
|
||||||
The controller manages it remotely as `pi@aegis`; unlike local desktop profiles, Aegis is
|
The controller manages it remotely as `pi@aegis`; unlike local desktop profiles, Aegis is
|
||||||
intentionally an SSH inventory target. `profile_aegis` manages rootful Podman Quadlets for AdGuard
|
intentionally an SSH inventory target. `profile_aegis` manages a rootful Podman Quadlet for AdGuard
|
||||||
Home and iCloudPD, persistent data under `/var/lib`, the Podman auto-update timer, LAN-restricted
|
Home, its persistent data under `/var/lib`, the Podman auto-update timer, LAN-restricted
|
||||||
firewalld rules, SSH key-only access for `pi`, the `nfs-utils` and `wireguard-tools` rpm-ostree layers,
|
firewalld rules, SSH key-only access for `pi`, the `nfs-utils` and `wireguard-tools` rpm-ostree layers,
|
||||||
and `wake-ikaros`. `wireguard_overlay` makes Aegis the internal endpoint and LAN gateway for Prometheus:
|
and `wake-ikaros`. `wireguard_overlay` makes Aegis the internal endpoint and LAN gateway for Prometheus:
|
||||||
it enables persistent IPv4 forwarding, installs a scoped WireGuard-to-LAN firewalld policy, and source-NATs
|
it enables persistent IPv4 forwarding, installs a scoped WireGuard-to-LAN firewalld policy, and source-NATs
|
||||||
@@ -203,9 +201,8 @@ opened and closed manually during initial setup. The profile disables the local
|
|||||||
stub and points `/etc/resolv.conf` to its full resolver data, freeing port 53 for AdGuard. LAN clients
|
stub and points `/etc/resolv.conf` to its full resolver data, freeing port 53 for AdGuard. LAN clients
|
||||||
may use AdGuard on Aegis, while Aegis itself uses the independent upstream DNS declared by
|
may use AdGuard on Aegis, while Aegis itself uses the independent upstream DNS declared by
|
||||||
`aegis_host_dns_servers`; this prevents Greenboot from depending on the AdGuard container during
|
`aegis_host_dns_servers`; this prevents Greenboot from depending on the AdGuard container during
|
||||||
startup. Reboot Aegis after changing its NetworkManager DNS profile. Define
|
startup. Reboot Aegis after changing its NetworkManager DNS profile. iCloudPD was retired from Aegis;
|
||||||
`vault_aegis_icloudpd_apple_id` in Vault before applying it. iCloudPD still requires interactive MFA
|
the Aegis role no longer contains iCloudPD tasks. Atlas iCloudPD config is Vault-backed; MFA is manual.
|
||||||
initialization after its first deployment.
|
|
||||||
|
|
||||||
New Aegis images create the `admin` account in Butane. Before configuring a newly imaged node, run its
|
New Aegis images create the `admin` account in Butane. Before configuring a newly imaged node, run its
|
||||||
first playbook execution with `-e ansible_user=admin`; the SSH hardening role then permits that same
|
first playbook execution with `-e ansible_user=admin`; the SSH hardening role then permits that same
|
||||||
@@ -275,7 +272,20 @@ Atlas temporarily hosts rootless Navidrome and Syncthing until Uranus replaces t
|
|||||||
Atlas' LAN address (`192.168.178.55`); WireGuard remains exclusively between Prometheus (`10.0.0.1`)
|
Atlas' LAN address (`192.168.178.55`); WireGuard remains exclusively between Prometheus (`10.0.0.1`)
|
||||||
and Aegis (`10.0.0.2`). Their state is initialized ex novo in `/zpool/services/data/navidrome` and
|
and Aegis (`10.0.0.2`). Their state is initialized ex novo in `/zpool/services/data/navidrome` and
|
||||||
`/zpool/services/data/syncthing`; no source application state is migrated. The music library at
|
`/zpool/services/data/syncthing`; no source application state is migrated. The music library at
|
||||||
`/zpool/media/music` is populated separately.
|
`/zpool/media/music` was populated separately from `/zpool/archive/Music` on 2026-09-30;
|
||||||
|
Navidrome completed its library scan. The rootless `atlas-music-sync.timer` copies new and changed
|
||||||
|
files daily at 00:45 Europe/Rome, without deleting destination-only files. Both ZFS datasets must
|
||||||
|
be mounted. Its first scheduled run succeeded on 2026-10-02. Some source playlists still contain
|
||||||
|
obsolete Windows paths.
|
||||||
|
|
||||||
|
The Gitea move from Prometheus to Atlas is tracked in
|
||||||
|
[`docs/atlas-gitea-migration.md`](docs/atlas-gitea-migration.md). The final consistent copy runs in
|
||||||
|
Atlas' dedicated dataset under `admin`'s rootless user Quadlet. Its pinned derived image uses an
|
||||||
|
internal Unix user named `gitea` (UID/GID 1000), while clone URLs keep `git@`. NPM remains on Prometheus and the primary
|
||||||
|
public HTTPS route serves Atlas. Public SSH/2222 authenticates the `ikaros` key and serves
|
||||||
|
`git ls-remote`; the operator also confirmed SSH pull and push. HTTPS login and Git writes were
|
||||||
|
confirmed on 2026-10-03. The old Gitea data remains on Prometheus, but its container
|
||||||
|
is absent from the desired stack.
|
||||||
|
|
||||||
The separate `wireguard_overlay` role manages `wg0` between Prometheus (`10.0.0.1`) and Aegis
|
The separate `wireguard_overlay` role manages `wg0` between Prometheus (`10.0.0.1`) and Aegis
|
||||||
(`10.0.0.2`), generating private keys once on their respective hosts and exchanging only public keys
|
(`10.0.0.2`), generating private keys once on their respective hosts and exchanging only public keys
|
||||||
@@ -299,9 +309,240 @@ The first real WireGuard run must include both peers. If Fedora IoT has just lay
|
|||||||
reboot Aegis manually and rerun the command without `--check`; the role then waits for a real peer
|
reboot Aegis manually and rerun the command without `--check`; the role then waits for a real peer
|
||||||
handshake.
|
handshake.
|
||||||
|
|
||||||
Snapshot retention, Syncthing topology, WireGuard/firewall validation, Prometheus backup pulls,
|
Atlas declares recursive, systemd-timed ZFS snapshots for the complete pool hierarchy: 24 hourly
|
||||||
encrypted Borg backups to a Hetzner Storage Box, USB backup, monitoring, and disaster-recovery tests
|
snapshots at minute 05, 30 daily snapshots at 00:15, 8 weekly snapshots on Sunday at 01:00, and 12
|
||||||
remain follow-up work. The detailed operational backlog is kept in `AGENTS.md`.
|
monthly snapshots on the first day at 02:00. The retention helper prunes only snapshots carrying its
|
||||||
|
managed `atlas-auto` prefix and never rolls back a dataset. The OpenZFS monthly scrub timer is scheduled
|
||||||
|
for the first Sunday at 03:00; the conflicting weekly scrub timer is disabled explicitly. The first recursive
|
||||||
|
hourly snapshot completed successfully on Atlas, and scheduled retention pruning was observed on
|
||||||
|
2026-09-30. The first monthly scrub still awaits runtime evidence. Validate this layer independently with:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
ANSIBLE_LOCAL_TEMP=/tmp/ansible-local \
|
||||||
|
ansible-playbook ansible/site.yml --limit atlas --tags snapshots,scrub --check --diff
|
||||||
|
```
|
||||||
|
|
||||||
|
Atlas also declares an encrypted Borg backup to the dedicated Hetzner Storage Box sub-account
|
||||||
|
`u660064-sub1`. The repository is the sub-account-relative `./borg-data` path and uses the explicitly
|
||||||
|
selected remote Borg 1.4 binary over SSH port 23. The ED25519 server key is pinned; a dedicated client
|
||||||
|
key is generated for the locked, non-login `borg` system account, and its private half never leaves
|
||||||
|
`/etc/atlas-borg`. The account has no sudo or supplementary groups and owns only its SSH identity,
|
||||||
|
passphrase, cache, and Borg state. Borg receives its passphrase through a mode `0600` file rendered from
|
||||||
|
`vault_atlas_borg_passphrase`.
|
||||||
|
|
||||||
|
The daily backup starts at 04:30 with up to 30 minutes of randomized delay. It creates a temporary,
|
||||||
|
recursive ZFS snapshot and reconstructs every dataset below `/zpool` as a read-only bind-mounted tree,
|
||||||
|
so parent and child datasets enter one consistent Borg archive. The wrapper recursively unmounts its
|
||||||
|
private source tree; a narrowly scoped `ExecStopPost` helper removes any remaining host-namespace ZFS
|
||||||
|
snapshot mounts and the named temporary snapshot after the backup process exits. Only the root wrapper
|
||||||
|
performs snapshot and mount operations; it launches the Borg client as `borg` with temporary read-search
|
||||||
|
capability and no ZFS, sudo, or pool-management privileges. Borg retains 30 daily, 8 weekly, and 12
|
||||||
|
monthly archives, then compacts the standard
|
||||||
|
read-write repository. A full metadata and repository check runs as `borg` on the fifteenth day of each
|
||||||
|
month at 06:00. Both operations use a common lock, journal logging, and bounded systemd retries.
|
||||||
|
New backup runs also log the create phase and a compact progress line at most once per minute: an
|
||||||
|
**estimated** percentage, dataset, files processed, and original/compressed/deduplicated bytes. The
|
||||||
|
denominator is the summed ZFS `logicalreferenced` size of the backup's own recursive snapshot, not a
|
||||||
|
Borg-reported total: the estimate can exceed 100% and does not cover retention, compaction, or checks.
|
||||||
|
Progress lines omit individual filenames; warnings may still name affected files.
|
||||||
|
Follow the current run with
|
||||||
|
`sudo journalctl -fu atlas-borg-backup.service` on Atlas; changes to the helper do not alter a run
|
||||||
|
already in progress.
|
||||||
|
|
||||||
|
Initial activation remains explicit:
|
||||||
|
|
||||||
|
1. Add a strong unique `vault_atlas_borg_passphrase` with `ansible-vault edit secrets/vault.yml`.
|
||||||
|
2. Generate and display only the dedicated public key with
|
||||||
|
`ansible-playbook ansible/site.yml --limit atlas --tags borg_key`.
|
||||||
|
3. Install that public key in the Hetzner sub-account, then apply with
|
||||||
|
`ansible-playbook ansible/site.yml --limit atlas --tags packages,borg`.
|
||||||
|
4. Copy the ignored `secrets/recovery/atlas-borg-repokey.export` file to genuinely offline storage.
|
||||||
|
The controller-side copy is not an offline backup by itself.
|
||||||
|
|
||||||
|
The role initializes only the missing `repokey` repository and never accepts an unpinned host key or
|
||||||
|
password authentication. It does not start the first backup manually. Validate the rendered state with:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
ANSIBLE_LOCAL_TEMP=/tmp/ansible-local \
|
||||||
|
ansible-playbook ansible/site.yml --limit atlas --tags packages,borg --check --diff
|
||||||
|
```
|
||||||
|
|
||||||
|
Atlas runtime activation is complete: the initial backup and repository check succeeded, a full restore
|
||||||
|
to a temporary directory was validated against the live `Archive` tree, the recovery-key export was copied
|
||||||
|
to offline storage, and the temporary snapshot and bind mounts were cleaned up.
|
||||||
|
The populated-pool archive on 2026-09-29 took 1 h 32 min for 2.18 TB original / 2.04 TB compressed
|
||||||
|
data, with a 13.49 GB deduplicated archive size. Retention and compaction succeeded, but a
|
||||||
|
`RuntimeDirectory` permission error prevented post-exit snapshot cleanup. After correction, the
|
||||||
|
2026-09-30 incremental archive completed in about 22 seconds, removed the stale and current temporary
|
||||||
|
snapshots, and ended with service status 0. The monitor reported 37% Storage Box quota used. These
|
||||||
|
observations do not predict the duration or compression ratio of future runs.
|
||||||
|
On 2026-09-25 a separate ZFS restore smoke test copied a small file from an automatic daily
|
||||||
|
`zpool/archive` snapshot to `/var/tmp`, then confirmed matching contents, ownership, mode, mtime and
|
||||||
|
POSIX ACL. The temporary copy and on-demand snapshot mount were removed; Borg kept running. This
|
||||||
|
does not validate a full dataset recovery.
|
||||||
|
|
||||||
|
The offline USB backup is deployed as a manual-only service (`atlas_manage_usb_backup: true`):
|
||||||
|
Ansible never formats, unlocks, mounts, backs up to, or schedules the disk. Atlas' existing USB disk was verified
|
||||||
|
read-only on 2026-09-23 as LUKS UUID `577b3c43-ea37-4611-81a9-39d555cdfbd4`, containing ext4 UUID
|
||||||
|
`758e2d2e-a427-4797-aad9-39c3a9f17c7e` through mapper `zpool-backup`. It was mounted at
|
||||||
|
`/mnt/zpool-backup` at inspection time. The service deliberately requires the verified mapper to be
|
||||||
|
**not mounted** before starting. When necessary, `systemd-ask-password` requests the LUKS passphrase
|
||||||
|
through the `systemctl start` password agent; it is piped directly to `cryptsetup` without saving it,
|
||||||
|
passing it as a command argument, or caching it. The service then mounts the disk privately, takes a recursive ZFS snapshot,
|
||||||
|
copies every dataset to a versioned `atlas/snapshots/<timestamp>/` directory using `rsync --link-dest`,
|
||||||
|
verifies the result with a checksum-based dry run, atomically updates `atlas/latest`, unmounts and closes
|
||||||
|
LUKS. A failed run never replaces `latest` or removes an earlier complete version. Borg and the USB
|
||||||
|
backup may run concurrently from separate snapshots; both reading the same pool can reduce throughput.
|
||||||
|
The USB copy preserves ACLs but not generic extended attributes; `security.selinux` is also intentionally
|
||||||
|
excluded because the target SELinux policy must recreate labels during a restore. Do not restore data into
|
||||||
|
service paths without relabeling. After restoring an explicit dataset path, apply its destination policy with:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
ansible-playbook ansible/site.yml --limit atlas --tags restorecon \
|
||||||
|
-e '{"atlas_restorecon_paths":["/zpool/archive"]}'
|
||||||
|
```
|
||||||
|
|
||||||
|
The task accepts only paths below the Atlas pool mount root, runs `restorecon -RFv` only for the paths
|
||||||
|
provided at invocation, and is otherwise a no-op. It must not be used on the whole pool during routine runs.
|
||||||
|
Old USB versions are not pruned automatically, to avoid deleting the only offline
|
||||||
|
copy without an explicitly chosen retention policy; capacity checks include an estimated transfer size
|
||||||
|
and a 10 GiB free-space reserve. The disk must be physically disconnected after a successful backup
|
||||||
|
to make the copy offline.
|
||||||
|
|
||||||
|
To check the USB backup and reminder configuration without starting a backup, run:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
ANSIBLE_LOCAL_TEMP=/tmp/ansible-local \
|
||||||
|
ansible-playbook ansible/site.yml --limit atlas --tags usb_backup,usb_reminder --check --diff
|
||||||
|
```
|
||||||
|
|
||||||
|
To validate a planned, explicit post-restore relabel operation without changing labels, run:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
ansible-playbook ansible/site.yml --limit atlas --tags restorecon --check \
|
||||||
|
-e '{"atlas_restorecon_paths":["/zpool/archive"]}'
|
||||||
|
```
|
||||||
|
|
||||||
|
Before the first **manual** service start, safely unmount the currently mounted
|
||||||
|
`/mnt/zpool-backup`; never run it on an arbitrary mounted disk. Future starts
|
||||||
|
can begin with the mapper closed: `sudo systemctl start atlas-usb-backup.service` prompts for the
|
||||||
|
passphrase interactively and then performs the backup. Neither the LUKS password nor a key file belongs
|
||||||
|
in Ansible. Inspect the run with
|
||||||
|
`sudo journalctl -fu atlas-usb-backup.service`. There is intentionally no timer. Independently test a
|
||||||
|
read-only mount and restore from `atlas/latest` into an empty temporary directory before marking the
|
||||||
|
USB recovery path complete. Only `atlas-usb-reminder.timer` is enabled, for the first Saturday of each
|
||||||
|
month at 10:00 Europe/Rome. Its warning notification uses the existing 45Drives Houston notifier.
|
||||||
|
A manual test confirmed a notification in 45Drives Alerts, **not** an email. The reminder service log
|
||||||
|
reports notification submission, not email delivery; the role does not depend on SMTP/OAuth settings.
|
||||||
|
The reminder never starts the backup. Check its schedule with
|
||||||
|
`systemctl list-timers atlas-usb-reminder.timer` and the result in 45Drives Alerts.
|
||||||
|
The timer was verified active with its first scheduled run at 2026-10-03 10:00 CEST. No email
|
||||||
|
delivery is claimed.
|
||||||
|
The first manual USB attempt on 2026-09-23 did not complete: rsync was denied while removing
|
||||||
|
`security.selinux` on the USB filesystem, then the interrupted service left its recursive
|
||||||
|
`atlas-usb-20260923T185748Z-2469168` snapshot and the `zpool-backup` LUKS mapper open. The
|
||||||
|
rsync xattr filter was deployed afterward. The incomplete USB directory was absent on inspection;
|
||||||
|
the exact failed snapshot was removed, the verified and unmounted mapper closed, and the service
|
||||||
|
failed state cleared. A final check found no remnant snapshot, mount, mapper, or staging directory.
|
||||||
|
The failed attempt was not a valid backup, and no USB restore had been tested at that point.
|
||||||
|
On 2026-09-24 a later run reported a checksum-verified, published USB version and closed the LUKS
|
||||||
|
mapper, but the service failed while destroying its temporary ZFS snapshot: OpenZFS still had
|
||||||
|
on-demand `.zfs/snapshot` mounts open in the host namespace. Those exact temporary snapshots were
|
||||||
|
unmounted normally and removed; no force or rollback was used. The backup service now records its
|
||||||
|
snapshot name and runs a narrowly scoped `ExecStopPost` cleanup after the private backup process
|
||||||
|
exits. The cleanup helper was tested with a disposable recursive snapshot and an active snapshot
|
||||||
|
mount. A complete run on 2026-09-24 later checksum-verified and published a new USB version; the
|
||||||
|
service ended successfully, the LUKS mapper closed, no temporary USB snapshot remained, and the pool
|
||||||
|
was healthy. On 2026-09-25 an independent restore test opened the configured USB disk read-only, mounted
|
||||||
|
ext4 with `ro,noload`, restored a 5,707,945-byte file from the published `atlas/latest` version to an
|
||||||
|
empty `/var/tmp` directory, and matched its content, owner, mode, size, mtime, and POSIX ACL against
|
||||||
|
the USB source. The test removed its temporary copy and mount, closed the LUKS mapper, and left the
|
||||||
|
pool healthy while Borg continued running. This is a file-level recovery smoke test, not a full dataset
|
||||||
|
or disaster-recovery exercise.
|
||||||
|
|
||||||
|
Atlas health monitoring runs every 30 minutes through `atlas-health-monitor.timer`. Its read-only probes
|
||||||
|
check pool/vdev state and errors, scrub/resilver status, four pool disks and the system NVMe via SMART,
|
||||||
|
disk and CPU temperatures, system/pool/snapshot space, local `zpool/backup` growth, and the Hetzner
|
||||||
|
Storage Box quota via `df -m` over the dedicated `borg` account's pinned-key SSH connection. The remote
|
||||||
|
query never opens the Borg repository or its lock. Growth alerts compare against a roughly 24-hour
|
||||||
|
baseline and therefore begin only after enough samples exist. The monitor also checks
|
||||||
|
maintenance/backup timer activation and freshness; systemd `OnFailure` hooks report snapshot,
|
||||||
|
scrub, Borg, USB, reminder, and monitoring services when they enter the failed state. An ongoing
|
||||||
|
Borg run is never restarted by the monitor; only a run exceeding 14 days raises a warning.
|
||||||
|
Thresholds and stable disk paths are declared in Atlas host variables. Alerts use the existing 45Drives
|
||||||
|
Houston notifier and repeated issues are deduplicated; **email delivery is not verified**. The
|
||||||
|
2026-09-25 live probe found no issues and a labelled test notification was submitted. On 2026-09-30
|
||||||
|
the monitor reported zero issues and 37% Storage Box quota used. The failed-job hook now passes the
|
||||||
|
literal systemd unit name; its expansion was verified without sending a false failure notification.
|
||||||
|
|
||||||
|
```bash
|
||||||
|
ansible-playbook ansible/site.yml --limit atlas --tags monitoring --check --diff
|
||||||
|
sudo /usr/local/libexec/atlas-health-monitor --dry-run
|
||||||
|
sudo journalctl -u atlas-health-monitor.service -n 100 --no-pager
|
||||||
|
systemctl list-timers atlas-health-monitor.timer
|
||||||
|
```
|
||||||
|
|
||||||
|
`--dry-run` sends no alerts and does not change monitor state. A real check is
|
||||||
|
`sudo systemctl start atlas-health-monitor.service`; do not start the backup services merely to test
|
||||||
|
monitoring. For a labelled 45Drives Alerts delivery test, use
|
||||||
|
`sudo /usr/local/libexec/atlas-health-monitor --test-notification`.
|
||||||
|
|
||||||
|
### Atlas systemd timers
|
||||||
|
|
||||||
|
All ten managed timers below are enabled. Times are local to Atlas (`Europe/Rome`); Borg and monitoring
|
||||||
|
add the indicated randomized delay. Every timer has `Persistent=true`, so a missed calendar run is
|
||||||
|
scheduled after the timer becomes active again.
|
||||||
|
|
||||||
|
| Timer | Schedule (`OnCalendar`) | Action |
|
||||||
|
| --- | --- | --- |
|
||||||
|
| `atlas-zfs-snapshot-hourly.timer` | `*-*-* *:05:00` — every hour at :05 | Recursive hourly snapshot and retention |
|
||||||
|
| `atlas-zfs-snapshot-daily.timer` | `*-*-* 00:15:00` — daily at 00:15 | Recursive daily snapshot and retention |
|
||||||
|
| `atlas-zfs-snapshot-weekly.timer` | `Sun *-*-* 01:00:00` — Sunday at 01:00 | Recursive weekly snapshot and retention |
|
||||||
|
| `atlas-zfs-snapshot-monthly.timer` | `*-*-01 02:00:00` — first day of the month at 02:00 | Recursive monthly snapshot and retention |
|
||||||
|
| `zfs-scrub-monthly@zpool.timer` | `Sun *-*-01..07 03:00:00` — first Sunday at 03:00 | ZFS scrub |
|
||||||
|
| `atlas-borg-backup.timer` | `*-*-* 04:30:00` — daily at 04:30, plus 0–30 min random delay | Encrypted offsite backup |
|
||||||
|
| `atlas-borg-check.timer` | `*-*-15 06:00:00` — 15th of the month at 06:00, plus 0–30 min random delay | Borg repository check |
|
||||||
|
| `atlas-usb-reminder.timer` | `Sat *-*-01..07 10:00:00 Europe/Rome` — first Saturday at 10:00 | 45Drives Alerts reminder only |
|
||||||
|
| `atlas-health-monitor.timer` | `*:0/30` — every half-hour, plus 0–5 min random delay | Read-only health checks |
|
||||||
|
| `atlas-prometheus-pull.timer` | `*-*-* 03:00:00 Europe/Rome` — daily at 03:00 | Pull and verify the prepared Prometheus backup |
|
||||||
|
|
||||||
|
`atlas-usb-backup.service` has **no timer**: the encrypted USB backup must be started manually.
|
||||||
|
The vendor's `zfs-scrub-weekly@zpool.timer` is intentionally disabled in favor of the monthly scrub.
|
||||||
|
The Prometheus export timer runs at 02:00 Europe/Rome. Its first scheduled export and Atlas pull
|
||||||
|
passed on 2026-10-01; a manual post-NPM-Quadlet export, pull, and temporary restore passed on
|
||||||
|
2026-10-03. The first scheduled cycle after that cutover remains to be observed. While a
|
||||||
|
Borg backup is still running, `systemctl list-timers` may show `-` for its next trigger; this does not
|
||||||
|
mean the timer has been disabled. Inspect the current schedule on Atlas with
|
||||||
|
`systemctl list-timers --all`.
|
||||||
|
|
||||||
|
A temporary Nextcloud deployment on Atlas is also planned before Uranus: it requires separately
|
||||||
|
declared persistent application, database, and cache storage, Vault-backed credentials, NPM-only
|
||||||
|
publishing through Aegis, and defined backup, upgrade, and eventual migration procedures. Do not deploy
|
||||||
|
it before the data-protection checklist is complete.
|
||||||
|
|
||||||
|
Atlas is the declared iCloud photo-ingestion host. Ansible manages the rootless Quadlet, a private
|
||||||
|
Vault-backed `icloudpd.conf`, photos under `/zpool/archive/Pictures/iCloudPD`, and separate state in
|
||||||
|
`zpool/services/data/icloudpd`. The service was started manually; Ansible does not enable automatic
|
||||||
|
startup or manage the password and MFA keyring. The operator initialized MFA interactively; on
|
||||||
|
2026-10-03 the initial photo/video download completed. Aegis iCloudPD, including its service data,
|
||||||
|
has been removed and verified; the Aegis role no longer manages it. Backup/restore and SMB access
|
||||||
|
for the new data remain unverified. The Photobook NFS export remains untouched. See
|
||||||
|
[`docs/atlas-icloudpd-migration.md`](docs/atlas-icloudpd-migration.md).
|
||||||
|
|
||||||
|
The first scheduled Prometheus backup runs and production-size disaster-recovery tests remain follow-up work. The prioritized
|
||||||
|
operational backlog is kept in `AGENTS.md`.
|
||||||
|
|
||||||
|
Priority 2 procedures and decisions are recorded in
|
||||||
|
[`docs/atlas-recovery.md`](docs/atlas-recovery.md),
|
||||||
|
[`docs/atlas-updates.md`](docs/atlas-updates.md), and
|
||||||
|
[`docs/atlas-sharing-decision.md`](docs/atlas-sharing-decision.md).
|
||||||
|
The provisional Atlas recovery objectives are RPO 24 hours and RTO 72 hours;
|
||||||
|
an isolated small-VM OS rebuild, pool import, Ansible reapplication, and
|
||||||
|
snapshot restore passed, but full-size recovery time is unmeasured. `Archive` (SMB) and
|
||||||
|
`photobook` (NFS) remain deliberately separate.
|
||||||
|
The Prometheus pull architecture and manual export/pull/restore evidence are in
|
||||||
|
[`docs/prometheus-backup.md`](docs/prometheus-backup.md). Both daily timers are
|
||||||
|
enabled; their first scheduled runs remain to be verified.
|
||||||
|
|
||||||
## How layering works
|
## How layering works
|
||||||
|
|
||||||
@@ -472,8 +713,9 @@ ansible-playbook ansible/site.yml --limit <host> --tags <tag1>,<tag2> --check --
|
|||||||
ansible-playbook ansible/site.yml --limit <host> --start-at-task "<task name>" --check --diff
|
ansible-playbook ansible/site.yml --limit <host> --start-at-task "<task name>" --check --diff
|
||||||
ansible-lint ansible/roles/<role>
|
ansible-lint ansible/roles/<role>
|
||||||
yamllint ansible/path/to/file.yml
|
yamllint ansible/path/to/file.yml
|
||||||
podman-compose -f /opt/docker/server/docker-compose.yml config
|
ansible-playbook ansible/site.yml --limit prometheus --tags server_legacy_cleanup --check --diff -e server_legacy_cleanup=true
|
||||||
ansible-playbook ansible/site.yml --limit atlas --tags storage,sharing,containers --check --diff
|
ansible-playbook ansible/site.yml --limit atlas --tags storage,sharing,containers --check --diff
|
||||||
|
ansible-playbook ansible/site.yml --limit atlas --tags music_sync --check --diff
|
||||||
```
|
```
|
||||||
|
|
||||||
## Tags
|
## Tags
|
||||||
|
|||||||
@@ -9,6 +9,7 @@ fedora_packages_base:
|
|||||||
- jq
|
- jq
|
||||||
- make
|
- make
|
||||||
- nodejs
|
- nodejs
|
||||||
|
- openssl
|
||||||
- ripgrep
|
- ripgrep
|
||||||
|
|
||||||
fedora_manage_docker_repo: true
|
fedora_manage_docker_repo: true
|
||||||
|
|||||||
@@ -6,6 +6,10 @@ effective_username: "{{ server_username }}"
|
|||||||
effective_user_group: "{{ server_user_group }}"
|
effective_user_group: "{{ server_user_group }}"
|
||||||
effective_user_home: "{{ server_user_home }}"
|
effective_user_home: "{{ server_user_home }}"
|
||||||
server_container_stack_dir: /opt/docker/server
|
server_container_stack_dir: /opt/docker/server
|
||||||
|
server_npm_quadlet_stage: false
|
||||||
|
server_npm_quadlet_cutover: false
|
||||||
|
server_legacy_stack_retired: false
|
||||||
|
server_legacy_cleanup: false
|
||||||
ai_agents: {}
|
ai_agents: {}
|
||||||
vim_plugins_enabled: false
|
vim_plugins_enabled: false
|
||||||
|
|
||||||
@@ -80,5 +84,36 @@ server_sshd_settings:
|
|||||||
|
|
||||||
server_sshd_allow_users:
|
server_sshd_allow_users:
|
||||||
- "{{ server_username }}"
|
- "{{ server_username }}"
|
||||||
|
server_backup_export_enabled: false
|
||||||
|
server_backup_username: prometheus-backup
|
||||||
|
server_backup_public_key_name: atlas-pull
|
||||||
|
server_backup_export_root: /var/lib/prometheus-backup-export
|
||||||
|
server_backup_rrsync_path: /usr/share/doc/rsync/support/rrsync
|
||||||
|
server_backup_export_calendar: "*-*-* 02:00:00 Europe/Rome"
|
||||||
|
server_backup_export_start_timer: false
|
||||||
|
# Ongoing public Gitea proxy configuration.
|
||||||
|
server_gitea_proxy_enabled: false
|
||||||
|
server_gitea_on_atlas: false
|
||||||
|
server_gitea_atlas_address: "{{ hostvars['atlas'].ansible_host }}"
|
||||||
|
server_gitea_npm_domains: []
|
||||||
|
server_gitea_ssh_public_port: 2222
|
||||||
|
server_gitea_ssh_target_port: 2222
|
||||||
|
server_backup_export_source_keep: 3
|
||||||
|
server_backup_export_paths: >-
|
||||||
|
{{ ['opt/npm/data', 'opt/npm/letsencrypt']
|
||||||
|
+ ([] if server_gitea_on_atlas | bool else ['opt/gitea/data', 'home/git/.ssh'])
|
||||||
|
+ ([] if server_legacy_stack_retired | bool else
|
||||||
|
['opt/docker/server/docker-compose.yml',
|
||||||
|
'etc/systemd/system/podman-compose-server.service'])
|
||||||
|
+ (['etc/containers/systemd/prometheus-npm.container',
|
||||||
|
'etc/containers/systemd/server-web.network']
|
||||||
|
if server_npm_quadlet_stage | bool else [])
|
||||||
|
+ ['etc/ssh/sshd_config', 'etc/ssh/sshd_config.d',
|
||||||
|
'etc/firewalld', 'etc/wireguard/wg0.conf'] }}
|
||||||
|
server_backup_export_excludes: >-
|
||||||
|
{{ ['opt/npm/data/logs']
|
||||||
|
+ ([] if server_gitea_on_atlas | bool else
|
||||||
|
['opt/gitea/data/gitea/log', 'opt/gitea/data/gitea/tmp',
|
||||||
|
'opt/gitea/data/gitea/sessions', 'opt/gitea/data/gitea/indexers']) }}
|
||||||
server_ssh_authorized_keys: []
|
server_ssh_authorized_keys: []
|
||||||
server_ssh_authorized_key_directory: "{{ server_user_home }}/.ssh/authorized_keys.d"
|
server_ssh_authorized_key_directory: "{{ server_user_home }}/.ssh/authorized_keys.d"
|
||||||
|
|||||||
@@ -12,9 +12,10 @@ workstation_dev_wsl_packages:
|
|||||||
- python3-pip
|
- python3-pip
|
||||||
- tmux
|
- tmux
|
||||||
|
|
||||||
# Java 11 and Maven are managed by Mise on this Fedora WSL profile. Keep their
|
# Java 11, Java 25 and Maven are managed by Mise on this Fedora WSL profile.
|
||||||
# versions pinned; update them deliberately.
|
# Keep their versions pinned; update them deliberately.
|
||||||
workstation_mise_java_version: temurin-11.0.31+11
|
workstation_mise_java_version: temurin-11.0.31+11
|
||||||
|
workstation_mise_java_25_version: 25.0.2
|
||||||
workstation_mise_maven_version: 3.9.16
|
workstation_mise_maven_version: 3.9.16
|
||||||
|
|
||||||
workstation_is_wsl: true
|
workstation_is_wsl: true
|
||||||
|
|||||||
@@ -42,5 +42,3 @@ aegis_ssh_authorized_keys:
|
|||||||
key: "ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIEH/7GJfGt0ZVmKeEzceoFkFkeCXFryKK9vAbaip+HCx nymph"
|
key: "ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIEH/7GJfGt0ZVmKeEzceoFkFkeCXFryKK9vAbaip+HCx nymph"
|
||||||
- name: siren
|
- name: siren
|
||||||
key: "ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIA95wYlzpfN3rjUhpMeP4KHn8I6ZrjQXoDTgwgRIa++b siren"
|
key: "ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIA95wYlzpfN3rjUhpMeP4KHn8I6ZrjQXoDTgwgRIa++b siren"
|
||||||
|
|
||||||
aegis_icloudpd_apple_id: "{{ vault_aegis_icloudpd_apple_id | default('') }}"
|
|
||||||
|
|||||||
@@ -49,8 +49,145 @@ atlas_zfs_backup_reservation: 500G
|
|||||||
atlas_zfs_dataset_photobook: media/photobook
|
atlas_zfs_dataset_photobook: media/photobook
|
||||||
atlas_mount_root: /zpool
|
atlas_mount_root: /zpool
|
||||||
atlas_manage_storage: true
|
atlas_manage_storage: true
|
||||||
|
atlas_manage_nextcloud: true
|
||||||
|
# Two local consistent bundles; long-term history stays in Borg/USB and ZFS.
|
||||||
|
atlas_nextcloud_backup_keep: 2
|
||||||
|
atlas_nextcloud_domain: cloud.fscotto.co
|
||||||
|
atlas_onlyoffice_domain: office.fscotto.co
|
||||||
|
# Resolved official amd64 images on 2026-10-03; updates are deliberate.
|
||||||
|
atlas_nextcloud_image: docker.io/library/nextcloud:33.0.9-apache@sha256:a97666d6ae931bde78a80cfba8abdf46d436d7b540f31895803f6fb0a012d689
|
||||||
|
atlas_nextcloud_postgres_image: docker.io/library/postgres:17-bookworm@sha256:639ab7ceb90e13123085b741fb31ef493fba25463002f6da665352e7b534b652
|
||||||
|
atlas_nextcloud_redis_image: docker.io/library/redis:7.4-bookworm@sha256:c6eabf748fc7a61dbb5a705c78bcf3d6377b1127a97d0ce965c11c44ba46896f
|
||||||
|
atlas_onlyoffice_image: docker.io/onlyoffice/documentserver:9.4.0.1@sha256:3ab6ebc7c605e5a32b7ae3ff19daed4925090245acc8100ce2230bd766c88212
|
||||||
|
atlas_nextcloud_users:
|
||||||
|
- username: fabio
|
||||||
|
display_name: Fabio
|
||||||
|
password: "{{ vault_nextcloud_fabio_password }}"
|
||||||
|
- username: chiara
|
||||||
|
display_name: Chiara
|
||||||
|
password: "{{ vault_nextcloud_chiara_password }}"
|
||||||
|
atlas_nextcloud_external_mounts:
|
||||||
|
- name: Documenti
|
||||||
|
user: fabio
|
||||||
|
source: /zpool/archive/Documents
|
||||||
|
target: /mnt/archive-documents
|
||||||
|
readonly: false
|
||||||
|
- name: Foto iCloud
|
||||||
|
user: fabio
|
||||||
|
source: /zpool/archive/Pictures/iCloudPD
|
||||||
|
target: /mnt/archive-icloud
|
||||||
|
readonly: true
|
||||||
|
atlas_nextcloud_apps:
|
||||||
|
- id: groupfolders
|
||||||
|
version: 21.0.9
|
||||||
|
url: https://github.com/nextcloud-releases/groupfolders/releases/download/v21.0.9/groupfolders-v21.0.9.tar.gz
|
||||||
|
checksum: sha256:d8b95f0778425f646f2311ba5b42d8e2fcfdf37dc2fd35fcac8d3f01bde38a21
|
||||||
|
- id: onlyoffice
|
||||||
|
version: 10.2.1
|
||||||
|
url: https://github.com/ONLYOFFICE/onlyoffice-nextcloud/releases/download/v10.2.1/onlyoffice.tar.gz
|
||||||
|
checksum: sha256:144998af0610ccd17ee8d7025e2f8001472da03f6dab90ff38039247825e3a9b
|
||||||
|
- id: contacts
|
||||||
|
version: 8.9.1
|
||||||
|
url: https://github.com/nextcloud-releases/contacts/releases/download/v8.9.1/contacts-v8.9.1.tar.gz
|
||||||
|
checksum: sha256:a25cdf448b192631b8e8eb7addc31b40382b33871b10521f4308ac5a6e0457bf
|
||||||
|
- id: calendar
|
||||||
|
version: 6.6.2
|
||||||
|
url: https://github.com/nextcloud-releases/calendar/releases/download/v6.6.2/calendar-v6.6.2.tar.gz
|
||||||
|
checksum: sha256:7e83632d4436d3037a34d1c73cbc06d5ccb6e8fc10f096a86a43a0515588529c
|
||||||
|
# Rootless Gitea was restored from the stopped-source export before production activation.
|
||||||
|
atlas_manage_gitea: true
|
||||||
|
atlas_gitea_production_enabled: true
|
||||||
|
atlas_gitea_public_domain: git.fscotto.co
|
||||||
|
atlas_prometheus_pull_start_timer: true
|
||||||
|
atlas_manage_zfs_snapshots: true
|
||||||
|
atlas_zfs_snapshot_prefix: atlas-auto
|
||||||
|
atlas_zfs_snapshot_policies:
|
||||||
|
- name: hourly
|
||||||
|
calendar: "*-*-* *:05:00"
|
||||||
|
keep: 24
|
||||||
|
- name: daily
|
||||||
|
calendar: "*-*-* 00:15:00"
|
||||||
|
keep: 30
|
||||||
|
- name: weekly
|
||||||
|
calendar: "Sun *-*-* 01:00:00"
|
||||||
|
keep: 8
|
||||||
|
- name: monthly
|
||||||
|
calendar: "*-*-01 02:00:00"
|
||||||
|
keep: 12
|
||||||
|
atlas_manage_zfs_scrub: true
|
||||||
|
atlas_zfs_scrub_calendar: "Sun *-*-01..07 03:00:00"
|
||||||
|
atlas_manage_borg_backup: true
|
||||||
|
atlas_borg_repository_host: u660064-sub1.your-storagebox.de
|
||||||
|
atlas_borg_repository_user: u660064-sub1
|
||||||
|
atlas_borg_repository_port: 23
|
||||||
|
atlas_borg_repository_path: ./borg-data
|
||||||
|
atlas_borg_remote_path: borg-1.4
|
||||||
|
# Verified against Hetzner's published ED25519 fingerprint on 2026-09-17:
|
||||||
|
# SHA256:XqONwb1S0zuj5A1CDxpOSuD2hnAArV1A3wKY7Z3sdgM
|
||||||
|
atlas_borg_host_key: >-
|
||||||
|
[u660064-sub1.your-storagebox.de]:23 ssh-ed25519
|
||||||
|
AAAAC3NzaC1lZDI1NTE5AAAAIICf9svRenC/PLKIL9nk6K/pxQgoiFC41wTNvoIncOxs
|
||||||
|
atlas_borg_backup_calendar: "*-*-* 04:30:00"
|
||||||
|
atlas_borg_check_calendar: "*-*-15 06:00:00"
|
||||||
|
atlas_borg_randomized_delay: 30m
|
||||||
|
atlas_borg_keep_daily: 30
|
||||||
|
atlas_borg_keep_weekly: 8
|
||||||
|
atlas_borg_keep_monthly: 12
|
||||||
|
atlas_manage_usb_backup: true
|
||||||
|
# Read-only lsblk verification on Atlas, 2026-09-23. Never store the LUKS password here.
|
||||||
|
atlas_usb_backup_luks_uuid: 577b3c43-ea37-4611-81a9-39d555cdfbd4
|
||||||
|
atlas_usb_backup_fs_uuid: 758e2d2e-a427-4797-aad9-39c3a9f17c7e
|
||||||
|
atlas_usb_backup_mapper_name: zpool-backup
|
||||||
|
atlas_manage_usb_reminder: true
|
||||||
|
atlas_usb_reminder_calendar: "Sat *-*-01..07 10:00:00 Europe/Rome"
|
||||||
|
atlas_manage_monitoring: true
|
||||||
|
atlas_manage_prometheus_backup_pull: true
|
||||||
|
# Prometheus ED25519 host key read through the controller's strict SSH trust on 2026-09-30.
|
||||||
|
# Fingerprint: SHA256:rfedk7DHI9mLB3UHk/4F3HHlSIiswtCAFsAXvfh6iXk
|
||||||
|
atlas_prometheus_ssh_host_key: >-
|
||||||
|
179.237.102.172 ssh-ed25519
|
||||||
|
AAAAC3NzaC1lZDI1NTE5AAAAIC4b+QXlPupoEx71W9NKs9tTeYjBqTkVMqbGB97nMNWv
|
||||||
|
# Physical pool disks and the system NVMe; the disconnected USB disk is intentionally excluded.
|
||||||
|
atlas_monitor_smart_devices:
|
||||||
|
- { name: pool-1, path: "{{ atlas_zpool_disks[0] }}", warning_c: 50, critical_c: 55 }
|
||||||
|
- { name: pool-2, path: "{{ atlas_zpool_disks[1] }}", warning_c: 50, critical_c: 55 }
|
||||||
|
- { name: pool-3, path: "{{ atlas_zpool_disks[2] }}", warning_c: 50, critical_c: 55 }
|
||||||
|
- { name: pool-4, path: "{{ atlas_zpool_disks[3] }}", warning_c: 50, critical_c: 55 }
|
||||||
|
- name: system-nvme
|
||||||
|
path: /dev/disk/by-id/nvme-Patriot_M.2_P320_256GB_P320ADB26011606111
|
||||||
|
warning_c: 70
|
||||||
|
critical_c: 85
|
||||||
|
atlas_monitor_timers:
|
||||||
|
- { name: atlas-zfs-snapshot-hourly.timer, max_age_hours: 3 }
|
||||||
|
- { name: atlas-zfs-snapshot-daily.timer, max_age_hours: 36 }
|
||||||
|
- { name: atlas-zfs-snapshot-weekly.timer, max_age_hours: 216 }
|
||||||
|
- { name: atlas-zfs-snapshot-monthly.timer, max_age_hours: 960 }
|
||||||
|
- { name: zfs-scrub-monthly@zpool.timer, max_age_hours: 960 }
|
||||||
|
- { name: atlas-borg-backup.timer, max_age_hours: 48 }
|
||||||
|
- { name: atlas-borg-check.timer, max_age_hours: 960 }
|
||||||
|
# The first manual USB reminder is not due until October; activation is checked, not age.
|
||||||
|
- { name: atlas-usb-reminder.timer, max_age_hours: 0 }
|
||||||
|
atlas_monitor_failure_units:
|
||||||
|
- atlas-zfs-snapshot@.service
|
||||||
|
- zfs-scrub@zpool.service
|
||||||
|
- atlas-borg-backup.service
|
||||||
|
- atlas-borg-check.service
|
||||||
|
- atlas-usb-backup.service
|
||||||
|
- atlas-usb-reminder.service
|
||||||
|
- atlas-health-monitor.service
|
||||||
|
atlas_monitor_remote_capacity:
|
||||||
|
user: "{{ atlas_borg_repository_user }}"
|
||||||
|
host: "{{ atlas_borg_repository_host }}"
|
||||||
|
run_as: "{{ atlas_borg_username }}"
|
||||||
|
ssh_wrapper: "{{ atlas_borg_ssh_wrapper_path }}"
|
||||||
|
warning_percent: 80
|
||||||
|
critical_percent: 90
|
||||||
|
growth_warning_gib_day: 500
|
||||||
atlas_manage_sharing: true
|
atlas_manage_sharing: true
|
||||||
atlas_manage_media_stack: false
|
atlas_manage_media_stack: false
|
||||||
|
# Planned after data-protection validation: move iCloudPD photo ingestion from
|
||||||
|
# Aegis to Atlas, with photos under /zpool/archive/Pictures/iCloudPD and
|
||||||
|
# application/MFA state in a separate dataset. Do not deploy or cut over yet.
|
||||||
|
|
||||||
# WireGuard is retired on Atlas. These rootless services are a temporary home
|
# WireGuard is retired on Atlas. These rootless services are a temporary home
|
||||||
# until Uranus replaces them.
|
# until Uranus replaces them.
|
||||||
@@ -60,6 +197,7 @@ backend_phase1_bind_address: "{{ ansible_host }}"
|
|||||||
backend_phase1_firewalld_zone: "{{ atlas_firewalld_zone }}"
|
backend_phase1_firewalld_zone: "{{ atlas_firewalld_zone }}"
|
||||||
backend_phase1_npm_source_ip: "{{ atlas_aegis_ip }}"
|
backend_phase1_npm_source_ip: "{{ atlas_aegis_ip }}"
|
||||||
backend_phase1_syncthing_native_subnet: "{{ atlas_lan_subnet }}"
|
backend_phase1_syncthing_native_subnet: "{{ atlas_lan_subnet }}"
|
||||||
|
backend_phase1_music_sync_enabled: true
|
||||||
|
|
||||||
rocky_manage_openzfs_repo: true
|
rocky_manage_openzfs_repo: true
|
||||||
rocky_manage_syncthing_binary: false
|
rocky_manage_syncthing_binary: false
|
||||||
@@ -69,13 +207,21 @@ rocky_podman_packages:
|
|||||||
|
|
||||||
host_packages:
|
host_packages:
|
||||||
- cockpit
|
- cockpit
|
||||||
|
- cockpit-podman
|
||||||
|
- cockpit-storaged
|
||||||
|
- realmd
|
||||||
|
- pcp
|
||||||
|
- python3-pcp
|
||||||
|
- cryptsetup
|
||||||
- nfs-utils
|
- nfs-utils
|
||||||
- policycoreutils
|
- policycoreutils
|
||||||
- policycoreutils-python-utils
|
- policycoreutils-python-utils
|
||||||
- python3-libselinux
|
- python3-libselinux
|
||||||
|
- setroubleshoot-server
|
||||||
- samba
|
- samba
|
||||||
- samba-client
|
- samba-client
|
||||||
- samba-common-tools
|
- samba-common-tools
|
||||||
|
- borgbackup
|
||||||
- zfs
|
- zfs
|
||||||
|
|
||||||
atlas_nfs_exports:
|
atlas_nfs_exports:
|
||||||
@@ -109,4 +255,5 @@ atlas_firewalld_rich_rules:
|
|||||||
host_enabled_services:
|
host_enabled_services:
|
||||||
- sshd
|
- sshd
|
||||||
- cockpit.socket
|
- cockpit.socket
|
||||||
|
- pmlogger.service
|
||||||
- zfs.target
|
- zfs.target
|
||||||
|
|||||||
@@ -6,7 +6,26 @@ ansible_port: 22
|
|||||||
ansible_ssh_private_key_file: /home/fscotto/.ssh/id_ed25519
|
ansible_ssh_private_key_file: /home/fscotto/.ssh/id_ed25519
|
||||||
|
|
||||||
server_username: rocky
|
server_username: rocky
|
||||||
server_duckdns_domain: fscotto
|
server_legacy_stack_retired: true
|
||||||
|
# Destructive deletion runs only with an explicit extra-var and cleanup tag.
|
||||||
|
server_legacy_cleanup: false
|
||||||
|
# Explicit opt-in cleanup; no data, volumes, networks or NPM images are removed.
|
||||||
|
server_legacy_image_cleanup: false
|
||||||
|
server_legacy_images:
|
||||||
|
- docker.gitea.com/gitea:1.25.2
|
||||||
|
- docker.io/deluan/navidrome:latest
|
||||||
|
- docker.io/library/postgres:13
|
||||||
|
server_npm_quadlet_stage: true
|
||||||
|
server_npm_quadlet_image: docker.io/jc21/nginx-proxy-manager@sha256:52b2c59994f3d36acfcf70a1626f29734df0ed8c71bacc0269f78b6f939858bb
|
||||||
|
# The stopped-source export and live Quadlet cutover passed on 2026-10-03.
|
||||||
|
server_npm_quadlet_cutover: true
|
||||||
|
server_backup_export_enabled: true
|
||||||
|
server_backup_export_start_timer: true
|
||||||
|
# Install the final-copy helper only; it is never run by a normal playbook invocation.
|
||||||
|
server_gitea_proxy_enabled: true
|
||||||
|
server_gitea_on_atlas: true
|
||||||
|
server_gitea_npm_domains:
|
||||||
|
- git.fscotto.duckdns.org
|
||||||
server_ssh_authorized_keys:
|
server_ssh_authorized_keys:
|
||||||
- name: ikaros
|
- name: ikaros
|
||||||
key: "ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAINrIxXjA3ffPwziKGR5gzc4gAoBehQPlnEMcXF4Wl0ZS ikaros"
|
key: "ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAINrIxXjA3ffPwziKGR5gzc4gAoBehQPlnEMcXF4Wl0ZS ikaros"
|
||||||
@@ -32,6 +51,12 @@ host_packages:
|
|||||||
- cockpit
|
- cockpit
|
||||||
- cockpit-navigator
|
- cockpit-navigator
|
||||||
- cockpit-podman
|
- cockpit-podman
|
||||||
|
- cockpit-storaged
|
||||||
|
- realmd
|
||||||
|
- pcp
|
||||||
|
- python3-pcp
|
||||||
|
- setroubleshoot-server
|
||||||
|
|
||||||
host_enabled_services:
|
host_enabled_services:
|
||||||
- cockpit.socket
|
- cockpit.socket
|
||||||
|
- pmlogger.service
|
||||||
|
|||||||
@@ -39,11 +39,41 @@
|
|||||||
state: enabled
|
state: enabled
|
||||||
when: "'workstation_dev_wsl' in group_names"
|
when: "'workstation_dev_wsl' in group_names"
|
||||||
|
|
||||||
|
- name: Install distribution signing keys for Fedora desktop codecs
|
||||||
|
tags: [packages, heic]
|
||||||
|
ansible.builtin.dnf:
|
||||||
|
name: distribution-gpg-keys
|
||||||
|
state: present
|
||||||
|
when: "'graphical_desktop' in group_names"
|
||||||
|
|
||||||
|
- name: Import RPM Fusion Free signing key for Fedora desktop codecs
|
||||||
|
tags: [packages, heic]
|
||||||
|
ansible.builtin.rpm_key:
|
||||||
|
key: /usr/share/distribution-gpg-keys/rpmfusion/RPM-GPG-KEY-rpmfusion-free-fedora-2020
|
||||||
|
fingerprint: E9A491A3DE247814E7E067EAE06F8ECDD651FF2E
|
||||||
|
state: present
|
||||||
|
when: "'graphical_desktop' in group_names"
|
||||||
|
|
||||||
|
- name: Enable RPM Fusion Free for Fedora desktop codecs
|
||||||
|
tags: [packages, heic]
|
||||||
|
ansible.builtin.dnf:
|
||||||
|
name: "https://download1.rpmfusion.org/free/fedora/rpmfusion-free-release-{{ ansible_facts['distribution_major_version'] }}.noarch.rpm"
|
||||||
|
state: present
|
||||||
|
when: "'graphical_desktop' in group_names"
|
||||||
|
|
||||||
- name: Refresh dnf package metadata
|
- name: Refresh dnf package metadata
|
||||||
tags: [packages]
|
tags: [packages]
|
||||||
ansible.builtin.dnf:
|
ansible.builtin.dnf:
|
||||||
update_cache: true
|
update_cache: true
|
||||||
|
|
||||||
|
- name: Install HEIC decoder on Fedora desktops
|
||||||
|
tags: [packages, heic]
|
||||||
|
ansible.builtin.dnf:
|
||||||
|
name: libheif-freeworld
|
||||||
|
state: present
|
||||||
|
update_cache: true
|
||||||
|
when: "'graphical_desktop' in group_names"
|
||||||
|
|
||||||
- name: Install packages on Fedora
|
- name: Install packages on Fedora
|
||||||
tags: [packages]
|
tags: [packages]
|
||||||
ansible.builtin.dnf:
|
ansible.builtin.dnf:
|
||||||
|
|||||||
@@ -8,10 +8,6 @@ aegis_network_connection_uuid: ""
|
|||||||
aegis_host_dns_servers: []
|
aegis_host_dns_servers: []
|
||||||
aegis_host_dns_search_domains: []
|
aegis_host_dns_search_domains: []
|
||||||
aegis_adguard_image: docker.io/adguard/adguardhome:latest
|
aegis_adguard_image: docker.io/adguard/adguardhome:latest
|
||||||
aegis_icloudpd_image: docker.io/boredazfcuk/icloudpd:latest
|
|
||||||
aegis_icloudpd_folder_structure: '{:%Y/%m/%d}'
|
|
||||||
aegis_icloudpd_synchronisation_interval: 86400
|
|
||||||
aegis_icloudpd_apple_id: ""
|
|
||||||
aegis_ikaros_mac_address: aa:bb:cc:dd:ee:ff
|
aegis_ikaros_mac_address: aa:bb:cc:dd:ee:ff
|
||||||
aegis_wol_port: 9
|
aegis_wol_port: 9
|
||||||
|
|
||||||
|
|||||||
@@ -9,13 +9,12 @@
|
|||||||
name: sshd.service
|
name: sshd.service
|
||||||
state: reloaded
|
state: reloaded
|
||||||
|
|
||||||
- name: Restart Aegis Quadlet services
|
- name: Restart Aegis AdGuard Quadlet
|
||||||
ansible.builtin.systemd:
|
ansible.builtin.systemd:
|
||||||
name: "{{ item }}"
|
name: "{{ item }}"
|
||||||
state: restarted
|
state: restarted
|
||||||
daemon_reload: true
|
daemon_reload: true
|
||||||
loop:
|
loop:
|
||||||
- adguardhome.service
|
- adguardhome.service
|
||||||
- icloudpd.service
|
|
||||||
loop_control:
|
loop_control:
|
||||||
label: "{{ item }}"
|
label: "{{ item }}"
|
||||||
|
|||||||
@@ -13,14 +13,6 @@
|
|||||||
msg: Reboot Aegis to activate the newly layered packages, then rerun the playbook.
|
msg: Reboot Aegis to activate the newly layered packages, then rerun the playbook.
|
||||||
when: aegis_layered_packages_result.needs_reboot | default(false)
|
when: aegis_layered_packages_result.needs_reboot | default(false)
|
||||||
|
|
||||||
- name: Require Aegis iCloudPD Apple ID
|
|
||||||
tags: [aegis, icloudpd]
|
|
||||||
ansible.builtin.assert:
|
|
||||||
that:
|
|
||||||
- aegis_icloudpd_apple_id | length > 0
|
|
||||||
fail_msg: Define vault_aegis_icloudpd_apple_id before applying the Aegis profile.
|
|
||||||
no_log: true
|
|
||||||
|
|
||||||
- name: Require completed Aegis network placeholders
|
- name: Require completed Aegis network placeholders
|
||||||
tags: [aegis, dns, firewall, network, services]
|
tags: [aegis, dns, firewall, network, services]
|
||||||
ansible.builtin.assert:
|
ansible.builtin.assert:
|
||||||
@@ -119,8 +111,6 @@
|
|||||||
loop:
|
loop:
|
||||||
- /var/lib/adguard/work
|
- /var/lib/adguard/work
|
||||||
- /var/lib/adguard/conf
|
- /var/lib/adguard/conf
|
||||||
- /var/lib/icloudpd/data
|
|
||||||
- /var/lib/icloudpd/config
|
|
||||||
|
|
||||||
- name: Create Quadlet configuration directory
|
- name: Create Quadlet configuration directory
|
||||||
tags: [aegis, containers]
|
tags: [aegis, containers]
|
||||||
@@ -142,12 +132,9 @@
|
|||||||
loop:
|
loop:
|
||||||
- src: adguardhome.container.j2
|
- src: adguardhome.container.j2
|
||||||
dest: adguardhome.container
|
dest: adguardhome.container
|
||||||
- src: icloudpd.container.j2
|
|
||||||
dest: icloudpd.container
|
|
||||||
loop_control:
|
loop_control:
|
||||||
label: "{{ item.dest }}"
|
label: "{{ item.dest }}"
|
||||||
no_log: "{{ item.dest == 'icloudpd.container' }}"
|
notify: Restart Aegis AdGuard Quadlet
|
||||||
notify: Restart Aegis Quadlet services
|
|
||||||
|
|
||||||
- name: Create Aegis systemd-resolved configuration directory
|
- name: Create Aegis systemd-resolved configuration directory
|
||||||
tags: [aegis, adguard, dns, services]
|
tags: [aegis, adguard, dns, services]
|
||||||
@@ -168,7 +155,7 @@
|
|||||||
mode: "0644"
|
mode: "0644"
|
||||||
notify:
|
notify:
|
||||||
- Restart Aegis systemd-resolved
|
- Restart Aegis systemd-resolved
|
||||||
- Restart Aegis Quadlet services
|
- Restart Aegis AdGuard Quadlet
|
||||||
|
|
||||||
- name: Point Aegis resolver at the full systemd-resolved configuration
|
- name: Point Aegis resolver at the full systemd-resolved configuration
|
||||||
tags: [aegis, adguard, dns, services]
|
tags: [aegis, adguard, dns, services]
|
||||||
@@ -361,7 +348,7 @@
|
|||||||
group: root
|
group: root
|
||||||
mode: "0755"
|
mode: "0755"
|
||||||
|
|
||||||
- name: Enable Aegis Quadlet services and automatic updates
|
- name: Enable Aegis AdGuard Quadlet and automatic updates
|
||||||
tags: [aegis, containers, services]
|
tags: [aegis, containers, services]
|
||||||
ansible.builtin.systemd:
|
ansible.builtin.systemd:
|
||||||
name: "{{ item }}"
|
name: "{{ item }}"
|
||||||
@@ -370,7 +357,6 @@
|
|||||||
daemon_reload: true
|
daemon_reload: true
|
||||||
loop:
|
loop:
|
||||||
- adguardhome.service
|
- adguardhome.service
|
||||||
- icloudpd.service
|
|
||||||
- podman-auto-update.timer
|
- podman-auto-update.timer
|
||||||
loop_control:
|
loop_control:
|
||||||
label: "{{ item }}"
|
label: "{{ item }}"
|
||||||
|
|||||||
@@ -1,20 +0,0 @@
|
|||||||
# Managed by Ansible. Do not edit manually.
|
|
||||||
[Unit]
|
|
||||||
Description=iCloud Photos Downloader
|
|
||||||
Wants=network-online.target
|
|
||||||
After=network-online.target
|
|
||||||
|
|
||||||
[Container]
|
|
||||||
Image={{ aegis_icloudpd_image }}
|
|
||||||
Environment=apple_id={{ aegis_icloudpd_apple_id }}
|
|
||||||
Environment=folder_structure={{ aegis_icloudpd_folder_structure }}
|
|
||||||
Environment=synchronisation_interval={{ aegis_icloudpd_synchronisation_interval }}
|
|
||||||
Volume=/var/lib/icloudpd/data:/home/root/iCloud:Z
|
|
||||||
Volume=/var/lib/icloudpd/config:/config:Z
|
|
||||||
AutoUpdate=registry
|
|
||||||
|
|
||||||
[Service]
|
|
||||||
Restart=always
|
|
||||||
|
|
||||||
[Install]
|
|
||||||
WantedBy=multi-user.target
|
|
||||||
@@ -1,5 +1,33 @@
|
|||||||
---
|
---
|
||||||
atlas_manage_storage: false
|
atlas_manage_storage: false
|
||||||
|
atlas_manage_nextcloud: false
|
||||||
|
atlas_nextcloud_root: "{{ atlas_app_data_mountpoint }}/nextcloud"
|
||||||
|
atlas_nextcloud_dataset: "{{ atlas_zfs_pool }}/services/data/nextcloud"
|
||||||
|
atlas_nextcloud_domain: ""
|
||||||
|
atlas_onlyoffice_domain: ""
|
||||||
|
atlas_nextcloud_http_port: 8080
|
||||||
|
atlas_onlyoffice_http_port: 8081
|
||||||
|
atlas_nextcloud_network_subnet: 10.90.10.0/24
|
||||||
|
atlas_nextcloud_network_gateway: 10.90.10.1
|
||||||
|
atlas_nextcloud_quadlet_dir: "{{ atlas_admin_home }}/.config/containers/systemd"
|
||||||
|
atlas_nextcloud_private_dir: "{{ atlas_admin_home }}/.config/atlas-nextcloud"
|
||||||
|
atlas_nextcloud_app_cache: "{{ atlas_admin_home }}/.cache/atlas-nextcloud-apps"
|
||||||
|
atlas_nextcloud_image: ""
|
||||||
|
atlas_nextcloud_postgres_image: ""
|
||||||
|
atlas_nextcloud_redis_image: ""
|
||||||
|
atlas_onlyoffice_image: ""
|
||||||
|
atlas_nextcloud_backup_root: "{{ atlas_mount_root }}/backup/nextcloud"
|
||||||
|
atlas_nextcloud_backup_keep: 2
|
||||||
|
atlas_nextcloud_admin: admin
|
||||||
|
atlas_nextcloud_users: []
|
||||||
|
atlas_nextcloud_apps: []
|
||||||
|
# Existing Archive directories; never import into the internal data namespace.
|
||||||
|
atlas_nextcloud_external_mounts: []
|
||||||
|
atlas_nextcloud_services:
|
||||||
|
- atlas-nextcloud-db.service
|
||||||
|
- atlas-nextcloud-redis.service
|
||||||
|
- atlas-nextcloud.service
|
||||||
|
- atlas-onlyoffice.service
|
||||||
atlas_manage_sharing: false
|
atlas_manage_sharing: false
|
||||||
# Destructive first-boot action; normally false once the pool exists.
|
# Destructive first-boot action; normally false once the pool exists.
|
||||||
atlas_create_pool: false
|
atlas_create_pool: false
|
||||||
@@ -61,6 +89,87 @@ atlas_zfs_backup_reservation: 500G
|
|||||||
atlas_zfs_dataset_photobook: media/photobook
|
atlas_zfs_dataset_photobook: media/photobook
|
||||||
atlas_mount_root: /CHANGEME_ATLAS_MOUNT_ROOT
|
atlas_mount_root: /CHANGEME_ATLAS_MOUNT_ROOT
|
||||||
|
|
||||||
|
atlas_manage_zfs_snapshots: false
|
||||||
|
atlas_zfs_snapshot_prefix: atlas-auto
|
||||||
|
atlas_zfs_snapshot_policies: []
|
||||||
|
atlas_manage_zfs_scrub: false
|
||||||
|
atlas_zfs_scrub_calendar: ""
|
||||||
|
|
||||||
|
atlas_manage_borg_backup: false
|
||||||
|
atlas_borg_username: borg
|
||||||
|
atlas_borg_group: borg
|
||||||
|
atlas_borg_home: /var/lib/atlas-borg
|
||||||
|
atlas_borg_repository_host: CHANGEME_BORG_HOST
|
||||||
|
atlas_borg_repository_user: CHANGEME_BORG_USER
|
||||||
|
atlas_borg_repository_port: 23
|
||||||
|
atlas_borg_repository_path: ./borg-data
|
||||||
|
atlas_borg_remote_path: borg-1.4
|
||||||
|
atlas_borg_host_key: ""
|
||||||
|
atlas_borg_ssh_private_key_path: /etc/atlas-borg/id_ed25519
|
||||||
|
atlas_borg_known_hosts_path: /etc/atlas-borg/known_hosts
|
||||||
|
atlas_borg_passphrase_path: /etc/atlas-borg/passphrase
|
||||||
|
atlas_borg_ssh_wrapper_path: /usr/local/libexec/atlas-borg-ssh
|
||||||
|
atlas_borg_passphrase: "{{ vault_atlas_borg_passphrase | default('') }}"
|
||||||
|
atlas_borg_encryption_mode: repokey
|
||||||
|
atlas_borg_archive_prefix: atlas
|
||||||
|
atlas_borg_snapshot_prefix: atlas-borg
|
||||||
|
atlas_borg_compression: auto,zstd,3
|
||||||
|
atlas_borg_backup_calendar: ""
|
||||||
|
atlas_borg_check_calendar: ""
|
||||||
|
atlas_borg_randomized_delay: 30m
|
||||||
|
atlas_borg_keep_daily: 30
|
||||||
|
atlas_borg_keep_weekly: 8
|
||||||
|
atlas_borg_keep_monthly: 12
|
||||||
|
atlas_borg_config_dir: /var/lib/atlas-borg
|
||||||
|
atlas_borg_cache_dir: /var/cache/atlas-borg
|
||||||
|
atlas_borg_lock_path: /var/lib/atlas-borg/backup.lock
|
||||||
|
atlas_borg_recovery_export_path: "{{ playbook_dir }}/../secrets/recovery/atlas-borg-repokey.export"
|
||||||
|
|
||||||
|
# Manual-only offline backup. No USB device is formatted or mounted by Ansible.
|
||||||
|
atlas_manage_usb_backup: false
|
||||||
|
atlas_usb_backup_luks_uuid: ""
|
||||||
|
atlas_usb_backup_fs_uuid: ""
|
||||||
|
atlas_usb_backup_mapper_name: atlas-usb-backup
|
||||||
|
atlas_usb_backup_min_free_bytes: 10737418240
|
||||||
|
atlas_usb_backup_snapshot_prefix: atlas-usb
|
||||||
|
atlas_manage_usb_reminder: false
|
||||||
|
atlas_usb_reminder_calendar: ""
|
||||||
|
atlas_usb_reminder_notifier: /opt/45drives/houston/houston-notify
|
||||||
|
|
||||||
|
# Read-only health probes and 45Drives Alerts; disabled outside Atlas host vars.
|
||||||
|
atlas_manage_monitoring: false
|
||||||
|
atlas_monitor_calendar: "*:0/30"
|
||||||
|
atlas_monitor_notifier: "{{ atlas_usb_reminder_notifier }}"
|
||||||
|
atlas_monitor_smart_devices: []
|
||||||
|
atlas_monitor_timers: []
|
||||||
|
atlas_monitor_failure_units: []
|
||||||
|
atlas_monitor_effective_timers: >-
|
||||||
|
{{ atlas_monitor_timers
|
||||||
|
+ ([{'name': 'atlas-prometheus-pull.timer', 'max_age_hours': 26}]
|
||||||
|
if atlas_manage_prometheus_backup_pull | bool and atlas_prometheus_pull_start_timer | bool
|
||||||
|
else []) }}
|
||||||
|
atlas_monitor_effective_failure_units: >-
|
||||||
|
{{ atlas_monitor_failure_units
|
||||||
|
+ (['atlas-prometheus-pull.service']
|
||||||
|
if atlas_manage_prometheus_backup_pull | bool else [])
|
||||||
|
+ (['atlas-nextcloud-backup.service', 'atlas-nextcloud-backup-recovery.service']
|
||||||
|
if atlas_manage_nextcloud | bool else []) }}
|
||||||
|
atlas_monitor_remote_capacity: {}
|
||||||
|
atlas_monitor_pool_warning_percent: 80
|
||||||
|
atlas_monitor_pool_critical_percent: 90
|
||||||
|
atlas_monitor_root_warning_percent: 80
|
||||||
|
atlas_monitor_root_critical_percent: 90
|
||||||
|
atlas_monitor_snapshot_warning_percent: 10
|
||||||
|
atlas_monitor_snapshot_critical_percent: 20
|
||||||
|
atlas_monitor_snapshot_growth_warning_gib_day: 100
|
||||||
|
atlas_monitor_backup_growth_warning_gib_day: 100
|
||||||
|
atlas_monitor_cpu_warning_c: 85
|
||||||
|
atlas_monitor_cpu_critical_c: 95
|
||||||
|
atlas_monitor_borg_max_runtime_days: 14
|
||||||
|
|
||||||
|
# Explicit post-restore relabeling only; never relabel datasets during ordinary runs.
|
||||||
|
atlas_restorecon_paths: []
|
||||||
|
|
||||||
atlas_archive_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_archive }}"
|
atlas_archive_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_archive }}"
|
||||||
atlas_services_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_services }}"
|
atlas_services_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_services }}"
|
||||||
atlas_app_data_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_app_data }}"
|
atlas_app_data_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_app_data }}"
|
||||||
@@ -71,8 +180,56 @@ atlas_music_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_music }}"
|
|||||||
atlas_backup_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_backup }}"
|
atlas_backup_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_backup }}"
|
||||||
atlas_host_backups_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_host_backups }}"
|
atlas_host_backups_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_host_backups }}"
|
||||||
atlas_backup_prometheus_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_backup_prometheus }}"
|
atlas_backup_prometheus_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_backup_prometheus }}"
|
||||||
|
atlas_manage_prometheus_backup_pull: false
|
||||||
|
atlas_prometheus_pull_ssh_dir: /etc/atlas-prometheus-pull
|
||||||
|
atlas_prometheus_pull_private_key_path: "{{ atlas_prometheus_pull_ssh_dir }}/id_ed25519"
|
||||||
|
atlas_prometheus_pull_known_hosts_path: "{{ atlas_prometheus_pull_ssh_dir }}/known_hosts"
|
||||||
|
atlas_prometheus_ssh_host_key: ""
|
||||||
|
atlas_prometheus_pull_source_user: prometheus-backup
|
||||||
|
atlas_prometheus_pull_source_port: 22
|
||||||
|
atlas_prometheus_pull_calendar: "*-*-* 03:00:00 Europe/Rome"
|
||||||
|
atlas_prometheus_pull_start_timer: false
|
||||||
|
atlas_prometheus_pull_keep_daily: 30
|
||||||
|
atlas_prometheus_pull_keep_weekly: 8
|
||||||
|
atlas_prometheus_pull_keep_monthly: 12
|
||||||
|
atlas_prometheus_pull_max_age_hours: 24
|
||||||
atlas_photobook_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_photobook }}"
|
atlas_photobook_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_photobook }}"
|
||||||
|
|
||||||
|
# Rootless Gitea runs in admin's user manager; the image maps internal gitea to UID/GID 1000.
|
||||||
|
atlas_manage_gitea: false
|
||||||
|
atlas_gitea_username: "{{ atlas_admin_username }}"
|
||||||
|
atlas_gitea_group: "{{ atlas_admin_group }}"
|
||||||
|
atlas_gitea_uid: "{{ atlas_admin_uid }}"
|
||||||
|
atlas_gitea_gid: "{{ atlas_admin_gid }}"
|
||||||
|
atlas_gitea_home: "{{ atlas_admin_home }}"
|
||||||
|
atlas_gitea_container_uid: 1000
|
||||||
|
atlas_gitea_container_gid: 1000
|
||||||
|
atlas_gitea_legacy_username: gitea
|
||||||
|
atlas_gitea_dataset: "{{ atlas_zfs_pool }}/services/data/gitea"
|
||||||
|
atlas_gitea_mountpoint: "{{ atlas_app_data_mountpoint }}/gitea"
|
||||||
|
atlas_gitea_quadlet_dir: "{{ atlas_gitea_home }}/.config/containers/systemd"
|
||||||
|
atlas_gitea_image: localhost/atlas-gitea:1.25.2-user-gitea-v1
|
||||||
|
atlas_gitea_image_build_dir: "{{ atlas_gitea_home }}/.local/share/atlas-gitea-image"
|
||||||
|
atlas_gitea_production_enabled: false
|
||||||
|
atlas_gitea_public_domain: ""
|
||||||
|
atlas_gitea_bind_address: "{{ ansible_host }}"
|
||||||
|
atlas_gitea_http_port: 3000
|
||||||
|
atlas_gitea_ssh_port: 2222
|
||||||
|
atlas_gitea_staging_bind_address: 127.0.0.1
|
||||||
|
atlas_gitea_staging_http_port: 3001
|
||||||
|
atlas_gitea_staging_ssh_port: 2223
|
||||||
|
|
||||||
|
# Declare storage and an inactive Quadlet only. The operator supplies the
|
||||||
|
# private configuration, handles MFA, and starts the user service manually.
|
||||||
|
atlas_icloudpd_dataset: "{{ atlas_zfs_pool }}/services/data/icloudpd"
|
||||||
|
atlas_icloudpd_state_dir: "{{ atlas_app_data_mountpoint }}/icloudpd"
|
||||||
|
atlas_icloudpd_config_dir: "{{ atlas_icloudpd_state_dir }}/config"
|
||||||
|
atlas_icloudpd_photos_dir: "{{ atlas_archive_mountpoint }}/Pictures/iCloudPD"
|
||||||
|
atlas_icloudpd_image: >-
|
||||||
|
docker.io/boredazfcuk/icloudpd@sha256:9966c31ddf0b5b306ac2410b4edd5d626806d96e80c92b83cbb689972dc9389f
|
||||||
|
atlas_icloudpd_quadlet_dir: "{{ atlas_admin_home }}/.config/containers/systemd"
|
||||||
|
atlas_icloudpd_timezone: Europe/Rome
|
||||||
|
|
||||||
atlas_45drives_repo_url: https://repo.45drives.com/repofiles/rocky/45drives-enterprise.repo
|
atlas_45drives_repo_url: https://repo.45drives.com/repofiles/rocky/45drives-enterprise.repo
|
||||||
atlas_45drives_repo_file: /etc/yum.repos.d/45drives-enterprise.repo
|
atlas_45drives_repo_file: /etc/yum.repos.d/45drives-enterprise.repo
|
||||||
atlas_45drives_packages:
|
atlas_45drives_packages:
|
||||||
|
|||||||
@@ -0,0 +1,10 @@
|
|||||||
|
FROM docker.gitea.com/gitea@sha256:f1943db2d2f1e447e857b3f0aee4ebb7b184500f86e5b80eae110fd435435906
|
||||||
|
|
||||||
|
# Preserve the official image's UID/GID, paths and entrypoint; change only the
|
||||||
|
# internal Unix identity. The host-side rootless owner is Atlas admin.
|
||||||
|
USER 0
|
||||||
|
RUN sed -i 's/^git:x:1000:1000:/gitea:x:1000:1000:/' /etc/passwd \
|
||||||
|
&& sed -i 's/^git:x:1000:/gitea:x:1000:/' /etc/group \
|
||||||
|
&& grep -q '^gitea:x:1000:1000:' /etc/passwd \
|
||||||
|
&& grep -q '^gitea:x:1000:' /etc/group
|
||||||
|
USER 1000:1000
|
||||||
59
ansible/roles/profile_atlas/files/atlas-borg-progress.py
Normal file
59
ansible/roles/profile_atlas/files/atlas-borg-progress.py
Normal file
@@ -0,0 +1,59 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Turn Borg's JSON progress stream into bounded, readable journal entries."""
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
import sys
|
||||||
|
import time
|
||||||
|
|
||||||
|
|
||||||
|
def size(value):
|
||||||
|
if not isinstance(value, (int, float)):
|
||||||
|
return "unknown"
|
||||||
|
return f"{value / (1024 ** 3):.2f} GiB"
|
||||||
|
|
||||||
|
|
||||||
|
parser = argparse.ArgumentParser()
|
||||||
|
parser.add_argument("--estimated-total-bytes", type=int, required=True)
|
||||||
|
args = parser.parse_args()
|
||||||
|
if args.estimated_total_bytes <= 0:
|
||||||
|
parser.error("estimated total must be positive")
|
||||||
|
|
||||||
|
last_progress = 0.0
|
||||||
|
for line in sys.stdin:
|
||||||
|
try:
|
||||||
|
event = json.loads(line)
|
||||||
|
except json.JSONDecodeError:
|
||||||
|
print(line.rstrip(), flush=True)
|
||||||
|
continue
|
||||||
|
|
||||||
|
kind = event.get("type")
|
||||||
|
if kind == "archive_progress":
|
||||||
|
now = time.monotonic()
|
||||||
|
if now - last_progress < 60 and not event.get("finished"):
|
||||||
|
continue
|
||||||
|
path = event.get("path") or ""
|
||||||
|
parts = path.split("/")
|
||||||
|
dataset = parts[1] if len(parts) > 1 and parts[0] == "source" else "unknown"
|
||||||
|
original_size = event.get("original_size")
|
||||||
|
if isinstance(original_size, (int, float)) and original_size >= 0:
|
||||||
|
percent = original_size / args.estimated_total_bytes * 100
|
||||||
|
estimated_progress = (
|
||||||
|
f"{percent:.1f}%" if percent < 100 else ">=100% (ZFS estimate exceeded)"
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
estimated_progress = "unknown"
|
||||||
|
print(
|
||||||
|
"Borg create progress: "
|
||||||
|
f"estimated={estimated_progress} dataset={dataset} "
|
||||||
|
f"files={event.get('nfiles', 'unknown')} "
|
||||||
|
f"original={size(original_size)} "
|
||||||
|
f"compressed={size(event.get('compressed_size'))} "
|
||||||
|
f"deduplicated={size(event.get('deduplicated_size'))}",
|
||||||
|
flush=True,
|
||||||
|
)
|
||||||
|
last_progress = now
|
||||||
|
elif kind == "log_message":
|
||||||
|
print(f"Borg {event.get('levelname', 'INFO')}: {event.get('message', '')}", flush=True)
|
||||||
|
elif kind == "progress_message" and event.get("message"):
|
||||||
|
print(f"Borg: {event['message']}", flush=True)
|
||||||
396
ansible/roles/profile_atlas/files/atlas-health-monitor.py
Normal file
396
ansible/roles/profile_atlas/files/atlas-health-monitor.py
Normal file
@@ -0,0 +1,396 @@
|
|||||||
|
#!/usr/bin/python3
|
||||||
|
"""Read-only Atlas health probes with deduplicated 45Drives Alerts."""
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import fcntl
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import re
|
||||||
|
import subprocess
|
||||||
|
import sys
|
||||||
|
import tempfile
|
||||||
|
import time
|
||||||
|
from datetime import datetime, timezone
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
|
||||||
|
CONFIG_PATH = Path("/etc/atlas-health-monitor.json")
|
||||||
|
STATE_DIR = Path("/var/lib/atlas-health-monitor")
|
||||||
|
STATE_PATH = STATE_DIR / "state.json"
|
||||||
|
GIB = 1024**3
|
||||||
|
|
||||||
|
|
||||||
|
def run(*argv, timeout=40):
|
||||||
|
return subprocess.run(argv, capture_output=True, text=True, timeout=timeout, check=False)
|
||||||
|
|
||||||
|
|
||||||
|
def issue(issues, key, severity, message):
|
||||||
|
issues[key] = {"severity": severity, "message": message}
|
||||||
|
|
||||||
|
|
||||||
|
def notify(config, event, severity, subject, message):
|
||||||
|
now = datetime.now(timezone.utc)
|
||||||
|
payload = {
|
||||||
|
"timestamp": now.isoformat(timespec="seconds"),
|
||||||
|
"unixtime": int(now.timestamp()),
|
||||||
|
"event": event,
|
||||||
|
"severity": severity,
|
||||||
|
"subject": subject,
|
||||||
|
"email_message": message,
|
||||||
|
}
|
||||||
|
result = run(config["notifier"], json.dumps(payload, ensure_ascii=False), timeout=30)
|
||||||
|
if result.returncode:
|
||||||
|
raise RuntimeError(f"45Drives notifier exited {result.returncode}: {result.stderr.strip()}")
|
||||||
|
|
||||||
|
|
||||||
|
def parse_fields(text):
|
||||||
|
return dict(line.split("=", 1) for line in text.splitlines() if "=" in line)
|
||||||
|
|
||||||
|
|
||||||
|
def systemd_fields(unit, *properties):
|
||||||
|
result = run("systemctl", "show", unit, *(f"-p{item}" for item in properties))
|
||||||
|
if result.returncode:
|
||||||
|
raise RuntimeError(f"systemctl show {unit} exited {result.returncode}")
|
||||||
|
return parse_fields(result.stdout)
|
||||||
|
|
||||||
|
|
||||||
|
def unix_time(text):
|
||||||
|
if not text or text == "n/a":
|
||||||
|
return None
|
||||||
|
result = run("date", "-d", text, "+%s")
|
||||||
|
if result.returncode:
|
||||||
|
raise ValueError(f"Cannot parse systemd timestamp: {text}")
|
||||||
|
return int(result.stdout.strip())
|
||||||
|
|
||||||
|
|
||||||
|
def check_pool(config, issues, measurements):
|
||||||
|
pool = config["pool"]
|
||||||
|
listing = run("zpool", "list", "-H", "-p", "-o", "size,alloc,capacity,health", pool)
|
||||||
|
if listing.returncode:
|
||||||
|
issue(issues, "pool.probe", "critical", f"Cannot query ZFS pool {pool}")
|
||||||
|
return
|
||||||
|
try:
|
||||||
|
size, alloc, capacity, health = listing.stdout.strip().split("\t")
|
||||||
|
size, alloc, capacity = int(size), int(alloc), int(capacity)
|
||||||
|
except (ValueError, TypeError):
|
||||||
|
issue(issues, "pool.probe", "critical", "Invalid ZFS pool capacity response")
|
||||||
|
return
|
||||||
|
measurements.update(pool_size_bytes=size, pool_alloc_bytes=alloc, pool_capacity_percent=capacity)
|
||||||
|
if health != "ONLINE":
|
||||||
|
issue(issues, "pool.health", "critical", f"ZFS pool {pool} state is {health}")
|
||||||
|
if capacity >= config["pool_critical_percent"]:
|
||||||
|
issue(issues, "pool.capacity", "critical", f"ZFS pool {pool} is {capacity}% full")
|
||||||
|
elif capacity >= config["pool_warning_percent"]:
|
||||||
|
issue(issues, "pool.capacity", "warning", f"ZFS pool {pool} is {capacity}% full")
|
||||||
|
|
||||||
|
status = run("zpool", "status", "-P", pool)
|
||||||
|
if status.returncode:
|
||||||
|
issue(issues, "pool.status", "critical", f"Cannot query detailed ZFS status for {pool}")
|
||||||
|
return
|
||||||
|
bad_vdevs = []
|
||||||
|
for line in status.stdout.splitlines():
|
||||||
|
match = re.match(r"^\s*(\S+)\s+(ONLINE|DEGRADED|FAULTED|OFFLINE|UNAVAIL|REMOVED)\s+(\d+)\s+(\d+)\s+(\d+)", line)
|
||||||
|
if match:
|
||||||
|
name, state, reads, writes, checksums = match.groups()
|
||||||
|
if state != "ONLINE" or any(int(value) for value in (reads, writes, checksums)):
|
||||||
|
bad_vdevs.append(f"{name}: {state}, READ={reads}, WRITE={writes}, CKSUM={checksums}")
|
||||||
|
if bad_vdevs:
|
||||||
|
issue(issues, "pool.vdevs", "critical", "ZFS vdev errors: " + "; ".join(bad_vdevs))
|
||||||
|
errors = re.search(r"^errors:\s*(.*)$", status.stdout, re.MULTILINE)
|
||||||
|
if not errors or errors.group(1).strip() != "No known data errors":
|
||||||
|
issue(issues, "pool.data_errors", "critical", "ZFS status reports data errors; inspect zpool status -v")
|
||||||
|
if re.search(r"^\s*scan:\s*resilver in progress", status.stdout, re.MULTILINE | re.IGNORECASE):
|
||||||
|
issue(issues, "pool.resilver", "warning", "ZFS resilver is in progress; inspect zpool status")
|
||||||
|
scan = re.search(r"^\s*scan:\s*(.*)$", status.stdout, re.MULTILINE)
|
||||||
|
if scan and re.search(r"\bwith [1-9][0-9]* errors\b", scan.group(1)):
|
||||||
|
issue(issues, "pool.scan_errors", "critical", f"ZFS scan reported errors: {scan.group(1)}")
|
||||||
|
|
||||||
|
|
||||||
|
def check_capacity(config, issues, measurements):
|
||||||
|
pool = config["pool"]
|
||||||
|
listing = run("zfs", "list", "-H", "-p", "-o", "name,usedbysnapshots", "-r", pool)
|
||||||
|
if listing.returncode:
|
||||||
|
issue(issues, "snapshot.probe", "warning", "Cannot query ZFS snapshot space")
|
||||||
|
else:
|
||||||
|
try:
|
||||||
|
snapshots = sum(int(line.split("\t")[1]) for line in listing.stdout.splitlines())
|
||||||
|
measurements["snapshots_bytes"] = snapshots
|
||||||
|
size = measurements.get("pool_size_bytes")
|
||||||
|
if size:
|
||||||
|
percent = snapshots * 100 // size
|
||||||
|
measurements["snapshots_percent"] = percent
|
||||||
|
if percent >= config["snapshot_critical_percent"]:
|
||||||
|
issue(issues, "snapshot.capacity", "critical", f"Snapshots use {percent}% of pool size")
|
||||||
|
elif percent >= config["snapshot_warning_percent"]:
|
||||||
|
issue(issues, "snapshot.capacity", "warning", f"Snapshots use {percent}% of pool size")
|
||||||
|
except (ValueError, IndexError):
|
||||||
|
issue(issues, "snapshot.probe", "warning", "Invalid ZFS snapshot-space response")
|
||||||
|
backup = run("zfs", "list", "-H", "-p", "-o", "used", config["backup_dataset"])
|
||||||
|
if backup.returncode:
|
||||||
|
issue(issues, "backup.capacity_probe", "warning", "Cannot query local backup dataset space")
|
||||||
|
else:
|
||||||
|
try:
|
||||||
|
measurements["backup_bytes"] = int(backup.stdout.strip())
|
||||||
|
except ValueError:
|
||||||
|
issue(issues, "backup.capacity_probe", "warning", "Invalid local backup space response")
|
||||||
|
|
||||||
|
try:
|
||||||
|
filesystem = os.statvfs("/")
|
||||||
|
total = filesystem.f_blocks * filesystem.f_frsize
|
||||||
|
available = filesystem.f_bavail * filesystem.f_frsize
|
||||||
|
used_percent = (total - available) * 100 // total
|
||||||
|
measurements["root_capacity_percent"] = used_percent
|
||||||
|
if used_percent >= config["root_critical_percent"]:
|
||||||
|
issue(issues, "root.capacity", "critical", f"Atlas system filesystem is {used_percent}% full")
|
||||||
|
elif used_percent >= config["root_warning_percent"]:
|
||||||
|
issue(issues, "root.capacity", "warning", f"Atlas system filesystem is {used_percent}% full")
|
||||||
|
except (OSError, ZeroDivisionError):
|
||||||
|
issue(issues, "root.capacity_probe", "warning", "Cannot query Atlas system filesystem space")
|
||||||
|
|
||||||
|
|
||||||
|
def check_remote_capacity(config, issues, measurements):
|
||||||
|
"""Query only the Storage Box quota; do not open or inspect the Borg repository."""
|
||||||
|
remote = config["remote_capacity"]
|
||||||
|
try:
|
||||||
|
result = run("runuser", "-u", remote["run_as"], "--", remote["ssh_wrapper"],
|
||||||
|
f"{remote['user']}@{remote['host']}", "df", "-m", timeout=65)
|
||||||
|
if result.returncode:
|
||||||
|
raise ValueError(f"SSH df exited {result.returncode}")
|
||||||
|
lines = result.stdout.strip().splitlines()
|
||||||
|
if len(lines) != 2:
|
||||||
|
raise ValueError("Unexpected Storage Box df output")
|
||||||
|
fields = lines[1].split()
|
||||||
|
if len(fields) < 5:
|
||||||
|
raise ValueError("Incomplete Storage Box df output")
|
||||||
|
total_mib, used_mib, available_mib = (int(value) for value in fields[1:4])
|
||||||
|
percent = int(fields[4].rstrip("%"))
|
||||||
|
if total_mib <= 0 or not 0 <= percent <= 100 or available_mib < 0:
|
||||||
|
raise ValueError("Invalid Storage Box quota values")
|
||||||
|
except (OSError, ValueError, subprocess.TimeoutExpired):
|
||||||
|
issue(issues, "remote.capacity_probe", "warning", "Cannot query Hetzner Storage Box quota via pinned-key SSH")
|
||||||
|
return
|
||||||
|
measurements.update(remote_capacity_percent=percent, remote_bytes=used_mib * 1024**2,
|
||||||
|
remote_available_bytes=available_mib * 1024**2)
|
||||||
|
if percent >= remote["critical_percent"]:
|
||||||
|
issue(issues, "remote.capacity", "critical", f"Hetzner Storage Box quota is {percent}% full")
|
||||||
|
elif percent >= remote["warning_percent"]:
|
||||||
|
issue(issues, "remote.capacity", "warning", f"Hetzner Storage Box quota is {percent}% full")
|
||||||
|
|
||||||
|
|
||||||
|
def check_smart(config, issues, measurements):
|
||||||
|
for device in config["smart_devices"]:
|
||||||
|
name, path = device["name"], device["path"]
|
||||||
|
try:
|
||||||
|
result = run("smartctl", "-j", "-a", path, timeout=60)
|
||||||
|
data = json.loads(result.stdout)
|
||||||
|
status = int(data.get("smartctl", {}).get("exit_status", result.returncode))
|
||||||
|
except (subprocess.TimeoutExpired, json.JSONDecodeError, ValueError) as exc:
|
||||||
|
issue(issues, f"smart.{name}.probe", "critical", f"SMART probe failed for {name}: {type(exc).__name__}")
|
||||||
|
continue
|
||||||
|
if status:
|
||||||
|
severity = "critical" if status & 0b00001111 else "warning"
|
||||||
|
issue(issues, f"smart.{name}.status", severity, f"SMART reported exit status {status} for {name}")
|
||||||
|
passed = data.get("smart_status", {}).get("passed")
|
||||||
|
if passed is False:
|
||||||
|
issue(issues, f"smart.{name}.health", "critical", f"SMART self-assessment failed for {name}")
|
||||||
|
elif passed is None:
|
||||||
|
issue(issues, f"smart.{name}.health", "warning", f"SMART self-assessment unavailable for {name}")
|
||||||
|
temperature = data.get("temperature", {}).get("current")
|
||||||
|
if isinstance(temperature, (int, float)):
|
||||||
|
measurements[f"smart_{name}_c"] = temperature
|
||||||
|
if temperature >= device["critical_c"]:
|
||||||
|
issue(issues, f"smart.{name}.temperature", "critical", f"{name} temperature is {temperature} C")
|
||||||
|
elif temperature >= device["warning_c"]:
|
||||||
|
issue(issues, f"smart.{name}.temperature", "warning", f"{name} temperature is {temperature} C")
|
||||||
|
else:
|
||||||
|
issue(issues, f"smart.{name}.temperature", "warning", f"Temperature unavailable for {name}")
|
||||||
|
for attribute in data.get("ata_smart_attributes", {}).get("table", []):
|
||||||
|
attribute_id = attribute.get("id")
|
||||||
|
if attribute_id in (5, 187, 197, 198):
|
||||||
|
raw = attribute.get("raw", {}).get("value", 0)
|
||||||
|
if isinstance(raw, int) and raw > 0:
|
||||||
|
severity = "critical" if attribute_id in (197, 198) else "warning"
|
||||||
|
issue(issues, f"smart.{name}.ata_{attribute_id}", severity,
|
||||||
|
f"{name} SMART attribute {attribute_id} raw count is {raw}")
|
||||||
|
nvme = data.get("nvme_smart_health_information_log", {})
|
||||||
|
if isinstance(nvme, dict):
|
||||||
|
if int(nvme.get("critical_warning", 0)):
|
||||||
|
issue(issues, f"smart.{name}.nvme_warning", "critical", f"{name} NVMe critical warning is nonzero")
|
||||||
|
if int(nvme.get("media_errors", 0)):
|
||||||
|
issue(issues, f"smart.{name}.nvme_media", "critical", f"{name} NVMe media errors are nonzero")
|
||||||
|
|
||||||
|
|
||||||
|
def check_cpu(config, issues, measurements):
|
||||||
|
sensors = []
|
||||||
|
for hwmon in Path("/sys/class/hwmon").glob("hwmon*"):
|
||||||
|
try:
|
||||||
|
if (hwmon / "name").read_text().strip() != "coretemp":
|
||||||
|
continue
|
||||||
|
sensors.extend(int(path.read_text().strip()) / 1000 for path in hwmon.glob("temp*_input"))
|
||||||
|
except (OSError, ValueError):
|
||||||
|
continue
|
||||||
|
if not sensors:
|
||||||
|
issue(issues, "cpu.temperature_probe", "warning", "CPU temperature sensors are unavailable")
|
||||||
|
return
|
||||||
|
hottest = max(sensors)
|
||||||
|
measurements["cpu_max_c"] = hottest
|
||||||
|
if hottest >= config["cpu_critical_c"]:
|
||||||
|
issue(issues, "cpu.temperature", "critical", f"CPU temperature is {hottest:g} C")
|
||||||
|
elif hottest >= config["cpu_warning_c"]:
|
||||||
|
issue(issues, "cpu.temperature", "warning", f"CPU temperature is {hottest:g} C")
|
||||||
|
|
||||||
|
|
||||||
|
def check_jobs(config, issues, measurements, now):
|
||||||
|
for timer in config["timers"]:
|
||||||
|
name = timer["name"]
|
||||||
|
try:
|
||||||
|
fields = systemd_fields(name, "ActiveState", "UnitFileState", "LastTriggerUSec", "ActiveEnterTimestamp")
|
||||||
|
if fields.get("ActiveState") != "active" or fields.get("UnitFileState") != "enabled":
|
||||||
|
issue(issues, f"timer.{name}", "critical", f"Timer {name} is not active and enabled")
|
||||||
|
max_age = int(timer["max_age_hours"]) * 3600
|
||||||
|
if max_age:
|
||||||
|
last = unix_time(fields.get("LastTriggerUSec"))
|
||||||
|
if last is None:
|
||||||
|
last = unix_time(fields.get("ActiveEnterTimestamp"))
|
||||||
|
if last is not None and now - last > max_age:
|
||||||
|
issue(issues, f"timer.{name}.stale", "warning",
|
||||||
|
f"Timer {name} has not fired in {int((now-last)/3600)} hours")
|
||||||
|
except (RuntimeError, ValueError, subprocess.TimeoutExpired):
|
||||||
|
issue(issues, f"timer.{name}.probe", "warning", f"Cannot query timer {name}")
|
||||||
|
for unit in config["failure_units"]:
|
||||||
|
if unit.endswith("@.service"):
|
||||||
|
continue
|
||||||
|
try:
|
||||||
|
fields = systemd_fields(unit, "ActiveState", "Result", "ExecMainStartTimestamp")
|
||||||
|
state = fields.get("ActiveState")
|
||||||
|
if state == "failed" or (state == "inactive" and fields.get("Result") not in (None, "", "success")):
|
||||||
|
issue(issues, f"service.{unit}", "critical", f"Service {unit} failed: {fields.get('Result')}")
|
||||||
|
if unit == "atlas-borg-backup.service" and fields.get("ActiveState") == "activating":
|
||||||
|
started = unix_time(fields.get("ExecMainStartTimestamp"))
|
||||||
|
if started is not None and now - started > config["borg_max_runtime_days"] * 86400:
|
||||||
|
issue(issues, "backup.borg_long_running", "warning",
|
||||||
|
"Borg has run longer than its configured limit")
|
||||||
|
except (RuntimeError, ValueError, subprocess.TimeoutExpired):
|
||||||
|
issue(issues, f"service.{unit}.probe", "warning", f"Cannot query service {unit}")
|
||||||
|
|
||||||
|
|
||||||
|
def check_growth(config, issues, measurements, samples, now):
|
||||||
|
previous = [sample for sample in samples if 20 * 3600 <= now - sample.get("time", now) <= 48 * 3600]
|
||||||
|
if previous:
|
||||||
|
baseline = min(previous, key=lambda sample: abs(now - sample["time"] - 86400))
|
||||||
|
days = (now - baseline["time"]) / 86400
|
||||||
|
for name, threshold in (("snapshots", config["snapshot_growth_warning_gib_day"]),
|
||||||
|
("backup", config["backup_growth_warning_gib_day"]),
|
||||||
|
("remote", config["remote_capacity"]["growth_warning_gib_day"])):
|
||||||
|
current, old = measurements.get(f"{name}_bytes"), baseline.get(f"{name}_bytes")
|
||||||
|
if isinstance(current, int) and isinstance(old, int) and days > 0:
|
||||||
|
growth_gib_day = (current - old) / GIB / days
|
||||||
|
measurements[f"{name}_growth_gib_day"] = round(growth_gib_day, 1)
|
||||||
|
if growth_gib_day >= threshold:
|
||||||
|
issue(issues, f"{name}.growth", "warning",
|
||||||
|
f"Local {name} usage grew {growth_gib_day:.1f} GiB/day over {days:.1f} days")
|
||||||
|
|
||||||
|
|
||||||
|
def allowed_failure_unit(config, unit):
|
||||||
|
for allowed in config["failure_units"]:
|
||||||
|
if allowed == unit:
|
||||||
|
return True
|
||||||
|
if allowed.endswith("@.service") and unit.startswith(allowed[:-9] + "@") and unit.endswith(".service"):
|
||||||
|
return True
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
def load_state():
|
||||||
|
if not STATE_PATH.exists():
|
||||||
|
return {"active": {}, "samples": []}
|
||||||
|
with STATE_PATH.open(encoding="utf-8") as stream:
|
||||||
|
state = json.load(stream)
|
||||||
|
if not isinstance(state.get("active"), dict) or not isinstance(state.get("samples"), list):
|
||||||
|
raise ValueError("Invalid Atlas monitor state; refusing to overwrite it")
|
||||||
|
return state
|
||||||
|
|
||||||
|
|
||||||
|
def save_state(state):
|
||||||
|
with tempfile.NamedTemporaryFile("w", dir=STATE_DIR, prefix=".state-", delete=False,
|
||||||
|
encoding="utf-8") as stream:
|
||||||
|
path = Path(stream.name)
|
||||||
|
os.chmod(path, 0o600)
|
||||||
|
json.dump(state, stream, sort_keys=True)
|
||||||
|
stream.write("\n")
|
||||||
|
stream.flush()
|
||||||
|
os.fsync(stream.fileno())
|
||||||
|
os.replace(path, STATE_PATH)
|
||||||
|
|
||||||
|
|
||||||
|
def main():
|
||||||
|
parser = argparse.ArgumentParser(description=__doc__)
|
||||||
|
parser.add_argument("--dry-run", action="store_true", help="probe without notifications or state changes")
|
||||||
|
parser.add_argument("--test-notification", action="store_true", help="submit a labelled test alert")
|
||||||
|
parser.add_argument("--job-failed", metavar="UNIT", help="notify about a failed configured service")
|
||||||
|
args = parser.parse_args()
|
||||||
|
with CONFIG_PATH.open(encoding="utf-8") as stream:
|
||||||
|
config = json.load(stream)
|
||||||
|
if args.test_notification:
|
||||||
|
notify(config, "atlas_monitor_test", "warning", "Test monitoraggio Atlas",
|
||||||
|
"Notifica di prova: il monitoraggio Atlas raggiunge 45Drives Alerts. Non conferma l'invio email.")
|
||||||
|
print("Atlas monitor test submitted to 45Drives Alerts; email delivery is not verified.")
|
||||||
|
return 0
|
||||||
|
if args.job_failed:
|
||||||
|
if not allowed_failure_unit(config, args.job_failed):
|
||||||
|
raise ValueError("Unconfigured Atlas failure unit")
|
||||||
|
notify(config, "atlas_job_failed", "critical", f"Job Atlas fallito: {args.job_failed}",
|
||||||
|
f"Il servizio {args.job_failed} e' fallito. Controlla: "
|
||||||
|
f"sudo journalctl -u {args.job_failed} -n 100 --no-pager")
|
||||||
|
print(f"Atlas job failure submitted to 45Drives Alerts: {args.job_failed}")
|
||||||
|
return 0
|
||||||
|
|
||||||
|
now = int(time.time())
|
||||||
|
issues, measurements = {}, {}
|
||||||
|
check_pool(config, issues, measurements)
|
||||||
|
check_capacity(config, issues, measurements)
|
||||||
|
check_remote_capacity(config, issues, measurements)
|
||||||
|
check_smart(config, issues, measurements)
|
||||||
|
check_cpu(config, issues, measurements)
|
||||||
|
check_jobs(config, issues, measurements, now)
|
||||||
|
if args.dry_run:
|
||||||
|
print(json.dumps({"issues": issues, "measurements": measurements}, sort_keys=True))
|
||||||
|
return 0
|
||||||
|
|
||||||
|
STATE_DIR.mkdir(mode=0o700, exist_ok=True)
|
||||||
|
with (STATE_DIR / "monitor.lock").open("w") as lock:
|
||||||
|
fcntl.flock(lock, fcntl.LOCK_EX | fcntl.LOCK_NB)
|
||||||
|
state = load_state()
|
||||||
|
check_growth(config, issues, measurements, state["samples"], now)
|
||||||
|
active, failed_notifications = state["active"], []
|
||||||
|
for key, details in issues.items():
|
||||||
|
old = active.get(key)
|
||||||
|
if old is None or old.get("severity") != details["severity"]:
|
||||||
|
try:
|
||||||
|
notify(config, "atlas_health_issue", details["severity"],
|
||||||
|
f"Atlas: {key}", details["message"])
|
||||||
|
active[key] = details
|
||||||
|
print(f"ALERT {details['severity']} {key}: {details['message']}", flush=True)
|
||||||
|
except (RuntimeError, subprocess.TimeoutExpired) as exc:
|
||||||
|
failed_notifications.append(key)
|
||||||
|
print(f"NOTIFICATION FAILED {key}: {exc}", file=sys.stderr, flush=True)
|
||||||
|
for key in set(active) - set(issues):
|
||||||
|
print(f"RECOVERED {key}", flush=True)
|
||||||
|
del active[key]
|
||||||
|
state["samples"] = [sample for sample in state["samples"] if now - sample.get("time", 0) < 48 * 3600]
|
||||||
|
state["samples"].append({"time": now, **{key: value for key, value in measurements.items()
|
||||||
|
if key in ("snapshots_bytes", "backup_bytes", "remote_bytes")}})
|
||||||
|
save_state(state)
|
||||||
|
print(f"Atlas health: issues={len(issues)} notifications_failed={len(failed_notifications)} "
|
||||||
|
f"pool={measurements.get('pool_capacity_percent', 'unknown')}% "
|
||||||
|
f"remote={measurements.get('remote_capacity_percent', 'unknown')}% "
|
||||||
|
f"snapshots={measurements.get('snapshots_bytes', 'unknown')} bytes "
|
||||||
|
f"backup={measurements.get('backup_bytes', 'unknown')} bytes", flush=True)
|
||||||
|
return 1 if failed_notifications else 0
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
try:
|
||||||
|
sys.exit(main())
|
||||||
|
except (OSError, RuntimeError, ValueError, subprocess.TimeoutExpired) as error:
|
||||||
|
print(f"Atlas health monitor failed: {error}", file=sys.stderr)
|
||||||
|
sys.exit(1)
|
||||||
56
ansible/roles/profile_atlas/files/atlas-prometheus-prune.py
Normal file
56
ansible/roles/profile_atlas/files/atlas-prometheus-prune.py
Normal file
@@ -0,0 +1,56 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Prune only verified, named Prometheus backup versions after publication."""
|
||||||
|
|
||||||
|
import datetime as dt
|
||||||
|
import pathlib
|
||||||
|
import re
|
||||||
|
import shutil
|
||||||
|
import sys
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> None:
|
||||||
|
if len(sys.argv) != 5:
|
||||||
|
raise SystemExit("Usage: atlas-prometheus-prune SNAPSHOTS DAILY WEEKLY MONTHLY")
|
||||||
|
root = pathlib.Path(sys.argv[1])
|
||||||
|
counts = [int(value) for value in sys.argv[2:]]
|
||||||
|
if not root.is_dir() or root.is_symlink() or min(counts) < 1:
|
||||||
|
raise SystemExit("Invalid backup directory or retention counts")
|
||||||
|
versions = []
|
||||||
|
for entry in root.iterdir():
|
||||||
|
if not entry.is_dir() or entry.is_symlink():
|
||||||
|
continue
|
||||||
|
if not re.fullmatch(r"[0-9]{8}T[0-9]{6}Z", entry.name):
|
||||||
|
continue
|
||||||
|
try:
|
||||||
|
when = dt.datetime.strptime(entry.name, "%Y%m%dT%H%M%SZ")
|
||||||
|
except ValueError:
|
||||||
|
continue
|
||||||
|
if not all((entry / name).is_file() for name in ("payload.tar", "payload.sha256", "metadata.json")):
|
||||||
|
continue
|
||||||
|
versions.append((when, entry))
|
||||||
|
versions.sort(reverse=True)
|
||||||
|
if not versions:
|
||||||
|
raise SystemExit("No published backup versions found; refusing to prune")
|
||||||
|
|
||||||
|
keep = {entry for _, entry in versions[: counts[0]]}
|
||||||
|
for count, key in (
|
||||||
|
(counts[1], lambda when: when.isocalendar()[:2]),
|
||||||
|
(counts[2], lambda when: (when.year, when.month)),
|
||||||
|
):
|
||||||
|
periods = set()
|
||||||
|
for when, entry in versions:
|
||||||
|
period = key(when)
|
||||||
|
if period in periods:
|
||||||
|
continue
|
||||||
|
periods.add(period)
|
||||||
|
keep.add(entry)
|
||||||
|
if len(periods) >= count:
|
||||||
|
break
|
||||||
|
|
||||||
|
for _, entry in versions:
|
||||||
|
if entry not in keep:
|
||||||
|
shutil.rmtree(entry)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
@@ -1,4 +1,14 @@
|
|||||||
---
|
---
|
||||||
|
- name: Reload Atlas admin user manager
|
||||||
|
become_user: "{{ atlas_admin_username }}"
|
||||||
|
ansible.builtin.systemd:
|
||||||
|
scope: user
|
||||||
|
daemon_reload: true
|
||||||
|
environment:
|
||||||
|
XDG_RUNTIME_DIR: "/run/user/{{ atlas_admin_uid }}"
|
||||||
|
DBUS_SESSION_BUS_ADDRESS: "unix:path=/run/user/{{ atlas_admin_uid }}/bus"
|
||||||
|
when: not ansible_check_mode
|
||||||
|
|
||||||
- name: Reload SSH service
|
- name: Reload SSH service
|
||||||
ansible.builtin.systemd:
|
ansible.builtin.systemd:
|
||||||
name: sshd
|
name: sshd
|
||||||
@@ -28,6 +38,18 @@
|
|||||||
name: smb
|
name: smb
|
||||||
state: restarted
|
state: restarted
|
||||||
|
|
||||||
|
- name: Restart Atlas Borg timers
|
||||||
|
ansible.builtin.systemd:
|
||||||
|
name: "{{ item }}"
|
||||||
|
state: restarted
|
||||||
|
daemon_reload: true
|
||||||
|
loop:
|
||||||
|
- atlas-borg-backup.timer
|
||||||
|
- atlas-borg-check.timer
|
||||||
|
when:
|
||||||
|
- atlas_manage_borg_backup | bool
|
||||||
|
- not ansible_check_mode
|
||||||
|
|
||||||
- name: Restart Atlas media Quadlets
|
- name: Restart Atlas media Quadlets
|
||||||
ansible.builtin.systemd:
|
ansible.builtin.systemd:
|
||||||
name: "{{ item }}"
|
name: "{{ item }}"
|
||||||
|
|||||||
538
ansible/roles/profile_atlas/tasks/borg_backup.yml
Normal file
538
ansible/roles/profile_atlas/tasks/borg_backup.yml
Normal file
@@ -0,0 +1,538 @@
|
|||||||
|
---
|
||||||
|
- name: Validate Atlas Borg backup configuration
|
||||||
|
tags: [atlas, storage, backup, borg, borg_key]
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- atlas_manage_storage | bool
|
||||||
|
- atlas_zfs_pool != 'CHANGEME_ZFS_POOL'
|
||||||
|
- atlas_mount_root.startswith('/')
|
||||||
|
- atlas_borg_username is match('^[a-z_][a-z0-9_-]*$')
|
||||||
|
- atlas_borg_group is match('^[a-z_][a-z0-9_-]*$')
|
||||||
|
- atlas_borg_username not in ['root', atlas_admin_username]
|
||||||
|
- atlas_borg_group != 'wheel'
|
||||||
|
- atlas_borg_home.startswith('/var/lib/')
|
||||||
|
- atlas_borg_repository_host is match('^[A-Za-z0-9.-]+$')
|
||||||
|
- atlas_borg_repository_user is match('^[A-Za-z0-9_-]+$')
|
||||||
|
- atlas_borg_repository_port | int > 0
|
||||||
|
- atlas_borg_repository_port | int < 65536
|
||||||
|
- atlas_borg_repository_path is match('^\./[A-Za-z0-9][A-Za-z0-9._/-]*$')
|
||||||
|
- "'/../' not in ('/' ~ atlas_borg_repository_path ~ '/')"
|
||||||
|
- atlas_borg_remote_path is match('^borg-[0-9]+\.[0-9]+$')
|
||||||
|
- atlas_borg_host_key.startswith(
|
||||||
|
'[' ~ atlas_borg_repository_host ~ ']:' ~ (atlas_borg_repository_port | string) ~ ' ssh-ed25519 '
|
||||||
|
)
|
||||||
|
- atlas_borg_ssh_private_key_path.startswith('/etc/atlas-borg/')
|
||||||
|
- atlas_borg_known_hosts_path.startswith('/etc/atlas-borg/')
|
||||||
|
- atlas_borg_passphrase_path.startswith('/etc/atlas-borg/')
|
||||||
|
- atlas_borg_ssh_wrapper_path.startswith('/usr/local/libexec/')
|
||||||
|
- atlas_borg_encryption_mode == 'repokey'
|
||||||
|
- atlas_borg_archive_prefix is match('^[a-z0-9][a-z0-9_-]*$')
|
||||||
|
- atlas_borg_snapshot_prefix is match('^[a-z0-9][a-z0-9_-]*$')
|
||||||
|
- atlas_borg_keep_daily | int > 0
|
||||||
|
- atlas_borg_keep_weekly | int > 0
|
||||||
|
- atlas_borg_keep_monthly | int > 0
|
||||||
|
fail_msg: >-
|
||||||
|
Atlas Borg needs a safe relative repository path, a pinned ED25519 host
|
||||||
|
key, positive retention counts, and valid dedicated SSH settings.
|
||||||
|
when: atlas_manage_borg_backup | bool
|
||||||
|
|
||||||
|
- name: Create the Atlas Borg system group
|
||||||
|
tags: [atlas, storage, backup, borg, borg_key]
|
||||||
|
ansible.builtin.group:
|
||||||
|
name: "{{ atlas_borg_group }}"
|
||||||
|
system: true
|
||||||
|
state: present
|
||||||
|
when: atlas_manage_borg_backup | bool
|
||||||
|
|
||||||
|
- name: Create the least-privilege Atlas Borg account
|
||||||
|
tags: [atlas, storage, backup, borg, borg_key]
|
||||||
|
ansible.builtin.user:
|
||||||
|
name: "{{ atlas_borg_username }}"
|
||||||
|
group: "{{ atlas_borg_group }}"
|
||||||
|
groups: []
|
||||||
|
append: false
|
||||||
|
comment: Atlas Borg backup service
|
||||||
|
home: "{{ atlas_borg_home }}"
|
||||||
|
create_home: false
|
||||||
|
shell: /sbin/nologin
|
||||||
|
password_lock: true
|
||||||
|
system: true
|
||||||
|
state: present
|
||||||
|
when: atlas_manage_borg_backup | bool
|
||||||
|
|
||||||
|
- name: Read Atlas Borg account group membership
|
||||||
|
tags: [atlas, storage, backup, borg, borg_key]
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv:
|
||||||
|
- id
|
||||||
|
- -nG
|
||||||
|
- "{{ atlas_borg_username }}"
|
||||||
|
register: atlas_borg_account_groups
|
||||||
|
changed_when: false
|
||||||
|
check_mode: false
|
||||||
|
when:
|
||||||
|
- atlas_manage_borg_backup | bool
|
||||||
|
- not ansible_check_mode
|
||||||
|
|
||||||
|
- name: Require the Atlas Borg account to have no supplementary groups
|
||||||
|
tags: [atlas, storage, backup, borg, borg_key]
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- atlas_borg_account_groups.stdout.split() == [atlas_borg_group]
|
||||||
|
fail_msg: >-
|
||||||
|
The Atlas Borg service account must belong only to its private primary
|
||||||
|
group and must never receive wheel or other supplementary membership.
|
||||||
|
when:
|
||||||
|
- atlas_manage_borg_backup | bool
|
||||||
|
- not ansible_check_mode
|
||||||
|
|
||||||
|
- name: Validate Atlas Borg systemd calendars
|
||||||
|
tags: [atlas, storage, backup, borg]
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv:
|
||||||
|
- systemd-analyze
|
||||||
|
- calendar
|
||||||
|
- "{{ item }}"
|
||||||
|
loop:
|
||||||
|
- "{{ atlas_borg_backup_calendar }}"
|
||||||
|
- "{{ atlas_borg_check_calendar }}"
|
||||||
|
changed_when: false
|
||||||
|
check_mode: false
|
||||||
|
when: atlas_manage_borg_backup | bool
|
||||||
|
|
||||||
|
- name: Create Atlas Borg configuration directory
|
||||||
|
tags: [atlas, storage, backup, borg, borg_key]
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: /etc/atlas-borg
|
||||||
|
state: directory
|
||||||
|
owner: root
|
||||||
|
group: "{{ atlas_borg_group }}"
|
||||||
|
mode: "0750"
|
||||||
|
when: atlas_manage_borg_backup | bool
|
||||||
|
|
||||||
|
- name: Generate the dedicated Atlas Borg SSH identity
|
||||||
|
tags: [atlas, storage, backup, borg, borg_key]
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv:
|
||||||
|
- ssh-keygen
|
||||||
|
- -q
|
||||||
|
- -t
|
||||||
|
- ed25519
|
||||||
|
- -N
|
||||||
|
- ""
|
||||||
|
- -C
|
||||||
|
- atlas-borg@atlas
|
||||||
|
- -f
|
||||||
|
- "{{ atlas_borg_ssh_private_key_path }}"
|
||||||
|
creates: "{{ atlas_borg_ssh_private_key_path }}"
|
||||||
|
when: atlas_manage_borg_backup | bool
|
||||||
|
|
||||||
|
- name: Protect the Atlas Borg private SSH identity
|
||||||
|
tags: [atlas, storage, backup, borg, borg_key]
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: "{{ atlas_borg_ssh_private_key_path }}"
|
||||||
|
owner: "{{ atlas_borg_username }}"
|
||||||
|
group: "{{ atlas_borg_group }}"
|
||||||
|
mode: "0600"
|
||||||
|
when:
|
||||||
|
- atlas_manage_borg_backup | bool
|
||||||
|
- not ansible_check_mode
|
||||||
|
|
||||||
|
- name: Set permissions on the Atlas Borg public SSH identity
|
||||||
|
tags: [atlas, storage, backup, borg, borg_key]
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: "{{ atlas_borg_ssh_private_key_path }}.pub"
|
||||||
|
owner: "{{ atlas_borg_username }}"
|
||||||
|
group: "{{ atlas_borg_group }}"
|
||||||
|
mode: "0644"
|
||||||
|
when:
|
||||||
|
- atlas_manage_borg_backup | bool
|
||||||
|
- not ansible_check_mode
|
||||||
|
|
||||||
|
- name: Read the dedicated Atlas Borg public SSH identity
|
||||||
|
tags: [atlas, storage, backup, borg, borg_key]
|
||||||
|
ansible.builtin.slurp:
|
||||||
|
src: "{{ atlas_borg_ssh_private_key_path }}.pub"
|
||||||
|
register: atlas_borg_public_key
|
||||||
|
when:
|
||||||
|
- atlas_manage_borg_backup | bool
|
||||||
|
- not ansible_check_mode
|
||||||
|
|
||||||
|
- name: Report the public SSH identity to install in the Hetzner sub-account
|
||||||
|
tags: [atlas, storage, backup, borg, borg_key]
|
||||||
|
ansible.builtin.debug:
|
||||||
|
msg: "{{ atlas_borg_public_key.content | b64decode | trim }}"
|
||||||
|
when:
|
||||||
|
- atlas_manage_borg_backup | bool
|
||||||
|
- not ansible_check_mode
|
||||||
|
|
||||||
|
- name: Pin the Hetzner Storage Box SSH host key
|
||||||
|
tags: [atlas, storage, backup, borg]
|
||||||
|
ansible.builtin.copy:
|
||||||
|
content: "{{ atlas_borg_host_key }}\n"
|
||||||
|
dest: "{{ atlas_borg_known_hosts_path }}"
|
||||||
|
owner: "{{ atlas_borg_username }}"
|
||||||
|
group: "{{ atlas_borg_group }}"
|
||||||
|
mode: "0600"
|
||||||
|
when: atlas_manage_borg_backup | bool
|
||||||
|
|
||||||
|
- name: Create Atlas Borg state directories
|
||||||
|
tags: [atlas, storage, backup, borg]
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: "{{ item }}"
|
||||||
|
state: directory
|
||||||
|
owner: "{{ atlas_borg_username }}"
|
||||||
|
group: "{{ atlas_borg_group }}"
|
||||||
|
mode: "0700"
|
||||||
|
loop:
|
||||||
|
- "{{ atlas_borg_config_dir }}"
|
||||||
|
- "{{ atlas_borg_cache_dir }}"
|
||||||
|
when: atlas_manage_borg_backup | bool
|
||||||
|
|
||||||
|
- name: Create the shared Atlas Borg operation lock
|
||||||
|
tags: [atlas, storage, backup, borg]
|
||||||
|
ansible.builtin.copy:
|
||||||
|
content: ""
|
||||||
|
dest: "{{ atlas_borg_lock_path }}"
|
||||||
|
owner: "{{ atlas_borg_username }}"
|
||||||
|
group: "{{ atlas_borg_group }}"
|
||||||
|
mode: "0600"
|
||||||
|
force: false
|
||||||
|
when: atlas_manage_borg_backup | bool
|
||||||
|
|
||||||
|
- name: Require the Atlas Borg encryption passphrase from Vault
|
||||||
|
tags: [atlas, storage, backup, borg]
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- atlas_borg_passphrase | length >= 20
|
||||||
|
fail_msg: >-
|
||||||
|
Define vault_atlas_borg_passphrase with a strong unique value in the
|
||||||
|
encrypted Vault before activating the Borg repository.
|
||||||
|
no_log: true
|
||||||
|
when: atlas_manage_borg_backup | bool
|
||||||
|
|
||||||
|
- name: Install the Atlas Borg passphrase
|
||||||
|
tags: [atlas, storage, backup, borg]
|
||||||
|
ansible.builtin.copy:
|
||||||
|
content: "{{ atlas_borg_passphrase }}\n"
|
||||||
|
dest: "{{ atlas_borg_passphrase_path }}"
|
||||||
|
owner: "{{ atlas_borg_username }}"
|
||||||
|
group: "{{ atlas_borg_group }}"
|
||||||
|
mode: "0600"
|
||||||
|
diff: false
|
||||||
|
no_log: true
|
||||||
|
when: atlas_manage_borg_backup | bool
|
||||||
|
|
||||||
|
- name: Install the Atlas Borg backup helper
|
||||||
|
tags: [atlas, storage, backup, borg, borg_logging]
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: atlas-borg-backup.sh.j2
|
||||||
|
dest: /usr/local/sbin/atlas-borg-backup
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0750"
|
||||||
|
when: atlas_manage_borg_backup | bool
|
||||||
|
|
||||||
|
- name: Install the Atlas Borg snapshot cleanup helper
|
||||||
|
tags: [atlas, storage, backup, borg, borg_logging]
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: atlas-borg-snapshot-cleanup.sh.j2
|
||||||
|
dest: /usr/local/sbin/atlas-borg-snapshot-cleanup
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0750"
|
||||||
|
when: atlas_manage_borg_backup | bool
|
||||||
|
|
||||||
|
- name: Install the Atlas Borg check helper
|
||||||
|
tags: [atlas, storage, backup, borg]
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: atlas-borg-check.sh.j2
|
||||||
|
dest: /usr/local/sbin/atlas-borg-check
|
||||||
|
owner: root
|
||||||
|
group: "{{ atlas_borg_group }}"
|
||||||
|
mode: "0750"
|
||||||
|
when: atlas_manage_borg_backup | bool
|
||||||
|
|
||||||
|
- name: Create the local libexec directory for the Atlas Borg SSH wrapper
|
||||||
|
tags: [atlas, storage, backup, borg, borg_logging]
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: "{{ atlas_borg_ssh_wrapper_path | dirname }}"
|
||||||
|
state: directory
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0755"
|
||||||
|
when: atlas_manage_borg_backup | bool
|
||||||
|
|
||||||
|
- name: Install the Atlas Borg progress formatter
|
||||||
|
tags: [atlas, storage, backup, borg, borg_logging]
|
||||||
|
ansible.builtin.copy:
|
||||||
|
src: atlas-borg-progress.py
|
||||||
|
dest: /usr/local/libexec/atlas-borg-progress
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0755"
|
||||||
|
when: atlas_manage_borg_backup | bool
|
||||||
|
|
||||||
|
- name: Install the capability-dropping Atlas Borg SSH wrapper
|
||||||
|
tags: [atlas, storage, backup, borg]
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: atlas-borg-ssh.sh.j2
|
||||||
|
dest: "{{ atlas_borg_ssh_wrapper_path }}"
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0755"
|
||||||
|
when: atlas_manage_borg_backup | bool
|
||||||
|
|
||||||
|
- name: Install Atlas Borg systemd units
|
||||||
|
tags: [atlas, storage, backup, borg, borg_logging]
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: "{{ item }}.j2"
|
||||||
|
dest: "/etc/systemd/system/{{ item }}"
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0644"
|
||||||
|
loop:
|
||||||
|
- atlas-borg-backup.service
|
||||||
|
- atlas-borg-backup.timer
|
||||||
|
- atlas-borg-check.service
|
||||||
|
- atlas-borg-check.timer
|
||||||
|
notify: Restart Atlas Borg timers
|
||||||
|
when: atlas_manage_borg_backup | bool
|
||||||
|
|
||||||
|
- name: Verify dedicated SSH access to the Hetzner Storage Box
|
||||||
|
tags: [atlas, storage, backup, borg]
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv:
|
||||||
|
- ssh
|
||||||
|
- -T
|
||||||
|
- -i
|
||||||
|
- "{{ atlas_borg_ssh_private_key_path }}"
|
||||||
|
- -p
|
||||||
|
- "{{ atlas_borg_repository_port | string }}"
|
||||||
|
- -o
|
||||||
|
- BatchMode=yes
|
||||||
|
- -o
|
||||||
|
- IdentitiesOnly=yes
|
||||||
|
- -o
|
||||||
|
- StrictHostKeyChecking=yes
|
||||||
|
- -o
|
||||||
|
- "UserKnownHostsFile={{ atlas_borg_known_hosts_path }}"
|
||||||
|
- "{{ atlas_borg_repository_user }}@{{ atlas_borg_repository_host }}"
|
||||||
|
- pwd
|
||||||
|
register: atlas_borg_ssh_probe
|
||||||
|
become: true
|
||||||
|
become_user: "{{ atlas_borg_username }}"
|
||||||
|
changed_when: false
|
||||||
|
failed_when: false
|
||||||
|
when:
|
||||||
|
- atlas_manage_borg_backup | bool
|
||||||
|
- not ansible_check_mode
|
||||||
|
|
||||||
|
- name: Require the dedicated public key on the Hetzner sub-account
|
||||||
|
tags: [atlas, storage, backup, borg]
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- atlas_borg_ssh_probe.rc == 0
|
||||||
|
fail_msg: >-
|
||||||
|
Install the reported Atlas Borg public key in the Hetzner sub-account
|
||||||
|
before rerunning the Borg tasks. Password authentication is never used.
|
||||||
|
when:
|
||||||
|
- atlas_manage_borg_backup | bool
|
||||||
|
- not ansible_check_mode
|
||||||
|
|
||||||
|
- name: Probe the remote Atlas Borg repository path
|
||||||
|
tags: [atlas, storage, backup, borg]
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv:
|
||||||
|
- ssh
|
||||||
|
- -T
|
||||||
|
- -i
|
||||||
|
- "{{ atlas_borg_ssh_private_key_path }}"
|
||||||
|
- -p
|
||||||
|
- "{{ atlas_borg_repository_port | string }}"
|
||||||
|
- -o
|
||||||
|
- BatchMode=yes
|
||||||
|
- -o
|
||||||
|
- IdentitiesOnly=yes
|
||||||
|
- -o
|
||||||
|
- StrictHostKeyChecking=yes
|
||||||
|
- -o
|
||||||
|
- "UserKnownHostsFile={{ atlas_borg_known_hosts_path }}"
|
||||||
|
- "{{ atlas_borg_repository_user }}@{{ atlas_borg_repository_host }}"
|
||||||
|
- stat
|
||||||
|
- "{{ atlas_borg_repository_path }}"
|
||||||
|
register: atlas_borg_repository_path_probe
|
||||||
|
become: true
|
||||||
|
become_user: "{{ atlas_borg_username }}"
|
||||||
|
changed_when: false
|
||||||
|
failed_when: false
|
||||||
|
when:
|
||||||
|
- atlas_manage_borg_backup | bool
|
||||||
|
- not ansible_check_mode
|
||||||
|
|
||||||
|
- name: Probe the Atlas Borg repository
|
||||||
|
tags: [atlas, storage, backup, borg]
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv:
|
||||||
|
- /usr/bin/borg
|
||||||
|
- --remote-path
|
||||||
|
- "{{ atlas_borg_remote_path }}"
|
||||||
|
- info
|
||||||
|
- >-
|
||||||
|
ssh://{{ atlas_borg_repository_user }}@{{ atlas_borg_repository_host }}:
|
||||||
|
{{- atlas_borg_repository_port }}/{{ atlas_borg_repository_path }}
|
||||||
|
environment:
|
||||||
|
BORG_CACHE_DIR: "{{ atlas_borg_cache_dir }}"
|
||||||
|
BORG_CONFIG_DIR: "{{ atlas_borg_config_dir }}"
|
||||||
|
BORG_PASSCOMMAND: "cat {{ atlas_borg_passphrase_path }}"
|
||||||
|
BORG_RSH: "{{ atlas_borg_ssh_wrapper_path }}"
|
||||||
|
register: atlas_borg_repository_probe
|
||||||
|
become: true
|
||||||
|
become_user: "{{ atlas_borg_username }}"
|
||||||
|
changed_when: false
|
||||||
|
failed_when: false
|
||||||
|
when:
|
||||||
|
- atlas_manage_borg_backup | bool
|
||||||
|
- not ansible_check_mode
|
||||||
|
- atlas_borg_repository_path_probe.rc == 0
|
||||||
|
|
||||||
|
- name: Reject an existing path that is not the configured Borg repository
|
||||||
|
tags: [atlas, storage, backup, borg]
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- atlas_borg_repository_probe.rc == 0
|
||||||
|
fail_msg: >-
|
||||||
|
The remote repository path already exists but Borg could not open it.
|
||||||
|
Refusing to initialize over existing data; verify the path, passphrase,
|
||||||
|
and repository state manually.
|
||||||
|
when:
|
||||||
|
- atlas_manage_borg_backup | bool
|
||||||
|
- not ansible_check_mode
|
||||||
|
- atlas_borg_repository_path_probe.rc == 0
|
||||||
|
|
||||||
|
- name: Initialize the encrypted Atlas Borg repository
|
||||||
|
tags: [atlas, storage, backup, borg]
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv:
|
||||||
|
- /usr/bin/borg
|
||||||
|
- --remote-path
|
||||||
|
- "{{ atlas_borg_remote_path }}"
|
||||||
|
- init
|
||||||
|
- --encryption
|
||||||
|
- "{{ atlas_borg_encryption_mode }}"
|
||||||
|
- >-
|
||||||
|
ssh://{{ atlas_borg_repository_user }}@{{ atlas_borg_repository_host }}:
|
||||||
|
{{- atlas_borg_repository_port }}/{{ atlas_borg_repository_path }}
|
||||||
|
environment:
|
||||||
|
BORG_CACHE_DIR: "{{ atlas_borg_cache_dir }}"
|
||||||
|
BORG_CONFIG_DIR: "{{ atlas_borg_config_dir }}"
|
||||||
|
BORG_PASSCOMMAND: "cat {{ atlas_borg_passphrase_path }}"
|
||||||
|
BORG_RSH: "{{ atlas_borg_ssh_wrapper_path }}"
|
||||||
|
become: true
|
||||||
|
become_user: "{{ atlas_borg_username }}"
|
||||||
|
when:
|
||||||
|
- atlas_manage_borg_backup | bool
|
||||||
|
- not ansible_check_mode
|
||||||
|
- atlas_borg_repository_path_probe.rc != 0
|
||||||
|
|
||||||
|
- name: Verify the encrypted Atlas Borg repository
|
||||||
|
tags: [atlas, storage, backup, borg]
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv:
|
||||||
|
- /usr/bin/borg
|
||||||
|
- --remote-path
|
||||||
|
- "{{ atlas_borg_remote_path }}"
|
||||||
|
- info
|
||||||
|
- >-
|
||||||
|
ssh://{{ atlas_borg_repository_user }}@{{ atlas_borg_repository_host }}:
|
||||||
|
{{- atlas_borg_repository_port }}/{{ atlas_borg_repository_path }}
|
||||||
|
environment:
|
||||||
|
BORG_CACHE_DIR: "{{ atlas_borg_cache_dir }}"
|
||||||
|
BORG_CONFIG_DIR: "{{ atlas_borg_config_dir }}"
|
||||||
|
BORG_PASSCOMMAND: "cat {{ atlas_borg_passphrase_path }}"
|
||||||
|
BORG_RSH: "{{ atlas_borg_ssh_wrapper_path }}"
|
||||||
|
become: true
|
||||||
|
become_user: "{{ atlas_borg_username }}"
|
||||||
|
changed_when: false
|
||||||
|
when:
|
||||||
|
- atlas_manage_borg_backup | bool
|
||||||
|
- not ansible_check_mode
|
||||||
|
|
||||||
|
- name: Check for the local Atlas Borg recovery-key export
|
||||||
|
tags: [atlas, storage, backup, borg]
|
||||||
|
ansible.builtin.stat:
|
||||||
|
path: "{{ atlas_borg_recovery_export_path }}"
|
||||||
|
register: atlas_borg_recovery_export
|
||||||
|
delegate_to: localhost
|
||||||
|
become: false
|
||||||
|
when:
|
||||||
|
- atlas_manage_borg_backup | bool
|
||||||
|
- not ansible_check_mode
|
||||||
|
|
||||||
|
- name: Export the Atlas Borg recovery key for offline preservation
|
||||||
|
tags: [atlas, storage, backup, borg]
|
||||||
|
when:
|
||||||
|
- atlas_manage_borg_backup | bool
|
||||||
|
- not ansible_check_mode
|
||||||
|
- not atlas_borg_recovery_export.stat.exists
|
||||||
|
no_log: true
|
||||||
|
block:
|
||||||
|
- name: Create the local recovery-material directory
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: "{{ atlas_borg_recovery_export_path | dirname }}"
|
||||||
|
state: directory
|
||||||
|
mode: "0700"
|
||||||
|
delegate_to: localhost
|
||||||
|
become: false
|
||||||
|
|
||||||
|
- name: Export the encrypted Borg repository key on Atlas
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv:
|
||||||
|
- /usr/bin/borg
|
||||||
|
- --remote-path
|
||||||
|
- "{{ atlas_borg_remote_path }}"
|
||||||
|
- key
|
||||||
|
- export
|
||||||
|
- >-
|
||||||
|
ssh://{{ atlas_borg_repository_user }}@{{ atlas_borg_repository_host }}:
|
||||||
|
{{- atlas_borg_repository_port }}/{{ atlas_borg_repository_path }}
|
||||||
|
- "{{ atlas_borg_config_dir }}/atlas-borg-repokey.export"
|
||||||
|
environment:
|
||||||
|
BORG_CACHE_DIR: "{{ atlas_borg_cache_dir }}"
|
||||||
|
BORG_CONFIG_DIR: "{{ atlas_borg_config_dir }}"
|
||||||
|
BORG_PASSCOMMAND: "cat {{ atlas_borg_passphrase_path }}"
|
||||||
|
BORG_RSH: "{{ atlas_borg_ssh_wrapper_path }}"
|
||||||
|
become: true
|
||||||
|
become_user: "{{ atlas_borg_username }}"
|
||||||
|
|
||||||
|
- name: Fetch the encrypted Borg recovery key from Atlas
|
||||||
|
ansible.builtin.fetch:
|
||||||
|
src: "{{ atlas_borg_config_dir }}/atlas-borg-repokey.export"
|
||||||
|
dest: "{{ atlas_borg_recovery_export_path }}"
|
||||||
|
flat: true
|
||||||
|
|
||||||
|
- name: Protect the local Borg recovery-key export
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: "{{ atlas_borg_recovery_export_path }}"
|
||||||
|
mode: "0600"
|
||||||
|
delegate_to: localhost
|
||||||
|
become: false
|
||||||
|
always:
|
||||||
|
- name: Remove the temporary recovery-key export from Atlas
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: "{{ atlas_borg_config_dir }}/atlas-borg-repokey.export"
|
||||||
|
state: absent
|
||||||
|
|
||||||
|
- name: Enable Atlas Borg backup and check timers
|
||||||
|
tags: [atlas, storage, backup, borg]
|
||||||
|
ansible.builtin.systemd:
|
||||||
|
name: "{{ item }}"
|
||||||
|
enabled: true
|
||||||
|
state: started
|
||||||
|
daemon_reload: true
|
||||||
|
loop:
|
||||||
|
- atlas-borg-backup.timer
|
||||||
|
- atlas-borg-check.timer
|
||||||
|
when:
|
||||||
|
- atlas_manage_borg_backup | bool
|
||||||
|
- not ansible_check_mode
|
||||||
159
ansible/roles/profile_atlas/tasks/gitea.yml
Normal file
159
ansible/roles/profile_atlas/tasks/gitea.yml
Normal file
@@ -0,0 +1,159 @@
|
|||||||
|
---
|
||||||
|
- name: Prepare the isolated rootless Atlas Gitea target
|
||||||
|
tags: [atlas, gitea]
|
||||||
|
when: atlas_manage_gitea | bool
|
||||||
|
block:
|
||||||
|
- name: Require the existing Atlas application-data dataset
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- atlas_manage_storage | bool
|
||||||
|
- atlas_gitea_dataset == atlas_zfs_pool ~ '/services/data/gitea'
|
||||||
|
- atlas_gitea_mountpoint == atlas_app_data_mountpoint ~ '/gitea'
|
||||||
|
- atlas_gitea_username == atlas_admin_username
|
||||||
|
- atlas_gitea_group == atlas_admin_group
|
||||||
|
- atlas_gitea_uid | int == atlas_admin_uid | int
|
||||||
|
- atlas_gitea_gid | int == atlas_admin_gid | int
|
||||||
|
- atlas_gitea_container_uid | int == 1000
|
||||||
|
- atlas_gitea_container_gid | int == 1000
|
||||||
|
- atlas_gitea_staging_bind_address == '127.0.0.1'
|
||||||
|
- not (atlas_gitea_production_enabled | bool) or atlas_manage_firewall | bool
|
||||||
|
- not (atlas_gitea_production_enabled | bool) or atlas_gitea_bind_address == ansible_host
|
||||||
|
fail_msg: >-
|
||||||
|
Rootless Gitea requires Atlas storage, the admin user manager, the
|
||||||
|
dedicated dataset, and loopback-only staging ports.
|
||||||
|
|
||||||
|
- name: Inspect the final-restore marker before production activation
|
||||||
|
ansible.builtin.stat:
|
||||||
|
path: "{{ atlas_gitea_mountpoint }}/.final-sha256"
|
||||||
|
register: atlas_gitea_final_marker
|
||||||
|
when: atlas_gitea_production_enabled | bool
|
||||||
|
|
||||||
|
- name: Refuse production activation without the final consistent restore
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- atlas_gitea_final_marker.stat.isreg | default(false)
|
||||||
|
fail_msg: Restore the final stopped-source Gitea export before enabling production.
|
||||||
|
when: atlas_gitea_production_enabled | bool
|
||||||
|
|
||||||
|
- name: Verify the production Gitea dataset belongs to admin
|
||||||
|
ansible.builtin.stat:
|
||||||
|
path: "{{ atlas_gitea_mountpoint }}"
|
||||||
|
register: atlas_gitea_dataset_owner
|
||||||
|
when: atlas_gitea_production_enabled | bool
|
||||||
|
|
||||||
|
- name: Refuse to overlap the legacy host-account service
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- atlas_gitea_dataset_owner.stat.uid | int == atlas_admin_uid | int
|
||||||
|
- atlas_gitea_dataset_owner.stat.gid | int == atlas_admin_gid | int
|
||||||
|
fail_msg: >-
|
||||||
|
The production dataset must already belong to admin before enabling
|
||||||
|
the Quadlet; normal provisioning must not chown an active legacy service.
|
||||||
|
when: atlas_gitea_production_enabled | bool
|
||||||
|
|
||||||
|
- name: Remove the retired account's parent-dataset traverse ACL
|
||||||
|
ansible.posix.acl:
|
||||||
|
path: "{{ item }}"
|
||||||
|
etype: user
|
||||||
|
entity: "{{ atlas_gitea_legacy_username }}"
|
||||||
|
state: absent
|
||||||
|
loop:
|
||||||
|
- "{{ atlas_services_mountpoint }}"
|
||||||
|
- "{{ atlas_app_data_mountpoint }}"
|
||||||
|
when: atlas_gitea_production_enabled | bool
|
||||||
|
|
||||||
|
- name: Enable POSIX ACLs only on the service-namespace parents
|
||||||
|
community.general.zfs:
|
||||||
|
name: "{{ item }}"
|
||||||
|
state: present
|
||||||
|
extra_zfs_properties:
|
||||||
|
acltype: posix
|
||||||
|
loop:
|
||||||
|
- "{{ atlas_zfs_pool }}/{{ atlas_zfs_dataset_services }}"
|
||||||
|
- "{{ atlas_zfs_pool }}/{{ atlas_zfs_dataset_app_data }}"
|
||||||
|
|
||||||
|
- name: Create the dedicated Gitea ZFS dataset
|
||||||
|
community.general.zfs:
|
||||||
|
name: "{{ atlas_gitea_dataset }}"
|
||||||
|
state: present
|
||||||
|
extra_zfs_properties:
|
||||||
|
compression: zstd
|
||||||
|
mountpoint: "{{ atlas_gitea_mountpoint }}"
|
||||||
|
|
||||||
|
- name: Restrict the Gitea dataset and create rootless volume paths
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: "{{ item }}"
|
||||||
|
state: directory
|
||||||
|
owner: "{{ atlas_gitea_username }}"
|
||||||
|
group: "{{ atlas_gitea_group }}"
|
||||||
|
mode: "0700"
|
||||||
|
loop:
|
||||||
|
- "{{ atlas_gitea_mountpoint }}"
|
||||||
|
- "{{ atlas_gitea_mountpoint }}/data"
|
||||||
|
- "{{ atlas_gitea_mountpoint }}/config"
|
||||||
|
- "{{ atlas_gitea_home }}/.config"
|
||||||
|
- "{{ atlas_gitea_home }}/.config/containers"
|
||||||
|
- "{{ atlas_gitea_quadlet_dir }}"
|
||||||
|
|
||||||
|
- name: Ensure lingering for the admin rootless account
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv:
|
||||||
|
- loginctl
|
||||||
|
- enable-linger
|
||||||
|
- "{{ atlas_gitea_username }}"
|
||||||
|
creates: "/var/lib/systemd/linger/{{ atlas_gitea_username }}"
|
||||||
|
|
||||||
|
- name: Start the admin rootless user manager
|
||||||
|
ansible.builtin.systemd:
|
||||||
|
name: "user@{{ atlas_gitea_uid }}.service"
|
||||||
|
state: started
|
||||||
|
when: not ansible_check_mode
|
||||||
|
|
||||||
|
- name: Prepare the admin-owned Gitea image
|
||||||
|
ansible.builtin.import_tasks: gitea_image.yml
|
||||||
|
|
||||||
|
- name: Render the rootless Gitea Quadlet
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: atlas-gitea.container.j2
|
||||||
|
dest: "{{ atlas_gitea_quadlet_dir }}/atlas-gitea.container"
|
||||||
|
owner: "{{ atlas_gitea_username }}"
|
||||||
|
group: "{{ atlas_gitea_group }}"
|
||||||
|
mode: "0644"
|
||||||
|
|
||||||
|
- name: Permit only Aegis to reach production Gitea HTTP and SSH
|
||||||
|
ansible.posix.firewalld:
|
||||||
|
rich_rule: >-
|
||||||
|
rule family="ipv4" source address="{{ atlas_aegis_ip }}"
|
||||||
|
port port="{{ item }}" protocol="tcp" accept
|
||||||
|
zone: "{{ atlas_firewalld_zone }}"
|
||||||
|
state: "{{ 'enabled' if atlas_gitea_production_enabled | bool else 'disabled' }}"
|
||||||
|
permanent: true
|
||||||
|
immediate: true
|
||||||
|
loop:
|
||||||
|
- "{{ atlas_gitea_http_port }}"
|
||||||
|
- "{{ atlas_gitea_ssh_port }}"
|
||||||
|
when: atlas_manage_firewall | bool
|
||||||
|
|
||||||
|
- name: Reload the rootless Gitea user manager without starting Gitea
|
||||||
|
become_user: "{{ atlas_gitea_username }}"
|
||||||
|
ansible.builtin.systemd:
|
||||||
|
scope: user
|
||||||
|
daemon_reload: true
|
||||||
|
environment:
|
||||||
|
XDG_RUNTIME_DIR: "/run/user/{{ atlas_gitea_uid }}"
|
||||||
|
DBUS_SESSION_BUS_ADDRESS: "unix:path=/run/user/{{ atlas_gitea_uid }}/bus"
|
||||||
|
when: not ansible_check_mode
|
||||||
|
|
||||||
|
- name: Start and enable the rootless Gitea user Quadlet after final restore
|
||||||
|
become_user: "{{ atlas_gitea_username }}"
|
||||||
|
ansible.builtin.systemd:
|
||||||
|
name: atlas-gitea.service
|
||||||
|
scope: user
|
||||||
|
state: started
|
||||||
|
enabled: true
|
||||||
|
environment:
|
||||||
|
XDG_RUNTIME_DIR: "/run/user/{{ atlas_gitea_uid }}"
|
||||||
|
DBUS_SESSION_BUS_ADDRESS: "unix:path=/run/user/{{ atlas_gitea_uid }}/bus"
|
||||||
|
when:
|
||||||
|
- atlas_gitea_production_enabled | bool
|
||||||
|
- not ansible_check_mode
|
||||||
51
ansible/roles/profile_atlas/tasks/gitea_image.yml
Normal file
51
ansible/roles/profile_atlas/tasks/gitea_image.yml
Normal file
@@ -0,0 +1,51 @@
|
|||||||
|
---
|
||||||
|
- name: Create the admin-owned Gitea image build directory
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: "{{ atlas_gitea_image_build_dir }}"
|
||||||
|
state: directory
|
||||||
|
owner: "{{ atlas_admin_username }}"
|
||||||
|
group: "{{ atlas_admin_group }}"
|
||||||
|
mode: "0700"
|
||||||
|
|
||||||
|
- name: Install the pinned rootless Gitea Containerfile
|
||||||
|
ansible.builtin.copy:
|
||||||
|
src: Containerfile.gitea-rootless
|
||||||
|
dest: "{{ atlas_gitea_image_build_dir }}/Containerfile"
|
||||||
|
owner: "{{ atlas_admin_username }}"
|
||||||
|
group: "{{ atlas_admin_group }}"
|
||||||
|
mode: "0644"
|
||||||
|
|
||||||
|
- name: Check the admin-owned Gitea image
|
||||||
|
become_user: "{{ atlas_admin_username }}"
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: [podman, image, exists, "{{ atlas_gitea_image }}"]
|
||||||
|
args:
|
||||||
|
chdir: "{{ atlas_gitea_image_build_dir }}"
|
||||||
|
environment:
|
||||||
|
HOME: "{{ atlas_admin_home }}"
|
||||||
|
XDG_RUNTIME_DIR: "/run/user/{{ atlas_admin_uid }}"
|
||||||
|
register: atlas_gitea_image_present
|
||||||
|
changed_when: false
|
||||||
|
failed_when: false
|
||||||
|
check_mode: false
|
||||||
|
|
||||||
|
- name: Build the pinned Gitea image with the internal gitea identity
|
||||||
|
become_user: "{{ atlas_admin_username }}"
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv:
|
||||||
|
- podman
|
||||||
|
- build
|
||||||
|
- --pull=always
|
||||||
|
- --tag
|
||||||
|
- "{{ atlas_gitea_image }}"
|
||||||
|
- --file
|
||||||
|
- Containerfile
|
||||||
|
- .
|
||||||
|
args:
|
||||||
|
chdir: "{{ atlas_gitea_image_build_dir }}"
|
||||||
|
environment:
|
||||||
|
HOME: "{{ atlas_admin_home }}"
|
||||||
|
XDG_RUNTIME_DIR: "/run/user/{{ atlas_admin_uid }}"
|
||||||
|
when:
|
||||||
|
- atlas_gitea_image_present.rc != 0
|
||||||
|
- not ansible_check_mode
|
||||||
58
ansible/roles/profile_atlas/tasks/gitea_public_domain.yml
Normal file
58
ansible/roles/profile_atlas/tasks/gitea_public_domain.yml
Normal file
@@ -0,0 +1,58 @@
|
|||||||
|
---
|
||||||
|
- name: Manage the public domain of the restored production Gitea
|
||||||
|
tags: [atlas, gitea, gitea_public_domain]
|
||||||
|
when:
|
||||||
|
- atlas_manage_gitea | bool
|
||||||
|
- atlas_gitea_production_enabled | bool
|
||||||
|
- atlas_gitea_public_domain | length > 0
|
||||||
|
block:
|
||||||
|
- name: Require an explicit public Gitea hostname
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- atlas_gitea_public_domain is match('^[a-zA-Z0-9][a-zA-Z0-9.-]*\.[a-zA-Z]{2,}$')
|
||||||
|
|
||||||
|
- name: Inspect the restored private Gitea configuration
|
||||||
|
ansible.builtin.stat:
|
||||||
|
path: "{{ atlas_gitea_mountpoint }}/config/app.ini"
|
||||||
|
follow: false
|
||||||
|
register: atlas_gitea_public_config
|
||||||
|
|
||||||
|
- name: Refuse to create or replace an unprepared Gitea configuration
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- atlas_gitea_public_config.stat.isreg | default(false)
|
||||||
|
- atlas_gitea_public_config.stat.uid | int == atlas_gitea_uid | int
|
||||||
|
- atlas_gitea_public_config.stat.mode == '0600'
|
||||||
|
|
||||||
|
# app.ini contains secrets: preserve all unrelated settings and suppress diffs.
|
||||||
|
- name: Set only the declared public Gitea server fields
|
||||||
|
community.general.ini_file:
|
||||||
|
path: "{{ atlas_gitea_mountpoint }}/config/app.ini"
|
||||||
|
section: server
|
||||||
|
option: "{{ item.option }}"
|
||||||
|
value: "{{ item.value }}"
|
||||||
|
create: false
|
||||||
|
backup: true
|
||||||
|
owner: "{{ atlas_gitea_username }}"
|
||||||
|
group: "{{ atlas_gitea_group }}"
|
||||||
|
mode: "0600"
|
||||||
|
loop:
|
||||||
|
- { option: DOMAIN, value: "{{ atlas_gitea_public_domain }}" }
|
||||||
|
- { option: ROOT_URL, value: "https://{{ atlas_gitea_public_domain }}/" }
|
||||||
|
- { option: SSH_DOMAIN, value: "{{ atlas_gitea_public_domain }}" }
|
||||||
|
register: atlas_gitea_public_domain_update
|
||||||
|
no_log: true
|
||||||
|
diff: false
|
||||||
|
|
||||||
|
- name: Restart only Gitea when its public configuration changes
|
||||||
|
become_user: "{{ atlas_gitea_username }}"
|
||||||
|
ansible.builtin.systemd:
|
||||||
|
name: atlas-gitea.service
|
||||||
|
scope: user
|
||||||
|
state: restarted
|
||||||
|
environment:
|
||||||
|
XDG_RUNTIME_DIR: "/run/user/{{ atlas_gitea_uid }}"
|
||||||
|
DBUS_SESSION_BUS_ADDRESS: "unix:path=/run/user/{{ atlas_gitea_uid }}/bus"
|
||||||
|
when:
|
||||||
|
- atlas_gitea_public_domain_update is changed
|
||||||
|
- not ansible_check_mode
|
||||||
188
ansible/roles/profile_atlas/tasks/icloudpd.yml
Normal file
188
ansible/roles/profile_atlas/tasks/icloudpd.yml
Normal file
@@ -0,0 +1,188 @@
|
|||||||
|
---
|
||||||
|
- name: Require exact Atlas iCloudPD paths and rootless identity
|
||||||
|
tags: [atlas, icloudpd]
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- atlas_manage_storage | bool
|
||||||
|
- atlas_icloudpd_dataset == atlas_zfs_pool ~ '/services/data/icloudpd'
|
||||||
|
- atlas_icloudpd_state_dir == atlas_app_data_mountpoint ~ '/icloudpd'
|
||||||
|
- atlas_icloudpd_config_dir == atlas_icloudpd_state_dir ~ '/config'
|
||||||
|
- atlas_icloudpd_photos_dir == atlas_archive_mountpoint ~ '/Pictures/iCloudPD'
|
||||||
|
- atlas_admin_uid | int == 1000
|
||||||
|
- atlas_admin_gid | int == 1000
|
||||||
|
- atlas_icloudpd_image is search('@sha256:[0-9a-f]{64}$')
|
||||||
|
fail_msg: Verify the fixed, separate Atlas iCloudPD photo and state paths.
|
||||||
|
|
||||||
|
- name: Declare rootless Atlas iCloudPD storage and boot-started Quadlet
|
||||||
|
tags: [atlas, icloudpd]
|
||||||
|
block:
|
||||||
|
- name: Inspect the existing Archive and application-data datasets
|
||||||
|
community.general.zfs_facts:
|
||||||
|
name: "{{ item.dataset }}"
|
||||||
|
properties: name,mounted,mountpoint
|
||||||
|
loop:
|
||||||
|
- dataset: "{{ atlas_zfs_pool }}/{{ atlas_zfs_dataset_archive }}"
|
||||||
|
mountpoint: "{{ atlas_archive_mountpoint }}"
|
||||||
|
- dataset: "{{ atlas_zfs_pool }}/{{ atlas_zfs_dataset_app_data }}"
|
||||||
|
mountpoint: "{{ atlas_app_data_mountpoint }}"
|
||||||
|
loop_control:
|
||||||
|
label: "{{ item.dataset }}"
|
||||||
|
register: atlas_icloudpd_parent_datasets
|
||||||
|
|
||||||
|
- name: Refuse missing or unmounted iCloudPD parent datasets
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- item.ansible_facts.ansible_zfs_datasets | length == 1
|
||||||
|
- item.ansible_facts.ansible_zfs_datasets[0].mounted == 'yes'
|
||||||
|
- item.ansible_facts.ansible_zfs_datasets[0].mountpoint == item.item.mountpoint
|
||||||
|
loop: "{{ atlas_icloudpd_parent_datasets.results }}"
|
||||||
|
loop_control:
|
||||||
|
label: "{{ item.item.dataset }}"
|
||||||
|
|
||||||
|
- name: Inspect the existing Pictures namespace and proposed target
|
||||||
|
ansible.builtin.stat:
|
||||||
|
path: "{{ item }}"
|
||||||
|
follow: false
|
||||||
|
loop:
|
||||||
|
- "{{ atlas_archive_mountpoint }}/Pictures"
|
||||||
|
- "{{ atlas_icloudpd_photos_dir }}"
|
||||||
|
- "{{ atlas_icloudpd_photos_dir }}/.atlas-icloudpd-managed"
|
||||||
|
register: atlas_icloudpd_photo_paths
|
||||||
|
|
||||||
|
- name: Refuse to adopt unrelated Pictures data or a symlink
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- atlas_icloudpd_photo_paths.results[0].stat.isdir | default(false)
|
||||||
|
- atlas_icloudpd_photo_paths.results[0].stat.uid | int == atlas_admin_uid | int
|
||||||
|
- >-
|
||||||
|
not atlas_icloudpd_photo_paths.results[1].stat.exists or
|
||||||
|
(atlas_icloudpd_photo_paths.results[1].stat.isdir | default(false) and
|
||||||
|
atlas_icloudpd_photo_paths.results[2].stat.isreg | default(false))
|
||||||
|
fail_msg: >-
|
||||||
|
Pictures must exist and be admin-owned; an existing iCloudPD target
|
||||||
|
must carry its managed marker. Never adopt or replace unrelated data.
|
||||||
|
|
||||||
|
- name: Create a dedicated ZFS dataset for iCloudPD configuration and MFA
|
||||||
|
community.general.zfs:
|
||||||
|
name: "{{ atlas_icloudpd_dataset }}"
|
||||||
|
state: present
|
||||||
|
extra_zfs_properties:
|
||||||
|
compression: zstd
|
||||||
|
mountpoint: "{{ atlas_icloudpd_state_dir }}"
|
||||||
|
|
||||||
|
- name: Restrict iCloudPD state and the new photo subtree
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: "{{ item.path }}"
|
||||||
|
state: directory
|
||||||
|
owner: "{{ atlas_admin_username }}"
|
||||||
|
group: "{{ atlas_admin_group }}"
|
||||||
|
mode: "{{ item.mode }}"
|
||||||
|
loop:
|
||||||
|
- path: "{{ atlas_icloudpd_state_dir }}"
|
||||||
|
mode: "0700"
|
||||||
|
- path: "{{ atlas_icloudpd_config_dir }}"
|
||||||
|
mode: "0700"
|
||||||
|
- path: "{{ atlas_icloudpd_photos_dir }}"
|
||||||
|
mode: "0750"
|
||||||
|
- path: "{{ atlas_icloudpd_quadlet_dir }}"
|
||||||
|
mode: "0700"
|
||||||
|
loop_control:
|
||||||
|
label: "{{ item.path }}"
|
||||||
|
|
||||||
|
- name: Mark only the newly managed iCloudPD photo subtree
|
||||||
|
ansible.builtin.copy:
|
||||||
|
content: "Atlas iCloudPD photo subtree; do not remove source photos.\n"
|
||||||
|
dest: "{{ atlas_icloudpd_photos_dir }}/.atlas-icloudpd-managed"
|
||||||
|
owner: "{{ atlas_admin_username }}"
|
||||||
|
group: "{{ atlas_admin_group }}"
|
||||||
|
mode: "0600"
|
||||||
|
force: false
|
||||||
|
|
||||||
|
- name: Install the image's required mounted-filesystem failsafe
|
||||||
|
ansible.builtin.copy:
|
||||||
|
content: ""
|
||||||
|
dest: "{{ atlas_icloudpd_photos_dir }}/.mounted"
|
||||||
|
owner: "{{ atlas_admin_username }}"
|
||||||
|
group: "{{ atlas_admin_group }}"
|
||||||
|
mode: "0644"
|
||||||
|
force: false
|
||||||
|
|
||||||
|
- name: Require the Vault-backed iCloudPD Apple ID
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- vault_atlas_icloudpd_apple_id is defined
|
||||||
|
- vault_atlas_icloudpd_apple_id | length > 0
|
||||||
|
- vault_atlas_icloudpd_apple_id != 'REPLACE_ME'
|
||||||
|
- vault_atlas_icloudpd_apple_id.splitlines() | length == 1
|
||||||
|
fail_msg: Configure the existing iCloudPD Apple ID in Vault.
|
||||||
|
no_log: true
|
||||||
|
|
||||||
|
- name: Seed private Atlas iCloudPD configuration when absent
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: atlas-icloudpd.conf.j2
|
||||||
|
dest: "{{ atlas_icloudpd_config_dir }}/icloudpd.conf"
|
||||||
|
owner: "{{ atlas_admin_username }}"
|
||||||
|
group: "{{ atlas_admin_group }}"
|
||||||
|
mode: "0600"
|
||||||
|
force: false
|
||||||
|
no_log: true
|
||||||
|
diff: false
|
||||||
|
|
||||||
|
- name: Keep declared iCloudPD options in the image-managed configuration
|
||||||
|
ansible.builtin.lineinfile:
|
||||||
|
path: "{{ atlas_icloudpd_config_dir }}/icloudpd.conf"
|
||||||
|
regexp: "^{{ item.key }}="
|
||||||
|
line: "{{ item.key }}={{ item.value }}"
|
||||||
|
owner: "{{ atlas_admin_username }}"
|
||||||
|
group: "{{ atlas_admin_group }}"
|
||||||
|
mode: "0600"
|
||||||
|
loop:
|
||||||
|
- {key: apple_id, value: "{{ vault_atlas_icloudpd_apple_id }}"}
|
||||||
|
- {key: authentication_type, value: MFA}
|
||||||
|
- {key: user, value: user}
|
||||||
|
- {key: user_id, value: "1000"}
|
||||||
|
- {key: group, value: group}
|
||||||
|
- {key: group_id, value: "1000"}
|
||||||
|
- {key: download_path, value: /home/user/iCloud}
|
||||||
|
- {key: folder_structure, value: "{:%Y/%m/%d}"}
|
||||||
|
- {key: directory_permissions, value: "750"}
|
||||||
|
- {key: file_permissions, value: "640"}
|
||||||
|
- {key: download_interval, value: "86400"}
|
||||||
|
- {key: auto_delete, value: "false"}
|
||||||
|
- {key: delete_after_download, value: "false"}
|
||||||
|
loop_control:
|
||||||
|
label: "{{ item.key }}"
|
||||||
|
no_log: true
|
||||||
|
diff: false
|
||||||
|
|
||||||
|
- name: Render the rootless Atlas iCloudPD Quadlet
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: atlas-icloudpd.container.j2
|
||||||
|
dest: "{{ atlas_icloudpd_quadlet_dir }}/atlas-icloudpd.container"
|
||||||
|
owner: "{{ atlas_admin_username }}"
|
||||||
|
group: "{{ atlas_admin_group }}"
|
||||||
|
mode: "0644"
|
||||||
|
register: atlas_icloudpd_quadlet
|
||||||
|
|
||||||
|
- name: Reload the Atlas admin user manager after iCloudPD Quadlet changes
|
||||||
|
become_user: "{{ atlas_admin_username }}"
|
||||||
|
ansible.builtin.systemd:
|
||||||
|
scope: user
|
||||||
|
daemon_reload: true
|
||||||
|
environment:
|
||||||
|
XDG_RUNTIME_DIR: "/run/user/{{ atlas_admin_uid }}"
|
||||||
|
DBUS_SESSION_BUS_ADDRESS: "unix:path=/run/user/{{ atlas_admin_uid }}/bus"
|
||||||
|
when:
|
||||||
|
- atlas_icloudpd_quadlet.changed
|
||||||
|
- not ansible_check_mode
|
||||||
|
|
||||||
|
- name: Keep the rootless Atlas iCloudPD service running
|
||||||
|
become_user: "{{ atlas_admin_username }}"
|
||||||
|
ansible.builtin.systemd:
|
||||||
|
name: atlas-icloudpd.service
|
||||||
|
scope: user
|
||||||
|
state: started
|
||||||
|
environment:
|
||||||
|
XDG_RUNTIME_DIR: "/run/user/{{ atlas_admin_uid }}"
|
||||||
|
DBUS_SESSION_BUS_ADDRESS: "unix:path=/run/user/{{ atlas_admin_uid }}/bus"
|
||||||
|
when: not ansible_check_mode
|
||||||
@@ -14,6 +14,42 @@
|
|||||||
- name: Import Atlas storage tasks
|
- name: Import Atlas storage tasks
|
||||||
ansible.builtin.import_tasks: storage.yml
|
ansible.builtin.import_tasks: storage.yml
|
||||||
|
|
||||||
|
- name: Import staged Atlas rootless Gitea tasks
|
||||||
|
ansible.builtin.import_tasks: gitea.yml
|
||||||
|
|
||||||
|
- name: Import the declared Atlas Gitea public domain
|
||||||
|
ansible.builtin.import_tasks: gitea_public_domain.yml
|
||||||
|
|
||||||
|
- name: Import Atlas Nextcloud steady-state stack
|
||||||
|
ansible.builtin.import_tasks: nextcloud.yml
|
||||||
|
|
||||||
|
- name: Import Atlas iCloudPD storage and boot-started Quadlet tasks
|
||||||
|
ansible.builtin.import_tasks: icloudpd.yml
|
||||||
|
|
||||||
|
- name: Import Atlas ZFS maintenance tasks
|
||||||
|
ansible.builtin.import_tasks: zfs_maintenance.yml
|
||||||
|
|
||||||
|
- name: Import Atlas Borg backup tasks
|
||||||
|
ansible.builtin.import_tasks: borg_backup.yml
|
||||||
|
|
||||||
|
- name: Import Atlas offline USB backup tasks
|
||||||
|
ansible.builtin.import_tasks: usb_backup.yml
|
||||||
|
|
||||||
|
- name: Import recurring Nextcloud backup preparation
|
||||||
|
ansible.builtin.import_tasks: nextcloud_backup.yml
|
||||||
|
|
||||||
|
- name: Import Atlas Prometheus backup pull identity tasks
|
||||||
|
ansible.builtin.import_tasks: prometheus_pull_identity.yml
|
||||||
|
|
||||||
|
- name: Import Atlas Prometheus backup pull job tasks
|
||||||
|
ansible.builtin.import_tasks: prometheus_pull_job.yml
|
||||||
|
|
||||||
|
- name: Import Atlas health monitoring tasks
|
||||||
|
ansible.builtin.import_tasks: monitoring.yml
|
||||||
|
|
||||||
|
- name: Import Atlas post-restore SELinux relabeling tasks
|
||||||
|
ansible.builtin.import_tasks: restorecon.yml
|
||||||
|
|
||||||
- name: Import Atlas file sharing tasks
|
- name: Import Atlas file sharing tasks
|
||||||
ansible.builtin.import_tasks: sharing.yml
|
ansible.builtin.import_tasks: sharing.yml
|
||||||
|
|
||||||
|
|||||||
201
ansible/roles/profile_atlas/tasks/monitoring.yml
Normal file
201
ansible/roles/profile_atlas/tasks/monitoring.yml
Normal file
@@ -0,0 +1,201 @@
|
|||||||
|
---
|
||||||
|
- name: Validate Atlas health monitoring policy
|
||||||
|
tags: [atlas, monitoring]
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- atlas_manage_storage | bool
|
||||||
|
- atlas_zfs_pool != 'CHANGEME_ZFS_POOL'
|
||||||
|
- atlas_monitor_calendar | length > 0
|
||||||
|
- atlas_monitor_smart_devices | length > 0
|
||||||
|
- atlas_monitor_effective_timers | length > 0
|
||||||
|
- atlas_monitor_effective_failure_units | length > 0
|
||||||
|
- atlas_monitor_remote_capacity.user == atlas_borg_repository_user
|
||||||
|
- atlas_monitor_remote_capacity.host == atlas_borg_repository_host
|
||||||
|
- atlas_monitor_remote_capacity.run_as == atlas_borg_username
|
||||||
|
- atlas_monitor_remote_capacity.ssh_wrapper == atlas_borg_ssh_wrapper_path
|
||||||
|
- >-
|
||||||
|
0 < atlas_monitor_remote_capacity.warning_percent | int
|
||||||
|
< atlas_monitor_remote_capacity.critical_percent | int < 100
|
||||||
|
- atlas_monitor_remote_capacity.growth_warning_gib_day | int > 0
|
||||||
|
- atlas_monitor_notifier.startswith('/opt/45drives/houston/')
|
||||||
|
- 0 < atlas_monitor_pool_warning_percent | int < atlas_monitor_pool_critical_percent | int < 100
|
||||||
|
- 0 < atlas_monitor_root_warning_percent | int < atlas_monitor_root_critical_percent | int < 100
|
||||||
|
- 0 < atlas_monitor_snapshot_warning_percent | int < atlas_monitor_snapshot_critical_percent | int < 100
|
||||||
|
- atlas_monitor_snapshot_growth_warning_gib_day | int > 0
|
||||||
|
- atlas_monitor_backup_growth_warning_gib_day | int > 0
|
||||||
|
- 0 < atlas_monitor_cpu_warning_c | int < atlas_monitor_cpu_critical_c | int
|
||||||
|
- atlas_monitor_borg_max_runtime_days | int > 0
|
||||||
|
fail_msg: >-
|
||||||
|
Atlas health monitoring needs real devices, job units, a valid calendar,
|
||||||
|
positive ordered thresholds, and the existing Houston notifier.
|
||||||
|
when: atlas_manage_monitoring | bool
|
||||||
|
|
||||||
|
- name: Validate monitored Atlas SMART devices
|
||||||
|
tags: [atlas, monitoring]
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- item.name is match('^[a-z0-9][a-z0-9_-]*$')
|
||||||
|
- item.path.startswith('/dev/disk/by-id/')
|
||||||
|
- 0 < item.warning_c | int < item.critical_c | int
|
||||||
|
fail_msg: "Every monitored disk needs a stable by-id path and ordered temperature thresholds."
|
||||||
|
loop: "{{ atlas_monitor_smart_devices }}"
|
||||||
|
loop_control:
|
||||||
|
label: "{{ item.name }}"
|
||||||
|
when: atlas_manage_monitoring | bool
|
||||||
|
|
||||||
|
- name: Validate monitored Atlas timer names and age thresholds
|
||||||
|
tags: [atlas, monitoring]
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- item.name is match('^[a-zA-Z0-9@_.-]+\\.timer$')
|
||||||
|
- item.max_age_hours | int >= 0
|
||||||
|
loop: "{{ atlas_monitor_effective_timers }}"
|
||||||
|
loop_control:
|
||||||
|
label: "{{ item.name }}"
|
||||||
|
when: atlas_manage_monitoring | bool
|
||||||
|
|
||||||
|
- name: Validate monitored Atlas failure unit names
|
||||||
|
tags: [atlas, monitoring]
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- item is match('^[a-zA-Z0-9@_.-]+\\.service$')
|
||||||
|
loop: "{{ atlas_monitor_effective_failure_units }}"
|
||||||
|
when: atlas_manage_monitoring | bool
|
||||||
|
|
||||||
|
- name: Validate Atlas health monitor calendar
|
||||||
|
tags: [atlas, monitoring]
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: [systemd-analyze, calendar, "{{ atlas_monitor_calendar }}"]
|
||||||
|
changed_when: false
|
||||||
|
check_mode: false
|
||||||
|
when: atlas_manage_monitoring | bool
|
||||||
|
|
||||||
|
- name: Install SMART tooling for Atlas health checks
|
||||||
|
tags: [atlas, monitoring, packages]
|
||||||
|
ansible.builtin.dnf:
|
||||||
|
name: smartmontools
|
||||||
|
state: present
|
||||||
|
when: atlas_manage_monitoring | bool
|
||||||
|
|
||||||
|
- name: Inspect the existing 45Drives notifier for monitoring
|
||||||
|
tags: [atlas, monitoring]
|
||||||
|
ansible.builtin.stat:
|
||||||
|
path: "{{ atlas_monitor_notifier }}"
|
||||||
|
register: atlas_monitor_notifier_file
|
||||||
|
when: atlas_manage_monitoring | bool
|
||||||
|
|
||||||
|
- name: Require the existing 45Drives notifier for monitoring
|
||||||
|
tags: [atlas, monitoring]
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- atlas_monitor_notifier_file.stat.executable | default(false)
|
||||||
|
fail_msg: "The existing 45Drives Houston notifier must be executable."
|
||||||
|
when: atlas_manage_monitoring | bool
|
||||||
|
|
||||||
|
- name: Create private Atlas health monitor state directory
|
||||||
|
tags: [atlas, monitoring]
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: /var/lib/atlas-health-monitor
|
||||||
|
state: directory
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0700"
|
||||||
|
when: atlas_manage_monitoring | bool
|
||||||
|
|
||||||
|
- name: Install Atlas health monitor configuration
|
||||||
|
tags: [atlas, monitoring]
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: atlas-health-monitor.json.j2
|
||||||
|
dest: /etc/atlas-health-monitor.json
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0600"
|
||||||
|
when: atlas_manage_monitoring | bool
|
||||||
|
|
||||||
|
- name: Install Atlas health monitor helper
|
||||||
|
tags: [atlas, monitoring]
|
||||||
|
ansible.builtin.copy:
|
||||||
|
src: atlas-health-monitor.py
|
||||||
|
dest: /usr/local/libexec/atlas-health-monitor
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0750"
|
||||||
|
when: atlas_manage_monitoring | bool
|
||||||
|
|
||||||
|
- name: Install Atlas health monitoring units
|
||||||
|
tags: [atlas, monitoring]
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: "{{ item }}.j2"
|
||||||
|
dest: "/etc/systemd/system/{{ item }}"
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0644"
|
||||||
|
loop:
|
||||||
|
- atlas-health-monitor.service
|
||||||
|
- atlas-health-monitor.timer
|
||||||
|
- atlas-monitor-failure@.service
|
||||||
|
when: atlas_manage_monitoring | bool
|
||||||
|
|
||||||
|
- name: Create failure hook directories for monitored Atlas jobs
|
||||||
|
tags: [atlas, monitoring]
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: "/etc/systemd/system/{{ item }}.d"
|
||||||
|
state: directory
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0755"
|
||||||
|
loop: "{{ atlas_monitor_effective_failure_units }}"
|
||||||
|
when: atlas_manage_monitoring | bool
|
||||||
|
|
||||||
|
- name: Notify 45Drives Alerts when an Atlas job fails
|
||||||
|
tags: [atlas, monitoring]
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: atlas-monitor-failure.conf.j2
|
||||||
|
dest: "/etc/systemd/system/{{ item }}.d/atlas-monitor.conf"
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0644"
|
||||||
|
loop: "{{ atlas_monitor_effective_failure_units }}"
|
||||||
|
when: atlas_manage_monitoring | bool
|
||||||
|
|
||||||
|
- name: Reload systemd after installing Atlas monitoring
|
||||||
|
tags: [atlas, monitoring]
|
||||||
|
ansible.builtin.systemd:
|
||||||
|
daemon_reload: true
|
||||||
|
when:
|
||||||
|
- atlas_manage_monitoring | bool
|
||||||
|
- not ansible_check_mode
|
||||||
|
|
||||||
|
- name: Enable the Atlas health monitoring timer
|
||||||
|
tags: [atlas, monitoring]
|
||||||
|
ansible.builtin.systemd:
|
||||||
|
name: atlas-health-monitor.timer
|
||||||
|
enabled: true
|
||||||
|
state: started
|
||||||
|
when:
|
||||||
|
- atlas_manage_monitoring | bool
|
||||||
|
- not ansible_check_mode
|
||||||
|
|
||||||
|
- name: Validate the deployed Atlas health monitoring units
|
||||||
|
tags: [atlas, monitoring]
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv:
|
||||||
|
- systemd-analyze
|
||||||
|
- verify
|
||||||
|
- atlas-health-monitor.service
|
||||||
|
- atlas-health-monitor.timer
|
||||||
|
- atlas-monitor-failure@.service
|
||||||
|
changed_when: false
|
||||||
|
when:
|
||||||
|
- atlas_manage_monitoring | bool
|
||||||
|
- not ansible_check_mode
|
||||||
|
|
||||||
|
- name: Probe Atlas health without sending notifications
|
||||||
|
tags: [atlas, monitoring]
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: [/usr/local/libexec/atlas-health-monitor, --dry-run]
|
||||||
|
register: atlas_monitor_dry_run
|
||||||
|
changed_when: false
|
||||||
|
when:
|
||||||
|
- atlas_manage_monitoring | bool
|
||||||
|
- not ansible_check_mode
|
||||||
328
ansible/roles/profile_atlas/tasks/nextcloud.yml
Normal file
328
ansible/roles/profile_atlas/tasks/nextcloud.yml
Normal file
@@ -0,0 +1,328 @@
|
|||||||
|
---
|
||||||
|
- name: Manage the empty Atlas Nextcloud and ONLYOFFICE stack
|
||||||
|
tags: [atlas, nextcloud]
|
||||||
|
when: atlas_manage_nextcloud | bool
|
||||||
|
block:
|
||||||
|
- name: Validate dedicated paths, domains and pinned images
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- atlas_manage_storage | bool
|
||||||
|
- atlas_manage_firewall | bool
|
||||||
|
- atlas_nextcloud_root == atlas_app_data_mountpoint ~ '/nextcloud'
|
||||||
|
- atlas_nextcloud_dataset == atlas_zfs_pool ~ '/services/data/nextcloud'
|
||||||
|
- atlas_nextcloud_domain is match('^[a-z0-9.-]+$')
|
||||||
|
- atlas_onlyoffice_domain is match('^[a-z0-9.-]+$')
|
||||||
|
- atlas_nextcloud_domain != atlas_onlyoffice_domain
|
||||||
|
- atlas_nextcloud_http_port | int > 1024
|
||||||
|
- atlas_onlyoffice_http_port | int > 1024
|
||||||
|
- atlas_nextcloud_http_port != atlas_onlyoffice_http_port
|
||||||
|
- "['calendar', 'contacts', 'onlyoffice', 'groupfolders'] | difference(atlas_nextcloud_apps | map(attribute='id') | list) | length == 0"
|
||||||
|
- atlas_nextcloud_users | length > 0
|
||||||
|
- atlas_nextcloud_admin not in (atlas_nextcloud_users | map(attribute='username') | list)
|
||||||
|
- atlas_nextcloud_users | map(attribute='username') | unique | list | length == atlas_nextcloud_users | length
|
||||||
|
- item is search('@sha256:[0-9a-f]{64}$')
|
||||||
|
loop:
|
||||||
|
- "{{ atlas_nextcloud_image }}"
|
||||||
|
- "{{ atlas_nextcloud_postgres_image }}"
|
||||||
|
- "{{ atlas_nextcloud_redis_image }}"
|
||||||
|
- "{{ atlas_onlyoffice_image }}"
|
||||||
|
|
||||||
|
- name: Require dedicated Vault secrets without exposing them
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- item | default('') is match('^[a-zA-Z0-9]{32,}$')
|
||||||
|
loop: >-
|
||||||
|
{{ [vault_nextcloud_database_password | default(''),
|
||||||
|
vault_nextcloud_redis_password | default(''),
|
||||||
|
vault_nextcloud_admin_password | default(''),
|
||||||
|
vault_nextcloud_onlyoffice_jwt | default('')] +
|
||||||
|
(atlas_nextcloud_users | map(attribute='password') | list) }}
|
||||||
|
no_log: true
|
||||||
|
|
||||||
|
- name: Prepare access to declared existing Archive directories
|
||||||
|
ansible.builtin.include_tasks: nextcloud_external_access.yml
|
||||||
|
when: atlas_nextcloud_external_mounts | length > 0
|
||||||
|
|
||||||
|
- name: Verify the existing application-data parent is mounted
|
||||||
|
community.general.zfs_facts:
|
||||||
|
name: "{{ atlas_zfs_pool }}/{{ atlas_zfs_dataset_app_data }}"
|
||||||
|
properties: name,mounted,mountpoint
|
||||||
|
register: atlas_nextcloud_parent
|
||||||
|
|
||||||
|
- name: Require the verified application-data parent
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- atlas_nextcloud_parent.ansible_facts.ansible_zfs_datasets | length == 1
|
||||||
|
- atlas_nextcloud_parent.ansible_facts.ansible_zfs_datasets[0].mounted == 'yes'
|
||||||
|
- atlas_nextcloud_parent.ansible_facts.ansible_zfs_datasets[0].mountpoint == atlas_app_data_mountpoint
|
||||||
|
|
||||||
|
- name: Create the dedicated Nextcloud namespace and component datasets
|
||||||
|
community.general.zfs:
|
||||||
|
name: "{{ atlas_nextcloud_dataset }}{{ item }}"
|
||||||
|
state: present
|
||||||
|
extra_zfs_properties:
|
||||||
|
compression: zstd
|
||||||
|
mountpoint: "{{ atlas_nextcloud_root }}{{ item }}"
|
||||||
|
loop: ['', /app, /files, /database, /cache, /office]
|
||||||
|
|
||||||
|
- name: Inspect component directories before seeding ownership
|
||||||
|
ansible.builtin.stat:
|
||||||
|
path: "{{ atlas_nextcloud_root }}{{ item }}"
|
||||||
|
follow: false
|
||||||
|
get_checksum: false
|
||||||
|
loop: [/app, /files, /database, /cache, /office]
|
||||||
|
register: atlas_nextcloud_component_paths
|
||||||
|
|
||||||
|
- name: Seed only root-owned new dataset roots without recursive ownership changes
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: "{{ item.stat.path }}"
|
||||||
|
state: directory
|
||||||
|
owner: "{{ atlas_admin_username }}"
|
||||||
|
group: "{{ atlas_admin_group }}"
|
||||||
|
mode: "0700"
|
||||||
|
loop: "{{ atlas_nextcloud_component_paths.results }}"
|
||||||
|
loop_control:
|
||||||
|
label: "{{ item.item }}"
|
||||||
|
when:
|
||||||
|
- item.stat.exists
|
||||||
|
- item.stat.uid | default(-1) | int == 0
|
||||||
|
|
||||||
|
- name: Ensure private rootless stack configuration directories exist
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: "{{ item.path }}"
|
||||||
|
state: directory
|
||||||
|
owner: "{{ atlas_admin_username }}"
|
||||||
|
group: "{{ atlas_admin_group }}"
|
||||||
|
mode: "{{ item.mode }}"
|
||||||
|
loop:
|
||||||
|
- {path: "{{ atlas_nextcloud_private_dir }}", mode: "0700"}
|
||||||
|
- {path: "{{ atlas_nextcloud_app_cache }}", mode: "0755"}
|
||||||
|
- {path: "{{ atlas_nextcloud_quadlet_dir }}", mode: "0700"}
|
||||||
|
- {path: "{{ atlas_admin_home }}/.config/systemd/user", mode: "0700"}
|
||||||
|
|
||||||
|
- name: Inspect the dedicated ONLYOFFICE bind directories
|
||||||
|
ansible.builtin.stat:
|
||||||
|
path: "{{ atlas_nextcloud_root }}/office/{{ item }}"
|
||||||
|
follow: false
|
||||||
|
get_checksum: false
|
||||||
|
loop: [data, lib, logs, database]
|
||||||
|
register: atlas_onlyoffice_bind_paths
|
||||||
|
|
||||||
|
- name: Create ONLYOFFICE bind directories only when absent
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: "{{ item.invocation.module_args.path }}"
|
||||||
|
state: directory
|
||||||
|
owner: "{{ atlas_admin_username }}"
|
||||||
|
group: "{{ atlas_admin_group }}"
|
||||||
|
mode: "0700"
|
||||||
|
loop: "{{ atlas_onlyoffice_bind_paths.results }}"
|
||||||
|
loop_control:
|
||||||
|
label: "{{ item.item }}"
|
||||||
|
when: not item.stat.exists
|
||||||
|
|
||||||
|
- name: Store private mounted password files inside a restricted host directory
|
||||||
|
ansible.builtin.copy:
|
||||||
|
content: "{{ item.value }}\n"
|
||||||
|
dest: "{{ atlas_nextcloud_private_dir }}/{{ item.name }}"
|
||||||
|
owner: "{{ atlas_admin_username }}"
|
||||||
|
group: "{{ atlas_admin_group }}"
|
||||||
|
mode: "0644"
|
||||||
|
loop:
|
||||||
|
- {name: postgres-password, value: "{{ vault_nextcloud_database_password }}"}
|
||||||
|
- {name: redis-password, value: "{{ vault_nextcloud_redis_password }}"}
|
||||||
|
- {name: admin-password, value: "{{ vault_nextcloud_admin_password }}"}
|
||||||
|
- {name: onlyoffice-jwt, value: "{{ vault_nextcloud_onlyoffice_jwt }}"}
|
||||||
|
no_log: true
|
||||||
|
diff: false
|
||||||
|
register: atlas_nextcloud_secret_files
|
||||||
|
|
||||||
|
- name: Render private Redis and ONLYOFFICE configuration
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: "{{ item.src }}"
|
||||||
|
dest: "{{ atlas_nextcloud_private_dir }}/{{ item.dest }}"
|
||||||
|
owner: "{{ atlas_admin_username }}"
|
||||||
|
group: "{{ atlas_admin_group }}"
|
||||||
|
mode: "{{ item.mode }}"
|
||||||
|
loop:
|
||||||
|
- {src: atlas-nextcloud-redis.conf.j2, dest: redis.conf, mode: "0644"}
|
||||||
|
- {src: atlas-onlyoffice.env.j2, dest: onlyoffice.env, mode: "0600"}
|
||||||
|
no_log: true
|
||||||
|
diff: false
|
||||||
|
register: atlas_nextcloud_private_configuration
|
||||||
|
|
||||||
|
- name: Download checksum-pinned compatible application releases
|
||||||
|
ansible.builtin.get_url:
|
||||||
|
url: "{{ item.url }}"
|
||||||
|
dest: "{{ atlas_nextcloud_app_cache }}/{{ item.id }}-{{ item.version }}.tar.gz"
|
||||||
|
checksum: "{{ item.checksum }}"
|
||||||
|
owner: "{{ atlas_admin_username }}"
|
||||||
|
group: "{{ atlas_admin_group }}"
|
||||||
|
mode: "0644"
|
||||||
|
loop: "{{ atlas_nextcloud_apps }}"
|
||||||
|
loop_control:
|
||||||
|
label: "{{ item.id }} {{ item.version }}"
|
||||||
|
when: not ansible_check_mode
|
||||||
|
|
||||||
|
- name: Admit only the Aegis gateway to the Nextcloud and Office HTTP listeners
|
||||||
|
ansible.posix.firewalld:
|
||||||
|
rich_rule: >-
|
||||||
|
rule family="ipv4" source address="{{ atlas_aegis_ip }}"
|
||||||
|
port port="{{ item }}" protocol="tcp" accept
|
||||||
|
zone: "{{ atlas_firewalld_zone }}"
|
||||||
|
state: enabled
|
||||||
|
permanent: true
|
||||||
|
immediate: true
|
||||||
|
loop: ["{{ atlas_nextcloud_http_port }}", "{{ atlas_onlyoffice_http_port }}"]
|
||||||
|
|
||||||
|
- name: Enable lingering for the declared rootless owner
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: [loginctl, enable-linger, "{{ atlas_admin_username }}"]
|
||||||
|
creates: "/var/lib/systemd/linger/{{ atlas_admin_username }}"
|
||||||
|
|
||||||
|
- name: Render Nextcloud component and network Quadlets
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: "{{ item }}.j2"
|
||||||
|
dest: "{{ atlas_nextcloud_quadlet_dir }}/{{ item }}"
|
||||||
|
owner: "{{ atlas_admin_username }}"
|
||||||
|
group: "{{ atlas_admin_group }}"
|
||||||
|
mode: "0644"
|
||||||
|
loop:
|
||||||
|
- atlas-nextcloud.network
|
||||||
|
- atlas-nextcloud-db.container
|
||||||
|
- atlas-nextcloud-redis.container
|
||||||
|
- atlas-nextcloud.container
|
||||||
|
- atlas-onlyoffice.container
|
||||||
|
register: atlas_nextcloud_quadlets
|
||||||
|
|
||||||
|
- name: Render recurring Nextcloud cron user units
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: "{{ item }}.j2"
|
||||||
|
dest: "{{ atlas_admin_home }}/.config/systemd/user/{{ item }}"
|
||||||
|
owner: "{{ atlas_admin_username }}"
|
||||||
|
group: "{{ atlas_admin_group }}"
|
||||||
|
mode: "0644"
|
||||||
|
loop: [atlas-nextcloud-cron.service, atlas-nextcloud-cron.timer,
|
||||||
|
atlas-nextcloud-external-scan.service, atlas-nextcloud-external-scan.timer]
|
||||||
|
register: atlas_nextcloud_cron_units
|
||||||
|
|
||||||
|
- name: Manage and verify rootless Nextcloud services
|
||||||
|
become_user: "{{ atlas_admin_username }}"
|
||||||
|
environment:
|
||||||
|
XDG_RUNTIME_DIR: "/run/user/{{ atlas_admin_uid }}"
|
||||||
|
DBUS_SESSION_BUS_ADDRESS: "unix:path=/run/user/{{ atlas_admin_uid }}/bus"
|
||||||
|
when: not ansible_check_mode
|
||||||
|
block:
|
||||||
|
- name: Pull the pinned images before starting services
|
||||||
|
containers.podman.podman_image:
|
||||||
|
name: "{{ item }}"
|
||||||
|
state: present
|
||||||
|
loop:
|
||||||
|
- "{{ atlas_nextcloud_image }}"
|
||||||
|
- "{{ atlas_nextcloud_postgres_image }}"
|
||||||
|
- "{{ atlas_nextcloud_redis_image }}"
|
||||||
|
- "{{ atlas_onlyoffice_image }}"
|
||||||
|
|
||||||
|
- name: Reload the user manager to generate component units
|
||||||
|
ansible.builtin.systemd:
|
||||||
|
scope: user
|
||||||
|
daemon_reload: true
|
||||||
|
|
||||||
|
- name: Start the declared Nextcloud and ONLYOFFICE services
|
||||||
|
ansible.builtin.systemd:
|
||||||
|
scope: user
|
||||||
|
name: "{{ item }}"
|
||||||
|
state: >-
|
||||||
|
{{ 'restarted' if (
|
||||||
|
atlas_nextcloud_quadlets.results |
|
||||||
|
selectattr('item', 'equalto', item | replace('.service', '.container')) |
|
||||||
|
selectattr('changed') | list | length > 0 or
|
||||||
|
atlas_nextcloud_quadlets.results |
|
||||||
|
selectattr('item', 'equalto', 'atlas-nextcloud.network') |
|
||||||
|
selectattr('changed') | list | length > 0 or
|
||||||
|
atlas_nextcloud_private_configuration is changed or
|
||||||
|
atlas_nextcloud_secret_files is changed) else 'started' }}
|
||||||
|
loop: "{{ atlas_nextcloud_services }}"
|
||||||
|
|
||||||
|
- name: Wait for the application configuration directory to be initialized
|
||||||
|
become: true
|
||||||
|
become_user: root
|
||||||
|
ansible.builtin.wait_for:
|
||||||
|
path: "{{ atlas_nextcloud_root }}/app/config/config.php"
|
||||||
|
timeout: 600
|
||||||
|
|
||||||
|
- name: Derive container web-user host IDs from the actual rootless maps
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv:
|
||||||
|
- podman
|
||||||
|
- unshare
|
||||||
|
- python3
|
||||||
|
- -c
|
||||||
|
- >-
|
||||||
|
import json;
|
||||||
|
print(json.dumps({k: next(int(b)+33-int(a) for a,b,n in
|
||||||
|
(l.split() for l in open('/proc/self/'+k+'_map'))
|
||||||
|
if int(a)<=33<int(a)+int(n)) for k in ['uid','gid']}))
|
||||||
|
register: atlas_nextcloud_web_mapping
|
||||||
|
changed_when: false
|
||||||
|
|
||||||
|
- name: Read the current application SELinux label without changing it
|
||||||
|
become: true
|
||||||
|
become_user: root
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: [stat, -c, '%C', "{{ atlas_nextcloud_root }}/app/config"]
|
||||||
|
register: atlas_nextcloud_config_label
|
||||||
|
changed_when: false
|
||||||
|
|
||||||
|
- name: Maintain the managed Nextcloud configuration include
|
||||||
|
become: true
|
||||||
|
become_user: root
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: atlas-nextcloud.config.php.j2
|
||||||
|
dest: "{{ atlas_nextcloud_root }}/app/config/atlas.config.php"
|
||||||
|
owner: "{{ (atlas_nextcloud_web_mapping.stdout | from_json).uid }}"
|
||||||
|
group: "{{ (atlas_nextcloud_web_mapping.stdout | from_json).gid }}"
|
||||||
|
mode: "0640"
|
||||||
|
seuser: "{{ atlas_nextcloud_config_label.stdout.split(':')[0] }}"
|
||||||
|
serole: "{{ atlas_nextcloud_config_label.stdout.split(':')[1] }}"
|
||||||
|
setype: "{{ atlas_nextcloud_config_label.stdout.split(':')[2] }}"
|
||||||
|
selevel: "{{ atlas_nextcloud_config_label.stdout.split(':')[3:] | join(':') }}"
|
||||||
|
diff: false
|
||||||
|
|
||||||
|
- name: Wait for Nextcloud to complete its initial installation
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, status, --output=json]
|
||||||
|
register: atlas_nextcloud_status
|
||||||
|
changed_when: false
|
||||||
|
retries: 60
|
||||||
|
delay: 10
|
||||||
|
until: >-
|
||||||
|
atlas_nextcloud_status.rc == 0 and
|
||||||
|
atlas_nextcloud_status.stdout.startswith('{') and
|
||||||
|
(atlas_nextcloud_status.stdout | from_json).installed | default(false)
|
||||||
|
|
||||||
|
- name: Import declared ongoing application and account configuration
|
||||||
|
ansible.builtin.include_tasks: nextcloud_application.yml
|
||||||
|
|
||||||
|
- name: Enable and start the recurring Nextcloud cron timer
|
||||||
|
ansible.builtin.systemd:
|
||||||
|
scope: user
|
||||||
|
name: atlas-nextcloud-cron.timer
|
||||||
|
state: "{{ 'restarted' if atlas_nextcloud_cron_units is changed else 'started' }}"
|
||||||
|
enabled: true
|
||||||
|
|
||||||
|
- name: Enable periodic targeted Archive discovery
|
||||||
|
ansible.builtin.systemd:
|
||||||
|
scope: user
|
||||||
|
name: atlas-nextcloud-external-scan.timer
|
||||||
|
state: "{{ 'restarted' if atlas_nextcloud_cron_units is changed else 'started' }}"
|
||||||
|
enabled: true
|
||||||
|
when: atlas_nextcloud_external_mounts | length > 0
|
||||||
|
|
||||||
|
- name: Verify ONLYOFFICE local health without publishing the domain
|
||||||
|
ansible.builtin.uri:
|
||||||
|
url: "http://127.0.0.1:{{ atlas_onlyoffice_http_port }}/healthcheck"
|
||||||
|
return_content: true
|
||||||
|
register: atlas_onlyoffice_health
|
||||||
|
retries: 60
|
||||||
|
delay: 10
|
||||||
|
until: atlas_onlyoffice_health.status | default(0) == 200 and atlas_onlyoffice_health.content | default('') | trim == 'true'
|
||||||
186
ansible/roles/profile_atlas/tasks/nextcloud_application.yml
Normal file
186
ansible/roles/profile_atlas/tasks/nextcloud_application.yml
Normal file
@@ -0,0 +1,186 @@
|
|||||||
|
---
|
||||||
|
- name: Inspect installed application state
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, app:list, --output=json]
|
||||||
|
register: atlas_nextcloud_current_apps
|
||||||
|
changed_when: false
|
||||||
|
|
||||||
|
- name: Record enabled and disabled application versions
|
||||||
|
ansible.builtin.set_fact:
|
||||||
|
atlas_nextcloud_installed_apps: >-
|
||||||
|
{{ (atlas_nextcloud_current_apps.stdout | from_json).enabled |
|
||||||
|
combine((atlas_nextcloud_current_apps.stdout | from_json).disabled) }}
|
||||||
|
|
||||||
|
- name: Refuse implicit application upgrades or downgrades
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- item.id not in atlas_nextcloud_installed_apps or atlas_nextcloud_installed_apps[item.id] == item.version
|
||||||
|
fail_msg: Application versions must be changed in a deliberate upgrade window.
|
||||||
|
loop: "{{ atlas_nextcloud_apps }}"
|
||||||
|
loop_control:
|
||||||
|
label: "{{ item.id }}"
|
||||||
|
|
||||||
|
- name: Install only absent checksum-verified application archives
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv:
|
||||||
|
- podman
|
||||||
|
- exec
|
||||||
|
- --user
|
||||||
|
- '33'
|
||||||
|
- atlas-nextcloud
|
||||||
|
- tar
|
||||||
|
- -xzf
|
||||||
|
- "/mnt/atlas-apps/{{ item.id }}-{{ item.version }}.tar.gz"
|
||||||
|
- -C
|
||||||
|
- /var/www/html/custom_apps
|
||||||
|
loop: "{{ atlas_nextcloud_apps }}"
|
||||||
|
loop_control:
|
||||||
|
label: "{{ item.id }}"
|
||||||
|
when: item.id not in atlas_nextcloud_installed_apps
|
||||||
|
changed_when: true
|
||||||
|
|
||||||
|
- name: Enable the declared applications
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, app:enable, "{{ item.id }}"]
|
||||||
|
loop: "{{ atlas_nextcloud_apps }}"
|
||||||
|
loop_control:
|
||||||
|
label: "{{ item.id }}"
|
||||||
|
when: item.id not in (atlas_nextcloud_current_apps.stdout | from_json).enabled
|
||||||
|
changed_when: true
|
||||||
|
|
||||||
|
- name: Inspect existing application users without exposing passwords
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, user:list, --output=json]
|
||||||
|
register: atlas_nextcloud_current_users
|
||||||
|
changed_when: false
|
||||||
|
|
||||||
|
- name: Ensure the two standard users exist without resetting existing passwords
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv:
|
||||||
|
- podman
|
||||||
|
- exec
|
||||||
|
- --user
|
||||||
|
- '33'
|
||||||
|
- --env
|
||||||
|
- OC_PASS
|
||||||
|
- atlas-nextcloud
|
||||||
|
- php
|
||||||
|
- occ
|
||||||
|
- user:add
|
||||||
|
- --password-from-env
|
||||||
|
- --display-name
|
||||||
|
- "{{ item.display_name }}"
|
||||||
|
- "{{ item.username }}"
|
||||||
|
environment:
|
||||||
|
OC_PASS: "{{ item.password }}"
|
||||||
|
loop: "{{ atlas_nextcloud_users }}"
|
||||||
|
when: item.username not in (atlas_nextcloud_current_users.stdout | from_json)
|
||||||
|
changed_when: true
|
||||||
|
no_log: true
|
||||||
|
diff: false
|
||||||
|
|
||||||
|
- name: Inspect standard-user group membership and quota
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, user:info, --output=json, "{{ item.username }}"]
|
||||||
|
loop: "{{ atlas_nextcloud_users }}"
|
||||||
|
loop_control:
|
||||||
|
label: "{{ item.username }}"
|
||||||
|
register: atlas_nextcloud_user_info
|
||||||
|
changed_when: false
|
||||||
|
no_log: true
|
||||||
|
|
||||||
|
- name: Require that family users are not administrators
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- "'admin' not in (item.stdout | from_json).groups"
|
||||||
|
loop: "{{ atlas_nextcloud_user_info.results }}"
|
||||||
|
no_log: true
|
||||||
|
|
||||||
|
- name: Maintain unlimited initial standard-user quotas
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, user:setting, "{{ item.item.username }}", files, quota, none]
|
||||||
|
loop: "{{ atlas_nextcloud_user_info.results }}"
|
||||||
|
when: (item.stdout | from_json).quota != 'none'
|
||||||
|
changed_when: true
|
||||||
|
no_log: true
|
||||||
|
|
||||||
|
- name: Inspect the family group
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, group:list, --output=json]
|
||||||
|
register: atlas_nextcloud_groups
|
||||||
|
changed_when: false
|
||||||
|
|
||||||
|
- name: Ensure the family group exists
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, group:add, famiglia]
|
||||||
|
when: "'famiglia' not in (atlas_nextcloud_groups.stdout | from_json)"
|
||||||
|
changed_when: true
|
||||||
|
|
||||||
|
- name: Ensure both standard users belong to the family group
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, group:adduser, famiglia, "{{ item.username }}"]
|
||||||
|
loop: "{{ atlas_nextcloud_users }}"
|
||||||
|
when: item.username not in ((atlas_nextcloud_groups.stdout | from_json).get('famiglia', []))
|
||||||
|
changed_when: true
|
||||||
|
no_log: true
|
||||||
|
|
||||||
|
- name: Inspect configured family folders
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, groupfolders:list, --output=json]
|
||||||
|
register: atlas_nextcloud_folders_before
|
||||||
|
changed_when: false
|
||||||
|
|
||||||
|
- name: Ensure a shared Famiglia folder exists
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, groupfolders:create, Famiglia]
|
||||||
|
when: >-
|
||||||
|
(atlas_nextcloud_folders_before.stdout | from_json |
|
||||||
|
selectattr('mountPoint', 'equalto', 'Famiglia') | list | length) == 0
|
||||||
|
changed_when: true
|
||||||
|
|
||||||
|
- name: Inspect the resulting family folder
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, groupfolders:list, --output=json]
|
||||||
|
register: atlas_nextcloud_folders_after
|
||||||
|
changed_when: false
|
||||||
|
|
||||||
|
- name: Select the existing family folder without changing unrelated folders
|
||||||
|
ansible.builtin.set_fact:
|
||||||
|
atlas_nextcloud_family_folder: >-
|
||||||
|
{{ atlas_nextcloud_folders_after.stdout | from_json |
|
||||||
|
selectattr('mountPoint', 'equalto', 'Famiglia') | first }}
|
||||||
|
|
||||||
|
- name: Maintain family read, create, write and delete permissions
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, groupfolders:group,
|
||||||
|
"{{ atlas_nextcloud_family_folder.id }}", famiglia, write, delete]
|
||||||
|
when: (atlas_nextcloud_family_folder.groups_list | default({}, true)).get('famiglia', 0) | int != 15
|
||||||
|
changed_when: true
|
||||||
|
|
||||||
|
- name: Inspect the background job mode
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, config:app:get, core, backgroundjobs_mode]
|
||||||
|
register: atlas_nextcloud_background_mode
|
||||||
|
changed_when: false
|
||||||
|
failed_when: atlas_nextcloud_background_mode.rc not in [0, 1]
|
||||||
|
|
||||||
|
- name: Maintain cron background processing
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, background:cron]
|
||||||
|
when: atlas_nextcloud_background_mode.stdout | trim != 'cron'
|
||||||
|
changed_when: true
|
||||||
|
|
||||||
|
- name: Enable shipped external storage support when required
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, app:enable, files_external]
|
||||||
|
when:
|
||||||
|
- atlas_nextcloud_external_mounts | length > 0
|
||||||
|
- "'files_external' not in (atlas_nextcloud_current_apps.stdout | from_json).enabled"
|
||||||
|
changed_when: true
|
||||||
|
|
||||||
|
- name: Maintain only the declared Archive mounts
|
||||||
|
ansible.builtin.include_tasks: nextcloud_external_mount.yml
|
||||||
|
loop: "{{ atlas_nextcloud_external_mounts }}"
|
||||||
|
loop_control:
|
||||||
|
loop_var: atlas_nextcloud_mount
|
||||||
|
label: "{{ atlas_nextcloud_mount.name }}"
|
||||||
55
ansible/roles/profile_atlas/tasks/nextcloud_backup.yml
Normal file
55
ansible/roles/profile_atlas/tasks/nextcloud_backup.yml
Normal file
@@ -0,0 +1,55 @@
|
|||||||
|
---
|
||||||
|
- name: Manage recurring consistent Nextcloud backup preparation
|
||||||
|
tags: [atlas, nextcloud_backup]
|
||||||
|
when: atlas_manage_nextcloud | bool
|
||||||
|
block:
|
||||||
|
- name: Validate private backup scope and local bundle retention
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- atlas_nextcloud_backup_root == atlas_mount_root ~ '/backup/nextcloud'
|
||||||
|
- atlas_nextcloud_backup_keep | int >= 2
|
||||||
|
- atlas_manage_borg_backup | bool
|
||||||
|
- atlas_manage_usb_backup | bool
|
||||||
|
|
||||||
|
- name: Install recurring backup helper with shell syntax validation
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: atlas-nextcloud-backup.sh.j2
|
||||||
|
dest: /usr/local/sbin/atlas-nextcloud-backup
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0750"
|
||||||
|
validate: /bin/bash -n %s
|
||||||
|
|
||||||
|
- name: Install Nextcloud backup preparation and boot recovery units
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: "{{ item }}.j2"
|
||||||
|
dest: "/etc/systemd/system/{{ item }}"
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0644"
|
||||||
|
loop: [atlas-nextcloud-backup.service, atlas-nextcloud-backup-recovery.service]
|
||||||
|
|
||||||
|
- name: Create backup dependency drop-in directories
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: "/etc/systemd/system/{{ item }}.d"
|
||||||
|
state: directory
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0755"
|
||||||
|
loop: [atlas-borg-backup.service, atlas-usb-backup.service]
|
||||||
|
|
||||||
|
- name: Require a fresh consistent bundle before offsite and manual USB backups
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: atlas-nextcloud-backup-dependency.conf.j2
|
||||||
|
dest: "/etc/systemd/system/{{ item }}.d/nextcloud.conf"
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0644"
|
||||||
|
loop: [atlas-borg-backup.service, atlas-usb-backup.service]
|
||||||
|
|
||||||
|
- name: Reload systemd and enable interruption recovery without running a backup
|
||||||
|
ansible.builtin.systemd:
|
||||||
|
daemon_reload: true
|
||||||
|
name: atlas-nextcloud-backup-recovery.service
|
||||||
|
enabled: true
|
||||||
|
when: not ansible_check_mode
|
||||||
100
ansible/roles/profile_atlas/tasks/nextcloud_external_access.yml
Normal file
100
ansible/roles/profile_atlas/tasks/nextcloud_external_access.yml
Normal file
@@ -0,0 +1,100 @@
|
|||||||
|
---
|
||||||
|
- name: Restrict external storage to explicit Archive directories
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- item.source in [atlas_archive_mountpoint ~ '/Documents', atlas_icloudpd_photos_dir]
|
||||||
|
- item.target is match('^/mnt/archive-[a-z]+$')
|
||||||
|
- item.name is match('^[A-Za-z][A-Za-z ]+$')
|
||||||
|
- item.readonly is boolean
|
||||||
|
- item.user in (atlas_nextcloud_users | map(attribute='username') | list)
|
||||||
|
- item.source != atlas_icloudpd_photos_dir or item.readonly
|
||||||
|
loop: "{{ atlas_nextcloud_external_mounts }}"
|
||||||
|
|
||||||
|
- name: Inspect existing sources without creating or moving data
|
||||||
|
ansible.builtin.stat:
|
||||||
|
path: "{{ item.source }}"
|
||||||
|
follow: false
|
||||||
|
loop: "{{ atlas_nextcloud_external_mounts }}"
|
||||||
|
register: atlas_nextcloud_external_sources
|
||||||
|
|
||||||
|
- name: Refuse missing sources and symlinks
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- item.stat.isdir | default(false)
|
||||||
|
- not (item.stat.islnk | default(false))
|
||||||
|
loop: "{{ atlas_nextcloud_external_sources.results }}"
|
||||||
|
loop_control:
|
||||||
|
label: "{{ item.item.source }}"
|
||||||
|
|
||||||
|
- name: Verify the Archive dataset before modifying its ACL capability
|
||||||
|
community.general.zfs_facts:
|
||||||
|
name: "{{ atlas_zfs_pool }}/{{ atlas_zfs_dataset_archive }}"
|
||||||
|
properties: name,mounted,mountpoint
|
||||||
|
register: atlas_nextcloud_external_dataset
|
||||||
|
|
||||||
|
- name: Refuse an absent or unmounted Archive dataset
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- atlas_nextcloud_external_dataset.ansible_facts.ansible_zfs_datasets | length == 1
|
||||||
|
- atlas_nextcloud_external_dataset.ansible_facts.ansible_zfs_datasets[0].mounted == 'yes'
|
||||||
|
- atlas_nextcloud_external_dataset.ansible_facts.ansible_zfs_datasets[0].mountpoint == atlas_archive_mountpoint
|
||||||
|
|
||||||
|
- name: Enable persistent POSIX ACL support on the verified Archive dataset
|
||||||
|
community.general.zfs:
|
||||||
|
name: "{{ atlas_zfs_pool }}/{{ atlas_zfs_dataset_archive }}"
|
||||||
|
state: present
|
||||||
|
extra_zfs_properties:
|
||||||
|
acltype: posix
|
||||||
|
|
||||||
|
- name: Derive actual rootless web UID for narrowly scoped Archive ACLs
|
||||||
|
become_user: "{{ atlas_admin_username }}"
|
||||||
|
environment:
|
||||||
|
XDG_RUNTIME_DIR: "/run/user/{{ atlas_admin_uid }}"
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv:
|
||||||
|
- podman
|
||||||
|
- unshare
|
||||||
|
- python3
|
||||||
|
- -c
|
||||||
|
- >-
|
||||||
|
print(next(int(b)+33-int(a) for a,b,n in
|
||||||
|
(l.split() for l in open('/proc/self/uid_map')) if int(a)<=33<int(a)+int(n)))
|
||||||
|
register: atlas_nextcloud_external_uid
|
||||||
|
changed_when: false
|
||||||
|
check_mode: false
|
||||||
|
|
||||||
|
- name: Grant web user access only inside the declared sources
|
||||||
|
ansible.posix.acl:
|
||||||
|
path: "{{ item.source }}"
|
||||||
|
entity: "{{ atlas_nextcloud_external_uid.stdout | trim }}"
|
||||||
|
etype: user
|
||||||
|
permissions: "{{ 'rX' if item.readonly else 'rwX' }}"
|
||||||
|
recursive: true
|
||||||
|
follow: false
|
||||||
|
state: present
|
||||||
|
loop: "{{ atlas_nextcloud_external_mounts }}"
|
||||||
|
|
||||||
|
- name: Inherit web access on new files and directories
|
||||||
|
ansible.posix.acl:
|
||||||
|
path: "{{ item.source }}"
|
||||||
|
entity: "{{ atlas_nextcloud_external_uid.stdout | trim }}"
|
||||||
|
etype: user
|
||||||
|
permissions: "{{ 'rX' if item.readonly else 'rwX' }}"
|
||||||
|
default: true
|
||||||
|
recursive: true
|
||||||
|
follow: false
|
||||||
|
state: present
|
||||||
|
loop: "{{ atlas_nextcloud_external_mounts }}"
|
||||||
|
|
||||||
|
- name: Preserve administrator access to documents created through Nextcloud
|
||||||
|
ansible.posix.acl:
|
||||||
|
path: "{{ item.source }}"
|
||||||
|
entity: "{{ atlas_admin_uid }}"
|
||||||
|
etype: user
|
||||||
|
permissions: rwX
|
||||||
|
default: true
|
||||||
|
recursive: true
|
||||||
|
follow: false
|
||||||
|
state: present
|
||||||
|
loop: "{{ atlas_nextcloud_external_mounts }}"
|
||||||
|
when: not item.readonly
|
||||||
@@ -0,0 +1,94 @@
|
|||||||
|
---
|
||||||
|
- name: Inspect current system mounts without exposing credentials
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, files_external:list, --output=json]
|
||||||
|
register: atlas_nextcloud_mount_list
|
||||||
|
changed_when: false
|
||||||
|
no_log: true
|
||||||
|
|
||||||
|
- name: Select only the matching mount name
|
||||||
|
ansible.builtin.set_fact:
|
||||||
|
atlas_nextcloud_matching_mounts: >-
|
||||||
|
{{ atlas_nextcloud_mount_list.stdout | from_json |
|
||||||
|
selectattr('mount_point', 'equalto', '/' ~ atlas_nextcloud_mount.name) | list }}
|
||||||
|
no_log: true
|
||||||
|
|
||||||
|
- name: Refuse duplicates or repurposing of existing unrelated storage
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- atlas_nextcloud_matching_mounts | length <= 1
|
||||||
|
- >-
|
||||||
|
atlas_nextcloud_matching_mounts | length == 0 or
|
||||||
|
(atlas_nextcloud_matching_mounts[0].configuration.datadir | default('') == atlas_nextcloud_mount.target
|
||||||
|
and atlas_nextcloud_matching_mounts[0].storage == '\\OC\\Files\\Storage\\Local')
|
||||||
|
fail_msg: Existing storage conflicts with the declared Archive mount; refusing an implicit replacement.
|
||||||
|
|
||||||
|
- name: Create an absent local mount restricted to its declared user
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv:
|
||||||
|
- podman
|
||||||
|
- exec
|
||||||
|
- --user
|
||||||
|
- '33'
|
||||||
|
- atlas-nextcloud
|
||||||
|
- php
|
||||||
|
- occ
|
||||||
|
- files_external:create
|
||||||
|
- "{{ atlas_nextcloud_mount.name }}"
|
||||||
|
- local
|
||||||
|
- null::null
|
||||||
|
- --config
|
||||||
|
- "datadir={{ atlas_nextcloud_mount.target }}"
|
||||||
|
- --applicable-user
|
||||||
|
- "{{ atlas_nextcloud_mount.user }}"
|
||||||
|
- --output=json
|
||||||
|
when: atlas_nextcloud_matching_mounts | length == 0
|
||||||
|
register: atlas_nextcloud_mount_created
|
||||||
|
changed_when: true
|
||||||
|
|
||||||
|
- name: Record the managed mount ID and options
|
||||||
|
ansible.builtin.set_fact:
|
||||||
|
atlas_nextcloud_mount_id: >-
|
||||||
|
{{ atlas_nextcloud_mount_created.stdout | trim if atlas_nextcloud_matching_mounts | length == 0
|
||||||
|
else atlas_nextcloud_matching_mounts[0].mount_id }}
|
||||||
|
atlas_nextcloud_mount_options: >-
|
||||||
|
{{ {} if atlas_nextcloud_matching_mounts | length == 0 else atlas_nextcloud_matching_mounts[0].options }}
|
||||||
|
|
||||||
|
- name: Restrict the managed mount to exactly its declared user
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: >-
|
||||||
|
{{ ['podman', 'exec', '--user', '33', 'atlas-nextcloud', 'php', 'occ',
|
||||||
|
'files_external:applicable', atlas_nextcloud_mount_id | string,
|
||||||
|
'--add-user=' ~ atlas_nextcloud_mount.user] +
|
||||||
|
(atlas_nextcloud_matching_mounts[0].applicable_groups |
|
||||||
|
map('regex_replace', '^', '--remove-group=') | list) +
|
||||||
|
(atlas_nextcloud_matching_mounts[0].applicable_users |
|
||||||
|
reject('equalto', atlas_nextcloud_mount.user) |
|
||||||
|
map('regex_replace', '^', '--remove-user=') | list) }}
|
||||||
|
when:
|
||||||
|
- atlas_nextcloud_matching_mounts | length > 0
|
||||||
|
- >-
|
||||||
|
atlas_nextcloud_matching_mounts[0].applicable_groups | length > 0 or
|
||||||
|
atlas_nextcloud_matching_mounts[0].applicable_users != [atlas_nextcloud_mount.user]
|
||||||
|
changed_when: true
|
||||||
|
|
||||||
|
- name: Maintain read-only photos and external change detection
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, files_external:option,
|
||||||
|
"{{ atlas_nextcloud_mount_id }}", "{{ item.key }}", "{{ item.value | to_json }}"]
|
||||||
|
loop:
|
||||||
|
- {key: readonly, value: "{{ atlas_nextcloud_mount.readonly }}"}
|
||||||
|
- {key: filesystem_check_changes, value: 1}
|
||||||
|
- {key: enable_sharing, value: false}
|
||||||
|
# Nextcloud persists option values as strings ("1" / "" for booleans).
|
||||||
|
when: >-
|
||||||
|
item.key not in atlas_nextcloud_mount_options or
|
||||||
|
atlas_nextcloud_mount_options[item.key] | string !=
|
||||||
|
(('1' if item.value else '') if item.value is boolean else item.value | string)
|
||||||
|
changed_when: true
|
||||||
|
|
||||||
|
- name: Verify the managed local storage is accessible
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, files_external:verify,
|
||||||
|
"{{ atlas_nextcloud_mount_id }}"]
|
||||||
|
changed_when: false
|
||||||
@@ -0,0 +1,66 @@
|
|||||||
|
---
|
||||||
|
- name: Validate Atlas Prometheus pull identity inputs
|
||||||
|
tags: [atlas, backup, prometheus_backup, prometheus_backup_key]
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- atlas_prometheus_pull_ssh_dir.startswith('/etc/')
|
||||||
|
- atlas_prometheus_pull_private_key_path.startswith(atlas_prometheus_pull_ssh_dir ~ '/')
|
||||||
|
- atlas_prometheus_pull_known_hosts_path.startswith(atlas_prometheus_pull_ssh_dir ~ '/')
|
||||||
|
- atlas_prometheus_ssh_host_key.startswith(
|
||||||
|
(hostvars['prometheus'].ansible_host | string) ~ ' ssh-ed25519 '
|
||||||
|
)
|
||||||
|
fail_msg: Pin the verified Prometheus ED25519 SSH host key before enabling the pull.
|
||||||
|
when: atlas_manage_prometheus_backup_pull | bool
|
||||||
|
|
||||||
|
- name: Create private Atlas Prometheus pull SSH directory
|
||||||
|
tags: [atlas, backup, prometheus_backup, prometheus_backup_key]
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: "{{ atlas_prometheus_pull_ssh_dir }}"
|
||||||
|
state: directory
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0700"
|
||||||
|
when: atlas_manage_prometheus_backup_pull | bool
|
||||||
|
|
||||||
|
- name: Generate Atlas-only Prometheus pull SSH identity
|
||||||
|
tags: [atlas, backup, prometheus_backup, prometheus_backup_key]
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv:
|
||||||
|
- ssh-keygen
|
||||||
|
- -q
|
||||||
|
- -t
|
||||||
|
- ed25519
|
||||||
|
- -N
|
||||||
|
- ""
|
||||||
|
- -C
|
||||||
|
- atlas-prometheus-pull@atlas
|
||||||
|
- -f
|
||||||
|
- "{{ atlas_prometheus_pull_private_key_path }}"
|
||||||
|
creates: "{{ atlas_prometheus_pull_private_key_path }}"
|
||||||
|
when: atlas_manage_prometheus_backup_pull | bool
|
||||||
|
|
||||||
|
- name: Protect Atlas-only Prometheus pull SSH identity
|
||||||
|
tags: [atlas, backup, prometheus_backup, prometheus_backup_key]
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: "{{ item.path }}"
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "{{ item.mode }}"
|
||||||
|
loop:
|
||||||
|
- { path: "{{ atlas_prometheus_pull_private_key_path }}", mode: "0600" }
|
||||||
|
- { path: "{{ atlas_prometheus_pull_private_key_path }}.pub", mode: "0644" }
|
||||||
|
loop_control:
|
||||||
|
label: "{{ item.path }}"
|
||||||
|
when:
|
||||||
|
- atlas_manage_prometheus_backup_pull | bool
|
||||||
|
- not ansible_check_mode
|
||||||
|
|
||||||
|
- name: Pin Prometheus SSH host key on Atlas
|
||||||
|
tags: [atlas, backup, prometheus_backup, prometheus_backup_key]
|
||||||
|
ansible.builtin.copy:
|
||||||
|
content: "{{ atlas_prometheus_ssh_host_key }}\n"
|
||||||
|
dest: "{{ atlas_prometheus_pull_known_hosts_path }}"
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0600"
|
||||||
|
when: atlas_manage_prometheus_backup_pull | bool
|
||||||
90
ansible/roles/profile_atlas/tasks/prometheus_pull_job.yml
Normal file
90
ansible/roles/profile_atlas/tasks/prometheus_pull_job.yml
Normal file
@@ -0,0 +1,90 @@
|
|||||||
|
---
|
||||||
|
- name: Validate Atlas Prometheus backup pull inputs
|
||||||
|
tags: [atlas, backup, prometheus_backup]
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- atlas_manage_storage | bool
|
||||||
|
- atlas_prometheus_pull_source_user is match('^[a-z_][a-z0-9_-]*$')
|
||||||
|
- atlas_prometheus_pull_source_port | int > 0
|
||||||
|
- atlas_prometheus_pull_source_port | int < 65536
|
||||||
|
- atlas_prometheus_pull_keep_daily | int > 0
|
||||||
|
- atlas_prometheus_pull_keep_weekly | int > 0
|
||||||
|
- atlas_prometheus_pull_keep_monthly | int > 0
|
||||||
|
- atlas_prometheus_pull_max_age_hours | int > 0
|
||||||
|
- atlas_backup_prometheus_mountpoint.startswith(atlas_mount_root ~ '/')
|
||||||
|
fail_msg: Define the Atlas backup destination, source account, and retention before enabling the pull.
|
||||||
|
when: atlas_manage_prometheus_backup_pull | bool
|
||||||
|
|
||||||
|
- name: Validate Atlas Prometheus backup pull calendar
|
||||||
|
tags: [atlas, backup, prometheus_backup]
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: [systemd-analyze, calendar, "{{ atlas_prometheus_pull_calendar }}"]
|
||||||
|
changed_when: false
|
||||||
|
check_mode: false
|
||||||
|
when: atlas_manage_prometheus_backup_pull | bool
|
||||||
|
|
||||||
|
- name: Create private Atlas Prometheus backup version directory
|
||||||
|
tags: [atlas, backup, prometheus_backup]
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: "{{ atlas_backup_prometheus_mountpoint }}/snapshots"
|
||||||
|
state: directory
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0700"
|
||||||
|
when: atlas_manage_prometheus_backup_pull | bool
|
||||||
|
|
||||||
|
- name: Install Atlas Prometheus backup pull helper
|
||||||
|
tags: [atlas, backup, prometheus_backup]
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: atlas-prometheus-pull.sh.j2
|
||||||
|
dest: /usr/local/sbin/atlas-prometheus-pull
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0750"
|
||||||
|
when: atlas_manage_prometheus_backup_pull | bool
|
||||||
|
|
||||||
|
- name: Install Atlas Prometheus backup retention helper
|
||||||
|
tags: [atlas, backup, prometheus_backup]
|
||||||
|
ansible.builtin.copy:
|
||||||
|
src: atlas-prometheus-prune.py
|
||||||
|
dest: /usr/local/libexec/atlas-prometheus-prune
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0750"
|
||||||
|
when: atlas_manage_prometheus_backup_pull | bool
|
||||||
|
|
||||||
|
- name: Install Atlas Prometheus backup pull systemd units
|
||||||
|
tags: [atlas, backup, prometheus_backup]
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: "{{ item }}.j2"
|
||||||
|
dest: "/etc/systemd/system/{{ item }}"
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0644"
|
||||||
|
loop:
|
||||||
|
- atlas-prometheus-pull.service
|
||||||
|
- atlas-prometheus-pull.timer
|
||||||
|
loop_control:
|
||||||
|
label: "{{ item }}"
|
||||||
|
register: atlas_prometheus_pull_units
|
||||||
|
when: atlas_manage_prometheus_backup_pull | bool
|
||||||
|
|
||||||
|
- name: Reload systemd after Atlas Prometheus pull unit changes
|
||||||
|
tags: [atlas, backup, prometheus_backup]
|
||||||
|
ansible.builtin.systemd:
|
||||||
|
daemon_reload: true
|
||||||
|
when:
|
||||||
|
- atlas_manage_prometheus_backup_pull | bool
|
||||||
|
- atlas_prometheus_pull_units is changed
|
||||||
|
- not ansible_check_mode
|
||||||
|
|
||||||
|
- name: Enable Atlas Prometheus pull timer only after explicit activation
|
||||||
|
tags: [atlas, backup, prometheus_backup]
|
||||||
|
ansible.builtin.systemd:
|
||||||
|
name: atlas-prometheus-pull.timer
|
||||||
|
enabled: true
|
||||||
|
state: started
|
||||||
|
when:
|
||||||
|
- atlas_manage_prometheus_backup_pull | bool
|
||||||
|
- atlas_prometheus_pull_start_timer | bool
|
||||||
|
- not ansible_check_mode
|
||||||
27
ansible/roles/profile_atlas/tasks/restorecon.yml
Normal file
27
ansible/roles/profile_atlas/tasks/restorecon.yml
Normal file
@@ -0,0 +1,27 @@
|
|||||||
|
---
|
||||||
|
- name: Validate requested Atlas post-restore relabel paths
|
||||||
|
tags: [atlas, restorecon, recovery]
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- item is string
|
||||||
|
- item.startswith(atlas_mount_root ~ '/')
|
||||||
|
- item != atlas_mount_root
|
||||||
|
fail_msg: >-
|
||||||
|
Post-restore relabeling accepts only explicit paths below the Atlas pool
|
||||||
|
mount root. Do not relabel the whole pool during routine provisioning.
|
||||||
|
loop: "{{ atlas_restorecon_paths }}"
|
||||||
|
when: atlas_restorecon_paths | length > 0
|
||||||
|
|
||||||
|
- name: Restore SELinux labels on explicitly restored Atlas paths
|
||||||
|
tags: [atlas, restorecon, recovery]
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv:
|
||||||
|
- restorecon
|
||||||
|
- -RFv
|
||||||
|
- "{{ item }}"
|
||||||
|
register: atlas_restorecon_result
|
||||||
|
changed_when: atlas_restorecon_result.stdout | length > 0
|
||||||
|
loop: "{{ atlas_restorecon_paths }}"
|
||||||
|
when:
|
||||||
|
- atlas_restorecon_paths | length > 0
|
||||||
|
- not ansible_check_mode
|
||||||
@@ -10,6 +10,7 @@
|
|||||||
properties:
|
properties:
|
||||||
compression: zstd
|
compression: zstd
|
||||||
mountpoint: "{{ atlas_archive_mountpoint }}"
|
mountpoint: "{{ atlas_archive_mountpoint }}"
|
||||||
|
acltype: posix
|
||||||
- name: "{{ atlas_zfs_pool }}/{{ atlas_zfs_dataset_services }}"
|
- name: "{{ atlas_zfs_pool }}/{{ atlas_zfs_dataset_services }}"
|
||||||
mountpoint: "{{ atlas_services_mountpoint }}"
|
mountpoint: "{{ atlas_services_mountpoint }}"
|
||||||
owner: "{{ atlas_admin_username }}"
|
owner: "{{ atlas_admin_username }}"
|
||||||
|
|||||||
145
ansible/roles/profile_atlas/tasks/usb_backup.yml
Normal file
145
ansible/roles/profile_atlas/tasks/usb_backup.yml
Normal file
@@ -0,0 +1,145 @@
|
|||||||
|
---
|
||||||
|
- name: Validate Atlas offline USB backup configuration
|
||||||
|
tags: [atlas, storage, backup, usb_backup]
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- atlas_manage_storage | bool
|
||||||
|
- atlas_zfs_pool != 'CHANGEME_ZFS_POOL'
|
||||||
|
- atlas_mount_root.startswith('/')
|
||||||
|
- atlas_usb_backup_luks_uuid is match('^[0-9a-fA-F]{8}(-[0-9a-fA-F]{4}){3}-[0-9a-fA-F]{12}$')
|
||||||
|
- atlas_usb_backup_fs_uuid is match('^[0-9a-fA-F]{8}(-[0-9a-fA-F]{4}){3}-[0-9a-fA-F]{12}$')
|
||||||
|
- atlas_usb_backup_luks_uuid != atlas_usb_backup_fs_uuid
|
||||||
|
- atlas_usb_backup_mapper_name is match('^[a-z][a-z0-9_-]*$')
|
||||||
|
- atlas_usb_backup_min_free_bytes | int > 0
|
||||||
|
- atlas_usb_backup_snapshot_prefix is match('^[a-z0-9][a-z0-9_-]*$')
|
||||||
|
- atlas_usb_backup_snapshot_prefix != atlas_borg_snapshot_prefix
|
||||||
|
- atlas_usb_backup_snapshot_prefix != atlas_zfs_snapshot_prefix
|
||||||
|
fail_msg: >-
|
||||||
|
The manual Atlas USB backup needs verified LUKS and ext4 UUIDs, a safe
|
||||||
|
mapper name, positive free-space reserve, and a unique snapshot prefix.
|
||||||
|
when: atlas_manage_usb_backup | bool
|
||||||
|
|
||||||
|
- name: Install rsync for the Atlas offline USB backup
|
||||||
|
tags: [atlas, storage, backup, usb_backup]
|
||||||
|
ansible.builtin.dnf:
|
||||||
|
name: rsync
|
||||||
|
state: present
|
||||||
|
when: atlas_manage_usb_backup | bool
|
||||||
|
|
||||||
|
- name: Install the manual Atlas offline USB backup helper
|
||||||
|
tags: [atlas, storage, backup, usb_backup]
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: atlas-usb-backup.sh.j2
|
||||||
|
dest: /usr/local/sbin/atlas-usb-backup
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0750"
|
||||||
|
when: atlas_manage_usb_backup | bool
|
||||||
|
|
||||||
|
- name: Install the Atlas USB snapshot cleanup helper
|
||||||
|
tags: [atlas, storage, backup, usb_backup]
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: atlas-usb-snapshot-cleanup.sh.j2
|
||||||
|
dest: /usr/local/sbin/atlas-usb-snapshot-cleanup
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0750"
|
||||||
|
when: atlas_manage_usb_backup | bool
|
||||||
|
|
||||||
|
- name: Install the manual Atlas offline USB backup service
|
||||||
|
tags: [atlas, storage, backup, usb_backup]
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: atlas-usb-backup.service.j2
|
||||||
|
dest: /etc/systemd/system/atlas-usb-backup.service
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0644"
|
||||||
|
when: atlas_manage_usb_backup | bool
|
||||||
|
|
||||||
|
- name: Reload systemd for the Atlas offline USB backup service
|
||||||
|
tags: [atlas, storage, backup, usb_backup]
|
||||||
|
ansible.builtin.systemd:
|
||||||
|
daemon_reload: true
|
||||||
|
when:
|
||||||
|
- atlas_manage_usb_backup | bool
|
||||||
|
- not ansible_check_mode
|
||||||
|
|
||||||
|
- name: Validate the 45Drives Atlas USB reminder configuration
|
||||||
|
tags: [atlas, backup, usb_reminder]
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- atlas_manage_usb_backup | bool
|
||||||
|
- atlas_usb_reminder_calendar | length > 0
|
||||||
|
- atlas_usb_reminder_notifier.startswith('/opt/45drives/houston/')
|
||||||
|
fail_msg: >-
|
||||||
|
Enable the manual USB backup and declare a systemd calendar before
|
||||||
|
enabling its 45Drives Alerts reminder.
|
||||||
|
when: atlas_manage_usb_reminder | bool
|
||||||
|
|
||||||
|
- name: Validate the Atlas USB reminder calendar
|
||||||
|
tags: [atlas, backup, usb_reminder]
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv:
|
||||||
|
- systemd-analyze
|
||||||
|
- calendar
|
||||||
|
- "{{ atlas_usb_reminder_calendar }}"
|
||||||
|
changed_when: false
|
||||||
|
check_mode: false
|
||||||
|
when: atlas_manage_usb_reminder | bool
|
||||||
|
|
||||||
|
- name: Inspect the existing 45Drives notifier
|
||||||
|
tags: [atlas, backup, usb_reminder]
|
||||||
|
ansible.builtin.stat:
|
||||||
|
path: "{{ atlas_usb_reminder_notifier }}"
|
||||||
|
register: atlas_usb_reminder_notifier_file
|
||||||
|
when: atlas_manage_usb_reminder | bool
|
||||||
|
|
||||||
|
- name: Require the configured 45Drives notifier for USB reminders
|
||||||
|
tags: [atlas, backup, usb_reminder]
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- atlas_usb_reminder_notifier_file.stat.executable | default(false)
|
||||||
|
fail_msg: >-
|
||||||
|
The existing 45Drives Houston notifier must be executable.
|
||||||
|
when: atlas_manage_usb_reminder | bool
|
||||||
|
|
||||||
|
- name: Install the 45Drives Atlas USB reminder helper
|
||||||
|
tags: [atlas, backup, usb_reminder]
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: atlas-usb-reminder.py.j2
|
||||||
|
dest: /usr/local/libexec/atlas-usb-reminder
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0750"
|
||||||
|
when: atlas_manage_usb_reminder | bool
|
||||||
|
|
||||||
|
- name: Install the 45Drives Atlas USB reminder service
|
||||||
|
tags: [atlas, backup, usb_reminder]
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: atlas-usb-reminder.service.j2
|
||||||
|
dest: /etc/systemd/system/atlas-usb-reminder.service
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0644"
|
||||||
|
when: atlas_manage_usb_reminder | bool
|
||||||
|
|
||||||
|
- name: Install the 45Drives Atlas USB reminder timer
|
||||||
|
tags: [atlas, backup, usb_reminder]
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: atlas-usb-reminder.timer.j2
|
||||||
|
dest: /etc/systemd/system/atlas-usb-reminder.timer
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0644"
|
||||||
|
when: atlas_manage_usb_reminder | bool
|
||||||
|
|
||||||
|
- name: Enable only the Atlas USB notification reminder timer
|
||||||
|
tags: [atlas, backup, usb_reminder]
|
||||||
|
ansible.builtin.systemd:
|
||||||
|
name: atlas-usb-reminder.timer
|
||||||
|
enabled: true
|
||||||
|
state: started
|
||||||
|
daemon_reload: true
|
||||||
|
when:
|
||||||
|
- atlas_manage_usb_reminder | bool
|
||||||
|
- not ansible_check_mode
|
||||||
175
ansible/roles/profile_atlas/tasks/zfs_maintenance.yml
Normal file
175
ansible/roles/profile_atlas/tasks/zfs_maintenance.yml
Normal file
@@ -0,0 +1,175 @@
|
|||||||
|
---
|
||||||
|
- name: Validate Atlas ZFS snapshot policy
|
||||||
|
tags: [atlas, storage, snapshots]
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- atlas_manage_storage | bool
|
||||||
|
- atlas_zfs_pool != 'CHANGEME_ZFS_POOL'
|
||||||
|
- atlas_zfs_snapshot_prefix is match('^[a-z0-9][a-z0-9_-]*$')
|
||||||
|
- atlas_zfs_snapshot_policies | length > 0
|
||||||
|
- >-
|
||||||
|
(atlas_zfs_snapshot_policies | map(attribute='name') | unique | list | length)
|
||||||
|
== (atlas_zfs_snapshot_policies | length)
|
||||||
|
fail_msg: >-
|
||||||
|
Enable Atlas storage and declare a non-empty snapshot policy with a safe
|
||||||
|
prefix and unique policy names before managing automatic snapshots.
|
||||||
|
when: atlas_manage_zfs_snapshots | bool
|
||||||
|
|
||||||
|
- name: Validate Atlas ZFS snapshot policy entries
|
||||||
|
tags: [atlas, storage, snapshots]
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- item.name is match('^[a-z][a-z0-9_-]*$')
|
||||||
|
- item.keep | int > 0
|
||||||
|
- item.calendar | length > 0
|
||||||
|
fail_msg: >-
|
||||||
|
Every Atlas snapshot policy needs a safe name, a positive retention
|
||||||
|
count, and a systemd calendar expression.
|
||||||
|
loop: "{{ atlas_zfs_snapshot_policies }}"
|
||||||
|
loop_control:
|
||||||
|
label: "{{ item.name | default('unnamed') }}"
|
||||||
|
when: atlas_manage_zfs_snapshots | bool
|
||||||
|
|
||||||
|
- name: Validate Atlas ZFS snapshot calendars
|
||||||
|
tags: [atlas, storage, snapshots]
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv:
|
||||||
|
- systemd-analyze
|
||||||
|
- calendar
|
||||||
|
- "{{ item.calendar }}"
|
||||||
|
loop: "{{ atlas_zfs_snapshot_policies }}"
|
||||||
|
loop_control:
|
||||||
|
label: "{{ item.name }}: {{ item.calendar }}"
|
||||||
|
changed_when: false
|
||||||
|
check_mode: false
|
||||||
|
when: atlas_manage_zfs_snapshots | bool
|
||||||
|
|
||||||
|
- name: Install Atlas ZFS snapshot and retention helper
|
||||||
|
tags: [atlas, storage, snapshots]
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: atlas-zfs-snapshot.sh.j2
|
||||||
|
dest: /usr/local/sbin/atlas-zfs-snapshot
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0750"
|
||||||
|
when: atlas_manage_zfs_snapshots | bool
|
||||||
|
|
||||||
|
- name: Install Atlas ZFS snapshot systemd service
|
||||||
|
tags: [atlas, storage, snapshots]
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: atlas-zfs-snapshot@.service.j2
|
||||||
|
dest: /etc/systemd/system/atlas-zfs-snapshot@.service
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0644"
|
||||||
|
when: atlas_manage_zfs_snapshots | bool
|
||||||
|
|
||||||
|
- name: Install Atlas ZFS snapshot systemd timers
|
||||||
|
tags: [atlas, storage, snapshots]
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: atlas-zfs-snapshot.timer.j2
|
||||||
|
dest: "/etc/systemd/system/atlas-zfs-snapshot-{{ item.name }}.timer"
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0644"
|
||||||
|
loop: "{{ atlas_zfs_snapshot_policies }}"
|
||||||
|
loop_control:
|
||||||
|
label: "{{ item.name }}"
|
||||||
|
when: atlas_manage_zfs_snapshots | bool
|
||||||
|
|
||||||
|
- name: Enable Atlas ZFS snapshot systemd timers
|
||||||
|
tags: [atlas, storage, snapshots]
|
||||||
|
ansible.builtin.systemd:
|
||||||
|
name: "atlas-zfs-snapshot-{{ item.name }}.timer"
|
||||||
|
enabled: true
|
||||||
|
state: started
|
||||||
|
daemon_reload: true
|
||||||
|
loop: "{{ atlas_zfs_snapshot_policies }}"
|
||||||
|
loop_control:
|
||||||
|
label: "{{ item.name }}"
|
||||||
|
when:
|
||||||
|
- atlas_manage_zfs_snapshots | bool
|
||||||
|
- not ansible_check_mode
|
||||||
|
|
||||||
|
- name: Validate Atlas ZFS scrub policy
|
||||||
|
tags: [atlas, storage, scrub]
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- atlas_manage_storage | bool
|
||||||
|
- atlas_zfs_pool != 'CHANGEME_ZFS_POOL'
|
||||||
|
- atlas_zfs_scrub_calendar | length > 0
|
||||||
|
fail_msg: >-
|
||||||
|
Enable Atlas storage and declare a systemd calendar expression before
|
||||||
|
managing periodic ZFS scrubs.
|
||||||
|
when: atlas_manage_zfs_scrub | bool
|
||||||
|
|
||||||
|
- name: Validate Atlas ZFS scrub calendar
|
||||||
|
tags: [atlas, storage, scrub]
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv:
|
||||||
|
- systemd-analyze
|
||||||
|
- calendar
|
||||||
|
- "{{ atlas_zfs_scrub_calendar }}"
|
||||||
|
changed_when: false
|
||||||
|
check_mode: false
|
||||||
|
when: atlas_manage_zfs_scrub | bool
|
||||||
|
|
||||||
|
- name: Require OpenZFS scrub systemd units
|
||||||
|
tags: [atlas, storage, scrub]
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv:
|
||||||
|
- systemctl
|
||||||
|
- cat
|
||||||
|
- "{{ item }}"
|
||||||
|
loop:
|
||||||
|
- "zfs-scrub@{{ atlas_zfs_pool }}.service"
|
||||||
|
- "zfs-scrub-monthly@{{ atlas_zfs_pool }}.timer"
|
||||||
|
- "zfs-scrub-weekly@{{ atlas_zfs_pool }}.timer"
|
||||||
|
loop_control:
|
||||||
|
label: "{{ item }}"
|
||||||
|
changed_when: false
|
||||||
|
check_mode: false
|
||||||
|
when: atlas_manage_zfs_scrub | bool
|
||||||
|
|
||||||
|
- name: Create Atlas ZFS scrub timer override directory
|
||||||
|
tags: [atlas, storage, scrub]
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: "/etc/systemd/system/zfs-scrub-monthly@{{ atlas_zfs_pool }}.timer.d"
|
||||||
|
state: directory
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0755"
|
||||||
|
when: atlas_manage_zfs_scrub | bool
|
||||||
|
|
||||||
|
- name: Configure Atlas ZFS monthly scrub schedule
|
||||||
|
tags: [atlas, storage, scrub]
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: atlas-zfs-scrub-timer.conf.j2
|
||||||
|
dest: >-
|
||||||
|
/etc/systemd/system/zfs-scrub-monthly@{{ atlas_zfs_pool }}.timer.d/override.conf
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0644"
|
||||||
|
when: atlas_manage_zfs_scrub | bool
|
||||||
|
|
||||||
|
- name: Disable the conflicting weekly OpenZFS scrub timer
|
||||||
|
tags: [atlas, storage, scrub]
|
||||||
|
ansible.builtin.systemd:
|
||||||
|
name: "zfs-scrub-weekly@{{ atlas_zfs_pool }}.timer"
|
||||||
|
enabled: false
|
||||||
|
state: stopped
|
||||||
|
daemon_reload: true
|
||||||
|
when:
|
||||||
|
- atlas_manage_zfs_scrub | bool
|
||||||
|
- not ansible_check_mode
|
||||||
|
|
||||||
|
- name: Enable the Atlas monthly OpenZFS scrub timer
|
||||||
|
tags: [atlas, storage, scrub]
|
||||||
|
ansible.builtin.systemd:
|
||||||
|
name: "zfs-scrub-monthly@{{ atlas_zfs_pool }}.timer"
|
||||||
|
enabled: true
|
||||||
|
state: started
|
||||||
|
daemon_reload: true
|
||||||
|
when:
|
||||||
|
- atlas_manage_zfs_scrub | bool
|
||||||
|
- not ansible_check_mode
|
||||||
@@ -0,0 +1,41 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Back up Atlas ZFS datasets to the encrypted Borg repository
|
||||||
|
Documentation=man:borg-create(1) man:borg-prune(1) man:borg-compact(1)
|
||||||
|
Requires=zfs.target
|
||||||
|
Wants=network-online.target
|
||||||
|
After=zfs.target network-online.target
|
||||||
|
StartLimitIntervalSec=6h
|
||||||
|
StartLimitBurst=3
|
||||||
|
ConditionFileIsExecutable=/usr/local/sbin/atlas-borg-backup
|
||||||
|
ConditionPathExists={{ atlas_borg_passphrase_path }}
|
||||||
|
ConditionPathExists={{ atlas_borg_ssh_private_key_path }}
|
||||||
|
ConditionPathExists={{ atlas_borg_known_hosts_path }}
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
Type=oneshot
|
||||||
|
ExecStart=/usr/local/sbin/atlas-borg-backup
|
||||||
|
ExecStopPost=+/usr/local/sbin/atlas-borg-snapshot-cleanup
|
||||||
|
User=root
|
||||||
|
Group=root
|
||||||
|
UMask=0077
|
||||||
|
SuccessExitStatus=1
|
||||||
|
Restart=on-failure
|
||||||
|
RestartSec=30m
|
||||||
|
TimeoutStartSec=infinity
|
||||||
|
RuntimeDirectory=atlas-borg
|
||||||
|
RuntimeDirectoryMode=0750
|
||||||
|
Nice=15
|
||||||
|
IOSchedulingClass=best-effort
|
||||||
|
IOSchedulingPriority=7
|
||||||
|
NoNewPrivileges=true
|
||||||
|
PrivateMounts=true
|
||||||
|
PrivateTmp=true
|
||||||
|
ProtectHome=true
|
||||||
|
ProtectSystem=strict
|
||||||
|
ReadWritePaths={{ atlas_borg_cache_dir }} {{ atlas_borg_config_dir }} /run/atlas-borg /run/lock
|
||||||
|
ProtectKernelTunables=true
|
||||||
|
ProtectKernelModules=true
|
||||||
|
ProtectControlGroups=true
|
||||||
|
RestrictAddressFamilies=AF_UNIX AF_INET AF_INET6
|
||||||
|
RestrictRealtime=true
|
||||||
|
LockPersonality=true
|
||||||
215
ansible/roles/profile_atlas/templates/atlas-borg-backup.sh.j2
Normal file
215
ansible/roles/profile_atlas/templates/atlas-borg-backup.sh.j2
Normal file
@@ -0,0 +1,215 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -Eeuo pipefail
|
||||||
|
|
||||||
|
export LC_ALL=C.utf8
|
||||||
|
export PATH=/usr/sbin:/usr/bin:/sbin:/bin
|
||||||
|
export BORG_CACHE_DIR={{ atlas_borg_cache_dir | quote }}
|
||||||
|
export BORG_CONFIG_DIR={{ atlas_borg_config_dir | quote }}
|
||||||
|
export BORG_PASSCOMMAND={{ ('cat ' ~ atlas_borg_passphrase_path) | quote }}
|
||||||
|
export BORG_RSH={{ atlas_borg_ssh_wrapper_path | quote }}
|
||||||
|
|
||||||
|
readonly pool={{ atlas_zfs_pool | quote }}
|
||||||
|
readonly mount_root={{ atlas_mount_root | quote }}
|
||||||
|
readonly repository={{ ('ssh://' ~ atlas_borg_repository_user ~ '@' ~ atlas_borg_repository_host
|
||||||
|
~ ':' ~ (atlas_borg_repository_port | string) ~ '/' ~ atlas_borg_repository_path) | quote }}
|
||||||
|
readonly remote_path={{ atlas_borg_remote_path | quote }}
|
||||||
|
readonly archive_prefix={{ atlas_borg_archive_prefix | quote }}
|
||||||
|
readonly snapshot_prefix={{ atlas_borg_snapshot_prefix | quote }}
|
||||||
|
readonly compression={{ atlas_borg_compression | quote }}
|
||||||
|
readonly stage=/run/atlas-borg/source
|
||||||
|
readonly snapshot_marker=/run/atlas-borg/snapshot-name
|
||||||
|
readonly borg_user={{ atlas_borg_username | quote }}
|
||||||
|
readonly borg_group={{ atlas_borg_group | quote }}
|
||||||
|
readonly borg_home={{ atlas_borg_home | quote }}
|
||||||
|
readonly borg_lock={{ atlas_borg_lock_path | quote }}
|
||||||
|
readonly progress_filter=/usr/local/libexec/atlas-borg-progress
|
||||||
|
|
||||||
|
snapshot_name=""
|
||||||
|
mounted_targets=()
|
||||||
|
|
||||||
|
# Invoked through the EXIT trap below.
|
||||||
|
# shellcheck disable=SC2329
|
||||||
|
cleanup() {
|
||||||
|
local status=$?
|
||||||
|
local cleanup_status=0
|
||||||
|
local index
|
||||||
|
local source_mount_failed=false
|
||||||
|
trap - EXIT HUP INT TERM
|
||||||
|
set +e
|
||||||
|
|
||||||
|
{% raw %}
|
||||||
|
for ((index = ${#mounted_targets[@]} - 1; index >= 0; index--)); do
|
||||||
|
{% endraw %}
|
||||||
|
if mountpoint -q "${mounted_targets[$index]}" && ! umount -R "${mounted_targets[$index]}"; then
|
||||||
|
printf 'Source snapshot mount cleanup failed: %s\n' "${mounted_targets[$index]}" >&2
|
||||||
|
source_mount_failed=true
|
||||||
|
fi
|
||||||
|
if mountpoint -q "${mounted_targets[$index]}"; then
|
||||||
|
printf 'Source snapshot mount is still active: %s\n' "${mounted_targets[$index]}" >&2
|
||||||
|
source_mount_failed=true
|
||||||
|
else
|
||||||
|
rmdir -- "${mounted_targets[$index]}" 2>/dev/null || true
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
if [[ "$source_mount_failed" == false ]]; then
|
||||||
|
if [[ -d "$stage" ]]; then
|
||||||
|
rmdir -- "$stage" 2>/dev/null || cleanup_status=2
|
||||||
|
fi
|
||||||
|
else
|
||||||
|
cleanup_status=2
|
||||||
|
printf 'Source bind mount cleanup failed; keeping the snapshot for recovery\n' >&2
|
||||||
|
fi
|
||||||
|
|
||||||
|
if ((status == 0 && cleanup_status != 0)); then
|
||||||
|
status=$cleanup_status
|
||||||
|
fi
|
||||||
|
exit "$status"
|
||||||
|
}
|
||||||
|
|
||||||
|
trap cleanup EXIT
|
||||||
|
trap 'exit 143' HUP INT TERM
|
||||||
|
|
||||||
|
run_as_borg() {
|
||||||
|
setpriv \
|
||||||
|
--reuid "$borg_user" \
|
||||||
|
--regid "$borg_group" \
|
||||||
|
--clear-groups \
|
||||||
|
--inh-caps=-all,+dac_read_search \
|
||||||
|
--ambient-caps=-all,+dac_read_search \
|
||||||
|
--bounding-set=-all,+dac_read_search \
|
||||||
|
-- env HOME="$borg_home" USER="$borg_user" LOGNAME="$borg_user" "$@"
|
||||||
|
}
|
||||||
|
|
||||||
|
exec 8>"$borg_lock"
|
||||||
|
flock 8
|
||||||
|
exec 9>/run/lock/atlas-zfs-snapshot.lock
|
||||||
|
|
||||||
|
zpool list -H -o name "$pool" >/dev/null
|
||||||
|
rm -rf "$stage"
|
||||||
|
mkdir -p "$stage"
|
||||||
|
# Keep systemd's root:root ownership of RuntimeDirectory: changing it makes
|
||||||
|
# ExecStopPost re-chown its contents, which SELinux denies for the marker.
|
||||||
|
setfacl -m "u:${borg_user}:rx" /run/atlas-borg
|
||||||
|
chown root:"$borg_group" "$stage"
|
||||||
|
chmod 0750 /run/atlas-borg "$stage"
|
||||||
|
|
||||||
|
flock 9
|
||||||
|
while IFS= read -r stale_snapshot; do
|
||||||
|
stale_suffix="${stale_snapshot#"${pool}@${snapshot_prefix}-"}"
|
||||||
|
if [[ "$stale_suffix" =~ ^[0-9]{8}T[0-9]{6}Z$ ]]; then
|
||||||
|
zfs destroy -r "$stale_snapshot"
|
||||||
|
printf 'Removed stale Borg source snapshot %s\n' "$stale_snapshot"
|
||||||
|
fi
|
||||||
|
done < <(
|
||||||
|
zfs list -H -t snapshot -o name -r "$pool" |
|
||||||
|
grep -E "^${pool}@${snapshot_prefix}-[0-9]{8}T[0-9]{6}Z$" || true
|
||||||
|
)
|
||||||
|
|
||||||
|
timestamp="$(date -u +%Y%m%dT%H%M%SZ)"
|
||||||
|
readonly timestamp
|
||||||
|
snapshot_name="${snapshot_prefix}-${timestamp}"
|
||||||
|
readonly snapshot_name
|
||||||
|
printf '%s\n' "$snapshot_name" >"$snapshot_marker"
|
||||||
|
zfs snapshot -r "${pool}@${snapshot_name}"
|
||||||
|
flock -u 9
|
||||||
|
printf 'Created recursive Borg source snapshot %s@%s\n' "$pool" "$snapshot_name"
|
||||||
|
|
||||||
|
while IFS=$'\t' read -r dataset dataset_mountpoint mounted; do
|
||||||
|
if [[ "$mounted" != yes ]]; then
|
||||||
|
printf 'Dataset %s is not mounted; refusing an incomplete backup\n' "$dataset" >&2
|
||||||
|
exit 65
|
||||||
|
fi
|
||||||
|
if [[ "$dataset_mountpoint" != "$mount_root" && "$dataset_mountpoint" != "$mount_root/"* ]]; then
|
||||||
|
printf 'Dataset %s has unexpected mountpoint %s\n' "$dataset" "$dataset_mountpoint" >&2
|
||||||
|
exit 65
|
||||||
|
fi
|
||||||
|
|
||||||
|
dataset_suffix="${dataset#"$pool"}"
|
||||||
|
source_path="${dataset_mountpoint}/.zfs/snapshot/${snapshot_name}"
|
||||||
|
target_path="${stage}${dataset_suffix}"
|
||||||
|
mkdir -p "$target_path"
|
||||||
|
mount --bind "$source_path" "$target_path"
|
||||||
|
mounted_targets+=("$target_path")
|
||||||
|
mount -o remount,bind,ro "$target_path"
|
||||||
|
done < <(zfs list -H -o name,mountpoint,mounted -s name -r "$pool")
|
||||||
|
|
||||||
|
estimated_source_bytes=0
|
||||||
|
while IFS=$'\t' read -r source_snapshot logical_bytes; do
|
||||||
|
if [[ "$source_snapshot" == *"@${snapshot_name}" ]]; then
|
||||||
|
[[ "$logical_bytes" =~ ^[0-9]+$ ]] || {
|
||||||
|
printf 'Invalid logical size for Borg source snapshot %s\n' "$source_snapshot" >&2
|
||||||
|
exit 74
|
||||||
|
}
|
||||||
|
estimated_source_bytes=$((estimated_source_bytes + logical_bytes))
|
||||||
|
fi
|
||||||
|
done < <(zfs list -H -p -t snapshot -o name,logicalreferenced -r "$pool")
|
||||||
|
((estimated_source_bytes > 0)) || {
|
||||||
|
printf 'Could not estimate the Borg source snapshot size\n' >&2
|
||||||
|
exit 74
|
||||||
|
}
|
||||||
|
printf 'Estimated Borg source logical size: %s bytes (ZFS; progress percentage is approximate)\n' \
|
||||||
|
"$estimated_source_bytes"
|
||||||
|
|
||||||
|
archive="${archive_prefix}-${timestamp}"
|
||||||
|
readonly archive
|
||||||
|
borg_status=0
|
||||||
|
|
||||||
|
printf 'Starting Borg archive %s from snapshot %s@%s\n' "$archive" "$pool" "$snapshot_name"
|
||||||
|
set +e
|
||||||
|
(
|
||||||
|
cd /run/atlas-borg
|
||||||
|
run_as_borg borg --remote-path "$remote_path" --lock-wait 600 --log-json --progress create \
|
||||||
|
--show-rc \
|
||||||
|
--stats \
|
||||||
|
--checkpoint-interval 900 \
|
||||||
|
--compression "$compression" \
|
||||||
|
"${repository}::${archive}" \
|
||||||
|
source 2>&1
|
||||||
|
) | /usr/bin/python3 -u "$progress_filter" --estimated-total-bytes "$estimated_source_bytes"
|
||||||
|
create_pipeline_status=("${PIPESTATUS[@]}")
|
||||||
|
set -e
|
||||||
|
create_status=${create_pipeline_status[0]}
|
||||||
|
if ((create_pipeline_status[1] != 0)); then
|
||||||
|
printf 'Borg progress logging failed with status %s\n' "${create_pipeline_status[1]}" >&2
|
||||||
|
exit 2
|
||||||
|
fi
|
||||||
|
if ((create_status >= 2)); then
|
||||||
|
exit "$create_status"
|
||||||
|
fi
|
||||||
|
borg_status=$create_status
|
||||||
|
|
||||||
|
printf 'Borg archive %s created; applying retention\n' "$archive"
|
||||||
|
set +e
|
||||||
|
run_as_borg borg --remote-path "$remote_path" --lock-wait 600 prune \
|
||||||
|
--show-rc \
|
||||||
|
--list \
|
||||||
|
--glob-archives "${archive_prefix}-*" \
|
||||||
|
--keep-daily {{ atlas_borg_keep_daily | int }} \
|
||||||
|
--keep-weekly {{ atlas_borg_keep_weekly | int }} \
|
||||||
|
--keep-monthly {{ atlas_borg_keep_monthly | int }} \
|
||||||
|
"$repository"
|
||||||
|
prune_status=$?
|
||||||
|
set -e
|
||||||
|
if ((prune_status >= 2)); then
|
||||||
|
exit "$prune_status"
|
||||||
|
fi
|
||||||
|
if ((prune_status > borg_status)); then
|
||||||
|
borg_status=$prune_status
|
||||||
|
fi
|
||||||
|
|
||||||
|
printf 'Borg retention complete; compacting repository\n'
|
||||||
|
set +e
|
||||||
|
run_as_borg borg --remote-path "$remote_path" --lock-wait 600 compact \
|
||||||
|
--show-rc \
|
||||||
|
"$repository"
|
||||||
|
compact_status=$?
|
||||||
|
set -e
|
||||||
|
if ((compact_status >= 2)); then
|
||||||
|
exit "$compact_status"
|
||||||
|
fi
|
||||||
|
if ((compact_status > borg_status)); then
|
||||||
|
borg_status=$compact_status
|
||||||
|
fi
|
||||||
|
|
||||||
|
printf 'Borg backup %s completed with status %s\n' "$archive" "$borg_status"
|
||||||
|
exit "$borg_status"
|
||||||
@@ -0,0 +1,12 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Schedule the encrypted Atlas Borg backup
|
||||||
|
|
||||||
|
[Timer]
|
||||||
|
OnCalendar={{ atlas_borg_backup_calendar }}
|
||||||
|
Persistent=true
|
||||||
|
RandomizedDelaySec={{ atlas_borg_randomized_delay }}
|
||||||
|
AccuracySec=1min
|
||||||
|
Unit=atlas-borg-backup.service
|
||||||
|
|
||||||
|
[Install]
|
||||||
|
WantedBy=timers.target
|
||||||
@@ -0,0 +1,32 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Check the encrypted Atlas Borg repository
|
||||||
|
Documentation=man:borg-check(1)
|
||||||
|
Wants=network-online.target
|
||||||
|
After=network-online.target atlas-borg-backup.service
|
||||||
|
ConditionFileIsExecutable=/usr/local/sbin/atlas-borg-check
|
||||||
|
ConditionPathExists={{ atlas_borg_passphrase_path }}
|
||||||
|
ConditionPathExists={{ atlas_borg_ssh_private_key_path }}
|
||||||
|
ConditionPathExists={{ atlas_borg_known_hosts_path }}
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
Type=oneshot
|
||||||
|
ExecStart=/usr/local/sbin/atlas-borg-check
|
||||||
|
User={{ atlas_borg_username }}
|
||||||
|
Group={{ atlas_borg_group }}
|
||||||
|
UMask=0077
|
||||||
|
SuccessExitStatus=1
|
||||||
|
TimeoutStartSec=infinity
|
||||||
|
Nice=15
|
||||||
|
IOSchedulingClass=best-effort
|
||||||
|
IOSchedulingPriority=7
|
||||||
|
NoNewPrivileges=true
|
||||||
|
PrivateTmp=true
|
||||||
|
ProtectHome=true
|
||||||
|
ProtectSystem=strict
|
||||||
|
ReadWritePaths={{ atlas_borg_cache_dir }} {{ atlas_borg_config_dir }}
|
||||||
|
ProtectKernelTunables=true
|
||||||
|
ProtectKernelModules=true
|
||||||
|
ProtectControlGroups=true
|
||||||
|
RestrictAddressFamilies=AF_UNIX AF_INET AF_INET6
|
||||||
|
RestrictRealtime=true
|
||||||
|
LockPersonality=true
|
||||||
23
ansible/roles/profile_atlas/templates/atlas-borg-check.sh.j2
Normal file
23
ansible/roles/profile_atlas/templates/atlas-borg-check.sh.j2
Normal file
@@ -0,0 +1,23 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -Eeuo pipefail
|
||||||
|
|
||||||
|
export LC_ALL=C
|
||||||
|
export PATH=/usr/sbin:/usr/bin:/sbin:/bin
|
||||||
|
export BORG_CACHE_DIR={{ atlas_borg_cache_dir | quote }}
|
||||||
|
export BORG_CONFIG_DIR={{ atlas_borg_config_dir | quote }}
|
||||||
|
export BORG_PASSCOMMAND={{ ('cat ' ~ atlas_borg_passphrase_path) | quote }}
|
||||||
|
export BORG_RSH={{ atlas_borg_ssh_wrapper_path | quote }}
|
||||||
|
|
||||||
|
readonly repository={{ ('ssh://' ~ atlas_borg_repository_user ~ '@' ~ atlas_borg_repository_host
|
||||||
|
~ ':' ~ (atlas_borg_repository_port | string) ~ '/' ~ atlas_borg_repository_path) | quote }}
|
||||||
|
readonly remote_path={{ atlas_borg_remote_path | quote }}
|
||||||
|
readonly archive_prefix={{ atlas_borg_archive_prefix | quote }}
|
||||||
|
readonly borg_lock={{ atlas_borg_lock_path | quote }}
|
||||||
|
|
||||||
|
exec 8>"$borg_lock"
|
||||||
|
flock 8
|
||||||
|
|
||||||
|
exec borg --remote-path "$remote_path" --lock-wait 600 check \
|
||||||
|
--show-rc \
|
||||||
|
--glob-archives "${archive_prefix}-*" \
|
||||||
|
"$repository"
|
||||||
@@ -0,0 +1,12 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Schedule checks of the encrypted Atlas Borg repository
|
||||||
|
|
||||||
|
[Timer]
|
||||||
|
OnCalendar={{ atlas_borg_check_calendar }}
|
||||||
|
Persistent=true
|
||||||
|
RandomizedDelaySec={{ atlas_borg_randomized_delay }}
|
||||||
|
AccuracySec=1min
|
||||||
|
Unit=atlas-borg-check.service
|
||||||
|
|
||||||
|
[Install]
|
||||||
|
WantedBy=timers.target
|
||||||
@@ -0,0 +1,51 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -Eeuo pipefail
|
||||||
|
|
||||||
|
export PATH=/usr/sbin:/usr/bin:/sbin:/bin
|
||||||
|
|
||||||
|
readonly pool={{ atlas_zfs_pool | quote }}
|
||||||
|
readonly mount_root={{ atlas_mount_root | quote }}
|
||||||
|
readonly snapshot_prefix={{ atlas_borg_snapshot_prefix | quote }}
|
||||||
|
readonly marker=/run/atlas-borg/snapshot-name
|
||||||
|
|
||||||
|
[[ -e "$marker" ]] || exit 0
|
||||||
|
[[ -f "$marker" && ! -L "$marker" ]] || {
|
||||||
|
printf 'Unsafe Atlas Borg snapshot marker; leaving snapshots unchanged\n' >&2
|
||||||
|
exit 2
|
||||||
|
}
|
||||||
|
IFS= read -r snapshot_name <"$marker"
|
||||||
|
[[ "$snapshot_name" =~ ^${snapshot_prefix}-[0-9]{8}T[0-9]{6}Z$ ]] || {
|
||||||
|
printf 'Invalid Atlas Borg snapshot marker; leaving snapshots unchanged\n' >&2
|
||||||
|
exit 2
|
||||||
|
}
|
||||||
|
|
||||||
|
exec 9>/run/lock/atlas-zfs-snapshot.lock
|
||||||
|
flock 9
|
||||||
|
if zfs list -H -t snapshot -o name "${pool}@${snapshot_name}" >/dev/null 2>&1; then
|
||||||
|
# The private bind mounts are gone, but ZFS may leave its on-demand
|
||||||
|
# .zfs/snapshot mounts in the host namespace until explicitly unmounted.
|
||||||
|
snapshot_mounts=()
|
||||||
|
snapshot_sources=()
|
||||||
|
while IFS=$'\t' read -r dataset dataset_mountpoint; do
|
||||||
|
[[ "$dataset_mountpoint" == "$mount_root" || "$dataset_mountpoint" == "$mount_root/"* ]] || continue
|
||||||
|
snapshot_mounts+=("${dataset_mountpoint}/.zfs/snapshot/${snapshot_name}")
|
||||||
|
snapshot_sources+=("${dataset}@${snapshot_name}")
|
||||||
|
done < <(zfs list -H -o name,mountpoint -s name -r "$pool")
|
||||||
|
|
||||||
|
{% raw %}
|
||||||
|
for ((index = ${#snapshot_mounts[@]} - 1; index >= 0; index--)); do
|
||||||
|
{% endraw %}
|
||||||
|
mounted_source="$(findmnt -rn -M "${snapshot_mounts[$index]}" -o SOURCE || true)"
|
||||||
|
[[ -n "$mounted_source" ]] || continue
|
||||||
|
[[ "$mounted_source" == "${snapshot_sources[$index]}" ]] || {
|
||||||
|
printf 'Unexpected source on Atlas Borg snapshot mount: %s\n' \
|
||||||
|
"${snapshot_mounts[$index]}" >&2
|
||||||
|
exit 2
|
||||||
|
}
|
||||||
|
umount "${snapshot_mounts[$index]}"
|
||||||
|
done
|
||||||
|
|
||||||
|
zfs destroy -r "${pool}@${snapshot_name}"
|
||||||
|
printf 'Removed recursive Atlas Borg source snapshot %s@%s after backup exit\n' \
|
||||||
|
"$pool" "$snapshot_name"
|
||||||
|
fi
|
||||||
19
ansible/roles/profile_atlas/templates/atlas-borg-ssh.sh.j2
Normal file
19
ansible/roles/profile_atlas/templates/atlas-borg-ssh.sh.j2
Normal file
@@ -0,0 +1,19 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
# Borg receives CAP_DAC_READ_SEARCH only for local snapshot traversal. Drop it
|
||||||
|
# before starting the network transport so SSH runs as the plain service user.
|
||||||
|
exec setpriv \
|
||||||
|
--inh-caps=-all \
|
||||||
|
--ambient-caps=-all \
|
||||||
|
-- /usr/bin/ssh \
|
||||||
|
-i {{ atlas_borg_ssh_private_key_path | quote }} \
|
||||||
|
-p {{ atlas_borg_repository_port | int }} \
|
||||||
|
-o BatchMode=yes \
|
||||||
|
-o IdentitiesOnly=yes \
|
||||||
|
-o StrictHostKeyChecking=yes \
|
||||||
|
-o UserKnownHostsFile={{ atlas_borg_known_hosts_path | quote }} \
|
||||||
|
-o ConnectTimeout=30 \
|
||||||
|
-o ServerAliveInterval=60 \
|
||||||
|
-o ServerAliveCountMax=3 \
|
||||||
|
"$@"
|
||||||
@@ -0,0 +1,30 @@
|
|||||||
|
# Managed by Ansible. Staging does not start automatically.
|
||||||
|
[Unit]
|
||||||
|
Description=Atlas rootless Gitea
|
||||||
|
RequiresMountsFor={{ atlas_gitea_mountpoint }}
|
||||||
|
|
||||||
|
[Container]
|
||||||
|
ContainerName=atlas-gitea
|
||||||
|
Image={{ atlas_gitea_image }}
|
||||||
|
UserNS=keep-id:uid={{ atlas_gitea_container_uid }},gid={{ atlas_gitea_container_gid }}
|
||||||
|
{% if atlas_gitea_production_enabled | bool %}
|
||||||
|
PublishPort={{ atlas_gitea_bind_address }}:{{ atlas_gitea_http_port }}:3000
|
||||||
|
PublishPort={{ atlas_gitea_bind_address }}:{{ atlas_gitea_ssh_port }}:2222
|
||||||
|
{% else %}
|
||||||
|
PublishPort={{ atlas_gitea_staging_bind_address }}:{{ atlas_gitea_staging_http_port }}:3000
|
||||||
|
PublishPort={{ atlas_gitea_staging_bind_address }}:{{ atlas_gitea_staging_ssh_port }}:2222
|
||||||
|
{% endif %}
|
||||||
|
Volume={{ atlas_gitea_mountpoint }}/data:/var/lib/gitea:Z
|
||||||
|
Volume={{ atlas_gitea_mountpoint }}/config:/etc/gitea:Z
|
||||||
|
NoNewPrivileges=true
|
||||||
|
DropCapability=all
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
Restart=on-failure
|
||||||
|
RestartSec=10
|
||||||
|
TimeoutStartSec=900
|
||||||
|
{% if atlas_gitea_production_enabled | bool %}
|
||||||
|
|
||||||
|
[Install]
|
||||||
|
WantedBy=default.target
|
||||||
|
{% endif %}
|
||||||
@@ -0,0 +1,20 @@
|
|||||||
|
{
|
||||||
|
"pool": {{ atlas_zfs_pool | to_json }},
|
||||||
|
"backup_dataset": {{ (atlas_zfs_pool ~ '/' ~ atlas_zfs_dataset_backup) | to_json }},
|
||||||
|
"notifier": {{ atlas_monitor_notifier | to_json }},
|
||||||
|
"smart_devices": {{ atlas_monitor_smart_devices | to_json }},
|
||||||
|
"timers": {{ atlas_monitor_effective_timers | to_json }},
|
||||||
|
"failure_units": {{ atlas_monitor_effective_failure_units | to_json }},
|
||||||
|
"remote_capacity": {{ atlas_monitor_remote_capacity | to_json }},
|
||||||
|
"pool_warning_percent": {{ atlas_monitor_pool_warning_percent | int }},
|
||||||
|
"pool_critical_percent": {{ atlas_monitor_pool_critical_percent | int }},
|
||||||
|
"root_warning_percent": {{ atlas_monitor_root_warning_percent | int }},
|
||||||
|
"root_critical_percent": {{ atlas_monitor_root_critical_percent | int }},
|
||||||
|
"snapshot_warning_percent": {{ atlas_monitor_snapshot_warning_percent | int }},
|
||||||
|
"snapshot_critical_percent": {{ atlas_monitor_snapshot_critical_percent | int }},
|
||||||
|
"snapshot_growth_warning_gib_day": {{ atlas_monitor_snapshot_growth_warning_gib_day | int }},
|
||||||
|
"backup_growth_warning_gib_day": {{ atlas_monitor_backup_growth_warning_gib_day | int }},
|
||||||
|
"cpu_warning_c": {{ atlas_monitor_cpu_warning_c | int }},
|
||||||
|
"cpu_critical_c": {{ atlas_monitor_cpu_critical_c | int }},
|
||||||
|
"borg_max_runtime_days": {{ atlas_monitor_borg_max_runtime_days | int }}
|
||||||
|
}
|
||||||
@@ -0,0 +1,25 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Check Atlas pool, disks, capacity, temperatures and maintenance jobs
|
||||||
|
Wants=houston-dbus.service network-online.target
|
||||||
|
After=zfs.target houston-dbus.service network-online.target
|
||||||
|
ConditionFileIsExecutable=/usr/local/libexec/atlas-health-monitor
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
Type=oneshot
|
||||||
|
ExecStart=/usr/local/libexec/atlas-health-monitor
|
||||||
|
User=root
|
||||||
|
Group=root
|
||||||
|
UMask=0077
|
||||||
|
StateDirectory=atlas-health-monitor
|
||||||
|
StateDirectoryMode=0700
|
||||||
|
NoNewPrivileges=true
|
||||||
|
PrivateTmp=true
|
||||||
|
ProtectHome=true
|
||||||
|
ProtectSystem=strict
|
||||||
|
ReadWritePaths=/var/lib/atlas-health-monitor
|
||||||
|
ProtectKernelTunables=true
|
||||||
|
ProtectKernelModules=true
|
||||||
|
ProtectControlGroups=true
|
||||||
|
RestrictAddressFamilies=AF_UNIX AF_INET AF_INET6
|
||||||
|
RestrictRealtime=true
|
||||||
|
LockPersonality=true
|
||||||
@@ -0,0 +1,11 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Schedule Atlas health checks
|
||||||
|
|
||||||
|
[Timer]
|
||||||
|
OnCalendar={{ atlas_monitor_calendar }}
|
||||||
|
Persistent=true
|
||||||
|
RandomizedDelaySec=5min
|
||||||
|
Unit=atlas-health-monitor.service
|
||||||
|
|
||||||
|
[Install]
|
||||||
|
WantedBy=timers.target
|
||||||
14
ansible/roles/profile_atlas/templates/atlas-icloudpd.conf.j2
Normal file
14
ansible/roles/profile_atlas/templates/atlas-icloudpd.conf.j2
Normal file
@@ -0,0 +1,14 @@
|
|||||||
|
# Managed by Ansible. Password, keyring and MFA cookies are stored separately in /config.
|
||||||
|
apple_id={{ vault_atlas_icloudpd_apple_id }}
|
||||||
|
authentication_type=MFA
|
||||||
|
user=user
|
||||||
|
user_id=1000
|
||||||
|
group=group
|
||||||
|
group_id=1000
|
||||||
|
download_path=/home/user/iCloud
|
||||||
|
folder_structure={:%Y/%m/%d}
|
||||||
|
directory_permissions=750
|
||||||
|
file_permissions=640
|
||||||
|
download_interval=86400
|
||||||
|
auto_delete=false
|
||||||
|
delete_after_download=false
|
||||||
@@ -0,0 +1,25 @@
|
|||||||
|
# Managed by Ansible. Start automatically with the lingering admin user manager.
|
||||||
|
[Unit]
|
||||||
|
Description=Atlas rootless iCloud Photos Downloader
|
||||||
|
RequiresMountsFor={{ atlas_icloudpd_state_dir }} {{ atlas_icloudpd_photos_dir }}
|
||||||
|
|
||||||
|
[Container]
|
||||||
|
ContainerName=atlas-icloudpd
|
||||||
|
Image={{ atlas_icloudpd_image }}
|
||||||
|
UserNS=keep-id:uid=1000,gid=1000
|
||||||
|
# The image initialises its unprivileged UID 1000 account as container root.
|
||||||
|
User=0
|
||||||
|
# Upstream launcher requires traceroute for its iCloud reachability check.
|
||||||
|
AddCapability=NET_RAW
|
||||||
|
Environment=TZ={{ atlas_icloudpd_timezone }}
|
||||||
|
Volume={{ atlas_icloudpd_photos_dir }}:/home/user/iCloud:z
|
||||||
|
Volume={{ atlas_icloudpd_config_dir }}:/config:Z
|
||||||
|
NoNewPrivileges=true
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
Restart=on-failure
|
||||||
|
RestartSec=300
|
||||||
|
TimeoutStartSec=900
|
||||||
|
|
||||||
|
[Install]
|
||||||
|
WantedBy=default.target
|
||||||
@@ -0,0 +1,2 @@
|
|||||||
|
[Unit]
|
||||||
|
OnFailure=atlas-monitor-failure@%n.service
|
||||||
@@ -0,0 +1,19 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Submit a 45Drives Alert for failed Atlas job %i
|
||||||
|
Requires=houston-dbus.service
|
||||||
|
After=houston-dbus.service
|
||||||
|
ConditionFileIsExecutable=/usr/local/libexec/atlas-health-monitor
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
Type=oneshot
|
||||||
|
ExecStart=/usr/local/libexec/atlas-health-monitor --job-failed %i
|
||||||
|
User=root
|
||||||
|
Group=root
|
||||||
|
UMask=0077
|
||||||
|
NoNewPrivileges=true
|
||||||
|
PrivateTmp=true
|
||||||
|
ProtectHome=true
|
||||||
|
ProtectSystem=strict
|
||||||
|
RestrictAddressFamilies=AF_UNIX
|
||||||
|
RestrictRealtime=true
|
||||||
|
LockPersonality=true
|
||||||
@@ -0,0 +1,3 @@
|
|||||||
|
[Unit]
|
||||||
|
Requires=atlas-nextcloud-backup.service
|
||||||
|
After=atlas-nextcloud-backup.service
|
||||||
@@ -0,0 +1,19 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Recover interrupted Nextcloud backup preparation after boot
|
||||||
|
Requires=zfs.target user@{{ atlas_admin_uid }}.service
|
||||||
|
After=zfs.target user@{{ atlas_admin_uid }}.service
|
||||||
|
{% if atlas_manage_monitoring | bool %}
|
||||||
|
OnFailure=atlas-monitor-failure@%n.service
|
||||||
|
{% endif %}
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
Type=oneshot
|
||||||
|
User=root
|
||||||
|
UMask=0077
|
||||||
|
StateDirectory=atlas-nextcloud-backup
|
||||||
|
StateDirectoryMode=0700
|
||||||
|
ExecStart=/usr/local/sbin/atlas-nextcloud-backup --recover
|
||||||
|
TimeoutStartSec=5min
|
||||||
|
|
||||||
|
[Install]
|
||||||
|
WantedBy=multi-user.target
|
||||||
@@ -0,0 +1,21 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Prepare a consistent Nextcloud bundle before Atlas backups
|
||||||
|
Requires=zfs.target user@{{ atlas_admin_uid }}.service
|
||||||
|
After=zfs.target user@{{ atlas_admin_uid }}.service atlas-nextcloud-backup-recovery.service
|
||||||
|
{% if atlas_manage_monitoring | bool %}
|
||||||
|
OnFailure=atlas-monitor-failure@%n.service
|
||||||
|
{% endif %}
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
Type=oneshot
|
||||||
|
User=root
|
||||||
|
UMask=0077
|
||||||
|
StateDirectory=atlas-nextcloud-backup
|
||||||
|
StateDirectoryMode=0700
|
||||||
|
ExecStart=/usr/local/sbin/atlas-nextcloud-backup
|
||||||
|
ExecStopPost=/usr/local/sbin/atlas-nextcloud-backup --recover
|
||||||
|
TimeoutStartSec=3h
|
||||||
|
TimeoutStopSec=5min
|
||||||
|
Nice=10
|
||||||
|
IOSchedulingClass=best-effort
|
||||||
|
IOSchedulingPriority=7
|
||||||
@@ -0,0 +1,133 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -Eeuo pipefail
|
||||||
|
export PATH=/usr/sbin:/usr/bin:/sbin:/bin
|
||||||
|
umask 077
|
||||||
|
readonly dataset={{ atlas_nextcloud_dataset | quote }}
|
||||||
|
readonly source_root={{ atlas_nextcloud_root | quote }}
|
||||||
|
readonly backup_root={{ atlas_nextcloud_backup_root | quote }}
|
||||||
|
readonly state=/var/lib/atlas-nextcloud-backup
|
||||||
|
readonly keep={{ atlas_nextcloud_backup_keep | int }}
|
||||||
|
readonly owner={{ atlas_admin_username | quote }}
|
||||||
|
readonly uid={{ atlas_admin_uid | int }}
|
||||||
|
|
||||||
|
user_run() {
|
||||||
|
runuser -u "$owner" -- env XDG_RUNTIME_DIR="/run/user/$uid" \
|
||||||
|
DBUS_SESSION_BUS_ADDRESS="unix:path=/run/user/$uid/bus" "$@"
|
||||||
|
}
|
||||||
|
occ() { user_run podman exec --user 33 atlas-nextcloud php occ "$@"; }
|
||||||
|
exec 8>/run/lock/atlas-nextcloud-backup.lock
|
||||||
|
flock 8
|
||||||
|
exec 9>/run/lock/atlas-zfs-snapshot.lock
|
||||||
|
mkdir -p "$state"
|
||||||
|
chmod 0700 "$state"
|
||||||
|
|
||||||
|
resume() {
|
||||||
|
[[ -e "$state/paused" ]] || return 0
|
||||||
|
user_run systemctl --user start atlas-nextcloud.service atlas-onlyoffice.service
|
||||||
|
local ready=false
|
||||||
|
for _ in {1..60}; do
|
||||||
|
if occ maintenance:mode --off >/dev/null 2>&1; then ready=true; break; fi
|
||||||
|
sleep 2
|
||||||
|
done
|
||||||
|
[[ "$ready" == true ]] || { echo 'Nextcloud resume failed; recovery marker retained' >&2; return 1; }
|
||||||
|
user_run systemctl --user start atlas-nextcloud-cron.timer
|
||||||
|
# Persist maintenance-off before clearing durable interruption ownership.
|
||||||
|
sync -f "$source_root/app"
|
||||||
|
rm "$state/paused"
|
||||||
|
sync -f "$state"
|
||||||
|
echo 'Nextcloud/Office resumed and cron timer restored'
|
||||||
|
}
|
||||||
|
|
||||||
|
recover() {
|
||||||
|
resume || return 1
|
||||||
|
[[ -e "$state/stamp" ]] || return 0
|
||||||
|
local stamp snapshot mount source
|
||||||
|
stamp=$(cat "$state/stamp")
|
||||||
|
[[ "$stamp" =~ ^[0-9]{8}T[0-9]{6}Z-[0-9]+$ ]] || return 65
|
||||||
|
snapshot="nc-backup-$stamp"
|
||||||
|
flock 9
|
||||||
|
for component in files app; do
|
||||||
|
mount="$source_root/$component/.zfs/snapshot/$snapshot"
|
||||||
|
source=$(findmnt -rn -M "$mount" -o SOURCE || true)
|
||||||
|
if [[ -n "$source" ]]; then
|
||||||
|
[[ "$source" == "$dataset/$component@$snapshot" ]] || return 65
|
||||||
|
umount "$mount" || return 1
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
if zfs list -H -t snapshot "$dataset@$snapshot" >/dev/null 2>&1; then
|
||||||
|
zfs destroy -r "$dataset@$snapshot" || return 1
|
||||||
|
fi
|
||||||
|
flock -u 9
|
||||||
|
# Only this job's private, unpublished staging directory can be removed.
|
||||||
|
rm -rf -- "$backup_root/.partial-$stamp"
|
||||||
|
rm "$state/stamp"
|
||||||
|
}
|
||||||
|
if [[ "${1:-}" == --recover ]]; then recover; exit; fi
|
||||||
|
recover
|
||||||
|
[[ "$(zfs get -H -o value mounted "$dataset")" == yes ]]
|
||||||
|
[[ "$(zfs get -H -o value mountpoint "$dataset")" == "$source_root" ]]
|
||||||
|
[[ "$(zfs get -H -o value mounted {{ (atlas_zfs_pool ~ '/backup') | quote }})" == yes ]]
|
||||||
|
[[ "$(zfs get -H -o value mountpoint {{ (atlas_zfs_pool ~ '/backup') | quote }})" == {{ (atlas_mount_root ~ '/backup') | quote }} ]]
|
||||||
|
for component in app files; do
|
||||||
|
[[ "$(zfs get -H -o value mounted "$dataset/$component")" == yes ]]
|
||||||
|
[[ "$(zfs get -H -o value mountpoint "$dataset/$component")" == "$source_root/$component" ]]
|
||||||
|
done
|
||||||
|
for unit in atlas-nextcloud.service atlas-onlyoffice.service atlas-nextcloud-cron.timer; do
|
||||||
|
user_run systemctl --user is-active --quiet "$unit"
|
||||||
|
done
|
||||||
|
occ status --output=json | python3 -c 'import json,sys; s=json.load(sys.stdin); assert s["installed"] and not s["maintenance"] and not s["needsDbUpgrade"]'
|
||||||
|
mkdir -p "$backup_root/versions"
|
||||||
|
chmod 0700 "$backup_root" "$backup_root/versions"
|
||||||
|
stamp="$(date -u +%Y%m%dT%H%M%SZ)-$$"
|
||||||
|
snapshot="nc-backup-$stamp"
|
||||||
|
stage="$backup_root/.partial-$stamp"
|
||||||
|
mkdir "$stage"
|
||||||
|
printf '%s\n' "$stamp" > "$state/stamp"
|
||||||
|
sync -f "$state"
|
||||||
|
cleanup() {
|
||||||
|
local rc=$?
|
||||||
|
trap - EXIT
|
||||||
|
if ! recover; then rc=1; fi
|
||||||
|
exit "$rc"
|
||||||
|
}
|
||||||
|
trap cleanup EXIT
|
||||||
|
trap 'exit 143' HUP INT TERM
|
||||||
|
# Wait for snapshot serialization before interrupting application availability.
|
||||||
|
flock 9
|
||||||
|
touch "$state/paused"
|
||||||
|
sync -f "$state"
|
||||||
|
user_run systemctl --user stop atlas-nextcloud-cron.timer atlas-nextcloud-cron.service
|
||||||
|
occ maintenance:mode --on
|
||||||
|
user_run systemctl --user stop atlas-onlyoffice.service atlas-nextcloud.service
|
||||||
|
user_run podman exec atlas-nextcloud-db pg_dumpall -U nextcloud --globals-only > "$stage/postgres-globals.sql"
|
||||||
|
user_run podman exec atlas-nextcloud-db pg_dump -U nextcloud -d nextcloud --format=custom > "$stage/database.dump"
|
||||||
|
zfs snapshot -r "$dataset@$snapshot"
|
||||||
|
flock -u 9
|
||||||
|
resume
|
||||||
|
# Copy immutable snapshot views; hashing and transfer never extend the outage.
|
||||||
|
previous=$(readlink -f "$backup_root/latest" 2>/dev/null || true)
|
||||||
|
for component in app files; do
|
||||||
|
args=(-aHAX)
|
||||||
|
if [[ "$previous" == "$backup_root/versions/"* && -d "$previous/$component" ]]; then
|
||||||
|
args+=("--link-dest=$previous/$component")
|
||||||
|
fi
|
||||||
|
if [[ "$component" == app ]]; then args+=(--exclude=/data); fi
|
||||||
|
rsync "${args[@]}" "$source_root/$component/.zfs/snapshot/$snapshot/" "$stage/$component/"
|
||||||
|
done
|
||||||
|
user_run podman exec -i atlas-nextcloud-db pg_restore --list < "$stage/database.dump" > "$stage/database-toc.txt"
|
||||||
|
user_run podman inspect --format '{% raw %}{{.ImageName}}{% endraw %}' atlas-nextcloud atlas-nextcloud-db atlas-nextcloud-redis atlas-onlyoffice > "$stage/images.txt"
|
||||||
|
(cd "$stage"; find app files -type f -exec sha256sum '{}' +; sha256sum database.dump postgres-globals.sql images.txt) > "$stage/SHA256SUMS"
|
||||||
|
(cd "$stage"; sha256sum --quiet --check SHA256SUMS)
|
||||||
|
printf 'snapshot=%s@%s\ncreated_utc=%s\n' "$dataset" "$snapshot" "$stamp" > "$stage/manifest.txt"
|
||||||
|
mv "$stage" "$backup_root/versions/$stamp"
|
||||||
|
ln -s "versions/$stamp" "$backup_root/.latest-$stamp"
|
||||||
|
mv -Tf "$backup_root/.latest-$stamp" "$backup_root/latest"
|
||||||
|
sync -f "$backup_root"
|
||||||
|
# Prune only timestamped job-owned versions after verified atomic publication.
|
||||||
|
mapfile -t versions < <(find "$backup_root/versions" -mindepth 1 -maxdepth 1 -type d -printf '%f\n' | grep -E '^[0-9]{8}T[0-9]{6}Z-[0-9]+$' | sort -r)
|
||||||
|
{% raw %}
|
||||||
|
for ((index=keep; index<${#versions[@]}; index++)); do
|
||||||
|
{% endraw %}
|
||||||
|
rm -rf -- "$backup_root/versions/${versions[$index]}"
|
||||||
|
done
|
||||||
|
echo "Published verified consistent Nextcloud bundle $stamp; local retention=$keep"
|
||||||
@@ -0,0 +1,10 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Atlas recurring Nextcloud background jobs
|
||||||
|
Requires=atlas-nextcloud.service
|
||||||
|
After=atlas-nextcloud.service
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
Type=oneshot
|
||||||
|
ExecStart=/usr/bin/podman exec --user 33 atlas-nextcloud php -f /var/www/html/cron.php
|
||||||
|
TimeoutStartSec=15min
|
||||||
|
NoNewPrivileges=true
|
||||||
@@ -0,0 +1,10 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Run Nextcloud background jobs every five minutes
|
||||||
|
|
||||||
|
[Timer]
|
||||||
|
OnBootSec=5min
|
||||||
|
OnUnitActiveSec=5min
|
||||||
|
Unit=atlas-nextcloud-cron.service
|
||||||
|
|
||||||
|
[Install]
|
||||||
|
WantedBy=timers.target
|
||||||
@@ -0,0 +1,27 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Atlas Nextcloud PostgreSQL
|
||||||
|
RequiresMountsFor={{ atlas_nextcloud_root }}/database
|
||||||
|
|
||||||
|
[Container]
|
||||||
|
ContainerName=atlas-nextcloud-db
|
||||||
|
Image={{ atlas_nextcloud_postgres_image }}
|
||||||
|
Network=atlas-nextcloud.network
|
||||||
|
NetworkAlias=atlas-nextcloud-db
|
||||||
|
Environment=POSTGRES_DB=nextcloud
|
||||||
|
Environment=POSTGRES_USER=nextcloud
|
||||||
|
Environment=POSTGRES_PASSWORD_FILE=/run/secrets/postgres-password
|
||||||
|
Volume={{ atlas_nextcloud_private_dir }}/postgres-password:/run/secrets/postgres-password:ro,z
|
||||||
|
Volume={{ atlas_nextcloud_root }}/database:/var/lib/postgresql/data:Z
|
||||||
|
PodmanArgs=--memory=1g
|
||||||
|
HealthCmd=pg_isready -U nextcloud -d nextcloud
|
||||||
|
HealthInterval=30s
|
||||||
|
HealthStartPeriod=60s
|
||||||
|
NoNewPrivileges=true
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
Restart=on-failure
|
||||||
|
RestartSec=10
|
||||||
|
TimeoutStartSec=900
|
||||||
|
|
||||||
|
[Install]
|
||||||
|
WantedBy=default.target
|
||||||
@@ -0,0 +1,12 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Discover existing Archive documents and iCloud photos in Nextcloud
|
||||||
|
Requires=atlas-nextcloud.service
|
||||||
|
After=atlas-nextcloud.service
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
Type=oneshot
|
||||||
|
{% for mount in atlas_nextcloud_external_mounts %}
|
||||||
|
ExecStart=/usr/bin/podman exec --user 33 atlas-nextcloud php occ files:scan "--path={{ mount.user }}/files/{{ mount.name }}" --quiet
|
||||||
|
{% endfor %}
|
||||||
|
TimeoutStartSec=90min
|
||||||
|
NoNewPrivileges=true
|
||||||
@@ -0,0 +1,10 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Periodic discovery of Archive changes made outside Nextcloud
|
||||||
|
|
||||||
|
[Timer]
|
||||||
|
OnBootSec=15min
|
||||||
|
OnUnitInactiveSec=1h
|
||||||
|
Unit=atlas-nextcloud-external-scan.service
|
||||||
|
|
||||||
|
[Install]
|
||||||
|
WantedBy=timers.target
|
||||||
@@ -0,0 +1,8 @@
|
|||||||
|
bind 0.0.0.0
|
||||||
|
protected-mode yes
|
||||||
|
port 6379
|
||||||
|
requirepass {{ vault_nextcloud_redis_password }}
|
||||||
|
maxmemory 128mb
|
||||||
|
maxmemory-policy noeviction
|
||||||
|
save ""
|
||||||
|
appendonly no
|
||||||
@@ -0,0 +1,22 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Atlas Nextcloud private Redis
|
||||||
|
RequiresMountsFor={{ atlas_nextcloud_root }}/cache
|
||||||
|
|
||||||
|
[Container]
|
||||||
|
ContainerName=atlas-nextcloud-redis
|
||||||
|
Image={{ atlas_nextcloud_redis_image }}
|
||||||
|
Network=atlas-nextcloud.network
|
||||||
|
NetworkAlias=atlas-nextcloud-redis
|
||||||
|
Volume={{ atlas_nextcloud_private_dir }}/redis.conf:/usr/local/etc/redis/atlas.conf:ro,z
|
||||||
|
Volume={{ atlas_nextcloud_root }}/cache:/data:Z
|
||||||
|
Exec=redis-server /usr/local/etc/redis/atlas.conf
|
||||||
|
PodmanArgs=--memory=256m
|
||||||
|
NoNewPrivileges=true
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
Restart=on-failure
|
||||||
|
RestartSec=10
|
||||||
|
TimeoutStartSec=900
|
||||||
|
|
||||||
|
[Install]
|
||||||
|
WantedBy=default.target
|
||||||
@@ -0,0 +1,23 @@
|
|||||||
|
<?php
|
||||||
|
// Managed ongoing application settings; never import or migrate user data.
|
||||||
|
$CONFIG = [
|
||||||
|
'trusted_domains' => ['{{ atlas_nextcloud_domain }}', 'atlas-nextcloud'],
|
||||||
|
'trusted_proxies' => ['{{ atlas_aegis_ip }}', '{{ atlas_nextcloud_network_gateway }}'],
|
||||||
|
'overwrite.cli.url' => 'https://{{ atlas_nextcloud_domain }}',
|
||||||
|
'overwritehost' => '{{ atlas_nextcloud_domain }}',
|
||||||
|
'overwriteprotocol' => 'https',
|
||||||
|
'allow_local_remote_servers' => true,
|
||||||
|
'default_quota' => 'none',
|
||||||
|
'skeletondirectory' => '',
|
||||||
|
'maintenance_window_start' => 1,
|
||||||
|
'default_phone_region' => 'IT',
|
||||||
|
'twofactor_enforced' => false,
|
||||||
|
'onlyoffice' => [
|
||||||
|
'DocumentServerUrl' => 'https://{{ atlas_onlyoffice_domain }}/',
|
||||||
|
'DocumentServerInternalUrl' => 'http://atlas-onlyoffice/',
|
||||||
|
'StorageUrl' => 'http://atlas-nextcloud/',
|
||||||
|
'jwt_secret' => trim(file_get_contents('/run/secrets/onlyoffice-jwt')),
|
||||||
|
'jwt_header' => 'AuthorizationJwt',
|
||||||
|
'allow_local_address' => true,
|
||||||
|
],
|
||||||
|
];
|
||||||
@@ -0,0 +1,46 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Atlas Nextcloud
|
||||||
|
Requires=atlas-nextcloud-db.service atlas-nextcloud-redis.service
|
||||||
|
After=atlas-nextcloud-db.service atlas-nextcloud-redis.service
|
||||||
|
RequiresMountsFor={{ atlas_nextcloud_root }}/app {{ atlas_nextcloud_root }}/files{% for mount in atlas_nextcloud_external_mounts %} {{ mount.source }}{% endfor %}
|
||||||
|
|
||||||
|
[Container]
|
||||||
|
ContainerName=atlas-nextcloud
|
||||||
|
Image={{ atlas_nextcloud_image }}
|
||||||
|
Network=atlas-nextcloud.network
|
||||||
|
NetworkAlias=atlas-nextcloud
|
||||||
|
PublishPort={{ ansible_host }}:{{ atlas_nextcloud_http_port }}:80
|
||||||
|
PublishPort=127.0.0.1:{{ atlas_nextcloud_http_port }}:80
|
||||||
|
Environment=POSTGRES_HOST=atlas-nextcloud-db
|
||||||
|
Environment=POSTGRES_DB=nextcloud
|
||||||
|
Environment=POSTGRES_USER=nextcloud
|
||||||
|
Environment=POSTGRES_PASSWORD_FILE=/run/secrets/postgres-password
|
||||||
|
Environment=NEXTCLOUD_ADMIN_USER={{ atlas_nextcloud_admin }}
|
||||||
|
Environment=NEXTCLOUD_ADMIN_PASSWORD_FILE=/run/secrets/admin-password
|
||||||
|
Environment="NEXTCLOUD_TRUSTED_DOMAINS={{ atlas_nextcloud_domain }} atlas-nextcloud"
|
||||||
|
Environment=REDIS_HOST=atlas-nextcloud-redis
|
||||||
|
Environment=REDIS_HOST_PASSWORD_FILE=/run/secrets/redis-password
|
||||||
|
Environment=APACHE_DISABLE_REWRITE_IP=1
|
||||||
|
Environment=PHP_MEMORY_LIMIT=512M
|
||||||
|
Environment=PHP_UPLOAD_LIMIT=2G
|
||||||
|
Volume={{ atlas_nextcloud_root }}/app:/var/www/html:Z
|
||||||
|
Volume={{ atlas_nextcloud_root }}/files:/var/www/html/data:Z
|
||||||
|
Volume={{ atlas_nextcloud_app_cache }}:/mnt/atlas-apps:ro,z
|
||||||
|
{% for mount in atlas_nextcloud_external_mounts %}
|
||||||
|
# Shared Archive label, not the private :Z label used for internal state.
|
||||||
|
Volume={{ mount.source }}:{{ mount.target }}:{{ 'ro' if mount.readonly else 'rw' }},z
|
||||||
|
{% endfor %}
|
||||||
|
Volume={{ atlas_nextcloud_private_dir }}/postgres-password:/run/secrets/postgres-password:ro,z
|
||||||
|
Volume={{ atlas_nextcloud_private_dir }}/admin-password:/run/secrets/admin-password:ro,z
|
||||||
|
Volume={{ atlas_nextcloud_private_dir }}/redis-password:/run/secrets/redis-password:ro,z
|
||||||
|
Volume={{ atlas_nextcloud_private_dir }}/onlyoffice-jwt:/run/secrets/onlyoffice-jwt:ro,z
|
||||||
|
PodmanArgs=--memory=2g
|
||||||
|
NoNewPrivileges=true
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
Restart=on-failure
|
||||||
|
RestartSec=10
|
||||||
|
TimeoutStartSec=900
|
||||||
|
|
||||||
|
[Install]
|
||||||
|
WantedBy=default.target
|
||||||
@@ -0,0 +1,5 @@
|
|||||||
|
# Managed by Ansible: private rootless application network, no host services.
|
||||||
|
[Network]
|
||||||
|
NetworkName=atlas-nextcloud
|
||||||
|
Subnet={{ atlas_nextcloud_network_subnet }}
|
||||||
|
Gateway={{ atlas_nextcloud_network_gateway }}
|
||||||
@@ -0,0 +1,26 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Atlas ONLYOFFICE Docs Community
|
||||||
|
RequiresMountsFor={{ atlas_nextcloud_root }}/office
|
||||||
|
|
||||||
|
[Container]
|
||||||
|
ContainerName=atlas-onlyoffice
|
||||||
|
Image={{ atlas_onlyoffice_image }}
|
||||||
|
Network=atlas-nextcloud.network
|
||||||
|
NetworkAlias=atlas-onlyoffice
|
||||||
|
PublishPort={{ ansible_host }}:{{ atlas_onlyoffice_http_port }}:80
|
||||||
|
PublishPort=127.0.0.1:{{ atlas_onlyoffice_http_port }}:80
|
||||||
|
EnvironmentFile={{ atlas_nextcloud_private_dir }}/onlyoffice.env
|
||||||
|
Volume={{ atlas_nextcloud_root }}/office/data:/var/www/onlyoffice/Data:Z
|
||||||
|
Volume={{ atlas_nextcloud_root }}/office/lib:/var/lib/onlyoffice:Z
|
||||||
|
Volume={{ atlas_nextcloud_root }}/office/logs:/var/log/onlyoffice:Z
|
||||||
|
Volume={{ atlas_nextcloud_root }}/office/database:/var/lib/postgresql:Z
|
||||||
|
PodmanArgs=--memory=4g --shm-size=256m
|
||||||
|
NoNewPrivileges=true
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
Restart=on-failure
|
||||||
|
RestartSec=10
|
||||||
|
TimeoutStartSec=1200
|
||||||
|
|
||||||
|
[Install]
|
||||||
|
WantedBy=default.target
|
||||||
@@ -0,0 +1,7 @@
|
|||||||
|
JWT_ENABLED=true
|
||||||
|
JWT_SECRET={{ vault_nextcloud_onlyoffice_jwt }}
|
||||||
|
JWT_HEADER=AuthorizationJwt
|
||||||
|
ALLOW_PRIVATE_IP_ADDRESS=true
|
||||||
|
ALLOW_META_IP_ADDRESS=false
|
||||||
|
USE_UNAUTHORIZED_STORAGE=false
|
||||||
|
WOPI_ENABLED=false
|
||||||
@@ -0,0 +1,19 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Pull a prepared read-only Prometheus backup to Atlas
|
||||||
|
RequiresMountsFor={{ atlas_backup_prometheus_mountpoint }}
|
||||||
|
Wants=network-online.target
|
||||||
|
After=network-online.target zfs.target
|
||||||
|
ConditionFileIsExecutable=/usr/local/sbin/atlas-prometheus-pull
|
||||||
|
ConditionPathExists={{ atlas_prometheus_pull_private_key_path }}
|
||||||
|
ConditionPathExists={{ atlas_prometheus_pull_known_hosts_path }}
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
Type=oneshot
|
||||||
|
ExecStart=/usr/local/sbin/atlas-prometheus-pull
|
||||||
|
User=root
|
||||||
|
Group=root
|
||||||
|
UMask=0077
|
||||||
|
TimeoutStartSec=infinity
|
||||||
|
Nice=15
|
||||||
|
IOSchedulingClass=best-effort
|
||||||
|
IOSchedulingPriority=7
|
||||||
@@ -0,0 +1,75 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -Eeuo pipefail
|
||||||
|
umask 077
|
||||||
|
|
||||||
|
backup_root={{ atlas_backup_prometheus_mountpoint | quote }}
|
||||||
|
snapshots="$backup_root/snapshots"
|
||||||
|
stage=''
|
||||||
|
exec 9>/run/lock/atlas-prometheus-pull.lock
|
||||||
|
flock -n 9 || { echo 'A Prometheus pull is already running' >&2; exit 1; }
|
||||||
|
|
||||||
|
cleanup() {
|
||||||
|
local rc=$?
|
||||||
|
trap - EXIT
|
||||||
|
if (( rc != 0 )) && [[ -n "$stage" && -d "$stage" ]]; then
|
||||||
|
rm -rf -- "$stage"
|
||||||
|
fi
|
||||||
|
exit "$rc"
|
||||||
|
}
|
||||||
|
trap cleanup EXIT
|
||||||
|
|
||||||
|
zpool list -H -o name {{ atlas_zfs_pool | quote }} >/dev/null
|
||||||
|
findmnt -rn --mountpoint "$backup_root" >/dev/null
|
||||||
|
stage=$(mktemp -d "$backup_root/.staging.XXXXXXXX")
|
||||||
|
ssh_cmd='/usr/bin/ssh -F /dev/null -o BatchMode=yes -o StrictHostKeyChecking=yes -o UserKnownHostsFile={{ atlas_prometheus_pull_known_hosts_path }} -o IdentitiesOnly=yes -i {{ atlas_prometheus_pull_private_key_path }} -p {{ atlas_prometheus_pull_source_port }}'
|
||||||
|
rsync -a --partial --delay-updates -e "$ssh_cmd" \
|
||||||
|
{{ (atlas_prometheus_pull_source_user ~ '@' ~ hostvars['prometheus'].ansible_host ~ ':current/') | quote }} \
|
||||||
|
"$stage/"
|
||||||
|
|
||||||
|
test -s "$stage/payload.tar"
|
||||||
|
test -s "$stage/payload.sha256"
|
||||||
|
test -s "$stage/metadata.json"
|
||||||
|
(cd "$stage" && sha256sum -c payload.sha256)
|
||||||
|
tar -tf "$stage/payload.tar" >/dev/null
|
||||||
|
stamp=$(python3 - "$stage/metadata.json" <<'PY'
|
||||||
|
import json
|
||||||
|
import datetime as dt
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
|
||||||
|
with open(sys.argv[1], encoding="utf-8") as stream:
|
||||||
|
metadata = json.load(stream)
|
||||||
|
stamp = metadata.get("created_utc", "")
|
||||||
|
if metadata.get("schema") != 1 or metadata.get("host") != "prometheus":
|
||||||
|
raise SystemExit("Unexpected Prometheus backup metadata")
|
||||||
|
if not re.fullmatch(r"[0-9]{8}T[0-9]{6}Z", stamp):
|
||||||
|
raise SystemExit("Invalid Prometheus backup timestamp")
|
||||||
|
created = dt.datetime.strptime(stamp, "%Y%m%dT%H%M%SZ").replace(tzinfo=dt.timezone.utc)
|
||||||
|
age = dt.datetime.now(dt.timezone.utc) - created
|
||||||
|
if age.total_seconds() < -300 or age > dt.timedelta(hours={{ atlas_prometheus_pull_max_age_hours }}):
|
||||||
|
raise SystemExit("Prometheus backup is outside the configured freshness window")
|
||||||
|
print(stamp)
|
||||||
|
PY
|
||||||
|
)
|
||||||
|
if [[ -e "$snapshots/$stamp" ]]; then
|
||||||
|
cmp "$stage/payload.sha256" "$snapshots/$stamp/payload.sha256"
|
||||||
|
cmp "$stage/metadata.json" "$snapshots/$stamp/metadata.json"
|
||||||
|
(cd "$snapshots/$stamp" && sha256sum -c payload.sha256)
|
||||||
|
rm -rf -- "${stage:?}"
|
||||||
|
stage=''
|
||||||
|
else
|
||||||
|
chown -R root:root "$stage"
|
||||||
|
chmod 0700 "$stage"
|
||||||
|
chmod 0600 "$stage/payload.tar" "$stage/payload.sha256" "$stage/metadata.json"
|
||||||
|
mv -- "$stage" "$snapshots/$stamp"
|
||||||
|
stage=''
|
||||||
|
fi
|
||||||
|
latest_link=$(readlink "$backup_root/latest" 2>/dev/null || true)
|
||||||
|
latest_stamp=${latest_link##*/}
|
||||||
|
if [[ -z "$latest_stamp" || "$stamp" > "$latest_stamp" ]]; then
|
||||||
|
ln -s "snapshots/$stamp" "$backup_root/.latest.new"
|
||||||
|
mv -Tf -- "$backup_root/.latest.new" "$backup_root/latest"
|
||||||
|
fi
|
||||||
|
python3 /usr/local/libexec/atlas-prometheus-prune "$snapshots" \
|
||||||
|
{{ atlas_prometheus_pull_keep_daily }} {{ atlas_prometheus_pull_keep_weekly }} {{ atlas_prometheus_pull_keep_monthly }}
|
||||||
|
echo "Verified and published Prometheus backup $stamp"
|
||||||
@@ -0,0 +1,10 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Schedule Atlas pull of prepared Prometheus backups
|
||||||
|
|
||||||
|
[Timer]
|
||||||
|
OnCalendar={{ atlas_prometheus_pull_calendar }}
|
||||||
|
Persistent=true
|
||||||
|
Unit=atlas-prometheus-pull.service
|
||||||
|
|
||||||
|
[Install]
|
||||||
|
WantedBy=timers.target
|
||||||
@@ -0,0 +1,30 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Run a manual, UUID-bound offline USB backup of Atlas ZFS datasets
|
||||||
|
Requires=zfs.target
|
||||||
|
After=zfs.target
|
||||||
|
ConditionFileIsExecutable=/usr/local/sbin/atlas-usb-backup
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
Type=oneshot
|
||||||
|
ExecStart=/usr/local/sbin/atlas-usb-backup
|
||||||
|
ExecStopPost=+/usr/local/sbin/atlas-usb-snapshot-cleanup
|
||||||
|
User=root
|
||||||
|
Group=root
|
||||||
|
UMask=0077
|
||||||
|
TimeoutStartSec=infinity
|
||||||
|
RuntimeDirectory=atlas-usb-backup
|
||||||
|
RuntimeDirectoryMode=0700
|
||||||
|
Nice=15
|
||||||
|
IOSchedulingClass=best-effort
|
||||||
|
IOSchedulingPriority=7
|
||||||
|
PrivateMounts=true
|
||||||
|
PrivateTmp=true
|
||||||
|
ProtectHome=true
|
||||||
|
ProtectSystem=strict
|
||||||
|
ReadWritePaths=/run/atlas-usb-backup /run/lock
|
||||||
|
ProtectKernelTunables=true
|
||||||
|
ProtectKernelModules=true
|
||||||
|
ProtectControlGroups=true
|
||||||
|
RestrictAddressFamilies=AF_UNIX
|
||||||
|
RestrictRealtime=true
|
||||||
|
LockPersonality=true
|
||||||
242
ansible/roles/profile_atlas/templates/atlas-usb-backup.sh.j2
Normal file
242
ansible/roles/profile_atlas/templates/atlas-usb-backup.sh.j2
Normal file
@@ -0,0 +1,242 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -Eeuo pipefail
|
||||||
|
|
||||||
|
export LC_ALL=C.utf8
|
||||||
|
export PATH=/usr/sbin:/usr/bin:/sbin:/bin
|
||||||
|
|
||||||
|
readonly pool={{ atlas_zfs_pool | quote }}
|
||||||
|
readonly mount_root={{ atlas_mount_root | quote }}
|
||||||
|
readonly luks_uuid={{ atlas_usb_backup_luks_uuid | quote }}
|
||||||
|
readonly fs_uuid={{ atlas_usb_backup_fs_uuid | quote }}
|
||||||
|
readonly mapper_name={{ atlas_usb_backup_mapper_name | quote }}
|
||||||
|
readonly snapshot_prefix={{ atlas_usb_backup_snapshot_prefix | quote }}
|
||||||
|
readonly min_free_bytes={{ atlas_usb_backup_min_free_bytes | int }}
|
||||||
|
readonly mapper="/dev/mapper/${mapper_name}"
|
||||||
|
readonly outer="/dev/disk/by-uuid/${luks_uuid}"
|
||||||
|
readonly runtime_dir=/run/atlas-usb-backup
|
||||||
|
readonly snapshot_marker="${runtime_dir}/snapshot-name"
|
||||||
|
readonly source_dir="${runtime_dir}/source"
|
||||||
|
readonly usb_mount="${runtime_dir}/target"
|
||||||
|
readonly backup_root="${usb_mount}/atlas"
|
||||||
|
|
||||||
|
snapshot_name=""
|
||||||
|
mapper_opened_by_script=false
|
||||||
|
usb_mounted=false
|
||||||
|
published=false
|
||||||
|
partial=""
|
||||||
|
mounted_targets=()
|
||||||
|
|
||||||
|
# shellcheck disable=SC2329
|
||||||
|
cleanup() {
|
||||||
|
local status=$?
|
||||||
|
local cleanup_status=0
|
||||||
|
local index
|
||||||
|
local source_mount_failed=false
|
||||||
|
trap - EXIT HUP INT TERM
|
||||||
|
set +e
|
||||||
|
|
||||||
|
if [[ -n "$partial" && "$published" == false && "$usb_mounted" == true ]]; then
|
||||||
|
rm -rf -- "$partial" || cleanup_status=2
|
||||||
|
fi
|
||||||
|
if [[ "$usb_mounted" == true ]]; then
|
||||||
|
umount "$usb_mount" || cleanup_status=2
|
||||||
|
fi
|
||||||
|
|
||||||
|
{% raw %}
|
||||||
|
for ((index = ${#mounted_targets[@]} - 1; index >= 0; index--)); do
|
||||||
|
{% endraw %}
|
||||||
|
if mountpoint -q "${mounted_targets[$index]}" && ! umount -R "${mounted_targets[$index]}"; then
|
||||||
|
printf 'Source snapshot mount cleanup failed: %s\n' "${mounted_targets[$index]}" >&2
|
||||||
|
source_mount_failed=true
|
||||||
|
fi
|
||||||
|
if mountpoint -q "${mounted_targets[$index]}"; then
|
||||||
|
printf 'Source snapshot mount is still active: %s\n' "${mounted_targets[$index]}" >&2
|
||||||
|
source_mount_failed=true
|
||||||
|
else
|
||||||
|
rmdir -- "${mounted_targets[$index]}" 2>/dev/null || true
|
||||||
|
fi
|
||||||
|
done
|
||||||
|
if [[ "$source_mount_failed" == false ]]; then
|
||||||
|
rmdir -- "$source_dir" 2>/dev/null || true
|
||||||
|
else
|
||||||
|
cleanup_status=2
|
||||||
|
printf 'Source bind mount cleanup failed; keeping the snapshot for recovery\n' >&2
|
||||||
|
fi
|
||||||
|
|
||||||
|
if [[ "$usb_mounted" == true || "$mapper_opened_by_script" == true ]] &&
|
||||||
|
! mountpoint -q "$usb_mount" &&
|
||||||
|
! findmnt -rn -S "$mapper" >/dev/null; then
|
||||||
|
cryptsetup close "$mapper_name" || cleanup_status=2
|
||||||
|
fi
|
||||||
|
rmdir -- "$usb_mount" 2>/dev/null || true
|
||||||
|
|
||||||
|
if ((status == 0 && cleanup_status != 0)); then
|
||||||
|
status=$cleanup_status
|
||||||
|
fi
|
||||||
|
exit "$status"
|
||||||
|
}
|
||||||
|
|
||||||
|
trap cleanup EXIT
|
||||||
|
trap 'exit 143' HUP INT TERM
|
||||||
|
|
||||||
|
exec 8>/run/lock/atlas-usb-backup.lock
|
||||||
|
flock -n 8 || { printf 'Atlas USB backup is already running\n' >&2; exit 75; }
|
||||||
|
exec 9>/run/lock/atlas-zfs-snapshot.lock
|
||||||
|
|
||||||
|
zpool list -H -o name "$pool" >/dev/null
|
||||||
|
[[ -b "$outer" ]] || { printf 'Configured LUKS UUID is not connected\n' >&2; exit 66; }
|
||||||
|
[[ "$(blkid -s TYPE -o value "$outer")" == crypto_LUKS ]] || {
|
||||||
|
printf 'Configured outer UUID is not a LUKS container\n' >&2
|
||||||
|
exit 65
|
||||||
|
}
|
||||||
|
[[ "$(blkid -s UUID -o value "$outer")" == "$luks_uuid" ]] || exit 65
|
||||||
|
if ! cryptsetup status "$mapper_name" >/dev/null; then
|
||||||
|
printf 'Requesting the LUKS passphrase for the configured USB disk\n'
|
||||||
|
systemd-ask-password -n --no-tty --timeout=300 \
|
||||||
|
--id="atlas-usb-backup:${luks_uuid}" \
|
||||||
|
'Atlas offline USB backup LUKS passphrase:' |
|
||||||
|
cryptsetup open --type luks2 --key-file - "$outer" "$mapper_name"
|
||||||
|
mapper_opened_by_script=true
|
||||||
|
fi
|
||||||
|
backing_device="$(cryptsetup status "$mapper_name" | awk '$1 == "device:" { print $2 }')"
|
||||||
|
[[ -n "$backing_device" && "$(readlink -f "$backing_device")" == "$(readlink -f "$outer")" ]] || {
|
||||||
|
printf 'The unlocked mapper does not belong to the configured LUKS UUID\n' >&2
|
||||||
|
exit 65
|
||||||
|
}
|
||||||
|
[[ "$(blkid -s TYPE -o value "$mapper")" == ext4 ]] || {
|
||||||
|
printf 'The unlocked USB filesystem is not ext4\n' >&2
|
||||||
|
exit 65
|
||||||
|
}
|
||||||
|
[[ "$(blkid -s UUID -o value "$mapper")" == "$fs_uuid" ]] || {
|
||||||
|
printf 'The unlocked USB filesystem UUID does not match\n' >&2
|
||||||
|
exit 65
|
||||||
|
}
|
||||||
|
if findmnt -rn -S "$mapper" >/dev/null; then
|
||||||
|
printf 'The USB filesystem is already mounted elsewhere\n' >&2
|
||||||
|
exit 65
|
||||||
|
fi
|
||||||
|
[[ ! -e "$source_dir" && ! -e "$usb_mount" ]] || {
|
||||||
|
printf 'USB backup staging directories already exist; inspect them manually\n' >&2
|
||||||
|
exit 65
|
||||||
|
}
|
||||||
|
|
||||||
|
mkdir -m 0700 "$usb_mount"
|
||||||
|
mount -t ext4 -o nodev,nosuid,noexec "$mapper" "$usb_mount"
|
||||||
|
usb_mounted=true
|
||||||
|
[[ "$(readlink -f "$(findmnt -nro SOURCE --target "$usb_mount")")" == "$(readlink -f "$mapper")" ]] || {
|
||||||
|
printf 'Mounted USB source does not match the verified mapper\n' >&2
|
||||||
|
exit 65
|
||||||
|
}
|
||||||
|
|
||||||
|
for path in "$backup_root" "$backup_root/snapshots"; do
|
||||||
|
[[ ! -L "$path" ]] || { printf 'Unsafe symlink in USB backup destination\n' >&2; exit 65; }
|
||||||
|
mkdir -p -- "$path"
|
||||||
|
[[ -d "$path" ]] || exit 65
|
||||||
|
chown root:root -- "$path"
|
||||||
|
chmod 0700 -- "$path"
|
||||||
|
done
|
||||||
|
|
||||||
|
free_bytes="$(df -B1 --output=avail "$usb_mount" | tail -n 1 | tr -d ' ')"
|
||||||
|
if ((free_bytes < min_free_bytes)); then
|
||||||
|
printf 'USB free space (%s bytes) is below the required reserve (%s bytes)\n' \
|
||||||
|
"$free_bytes" "$min_free_bytes" >&2
|
||||||
|
exit 73
|
||||||
|
fi
|
||||||
|
|
||||||
|
mkdir -m 0700 "$source_dir"
|
||||||
|
flock 9
|
||||||
|
timestamp="$(date -u +%Y%m%dT%H%M%SZ)"
|
||||||
|
snapshot_name="${snapshot_prefix}-${timestamp}-$$"
|
||||||
|
printf '%s\n' "$snapshot_name" >"$snapshot_marker"
|
||||||
|
zfs snapshot -r "${pool}@${snapshot_name}"
|
||||||
|
flock -u 9
|
||||||
|
printf 'Created recursive USB source snapshot %s@%s\n' "$pool" "$snapshot_name"
|
||||||
|
|
||||||
|
while IFS=$'\t' read -r dataset dataset_mountpoint mounted; do
|
||||||
|
if [[ "$mounted" != yes ]]; then
|
||||||
|
printf 'Dataset %s is not mounted; refusing an incomplete backup\n' "$dataset" >&2
|
||||||
|
exit 65
|
||||||
|
fi
|
||||||
|
if [[ "$dataset_mountpoint" != "$mount_root" && "$dataset_mountpoint" != "$mount_root/"* ]]; then
|
||||||
|
printf 'Dataset %s has unexpected mountpoint %s\n' "$dataset" "$dataset_mountpoint" >&2
|
||||||
|
exit 65
|
||||||
|
fi
|
||||||
|
dataset_suffix="${dataset#"$pool"}"
|
||||||
|
source_path="${dataset_mountpoint}/.zfs/snapshot/${snapshot_name}"
|
||||||
|
target_path="${source_dir}${dataset_suffix}"
|
||||||
|
mkdir -p "$target_path"
|
||||||
|
mount --bind "$source_path" "$target_path"
|
||||||
|
mounted_targets+=("$target_path")
|
||||||
|
mount -o remount,bind,ro "$target_path"
|
||||||
|
done < <(zfs list -H -o name,mountpoint,mounted -s name -r "$pool")
|
||||||
|
|
||||||
|
previous=""
|
||||||
|
if [[ -e "$backup_root/latest" || -L "$backup_root/latest" ]]; then
|
||||||
|
[[ -L "$backup_root/latest" ]] || { printf 'latest is not a symlink\n' >&2; exit 65; }
|
||||||
|
previous="$(readlink -e "$backup_root/latest")"
|
||||||
|
[[ -n "$previous" && "$previous" == "$backup_root/snapshots/"* && -d "$previous" ]] || {
|
||||||
|
printf 'latest does not point to a complete snapshot on the USB disk\n' >&2
|
||||||
|
exit 65
|
||||||
|
}
|
||||||
|
fi
|
||||||
|
|
||||||
|
backup_name="${timestamp}-$$"
|
||||||
|
candidate_partial="${backup_root}/snapshots/.incomplete-${backup_name}"
|
||||||
|
complete="${backup_root}/snapshots/${backup_name}"
|
||||||
|
[[ ! -e "$candidate_partial" && ! -L "$candidate_partial" && ! -e "$complete" && ! -L "$complete" ]] || exit 65
|
||||||
|
mkdir -m 0700 "$candidate_partial"
|
||||||
|
partial="$candidate_partial"
|
||||||
|
|
||||||
|
printf 'Copying the consistent pool tree to USB backup %s\n' "$backup_name"
|
||||||
|
# Preserve POSIX ACLs, ownership, modes, timestamps, hard links, and sparse
|
||||||
|
# files. Do not preserve generic xattrs: Rocky 9's rsync 3.2.7 fails when
|
||||||
|
# combining xattrs with --link-dest, while SELinux labels were intentionally
|
||||||
|
# excluded because restores must relabel for their destination host.
|
||||||
|
rsync_args=(-aHAS --numeric-ids "--info=progress2,stats2")
|
||||||
|
estimate_args=(-aHAS --numeric-ids --dry-run --stats)
|
||||||
|
if [[ -n "$previous" ]]; then
|
||||||
|
rsync_args+=("--link-dest=$previous")
|
||||||
|
estimate_args+=("--link-dest=$previous")
|
||||||
|
fi
|
||||||
|
|
||||||
|
# The rsync dry run estimates changed file bytes after link-dest deduplication.
|
||||||
|
# Metadata and filesystem allocation still require the separate free-space reserve.
|
||||||
|
estimate="$(rsync "${estimate_args[@]}" "${source_dir}/" "${partial}/")"
|
||||||
|
transfer_bytes="$(printf '%s\n' "$estimate" | awk -F: \
|
||||||
|
'/^Total transferred file size:/ { gsub(/[^0-9]/, "", $2); print $2 }')"
|
||||||
|
[[ "$transfer_bytes" =~ ^[0-9]+$ ]] || {
|
||||||
|
printf 'Could not determine the USB transfer size\n' >&2
|
||||||
|
exit 74
|
||||||
|
}
|
||||||
|
if ((free_bytes - transfer_bytes < min_free_bytes)); then
|
||||||
|
printf 'Insufficient USB space: %s bytes free, %s estimated transfer, %s reserved\n' \
|
||||||
|
"$free_bytes" "$transfer_bytes" "$min_free_bytes" >&2
|
||||||
|
exit 73
|
||||||
|
fi
|
||||||
|
|
||||||
|
rsync "${rsync_args[@]}" "${source_dir}/" "${partial}/"
|
||||||
|
|
||||||
|
printf 'Verifying USB backup %s with a checksum-based dry run\n' "$backup_name"
|
||||||
|
verification="${runtime_dir}/verification.out"
|
||||||
|
rsync -aHAS --numeric-ids \
|
||||||
|
--checksum --dry-run --delete --itemize-changes \
|
||||||
|
"${source_dir}/" "${partial}/" >"$verification"
|
||||||
|
if [[ -s "$verification" ]]; then
|
||||||
|
printf 'USB verification found mismatches; refusing to publish the backup\n' >&2
|
||||||
|
exit 74
|
||||||
|
fi
|
||||||
|
|
||||||
|
free_bytes="$(df -B1 --output=avail "$usb_mount" | tail -n 1 | tr -d ' ')"
|
||||||
|
if ((free_bytes < min_free_bytes)); then
|
||||||
|
printf 'USB backup completed below the free-space reserve; refusing to publish it\n' >&2
|
||||||
|
exit 73
|
||||||
|
fi
|
||||||
|
|
||||||
|
mv -- "$partial" "$complete"
|
||||||
|
partial=""
|
||||||
|
ln -s "snapshots/${backup_name}" "${backup_root}/.latest-${backup_name}"
|
||||||
|
mv -Tf -- "${backup_root}/.latest-${backup_name}" "${backup_root}/latest"
|
||||||
|
published=true
|
||||||
|
sync -f "$complete"
|
||||||
|
sync -f "$backup_root"
|
||||||
|
printf 'USB backup %s verified and published; unmounting and closing LUKS\n' "$backup_name"
|
||||||
@@ -0,0 +1,31 @@
|
|||||||
|
#!/usr/bin/python3
|
||||||
|
"""Submit a manual-backup reminder through Atlas' existing Houston notifier."""
|
||||||
|
|
||||||
|
import json
|
||||||
|
import subprocess
|
||||||
|
from datetime import datetime, timezone
|
||||||
|
|
||||||
|
|
||||||
|
now = datetime.now(timezone.utc)
|
||||||
|
message = {
|
||||||
|
"timestamp": now.isoformat(timespec="seconds"),
|
||||||
|
"unixtime": int(now.timestamp()),
|
||||||
|
"event": "atlas_usb_backup_reminder",
|
||||||
|
"severity": "warning",
|
||||||
|
"subject": "Promemoria backup USB offline Atlas",
|
||||||
|
"email_message": (
|
||||||
|
"Collega il disco USB di backup ad Atlas ed esegui manualmente il backup offline.\n"
|
||||||
|
"Il promemoria non avvia il backup. Controlla che il disco non sia\n"
|
||||||
|
"montato; poi esegui:\n\n"
|
||||||
|
" sudo systemctl start atlas-usb-backup.service\n\n"
|
||||||
|
"Verifica l'esito con:\n"
|
||||||
|
" sudo journalctl -u atlas-usb-backup.service -n 100 --no-pager\n\n"
|
||||||
|
"Dopo la riuscita, scollega fisicamente il disco."
|
||||||
|
),
|
||||||
|
}
|
||||||
|
|
||||||
|
subprocess.run(
|
||||||
|
[{{ atlas_usb_reminder_notifier | to_json }}, json.dumps(message)],
|
||||||
|
check=True,
|
||||||
|
)
|
||||||
|
print("Atlas USB backup reminder submitted to 45Drives Alerts; email delivery is not verified.", flush=True)
|
||||||
@@ -0,0 +1,22 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=45Drives Alerts reminder to run the manual Atlas offline USB backup
|
||||||
|
Requires=houston-dbus.service
|
||||||
|
After=houston-dbus.service
|
||||||
|
ConditionFileIsExecutable=/usr/local/libexec/atlas-usb-reminder
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
Type=oneshot
|
||||||
|
ExecStart=/usr/local/libexec/atlas-usb-reminder
|
||||||
|
User=root
|
||||||
|
Group=root
|
||||||
|
UMask=0077
|
||||||
|
NoNewPrivileges=true
|
||||||
|
PrivateTmp=true
|
||||||
|
ProtectHome=true
|
||||||
|
ProtectSystem=strict
|
||||||
|
ProtectKernelTunables=true
|
||||||
|
ProtectKernelModules=true
|
||||||
|
ProtectControlGroups=true
|
||||||
|
RestrictAddressFamilies=AF_UNIX
|
||||||
|
RestrictRealtime=true
|
||||||
|
LockPersonality=true
|
||||||
@@ -0,0 +1,10 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Remind the administrator to run the manual Atlas offline USB backup
|
||||||
|
|
||||||
|
[Timer]
|
||||||
|
OnCalendar={{ atlas_usb_reminder_calendar }}
|
||||||
|
Persistent=true
|
||||||
|
Unit=atlas-usb-reminder.service
|
||||||
|
|
||||||
|
[Install]
|
||||||
|
WantedBy=timers.target
|
||||||
@@ -0,0 +1,51 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -Eeuo pipefail
|
||||||
|
|
||||||
|
export PATH=/usr/sbin:/usr/bin:/sbin:/bin
|
||||||
|
|
||||||
|
readonly pool={{ atlas_zfs_pool | quote }}
|
||||||
|
readonly mount_root={{ atlas_mount_root | quote }}
|
||||||
|
readonly snapshot_prefix={{ atlas_usb_backup_snapshot_prefix | quote }}
|
||||||
|
readonly marker=/run/atlas-usb-backup/snapshot-name
|
||||||
|
|
||||||
|
[[ -e "$marker" ]] || exit 0
|
||||||
|
[[ -f "$marker" && ! -L "$marker" ]] || {
|
||||||
|
printf 'Unsafe Atlas USB snapshot marker; leaving snapshots unchanged\n' >&2
|
||||||
|
exit 2
|
||||||
|
}
|
||||||
|
IFS= read -r snapshot_name <"$marker"
|
||||||
|
[[ "$snapshot_name" =~ ^${snapshot_prefix}-[0-9]{8}T[0-9]{6}Z-[0-9]+$ ]] || {
|
||||||
|
printf 'Invalid Atlas USB snapshot marker; leaving snapshots unchanged\n' >&2
|
||||||
|
exit 2
|
||||||
|
}
|
||||||
|
|
||||||
|
exec 9>/run/lock/atlas-zfs-snapshot.lock
|
||||||
|
flock 9
|
||||||
|
if zfs list -H -t snapshot -o name "${pool}@${snapshot_name}" >/dev/null 2>&1; then
|
||||||
|
# ZFS can leave its on-demand .zfs/snapshot mounts in the host namespace
|
||||||
|
# even after the backup's private bind mounts and process have exited.
|
||||||
|
snapshot_mounts=()
|
||||||
|
snapshot_sources=()
|
||||||
|
while IFS=$'\t' read -r dataset dataset_mountpoint; do
|
||||||
|
[[ "$dataset_mountpoint" == "$mount_root" || "$dataset_mountpoint" == "$mount_root/"* ]] || continue
|
||||||
|
snapshot_mounts+=("${dataset_mountpoint}/.zfs/snapshot/${snapshot_name}")
|
||||||
|
snapshot_sources+=("${dataset}@${snapshot_name}")
|
||||||
|
done < <(zfs list -H -o name,mountpoint -s name -r "$pool")
|
||||||
|
|
||||||
|
{% raw %}
|
||||||
|
for ((index = ${#snapshot_mounts[@]} - 1; index >= 0; index--)); do
|
||||||
|
{% endraw %}
|
||||||
|
mounted_source="$(findmnt -rn -M "${snapshot_mounts[$index]}" -o SOURCE || true)"
|
||||||
|
[[ -n "$mounted_source" ]] || continue
|
||||||
|
[[ "$mounted_source" == "${snapshot_sources[$index]}" ]] || {
|
||||||
|
printf 'Unexpected source on Atlas USB snapshot mount: %s\n' \
|
||||||
|
"${snapshot_mounts[$index]}" >&2
|
||||||
|
exit 2
|
||||||
|
}
|
||||||
|
umount "${snapshot_mounts[$index]}"
|
||||||
|
done
|
||||||
|
|
||||||
|
zfs destroy -r "${pool}@${snapshot_name}"
|
||||||
|
printf 'Removed recursive Atlas USB source snapshot %s@%s after backup exit\n' \
|
||||||
|
"$pool" "$snapshot_name"
|
||||||
|
fi
|
||||||
@@ -0,0 +1,6 @@
|
|||||||
|
[Timer]
|
||||||
|
OnCalendar=
|
||||||
|
OnCalendar={{ atlas_zfs_scrub_calendar }}
|
||||||
|
Persistent=true
|
||||||
|
RandomizedDelaySec=0
|
||||||
|
AccuracySec=1min
|
||||||
@@ -0,0 +1,72 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -euo pipefail
|
||||||
|
|
||||||
|
export LC_ALL=C
|
||||||
|
export PATH=/usr/sbin:/usr/bin:/sbin:/bin
|
||||||
|
|
||||||
|
if [[ $# -ne 1 ]]; then
|
||||||
|
printf 'Usage: %s <policy>\n' "$0" >&2
|
||||||
|
exit 64
|
||||||
|
fi
|
||||||
|
|
||||||
|
readonly pool={{ atlas_zfs_pool | quote }}
|
||||||
|
readonly prefix={{ atlas_zfs_snapshot_prefix | quote }}
|
||||||
|
readonly period="$1"
|
||||||
|
|
||||||
|
case "$period" in
|
||||||
|
{% for policy in atlas_zfs_snapshot_policies %}
|
||||||
|
{{ policy.name | quote }})
|
||||||
|
keep={{ policy.keep | int }}
|
||||||
|
;;
|
||||||
|
{% endfor %}
|
||||||
|
*)
|
||||||
|
printf 'Unknown Atlas ZFS snapshot policy: %s\n' "$period" >&2
|
||||||
|
exit 64
|
||||||
|
;;
|
||||||
|
esac
|
||||||
|
readonly keep
|
||||||
|
|
||||||
|
zpool list -H -o name "$pool" >/dev/null
|
||||||
|
|
||||||
|
exec 9>/run/lock/atlas-zfs-snapshot.lock
|
||||||
|
flock 9
|
||||||
|
|
||||||
|
timestamp="$(date -u +%Y%m%dT%H%M%SZ)"
|
||||||
|
readonly timestamp
|
||||||
|
readonly snapshot_prefix="${pool}@${prefix}-${period}-"
|
||||||
|
readonly snapshot="${snapshot_prefix}${timestamp}"
|
||||||
|
|
||||||
|
zfs snapshot -r "$snapshot"
|
||||||
|
printf 'Created recursive ZFS snapshot %s\n' "$snapshot"
|
||||||
|
|
||||||
|
snapshot_listing="$(zfs list -H -t snapshot -o name -s creation -r "$pool")"
|
||||||
|
managed_snapshots=()
|
||||||
|
while IFS= read -r snapshot_name; do
|
||||||
|
if [[ "$snapshot_name" == "$snapshot_prefix"* ]]; then
|
||||||
|
snapshot_suffix="${snapshot_name#"$snapshot_prefix"}"
|
||||||
|
if [[ "$snapshot_suffix" =~ ^[0-9]{8}T[0-9]{6}Z$ ]]; then
|
||||||
|
managed_snapshots+=("$snapshot_name")
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
done <<< "$snapshot_listing"
|
||||||
|
|
||||||
|
{% raw %}
|
||||||
|
managed_snapshot_count="${#managed_snapshots[@]}"
|
||||||
|
{% endraw %}
|
||||||
|
prune_count=$((managed_snapshot_count - keep))
|
||||||
|
if ((prune_count <= 0)); then
|
||||||
|
printf 'Retaining %d of %d managed %s snapshots\n' \
|
||||||
|
"$managed_snapshot_count" "$keep" "$period"
|
||||||
|
exit 0
|
||||||
|
fi
|
||||||
|
|
||||||
|
for ((index = 0; index < prune_count; index++)); do
|
||||||
|
candidate="${managed_snapshots[$index]}"
|
||||||
|
if [[ "$candidate" != "$snapshot_prefix"* ]]; then
|
||||||
|
printf 'Refusing to destroy unexpected snapshot: %s\n' "$candidate" >&2
|
||||||
|
exit 65
|
||||||
|
fi
|
||||||
|
|
||||||
|
zfs destroy -r "$candidate"
|
||||||
|
printf 'Pruned recursive ZFS snapshot %s\n' "$candidate"
|
||||||
|
done
|
||||||
@@ -0,0 +1,11 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Schedule {{ item.name }} ZFS snapshots for {{ atlas_zfs_pool }}
|
||||||
|
|
||||||
|
[Timer]
|
||||||
|
OnCalendar={{ item.calendar }}
|
||||||
|
Persistent=true
|
||||||
|
AccuracySec=1min
|
||||||
|
Unit=atlas-zfs-snapshot@{{ item.name }}.service
|
||||||
|
|
||||||
|
[Install]
|
||||||
|
WantedBy=timers.target
|
||||||
@@ -0,0 +1,25 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Create and retain %i ZFS snapshots for {{ atlas_zfs_pool }}
|
||||||
|
Documentation=man:zfs-snapshot(8) man:zfs-destroy(8)
|
||||||
|
Requires=zfs.target
|
||||||
|
After=zfs.target
|
||||||
|
ConditionFileIsExecutable=/usr/local/sbin/atlas-zfs-snapshot
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
Type=oneshot
|
||||||
|
ExecStart=/usr/local/sbin/atlas-zfs-snapshot %i
|
||||||
|
User=root
|
||||||
|
Group=root
|
||||||
|
UMask=0077
|
||||||
|
Nice=10
|
||||||
|
IOSchedulingClass=best-effort
|
||||||
|
IOSchedulingPriority=7
|
||||||
|
NoNewPrivileges=true
|
||||||
|
PrivateTmp=true
|
||||||
|
ProtectHome=true
|
||||||
|
ProtectSystem=strict
|
||||||
|
ProtectKernelTunables=true
|
||||||
|
ProtectKernelModules=true
|
||||||
|
ProtectControlGroups=true
|
||||||
|
RestrictRealtime=true
|
||||||
|
LockPersonality=true
|
||||||
@@ -30,3 +30,7 @@ backend_phase1_timezone: Europe/Rome
|
|||||||
backend_phase1_services:
|
backend_phase1_services:
|
||||||
- atlas-navidrome.service
|
- atlas-navidrome.service
|
||||||
- atlas-syncthing.service
|
- atlas-syncthing.service
|
||||||
|
backend_phase1_music_sync_enabled: false
|
||||||
|
backend_phase1_music_source_dir: "{{ backend_phase1_archive_dir }}/Music"
|
||||||
|
backend_phase1_music_sync_calendar: "*-*-* 00:45:00 Europe/Rome"
|
||||||
|
backend_phase1_user_systemd_dir: "{{ backend_phase1_user_home }}/.config/systemd/user"
|
||||||
|
|||||||
@@ -18,21 +18,28 @@
|
|||||||
- backend_phase1_app_data_root.startswith('/')
|
- backend_phase1_app_data_root.startswith('/')
|
||||||
- backend_phase1_navidrome_data_dir.startswith(backend_phase1_app_data_root + '/')
|
- backend_phase1_navidrome_data_dir.startswith(backend_phase1_app_data_root + '/')
|
||||||
- backend_phase1_syncthing_root.startswith(backend_phase1_app_data_root + '/')
|
- backend_phase1_syncthing_root.startswith(backend_phase1_app_data_root + '/')
|
||||||
|
- >-
|
||||||
|
not (backend_phase1_music_sync_enabled | bool) or
|
||||||
|
(backend_phase1_music_source_dir.startswith(backend_phase1_archive_dir + '/')
|
||||||
|
and backend_phase1_music_sync_calendar | length > 0)
|
||||||
fail_msg: >-
|
fail_msg: >-
|
||||||
Disable the rootful media-stack gate and provide the Atlas LAN bind
|
Disable the rootful media-stack gate and provide the Atlas LAN bind
|
||||||
address, firewall sources, and absolute ZFS-backed paths before
|
address, firewall sources, and absolute ZFS-backed paths before
|
||||||
enabling phase one. This role does not manage Prometheus or migrate
|
enabling phase one. This role does not manage Prometheus or migrate
|
||||||
application data.
|
application data.
|
||||||
|
tags: [music_sync]
|
||||||
|
|
||||||
- name: Read the rootless service account
|
- name: Read the rootless service account
|
||||||
ansible.builtin.getent:
|
ansible.builtin.getent:
|
||||||
database: passwd
|
database: passwd
|
||||||
key: "{{ backend_phase1_username }}"
|
key: "{{ backend_phase1_username }}"
|
||||||
|
tags: [music_sync]
|
||||||
|
|
||||||
- name: Record rootless service account IDs
|
- name: Record rootless service account IDs
|
||||||
ansible.builtin.set_fact:
|
ansible.builtin.set_fact:
|
||||||
backend_phase1_uid: "{{ ansible_facts['getent_passwd'][backend_phase1_username][1] }}"
|
backend_phase1_uid: "{{ ansible_facts['getent_passwd'][backend_phase1_username][1] }}"
|
||||||
backend_phase1_gid: "{{ ansible_facts['getent_passwd'][backend_phase1_username][2] }}"
|
backend_phase1_gid: "{{ ansible_facts['getent_passwd'][backend_phase1_username][2] }}"
|
||||||
|
tags: [music_sync]
|
||||||
|
|
||||||
- name: Read system service state before starting rootless Syncthing
|
- name: Read system service state before starting rootless Syncthing
|
||||||
ansible.builtin.service_facts:
|
ansible.builtin.service_facts:
|
||||||
@@ -65,6 +72,7 @@
|
|||||||
loop_control:
|
loop_control:
|
||||||
label: "{{ item.dataset }}"
|
label: "{{ item.dataset }}"
|
||||||
register: backend_phase1_zfs_facts
|
register: backend_phase1_zfs_facts
|
||||||
|
tags: [music_sync]
|
||||||
|
|
||||||
- name: Require mounted datasets at the declared paths
|
- name: Require mounted datasets at the declared paths
|
||||||
ansible.builtin.assert:
|
ansible.builtin.assert:
|
||||||
@@ -79,6 +87,23 @@
|
|||||||
loop: "{{ backend_phase1_zfs_facts.results }}"
|
loop: "{{ backend_phase1_zfs_facts.results }}"
|
||||||
loop_control:
|
loop_control:
|
||||||
label: "{{ item.item.dataset }}"
|
label: "{{ item.item.dataset }}"
|
||||||
|
tags: [music_sync]
|
||||||
|
|
||||||
|
- name: Inspect the music copy source
|
||||||
|
ansible.builtin.stat:
|
||||||
|
path: "{{ backend_phase1_music_source_dir }}"
|
||||||
|
register: backend_phase1_music_source_stat
|
||||||
|
when: backend_phase1_music_sync_enabled | bool
|
||||||
|
tags: [music_sync]
|
||||||
|
|
||||||
|
- name: Require an existing music source directory
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- backend_phase1_music_source_stat.stat.isdir | default(false)
|
||||||
|
fail_msg: >-
|
||||||
|
{{ backend_phase1_music_source_dir }} must exist before enabling the daily music copy.
|
||||||
|
when: backend_phase1_music_sync_enabled | bool
|
||||||
|
tags: [music_sync]
|
||||||
|
|
||||||
- name: Enable lingering for the rootless service account
|
- name: Enable lingering for the rootless service account
|
||||||
ansible.builtin.command:
|
ansible.builtin.command:
|
||||||
@@ -87,12 +112,14 @@
|
|||||||
- enable-linger
|
- enable-linger
|
||||||
- "{{ backend_phase1_username }}"
|
- "{{ backend_phase1_username }}"
|
||||||
creates: "/var/lib/systemd/linger/{{ backend_phase1_username }}"
|
creates: "/var/lib/systemd/linger/{{ backend_phase1_username }}"
|
||||||
|
tags: [music_sync]
|
||||||
|
|
||||||
- name: Start the rootless user systemd manager
|
- name: Start the rootless user systemd manager
|
||||||
ansible.builtin.systemd:
|
ansible.builtin.systemd:
|
||||||
name: "user@{{ backend_phase1_uid }}.service"
|
name: "user@{{ backend_phase1_uid }}.service"
|
||||||
state: started
|
state: started
|
||||||
when: not ansible_check_mode
|
when: not ansible_check_mode
|
||||||
|
tags: [music_sync]
|
||||||
|
|
||||||
- name: Create rootless Quadlet and application directories
|
- name: Create rootless Quadlet and application directories
|
||||||
ansible.builtin.file:
|
ansible.builtin.file:
|
||||||
@@ -113,6 +140,43 @@
|
|||||||
loop_control:
|
loop_control:
|
||||||
label: "{{ item.path }}"
|
label: "{{ item.path }}"
|
||||||
|
|
||||||
|
- name: Install rsync for the daily music copy
|
||||||
|
ansible.builtin.dnf:
|
||||||
|
name: rsync
|
||||||
|
state: present
|
||||||
|
when: backend_phase1_music_sync_enabled | bool
|
||||||
|
tags: [music_sync]
|
||||||
|
|
||||||
|
- name: Create the rootless user systemd directory
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: "{{ backend_phase1_user_systemd_dir }}"
|
||||||
|
state: directory
|
||||||
|
owner: "{{ backend_phase1_username }}"
|
||||||
|
group: "{{ backend_phase1_user_group }}"
|
||||||
|
mode: "0700"
|
||||||
|
when: backend_phase1_music_sync_enabled | bool
|
||||||
|
tags: [music_sync]
|
||||||
|
|
||||||
|
- name: Install the daily music copy service
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: atlas-music-sync.service.j2
|
||||||
|
dest: "{{ backend_phase1_user_systemd_dir }}/atlas-music-sync.service"
|
||||||
|
owner: "{{ backend_phase1_username }}"
|
||||||
|
group: "{{ backend_phase1_user_group }}"
|
||||||
|
mode: "0644"
|
||||||
|
when: backend_phase1_music_sync_enabled | bool
|
||||||
|
tags: [music_sync]
|
||||||
|
|
||||||
|
- name: Install the daily music copy timer
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: atlas-music-sync.timer.j2
|
||||||
|
dest: "{{ backend_phase1_user_systemd_dir }}/atlas-music-sync.timer"
|
||||||
|
owner: "{{ backend_phase1_username }}"
|
||||||
|
group: "{{ backend_phase1_user_group }}"
|
||||||
|
mode: "0644"
|
||||||
|
when: backend_phase1_music_sync_enabled | bool
|
||||||
|
tags: [music_sync]
|
||||||
|
|
||||||
- name: Render the rootless Navidrome Quadlet
|
- name: Render the rootless Navidrome Quadlet
|
||||||
ansible.builtin.template:
|
ansible.builtin.template:
|
||||||
src: atlas-navidrome.container.j2
|
src: atlas-navidrome.container.j2
|
||||||
@@ -140,6 +204,7 @@
|
|||||||
XDG_RUNTIME_DIR: "/run/user/{{ backend_phase1_uid }}"
|
XDG_RUNTIME_DIR: "/run/user/{{ backend_phase1_uid }}"
|
||||||
DBUS_SESSION_BUS_ADDRESS: "unix:path=/run/user/{{ backend_phase1_uid }}/bus"
|
DBUS_SESSION_BUS_ADDRESS: "unix:path=/run/user/{{ backend_phase1_uid }}/bus"
|
||||||
when: not ansible_check_mode
|
when: not ansible_check_mode
|
||||||
|
tags: [music_sync]
|
||||||
|
|
||||||
- name: Permit NPM access to phase-one web interfaces through Aegis
|
- name: Permit NPM access to phase-one web interfaces through Aegis
|
||||||
ansible.posix.firewalld:
|
ansible.posix.firewalld:
|
||||||
@@ -186,3 +251,19 @@
|
|||||||
when:
|
when:
|
||||||
- backend_phase1_start_services | bool
|
- backend_phase1_start_services | bool
|
||||||
- not ansible_check_mode
|
- not ansible_check_mode
|
||||||
|
|
||||||
|
- name: Enable the daily music copy timer
|
||||||
|
become_user: "{{ backend_phase1_username }}"
|
||||||
|
ansible.builtin.systemd:
|
||||||
|
name: atlas-music-sync.timer
|
||||||
|
scope: user
|
||||||
|
state: started
|
||||||
|
enabled: true
|
||||||
|
daemon_reload: true
|
||||||
|
environment:
|
||||||
|
XDG_RUNTIME_DIR: "/run/user/{{ backend_phase1_uid }}"
|
||||||
|
DBUS_SESSION_BUS_ADDRESS: "unix:path=/run/user/{{ backend_phase1_uid }}/bus"
|
||||||
|
when:
|
||||||
|
- backend_phase1_music_sync_enabled | bool
|
||||||
|
- not ansible_check_mode
|
||||||
|
tags: [music_sync]
|
||||||
|
|||||||
@@ -0,0 +1,10 @@
|
|||||||
|
# Managed by Ansible. Do not edit manually.
|
||||||
|
[Unit]
|
||||||
|
Description=Copy Atlas Archive music to the Navidrome library
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
Type=oneshot
|
||||||
|
ExecStartPre=/usr/bin/mountpoint -q {{ backend_phase1_archive_dir }}
|
||||||
|
ExecStartPre=/usr/bin/mountpoint -q {{ backend_phase1_music_dir }}
|
||||||
|
ExecStartPre=/usr/bin/test -d {{ backend_phase1_music_source_dir }}
|
||||||
|
ExecStart=/usr/bin/rsync -aH --no-perms --no-owner --no-group --delay-updates --stats -- {{ backend_phase1_music_source_dir }}/ {{ backend_phase1_music_dir }}/
|
||||||
@@ -0,0 +1,11 @@
|
|||||||
|
# Managed by Ansible. Do not edit manually.
|
||||||
|
[Unit]
|
||||||
|
Description=Schedule the daily Atlas Navidrome music copy
|
||||||
|
|
||||||
|
[Timer]
|
||||||
|
OnCalendar={{ backend_phase1_music_sync_calendar }}
|
||||||
|
Persistent=true
|
||||||
|
Unit=atlas-music-sync.service
|
||||||
|
|
||||||
|
[Install]
|
||||||
|
WantedBy=timers.target
|
||||||
103
ansible/roles/profile_server/tasks/backup_export_identity.yml
Normal file
103
ansible/roles/profile_server/tasks/backup_export_identity.yml
Normal file
@@ -0,0 +1,103 @@
|
|||||||
|
---
|
||||||
|
- name: Validate Prometheus backup export identity inputs
|
||||||
|
tags: [services, backup, prometheus_backup]
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- inventory_hostname == 'prometheus'
|
||||||
|
- server_backup_username is match('^[a-z_][a-z0-9_-]*$')
|
||||||
|
- server_backup_username not in ['root', server_username]
|
||||||
|
- server_backup_export_root.startswith('/var/lib/')
|
||||||
|
- server_backup_public_key_name is match('^[a-z0-9_-]+$')
|
||||||
|
- hostvars['atlas'].atlas_manage_prometheus_backup_pull | default(false) | bool
|
||||||
|
fail_msg: Enable Atlas and Prometheus backup roles together with dedicated identity settings.
|
||||||
|
when: server_backup_export_enabled | bool
|
||||||
|
|
||||||
|
- name: Create dedicated Prometheus backup export group
|
||||||
|
tags: [services, backup, prometheus_backup]
|
||||||
|
ansible.builtin.group:
|
||||||
|
name: "{{ server_backup_username }}"
|
||||||
|
system: true
|
||||||
|
state: present
|
||||||
|
when: server_backup_export_enabled | bool
|
||||||
|
|
||||||
|
- name: Create locked Prometheus backup export account
|
||||||
|
tags: [services, backup, prometheus_backup]
|
||||||
|
ansible.builtin.user:
|
||||||
|
name: "{{ server_backup_username }}"
|
||||||
|
group: "{{ server_backup_username }}"
|
||||||
|
groups: []
|
||||||
|
append: false
|
||||||
|
comment: Read-only prepared backup export for Atlas
|
||||||
|
home: "{{ server_backup_export_root }}"
|
||||||
|
create_home: false
|
||||||
|
shell: /bin/bash
|
||||||
|
password_lock: true
|
||||||
|
system: true
|
||||||
|
state: present
|
||||||
|
when: server_backup_export_enabled | bool
|
||||||
|
|
||||||
|
- name: Require restricted rrsync helper on Prometheus
|
||||||
|
tags: [services, backup, prometheus_backup]
|
||||||
|
ansible.builtin.stat:
|
||||||
|
path: "{{ server_backup_rrsync_path }}"
|
||||||
|
register: server_backup_rrsync_file
|
||||||
|
when: server_backup_export_enabled | bool
|
||||||
|
|
||||||
|
- name: Validate restricted rrsync helper
|
||||||
|
tags: [services, backup, prometheus_backup]
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- server_backup_rrsync_file.stat.exists
|
||||||
|
- server_backup_rrsync_file.stat.isreg
|
||||||
|
- server_backup_rrsync_file.stat.pw_name == 'root'
|
||||||
|
fail_msg: Rocky rsync must provide the root-owned rrsync support script.
|
||||||
|
when: server_backup_export_enabled | bool
|
||||||
|
|
||||||
|
- name: Create prepared backup export root
|
||||||
|
tags: [services, backup, prometheus_backup]
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: "{{ server_backup_export_root }}"
|
||||||
|
state: directory
|
||||||
|
owner: root
|
||||||
|
group: "{{ server_backup_username }}"
|
||||||
|
mode: "0750"
|
||||||
|
when: server_backup_export_enabled | bool
|
||||||
|
|
||||||
|
- name: Create restricted Prometheus backup SSH directories
|
||||||
|
tags: [services, backup, prometheus_backup]
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: "{{ item }}"
|
||||||
|
state: directory
|
||||||
|
owner: root
|
||||||
|
group: "{{ server_backup_username }}"
|
||||||
|
mode: "0750"
|
||||||
|
loop:
|
||||||
|
- "{{ server_backup_export_root }}/.ssh"
|
||||||
|
- "{{ server_backup_export_root }}/.ssh/authorized_keys.d"
|
||||||
|
when: server_backup_export_enabled | bool
|
||||||
|
|
||||||
|
- name: Read Atlas public key for Prometheus backup pull
|
||||||
|
tags: [services, backup, prometheus_backup]
|
||||||
|
ansible.builtin.slurp:
|
||||||
|
src: "{{ hostvars['atlas'].atlas_prometheus_pull_private_key_path | default('/etc/atlas-prometheus-pull/id_ed25519') }}.pub"
|
||||||
|
delegate_to: atlas
|
||||||
|
become: true
|
||||||
|
register: server_backup_atlas_public_key
|
||||||
|
when:
|
||||||
|
- server_backup_export_enabled | bool
|
||||||
|
- not ansible_check_mode
|
||||||
|
|
||||||
|
- name: Authorize only restricted read-only backup access from Atlas
|
||||||
|
tags: [services, backup, prometheus_backup]
|
||||||
|
ansible.builtin.copy:
|
||||||
|
content: >-
|
||||||
|
{{ 'command="/usr/bin/python3 ' ~ server_backup_rrsync_path ~ ' -ro '
|
||||||
|
~ server_backup_export_root ~ '/versions",restrict '
|
||||||
|
~ (server_backup_atlas_public_key.content | b64decode | trim) ~ '\n' }}
|
||||||
|
dest: "{{ server_backup_export_root }}/.ssh/authorized_keys.d/{{ server_backup_public_key_name }}"
|
||||||
|
owner: root
|
||||||
|
group: "{{ server_backup_username }}"
|
||||||
|
mode: "0640"
|
||||||
|
when:
|
||||||
|
- server_backup_export_enabled | bool
|
||||||
|
- not ansible_check_mode
|
||||||
91
ansible/roles/profile_server/tasks/backup_export_job.yml
Normal file
91
ansible/roles/profile_server/tasks/backup_export_job.yml
Normal file
@@ -0,0 +1,91 @@
|
|||||||
|
---
|
||||||
|
- name: Validate Prometheus backup export job inputs
|
||||||
|
tags: [services, backup, prometheus_backup]
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- server_backup_export_source_keep | int >= 2
|
||||||
|
- server_backup_export_paths | length > 0
|
||||||
|
- server_backup_export_paths | unique | length == server_backup_export_paths | length
|
||||||
|
- >-
|
||||||
|
server_backup_export_paths
|
||||||
|
| select('match', '^[a-zA-Z0-9][a-zA-Z0-9._/-]*$') | list | length
|
||||||
|
== server_backup_export_paths | length
|
||||||
|
- >-
|
||||||
|
server_backup_export_paths
|
||||||
|
| reject('search', '(^|/)\.\.(/|$)') | list | length
|
||||||
|
== server_backup_export_paths | length
|
||||||
|
- >-
|
||||||
|
server_backup_export_excludes
|
||||||
|
| select('match', '^[a-zA-Z0-9][a-zA-Z0-9._/-]*$') | list | length
|
||||||
|
== server_backup_export_excludes | length
|
||||||
|
- >-
|
||||||
|
server_backup_export_excludes
|
||||||
|
| reject('search', '(^|/)\.\.(/|$)') | list | length
|
||||||
|
== server_backup_export_excludes | length
|
||||||
|
fail_msg: Define safe relative paths and at least two prepared export versions.
|
||||||
|
when: server_backup_export_enabled | bool
|
||||||
|
|
||||||
|
- name: Validate Prometheus backup export calendar
|
||||||
|
tags: [services, backup, prometheus_backup]
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: [systemd-analyze, calendar, "{{ server_backup_export_calendar }}"]
|
||||||
|
changed_when: false
|
||||||
|
check_mode: false
|
||||||
|
when: server_backup_export_enabled | bool
|
||||||
|
|
||||||
|
- name: Ensure prepared Prometheus backup versions directory exists
|
||||||
|
tags: [services, backup, prometheus_backup]
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: "{{ server_backup_export_root }}/versions"
|
||||||
|
state: directory
|
||||||
|
owner: root
|
||||||
|
group: "{{ server_backup_username }}"
|
||||||
|
mode: "0750"
|
||||||
|
when: server_backup_export_enabled | bool
|
||||||
|
|
||||||
|
- name: Install Prometheus backup export helper
|
||||||
|
tags: [services, backup, prometheus_backup, gitea_cutover, npm_quadlet_backup]
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: prometheus-backup-export.sh.j2
|
||||||
|
dest: /usr/local/sbin/prometheus-backup-export
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0750"
|
||||||
|
validate: "bash -n %s"
|
||||||
|
when: server_backup_export_enabled | bool
|
||||||
|
|
||||||
|
- name: Install Prometheus backup export systemd units
|
||||||
|
tags: [services, backup, prometheus_backup]
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: "{{ item }}.j2"
|
||||||
|
dest: "/etc/systemd/system/{{ item }}"
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0644"
|
||||||
|
loop:
|
||||||
|
- prometheus-backup-export.service
|
||||||
|
- prometheus-backup-export.timer
|
||||||
|
loop_control:
|
||||||
|
label: "{{ item }}"
|
||||||
|
register: server_backup_export_units
|
||||||
|
when: server_backup_export_enabled | bool
|
||||||
|
|
||||||
|
- name: Reload systemd after Prometheus backup export unit changes
|
||||||
|
tags: [services, backup, prometheus_backup]
|
||||||
|
ansible.builtin.systemd:
|
||||||
|
daemon_reload: true
|
||||||
|
when:
|
||||||
|
- server_backup_export_enabled | bool
|
||||||
|
- server_backup_export_units is changed
|
||||||
|
- not ansible_check_mode
|
||||||
|
|
||||||
|
- name: Enable Prometheus backup export timer only after explicit activation
|
||||||
|
tags: [services, backup, prometheus_backup]
|
||||||
|
ansible.builtin.systemd:
|
||||||
|
name: prometheus-backup-export.timer
|
||||||
|
enabled: true
|
||||||
|
state: started
|
||||||
|
when:
|
||||||
|
- server_backup_export_enabled | bool
|
||||||
|
- server_backup_export_start_timer | bool
|
||||||
|
- not ansible_check_mode
|
||||||
@@ -1,33 +0,0 @@
|
|||||||
---
|
|
||||||
- name: Require DuckDNS domain and Vault token before deployment
|
|
||||||
ansible.builtin.assert:
|
|
||||||
that:
|
|
||||||
- >-
|
|
||||||
server_duckdns_domain | default('') is
|
|
||||||
regex('[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?', match_type='fullmatch')
|
|
||||||
- >-
|
|
||||||
vault_duckdns_token | default('') is
|
|
||||||
regex('[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}', match_type='fullmatch')
|
|
||||||
fail_msg: >-
|
|
||||||
Define server_duckdns_domain in host_vars and the rotated vault_duckdns_token
|
|
||||||
in encrypted Vault or an untracked local vars file before deploying DuckDNS.
|
|
||||||
no_log: true
|
|
||||||
|
|
||||||
- name: Ensure private DuckDNS directory exists
|
|
||||||
ansible.builtin.file:
|
|
||||||
path: "{{ server_user_home }}/duckdns"
|
|
||||||
state: directory
|
|
||||||
owner: "{{ server_username }}"
|
|
||||||
group: "{{ server_user_group }}"
|
|
||||||
mode: "0700"
|
|
||||||
|
|
||||||
- name: Render DuckDNS updater with the Vault token
|
|
||||||
ansible.builtin.template:
|
|
||||||
src: duck.sh.j2
|
|
||||||
dest: "{{ server_user_home }}/duckdns/duck.sh"
|
|
||||||
owner: "{{ server_username }}"
|
|
||||||
group: "{{ server_user_group }}"
|
|
||||||
mode: "0700"
|
|
||||||
validate: /bin/sh -n %s
|
|
||||||
no_log: true
|
|
||||||
diff: false
|
|
||||||
59
ansible/roles/profile_server/tasks/gitea_npm_proxy.yml
Normal file
59
ansible/roles/profile_server/tasks/gitea_npm_proxy.yml
Normal file
@@ -0,0 +1,59 @@
|
|||||||
|
---
|
||||||
|
- name: Validate the NPM Gitea cutover override
|
||||||
|
tags: [services, gitea_cutover]
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- server_gitea_proxy_enabled | bool
|
||||||
|
- server_gitea_npm_domains | length > 0
|
||||||
|
- server_gitea_npm_domains | select('match', '^[a-zA-Z0-9.-]+$') | list | length == server_gitea_npm_domains | length
|
||||||
|
fail_msg: Declare the exact NPM Gitea hostnames before enabling the Atlas upstream.
|
||||||
|
when: server_gitea_on_atlas | bool
|
||||||
|
|
||||||
|
- name: Ensure the NPM custom configuration directory exists
|
||||||
|
tags: [services, gitea_cutover]
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: /opt/npm/data/nginx/custom
|
||||||
|
state: directory
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0755"
|
||||||
|
when: server_gitea_proxy_enabled | bool
|
||||||
|
|
||||||
|
- name: Render the Gitea-only NPM runtime upstream override
|
||||||
|
tags: [services, gitea_cutover]
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: prometheus-gitea-npm-proxy.conf.j2
|
||||||
|
dest: /opt/npm/data/nginx/custom/server_proxy.conf
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0644"
|
||||||
|
register: server_gitea_npm_override
|
||||||
|
when: server_gitea_on_atlas | bool
|
||||||
|
|
||||||
|
- name: Remove the Gitea NPM override when source routing is selected
|
||||||
|
tags: [services, gitea_cutover]
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: /opt/npm/data/nginx/custom/server_proxy.conf
|
||||||
|
state: absent
|
||||||
|
when:
|
||||||
|
- server_gitea_proxy_enabled | bool
|
||||||
|
- not server_gitea_on_atlas | bool
|
||||||
|
|
||||||
|
- name: Validate NPM configuration after a Gitea upstream change
|
||||||
|
tags: [services, gitea_cutover]
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: [podman, exec, nginx-proxy-manager, nginx, -t]
|
||||||
|
changed_when: false
|
||||||
|
when:
|
||||||
|
- server_gitea_on_atlas | bool
|
||||||
|
- server_gitea_npm_override is changed
|
||||||
|
- not ansible_check_mode
|
||||||
|
|
||||||
|
- name: Reload NPM after validating the Gitea upstream change
|
||||||
|
tags: [services, gitea_cutover]
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: [podman, exec, nginx-proxy-manager, nginx, -s, reload]
|
||||||
|
when:
|
||||||
|
- server_gitea_on_atlas | bool
|
||||||
|
- server_gitea_npm_override is changed
|
||||||
|
- not ansible_check_mode
|
||||||
61
ansible/roles/profile_server/tasks/gitea_ssh_proxy.yml
Normal file
61
ansible/roles/profile_server/tasks/gitea_ssh_proxy.yml
Normal file
@@ -0,0 +1,61 @@
|
|||||||
|
---
|
||||||
|
- name: Validate the Prometheus Gitea SSH cutover inputs
|
||||||
|
tags: [services, gitea_cutover]
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- server_gitea_proxy_enabled | bool
|
||||||
|
- server_gitea_atlas_address is match('^[0-9]{1,3}(\.[0-9]{1,3}){3}$')
|
||||||
|
- server_gitea_ssh_public_port | int > 1024
|
||||||
|
- server_gitea_ssh_public_port | int < 65536
|
||||||
|
- server_gitea_ssh_target_port | int > 1024
|
||||||
|
- server_gitea_ssh_target_port | int < 65536
|
||||||
|
- server_gitea_ssh_public_port | int != 22
|
||||||
|
fail_msg: Keep administrative SSH on 22 and provide the Atlas rootless Gitea SSH endpoint.
|
||||||
|
when: server_gitea_on_atlas | bool
|
||||||
|
|
||||||
|
- name: Install the Gitea SSH socket proxy units without activating them
|
||||||
|
tags: [services, gitea_cutover]
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: "{{ item }}.j2"
|
||||||
|
dest: "/etc/systemd/system/{{ item }}"
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0644"
|
||||||
|
loop:
|
||||||
|
- prometheus-gitea-ssh-proxy.socket
|
||||||
|
- prometheus-gitea-ssh-proxy.service
|
||||||
|
loop_control:
|
||||||
|
label: "{{ item }}"
|
||||||
|
register: server_gitea_ssh_proxy_units
|
||||||
|
when: server_gitea_proxy_enabled | bool
|
||||||
|
|
||||||
|
- name: Reload systemd after Gitea SSH proxy unit changes
|
||||||
|
tags: [services, gitea_cutover]
|
||||||
|
ansible.builtin.systemd:
|
||||||
|
daemon_reload: true
|
||||||
|
when:
|
||||||
|
- server_gitea_proxy_enabled | bool
|
||||||
|
- server_gitea_ssh_proxy_units is changed
|
||||||
|
- not ansible_check_mode
|
||||||
|
|
||||||
|
- name: Manage the public Gitea SSH socket separately from administrative SSH
|
||||||
|
tags: [services, gitea_cutover]
|
||||||
|
ansible.builtin.systemd:
|
||||||
|
name: prometheus-gitea-ssh-proxy.socket
|
||||||
|
state: "{{ 'started' if server_gitea_on_atlas | bool else 'stopped' }}"
|
||||||
|
enabled: "{{ server_gitea_on_atlas | bool }}"
|
||||||
|
when:
|
||||||
|
- server_gitea_proxy_enabled | bool
|
||||||
|
- not ansible_check_mode
|
||||||
|
|
||||||
|
- name: Open only the public Gitea SSH port after cutover
|
||||||
|
tags: [services, gitea_cutover]
|
||||||
|
ansible.posix.firewalld:
|
||||||
|
port: "{{ server_gitea_ssh_public_port }}/tcp"
|
||||||
|
zone: "{{ server_firewalld_zone }}"
|
||||||
|
state: "{{ 'enabled' if server_gitea_on_atlas | bool else 'disabled' }}"
|
||||||
|
permanent: true
|
||||||
|
immediate: true
|
||||||
|
when:
|
||||||
|
- server_gitea_proxy_enabled | bool
|
||||||
|
- server_firewall_backend == 'firewalld'
|
||||||
112
ansible/roles/profile_server/tasks/legacy_cleanup.yml
Normal file
112
ansible/roles/profile_server/tasks/legacy_cleanup.yml
Normal file
@@ -0,0 +1,112 @@
|
|||||||
|
---
|
||||||
|
- name: Require explicit retirement of the migrated Prometheus source
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- inventory_hostname == 'prometheus'
|
||||||
|
- server_legacy_stack_retired | bool
|
||||||
|
- server_gitea_on_atlas | bool
|
||||||
|
- server_npm_quadlet_cutover | bool
|
||||||
|
- server_backup_export_enabled | bool
|
||||||
|
|
||||||
|
- name: Verify legacy paths have no mounts or container users
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv:
|
||||||
|
- python3
|
||||||
|
- -c
|
||||||
|
- |
|
||||||
|
import json, os, pathlib, subprocess
|
||||||
|
def run(*args):
|
||||||
|
return subprocess.check_output(args, text=True).strip()
|
||||||
|
paths = ['/opt/gitea', '/home/git/.ssh', '/opt/navidrome',
|
||||||
|
'/opt/postgres', '/opt/music', '/opt/containerd', '/opt/docker']
|
||||||
|
mounts = json.loads(run('findmnt', '--json', '--list', '-o', 'TARGET'))['filesystems']
|
||||||
|
for path in paths:
|
||||||
|
assert os.path.realpath(path) == path, 'Symlink in cleanup path: ' + path
|
||||||
|
for mount in mounts:
|
||||||
|
target = mount['target']
|
||||||
|
assert target != path and not target.startswith(path + '/'), 'Mounted cleanup path: ' + path
|
||||||
|
ids = run('podman', 'ps', '-aq').split()
|
||||||
|
containers = json.loads(run('podman', 'inspect', *ids)) if ids else []
|
||||||
|
for container in containers:
|
||||||
|
assert container['Name'].lstrip('/') == 'nginx-proxy-manager', 'Unexpected container; review before cleanup'
|
||||||
|
for mount in container.get('Mounts', []):
|
||||||
|
source = os.path.realpath(mount['Source'])
|
||||||
|
for path in paths:
|
||||||
|
assert source != path and not source.startswith(path + '/'), 'Container uses cleanup path: ' + path
|
||||||
|
for path in ['/opt/music', '/opt/containerd']:
|
||||||
|
if os.path.isdir(path):
|
||||||
|
for entry in pathlib.Path(path).rglob('*'):
|
||||||
|
assert entry.is_dir() and not entry.is_symlink(), 'Unexpected file in empty legacy path: ' + str(entry)
|
||||||
|
if os.path.isdir('/opt/docker'):
|
||||||
|
allowed = {'/opt/docker/server', '/opt/docker/server/docker-compose.yml'}
|
||||||
|
for entry in pathlib.Path('/opt/docker').rglob('*'):
|
||||||
|
assert str(entry) in allowed and not entry.is_symlink(), 'Unexpected legacy Docker content: ' + str(entry)
|
||||||
|
assert run('systemctl', 'is-active', 'prometheus-npm.service') == 'active'
|
||||||
|
assert subprocess.run(['systemctl', 'is-active', '--quiet', 'podman-compose-server.service']).returncode != 0
|
||||||
|
assert subprocess.run(['systemctl', 'is-active', '--quiet', 'prometheus-backup-export.service']).returncode != 0
|
||||||
|
print('Legacy cleanup preflight passed')
|
||||||
|
changed_when: false
|
||||||
|
check_mode: false
|
||||||
|
|
||||||
|
- name: Require the updated backup configuration before deleting fallback files
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv:
|
||||||
|
- python3
|
||||||
|
- -c
|
||||||
|
- |
|
||||||
|
import pathlib, subprocess
|
||||||
|
unit = subprocess.check_output(['systemctl', 'show', 'prometheus-backup-export.service',
|
||||||
|
'-p', 'RequiresMountsFor', '--value'], text=True)
|
||||||
|
assert '/opt/gitea' not in unit, 'Backup unit still depends on legacy Gitea'
|
||||||
|
helper = pathlib.Path('/usr/local/sbin/prometheus-backup-export').read_text()
|
||||||
|
assert 'podman-compose-server' not in helper and 'opt/docker/server' not in helper
|
||||||
|
subprocess.run(['bash', '-n', '/usr/local/sbin/prometheus-backup-export'], check=True)
|
||||||
|
changed_when: false
|
||||||
|
when: not ansible_check_mode
|
||||||
|
|
||||||
|
- name: Delete only the explicitly approved legacy data and fallback files
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: "{{ item }}"
|
||||||
|
state: absent
|
||||||
|
loop:
|
||||||
|
- /opt/gitea
|
||||||
|
- /home/git/.ssh
|
||||||
|
- /opt/navidrome
|
||||||
|
- /opt/postgres
|
||||||
|
- /opt/music
|
||||||
|
- /opt/containerd
|
||||||
|
- /opt/docker
|
||||||
|
- /usr/local/sbin/prometheus-gitea-final-export
|
||||||
|
- /etc/systemd/system/podman-compose-server.service
|
||||||
|
register: server_legacy_deleted
|
||||||
|
diff: false
|
||||||
|
|
||||||
|
- name: Reload systemd after removing the inactive legacy unit
|
||||||
|
ansible.builtin.systemd:
|
||||||
|
daemon_reload: true
|
||||||
|
when:
|
||||||
|
- server_legacy_deleted is changed
|
||||||
|
- not ansible_check_mode
|
||||||
|
|
||||||
|
- name: Inspect the obsolete Git home without following symlinks
|
||||||
|
ansible.builtin.stat:
|
||||||
|
path: /home/git
|
||||||
|
follow: false
|
||||||
|
register: server_legacy_git_home
|
||||||
|
|
||||||
|
- name: Require the obsolete Git account to be absent before removing its empty home
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: [getent, passwd, git]
|
||||||
|
register: server_legacy_git_account
|
||||||
|
changed_when: false
|
||||||
|
failed_when: server_legacy_git_account.rc != 2
|
||||||
|
check_mode: false
|
||||||
|
when: server_legacy_git_home.stat.exists
|
||||||
|
|
||||||
|
# rmdir refuses any nonempty directory; never recursively delete this parent.
|
||||||
|
- name: Remove only the empty obsolete Git home
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: [rmdir, /home/git]
|
||||||
|
register: server_legacy_git_home_removed
|
||||||
|
changed_when: server_legacy_git_home_removed.rc == 0
|
||||||
|
when: server_legacy_git_home.stat.exists
|
||||||
33
ansible/roles/profile_server/tasks/legacy_image_cleanup.yml
Normal file
33
ansible/roles/profile_server/tasks/legacy_image_cleanup.yml
Normal file
@@ -0,0 +1,33 @@
|
|||||||
|
---
|
||||||
|
- name: Require the migrated Prometheus topology for image cleanup
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- inventory_hostname == 'prometheus'
|
||||||
|
- server_gitea_on_atlas | bool
|
||||||
|
- server_npm_quadlet_cutover | bool
|
||||||
|
- server_legacy_images | default([]) | length > 0
|
||||||
|
- >-
|
||||||
|
server_legacy_images | difference([
|
||||||
|
'docker.gitea.com/gitea:1.25.2',
|
||||||
|
'docker.io/deluan/navidrome:latest',
|
||||||
|
'docker.io/library/postgres:13']) | length == 0
|
||||||
|
|
||||||
|
- name: Check whether the explicitly selected legacy images exist
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: [podman, image, exists, "{{ item }}"]
|
||||||
|
loop: "{{ server_legacy_images }}"
|
||||||
|
register: server_legacy_image_presence
|
||||||
|
changed_when: false
|
||||||
|
failed_when: server_legacy_image_presence.rc not in [0, 1]
|
||||||
|
check_mode: false
|
||||||
|
|
||||||
|
# No --force: Podman must refuse images referenced by any existing container.
|
||||||
|
- name: Remove only unused explicitly selected legacy images
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: [podman, image, rm, "{{ item.item }}"]
|
||||||
|
loop: "{{ server_legacy_image_presence.results }}"
|
||||||
|
loop_control:
|
||||||
|
label: "{{ item.item }}"
|
||||||
|
when: item.rc == 0
|
||||||
|
register: server_legacy_image_removal
|
||||||
|
changed_when: server_legacy_image_removal.rc == 0
|
||||||
@@ -8,10 +8,6 @@
|
|||||||
fail_msg: >-
|
fail_msg: >-
|
||||||
server_firewall_backend must be firewalld for the Rocky server profile.
|
server_firewall_backend must be firewalld for the Rocky server profile.
|
||||||
|
|
||||||
- name: Configure DuckDNS updater
|
|
||||||
tags: [dotfiles, dotfiles:server, duckdns]
|
|
||||||
ansible.builtin.import_tasks: duckdns.yml
|
|
||||||
|
|
||||||
- name: Ensure server directories exist
|
- name: Ensure server directories exist
|
||||||
tags: [dotfiles, services]
|
tags: [dotfiles, services]
|
||||||
ansible.builtin.file:
|
ansible.builtin.file:
|
||||||
@@ -23,6 +19,9 @@
|
|||||||
loop: "{{ server_directories | default([]) }}"
|
loop: "{{ server_directories | default([]) }}"
|
||||||
loop_control:
|
loop_control:
|
||||||
label: "{{ item.path }}"
|
label: "{{ item.path }}"
|
||||||
|
when:
|
||||||
|
- item.path != '/opt/gitea/data' or not server_gitea_on_atlas | bool
|
||||||
|
- item.path != server_container_stack_dir or not server_legacy_stack_retired | bool
|
||||||
|
|
||||||
- name: Copy server dotfiles
|
- name: Copy server dotfiles
|
||||||
tags: [dotfiles, dotfiles:server]
|
tags: [dotfiles, dotfiles:server]
|
||||||
@@ -37,7 +36,7 @@
|
|||||||
label: "{{ item.dest }}"
|
label: "{{ item.dest }}"
|
||||||
|
|
||||||
- name: Render server templates
|
- name: Render server templates
|
||||||
tags: [dotfiles, dotfiles:server]
|
tags: [dotfiles, dotfiles:server, gitea_cutover]
|
||||||
ansible.builtin.template:
|
ansible.builtin.template:
|
||||||
src: "{{ item.src }}"
|
src: "{{ item.src }}"
|
||||||
dest: "{{ item.dest if item.dest.startswith('/') else server_user_home ~ '/' ~ item.dest }}"
|
dest: "{{ item.dest if item.dest.startswith('/') else server_user_home ~ '/' ~ item.dest }}"
|
||||||
@@ -48,10 +47,38 @@
|
|||||||
loop_control:
|
loop_control:
|
||||||
label: "{{ item.dest }}"
|
label: "{{ item.dest }}"
|
||||||
no_log: "{{ item.no_log | default(false) }}"
|
no_log: "{{ item.no_log | default(false) }}"
|
||||||
|
when: item.src != 'server/docker-compose.yml.j2' or not server_legacy_stack_retired | bool
|
||||||
|
|
||||||
- name: Manage Podman Compose stack
|
- name: Manage Podman Compose stack
|
||||||
tags: [services, podman]
|
tags: [services, podman]
|
||||||
ansible.builtin.include_tasks: podman-compose.yml
|
ansible.builtin.include_tasks: podman-compose.yml
|
||||||
|
when: not server_legacy_stack_retired | bool
|
||||||
|
|
||||||
|
- name: Import staged NPM Quadlet tasks
|
||||||
|
ansible.builtin.import_tasks: npm_quadlet.yml
|
||||||
|
|
||||||
|
- name: Import explicit legacy server image cleanup
|
||||||
|
ansible.builtin.import_tasks: legacy_image_cleanup.yml
|
||||||
|
tags: [never, server_image_cleanup]
|
||||||
|
when: server_legacy_image_cleanup | default(false) | bool
|
||||||
|
|
||||||
|
- name: Import Prometheus backup export identity tasks
|
||||||
|
ansible.builtin.import_tasks: backup_export_identity.yml
|
||||||
|
|
||||||
|
- name: Import Prometheus backup export job tasks
|
||||||
|
ansible.builtin.import_tasks: backup_export_job.yml
|
||||||
|
tags: [server_legacy_cleanup]
|
||||||
|
|
||||||
|
- name: Import explicitly approved legacy server data cleanup
|
||||||
|
ansible.builtin.import_tasks: legacy_cleanup.yml
|
||||||
|
tags: [never, server_legacy_cleanup]
|
||||||
|
when: server_legacy_cleanup | bool
|
||||||
|
|
||||||
|
- name: Import Prometheus Gitea SSH proxy tasks
|
||||||
|
ansible.builtin.import_tasks: gitea_ssh_proxy.yml
|
||||||
|
|
||||||
|
- name: Import Prometheus Gitea NPM proxy override tasks
|
||||||
|
ansible.builtin.import_tasks: gitea_npm_proxy.yml
|
||||||
|
|
||||||
- name: Ensure server SSH authorized key fragments directory exists
|
- name: Ensure server SSH authorized key fragments directory exists
|
||||||
tags: [services, ssh]
|
tags: [services, ssh]
|
||||||
@@ -77,13 +104,17 @@
|
|||||||
when: server_ssh_authorized_keys | length > 0
|
when: server_ssh_authorized_keys | length > 0
|
||||||
|
|
||||||
- name: Configure server SSH authorized key fragments
|
- name: Configure server SSH authorized key fragments
|
||||||
tags: [services, ssh]
|
tags: [services, ssh, prometheus_backup]
|
||||||
ansible.builtin.lineinfile:
|
ansible.builtin.lineinfile:
|
||||||
path: /etc/ssh/sshd_config
|
path: /etc/ssh/sshd_config
|
||||||
regexp: '^\s*AuthorizedKeysFile\s+'
|
regexp: '^\s*AuthorizedKeysFile\s+'
|
||||||
line: >-
|
line: >-
|
||||||
AuthorizedKeysFile {{ server_ssh_authorized_keys | map(attribute='name')
|
AuthorizedKeysFile {{
|
||||||
| map('regex_replace', '^', '%h/.ssh/authorized_keys.d/') | join(' ') }}
|
((server_ssh_authorized_keys | map(attribute='name')
|
||||||
|
| map('regex_replace', '^', '%h/.ssh/authorized_keys.d/') | list)
|
||||||
|
+ (['%h/.ssh/authorized_keys.d/' ~ server_backup_public_key_name]
|
||||||
|
if server_backup_export_enabled | bool else [])) | join(' ')
|
||||||
|
}}
|
||||||
state: present
|
state: present
|
||||||
validate: "sshd -t -f %s"
|
validate: "sshd -t -f %s"
|
||||||
notify: Reload SSH service
|
notify: Reload SSH service
|
||||||
@@ -100,11 +131,13 @@
|
|||||||
notify: Reload SSH service
|
notify: Reload SSH service
|
||||||
|
|
||||||
- name: Restrict SSH login to allowed users on server
|
- name: Restrict SSH login to allowed users on server
|
||||||
tags: [services]
|
tags: [services, prometheus_backup]
|
||||||
ansible.builtin.lineinfile:
|
ansible.builtin.lineinfile:
|
||||||
path: /etc/ssh/sshd_config
|
path: /etc/ssh/sshd_config
|
||||||
regexp: '^\s*AllowUsers\s+'
|
regexp: '^\s*AllowUsers\s+'
|
||||||
line: "AllowUsers {{ server_sshd_allow_users | join(' ') }}"
|
line: >-
|
||||||
|
AllowUsers {{ (server_sshd_allow_users
|
||||||
|
+ ([server_backup_username] if server_backup_export_enabled | bool else [])) | join(' ') }}
|
||||||
state: present
|
state: present
|
||||||
validate: "sshd -t -f %s"
|
validate: "sshd -t -f %s"
|
||||||
notify: Reload SSH service
|
notify: Reload SSH service
|
||||||
|
|||||||
71
ansible/roles/profile_server/tasks/npm_quadlet.yml
Normal file
71
ansible/roles/profile_server/tasks/npm_quadlet.yml
Normal file
@@ -0,0 +1,71 @@
|
|||||||
|
---
|
||||||
|
- name: Require staged NPM Quadlet for an active cutover
|
||||||
|
tags: [services, npm_quadlet]
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- not server_npm_quadlet_cutover | bool or server_npm_quadlet_stage | bool
|
||||||
|
fail_msg: The NPM Quadlet cutover requires the staged container and network.
|
||||||
|
|
||||||
|
- name: Validate staged NPM Quadlet inputs
|
||||||
|
tags: [services, npm_quadlet]
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- server_npm_quadlet_image is defined
|
||||||
|
- server_npm_quadlet_image is match('^docker\.io/jc21/nginx-proxy-manager@sha256:[a-f0-9]{64}$')
|
||||||
|
- server_gitea_on_atlas | bool
|
||||||
|
fail_msg: Stage the exact running NPM image only after Gitea has left Compose.
|
||||||
|
when: server_npm_quadlet_stage | bool
|
||||||
|
|
||||||
|
- name: Ensure rootful Quadlet directory exists for NPM
|
||||||
|
tags: [services, npm_quadlet]
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: /etc/containers/systemd
|
||||||
|
state: directory
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0755"
|
||||||
|
when: server_npm_quadlet_stage | bool
|
||||||
|
|
||||||
|
- name: Render staged NPM container and network Quadlets
|
||||||
|
tags: [services, npm_quadlet]
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: "{{ item }}.j2"
|
||||||
|
dest: "/etc/containers/systemd/{{ item }}"
|
||||||
|
owner: root
|
||||||
|
group: root
|
||||||
|
mode: "0644"
|
||||||
|
loop:
|
||||||
|
- prometheus-npm.container
|
||||||
|
- server-web.network
|
||||||
|
loop_control:
|
||||||
|
label: "{{ item }}"
|
||||||
|
register: server_npm_quadlet_units
|
||||||
|
when: server_npm_quadlet_stage | bool
|
||||||
|
|
||||||
|
- name: Reload systemd after staging NPM Quadlets
|
||||||
|
tags: [services, npm_quadlet]
|
||||||
|
ansible.builtin.systemd:
|
||||||
|
daemon_reload: true
|
||||||
|
when:
|
||||||
|
- server_npm_quadlet_stage | bool
|
||||||
|
- server_npm_quadlet_units is changed
|
||||||
|
- not ansible_check_mode
|
||||||
|
|
||||||
|
- name: Verify the staged NPM Quadlet was generated
|
||||||
|
tags: [services, npm_quadlet]
|
||||||
|
ansible.builtin.command:
|
||||||
|
argv: [systemctl, show, prometheus-npm.service, --property=LoadState, --value]
|
||||||
|
register: server_npm_quadlet_load_state
|
||||||
|
changed_when: false
|
||||||
|
when:
|
||||||
|
- server_npm_quadlet_stage | bool
|
||||||
|
- not ansible_check_mode
|
||||||
|
|
||||||
|
- name: Reject an invalid staged NPM Quadlet
|
||||||
|
tags: [services, npm_quadlet]
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that: server_npm_quadlet_load_state.stdout == 'loaded'
|
||||||
|
fail_msg: Quadlet generator did not produce prometheus-npm.service.
|
||||||
|
when:
|
||||||
|
- server_npm_quadlet_stage | bool
|
||||||
|
- not ansible_check_mode
|
||||||
@@ -1,24 +0,0 @@
|
|||||||
#!/bin/sh
|
|
||||||
# Managed by Ansible. Contains a Vault token; never copy this file into Git.
|
|
||||||
set -eu
|
|
||||||
umask 077
|
|
||||||
|
|
||||||
log_file={{ (server_user_home ~ '/duckdns/duck.log') | quote }}
|
|
||||||
|
|
||||||
# Keep the token out of process arguments and verify the HTTPS certificate.
|
|
||||||
if ! response=$(curl --fail --silent --show-error --connect-timeout 10 --max-time 30 --config - <<'DUCKDNS_CONFIG'
|
|
||||||
url = "https://www.duckdns.org/update?domains={{ server_duckdns_domain }}&token={{ vault_duckdns_token }}&ip="
|
|
||||||
DUCKDNS_CONFIG
|
|
||||||
); then
|
|
||||||
printf 'ERROR\n' > "$log_file"
|
|
||||||
exit 1
|
|
||||||
fi
|
|
||||||
|
|
||||||
case "$response" in
|
|
||||||
OK) printf 'OK\n' > "$log_file" ;;
|
|
||||||
*)
|
|
||||||
printf 'KO\n' > "$log_file"
|
|
||||||
printf 'DuckDNS update failed.\n' >&2
|
|
||||||
exit 1
|
|
||||||
;;
|
|
||||||
esac
|
|
||||||
@@ -0,0 +1,15 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Prepare a read-only Prometheus application backup for Atlas
|
||||||
|
RequiresMountsFor=/opt/npm {% if not server_gitea_on_atlas | bool %}/opt/gitea {% endif %}{{ server_backup_export_root }}
|
||||||
|
ConditionFileIsExecutable=/usr/local/sbin/prometheus-backup-export
|
||||||
|
|
||||||
|
[Service]
|
||||||
|
Type=oneshot
|
||||||
|
ExecStart=/usr/local/sbin/prometheus-backup-export
|
||||||
|
User=root
|
||||||
|
Group=root
|
||||||
|
UMask=0077
|
||||||
|
TimeoutStartSec=infinity
|
||||||
|
Nice=10
|
||||||
|
IOSchedulingClass=best-effort
|
||||||
|
IOSchedulingPriority=7
|
||||||
@@ -0,0 +1,118 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
set -Eeuo pipefail
|
||||||
|
umask 077
|
||||||
|
|
||||||
|
export_root={{ server_backup_export_root | quote }}
|
||||||
|
versions="$export_root/versions"
|
||||||
|
{% if server_legacy_stack_retired | bool %}
|
||||||
|
stack_unit=prometheus-npm.service
|
||||||
|
{% else %}
|
||||||
|
stack_unit=''
|
||||||
|
compose_active=false
|
||||||
|
quadlet_active=false
|
||||||
|
systemctl is-active --quiet podman-compose-server.service && compose_active=true
|
||||||
|
systemctl is-active --quiet prometheus-npm.service && quadlet_active=true
|
||||||
|
if [[ "$compose_active" == "$quadlet_active" ]]; then
|
||||||
|
echo 'Expected exactly one active NPM service (Compose or Quadlet)' >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
if "$quadlet_active"; then
|
||||||
|
stack_unit=prometheus-npm.service
|
||||||
|
else
|
||||||
|
stack_unit=podman-compose-server.service
|
||||||
|
fi
|
||||||
|
{% endif %}
|
||||||
|
stamp=$(date -u +%Y%m%dT%H%M%SZ)
|
||||||
|
stage=''
|
||||||
|
stack_stopped=false
|
||||||
|
|
||||||
|
exec 9>/run/lock/prometheus-backup-export.lock
|
||||||
|
flock -n 9 || { echo 'A backup export is already running' >&2; exit 1; }
|
||||||
|
|
||||||
|
cleanup() {
|
||||||
|
local rc=$?
|
||||||
|
trap - EXIT
|
||||||
|
if "$stack_stopped"; then
|
||||||
|
if systemctl is-active --quiet "$stack_unit"; then
|
||||||
|
systemctl restart "$stack_unit" || rc=1
|
||||||
|
else
|
||||||
|
systemctl start "$stack_unit" || rc=1
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
if (( rc != 0 )) && [[ -n "$stage" && -d "$stage" ]]; then
|
||||||
|
rm -rf -- "$stage"
|
||||||
|
fi
|
||||||
|
exit "$rc"
|
||||||
|
}
|
||||||
|
trap cleanup EXIT
|
||||||
|
trap 'exit 129' HUP
|
||||||
|
trap 'exit 130' INT
|
||||||
|
trap 'exit 143' TERM
|
||||||
|
|
||||||
|
systemctl is-active --quiet "$stack_unit" || {
|
||||||
|
echo "The managed NPM unit $stack_unit must be active before preparing a backup" >&2
|
||||||
|
exit 1
|
||||||
|
}
|
||||||
|
|
||||||
|
paths=(
|
||||||
|
{% for path in server_backup_export_paths %}
|
||||||
|
{{ path | quote }}
|
||||||
|
{% endfor %}
|
||||||
|
)
|
||||||
|
excludes=(
|
||||||
|
{% for path in server_backup_export_excludes %}
|
||||||
|
--exclude={{ path | quote }}
|
||||||
|
{% endfor %}
|
||||||
|
)
|
||||||
|
for path in "${paths[@]}"; do
|
||||||
|
[[ -e "/$path" ]] || { echo "Required backup path missing: /$path" >&2; exit 1; }
|
||||||
|
done
|
||||||
|
[[ ! -e "$versions/$stamp" ]] || { echo "Export version already exists: $stamp" >&2; exit 1; }
|
||||||
|
stage=$(mktemp -d "$export_root/.staging.XXXXXXXX")
|
||||||
|
|
||||||
|
# SQLite databases and their accompanying files are copied while both
|
||||||
|
# managed containers are stopped. The EXIT trap restarts the stack on error.
|
||||||
|
stack_stopped=true
|
||||||
|
systemctl stop "$stack_unit"
|
||||||
|
tar --acls --xattrs --selinux "${excludes[@]}" -C / -cf "$stage/payload.tar" "${paths[@]}"
|
||||||
|
systemctl start "$stack_unit"
|
||||||
|
for container in nginx-proxy-manager{% if not server_gitea_on_atlas | bool %} gitea{% endif %}; do
|
||||||
|
running=false
|
||||||
|
for _ in {1..30}; do
|
||||||
|
if [[ $(podman inspect --format '{{ '{{.State.Running}}' }}' "$container" 2>/dev/null) == true ]]; then
|
||||||
|
running=true
|
||||||
|
break
|
||||||
|
fi
|
||||||
|
sleep 2
|
||||||
|
done
|
||||||
|
"$running" || { echo "Container did not restart: $container" >&2; exit 1; }
|
||||||
|
done
|
||||||
|
ready=false
|
||||||
|
for _ in {1..60}; do
|
||||||
|
if curl -fsS --connect-timeout 2 --max-time 3 -o /dev/null http://127.0.0.1:81/; then
|
||||||
|
ready=true
|
||||||
|
break
|
||||||
|
fi
|
||||||
|
sleep 2
|
||||||
|
done
|
||||||
|
"$ready" || { echo 'NPM administration did not become ready after backup' >&2; exit 1; }
|
||||||
|
stack_stopped=false
|
||||||
|
|
||||||
|
tar -tf "$stage/payload.tar" >/dev/null
|
||||||
|
(cd "$stage" && sha256sum payload.tar >payload.sha256)
|
||||||
|
printf '{"schema":1,"host":"prometheus","created_utc":"%s"}\n' "$stamp" >"$stage/metadata.json"
|
||||||
|
chown root:{{ server_backup_username }} "$stage" "$stage/payload.tar" "$stage/payload.sha256" "$stage/metadata.json"
|
||||||
|
chmod 0750 "$stage"
|
||||||
|
chmod 0640 "$stage/payload.tar" "$stage/payload.sha256" "$stage/metadata.json"
|
||||||
|
mv -- "$stage" "$versions/$stamp"
|
||||||
|
stage=''
|
||||||
|
ln -s "$stamp" "$versions/.current.new"
|
||||||
|
mv -Tf -- "$versions/.current.new" "$versions/current"
|
||||||
|
|
||||||
|
# Keep a small source-side safety window; Atlas owns long-term retention.
|
||||||
|
mapfile -t old_versions < <(find "$versions" -mindepth 1 -maxdepth 1 -type d \
|
||||||
|
-printf '%f\n' | grep -E '^[0-9]{8}T[0-9]{6}Z$' | sort -r | tail -n +{{ server_backup_export_source_keep + 1 }})
|
||||||
|
for old in "${old_versions[@]}"; do
|
||||||
|
rm -rf -- "${versions:?}/$old"
|
||||||
|
done
|
||||||
|
echo "Prepared Prometheus backup export $stamp"
|
||||||
@@ -0,0 +1,10 @@
|
|||||||
|
[Unit]
|
||||||
|
Description=Prepare daily Prometheus application backup for Atlas
|
||||||
|
|
||||||
|
[Timer]
|
||||||
|
OnCalendar={{ server_backup_export_calendar }}
|
||||||
|
Persistent=false
|
||||||
|
Unit=prometheus-backup-export.service
|
||||||
|
|
||||||
|
[Install]
|
||||||
|
WantedBy=timers.target
|
||||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user