mirror of
https://github.com/fscotto/infra.git
synced 2026-10-03 21:39:50 +00:00
Compare commits
46 Commits
4c10af3187
...
main
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
def3dbf313 | ||
|
|
a00602973c | ||
|
|
18eb2d2eb2 | ||
|
|
269fb13665 | ||
|
|
2dfe766b7b | ||
|
|
755f24bc72 | ||
|
|
7bc7f0e645 | ||
|
|
1577eec19d | ||
|
|
e30683c3d1 | ||
|
|
4bd6aafb53 | ||
|
|
9e76309833 | ||
|
|
ed3fee06e8 | ||
|
|
309d64b4ed | ||
|
|
dd33a4f55d | ||
|
|
0028fe8c4d | ||
|
|
12037fcc9a | ||
|
|
3f9a626759 | ||
|
|
31fedb8d44 | ||
|
|
9b5ee77905 | ||
|
|
54fb7d46d7 | ||
|
|
a609e68f42 | ||
|
|
06d3b175cb | ||
|
|
256d758b1a | ||
|
|
9d0013769c | ||
|
|
5b0f415163 | ||
|
|
3401b6137d | ||
|
|
bae6a9f554 | ||
|
|
c37483ba38 | ||
|
|
f491362365 | ||
|
|
5a1047adde | ||
|
|
802cb8c7ba | ||
|
|
0144600a4a | ||
|
|
3d2ef02c98 | ||
|
|
8844d00e24 | ||
|
|
9798fe3a12 | ||
|
|
702283b430 | ||
|
|
797087c66f | ||
|
|
fa1c8c0b82 | ||
|
|
0b6efc9ad8 | ||
|
|
361ee77d72 | ||
|
|
d4e40d423a | ||
|
|
0a5c2ac1a4 | ||
|
|
48a7f57f7e | ||
|
|
defa98c968 | ||
|
|
21e41f4fc1 | ||
|
|
e10c6694f8 |
367
AGENTS.md
367
AGENTS.md
@@ -25,6 +25,9 @@ Ansible-driven personal infrastructure repo for Fedora and Void desktops, Fedora
|
||||
- Preserve layering `all -> platform -> role -> desktop -> host`.
|
||||
- Keep `ansible/site.yml` small; orchestration belongs there, implementation belongs in roles.
|
||||
- Prefer minimal, targeted edits. Preserve idempotency and existing ordering.
|
||||
- Keep completed one-time cleanup operations out of the playbook. Execute them directly
|
||||
with explicit authorization; retain only the ongoing desired-state configuration and
|
||||
historical documentation, not permanent cleanup flags or tasks.
|
||||
- Use Git Flow branch prefixes: `feature/` for new functionality, `bugfix/` for non-urgent fixes,
|
||||
`hotfix/` for urgent production fixes, `release/` for release preparation, and `support/` for
|
||||
maintained release lines. Do not use abbreviated prefixes such as `feat/`.
|
||||
@@ -54,18 +57,42 @@ Ansible-driven personal infrastructure repo for Fedora and Void desktops, Fedora
|
||||
- Emacs is disabled by default; temporary Emacs check: `ansible-playbook ansible/site.yml --limit <host> --tags emacs --check --diff -e emacs_enabled=true`
|
||||
- AI coding agents: `ansible-playbook ansible/site.yml --limit <host> --tags ai_agents --check --diff`
|
||||
- Mail bootstrap: `sh -n scripts/bootstrap_mail.sh` and `shellcheck scripts/bootstrap_mail.sh`
|
||||
- Server compose render: `podman-compose -f /opt/docker/server/docker-compose.yml config` and `systemctl status podman-compose-server`
|
||||
- Server NPM Quadlet: `systemctl status prometheus-npm.service`; the Compose fallback is retired.
|
||||
- Explicit Prometheus legacy cleanup (destructive only without check mode):
|
||||
`ansible-playbook ansible/site.yml --limit prometheus --tags server_legacy_cleanup --check --diff -e server_legacy_cleanup=true`
|
||||
- Atlas media stack:
|
||||
`ansible-playbook ansible/site.yml --limit atlas --tags storage,sharing,containers --check --diff`
|
||||
- Atlas rootless Gitea staging (does not start Gitea):
|
||||
`ansible-playbook ansible/site.yml --limit atlas --tags gitea --check --diff`
|
||||
- Atlas canonical Gitea domain (restarts only Gitea on a real configuration change):
|
||||
`ansible-playbook ansible/site.yml --limit atlas --tags gitea_public_domain --check --diff`
|
||||
- Atlas Nextcloud/ONLYOFFICE steady state:
|
||||
`ansible-playbook ansible/site.yml --limit atlas --tags nextcloud --check --diff`
|
||||
- Atlas iCloudPD storage and boot-started Quadlet:
|
||||
`ansible-playbook ansible/site.yml --limit atlas --tags icloudpd --check --diff`
|
||||
- Ongoing Gitea proxy configuration:
|
||||
`ansible-playbook ansible/site.yml --limit prometheus --tags gitea_cutover,prometheus_backup --check --diff -e server_gitea_on_atlas=true`
|
||||
and `ansible-playbook ansible/site.yml --limit atlas --tags gitea --check --diff`
|
||||
- Atlas daily Navidrome music copy:
|
||||
`ansible-playbook ansible/site.yml --limit atlas --tags music_sync --check --diff`
|
||||
- Atlas network/share hardening:
|
||||
`ansible-playbook ansible/site.yml --limit atlas --tags hardening,sharing --check --diff`
|
||||
- Atlas ZFS snapshot retention and scrub timers:
|
||||
`ansible-playbook ansible/site.yml --limit atlas --tags snapshots,scrub --check --diff`
|
||||
- Atlas encrypted Borg backup:
|
||||
`ansible-playbook ansible/site.yml --limit atlas --tags packages,borg --check --diff`
|
||||
- Atlas Borg progress logging only:
|
||||
`ansible-playbook ansible/site.yml --limit atlas --tags borg_logging --check --diff`
|
||||
- Atlas manual offline USB backup and 45Drives Alerts reminder:
|
||||
`ansible-playbook ansible/site.yml --limit atlas --tags usb_backup,usb_reminder --check --diff`
|
||||
- Atlas pool, disk, capacity, temperature, and job monitoring:
|
||||
`ansible-playbook ansible/site.yml --limit atlas --tags monitoring --check --diff`
|
||||
- Atlas explicit post-restore SELinux relabeling:
|
||||
`ansible-playbook ansible/site.yml --limit atlas --tags restorecon --check -e '{"atlas_restorecon_paths":["/zpool/archive"]}'`
|
||||
- Prometheus/Aegis WireGuard gateway:
|
||||
`ansible-playbook ansible/site.yml --limit prometheus,aegis --tags wireguard --check --diff`
|
||||
- DuckDNS config only: `ansible-playbook ansible/site.yml --limit prometheus --tags duckdns --check --diff`
|
||||
- Prometheus NPM Quadlet steady state (does not perform a cutover):
|
||||
`ansible-playbook ansible/site.yml --limit prometheus --tags npm_quadlet --check --diff`
|
||||
|
||||
## Conventions
|
||||
- Use FQCN Ansible modules.
|
||||
@@ -109,21 +136,29 @@ The dotfile vars follow the same split: `desktop_common_dotfiles` carries mode-i
|
||||
- Windows applications are installed manually and are not managed from the WSL profile.
|
||||
|
||||
## Rocky Server Notes
|
||||
- DuckDNS is rendered by `profile_server` from host-local `server_duckdns_domain` and
|
||||
`vault_duckdns_token`. Keep the rotated token in encrypted Vault or untracked local vars, never in
|
||||
dotfiles. The private `~/duckdns/duck.sh` keeps the existing entrypoint; rendering uses `no_log`
|
||||
and disables diffs. Provisioning does not execute the updater or change its external schedule.
|
||||
- DuckDNS support is removed from the server profile, not feature-gated. No updater tasks,
|
||||
templates or enablement variables remain. Prometheus uses its static IP and `fscotto.co`;
|
||||
the local updater, log and cron job were already retired. External DuckDNS account/name
|
||||
and existing encrypted token are outside this removal and remain untouched.
|
||||
- `rocky_server` is a child of both `platform_rocky` and `server`; `prometheus` is its active target.
|
||||
- The target must already provide `server_username` with local sudo access before the profile runs.
|
||||
- The Rocky profile installs Podman and podman-compose, uses firewalld, preserves SELinux enforcement, and renders the
|
||||
existing Nginx Proxy Manager/Gitea Compose stack with a `podman-compose-server` systemd unit. PostgreSQL and
|
||||
Navidrome are no longer part of the desired Prometheus configuration. The role does not stop or remove legacy
|
||||
containers, delete `/opt/postgres/data`, start the Compose stack, update DNS, or cut over traffic.
|
||||
- The Rocky profile installs Podman and podman-compose. Prometheus explicitly retires the legacy
|
||||
Compose unit, files and final-export helper with `server_legacy_stack_retired: true`.
|
||||
Its approved opt-in cleanup removed old application data on 2026-10-03; normal runs do not
|
||||
delete data or recreate the retired files. On Prometheus, Nginx Proxy Manager is now the rootful
|
||||
`prometheus-npm.service` Quadlet with a pinned image digest and the existing `/opt/npm/data` and
|
||||
`/opt/npm/letsencrypt` bind mounts. The rootful `server_web` bridge remains `10.89.0.0/24`.
|
||||
Gitea runs on Atlas; PostgreSQL and Navidrome are absent from the desired Prometheus stack.
|
||||
Normal runs do not delete legacy data, update DNS, or perform an implicit cutover;
|
||||
destructive cleanup requires its explicit tag and opt-in extra-var.
|
||||
- Firewalld enables SSH, Cockpit (`9090/tcp`), HTTP and HTTPS. Nginx Proxy Manager publishes `80/tcp` and
|
||||
`443/tcp`; bind its administration interface only to `127.0.0.1:81` and use `npm-tunnel` from Ikaros or Nymph.
|
||||
Nextcloud remains disabled; do not provision `/srv/nextcloud` directories.
|
||||
- `scripts/migrate_prometheus_data.sh` is the separate, source-host-run NPM/Gitea migration path. It dry-runs by
|
||||
default and requires explicit source-stack quiescing before copying persistent Docker data with rsync.
|
||||
- The completed Ubuntu-to-Rocky data migration script and its operational instructions
|
||||
have been removed; current provisioning does not provide that one-time migration path.
|
||||
- Completed Gitea owner-migration, migration-restore and final-export tasks, helpers and flags
|
||||
are removed. Current Gitea marker/ownership checks, recurring backups and proxy configuration
|
||||
remain intact. `server_gitea_proxy_enabled` controls ongoing proxy management only.
|
||||
- Atlas-only OpenZFS, NFS, Samba, and Syncthing stay selected through Atlas host variables and must not
|
||||
leak into `rocky_server`. Cockpit plus its Navigator and Podman extensions are selected explicitly for
|
||||
Prometheus through its host variables.
|
||||
@@ -156,7 +191,9 @@ The dotfile vars follow the same split: `desktop_common_dotfiles` carries mode-i
|
||||
- `profile_backend_phase1` temporarily runs rootless Navidrome and Syncthing on Atlas until Uranus replaces
|
||||
them. It binds only to Atlas' LAN IP, never `wg0`; Navidrome and the Syncthing GUI admit only Aegis as
|
||||
the source-NAT gateway, while native Syncthing ports admit the configured LAN. It initializes fresh
|
||||
state only and never migrates or deletes source application data.
|
||||
state only and never migrates or deletes source application data. The enabled rootless
|
||||
`atlas-music-sync.timer` copies `/zpool/archive/Music` to `/zpool/media/music` daily at 00:45
|
||||
Europe/Rome without deleting destination files; it requires both datasets to be mounted.
|
||||
- `wireguard_overlay` manages `wg0` between Prometheus (`10.0.0.1`) and Aegis (`10.0.0.2`). It persists private
|
||||
keys only on their respective hosts, exchanges only derived public keys through Ansible, and verifies a real peer
|
||||
handshake. Prometheus opens `51820/udp`; Aegis is the LAN gateway. Its persistent IPv4 forwarding, narrowly scoped
|
||||
@@ -172,52 +209,298 @@ Prometheus--Aegis WireGuard gateway are operational. The gateway handshake, forw
|
||||
and TCP reachability to Atlas were verified. Temporary Navidrome and Syncthing are available through their
|
||||
manual NPM Proxy Hosts; Syncthing uses `/data/Org` backed by the SMB-shared Archive dataset. Aegis has also
|
||||
validated NFSv4.2 read, write, delete, and `all_squash` mapping to UID/GID `1100` end-to-end. The ZFS
|
||||
snapshot timers are active and the first recursive hourly snapshot completed successfully; the first
|
||||
scheduled retention prune and monthly scrub remain runtime checks.
|
||||
snapshot timers are active; a recursive hourly snapshot and scheduled retention prune completed
|
||||
successfully. The first monthly scrub remains a runtime check.
|
||||
|
||||
### Priority 1 - Data protection
|
||||
- [x] Deploy Ansible-managed recursive ZFS snapshots with 24 hourly, 30 daily, 8 weekly, and 12 monthly
|
||||
generations, plus a monthly scrub on the first Sunday at 03:00. The timers and first hourly snapshot were
|
||||
verified on Atlas. Still observe the first scheduled retention prune and scrub; Cockpit Scheduler is for
|
||||
visibility or manual operations only, and snapshot rollback is never automated.
|
||||
verified on Atlas. Cockpit Scheduler is for visibility or manual operations only, and snapshot
|
||||
rollback is never automated.
|
||||
- [ ] Verify the first monthly ZFS scrub from its actual service result. Scheduled retention pruning
|
||||
was observed on 2026-09-30; timer activation alone does not establish a successful scrub.
|
||||
- [x] Activate and validate the encrypted offsite Borg backup to the Hetzner Storage Box. Atlas uses the
|
||||
dedicated SSH identity, pinned ED25519 host key, Vault-backed `repokey` encryption, and a locked
|
||||
non-login `borg` account with no sudo or supplementary groups. The initial snapshot-consistent backup,
|
||||
Borg repository check, and temporary-directory restore completed successfully; the restored `Archive`
|
||||
tree matched the live data, and temporary snapshots and mounts were removed. The exported recovery key
|
||||
was copied offline. Daily backup retries and logging, 30 daily, 8 weekly and 12 monthly archives,
|
||||
compaction, and monthly repository checks are enabled.
|
||||
- [ ] Add the UUID-bound offline USB backup with versioned rsync, locking, capacity checks, verification,
|
||||
safe unmounting and a tested restore procedure; never trigger it for an arbitrary USB disk.
|
||||
- [ ] Test restores independently from a ZFS snapshot, Borg, and the offline USB backup before relying on
|
||||
any backup path.
|
||||
- [ ] Add monitoring and alerting for pool health, scrub/resilver, SMART data, temperatures, free space,
|
||||
snapshot/backup capacity growth, and failed maintenance or backup timers.
|
||||
compaction, and monthly repository checks are enabled. Runs report a ZFS-based estimated percentage.
|
||||
On 2026-09-30 a successful incremental run also removed the stale 2026-09-29 snapshot and its own
|
||||
temporary snapshot after exit; the earlier `RuntimeDirectory` cleanup failure is resolved.
|
||||
- [x] Populate `/zpool/archive` with the currently available data so offsite and offline backup tests run
|
||||
against a representative load.
|
||||
- [x] Evaluate Borg against the populated pool. The 2026-09-29 archive took 1 h 32 min for 2.18 TB
|
||||
original / 2.04 TB compressed data, with 13.49 GB deduplicated size; retention and compaction
|
||||
succeeded. On 2026-09-30 a subsequent incremental archive completed in about 22 seconds with
|
||||
successful cleanup. The monitor reported 37% Storage Box quota used. These are observed runs, not
|
||||
a guarantee of future duration or compression ratio.
|
||||
- [x] Add the UUID-bound offline USB backup with versioned rsync, locking, capacity checks, verification,
|
||||
safe unmounting and a tested restore procedure; never trigger it for an arbitrary USB disk. The
|
||||
LUKS/ext4 identities were read-only verified; the manual service and 45Drives Alerts reminder timer were
|
||||
deployed on Atlas. Interactive LUKS unlock is part of the manual service; only the reminder is
|
||||
scheduled for the first Saturday of each month at 10:00 Europe/Rome via the existing 45Drives
|
||||
notifier. A manual test produced an Alerts notification, not an email. The first USB attempt failed
|
||||
on a `security.selinux` xattr and was interrupted; the xattr filter is deployed and the temporary
|
||||
recursive snapshot and open LUKS mapper were cleaned up. A later run reported checksum verification
|
||||
and published the USB version, but failed while removing host-namespace ZFS snapshot mounts. Those
|
||||
exact mounts and snapshots were cleaned up. An `ExecStopPost` helper now removes only the named
|
||||
temporary snapshot after the backup process exits. A new full run checksum-verified and published a
|
||||
USB version; the service ended successfully, the mapper closed, no temporary USB snapshot remained,
|
||||
and the pool was healthy. On 2026-09-25 an independent, read-only USB restore test copied one file from
|
||||
the published `atlas/latest` version into `/var/tmp` and matched contents, owner, mode, size, mtime and
|
||||
POSIX ACL. The temporary copy and mount were removed, the mapper closed, and the pool remained healthy.
|
||||
- [x] Test restores independently from a ZFS snapshot, Borg, and the offline USB backup before relying on
|
||||
any backup path. The earlier Borg temporary-directory restore passed. On 2026-09-25 a separate,
|
||||
read-only ZFS snapshot test restored one file to `/var/tmp`, confirmed matching contents, ownership,
|
||||
mode, mtime and ACL, then removed its temporary copy and on-demand mount. This is a file-level smoke
|
||||
test, not full dataset recovery. An independent USB file restore passed on 2026-09-25 with matching
|
||||
content and metadata; the later scaled OS-rebuild rehearsal is documented under Priority 2.
|
||||
- [x] Add monitoring and alerting for pool health, scrub/resilver, SMART data, temperatures, free space,
|
||||
snapshot/local-backup growth, Hetzner Storage Box quota, and failed maintenance or backup timers.
|
||||
The half-hourly Atlas health monitor and systemd final-failure hooks are deployed. A live probe
|
||||
found no issues; the service and timer succeeded, and a labelled 45Drives Alerts test notification
|
||||
was submitted. Alerts are deduplicated; email delivery is not claimed. The Storage Box quota probe
|
||||
runs `df -m` over the dedicated pinned-key SSH identity and does not open the Borg repository.
|
||||
The failed-job hook was corrected to pass the literal systemd unit name; its expansion was verified
|
||||
on Atlas, but a new real failure notification has not been deliberately triggered.
|
||||
|
||||
### Priority 2 - NAS operability and recovery
|
||||
- [ ] Document and test disaster recovery: rebuild Atlas with Ansible, import the existing pool, restore
|
||||
from snapshot/USB/Hetzner, preserve Vault and Borg recovery material offline, and define RPO/RTO.
|
||||
- [ ] Define a controlled Rocky kernel/OpenZFS update and reboot procedure.
|
||||
- [ ] Add the Atlas-initiated least-privilege Prometheus backup pull: Prometheus exposes only prepared
|
||||
- [x] Document and test disaster recovery in `docs/atlas-recovery.md`: the operator confirmed Vault
|
||||
and Borg recovery material is available offline; provisional targets are RPO 24h/RTO 72h. On
|
||||
2026-09-30 an isolated small Rocky VM was rebuilt with the Atlas Ansible roles, imported its
|
||||
preserved RAIDZ2 pool without force/rewind, and restored a file from the preserved snapshot;
|
||||
the second Ansible run was idempotent. Earlier independent production ZFS, USB, and Borg file
|
||||
restore tests remain separate evidence. A production-size full restore, unclean import, and
|
||||
measured 24h/72h compliance are not claimed.
|
||||
- [x] Define a controlled Rocky kernel/OpenZFS update and reboot procedure in
|
||||
`docs/atlas-updates.md`. The first real change-window execution is not yet
|
||||
validated; the procedure never reboots automatically or upgrades pool features.
|
||||
- [x] Add the Atlas-initiated least-privilege Prometheus backup pull: Prometheus exposes only prepared
|
||||
read-only dumps through a dedicated account and Atlas retains the private SSH key, pinned host key,
|
||||
atomic pull, verification, retention and systemd service/timer.
|
||||
- [ ] Decide whether a common SMB/NFS namespace is required. `Archive` (SMB) and `photobook` (NFS) are
|
||||
intentionally distinct today; only if a shared namespace is selected, finalize its UID/GID, group,
|
||||
and POSIX ACL model and test the same files through both protocols.
|
||||
atomic pull, verification, retention and systemd service/timer. The dedicated key/account and unit
|
||||
files are deployed; live read-only SSH, shell denial, and write denial were verified. On 2026-09-30
|
||||
a manual export, Atlas pull, checksum verification, and temporary restore passed; both SQLite
|
||||
databases passed integrity checks and a restored Git repository passed `git fsck`. Both daily
|
||||
timers are enabled for 02:00/03:00 Europe/Rome. On 2026-10-01 their first scheduled export and
|
||||
pull succeeded: Atlas verified the payload checksum and published `20261001T000001Z` as `latest`.
|
||||
- [x] Decide whether a common SMB/NFS namespace is required: no. `Archive` (SMB) and `photobook` (NFS)
|
||||
remain intentionally distinct; `docs/atlas-sharing-decision.md` records the decision. No ACL or export
|
||||
change is authorized by this decision.
|
||||
|
||||
### Priority 3 - Service expansion
|
||||
- [ ] After data protection and recovery are validated, populate `/zpool/media/music` and validate Navidrome.
|
||||
- [ ] Design and deploy Nextcloud as another explicitly temporary Atlas service before Uranus. Give it
|
||||
separate persistent application, database, and cache storage; keep credentials in Vault; publish it only
|
||||
through NPM over the Prometheus--Aegis gateway; and define backup, upgrade, and eventual Uranus-migration
|
||||
procedures before exposing user data. Do not deploy Nextcloud before the data-protection checklist is complete.
|
||||
- [x] Populate `/zpool/media/music` and validate Navidrome. On 2026-09-30, 21,158 files
|
||||
(93,937,810,350 regular-file bytes) were copied from `/zpool/archive/Music` using a temporary
|
||||
ZFS snapshot; a checksum-based rsync dry run found no differences or extra files. Navidrome saw
|
||||
all files through its read-only mount, completed a scan, indexed 18,168 tracks, and responded
|
||||
over HTTP. Some imported playlists still reference obsolete Windows paths. The source was left
|
||||
intact and the temporary snapshot was removed.
|
||||
- [x] Schedule a daily, non-deleting copy from `Archive/Music` to the separate Navidrome music
|
||||
dataset. The rootless `atlas-music-sync.timer` is enabled for 00:45 Europe/Rome; a manual
|
||||
idempotent service run succeeded on 2026-10-01. The first scheduled run triggered at
|
||||
00:45 CEST on 2026-10-02 and exited successfully (`Result=success`, status 0); the next
|
||||
run is scheduled for 2026-10-03 00:45 CEST.
|
||||
- [x] Design the staged Prometheus-to-Atlas Gitea migration in `docs/atlas-gitea-migration.md`.
|
||||
The approved topology keeps NPM on Prometheus and moves HTTPS and public SSH (TCP/2222) together;
|
||||
Gitea runs as an `admin`-owned rootless user Quadlet on Atlas with an internal `gitea` user.
|
||||
The rootful-to-rootless data-layout
|
||||
conversion passed an isolated restore rehearsal. The later partial cutover is tracked below.
|
||||
- [x] Prepare the dedicated Atlas Gitea dataset, non-login UID/GID 1101 with a separate rootless Podman
|
||||
sub-ID range, and disabled user Quadlet. On 2026-10-01 the targeted Ansible run and a second idempotent
|
||||
run passed; the generated unit was inactive, with no staging HTTP/SSH listener. POSIX ACLs on only the
|
||||
service-namespace parents grant this account traversal without access to sibling datasets.
|
||||
- [x] Perform an isolated rootless restore rehearsal from the verified Prometheus backup. On 2026-10-01
|
||||
the SHA-256-checked selective extraction and path/SSH conversion succeeded; SQLite `quick_check`
|
||||
passed, all 33 repositories passed `git fsck`, and source/target public SSH host-key fingerprints
|
||||
matched. The pinned rootless image answered HTTP and listened on internal SSH/2222 with
|
||||
`--network none`; the temporary container was removed and the Quadlet stayed inactive. A second
|
||||
restore run made no changes. This is a rehearsal copy, not the final consistent cutover copy.
|
||||
- [x] Verify ZFS and Borg coverage of the staged Gitea dataset. On 2026-10-01 the managed recursive
|
||||
hourly snapshot `atlas-auto-hourly-20261001T193401Z` included it, and the managed incremental
|
||||
Borg archive `atlas-20261001T193420Z` included its database. A private one-file restore from
|
||||
each independently matched the staged database and passed SQLite `quick_check`; temporary files
|
||||
and snapshot mounts were removed, the Borg service ended successfully, and the pool was healthy.
|
||||
- [x] Include the new Gitea dataset in a UUID-bound offline USB version and test a file restore
|
||||
before accepting production writes. The operator's 2026-10-01 manual run published version
|
||||
`20261001T201220Z-254397` successfully on 2026-10-02. Its Gitea database was restored to a
|
||||
temporary directory from a read-only mount: contents, owner, group, mode, size, mtime and POSIX
|
||||
ACL matched, and SQLite `quick_check` passed. Temporary files and mounts were removed, LUKS
|
||||
was closed, and the pool remained healthy. A redundant run was stopped during verification;
|
||||
its temporary snapshot was cleaned up and the service's resulting failed state was reset.
|
||||
- [x] Install a separate opt-in final Gitea export helper on Prometheus. Its 2026-10-01 targeted
|
||||
deployment and `bash -n` passed while Gitea and NPM stayed running. It refuses an active export
|
||||
timer, stops only Gitea, verifies SQLite, publishes a checksum-verified Gitea-only version for
|
||||
Atlas' existing pull, and leaves the source stopped on success. It was invoked on 2026-10-02
|
||||
after the export timer was stopped; version `20261002T071525Z` was pulled and verified on Atlas.
|
||||
- [x] Prepare the Atlas final-restore gate without replacing the rehearsal: it accepts only a
|
||||
checksum-verified `gitea-cutover` export, refuses a running target, stages and validates the new
|
||||
layout before replacing the marked rehearsal, and rolls back a failed swap. Synthetic success
|
||||
and rollback tests passed on 2026-10-01. On 2026-10-02 the final gate replaced the rehearsal;
|
||||
SQLite `quick_check`, all 33 repository `git fsck` checks, checksum and SSH host-key comparison passed.
|
||||
- [x] Start the rootless Atlas Gitea Quadlet and move the primary HTTPS route. On 2026-10-02 Atlas
|
||||
answered HTTP 200 through the Aegis gateway. NPM stayed on Prometheus; its variable upstream
|
||||
required a managed Nginx `server_proxy.conf` override because runtime DNS ignores Compose
|
||||
`extra_hosts`. The primary public HTTPS page and API returned 200, and `git ls-remote` succeeded
|
||||
for a representative repository after NPM restart; the Navidrome and Syncthing Proxy Hosts also
|
||||
responded. The source
|
||||
Gitea container was removed from the desired Compose stack without deleting its data; the
|
||||
Prometheus backup export timer resumed for NPM only. A post-cutover recursive ZFS snapshot and
|
||||
encrypted Borg archive `atlas-20261002T073044Z` completed successfully.
|
||||
- [x] Move the live Gitea Quadlet and dataset from the legacy host `gitea` account to `admin`
|
||||
after a disposable snapshot-copy test of the pinned derived image. On 2026-10-02 the explicit
|
||||
outage run stopped only legacy Gitea, made safety snapshot
|
||||
`zpool/services/data/gitea@gitea-owner-migration-20261002T100104`, changed dataset ownership,
|
||||
and validated loopback staging (HTTP 200, internal `gitea` UID/GID 1000, SQLite `quick_check`)
|
||||
before promoting the `admin` Quadlet. Production LAN and public HTTPS returned 200; Navidrome
|
||||
and Syncthing remained active, the pool was healthy, and the normal Gitea run changed nothing.
|
||||
The old host account and data on Prometheus remain preserved; the old Atlas Quadlet and its
|
||||
parent-dataset traverse ACL were removed. A subsequent normal run changed nothing.
|
||||
- [x] Validate public Gitea SSH/2222 and an authenticated read from Ikaros. After the VPS
|
||||
firewall was opened on 2026-10-02, TCP/2222 connected, the public ED25519 host-key
|
||||
fingerprint matched Atlas, Gitea authenticated `fscotto` using the `ikaros` key, and
|
||||
`git ls-remote` returned HEAD for `fscotto/infra.git` over public SSH.
|
||||
- [x] Validate authenticated SSH pull and push. On 2026-10-02 the operator reported both
|
||||
operations working through the public SSH endpoint; the earlier agent-run `git ls-remote`
|
||||
remains the independent read-only check. The agent did not perform a test push.
|
||||
- [x] Validate Gitea login and write via HTTPS. On 2026-10-03 the operator confirmed
|
||||
authenticated web login and Git clone/pull/push through the public HTTPS endpoint. Do not
|
||||
restart the stale source Gitea after Atlas has accepted writes.
|
||||
- [x] Design and deploy the empty temporary Atlas Nextcloud/ONLYOFFICE stack on 2026-10-03.
|
||||
The operator explicitly authorized empty internal service startup before the first scrub;
|
||||
this does not close the scrub or protection checks. Four rootless Quadlets, separate
|
||||
component datasets, pinned images/apps, Vault secrets, standard fabio/chiara users, a
|
||||
separate application admin and the Famiglia folder are deployed. Cron and internal Office
|
||||
connection checks succeeded; repeat deployment changed nothing. See `docs/atlas-nextcloud.md`.
|
||||
- [x] Complete the authorized empty-stack public cutover on 2026-10-03 after operator
|
||||
DNS/NPM configuration. Both hostnames passed TLS and HTTPS redirects; authenticated
|
||||
web login, WebDAV, private-file isolation, Famiglia cross-user create/read/update/delete and
|
||||
CalDAV/CardDAV discovery passed. The Office connector and public health/API asset passed.
|
||||
Temporary test files were removed; no iCloud data was imported.
|
||||
- [ ] Complete Nextcloud desktop/mobile editing and synchronization acceptance, and
|
||||
application-consistent backup/restore validation. Close the first actual scrub and
|
||||
protection checks before importing family data.
|
||||
iCloud migration and future Uranus transfer remain separate operations, not playbook flags.
|
||||
- [x] Move Gitea canonical HTTPS and SSH hostname to `git.fscotto.co` on
|
||||
2026-10-03 through Ansible. Only Gitea restarted; second run changed nothing.
|
||||
HTTPS and authenticated SSH reads returned the same repository HEAD.
|
||||
The new NPM hostnames passed TLS/HTTP checks; old DuckDNS Proxy Hosts were
|
||||
observed disabled. Details are in `docs/domain-fscotto-co.md`.
|
||||
- [x] Confirm login on the new Gitea hostname and update remaining client remotes/integrations.
|
||||
The operator confirmed completion on 2026-10-03; the agent did not perform a test push.
|
||||
- [x] Remove obsolete DuckDNS NPM Proxy Hosts, unused certificates and the old upstream override.
|
||||
The operator confirmed completion on 2026-10-03; no new agent runtime check was performed.
|
||||
- [x] Review and remove completed one-time procedures from the playbook.
|
||||
The operator confirmed completion on 2026-10-03.
|
||||
- [x] Retire Prometheus' local DuckDNS updater on 2026-10-03 through Ansible:
|
||||
the five-minute cron entry and private updater/log directory were removed.
|
||||
Provisioning support was subsequently removed entirely; repeat cleanup changed nothing. HTTPS services, private NPM
|
||||
administration and the export timer stayed healthy. The external name and Vault token
|
||||
remain untouched for possible future use on a local host.
|
||||
- [ ] Keep `atlas_manage_media_stack` disabled until the future Immich deployment has validated `/dev/dri`,
|
||||
container paths, and the required Vault database secret.
|
||||
|
||||
### Priority 4 - Optional workflows
|
||||
- [ ] Optionally design iCloud photo ingestion and an Aegis persistent NFS mount as a separate workflow
|
||||
after the storage and backup layers are validated; do not make either a dependency of the Atlas
|
||||
baseline.
|
||||
- [x] Deploy the declared Atlas iCloudPD state dataset and inactive rootless `admin` Quadlet.
|
||||
Photos belong under `/zpool/archive/Pictures/iCloudPD`; private config/MFA state belongs in
|
||||
`zpool/services/data/icloudpd`. Photobook remains reserved for Immich. Ansible now renders
|
||||
`icloudpd.conf` with the Apple ID from the existing Vault key, but does not store the password
|
||||
or manage MFA. Automatic startup was approved on 2026-10-03; the Quadlet now
|
||||
uses `WantedBy=default.target` and Ansible keeps the service running.
|
||||
The isolated no-network layout test is documented in
|
||||
`docs/atlas-icloudpd-migration.md`. On 2026-10-02 Atlas deployment and a second idempotent run
|
||||
passed; no app config existed at deployment. A manual first start on 2026-10-02 generated
|
||||
`icloudpd.conf`; an Ansible run then replaced it with a private mode-0600 Vault-backed template
|
||||
and an idempotent second run. The image later expanded the config, so Ansible now seeds it
|
||||
only when absent and maintains the declared fields. Its launcher requires `traceroute`; the
|
||||
rootless Quadlet grants only `NET_RAW`, tested in isolation and after restart. The service
|
||||
was subsequently initialized interactively; initial ingestion is tracked below.
|
||||
- [x] Retire Aegis iCloudPD completely. The operator authorized deleting its Quadlet,
|
||||
`/var/lib/icloudpd` data, and MFA state despite an unaudited container overlay. After two
|
||||
interactive-sudo runs on 2026-10-02, the unit is `not-found`/`inactive`, the Quadlet and state
|
||||
directory are absent, and AdGuard remains active. The temporary retirement tasks have since
|
||||
been removed from the Aegis role; it no longer manages iCloudPD.
|
||||
- [x] Validate Atlas iCloudPD authentication and initial ingestion. On 2026-10-03 the active
|
||||
rootless service logged `All photos and videos have been downloaded` at 02:16 and reported
|
||||
completion for the user. The destination held 11,658 files (86,020,430,015 bytes); the preceding 24h
|
||||
logs showed download activity without authentication failures or errors. A later read-only check
|
||||
found the service still active. This confirms the initial download, not the next daily cycle.
|
||||
- [x] Declare HEIC decoding for Fedora graphical desktops without converting the originals on Atlas.
|
||||
The Fedora role installs RPM Fusion Free with a pinned signing-key fingerprint and
|
||||
`libheif-freeworld` on Ikaros and Nymph. The package was confirmed installed on Ikaros on
|
||||
2026-10-03; Nymph deployment and an actual image-opening test were not observed.
|
||||
- [ ] Validate Atlas iCloudPD filesystem/SELinux/SMB access, the next daily sync, ZFS/Borg/USB
|
||||
backup inclusion, and isolated restore of photos and private state. A recursive hourly snapshot
|
||||
of `zpool/archive` exists after ingestion, but no iCloudPD-specific backup version or restore
|
||||
has been verified. The first monthly scrub remains a separate open data-protection check.
|
||||
|
||||
## Prometheus NPM Quadlet cutover
|
||||
- [x] Stage a rootful NPM Quadlet using the exact running image and the existing data/certificate
|
||||
mounts, bridge subnet, public HTTP/HTTPS ports, and loopback-only administration port.
|
||||
The generated service depends on `server-web-network.service` and is wanted by `multi-user.target`.
|
||||
- [x] Take and verify the stopped-source export before switching owners. Version
|
||||
`20261003T091009Z` was pulled to Atlas and its NPM SQLite database checked in isolation.
|
||||
- [x] Cut over NPM to `prometheus-npm.service` on 2026-10-03. The legacy Compose unit is inactive
|
||||
and disabled; the Quadlet is active with zero recorded restarts. Public Gitea and Syncthing
|
||||
HTTPS returned 200 with valid TLS, while public TCP/81 remained unreachable.
|
||||
- [x] Validate the post-cutover backup path. The export and Atlas pull published
|
||||
`20261003T091633Z`; checksum, SQLite `quick_check`, ten proxy hosts, six certificate records,
|
||||
both Quadlet files were present, and the complete Let's Encrypt tree (70 regular files plus
|
||||
12 symlinks) matched the live data. A targeted normal Ansible run changed nothing. Details and rollback
|
||||
boundaries are in `docs/prometheus-npm-quadlet.md`.
|
||||
- [x] Remove only unused Gitea, Navidrome and PostgreSQL images with opt-in
|
||||
Ansible tasks on 2026-10-03. Second run changed nothing; NPM stayed active
|
||||
with zero restarts, HTTP/HTTPS passed, backup timer and SSH proxy stayed active.
|
||||
This image-only step preserved data and fallback; the later approved deletion is tracked below. Validation:
|
||||
`ansible-playbook ansible/site.yml --limit prometheus --tags server_image_cleanup --check --diff -e server_legacy_image_cleanup=true`
|
||||
- [x] Complete explicitly approved old-data and Compose fallback removal on 2026-10-03.
|
||||
Backup paths and mount dependencies were reconciled before deletion; repeat cleanup changed
|
||||
nothing. The normal Compose/template/helper check did not recreate retired files.
|
||||
A separately approved manual export/pull published `20261003T112906Z`; checksum and isolated
|
||||
SQLite restore passed with ten proxy hosts and both Quadlet definitions. NPM, primary HTTPS,
|
||||
WireGuard, SSH proxy and backup timer remained healthy; existing backup archives were preserved.
|
||||
- [x] Retire the unused secondary Gitea hostname `git.ov-ad3410.infomaniak.ch`
|
||||
on 2026-10-03. Its NPM Proxy Host was already soft-deleted and had no
|
||||
associated certificate. Its Ansible domain and runtime override were removed;
|
||||
nginx -t and reload passed without restarting NPM. Primary HTTPS returned 200
|
||||
with valid TLS. At that step only `git.fscotto.duckdns.org` remained declared;
|
||||
the subsequent domain transition and operator-confirmed cleanup are tracked above.
|
||||
- [ ] Observe the first scheduled export and Atlas pull after the cutover; the manual end-to-end
|
||||
cycle passed, but the next unattended cycle has not yet occurred.
|
||||
|
||||
## Cerberus Management Node (Deferred)
|
||||
`cerberus` is postponed until the office in the new house is physically set up. It is not an inventory
|
||||
host and this section is a design and implementation backlog, not authorization to provision it early.
|
||||
|
||||
The planned node is a Lenovo ThinkCentre M700 Tiny with an Intel Core i3-6100T, 8 GB RAM, a 256 GB SSD,
|
||||
and native 1 Gbps Ethernet. It will connect to a multi-input KVM switch using a passive DisplayPort-to-HDMI
|
||||
cable, sharing the monitor and peripherals with Ikaros. Fedora Sericea (immutable Fedora with the Sway
|
||||
Wayland compositor) is the intended OS. Cerberus is an isolated management plane: a dedicated Toolbox
|
||||
environment will run Ansible for future `uranus` cluster provisioning. Rootless Podman will host Grafana,
|
||||
Prometheus, and Loki. The 256 GB local SSD is the hot tier retaining metrics and logs for 30 days; scheduled,
|
||||
validated exports of older historical data will use a dedicated Atlas NFS dataset as cold storage.
|
||||
|
||||
### Implementation plan
|
||||
- [ ] Confirm the office, KVM switch, passive DisplayPort-to-HDMI path, shared monitor/peripherals, and native
|
||||
1 Gbps Ethernet are physically operational before adding Cerberus to inventory.
|
||||
- [ ] Install and update Fedora Sericea with Sway; document the immutable-host lifecycle and keep host changes
|
||||
declarative rather than treating the base OS as a mutable workstation.
|
||||
- [ ] Model Cerberus as its own host with independent platform, role, desktop, network, and storage inputs;
|
||||
do not repurpose Ikaros variables or make it a Uranus cluster member.
|
||||
- [ ] Provision an isolated Toolbox-based Ansible controller with the required collections and a reproducible
|
||||
project checkout; define its least-privilege SSH access, known-host handling, and Vault workflow without
|
||||
storing secrets in the image or repository.
|
||||
- [ ] Define the explicit Uranus provisioning workflow from Cerberus, including inventory boundaries,
|
||||
validation-only runs, and separate approval for any destructive cluster operation.
|
||||
- [ ] Design rootless Podman/Quadlet services for Grafana, Prometheus, and Loki, including persistent local
|
||||
state, service ownership, LAN exposure/authentication, resource limits, updates, and backups.
|
||||
- [ ] Size and enforce a 30-day local hot-retention policy for metrics and logs on the 256 GB SSD; validate
|
||||
actual disk growth and alert before capacity exhaustion.
|
||||
- [ ] Create and validate a dedicated Atlas NFS cold-storage dataset and least-privilege export for Cerberus;
|
||||
do not use a broad existing share or couple it to unrelated Atlas application state.
|
||||
- [ ] Implement scheduled, idempotent exports of data older than 30 days to the Atlas NFS cold tier, with
|
||||
locking, capacity checks, integrity verification, retention rules, failure monitoring, and a tested restore.
|
||||
- [ ] Validate management-plane recovery: rebuild Cerberus, restore observability history from Atlas, and
|
||||
confirm that Uranus provisioning can resume without depending on unreproducible local state.
|
||||
|
||||
## Coding Agent Notes
|
||||
- Shared agent definitions and lifecycle flags live in `ai_agents` in `ansible/inventory/group_vars/all.yml`.
|
||||
@@ -258,5 +541,5 @@ scheduled retention prune and monthly scrub remain runtime checks.
|
||||
`/etc/resolv.conf` linked to `/run/systemd/resolve/resolv.conf`. LAN clients may use AdGuard, but
|
||||
Aegis must use the independent upstream DNS declared by `aegis_host_dns_servers` so Greenboot does
|
||||
not depend on the AdGuard container during startup.
|
||||
- iCloudPD requires post-deployment interactive MFA initialization; its cookie/configuration state is
|
||||
persisted in `/var/lib/icloudpd/config`.
|
||||
- Aegis iCloudPD has been retired and is no longer managed by this role. Its service, Quadlet,
|
||||
data, and MFA state were removed with the operator's explicit authorization.
|
||||
|
||||
411
README.it.md
411
README.it.md
@@ -95,6 +95,27 @@ Nota sullo stato attuale del playbook principale:
|
||||
- `ansible/site.yml` applica il profilo server Rocky a `prometheus` con DNF, systemd, dotfiles server e firewalld
|
||||
- `ansible/site.yml` applica il profilo NAS Rocky su `atlas` tramite SSH remoto
|
||||
|
||||
## Nodo pianificato e posticipato: Cerberus
|
||||
|
||||
`cerberus` e un nodo di management **posticipato**, in attesa dell'allestimento
|
||||
fisico dell'ufficio nella nuova casa. Non e ancora presente nell'inventory e non
|
||||
esistono ruoli o playbook che lo prendano come target.
|
||||
|
||||
L'hardware previsto e un Lenovo ThinkCentre M700 Tiny (Intel Core i3-6100T,
|
||||
8 GB di RAM e SSD da 256 GB) con Ethernet nativa a 1 Gbps. Condividera monitor
|
||||
e periferiche di Ikaros tramite uno switch KVM a ingressi multipli, usando un
|
||||
cavo passivo DisplayPort-HDMI per il collegamento video. Il sistema operativo
|
||||
previsto e Fedora Sericea, la variante Fedora immutabile con compositor Wayland
|
||||
Sway.
|
||||
|
||||
Cerberus sara un management plane isolato: Ansible verra eseguito in un ambiente
|
||||
Toolbox dedicato per il provisioning del futuro cluster `uranus`, anziche da
|
||||
Ikaros o da un host non gestito. Lo stack di osservabilita rootless Podman
|
||||
eseguira Grafana, Prometheus e Loki. L'SSD locale sara l'hot storage, con
|
||||
metriche e log conservati per 30 giorni; esportazioni programmate trasferiranno
|
||||
i dati storici piu vecchi su un dataset Atlas montato via NFS come cold storage.
|
||||
Il piano di implementazione, con prerequisiti espliciti, e in `AGENTS.md`.
|
||||
|
||||
## Desktop
|
||||
|
||||
Target operativi:
|
||||
@@ -161,6 +182,10 @@ Le applicazioni Windows sono installate e gestite manualmente; il profilo WSL no
|
||||
|
||||
## Server
|
||||
|
||||
La migrazione dei servizi pubblici a `fscotto.co`, la gestione Ansible
|
||||
degli URL Gitea e i passaggi ancora aperti per ritirare DuckDNS sono in
|
||||
[`docs/domain-fscotto-co.md`](docs/domain-fscotto-co.md).
|
||||
|
||||
Sistema operativo:
|
||||
|
||||
- Rocky Linux 9
|
||||
@@ -180,51 +205,36 @@ Lo stato attuale del profilo server include:
|
||||
- installazione pacchetti Rocky via DNF, EPEL e CRB
|
||||
- installazione di Podman e podman-compose
|
||||
- abilitazione dei servizi systemd dichiarati in inventory/group vars
|
||||
- copia dei dotfiles server e rendering del `docker-compose.yml` per Nginx Proxy Manager e Gitea,
|
||||
piu l'unita `podman-compose-server` (attivazione manuale)
|
||||
- copia dei dotfiles server e rendering del Quadlet rootful `prometheus-npm.service` per Nginx Proxy
|
||||
Manager; il vecchio fallback Compose è stato rimosso con autorizzazione esplicita
|
||||
- attivazione di firewalld con SSH, Cockpit (`9090/tcp`), HTTP e HTTPS abilitati
|
||||
- Syncthing escluso dal profilo server Rocky
|
||||
|
||||
Il Compose desiderato su Prometheus non include piu Navidrome ne il database PostgreSQL obsoleto.
|
||||
Navidrome e Syncthing appartengono ad Atlas; Navidrome ufficiale usa invece SQLite. Il profilo non
|
||||
arresta o rimuove automaticamente eventuali container legacy e non elimina `/opt/postgres/data`.
|
||||
Il 2026-10-03 la pulizia opt-in autorizzata ha rimosso dati e immagini precedenti di Gitea,
|
||||
Navidrome e PostgreSQL, directory obsolete vuote, helper finale Gitea e fallback Compose NPM.
|
||||
I servizi migrati restano su Atlas. `server_legacy_stack_retired: true` evita che i normali task
|
||||
ricreino i residui; la cancellazione richiede `--tags server_legacy_cleanup` e
|
||||
`-e server_legacy_cleanup=true`. NPM attivo e archivi di backup restano intatti.
|
||||
Export, pull Atlas e restore SQLite isolato post-pulizia sono riusciti; il primo ciclo automatico
|
||||
resta da osservare. Evidenze e confini del recovery:
|
||||
[`docs/prometheus-npm-quadlet.md`](docs/prometheus-npm-quadlet.md).
|
||||
|
||||
Nginx Proxy Manager pubblica solo `80/tcp` e `443/tcp`; la sua interfaccia di amministrazione e
|
||||
associata a `127.0.0.1:81` ed e raggiungibile da Ikaros o Nymph con l'alias Bash `npm-tunnel`.
|
||||
Nextcloud resta disabilitato e il profilo non crea directory `/srv/nextcloud`.
|
||||
|
||||
La fase 1 su Atlas non modifica questo deployment NPM ne i suoi dati persistenti. Dopo aver attivato
|
||||
WireGuard e i servizi Atlas, configurare i proxy host NPM correnti con upstream Navidrome
|
||||
`http://10.0.0.2:4533` e upstream per la GUI Syncthing `http://10.0.0.2:8384`. Solo la GUI web di
|
||||
Syncthing usa NPM; il traffico di sincronizzazione resta sulle porte native pubblicate esplicitamente solo
|
||||
sull'indirizzo WireGuard di Atlas. Configurare l'autenticazione Syncthing e una policy di accesso NPM adeguata prima di pubblicare la GUI.
|
||||
La fase 1 su Atlas non modifica i dati persistenti NPM. I proxy host NPM usano gli upstream LAN
|
||||
`http://192.168.178.55:4533` per Navidrome e `http://192.168.178.55:8384` per la GUI Syncthing;
|
||||
Prometheus li raggiunge attraverso Aegis come gateway WireGuard. Solo la GUI web di Syncthing usa
|
||||
NPM; il traffico di sincronizzazione resta sulle porte native esposte sulla LAN dichiarata.
|
||||
Mantenere l'autenticazione Syncthing e una policy di accesso NPM adeguata.
|
||||
|
||||
### DuckDNS
|
||||
### Rimozione DuckDNS
|
||||
|
||||
`profile_server` genera `~/duckdns/duck.sh` con permessi `0700`, mantenendo il percorso dello
|
||||
script e `duck.log`. Definire `server_duckdns_domain` negli host vars del server e salvare il
|
||||
**nuovo token rigenerato** in `vault_duckdns_token`, nel Vault cifrato `secrets/vault.yml`
|
||||
(`ansible-vault edit secrets/vault.yml`) oppure negli override non versionati `secrets/vault.local.yml`.
|
||||
Non committare lo script generato e non passare il token sulla riga di comando. Il rendering
|
||||
nasconde output e diff sensibili; lo script verifica TLS e passa il token a curl tramite stdin.
|
||||
Il playbook non esegue lo script e non modifica la sua schedulazione esterna.
|
||||
|
||||
```bash
|
||||
ansible-playbook ansible/site.yml --limit prometheus --tags duckdns --check --diff
|
||||
ansible-playbook ansible/site.yml --limit prometheus --tags duckdns
|
||||
```
|
||||
|
||||
La cancellazione dalla cronologia non revoca il token: rigenerarlo sul pannello DuckDNS.
|
||||
Dopo la bonifica, riclonare gli altri checkout senza unire nuovamente la vecchia storia;
|
||||
salvare separatamente eventuali modifiche non committate senza copiare segreti.
|
||||
|
||||
### Migrazione dati
|
||||
|
||||
Dopo il provisioning Rocky, eseguire `scripts/migrate_prometheus_data.sh` **sul server Ubuntu
|
||||
sorgente**. Lo script usa rsync, e in dry-run di default; richiede `--quiesce-source --execute` per
|
||||
fermare lo stack sorgente e copiare in modo consistente soltanto i dati di Nginx Proxy Manager e
|
||||
Gitea. Non sposta Navidrome o Syncthing, non avvia container, non cancella dati e non esegue il
|
||||
cutover.
|
||||
Il supporto DuckDNS è stato rimosso dal profilo server: non restano task, template,
|
||||
variabili o flag di abilitazione. Prometheus usa IP statico e `fscotto.co`.
|
||||
Updater locale, log e cron erano già stati rimossi. Nome/account DuckDNS esterni
|
||||
ed eventuale token cifrato esistente restano invariati per un possibile uso futuro.
|
||||
|
||||
Utente del profilo server:
|
||||
|
||||
@@ -246,90 +256,282 @@ ansible-playbook ansible/site.yml --limit prometheus -e server_username=myuser -
|
||||
|
||||
## NAS
|
||||
|
||||
`atlas` e un NAS Rocky Linux 9 raggiunto tramite SSH. Normalmente il pool ZFS esiste gia e il profilo
|
||||
gestisce solo i dataset figli. Un bootstrap RAIDZ2 una tantum e disponibile solo con conferma esplicita
|
||||
(`atlas_create_pool=true`) e quattro percorsi reali e verificati `/dev/disk/by-id/...` in
|
||||
`atlas_zpool_disks`. Non partiziona, forza, distrugge, esegue rollback o modifica il layout vdev di un
|
||||
pool esistente. I client Linux usano NFSv4, quelli Windows/WSL SMB; entrambi restano limitati alla LAN
|
||||
configurata.
|
||||
`atlas` è un NAS Rocky Linux 9 raggiunto via SSH. Normalmente il pool esiste già e il profilo gestisce
|
||||
solo i dataset figli. La creazione iniziale del RAIDZ2 richiede esplicitamente `atlas_create_pool=true`
|
||||
e quattro percorsi `/dev/disk/by-id/...` verificati in `atlas_zpool_disks`. Il ruolo non partiziona,
|
||||
forza, distrugge, ripristina né modifica il layout vdev di un pool esistente. I client Linux usano NFSv4,
|
||||
quelli Windows/WSL SMB; l'accesso è limitato alla LAN configurata.
|
||||
|
||||
Per il primo avvio fornire `vault_atlas_authorized_ssh_keys`, `vault_atlas_admin_password_hash`,
|
||||
`vault_atlas_samba_password` e `vault_atlas_immich_db_password`. Eseguire il bootstrap tramite
|
||||
l'amministratore esistente:
|
||||
Per il primo avvio servono `vault_atlas_admin_password_hash`, `vault_atlas_samba_password` e
|
||||
`vault_atlas_immich_db_password`; il primo è un hash compatibile con `/etc/shadow`, non una password
|
||||
Cockpit in chiaro. Il bootstrap usa l'amministratore preesistente:
|
||||
|
||||
```bash
|
||||
ansible-playbook ansible/site.yml --limit atlas \
|
||||
-e atlas_connection_username=<existing-admin>
|
||||
```
|
||||
|
||||
`vault_atlas_admin_password_hash` deve essere un hash compatibile con `/etc/shadow`, non una
|
||||
password Cockpit in chiaro. Le esecuzioni successive usano `atlas_admin_username`. Atlas dichiara
|
||||
abilitati storage, condivisioni e regole firewall LAN. Prima della prima applicazione verificare pool e
|
||||
mountpoint esistenti, subnet LAN e zona firewalld attiva. `atlas_manage_media_stack` resta disabilitato
|
||||
finche non saranno validati `/dev/dri`, i percorsi dei container e il segreto del database Immich.
|
||||
Le esecuzioni successive usano `atlas_admin_username`. Storage, condivisioni e firewall LAN sono
|
||||
abilitati; prima dell'applicazione verificare pool, mountpoint, subnet e zona firewalld. La creazione
|
||||
del pool è protetta da un gate esplicito e avviene solo se è assente. Atlas non fa più parte della VPN
|
||||
WireGuard: la vecchia interfaccia è stata ritirata manualmente dopo la verifica del collegamento tra
|
||||
Prometheus e Aegis. Le chiavi SSH autorizzate sono in file separati sotto
|
||||
`~/.ssh/authorized_keys.d/`. `atlas_manage_media_stack` resta disabilitato finché `/dev/dri`, percorsi
|
||||
dei container e segreto del database Immich non sono validati.
|
||||
|
||||
Con la gestione storage attiva, Atlas crea l'intera gerarchia sotto il pool `zpool` esistente o creato esplicitamente:
|
||||
`work`, `archive`, `archive/app_data`, i dataset applicativi separati
|
||||
`archive/app_data/navidrome` e `archive/app_data/syncthing`, `media`, `media/music`,
|
||||
`media/photobook`, `backups`, `backups/services` e `backup_prometheus`. I dataset applicativi e
|
||||
di archivio usano `zstd`; media, Syncthing e backup dei servizi usano `lz4`;
|
||||
`backups/services` mantiene inoltre una `refreservation` di `500G`.
|
||||
Atlas impone SELinux targeted in modo persistente e segnala, senza avviarlo, l’eventuale reboot necessario per attivarlo. Assegna esplicitamente l’interfaccia LAN primaria alla zona firewalld gestita e applica hardening persistente del kernel di rete: rifiuta redirect e source-route, registra i martian, usa reverse-path filtering loose per WireGuard e disabilita il forwarding IPv4. SSH consente solo l’amministratore dichiarato tramite chiave pubblica; root, password, agent e forwarding
|
||||
remoto sono disabilitati, mentre il forwarding locale resta disponibile per tunnel amministrativi privati. SMB3 pubblica `Archive` solo agli account Samba configurati con password in Vault e
|
||||
ammette la LAN configurata su SMB3 cifrato e firmato, esclusivamente su TCP/445. NFSv4 esporta soltanto
|
||||
`media/photobook` all'IP configurato di Aegis su TCP/2049, con `all_squash` verso UID/GID anonimi `1100`.
|
||||
Sotto `zpool` Atlas crea `archive` (SMB), `services/data` con i dataset applicativi
|
||||
`services/data/navidrome` e `services/data/syncthing`, `media`, `media/music`, `media/photobook` e
|
||||
`backup/hosts/prometheus`. Archivio e applicazioni usano `zstd`; media, Syncthing e backup host usano
|
||||
`lz4`. `backup` ha una riserva di `500G` che copre i discendenti. SELinux targeted è persistente;
|
||||
l'eventuale riavvio necessario viene segnalato, non eseguito. Atlas assegna l'interfaccia primaria
|
||||
alla zona firewalld gestita, rifiuta redirect e source route, registra i martian, mantiene il reverse-path
|
||||
filter loose e disabilita il forwarding IPv4. SSH consente soltanto l'amministratore dichiarato con
|
||||
chiave pubblica: root, password, agent forwarding e remote forwarding sono disabilitati, mentre il
|
||||
forwarding locale resta disponibile per i tunnel amministrativi. SMB3 espone `Archive` agli account
|
||||
autorizzati da Vault sulla LAN, solo su TCP/445 con cifratura e firma obbligatorie. NFSv4 espone
|
||||
soltanto `media/photobook` all'IP di Aegis su TCP/2049, con `all_squash` verso UID/GID `1100`.
|
||||
|
||||
L'account di sistema `immich` usa UID/GID `1100`, shell senza login, nessuna appartenenza a `wheel` e
|
||||
i gruppi supplementari `video` e `render`. I Quadlet rootful di Immich Server, ML, cache compatibile
|
||||
Redis, PostgreSQL e NPM condividono una rete Podman. Immich viene eseguito come `1100:1100`; Server e
|
||||
ML ricevono `/dev/dri` e Photobook e montato in sola lettura su `/external/photobook`. NPM pubblica `80` e
|
||||
`443`, mentre l'amministrazione resta vincolata a `127.0.0.1:81` per l'accesso tramite tunnel SSH.
|
||||
L'account di sistema `immich` usa UID/GID `1100`, non ha shell di login né gruppo `wheel` e riceve i
|
||||
gruppi `video` e `render`. Lo stack Immich futuro prevede Quadlet rootful per Server, ML, cache,
|
||||
PostgreSQL e NPM su una rete Podman comune. Immich gira come `1100:1100`, Server e ML ricevono
|
||||
`/dev/dri` e Photobook è montato in sola lettura su `/external/photobook`. NPM pubblica `80` e `443`;
|
||||
l'interfaccia amministrativa resta su `127.0.0.1:81`, raggiungibile via tunnel SSH.
|
||||
|
||||
La fase 1 e limitata ai Quadlet utente rootless di Navidrome e Syncthing su Atlas. E abilitata nella
|
||||
configurazione host di Atlas e puo essere impostata a `false` solo per una sospensione intenzionale. Navidrome ufficiale `0.63.2` usa il database SQLite sotto `/data` e
|
||||
non supporta `ND_DATABASE_URL` ne un backend PostgreSQL esterno. Il servizio obsoleto `navidromedb`
|
||||
e quindi rimosso da Prometheus invece di essere replicato su Atlas. Il ruolo deriva i percorsi dal
|
||||
pool `zpool`, montato in `/zpool`: musica in sola lettura da `/zpool/media/music`, stato
|
||||
applicativo Navidrome e `navidrome.db` in `/zpool/archive/app_data/navidrome` e dati Syncthing in
|
||||
`/zpool/archive/app_data/syncthing`. `profile_atlas` crea questi dataset quando
|
||||
`atlas_manage_storage` e attivo; il ruolo backend verifica i mountpoint esatti prima di avviare i
|
||||
container. Il ruolo backend non crea mai il pool. Il ruolo separato `wireguard_overlay`
|
||||
gestisce `wg0` tra Prometheus (`10.0.0.1`) e Atlas (`10.0.0.2`), genera una sola volta le chiavi
|
||||
private sui rispettivi host e scambia tramite Ansible soltanto quelle pubbliche. Solo Prometheus apre
|
||||
pubblicamente `51820/udp`. Le porte backend sono ammesse esclusivamente nella zona firewalld WireGuard.
|
||||
Atlas ospita temporaneamente Navidrome e Syncthing rootless fino alla sostituzione con Uranus. I
|
||||
servizi sono inizializzati **ex novo**, senza migrare lo stato precedente, rispettivamente sotto
|
||||
`/zpool/services/data/navidrome` e `/zpool/services/data/syncthing`; la musica in
|
||||
`/zpool/media/music` è stata popolata separatamente da `/zpool/archive/Music` il 2026-09-30;
|
||||
Navidrome ha completato la scansione. Il timer rootless `atlas-music-sync.timer` copia i file nuovi
|
||||
o modificati ogni giorno alle 00:45 Europe/Rome, senza eliminare quelli presenti solo nella
|
||||
destinazione; entrambi i dataset ZFS devono essere montati. La prima esecuzione schedulata è
|
||||
riuscita il 2026-10-02. Alcune playlist originali contengono
|
||||
ancora vecchi percorsi Windows. I servizi sono vincolati all'indirizzo LAN di Atlas
|
||||
(`192.168.178.55`), mai a WireGuard. `wireguard_overlay` collega invece Prometheus (`10.0.0.1`)
|
||||
e Aegis (`10.0.0.2`): le chiavi private restano sui rispettivi host e Ansible scambia solo le pubbliche.
|
||||
Prometheus apre `51820/udp`; Aegis inoltra soltanto il traffico overlay→LAN dichiarato e applica
|
||||
source NAT, evitando interfacce VPN su Atlas/Uranus e route statiche sul router. Navidrome (`4533/tcp`)
|
||||
e la GUI Syncthing (`8384/tcp`) ammettono solo Aegis, mentre le porte native Syncthing sono limitate
|
||||
alla LAN. Dopo la verifica dei servizi, configurare manualmente i Proxy Host NPM verso
|
||||
`http://192.168.178.55:4533` e `http://192.168.178.55:8384`. Il peer Prometheus include la LAN
|
||||
negli `AllowedIPs`; aggiungere la VIP Uranus quando esisterà. Dopo il reload di firewalld, Ansible
|
||||
ricarica le reti Podman rootful di Prometheus per conservare DNS e connettività del proxy.
|
||||
|
||||
`backend_phase1_start_services` resta falso durante il trasferimento dello stato applicativo, quindi
|
||||
la prima esecuzione reale del backend genera i Quadlet senza creare un database Atlas vuoto. Dopo aver
|
||||
arrestato Navidrome su Prometheus, copiare l'intera directory `/opt/navidrome/data/` in
|
||||
`/zpool/archive/app_data/navidrome/`, preservando `navidrome.db` e gli eventuali file SQLite laterali.
|
||||
Impostare quindi questa variabile a vero e rieseguire il ruolo per abilitare e avviare Navidrome e
|
||||
Syncthing. Il playbook non copia e non elimina mai i dati applicativi.
|
||||
La migrazione Gitea da Prometheus ad Atlas è descritta in
|
||||
[`docs/atlas-gitea-migration.md`](docs/atlas-gitea-migration.md). Gitea usa un Quadlet rootless
|
||||
di `admin` su un dataset dedicato; l'immagine derivata mantiene UID/GID 1000 ma chiama l'utente
|
||||
interno `gitea`. NPM resta su Prometheus e l'HTTPS pubblico primario serve Atlas. L'SSH pubblico
|
||||
su TCP/2222 autentica la chiave `ikaros` e un `git ls-remote` è riuscito; l'operatore ha
|
||||
confermato pull e push SSH. Login e scrittura Git via HTTPS sono stati confermati il 2026-10-03. I dati sorgente restano
|
||||
conservati su Prometheus senza avviarne il vecchio container.
|
||||
|
||||
Validare e generare i servizi Atlas con:
|
||||
Validare il gateway con:
|
||||
|
||||
```bash
|
||||
ANSIBLE_LOCAL_TEMP=/tmp/ansible-local \
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags storage
|
||||
|
||||
ANSIBLE_LOCAL_TEMP=/tmp/ansible-local \
|
||||
ansible-playbook ansible/site.yml --limit prometheus,atlas --tags wireguard
|
||||
|
||||
ANSIBLE_LOCAL_TEMP=/tmp/ansible-local \
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags backend_phase1 --check --diff
|
||||
|
||||
ANSIBLE_LOCAL_TEMP=/tmp/ansible-local \
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags backend_phase1
|
||||
ansible-playbook ansible/site.yml --limit prometheus,aegis --tags wireguard --check --diff
|
||||
```
|
||||
|
||||
Per il cutover, arrestare il vecchio Navidrome prima di copiare la sua directory dati, verificare
|
||||
l'ownership dell'account `admin` su Atlas e confermare la presenza del database SQLite copiato prima
|
||||
di impostare `backend_phase1_start_services: true` in `host_vars/atlas.yml`. Conservare i dati sorgente
|
||||
e il container legacy `navidromedb` fermo finche Navidrome su Atlas e una prova di restore non sono
|
||||
stati validati.
|
||||
La prima esecuzione reale WireGuard deve includere entrambi i peer. Se Aegis ha appena installato il
|
||||
layer `wireguard-tools`, riavviarlo manualmente e rieseguire senza `--check`: il ruolo attende un
|
||||
handshake effettivo.
|
||||
|
||||
Restano da completare retention delle snapshot, topologia Syncthing, validazione WireGuard/firewall,
|
||||
pull di backup da Prometheus, backup cifrati con Borg su una Hetzner Storage Box, backup USB,
|
||||
monitoraggio e test di disaster recovery. Il backlog operativo dettagliato e in `AGENTS.md`.
|
||||
Gli snapshot ZFS ricorsivi coprono l'intero pool: 24 orari al minuto 05, 30 giornalieri alle 00:15,
|
||||
8 settimanali la domenica alle 01:00 e 12 mensili il primo giorno alle 02:00. La retention elimina
|
||||
solo gli snapshot con prefisso gestito `atlas-auto` e non esegue rollback. Lo scrub OpenZFS mensile è
|
||||
previsto la prima domenica alle 03:00; il timer settimanale incompatibile è disabilitato. Il primo
|
||||
snapshot orario ricorsivo è riuscito e la pulizia pianificata della retention è stata osservata il
|
||||
2026-09-30. Il primo scrub mensile richiede ancora una verifica a runtime.
|
||||
|
||||
```bash
|
||||
ANSIBLE_LOCAL_TEMP=/tmp/ansible-local \
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags snapshots,scrub --check --diff
|
||||
```
|
||||
|
||||
Il backup Borg cifrato usa il sub-account Hetzner `u660064-sub1`, il repository relativo `./borg-data`
|
||||
e Borg remoto 1.4 su SSH porta 23. La chiave ED25519 del server è fissata; una chiave client dedicata
|
||||
appartiene all'account `borg`, bloccato e senza login, sudo o gruppi supplementari. La chiave privata
|
||||
resta in `/etc/atlas-borg`; la passphrase proviene da `vault_atlas_borg_passphrase` ed è resa in un
|
||||
file `0600`. Solo il wrapper root crea snapshot e mount; avvia il client come `borg` con il minimo
|
||||
accesso temporaneo in lettura, senza concedergli gestione ZFS o sudo.
|
||||
|
||||
Il backup giornaliero parte alle 04:30 con un ritardo casuale fino a 30 minuti. Crea uno snapshot ZFS
|
||||
ricorsivo temporaneo e ricostruisce tutti i dataset sotto `/zpool` in un albero di bind mount in sola
|
||||
lettura, per inserirli in un unico archivio coerente. Il wrapper smonta ricorsivamente l'albero privato;
|
||||
un helper `ExecStopPost` mirato rimuove eventuali mount dello snapshot nel namespace host e lo snapshot
|
||||
temporaneo dopo l'uscita del processo. Borg conserva 30 archivi giornalieri, 8 settimanali e 12
|
||||
mensili, poi compatta il repository. Il controllo completo di metadati e repository si svolge il 15
|
||||
di ogni mese alle 06:00. Le operazioni usano un lock comune, journal e retry systemd limitati. Le
|
||||
nuove esecuzioni riportano al massimo una riga di avanzamento al minuto: percentuale **stimata**,
|
||||
dataset, file elaborati e byte originali/compressi/deduplicati. Il denominatore è la somma dei
|
||||
`logicalreferenced` ZFS dello snapshot, non un totale Borg: può superare il 100% e non comprende
|
||||
retention, compattazione o controlli. Le righe di progresso non riportano i nomi dei file; eventuali
|
||||
warning possono farlo. Seguire il job con `sudo journalctl -fu atlas-borg-backup.service`; modifiche
|
||||
all'helper non cambiano un'esecuzione già avviata.
|
||||
|
||||
Attivazione iniziale esplicita:
|
||||
|
||||
1. Inserire una passphrase unica in `secrets/vault.yml` con `ansible-vault edit`.
|
||||
2. Generare e mostrare solo la chiave pubblica con
|
||||
`ansible-playbook ansible/site.yml --limit atlas --tags borg_key`.
|
||||
3. Installarla nel sub-account Hetzner, poi applicare con
|
||||
`ansible-playbook ansible/site.yml --limit atlas --tags packages,borg`.
|
||||
4. Copiare `secrets/recovery/atlas-borg-repokey.export` su un supporto davvero offline: la copia
|
||||
locale ignorata da Git non è di per sé un backup offline.
|
||||
|
||||
Il ruolo inizializza solo un repository `repokey` assente, non accetta password SSH né host key non
|
||||
fissate e non avvia manualmente il primo backup. Validazione:
|
||||
|
||||
```bash
|
||||
ANSIBLE_LOCAL_TEMP=/tmp/ansible-local \
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags packages,borg --check --diff
|
||||
```
|
||||
|
||||
L'attivazione iniziale è riuscita: backup e controllo del repository, restore completo in una
|
||||
directory temporanea confrontato con l'albero `Archive`, esportazione offline della chiave di recupero
|
||||
e pulizia di snapshot/mount temporanei. Il 2026-09-25 un test separato da snapshot ZFS giornaliero ha
|
||||
copiato un file di `/zpool/archive` in `/var/tmp`, verificando contenuto, proprietario, modalità,
|
||||
mtime e ACL POSIX; copia e mount temporanei sono stati rimossi senza interrompere Borg. Non è un test
|
||||
di ripristino dell'intero dataset.
|
||||
|
||||
L'archivio del pool popolato del 2026-09-29 ha richiesto 1 h 32 min per 2,18 TB originali / 2,04 TB
|
||||
compressi, con 13,49 GB di dimensione deduplicata. Retention e compattazione sono riuscite, ma un
|
||||
errore di permessi su `RuntimeDirectory` ha impedito la pulizia dello snapshot dopo il job. Dopo la
|
||||
correzione, l'archivio incrementale del 2026-09-30 è terminato in circa 22 secondi, ha rimosso lo
|
||||
snapshot residuo e quello corrente ed è terminato con stato 0. Il monitor ha rilevato il 37% della
|
||||
quota Storage Box utilizzata. Questi risultati non predicono durata o compressione dei prossimi run.
|
||||
|
||||
Il backup USB offline è distribuito come **servizio solo manuale** (`atlas_manage_usb_backup: true`):
|
||||
Ansible non formatta, sblocca, monta né avvia automaticamente il disco. Il disco esistente è stato
|
||||
verificato in sola lettura il 2026-09-23: UUID LUKS `577b3c43-ea37-4611-81a9-39d555cdfbd4`,
|
||||
UUID ext4 interno `758e2d2e-a427-4797-aad9-39c3a9f17c7e`, mapper `zpool-backup`. All'ispezione
|
||||
era montato in `/mnt/zpool-backup`; il servizio richiede invece che il mapper **non sia montato** prima
|
||||
dell'avvio. Se serve, `systemd-ask-password` chiede interattivamente la passphrase LUKS tramite
|
||||
l'agente di `systemctl start` e la passa direttamente a `cryptsetup`, senza salvarla, esporla negli
|
||||
argomenti o memorizzarla nella cache. Lo script monta il disco privatamente, crea uno snapshot ZFS
|
||||
ricorsivo, copia tutti i dataset in `atlas/snapshots/<timestamp>/` con `rsync --link-dest`, verifica
|
||||
con un dry-run basato sui checksum, aggiorna atomicamente `atlas/latest`, smonta e chiude LUKS. Un
|
||||
errore non sostituisce `latest` né cancella versioni complete precedenti. Borg e USB possono operare
|
||||
contemporaneamente su snapshot distinti, ma la lettura concorrente può ridurre il throughput.
|
||||
|
||||
La copia USB conserva le ACL ma non gli attributi estesi generici, compreso `security.selinux`: la
|
||||
policy della destinazione deve ricreare le etichette dopo un restore. Per un percorso esplicito:
|
||||
|
||||
```bash
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags restorecon \
|
||||
-e '{"atlas_restorecon_paths":["/zpool/archive"]}'
|
||||
```
|
||||
|
||||
Il task accetta solo percorsi sotto la radice del pool Atlas, esegue `restorecon -RFv` solo su quelli
|
||||
indicati ed è altrimenti inattivo; non va lanciato sull'intero pool durante i run ordinari. Le vecchie
|
||||
versioni USB non vengono eliminate automaticamente senza una retention deliberata. Il controllo di
|
||||
capacità include il trasferimento stimato e una riserva libera di 10 GiB. Dopo un backup riuscito,
|
||||
scollegare fisicamente il disco per renderlo davvero offline.
|
||||
|
||||
Validare la configurazione senza avviare il backup e, separatamente, un eventuale relabel pianificato:
|
||||
|
||||
```bash
|
||||
ANSIBLE_LOCAL_TEMP=/tmp/ansible-local \
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags usb_backup,usb_reminder --check --diff
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags restorecon --check \
|
||||
-e '{"atlas_restorecon_paths":["/zpool/archive"]}'
|
||||
```
|
||||
|
||||
Prima dell'avvio manuale smontare in sicurezza `/mnt/zpool-backup`, se ancora montato. Con il mapper
|
||||
chiuso, `sudo systemctl start atlas-usb-backup.service` chiede la passphrase e avvia il backup; né la
|
||||
password LUKS né un keyfile vanno in Ansible. Seguire con
|
||||
`sudo journalctl -fu atlas-usb-backup.service`. **Non esiste un timer di backup USB.** Soltanto
|
||||
`atlas-usb-reminder.timer` è schedulato il primo sabato del mese alle 10:00 `Europe/Rome`: invia un
|
||||
promemoria al notifier 45Drives Houston, senza avviare il backup. Un test manuale ha prodotto una
|
||||
notifica in 45Drives Alerts, **non un'email**; il log conferma l'invio della notifica, non la consegna
|
||||
di posta. Il primo evento pianificato era il 2026-10-03 alle 10:00 CEST. Controllare timer e risultato
|
||||
con `systemctl list-timers atlas-usb-reminder.timer` e in 45Drives Alerts.
|
||||
|
||||
Il primo tentativo USB del 2026-09-23 fallì su `security.selinux` e, dopo l'interruzione, lasciò
|
||||
snapshot e mapper aperti. Applicato il filtro rsync, furono rimossi lo snapshot fallito, il mapper
|
||||
smontato e lo stato failed; non rimase una copia valida di quel tentativo. Un run del 2026-09-24
|
||||
pubblicò una versione verificata ma fallì nella distruzione dello snapshot a causa di mount
|
||||
`.zfs/snapshot` aperti nel namespace host. Dopo la pulizia non forzata, è stato aggiunto un helper
|
||||
`ExecStopPost` mirato e testato con uno snapshot usa-e-getta. Un run successivo del 2026-09-24 ha
|
||||
verificato i checksum, pubblicato la versione ed è terminato con successo: mapper chiuso, nessuno
|
||||
snapshot USB temporaneo e pool sano. Il 2026-09-25 un test di restore indipendente ha aperto il disco
|
||||
in sola lettura, montato ext4 con `ro,noload`, copiato un file di 5.707.945 byte da `atlas/latest` in
|
||||
una directory vuota sotto `/var/tmp` e confrontato contenuto, proprietario, modalità, dimensione,
|
||||
mtime e ACL POSIX. Il test ha rimosso copia e mount temporanei, chiuso LUKS e lasciato il pool sano
|
||||
mentre Borg continuava. È un test su file, non un esercizio completo di disaster recovery.
|
||||
|
||||
Il monitoraggio Atlas è eseguito ogni 30 minuti da `atlas-health-monitor.timer`. Sonde in sola
|
||||
lettura controllano stato/errori del pool e dei vdev, scrub/resilver, SMART dei quattro dischi del
|
||||
pool e dell'NVMe di sistema, temperature dei dischi e CPU, spazio di sistema/pool/snapshot, crescita
|
||||
di `zpool/backup` e quota Hetzner tramite `df -m` via SSH con l'account `borg` e la chiave fissata.
|
||||
La query remota non apre il repository Borg né il suo lock. Gli alert di crescita richiedono una
|
||||
baseline di circa 24 ore. Sono controllati anche attivazione e freschezza dei timer; hook systemd
|
||||
`OnFailure` segnalano errori di snapshot, scrub, Borg, USB, promemoria e monitoraggio. Il monitor non
|
||||
riavvia Borg; avvisa solo se un run supera 14 giorni. Soglie e percorsi stabili dei dischi sono nelle
|
||||
variabili host. Gli avvisi usano 45Drives Houston con deduplicazione; **la consegna email non è stata
|
||||
verificata**. Il controllo live del 2026-09-25 non ha trovato problemi e ha inviato una notifica di
|
||||
prova. Il 2026-09-30 il monitor ha rilevato zero problemi e una quota Storage Box occupata al 37%.
|
||||
L'hook per i job falliti ora passa il nome letterale della unità systemd; l'espansione è stata
|
||||
verificata senza inviare un falso allarme.
|
||||
|
||||
```bash
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags monitoring --check --diff
|
||||
sudo /usr/local/libexec/atlas-health-monitor --dry-run
|
||||
sudo journalctl -u atlas-health-monitor.service -n 100 --no-pager
|
||||
systemctl list-timers atlas-health-monitor.timer
|
||||
```
|
||||
|
||||
`--dry-run` non invia alert e non modifica lo stato del monitor. Un controllo reale si avvia con
|
||||
`sudo systemctl start atlas-health-monitor.service`, senza avviare servizi di backup. Per una prova
|
||||
etichettata di 45Drives Alerts usare
|
||||
`sudo /usr/local/libexec/atlas-health-monitor --test-notification`.
|
||||
|
||||
### Timer systemd di Atlas
|
||||
|
||||
Tutti i dieci timer gestiti sono abilitati. Gli orari sono locali ad Atlas (`Europe/Rome`); Borg e
|
||||
monitoraggio aggiungono il ritardo casuale indicato. Tutti hanno `Persistent=true`: un evento perso
|
||||
viene recuperato quando il timer torna attivo.
|
||||
|
||||
| Timer | Pianificazione (`OnCalendar`) | Azione |
|
||||
| --- | --- | --- |
|
||||
| `atlas-zfs-snapshot-hourly.timer` | `*-*-* *:05:00` — ogni ora al minuto 05 | Snapshot ricorsivo orario e retention |
|
||||
| `atlas-zfs-snapshot-daily.timer` | `*-*-* 00:15:00` — ogni giorno alle 00:15 | Snapshot ricorsivo giornaliero e retention |
|
||||
| `atlas-zfs-snapshot-weekly.timer` | `Sun *-*-* 01:00:00` — domenica alle 01:00 | Snapshot ricorsivo settimanale e retention |
|
||||
| `atlas-zfs-snapshot-monthly.timer` | `*-*-01 02:00:00` — primo giorno del mese alle 02:00 | Snapshot ricorsivo mensile e retention |
|
||||
| `zfs-scrub-monthly@zpool.timer` | `Sun *-*-01..07 03:00:00` — prima domenica alle 03:00 | Scrub ZFS |
|
||||
| `atlas-borg-backup.timer` | `*-*-* 04:30:00` — ogni giorno alle 04:30, più 0–30 min casuali | Backup cifrato offsite |
|
||||
| `atlas-borg-check.timer` | `*-*-15 06:00:00` — giorno 15 alle 06:00, più 0–30 min casuali | Controllo repository Borg |
|
||||
| `atlas-usb-reminder.timer` | `Sat *-*-01..07 10:00:00 Europe/Rome` — primo sabato alle 10:00 | Solo promemoria 45Drives Alerts |
|
||||
| `atlas-health-monitor.timer` | `*:0/30` — ogni mezz'ora, più 0–5 min casuali | Controlli di salute in sola lettura |
|
||||
| `atlas-prometheus-pull.timer` | `*-*-* 03:00:00 Europe/Rome` — ogni giorno alle 03:00 | Pull e verifica del backup preparato su Prometheus |
|
||||
|
||||
`atlas-usb-backup.service` **non ha timer** e va avviato manualmente. Il timer del fornitore
|
||||
`zfs-scrub-weekly@zpool.timer` è disabilitato a favore dello scrub mensile. Il timer di preparazione
|
||||
su Prometheus è attivo alle 02:00 Europe/Rome; il primo ciclo pianificato è riuscito il 2026-10-01.
|
||||
Un export, pull e ripristino temporaneo post-cutover NPM Quadlet sono riusciti il 2026-10-03;
|
||||
il primo ciclo pianificato dopo quel cutover resta da osservare. Durante un backup Borg attivo,
|
||||
`systemctl list-timers` può mostrare `-` per il prossimo evento senza che il timer sia disabilitato.
|
||||
Per vedere la pianificazione corrente: `systemctl list-timers --all` su Atlas.
|
||||
|
||||
Nextcloud è previsto come servizio temporaneo su Atlas prima di Uranus, ma solo dopo la validazione
|
||||
della protezione dei dati: richiede storage applicativo, database e cache separati, segreti Vault,
|
||||
pubblicazione solo tramite NPM e Aegis, procedure di backup, aggiornamento e migrazione. Non
|
||||
distribuirlo prima di completare la checklist di protezione dei dati.
|
||||
|
||||
Atlas è la destinazione dichiarata per iCloudPD. Ansible gestisce dataset, Quadlet rootless e
|
||||
`icloudpd.conf` privato con Apple ID dal Vault: foto in `/zpool/archive/Pictures/iCloudPD`,
|
||||
stato in `zpool/services/data/icloudpd`. Il primo avvio è stato manuale; password e MFA restano
|
||||
gestiti interattivamente, senza avvio automatico al boot. L'inizializzazione è stata completata e
|
||||
il download iniziale di foto e video è terminato il 2026-10-03. Su Aegis
|
||||
il servizio, il Quadlet e `/var/lib/icloudpd` sono stati rimossi e verificati; il playbook Aegis
|
||||
non contiene più task iCloudPD. L'accesso SMB e il ripristino dai backup dei nuovi dati restano
|
||||
da verificare. L'export NFS Photobook resta
|
||||
invariato. Dettagli in [`docs/atlas-icloudpd-migration.md`](docs/atlas-icloudpd-migration.md).
|
||||
|
||||
Il primo ciclo pianificato del backup di Prometheus e una prova di disaster recovery a dimensione reale
|
||||
restano da verificare. Il 2026-09-30 una VM Rocky isolata ha superato ricostruzione OS con Ansible,
|
||||
import del pool RAIDZ2 fittizio e ripristino da snapshot; RPO 24 ore/RTO 72 ore restano obiettivi
|
||||
provvisori, non tempi misurati. Dettagli e limiti sono in `docs/atlas-recovery.md`. Il backlog
|
||||
prioritizzato è in `AGENTS.md`.
|
||||
|
||||
---
|
||||
|
||||
@@ -426,8 +628,8 @@ Questo significa che, allo stato attuale:
|
||||
- `deadalus` riceve il profilo Fedora WSL tramite play dev dedicati
|
||||
- il server Rocky (`prometheus`) e gestito con pacchetti, servizi, dotfiles server e firewalld
|
||||
- il NAS Rocky (`atlas`) usa un pool ZFS gia esistente, condivisioni NFSv4/SMB limitate alla LAN e Cockpit/45Drives
|
||||
- lo stack Compose server include soltanto `gitea` e `nginx-proxy-manager`; Navidrome e Syncthing
|
||||
della fase 1 sono Quadlet rootless su Atlas
|
||||
- NPM è un Quadlet rootful su Prometheus, mentre Gitea, Navidrome e Syncthing sono Quadlet
|
||||
rootless su Atlas; il fallback Compose server è stato rimosso
|
||||
|
||||
# Dotfiles
|
||||
|
||||
@@ -533,9 +735,10 @@ ansible-playbook ansible/site.yml --limit <host> --tags <tag1>,<tag2> --check --
|
||||
ansible-playbook ansible/site.yml --limit <host> --start-at-task "<task name>" --check --diff
|
||||
ansible-lint ansible/roles/<role>
|
||||
yamllint ansible/path/to/file.yml
|
||||
podman-compose -f /opt/docker/server/docker-compose.yml config
|
||||
ansible-playbook ansible/site.yml --limit prometheus --tags server_legacy_cleanup --check --diff -e server_legacy_cleanup=true
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags storage,sharing,containers --check --diff
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags backend_phase1 --check --diff
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags music_sync --check --diff
|
||||
```
|
||||
|
||||
## Tag supportati dal playbook
|
||||
|
||||
306
README.md
306
README.md
@@ -67,6 +67,27 @@ The official ChatGPT desktop RPM is enabled only on `ikaros` and `nymph`. The
|
||||
playbook configures OpenAI's signed RPM repository and imports its pinned RPM
|
||||
signing key before installation; subsequent updates are handled by DNF.
|
||||
|
||||
## Deferred planned node: Cerberus
|
||||
|
||||
`cerberus` is a **postponed** management-plane node, pending the physical setup
|
||||
of the office in the new house. It is not yet an inventory host and no role or
|
||||
playbook targets it.
|
||||
|
||||
The planned hardware is a Lenovo ThinkCentre M700 Tiny (Intel Core i3-6100T,
|
||||
8 GB RAM, and a 256 GB SSD) with native 1 Gbps Ethernet. It will share Ikaros'
|
||||
monitor and peripherals through a multi-input KVM switch, using a passive
|
||||
DisplayPort-to-HDMI cable for its video connection. Fedora Sericea, the
|
||||
immutable Fedora variant with the Sway Wayland compositor, is the intended
|
||||
operating system.
|
||||
|
||||
Cerberus will be an isolated management plane: Ansible will run from a
|
||||
dedicated Toolbox environment to provision the future `uranus` cluster, rather
|
||||
than from Ikaros or an unmanaged host. Its rootless Podman observability stack
|
||||
will run Grafana, Prometheus, and Loki. The local SSD is the hot tier and
|
||||
retains metrics and logs for 30 days; scheduled exports will place older
|
||||
historical data on an NFS-mounted Atlas dataset as the cold tier. The detailed,
|
||||
implementation-gated plan is maintained in `AGENTS.md`.
|
||||
|
||||
## Desktop profiles
|
||||
|
||||
- `ikaros`: stable Fedora Workstation + GNOME desktop.
|
||||
@@ -104,16 +125,26 @@ That gives it Fedora packages through DNF, Docker from the official repository,
|
||||
|
||||
## Server
|
||||
|
||||
The public service domain transition to `fscotto.co`, Gitea canonical URL
|
||||
management, and remaining DuckDNS retirement steps are documented in
|
||||
[`docs/domain-fscotto-co.md`](docs/domain-fscotto-co.md).
|
||||
|
||||
`prometheus` is the Rocky Linux 9 server. It has no graphical environment and gets server-specific
|
||||
dotfiles and templates. The profile provisions configuration only: it does not transfer data, start
|
||||
the Compose stack, update DNS, or perform a cutover.
|
||||
dotfiles and templates. The profile does not transfer application data, update DNS, or perform an
|
||||
implicit service cutover.
|
||||
|
||||
The server profile installs platform-specific packages, Podman and podman-compose, declared systemd
|
||||
services, and firewalld. The manually activated `podman-compose-server` unit contains the existing
|
||||
Nginx Proxy Manager and Gitea services. The desired Compose file no longer includes Navidrome,
|
||||
Syncthing, or the obsolete Navidrome PostgreSQL database; their temporary Atlas deployment is managed
|
||||
by `profile_backend_phase1`. Applying the profile does not stop or remove legacy containers and does
|
||||
not delete `/opt/postgres/data`.
|
||||
services, and firewalld. Nginx Proxy Manager runs as the rootful `prometheus-npm.service` Quadlet.
|
||||
On 2026-10-03 the operator-approved opt-in cleanup removed old Gitea, Navidrome and PostgreSQL
|
||||
data/images, empty legacy directories, the Gitea final-export helper and the Compose rollback files.
|
||||
The migrated services stay on Atlas. `server_legacy_stack_retired: true` prevents normal runs from
|
||||
recreating retired files. Data deletion requires `--tags server_legacy_cleanup` and
|
||||
`-e server_legacy_cleanup=true`; image-only cleanup has its own `server_image_cleanup` tag and flag.
|
||||
Active NPM resources and existing backup archives remain preserved.
|
||||
The post-cleanup export/pull and isolated SQLite restore passed; the first unattended cycle remains
|
||||
pending. Evidence and recovery boundaries:
|
||||
[`docs/prometheus-npm-quadlet.md`](docs/prometheus-npm-quadlet.md).
|
||||
|
||||
|
||||
Firewalld enables SSH, Cockpit (`9090/tcp`), HTTP and HTTPS. Nginx Proxy Manager publishes only
|
||||
`80/tcp` and `443/tcp`; its administration interface is bound to `127.0.0.1:81` and can be reached
|
||||
@@ -138,45 +169,12 @@ The target must already provide `server_username` with local sudo access.
|
||||
Prometheus authorizes its declared SSH public keys through separate files below
|
||||
`~/.ssh/authorized_keys.d/`, while `sshd` is configured to read those files directly.
|
||||
|
||||
### DuckDNS
|
||||
### DuckDNS retirement
|
||||
|
||||
`profile_server` renders `~/duckdns/duck.sh` with mode `0700`, keeping the existing updater path
|
||||
and `duck.log`. Set `server_duckdns_domain` in the server's host vars and store the **rotated**
|
||||
`vault_duckdns_token` in encrypted `secrets/vault.yml` (using `ansible-vault edit secrets/vault.yml`)
|
||||
or untracked `secrets/vault.local.yml`. Never commit the rendered script or put the token on a
|
||||
command line. Rendering hides secret output/diffs; the updater verifies TLS and passes the token
|
||||
to curl through stdin. The playbook neither runs the updater nor changes its external schedule.
|
||||
|
||||
```bash
|
||||
ansible-playbook ansible/site.yml --limit prometheus --tags duckdns --check --diff
|
||||
ansible-playbook ansible/site.yml --limit prometheus --tags duckdns
|
||||
```
|
||||
|
||||
An exposed token must be revoked/regenerated on DuckDNS: deleting it from Git history does not
|
||||
revoke it. After a history cleanup, re-clone other checkouts rather than merging the old history
|
||||
back in; preserve any uncommitted work separately without copying secrets.
|
||||
|
||||
### Data migration
|
||||
|
||||
Provision Rocky first, then run the migration script **on the retired Ubuntu source host**. It is
|
||||
dry-run by default and requires an explicit source-stack stop before it can copy application data:
|
||||
|
||||
```bash
|
||||
sudo ./scripts/migrate_prometheus_data.sh \
|
||||
--destination rocky@179.237.102.172 \
|
||||
--identity /root/.ssh/id_ed25519
|
||||
|
||||
sudo ./scripts/migrate_prometheus_data.sh \
|
||||
--destination rocky@179.237.102.172 \
|
||||
--identity /root/.ssh/id_ed25519 \
|
||||
--quiesce-source --execute
|
||||
```
|
||||
|
||||
The script copies only Nginx Proxy Manager and Gitea data. It does not delete data, move
|
||||
Navidrome/Syncthing, copy `/home/git/.ssh`, start containers, update DNS, or perform a cutover. The
|
||||
destination SSH host key must already be trusted and the destination account needs passwordless sudo
|
||||
for `rsync`. It preserves ACLs but not extended attributes, so source SELinux labels are not
|
||||
transferred; the Rocky Compose bind mounts apply their own `:Z` labels when containers start.
|
||||
DuckDNS support has been removed from the server profile: no tasks, templates,
|
||||
variables or enablement flags remain. Prometheus uses its static IP and `fscotto.co`.
|
||||
The local updater, log and cron job were already removed. The external DuckDNS
|
||||
name/account and existing encrypted token remain untouched for possible future use.
|
||||
|
||||
## DNS Filter
|
||||
|
||||
@@ -189,8 +187,8 @@ ansible/bootstrap/generate-aegis-ign.sh --write IMAGE DEVICE
|
||||
```
|
||||
|
||||
The controller manages it remotely as `pi@aegis`; unlike local desktop profiles, Aegis is
|
||||
intentionally an SSH inventory target. `profile_aegis` manages rootful Podman Quadlets for AdGuard
|
||||
Home and iCloudPD, persistent data under `/var/lib`, the Podman auto-update timer, LAN-restricted
|
||||
intentionally an SSH inventory target. `profile_aegis` manages a rootful Podman Quadlet for AdGuard
|
||||
Home, its persistent data under `/var/lib`, the Podman auto-update timer, LAN-restricted
|
||||
firewalld rules, SSH key-only access for `pi`, the `nfs-utils` and `wireguard-tools` rpm-ostree layers,
|
||||
and `wake-ikaros`. `wireguard_overlay` makes Aegis the internal endpoint and LAN gateway for Prometheus:
|
||||
it enables persistent IPv4 forwarding, installs a scoped WireGuard-to-LAN firewalld policy, and source-NATs
|
||||
@@ -203,9 +201,8 @@ opened and closed manually during initial setup. The profile disables the local
|
||||
stub and points `/etc/resolv.conf` to its full resolver data, freeing port 53 for AdGuard. LAN clients
|
||||
may use AdGuard on Aegis, while Aegis itself uses the independent upstream DNS declared by
|
||||
`aegis_host_dns_servers`; this prevents Greenboot from depending on the AdGuard container during
|
||||
startup. Reboot Aegis after changing its NetworkManager DNS profile. Define
|
||||
`vault_aegis_icloudpd_apple_id` in Vault before applying it. iCloudPD still requires interactive MFA
|
||||
initialization after its first deployment.
|
||||
startup. Reboot Aegis after changing its NetworkManager DNS profile. iCloudPD was retired from Aegis;
|
||||
the Aegis role no longer contains iCloudPD tasks. Atlas iCloudPD config is Vault-backed; MFA is manual.
|
||||
|
||||
New Aegis images create the `admin` account in Butane. Before configuring a newly imaged node, run its
|
||||
first playbook execution with `-e ansible_user=admin`; the SSH hardening role then permits that same
|
||||
@@ -275,7 +272,20 @@ Atlas temporarily hosts rootless Navidrome and Syncthing until Uranus replaces t
|
||||
Atlas' LAN address (`192.168.178.55`); WireGuard remains exclusively between Prometheus (`10.0.0.1`)
|
||||
and Aegis (`10.0.0.2`). Their state is initialized ex novo in `/zpool/services/data/navidrome` and
|
||||
`/zpool/services/data/syncthing`; no source application state is migrated. The music library at
|
||||
`/zpool/media/music` is populated separately.
|
||||
`/zpool/media/music` was populated separately from `/zpool/archive/Music` on 2026-09-30;
|
||||
Navidrome completed its library scan. The rootless `atlas-music-sync.timer` copies new and changed
|
||||
files daily at 00:45 Europe/Rome, without deleting destination-only files. Both ZFS datasets must
|
||||
be mounted. Its first scheduled run succeeded on 2026-10-02. Some source playlists still contain
|
||||
obsolete Windows paths.
|
||||
|
||||
The Gitea move from Prometheus to Atlas is tracked in
|
||||
[`docs/atlas-gitea-migration.md`](docs/atlas-gitea-migration.md). The final consistent copy runs in
|
||||
Atlas' dedicated dataset under `admin`'s rootless user Quadlet. Its pinned derived image uses an
|
||||
internal Unix user named `gitea` (UID/GID 1000), while clone URLs keep `git@`. NPM remains on Prometheus and the primary
|
||||
public HTTPS route serves Atlas. Public SSH/2222 authenticates the `ikaros` key and serves
|
||||
`git ls-remote`; the operator also confirmed SSH pull and push. HTTPS login and Git writes were
|
||||
confirmed on 2026-10-03. The old Gitea data remains on Prometheus, but its container
|
||||
is absent from the desired stack.
|
||||
|
||||
The separate `wireguard_overlay` role manages `wg0` between Prometheus (`10.0.0.1`) and Aegis
|
||||
(`10.0.0.2`), generating private keys once on their respective hosts and exchanging only public keys
|
||||
@@ -304,8 +314,8 @@ snapshots at minute 05, 30 daily snapshots at 00:15, 8 weekly snapshots on Sunda
|
||||
monthly snapshots on the first day at 02:00. The retention helper prunes only snapshots carrying its
|
||||
managed `atlas-auto` prefix and never rolls back a dataset. The OpenZFS monthly scrub timer is scheduled
|
||||
for the first Sunday at 03:00; the conflicting weekly scrub timer is disabled explicitly. The first recursive
|
||||
hourly snapshot completed successfully on Atlas; retention pruning and the first scheduled scrub still await
|
||||
live runtime evidence. Validate this layer independently with:
|
||||
hourly snapshot completed successfully on Atlas, and scheduled retention pruning was observed on
|
||||
2026-09-30. The first monthly scrub still awaits runtime evidence. Validate this layer independently with:
|
||||
|
||||
```bash
|
||||
ANSIBLE_LOCAL_TEMP=/tmp/ansible-local \
|
||||
@@ -322,12 +332,22 @@ passphrase, cache, and Borg state. Borg receives its passphrase through a mode `
|
||||
|
||||
The daily backup starts at 04:30 with up to 30 minutes of randomized delay. It creates a temporary,
|
||||
recursive ZFS snapshot and reconstructs every dataset below `/zpool` as a read-only bind-mounted tree,
|
||||
so parent and child datasets enter one consistent Borg archive. Cleanup always removes the temporary
|
||||
mounts and managed snapshot. Only the root wrapper performs snapshot and mount operations; it launches
|
||||
the Borg client as `borg` with temporary read-search capability and no ZFS, sudo, or pool-management
|
||||
privileges. Borg retains 30 daily, 8 weekly, and 12 monthly archives, then compacts the standard
|
||||
so parent and child datasets enter one consistent Borg archive. The wrapper recursively unmounts its
|
||||
private source tree; a narrowly scoped `ExecStopPost` helper removes any remaining host-namespace ZFS
|
||||
snapshot mounts and the named temporary snapshot after the backup process exits. Only the root wrapper
|
||||
performs snapshot and mount operations; it launches the Borg client as `borg` with temporary read-search
|
||||
capability and no ZFS, sudo, or pool-management privileges. Borg retains 30 daily, 8 weekly, and 12
|
||||
monthly archives, then compacts the standard
|
||||
read-write repository. A full metadata and repository check runs as `borg` on the fifteenth day of each
|
||||
month at 06:00. Both operations use a common lock, journal logging, and bounded systemd retries.
|
||||
New backup runs also log the create phase and a compact progress line at most once per minute: an
|
||||
**estimated** percentage, dataset, files processed, and original/compressed/deduplicated bytes. The
|
||||
denominator is the summed ZFS `logicalreferenced` size of the backup's own recursive snapshot, not a
|
||||
Borg-reported total: the estimate can exceed 100% and does not cover retention, compaction, or checks.
|
||||
Progress lines omit individual filenames; warnings may still name affected files.
|
||||
Follow the current run with
|
||||
`sudo journalctl -fu atlas-borg-backup.service` on Atlas; changes to the helper do not alter a run
|
||||
already in progress.
|
||||
|
||||
Initial activation remains explicit:
|
||||
|
||||
@@ -350,14 +370,179 @@ ansible-playbook ansible/site.yml --limit atlas --tags packages,borg --check --d
|
||||
Atlas runtime activation is complete: the initial backup and repository check succeeded, a full restore
|
||||
to a temporary directory was validated against the live `Archive` tree, the recovery-key export was copied
|
||||
to offline storage, and the temporary snapshot and bind mounts were cleaned up.
|
||||
The populated-pool archive on 2026-09-29 took 1 h 32 min for 2.18 TB original / 2.04 TB compressed
|
||||
data, with a 13.49 GB deduplicated archive size. Retention and compaction succeeded, but a
|
||||
`RuntimeDirectory` permission error prevented post-exit snapshot cleanup. After correction, the
|
||||
2026-09-30 incremental archive completed in about 22 seconds, removed the stale and current temporary
|
||||
snapshots, and ended with service status 0. The monitor reported 37% Storage Box quota used. These
|
||||
observations do not predict the duration or compression ratio of future runs.
|
||||
On 2026-09-25 a separate ZFS restore smoke test copied a small file from an automatic daily
|
||||
`zpool/archive` snapshot to `/var/tmp`, then confirmed matching contents, ownership, mode, mtime and
|
||||
POSIX ACL. The temporary copy and on-demand snapshot mount were removed; Borg kept running. This
|
||||
does not validate a full dataset recovery.
|
||||
|
||||
The offline USB backup is deployed as a manual-only service (`atlas_manage_usb_backup: true`):
|
||||
Ansible never formats, unlocks, mounts, backs up to, or schedules the disk. Atlas' existing USB disk was verified
|
||||
read-only on 2026-09-23 as LUKS UUID `577b3c43-ea37-4611-81a9-39d555cdfbd4`, containing ext4 UUID
|
||||
`758e2d2e-a427-4797-aad9-39c3a9f17c7e` through mapper `zpool-backup`. It was mounted at
|
||||
`/mnt/zpool-backup` at inspection time. The service deliberately requires the verified mapper to be
|
||||
**not mounted** before starting. When necessary, `systemd-ask-password` requests the LUKS passphrase
|
||||
through the `systemctl start` password agent; it is piped directly to `cryptsetup` without saving it,
|
||||
passing it as a command argument, or caching it. The service then mounts the disk privately, takes a recursive ZFS snapshot,
|
||||
copies every dataset to a versioned `atlas/snapshots/<timestamp>/` directory using `rsync --link-dest`,
|
||||
verifies the result with a checksum-based dry run, atomically updates `atlas/latest`, unmounts and closes
|
||||
LUKS. A failed run never replaces `latest` or removes an earlier complete version. Borg and the USB
|
||||
backup may run concurrently from separate snapshots; both reading the same pool can reduce throughput.
|
||||
The USB copy preserves ACLs but not generic extended attributes; `security.selinux` is also intentionally
|
||||
excluded because the target SELinux policy must recreate labels during a restore. Do not restore data into
|
||||
service paths without relabeling. After restoring an explicit dataset path, apply its destination policy with:
|
||||
|
||||
```bash
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags restorecon \
|
||||
-e '{"atlas_restorecon_paths":["/zpool/archive"]}'
|
||||
```
|
||||
|
||||
The task accepts only paths below the Atlas pool mount root, runs `restorecon -RFv` only for the paths
|
||||
provided at invocation, and is otherwise a no-op. It must not be used on the whole pool during routine runs.
|
||||
Old USB versions are not pruned automatically, to avoid deleting the only offline
|
||||
copy without an explicitly chosen retention policy; capacity checks include an estimated transfer size
|
||||
and a 10 GiB free-space reserve. The disk must be physically disconnected after a successful backup
|
||||
to make the copy offline.
|
||||
|
||||
To check the USB backup and reminder configuration without starting a backup, run:
|
||||
|
||||
```bash
|
||||
ANSIBLE_LOCAL_TEMP=/tmp/ansible-local \
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags usb_backup,usb_reminder --check --diff
|
||||
```
|
||||
|
||||
To validate a planned, explicit post-restore relabel operation without changing labels, run:
|
||||
|
||||
```bash
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags restorecon --check \
|
||||
-e '{"atlas_restorecon_paths":["/zpool/archive"]}'
|
||||
```
|
||||
|
||||
Before the first **manual** service start, safely unmount the currently mounted
|
||||
`/mnt/zpool-backup`; never run it on an arbitrary mounted disk. Future starts
|
||||
can begin with the mapper closed: `sudo systemctl start atlas-usb-backup.service` prompts for the
|
||||
passphrase interactively and then performs the backup. Neither the LUKS password nor a key file belongs
|
||||
in Ansible. Inspect the run with
|
||||
`sudo journalctl -fu atlas-usb-backup.service`. There is intentionally no timer. Independently test a
|
||||
read-only mount and restore from `atlas/latest` into an empty temporary directory before marking the
|
||||
USB recovery path complete. Only `atlas-usb-reminder.timer` is enabled, for the first Saturday of each
|
||||
month at 10:00 Europe/Rome. Its warning notification uses the existing 45Drives Houston notifier.
|
||||
A manual test confirmed a notification in 45Drives Alerts, **not** an email. The reminder service log
|
||||
reports notification submission, not email delivery; the role does not depend on SMTP/OAuth settings.
|
||||
The reminder never starts the backup. Check its schedule with
|
||||
`systemctl list-timers atlas-usb-reminder.timer` and the result in 45Drives Alerts.
|
||||
The timer was verified active with its first scheduled run at 2026-10-03 10:00 CEST. No email
|
||||
delivery is claimed.
|
||||
The first manual USB attempt on 2026-09-23 did not complete: rsync was denied while removing
|
||||
`security.selinux` on the USB filesystem, then the interrupted service left its recursive
|
||||
`atlas-usb-20260923T185748Z-2469168` snapshot and the `zpool-backup` LUKS mapper open. The
|
||||
rsync xattr filter was deployed afterward. The incomplete USB directory was absent on inspection;
|
||||
the exact failed snapshot was removed, the verified and unmounted mapper closed, and the service
|
||||
failed state cleared. A final check found no remnant snapshot, mount, mapper, or staging directory.
|
||||
The failed attempt was not a valid backup, and no USB restore had been tested at that point.
|
||||
On 2026-09-24 a later run reported a checksum-verified, published USB version and closed the LUKS
|
||||
mapper, but the service failed while destroying its temporary ZFS snapshot: OpenZFS still had
|
||||
on-demand `.zfs/snapshot` mounts open in the host namespace. Those exact temporary snapshots were
|
||||
unmounted normally and removed; no force or rollback was used. The backup service now records its
|
||||
snapshot name and runs a narrowly scoped `ExecStopPost` cleanup after the private backup process
|
||||
exits. The cleanup helper was tested with a disposable recursive snapshot and an active snapshot
|
||||
mount. A complete run on 2026-09-24 later checksum-verified and published a new USB version; the
|
||||
service ended successfully, the LUKS mapper closed, no temporary USB snapshot remained, and the pool
|
||||
was healthy. On 2026-09-25 an independent restore test opened the configured USB disk read-only, mounted
|
||||
ext4 with `ro,noload`, restored a 5,707,945-byte file from the published `atlas/latest` version to an
|
||||
empty `/var/tmp` directory, and matched its content, owner, mode, size, mtime, and POSIX ACL against
|
||||
the USB source. The test removed its temporary copy and mount, closed the LUKS mapper, and left the
|
||||
pool healthy while Borg continued running. This is a file-level recovery smoke test, not a full dataset
|
||||
or disaster-recovery exercise.
|
||||
|
||||
Atlas health monitoring runs every 30 minutes through `atlas-health-monitor.timer`. Its read-only probes
|
||||
check pool/vdev state and errors, scrub/resilver status, four pool disks and the system NVMe via SMART,
|
||||
disk and CPU temperatures, system/pool/snapshot space, local `zpool/backup` growth, and the Hetzner
|
||||
Storage Box quota via `df -m` over the dedicated `borg` account's pinned-key SSH connection. The remote
|
||||
query never opens the Borg repository or its lock. Growth alerts compare against a roughly 24-hour
|
||||
baseline and therefore begin only after enough samples exist. The monitor also checks
|
||||
maintenance/backup timer activation and freshness; systemd `OnFailure` hooks report snapshot,
|
||||
scrub, Borg, USB, reminder, and monitoring services when they enter the failed state. An ongoing
|
||||
Borg run is never restarted by the monitor; only a run exceeding 14 days raises a warning.
|
||||
Thresholds and stable disk paths are declared in Atlas host variables. Alerts use the existing 45Drives
|
||||
Houston notifier and repeated issues are deduplicated; **email delivery is not verified**. The
|
||||
2026-09-25 live probe found no issues and a labelled test notification was submitted. On 2026-09-30
|
||||
the monitor reported zero issues and 37% Storage Box quota used. The failed-job hook now passes the
|
||||
literal systemd unit name; its expansion was verified without sending a false failure notification.
|
||||
|
||||
```bash
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags monitoring --check --diff
|
||||
sudo /usr/local/libexec/atlas-health-monitor --dry-run
|
||||
sudo journalctl -u atlas-health-monitor.service -n 100 --no-pager
|
||||
systemctl list-timers atlas-health-monitor.timer
|
||||
```
|
||||
|
||||
`--dry-run` sends no alerts and does not change monitor state. A real check is
|
||||
`sudo systemctl start atlas-health-monitor.service`; do not start the backup services merely to test
|
||||
monitoring. For a labelled 45Drives Alerts delivery test, use
|
||||
`sudo /usr/local/libexec/atlas-health-monitor --test-notification`.
|
||||
|
||||
### Atlas systemd timers
|
||||
|
||||
All ten managed timers below are enabled. Times are local to Atlas (`Europe/Rome`); Borg and monitoring
|
||||
add the indicated randomized delay. Every timer has `Persistent=true`, so a missed calendar run is
|
||||
scheduled after the timer becomes active again.
|
||||
|
||||
| Timer | Schedule (`OnCalendar`) | Action |
|
||||
| --- | --- | --- |
|
||||
| `atlas-zfs-snapshot-hourly.timer` | `*-*-* *:05:00` — every hour at :05 | Recursive hourly snapshot and retention |
|
||||
| `atlas-zfs-snapshot-daily.timer` | `*-*-* 00:15:00` — daily at 00:15 | Recursive daily snapshot and retention |
|
||||
| `atlas-zfs-snapshot-weekly.timer` | `Sun *-*-* 01:00:00` — Sunday at 01:00 | Recursive weekly snapshot and retention |
|
||||
| `atlas-zfs-snapshot-monthly.timer` | `*-*-01 02:00:00` — first day of the month at 02:00 | Recursive monthly snapshot and retention |
|
||||
| `zfs-scrub-monthly@zpool.timer` | `Sun *-*-01..07 03:00:00` — first Sunday at 03:00 | ZFS scrub |
|
||||
| `atlas-borg-backup.timer` | `*-*-* 04:30:00` — daily at 04:30, plus 0–30 min random delay | Encrypted offsite backup |
|
||||
| `atlas-borg-check.timer` | `*-*-15 06:00:00` — 15th of the month at 06:00, plus 0–30 min random delay | Borg repository check |
|
||||
| `atlas-usb-reminder.timer` | `Sat *-*-01..07 10:00:00 Europe/Rome` — first Saturday at 10:00 | 45Drives Alerts reminder only |
|
||||
| `atlas-health-monitor.timer` | `*:0/30` — every half-hour, plus 0–5 min random delay | Read-only health checks |
|
||||
| `atlas-prometheus-pull.timer` | `*-*-* 03:00:00 Europe/Rome` — daily at 03:00 | Pull and verify the prepared Prometheus backup |
|
||||
|
||||
`atlas-usb-backup.service` has **no timer**: the encrypted USB backup must be started manually.
|
||||
The vendor's `zfs-scrub-weekly@zpool.timer` is intentionally disabled in favor of the monthly scrub.
|
||||
The Prometheus export timer runs at 02:00 Europe/Rome. Its first scheduled export and Atlas pull
|
||||
passed on 2026-10-01; a manual post-NPM-Quadlet export, pull, and temporary restore passed on
|
||||
2026-10-03. The first scheduled cycle after that cutover remains to be observed. While a
|
||||
Borg backup is still running, `systemctl list-timers` may show `-` for its next trigger; this does not
|
||||
mean the timer has been disabled. Inspect the current schedule on Atlas with
|
||||
`systemctl list-timers --all`.
|
||||
|
||||
A temporary Nextcloud deployment on Atlas is also planned before Uranus: it requires separately
|
||||
declared persistent application, database, and cache storage, Vault-backed credentials, NPM-only
|
||||
publishing through Aegis, and defined backup, upgrade, and eventual migration procedures. Do not deploy
|
||||
it before the data-protection checklist is complete.
|
||||
|
||||
Prometheus backup pulls, USB backup, monitoring, and disaster-recovery tests remain follow-up work. The
|
||||
prioritized operational backlog is kept in `AGENTS.md`.
|
||||
Atlas is the declared iCloud photo-ingestion host. Ansible manages the rootless Quadlet, a private
|
||||
Vault-backed `icloudpd.conf`, photos under `/zpool/archive/Pictures/iCloudPD`, and separate state in
|
||||
`zpool/services/data/icloudpd`. The service was started manually; Ansible does not enable automatic
|
||||
startup or manage the password and MFA keyring. The operator initialized MFA interactively; on
|
||||
2026-10-03 the initial photo/video download completed. Aegis iCloudPD, including its service data,
|
||||
has been removed and verified; the Aegis role no longer manages it. Backup/restore and SMB access
|
||||
for the new data remain unverified. The Photobook NFS export remains untouched. See
|
||||
[`docs/atlas-icloudpd-migration.md`](docs/atlas-icloudpd-migration.md).
|
||||
|
||||
The first scheduled Prometheus backup runs and production-size disaster-recovery tests remain follow-up work. The prioritized
|
||||
operational backlog is kept in `AGENTS.md`.
|
||||
|
||||
Priority 2 procedures and decisions are recorded in
|
||||
[`docs/atlas-recovery.md`](docs/atlas-recovery.md),
|
||||
[`docs/atlas-updates.md`](docs/atlas-updates.md), and
|
||||
[`docs/atlas-sharing-decision.md`](docs/atlas-sharing-decision.md).
|
||||
The provisional Atlas recovery objectives are RPO 24 hours and RTO 72 hours;
|
||||
an isolated small-VM OS rebuild, pool import, Ansible reapplication, and
|
||||
snapshot restore passed, but full-size recovery time is unmeasured. `Archive` (SMB) and
|
||||
`photobook` (NFS) remain deliberately separate.
|
||||
The Prometheus pull architecture and manual export/pull/restore evidence are in
|
||||
[`docs/prometheus-backup.md`](docs/prometheus-backup.md). Both daily timers are
|
||||
enabled; their first scheduled runs remain to be verified.
|
||||
|
||||
## How layering works
|
||||
|
||||
@@ -528,8 +713,9 @@ ansible-playbook ansible/site.yml --limit <host> --tags <tag1>,<tag2> --check --
|
||||
ansible-playbook ansible/site.yml --limit <host> --start-at-task "<task name>" --check --diff
|
||||
ansible-lint ansible/roles/<role>
|
||||
yamllint ansible/path/to/file.yml
|
||||
podman-compose -f /opt/docker/server/docker-compose.yml config
|
||||
ansible-playbook ansible/site.yml --limit prometheus --tags server_legacy_cleanup --check --diff -e server_legacy_cleanup=true
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags storage,sharing,containers --check --diff
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags music_sync --check --diff
|
||||
```
|
||||
|
||||
## Tags
|
||||
|
||||
@@ -6,6 +6,10 @@ effective_username: "{{ server_username }}"
|
||||
effective_user_group: "{{ server_user_group }}"
|
||||
effective_user_home: "{{ server_user_home }}"
|
||||
server_container_stack_dir: /opt/docker/server
|
||||
server_npm_quadlet_stage: false
|
||||
server_npm_quadlet_cutover: false
|
||||
server_legacy_stack_retired: false
|
||||
server_legacy_cleanup: false
|
||||
ai_agents: {}
|
||||
vim_plugins_enabled: false
|
||||
|
||||
@@ -80,5 +84,36 @@ server_sshd_settings:
|
||||
|
||||
server_sshd_allow_users:
|
||||
- "{{ server_username }}"
|
||||
server_backup_export_enabled: false
|
||||
server_backup_username: prometheus-backup
|
||||
server_backup_public_key_name: atlas-pull
|
||||
server_backup_export_root: /var/lib/prometheus-backup-export
|
||||
server_backup_rrsync_path: /usr/share/doc/rsync/support/rrsync
|
||||
server_backup_export_calendar: "*-*-* 02:00:00 Europe/Rome"
|
||||
server_backup_export_start_timer: false
|
||||
# Ongoing public Gitea proxy configuration.
|
||||
server_gitea_proxy_enabled: false
|
||||
server_gitea_on_atlas: false
|
||||
server_gitea_atlas_address: "{{ hostvars['atlas'].ansible_host }}"
|
||||
server_gitea_npm_domains: []
|
||||
server_gitea_ssh_public_port: 2222
|
||||
server_gitea_ssh_target_port: 2222
|
||||
server_backup_export_source_keep: 3
|
||||
server_backup_export_paths: >-
|
||||
{{ ['opt/npm/data', 'opt/npm/letsencrypt']
|
||||
+ ([] if server_gitea_on_atlas | bool else ['opt/gitea/data', 'home/git/.ssh'])
|
||||
+ ([] if server_legacy_stack_retired | bool else
|
||||
['opt/docker/server/docker-compose.yml',
|
||||
'etc/systemd/system/podman-compose-server.service'])
|
||||
+ (['etc/containers/systemd/prometheus-npm.container',
|
||||
'etc/containers/systemd/server-web.network']
|
||||
if server_npm_quadlet_stage | bool else [])
|
||||
+ ['etc/ssh/sshd_config', 'etc/ssh/sshd_config.d',
|
||||
'etc/firewalld', 'etc/wireguard/wg0.conf'] }}
|
||||
server_backup_export_excludes: >-
|
||||
{{ ['opt/npm/data/logs']
|
||||
+ ([] if server_gitea_on_atlas | bool else
|
||||
['opt/gitea/data/gitea/log', 'opt/gitea/data/gitea/tmp',
|
||||
'opt/gitea/data/gitea/sessions', 'opt/gitea/data/gitea/indexers']) }}
|
||||
server_ssh_authorized_keys: []
|
||||
server_ssh_authorized_key_directory: "{{ server_user_home }}/.ssh/authorized_keys.d"
|
||||
|
||||
@@ -42,5 +42,3 @@ aegis_ssh_authorized_keys:
|
||||
key: "ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIEH/7GJfGt0ZVmKeEzceoFkFkeCXFryKK9vAbaip+HCx nymph"
|
||||
- name: siren
|
||||
key: "ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAIA95wYlzpfN3rjUhpMeP4KHn8I6ZrjQXoDTgwgRIa++b siren"
|
||||
|
||||
aegis_icloudpd_apple_id: "{{ vault_aegis_icloudpd_apple_id | default('') }}"
|
||||
|
||||
@@ -49,6 +49,43 @@ atlas_zfs_backup_reservation: 500G
|
||||
atlas_zfs_dataset_photobook: media/photobook
|
||||
atlas_mount_root: /zpool
|
||||
atlas_manage_storage: true
|
||||
atlas_manage_nextcloud: true
|
||||
atlas_nextcloud_domain: cloud.fscotto.co
|
||||
atlas_onlyoffice_domain: office.fscotto.co
|
||||
# Resolved official amd64 images on 2026-10-03; updates are deliberate.
|
||||
atlas_nextcloud_image: docker.io/library/nextcloud:33.0.9-apache@sha256:a97666d6ae931bde78a80cfba8abdf46d436d7b540f31895803f6fb0a012d689
|
||||
atlas_nextcloud_postgres_image: docker.io/library/postgres:17-bookworm@sha256:639ab7ceb90e13123085b741fb31ef493fba25463002f6da665352e7b534b652
|
||||
atlas_nextcloud_redis_image: docker.io/library/redis:7.4-bookworm@sha256:c6eabf748fc7a61dbb5a705c78bcf3d6377b1127a97d0ce965c11c44ba46896f
|
||||
atlas_onlyoffice_image: docker.io/onlyoffice/documentserver:9.4.0.1@sha256:3ab6ebc7c605e5a32b7ae3ff19daed4925090245acc8100ce2230bd766c88212
|
||||
atlas_nextcloud_users:
|
||||
- username: fabio
|
||||
display_name: Fabio
|
||||
password: "{{ vault_nextcloud_fabio_password }}"
|
||||
- username: chiara
|
||||
display_name: Chiara
|
||||
password: "{{ vault_nextcloud_chiara_password }}"
|
||||
atlas_nextcloud_apps:
|
||||
- id: groupfolders
|
||||
version: 21.0.9
|
||||
url: https://github.com/nextcloud-releases/groupfolders/releases/download/v21.0.9/groupfolders-v21.0.9.tar.gz
|
||||
checksum: sha256:d8b95f0778425f646f2311ba5b42d8e2fcfdf37dc2fd35fcac8d3f01bde38a21
|
||||
- id: onlyoffice
|
||||
version: 10.2.1
|
||||
url: https://github.com/ONLYOFFICE/onlyoffice-nextcloud/releases/download/v10.2.1/onlyoffice.tar.gz
|
||||
checksum: sha256:144998af0610ccd17ee8d7025e2f8001472da03f6dab90ff38039247825e3a9b
|
||||
- id: contacts
|
||||
version: 8.9.1
|
||||
url: https://github.com/nextcloud-releases/contacts/releases/download/v8.9.1/contacts-v8.9.1.tar.gz
|
||||
checksum: sha256:a25cdf448b192631b8e8eb7addc31b40382b33871b10521f4308ac5a6e0457bf
|
||||
- id: calendar
|
||||
version: 6.6.2
|
||||
url: https://github.com/nextcloud-releases/calendar/releases/download/v6.6.2/calendar-v6.6.2.tar.gz
|
||||
checksum: sha256:7e83632d4436d3037a34d1c73cbc06d5ccb6e8fc10f096a86a43a0515588529c
|
||||
# Rootless Gitea was restored from the stopped-source export before production activation.
|
||||
atlas_manage_gitea: true
|
||||
atlas_gitea_production_enabled: true
|
||||
atlas_gitea_public_domain: git.fscotto.co
|
||||
atlas_prometheus_pull_start_timer: true
|
||||
atlas_manage_zfs_snapshots: true
|
||||
atlas_zfs_snapshot_prefix: atlas-auto
|
||||
atlas_zfs_snapshot_policies:
|
||||
@@ -83,8 +120,61 @@ atlas_borg_randomized_delay: 30m
|
||||
atlas_borg_keep_daily: 30
|
||||
atlas_borg_keep_weekly: 8
|
||||
atlas_borg_keep_monthly: 12
|
||||
atlas_manage_usb_backup: true
|
||||
# Read-only lsblk verification on Atlas, 2026-09-23. Never store the LUKS password here.
|
||||
atlas_usb_backup_luks_uuid: 577b3c43-ea37-4611-81a9-39d555cdfbd4
|
||||
atlas_usb_backup_fs_uuid: 758e2d2e-a427-4797-aad9-39c3a9f17c7e
|
||||
atlas_usb_backup_mapper_name: zpool-backup
|
||||
atlas_manage_usb_reminder: true
|
||||
atlas_usb_reminder_calendar: "Sat *-*-01..07 10:00:00 Europe/Rome"
|
||||
atlas_manage_monitoring: true
|
||||
atlas_manage_prometheus_backup_pull: true
|
||||
# Prometheus ED25519 host key read through the controller's strict SSH trust on 2026-09-30.
|
||||
# Fingerprint: SHA256:rfedk7DHI9mLB3UHk/4F3HHlSIiswtCAFsAXvfh6iXk
|
||||
atlas_prometheus_ssh_host_key: >-
|
||||
179.237.102.172 ssh-ed25519
|
||||
AAAAC3NzaC1lZDI1NTE5AAAAIC4b+QXlPupoEx71W9NKs9tTeYjBqTkVMqbGB97nMNWv
|
||||
# Physical pool disks and the system NVMe; the disconnected USB disk is intentionally excluded.
|
||||
atlas_monitor_smart_devices:
|
||||
- { name: pool-1, path: "{{ atlas_zpool_disks[0] }}", warning_c: 50, critical_c: 55 }
|
||||
- { name: pool-2, path: "{{ atlas_zpool_disks[1] }}", warning_c: 50, critical_c: 55 }
|
||||
- { name: pool-3, path: "{{ atlas_zpool_disks[2] }}", warning_c: 50, critical_c: 55 }
|
||||
- { name: pool-4, path: "{{ atlas_zpool_disks[3] }}", warning_c: 50, critical_c: 55 }
|
||||
- name: system-nvme
|
||||
path: /dev/disk/by-id/nvme-Patriot_M.2_P320_256GB_P320ADB26011606111
|
||||
warning_c: 70
|
||||
critical_c: 85
|
||||
atlas_monitor_timers:
|
||||
- { name: atlas-zfs-snapshot-hourly.timer, max_age_hours: 3 }
|
||||
- { name: atlas-zfs-snapshot-daily.timer, max_age_hours: 36 }
|
||||
- { name: atlas-zfs-snapshot-weekly.timer, max_age_hours: 216 }
|
||||
- { name: atlas-zfs-snapshot-monthly.timer, max_age_hours: 960 }
|
||||
- { name: zfs-scrub-monthly@zpool.timer, max_age_hours: 960 }
|
||||
- { name: atlas-borg-backup.timer, max_age_hours: 48 }
|
||||
- { name: atlas-borg-check.timer, max_age_hours: 960 }
|
||||
# The first manual USB reminder is not due until October; activation is checked, not age.
|
||||
- { name: atlas-usb-reminder.timer, max_age_hours: 0 }
|
||||
atlas_monitor_failure_units:
|
||||
- atlas-zfs-snapshot@.service
|
||||
- zfs-scrub@zpool.service
|
||||
- atlas-borg-backup.service
|
||||
- atlas-borg-check.service
|
||||
- atlas-usb-backup.service
|
||||
- atlas-usb-reminder.service
|
||||
- atlas-health-monitor.service
|
||||
atlas_monitor_remote_capacity:
|
||||
user: "{{ atlas_borg_repository_user }}"
|
||||
host: "{{ atlas_borg_repository_host }}"
|
||||
run_as: "{{ atlas_borg_username }}"
|
||||
ssh_wrapper: "{{ atlas_borg_ssh_wrapper_path }}"
|
||||
warning_percent: 80
|
||||
critical_percent: 90
|
||||
growth_warning_gib_day: 500
|
||||
atlas_manage_sharing: true
|
||||
atlas_manage_media_stack: false
|
||||
# Planned after data-protection validation: move iCloudPD photo ingestion from
|
||||
# Aegis to Atlas, with photos under /zpool/archive/Pictures/iCloudPD and
|
||||
# application/MFA state in a separate dataset. Do not deploy or cut over yet.
|
||||
|
||||
# WireGuard is retired on Atlas. These rootless services are a temporary home
|
||||
# until Uranus replaces them.
|
||||
@@ -94,6 +184,7 @@ backend_phase1_bind_address: "{{ ansible_host }}"
|
||||
backend_phase1_firewalld_zone: "{{ atlas_firewalld_zone }}"
|
||||
backend_phase1_npm_source_ip: "{{ atlas_aegis_ip }}"
|
||||
backend_phase1_syncthing_native_subnet: "{{ atlas_lan_subnet }}"
|
||||
backend_phase1_music_sync_enabled: true
|
||||
|
||||
rocky_manage_openzfs_repo: true
|
||||
rocky_manage_syncthing_binary: false
|
||||
@@ -103,10 +194,17 @@ rocky_podman_packages:
|
||||
|
||||
host_packages:
|
||||
- cockpit
|
||||
- cockpit-podman
|
||||
- cockpit-storaged
|
||||
- realmd
|
||||
- pcp
|
||||
- python3-pcp
|
||||
- cryptsetup
|
||||
- nfs-utils
|
||||
- policycoreutils
|
||||
- policycoreutils-python-utils
|
||||
- python3-libselinux
|
||||
- setroubleshoot-server
|
||||
- samba
|
||||
- samba-client
|
||||
- samba-common-tools
|
||||
@@ -144,4 +242,5 @@ atlas_firewalld_rich_rules:
|
||||
host_enabled_services:
|
||||
- sshd
|
||||
- cockpit.socket
|
||||
- pmlogger.service
|
||||
- zfs.target
|
||||
|
||||
@@ -6,7 +6,26 @@ ansible_port: 22
|
||||
ansible_ssh_private_key_file: /home/fscotto/.ssh/id_ed25519
|
||||
|
||||
server_username: rocky
|
||||
server_duckdns_domain: fscotto
|
||||
server_legacy_stack_retired: true
|
||||
# Destructive deletion runs only with an explicit extra-var and cleanup tag.
|
||||
server_legacy_cleanup: false
|
||||
# Explicit opt-in cleanup; no data, volumes, networks or NPM images are removed.
|
||||
server_legacy_image_cleanup: false
|
||||
server_legacy_images:
|
||||
- docker.gitea.com/gitea:1.25.2
|
||||
- docker.io/deluan/navidrome:latest
|
||||
- docker.io/library/postgres:13
|
||||
server_npm_quadlet_stage: true
|
||||
server_npm_quadlet_image: docker.io/jc21/nginx-proxy-manager@sha256:52b2c59994f3d36acfcf70a1626f29734df0ed8c71bacc0269f78b6f939858bb
|
||||
# The stopped-source export and live Quadlet cutover passed on 2026-10-03.
|
||||
server_npm_quadlet_cutover: true
|
||||
server_backup_export_enabled: true
|
||||
server_backup_export_start_timer: true
|
||||
# Install the final-copy helper only; it is never run by a normal playbook invocation.
|
||||
server_gitea_proxy_enabled: true
|
||||
server_gitea_on_atlas: true
|
||||
server_gitea_npm_domains:
|
||||
- git.fscotto.duckdns.org
|
||||
server_ssh_authorized_keys:
|
||||
- name: ikaros
|
||||
key: "ssh-ed25519 AAAAC3NzaC1lZDI1NTE5AAAAINrIxXjA3ffPwziKGR5gzc4gAoBehQPlnEMcXF4Wl0ZS ikaros"
|
||||
@@ -32,6 +51,12 @@ host_packages:
|
||||
- cockpit
|
||||
- cockpit-navigator
|
||||
- cockpit-podman
|
||||
- cockpit-storaged
|
||||
- realmd
|
||||
- pcp
|
||||
- python3-pcp
|
||||
- setroubleshoot-server
|
||||
|
||||
host_enabled_services:
|
||||
- cockpit.socket
|
||||
- pmlogger.service
|
||||
|
||||
@@ -39,11 +39,41 @@
|
||||
state: enabled
|
||||
when: "'workstation_dev_wsl' in group_names"
|
||||
|
||||
- name: Install distribution signing keys for Fedora desktop codecs
|
||||
tags: [packages, heic]
|
||||
ansible.builtin.dnf:
|
||||
name: distribution-gpg-keys
|
||||
state: present
|
||||
when: "'graphical_desktop' in group_names"
|
||||
|
||||
- name: Import RPM Fusion Free signing key for Fedora desktop codecs
|
||||
tags: [packages, heic]
|
||||
ansible.builtin.rpm_key:
|
||||
key: /usr/share/distribution-gpg-keys/rpmfusion/RPM-GPG-KEY-rpmfusion-free-fedora-2020
|
||||
fingerprint: E9A491A3DE247814E7E067EAE06F8ECDD651FF2E
|
||||
state: present
|
||||
when: "'graphical_desktop' in group_names"
|
||||
|
||||
- name: Enable RPM Fusion Free for Fedora desktop codecs
|
||||
tags: [packages, heic]
|
||||
ansible.builtin.dnf:
|
||||
name: "https://download1.rpmfusion.org/free/fedora/rpmfusion-free-release-{{ ansible_facts['distribution_major_version'] }}.noarch.rpm"
|
||||
state: present
|
||||
when: "'graphical_desktop' in group_names"
|
||||
|
||||
- name: Refresh dnf package metadata
|
||||
tags: [packages]
|
||||
ansible.builtin.dnf:
|
||||
update_cache: true
|
||||
|
||||
- name: Install HEIC decoder on Fedora desktops
|
||||
tags: [packages, heic]
|
||||
ansible.builtin.dnf:
|
||||
name: libheif-freeworld
|
||||
state: present
|
||||
update_cache: true
|
||||
when: "'graphical_desktop' in group_names"
|
||||
|
||||
- name: Install packages on Fedora
|
||||
tags: [packages]
|
||||
ansible.builtin.dnf:
|
||||
|
||||
@@ -8,10 +8,6 @@ aegis_network_connection_uuid: ""
|
||||
aegis_host_dns_servers: []
|
||||
aegis_host_dns_search_domains: []
|
||||
aegis_adguard_image: docker.io/adguard/adguardhome:latest
|
||||
aegis_icloudpd_image: docker.io/boredazfcuk/icloudpd:latest
|
||||
aegis_icloudpd_folder_structure: '{:%Y/%m/%d}'
|
||||
aegis_icloudpd_synchronisation_interval: 86400
|
||||
aegis_icloudpd_apple_id: ""
|
||||
aegis_ikaros_mac_address: aa:bb:cc:dd:ee:ff
|
||||
aegis_wol_port: 9
|
||||
|
||||
|
||||
@@ -9,13 +9,12 @@
|
||||
name: sshd.service
|
||||
state: reloaded
|
||||
|
||||
- name: Restart Aegis Quadlet services
|
||||
- name: Restart Aegis AdGuard Quadlet
|
||||
ansible.builtin.systemd:
|
||||
name: "{{ item }}"
|
||||
state: restarted
|
||||
daemon_reload: true
|
||||
loop:
|
||||
- adguardhome.service
|
||||
- icloudpd.service
|
||||
loop_control:
|
||||
label: "{{ item }}"
|
||||
|
||||
@@ -13,14 +13,6 @@
|
||||
msg: Reboot Aegis to activate the newly layered packages, then rerun the playbook.
|
||||
when: aegis_layered_packages_result.needs_reboot | default(false)
|
||||
|
||||
- name: Require Aegis iCloudPD Apple ID
|
||||
tags: [aegis, icloudpd]
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- aegis_icloudpd_apple_id | length > 0
|
||||
fail_msg: Define vault_aegis_icloudpd_apple_id before applying the Aegis profile.
|
||||
no_log: true
|
||||
|
||||
- name: Require completed Aegis network placeholders
|
||||
tags: [aegis, dns, firewall, network, services]
|
||||
ansible.builtin.assert:
|
||||
@@ -119,8 +111,6 @@
|
||||
loop:
|
||||
- /var/lib/adguard/work
|
||||
- /var/lib/adguard/conf
|
||||
- /var/lib/icloudpd/data
|
||||
- /var/lib/icloudpd/config
|
||||
|
||||
- name: Create Quadlet configuration directory
|
||||
tags: [aegis, containers]
|
||||
@@ -142,12 +132,9 @@
|
||||
loop:
|
||||
- src: adguardhome.container.j2
|
||||
dest: adguardhome.container
|
||||
- src: icloudpd.container.j2
|
||||
dest: icloudpd.container
|
||||
loop_control:
|
||||
label: "{{ item.dest }}"
|
||||
no_log: "{{ item.dest == 'icloudpd.container' }}"
|
||||
notify: Restart Aegis Quadlet services
|
||||
notify: Restart Aegis AdGuard Quadlet
|
||||
|
||||
- name: Create Aegis systemd-resolved configuration directory
|
||||
tags: [aegis, adguard, dns, services]
|
||||
@@ -168,7 +155,7 @@
|
||||
mode: "0644"
|
||||
notify:
|
||||
- Restart Aegis systemd-resolved
|
||||
- Restart Aegis Quadlet services
|
||||
- Restart Aegis AdGuard Quadlet
|
||||
|
||||
- name: Point Aegis resolver at the full systemd-resolved configuration
|
||||
tags: [aegis, adguard, dns, services]
|
||||
@@ -361,7 +348,7 @@
|
||||
group: root
|
||||
mode: "0755"
|
||||
|
||||
- name: Enable Aegis Quadlet services and automatic updates
|
||||
- name: Enable Aegis AdGuard Quadlet and automatic updates
|
||||
tags: [aegis, containers, services]
|
||||
ansible.builtin.systemd:
|
||||
name: "{{ item }}"
|
||||
@@ -370,7 +357,6 @@
|
||||
daemon_reload: true
|
||||
loop:
|
||||
- adguardhome.service
|
||||
- icloudpd.service
|
||||
- podman-auto-update.timer
|
||||
loop_control:
|
||||
label: "{{ item }}"
|
||||
|
||||
@@ -1,20 +0,0 @@
|
||||
# Managed by Ansible. Do not edit manually.
|
||||
[Unit]
|
||||
Description=iCloud Photos Downloader
|
||||
Wants=network-online.target
|
||||
After=network-online.target
|
||||
|
||||
[Container]
|
||||
Image={{ aegis_icloudpd_image }}
|
||||
Environment=apple_id={{ aegis_icloudpd_apple_id }}
|
||||
Environment=folder_structure={{ aegis_icloudpd_folder_structure }}
|
||||
Environment=synchronisation_interval={{ aegis_icloudpd_synchronisation_interval }}
|
||||
Volume=/var/lib/icloudpd/data:/home/root/iCloud:Z
|
||||
Volume=/var/lib/icloudpd/config:/config:Z
|
||||
AutoUpdate=registry
|
||||
|
||||
[Service]
|
||||
Restart=always
|
||||
|
||||
[Install]
|
||||
WantedBy=multi-user.target
|
||||
@@ -1,5 +1,29 @@
|
||||
---
|
||||
atlas_manage_storage: false
|
||||
atlas_manage_nextcloud: false
|
||||
atlas_nextcloud_root: "{{ atlas_app_data_mountpoint }}/nextcloud"
|
||||
atlas_nextcloud_dataset: "{{ atlas_zfs_pool }}/services/data/nextcloud"
|
||||
atlas_nextcloud_domain: ""
|
||||
atlas_onlyoffice_domain: ""
|
||||
atlas_nextcloud_http_port: 8080
|
||||
atlas_onlyoffice_http_port: 8081
|
||||
atlas_nextcloud_network_subnet: 10.90.10.0/24
|
||||
atlas_nextcloud_network_gateway: 10.90.10.1
|
||||
atlas_nextcloud_quadlet_dir: "{{ atlas_admin_home }}/.config/containers/systemd"
|
||||
atlas_nextcloud_private_dir: "{{ atlas_admin_home }}/.config/atlas-nextcloud"
|
||||
atlas_nextcloud_app_cache: "{{ atlas_admin_home }}/.cache/atlas-nextcloud-apps"
|
||||
atlas_nextcloud_image: ""
|
||||
atlas_nextcloud_postgres_image: ""
|
||||
atlas_nextcloud_redis_image: ""
|
||||
atlas_onlyoffice_image: ""
|
||||
atlas_nextcloud_admin: admin
|
||||
atlas_nextcloud_users: []
|
||||
atlas_nextcloud_apps: []
|
||||
atlas_nextcloud_services:
|
||||
- atlas-nextcloud-db.service
|
||||
- atlas-nextcloud-redis.service
|
||||
- atlas-nextcloud.service
|
||||
- atlas-onlyoffice.service
|
||||
atlas_manage_sharing: false
|
||||
# Destructive first-boot action; normally false once the pool exists.
|
||||
atlas_create_pool: false
|
||||
@@ -97,6 +121,49 @@ atlas_borg_cache_dir: /var/cache/atlas-borg
|
||||
atlas_borg_lock_path: /var/lib/atlas-borg/backup.lock
|
||||
atlas_borg_recovery_export_path: "{{ playbook_dir }}/../secrets/recovery/atlas-borg-repokey.export"
|
||||
|
||||
# Manual-only offline backup. No USB device is formatted or mounted by Ansible.
|
||||
atlas_manage_usb_backup: false
|
||||
atlas_usb_backup_luks_uuid: ""
|
||||
atlas_usb_backup_fs_uuid: ""
|
||||
atlas_usb_backup_mapper_name: atlas-usb-backup
|
||||
atlas_usb_backup_min_free_bytes: 10737418240
|
||||
atlas_usb_backup_snapshot_prefix: atlas-usb
|
||||
atlas_manage_usb_reminder: false
|
||||
atlas_usb_reminder_calendar: ""
|
||||
atlas_usb_reminder_notifier: /opt/45drives/houston/houston-notify
|
||||
|
||||
# Read-only health probes and 45Drives Alerts; disabled outside Atlas host vars.
|
||||
atlas_manage_monitoring: false
|
||||
atlas_monitor_calendar: "*:0/30"
|
||||
atlas_monitor_notifier: "{{ atlas_usb_reminder_notifier }}"
|
||||
atlas_monitor_smart_devices: []
|
||||
atlas_monitor_timers: []
|
||||
atlas_monitor_failure_units: []
|
||||
atlas_monitor_effective_timers: >-
|
||||
{{ atlas_monitor_timers
|
||||
+ ([{'name': 'atlas-prometheus-pull.timer', 'max_age_hours': 26}]
|
||||
if atlas_manage_prometheus_backup_pull | bool and atlas_prometheus_pull_start_timer | bool
|
||||
else []) }}
|
||||
atlas_monitor_effective_failure_units: >-
|
||||
{{ atlas_monitor_failure_units
|
||||
+ (['atlas-prometheus-pull.service']
|
||||
if atlas_manage_prometheus_backup_pull | bool else []) }}
|
||||
atlas_monitor_remote_capacity: {}
|
||||
atlas_monitor_pool_warning_percent: 80
|
||||
atlas_monitor_pool_critical_percent: 90
|
||||
atlas_monitor_root_warning_percent: 80
|
||||
atlas_monitor_root_critical_percent: 90
|
||||
atlas_monitor_snapshot_warning_percent: 10
|
||||
atlas_monitor_snapshot_critical_percent: 20
|
||||
atlas_monitor_snapshot_growth_warning_gib_day: 100
|
||||
atlas_monitor_backup_growth_warning_gib_day: 100
|
||||
atlas_monitor_cpu_warning_c: 85
|
||||
atlas_monitor_cpu_critical_c: 95
|
||||
atlas_monitor_borg_max_runtime_days: 14
|
||||
|
||||
# Explicit post-restore relabeling only; never relabel datasets during ordinary runs.
|
||||
atlas_restorecon_paths: []
|
||||
|
||||
atlas_archive_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_archive }}"
|
||||
atlas_services_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_services }}"
|
||||
atlas_app_data_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_app_data }}"
|
||||
@@ -107,8 +174,56 @@ atlas_music_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_music }}"
|
||||
atlas_backup_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_backup }}"
|
||||
atlas_host_backups_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_host_backups }}"
|
||||
atlas_backup_prometheus_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_backup_prometheus }}"
|
||||
atlas_manage_prometheus_backup_pull: false
|
||||
atlas_prometheus_pull_ssh_dir: /etc/atlas-prometheus-pull
|
||||
atlas_prometheus_pull_private_key_path: "{{ atlas_prometheus_pull_ssh_dir }}/id_ed25519"
|
||||
atlas_prometheus_pull_known_hosts_path: "{{ atlas_prometheus_pull_ssh_dir }}/known_hosts"
|
||||
atlas_prometheus_ssh_host_key: ""
|
||||
atlas_prometheus_pull_source_user: prometheus-backup
|
||||
atlas_prometheus_pull_source_port: 22
|
||||
atlas_prometheus_pull_calendar: "*-*-* 03:00:00 Europe/Rome"
|
||||
atlas_prometheus_pull_start_timer: false
|
||||
atlas_prometheus_pull_keep_daily: 30
|
||||
atlas_prometheus_pull_keep_weekly: 8
|
||||
atlas_prometheus_pull_keep_monthly: 12
|
||||
atlas_prometheus_pull_max_age_hours: 24
|
||||
atlas_photobook_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_photobook }}"
|
||||
|
||||
# Rootless Gitea runs in admin's user manager; the image maps internal gitea to UID/GID 1000.
|
||||
atlas_manage_gitea: false
|
||||
atlas_gitea_username: "{{ atlas_admin_username }}"
|
||||
atlas_gitea_group: "{{ atlas_admin_group }}"
|
||||
atlas_gitea_uid: "{{ atlas_admin_uid }}"
|
||||
atlas_gitea_gid: "{{ atlas_admin_gid }}"
|
||||
atlas_gitea_home: "{{ atlas_admin_home }}"
|
||||
atlas_gitea_container_uid: 1000
|
||||
atlas_gitea_container_gid: 1000
|
||||
atlas_gitea_legacy_username: gitea
|
||||
atlas_gitea_dataset: "{{ atlas_zfs_pool }}/services/data/gitea"
|
||||
atlas_gitea_mountpoint: "{{ atlas_app_data_mountpoint }}/gitea"
|
||||
atlas_gitea_quadlet_dir: "{{ atlas_gitea_home }}/.config/containers/systemd"
|
||||
atlas_gitea_image: localhost/atlas-gitea:1.25.2-user-gitea-v1
|
||||
atlas_gitea_image_build_dir: "{{ atlas_gitea_home }}/.local/share/atlas-gitea-image"
|
||||
atlas_gitea_production_enabled: false
|
||||
atlas_gitea_public_domain: ""
|
||||
atlas_gitea_bind_address: "{{ ansible_host }}"
|
||||
atlas_gitea_http_port: 3000
|
||||
atlas_gitea_ssh_port: 2222
|
||||
atlas_gitea_staging_bind_address: 127.0.0.1
|
||||
atlas_gitea_staging_http_port: 3001
|
||||
atlas_gitea_staging_ssh_port: 2223
|
||||
|
||||
# Declare storage and an inactive Quadlet only. The operator supplies the
|
||||
# private configuration, handles MFA, and starts the user service manually.
|
||||
atlas_icloudpd_dataset: "{{ atlas_zfs_pool }}/services/data/icloudpd"
|
||||
atlas_icloudpd_state_dir: "{{ atlas_app_data_mountpoint }}/icloudpd"
|
||||
atlas_icloudpd_config_dir: "{{ atlas_icloudpd_state_dir }}/config"
|
||||
atlas_icloudpd_photos_dir: "{{ atlas_archive_mountpoint }}/Pictures/iCloudPD"
|
||||
atlas_icloudpd_image: >-
|
||||
docker.io/boredazfcuk/icloudpd@sha256:9966c31ddf0b5b306ac2410b4edd5d626806d96e80c92b83cbb689972dc9389f
|
||||
atlas_icloudpd_quadlet_dir: "{{ atlas_admin_home }}/.config/containers/systemd"
|
||||
atlas_icloudpd_timezone: Europe/Rome
|
||||
|
||||
atlas_45drives_repo_url: https://repo.45drives.com/repofiles/rocky/45drives-enterprise.repo
|
||||
atlas_45drives_repo_file: /etc/yum.repos.d/45drives-enterprise.repo
|
||||
atlas_45drives_packages:
|
||||
|
||||
@@ -0,0 +1,10 @@
|
||||
FROM docker.gitea.com/gitea@sha256:f1943db2d2f1e447e857b3f0aee4ebb7b184500f86e5b80eae110fd435435906
|
||||
|
||||
# Preserve the official image's UID/GID, paths and entrypoint; change only the
|
||||
# internal Unix identity. The host-side rootless owner is Atlas admin.
|
||||
USER 0
|
||||
RUN sed -i 's/^git:x:1000:1000:/gitea:x:1000:1000:/' /etc/passwd \
|
||||
&& sed -i 's/^git:x:1000:/gitea:x:1000:/' /etc/group \
|
||||
&& grep -q '^gitea:x:1000:1000:' /etc/passwd \
|
||||
&& grep -q '^gitea:x:1000:' /etc/group
|
||||
USER 1000:1000
|
||||
59
ansible/roles/profile_atlas/files/atlas-borg-progress.py
Normal file
59
ansible/roles/profile_atlas/files/atlas-borg-progress.py
Normal file
@@ -0,0 +1,59 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Turn Borg's JSON progress stream into bounded, readable journal entries."""
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import sys
|
||||
import time
|
||||
|
||||
|
||||
def size(value):
|
||||
if not isinstance(value, (int, float)):
|
||||
return "unknown"
|
||||
return f"{value / (1024 ** 3):.2f} GiB"
|
||||
|
||||
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--estimated-total-bytes", type=int, required=True)
|
||||
args = parser.parse_args()
|
||||
if args.estimated_total_bytes <= 0:
|
||||
parser.error("estimated total must be positive")
|
||||
|
||||
last_progress = 0.0
|
||||
for line in sys.stdin:
|
||||
try:
|
||||
event = json.loads(line)
|
||||
except json.JSONDecodeError:
|
||||
print(line.rstrip(), flush=True)
|
||||
continue
|
||||
|
||||
kind = event.get("type")
|
||||
if kind == "archive_progress":
|
||||
now = time.monotonic()
|
||||
if now - last_progress < 60 and not event.get("finished"):
|
||||
continue
|
||||
path = event.get("path") or ""
|
||||
parts = path.split("/")
|
||||
dataset = parts[1] if len(parts) > 1 and parts[0] == "source" else "unknown"
|
||||
original_size = event.get("original_size")
|
||||
if isinstance(original_size, (int, float)) and original_size >= 0:
|
||||
percent = original_size / args.estimated_total_bytes * 100
|
||||
estimated_progress = (
|
||||
f"{percent:.1f}%" if percent < 100 else ">=100% (ZFS estimate exceeded)"
|
||||
)
|
||||
else:
|
||||
estimated_progress = "unknown"
|
||||
print(
|
||||
"Borg create progress: "
|
||||
f"estimated={estimated_progress} dataset={dataset} "
|
||||
f"files={event.get('nfiles', 'unknown')} "
|
||||
f"original={size(original_size)} "
|
||||
f"compressed={size(event.get('compressed_size'))} "
|
||||
f"deduplicated={size(event.get('deduplicated_size'))}",
|
||||
flush=True,
|
||||
)
|
||||
last_progress = now
|
||||
elif kind == "log_message":
|
||||
print(f"Borg {event.get('levelname', 'INFO')}: {event.get('message', '')}", flush=True)
|
||||
elif kind == "progress_message" and event.get("message"):
|
||||
print(f"Borg: {event['message']}", flush=True)
|
||||
396
ansible/roles/profile_atlas/files/atlas-health-monitor.py
Normal file
396
ansible/roles/profile_atlas/files/atlas-health-monitor.py
Normal file
@@ -0,0 +1,396 @@
|
||||
#!/usr/bin/python3
|
||||
"""Read-only Atlas health probes with deduplicated 45Drives Alerts."""
|
||||
|
||||
import argparse
|
||||
import fcntl
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
import time
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
CONFIG_PATH = Path("/etc/atlas-health-monitor.json")
|
||||
STATE_DIR = Path("/var/lib/atlas-health-monitor")
|
||||
STATE_PATH = STATE_DIR / "state.json"
|
||||
GIB = 1024**3
|
||||
|
||||
|
||||
def run(*argv, timeout=40):
|
||||
return subprocess.run(argv, capture_output=True, text=True, timeout=timeout, check=False)
|
||||
|
||||
|
||||
def issue(issues, key, severity, message):
|
||||
issues[key] = {"severity": severity, "message": message}
|
||||
|
||||
|
||||
def notify(config, event, severity, subject, message):
|
||||
now = datetime.now(timezone.utc)
|
||||
payload = {
|
||||
"timestamp": now.isoformat(timespec="seconds"),
|
||||
"unixtime": int(now.timestamp()),
|
||||
"event": event,
|
||||
"severity": severity,
|
||||
"subject": subject,
|
||||
"email_message": message,
|
||||
}
|
||||
result = run(config["notifier"], json.dumps(payload, ensure_ascii=False), timeout=30)
|
||||
if result.returncode:
|
||||
raise RuntimeError(f"45Drives notifier exited {result.returncode}: {result.stderr.strip()}")
|
||||
|
||||
|
||||
def parse_fields(text):
|
||||
return dict(line.split("=", 1) for line in text.splitlines() if "=" in line)
|
||||
|
||||
|
||||
def systemd_fields(unit, *properties):
|
||||
result = run("systemctl", "show", unit, *(f"-p{item}" for item in properties))
|
||||
if result.returncode:
|
||||
raise RuntimeError(f"systemctl show {unit} exited {result.returncode}")
|
||||
return parse_fields(result.stdout)
|
||||
|
||||
|
||||
def unix_time(text):
|
||||
if not text or text == "n/a":
|
||||
return None
|
||||
result = run("date", "-d", text, "+%s")
|
||||
if result.returncode:
|
||||
raise ValueError(f"Cannot parse systemd timestamp: {text}")
|
||||
return int(result.stdout.strip())
|
||||
|
||||
|
||||
def check_pool(config, issues, measurements):
|
||||
pool = config["pool"]
|
||||
listing = run("zpool", "list", "-H", "-p", "-o", "size,alloc,capacity,health", pool)
|
||||
if listing.returncode:
|
||||
issue(issues, "pool.probe", "critical", f"Cannot query ZFS pool {pool}")
|
||||
return
|
||||
try:
|
||||
size, alloc, capacity, health = listing.stdout.strip().split("\t")
|
||||
size, alloc, capacity = int(size), int(alloc), int(capacity)
|
||||
except (ValueError, TypeError):
|
||||
issue(issues, "pool.probe", "critical", "Invalid ZFS pool capacity response")
|
||||
return
|
||||
measurements.update(pool_size_bytes=size, pool_alloc_bytes=alloc, pool_capacity_percent=capacity)
|
||||
if health != "ONLINE":
|
||||
issue(issues, "pool.health", "critical", f"ZFS pool {pool} state is {health}")
|
||||
if capacity >= config["pool_critical_percent"]:
|
||||
issue(issues, "pool.capacity", "critical", f"ZFS pool {pool} is {capacity}% full")
|
||||
elif capacity >= config["pool_warning_percent"]:
|
||||
issue(issues, "pool.capacity", "warning", f"ZFS pool {pool} is {capacity}% full")
|
||||
|
||||
status = run("zpool", "status", "-P", pool)
|
||||
if status.returncode:
|
||||
issue(issues, "pool.status", "critical", f"Cannot query detailed ZFS status for {pool}")
|
||||
return
|
||||
bad_vdevs = []
|
||||
for line in status.stdout.splitlines():
|
||||
match = re.match(r"^\s*(\S+)\s+(ONLINE|DEGRADED|FAULTED|OFFLINE|UNAVAIL|REMOVED)\s+(\d+)\s+(\d+)\s+(\d+)", line)
|
||||
if match:
|
||||
name, state, reads, writes, checksums = match.groups()
|
||||
if state != "ONLINE" or any(int(value) for value in (reads, writes, checksums)):
|
||||
bad_vdevs.append(f"{name}: {state}, READ={reads}, WRITE={writes}, CKSUM={checksums}")
|
||||
if bad_vdevs:
|
||||
issue(issues, "pool.vdevs", "critical", "ZFS vdev errors: " + "; ".join(bad_vdevs))
|
||||
errors = re.search(r"^errors:\s*(.*)$", status.stdout, re.MULTILINE)
|
||||
if not errors or errors.group(1).strip() != "No known data errors":
|
||||
issue(issues, "pool.data_errors", "critical", "ZFS status reports data errors; inspect zpool status -v")
|
||||
if re.search(r"^\s*scan:\s*resilver in progress", status.stdout, re.MULTILINE | re.IGNORECASE):
|
||||
issue(issues, "pool.resilver", "warning", "ZFS resilver is in progress; inspect zpool status")
|
||||
scan = re.search(r"^\s*scan:\s*(.*)$", status.stdout, re.MULTILINE)
|
||||
if scan and re.search(r"\bwith [1-9][0-9]* errors\b", scan.group(1)):
|
||||
issue(issues, "pool.scan_errors", "critical", f"ZFS scan reported errors: {scan.group(1)}")
|
||||
|
||||
|
||||
def check_capacity(config, issues, measurements):
|
||||
pool = config["pool"]
|
||||
listing = run("zfs", "list", "-H", "-p", "-o", "name,usedbysnapshots", "-r", pool)
|
||||
if listing.returncode:
|
||||
issue(issues, "snapshot.probe", "warning", "Cannot query ZFS snapshot space")
|
||||
else:
|
||||
try:
|
||||
snapshots = sum(int(line.split("\t")[1]) for line in listing.stdout.splitlines())
|
||||
measurements["snapshots_bytes"] = snapshots
|
||||
size = measurements.get("pool_size_bytes")
|
||||
if size:
|
||||
percent = snapshots * 100 // size
|
||||
measurements["snapshots_percent"] = percent
|
||||
if percent >= config["snapshot_critical_percent"]:
|
||||
issue(issues, "snapshot.capacity", "critical", f"Snapshots use {percent}% of pool size")
|
||||
elif percent >= config["snapshot_warning_percent"]:
|
||||
issue(issues, "snapshot.capacity", "warning", f"Snapshots use {percent}% of pool size")
|
||||
except (ValueError, IndexError):
|
||||
issue(issues, "snapshot.probe", "warning", "Invalid ZFS snapshot-space response")
|
||||
backup = run("zfs", "list", "-H", "-p", "-o", "used", config["backup_dataset"])
|
||||
if backup.returncode:
|
||||
issue(issues, "backup.capacity_probe", "warning", "Cannot query local backup dataset space")
|
||||
else:
|
||||
try:
|
||||
measurements["backup_bytes"] = int(backup.stdout.strip())
|
||||
except ValueError:
|
||||
issue(issues, "backup.capacity_probe", "warning", "Invalid local backup space response")
|
||||
|
||||
try:
|
||||
filesystem = os.statvfs("/")
|
||||
total = filesystem.f_blocks * filesystem.f_frsize
|
||||
available = filesystem.f_bavail * filesystem.f_frsize
|
||||
used_percent = (total - available) * 100 // total
|
||||
measurements["root_capacity_percent"] = used_percent
|
||||
if used_percent >= config["root_critical_percent"]:
|
||||
issue(issues, "root.capacity", "critical", f"Atlas system filesystem is {used_percent}% full")
|
||||
elif used_percent >= config["root_warning_percent"]:
|
||||
issue(issues, "root.capacity", "warning", f"Atlas system filesystem is {used_percent}% full")
|
||||
except (OSError, ZeroDivisionError):
|
||||
issue(issues, "root.capacity_probe", "warning", "Cannot query Atlas system filesystem space")
|
||||
|
||||
|
||||
def check_remote_capacity(config, issues, measurements):
|
||||
"""Query only the Storage Box quota; do not open or inspect the Borg repository."""
|
||||
remote = config["remote_capacity"]
|
||||
try:
|
||||
result = run("runuser", "-u", remote["run_as"], "--", remote["ssh_wrapper"],
|
||||
f"{remote['user']}@{remote['host']}", "df", "-m", timeout=65)
|
||||
if result.returncode:
|
||||
raise ValueError(f"SSH df exited {result.returncode}")
|
||||
lines = result.stdout.strip().splitlines()
|
||||
if len(lines) != 2:
|
||||
raise ValueError("Unexpected Storage Box df output")
|
||||
fields = lines[1].split()
|
||||
if len(fields) < 5:
|
||||
raise ValueError("Incomplete Storage Box df output")
|
||||
total_mib, used_mib, available_mib = (int(value) for value in fields[1:4])
|
||||
percent = int(fields[4].rstrip("%"))
|
||||
if total_mib <= 0 or not 0 <= percent <= 100 or available_mib < 0:
|
||||
raise ValueError("Invalid Storage Box quota values")
|
||||
except (OSError, ValueError, subprocess.TimeoutExpired):
|
||||
issue(issues, "remote.capacity_probe", "warning", "Cannot query Hetzner Storage Box quota via pinned-key SSH")
|
||||
return
|
||||
measurements.update(remote_capacity_percent=percent, remote_bytes=used_mib * 1024**2,
|
||||
remote_available_bytes=available_mib * 1024**2)
|
||||
if percent >= remote["critical_percent"]:
|
||||
issue(issues, "remote.capacity", "critical", f"Hetzner Storage Box quota is {percent}% full")
|
||||
elif percent >= remote["warning_percent"]:
|
||||
issue(issues, "remote.capacity", "warning", f"Hetzner Storage Box quota is {percent}% full")
|
||||
|
||||
|
||||
def check_smart(config, issues, measurements):
|
||||
for device in config["smart_devices"]:
|
||||
name, path = device["name"], device["path"]
|
||||
try:
|
||||
result = run("smartctl", "-j", "-a", path, timeout=60)
|
||||
data = json.loads(result.stdout)
|
||||
status = int(data.get("smartctl", {}).get("exit_status", result.returncode))
|
||||
except (subprocess.TimeoutExpired, json.JSONDecodeError, ValueError) as exc:
|
||||
issue(issues, f"smart.{name}.probe", "critical", f"SMART probe failed for {name}: {type(exc).__name__}")
|
||||
continue
|
||||
if status:
|
||||
severity = "critical" if status & 0b00001111 else "warning"
|
||||
issue(issues, f"smart.{name}.status", severity, f"SMART reported exit status {status} for {name}")
|
||||
passed = data.get("smart_status", {}).get("passed")
|
||||
if passed is False:
|
||||
issue(issues, f"smart.{name}.health", "critical", f"SMART self-assessment failed for {name}")
|
||||
elif passed is None:
|
||||
issue(issues, f"smart.{name}.health", "warning", f"SMART self-assessment unavailable for {name}")
|
||||
temperature = data.get("temperature", {}).get("current")
|
||||
if isinstance(temperature, (int, float)):
|
||||
measurements[f"smart_{name}_c"] = temperature
|
||||
if temperature >= device["critical_c"]:
|
||||
issue(issues, f"smart.{name}.temperature", "critical", f"{name} temperature is {temperature} C")
|
||||
elif temperature >= device["warning_c"]:
|
||||
issue(issues, f"smart.{name}.temperature", "warning", f"{name} temperature is {temperature} C")
|
||||
else:
|
||||
issue(issues, f"smart.{name}.temperature", "warning", f"Temperature unavailable for {name}")
|
||||
for attribute in data.get("ata_smart_attributes", {}).get("table", []):
|
||||
attribute_id = attribute.get("id")
|
||||
if attribute_id in (5, 187, 197, 198):
|
||||
raw = attribute.get("raw", {}).get("value", 0)
|
||||
if isinstance(raw, int) and raw > 0:
|
||||
severity = "critical" if attribute_id in (197, 198) else "warning"
|
||||
issue(issues, f"smart.{name}.ata_{attribute_id}", severity,
|
||||
f"{name} SMART attribute {attribute_id} raw count is {raw}")
|
||||
nvme = data.get("nvme_smart_health_information_log", {})
|
||||
if isinstance(nvme, dict):
|
||||
if int(nvme.get("critical_warning", 0)):
|
||||
issue(issues, f"smart.{name}.nvme_warning", "critical", f"{name} NVMe critical warning is nonzero")
|
||||
if int(nvme.get("media_errors", 0)):
|
||||
issue(issues, f"smart.{name}.nvme_media", "critical", f"{name} NVMe media errors are nonzero")
|
||||
|
||||
|
||||
def check_cpu(config, issues, measurements):
|
||||
sensors = []
|
||||
for hwmon in Path("/sys/class/hwmon").glob("hwmon*"):
|
||||
try:
|
||||
if (hwmon / "name").read_text().strip() != "coretemp":
|
||||
continue
|
||||
sensors.extend(int(path.read_text().strip()) / 1000 for path in hwmon.glob("temp*_input"))
|
||||
except (OSError, ValueError):
|
||||
continue
|
||||
if not sensors:
|
||||
issue(issues, "cpu.temperature_probe", "warning", "CPU temperature sensors are unavailable")
|
||||
return
|
||||
hottest = max(sensors)
|
||||
measurements["cpu_max_c"] = hottest
|
||||
if hottest >= config["cpu_critical_c"]:
|
||||
issue(issues, "cpu.temperature", "critical", f"CPU temperature is {hottest:g} C")
|
||||
elif hottest >= config["cpu_warning_c"]:
|
||||
issue(issues, "cpu.temperature", "warning", f"CPU temperature is {hottest:g} C")
|
||||
|
||||
|
||||
def check_jobs(config, issues, measurements, now):
|
||||
for timer in config["timers"]:
|
||||
name = timer["name"]
|
||||
try:
|
||||
fields = systemd_fields(name, "ActiveState", "UnitFileState", "LastTriggerUSec", "ActiveEnterTimestamp")
|
||||
if fields.get("ActiveState") != "active" or fields.get("UnitFileState") != "enabled":
|
||||
issue(issues, f"timer.{name}", "critical", f"Timer {name} is not active and enabled")
|
||||
max_age = int(timer["max_age_hours"]) * 3600
|
||||
if max_age:
|
||||
last = unix_time(fields.get("LastTriggerUSec"))
|
||||
if last is None:
|
||||
last = unix_time(fields.get("ActiveEnterTimestamp"))
|
||||
if last is not None and now - last > max_age:
|
||||
issue(issues, f"timer.{name}.stale", "warning",
|
||||
f"Timer {name} has not fired in {int((now-last)/3600)} hours")
|
||||
except (RuntimeError, ValueError, subprocess.TimeoutExpired):
|
||||
issue(issues, f"timer.{name}.probe", "warning", f"Cannot query timer {name}")
|
||||
for unit in config["failure_units"]:
|
||||
if unit.endswith("@.service"):
|
||||
continue
|
||||
try:
|
||||
fields = systemd_fields(unit, "ActiveState", "Result", "ExecMainStartTimestamp")
|
||||
state = fields.get("ActiveState")
|
||||
if state == "failed" or (state == "inactive" and fields.get("Result") not in (None, "", "success")):
|
||||
issue(issues, f"service.{unit}", "critical", f"Service {unit} failed: {fields.get('Result')}")
|
||||
if unit == "atlas-borg-backup.service" and fields.get("ActiveState") == "activating":
|
||||
started = unix_time(fields.get("ExecMainStartTimestamp"))
|
||||
if started is not None and now - started > config["borg_max_runtime_days"] * 86400:
|
||||
issue(issues, "backup.borg_long_running", "warning",
|
||||
"Borg has run longer than its configured limit")
|
||||
except (RuntimeError, ValueError, subprocess.TimeoutExpired):
|
||||
issue(issues, f"service.{unit}.probe", "warning", f"Cannot query service {unit}")
|
||||
|
||||
|
||||
def check_growth(config, issues, measurements, samples, now):
|
||||
previous = [sample for sample in samples if 20 * 3600 <= now - sample.get("time", now) <= 48 * 3600]
|
||||
if previous:
|
||||
baseline = min(previous, key=lambda sample: abs(now - sample["time"] - 86400))
|
||||
days = (now - baseline["time"]) / 86400
|
||||
for name, threshold in (("snapshots", config["snapshot_growth_warning_gib_day"]),
|
||||
("backup", config["backup_growth_warning_gib_day"]),
|
||||
("remote", config["remote_capacity"]["growth_warning_gib_day"])):
|
||||
current, old = measurements.get(f"{name}_bytes"), baseline.get(f"{name}_bytes")
|
||||
if isinstance(current, int) and isinstance(old, int) and days > 0:
|
||||
growth_gib_day = (current - old) / GIB / days
|
||||
measurements[f"{name}_growth_gib_day"] = round(growth_gib_day, 1)
|
||||
if growth_gib_day >= threshold:
|
||||
issue(issues, f"{name}.growth", "warning",
|
||||
f"Local {name} usage grew {growth_gib_day:.1f} GiB/day over {days:.1f} days")
|
||||
|
||||
|
||||
def allowed_failure_unit(config, unit):
|
||||
for allowed in config["failure_units"]:
|
||||
if allowed == unit:
|
||||
return True
|
||||
if allowed.endswith("@.service") and unit.startswith(allowed[:-9] + "@") and unit.endswith(".service"):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def load_state():
|
||||
if not STATE_PATH.exists():
|
||||
return {"active": {}, "samples": []}
|
||||
with STATE_PATH.open(encoding="utf-8") as stream:
|
||||
state = json.load(stream)
|
||||
if not isinstance(state.get("active"), dict) or not isinstance(state.get("samples"), list):
|
||||
raise ValueError("Invalid Atlas monitor state; refusing to overwrite it")
|
||||
return state
|
||||
|
||||
|
||||
def save_state(state):
|
||||
with tempfile.NamedTemporaryFile("w", dir=STATE_DIR, prefix=".state-", delete=False,
|
||||
encoding="utf-8") as stream:
|
||||
path = Path(stream.name)
|
||||
os.chmod(path, 0o600)
|
||||
json.dump(state, stream, sort_keys=True)
|
||||
stream.write("\n")
|
||||
stream.flush()
|
||||
os.fsync(stream.fileno())
|
||||
os.replace(path, STATE_PATH)
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--dry-run", action="store_true", help="probe without notifications or state changes")
|
||||
parser.add_argument("--test-notification", action="store_true", help="submit a labelled test alert")
|
||||
parser.add_argument("--job-failed", metavar="UNIT", help="notify about a failed configured service")
|
||||
args = parser.parse_args()
|
||||
with CONFIG_PATH.open(encoding="utf-8") as stream:
|
||||
config = json.load(stream)
|
||||
if args.test_notification:
|
||||
notify(config, "atlas_monitor_test", "warning", "Test monitoraggio Atlas",
|
||||
"Notifica di prova: il monitoraggio Atlas raggiunge 45Drives Alerts. Non conferma l'invio email.")
|
||||
print("Atlas monitor test submitted to 45Drives Alerts; email delivery is not verified.")
|
||||
return 0
|
||||
if args.job_failed:
|
||||
if not allowed_failure_unit(config, args.job_failed):
|
||||
raise ValueError("Unconfigured Atlas failure unit")
|
||||
notify(config, "atlas_job_failed", "critical", f"Job Atlas fallito: {args.job_failed}",
|
||||
f"Il servizio {args.job_failed} e' fallito. Controlla: "
|
||||
f"sudo journalctl -u {args.job_failed} -n 100 --no-pager")
|
||||
print(f"Atlas job failure submitted to 45Drives Alerts: {args.job_failed}")
|
||||
return 0
|
||||
|
||||
now = int(time.time())
|
||||
issues, measurements = {}, {}
|
||||
check_pool(config, issues, measurements)
|
||||
check_capacity(config, issues, measurements)
|
||||
check_remote_capacity(config, issues, measurements)
|
||||
check_smart(config, issues, measurements)
|
||||
check_cpu(config, issues, measurements)
|
||||
check_jobs(config, issues, measurements, now)
|
||||
if args.dry_run:
|
||||
print(json.dumps({"issues": issues, "measurements": measurements}, sort_keys=True))
|
||||
return 0
|
||||
|
||||
STATE_DIR.mkdir(mode=0o700, exist_ok=True)
|
||||
with (STATE_DIR / "monitor.lock").open("w") as lock:
|
||||
fcntl.flock(lock, fcntl.LOCK_EX | fcntl.LOCK_NB)
|
||||
state = load_state()
|
||||
check_growth(config, issues, measurements, state["samples"], now)
|
||||
active, failed_notifications = state["active"], []
|
||||
for key, details in issues.items():
|
||||
old = active.get(key)
|
||||
if old is None or old.get("severity") != details["severity"]:
|
||||
try:
|
||||
notify(config, "atlas_health_issue", details["severity"],
|
||||
f"Atlas: {key}", details["message"])
|
||||
active[key] = details
|
||||
print(f"ALERT {details['severity']} {key}: {details['message']}", flush=True)
|
||||
except (RuntimeError, subprocess.TimeoutExpired) as exc:
|
||||
failed_notifications.append(key)
|
||||
print(f"NOTIFICATION FAILED {key}: {exc}", file=sys.stderr, flush=True)
|
||||
for key in set(active) - set(issues):
|
||||
print(f"RECOVERED {key}", flush=True)
|
||||
del active[key]
|
||||
state["samples"] = [sample for sample in state["samples"] if now - sample.get("time", 0) < 48 * 3600]
|
||||
state["samples"].append({"time": now, **{key: value for key, value in measurements.items()
|
||||
if key in ("snapshots_bytes", "backup_bytes", "remote_bytes")}})
|
||||
save_state(state)
|
||||
print(f"Atlas health: issues={len(issues)} notifications_failed={len(failed_notifications)} "
|
||||
f"pool={measurements.get('pool_capacity_percent', 'unknown')}% "
|
||||
f"remote={measurements.get('remote_capacity_percent', 'unknown')}% "
|
||||
f"snapshots={measurements.get('snapshots_bytes', 'unknown')} bytes "
|
||||
f"backup={measurements.get('backup_bytes', 'unknown')} bytes", flush=True)
|
||||
return 1 if failed_notifications else 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
try:
|
||||
sys.exit(main())
|
||||
except (OSError, RuntimeError, ValueError, subprocess.TimeoutExpired) as error:
|
||||
print(f"Atlas health monitor failed: {error}", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
56
ansible/roles/profile_atlas/files/atlas-prometheus-prune.py
Normal file
56
ansible/roles/profile_atlas/files/atlas-prometheus-prune.py
Normal file
@@ -0,0 +1,56 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Prune only verified, named Prometheus backup versions after publication."""
|
||||
|
||||
import datetime as dt
|
||||
import pathlib
|
||||
import re
|
||||
import shutil
|
||||
import sys
|
||||
|
||||
|
||||
def main() -> None:
|
||||
if len(sys.argv) != 5:
|
||||
raise SystemExit("Usage: atlas-prometheus-prune SNAPSHOTS DAILY WEEKLY MONTHLY")
|
||||
root = pathlib.Path(sys.argv[1])
|
||||
counts = [int(value) for value in sys.argv[2:]]
|
||||
if not root.is_dir() or root.is_symlink() or min(counts) < 1:
|
||||
raise SystemExit("Invalid backup directory or retention counts")
|
||||
versions = []
|
||||
for entry in root.iterdir():
|
||||
if not entry.is_dir() or entry.is_symlink():
|
||||
continue
|
||||
if not re.fullmatch(r"[0-9]{8}T[0-9]{6}Z", entry.name):
|
||||
continue
|
||||
try:
|
||||
when = dt.datetime.strptime(entry.name, "%Y%m%dT%H%M%SZ")
|
||||
except ValueError:
|
||||
continue
|
||||
if not all((entry / name).is_file() for name in ("payload.tar", "payload.sha256", "metadata.json")):
|
||||
continue
|
||||
versions.append((when, entry))
|
||||
versions.sort(reverse=True)
|
||||
if not versions:
|
||||
raise SystemExit("No published backup versions found; refusing to prune")
|
||||
|
||||
keep = {entry for _, entry in versions[: counts[0]]}
|
||||
for count, key in (
|
||||
(counts[1], lambda when: when.isocalendar()[:2]),
|
||||
(counts[2], lambda when: (when.year, when.month)),
|
||||
):
|
||||
periods = set()
|
||||
for when, entry in versions:
|
||||
period = key(when)
|
||||
if period in periods:
|
||||
continue
|
||||
periods.add(period)
|
||||
keep.add(entry)
|
||||
if len(periods) >= count:
|
||||
break
|
||||
|
||||
for _, entry in versions:
|
||||
if entry not in keep:
|
||||
shutil.rmtree(entry)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1,4 +1,14 @@
|
||||
---
|
||||
- name: Reload Atlas admin user manager
|
||||
become_user: "{{ atlas_admin_username }}"
|
||||
ansible.builtin.systemd:
|
||||
scope: user
|
||||
daemon_reload: true
|
||||
environment:
|
||||
XDG_RUNTIME_DIR: "/run/user/{{ atlas_admin_uid }}"
|
||||
DBUS_SESSION_BUS_ADDRESS: "unix:path=/run/user/{{ atlas_admin_uid }}/bus"
|
||||
when: not ansible_check_mode
|
||||
|
||||
- name: Reload SSH service
|
||||
ansible.builtin.systemd:
|
||||
name: sshd
|
||||
|
||||
@@ -224,7 +224,7 @@
|
||||
when: atlas_manage_borg_backup | bool
|
||||
|
||||
- name: Install the Atlas Borg backup helper
|
||||
tags: [atlas, storage, backup, borg]
|
||||
tags: [atlas, storage, backup, borg, borg_logging]
|
||||
ansible.builtin.template:
|
||||
src: atlas-borg-backup.sh.j2
|
||||
dest: /usr/local/sbin/atlas-borg-backup
|
||||
@@ -233,6 +233,16 @@
|
||||
mode: "0750"
|
||||
when: atlas_manage_borg_backup | bool
|
||||
|
||||
- name: Install the Atlas Borg snapshot cleanup helper
|
||||
tags: [atlas, storage, backup, borg, borg_logging]
|
||||
ansible.builtin.template:
|
||||
src: atlas-borg-snapshot-cleanup.sh.j2
|
||||
dest: /usr/local/sbin/atlas-borg-snapshot-cleanup
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0750"
|
||||
when: atlas_manage_borg_backup | bool
|
||||
|
||||
- name: Install the Atlas Borg check helper
|
||||
tags: [atlas, storage, backup, borg]
|
||||
ansible.builtin.template:
|
||||
@@ -244,7 +254,7 @@
|
||||
when: atlas_manage_borg_backup | bool
|
||||
|
||||
- name: Create the local libexec directory for the Atlas Borg SSH wrapper
|
||||
tags: [atlas, storage, backup, borg]
|
||||
tags: [atlas, storage, backup, borg, borg_logging]
|
||||
ansible.builtin.file:
|
||||
path: "{{ atlas_borg_ssh_wrapper_path | dirname }}"
|
||||
state: directory
|
||||
@@ -253,6 +263,16 @@
|
||||
mode: "0755"
|
||||
when: atlas_manage_borg_backup | bool
|
||||
|
||||
- name: Install the Atlas Borg progress formatter
|
||||
tags: [atlas, storage, backup, borg, borg_logging]
|
||||
ansible.builtin.copy:
|
||||
src: atlas-borg-progress.py
|
||||
dest: /usr/local/libexec/atlas-borg-progress
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0755"
|
||||
when: atlas_manage_borg_backup | bool
|
||||
|
||||
- name: Install the capability-dropping Atlas Borg SSH wrapper
|
||||
tags: [atlas, storage, backup, borg]
|
||||
ansible.builtin.template:
|
||||
@@ -264,7 +284,7 @@
|
||||
when: atlas_manage_borg_backup | bool
|
||||
|
||||
- name: Install Atlas Borg systemd units
|
||||
tags: [atlas, storage, backup, borg]
|
||||
tags: [atlas, storage, backup, borg, borg_logging]
|
||||
ansible.builtin.template:
|
||||
src: "{{ item }}.j2"
|
||||
dest: "/etc/systemd/system/{{ item }}"
|
||||
|
||||
159
ansible/roles/profile_atlas/tasks/gitea.yml
Normal file
159
ansible/roles/profile_atlas/tasks/gitea.yml
Normal file
@@ -0,0 +1,159 @@
|
||||
---
|
||||
- name: Prepare the isolated rootless Atlas Gitea target
|
||||
tags: [atlas, gitea]
|
||||
when: atlas_manage_gitea | bool
|
||||
block:
|
||||
- name: Require the existing Atlas application-data dataset
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- atlas_manage_storage | bool
|
||||
- atlas_gitea_dataset == atlas_zfs_pool ~ '/services/data/gitea'
|
||||
- atlas_gitea_mountpoint == atlas_app_data_mountpoint ~ '/gitea'
|
||||
- atlas_gitea_username == atlas_admin_username
|
||||
- atlas_gitea_group == atlas_admin_group
|
||||
- atlas_gitea_uid | int == atlas_admin_uid | int
|
||||
- atlas_gitea_gid | int == atlas_admin_gid | int
|
||||
- atlas_gitea_container_uid | int == 1000
|
||||
- atlas_gitea_container_gid | int == 1000
|
||||
- atlas_gitea_staging_bind_address == '127.0.0.1'
|
||||
- not (atlas_gitea_production_enabled | bool) or atlas_manage_firewall | bool
|
||||
- not (atlas_gitea_production_enabled | bool) or atlas_gitea_bind_address == ansible_host
|
||||
fail_msg: >-
|
||||
Rootless Gitea requires Atlas storage, the admin user manager, the
|
||||
dedicated dataset, and loopback-only staging ports.
|
||||
|
||||
- name: Inspect the final-restore marker before production activation
|
||||
ansible.builtin.stat:
|
||||
path: "{{ atlas_gitea_mountpoint }}/.final-sha256"
|
||||
register: atlas_gitea_final_marker
|
||||
when: atlas_gitea_production_enabled | bool
|
||||
|
||||
- name: Refuse production activation without the final consistent restore
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- atlas_gitea_final_marker.stat.isreg | default(false)
|
||||
fail_msg: Restore the final stopped-source Gitea export before enabling production.
|
||||
when: atlas_gitea_production_enabled | bool
|
||||
|
||||
- name: Verify the production Gitea dataset belongs to admin
|
||||
ansible.builtin.stat:
|
||||
path: "{{ atlas_gitea_mountpoint }}"
|
||||
register: atlas_gitea_dataset_owner
|
||||
when: atlas_gitea_production_enabled | bool
|
||||
|
||||
- name: Refuse to overlap the legacy host-account service
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- atlas_gitea_dataset_owner.stat.uid | int == atlas_admin_uid | int
|
||||
- atlas_gitea_dataset_owner.stat.gid | int == atlas_admin_gid | int
|
||||
fail_msg: >-
|
||||
The production dataset must already belong to admin before enabling
|
||||
the Quadlet; normal provisioning must not chown an active legacy service.
|
||||
when: atlas_gitea_production_enabled | bool
|
||||
|
||||
- name: Remove the retired account's parent-dataset traverse ACL
|
||||
ansible.posix.acl:
|
||||
path: "{{ item }}"
|
||||
etype: user
|
||||
entity: "{{ atlas_gitea_legacy_username }}"
|
||||
state: absent
|
||||
loop:
|
||||
- "{{ atlas_services_mountpoint }}"
|
||||
- "{{ atlas_app_data_mountpoint }}"
|
||||
when: atlas_gitea_production_enabled | bool
|
||||
|
||||
- name: Enable POSIX ACLs only on the service-namespace parents
|
||||
community.general.zfs:
|
||||
name: "{{ item }}"
|
||||
state: present
|
||||
extra_zfs_properties:
|
||||
acltype: posix
|
||||
loop:
|
||||
- "{{ atlas_zfs_pool }}/{{ atlas_zfs_dataset_services }}"
|
||||
- "{{ atlas_zfs_pool }}/{{ atlas_zfs_dataset_app_data }}"
|
||||
|
||||
- name: Create the dedicated Gitea ZFS dataset
|
||||
community.general.zfs:
|
||||
name: "{{ atlas_gitea_dataset }}"
|
||||
state: present
|
||||
extra_zfs_properties:
|
||||
compression: zstd
|
||||
mountpoint: "{{ atlas_gitea_mountpoint }}"
|
||||
|
||||
- name: Restrict the Gitea dataset and create rootless volume paths
|
||||
ansible.builtin.file:
|
||||
path: "{{ item }}"
|
||||
state: directory
|
||||
owner: "{{ atlas_gitea_username }}"
|
||||
group: "{{ atlas_gitea_group }}"
|
||||
mode: "0700"
|
||||
loop:
|
||||
- "{{ atlas_gitea_mountpoint }}"
|
||||
- "{{ atlas_gitea_mountpoint }}/data"
|
||||
- "{{ atlas_gitea_mountpoint }}/config"
|
||||
- "{{ atlas_gitea_home }}/.config"
|
||||
- "{{ atlas_gitea_home }}/.config/containers"
|
||||
- "{{ atlas_gitea_quadlet_dir }}"
|
||||
|
||||
- name: Ensure lingering for the admin rootless account
|
||||
ansible.builtin.command:
|
||||
argv:
|
||||
- loginctl
|
||||
- enable-linger
|
||||
- "{{ atlas_gitea_username }}"
|
||||
creates: "/var/lib/systemd/linger/{{ atlas_gitea_username }}"
|
||||
|
||||
- name: Start the admin rootless user manager
|
||||
ansible.builtin.systemd:
|
||||
name: "user@{{ atlas_gitea_uid }}.service"
|
||||
state: started
|
||||
when: not ansible_check_mode
|
||||
|
||||
- name: Prepare the admin-owned Gitea image
|
||||
ansible.builtin.import_tasks: gitea_image.yml
|
||||
|
||||
- name: Render the rootless Gitea Quadlet
|
||||
ansible.builtin.template:
|
||||
src: atlas-gitea.container.j2
|
||||
dest: "{{ atlas_gitea_quadlet_dir }}/atlas-gitea.container"
|
||||
owner: "{{ atlas_gitea_username }}"
|
||||
group: "{{ atlas_gitea_group }}"
|
||||
mode: "0644"
|
||||
|
||||
- name: Permit only Aegis to reach production Gitea HTTP and SSH
|
||||
ansible.posix.firewalld:
|
||||
rich_rule: >-
|
||||
rule family="ipv4" source address="{{ atlas_aegis_ip }}"
|
||||
port port="{{ item }}" protocol="tcp" accept
|
||||
zone: "{{ atlas_firewalld_zone }}"
|
||||
state: "{{ 'enabled' if atlas_gitea_production_enabled | bool else 'disabled' }}"
|
||||
permanent: true
|
||||
immediate: true
|
||||
loop:
|
||||
- "{{ atlas_gitea_http_port }}"
|
||||
- "{{ atlas_gitea_ssh_port }}"
|
||||
when: atlas_manage_firewall | bool
|
||||
|
||||
- name: Reload the rootless Gitea user manager without starting Gitea
|
||||
become_user: "{{ atlas_gitea_username }}"
|
||||
ansible.builtin.systemd:
|
||||
scope: user
|
||||
daemon_reload: true
|
||||
environment:
|
||||
XDG_RUNTIME_DIR: "/run/user/{{ atlas_gitea_uid }}"
|
||||
DBUS_SESSION_BUS_ADDRESS: "unix:path=/run/user/{{ atlas_gitea_uid }}/bus"
|
||||
when: not ansible_check_mode
|
||||
|
||||
- name: Start and enable the rootless Gitea user Quadlet after final restore
|
||||
become_user: "{{ atlas_gitea_username }}"
|
||||
ansible.builtin.systemd:
|
||||
name: atlas-gitea.service
|
||||
scope: user
|
||||
state: started
|
||||
enabled: true
|
||||
environment:
|
||||
XDG_RUNTIME_DIR: "/run/user/{{ atlas_gitea_uid }}"
|
||||
DBUS_SESSION_BUS_ADDRESS: "unix:path=/run/user/{{ atlas_gitea_uid }}/bus"
|
||||
when:
|
||||
- atlas_gitea_production_enabled | bool
|
||||
- not ansible_check_mode
|
||||
51
ansible/roles/profile_atlas/tasks/gitea_image.yml
Normal file
51
ansible/roles/profile_atlas/tasks/gitea_image.yml
Normal file
@@ -0,0 +1,51 @@
|
||||
---
|
||||
- name: Create the admin-owned Gitea image build directory
|
||||
ansible.builtin.file:
|
||||
path: "{{ atlas_gitea_image_build_dir }}"
|
||||
state: directory
|
||||
owner: "{{ atlas_admin_username }}"
|
||||
group: "{{ atlas_admin_group }}"
|
||||
mode: "0700"
|
||||
|
||||
- name: Install the pinned rootless Gitea Containerfile
|
||||
ansible.builtin.copy:
|
||||
src: Containerfile.gitea-rootless
|
||||
dest: "{{ atlas_gitea_image_build_dir }}/Containerfile"
|
||||
owner: "{{ atlas_admin_username }}"
|
||||
group: "{{ atlas_admin_group }}"
|
||||
mode: "0644"
|
||||
|
||||
- name: Check the admin-owned Gitea image
|
||||
become_user: "{{ atlas_admin_username }}"
|
||||
ansible.builtin.command:
|
||||
argv: [podman, image, exists, "{{ atlas_gitea_image }}"]
|
||||
args:
|
||||
chdir: "{{ atlas_gitea_image_build_dir }}"
|
||||
environment:
|
||||
HOME: "{{ atlas_admin_home }}"
|
||||
XDG_RUNTIME_DIR: "/run/user/{{ atlas_admin_uid }}"
|
||||
register: atlas_gitea_image_present
|
||||
changed_when: false
|
||||
failed_when: false
|
||||
check_mode: false
|
||||
|
||||
- name: Build the pinned Gitea image with the internal gitea identity
|
||||
become_user: "{{ atlas_admin_username }}"
|
||||
ansible.builtin.command:
|
||||
argv:
|
||||
- podman
|
||||
- build
|
||||
- --pull=always
|
||||
- --tag
|
||||
- "{{ atlas_gitea_image }}"
|
||||
- --file
|
||||
- Containerfile
|
||||
- .
|
||||
args:
|
||||
chdir: "{{ atlas_gitea_image_build_dir }}"
|
||||
environment:
|
||||
HOME: "{{ atlas_admin_home }}"
|
||||
XDG_RUNTIME_DIR: "/run/user/{{ atlas_admin_uid }}"
|
||||
when:
|
||||
- atlas_gitea_image_present.rc != 0
|
||||
- not ansible_check_mode
|
||||
58
ansible/roles/profile_atlas/tasks/gitea_public_domain.yml
Normal file
58
ansible/roles/profile_atlas/tasks/gitea_public_domain.yml
Normal file
@@ -0,0 +1,58 @@
|
||||
---
|
||||
- name: Manage the public domain of the restored production Gitea
|
||||
tags: [atlas, gitea, gitea_public_domain]
|
||||
when:
|
||||
- atlas_manage_gitea | bool
|
||||
- atlas_gitea_production_enabled | bool
|
||||
- atlas_gitea_public_domain | length > 0
|
||||
block:
|
||||
- name: Require an explicit public Gitea hostname
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- atlas_gitea_public_domain is match('^[a-zA-Z0-9][a-zA-Z0-9.-]*\.[a-zA-Z]{2,}$')
|
||||
|
||||
- name: Inspect the restored private Gitea configuration
|
||||
ansible.builtin.stat:
|
||||
path: "{{ atlas_gitea_mountpoint }}/config/app.ini"
|
||||
follow: false
|
||||
register: atlas_gitea_public_config
|
||||
|
||||
- name: Refuse to create or replace an unprepared Gitea configuration
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- atlas_gitea_public_config.stat.isreg | default(false)
|
||||
- atlas_gitea_public_config.stat.uid | int == atlas_gitea_uid | int
|
||||
- atlas_gitea_public_config.stat.mode == '0600'
|
||||
|
||||
# app.ini contains secrets: preserve all unrelated settings and suppress diffs.
|
||||
- name: Set only the declared public Gitea server fields
|
||||
community.general.ini_file:
|
||||
path: "{{ atlas_gitea_mountpoint }}/config/app.ini"
|
||||
section: server
|
||||
option: "{{ item.option }}"
|
||||
value: "{{ item.value }}"
|
||||
create: false
|
||||
backup: true
|
||||
owner: "{{ atlas_gitea_username }}"
|
||||
group: "{{ atlas_gitea_group }}"
|
||||
mode: "0600"
|
||||
loop:
|
||||
- { option: DOMAIN, value: "{{ atlas_gitea_public_domain }}" }
|
||||
- { option: ROOT_URL, value: "https://{{ atlas_gitea_public_domain }}/" }
|
||||
- { option: SSH_DOMAIN, value: "{{ atlas_gitea_public_domain }}" }
|
||||
register: atlas_gitea_public_domain_update
|
||||
no_log: true
|
||||
diff: false
|
||||
|
||||
- name: Restart only Gitea when its public configuration changes
|
||||
become_user: "{{ atlas_gitea_username }}"
|
||||
ansible.builtin.systemd:
|
||||
name: atlas-gitea.service
|
||||
scope: user
|
||||
state: restarted
|
||||
environment:
|
||||
XDG_RUNTIME_DIR: "/run/user/{{ atlas_gitea_uid }}"
|
||||
DBUS_SESSION_BUS_ADDRESS: "unix:path=/run/user/{{ atlas_gitea_uid }}/bus"
|
||||
when:
|
||||
- atlas_gitea_public_domain_update is changed
|
||||
- not ansible_check_mode
|
||||
188
ansible/roles/profile_atlas/tasks/icloudpd.yml
Normal file
188
ansible/roles/profile_atlas/tasks/icloudpd.yml
Normal file
@@ -0,0 +1,188 @@
|
||||
---
|
||||
- name: Require exact Atlas iCloudPD paths and rootless identity
|
||||
tags: [atlas, icloudpd]
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- atlas_manage_storage | bool
|
||||
- atlas_icloudpd_dataset == atlas_zfs_pool ~ '/services/data/icloudpd'
|
||||
- atlas_icloudpd_state_dir == atlas_app_data_mountpoint ~ '/icloudpd'
|
||||
- atlas_icloudpd_config_dir == atlas_icloudpd_state_dir ~ '/config'
|
||||
- atlas_icloudpd_photos_dir == atlas_archive_mountpoint ~ '/Pictures/iCloudPD'
|
||||
- atlas_admin_uid | int == 1000
|
||||
- atlas_admin_gid | int == 1000
|
||||
- atlas_icloudpd_image is search('@sha256:[0-9a-f]{64}$')
|
||||
fail_msg: Verify the fixed, separate Atlas iCloudPD photo and state paths.
|
||||
|
||||
- name: Declare rootless Atlas iCloudPD storage and boot-started Quadlet
|
||||
tags: [atlas, icloudpd]
|
||||
block:
|
||||
- name: Inspect the existing Archive and application-data datasets
|
||||
community.general.zfs_facts:
|
||||
name: "{{ item.dataset }}"
|
||||
properties: name,mounted,mountpoint
|
||||
loop:
|
||||
- dataset: "{{ atlas_zfs_pool }}/{{ atlas_zfs_dataset_archive }}"
|
||||
mountpoint: "{{ atlas_archive_mountpoint }}"
|
||||
- dataset: "{{ atlas_zfs_pool }}/{{ atlas_zfs_dataset_app_data }}"
|
||||
mountpoint: "{{ atlas_app_data_mountpoint }}"
|
||||
loop_control:
|
||||
label: "{{ item.dataset }}"
|
||||
register: atlas_icloudpd_parent_datasets
|
||||
|
||||
- name: Refuse missing or unmounted iCloudPD parent datasets
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- item.ansible_facts.ansible_zfs_datasets | length == 1
|
||||
- item.ansible_facts.ansible_zfs_datasets[0].mounted == 'yes'
|
||||
- item.ansible_facts.ansible_zfs_datasets[0].mountpoint == item.item.mountpoint
|
||||
loop: "{{ atlas_icloudpd_parent_datasets.results }}"
|
||||
loop_control:
|
||||
label: "{{ item.item.dataset }}"
|
||||
|
||||
- name: Inspect the existing Pictures namespace and proposed target
|
||||
ansible.builtin.stat:
|
||||
path: "{{ item }}"
|
||||
follow: false
|
||||
loop:
|
||||
- "{{ atlas_archive_mountpoint }}/Pictures"
|
||||
- "{{ atlas_icloudpd_photos_dir }}"
|
||||
- "{{ atlas_icloudpd_photos_dir }}/.atlas-icloudpd-managed"
|
||||
register: atlas_icloudpd_photo_paths
|
||||
|
||||
- name: Refuse to adopt unrelated Pictures data or a symlink
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- atlas_icloudpd_photo_paths.results[0].stat.isdir | default(false)
|
||||
- atlas_icloudpd_photo_paths.results[0].stat.uid | int == atlas_admin_uid | int
|
||||
- >-
|
||||
not atlas_icloudpd_photo_paths.results[1].stat.exists or
|
||||
(atlas_icloudpd_photo_paths.results[1].stat.isdir | default(false) and
|
||||
atlas_icloudpd_photo_paths.results[2].stat.isreg | default(false))
|
||||
fail_msg: >-
|
||||
Pictures must exist and be admin-owned; an existing iCloudPD target
|
||||
must carry its managed marker. Never adopt or replace unrelated data.
|
||||
|
||||
- name: Create a dedicated ZFS dataset for iCloudPD configuration and MFA
|
||||
community.general.zfs:
|
||||
name: "{{ atlas_icloudpd_dataset }}"
|
||||
state: present
|
||||
extra_zfs_properties:
|
||||
compression: zstd
|
||||
mountpoint: "{{ atlas_icloudpd_state_dir }}"
|
||||
|
||||
- name: Restrict iCloudPD state and the new photo subtree
|
||||
ansible.builtin.file:
|
||||
path: "{{ item.path }}"
|
||||
state: directory
|
||||
owner: "{{ atlas_admin_username }}"
|
||||
group: "{{ atlas_admin_group }}"
|
||||
mode: "{{ item.mode }}"
|
||||
loop:
|
||||
- path: "{{ atlas_icloudpd_state_dir }}"
|
||||
mode: "0700"
|
||||
- path: "{{ atlas_icloudpd_config_dir }}"
|
||||
mode: "0700"
|
||||
- path: "{{ atlas_icloudpd_photos_dir }}"
|
||||
mode: "0750"
|
||||
- path: "{{ atlas_icloudpd_quadlet_dir }}"
|
||||
mode: "0700"
|
||||
loop_control:
|
||||
label: "{{ item.path }}"
|
||||
|
||||
- name: Mark only the newly managed iCloudPD photo subtree
|
||||
ansible.builtin.copy:
|
||||
content: "Atlas iCloudPD photo subtree; do not remove source photos.\n"
|
||||
dest: "{{ atlas_icloudpd_photos_dir }}/.atlas-icloudpd-managed"
|
||||
owner: "{{ atlas_admin_username }}"
|
||||
group: "{{ atlas_admin_group }}"
|
||||
mode: "0600"
|
||||
force: false
|
||||
|
||||
- name: Install the image's required mounted-filesystem failsafe
|
||||
ansible.builtin.copy:
|
||||
content: ""
|
||||
dest: "{{ atlas_icloudpd_photos_dir }}/.mounted"
|
||||
owner: "{{ atlas_admin_username }}"
|
||||
group: "{{ atlas_admin_group }}"
|
||||
mode: "0644"
|
||||
force: false
|
||||
|
||||
- name: Require the Vault-backed iCloudPD Apple ID
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- vault_atlas_icloudpd_apple_id is defined
|
||||
- vault_atlas_icloudpd_apple_id | length > 0
|
||||
- vault_atlas_icloudpd_apple_id != 'REPLACE_ME'
|
||||
- vault_atlas_icloudpd_apple_id.splitlines() | length == 1
|
||||
fail_msg: Configure the existing iCloudPD Apple ID in Vault.
|
||||
no_log: true
|
||||
|
||||
- name: Seed private Atlas iCloudPD configuration when absent
|
||||
ansible.builtin.template:
|
||||
src: atlas-icloudpd.conf.j2
|
||||
dest: "{{ atlas_icloudpd_config_dir }}/icloudpd.conf"
|
||||
owner: "{{ atlas_admin_username }}"
|
||||
group: "{{ atlas_admin_group }}"
|
||||
mode: "0600"
|
||||
force: false
|
||||
no_log: true
|
||||
diff: false
|
||||
|
||||
- name: Keep declared iCloudPD options in the image-managed configuration
|
||||
ansible.builtin.lineinfile:
|
||||
path: "{{ atlas_icloudpd_config_dir }}/icloudpd.conf"
|
||||
regexp: "^{{ item.key }}="
|
||||
line: "{{ item.key }}={{ item.value }}"
|
||||
owner: "{{ atlas_admin_username }}"
|
||||
group: "{{ atlas_admin_group }}"
|
||||
mode: "0600"
|
||||
loop:
|
||||
- {key: apple_id, value: "{{ vault_atlas_icloudpd_apple_id }}"}
|
||||
- {key: authentication_type, value: MFA}
|
||||
- {key: user, value: user}
|
||||
- {key: user_id, value: "1000"}
|
||||
- {key: group, value: group}
|
||||
- {key: group_id, value: "1000"}
|
||||
- {key: download_path, value: /home/user/iCloud}
|
||||
- {key: folder_structure, value: "{:%Y/%m/%d}"}
|
||||
- {key: directory_permissions, value: "750"}
|
||||
- {key: file_permissions, value: "640"}
|
||||
- {key: download_interval, value: "86400"}
|
||||
- {key: auto_delete, value: "false"}
|
||||
- {key: delete_after_download, value: "false"}
|
||||
loop_control:
|
||||
label: "{{ item.key }}"
|
||||
no_log: true
|
||||
diff: false
|
||||
|
||||
- name: Render the rootless Atlas iCloudPD Quadlet
|
||||
ansible.builtin.template:
|
||||
src: atlas-icloudpd.container.j2
|
||||
dest: "{{ atlas_icloudpd_quadlet_dir }}/atlas-icloudpd.container"
|
||||
owner: "{{ atlas_admin_username }}"
|
||||
group: "{{ atlas_admin_group }}"
|
||||
mode: "0644"
|
||||
register: atlas_icloudpd_quadlet
|
||||
|
||||
- name: Reload the Atlas admin user manager after iCloudPD Quadlet changes
|
||||
become_user: "{{ atlas_admin_username }}"
|
||||
ansible.builtin.systemd:
|
||||
scope: user
|
||||
daemon_reload: true
|
||||
environment:
|
||||
XDG_RUNTIME_DIR: "/run/user/{{ atlas_admin_uid }}"
|
||||
DBUS_SESSION_BUS_ADDRESS: "unix:path=/run/user/{{ atlas_admin_uid }}/bus"
|
||||
when:
|
||||
- atlas_icloudpd_quadlet.changed
|
||||
- not ansible_check_mode
|
||||
|
||||
- name: Keep the rootless Atlas iCloudPD service running
|
||||
become_user: "{{ atlas_admin_username }}"
|
||||
ansible.builtin.systemd:
|
||||
name: atlas-icloudpd.service
|
||||
scope: user
|
||||
state: started
|
||||
environment:
|
||||
XDG_RUNTIME_DIR: "/run/user/{{ atlas_admin_uid }}"
|
||||
DBUS_SESSION_BUS_ADDRESS: "unix:path=/run/user/{{ atlas_admin_uid }}/bus"
|
||||
when: not ansible_check_mode
|
||||
@@ -14,12 +14,39 @@
|
||||
- name: Import Atlas storage tasks
|
||||
ansible.builtin.import_tasks: storage.yml
|
||||
|
||||
- name: Import staged Atlas rootless Gitea tasks
|
||||
ansible.builtin.import_tasks: gitea.yml
|
||||
|
||||
- name: Import the declared Atlas Gitea public domain
|
||||
ansible.builtin.import_tasks: gitea_public_domain.yml
|
||||
|
||||
- name: Import Atlas Nextcloud steady-state stack
|
||||
ansible.builtin.import_tasks: nextcloud.yml
|
||||
|
||||
- name: Import Atlas iCloudPD storage and boot-started Quadlet tasks
|
||||
ansible.builtin.import_tasks: icloudpd.yml
|
||||
|
||||
- name: Import Atlas ZFS maintenance tasks
|
||||
ansible.builtin.import_tasks: zfs_maintenance.yml
|
||||
|
||||
- name: Import Atlas Borg backup tasks
|
||||
ansible.builtin.import_tasks: borg_backup.yml
|
||||
|
||||
- name: Import Atlas offline USB backup tasks
|
||||
ansible.builtin.import_tasks: usb_backup.yml
|
||||
|
||||
- name: Import Atlas Prometheus backup pull identity tasks
|
||||
ansible.builtin.import_tasks: prometheus_pull_identity.yml
|
||||
|
||||
- name: Import Atlas Prometheus backup pull job tasks
|
||||
ansible.builtin.import_tasks: prometheus_pull_job.yml
|
||||
|
||||
- name: Import Atlas health monitoring tasks
|
||||
ansible.builtin.import_tasks: monitoring.yml
|
||||
|
||||
- name: Import Atlas post-restore SELinux relabeling tasks
|
||||
ansible.builtin.import_tasks: restorecon.yml
|
||||
|
||||
- name: Import Atlas file sharing tasks
|
||||
ansible.builtin.import_tasks: sharing.yml
|
||||
|
||||
|
||||
201
ansible/roles/profile_atlas/tasks/monitoring.yml
Normal file
201
ansible/roles/profile_atlas/tasks/monitoring.yml
Normal file
@@ -0,0 +1,201 @@
|
||||
---
|
||||
- name: Validate Atlas health monitoring policy
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- atlas_manage_storage | bool
|
||||
- atlas_zfs_pool != 'CHANGEME_ZFS_POOL'
|
||||
- atlas_monitor_calendar | length > 0
|
||||
- atlas_monitor_smart_devices | length > 0
|
||||
- atlas_monitor_effective_timers | length > 0
|
||||
- atlas_monitor_effective_failure_units | length > 0
|
||||
- atlas_monitor_remote_capacity.user == atlas_borg_repository_user
|
||||
- atlas_monitor_remote_capacity.host == atlas_borg_repository_host
|
||||
- atlas_monitor_remote_capacity.run_as == atlas_borg_username
|
||||
- atlas_monitor_remote_capacity.ssh_wrapper == atlas_borg_ssh_wrapper_path
|
||||
- >-
|
||||
0 < atlas_monitor_remote_capacity.warning_percent | int
|
||||
< atlas_monitor_remote_capacity.critical_percent | int < 100
|
||||
- atlas_monitor_remote_capacity.growth_warning_gib_day | int > 0
|
||||
- atlas_monitor_notifier.startswith('/opt/45drives/houston/')
|
||||
- 0 < atlas_monitor_pool_warning_percent | int < atlas_monitor_pool_critical_percent | int < 100
|
||||
- 0 < atlas_monitor_root_warning_percent | int < atlas_monitor_root_critical_percent | int < 100
|
||||
- 0 < atlas_monitor_snapshot_warning_percent | int < atlas_monitor_snapshot_critical_percent | int < 100
|
||||
- atlas_monitor_snapshot_growth_warning_gib_day | int > 0
|
||||
- atlas_monitor_backup_growth_warning_gib_day | int > 0
|
||||
- 0 < atlas_monitor_cpu_warning_c | int < atlas_monitor_cpu_critical_c | int
|
||||
- atlas_monitor_borg_max_runtime_days | int > 0
|
||||
fail_msg: >-
|
||||
Atlas health monitoring needs real devices, job units, a valid calendar,
|
||||
positive ordered thresholds, and the existing Houston notifier.
|
||||
when: atlas_manage_monitoring | bool
|
||||
|
||||
- name: Validate monitored Atlas SMART devices
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- item.name is match('^[a-z0-9][a-z0-9_-]*$')
|
||||
- item.path.startswith('/dev/disk/by-id/')
|
||||
- 0 < item.warning_c | int < item.critical_c | int
|
||||
fail_msg: "Every monitored disk needs a stable by-id path and ordered temperature thresholds."
|
||||
loop: "{{ atlas_monitor_smart_devices }}"
|
||||
loop_control:
|
||||
label: "{{ item.name }}"
|
||||
when: atlas_manage_monitoring | bool
|
||||
|
||||
- name: Validate monitored Atlas timer names and age thresholds
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- item.name is match('^[a-zA-Z0-9@_.-]+\\.timer$')
|
||||
- item.max_age_hours | int >= 0
|
||||
loop: "{{ atlas_monitor_effective_timers }}"
|
||||
loop_control:
|
||||
label: "{{ item.name }}"
|
||||
when: atlas_manage_monitoring | bool
|
||||
|
||||
- name: Validate monitored Atlas failure unit names
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- item is match('^[a-zA-Z0-9@_.-]+\\.service$')
|
||||
loop: "{{ atlas_monitor_effective_failure_units }}"
|
||||
when: atlas_manage_monitoring | bool
|
||||
|
||||
- name: Validate Atlas health monitor calendar
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.command:
|
||||
argv: [systemd-analyze, calendar, "{{ atlas_monitor_calendar }}"]
|
||||
changed_when: false
|
||||
check_mode: false
|
||||
when: atlas_manage_monitoring | bool
|
||||
|
||||
- name: Install SMART tooling for Atlas health checks
|
||||
tags: [atlas, monitoring, packages]
|
||||
ansible.builtin.dnf:
|
||||
name: smartmontools
|
||||
state: present
|
||||
when: atlas_manage_monitoring | bool
|
||||
|
||||
- name: Inspect the existing 45Drives notifier for monitoring
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.stat:
|
||||
path: "{{ atlas_monitor_notifier }}"
|
||||
register: atlas_monitor_notifier_file
|
||||
when: atlas_manage_monitoring | bool
|
||||
|
||||
- name: Require the existing 45Drives notifier for monitoring
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- atlas_monitor_notifier_file.stat.executable | default(false)
|
||||
fail_msg: "The existing 45Drives Houston notifier must be executable."
|
||||
when: atlas_manage_monitoring | bool
|
||||
|
||||
- name: Create private Atlas health monitor state directory
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.file:
|
||||
path: /var/lib/atlas-health-monitor
|
||||
state: directory
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0700"
|
||||
when: atlas_manage_monitoring | bool
|
||||
|
||||
- name: Install Atlas health monitor configuration
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.template:
|
||||
src: atlas-health-monitor.json.j2
|
||||
dest: /etc/atlas-health-monitor.json
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0600"
|
||||
when: atlas_manage_monitoring | bool
|
||||
|
||||
- name: Install Atlas health monitor helper
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.copy:
|
||||
src: atlas-health-monitor.py
|
||||
dest: /usr/local/libexec/atlas-health-monitor
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0750"
|
||||
when: atlas_manage_monitoring | bool
|
||||
|
||||
- name: Install Atlas health monitoring units
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.template:
|
||||
src: "{{ item }}.j2"
|
||||
dest: "/etc/systemd/system/{{ item }}"
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0644"
|
||||
loop:
|
||||
- atlas-health-monitor.service
|
||||
- atlas-health-monitor.timer
|
||||
- atlas-monitor-failure@.service
|
||||
when: atlas_manage_monitoring | bool
|
||||
|
||||
- name: Create failure hook directories for monitored Atlas jobs
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.file:
|
||||
path: "/etc/systemd/system/{{ item }}.d"
|
||||
state: directory
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0755"
|
||||
loop: "{{ atlas_monitor_effective_failure_units }}"
|
||||
when: atlas_manage_monitoring | bool
|
||||
|
||||
- name: Notify 45Drives Alerts when an Atlas job fails
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.template:
|
||||
src: atlas-monitor-failure.conf.j2
|
||||
dest: "/etc/systemd/system/{{ item }}.d/atlas-monitor.conf"
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0644"
|
||||
loop: "{{ atlas_monitor_effective_failure_units }}"
|
||||
when: atlas_manage_monitoring | bool
|
||||
|
||||
- name: Reload systemd after installing Atlas monitoring
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.systemd:
|
||||
daemon_reload: true
|
||||
when:
|
||||
- atlas_manage_monitoring | bool
|
||||
- not ansible_check_mode
|
||||
|
||||
- name: Enable the Atlas health monitoring timer
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.systemd:
|
||||
name: atlas-health-monitor.timer
|
||||
enabled: true
|
||||
state: started
|
||||
when:
|
||||
- atlas_manage_monitoring | bool
|
||||
- not ansible_check_mode
|
||||
|
||||
- name: Validate the deployed Atlas health monitoring units
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.command:
|
||||
argv:
|
||||
- systemd-analyze
|
||||
- verify
|
||||
- atlas-health-monitor.service
|
||||
- atlas-health-monitor.timer
|
||||
- atlas-monitor-failure@.service
|
||||
changed_when: false
|
||||
when:
|
||||
- atlas_manage_monitoring | bool
|
||||
- not ansible_check_mode
|
||||
|
||||
- name: Probe Atlas health without sending notifications
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.command:
|
||||
argv: [/usr/local/libexec/atlas-health-monitor, --dry-run]
|
||||
register: atlas_monitor_dry_run
|
||||
changed_when: false
|
||||
when:
|
||||
- atlas_manage_monitoring | bool
|
||||
- not ansible_check_mode
|
||||
309
ansible/roles/profile_atlas/tasks/nextcloud.yml
Normal file
309
ansible/roles/profile_atlas/tasks/nextcloud.yml
Normal file
@@ -0,0 +1,309 @@
|
||||
---
|
||||
- name: Manage the empty Atlas Nextcloud and ONLYOFFICE stack
|
||||
tags: [atlas, nextcloud]
|
||||
when: atlas_manage_nextcloud | bool
|
||||
block:
|
||||
- name: Validate dedicated paths, domains and pinned images
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- atlas_manage_storage | bool
|
||||
- atlas_manage_firewall | bool
|
||||
- atlas_nextcloud_root == atlas_app_data_mountpoint ~ '/nextcloud'
|
||||
- atlas_nextcloud_dataset == atlas_zfs_pool ~ '/services/data/nextcloud'
|
||||
- atlas_nextcloud_domain is match('^[a-z0-9.-]+$')
|
||||
- atlas_onlyoffice_domain is match('^[a-z0-9.-]+$')
|
||||
- atlas_nextcloud_domain != atlas_onlyoffice_domain
|
||||
- atlas_nextcloud_http_port | int > 1024
|
||||
- atlas_onlyoffice_http_port | int > 1024
|
||||
- atlas_nextcloud_http_port != atlas_onlyoffice_http_port
|
||||
- "['calendar', 'contacts', 'onlyoffice', 'groupfolders'] | difference(atlas_nextcloud_apps | map(attribute='id') | list) | length == 0"
|
||||
- atlas_nextcloud_users | length > 0
|
||||
- atlas_nextcloud_admin not in (atlas_nextcloud_users | map(attribute='username') | list)
|
||||
- atlas_nextcloud_users | map(attribute='username') | unique | list | length == atlas_nextcloud_users | length
|
||||
- item is search('@sha256:[0-9a-f]{64}$')
|
||||
loop:
|
||||
- "{{ atlas_nextcloud_image }}"
|
||||
- "{{ atlas_nextcloud_postgres_image }}"
|
||||
- "{{ atlas_nextcloud_redis_image }}"
|
||||
- "{{ atlas_onlyoffice_image }}"
|
||||
|
||||
- name: Require dedicated Vault secrets without exposing them
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- item | default('') is match('^[a-zA-Z0-9]{32,}$')
|
||||
loop: >-
|
||||
{{ [vault_nextcloud_database_password | default(''),
|
||||
vault_nextcloud_redis_password | default(''),
|
||||
vault_nextcloud_admin_password | default(''),
|
||||
vault_nextcloud_onlyoffice_jwt | default('')] +
|
||||
(atlas_nextcloud_users | map(attribute='password') | list) }}
|
||||
no_log: true
|
||||
|
||||
- name: Verify the existing application-data parent is mounted
|
||||
community.general.zfs_facts:
|
||||
name: "{{ atlas_zfs_pool }}/{{ atlas_zfs_dataset_app_data }}"
|
||||
properties: name,mounted,mountpoint
|
||||
register: atlas_nextcloud_parent
|
||||
|
||||
- name: Require the verified application-data parent
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- atlas_nextcloud_parent.ansible_facts.ansible_zfs_datasets | length == 1
|
||||
- atlas_nextcloud_parent.ansible_facts.ansible_zfs_datasets[0].mounted == 'yes'
|
||||
- atlas_nextcloud_parent.ansible_facts.ansible_zfs_datasets[0].mountpoint == atlas_app_data_mountpoint
|
||||
|
||||
- name: Create the dedicated Nextcloud namespace and component datasets
|
||||
community.general.zfs:
|
||||
name: "{{ atlas_nextcloud_dataset }}{{ item }}"
|
||||
state: present
|
||||
extra_zfs_properties:
|
||||
compression: zstd
|
||||
mountpoint: "{{ atlas_nextcloud_root }}{{ item }}"
|
||||
loop: ['', /app, /files, /database, /cache, /office]
|
||||
|
||||
- name: Inspect component directories before seeding ownership
|
||||
ansible.builtin.stat:
|
||||
path: "{{ atlas_nextcloud_root }}{{ item }}"
|
||||
follow: false
|
||||
get_checksum: false
|
||||
loop: [/app, /files, /database, /cache, /office]
|
||||
register: atlas_nextcloud_component_paths
|
||||
|
||||
- name: Seed only root-owned new dataset roots without recursive ownership changes
|
||||
ansible.builtin.file:
|
||||
path: "{{ item.stat.path }}"
|
||||
state: directory
|
||||
owner: "{{ atlas_admin_username }}"
|
||||
group: "{{ atlas_admin_group }}"
|
||||
mode: "0700"
|
||||
loop: "{{ atlas_nextcloud_component_paths.results }}"
|
||||
loop_control:
|
||||
label: "{{ item.item }}"
|
||||
when:
|
||||
- item.stat.exists
|
||||
- item.stat.uid | default(-1) | int == 0
|
||||
|
||||
- name: Ensure private rootless stack configuration directories exist
|
||||
ansible.builtin.file:
|
||||
path: "{{ item.path }}"
|
||||
state: directory
|
||||
owner: "{{ atlas_admin_username }}"
|
||||
group: "{{ atlas_admin_group }}"
|
||||
mode: "{{ item.mode }}"
|
||||
loop:
|
||||
- {path: "{{ atlas_nextcloud_private_dir }}", mode: "0700"}
|
||||
- {path: "{{ atlas_nextcloud_app_cache }}", mode: "0755"}
|
||||
- {path: "{{ atlas_nextcloud_quadlet_dir }}", mode: "0700"}
|
||||
- {path: "{{ atlas_admin_home }}/.config/systemd/user", mode: "0700"}
|
||||
|
||||
- name: Inspect the dedicated ONLYOFFICE bind directories
|
||||
ansible.builtin.stat:
|
||||
path: "{{ atlas_nextcloud_root }}/office/{{ item }}"
|
||||
follow: false
|
||||
get_checksum: false
|
||||
loop: [data, lib, logs, database]
|
||||
register: atlas_onlyoffice_bind_paths
|
||||
|
||||
- name: Create ONLYOFFICE bind directories only when absent
|
||||
ansible.builtin.file:
|
||||
path: "{{ item.invocation.module_args.path }}"
|
||||
state: directory
|
||||
owner: "{{ atlas_admin_username }}"
|
||||
group: "{{ atlas_admin_group }}"
|
||||
mode: "0700"
|
||||
loop: "{{ atlas_onlyoffice_bind_paths.results }}"
|
||||
loop_control:
|
||||
label: "{{ item.item }}"
|
||||
when: not item.stat.exists
|
||||
|
||||
- name: Store private mounted password files inside a restricted host directory
|
||||
ansible.builtin.copy:
|
||||
content: "{{ item.value }}\n"
|
||||
dest: "{{ atlas_nextcloud_private_dir }}/{{ item.name }}"
|
||||
owner: "{{ atlas_admin_username }}"
|
||||
group: "{{ atlas_admin_group }}"
|
||||
mode: "0644"
|
||||
loop:
|
||||
- {name: postgres-password, value: "{{ vault_nextcloud_database_password }}"}
|
||||
- {name: redis-password, value: "{{ vault_nextcloud_redis_password }}"}
|
||||
- {name: admin-password, value: "{{ vault_nextcloud_admin_password }}"}
|
||||
- {name: onlyoffice-jwt, value: "{{ vault_nextcloud_onlyoffice_jwt }}"}
|
||||
no_log: true
|
||||
diff: false
|
||||
register: atlas_nextcloud_secret_files
|
||||
|
||||
- name: Render private Redis and ONLYOFFICE configuration
|
||||
ansible.builtin.template:
|
||||
src: "{{ item.src }}"
|
||||
dest: "{{ atlas_nextcloud_private_dir }}/{{ item.dest }}"
|
||||
owner: "{{ atlas_admin_username }}"
|
||||
group: "{{ atlas_admin_group }}"
|
||||
mode: "{{ item.mode }}"
|
||||
loop:
|
||||
- {src: atlas-nextcloud-redis.conf.j2, dest: redis.conf, mode: "0644"}
|
||||
- {src: atlas-onlyoffice.env.j2, dest: onlyoffice.env, mode: "0600"}
|
||||
no_log: true
|
||||
diff: false
|
||||
register: atlas_nextcloud_private_configuration
|
||||
|
||||
- name: Download checksum-pinned compatible application releases
|
||||
ansible.builtin.get_url:
|
||||
url: "{{ item.url }}"
|
||||
dest: "{{ atlas_nextcloud_app_cache }}/{{ item.id }}-{{ item.version }}.tar.gz"
|
||||
checksum: "{{ item.checksum }}"
|
||||
owner: "{{ atlas_admin_username }}"
|
||||
group: "{{ atlas_admin_group }}"
|
||||
mode: "0644"
|
||||
loop: "{{ atlas_nextcloud_apps }}"
|
||||
loop_control:
|
||||
label: "{{ item.id }} {{ item.version }}"
|
||||
when: not ansible_check_mode
|
||||
|
||||
- name: Admit only the Aegis gateway to the Nextcloud and Office HTTP listeners
|
||||
ansible.posix.firewalld:
|
||||
rich_rule: >-
|
||||
rule family="ipv4" source address="{{ atlas_aegis_ip }}"
|
||||
port port="{{ item }}" protocol="tcp" accept
|
||||
zone: "{{ atlas_firewalld_zone }}"
|
||||
state: enabled
|
||||
permanent: true
|
||||
immediate: true
|
||||
loop: ["{{ atlas_nextcloud_http_port }}", "{{ atlas_onlyoffice_http_port }}"]
|
||||
|
||||
- name: Enable lingering for the declared rootless owner
|
||||
ansible.builtin.command:
|
||||
argv: [loginctl, enable-linger, "{{ atlas_admin_username }}"]
|
||||
creates: "/var/lib/systemd/linger/{{ atlas_admin_username }}"
|
||||
|
||||
- name: Render Nextcloud component and network Quadlets
|
||||
ansible.builtin.template:
|
||||
src: "{{ item }}.j2"
|
||||
dest: "{{ atlas_nextcloud_quadlet_dir }}/{{ item }}"
|
||||
owner: "{{ atlas_admin_username }}"
|
||||
group: "{{ atlas_admin_group }}"
|
||||
mode: "0644"
|
||||
loop:
|
||||
- atlas-nextcloud.network
|
||||
- atlas-nextcloud-db.container
|
||||
- atlas-nextcloud-redis.container
|
||||
- atlas-nextcloud.container
|
||||
- atlas-onlyoffice.container
|
||||
register: atlas_nextcloud_quadlets
|
||||
|
||||
- name: Render recurring Nextcloud cron user units
|
||||
ansible.builtin.template:
|
||||
src: "{{ item }}.j2"
|
||||
dest: "{{ atlas_admin_home }}/.config/systemd/user/{{ item }}"
|
||||
owner: "{{ atlas_admin_username }}"
|
||||
group: "{{ atlas_admin_group }}"
|
||||
mode: "0644"
|
||||
loop: [atlas-nextcloud-cron.service, atlas-nextcloud-cron.timer]
|
||||
register: atlas_nextcloud_cron_units
|
||||
|
||||
- name: Manage and verify rootless Nextcloud services
|
||||
become_user: "{{ atlas_admin_username }}"
|
||||
environment:
|
||||
XDG_RUNTIME_DIR: "/run/user/{{ atlas_admin_uid }}"
|
||||
DBUS_SESSION_BUS_ADDRESS: "unix:path=/run/user/{{ atlas_admin_uid }}/bus"
|
||||
when: not ansible_check_mode
|
||||
block:
|
||||
- name: Pull the pinned images before starting services
|
||||
containers.podman.podman_image:
|
||||
name: "{{ item }}"
|
||||
state: present
|
||||
loop:
|
||||
- "{{ atlas_nextcloud_image }}"
|
||||
- "{{ atlas_nextcloud_postgres_image }}"
|
||||
- "{{ atlas_nextcloud_redis_image }}"
|
||||
- "{{ atlas_onlyoffice_image }}"
|
||||
|
||||
- name: Reload the user manager to generate component units
|
||||
ansible.builtin.systemd:
|
||||
scope: user
|
||||
daemon_reload: true
|
||||
|
||||
- name: Start the declared Nextcloud and ONLYOFFICE services
|
||||
ansible.builtin.systemd:
|
||||
scope: user
|
||||
name: "{{ item }}"
|
||||
state: >-
|
||||
{{ 'restarted' if (atlas_nextcloud_quadlets is changed or
|
||||
atlas_nextcloud_private_configuration is changed or
|
||||
atlas_nextcloud_secret_files is changed) else 'started' }}
|
||||
loop: "{{ atlas_nextcloud_services }}"
|
||||
|
||||
- name: Wait for the application configuration directory to be initialized
|
||||
become: true
|
||||
become_user: root
|
||||
ansible.builtin.wait_for:
|
||||
path: "{{ atlas_nextcloud_root }}/app/config/config.php"
|
||||
timeout: 600
|
||||
|
||||
- name: Derive container web-user host IDs from the actual rootless maps
|
||||
ansible.builtin.command:
|
||||
argv:
|
||||
- podman
|
||||
- unshare
|
||||
- python3
|
||||
- -c
|
||||
- >-
|
||||
import json;
|
||||
print(json.dumps({k: next(int(b)+33-int(a) for a,b,n in
|
||||
(l.split() for l in open('/proc/self/'+k+'_map'))
|
||||
if int(a)<=33<int(a)+int(n)) for k in ['uid','gid']}))
|
||||
register: atlas_nextcloud_web_mapping
|
||||
changed_when: false
|
||||
|
||||
- name: Read the current application SELinux label without changing it
|
||||
become: true
|
||||
become_user: root
|
||||
ansible.builtin.command:
|
||||
argv: [stat, -c, '%C', "{{ atlas_nextcloud_root }}/app/config"]
|
||||
register: atlas_nextcloud_config_label
|
||||
changed_when: false
|
||||
|
||||
- name: Maintain the managed Nextcloud configuration include
|
||||
become: true
|
||||
become_user: root
|
||||
ansible.builtin.template:
|
||||
src: atlas-nextcloud.config.php.j2
|
||||
dest: "{{ atlas_nextcloud_root }}/app/config/atlas.config.php"
|
||||
owner: "{{ (atlas_nextcloud_web_mapping.stdout | from_json).uid }}"
|
||||
group: "{{ (atlas_nextcloud_web_mapping.stdout | from_json).gid }}"
|
||||
mode: "0640"
|
||||
seuser: "{{ atlas_nextcloud_config_label.stdout.split(':')[0] }}"
|
||||
serole: "{{ atlas_nextcloud_config_label.stdout.split(':')[1] }}"
|
||||
setype: "{{ atlas_nextcloud_config_label.stdout.split(':')[2] }}"
|
||||
selevel: "{{ atlas_nextcloud_config_label.stdout.split(':')[3:] | join(':') }}"
|
||||
diff: false
|
||||
|
||||
- name: Wait for Nextcloud to complete its initial installation
|
||||
ansible.builtin.command:
|
||||
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, status, --output=json]
|
||||
register: atlas_nextcloud_status
|
||||
changed_when: false
|
||||
retries: 60
|
||||
delay: 10
|
||||
until: >-
|
||||
atlas_nextcloud_status.rc == 0 and
|
||||
atlas_nextcloud_status.stdout.startswith('{') and
|
||||
(atlas_nextcloud_status.stdout | from_json).installed | default(false)
|
||||
|
||||
- name: Import declared ongoing application and account configuration
|
||||
ansible.builtin.include_tasks: nextcloud_application.yml
|
||||
|
||||
- name: Enable and start the recurring Nextcloud cron timer
|
||||
ansible.builtin.systemd:
|
||||
scope: user
|
||||
name: atlas-nextcloud-cron.timer
|
||||
state: "{{ 'restarted' if atlas_nextcloud_cron_units is changed else 'started' }}"
|
||||
enabled: true
|
||||
|
||||
- name: Verify ONLYOFFICE local health without publishing the domain
|
||||
ansible.builtin.uri:
|
||||
url: "http://127.0.0.1:{{ atlas_onlyoffice_http_port }}/healthcheck"
|
||||
return_content: true
|
||||
register: atlas_onlyoffice_health
|
||||
retries: 60
|
||||
delay: 10
|
||||
until: atlas_onlyoffice_health.status | default(0) == 200 and atlas_onlyoffice_health.content | default('') | trim == 'true'
|
||||
171
ansible/roles/profile_atlas/tasks/nextcloud_application.yml
Normal file
171
ansible/roles/profile_atlas/tasks/nextcloud_application.yml
Normal file
@@ -0,0 +1,171 @@
|
||||
---
|
||||
- name: Inspect installed application state
|
||||
ansible.builtin.command:
|
||||
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, app:list, --output=json]
|
||||
register: atlas_nextcloud_current_apps
|
||||
changed_when: false
|
||||
|
||||
- name: Record enabled and disabled application versions
|
||||
ansible.builtin.set_fact:
|
||||
atlas_nextcloud_installed_apps: >-
|
||||
{{ (atlas_nextcloud_current_apps.stdout | from_json).enabled |
|
||||
combine((atlas_nextcloud_current_apps.stdout | from_json).disabled) }}
|
||||
|
||||
- name: Refuse implicit application upgrades or downgrades
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- item.id not in atlas_nextcloud_installed_apps or atlas_nextcloud_installed_apps[item.id] == item.version
|
||||
fail_msg: Application versions must be changed in a deliberate upgrade window.
|
||||
loop: "{{ atlas_nextcloud_apps }}"
|
||||
loop_control:
|
||||
label: "{{ item.id }}"
|
||||
|
||||
- name: Install only absent checksum-verified application archives
|
||||
ansible.builtin.command:
|
||||
argv:
|
||||
- podman
|
||||
- exec
|
||||
- --user
|
||||
- '33'
|
||||
- atlas-nextcloud
|
||||
- tar
|
||||
- -xzf
|
||||
- "/mnt/atlas-apps/{{ item.id }}-{{ item.version }}.tar.gz"
|
||||
- -C
|
||||
- /var/www/html/custom_apps
|
||||
loop: "{{ atlas_nextcloud_apps }}"
|
||||
loop_control:
|
||||
label: "{{ item.id }}"
|
||||
when: item.id not in atlas_nextcloud_installed_apps
|
||||
changed_when: true
|
||||
|
||||
- name: Enable the declared applications
|
||||
ansible.builtin.command:
|
||||
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, app:enable, "{{ item.id }}"]
|
||||
loop: "{{ atlas_nextcloud_apps }}"
|
||||
loop_control:
|
||||
label: "{{ item.id }}"
|
||||
when: item.id not in (atlas_nextcloud_current_apps.stdout | from_json).enabled
|
||||
changed_when: true
|
||||
|
||||
- name: Inspect existing application users without exposing passwords
|
||||
ansible.builtin.command:
|
||||
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, user:list, --output=json]
|
||||
register: atlas_nextcloud_current_users
|
||||
changed_when: false
|
||||
|
||||
- name: Ensure the two standard users exist without resetting existing passwords
|
||||
ansible.builtin.command:
|
||||
argv:
|
||||
- podman
|
||||
- exec
|
||||
- --user
|
||||
- '33'
|
||||
- --env
|
||||
- OC_PASS
|
||||
- atlas-nextcloud
|
||||
- php
|
||||
- occ
|
||||
- user:add
|
||||
- --password-from-env
|
||||
- --display-name
|
||||
- "{{ item.display_name }}"
|
||||
- "{{ item.username }}"
|
||||
environment:
|
||||
OC_PASS: "{{ item.password }}"
|
||||
loop: "{{ atlas_nextcloud_users }}"
|
||||
when: item.username not in (atlas_nextcloud_current_users.stdout | from_json)
|
||||
changed_when: true
|
||||
no_log: true
|
||||
diff: false
|
||||
|
||||
- name: Inspect standard-user group membership and quota
|
||||
ansible.builtin.command:
|
||||
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, user:info, --output=json, "{{ item.username }}"]
|
||||
loop: "{{ atlas_nextcloud_users }}"
|
||||
loop_control:
|
||||
label: "{{ item.username }}"
|
||||
register: atlas_nextcloud_user_info
|
||||
changed_when: false
|
||||
no_log: true
|
||||
|
||||
- name: Require that family users are not administrators
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- "'admin' not in (item.stdout | from_json).groups"
|
||||
loop: "{{ atlas_nextcloud_user_info.results }}"
|
||||
no_log: true
|
||||
|
||||
- name: Maintain unlimited initial standard-user quotas
|
||||
ansible.builtin.command:
|
||||
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, user:setting, "{{ item.item.username }}", files, quota, none]
|
||||
loop: "{{ atlas_nextcloud_user_info.results }}"
|
||||
when: (item.stdout | from_json).quota != 'none'
|
||||
changed_when: true
|
||||
no_log: true
|
||||
|
||||
- name: Inspect the family group
|
||||
ansible.builtin.command:
|
||||
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, group:list, --output=json]
|
||||
register: atlas_nextcloud_groups
|
||||
changed_when: false
|
||||
|
||||
- name: Ensure the family group exists
|
||||
ansible.builtin.command:
|
||||
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, group:add, famiglia]
|
||||
when: "'famiglia' not in (atlas_nextcloud_groups.stdout | from_json)"
|
||||
changed_when: true
|
||||
|
||||
- name: Ensure both standard users belong to the family group
|
||||
ansible.builtin.command:
|
||||
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, group:adduser, famiglia, "{{ item.username }}"]
|
||||
loop: "{{ atlas_nextcloud_users }}"
|
||||
when: item.username not in ((atlas_nextcloud_groups.stdout | from_json).get('famiglia', []))
|
||||
changed_when: true
|
||||
no_log: true
|
||||
|
||||
- name: Inspect configured family folders
|
||||
ansible.builtin.command:
|
||||
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, groupfolders:list, --output=json]
|
||||
register: atlas_nextcloud_folders_before
|
||||
changed_when: false
|
||||
|
||||
- name: Ensure a shared Famiglia folder exists
|
||||
ansible.builtin.command:
|
||||
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, groupfolders:create, Famiglia]
|
||||
when: >-
|
||||
(atlas_nextcloud_folders_before.stdout | from_json |
|
||||
selectattr('mountPoint', 'equalto', 'Famiglia') | list | length) == 0
|
||||
changed_when: true
|
||||
|
||||
- name: Inspect the resulting family folder
|
||||
ansible.builtin.command:
|
||||
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, groupfolders:list, --output=json]
|
||||
register: atlas_nextcloud_folders_after
|
||||
changed_when: false
|
||||
|
||||
- name: Select the existing family folder without changing unrelated folders
|
||||
ansible.builtin.set_fact:
|
||||
atlas_nextcloud_family_folder: >-
|
||||
{{ atlas_nextcloud_folders_after.stdout | from_json |
|
||||
selectattr('mountPoint', 'equalto', 'Famiglia') | first }}
|
||||
|
||||
- name: Maintain family read, create, write and delete permissions
|
||||
ansible.builtin.command:
|
||||
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, groupfolders:group,
|
||||
"{{ atlas_nextcloud_family_folder.id }}", famiglia, write, delete]
|
||||
when: (atlas_nextcloud_family_folder.groups_list | default({}, true)).get('famiglia', 0) | int != 15
|
||||
changed_when: true
|
||||
|
||||
- name: Inspect the background job mode
|
||||
ansible.builtin.command:
|
||||
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, config:app:get, core, backgroundjobs_mode]
|
||||
register: atlas_nextcloud_background_mode
|
||||
changed_when: false
|
||||
failed_when: atlas_nextcloud_background_mode.rc not in [0, 1]
|
||||
|
||||
- name: Maintain cron background processing
|
||||
ansible.builtin.command:
|
||||
argv: [podman, exec, --user, '33', atlas-nextcloud, php, occ, background:cron]
|
||||
when: atlas_nextcloud_background_mode.stdout | trim != 'cron'
|
||||
changed_when: true
|
||||
@@ -0,0 +1,66 @@
|
||||
---
|
||||
- name: Validate Atlas Prometheus pull identity inputs
|
||||
tags: [atlas, backup, prometheus_backup, prometheus_backup_key]
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- atlas_prometheus_pull_ssh_dir.startswith('/etc/')
|
||||
- atlas_prometheus_pull_private_key_path.startswith(atlas_prometheus_pull_ssh_dir ~ '/')
|
||||
- atlas_prometheus_pull_known_hosts_path.startswith(atlas_prometheus_pull_ssh_dir ~ '/')
|
||||
- atlas_prometheus_ssh_host_key.startswith(
|
||||
(hostvars['prometheus'].ansible_host | string) ~ ' ssh-ed25519 '
|
||||
)
|
||||
fail_msg: Pin the verified Prometheus ED25519 SSH host key before enabling the pull.
|
||||
when: atlas_manage_prometheus_backup_pull | bool
|
||||
|
||||
- name: Create private Atlas Prometheus pull SSH directory
|
||||
tags: [atlas, backup, prometheus_backup, prometheus_backup_key]
|
||||
ansible.builtin.file:
|
||||
path: "{{ atlas_prometheus_pull_ssh_dir }}"
|
||||
state: directory
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0700"
|
||||
when: atlas_manage_prometheus_backup_pull | bool
|
||||
|
||||
- name: Generate Atlas-only Prometheus pull SSH identity
|
||||
tags: [atlas, backup, prometheus_backup, prometheus_backup_key]
|
||||
ansible.builtin.command:
|
||||
argv:
|
||||
- ssh-keygen
|
||||
- -q
|
||||
- -t
|
||||
- ed25519
|
||||
- -N
|
||||
- ""
|
||||
- -C
|
||||
- atlas-prometheus-pull@atlas
|
||||
- -f
|
||||
- "{{ atlas_prometheus_pull_private_key_path }}"
|
||||
creates: "{{ atlas_prometheus_pull_private_key_path }}"
|
||||
when: atlas_manage_prometheus_backup_pull | bool
|
||||
|
||||
- name: Protect Atlas-only Prometheus pull SSH identity
|
||||
tags: [atlas, backup, prometheus_backup, prometheus_backup_key]
|
||||
ansible.builtin.file:
|
||||
path: "{{ item.path }}"
|
||||
owner: root
|
||||
group: root
|
||||
mode: "{{ item.mode }}"
|
||||
loop:
|
||||
- { path: "{{ atlas_prometheus_pull_private_key_path }}", mode: "0600" }
|
||||
- { path: "{{ atlas_prometheus_pull_private_key_path }}.pub", mode: "0644" }
|
||||
loop_control:
|
||||
label: "{{ item.path }}"
|
||||
when:
|
||||
- atlas_manage_prometheus_backup_pull | bool
|
||||
- not ansible_check_mode
|
||||
|
||||
- name: Pin Prometheus SSH host key on Atlas
|
||||
tags: [atlas, backup, prometheus_backup, prometheus_backup_key]
|
||||
ansible.builtin.copy:
|
||||
content: "{{ atlas_prometheus_ssh_host_key }}\n"
|
||||
dest: "{{ atlas_prometheus_pull_known_hosts_path }}"
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0600"
|
||||
when: atlas_manage_prometheus_backup_pull | bool
|
||||
90
ansible/roles/profile_atlas/tasks/prometheus_pull_job.yml
Normal file
90
ansible/roles/profile_atlas/tasks/prometheus_pull_job.yml
Normal file
@@ -0,0 +1,90 @@
|
||||
---
|
||||
- name: Validate Atlas Prometheus backup pull inputs
|
||||
tags: [atlas, backup, prometheus_backup]
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- atlas_manage_storage | bool
|
||||
- atlas_prometheus_pull_source_user is match('^[a-z_][a-z0-9_-]*$')
|
||||
- atlas_prometheus_pull_source_port | int > 0
|
||||
- atlas_prometheus_pull_source_port | int < 65536
|
||||
- atlas_prometheus_pull_keep_daily | int > 0
|
||||
- atlas_prometheus_pull_keep_weekly | int > 0
|
||||
- atlas_prometheus_pull_keep_monthly | int > 0
|
||||
- atlas_prometheus_pull_max_age_hours | int > 0
|
||||
- atlas_backup_prometheus_mountpoint.startswith(atlas_mount_root ~ '/')
|
||||
fail_msg: Define the Atlas backup destination, source account, and retention before enabling the pull.
|
||||
when: atlas_manage_prometheus_backup_pull | bool
|
||||
|
||||
- name: Validate Atlas Prometheus backup pull calendar
|
||||
tags: [atlas, backup, prometheus_backup]
|
||||
ansible.builtin.command:
|
||||
argv: [systemd-analyze, calendar, "{{ atlas_prometheus_pull_calendar }}"]
|
||||
changed_when: false
|
||||
check_mode: false
|
||||
when: atlas_manage_prometheus_backup_pull | bool
|
||||
|
||||
- name: Create private Atlas Prometheus backup version directory
|
||||
tags: [atlas, backup, prometheus_backup]
|
||||
ansible.builtin.file:
|
||||
path: "{{ atlas_backup_prometheus_mountpoint }}/snapshots"
|
||||
state: directory
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0700"
|
||||
when: atlas_manage_prometheus_backup_pull | bool
|
||||
|
||||
- name: Install Atlas Prometheus backup pull helper
|
||||
tags: [atlas, backup, prometheus_backup]
|
||||
ansible.builtin.template:
|
||||
src: atlas-prometheus-pull.sh.j2
|
||||
dest: /usr/local/sbin/atlas-prometheus-pull
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0750"
|
||||
when: atlas_manage_prometheus_backup_pull | bool
|
||||
|
||||
- name: Install Atlas Prometheus backup retention helper
|
||||
tags: [atlas, backup, prometheus_backup]
|
||||
ansible.builtin.copy:
|
||||
src: atlas-prometheus-prune.py
|
||||
dest: /usr/local/libexec/atlas-prometheus-prune
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0750"
|
||||
when: atlas_manage_prometheus_backup_pull | bool
|
||||
|
||||
- name: Install Atlas Prometheus backup pull systemd units
|
||||
tags: [atlas, backup, prometheus_backup]
|
||||
ansible.builtin.template:
|
||||
src: "{{ item }}.j2"
|
||||
dest: "/etc/systemd/system/{{ item }}"
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0644"
|
||||
loop:
|
||||
- atlas-prometheus-pull.service
|
||||
- atlas-prometheus-pull.timer
|
||||
loop_control:
|
||||
label: "{{ item }}"
|
||||
register: atlas_prometheus_pull_units
|
||||
when: atlas_manage_prometheus_backup_pull | bool
|
||||
|
||||
- name: Reload systemd after Atlas Prometheus pull unit changes
|
||||
tags: [atlas, backup, prometheus_backup]
|
||||
ansible.builtin.systemd:
|
||||
daemon_reload: true
|
||||
when:
|
||||
- atlas_manage_prometheus_backup_pull | bool
|
||||
- atlas_prometheus_pull_units is changed
|
||||
- not ansible_check_mode
|
||||
|
||||
- name: Enable Atlas Prometheus pull timer only after explicit activation
|
||||
tags: [atlas, backup, prometheus_backup]
|
||||
ansible.builtin.systemd:
|
||||
name: atlas-prometheus-pull.timer
|
||||
enabled: true
|
||||
state: started
|
||||
when:
|
||||
- atlas_manage_prometheus_backup_pull | bool
|
||||
- atlas_prometheus_pull_start_timer | bool
|
||||
- not ansible_check_mode
|
||||
27
ansible/roles/profile_atlas/tasks/restorecon.yml
Normal file
27
ansible/roles/profile_atlas/tasks/restorecon.yml
Normal file
@@ -0,0 +1,27 @@
|
||||
---
|
||||
- name: Validate requested Atlas post-restore relabel paths
|
||||
tags: [atlas, restorecon, recovery]
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- item is string
|
||||
- item.startswith(atlas_mount_root ~ '/')
|
||||
- item != atlas_mount_root
|
||||
fail_msg: >-
|
||||
Post-restore relabeling accepts only explicit paths below the Atlas pool
|
||||
mount root. Do not relabel the whole pool during routine provisioning.
|
||||
loop: "{{ atlas_restorecon_paths }}"
|
||||
when: atlas_restorecon_paths | length > 0
|
||||
|
||||
- name: Restore SELinux labels on explicitly restored Atlas paths
|
||||
tags: [atlas, restorecon, recovery]
|
||||
ansible.builtin.command:
|
||||
argv:
|
||||
- restorecon
|
||||
- -RFv
|
||||
- "{{ item }}"
|
||||
register: atlas_restorecon_result
|
||||
changed_when: atlas_restorecon_result.stdout | length > 0
|
||||
loop: "{{ atlas_restorecon_paths }}"
|
||||
when:
|
||||
- atlas_restorecon_paths | length > 0
|
||||
- not ansible_check_mode
|
||||
145
ansible/roles/profile_atlas/tasks/usb_backup.yml
Normal file
145
ansible/roles/profile_atlas/tasks/usb_backup.yml
Normal file
@@ -0,0 +1,145 @@
|
||||
---
|
||||
- name: Validate Atlas offline USB backup configuration
|
||||
tags: [atlas, storage, backup, usb_backup]
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- atlas_manage_storage | bool
|
||||
- atlas_zfs_pool != 'CHANGEME_ZFS_POOL'
|
||||
- atlas_mount_root.startswith('/')
|
||||
- atlas_usb_backup_luks_uuid is match('^[0-9a-fA-F]{8}(-[0-9a-fA-F]{4}){3}-[0-9a-fA-F]{12}$')
|
||||
- atlas_usb_backup_fs_uuid is match('^[0-9a-fA-F]{8}(-[0-9a-fA-F]{4}){3}-[0-9a-fA-F]{12}$')
|
||||
- atlas_usb_backup_luks_uuid != atlas_usb_backup_fs_uuid
|
||||
- atlas_usb_backup_mapper_name is match('^[a-z][a-z0-9_-]*$')
|
||||
- atlas_usb_backup_min_free_bytes | int > 0
|
||||
- atlas_usb_backup_snapshot_prefix is match('^[a-z0-9][a-z0-9_-]*$')
|
||||
- atlas_usb_backup_snapshot_prefix != atlas_borg_snapshot_prefix
|
||||
- atlas_usb_backup_snapshot_prefix != atlas_zfs_snapshot_prefix
|
||||
fail_msg: >-
|
||||
The manual Atlas USB backup needs verified LUKS and ext4 UUIDs, a safe
|
||||
mapper name, positive free-space reserve, and a unique snapshot prefix.
|
||||
when: atlas_manage_usb_backup | bool
|
||||
|
||||
- name: Install rsync for the Atlas offline USB backup
|
||||
tags: [atlas, storage, backup, usb_backup]
|
||||
ansible.builtin.dnf:
|
||||
name: rsync
|
||||
state: present
|
||||
when: atlas_manage_usb_backup | bool
|
||||
|
||||
- name: Install the manual Atlas offline USB backup helper
|
||||
tags: [atlas, storage, backup, usb_backup]
|
||||
ansible.builtin.template:
|
||||
src: atlas-usb-backup.sh.j2
|
||||
dest: /usr/local/sbin/atlas-usb-backup
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0750"
|
||||
when: atlas_manage_usb_backup | bool
|
||||
|
||||
- name: Install the Atlas USB snapshot cleanup helper
|
||||
tags: [atlas, storage, backup, usb_backup]
|
||||
ansible.builtin.template:
|
||||
src: atlas-usb-snapshot-cleanup.sh.j2
|
||||
dest: /usr/local/sbin/atlas-usb-snapshot-cleanup
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0750"
|
||||
when: atlas_manage_usb_backup | bool
|
||||
|
||||
- name: Install the manual Atlas offline USB backup service
|
||||
tags: [atlas, storage, backup, usb_backup]
|
||||
ansible.builtin.template:
|
||||
src: atlas-usb-backup.service.j2
|
||||
dest: /etc/systemd/system/atlas-usb-backup.service
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0644"
|
||||
when: atlas_manage_usb_backup | bool
|
||||
|
||||
- name: Reload systemd for the Atlas offline USB backup service
|
||||
tags: [atlas, storage, backup, usb_backup]
|
||||
ansible.builtin.systemd:
|
||||
daemon_reload: true
|
||||
when:
|
||||
- atlas_manage_usb_backup | bool
|
||||
- not ansible_check_mode
|
||||
|
||||
- name: Validate the 45Drives Atlas USB reminder configuration
|
||||
tags: [atlas, backup, usb_reminder]
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- atlas_manage_usb_backup | bool
|
||||
- atlas_usb_reminder_calendar | length > 0
|
||||
- atlas_usb_reminder_notifier.startswith('/opt/45drives/houston/')
|
||||
fail_msg: >-
|
||||
Enable the manual USB backup and declare a systemd calendar before
|
||||
enabling its 45Drives Alerts reminder.
|
||||
when: atlas_manage_usb_reminder | bool
|
||||
|
||||
- name: Validate the Atlas USB reminder calendar
|
||||
tags: [atlas, backup, usb_reminder]
|
||||
ansible.builtin.command:
|
||||
argv:
|
||||
- systemd-analyze
|
||||
- calendar
|
||||
- "{{ atlas_usb_reminder_calendar }}"
|
||||
changed_when: false
|
||||
check_mode: false
|
||||
when: atlas_manage_usb_reminder | bool
|
||||
|
||||
- name: Inspect the existing 45Drives notifier
|
||||
tags: [atlas, backup, usb_reminder]
|
||||
ansible.builtin.stat:
|
||||
path: "{{ atlas_usb_reminder_notifier }}"
|
||||
register: atlas_usb_reminder_notifier_file
|
||||
when: atlas_manage_usb_reminder | bool
|
||||
|
||||
- name: Require the configured 45Drives notifier for USB reminders
|
||||
tags: [atlas, backup, usb_reminder]
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- atlas_usb_reminder_notifier_file.stat.executable | default(false)
|
||||
fail_msg: >-
|
||||
The existing 45Drives Houston notifier must be executable.
|
||||
when: atlas_manage_usb_reminder | bool
|
||||
|
||||
- name: Install the 45Drives Atlas USB reminder helper
|
||||
tags: [atlas, backup, usb_reminder]
|
||||
ansible.builtin.template:
|
||||
src: atlas-usb-reminder.py.j2
|
||||
dest: /usr/local/libexec/atlas-usb-reminder
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0750"
|
||||
when: atlas_manage_usb_reminder | bool
|
||||
|
||||
- name: Install the 45Drives Atlas USB reminder service
|
||||
tags: [atlas, backup, usb_reminder]
|
||||
ansible.builtin.template:
|
||||
src: atlas-usb-reminder.service.j2
|
||||
dest: /etc/systemd/system/atlas-usb-reminder.service
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0644"
|
||||
when: atlas_manage_usb_reminder | bool
|
||||
|
||||
- name: Install the 45Drives Atlas USB reminder timer
|
||||
tags: [atlas, backup, usb_reminder]
|
||||
ansible.builtin.template:
|
||||
src: atlas-usb-reminder.timer.j2
|
||||
dest: /etc/systemd/system/atlas-usb-reminder.timer
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0644"
|
||||
when: atlas_manage_usb_reminder | bool
|
||||
|
||||
- name: Enable only the Atlas USB notification reminder timer
|
||||
tags: [atlas, backup, usb_reminder]
|
||||
ansible.builtin.systemd:
|
||||
name: atlas-usb-reminder.timer
|
||||
enabled: true
|
||||
state: started
|
||||
daemon_reload: true
|
||||
when:
|
||||
- atlas_manage_usb_reminder | bool
|
||||
- not ansible_check_mode
|
||||
@@ -14,6 +14,7 @@ ConditionPathExists={{ atlas_borg_known_hosts_path }}
|
||||
[Service]
|
||||
Type=oneshot
|
||||
ExecStart=/usr/local/sbin/atlas-borg-backup
|
||||
ExecStopPost=+/usr/local/sbin/atlas-borg-snapshot-cleanup
|
||||
User=root
|
||||
Group=root
|
||||
UMask=0077
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
#!/usr/bin/env bash
|
||||
set -Eeuo pipefail
|
||||
|
||||
export LC_ALL=C
|
||||
export LC_ALL=C.utf8
|
||||
export PATH=/usr/sbin:/usr/bin:/sbin:/bin
|
||||
export BORG_CACHE_DIR={{ atlas_borg_cache_dir | quote }}
|
||||
export BORG_CONFIG_DIR={{ atlas_borg_config_dir | quote }}
|
||||
@@ -17,13 +17,14 @@ readonly archive_prefix={{ atlas_borg_archive_prefix | quote }}
|
||||
readonly snapshot_prefix={{ atlas_borg_snapshot_prefix | quote }}
|
||||
readonly compression={{ atlas_borg_compression | quote }}
|
||||
readonly stage=/run/atlas-borg/source
|
||||
readonly snapshot_marker=/run/atlas-borg/snapshot-name
|
||||
readonly borg_user={{ atlas_borg_username | quote }}
|
||||
readonly borg_group={{ atlas_borg_group | quote }}
|
||||
readonly borg_home={{ atlas_borg_home | quote }}
|
||||
readonly borg_lock={{ atlas_borg_lock_path | quote }}
|
||||
readonly progress_filter=/usr/local/libexec/atlas-borg-progress
|
||||
|
||||
snapshot_name=""
|
||||
snapshot_created=false
|
||||
mounted_targets=()
|
||||
|
||||
# Invoked through the EXIT trap below.
|
||||
@@ -32,22 +33,31 @@ cleanup() {
|
||||
local status=$?
|
||||
local cleanup_status=0
|
||||
local index
|
||||
local source_mount_failed=false
|
||||
trap - EXIT HUP INT TERM
|
||||
set +e
|
||||
|
||||
{% raw %}
|
||||
for ((index = ${#mounted_targets[@]} - 1; index >= 0; index--)); do
|
||||
{% endraw %}
|
||||
if mountpoint -q "${mounted_targets[$index]}" && ! umount -R "${mounted_targets[$index]}"; then
|
||||
printf 'Source snapshot mount cleanup failed: %s\n' "${mounted_targets[$index]}" >&2
|
||||
source_mount_failed=true
|
||||
fi
|
||||
if mountpoint -q "${mounted_targets[$index]}"; then
|
||||
umount "${mounted_targets[$index]}" || cleanup_status=2
|
||||
printf 'Source snapshot mount is still active: %s\n' "${mounted_targets[$index]}" >&2
|
||||
source_mount_failed=true
|
||||
else
|
||||
rmdir -- "${mounted_targets[$index]}" 2>/dev/null || true
|
||||
fi
|
||||
done
|
||||
rm -rf "$stage" || cleanup_status=2
|
||||
|
||||
if [[ "$snapshot_created" == true ]]; then
|
||||
flock 9
|
||||
zfs destroy -r "${pool}@${snapshot_name}" || cleanup_status=2
|
||||
flock -u 9
|
||||
if [[ "$source_mount_failed" == false ]]; then
|
||||
if [[ -d "$stage" ]]; then
|
||||
rmdir -- "$stage" 2>/dev/null || cleanup_status=2
|
||||
fi
|
||||
else
|
||||
cleanup_status=2
|
||||
printf 'Source bind mount cleanup failed; keeping the snapshot for recovery\n' >&2
|
||||
fi
|
||||
|
||||
if ((status == 0 && cleanup_status != 0)); then
|
||||
@@ -77,7 +87,10 @@ exec 9>/run/lock/atlas-zfs-snapshot.lock
|
||||
zpool list -H -o name "$pool" >/dev/null
|
||||
rm -rf "$stage"
|
||||
mkdir -p "$stage"
|
||||
chown root:"$borg_group" /run/atlas-borg "$stage"
|
||||
# Keep systemd's root:root ownership of RuntimeDirectory: changing it makes
|
||||
# ExecStopPost re-chown its contents, which SELinux denies for the marker.
|
||||
setfacl -m "u:${borg_user}:rx" /run/atlas-borg
|
||||
chown root:"$borg_group" "$stage"
|
||||
chmod 0750 /run/atlas-borg "$stage"
|
||||
|
||||
flock 9
|
||||
@@ -96,8 +109,8 @@ timestamp="$(date -u +%Y%m%dT%H%M%SZ)"
|
||||
readonly timestamp
|
||||
snapshot_name="${snapshot_prefix}-${timestamp}"
|
||||
readonly snapshot_name
|
||||
printf '%s\n' "$snapshot_name" >"$snapshot_marker"
|
||||
zfs snapshot -r "${pool}@${snapshot_name}"
|
||||
snapshot_created=true
|
||||
flock -u 9
|
||||
printf 'Created recursive Borg source snapshot %s@%s\n' "$pool" "$snapshot_name"
|
||||
|
||||
@@ -116,32 +129,56 @@ while IFS=$'\t' read -r dataset dataset_mountpoint mounted; do
|
||||
target_path="${stage}${dataset_suffix}"
|
||||
mkdir -p "$target_path"
|
||||
mount --bind "$source_path" "$target_path"
|
||||
mount -o remount,bind,ro "$target_path"
|
||||
mounted_targets+=("$target_path")
|
||||
mount -o remount,bind,ro "$target_path"
|
||||
done < <(zfs list -H -o name,mountpoint,mounted -s name -r "$pool")
|
||||
|
||||
estimated_source_bytes=0
|
||||
while IFS=$'\t' read -r source_snapshot logical_bytes; do
|
||||
if [[ "$source_snapshot" == *"@${snapshot_name}" ]]; then
|
||||
[[ "$logical_bytes" =~ ^[0-9]+$ ]] || {
|
||||
printf 'Invalid logical size for Borg source snapshot %s\n' "$source_snapshot" >&2
|
||||
exit 74
|
||||
}
|
||||
estimated_source_bytes=$((estimated_source_bytes + logical_bytes))
|
||||
fi
|
||||
done < <(zfs list -H -p -t snapshot -o name,logicalreferenced -r "$pool")
|
||||
((estimated_source_bytes > 0)) || {
|
||||
printf 'Could not estimate the Borg source snapshot size\n' >&2
|
||||
exit 74
|
||||
}
|
||||
printf 'Estimated Borg source logical size: %s bytes (ZFS; progress percentage is approximate)\n' \
|
||||
"$estimated_source_bytes"
|
||||
|
||||
archive="${archive_prefix}-${timestamp}"
|
||||
readonly archive
|
||||
borg_status=0
|
||||
|
||||
printf 'Starting Borg archive %s from snapshot %s@%s\n' "$archive" "$pool" "$snapshot_name"
|
||||
set +e
|
||||
(
|
||||
cd /run/atlas-borg
|
||||
run_as_borg borg --remote-path "$remote_path" --lock-wait 600 create \
|
||||
run_as_borg borg --remote-path "$remote_path" --lock-wait 600 --log-json --progress create \
|
||||
--show-rc \
|
||||
--stats \
|
||||
--checkpoint-interval 900 \
|
||||
--compression "$compression" \
|
||||
"${repository}::${archive}" \
|
||||
source
|
||||
)
|
||||
create_status=$?
|
||||
source 2>&1
|
||||
) | /usr/bin/python3 -u "$progress_filter" --estimated-total-bytes "$estimated_source_bytes"
|
||||
create_pipeline_status=("${PIPESTATUS[@]}")
|
||||
set -e
|
||||
create_status=${create_pipeline_status[0]}
|
||||
if ((create_pipeline_status[1] != 0)); then
|
||||
printf 'Borg progress logging failed with status %s\n' "${create_pipeline_status[1]}" >&2
|
||||
exit 2
|
||||
fi
|
||||
if ((create_status >= 2)); then
|
||||
exit "$create_status"
|
||||
fi
|
||||
borg_status=$create_status
|
||||
|
||||
printf 'Borg archive %s created; applying retention\n' "$archive"
|
||||
set +e
|
||||
run_as_borg borg --remote-path "$remote_path" --lock-wait 600 prune \
|
||||
--show-rc \
|
||||
@@ -160,6 +197,7 @@ if ((prune_status > borg_status)); then
|
||||
borg_status=$prune_status
|
||||
fi
|
||||
|
||||
printf 'Borg retention complete; compacting repository\n'
|
||||
set +e
|
||||
run_as_borg borg --remote-path "$remote_path" --lock-wait 600 compact \
|
||||
--show-rc \
|
||||
@@ -173,4 +211,5 @@ if ((compact_status > borg_status)); then
|
||||
borg_status=$compact_status
|
||||
fi
|
||||
|
||||
printf 'Borg backup %s completed with status %s\n' "$archive" "$borg_status"
|
||||
exit "$borg_status"
|
||||
|
||||
@@ -0,0 +1,51 @@
|
||||
#!/usr/bin/env bash
|
||||
set -Eeuo pipefail
|
||||
|
||||
export PATH=/usr/sbin:/usr/bin:/sbin:/bin
|
||||
|
||||
readonly pool={{ atlas_zfs_pool | quote }}
|
||||
readonly mount_root={{ atlas_mount_root | quote }}
|
||||
readonly snapshot_prefix={{ atlas_borg_snapshot_prefix | quote }}
|
||||
readonly marker=/run/atlas-borg/snapshot-name
|
||||
|
||||
[[ -e "$marker" ]] || exit 0
|
||||
[[ -f "$marker" && ! -L "$marker" ]] || {
|
||||
printf 'Unsafe Atlas Borg snapshot marker; leaving snapshots unchanged\n' >&2
|
||||
exit 2
|
||||
}
|
||||
IFS= read -r snapshot_name <"$marker"
|
||||
[[ "$snapshot_name" =~ ^${snapshot_prefix}-[0-9]{8}T[0-9]{6}Z$ ]] || {
|
||||
printf 'Invalid Atlas Borg snapshot marker; leaving snapshots unchanged\n' >&2
|
||||
exit 2
|
||||
}
|
||||
|
||||
exec 9>/run/lock/atlas-zfs-snapshot.lock
|
||||
flock 9
|
||||
if zfs list -H -t snapshot -o name "${pool}@${snapshot_name}" >/dev/null 2>&1; then
|
||||
# The private bind mounts are gone, but ZFS may leave its on-demand
|
||||
# .zfs/snapshot mounts in the host namespace until explicitly unmounted.
|
||||
snapshot_mounts=()
|
||||
snapshot_sources=()
|
||||
while IFS=$'\t' read -r dataset dataset_mountpoint; do
|
||||
[[ "$dataset_mountpoint" == "$mount_root" || "$dataset_mountpoint" == "$mount_root/"* ]] || continue
|
||||
snapshot_mounts+=("${dataset_mountpoint}/.zfs/snapshot/${snapshot_name}")
|
||||
snapshot_sources+=("${dataset}@${snapshot_name}")
|
||||
done < <(zfs list -H -o name,mountpoint -s name -r "$pool")
|
||||
|
||||
{% raw %}
|
||||
for ((index = ${#snapshot_mounts[@]} - 1; index >= 0; index--)); do
|
||||
{% endraw %}
|
||||
mounted_source="$(findmnt -rn -M "${snapshot_mounts[$index]}" -o SOURCE || true)"
|
||||
[[ -n "$mounted_source" ]] || continue
|
||||
[[ "$mounted_source" == "${snapshot_sources[$index]}" ]] || {
|
||||
printf 'Unexpected source on Atlas Borg snapshot mount: %s\n' \
|
||||
"${snapshot_mounts[$index]}" >&2
|
||||
exit 2
|
||||
}
|
||||
umount "${snapshot_mounts[$index]}"
|
||||
done
|
||||
|
||||
zfs destroy -r "${pool}@${snapshot_name}"
|
||||
printf 'Removed recursive Atlas Borg source snapshot %s@%s after backup exit\n' \
|
||||
"$pool" "$snapshot_name"
|
||||
fi
|
||||
@@ -0,0 +1,30 @@
|
||||
# Managed by Ansible. Staging does not start automatically.
|
||||
[Unit]
|
||||
Description=Atlas rootless Gitea
|
||||
RequiresMountsFor={{ atlas_gitea_mountpoint }}
|
||||
|
||||
[Container]
|
||||
ContainerName=atlas-gitea
|
||||
Image={{ atlas_gitea_image }}
|
||||
UserNS=keep-id:uid={{ atlas_gitea_container_uid }},gid={{ atlas_gitea_container_gid }}
|
||||
{% if atlas_gitea_production_enabled | bool %}
|
||||
PublishPort={{ atlas_gitea_bind_address }}:{{ atlas_gitea_http_port }}:3000
|
||||
PublishPort={{ atlas_gitea_bind_address }}:{{ atlas_gitea_ssh_port }}:2222
|
||||
{% else %}
|
||||
PublishPort={{ atlas_gitea_staging_bind_address }}:{{ atlas_gitea_staging_http_port }}:3000
|
||||
PublishPort={{ atlas_gitea_staging_bind_address }}:{{ atlas_gitea_staging_ssh_port }}:2222
|
||||
{% endif %}
|
||||
Volume={{ atlas_gitea_mountpoint }}/data:/var/lib/gitea:Z
|
||||
Volume={{ atlas_gitea_mountpoint }}/config:/etc/gitea:Z
|
||||
NoNewPrivileges=true
|
||||
DropCapability=all
|
||||
|
||||
[Service]
|
||||
Restart=on-failure
|
||||
RestartSec=10
|
||||
TimeoutStartSec=900
|
||||
{% if atlas_gitea_production_enabled | bool %}
|
||||
|
||||
[Install]
|
||||
WantedBy=default.target
|
||||
{% endif %}
|
||||
@@ -0,0 +1,20 @@
|
||||
{
|
||||
"pool": {{ atlas_zfs_pool | to_json }},
|
||||
"backup_dataset": {{ (atlas_zfs_pool ~ '/' ~ atlas_zfs_dataset_backup) | to_json }},
|
||||
"notifier": {{ atlas_monitor_notifier | to_json }},
|
||||
"smart_devices": {{ atlas_monitor_smart_devices | to_json }},
|
||||
"timers": {{ atlas_monitor_effective_timers | to_json }},
|
||||
"failure_units": {{ atlas_monitor_effective_failure_units | to_json }},
|
||||
"remote_capacity": {{ atlas_monitor_remote_capacity | to_json }},
|
||||
"pool_warning_percent": {{ atlas_monitor_pool_warning_percent | int }},
|
||||
"pool_critical_percent": {{ atlas_monitor_pool_critical_percent | int }},
|
||||
"root_warning_percent": {{ atlas_monitor_root_warning_percent | int }},
|
||||
"root_critical_percent": {{ atlas_monitor_root_critical_percent | int }},
|
||||
"snapshot_warning_percent": {{ atlas_monitor_snapshot_warning_percent | int }},
|
||||
"snapshot_critical_percent": {{ atlas_monitor_snapshot_critical_percent | int }},
|
||||
"snapshot_growth_warning_gib_day": {{ atlas_monitor_snapshot_growth_warning_gib_day | int }},
|
||||
"backup_growth_warning_gib_day": {{ atlas_monitor_backup_growth_warning_gib_day | int }},
|
||||
"cpu_warning_c": {{ atlas_monitor_cpu_warning_c | int }},
|
||||
"cpu_critical_c": {{ atlas_monitor_cpu_critical_c | int }},
|
||||
"borg_max_runtime_days": {{ atlas_monitor_borg_max_runtime_days | int }}
|
||||
}
|
||||
@@ -0,0 +1,25 @@
|
||||
[Unit]
|
||||
Description=Check Atlas pool, disks, capacity, temperatures and maintenance jobs
|
||||
Wants=houston-dbus.service network-online.target
|
||||
After=zfs.target houston-dbus.service network-online.target
|
||||
ConditionFileIsExecutable=/usr/local/libexec/atlas-health-monitor
|
||||
|
||||
[Service]
|
||||
Type=oneshot
|
||||
ExecStart=/usr/local/libexec/atlas-health-monitor
|
||||
User=root
|
||||
Group=root
|
||||
UMask=0077
|
||||
StateDirectory=atlas-health-monitor
|
||||
StateDirectoryMode=0700
|
||||
NoNewPrivileges=true
|
||||
PrivateTmp=true
|
||||
ProtectHome=true
|
||||
ProtectSystem=strict
|
||||
ReadWritePaths=/var/lib/atlas-health-monitor
|
||||
ProtectKernelTunables=true
|
||||
ProtectKernelModules=true
|
||||
ProtectControlGroups=true
|
||||
RestrictAddressFamilies=AF_UNIX AF_INET AF_INET6
|
||||
RestrictRealtime=true
|
||||
LockPersonality=true
|
||||
@@ -0,0 +1,11 @@
|
||||
[Unit]
|
||||
Description=Schedule Atlas health checks
|
||||
|
||||
[Timer]
|
||||
OnCalendar={{ atlas_monitor_calendar }}
|
||||
Persistent=true
|
||||
RandomizedDelaySec=5min
|
||||
Unit=atlas-health-monitor.service
|
||||
|
||||
[Install]
|
||||
WantedBy=timers.target
|
||||
14
ansible/roles/profile_atlas/templates/atlas-icloudpd.conf.j2
Normal file
14
ansible/roles/profile_atlas/templates/atlas-icloudpd.conf.j2
Normal file
@@ -0,0 +1,14 @@
|
||||
# Managed by Ansible. Password, keyring and MFA cookies are stored separately in /config.
|
||||
apple_id={{ vault_atlas_icloudpd_apple_id }}
|
||||
authentication_type=MFA
|
||||
user=user
|
||||
user_id=1000
|
||||
group=group
|
||||
group_id=1000
|
||||
download_path=/home/user/iCloud
|
||||
folder_structure={:%Y/%m/%d}
|
||||
directory_permissions=750
|
||||
file_permissions=640
|
||||
download_interval=86400
|
||||
auto_delete=false
|
||||
delete_after_download=false
|
||||
@@ -0,0 +1,25 @@
|
||||
# Managed by Ansible. Start automatically with the lingering admin user manager.
|
||||
[Unit]
|
||||
Description=Atlas rootless iCloud Photos Downloader
|
||||
RequiresMountsFor={{ atlas_icloudpd_state_dir }} {{ atlas_icloudpd_photos_dir }}
|
||||
|
||||
[Container]
|
||||
ContainerName=atlas-icloudpd
|
||||
Image={{ atlas_icloudpd_image }}
|
||||
UserNS=keep-id:uid=1000,gid=1000
|
||||
# The image initialises its unprivileged UID 1000 account as container root.
|
||||
User=0
|
||||
# Upstream launcher requires traceroute for its iCloud reachability check.
|
||||
AddCapability=NET_RAW
|
||||
Environment=TZ={{ atlas_icloudpd_timezone }}
|
||||
Volume={{ atlas_icloudpd_photos_dir }}:/home/user/iCloud:z
|
||||
Volume={{ atlas_icloudpd_config_dir }}:/config:Z
|
||||
NoNewPrivileges=true
|
||||
|
||||
[Service]
|
||||
Restart=on-failure
|
||||
RestartSec=300
|
||||
TimeoutStartSec=900
|
||||
|
||||
[Install]
|
||||
WantedBy=default.target
|
||||
@@ -0,0 +1,2 @@
|
||||
[Unit]
|
||||
OnFailure=atlas-monitor-failure@%n.service
|
||||
@@ -0,0 +1,19 @@
|
||||
[Unit]
|
||||
Description=Submit a 45Drives Alert for failed Atlas job %i
|
||||
Requires=houston-dbus.service
|
||||
After=houston-dbus.service
|
||||
ConditionFileIsExecutable=/usr/local/libexec/atlas-health-monitor
|
||||
|
||||
[Service]
|
||||
Type=oneshot
|
||||
ExecStart=/usr/local/libexec/atlas-health-monitor --job-failed %i
|
||||
User=root
|
||||
Group=root
|
||||
UMask=0077
|
||||
NoNewPrivileges=true
|
||||
PrivateTmp=true
|
||||
ProtectHome=true
|
||||
ProtectSystem=strict
|
||||
RestrictAddressFamilies=AF_UNIX
|
||||
RestrictRealtime=true
|
||||
LockPersonality=true
|
||||
@@ -0,0 +1,10 @@
|
||||
[Unit]
|
||||
Description=Atlas recurring Nextcloud background jobs
|
||||
Requires=atlas-nextcloud.service
|
||||
After=atlas-nextcloud.service
|
||||
|
||||
[Service]
|
||||
Type=oneshot
|
||||
ExecStart=/usr/bin/podman exec --user 33 atlas-nextcloud php -f /var/www/html/cron.php
|
||||
TimeoutStartSec=15min
|
||||
NoNewPrivileges=true
|
||||
@@ -0,0 +1,10 @@
|
||||
[Unit]
|
||||
Description=Run Nextcloud background jobs every five minutes
|
||||
|
||||
[Timer]
|
||||
OnBootSec=5min
|
||||
OnUnitActiveSec=5min
|
||||
Unit=atlas-nextcloud-cron.service
|
||||
|
||||
[Install]
|
||||
WantedBy=timers.target
|
||||
@@ -0,0 +1,27 @@
|
||||
[Unit]
|
||||
Description=Atlas Nextcloud PostgreSQL
|
||||
RequiresMountsFor={{ atlas_nextcloud_root }}/database
|
||||
|
||||
[Container]
|
||||
ContainerName=atlas-nextcloud-db
|
||||
Image={{ atlas_nextcloud_postgres_image }}
|
||||
Network=atlas-nextcloud.network
|
||||
NetworkAlias=atlas-nextcloud-db
|
||||
Environment=POSTGRES_DB=nextcloud
|
||||
Environment=POSTGRES_USER=nextcloud
|
||||
Environment=POSTGRES_PASSWORD_FILE=/run/secrets/postgres-password
|
||||
Volume={{ atlas_nextcloud_private_dir }}/postgres-password:/run/secrets/postgres-password:ro,z
|
||||
Volume={{ atlas_nextcloud_root }}/database:/var/lib/postgresql/data:Z
|
||||
PodmanArgs=--memory=1g
|
||||
HealthCmd=pg_isready -U nextcloud -d nextcloud
|
||||
HealthInterval=30s
|
||||
HealthStartPeriod=60s
|
||||
NoNewPrivileges=true
|
||||
|
||||
[Service]
|
||||
Restart=on-failure
|
||||
RestartSec=10
|
||||
TimeoutStartSec=900
|
||||
|
||||
[Install]
|
||||
WantedBy=default.target
|
||||
@@ -0,0 +1,8 @@
|
||||
bind 0.0.0.0
|
||||
protected-mode yes
|
||||
port 6379
|
||||
requirepass {{ vault_nextcloud_redis_password }}
|
||||
maxmemory 128mb
|
||||
maxmemory-policy noeviction
|
||||
save ""
|
||||
appendonly no
|
||||
@@ -0,0 +1,22 @@
|
||||
[Unit]
|
||||
Description=Atlas Nextcloud private Redis
|
||||
RequiresMountsFor={{ atlas_nextcloud_root }}/cache
|
||||
|
||||
[Container]
|
||||
ContainerName=atlas-nextcloud-redis
|
||||
Image={{ atlas_nextcloud_redis_image }}
|
||||
Network=atlas-nextcloud.network
|
||||
NetworkAlias=atlas-nextcloud-redis
|
||||
Volume={{ atlas_nextcloud_private_dir }}/redis.conf:/usr/local/etc/redis/atlas.conf:ro,z
|
||||
Volume={{ atlas_nextcloud_root }}/cache:/data:Z
|
||||
Exec=redis-server /usr/local/etc/redis/atlas.conf
|
||||
PodmanArgs=--memory=256m
|
||||
NoNewPrivileges=true
|
||||
|
||||
[Service]
|
||||
Restart=on-failure
|
||||
RestartSec=10
|
||||
TimeoutStartSec=900
|
||||
|
||||
[Install]
|
||||
WantedBy=default.target
|
||||
@@ -0,0 +1,23 @@
|
||||
<?php
|
||||
// Managed ongoing application settings; never import or migrate user data.
|
||||
$CONFIG = [
|
||||
'trusted_domains' => ['{{ atlas_nextcloud_domain }}', 'atlas-nextcloud'],
|
||||
'trusted_proxies' => ['{{ atlas_aegis_ip }}', '{{ atlas_nextcloud_network_gateway }}'],
|
||||
'overwrite.cli.url' => 'https://{{ atlas_nextcloud_domain }}',
|
||||
'overwritehost' => '{{ atlas_nextcloud_domain }}',
|
||||
'overwriteprotocol' => 'https',
|
||||
'allow_local_remote_servers' => true,
|
||||
'default_quota' => 'none',
|
||||
'skeletondirectory' => '',
|
||||
'maintenance_window_start' => 1,
|
||||
'default_phone_region' => 'IT',
|
||||
'twofactor_enforced' => false,
|
||||
'onlyoffice' => [
|
||||
'DocumentServerUrl' => 'https://{{ atlas_onlyoffice_domain }}/',
|
||||
'DocumentServerInternalUrl' => 'http://atlas-onlyoffice/',
|
||||
'StorageUrl' => 'http://atlas-nextcloud/',
|
||||
'jwt_secret' => trim(file_get_contents('/run/secrets/onlyoffice-jwt')),
|
||||
'jwt_header' => 'AuthorizationJwt',
|
||||
'allow_local_address' => true,
|
||||
],
|
||||
];
|
||||
@@ -0,0 +1,42 @@
|
||||
[Unit]
|
||||
Description=Atlas Nextcloud
|
||||
Requires=atlas-nextcloud-db.service atlas-nextcloud-redis.service
|
||||
After=atlas-nextcloud-db.service atlas-nextcloud-redis.service
|
||||
RequiresMountsFor={{ atlas_nextcloud_root }}/app {{ atlas_nextcloud_root }}/files
|
||||
|
||||
[Container]
|
||||
ContainerName=atlas-nextcloud
|
||||
Image={{ atlas_nextcloud_image }}
|
||||
Network=atlas-nextcloud.network
|
||||
NetworkAlias=atlas-nextcloud
|
||||
PublishPort={{ ansible_host }}:{{ atlas_nextcloud_http_port }}:80
|
||||
PublishPort=127.0.0.1:{{ atlas_nextcloud_http_port }}:80
|
||||
Environment=POSTGRES_HOST=atlas-nextcloud-db
|
||||
Environment=POSTGRES_DB=nextcloud
|
||||
Environment=POSTGRES_USER=nextcloud
|
||||
Environment=POSTGRES_PASSWORD_FILE=/run/secrets/postgres-password
|
||||
Environment=NEXTCLOUD_ADMIN_USER={{ atlas_nextcloud_admin }}
|
||||
Environment=NEXTCLOUD_ADMIN_PASSWORD_FILE=/run/secrets/admin-password
|
||||
Environment="NEXTCLOUD_TRUSTED_DOMAINS={{ atlas_nextcloud_domain }} atlas-nextcloud"
|
||||
Environment=REDIS_HOST=atlas-nextcloud-redis
|
||||
Environment=REDIS_HOST_PASSWORD_FILE=/run/secrets/redis-password
|
||||
Environment=APACHE_DISABLE_REWRITE_IP=1
|
||||
Environment=PHP_MEMORY_LIMIT=512M
|
||||
Environment=PHP_UPLOAD_LIMIT=2G
|
||||
Volume={{ atlas_nextcloud_root }}/app:/var/www/html:Z
|
||||
Volume={{ atlas_nextcloud_root }}/files:/var/www/html/data:Z
|
||||
Volume={{ atlas_nextcloud_app_cache }}:/mnt/atlas-apps:ro,z
|
||||
Volume={{ atlas_nextcloud_private_dir }}/postgres-password:/run/secrets/postgres-password:ro,z
|
||||
Volume={{ atlas_nextcloud_private_dir }}/admin-password:/run/secrets/admin-password:ro,z
|
||||
Volume={{ atlas_nextcloud_private_dir }}/redis-password:/run/secrets/redis-password:ro,z
|
||||
Volume={{ atlas_nextcloud_private_dir }}/onlyoffice-jwt:/run/secrets/onlyoffice-jwt:ro,z
|
||||
PodmanArgs=--memory=2g
|
||||
NoNewPrivileges=true
|
||||
|
||||
[Service]
|
||||
Restart=on-failure
|
||||
RestartSec=10
|
||||
TimeoutStartSec=900
|
||||
|
||||
[Install]
|
||||
WantedBy=default.target
|
||||
@@ -0,0 +1,5 @@
|
||||
# Managed by Ansible: private rootless application network, no host services.
|
||||
[Network]
|
||||
NetworkName=atlas-nextcloud
|
||||
Subnet={{ atlas_nextcloud_network_subnet }}
|
||||
Gateway={{ atlas_nextcloud_network_gateway }}
|
||||
@@ -0,0 +1,26 @@
|
||||
[Unit]
|
||||
Description=Atlas ONLYOFFICE Docs Community
|
||||
RequiresMountsFor={{ atlas_nextcloud_root }}/office
|
||||
|
||||
[Container]
|
||||
ContainerName=atlas-onlyoffice
|
||||
Image={{ atlas_onlyoffice_image }}
|
||||
Network=atlas-nextcloud.network
|
||||
NetworkAlias=atlas-onlyoffice
|
||||
PublishPort={{ ansible_host }}:{{ atlas_onlyoffice_http_port }}:80
|
||||
PublishPort=127.0.0.1:{{ atlas_onlyoffice_http_port }}:80
|
||||
EnvironmentFile={{ atlas_nextcloud_private_dir }}/onlyoffice.env
|
||||
Volume={{ atlas_nextcloud_root }}/office/data:/var/www/onlyoffice/Data:Z
|
||||
Volume={{ atlas_nextcloud_root }}/office/lib:/var/lib/onlyoffice:Z
|
||||
Volume={{ atlas_nextcloud_root }}/office/logs:/var/log/onlyoffice:Z
|
||||
Volume={{ atlas_nextcloud_root }}/office/database:/var/lib/postgresql:Z
|
||||
PodmanArgs=--memory=4g --shm-size=256m
|
||||
NoNewPrivileges=true
|
||||
|
||||
[Service]
|
||||
Restart=on-failure
|
||||
RestartSec=10
|
||||
TimeoutStartSec=1200
|
||||
|
||||
[Install]
|
||||
WantedBy=default.target
|
||||
@@ -0,0 +1,7 @@
|
||||
JWT_ENABLED=true
|
||||
JWT_SECRET={{ vault_nextcloud_onlyoffice_jwt }}
|
||||
JWT_HEADER=AuthorizationJwt
|
||||
ALLOW_PRIVATE_IP_ADDRESS=true
|
||||
ALLOW_META_IP_ADDRESS=false
|
||||
USE_UNAUTHORIZED_STORAGE=false
|
||||
WOPI_ENABLED=false
|
||||
@@ -0,0 +1,19 @@
|
||||
[Unit]
|
||||
Description=Pull a prepared read-only Prometheus backup to Atlas
|
||||
RequiresMountsFor={{ atlas_backup_prometheus_mountpoint }}
|
||||
Wants=network-online.target
|
||||
After=network-online.target zfs.target
|
||||
ConditionFileIsExecutable=/usr/local/sbin/atlas-prometheus-pull
|
||||
ConditionPathExists={{ atlas_prometheus_pull_private_key_path }}
|
||||
ConditionPathExists={{ atlas_prometheus_pull_known_hosts_path }}
|
||||
|
||||
[Service]
|
||||
Type=oneshot
|
||||
ExecStart=/usr/local/sbin/atlas-prometheus-pull
|
||||
User=root
|
||||
Group=root
|
||||
UMask=0077
|
||||
TimeoutStartSec=infinity
|
||||
Nice=15
|
||||
IOSchedulingClass=best-effort
|
||||
IOSchedulingPriority=7
|
||||
@@ -0,0 +1,75 @@
|
||||
#!/usr/bin/env bash
|
||||
set -Eeuo pipefail
|
||||
umask 077
|
||||
|
||||
backup_root={{ atlas_backup_prometheus_mountpoint | quote }}
|
||||
snapshots="$backup_root/snapshots"
|
||||
stage=''
|
||||
exec 9>/run/lock/atlas-prometheus-pull.lock
|
||||
flock -n 9 || { echo 'A Prometheus pull is already running' >&2; exit 1; }
|
||||
|
||||
cleanup() {
|
||||
local rc=$?
|
||||
trap - EXIT
|
||||
if (( rc != 0 )) && [[ -n "$stage" && -d "$stage" ]]; then
|
||||
rm -rf -- "$stage"
|
||||
fi
|
||||
exit "$rc"
|
||||
}
|
||||
trap cleanup EXIT
|
||||
|
||||
zpool list -H -o name {{ atlas_zfs_pool | quote }} >/dev/null
|
||||
findmnt -rn --mountpoint "$backup_root" >/dev/null
|
||||
stage=$(mktemp -d "$backup_root/.staging.XXXXXXXX")
|
||||
ssh_cmd='/usr/bin/ssh -F /dev/null -o BatchMode=yes -o StrictHostKeyChecking=yes -o UserKnownHostsFile={{ atlas_prometheus_pull_known_hosts_path }} -o IdentitiesOnly=yes -i {{ atlas_prometheus_pull_private_key_path }} -p {{ atlas_prometheus_pull_source_port }}'
|
||||
rsync -a --partial --delay-updates -e "$ssh_cmd" \
|
||||
{{ (atlas_prometheus_pull_source_user ~ '@' ~ hostvars['prometheus'].ansible_host ~ ':current/') | quote }} \
|
||||
"$stage/"
|
||||
|
||||
test -s "$stage/payload.tar"
|
||||
test -s "$stage/payload.sha256"
|
||||
test -s "$stage/metadata.json"
|
||||
(cd "$stage" && sha256sum -c payload.sha256)
|
||||
tar -tf "$stage/payload.tar" >/dev/null
|
||||
stamp=$(python3 - "$stage/metadata.json" <<'PY'
|
||||
import json
|
||||
import datetime as dt
|
||||
import re
|
||||
import sys
|
||||
|
||||
with open(sys.argv[1], encoding="utf-8") as stream:
|
||||
metadata = json.load(stream)
|
||||
stamp = metadata.get("created_utc", "")
|
||||
if metadata.get("schema") != 1 or metadata.get("host") != "prometheus":
|
||||
raise SystemExit("Unexpected Prometheus backup metadata")
|
||||
if not re.fullmatch(r"[0-9]{8}T[0-9]{6}Z", stamp):
|
||||
raise SystemExit("Invalid Prometheus backup timestamp")
|
||||
created = dt.datetime.strptime(stamp, "%Y%m%dT%H%M%SZ").replace(tzinfo=dt.timezone.utc)
|
||||
age = dt.datetime.now(dt.timezone.utc) - created
|
||||
if age.total_seconds() < -300 or age > dt.timedelta(hours={{ atlas_prometheus_pull_max_age_hours }}):
|
||||
raise SystemExit("Prometheus backup is outside the configured freshness window")
|
||||
print(stamp)
|
||||
PY
|
||||
)
|
||||
if [[ -e "$snapshots/$stamp" ]]; then
|
||||
cmp "$stage/payload.sha256" "$snapshots/$stamp/payload.sha256"
|
||||
cmp "$stage/metadata.json" "$snapshots/$stamp/metadata.json"
|
||||
(cd "$snapshots/$stamp" && sha256sum -c payload.sha256)
|
||||
rm -rf -- "${stage:?}"
|
||||
stage=''
|
||||
else
|
||||
chown -R root:root "$stage"
|
||||
chmod 0700 "$stage"
|
||||
chmod 0600 "$stage/payload.tar" "$stage/payload.sha256" "$stage/metadata.json"
|
||||
mv -- "$stage" "$snapshots/$stamp"
|
||||
stage=''
|
||||
fi
|
||||
latest_link=$(readlink "$backup_root/latest" 2>/dev/null || true)
|
||||
latest_stamp=${latest_link##*/}
|
||||
if [[ -z "$latest_stamp" || "$stamp" > "$latest_stamp" ]]; then
|
||||
ln -s "snapshots/$stamp" "$backup_root/.latest.new"
|
||||
mv -Tf -- "$backup_root/.latest.new" "$backup_root/latest"
|
||||
fi
|
||||
python3 /usr/local/libexec/atlas-prometheus-prune "$snapshots" \
|
||||
{{ atlas_prometheus_pull_keep_daily }} {{ atlas_prometheus_pull_keep_weekly }} {{ atlas_prometheus_pull_keep_monthly }}
|
||||
echo "Verified and published Prometheus backup $stamp"
|
||||
@@ -0,0 +1,10 @@
|
||||
[Unit]
|
||||
Description=Schedule Atlas pull of prepared Prometheus backups
|
||||
|
||||
[Timer]
|
||||
OnCalendar={{ atlas_prometheus_pull_calendar }}
|
||||
Persistent=true
|
||||
Unit=atlas-prometheus-pull.service
|
||||
|
||||
[Install]
|
||||
WantedBy=timers.target
|
||||
@@ -0,0 +1,30 @@
|
||||
[Unit]
|
||||
Description=Run a manual, UUID-bound offline USB backup of Atlas ZFS datasets
|
||||
Requires=zfs.target
|
||||
After=zfs.target
|
||||
ConditionFileIsExecutable=/usr/local/sbin/atlas-usb-backup
|
||||
|
||||
[Service]
|
||||
Type=oneshot
|
||||
ExecStart=/usr/local/sbin/atlas-usb-backup
|
||||
ExecStopPost=+/usr/local/sbin/atlas-usb-snapshot-cleanup
|
||||
User=root
|
||||
Group=root
|
||||
UMask=0077
|
||||
TimeoutStartSec=infinity
|
||||
RuntimeDirectory=atlas-usb-backup
|
||||
RuntimeDirectoryMode=0700
|
||||
Nice=15
|
||||
IOSchedulingClass=best-effort
|
||||
IOSchedulingPriority=7
|
||||
PrivateMounts=true
|
||||
PrivateTmp=true
|
||||
ProtectHome=true
|
||||
ProtectSystem=strict
|
||||
ReadWritePaths=/run/atlas-usb-backup /run/lock
|
||||
ProtectKernelTunables=true
|
||||
ProtectKernelModules=true
|
||||
ProtectControlGroups=true
|
||||
RestrictAddressFamilies=AF_UNIX
|
||||
RestrictRealtime=true
|
||||
LockPersonality=true
|
||||
242
ansible/roles/profile_atlas/templates/atlas-usb-backup.sh.j2
Normal file
242
ansible/roles/profile_atlas/templates/atlas-usb-backup.sh.j2
Normal file
@@ -0,0 +1,242 @@
|
||||
#!/usr/bin/env bash
|
||||
set -Eeuo pipefail
|
||||
|
||||
export LC_ALL=C.utf8
|
||||
export PATH=/usr/sbin:/usr/bin:/sbin:/bin
|
||||
|
||||
readonly pool={{ atlas_zfs_pool | quote }}
|
||||
readonly mount_root={{ atlas_mount_root | quote }}
|
||||
readonly luks_uuid={{ atlas_usb_backup_luks_uuid | quote }}
|
||||
readonly fs_uuid={{ atlas_usb_backup_fs_uuid | quote }}
|
||||
readonly mapper_name={{ atlas_usb_backup_mapper_name | quote }}
|
||||
readonly snapshot_prefix={{ atlas_usb_backup_snapshot_prefix | quote }}
|
||||
readonly min_free_bytes={{ atlas_usb_backup_min_free_bytes | int }}
|
||||
readonly mapper="/dev/mapper/${mapper_name}"
|
||||
readonly outer="/dev/disk/by-uuid/${luks_uuid}"
|
||||
readonly runtime_dir=/run/atlas-usb-backup
|
||||
readonly snapshot_marker="${runtime_dir}/snapshot-name"
|
||||
readonly source_dir="${runtime_dir}/source"
|
||||
readonly usb_mount="${runtime_dir}/target"
|
||||
readonly backup_root="${usb_mount}/atlas"
|
||||
|
||||
snapshot_name=""
|
||||
mapper_opened_by_script=false
|
||||
usb_mounted=false
|
||||
published=false
|
||||
partial=""
|
||||
mounted_targets=()
|
||||
|
||||
# shellcheck disable=SC2329
|
||||
cleanup() {
|
||||
local status=$?
|
||||
local cleanup_status=0
|
||||
local index
|
||||
local source_mount_failed=false
|
||||
trap - EXIT HUP INT TERM
|
||||
set +e
|
||||
|
||||
if [[ -n "$partial" && "$published" == false && "$usb_mounted" == true ]]; then
|
||||
rm -rf -- "$partial" || cleanup_status=2
|
||||
fi
|
||||
if [[ "$usb_mounted" == true ]]; then
|
||||
umount "$usb_mount" || cleanup_status=2
|
||||
fi
|
||||
|
||||
{% raw %}
|
||||
for ((index = ${#mounted_targets[@]} - 1; index >= 0; index--)); do
|
||||
{% endraw %}
|
||||
if mountpoint -q "${mounted_targets[$index]}" && ! umount -R "${mounted_targets[$index]}"; then
|
||||
printf 'Source snapshot mount cleanup failed: %s\n' "${mounted_targets[$index]}" >&2
|
||||
source_mount_failed=true
|
||||
fi
|
||||
if mountpoint -q "${mounted_targets[$index]}"; then
|
||||
printf 'Source snapshot mount is still active: %s\n' "${mounted_targets[$index]}" >&2
|
||||
source_mount_failed=true
|
||||
else
|
||||
rmdir -- "${mounted_targets[$index]}" 2>/dev/null || true
|
||||
fi
|
||||
done
|
||||
if [[ "$source_mount_failed" == false ]]; then
|
||||
rmdir -- "$source_dir" 2>/dev/null || true
|
||||
else
|
||||
cleanup_status=2
|
||||
printf 'Source bind mount cleanup failed; keeping the snapshot for recovery\n' >&2
|
||||
fi
|
||||
|
||||
if [[ "$usb_mounted" == true || "$mapper_opened_by_script" == true ]] &&
|
||||
! mountpoint -q "$usb_mount" &&
|
||||
! findmnt -rn -S "$mapper" >/dev/null; then
|
||||
cryptsetup close "$mapper_name" || cleanup_status=2
|
||||
fi
|
||||
rmdir -- "$usb_mount" 2>/dev/null || true
|
||||
|
||||
if ((status == 0 && cleanup_status != 0)); then
|
||||
status=$cleanup_status
|
||||
fi
|
||||
exit "$status"
|
||||
}
|
||||
|
||||
trap cleanup EXIT
|
||||
trap 'exit 143' HUP INT TERM
|
||||
|
||||
exec 8>/run/lock/atlas-usb-backup.lock
|
||||
flock -n 8 || { printf 'Atlas USB backup is already running\n' >&2; exit 75; }
|
||||
exec 9>/run/lock/atlas-zfs-snapshot.lock
|
||||
|
||||
zpool list -H -o name "$pool" >/dev/null
|
||||
[[ -b "$outer" ]] || { printf 'Configured LUKS UUID is not connected\n' >&2; exit 66; }
|
||||
[[ "$(blkid -s TYPE -o value "$outer")" == crypto_LUKS ]] || {
|
||||
printf 'Configured outer UUID is not a LUKS container\n' >&2
|
||||
exit 65
|
||||
}
|
||||
[[ "$(blkid -s UUID -o value "$outer")" == "$luks_uuid" ]] || exit 65
|
||||
if ! cryptsetup status "$mapper_name" >/dev/null; then
|
||||
printf 'Requesting the LUKS passphrase for the configured USB disk\n'
|
||||
systemd-ask-password -n --no-tty --timeout=300 \
|
||||
--id="atlas-usb-backup:${luks_uuid}" \
|
||||
'Atlas offline USB backup LUKS passphrase:' |
|
||||
cryptsetup open --type luks2 --key-file - "$outer" "$mapper_name"
|
||||
mapper_opened_by_script=true
|
||||
fi
|
||||
backing_device="$(cryptsetup status "$mapper_name" | awk '$1 == "device:" { print $2 }')"
|
||||
[[ -n "$backing_device" && "$(readlink -f "$backing_device")" == "$(readlink -f "$outer")" ]] || {
|
||||
printf 'The unlocked mapper does not belong to the configured LUKS UUID\n' >&2
|
||||
exit 65
|
||||
}
|
||||
[[ "$(blkid -s TYPE -o value "$mapper")" == ext4 ]] || {
|
||||
printf 'The unlocked USB filesystem is not ext4\n' >&2
|
||||
exit 65
|
||||
}
|
||||
[[ "$(blkid -s UUID -o value "$mapper")" == "$fs_uuid" ]] || {
|
||||
printf 'The unlocked USB filesystem UUID does not match\n' >&2
|
||||
exit 65
|
||||
}
|
||||
if findmnt -rn -S "$mapper" >/dev/null; then
|
||||
printf 'The USB filesystem is already mounted elsewhere\n' >&2
|
||||
exit 65
|
||||
fi
|
||||
[[ ! -e "$source_dir" && ! -e "$usb_mount" ]] || {
|
||||
printf 'USB backup staging directories already exist; inspect them manually\n' >&2
|
||||
exit 65
|
||||
}
|
||||
|
||||
mkdir -m 0700 "$usb_mount"
|
||||
mount -t ext4 -o nodev,nosuid,noexec "$mapper" "$usb_mount"
|
||||
usb_mounted=true
|
||||
[[ "$(readlink -f "$(findmnt -nro SOURCE --target "$usb_mount")")" == "$(readlink -f "$mapper")" ]] || {
|
||||
printf 'Mounted USB source does not match the verified mapper\n' >&2
|
||||
exit 65
|
||||
}
|
||||
|
||||
for path in "$backup_root" "$backup_root/snapshots"; do
|
||||
[[ ! -L "$path" ]] || { printf 'Unsafe symlink in USB backup destination\n' >&2; exit 65; }
|
||||
mkdir -p -- "$path"
|
||||
[[ -d "$path" ]] || exit 65
|
||||
chown root:root -- "$path"
|
||||
chmod 0700 -- "$path"
|
||||
done
|
||||
|
||||
free_bytes="$(df -B1 --output=avail "$usb_mount" | tail -n 1 | tr -d ' ')"
|
||||
if ((free_bytes < min_free_bytes)); then
|
||||
printf 'USB free space (%s bytes) is below the required reserve (%s bytes)\n' \
|
||||
"$free_bytes" "$min_free_bytes" >&2
|
||||
exit 73
|
||||
fi
|
||||
|
||||
mkdir -m 0700 "$source_dir"
|
||||
flock 9
|
||||
timestamp="$(date -u +%Y%m%dT%H%M%SZ)"
|
||||
snapshot_name="${snapshot_prefix}-${timestamp}-$$"
|
||||
printf '%s\n' "$snapshot_name" >"$snapshot_marker"
|
||||
zfs snapshot -r "${pool}@${snapshot_name}"
|
||||
flock -u 9
|
||||
printf 'Created recursive USB source snapshot %s@%s\n' "$pool" "$snapshot_name"
|
||||
|
||||
while IFS=$'\t' read -r dataset dataset_mountpoint mounted; do
|
||||
if [[ "$mounted" != yes ]]; then
|
||||
printf 'Dataset %s is not mounted; refusing an incomplete backup\n' "$dataset" >&2
|
||||
exit 65
|
||||
fi
|
||||
if [[ "$dataset_mountpoint" != "$mount_root" && "$dataset_mountpoint" != "$mount_root/"* ]]; then
|
||||
printf 'Dataset %s has unexpected mountpoint %s\n' "$dataset" "$dataset_mountpoint" >&2
|
||||
exit 65
|
||||
fi
|
||||
dataset_suffix="${dataset#"$pool"}"
|
||||
source_path="${dataset_mountpoint}/.zfs/snapshot/${snapshot_name}"
|
||||
target_path="${source_dir}${dataset_suffix}"
|
||||
mkdir -p "$target_path"
|
||||
mount --bind "$source_path" "$target_path"
|
||||
mounted_targets+=("$target_path")
|
||||
mount -o remount,bind,ro "$target_path"
|
||||
done < <(zfs list -H -o name,mountpoint,mounted -s name -r "$pool")
|
||||
|
||||
previous=""
|
||||
if [[ -e "$backup_root/latest" || -L "$backup_root/latest" ]]; then
|
||||
[[ -L "$backup_root/latest" ]] || { printf 'latest is not a symlink\n' >&2; exit 65; }
|
||||
previous="$(readlink -e "$backup_root/latest")"
|
||||
[[ -n "$previous" && "$previous" == "$backup_root/snapshots/"* && -d "$previous" ]] || {
|
||||
printf 'latest does not point to a complete snapshot on the USB disk\n' >&2
|
||||
exit 65
|
||||
}
|
||||
fi
|
||||
|
||||
backup_name="${timestamp}-$$"
|
||||
candidate_partial="${backup_root}/snapshots/.incomplete-${backup_name}"
|
||||
complete="${backup_root}/snapshots/${backup_name}"
|
||||
[[ ! -e "$candidate_partial" && ! -L "$candidate_partial" && ! -e "$complete" && ! -L "$complete" ]] || exit 65
|
||||
mkdir -m 0700 "$candidate_partial"
|
||||
partial="$candidate_partial"
|
||||
|
||||
printf 'Copying the consistent pool tree to USB backup %s\n' "$backup_name"
|
||||
# Preserve POSIX ACLs, ownership, modes, timestamps, hard links, and sparse
|
||||
# files. Do not preserve generic xattrs: Rocky 9's rsync 3.2.7 fails when
|
||||
# combining xattrs with --link-dest, while SELinux labels were intentionally
|
||||
# excluded because restores must relabel for their destination host.
|
||||
rsync_args=(-aHAS --numeric-ids "--info=progress2,stats2")
|
||||
estimate_args=(-aHAS --numeric-ids --dry-run --stats)
|
||||
if [[ -n "$previous" ]]; then
|
||||
rsync_args+=("--link-dest=$previous")
|
||||
estimate_args+=("--link-dest=$previous")
|
||||
fi
|
||||
|
||||
# The rsync dry run estimates changed file bytes after link-dest deduplication.
|
||||
# Metadata and filesystem allocation still require the separate free-space reserve.
|
||||
estimate="$(rsync "${estimate_args[@]}" "${source_dir}/" "${partial}/")"
|
||||
transfer_bytes="$(printf '%s\n' "$estimate" | awk -F: \
|
||||
'/^Total transferred file size:/ { gsub(/[^0-9]/, "", $2); print $2 }')"
|
||||
[[ "$transfer_bytes" =~ ^[0-9]+$ ]] || {
|
||||
printf 'Could not determine the USB transfer size\n' >&2
|
||||
exit 74
|
||||
}
|
||||
if ((free_bytes - transfer_bytes < min_free_bytes)); then
|
||||
printf 'Insufficient USB space: %s bytes free, %s estimated transfer, %s reserved\n' \
|
||||
"$free_bytes" "$transfer_bytes" "$min_free_bytes" >&2
|
||||
exit 73
|
||||
fi
|
||||
|
||||
rsync "${rsync_args[@]}" "${source_dir}/" "${partial}/"
|
||||
|
||||
printf 'Verifying USB backup %s with a checksum-based dry run\n' "$backup_name"
|
||||
verification="${runtime_dir}/verification.out"
|
||||
rsync -aHAS --numeric-ids \
|
||||
--checksum --dry-run --delete --itemize-changes \
|
||||
"${source_dir}/" "${partial}/" >"$verification"
|
||||
if [[ -s "$verification" ]]; then
|
||||
printf 'USB verification found mismatches; refusing to publish the backup\n' >&2
|
||||
exit 74
|
||||
fi
|
||||
|
||||
free_bytes="$(df -B1 --output=avail "$usb_mount" | tail -n 1 | tr -d ' ')"
|
||||
if ((free_bytes < min_free_bytes)); then
|
||||
printf 'USB backup completed below the free-space reserve; refusing to publish it\n' >&2
|
||||
exit 73
|
||||
fi
|
||||
|
||||
mv -- "$partial" "$complete"
|
||||
partial=""
|
||||
ln -s "snapshots/${backup_name}" "${backup_root}/.latest-${backup_name}"
|
||||
mv -Tf -- "${backup_root}/.latest-${backup_name}" "${backup_root}/latest"
|
||||
published=true
|
||||
sync -f "$complete"
|
||||
sync -f "$backup_root"
|
||||
printf 'USB backup %s verified and published; unmounting and closing LUKS\n' "$backup_name"
|
||||
@@ -0,0 +1,31 @@
|
||||
#!/usr/bin/python3
|
||||
"""Submit a manual-backup reminder through Atlas' existing Houston notifier."""
|
||||
|
||||
import json
|
||||
import subprocess
|
||||
from datetime import datetime, timezone
|
||||
|
||||
|
||||
now = datetime.now(timezone.utc)
|
||||
message = {
|
||||
"timestamp": now.isoformat(timespec="seconds"),
|
||||
"unixtime": int(now.timestamp()),
|
||||
"event": "atlas_usb_backup_reminder",
|
||||
"severity": "warning",
|
||||
"subject": "Promemoria backup USB offline Atlas",
|
||||
"email_message": (
|
||||
"Collega il disco USB di backup ad Atlas ed esegui manualmente il backup offline.\n"
|
||||
"Il promemoria non avvia il backup. Controlla che il disco non sia\n"
|
||||
"montato; poi esegui:\n\n"
|
||||
" sudo systemctl start atlas-usb-backup.service\n\n"
|
||||
"Verifica l'esito con:\n"
|
||||
" sudo journalctl -u atlas-usb-backup.service -n 100 --no-pager\n\n"
|
||||
"Dopo la riuscita, scollega fisicamente il disco."
|
||||
),
|
||||
}
|
||||
|
||||
subprocess.run(
|
||||
[{{ atlas_usb_reminder_notifier | to_json }}, json.dumps(message)],
|
||||
check=True,
|
||||
)
|
||||
print("Atlas USB backup reminder submitted to 45Drives Alerts; email delivery is not verified.", flush=True)
|
||||
@@ -0,0 +1,22 @@
|
||||
[Unit]
|
||||
Description=45Drives Alerts reminder to run the manual Atlas offline USB backup
|
||||
Requires=houston-dbus.service
|
||||
After=houston-dbus.service
|
||||
ConditionFileIsExecutable=/usr/local/libexec/atlas-usb-reminder
|
||||
|
||||
[Service]
|
||||
Type=oneshot
|
||||
ExecStart=/usr/local/libexec/atlas-usb-reminder
|
||||
User=root
|
||||
Group=root
|
||||
UMask=0077
|
||||
NoNewPrivileges=true
|
||||
PrivateTmp=true
|
||||
ProtectHome=true
|
||||
ProtectSystem=strict
|
||||
ProtectKernelTunables=true
|
||||
ProtectKernelModules=true
|
||||
ProtectControlGroups=true
|
||||
RestrictAddressFamilies=AF_UNIX
|
||||
RestrictRealtime=true
|
||||
LockPersonality=true
|
||||
@@ -0,0 +1,10 @@
|
||||
[Unit]
|
||||
Description=Remind the administrator to run the manual Atlas offline USB backup
|
||||
|
||||
[Timer]
|
||||
OnCalendar={{ atlas_usb_reminder_calendar }}
|
||||
Persistent=true
|
||||
Unit=atlas-usb-reminder.service
|
||||
|
||||
[Install]
|
||||
WantedBy=timers.target
|
||||
@@ -0,0 +1,51 @@
|
||||
#!/usr/bin/env bash
|
||||
set -Eeuo pipefail
|
||||
|
||||
export PATH=/usr/sbin:/usr/bin:/sbin:/bin
|
||||
|
||||
readonly pool={{ atlas_zfs_pool | quote }}
|
||||
readonly mount_root={{ atlas_mount_root | quote }}
|
||||
readonly snapshot_prefix={{ atlas_usb_backup_snapshot_prefix | quote }}
|
||||
readonly marker=/run/atlas-usb-backup/snapshot-name
|
||||
|
||||
[[ -e "$marker" ]] || exit 0
|
||||
[[ -f "$marker" && ! -L "$marker" ]] || {
|
||||
printf 'Unsafe Atlas USB snapshot marker; leaving snapshots unchanged\n' >&2
|
||||
exit 2
|
||||
}
|
||||
IFS= read -r snapshot_name <"$marker"
|
||||
[[ "$snapshot_name" =~ ^${snapshot_prefix}-[0-9]{8}T[0-9]{6}Z-[0-9]+$ ]] || {
|
||||
printf 'Invalid Atlas USB snapshot marker; leaving snapshots unchanged\n' >&2
|
||||
exit 2
|
||||
}
|
||||
|
||||
exec 9>/run/lock/atlas-zfs-snapshot.lock
|
||||
flock 9
|
||||
if zfs list -H -t snapshot -o name "${pool}@${snapshot_name}" >/dev/null 2>&1; then
|
||||
# ZFS can leave its on-demand .zfs/snapshot mounts in the host namespace
|
||||
# even after the backup's private bind mounts and process have exited.
|
||||
snapshot_mounts=()
|
||||
snapshot_sources=()
|
||||
while IFS=$'\t' read -r dataset dataset_mountpoint; do
|
||||
[[ "$dataset_mountpoint" == "$mount_root" || "$dataset_mountpoint" == "$mount_root/"* ]] || continue
|
||||
snapshot_mounts+=("${dataset_mountpoint}/.zfs/snapshot/${snapshot_name}")
|
||||
snapshot_sources+=("${dataset}@${snapshot_name}")
|
||||
done < <(zfs list -H -o name,mountpoint -s name -r "$pool")
|
||||
|
||||
{% raw %}
|
||||
for ((index = ${#snapshot_mounts[@]} - 1; index >= 0; index--)); do
|
||||
{% endraw %}
|
||||
mounted_source="$(findmnt -rn -M "${snapshot_mounts[$index]}" -o SOURCE || true)"
|
||||
[[ -n "$mounted_source" ]] || continue
|
||||
[[ "$mounted_source" == "${snapshot_sources[$index]}" ]] || {
|
||||
printf 'Unexpected source on Atlas USB snapshot mount: %s\n' \
|
||||
"${snapshot_mounts[$index]}" >&2
|
||||
exit 2
|
||||
}
|
||||
umount "${snapshot_mounts[$index]}"
|
||||
done
|
||||
|
||||
zfs destroy -r "${pool}@${snapshot_name}"
|
||||
printf 'Removed recursive Atlas USB source snapshot %s@%s after backup exit\n' \
|
||||
"$pool" "$snapshot_name"
|
||||
fi
|
||||
@@ -30,3 +30,7 @@ backend_phase1_timezone: Europe/Rome
|
||||
backend_phase1_services:
|
||||
- atlas-navidrome.service
|
||||
- atlas-syncthing.service
|
||||
backend_phase1_music_sync_enabled: false
|
||||
backend_phase1_music_source_dir: "{{ backend_phase1_archive_dir }}/Music"
|
||||
backend_phase1_music_sync_calendar: "*-*-* 00:45:00 Europe/Rome"
|
||||
backend_phase1_user_systemd_dir: "{{ backend_phase1_user_home }}/.config/systemd/user"
|
||||
|
||||
@@ -18,21 +18,28 @@
|
||||
- backend_phase1_app_data_root.startswith('/')
|
||||
- backend_phase1_navidrome_data_dir.startswith(backend_phase1_app_data_root + '/')
|
||||
- backend_phase1_syncthing_root.startswith(backend_phase1_app_data_root + '/')
|
||||
- >-
|
||||
not (backend_phase1_music_sync_enabled | bool) or
|
||||
(backend_phase1_music_source_dir.startswith(backend_phase1_archive_dir + '/')
|
||||
and backend_phase1_music_sync_calendar | length > 0)
|
||||
fail_msg: >-
|
||||
Disable the rootful media-stack gate and provide the Atlas LAN bind
|
||||
address, firewall sources, and absolute ZFS-backed paths before
|
||||
enabling phase one. This role does not manage Prometheus or migrate
|
||||
application data.
|
||||
tags: [music_sync]
|
||||
|
||||
- name: Read the rootless service account
|
||||
ansible.builtin.getent:
|
||||
database: passwd
|
||||
key: "{{ backend_phase1_username }}"
|
||||
tags: [music_sync]
|
||||
|
||||
- name: Record rootless service account IDs
|
||||
ansible.builtin.set_fact:
|
||||
backend_phase1_uid: "{{ ansible_facts['getent_passwd'][backend_phase1_username][1] }}"
|
||||
backend_phase1_gid: "{{ ansible_facts['getent_passwd'][backend_phase1_username][2] }}"
|
||||
tags: [music_sync]
|
||||
|
||||
- name: Read system service state before starting rootless Syncthing
|
||||
ansible.builtin.service_facts:
|
||||
@@ -65,6 +72,7 @@
|
||||
loop_control:
|
||||
label: "{{ item.dataset }}"
|
||||
register: backend_phase1_zfs_facts
|
||||
tags: [music_sync]
|
||||
|
||||
- name: Require mounted datasets at the declared paths
|
||||
ansible.builtin.assert:
|
||||
@@ -79,6 +87,23 @@
|
||||
loop: "{{ backend_phase1_zfs_facts.results }}"
|
||||
loop_control:
|
||||
label: "{{ item.item.dataset }}"
|
||||
tags: [music_sync]
|
||||
|
||||
- name: Inspect the music copy source
|
||||
ansible.builtin.stat:
|
||||
path: "{{ backend_phase1_music_source_dir }}"
|
||||
register: backend_phase1_music_source_stat
|
||||
when: backend_phase1_music_sync_enabled | bool
|
||||
tags: [music_sync]
|
||||
|
||||
- name: Require an existing music source directory
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- backend_phase1_music_source_stat.stat.isdir | default(false)
|
||||
fail_msg: >-
|
||||
{{ backend_phase1_music_source_dir }} must exist before enabling the daily music copy.
|
||||
when: backend_phase1_music_sync_enabled | bool
|
||||
tags: [music_sync]
|
||||
|
||||
- name: Enable lingering for the rootless service account
|
||||
ansible.builtin.command:
|
||||
@@ -87,12 +112,14 @@
|
||||
- enable-linger
|
||||
- "{{ backend_phase1_username }}"
|
||||
creates: "/var/lib/systemd/linger/{{ backend_phase1_username }}"
|
||||
tags: [music_sync]
|
||||
|
||||
- name: Start the rootless user systemd manager
|
||||
ansible.builtin.systemd:
|
||||
name: "user@{{ backend_phase1_uid }}.service"
|
||||
state: started
|
||||
when: not ansible_check_mode
|
||||
tags: [music_sync]
|
||||
|
||||
- name: Create rootless Quadlet and application directories
|
||||
ansible.builtin.file:
|
||||
@@ -113,6 +140,43 @@
|
||||
loop_control:
|
||||
label: "{{ item.path }}"
|
||||
|
||||
- name: Install rsync for the daily music copy
|
||||
ansible.builtin.dnf:
|
||||
name: rsync
|
||||
state: present
|
||||
when: backend_phase1_music_sync_enabled | bool
|
||||
tags: [music_sync]
|
||||
|
||||
- name: Create the rootless user systemd directory
|
||||
ansible.builtin.file:
|
||||
path: "{{ backend_phase1_user_systemd_dir }}"
|
||||
state: directory
|
||||
owner: "{{ backend_phase1_username }}"
|
||||
group: "{{ backend_phase1_user_group }}"
|
||||
mode: "0700"
|
||||
when: backend_phase1_music_sync_enabled | bool
|
||||
tags: [music_sync]
|
||||
|
||||
- name: Install the daily music copy service
|
||||
ansible.builtin.template:
|
||||
src: atlas-music-sync.service.j2
|
||||
dest: "{{ backend_phase1_user_systemd_dir }}/atlas-music-sync.service"
|
||||
owner: "{{ backend_phase1_username }}"
|
||||
group: "{{ backend_phase1_user_group }}"
|
||||
mode: "0644"
|
||||
when: backend_phase1_music_sync_enabled | bool
|
||||
tags: [music_sync]
|
||||
|
||||
- name: Install the daily music copy timer
|
||||
ansible.builtin.template:
|
||||
src: atlas-music-sync.timer.j2
|
||||
dest: "{{ backend_phase1_user_systemd_dir }}/atlas-music-sync.timer"
|
||||
owner: "{{ backend_phase1_username }}"
|
||||
group: "{{ backend_phase1_user_group }}"
|
||||
mode: "0644"
|
||||
when: backend_phase1_music_sync_enabled | bool
|
||||
tags: [music_sync]
|
||||
|
||||
- name: Render the rootless Navidrome Quadlet
|
||||
ansible.builtin.template:
|
||||
src: atlas-navidrome.container.j2
|
||||
@@ -140,6 +204,7 @@
|
||||
XDG_RUNTIME_DIR: "/run/user/{{ backend_phase1_uid }}"
|
||||
DBUS_SESSION_BUS_ADDRESS: "unix:path=/run/user/{{ backend_phase1_uid }}/bus"
|
||||
when: not ansible_check_mode
|
||||
tags: [music_sync]
|
||||
|
||||
- name: Permit NPM access to phase-one web interfaces through Aegis
|
||||
ansible.posix.firewalld:
|
||||
@@ -186,3 +251,19 @@
|
||||
when:
|
||||
- backend_phase1_start_services | bool
|
||||
- not ansible_check_mode
|
||||
|
||||
- name: Enable the daily music copy timer
|
||||
become_user: "{{ backend_phase1_username }}"
|
||||
ansible.builtin.systemd:
|
||||
name: atlas-music-sync.timer
|
||||
scope: user
|
||||
state: started
|
||||
enabled: true
|
||||
daemon_reload: true
|
||||
environment:
|
||||
XDG_RUNTIME_DIR: "/run/user/{{ backend_phase1_uid }}"
|
||||
DBUS_SESSION_BUS_ADDRESS: "unix:path=/run/user/{{ backend_phase1_uid }}/bus"
|
||||
when:
|
||||
- backend_phase1_music_sync_enabled | bool
|
||||
- not ansible_check_mode
|
||||
tags: [music_sync]
|
||||
|
||||
@@ -0,0 +1,10 @@
|
||||
# Managed by Ansible. Do not edit manually.
|
||||
[Unit]
|
||||
Description=Copy Atlas Archive music to the Navidrome library
|
||||
|
||||
[Service]
|
||||
Type=oneshot
|
||||
ExecStartPre=/usr/bin/mountpoint -q {{ backend_phase1_archive_dir }}
|
||||
ExecStartPre=/usr/bin/mountpoint -q {{ backend_phase1_music_dir }}
|
||||
ExecStartPre=/usr/bin/test -d {{ backend_phase1_music_source_dir }}
|
||||
ExecStart=/usr/bin/rsync -aH --no-perms --no-owner --no-group --delay-updates --stats -- {{ backend_phase1_music_source_dir }}/ {{ backend_phase1_music_dir }}/
|
||||
@@ -0,0 +1,11 @@
|
||||
# Managed by Ansible. Do not edit manually.
|
||||
[Unit]
|
||||
Description=Schedule the daily Atlas Navidrome music copy
|
||||
|
||||
[Timer]
|
||||
OnCalendar={{ backend_phase1_music_sync_calendar }}
|
||||
Persistent=true
|
||||
Unit=atlas-music-sync.service
|
||||
|
||||
[Install]
|
||||
WantedBy=timers.target
|
||||
103
ansible/roles/profile_server/tasks/backup_export_identity.yml
Normal file
103
ansible/roles/profile_server/tasks/backup_export_identity.yml
Normal file
@@ -0,0 +1,103 @@
|
||||
---
|
||||
- name: Validate Prometheus backup export identity inputs
|
||||
tags: [services, backup, prometheus_backup]
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- inventory_hostname == 'prometheus'
|
||||
- server_backup_username is match('^[a-z_][a-z0-9_-]*$')
|
||||
- server_backup_username not in ['root', server_username]
|
||||
- server_backup_export_root.startswith('/var/lib/')
|
||||
- server_backup_public_key_name is match('^[a-z0-9_-]+$')
|
||||
- hostvars['atlas'].atlas_manage_prometheus_backup_pull | default(false) | bool
|
||||
fail_msg: Enable Atlas and Prometheus backup roles together with dedicated identity settings.
|
||||
when: server_backup_export_enabled | bool
|
||||
|
||||
- name: Create dedicated Prometheus backup export group
|
||||
tags: [services, backup, prometheus_backup]
|
||||
ansible.builtin.group:
|
||||
name: "{{ server_backup_username }}"
|
||||
system: true
|
||||
state: present
|
||||
when: server_backup_export_enabled | bool
|
||||
|
||||
- name: Create locked Prometheus backup export account
|
||||
tags: [services, backup, prometheus_backup]
|
||||
ansible.builtin.user:
|
||||
name: "{{ server_backup_username }}"
|
||||
group: "{{ server_backup_username }}"
|
||||
groups: []
|
||||
append: false
|
||||
comment: Read-only prepared backup export for Atlas
|
||||
home: "{{ server_backup_export_root }}"
|
||||
create_home: false
|
||||
shell: /bin/bash
|
||||
password_lock: true
|
||||
system: true
|
||||
state: present
|
||||
when: server_backup_export_enabled | bool
|
||||
|
||||
- name: Require restricted rrsync helper on Prometheus
|
||||
tags: [services, backup, prometheus_backup]
|
||||
ansible.builtin.stat:
|
||||
path: "{{ server_backup_rrsync_path }}"
|
||||
register: server_backup_rrsync_file
|
||||
when: server_backup_export_enabled | bool
|
||||
|
||||
- name: Validate restricted rrsync helper
|
||||
tags: [services, backup, prometheus_backup]
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- server_backup_rrsync_file.stat.exists
|
||||
- server_backup_rrsync_file.stat.isreg
|
||||
- server_backup_rrsync_file.stat.pw_name == 'root'
|
||||
fail_msg: Rocky rsync must provide the root-owned rrsync support script.
|
||||
when: server_backup_export_enabled | bool
|
||||
|
||||
- name: Create prepared backup export root
|
||||
tags: [services, backup, prometheus_backup]
|
||||
ansible.builtin.file:
|
||||
path: "{{ server_backup_export_root }}"
|
||||
state: directory
|
||||
owner: root
|
||||
group: "{{ server_backup_username }}"
|
||||
mode: "0750"
|
||||
when: server_backup_export_enabled | bool
|
||||
|
||||
- name: Create restricted Prometheus backup SSH directories
|
||||
tags: [services, backup, prometheus_backup]
|
||||
ansible.builtin.file:
|
||||
path: "{{ item }}"
|
||||
state: directory
|
||||
owner: root
|
||||
group: "{{ server_backup_username }}"
|
||||
mode: "0750"
|
||||
loop:
|
||||
- "{{ server_backup_export_root }}/.ssh"
|
||||
- "{{ server_backup_export_root }}/.ssh/authorized_keys.d"
|
||||
when: server_backup_export_enabled | bool
|
||||
|
||||
- name: Read Atlas public key for Prometheus backup pull
|
||||
tags: [services, backup, prometheus_backup]
|
||||
ansible.builtin.slurp:
|
||||
src: "{{ hostvars['atlas'].atlas_prometheus_pull_private_key_path | default('/etc/atlas-prometheus-pull/id_ed25519') }}.pub"
|
||||
delegate_to: atlas
|
||||
become: true
|
||||
register: server_backup_atlas_public_key
|
||||
when:
|
||||
- server_backup_export_enabled | bool
|
||||
- not ansible_check_mode
|
||||
|
||||
- name: Authorize only restricted read-only backup access from Atlas
|
||||
tags: [services, backup, prometheus_backup]
|
||||
ansible.builtin.copy:
|
||||
content: >-
|
||||
{{ 'command="/usr/bin/python3 ' ~ server_backup_rrsync_path ~ ' -ro '
|
||||
~ server_backup_export_root ~ '/versions",restrict '
|
||||
~ (server_backup_atlas_public_key.content | b64decode | trim) ~ '\n' }}
|
||||
dest: "{{ server_backup_export_root }}/.ssh/authorized_keys.d/{{ server_backup_public_key_name }}"
|
||||
owner: root
|
||||
group: "{{ server_backup_username }}"
|
||||
mode: "0640"
|
||||
when:
|
||||
- server_backup_export_enabled | bool
|
||||
- not ansible_check_mode
|
||||
91
ansible/roles/profile_server/tasks/backup_export_job.yml
Normal file
91
ansible/roles/profile_server/tasks/backup_export_job.yml
Normal file
@@ -0,0 +1,91 @@
|
||||
---
|
||||
- name: Validate Prometheus backup export job inputs
|
||||
tags: [services, backup, prometheus_backup]
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- server_backup_export_source_keep | int >= 2
|
||||
- server_backup_export_paths | length > 0
|
||||
- server_backup_export_paths | unique | length == server_backup_export_paths | length
|
||||
- >-
|
||||
server_backup_export_paths
|
||||
| select('match', '^[a-zA-Z0-9][a-zA-Z0-9._/-]*$') | list | length
|
||||
== server_backup_export_paths | length
|
||||
- >-
|
||||
server_backup_export_paths
|
||||
| reject('search', '(^|/)\.\.(/|$)') | list | length
|
||||
== server_backup_export_paths | length
|
||||
- >-
|
||||
server_backup_export_excludes
|
||||
| select('match', '^[a-zA-Z0-9][a-zA-Z0-9._/-]*$') | list | length
|
||||
== server_backup_export_excludes | length
|
||||
- >-
|
||||
server_backup_export_excludes
|
||||
| reject('search', '(^|/)\.\.(/|$)') | list | length
|
||||
== server_backup_export_excludes | length
|
||||
fail_msg: Define safe relative paths and at least two prepared export versions.
|
||||
when: server_backup_export_enabled | bool
|
||||
|
||||
- name: Validate Prometheus backup export calendar
|
||||
tags: [services, backup, prometheus_backup]
|
||||
ansible.builtin.command:
|
||||
argv: [systemd-analyze, calendar, "{{ server_backup_export_calendar }}"]
|
||||
changed_when: false
|
||||
check_mode: false
|
||||
when: server_backup_export_enabled | bool
|
||||
|
||||
- name: Ensure prepared Prometheus backup versions directory exists
|
||||
tags: [services, backup, prometheus_backup]
|
||||
ansible.builtin.file:
|
||||
path: "{{ server_backup_export_root }}/versions"
|
||||
state: directory
|
||||
owner: root
|
||||
group: "{{ server_backup_username }}"
|
||||
mode: "0750"
|
||||
when: server_backup_export_enabled | bool
|
||||
|
||||
- name: Install Prometheus backup export helper
|
||||
tags: [services, backup, prometheus_backup, gitea_cutover, npm_quadlet_backup]
|
||||
ansible.builtin.template:
|
||||
src: prometheus-backup-export.sh.j2
|
||||
dest: /usr/local/sbin/prometheus-backup-export
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0750"
|
||||
validate: "bash -n %s"
|
||||
when: server_backup_export_enabled | bool
|
||||
|
||||
- name: Install Prometheus backup export systemd units
|
||||
tags: [services, backup, prometheus_backup]
|
||||
ansible.builtin.template:
|
||||
src: "{{ item }}.j2"
|
||||
dest: "/etc/systemd/system/{{ item }}"
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0644"
|
||||
loop:
|
||||
- prometheus-backup-export.service
|
||||
- prometheus-backup-export.timer
|
||||
loop_control:
|
||||
label: "{{ item }}"
|
||||
register: server_backup_export_units
|
||||
when: server_backup_export_enabled | bool
|
||||
|
||||
- name: Reload systemd after Prometheus backup export unit changes
|
||||
tags: [services, backup, prometheus_backup]
|
||||
ansible.builtin.systemd:
|
||||
daemon_reload: true
|
||||
when:
|
||||
- server_backup_export_enabled | bool
|
||||
- server_backup_export_units is changed
|
||||
- not ansible_check_mode
|
||||
|
||||
- name: Enable Prometheus backup export timer only after explicit activation
|
||||
tags: [services, backup, prometheus_backup]
|
||||
ansible.builtin.systemd:
|
||||
name: prometheus-backup-export.timer
|
||||
enabled: true
|
||||
state: started
|
||||
when:
|
||||
- server_backup_export_enabled | bool
|
||||
- server_backup_export_start_timer | bool
|
||||
- not ansible_check_mode
|
||||
@@ -1,33 +0,0 @@
|
||||
---
|
||||
- name: Require DuckDNS domain and Vault token before deployment
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- >-
|
||||
server_duckdns_domain | default('') is
|
||||
regex('[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?', match_type='fullmatch')
|
||||
- >-
|
||||
vault_duckdns_token | default('') is
|
||||
regex('[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}', match_type='fullmatch')
|
||||
fail_msg: >-
|
||||
Define server_duckdns_domain in host_vars and the rotated vault_duckdns_token
|
||||
in encrypted Vault or an untracked local vars file before deploying DuckDNS.
|
||||
no_log: true
|
||||
|
||||
- name: Ensure private DuckDNS directory exists
|
||||
ansible.builtin.file:
|
||||
path: "{{ server_user_home }}/duckdns"
|
||||
state: directory
|
||||
owner: "{{ server_username }}"
|
||||
group: "{{ server_user_group }}"
|
||||
mode: "0700"
|
||||
|
||||
- name: Render DuckDNS updater with the Vault token
|
||||
ansible.builtin.template:
|
||||
src: duck.sh.j2
|
||||
dest: "{{ server_user_home }}/duckdns/duck.sh"
|
||||
owner: "{{ server_username }}"
|
||||
group: "{{ server_user_group }}"
|
||||
mode: "0700"
|
||||
validate: /bin/sh -n %s
|
||||
no_log: true
|
||||
diff: false
|
||||
59
ansible/roles/profile_server/tasks/gitea_npm_proxy.yml
Normal file
59
ansible/roles/profile_server/tasks/gitea_npm_proxy.yml
Normal file
@@ -0,0 +1,59 @@
|
||||
---
|
||||
- name: Validate the NPM Gitea cutover override
|
||||
tags: [services, gitea_cutover]
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- server_gitea_proxy_enabled | bool
|
||||
- server_gitea_npm_domains | length > 0
|
||||
- server_gitea_npm_domains | select('match', '^[a-zA-Z0-9.-]+$') | list | length == server_gitea_npm_domains | length
|
||||
fail_msg: Declare the exact NPM Gitea hostnames before enabling the Atlas upstream.
|
||||
when: server_gitea_on_atlas | bool
|
||||
|
||||
- name: Ensure the NPM custom configuration directory exists
|
||||
tags: [services, gitea_cutover]
|
||||
ansible.builtin.file:
|
||||
path: /opt/npm/data/nginx/custom
|
||||
state: directory
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0755"
|
||||
when: server_gitea_proxy_enabled | bool
|
||||
|
||||
- name: Render the Gitea-only NPM runtime upstream override
|
||||
tags: [services, gitea_cutover]
|
||||
ansible.builtin.template:
|
||||
src: prometheus-gitea-npm-proxy.conf.j2
|
||||
dest: /opt/npm/data/nginx/custom/server_proxy.conf
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0644"
|
||||
register: server_gitea_npm_override
|
||||
when: server_gitea_on_atlas | bool
|
||||
|
||||
- name: Remove the Gitea NPM override when source routing is selected
|
||||
tags: [services, gitea_cutover]
|
||||
ansible.builtin.file:
|
||||
path: /opt/npm/data/nginx/custom/server_proxy.conf
|
||||
state: absent
|
||||
when:
|
||||
- server_gitea_proxy_enabled | bool
|
||||
- not server_gitea_on_atlas | bool
|
||||
|
||||
- name: Validate NPM configuration after a Gitea upstream change
|
||||
tags: [services, gitea_cutover]
|
||||
ansible.builtin.command:
|
||||
argv: [podman, exec, nginx-proxy-manager, nginx, -t]
|
||||
changed_when: false
|
||||
when:
|
||||
- server_gitea_on_atlas | bool
|
||||
- server_gitea_npm_override is changed
|
||||
- not ansible_check_mode
|
||||
|
||||
- name: Reload NPM after validating the Gitea upstream change
|
||||
tags: [services, gitea_cutover]
|
||||
ansible.builtin.command:
|
||||
argv: [podman, exec, nginx-proxy-manager, nginx, -s, reload]
|
||||
when:
|
||||
- server_gitea_on_atlas | bool
|
||||
- server_gitea_npm_override is changed
|
||||
- not ansible_check_mode
|
||||
61
ansible/roles/profile_server/tasks/gitea_ssh_proxy.yml
Normal file
61
ansible/roles/profile_server/tasks/gitea_ssh_proxy.yml
Normal file
@@ -0,0 +1,61 @@
|
||||
---
|
||||
- name: Validate the Prometheus Gitea SSH cutover inputs
|
||||
tags: [services, gitea_cutover]
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- server_gitea_proxy_enabled | bool
|
||||
- server_gitea_atlas_address is match('^[0-9]{1,3}(\.[0-9]{1,3}){3}$')
|
||||
- server_gitea_ssh_public_port | int > 1024
|
||||
- server_gitea_ssh_public_port | int < 65536
|
||||
- server_gitea_ssh_target_port | int > 1024
|
||||
- server_gitea_ssh_target_port | int < 65536
|
||||
- server_gitea_ssh_public_port | int != 22
|
||||
fail_msg: Keep administrative SSH on 22 and provide the Atlas rootless Gitea SSH endpoint.
|
||||
when: server_gitea_on_atlas | bool
|
||||
|
||||
- name: Install the Gitea SSH socket proxy units without activating them
|
||||
tags: [services, gitea_cutover]
|
||||
ansible.builtin.template:
|
||||
src: "{{ item }}.j2"
|
||||
dest: "/etc/systemd/system/{{ item }}"
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0644"
|
||||
loop:
|
||||
- prometheus-gitea-ssh-proxy.socket
|
||||
- prometheus-gitea-ssh-proxy.service
|
||||
loop_control:
|
||||
label: "{{ item }}"
|
||||
register: server_gitea_ssh_proxy_units
|
||||
when: server_gitea_proxy_enabled | bool
|
||||
|
||||
- name: Reload systemd after Gitea SSH proxy unit changes
|
||||
tags: [services, gitea_cutover]
|
||||
ansible.builtin.systemd:
|
||||
daemon_reload: true
|
||||
when:
|
||||
- server_gitea_proxy_enabled | bool
|
||||
- server_gitea_ssh_proxy_units is changed
|
||||
- not ansible_check_mode
|
||||
|
||||
- name: Manage the public Gitea SSH socket separately from administrative SSH
|
||||
tags: [services, gitea_cutover]
|
||||
ansible.builtin.systemd:
|
||||
name: prometheus-gitea-ssh-proxy.socket
|
||||
state: "{{ 'started' if server_gitea_on_atlas | bool else 'stopped' }}"
|
||||
enabled: "{{ server_gitea_on_atlas | bool }}"
|
||||
when:
|
||||
- server_gitea_proxy_enabled | bool
|
||||
- not ansible_check_mode
|
||||
|
||||
- name: Open only the public Gitea SSH port after cutover
|
||||
tags: [services, gitea_cutover]
|
||||
ansible.posix.firewalld:
|
||||
port: "{{ server_gitea_ssh_public_port }}/tcp"
|
||||
zone: "{{ server_firewalld_zone }}"
|
||||
state: "{{ 'enabled' if server_gitea_on_atlas | bool else 'disabled' }}"
|
||||
permanent: true
|
||||
immediate: true
|
||||
when:
|
||||
- server_gitea_proxy_enabled | bool
|
||||
- server_firewall_backend == 'firewalld'
|
||||
112
ansible/roles/profile_server/tasks/legacy_cleanup.yml
Normal file
112
ansible/roles/profile_server/tasks/legacy_cleanup.yml
Normal file
@@ -0,0 +1,112 @@
|
||||
---
|
||||
- name: Require explicit retirement of the migrated Prometheus source
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- inventory_hostname == 'prometheus'
|
||||
- server_legacy_stack_retired | bool
|
||||
- server_gitea_on_atlas | bool
|
||||
- server_npm_quadlet_cutover | bool
|
||||
- server_backup_export_enabled | bool
|
||||
|
||||
- name: Verify legacy paths have no mounts or container users
|
||||
ansible.builtin.command:
|
||||
argv:
|
||||
- python3
|
||||
- -c
|
||||
- |
|
||||
import json, os, pathlib, subprocess
|
||||
def run(*args):
|
||||
return subprocess.check_output(args, text=True).strip()
|
||||
paths = ['/opt/gitea', '/home/git/.ssh', '/opt/navidrome',
|
||||
'/opt/postgres', '/opt/music', '/opt/containerd', '/opt/docker']
|
||||
mounts = json.loads(run('findmnt', '--json', '--list', '-o', 'TARGET'))['filesystems']
|
||||
for path in paths:
|
||||
assert os.path.realpath(path) == path, 'Symlink in cleanup path: ' + path
|
||||
for mount in mounts:
|
||||
target = mount['target']
|
||||
assert target != path and not target.startswith(path + '/'), 'Mounted cleanup path: ' + path
|
||||
ids = run('podman', 'ps', '-aq').split()
|
||||
containers = json.loads(run('podman', 'inspect', *ids)) if ids else []
|
||||
for container in containers:
|
||||
assert container['Name'].lstrip('/') == 'nginx-proxy-manager', 'Unexpected container; review before cleanup'
|
||||
for mount in container.get('Mounts', []):
|
||||
source = os.path.realpath(mount['Source'])
|
||||
for path in paths:
|
||||
assert source != path and not source.startswith(path + '/'), 'Container uses cleanup path: ' + path
|
||||
for path in ['/opt/music', '/opt/containerd']:
|
||||
if os.path.isdir(path):
|
||||
for entry in pathlib.Path(path).rglob('*'):
|
||||
assert entry.is_dir() and not entry.is_symlink(), 'Unexpected file in empty legacy path: ' + str(entry)
|
||||
if os.path.isdir('/opt/docker'):
|
||||
allowed = {'/opt/docker/server', '/opt/docker/server/docker-compose.yml'}
|
||||
for entry in pathlib.Path('/opt/docker').rglob('*'):
|
||||
assert str(entry) in allowed and not entry.is_symlink(), 'Unexpected legacy Docker content: ' + str(entry)
|
||||
assert run('systemctl', 'is-active', 'prometheus-npm.service') == 'active'
|
||||
assert subprocess.run(['systemctl', 'is-active', '--quiet', 'podman-compose-server.service']).returncode != 0
|
||||
assert subprocess.run(['systemctl', 'is-active', '--quiet', 'prometheus-backup-export.service']).returncode != 0
|
||||
print('Legacy cleanup preflight passed')
|
||||
changed_when: false
|
||||
check_mode: false
|
||||
|
||||
- name: Require the updated backup configuration before deleting fallback files
|
||||
ansible.builtin.command:
|
||||
argv:
|
||||
- python3
|
||||
- -c
|
||||
- |
|
||||
import pathlib, subprocess
|
||||
unit = subprocess.check_output(['systemctl', 'show', 'prometheus-backup-export.service',
|
||||
'-p', 'RequiresMountsFor', '--value'], text=True)
|
||||
assert '/opt/gitea' not in unit, 'Backup unit still depends on legacy Gitea'
|
||||
helper = pathlib.Path('/usr/local/sbin/prometheus-backup-export').read_text()
|
||||
assert 'podman-compose-server' not in helper and 'opt/docker/server' not in helper
|
||||
subprocess.run(['bash', '-n', '/usr/local/sbin/prometheus-backup-export'], check=True)
|
||||
changed_when: false
|
||||
when: not ansible_check_mode
|
||||
|
||||
- name: Delete only the explicitly approved legacy data and fallback files
|
||||
ansible.builtin.file:
|
||||
path: "{{ item }}"
|
||||
state: absent
|
||||
loop:
|
||||
- /opt/gitea
|
||||
- /home/git/.ssh
|
||||
- /opt/navidrome
|
||||
- /opt/postgres
|
||||
- /opt/music
|
||||
- /opt/containerd
|
||||
- /opt/docker
|
||||
- /usr/local/sbin/prometheus-gitea-final-export
|
||||
- /etc/systemd/system/podman-compose-server.service
|
||||
register: server_legacy_deleted
|
||||
diff: false
|
||||
|
||||
- name: Reload systemd after removing the inactive legacy unit
|
||||
ansible.builtin.systemd:
|
||||
daemon_reload: true
|
||||
when:
|
||||
- server_legacy_deleted is changed
|
||||
- not ansible_check_mode
|
||||
|
||||
- name: Inspect the obsolete Git home without following symlinks
|
||||
ansible.builtin.stat:
|
||||
path: /home/git
|
||||
follow: false
|
||||
register: server_legacy_git_home
|
||||
|
||||
- name: Require the obsolete Git account to be absent before removing its empty home
|
||||
ansible.builtin.command:
|
||||
argv: [getent, passwd, git]
|
||||
register: server_legacy_git_account
|
||||
changed_when: false
|
||||
failed_when: server_legacy_git_account.rc != 2
|
||||
check_mode: false
|
||||
when: server_legacy_git_home.stat.exists
|
||||
|
||||
# rmdir refuses any nonempty directory; never recursively delete this parent.
|
||||
- name: Remove only the empty obsolete Git home
|
||||
ansible.builtin.command:
|
||||
argv: [rmdir, /home/git]
|
||||
register: server_legacy_git_home_removed
|
||||
changed_when: server_legacy_git_home_removed.rc == 0
|
||||
when: server_legacy_git_home.stat.exists
|
||||
33
ansible/roles/profile_server/tasks/legacy_image_cleanup.yml
Normal file
33
ansible/roles/profile_server/tasks/legacy_image_cleanup.yml
Normal file
@@ -0,0 +1,33 @@
|
||||
---
|
||||
- name: Require the migrated Prometheus topology for image cleanup
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- inventory_hostname == 'prometheus'
|
||||
- server_gitea_on_atlas | bool
|
||||
- server_npm_quadlet_cutover | bool
|
||||
- server_legacy_images | default([]) | length > 0
|
||||
- >-
|
||||
server_legacy_images | difference([
|
||||
'docker.gitea.com/gitea:1.25.2',
|
||||
'docker.io/deluan/navidrome:latest',
|
||||
'docker.io/library/postgres:13']) | length == 0
|
||||
|
||||
- name: Check whether the explicitly selected legacy images exist
|
||||
ansible.builtin.command:
|
||||
argv: [podman, image, exists, "{{ item }}"]
|
||||
loop: "{{ server_legacy_images }}"
|
||||
register: server_legacy_image_presence
|
||||
changed_when: false
|
||||
failed_when: server_legacy_image_presence.rc not in [0, 1]
|
||||
check_mode: false
|
||||
|
||||
# No --force: Podman must refuse images referenced by any existing container.
|
||||
- name: Remove only unused explicitly selected legacy images
|
||||
ansible.builtin.command:
|
||||
argv: [podman, image, rm, "{{ item.item }}"]
|
||||
loop: "{{ server_legacy_image_presence.results }}"
|
||||
loop_control:
|
||||
label: "{{ item.item }}"
|
||||
when: item.rc == 0
|
||||
register: server_legacy_image_removal
|
||||
changed_when: server_legacy_image_removal.rc == 0
|
||||
@@ -8,10 +8,6 @@
|
||||
fail_msg: >-
|
||||
server_firewall_backend must be firewalld for the Rocky server profile.
|
||||
|
||||
- name: Configure DuckDNS updater
|
||||
tags: [dotfiles, dotfiles:server, duckdns]
|
||||
ansible.builtin.import_tasks: duckdns.yml
|
||||
|
||||
- name: Ensure server directories exist
|
||||
tags: [dotfiles, services]
|
||||
ansible.builtin.file:
|
||||
@@ -23,6 +19,9 @@
|
||||
loop: "{{ server_directories | default([]) }}"
|
||||
loop_control:
|
||||
label: "{{ item.path }}"
|
||||
when:
|
||||
- item.path != '/opt/gitea/data' or not server_gitea_on_atlas | bool
|
||||
- item.path != server_container_stack_dir or not server_legacy_stack_retired | bool
|
||||
|
||||
- name: Copy server dotfiles
|
||||
tags: [dotfiles, dotfiles:server]
|
||||
@@ -37,7 +36,7 @@
|
||||
label: "{{ item.dest }}"
|
||||
|
||||
- name: Render server templates
|
||||
tags: [dotfiles, dotfiles:server]
|
||||
tags: [dotfiles, dotfiles:server, gitea_cutover]
|
||||
ansible.builtin.template:
|
||||
src: "{{ item.src }}"
|
||||
dest: "{{ item.dest if item.dest.startswith('/') else server_user_home ~ '/' ~ item.dest }}"
|
||||
@@ -48,10 +47,38 @@
|
||||
loop_control:
|
||||
label: "{{ item.dest }}"
|
||||
no_log: "{{ item.no_log | default(false) }}"
|
||||
when: item.src != 'server/docker-compose.yml.j2' or not server_legacy_stack_retired | bool
|
||||
|
||||
- name: Manage Podman Compose stack
|
||||
tags: [services, podman]
|
||||
ansible.builtin.include_tasks: podman-compose.yml
|
||||
when: not server_legacy_stack_retired | bool
|
||||
|
||||
- name: Import staged NPM Quadlet tasks
|
||||
ansible.builtin.import_tasks: npm_quadlet.yml
|
||||
|
||||
- name: Import explicit legacy server image cleanup
|
||||
ansible.builtin.import_tasks: legacy_image_cleanup.yml
|
||||
tags: [never, server_image_cleanup]
|
||||
when: server_legacy_image_cleanup | default(false) | bool
|
||||
|
||||
- name: Import Prometheus backup export identity tasks
|
||||
ansible.builtin.import_tasks: backup_export_identity.yml
|
||||
|
||||
- name: Import Prometheus backup export job tasks
|
||||
ansible.builtin.import_tasks: backup_export_job.yml
|
||||
tags: [server_legacy_cleanup]
|
||||
|
||||
- name: Import explicitly approved legacy server data cleanup
|
||||
ansible.builtin.import_tasks: legacy_cleanup.yml
|
||||
tags: [never, server_legacy_cleanup]
|
||||
when: server_legacy_cleanup | bool
|
||||
|
||||
- name: Import Prometheus Gitea SSH proxy tasks
|
||||
ansible.builtin.import_tasks: gitea_ssh_proxy.yml
|
||||
|
||||
- name: Import Prometheus Gitea NPM proxy override tasks
|
||||
ansible.builtin.import_tasks: gitea_npm_proxy.yml
|
||||
|
||||
- name: Ensure server SSH authorized key fragments directory exists
|
||||
tags: [services, ssh]
|
||||
@@ -77,13 +104,17 @@
|
||||
when: server_ssh_authorized_keys | length > 0
|
||||
|
||||
- name: Configure server SSH authorized key fragments
|
||||
tags: [services, ssh]
|
||||
tags: [services, ssh, prometheus_backup]
|
||||
ansible.builtin.lineinfile:
|
||||
path: /etc/ssh/sshd_config
|
||||
regexp: '^\s*AuthorizedKeysFile\s+'
|
||||
line: >-
|
||||
AuthorizedKeysFile {{ server_ssh_authorized_keys | map(attribute='name')
|
||||
| map('regex_replace', '^', '%h/.ssh/authorized_keys.d/') | join(' ') }}
|
||||
AuthorizedKeysFile {{
|
||||
((server_ssh_authorized_keys | map(attribute='name')
|
||||
| map('regex_replace', '^', '%h/.ssh/authorized_keys.d/') | list)
|
||||
+ (['%h/.ssh/authorized_keys.d/' ~ server_backup_public_key_name]
|
||||
if server_backup_export_enabled | bool else [])) | join(' ')
|
||||
}}
|
||||
state: present
|
||||
validate: "sshd -t -f %s"
|
||||
notify: Reload SSH service
|
||||
@@ -100,11 +131,13 @@
|
||||
notify: Reload SSH service
|
||||
|
||||
- name: Restrict SSH login to allowed users on server
|
||||
tags: [services]
|
||||
tags: [services, prometheus_backup]
|
||||
ansible.builtin.lineinfile:
|
||||
path: /etc/ssh/sshd_config
|
||||
regexp: '^\s*AllowUsers\s+'
|
||||
line: "AllowUsers {{ server_sshd_allow_users | join(' ') }}"
|
||||
line: >-
|
||||
AllowUsers {{ (server_sshd_allow_users
|
||||
+ ([server_backup_username] if server_backup_export_enabled | bool else [])) | join(' ') }}
|
||||
state: present
|
||||
validate: "sshd -t -f %s"
|
||||
notify: Reload SSH service
|
||||
|
||||
71
ansible/roles/profile_server/tasks/npm_quadlet.yml
Normal file
71
ansible/roles/profile_server/tasks/npm_quadlet.yml
Normal file
@@ -0,0 +1,71 @@
|
||||
---
|
||||
- name: Require staged NPM Quadlet for an active cutover
|
||||
tags: [services, npm_quadlet]
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- not server_npm_quadlet_cutover | bool or server_npm_quadlet_stage | bool
|
||||
fail_msg: The NPM Quadlet cutover requires the staged container and network.
|
||||
|
||||
- name: Validate staged NPM Quadlet inputs
|
||||
tags: [services, npm_quadlet]
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- server_npm_quadlet_image is defined
|
||||
- server_npm_quadlet_image is match('^docker\.io/jc21/nginx-proxy-manager@sha256:[a-f0-9]{64}$')
|
||||
- server_gitea_on_atlas | bool
|
||||
fail_msg: Stage the exact running NPM image only after Gitea has left Compose.
|
||||
when: server_npm_quadlet_stage | bool
|
||||
|
||||
- name: Ensure rootful Quadlet directory exists for NPM
|
||||
tags: [services, npm_quadlet]
|
||||
ansible.builtin.file:
|
||||
path: /etc/containers/systemd
|
||||
state: directory
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0755"
|
||||
when: server_npm_quadlet_stage | bool
|
||||
|
||||
- name: Render staged NPM container and network Quadlets
|
||||
tags: [services, npm_quadlet]
|
||||
ansible.builtin.template:
|
||||
src: "{{ item }}.j2"
|
||||
dest: "/etc/containers/systemd/{{ item }}"
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0644"
|
||||
loop:
|
||||
- prometheus-npm.container
|
||||
- server-web.network
|
||||
loop_control:
|
||||
label: "{{ item }}"
|
||||
register: server_npm_quadlet_units
|
||||
when: server_npm_quadlet_stage | bool
|
||||
|
||||
- name: Reload systemd after staging NPM Quadlets
|
||||
tags: [services, npm_quadlet]
|
||||
ansible.builtin.systemd:
|
||||
daemon_reload: true
|
||||
when:
|
||||
- server_npm_quadlet_stage | bool
|
||||
- server_npm_quadlet_units is changed
|
||||
- not ansible_check_mode
|
||||
|
||||
- name: Verify the staged NPM Quadlet was generated
|
||||
tags: [services, npm_quadlet]
|
||||
ansible.builtin.command:
|
||||
argv: [systemctl, show, prometheus-npm.service, --property=LoadState, --value]
|
||||
register: server_npm_quadlet_load_state
|
||||
changed_when: false
|
||||
when:
|
||||
- server_npm_quadlet_stage | bool
|
||||
- not ansible_check_mode
|
||||
|
||||
- name: Reject an invalid staged NPM Quadlet
|
||||
tags: [services, npm_quadlet]
|
||||
ansible.builtin.assert:
|
||||
that: server_npm_quadlet_load_state.stdout == 'loaded'
|
||||
fail_msg: Quadlet generator did not produce prometheus-npm.service.
|
||||
when:
|
||||
- server_npm_quadlet_stage | bool
|
||||
- not ansible_check_mode
|
||||
@@ -1,24 +0,0 @@
|
||||
#!/bin/sh
|
||||
# Managed by Ansible. Contains a Vault token; never copy this file into Git.
|
||||
set -eu
|
||||
umask 077
|
||||
|
||||
log_file={{ (server_user_home ~ '/duckdns/duck.log') | quote }}
|
||||
|
||||
# Keep the token out of process arguments and verify the HTTPS certificate.
|
||||
if ! response=$(curl --fail --silent --show-error --connect-timeout 10 --max-time 30 --config - <<'DUCKDNS_CONFIG'
|
||||
url = "https://www.duckdns.org/update?domains={{ server_duckdns_domain }}&token={{ vault_duckdns_token }}&ip="
|
||||
DUCKDNS_CONFIG
|
||||
); then
|
||||
printf 'ERROR\n' > "$log_file"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
case "$response" in
|
||||
OK) printf 'OK\n' > "$log_file" ;;
|
||||
*)
|
||||
printf 'KO\n' > "$log_file"
|
||||
printf 'DuckDNS update failed.\n' >&2
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
@@ -0,0 +1,15 @@
|
||||
[Unit]
|
||||
Description=Prepare a read-only Prometheus application backup for Atlas
|
||||
RequiresMountsFor=/opt/npm {% if not server_gitea_on_atlas | bool %}/opt/gitea {% endif %}{{ server_backup_export_root }}
|
||||
ConditionFileIsExecutable=/usr/local/sbin/prometheus-backup-export
|
||||
|
||||
[Service]
|
||||
Type=oneshot
|
||||
ExecStart=/usr/local/sbin/prometheus-backup-export
|
||||
User=root
|
||||
Group=root
|
||||
UMask=0077
|
||||
TimeoutStartSec=infinity
|
||||
Nice=10
|
||||
IOSchedulingClass=best-effort
|
||||
IOSchedulingPriority=7
|
||||
@@ -0,0 +1,118 @@
|
||||
#!/usr/bin/env bash
|
||||
set -Eeuo pipefail
|
||||
umask 077
|
||||
|
||||
export_root={{ server_backup_export_root | quote }}
|
||||
versions="$export_root/versions"
|
||||
{% if server_legacy_stack_retired | bool %}
|
||||
stack_unit=prometheus-npm.service
|
||||
{% else %}
|
||||
stack_unit=''
|
||||
compose_active=false
|
||||
quadlet_active=false
|
||||
systemctl is-active --quiet podman-compose-server.service && compose_active=true
|
||||
systemctl is-active --quiet prometheus-npm.service && quadlet_active=true
|
||||
if [[ "$compose_active" == "$quadlet_active" ]]; then
|
||||
echo 'Expected exactly one active NPM service (Compose or Quadlet)' >&2
|
||||
exit 1
|
||||
fi
|
||||
if "$quadlet_active"; then
|
||||
stack_unit=prometheus-npm.service
|
||||
else
|
||||
stack_unit=podman-compose-server.service
|
||||
fi
|
||||
{% endif %}
|
||||
stamp=$(date -u +%Y%m%dT%H%M%SZ)
|
||||
stage=''
|
||||
stack_stopped=false
|
||||
|
||||
exec 9>/run/lock/prometheus-backup-export.lock
|
||||
flock -n 9 || { echo 'A backup export is already running' >&2; exit 1; }
|
||||
|
||||
cleanup() {
|
||||
local rc=$?
|
||||
trap - EXIT
|
||||
if "$stack_stopped"; then
|
||||
if systemctl is-active --quiet "$stack_unit"; then
|
||||
systemctl restart "$stack_unit" || rc=1
|
||||
else
|
||||
systemctl start "$stack_unit" || rc=1
|
||||
fi
|
||||
fi
|
||||
if (( rc != 0 )) && [[ -n "$stage" && -d "$stage" ]]; then
|
||||
rm -rf -- "$stage"
|
||||
fi
|
||||
exit "$rc"
|
||||
}
|
||||
trap cleanup EXIT
|
||||
trap 'exit 129' HUP
|
||||
trap 'exit 130' INT
|
||||
trap 'exit 143' TERM
|
||||
|
||||
systemctl is-active --quiet "$stack_unit" || {
|
||||
echo "The managed NPM unit $stack_unit must be active before preparing a backup" >&2
|
||||
exit 1
|
||||
}
|
||||
|
||||
paths=(
|
||||
{% for path in server_backup_export_paths %}
|
||||
{{ path | quote }}
|
||||
{% endfor %}
|
||||
)
|
||||
excludes=(
|
||||
{% for path in server_backup_export_excludes %}
|
||||
--exclude={{ path | quote }}
|
||||
{% endfor %}
|
||||
)
|
||||
for path in "${paths[@]}"; do
|
||||
[[ -e "/$path" ]] || { echo "Required backup path missing: /$path" >&2; exit 1; }
|
||||
done
|
||||
[[ ! -e "$versions/$stamp" ]] || { echo "Export version already exists: $stamp" >&2; exit 1; }
|
||||
stage=$(mktemp -d "$export_root/.staging.XXXXXXXX")
|
||||
|
||||
# SQLite databases and their accompanying files are copied while both
|
||||
# managed containers are stopped. The EXIT trap restarts the stack on error.
|
||||
stack_stopped=true
|
||||
systemctl stop "$stack_unit"
|
||||
tar --acls --xattrs --selinux "${excludes[@]}" -C / -cf "$stage/payload.tar" "${paths[@]}"
|
||||
systemctl start "$stack_unit"
|
||||
for container in nginx-proxy-manager{% if not server_gitea_on_atlas | bool %} gitea{% endif %}; do
|
||||
running=false
|
||||
for _ in {1..30}; do
|
||||
if [[ $(podman inspect --format '{{ '{{.State.Running}}' }}' "$container" 2>/dev/null) == true ]]; then
|
||||
running=true
|
||||
break
|
||||
fi
|
||||
sleep 2
|
||||
done
|
||||
"$running" || { echo "Container did not restart: $container" >&2; exit 1; }
|
||||
done
|
||||
ready=false
|
||||
for _ in {1..60}; do
|
||||
if curl -fsS --connect-timeout 2 --max-time 3 -o /dev/null http://127.0.0.1:81/; then
|
||||
ready=true
|
||||
break
|
||||
fi
|
||||
sleep 2
|
||||
done
|
||||
"$ready" || { echo 'NPM administration did not become ready after backup' >&2; exit 1; }
|
||||
stack_stopped=false
|
||||
|
||||
tar -tf "$stage/payload.tar" >/dev/null
|
||||
(cd "$stage" && sha256sum payload.tar >payload.sha256)
|
||||
printf '{"schema":1,"host":"prometheus","created_utc":"%s"}\n' "$stamp" >"$stage/metadata.json"
|
||||
chown root:{{ server_backup_username }} "$stage" "$stage/payload.tar" "$stage/payload.sha256" "$stage/metadata.json"
|
||||
chmod 0750 "$stage"
|
||||
chmod 0640 "$stage/payload.tar" "$stage/payload.sha256" "$stage/metadata.json"
|
||||
mv -- "$stage" "$versions/$stamp"
|
||||
stage=''
|
||||
ln -s "$stamp" "$versions/.current.new"
|
||||
mv -Tf -- "$versions/.current.new" "$versions/current"
|
||||
|
||||
# Keep a small source-side safety window; Atlas owns long-term retention.
|
||||
mapfile -t old_versions < <(find "$versions" -mindepth 1 -maxdepth 1 -type d \
|
||||
-printf '%f\n' | grep -E '^[0-9]{8}T[0-9]{6}Z$' | sort -r | tail -n +{{ server_backup_export_source_keep + 1 }})
|
||||
for old in "${old_versions[@]}"; do
|
||||
rm -rf -- "${versions:?}/$old"
|
||||
done
|
||||
echo "Prepared Prometheus backup export $stamp"
|
||||
@@ -0,0 +1,10 @@
|
||||
[Unit]
|
||||
Description=Prepare daily Prometheus application backup for Atlas
|
||||
|
||||
[Timer]
|
||||
OnCalendar={{ server_backup_export_calendar }}
|
||||
Persistent=false
|
||||
Unit=prometheus-backup-export.service
|
||||
|
||||
[Install]
|
||||
WantedBy=timers.target
|
||||
@@ -0,0 +1,6 @@
|
||||
# Managed by Ansible. NPM's variable proxy upstream uses Nginx DNS, not /etc/hosts.
|
||||
{% for domain in server_gitea_npm_domains %}
|
||||
if ($host = {{ domain }}) {
|
||||
set $server {{ server_gitea_atlas_address }};
|
||||
}
|
||||
{% endfor %}
|
||||
@@ -0,0 +1,12 @@
|
||||
[Unit]
|
||||
Description=Forward public Gitea SSH to Atlas through Aegis
|
||||
Requires=prometheus-gitea-ssh-proxy.socket
|
||||
After=network-online.target wg-quick@wg0.service
|
||||
|
||||
[Service]
|
||||
ExecStart=/usr/lib/systemd/systemd-socket-proxyd {{ server_gitea_atlas_address }}:{{ server_gitea_ssh_target_port }}
|
||||
DynamicUser=true
|
||||
NoNewPrivileges=true
|
||||
ProtectSystem=strict
|
||||
ProtectHome=true
|
||||
PrivateTmp=true
|
||||
@@ -0,0 +1,9 @@
|
||||
[Unit]
|
||||
Description=Public Gitea SSH socket on Prometheus
|
||||
|
||||
[Socket]
|
||||
ListenStream=0.0.0.0:{{ server_gitea_ssh_public_port }}
|
||||
NoDelay=true
|
||||
|
||||
[Install]
|
||||
WantedBy=sockets.target
|
||||
@@ -0,0 +1,26 @@
|
||||
[Unit]
|
||||
Description=Nginx Proxy Manager on Prometheus
|
||||
RequiresMountsFor=/opt/npm/data /opt/npm/letsencrypt
|
||||
|
||||
[Container]
|
||||
Image={{ server_npm_quadlet_image }}
|
||||
ContainerName=nginx-proxy-manager
|
||||
Network=server-web.network
|
||||
NetworkAlias=nginx-proxy-manager
|
||||
AddHost=host.containers.internal:host-gateway
|
||||
PublishPort=80:80
|
||||
PublishPort=443:443
|
||||
PublishPort=127.0.0.1:81:81
|
||||
Volume=/opt/npm/data:/data
|
||||
Volume=/opt/npm/letsencrypt:/etc/letsencrypt
|
||||
Pull=missing
|
||||
|
||||
[Service]
|
||||
Restart=always
|
||||
TimeoutStartSec=180
|
||||
TimeoutStopSec=120
|
||||
|
||||
{% if server_npm_quadlet_cutover | bool %}
|
||||
[Install]
|
||||
WantedBy=multi-user.target
|
||||
{% endif %}
|
||||
@@ -0,0 +1,5 @@
|
||||
[Network]
|
||||
NetworkName=server_web
|
||||
Driver=bridge
|
||||
Subnet=10.89.0.0/24
|
||||
Gateway=10.89.0.1
|
||||
@@ -4,7 +4,7 @@ name: server
|
||||
|
||||
services:
|
||||
nginx-proxy-manager:
|
||||
image: docker.io/jc21/nginx-proxy-manager:latest
|
||||
image: {{ server_npm_quadlet_image if server_npm_quadlet_stage | bool else 'docker.io/jc21/nginx-proxy-manager:latest' }}
|
||||
container_name: nginx-proxy-manager
|
||||
restart: unless-stopped
|
||||
ports:
|
||||
@@ -38,6 +38,7 @@ services:
|
||||
# networks:
|
||||
# - web
|
||||
|
||||
{% if not server_gitea_on_atlas | bool %}
|
||||
gitea:
|
||||
image: docker.gitea.com/gitea:1.25.2
|
||||
container_name: gitea
|
||||
@@ -55,6 +56,7 @@ services:
|
||||
ports:
|
||||
- "3000:3000"
|
||||
- "127.0.0.1:222:22"
|
||||
{% endif %}
|
||||
|
||||
|
||||
networks:
|
||||
|
||||
74
docs/atlas-dr-lab.md
Normal file
74
docs/atlas-dr-lab.md
Normal file
@@ -0,0 +1,74 @@
|
||||
# Isolated Atlas DR lab
|
||||
|
||||
This is a **scaled rehearsal**, not a substitute for a full-data restore. The
|
||||
`atlas-dr-lab` libvirt VM on Ikaros was left **shut off** on 2026-09-30. Its
|
||||
persistent volumes are in the default libvirt pool: the current 30 GiB OS
|
||||
volume `atlas-dr-lab-os-rebuild2.qcow2`, the pre-rebuild OS volume
|
||||
`atlas-dr-lab-os.qcow2`, and four independent 4 GiB
|
||||
`atlas-dr-lab-data{1,2,3,4}.qcow2` volumes. The VM uses libvirt's `default`
|
||||
NAT network (last DHCP address `192.168.122.168`), 2 vCPU, and 4 GiB RAM.
|
||||
The data disks have `virtio-atlasdrdata{1,2,3,4}` serials. No physical disk or
|
||||
production Atlas storage is attached. The VM has no autostart.
|
||||
|
||||
## Rebuild inputs and isolation
|
||||
|
||||
- Use Rocky's **9.8 GenericCloud Base x86_64** image
|
||||
`Rocky-9-GenericCloud-Base-9.8-20260525.0.x86_64.qcow2` from
|
||||
`https://download.rockylinux.org/pub/rocky/9.8/images/x86_64/`.
|
||||
Verify its `.CHECKSUM` file; the observed SHA-256 was
|
||||
`92c206cc6f790c61583247eefe87890f8828420662c17cacf247cec78ab4eec8`.
|
||||
- Use a dedicated lab-only inventory merged **after** the repository
|
||||
inventory, and always `--limit atlas_dr_lab`. The temporary 2026-09-30
|
||||
inventory/playbook and logs are in `/tmp/atlas-dr-lab-image/`; copy a
|
||||
sanitized inventory to durable private storage before `/tmp` is cleared if
|
||||
the lab will be repeated. Never reuse `host_vars/atlas.yml`, production
|
||||
Vault secrets, or production disk by-id paths for the lab.
|
||||
- The lab host belongs to `platform_rocky` and `atlas`. It uses `dradmin`
|
||||
(UID/GID 1000) with the operator's **public** SSH key and a random,
|
||||
unknown password hash, the libvirt DHCP address, pool `zpool`, mount root
|
||||
`/zpool`, the four `virtio-atlasdrdata*` by-id paths, a 1 GiB backup
|
||||
reservation, and `rocky_manage_openzfs_repo: true` with only `zfs` in
|
||||
`host_packages`. The following gates remain false: sharing, firewall,
|
||||
media stack, ZFS timers, Borg, USB, monitoring, and Prometheus pull.
|
||||
`atlas_manage_storage` is true. Set `atlas_create_pool: true` **only for the
|
||||
first disposable pool creation**, then set it false before any later run.
|
||||
- A minimal lab playbook selects `atlas_dr_lab`, `become: true`, and the
|
||||
existing `packages_rocky` and `profile_atlas` roles. Use a separate
|
||||
`ANSIBLE_CONFIG` without the production Vault password script, and keep
|
||||
host-key checking on with a lab-specific known-hosts file. The 2026-09-30
|
||||
runs used `-i ansible/inventory/hosts.yml -i <lab-inventory.yml>` and
|
||||
`--limit atlas_dr_lab` throughout.
|
||||
|
||||
## Rehearsal and narrow checks
|
||||
|
||||
1. Before any pool operation, compare `virsh -c qemu:///system domblklist
|
||||
atlas-dr-lab` with the four intended qcow2 paths, and in the guest compare
|
||||
`/dev/disk/by-id/virtio-atlasdrdata*` with `lsblk`. Do not proceed if a
|
||||
physical disk or production identity appears.
|
||||
2. For a first-time disposable build only, run the lab playbook with
|
||||
`--tags pool` and `atlas_create_pool: true`, then immediately set the gate
|
||||
false. Run the full lab playbook and check `zpool status -P zpool`,
|
||||
`zfs list -r zpool`, SELinux, and failed systemd units.
|
||||
3. Write a non-sensitive canary under the lab `/zpool/archive` and snapshot
|
||||
it. Record the pool GUID and canary SHA-256. Export the lab pool cleanly,
|
||||
shut down the VM, and replace **only the OS volume** with a fresh verified
|
||||
Rocky image. Preserve all four data volumes. Reconfigure cloud-init for a
|
||||
new instance; the seed CD-ROM must use **SATA**. The SCSI seed attachment
|
||||
tried during this rehearsal was not detected by cloud-init and was
|
||||
replaced with a SATA attachment before proceeding.
|
||||
4. On the new OS, apply `packages_rocky` to reinstall OpenZFS. First run
|
||||
`zpool import -d /dev/disk/by-id` **without importing**, compare GUID and
|
||||
vdev membership, then use ordinary `zpool import -d /dev/disk/by-id zpool`.
|
||||
Do not use `-f`, `-F`, `-X`, rollback, or pool creation.
|
||||
5. Reapply `profile_atlas` with the lab gates and `atlas_create_pool: false`.
|
||||
Verify the canary, restored snapshot file in an empty temporary directory,
|
||||
dataset hierarchy, SELinux, and pool health. A second full playbook run
|
||||
should report `changed=0`. Remove temporary restored files and shut down
|
||||
the VM after testing.
|
||||
|
||||
The observed 2026-09-30 pool GUID was `8880368391795119587`; the canary
|
||||
SHA-256 was `949701c7a95fadae1fddc21abe846c4312212dbfeb7477948f3188fc3ec34a78`.
|
||||
The post-rebuild Ansible run succeeded, a repeat run reported `changed=0`,
|
||||
12 datasets and the original snapshot were present, and the pool was healthy.
|
||||
The snapshot-restored file matched content and basic metadata. See
|
||||
[`atlas-recovery.md`](atlas-recovery.md) for the production runbook and limits.
|
||||
246
docs/atlas-gitea-migration.md
Normal file
246
docs/atlas-gitea-migration.md
Normal file
@@ -0,0 +1,246 @@
|
||||
# Gitea migration from Prometheus to Atlas
|
||||
|
||||
Historical record: the completed owner-migration, migration-restore and final-export
|
||||
tasks, helpers and flags have been removed from the repository. Commands below
|
||||
record past execution, not currently supported migration entry points. Current
|
||||
service safety checks, recurring backups and proxy configuration remain managed.
|
||||
|
||||
The later 2026-10-03 canonical-domain change to `git.fscotto.co` is recorded
|
||||
in `docs/domain-fscotto-co.md`. Public SSH remains on TCP/2222; the old
|
||||
DuckDNS Proxy Host was observed disabled. Earlier domain references below
|
||||
describe migration evidence, not the current canonical URL.
|
||||
|
||||
This records the staged migration and its observed partial cutover. Gitea is
|
||||
temporary on Atlas until Uranus; NPM remains on Prometheus. On 2026-10-03
|
||||
the operator explicitly approved removal of the old Prometheus Gitea data,
|
||||
SSH fragment and final-export helper. NPM now uses a rootful Quadlet with no
|
||||
installed Compose fallback. The source-retention and rollback steps below
|
||||
are historical migration gates, not current recovery instructions.
|
||||
Existing backup archives were preserved; use current Atlas data and verified
|
||||
backups for recovery. Do not recreate or restart stale source Gitea.
|
||||
|
||||
## Observed source before cutover and chosen topology (2026-10-01)
|
||||
|
||||
- Prometheus runs the rootful `docker.gitea.com/gitea:1.25.2` image in its
|
||||
managed Compose stack. `/opt/gitea/data` is about 280 MiB, uses SQLite,
|
||||
and contains 33 repositories. A live read-only SQLite `quick_check` passed.
|
||||
`/home/git/.ssh` is a separate small bind mount; `/opt/gitea/data/ssh`
|
||||
contains the existing SSH host keys. Neither tree may be discarded.
|
||||
- Gitea answers HTTP 200 on Prometheus port 3000. NPM currently forwards
|
||||
`git.fscotto.duckdns.org` and `git.ov-ad3410.infomaniak.ch` to the Compose
|
||||
hostname `gitea:3000`. Public DNS resolves to Prometheus. The container's
|
||||
SSH port is bound only to `127.0.0.1:222`; this is not a public Gitea SSH
|
||||
listener. Prometheus' public port 22 remains administrative SSH.
|
||||
- Atlas has a healthy pool and a verified, private Prometheus backup under
|
||||
`/zpool/backup/hosts/prometheus/latest`. The 2026-10-01 scheduled export
|
||||
and pull succeeded. The intended target is a separate
|
||||
`/zpool/services/data/gitea` dataset, not `Archive` or the backup dataset.
|
||||
- The approved cutover keeps NPM on Prometheus, changes the two HTTP Proxy
|
||||
Hosts' effective upstream to Atlas over the Prometheus--Aegis gateway, and offers public Gitea
|
||||
SSH on port 2222 via the same gateway. Prometheus port 22 is unchanged.
|
||||
HTTPS and SSH must be validated together before declaring cutover.
|
||||
- The initial staging ran as a **rootless user Quadlet** under a dedicated,
|
||||
non-login Atlas account, using the pinned `1.25.2-rootless` image. This was an explicit
|
||||
rootful-to-rootless **data-layout conversion**, not a drop-in image swap:
|
||||
the target mounts `/var/lib/gitea` and `/etc/gitea`, and uses Gitea's
|
||||
built-in SSH server instead of the source image's OpenSSH daemon. Keep the
|
||||
application version unchanged until the conversion has passed an isolated
|
||||
restore test. The host's rootful Quadlet directory must not be used.
|
||||
|
||||
## Phase 1: prepare without traffic changes
|
||||
|
||||
Preparation completed on 2026-10-01: Ansible created
|
||||
`zpool/services/data/gitea`, a dedicated non-login `gitea` account (UID/GID
|
||||
1101), separate subordinate IDs, parent-dataset traverse ACLs, and an inactive
|
||||
user Quadlet under `/var/lib/atlas-gitea/.config/containers/systemd/`. The
|
||||
Quadlet has no `[Install]` section and, until the final cutover, binds only
|
||||
loopback staging ports 3001/2223 if started manually. A second targeted
|
||||
Ansible run changed nothing; the generated service was inactive and neither
|
||||
staging port listened.
|
||||
|
||||
The explicit rehearsal is managed by:
|
||||
|
||||
```bash
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags gitea_restore \
|
||||
-e atlas_gitea_restore_test=true
|
||||
```
|
||||
|
||||
On 2026-10-01 this selected the latest verified Prometheus backup, checked its
|
||||
SHA-256, extracted only `opt/gitea/data`, moved `app.ini` into the rootless
|
||||
config mount, rewrote `/data/` paths, enabled built-in SSH on internal port
|
||||
2222, and retained the three source SSH host-key pairs. SQLite `quick_check`
|
||||
passed, all 33 restored repositories passed `git fsck`, and each source/target
|
||||
public host-key fingerprint matched. A temporary `1.25.2-rootless` container
|
||||
with `--network none` answered HTTP internally and listened on internal
|
||||
SSH/2222. The container was removed; the user Quadlet remains inactive, with
|
||||
no staging listener. The second restore run changed nothing. This copy is
|
||||
deliberately stale once new source writes occur and **must not** be used as the
|
||||
final cutover copy.
|
||||
|
||||
Target backup checks on 2026-10-01: the managed recursive hourly ZFS snapshot
|
||||
`atlas-auto-hourly-20261001T193401Z` contains the new dataset. The managed
|
||||
Borg service completed archive `atlas-20261001T193420Z`, whose contents list
|
||||
includes the staged Gitea database. A separate one-file restore from each
|
||||
source into private `/var/tmp` directories matched the live staged database
|
||||
and passed SQLite `quick_check`. Temporary files and the on-demand snapshot
|
||||
mount were removed; the Borg temporary snapshot was cleaned up and the pool
|
||||
remained healthy. This is file-level proof, **not** a full Gitea recovery.
|
||||
The operator's UUID-bound offline USB run published version
|
||||
`20261001T201220Z-254397` on 2026-10-02. A separate read-only mount and
|
||||
temporary restore of `services/data/gitea/data/gitea/gitea.db` matched
|
||||
contents, owner, group, mode, size, mtime and POSIX ACL; SQLite
|
||||
`quick_check` returned `ok`. The temporary mount and copy were removed,
|
||||
LUKS was closed, and the pool was healthy. This is a file-level restore test,
|
||||
not a complete Gitea recovery rehearsal from USB.
|
||||
|
||||
1. Provision a dedicated target dataset and non-login service identity via
|
||||
Ansible, keeping UID/GID distinct from Atlas' reserved Immich `1100`.
|
||||
Install the user Quadlet in that identity's
|
||||
`~/.config/containers/systemd/`, **without** an `[Install]` section;
|
||||
do not enable, start, or expose it yet.
|
||||
2. Verify the selected Atlas backup SHA-256 and metadata, then extract **only**
|
||||
`opt/gitea/data` to private staging. Keep `home/git/.ssh` in the source
|
||||
backup for rollback; the rootless image does not consume its OpenSSH mount.
|
||||
Never unpack NPM,
|
||||
WireGuard, or other host configuration from this sensitive tarball into a
|
||||
live namespace. Convert the rootful `/data` tree on a disposable copy:
|
||||
place application data under `/var/lib/gitea`, move `app.ini` to
|
||||
`/etc/gitea`, and rewrite every absolute `/data/...` path for the new
|
||||
layout. Enable `START_SSH_SERVER`, use internal SSH port 2222, and retain
|
||||
the source host-key pairs for the built-in server only after verifying
|
||||
their fingerprints and compatibility. Do not rely on the old
|
||||
`/home/git/.ssh` OpenSSH mount in the rootless image. Set only the target
|
||||
copy's ownership and path-scoped SELinux labels.
|
||||
3. Validate SQLite integrity, repository count and representative `git fsck`,
|
||||
LFS/attachment presence, permissions, and an isolated rootless test
|
||||
container with no production ingress or outbound network. Because the
|
||||
source stays active, this is a rehearsal copy, not the final cutover copy.
|
||||
Regenerate Git hooks if the changed installation path requires it.
|
||||
4. ZFS, Borg and UUID-bound offline USB inclusion and one-file restores have
|
||||
passed. These do not replace the final consistent source copy.
|
||||
|
||||
## Phase 2: explicit final cutover
|
||||
|
||||
The opt-in `/usr/local/sbin/prometheus-gitea-final-export` helper was installed
|
||||
on 2026-10-01 and passed `bash -n`. It refuses to
|
||||
run while the scheduled Prometheus export timer is active. When explicitly
|
||||
triggered, it stops only the source Gitea container, checks SQLite, publishes
|
||||
a checksum-verified Gitea-only version for Atlas' existing pull, and leaves
|
||||
the source stopped on success. NPM remains running. A failure before
|
||||
completion restarts source Gitea. Its Ansible gate is
|
||||
`--tags gitea_final_export -e server_gitea_final_export=true`.
|
||||
After Atlas pulls that version, its separate
|
||||
`--tags gitea_final_restore -e atlas_gitea_final_restore=true` gate accepts
|
||||
only metadata marked `gitea-cutover`, validates a private staged replacement,
|
||||
and swaps it for the marked rehearsal. The swap and its rollback path passed
|
||||
synthetic tests on 2026-10-01; the live gate succeeded on 2026-10-02.
|
||||
|
||||
On 2026-10-02 the operator approved the outage. The final stopped-source
|
||||
export `20261002T071525Z` passed the Atlas pull checksum; the guarded restore
|
||||
replaced the rehearsal. SQLite `quick_check`, all 33 repository `git fsck`
|
||||
checks, and the source/target SSH host-key comparison passed. The rootless
|
||||
Atlas Quadlet serves LAN HTTP/3000 and SSH/2222, reachable from Prometheus
|
||||
through Aegis; its firewall admits only Aegis. The final marker gates startup.
|
||||
|
||||
Prometheus now runs the NPM-only Compose stack. Both NPM database records still
|
||||
say `gitea:3000`, but Nginx evaluates this variable upstream through its
|
||||
runtime DNS resolver, which **does not** use a Compose `extra_hosts` alias.
|
||||
The initial alias attempt returned 502. A managed `server_proxy.conf` override
|
||||
sets `$server` to Atlas' IP for only the two declared Gitea domains; it passed
|
||||
`nginx -t` and primary HTTPS/API returned 200 after a clean NPM restart
|
||||
without the alias; a representative public `git ls-remote` also succeeded.
|
||||
Navidrome and Syncthing Proxy Hosts still responded. No NPM SQLite records
|
||||
or credentials were changed. The
|
||||
secondary hostname `git.ov-ad3410.infomaniak.ch` did not resolve from Ikaros
|
||||
and had no generated NPM config file at the time of inspection.
|
||||
|
||||
On 2026-10-03 the operator retired this unused secondary hostname. Its NPM
|
||||
Proxy Host was already soft-deleted; Ansible now declares only
|
||||
`git.fscotto.duckdns.org` and removes the secondary runtime override.
|
||||
|
||||
Prometheus' public TCP/2222 socket proxies to Atlas without changing admin
|
||||
SSH/22. The local socket presents the preserved Gitea ED25519 host key, but
|
||||
an external TCP/2222 connection from Ikaros initially timed out. During that
|
||||
test no SYN reached Prometheus `eth0`; its socket and firewalld port were active.
|
||||
After the VPS firewall was opened later on 2026-10-02, the public port connected,
|
||||
its ED25519 host-key fingerprint matched Atlas, Gitea authenticated the `ikaros`
|
||||
key as `fscotto`, and a public SSH `git ls-remote` for `fscotto/infra.git`
|
||||
returned HEAD. The operator subsequently reported successful authenticated
|
||||
SSH pull and push; the agent did not perform a write test. HTTPS write/login
|
||||
remain untested. Do not
|
||||
restart the stale source after public HTTPS has accepted target writes.
|
||||
|
||||
The Prometheus export timer resumed with NPM-only paths. A recursive ZFS
|
||||
snapshot at `20261002T073032Z` and encrypted Borg archive
|
||||
`atlas-20261002T073044Z` captured the Atlas target after cutover; Borg exited
|
||||
successfully, cleaned its temporary snapshot, and the pool was healthy.
|
||||
|
||||
## Corrected Atlas service owner (2026-10-02)
|
||||
|
||||
The operator required the host Quadlet to belong to `admin`, while the Unix
|
||||
user **inside** the container must be named `gitea`. The pinned derived
|
||||
`Containerfile.gitea-rootless` changes only the base image's UID/GID 1000
|
||||
passwd/group names from `git` to `gitea`; it retains the rootless image's
|
||||
paths and entrypoint. Gitea's `RUN_USER` is `gitea`, while its built-in SSH
|
||||
user and advertised clone user remain `git`, preserving `git@` URLs. The
|
||||
selective restore helper now generates the same three settings for any future
|
||||
explicit restore, instead of recreating a `RUN_USER = git` target.
|
||||
|
||||
A disposable, loopback-only container using a copy of a Gitea ZFS snapshot
|
||||
passed HTTP, SQLite, internal-user and SSH host-key checks without touching
|
||||
live data. After explicit outage approval, the opt-in
|
||||
`--tags gitea_owner_migration -e atlas_gitea_owner_migration=true` run stopped
|
||||
the old user service, took safety snapshot
|
||||
`zpool/services/data/gitea@gitea-owner-migration-20261002T100104`, transferred
|
||||
only the Gitea dataset to `admin`, tested an `admin` staging Quadlet on
|
||||
loopback, then promoted it to the production LAN ports. The old Atlas Quadlet
|
||||
was removed. The old host `gitea` account and its sub-ID range are retained
|
||||
for a deliberate rollback; they must not restart stale Gitea. The parent
|
||||
traverse ACL is removed by the normal Gitea role once the new owner is live.
|
||||
|
||||
The new service returned HTTP 200 locally and through public primary HTTPS;
|
||||
Navidrome and Syncthing remained active under `admin`, the pool was healthy,
|
||||
and a second normal Gitea Ansible run was idempotent. This does **not** close
|
||||
the separate external TCP/2222 or authenticated clone/push validation gap.
|
||||
|
||||
1. Agree on an outage and record source/target versions, pool health, the
|
||||
latest backups, SSH host-key fingerprints, and both current NPM routes.
|
||||
Stop the Prometheus export timer for the change window so it cannot
|
||||
restart the old Compose stack unexpectedly.
|
||||
2. Quiesce source writes with the final-export helper: it stops Gitea before
|
||||
the consistent export and leaves it stopped after success. Pull that export
|
||||
to Atlas and verify checksum and timestamp. Keep
|
||||
`/opt/gitea/data` and `/home/git/.ssh` intact for rollback. Do not allow
|
||||
source Gitea to restart after accepting writes on Atlas.
|
||||
3. Restore the final Gitea-only payload to the target and repeat integrity
|
||||
checks. Verify its advertised SSH port is 2222, its existing HTTPS
|
||||
`ROOT_URL`, repositories, LFS/attachments, and SSH host-key identity. Enable
|
||||
the production Atlas Quadlet only after the final-restore marker exists;
|
||||
its firewall permits only Aegis to reach HTTP and SSH. Validate local HTTP
|
||||
and the target service before switching NPM.
|
||||
4. Enable the public TCP/2222 socket proxy on Prometheus to Atlas over Aegis
|
||||
without changing administrative TCP/22. Switch Prometheus to the desired
|
||||
NPM-only Compose stack and use the managed Gitea-only NPM runtime upstream
|
||||
override. Do not use Compose `extra_hosts`: Nginx bypasses it for the
|
||||
variable upstream. The old Gitea data stays intact. Do not change public DNS.
|
||||
5. Test HTTPS login, representative clone/push, LFS, and public SSH clone/push
|
||||
on port 2222 from outside the Atlas LAN. Record the last source write and
|
||||
first healthy target service times; do not claim RPO/RTO without measuring.
|
||||
6. Resume the Prometheus NPM-only backup export timer after the desired stack
|
||||
is active and verify its next result. Verify the next Atlas snapshot/Borg
|
||||
run covers Gitea and test a restored target copy. Do not delete old source
|
||||
data.
|
||||
|
||||
## Rollback gate
|
||||
|
||||
Before Atlas accepts writes, restore the old Compose definition and remove the
|
||||
NPM override, disable the public 2222 proxy, and restart the unchanged source
|
||||
Gitea if target validation fails. **After Atlas accepts writes, do not blindly restart the source:** its
|
||||
SQLite database and repositories are stale. Quiesce Atlas, capture its new
|
||||
data, and decide a reverse migration or an extended outage explicitly.
|
||||
|
||||
Upstream references: [rootful container layout](https://docs.gitea.com/1.25/installation/install-with-docker/),
|
||||
[rootless image layout and incompatibility](https://docs.gitea.com/installation/install-with-docker-rootless/),
|
||||
[rootless Podman Quadlet](https://docs.gitea.com/installation/install-with-podman-quadlet/),
|
||||
[standard-image conversion](https://docs.gitea.com/1.24/installation/install-with-docker-rootless/),
|
||||
and [restore and hook regeneration](https://docs.gitea.com/1.26/administration/backup-and-restore/).
|
||||
173
docs/atlas-icloudpd-migration.md
Normal file
173
docs/atlas-icloudpd-migration.md
Normal file
@@ -0,0 +1,173 @@
|
||||
# iCloudPD: Aegis to Atlas
|
||||
|
||||
Atlas is the temporary ingestion host until Uranus. Aegis iCloudPD and its
|
||||
state were retired. Ansible declares Atlas storage, the rootless Quadlet,
|
||||
and a private `icloudpd.conf` with the Apple ID from the existing Vault key.
|
||||
The password, keyring and MFA cookies remain application-managed; initialization
|
||||
is interactive.
|
||||
Do not place cookies, keyring files, passwords, or the Apple ID in this document,
|
||||
unencrypted repository content, or a terminal transcript.
|
||||
|
||||
## Historical source and current destination (2026-10-02)
|
||||
|
||||
- Before retirement, Aegis' rootful `icloudpd.service` was active (no reported restarts, running
|
||||
since 2026-07-25), but its declared data bind `/var/lib/icloudpd/data`
|
||||
has **zero top-level entries** and is 4 KiB as observed on 2026-10-02.
|
||||
Its persistent config has two top-level entries. `pi` cannot run passwordless
|
||||
sudo, so the container's internal filesystem and root-only state have **not**
|
||||
been audited. Do not conclude there are no photos to preserve: they could be
|
||||
inside the container overlay because the declared bind targets the wrong
|
||||
home. The current
|
||||
Quadlet mounts that data directory at `/home/root/iCloud`; the image's
|
||||
documented default is `/home/user/iCloud` with its default `user=user`.
|
||||
- The non-secret `folder_structure` value in the persisted Aegis config is a
|
||||
systemd generator path, **not** `{:%Y/%m/%d}`. The Quadlet passes percent
|
||||
characters in `Environment=` without systemd escaping; that is the likely
|
||||
cause. A running unit therefore does not prove that Aegis ingests photos.
|
||||
Do not copy this config or assume that its MFA state is usable on Atlas.
|
||||
- Atlas' `zpool` is healthy. `/zpool/archive/Pictures` already contains about
|
||||
25 GiB of unrelated data; iCloudPD gets only a new managed
|
||||
`/zpool/archive/Pictures/iCloudPD` subtree. Both that subtree and
|
||||
`zpool/services/data/icloudpd` were created on 2026-10-02. Never rsync with `--delete` into
|
||||
Pictures or adopt its existing contents. `/zpool/media/photobook` is reserved
|
||||
for Immich and remains untouched, including its Aegis-only NFS export.
|
||||
|
||||
The upstream image documents `/config/icloudpd.conf` as its primary
|
||||
configuration (environment configuration is deprecated), an exact
|
||||
`/home/${user}/iCloud/.mounted` failsafe, and an interactive `--Initialise`
|
||||
step for keyring and MFA cookies. The configuration must use the same download
|
||||
path, user/UID, and folder format as the bind mounts. References:
|
||||
[image configuration](https://github.com/boredazfcuk/docker-icloudpd/blob/master/CONFIGURATION.md),
|
||||
[Podman user namespaces](https://docs.podman.io/en/latest/markdown/podman-pod.unit.5.html).
|
||||
|
||||
## Declared Atlas target
|
||||
|
||||
| Item | Location or policy |
|
||||
| --- | --- |
|
||||
| Downloaded photos | `/zpool/archive/Pictures/iCloudPD`, a new managed subtree of the SMB `Archive` dataset |
|
||||
| Config, keyring, MFA cookies | `zpool/services/data/icloudpd` at `/zpool/services/data/icloudpd/config`, outside Archive |
|
||||
| Host service owner | `admin` rootless user manager; no rootful Quadlet or published port |
|
||||
| Container identity | Entry process root in its user namespace; downloader UID/GID 1000 maps to host `admin` |
|
||||
| Image | Digest-pinned `docker.io/boredazfcuk/icloudpd`, with no registry auto-update |
|
||||
| SELinux | Private `:Z` config bind; shared `:z` photo bind because Archive is also exposed through SMB and used by Syncthing. The label and SMB behavior require runtime testing. |
|
||||
| Access | The new subtree is `admin:admin` mode 0750. No Photobook ownership, ACL, or export changes. |
|
||||
| Sync policy | Daily interval; explicit directory/file modes 750/640; no iCloud deletion and no deletion of destination-only files |
|
||||
|
||||
The photo subtree receives a managed marker and the image's `.mounted` file.
|
||||
An existing unmarked path is refused rather than taken over. The existing
|
||||
Pictures tree is not chowned or emptied. The Quadlet now has `[Install]` with
|
||||
`WantedBy=default.target`, so the lingering admin user manager starts it at boot.
|
||||
Ansible keeps the service running. Ansible renders a mode-0600
|
||||
`icloudpd.conf` with `no_log` and no diff, but does not pull the image,
|
||||
initialize MFA, or run a cutover task. Boot startup was approved on 2026-10-03
|
||||
after a reboot left the previously manual-started service inactive.
|
||||
|
||||
The previous gated check-mode tests and isolated Quadlet-generator test proved
|
||||
only the proposed layout; they predate the simplified declarative role. They
|
||||
were not a production deployment or an authentication test.
|
||||
|
||||
## Evidence already gathered without production writes
|
||||
|
||||
The digest-pinned image was pulled into **admin's** Atlas Podman store. An
|
||||
isolated `/var/tmp` test ran with no network, a fake Apple ID, private temporary
|
||||
config/photo mounts, `keep-id:uid=1000,gid=1000`, and no new privileges. Both
|
||||
container root and UID 1000 wrote to the mounts; UID
|
||||
1000's files mapped to host `admin`. A short-lived container remained running,
|
||||
retained the intended `/home/user/iCloud` and literal `{:%Y/%m/%d}` config,
|
||||
and saw an admin-owned `.mounted` marker. The container and temporary files
|
||||
were removed. A second isolated test showed that dropping **all** container
|
||||
capabilities prevents its root entrypoint from reading an admin-owned 0600
|
||||
config; with the default rootless user-namespace capabilities it could read
|
||||
and write that file. The Quadlet retains `NoNewPrivileges=true` but does not
|
||||
drop every capability. This proves only the container layout and namespace mapping,
|
||||
**not** Apple authentication, a real download, SMB visibility, scheduled
|
||||
operation, backup coverage, or recovery.
|
||||
|
||||
The earlier disposable Photobook ACL test is superseded by the operator's
|
||||
clarification that Photobook belongs to Immich. It is not evidence for the
|
||||
current Archive destination, and the proposed Photobook ACL change was never
|
||||
deployed.
|
||||
|
||||
Backup path review on 2026-10-02: the managed Borg and USB scripts snapshot
|
||||
the pool recursively and bind every mounted child dataset, so both
|
||||
`archive` and the proposed `services/data/icloudpd` fall within their
|
||||
declared source scope. Borg's runner switches to the dedicated `borg` account
|
||||
with only `CAP_DAC_READ_SEARCH`; a read-only check using those exact `setpriv`
|
||||
capability flags could traverse/read Archive, whereas plain
|
||||
`sudo -u borg` could not. USB copies as root and preserves POSIX ACLs, but not
|
||||
generic xattrs/SELinux labels. **This was scope and permission evidence, not a
|
||||
completed backup or restore of iCloudPD data**, which did not exist at the time.
|
||||
|
||||
## Validation status and remaining checks
|
||||
|
||||
- Aegis retirement is complete: `icloudpd.service` is `not-found`/`inactive`,
|
||||
the rootful Quadlet and `/var/lib/icloudpd` are absent, and AdGuard is active.
|
||||
The temporary retirement tasks are no longer in the Aegis role. The Podman
|
||||
image cache may remain; it is not service data.
|
||||
- Atlas storage and the `admin` Quadlet are deployed. The second Ansible
|
||||
run changed nothing and did not start the service; a later manual start
|
||||
generated the config. `/zpool/media/photobook` was unchanged.
|
||||
- The image generated `/zpool/services/data/icloudpd/config/icloudpd.conf`
|
||||
on first start. Ansible replaced that default file with a private template
|
||||
using the Apple ID already in Vault. The operator initialized password
|
||||
and MFA interactively; never put credentials or codes in the repository,
|
||||
chat, or Ansible extra-vars. Automatic boot startup was separately approved
|
||||
on 2026-10-03; this does not change the interactive MFA procedure.
|
||||
- Initial ingestion completed on 2026-10-03. Still check folder structure,
|
||||
ownership, SELinux and SMB access, no unintended deletions, the next daily
|
||||
cycle, completed Borg and USB versions, and isolated restore of photos and
|
||||
private state. A recursive hourly `zpool/archive` snapshot exists after
|
||||
ingestion, but no iCloudPD-specific backup restore has passed. The first
|
||||
real scrub and measured recovery targets are separate open items.
|
||||
|
||||
On 2026-10-02 Atlas storage and the inactive Quadlet were deployed; a second
|
||||
Ansible run made zero changes. The generated service was inactive, and no
|
||||
`icloudpd.conf` existed. Two interactive-sudo Aegis runs removed its service,
|
||||
Quadlet and `/var/lib/icloudpd`, then cleared the failed-unit record left by a
|
||||
SIGKILL during shutdown. Read-only verification found `LoadState=not-found`,
|
||||
`ActiveState=inactive`, both paths absent, and AdGuard active.
|
||||
|
||||
On 2026-10-02 the operator requested the first manual start. The rootless
|
||||
service stayed active, and the image generated `icloudpd.conf` under the
|
||||
private config dataset. Its mode was tightened from 0644 to 0600. The generated
|
||||
`apple_id` field is empty; no MFA or download is verified. The service has no
|
||||
boot-time install target, so it is not configured for automatic startup.
|
||||
|
||||
The 2026-10-02 Atlas `icloudpd` run rendered the Vault-backed template without
|
||||
printing its contents; the second run made zero changes. File owner is
|
||||
`admin:admin`, mode 0600, and the Apple ID field is nonempty. The rootless
|
||||
service remained active with zero restarts. At that point keyring initialization,
|
||||
cookie creation and a real download were unverified. The template now reads
|
||||
`vault_atlas_icloudpd_apple_id`, which is already present in the encrypted
|
||||
Vault; no password or MFA code was added to the template.
|
||||
|
||||
The attempted interactive initialization then lost its container. Diagnosis
|
||||
found that the image launcher requires `traceroute` to pass its iCloud
|
||||
reachability check. Rootless Podman without `NET_RAW` returned `Operation not
|
||||
permitted` despite working Atlas/container DNS and host HTTPS. An isolated
|
||||
container with only `CAP_NET_RAW` passed the same check. The Quadlet now grants
|
||||
that single capability while keeping `NoNewPrivileges=true`; a manual restart
|
||||
passed `traceroute`, and the app stayed running. Logs then showed only the missing
|
||||
keyring and a wait for `--Initialise` again. The app expanded the generated config
|
||||
on startup, so Ansible now seeds it only when absent and idempotently maintains
|
||||
only its declared options. A second live Ansible run made zero changes. At
|
||||
that point MFA, actual ingestion, and backup/restore were unverified.
|
||||
|
||||
On 2026-10-03, after interactive initialization, the rootless service was
|
||||
active and the previous 24h of logs showed download activity with no
|
||||
authentication failures or errors. At 02:16 the application reported `All
|
||||
photos and videos have been downloaded` and `Download complete for user`.
|
||||
The destination contained 11,658 files totaling 86,020,430,015 bytes; this
|
||||
is a filesystem file count, not a count of distinct iCloud assets. A later
|
||||
read-only check found the service still active. This closes initial
|
||||
authentication and ingestion only: a subsequent daily cycle and end-to-end
|
||||
recovery of the new photos and private state remain untested.
|
||||
|
||||
On 2026-10-03 Atlas rebooted at 10:17 CEST; iCloudPD stayed inactive because
|
||||
its Quadlet had no install target. A manual start restored the running service
|
||||
and the application began listing iCloud files. The operator then approved
|
||||
persistent boot startup. The managed Quadlet now declares
|
||||
`WantedBy=default.target`; the live generator created
|
||||
`default.target.wants/atlas-icloudpd.service`, admin has `Linger=yes`, and the
|
||||
service remained active with zero restarts. No NAS reboot was performed to
|
||||
test this change; actual post-reboot startup remains untested.
|
||||
105
docs/atlas-nextcloud-design.md
Normal file
105
docs/atlas-nextcloud-design.md
Normal file
@@ -0,0 +1,105 @@
|
||||
# Nextcloud on Atlas — design draft
|
||||
|
||||
Status: the empty stack was deployed on 2026-10-03, explicitly before the first
|
||||
scrub. The operator configured DNS/NPM and authorized public cutover; public TLS,
|
||||
DAV and cross-user file checks passed. Client editing/sync acceptance and consistent
|
||||
backup/restore validation remain open before family data. iCloud import remains a
|
||||
separate operation. See `docs/atlas-nextcloud.md` for observed runtime state.
|
||||
|
||||
## Confirmed requirements
|
||||
|
||||
- Three family members are the eventual scope; provision only two standard user
|
||||
accounts initially, `fabio` and `chiara`, with the third family user deferred.
|
||||
Each initial user has a private file space and no administrator privileges.
|
||||
- Add a separate Nextcloud application administrator account named `admin`, for
|
||||
administration rather than daily document use. This is distinct from Atlas'
|
||||
host account of the same name; credentials must not be reused.
|
||||
- 2FA is optional, not enforced for the accounts. Offer enrollment and recovery
|
||||
codes; encourage it for the administrator without silently imposing it.
|
||||
- The three application accounts have been created in the empty deployment.
|
||||
- No SMTP service is available. Initial deployment will not configure outbound
|
||||
email or provision a mail server. Email notifications and email-based password
|
||||
recovery are unavailable until SMTP is explicitly added. Document administrator-
|
||||
assisted recovery for standard users and a private host-side admin recovery
|
||||
procedure; do not expose a recovery endpoint or store plaintext passwords.
|
||||
- Both initial users may add, edit and delete files in the shared `Famiglia`
|
||||
folder. This does not imply sharing personal calendars or contacts.
|
||||
- No initial per-user Nextcloud storage quota for `fabio` or `chiara`. Available
|
||||
space is still bounded by the physical pool and any separately approved dataset
|
||||
limits; monitor capacity and do not describe this as unlimited physical storage.
|
||||
- Files, calendars, contacts and Office document editing in the browser.
|
||||
- iPhone/iPad, Windows and Linux clients.
|
||||
- Migrate iCloud Drive files, calendars and contacts. The operator estimates
|
||||
approximately 50 GB of iCloud Drive files, excluding iCloudPD photos; this is
|
||||
an estimate, not a measured inventory. The files include a mix of Fabio's and
|
||||
Chiara's data. Migration is explicitly deferred to a separate later operation;
|
||||
initial deployment must not import iCloud files, calendars or contacts.
|
||||
Per-account mapping will be decided at migration time. Do not assume ongoing
|
||||
two-way synchronization with iCloud or extend this scope to iCloud Photos.
|
||||
- ONLYOFFICE is the chosen editor: browser editing on desktop and the existing
|
||||
ONLYOFFICE app on iPhone/iPad. Mobile browser editing is not required.
|
||||
- Temporary Atlas hosting, with eventual migration to Uranus.
|
||||
- Completed one-time imports/migrations stay outside the steady-state playbook.
|
||||
|
||||
## Implemented architecture — public acceptance pending
|
||||
|
||||
- Nextcloud application with Files, Calendar, Contacts and an Office connector.
|
||||
- PostgreSQL database and Redis for locking/cache; deployed versions and pinned
|
||||
image digests are declared in Atlas host vars and documented in the runbook.
|
||||
- Dedicated ONLYOFFICE Docs service and its Nextcloud connector. Test real
|
||||
DOCX/XLSX/PPTX files in desktop browsers and opening/editing/saving through
|
||||
the mobile ONLYOFFICE app before acceptance. Community Edition is deployed;
|
||||
internal connector checks passed, but browser/mobile acceptance is still pending.
|
||||
- Explicit Podman Quadlets managed by Ansible, preferably rootless like existing
|
||||
Atlas services, subject to image/user namespace/SELinux validation.
|
||||
- Separate persistent application/configuration, user files, database and cache
|
||||
storage in the service namespace. Do not expose the managed Nextcloud data
|
||||
directory as a writable SMB share or let Syncthing modify it directly.
|
||||
- Approved names: `cloud.fscotto.co` for Nextcloud and `office.fscotto.co` for
|
||||
ONLYOFFICE Docs. The operator configured DNS, certificates and NPM hosts;
|
||||
public endpoint and routing checks passed on 2026-10-03.
|
||||
- Public HTTPS through Prometheus NPM and the existing Aegis gateway only.
|
||||
No public database/cache ports or directly exposed administrative interfaces.
|
||||
- Office/Nextcloud callback routing, WebSockets, trusted proxies, JWT authentication
|
||||
and upload limits must be tested end to end before publication.
|
||||
- Credentials remain in Vault; never enter passwords or private keys in chat.
|
||||
|
||||
## Office decision
|
||||
|
||||
The operator already uses ONLYOFFICE on mobile and desktop and selected it for
|
||||
this project. Desktop browser editing will use ONLYOFFICE Docs integrated with
|
||||
Nextcloud; mobile editing will use the existing ONLYOFFICE app. The limitation
|
||||
on Community mobile web editors does not conflict with that requirement.
|
||||
App integration, permissions, document fidelity and reliable saves still require
|
||||
acceptance tests; the app is not treated as proof of server-side compatibility.
|
||||
|
||||
## Data protection and rollout gates
|
||||
|
||||
- The operator explicitly authorized this empty deployment before the first scrub.
|
||||
Close the data-protection checks before accepting live family data; this limited
|
||||
exception does not mark the scrub or recovery checks complete.
|
||||
- Re-check free RAM/CPU/storage and existing workload before choosing limits or quotas.
|
||||
- Design consistent backups covering configuration, custom apps/themes, user files
|
||||
and the database. ZFS snapshots alone do not establish application consistency.
|
||||
- Define a coordinated maintenance/background-job pause and database dump/snapshot
|
||||
procedure for recurring backups, with failure cleanup and monitoring.
|
||||
- Confirm ZFS/Borg/USB coverage and independently restore into an isolated environment
|
||||
before importing family data.
|
||||
- Define deliberate upgrades and rollback boundaries; do not roll back a database
|
||||
independently of its matching application/data backup.
|
||||
- Start with a test account and representative documents; migrate iCloud content
|
||||
explicitly only after client, sharing, Office and recovery tests pass.
|
||||
- Plan Uranus transfer separately; do not add permanent one-time migration flags.
|
||||
|
||||
## Next decisions, one at a time
|
||||
|
||||
1. Validate desktop Office editing/saving, calendar/contact synchronization and
|
||||
mobile ONLYOFFICE app integration; public empty-stack cutover is verified.
|
||||
2. Complete protection gates and application-consistent backup/recovery tests.
|
||||
3. Plan the deferred iCloud migration when explicitly requested.
|
||||
|
||||
## Primary references
|
||||
|
||||
- [Nextcloud Office installation](https://docs.nextcloud.com/server/stable/admin_manual/office/installation.html)
|
||||
- [ONLYOFFICE mobile web editor restrictions](https://helpcenter.onlyoffice.com/mobile/android/mobile-web-editors/overview.aspx)
|
||||
- [Nextcloud backup requirements](https://docs.nextcloud.com/server/stable/admin_manual/maintenance/backup.html)
|
||||
127
docs/atlas-nextcloud.md
Normal file
127
docs/atlas-nextcloud.md
Normal file
@@ -0,0 +1,127 @@
|
||||
# Atlas Nextcloud — public empty-stack cutover
|
||||
|
||||
## Observed state, 2026-10-03
|
||||
|
||||
The operator explicitly approved an empty deployment before the first monthly
|
||||
scrub, and subsequently authorized public cutover. No iCloud files, calendars
|
||||
or contacts have been imported. Public empty-stack validation is not acceptance
|
||||
of production data before the outstanding protection and recovery checks.
|
||||
|
||||
Ansible manages the steady state through `profile_atlas` and the host-local
|
||||
`atlas_manage_nextcloud: true` declaration. No migration/import flags or helpers
|
||||
were added. An actual repeat run returned `changed=0`, with no failures.
|
||||
|
||||
- Rootless `admin` Quadlets: Nextcloud 33.0.9, PostgreSQL 17.11, Redis 7.4.11 and
|
||||
ONLYOFFICE Docs Community 9.4.0.129 (image tag 9.4.0.1), on a dedicated network.
|
||||
- Images are pinned by digest; Calendar 6.6.2, Contacts 8.9.1, ONLYOFFICE connector
|
||||
10.2.1 and Team Folders 21.0.9 archives are pinned by version and SHA-256.
|
||||
- Dedicated ZFS namespace: `zpool/services/data/nextcloud`, with separate `app`,
|
||||
`files`, `database`, `cache` and `office` datasets. No writable SMB/Syncthing
|
||||
access to the Nextcloud-managed file namespace is provided.
|
||||
- The `admin` Nextcloud account is an application administrator, distinct from
|
||||
the host account. `fabio` and `chiara` are standard users in `famiglia`, each
|
||||
with no initial quota. Team folder `Famiglia` has unlimited quota and group
|
||||
permission mask 15 (read/create/update/delete, not additional re-sharing).
|
||||
- Optional TOTP is available; 2FA is not enforced. SMTP is not configured.
|
||||
- The five-minute user cron timer is active; a manual service run succeeded.
|
||||
Its `Type=oneshot` means a recurring short-lived job, not a one-time migration.
|
||||
- Component memory ceilings are Nextcloud 2 GiB, ONLYOFFICE 4 GiB, PostgreSQL
|
||||
1 GiB and Redis 256 MiB; these are ceilings, not reserved memory or load-test results.
|
||||
|
||||
Nextcloud reported installed, no maintenance mode and no pending DB upgrade.
|
||||
PostgreSQL was healthy; ONLYOFFICE `/healthcheck` returned `true`. The connector's
|
||||
`onlyoffice:documentserver --check` succeeded using internal routing. JWT is
|
||||
enabled and matches the dedicated secret; neither privileged containers nor
|
||||
container-engine socket mounts are used.
|
||||
|
||||
NPM on Prometheus reached both upstreams through the Aegis gateway. Direct LAN
|
||||
connections from Ikaros to 8080/8081 were blocked, and PostgreSQL/Redis had no
|
||||
published host ports. Existing Git, Music and Syncthing HTTPS returned 200 with
|
||||
valid TLS. NPM and its backup export timer stayed active; the pool remained healthy.
|
||||
|
||||
After operator DNS/NPM configuration, both public hostnames resolved to the VPS.
|
||||
HTTPS and HTTP-to-HTTPS redirects passed with valid certificates. Both Proxy Hosts
|
||||
were enabled with Force SSL and WebSocket support. Public Office health and its
|
||||
browser API asset returned 200; the connector check also passed. Actual browser
|
||||
editing/saving and native mobile client use remain operator acceptance tests.
|
||||
|
||||
Public session-based web login and authenticated WebDAV succeeded for admin,
|
||||
fabio and chiara. CalDAV/CardDAV
|
||||
discovery redirected to the DAV endpoint; Fabio's calendar/address-book collections
|
||||
answered PROPFIND. A uniquely named private test file was inaccessible to Chiara.
|
||||
Fabio created a test file in Famiglia; Chiara read, edited and deleted it, and Fabio
|
||||
read the updated contents. All temporary test files were removed. These are HTTP
|
||||
protocol checks, not device synchronization or large-upload acceptance evidence.
|
||||
|
||||
## Operator DNS and NPM configuration
|
||||
|
||||
Namecheap: add CNAMEs `cloud` and `office` to `fscotto.co`. Do not change the blog,
|
||||
mail records or apex IP.
|
||||
|
||||
| NPM hostname | Scheme | Upstream | Port |
|
||||
| --- | --- | --- | --- |
|
||||
| cloud.fscotto.co | http | 192.168.178.55 | 8080 |
|
||||
| office.fscotto.co | http | 192.168.178.55 | 8081 |
|
||||
|
||||
For each host, obtain a certificate for its hostname, enable Force SSL and
|
||||
WebSocket support. Keep NPM administration loopback-only; do not expose port 81.
|
||||
Nextcloud's declared upload ceiling is 2 GiB; align the proxy request-size and
|
||||
timeout settings rather than claiming large uploads work before testing them.
|
||||
Verify CalDAV/CardDAV `.well-known` redirects to `/remote.php/dav/` through NPM.
|
||||
Never disable certificate verification to make Office work.
|
||||
|
||||
The browser-facing Office URL is `https://office.fscotto.co/`; server-side routes
|
||||
use `http://atlas-onlyoffice/` and `http://atlas-nextcloud/` on the private network.
|
||||
These internal routes require explicit local-address permission in the connector
|
||||
and ONLYOFFICE. Metadata-address access remains disabled. Nextcloud trusts only
|
||||
the declared Aegis address and rootless network gateway, not arbitrary proxies.
|
||||
|
||||
## Secrets and administration
|
||||
|
||||
Six unique secrets were generated into the existing encrypted `secrets/vault.yml`:
|
||||
database, Redis, Office JWT and initial passwords for `admin`, `fabio`, `chiara`.
|
||||
Use the local Vault editor to retrieve them; do not paste them in chat.
|
||||
Account provisioning never resets an existing user's password. After a user
|
||||
changes it, the initial Vault password is not necessarily their current password.
|
||||
Database secret rotation needs a coordinated role-password update, not just an
|
||||
edited initialization file. Image/app upgrades likewise require a deliberate window.
|
||||
|
||||
Host configuration lives below `/home/admin/.config/atlas-nextcloud` with a 0700
|
||||
parent. Mounted individual secret files are readable by their container consumers,
|
||||
but a different host user was verified unable to read them through the parent.
|
||||
Nextcloud's managed PHP include inherits the live container SELinux category;
|
||||
neither global relabeling nor disabling SELinux is used.
|
||||
|
||||
```bash
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags nextcloud --check --diff
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags nextcloud
|
||||
```
|
||||
|
||||
Dry-run skips initial downloads, image pulls and runtime account/app commands;
|
||||
it is not proof of an installed or healthy stack. The deployed repeat run is
|
||||
the current idempotence evidence.
|
||||
|
||||
## Gates before family data and full client acceptance
|
||||
|
||||
- Verify the first actual scrub and the outstanding protection checks.
|
||||
- Public TLS, redirects, web login and WebDAV passed. Complete calendar/contact
|
||||
synchronization and Office editing/saving from a desktop.
|
||||
- Test opening, editing and saving from the iPhone/iPad ONLYOFFICE app; mobile
|
||||
browser editing is not a requirement. No such client test is claimed yet.
|
||||
- Private-space isolation and cross-user shared writes/deletes passed the public
|
||||
smoke test above; complete normal client acceptance as well.
|
||||
- Integrate and test application-consistent database/files backups before import.
|
||||
The new datasets fall beneath existing recursive snapshot/backup scope, but
|
||||
that alone does not verify a new Borg/USB version or a consistent Nextcloud restore.
|
||||
- For a consistent backup, coordinate pending Office saves, pause cron and writes,
|
||||
take a verified PostgreSQL dump and matching application/files snapshot, and
|
||||
resume services promptly even on failure. Extend recurring backup procedures,
|
||||
not the steady-state playbook with one-time migration tasks. Restore into an
|
||||
isolated environment using matching image/app versions, config, files and DB.
|
||||
- Confirm encrypted Vault/recovery material is available offline. Without SMTP,
|
||||
recovery for standard accounts is administrator-assisted; a forgotten admin
|
||||
password can be reset through the private host-side `occ` CLI.
|
||||
- Select versions deliberately for upgrades. Do not downgrade the application
|
||||
against an upgraded database; use matching tested backups for recovery.
|
||||
- Future Uranus migration and iCloud import are separate, explicitly authorized
|
||||
operations. No source data deletion or automatic cross-system cutover is provided.
|
||||
141
docs/atlas-recovery.md
Normal file
141
docs/atlas-recovery.md
Normal file
@@ -0,0 +1,141 @@
|
||||
# Atlas recovery runbook
|
||||
|
||||
This runbook is for a **replacement Rocky Linux 9 installation**, not a normal
|
||||
playbook run. A scaled whole-OS rebuild with a disposable pool passed in an
|
||||
isolated VM on 2026-09-30, but no production-size whole-host recovery has been
|
||||
tested. The existing production pool must be imported, never created or
|
||||
rewritten. The provisional targets are **RPO 24 hours**
|
||||
and **RTO 72 hours**, for Archive and Atlas services alike. They are planning
|
||||
objectives, not demonstrated recovery times. The manual USB cadence may leave
|
||||
an older copy; a recent Borg archive is needed to meet the RPO after total
|
||||
pool loss.
|
||||
|
||||
## Before an incident
|
||||
|
||||
- Keep an offline copy of the encrypted Ansible Vault, its unlock material,
|
||||
the exported Borg repository key, and the Borg passphrase. Do not store
|
||||
unlock material in this repository or in a recovery command line.
|
||||
On 2026-09-30 the operator confirmed these are available independently of
|
||||
Atlas and the Ansible controller; their usability has not been tested here.
|
||||
- Keep the Atlas installation media and a reproducible checkout of this
|
||||
repository available independently of Atlas. Record the exact Git revision
|
||||
used for a successful deployment.
|
||||
- Record the pool's current disk identities with `zpool status -P zpool` and
|
||||
`lsblk -o NAME,SIZE,MODEL,SERIAL,FSTYPE,UUID`. Compare these with
|
||||
`atlas_zpool_disks` before touching a replacement host. The `host_vars`
|
||||
values are historical identifiers, not evidence that a newly attached disk
|
||||
is the same device.
|
||||
- Verify that the latest hourly/daily snapshots, Borg archive, and offline USB
|
||||
version exist and note their timestamps. A timer being enabled is not proof
|
||||
that a backup completed.
|
||||
|
||||
## Incident gate
|
||||
|
||||
1. Identify whether the fault is the OS disk, one or more pool disks, accidental
|
||||
deletion, or an unavailable host. Preserve failed media when possible.
|
||||
2. Stop writes to affected services and capture the last known good backup
|
||||
timestamps. Do not run `zpool create`, `zpool destroy`, `zfs rollback`,
|
||||
`zpool import -F`, `zpool import -X`, `zpool import -f`, or disk formatting
|
||||
as a diagnostic shortcut.
|
||||
3. Choose one recovery source below. Do not merge several sources into the
|
||||
production namespace without comparing their timestamps and content.
|
||||
|
||||
## Rebuild the OS and import the existing pool
|
||||
|
||||
1. Install Rocky Linux 9 on a **separate system disk**. Configure basic network,
|
||||
SSH, a temporary sudo administrator, SELinux enforcing, and the current
|
||||
OpenZFS kmod repository. Keep the pool drives untouched.
|
||||
2. Run read-only identification: `lsblk -f`, `zpool import`, and
|
||||
`zpool import -d /dev/disk/by-id`. Check the pool GUID, vdev layout, and
|
||||
stable drive identities against the incident record. If any differ, stop.
|
||||
3. Import only after matching the expected pool and host ownership. A pool
|
||||
cleanly exported from the old host can be imported with
|
||||
`zpool import -d /dev/disk/by-id zpool`. If it reports that the pool is
|
||||
active elsewhere or needs a rewind/force, stop and investigate rather than
|
||||
adding flags. Verify with `zpool status -v zpool`, `zfs list -r zpool`,
|
||||
`zfs get -r mountpoint,canmount zpool`, and `findmnt -R /zpool`.
|
||||
4. Leave `atlas_create_pool: false`. Ensure `host_vars/atlas.yml` reflects the
|
||||
replacement host's actual SSH address and disk identities before running
|
||||
Ansible. Apply `ansible/site.yml --limit atlas` with the bootstrap admin
|
||||
connection override as documented in the Atlas setup section of README.
|
||||
This may start shares/services, so keep clients disconnected or services
|
||||
gated until data and permissions are verified.
|
||||
5. Check `getenforce`, `zpool status -v zpool`, `systemctl --failed`, SSH,
|
||||
firewalld, Cockpit, NFS, SMB, and the backup/monitoring timers. Do not
|
||||
report recovery complete on the basis of Ansible success alone.
|
||||
|
||||
## Choose the data source
|
||||
|
||||
- **Local snapshot, pool intact:** inspect `zfs list -t snapshot -r zpool`.
|
||||
Mount/access the chosen snapshot read-only and copy selected files to an
|
||||
empty staging directory; compare content, owner, mode, mtime, and POSIX ACL.
|
||||
Move into the live namespace only after an operator-approved scope review.
|
||||
Do not use an automatic rollback: it can discard newer changes in the
|
||||
dataset and descendants.
|
||||
- **Offline USB:** verify the configured LUKS and ext4 UUIDs from
|
||||
`host_vars/atlas.yml` before unlocking. Mount ext4 read-only with `ro,noload`,
|
||||
use only a published `atlas/latest` version, and restore to an empty staging
|
||||
directory. Compare checksums and metadata. The USB copy intentionally omits
|
||||
generic xattrs and SELinux labels; relabel only the restored destination.
|
||||
Never run the backup service to perform a restore.
|
||||
- **Hetzner Borg:** use the dedicated pinned host key, repository path,
|
||||
offline exported recovery key, and Vault-backed passphrase. List archives
|
||||
and extract a selected archive into an empty staging directory, never the
|
||||
live `/zpool` tree. A repository check and sample restore were previously
|
||||
performed; that does not prove this incident's archive is complete. Compare
|
||||
content and metadata before publication. Avoid `borg break-lock` while any
|
||||
backup/check job may still be active.
|
||||
|
||||
After publishing restored files, run the explicit Ansible `restorecon` tag only
|
||||
for the paths actually restored, for example:
|
||||
|
||||
```bash
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags restorecon \
|
||||
-e '{"atlas_restorecon_paths":["/zpool/archive"]}'
|
||||
```
|
||||
|
||||
Then check ownership/ACLs, application-specific integrity, SMB/NFS client
|
||||
access, backup service health, and `zpool status -v zpool`. Reconnect clients
|
||||
only after these checks pass. Record the last recoverable timestamp (actual
|
||||
RPO) and elapsed service outage (actual RTO) in the incident log.
|
||||
|
||||
## Scaled isolated rehearsal (2026-09-30)
|
||||
|
||||
The lab setup, repeatable checks, and preserved VM state are recorded in
|
||||
[`atlas-dr-lab.md`](atlas-dr-lab.md).
|
||||
|
||||
On Ikaros, a local libvirt `atlas-dr-lab` VM used a 30 GiB Rocky 9.8 system
|
||||
disk and four separate, disposable 4 GiB virtio data disks with stable
|
||||
`/dev/disk/by-id` identities. The official Rocky cloud image matched its
|
||||
published SHA-256. The lab inventory was separate from production, used a
|
||||
fresh lab-only password hash and the operator's public SSH key, and disabled
|
||||
sharing, Borg, USB backup, monitoring, media services, and the Prometheus pull.
|
||||
No production disk, Vault secret, or production data was attached or copied.
|
||||
|
||||
1. The existing `packages_rocky` and `profile_atlas` roles installed OpenZFS,
|
||||
created a RAIDZ2 `zpool` through the explicit one-time pool gate, and built
|
||||
all 12 declared datasets with a lab-sized 1 GiB backup reservation. The
|
||||
pool creation gate was set false immediately afterward.
|
||||
2. A 4 MiB canary file was written under the lab `archive` dataset and a ZFS
|
||||
snapshot created. The pool was cleanly exported and the VM shut down.
|
||||
3. Only the system-disk volume was replaced by a fresh Rocky cloud image;
|
||||
the four virtio data volumes were retained. Ansible reinstalled OpenZFS.
|
||||
Read-only `zpool import -d /dev/disk/by-id` showed the expected RAIDZ2
|
||||
topology and pool GUID `8880368391795119587` before an ordinary import
|
||||
without `-f`, rewind, or rollback.
|
||||
4. The imported pool was healthy. The canary SHA-256 matched its pre-rebuild
|
||||
value. `profile_atlas` completed against the imported pool and a second
|
||||
run reported `changed=0`. A file restored from the preserved snapshot into
|
||||
`/var/tmp` matched SHA-256, owner, group, mode, size, and mtime; the temporary
|
||||
copy was removed. Final checks found SELinux Enforcing, 12 datasets, the
|
||||
snapshot, no failed units, and a healthy pool. The VM was shut down while
|
||||
retaining its disposable volumes for a future rehearsal.
|
||||
|
||||
This proves the **sequence** for a cleanly exported, small pool and the tested
|
||||
Ansible subset, not recovery duration or capacity at 2 TB. The earlier
|
||||
2026-09-25 independent production ZFS/USB file restores and the earlier Borg
|
||||
temporary-directory restore remain separate evidence. The VM did not restore
|
||||
production USB/Borg archives, exercise services with production data, test an
|
||||
unclean import, or prove the provisional RPO/RTO. Before relying on 24h/72h,
|
||||
measure a representative full restore and service cutover in a suitably sized
|
||||
future change window. Never use the production Atlas pool for a rehearsal.
|
||||
20
docs/atlas-sharing-decision.md
Normal file
20
docs/atlas-sharing-decision.md
Normal file
@@ -0,0 +1,20 @@
|
||||
# Atlas SMB/NFS namespace decision
|
||||
|
||||
Decision date: 2026-09-30. Keep the current namespaces **separate**.
|
||||
|
||||
- `/zpool/archive` is the SMB3 `Archive` share for authorized Samba accounts.
|
||||
- `/zpool/media/photobook` is the Aegis-only NFSv4 export, `all_squash`-mapped
|
||||
to UID/GID `1100`.
|
||||
- No new dual-protocol namespace, broad export, group, or ACL model is needed.
|
||||
Existing permissions and client access remain unchanged.
|
||||
|
||||
The two paths serve different ownership and exposure needs. A common namespace
|
||||
would expand the permissions design and require same-file SMB/NFS interoperability
|
||||
testing without a present requirement. Revisit only when a specific workflow
|
||||
needs both protocols on the same files; then decide UID/GID, group, POSIX ACL,
|
||||
SELinux policy and client behavior before changing exports or permissions.
|
||||
|
||||
Read-only Atlas verification on 2026-09-30 confirmed that Samba `Archive` points
|
||||
to `/zpool/archive`, NFS exports `/zpool/media/photobook` only to
|
||||
`192.168.178.54` with `all_squash` and anonymous UID/GID `1100`, both datasets
|
||||
are distinct, and `zpool` is healthy. No sharing configuration was changed.
|
||||
63
docs/atlas-updates.md
Normal file
63
docs/atlas-updates.md
Normal file
@@ -0,0 +1,63 @@
|
||||
# Atlas Rocky/OpenZFS update and reboot procedure (draft)
|
||||
|
||||
This is an operator-controlled maintenance procedure. The playbook does not
|
||||
reboot Atlas, replace a pool device, or perform a pool feature upgrade.
|
||||
|
||||
## Preflight
|
||||
|
||||
1. Schedule an outage and confirm no Borg, USB, snapshot, scrub, or resilver
|
||||
job is active. A service in `activating` is still active; do not interrupt it.
|
||||
2. Check `zpool status -v zpool` (including scrub status), `zfs list -r zpool`,
|
||||
`systemctl --failed`, and `systemctl list-timers --all`. Resolve pool errors
|
||||
first. Record current `uname -r`, `modinfo zfs | grep '^version:'`,
|
||||
`rpm -q kernel-core kmod-zfs zfs`, and the current boot entry.
|
||||
3. Confirm a recent successful Borg archive and a usable snapshot. Confirm
|
||||
the latest published offline USB version and its physical availability;
|
||||
do not start a USB backup merely to satisfy a checklist without capacity,
|
||||
UUID, and operator checks. Record timestamps, not just timer state.
|
||||
4. Ensure console/KVM or another independent recovery route is available.
|
||||
Check free space in `/boot` and the root filesystem. Review proposed DNF
|
||||
transactions before consenting to package changes.
|
||||
|
||||
## Change window
|
||||
|
||||
1. Stop client writes and quiesce stateful applications deliberately. Record
|
||||
which services were stopped; do not assume `ansible-playbook --check` does
|
||||
this. Avoid updating during a running scrub or backup.
|
||||
2. Use `dnf upgrade --assumeno` first to review the kernel, `kmod-zfs`, `zfs`,
|
||||
and dependencies. Confirm a matching kmod will be available for the target
|
||||
kernel. If compatibility is uncertain, defer the update.
|
||||
3. Apply the approved DNF transaction. Do not run `zpool upgrade` or enable
|
||||
new pool feature flags as part of ordinary OS maintenance; that can remove
|
||||
downgrade options. Preserve at least one known-good boot entry.
|
||||
4. Reboot **manually** during the agreed outage. Ansible must not trigger it.
|
||||
|
||||
## Post-boot gate
|
||||
|
||||
1. Verify `uname -r`, `modinfo zfs`, `rpm -q kernel-core kmod-zfs zfs`,
|
||||
`zpool status -v zpool`, `zfs list -r zpool`, and `findmnt -R /zpool`.
|
||||
2. Verify SELinux remains enforcing; inspect `systemctl --failed` and the
|
||||
journal for ZFS, mount, SSH, NFS, SMB, Cockpit, Podman, and backup errors.
|
||||
3. Validate a read-only file listing through SMB and an NFS client access
|
||||
check before reopening writes. Check the rootless temporary services and
|
||||
all backup/monitoring timers. Run the Atlas health monitor in `--dry-run`
|
||||
mode, then a real check after inspection.
|
||||
4. Re-enable clients and record versions, downtime, anomalies, and next
|
||||
successful snapshot/Borg run. A green boot alone is not a completed update.
|
||||
|
||||
## Failure response
|
||||
|
||||
If the new kernel cannot load ZFS, boot the previous known-good kernel from
|
||||
the console and inspect package/kmod matching before trying another reboot.
|
||||
Do not force-import, rewind, clear errors, or upgrade pool features to make a
|
||||
failed OS update appear successful. Preserve logs and stop for a recovery
|
||||
decision if the pool does not import cleanly.
|
||||
|
||||
The procedure-definition item is complete, but the procedure is **not yet
|
||||
rehearsed** on a replacement host or during a real Atlas update. Record the
|
||||
first controlled execution and its post-boot evidence separately.
|
||||
|
||||
Read-only preflight on 2026-09-30 observed kernel
|
||||
`5.14.0-687.52.1.el9_8.x86_64`, ZFS module/package `2.2.11-1`, a healthy
|
||||
`zpool`, enforcing SELinux, and no failed systemd units. This did not review
|
||||
an upgrade transaction, stop services, or reboot the host.
|
||||
82
docs/domain-fscotto-co.md
Normal file
82
docs/domain-fscotto-co.md
Normal file
@@ -0,0 +1,82 @@
|
||||
# fscotto.co domain transition
|
||||
|
||||
## Observed state (2026-10-03)
|
||||
|
||||
Namecheap remains the DNS provider. The operator moved GitHub Pages to
|
||||
`blog.fscotto.co` in `fscotto/fscotto.github.io`, aligned Hugo and Pages
|
||||
settings, and changed the apex A record to `179.237.102.172`. The blog
|
||||
remains a CNAME to `fscotto.github.io`; mail records were left unchanged.
|
||||
A new Hugo deployment and cache clearing resolved the initial stale DNS
|
||||
and generated URLs. Blog HTTPS returned 200 with valid TLS.
|
||||
|
||||
The `git`, `music` and `syncthing` subdomains are CNAMEs to `fscotto.co`.
|
||||
The operator added NPM Proxy Hosts with certificates, WebSocket support
|
||||
and Force SSL:
|
||||
|
||||
| Hostname | HTTP upstream |
|
||||
| --- | --- |
|
||||
| git.fscotto.co | 192.168.178.55:3000 |
|
||||
| music.fscotto.co | 192.168.178.55:4533 |
|
||||
| syncthing.fscotto.co | 192.168.178.55:8384 |
|
||||
|
||||
All three redirected HTTP to HTTPS and returned final HTTPS 200 with valid
|
||||
TLS. Only the Syncthing GUI uses NPM; native synchronization is unchanged.
|
||||
NPM administration remains loopback-only on port 81 via SSH tunnel.
|
||||
|
||||
## Gitea canonical hostname
|
||||
|
||||
Atlas declares `atlas_gitea_public_domain: git.fscotto.co`. Ansible manages
|
||||
only `[server] DOMAIN`, `ROOT_URL` and `SSH_DOMAIN` in the existing private
|
||||
app.ini, preserving unrelated settings and mode 0600. Private configuration
|
||||
backups are created; diffs and secret-bearing results are suppressed.
|
||||
Only Gitea restarts when these fields change; a repeat run changed nothing.
|
||||
|
||||
HTTPS uses `https://git.fscotto.co/`; public SSH remains TCP/2222.
|
||||
Agent read-only checks returned the same HEAD from `fscotto/infra.git`
|
||||
over HTTPS and authenticated SSH. SSH host identity was checked against
|
||||
the already-trusted old endpoint key. No test push or user-authenticated
|
||||
web login was performed by the agent.
|
||||
|
||||
```bash
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags gitea_public_domain --check --diff
|
||||
```
|
||||
|
||||
Client remotes do not update automatically. Update them deliberately after
|
||||
checking repository paths; integrations and webhooks are separate operations.
|
||||
For the verified infrastructure repository only:
|
||||
|
||||
```bash
|
||||
git remote set-url origin ssh://git@git.fscotto.co:2222/fscotto/infra.git
|
||||
```
|
||||
|
||||
Do not copy this path into unrelated clones. Verify Gitea's known SSH key
|
||||
before accepting the new hostname's identity.
|
||||
|
||||
## Local DuckDNS retirement
|
||||
|
||||
DuckDNS support has been removed entirely from the server profile. On 2026-10-03 the explicit
|
||||
Ansible cleanup removed the five-minute rocky cron entry and the private
|
||||
`~/duckdns` directory containing only `duck.sh` and `duck.log`. The temporary
|
||||
cleanup tasks and flag were subsequently removed from the playbook at the
|
||||
operator's request. The updater provisioning tasks, template, variables and
|
||||
enablement flag were also removed; there is no retained opt-in support.
|
||||
The external DuckDNS name, Vault token, disabled NPM hosts and certificates
|
||||
remain untouched for a separate future decision.
|
||||
Before removing the temporary cleanup tasks, the repeat cleanup changed nothing.
|
||||
The cron table had no remaining entries, NPM and the export timer were active,
|
||||
and NPM administration still listened only on `127.0.0.1:81`.
|
||||
|
||||
## Operator-confirmed transition completion
|
||||
|
||||
On 2026-10-03 the operator confirmed completion of:
|
||||
|
||||
- Web login on the new Gitea hostname.
|
||||
- Updates to remaining Git remotes, webhooks and integrations.
|
||||
- Removal of obsolete DuckDNS NPM Proxy Hosts, unused certificates and the old upstream override.
|
||||
- Review and removal of completed one-time procedures from the playbook.
|
||||
|
||||
These are operator confirmations, not new agent runtime checks or a test push.
|
||||
At the earlier inspection the three old DuckDNS Proxy Hosts were disabled,
|
||||
not deleted; that observation predates the confirmed cleanup. Existing backup
|
||||
archives remain preserved. DNS/Pages/NPM changes were operator actions;
|
||||
the Gitea application configuration change was deployed through Ansible.
|
||||
131
docs/prometheus-backup.md
Normal file
131
docs/prometheus-backup.md
Normal file
@@ -0,0 +1,131 @@
|
||||
# Prometheus to Atlas backup pull
|
||||
|
||||
The playbook and both hosts have the dedicated identity, restricted SSH
|
||||
access, helpers, and systemd units. A manual export, pull, and temporary
|
||||
restore passed on 2026-09-30. The first scheduled export and pull passed on
|
||||
2026-10-01. After NPM moved to its Quadlet, another manual export, pull, and
|
||||
isolated restore passed on 2026-10-03. The first scheduled cycle after that
|
||||
cutover is still pending. See `docs/prometheus-npm-quadlet.md`.
|
||||
|
||||
## Declared design
|
||||
|
||||
- Prometheus prepares a tar archive of Nginx Proxy Manager data and certificates,
|
||||
its active Quadlet and network definitions,
|
||||
and SSH/firewalld/WireGuard configuration. Gitea now runs on Atlas and is no
|
||||
longer included in new Prometheus exports. NPM access logs are excluded.
|
||||
The archive contains credentials, certificates, and the WireGuard private
|
||||
key: protect both copies accordingly.
|
||||
- The approved consistency mode stops the NPM Quadlet for local tar creation
|
||||
at 02:00 Europe/Rome, then restarts it even if archiving fails. After the
|
||||
approved legacy cleanup, the helper requires the Quadlet active and has
|
||||
no Compose dependency. A manual test outside that window requires separate approval.
|
||||
- Prometheus publishes the archive with its checksum as a versioned, read-only
|
||||
source under `/var/lib/prometheus-backup-export`. A locked service account
|
||||
has no sudo or supplementary groups. Its only authorized SSH key is forced
|
||||
through Rocky's `rrsync -ro`; root owns the key file and export directories,
|
||||
so the account cannot add an unrestricted key or change prepared data.
|
||||
- Atlas generates and retains the private Ed25519 identity under
|
||||
`/etc/atlas-prometheus-pull`. Its pinned Prometheus host key came through
|
||||
the controller's already strict SSH trust; the observed fingerprint was
|
||||
`SHA256:rfedk7DHI9mLB3UHk/4F3HHlSIiswtCAFsAXvfh6iXk` on 2026-09-30.
|
||||
Atlas pulls only the prepared `current/` version, verifies SHA-256, tar
|
||||
readability, metadata, and source freshness, then publishes atomically
|
||||
below `/zpool/backup/hosts/prometheus/snapshots`. Long-term retention runs
|
||||
only after publication. A local `rrsync` fixture verified the in-tree
|
||||
`current` symlink. A live Atlas-to-Prometheus SSH test verified that the
|
||||
account could list only the prepared versions directory,
|
||||
cannot obtain a shell, and cannot write to the export. The key is restricted
|
||||
to `/var/lib/prometheus-backup-export/versions`, not the account's `.ssh`.
|
||||
- Approved source preparation is 02:00 Europe/Rome, pull 03:00, three source
|
||||
versions, and 30 daily/8 weekly/12 monthly Atlas versions. The source
|
||||
timer is non-persistent to avoid an unexpected outage after a missed run.
|
||||
Atlas rejects a prepared source older than 24 hours.
|
||||
- The Atlas pull joins the existing health monitor's timer/failure checks
|
||||
only when enabled. Its failure hook uses 45Drives Alerts; email delivery
|
||||
is not claimed. A failed source preparation should produce a stale-source
|
||||
pull failure, not a silently successful reuse of an old archive.
|
||||
|
||||
## Activation and verification
|
||||
|
||||
1. The user confirmed downtime/consistency mode, schedule, retention, and
|
||||
targeted configuration scope. Review the tar path list and exclusions
|
||||
against the actual containers.
|
||||
2. The identity and units are deployed. Re-run the targeted
|
||||
check, confirm the Atlas public key remains only the restricted Prometheus
|
||||
account's key, and verify `sshd -T -C user=prometheus-backup,...` plus
|
||||
read-only SSH denial tests after any SSH configuration change.
|
||||
3. During an agreed window, start the Prometheus export service manually.
|
||||
Confirm the active NPM service is healthy afterward, inspect the archive
|
||||
without exposing file contents, and verify the checksum/metadata.
|
||||
4. Start the Atlas pull service manually. Confirm the SSH host pin, source
|
||||
freshness, checksum, tar listing, published `latest`, retention behavior,
|
||||
clean temporary directories, and healthy pool.
|
||||
5. Independently restore the selected archive to an empty staging directory
|
||||
(never `/`) and compare NPM SQLite, data, active Quadlet files, certificates,
|
||||
permissions, and representative files. Historical pre-Gitea-cutover
|
||||
versions also include Gitea repositories; current versions do not. Test
|
||||
application startup only in an isolated environment or an approved restore
|
||||
window.
|
||||
6. Both timers are enabled. Verify their calendars and the next actual run
|
||||
after any service-ownership change. A successful manual test is not proof
|
||||
of a later scheduled cycle.
|
||||
|
||||
Narrow static validation:
|
||||
|
||||
```bash
|
||||
ANSIBLE_LOCAL_TEMP=/tmp/ansible-local \
|
||||
ansible-playbook ansible/site.yml --syntax-check
|
||||
ANSIBLE_LOCAL_TEMP=/tmp/ansible-local \
|
||||
ansible-playbook ansible/site.yml --limit prometheus,atlas \
|
||||
--tags prometheus_backup --check --diff
|
||||
```
|
||||
|
||||
Do not run the export service as part of a routine playbook deployment. The
|
||||
service restart and any restore/cutover require separate operator decisions.
|
||||
|
||||
On 2026-09-30 the initial targeted `--check --diff` run ended `changed=0`
|
||||
with gates false. After enabling **implementation only**, a targeted real run
|
||||
installed the identities and units; both timers were confirmed `disabled` and
|
||||
`inactive`, the Compose stack stayed active, and the new account was locked
|
||||
with no supplementary groups. No application was stopped.
|
||||
The rendered shell helpers passed `bash -n` and ShellCheck; the retention
|
||||
helper passed an isolated 400-version fixture. These static/isolated checks
|
||||
were followed by live SSH, export, pull, and temporary restore checks.
|
||||
Read-only preflight on 2026-09-30 found the Compose service active, all
|
||||
declared source paths present, both timers inactive, and no prepared versions.
|
||||
The source filesystem had about 6.0 GB free. Of the 2.1 GB NPM data tree,
|
||||
2.1 GB was excluded access logs, so the expected archive is much smaller than
|
||||
the raw tree size; capacity still needs verification after actual exports.
|
||||
The manual export produced a 285,777,920-byte tar (273 MiB allocated at the
|
||||
source), and Prometheus retained about 5.8 GB free. NPM and Gitea restarted;
|
||||
both containers were running and their local HTTP endpoints returned 200.
|
||||
Atlas pulled the same version, verified SHA-256, published `latest`, and kept
|
||||
the pool healthy. A full extract to `/var/tmp` yielded 4,747 files; both
|
||||
SQLite databases passed `PRAGMA integrity_check`, and one restored Gitea Git
|
||||
repository passed `git fsck`. The temporary restore directory was removed.
|
||||
This did not test application startup on an isolated host.
|
||||
|
||||
After these checks, Ansible enabled the Prometheus 02:00 Europe/Rome export
|
||||
timer and Atlas 03:00 Europe/Rome pull timer. Their first scheduled run passed
|
||||
on 2026-10-01; Atlas verified and published `20261001T000001Z` as `latest`.
|
||||
Atlas' health monitor includes the pull timer.
|
||||
|
||||
On 2026-10-03 the stopped-source version `20261003T091009Z` was verified and
|
||||
pulled before the NPM cutover. The post-cutover version `20261003T091633Z`
|
||||
was exported by the Quadlet-aware helper, checksum-verified, pulled to Atlas,
|
||||
and restored to an isolated temporary directory. NPM SQLite `quick_check`
|
||||
passed with ten proxy hosts and six certificate records. The archive contains
|
||||
both Quadlet definitions. A manifest of all 70 regular Let's Encrypt files
|
||||
and 12 symlinks, including content hashes and link targets, matched the live
|
||||
Prometheus tree. No private key or secret content was printed. The next
|
||||
scheduled export/pull is still pending observation.
|
||||
|
||||
## Post-cleanup validation (2026-10-03)
|
||||
|
||||
The operator-approved removal of legacy data and Compose fallback also
|
||||
removed those backup input paths and the obsolete Gitea mount dependency.
|
||||
A separately approved export and Atlas pull published `20261003T112906Z`.
|
||||
Both SHA-256 checks passed; an isolated SQLite restore passed `quick_check`
|
||||
and contained ten proxy hosts. Both active Quadlet definitions were present;
|
||||
retired paths were absent. Existing backup archives were not deleted by cleanup.
|
||||
The first scheduled cycle after these changes remains unverified.
|
||||
145
docs/prometheus-npm-quadlet.md
Normal file
145
docs/prometheus-npm-quadlet.md
Normal file
@@ -0,0 +1,145 @@
|
||||
# Prometheus NPM Quadlet cutover
|
||||
|
||||
## Current state (2026-10-03)
|
||||
|
||||
Nginx Proxy Manager runs as the **rootful** generated
|
||||
`prometheus-npm.service` on Prometheus. The Quadlet files are
|
||||
`/etc/containers/systemd/prometheus-npm.container` and
|
||||
`/etc/containers/systemd/server-web.network`; the image is pinned by digest
|
||||
in `ansible/inventory/host_vars/prometheus.yml`. The generated service is
|
||||
wanted by `multi-user.target` and requires the generated network service.
|
||||
The old Compose unit, Compose file and Gitea final-export helper were
|
||||
removed by the operator-approved cleanup on 2026-10-03. The retired
|
||||
application data and empty legacy directories were also removed.
|
||||
Prometheus host vars set `server_legacy_stack_retired: true` so normal runs
|
||||
do not recreate those files. Destructive deletion still requires a separate
|
||||
cleanup tag and explicit extra-var.
|
||||
|
||||
There was **no data copy** in this cutover. The Quadlet reuses the existing
|
||||
`/opt/npm/data:/data` and `/opt/npm/letsencrypt:/etc/letsencrypt` bind mounts
|
||||
with the same container name and `server_web` bridge (`10.89.0.0/24`). Ports
|
||||
80 and 443 remain public; administration port 81 remains bound to
|
||||
`127.0.0.1`. Gitea stays on Atlas, and NPM remains on Prometheus. The
|
||||
Compose fallback is no longer installed. The Quadlet uses `Pull=missing`,
|
||||
not an automatic floating-tag update.
|
||||
|
||||
## Cutover and recovery boundaries
|
||||
|
||||
The separate `scripts/cutover_prometheus_npm_quadlet.sh` was run **once** in
|
||||
the approved outage window, after source backup version
|
||||
`20261003T091009Z` was checksum-verified and pulled to Atlas. Its preflight
|
||||
required exactly the Compose owner, an inactive generated Quadlet, the
|
||||
expected image, and the current backup version. The execution held the
|
||||
backup-export lock, stopped the export timer, stopped and disabled Compose,
|
||||
started the Quadlet, checked the exact image ID, SQLite database counts,
|
||||
certificate content, Nginx configuration, and local Gitea/Syncthing HTTPS,
|
||||
then restarted the timer. Its failure trap would have restarted Compose.
|
||||
**Do not rerun that forward-cutover script after success**: its preconditions
|
||||
intentionally reject an active Quadlet.
|
||||
|
||||
Recovery is now a Quadlet rebuild and restoration from a verified Atlas
|
||||
backup, with an explicit outage decision before replacing live NPM state.
|
||||
The old Compose owner is no longer installed; reintroducing it would require
|
||||
a separately reviewed configuration and outage plan. The historical
|
||||
in-window rollback trap is not a supported post-cleanup rollback procedure.
|
||||
Do not restore an old database over a live instance or remove NPM bind mounts.
|
||||
|
||||
## Verified evidence
|
||||
|
||||
- Immediately after cutover, `prometheus-npm.service` was active with zero
|
||||
recorded restarts; Compose was inactive/disabled. The generated
|
||||
`multi-user.target.wants` link and network dependency were present. An
|
||||
actual reboot has not been performed solely for this test.
|
||||
- The running image ID matched the prior Compose image. Podman showed the
|
||||
original two bind mounts, `server_web`, public 80/443, and loopback-only 81.
|
||||
External HTTPS to Gitea and Syncthing returned 200 with TLS verification
|
||||
result 0. External access to TCP/81 timed out.
|
||||
- The first **manual post-cutover** export `20261003T091633Z` succeeded with
|
||||
the Quadlet as its active owner. The Atlas pull published that version;
|
||||
its SHA-256 payload check passed. An isolated restore passed NPM SQLite
|
||||
`quick_check` with ten proxy hosts and six certificate records. Both
|
||||
Quadlet definitions were present in the tar archive.
|
||||
- A path/content manifest of all 70 regular Let's Encrypt files and the
|
||||
path/target manifest of all 12 symlinks in the Atlas archive exactly
|
||||
matched the live Prometheus tree (aggregate SHA-256
|
||||
`ce0965fbd3ff44bb8502ed9f314e0131edd86d822039de115b39f6a2273c2da8`).
|
||||
The earlier apparent 70-vs-82 count was only a regular-file-versus-symlink
|
||||
counting difference, not missing certificate data. No certificate key
|
||||
contents were exposed during comparison.
|
||||
- The targeted `--tags npm_quadlet` normal Ansible run completed with
|
||||
`changed=0`, and the backup export timer remained active/enabled.
|
||||
|
||||
The first unattended 02:00 Europe/Rome export and 03:00 Atlas pull **after**
|
||||
this cutover have not yet occurred. Check their service results and the
|
||||
published version after the next cycle; the successful manual cycle proves
|
||||
the new path works but not its next scheduled execution.
|
||||
|
||||
```bash
|
||||
ANSIBLE_LOCAL_TEMP=/tmp/ansible-local \
|
||||
ansible-playbook ansible/site.yml --limit prometheus --tags npm_quadlet --check --diff
|
||||
sudo systemctl status prometheus-npm.service prometheus-backup-export.timer
|
||||
sudo systemctl show podman-compose-server.service -p LoadState # expected: not-found
|
||||
```
|
||||
|
||||
The backup archive includes credentials, certificates, and WireGuard
|
||||
configuration. Do not publish it or print its contents in diagnostics; see
|
||||
`docs/prometheus-backup.md` for the restricted pull and restore procedure.
|
||||
|
||||
## Selective legacy image cleanup
|
||||
|
||||
On 2026-10-03 opt-in Ansible tasks removed only the unused Gitea 1.25.2,
|
||||
Navidrome latest and PostgreSQL 13 rootful images, without force or global
|
||||
prune. Podman refuses images referenced by existing containers. The second
|
||||
run changed nothing. NPM remained active with zero restarts; local admin
|
||||
and public Gitea HTTPS returned 200. Backup timer and SSH proxy stayed active.
|
||||
|
||||
Validation:
|
||||
```bash
|
||||
ansible-playbook ansible/site.yml --limit prometheus --tags server_image_cleanup --check --diff -e server_legacy_image_cleanup=true
|
||||
```
|
||||
|
||||
The image cleanup defaults to disabled and carries the `never` tag.
|
||||
Check mode probes image presence but skips removal; it does not prove
|
||||
Podman would accept deletion. It never removes NPM resources.
|
||||
|
||||
## Approved legacy data and fallback cleanup
|
||||
|
||||
The operator explicitly approved deletion on 2026-10-03. The separate
|
||||
`server_legacy_cleanup` tasks removed `/opt/gitea`, `/home/git/.ssh`,
|
||||
`/opt/navidrome`, `/opt/postgres`, `/opt/music`, `/opt/containerd`,
|
||||
`/opt/docker`, the old Compose unit and the final Gitea export helper.
|
||||
The empty `/home/git` parent is removed only with `rmdir`, after confirming
|
||||
the Git account is absent. Guards reject symlinked paths, nested mounts,
|
||||
unexpected containers, container users of these paths, unexpected content
|
||||
in the empty legacy trees, and an active Compose or export service.
|
||||
The second cleanup run changed nothing.
|
||||
|
||||
Before deletion, Ansible removed obsolete backup input paths and the
|
||||
Gitea mount dependency. Normal Compose/template/final-export task checks
|
||||
changed nothing and did not recreate the retired files. Deletion is opt-in:
|
||||
|
||||
```bash
|
||||
ansible-playbook ansible/site.yml --limit prometheus --tags server_legacy_cleanup --check --diff -e server_legacy_cleanup=true
|
||||
```
|
||||
|
||||
Remove check mode only for approved deletion. No active NPM data, certificate,
|
||||
image, network, volume, SSH proxy, WireGuard configuration or backup archive
|
||||
is removed. No services were restarted by the cleanup.
|
||||
|
||||
After separate approval for the brief managed NPM pause, the new export
|
||||
`20261003T112906Z` completed successfully and was pulled to Atlas. SHA-256
|
||||
passed on both hosts; an isolated SQLite restore passed `quick_check` and
|
||||
contained ten proxy hosts. Both Quadlet definitions were present, and
|
||||
retired paths were absent. Temporary restore files were removed.
|
||||
NPM was active with zero automatic restarts; primary public Gitea HTTPS
|
||||
returned 200 with valid TLS. Backup timer, SSH proxy and WireGuard stayed active.
|
||||
The first scheduled post-cleanup cycle remains unverified.
|
||||
|
||||
After separate operator approval on 2026-10-03, the unused secondary hostname
|
||||
`git.ov-ad3410.infomaniak.ch` was removed from the declared domains and
|
||||
the managed NPM runtime override. Its Proxy Host (id 10) was already
|
||||
soft-deleted, with no generated config or associated certificate. Historical
|
||||
deleted records and backup archives are preserved; no DNS changes were made.
|
||||
Only `git.fscotto.duckdns.org` remains declared for the Gitea override.
|
||||
Nginx validation and reload passed without restarting NPM; the primary
|
||||
public HTTPS endpoint returned 200 with valid TLS.
|
||||
106
scripts/cutover_prometheus_npm_quadlet.sh
Normal file
106
scripts/cutover_prometheus_npm_quadlet.sh
Normal file
@@ -0,0 +1,106 @@
|
||||
#!/usr/bin/env bash
|
||||
# Run on Prometheus as root with the exact verified source-export version.
|
||||
set -Eeuo pipefail
|
||||
|
||||
expected_export=${1:?Pass the verified Prometheus backup export version}
|
||||
mode=${2:---preflight}
|
||||
[[ $expected_export =~ ^[0-9]{8}T[0-9]{6}Z$ ]] || exit 2
|
||||
[[ $mode == --preflight || $mode == --execute ]] || exit 2
|
||||
[[ $EUID -eq 0 ]] || { echo 'Run as root on Prometheus' >&2; exit 2; }
|
||||
|
||||
compose_unit=podman-compose-server.service
|
||||
quadlet_unit=prometheus-npm.service
|
||||
backup_timer=prometheus-backup-export.timer
|
||||
versions=/var/lib/prometheus-backup-export/versions
|
||||
quadlet_file=/etc/containers/systemd/prometheus-npm.container
|
||||
|
||||
exec 9>/run/lock/prometheus-backup-export.lock
|
||||
flock -n 9 || { echo 'Backup/export lock is busy' >&2; exit 1; }
|
||||
|
||||
systemctl is-active --quiet "$compose_unit"
|
||||
if systemctl is-active --quiet "$quadlet_unit"; then
|
||||
echo 'NPM Quadlet is already active; refusing overlapping cutover' >&2
|
||||
exit 1
|
||||
fi
|
||||
[[ $(systemctl show "$quadlet_unit" -p LoadState --value) == loaded ]]
|
||||
[[ $(systemctl is-enabled "$compose_unit") == enabled ]]
|
||||
[[ $(readlink "$versions/current") == "$expected_export" ]]
|
||||
image=$(sed -n 's/^Image=//p' "$quadlet_file")
|
||||
[[ $image =~ ^docker\.io/jc21/nginx-proxy-manager@sha256:[a-f0-9]{64}$ ]]
|
||||
podman image exists "$image"
|
||||
(cd "$versions/current" && sha256sum -c payload.sha256 && tar -tf payload.tar >/dev/null)
|
||||
curl -fsS --connect-timeout 2 --max-time 5 -o /dev/null http://127.0.0.1:81/
|
||||
old_image=$(podman inspect nginx-proxy-manager --format '{{.Image}}')
|
||||
|
||||
data_signature() {
|
||||
python3 - <<'PY'
|
||||
import hashlib, os, sqlite3
|
||||
db = sqlite3.connect('file:/opt/npm/data/database.sqlite?mode=ro', uri=True)
|
||||
assert db.execute('pragma quick_check').fetchone()[0] == 'ok'
|
||||
counts = [db.execute('select count(*) from ' + table).fetchone()[0]
|
||||
for table in ('proxy_host', 'certificate', 'user')]
|
||||
db.close()
|
||||
digest = hashlib.sha256()
|
||||
for root, dirs, files in os.walk('/opt/npm/letsencrypt'):
|
||||
dirs.sort()
|
||||
for name in sorted(files):
|
||||
path = os.path.join(root, name)
|
||||
with open(path, 'rb') as stream:
|
||||
digest.update(path.encode() + b'\0' + stream.read())
|
||||
print(*counts, digest.hexdigest())
|
||||
PY
|
||||
}
|
||||
before=$(data_signature)
|
||||
if [[ $mode == --preflight ]]; then
|
||||
echo 'NPM Quadlet cutover preflight passed; no service was changed'
|
||||
exit 0
|
||||
fi
|
||||
|
||||
stopped_old=false
|
||||
rollback() {
|
||||
rc=$?
|
||||
trap - EXIT
|
||||
if (( rc != 0 )) && "$stopped_old"; then
|
||||
echo 'NPM Quadlet cutover failed; restoring Compose' >&2
|
||||
systemctl stop "$quadlet_unit" || true
|
||||
systemctl enable "$compose_unit" || true
|
||||
systemctl start "$compose_unit" || true
|
||||
systemctl start "$backup_timer" || true
|
||||
curl -fsS --connect-timeout 2 --max-time 10 -o /dev/null http://127.0.0.1:81/ || true
|
||||
fi
|
||||
exit "$rc"
|
||||
}
|
||||
trap rollback EXIT
|
||||
|
||||
stopped_old=true
|
||||
systemctl stop "$backup_timer"
|
||||
systemctl stop "$compose_unit"
|
||||
if podman container exists nginx-proxy-manager; then
|
||||
echo 'Compose left the NPM container behind; refusing duplicate ownership' >&2
|
||||
exit 1
|
||||
fi
|
||||
systemctl disable "$compose_unit"
|
||||
systemctl start "$quadlet_unit"
|
||||
|
||||
ready=false
|
||||
for _ in {1..60}; do
|
||||
if curl -fsS --connect-timeout 2 --max-time 3 -o /dev/null http://127.0.0.1:81/; then
|
||||
ready=true
|
||||
break
|
||||
fi
|
||||
sleep 2
|
||||
done
|
||||
"$ready"
|
||||
systemctl is-active --quiet "$quadlet_unit"
|
||||
[[ $(podman inspect nginx-proxy-manager --format '{{.Image}}') == "$old_image" ]]
|
||||
podman exec nginx-proxy-manager nginx -t
|
||||
[[ $(data_signature) == "$before" ]]
|
||||
for hostname in git.fscotto.duckdns.org syncthing.fscotto.duckdns.org; do
|
||||
status=$(curl -ksS --connect-timeout 3 --max-time 10 \
|
||||
--resolve "$hostname:443:127.0.0.1" -o /dev/null -w '%{http_code}' \
|
||||
"https://$hostname/")
|
||||
[[ $status == 200 ]]
|
||||
done
|
||||
systemctl start "$backup_timer"
|
||||
stopped_old=false
|
||||
echo 'NPM Quadlet cutover passed local application and data checks'
|
||||
@@ -1,161 +0,0 @@
|
||||
#!/usr/bin/env sh
|
||||
|
||||
# Copy the persistent NPM and Gitea data from the retired Ubuntu server to the Rocky
|
||||
# replacement. Run this script on the Ubuntu source as root. It is a dry run
|
||||
# unless --execute and --quiesce-source are both supplied. Extended attributes
|
||||
# are deliberately not copied: Rocky must assign its own SELinux labels.
|
||||
|
||||
set -eu
|
||||
|
||||
SOURCE_COMPOSE_FILE=/opt/docker/server/docker-compose.yml
|
||||
DESTINATION=
|
||||
IDENTITY_FILE=
|
||||
EXECUTE=false
|
||||
QUIESCE_SOURCE=false
|
||||
|
||||
DATA_PATHS='
|
||||
/opt/npm/data
|
||||
/opt/npm/letsencrypt
|
||||
/opt/gitea/data
|
||||
'
|
||||
|
||||
usage() {
|
||||
cat <<'EOF'
|
||||
Usage: sudo ./scripts/migrate_prometheus_data.sh --destination USER@HOST [options]
|
||||
|
||||
Copies persistent Nginx Proxy Manager and Gitea data to the Rocky server with
|
||||
rsync. The destination Docker containers must be stopped.
|
||||
|
||||
Options:
|
||||
--destination USER@HOST Rocky SSH destination (required).
|
||||
--identity PATH SSH private key readable by root on the source host.
|
||||
--source-compose PATH Source Compose file (default: /opt/docker/server/docker-compose.yml).
|
||||
--quiesce-source Stop the source Compose stack before copying.
|
||||
--execute Perform the transfer; otherwise only show changes.
|
||||
-h, --help Show this help.
|
||||
|
||||
The script never deletes source data, destination-only files, containers, or
|
||||
volumes. It intentionally excludes Syncthing and /home/git/.ssh.
|
||||
EOF
|
||||
}
|
||||
|
||||
fail() {
|
||||
printf 'Error: %s\n' "$1" >&2
|
||||
exit 1
|
||||
}
|
||||
|
||||
require_command() {
|
||||
command -v "$1" >/dev/null 2>&1 || fail "required command not found: $1"
|
||||
}
|
||||
|
||||
while [ "$#" -gt 0 ]; do
|
||||
case "$1" in
|
||||
--destination)
|
||||
[ "$#" -ge 2 ] || fail '--destination requires USER@HOST'
|
||||
DESTINATION=$2
|
||||
shift 2
|
||||
;;
|
||||
--identity)
|
||||
[ "$#" -ge 2 ] || fail '--identity requires a path'
|
||||
IDENTITY_FILE=$2
|
||||
shift 2
|
||||
;;
|
||||
--source-compose)
|
||||
[ "$#" -ge 2 ] || fail '--source-compose requires a path'
|
||||
SOURCE_COMPOSE_FILE=$2
|
||||
shift 2
|
||||
;;
|
||||
--quiesce-source)
|
||||
QUIESCE_SOURCE=true
|
||||
shift
|
||||
;;
|
||||
--execute)
|
||||
EXECUTE=true
|
||||
shift
|
||||
;;
|
||||
-h|--help)
|
||||
usage
|
||||
exit 0
|
||||
;;
|
||||
*)
|
||||
fail "unknown option: $1"
|
||||
;;
|
||||
esac
|
||||
done
|
||||
|
||||
[ "$(id -u)" -eq 0 ] || fail 'run this script with sudo on the Ubuntu source host'
|
||||
[ -n "$DESTINATION" ] || fail '--destination is required'
|
||||
|
||||
if [ -n "$IDENTITY_FILE" ]; then
|
||||
[ -r "$IDENTITY_FILE" ] || fail "SSH identity is not readable: $IDENTITY_FILE"
|
||||
case "$IDENTITY_FILE" in
|
||||
*' '*|*"$(printf '\t')"*) fail 'SSH identity paths must not contain whitespace' ;;
|
||||
esac
|
||||
fi
|
||||
|
||||
if [ "$EXECUTE" = true ] && [ "$QUIESCE_SOURCE" != true ]; then
|
||||
fail '--execute requires --quiesce-source to keep application data consistent'
|
||||
fi
|
||||
|
||||
require_command rsync
|
||||
require_command ssh
|
||||
|
||||
SSH_COMMAND='ssh -o BatchMode=yes'
|
||||
if [ -n "$IDENTITY_FILE" ]; then
|
||||
SSH_COMMAND="$SSH_COMMAND -i $IDENTITY_FILE"
|
||||
fi
|
||||
|
||||
run_ssh() {
|
||||
# shellcheck disable=SC2086
|
||||
$SSH_COMMAND "$DESTINATION" "$@"
|
||||
}
|
||||
|
||||
printf 'Destination: %s\n' "$DESTINATION"
|
||||
printf 'Mode: %s\n' "$( [ "$EXECUTE" = true ] && printf execute || printf dry-run )"
|
||||
printf 'Data paths:\n%s\n' "$DATA_PATHS"
|
||||
|
||||
run_ssh 'sudo -n true' || fail 'destination sudo must be passwordless for this transfer'
|
||||
run_ssh 'sudo -n docker info >/dev/null' \
|
||||
|| fail 'destination Docker daemon is unavailable'
|
||||
if run_ssh 'sudo -n docker ps -q | grep -q .'; then
|
||||
fail 'destination Docker containers must be stopped before migration'
|
||||
fi
|
||||
|
||||
for path in $DATA_PATHS; do
|
||||
[ -d "$path" ] || fail "source directory is missing: $path"
|
||||
run_ssh "sudo -n test -d $path" || fail "destination directory is missing: $path"
|
||||
done
|
||||
|
||||
if [ "$QUIESCE_SOURCE" = true ]; then
|
||||
require_command docker
|
||||
[ -f "$SOURCE_COMPOSE_FILE" ] || fail "source Compose file is missing: $SOURCE_COMPOSE_FILE"
|
||||
|
||||
if [ "$EXECUTE" = true ]; then
|
||||
printf 'Stopping source Compose stack...\n'
|
||||
docker compose -f "$SOURCE_COMPOSE_FILE" stop
|
||||
else
|
||||
printf 'Dry-run: source Compose stack would be stopped.\n'
|
||||
fi
|
||||
fi
|
||||
|
||||
for path in $DATA_PATHS; do
|
||||
printf '\nSyncing %s\n' "$path"
|
||||
if [ "$EXECUTE" = true ]; then
|
||||
rsync -aHA --numeric-ids --itemize-changes --human-readable --partial \
|
||||
--rsync-path='sudo -n rsync' -e "$SSH_COMMAND" "$path/" "$DESTINATION:$path/"
|
||||
else
|
||||
rsync -aHA --numeric-ids --itemize-changes --human-readable --partial --dry-run \
|
||||
--rsync-path='sudo -n rsync' -e "$SSH_COMMAND" "$path/" "$DESTINATION:$path/"
|
||||
fi
|
||||
done
|
||||
|
||||
if [ "$EXECUTE" = true ]; then
|
||||
printf '\nVerifying source-to-destination parity...\n'
|
||||
for path in $DATA_PATHS; do
|
||||
rsync -aHA --numeric-ids --itemize-changes --dry-run \
|
||||
--rsync-path='sudo -n rsync' -e "$SSH_COMMAND" "$path/" "$DESTINATION:$path/"
|
||||
done
|
||||
printf '\nTransfer completed. Keep the source stack stopped until application validation on Rocky succeeds.\n'
|
||||
else
|
||||
printf '\nDry-run completed. Re-run with --quiesce-source --execute after reviewing the changes.\n'
|
||||
fi
|
||||
@@ -1,83 +1,101 @@
|
||||
$ANSIBLE_VAULT;1.1;AES256
|
||||
61353065386233646137323235306631353635663530363237636231316265643562353465323430
|
||||
6165646466623962313835313537633137633766373930380a316335323962616265643136346666
|
||||
63336133336131346336383534356637623831363138323165633262386333363535393365383233
|
||||
6234393835653439370a313963313365373633323464343263383661383336363662633133643232
|
||||
34366634383862363635653034313531623330396639616462343630326162316535643465653532
|
||||
36326534333637376462353561343964633636366331363833313263353133383636623537303663
|
||||
35393032316439336666343161653439643638376134363535656262343963393365623432336433
|
||||
35383934313762313037326430316666363731666231336534326661353034333063643364343230
|
||||
65333739303566366263333565333465613136646237623937393733623438613832393634663463
|
||||
39376131313234333039633735613233373931613232653036663665316636303961653834366339
|
||||
36353730316132316233303964303839363161346564396163336137663134353062363733656430
|
||||
37643339326661653031376265646132623162373562393437373437313732396537383939333666
|
||||
62353036316633306666313461663033303830393765396131643035353730383931646239663935
|
||||
32626461316364386135303761383837613063336466363162323332663764616464373565383231
|
||||
61346463336566346533326535376439643133613762383633396131323632356533636139336365
|
||||
62393838316634623932643034376631333539343965383436613364643962363834346337353334
|
||||
32656439366439313734353963343133333533653839613632323338336131373566613835393536
|
||||
31663433616334373432376531346435336530303936356461303163646463613661643161313661
|
||||
66663866343565616631616338353737356164353562366164383736346131666662623132333466
|
||||
39383865653631373232393433663430643961646265386166333137643966303834363262373636
|
||||
62396434373363353636376133666133663162653265313139313732353639336232333862643036
|
||||
64386231336561396537326139346566306434633934343038663165396665363032383466633662
|
||||
62336163633964363435386630343966333162333730336138333239646631633132663931376462
|
||||
33663139356261313065376636613930353735396131306538306664646135636336643032623131
|
||||
38346264333331353633326535326431626563323036313665643337353563333339646430386564
|
||||
31613435383036313430316366323636663735326336393338353835323861333564363832656462
|
||||
35336435623261326363633033316130393062616339353263643062633331646137376135656365
|
||||
35636139336564346164616235616431326531333433646330386134323932373339646536356464
|
||||
66343533326534326165323564663533653666633035343163633832393361336462343937623165
|
||||
62383931326630363036396333313931393836366439653433623165666166356338653364336534
|
||||
35333936653833386163633738326164386166613561333530633937343230363366333662666539
|
||||
39333361633933663735303438663239303536363433313962643137386533633539326365383765
|
||||
37636538386339333935386132353265353031643662616330316463623661663738353433313830
|
||||
36373963633166333464653338343830373063323536383364393033393235326639613662343737
|
||||
38663362636331343061646465313237313431373433353361353265333766633463353632646536
|
||||
31323231306138323031396630656538363930373439336234343963616334363632653738316465
|
||||
63653938373830336362313238656266613362636634616537653863336132343931616262396130
|
||||
66393239303866656232653832343132366537333537343635666563343639323433383163613335
|
||||
39613533376634316133633430303535306266656333626264343733666335393661666561396633
|
||||
39346265316137326465326635396362333565393133623637633132616232326263663662343137
|
||||
33363733306135363361643031306265363733656362386666306334333035393839636533343363
|
||||
35396638616636633639343930373136376339346162393061393765363837646365383866636131
|
||||
33653465666239393133616232636231333332396138376332393664343364643835306530393238
|
||||
34663237303530303837663535646263393931373531393039356336316561653130356262636562
|
||||
38336362326639653237626634376334666565653036353236313634376364626338646538386536
|
||||
38626636386466373566646166393963643164343536373236396138303532393161363335386638
|
||||
32633032393737626363613463323366366637616361313537356136626661626633613739323338
|
||||
35383963666431343566356562333234663936376562616638636261303466633539376334303331
|
||||
39303834663234663063356233313962326664383839393832303462643636393034383434303465
|
||||
64333635376135326333356435373734643430623736373234643335343130383066326436356664
|
||||
63346663326364343634303930343338336139313864316165366232643537366635653764353763
|
||||
31363863633261643263303433373330366161323166366462336332313135366338393334653764
|
||||
66353733653137663835663731373364613030373334663061313433373861613665363236633130
|
||||
65613965366636343465336533613438373466383737373366653965633437323562643966396431
|
||||
39303033643438633762633263326132663466643438656366363431616237633031333936313831
|
||||
30323930383233313032323638356333626230333764363662313662646536643839353032353462
|
||||
30326166653937353130623133303533343934633565393831623033303234316330353432313266
|
||||
30636536633933376365623665616262663236383731633633346232613366333137396139306363
|
||||
35633336643266326335303261666666653536666630613639376336373237646134306462616537
|
||||
33343561373162666332613634643837343566646161373065366637653135613632353334636363
|
||||
63363232303963646530333366663862323264326536643337323266396566316233613630303637
|
||||
66646366376466373931613734363931316230323063373666653062373364396433633762633762
|
||||
38613933323733653238383935623230383562646563363833653838636165626365646537383639
|
||||
33666535656363393562316336633439636138373365623431393965653765306138646234663938
|
||||
65653133663663393731646337386535333261643932336132396237323930306136643534353930
|
||||
65636438396432623034626561613137336138623265393064383034623863303166356138393564
|
||||
37373164626634653662326234333539663735323464613334616130643937373730363263633366
|
||||
31393937326432386165343338313031376565313866363731643534313233303064373935303538
|
||||
31343832336230393636653432653162336361383963633766343461653466316337353931333363
|
||||
63313137303564336630343937356564643763383764613362366634373362666465626334336539
|
||||
64366533376165306532343461613265366266383862323032333465336161663161376630316465
|
||||
30306562666163646235656664653635366461366435663961623635383437663564356563346462
|
||||
31636234663765623838333237393239373564366262613637363938653463396530613963643837
|
||||
38636634376637366332623035313465393762653865623130336263343663303066366135616639
|
||||
63333964356466613038303263366462346261353030646532366361393965306435613131316463
|
||||
65366266376637323764643239323730366565633335666638666334663635373961303637383861
|
||||
35313431646434656562333937663837393038386361616630626532636339306432353434656165
|
||||
33663261383166386432383465666136376237346565303164363461666663346130346162316338
|
||||
62373061353034316234303835663439396434343738303764376665336239626238386436386234
|
||||
61306166383637366266393730323732386163366261393630336431633862353761343763363665
|
||||
61323039396234393835303633363339373633653334343766653032313230343464326664356566
|
||||
3462623830666664626633373966363866333337383730313066
|
||||
36616436366637373963326235323736623235633666353235383933663230616532613131636466
|
||||
3132633562663861353835633633653764376634636638620a636662316234316164626635646539
|
||||
63343233373531373833626437656630363330363932353136653834313830646431343961386237
|
||||
3166323135376665620a383334616634356361326134313930613266333136393238366566343233
|
||||
33366335353639316164346239336539636335393130663261333065363733323163613437396332
|
||||
35663339623338363737383332396238346430353730356632623964323134663434336363613564
|
||||
63306464623331643738666234343162643630353061363231313933633733626165333763653461
|
||||
39663066376637623939383964333663306137663433313334313132323465623534666133393533
|
||||
36633036376538623764363165663861383135663437343230366165636530663165643538376161
|
||||
62653335666463653538356635333339353165336333306462373233316438386539613361383039
|
||||
34663336636565343035626238633139356638636535373239386463663738633036383861633062
|
||||
35313735333530306666313966333061326338393533333936633634633136353237643464376563
|
||||
35626266363237613037663934666538356639366637643037386336316131343965616137336330
|
||||
64623432663033653066353661613366313065366264663138643965346363626562366433326461
|
||||
65316661346631343330633033326630306536633831366231363066323861366662363861666364
|
||||
38623438633235646430613935363932386237303132343236303439633939373862366565313864
|
||||
35306332636562636466333739343663343762343163343738646234353638303134643763636639
|
||||
36613662633135303733376333613235333637646661326235373732306139363363623632666262
|
||||
38366436613733316465623438343334333861313161363131376132613232376663623230623533
|
||||
38303061386639616631383636303966666338353865626464363434353661393665613862303130
|
||||
62303664653362336433356239626661353864363537346234613331376331313038633138363565
|
||||
31616232653265343430646537373835643163396530353832366337663363386635306665643432
|
||||
66393838363266383230363633313235376130356436633137636637666562383165643862313931
|
||||
63346633343334333662363334373865653232623938363162363362646361383961376532626339
|
||||
30643537356436346161353161626232303962396463323037653235343633643261396134373061
|
||||
61323463653962363639373531366130326431353635346463396434393336313730373431316334
|
||||
37313032666231383536363535326239383363346137363037653930373261326338303936663234
|
||||
36326635346465316233363266383337343335653239393830356262346530363734383532303936
|
||||
38633065633135666438333832333336636365326430656534313332356662356165616563333035
|
||||
35363833363636346430356461306337396561366536326139623131303638333733616663653336
|
||||
65623062626366386364343036386633626236363638393565323163623936663930363864656264
|
||||
61323566376464356532316366633663623031613439653635323339363730366231326531303163
|
||||
62313638393962653064303934663436376335663763333965366230323466646463653665656466
|
||||
62643164616331636464613934376335353437653662363433363533613633633536346662656339
|
||||
62353565303464373438383234353237636239313062643036383161303735386539613533383334
|
||||
63383065613236316633623936383130353466383865376336393733663434663636333463336334
|
||||
37356233306333366463303839643363393463636630306632326339646661643162323334633331
|
||||
66613731313733646362396534356236363361363330383230303731356261333336653930303161
|
||||
35353336326438376563616534616361353233373232303034623465656261326664393962326632
|
||||
36373936393261616338396630383034323462646664623566663064316438363065646330353362
|
||||
32633566666263333863383264363762323964356430336539623633643537336538353037396566
|
||||
63396537626465363531393161653939633461366231326234663161646364616338636236313332
|
||||
35646664333763623532306637383961623538643164633939303561316262316463646665353633
|
||||
66313164646134646132653338356531303435623130343864326236353939356433396164336236
|
||||
36623066396435323532356663326163636637346463626235616132353932326438303233393830
|
||||
64326563303365646664376337303539643032363537633139623665346130636631386662373762
|
||||
66336565303334303561386134343730306566303036313933613134366238303636316238363165
|
||||
66373531633430363730376236353939626137323862623538356233616363376330366433633032
|
||||
37363538393331376665623230623233343065653139313431323966326238636663633030393734
|
||||
32643635326266646636663539353537623062633130386532373638396366353038663861333033
|
||||
63343031383238306139653932366433346564643233323937316134306666623030623137313736
|
||||
39623537346564623236353131656465326632303038366261626661333931323265396262636661
|
||||
35313739306432323034643832623831373831666231643862613736393135383561386365323835
|
||||
35623863396339653038316262303263313262616361666666343331393666663530363764643639
|
||||
31353564633835323031303835636261613839353031373334366335323465326536323762626633
|
||||
37343633376666323963336436623533346261396438343336663630366434383961393738383263
|
||||
65383036333031643336393835303835363733663634653463313639313939636539386634663464
|
||||
32643034383835333533343434656234343134313934323462643631653337383536363165613835
|
||||
63393437653261653966313237633939626330316631633335386235346465663332336337613865
|
||||
30626332353130326430316266363062353636356663663439346662313461393835663864656561
|
||||
62633131323937656239383531373863393865386265663038346535616463326630646565663463
|
||||
64393861366635656130386434633431393661656438333832633366333730643639333036613935
|
||||
31363764396466333964343136363630386530343662656362316137306634383032363962616530
|
||||
38393731346236353530626263336366376466343430316235653363396565656435323531393438
|
||||
61356464333539636637363632626661663634333331643734316230663736333134383664376231
|
||||
37613339613266663831613030633466326439323635626638343430383230333639333561646431
|
||||
66313634373365373137613134373635333535333164353134623937633066613330393430633438
|
||||
35366261633739623963623331373262313865326264386334633630653263343637343633313366
|
||||
33303036623333633365373830333465333931633761636366323939363463303239363461333139
|
||||
33396663376335646137393436323463383461336233646236306331636361653964346536666637
|
||||
36376133326431333234613435613535316263313364396362386537343565393533356564633338
|
||||
64363261323838343531326234343138303133626636653732313234383662326131313431323332
|
||||
32636539323033376434323939666437373936383763323762636439323836656432303833363362
|
||||
35646235653839303838363936613166643662393131373438633265316136663264316332663638
|
||||
32613535643565303566376530373065646333356136613462666465396566313933323261663736
|
||||
66396565396261306139396364393962393361326363666439333566376561366466626335363461
|
||||
34356232353664346234316330363962656332396236383136353461333234313662633234346565
|
||||
34376365356133396239383333643163386133316461633032373035323131663139336339346633
|
||||
32373633663231373361393762396632383738616330333038646439336532303461663430613133
|
||||
37633833393762303035353566633736353130663136626666613061383732326233303831386466
|
||||
38313162623836373533326361313031303636393564656634393263306262376336303663363933
|
||||
30643266626632366333323434383063356363646264363133306566316533356438633135336130
|
||||
62663335386335636364313234633965373961353135373339316337626665323761336133653364
|
||||
33393733383330656432356231313236646163666565373666633637373765346636336235316534
|
||||
64613536333433626261646333373539383862366334376137373862323232653362346431386164
|
||||
36643536613133396162653132616134663538393566323363353038383464663638303865376336
|
||||
30333833313764643130366533646234343339356562663036373137356565643762306261316632
|
||||
33323134633562303263383931623565383766653536353565303266353862643234346637653132
|
||||
63646130646339663035333963323366373331616462613236623133646239363134333165646133
|
||||
64643433373134656161653130323537613361643731653938623036383331633861666332376361
|
||||
39373962653630326561323662303664636161386461383833363865663935303132353637386633
|
||||
37346566626439323863393064643765636337616231363066636539306439356632633032663065
|
||||
66363839616430666233373033376362623862383066396565633632306534623036626335393039
|
||||
32343132616465383961373432336233376339393863663136663435303266333038333566313665
|
||||
61303561376366306331633730616265343662333833633533643465373663663634636632666234
|
||||
31373332323234376430306538386138316431623133626636633034333735303337663335646461
|
||||
61653439653663653930666332313334623264323539613037323534666137616165373865306531
|
||||
36306137663164316534373738383865363333316363323538356139646139363064383666626536
|
||||
66653363393433316335623063633436353761313065636631623366646633353735326362366162
|
||||
64613736343363333834
|
||||
|
||||
@@ -1,17 +1,20 @@
|
||||
---
|
||||
vault_duckdns_token: "CHANGEME"
|
||||
vault_personal_full_name: "REPLACE_ME"
|
||||
vault_git_email: "REPLACE_ME"
|
||||
vault_git_signing_key: "REPLACE_ME"
|
||||
vault_icloud_email: "REPLACE_ME"
|
||||
vault_protonmail_email: "REPLACE_ME"
|
||||
vault_icloud_mail_password: "REPLACE_ME"
|
||||
vault_git_work_email: "REPLACE_ME"
|
||||
vault_git_work_gpg: "REPLACE_ME"
|
||||
vault_openai_api_key: "REPLACE_ME"
|
||||
vault_ikaros_authorized_ssh_keys:
|
||||
- "ssh-ed25519 REPLACE_ME"
|
||||
vault_atlas_icloudpd_apple_id: "REPLACE_ME"
|
||||
vault_atlas_admin_password_hash: "REPLACE_WITH_A_SHADOW_COMPATIBLE_HASH"
|
||||
vault_atlas_samba_password: "REPLACE_ME"
|
||||
vault_atlas_immich_db_password: "REPLACE_ME"
|
||||
vault_nextcloud_database_password: "REPLACE_WITH_A_UNIQUE_RANDOM_ALPHANUMERIC_SECRET"
|
||||
vault_nextcloud_redis_password: "REPLACE_WITH_A_UNIQUE_RANDOM_ALPHANUMERIC_SECRET"
|
||||
vault_nextcloud_onlyoffice_jwt: "REPLACE_WITH_A_UNIQUE_RANDOM_ALPHANUMERIC_SECRET"
|
||||
vault_nextcloud_admin_password: "REPLACE_WITH_A_UNIQUE_RANDOM_ALPHANUMERIC_SECRET"
|
||||
vault_nextcloud_fabio_password: "REPLACE_WITH_A_UNIQUE_RANDOM_ALPHANUMERIC_SECRET"
|
||||
vault_nextcloud_chiara_password: "REPLACE_WITH_A_UNIQUE_RANDOM_ALPHANUMERIC_SECRET"
|
||||
vault_atlas_borg_passphrase: "REPLACE_WITH_A_STRONG_UNIQUE_PASSPHRASE"
|
||||
|
||||
Reference in New Issue
Block a user