From 7bc7f0e645067e7e5f84e3943d70c03ded3403e7 Mon Sep 17 00:00:00 2001 From: Fabio Scotto di Santolo Date: Sat, 3 Oct 2026 11:53:12 +0200 Subject: [PATCH] Feature/prometheus npm quadlet (#15) * Stage Prometheus NPM Quadlet with backup-safe cutover * Complete Prometheus NPM Quadlet cutover --- AGENTS.md | 32 +++++- README.it.md | 29 ++--- README.md | 27 +++-- ansible/inventory/group_vars/server.yml | 9 +- ansible/inventory/host_vars/prometheus.yml | 4 + .../tasks/backup_export_job.yml | 2 +- ansible/roles/profile_server/tasks/main.yml | 3 + .../profile_server/tasks/npm_quadlet.yml | 71 ++++++++++++ .../templates/prometheus-backup-export.sh.j2 | 26 ++++- .../templates/prometheus-npm.container.j2 | 26 +++++ .../templates/server-web.network.j2 | 5 + .../templates/server/docker-compose.yml.j2 | 2 +- docs/prometheus-backup.md | 59 ++++++---- docs/prometheus-npm-quadlet.md | 90 +++++++++++++++ scripts/cutover_prometheus_npm_quadlet.sh | 106 ++++++++++++++++++ 15 files changed, 433 insertions(+), 58 deletions(-) create mode 100644 ansible/roles/profile_server/tasks/npm_quadlet.yml create mode 100644 ansible/roles/profile_server/templates/prometheus-npm.container.j2 create mode 100644 ansible/roles/profile_server/templates/server-web.network.j2 create mode 100644 docs/prometheus-npm-quadlet.md create mode 100644 scripts/cutover_prometheus_npm_quadlet.sh diff --git a/AGENTS.md b/AGENTS.md index db34cde..629fb06 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -54,7 +54,8 @@ Ansible-driven personal infrastructure repo for Fedora and Void desktops, Fedora - Emacs is disabled by default; temporary Emacs check: `ansible-playbook ansible/site.yml --limit --tags emacs --check --diff -e emacs_enabled=true` - AI coding agents: `ansible-playbook ansible/site.yml --limit --tags ai_agents --check --diff` - Mail bootstrap: `sh -n scripts/bootstrap_mail.sh` and `shellcheck scripts/bootstrap_mail.sh` - - Server compose render: `podman-compose -f /opt/docker/server/docker-compose.yml config` and `systemctl status podman-compose-server` + - Server NPM Quadlet: `systemctl status prometheus-npm.service`; disabled Compose fallback render: + `podman-compose -f /opt/docker/server/docker-compose.yml config` - Atlas media stack: `ansible-playbook ansible/site.yml --limit atlas --tags storage,sharing,containers --check --diff` - Atlas rootless Gitea staging (does not start Gitea): @@ -90,6 +91,8 @@ Ansible-driven personal infrastructure repo for Fedora and Void desktops, Fedora `ansible-playbook ansible/site.yml --limit atlas --tags restorecon --check -e '{"atlas_restorecon_paths":["/zpool/archive"]}'` - Prometheus/Aegis WireGuard gateway: `ansible-playbook ansible/site.yml --limit prometheus,aegis --tags wireguard --check --diff` + - Prometheus NPM Quadlet steady state (does not perform a cutover): + `ansible-playbook ansible/site.yml --limit prometheus --tags npm_quadlet --check --diff` - DuckDNS config only: `ansible-playbook ansible/site.yml --limit prometheus --tags duckdns --check --diff` ## Conventions @@ -140,10 +143,12 @@ The dotfile vars follow the same split: `desktop_common_dotfiles` carries mode-i and disables diffs. Provisioning does not execute the updater or change its external schedule. - `rocky_server` is a child of both `platform_rocky` and `server`; `prometheus` is its active target. - The target must already provide `server_username` with local sudo access before the profile runs. -- The Rocky profile installs Podman and podman-compose, uses firewalld, preserves SELinux enforcement, and renders the - existing Nginx Proxy Manager/Gitea Compose stack with a `podman-compose-server` systemd unit. PostgreSQL and - Navidrome are no longer part of the desired Prometheus configuration. The role does not stop or remove legacy - containers, delete `/opt/postgres/data`, start the Compose stack, update DNS, or cut over traffic. +- The Rocky profile installs Podman and podman-compose and renders the disabled legacy + `podman-compose-server` unit for rollback. On Prometheus, Nginx Proxy Manager is now the rootful + `prometheus-npm.service` Quadlet with a pinned image digest and the existing `/opt/npm/data` and + `/opt/npm/letsencrypt` bind mounts. The rootful `server_web` bridge remains `10.89.0.0/24`. + Gitea runs on Atlas; PostgreSQL and Navidrome are absent from the desired Prometheus stack. + The profile does not delete legacy data, update DNS, or perform an implicit cutover. - Firewalld enables SSH, Cockpit (`9090/tcp`), HTTP and HTTPS. Nginx Proxy Manager publishes `80/tcp` and `443/tcp`; bind its administration interface only to `127.0.0.1:81` and use `npm-tunnel` from Ikaros or Nymph. Nextcloud remains disabled; do not provision `/srv/nextcloud` directories. @@ -394,6 +399,23 @@ successfully. The first monthly scrub remains a runtime check. of `zpool/archive` exists after ingestion, but no iCloudPD-specific backup version or restore has been verified. The first monthly scrub remains a separate open data-protection check. +## Prometheus NPM Quadlet cutover +- [x] Stage a rootful NPM Quadlet using the exact running image and the existing data/certificate + mounts, bridge subnet, public HTTP/HTTPS ports, and loopback-only administration port. + The generated service depends on `server-web-network.service` and is wanted by `multi-user.target`. +- [x] Take and verify the stopped-source export before switching owners. Version + `20261003T091009Z` was pulled to Atlas and its NPM SQLite database checked in isolation. +- [x] Cut over NPM to `prometheus-npm.service` on 2026-10-03. The legacy Compose unit is inactive + and disabled; the Quadlet is active with zero recorded restarts. Public Gitea and Syncthing + HTTPS returned 200 with valid TLS, while public TCP/81 remained unreachable. +- [x] Validate the post-cutover backup path. The export and Atlas pull published + `20261003T091633Z`; checksum, SQLite `quick_check`, ten proxy hosts, six certificate records, + both Quadlet files were present, and the complete Let's Encrypt tree (70 regular files plus + 12 symlinks) matched the live data. A targeted normal Ansible run changed nothing. Details and rollback + boundaries are in `docs/prometheus-npm-quadlet.md`. +- [ ] Observe the first scheduled export and Atlas pull after the cutover; the manual end-to-end + cycle passed, but the next unattended cycle has not yet occurred. + ## Cerberus Management Node (Deferred) `cerberus` is postponed until the office in the new house is physically set up. It is not an inventory host and this section is a design and implementation backlog, not authorization to provision it early. diff --git a/README.it.md b/README.it.md index 1e25e98..e0da59f 100644 --- a/README.it.md +++ b/README.it.md @@ -201,24 +201,26 @@ Lo stato attuale del profilo server include: - installazione pacchetti Rocky via DNF, EPEL e CRB - installazione di Podman e podman-compose - abilitazione dei servizi systemd dichiarati in inventory/group vars -- copia dei dotfiles server e rendering del `docker-compose.yml` per Nginx Proxy Manager e Gitea, - piu l'unita `podman-compose-server` (attivazione manuale) +- copia dei dotfiles server e rendering del Quadlet rootful `prometheus-npm.service` per Nginx Proxy + Manager; il vecchio `podman-compose-server` resta disabilitato soltanto per un rollback controllato - attivazione di firewalld con SSH, Cockpit (`9090/tcp`), HTTP e HTTPS abilitati - Syncthing escluso dal profilo server Rocky -Il Compose desiderato su Prometheus non include piu Navidrome ne il database PostgreSQL obsoleto. +Il Compose desiderato su Prometheus non include piu Gitea, Navidrome ne il database PostgreSQL obsoleto. Navidrome e Syncthing appartengono ad Atlas; Navidrome ufficiale usa invece SQLite. Il profilo non arresta o rimuove automaticamente eventuali container legacy e non elimina `/opt/postgres/data`. +Il cutover NPM, le verifiche dei dati e del backup e i limiti del rollback sono documentati in +[`docs/prometheus-npm-quadlet.md`](docs/prometheus-npm-quadlet.md). Nginx Proxy Manager pubblica solo `80/tcp` e `443/tcp`; la sua interfaccia di amministrazione e associata a `127.0.0.1:81` ed e raggiungibile da Ikaros o Nymph con l'alias Bash `npm-tunnel`. Nextcloud resta disabilitato e il profilo non crea directory `/srv/nextcloud`. -La fase 1 su Atlas non modifica questo deployment NPM ne i suoi dati persistenti. Dopo aver attivato -WireGuard e i servizi Atlas, configurare i proxy host NPM correnti con upstream Navidrome -`http://10.0.0.2:4533` e upstream per la GUI Syncthing `http://10.0.0.2:8384`. Solo la GUI web di -Syncthing usa NPM; il traffico di sincronizzazione resta sulle porte native pubblicate esplicitamente solo -sull'indirizzo WireGuard di Atlas. Configurare l'autenticazione Syncthing e una policy di accesso NPM adeguata prima di pubblicare la GUI. +La fase 1 su Atlas non modifica i dati persistenti NPM. I proxy host NPM usano gli upstream LAN +`http://192.168.178.55:4533` per Navidrome e `http://192.168.178.55:8384` per la GUI Syncthing; +Prometheus li raggiunge attraverso Aegis come gateway WireGuard. Solo la GUI web di Syncthing usa +NPM; il traffico di sincronizzazione resta sulle porte native esposte sulla LAN dichiarata. +Mantenere l'autenticazione Syncthing e una policy di accesso NPM adeguata. ### DuckDNS @@ -332,7 +334,7 @@ La migrazione Gitea da Prometheus ad Atlas è descritta in di `admin` su un dataset dedicato; l'immagine derivata mantiene UID/GID 1000 ma chiama l'utente interno `gitea`. NPM resta su Prometheus e l'HTTPS pubblico primario serve Atlas. L'SSH pubblico su TCP/2222 autentica la chiave `ikaros` e un `git ls-remote` è riuscito; l'operatore ha -confermato pull e push SSH. Resta da provare la scrittura via HTTPS. I dati sorgente restano +confermato pull e push SSH. Login e scrittura Git via HTTPS sono stati confermati il 2026-10-03. I dati sorgente restano conservati su Prometheus senza avviarne il vecchio container. Validare il gateway con: @@ -517,8 +519,9 @@ viene recuperato quando il timer torna attivo. `atlas-usb-backup.service` **non ha timer** e va avviato manualmente. Il timer del fornitore `zfs-scrub-weekly@zpool.timer` è disabilitato a favore dello scrub mensile. Il timer di preparazione -su Prometheus è attivo alle 02:00 Europe/Rome; export, pull e ripristino temporaneo manuali sono -riusciti il 2026-09-30, ma il primo ciclo pianificato va ancora verificato. Durante un backup Borg attivo, +su Prometheus è attivo alle 02:00 Europe/Rome; il primo ciclo pianificato è riuscito il 2026-10-01. +Un export, pull e ripristino temporaneo post-cutover NPM Quadlet sono riusciti il 2026-10-03; +il primo ciclo pianificato dopo quel cutover resta da osservare. Durante un backup Borg attivo, `systemctl list-timers` può mostrare `-` per il prossimo evento senza che il timer sia disabilitato. Per vedere la pianificazione corrente: `systemctl list-timers --all` su Atlas. @@ -638,8 +641,8 @@ Questo significa che, allo stato attuale: - `deadalus` riceve il profilo Fedora WSL tramite play dev dedicati - il server Rocky (`prometheus`) e gestito con pacchetti, servizi, dotfiles server e firewalld - il NAS Rocky (`atlas`) usa un pool ZFS gia esistente, condivisioni NFSv4/SMB limitate alla LAN e Cockpit/45Drives -- lo stack Compose server include soltanto `gitea` e `nginx-proxy-manager`; Navidrome e Syncthing - della fase 1 sono Quadlet rootless su Atlas +- NPM è un Quadlet rootful su Prometheus, mentre Gitea, Navidrome e Syncthing sono Quadlet + rootless su Atlas; il Compose server resta disabilitato per rollback # Dotfiles diff --git a/README.md b/README.md index 0dba351..314eae7 100644 --- a/README.md +++ b/README.md @@ -126,15 +126,17 @@ That gives it Fedora packages through DNF, Docker from the official repository, ## Server `prometheus` is the Rocky Linux 9 server. It has no graphical environment and gets server-specific -dotfiles and templates. The profile provisions configuration only: it does not transfer data, start -the Compose stack, update DNS, or perform a cutover. +dotfiles and templates. The profile does not transfer application data, update DNS, or perform an +implicit service cutover. The server profile installs platform-specific packages, Podman and podman-compose, declared systemd -services, and firewalld. The manually activated `podman-compose-server` unit contains the existing -Nginx Proxy Manager and Gitea services. The desired Compose file no longer includes Navidrome, -Syncthing, or the obsolete Navidrome PostgreSQL database; their temporary Atlas deployment is managed -by `profile_backend_phase1`. Applying the profile does not stop or remove legacy containers and does -not delete `/opt/postgres/data`. +services, and firewalld. Nginx Proxy Manager now runs as the rootful `prometheus-npm.service` Quadlet; +the old `podman-compose-server` unit is disabled and retained only for controlled rollback. The +Compose file no longer includes Gitea, Navidrome, Syncthing, or the obsolete Navidrome PostgreSQL +database. The temporary Navidrome and Syncthing deployment on Atlas is managed by +`profile_backend_phase1`. Applying the profile does not delete legacy data or `/opt/postgres/data`. +The NPM cutover, data checks, backup evidence, and rollback boundaries are documented in +[`docs/prometheus-npm-quadlet.md`](docs/prometheus-npm-quadlet.md). Firewalld enables SSH, Cockpit (`9090/tcp`), HTTP and HTTPS. Nginx Proxy Manager publishes only `80/tcp` and `443/tcp`; its administration interface is bound to `127.0.0.1:81` and can be reached @@ -305,9 +307,9 @@ The Gitea move from Prometheus to Atlas is tracked in [`docs/atlas-gitea-migration.md`](docs/atlas-gitea-migration.md). The final consistent copy runs in Atlas' dedicated dataset under `admin`'s rootless user Quadlet. Its pinned derived image uses an internal Unix user named `gitea` (UID/GID 1000), while clone URLs keep `git@`. NPM remains on Prometheus and the primary -public HTTPS route serves Atlas. Public SSH/2222 now authenticates the `ikaros` key and serves -read-only `git ls-remote`; the operator also confirmed SSH pull and push. HTTPS writes remain -untested. The old Gitea data remains on Prometheus, but its container +public HTTPS route serves Atlas. Public SSH/2222 authenticates the `ikaros` key and serves +`git ls-remote`; the operator also confirmed SSH pull and push. HTTPS login and Git writes were +confirmed on 2026-10-03. The old Gitea data remains on Prometheus, but its container is absent from the desired stack. The separate `wireguard_overlay` role manages `wg0` between Prometheus (`10.0.0.1`) and Aegis @@ -531,8 +533,9 @@ scheduled after the timer becomes active again. `atlas-usb-backup.service` has **no timer**: the encrypted USB backup must be started manually. The vendor's `zfs-scrub-weekly@zpool.timer` is intentionally disabled in favor of the monthly scrub. -The Prometheus export timer runs at 02:00 Europe/Rome; its first scheduled run and the Atlas pull -remain to be observed. A manual export, pull, and temporary restore passed. While a +The Prometheus export timer runs at 02:00 Europe/Rome. Its first scheduled export and Atlas pull +passed on 2026-10-01; a manual post-NPM-Quadlet export, pull, and temporary restore passed on +2026-10-03. The first scheduled cycle after that cutover remains to be observed. While a Borg backup is still running, `systemctl list-timers` may show `-` for its next trigger; this does not mean the timer has been disabled. Inspect the current schedule on Atlas with `systemctl list-timers --all`. diff --git a/ansible/inventory/group_vars/server.yml b/ansible/inventory/group_vars/server.yml index 360acdd..8208d4b 100644 --- a/ansible/inventory/group_vars/server.yml +++ b/ansible/inventory/group_vars/server.yml @@ -6,6 +6,8 @@ effective_username: "{{ server_username }}" effective_user_group: "{{ server_user_group }}" effective_user_home: "{{ server_user_home }}" server_container_stack_dir: /opt/docker/server +server_npm_quadlet_stage: false +server_npm_quadlet_cutover: false ai_agents: {} vim_plugins_enabled: false @@ -100,8 +102,11 @@ server_backup_export_paths: >- {{ ['opt/npm/data', 'opt/npm/letsencrypt'] + ([] if server_gitea_on_atlas | bool else ['opt/gitea/data', 'home/git/.ssh']) + ['opt/docker/server/docker-compose.yml', - 'etc/systemd/system/podman-compose-server.service', - 'etc/ssh/sshd_config', 'etc/ssh/sshd_config.d', + 'etc/systemd/system/podman-compose-server.service'] + + (['etc/containers/systemd/prometheus-npm.container', + 'etc/containers/systemd/server-web.network'] + if server_npm_quadlet_stage | bool else []) + + ['etc/ssh/sshd_config', 'etc/ssh/sshd_config.d', 'etc/firewalld', 'etc/wireguard/wg0.conf'] }} server_backup_export_excludes: >- {{ ['opt/npm/data/logs'] diff --git a/ansible/inventory/host_vars/prometheus.yml b/ansible/inventory/host_vars/prometheus.yml index a5c8590..1046f20 100644 --- a/ansible/inventory/host_vars/prometheus.yml +++ b/ansible/inventory/host_vars/prometheus.yml @@ -6,6 +6,10 @@ ansible_port: 22 ansible_ssh_private_key_file: /home/fscotto/.ssh/id_ed25519 server_username: rocky +server_npm_quadlet_stage: true +server_npm_quadlet_image: docker.io/jc21/nginx-proxy-manager@sha256:52b2c59994f3d36acfcf70a1626f29734df0ed8c71bacc0269f78b6f939858bb +# The stopped-source export and live Quadlet cutover passed on 2026-10-03. +server_npm_quadlet_cutover: true server_backup_export_enabled: true server_backup_export_start_timer: true # Install the final-copy helper only; it is never run by a normal playbook invocation. diff --git a/ansible/roles/profile_server/tasks/backup_export_job.yml b/ansible/roles/profile_server/tasks/backup_export_job.yml index ebed1fa..6b55d82 100644 --- a/ansible/roles/profile_server/tasks/backup_export_job.yml +++ b/ansible/roles/profile_server/tasks/backup_export_job.yml @@ -44,7 +44,7 @@ when: server_backup_export_enabled | bool - name: Install Prometheus backup export helper - tags: [services, backup, prometheus_backup, gitea_cutover] + tags: [services, backup, prometheus_backup, gitea_cutover, npm_quadlet_backup] ansible.builtin.template: src: prometheus-backup-export.sh.j2 dest: /usr/local/sbin/prometheus-backup-export diff --git a/ansible/roles/profile_server/tasks/main.yml b/ansible/roles/profile_server/tasks/main.yml index 59e7c1b..09f7a13 100644 --- a/ansible/roles/profile_server/tasks/main.yml +++ b/ansible/roles/profile_server/tasks/main.yml @@ -53,6 +53,9 @@ tags: [services, podman] ansible.builtin.include_tasks: podman-compose.yml +- name: Import staged NPM Quadlet tasks + ansible.builtin.import_tasks: npm_quadlet.yml + - name: Import Prometheus backup export identity tasks ansible.builtin.import_tasks: backup_export_identity.yml diff --git a/ansible/roles/profile_server/tasks/npm_quadlet.yml b/ansible/roles/profile_server/tasks/npm_quadlet.yml new file mode 100644 index 0000000..1d9fc91 --- /dev/null +++ b/ansible/roles/profile_server/tasks/npm_quadlet.yml @@ -0,0 +1,71 @@ +--- +- name: Require staged NPM Quadlet for an active cutover + tags: [services, npm_quadlet] + ansible.builtin.assert: + that: + - not server_npm_quadlet_cutover | bool or server_npm_quadlet_stage | bool + fail_msg: The NPM Quadlet cutover requires the staged container and network. + +- name: Validate staged NPM Quadlet inputs + tags: [services, npm_quadlet] + ansible.builtin.assert: + that: + - server_npm_quadlet_image is defined + - server_npm_quadlet_image is match('^docker\.io/jc21/nginx-proxy-manager@sha256:[a-f0-9]{64}$') + - server_gitea_on_atlas | bool + fail_msg: Stage the exact running NPM image only after Gitea has left Compose. + when: server_npm_quadlet_stage | bool + +- name: Ensure rootful Quadlet directory exists for NPM + tags: [services, npm_quadlet] + ansible.builtin.file: + path: /etc/containers/systemd + state: directory + owner: root + group: root + mode: "0755" + when: server_npm_quadlet_stage | bool + +- name: Render staged NPM container and network Quadlets + tags: [services, npm_quadlet] + ansible.builtin.template: + src: "{{ item }}.j2" + dest: "/etc/containers/systemd/{{ item }}" + owner: root + group: root + mode: "0644" + loop: + - prometheus-npm.container + - server-web.network + loop_control: + label: "{{ item }}" + register: server_npm_quadlet_units + when: server_npm_quadlet_stage | bool + +- name: Reload systemd after staging NPM Quadlets + tags: [services, npm_quadlet] + ansible.builtin.systemd: + daemon_reload: true + when: + - server_npm_quadlet_stage | bool + - server_npm_quadlet_units is changed + - not ansible_check_mode + +- name: Verify the staged NPM Quadlet was generated + tags: [services, npm_quadlet] + ansible.builtin.command: + argv: [systemctl, show, prometheus-npm.service, --property=LoadState, --value] + register: server_npm_quadlet_load_state + changed_when: false + when: + - server_npm_quadlet_stage | bool + - not ansible_check_mode + +- name: Reject an invalid staged NPM Quadlet + tags: [services, npm_quadlet] + ansible.builtin.assert: + that: server_npm_quadlet_load_state.stdout == 'loaded' + fail_msg: Quadlet generator did not produce prometheus-npm.service. + when: + - server_npm_quadlet_stage | bool + - not ansible_check_mode diff --git a/ansible/roles/profile_server/templates/prometheus-backup-export.sh.j2 b/ansible/roles/profile_server/templates/prometheus-backup-export.sh.j2 index 6a726bf..ef5f77d 100644 --- a/ansible/roles/profile_server/templates/prometheus-backup-export.sh.j2 +++ b/ansible/roles/profile_server/templates/prometheus-backup-export.sh.j2 @@ -4,7 +4,20 @@ umask 077 export_root={{ server_backup_export_root | quote }} versions="$export_root/versions" -stack_unit=podman-compose-server.service +stack_unit='' +compose_active=false +quadlet_active=false +systemctl is-active --quiet podman-compose-server.service && compose_active=true +systemctl is-active --quiet prometheus-npm.service && quadlet_active=true +if [[ "$compose_active" == "$quadlet_active" ]]; then + echo 'Expected exactly one active NPM service (Compose or Quadlet)' >&2 + exit 1 +fi +if "$quadlet_active"; then + stack_unit=prometheus-npm.service +else + stack_unit=podman-compose-server.service +fi stamp=$(date -u +%Y%m%dT%H%M%SZ) stage='' stack_stopped=false @@ -33,7 +46,7 @@ trap 'exit 130' INT trap 'exit 143' TERM systemctl is-active --quiet "$stack_unit" || { - echo 'The managed Compose stack must be active before preparing a backup' >&2 + echo "The managed NPM unit $stack_unit must be active before preparing a backup" >&2 exit 1 } @@ -70,6 +83,15 @@ for container in nginx-proxy-manager{% if not server_gitea_on_atlas | bool %} gi done "$running" || { echo "Container did not restart: $container" >&2; exit 1; } done +ready=false +for _ in {1..60}; do + if curl -fsS --connect-timeout 2 --max-time 3 -o /dev/null http://127.0.0.1:81/; then + ready=true + break + fi + sleep 2 +done +"$ready" || { echo 'NPM administration did not become ready after backup' >&2; exit 1; } stack_stopped=false tar -tf "$stage/payload.tar" >/dev/null diff --git a/ansible/roles/profile_server/templates/prometheus-npm.container.j2 b/ansible/roles/profile_server/templates/prometheus-npm.container.j2 new file mode 100644 index 0000000..e477969 --- /dev/null +++ b/ansible/roles/profile_server/templates/prometheus-npm.container.j2 @@ -0,0 +1,26 @@ +[Unit] +Description=Nginx Proxy Manager on Prometheus +RequiresMountsFor=/opt/npm/data /opt/npm/letsencrypt + +[Container] +Image={{ server_npm_quadlet_image }} +ContainerName=nginx-proxy-manager +Network=server-web.network +NetworkAlias=nginx-proxy-manager +AddHost=host.containers.internal:host-gateway +PublishPort=80:80 +PublishPort=443:443 +PublishPort=127.0.0.1:81:81 +Volume=/opt/npm/data:/data +Volume=/opt/npm/letsencrypt:/etc/letsencrypt +Pull=missing + +[Service] +Restart=always +TimeoutStartSec=180 +TimeoutStopSec=120 + +{% if server_npm_quadlet_cutover | bool %} +[Install] +WantedBy=multi-user.target +{% endif %} diff --git a/ansible/roles/profile_server/templates/server-web.network.j2 b/ansible/roles/profile_server/templates/server-web.network.j2 new file mode 100644 index 0000000..036bab5 --- /dev/null +++ b/ansible/roles/profile_server/templates/server-web.network.j2 @@ -0,0 +1,5 @@ +[Network] +NetworkName=server_web +Driver=bridge +Subnet=10.89.0.0/24 +Gateway=10.89.0.1 diff --git a/ansible/templates/server/docker-compose.yml.j2 b/ansible/templates/server/docker-compose.yml.j2 index 45055b1..b615ffa 100644 --- a/ansible/templates/server/docker-compose.yml.j2 +++ b/ansible/templates/server/docker-compose.yml.j2 @@ -4,7 +4,7 @@ name: server services: nginx-proxy-manager: - image: docker.io/jc21/nginx-proxy-manager:latest + image: {{ server_npm_quadlet_image if server_npm_quadlet_stage | bool else 'docker.io/jc21/nginx-proxy-manager:latest' }} container_name: nginx-proxy-manager restart: unless-stopped ports: diff --git a/docs/prometheus-backup.md b/docs/prometheus-backup.md index 37468cc..97504c8 100644 --- a/docs/prometheus-backup.md +++ b/docs/prometheus-backup.md @@ -2,19 +2,23 @@ The playbook and both hosts have the dedicated identity, restricted SSH access, helpers, and systemd units. A manual export, pull, and temporary -restore passed on 2026-09-30. Both timers are enabled; their first scheduled -runs are pending, so daily operation is not yet verified. +restore passed on 2026-09-30. The first scheduled export and pull passed on +2026-10-01. After NPM moved to its Quadlet, another manual export, pull, and +isolated restore passed on 2026-10-03. The first scheduled cycle after that +cutover is still pending. See `docs/prometheus-npm-quadlet.md`. ## Declared design -- Prometheus prepares a tar archive of Nginx Proxy Manager and Gitea data, - their managed Compose configuration, SSH/firewalld/WireGuard configuration, - and the Gitea SSH path. NPM access logs and regenerable Gitea logs, sessions, - temporary files, and indexers are excluded. The archive contains credentials, - certificates, and the WireGuard private key: protect both copies accordingly. -- The approved consistency mode stops the managed Compose stack for the local - tar creation at 02:00 Europe/Rome, then restarts it even if archiving fails. - A manual test outside that window requires separate approval. +- Prometheus prepares a tar archive of Nginx Proxy Manager data and certificates, + its active Quadlet and network definitions, the disabled Compose fallback, + and SSH/firewalld/WireGuard configuration. Gitea now runs on Atlas and is no + longer included in new Prometheus exports. NPM access logs are excluded. + The archive contains credentials, certificates, and the WireGuard private + key: protect both copies accordingly. +- The approved consistency mode stops the one active NPM service (Quadlet now, + Compose before cutover) for local tar creation at 02:00 Europe/Rome, then + restarts it even if archiving fails. The helper refuses both services active + or both inactive. A manual test outside that window requires separate approval. - Prometheus publishes the archive with its checksum as a versioned, read-only source under `/var/lib/prometheus-backup-export`. A locked service account has no sudo or supplementary groups. Its only authorized SSH key is forced @@ -51,18 +55,20 @@ runs are pending, so daily operation is not yet verified. account's key, and verify `sshd -T -C user=prometheus-backup,...` plus read-only SSH denial tests after any SSH configuration change. 3. During an agreed window, start the Prometheus export service manually. - Confirm Compose is healthy afterward, inspect the archive without exposing - file contents, and verify the checksum/metadata. + Confirm the active NPM service is healthy afterward, inspect the archive + without exposing file contents, and verify the checksum/metadata. 4. Start the Atlas pull service manually. Confirm the SSH host pin, source freshness, checksum, tar listing, published `latest`, retention behavior, clean temporary directories, and healthy pool. 5. Independently restore the selected archive to an empty staging directory - (never `/`) and compare the SQLite databases, Git repositories, NPM data, - Compose file, permissions, and representative files. Test application - startup only in an isolated environment or an approved restore window. -6. The two timers were enabled after the manual test. Verify their calendars - and the next actual run. A successful manual test is not proof of scheduled - operation. + (never `/`) and compare NPM SQLite, data, active Quadlet files, certificates, + permissions, and representative files. Historical pre-Gitea-cutover + versions also include Gitea repositories; current versions do not. Test + application startup only in an isolated environment or an approved restore + window. +6. Both timers are enabled. Verify their calendars and the next actual run + after any service-ownership change. A successful manual test is not proof + of a later scheduled cycle. Narrow static validation: @@ -100,7 +106,16 @@ repository passed `git fsck`. The temporary restore directory was removed. This did not test application startup on an isolated host. After these checks, Ansible enabled the Prometheus 02:00 Europe/Rome export -timer and Atlas 03:00 Europe/Rome pull timer. The next scheduled occurrences -were displayed for 2026-10-01. Atlas' health monitor now includes the pull -timer. Check both actual service results after the first scheduled run before -claiming unattended operation. +timer and Atlas 03:00 Europe/Rome pull timer. Their first scheduled run passed +on 2026-10-01; Atlas verified and published `20261001T000001Z` as `latest`. +Atlas' health monitor includes the pull timer. + +On 2026-10-03 the stopped-source version `20261003T091009Z` was verified and +pulled before the NPM cutover. The post-cutover version `20261003T091633Z` +was exported by the Quadlet-aware helper, checksum-verified, pulled to Atlas, +and restored to an isolated temporary directory. NPM SQLite `quick_check` +passed with ten proxy hosts and six certificate records. The archive contains +both Quadlet definitions. A manifest of all 70 regular Let's Encrypt files +and 12 symlinks, including content hashes and link targets, matched the live +Prometheus tree. No private key or secret content was printed. The next +scheduled export/pull is still pending observation. diff --git a/docs/prometheus-npm-quadlet.md b/docs/prometheus-npm-quadlet.md new file mode 100644 index 0000000..61b772c --- /dev/null +++ b/docs/prometheus-npm-quadlet.md @@ -0,0 +1,90 @@ +# Prometheus NPM Quadlet cutover + +## Current state (2026-10-03) + +Nginx Proxy Manager runs as the **rootful** generated +`prometheus-npm.service` on Prometheus. The Quadlet files are +`/etc/containers/systemd/prometheus-npm.container` and +`/etc/containers/systemd/server-web.network`; the image is pinned by digest +in `ansible/inventory/host_vars/prometheus.yml`. The generated service is +wanted by `multi-user.target` and requires the generated network service. +The old `podman-compose-server.service` is inactive and disabled. Its unit +and Compose file remain as a rollback option, not as another active owner. +Do not start both units or run `podman-compose down` while the Quadlet owns +the shared `server_web` network. + +There was **no data copy** in this cutover. The Quadlet reuses the existing +`/opt/npm/data:/data` and `/opt/npm/letsencrypt:/etc/letsencrypt` bind mounts +with the same container name and `server_web` bridge (`10.89.0.0/24`). Ports +80 and 443 remain public; administration port 81 remains bound to +`127.0.0.1`. Gitea stays on Atlas, and NPM remains on Prometheus. The +Prometheus Compose file is retained with the same pinned NPM image for a +controlled fallback. The Quadlet uses `Pull=missing`, not an automatic +floating-tag update. + +## Cutover and recovery boundaries + +The separate `scripts/cutover_prometheus_npm_quadlet.sh` was run **once** in +the approved outage window, after source backup version +`20261003T091009Z` was checksum-verified and pulled to Atlas. Its preflight +required exactly the Compose owner, an inactive generated Quadlet, the +expected image, and the current backup version. The execution held the +backup-export lock, stopped the export timer, stopped and disabled Compose, +started the Quadlet, checked the exact image ID, SQLite database counts, +certificate content, Nginx configuration, and local Gitea/Syncthing HTTPS, +then restarted the timer. Its failure trap would have restarted Compose. +**Do not rerun that forward-cutover script after success**: its preconditions +intentionally reject an active Quadlet. + +A future rollback is a separate outage decision, not an ordinary Ansible run. +First verify a usable recent Atlas backup and stop the export timer. Stop +the Quadlet and verify that its container is gone before allowing Compose +to own the same name, mounts, network, and ports; use the pinned Compose +configuration, then validate NPM/HTTPS and restart the timer. Set +`server_npm_quadlet_cutover: false` only as part of that controlled rollback. +Do not run the two owners concurrently, restore an old NPM database over a +live instance, or delete either bind mount. This reverse procedure has not +been exercised on production; the forward script's in-window rollback path +is not evidence of a later reverse cutover. + +## Verified evidence + +- Immediately after cutover, `prometheus-npm.service` was active with zero + recorded restarts; Compose was inactive/disabled. The generated + `multi-user.target.wants` link and network dependency were present. An + actual reboot has not been performed solely for this test. +- The running image ID matched the prior Compose image. Podman showed the + original two bind mounts, `server_web`, public 80/443, and loopback-only 81. + External HTTPS to Gitea and Syncthing returned 200 with TLS verification + result 0. External access to TCP/81 timed out. +- The first **manual post-cutover** export `20261003T091633Z` succeeded with + the Quadlet as its active owner. The Atlas pull published that version; + its SHA-256 payload check passed. An isolated restore passed NPM SQLite + `quick_check` with ten proxy hosts and six certificate records. Both + Quadlet definitions were present in the tar archive. +- A path/content manifest of all 70 regular Let's Encrypt files and the + path/target manifest of all 12 symlinks in the Atlas archive exactly + matched the live Prometheus tree (aggregate SHA-256 + `ce0965fbd3ff44bb8502ed9f314e0131edd86d822039de115b39f6a2273c2da8`). + The earlier apparent 70-vs-82 count was only a regular-file-versus-symlink + counting difference, not missing certificate data. No certificate key + contents were exposed during comparison. +- The targeted `--tags npm_quadlet` normal Ansible run completed with + `changed=0`, and the backup export timer remained active/enabled. + +The first unattended 02:00 Europe/Rome export and 03:00 Atlas pull **after** +this cutover have not yet occurred. Check their service results and the +published version after the next cycle; the successful manual cycle proves +the new path works but not its next scheduled execution. + +```bash +ANSIBLE_LOCAL_TEMP=/tmp/ansible-local \ +ansible-playbook ansible/site.yml --limit prometheus --tags npm_quadlet --check --diff +sudo systemctl status prometheus-npm.service prometheus-backup-export.timer +sudo systemctl is-active podman-compose-server.service +sudo systemctl is-enabled podman-compose-server.service +``` + +The backup archive includes credentials, certificates, and WireGuard +configuration. Do not publish it or print its contents in diagnostics; see +`docs/prometheus-backup.md` for the restricted pull and restore procedure. diff --git a/scripts/cutover_prometheus_npm_quadlet.sh b/scripts/cutover_prometheus_npm_quadlet.sh new file mode 100644 index 0000000..2e13cd3 --- /dev/null +++ b/scripts/cutover_prometheus_npm_quadlet.sh @@ -0,0 +1,106 @@ +#!/usr/bin/env bash +# Run on Prometheus as root with the exact verified source-export version. +set -Eeuo pipefail + +expected_export=${1:?Pass the verified Prometheus backup export version} +mode=${2:---preflight} +[[ $expected_export =~ ^[0-9]{8}T[0-9]{6}Z$ ]] || exit 2 +[[ $mode == --preflight || $mode == --execute ]] || exit 2 +[[ $EUID -eq 0 ]] || { echo 'Run as root on Prometheus' >&2; exit 2; } + +compose_unit=podman-compose-server.service +quadlet_unit=prometheus-npm.service +backup_timer=prometheus-backup-export.timer +versions=/var/lib/prometheus-backup-export/versions +quadlet_file=/etc/containers/systemd/prometheus-npm.container + +exec 9>/run/lock/prometheus-backup-export.lock +flock -n 9 || { echo 'Backup/export lock is busy' >&2; exit 1; } + +systemctl is-active --quiet "$compose_unit" +if systemctl is-active --quiet "$quadlet_unit"; then + echo 'NPM Quadlet is already active; refusing overlapping cutover' >&2 + exit 1 +fi +[[ $(systemctl show "$quadlet_unit" -p LoadState --value) == loaded ]] +[[ $(systemctl is-enabled "$compose_unit") == enabled ]] +[[ $(readlink "$versions/current") == "$expected_export" ]] +image=$(sed -n 's/^Image=//p' "$quadlet_file") +[[ $image =~ ^docker\.io/jc21/nginx-proxy-manager@sha256:[a-f0-9]{64}$ ]] +podman image exists "$image" +(cd "$versions/current" && sha256sum -c payload.sha256 && tar -tf payload.tar >/dev/null) +curl -fsS --connect-timeout 2 --max-time 5 -o /dev/null http://127.0.0.1:81/ +old_image=$(podman inspect nginx-proxy-manager --format '{{.Image}}') + +data_signature() { + python3 - <<'PY' +import hashlib, os, sqlite3 +db = sqlite3.connect('file:/opt/npm/data/database.sqlite?mode=ro', uri=True) +assert db.execute('pragma quick_check').fetchone()[0] == 'ok' +counts = [db.execute('select count(*) from ' + table).fetchone()[0] + for table in ('proxy_host', 'certificate', 'user')] +db.close() +digest = hashlib.sha256() +for root, dirs, files in os.walk('/opt/npm/letsencrypt'): + dirs.sort() + for name in sorted(files): + path = os.path.join(root, name) + with open(path, 'rb') as stream: + digest.update(path.encode() + b'\0' + stream.read()) +print(*counts, digest.hexdigest()) +PY +} +before=$(data_signature) +if [[ $mode == --preflight ]]; then + echo 'NPM Quadlet cutover preflight passed; no service was changed' + exit 0 +fi + +stopped_old=false +rollback() { + rc=$? + trap - EXIT + if (( rc != 0 )) && "$stopped_old"; then + echo 'NPM Quadlet cutover failed; restoring Compose' >&2 + systemctl stop "$quadlet_unit" || true + systemctl enable "$compose_unit" || true + systemctl start "$compose_unit" || true + systemctl start "$backup_timer" || true + curl -fsS --connect-timeout 2 --max-time 10 -o /dev/null http://127.0.0.1:81/ || true + fi + exit "$rc" +} +trap rollback EXIT + +stopped_old=true +systemctl stop "$backup_timer" +systemctl stop "$compose_unit" +if podman container exists nginx-proxy-manager; then + echo 'Compose left the NPM container behind; refusing duplicate ownership' >&2 + exit 1 +fi +systemctl disable "$compose_unit" +systemctl start "$quadlet_unit" + +ready=false +for _ in {1..60}; do + if curl -fsS --connect-timeout 2 --max-time 3 -o /dev/null http://127.0.0.1:81/; then + ready=true + break + fi + sleep 2 +done +"$ready" +systemctl is-active --quiet "$quadlet_unit" +[[ $(podman inspect nginx-proxy-manager --format '{{.Image}}') == "$old_image" ]] +podman exec nginx-proxy-manager nginx -t +[[ $(data_signature) == "$before" ]] +for hostname in git.fscotto.duckdns.org syncthing.fscotto.duckdns.org; do + status=$(curl -ksS --connect-timeout 3 --max-time 10 \ + --resolve "$hostname:443:127.0.0.1" -o /dev/null -w '%{http_code}' \ + "https://$hostname/") + [[ $status == 200 ]] +done +systemctl start "$backup_timer" +stopped_old=false +echo 'NPM Quadlet cutover passed local application and data checks'