mirror of
https://github.com/fscotto/infra.git
synced 2026-09-27 19:03:47 +00:00
Compare commits
11 Commits
e837b0059b
...
main
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
fa1c8c0b82 | ||
|
|
0b6efc9ad8 | ||
|
|
361ee77d72 | ||
|
|
d4e40d423a | ||
|
|
0a5c2ac1a4 | ||
|
|
48a7f57f7e | ||
|
|
defa98c968 | ||
|
|
21e41f4fc1 | ||
|
|
e10c6694f8 | ||
|
|
4c10af3187 | ||
|
|
3ac732751c |
93
AGENTS.md
93
AGENTS.md
@@ -63,6 +63,14 @@ Ansible-driven personal infrastructure repo for Fedora and Void desktops, Fedora
|
||||
`ansible-playbook ansible/site.yml --limit atlas --tags snapshots,scrub --check --diff`
|
||||
- Atlas encrypted Borg backup:
|
||||
`ansible-playbook ansible/site.yml --limit atlas --tags packages,borg --check --diff`
|
||||
- Atlas Borg progress logging only:
|
||||
`ansible-playbook ansible/site.yml --limit atlas --tags borg_logging --check --diff`
|
||||
- Atlas manual offline USB backup and 45Drives Alerts reminder:
|
||||
`ansible-playbook ansible/site.yml --limit atlas --tags usb_backup,usb_reminder --check --diff`
|
||||
- Atlas pool, disk, capacity, temperature, and job monitoring:
|
||||
`ansible-playbook ansible/site.yml --limit atlas --tags monitoring --check --diff`
|
||||
- Atlas explicit post-restore SELinux relabeling:
|
||||
`ansible-playbook ansible/site.yml --limit atlas --tags restorecon --check -e '{"atlas_restorecon_paths":["/zpool/archive"]}'`
|
||||
- Prometheus/Aegis WireGuard gateway:
|
||||
`ansible-playbook ansible/site.yml --limit prometheus,aegis --tags wireguard --check --diff`
|
||||
- DuckDNS config only: `ansible-playbook ansible/site.yml --limit prometheus --tags duckdns --check --diff`
|
||||
@@ -186,13 +194,41 @@ scheduled retention prune and monthly scrub remain runtime checks.
|
||||
Borg repository check, and temporary-directory restore completed successfully; the restored `Archive`
|
||||
tree matched the live data, and temporary snapshots and mounts were removed. The exported recovery key
|
||||
was copied offline. Daily backup retries and logging, 30 daily, 8 weekly and 12 monthly archives,
|
||||
compaction, and monthly repository checks are enabled.
|
||||
- [ ] Add the UUID-bound offline USB backup with versioned rsync, locking, capacity checks, verification,
|
||||
safe unmounting and a tested restore procedure; never trigger it for an arbitrary USB disk.
|
||||
- [ ] Test restores independently from a ZFS snapshot, Borg, and the offline USB backup before relying on
|
||||
any backup path.
|
||||
- [ ] Add monitoring and alerting for pool health, scrub/resilver, SMART data, temperatures, free space,
|
||||
snapshot/backup capacity growth, and failed maintenance or backup timers.
|
||||
compaction, and monthly repository checks are enabled. Future runs report a ZFS-based estimated
|
||||
percentage, and a post-exit helper handles host-namespace temporary snapshot cleanup. The active
|
||||
run predates the new progress logging and still requires an observed final cleanup result.
|
||||
- [x] Populate `/zpool/archive` with the currently available data so offsite and offline backup tests run
|
||||
against a representative load.
|
||||
- [ ] Run and evaluate Borg against the populated pool: duration, repository capacity, deduplication, and
|
||||
a subsequent incremental archive must be observed before considering the offsite path fully validated.
|
||||
- [x] Add the UUID-bound offline USB backup with versioned rsync, locking, capacity checks, verification,
|
||||
safe unmounting and a tested restore procedure; never trigger it for an arbitrary USB disk. The
|
||||
LUKS/ext4 identities were read-only verified; the manual service and 45Drives Alerts reminder timer were
|
||||
deployed on Atlas. Interactive LUKS unlock is part of the manual service; only the reminder is
|
||||
scheduled for the first Saturday of each month at 10:00 Europe/Rome via the existing 45Drives
|
||||
notifier. A manual test produced an Alerts notification, not an email. The first USB attempt failed
|
||||
on a `security.selinux` xattr and was interrupted; the xattr filter is deployed and the temporary
|
||||
recursive snapshot and open LUKS mapper were cleaned up. A later run reported checksum verification
|
||||
and published the USB version, but failed while removing host-namespace ZFS snapshot mounts. Those
|
||||
exact mounts and snapshots were cleaned up. An `ExecStopPost` helper now removes only the named
|
||||
temporary snapshot after the backup process exits. A new full run checksum-verified and published a
|
||||
USB version; the service ended successfully, the mapper closed, no temporary USB snapshot remained,
|
||||
and the pool was healthy. On 2026-09-25 an independent, read-only USB restore test copied one file from
|
||||
the published `atlas/latest` version into `/var/tmp` and matched contents, owner, mode, size, mtime and
|
||||
POSIX ACL. The temporary copy and mount were removed, the mapper closed, and the pool remained healthy.
|
||||
- [x] Test restores independently from a ZFS snapshot, Borg, and the offline USB backup before relying on
|
||||
any backup path. The earlier Borg temporary-directory restore passed. On 2026-09-25 a separate,
|
||||
read-only ZFS snapshot test restored one file to `/var/tmp`, confirmed matching contents, ownership,
|
||||
mode, mtime and ACL, then removed its temporary copy and on-demand mount. This is a file-level smoke
|
||||
test, not full dataset recovery. An independent USB file restore passed on 2026-09-25 with matching
|
||||
content and metadata; full disaster recovery remains a separate Priority 2 task.
|
||||
- [x] Add monitoring and alerting for pool health, scrub/resilver, SMART data, temperatures, free space,
|
||||
snapshot/local-backup growth, Hetzner Storage Box quota, and failed maintenance or backup timers.
|
||||
The half-hourly Atlas health monitor and systemd final-failure hooks are deployed. A live probe
|
||||
found no issues; the service and timer succeeded, and a labelled 45Drives Alerts test notification
|
||||
was submitted. Alerts are deduplicated; email delivery is not claimed. The Storage Box quota probe
|
||||
runs `df -m` over the dedicated pinned-key SSH identity and does not open the Borg repository.
|
||||
Detailed archive size and deduplication remain part of the pending Borg evaluation.
|
||||
|
||||
### Priority 2 - NAS operability and recovery
|
||||
- [ ] Document and test disaster recovery: rebuild Atlas with Ansible, import the existing pool, restore
|
||||
@@ -215,9 +251,46 @@ scheduled retention prune and monthly scrub remain runtime checks.
|
||||
container paths, and the required Vault database secret.
|
||||
|
||||
### Priority 4 - Optional workflows
|
||||
- [ ] Optionally design iCloud photo ingestion and an Aegis persistent NFS mount as a separate workflow
|
||||
after the storage and backup layers are validated; do not make either a dependency of the Atlas
|
||||
baseline.
|
||||
- [ ] After data protection is validated, move iCloudPD photo ingestion from Aegis to Atlas as a
|
||||
temporary service until Uranus is ready. Plan to store photos in `/zpool/archive/Pictures` and
|
||||
persistent application/MFA state outside `Archive`; validate permissions, SELinux, backups and
|
||||
recovery before cutover. Keep the current Aegis service and Photobook NFS export unchanged until
|
||||
the Atlas workflow is tested, then retire them explicitly if no longer needed.
|
||||
|
||||
## Cerberus Management Node (Deferred)
|
||||
`cerberus` is postponed until the office in the new house is physically set up. It is not an inventory
|
||||
host and this section is a design and implementation backlog, not authorization to provision it early.
|
||||
|
||||
The planned node is a Lenovo ThinkCentre M700 Tiny with an Intel Core i3-6100T, 8 GB RAM, a 256 GB SSD,
|
||||
and native 1 Gbps Ethernet. It will connect to a multi-input KVM switch using a passive DisplayPort-to-HDMI
|
||||
cable, sharing the monitor and peripherals with Ikaros. Fedora Sericea (immutable Fedora with the Sway
|
||||
Wayland compositor) is the intended OS. Cerberus is an isolated management plane: a dedicated Toolbox
|
||||
environment will run Ansible for future `uranus` cluster provisioning. Rootless Podman will host Grafana,
|
||||
Prometheus, and Loki. The 256 GB local SSD is the hot tier retaining metrics and logs for 30 days; scheduled,
|
||||
validated exports of older historical data will use a dedicated Atlas NFS dataset as cold storage.
|
||||
|
||||
### Implementation plan
|
||||
- [ ] Confirm the office, KVM switch, passive DisplayPort-to-HDMI path, shared monitor/peripherals, and native
|
||||
1 Gbps Ethernet are physically operational before adding Cerberus to inventory.
|
||||
- [ ] Install and update Fedora Sericea with Sway; document the immutable-host lifecycle and keep host changes
|
||||
declarative rather than treating the base OS as a mutable workstation.
|
||||
- [ ] Model Cerberus as its own host with independent platform, role, desktop, network, and storage inputs;
|
||||
do not repurpose Ikaros variables or make it a Uranus cluster member.
|
||||
- [ ] Provision an isolated Toolbox-based Ansible controller with the required collections and a reproducible
|
||||
project checkout; define its least-privilege SSH access, known-host handling, and Vault workflow without
|
||||
storing secrets in the image or repository.
|
||||
- [ ] Define the explicit Uranus provisioning workflow from Cerberus, including inventory boundaries,
|
||||
validation-only runs, and separate approval for any destructive cluster operation.
|
||||
- [ ] Design rootless Podman/Quadlet services for Grafana, Prometheus, and Loki, including persistent local
|
||||
state, service ownership, LAN exposure/authentication, resource limits, updates, and backups.
|
||||
- [ ] Size and enforce a 30-day local hot-retention policy for metrics and logs on the 256 GB SSD; validate
|
||||
actual disk growth and alert before capacity exhaustion.
|
||||
- [ ] Create and validate a dedicated Atlas NFS cold-storage dataset and least-privilege export for Cerberus;
|
||||
do not use a broad existing share or couple it to unrelated Atlas application state.
|
||||
- [ ] Implement scheduled, idempotent exports of data older than 30 days to the Atlas NFS cold tier, with
|
||||
locking, capacity checks, integrity verification, retention rules, failure monitoring, and a tested restore.
|
||||
- [ ] Validate management-plane recovery: rebuild Cerberus, restore observability history from Atlas, and
|
||||
confirm that Uranus provisioning can resume without depending on unreproducible local state.
|
||||
|
||||
## Coding Agent Notes
|
||||
- Shared agent definitions and lifecycle flags live in `ai_agents` in `ansible/inventory/group_vars/all.yml`.
|
||||
|
||||
317
README.it.md
317
README.it.md
@@ -95,6 +95,27 @@ Nota sullo stato attuale del playbook principale:
|
||||
- `ansible/site.yml` applica il profilo server Rocky a `prometheus` con DNF, systemd, dotfiles server e firewalld
|
||||
- `ansible/site.yml` applica il profilo NAS Rocky su `atlas` tramite SSH remoto
|
||||
|
||||
## Nodo pianificato e posticipato: Cerberus
|
||||
|
||||
`cerberus` e un nodo di management **posticipato**, in attesa dell'allestimento
|
||||
fisico dell'ufficio nella nuova casa. Non e ancora presente nell'inventory e non
|
||||
esistono ruoli o playbook che lo prendano come target.
|
||||
|
||||
L'hardware previsto e un Lenovo ThinkCentre M700 Tiny (Intel Core i3-6100T,
|
||||
8 GB di RAM e SSD da 256 GB) con Ethernet nativa a 1 Gbps. Condividera monitor
|
||||
e periferiche di Ikaros tramite uno switch KVM a ingressi multipli, usando un
|
||||
cavo passivo DisplayPort-HDMI per il collegamento video. Il sistema operativo
|
||||
previsto e Fedora Sericea, la variante Fedora immutabile con compositor Wayland
|
||||
Sway.
|
||||
|
||||
Cerberus sara un management plane isolato: Ansible verra eseguito in un ambiente
|
||||
Toolbox dedicato per il provisioning del futuro cluster `uranus`, anziche da
|
||||
Ikaros o da un host non gestito. Lo stack di osservabilita rootless Podman
|
||||
eseguira Grafana, Prometheus e Loki. L'SSD locale sara l'hot storage, con
|
||||
metriche e log conservati per 30 giorni; esportazioni programmate trasferiranno
|
||||
i dati storici piu vecchi su un dataset Atlas montato via NFS come cold storage.
|
||||
Il piano di implementazione, con prerequisiti espliciti, e in `AGENTS.md`.
|
||||
|
||||
## Desktop
|
||||
|
||||
Target operativi:
|
||||
@@ -246,90 +267,252 @@ ansible-playbook ansible/site.yml --limit prometheus -e server_username=myuser -
|
||||
|
||||
## NAS
|
||||
|
||||
`atlas` e un NAS Rocky Linux 9 raggiunto tramite SSH. Normalmente il pool ZFS esiste gia e il profilo
|
||||
gestisce solo i dataset figli. Un bootstrap RAIDZ2 una tantum e disponibile solo con conferma esplicita
|
||||
(`atlas_create_pool=true`) e quattro percorsi reali e verificati `/dev/disk/by-id/...` in
|
||||
`atlas_zpool_disks`. Non partiziona, forza, distrugge, esegue rollback o modifica il layout vdev di un
|
||||
pool esistente. I client Linux usano NFSv4, quelli Windows/WSL SMB; entrambi restano limitati alla LAN
|
||||
configurata.
|
||||
`atlas` è un NAS Rocky Linux 9 raggiunto via SSH. Normalmente il pool esiste già e il profilo gestisce
|
||||
solo i dataset figli. La creazione iniziale del RAIDZ2 richiede esplicitamente `atlas_create_pool=true`
|
||||
e quattro percorsi `/dev/disk/by-id/...` verificati in `atlas_zpool_disks`. Il ruolo non partiziona,
|
||||
forza, distrugge, ripristina né modifica il layout vdev di un pool esistente. I client Linux usano NFSv4,
|
||||
quelli Windows/WSL SMB; l'accesso è limitato alla LAN configurata.
|
||||
|
||||
Per il primo avvio fornire `vault_atlas_authorized_ssh_keys`, `vault_atlas_admin_password_hash`,
|
||||
`vault_atlas_samba_password` e `vault_atlas_immich_db_password`. Eseguire il bootstrap tramite
|
||||
l'amministratore esistente:
|
||||
Per il primo avvio servono `vault_atlas_admin_password_hash`, `vault_atlas_samba_password` e
|
||||
`vault_atlas_immich_db_password`; il primo è un hash compatibile con `/etc/shadow`, non una password
|
||||
Cockpit in chiaro. Il bootstrap usa l'amministratore preesistente:
|
||||
|
||||
```bash
|
||||
ansible-playbook ansible/site.yml --limit atlas \
|
||||
-e atlas_connection_username=<existing-admin>
|
||||
```
|
||||
|
||||
`vault_atlas_admin_password_hash` deve essere un hash compatibile con `/etc/shadow`, non una
|
||||
password Cockpit in chiaro. Le esecuzioni successive usano `atlas_admin_username`. Atlas dichiara
|
||||
abilitati storage, condivisioni e regole firewall LAN. Prima della prima applicazione verificare pool e
|
||||
mountpoint esistenti, subnet LAN e zona firewalld attiva. `atlas_manage_media_stack` resta disabilitato
|
||||
finche non saranno validati `/dev/dri`, i percorsi dei container e il segreto del database Immich.
|
||||
Le esecuzioni successive usano `atlas_admin_username`. Storage, condivisioni e firewall LAN sono
|
||||
abilitati; prima dell'applicazione verificare pool, mountpoint, subnet e zona firewalld. La creazione
|
||||
del pool è protetta da un gate esplicito e avviene solo se è assente. Atlas non fa più parte della VPN
|
||||
WireGuard: la vecchia interfaccia è stata ritirata manualmente dopo la verifica del collegamento tra
|
||||
Prometheus e Aegis. Le chiavi SSH autorizzate sono in file separati sotto
|
||||
`~/.ssh/authorized_keys.d/`. `atlas_manage_media_stack` resta disabilitato finché `/dev/dri`, percorsi
|
||||
dei container e segreto del database Immich non sono validati.
|
||||
|
||||
Con la gestione storage attiva, Atlas crea l'intera gerarchia sotto il pool `zpool` esistente o creato esplicitamente:
|
||||
`work`, `archive`, `archive/app_data`, i dataset applicativi separati
|
||||
`archive/app_data/navidrome` e `archive/app_data/syncthing`, `media`, `media/music`,
|
||||
`media/photobook`, `backups`, `backups/services` e `backup_prometheus`. I dataset applicativi e
|
||||
di archivio usano `zstd`; media, Syncthing e backup dei servizi usano `lz4`;
|
||||
`backups/services` mantiene inoltre una `refreservation` di `500G`.
|
||||
Atlas impone SELinux targeted in modo persistente e segnala, senza avviarlo, l’eventuale reboot necessario per attivarlo. Assegna esplicitamente l’interfaccia LAN primaria alla zona firewalld gestita e applica hardening persistente del kernel di rete: rifiuta redirect e source-route, registra i martian, usa reverse-path filtering loose per WireGuard e disabilita il forwarding IPv4. SSH consente solo l’amministratore dichiarato tramite chiave pubblica; root, password, agent e forwarding
|
||||
remoto sono disabilitati, mentre il forwarding locale resta disponibile per tunnel amministrativi privati. SMB3 pubblica `Archive` solo agli account Samba configurati con password in Vault e
|
||||
ammette la LAN configurata su SMB3 cifrato e firmato, esclusivamente su TCP/445. NFSv4 esporta soltanto
|
||||
`media/photobook` all'IP configurato di Aegis su TCP/2049, con `all_squash` verso UID/GID anonimi `1100`.
|
||||
Sotto `zpool` Atlas crea `archive` (SMB), `services/data` con i dataset applicativi
|
||||
`services/data/navidrome` e `services/data/syncthing`, `media`, `media/music`, `media/photobook` e
|
||||
`backup/hosts/prometheus`. Archivio e applicazioni usano `zstd`; media, Syncthing e backup host usano
|
||||
`lz4`. `backup` ha una riserva di `500G` che copre i discendenti. SELinux targeted è persistente;
|
||||
l'eventuale riavvio necessario viene segnalato, non eseguito. Atlas assegna l'interfaccia primaria
|
||||
alla zona firewalld gestita, rifiuta redirect e source route, registra i martian, mantiene il reverse-path
|
||||
filter loose e disabilita il forwarding IPv4. SSH consente soltanto l'amministratore dichiarato con
|
||||
chiave pubblica: root, password, agent forwarding e remote forwarding sono disabilitati, mentre il
|
||||
forwarding locale resta disponibile per i tunnel amministrativi. SMB3 espone `Archive` agli account
|
||||
autorizzati da Vault sulla LAN, solo su TCP/445 con cifratura e firma obbligatorie. NFSv4 espone
|
||||
soltanto `media/photobook` all'IP di Aegis su TCP/2049, con `all_squash` verso UID/GID `1100`.
|
||||
|
||||
L'account di sistema `immich` usa UID/GID `1100`, shell senza login, nessuna appartenenza a `wheel` e
|
||||
i gruppi supplementari `video` e `render`. I Quadlet rootful di Immich Server, ML, cache compatibile
|
||||
Redis, PostgreSQL e NPM condividono una rete Podman. Immich viene eseguito come `1100:1100`; Server e
|
||||
ML ricevono `/dev/dri` e Photobook e montato in sola lettura su `/external/photobook`. NPM pubblica `80` e
|
||||
`443`, mentre l'amministrazione resta vincolata a `127.0.0.1:81` per l'accesso tramite tunnel SSH.
|
||||
L'account di sistema `immich` usa UID/GID `1100`, non ha shell di login né gruppo `wheel` e riceve i
|
||||
gruppi `video` e `render`. Lo stack Immich futuro prevede Quadlet rootful per Server, ML, cache,
|
||||
PostgreSQL e NPM su una rete Podman comune. Immich gira come `1100:1100`, Server e ML ricevono
|
||||
`/dev/dri` e Photobook è montato in sola lettura su `/external/photobook`. NPM pubblica `80` e `443`;
|
||||
l'interfaccia amministrativa resta su `127.0.0.1:81`, raggiungibile via tunnel SSH.
|
||||
|
||||
La fase 1 e limitata ai Quadlet utente rootless di Navidrome e Syncthing su Atlas. E abilitata nella
|
||||
configurazione host di Atlas e puo essere impostata a `false` solo per una sospensione intenzionale. Navidrome ufficiale `0.63.2` usa il database SQLite sotto `/data` e
|
||||
non supporta `ND_DATABASE_URL` ne un backend PostgreSQL esterno. Il servizio obsoleto `navidromedb`
|
||||
e quindi rimosso da Prometheus invece di essere replicato su Atlas. Il ruolo deriva i percorsi dal
|
||||
pool `zpool`, montato in `/zpool`: musica in sola lettura da `/zpool/media/music`, stato
|
||||
applicativo Navidrome e `navidrome.db` in `/zpool/archive/app_data/navidrome` e dati Syncthing in
|
||||
`/zpool/archive/app_data/syncthing`. `profile_atlas` crea questi dataset quando
|
||||
`atlas_manage_storage` e attivo; il ruolo backend verifica i mountpoint esatti prima di avviare i
|
||||
container. Il ruolo backend non crea mai il pool. Il ruolo separato `wireguard_overlay`
|
||||
gestisce `wg0` tra Prometheus (`10.0.0.1`) e Atlas (`10.0.0.2`), genera una sola volta le chiavi
|
||||
private sui rispettivi host e scambia tramite Ansible soltanto quelle pubbliche. Solo Prometheus apre
|
||||
pubblicamente `51820/udp`. Le porte backend sono ammesse esclusivamente nella zona firewalld WireGuard.
|
||||
Atlas ospita temporaneamente Navidrome e Syncthing rootless fino alla sostituzione con Uranus. I
|
||||
servizi sono inizializzati **ex novo**, senza migrare lo stato precedente, rispettivamente sotto
|
||||
`/zpool/services/data/navidrome` e `/zpool/services/data/syncthing`; la musica in
|
||||
`/zpool/media/music` viene popolata separatamente. Sono vincolati all'indirizzo LAN di Atlas
|
||||
(`192.168.178.55`), mai a WireGuard. `wireguard_overlay` collega invece Prometheus (`10.0.0.1`)
|
||||
e Aegis (`10.0.0.2`): le chiavi private restano sui rispettivi host e Ansible scambia solo le pubbliche.
|
||||
Prometheus apre `51820/udp`; Aegis inoltra soltanto il traffico overlay→LAN dichiarato e applica
|
||||
source NAT, evitando interfacce VPN su Atlas/Uranus e route statiche sul router. Navidrome (`4533/tcp`)
|
||||
e la GUI Syncthing (`8384/tcp`) ammettono solo Aegis, mentre le porte native Syncthing sono limitate
|
||||
alla LAN. Dopo la verifica dei servizi, configurare manualmente i Proxy Host NPM verso
|
||||
`http://192.168.178.55:4533` e `http://192.168.178.55:8384`. Il peer Prometheus include la LAN
|
||||
negli `AllowedIPs`; aggiungere la VIP Uranus quando esisterà. Dopo il reload di firewalld, Ansible
|
||||
ricarica le reti Podman rootful di Prometheus per conservare DNS e connettività del proxy.
|
||||
|
||||
`backend_phase1_start_services` resta falso durante il trasferimento dello stato applicativo, quindi
|
||||
la prima esecuzione reale del backend genera i Quadlet senza creare un database Atlas vuoto. Dopo aver
|
||||
arrestato Navidrome su Prometheus, copiare l'intera directory `/opt/navidrome/data/` in
|
||||
`/zpool/archive/app_data/navidrome/`, preservando `navidrome.db` e gli eventuali file SQLite laterali.
|
||||
Impostare quindi questa variabile a vero e rieseguire il ruolo per abilitare e avviare Navidrome e
|
||||
Syncthing. Il playbook non copia e non elimina mai i dati applicativi.
|
||||
|
||||
Validare e generare i servizi Atlas con:
|
||||
Validare il gateway con:
|
||||
|
||||
```bash
|
||||
ANSIBLE_LOCAL_TEMP=/tmp/ansible-local \
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags storage
|
||||
|
||||
ANSIBLE_LOCAL_TEMP=/tmp/ansible-local \
|
||||
ansible-playbook ansible/site.yml --limit prometheus,atlas --tags wireguard
|
||||
|
||||
ANSIBLE_LOCAL_TEMP=/tmp/ansible-local \
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags backend_phase1 --check --diff
|
||||
|
||||
ANSIBLE_LOCAL_TEMP=/tmp/ansible-local \
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags backend_phase1
|
||||
ansible-playbook ansible/site.yml --limit prometheus,aegis --tags wireguard --check --diff
|
||||
```
|
||||
|
||||
Per il cutover, arrestare il vecchio Navidrome prima di copiare la sua directory dati, verificare
|
||||
l'ownership dell'account `admin` su Atlas e confermare la presenza del database SQLite copiato prima
|
||||
di impostare `backend_phase1_start_services: true` in `host_vars/atlas.yml`. Conservare i dati sorgente
|
||||
e il container legacy `navidromedb` fermo finche Navidrome su Atlas e una prova di restore non sono
|
||||
stati validati.
|
||||
La prima esecuzione reale WireGuard deve includere entrambi i peer. Se Aegis ha appena installato il
|
||||
layer `wireguard-tools`, riavviarlo manualmente e rieseguire senza `--check`: il ruolo attende un
|
||||
handshake effettivo.
|
||||
|
||||
Restano da completare retention delle snapshot, topologia Syncthing, validazione WireGuard/firewall,
|
||||
pull di backup da Prometheus, backup cifrati con Borg su una Hetzner Storage Box, backup USB,
|
||||
monitoraggio e test di disaster recovery. Il backlog operativo dettagliato e in `AGENTS.md`.
|
||||
Gli snapshot ZFS ricorsivi coprono l'intero pool: 24 orari al minuto 05, 30 giornalieri alle 00:15,
|
||||
8 settimanali la domenica alle 01:00 e 12 mensili il primo giorno alle 02:00. La retention elimina
|
||||
solo gli snapshot con prefisso gestito `atlas-auto` e non esegue rollback. Lo scrub OpenZFS mensile è
|
||||
previsto la prima domenica alle 03:00; il timer settimanale incompatibile è disabilitato. Il primo
|
||||
snapshot orario ricorsivo è riuscito; la prima pulizia pianificata e il primo scrub schedulato
|
||||
richiedono ancora una verifica a runtime.
|
||||
|
||||
```bash
|
||||
ANSIBLE_LOCAL_TEMP=/tmp/ansible-local \
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags snapshots,scrub --check --diff
|
||||
```
|
||||
|
||||
Il backup Borg cifrato usa il sub-account Hetzner `u660064-sub1`, il repository relativo `./borg-data`
|
||||
e Borg remoto 1.4 su SSH porta 23. La chiave ED25519 del server è fissata; una chiave client dedicata
|
||||
appartiene all'account `borg`, bloccato e senza login, sudo o gruppi supplementari. La chiave privata
|
||||
resta in `/etc/atlas-borg`; la passphrase proviene da `vault_atlas_borg_passphrase` ed è resa in un
|
||||
file `0600`. Solo il wrapper root crea snapshot e mount; avvia il client come `borg` con il minimo
|
||||
accesso temporaneo in lettura, senza concedergli gestione ZFS o sudo.
|
||||
|
||||
Il backup giornaliero parte alle 04:30 con un ritardo casuale fino a 30 minuti. Crea uno snapshot ZFS
|
||||
ricorsivo temporaneo e ricostruisce tutti i dataset sotto `/zpool` in un albero di bind mount in sola
|
||||
lettura, per inserirli in un unico archivio coerente. Il wrapper smonta ricorsivamente l'albero privato;
|
||||
un helper `ExecStopPost` mirato rimuove eventuali mount dello snapshot nel namespace host e lo snapshot
|
||||
temporaneo dopo l'uscita del processo. Borg conserva 30 archivi giornalieri, 8 settimanali e 12
|
||||
mensili, poi compatta il repository. Il controllo completo di metadati e repository si svolge il 15
|
||||
di ogni mese alle 06:00. Le operazioni usano un lock comune, journal e retry systemd limitati. Le
|
||||
nuove esecuzioni riportano al massimo una riga di avanzamento al minuto: percentuale **stimata**,
|
||||
dataset, file elaborati e byte originali/compressi/deduplicati. Il denominatore è la somma dei
|
||||
`logicalreferenced` ZFS dello snapshot, non un totale Borg: può superare il 100% e non comprende
|
||||
retention, compattazione o controlli. Le righe di progresso non riportano i nomi dei file; eventuali
|
||||
warning possono farlo. Seguire il job con `sudo journalctl -fu atlas-borg-backup.service`; modifiche
|
||||
all'helper non cambiano un'esecuzione già avviata.
|
||||
|
||||
Attivazione iniziale esplicita:
|
||||
|
||||
1. Inserire una passphrase unica in `secrets/vault.yml` con `ansible-vault edit`.
|
||||
2. Generare e mostrare solo la chiave pubblica con
|
||||
`ansible-playbook ansible/site.yml --limit atlas --tags borg_key`.
|
||||
3. Installarla nel sub-account Hetzner, poi applicare con
|
||||
`ansible-playbook ansible/site.yml --limit atlas --tags packages,borg`.
|
||||
4. Copiare `secrets/recovery/atlas-borg-repokey.export` su un supporto davvero offline: la copia
|
||||
locale ignorata da Git non è di per sé un backup offline.
|
||||
|
||||
Il ruolo inizializza solo un repository `repokey` assente, non accetta password SSH né host key non
|
||||
fissate e non avvia manualmente il primo backup. Validazione:
|
||||
|
||||
```bash
|
||||
ANSIBLE_LOCAL_TEMP=/tmp/ansible-local \
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags packages,borg --check --diff
|
||||
```
|
||||
|
||||
L'attivazione iniziale è riuscita: backup e controllo del repository, restore completo in una
|
||||
directory temporanea confrontato con l'albero `Archive`, esportazione offline della chiave di recupero
|
||||
e pulizia di snapshot/mount temporanei. Il 2026-09-25 un test separato da snapshot ZFS giornaliero ha
|
||||
copiato un file di `/zpool/archive` in `/var/tmp`, verificando contenuto, proprietario, modalità,
|
||||
mtime e ACL POSIX; copia e mount temporanei sono stati rimossi senza interrompere Borg. Non è un test
|
||||
di ripristino dell'intero dataset.
|
||||
|
||||
Il backup USB offline è distribuito come **servizio solo manuale** (`atlas_manage_usb_backup: true`):
|
||||
Ansible non formatta, sblocca, monta né avvia automaticamente il disco. Il disco esistente è stato
|
||||
verificato in sola lettura il 2026-09-23: UUID LUKS `577b3c43-ea37-4611-81a9-39d555cdfbd4`,
|
||||
UUID ext4 interno `758e2d2e-a427-4797-aad9-39c3a9f17c7e`, mapper `zpool-backup`. All'ispezione
|
||||
era montato in `/mnt/zpool-backup`; il servizio richiede invece che il mapper **non sia montato** prima
|
||||
dell'avvio. Se serve, `systemd-ask-password` chiede interattivamente la passphrase LUKS tramite
|
||||
l'agente di `systemctl start` e la passa direttamente a `cryptsetup`, senza salvarla, esporla negli
|
||||
argomenti o memorizzarla nella cache. Lo script monta il disco privatamente, crea uno snapshot ZFS
|
||||
ricorsivo, copia tutti i dataset in `atlas/snapshots/<timestamp>/` con `rsync --link-dest`, verifica
|
||||
con un dry-run basato sui checksum, aggiorna atomicamente `atlas/latest`, smonta e chiude LUKS. Un
|
||||
errore non sostituisce `latest` né cancella versioni complete precedenti. Borg e USB possono operare
|
||||
contemporaneamente su snapshot distinti, ma la lettura concorrente può ridurre il throughput.
|
||||
|
||||
La copia USB conserva le ACL ma non gli attributi estesi generici, compreso `security.selinux`: la
|
||||
policy della destinazione deve ricreare le etichette dopo un restore. Per un percorso esplicito:
|
||||
|
||||
```bash
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags restorecon \
|
||||
-e '{"atlas_restorecon_paths":["/zpool/archive"]}'
|
||||
```
|
||||
|
||||
Il task accetta solo percorsi sotto la radice del pool Atlas, esegue `restorecon -RFv` solo su quelli
|
||||
indicati ed è altrimenti inattivo; non va lanciato sull'intero pool durante i run ordinari. Le vecchie
|
||||
versioni USB non vengono eliminate automaticamente senza una retention deliberata. Il controllo di
|
||||
capacità include il trasferimento stimato e una riserva libera di 10 GiB. Dopo un backup riuscito,
|
||||
scollegare fisicamente il disco per renderlo davvero offline.
|
||||
|
||||
Validare la configurazione senza avviare il backup e, separatamente, un eventuale relabel pianificato:
|
||||
|
||||
```bash
|
||||
ANSIBLE_LOCAL_TEMP=/tmp/ansible-local \
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags usb_backup,usb_reminder --check --diff
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags restorecon --check \
|
||||
-e '{"atlas_restorecon_paths":["/zpool/archive"]}'
|
||||
```
|
||||
|
||||
Prima dell'avvio manuale smontare in sicurezza `/mnt/zpool-backup`, se ancora montato. Con il mapper
|
||||
chiuso, `sudo systemctl start atlas-usb-backup.service` chiede la passphrase e avvia il backup; né la
|
||||
password LUKS né un keyfile vanno in Ansible. Seguire con
|
||||
`sudo journalctl -fu atlas-usb-backup.service`. **Non esiste un timer di backup USB.** Soltanto
|
||||
`atlas-usb-reminder.timer` è schedulato il primo sabato del mese alle 10:00 `Europe/Rome`: invia un
|
||||
promemoria al notifier 45Drives Houston, senza avviare il backup. Un test manuale ha prodotto una
|
||||
notifica in 45Drives Alerts, **non un'email**; il log conferma l'invio della notifica, non la consegna
|
||||
di posta. Il primo evento pianificato era il 2026-10-03 alle 10:00 CEST. Controllare timer e risultato
|
||||
con `systemctl list-timers atlas-usb-reminder.timer` e in 45Drives Alerts.
|
||||
|
||||
Il primo tentativo USB del 2026-09-23 fallì su `security.selinux` e, dopo l'interruzione, lasciò
|
||||
snapshot e mapper aperti. Applicato il filtro rsync, furono rimossi lo snapshot fallito, il mapper
|
||||
smontato e lo stato failed; non rimase una copia valida di quel tentativo. Un run del 2026-09-24
|
||||
pubblicò una versione verificata ma fallì nella distruzione dello snapshot a causa di mount
|
||||
`.zfs/snapshot` aperti nel namespace host. Dopo la pulizia non forzata, è stato aggiunto un helper
|
||||
`ExecStopPost` mirato e testato con uno snapshot usa-e-getta. Un run successivo del 2026-09-24 ha
|
||||
verificato i checksum, pubblicato la versione ed è terminato con successo: mapper chiuso, nessuno
|
||||
snapshot USB temporaneo e pool sano. Il 2026-09-25 un test di restore indipendente ha aperto il disco
|
||||
in sola lettura, montato ext4 con `ro,noload`, copiato un file di 5.707.945 byte da `atlas/latest` in
|
||||
una directory vuota sotto `/var/tmp` e confrontato contenuto, proprietario, modalità, dimensione,
|
||||
mtime e ACL POSIX. Il test ha rimosso copia e mount temporanei, chiuso LUKS e lasciato il pool sano
|
||||
mentre Borg continuava. È un test su file, non un esercizio completo di disaster recovery.
|
||||
|
||||
Il monitoraggio Atlas è eseguito ogni 30 minuti da `atlas-health-monitor.timer`. Sonde in sola
|
||||
lettura controllano stato/errori del pool e dei vdev, scrub/resilver, SMART dei quattro dischi del
|
||||
pool e dell'NVMe di sistema, temperature dei dischi e CPU, spazio di sistema/pool/snapshot, crescita
|
||||
di `zpool/backup` e quota Hetzner tramite `df -m` via SSH con l'account `borg` e la chiave fissata.
|
||||
La query remota non apre il repository Borg né il suo lock. Gli alert di crescita richiedono una
|
||||
baseline di circa 24 ore. Sono controllati anche attivazione e freschezza dei timer; hook systemd
|
||||
`OnFailure` segnalano errori di snapshot, scrub, Borg, USB, promemoria e monitoraggio. Il monitor non
|
||||
riavvia Borg; avvisa solo se un run supera 14 giorni. Soglie e percorsi stabili dei dischi sono nelle
|
||||
variabili host. Gli avvisi usano 45Drives Houston con deduplicazione; **la consegna email non è stata
|
||||
verificata**. Il controllo live del 2026-09-25 non ha trovato problemi; la notifica di prova è stata
|
||||
inviata e lo Storage Box risultava occupato al 22%. Dimensione dell'archivio Borg e deduplicazione
|
||||
dettagliata richiedono ancora la fine del backup in corso.
|
||||
|
||||
```bash
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags monitoring --check --diff
|
||||
sudo /usr/local/libexec/atlas-health-monitor --dry-run
|
||||
sudo journalctl -u atlas-health-monitor.service -n 100 --no-pager
|
||||
systemctl list-timers atlas-health-monitor.timer
|
||||
```
|
||||
|
||||
`--dry-run` non invia alert e non modifica lo stato del monitor. Un controllo reale si avvia con
|
||||
`sudo systemctl start atlas-health-monitor.service`, senza avviare servizi di backup. Per una prova
|
||||
etichettata di 45Drives Alerts usare
|
||||
`sudo /usr/local/libexec/atlas-health-monitor --test-notification`.
|
||||
|
||||
### Timer systemd di Atlas
|
||||
|
||||
Tutti i nove timer gestiti sono abilitati. Gli orari sono locali ad Atlas (`Europe/Rome`); Borg e
|
||||
monitoraggio aggiungono il ritardo casuale indicato. Tutti hanno `Persistent=true`: un evento perso
|
||||
viene recuperato quando il timer torna attivo.
|
||||
|
||||
| Timer | Pianificazione (`OnCalendar`) | Azione |
|
||||
| --- | --- | --- |
|
||||
| `atlas-zfs-snapshot-hourly.timer` | `*-*-* *:05:00` — ogni ora al minuto 05 | Snapshot ricorsivo orario e retention |
|
||||
| `atlas-zfs-snapshot-daily.timer` | `*-*-* 00:15:00` — ogni giorno alle 00:15 | Snapshot ricorsivo giornaliero e retention |
|
||||
| `atlas-zfs-snapshot-weekly.timer` | `Sun *-*-* 01:00:00` — domenica alle 01:00 | Snapshot ricorsivo settimanale e retention |
|
||||
| `atlas-zfs-snapshot-monthly.timer` | `*-*-01 02:00:00` — primo giorno del mese alle 02:00 | Snapshot ricorsivo mensile e retention |
|
||||
| `zfs-scrub-monthly@zpool.timer` | `Sun *-*-01..07 03:00:00` — prima domenica alle 03:00 | Scrub ZFS |
|
||||
| `atlas-borg-backup.timer` | `*-*-* 04:30:00` — ogni giorno alle 04:30, più 0–30 min casuali | Backup cifrato offsite |
|
||||
| `atlas-borg-check.timer` | `*-*-15 06:00:00` — giorno 15 alle 06:00, più 0–30 min casuali | Controllo repository Borg |
|
||||
| `atlas-usb-reminder.timer` | `Sat *-*-01..07 10:00:00 Europe/Rome` — primo sabato alle 10:00 | Solo promemoria 45Drives Alerts |
|
||||
| `atlas-health-monitor.timer` | `*:0/30` — ogni mezz'ora, più 0–5 min casuali | Controlli di salute in sola lettura |
|
||||
|
||||
`atlas-usb-backup.service` **non ha timer** e va avviato manualmente. Il timer del fornitore
|
||||
`zfs-scrub-weekly@zpool.timer` è disabilitato a favore dello scrub mensile. Il futuro pull del backup
|
||||
Prometheus non ha ancora un timer, perché non è implementato. Durante un backup Borg attivo,
|
||||
`systemctl list-timers` può mostrare `-` per il prossimo evento senza che il timer sia disabilitato.
|
||||
Per vedere la pianificazione corrente: `systemctl list-timers --all` su Atlas.
|
||||
|
||||
Nextcloud è previsto come servizio temporaneo su Atlas prima di Uranus, ma solo dopo la validazione
|
||||
della protezione dei dati: richiede storage applicativo, database e cache separati, segreti Vault,
|
||||
pubblicazione solo tramite NPM e Aegis, procedure di backup, aggiornamento e migrazione. Non
|
||||
distribuirlo prima di completare la checklist di protezione dei dati.
|
||||
|
||||
La destinazione futura per l'importazione foto iCloud è Atlas, non Aegis. Dopo la validazione dei
|
||||
backup, pianificare una migrazione esplicita di iCloudPD con foto sotto `/zpool/archive/Pictures` e
|
||||
stato applicativo/MFA fuori da `Archive`; testare permessi, SELinux, backup e restore prima del
|
||||
cutover. L'attuale iCloudPD su Aegis e l'export NFS Photobook restano configurati fino
|
||||
all'approvazione e alla verifica di questa migrazione separata. Anche il servizio Atlas sarà
|
||||
temporaneo in attesa di Uranus.
|
||||
|
||||
Il pull dei backup di Prometheus, la valutazione delle dimensioni degli archivi Borg e i test completi
|
||||
di disaster recovery restano da fare. Il backlog prioritizzato è in `AGENTS.md`.
|
||||
|
||||
---
|
||||
|
||||
|
||||
183
README.md
183
README.md
@@ -67,6 +67,27 @@ The official ChatGPT desktop RPM is enabled only on `ikaros` and `nymph`. The
|
||||
playbook configures OpenAI's signed RPM repository and imports its pinned RPM
|
||||
signing key before installation; subsequent updates are handled by DNF.
|
||||
|
||||
## Deferred planned node: Cerberus
|
||||
|
||||
`cerberus` is a **postponed** management-plane node, pending the physical setup
|
||||
of the office in the new house. It is not yet an inventory host and no role or
|
||||
playbook targets it.
|
||||
|
||||
The planned hardware is a Lenovo ThinkCentre M700 Tiny (Intel Core i3-6100T,
|
||||
8 GB RAM, and a 256 GB SSD) with native 1 Gbps Ethernet. It will share Ikaros'
|
||||
monitor and peripherals through a multi-input KVM switch, using a passive
|
||||
DisplayPort-to-HDMI cable for its video connection. Fedora Sericea, the
|
||||
immutable Fedora variant with the Sway Wayland compositor, is the intended
|
||||
operating system.
|
||||
|
||||
Cerberus will be an isolated management plane: Ansible will run from a
|
||||
dedicated Toolbox environment to provision the future `uranus` cluster, rather
|
||||
than from Ikaros or an unmanaged host. Its rootless Podman observability stack
|
||||
will run Grafana, Prometheus, and Loki. The local SSD is the hot tier and
|
||||
retains metrics and logs for 30 days; scheduled exports will place older
|
||||
historical data on an NFS-mounted Atlas dataset as the cold tier. The detailed,
|
||||
implementation-gated plan is maintained in `AGENTS.md`.
|
||||
|
||||
## Desktop profiles
|
||||
|
||||
- `ikaros`: stable Fedora Workstation + GNOME desktop.
|
||||
@@ -322,12 +343,22 @@ passphrase, cache, and Borg state. Borg receives its passphrase through a mode `
|
||||
|
||||
The daily backup starts at 04:30 with up to 30 minutes of randomized delay. It creates a temporary,
|
||||
recursive ZFS snapshot and reconstructs every dataset below `/zpool` as a read-only bind-mounted tree,
|
||||
so parent and child datasets enter one consistent Borg archive. Cleanup always removes the temporary
|
||||
mounts and managed snapshot. Only the root wrapper performs snapshot and mount operations; it launches
|
||||
the Borg client as `borg` with temporary read-search capability and no ZFS, sudo, or pool-management
|
||||
privileges. Borg retains 30 daily, 8 weekly, and 12 monthly archives, then compacts the standard
|
||||
so parent and child datasets enter one consistent Borg archive. The wrapper recursively unmounts its
|
||||
private source tree; a narrowly scoped `ExecStopPost` helper removes any remaining host-namespace ZFS
|
||||
snapshot mounts and the named temporary snapshot after the backup process exits. Only the root wrapper
|
||||
performs snapshot and mount operations; it launches the Borg client as `borg` with temporary read-search
|
||||
capability and no ZFS, sudo, or pool-management privileges. Borg retains 30 daily, 8 weekly, and 12
|
||||
monthly archives, then compacts the standard
|
||||
read-write repository. A full metadata and repository check runs as `borg` on the fifteenth day of each
|
||||
month at 06:00. Both operations use a common lock, journal logging, and bounded systemd retries.
|
||||
New backup runs also log the create phase and a compact progress line at most once per minute: an
|
||||
**estimated** percentage, dataset, files processed, and original/compressed/deduplicated bytes. The
|
||||
denominator is the summed ZFS `logicalreferenced` size of the backup's own recursive snapshot, not a
|
||||
Borg-reported total: the estimate can exceed 100% and does not cover retention, compaction, or checks.
|
||||
Progress lines omit individual filenames; warnings may still name affected files.
|
||||
Follow the current run with
|
||||
`sudo journalctl -fu atlas-borg-backup.service` on Atlas; changes to the helper do not alter a run
|
||||
already in progress.
|
||||
|
||||
Initial activation remains explicit:
|
||||
|
||||
@@ -350,14 +381,154 @@ ansible-playbook ansible/site.yml --limit atlas --tags packages,borg --check --d
|
||||
Atlas runtime activation is complete: the initial backup and repository check succeeded, a full restore
|
||||
to a temporary directory was validated against the live `Archive` tree, the recovery-key export was copied
|
||||
to offline storage, and the temporary snapshot and bind mounts were cleaned up.
|
||||
On 2026-09-25 a separate ZFS restore smoke test copied a small file from an automatic daily
|
||||
`zpool/archive` snapshot to `/var/tmp`, then confirmed matching contents, ownership, mode, mtime and
|
||||
POSIX ACL. The temporary copy and on-demand snapshot mount were removed; Borg kept running. This
|
||||
does not validate a full dataset recovery.
|
||||
|
||||
The offline USB backup is deployed as a manual-only service (`atlas_manage_usb_backup: true`):
|
||||
Ansible never formats, unlocks, mounts, backs up to, or schedules the disk. Atlas' existing USB disk was verified
|
||||
read-only on 2026-09-23 as LUKS UUID `577b3c43-ea37-4611-81a9-39d555cdfbd4`, containing ext4 UUID
|
||||
`758e2d2e-a427-4797-aad9-39c3a9f17c7e` through mapper `zpool-backup`. It was mounted at
|
||||
`/mnt/zpool-backup` at inspection time. The service deliberately requires the verified mapper to be
|
||||
**not mounted** before starting. When necessary, `systemd-ask-password` requests the LUKS passphrase
|
||||
through the `systemctl start` password agent; it is piped directly to `cryptsetup` without saving it,
|
||||
passing it as a command argument, or caching it. The service then mounts the disk privately, takes a recursive ZFS snapshot,
|
||||
copies every dataset to a versioned `atlas/snapshots/<timestamp>/` directory using `rsync --link-dest`,
|
||||
verifies the result with a checksum-based dry run, atomically updates `atlas/latest`, unmounts and closes
|
||||
LUKS. A failed run never replaces `latest` or removes an earlier complete version. Borg and the USB
|
||||
backup may run concurrently from separate snapshots; both reading the same pool can reduce throughput.
|
||||
The USB copy preserves ACLs but not generic extended attributes; `security.selinux` is also intentionally
|
||||
excluded because the target SELinux policy must recreate labels during a restore. Do not restore data into
|
||||
service paths without relabeling. After restoring an explicit dataset path, apply its destination policy with:
|
||||
|
||||
```bash
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags restorecon \
|
||||
-e '{"atlas_restorecon_paths":["/zpool/archive"]}'
|
||||
```
|
||||
|
||||
The task accepts only paths below the Atlas pool mount root, runs `restorecon -RFv` only for the paths
|
||||
provided at invocation, and is otherwise a no-op. It must not be used on the whole pool during routine runs.
|
||||
Old USB versions are not pruned automatically, to avoid deleting the only offline
|
||||
copy without an explicitly chosen retention policy; capacity checks include an estimated transfer size
|
||||
and a 10 GiB free-space reserve. The disk must be physically disconnected after a successful backup
|
||||
to make the copy offline.
|
||||
|
||||
To check the USB backup and reminder configuration without starting a backup, run:
|
||||
|
||||
```bash
|
||||
ANSIBLE_LOCAL_TEMP=/tmp/ansible-local \
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags usb_backup,usb_reminder --check --diff
|
||||
```
|
||||
|
||||
To validate a planned, explicit post-restore relabel operation without changing labels, run:
|
||||
|
||||
```bash
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags restorecon --check \
|
||||
-e '{"atlas_restorecon_paths":["/zpool/archive"]}'
|
||||
```
|
||||
|
||||
Before the first **manual** service start, safely unmount the currently mounted
|
||||
`/mnt/zpool-backup`; never run it on an arbitrary mounted disk. Future starts
|
||||
can begin with the mapper closed: `sudo systemctl start atlas-usb-backup.service` prompts for the
|
||||
passphrase interactively and then performs the backup. Neither the LUKS password nor a key file belongs
|
||||
in Ansible. Inspect the run with
|
||||
`sudo journalctl -fu atlas-usb-backup.service`. There is intentionally no timer. Independently test a
|
||||
read-only mount and restore from `atlas/latest` into an empty temporary directory before marking the
|
||||
USB recovery path complete. Only `atlas-usb-reminder.timer` is enabled, for the first Saturday of each
|
||||
month at 10:00 Europe/Rome. Its warning notification uses the existing 45Drives Houston notifier.
|
||||
A manual test confirmed a notification in 45Drives Alerts, **not** an email. The reminder service log
|
||||
reports notification submission, not email delivery; the role does not depend on SMTP/OAuth settings.
|
||||
The reminder never starts the backup. Check its schedule with
|
||||
`systemctl list-timers atlas-usb-reminder.timer` and the result in 45Drives Alerts.
|
||||
The timer was verified active with its first scheduled run at 2026-10-03 10:00 CEST. No email
|
||||
delivery is claimed.
|
||||
The first manual USB attempt on 2026-09-23 did not complete: rsync was denied while removing
|
||||
`security.selinux` on the USB filesystem, then the interrupted service left its recursive
|
||||
`atlas-usb-20260923T185748Z-2469168` snapshot and the `zpool-backup` LUKS mapper open. The
|
||||
rsync xattr filter was deployed afterward. The incomplete USB directory was absent on inspection;
|
||||
the exact failed snapshot was removed, the verified and unmounted mapper closed, and the service
|
||||
failed state cleared. A final check found no remnant snapshot, mount, mapper, or staging directory.
|
||||
The failed attempt was not a valid backup, and no USB restore had been tested at that point.
|
||||
On 2026-09-24 a later run reported a checksum-verified, published USB version and closed the LUKS
|
||||
mapper, but the service failed while destroying its temporary ZFS snapshot: OpenZFS still had
|
||||
on-demand `.zfs/snapshot` mounts open in the host namespace. Those exact temporary snapshots were
|
||||
unmounted normally and removed; no force or rollback was used. The backup service now records its
|
||||
snapshot name and runs a narrowly scoped `ExecStopPost` cleanup after the private backup process
|
||||
exits. The cleanup helper was tested with a disposable recursive snapshot and an active snapshot
|
||||
mount. A complete run on 2026-09-24 later checksum-verified and published a new USB version; the
|
||||
service ended successfully, the LUKS mapper closed, no temporary USB snapshot remained, and the pool
|
||||
was healthy. On 2026-09-25 an independent restore test opened the configured USB disk read-only, mounted
|
||||
ext4 with `ro,noload`, restored a 5,707,945-byte file from the published `atlas/latest` version to an
|
||||
empty `/var/tmp` directory, and matched its content, owner, mode, size, mtime, and POSIX ACL against
|
||||
the USB source. The test removed its temporary copy and mount, closed the LUKS mapper, and left the
|
||||
pool healthy while Borg continued running. This is a file-level recovery smoke test, not a full dataset
|
||||
or disaster-recovery exercise.
|
||||
|
||||
Atlas health monitoring runs every 30 minutes through `atlas-health-monitor.timer`. Its read-only probes
|
||||
check pool/vdev state and errors, scrub/resilver status, four pool disks and the system NVMe via SMART,
|
||||
disk and CPU temperatures, system/pool/snapshot space, local `zpool/backup` growth, and the Hetzner
|
||||
Storage Box quota via `df -m` over the dedicated `borg` account's pinned-key SSH connection. The remote
|
||||
query never opens the Borg repository or its lock. Growth alerts compare against a roughly 24-hour
|
||||
baseline and therefore begin only after enough samples exist. The monitor also checks
|
||||
maintenance/backup timer activation and freshness; systemd `OnFailure` hooks report snapshot,
|
||||
scrub, Borg, USB, reminder, and monitoring services when they enter the failed state. The ongoing
|
||||
initial Borg run is never restarted by the monitor; only a run exceeding 14 days raises a warning.
|
||||
Thresholds and stable disk paths are declared in Atlas host variables. Alerts use the existing 45Drives
|
||||
Houston notifier and repeated issues are deduplicated; **email delivery is not verified**. The
|
||||
2026-09-25 live probe found no issues and a labelled test notification was submitted. The Storage Box
|
||||
reported 22% used. Detailed Borg archive size and deduplication still require the active run to finish.
|
||||
|
||||
```bash
|
||||
ansible-playbook ansible/site.yml --limit atlas --tags monitoring --check --diff
|
||||
sudo /usr/local/libexec/atlas-health-monitor --dry-run
|
||||
sudo journalctl -u atlas-health-monitor.service -n 100 --no-pager
|
||||
systemctl list-timers atlas-health-monitor.timer
|
||||
```
|
||||
|
||||
`--dry-run` sends no alerts and does not change monitor state. A real check is
|
||||
`sudo systemctl start atlas-health-monitor.service`; do not start the backup services merely to test
|
||||
monitoring. For a labelled 45Drives Alerts delivery test, use
|
||||
`sudo /usr/local/libexec/atlas-health-monitor --test-notification`.
|
||||
|
||||
### Atlas systemd timers
|
||||
|
||||
All nine managed timers below are enabled. Times are local to Atlas (`Europe/Rome`); Borg and monitoring
|
||||
add the indicated randomized delay. Every timer has `Persistent=true`, so a missed calendar run is
|
||||
scheduled after the timer becomes active again.
|
||||
|
||||
| Timer | Schedule (`OnCalendar`) | Action |
|
||||
| --- | --- | --- |
|
||||
| `atlas-zfs-snapshot-hourly.timer` | `*-*-* *:05:00` — every hour at :05 | Recursive hourly snapshot and retention |
|
||||
| `atlas-zfs-snapshot-daily.timer` | `*-*-* 00:15:00` — daily at 00:15 | Recursive daily snapshot and retention |
|
||||
| `atlas-zfs-snapshot-weekly.timer` | `Sun *-*-* 01:00:00` — Sunday at 01:00 | Recursive weekly snapshot and retention |
|
||||
| `atlas-zfs-snapshot-monthly.timer` | `*-*-01 02:00:00` — first day of the month at 02:00 | Recursive monthly snapshot and retention |
|
||||
| `zfs-scrub-monthly@zpool.timer` | `Sun *-*-01..07 03:00:00` — first Sunday at 03:00 | ZFS scrub |
|
||||
| `atlas-borg-backup.timer` | `*-*-* 04:30:00` — daily at 04:30, plus 0–30 min random delay | Encrypted offsite backup |
|
||||
| `atlas-borg-check.timer` | `*-*-15 06:00:00` — 15th of the month at 06:00, plus 0–30 min random delay | Borg repository check |
|
||||
| `atlas-usb-reminder.timer` | `Sat *-*-01..07 10:00:00 Europe/Rome` — first Saturday at 10:00 | 45Drives Alerts reminder only |
|
||||
| `atlas-health-monitor.timer` | `*:0/30` — every half-hour, plus 0–5 min random delay | Read-only health checks |
|
||||
|
||||
`atlas-usb-backup.service` has **no timer**: the encrypted USB backup must be started manually.
|
||||
The vendor's `zfs-scrub-weekly@zpool.timer` is intentionally disabled in favor of the monthly scrub.
|
||||
The future Prometheus backup pull has no timer yet because that workflow is not implemented. While a
|
||||
Borg backup is still running, `systemctl list-timers` may show `-` for its next trigger; this does not
|
||||
mean the timer has been disabled. Inspect the current schedule on Atlas with
|
||||
`systemctl list-timers --all`.
|
||||
|
||||
A temporary Nextcloud deployment on Atlas is also planned before Uranus: it requires separately
|
||||
declared persistent application, database, and cache storage, Vault-backed credentials, NPM-only
|
||||
publishing through Aegis, and defined backup, upgrade, and eventual migration procedures. Do not deploy
|
||||
it before the data-protection checklist is complete.
|
||||
|
||||
Prometheus backup pulls, USB backup, monitoring, and disaster-recovery tests remain follow-up work. The
|
||||
prioritized operational backlog is kept in `AGENTS.md`.
|
||||
The desired future iCloud photo-ingestion host is Atlas, not Aegis. After data-protection validation,
|
||||
plan an explicit iCloudPD migration with photos under `/zpool/archive/Pictures` and application/MFA
|
||||
state outside `Archive`, then test permissions, SELinux, backups and recovery before cutting over.
|
||||
The current Aegis iCloudPD service and Atlas Photobook NFS export remain configured until that
|
||||
separate migration is approved and validated; the eventual Atlas service is temporary until Uranus.
|
||||
|
||||
Prometheus backup pulls, Borg archive-size evaluation, and full disaster-recovery tests remain follow-up
|
||||
work. The prioritized operational backlog is kept in `AGENTS.md`.
|
||||
|
||||
## How layering works
|
||||
|
||||
|
||||
@@ -9,6 +9,7 @@ fedora_packages_base:
|
||||
- jq
|
||||
- make
|
||||
- nodejs
|
||||
- openssl
|
||||
- ripgrep
|
||||
|
||||
fedora_manage_docker_repo: true
|
||||
|
||||
@@ -12,9 +12,10 @@ workstation_dev_wsl_packages:
|
||||
- python3-pip
|
||||
- tmux
|
||||
|
||||
# Java 11 and Maven are managed by Mise on this Fedora WSL profile. Keep their
|
||||
# versions pinned; update them deliberately.
|
||||
# Java 11, Java 25 and Maven are managed by Mise on this Fedora WSL profile.
|
||||
# Keep their versions pinned; update them deliberately.
|
||||
workstation_mise_java_version: temurin-11.0.31+11
|
||||
workstation_mise_java_25_version: 25.0.2
|
||||
workstation_mise_maven_version: 3.9.16
|
||||
|
||||
workstation_is_wsl: true
|
||||
|
||||
@@ -83,8 +83,55 @@ atlas_borg_randomized_delay: 30m
|
||||
atlas_borg_keep_daily: 30
|
||||
atlas_borg_keep_weekly: 8
|
||||
atlas_borg_keep_monthly: 12
|
||||
atlas_manage_usb_backup: true
|
||||
# Read-only lsblk verification on Atlas, 2026-09-23. Never store the LUKS password here.
|
||||
atlas_usb_backup_luks_uuid: 577b3c43-ea37-4611-81a9-39d555cdfbd4
|
||||
atlas_usb_backup_fs_uuid: 758e2d2e-a427-4797-aad9-39c3a9f17c7e
|
||||
atlas_usb_backup_mapper_name: zpool-backup
|
||||
atlas_manage_usb_reminder: true
|
||||
atlas_usb_reminder_calendar: "Sat *-*-01..07 10:00:00 Europe/Rome"
|
||||
atlas_manage_monitoring: true
|
||||
# Physical pool disks and the system NVMe; the disconnected USB disk is intentionally excluded.
|
||||
atlas_monitor_smart_devices:
|
||||
- { name: pool-1, path: "{{ atlas_zpool_disks[0] }}", warning_c: 50, critical_c: 55 }
|
||||
- { name: pool-2, path: "{{ atlas_zpool_disks[1] }}", warning_c: 50, critical_c: 55 }
|
||||
- { name: pool-3, path: "{{ atlas_zpool_disks[2] }}", warning_c: 50, critical_c: 55 }
|
||||
- { name: pool-4, path: "{{ atlas_zpool_disks[3] }}", warning_c: 50, critical_c: 55 }
|
||||
- name: system-nvme
|
||||
path: /dev/disk/by-id/nvme-Patriot_M.2_P320_256GB_P320ADB26011606111
|
||||
warning_c: 70
|
||||
critical_c: 85
|
||||
atlas_monitor_timers:
|
||||
- { name: atlas-zfs-snapshot-hourly.timer, max_age_hours: 3 }
|
||||
- { name: atlas-zfs-snapshot-daily.timer, max_age_hours: 36 }
|
||||
- { name: atlas-zfs-snapshot-weekly.timer, max_age_hours: 216 }
|
||||
- { name: atlas-zfs-snapshot-monthly.timer, max_age_hours: 960 }
|
||||
- { name: zfs-scrub-monthly@zpool.timer, max_age_hours: 960 }
|
||||
- { name: atlas-borg-backup.timer, max_age_hours: 48 }
|
||||
- { name: atlas-borg-check.timer, max_age_hours: 960 }
|
||||
# The first manual USB reminder is not due until October; activation is checked, not age.
|
||||
- { name: atlas-usb-reminder.timer, max_age_hours: 0 }
|
||||
atlas_monitor_failure_units:
|
||||
- atlas-zfs-snapshot@.service
|
||||
- zfs-scrub@zpool.service
|
||||
- atlas-borg-backup.service
|
||||
- atlas-borg-check.service
|
||||
- atlas-usb-backup.service
|
||||
- atlas-usb-reminder.service
|
||||
- atlas-health-monitor.service
|
||||
atlas_monitor_remote_capacity:
|
||||
user: "{{ atlas_borg_repository_user }}"
|
||||
host: "{{ atlas_borg_repository_host }}"
|
||||
run_as: "{{ atlas_borg_username }}"
|
||||
ssh_wrapper: "{{ atlas_borg_ssh_wrapper_path }}"
|
||||
warning_percent: 80
|
||||
critical_percent: 90
|
||||
growth_warning_gib_day: 500
|
||||
atlas_manage_sharing: true
|
||||
atlas_manage_media_stack: false
|
||||
# Planned after data-protection validation: move iCloudPD photo ingestion from
|
||||
# Aegis to Atlas, with photos under /zpool/archive/Pictures and persistent
|
||||
# application/MFA state outside Archive. Do not deploy or cut over yet.
|
||||
|
||||
# WireGuard is retired on Atlas. These rootless services are a temporary home
|
||||
# until Uranus replaces them.
|
||||
@@ -103,10 +150,17 @@ rocky_podman_packages:
|
||||
|
||||
host_packages:
|
||||
- cockpit
|
||||
- cockpit-podman
|
||||
- cockpit-storaged
|
||||
- realmd
|
||||
- pcp
|
||||
- python3-pcp
|
||||
- cryptsetup
|
||||
- nfs-utils
|
||||
- policycoreutils
|
||||
- policycoreutils-python-utils
|
||||
- python3-libselinux
|
||||
- setroubleshoot-server
|
||||
- samba
|
||||
- samba-client
|
||||
- samba-common-tools
|
||||
@@ -144,4 +198,5 @@ atlas_firewalld_rich_rules:
|
||||
host_enabled_services:
|
||||
- sshd
|
||||
- cockpit.socket
|
||||
- pmlogger.service
|
||||
- zfs.target
|
||||
|
||||
@@ -32,6 +32,12 @@ host_packages:
|
||||
- cockpit
|
||||
- cockpit-navigator
|
||||
- cockpit-podman
|
||||
- cockpit-storaged
|
||||
- realmd
|
||||
- pcp
|
||||
- python3-pcp
|
||||
- setroubleshoot-server
|
||||
|
||||
host_enabled_services:
|
||||
- cockpit.socket
|
||||
- pmlogger.service
|
||||
|
||||
@@ -97,6 +97,40 @@ atlas_borg_cache_dir: /var/cache/atlas-borg
|
||||
atlas_borg_lock_path: /var/lib/atlas-borg/backup.lock
|
||||
atlas_borg_recovery_export_path: "{{ playbook_dir }}/../secrets/recovery/atlas-borg-repokey.export"
|
||||
|
||||
# Manual-only offline backup. No USB device is formatted or mounted by Ansible.
|
||||
atlas_manage_usb_backup: false
|
||||
atlas_usb_backup_luks_uuid: ""
|
||||
atlas_usb_backup_fs_uuid: ""
|
||||
atlas_usb_backup_mapper_name: atlas-usb-backup
|
||||
atlas_usb_backup_min_free_bytes: 10737418240
|
||||
atlas_usb_backup_snapshot_prefix: atlas-usb
|
||||
atlas_manage_usb_reminder: false
|
||||
atlas_usb_reminder_calendar: ""
|
||||
atlas_usb_reminder_notifier: /opt/45drives/houston/houston-notify
|
||||
|
||||
# Read-only health probes and 45Drives Alerts; disabled outside Atlas host vars.
|
||||
atlas_manage_monitoring: false
|
||||
atlas_monitor_calendar: "*:0/30"
|
||||
atlas_monitor_notifier: "{{ atlas_usb_reminder_notifier }}"
|
||||
atlas_monitor_smart_devices: []
|
||||
atlas_monitor_timers: []
|
||||
atlas_monitor_failure_units: []
|
||||
atlas_monitor_remote_capacity: {}
|
||||
atlas_monitor_pool_warning_percent: 80
|
||||
atlas_monitor_pool_critical_percent: 90
|
||||
atlas_monitor_root_warning_percent: 80
|
||||
atlas_monitor_root_critical_percent: 90
|
||||
atlas_monitor_snapshot_warning_percent: 10
|
||||
atlas_monitor_snapshot_critical_percent: 20
|
||||
atlas_monitor_snapshot_growth_warning_gib_day: 100
|
||||
atlas_monitor_backup_growth_warning_gib_day: 100
|
||||
atlas_monitor_cpu_warning_c: 85
|
||||
atlas_monitor_cpu_critical_c: 95
|
||||
atlas_monitor_borg_max_runtime_days: 14
|
||||
|
||||
# Explicit post-restore relabeling only; never relabel datasets during ordinary runs.
|
||||
atlas_restorecon_paths: []
|
||||
|
||||
atlas_archive_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_archive }}"
|
||||
atlas_services_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_services }}"
|
||||
atlas_app_data_mountpoint: "{{ atlas_mount_root }}/{{ atlas_zfs_dataset_app_data }}"
|
||||
|
||||
59
ansible/roles/profile_atlas/files/atlas-borg-progress.py
Normal file
59
ansible/roles/profile_atlas/files/atlas-borg-progress.py
Normal file
@@ -0,0 +1,59 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Turn Borg's JSON progress stream into bounded, readable journal entries."""
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import sys
|
||||
import time
|
||||
|
||||
|
||||
def size(value):
|
||||
if not isinstance(value, (int, float)):
|
||||
return "unknown"
|
||||
return f"{value / (1024 ** 3):.2f} GiB"
|
||||
|
||||
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--estimated-total-bytes", type=int, required=True)
|
||||
args = parser.parse_args()
|
||||
if args.estimated_total_bytes <= 0:
|
||||
parser.error("estimated total must be positive")
|
||||
|
||||
last_progress = 0.0
|
||||
for line in sys.stdin:
|
||||
try:
|
||||
event = json.loads(line)
|
||||
except json.JSONDecodeError:
|
||||
print(line.rstrip(), flush=True)
|
||||
continue
|
||||
|
||||
kind = event.get("type")
|
||||
if kind == "archive_progress":
|
||||
now = time.monotonic()
|
||||
if now - last_progress < 60 and not event.get("finished"):
|
||||
continue
|
||||
path = event.get("path") or ""
|
||||
parts = path.split("/")
|
||||
dataset = parts[1] if len(parts) > 1 and parts[0] == "source" else "unknown"
|
||||
original_size = event.get("original_size")
|
||||
if isinstance(original_size, (int, float)) and original_size >= 0:
|
||||
percent = original_size / args.estimated_total_bytes * 100
|
||||
estimated_progress = (
|
||||
f"{percent:.1f}%" if percent < 100 else ">=100% (ZFS estimate exceeded)"
|
||||
)
|
||||
else:
|
||||
estimated_progress = "unknown"
|
||||
print(
|
||||
"Borg create progress: "
|
||||
f"estimated={estimated_progress} dataset={dataset} "
|
||||
f"files={event.get('nfiles', 'unknown')} "
|
||||
f"original={size(original_size)} "
|
||||
f"compressed={size(event.get('compressed_size'))} "
|
||||
f"deduplicated={size(event.get('deduplicated_size'))}",
|
||||
flush=True,
|
||||
)
|
||||
last_progress = now
|
||||
elif kind == "log_message":
|
||||
print(f"Borg {event.get('levelname', 'INFO')}: {event.get('message', '')}", flush=True)
|
||||
elif kind == "progress_message" and event.get("message"):
|
||||
print(f"Borg: {event['message']}", flush=True)
|
||||
396
ansible/roles/profile_atlas/files/atlas-health-monitor.py
Normal file
396
ansible/roles/profile_atlas/files/atlas-health-monitor.py
Normal file
@@ -0,0 +1,396 @@
|
||||
#!/usr/bin/python3
|
||||
"""Read-only Atlas health probes with deduplicated 45Drives Alerts."""
|
||||
|
||||
import argparse
|
||||
import fcntl
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
import time
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
CONFIG_PATH = Path("/etc/atlas-health-monitor.json")
|
||||
STATE_DIR = Path("/var/lib/atlas-health-monitor")
|
||||
STATE_PATH = STATE_DIR / "state.json"
|
||||
GIB = 1024**3
|
||||
|
||||
|
||||
def run(*argv, timeout=40):
|
||||
return subprocess.run(argv, capture_output=True, text=True, timeout=timeout, check=False)
|
||||
|
||||
|
||||
def issue(issues, key, severity, message):
|
||||
issues[key] = {"severity": severity, "message": message}
|
||||
|
||||
|
||||
def notify(config, event, severity, subject, message):
|
||||
now = datetime.now(timezone.utc)
|
||||
payload = {
|
||||
"timestamp": now.isoformat(timespec="seconds"),
|
||||
"unixtime": int(now.timestamp()),
|
||||
"event": event,
|
||||
"severity": severity,
|
||||
"subject": subject,
|
||||
"email_message": message,
|
||||
}
|
||||
result = run(config["notifier"], json.dumps(payload, ensure_ascii=False), timeout=30)
|
||||
if result.returncode:
|
||||
raise RuntimeError(f"45Drives notifier exited {result.returncode}: {result.stderr.strip()}")
|
||||
|
||||
|
||||
def parse_fields(text):
|
||||
return dict(line.split("=", 1) for line in text.splitlines() if "=" in line)
|
||||
|
||||
|
||||
def systemd_fields(unit, *properties):
|
||||
result = run("systemctl", "show", unit, *(f"-p{item}" for item in properties))
|
||||
if result.returncode:
|
||||
raise RuntimeError(f"systemctl show {unit} exited {result.returncode}")
|
||||
return parse_fields(result.stdout)
|
||||
|
||||
|
||||
def unix_time(text):
|
||||
if not text or text == "n/a":
|
||||
return None
|
||||
result = run("date", "-d", text, "+%s")
|
||||
if result.returncode:
|
||||
raise ValueError(f"Cannot parse systemd timestamp: {text}")
|
||||
return int(result.stdout.strip())
|
||||
|
||||
|
||||
def check_pool(config, issues, measurements):
|
||||
pool = config["pool"]
|
||||
listing = run("zpool", "list", "-H", "-p", "-o", "size,alloc,capacity,health", pool)
|
||||
if listing.returncode:
|
||||
issue(issues, "pool.probe", "critical", f"Cannot query ZFS pool {pool}")
|
||||
return
|
||||
try:
|
||||
size, alloc, capacity, health = listing.stdout.strip().split("\t")
|
||||
size, alloc, capacity = int(size), int(alloc), int(capacity)
|
||||
except (ValueError, TypeError):
|
||||
issue(issues, "pool.probe", "critical", "Invalid ZFS pool capacity response")
|
||||
return
|
||||
measurements.update(pool_size_bytes=size, pool_alloc_bytes=alloc, pool_capacity_percent=capacity)
|
||||
if health != "ONLINE":
|
||||
issue(issues, "pool.health", "critical", f"ZFS pool {pool} state is {health}")
|
||||
if capacity >= config["pool_critical_percent"]:
|
||||
issue(issues, "pool.capacity", "critical", f"ZFS pool {pool} is {capacity}% full")
|
||||
elif capacity >= config["pool_warning_percent"]:
|
||||
issue(issues, "pool.capacity", "warning", f"ZFS pool {pool} is {capacity}% full")
|
||||
|
||||
status = run("zpool", "status", "-P", pool)
|
||||
if status.returncode:
|
||||
issue(issues, "pool.status", "critical", f"Cannot query detailed ZFS status for {pool}")
|
||||
return
|
||||
bad_vdevs = []
|
||||
for line in status.stdout.splitlines():
|
||||
match = re.match(r"^\s*(\S+)\s+(ONLINE|DEGRADED|FAULTED|OFFLINE|UNAVAIL|REMOVED)\s+(\d+)\s+(\d+)\s+(\d+)", line)
|
||||
if match:
|
||||
name, state, reads, writes, checksums = match.groups()
|
||||
if state != "ONLINE" or any(int(value) for value in (reads, writes, checksums)):
|
||||
bad_vdevs.append(f"{name}: {state}, READ={reads}, WRITE={writes}, CKSUM={checksums}")
|
||||
if bad_vdevs:
|
||||
issue(issues, "pool.vdevs", "critical", "ZFS vdev errors: " + "; ".join(bad_vdevs))
|
||||
errors = re.search(r"^errors:\s*(.*)$", status.stdout, re.MULTILINE)
|
||||
if not errors or errors.group(1).strip() != "No known data errors":
|
||||
issue(issues, "pool.data_errors", "critical", "ZFS status reports data errors; inspect zpool status -v")
|
||||
if re.search(r"^\s*scan:\s*resilver in progress", status.stdout, re.MULTILINE | re.IGNORECASE):
|
||||
issue(issues, "pool.resilver", "warning", "ZFS resilver is in progress; inspect zpool status")
|
||||
scan = re.search(r"^\s*scan:\s*(.*)$", status.stdout, re.MULTILINE)
|
||||
if scan and re.search(r"\bwith [1-9][0-9]* errors\b", scan.group(1)):
|
||||
issue(issues, "pool.scan_errors", "critical", f"ZFS scan reported errors: {scan.group(1)}")
|
||||
|
||||
|
||||
def check_capacity(config, issues, measurements):
|
||||
pool = config["pool"]
|
||||
listing = run("zfs", "list", "-H", "-p", "-o", "name,usedbysnapshots", "-r", pool)
|
||||
if listing.returncode:
|
||||
issue(issues, "snapshot.probe", "warning", "Cannot query ZFS snapshot space")
|
||||
else:
|
||||
try:
|
||||
snapshots = sum(int(line.split("\t")[1]) for line in listing.stdout.splitlines())
|
||||
measurements["snapshots_bytes"] = snapshots
|
||||
size = measurements.get("pool_size_bytes")
|
||||
if size:
|
||||
percent = snapshots * 100 // size
|
||||
measurements["snapshots_percent"] = percent
|
||||
if percent >= config["snapshot_critical_percent"]:
|
||||
issue(issues, "snapshot.capacity", "critical", f"Snapshots use {percent}% of pool size")
|
||||
elif percent >= config["snapshot_warning_percent"]:
|
||||
issue(issues, "snapshot.capacity", "warning", f"Snapshots use {percent}% of pool size")
|
||||
except (ValueError, IndexError):
|
||||
issue(issues, "snapshot.probe", "warning", "Invalid ZFS snapshot-space response")
|
||||
backup = run("zfs", "list", "-H", "-p", "-o", "used", config["backup_dataset"])
|
||||
if backup.returncode:
|
||||
issue(issues, "backup.capacity_probe", "warning", "Cannot query local backup dataset space")
|
||||
else:
|
||||
try:
|
||||
measurements["backup_bytes"] = int(backup.stdout.strip())
|
||||
except ValueError:
|
||||
issue(issues, "backup.capacity_probe", "warning", "Invalid local backup space response")
|
||||
|
||||
try:
|
||||
filesystem = os.statvfs("/")
|
||||
total = filesystem.f_blocks * filesystem.f_frsize
|
||||
available = filesystem.f_bavail * filesystem.f_frsize
|
||||
used_percent = (total - available) * 100 // total
|
||||
measurements["root_capacity_percent"] = used_percent
|
||||
if used_percent >= config["root_critical_percent"]:
|
||||
issue(issues, "root.capacity", "critical", f"Atlas system filesystem is {used_percent}% full")
|
||||
elif used_percent >= config["root_warning_percent"]:
|
||||
issue(issues, "root.capacity", "warning", f"Atlas system filesystem is {used_percent}% full")
|
||||
except (OSError, ZeroDivisionError):
|
||||
issue(issues, "root.capacity_probe", "warning", "Cannot query Atlas system filesystem space")
|
||||
|
||||
|
||||
def check_remote_capacity(config, issues, measurements):
|
||||
"""Query only the Storage Box quota; do not open or inspect the Borg repository."""
|
||||
remote = config["remote_capacity"]
|
||||
try:
|
||||
result = run("runuser", "-u", remote["run_as"], "--", remote["ssh_wrapper"],
|
||||
f"{remote['user']}@{remote['host']}", "df", "-m", timeout=65)
|
||||
if result.returncode:
|
||||
raise ValueError(f"SSH df exited {result.returncode}")
|
||||
lines = result.stdout.strip().splitlines()
|
||||
if len(lines) != 2:
|
||||
raise ValueError("Unexpected Storage Box df output")
|
||||
fields = lines[1].split()
|
||||
if len(fields) < 5:
|
||||
raise ValueError("Incomplete Storage Box df output")
|
||||
total_mib, used_mib, available_mib = (int(value) for value in fields[1:4])
|
||||
percent = int(fields[4].rstrip("%"))
|
||||
if total_mib <= 0 or not 0 <= percent <= 100 or available_mib < 0:
|
||||
raise ValueError("Invalid Storage Box quota values")
|
||||
except (OSError, ValueError, subprocess.TimeoutExpired):
|
||||
issue(issues, "remote.capacity_probe", "warning", "Cannot query Hetzner Storage Box quota via pinned-key SSH")
|
||||
return
|
||||
measurements.update(remote_capacity_percent=percent, remote_bytes=used_mib * 1024**2,
|
||||
remote_available_bytes=available_mib * 1024**2)
|
||||
if percent >= remote["critical_percent"]:
|
||||
issue(issues, "remote.capacity", "critical", f"Hetzner Storage Box quota is {percent}% full")
|
||||
elif percent >= remote["warning_percent"]:
|
||||
issue(issues, "remote.capacity", "warning", f"Hetzner Storage Box quota is {percent}% full")
|
||||
|
||||
|
||||
def check_smart(config, issues, measurements):
|
||||
for device in config["smart_devices"]:
|
||||
name, path = device["name"], device["path"]
|
||||
try:
|
||||
result = run("smartctl", "-j", "-a", path, timeout=60)
|
||||
data = json.loads(result.stdout)
|
||||
status = int(data.get("smartctl", {}).get("exit_status", result.returncode))
|
||||
except (subprocess.TimeoutExpired, json.JSONDecodeError, ValueError) as exc:
|
||||
issue(issues, f"smart.{name}.probe", "critical", f"SMART probe failed for {name}: {type(exc).__name__}")
|
||||
continue
|
||||
if status:
|
||||
severity = "critical" if status & 0b00001111 else "warning"
|
||||
issue(issues, f"smart.{name}.status", severity, f"SMART reported exit status {status} for {name}")
|
||||
passed = data.get("smart_status", {}).get("passed")
|
||||
if passed is False:
|
||||
issue(issues, f"smart.{name}.health", "critical", f"SMART self-assessment failed for {name}")
|
||||
elif passed is None:
|
||||
issue(issues, f"smart.{name}.health", "warning", f"SMART self-assessment unavailable for {name}")
|
||||
temperature = data.get("temperature", {}).get("current")
|
||||
if isinstance(temperature, (int, float)):
|
||||
measurements[f"smart_{name}_c"] = temperature
|
||||
if temperature >= device["critical_c"]:
|
||||
issue(issues, f"smart.{name}.temperature", "critical", f"{name} temperature is {temperature} C")
|
||||
elif temperature >= device["warning_c"]:
|
||||
issue(issues, f"smart.{name}.temperature", "warning", f"{name} temperature is {temperature} C")
|
||||
else:
|
||||
issue(issues, f"smart.{name}.temperature", "warning", f"Temperature unavailable for {name}")
|
||||
for attribute in data.get("ata_smart_attributes", {}).get("table", []):
|
||||
attribute_id = attribute.get("id")
|
||||
if attribute_id in (5, 187, 197, 198):
|
||||
raw = attribute.get("raw", {}).get("value", 0)
|
||||
if isinstance(raw, int) and raw > 0:
|
||||
severity = "critical" if attribute_id in (197, 198) else "warning"
|
||||
issue(issues, f"smart.{name}.ata_{attribute_id}", severity,
|
||||
f"{name} SMART attribute {attribute_id} raw count is {raw}")
|
||||
nvme = data.get("nvme_smart_health_information_log", {})
|
||||
if isinstance(nvme, dict):
|
||||
if int(nvme.get("critical_warning", 0)):
|
||||
issue(issues, f"smart.{name}.nvme_warning", "critical", f"{name} NVMe critical warning is nonzero")
|
||||
if int(nvme.get("media_errors", 0)):
|
||||
issue(issues, f"smart.{name}.nvme_media", "critical", f"{name} NVMe media errors are nonzero")
|
||||
|
||||
|
||||
def check_cpu(config, issues, measurements):
|
||||
sensors = []
|
||||
for hwmon in Path("/sys/class/hwmon").glob("hwmon*"):
|
||||
try:
|
||||
if (hwmon / "name").read_text().strip() != "coretemp":
|
||||
continue
|
||||
sensors.extend(int(path.read_text().strip()) / 1000 for path in hwmon.glob("temp*_input"))
|
||||
except (OSError, ValueError):
|
||||
continue
|
||||
if not sensors:
|
||||
issue(issues, "cpu.temperature_probe", "warning", "CPU temperature sensors are unavailable")
|
||||
return
|
||||
hottest = max(sensors)
|
||||
measurements["cpu_max_c"] = hottest
|
||||
if hottest >= config["cpu_critical_c"]:
|
||||
issue(issues, "cpu.temperature", "critical", f"CPU temperature is {hottest:g} C")
|
||||
elif hottest >= config["cpu_warning_c"]:
|
||||
issue(issues, "cpu.temperature", "warning", f"CPU temperature is {hottest:g} C")
|
||||
|
||||
|
||||
def check_jobs(config, issues, measurements, now):
|
||||
for timer in config["timers"]:
|
||||
name = timer["name"]
|
||||
try:
|
||||
fields = systemd_fields(name, "ActiveState", "UnitFileState", "LastTriggerUSec", "ActiveEnterTimestamp")
|
||||
if fields.get("ActiveState") != "active" or fields.get("UnitFileState") != "enabled":
|
||||
issue(issues, f"timer.{name}", "critical", f"Timer {name} is not active and enabled")
|
||||
max_age = int(timer["max_age_hours"]) * 3600
|
||||
if max_age:
|
||||
last = unix_time(fields.get("LastTriggerUSec"))
|
||||
if last is None:
|
||||
last = unix_time(fields.get("ActiveEnterTimestamp"))
|
||||
if last is not None and now - last > max_age:
|
||||
issue(issues, f"timer.{name}.stale", "warning",
|
||||
f"Timer {name} has not fired in {int((now-last)/3600)} hours")
|
||||
except (RuntimeError, ValueError, subprocess.TimeoutExpired):
|
||||
issue(issues, f"timer.{name}.probe", "warning", f"Cannot query timer {name}")
|
||||
for unit in config["failure_units"]:
|
||||
if unit.endswith("@.service"):
|
||||
continue
|
||||
try:
|
||||
fields = systemd_fields(unit, "ActiveState", "Result", "ExecMainStartTimestamp")
|
||||
state = fields.get("ActiveState")
|
||||
if state == "failed" or (state == "inactive" and fields.get("Result") not in (None, "", "success")):
|
||||
issue(issues, f"service.{unit}", "critical", f"Service {unit} failed: {fields.get('Result')}")
|
||||
if unit == "atlas-borg-backup.service" and fields.get("ActiveState") == "activating":
|
||||
started = unix_time(fields.get("ExecMainStartTimestamp"))
|
||||
if started is not None and now - started > config["borg_max_runtime_days"] * 86400:
|
||||
issue(issues, "backup.borg_long_running", "warning",
|
||||
"Borg has run longer than its configured limit")
|
||||
except (RuntimeError, ValueError, subprocess.TimeoutExpired):
|
||||
issue(issues, f"service.{unit}.probe", "warning", f"Cannot query service {unit}")
|
||||
|
||||
|
||||
def check_growth(config, issues, measurements, samples, now):
|
||||
previous = [sample for sample in samples if 20 * 3600 <= now - sample.get("time", now) <= 48 * 3600]
|
||||
if previous:
|
||||
baseline = min(previous, key=lambda sample: abs(now - sample["time"] - 86400))
|
||||
days = (now - baseline["time"]) / 86400
|
||||
for name, threshold in (("snapshots", config["snapshot_growth_warning_gib_day"]),
|
||||
("backup", config["backup_growth_warning_gib_day"]),
|
||||
("remote", config["remote_capacity"]["growth_warning_gib_day"])):
|
||||
current, old = measurements.get(f"{name}_bytes"), baseline.get(f"{name}_bytes")
|
||||
if isinstance(current, int) and isinstance(old, int) and days > 0:
|
||||
growth_gib_day = (current - old) / GIB / days
|
||||
measurements[f"{name}_growth_gib_day"] = round(growth_gib_day, 1)
|
||||
if growth_gib_day >= threshold:
|
||||
issue(issues, f"{name}.growth", "warning",
|
||||
f"Local {name} usage grew {growth_gib_day:.1f} GiB/day over {days:.1f} days")
|
||||
|
||||
|
||||
def allowed_failure_unit(config, unit):
|
||||
for allowed in config["failure_units"]:
|
||||
if allowed == unit:
|
||||
return True
|
||||
if allowed.endswith("@.service") and unit.startswith(allowed[:-9] + "@") and unit.endswith(".service"):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def load_state():
|
||||
if not STATE_PATH.exists():
|
||||
return {"active": {}, "samples": []}
|
||||
with STATE_PATH.open(encoding="utf-8") as stream:
|
||||
state = json.load(stream)
|
||||
if not isinstance(state.get("active"), dict) or not isinstance(state.get("samples"), list):
|
||||
raise ValueError("Invalid Atlas monitor state; refusing to overwrite it")
|
||||
return state
|
||||
|
||||
|
||||
def save_state(state):
|
||||
with tempfile.NamedTemporaryFile("w", dir=STATE_DIR, prefix=".state-", delete=False,
|
||||
encoding="utf-8") as stream:
|
||||
path = Path(stream.name)
|
||||
os.chmod(path, 0o600)
|
||||
json.dump(state, stream, sort_keys=True)
|
||||
stream.write("\n")
|
||||
stream.flush()
|
||||
os.fsync(stream.fileno())
|
||||
os.replace(path, STATE_PATH)
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--dry-run", action="store_true", help="probe without notifications or state changes")
|
||||
parser.add_argument("--test-notification", action="store_true", help="submit a labelled test alert")
|
||||
parser.add_argument("--job-failed", metavar="UNIT", help="notify about a failed configured service")
|
||||
args = parser.parse_args()
|
||||
with CONFIG_PATH.open(encoding="utf-8") as stream:
|
||||
config = json.load(stream)
|
||||
if args.test_notification:
|
||||
notify(config, "atlas_monitor_test", "warning", "Test monitoraggio Atlas",
|
||||
"Notifica di prova: il monitoraggio Atlas raggiunge 45Drives Alerts. Non conferma l'invio email.")
|
||||
print("Atlas monitor test submitted to 45Drives Alerts; email delivery is not verified.")
|
||||
return 0
|
||||
if args.job_failed:
|
||||
if not allowed_failure_unit(config, args.job_failed):
|
||||
raise ValueError("Unconfigured Atlas failure unit")
|
||||
notify(config, "atlas_job_failed", "critical", f"Job Atlas fallito: {args.job_failed}",
|
||||
f"Il servizio {args.job_failed} e' fallito. Controlla: "
|
||||
f"sudo journalctl -u {args.job_failed} -n 100 --no-pager")
|
||||
print(f"Atlas job failure submitted to 45Drives Alerts: {args.job_failed}")
|
||||
return 0
|
||||
|
||||
now = int(time.time())
|
||||
issues, measurements = {}, {}
|
||||
check_pool(config, issues, measurements)
|
||||
check_capacity(config, issues, measurements)
|
||||
check_remote_capacity(config, issues, measurements)
|
||||
check_smart(config, issues, measurements)
|
||||
check_cpu(config, issues, measurements)
|
||||
check_jobs(config, issues, measurements, now)
|
||||
if args.dry_run:
|
||||
print(json.dumps({"issues": issues, "measurements": measurements}, sort_keys=True))
|
||||
return 0
|
||||
|
||||
STATE_DIR.mkdir(mode=0o700, exist_ok=True)
|
||||
with (STATE_DIR / "monitor.lock").open("w") as lock:
|
||||
fcntl.flock(lock, fcntl.LOCK_EX | fcntl.LOCK_NB)
|
||||
state = load_state()
|
||||
check_growth(config, issues, measurements, state["samples"], now)
|
||||
active, failed_notifications = state["active"], []
|
||||
for key, details in issues.items():
|
||||
old = active.get(key)
|
||||
if old is None or old.get("severity") != details["severity"]:
|
||||
try:
|
||||
notify(config, "atlas_health_issue", details["severity"],
|
||||
f"Atlas: {key}", details["message"])
|
||||
active[key] = details
|
||||
print(f"ALERT {details['severity']} {key}: {details['message']}", flush=True)
|
||||
except (RuntimeError, subprocess.TimeoutExpired) as exc:
|
||||
failed_notifications.append(key)
|
||||
print(f"NOTIFICATION FAILED {key}: {exc}", file=sys.stderr, flush=True)
|
||||
for key in set(active) - set(issues):
|
||||
print(f"RECOVERED {key}", flush=True)
|
||||
del active[key]
|
||||
state["samples"] = [sample for sample in state["samples"] if now - sample.get("time", 0) < 48 * 3600]
|
||||
state["samples"].append({"time": now, **{key: value for key, value in measurements.items()
|
||||
if key in ("snapshots_bytes", "backup_bytes", "remote_bytes")}})
|
||||
save_state(state)
|
||||
print(f"Atlas health: issues={len(issues)} notifications_failed={len(failed_notifications)} "
|
||||
f"pool={measurements.get('pool_capacity_percent', 'unknown')}% "
|
||||
f"remote={measurements.get('remote_capacity_percent', 'unknown')}% "
|
||||
f"snapshots={measurements.get('snapshots_bytes', 'unknown')} bytes "
|
||||
f"backup={measurements.get('backup_bytes', 'unknown')} bytes", flush=True)
|
||||
return 1 if failed_notifications else 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
try:
|
||||
sys.exit(main())
|
||||
except (OSError, RuntimeError, ValueError, subprocess.TimeoutExpired) as error:
|
||||
print(f"Atlas health monitor failed: {error}", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
@@ -224,7 +224,7 @@
|
||||
when: atlas_manage_borg_backup | bool
|
||||
|
||||
- name: Install the Atlas Borg backup helper
|
||||
tags: [atlas, storage, backup, borg]
|
||||
tags: [atlas, storage, backup, borg, borg_logging]
|
||||
ansible.builtin.template:
|
||||
src: atlas-borg-backup.sh.j2
|
||||
dest: /usr/local/sbin/atlas-borg-backup
|
||||
@@ -233,6 +233,16 @@
|
||||
mode: "0750"
|
||||
when: atlas_manage_borg_backup | bool
|
||||
|
||||
- name: Install the Atlas Borg snapshot cleanup helper
|
||||
tags: [atlas, storage, backup, borg, borg_logging]
|
||||
ansible.builtin.template:
|
||||
src: atlas-borg-snapshot-cleanup.sh.j2
|
||||
dest: /usr/local/sbin/atlas-borg-snapshot-cleanup
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0750"
|
||||
when: atlas_manage_borg_backup | bool
|
||||
|
||||
- name: Install the Atlas Borg check helper
|
||||
tags: [atlas, storage, backup, borg]
|
||||
ansible.builtin.template:
|
||||
@@ -244,7 +254,7 @@
|
||||
when: atlas_manage_borg_backup | bool
|
||||
|
||||
- name: Create the local libexec directory for the Atlas Borg SSH wrapper
|
||||
tags: [atlas, storage, backup, borg]
|
||||
tags: [atlas, storage, backup, borg, borg_logging]
|
||||
ansible.builtin.file:
|
||||
path: "{{ atlas_borg_ssh_wrapper_path | dirname }}"
|
||||
state: directory
|
||||
@@ -253,6 +263,16 @@
|
||||
mode: "0755"
|
||||
when: atlas_manage_borg_backup | bool
|
||||
|
||||
- name: Install the Atlas Borg progress formatter
|
||||
tags: [atlas, storage, backup, borg, borg_logging]
|
||||
ansible.builtin.copy:
|
||||
src: atlas-borg-progress.py
|
||||
dest: /usr/local/libexec/atlas-borg-progress
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0755"
|
||||
when: atlas_manage_borg_backup | bool
|
||||
|
||||
- name: Install the capability-dropping Atlas Borg SSH wrapper
|
||||
tags: [atlas, storage, backup, borg]
|
||||
ansible.builtin.template:
|
||||
@@ -264,7 +284,7 @@
|
||||
when: atlas_manage_borg_backup | bool
|
||||
|
||||
- name: Install Atlas Borg systemd units
|
||||
tags: [atlas, storage, backup, borg]
|
||||
tags: [atlas, storage, backup, borg, borg_logging]
|
||||
ansible.builtin.template:
|
||||
src: "{{ item }}.j2"
|
||||
dest: "/etc/systemd/system/{{ item }}"
|
||||
|
||||
@@ -20,6 +20,15 @@
|
||||
- name: Import Atlas Borg backup tasks
|
||||
ansible.builtin.import_tasks: borg_backup.yml
|
||||
|
||||
- name: Import Atlas offline USB backup tasks
|
||||
ansible.builtin.import_tasks: usb_backup.yml
|
||||
|
||||
- name: Import Atlas health monitoring tasks
|
||||
ansible.builtin.import_tasks: monitoring.yml
|
||||
|
||||
- name: Import Atlas post-restore SELinux relabeling tasks
|
||||
ansible.builtin.import_tasks: restorecon.yml
|
||||
|
||||
- name: Import Atlas file sharing tasks
|
||||
ansible.builtin.import_tasks: sharing.yml
|
||||
|
||||
|
||||
201
ansible/roles/profile_atlas/tasks/monitoring.yml
Normal file
201
ansible/roles/profile_atlas/tasks/monitoring.yml
Normal file
@@ -0,0 +1,201 @@
|
||||
---
|
||||
- name: Validate Atlas health monitoring policy
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- atlas_manage_storage | bool
|
||||
- atlas_zfs_pool != 'CHANGEME_ZFS_POOL'
|
||||
- atlas_monitor_calendar | length > 0
|
||||
- atlas_monitor_smart_devices | length > 0
|
||||
- atlas_monitor_timers | length > 0
|
||||
- atlas_monitor_failure_units | length > 0
|
||||
- atlas_monitor_remote_capacity.user == atlas_borg_repository_user
|
||||
- atlas_monitor_remote_capacity.host == atlas_borg_repository_host
|
||||
- atlas_monitor_remote_capacity.run_as == atlas_borg_username
|
||||
- atlas_monitor_remote_capacity.ssh_wrapper == atlas_borg_ssh_wrapper_path
|
||||
- >-
|
||||
0 < atlas_monitor_remote_capacity.warning_percent | int
|
||||
< atlas_monitor_remote_capacity.critical_percent | int < 100
|
||||
- atlas_monitor_remote_capacity.growth_warning_gib_day | int > 0
|
||||
- atlas_monitor_notifier.startswith('/opt/45drives/houston/')
|
||||
- 0 < atlas_monitor_pool_warning_percent | int < atlas_monitor_pool_critical_percent | int < 100
|
||||
- 0 < atlas_monitor_root_warning_percent | int < atlas_monitor_root_critical_percent | int < 100
|
||||
- 0 < atlas_monitor_snapshot_warning_percent | int < atlas_monitor_snapshot_critical_percent | int < 100
|
||||
- atlas_monitor_snapshot_growth_warning_gib_day | int > 0
|
||||
- atlas_monitor_backup_growth_warning_gib_day | int > 0
|
||||
- 0 < atlas_monitor_cpu_warning_c | int < atlas_monitor_cpu_critical_c | int
|
||||
- atlas_monitor_borg_max_runtime_days | int > 0
|
||||
fail_msg: >-
|
||||
Atlas health monitoring needs real devices, job units, a valid calendar,
|
||||
positive ordered thresholds, and the existing Houston notifier.
|
||||
when: atlas_manage_monitoring | bool
|
||||
|
||||
- name: Validate monitored Atlas SMART devices
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- item.name is match('^[a-z0-9][a-z0-9_-]*$')
|
||||
- item.path.startswith('/dev/disk/by-id/')
|
||||
- 0 < item.warning_c | int < item.critical_c | int
|
||||
fail_msg: "Every monitored disk needs a stable by-id path and ordered temperature thresholds."
|
||||
loop: "{{ atlas_monitor_smart_devices }}"
|
||||
loop_control:
|
||||
label: "{{ item.name }}"
|
||||
when: atlas_manage_monitoring | bool
|
||||
|
||||
- name: Validate monitored Atlas timer names and age thresholds
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- item.name is match('^[a-zA-Z0-9@_.-]+\\.timer$')
|
||||
- item.max_age_hours | int >= 0
|
||||
loop: "{{ atlas_monitor_timers }}"
|
||||
loop_control:
|
||||
label: "{{ item.name }}"
|
||||
when: atlas_manage_monitoring | bool
|
||||
|
||||
- name: Validate monitored Atlas failure unit names
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- item is match('^[a-zA-Z0-9@_.-]+\\.service$')
|
||||
loop: "{{ atlas_monitor_failure_units }}"
|
||||
when: atlas_manage_monitoring | bool
|
||||
|
||||
- name: Validate Atlas health monitor calendar
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.command:
|
||||
argv: [systemd-analyze, calendar, "{{ atlas_monitor_calendar }}"]
|
||||
changed_when: false
|
||||
check_mode: false
|
||||
when: atlas_manage_monitoring | bool
|
||||
|
||||
- name: Install SMART tooling for Atlas health checks
|
||||
tags: [atlas, monitoring, packages]
|
||||
ansible.builtin.dnf:
|
||||
name: smartmontools
|
||||
state: present
|
||||
when: atlas_manage_monitoring | bool
|
||||
|
||||
- name: Inspect the existing 45Drives notifier for monitoring
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.stat:
|
||||
path: "{{ atlas_monitor_notifier }}"
|
||||
register: atlas_monitor_notifier_file
|
||||
when: atlas_manage_monitoring | bool
|
||||
|
||||
- name: Require the existing 45Drives notifier for monitoring
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- atlas_monitor_notifier_file.stat.executable | default(false)
|
||||
fail_msg: "The existing 45Drives Houston notifier must be executable."
|
||||
when: atlas_manage_monitoring | bool
|
||||
|
||||
- name: Create private Atlas health monitor state directory
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.file:
|
||||
path: /var/lib/atlas-health-monitor
|
||||
state: directory
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0700"
|
||||
when: atlas_manage_monitoring | bool
|
||||
|
||||
- name: Install Atlas health monitor configuration
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.template:
|
||||
src: atlas-health-monitor.json.j2
|
||||
dest: /etc/atlas-health-monitor.json
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0600"
|
||||
when: atlas_manage_monitoring | bool
|
||||
|
||||
- name: Install Atlas health monitor helper
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.copy:
|
||||
src: atlas-health-monitor.py
|
||||
dest: /usr/local/libexec/atlas-health-monitor
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0750"
|
||||
when: atlas_manage_monitoring | bool
|
||||
|
||||
- name: Install Atlas health monitoring units
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.template:
|
||||
src: "{{ item }}.j2"
|
||||
dest: "/etc/systemd/system/{{ item }}"
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0644"
|
||||
loop:
|
||||
- atlas-health-monitor.service
|
||||
- atlas-health-monitor.timer
|
||||
- atlas-monitor-failure@.service
|
||||
when: atlas_manage_monitoring | bool
|
||||
|
||||
- name: Create failure hook directories for monitored Atlas jobs
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.file:
|
||||
path: "/etc/systemd/system/{{ item }}.d"
|
||||
state: directory
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0755"
|
||||
loop: "{{ atlas_monitor_failure_units }}"
|
||||
when: atlas_manage_monitoring | bool
|
||||
|
||||
- name: Notify 45Drives Alerts when an Atlas job fails
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.template:
|
||||
src: atlas-monitor-failure.conf.j2
|
||||
dest: "/etc/systemd/system/{{ item }}.d/atlas-monitor.conf"
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0644"
|
||||
loop: "{{ atlas_monitor_failure_units }}"
|
||||
when: atlas_manage_monitoring | bool
|
||||
|
||||
- name: Reload systemd after installing Atlas monitoring
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.systemd:
|
||||
daemon_reload: true
|
||||
when:
|
||||
- atlas_manage_monitoring | bool
|
||||
- not ansible_check_mode
|
||||
|
||||
- name: Enable the Atlas health monitoring timer
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.systemd:
|
||||
name: atlas-health-monitor.timer
|
||||
enabled: true
|
||||
state: started
|
||||
when:
|
||||
- atlas_manage_monitoring | bool
|
||||
- not ansible_check_mode
|
||||
|
||||
- name: Validate the deployed Atlas health monitoring units
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.command:
|
||||
argv:
|
||||
- systemd-analyze
|
||||
- verify
|
||||
- atlas-health-monitor.service
|
||||
- atlas-health-monitor.timer
|
||||
- atlas-monitor-failure@.service
|
||||
changed_when: false
|
||||
when:
|
||||
- atlas_manage_monitoring | bool
|
||||
- not ansible_check_mode
|
||||
|
||||
- name: Probe Atlas health without sending notifications
|
||||
tags: [atlas, monitoring]
|
||||
ansible.builtin.command:
|
||||
argv: [/usr/local/libexec/atlas-health-monitor, --dry-run]
|
||||
register: atlas_monitor_dry_run
|
||||
changed_when: false
|
||||
when:
|
||||
- atlas_manage_monitoring | bool
|
||||
- not ansible_check_mode
|
||||
27
ansible/roles/profile_atlas/tasks/restorecon.yml
Normal file
27
ansible/roles/profile_atlas/tasks/restorecon.yml
Normal file
@@ -0,0 +1,27 @@
|
||||
---
|
||||
- name: Validate requested Atlas post-restore relabel paths
|
||||
tags: [atlas, restorecon, recovery]
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- item is string
|
||||
- item.startswith(atlas_mount_root ~ '/')
|
||||
- item != atlas_mount_root
|
||||
fail_msg: >-
|
||||
Post-restore relabeling accepts only explicit paths below the Atlas pool
|
||||
mount root. Do not relabel the whole pool during routine provisioning.
|
||||
loop: "{{ atlas_restorecon_paths }}"
|
||||
when: atlas_restorecon_paths | length > 0
|
||||
|
||||
- name: Restore SELinux labels on explicitly restored Atlas paths
|
||||
tags: [atlas, restorecon, recovery]
|
||||
ansible.builtin.command:
|
||||
argv:
|
||||
- restorecon
|
||||
- -RFv
|
||||
- "{{ item }}"
|
||||
register: atlas_restorecon_result
|
||||
changed_when: atlas_restorecon_result.stdout | length > 0
|
||||
loop: "{{ atlas_restorecon_paths }}"
|
||||
when:
|
||||
- atlas_restorecon_paths | length > 0
|
||||
- not ansible_check_mode
|
||||
145
ansible/roles/profile_atlas/tasks/usb_backup.yml
Normal file
145
ansible/roles/profile_atlas/tasks/usb_backup.yml
Normal file
@@ -0,0 +1,145 @@
|
||||
---
|
||||
- name: Validate Atlas offline USB backup configuration
|
||||
tags: [atlas, storage, backup, usb_backup]
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- atlas_manage_storage | bool
|
||||
- atlas_zfs_pool != 'CHANGEME_ZFS_POOL'
|
||||
- atlas_mount_root.startswith('/')
|
||||
- atlas_usb_backup_luks_uuid is match('^[0-9a-fA-F]{8}(-[0-9a-fA-F]{4}){3}-[0-9a-fA-F]{12}$')
|
||||
- atlas_usb_backup_fs_uuid is match('^[0-9a-fA-F]{8}(-[0-9a-fA-F]{4}){3}-[0-9a-fA-F]{12}$')
|
||||
- atlas_usb_backup_luks_uuid != atlas_usb_backup_fs_uuid
|
||||
- atlas_usb_backup_mapper_name is match('^[a-z][a-z0-9_-]*$')
|
||||
- atlas_usb_backup_min_free_bytes | int > 0
|
||||
- atlas_usb_backup_snapshot_prefix is match('^[a-z0-9][a-z0-9_-]*$')
|
||||
- atlas_usb_backup_snapshot_prefix != atlas_borg_snapshot_prefix
|
||||
- atlas_usb_backup_snapshot_prefix != atlas_zfs_snapshot_prefix
|
||||
fail_msg: >-
|
||||
The manual Atlas USB backup needs verified LUKS and ext4 UUIDs, a safe
|
||||
mapper name, positive free-space reserve, and a unique snapshot prefix.
|
||||
when: atlas_manage_usb_backup | bool
|
||||
|
||||
- name: Install rsync for the Atlas offline USB backup
|
||||
tags: [atlas, storage, backup, usb_backup]
|
||||
ansible.builtin.dnf:
|
||||
name: rsync
|
||||
state: present
|
||||
when: atlas_manage_usb_backup | bool
|
||||
|
||||
- name: Install the manual Atlas offline USB backup helper
|
||||
tags: [atlas, storage, backup, usb_backup]
|
||||
ansible.builtin.template:
|
||||
src: atlas-usb-backup.sh.j2
|
||||
dest: /usr/local/sbin/atlas-usb-backup
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0750"
|
||||
when: atlas_manage_usb_backup | bool
|
||||
|
||||
- name: Install the Atlas USB snapshot cleanup helper
|
||||
tags: [atlas, storage, backup, usb_backup]
|
||||
ansible.builtin.template:
|
||||
src: atlas-usb-snapshot-cleanup.sh.j2
|
||||
dest: /usr/local/sbin/atlas-usb-snapshot-cleanup
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0750"
|
||||
when: atlas_manage_usb_backup | bool
|
||||
|
||||
- name: Install the manual Atlas offline USB backup service
|
||||
tags: [atlas, storage, backup, usb_backup]
|
||||
ansible.builtin.template:
|
||||
src: atlas-usb-backup.service.j2
|
||||
dest: /etc/systemd/system/atlas-usb-backup.service
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0644"
|
||||
when: atlas_manage_usb_backup | bool
|
||||
|
||||
- name: Reload systemd for the Atlas offline USB backup service
|
||||
tags: [atlas, storage, backup, usb_backup]
|
||||
ansible.builtin.systemd:
|
||||
daemon_reload: true
|
||||
when:
|
||||
- atlas_manage_usb_backup | bool
|
||||
- not ansible_check_mode
|
||||
|
||||
- name: Validate the 45Drives Atlas USB reminder configuration
|
||||
tags: [atlas, backup, usb_reminder]
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- atlas_manage_usb_backup | bool
|
||||
- atlas_usb_reminder_calendar | length > 0
|
||||
- atlas_usb_reminder_notifier.startswith('/opt/45drives/houston/')
|
||||
fail_msg: >-
|
||||
Enable the manual USB backup and declare a systemd calendar before
|
||||
enabling its 45Drives Alerts reminder.
|
||||
when: atlas_manage_usb_reminder | bool
|
||||
|
||||
- name: Validate the Atlas USB reminder calendar
|
||||
tags: [atlas, backup, usb_reminder]
|
||||
ansible.builtin.command:
|
||||
argv:
|
||||
- systemd-analyze
|
||||
- calendar
|
||||
- "{{ atlas_usb_reminder_calendar }}"
|
||||
changed_when: false
|
||||
check_mode: false
|
||||
when: atlas_manage_usb_reminder | bool
|
||||
|
||||
- name: Inspect the existing 45Drives notifier
|
||||
tags: [atlas, backup, usb_reminder]
|
||||
ansible.builtin.stat:
|
||||
path: "{{ atlas_usb_reminder_notifier }}"
|
||||
register: atlas_usb_reminder_notifier_file
|
||||
when: atlas_manage_usb_reminder | bool
|
||||
|
||||
- name: Require the configured 45Drives notifier for USB reminders
|
||||
tags: [atlas, backup, usb_reminder]
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- atlas_usb_reminder_notifier_file.stat.executable | default(false)
|
||||
fail_msg: >-
|
||||
The existing 45Drives Houston notifier must be executable.
|
||||
when: atlas_manage_usb_reminder | bool
|
||||
|
||||
- name: Install the 45Drives Atlas USB reminder helper
|
||||
tags: [atlas, backup, usb_reminder]
|
||||
ansible.builtin.template:
|
||||
src: atlas-usb-reminder.py.j2
|
||||
dest: /usr/local/libexec/atlas-usb-reminder
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0750"
|
||||
when: atlas_manage_usb_reminder | bool
|
||||
|
||||
- name: Install the 45Drives Atlas USB reminder service
|
||||
tags: [atlas, backup, usb_reminder]
|
||||
ansible.builtin.template:
|
||||
src: atlas-usb-reminder.service.j2
|
||||
dest: /etc/systemd/system/atlas-usb-reminder.service
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0644"
|
||||
when: atlas_manage_usb_reminder | bool
|
||||
|
||||
- name: Install the 45Drives Atlas USB reminder timer
|
||||
tags: [atlas, backup, usb_reminder]
|
||||
ansible.builtin.template:
|
||||
src: atlas-usb-reminder.timer.j2
|
||||
dest: /etc/systemd/system/atlas-usb-reminder.timer
|
||||
owner: root
|
||||
group: root
|
||||
mode: "0644"
|
||||
when: atlas_manage_usb_reminder | bool
|
||||
|
||||
- name: Enable only the Atlas USB notification reminder timer
|
||||
tags: [atlas, backup, usb_reminder]
|
||||
ansible.builtin.systemd:
|
||||
name: atlas-usb-reminder.timer
|
||||
enabled: true
|
||||
state: started
|
||||
daemon_reload: true
|
||||
when:
|
||||
- atlas_manage_usb_reminder | bool
|
||||
- not ansible_check_mode
|
||||
@@ -14,6 +14,7 @@ ConditionPathExists={{ atlas_borg_known_hosts_path }}
|
||||
[Service]
|
||||
Type=oneshot
|
||||
ExecStart=/usr/local/sbin/atlas-borg-backup
|
||||
ExecStopPost=+/usr/local/sbin/atlas-borg-snapshot-cleanup
|
||||
User=root
|
||||
Group=root
|
||||
UMask=0077
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
#!/usr/bin/env bash
|
||||
set -Eeuo pipefail
|
||||
|
||||
export LC_ALL=C
|
||||
export LC_ALL=C.utf8
|
||||
export PATH=/usr/sbin:/usr/bin:/sbin:/bin
|
||||
export BORG_CACHE_DIR={{ atlas_borg_cache_dir | quote }}
|
||||
export BORG_CONFIG_DIR={{ atlas_borg_config_dir | quote }}
|
||||
@@ -17,13 +17,14 @@ readonly archive_prefix={{ atlas_borg_archive_prefix | quote }}
|
||||
readonly snapshot_prefix={{ atlas_borg_snapshot_prefix | quote }}
|
||||
readonly compression={{ atlas_borg_compression | quote }}
|
||||
readonly stage=/run/atlas-borg/source
|
||||
readonly snapshot_marker=/run/atlas-borg/snapshot-name
|
||||
readonly borg_user={{ atlas_borg_username | quote }}
|
||||
readonly borg_group={{ atlas_borg_group | quote }}
|
||||
readonly borg_home={{ atlas_borg_home | quote }}
|
||||
readonly borg_lock={{ atlas_borg_lock_path | quote }}
|
||||
readonly progress_filter=/usr/local/libexec/atlas-borg-progress
|
||||
|
||||
snapshot_name=""
|
||||
snapshot_created=false
|
||||
mounted_targets=()
|
||||
|
||||
# Invoked through the EXIT trap below.
|
||||
@@ -32,22 +33,31 @@ cleanup() {
|
||||
local status=$?
|
||||
local cleanup_status=0
|
||||
local index
|
||||
local source_mount_failed=false
|
||||
trap - EXIT HUP INT TERM
|
||||
set +e
|
||||
|
||||
{% raw %}
|
||||
for ((index = ${#mounted_targets[@]} - 1; index >= 0; index--)); do
|
||||
{% endraw %}
|
||||
if mountpoint -q "${mounted_targets[$index]}" && ! umount -R "${mounted_targets[$index]}"; then
|
||||
printf 'Source snapshot mount cleanup failed: %s\n' "${mounted_targets[$index]}" >&2
|
||||
source_mount_failed=true
|
||||
fi
|
||||
if mountpoint -q "${mounted_targets[$index]}"; then
|
||||
umount "${mounted_targets[$index]}" || cleanup_status=2
|
||||
printf 'Source snapshot mount is still active: %s\n' "${mounted_targets[$index]}" >&2
|
||||
source_mount_failed=true
|
||||
else
|
||||
rmdir -- "${mounted_targets[$index]}" 2>/dev/null || true
|
||||
fi
|
||||
done
|
||||
rm -rf "$stage" || cleanup_status=2
|
||||
|
||||
if [[ "$snapshot_created" == true ]]; then
|
||||
flock 9
|
||||
zfs destroy -r "${pool}@${snapshot_name}" || cleanup_status=2
|
||||
flock -u 9
|
||||
if [[ "$source_mount_failed" == false ]]; then
|
||||
if [[ -d "$stage" ]]; then
|
||||
rmdir -- "$stage" 2>/dev/null || cleanup_status=2
|
||||
fi
|
||||
else
|
||||
cleanup_status=2
|
||||
printf 'Source bind mount cleanup failed; keeping the snapshot for recovery\n' >&2
|
||||
fi
|
||||
|
||||
if ((status == 0 && cleanup_status != 0)); then
|
||||
@@ -96,8 +106,8 @@ timestamp="$(date -u +%Y%m%dT%H%M%SZ)"
|
||||
readonly timestamp
|
||||
snapshot_name="${snapshot_prefix}-${timestamp}"
|
||||
readonly snapshot_name
|
||||
printf '%s\n' "$snapshot_name" >"$snapshot_marker"
|
||||
zfs snapshot -r "${pool}@${snapshot_name}"
|
||||
snapshot_created=true
|
||||
flock -u 9
|
||||
printf 'Created recursive Borg source snapshot %s@%s\n' "$pool" "$snapshot_name"
|
||||
|
||||
@@ -116,32 +126,56 @@ while IFS=$'\t' read -r dataset dataset_mountpoint mounted; do
|
||||
target_path="${stage}${dataset_suffix}"
|
||||
mkdir -p "$target_path"
|
||||
mount --bind "$source_path" "$target_path"
|
||||
mount -o remount,bind,ro "$target_path"
|
||||
mounted_targets+=("$target_path")
|
||||
mount -o remount,bind,ro "$target_path"
|
||||
done < <(zfs list -H -o name,mountpoint,mounted -s name -r "$pool")
|
||||
|
||||
estimated_source_bytes=0
|
||||
while IFS=$'\t' read -r source_snapshot logical_bytes; do
|
||||
if [[ "$source_snapshot" == *"@${snapshot_name}" ]]; then
|
||||
[[ "$logical_bytes" =~ ^[0-9]+$ ]] || {
|
||||
printf 'Invalid logical size for Borg source snapshot %s\n' "$source_snapshot" >&2
|
||||
exit 74
|
||||
}
|
||||
estimated_source_bytes=$((estimated_source_bytes + logical_bytes))
|
||||
fi
|
||||
done < <(zfs list -H -p -t snapshot -o name,logicalreferenced -r "$pool")
|
||||
((estimated_source_bytes > 0)) || {
|
||||
printf 'Could not estimate the Borg source snapshot size\n' >&2
|
||||
exit 74
|
||||
}
|
||||
printf 'Estimated Borg source logical size: %s bytes (ZFS; progress percentage is approximate)\n' \
|
||||
"$estimated_source_bytes"
|
||||
|
||||
archive="${archive_prefix}-${timestamp}"
|
||||
readonly archive
|
||||
borg_status=0
|
||||
|
||||
printf 'Starting Borg archive %s from snapshot %s@%s\n' "$archive" "$pool" "$snapshot_name"
|
||||
set +e
|
||||
(
|
||||
cd /run/atlas-borg
|
||||
run_as_borg borg --remote-path "$remote_path" --lock-wait 600 create \
|
||||
run_as_borg borg --remote-path "$remote_path" --lock-wait 600 --log-json --progress create \
|
||||
--show-rc \
|
||||
--stats \
|
||||
--checkpoint-interval 900 \
|
||||
--compression "$compression" \
|
||||
"${repository}::${archive}" \
|
||||
source
|
||||
)
|
||||
create_status=$?
|
||||
source 2>&1
|
||||
) | /usr/bin/python3 -u "$progress_filter" --estimated-total-bytes "$estimated_source_bytes"
|
||||
create_pipeline_status=("${PIPESTATUS[@]}")
|
||||
set -e
|
||||
create_status=${create_pipeline_status[0]}
|
||||
if ((create_pipeline_status[1] != 0)); then
|
||||
printf 'Borg progress logging failed with status %s\n' "${create_pipeline_status[1]}" >&2
|
||||
exit 2
|
||||
fi
|
||||
if ((create_status >= 2)); then
|
||||
exit "$create_status"
|
||||
fi
|
||||
borg_status=$create_status
|
||||
|
||||
printf 'Borg archive %s created; applying retention\n' "$archive"
|
||||
set +e
|
||||
run_as_borg borg --remote-path "$remote_path" --lock-wait 600 prune \
|
||||
--show-rc \
|
||||
@@ -160,6 +194,7 @@ if ((prune_status > borg_status)); then
|
||||
borg_status=$prune_status
|
||||
fi
|
||||
|
||||
printf 'Borg retention complete; compacting repository\n'
|
||||
set +e
|
||||
run_as_borg borg --remote-path "$remote_path" --lock-wait 600 compact \
|
||||
--show-rc \
|
||||
@@ -173,4 +208,5 @@ if ((compact_status > borg_status)); then
|
||||
borg_status=$compact_status
|
||||
fi
|
||||
|
||||
printf 'Borg backup %s completed with status %s\n' "$archive" "$borg_status"
|
||||
exit "$borg_status"
|
||||
|
||||
@@ -0,0 +1,51 @@
|
||||
#!/usr/bin/env bash
|
||||
set -Eeuo pipefail
|
||||
|
||||
export PATH=/usr/sbin:/usr/bin:/sbin:/bin
|
||||
|
||||
readonly pool={{ atlas_zfs_pool | quote }}
|
||||
readonly mount_root={{ atlas_mount_root | quote }}
|
||||
readonly snapshot_prefix={{ atlas_borg_snapshot_prefix | quote }}
|
||||
readonly marker=/run/atlas-borg/snapshot-name
|
||||
|
||||
[[ -e "$marker" ]] || exit 0
|
||||
[[ -f "$marker" && ! -L "$marker" ]] || {
|
||||
printf 'Unsafe Atlas Borg snapshot marker; leaving snapshots unchanged\n' >&2
|
||||
exit 2
|
||||
}
|
||||
IFS= read -r snapshot_name <"$marker"
|
||||
[[ "$snapshot_name" =~ ^${snapshot_prefix}-[0-9]{8}T[0-9]{6}Z$ ]] || {
|
||||
printf 'Invalid Atlas Borg snapshot marker; leaving snapshots unchanged\n' >&2
|
||||
exit 2
|
||||
}
|
||||
|
||||
exec 9>/run/lock/atlas-zfs-snapshot.lock
|
||||
flock 9
|
||||
if zfs list -H -t snapshot -o name "${pool}@${snapshot_name}" >/dev/null 2>&1; then
|
||||
# The private bind mounts are gone, but ZFS may leave its on-demand
|
||||
# .zfs/snapshot mounts in the host namespace until explicitly unmounted.
|
||||
snapshot_mounts=()
|
||||
snapshot_sources=()
|
||||
while IFS=$'\t' read -r dataset dataset_mountpoint; do
|
||||
[[ "$dataset_mountpoint" == "$mount_root" || "$dataset_mountpoint" == "$mount_root/"* ]] || continue
|
||||
snapshot_mounts+=("${dataset_mountpoint}/.zfs/snapshot/${snapshot_name}")
|
||||
snapshot_sources+=("${dataset}@${snapshot_name}")
|
||||
done < <(zfs list -H -o name,mountpoint -s name -r "$pool")
|
||||
|
||||
{% raw %}
|
||||
for ((index = ${#snapshot_mounts[@]} - 1; index >= 0; index--)); do
|
||||
{% endraw %}
|
||||
mounted_source="$(findmnt -rn -M "${snapshot_mounts[$index]}" -o SOURCE || true)"
|
||||
[[ -n "$mounted_source" ]] || continue
|
||||
[[ "$mounted_source" == "${snapshot_sources[$index]}" ]] || {
|
||||
printf 'Unexpected source on Atlas Borg snapshot mount: %s\n' \
|
||||
"${snapshot_mounts[$index]}" >&2
|
||||
exit 2
|
||||
}
|
||||
umount "${snapshot_mounts[$index]}"
|
||||
done
|
||||
|
||||
zfs destroy -r "${pool}@${snapshot_name}"
|
||||
printf 'Removed recursive Atlas Borg source snapshot %s@%s after backup exit\n' \
|
||||
"$pool" "$snapshot_name"
|
||||
fi
|
||||
@@ -0,0 +1,20 @@
|
||||
{
|
||||
"pool": {{ atlas_zfs_pool | to_json }},
|
||||
"backup_dataset": {{ (atlas_zfs_pool ~ '/' ~ atlas_zfs_dataset_backup) | to_json }},
|
||||
"notifier": {{ atlas_monitor_notifier | to_json }},
|
||||
"smart_devices": {{ atlas_monitor_smart_devices | to_json }},
|
||||
"timers": {{ atlas_monitor_timers | to_json }},
|
||||
"failure_units": {{ atlas_monitor_failure_units | to_json }},
|
||||
"remote_capacity": {{ atlas_monitor_remote_capacity | to_json }},
|
||||
"pool_warning_percent": {{ atlas_monitor_pool_warning_percent | int }},
|
||||
"pool_critical_percent": {{ atlas_monitor_pool_critical_percent | int }},
|
||||
"root_warning_percent": {{ atlas_monitor_root_warning_percent | int }},
|
||||
"root_critical_percent": {{ atlas_monitor_root_critical_percent | int }},
|
||||
"snapshot_warning_percent": {{ atlas_monitor_snapshot_warning_percent | int }},
|
||||
"snapshot_critical_percent": {{ atlas_monitor_snapshot_critical_percent | int }},
|
||||
"snapshot_growth_warning_gib_day": {{ atlas_monitor_snapshot_growth_warning_gib_day | int }},
|
||||
"backup_growth_warning_gib_day": {{ atlas_monitor_backup_growth_warning_gib_day | int }},
|
||||
"cpu_warning_c": {{ atlas_monitor_cpu_warning_c | int }},
|
||||
"cpu_critical_c": {{ atlas_monitor_cpu_critical_c | int }},
|
||||
"borg_max_runtime_days": {{ atlas_monitor_borg_max_runtime_days | int }}
|
||||
}
|
||||
@@ -0,0 +1,25 @@
|
||||
[Unit]
|
||||
Description=Check Atlas pool, disks, capacity, temperatures and maintenance jobs
|
||||
Wants=houston-dbus.service network-online.target
|
||||
After=zfs.target houston-dbus.service network-online.target
|
||||
ConditionFileIsExecutable=/usr/local/libexec/atlas-health-monitor
|
||||
|
||||
[Service]
|
||||
Type=oneshot
|
||||
ExecStart=/usr/local/libexec/atlas-health-monitor
|
||||
User=root
|
||||
Group=root
|
||||
UMask=0077
|
||||
StateDirectory=atlas-health-monitor
|
||||
StateDirectoryMode=0700
|
||||
NoNewPrivileges=true
|
||||
PrivateTmp=true
|
||||
ProtectHome=true
|
||||
ProtectSystem=strict
|
||||
ReadWritePaths=/var/lib/atlas-health-monitor
|
||||
ProtectKernelTunables=true
|
||||
ProtectKernelModules=true
|
||||
ProtectControlGroups=true
|
||||
RestrictAddressFamilies=AF_UNIX AF_INET AF_INET6
|
||||
RestrictRealtime=true
|
||||
LockPersonality=true
|
||||
@@ -0,0 +1,11 @@
|
||||
[Unit]
|
||||
Description=Schedule Atlas health checks
|
||||
|
||||
[Timer]
|
||||
OnCalendar={{ atlas_monitor_calendar }}
|
||||
Persistent=true
|
||||
RandomizedDelaySec=5min
|
||||
Unit=atlas-health-monitor.service
|
||||
|
||||
[Install]
|
||||
WantedBy=timers.target
|
||||
@@ -0,0 +1,2 @@
|
||||
[Unit]
|
||||
OnFailure=atlas-monitor-failure@%n.service
|
||||
@@ -0,0 +1,19 @@
|
||||
[Unit]
|
||||
Description=Submit a 45Drives Alert for failed Atlas job %I
|
||||
Requires=houston-dbus.service
|
||||
After=houston-dbus.service
|
||||
ConditionFileIsExecutable=/usr/local/libexec/atlas-health-monitor
|
||||
|
||||
[Service]
|
||||
Type=oneshot
|
||||
ExecStart=/usr/local/libexec/atlas-health-monitor --job-failed %I
|
||||
User=root
|
||||
Group=root
|
||||
UMask=0077
|
||||
NoNewPrivileges=true
|
||||
PrivateTmp=true
|
||||
ProtectHome=true
|
||||
ProtectSystem=strict
|
||||
RestrictAddressFamilies=AF_UNIX
|
||||
RestrictRealtime=true
|
||||
LockPersonality=true
|
||||
@@ -0,0 +1,30 @@
|
||||
[Unit]
|
||||
Description=Run a manual, UUID-bound offline USB backup of Atlas ZFS datasets
|
||||
Requires=zfs.target
|
||||
After=zfs.target
|
||||
ConditionFileIsExecutable=/usr/local/sbin/atlas-usb-backup
|
||||
|
||||
[Service]
|
||||
Type=oneshot
|
||||
ExecStart=/usr/local/sbin/atlas-usb-backup
|
||||
ExecStopPost=+/usr/local/sbin/atlas-usb-snapshot-cleanup
|
||||
User=root
|
||||
Group=root
|
||||
UMask=0077
|
||||
TimeoutStartSec=infinity
|
||||
RuntimeDirectory=atlas-usb-backup
|
||||
RuntimeDirectoryMode=0700
|
||||
Nice=15
|
||||
IOSchedulingClass=best-effort
|
||||
IOSchedulingPriority=7
|
||||
PrivateMounts=true
|
||||
PrivateTmp=true
|
||||
ProtectHome=true
|
||||
ProtectSystem=strict
|
||||
ReadWritePaths=/run/atlas-usb-backup /run/lock
|
||||
ProtectKernelTunables=true
|
||||
ProtectKernelModules=true
|
||||
ProtectControlGroups=true
|
||||
RestrictAddressFamilies=AF_UNIX
|
||||
RestrictRealtime=true
|
||||
LockPersonality=true
|
||||
242
ansible/roles/profile_atlas/templates/atlas-usb-backup.sh.j2
Normal file
242
ansible/roles/profile_atlas/templates/atlas-usb-backup.sh.j2
Normal file
@@ -0,0 +1,242 @@
|
||||
#!/usr/bin/env bash
|
||||
set -Eeuo pipefail
|
||||
|
||||
export LC_ALL=C.utf8
|
||||
export PATH=/usr/sbin:/usr/bin:/sbin:/bin
|
||||
|
||||
readonly pool={{ atlas_zfs_pool | quote }}
|
||||
readonly mount_root={{ atlas_mount_root | quote }}
|
||||
readonly luks_uuid={{ atlas_usb_backup_luks_uuid | quote }}
|
||||
readonly fs_uuid={{ atlas_usb_backup_fs_uuid | quote }}
|
||||
readonly mapper_name={{ atlas_usb_backup_mapper_name | quote }}
|
||||
readonly snapshot_prefix={{ atlas_usb_backup_snapshot_prefix | quote }}
|
||||
readonly min_free_bytes={{ atlas_usb_backup_min_free_bytes | int }}
|
||||
readonly mapper="/dev/mapper/${mapper_name}"
|
||||
readonly outer="/dev/disk/by-uuid/${luks_uuid}"
|
||||
readonly runtime_dir=/run/atlas-usb-backup
|
||||
readonly snapshot_marker="${runtime_dir}/snapshot-name"
|
||||
readonly source_dir="${runtime_dir}/source"
|
||||
readonly usb_mount="${runtime_dir}/target"
|
||||
readonly backup_root="${usb_mount}/atlas"
|
||||
|
||||
snapshot_name=""
|
||||
mapper_opened_by_script=false
|
||||
usb_mounted=false
|
||||
published=false
|
||||
partial=""
|
||||
mounted_targets=()
|
||||
|
||||
# shellcheck disable=SC2329
|
||||
cleanup() {
|
||||
local status=$?
|
||||
local cleanup_status=0
|
||||
local index
|
||||
local source_mount_failed=false
|
||||
trap - EXIT HUP INT TERM
|
||||
set +e
|
||||
|
||||
if [[ -n "$partial" && "$published" == false && "$usb_mounted" == true ]]; then
|
||||
rm -rf -- "$partial" || cleanup_status=2
|
||||
fi
|
||||
if [[ "$usb_mounted" == true ]]; then
|
||||
umount "$usb_mount" || cleanup_status=2
|
||||
fi
|
||||
|
||||
{% raw %}
|
||||
for ((index = ${#mounted_targets[@]} - 1; index >= 0; index--)); do
|
||||
{% endraw %}
|
||||
if mountpoint -q "${mounted_targets[$index]}" && ! umount -R "${mounted_targets[$index]}"; then
|
||||
printf 'Source snapshot mount cleanup failed: %s\n' "${mounted_targets[$index]}" >&2
|
||||
source_mount_failed=true
|
||||
fi
|
||||
if mountpoint -q "${mounted_targets[$index]}"; then
|
||||
printf 'Source snapshot mount is still active: %s\n' "${mounted_targets[$index]}" >&2
|
||||
source_mount_failed=true
|
||||
else
|
||||
rmdir -- "${mounted_targets[$index]}" 2>/dev/null || true
|
||||
fi
|
||||
done
|
||||
if [[ "$source_mount_failed" == false ]]; then
|
||||
rmdir -- "$source_dir" 2>/dev/null || true
|
||||
else
|
||||
cleanup_status=2
|
||||
printf 'Source bind mount cleanup failed; keeping the snapshot for recovery\n' >&2
|
||||
fi
|
||||
|
||||
if [[ "$usb_mounted" == true || "$mapper_opened_by_script" == true ]] &&
|
||||
! mountpoint -q "$usb_mount" &&
|
||||
! findmnt -rn -S "$mapper" >/dev/null; then
|
||||
cryptsetup close "$mapper_name" || cleanup_status=2
|
||||
fi
|
||||
rmdir -- "$usb_mount" 2>/dev/null || true
|
||||
|
||||
if ((status == 0 && cleanup_status != 0)); then
|
||||
status=$cleanup_status
|
||||
fi
|
||||
exit "$status"
|
||||
}
|
||||
|
||||
trap cleanup EXIT
|
||||
trap 'exit 143' HUP INT TERM
|
||||
|
||||
exec 8>/run/lock/atlas-usb-backup.lock
|
||||
flock -n 8 || { printf 'Atlas USB backup is already running\n' >&2; exit 75; }
|
||||
exec 9>/run/lock/atlas-zfs-snapshot.lock
|
||||
|
||||
zpool list -H -o name "$pool" >/dev/null
|
||||
[[ -b "$outer" ]] || { printf 'Configured LUKS UUID is not connected\n' >&2; exit 66; }
|
||||
[[ "$(blkid -s TYPE -o value "$outer")" == crypto_LUKS ]] || {
|
||||
printf 'Configured outer UUID is not a LUKS container\n' >&2
|
||||
exit 65
|
||||
}
|
||||
[[ "$(blkid -s UUID -o value "$outer")" == "$luks_uuid" ]] || exit 65
|
||||
if ! cryptsetup status "$mapper_name" >/dev/null; then
|
||||
printf 'Requesting the LUKS passphrase for the configured USB disk\n'
|
||||
systemd-ask-password -n --no-tty --timeout=300 \
|
||||
--id="atlas-usb-backup:${luks_uuid}" \
|
||||
'Atlas offline USB backup LUKS passphrase:' |
|
||||
cryptsetup open --type luks2 --key-file - "$outer" "$mapper_name"
|
||||
mapper_opened_by_script=true
|
||||
fi
|
||||
backing_device="$(cryptsetup status "$mapper_name" | awk '$1 == "device:" { print $2 }')"
|
||||
[[ -n "$backing_device" && "$(readlink -f "$backing_device")" == "$(readlink -f "$outer")" ]] || {
|
||||
printf 'The unlocked mapper does not belong to the configured LUKS UUID\n' >&2
|
||||
exit 65
|
||||
}
|
||||
[[ "$(blkid -s TYPE -o value "$mapper")" == ext4 ]] || {
|
||||
printf 'The unlocked USB filesystem is not ext4\n' >&2
|
||||
exit 65
|
||||
}
|
||||
[[ "$(blkid -s UUID -o value "$mapper")" == "$fs_uuid" ]] || {
|
||||
printf 'The unlocked USB filesystem UUID does not match\n' >&2
|
||||
exit 65
|
||||
}
|
||||
if findmnt -rn -S "$mapper" >/dev/null; then
|
||||
printf 'The USB filesystem is already mounted elsewhere\n' >&2
|
||||
exit 65
|
||||
fi
|
||||
[[ ! -e "$source_dir" && ! -e "$usb_mount" ]] || {
|
||||
printf 'USB backup staging directories already exist; inspect them manually\n' >&2
|
||||
exit 65
|
||||
}
|
||||
|
||||
mkdir -m 0700 "$usb_mount"
|
||||
mount -t ext4 -o nodev,nosuid,noexec "$mapper" "$usb_mount"
|
||||
usb_mounted=true
|
||||
[[ "$(readlink -f "$(findmnt -nro SOURCE --target "$usb_mount")")" == "$(readlink -f "$mapper")" ]] || {
|
||||
printf 'Mounted USB source does not match the verified mapper\n' >&2
|
||||
exit 65
|
||||
}
|
||||
|
||||
for path in "$backup_root" "$backup_root/snapshots"; do
|
||||
[[ ! -L "$path" ]] || { printf 'Unsafe symlink in USB backup destination\n' >&2; exit 65; }
|
||||
mkdir -p -- "$path"
|
||||
[[ -d "$path" ]] || exit 65
|
||||
chown root:root -- "$path"
|
||||
chmod 0700 -- "$path"
|
||||
done
|
||||
|
||||
free_bytes="$(df -B1 --output=avail "$usb_mount" | tail -n 1 | tr -d ' ')"
|
||||
if ((free_bytes < min_free_bytes)); then
|
||||
printf 'USB free space (%s bytes) is below the required reserve (%s bytes)\n' \
|
||||
"$free_bytes" "$min_free_bytes" >&2
|
||||
exit 73
|
||||
fi
|
||||
|
||||
mkdir -m 0700 "$source_dir"
|
||||
flock 9
|
||||
timestamp="$(date -u +%Y%m%dT%H%M%SZ)"
|
||||
snapshot_name="${snapshot_prefix}-${timestamp}-$$"
|
||||
printf '%s\n' "$snapshot_name" >"$snapshot_marker"
|
||||
zfs snapshot -r "${pool}@${snapshot_name}"
|
||||
flock -u 9
|
||||
printf 'Created recursive USB source snapshot %s@%s\n' "$pool" "$snapshot_name"
|
||||
|
||||
while IFS=$'\t' read -r dataset dataset_mountpoint mounted; do
|
||||
if [[ "$mounted" != yes ]]; then
|
||||
printf 'Dataset %s is not mounted; refusing an incomplete backup\n' "$dataset" >&2
|
||||
exit 65
|
||||
fi
|
||||
if [[ "$dataset_mountpoint" != "$mount_root" && "$dataset_mountpoint" != "$mount_root/"* ]]; then
|
||||
printf 'Dataset %s has unexpected mountpoint %s\n' "$dataset" "$dataset_mountpoint" >&2
|
||||
exit 65
|
||||
fi
|
||||
dataset_suffix="${dataset#"$pool"}"
|
||||
source_path="${dataset_mountpoint}/.zfs/snapshot/${snapshot_name}"
|
||||
target_path="${source_dir}${dataset_suffix}"
|
||||
mkdir -p "$target_path"
|
||||
mount --bind "$source_path" "$target_path"
|
||||
mounted_targets+=("$target_path")
|
||||
mount -o remount,bind,ro "$target_path"
|
||||
done < <(zfs list -H -o name,mountpoint,mounted -s name -r "$pool")
|
||||
|
||||
previous=""
|
||||
if [[ -e "$backup_root/latest" || -L "$backup_root/latest" ]]; then
|
||||
[[ -L "$backup_root/latest" ]] || { printf 'latest is not a symlink\n' >&2; exit 65; }
|
||||
previous="$(readlink -e "$backup_root/latest")"
|
||||
[[ -n "$previous" && "$previous" == "$backup_root/snapshots/"* && -d "$previous" ]] || {
|
||||
printf 'latest does not point to a complete snapshot on the USB disk\n' >&2
|
||||
exit 65
|
||||
}
|
||||
fi
|
||||
|
||||
backup_name="${timestamp}-$$"
|
||||
candidate_partial="${backup_root}/snapshots/.incomplete-${backup_name}"
|
||||
complete="${backup_root}/snapshots/${backup_name}"
|
||||
[[ ! -e "$candidate_partial" && ! -L "$candidate_partial" && ! -e "$complete" && ! -L "$complete" ]] || exit 65
|
||||
mkdir -m 0700 "$candidate_partial"
|
||||
partial="$candidate_partial"
|
||||
|
||||
printf 'Copying the consistent pool tree to USB backup %s\n' "$backup_name"
|
||||
# Preserve POSIX ACLs, ownership, modes, timestamps, hard links, and sparse
|
||||
# files. Do not preserve generic xattrs: Rocky 9's rsync 3.2.7 fails when
|
||||
# combining xattrs with --link-dest, while SELinux labels were intentionally
|
||||
# excluded because restores must relabel for their destination host.
|
||||
rsync_args=(-aHAS --numeric-ids "--info=progress2,stats2")
|
||||
estimate_args=(-aHAS --numeric-ids --dry-run --stats)
|
||||
if [[ -n "$previous" ]]; then
|
||||
rsync_args+=("--link-dest=$previous")
|
||||
estimate_args+=("--link-dest=$previous")
|
||||
fi
|
||||
|
||||
# The rsync dry run estimates changed file bytes after link-dest deduplication.
|
||||
# Metadata and filesystem allocation still require the separate free-space reserve.
|
||||
estimate="$(rsync "${estimate_args[@]}" "${source_dir}/" "${partial}/")"
|
||||
transfer_bytes="$(printf '%s\n' "$estimate" | awk -F: \
|
||||
'/^Total transferred file size:/ { gsub(/[^0-9]/, "", $2); print $2 }')"
|
||||
[[ "$transfer_bytes" =~ ^[0-9]+$ ]] || {
|
||||
printf 'Could not determine the USB transfer size\n' >&2
|
||||
exit 74
|
||||
}
|
||||
if ((free_bytes - transfer_bytes < min_free_bytes)); then
|
||||
printf 'Insufficient USB space: %s bytes free, %s estimated transfer, %s reserved\n' \
|
||||
"$free_bytes" "$transfer_bytes" "$min_free_bytes" >&2
|
||||
exit 73
|
||||
fi
|
||||
|
||||
rsync "${rsync_args[@]}" "${source_dir}/" "${partial}/"
|
||||
|
||||
printf 'Verifying USB backup %s with a checksum-based dry run\n' "$backup_name"
|
||||
verification="${runtime_dir}/verification.out"
|
||||
rsync -aHAS --numeric-ids \
|
||||
--checksum --dry-run --delete --itemize-changes \
|
||||
"${source_dir}/" "${partial}/" >"$verification"
|
||||
if [[ -s "$verification" ]]; then
|
||||
printf 'USB verification found mismatches; refusing to publish the backup\n' >&2
|
||||
exit 74
|
||||
fi
|
||||
|
||||
free_bytes="$(df -B1 --output=avail "$usb_mount" | tail -n 1 | tr -d ' ')"
|
||||
if ((free_bytes < min_free_bytes)); then
|
||||
printf 'USB backup completed below the free-space reserve; refusing to publish it\n' >&2
|
||||
exit 73
|
||||
fi
|
||||
|
||||
mv -- "$partial" "$complete"
|
||||
partial=""
|
||||
ln -s "snapshots/${backup_name}" "${backup_root}/.latest-${backup_name}"
|
||||
mv -Tf -- "${backup_root}/.latest-${backup_name}" "${backup_root}/latest"
|
||||
published=true
|
||||
sync -f "$complete"
|
||||
sync -f "$backup_root"
|
||||
printf 'USB backup %s verified and published; unmounting and closing LUKS\n' "$backup_name"
|
||||
@@ -0,0 +1,31 @@
|
||||
#!/usr/bin/python3
|
||||
"""Submit a manual-backup reminder through Atlas' existing Houston notifier."""
|
||||
|
||||
import json
|
||||
import subprocess
|
||||
from datetime import datetime, timezone
|
||||
|
||||
|
||||
now = datetime.now(timezone.utc)
|
||||
message = {
|
||||
"timestamp": now.isoformat(timespec="seconds"),
|
||||
"unixtime": int(now.timestamp()),
|
||||
"event": "atlas_usb_backup_reminder",
|
||||
"severity": "warning",
|
||||
"subject": "Promemoria backup USB offline Atlas",
|
||||
"email_message": (
|
||||
"Collega il disco USB di backup ad Atlas ed esegui manualmente il backup offline.\n"
|
||||
"Il promemoria non avvia il backup. Controlla che il disco non sia\n"
|
||||
"montato; poi esegui:\n\n"
|
||||
" sudo systemctl start atlas-usb-backup.service\n\n"
|
||||
"Verifica l'esito con:\n"
|
||||
" sudo journalctl -u atlas-usb-backup.service -n 100 --no-pager\n\n"
|
||||
"Dopo la riuscita, scollega fisicamente il disco."
|
||||
),
|
||||
}
|
||||
|
||||
subprocess.run(
|
||||
[{{ atlas_usb_reminder_notifier | to_json }}, json.dumps(message)],
|
||||
check=True,
|
||||
)
|
||||
print("Atlas USB backup reminder submitted to 45Drives Alerts; email delivery is not verified.", flush=True)
|
||||
@@ -0,0 +1,22 @@
|
||||
[Unit]
|
||||
Description=45Drives Alerts reminder to run the manual Atlas offline USB backup
|
||||
Requires=houston-dbus.service
|
||||
After=houston-dbus.service
|
||||
ConditionFileIsExecutable=/usr/local/libexec/atlas-usb-reminder
|
||||
|
||||
[Service]
|
||||
Type=oneshot
|
||||
ExecStart=/usr/local/libexec/atlas-usb-reminder
|
||||
User=root
|
||||
Group=root
|
||||
UMask=0077
|
||||
NoNewPrivileges=true
|
||||
PrivateTmp=true
|
||||
ProtectHome=true
|
||||
ProtectSystem=strict
|
||||
ProtectKernelTunables=true
|
||||
ProtectKernelModules=true
|
||||
ProtectControlGroups=true
|
||||
RestrictAddressFamilies=AF_UNIX
|
||||
RestrictRealtime=true
|
||||
LockPersonality=true
|
||||
@@ -0,0 +1,10 @@
|
||||
[Unit]
|
||||
Description=Remind the administrator to run the manual Atlas offline USB backup
|
||||
|
||||
[Timer]
|
||||
OnCalendar={{ atlas_usb_reminder_calendar }}
|
||||
Persistent=true
|
||||
Unit=atlas-usb-reminder.service
|
||||
|
||||
[Install]
|
||||
WantedBy=timers.target
|
||||
@@ -0,0 +1,51 @@
|
||||
#!/usr/bin/env bash
|
||||
set -Eeuo pipefail
|
||||
|
||||
export PATH=/usr/sbin:/usr/bin:/sbin:/bin
|
||||
|
||||
readonly pool={{ atlas_zfs_pool | quote }}
|
||||
readonly mount_root={{ atlas_mount_root | quote }}
|
||||
readonly snapshot_prefix={{ atlas_usb_backup_snapshot_prefix | quote }}
|
||||
readonly marker=/run/atlas-usb-backup/snapshot-name
|
||||
|
||||
[[ -e "$marker" ]] || exit 0
|
||||
[[ -f "$marker" && ! -L "$marker" ]] || {
|
||||
printf 'Unsafe Atlas USB snapshot marker; leaving snapshots unchanged\n' >&2
|
||||
exit 2
|
||||
}
|
||||
IFS= read -r snapshot_name <"$marker"
|
||||
[[ "$snapshot_name" =~ ^${snapshot_prefix}-[0-9]{8}T[0-9]{6}Z-[0-9]+$ ]] || {
|
||||
printf 'Invalid Atlas USB snapshot marker; leaving snapshots unchanged\n' >&2
|
||||
exit 2
|
||||
}
|
||||
|
||||
exec 9>/run/lock/atlas-zfs-snapshot.lock
|
||||
flock 9
|
||||
if zfs list -H -t snapshot -o name "${pool}@${snapshot_name}" >/dev/null 2>&1; then
|
||||
# ZFS can leave its on-demand .zfs/snapshot mounts in the host namespace
|
||||
# even after the backup's private bind mounts and process have exited.
|
||||
snapshot_mounts=()
|
||||
snapshot_sources=()
|
||||
while IFS=$'\t' read -r dataset dataset_mountpoint; do
|
||||
[[ "$dataset_mountpoint" == "$mount_root" || "$dataset_mountpoint" == "$mount_root/"* ]] || continue
|
||||
snapshot_mounts+=("${dataset_mountpoint}/.zfs/snapshot/${snapshot_name}")
|
||||
snapshot_sources+=("${dataset}@${snapshot_name}")
|
||||
done < <(zfs list -H -o name,mountpoint -s name -r "$pool")
|
||||
|
||||
{% raw %}
|
||||
for ((index = ${#snapshot_mounts[@]} - 1; index >= 0; index--)); do
|
||||
{% endraw %}
|
||||
mounted_source="$(findmnt -rn -M "${snapshot_mounts[$index]}" -o SOURCE || true)"
|
||||
[[ -n "$mounted_source" ]] || continue
|
||||
[[ "$mounted_source" == "${snapshot_sources[$index]}" ]] || {
|
||||
printf 'Unexpected source on Atlas USB snapshot mount: %s\n' \
|
||||
"${snapshot_mounts[$index]}" >&2
|
||||
exit 2
|
||||
}
|
||||
umount "${snapshot_mounts[$index]}"
|
||||
done
|
||||
|
||||
zfs destroy -r "${pool}@${snapshot_name}"
|
||||
printf 'Removed recursive Atlas USB source snapshot %s@%s after backup exit\n' \
|
||||
"$pool" "$snapshot_name"
|
||||
fi
|
||||
@@ -66,6 +66,17 @@
|
||||
changed_when: false
|
||||
failed_when: false
|
||||
|
||||
- name: Check whether the pinned Java 25 version is installed with Mise
|
||||
tags: [packages, mise, java, wsl]
|
||||
ansible.builtin.command:
|
||||
cmd: "mise where java@{{ workstation_mise_java_25_version }}"
|
||||
become_user: "{{ username }}"
|
||||
environment:
|
||||
HOME: "{{ user_home }}"
|
||||
register: workstation_mise_java_25_where
|
||||
changed_when: false
|
||||
failed_when: false
|
||||
|
||||
- name: Check whether the pinned Maven version is installed with Mise
|
||||
tags: [packages, mise, maven, wsl]
|
||||
ansible.builtin.command:
|
||||
@@ -86,6 +97,7 @@
|
||||
HOME: "{{ user_home }}"
|
||||
when: >-
|
||||
workstation_mise_java_where.rc != 0 or
|
||||
workstation_mise_java_25_where.rc != 0 or
|
||||
workstation_mise_maven_where.rc != 0
|
||||
|
||||
- name: Ensure WSL boot configuration file exists
|
||||
|
||||
@@ -1,3 +1,6 @@
|
||||
[tools]
|
||||
java = "temurin-11.0.31+11"
|
||||
java = ["25.0.2", "temurin-11.0.31+11"]
|
||||
maven = "3.9.16"
|
||||
|
||||
[env]
|
||||
JAVA_HOME = "{{ env.HOME }}/.local/share/mise/installs/java/25.0.2"
|
||||
|
||||
@@ -3,14 +3,12 @@ vault_duckdns_token: "CHANGEME"
|
||||
vault_personal_full_name: "REPLACE_ME"
|
||||
vault_git_email: "REPLACE_ME"
|
||||
vault_git_signing_key: "REPLACE_ME"
|
||||
vault_icloud_email: "REPLACE_ME"
|
||||
vault_protonmail_email: "REPLACE_ME"
|
||||
vault_icloud_mail_password: "REPLACE_ME"
|
||||
vault_git_work_email: "REPLACE_ME"
|
||||
vault_git_work_gpg: "REPLACE_ME"
|
||||
vault_openai_api_key: "REPLACE_ME"
|
||||
vault_ikaros_authorized_ssh_keys:
|
||||
- "ssh-ed25519 REPLACE_ME"
|
||||
vault_aegis_icloudpd_apple_id: "REPLACE_ME"
|
||||
vault_atlas_admin_password_hash: "REPLACE_WITH_A_SHADOW_COMPATIBLE_HASH"
|
||||
vault_atlas_samba_password: "REPLACE_ME"
|
||||
vault_atlas_immich_db_password: "REPLACE_ME"
|
||||
|
||||
Reference in New Issue
Block a user